Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
91 changes: 72 additions & 19 deletions api/services/table_actions.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,8 +108,7 @@ class RoleGate:
DELETE: RoleGate(DELETE_PERM, "Only Data maintainers and Table admins can delete"),
}

# The most Tables one request may name, per action; an action not listed has
# no ceiling yet (the Dataset actions get theirs in #2565). The dashboard is
# The most Tables one request may name, per action. The dashboard is
# synchronous by design (no task queue), so a request has to finish inside
# the host's timeout; mass work stays possible through the API, one call per
# Table. Over the ceiling the preflight reads nothing and nothing can be
Expand Down Expand Up @@ -144,9 +143,25 @@ class RoleGate:
# a safety factor of 5 against the 300 s, which covers production's OEDB
# sitting on another host and its larger buffer pool. Typical: 1-2 s. Not
# covered: a drop waiting for a lock another session holds on that Table.
#
# Adding to and removing from a Dataset: 2,500 each. The same host limit
# (300 s). Measured locally with ``benchmarks/tables_tab/dataset_cost.py``
# (Postgres 14, batches of 100, 400 and 1,000, three rounds), per Table,
# against 6 / 60 / 500 KB of metadata: adding 2.5-3.0 / 2.7-3.3 / 5.0-6.8
# ms, removing 0.8-1.2 / 1.1-1.4 / 3.3-3.9 ms; the preflight with a Dataset
# chosen is 0.1-0.3 / 0.4-0.7 / 3.0-3.5 ms of that. Adding is the dearer: it
# writes the membership and seeds the Table's Topics into the Dataset.
# Neither saves the Table, so the metadata costs only its decoding in the
# preflight. At the worst 6.8 ms, 2,500 Tables take about 17 s: a safety
# factor of about 17 against the 300 s, more than publish's, because this
# ceiling is set by what a curator needs rather than by time: "select all"
# on the largest account (2,068 Tables), then "Add to dataset", is one
# request.
CEILINGS = {
PUBLISH: 1000,
UNPUBLISH: 1000,
DATASET_ADD: 2500,
DATASET_REMOVE: 2500,
DELETE: 50,
}

Expand Down Expand Up @@ -224,6 +239,10 @@ class Preflight:
consequences: dict = field(default_factory=dict)
subject: str = ""
confirmation: str = ""
# every name sent, once each, in the order sent: what a bulk dialog
# re-checks when the user chooses a Dataset, so the left-out groups stay
# complete although the form posts only the eligible names
requested: list = field(default_factory=list)
# The Dataset actions only: the user's own Datasets the action could
# change for these Tables, and the one it is about (None until chosen).
datasets: list = field(default_factory=list)
Expand Down Expand Up @@ -252,6 +271,17 @@ def over_ceiling(self) -> bool:
def ceiling_message(self) -> str:
return _ceiling_message(self.action, self.ceiling, self.total)

@property
def ceiling_rule(self) -> str:
"""The ceiling as the dialog states it before anything exceeds it,
"" where the action has none."""
return _ceiling_rule(self.action, self.ceiling) if self.ceiling else ""

@property
def dataset_title(self) -> str:
"""What the chosen Dataset is called, "" while none is chosen."""
return dataset_title(self.dataset) if self.dataset else ""


@dataclass(frozen=True)
class Outcome:
Expand Down Expand Up @@ -321,11 +351,23 @@ def _gate_reason(table) -> str:
return "Fails the Publish gate: " + ", ".join(failed)


# What an action is called at the start of a sentence, as in the ceiling's
# "Delete takes at most 50 tables at a time".
ACTION_NAMES = {
PUBLISH: "Publish",
UNPUBLISH: "Unpublish",
DELETE: "Delete",
DATASET_ADD: "Adding to a dataset",
DATASET_REMOVE: "Removing from a dataset",
}


def _ceiling_rule(action, ceiling) -> str:
return f"{ACTION_NAMES[action]} takes at most {ceiling:,} tables at a time."


def _ceiling_message(action, ceiling, total) -> str:
return (
f"{action.capitalize()} takes at most {ceiling:,} tables at a time; "
f"you selected {total:,}."
)
return f"{_ceiling_rule(action, ceiling)[:-1]}; you selected {total:,}."


def _confirmation(action, eligible) -> str:
Expand Down Expand Up @@ -380,17 +422,25 @@ def _others_datasets(user, tables) -> list:
def _delete_consequences(user, tables) -> dict:
"""What deleting ``tables`` breaks, for the dialog: the Datasets they
leave (the user's own by name, other people's by owner and name, since
their membership goes silently), which are published, their Review
state, an active embargo, and whether knowledge-graph links may point at
them. Five queries whatever the number of Tables."""
their membership goes silently), each with how many of ``tables`` leave
it, which are published, their Review state, an active embargo, and
whether knowledge-graph links may point at them. A batch is shown
counted rather than listed per Table, so the review states are counted
here too. Three queries whatever the number of Tables."""
names = [table.name for table in tables]
published = [table for table in tables if table.is_publish]
datasets = (
Dataset.objects.filter(tables__in=tables)
.filter(pk__in=visible_datasets(user).values("pk"))
.order_by("name")
.distinct()
Dataset.objects.filter(pk__in=visible_datasets(user).values("pk"))
.annotate(leaving=Count("tables", filter=Q(tables__in=tables)))
.filter(leaving__gt=0)
.values_list("creator_id", "creator__name", "name", "leaving")
)
own, others = [], []
for creator, owner, name, leaving in datasets:
if creator == user.pk:
own.append((name, leaving))
else:
others.append((owner or "Unknown owner", name, leaving))
reviews = {}
for name, finished in PeerReview.objects.filter(table__in=names).values_list(
"table", "is_finished"
Expand All @@ -402,16 +452,17 @@ def _delete_consequences(user, tables) -> dict:
.values_list("table__name", "date_ended")
)
by_name = {table.name: table for table in tables}
finished = sum(1 for state in reviews.values() if state)
return {
"published": published,
"own_datasets": list(
datasets.filter(creator=user).values_list("name", flat=True)
),
"others_datasets": _others_datasets(user, tables),
"own_datasets": sorted(own),
"others_datasets": sorted(others),
"reviewed": [
(by_name[name], "Reviewed" if finished else "In review")
for name, finished in sorted(reviews.items())
(by_name[name], "Reviewed" if state else "In review")
for name, state in sorted(reviews.items())
],
"finished_reviews": finished,
"open_reviews": len(reviews) - finished,
"embargoed": [
(by_name[name], until) for name, until in sorted(embargoes.items())
],
Expand Down Expand Up @@ -502,6 +553,7 @@ def preflight(user, action, names, params=None) -> Preflight:
left_out=[],
ceiling=ceiling,
subject=f"{len(names)} tables",
requested=names,
)
found = {table.name: table for table in Table.objects.filter(name__in=names)}
levels = table_levels(user, found.values())
Expand Down Expand Up @@ -559,6 +611,7 @@ def preflight(user, action, names, params=None) -> Preflight:
confirmation=_confirmation(action, eligible),
datasets=datasets,
dataset=dataset,
requested=names,
)


Expand Down
35 changes: 35 additions & 0 deletions benchmarks/tables_tab/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,9 @@ take:
delete ceiling `CEILINGS["delete"]` in `api/services/table_actions.py`.
- **`publish_cost.py`**: what publishing and unpublishing one Table cost
(#2564), which sets `CEILINGS["publish"]` and `CEILINGS["unpublish"]`.
- **`dataset_cost.py`**: what adding a Table to a Dataset and removing it cost
(#2565), which sets `CEILINGS["dataset_add"]` and
`CEILINGS["dataset_remove"]`.

Both use accounts shaped like production (`seed.py`, a port of the WF-06
prototype's generator, sized from WF-01's census): `p90` (130 Tables), `max`
Expand Down Expand Up @@ -137,3 +140,35 @@ decoding the metadata is what it pays for. Both writes save the whole row, which
is why they grow with the metadata. At the worst 34 ms, 1,000 Tables take 34 s
against production's 300 s timeout: the ceiling is 1,000 for both, a safety
factor of about 9. The reasoning is beside the constant.

## Dataset add and remove cost

Same throwaway database; Dataset membership lives in Django only, so no OEDB
table is created.

```bash
python -m benchmarks.tables_tab.dataset_cost
python -m benchmarks.tables_tab.dataset_cost --tables 100,400,1000 --metadata-kb 6,60,500
```

For each metadata size and batch size it creates Tables (half drafts, half
published) with a Data editor grant and two Topics each, and one Dataset of the
user's own, then times the preflight with that Dataset chosen, adding every
Table and removing them again, through `table_actions`. Results append to
`benchmarks/results/tables_tab_dataset.csv`.

Measured 2026-10-03, local Postgres 14, batches of 100, 400 and 1,000, three
rounds, per Table:

| metadata per Table | preflight | add | remove |
| ------------------ | ---------- | ---------- | ---------- |
| 6 KB | 0.1-0.3 ms | 2.5-3.0 ms | 0.8-1.2 ms |
| 60 KB | 0.4-0.7 ms | 2.7-3.3 ms | 1.1-1.4 ms |
| 500 KB | 3.0-3.5 ms | 5.0-6.8 ms | 3.3-3.9 ms |

Adding writes the membership and seeds the Table's Topics into the Dataset;
neither write saves the Table, so the metadata costs only its decoding in the
preflight. At the worst 6.8 ms, 2,500 Tables take about 17 s against
production's 300 s timeout: the ceiling is 2,500 for both, a safety factor of
about 17, set so that "select all" on the largest account (2,068 Tables) is one
request. The reasoning is beside the constant.
171 changes: 171 additions & 0 deletions benchmarks/tables_tab/dataset_cost.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,171 @@
"""What adding a Table to a Dataset and removing it cost, to set their ceilings (#2565).

# the default: batches of 100 and 400 Tables, 6 KB and 60 KB of metadata
python -m benchmarks.tables_tab.dataset_cost

python -m benchmarks.tables_tab.dataset_cost --tables 1000 --metadata-kb 500

Spec #2551 owes these numbers: the dashboard adds a selection to one of the
user's own Datasets, or removes it, in one request, with no task queue, so
the most Tables one request may name (``CEILINGS["dataset_add"]`` and
``CEILINGS["dataset_remove"]`` in ``api/services/table_actions.py``) has to
finish inside the host's request timeout, with a stated safety factor.

Like ``publish_cost.py`` this never touches production or a developer
database: it asks Django's test runner for a throwaway database. Dataset
membership lives in Django only, so no OEDB table is created.

For each metadata size and batch size it creates that many Tables shaped
like real ones (the metadata, a Data editor grant, two Topics each, which
adding seeds into the Dataset), half of them drafts (the curation rule reads
the grant for those) and half published, and one Dataset of the user's own,
then times what a bulk add and a bulk remove do, through the service the
dashboard calls:

- ``check``: the preflight the dialog shows once a Dataset is chosen
(``table_actions.preflight`` with ``dataset``), which also runs the
curation rule and the membership split;
- ``add``: ``table_actions.execute`` adding every Table;
- ``remove``: ``table_actions.execute`` removing them again.

Each is reported per Table: the whole call divided by the Tables in it.
"""

from __future__ import annotations

import argparse
import csv
import logging
import time
import uuid
from datetime import datetime, timezone
from pathlib import Path

from benchmarks.tables_tab.publish_cost import metadata
from benchmarks.tables_tab.run import bootstrap

DEFAULT_RESULTS = Path("benchmarks/results/tables_tab_dataset.csv")


def parse_args(argv=None):
p = argparse.ArgumentParser(
prog="python -m benchmarks.tables_tab.dataset_cost",
description="Measure what adding a Table to a Dataset and removing it cost.",
)
p.add_argument("--tables", default="100,400")
p.add_argument("--metadata-kb", default="6,60")
p.add_argument("--results", type=Path, default=DEFAULT_RESULTS)
p.add_argument("--no-results", action="store_true")
return p.parse_args(argv)


def main(argv=None) -> int:
args = parse_args(argv)
runner, old_config = bootstrap()
# one line per Table is the service's record, not this measurement's
logging.getLogger("oeplatform.table_actions").setLevel(logging.WARNING)

from django.db import connection

from api.services import table_actions
from dataedit.models import Dataset, Table, Topic
from login.models import WRITE_PERM, UserPermission, myuser

owner = myuser.objects.create(
name=f"bench_dataset_{uuid.uuid4().hex[:6]}",
email=f"bench_{uuid.uuid4().hex[:6]}@example.org",
did_agree=True,
is_mail_verified=True,
)
topics = [
Topic.objects.get_or_create(name=name)[0]
for name in ("bench_topic_a", "bench_topic_b")
]
for action in table_actions.DATASET_ACTIONS:
table_actions.CEILINGS[action] = None
stamp = datetime.now(timezone.utc).isoformat(timespec="seconds")
results = []

def per_table(fn, n):
connection.close() # a fresh connection, like a request's
start = time.perf_counter()
fn()
return round((time.perf_counter() - start) * 1000 / n, 2)

try:
for kb in [int(k) for k in args.metadata_kb.split(",")]:
document = metadata(kb)
for n in [int(t) for t in args.tables.split(",")]:
names = [f"bench_ds_{uuid.uuid4().hex[:10]}" for _ in range(n)]
tables = Table.objects.bulk_create(
Table(name=name, oemetadata=document, is_publish=i % 2 == 0)
for i, name in enumerate(names)
)
UserPermission.objects.bulk_create(
UserPermission(holder=owner, table=table, level=WRITE_PERM)
for table in tables
)
for topic in topics:
topic.tables.add(*tables)
dataset = Dataset.objects.create(
name=f"bench_ds_{uuid.uuid4().hex[:8]}",
metadata={"title": "Bench"},
creator=owner,
)
params = {"dataset": dataset.name}
problem = table_actions.preflight(owner, "dataset_add", names, params)
assert len(problem.eligible) == n, problem.left_out

check = per_table(
lambda: table_actions.preflight(
owner, "dataset_add", names, params
),
n,
)
add = per_table(
lambda: table_actions.execute(
owner, "dataset_add", names, params, via="benchmark"
),
n,
)
assert dataset.tables.count() == n
remove = per_table(
lambda: table_actions.execute(
owner, "dataset_remove", names, params, via="benchmark"
),
n,
)
assert dataset.tables.count() == 0
row = {
"run_utc": stamp,
"metadata_kb": kb,
"tables": n,
"check_per_table_ms": check,
"add_per_table_ms": add,
"remove_per_table_ms": remove,
}
results.append(row)
print(
"{metadata_kb:>4} KB x {tables:>4}: per Table check "
"{check_per_table_ms} ms, add {add_per_table_ms} ms, "
"remove {remove_per_table_ms} ms".format(**row)
)
dataset.delete()
Table.objects.filter(name__in=names).delete()
finally:
runner.teardown_databases(old_config)

if results and not args.no_results:
args.results.parent.mkdir(parents=True, exist_ok=True)
exists = args.results.exists()
with args.results.open("a", newline="", encoding="utf-8") as fh:
writer = csv.DictWriter(fh, fieldnames=list(results[0]))
if not exists:
writer.writeheader()
writer.writerows(results)
print(f"\nappended {len(results)} rows to {args.results}")
return 0


if __name__ == "__main__":
raise SystemExit(main())
Loading
Loading