Repository navigation
Expand file tree
/
Copy pathassign_form_ids.py
More file actions
728 lines (655 loc) · 35.2 KB
/
Copy pathassign_form_ids.py
File metadata and controls
728 lines (655 loc) · 35.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
#!/usr/bin/env python3
"""Assign durable, content-independent public IDs to form nodes.
The legacy importer numbers reflexes from file order and etymon-local counters. Those IDs change
when files are inserted, records are reordered, or a form is assigned to a different etymon. This
post-unification pass gives every form an opaque ``f_…`` ID and rewrites the completed graph
atomically, except where the entry has a native CDIAL or DEDR identifier. Those dictionary IDs
remain unchanged because they are stable source identifiers and useful citations.
Identity lives in ``data/form-identities.csv``. A fingerprint is only a reconciliation aid: it is
computed from source transcription and provenance, never used again as the identity once a form
has entered the registry. Consequently profile/normalisation changes do not change IDs. When a
source supplies no immutable record key, a simultaneous source-text edit and row reorder may need
manual registry reconciliation; the script refuses ambiguous matches rather than silently moving
an ID to another form.
Run after ``unify_cldf.py`` and before ``concepts.py`` / ``align.py``.
"""
from __future__ import annotations
import argparse
import base64
import csv
import hashlib
import re
from collections import defaultdict
from pathlib import Path
import etymology_assignments as overlay
from edges_build import validate_edge_dicts
from form_note_policy import apply_form_note_policy
ROOT = Path(__file__).resolve().parent
FORMS = ROOT / "cldf/forms.csv"
REGISTRY = ROOT / "data/form-identities.csv"
ALIASES = ROOT / "cldf/form-id-aliases.csv"
# Curated etymology decisions live in per-source sidecars (see etymology_assignments.py); a single
# overlay file is still accepted through --assignments for tests and one-off tools.
SOURCE_KEYS = ROOT / "cldf/form-source-keys.csv"
GRAPH_FILE_COLUMNS = {
"edges.csv": ("Child_ID", "Parent_ID"),
# build intermediate; present only under `unify_cldf.py --legacy-cols` (else a no-op)
"derivation.csv": ("Child_ID", "Parent_ID"),
"forms-legacy.csv": ("ID", "Origin_ID", "Redirect", "Variant_Of", "Borrowed_From"),
"merges.csv": ("Addendum_ID", "Main_ID"),
# Optional structured source-prose sidecar consumed by jambu-static's DB builder.
"entry-texts.csv": ("Form_ID",),
# Records which evidence row supplied each DEDR display head. Keep the pointer durable even
# when its source form receives an opaque public ID in this pass.
"pdr-headword-audit.csv": ("Source_Form_ID",),
# Article-level cross-family links normally retain native DEDR/CDIAL IDs, but keeping them in
# the generic rewrite map makes the sidecar safe if either endpoint is ever canonicalized.
"comparisons.csv": ("Entry_ID", "Compared_Entry_ID"),
# These are normally regenerated later in the pipeline. Rewriting them too keeps a checkout
# internally consistent immediately after the one-time ID migration.
"form_concepts.csv": ("Form_ID",),
"alignments.csv": ("Form_ID", "Origin_ID"),
}
REGISTRY_FIELDS = [
"Form_ID", "Legacy_ID", "Source_Key", "Fingerprint", "Source", "Language_ID", "Original",
"Gloss", "Status",
]
ASSIGNMENT_FIELDS = ["Form_ID", "Etymon_ID", "Kind", "Rank", "Status", "Source", "Notes", "Pos"]
EDGES_FIELDS = ["Child_ID", "Parent_ID", "Kind", "Rank", "Pos", "Source", "Note"]
def normalized(value: str) -> str:
return " ".join((value or "").strip().split())
def source_identity(value: str) -> str:
"""Citation locators are mutable metadata, not part of a form's identity."""
bare = re.sub(r"\[[^\]]*\]", "", value or "")
return ";".join(sorted(filter(None, (normalized(item) for item in bare.split(";")))))
def primary_source(value: str) -> str:
"""The dictionary or survey a row comes from; supporting citations may be added later."""
return normalized(re.sub(r"\[[^\]]*\]", "", (value or "").split(";", 1)[0]))
def fingerprint(row: dict[str, str], source_key: str = "") -> str:
"""Fingerprint raw-ish provenance, deliberately excluding generated transcription and graph."""
if source_key:
return hashlib.blake2b(
("jambu-source-key-v1\x1f" + source_key).encode("utf-8"), digest_size=16
).hexdigest()
identity = "\x1f".join(
(
source_identity(row.get("Source", "")),
normalized(row.get("Language_ID", "")),
normalized(row.get("Original", "") or row.get("Form", "")),
normalized(row.get("Gloss", "")),
normalized(row.get("Native", "")),
)
)
return hashlib.blake2b(identity.encode("utf-8"), digest_size=16).hexdigest()
def mint_id(fp: str, discriminator: str, used: set[str]) -> str:
nonce = 0
while True:
seed = f"jambu-form-v1\x1f{fp}\x1f{discriminator}\x1f{nonce}".encode()
# 64 bits gives a ~1 in 7.4 billion birthday-collision chance at 370k forms; the explicit
# used-set check makes a collision harmless while keeping URLs substantially shorter.
token = base64.b32encode(hashlib.blake2b(seed, digest_size=8).digest()).decode()
candidate = "f_" + token.rstrip("=").lower()
if candidate not in used:
return candidate
nonce += 1
def read_rows(path: Path) -> tuple[list[str], list[dict[str, str]]]:
if not path.exists():
return [], []
with path.open(encoding="utf-8", newline="") as handle:
reader = csv.DictReader(handle)
return list(reader.fieldnames or []), list(reader)
def write_rows(path: Path, fields: list[str], rows: list[dict[str, str]]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8", newline="") as handle:
writer = csv.DictWriter(handle, fieldnames=fields, extrasaction="ignore", lineterminator="\n")
writer.writeheader()
writer.writerows(rows)
def has_dictionary_entry_id(row: dict[str, str]) -> bool:
"""Keep only native CDIAL and DEDR entry identifiers as public IDs."""
form_id = row.get("ID", "")
language_id = row.get("Language_ID", "")
sources = {part.strip() for part in row.get("Source", "").split(";")}
is_cdial = language_id == "Indo-Aryan" and (
bool(re.fullmatch(r"\d+[a-z]?", form_id))
or (row.get("Status") == "entry" and "CDIAL" in sources)
)
is_dedr = language_id == "PDr" and bool(re.fullmatch(r"d\d+", form_id))
return is_cdial or is_dedr
def legacy_position(legacy_id: str) -> tuple[str, int] | None:
"""Split a positional legacy ID such as ``0-137060`` into its layer and row index."""
layer, sep, index = (legacy_id or "").rpartition("-")
return (layer, int(index)) if sep and index.isdigit() else None
def nearest_legacy_position(
old_id: str, candidates: list[dict[str, str]], max_distance: int | None = None
) -> dict[str, str] | None:
"""Among same-fingerprint registry rows, the one whose legacy position lies closest."""
position = legacy_position(old_id)
if position is None:
return None
ranked = []
for candidate in candidates:
candidate_position = legacy_position(candidate.get("Legacy_ID", ""))
if candidate_position and candidate_position[0] == position[0]:
ranked.append((abs(candidate_position[1] - position[1]), candidate.get("Status") != "active", candidate))
if not ranked:
return None
ranked.sort(key=lambda item: item[:2])
# only decide when the nearest is unambiguous; an active identity outranks a retired
# tombstone left at the same position by an earlier build
if len(ranked) > 1 and ranked[0][:2] == ranked[1][:2]:
return None
if max_distance is not None and ranked[0][0] > max_distance:
return None
return ranked[0][2]
def folded_node_split(registry_row: dict[str, str], row: dict[str, str]) -> bool:
"""True when ``row`` is (one of) the source record(s) behind ``registry_row``: same
provenance and language, and its source form is the registry Original or one of the
``; ``-joined Originals of a node make_cldf folded from several survey sites."""
original = row.get("Original", "") or row.get("Form", "")
registry_sources = set(source_identity(registry_row.get("Source", "")).split(";"))
row_sources = set(source_identity(row.get("Source", "")).split(";"))
registry_parts = set(registry_row.get("Original", "").split("; "))
row_parts = set(original.split("; "))
if registry_row.get("Language_ID") != row.get("Language_ID", ""):
return False
# the node split: this row is one of the sites the registry node had folded
if row_parts <= registry_parts and row_sources <= registry_sources:
return True
# the node grew: the registry record (folded or not) is now inside this row
return registry_parts <= row_parts and registry_sources <= row_sources
def assign_ids(
forms: list[dict[str, str]], registry: list[dict[str, str]], source_keys: dict[str, str] | None = None
) -> tuple[dict[str, str], list[dict[str, str]]]:
source_keys = source_keys or {}
by_form_id = {row["Form_ID"]: row for row in registry if row.get("Form_ID")}
# A retired tombstone can legitimately retain a positional legacy ID that was later reused.
# Prefer the active identity regardless of CSV sort order; otherwise a retired row can steal
# the lookup and cause curated assignments to appear to reference a missing form.
by_legacy: dict[str, dict[str, str]] = {}
for row in registry:
legacy_id = row.get("Legacy_ID", "")
if legacy_id and (
legacy_id not in by_legacy or row.get("Status") == "active"
):
by_legacy[legacy_id] = row
by_fp: dict[str, list[dict[str, str]]] = defaultdict(list)
by_source_key: dict[str, list[dict[str, str]]] = defaultdict(list)
# provenance + language + source form, without the gloss: the durable identity of a record
# whose gloss was corrected and whose generated position moved
by_record: dict[tuple[str, str, str], list[dict[str, str]]] = defaultdict(list)
for row in registry:
if row.get("Source_Key"):
by_source_key[row["Source_Key"]].append(row)
if row.get("Fingerprint"):
by_fp[row["Fingerprint"]].append(row)
by_record[(
primary_source(row.get("Source", "")), normalized(row.get("Language_ID", "")),
normalized(row.get("Original", "")),
)].append(row)
used = set(by_form_id)
claimed: set[str] = set()
mapping: dict[str, str] = {}
snapshots: dict[str, dict[str, str]] = {}
for row in forms:
old_id = row["ID"]
if has_dictionary_entry_id(row):
continue
source_key = source_keys.get(old_id, "")
fp = fingerprint(row, source_key)
match = by_form_id.get(old_id)
legacy_match = by_legacy.get(old_id)
if not match and source_key:
# Some historical snapshots gained keys without updating their old
# content fingerprint. The immutable key still identifies the row.
candidates = [candidate for candidate in by_source_key.get(source_key, [])
if candidate["Form_ID"] not in claimed]
active_candidates = [candidate for candidate in candidates if candidate.get("Status") == "active"]
candidates = active_candidates or candidates
if len(candidates) == 1:
match = candidates[0]
if not match and legacy_match and legacy_match.get("Fingerprint") == fp:
match = legacy_match
if (
not match and not source_key and legacy_match
and legacy_match.get("Status") == "active"
and ("; " in legacy_match.get("Original", "") or ";" in row.get("Source", ""))
and folded_node_split(legacy_match, row)
):
# A node that make_cldf had folded from several survey sites carries their
# Originals joined by "; ". When a transcription change stops those sites from
# sharing a display form, the site at the old position keeps the public ID and
# the others are minted afresh; this must beat the fingerprint fallback, which
# would otherwise revive the tombstone the fold had retired.
match = legacy_match
if not match and not source_key:
# The same source record at the nearest generated position keeps its public ID
# even when its gloss changed (a parser now fills it), it gained supporting
# citations, and rows above it were removed. Prefer this over a fingerprint match,
# which could otherwise revive a retired tombstone from an earlier gloss.
record_key = (
primary_source(row.get("Source", "")), normalized(row.get("Language_ID", "")),
normalized(row.get("Original", "") or row.get("Form", "")),
)
candidates = [candidate for candidate in by_record.get(record_key, []) if candidate["Form_ID"] not in claimed]
active = [candidate for candidate in candidates if candidate.get("Status") == "active"]
if active:
match = nearest_legacy_position(old_id, active, max_distance=5000)
if not match:
candidates = [candidate for candidate in by_fp.get(fp, []) if candidate["Form_ID"] not in claimed]
if len(candidates) == 1:
match = candidates[0]
elif len(candidates) > 1:
# Homographs with the same gloss in different articles (CDIAL H. saṛak 'road' under
# 12269 and 13577) share a fingerprint. When rows above them were removed, their
# positional legacy IDs no longer match exactly, but the nearest legacy position
# still identifies each record; minting new IDs here would orphan curated
# etymologies that reference the old ones.
match = nearest_legacy_position(old_id, candidates)
if not match and source_key:
# A source may gain immutable record keys after it has already shipped with
# provenance-based identities. Preserve the existing public ID when the old
# fingerprint identifies exactly one unclaimed registry row; the snapshot written
# below upgrades that identity to the source key. Never guess among duplicate
# legacy attestations (for example, identical forms at multiple survey sites).
legacy_fp = fingerprint(row)
candidates = [
candidate
for candidate in by_fp.get(legacy_fp, [])
if candidate["Form_ID"] not in claimed
]
if len(candidates) == 1:
match = candidates[0]
if not match and legacy_match:
# This preserves a corrected source row when its old generated position did not move.
registry_source_key = legacy_match.get("Source_Key", "")
if source_key and registry_source_key:
# Once both records have immutable keys, a reused positional ID must never
# capture a neighbouring homograph after rows are inserted or reordered.
same_source_record = registry_source_key == source_key
else:
# Older sources without immutable keys still need editorial corrections to a
# gloss to preserve their public ID when provenance, language and source form
# identify the same record at the same legacy position.
same_source_record = folded_node_split(legacy_match, row)
if same_source_record:
match = legacy_match
if match:
form_id = match["Form_ID"]
if form_id in claimed:
match = None
if not match:
form_id = mint_id(fp, old_id, used)
used.add(form_id)
claimed.add(form_id)
mapping[old_id] = form_id
snapshots[form_id] = {
"Form_ID": form_id,
"Legacy_ID": old_id if not old_id.startswith("f_") else by_form_id.get(old_id, {}).get("Legacy_ID", ""),
"Source_Key": source_key or (match.get("Source_Key", "") if match else ""),
"Fingerprint": fp,
"Source": row.get("Source", ""),
"Language_ID": row.get("Language_ID", ""),
"Original": row.get("Original", "") or row.get("Form", ""),
"Gloss": row.get("Gloss", ""),
"Status": "active",
}
# The reverse of a split: a transcription change can make make_cldf fold survey sites that
# used to be separate nodes into one ("; "-joined Originals). The sites that lost their own
# node are aliased to the survivor so their curated etymologies follow them.
by_site: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
for candidate in registry:
if candidate.get("Status") == "active" and candidate["Form_ID"] not in claimed:
for part in candidate.get("Original", "").split("; "):
by_site[(normalized(candidate.get("Language_ID", "")), normalized(part))].append(candidate)
for row in forms:
original = row.get("Original", "") or row.get("Form", "")
if ("; " not in original and ";" not in row.get("Source", "")) or has_dictionary_entry_id(row):
continue
survivor = mapping.get(row["ID"])
if not survivor:
continue
sources = set(source_identity(row.get("Source", "")).split(";"))
parts = {normalized(part) for part in original.split("; ")}
for part in parts:
for candidate in by_site.get((normalized(row.get("Language_ID", "")), part), []):
if candidate["Form_ID"] in claimed:
continue
# every site the old node carried is now inside this node
if (
{normalized(p) for p in candidate.get("Original", "").split("; ")} <= parts
and set(source_identity(candidate.get("Source", "")).split(";")) <= sources
):
mapping[candidate["Form_ID"]] = survivor
claimed.add(candidate["Form_ID"])
for old in registry:
if old.get("Form_ID") not in snapshots:
tombstone = dict(old)
tombstone["Status"] = "retired"
snapshots[old["Form_ID"]] = tombstone
return mapping, sorted(snapshots.values(), key=lambda row: row["Form_ID"])
def rewrite_graph_file(path: Path, columns: tuple[str, ...], mapping: dict[str, str]) -> None:
fields, rows = read_rows(path)
if not fields:
return
for row in rows:
for column in columns:
row[column] = mapping.get(row.get(column, ""), row.get(column, ""))
write_rows(path, fields, rows)
ACCEPTED = {"accepted", "yes", "active"}
REJECTED = {"rejected", "no"}
# A dictionary sub-entry is identified as ``<entry>-<n>`` (CDIAL 103-2 under
# etymon 103), which is exactly the shape of the positional pre-ID
# ``<file>-<row>`` that make_cldf mints for source rows. Those two namespaces
# overlap, and inserting one source file re-issues every later pre-ID, so a
# retired sub-entry id can be picked up by an unrelated row and then aliased to
# it. Following that alias silently moves a curated etymology onto a different
# word: Zargari ``pani`` 'water' once acquired CDIAL 103 ``aŋkapāli`` 'embrace'
# this way. An assignment that names a sub-entry of its own etymon can only
# ever have meant that sub-entry, so when the sub-entry is gone the assignment
# is stale and is dropped rather than redirected.
SUBENTRY_SUFFIX = re.compile(r"-\d+[a-z]*$")
def is_retired_subentry(assignment: dict[str, str], active_ids: set[str]) -> bool:
form_id = (assignment.get("Form_ID") or "").strip()
etymon_id = (assignment.get("Etymon_ID") or "").strip()
if not form_id or not etymon_id or form_id in active_ids:
return False
remainder = form_id[len(etymon_id):]
return form_id.startswith(etymon_id) and bool(SUBENTRY_SUFFIX.fullmatch(remainder))
def drop_stale_subentry_assignments(
assignments: list[dict[str, str]], active_ids: set[str]
) -> tuple[list[dict[str, str]], list[dict[str, str]]]:
kept, stale = [], []
for assignment in assignments:
(stale if is_retired_subentry(assignment, active_ids) else kept).append(assignment)
return kept, stale
def migrate_assignment_schema(assignments: list[dict[str, str]]) -> None:
"""One-time upgrade of legacy overlay rows (Relation column, implicit rank 1)."""
for row in assignments:
if "Kind" not in row or not row.get("Kind"):
row["Kind"] = (row.get("Relation") or "reflex").strip() or "reflex"
if not row.get("Rank"):
row["Rank"] = "1"
def validate_assignments(forms: list[dict[str, str]], assignments: list[dict[str, str]]) -> None:
"""Hard-fail before any file is mutated (same contract as the legacy overlay)."""
by_id = {row["ID"]: row for row in forms}
linkable = {row["ID"] for row in forms if row.get("Status") != "unlinked"}
# A base can gain its ancestry in this overlay before a derivative uses it.
# Resolve chains to existing linkable nodes without depending on CSV order;
# an unassigned base or a cycle of unlinked nodes is still not a valid target.
pending = [
(a.get("Form_ID", "").strip(), a.get("Etymon_ID", "").strip())
for a in assignments
if a.get("Status", "accepted").strip().lower() in ACCEPTED
and a.get("Rank", "1") == "1"
and a.get("Kind") in {"reflex", "borrowed", "derived", "component"}
]
parents = defaultdict(set)
for child, parent in pending:
parents[child].add(parent)
while True:
additions = {child for child, targets in parents.items()
if child in by_id and targets <= linkable} - linkable
if not additions:
break
linkable.update(additions)
for assignment in assignments:
status = assignment.get("Status", "accepted").strip().lower()
if status not in ACCEPTED | REJECTED:
raise ValueError(f"unsupported assignment status {status!r}")
form_id = assignment.get("Form_ID", "").strip()
etymon_id = assignment.get("Etymon_ID", "").strip()
if form_id not in by_id:
raise ValueError(f"etymology assignment references missing form {form_id}")
if status in REJECTED:
continue
if etymon_id not in linkable:
raise ValueError(f"etymology assignment for {form_id} references missing etymon {etymon_id}")
if assignment.get("Kind") not in {"reflex", "borrowed", "variant", "derived", "component"}:
raise ValueError(f"unsupported assignment kind {assignment.get('Kind')!r} for {form_id}")
if not re.fullmatch(r"[1-9]\d*", assignment.get("Rank", "1")):
raise ValueError(f"bad assignment rank {assignment.get('Rank')!r} for {form_id}")
pos = assignment.get("Pos", "")
if assignment.get("Kind") == "component":
if not re.fullmatch(r"[1-9]\d*", pos):
raise ValueError(f"bad component position {pos!r} for {form_id}")
elif pos:
raise ValueError(f"position on non-component assignment for {form_id}")
# Legacy modelled a dictionary headword as an entry row plus an attested row beneath it;
# both collapse onto one node here, so an importer resolving that pair emits a link from
# the node to itself. Installing it makes the node its own etymon and drops it from the
# headword list — 14,506 CDIAL entries vanished this way.
if form_id == etymon_id:
raise ValueError(f"etymology assignment for {form_id} points at itself")
component_positions = defaultdict(list)
for a in assignments:
if a.get("Status", "accepted").strip().lower() in ACCEPTED and a.get("Kind") == "component":
component_positions[a["Form_ID"]].append(int(a["Pos"]))
for child, positions in component_positions.items():
if len(positions) < 2 or sorted(positions) != list(range(1, len(positions) + 1)):
raise ValueError(f"component Pos not contiguous for {child}: {positions}")
def apply_assignments(
edges_path: Path, forms: list[dict[str, str]], assignments: list[dict[str, str]]
) -> int:
"""Patch the curated overlay into cldf/edges.csv (rank-1 upserts replace the accepted
etymology; rank≥2 upserts add hypotheses; rejected rows delete generated non-primary edges).
A form gaining a rank-1 edge stops being `unlinked`."""
fields, edges = read_rows(edges_path)
if not fields:
raise ValueError(f"{edges_path} missing — run unify_cldf.py first")
by_form = {row["ID"]: row for row in forms}
rank1_by_child = {}
for edge in edges:
if edge.get("Rank") == "1" and edge.get("Kind") in {"reflex", "borrowed", "variant"}:
rank1_by_child[edge["Child_ID"]] = edge
# (child, parent) → its edges in table order: every lookup below is by that pair
by_pair: dict[tuple[str, str], list[dict[str, str]]] = defaultdict(list)
for edge in edges:
by_pair[(edge["Child_ID"], edge["Parent_ID"])].append(edge)
dropped: set[int] = set() # id() of rejected edges, filtered out once at the end
def add_edge(edge: dict[str, str]) -> None:
edges.append(edge)
by_pair[(edge["Child_ID"], edge["Parent_ID"])].append(edge)
changed = 0
for assignment in assignments:
status = assignment.get("Status", "accepted").strip().lower()
form_id = assignment.get("Form_ID", "").strip()
etymon_id = assignment.get("Etymon_ID", "").strip()
kind = assignment.get("Kind", "reflex")
rank = assignment.get("Rank", "1")
if status in REJECTED:
group = by_pair.get((form_id, etymon_id), [])
doomed = [e for e in group if e["Rank"] != "1"]
if doomed:
dropped.update(id(e) for e in doomed)
by_pair[(form_id, etymon_id)] = [e for e in group if e["Rank"] == "1"]
changed += len(doomed)
continue
if kind in {"derived", "component"}:
# Derivations are non-attestation edges, so do not put them in the
# reflex/loan rank-1 index or replace an unrelated ancestry edge.
match = next((e for e in by_pair.get((form_id, etymon_id), ()) if
e["Kind"] == kind and e["Rank"] == rank
and e.get("Pos", "") == assignment.get("Pos", "")), None)
if match is None:
add_edge(dict(
Child_ID=form_id, Parent_ID=etymon_id, Kind=kind, Rank=rank,
Pos=assignment.get("Pos", ""), Source=assignment.get("Source", ""),
Note=assignment.get("Notes", ""),
))
changed += 1
else:
values = dict(Source=assignment.get("Source", ""),
Note=assignment.get("Notes", ""))
if any(match.get(k, "") != v for k, v in values.items()):
match.update(values)
changed += 1
row = by_form.get(form_id)
if rank == "1" and row is not None and row.get("Status") in ("unlinked", "entry"):
row["Status"] = ""
changed += 1
continue
if rank == "1":
existing = rank1_by_child.get(form_id)
if existing is not None:
if (existing["Parent_ID"], existing["Kind"]) != (etymon_id, kind):
existing.update(
Parent_ID=etymon_id, Kind=kind,
Source=assignment.get("Source", ""), Note="",
)
changed += 1
else:
edge = dict(
Child_ID=form_id, Parent_ID=etymon_id, Kind=kind, Rank="1", Pos="",
Source=assignment.get("Source", ""), Note="",
)
add_edge(edge)
rank1_by_child[form_id] = edge
changed += 1
# An accepted rank-1 edge makes the node attested, so clear whichever parentless
# Status it carried and keep forms.csv agreeing with edges.csv. `entry` applies to
# dictionary sub-entries (CDIAL 9017-2) that the overlay re-homes onto their head.
row = by_form.get(form_id)
if row is not None and row.get("Status") in ("unlinked", "entry"):
row["Status"] = ""
changed += 1
else:
match = [e for e in by_pair.get((form_id, etymon_id), ()) if e["Rank"] != "1"]
if match:
for e in match:
if e.get("Note", "").startswith("review:") or e.get("Kind") != kind:
e.update(Kind=kind, Rank=rank, Source=assignment.get("Source", ""), Note="")
changed += 1
else:
add_edge(dict(
Child_ID=form_id, Parent_ID=etymon_id, Kind=kind, Rank=rank, Pos="",
Source=assignment.get("Source", ""), Note="",
))
changed += 1
if dropped:
edges = [e for e in edges if id(e) not in dropped]
edges.sort(key=lambda e: (
e["Child_ID"], e["Kind"], int(e["Rank"] or 1), int(e["Pos"] or 0), e["Parent_ID"]
))
# The last writer of cldf/edges.csv re-checks the shipped table against the same contract
# `edges_build` enforces when it first derives the graph — otherwise overlay-only breakage
# (self-edges, a headword given a parent) reaches the browser DB unnoticed.
validate_edge_dicts(edges, {row["ID"]: row.get("Status", "") for row in forms})
write_rows(edges_path, EDGES_FIELDS, edges)
return changed
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--forms", type=Path, default=FORMS)
parser.add_argument("--registry", type=Path, default=REGISTRY)
parser.add_argument("--aliases", type=Path, default=ALIASES)
parser.add_argument(
"--assignments", type=Path, default=None,
help="a single overlay CSV instead of the per-source sidecars (tests / one-off tools)",
)
parser.add_argument("--source-keys", type=Path, default=SOURCE_KEYS)
parser.add_argument(
"--fresh", action="store_true",
help="replace a just-created registry during an unshipped migration (never use after curation)",
)
args = parser.parse_args()
fields, forms = read_rows(args.forms)
if not forms or "Status" not in fields or "Redirect" not in fields:
raise ValueError(f"{args.forms} is not a unified Jambu forms table (edge-model format)")
_, registry = read_rows(args.registry)
if args.fresh:
reverse = {
row["Form_ID"]: row["Legacy_ID"]
for row in registry
if row.get("Form_ID") and row.get("Legacy_ID") and row.get("Status") == "active"
}
for row in forms:
row["ID"] = reverse.get(row["ID"], row["ID"])
row["Redirect"] = reverse.get(row.get("Redirect", ""), row.get("Redirect", ""))
for name, columns in GRAPH_FILE_COLUMNS.items():
rewrite_graph_file(args.forms.parent / name, columns, reverse)
registry = []
_, source_key_rows = read_rows(args.source_keys)
source_keys = {
row["Legacy_ID"]: row["Source_Key"]
for row in source_key_rows
if row.get("Legacy_ID") and row.get("Source_Key")
}
mapping, next_registry = assign_ids(forms, registry, source_keys)
_, old_aliases = read_rows(args.aliases)
if args.fresh:
old_aliases = []
aliases = {row["Legacy_ID"]: row["Form_ID"] for row in old_aliases if row.get("Legacy_ID")}
aliases.update({old: new for old, new in mapping.items() if old != new})
for row in forms:
row["ID"] = mapping.get(row["ID"], row["ID"])
row["Redirect"] = mapping.get(row.get("Redirect", ""), row.get("Redirect", ""))
active_ids = {row["ID"] for row in forms}
# A policy migration may restore a source-owned dictionary ID that an earlier run aliased to
# an opaque ID. Prefer the now-active dictionary ID and discard that obsolete redirect.
restored_ids = {
form_id: legacy
for legacy, form_id in aliases.items()
if legacy in active_ids and form_id not in active_ids
}
aliases = {legacy: form_id for legacy, form_id in aliases.items() if legacy not in active_ids}
from source_key_aliases import apply_source_key_aliases
apply_source_key_aliases(aliases, registry, next_registry, active_ids)
if args.assignments is not None:
if not args.assignments.exists():
write_rows(args.assignments, ASSIGNMENT_FIELDS, [])
_, assignments = read_rows(args.assignments)
else:
assignments = overlay.read_assignments()
assignments, stale = drop_stale_subentry_assignments(assignments, active_ids)
for assignment in assignments:
for column in ("Form_ID", "Etymon_ID"):
value = restored_ids.get(assignment.get(column, ""), assignment.get(column, ""))
assignment[column] = value if value in active_ids else aliases.get(value, value)
migrate_assignment_schema(assignments)
validate_assignments(forms, assignments)
# Do not mutate sidecar graph files until every assignment has validated. A bad assignment
# must leave the whole pre-ID build intact rather than producing a half-rewritten graph.
for name, columns in GRAPH_FILE_COLUMNS.items():
rewrite_graph_file(args.forms.parent / name, columns, mapping)
changed = apply_assignments(args.forms.parent / "edges.csv", forms, assignments)
from nuristani_grouping import apply_to_build
grouped = apply_to_build(forms, args.forms.parent / "edges.csv", aliases)
# Local source rows deliberately retain extraction and review prose for auditing. Apply the
# public-note boundary only after identity reconciliation, so promoting citation locators or
# hiding provenance cannot alter fingerprints, aliases, or durable form IDs.
for row in forms:
notes, source, etymology, promoted_tags = apply_form_note_policy(
row.get("Description", ""), row.get("Source", ""), row.get("Etymology", "")
)
row["Description"] = notes
row["Source"] = source
row["Etymology"] = etymology
row["Tags"] = " ".join(
dict.fromkeys(filter(None, [*row.get("Tags", "").split(), *promoted_tags]))
)
write_rows(args.forms, fields, forms)
write_rows(args.registry, REGISTRY_FIELDS, next_registry)
if args.assignments is not None:
write_rows(args.assignments, ASSIGNMENT_FIELDS, assignments)
else:
# Every row returns to the sidecar it was read from. Rows from the inbox
# (data/other/forms/etymologies/_pending.csv) are filed under the source that owns their
# child, resolved against the registry this build has just produced.
overlay.write_assignments(assignments, overlay.SidecarResolver(next_registry))
write_rows(
args.aliases,
["Legacy_ID", "Form_ID"],
[
{"Legacy_ID": legacy, "Form_ID": form_id}
for legacy, form_id in sorted(aliases.items())
if legacy != form_id
],
)
print(
f"assigned {len(mapping):,} durable form IDs; "
f"preserved {len(aliases):,} aliases; applied {changed:,} etymology assignments; "
f"updated {len(grouped):,} Nuristani/CDIAL grouping records"
+ (f"; dropped {len(stale):,} assignments on retired dictionary sub-entries"
if stale else "")
)
if __name__ == "__main__":
main()