Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Build stratified validation datasets for cognate detection ML training.3 4Reads lexicon TSVs and cognate-pair TSVs from data/training/, builds a5phylogenetic tree of language relationships, and generates stratified6validation sets split by phylogenetic distance, timespan, family, and7concept domain (religious terms).8 9Output goes to data/training/validation/.10"""11 12from __future__ import annotations13 14import csv15import json16import random17import sys18from collections import defaultdict19from pathlib import Path20from typing import Any21 22# Ensure cognate_pipeline package is importable23sys.path.insert(24 0,25 str(Path(__file__).resolve().parent.parent / "cognate_pipeline" / "src"),26)27from cognate_pipeline.normalise.sound_class import ipa_to_sound_class28 29# ---------------------------------------------------------------------------30# Paths31# ---------------------------------------------------------------------------32 33REPO_ROOT = Path(__file__).resolve().parent.parent34TRAINING_DIR = REPO_ROOT / "data" / "training"35LEXICONS_DIR = TRAINING_DIR / "lexicons"36COGNATE_DIR = TRAINING_DIR / "cognate_pairs"37OUTPUT_DIR = TRAINING_DIR / "validation"38FAMILY_MAP_PATH = (39 REPO_ROOT40 / "cognate_pipeline"41 / "src"42 / "cognate_pipeline"43 / "cognate"44 / "family_map.json"45)46 47PAIR_CAP = 50_00048SEED = 4249MAX_PAIRS_PER_CONCEPT_PER_LEVEL = 10050MAX_CROSS_FAMILY_PAIRS_PER_CONCEPT = 5051TRUE_NEG_SAMPLE_ATTEMPTS = 2_000_00052 53# ---------------------------------------------------------------------------54# TSV field names for output55# ---------------------------------------------------------------------------56 57OUTPUT_FIELDS = [58 "Lang_A",59 "Word_A",60 "IPA_A",61 "SCA_A",62 "Lang_B",63 "Word_B",64 "IPA_B",65 "SCA_B",66 "Concept_ID",67 "Label",68 "Phylo_Dist",69 "Timespan",70 "Score",71 "Source",72]73 74# ---------------------------------------------------------------------------75# Era classification76# ---------------------------------------------------------------------------77 78ANCIENT: set[str] = {79 "grc", "lat", "san", "ave", "got", "akk", "egy", "phn", "uga",80 "sux", "hit", "osc", "xum", "gmy", "sga", "chu", "och", "obr",81 "cop", "arc", "syc", "ett",82}83 84MEDIEVAL: set[str] = {85 "ang", "enm", "fro", "osp", "non", "goh", "dum", "mga", "wlm",86 "orv", "otk", "ota", "okm", "kaw", "mnc", "bod",87}88 89 90def classify_era(iso: str) -> str:91 """Return 'ancient', 'medieval', or 'modern'."""92 if iso in ANCIENT:93 return "ancient"94 if iso in MEDIEVAL:95 return "medieval"96 return "modern"97 98 99def get_timespan(iso_a: str, iso_b: str) -> str:100 """Return one of the four canonical timespan buckets."""101 era_a = classify_era(iso_a)102 era_b = classify_era(iso_b)103 eras = frozenset((era_a, era_b))104 if eras == {"ancient"}:105 return "ancient_ancient"106 if eras == {"modern"}:107 return "modern_modern"108 if eras == {"medieval"}:109 return "medieval_modern"110 if "ancient" in eras and "medieval" in eras:111 return "ancient_modern"112 if "ancient" in eras and "modern" in eras:113 return "ancient_modern"114 # medieval + modern115 return "medieval_modern"116 117 118# ---------------------------------------------------------------------------119# Religious concepts — organised by sub-domain (generous classification)120# ---------------------------------------------------------------------------121 122RELIGIOUS_CORE: set[str] = {123 "DEITY", "DEITY/GOD", "GOD", "SPIRIT", "TEMPLE", "ALTAR", "SACRIFICE",124 "WORSHIP", "PRAY", "PRIEST", "HOLY", "PREACH", "BLESS", "CHURCH",125 "MOSQUE", "SOUL", "RELIGION", "IDOL", "MINISTER",126}127 128RELIGIOUS_SUPERNATURAL: set[str] = {129 "GHOST", "DEMON", "MAGIC", "SORCERER", "MAGICIAN", "OMEN",130 "ELF OR FAIRY", "FAIRY TALE", "DREAM (SOMETHING)", "DREAM",131}132 133RELIGIOUS_MORAL: set[str] = {134 "SIN", "BAD OR EVIL", "EVIL", "BELIEVE", "TRUTH", "SHAME",135 "GUILTY", "CRIME", "ADULTERY", "PITY", "FORGIVE", "INNOCENT",136 "ACCUSE", "CONDEMN", "JUDGE", "JUDGMENT", "LAW", "PUNISHMENT",137 "WITNESS", "POOR", "RICH", "FAITHFUL", "GOOD",138}139 140RELIGIOUS_RITUAL: set[str] = {141 "CURSE", "FAST", "CIRCUMCISION", "INITIATION CEREMONY", "WEDDING",142 "OATH", "SWEAR", "CUSTOM", "BURY", "GRAVE", "CORPSE", "DANCE",143 "DRUM", "SONG",144}145 146RELIGIOUS_VERBS: set[str] = {147 "GIVE", "DONATE", "KNEEL", "BURN (SOMETHING)", "KILL", "POUR",148 "FEED", "SHARE", "INVITE", "COMMAND", "PROMISE", "OBEY",149 "HELP", "PROTECT", "DEFEND", "RESCUE", "HOPE (SOMETHING)",150 "FEAR (BE AFRAID)", "FEAR (FRIGHT)", "LOVE",151}152 153RELIGIOUS_COSMIC: set[str] = {154 "HEAVEN", "HELL", "LIGHTNING", "THUNDER", "FIRE", "SUN", "MOON",155 "STAR", "SKY", "RAINBOW", "EARTHQUAKE", "WORLD", "LIFE",156 "DEATH", "BE DEAD OR DIE", "BE ALIVE", "BE BORN",157 "ANCESTORS", "DESCENDANTS",158}159 160RELIGIOUS_PLACES: set[str] = {161 "TEMPLE", "CHURCH", "MOSQUE", "CAVE", "MOUNTAIN", "MOUNTAIN OR HILL",162 "SPRING OR WELL", "GARDEN", "FOREST", "STONE", "STONE OR ROCK",163 "VILLAGE", "TOWN", "COUNTRY", "ISLAND", "RIVER", "SEA",164 "NATIVE COUNTRY",165}166 167# Additional numeric IDs from cross-lingual datasets that map to religious concepts168_RELIGIOUS_NUMERIC_IDS: set[str] = {169 "3231", "53", "911", "853", "1103", "257", "24", "852", "1702", "304",170 "391", "8", "303", "1565", "878", "1973", "1945", "392", "2137", "1175",171 "107", "1349", "1603", "811", "2971", "661", "1944",172}173 174# Union of all sub-domains — the generous set175RELIGIOUS_ALL: set[str] = (176 RELIGIOUS_CORE177 | RELIGIOUS_SUPERNATURAL178 | RELIGIOUS_MORAL179 | RELIGIOUS_RITUAL180 | RELIGIOUS_VERBS181 | RELIGIOUS_COSMIC182 | RELIGIOUS_PLACES183 | _RELIGIOUS_NUMERIC_IDS184)185 186# Sub-domain name → concept set mapping (for generating sub-domain files)187RELIGIOUS_SUBDOMAINS: dict[str, set[str]] = {188 "core_religious": RELIGIOUS_CORE,189 "supernatural": RELIGIOUS_SUPERNATURAL,190 "moral_ethical": RELIGIOUS_MORAL,191 "ritual_ceremony": RELIGIOUS_RITUAL,192 "religious_verbs": RELIGIOUS_VERBS,193 "cosmic_spiritual": RELIGIOUS_COSMIC,194 "sacred_places": RELIGIOUS_PLACES,195}196 197# Pre-compute uppercase set for fast case-insensitive matching198_RELIGIOUS_ALL_UPPER: set[str] = {c.upper() for c in RELIGIOUS_ALL}199 200 201def is_religious(concept_id: str) -> bool:202 """Return True if *concept_id* refers to a religious concept."""203 if concept_id in RELIGIOUS_ALL:204 return True205 return concept_id.upper() in _RELIGIOUS_ALL_UPPER206 207 208def _in_subdomain(concept_id: str, subdomain_set: set[str]) -> bool:209 """Return True if *concept_id* belongs to the given sub-domain set."""210 if concept_id in subdomain_set:211 return True212 return concept_id.upper() in {c.upper() for c in subdomain_set}213 214 215# ---------------------------------------------------------------------------216# Top families217# ---------------------------------------------------------------------------218 219TOP_FAMILIES = [220 "germanic", "italic", "balto_slavic", "indo_iranian", "hellenic",221 "celtic", "uralic", "turkic", "sino_tibetan", "austronesian",222 "semitic", "dravidian", "japonic", "koreanic", "kartvelian",223]224 225# Map family_map values that differ from the tree's branch names226FAMILY_ALIAS = {227 "slavic": "balto_slavic",228 "baltic": "balto_slavic",229}230 231# ---------------------------------------------------------------------------232# SCA similarity (standalone, mirrors baseline_levenshtein.py)233# ---------------------------------------------------------------------------234 235_VOWELS = set("AEIOU")236_LABIALS = {"P", "B", "M"}237_CORONALS = {"T", "D", "N", "S", "L", "R"}238_VELARS = {"K", "G"}239_LARYNGEALS = {"H"}240_GLIDES = {"W", "Y"}241_NATURAL_CLASSES = [_VOWELS, _LABIALS, _CORONALS, _VELARS, _LARYNGEALS, _GLIDES]242 243 244def _substitution_cost(a: str, b: str) -> float:245 if a == b:246 return 0.0247 for cls in _NATURAL_CLASSES:248 if a in cls and b in cls:249 return 0.3250 return 1.0251 252 253def weighted_levenshtein(s1: str, s2: str) -> float:254 n, m = len(s1), len(s2)255 if n == 0:256 return m * 0.5257 if m == 0:258 return n * 0.5259 dp = [[0.0] * (m + 1) for _ in range(n + 1)]260 for i in range(n + 1):261 dp[i][0] = i * 0.5262 for j in range(m + 1):263 dp[0][j] = j * 0.5264 for i in range(1, n + 1):265 for j in range(1, m + 1):266 sub = _substitution_cost(s1[i - 1], s2[j - 1])267 dp[i][j] = min(268 dp[i - 1][j] + 0.5,269 dp[i][j - 1] + 0.5,270 dp[i - 1][j - 1] + sub,271 )272 return dp[n][m]273 274 275def normalised_similarity(s1: str, s2: str) -> float:276 if not s1 and not s2:277 return 1.0278 max_len = max(len(s1), len(s2))279 dist = weighted_levenshtein(s1, s2)280 return 1.0 - (dist / max_len) if max_len > 0 else 1.0281 282 283# ---------------------------------------------------------------------------284# Phylogenetic tree definition285# ---------------------------------------------------------------------------286 287def build_raw_tree() -> dict[str, Any]:288 """Return the hard-coded phylogenetic tree.289 290 Leaf-group values are either lists of ISO codes or the sentinel291 ``"__from_family_map__"`` which is resolved later.292 """293 return {294 "indo_european": {295 "germanic": {296 "west_germanic": {297 "anglo_frisian": ["eng", "ang", "enm", "fry", "frr", "ofs"],298 "franconian": ["nld", "dum", "lim", "afr"],299 "high_german": ["deu", "goh", "gsw", "bar", "ltz", "yid"],300 },301 "north_germanic": [302 "swe", "dan", "nor", "nno", "nob", "isl", "fao", "non",303 ],304 "east_germanic": ["got"],305 },306 "italic": {307 "romance": {308 "ibero_romance": ["spa", "por", "cat", "glg", "osp"],309 "gallo_romance": ["fra", "oci", "fro"],310 "italo_dalmatian": ["ita", "nap", "scn", "dlm", "cos"],311 "eastern_romance": ["ron", "rup"],312 },313 "latino_faliscan": ["lat", "osc", "xum"],314 },315 "celtic": {316 "goidelic": ["gle", "gla", "sga", "mga"],317 "brythonic": ["cym", "bre", "cor", "wlm"],318 },319 "balto_slavic": {320 "baltic": ["lit", "lav", "ltg"],321 "east_slavic": ["rus", "ukr", "bel", "orv"],322 "west_slavic": ["pol", "ces", "slk", "dsb", "hsb", "csb", "pox"],323 "south_slavic": ["bul", "mkd", "hrv", "slv", "hbs", "chu"],324 },325 "hellenic": ["ell", "grc", "gmy"],326 "indo_iranian": {327 "iranian": [328 "fas", "pes", "oss", "kmr", "ckb", "pbu", "tgk", "ave",329 "zza",330 ],331 "indic": [332 "hin", "ben", "san", "guj", "mar", "pan", "sin", "urd",333 "asm", "nep", "rom", "rmn",334 ],335 },336 "armenian": ["hye"],337 "albanian": ["sqi"],338 "anatolian": ["hit"],339 },340 "uralic": {341 "finnic": [342 "fin", "est", "ekk", "krl", "olo", "vep", "vot", "izh", "liv",343 ],344 "ugric": ["hun", "mns", "kca"],345 "samic": ["sme", "sma", "smj", "smn", "sms", "sjd"],346 "mordvinic": ["myv", "mdf"],347 "permic": ["kpv", "koi", "udm"],348 "mari": ["mhr", "mrj"],349 "samoyedic": ["yrk", "enf", "sel", "nio"],350 },351 "turkic": {352 "oghuz": ["tur", "aze", "azj", "ota", "otk"],353 "kipchak": ["kaz", "kir", "tat", "bak"],354 "siberian": ["sah", "tyv"],355 "karluk": ["uzb", "uzn"],356 "oghur": ["chv"],357 },358 "sino_tibetan": {359 "sinitic": ["zho", "cmn", "yue", "och"],360 "tibeto_burman": ["bod", "mya", "obr", "new", "lif"],361 },362 "austronesian": {363 "malayo_polynesian": "__from_family_map__",364 },365 "semitic": [366 "heb", "arb", "ara", "amh", "mlt", "syc", "arc", "akk", "phn",367 "uga",368 ],369 "dravidian": ["tam", "tel", "kan", "mal"],370 "japonic": ["jpn"],371 "koreanic": ["kor", "jje", "okm"],372 "kartvelian": ["kat", "lzz"],373 }374 375 376# ---------------------------------------------------------------------------377# Tree resolution helpers378# ---------------------------------------------------------------------------379 380def _collect_isos_from_tree(node: Any) -> set[str]:381 """Recursively collect all ISO codes that already appear in *node*."""382 if isinstance(node, list):383 return set(node)384 if isinstance(node, str):385 if node == "__from_family_map__":386 return set()387 return {node}388 isos: set[str] = set()389 for v in node.values():390 isos |= _collect_isos_from_tree(v)391 return isos392 393 394def resolve_tree(tree: dict[str, Any], family_map: dict[str, str]) -> dict[str, Any]:395 """Replace ``"__from_family_map__"`` sentinels and add catch-all groups.396 397 Returns a new tree (original is not mutated).398 """399 tree = _deep_copy_tree(tree)400 401 # Phase 1: resolve sentinels -----------------------------------------402 _resolve_sentinels(tree, family_map)403 404 # Phase 2: add catch-all for languages in family_map but not in tree --405 present = _collect_isos_from_tree(tree)406 extras: dict[str, list[str]] = defaultdict(list)407 for iso, fam in family_map.items():408 if iso in present:409 continue410 canonical = FAMILY_ALIAS.get(fam, fam)411 extras[canonical].append(iso)412 413 for fam, isos in extras.items():414 if fam not in tree:415 tree[fam] = sorted(isos)416 else:417 # Family exists as a top-level node — add under an418 # "other_{fam}" subgroup so we don't clobber existing structure.419 node = tree[fam]420 if isinstance(node, dict):421 existing = _collect_isos_from_tree(node)422 new_isos = [i for i in isos if i not in existing]423 if new_isos:424 node[f"other_{fam}"] = sorted(new_isos)425 elif isinstance(node, list):426 existing = set(node)427 for iso in isos:428 if iso not in existing:429 node.append(iso)430 # If the node is a single string, wrap it431 elif isinstance(node, str) and node != "__from_family_map__":432 tree[fam] = [node] + sorted(isos)433 434 return tree435 436 437def _deep_copy_tree(node: Any) -> Any:438 if isinstance(node, dict):439 return {k: _deep_copy_tree(v) for k, v in node.items()}440 if isinstance(node, list):441 return list(node)442 return node443 444 445def _resolve_sentinels(node: Any, family_map: dict[str, str]) -> None:446 """In-place replacement of ``"__from_family_map__"`` values."""447 if not isinstance(node, dict):448 return449 for key, val in list(node.items()):450 if val == "__from_family_map__":451 # key is the family name that should match family_map values452 # For "malayo_polynesian" under "austronesian", pull all453 # family_map entries mapped to "austronesian".454 parent_family = _find_parent_family(node, key)455 if parent_family is None:456 parent_family = key457 isos = sorted(458 iso for iso, fam in family_map.items() if fam == parent_family459 )460 node[key] = isos if isos else []461 elif isinstance(val, dict):462 _resolve_sentinels(val, family_map)463 464 465def _find_parent_family(node: dict, child_key: str) -> str | None: # noqa: ARG001466 """Heuristic: the sentinel is typically placed one level below the467 actual family name. Walk the raw tree keys for a match. For our468 tree, ``malayo_polynesian`` is under ``austronesian``, so we return469 ``austronesian``."""470 # We rely on the caller context; this is called from _resolve_sentinels471 # which walks the tree recursively. At the point we find the sentinel472 # the *node* dict is ``{"malayo_polynesian": "__from_family_map__"}``,473 # and we need the grandparent key. Since we don't track the parent key474 # inside the recursive walk, we use a simpler approach: just look up in475 # a mapping.476 _SENTINEL_PARENT: dict[str, str] = {477 "malayo_polynesian": "austronesian",478 }479 return _SENTINEL_PARENT.get(child_key)480 481 482# ---------------------------------------------------------------------------483# Language path index & phylo distance484# ---------------------------------------------------------------------------485 486def build_lang_paths(487 tree: dict[str, Any],488) -> dict[str, list[str]]:489 """Map each ISO code to its full path from root to its leaf group.490 491 For ``eng`` inside ``indo_european > germanic > west_germanic >492 anglo_frisian`` the path is493 ``["indo_european", "germanic", "west_germanic", "anglo_frisian"]``.494 """495 paths: dict[str, list[str]] = {}496 497 def _walk(node: Any, prefix: list[str]) -> None:498 if isinstance(node, list):499 for iso in node:500 paths[iso] = list(prefix)501 elif isinstance(node, dict):502 for key, child in node.items():503 _walk(child, prefix + [key])504 elif isinstance(node, str):505 # Single ISO as a leaf506 paths[node] = list(prefix)507 508 _walk(tree, [])509 return paths510 511 512def compute_distance(513 lang_a: str,514 lang_b: str,515 lang_paths: dict[str, list[str]],516) -> tuple[int, str]:517 """Return (edge_count, level_label) between two languages.518 519 Level mapping:520 L1 — same leaf group (same last path element)521 L2 — LCA is two levels above leaf522 L3 — LCA is deeper within same top-level family523 L4 — different top-level families or not in tree524 """525 pa = lang_paths.get(lang_a)526 pb = lang_paths.get(lang_b)527 if pa is None or pb is None:528 return (99, "L4")529 if not pa or not pb:530 return (99, "L4")531 532 # Find LCA depth533 lca_depth = 0534 for i, (a, b) in enumerate(zip(pa, pb)):535 if a != b:536 break537 lca_depth = i + 1538 else:539 # Exhausted the shorter path without mismatch540 lca_depth = min(len(pa), len(pb))541 542 if lca_depth == 0:543 # Different top-level families544 return (len(pa) + len(pb), "L4")545 546 depth_a = len(pa)547 depth_b = len(pb)548 edge_count = (depth_a - lca_depth) + (depth_b - lca_depth)549 550 # Determine level from the relationship between paths.551 #552 # L1 — same leaf group: paths are identical (both languages listed under553 # the same terminal node, e.g. both in "anglo_frisian").554 # L2 — same branch, different leaf group: LCA is the *parent* of the555 # leaf groups (e.g. eng in anglo_frisian, deu in high_german; LCA556 # is west_germanic) OR LCA is one more level up within the same557 # major branch. Concretely: the paths diverge at the last or558 # second-to-last position relative to the *shorter* path.559 # L3 — same top-level family, deeper divergence (e.g. eng vs fra both in560 # indo_european but in different major branches).561 # L4 — different top-level families (already handled above).562 563 if pa == pb:564 # Same leaf group565 return (edge_count, "L1")566 567 # Classification based on how deep the shared ancestry is:568 # L2 — same branch: they share at least 2 path elements569 # e.g. both under [indo_european, germanic, ...]570 # or both under [uralic, finnic, ...]571 # L3 — same top-level family but different branches:572 # they share only 1 path element (the super-family)573 # e.g. [indo_european, germanic, ...] vs [indo_european, italic, ...]574 if lca_depth >= 2:575 return (edge_count, "L2")576 # lca_depth == 1: same top-level family, different branches577 return (edge_count, "L3")578 579 580def get_top_family(iso: str, lang_paths: dict[str, list[str]], family_map: dict[str, str]) -> str:581 """Return the top-level family name for *iso*.582 583 Tries the tree path first; falls back to family_map.584 """585 path = lang_paths.get(iso)586 if path:587 # The first element is the top-level family node; but we want the588 # *branch* name that corresponds to TOP_FAMILIES. The branch is589 # typically the second element (e.g. "germanic" under "indo_european").590 # For top-level families like "uralic", the first element IS the family.591 if len(path) >= 2 and path[1] in _TOP_FAMILY_SET:592 return path[1]593 if path[0] in _TOP_FAMILY_SET:594 return path[0]595 # Search path for any matching family name596 for segment in path:597 if segment in _TOP_FAMILY_SET:598 return segment599 # Return the first path element as family600 return path[0] if path else "unknown"601 602 # Fallback: family_map603 fam = family_map.get(iso, "unknown")604 return FAMILY_ALIAS.get(fam, fam)605 606 607_TOP_FAMILY_SET = set(TOP_FAMILIES)608 609# ---------------------------------------------------------------------------610# Data loading611# ---------------------------------------------------------------------------612 613LexEntry = tuple[str, str, str] # (word, ipa, sca)614 615 616def load_lexicons(617 lexicons_dir: Path,618) -> dict[tuple[str, str], list[LexEntry]]:619 """Load all lexicon TSVs into ``{(iso, concept_id): [(word, ipa, sca), ...]}``.620 621 Entries with ``Concept_ID`` of ``"-"`` or empty are skipped.622 """623 lexicon: dict[tuple[str, str], list[LexEntry]] = defaultdict(list)624 files = sorted(lexicons_dir.glob("*.tsv"))625 total_entries = 0626 skipped_no_concept = 0627 628 for i, fp in enumerate(files, 1):629 iso = fp.stem630 if i % 100 == 0 or i == len(files):631 print(f" Loading lexicon {i}/{len(files)} ({iso}) ...")632 try:633 with fp.open(encoding="utf-8") as fh:634 reader = csv.DictReader(fh, delimiter="\t")635 for row in reader:636 cid = row.get("Concept_ID", "").strip()637 if cid in ("", "-"):638 skipped_no_concept += 1639 continue640 word = row.get("Word", "").strip()641 ipa = row.get("IPA", "").strip()642 sca = row.get("SCA", "").strip()643 if not word and not ipa and not sca:644 continue645 lexicon[(iso, cid)].append((word, ipa, sca))646 total_entries += 1647 except Exception as exc:648 print(f" WARNING: failed to read {fp.name}: {exc}", file=sys.stderr)649 650 print(f" Loaded {total_entries:,} entries across {len(files)} lexicons "651 f"(skipped {skipped_no_concept:,} without concept).")652 return lexicon653 654 655def build_concept_index(656 lexicon: dict[tuple[str, str], list[LexEntry]],657) -> dict[str, list[str]]:658 """Map each concept_id to the list of ISO codes that have entries for it."""659 concept_langs: dict[str, set[str]] = defaultdict(set)660 for (iso, cid) in lexicon:661 concept_langs[cid].add(iso)662 return {cid: sorted(isos) for cid, isos in concept_langs.items()}663 664 665def load_cognate_pairs(path: Path) -> list[dict[str, str]]:666 """Load a cognate-pairs TSV file into a list of dicts."""667 rows: list[dict[str, str]] = []668 if not path.exists():669 print(f" WARNING: {path} not found, skipping.")670 return rows671 with path.open(encoding="utf-8") as fh:672 reader = csv.DictReader(fh, delimiter="\t")673 for row in reader:674 rows.append(dict(row))675 print(f" Loaded {len(rows):,} pairs from {path.name}")676 return rows677 678 679# ---------------------------------------------------------------------------680# Pair record builder681# ---------------------------------------------------------------------------682 683def make_pair_record(684 lang_a: str,685 word_a: str,686 ipa_a: str,687 sca_a: str,688 lang_b: str,689 word_b: str,690 ipa_b: str,691 sca_b: str,692 concept_id: str,693 label: str,694 lang_paths: dict[str, list[str]],695 source: str = "lexicon",696 score_override: float | None = None,697) -> dict[str, str]:698 """Build a single output-row dict."""699 _, level = compute_distance(lang_a, lang_b, lang_paths)700 ts = get_timespan(lang_a, lang_b)701 if score_override is not None:702 score = score_override703 else:704 score = normalised_similarity(sca_a, sca_b) if sca_a and sca_b else 0.0705 return {706 "Lang_A": lang_a,707 "Word_A": word_a,708 "IPA_A": ipa_a,709 "SCA_A": sca_a,710 "Lang_B": lang_b,711 "Word_B": word_b,712 "IPA_B": ipa_b,713 "SCA_B": sca_b,714 "Concept_ID": concept_id,715 "Label": label,716 "Phylo_Dist": level,717 "Timespan": ts,718 "Score": f"{score:.4f}",719 "Source": source,720 }721 722 723# ---------------------------------------------------------------------------724# TSV writing helper725# ---------------------------------------------------------------------------726 727def write_pairs_tsv(path: Path, pairs: list[dict[str, str]]) -> None:728 """Write a list of pair dicts as a TSV file."""729 path.parent.mkdir(parents=True, exist_ok=True)730 with path.open("w", encoding="utf-8", newline="") as fh:731 writer = csv.DictWriter(fh, fieldnames=OUTPUT_FIELDS, delimiter="\t",732 extrasaction="ignore")733 writer.writeheader()734 for row in pairs:735 writer.writerow(row)736 print(f" Wrote {len(pairs):,} pairs to {path.relative_to(REPO_ROOT)}")737 738 739# ---------------------------------------------------------------------------740# Step 3: True cognate pair generation741# ---------------------------------------------------------------------------742 743def generate_true_cognates(744 lexicon: dict[tuple[str, str], list[LexEntry]],745 concept_langs: dict[str, list[str]],746 lang_paths: dict[str, list[str]],747 family_map: dict[str, str],748 inherited_pairs: list[dict[str, str]],749) -> tuple[list[dict[str, str]], list[dict[str, str]], list[dict[str, str]]]:750 """Generate true cognate pairs at L1, L2, L3 levels.751 752 Returns (l1_pairs, l2_pairs, l3_pairs).753 """754 l1: list[dict[str, str]] = []755 l2: list[dict[str, str]] = []756 l3: list[dict[str, str]] = []757 758 buckets = {"L1": l1, "L2": l2, "L3": l3}759 thresholds = {"L1": 0.5, "L2": 0.5, "L3": 0.4}760 761 # Step 3a: incorporate expert pairs FIRST (they have priority)762 print("Step 3a: Adding expert cognate pairs ...")763 expert_added = 0764 for row in inherited_pairs:765 if all(len(b) >= PAIR_CAP for b in buckets.values()):766 break767 768 lang_a = row.get("Lang_A", "")769 lang_b = row.get("Lang_B", "")770 if not lang_a or not lang_b:771 continue772 773 _, level = compute_distance(lang_a, lang_b, lang_paths)774 if level not in buckets or len(buckets[level]) >= PAIR_CAP:775 continue776 777 cid = row.get("Concept_ID", "")778 ipa_a = row.get("IPA_A", "")779 ipa_b = row.get("IPA_B", "")780 word_a = row.get("Word_A", "")781 word_b = row.get("Word_B", "")782 sca_a = _lookup_sca(lexicon, lang_a, cid, word_a, ipa_a)783 sca_b = _lookup_sca(lexicon, lang_b, cid, word_b, ipa_b)784 785 score_str = row.get("Score", "0")786 try:787 score = float(score_str)788 except (ValueError, TypeError):789 score = 0.0790 791 if sca_a and sca_b:792 sca_sim = normalised_similarity(sca_a, sca_b)793 else:794 sca_sim = score795 796 rec = make_pair_record(797 lang_a, word_a, ipa_a, sca_a,798 lang_b, word_b, ipa_b, sca_b,799 cid, "true_cognate", lang_paths,800 source=row.get("Source", "expert"),801 score_override=sca_sim,802 )803 rec["Phylo_Dist"] = level804 buckets[level].append(rec)805 expert_added += 1806 807 print(f" Added {expert_added:,} expert pairs "808 f"(L1={len(l1):,} L2={len(l2):,} L3={len(l3):,})")809 810 # Step 3b: Fill remaining slots from lexicon data811 print("Step 3b: Generating true cognates from lexicon data ...")812 813 concepts_processed = 0814 total_concepts = len(concept_langs)815 816 for cid, langs in concept_langs.items():817 concepts_processed += 1818 if concepts_processed % 500 == 0:819 print(f" Concept {concepts_processed}/{total_concepts} "820 f"(L1={len(l1):,} L2={len(l2):,} L3={len(l3):,})")821 822 # Early termination823 if all(len(b) >= PAIR_CAP for b in buckets.values()):824 break825 826 # Group languages by top-level family827 family_groups: dict[str, list[str]] = defaultdict(list)828 for iso in langs:829 fam = get_top_family(iso, lang_paths, family_map)830 family_groups[fam].append(iso)831 832 # Within each family, generate pairs833 for fam, fam_langs in family_groups.items():834 if len(fam_langs) < 2:835 continue836 837 # Sample pairs if too many languages838 if len(fam_langs) > 50:839 sampled = random.sample(fam_langs, 50)840 else:841 sampled = fam_langs842 843 pair_count_this_concept: dict[str, int] = {"L1": 0, "L2": 0, "L3": 0}844 845 for i in range(len(sampled)):846 for j in range(i + 1, len(sampled)):847 iso_a = sampled[i]848 iso_b = sampled[j]849 if iso_a == iso_b:850 continue851 852 _, level = compute_distance(iso_a, iso_b, lang_paths)853 if level not in buckets or len(buckets[level]) >= PAIR_CAP:854 continue855 if pair_count_this_concept.get(level, 0) >= MAX_PAIRS_PER_CONCEPT_PER_LEVEL:856 continue857 858 thresh = thresholds[level]859 entries_a = lexicon.get((iso_a, cid), [])860 entries_b = lexicon.get((iso_b, cid), [])861 if not entries_a or not entries_b:862 continue863 864 # Pick one entry from each language (first with SCA)865 ea = _pick_best_entry(entries_a)866 eb = _pick_best_entry(entries_b)867 if ea is None or eb is None:868 continue869 870 sca_sim = normalised_similarity(ea[2], eb[2]) if ea[2] and eb[2] else 0.0871 if sca_sim < thresh:872 continue873 874 rec = make_pair_record(875 iso_a, ea[0], ea[1], ea[2],876 iso_b, eb[0], eb[1], eb[2],877 cid, "true_cognate", lang_paths,878 source="lexicon",879 score_override=sca_sim,880 )881 rec["Phylo_Dist"] = level # ensure correct level882 buckets[level].append(rec)883 pair_count_this_concept[level] = pair_count_this_concept.get(level, 0) + 1884 885 print(f" From lexicon: L1={len(l1):,} L2={len(l2):,} L3={len(l3):,}")886 887 # Cap each bucket888 for level_name, bucket in buckets.items():889 if len(bucket) > PAIR_CAP:890 random.shuffle(bucket)891 buckets[level_name] = bucket[:PAIR_CAP]892 if level_name == "L1":893 l1[:] = buckets[level_name]894 elif level_name == "L2":895 l2[:] = buckets[level_name]896 else:897 l3[:] = buckets[level_name]898 899 print(f" Final: L1={len(l1):,} L2={len(l2):,} L3={len(l3):,}")900 return l1, l2, l3901 902 903def _pick_best_entry(entries: list[LexEntry]) -> LexEntry | None:904 """Pick an entry preferring ones with non-empty SCA."""905 for e in entries:906 if e[2]: # has SCA907 return e908 return entries[0] if entries else None909 910 911def _lookup_sca(912 lexicon: dict[tuple[str, str], list[LexEntry]],913 iso: str,914 concept_id: str,915 word: str,916 ipa: str = "",917) -> str:918 """Try to find the SCA encoding for a given word from the lexicon.919 920 Falls back to computing SCA on-the-fly from IPA if all lookups fail.921 """922 entries = lexicon.get((iso, concept_id), [])923 # Exact word match first924 for e in entries:925 if e[0] == word and e[2]:926 return e[2]927 # Any entry with SCA for this (iso, concept)928 for e in entries:929 if e[2]:930 return e[2]931 # Use the word index for cross-concept fallback (fast)932 sca = _word_sca_index.get((iso, word), "")933 if sca:934 return sca935 # Last resort: compute from IPA on-the-fly936 if ipa:937 return ipa_to_sound_class(ipa)938 return ""939 940 941# ---------------------------------------------------------------------------942# Step 4: False positives943# ---------------------------------------------------------------------------944 945def generate_false_positives(946 lexicon: dict[tuple[str, str], list[LexEntry]],947 concept_langs: dict[str, list[str]],948 lang_paths: dict[str, list[str]],949 family_map: dict[str, str],950 similarity_pairs: list[dict[str, str]],951) -> list[dict[str, str]]:952 """Cross-family pairs with same concept and SCA similarity >= 0.5."""953 print("Step 4: Generating false positives ...")954 fps: list[dict[str, str]] = []955 956 # 4a: From lexicon — cross-family same-concept pairs957 concepts_processed = 0958 for cid, langs in concept_langs.items():959 if len(fps) >= PAIR_CAP:960 break961 concepts_processed += 1962 if concepts_processed % 500 == 0:963 print(f" Concept {concepts_processed}/{len(concept_langs)} "964 f"(fps={len(fps):,})")965 966 # Group by family967 family_groups: dict[str, list[str]] = defaultdict(list)968 for iso in langs:969 fam = get_top_family(iso, lang_paths, family_map)970 family_groups[fam].append(iso)971 972 families = list(family_groups.keys())973 if len(families) < 2:974 continue975 976 # Cross-family pairs977 pair_count = 0978 for fi in range(len(families)):979 for fj in range(fi + 1, len(families)):980 if len(fps) >= PAIR_CAP:981 break982 if pair_count >= MAX_CROSS_FAMILY_PAIRS_PER_CONCEPT:983 break984 985 fam_a_langs = family_groups[families[fi]]986 fam_b_langs = family_groups[families[fj]]987 988 # Sample one from each989 iso_a = random.choice(fam_a_langs)990 iso_b = random.choice(fam_b_langs)991 992 entries_a = lexicon.get((iso_a, cid), [])993 entries_b = lexicon.get((iso_b, cid), [])994 if not entries_a or not entries_b:995 continue996 997 ea = _pick_best_entry(entries_a)998 eb = _pick_best_entry(entries_b)999 if ea is None or eb is None:1000 continue1001 1002 sca_sim = normalised_similarity(ea[2], eb[2]) if ea[2] and eb[2] else 0.01003 if sca_sim < 0.5:1004 continue1005 1006 rec = make_pair_record(1007 iso_a, ea[0], ea[1], ea[2],1008 iso_b, eb[0], eb[1], eb[2],1009 cid, "false_positive", lang_paths,1010 source="lexicon",1011 score_override=sca_sim,1012 )1013 fps.append(rec)1014 pair_count += 11015 1016 print(f" From lexicon: {len(fps):,} false positives")1017 1018 # 4b: From similarity_pairs file1019 sim_added = 01020 for row in similarity_pairs:1021 if len(fps) >= PAIR_CAP:1022 break1023 lang_a = row.get("Lang_A", "")1024 lang_b = row.get("Lang_B", "")1025 if not lang_a or not lang_b:1026 continue1027 1028 fam_a = get_top_family(lang_a, lang_paths, family_map)1029 fam_b = get_top_family(lang_b, lang_paths, family_map)1030 if fam_a == fam_b:1031 continue # only want cross-family1032 1033 cid = row.get("Concept_ID", "")1034 sca_a = _lookup_sca(lexicon, lang_a, cid, row.get("Word_A", ""),1035 row.get("IPA_A", ""))1036 sca_b = _lookup_sca(lexicon, lang_b, cid, row.get("Word_B", ""),1037 row.get("IPA_B", ""))1038 1039 score_str = row.get("Score", "0")1040 try:1041 score = float(score_str)1042 except (ValueError, TypeError):1043 score = 0.01044 1045 if sca_a and sca_b:1046 sca_sim = normalised_similarity(sca_a, sca_b)1047 else:1048 sca_sim = score1049 1050 if sca_sim < 0.5:1051 continue1052 1053 rec = make_pair_record(1054 lang_a, row.get("Word_A", ""), row.get("IPA_A", ""), sca_a,1055 lang_b, row.get("Word_B", ""), row.get("IPA_B", ""), sca_b,1056 cid, "false_positive", lang_paths,1057 source=row.get("Source", "similarity"),1058 score_override=sca_sim,1059 )1060 fps.append(rec)1061 sim_added += 11062 1063 print(f" Added {sim_added:,} from similarity pairs")1064 1065 if len(fps) > PAIR_CAP:1066 random.shuffle(fps)1067 fps = fps[:PAIR_CAP]1068 1069 print(f" Final false positives: {len(fps):,}")1070 return fps1071 1072 1073# ---------------------------------------------------------------------------1074# Step 5: True negatives1075# ---------------------------------------------------------------------------1076 1077def generate_true_negatives(1078 lexicon: dict[tuple[str, str], list[LexEntry]],1079 lang_paths: dict[str, list[str]],1080 family_map: dict[str, str],1081) -> list[dict[str, str]]:1082 """Random cross-family pairs with different concepts and SCA sim < 0.3."""1083 print("Step 5: Generating true negatives ...")1084 negs: list[dict[str, str]] = []1085 1086 # Build index of all (iso, concept_id) keys for sampling1087 all_keys = list(lexicon.keys())1088 if len(all_keys) < 2:1089 print(" Not enough lexicon entries for true negatives.")1090 return negs1091 1092 attempts = 01093 while len(negs) < PAIR_CAP and attempts < TRUE_NEG_SAMPLE_ATTEMPTS:1094 attempts += 11095 if attempts % 50_000 == 0:1096 print(f" Attempt {attempts:,}, negatives so far: {len(negs):,}")1097 1098 # Pick two random entries1099 key_a = random.choice(all_keys)1100 key_b = random.choice(all_keys)1101 iso_a, cid_a = key_a1102 iso_b, cid_b = key_b1103 1104 # Must be different concepts1105 if cid_a == cid_b:1106 continue1107 # Must be different families1108 fam_a = get_top_family(iso_a, lang_paths, family_map)1109 fam_b = get_top_family(iso_b, lang_paths, family_map)1110 if fam_a == fam_b:1111 continue1112 # Must be different languages1113 if iso_a == iso_b:1114 continue1115 1116 entries_a = lexicon[key_a]1117 entries_b = lexicon[key_b]1118 ea = _pick_best_entry(entries_a)1119 eb = _pick_best_entry(entries_b)1120 if ea is None or eb is None:1121 continue1122 if not ea[2] or not eb[2]:1123 continue1124 1125 sca_sim = normalised_similarity(ea[2], eb[2])1126 if sca_sim >= 0.4:1127 continue1128 1129 # Use concept_a for the record (arbitrary; both concepts are different)1130 rec = make_pair_record(1131 iso_a, ea[0], ea[1], ea[2],1132 iso_b, eb[0], eb[1], eb[2],1133 f"{cid_a} / {cid_b}", "true_negative", lang_paths,1134 source="random_sample",1135 score_override=sca_sim,1136 )1137 negs.append(rec)1138 1139 print(f" Generated {len(negs):,} true negatives in {attempts:,} attempts")1140 return negs1141 1142 1143# ---------------------------------------------------------------------------1144# Step 6: Borrowings1145# ---------------------------------------------------------------------------1146 1147def generate_borrowings(1148 borrowing_pairs: list[dict[str, str]],1149 lexicon: dict[tuple[str, str], list[LexEntry]],1150 lang_paths: dict[str, list[str]],1151) -> list[dict[str, str]]:1152 """Process borrowing pairs from WOLD."""1153 print("Step 6: Processing borrowing pairs ...")1154 rows: list[dict[str, str]] = []1155 1156 for row in borrowing_pairs:1157 lang_a = row.get("Lang_A", "")1158 lang_b = row.get("Lang_B", "")1159 if not lang_a or not lang_b:1160 continue1161 1162 cid = row.get("Concept_ID", "")1163 word_a = row.get("Word_A", "")1164 word_b = row.get("Word_B", "")1165 ipa_a = row.get("IPA_A", "")1166 ipa_b = row.get("IPA_B", "")1167 1168 sca_a = _lookup_sca(lexicon, lang_a, cid, word_a, ipa_a)1169 sca_b = _lookup_sca(lexicon, lang_b, cid, word_b, ipa_b)1170 1171 score_str = row.get("Score", "0")1172 try:1173 score = float(score_str)1174 except (ValueError, TypeError):1175 score = 0.01176 1177 if sca_a and sca_b:1178 sca_sim = normalised_similarity(sca_a, sca_b)1179 else:1180 sca_sim = score1181 1182 rec = make_pair_record(1183 lang_a, word_a, ipa_a, sca_a,1184 lang_b, word_b, ipa_b, sca_b,1185 cid, "borrowing", lang_paths,1186 source=row.get("Source", "wold"),1187 score_override=sca_sim,1188 )1189 rows.append(rec)1190 1191 print(f" Processed {len(rows):,} borrowing pairs")1192 return rows1193 1194 1195# ---------------------------------------------------------------------------1196# Step 7: Religious subsets1197# ---------------------------------------------------------------------------1198 1199def generate_religious_pairs(1200 all_pairs: list[dict[str, str]],