Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
build_validation_sets.py1729 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Build stratified validation datasets for cognate detection ML training.3 4Reads lexicon TSVs and cognate-pair TSVs from data/training/, builds a5phylogenetic tree of language relationships, and generates stratified6validation sets split by phylogenetic distance, timespan, family, and7concept domain (religious terms).8 9Output goes to data/training/validation/.10"""11 12from __future__ import annotations13 14import csv15import json16import random17import sys18from collections import defaultdict19from pathlib import Path20from typing import Any21 22# Ensure cognate_pipeline package is importable23sys.path.insert(24    0,25    str(Path(__file__).resolve().parent.parent / "cognate_pipeline" / "src"),26)27from cognate_pipeline.normalise.sound_class import ipa_to_sound_class28 29# ---------------------------------------------------------------------------30# Paths31# ---------------------------------------------------------------------------32 33REPO_ROOT = Path(__file__).resolve().parent.parent34TRAINING_DIR = REPO_ROOT / "data" / "training"35LEXICONS_DIR = TRAINING_DIR / "lexicons"36COGNATE_DIR = TRAINING_DIR / "cognate_pairs"37OUTPUT_DIR = TRAINING_DIR / "validation"38FAMILY_MAP_PATH = (39    REPO_ROOT40    / "cognate_pipeline"41    / "src"42    / "cognate_pipeline"43    / "cognate"44    / "family_map.json"45)46 47PAIR_CAP = 50_00048SEED = 4249MAX_PAIRS_PER_CONCEPT_PER_LEVEL = 10050MAX_CROSS_FAMILY_PAIRS_PER_CONCEPT = 5051TRUE_NEG_SAMPLE_ATTEMPTS = 2_000_00052 53# ---------------------------------------------------------------------------54# TSV field names for output55# ---------------------------------------------------------------------------56 57OUTPUT_FIELDS = [58    "Lang_A",59    "Word_A",60    "IPA_A",61    "SCA_A",62    "Lang_B",63    "Word_B",64    "IPA_B",65    "SCA_B",66    "Concept_ID",67    "Label",68    "Phylo_Dist",69    "Timespan",70    "Score",71    "Source",72]73 74# ---------------------------------------------------------------------------75# Era classification76# ---------------------------------------------------------------------------77 78ANCIENT: set[str] = {79    "grc", "lat", "san", "ave", "got", "akk", "egy", "phn", "uga",80    "sux", "hit", "osc", "xum", "gmy", "sga", "chu", "och", "obr",81    "cop", "arc", "syc", "ett",82}83 84MEDIEVAL: set[str] = {85    "ang", "enm", "fro", "osp", "non", "goh", "dum", "mga", "wlm",86    "orv", "otk", "ota", "okm", "kaw", "mnc", "bod",87}88 89 90def classify_era(iso: str) -> str:91    """Return 'ancient', 'medieval', or 'modern'."""92    if iso in ANCIENT:93        return "ancient"94    if iso in MEDIEVAL:95        return "medieval"96    return "modern"97 98 99def get_timespan(iso_a: str, iso_b: str) -> str:100    """Return one of the four canonical timespan buckets."""101    era_a = classify_era(iso_a)102    era_b = classify_era(iso_b)103    eras = frozenset((era_a, era_b))104    if eras == {"ancient"}:105        return "ancient_ancient"106    if eras == {"modern"}:107        return "modern_modern"108    if eras == {"medieval"}:109        return "medieval_modern"110    if "ancient" in eras and "medieval" in eras:111        return "ancient_modern"112    if "ancient" in eras and "modern" in eras:113        return "ancient_modern"114    # medieval + modern115    return "medieval_modern"116 117 118# ---------------------------------------------------------------------------119# Religious concepts — organised by sub-domain (generous classification)120# ---------------------------------------------------------------------------121 122RELIGIOUS_CORE: set[str] = {123    "DEITY", "DEITY/GOD", "GOD", "SPIRIT", "TEMPLE", "ALTAR", "SACRIFICE",124    "WORSHIP", "PRAY", "PRIEST", "HOLY", "PREACH", "BLESS", "CHURCH",125    "MOSQUE", "SOUL", "RELIGION", "IDOL", "MINISTER",126}127 128RELIGIOUS_SUPERNATURAL: set[str] = {129    "GHOST", "DEMON", "MAGIC", "SORCERER", "MAGICIAN", "OMEN",130    "ELF OR FAIRY", "FAIRY TALE", "DREAM (SOMETHING)", "DREAM",131}132 133RELIGIOUS_MORAL: set[str] = {134    "SIN", "BAD OR EVIL", "EVIL", "BELIEVE", "TRUTH", "SHAME",135    "GUILTY", "CRIME", "ADULTERY", "PITY", "FORGIVE", "INNOCENT",136    "ACCUSE", "CONDEMN", "JUDGE", "JUDGMENT", "LAW", "PUNISHMENT",137    "WITNESS", "POOR", "RICH", "FAITHFUL", "GOOD",138}139 140RELIGIOUS_RITUAL: set[str] = {141    "CURSE", "FAST", "CIRCUMCISION", "INITIATION CEREMONY", "WEDDING",142    "OATH", "SWEAR", "CUSTOM", "BURY", "GRAVE", "CORPSE", "DANCE",143    "DRUM", "SONG",144}145 146RELIGIOUS_VERBS: set[str] = {147    "GIVE", "DONATE", "KNEEL", "BURN (SOMETHING)", "KILL", "POUR",148    "FEED", "SHARE", "INVITE", "COMMAND", "PROMISE", "OBEY",149    "HELP", "PROTECT", "DEFEND", "RESCUE", "HOPE (SOMETHING)",150    "FEAR (BE AFRAID)", "FEAR (FRIGHT)", "LOVE",151}152 153RELIGIOUS_COSMIC: set[str] = {154    "HEAVEN", "HELL", "LIGHTNING", "THUNDER", "FIRE", "SUN", "MOON",155    "STAR", "SKY", "RAINBOW", "EARTHQUAKE", "WORLD", "LIFE",156    "DEATH", "BE DEAD OR DIE", "BE ALIVE", "BE BORN",157    "ANCESTORS", "DESCENDANTS",158}159 160RELIGIOUS_PLACES: set[str] = {161    "TEMPLE", "CHURCH", "MOSQUE", "CAVE", "MOUNTAIN", "MOUNTAIN OR HILL",162    "SPRING OR WELL", "GARDEN", "FOREST", "STONE", "STONE OR ROCK",163    "VILLAGE", "TOWN", "COUNTRY", "ISLAND", "RIVER", "SEA",164    "NATIVE COUNTRY",165}166 167# Additional numeric IDs from cross-lingual datasets that map to religious concepts168_RELIGIOUS_NUMERIC_IDS: set[str] = {169    "3231", "53", "911", "853", "1103", "257", "24", "852", "1702", "304",170    "391", "8", "303", "1565", "878", "1973", "1945", "392", "2137", "1175",171    "107", "1349", "1603", "811", "2971", "661", "1944",172}173 174# Union of all sub-domains — the generous set175RELIGIOUS_ALL: set[str] = (176    RELIGIOUS_CORE177    | RELIGIOUS_SUPERNATURAL178    | RELIGIOUS_MORAL179    | RELIGIOUS_RITUAL180    | RELIGIOUS_VERBS181    | RELIGIOUS_COSMIC182    | RELIGIOUS_PLACES183    | _RELIGIOUS_NUMERIC_IDS184)185 186# Sub-domain name → concept set mapping (for generating sub-domain files)187RELIGIOUS_SUBDOMAINS: dict[str, set[str]] = {188    "core_religious": RELIGIOUS_CORE,189    "supernatural": RELIGIOUS_SUPERNATURAL,190    "moral_ethical": RELIGIOUS_MORAL,191    "ritual_ceremony": RELIGIOUS_RITUAL,192    "religious_verbs": RELIGIOUS_VERBS,193    "cosmic_spiritual": RELIGIOUS_COSMIC,194    "sacred_places": RELIGIOUS_PLACES,195}196 197# Pre-compute uppercase set for fast case-insensitive matching198_RELIGIOUS_ALL_UPPER: set[str] = {c.upper() for c in RELIGIOUS_ALL}199 200 201def is_religious(concept_id: str) -> bool:202    """Return True if *concept_id* refers to a religious concept."""203    if concept_id in RELIGIOUS_ALL:204        return True205    return concept_id.upper() in _RELIGIOUS_ALL_UPPER206 207 208def _in_subdomain(concept_id: str, subdomain_set: set[str]) -> bool:209    """Return True if *concept_id* belongs to the given sub-domain set."""210    if concept_id in subdomain_set:211        return True212    return concept_id.upper() in {c.upper() for c in subdomain_set}213 214 215# ---------------------------------------------------------------------------216# Top families217# ---------------------------------------------------------------------------218 219TOP_FAMILIES = [220    "germanic", "italic", "balto_slavic", "indo_iranian", "hellenic",221    "celtic", "uralic", "turkic", "sino_tibetan", "austronesian",222    "semitic", "dravidian", "japonic", "koreanic", "kartvelian",223]224 225# Map family_map values that differ from the tree's branch names226FAMILY_ALIAS = {227    "slavic": "balto_slavic",228    "baltic": "balto_slavic",229}230 231# ---------------------------------------------------------------------------232# SCA similarity (standalone, mirrors baseline_levenshtein.py)233# ---------------------------------------------------------------------------234 235_VOWELS = set("AEIOU")236_LABIALS = {"P", "B", "M"}237_CORONALS = {"T", "D", "N", "S", "L", "R"}238_VELARS = {"K", "G"}239_LARYNGEALS = {"H"}240_GLIDES = {"W", "Y"}241_NATURAL_CLASSES = [_VOWELS, _LABIALS, _CORONALS, _VELARS, _LARYNGEALS, _GLIDES]242 243 244def _substitution_cost(a: str, b: str) -> float:245    if a == b:246        return 0.0247    for cls in _NATURAL_CLASSES:248        if a in cls and b in cls:249            return 0.3250    return 1.0251 252 253def weighted_levenshtein(s1: str, s2: str) -> float:254    n, m = len(s1), len(s2)255    if n == 0:256        return m * 0.5257    if m == 0:258        return n * 0.5259    dp = [[0.0] * (m + 1) for _ in range(n + 1)]260    for i in range(n + 1):261        dp[i][0] = i * 0.5262    for j in range(m + 1):263        dp[0][j] = j * 0.5264    for i in range(1, n + 1):265        for j in range(1, m + 1):266            sub = _substitution_cost(s1[i - 1], s2[j - 1])267            dp[i][j] = min(268                dp[i - 1][j] + 0.5,269                dp[i][j - 1] + 0.5,270                dp[i - 1][j - 1] + sub,271            )272    return dp[n][m]273 274 275def normalised_similarity(s1: str, s2: str) -> float:276    if not s1 and not s2:277        return 1.0278    max_len = max(len(s1), len(s2))279    dist = weighted_levenshtein(s1, s2)280    return 1.0 - (dist / max_len) if max_len > 0 else 1.0281 282 283# ---------------------------------------------------------------------------284# Phylogenetic tree definition285# ---------------------------------------------------------------------------286 287def build_raw_tree() -> dict[str, Any]:288    """Return the hard-coded phylogenetic tree.289 290    Leaf-group values are either lists of ISO codes or the sentinel291    ``"__from_family_map__"`` which is resolved later.292    """293    return {294        "indo_european": {295            "germanic": {296                "west_germanic": {297                    "anglo_frisian": ["eng", "ang", "enm", "fry", "frr", "ofs"],298                    "franconian": ["nld", "dum", "lim", "afr"],299                    "high_german": ["deu", "goh", "gsw", "bar", "ltz", "yid"],300                },301                "north_germanic": [302                    "swe", "dan", "nor", "nno", "nob", "isl", "fao", "non",303                ],304                "east_germanic": ["got"],305            },306            "italic": {307                "romance": {308                    "ibero_romance": ["spa", "por", "cat", "glg", "osp"],309                    "gallo_romance": ["fra", "oci", "fro"],310                    "italo_dalmatian": ["ita", "nap", "scn", "dlm", "cos"],311                    "eastern_romance": ["ron", "rup"],312                },313                "latino_faliscan": ["lat", "osc", "xum"],314            },315            "celtic": {316                "goidelic": ["gle", "gla", "sga", "mga"],317                "brythonic": ["cym", "bre", "cor", "wlm"],318            },319            "balto_slavic": {320                "baltic": ["lit", "lav", "ltg"],321                "east_slavic": ["rus", "ukr", "bel", "orv"],322                "west_slavic": ["pol", "ces", "slk", "dsb", "hsb", "csb", "pox"],323                "south_slavic": ["bul", "mkd", "hrv", "slv", "hbs", "chu"],324            },325            "hellenic": ["ell", "grc", "gmy"],326            "indo_iranian": {327                "iranian": [328                    "fas", "pes", "oss", "kmr", "ckb", "pbu", "tgk", "ave",329                    "zza",330                ],331                "indic": [332                    "hin", "ben", "san", "guj", "mar", "pan", "sin", "urd",333                    "asm", "nep", "rom", "rmn",334                ],335            },336            "armenian": ["hye"],337            "albanian": ["sqi"],338            "anatolian": ["hit"],339        },340        "uralic": {341            "finnic": [342                "fin", "est", "ekk", "krl", "olo", "vep", "vot", "izh", "liv",343            ],344            "ugric": ["hun", "mns", "kca"],345            "samic": ["sme", "sma", "smj", "smn", "sms", "sjd"],346            "mordvinic": ["myv", "mdf"],347            "permic": ["kpv", "koi", "udm"],348            "mari": ["mhr", "mrj"],349            "samoyedic": ["yrk", "enf", "sel", "nio"],350        },351        "turkic": {352            "oghuz": ["tur", "aze", "azj", "ota", "otk"],353            "kipchak": ["kaz", "kir", "tat", "bak"],354            "siberian": ["sah", "tyv"],355            "karluk": ["uzb", "uzn"],356            "oghur": ["chv"],357        },358        "sino_tibetan": {359            "sinitic": ["zho", "cmn", "yue", "och"],360            "tibeto_burman": ["bod", "mya", "obr", "new", "lif"],361        },362        "austronesian": {363            "malayo_polynesian": "__from_family_map__",364        },365        "semitic": [366            "heb", "arb", "ara", "amh", "mlt", "syc", "arc", "akk", "phn",367            "uga",368        ],369        "dravidian": ["tam", "tel", "kan", "mal"],370        "japonic": ["jpn"],371        "koreanic": ["kor", "jje", "okm"],372        "kartvelian": ["kat", "lzz"],373    }374 375 376# ---------------------------------------------------------------------------377# Tree resolution helpers378# ---------------------------------------------------------------------------379 380def _collect_isos_from_tree(node: Any) -> set[str]:381    """Recursively collect all ISO codes that already appear in *node*."""382    if isinstance(node, list):383        return set(node)384    if isinstance(node, str):385        if node == "__from_family_map__":386            return set()387        return {node}388    isos: set[str] = set()389    for v in node.values():390        isos |= _collect_isos_from_tree(v)391    return isos392 393 394def resolve_tree(tree: dict[str, Any], family_map: dict[str, str]) -> dict[str, Any]:395    """Replace ``"__from_family_map__"`` sentinels and add catch-all groups.396 397    Returns a new tree (original is not mutated).398    """399    tree = _deep_copy_tree(tree)400 401    # Phase 1: resolve sentinels -----------------------------------------402    _resolve_sentinels(tree, family_map)403 404    # Phase 2: add catch-all for languages in family_map but not in tree --405    present = _collect_isos_from_tree(tree)406    extras: dict[str, list[str]] = defaultdict(list)407    for iso, fam in family_map.items():408        if iso in present:409            continue410        canonical = FAMILY_ALIAS.get(fam, fam)411        extras[canonical].append(iso)412 413    for fam, isos in extras.items():414        if fam not in tree:415            tree[fam] = sorted(isos)416        else:417            # Family exists as a top-level node — add under an418            # "other_{fam}" subgroup so we don't clobber existing structure.419            node = tree[fam]420            if isinstance(node, dict):421                existing = _collect_isos_from_tree(node)422                new_isos = [i for i in isos if i not in existing]423                if new_isos:424                    node[f"other_{fam}"] = sorted(new_isos)425            elif isinstance(node, list):426                existing = set(node)427                for iso in isos:428                    if iso not in existing:429                        node.append(iso)430            # If the node is a single string, wrap it431            elif isinstance(node, str) and node != "__from_family_map__":432                tree[fam] = [node] + sorted(isos)433 434    return tree435 436 437def _deep_copy_tree(node: Any) -> Any:438    if isinstance(node, dict):439        return {k: _deep_copy_tree(v) for k, v in node.items()}440    if isinstance(node, list):441        return list(node)442    return node443 444 445def _resolve_sentinels(node: Any, family_map: dict[str, str]) -> None:446    """In-place replacement of ``"__from_family_map__"`` values."""447    if not isinstance(node, dict):448        return449    for key, val in list(node.items()):450        if val == "__from_family_map__":451            # key is the family name that should match family_map values452            # For "malayo_polynesian" under "austronesian", pull all453            # family_map entries mapped to "austronesian".454            parent_family = _find_parent_family(node, key)455            if parent_family is None:456                parent_family = key457            isos = sorted(458                iso for iso, fam in family_map.items() if fam == parent_family459            )460            node[key] = isos if isos else []461        elif isinstance(val, dict):462            _resolve_sentinels(val, family_map)463 464 465def _find_parent_family(node: dict, child_key: str) -> str | None:  # noqa: ARG001466    """Heuristic: the sentinel is typically placed one level below the467    actual family name.  Walk the raw tree keys for a match.  For our468    tree, ``malayo_polynesian`` is under ``austronesian``, so we return469    ``austronesian``."""470    # We rely on the caller context; this is called from _resolve_sentinels471    # which walks the tree recursively.  At the point we find the sentinel472    # the *node* dict is ``{"malayo_polynesian": "__from_family_map__"}``,473    # and we need the grandparent key.  Since we don't track the parent key474    # inside the recursive walk, we use a simpler approach: just look up in475    # a mapping.476    _SENTINEL_PARENT: dict[str, str] = {477        "malayo_polynesian": "austronesian",478    }479    return _SENTINEL_PARENT.get(child_key)480 481 482# ---------------------------------------------------------------------------483# Language path index & phylo distance484# ---------------------------------------------------------------------------485 486def build_lang_paths(487    tree: dict[str, Any],488) -> dict[str, list[str]]:489    """Map each ISO code to its full path from root to its leaf group.490 491    For ``eng`` inside ``indo_european > germanic > west_germanic >492    anglo_frisian`` the path is493    ``["indo_european", "germanic", "west_germanic", "anglo_frisian"]``.494    """495    paths: dict[str, list[str]] = {}496 497    def _walk(node: Any, prefix: list[str]) -> None:498        if isinstance(node, list):499            for iso in node:500                paths[iso] = list(prefix)501        elif isinstance(node, dict):502            for key, child in node.items():503                _walk(child, prefix + [key])504        elif isinstance(node, str):505            # Single ISO as a leaf506            paths[node] = list(prefix)507 508    _walk(tree, [])509    return paths510 511 512def compute_distance(513    lang_a: str,514    lang_b: str,515    lang_paths: dict[str, list[str]],516) -> tuple[int, str]:517    """Return (edge_count, level_label) between two languages.518 519    Level mapping:520        L1 — same leaf group (same last path element)521        L2 — LCA is two levels above leaf522        L3 — LCA is deeper within same top-level family523        L4 — different top-level families or not in tree524    """525    pa = lang_paths.get(lang_a)526    pb = lang_paths.get(lang_b)527    if pa is None or pb is None:528        return (99, "L4")529    if not pa or not pb:530        return (99, "L4")531 532    # Find LCA depth533    lca_depth = 0534    for i, (a, b) in enumerate(zip(pa, pb)):535        if a != b:536            break537        lca_depth = i + 1538    else:539        # Exhausted the shorter path without mismatch540        lca_depth = min(len(pa), len(pb))541 542    if lca_depth == 0:543        # Different top-level families544        return (len(pa) + len(pb), "L4")545 546    depth_a = len(pa)547    depth_b = len(pb)548    edge_count = (depth_a - lca_depth) + (depth_b - lca_depth)549 550    # Determine level from the relationship between paths.551    #552    # L1 — same leaf group: paths are identical (both languages listed under553    #       the same terminal node, e.g. both in "anglo_frisian").554    # L2 — same branch, different leaf group: LCA is the *parent* of the555    #       leaf groups (e.g. eng in anglo_frisian, deu in high_german; LCA556    #       is west_germanic) OR LCA is one more level up within the same557    #       major branch.  Concretely: the paths diverge at the last or558    #       second-to-last position relative to the *shorter* path.559    # L3 — same top-level family, deeper divergence (e.g. eng vs fra both in560    #       indo_european but in different major branches).561    # L4 — different top-level families (already handled above).562 563    if pa == pb:564        # Same leaf group565        return (edge_count, "L1")566 567    # Classification based on how deep the shared ancestry is:568    #   L2 — same branch: they share at least 2 path elements569    #         e.g. both under [indo_european, germanic, ...]570    #         or both under [uralic, finnic, ...]571    #   L3 — same top-level family but different branches:572    #         they share only 1 path element (the super-family)573    #         e.g. [indo_european, germanic, ...] vs [indo_european, italic, ...]574    if lca_depth >= 2:575        return (edge_count, "L2")576    # lca_depth == 1: same top-level family, different branches577    return (edge_count, "L3")578 579 580def get_top_family(iso: str, lang_paths: dict[str, list[str]], family_map: dict[str, str]) -> str:581    """Return the top-level family name for *iso*.582 583    Tries the tree path first; falls back to family_map.584    """585    path = lang_paths.get(iso)586    if path:587        # The first element is the top-level family node; but we want the588        # *branch* name that corresponds to TOP_FAMILIES.  The branch is589        # typically the second element (e.g. "germanic" under "indo_european").590        # For top-level families like "uralic", the first element IS the family.591        if len(path) >= 2 and path[1] in _TOP_FAMILY_SET:592            return path[1]593        if path[0] in _TOP_FAMILY_SET:594            return path[0]595        # Search path for any matching family name596        for segment in path:597            if segment in _TOP_FAMILY_SET:598                return segment599        # Return the first path element as family600        return path[0] if path else "unknown"601 602    # Fallback: family_map603    fam = family_map.get(iso, "unknown")604    return FAMILY_ALIAS.get(fam, fam)605 606 607_TOP_FAMILY_SET = set(TOP_FAMILIES)608 609# ---------------------------------------------------------------------------610# Data loading611# ---------------------------------------------------------------------------612 613LexEntry = tuple[str, str, str]  # (word, ipa, sca)614 615 616def load_lexicons(617    lexicons_dir: Path,618) -> dict[tuple[str, str], list[LexEntry]]:619    """Load all lexicon TSVs into ``{(iso, concept_id): [(word, ipa, sca), ...]}``.620 621    Entries with ``Concept_ID`` of ``"-"`` or empty are skipped.622    """623    lexicon: dict[tuple[str, str], list[LexEntry]] = defaultdict(list)624    files = sorted(lexicons_dir.glob("*.tsv"))625    total_entries = 0626    skipped_no_concept = 0627 628    for i, fp in enumerate(files, 1):629        iso = fp.stem630        if i % 100 == 0 or i == len(files):631            print(f"  Loading lexicon {i}/{len(files)} ({iso}) ...")632        try:633            with fp.open(encoding="utf-8") as fh:634                reader = csv.DictReader(fh, delimiter="\t")635                for row in reader:636                    cid = row.get("Concept_ID", "").strip()637                    if cid in ("", "-"):638                        skipped_no_concept += 1639                        continue640                    word = row.get("Word", "").strip()641                    ipa = row.get("IPA", "").strip()642                    sca = row.get("SCA", "").strip()643                    if not word and not ipa and not sca:644                        continue645                    lexicon[(iso, cid)].append((word, ipa, sca))646                    total_entries += 1647        except Exception as exc:648            print(f"  WARNING: failed to read {fp.name}: {exc}", file=sys.stderr)649 650    print(f"  Loaded {total_entries:,} entries across {len(files)} lexicons "651          f"(skipped {skipped_no_concept:,} without concept).")652    return lexicon653 654 655def build_concept_index(656    lexicon: dict[tuple[str, str], list[LexEntry]],657) -> dict[str, list[str]]:658    """Map each concept_id to the list of ISO codes that have entries for it."""659    concept_langs: dict[str, set[str]] = defaultdict(set)660    for (iso, cid) in lexicon:661        concept_langs[cid].add(iso)662    return {cid: sorted(isos) for cid, isos in concept_langs.items()}663 664 665def load_cognate_pairs(path: Path) -> list[dict[str, str]]:666    """Load a cognate-pairs TSV file into a list of dicts."""667    rows: list[dict[str, str]] = []668    if not path.exists():669        print(f"  WARNING: {path} not found, skipping.")670        return rows671    with path.open(encoding="utf-8") as fh:672        reader = csv.DictReader(fh, delimiter="\t")673        for row in reader:674            rows.append(dict(row))675    print(f"  Loaded {len(rows):,} pairs from {path.name}")676    return rows677 678 679# ---------------------------------------------------------------------------680# Pair record builder681# ---------------------------------------------------------------------------682 683def make_pair_record(684    lang_a: str,685    word_a: str,686    ipa_a: str,687    sca_a: str,688    lang_b: str,689    word_b: str,690    ipa_b: str,691    sca_b: str,692    concept_id: str,693    label: str,694    lang_paths: dict[str, list[str]],695    source: str = "lexicon",696    score_override: float | None = None,697) -> dict[str, str]:698    """Build a single output-row dict."""699    _, level = compute_distance(lang_a, lang_b, lang_paths)700    ts = get_timespan(lang_a, lang_b)701    if score_override is not None:702        score = score_override703    else:704        score = normalised_similarity(sca_a, sca_b) if sca_a and sca_b else 0.0705    return {706        "Lang_A": lang_a,707        "Word_A": word_a,708        "IPA_A": ipa_a,709        "SCA_A": sca_a,710        "Lang_B": lang_b,711        "Word_B": word_b,712        "IPA_B": ipa_b,713        "SCA_B": sca_b,714        "Concept_ID": concept_id,715        "Label": label,716        "Phylo_Dist": level,717        "Timespan": ts,718        "Score": f"{score:.4f}",719        "Source": source,720    }721 722 723# ---------------------------------------------------------------------------724# TSV writing helper725# ---------------------------------------------------------------------------726 727def write_pairs_tsv(path: Path, pairs: list[dict[str, str]]) -> None:728    """Write a list of pair dicts as a TSV file."""729    path.parent.mkdir(parents=True, exist_ok=True)730    with path.open("w", encoding="utf-8", newline="") as fh:731        writer = csv.DictWriter(fh, fieldnames=OUTPUT_FIELDS, delimiter="\t",732                                extrasaction="ignore")733        writer.writeheader()734        for row in pairs:735            writer.writerow(row)736    print(f"  Wrote {len(pairs):,} pairs to {path.relative_to(REPO_ROOT)}")737 738 739# ---------------------------------------------------------------------------740# Step 3: True cognate pair generation741# ---------------------------------------------------------------------------742 743def generate_true_cognates(744    lexicon: dict[tuple[str, str], list[LexEntry]],745    concept_langs: dict[str, list[str]],746    lang_paths: dict[str, list[str]],747    family_map: dict[str, str],748    inherited_pairs: list[dict[str, str]],749) -> tuple[list[dict[str, str]], list[dict[str, str]], list[dict[str, str]]]:750    """Generate true cognate pairs at L1, L2, L3 levels.751 752    Returns (l1_pairs, l2_pairs, l3_pairs).753    """754    l1: list[dict[str, str]] = []755    l2: list[dict[str, str]] = []756    l3: list[dict[str, str]] = []757 758    buckets = {"L1": l1, "L2": l2, "L3": l3}759    thresholds = {"L1": 0.5, "L2": 0.5, "L3": 0.4}760 761    # Step 3a: incorporate expert pairs FIRST (they have priority)762    print("Step 3a: Adding expert cognate pairs ...")763    expert_added = 0764    for row in inherited_pairs:765        if all(len(b) >= PAIR_CAP for b in buckets.values()):766            break767 768        lang_a = row.get("Lang_A", "")769        lang_b = row.get("Lang_B", "")770        if not lang_a or not lang_b:771            continue772 773        _, level = compute_distance(lang_a, lang_b, lang_paths)774        if level not in buckets or len(buckets[level]) >= PAIR_CAP:775            continue776 777        cid = row.get("Concept_ID", "")778        ipa_a = row.get("IPA_A", "")779        ipa_b = row.get("IPA_B", "")780        word_a = row.get("Word_A", "")781        word_b = row.get("Word_B", "")782        sca_a = _lookup_sca(lexicon, lang_a, cid, word_a, ipa_a)783        sca_b = _lookup_sca(lexicon, lang_b, cid, word_b, ipa_b)784 785        score_str = row.get("Score", "0")786        try:787            score = float(score_str)788        except (ValueError, TypeError):789            score = 0.0790 791        if sca_a and sca_b:792            sca_sim = normalised_similarity(sca_a, sca_b)793        else:794            sca_sim = score795 796        rec = make_pair_record(797            lang_a, word_a, ipa_a, sca_a,798            lang_b, word_b, ipa_b, sca_b,799            cid, "true_cognate", lang_paths,800            source=row.get("Source", "expert"),801            score_override=sca_sim,802        )803        rec["Phylo_Dist"] = level804        buckets[level].append(rec)805        expert_added += 1806 807    print(f"  Added {expert_added:,} expert pairs "808          f"(L1={len(l1):,} L2={len(l2):,} L3={len(l3):,})")809 810    # Step 3b: Fill remaining slots from lexicon data811    print("Step 3b: Generating true cognates from lexicon data ...")812 813    concepts_processed = 0814    total_concepts = len(concept_langs)815 816    for cid, langs in concept_langs.items():817        concepts_processed += 1818        if concepts_processed % 500 == 0:819            print(f"  Concept {concepts_processed}/{total_concepts} "820                  f"(L1={len(l1):,} L2={len(l2):,} L3={len(l3):,})")821 822        # Early termination823        if all(len(b) >= PAIR_CAP for b in buckets.values()):824            break825 826        # Group languages by top-level family827        family_groups: dict[str, list[str]] = defaultdict(list)828        for iso in langs:829            fam = get_top_family(iso, lang_paths, family_map)830            family_groups[fam].append(iso)831 832        # Within each family, generate pairs833        for fam, fam_langs in family_groups.items():834            if len(fam_langs) < 2:835                continue836 837            # Sample pairs if too many languages838            if len(fam_langs) > 50:839                sampled = random.sample(fam_langs, 50)840            else:841                sampled = fam_langs842 843            pair_count_this_concept: dict[str, int] = {"L1": 0, "L2": 0, "L3": 0}844 845            for i in range(len(sampled)):846                for j in range(i + 1, len(sampled)):847                    iso_a = sampled[i]848                    iso_b = sampled[j]849                    if iso_a == iso_b:850                        continue851 852                    _, level = compute_distance(iso_a, iso_b, lang_paths)853                    if level not in buckets or len(buckets[level]) >= PAIR_CAP:854                        continue855                    if pair_count_this_concept.get(level, 0) >= MAX_PAIRS_PER_CONCEPT_PER_LEVEL:856                        continue857 858                    thresh = thresholds[level]859                    entries_a = lexicon.get((iso_a, cid), [])860                    entries_b = lexicon.get((iso_b, cid), [])861                    if not entries_a or not entries_b:862                        continue863 864                    # Pick one entry from each language (first with SCA)865                    ea = _pick_best_entry(entries_a)866                    eb = _pick_best_entry(entries_b)867                    if ea is None or eb is None:868                        continue869 870                    sca_sim = normalised_similarity(ea[2], eb[2]) if ea[2] and eb[2] else 0.0871                    if sca_sim < thresh:872                        continue873 874                    rec = make_pair_record(875                        iso_a, ea[0], ea[1], ea[2],876                        iso_b, eb[0], eb[1], eb[2],877                        cid, "true_cognate", lang_paths,878                        source="lexicon",879                        score_override=sca_sim,880                    )881                    rec["Phylo_Dist"] = level  # ensure correct level882                    buckets[level].append(rec)883                    pair_count_this_concept[level] = pair_count_this_concept.get(level, 0) + 1884 885    print(f"  From lexicon: L1={len(l1):,} L2={len(l2):,} L3={len(l3):,}")886 887    # Cap each bucket888    for level_name, bucket in buckets.items():889        if len(bucket) > PAIR_CAP:890            random.shuffle(bucket)891            buckets[level_name] = bucket[:PAIR_CAP]892            if level_name == "L1":893                l1[:] = buckets[level_name]894            elif level_name == "L2":895                l2[:] = buckets[level_name]896            else:897                l3[:] = buckets[level_name]898 899    print(f"  Final: L1={len(l1):,} L2={len(l2):,} L3={len(l3):,}")900    return l1, l2, l3901 902 903def _pick_best_entry(entries: list[LexEntry]) -> LexEntry | None:904    """Pick an entry preferring ones with non-empty SCA."""905    for e in entries:906        if e[2]:  # has SCA907            return e908    return entries[0] if entries else None909 910 911def _lookup_sca(912    lexicon: dict[tuple[str, str], list[LexEntry]],913    iso: str,914    concept_id: str,915    word: str,916    ipa: str = "",917) -> str:918    """Try to find the SCA encoding for a given word from the lexicon.919 920    Falls back to computing SCA on-the-fly from IPA if all lookups fail.921    """922    entries = lexicon.get((iso, concept_id), [])923    # Exact word match first924    for e in entries:925        if e[0] == word and e[2]:926            return e[2]927    # Any entry with SCA for this (iso, concept)928    for e in entries:929        if e[2]:930            return e[2]931    # Use the word index for cross-concept fallback (fast)932    sca = _word_sca_index.get((iso, word), "")933    if sca:934        return sca935    # Last resort: compute from IPA on-the-fly936    if ipa:937        return ipa_to_sound_class(ipa)938    return ""939 940 941# ---------------------------------------------------------------------------942# Step 4: False positives943# ---------------------------------------------------------------------------944 945def generate_false_positives(946    lexicon: dict[tuple[str, str], list[LexEntry]],947    concept_langs: dict[str, list[str]],948    lang_paths: dict[str, list[str]],949    family_map: dict[str, str],950    similarity_pairs: list[dict[str, str]],951) -> list[dict[str, str]]:952    """Cross-family pairs with same concept and SCA similarity >= 0.5."""953    print("Step 4: Generating false positives ...")954    fps: list[dict[str, str]] = []955 956    # 4a: From lexicon — cross-family same-concept pairs957    concepts_processed = 0958    for cid, langs in concept_langs.items():959        if len(fps) >= PAIR_CAP:960            break961        concepts_processed += 1962        if concepts_processed % 500 == 0:963            print(f"  Concept {concepts_processed}/{len(concept_langs)} "964                  f"(fps={len(fps):,})")965 966        # Group by family967        family_groups: dict[str, list[str]] = defaultdict(list)968        for iso in langs:969            fam = get_top_family(iso, lang_paths, family_map)970            family_groups[fam].append(iso)971 972        families = list(family_groups.keys())973        if len(families) < 2:974            continue975 976        # Cross-family pairs977        pair_count = 0978        for fi in range(len(families)):979            for fj in range(fi + 1, len(families)):980                if len(fps) >= PAIR_CAP:981                    break982                if pair_count >= MAX_CROSS_FAMILY_PAIRS_PER_CONCEPT:983                    break984 985                fam_a_langs = family_groups[families[fi]]986                fam_b_langs = family_groups[families[fj]]987 988                # Sample one from each989                iso_a = random.choice(fam_a_langs)990                iso_b = random.choice(fam_b_langs)991 992                entries_a = lexicon.get((iso_a, cid), [])993                entries_b = lexicon.get((iso_b, cid), [])994                if not entries_a or not entries_b:995                    continue996 997                ea = _pick_best_entry(entries_a)998                eb = _pick_best_entry(entries_b)999                if ea is None or eb is None:1000                    continue1001 1002                sca_sim = normalised_similarity(ea[2], eb[2]) if ea[2] and eb[2] else 0.01003                if sca_sim < 0.5:1004                    continue1005 1006                rec = make_pair_record(1007                    iso_a, ea[0], ea[1], ea[2],1008                    iso_b, eb[0], eb[1], eb[2],1009                    cid, "false_positive", lang_paths,1010                    source="lexicon",1011                    score_override=sca_sim,1012                )1013                fps.append(rec)1014                pair_count += 11015 1016    print(f"  From lexicon: {len(fps):,} false positives")1017 1018    # 4b: From similarity_pairs file1019    sim_added = 01020    for row in similarity_pairs:1021        if len(fps) >= PAIR_CAP:1022            break1023        lang_a = row.get("Lang_A", "")1024        lang_b = row.get("Lang_B", "")1025        if not lang_a or not lang_b:1026            continue1027 1028        fam_a = get_top_family(lang_a, lang_paths, family_map)1029        fam_b = get_top_family(lang_b, lang_paths, family_map)1030        if fam_a == fam_b:1031            continue  # only want cross-family1032 1033        cid = row.get("Concept_ID", "")1034        sca_a = _lookup_sca(lexicon, lang_a, cid, row.get("Word_A", ""),1035                            row.get("IPA_A", ""))1036        sca_b = _lookup_sca(lexicon, lang_b, cid, row.get("Word_B", ""),1037                            row.get("IPA_B", ""))1038 1039        score_str = row.get("Score", "0")1040        try:1041            score = float(score_str)1042        except (ValueError, TypeError):1043            score = 0.01044 1045        if sca_a and sca_b:1046            sca_sim = normalised_similarity(sca_a, sca_b)1047        else:1048            sca_sim = score1049 1050        if sca_sim < 0.5:1051            continue1052 1053        rec = make_pair_record(1054            lang_a, row.get("Word_A", ""), row.get("IPA_A", ""), sca_a,1055            lang_b, row.get("Word_B", ""), row.get("IPA_B", ""), sca_b,1056            cid, "false_positive", lang_paths,1057            source=row.get("Source", "similarity"),1058            score_override=sca_sim,1059        )1060        fps.append(rec)1061        sim_added += 11062 1063    print(f"  Added {sim_added:,} from similarity pairs")1064 1065    if len(fps) > PAIR_CAP:1066        random.shuffle(fps)1067        fps = fps[:PAIR_CAP]1068 1069    print(f"  Final false positives: {len(fps):,}")1070    return fps1071 1072 1073# ---------------------------------------------------------------------------1074# Step 5: True negatives1075# ---------------------------------------------------------------------------1076 1077def generate_true_negatives(1078    lexicon: dict[tuple[str, str], list[LexEntry]],1079    lang_paths: dict[str, list[str]],1080    family_map: dict[str, str],1081) -> list[dict[str, str]]:1082    """Random cross-family pairs with different concepts and SCA sim < 0.3."""1083    print("Step 5: Generating true negatives ...")1084    negs: list[dict[str, str]] = []1085 1086    # Build index of all (iso, concept_id) keys for sampling1087    all_keys = list(lexicon.keys())1088    if len(all_keys) < 2:1089        print("  Not enough lexicon entries for true negatives.")1090        return negs1091 1092    attempts = 01093    while len(negs) < PAIR_CAP and attempts < TRUE_NEG_SAMPLE_ATTEMPTS:1094        attempts += 11095        if attempts % 50_000 == 0:1096            print(f"  Attempt {attempts:,}, negatives so far: {len(negs):,}")1097 1098        # Pick two random entries1099        key_a = random.choice(all_keys)1100        key_b = random.choice(all_keys)1101        iso_a, cid_a = key_a1102        iso_b, cid_b = key_b1103 1104        # Must be different concepts1105        if cid_a == cid_b:1106            continue1107        # Must be different families1108        fam_a = get_top_family(iso_a, lang_paths, family_map)1109        fam_b = get_top_family(iso_b, lang_paths, family_map)1110        if fam_a == fam_b:1111            continue1112        # Must be different languages1113        if iso_a == iso_b:1114            continue1115 1116        entries_a = lexicon[key_a]1117        entries_b = lexicon[key_b]1118        ea = _pick_best_entry(entries_a)1119        eb = _pick_best_entry(entries_b)1120        if ea is None or eb is None:1121            continue1122        if not ea[2] or not eb[2]:1123            continue1124 1125        sca_sim = normalised_similarity(ea[2], eb[2])1126        if sca_sim >= 0.4:1127            continue1128 1129        # Use concept_a for the record (arbitrary; both concepts are different)1130        rec = make_pair_record(1131            iso_a, ea[0], ea[1], ea[2],1132            iso_b, eb[0], eb[1], eb[2],1133            f"{cid_a} / {cid_b}", "true_negative", lang_paths,1134            source="random_sample",1135            score_override=sca_sim,1136        )1137        negs.append(rec)1138 1139    print(f"  Generated {len(negs):,} true negatives in {attempts:,} attempts")1140    return negs1141 1142 1143# ---------------------------------------------------------------------------1144# Step 6: Borrowings1145# ---------------------------------------------------------------------------1146 1147def generate_borrowings(1148    borrowing_pairs: list[dict[str, str]],1149    lexicon: dict[tuple[str, str], list[LexEntry]],1150    lang_paths: dict[str, list[str]],1151) -> list[dict[str, str]]:1152    """Process borrowing pairs from WOLD."""1153    print("Step 6: Processing borrowing pairs ...")1154    rows: list[dict[str, str]] = []1155 1156    for row in borrowing_pairs:1157        lang_a = row.get("Lang_A", "")1158        lang_b = row.get("Lang_B", "")1159        if not lang_a or not lang_b:1160            continue1161 1162        cid = row.get("Concept_ID", "")1163        word_a = row.get("Word_A", "")1164        word_b = row.get("Word_B", "")1165        ipa_a = row.get("IPA_A", "")1166        ipa_b = row.get("IPA_B", "")1167 1168        sca_a = _lookup_sca(lexicon, lang_a, cid, word_a, ipa_a)1169        sca_b = _lookup_sca(lexicon, lang_b, cid, word_b, ipa_b)1170 1171        score_str = row.get("Score", "0")1172        try:1173            score = float(score_str)1174        except (ValueError, TypeError):1175            score = 0.01176 1177        if sca_a and sca_b:1178            sca_sim = normalised_similarity(sca_a, sca_b)1179        else:1180            sca_sim = score1181 1182        rec = make_pair_record(1183            lang_a, word_a, ipa_a, sca_a,1184            lang_b, word_b, ipa_b, sca_b,1185            cid, "borrowing", lang_paths,1186            source=row.get("Source", "wold"),1187            score_override=sca_sim,1188        )1189        rows.append(rec)1190 1191    print(f"  Processed {len(rows):,} borrowing pairs")1192    return rows1193 1194 1195# ---------------------------------------------------------------------------1196# Step 7: Religious subsets1197# ---------------------------------------------------------------------------1198 1199def generate_religious_pairs(1200    all_pairs: list[dict[str, str]],

Showing the first 1,200 of 1729 lines. Download the file for the rest.