Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
build_linear_b_dataset.py797 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Build the Linear B (Mycenaean Greek) dataset from downloaded raw data.3 4Parses and combines data from:5  1. Unicode UCD — Sign inventory (88 syllabograms + 123 ideograms)6  2. jhnwnstd/shannon — Linear B Lexicon (2,747 entries)7  3. Wiktionary — Mycenaean Greek lemmas (~435 entries with IPA)8  4. IE-CoR — Existing 43 Mycenaean Greek (gmy) words with expert IPA9 10Output files:11  data/linear_b/linear_b_signs.tsv — Full sign inventory12  data/linear_b/sign_to_ipa.json — Sign transliteration → IPA mapping13  data/linear_b/linear_b_words.tsv — Word list (Word, IPA, SCA, Source, Concept_ID, Cognate_Set_ID)14  data/linear_b/README.md — Documentation15 16Transliteration → IPA mapping:17  Reference: Ventris & Chadwick (1973) "Documents in Mycenaean Greek", 2nd ed.18  The Linear B syllabary encodes CV syllables. The conventional transliteration19  uses Latin characters that are near-IPA with these systematic differences:20    q = /kʷ/ (labiovelar stop)21    z = /ts/ or /dz/ (affricate, exact value debated)22    j = /j/ (palatal glide)23    w = /w/ (labial glide)24    p2 = /pʰ/ (aspirated p)25    t2 = /tʰ/ (aspirated t)  — actually written as "pu2" etc. in convention26 27Usage:28    python scripts/build_linear_b_dataset.py29"""30 31from __future__ import annotations32 33import csv34import io35import json36import re37import sys38import unicodedata39from collections import OrderedDict40from pathlib import Path41 42sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")43sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")44 45ROOT = Path(__file__).resolve().parent.parent46RAW_DIR = ROOT / "data" / "training" / "raw" / "linear_b"47OUT_DIR = ROOT / "data" / "linear_b"48 49# ── Linear B Unicode ranges ──50LINB_SYLLABARY_START = 0x1000051LINB_SYLLABARY_END = 0x1007F52LINB_IDEOGRAM_START = 0x1008053LINB_IDEOGRAM_END = 0x100FF54 55# ── Transliteration → IPA mapping ──56# Reference: Ventris & Chadwick (1973), "Documents in Mycenaean Greek", 2nd ed.57# Palmer (1963), "The Interpretation of Mycenaean Greek Texts"58# Hooker (1980), "Linear B: An Introduction"59#60# The conventional transliteration values are based on the Ventris decipherment61# (1952) and CIPEM standard. Most consonants map directly; the key differences are:62#   - q-series represents labiovelars /kʷ/, not /k/63#   - z-series represents affricates, transcribed as /ts/ (Hooker 1980: p.68)64#   - j represents /j/ (palatal approximant)65#   - w represents /w/ (labio-velar approximant)66#67# The "2" variants (a2, a3, pu2, etc.) represent:68#   - a2 = /ha/ (initial aspiration)69#   - a3 = /ai/ (diphthong)70#   - pu2 = /pʰu/ (aspirated)71#   - ra2 = /rja/ (palatalized)72#   - ro2 = /rjo/73#   - ta2 = /tja/74#   - nwa = /nwa/75#76# For undeciphered signs (*18, *19, etc.), IPA is left as "-".77 78TRANSLIT_TO_IPA = {79    # Pure vowels80    "a": "a", "e": "e", "i": "i", "o": "o", "u": "u",81    # d-series82    "da": "da", "de": "de", "di": "di", "do": "do", "du": "du",83    # j-series (palatal glide)84    "ja": "ja", "je": "je", "jo": "jo", "ju": "ju",85    # k-series86    "ka": "ka", "ke": "ke", "ki": "ki", "ko": "ko", "ku": "ku",87    # m-series88    "ma": "ma", "me": "me", "mi": "mi", "mo": "mo", "mu": "mu",89    # n-series90    "na": "na", "ne": "ne", "ni": "ni", "no": "no", "nu": "nu",91    # p-series92    "pa": "pa", "pe": "pe", "pi": "pi", "po": "po", "pu": "pu",93    # q-series (labiovelars)94    "qa": "kʷa", "qe": "kʷe", "qi": "kʷi", "qo": "kʷo",95    # r-series (covers both /r/ and /l/ — Linear B does not distinguish)96    "ra": "ra", "re": "re", "ri": "ri", "ro": "ro", "ru": "ru",97    # s-series98    "sa": "sa", "se": "se", "si": "si", "so": "so", "su": "su",99    # t-series100    "ta": "ta", "te": "te", "ti": "ti", "to": "to", "tu": "tu",101    # w-series102    "wa": "wa", "we": "we", "wi": "wi", "wo": "wo",103    # z-series (affricates: Hooker 1980, p.68)104    "za": "tsa", "ze": "tse", "zi": "tsi", "zo": "tso", "zu": "tsu",105    # Special/variant signs106    "a2": "ha", "a3": "ai",107    "nwa": "nwa",108    "pu2": "pʰu",109    "ra2": "rja", "ra3": "rai",110    "ro2": "rjo",111    "ta2": "tja",112    "two": "two",113    "dwe": "dwe",114    "dwo": "dwo",115    "twe": "twe",116    # Undeciphered signs — no IPA117}118 119 120def parse_unicode_signs(ucd_path: Path) -> list[dict]:121    """Parse Linear B signs from UnicodeData.txt.122 123    Each line has format: codepoint;name;category;...124    We extract signs in U+10000-U+100FF range.125    """126    signs = []127    with open(ucd_path, encoding="utf-8") as f:128        for line in f:129            parts = line.strip().split(";")130            if len(parts) < 2:131                continue132            cp_hex = parts[0]133            name = parts[1]134            cp = int(cp_hex, 16)135 136            if LINB_SYLLABARY_START <= cp <= LINB_SYLLABARY_END:137                sign_type = "syllabogram"138            elif LINB_IDEOGRAM_START <= cp <= LINB_IDEOGRAM_END:139                sign_type = "ideogram"140            else:141                continue142 143            # Parse Bennett number and phonetic value from name144            # Format: "LINEAR B SYLLABLE B008 A" or "LINEAR B IDEOGRAM B100 MAN"145            bennett = ""146            phonetic = ""147            m = re.match(r"LINEAR B (?:SYLLABLE|SYMBOL) (B\d+)\s*(.*)", name)148            if m:149                bennett = m.group(1)150                phonetic = m.group(2).strip().lower() if m.group(2) else ""151            else:152                m = re.match(r"LINEAR B IDEOGRAM (B\d+\w*)\s*(.*)", name)153                if m:154                    bennett = m.group(1)155                    phonetic = m.group(2).strip() if m.group(2) else ""156 157            # Get IPA from transliteration158            ipa = TRANSLIT_TO_IPA.get(phonetic, "-") if phonetic else "-"159 160            signs.append({161                "Codepoint": f"U+{cp_hex}",162                "Unicode_Char": chr(cp),163                "Bennett_Number": bennett,164                "Name": name,165                "Type": sign_type,166                "Transliteration": phonetic if phonetic else "-",167                "IPA": ipa,168            })169 170    return signs171 172 173def parse_shannon_lexicon(csv_path: Path) -> list[dict]:174    """Parse jhnwnstd/shannon Linear_B_Lexicon.csv.175 176    Columns: word (Unicode), transcription (Latin), definition (scholarly notes)177    We extract: transliteration, clean definition, and classify as common/proper noun.178    """179    entries = []180    with open(csv_path, encoding="utf-8") as f:181        reader = csv.DictReader(f)182        for row in reader:183            word_unicode = row.get("word", "").strip()184            translit = row.get("transcription", "").strip()185            definition = row.get("definition", "").strip()186 187            if not translit:188                continue189 190            # Classify: is this a common noun or anthroponym/toponym?191            def_lower = definition.lower()192            is_anthroponym = "anthroponym" in def_lower and ":" not in def_lower.split("anthroponym")[0][-20:]193            is_toponym = "toponym" in def_lower and ":" not in def_lower.split("toponym")[0][-20:]194 195            # Try to extract a clean gloss from the definition196            # Patterns:197            #   "Chadwick & Ventris 1973: anthroponym" → type=proper, gloss=anthroponym198            #   "Chadwick & Ventris 1973: figs" → type=common, gloss=figs199            gloss = ""200            # Look for meaning after first colon201            colon_parts = definition.split(":", 1)202            if len(colon_parts) > 1:203                after_colon = colon_parts[1].strip()204                # Take the first meaningful phrase (up to next reference or semicolon)205                # Clean up common noise206                gloss_match = re.match(207                    r"([\w\s,/()?.!'\-]+?)(?:\s+(?:Chadwick|McArthur|Witczak|van |Palmer|"208                    r"Ruijgh|Bernabé|Appears|KN|PY|MY|TH|TI))",209                    after_colon,210                )211                if gloss_match:212                    gloss = gloss_match.group(1).strip().rstrip(",;.")213                else:214                    # Take first 80 chars as fallback215                    gloss = after_colon[:80].strip()216                    # Cut at first reference-like pattern217                    for cutoff in ["Chadwick", "McArthur", "Ventris", "John and"]:218                        if cutoff in gloss:219                            gloss = gloss[: gloss.index(cutoff)].strip().rstrip(",;.")220                            break221 222            # Determine word type223            if "anthroponym" in gloss.lower():224                word_type = "anthroponym"225            elif "toponym" in gloss.lower():226                word_type = "toponym"227            elif "theonym" in gloss.lower():228                word_type = "theonym"229            elif "ethnic" in gloss.lower():230                word_type = "ethnic"231            elif not gloss or gloss.lower() in ("meaning obscure", "meaning unknown",232                                                 "meaning uncertain", "hapax"):233                word_type = "unknown"234            else:235                word_type = "common"236 237            entries.append({238                "Word_Unicode": word_unicode,239                "Transliteration": translit,240                "Gloss": gloss,241                "Word_Type": word_type,242                "Source": "shannon_lexicon",243            })244 245    return entries246 247 248def unicode_to_translit(title: str) -> str:249    """Convert Linear B Unicode characters in a title to transliteration.250 251    Uses Python's unicodedata to get character names, then extracts the252    phonetic value from names like "LINEAR B SYLLABLE B008 A" → "a".253    """254    parts = []255    for ch in title:256        cp = ord(ch)257        if LINB_SYLLABARY_START <= cp <= LINB_SYLLABARY_END:258            try:259                name = unicodedata.name(ch, "")260                m = re.match(r"LINEAR B (?:SYLLABLE|SYMBOL) B\d+\s*(.*)", name)261                if m and m.group(1):262                    parts.append(m.group(1).strip().lower())263                else:264                    # Undeciphered symbol265                    m2 = re.match(r"LINEAR B SYMBOL (B\d+)", name)266                    if m2:267                        parts.append(f"*{m2.group(1)[1:]}")268            except ValueError:269                pass270        elif LINB_IDEOGRAM_START <= cp <= LINB_IDEOGRAM_END:271            # Ideograms — skip or mark272            try:273                name = unicodedata.name(ch, "")274                m = re.match(r"LINEAR B IDEOGRAM (B\d+\w*)\s*(.*)", name)275                if m:276                    parts.append(f"[{m.group(2).strip() or m.group(1)}]")277            except ValueError:278                pass279        # Skip non-Linear B characters (spaces, combining marks, etc.)280    return "-".join(parts) if parts else ""281 282 283def parse_wiktionary_lemmas(json_path: Path) -> list[dict]:284    """Parse Wiktionary Mycenaean Greek lemma data.285 286    Extract from wikitext:287      - ts= parameter → IPA transcription288      - # [[gloss]] → English meaning289      - head template → part of speech290    """291    with open(json_path, encoding="utf-8") as f:292        lemmas = json.load(f)293 294    entries = []295    for lemma in lemmas:296        title = lemma["title"]297        wikitext = lemma["wikitext"]298 299        # Skip if not Mycenaean Greek300        if "==Mycenaean Greek==" not in wikitext:301            continue302 303        # Convert Unicode title to transliteration304        title_translit = unicode_to_translit(title)305 306        # Skip ideogram-only entries (titles that are purely ideograms or *NNN)307        if not title_translit or all(308            p.startswith("[") or p.startswith("*") for p in title_translit.split("-") if p309        ):310            # Check if it has a tr= parameter we could use instead311            tr_check = re.search(r"\|tr=([^|}]+)", wikitext)312            if not tr_check:313                continue314 315        # Extract IPA from ts= parameter — ONLY from {{head|gmy|...}} template,316        # NOT from {{quote|gmy|...}} tablet quotation contexts.317        # The head template has format: {{head|gmy|noun|ts=VALUE}}318        # Quote templates have format: {{quote|gmy|...|ts=FULL_SENTENCE}}319        ipa = ""320        head_match = re.search(r"\{\{(?:head|h)\|gmy\|[^}]*\|ts=([^|}]+)", wikitext)321        if head_match:322            ipa = head_match.group(1).strip()323            # Sanity check: headword IPA should be a single word, not a sentence324            # If it contains spaces or <br>, it's a tablet quotation that leaked in325            if " " in ipa or "<br>" in ipa:326                ipa = ""327 328        # Extract transliteration: prefer Unicode title conversion, fallback to tr= from head template.329        # IMPORTANT: Do NOT use a global tr= search — {{quote-book}} templates contain330        # tr= with full tablet transliterations (e.g., "o-di-do-si du-ru-to-mo / ..."),331        # which are NOT the headword transliteration. Only use tr= from {{head|gmy|...}}.332        translit = title_translit  # Primary: Unicode character names → transliteration333        if not translit:334            # Fallback: tr= from head template only335            head_tr_match = re.search(r"\{\{(?:head|h)\|gmy\|[^}]*\|tr=([^|}]+)", wikitext)336            if head_tr_match:337                translit = head_tr_match.group(1).strip()338 339        # Clean transliteration: remove tablet context, bold markers, etc.340        # Wiktionary titles sometimes embed context like "'''di-wo''' u-ta-jo-jo"341        if translit:342            # Remove wikitext bold markers343            translit = translit.replace("'''", "")344            # If transliteration contains spaces (tablet context), take first word only345            if " " in translit:346                translit = translit.split()[0]347            # Remove trailing punctuation348            translit = translit.strip(".,;:!?")349            # Skip if still contains non-transliteration characters350            if re.search(r"[<>\[\]{}|=]", translit):351                continue352 353        # Skip entries with no usable transliteration354        if not translit or translit == "-":355            continue356 357        # Skip pure ideogram/logogram entries (*NNN without syllabic content)358        # These are ideograms like *142, *150, etc. that have no phonetic reading359        translit_parts = [p for p in translit.split("-") if p]360        syllabic_parts = [p for p in translit_parts361                          if not p.startswith("*") and not p.startswith("[")]362        if not syllabic_parts:363            continue  # Skip: no syllabic content at all364 365        # Extract gloss from definition lines (# [[word]] or # text)366        glosses = []367        for line in wikitext.split("\n"):368            line = line.strip()369            if line.startswith("# ") and not line.startswith("# {{def-uncertain"):370                # Clean wikitext markup371                gloss = line[2:]372                # Remove templates but preserve content for some373                gloss = re.sub(r"\{\{l\|en\|([^|}]+)[^}]*\}\}", r"\1", gloss)374                gloss = re.sub(r"\{\{[^}]*\}\}", "", gloss)375                # Remove links but keep text: [[word|display]] → display, [[word]] → word376                gloss = re.sub(r"\[\[(?:[^|\]]*\|)?([^\]]*)\]\]", r"\1", gloss)377                # Remove remaining markup378                gloss = re.sub(r"['\[\]]", "", gloss)379                # Remove wikitext remnants like }}, {{, etc.380                gloss = re.sub(r"\}\}|\{\{", "", gloss)381                # Remove leading/trailing whitespace and orphaned punctuation382                gloss = gloss.strip().strip(".,;:")383                if gloss and len(gloss) > 1:384                    glosses.append(gloss)385 386        # Extract part of speech387        pos = ""388        pos_match = re.search(r"\{\{head\|gmy\|(\w+)", wikitext)389        if pos_match:390            pos = pos_match.group(1)391 392        # Extract etymology cognates (useful for Concept_ID mapping)393        cognates = []394        cog_matches = re.finditer(r"\{\{cog\|grc\|([^|}]+)", wikitext)395        for m in cog_matches:396            cognates.append(m.group(1))397 398        gloss_text = "; ".join(glosses) if glosses else "-"399 400        # Determine word type from POS and content401        word_type = "common"402        if pos == "proper noun":403            word_type = "proper"404        elif "toponym" in gloss_text.lower():405            word_type = "toponym"406        elif "anthroponym" in gloss_text.lower():407            word_type = "anthroponym"408 409        entries.append({410            "Title_Unicode": title,411            "Transliteration": translit,412            "IPA": ipa,413            "Gloss": gloss_text,414            "POS": pos,415            "Word_Type": word_type,416            "Greek_Cognate": cognates[0] if cognates else "-",417            "Source": "wiktionary_gmy",418        })419 420    return entries421 422 423def transliterate_to_ipa(translit: str) -> str:424    """Convert Linear B transliteration to IPA.425 426    Reference: Ventris & Chadwick (1973), Hooker (1980)427 428    Linear B transliterations use the format: syllable-syllable-syllable429    where each syllable is a CV value from the Ventris grid.430    E.g., "a-ke-ro" → "akero", "pa-ka-na" → "pakana"431    """432    if not translit or translit == "-":433        return "-"434 435    # Remove leading/trailing hyphens and whitespace436    translit = translit.strip().strip("-")437 438    # Split on hyphens439    syllables = translit.split("-")440 441    ipa_parts = []442    for syl in syllables:443        syl = syl.strip().lower()444        if not syl:445            continue446        # Check for undeciphered signs (*18, *47, etc.)447        if syl.startswith("*"):448            ipa_parts.append("?")449            continue450        # Look up in mapping451        if syl in TRANSLIT_TO_IPA:452            ipa_parts.append(TRANSLIT_TO_IPA[syl])453        else:454            # Unknown syllable — keep as-is (it may already be a valid value)455            ipa_parts.append(syl)456 457    return "".join(ipa_parts)458 459 460def load_iecor_gmy_words() -> list[dict]:461    """Load existing Mycenaean Greek (gmy) words from cognate pairs Parquet."""462    try:463        import pyarrow.parquet as pq464        import pyarrow.compute as pc465    except ImportError:466        print("  [WARN] pyarrow not available, skipping IE-CoR data")467        return []468 469    parquet_path = ROOT / "data" / "training" / "cognate_pairs" / "cognate_pairs_inherited.parquet"470    if not parquet_path.exists():471        return []472 473    t = pq.read_table(parquet_path)474    mask_a = pc.equal(t["Lang_A"], "gmy")475    mask_b = pc.equal(t["Lang_B"], "gmy")476 477    words = {}  # translit → {ipa, concept_ids}478 479    # Extract from Lang_A side480    gmy_a = t.filter(mask_a)481    for i in range(gmy_a.num_rows):482        w = gmy_a.column("Word_A")[i].as_py()483        ipa = gmy_a.column("IPA_A")[i].as_py()484        cid = gmy_a.column("Concept_ID")[i].as_py()485        if w and w != "-":486            if w not in words:487                words[w] = {"ipa": ipa or "-", "concept_ids": set()}488            if cid and cid != "-":489                words[w]["concept_ids"].add(cid)490 491    # Extract from Lang_B side492    gmy_b = t.filter(mask_b)493    for i in range(gmy_b.num_rows):494        w = gmy_b.column("Word_B")[i].as_py()495        ipa = gmy_b.column("IPA_B")[i].as_py()496        cid = gmy_b.column("Concept_ID")[i].as_py()497        if w and w != "-":498            if w not in words:499                words[w] = {"ipa": ipa or "-", "concept_ids": set()}500            if cid and cid != "-":501                words[w]["concept_ids"].add(cid)502 503    result = []504    for translit, data in words.items():505        result.append({506            "Transliteration": translit,507            "IPA": data["ipa"],508            "Concept_IDs": ",".join(sorted(data["concept_ids"])),509            "Source": "iecor",510        })511 512    return result513 514 515def build_sign_inventory(signs: list[dict]) -> None:516    """Write sign inventory TSV and sign_to_ipa.json."""517    OUT_DIR.mkdir(parents=True, exist_ok=True)518 519    # TSV520    tsv_path = OUT_DIR / "linear_b_signs.tsv"521    cols = ["Codepoint", "Unicode_Char", "Bennett_Number", "Name", "Type",522            "Transliteration", "IPA"]523    with open(tsv_path, "w", encoding="utf-8", newline="") as f:524        writer = csv.DictWriter(f, fieldnames=cols, delimiter="\t")525        writer.writeheader()526        for sign in signs:527            writer.writerow(sign)528    print(f"  Signs TSV: {len(signs)} signs → {tsv_path}")529 530    # sign_to_ipa.json (only syllabograms with phonetic values)531    sign_map = OrderedDict()532    for sign in signs:533        if sign["Type"] == "syllabogram" and sign["Transliteration"] != "-":534            sign_map[sign["Transliteration"]] = sign["IPA"]535    json_path = OUT_DIR / "sign_to_ipa.json"536    json_path.write_text(json.dumps(sign_map, ensure_ascii=False, indent=2), encoding="utf-8")537    print(f"  sign_to_ipa.json: {len(sign_map)} mappings → {json_path}")538 539    # Stats540    syllabograms = [s for s in signs if s["Type"] == "syllabogram"]541    ideograms = [s for s in signs if s["Type"] == "ideogram"]542    with_phonetic = [s for s in syllabograms if s["Transliteration"] != "-"]543    print(f"  Syllabograms: {len(syllabograms)} ({len(with_phonetic)} with phonetic values)")544    print(f"  Ideograms: {len(ideograms)}")545 546 547def build_word_list(548    shannon_entries: list[dict],549    wiktionary_entries: list[dict],550    iecor_entries: list[dict],551) -> None:552    """Merge all word sources and write linear_b_words.tsv."""553    # Priority order for IPA: IE-CoR (expert) > Wiktionary (ts=) > transliteration conversion554    # Priority order for glosses: Wiktionary > Shannon > IE-CoR (no glosses)555 556    # Index IE-CoR by transliteration557    iecor_by_translit = {}558    for e in iecor_entries:559        t = e["Transliteration"]560        iecor_by_translit[t] = e561 562    # Index Wiktionary by transliteration563    wikt_by_translit = {}564    for e in wiktionary_entries:565        t = e["Transliteration"]566        if t:567            wikt_by_translit[t] = e568 569    # Build merged word list570    # Key: transliteration (hyphenated form like "a-ke-ro")571    all_words = {}  # translit → merged dict572 573    # 1. Start with Shannon entries (largest source)574    for e in shannon_entries:575        t = e["Transliteration"]576        if t not in all_words:577            all_words[t] = {578                "Transliteration": t,579                "Gloss": e["Gloss"],580                "Word_Type": e["Word_Type"],581                "IPA": "-",582                "Source": "shannon_lexicon",583                "Concept_ID": "-",584                "Cognate_Set_ID": "-",585            }586 587    # 2. Merge Wiktionary (better glosses, has IPA)588    for e in wiktionary_entries:589        t = e["Transliteration"]590        if not t:591            continue592        # Skip non-standard transliterations from Wiktionary:593        # - Single letters (measure symbols like L, N, P, Q, S, T, V, Z)594        # - ALL-CAPS abbreviations (AES, KAPO, etc.) — these are ideogram labels595        # - Entries that look like Greek or modern language forms596        if len(t) <= 2 and t.isalpha() and "-" not in t:597            continue598        if t.isupper() and len(t) <= 6:599            continue600        # Valid transliterations use lowercase with hyphens (a-ke-ro)601        # or start with * for undeciphered signs602        if not re.match(r'^[\-a-z0-9*]+$', t.replace("-", "")):603            continue604        if t in all_words:605            # Update gloss if Wiktionary has a better one606            if e["Gloss"] != "-":607                all_words[t]["Gloss"] = e["Gloss"]608            if e["IPA"]:609                all_words[t]["IPA"] = e["IPA"]610            all_words[t]["Source"] = "wiktionary_gmy"611            if e["Word_Type"] != "common":612                all_words[t]["Word_Type"] = e["Word_Type"]613        else:614            all_words[t] = {615                "Transliteration": t,616                "Gloss": e["Gloss"],617                "Word_Type": e["Word_Type"],618                "IPA": e["IPA"] if e["IPA"] else "-",619                "Source": "wiktionary_gmy",620                "Concept_ID": "-",621                "Cognate_Set_ID": "-",622            }623 624    # 3. Merge IE-CoR (best IPA, has concept IDs)625    for e in iecor_entries:626        t = e["Transliteration"]627        if t in all_words:628            # IE-CoR IPA takes priority (expert reconstructions)629            if e["IPA"] and e["IPA"] != "-":630                all_words[t]["IPA"] = e["IPA"]631            if e["Concept_IDs"]:632                all_words[t]["Concept_ID"] = e["Concept_IDs"]633            # Mark as having IE-CoR data634            all_words[t]["Source"] = "iecor+" + all_words[t]["Source"]635        else:636            all_words[t] = {637                "Transliteration": t,638                "Gloss": "-",639                "Word_Type": "common",640                "IPA": e["IPA"],641                "Source": "iecor",642                "Concept_ID": e.get("Concept_IDs", "-"),643                "Cognate_Set_ID": "-",644            }645 646    # 4. For entries without IPA, generate from transliteration647    for t, entry in all_words.items():648        if entry["IPA"] == "-" or not entry["IPA"]:649            entry["IPA"] = transliterate_to_ipa(t)650            if entry["IPA"] != "-":651                entry["IPA_Source"] = "translit_conversion"652            else:653                entry["IPA_Source"] = "none"654        else:655            entry["IPA_Source"] = "expert"656 657    # 5. Compute SCA (Sound Class Alphabet) encoding658    try:659        sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))660        from cognate_pipeline.normalise.sound_class import ipa_to_sound_class661        has_sca = True662    except ImportError:663        has_sca = False664        print("  [WARN] cognate_pipeline not available, SCA will be computed from IPA directly")665 666    for entry in all_words.values():667        if has_sca and entry["IPA"] != "-":668            try:669                entry["SCA"] = ipa_to_sound_class(entry["IPA"])670            except Exception:671                entry["SCA"] = entry["IPA"].upper()672        elif entry["IPA"] != "-":673            # Simple uppercase fallback674            entry["SCA"] = entry["IPA"].upper()675        else:676            entry["SCA"] = "-"677 678    # Write output679    OUT_DIR.mkdir(parents=True, exist_ok=True)680    tsv_path = OUT_DIR / "linear_b_words.tsv"681    cols = ["Word", "IPA", "SCA", "Source", "Concept_ID", "Cognate_Set_ID",682            "Gloss", "Word_Type", "IPA_Source"]683 684    # Sort: common nouns first, then by transliteration685    type_order = {"common": 0, "unknown": 1, "theonym": 2, "ethnic": 3,686                  "proper": 4, "toponym": 5, "anthroponym": 6}687    sorted_entries = sorted(688        all_words.values(),689        key=lambda e: (type_order.get(e["Word_Type"], 9), e["Transliteration"]),690    )691 692    with open(tsv_path, "w", encoding="utf-8", newline="") as f:693        writer = csv.DictWriter(f, fieldnames=cols, delimiter="\t",694                                extrasaction="ignore")695        writer.writeheader()696        for entry in sorted_entries:697            writer.writerow({698                "Word": entry["Transliteration"],699                "IPA": entry["IPA"],700                "SCA": entry["SCA"],701                "Source": entry["Source"],702                "Concept_ID": entry["Concept_ID"],703                "Cognate_Set_ID": entry["Cognate_Set_ID"],704                "Gloss": entry["Gloss"],705                "Word_Type": entry["Word_Type"],706                "IPA_Source": entry.get("IPA_Source", "unknown"),707            })708 709    # Statistics710    total = len(sorted_entries)711    common = sum(1 for e in sorted_entries if e["Word_Type"] == "common")712    proper = total - common713    with_expert_ipa = sum(1 for e in sorted_entries if e.get("IPA_Source") == "expert")714    with_translit_ipa = sum(1 for e in sorted_entries if e.get("IPA_Source") == "translit_conversion")715 716    print(f"\n  Words TSV: {total} entries → {tsv_path}")717    print(f"  Common nouns: {common}")718    print(f"  Proper nouns (names/places): {proper}")719    print(f"  IPA from expert sources: {with_expert_ipa}")720    print(f"  IPA from transliteration conversion: {with_translit_ipa}")721    print(f"  No IPA: {total - with_expert_ipa - with_translit_ipa}")722 723    # Source distribution724    src_counts = {}725    for e in sorted_entries:726        s = e["Source"]727        src_counts[s] = src_counts.get(s, 0) + 1728    print(f"\n  Source distribution:")729    for src, count in sorted(src_counts.items(), key=lambda x: -x[1]):730        print(f"    {src}: {count}")731 732    return sorted_entries733 734 735def main():736    print("=" * 70)737    print("LINEAR B DATASET BUILD")738    print("=" * 70)739 740    # 1. Parse Unicode sign inventory741    print("\n[1/4] Parsing Unicode UCD for Linear B signs...")742    ucd_path = RAW_DIR / "UnicodeData.txt"743    if not ucd_path.exists():744        print("  ERROR: UnicodeData.txt not found. Run ingest_linear_b.py first.")745        sys.exit(1)746    signs = parse_unicode_signs(ucd_path)747    build_sign_inventory(signs)748 749    # 2. Parse Shannon lexicon750    print("\n[2/4] Parsing Shannon Linear B Lexicon...")751    shannon_path = RAW_DIR / "shannon_Linear_B_Lexicon.csv"752    if not shannon_path.exists():753        print("  ERROR: shannon_Linear_B_Lexicon.csv not found. Run ingest_linear_b.py first.")754        sys.exit(1)755    shannon_entries = parse_shannon_lexicon(shannon_path)756    print(f"  Parsed {len(shannon_entries)} entries")757    type_counts = {}758    for e in shannon_entries:759        type_counts[e["Word_Type"]] = type_counts.get(e["Word_Type"], 0) + 1760    for wt, c in sorted(type_counts.items(), key=lambda x: -x[1]):761        print(f"    {wt}: {c}")762 763    # 3. Parse Wiktionary lemmas764    print("\n[3/4] Parsing Wiktionary Mycenaean Greek lemmas...")765    wikt_path = RAW_DIR / "wiktionary_gmy_lemmas.json"766    if not wikt_path.exists():767        print("  ERROR: wiktionary_gmy_lemmas.json not found. Run ingest_linear_b.py first.")768        sys.exit(1)769    wiktionary_entries = parse_wiktionary_lemmas(wikt_path)770    print(f"  Parsed {len(wiktionary_entries)} entries")771    with_ipa = sum(1 for e in wiktionary_entries if e["IPA"])772    with_translit = sum(1 for e in wiktionary_entries if e["Transliteration"])773    with_gloss = sum(1 for e in wiktionary_entries if e["Gloss"] != "-")774    print(f"    With IPA (ts=): {with_ipa}")775    print(f"    With transliteration: {with_translit}")776    print(f"    With gloss: {with_gloss}")777 778    # 4. Load IE-CoR existing data779    print("\n[4/4] Loading IE-CoR Mycenaean Greek data...")780    iecor_entries = load_iecor_gmy_words()781    print(f"  Loaded {len(iecor_entries)} entries from cognate pairs")782 783    # 5. Merge and build word list784    print("\n[BUILD] Merging all sources...")785    entries = build_word_list(shannon_entries, wiktionary_entries, iecor_entries)786 787    print("\n" + "=" * 70)788    print("BUILD COMPLETE")789    print("=" * 70)790    print(f"\nOutput directory: {OUT_DIR}")791    for p in sorted(OUT_DIR.iterdir()):792        print(f"  {p.name}: {p.stat().st_size:,} bytes")793 794 795if __name__ == "__main__":796    main()797