Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Build the Linear B (Mycenaean Greek) dataset from downloaded raw data.3 4Parses and combines data from:5 1. Unicode UCD — Sign inventory (88 syllabograms + 123 ideograms)6 2. jhnwnstd/shannon — Linear B Lexicon (2,747 entries)7 3. Wiktionary — Mycenaean Greek lemmas (~435 entries with IPA)8 4. IE-CoR — Existing 43 Mycenaean Greek (gmy) words with expert IPA9 10Output files:11 data/linear_b/linear_b_signs.tsv — Full sign inventory12 data/linear_b/sign_to_ipa.json — Sign transliteration → IPA mapping13 data/linear_b/linear_b_words.tsv — Word list (Word, IPA, SCA, Source, Concept_ID, Cognate_Set_ID)14 data/linear_b/README.md — Documentation15 16Transliteration → IPA mapping:17 Reference: Ventris & Chadwick (1973) "Documents in Mycenaean Greek", 2nd ed.18 The Linear B syllabary encodes CV syllables. The conventional transliteration19 uses Latin characters that are near-IPA with these systematic differences:20 q = /kʷ/ (labiovelar stop)21 z = /ts/ or /dz/ (affricate, exact value debated)22 j = /j/ (palatal glide)23 w = /w/ (labial glide)24 p2 = /pʰ/ (aspirated p)25 t2 = /tʰ/ (aspirated t) — actually written as "pu2" etc. in convention26 27Usage:28 python scripts/build_linear_b_dataset.py29"""30 31from __future__ import annotations32 33import csv34import io35import json36import re37import sys38import unicodedata39from collections import OrderedDict40from pathlib import Path41 42sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")43sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")44 45ROOT = Path(__file__).resolve().parent.parent46RAW_DIR = ROOT / "data" / "training" / "raw" / "linear_b"47OUT_DIR = ROOT / "data" / "linear_b"48 49# ── Linear B Unicode ranges ──50LINB_SYLLABARY_START = 0x1000051LINB_SYLLABARY_END = 0x1007F52LINB_IDEOGRAM_START = 0x1008053LINB_IDEOGRAM_END = 0x100FF54 55# ── Transliteration → IPA mapping ──56# Reference: Ventris & Chadwick (1973), "Documents in Mycenaean Greek", 2nd ed.57# Palmer (1963), "The Interpretation of Mycenaean Greek Texts"58# Hooker (1980), "Linear B: An Introduction"59#60# The conventional transliteration values are based on the Ventris decipherment61# (1952) and CIPEM standard. Most consonants map directly; the key differences are:62# - q-series represents labiovelars /kʷ/, not /k/63# - z-series represents affricates, transcribed as /ts/ (Hooker 1980: p.68)64# - j represents /j/ (palatal approximant)65# - w represents /w/ (labio-velar approximant)66#67# The "2" variants (a2, a3, pu2, etc.) represent:68# - a2 = /ha/ (initial aspiration)69# - a3 = /ai/ (diphthong)70# - pu2 = /pʰu/ (aspirated)71# - ra2 = /rja/ (palatalized)72# - ro2 = /rjo/73# - ta2 = /tja/74# - nwa = /nwa/75#76# For undeciphered signs (*18, *19, etc.), IPA is left as "-".77 78TRANSLIT_TO_IPA = {79 # Pure vowels80 "a": "a", "e": "e", "i": "i", "o": "o", "u": "u",81 # d-series82 "da": "da", "de": "de", "di": "di", "do": "do", "du": "du",83 # j-series (palatal glide)84 "ja": "ja", "je": "je", "jo": "jo", "ju": "ju",85 # k-series86 "ka": "ka", "ke": "ke", "ki": "ki", "ko": "ko", "ku": "ku",87 # m-series88 "ma": "ma", "me": "me", "mi": "mi", "mo": "mo", "mu": "mu",89 # n-series90 "na": "na", "ne": "ne", "ni": "ni", "no": "no", "nu": "nu",91 # p-series92 "pa": "pa", "pe": "pe", "pi": "pi", "po": "po", "pu": "pu",93 # q-series (labiovelars)94 "qa": "kʷa", "qe": "kʷe", "qi": "kʷi", "qo": "kʷo",95 # r-series (covers both /r/ and /l/ — Linear B does not distinguish)96 "ra": "ra", "re": "re", "ri": "ri", "ro": "ro", "ru": "ru",97 # s-series98 "sa": "sa", "se": "se", "si": "si", "so": "so", "su": "su",99 # t-series100 "ta": "ta", "te": "te", "ti": "ti", "to": "to", "tu": "tu",101 # w-series102 "wa": "wa", "we": "we", "wi": "wi", "wo": "wo",103 # z-series (affricates: Hooker 1980, p.68)104 "za": "tsa", "ze": "tse", "zi": "tsi", "zo": "tso", "zu": "tsu",105 # Special/variant signs106 "a2": "ha", "a3": "ai",107 "nwa": "nwa",108 "pu2": "pʰu",109 "ra2": "rja", "ra3": "rai",110 "ro2": "rjo",111 "ta2": "tja",112 "two": "two",113 "dwe": "dwe",114 "dwo": "dwo",115 "twe": "twe",116 # Undeciphered signs — no IPA117}118 119 120def parse_unicode_signs(ucd_path: Path) -> list[dict]:121 """Parse Linear B signs from UnicodeData.txt.122 123 Each line has format: codepoint;name;category;...124 We extract signs in U+10000-U+100FF range.125 """126 signs = []127 with open(ucd_path, encoding="utf-8") as f:128 for line in f:129 parts = line.strip().split(";")130 if len(parts) < 2:131 continue132 cp_hex = parts[0]133 name = parts[1]134 cp = int(cp_hex, 16)135 136 if LINB_SYLLABARY_START <= cp <= LINB_SYLLABARY_END:137 sign_type = "syllabogram"138 elif LINB_IDEOGRAM_START <= cp <= LINB_IDEOGRAM_END:139 sign_type = "ideogram"140 else:141 continue142 143 # Parse Bennett number and phonetic value from name144 # Format: "LINEAR B SYLLABLE B008 A" or "LINEAR B IDEOGRAM B100 MAN"145 bennett = ""146 phonetic = ""147 m = re.match(r"LINEAR B (?:SYLLABLE|SYMBOL) (B\d+)\s*(.*)", name)148 if m:149 bennett = m.group(1)150 phonetic = m.group(2).strip().lower() if m.group(2) else ""151 else:152 m = re.match(r"LINEAR B IDEOGRAM (B\d+\w*)\s*(.*)", name)153 if m:154 bennett = m.group(1)155 phonetic = m.group(2).strip() if m.group(2) else ""156 157 # Get IPA from transliteration158 ipa = TRANSLIT_TO_IPA.get(phonetic, "-") if phonetic else "-"159 160 signs.append({161 "Codepoint": f"U+{cp_hex}",162 "Unicode_Char": chr(cp),163 "Bennett_Number": bennett,164 "Name": name,165 "Type": sign_type,166 "Transliteration": phonetic if phonetic else "-",167 "IPA": ipa,168 })169 170 return signs171 172 173def parse_shannon_lexicon(csv_path: Path) -> list[dict]:174 """Parse jhnwnstd/shannon Linear_B_Lexicon.csv.175 176 Columns: word (Unicode), transcription (Latin), definition (scholarly notes)177 We extract: transliteration, clean definition, and classify as common/proper noun.178 """179 entries = []180 with open(csv_path, encoding="utf-8") as f:181 reader = csv.DictReader(f)182 for row in reader:183 word_unicode = row.get("word", "").strip()184 translit = row.get("transcription", "").strip()185 definition = row.get("definition", "").strip()186 187 if not translit:188 continue189 190 # Classify: is this a common noun or anthroponym/toponym?191 def_lower = definition.lower()192 is_anthroponym = "anthroponym" in def_lower and ":" not in def_lower.split("anthroponym")[0][-20:]193 is_toponym = "toponym" in def_lower and ":" not in def_lower.split("toponym")[0][-20:]194 195 # Try to extract a clean gloss from the definition196 # Patterns:197 # "Chadwick & Ventris 1973: anthroponym" → type=proper, gloss=anthroponym198 # "Chadwick & Ventris 1973: figs" → type=common, gloss=figs199 gloss = ""200 # Look for meaning after first colon201 colon_parts = definition.split(":", 1)202 if len(colon_parts) > 1:203 after_colon = colon_parts[1].strip()204 # Take the first meaningful phrase (up to next reference or semicolon)205 # Clean up common noise206 gloss_match = re.match(207 r"([\w\s,/()?.!'\-]+?)(?:\s+(?:Chadwick|McArthur|Witczak|van |Palmer|"208 r"Ruijgh|Bernabé|Appears|KN|PY|MY|TH|TI))",209 after_colon,210 )211 if gloss_match:212 gloss = gloss_match.group(1).strip().rstrip(",;.")213 else:214 # Take first 80 chars as fallback215 gloss = after_colon[:80].strip()216 # Cut at first reference-like pattern217 for cutoff in ["Chadwick", "McArthur", "Ventris", "John and"]:218 if cutoff in gloss:219 gloss = gloss[: gloss.index(cutoff)].strip().rstrip(",;.")220 break221 222 # Determine word type223 if "anthroponym" in gloss.lower():224 word_type = "anthroponym"225 elif "toponym" in gloss.lower():226 word_type = "toponym"227 elif "theonym" in gloss.lower():228 word_type = "theonym"229 elif "ethnic" in gloss.lower():230 word_type = "ethnic"231 elif not gloss or gloss.lower() in ("meaning obscure", "meaning unknown",232 "meaning uncertain", "hapax"):233 word_type = "unknown"234 else:235 word_type = "common"236 237 entries.append({238 "Word_Unicode": word_unicode,239 "Transliteration": translit,240 "Gloss": gloss,241 "Word_Type": word_type,242 "Source": "shannon_lexicon",243 })244 245 return entries246 247 248def unicode_to_translit(title: str) -> str:249 """Convert Linear B Unicode characters in a title to transliteration.250 251 Uses Python's unicodedata to get character names, then extracts the252 phonetic value from names like "LINEAR B SYLLABLE B008 A" → "a".253 """254 parts = []255 for ch in title:256 cp = ord(ch)257 if LINB_SYLLABARY_START <= cp <= LINB_SYLLABARY_END:258 try:259 name = unicodedata.name(ch, "")260 m = re.match(r"LINEAR B (?:SYLLABLE|SYMBOL) B\d+\s*(.*)", name)261 if m and m.group(1):262 parts.append(m.group(1).strip().lower())263 else:264 # Undeciphered symbol265 m2 = re.match(r"LINEAR B SYMBOL (B\d+)", name)266 if m2:267 parts.append(f"*{m2.group(1)[1:]}")268 except ValueError:269 pass270 elif LINB_IDEOGRAM_START <= cp <= LINB_IDEOGRAM_END:271 # Ideograms — skip or mark272 try:273 name = unicodedata.name(ch, "")274 m = re.match(r"LINEAR B IDEOGRAM (B\d+\w*)\s*(.*)", name)275 if m:276 parts.append(f"[{m.group(2).strip() or m.group(1)}]")277 except ValueError:278 pass279 # Skip non-Linear B characters (spaces, combining marks, etc.)280 return "-".join(parts) if parts else ""281 282 283def parse_wiktionary_lemmas(json_path: Path) -> list[dict]:284 """Parse Wiktionary Mycenaean Greek lemma data.285 286 Extract from wikitext:287 - ts= parameter → IPA transcription288 - # [[gloss]] → English meaning289 - head template → part of speech290 """291 with open(json_path, encoding="utf-8") as f:292 lemmas = json.load(f)293 294 entries = []295 for lemma in lemmas:296 title = lemma["title"]297 wikitext = lemma["wikitext"]298 299 # Skip if not Mycenaean Greek300 if "==Mycenaean Greek==" not in wikitext:301 continue302 303 # Convert Unicode title to transliteration304 title_translit = unicode_to_translit(title)305 306 # Skip ideogram-only entries (titles that are purely ideograms or *NNN)307 if not title_translit or all(308 p.startswith("[") or p.startswith("*") for p in title_translit.split("-") if p309 ):310 # Check if it has a tr= parameter we could use instead311 tr_check = re.search(r"\|tr=([^|}]+)", wikitext)312 if not tr_check:313 continue314 315 # Extract IPA from ts= parameter — ONLY from {{head|gmy|...}} template,316 # NOT from {{quote|gmy|...}} tablet quotation contexts.317 # The head template has format: {{head|gmy|noun|ts=VALUE}}318 # Quote templates have format: {{quote|gmy|...|ts=FULL_SENTENCE}}319 ipa = ""320 head_match = re.search(r"\{\{(?:head|h)\|gmy\|[^}]*\|ts=([^|}]+)", wikitext)321 if head_match:322 ipa = head_match.group(1).strip()323 # Sanity check: headword IPA should be a single word, not a sentence324 # If it contains spaces or <br>, it's a tablet quotation that leaked in325 if " " in ipa or "<br>" in ipa:326 ipa = ""327 328 # Extract transliteration: prefer Unicode title conversion, fallback to tr= from head template.329 # IMPORTANT: Do NOT use a global tr= search — {{quote-book}} templates contain330 # tr= with full tablet transliterations (e.g., "o-di-do-si du-ru-to-mo / ..."),331 # which are NOT the headword transliteration. Only use tr= from {{head|gmy|...}}.332 translit = title_translit # Primary: Unicode character names → transliteration333 if not translit:334 # Fallback: tr= from head template only335 head_tr_match = re.search(r"\{\{(?:head|h)\|gmy\|[^}]*\|tr=([^|}]+)", wikitext)336 if head_tr_match:337 translit = head_tr_match.group(1).strip()338 339 # Clean transliteration: remove tablet context, bold markers, etc.340 # Wiktionary titles sometimes embed context like "'''di-wo''' u-ta-jo-jo"341 if translit:342 # Remove wikitext bold markers343 translit = translit.replace("'''", "")344 # If transliteration contains spaces (tablet context), take first word only345 if " " in translit:346 translit = translit.split()[0]347 # Remove trailing punctuation348 translit = translit.strip(".,;:!?")349 # Skip if still contains non-transliteration characters350 if re.search(r"[<>\[\]{}|=]", translit):351 continue352 353 # Skip entries with no usable transliteration354 if not translit or translit == "-":355 continue356 357 # Skip pure ideogram/logogram entries (*NNN without syllabic content)358 # These are ideograms like *142, *150, etc. that have no phonetic reading359 translit_parts = [p for p in translit.split("-") if p]360 syllabic_parts = [p for p in translit_parts361 if not p.startswith("*") and not p.startswith("[")]362 if not syllabic_parts:363 continue # Skip: no syllabic content at all364 365 # Extract gloss from definition lines (# [[word]] or # text)366 glosses = []367 for line in wikitext.split("\n"):368 line = line.strip()369 if line.startswith("# ") and not line.startswith("# {{def-uncertain"):370 # Clean wikitext markup371 gloss = line[2:]372 # Remove templates but preserve content for some373 gloss = re.sub(r"\{\{l\|en\|([^|}]+)[^}]*\}\}", r"\1", gloss)374 gloss = re.sub(r"\{\{[^}]*\}\}", "", gloss)375 # Remove links but keep text: [[word|display]] → display, [[word]] → word376 gloss = re.sub(r"\[\[(?:[^|\]]*\|)?([^\]]*)\]\]", r"\1", gloss)377 # Remove remaining markup378 gloss = re.sub(r"['\[\]]", "", gloss)379 # Remove wikitext remnants like }}, {{, etc.380 gloss = re.sub(r"\}\}|\{\{", "", gloss)381 # Remove leading/trailing whitespace and orphaned punctuation382 gloss = gloss.strip().strip(".,;:")383 if gloss and len(gloss) > 1:384 glosses.append(gloss)385 386 # Extract part of speech387 pos = ""388 pos_match = re.search(r"\{\{head\|gmy\|(\w+)", wikitext)389 if pos_match:390 pos = pos_match.group(1)391 392 # Extract etymology cognates (useful for Concept_ID mapping)393 cognates = []394 cog_matches = re.finditer(r"\{\{cog\|grc\|([^|}]+)", wikitext)395 for m in cog_matches:396 cognates.append(m.group(1))397 398 gloss_text = "; ".join(glosses) if glosses else "-"399 400 # Determine word type from POS and content401 word_type = "common"402 if pos == "proper noun":403 word_type = "proper"404 elif "toponym" in gloss_text.lower():405 word_type = "toponym"406 elif "anthroponym" in gloss_text.lower():407 word_type = "anthroponym"408 409 entries.append({410 "Title_Unicode": title,411 "Transliteration": translit,412 "IPA": ipa,413 "Gloss": gloss_text,414 "POS": pos,415 "Word_Type": word_type,416 "Greek_Cognate": cognates[0] if cognates else "-",417 "Source": "wiktionary_gmy",418 })419 420 return entries421 422 423def transliterate_to_ipa(translit: str) -> str:424 """Convert Linear B transliteration to IPA.425 426 Reference: Ventris & Chadwick (1973), Hooker (1980)427 428 Linear B transliterations use the format: syllable-syllable-syllable429 where each syllable is a CV value from the Ventris grid.430 E.g., "a-ke-ro" → "akero", "pa-ka-na" → "pakana"431 """432 if not translit or translit == "-":433 return "-"434 435 # Remove leading/trailing hyphens and whitespace436 translit = translit.strip().strip("-")437 438 # Split on hyphens439 syllables = translit.split("-")440 441 ipa_parts = []442 for syl in syllables:443 syl = syl.strip().lower()444 if not syl:445 continue446 # Check for undeciphered signs (*18, *47, etc.)447 if syl.startswith("*"):448 ipa_parts.append("?")449 continue450 # Look up in mapping451 if syl in TRANSLIT_TO_IPA:452 ipa_parts.append(TRANSLIT_TO_IPA[syl])453 else:454 # Unknown syllable — keep as-is (it may already be a valid value)455 ipa_parts.append(syl)456 457 return "".join(ipa_parts)458 459 460def load_iecor_gmy_words() -> list[dict]:461 """Load existing Mycenaean Greek (gmy) words from cognate pairs Parquet."""462 try:463 import pyarrow.parquet as pq464 import pyarrow.compute as pc465 except ImportError:466 print(" [WARN] pyarrow not available, skipping IE-CoR data")467 return []468 469 parquet_path = ROOT / "data" / "training" / "cognate_pairs" / "cognate_pairs_inherited.parquet"470 if not parquet_path.exists():471 return []472 473 t = pq.read_table(parquet_path)474 mask_a = pc.equal(t["Lang_A"], "gmy")475 mask_b = pc.equal(t["Lang_B"], "gmy")476 477 words = {} # translit → {ipa, concept_ids}478 479 # Extract from Lang_A side480 gmy_a = t.filter(mask_a)481 for i in range(gmy_a.num_rows):482 w = gmy_a.column("Word_A")[i].as_py()483 ipa = gmy_a.column("IPA_A")[i].as_py()484 cid = gmy_a.column("Concept_ID")[i].as_py()485 if w and w != "-":486 if w not in words:487 words[w] = {"ipa": ipa or "-", "concept_ids": set()}488 if cid and cid != "-":489 words[w]["concept_ids"].add(cid)490 491 # Extract from Lang_B side492 gmy_b = t.filter(mask_b)493 for i in range(gmy_b.num_rows):494 w = gmy_b.column("Word_B")[i].as_py()495 ipa = gmy_b.column("IPA_B")[i].as_py()496 cid = gmy_b.column("Concept_ID")[i].as_py()497 if w and w != "-":498 if w not in words:499 words[w] = {"ipa": ipa or "-", "concept_ids": set()}500 if cid and cid != "-":501 words[w]["concept_ids"].add(cid)502 503 result = []504 for translit, data in words.items():505 result.append({506 "Transliteration": translit,507 "IPA": data["ipa"],508 "Concept_IDs": ",".join(sorted(data["concept_ids"])),509 "Source": "iecor",510 })511 512 return result513 514 515def build_sign_inventory(signs: list[dict]) -> None:516 """Write sign inventory TSV and sign_to_ipa.json."""517 OUT_DIR.mkdir(parents=True, exist_ok=True)518 519 # TSV520 tsv_path = OUT_DIR / "linear_b_signs.tsv"521 cols = ["Codepoint", "Unicode_Char", "Bennett_Number", "Name", "Type",522 "Transliteration", "IPA"]523 with open(tsv_path, "w", encoding="utf-8", newline="") as f:524 writer = csv.DictWriter(f, fieldnames=cols, delimiter="\t")525 writer.writeheader()526 for sign in signs:527 writer.writerow(sign)528 print(f" Signs TSV: {len(signs)} signs → {tsv_path}")529 530 # sign_to_ipa.json (only syllabograms with phonetic values)531 sign_map = OrderedDict()532 for sign in signs:533 if sign["Type"] == "syllabogram" and sign["Transliteration"] != "-":534 sign_map[sign["Transliteration"]] = sign["IPA"]535 json_path = OUT_DIR / "sign_to_ipa.json"536 json_path.write_text(json.dumps(sign_map, ensure_ascii=False, indent=2), encoding="utf-8")537 print(f" sign_to_ipa.json: {len(sign_map)} mappings → {json_path}")538 539 # Stats540 syllabograms = [s for s in signs if s["Type"] == "syllabogram"]541 ideograms = [s for s in signs if s["Type"] == "ideogram"]542 with_phonetic = [s for s in syllabograms if s["Transliteration"] != "-"]543 print(f" Syllabograms: {len(syllabograms)} ({len(with_phonetic)} with phonetic values)")544 print(f" Ideograms: {len(ideograms)}")545 546 547def build_word_list(548 shannon_entries: list[dict],549 wiktionary_entries: list[dict],550 iecor_entries: list[dict],551) -> None:552 """Merge all word sources and write linear_b_words.tsv."""553 # Priority order for IPA: IE-CoR (expert) > Wiktionary (ts=) > transliteration conversion554 # Priority order for glosses: Wiktionary > Shannon > IE-CoR (no glosses)555 556 # Index IE-CoR by transliteration557 iecor_by_translit = {}558 for e in iecor_entries:559 t = e["Transliteration"]560 iecor_by_translit[t] = e561 562 # Index Wiktionary by transliteration563 wikt_by_translit = {}564 for e in wiktionary_entries:565 t = e["Transliteration"]566 if t:567 wikt_by_translit[t] = e568 569 # Build merged word list570 # Key: transliteration (hyphenated form like "a-ke-ro")571 all_words = {} # translit → merged dict572 573 # 1. Start with Shannon entries (largest source)574 for e in shannon_entries:575 t = e["Transliteration"]576 if t not in all_words:577 all_words[t] = {578 "Transliteration": t,579 "Gloss": e["Gloss"],580 "Word_Type": e["Word_Type"],581 "IPA": "-",582 "Source": "shannon_lexicon",583 "Concept_ID": "-",584 "Cognate_Set_ID": "-",585 }586 587 # 2. Merge Wiktionary (better glosses, has IPA)588 for e in wiktionary_entries:589 t = e["Transliteration"]590 if not t:591 continue592 # Skip non-standard transliterations from Wiktionary:593 # - Single letters (measure symbols like L, N, P, Q, S, T, V, Z)594 # - ALL-CAPS abbreviations (AES, KAPO, etc.) — these are ideogram labels595 # - Entries that look like Greek or modern language forms596 if len(t) <= 2 and t.isalpha() and "-" not in t:597 continue598 if t.isupper() and len(t) <= 6:599 continue600 # Valid transliterations use lowercase with hyphens (a-ke-ro)601 # or start with * for undeciphered signs602 if not re.match(r'^[\-a-z0-9*]+$', t.replace("-", "")):603 continue604 if t in all_words:605 # Update gloss if Wiktionary has a better one606 if e["Gloss"] != "-":607 all_words[t]["Gloss"] = e["Gloss"]608 if e["IPA"]:609 all_words[t]["IPA"] = e["IPA"]610 all_words[t]["Source"] = "wiktionary_gmy"611 if e["Word_Type"] != "common":612 all_words[t]["Word_Type"] = e["Word_Type"]613 else:614 all_words[t] = {615 "Transliteration": t,616 "Gloss": e["Gloss"],617 "Word_Type": e["Word_Type"],618 "IPA": e["IPA"] if e["IPA"] else "-",619 "Source": "wiktionary_gmy",620 "Concept_ID": "-",621 "Cognate_Set_ID": "-",622 }623 624 # 3. Merge IE-CoR (best IPA, has concept IDs)625 for e in iecor_entries:626 t = e["Transliteration"]627 if t in all_words:628 # IE-CoR IPA takes priority (expert reconstructions)629 if e["IPA"] and e["IPA"] != "-":630 all_words[t]["IPA"] = e["IPA"]631 if e["Concept_IDs"]:632 all_words[t]["Concept_ID"] = e["Concept_IDs"]633 # Mark as having IE-CoR data634 all_words[t]["Source"] = "iecor+" + all_words[t]["Source"]635 else:636 all_words[t] = {637 "Transliteration": t,638 "Gloss": "-",639 "Word_Type": "common",640 "IPA": e["IPA"],641 "Source": "iecor",642 "Concept_ID": e.get("Concept_IDs", "-"),643 "Cognate_Set_ID": "-",644 }645 646 # 4. For entries without IPA, generate from transliteration647 for t, entry in all_words.items():648 if entry["IPA"] == "-" or not entry["IPA"]:649 entry["IPA"] = transliterate_to_ipa(t)650 if entry["IPA"] != "-":651 entry["IPA_Source"] = "translit_conversion"652 else:653 entry["IPA_Source"] = "none"654 else:655 entry["IPA_Source"] = "expert"656 657 # 5. Compute SCA (Sound Class Alphabet) encoding658 try:659 sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))660 from cognate_pipeline.normalise.sound_class import ipa_to_sound_class661 has_sca = True662 except ImportError:663 has_sca = False664 print(" [WARN] cognate_pipeline not available, SCA will be computed from IPA directly")665 666 for entry in all_words.values():667 if has_sca and entry["IPA"] != "-":668 try:669 entry["SCA"] = ipa_to_sound_class(entry["IPA"])670 except Exception:671 entry["SCA"] = entry["IPA"].upper()672 elif entry["IPA"] != "-":673 # Simple uppercase fallback674 entry["SCA"] = entry["IPA"].upper()675 else:676 entry["SCA"] = "-"677 678 # Write output679 OUT_DIR.mkdir(parents=True, exist_ok=True)680 tsv_path = OUT_DIR / "linear_b_words.tsv"681 cols = ["Word", "IPA", "SCA", "Source", "Concept_ID", "Cognate_Set_ID",682 "Gloss", "Word_Type", "IPA_Source"]683 684 # Sort: common nouns first, then by transliteration685 type_order = {"common": 0, "unknown": 1, "theonym": 2, "ethnic": 3,686 "proper": 4, "toponym": 5, "anthroponym": 6}687 sorted_entries = sorted(688 all_words.values(),689 key=lambda e: (type_order.get(e["Word_Type"], 9), e["Transliteration"]),690 )691 692 with open(tsv_path, "w", encoding="utf-8", newline="") as f:693 writer = csv.DictWriter(f, fieldnames=cols, delimiter="\t",694 extrasaction="ignore")695 writer.writeheader()696 for entry in sorted_entries:697 writer.writerow({698 "Word": entry["Transliteration"],699 "IPA": entry["IPA"],700 "SCA": entry["SCA"],701 "Source": entry["Source"],702 "Concept_ID": entry["Concept_ID"],703 "Cognate_Set_ID": entry["Cognate_Set_ID"],704 "Gloss": entry["Gloss"],705 "Word_Type": entry["Word_Type"],706 "IPA_Source": entry.get("IPA_Source", "unknown"),707 })708 709 # Statistics710 total = len(sorted_entries)711 common = sum(1 for e in sorted_entries if e["Word_Type"] == "common")712 proper = total - common713 with_expert_ipa = sum(1 for e in sorted_entries if e.get("IPA_Source") == "expert")714 with_translit_ipa = sum(1 for e in sorted_entries if e.get("IPA_Source") == "translit_conversion")715 716 print(f"\n Words TSV: {total} entries → {tsv_path}")717 print(f" Common nouns: {common}")718 print(f" Proper nouns (names/places): {proper}")719 print(f" IPA from expert sources: {with_expert_ipa}")720 print(f" IPA from transliteration conversion: {with_translit_ipa}")721 print(f" No IPA: {total - with_expert_ipa - with_translit_ipa}")722 723 # Source distribution724 src_counts = {}725 for e in sorted_entries:726 s = e["Source"]727 src_counts[s] = src_counts.get(s, 0) + 1728 print(f"\n Source distribution:")729 for src, count in sorted(src_counts.items(), key=lambda x: -x[1]):730 print(f" {src}: {count}")731 732 return sorted_entries733 734 735def main():736 print("=" * 70)737 print("LINEAR B DATASET BUILD")738 print("=" * 70)739 740 # 1. Parse Unicode sign inventory741 print("\n[1/4] Parsing Unicode UCD for Linear B signs...")742 ucd_path = RAW_DIR / "UnicodeData.txt"743 if not ucd_path.exists():744 print(" ERROR: UnicodeData.txt not found. Run ingest_linear_b.py first.")745 sys.exit(1)746 signs = parse_unicode_signs(ucd_path)747 build_sign_inventory(signs)748 749 # 2. Parse Shannon lexicon750 print("\n[2/4] Parsing Shannon Linear B Lexicon...")751 shannon_path = RAW_DIR / "shannon_Linear_B_Lexicon.csv"752 if not shannon_path.exists():753 print(" ERROR: shannon_Linear_B_Lexicon.csv not found. Run ingest_linear_b.py first.")754 sys.exit(1)755 shannon_entries = parse_shannon_lexicon(shannon_path)756 print(f" Parsed {len(shannon_entries)} entries")757 type_counts = {}758 for e in shannon_entries:759 type_counts[e["Word_Type"]] = type_counts.get(e["Word_Type"], 0) + 1760 for wt, c in sorted(type_counts.items(), key=lambda x: -x[1]):761 print(f" {wt}: {c}")762 763 # 3. Parse Wiktionary lemmas764 print("\n[3/4] Parsing Wiktionary Mycenaean Greek lemmas...")765 wikt_path = RAW_DIR / "wiktionary_gmy_lemmas.json"766 if not wikt_path.exists():767 print(" ERROR: wiktionary_gmy_lemmas.json not found. Run ingest_linear_b.py first.")768 sys.exit(1)769 wiktionary_entries = parse_wiktionary_lemmas(wikt_path)770 print(f" Parsed {len(wiktionary_entries)} entries")771 with_ipa = sum(1 for e in wiktionary_entries if e["IPA"])772 with_translit = sum(1 for e in wiktionary_entries if e["Transliteration"])773 with_gloss = sum(1 for e in wiktionary_entries if e["Gloss"] != "-")774 print(f" With IPA (ts=): {with_ipa}")775 print(f" With transliteration: {with_translit}")776 print(f" With gloss: {with_gloss}")777 778 # 4. Load IE-CoR existing data779 print("\n[4/4] Loading IE-CoR Mycenaean Greek data...")780 iecor_entries = load_iecor_gmy_words()781 print(f" Loaded {len(iecor_entries)} entries from cognate pairs")782 783 # 5. Merge and build word list784 print("\n[BUILD] Merging all sources...")785 entries = build_word_list(shannon_entries, wiktionary_entries, iecor_entries)786 787 print("\n" + "=" * 70)788 print("BUILD COMPLETE")789 print("=" * 70)790 print(f"\nOutput directory: {OUT_DIR}")791 for p in sorted(OUT_DIR.iterdir()):792 print(f" {p.name}: {p.stat().st_size:,} bytes")793 794 795if __name__ == "__main__":796 main()797 