Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Update data/training/metadata/languages.tsv from actual TSV files on disk.3 4Scans all .tsv files in data/training/lexicons/, counts entries, extracts5unique sources, and merges into the metadata file. Does NOT remove existing6entries — only adds missing ones or updates counts/sources for existing ones.7 8Iron Rule: This script does NOT add data. It only updates metadata bookkeeping.9 10Usage:11 python scripts/update_metadata.py [--dry-run]12"""13 14from __future__ import annotations15 16import argparse17import logging18import sys19from pathlib import Path20 21ROOT = Path(__file__).resolve().parent.parent22LEXICON_DIR = ROOT / "data" / "training" / "lexicons"23METADATA_PATH = ROOT / "data" / "training" / "metadata" / "languages.tsv"24 25logger = logging.getLogger(__name__)26 27# Known ancient language metadata (ISO -> (Family, Glottocode))28# Sourced from Glottolog and standard comparative linguistics references29ANCIENT_LANG_INFO: dict[str, tuple[str, str]] = {30 "hit": ("anatolian", "hitt1242"),31 "xlw": ("anatolian", "hier1240"),32 "xur": ("hurro_urartian", "urar1245"),33 "xhu": ("hurro_urartian", "hurr1240"),34 "uga": ("semitic", "ugar1238"),35 "phn": ("semitic", "phoe1239"),36 "elx": ("isolates", "elam1244"),37 "ave": ("indo_iranian", "aves1237"),38 "peo": ("indo_iranian", "oldp1254"),39 "ett": ("tyrsenian", "etru1241"),40 "xlc": ("anatolian", "lyci1241"),41 "xld": ("anatolian", "lydi1241"),42 "xcr": ("anatolian", "cari1275"),43 "xpg": ("paleo_balkan", "phry1239"),44 "xle": ("anatolian", "leps1239"),45 "xrr": ("tyrsenian", "raet1238"),46 "cms": ("anatolian", "cimm1244"),47 "txb": ("tocharian", "tokh1243"),48 "xto": ("tocharian", "tokh1242"),49 "ine-pro": ("indo_european", "indo1319"),50 "sem-pro": ("afroasiatic_semitic", "semi1276"),51 "ccs-pro": ("kartvelian", "kart1248"),52 "dra-pro": ("dravidian", "drav1251"),53 # Phase 6-7 additions54 "cop": ("afroasiatic_egyptian", "copt1239"),55 "pli": ("indo_aryan", "pali1273"),56 "xcl": ("armenian", "clas1249"),57 "ang": ("germanic", "olde1238"),58 "gez": ("afroasiatic_semitic", "geez1241"),59 "hbo": ("afroasiatic_semitic", "anci1244"),60 "xht": ("isolates", "hatt1246"),61 "osc": ("italic", "osca1245"),62 "xum": ("italic", "umbr1253"),63 "xve": ("italic", "vene1257"),64 "sga": ("celtic", "oldi1245"),65 "nci": ("uto_aztecan", "clas1250"),66 "ojp": ("japonic", "oldj1239"),67 "pal": ("indo_iranian", "midd1359"),68 "sog": ("indo_iranian", "sogd1245"),69 "xtg": ("celtic", "gaul1239"),70 "gem-pro": ("germanic", ""),71 "cel-pro": ("celtic", ""),72 "urj-pro": ("uralic", ""),73 "bnt-pro": ("atlantic_congo", ""),74 "sit-pro": ("sino_tibetan", ""),75 # Phase 8 Batch 176 "sla-pro": ("balto_slavic", ""),77 "trk-pro": ("turkic", ""),78 "itc-pro": ("italic", ""),79 "jpx-pro": ("japonic", ""),80 "ira-pro": ("indo_iranian", ""),81 "xfa": ("italic", "fali1291"),82 "xlp": ("celtic", "lepo1240"),83 "xce": ("celtic", "celt1248"),84 "xsa": ("afroasiatic_semitic", "olda1245"),85 # Phase 8 Batch 286 "alg-pro": ("algic", ""),87 "sqj-pro": ("albanian", ""),88 "aav-pro": ("austroasiatic", ""),89 "poz-pol-pro": ("austronesian", ""),90 "tai-pro": ("kra_dai", ""),91 "xto-pro": ("tocharian", ""),92 "poz-oce-pro": ("austronesian", ""),93 "xgn-pro": ("mongolic", ""),94 "xmr": ("nilo_saharan", "mero1241"),95 "obm": ("afroasiatic_semitic", "moab1234"),96 # Phase 8 Batch 397 "myn-pro": ("mayan", ""),98 "afa-pro": ("afroasiatic", ""),99 "xib": ("isolates", "iber1248"),100 # Phase 8 Eblaite101 "xeb": ("afroasiatic_semitic", "ebla1241"),102}103 104 105def scan_lexicons() -> dict[str, dict]:106 """Scan all TSV files and return {iso: {entries, sources}}."""107 result = {}108 for tsv_path in sorted(LEXICON_DIR.glob("*.tsv")):109 iso = tsv_path.stem110 with open(tsv_path, "r", encoding="utf-8") as f:111 lines = f.readlines()112 if len(lines) <= 1:113 continue114 entries = len(lines) - 1 # subtract header115 # Extract unique sources from column 3 (0-indexed)116 sources = set()117 for line in lines[1:]:118 parts = line.rstrip("\n").split("\t")119 if len(parts) >= 4 and parts[3] and parts[3] != "-":120 sources.add(parts[3])121 result[iso] = {122 "entries": entries,123 "sources": ",".join(sorted(sources)) if sources else "unknown",124 }125 return result126 127 128def load_metadata() -> tuple[str, dict[str, dict]]:129 """Load existing metadata. Returns (header, {iso: {family, glottocode, entries, sources}})."""130 existing = {}131 header = "ISO\tFamily\tGlottocode\tEntries\tSources\n"132 if METADATA_PATH.exists():133 with open(METADATA_PATH, "r", encoding="utf-8") as f:134 lines = f.readlines()135 if lines:136 header = lines[0]137 for line in lines[1:]:138 parts = line.rstrip("\n").split("\t")139 if len(parts) >= 5:140 existing[parts[0]] = {141 "family": parts[1],142 "glottocode": parts[2],143 "entries": int(parts[3]) if parts[3].isdigit() else 0,144 "sources": parts[4],145 }146 return header, existing147 148 149def main():150 parser = argparse.ArgumentParser(description="Update languages.tsv metadata")151 parser.add_argument("--dry-run", action="store_true", help="Preview without writing")152 args = parser.parse_args()153 154 logging.basicConfig(155 level=logging.INFO,156 format="%(asctime)s %(levelname)s: %(message)s",157 datefmt="%H:%M:%S",158 )159 160 header, existing = load_metadata()161 disk = scan_lexicons()162 163 added = 0164 updated = 0165 166 for iso, info in disk.items():167 if iso in existing:168 # Update entry count and sources if they've changed169 old_entries = existing[iso]["entries"]170 if old_entries != info["entries"] or existing[iso]["sources"] != info["sources"]:171 existing[iso]["entries"] = info["entries"]172 existing[iso]["sources"] = info["sources"]173 updated += 1174 logger.info("UPDATE %s: %d -> %d entries", iso, old_entries, info["entries"])175 else:176 # New language — look up family/glottocode from our table177 if iso in ANCIENT_LANG_INFO:178 family, glottocode = ANCIENT_LANG_INFO[iso]179 else:180 family = "unknown"181 glottocode = ""182 existing[iso] = {183 "family": family,184 "glottocode": glottocode,185 "entries": info["entries"],186 "sources": info["sources"],187 }188 added += 1189 logger.info("ADD %s: %d entries (%s)", iso, info["entries"], family)190 191 # Sort by entry count descending (same as original)192 sorted_isos = sorted(existing.keys(), key=lambda x: existing[x]["entries"], reverse=True)193 194 if not args.dry_run and (added > 0 or updated > 0):195 METADATA_PATH.parent.mkdir(parents=True, exist_ok=True)196 with open(METADATA_PATH, "w", encoding="utf-8") as f:197 f.write(header)198 for iso in sorted_isos:199 e = existing[iso]200 f.write(f"{iso}\t{e['family']}\t{e['glottocode']}\t{e['entries']}\t{e['sources']}\n")201 202 print(f"\n{'DRY RUN: ' if args.dry_run else ''}Metadata update complete:")203 print(f" Added: {added}")204 print(f" Updated: {updated}")205 print(f" Total: {len(existing)}")206 207 208if __name__ == "__main__":209 main()210 