Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
update_metadata.py210 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Update data/training/metadata/languages.tsv from actual TSV files on disk.3 4Scans all .tsv files in data/training/lexicons/, counts entries, extracts5unique sources, and merges into the metadata file. Does NOT remove existing6entries — only adds missing ones or updates counts/sources for existing ones.7 8Iron Rule: This script does NOT add data. It only updates metadata bookkeeping.9 10Usage:11    python scripts/update_metadata.py [--dry-run]12"""13 14from __future__ import annotations15 16import argparse17import logging18import sys19from pathlib import Path20 21ROOT = Path(__file__).resolve().parent.parent22LEXICON_DIR = ROOT / "data" / "training" / "lexicons"23METADATA_PATH = ROOT / "data" / "training" / "metadata" / "languages.tsv"24 25logger = logging.getLogger(__name__)26 27# Known ancient language metadata (ISO -> (Family, Glottocode))28# Sourced from Glottolog and standard comparative linguistics references29ANCIENT_LANG_INFO: dict[str, tuple[str, str]] = {30    "hit": ("anatolian", "hitt1242"),31    "xlw": ("anatolian", "hier1240"),32    "xur": ("hurro_urartian", "urar1245"),33    "xhu": ("hurro_urartian", "hurr1240"),34    "uga": ("semitic", "ugar1238"),35    "phn": ("semitic", "phoe1239"),36    "elx": ("isolates", "elam1244"),37    "ave": ("indo_iranian", "aves1237"),38    "peo": ("indo_iranian", "oldp1254"),39    "ett": ("tyrsenian", "etru1241"),40    "xlc": ("anatolian", "lyci1241"),41    "xld": ("anatolian", "lydi1241"),42    "xcr": ("anatolian", "cari1275"),43    "xpg": ("paleo_balkan", "phry1239"),44    "xle": ("anatolian", "leps1239"),45    "xrr": ("tyrsenian", "raet1238"),46    "cms": ("anatolian", "cimm1244"),47    "txb": ("tocharian", "tokh1243"),48    "xto": ("tocharian", "tokh1242"),49    "ine-pro": ("indo_european", "indo1319"),50    "sem-pro": ("afroasiatic_semitic", "semi1276"),51    "ccs-pro": ("kartvelian", "kart1248"),52    "dra-pro": ("dravidian", "drav1251"),53    # Phase 6-7 additions54    "cop": ("afroasiatic_egyptian", "copt1239"),55    "pli": ("indo_aryan", "pali1273"),56    "xcl": ("armenian", "clas1249"),57    "ang": ("germanic", "olde1238"),58    "gez": ("afroasiatic_semitic", "geez1241"),59    "hbo": ("afroasiatic_semitic", "anci1244"),60    "xht": ("isolates", "hatt1246"),61    "osc": ("italic", "osca1245"),62    "xum": ("italic", "umbr1253"),63    "xve": ("italic", "vene1257"),64    "sga": ("celtic", "oldi1245"),65    "nci": ("uto_aztecan", "clas1250"),66    "ojp": ("japonic", "oldj1239"),67    "pal": ("indo_iranian", "midd1359"),68    "sog": ("indo_iranian", "sogd1245"),69    "xtg": ("celtic", "gaul1239"),70    "gem-pro": ("germanic", ""),71    "cel-pro": ("celtic", ""),72    "urj-pro": ("uralic", ""),73    "bnt-pro": ("atlantic_congo", ""),74    "sit-pro": ("sino_tibetan", ""),75    # Phase 8 Batch 176    "sla-pro": ("balto_slavic", ""),77    "trk-pro": ("turkic", ""),78    "itc-pro": ("italic", ""),79    "jpx-pro": ("japonic", ""),80    "ira-pro": ("indo_iranian", ""),81    "xfa": ("italic", "fali1291"),82    "xlp": ("celtic", "lepo1240"),83    "xce": ("celtic", "celt1248"),84    "xsa": ("afroasiatic_semitic", "olda1245"),85    # Phase 8 Batch 286    "alg-pro": ("algic", ""),87    "sqj-pro": ("albanian", ""),88    "aav-pro": ("austroasiatic", ""),89    "poz-pol-pro": ("austronesian", ""),90    "tai-pro": ("kra_dai", ""),91    "xto-pro": ("tocharian", ""),92    "poz-oce-pro": ("austronesian", ""),93    "xgn-pro": ("mongolic", ""),94    "xmr": ("nilo_saharan", "mero1241"),95    "obm": ("afroasiatic_semitic", "moab1234"),96    # Phase 8 Batch 397    "myn-pro": ("mayan", ""),98    "afa-pro": ("afroasiatic", ""),99    "xib": ("isolates", "iber1248"),100    # Phase 8 Eblaite101    "xeb": ("afroasiatic_semitic", "ebla1241"),102}103 104 105def scan_lexicons() -> dict[str, dict]:106    """Scan all TSV files and return {iso: {entries, sources}}."""107    result = {}108    for tsv_path in sorted(LEXICON_DIR.glob("*.tsv")):109        iso = tsv_path.stem110        with open(tsv_path, "r", encoding="utf-8") as f:111            lines = f.readlines()112        if len(lines) <= 1:113            continue114        entries = len(lines) - 1  # subtract header115        # Extract unique sources from column 3 (0-indexed)116        sources = set()117        for line in lines[1:]:118            parts = line.rstrip("\n").split("\t")119            if len(parts) >= 4 and parts[3] and parts[3] != "-":120                sources.add(parts[3])121        result[iso] = {122            "entries": entries,123            "sources": ",".join(sorted(sources)) if sources else "unknown",124        }125    return result126 127 128def load_metadata() -> tuple[str, dict[str, dict]]:129    """Load existing metadata. Returns (header, {iso: {family, glottocode, entries, sources}})."""130    existing = {}131    header = "ISO\tFamily\tGlottocode\tEntries\tSources\n"132    if METADATA_PATH.exists():133        with open(METADATA_PATH, "r", encoding="utf-8") as f:134            lines = f.readlines()135        if lines:136            header = lines[0]137            for line in lines[1:]:138                parts = line.rstrip("\n").split("\t")139                if len(parts) >= 5:140                    existing[parts[0]] = {141                        "family": parts[1],142                        "glottocode": parts[2],143                        "entries": int(parts[3]) if parts[3].isdigit() else 0,144                        "sources": parts[4],145                    }146    return header, existing147 148 149def main():150    parser = argparse.ArgumentParser(description="Update languages.tsv metadata")151    parser.add_argument("--dry-run", action="store_true", help="Preview without writing")152    args = parser.parse_args()153 154    logging.basicConfig(155        level=logging.INFO,156        format="%(asctime)s %(levelname)s: %(message)s",157        datefmt="%H:%M:%S",158    )159 160    header, existing = load_metadata()161    disk = scan_lexicons()162 163    added = 0164    updated = 0165 166    for iso, info in disk.items():167        if iso in existing:168            # Update entry count and sources if they've changed169            old_entries = existing[iso]["entries"]170            if old_entries != info["entries"] or existing[iso]["sources"] != info["sources"]:171                existing[iso]["entries"] = info["entries"]172                existing[iso]["sources"] = info["sources"]173                updated += 1174                logger.info("UPDATE %s: %d -> %d entries", iso, old_entries, info["entries"])175        else:176            # New language — look up family/glottocode from our table177            if iso in ANCIENT_LANG_INFO:178                family, glottocode = ANCIENT_LANG_INFO[iso]179            else:180                family = "unknown"181                glottocode = ""182            existing[iso] = {183                "family": family,184                "glottocode": glottocode,185                "entries": info["entries"],186                "sources": info["sources"],187            }188            added += 1189            logger.info("ADD %s: %d entries (%s)", iso, info["entries"], family)190 191    # Sort by entry count descending (same as original)192    sorted_isos = sorted(existing.keys(), key=lambda x: existing[x]["entries"], reverse=True)193 194    if not args.dry_run and (added > 0 or updated > 0):195        METADATA_PATH.parent.mkdir(parents=True, exist_ok=True)196        with open(METADATA_PATH, "w", encoding="utf-8") as f:197            f.write(header)198            for iso in sorted_isos:199                e = existing[iso]200                f.write(f"{iso}\t{e['family']}\t{e['glottocode']}\t{e['entries']}\t{e['sources']}\n")201 202    print(f"\n{'DRY RUN: ' if args.dry_run else ''}Metadata update complete:")203    print(f"  Added:   {added}")204    print(f"  Updated: {updated}")205    print(f"  Total:   {len(existing)}")206 207 208if __name__ == "__main__":209    main()210