Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
ingest_cldf_comparative.py264 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Ingest word forms from CLDF comparative datasets (lexibank).3 4Supports multiple CLDF Wordlist datasets from lexibank GitHub repos.5Each dataset has forms.csv with standard columns: Language_ID, Form, etc.6 7Sources:8  - diacl (Carling 2017): Gaulish (xtg), and others9  - iecor (IE-CoR): Sogdian (sog), and others10 11Iron Rule: Data comes from downloaded CSV files. No hardcoded word lists.12 13Usage:14    python scripts/ingest_cldf_comparative.py [--dataset NAME] [--language ISO] [--dry-run]15"""16 17from __future__ import annotations18 19import argparse20import csv21import io22import json23import logging24import re25import sys26import unicodedata27import urllib.request28from pathlib import Path29 30sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")31sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")32 33ROOT = Path(__file__).resolve().parent.parent34sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))35sys.path.insert(0, str(ROOT / "scripts"))36 37from cognate_pipeline.normalise.sound_class import ipa_to_sound_class  # noqa: E40238from transliteration_maps import transliterate  # noqa: E40239 40logger = logging.getLogger(__name__)41 42LEXICON_DIR = ROOT / "data" / "training" / "lexicons"43AUDIT_TRAIL_DIR = ROOT / "data" / "training" / "audit_trails"44RAW_DIR = ROOT / "data" / "training" / "raw"45 46# Dataset configurations47DATASETS = {48    "diacl": {49        "repo": "lexibank/diacl",50        "branch": "master",51        "forms_path": "cldf/forms.csv",52        "languages_path": "cldf/languages.csv",53        "language_id_field": "Language_ID",54        "form_field": "Form",55        "targets": {56            "xtg": {"language_ids": ["39200"], "source_tag": "diacl"},57        },58    },59    "iecor": {60        "repo": "lexibank/iecor",61        "branch": "master",62        "forms_path": "cldf/forms.csv",63        "languages_path": "cldf/languages.csv",64        "language_id_field": "Language_ID",65        "form_field": "Form",66        "targets": {67            "sog": {"language_ids": ["271"], "source_tag": "iecor"},68        },69    },70}71 72USER_AGENT = "PhaiPhon/1.0 (ancient-scripts-datasets)"73 74 75def download_csv(repo: str, branch: str, path: str, local_path: Path) -> None:76    """Download a CSV file from GitHub raw."""77    local_path.parent.mkdir(parents=True, exist_ok=True)78    if local_path.exists():79        logger.info("Cached: %s (%d bytes)", local_path.name, local_path.stat().st_size)80        return81    url = f"https://raw.githubusercontent.com/{repo}/{branch}/{path}"82    logger.info("Downloading %s ...", url)83    req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})84    with urllib.request.urlopen(req, timeout=120) as resp:85        data = resp.read()86    with open(local_path, "wb") as f:87        f.write(data)88    logger.info("Downloaded %s (%d bytes)", local_path.name, len(data))89 90 91def load_existing_words(tsv_path: Path) -> set[str]:92    """Load existing Word column values."""93    existing = set()94    if tsv_path.exists():95        with open(tsv_path, "r", encoding="utf-8") as f:96            for line in f:97                if line.startswith("Word\t"):98                    continue99                word = line.split("\t")[0]100                existing.add(word)101    return existing102 103 104def extract_forms(forms_csv: Path, language_ids: list[str],105                  lang_id_field: str, form_field: str) -> list[dict]:106    """Extract word forms for specific language IDs from a CLDF forms.csv."""107    entries = []108    lang_ids_lower = {lid.lower() for lid in language_ids}109 110    with open(forms_csv, "r", encoding="utf-8") as f:111        reader = csv.DictReader(f)112        for row in reader:113            lid = row.get(lang_id_field, "").lower()114            if lid not in lang_ids_lower:115                continue116 117            form = row.get(form_field, "").strip()118            if not form:119                continue120 121            # Clean form122            form = re.sub(r"^\*+", "", form)  # Remove reconstruction asterisk123            form = re.sub(r"\(.+?\)", "", form)  # Remove parenthetical124            form = unicodedata.normalize("NFC", form.strip())125 126            if not form or len(form) < 2 or len(form) > 50:127                continue128 129            entries.append({130                "word": form,131                "parameter_id": row.get("Parameter_ID", ""),132            })133 134    return entries135 136 137def ingest_language(iso: str, dataset_name: str, config: dict,138                    target: dict, dry_run: bool = False) -> dict:139    """Ingest a single language from a CLDF dataset."""140    # Download forms.csv141    cache_dir = RAW_DIR / f"cldf_{dataset_name}"142    forms_local = cache_dir / "forms.csv"143    download_csv(config["repo"], config["branch"],144                 config["forms_path"], forms_local)145 146    tsv_path = LEXICON_DIR / f"{iso}.tsv"147    existing = load_existing_words(tsv_path)148    logger.info("%s: %d existing entries", iso, len(existing))149 150    # Extract forms151    entries = extract_forms(152        forms_local, target["language_ids"],153        config["language_id_field"], config["form_field"],154    )155    logger.info("%s: %d forms in CLDF", iso, len(entries))156 157    # Process158    new_entries = []159    audit_trail = []160    skipped = 0161    seen = set(existing)162 163    for entry in entries:164        word = entry["word"]165        if word in seen:166            skipped += 1167            continue168 169        try:170            ipa = transliterate(word, iso)171        except Exception:172            ipa = word173 174        if not ipa:175            ipa = word176 177        try:178            sca = ipa_to_sound_class(ipa)179        except Exception:180            sca = ""181 182        new_entries.append({183            "word": word,184            "ipa": ipa,185            "sca": sca,186        })187        seen.add(word)188 189        audit_trail.append({190            "word": word,191            "ipa": ipa,192            "concept": entry["parameter_id"],193            "source": target["source_tag"],194        })195 196    logger.info("%s: %d new, %d skipped", iso, len(new_entries), skipped)197 198    if dry_run:199        return {200            "iso": iso, "dataset": dataset_name,201            "cldf_forms": len(entries), "existing": len(existing),202            "new": len(new_entries), "total": len(seen),203        }204 205    # Write to TSV206    if new_entries:207        LEXICON_DIR.mkdir(parents=True, exist_ok=True)208        if not tsv_path.exists():209            with open(tsv_path, "w", encoding="utf-8") as f:210                f.write("Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID\n")211 212        with open(tsv_path, "a", encoding="utf-8") as f:213            for e in new_entries:214                f.write(f"{e['word']}\t{e['ipa']}\t{e['sca']}\t{target['source_tag']}\t-\t-\n")215 216    # Save audit trail217    if audit_trail:218        AUDIT_TRAIL_DIR.mkdir(parents=True, exist_ok=True)219        audit_path = AUDIT_TRAIL_DIR / f"cldf_{dataset_name}_ingest_{iso}.jsonl"220        with open(audit_path, "w", encoding="utf-8") as f:221            for r in audit_trail:222                f.write(json.dumps(r, ensure_ascii=False) + "\n")223 224    return {225        "iso": iso, "dataset": dataset_name,226        "cldf_forms": len(entries), "existing": len(existing),227        "new": len(new_entries), "total": len(seen),228    }229 230 231def main():232    parser = argparse.ArgumentParser(description="Ingest from CLDF comparative datasets")233    parser.add_argument("--dataset", "-d", help="Specific dataset (default: all)")234    parser.add_argument("--language", "-l", help="Specific ISO code (default: all targets)")235    parser.add_argument("--dry-run", action="store_true")236    args = parser.parse_args()237 238    logging.basicConfig(239        level=logging.INFO,240        format="%(asctime)s %(levelname)s: %(message)s",241        datefmt="%H:%M:%S",242    )243 244    results = []245    for ds_name, config in DATASETS.items():246        if args.dataset and args.dataset != ds_name:247            continue248        for iso, target in config["targets"].items():249            if args.language and args.language != iso:250                continue251            result = ingest_language(iso, ds_name, config, target, dry_run=args.dry_run)252            results.append(result)253 254    print(f"\n{'DRY RUN: ' if args.dry_run else ''}CLDF Comparative Ingestion:")255    print("=" * 60)256    for r in results:257        print(f"  {r['iso']:8s} ({r['dataset']:8s}) cldf={r['cldf_forms']:>5d}, "258              f"existing={r.get('existing', 0):>5d}, new={r['new']:>5d}, total={r['total']:>5d}")259    print("=" * 60)260 261 262if __name__ == "__main__":263    main()264