Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Ingest word forms from CLDF comparative datasets (lexibank).3 4Supports multiple CLDF Wordlist datasets from lexibank GitHub repos.5Each dataset has forms.csv with standard columns: Language_ID, Form, etc.6 7Sources:8 - diacl (Carling 2017): Gaulish (xtg), and others9 - iecor (IE-CoR): Sogdian (sog), and others10 11Iron Rule: Data comes from downloaded CSV files. No hardcoded word lists.12 13Usage:14 python scripts/ingest_cldf_comparative.py [--dataset NAME] [--language ISO] [--dry-run]15"""16 17from __future__ import annotations18 19import argparse20import csv21import io22import json23import logging24import re25import sys26import unicodedata27import urllib.request28from pathlib import Path29 30sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")31sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")32 33ROOT = Path(__file__).resolve().parent.parent34sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))35sys.path.insert(0, str(ROOT / "scripts"))36 37from cognate_pipeline.normalise.sound_class import ipa_to_sound_class # noqa: E40238from transliteration_maps import transliterate # noqa: E40239 40logger = logging.getLogger(__name__)41 42LEXICON_DIR = ROOT / "data" / "training" / "lexicons"43AUDIT_TRAIL_DIR = ROOT / "data" / "training" / "audit_trails"44RAW_DIR = ROOT / "data" / "training" / "raw"45 46# Dataset configurations47DATASETS = {48 "diacl": {49 "repo": "lexibank/diacl",50 "branch": "master",51 "forms_path": "cldf/forms.csv",52 "languages_path": "cldf/languages.csv",53 "language_id_field": "Language_ID",54 "form_field": "Form",55 "targets": {56 "xtg": {"language_ids": ["39200"], "source_tag": "diacl"},57 },58 },59 "iecor": {60 "repo": "lexibank/iecor",61 "branch": "master",62 "forms_path": "cldf/forms.csv",63 "languages_path": "cldf/languages.csv",64 "language_id_field": "Language_ID",65 "form_field": "Form",66 "targets": {67 "sog": {"language_ids": ["271"], "source_tag": "iecor"},68 },69 },70}71 72USER_AGENT = "PhaiPhon/1.0 (ancient-scripts-datasets)"73 74 75def download_csv(repo: str, branch: str, path: str, local_path: Path) -> None:76 """Download a CSV file from GitHub raw."""77 local_path.parent.mkdir(parents=True, exist_ok=True)78 if local_path.exists():79 logger.info("Cached: %s (%d bytes)", local_path.name, local_path.stat().st_size)80 return81 url = f"https://raw.githubusercontent.com/{repo}/{branch}/{path}"82 logger.info("Downloading %s ...", url)83 req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})84 with urllib.request.urlopen(req, timeout=120) as resp:85 data = resp.read()86 with open(local_path, "wb") as f:87 f.write(data)88 logger.info("Downloaded %s (%d bytes)", local_path.name, len(data))89 90 91def load_existing_words(tsv_path: Path) -> set[str]:92 """Load existing Word column values."""93 existing = set()94 if tsv_path.exists():95 with open(tsv_path, "r", encoding="utf-8") as f:96 for line in f:97 if line.startswith("Word\t"):98 continue99 word = line.split("\t")[0]100 existing.add(word)101 return existing102 103 104def extract_forms(forms_csv: Path, language_ids: list[str],105 lang_id_field: str, form_field: str) -> list[dict]:106 """Extract word forms for specific language IDs from a CLDF forms.csv."""107 entries = []108 lang_ids_lower = {lid.lower() for lid in language_ids}109 110 with open(forms_csv, "r", encoding="utf-8") as f:111 reader = csv.DictReader(f)112 for row in reader:113 lid = row.get(lang_id_field, "").lower()114 if lid not in lang_ids_lower:115 continue116 117 form = row.get(form_field, "").strip()118 if not form:119 continue120 121 # Clean form122 form = re.sub(r"^\*+", "", form) # Remove reconstruction asterisk123 form = re.sub(r"\(.+?\)", "", form) # Remove parenthetical124 form = unicodedata.normalize("NFC", form.strip())125 126 if not form or len(form) < 2 or len(form) > 50:127 continue128 129 entries.append({130 "word": form,131 "parameter_id": row.get("Parameter_ID", ""),132 })133 134 return entries135 136 137def ingest_language(iso: str, dataset_name: str, config: dict,138 target: dict, dry_run: bool = False) -> dict:139 """Ingest a single language from a CLDF dataset."""140 # Download forms.csv141 cache_dir = RAW_DIR / f"cldf_{dataset_name}"142 forms_local = cache_dir / "forms.csv"143 download_csv(config["repo"], config["branch"],144 config["forms_path"], forms_local)145 146 tsv_path = LEXICON_DIR / f"{iso}.tsv"147 existing = load_existing_words(tsv_path)148 logger.info("%s: %d existing entries", iso, len(existing))149 150 # Extract forms151 entries = extract_forms(152 forms_local, target["language_ids"],153 config["language_id_field"], config["form_field"],154 )155 logger.info("%s: %d forms in CLDF", iso, len(entries))156 157 # Process158 new_entries = []159 audit_trail = []160 skipped = 0161 seen = set(existing)162 163 for entry in entries:164 word = entry["word"]165 if word in seen:166 skipped += 1167 continue168 169 try:170 ipa = transliterate(word, iso)171 except Exception:172 ipa = word173 174 if not ipa:175 ipa = word176 177 try:178 sca = ipa_to_sound_class(ipa)179 except Exception:180 sca = ""181 182 new_entries.append({183 "word": word,184 "ipa": ipa,185 "sca": sca,186 })187 seen.add(word)188 189 audit_trail.append({190 "word": word,191 "ipa": ipa,192 "concept": entry["parameter_id"],193 "source": target["source_tag"],194 })195 196 logger.info("%s: %d new, %d skipped", iso, len(new_entries), skipped)197 198 if dry_run:199 return {200 "iso": iso, "dataset": dataset_name,201 "cldf_forms": len(entries), "existing": len(existing),202 "new": len(new_entries), "total": len(seen),203 }204 205 # Write to TSV206 if new_entries:207 LEXICON_DIR.mkdir(parents=True, exist_ok=True)208 if not tsv_path.exists():209 with open(tsv_path, "w", encoding="utf-8") as f:210 f.write("Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID\n")211 212 with open(tsv_path, "a", encoding="utf-8") as f:213 for e in new_entries:214 f.write(f"{e['word']}\t{e['ipa']}\t{e['sca']}\t{target['source_tag']}\t-\t-\n")215 216 # Save audit trail217 if audit_trail:218 AUDIT_TRAIL_DIR.mkdir(parents=True, exist_ok=True)219 audit_path = AUDIT_TRAIL_DIR / f"cldf_{dataset_name}_ingest_{iso}.jsonl"220 with open(audit_path, "w", encoding="utf-8") as f:221 for r in audit_trail:222 f.write(json.dumps(r, ensure_ascii=False) + "\n")223 224 return {225 "iso": iso, "dataset": dataset_name,226 "cldf_forms": len(entries), "existing": len(existing),227 "new": len(new_entries), "total": len(seen),228 }229 230 231def main():232 parser = argparse.ArgumentParser(description="Ingest from CLDF comparative datasets")233 parser.add_argument("--dataset", "-d", help="Specific dataset (default: all)")234 parser.add_argument("--language", "-l", help="Specific ISO code (default: all targets)")235 parser.add_argument("--dry-run", action="store_true")236 args = parser.parse_args()237 238 logging.basicConfig(239 level=logging.INFO,240 format="%(asctime)s %(levelname)s: %(message)s",241 datefmt="%H:%M:%S",242 )243 244 results = []245 for ds_name, config in DATASETS.items():246 if args.dataset and args.dataset != ds_name:247 continue248 for iso, target in config["targets"].items():249 if args.language and args.language != iso:250 continue251 result = ingest_language(iso, ds_name, config, target, dry_run=args.dry_run)252 results.append(result)253 254 print(f"\n{'DRY RUN: ' if args.dry_run else ''}CLDF Comparative Ingestion:")255 print("=" * 60)256 for r in results:257 print(f" {r['iso']:8s} ({r['dataset']:8s}) cldf={r['cldf_forms']:>5d}, "258 f"existing={r.get('existing', 0):>5d}, new={r['new']:>5d}, total={r['total']:>5d}")259 print("=" * 60)260 261 262if __name__ == "__main__":263 main()264 