Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Tag Sumerogram/logogram entries in cuneiform-language TSV files.3 4Sumerograms are uppercase Sumerian logograms used as shorthand in cuneiform5texts (e.g., LUGAL = king, URU = city, DINGIR = god). They are NOT phonemic6data in the target language. This script tags them in the Concept_ID field7with a 'sumerogram:' prefix so downstream pipelines can filter if needed.8 9The entries are PRESERVED (not deleted) because they are legitimate scholarly10data — they just shouldn't be used for phonetic comparison.11 12Iron Rule: This script does NOT add data. It only modifies Concept_ID tags.13 14Usage:15 python scripts/tag_sumerograms.py [--dry-run] [--language ISO]16"""17 18from __future__ import annotations19 20import argparse21import json22import logging23import re24import sys25from datetime import datetime, timezone26from pathlib import Path27 28ROOT = Path(__file__).resolve().parent.parent29LEXICON_DIR = ROOT / "data" / "training" / "lexicons"30AUDIT_TRAIL_DIR = ROOT / "data" / "training" / "audit_trails"31 32logger = logging.getLogger(__name__)33 34# Languages written in cuneiform that may contain Sumerograms35CUNEIFORM_LANGUAGES = ["hit", "xlw", "xur", "xhu"]36 37# Known Sumerogram patterns38KNOWN_SUMEROGRAMS = {39 "LUGAL", "URU", "DINGIR", "LU", "MUNUS", "GIS", "KUR", "E",40 "AN", "EN", "NIN", "GAL", "TUR", "DUG", "KU6", "ITUD", "ZAG",41 "SAG", "SU", "KA", "UD", "GI", "MES", "HI.A", "KASKAL",42 "NINDA", "A", "KI", "GU4", "UDU", "ANSE", "GIR", "EDIN",43 "SILA", "NUMUN", "NAM", "ALAN", "GUB", "TUG", "SIG", "DUB",44}45 46 47def is_sumerogram(word: str) -> bool:48 """Detect cuneiform Sumerograms (uppercase sign names)."""49 stripped = word.strip("-").strip()50 if not stripped:51 return False52 53 # Pattern 1: All uppercase ASCII, length >= 2 (e.g., LUGAL, URU)54 if stripped.isupper() and stripped.replace(".", "").replace("-", "").isascii() and len(stripped) >= 2:55 return True56 57 # Pattern 2: Compound Sumerograms (e.g., MUNUS.LUGAL, ANSE.KUR.RA)58 if re.match(r"^[A-Z]+(\.[A-Z]+)+$", stripped):59 return True60 61 # Pattern 3: Sumerogram with number (e.g., KU6, AN2)62 if re.match(r"^[A-Z]+\d+$", stripped):63 return True64 65 # Pattern 4: Known Sumerogram followed by Luwian/Hittite suffix66 # e.g., LUGAL-us, URU-as, DINGIR-LIM67 parts = stripped.split("-")68 if parts and parts[0] in KNOWN_SUMEROGRAMS:69 return True70 71 # Pattern 5: Mixed SUMEROGRAM-phonetic (e.g., GEŠTU-, ANŠE.KUR.RA-u-)72 base = parts[0] if parts else stripped73 if base.isupper() and base.isascii() and len(base) >= 2:74 # Check if the base part is all uppercase (Sumerogram base)75 return True76 77 return False78 79 80def tag_tsv(iso: str, tsv_path: Path, dry_run: bool = False) -> dict:81 """Tag Sumerogram entries in a single TSV file."""82 with open(tsv_path, "r", encoding="utf-8") as f:83 lines = f.readlines()84 85 if not lines:86 return {"iso": iso, "total": 0, "tagged": 0}87 88 header = lines[0]89 output = [header]90 tagged_entries = []91 92 for line in lines[1:]:93 parts = line.rstrip("\n").split("\t")94 if len(parts) < 6:95 output.append(line)96 continue97 98 word = parts[0]99 concept_id = parts[4]100 101 if is_sumerogram(word) and not concept_id.startswith("sumerogram:"):102 # Tag it103 if concept_id and concept_id != "-":104 new_concept = f"sumerogram:{concept_id}"105 else:106 new_concept = "sumerogram:unknown"107 parts[4] = new_concept108 tagged_entries.append({"word": word, "old_concept": concept_id, "new_concept": new_concept})109 output.append("\t".join(parts) + "\n")110 else:111 output.append(line)112 113 if tagged_entries and not dry_run:114 with open(tsv_path, "w", encoding="utf-8") as f:115 f.writelines(output)116 117 AUDIT_TRAIL_DIR.mkdir(parents=True, exist_ok=True)118 audit_path = AUDIT_TRAIL_DIR / f"tag_sumerograms_{iso}_{datetime.now(timezone.utc).strftime('%Y%m%d')}.jsonl"119 with open(audit_path, "w", encoding="utf-8") as f:120 for r in tagged_entries:121 f.write(json.dumps(r, ensure_ascii=False) + "\n")122 123 return {124 "iso": iso,125 "total": len(lines) - 1,126 "tagged": len(tagged_entries),127 }128 129 130def main():131 parser = argparse.ArgumentParser(description="Tag Sumerograms in cuneiform-language TSVs")132 parser.add_argument("--language", "-l", help="ISO code (default: all cuneiform languages)")133 parser.add_argument("--dry-run", action="store_true", help="Preview without writing")134 args = parser.parse_args()135 136 logging.basicConfig(137 level=logging.INFO,138 format="%(asctime)s %(levelname)s: %(message)s",139 datefmt="%H:%M:%S",140 )141 142 if args.language:143 isos = [args.language]144 else:145 isos = CUNEIFORM_LANGUAGES146 147 results = []148 for iso in isos:149 tsv_path = LEXICON_DIR / f"{iso}.tsv"150 if not tsv_path.exists():151 logger.info("SKIP %s: file not found", iso)152 continue153 result = tag_tsv(iso, tsv_path, dry_run=args.dry_run)154 results.append(result)155 156 print("\n" + "=" * 60)157 print(f"TAG SUMEROGRAMS {'(DRY RUN)' if args.dry_run else ''}")158 print("=" * 60)159 total_tagged = 0160 for r in results:161 print(f" {r['iso']:10s} total={r['total']}, tagged={r['tagged']}")162 total_tagged += r["tagged"]163 print(f"\n Total tagged: {total_tagged}")164 print("=" * 60)165 166 167if __name__ == "__main__":168 main()169 