Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
tag_sumerograms.py169 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Tag Sumerogram/logogram entries in cuneiform-language TSV files.3 4Sumerograms are uppercase Sumerian logograms used as shorthand in cuneiform5texts (e.g., LUGAL = king, URU = city, DINGIR = god). They are NOT phonemic6data in the target language. This script tags them in the Concept_ID field7with a 'sumerogram:' prefix so downstream pipelines can filter if needed.8 9The entries are PRESERVED (not deleted) because they are legitimate scholarly10data — they just shouldn't be used for phonetic comparison.11 12Iron Rule: This script does NOT add data. It only modifies Concept_ID tags.13 14Usage:15    python scripts/tag_sumerograms.py [--dry-run] [--language ISO]16"""17 18from __future__ import annotations19 20import argparse21import json22import logging23import re24import sys25from datetime import datetime, timezone26from pathlib import Path27 28ROOT = Path(__file__).resolve().parent.parent29LEXICON_DIR = ROOT / "data" / "training" / "lexicons"30AUDIT_TRAIL_DIR = ROOT / "data" / "training" / "audit_trails"31 32logger = logging.getLogger(__name__)33 34# Languages written in cuneiform that may contain Sumerograms35CUNEIFORM_LANGUAGES = ["hit", "xlw", "xur", "xhu"]36 37# Known Sumerogram patterns38KNOWN_SUMEROGRAMS = {39    "LUGAL", "URU", "DINGIR", "LU", "MUNUS", "GIS", "KUR", "E",40    "AN", "EN", "NIN", "GAL", "TUR", "DUG", "KU6", "ITUD", "ZAG",41    "SAG", "SU", "KA", "UD", "GI", "MES", "HI.A", "KASKAL",42    "NINDA", "A", "KI", "GU4", "UDU", "ANSE", "GIR", "EDIN",43    "SILA", "NUMUN", "NAM", "ALAN", "GUB", "TUG", "SIG", "DUB",44}45 46 47def is_sumerogram(word: str) -> bool:48    """Detect cuneiform Sumerograms (uppercase sign names)."""49    stripped = word.strip("-").strip()50    if not stripped:51        return False52 53    # Pattern 1: All uppercase ASCII, length >= 2 (e.g., LUGAL, URU)54    if stripped.isupper() and stripped.replace(".", "").replace("-", "").isascii() and len(stripped) >= 2:55        return True56 57    # Pattern 2: Compound Sumerograms (e.g., MUNUS.LUGAL, ANSE.KUR.RA)58    if re.match(r"^[A-Z]+(\.[A-Z]+)+$", stripped):59        return True60 61    # Pattern 3: Sumerogram with number (e.g., KU6, AN2)62    if re.match(r"^[A-Z]+\d+$", stripped):63        return True64 65    # Pattern 4: Known Sumerogram followed by Luwian/Hittite suffix66    # e.g., LUGAL-us, URU-as, DINGIR-LIM67    parts = stripped.split("-")68    if parts and parts[0] in KNOWN_SUMEROGRAMS:69        return True70 71    # Pattern 5: Mixed SUMEROGRAM-phonetic (e.g., GEŠTU-, ANŠE.KUR.RA-u-)72    base = parts[0] if parts else stripped73    if base.isupper() and base.isascii() and len(base) >= 2:74        # Check if the base part is all uppercase (Sumerogram base)75        return True76 77    return False78 79 80def tag_tsv(iso: str, tsv_path: Path, dry_run: bool = False) -> dict:81    """Tag Sumerogram entries in a single TSV file."""82    with open(tsv_path, "r", encoding="utf-8") as f:83        lines = f.readlines()84 85    if not lines:86        return {"iso": iso, "total": 0, "tagged": 0}87 88    header = lines[0]89    output = [header]90    tagged_entries = []91 92    for line in lines[1:]:93        parts = line.rstrip("\n").split("\t")94        if len(parts) < 6:95            output.append(line)96            continue97 98        word = parts[0]99        concept_id = parts[4]100 101        if is_sumerogram(word) and not concept_id.startswith("sumerogram:"):102            # Tag it103            if concept_id and concept_id != "-":104                new_concept = f"sumerogram:{concept_id}"105            else:106                new_concept = "sumerogram:unknown"107            parts[4] = new_concept108            tagged_entries.append({"word": word, "old_concept": concept_id, "new_concept": new_concept})109            output.append("\t".join(parts) + "\n")110        else:111            output.append(line)112 113    if tagged_entries and not dry_run:114        with open(tsv_path, "w", encoding="utf-8") as f:115            f.writelines(output)116 117        AUDIT_TRAIL_DIR.mkdir(parents=True, exist_ok=True)118        audit_path = AUDIT_TRAIL_DIR / f"tag_sumerograms_{iso}_{datetime.now(timezone.utc).strftime('%Y%m%d')}.jsonl"119        with open(audit_path, "w", encoding="utf-8") as f:120            for r in tagged_entries:121                f.write(json.dumps(r, ensure_ascii=False) + "\n")122 123    return {124        "iso": iso,125        "total": len(lines) - 1,126        "tagged": len(tagged_entries),127    }128 129 130def main():131    parser = argparse.ArgumentParser(description="Tag Sumerograms in cuneiform-language TSVs")132    parser.add_argument("--language", "-l", help="ISO code (default: all cuneiform languages)")133    parser.add_argument("--dry-run", action="store_true", help="Preview without writing")134    args = parser.parse_args()135 136    logging.basicConfig(137        level=logging.INFO,138        format="%(asctime)s %(levelname)s: %(message)s",139        datefmt="%H:%M:%S",140    )141 142    if args.language:143        isos = [args.language]144    else:145        isos = CUNEIFORM_LANGUAGES146 147    results = []148    for iso in isos:149        tsv_path = LEXICON_DIR / f"{iso}.tsv"150        if not tsv_path.exists():151            logger.info("SKIP %s: file not found", iso)152            continue153        result = tag_tsv(iso, tsv_path, dry_run=args.dry_run)154        results.append(result)155 156    print("\n" + "=" * 60)157    print(f"TAG SUMEROGRAMS {'(DRY RUN)' if args.dry_run else ''}")158    print("=" * 60)159    total_tagged = 0160    for r in results:161        print(f"  {r['iso']:10s} total={r['total']}, tagged={r['tagged']}")162        total_tagged += r["tagged"]163    print(f"\n  Total tagged: {total_tagged}")164    print("=" * 60)165 166 167if __name__ == "__main__":168    main()169