Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
reprocess_ipa.py259 linesDownload Raw Back to scripts
1#!/usr/bin/env python3
2"""Re-process IPA and SCA columns for ancient language TSV files.
3
4Uses the updated transliteration maps (with NFC normalization, dedicated
5Tocharian/Luwian/Hurrian/Etruscan maps, Avestan script chars, etc.) to
6regenerate the IPA column from the Word column, then recomputes SCA.
7
8Preserves: Word, Source, Concept_ID, Cognate_Set_ID columns unchanged.
9Updates:   IPA, SCA columns using current transliteration_maps.transliterate()
10           and cognate_pipeline.normalise.sound_class.ipa_to_sound_class().
11
12Usage:
13    python scripts/reprocess_ipa.py [--dry-run] [--language ISO]
14"""
15
16from __future__ import annotations
17
18import argparse
19import io
20import logging
21import sys
22from pathlib import Path
23
24# Fix Windows encoding
25sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")
26sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")
27
28ROOT = Path(__file__).resolve().parent.parent
29sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))
30sys.path.insert(0, str(ROOT / "scripts"))
31
32from transliteration_maps import transliterate  # noqa: E402
33from cognate_pipeline.normalise.sound_class import ipa_to_sound_class  # noqa: E402
34
35logger = logging.getLogger(__name__)
36
37LEXICON_DIR = ROOT / "data" / "training" / "lexicons"
38
39# Ancient languages with transliteration maps
40ANCIENT_LANGUAGES = [
41    "hit", "uga", "phn", "xur", "elx", "xlc", "xld", "xcr",
42    "ave", "peo", "ine-pro", "sem-pro", "ccs-pro", "dra-pro",
43    "xpg", "xle", "xrr", "cms", "xlw", "xhu", "ett", "txb", "xto",
44    "non", "got", "chu", "akk", "sux", "gmy",
45    # Tier 2
46    "cop", "pli", "xcl", "ang", "gez", "hbo", "xht",
47    # Tier 3
48    "osc", "xum", "xve", "sga", "xeb", "nci",
49    "ojp", "pal", "sog", "xtg", "xfa", "xlp",
50    # Proto-languages
51    "gem-pro", "cel-pro", "urj-pro", "bnt-pro", "sit-pro",
52    "sla-pro", "trk-pro", "itc-pro", "jpx-pro", "ira-pro",
53    "xce",
54    "xsa",
55    # Phase 8 P1 proto-languages
56    "alg-pro", "sqj-pro", "aav-pro", "poz-pol-pro",
57    "tai-pro", "xto-pro", "poz-oce-pro", "xgn-pro",
58    # Phase 8 additional ancient languages
59    "obm",
60    "xmr",
61    # Batch 3: P2 proto-languages + Iberian
62    "myn-pro",
63    "afa-pro",
64    "xib",
65]
66
67# Map from TSV filename ISO to transliteration map ISO
68# (some TSV files use e.g. "ine-pro" but ALL_MAPS uses "ine")
69ISO_TO_MAP_ISO = {
70    "ine-pro": "ine",
71    "sem-pro": "sem",
72    "ccs-pro": "ccs",
73    "dra-pro": "dra",
74}
75
76HEADER = "Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID\n"
77
78
79def reprocess_file(iso: str, dry_run: bool = False) -> dict:
80    """Re-process a single TSV file."""
81    tsv_path = LEXICON_DIR / f"{iso}.tsv"
82    if not tsv_path.exists():
83        logger.warning("File not found: %s", tsv_path)
84        return {"iso": iso, "status": "not_found"}
85
86    map_iso = ISO_TO_MAP_ISO.get(iso, iso)
87
88    with open(tsv_path, "r", encoding="utf-8") as f:
89        lines = f.readlines()
90
91    # Parse header
92    has_header = lines and lines[0].startswith("Word\t")
93    data_lines = lines[1:] if has_header else lines
94
95    entries = []
96    total = 0
97    old_identity = 0
98    new_identity = 0
99    changed = 0
100    errors = 0
101
102    for line in data_lines:
103        line = line.rstrip("\n\r")
104        if not line.strip():
105            continue
106
107        parts = line.split("\t")
108        if len(parts) < 6:
109            # Pad with defaults if needed
110            while len(parts) < 6:
111                parts.append("-")
112
113        word = parts[0]
114        old_ipa = parts[1]
115        old_sca = parts[2]
116        source = parts[3]
117        concept_id = parts[4]
118        cognate_set_id = parts[5]
119
120        total += 1
121
122        if word == old_ipa:
123            old_identity += 1
124
125        # Re-transliterate using updated maps
126        try:
127            candidate_ipa = transliterate(word, map_iso)
128        except Exception as e:
129            logger.warning("Transliteration error for %s '%s': %s", iso, word, e)
130            candidate_ipa = word  # treat as identity (no conversion)
131            errors += 1
132
133        # NEVER-REGRESS RULE:
134        # - If candidate converts the word (candidate != word): use it
135        #   (the updated map handled these characters)
136        # - If candidate is identity (candidate == word) but old IPA wasn't:
137        #   KEEP old IPA (the old extraction had a conversion we'd lose)
138        # - If both are identity: nothing we can do, keep identity
139        if candidate_ipa != word:
140            # New map successfully converts — use it
141            final_ipa = candidate_ipa
142        elif old_ipa != word:
143            # New map can't convert, but old IPA was non-identity — keep old
144            final_ipa = old_ipa
145        else:
146            # Both identity — no improvement possible
147            final_ipa = word
148
149        # Re-compute SCA from the chosen IPA
150        try:
151            new_sca = ipa_to_sound_class(final_ipa)
152        except Exception:
153            new_sca = old_sca
154
155        if final_ipa == word:
156            new_identity += 1
157
158        if final_ipa != old_ipa:
159            changed += 1
160
161        entries.append({
162            "word": word,
163            "ipa": final_ipa,
164            "sca": new_sca,
165            "source": source,
166            "concept_id": concept_id,
167            "cognate_set_id": cognate_set_id,
168        })
169
170    old_pct = (old_identity / total * 100) if total > 0 else 0
171    new_pct = (new_identity / total * 100) if total > 0 else 0
172
173    result = {
174        "iso": iso,
175        "total": total,
176        "old_identity": old_identity,
177        "new_identity": new_identity,
178        "old_identity_pct": old_pct,
179        "new_identity_pct": new_pct,
180        "changed": changed,
181        "errors": errors,
182        "status": "dry_run" if dry_run else "written",
183    }
184
185    if not dry_run and entries:
186        with open(tsv_path, "w", encoding="utf-8") as f:
187            f.write(HEADER)
188            for e in entries:
189                f.write(
190                    f"{e['word']}\t{e['ipa']}\t{e['sca']}\t"
191                    f"{e['source']}\t{e['concept_id']}\t{e['cognate_set_id']}\n"
192                )
193
194    return result
195
196
197def main():
198    parser = argparse.ArgumentParser(description="Re-process IPA/SCA for ancient languages")
199    parser.add_argument("--dry-run", action="store_true",
200                        help="Show what would change without writing files")
201    parser.add_argument("--language", "-l",
202                        help="Process only this ISO code (default: all ancient)")
203    args = parser.parse_args()
204
205    logging.basicConfig(
206        level=logging.INFO,
207        format="%(asctime)s %(levelname)s: %(message)s",
208        datefmt="%H:%M:%S",
209    )
210
211    if args.language:
212        if args.language not in ANCIENT_LANGUAGES:
213            print(f"Unknown language: {args.language}", file=sys.stderr)
214            sys.exit(1)
215        languages = [args.language]
216    else:
217        languages = ANCIENT_LANGUAGES
218
219    mode = "DRY RUN" if args.dry_run else "LIVE"
220    print(f"{'=' * 70}")
221    print(f"IPA Re-processing ({mode})")
222    print(f"Languages: {len(languages)}")
223    print(f"{'=' * 70}")
224    print()
225    print(f"{'ISO':10s} {'Total':>6s} {'OldId%':>7s} {'NewId%':>7s} {'Changed':>8s} {'Errors':>7s}")
226    print("-" * 50)
227
228    results = []
229    for iso in languages:
230        result = reprocess_file(iso, dry_run=args.dry_run)
231        results.append(result)
232        if result["status"] == "not_found":
233            print(f"{iso:10s} NOT FOUND")
234        else:
235            print(
236                f"{iso:10s} {result['total']:6d} "
237                f"{result['old_identity_pct']:6.1f}% "
238                f"{result['new_identity_pct']:6.1f}% "
239                f"{result['changed']:8d} "
240                f"{result['errors']:7d}"
241            )
242
243    print()
244    print(f"{'=' * 70}")
245    print("SUMMARY")
246    total_entries = sum(r.get("total", 0) for r in results)
247    total_changed = sum(r.get("changed", 0) for r in results)
248    total_errors = sum(r.get("errors", 0) for r in results)
249    improved = [r for r in results if r.get("new_identity_pct", 100) < r.get("old_identity_pct", 0)]
250    print(f"  Total entries:  {total_entries}")
251    print(f"  Total changed:  {total_changed}")
252    print(f"  Total errors:   {total_errors}")
253    print(f"  Languages improved: {len(improved)}/{len(results)}")
254    print(f"{'=' * 70}")
255
256
257if __name__ == "__main__":
258    main()
259