Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python3
2"""Re-process IPA and SCA columns for ancient language TSV files.
3
4Uses the updated transliteration maps (with NFC normalization, dedicated
5Tocharian/Luwian/Hurrian/Etruscan maps, Avestan script chars, etc.) to
6regenerate the IPA column from the Word column, then recomputes SCA.
7
8Preserves: Word, Source, Concept_ID, Cognate_Set_ID columns unchanged.
9Updates: IPA, SCA columns using current transliteration_maps.transliterate()
10 and cognate_pipeline.normalise.sound_class.ipa_to_sound_class().
11
12Usage:
13 python scripts/reprocess_ipa.py [--dry-run] [--language ISO]
14"""
15
16from __future__ import annotations
17
18import argparse
19import io
20import logging
21import sys
22from pathlib import Path
23
24# Fix Windows encoding
25sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")
26sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")
27
28ROOT = Path(__file__).resolve().parent.parent
29sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))
30sys.path.insert(0, str(ROOT / "scripts"))
31
32from transliteration_maps import transliterate # noqa: E402
33from cognate_pipeline.normalise.sound_class import ipa_to_sound_class # noqa: E402
34
35logger = logging.getLogger(__name__)
36
37LEXICON_DIR = ROOT / "data" / "training" / "lexicons"
38
39# Ancient languages with transliteration maps
40ANCIENT_LANGUAGES = [
41 "hit", "uga", "phn", "xur", "elx", "xlc", "xld", "xcr",
42 "ave", "peo", "ine-pro", "sem-pro", "ccs-pro", "dra-pro",
43 "xpg", "xle", "xrr", "cms", "xlw", "xhu", "ett", "txb", "xto",
44 "non", "got", "chu", "akk", "sux", "gmy",
45 # Tier 2
46 "cop", "pli", "xcl", "ang", "gez", "hbo", "xht",
47 # Tier 3
48 "osc", "xum", "xve", "sga", "xeb", "nci",
49 "ojp", "pal", "sog", "xtg", "xfa", "xlp",
50 # Proto-languages
51 "gem-pro", "cel-pro", "urj-pro", "bnt-pro", "sit-pro",
52 "sla-pro", "trk-pro", "itc-pro", "jpx-pro", "ira-pro",
53 "xce",
54 "xsa",
55 # Phase 8 P1 proto-languages
56 "alg-pro", "sqj-pro", "aav-pro", "poz-pol-pro",
57 "tai-pro", "xto-pro", "poz-oce-pro", "xgn-pro",
58 # Phase 8 additional ancient languages
59 "obm",
60 "xmr",
61 # Batch 3: P2 proto-languages + Iberian
62 "myn-pro",
63 "afa-pro",
64 "xib",
65]
66
67# Map from TSV filename ISO to transliteration map ISO
68# (some TSV files use e.g. "ine-pro" but ALL_MAPS uses "ine")
69ISO_TO_MAP_ISO = {
70 "ine-pro": "ine",
71 "sem-pro": "sem",
72 "ccs-pro": "ccs",
73 "dra-pro": "dra",
74}
75
76HEADER = "Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID\n"
77
78
79def reprocess_file(iso: str, dry_run: bool = False) -> dict:
80 """Re-process a single TSV file."""
81 tsv_path = LEXICON_DIR / f"{iso}.tsv"
82 if not tsv_path.exists():
83 logger.warning("File not found: %s", tsv_path)
84 return {"iso": iso, "status": "not_found"}
85
86 map_iso = ISO_TO_MAP_ISO.get(iso, iso)
87
88 with open(tsv_path, "r", encoding="utf-8") as f:
89 lines = f.readlines()
90
91 # Parse header
92 has_header = lines and lines[0].startswith("Word\t")
93 data_lines = lines[1:] if has_header else lines
94
95 entries = []
96 total = 0
97 old_identity = 0
98 new_identity = 0
99 changed = 0
100 errors = 0
101
102 for line in data_lines:
103 line = line.rstrip("\n\r")
104 if not line.strip():
105 continue
106
107 parts = line.split("\t")
108 if len(parts) < 6:
109 # Pad with defaults if needed
110 while len(parts) < 6:
111 parts.append("-")
112
113 word = parts[0]
114 old_ipa = parts[1]
115 old_sca = parts[2]
116 source = parts[3]
117 concept_id = parts[4]
118 cognate_set_id = parts[5]
119
120 total += 1
121
122 if word == old_ipa:
123 old_identity += 1
124
125 # Re-transliterate using updated maps
126 try:
127 candidate_ipa = transliterate(word, map_iso)
128 except Exception as e:
129 logger.warning("Transliteration error for %s '%s': %s", iso, word, e)
130 candidate_ipa = word # treat as identity (no conversion)
131 errors += 1
132
133 # NEVER-REGRESS RULE:
134 # - If candidate converts the word (candidate != word): use it
135 # (the updated map handled these characters)
136 # - If candidate is identity (candidate == word) but old IPA wasn't:
137 # KEEP old IPA (the old extraction had a conversion we'd lose)
138 # - If both are identity: nothing we can do, keep identity
139 if candidate_ipa != word:
140 # New map successfully converts — use it
141 final_ipa = candidate_ipa
142 elif old_ipa != word:
143 # New map can't convert, but old IPA was non-identity — keep old
144 final_ipa = old_ipa
145 else:
146 # Both identity — no improvement possible
147 final_ipa = word
148
149 # Re-compute SCA from the chosen IPA
150 try:
151 new_sca = ipa_to_sound_class(final_ipa)
152 except Exception:
153 new_sca = old_sca
154
155 if final_ipa == word:
156 new_identity += 1
157
158 if final_ipa != old_ipa:
159 changed += 1
160
161 entries.append({
162 "word": word,
163 "ipa": final_ipa,
164 "sca": new_sca,
165 "source": source,
166 "concept_id": concept_id,
167 "cognate_set_id": cognate_set_id,
168 })
169
170 old_pct = (old_identity / total * 100) if total > 0 else 0
171 new_pct = (new_identity / total * 100) if total > 0 else 0
172
173 result = {
174 "iso": iso,
175 "total": total,
176 "old_identity": old_identity,
177 "new_identity": new_identity,
178 "old_identity_pct": old_pct,
179 "new_identity_pct": new_pct,
180 "changed": changed,
181 "errors": errors,
182 "status": "dry_run" if dry_run else "written",
183 }
184
185 if not dry_run and entries:
186 with open(tsv_path, "w", encoding="utf-8") as f:
187 f.write(HEADER)
188 for e in entries:
189 f.write(
190 f"{e['word']}\t{e['ipa']}\t{e['sca']}\t"
191 f"{e['source']}\t{e['concept_id']}\t{e['cognate_set_id']}\n"
192 )
193
194 return result
195
196
197def main():
198 parser = argparse.ArgumentParser(description="Re-process IPA/SCA for ancient languages")
199 parser.add_argument("--dry-run", action="store_true",
200 help="Show what would change without writing files")
201 parser.add_argument("--language", "-l",
202 help="Process only this ISO code (default: all ancient)")
203 args = parser.parse_args()
204
205 logging.basicConfig(
206 level=logging.INFO,
207 format="%(asctime)s %(levelname)s: %(message)s",
208 datefmt="%H:%M:%S",
209 )
210
211 if args.language:
212 if args.language not in ANCIENT_LANGUAGES:
213 print(f"Unknown language: {args.language}", file=sys.stderr)
214 sys.exit(1)
215 languages = [args.language]
216 else:
217 languages = ANCIENT_LANGUAGES
218
219 mode = "DRY RUN" if args.dry_run else "LIVE"
220 print(f"{'=' * 70}")
221 print(f"IPA Re-processing ({mode})")
222 print(f"Languages: {len(languages)}")
223 print(f"{'=' * 70}")
224 print()
225 print(f"{'ISO':10s} {'Total':>6s} {'OldId%':>7s} {'NewId%':>7s} {'Changed':>8s} {'Errors':>7s}")
226 print("-" * 50)
227
228 results = []
229 for iso in languages:
230 result = reprocess_file(iso, dry_run=args.dry_run)
231 results.append(result)
232 if result["status"] == "not_found":
233 print(f"{iso:10s} NOT FOUND")
234 else:
235 print(
236 f"{iso:10s} {result['total']:6d} "
237 f"{result['old_identity_pct']:6.1f}% "
238 f"{result['new_identity_pct']:6.1f}% "
239 f"{result['changed']:8d} "
240 f"{result['errors']:7d}"
241 )
242
243 print()
244 print(f"{'=' * 70}")
245 print("SUMMARY")
246 total_entries = sum(r.get("total", 0) for r in results)
247 total_changed = sum(r.get("changed", 0) for r in results)
248 total_errors = sum(r.get("errors", 0) for r in results)
249 improved = [r for r in results if r.get("new_identity_pct", 100) < r.get("old_identity_pct", 0)]
250 print(f" Total entries: {total_entries}")
251 print(f" Total changed: {total_changed}")
252 print(f" Total errors: {total_errors}")
253 print(f" Languages improved: {len(improved)}/{len(results)}")
254 print(f"{'=' * 70}")
255
256
257if __name__ == "__main__":
258 main()
259 