Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Normalize IPA and recompute SCA for all lexicon entries.3 4For every entry in data/training/lexicons/:51. Unicode NFC normalization62. Strip suprasegmentals (stress marks, tone diacritics)73. Validate through tokenize_ipa() and ipa_to_sound_class()84. Flag entries where SCA produces '0' (unknown segments)95. Recompute and store SCA string10 11Handles source-specific quirks:12- WikiPron: multiple pronunciations separated by ', ' -> split into rows13- NorthEuraLex: segments already joined14- ABVD: orthographic forms treated as-is15"""16 17from __future__ import annotations18 19import csv20import sys21import unicodedata22from pathlib import Path23 24ROOT = Path(__file__).resolve().parent.parent25sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))26 27from cognate_pipeline.normalise.sound_class import (28 ipa_to_sound_class,29 tokenize_ipa,30)31 32LEXICONS_DIR = ROOT / "data" / "training" / "lexicons"33METADATA_DIR = ROOT / "data" / "training" / "metadata"34 35HEADER = "Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID\n"36 37# Characters to strip from IPA (suprasegmentals, tone marks)38STRIP_CHARS = {39 "\u02c8", # primary stress ˈ40 "\u02cc", # secondary stress ˌ41 "\u02d0", # length mark ː (keep — SCA ignores it)42 "\u02d1", # half-length ˑ43 "\u0300", # combining grave (tone)44 "\u0301", # combining acute (tone)45 "\u0302", # combining circumflex (tone)46 "\u0303", # combining tilde (nasalization — keep for now)47 "\u0304", # combining macron (tone)48 "\u030b", # combining double acute (tone)49 "\u030c", # combining caron (tone)50 "\u030f", # combining double grave (tone)51}52 53# Only strip tone diacritics, not nasalization54TONE_STRIP = {55 "\u0300", "\u0301", "\u0302", "\u0304", "\u030b", "\u030c", "\u030f",56}57 58 59def normalize_entry_ipa(ipa: str) -> str:60 """Normalize a single IPA string."""61 # NFC first62 ipa = unicodedata.normalize("NFC", ipa)63 # Strip stress marks64 ipa = ipa.replace("\u02c8", "").replace("\u02cc", "")65 # Strip syllable breaks66 ipa = ipa.replace(".", "")67 # Strip tone diacritics68 for ch in TONE_STRIP:69 ipa = ipa.replace(ch, "")70 return ipa.strip()71 72 73def process_lexicon(path: Path) -> tuple[int, int, int, int]:74 """Process a single lexicon file. Returns (total, valid, unknown_segments, dupes)."""75 if not path.exists():76 return 0, 0, 0, 077 78 entries = []79 with open(path, encoding="utf-8", newline="") as f:80 reader = csv.reader(f, delimiter="\t")81 header = next(reader, None)82 if header is None:83 return 0, 0, 0, 084 for row in reader:85 if len(row) < 6:86 row.extend(["-"] * (6 - len(row)))87 entries.append(row)88 89 # Process entries90 normalized = []91 seen = set()92 total = 093 unknown_segments = 094 dupes = 095 96 for row in entries:97 word, ipa, sca_old, source, concept_id, cognate_set_id = (98 row[0], row[1], row[2], row[3], row[4], row[5]99 )100 101 # Handle multiple pronunciations (WikiPron sometimes has them)102 ipa_variants = [ipa]103 if ", " in ipa and source == "wikipron":104 ipa_variants = [v.strip() for v in ipa.split(", ") if v.strip()]105 106 for ipa_v in ipa_variants:107 total += 1108 ipa_clean = normalize_entry_ipa(ipa_v)109 if not ipa_clean:110 continue111 112 # Dedup key113 key = (word, ipa_clean)114 if key in seen:115 dupes += 1116 continue117 seen.add(key)118 119 # Recompute SCA120 sca = ipa_to_sound_class(ipa_clean)121 122 # Check for unknowns123 if "0" in sca:124 unknown_segments += sca.count("0")125 126 normalized.append((word, ipa_clean, sca, source, concept_id, cognate_set_id))127 128 # Write back129 with open(path, "w", encoding="utf-8", newline="") as f:130 f.write(HEADER)131 for word, ipa, sca, source, concept_id, cognate_set_id in sorted(132 normalized, key=lambda x: x[0].lower()133 ):134 f.write(f"{word}\t{ipa}\t{sca}\t{source}\t{concept_id}\t{cognate_set_id}\n")135 136 return total, len(normalized), unknown_segments, dupes137 138 139def main():140 print("=" * 80)141 print("IPA Normalization & SCA Recomputation")142 print("=" * 80)143 144 if not LEXICONS_DIR.exists():145 print("No lexicons directory found. Run ingestion scripts first.")146 return147 148 files = sorted(LEXICONS_DIR.glob("*.tsv"))149 print(f"\n Found {len(files)} lexicon files")150 151 grand_total = 0152 grand_valid = 0153 grand_unknown = 0154 grand_dupes = 0155 total_sca_chars = 0156 problem_langs = []157 158 for path in files:159 total, valid, unknown, dupes = process_lexicon(path)160 grand_total += total161 grand_valid += valid162 grand_unknown += unknown163 grand_dupes += dupes164 165 # Estimate total SCA chars for percentage166 # (approximation: valid entries * avg SCA length ~5)167 if valid > 0:168 # Read actual SCA lengths169 with open(path, encoding="utf-8", newline="") as f:170 reader = csv.reader(f, delimiter="\t")171 next(reader)172 for row in reader:173 if len(row) >= 3:174 total_sca_chars += len(row[2])175 176 if unknown > 0 and valid > 0:177 unknown_rate = unknown / max(total_sca_chars, 1)178 if unknown_rate > 0.05:179 problem_langs.append((path.stem, unknown, valid))180 181 print(f"\n{'=' * 80}")182 print(f"SUMMARY")183 print(f"{'=' * 80}")184 print(f" Total entries processed: {grand_total:,}")185 print(f" Valid entries retained: {grand_valid:,}")186 print(f" Duplicates removed: {grand_dupes:,}")187 if total_sca_chars > 0:188 pct = 100 * grand_unknown / total_sca_chars189 print(f" SCA unknown segments: {grand_unknown:,} / {total_sca_chars:,} ({pct:.2f}%)")190 191 if problem_langs:192 print(f"\n Languages with >5% SCA unknowns:")193 for lang, unk, valid in problem_langs:194 print(f" {lang}: {unk} unknown segments in {valid} entries")195 196 print("\nDone!")197 198 199if __name__ == "__main__":200 main()201 