Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""3Cross-reference lexicon entries sourced from Wikipron against the actual4Wikipron source TSVs to verify every entry is real.5 6Wikipron source: sources/wikipron/data/scrape/tsv/<iso>_<script>_<type>.tsv7 - Two columns (no header): word\tpronunciation (space-separated phones)8 9Lexicon files: data/training/lexicons/<iso>.tsv10 - Header: Word IPA SCA Source Concept_ID Cognate_Set_ID11 - We only check rows where Source == "wikipron"12"""13 14import os15import csv16import sys17from pathlib import Path18from collections import defaultdict19 20BASE = Path(r"C:\Users\alvin\hf-ancient-scripts")21WIKIPRON_TSV_DIR = BASE / "sources" / "wikipron" / "data" / "scrape" / "tsv"22LEXICON_DIR = BASE / "data" / "training" / "lexicons"23 24def load_wikipron_words():25 """Load all Wikipron source TSVs and build a lookup: iso_code -> set(words)."""26 iso_words = defaultdict(set)27 tsv_count = 028 entry_count = 029 30 for fname in sorted(os.listdir(WIKIPRON_TSV_DIR)):31 if not fname.endswith(".tsv"):32 continue33 # Extract ISO code from filename like "aar_latn_broad.tsv"34 iso = fname.split("_")[0]35 fpath = WIKIPRON_TSV_DIR / fname36 tsv_count += 137 38 with open(fpath, "r", encoding="utf-8") as f:39 for line in f:40 line = line.rstrip("\n\r")41 if not line:42 continue43 parts = line.split("\t")44 if len(parts) >= 1:45 word = parts[0].strip()46 if word:47 iso_words[iso].add(word)48 entry_count += 149 50 print(f"[Wikipron Source] Loaded {tsv_count} TSV files covering {len(iso_words)} languages, {entry_count:,} total word entries")51 return iso_words52 53 54def verify_lexicons(iso_words):55 """For each lexicon file, check that every Source=wikipron entry exists in the source data."""56 57 results = {} # iso -> {total, found, missing, missing_words}58 total_checked = 059 total_found = 060 total_missing = 061 langs_with_wikipron = 062 langs_no_source_match = []63 64 lexicon_files = sorted([f for f in os.listdir(LEXICON_DIR) if f.endswith(".tsv")])65 66 for fname in lexicon_files:67 iso = fname.replace(".tsv", "")68 fpath = LEXICON_DIR / fname69 70 # Read lexicon entries with Source="wikipron"71 wp_entries = []72 try:73 with open(fpath, "r", encoding="utf-8") as f:74 reader = csv.DictReader(f, delimiter="\t")75 for row in reader:76 if row.get("Source", "").strip() == "wikipron":77 wp_entries.append(row["Word"].strip())78 except Exception as e:79 print(f" ERROR reading {fname}: {e}", file=sys.stderr)80 continue81 82 if not wp_entries:83 continue # No wikipron entries in this lexicon84 85 langs_with_wikipron += 186 87 # Get the Wikipron source words for this ISO code88 source_words = iso_words.get(iso, set())89 90 if not source_words:91 # No Wikipron source file found for this language92 langs_no_source_match.append((iso, len(wp_entries)))93 94 found = 095 missing = 096 missing_words = []97 98 # Deduplicate: a word may appear multiple times in the lexicon (different pronunciations)99 unique_words = set(wp_entries)100 101 for word in sorted(unique_words):102 if word in source_words:103 found += 1104 else:105 missing += 1106 if len(missing_words) < 20: # Cap examples107 missing_words.append(word)108 109 results[iso] = {110 "total_rows": len(wp_entries),111 "unique_words": len(unique_words),112 "found": found,113 "missing": missing,114 "missing_words": missing_words,115 "has_source": bool(source_words),116 "source_size": len(source_words),117 }118 119 total_checked += len(unique_words)120 total_found += found121 total_missing += missing122 123 return results, total_checked, total_found, total_missing, langs_with_wikipron, langs_no_source_match124 125 126def main():127 print("=" * 80)128 print("WIKIPRON CROSS-REFERENCE VERIFICATION")129 print("=" * 80)130 print()131 132 # Step 1: Load Wikipron source data133 print("--- Step 1: Loading Wikipron source data ---")134 iso_words = load_wikipron_words()135 print()136 137 # Step 2: Verify lexicons138 print("--- Step 2: Verifying lexicon entries ---")139 results, total_checked, total_found, total_missing, langs_with_wp, langs_no_source = verify_lexicons(iso_words)140 print()141 142 # Step 3: Summary143 print("=" * 80)144 print("SUMMARY")145 print("=" * 80)146 print(f"Languages with Source=wikipron entries: {langs_with_wp}")147 print(f"Total unique words checked: {total_checked:,}")148 print(f"Words verified in Wikipron source: {total_found:,}")149 print(f"Words NOT found in Wikipron source: {total_missing:,}")150 if total_checked > 0:151 print(f"Overall match rate: {100 * total_found / total_checked:.2f}%")152 print()153 154 # Step 4: Languages with NO Wikipron source file at all155 if langs_no_source:156 print("=" * 80)157 print(f"LANGUAGES WITH NO WIKIPRON SOURCE FILE ({len(langs_no_source)} languages)")158 print("=" * 80)159 print(f"{'ISO':<8} {'Lexicon Entries':>15}")160 print("-" * 25)161 for iso, count in sorted(langs_no_source, key=lambda x: -x[1]):162 print(f"{iso:<8} {count:>15,}")163 print()164 165 # Step 5: Per-language breakdown (all languages)166 print("=" * 80)167 print("PER-LANGUAGE RESULTS (all languages with wikipron entries)")168 print("=" * 80)169 print(f"{'ISO':<8} {'Unique':>8} {'Found':>8} {'Missing':>8} {'Match%':>8} {'SrcSize':>8} {'HasSrc':>6}")170 print("-" * 60)171 172 for iso in sorted(results.keys()):173 r = results[iso]174 match_pct = 100 * r["found"] / r["unique_words"] if r["unique_words"] > 0 else 0.0175 print(f"{iso:<8} {r['unique_words']:>8} {r['found']:>8} {r['missing']:>8} {match_pct:>7.1f}% {r['source_size']:>8} {'Y' if r['has_source'] else 'N':>6}")176 print()177 178 # Step 6: Languages with >5% mismatch rate179 high_mismatch = []180 for iso, r in results.items():181 if r["unique_words"] > 0:182 mismatch_pct = 100 * r["missing"] / r["unique_words"]183 if mismatch_pct > 5.0:184 high_mismatch.append((iso, r, mismatch_pct))185 186 high_mismatch.sort(key=lambda x: -x[2])187 188 print("=" * 80)189 print(f"LANGUAGES WITH >5% MISMATCH RATE ({len(high_mismatch)} languages)")190 print("=" * 80)191 if not high_mismatch:192 print(" (None - all languages have <=5% mismatch rate)")193 else:194 print(f"{'ISO':<8} {'Unique':>8} {'Missing':>8} {'Mismatch%':>10} {'HasSrc':>6} Sample Missing Words")195 print("-" * 100)196 for iso, r, mm_pct in high_mismatch:197 samples = ", ".join(r["missing_words"][:10])198 print(f"{iso:<8} {r['unique_words']:>8} {r['missing']:>8} {mm_pct:>9.1f}% {'Y' if r['has_source'] else 'N':>6} {samples}")199 print()200 201 # Step 7: Detailed missing-word examples for top mismatchers202 print("=" * 80)203 print("DETAILED MISSING WORDS (top 30 languages by mismatch count)")204 print("=" * 80)205 by_missing_count = sorted(results.items(), key=lambda x: -x[1]["missing"])206 for iso, r in by_missing_count[:30]:207 if r["missing"] == 0:208 break209 samples = r["missing_words"][:20]210 print(f"\n{iso} ({r['missing']} missing / {r['unique_words']} unique, source_size={r['source_size']}, has_source={'Y' if r['has_source'] else 'N'}):")211 for w in samples:212 print(f" {w}")213 214 # Final verdict215 print()216 print("=" * 80)217 if total_checked > 0:218 overall = 100 * total_found / total_checked219 print(f"FINAL VERDICT: {overall:.2f}% of all wikipron-sourced lexicon words verified against source data")220 else:221 print("FINAL VERDICT: No wikipron entries found to verify")222 print("=" * 80)223 224 225if __name__ == "__main__":226 main()227 