Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
verify_wikipron.py227 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""3Cross-reference lexicon entries sourced from Wikipron against the actual4Wikipron source TSVs to verify every entry is real.5 6Wikipron source: sources/wikipron/data/scrape/tsv/<iso>_<script>_<type>.tsv7  - Two columns (no header): word\tpronunciation (space-separated phones)8 9Lexicon files: data/training/lexicons/<iso>.tsv10  - Header: Word  IPA  SCA  Source  Concept_ID  Cognate_Set_ID11  - We only check rows where Source == "wikipron"12"""13 14import os15import csv16import sys17from pathlib import Path18from collections import defaultdict19 20BASE = Path(r"C:\Users\alvin\hf-ancient-scripts")21WIKIPRON_TSV_DIR = BASE / "sources" / "wikipron" / "data" / "scrape" / "tsv"22LEXICON_DIR = BASE / "data" / "training" / "lexicons"23 24def load_wikipron_words():25    """Load all Wikipron source TSVs and build a lookup: iso_code -> set(words)."""26    iso_words = defaultdict(set)27    tsv_count = 028    entry_count = 029 30    for fname in sorted(os.listdir(WIKIPRON_TSV_DIR)):31        if not fname.endswith(".tsv"):32            continue33        # Extract ISO code from filename like "aar_latn_broad.tsv"34        iso = fname.split("_")[0]35        fpath = WIKIPRON_TSV_DIR / fname36        tsv_count += 137 38        with open(fpath, "r", encoding="utf-8") as f:39            for line in f:40                line = line.rstrip("\n\r")41                if not line:42                    continue43                parts = line.split("\t")44                if len(parts) >= 1:45                    word = parts[0].strip()46                    if word:47                        iso_words[iso].add(word)48                        entry_count += 149 50    print(f"[Wikipron Source] Loaded {tsv_count} TSV files covering {len(iso_words)} languages, {entry_count:,} total word entries")51    return iso_words52 53 54def verify_lexicons(iso_words):55    """For each lexicon file, check that every Source=wikipron entry exists in the source data."""56 57    results = {}  # iso -> {total, found, missing, missing_words}58    total_checked = 059    total_found = 060    total_missing = 061    langs_with_wikipron = 062    langs_no_source_match = []63 64    lexicon_files = sorted([f for f in os.listdir(LEXICON_DIR) if f.endswith(".tsv")])65 66    for fname in lexicon_files:67        iso = fname.replace(".tsv", "")68        fpath = LEXICON_DIR / fname69 70        # Read lexicon entries with Source="wikipron"71        wp_entries = []72        try:73            with open(fpath, "r", encoding="utf-8") as f:74                reader = csv.DictReader(f, delimiter="\t")75                for row in reader:76                    if row.get("Source", "").strip() == "wikipron":77                        wp_entries.append(row["Word"].strip())78        except Exception as e:79            print(f"  ERROR reading {fname}: {e}", file=sys.stderr)80            continue81 82        if not wp_entries:83            continue  # No wikipron entries in this lexicon84 85        langs_with_wikipron += 186 87        # Get the Wikipron source words for this ISO code88        source_words = iso_words.get(iso, set())89 90        if not source_words:91            # No Wikipron source file found for this language92            langs_no_source_match.append((iso, len(wp_entries)))93 94        found = 095        missing = 096        missing_words = []97 98        # Deduplicate: a word may appear multiple times in the lexicon (different pronunciations)99        unique_words = set(wp_entries)100 101        for word in sorted(unique_words):102            if word in source_words:103                found += 1104            else:105                missing += 1106                if len(missing_words) < 20:  # Cap examples107                    missing_words.append(word)108 109        results[iso] = {110            "total_rows": len(wp_entries),111            "unique_words": len(unique_words),112            "found": found,113            "missing": missing,114            "missing_words": missing_words,115            "has_source": bool(source_words),116            "source_size": len(source_words),117        }118 119        total_checked += len(unique_words)120        total_found += found121        total_missing += missing122 123    return results, total_checked, total_found, total_missing, langs_with_wikipron, langs_no_source_match124 125 126def main():127    print("=" * 80)128    print("WIKIPRON CROSS-REFERENCE VERIFICATION")129    print("=" * 80)130    print()131 132    # Step 1: Load Wikipron source data133    print("--- Step 1: Loading Wikipron source data ---")134    iso_words = load_wikipron_words()135    print()136 137    # Step 2: Verify lexicons138    print("--- Step 2: Verifying lexicon entries ---")139    results, total_checked, total_found, total_missing, langs_with_wp, langs_no_source = verify_lexicons(iso_words)140    print()141 142    # Step 3: Summary143    print("=" * 80)144    print("SUMMARY")145    print("=" * 80)146    print(f"Languages with Source=wikipron entries:  {langs_with_wp}")147    print(f"Total unique words checked:              {total_checked:,}")148    print(f"Words verified in Wikipron source:       {total_found:,}")149    print(f"Words NOT found in Wikipron source:      {total_missing:,}")150    if total_checked > 0:151        print(f"Overall match rate:                      {100 * total_found / total_checked:.2f}%")152    print()153 154    # Step 4: Languages with NO Wikipron source file at all155    if langs_no_source:156        print("=" * 80)157        print(f"LANGUAGES WITH NO WIKIPRON SOURCE FILE ({len(langs_no_source)} languages)")158        print("=" * 80)159        print(f"{'ISO':<8} {'Lexicon Entries':>15}")160        print("-" * 25)161        for iso, count in sorted(langs_no_source, key=lambda x: -x[1]):162            print(f"{iso:<8} {count:>15,}")163        print()164 165    # Step 5: Per-language breakdown (all languages)166    print("=" * 80)167    print("PER-LANGUAGE RESULTS (all languages with wikipron entries)")168    print("=" * 80)169    print(f"{'ISO':<8} {'Unique':>8} {'Found':>8} {'Missing':>8} {'Match%':>8} {'SrcSize':>8} {'HasSrc':>6}")170    print("-" * 60)171 172    for iso in sorted(results.keys()):173        r = results[iso]174        match_pct = 100 * r["found"] / r["unique_words"] if r["unique_words"] > 0 else 0.0175        print(f"{iso:<8} {r['unique_words']:>8} {r['found']:>8} {r['missing']:>8} {match_pct:>7.1f}% {r['source_size']:>8} {'Y' if r['has_source'] else 'N':>6}")176    print()177 178    # Step 6: Languages with >5% mismatch rate179    high_mismatch = []180    for iso, r in results.items():181        if r["unique_words"] > 0:182            mismatch_pct = 100 * r["missing"] / r["unique_words"]183            if mismatch_pct > 5.0:184                high_mismatch.append((iso, r, mismatch_pct))185 186    high_mismatch.sort(key=lambda x: -x[2])187 188    print("=" * 80)189    print(f"LANGUAGES WITH >5% MISMATCH RATE ({len(high_mismatch)} languages)")190    print("=" * 80)191    if not high_mismatch:192        print("  (None - all languages have <=5% mismatch rate)")193    else:194        print(f"{'ISO':<8} {'Unique':>8} {'Missing':>8} {'Mismatch%':>10} {'HasSrc':>6}  Sample Missing Words")195        print("-" * 100)196        for iso, r, mm_pct in high_mismatch:197            samples = ", ".join(r["missing_words"][:10])198            print(f"{iso:<8} {r['unique_words']:>8} {r['missing']:>8} {mm_pct:>9.1f}% {'Y' if r['has_source'] else 'N':>6}  {samples}")199    print()200 201    # Step 7: Detailed missing-word examples for top mismatchers202    print("=" * 80)203    print("DETAILED MISSING WORDS (top 30 languages by mismatch count)")204    print("=" * 80)205    by_missing_count = sorted(results.items(), key=lambda x: -x[1]["missing"])206    for iso, r in by_missing_count[:30]:207        if r["missing"] == 0:208            break209        samples = r["missing_words"][:20]210        print(f"\n{iso} ({r['missing']} missing / {r['unique_words']} unique, source_size={r['source_size']}, has_source={'Y' if r['has_source'] else 'N'}):")211        for w in samples:212            print(f"    {w}")213 214    # Final verdict215    print()216    print("=" * 80)217    if total_checked > 0:218        overall = 100 * total_found / total_checked219        print(f"FINAL VERDICT: {overall:.2f}% of all wikipron-sourced lexicon words verified against source data")220    else:221        print("FINAL VERDICT: No wikipron entries found to verify")222    print("=" * 80)223 224 225if __name__ == "__main__":226    main()227