Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
normalize_lexicons.py201 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Normalize IPA and recompute SCA for all lexicon entries.3 4For every entry in data/training/lexicons/:51. Unicode NFC normalization62. Strip suprasegmentals (stress marks, tone diacritics)73. Validate through tokenize_ipa() and ipa_to_sound_class()84. Flag entries where SCA produces '0' (unknown segments)95. Recompute and store SCA string10 11Handles source-specific quirks:12- WikiPron: multiple pronunciations separated by ', ' -> split into rows13- NorthEuraLex: segments already joined14- ABVD: orthographic forms treated as-is15"""16 17from __future__ import annotations18 19import csv20import sys21import unicodedata22from pathlib import Path23 24ROOT = Path(__file__).resolve().parent.parent25sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))26 27from cognate_pipeline.normalise.sound_class import (28    ipa_to_sound_class,29    tokenize_ipa,30)31 32LEXICONS_DIR = ROOT / "data" / "training" / "lexicons"33METADATA_DIR = ROOT / "data" / "training" / "metadata"34 35HEADER = "Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID\n"36 37# Characters to strip from IPA (suprasegmentals, tone marks)38STRIP_CHARS = {39    "\u02c8",  # primary stress ˈ40    "\u02cc",  # secondary stress ˌ41    "\u02d0",  # length mark ː (keep — SCA ignores it)42    "\u02d1",  # half-length ˑ43    "\u0300",  # combining grave (tone)44    "\u0301",  # combining acute (tone)45    "\u0302",  # combining circumflex (tone)46    "\u0303",  # combining tilde (nasalization — keep for now)47    "\u0304",  # combining macron (tone)48    "\u030b",  # combining double acute (tone)49    "\u030c",  # combining caron (tone)50    "\u030f",  # combining double grave (tone)51}52 53# Only strip tone diacritics, not nasalization54TONE_STRIP = {55    "\u0300", "\u0301", "\u0302", "\u0304", "\u030b", "\u030c", "\u030f",56}57 58 59def normalize_entry_ipa(ipa: str) -> str:60    """Normalize a single IPA string."""61    # NFC first62    ipa = unicodedata.normalize("NFC", ipa)63    # Strip stress marks64    ipa = ipa.replace("\u02c8", "").replace("\u02cc", "")65    # Strip syllable breaks66    ipa = ipa.replace(".", "")67    # Strip tone diacritics68    for ch in TONE_STRIP:69        ipa = ipa.replace(ch, "")70    return ipa.strip()71 72 73def process_lexicon(path: Path) -> tuple[int, int, int, int]:74    """Process a single lexicon file. Returns (total, valid, unknown_segments, dupes)."""75    if not path.exists():76        return 0, 0, 0, 077 78    entries = []79    with open(path, encoding="utf-8", newline="") as f:80        reader = csv.reader(f, delimiter="\t")81        header = next(reader, None)82        if header is None:83            return 0, 0, 0, 084        for row in reader:85            if len(row) < 6:86                row.extend(["-"] * (6 - len(row)))87            entries.append(row)88 89    # Process entries90    normalized = []91    seen = set()92    total = 093    unknown_segments = 094    dupes = 095 96    for row in entries:97        word, ipa, sca_old, source, concept_id, cognate_set_id = (98            row[0], row[1], row[2], row[3], row[4], row[5]99        )100 101        # Handle multiple pronunciations (WikiPron sometimes has them)102        ipa_variants = [ipa]103        if ", " in ipa and source == "wikipron":104            ipa_variants = [v.strip() for v in ipa.split(", ") if v.strip()]105 106        for ipa_v in ipa_variants:107            total += 1108            ipa_clean = normalize_entry_ipa(ipa_v)109            if not ipa_clean:110                continue111 112            # Dedup key113            key = (word, ipa_clean)114            if key in seen:115                dupes += 1116                continue117            seen.add(key)118 119            # Recompute SCA120            sca = ipa_to_sound_class(ipa_clean)121 122            # Check for unknowns123            if "0" in sca:124                unknown_segments += sca.count("0")125 126            normalized.append((word, ipa_clean, sca, source, concept_id, cognate_set_id))127 128    # Write back129    with open(path, "w", encoding="utf-8", newline="") as f:130        f.write(HEADER)131        for word, ipa, sca, source, concept_id, cognate_set_id in sorted(132            normalized, key=lambda x: x[0].lower()133        ):134            f.write(f"{word}\t{ipa}\t{sca}\t{source}\t{concept_id}\t{cognate_set_id}\n")135 136    return total, len(normalized), unknown_segments, dupes137 138 139def main():140    print("=" * 80)141    print("IPA Normalization & SCA Recomputation")142    print("=" * 80)143 144    if not LEXICONS_DIR.exists():145        print("No lexicons directory found. Run ingestion scripts first.")146        return147 148    files = sorted(LEXICONS_DIR.glob("*.tsv"))149    print(f"\n  Found {len(files)} lexicon files")150 151    grand_total = 0152    grand_valid = 0153    grand_unknown = 0154    grand_dupes = 0155    total_sca_chars = 0156    problem_langs = []157 158    for path in files:159        total, valid, unknown, dupes = process_lexicon(path)160        grand_total += total161        grand_valid += valid162        grand_unknown += unknown163        grand_dupes += dupes164 165        # Estimate total SCA chars for percentage166        # (approximation: valid entries * avg SCA length ~5)167        if valid > 0:168            # Read actual SCA lengths169            with open(path, encoding="utf-8", newline="") as f:170                reader = csv.reader(f, delimiter="\t")171                next(reader)172                for row in reader:173                    if len(row) >= 3:174                        total_sca_chars += len(row[2])175 176        if unknown > 0 and valid > 0:177            unknown_rate = unknown / max(total_sca_chars, 1)178            if unknown_rate > 0.05:179                problem_langs.append((path.stem, unknown, valid))180 181    print(f"\n{'=' * 80}")182    print(f"SUMMARY")183    print(f"{'=' * 80}")184    print(f"  Total entries processed: {grand_total:,}")185    print(f"  Valid entries retained:  {grand_valid:,}")186    print(f"  Duplicates removed:      {grand_dupes:,}")187    if total_sca_chars > 0:188        pct = 100 * grand_unknown / total_sca_chars189        print(f"  SCA unknown segments:    {grand_unknown:,} / {total_sca_chars:,} ({pct:.2f}%)")190 191    if problem_langs:192        print(f"\n  Languages with >5% SCA unknowns:")193        for lang, unk, valid in problem_langs:194            print(f"    {lang}: {unk} unknown segments in {valid} entries")195 196    print("\nDone!")197 198 199if __name__ == "__main__":200    main()201