Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
build_linear_a_validation.py170 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""3Build the Linear A Phonotactics Validation dataset.4 5Criteria for inclusion:6  1. Both Lang_A and Lang_B have Linear_A_Score >= 0.807     (open syllable ratio, no clusters, final vowels, CV structure)8  2. Confidence = "certain" (expert-determined, not contested)9  3. Phono_Quality = "strong" (SCA Score >= 0.5, clear phonological similarity)10 11This produces a gold-standard dataset of proven cognate pairs between12languages whose phonotactics resemble Linear A. Used to validate13cognate detection models before applying them to Linear A.14 15Input:16  - analysis/typology_linear_a.tsv (Linear A similarity scores per language)17  - data/training/cognate_pairs/cognate_pairs_inherited.parquet (full dataset)18Output:19  - data/training/cognate_pairs/linear_a_phonotactics_validation.parquet20"""21import csv22import sys23from pathlib import Path24 25if sys.stdout.encoding != 'utf-8':26    sys.stdout.reconfigure(encoding='utf-8')27 28THRESHOLD = 0.8029 30 31def main():32    hf_dir = Path(__file__).parent.parent33 34    # 1. Load typology scores35    print('Loading Linear A typology scores...')36    lang_scores = {}37    with open(hf_dir / 'analysis' / 'typology_linear_a.tsv', encoding='utf-8') as f:38        for row in csv.DictReader(f, delimiter='\t'):39            lang_scores[row['Language']] = float(row['Linear_A_Score'])40 41    qualified_langs = {l for l, s in lang_scores.items() if s >= THRESHOLD}42    print(f'  Languages with score >= {THRESHOLD}: {len(qualified_langs)}')43 44    # 2. Load full Parquet45    import pyarrow as pa46    import pyarrow.parquet as pq47    import pyarrow.compute as pc48 49    print('Loading full inherited Parquet...')50    table = pq.read_table(51        hf_dir / 'data' / 'training' / 'cognate_pairs' / 'cognate_pairs_inherited.parquet'52    )53    print(f'  Total rows: {table.num_rows:,}')54 55    # 3. Filter: both languages qualified, certain confidence, strong phono56    print('Filtering...')57    lang_a = table['Lang_A'].to_pylist()58    lang_b = table['Lang_B'].to_pylist()59    conf = table['Confidence'].to_pylist()60    phono = table['Phono_Quality'].to_pylist()61 62    mask = []63    for i in range(len(lang_a)):64        keep = (65            lang_a[i] in qualified_langs66            and lang_b[i] in qualified_langs67            and conf[i] == 'certain'68            and phono[i] == 'strong'69        )70        mask.append(keep)71 72    mask_arr = pa.array(mask)73    filtered = table.filter(mask_arr)74    print(f'  Filtered rows: {filtered.num_rows:,}')75 76    # 4. Add Linear_A_Score columns for both languages77    scores_a = [lang_scores.get(la, 0.0) for la, m in zip(lang_a, mask) if m]78    scores_b = [lang_scores.get(lb, 0.0) for lb, m in zip(lang_b, mask) if m]79 80    filtered = filtered.append_column(81        'Linear_A_Score_A', pa.array([round(s, 4) for s in scores_a], type=pa.float64())82    )83    filtered = filtered.append_column(84        'Linear_A_Score_B', pa.array([round(s, 4) for s in scores_b], type=pa.float64())85    )86 87    # 5. Write output88    out_path = hf_dir / 'data' / 'training' / 'cognate_pairs' / 'linear_a_phonotactics_validation.parquet'89    pq.write_table(filtered, str(out_path), compression='zstd', compression_level=3)90 91    import os92    size = os.path.getsize(out_path)93    print(f'\n  Written to: {out_path}')94    print(f'  Size: {size/1024/1024:.1f} MB')95 96    # 6. Statistics97    print(f'\n=== VALIDATION DATASET STATISTICS ===')98    print(f'Total pairs: {filtered.num_rows:,}')99 100    # Unique languages101    langs = set()102    fa = filtered['Lang_A'].to_pylist()103    fb = filtered['Lang_B'].to_pylist()104    for a in fa:105        langs.add(a)106    for b in fb:107        langs.add(b)108    print(f'Unique languages: {len(langs)}')109 110    # Load family map111    t2 = pq.read_table(str(hf_dir / 'data' / 'training' / 'metadata' / 'languages.parquet'))112    iso_to_family = dict(zip(t2['ISO'].to_pylist(), t2['Family'].to_pylist()))113 114    # Family distribution115    fam_counts = {}116    for l in langs:117        fam = iso_to_family.get(l, 'unknown')118        fam_counts[fam] = fam_counts.get(fam, 0) + 1119    print(f'\nLanguage families:')120    for fam, c in sorted(fam_counts.items(), key=lambda x: -x[1]):121        print(f'  {fam}: {c}')122 123    # Source distribution124    sources = filtered['Source'].to_pylist()125    src_counts = {}126    for s in sources:127        src_counts[s] = src_counts.get(s, 0) + 1128    print(f'\nSource distribution:')129    for src, c in sorted(src_counts.items(), key=lambda x: -x[1]):130        print(f'  {src}: {c:,}')131 132    # Score distribution133    scores = filtered['Score'].to_pylist()134    float_scores = [float(s) for s in scores if s and s != '-1']135    if float_scores:136        print(f'\nSCA Score distribution:')137        print(f'  Min: {min(float_scores):.4f}')138        print(f'  Max: {max(float_scores):.4f}')139        print(f'  Mean: {sum(float_scores)/len(float_scores):.4f}')140        bins = [(0.5, 0.6), (0.6, 0.7), (0.7, 0.8), (0.8, 0.9), (0.9, 1.01)]141        for lo, hi in bins:142            n = sum(1 for s in float_scores if lo <= s < hi)143            print(f'  [{lo:.1f}, {hi:.1f}): {n:,}')144 145    # Load names for top contributing languages146    iso_to_name = {}147    glot_path = hf_dir / 'data' / 'training' / 'raw' / 'glottolog_cldf' / 'languages.csv'148    with open(glot_path, encoding='utf-8') as f:149        for row in csv.DictReader(f):150            iso = row.get('ISO639P3code', '').strip()151            name = row.get('Name', '').strip()152            if iso and name:153                iso_to_name[iso] = name154 155    # Top 20 languages by pair count156    lang_pair_counts = {}157    for a, b in zip(fa, fb):158        lang_pair_counts[a] = lang_pair_counts.get(a, 0) + 1159        lang_pair_counts[b] = lang_pair_counts.get(b, 0) + 1160    print(f'\nTop 20 languages by pair count:')161    for lang, c in sorted(lang_pair_counts.items(), key=lambda x: -x[1])[:20]:162        name = iso_to_name.get(lang, '?')163        fam = iso_to_family.get(lang, '?')164        score = lang_scores.get(lang, 0)165        print(f'  {lang} ({name}) - {fam} - LA score {score:.4f} - {c:,} pairs')166 167 168if __name__ == '__main__':169    main()170