Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""3Build the Linear A Phonotactics Validation dataset.4 5Criteria for inclusion:6 1. Both Lang_A and Lang_B have Linear_A_Score >= 0.807 (open syllable ratio, no clusters, final vowels, CV structure)8 2. Confidence = "certain" (expert-determined, not contested)9 3. Phono_Quality = "strong" (SCA Score >= 0.5, clear phonological similarity)10 11This produces a gold-standard dataset of proven cognate pairs between12languages whose phonotactics resemble Linear A. Used to validate13cognate detection models before applying them to Linear A.14 15Input:16 - analysis/typology_linear_a.tsv (Linear A similarity scores per language)17 - data/training/cognate_pairs/cognate_pairs_inherited.parquet (full dataset)18Output:19 - data/training/cognate_pairs/linear_a_phonotactics_validation.parquet20"""21import csv22import sys23from pathlib import Path24 25if sys.stdout.encoding != 'utf-8':26 sys.stdout.reconfigure(encoding='utf-8')27 28THRESHOLD = 0.8029 30 31def main():32 hf_dir = Path(__file__).parent.parent33 34 # 1. Load typology scores35 print('Loading Linear A typology scores...')36 lang_scores = {}37 with open(hf_dir / 'analysis' / 'typology_linear_a.tsv', encoding='utf-8') as f:38 for row in csv.DictReader(f, delimiter='\t'):39 lang_scores[row['Language']] = float(row['Linear_A_Score'])40 41 qualified_langs = {l for l, s in lang_scores.items() if s >= THRESHOLD}42 print(f' Languages with score >= {THRESHOLD}: {len(qualified_langs)}')43 44 # 2. Load full Parquet45 import pyarrow as pa46 import pyarrow.parquet as pq47 import pyarrow.compute as pc48 49 print('Loading full inherited Parquet...')50 table = pq.read_table(51 hf_dir / 'data' / 'training' / 'cognate_pairs' / 'cognate_pairs_inherited.parquet'52 )53 print(f' Total rows: {table.num_rows:,}')54 55 # 3. Filter: both languages qualified, certain confidence, strong phono56 print('Filtering...')57 lang_a = table['Lang_A'].to_pylist()58 lang_b = table['Lang_B'].to_pylist()59 conf = table['Confidence'].to_pylist()60 phono = table['Phono_Quality'].to_pylist()61 62 mask = []63 for i in range(len(lang_a)):64 keep = (65 lang_a[i] in qualified_langs66 and lang_b[i] in qualified_langs67 and conf[i] == 'certain'68 and phono[i] == 'strong'69 )70 mask.append(keep)71 72 mask_arr = pa.array(mask)73 filtered = table.filter(mask_arr)74 print(f' Filtered rows: {filtered.num_rows:,}')75 76 # 4. Add Linear_A_Score columns for both languages77 scores_a = [lang_scores.get(la, 0.0) for la, m in zip(lang_a, mask) if m]78 scores_b = [lang_scores.get(lb, 0.0) for lb, m in zip(lang_b, mask) if m]79 80 filtered = filtered.append_column(81 'Linear_A_Score_A', pa.array([round(s, 4) for s in scores_a], type=pa.float64())82 )83 filtered = filtered.append_column(84 'Linear_A_Score_B', pa.array([round(s, 4) for s in scores_b], type=pa.float64())85 )86 87 # 5. Write output88 out_path = hf_dir / 'data' / 'training' / 'cognate_pairs' / 'linear_a_phonotactics_validation.parquet'89 pq.write_table(filtered, str(out_path), compression='zstd', compression_level=3)90 91 import os92 size = os.path.getsize(out_path)93 print(f'\n Written to: {out_path}')94 print(f' Size: {size/1024/1024:.1f} MB')95 96 # 6. Statistics97 print(f'\n=== VALIDATION DATASET STATISTICS ===')98 print(f'Total pairs: {filtered.num_rows:,}')99 100 # Unique languages101 langs = set()102 fa = filtered['Lang_A'].to_pylist()103 fb = filtered['Lang_B'].to_pylist()104 for a in fa:105 langs.add(a)106 for b in fb:107 langs.add(b)108 print(f'Unique languages: {len(langs)}')109 110 # Load family map111 t2 = pq.read_table(str(hf_dir / 'data' / 'training' / 'metadata' / 'languages.parquet'))112 iso_to_family = dict(zip(t2['ISO'].to_pylist(), t2['Family'].to_pylist()))113 114 # Family distribution115 fam_counts = {}116 for l in langs:117 fam = iso_to_family.get(l, 'unknown')118 fam_counts[fam] = fam_counts.get(fam, 0) + 1119 print(f'\nLanguage families:')120 for fam, c in sorted(fam_counts.items(), key=lambda x: -x[1]):121 print(f' {fam}: {c}')122 123 # Source distribution124 sources = filtered['Source'].to_pylist()125 src_counts = {}126 for s in sources:127 src_counts[s] = src_counts.get(s, 0) + 1128 print(f'\nSource distribution:')129 for src, c in sorted(src_counts.items(), key=lambda x: -x[1]):130 print(f' {src}: {c:,}')131 132 # Score distribution133 scores = filtered['Score'].to_pylist()134 float_scores = [float(s) for s in scores if s and s != '-1']135 if float_scores:136 print(f'\nSCA Score distribution:')137 print(f' Min: {min(float_scores):.4f}')138 print(f' Max: {max(float_scores):.4f}')139 print(f' Mean: {sum(float_scores)/len(float_scores):.4f}')140 bins = [(0.5, 0.6), (0.6, 0.7), (0.7, 0.8), (0.8, 0.9), (0.9, 1.01)]141 for lo, hi in bins:142 n = sum(1 for s in float_scores if lo <= s < hi)143 print(f' [{lo:.1f}, {hi:.1f}): {n:,}')144 145 # Load names for top contributing languages146 iso_to_name = {}147 glot_path = hf_dir / 'data' / 'training' / 'raw' / 'glottolog_cldf' / 'languages.csv'148 with open(glot_path, encoding='utf-8') as f:149 for row in csv.DictReader(f):150 iso = row.get('ISO639P3code', '').strip()151 name = row.get('Name', '').strip()152 if iso and name:153 iso_to_name[iso] = name154 155 # Top 20 languages by pair count156 lang_pair_counts = {}157 for a, b in zip(fa, fb):158 lang_pair_counts[a] = lang_pair_counts.get(a, 0) + 1159 lang_pair_counts[b] = lang_pair_counts.get(b, 0) + 1160 print(f'\nTop 20 languages by pair count:')161 for lang, c in sorted(lang_pair_counts.items(), key=lambda x: -x[1])[:20]:162 name = iso_to_name.get(lang, '?')163 fam = iso_to_family.get(lang, '?')164 score = lang_scores.get(lang, 0)165 print(f' {lang} ({name}) - {fam} - LA score {score:.4f} - {c:,} pairs')166 167 168if __name__ == '__main__':169 main()170 