Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Comprehensive validation suite for ancient language TSV datasets.3 4Checks every lexicon TSV in data/training/lexicons/ against header format,5data-quality invariants, SCA consistency, metadata entry counts, and6artifact patterns. Designed to run on Windows (handles encoding) and in CI.7 8Usage:9 python scripts/validate_all.py # validate all files10 python scripts/validate_all.py --verbose # show individual failing entries11 python scripts/validate_all.py --file hittite.tsv # single file12"""13 14from __future__ import annotations15 16import argparse17import io18import os19import re20import sys21from pathlib import Path22 23# ---------------------------------------------------------------------------24# Paths25# ---------------------------------------------------------------------------26ROOT = Path(__file__).resolve().parent.parent27LEXICON_DIR = ROOT / "data" / "training" / "lexicons"28METADATA_PATH = ROOT / "data" / "training" / "metadata" / "languages.tsv"29 30# ---------------------------------------------------------------------------31# Ensure cognate_pipeline importable32# ---------------------------------------------------------------------------33sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))34 35try:36 from cognate_pipeline.normalise.sound_class import ipa_to_sound_class # type: ignore37 38 HAS_SCA = True39except ImportError:40 HAS_SCA = False41 42# ---------------------------------------------------------------------------43# Constants44# ---------------------------------------------------------------------------45EXPECTED_HEADER = "Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID"46EXPECTED_COLUMNS = 647 48ARTIFACT_RE = re.compile(49 r"inprogress|phoneticvalue|<[^>]+>|&[a-z]+;|\xa0",50 re.IGNORECASE,51)52 53 54# ---------------------------------------------------------------------------55# Result containers56# ---------------------------------------------------------------------------57class FileResult:58 """Accumulates pass / warn / fail status for a single TSV file."""59 60 def __init__(self, path: Path) -> None:61 self.path = path62 self.errors: list[str] = []63 self.warnings: list[str] = []64 65 def error(self, msg: str) -> None:66 self.errors.append(msg)67 68 def warn(self, msg: str) -> None:69 self.warnings.append(msg)70 71 @property72 def status(self) -> str:73 if self.errors:74 return "FAIL"75 if self.warnings:76 return "WARN"77 return "PASS"78 79 80# ---------------------------------------------------------------------------81# Metadata loader82# ---------------------------------------------------------------------------83def load_metadata() -> dict[str, int]:84 """Return {iso_code: expected_entry_count} from languages.tsv."""85 counts: dict[str, int] = {}86 if not METADATA_PATH.exists():87 return counts88 with open(METADATA_PATH, encoding="utf-8") as fh:89 header = fh.readline().strip()90 cols = header.split("\t")91 try:92 iso_idx = cols.index("ISO")93 entries_idx = cols.index("Entries")94 except ValueError:95 return counts96 for line in fh:97 parts = line.strip().split("\t")98 if len(parts) > max(iso_idx, entries_idx):99 iso = parts[iso_idx].strip()100 try:101 counts[iso] = int(parts[entries_idx].strip())102 except ValueError:103 pass104 return counts105 106 107# ---------------------------------------------------------------------------108# Core validation109# ---------------------------------------------------------------------------110def validate_file(tsv_path: Path, metadata: dict[str, int], verbose: bool) -> FileResult:111 """Run all checks on a single lexicon TSV and return a FileResult."""112 result = FileResult(tsv_path)113 114 # --- Read file ----------------------------------------------------------115 try:116 with open(tsv_path, encoding="utf-8") as fh:117 raw_lines = fh.readlines()118 except Exception as exc:119 result.error(f"Cannot read file: {exc}")120 return result121 122 if not raw_lines:123 result.error("File is empty")124 return result125 126 # --- 1. Header validation -----------------------------------------------127 header_line = raw_lines[0].rstrip("\n\r")128 if header_line != EXPECTED_HEADER:129 result.error(f"Bad header: got {header_line!r}")130 131 # --- Parse rows ---------------------------------------------------------132 rows: list[dict[str, str]] = []133 for lineno, raw in enumerate(raw_lines[1:], start=2):134 parts = raw.rstrip("\n\r").split("\t")135 if len(parts) != EXPECTED_COLUMNS:136 result.error(137 f"Line {lineno}: expected {EXPECTED_COLUMNS} columns, got {len(parts)}"138 )139 continue140 rows.append(141 {142 "Word": parts[0],143 "IPA": parts[1],144 "SCA": parts[2],145 "Source": parts[3],146 "Concept_ID": parts[4],147 "Cognate_Set_ID": parts[5],148 "_lineno": str(lineno),149 }150 )151 152 if not rows:153 result.error("No data rows found")154 return result155 156 # --- 2. No empty IPA ----------------------------------------------------157 empty_ipa = [r for r in rows if not r["IPA"].strip()]158 if empty_ipa:159 result.error(f"Empty IPA: {len(empty_ipa)} entries")160 if verbose:161 for r in empty_ipa[:20]:162 result.error(f" line {r['_lineno']}: Word={r['Word']!r}")163 164 # --- 3. No duplicate words ----------------------------------------------165 seen_words: dict[str, int] = {}166 duplicates: list[tuple[str, int, int]] = []167 for r in rows:168 w = r["Word"]169 if w in seen_words:170 duplicates.append((w, seen_words[w], int(r["_lineno"])))171 else:172 seen_words[w] = int(r["_lineno"])173 if duplicates:174 result.error(f"Duplicate words: {len(duplicates)} duplicates")175 if verbose:176 for word, first, second in duplicates[:20]:177 result.error(f" {word!r}: lines {first} and {second}")178 179 # --- 4. SCA consistency -------------------------------------------------180 if HAS_SCA:181 sca_mismatches: list[tuple[str, str, str, str]] = []182 for r in rows:183 sca_val = r["SCA"].strip()184 ipa_val = r["IPA"].strip()185 if not sca_val or not ipa_val:186 continue187 try:188 expected_sca = ipa_to_sound_class(ipa_val)189 except Exception:190 continue191 if expected_sca and expected_sca != sca_val:192 sca_mismatches.append(193 (r["_lineno"], r["Word"], sca_val, expected_sca)194 )195 if sca_mismatches:196 result.error(f"SCA mismatches: {len(sca_mismatches)} entries")197 if verbose:198 for lineno, word, got, expected in sca_mismatches[:20]:199 result.error(200 f" line {lineno}: {word!r} SCA={got!r} expected={expected!r}"201 )202 203 # --- 5. Spurious "0" in SCA (warning) -----------------------------------204 zero_sca = [r for r in rows if r["SCA"].strip() == "0"]205 if zero_sca:206 result.warn(f'SCA == "0": {len(zero_sca)} entries')207 if verbose:208 for r in zero_sca[:20]:209 result.warn(f" line {r['_lineno']}: Word={r['Word']!r}")210 211 # --- 6. Non-empty Source ------------------------------------------------212 empty_source = [r for r in rows if not r["Source"].strip()]213 if empty_source:214 result.error(f"Empty Source: {len(empty_source)} entries")215 if verbose:216 for r in empty_source[:20]:217 result.error(218 f" line {r['_lineno']}: Word={r['Word']!r}"219 )220 221 # --- 7. Entry count accuracy --------------------------------------------222 stem = tsv_path.stem # e.g. "hittite" from hittite.tsv223 # Try matching by ISO code (stem) against metadata224 actual_count = len(rows)225 matched_iso: str | None = None226 # Prefer exact match first, then prefix match227 if stem in metadata:228 matched_iso = stem229 expected_count = metadata[stem]230 if actual_count != expected_count:231 result.error(232 f"Entry count mismatch: TSV has {actual_count} rows, "233 f"metadata ({stem}) says {expected_count}"234 )235 else:236 for iso, expected_count in metadata.items():237 if iso.lower() == stem.lower() or stem.lower().startswith(iso.lower()):238 matched_iso = iso239 if actual_count != expected_count:240 result.error(241 f"Entry count mismatch: TSV has {actual_count} rows, "242 f"metadata ({iso}) says {expected_count}"243 )244 break245 # Also try matching by filename without extension246 if matched_iso is None and stem in metadata:247 expected_count = metadata[stem]248 if actual_count != expected_count:249 result.error(250 f"Entry count mismatch: TSV has {actual_count} rows, "251 f"metadata ({stem}) says {expected_count}"252 )253 254 # --- 8. Artifact patterns -----------------------------------------------255 artifact_hits: list[tuple[str, str, str]] = []256 for r in rows:257 for field in ("Word", "IPA", "SCA", "Source"):258 if ARTIFACT_RE.search(r[field]):259 artifact_hits.append((r["_lineno"], field, r[field]))260 if artifact_hits:261 result.warn(f"Artifact patterns detected: {len(artifact_hits)} hits")262 if verbose:263 for lineno, field, val in artifact_hits[:20]:264 result.warn(f" line {lineno}: {field}={val!r}")265 266 return result267 268 269# ---------------------------------------------------------------------------270# Main271# ---------------------------------------------------------------------------272def main() -> int:273 # Handle Windows encoding for stdout / stderr274 if sys.platform == "win32":275 if hasattr(sys.stdout, "buffer"):276 sys.stdout.reconfigure(encoding="utf-8", errors="replace")277 if hasattr(sys.stderr, "buffer"):278 sys.stderr.reconfigure(encoding="utf-8", errors="replace")279 280 parser = argparse.ArgumentParser(281 description="Validate ancient language TSV datasets."282 )283 parser.add_argument(284 "--verbose", "-v", action="store_true", help="Show individual failing entries"285 )286 parser.add_argument(287 "--file",288 "-f",289 type=str,290 default=None,291 help="Validate a single file (name or path relative to lexicons/)",292 )293 args = parser.parse_args()294 295 # Discover files296 if args.file:297 target = Path(args.file)298 if not target.is_absolute():299 target = LEXICON_DIR / target300 if not target.suffix:301 target = target.with_suffix(".tsv")302 if not target.exists():303 print(f"ERROR: File not found: {target}", file=sys.stderr)304 return 1305 tsv_files = [target]306 else:307 if not LEXICON_DIR.is_dir():308 print(f"ERROR: Lexicon directory not found: {LEXICON_DIR}", file=sys.stderr)309 return 1310 tsv_files = sorted(LEXICON_DIR.glob("*.tsv"))311 if not tsv_files:312 print(f"ERROR: No TSV files found in {LEXICON_DIR}", file=sys.stderr)313 return 1314 315 # Load metadata316 metadata = load_metadata()317 if not metadata:318 print(319 f"WARNING: Could not load metadata from {METADATA_PATH}; "320 "skipping entry-count checks.",321 file=sys.stderr,322 )323 324 if not HAS_SCA:325 print(326 "WARNING: Could not import ipa_to_sound_class; "327 "skipping SCA consistency checks.",328 file=sys.stderr,329 )330 331 # Run validation332 results: list[FileResult] = []333 for tsv in tsv_files:334 res = validate_file(tsv, metadata, args.verbose)335 results.append(res)336 337 # ---------------------------------------------------------------------------338 # Output339 # ---------------------------------------------------------------------------340 total_pass = 0341 total_warn = 0342 total_fail = 0343 344 for res in results:345 status = res.status346 name = res.path.name347 348 if status == "PASS":349 total_pass += 1350 print(f" PASS {name}")351 elif status == "WARN":352 total_warn += 1353 print(f" WARN {name} ({len(res.warnings)} warning(s))")354 for w in res.warnings:355 print(f" {w}")356 else:357 total_fail += 1358 print(f" FAIL {name} ({len(res.errors)} error(s), {len(res.warnings)} warning(s))")359 for e in res.errors:360 print(f" [E] {e}")361 for w in res.warnings:362 print(f" [W] {w}")363 364 # Summary365 total = len(results)366 print()367 print("=" * 60)368 print(f"SUMMARY: {total} file(s) validated")369 print(f" PASS: {total_pass} WARN: {total_warn} FAIL: {total_fail}")370 print("=" * 60)371 372 if total_fail > 0:373 print("\nResult: FAILED (errors found)")374 return 1375 elif total_warn > 0:376 print("\nResult: PASSED with warnings")377 return 0378 else:379 print("\nResult: PASSED")380 return 0381 382 383if __name__ == "__main__":384 sys.exit(main())385 