Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
validate_all.py385 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Comprehensive validation suite for ancient language TSV datasets.3 4Checks every lexicon TSV in data/training/lexicons/ against header format,5data-quality invariants, SCA consistency, metadata entry counts, and6artifact patterns.  Designed to run on Windows (handles encoding) and in CI.7 8Usage:9    python scripts/validate_all.py              # validate all files10    python scripts/validate_all.py --verbose    # show individual failing entries11    python scripts/validate_all.py --file hittite.tsv   # single file12"""13 14from __future__ import annotations15 16import argparse17import io18import os19import re20import sys21from pathlib import Path22 23# ---------------------------------------------------------------------------24# Paths25# ---------------------------------------------------------------------------26ROOT = Path(__file__).resolve().parent.parent27LEXICON_DIR = ROOT / "data" / "training" / "lexicons"28METADATA_PATH = ROOT / "data" / "training" / "metadata" / "languages.tsv"29 30# ---------------------------------------------------------------------------31# Ensure cognate_pipeline importable32# ---------------------------------------------------------------------------33sys.path.insert(0, str(ROOT / "cognate_pipeline" / "src"))34 35try:36    from cognate_pipeline.normalise.sound_class import ipa_to_sound_class  # type: ignore37 38    HAS_SCA = True39except ImportError:40    HAS_SCA = False41 42# ---------------------------------------------------------------------------43# Constants44# ---------------------------------------------------------------------------45EXPECTED_HEADER = "Word\tIPA\tSCA\tSource\tConcept_ID\tCognate_Set_ID"46EXPECTED_COLUMNS = 647 48ARTIFACT_RE = re.compile(49    r"inprogress|phoneticvalue|<[^>]+>|&[a-z]+;|\xa0",50    re.IGNORECASE,51)52 53 54# ---------------------------------------------------------------------------55# Result containers56# ---------------------------------------------------------------------------57class FileResult:58    """Accumulates pass / warn / fail status for a single TSV file."""59 60    def __init__(self, path: Path) -> None:61        self.path = path62        self.errors: list[str] = []63        self.warnings: list[str] = []64 65    def error(self, msg: str) -> None:66        self.errors.append(msg)67 68    def warn(self, msg: str) -> None:69        self.warnings.append(msg)70 71    @property72    def status(self) -> str:73        if self.errors:74            return "FAIL"75        if self.warnings:76            return "WARN"77        return "PASS"78 79 80# ---------------------------------------------------------------------------81# Metadata loader82# ---------------------------------------------------------------------------83def load_metadata() -> dict[str, int]:84    """Return {iso_code: expected_entry_count} from languages.tsv."""85    counts: dict[str, int] = {}86    if not METADATA_PATH.exists():87        return counts88    with open(METADATA_PATH, encoding="utf-8") as fh:89        header = fh.readline().strip()90        cols = header.split("\t")91        try:92            iso_idx = cols.index("ISO")93            entries_idx = cols.index("Entries")94        except ValueError:95            return counts96        for line in fh:97            parts = line.strip().split("\t")98            if len(parts) > max(iso_idx, entries_idx):99                iso = parts[iso_idx].strip()100                try:101                    counts[iso] = int(parts[entries_idx].strip())102                except ValueError:103                    pass104    return counts105 106 107# ---------------------------------------------------------------------------108# Core validation109# ---------------------------------------------------------------------------110def validate_file(tsv_path: Path, metadata: dict[str, int], verbose: bool) -> FileResult:111    """Run all checks on a single lexicon TSV and return a FileResult."""112    result = FileResult(tsv_path)113 114    # --- Read file ----------------------------------------------------------115    try:116        with open(tsv_path, encoding="utf-8") as fh:117            raw_lines = fh.readlines()118    except Exception as exc:119        result.error(f"Cannot read file: {exc}")120        return result121 122    if not raw_lines:123        result.error("File is empty")124        return result125 126    # --- 1. Header validation -----------------------------------------------127    header_line = raw_lines[0].rstrip("\n\r")128    if header_line != EXPECTED_HEADER:129        result.error(f"Bad header: got {header_line!r}")130 131    # --- Parse rows ---------------------------------------------------------132    rows: list[dict[str, str]] = []133    for lineno, raw in enumerate(raw_lines[1:], start=2):134        parts = raw.rstrip("\n\r").split("\t")135        if len(parts) != EXPECTED_COLUMNS:136            result.error(137                f"Line {lineno}: expected {EXPECTED_COLUMNS} columns, got {len(parts)}"138            )139            continue140        rows.append(141            {142                "Word": parts[0],143                "IPA": parts[1],144                "SCA": parts[2],145                "Source": parts[3],146                "Concept_ID": parts[4],147                "Cognate_Set_ID": parts[5],148                "_lineno": str(lineno),149            }150        )151 152    if not rows:153        result.error("No data rows found")154        return result155 156    # --- 2. No empty IPA ----------------------------------------------------157    empty_ipa = [r for r in rows if not r["IPA"].strip()]158    if empty_ipa:159        result.error(f"Empty IPA: {len(empty_ipa)} entries")160        if verbose:161            for r in empty_ipa[:20]:162                result.error(f"  line {r['_lineno']}: Word={r['Word']!r}")163 164    # --- 3. No duplicate words ----------------------------------------------165    seen_words: dict[str, int] = {}166    duplicates: list[tuple[str, int, int]] = []167    for r in rows:168        w = r["Word"]169        if w in seen_words:170            duplicates.append((w, seen_words[w], int(r["_lineno"])))171        else:172            seen_words[w] = int(r["_lineno"])173    if duplicates:174        result.error(f"Duplicate words: {len(duplicates)} duplicates")175        if verbose:176            for word, first, second in duplicates[:20]:177                result.error(f"  {word!r}: lines {first} and {second}")178 179    # --- 4. SCA consistency -------------------------------------------------180    if HAS_SCA:181        sca_mismatches: list[tuple[str, str, str, str]] = []182        for r in rows:183            sca_val = r["SCA"].strip()184            ipa_val = r["IPA"].strip()185            if not sca_val or not ipa_val:186                continue187            try:188                expected_sca = ipa_to_sound_class(ipa_val)189            except Exception:190                continue191            if expected_sca and expected_sca != sca_val:192                sca_mismatches.append(193                    (r["_lineno"], r["Word"], sca_val, expected_sca)194                )195        if sca_mismatches:196            result.error(f"SCA mismatches: {len(sca_mismatches)} entries")197            if verbose:198                for lineno, word, got, expected in sca_mismatches[:20]:199                    result.error(200                        f"  line {lineno}: {word!r} SCA={got!r} expected={expected!r}"201                    )202 203    # --- 5. Spurious "0" in SCA (warning) -----------------------------------204    zero_sca = [r for r in rows if r["SCA"].strip() == "0"]205    if zero_sca:206        result.warn(f'SCA == "0": {len(zero_sca)} entries')207        if verbose:208            for r in zero_sca[:20]:209                result.warn(f"  line {r['_lineno']}: Word={r['Word']!r}")210 211    # --- 6. Non-empty Source ------------------------------------------------212    empty_source = [r for r in rows if not r["Source"].strip()]213    if empty_source:214        result.error(f"Empty Source: {len(empty_source)} entries")215        if verbose:216            for r in empty_source[:20]:217                result.error(218                    f"  line {r['_lineno']}: Word={r['Word']!r}"219                )220 221    # --- 7. Entry count accuracy --------------------------------------------222    stem = tsv_path.stem  # e.g. "hittite" from hittite.tsv223    # Try matching by ISO code (stem) against metadata224    actual_count = len(rows)225    matched_iso: str | None = None226    # Prefer exact match first, then prefix match227    if stem in metadata:228        matched_iso = stem229        expected_count = metadata[stem]230        if actual_count != expected_count:231            result.error(232                f"Entry count mismatch: TSV has {actual_count} rows, "233                f"metadata ({stem}) says {expected_count}"234            )235    else:236        for iso, expected_count in metadata.items():237            if iso.lower() == stem.lower() or stem.lower().startswith(iso.lower()):238                matched_iso = iso239                if actual_count != expected_count:240                    result.error(241                        f"Entry count mismatch: TSV has {actual_count} rows, "242                        f"metadata ({iso}) says {expected_count}"243                    )244                break245    # Also try matching by filename without extension246    if matched_iso is None and stem in metadata:247        expected_count = metadata[stem]248        if actual_count != expected_count:249            result.error(250                f"Entry count mismatch: TSV has {actual_count} rows, "251                f"metadata ({stem}) says {expected_count}"252            )253 254    # --- 8. Artifact patterns -----------------------------------------------255    artifact_hits: list[tuple[str, str, str]] = []256    for r in rows:257        for field in ("Word", "IPA", "SCA", "Source"):258            if ARTIFACT_RE.search(r[field]):259                artifact_hits.append((r["_lineno"], field, r[field]))260    if artifact_hits:261        result.warn(f"Artifact patterns detected: {len(artifact_hits)} hits")262        if verbose:263            for lineno, field, val in artifact_hits[:20]:264                result.warn(f"  line {lineno}: {field}={val!r}")265 266    return result267 268 269# ---------------------------------------------------------------------------270# Main271# ---------------------------------------------------------------------------272def main() -> int:273    # Handle Windows encoding for stdout / stderr274    if sys.platform == "win32":275        if hasattr(sys.stdout, "buffer"):276            sys.stdout.reconfigure(encoding="utf-8", errors="replace")277        if hasattr(sys.stderr, "buffer"):278            sys.stderr.reconfigure(encoding="utf-8", errors="replace")279 280    parser = argparse.ArgumentParser(281        description="Validate ancient language TSV datasets."282    )283    parser.add_argument(284        "--verbose", "-v", action="store_true", help="Show individual failing entries"285    )286    parser.add_argument(287        "--file",288        "-f",289        type=str,290        default=None,291        help="Validate a single file (name or path relative to lexicons/)",292    )293    args = parser.parse_args()294 295    # Discover files296    if args.file:297        target = Path(args.file)298        if not target.is_absolute():299            target = LEXICON_DIR / target300        if not target.suffix:301            target = target.with_suffix(".tsv")302        if not target.exists():303            print(f"ERROR: File not found: {target}", file=sys.stderr)304            return 1305        tsv_files = [target]306    else:307        if not LEXICON_DIR.is_dir():308            print(f"ERROR: Lexicon directory not found: {LEXICON_DIR}", file=sys.stderr)309            return 1310        tsv_files = sorted(LEXICON_DIR.glob("*.tsv"))311        if not tsv_files:312            print(f"ERROR: No TSV files found in {LEXICON_DIR}", file=sys.stderr)313            return 1314 315    # Load metadata316    metadata = load_metadata()317    if not metadata:318        print(319            f"WARNING: Could not load metadata from {METADATA_PATH}; "320            "skipping entry-count checks.",321            file=sys.stderr,322        )323 324    if not HAS_SCA:325        print(326            "WARNING: Could not import ipa_to_sound_class; "327            "skipping SCA consistency checks.",328            file=sys.stderr,329        )330 331    # Run validation332    results: list[FileResult] = []333    for tsv in tsv_files:334        res = validate_file(tsv, metadata, args.verbose)335        results.append(res)336 337    # ---------------------------------------------------------------------------338    # Output339    # ---------------------------------------------------------------------------340    total_pass = 0341    total_warn = 0342    total_fail = 0343 344    for res in results:345        status = res.status346        name = res.path.name347 348        if status == "PASS":349            total_pass += 1350            print(f"  PASS  {name}")351        elif status == "WARN":352            total_warn += 1353            print(f"  WARN  {name}  ({len(res.warnings)} warning(s))")354            for w in res.warnings:355                print(f"        {w}")356        else:357            total_fail += 1358            print(f"  FAIL  {name}  ({len(res.errors)} error(s), {len(res.warnings)} warning(s))")359            for e in res.errors:360                print(f"    [E] {e}")361            for w in res.warnings:362                print(f"    [W] {w}")363 364    # Summary365    total = len(results)366    print()367    print("=" * 60)368    print(f"SUMMARY: {total} file(s) validated")369    print(f"  PASS: {total_pass}    WARN: {total_warn}    FAIL: {total_fail}")370    print("=" * 60)371 372    if total_fail > 0:373        print("\nResult: FAILED (errors found)")374        return 1375    elif total_warn > 0:376        print("\nResult: PASSED with warnings")377        return 0378    else:379        print("\nResult: PASSED")380        return 0381 382 383if __name__ == "__main__":384    sys.exit(main())385