Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
clean_artifacts.py153 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Remove known data processing artifacts from ancient language TSV files.3 4Iron Rule: This script does NOT add data. It only removes entries that match5known artifact patterns (processing placeholders, wrong-language entries,6encoding errors). All removals are logged to an audit trail.7 8Usage:9    python scripts/clean_artifacts.py [--dry-run] [--language ISO]10"""11 12from __future__ import annotations13 14import argparse15import json16import logging17import re18import sys19from datetime import datetime, timezone20from pathlib import Path21 22ROOT = Path(__file__).resolve().parent.parent23LEXICON_DIR = ROOT / "data" / "training" / "lexicons"24AUDIT_TRAIL_DIR = ROOT / "data" / "training" / "audit_trails"25 26logger = logging.getLogger(__name__)27 28# Known artifact patterns: (iso, word_pattern, reason)29ARTIFACT_PATTERNS: list[tuple[str | None, re.Pattern, str]] = [30    # Processing placeholders found in Avestan31    (None, re.compile(r"^inprogress$", re.IGNORECASE), "processing placeholder"),32    (None, re.compile(r"^phoneticvalue$", re.IGNORECASE), "processing placeholder"),33    (None, re.compile(r"^testentry$", re.IGNORECASE), "processing placeholder"),34    (None, re.compile(r"^placeholder$", re.IGNORECASE), "processing placeholder"),35    (None, re.compile(r"^TODO$"), "processing placeholder"),36]37 38# Known cross-language contamination: (iso, exact_word, reason)39CONTAMINATION: list[tuple[str, str, str]] = [40    ("hit", "xshap", "Avestan word (xshap- = night), not Hittite"),41]42 43 44def is_artifact(iso: str, word: str) -> str | None:45    """Return reason string if word is a known artifact, else None."""46    for pattern_iso, pattern, reason in ARTIFACT_PATTERNS:47        if pattern_iso is not None and pattern_iso != iso:48            continue49        if pattern.match(word):50            return reason51 52    for cont_iso, cont_word, reason in CONTAMINATION:53        if iso == cont_iso and word == cont_word:54            return reason55 56    return None57 58 59def clean_tsv(iso: str, tsv_path: Path, dry_run: bool = False) -> dict:60    """Clean a single TSV file. Returns stats dict."""61    with open(tsv_path, "r", encoding="utf-8") as f:62        lines = f.readlines()63 64    if not lines:65        return {"iso": iso, "total": 0, "removed": 0, "kept": 0}66 67    header = lines[0]68    kept = [header]69    removed = []70 71    for line in lines[1:]:72        parts = line.rstrip("\n").split("\t")73        if not parts:74            continue75        word = parts[0]76        reason = is_artifact(iso, word)77        if reason:78            removed.append({"word": word, "reason": reason, "line": line.rstrip("\n")})79            logger.warning("REMOVE %s/%s: %s", iso, word, reason)80        else:81            kept.append(line)82 83    if removed and not dry_run:84        # Write cleaned file85        with open(tsv_path, "w", encoding="utf-8") as f:86            f.writelines(kept)87 88        # Write audit trail89        AUDIT_TRAIL_DIR.mkdir(parents=True, exist_ok=True)90        audit_path = AUDIT_TRAIL_DIR / f"clean_artifacts_{iso}_{datetime.now(timezone.utc).strftime('%Y%m%d')}.jsonl"91        with open(audit_path, "w", encoding="utf-8") as f:92            for r in removed:93                f.write(json.dumps(r, ensure_ascii=False) + "\n")94 95    return {96        "iso": iso,97        "total": len(lines) - 1,98        "removed": len(removed),99        "kept": len(kept) - 1,100    }101 102 103ANCIENT_ISOS = [104    "ave", "txb", "xlw", "ine-pro", "xlc", "ett", "xur", "xld",105    "xcr", "ccs-pro", "peo", "xto", "dra-pro", "sem-pro", "uga",106    "hit", "xhu", "elx", "xrr", "phn", "xpg", "cms", "xle",107]108 109 110def main():111    parser = argparse.ArgumentParser(description="Remove known artifacts from TSV files")112    parser.add_argument("--language", "-l", help="ISO code (default: all ancient)")113    parser.add_argument("--dry-run", action="store_true", help="Preview without writing")114    args = parser.parse_args()115 116    logging.basicConfig(117        level=logging.INFO,118        format="%(asctime)s %(levelname)s: %(message)s",119        datefmt="%H:%M:%S",120    )121 122    if args.language:123        isos = [args.language]124    else:125        isos = ANCIENT_ISOS126 127    results = []128    for iso in isos:129        tsv_path = LEXICON_DIR / f"{iso}.tsv"130        if not tsv_path.exists():131            logger.info("SKIP %s: file not found", iso)132            continue133        result = clean_tsv(iso, tsv_path, dry_run=args.dry_run)134        results.append(result)135 136    print("\n" + "=" * 60)137    print(f"CLEAN ARTIFACTS {'(DRY RUN)' if args.dry_run else ''}")138    print("=" * 60)139    total_removed = 0140    for r in results:141        if r["removed"] > 0:142            print(f"  {r['iso']:10s} total={r['total']}, removed={r['removed']}, kept={r['kept']}")143            total_removed += r["removed"]144    if total_removed == 0:145        print("  No artifacts found.")146    else:147        print(f"\n  Total removed: {total_removed}")148    print("=" * 60)149 150 151if __name__ == "__main__":152    main()153