Nacryos/ancient-scripts-datasets
Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.
1531
1#!/usr/bin/env python32"""Remove known data processing artifacts from ancient language TSV files.3 4Iron Rule: This script does NOT add data. It only removes entries that match5known artifact patterns (processing placeholders, wrong-language entries,6encoding errors). All removals are logged to an audit trail.7 8Usage:9 python scripts/clean_artifacts.py [--dry-run] [--language ISO]10"""11 12from __future__ import annotations13 14import argparse15import json16import logging17import re18import sys19from datetime import datetime, timezone20from pathlib import Path21 22ROOT = Path(__file__).resolve().parent.parent23LEXICON_DIR = ROOT / "data" / "training" / "lexicons"24AUDIT_TRAIL_DIR = ROOT / "data" / "training" / "audit_trails"25 26logger = logging.getLogger(__name__)27 28# Known artifact patterns: (iso, word_pattern, reason)29ARTIFACT_PATTERNS: list[tuple[str | None, re.Pattern, str]] = [30 # Processing placeholders found in Avestan31 (None, re.compile(r"^inprogress$", re.IGNORECASE), "processing placeholder"),32 (None, re.compile(r"^phoneticvalue$", re.IGNORECASE), "processing placeholder"),33 (None, re.compile(r"^testentry$", re.IGNORECASE), "processing placeholder"),34 (None, re.compile(r"^placeholder$", re.IGNORECASE), "processing placeholder"),35 (None, re.compile(r"^TODO$"), "processing placeholder"),36]37 38# Known cross-language contamination: (iso, exact_word, reason)39CONTAMINATION: list[tuple[str, str, str]] = [40 ("hit", "xshap", "Avestan word (xshap- = night), not Hittite"),41]42 43 44def is_artifact(iso: str, word: str) -> str | None:45 """Return reason string if word is a known artifact, else None."""46 for pattern_iso, pattern, reason in ARTIFACT_PATTERNS:47 if pattern_iso is not None and pattern_iso != iso:48 continue49 if pattern.match(word):50 return reason51 52 for cont_iso, cont_word, reason in CONTAMINATION:53 if iso == cont_iso and word == cont_word:54 return reason55 56 return None57 58 59def clean_tsv(iso: str, tsv_path: Path, dry_run: bool = False) -> dict:60 """Clean a single TSV file. Returns stats dict."""61 with open(tsv_path, "r", encoding="utf-8") as f:62 lines = f.readlines()63 64 if not lines:65 return {"iso": iso, "total": 0, "removed": 0, "kept": 0}66 67 header = lines[0]68 kept = [header]69 removed = []70 71 for line in lines[1:]:72 parts = line.rstrip("\n").split("\t")73 if not parts:74 continue75 word = parts[0]76 reason = is_artifact(iso, word)77 if reason:78 removed.append({"word": word, "reason": reason, "line": line.rstrip("\n")})79 logger.warning("REMOVE %s/%s: %s", iso, word, reason)80 else:81 kept.append(line)82 83 if removed and not dry_run:84 # Write cleaned file85 with open(tsv_path, "w", encoding="utf-8") as f:86 f.writelines(kept)87 88 # Write audit trail89 AUDIT_TRAIL_DIR.mkdir(parents=True, exist_ok=True)90 audit_path = AUDIT_TRAIL_DIR / f"clean_artifacts_{iso}_{datetime.now(timezone.utc).strftime('%Y%m%d')}.jsonl"91 with open(audit_path, "w", encoding="utf-8") as f:92 for r in removed:93 f.write(json.dumps(r, ensure_ascii=False) + "\n")94 95 return {96 "iso": iso,97 "total": len(lines) - 1,98 "removed": len(removed),99 "kept": len(kept) - 1,100 }101 102 103ANCIENT_ISOS = [104 "ave", "txb", "xlw", "ine-pro", "xlc", "ett", "xur", "xld",105 "xcr", "ccs-pro", "peo", "xto", "dra-pro", "sem-pro", "uga",106 "hit", "xhu", "elx", "xrr", "phn", "xpg", "cms", "xle",107]108 109 110def main():111 parser = argparse.ArgumentParser(description="Remove known artifacts from TSV files")112 parser.add_argument("--language", "-l", help="ISO code (default: all ancient)")113 parser.add_argument("--dry-run", action="store_true", help="Preview without writing")114 args = parser.parse_args()115 116 logging.basicConfig(117 level=logging.INFO,118 format="%(asctime)s %(levelname)s: %(message)s",119 datefmt="%H:%M:%S",120 )121 122 if args.language:123 isos = [args.language]124 else:125 isos = ANCIENT_ISOS126 127 results = []128 for iso in isos:129 tsv_path = LEXICON_DIR / f"{iso}.tsv"130 if not tsv_path.exists():131 logger.info("SKIP %s: file not found", iso)132 continue133 result = clean_tsv(iso, tsv_path, dry_run=args.dry_run)134 results.append(result)135 136 print("\n" + "=" * 60)137 print(f"CLEAN ARTIFACTS {'(DRY RUN)' if args.dry_run else ''}")138 print("=" * 60)139 total_removed = 0140 for r in results:141 if r["removed"] > 0:142 print(f" {r['iso']:10s} total={r['total']}, removed={r['removed']}, kept={r['kept']}")143 total_removed += r["removed"]144 if total_removed == 0:145 print(" No artifacts found.")146 else:147 print(f"\n Total removed: {total_removed}")148 print("=" * 60)149 150 151if __name__ == "__main__":152 main()153 