Team Ai
Datasetpublic

Nacryos/ancient-scripts-datasets

Ancient Scripts Decipherment Datasets Collated datasets for the paper: Deciphering Undersegmented Ancient Scripts Using Phonetic Prior Jiaming Luo, Frederik Hartmann, Enrico Santus, Regina Barzilay, Yuan Cao Transactions of the Association for Computational Linguistics, 2021 arXiv:2010.11054 This repository gathers the training datasets used in the paper — both those hosted in the authors' GitHub repos and the external cited sources. Repository Structure data/… See the full description on the dataset page: https://huggingface.co/datasets/Nacryos/ancient-scripts-datasets.

sourceHugging Facecc-by-sa-4.0updated 7mo agoView on Hugging Face
1likes531downloads
ingest_linear_b.py281 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Ingest Linear B (Mycenaean Greek) data from CC-BY-SA compatible sources.3 4Sources:5  1. Unicode UCD — Sign inventory (88 syllabograms + 123 ideograms)6     URL: https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt7     License: Unicode Terms (permissive, CC-BY-SA-4.0 compatible)8 9  2. Wiktionary — Mycenaean Greek lemmas (~435 entries)10     URL: https://en.wiktionary.org/w/api.php (MediaWiki API)11     License: CC-BY-SA-3.0+12 13  3. jhnwnstd/shannon — Linear B Lexicon (2,747 entries, MIT license)14     URL: https://raw.githubusercontent.com/jhnwnstd/shannon/main/Linear_B_Lexicon.csv15     License: MIT16 17Iron Rule: Data comes from downloaded files/API responses. No hardcoded word lists.18 19Usage:20    python scripts/ingest_linear_b.py [--dry-run]21"""22 23from __future__ import annotations24 25import argparse26import csv27import io28import json29import logging30import os31import re32import sys33import time34import urllib.error35import urllib.parse36import urllib.request37from pathlib import Path38 39sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")40sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8")41 42ROOT = Path(__file__).resolve().parent.parent43 44logger = logging.getLogger(__name__)45 46RAW_DIR = ROOT / "data" / "training" / "raw" / "linear_b"47 48# ── Source URLs ──49 50UNICODE_DATA_URL = "https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt"51 52WIKTIONARY_API = "https://en.wiktionary.org/w/api.php"53 54SHANNON_LEXICON_URL = (55    "https://raw.githubusercontent.com/jhnwnstd/shannon/main/Linear_B_Lexicon.csv"56)57 58# Linear B Unicode ranges59LINB_SYLLABARY_START = 0x1000060LINB_SYLLABARY_END = 0x1007F61LINB_IDEOGRAM_START = 0x1008062LINB_IDEOGRAM_END = 0x100FF63 64 65def download_file(url: str, dest: Path, label: str) -> bool:66    """Download a file, skipping if already present and non-empty."""67    if dest.exists() and dest.stat().st_size > 0:68        logger.info(f"  {label}: already exists ({dest.stat().st_size:,} bytes)")69        return True70    logger.info(f"  {label}: downloading from {url}")71    try:72        req = urllib.request.Request(url, headers={"User-Agent": "LinearB-Ingestion/1.0"})73        with urllib.request.urlopen(req, timeout=60) as resp:74            data = resp.read()75        dest.parent.mkdir(parents=True, exist_ok=True)76        dest.write_bytes(data)77        logger.info(f"  {label}: downloaded {len(data):,} bytes")78        return True79    except (urllib.error.URLError, urllib.error.HTTPError, OSError) as e:80        logger.error(f"  {label}: DOWNLOAD FAILED — {e}")81        return False82 83 84def download_unicode_data(dry_run: bool = False) -> Path:85    """Download UnicodeData.txt."""86    dest = RAW_DIR / "UnicodeData.txt"87    if dry_run:88        logger.info(f"  [DRY RUN] Would download {UNICODE_DATA_URL}")89        return dest90    download_file(UNICODE_DATA_URL, dest, "UnicodeData.txt")91    return dest92 93 94def download_shannon_lexicon(dry_run: bool = False) -> Path:95    """Download jhnwnstd/shannon Linear_B_Lexicon.csv."""96    dest = RAW_DIR / "shannon_Linear_B_Lexicon.csv"97    if dry_run:98        logger.info(f"  [DRY RUN] Would download {SHANNON_LEXICON_URL}")99        return dest100    download_file(SHANNON_LEXICON_URL, dest, "shannon_lexicon")101    return dest102 103 104def download_wiktionary_lemmas(dry_run: bool = False) -> Path:105    """Download all Mycenaean Greek lemmas from Wiktionary API."""106    dest = RAW_DIR / "wiktionary_gmy_lemmas.json"107    if dest.exists() and dest.stat().st_size > 0:108        logger.info(f"  wiktionary: already exists ({dest.stat().st_size:,} bytes)")109        return dest110    if dry_run:111        logger.info(f"  [DRY RUN] Would fetch Wiktionary gmy lemmas")112        return dest113 114    all_titles = []115    cmcontinue = None116    page = 0117 118    while True:119        page += 1120        params = {121            "action": "query",122            "list": "categorymembers",123            "cmtitle": "Category:Mycenaean_Greek_lemmas",124            "cmlimit": "500",125            "format": "json",126        }127        if cmcontinue:128            params["cmcontinue"] = cmcontinue129 130        url = f"{WIKTIONARY_API}?{urllib.parse.urlencode(params)}"131        logger.info(f"  wiktionary: page {page}, {len(all_titles)} titles so far...")132 133        req = urllib.request.Request(url, headers={"User-Agent": "LinearB-Ingestion/1.0"})134        with urllib.request.urlopen(req, timeout=30) as resp:135            data = json.loads(resp.read().decode("utf-8"))136 137        members = data.get("query", {}).get("categorymembers", [])138        for m in members:139            all_titles.append(m["title"])140 141        # Check for continuation142        cont = data.get("continue", {})143        if "cmcontinue" in cont:144            cmcontinue = cont["cmcontinue"]145            time.sleep(0.5)  # Be polite to Wiktionary146        else:147            break148 149    logger.info(f"  wiktionary: fetched {len(all_titles)} lemma titles")150 151    # Now fetch content for each lemma (in batches of 50)152    lemma_data = []153    batch_size = 50154    for i in range(0, len(all_titles), batch_size):155        batch = all_titles[i : i + batch_size]156        titles_param = "|".join(batch)157        params = {158            "action": "query",159            "titles": titles_param,160            "prop": "revisions",161            "rvprop": "content",162            "rvslots": "main",163            "format": "json",164        }165        url = f"{WIKTIONARY_API}?{urllib.parse.urlencode(params)}"166        req = urllib.request.Request(url, headers={"User-Agent": "LinearB-Ingestion/1.0"})167 168        try:169            with urllib.request.urlopen(req, timeout=30) as resp:170                data = json.loads(resp.read().decode("utf-8"))171 172            pages = data.get("query", {}).get("pages", {})173            for page_id, page_data in pages.items():174                title = page_data.get("title", "")175                revisions = page_data.get("revisions", [])176                if revisions:177                    content = revisions[0].get("slots", {}).get("main", {}).get("*", "")178                    lemma_data.append({"title": title, "wikitext": content})179        except Exception as e:180            logger.warning(f"  wiktionary batch {i//batch_size}: {e}")181 182        if i + batch_size < len(all_titles):183            time.sleep(1.0)  # Rate limit184 185    dest.parent.mkdir(parents=True, exist_ok=True)186    dest.write_text(json.dumps(lemma_data, ensure_ascii=False, indent=2), encoding="utf-8")187    logger.info(f"  wiktionary: saved {len(lemma_data)} lemma entries to {dest}")188    return dest189 190 191def download_wiktionary_swadesh(dry_run: bool = False) -> Path:192    """Download Mycenaean Greek Swadesh list from Wiktionary."""193    dest = RAW_DIR / "wiktionary_gmy_swadesh.json"194    if dest.exists() and dest.stat().st_size > 0:195        logger.info(f"  swadesh: already exists ({dest.stat().st_size:,} bytes)")196        return dest197    if dry_run:198        logger.info(f"  [DRY RUN] Would fetch Wiktionary gmy Swadesh list")199        return dest200 201    params = {202        "action": "query",203        "titles": "Appendix:Mycenaean_Greek_Swadesh_list",204        "prop": "revisions",205        "rvprop": "content",206        "rvslots": "main",207        "format": "json",208    }209    url = f"{WIKTIONARY_API}?{urllib.parse.urlencode(params)}"210    req = urllib.request.Request(url, headers={"User-Agent": "LinearB-Ingestion/1.0"})211 212    with urllib.request.urlopen(req, timeout=30) as resp:213        data = json.loads(resp.read().decode("utf-8"))214 215    pages = data.get("query", {}).get("pages", {})216    for page_id, page_data in pages.items():217        content = page_data.get("revisions", [{}])[0].get("slots", {}).get("main", {}).get("*", "")218 219    dest.parent.mkdir(parents=True, exist_ok=True)220    dest.write_text(json.dumps({"title": "Mycenaean_Greek_Swadesh_list", "wikitext": content},221                               ensure_ascii=False, indent=2), encoding="utf-8")222    logger.info(f"  swadesh: saved to {dest}")223    return dest224 225 226def main():227    parser = argparse.ArgumentParser(description="Ingest Linear B data from open sources")228    parser.add_argument("--dry-run", action="store_true", help="Show what would be downloaded")229    args = parser.parse_args()230 231    logging.basicConfig(232        level=logging.INFO,233        format="%(asctime)s %(levelname)s %(message)s",234        datefmt="%H:%M:%S",235    )236 237    RAW_DIR.mkdir(parents=True, exist_ok=True)238 239    print("=" * 70)240    print("LINEAR B DATA INGESTION")241    print("=" * 70)242 243    # 1. Unicode Data244    print("\n[1/4] Unicode UCD (sign inventory)")245    ucd_path = download_unicode_data(args.dry_run)246 247    # 2. Shannon lexicon248    print("\n[2/4] jhnwnstd/shannon Linear B Lexicon (MIT)")249    shannon_path = download_shannon_lexicon(args.dry_run)250 251    # 3. Wiktionary lemmas252    print("\n[3/4] Wiktionary Mycenaean Greek lemmas (CC-BY-SA)")253    wikt_path = download_wiktionary_lemmas(args.dry_run)254 255    # 4. Wiktionary Swadesh list256    print("\n[4/4] Wiktionary Mycenaean Greek Swadesh list")257    swadesh_path = download_wiktionary_swadesh(args.dry_run)258 259    print("\n" + "=" * 70)260    print("INGESTION COMPLETE")261    print("=" * 70)262 263    # Verify files264    for label, path in [265        ("UnicodeData.txt", ucd_path),266        ("Shannon Lexicon", shannon_path),267        ("Wiktionary Lemmas", wikt_path),268        ("Wiktionary Swadesh", swadesh_path),269    ]:270        if path.exists():271            size = path.stat().st_size272            print(f"  {label}: {path.name} ({size:,} bytes)")273        else:274            print(f"  {label}: NOT DOWNLOADED")275 276    print(f"\nAll raw data saved to: {RAW_DIR}")277 278 279if __name__ == "__main__":280    main()281