Team Ai
Datasetpublic

uv-scripts/ocr

OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.

sourceHugging Faceupdated 10d agoView on Hugging Face
163likes6.5kdownloads
lfm2-extract.py337 linesDownload Raw Back to root
1# /// script2# requires-python = ">=3.11"3# dependencies = [4#     "datasets>=4.0.0",5#     "huggingface-hub",6#     "transformers",7#     "tqdm",8#     "toolz",9# ]10#11# [tool.hf-jobs]12# image = "vllm/vllm-openai:v0.29.0"13# python = "/usr/bin/python3"14# env = { PYTHONPATH = "/usr/local/lib/python3.12/dist-packages" }15# flavor = "a10g-small"16# secrets = ["HF_TOKEN"]17# ///18"""19Extract structured data (JSON / XML / YAML) from text using LiquidAI's LFM2-1.2B-Extract.20 21LFM2-1.2B-Extract is a compact 1.2B text-only model purpose-built for turning unstructured22documents into structured data: give it a schema, it returns JSON, XML, or YAML. It reports23beating Gemma 3 27B (22x larger) on syntax validity / format accuracy / faithfulness, and24is multilingual (en, ar, zh, fr, de, ja, ko, pt, es).25 26This is the *text* counterpart to `lfm2-vl-extract.py` (which extracts from images). Pair them:27OCR a page to markdown with one of the OCR recipes, then extract fields from that text here.28 29Pass `--schema` as inline text/JSON, a URL, or a file path describing the structure to extract:30 31    --schema '{"invoice_number": "string", "total": "number", "line_items": "array"}'32 33Model:  https://huggingface.co/LiquidAI/LFM2-1.2B-Extract34Docs:   https://docs.liquid.ai/deployment/gpu-inference/vllm35 36HF Jobs: the `[tool.hf-jobs]` header above pins the vLLM image (vLLM + torch come from the37image, not the deps), the GPU flavor and the HF_TOKEN secret, so no flags are needed38(requires `hf` CLI 1.32+):39 40    hf jobs uv run \41        https://huggingface.co/datasets/uv-scripts/ocr/raw/main/lfm2-extract.py \42        INPUT OUTPUT --text-column text --schema '{"field": "description"}'43 44On your own GPU, supply vLLM yourself:45 46    uv run --with vllm==0.29.0 lfm2-extract.py INPUT OUTPUT --schema '{"field": "description"}'47 48FlashInfer sampling is disabled (see below) so the engine never JIT-compiles a kernel that49needs nvcc.50"""51 52import argparse53import json54import logging55import os56import sys57from datetime import datetime, timezone58from typing import List, Optional59from urllib.request import urlopen60 61# Disable vLLM's FlashInfer sampler before the engine starts: it JIT-compiles at warmup and62# needs nvcc (absent from the default uv image). Harmless for greedy decoding.63os.environ.setdefault("VLLM_USE_FLASHINFER_SAMPLER", "0")64 65import torch66from datasets import load_dataset67from huggingface_hub import DatasetCard, login68from toolz import partition_all69from tqdm import tqdm70from vllm import LLM, SamplingParams71 72logging.basicConfig(level=logging.INFO)73logger = logging.getLogger(__name__)74 75DEFAULT_MODEL = "LiquidAI/LFM2-1.2B-Extract"76FORMATS = {"json": "JSON", "xml": "XML", "yaml": "YAML"}77 78 79def check_cuda_availability() -> None:80    if not torch.cuda.is_available():81        logger.error("CUDA is not available. This script requires a GPU.")82        logger.error("Run on Hugging Face Jobs with: hf jobs uv run lfm2-extract.py ...")83        sys.exit(1)84    logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name()}")85 86 87def ensure_output_columns_free(dataset, columns, overwrite=False):88    """Fail fast if an output column would collide with an existing input column.89 90    Adding a column that already exists silently overwrites it (e.g. a ground-truth91    `text`/`markdown` column) or crashes on push with a duplicate-column error only92    *after* inference has run. Catch it up front. With overwrite=True, drop the clashing93    column(s) here instead (logged) so the later add_column is clean.94    """95    clash = [c for c in columns if c in dataset.column_names]96    if not clash:97        return dataset98    if overwrite:99        logger.warning(f"--overwrite: replacing existing column(s) {clash}")100        return dataset.remove_columns(clash)101    logger.error(102        f"Output column(s) {clash} already exist in the input dataset "103        f"(columns: {dataset.column_names})."104    )105    logger.error("Choose a different --output-column, or pass --overwrite to replace them.")106    sys.exit(1)107 108 109def load_text_arg(value: str) -> str:110    """Resolve --schema (inline text/JSON, URL, or file path) into a string."""111    text = value.strip()112    if text.startswith("http://") or text.startswith("https://"):113        logger.info(f"Loading schema from URL: {text}")114        return urlopen(text).read().decode("utf-8").strip()115    if os.path.exists(text):116        logger.info(f"Loading schema from file: {text}")117        with open(text) as f:118            return f.read().strip()119    return text120 121 122def build_system_prompt(schema_text: str, fmt: str) -> str:123    return f"Return data as a {FORMATS[fmt]} object with the following schema:\n\n{schema_text}"124 125 126def parse_output(text: str, fmt: str) -> tuple[str, bool]:127    """Strip code fences; for JSON, validate. Returns (cleaned_text, is_valid)."""128    stripped = text.strip()129    if stripped.startswith("```"):130        stripped = stripped.split("\n", 1)[-1]131        if stripped.endswith("```"):132            stripped = stripped.rsplit("```", 1)[0]133    stripped = stripped.strip()134    if fmt == "json":135        try:136            return json.dumps(json.loads(stripped), ensure_ascii=False), True137        except (json.JSONDecodeError, ValueError):138            return stripped, False139    return stripped, True  # xml/yaml: store as-is (no strict validator)140 141 142def main(143    input_dataset: str,144    output_dataset: str,145    schema: str,146    text_column: str = "text",147    output_column: str = "extraction",148    overwrite: bool = False,149    output_format: str = "json",150    split: str = "train",151    max_samples: Optional[int] = None,152    shuffle: bool = False,153    seed: int = 42,154    batch_size: int = 32,155    model: str = DEFAULT_MODEL,156    max_model_len: int = 8192,157    max_tokens: int = 4096,158    private: bool = False,159    hf_token: Optional[str] = None,160) -> None:161    check_cuda_availability()162    if output_format not in FORMATS:163        logger.error(f"--format must be one of {list(FORMATS)}; got {output_format}")164        sys.exit(1)165 166    HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")167    if HF_TOKEN:168        login(token=HF_TOKEN)169 170    schema_text = load_text_arg(schema)171    system_prompt = build_system_prompt(schema_text, output_format)172 173    logger.info(f"Loading dataset: {input_dataset} (split={split})")174    dataset = load_dataset(input_dataset, split=split)175 176    # Fail fast if the output column would collide with an existing input column177    dataset = ensure_output_columns_free(dataset, [output_column], overwrite=overwrite)178 179    if shuffle:180        dataset = dataset.shuffle(seed=seed)181    if max_samples:182        dataset = dataset.select(range(min(max_samples, len(dataset))))183    logger.info(f"Processing {len(dataset)} examples; format={output_format}")184 185    if text_column not in dataset.column_names:186        logger.error(f"Text column '{text_column}' not found. Columns: {dataset.column_names}")187        sys.exit(1)188 189    logger.info(f"Loading model: {model}")190    llm = LLM(model=model, max_model_len=max_model_len, enforce_eager=True)191    sampling_params = SamplingParams(temperature=0.0, max_tokens=max_tokens)192 193    all_outputs: List[str] = []194    n_valid = 0195    texts = dataset[text_column]196    for batch in tqdm(list(partition_all(batch_size, texts)), desc="Extracting"):197        batch_messages = [198            [199                {"role": "system", "content": system_prompt},200                {"role": "user", "content": str(doc)},201            ]202            for doc in batch203        ]204        outputs = llm.chat(batch_messages, sampling_params)205        for out in outputs:206            cleaned, ok = parse_output(out.outputs[0].text, output_format)207            n_valid += int(ok)208            all_outputs.append(cleaned)209 210    logger.info(f"Valid {output_format.upper()}: {n_valid}/{len(all_outputs)}")211    dataset = dataset.add_column(output_column, all_outputs)212 213    inference_entry = {214        "model": model,215        "column_name": output_column,216        "task": "structured extraction",217        "format": output_format,218        "timestamp": datetime.now(timezone.utc).isoformat(),219        "script": "lfm2-extract.py",220    }221    if "inference_info" in dataset.column_names:222        def update_info(example):223            try:224                existing = json.loads(example["inference_info"]) if example["inference_info"] else []225            except (json.JSONDecodeError, TypeError):226                existing = []227            existing.append(inference_entry)228            return {"inference_info": json.dumps(existing)}229        dataset = dataset.map(update_info)230    else:231        dataset = dataset.add_column(232            "inference_info", [json.dumps([inference_entry])] * len(dataset)233        )234 235    logger.info(f"Pushing to {output_dataset}")236    dataset.push_to_hub(output_dataset, private=private, token=HF_TOKEN)237 238    card_text = f"""---239tags:240- uv-script241- extraction242- lfm2243- {output_format}244---245 246# Structured extraction with LFM2-1.2B-Extract247 248`{output_format.upper()}` extracted from the `{text_column}` column of249[{input_dataset}](https://huggingface.co/datasets/{input_dataset})250using [{model}](https://huggingface.co/{model}).251 252- **Source**: `{input_dataset}` (split `{split}`, column `{text_column}`)253- **Model**: `{model}`254- **Format**: `{output_format}`255- **Output column**: `{output_column}`256- **Valid {output_format.upper()}**: {n_valid}/{len(all_outputs)}257- **Date**: {datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")}258 259Generated with the [uv-scripts/ocr](https://huggingface.co/datasets/uv-scripts/ocr) `lfm2-extract.py` script.260"""261    try:262        DatasetCard(card_text).push_to_hub(output_dataset, token=HF_TOKEN)263    except Exception as e:264        logger.warning(f"Could not push dataset card: {e}")265 266    logger.info("Done! Extraction complete.")267    logger.info(f"Dataset: https://huggingface.co/datasets/{output_dataset}")268 269 270if __name__ == "__main__":271    if len(sys.argv) == 1:272        print("LFM2-1.2B-Extract — structured extraction (JSON/XML/YAML) from text")273        print("\nUsage:")274        print("  uv run --with vllm==0.29.0 lfm2-extract.py INPUT OUTPUT --schema SCHEMA [--text-column text] [--format json]")275        print("\nExample:")276        print('  uv run --with vllm==0.29.0 lfm2-extract.py my-docs my-fields \\')277        print('    --text-column markdown \\')278        print('    --schema \'{"title": "the title", "date": "any date", "summary": "one sentence"}\'')279        print("\n  --schema accepts inline text/JSON, a URL, or a file path.")280        print("\nOn HF Jobs (hf CLI 1.32+; image/flavor/secrets come from the script header):")281        print("  hf jobs uv run lfm2-extract.py INPUT OUTPUT --schema SCHEMA")282        print("\nFor full help: uv run --with vllm==0.29.0 lfm2-extract.py --help")283        sys.exit(0)284 285    parser = argparse.ArgumentParser(286        description="Structured extraction (JSON/XML/YAML) from text using LFM2-1.2B-Extract",287    )288    parser.add_argument("input_dataset", help="Input dataset ID (with a text column)")289    parser.add_argument("output_dataset", help="Output dataset ID")290    parser.add_argument(291        "--schema", required=True,292        help="Structure to extract: inline text/JSON, a URL, or a file path",293    )294    parser.add_argument("--text-column", default="text", help="Text column (default: text)")295    parser.add_argument("--output-column", default="extraction", help="Output column (default: extraction)")296    parser.add_argument(297        "--overwrite",298        action="store_true",299        help="Replace the output column if it already exists in the input dataset "300        "(default: error out to avoid clobbering an existing column).",301    )302    parser.add_argument(303        "--format", dest="output_format", default="json", choices=list(FORMATS),304        help="Output format (default: json)",305    )306    parser.add_argument("--split", default="train", help="Dataset split (default: train)")307    parser.add_argument("--max-samples", type=int, help="Limit number of samples")308    parser.add_argument("--shuffle", action="store_true", help="Shuffle before sampling")309    parser.add_argument("--seed", type=int, default=42, help="Shuffle seed (default: 42)")310    parser.add_argument("--batch-size", type=int, default=32, help="Batch size (default: 32)")311    parser.add_argument("--model", default=DEFAULT_MODEL, help=f"Model (default: {DEFAULT_MODEL})")312    parser.add_argument("--max-model-len", type=int, default=8192, help="Max context length (default: 8192)")313    parser.add_argument("--max-tokens", type=int, default=4096, help="Max output tokens (default: 4096)")314    parser.add_argument("--private", action="store_true", help="Make output dataset private")315    parser.add_argument("--hf-token", help="HF token (or set HF_TOKEN)")316    args = parser.parse_args()317 318    main(319        input_dataset=args.input_dataset,320        output_dataset=args.output_dataset,321        schema=args.schema,322        text_column=args.text_column,323        output_column=args.output_column,324        overwrite=args.overwrite,325        output_format=args.output_format,326        split=args.split,327        max_samples=args.max_samples,328        shuffle=args.shuffle,329        seed=args.seed,330        batch_size=args.batch_size,331        model=args.model,332        max_model_len=args.max_model_len,333        max_tokens=args.max_tokens,334        private=args.private,335        hf_token=args.hf_token,336    )337