uv-scripts/ocr
OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.
1636.5k
1# /// script2# requires-python = ">=3.11"3# dependencies = [4# "datasets>=4.0.0",5# "huggingface-hub",6# "transformers",7# "tqdm",8# "toolz",9# ]10#11# [tool.hf-jobs]12# image = "vllm/vllm-openai:v0.29.0"13# python = "/usr/bin/python3"14# env = { PYTHONPATH = "/usr/local/lib/python3.12/dist-packages" }15# flavor = "a10g-small"16# secrets = ["HF_TOKEN"]17# ///18"""19Extract structured data (JSON / XML / YAML) from text using LiquidAI's LFM2-1.2B-Extract.20 21LFM2-1.2B-Extract is a compact 1.2B text-only model purpose-built for turning unstructured22documents into structured data: give it a schema, it returns JSON, XML, or YAML. It reports23beating Gemma 3 27B (22x larger) on syntax validity / format accuracy / faithfulness, and24is multilingual (en, ar, zh, fr, de, ja, ko, pt, es).25 26This is the *text* counterpart to `lfm2-vl-extract.py` (which extracts from images). Pair them:27OCR a page to markdown with one of the OCR recipes, then extract fields from that text here.28 29Pass `--schema` as inline text/JSON, a URL, or a file path describing the structure to extract:30 31 --schema '{"invoice_number": "string", "total": "number", "line_items": "array"}'32 33Model: https://huggingface.co/LiquidAI/LFM2-1.2B-Extract34Docs: https://docs.liquid.ai/deployment/gpu-inference/vllm35 36HF Jobs: the `[tool.hf-jobs]` header above pins the vLLM image (vLLM + torch come from the37image, not the deps), the GPU flavor and the HF_TOKEN secret, so no flags are needed38(requires `hf` CLI 1.32+):39 40 hf jobs uv run \41 https://huggingface.co/datasets/uv-scripts/ocr/raw/main/lfm2-extract.py \42 INPUT OUTPUT --text-column text --schema '{"field": "description"}'43 44On your own GPU, supply vLLM yourself:45 46 uv run --with vllm==0.29.0 lfm2-extract.py INPUT OUTPUT --schema '{"field": "description"}'47 48FlashInfer sampling is disabled (see below) so the engine never JIT-compiles a kernel that49needs nvcc.50"""51 52import argparse53import json54import logging55import os56import sys57from datetime import datetime, timezone58from typing import List, Optional59from urllib.request import urlopen60 61# Disable vLLM's FlashInfer sampler before the engine starts: it JIT-compiles at warmup and62# needs nvcc (absent from the default uv image). Harmless for greedy decoding.63os.environ.setdefault("VLLM_USE_FLASHINFER_SAMPLER", "0")64 65import torch66from datasets import load_dataset67from huggingface_hub import DatasetCard, login68from toolz import partition_all69from tqdm import tqdm70from vllm import LLM, SamplingParams71 72logging.basicConfig(level=logging.INFO)73logger = logging.getLogger(__name__)74 75DEFAULT_MODEL = "LiquidAI/LFM2-1.2B-Extract"76FORMATS = {"json": "JSON", "xml": "XML", "yaml": "YAML"}77 78 79def check_cuda_availability() -> None:80 if not torch.cuda.is_available():81 logger.error("CUDA is not available. This script requires a GPU.")82 logger.error("Run on Hugging Face Jobs with: hf jobs uv run lfm2-extract.py ...")83 sys.exit(1)84 logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name()}")85 86 87def ensure_output_columns_free(dataset, columns, overwrite=False):88 """Fail fast if an output column would collide with an existing input column.89 90 Adding a column that already exists silently overwrites it (e.g. a ground-truth91 `text`/`markdown` column) or crashes on push with a duplicate-column error only92 *after* inference has run. Catch it up front. With overwrite=True, drop the clashing93 column(s) here instead (logged) so the later add_column is clean.94 """95 clash = [c for c in columns if c in dataset.column_names]96 if not clash:97 return dataset98 if overwrite:99 logger.warning(f"--overwrite: replacing existing column(s) {clash}")100 return dataset.remove_columns(clash)101 logger.error(102 f"Output column(s) {clash} already exist in the input dataset "103 f"(columns: {dataset.column_names})."104 )105 logger.error("Choose a different --output-column, or pass --overwrite to replace them.")106 sys.exit(1)107 108 109def load_text_arg(value: str) -> str:110 """Resolve --schema (inline text/JSON, URL, or file path) into a string."""111 text = value.strip()112 if text.startswith("http://") or text.startswith("https://"):113 logger.info(f"Loading schema from URL: {text}")114 return urlopen(text).read().decode("utf-8").strip()115 if os.path.exists(text):116 logger.info(f"Loading schema from file: {text}")117 with open(text) as f:118 return f.read().strip()119 return text120 121 122def build_system_prompt(schema_text: str, fmt: str) -> str:123 return f"Return data as a {FORMATS[fmt]} object with the following schema:\n\n{schema_text}"124 125 126def parse_output(text: str, fmt: str) -> tuple[str, bool]:127 """Strip code fences; for JSON, validate. Returns (cleaned_text, is_valid)."""128 stripped = text.strip()129 if stripped.startswith("```"):130 stripped = stripped.split("\n", 1)[-1]131 if stripped.endswith("```"):132 stripped = stripped.rsplit("```", 1)[0]133 stripped = stripped.strip()134 if fmt == "json":135 try:136 return json.dumps(json.loads(stripped), ensure_ascii=False), True137 except (json.JSONDecodeError, ValueError):138 return stripped, False139 return stripped, True # xml/yaml: store as-is (no strict validator)140 141 142def main(143 input_dataset: str,144 output_dataset: str,145 schema: str,146 text_column: str = "text",147 output_column: str = "extraction",148 overwrite: bool = False,149 output_format: str = "json",150 split: str = "train",151 max_samples: Optional[int] = None,152 shuffle: bool = False,153 seed: int = 42,154 batch_size: int = 32,155 model: str = DEFAULT_MODEL,156 max_model_len: int = 8192,157 max_tokens: int = 4096,158 private: bool = False,159 hf_token: Optional[str] = None,160) -> None:161 check_cuda_availability()162 if output_format not in FORMATS:163 logger.error(f"--format must be one of {list(FORMATS)}; got {output_format}")164 sys.exit(1)165 166 HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")167 if HF_TOKEN:168 login(token=HF_TOKEN)169 170 schema_text = load_text_arg(schema)171 system_prompt = build_system_prompt(schema_text, output_format)172 173 logger.info(f"Loading dataset: {input_dataset} (split={split})")174 dataset = load_dataset(input_dataset, split=split)175 176 # Fail fast if the output column would collide with an existing input column177 dataset = ensure_output_columns_free(dataset, [output_column], overwrite=overwrite)178 179 if shuffle:180 dataset = dataset.shuffle(seed=seed)181 if max_samples:182 dataset = dataset.select(range(min(max_samples, len(dataset))))183 logger.info(f"Processing {len(dataset)} examples; format={output_format}")184 185 if text_column not in dataset.column_names:186 logger.error(f"Text column '{text_column}' not found. Columns: {dataset.column_names}")187 sys.exit(1)188 189 logger.info(f"Loading model: {model}")190 llm = LLM(model=model, max_model_len=max_model_len, enforce_eager=True)191 sampling_params = SamplingParams(temperature=0.0, max_tokens=max_tokens)192 193 all_outputs: List[str] = []194 n_valid = 0195 texts = dataset[text_column]196 for batch in tqdm(list(partition_all(batch_size, texts)), desc="Extracting"):197 batch_messages = [198 [199 {"role": "system", "content": system_prompt},200 {"role": "user", "content": str(doc)},201 ]202 for doc in batch203 ]204 outputs = llm.chat(batch_messages, sampling_params)205 for out in outputs:206 cleaned, ok = parse_output(out.outputs[0].text, output_format)207 n_valid += int(ok)208 all_outputs.append(cleaned)209 210 logger.info(f"Valid {output_format.upper()}: {n_valid}/{len(all_outputs)}")211 dataset = dataset.add_column(output_column, all_outputs)212 213 inference_entry = {214 "model": model,215 "column_name": output_column,216 "task": "structured extraction",217 "format": output_format,218 "timestamp": datetime.now(timezone.utc).isoformat(),219 "script": "lfm2-extract.py",220 }221 if "inference_info" in dataset.column_names:222 def update_info(example):223 try:224 existing = json.loads(example["inference_info"]) if example["inference_info"] else []225 except (json.JSONDecodeError, TypeError):226 existing = []227 existing.append(inference_entry)228 return {"inference_info": json.dumps(existing)}229 dataset = dataset.map(update_info)230 else:231 dataset = dataset.add_column(232 "inference_info", [json.dumps([inference_entry])] * len(dataset)233 )234 235 logger.info(f"Pushing to {output_dataset}")236 dataset.push_to_hub(output_dataset, private=private, token=HF_TOKEN)237 238 card_text = f"""---239tags:240- uv-script241- extraction242- lfm2243- {output_format}244---245 246# Structured extraction with LFM2-1.2B-Extract247 248`{output_format.upper()}` extracted from the `{text_column}` column of249[{input_dataset}](https://huggingface.co/datasets/{input_dataset})250using [{model}](https://huggingface.co/{model}).251 252- **Source**: `{input_dataset}` (split `{split}`, column `{text_column}`)253- **Model**: `{model}`254- **Format**: `{output_format}`255- **Output column**: `{output_column}`256- **Valid {output_format.upper()}**: {n_valid}/{len(all_outputs)}257- **Date**: {datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")}258 259Generated with the [uv-scripts/ocr](https://huggingface.co/datasets/uv-scripts/ocr) `lfm2-extract.py` script.260"""261 try:262 DatasetCard(card_text).push_to_hub(output_dataset, token=HF_TOKEN)263 except Exception as e:264 logger.warning(f"Could not push dataset card: {e}")265 266 logger.info("Done! Extraction complete.")267 logger.info(f"Dataset: https://huggingface.co/datasets/{output_dataset}")268 269 270if __name__ == "__main__":271 if len(sys.argv) == 1:272 print("LFM2-1.2B-Extract — structured extraction (JSON/XML/YAML) from text")273 print("\nUsage:")274 print(" uv run --with vllm==0.29.0 lfm2-extract.py INPUT OUTPUT --schema SCHEMA [--text-column text] [--format json]")275 print("\nExample:")276 print(' uv run --with vllm==0.29.0 lfm2-extract.py my-docs my-fields \\')277 print(' --text-column markdown \\')278 print(' --schema \'{"title": "the title", "date": "any date", "summary": "one sentence"}\'')279 print("\n --schema accepts inline text/JSON, a URL, or a file path.")280 print("\nOn HF Jobs (hf CLI 1.32+; image/flavor/secrets come from the script header):")281 print(" hf jobs uv run lfm2-extract.py INPUT OUTPUT --schema SCHEMA")282 print("\nFor full help: uv run --with vllm==0.29.0 lfm2-extract.py --help")283 sys.exit(0)284 285 parser = argparse.ArgumentParser(286 description="Structured extraction (JSON/XML/YAML) from text using LFM2-1.2B-Extract",287 )288 parser.add_argument("input_dataset", help="Input dataset ID (with a text column)")289 parser.add_argument("output_dataset", help="Output dataset ID")290 parser.add_argument(291 "--schema", required=True,292 help="Structure to extract: inline text/JSON, a URL, or a file path",293 )294 parser.add_argument("--text-column", default="text", help="Text column (default: text)")295 parser.add_argument("--output-column", default="extraction", help="Output column (default: extraction)")296 parser.add_argument(297 "--overwrite",298 action="store_true",299 help="Replace the output column if it already exists in the input dataset "300 "(default: error out to avoid clobbering an existing column).",301 )302 parser.add_argument(303 "--format", dest="output_format", default="json", choices=list(FORMATS),304 help="Output format (default: json)",305 )306 parser.add_argument("--split", default="train", help="Dataset split (default: train)")307 parser.add_argument("--max-samples", type=int, help="Limit number of samples")308 parser.add_argument("--shuffle", action="store_true", help="Shuffle before sampling")309 parser.add_argument("--seed", type=int, default=42, help="Shuffle seed (default: 42)")310 parser.add_argument("--batch-size", type=int, default=32, help="Batch size (default: 32)")311 parser.add_argument("--model", default=DEFAULT_MODEL, help=f"Model (default: {DEFAULT_MODEL})")312 parser.add_argument("--max-model-len", type=int, default=8192, help="Max context length (default: 8192)")313 parser.add_argument("--max-tokens", type=int, default=4096, help="Max output tokens (default: 4096)")314 parser.add_argument("--private", action="store_true", help="Make output dataset private")315 parser.add_argument("--hf-token", help="HF token (or set HF_TOKEN)")316 args = parser.parse_args()317 318 main(319 input_dataset=args.input_dataset,320 output_dataset=args.output_dataset,321 schema=args.schema,322 text_column=args.text_column,323 output_column=args.output_column,324 overwrite=args.overwrite,325 output_format=args.output_format,326 split=args.split,327 max_samples=args.max_samples,328 shuffle=args.shuffle,329 seed=args.seed,330 batch_size=args.batch_size,331 model=args.model,332 max_model_len=args.max_model_len,333 max_tokens=args.max_tokens,334 private=args.private,335 hf_token=args.hf_token,336 )337 