uv-scripts/ocr
OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.
1636.5k
1# /// script2# requires-python = ">=3.11"3# dependencies = [4# "datasets>=4.0.0",5# "huggingface-hub",6# "pillow",7# "tqdm",8# "toolz",9# ]10#11# [tool.hf-jobs]12# image = "vllm/vllm-openai:v0.22.1"13# python = "/usr/bin/python3"14# env = { PYTHONPATH = "/usr/local/lib/python3.12/dist-packages" }15# flavor = "a10g-small"16# secrets = ["HF_TOKEN"]17# ///18 19"""20Convert document images to markdown using LightOnOCR-2 with vLLM.21 22LightOnOCR-2 is a compact 1B multilingual OCR model optimized for production speed.23Combines Pixtral ViT encoder with Qwen3 language model for efficient document parsing.24Uses Reinforcement Learning with Verifiable Rewards (RLVR) for improved quality.25 26Run on HF Jobs. vLLM and torch come from the vllm/vllm-openai:v0.22.1 image declared27in the [tool.hf-jobs] header (`hf` CLI 1.32+), which also sets the hardware and the28HF_TOKEN secret. The tag is pinned: unpinned vLLM 0.29/0.30 with current transformers29fails to import LightOnOCR-2 (PixtralRotaryEmbedding). Pass --timeout for a long run:30 31 hf jobs uv run --timeout 1h \\32 https://huggingface.co/datasets/uv-scripts/ocr/raw/main/lighton-ocr2.py \\33 <input-dataset> <output-dataset>34 35To run on your own GPU, add the engine: `uv run --with vllm==0.22.1 lighton-ocr2.py ...`.36 37Features:38- ⚡ Fastest: 42.8 pages/sec on H100 GPU (7× faster than v1)39- 🎯 High accuracy: 83.2 ± 0.9% on OlmOCR-Bench (+7.1% vs v1)40- 🧠 RLVR trained: Eliminates repetition loops and formatting errors41- 📚 Better training: 2.5× larger dataset with cleaner annotations42- 🌍 Multilingual with European language optimization43- 📐 LaTeX formula recognition44- 📊 Table extraction (markdown format)45- 📝 Document structure preservation46- 💪 Production-ready: Outperforms models 9× larger47 48Model: lightonai/LightOnOCR-2-1B49vLLM: vllm/vllm-openai:v0.22.1 image (see the header)50Performance: 83.2 ± 0.9% on OlmOCR-Bench51"""52 53import argparse54import base6455import io56import json57import logging58import os59import sys60from typing import Any, Dict, List, Union61from datetime import datetime62 63import torch64from datasets import load_dataset65from huggingface_hub import DatasetCard, login66from PIL import Image67from toolz import partition_all68from tqdm.auto import tqdm69# Disable vLLM's FlashInfer sampler: it JIT-compiles a CUDA kernel needing nvcc, which the70# default uv-script image lacks (engine init then crashes). Greedy OCR doesn't use it; this71# lets the plain default-image command work. On the vllm/vllm-openai image it's a harmless no-op.72os.environ.setdefault("VLLM_USE_FLASHINFER_SAMPLER", "0")73from vllm import LLM, SamplingParams74 75logging.basicConfig(level=logging.INFO)76logger = logging.getLogger(__name__)77 78 79# LightOnOCR-2 model (single variant)80MODEL = "lightonai/LightOnOCR-2-1B"81 82 83def check_cuda_availability():84 """Check if CUDA is available and exit if not."""85 if not torch.cuda.is_available():86 logger.error("CUDA is not available. This script requires a GPU.")87 logger.error("Please run on a machine with a CUDA-capable GPU.")88 sys.exit(1)89 else:90 logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name(0)}")91 92 93def ensure_output_columns_free(dataset, columns, overwrite=False):94 """Fail fast if an output column would collide with an existing input column.95 96 Adding a column that already exists silently overwrites it (e.g. a ground-truth97 `text`/`markdown` column) or crashes on push with a duplicate-column error only98 *after* inference has run. Catch it up front. With overwrite=True, drop the clashing99 column(s) here instead (logged) so the later add_column is clean.100 """101 clash = [c for c in columns if c in dataset.column_names]102 if not clash:103 return dataset104 if overwrite:105 logger.warning(f"--overwrite: replacing existing column(s) {clash}")106 return dataset.remove_columns(clash)107 logger.error(108 f"Output column(s) {clash} already exist in the input dataset "109 f"(columns: {dataset.column_names})."110 )111 logger.error("Choose a different --output-column, or pass --overwrite to replace them.")112 sys.exit(1)113 114 115def resize_image_to_target(image: Image.Image, target_size: int = 1540) -> Image.Image:116 """117 Resize image so longest dimension is target_size while maintaining aspect ratio.118 119 LightOnOCR-2 was trained with images at 1540px max resolution and 200 DPI.120 """121 width, height = image.size122 123 # If image is already smaller, don't upscale124 if max(width, height) <= target_size:125 return image126 127 # Calculate new dimensions maintaining aspect ratio128 if width > height:129 new_width = target_size130 new_height = int(height * (target_size / width))131 else:132 new_height = target_size133 new_width = int(width * (target_size / height))134 135 return image.resize((new_width, new_height), Image.Resampling.LANCZOS)136 137 138def make_ocr_message(139 image: Union[Image.Image, Dict[str, Any], str],140 resize: bool = True,141 target_size: int = 1540,142) -> List[Dict]:143 """144 Create chat message for OCR processing.145 146 LightOnOCR-2 was trained with 1540px max resolution at 200 DPI for optimal results.147 Unlike v1, LightOnOCR-2 does NOT use an empty text prefix - just the image.148 """149 # Convert to PIL Image if needed150 if isinstance(image, Image.Image):151 pil_img = image152 elif isinstance(image, dict) and "bytes" in image:153 pil_img = Image.open(io.BytesIO(image["bytes"]))154 elif isinstance(image, str):155 pil_img = Image.open(image)156 else:157 raise ValueError(f"Unsupported image type: {type(image)}")158 159 # Convert to RGB160 pil_img = pil_img.convert("RGB")161 162 # Resize to optimal dimensions for LightOnOCR-2163 if resize:164 pil_img = resize_image_to_target(pil_img, target_size)165 logger.debug(f"Resized image to {pil_img.size}")166 167 # Convert to base64 data URI168 buf = io.BytesIO()169 pil_img.save(buf, format="PNG")170 data_uri = f"data:image/png;base64,{base64.b64encode(buf.getvalue()).decode()}"171 172 # LightOnOCR-2 uses message format with ONLY the image (no text prefix)173 return [174 {175 "role": "user",176 "content": [177 {"type": "image_url", "image_url": {"url": data_uri}},178 ],179 }180 ]181 182 183def create_dataset_card(184 source_dataset: str,185 model: str,186 num_samples: int,187 processing_time: str,188 batch_size: int,189 max_model_len: int,190 max_tokens: int,191 gpu_memory_utilization: float,192 temperature: float,193 top_p: float,194 target_size: int,195 image_column: str = "image",196 split: str = "train",197) -> str:198 """Create a dataset card documenting the OCR process."""199 model_name = model.split("/")[-1]200 201 return f"""---202tags:203- ocr204- document-processing205- lighton-ocr-2206- markdown207- uv-script208- generated209---210 211# Document OCR using {model_name}212 213This dataset contains OCR results from images in [{source_dataset}](https://huggingface.co/datasets/{source_dataset}) using LightOnOCR-2, a fast and compact 1B OCR model trained with RLVR.214 215## Processing Details216 217- **Source Dataset**: [{source_dataset}](https://huggingface.co/datasets/{source_dataset})218- **Model**: [{model}](https://huggingface.co/{model})219- **Number of Samples**: {num_samples:,}220- **Processing Time**: {processing_time}221- **Processing Date**: {datetime.now().strftime("%Y-%m-%d %H:%M UTC")}222 223### Configuration224 225- **Image Column**: `{image_column}`226- **Output Column**: `markdown`227- **Dataset Split**: `{split}`228- **Batch Size**: {batch_size}229- **Target Image Size**: {target_size}px (longest dimension)230- **Max Model Length**: {max_model_len:,} tokens231- **Max Output Tokens**: {max_tokens:,}232- **Temperature**: {temperature}233- **Top P**: {top_p}234- **GPU Memory Utilization**: {gpu_memory_utilization:.1%}235 236## Model Information237 238LightOnOCR-2 is a next-generation fast, compact OCR model that excels at:239- ⚡ **Fastest Speed** - 42.8 pages/second on H100 GPU (7× faster than v1)240- 🎯 **High Accuracy** - 83.2 ± 0.9% on OlmOCR-Bench (+7.1% vs v1)241- 🧠 **RLVR Training** - Eliminates repetition loops and formatting errors242- 📚 **Better Dataset** - 2.5× larger training data with cleaner annotations243- 📐 **LaTeX formulas** - Mathematical notation in LaTeX format244- 📊 **Tables** - Extracted and formatted as markdown245- 📝 **Document structure** - Hierarchy and layout preservation246- 🌍 **Multilingual** - Optimized for European languages247- 💪 **Production-ready** - Outperforms models 9× larger248 249### Key Improvements over v1250 251- **7.5× faster**: 42.8 vs 5.71 pages/sec on H100252- **+7.1% accuracy**: 83.2% vs 76.1% on benchmarks253- **Better quality**: RLVR training eliminates common OCR errors254- **Cleaner output**: No repetition loops or formatting glitches255- **Simpler**: Single model (no vocabulary variants)256 257## Dataset Structure258 259The dataset contains all original columns plus:260- `markdown`: The extracted text in markdown format with LaTeX formulas261- `inference_info`: JSON list tracking all OCR models applied to this dataset262 263## Usage264 265```python266from datasets import load_dataset267import json268 269# Load the dataset270dataset = load_dataset("{{output_dataset_id}}", split="{split}")271 272# Access the markdown text273for example in dataset:274 print(example["markdown"])275 break276 277# View all OCR models applied to this dataset278inference_info = json.loads(dataset[0]["inference_info"])279for info in inference_info:280 print(f"Column: {{info['column_name']}} - Model: {{info['model_id']}}")281```282 283## Reproduction284 285This dataset was generated using the [uv-scripts/ocr](https://huggingface.co/datasets/uv-scripts/ocr) LightOnOCR-2 script:286 287```bash288uv run https://huggingface.co/datasets/uv-scripts/ocr/raw/main/lighton-ocr2.py \\289 {source_dataset} \\290 <output-dataset> \\291 --image-column {image_column} \\292 --batch-size {batch_size}293```294 295## Performance296 297- **Processing Speed**: ~{num_samples / (float(processing_time.split()[0]) * 60):.2f} images/second298- **Benchmark Score**: 83.2 ± 0.9% on OlmOCR-Bench299- **Training**: RLVR (Reinforcement Learning with Verifiable Rewards)300 301Generated with 🤖 [UV Scripts](https://huggingface.co/uv-scripts)302"""303 304 305def main(306 input_dataset: str,307 output_dataset: str,308 image_column: str = "image",309 batch_size: int = 16,310 max_model_len: int = 8192,311 max_tokens: int = 4096,312 temperature: float = 0.2,313 top_p: float = 0.9,314 gpu_memory_utilization: float = 0.8,315 target_size: int = 1540,316 no_resize: bool = False,317 hf_token: str = None,318 split: str = "train",319 max_samples: int = None,320 private: bool = False,321 shuffle: bool = False,322 seed: int = 42,323 output_column: str = "markdown",324 overwrite: bool = False,325 config: str = None,326 create_pr: bool = False,327 verbose: bool = False,328):329 """Process images from HF dataset through LightOnOCR-2 model."""330 331 # Check CUDA availability first332 check_cuda_availability()333 334 # Track processing start time335 start_time = datetime.now()336 337 # Login to HF if token provided338 HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")339 if HF_TOKEN:340 login(token=HF_TOKEN)341 342 logger.info(f"Using model: {MODEL}")343 344 # Load dataset345 logger.info(f"Loading dataset: {input_dataset}")346 dataset = load_dataset(input_dataset, split=split)347 348 # Validate image column349 if image_column not in dataset.column_names:350 raise ValueError(351 f"Column '{image_column}' not found. Available: {dataset.column_names}"352 )353 354 # Fail fast if the output column would collide with an existing input column355 dataset = ensure_output_columns_free(dataset, [output_column], overwrite=overwrite)356 357 # Shuffle if requested358 if shuffle:359 logger.info(f"Shuffling dataset with seed {seed}")360 dataset = dataset.shuffle(seed=seed)361 362 # Limit samples if requested363 if max_samples:364 dataset = dataset.select(range(min(max_samples, len(dataset))))365 logger.info(f"Limited to {len(dataset)} samples")366 367 # Initialize vLLM model368 logger.info("Initializing vLLM with LightOnOCR-2")369 logger.info("This may take a few minutes on first run...")370 llm = LLM(371 model=MODEL,372 trust_remote_code=True,373 max_model_len=max_model_len,374 gpu_memory_utilization=gpu_memory_utilization,375 limit_mm_per_prompt={"image": 1}, # One image per prompt376 enforce_eager=False, # Use torch.compile for better performance377 )378 379 # LightOnOCR-2 recommended sampling parameters380 sampling_params = SamplingParams(381 temperature=temperature,382 top_p=top_p,383 max_tokens=max_tokens,384 )385 386 logger.info(f"Processing {len(dataset)} images in batches of {batch_size}")387 logger.info(f"Output will be written to column: {output_column}")388 if not no_resize:389 logger.info(f"Images will be resized to {target_size}px (longest dimension)")390 391 # Process images in batches392 all_outputs = []393 394 for batch_indices in tqdm(395 partition_all(batch_size, range(len(dataset))),396 total=(len(dataset) + batch_size - 1) // batch_size,397 desc="LightOnOCR-2 processing",398 ):399 batch_indices = list(batch_indices)400 batch_images = [dataset[i][image_column] for i in batch_indices]401 402 try:403 # Create messages for batch404 batch_messages = [405 make_ocr_message(img, resize=not no_resize, target_size=target_size)406 for img in batch_images407 ]408 409 # Process with vLLM410 outputs = llm.chat(batch_messages, sampling_params)411 412 # Extract outputs413 for output in outputs:414 text = output.outputs[0].text.strip()415 all_outputs.append(text)416 417 except Exception as e:418 logger.error(f"Error processing batch: {e}")419 # Add error placeholders for failed batch420 all_outputs.extend(["[OCR ERROR]"] * len(batch_images))421 422 # Calculate processing time423 processing_duration = datetime.now() - start_time424 processing_time_str = f"{processing_duration.total_seconds() / 60:.1f} min"425 426 # Add output column to dataset427 logger.info(f"Adding '{output_column}' column to dataset")428 dataset = dataset.add_column(output_column, all_outputs)429 430 # Handle inference_info tracking (for multi-model comparisons)431 inference_entry = {432 "model_id": MODEL,433 "model_name": "LightOnOCR-2",434 "column_name": output_column,435 "timestamp": datetime.now().isoformat(),436 "temperature": temperature,437 "top_p": top_p,438 "max_tokens": max_tokens,439 "target_size": target_size if not no_resize else "original",440 }441 442 if "inference_info" in dataset.column_names:443 # Append to existing inference info444 logger.info("Updating existing inference_info column")445 446 def update_inference_info(example):447 try:448 existing_info = (449 json.loads(example["inference_info"])450 if example["inference_info"]451 else []452 )453 except (json.JSONDecodeError, TypeError):454 existing_info = []455 456 existing_info.append(inference_entry)457 return {"inference_info": json.dumps(existing_info)}458 459 dataset = dataset.map(update_inference_info)460 else:461 # Create new inference_info column462 logger.info("Creating new inference_info column")463 inference_list = [json.dumps([inference_entry])] * len(dataset)464 dataset = dataset.add_column("inference_info", inference_list)465 466 # Push to hub467 logger.info(f"Pushing to {output_dataset}")468 dataset.push_to_hub(469 output_dataset,470 private=private,471 token=HF_TOKEN,472 **({"config_name": config} if config else {}),473 create_pr=create_pr,474 commit_message=f"Add {MODEL} OCR results ({len(dataset)} samples)"475 + (f" [{config}]" if config else ""),476 )477 478 # Create and push dataset card479 logger.info("Creating dataset card")480 card_content = create_dataset_card(481 source_dataset=input_dataset,482 model=MODEL,483 num_samples=len(dataset),484 processing_time=processing_time_str,485 batch_size=batch_size,486 max_model_len=max_model_len,487 max_tokens=max_tokens,488 gpu_memory_utilization=gpu_memory_utilization,489 temperature=temperature,490 top_p=top_p,491 target_size=target_size,492 image_column=image_column,493 split=split,494 )495 496 card = DatasetCard(card_content)497 card.push_to_hub(output_dataset, token=HF_TOKEN)498 499 logger.info("✅ LightOnOCR-2 processing complete!")500 logger.info(501 f"Dataset available at: https://huggingface.co/datasets/{output_dataset}"502 )503 logger.info(f"Processing time: {processing_time_str}")504 logger.info(505 f"Processing speed: {len(dataset) / processing_duration.total_seconds():.2f} images/sec"506 )507 508 if verbose:509 import importlib.metadata510 511 logger.info("--- Resolved package versions ---")512 for pkg in ["vllm", "transformers", "torch", "datasets", "pyarrow", "pillow"]:513 try:514 logger.info(f" {pkg}=={importlib.metadata.version(pkg)}")515 except importlib.metadata.PackageNotFoundError:516 logger.info(f" {pkg}: not installed")517 logger.info("--- End versions ---")518 519 520if __name__ == "__main__":521 # Show example usage if no arguments522 if len(sys.argv) == 1:523 print("=" * 80)524 print("LightOnOCR-2 Document Processing")525 print("=" * 80)526 print("\nNext-generation 1B OCR model with RLVR training")527 print("\nFeatures:")528 print("- ⚡ Fastest processing: 42.8 pages/sec on H100 (7× faster than v1)")529 print("- 🎯 High accuracy: 83.2 ± 0.9% on OlmOCR-Bench (+7.1% vs v1)")530 print("- 🧠 RLVR trained: No repetition loops or formatting errors")531 print("- 📚 Better training: 2.5× larger dataset with cleaner annotations")532 print("- 🌍 Multilingual with European language optimization")533 print("- 📐 LaTeX formula recognition")534 print("- 📊 Table extraction (markdown format)")535 print("- 💪 Production-ready: Outperforms models 9× larger")536 print("\nExample usage:")537 print("\n1. Basic OCR:")538 print(" uv run lighton-ocr2.py input-dataset output-dataset")539 print("\n2. Custom batch size for performance:")540 print(" uv run lighton-ocr2.py docs results --batch-size 32")541 print("\n3. Test with small sample:")542 print(" uv run lighton-ocr2.py large-dataset test --max-samples 50 --shuffle")543 print("\n4. Original image size (no resize):")544 print(" uv run lighton-ocr2.py docs output --no-resize")545 print("\n5. Running on HF Jobs:")546 print(" (image, hardware and HF_TOKEN come from the script's [tool.hf-jobs] header)")547 print(" hf jobs uv run \\")548 print(549 " https://huggingface.co/datasets/uv-scripts/ocr/raw/main/lighton-ocr2.py \\"550 )551 print(" input-dataset output-dataset --batch-size 32")552 print("\n" + "=" * 80)553 print("\nKey Improvements over v1:")554 print(" - 7.5× faster processing speed")555 print(" - 7.1% higher accuracy on benchmarks")556 print(" - Eliminates repetition loops and formatting errors")557 print(" - Simpler: single model (no vocabulary variants)")558 print("\nFor full help, run: uv run lighton-ocr2.py --help")559 sys.exit(0)560 561 parser = argparse.ArgumentParser(562 description="Document OCR using LightOnOCR-2 (next-gen 1B model with RLVR)",563 formatter_class=argparse.RawDescriptionHelpFormatter,564 epilog="""565Key Improvements over v1:566 - 7.5× faster: 42.8 vs 5.71 pages/sec on H100567 - +7.1% accuracy: 83.2% vs 76.1% on benchmarks568 - Better quality: RLVR training eliminates repetition loops569 - Cleaner output: No formatting glitches570 - Simpler: Single model (no vocabulary variants)571 572Examples:573 # Basic text OCR574 uv run lighton-ocr2.py my-docs analyzed-docs575 576 # Test with random sampling577 uv run lighton-ocr2.py large-dataset test --max-samples 50 --shuffle578 579 # Custom batch size for GPU optimization580 uv run lighton-ocr2.py dataset output --batch-size 32 --gpu-memory-utilization 0.9581 """,582 )583 584 parser.add_argument("input_dataset", help="Input dataset ID from Hugging Face Hub")585 parser.add_argument("output_dataset", help="Output dataset ID for Hugging Face Hub")586 parser.add_argument(587 "--image-column",588 default="image",589 help="Column containing images (default: image)",590 )591 parser.add_argument(592 "--batch-size",593 type=int,594 default=16,595 help="Batch size for processing (default: 16)",596 )597 parser.add_argument(598 "--max-model-len",599 type=int,600 default=16384,601 help=(602 "Maximum model context length (default: 16384). A full page resized to "603 "1540px is ~6k image tokens; with --max-tokens 4096 output that overflows "604 "the old 8192 default at admission and vLLM rejects the request."605 ),606 )607 parser.add_argument(608 "--max-tokens",609 type=int,610 default=4096,611 help="Maximum tokens to generate (default: 4096, recommended for arXiv papers)",612 )613 parser.add_argument(614 "--temperature",615 type=float,616 default=0.2,617 help="Sampling temperature (default: 0.2)",618 )619 parser.add_argument(620 "--top-p",621 type=float,622 default=0.9,623 help="Top-p sampling parameter (default: 0.9)",624 )625 parser.add_argument(626 "--gpu-memory-utilization",627 type=float,628 default=0.8,629 help="GPU memory utilization (default: 0.8)",630 )631 parser.add_argument(632 "--target-size",633 type=int,634 default=1540,635 help="Target size for longest image dimension in pixels (default: 1540, matching training)",636 )637 parser.add_argument(638 "--no-resize",639 action="store_true",640 help="Don't resize images (use original size)",641 )642 parser.add_argument("--hf-token", help="Hugging Face API token")643 parser.add_argument(644 "--split", default="train", help="Dataset split to use (default: train)"645 )646 parser.add_argument(647 "--max-samples",648 type=int,649 help="Maximum number of samples to process (for testing)",650 )651 parser.add_argument(652 "--private", action="store_true", help="Make output dataset private"653 )654 parser.add_argument(655 "--config",656 help="Config/subset name when pushing to Hub (for benchmarking multiple models in one repo)",657 )658 parser.add_argument(659 "--create-pr",660 action="store_true",661 help="Create a pull request instead of pushing directly (for parallel benchmarking)",662 )663 parser.add_argument(664 "--shuffle", action="store_true", help="Shuffle dataset before processing"665 )666 parser.add_argument(667 "--seed",668 type=int,669 default=42,670 help="Random seed for shuffling (default: 42)",671 )672 parser.add_argument(673 "--output-column",674 default="markdown",675 help="Column name for output text (default: markdown)",676 )677 parser.add_argument(678 "--overwrite",679 action="store_true",680 help="Replace the output column if it already exists in the input dataset "681 "(default: error out to avoid clobbering an existing column).",682 )683 parser.add_argument(684 "--verbose",685 action="store_true",686 help="Log resolved package versions after processing (useful for pinning deps)",687 )688 689 args = parser.parse_args()690 691 main(692 input_dataset=args.input_dataset,693 output_dataset=args.output_dataset,694 image_column=args.image_column,695 batch_size=args.batch_size,696 max_model_len=args.max_model_len,697 max_tokens=args.max_tokens,698 temperature=args.temperature,699 top_p=args.top_p,700 gpu_memory_utilization=args.gpu_memory_utilization,701 target_size=args.target_size,702 no_resize=args.no_resize,703 hf_token=args.hf_token,704 split=args.split,705 max_samples=args.max_samples,706 private=args.private,707 shuffle=args.shuffle,708 seed=args.seed,709 output_column=args.output_column,710 overwrite=args.overwrite,711 config=args.config,712 create_pr=args.create_pr,713 verbose=args.verbose,714 )715 