uv-scripts/ocr
OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First⦠See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.
1636.5k
1# /// script2# requires-python = ">=3.11"3# dependencies = [4# "datasets>=4.0.0",5# "huggingface-hub",6# "pillow",7# "torch",8# "torchvision",9# "transformers==4.46.3",10# "tokenizers==0.20.3",11# "tqdm",12# "addict",13# "matplotlib",14# "einops",15# "easydict",16# ]17#18# ///19 20"""21UNSUPPORTED: Broken (2026-09-23): every output row is the string 'None' (model.infer returns None). Use deepseek-ocr-vllm.py. See models.json (`support`).22 23Convert document images to markdown using DeepSeek-OCR with Transformers.24 25This script processes images through the DeepSeek-OCR model to extract26text and structure as markdown, using the official Transformers API.27 28Features:29- Multiple resolution modes (Tiny/Small/Base/Large/Gundam)30- LaTeX equation recognition31- Table extraction and formatting32- Document structure preservation33- Image grounding and descriptions34- Multilingual support35 36Note: This script processes images sequentially (no batching) using the37official transformers API. It's slower than vLLM-based scripts but uses38the well-supported official implementation.39"""40 41import argparse42import json43import logging44import os45import shutil46import sys47from datetime import datetime48from pathlib import Path49from typing import Optional50 51import torch52from datasets import load_dataset53from huggingface_hub import DatasetCard, login54from PIL import Image55from tqdm.auto import tqdm56from transformers import AutoModel, AutoTokenizer57 58logging.basicConfig(level=logging.INFO)59logger = logging.getLogger(__name__)60 61# Resolution mode presets62RESOLUTION_MODES = {63 "tiny": {"base_size": 512, "image_size": 512, "crop_mode": False},64 "small": {"base_size": 640, "image_size": 640, "crop_mode": False},65 "base": {"base_size": 1024, "image_size": 1024, "crop_mode": False},66 "large": {"base_size": 1280, "image_size": 1280, "crop_mode": False},67 "gundam": {"base_size": 1024, "image_size": 640, "crop_mode": True}, # Dynamic resolution68}69 70 71def check_cuda_availability():72 """Check if CUDA is available and exit if not."""73 if not torch.cuda.is_available():74 logger.error("CUDA is not available. This script requires a GPU.")75 logger.error("Please run on a machine with a CUDA-capable GPU.")76 sys.exit(1)77 else:78 logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name(0)}")79 80 81def ensure_output_columns_free(dataset, columns, overwrite=False):82 """Fail fast if an output column would collide with an existing input column.83 84 Adding a column that already exists silently overwrites it (e.g. a ground-truth85 `text`/`markdown` column) or crashes on push with a duplicate-column error only86 *after* inference has run. Catch it up front. With overwrite=True, drop the clashing87 column(s) here instead (logged) so the later add_column is clean.88 """89 clash = [c for c in columns if c in dataset.column_names]90 if not clash:91 return dataset92 if overwrite:93 logger.warning(f"--overwrite: replacing existing column(s) {clash}")94 return dataset.remove_columns(clash)95 logger.error(96 f"Output column(s) {clash} already exist in the input dataset "97 f"(columns: {dataset.column_names})."98 )99 logger.error("Choose a different --output-column, or pass --overwrite to replace them.")100 sys.exit(1)101 102 103def create_dataset_card(104 source_dataset: str,105 model: str,106 num_samples: int,107 processing_time: str,108 resolution_mode: str,109 base_size: int,110 image_size: int,111 crop_mode: bool,112 image_column: str = "image",113 split: str = "train",114) -> str:115 """Create a dataset card documenting the OCR process."""116 model_name = model.split("/")[-1]117 118 return f"""---119tags:120- ocr121- document-processing122- deepseek123- deepseek-ocr124- markdown125- uv-script126- generated127---128 129# Document OCR using {model_name}130 131This dataset contains markdown-formatted OCR results from images in [{source_dataset}](https://huggingface.co/datasets/{source_dataset}) using DeepSeek-OCR.132 133## Processing Details134 135- **Source Dataset**: [{source_dataset}](https://huggingface.co/datasets/{source_dataset})136- **Model**: [{model}](https://huggingface.co/{model})137- **Number of Samples**: {num_samples:,}138- **Processing Time**: {processing_time}139- **Processing Date**: {datetime.now().strftime("%Y-%m-%d %H:%M UTC")}140 141### Configuration142 143- **Image Column**: `{image_column}`144- **Output Column**: `markdown`145- **Dataset Split**: `{split}`146- **Resolution Mode**: {resolution_mode}147- **Base Size**: {base_size}148- **Image Size**: {image_size}149- **Crop Mode**: {crop_mode}150 151## Model Information152 153DeepSeek-OCR is a state-of-the-art document OCR model that excels at:154- π **LaTeX equations** - Mathematical formulas preserved in LaTeX format155- π **Tables** - Extracted and formatted as HTML/markdown156- π **Document structure** - Headers, lists, and formatting maintained157- πΌοΈ **Image grounding** - Spatial layout and bounding box information158- π **Complex layouts** - Multi-column and hierarchical structures159- π **Multilingual** - Supports multiple languages160 161### Resolution Modes162 163- **Tiny** (512Γ512): Fast processing, 64 vision tokens164- **Small** (640Γ640): Balanced speed/quality, 100 vision tokens165- **Base** (1024Γ1024): High quality, 256 vision tokens166- **Large** (1280Γ1280): Maximum quality, 400 vision tokens167- **Gundam** (dynamic): Adaptive multi-tile processing for large documents168 169## Dataset Structure170 171The dataset contains all original columns plus:172- `markdown`: The extracted text in markdown format with preserved structure173- `inference_info`: JSON list tracking all OCR models applied to this dataset174 175## Usage176 177```python178from datasets import load_dataset179import json180 181# Load the dataset182dataset = load_dataset("{{{{output_dataset_id}}}}", split="{split}")183 184# Access the markdown text185for example in dataset:186 print(example["markdown"])187 break188 189# View all OCR models applied to this dataset190inference_info = json.loads(dataset[0]["inference_info"])191for info in inference_info:192 print(f"Column: {{{{info['column_name']}}}} - Model: {{{{info['model_id']}}}}")193```194 195## Reproduction196 197This dataset was generated using the [uv-scripts/ocr](https://huggingface.co/datasets/uv-scripts/ocr) DeepSeek OCR script:198 199```bash200uv run https://huggingface.co/datasets/uv-scripts/ocr/raw/main/deepseek-ocr.py \\201 {source_dataset} \\202 <output-dataset> \\203 --resolution-mode {resolution_mode} \\204 --image-column {image_column}205```206 207## Performance208 209- **Processing Speed**: ~{num_samples / (float(processing_time.split()[0]) * 60):.1f} images/second210- **Processing Method**: Sequential (Transformers API, no batching)211 212Note: This uses the official Transformers implementation. For faster batch processing,213consider using the vLLM version once DeepSeek-OCR is officially supported by vLLM.214 215Generated with π€ [UV Scripts](https://huggingface.co/uv-scripts)216"""217 218 219def process_single_image(220 model,221 tokenizer,222 image: Image.Image,223 prompt: str,224 base_size: int,225 image_size: int,226 crop_mode: bool,227 temp_image_path: str,228 temp_output_dir: str,229) -> str:230 """Process a single image through DeepSeek-OCR."""231 # Convert to RGB if needed232 if image.mode != "RGB":233 image = image.convert("RGB")234 235 # Save to temp file (model.infer expects a file path)236 image.save(temp_image_path, format="PNG")237 238 # Run inference239 result = model.infer(240 tokenizer,241 prompt=prompt,242 image_file=temp_image_path,243 output_path=temp_output_dir, # Need real directory path244 base_size=base_size,245 image_size=image_size,246 crop_mode=crop_mode,247 save_results=False,248 test_compress=False,249 )250 251 return result if isinstance(result, str) else str(result)252 253 254def main(255 input_dataset: str,256 output_dataset: str,257 image_column: str = "image",258 model: str = "deepseek-ai/DeepSeek-OCR",259 resolution_mode: str = "gundam",260 base_size: Optional[int] = None,261 image_size: Optional[int] = None,262 crop_mode: Optional[bool] = None,263 prompt: str = "<image>\n<|grounding|>Convert the document to markdown.",264 hf_token: str = None,265 split: str = "train",266 max_samples: int = None,267 private: bool = False,268 shuffle: bool = False,269 seed: int = 42,270 output_column: str = "markdown",271 overwrite: bool = False,272):273 """Process images from HF dataset through DeepSeek-OCR model."""274 275 # Check CUDA availability first276 check_cuda_availability()277 278 # Track processing start time279 start_time = datetime.now()280 281 282 283 # Login to HF if token provided284 HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")285 if HF_TOKEN:286 login(token=HF_TOKEN)287 288 # Determine resolution settings289 if resolution_mode in RESOLUTION_MODES:290 mode_config = RESOLUTION_MODES[resolution_mode]291 final_base_size = base_size if base_size is not None else mode_config["base_size"]292 final_image_size = image_size if image_size is not None else mode_config["image_size"]293 final_crop_mode = crop_mode if crop_mode is not None else mode_config["crop_mode"]294 logger.info(f"Using resolution mode: {resolution_mode}")295 else:296 # Custom mode - require all parameters297 if base_size is None or image_size is None or crop_mode is None:298 raise ValueError(299 f"Invalid resolution mode '{resolution_mode}'. "300 f"Use one of {list(RESOLUTION_MODES.keys())} or specify "301 f"--base-size, --image-size, and --crop-mode manually."302 )303 final_base_size = base_size304 final_image_size = image_size305 final_crop_mode = crop_mode306 resolution_mode = "custom"307 308 logger.info(309 f"Resolution: base_size={final_base_size}, "310 f"image_size={final_image_size}, crop_mode={final_crop_mode}"311 )312 313 # Load dataset314 logger.info(f"Loading dataset: {input_dataset}")315 dataset = load_dataset(input_dataset, split=split)316 317 # Validate image column318 if image_column not in dataset.column_names:319 raise ValueError(320 f"Column '{image_column}' not found. Available: {dataset.column_names}"321 )322 323 # Fail fast if the output column would collide with an existing input column324 dataset = ensure_output_columns_free(dataset, [output_column], overwrite=overwrite)325 326 # Shuffle if requested327 if shuffle:328 logger.info(f"Shuffling dataset with seed {seed}")329 dataset = dataset.shuffle(seed=seed)330 331 # Limit samples if requested332 if max_samples:333 dataset = dataset.select(range(min(max_samples, len(dataset))))334 logger.info(f"Limited to {len(dataset)} samples")335 336 # Initialize model337 logger.info(f"Loading model: {model}")338 tokenizer = AutoTokenizer.from_pretrained(model, trust_remote_code=True)339 340 try:341 model_obj = AutoModel.from_pretrained(342 model,343 _attn_implementation="flash_attention_2",344 trust_remote_code=True,345 use_safetensors=True,346 )347 except Exception as e:348 logger.warning(f"Failed to load with flash_attention_2: {e}")349 logger.info("Falling back to standard attention...")350 model_obj = AutoModel.from_pretrained(351 model,352 trust_remote_code=True,353 use_safetensors=True,354 )355 356 model_obj = model_obj.eval().cuda().to(torch.bfloat16)357 logger.info("Model loaded successfully")358 359 # Process images sequentially360 all_markdown = []361 362 logger.info(f"Processing {len(dataset)} images (sequential, no batching)")363 logger.info("Note: This may be slower than vLLM-based scripts")364 365 # Create temp directories for image files and output (simple local dirs)366 temp_dir = Path("temp_images")367 temp_dir.mkdir(exist_ok=True)368 temp_image_path = str(temp_dir / "temp_image.png")369 370 temp_output_dir = Path("temp_output")371 temp_output_dir.mkdir(exist_ok=True)372 373 try:374 for i in tqdm(range(len(dataset)), desc="OCR processing"):375 try:376 image = dataset[i][image_column]377 378 # Handle different image formats379 if isinstance(image, dict) and "bytes" in image:380 from io import BytesIO381 image = Image.open(BytesIO(image["bytes"]))382 elif isinstance(image, str):383 image = Image.open(image)384 elif not isinstance(image, Image.Image):385 raise ValueError(f"Unsupported image type: {type(image)}")386 387 # Process image388 result = process_single_image(389 model_obj,390 tokenizer,391 image,392 prompt,393 final_base_size,394 final_image_size,395 final_crop_mode,396 temp_image_path,397 str(temp_output_dir),398 )399 400 all_markdown.append(result)401 402 except Exception as e:403 logger.error(f"Error processing image {i}: {e}")404 all_markdown.append("[OCR FAILED]")405 406 finally:407 # Clean up temp directories408 try:409 shutil.rmtree(temp_dir)410 shutil.rmtree(temp_output_dir)411 except Exception:412 pass413 414 # Add output column to dataset415 logger.info(f"Adding '{output_column}' column to dataset")416 dataset = dataset.add_column(output_column, all_markdown)417 418 # Handle inference_info tracking419 logger.info("Updating inference_info...")420 421 # Check for existing inference_info422 if "inference_info" in dataset.column_names:423 try:424 existing_info = json.loads(dataset[0]["inference_info"])425 if not isinstance(existing_info, list):426 existing_info = [existing_info]427 except (json.JSONDecodeError, TypeError):428 existing_info = []429 dataset = dataset.remove_columns(["inference_info"])430 else:431 existing_info = []432 433 # Add new inference info434 new_info = {435 "column_name": output_column,436 "model_id": model,437 "processing_date": datetime.now().isoformat(),438 "resolution_mode": resolution_mode,439 "base_size": final_base_size,440 "image_size": final_image_size,441 "crop_mode": final_crop_mode,442 "prompt": prompt,443 "script": "deepseek-ocr.py",444 "script_version": "1.0.0",445 "script_url": "https://huggingface.co/datasets/uv-scripts/ocr/raw/main/deepseek-ocr.py",446 "implementation": "transformers (sequential)",447 }448 existing_info.append(new_info)449 450 # Add updated inference_info column451 info_json = json.dumps(existing_info, ensure_ascii=False)452 dataset = dataset.add_column("inference_info", [info_json] * len(dataset))453 454 # Push to hub455 logger.info(f"Pushing to {output_dataset}")456 dataset.push_to_hub(output_dataset, private=private, token=HF_TOKEN)457 458 # Calculate processing time459 end_time = datetime.now()460 processing_duration = end_time - start_time461 processing_time = f"{processing_duration.total_seconds() / 60:.1f} minutes"462 463 # Create and push dataset card464 logger.info("Creating dataset card...")465 card_content = create_dataset_card(466 source_dataset=input_dataset,467 model=model,468 num_samples=len(dataset),469 processing_time=processing_time,470 resolution_mode=resolution_mode,471 base_size=final_base_size,472 image_size=final_image_size,473 crop_mode=final_crop_mode,474 image_column=image_column,475 split=split,476 )477 478 card = DatasetCard(card_content)479 card.push_to_hub(output_dataset, token=HF_TOKEN)480 logger.info("β
Dataset card created and pushed!")481 482 logger.info("β
OCR conversion complete!")483 logger.info(484 f"Dataset available at: https://huggingface.co/datasets/{output_dataset}"485 )486 487 488if __name__ == "__main__":489 # Show example usage if no arguments490 if len(sys.argv) == 1:491 print("=" * 80)492 print("DeepSeek-OCR to Markdown Converter (Transformers)")493 print("=" * 80)494 print("\nThis script converts document images to markdown using")495 print("DeepSeek-OCR with the official Transformers API.")496 print("\nFeatures:")497 print("- Multiple resolution modes (Tiny/Small/Base/Large/Gundam)")498 print("- LaTeX equation recognition")499 print("- Table extraction and formatting")500 print("- Document structure preservation")501 print("- Image grounding and spatial layout")502 print("- Multilingual support")503 print("\nNote: Sequential processing (no batching). Slower than vLLM scripts.")504 print("\nExample usage:")505 print("\n1. Basic OCR conversion (Gundam mode - dynamic resolution):")506 print(" uv run deepseek-ocr.py document-images markdown-docs")507 print("\n2. High quality mode (Large - 1280Γ1280):")508 print(" uv run deepseek-ocr.py scanned-pdfs extracted-text --resolution-mode large")509 print("\n3. Fast processing (Tiny - 512Γ512):")510 print(" uv run deepseek-ocr.py quick-test output --resolution-mode tiny")511 print("\n4. Process a subset for testing:")512 print(" uv run deepseek-ocr.py large-dataset test-output --max-samples 10")513 print("\n5. Custom resolution:")514 print(" uv run deepseek-ocr.py dataset output \\")515 print(" --base-size 1024 --image-size 640 --crop-mode")516 print("\n6. Running on HF Jobs:")517 print(" hf jobs uv run --flavor l4x1 \\")518 print(' --secrets HF_TOKEN \\')519 print(" https://huggingface.co/datasets/uv-scripts/ocr/raw/main/deepseek-ocr.py \\")520 print(" your-document-dataset \\")521 print(" your-markdown-output")522 print("\n" + "=" * 80)523 print("\nFor full help, run: uv run deepseek-ocr.py --help")524 sys.exit(0)525 526 parser = argparse.ArgumentParser(527 description="OCR images to markdown using DeepSeek-OCR (Transformers)",528 formatter_class=argparse.RawDescriptionHelpFormatter,529 epilog="""530Resolution Modes:531 tiny 512Γ512 pixels, fast processing (64 vision tokens)532 small 640Γ640 pixels, balanced (100 vision tokens)533 base 1024Γ1024 pixels, high quality (256 vision tokens)534 large 1280Γ1280 pixels, maximum quality (400 vision tokens)535 gundam Dynamic multi-tile processing (adaptive)536 537Examples:538 # Basic usage with default Gundam mode539 uv run deepseek-ocr.py my-images-dataset ocr-results540 541 # High quality processing542 uv run deepseek-ocr.py documents extracted-text --resolution-mode large543 544 # Fast processing for testing545 uv run deepseek-ocr.py dataset output --resolution-mode tiny --max-samples 100546 547 # Custom resolution settings548 uv run deepseek-ocr.py dataset output --base-size 1024 --image-size 640 --crop-mode549 """,550 )551 552 parser.add_argument("input_dataset", help="Input dataset ID from Hugging Face Hub")553 parser.add_argument("output_dataset", help="Output dataset ID for Hugging Face Hub")554 parser.add_argument(555 "--image-column",556 default="image",557 help="Column containing images (default: image)",558 )559 parser.add_argument(560 "--model",561 default="deepseek-ai/DeepSeek-OCR",562 help="Model to use (default: deepseek-ai/DeepSeek-OCR)",563 )564 parser.add_argument(565 "--resolution-mode",566 default="gundam",567 choices=list(RESOLUTION_MODES.keys()) + ["custom"],568 help="Resolution mode preset (default: gundam)",569 )570 parser.add_argument(571 "--base-size",572 type=int,573 help="Base resolution size (overrides resolution-mode)",574 )575 parser.add_argument(576 "--image-size",577 type=int,578 help="Image tile size (overrides resolution-mode)",579 )580 parser.add_argument(581 "--crop-mode",582 action="store_true",583 help="Enable dynamic multi-tile cropping (overrides resolution-mode)",584 )585 parser.add_argument(586 "--prompt",587 default="<image>\n<|grounding|>Convert the document to markdown.",588 help="Prompt for OCR (default: grounding markdown conversion)",589 )590 parser.add_argument("--hf-token", help="Hugging Face API token")591 parser.add_argument(592 "--split", default="train", help="Dataset split to use (default: train)"593 )594 parser.add_argument(595 "--max-samples",596 type=int,597 help="Maximum number of samples to process (for testing)",598 )599 parser.add_argument(600 "--private", action="store_true", help="Make output dataset private"601 )602 parser.add_argument(603 "--shuffle",604 action="store_true",605 help="Shuffle the dataset before processing (useful for random sampling)",606 )607 parser.add_argument(608 "--seed",609 type=int,610 default=42,611 help="Random seed for shuffling (default: 42)",612 )613 parser.add_argument(614 "--output-column",615 default="markdown",616 help="Column name for the OCR output text (default: markdown)",617 )618 parser.add_argument(619 "--overwrite",620 action="store_true",621 help="Replace the output column if it already exists in the input dataset "622 "(default: error out to avoid clobbering an existing column).",623 )624 625 args = parser.parse_args()626 627 main(628 input_dataset=args.input_dataset,629 output_dataset=args.output_dataset,630 image_column=args.image_column,631 model=args.model,632 resolution_mode=args.resolution_mode,633 base_size=args.base_size,634 image_size=args.image_size,635 crop_mode=args.crop_mode if args.crop_mode else None,636 prompt=args.prompt,637 hf_token=args.hf_token,638 split=args.split,639 max_samples=args.max_samples,640 private=args.private,641 shuffle=args.shuffle,642 seed=args.seed,643 output_column=args.output_column,644 overwrite=args.overwrite,645 )646 