Team Ai
Datasetpublic

uv-scripts/ocr

OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.

sourceHugging Faceupdated 10d agoView on Hugging Face
163likes6.5kdownloads
deepseek-ocr.py646 linesDownload Raw Back to root
1# /// script2# requires-python = ">=3.11"3# dependencies = [4#     "datasets>=4.0.0",5#     "huggingface-hub",6#     "pillow",7#     "torch",8#     "torchvision",9#     "transformers==4.46.3",10#     "tokenizers==0.20.3",11#     "tqdm",12#     "addict",13#     "matplotlib",14#     "einops",15#     "easydict",16# ]17#18# ///19 20"""21UNSUPPORTED: Broken (2026-09-23): every output row is the string 'None' (model.infer returns None). Use deepseek-ocr-vllm.py. See models.json (`support`).22 23Convert document images to markdown using DeepSeek-OCR with Transformers.24 25This script processes images through the DeepSeek-OCR model to extract26text and structure as markdown, using the official Transformers API.27 28Features:29- Multiple resolution modes (Tiny/Small/Base/Large/Gundam)30- LaTeX equation recognition31- Table extraction and formatting32- Document structure preservation33- Image grounding and descriptions34- Multilingual support35 36Note: This script processes images sequentially (no batching) using the37official transformers API. It's slower than vLLM-based scripts but uses38the well-supported official implementation.39"""40 41import argparse42import json43import logging44import os45import shutil46import sys47from datetime import datetime48from pathlib import Path49from typing import Optional50 51import torch52from datasets import load_dataset53from huggingface_hub import DatasetCard, login54from PIL import Image55from tqdm.auto import tqdm56from transformers import AutoModel, AutoTokenizer57 58logging.basicConfig(level=logging.INFO)59logger = logging.getLogger(__name__)60 61# Resolution mode presets62RESOLUTION_MODES = {63    "tiny": {"base_size": 512, "image_size": 512, "crop_mode": False},64    "small": {"base_size": 640, "image_size": 640, "crop_mode": False},65    "base": {"base_size": 1024, "image_size": 1024, "crop_mode": False},66    "large": {"base_size": 1280, "image_size": 1280, "crop_mode": False},67    "gundam": {"base_size": 1024, "image_size": 640, "crop_mode": True},  # Dynamic resolution68}69 70 71def check_cuda_availability():72    """Check if CUDA is available and exit if not."""73    if not torch.cuda.is_available():74        logger.error("CUDA is not available. This script requires a GPU.")75        logger.error("Please run on a machine with a CUDA-capable GPU.")76        sys.exit(1)77    else:78        logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name(0)}")79 80 81def ensure_output_columns_free(dataset, columns, overwrite=False):82    """Fail fast if an output column would collide with an existing input column.83 84    Adding a column that already exists silently overwrites it (e.g. a ground-truth85    `text`/`markdown` column) or crashes on push with a duplicate-column error only86    *after* inference has run. Catch it up front. With overwrite=True, drop the clashing87    column(s) here instead (logged) so the later add_column is clean.88    """89    clash = [c for c in columns if c in dataset.column_names]90    if not clash:91        return dataset92    if overwrite:93        logger.warning(f"--overwrite: replacing existing column(s) {clash}")94        return dataset.remove_columns(clash)95    logger.error(96        f"Output column(s) {clash} already exist in the input dataset "97        f"(columns: {dataset.column_names})."98    )99    logger.error("Choose a different --output-column, or pass --overwrite to replace them.")100    sys.exit(1)101 102 103def create_dataset_card(104    source_dataset: str,105    model: str,106    num_samples: int,107    processing_time: str,108    resolution_mode: str,109    base_size: int,110    image_size: int,111    crop_mode: bool,112    image_column: str = "image",113    split: str = "train",114) -> str:115    """Create a dataset card documenting the OCR process."""116    model_name = model.split("/")[-1]117 118    return f"""---119tags:120- ocr121- document-processing122- deepseek123- deepseek-ocr124- markdown125- uv-script126- generated127---128 129# Document OCR using {model_name}130 131This dataset contains markdown-formatted OCR results from images in [{source_dataset}](https://huggingface.co/datasets/{source_dataset}) using DeepSeek-OCR.132 133## Processing Details134 135- **Source Dataset**: [{source_dataset}](https://huggingface.co/datasets/{source_dataset})136- **Model**: [{model}](https://huggingface.co/{model})137- **Number of Samples**: {num_samples:,}138- **Processing Time**: {processing_time}139- **Processing Date**: {datetime.now().strftime("%Y-%m-%d %H:%M UTC")}140 141### Configuration142 143- **Image Column**: `{image_column}`144- **Output Column**: `markdown`145- **Dataset Split**: `{split}`146- **Resolution Mode**: {resolution_mode}147- **Base Size**: {base_size}148- **Image Size**: {image_size}149- **Crop Mode**: {crop_mode}150 151## Model Information152 153DeepSeek-OCR is a state-of-the-art document OCR model that excels at:154- πŸ“ **LaTeX equations** - Mathematical formulas preserved in LaTeX format155- πŸ“Š **Tables** - Extracted and formatted as HTML/markdown156- πŸ“ **Document structure** - Headers, lists, and formatting maintained157- πŸ–ΌοΈ **Image grounding** - Spatial layout and bounding box information158- πŸ” **Complex layouts** - Multi-column and hierarchical structures159- 🌍 **Multilingual** - Supports multiple languages160 161### Resolution Modes162 163- **Tiny** (512Γ—512): Fast processing, 64 vision tokens164- **Small** (640Γ—640): Balanced speed/quality, 100 vision tokens165- **Base** (1024Γ—1024): High quality, 256 vision tokens166- **Large** (1280Γ—1280): Maximum quality, 400 vision tokens167- **Gundam** (dynamic): Adaptive multi-tile processing for large documents168 169## Dataset Structure170 171The dataset contains all original columns plus:172- `markdown`: The extracted text in markdown format with preserved structure173- `inference_info`: JSON list tracking all OCR models applied to this dataset174 175## Usage176 177```python178from datasets import load_dataset179import json180 181# Load the dataset182dataset = load_dataset("{{{{output_dataset_id}}}}", split="{split}")183 184# Access the markdown text185for example in dataset:186    print(example["markdown"])187    break188 189# View all OCR models applied to this dataset190inference_info = json.loads(dataset[0]["inference_info"])191for info in inference_info:192    print(f"Column: {{{{info['column_name']}}}} - Model: {{{{info['model_id']}}}}")193```194 195## Reproduction196 197This dataset was generated using the [uv-scripts/ocr](https://huggingface.co/datasets/uv-scripts/ocr) DeepSeek OCR script:198 199```bash200uv run https://huggingface.co/datasets/uv-scripts/ocr/raw/main/deepseek-ocr.py \\201    {source_dataset} \\202    <output-dataset> \\203    --resolution-mode {resolution_mode} \\204    --image-column {image_column}205```206 207## Performance208 209- **Processing Speed**: ~{num_samples / (float(processing_time.split()[0]) * 60):.1f} images/second210- **Processing Method**: Sequential (Transformers API, no batching)211 212Note: This uses the official Transformers implementation. For faster batch processing,213consider using the vLLM version once DeepSeek-OCR is officially supported by vLLM.214 215Generated with πŸ€– [UV Scripts](https://huggingface.co/uv-scripts)216"""217 218 219def process_single_image(220    model,221    tokenizer,222    image: Image.Image,223    prompt: str,224    base_size: int,225    image_size: int,226    crop_mode: bool,227    temp_image_path: str,228    temp_output_dir: str,229) -> str:230    """Process a single image through DeepSeek-OCR."""231    # Convert to RGB if needed232    if image.mode != "RGB":233        image = image.convert("RGB")234 235    # Save to temp file (model.infer expects a file path)236    image.save(temp_image_path, format="PNG")237 238    # Run inference239    result = model.infer(240        tokenizer,241        prompt=prompt,242        image_file=temp_image_path,243        output_path=temp_output_dir,  # Need real directory path244        base_size=base_size,245        image_size=image_size,246        crop_mode=crop_mode,247        save_results=False,248        test_compress=False,249    )250 251    return result if isinstance(result, str) else str(result)252 253 254def main(255    input_dataset: str,256    output_dataset: str,257    image_column: str = "image",258    model: str = "deepseek-ai/DeepSeek-OCR",259    resolution_mode: str = "gundam",260    base_size: Optional[int] = None,261    image_size: Optional[int] = None,262    crop_mode: Optional[bool] = None,263    prompt: str = "<image>\n<|grounding|>Convert the document to markdown.",264    hf_token: str = None,265    split: str = "train",266    max_samples: int = None,267    private: bool = False,268    shuffle: bool = False,269    seed: int = 42,270    output_column: str = "markdown",271    overwrite: bool = False,272):273    """Process images from HF dataset through DeepSeek-OCR model."""274 275    # Check CUDA availability first276    check_cuda_availability()277 278    # Track processing start time279    start_time = datetime.now()280 281 282 283    # Login to HF if token provided284    HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")285    if HF_TOKEN:286        login(token=HF_TOKEN)287 288    # Determine resolution settings289    if resolution_mode in RESOLUTION_MODES:290        mode_config = RESOLUTION_MODES[resolution_mode]291        final_base_size = base_size if base_size is not None else mode_config["base_size"]292        final_image_size = image_size if image_size is not None else mode_config["image_size"]293        final_crop_mode = crop_mode if crop_mode is not None else mode_config["crop_mode"]294        logger.info(f"Using resolution mode: {resolution_mode}")295    else:296        # Custom mode - require all parameters297        if base_size is None or image_size is None or crop_mode is None:298            raise ValueError(299                f"Invalid resolution mode '{resolution_mode}'. "300                f"Use one of {list(RESOLUTION_MODES.keys())} or specify "301                f"--base-size, --image-size, and --crop-mode manually."302            )303        final_base_size = base_size304        final_image_size = image_size305        final_crop_mode = crop_mode306        resolution_mode = "custom"307 308    logger.info(309        f"Resolution: base_size={final_base_size}, "310        f"image_size={final_image_size}, crop_mode={final_crop_mode}"311    )312 313    # Load dataset314    logger.info(f"Loading dataset: {input_dataset}")315    dataset = load_dataset(input_dataset, split=split)316 317    # Validate image column318    if image_column not in dataset.column_names:319        raise ValueError(320            f"Column '{image_column}' not found. Available: {dataset.column_names}"321        )322 323    # Fail fast if the output column would collide with an existing input column324    dataset = ensure_output_columns_free(dataset, [output_column], overwrite=overwrite)325 326    # Shuffle if requested327    if shuffle:328        logger.info(f"Shuffling dataset with seed {seed}")329        dataset = dataset.shuffle(seed=seed)330 331    # Limit samples if requested332    if max_samples:333        dataset = dataset.select(range(min(max_samples, len(dataset))))334        logger.info(f"Limited to {len(dataset)} samples")335 336    # Initialize model337    logger.info(f"Loading model: {model}")338    tokenizer = AutoTokenizer.from_pretrained(model, trust_remote_code=True)339 340    try:341        model_obj = AutoModel.from_pretrained(342            model,343            _attn_implementation="flash_attention_2",344            trust_remote_code=True,345            use_safetensors=True,346        )347    except Exception as e:348        logger.warning(f"Failed to load with flash_attention_2: {e}")349        logger.info("Falling back to standard attention...")350        model_obj = AutoModel.from_pretrained(351            model,352            trust_remote_code=True,353            use_safetensors=True,354        )355 356    model_obj = model_obj.eval().cuda().to(torch.bfloat16)357    logger.info("Model loaded successfully")358 359    # Process images sequentially360    all_markdown = []361 362    logger.info(f"Processing {len(dataset)} images (sequential, no batching)")363    logger.info("Note: This may be slower than vLLM-based scripts")364 365    # Create temp directories for image files and output (simple local dirs)366    temp_dir = Path("temp_images")367    temp_dir.mkdir(exist_ok=True)368    temp_image_path = str(temp_dir / "temp_image.png")369 370    temp_output_dir = Path("temp_output")371    temp_output_dir.mkdir(exist_ok=True)372 373    try:374        for i in tqdm(range(len(dataset)), desc="OCR processing"):375            try:376                image = dataset[i][image_column]377 378                # Handle different image formats379                if isinstance(image, dict) and "bytes" in image:380                    from io import BytesIO381                    image = Image.open(BytesIO(image["bytes"]))382                elif isinstance(image, str):383                    image = Image.open(image)384                elif not isinstance(image, Image.Image):385                    raise ValueError(f"Unsupported image type: {type(image)}")386 387                # Process image388                result = process_single_image(389                    model_obj,390                    tokenizer,391                    image,392                    prompt,393                    final_base_size,394                    final_image_size,395                    final_crop_mode,396                    temp_image_path,397                    str(temp_output_dir),398                )399 400                all_markdown.append(result)401 402            except Exception as e:403                logger.error(f"Error processing image {i}: {e}")404                all_markdown.append("[OCR FAILED]")405 406    finally:407        # Clean up temp directories408        try:409            shutil.rmtree(temp_dir)410            shutil.rmtree(temp_output_dir)411        except Exception:412            pass413 414    # Add output column to dataset415    logger.info(f"Adding '{output_column}' column to dataset")416    dataset = dataset.add_column(output_column, all_markdown)417 418    # Handle inference_info tracking419    logger.info("Updating inference_info...")420 421    # Check for existing inference_info422    if "inference_info" in dataset.column_names:423        try:424            existing_info = json.loads(dataset[0]["inference_info"])425            if not isinstance(existing_info, list):426                existing_info = [existing_info]427        except (json.JSONDecodeError, TypeError):428            existing_info = []429        dataset = dataset.remove_columns(["inference_info"])430    else:431        existing_info = []432 433    # Add new inference info434    new_info = {435        "column_name": output_column,436        "model_id": model,437        "processing_date": datetime.now().isoformat(),438        "resolution_mode": resolution_mode,439        "base_size": final_base_size,440        "image_size": final_image_size,441        "crop_mode": final_crop_mode,442        "prompt": prompt,443        "script": "deepseek-ocr.py",444        "script_version": "1.0.0",445        "script_url": "https://huggingface.co/datasets/uv-scripts/ocr/raw/main/deepseek-ocr.py",446        "implementation": "transformers (sequential)",447    }448    existing_info.append(new_info)449 450    # Add updated inference_info column451    info_json = json.dumps(existing_info, ensure_ascii=False)452    dataset = dataset.add_column("inference_info", [info_json] * len(dataset))453 454    # Push to hub455    logger.info(f"Pushing to {output_dataset}")456    dataset.push_to_hub(output_dataset, private=private, token=HF_TOKEN)457 458    # Calculate processing time459    end_time = datetime.now()460    processing_duration = end_time - start_time461    processing_time = f"{processing_duration.total_seconds() / 60:.1f} minutes"462 463    # Create and push dataset card464    logger.info("Creating dataset card...")465    card_content = create_dataset_card(466        source_dataset=input_dataset,467        model=model,468        num_samples=len(dataset),469        processing_time=processing_time,470        resolution_mode=resolution_mode,471        base_size=final_base_size,472        image_size=final_image_size,473        crop_mode=final_crop_mode,474        image_column=image_column,475        split=split,476    )477 478    card = DatasetCard(card_content)479    card.push_to_hub(output_dataset, token=HF_TOKEN)480    logger.info("βœ… Dataset card created and pushed!")481 482    logger.info("βœ… OCR conversion complete!")483    logger.info(484        f"Dataset available at: https://huggingface.co/datasets/{output_dataset}"485    )486 487 488if __name__ == "__main__":489    # Show example usage if no arguments490    if len(sys.argv) == 1:491        print("=" * 80)492        print("DeepSeek-OCR to Markdown Converter (Transformers)")493        print("=" * 80)494        print("\nThis script converts document images to markdown using")495        print("DeepSeek-OCR with the official Transformers API.")496        print("\nFeatures:")497        print("- Multiple resolution modes (Tiny/Small/Base/Large/Gundam)")498        print("- LaTeX equation recognition")499        print("- Table extraction and formatting")500        print("- Document structure preservation")501        print("- Image grounding and spatial layout")502        print("- Multilingual support")503        print("\nNote: Sequential processing (no batching). Slower than vLLM scripts.")504        print("\nExample usage:")505        print("\n1. Basic OCR conversion (Gundam mode - dynamic resolution):")506        print("   uv run deepseek-ocr.py document-images markdown-docs")507        print("\n2. High quality mode (Large - 1280Γ—1280):")508        print("   uv run deepseek-ocr.py scanned-pdfs extracted-text --resolution-mode large")509        print("\n3. Fast processing (Tiny - 512Γ—512):")510        print("   uv run deepseek-ocr.py quick-test output --resolution-mode tiny")511        print("\n4. Process a subset for testing:")512        print("   uv run deepseek-ocr.py large-dataset test-output --max-samples 10")513        print("\n5. Custom resolution:")514        print("   uv run deepseek-ocr.py dataset output \\")515        print("       --base-size 1024 --image-size 640 --crop-mode")516        print("\n6. Running on HF Jobs:")517        print("   hf jobs uv run --flavor l4x1 \\")518        print('     --secrets HF_TOKEN \\')519        print("     https://huggingface.co/datasets/uv-scripts/ocr/raw/main/deepseek-ocr.py \\")520        print("       your-document-dataset \\")521        print("       your-markdown-output")522        print("\n" + "=" * 80)523        print("\nFor full help, run: uv run deepseek-ocr.py --help")524        sys.exit(0)525 526    parser = argparse.ArgumentParser(527        description="OCR images to markdown using DeepSeek-OCR (Transformers)",528        formatter_class=argparse.RawDescriptionHelpFormatter,529        epilog="""530Resolution Modes:531  tiny      512Γ—512 pixels, fast processing (64 vision tokens)532  small     640Γ—640 pixels, balanced (100 vision tokens)533  base      1024Γ—1024 pixels, high quality (256 vision tokens)534  large     1280Γ—1280 pixels, maximum quality (400 vision tokens)535  gundam    Dynamic multi-tile processing (adaptive)536 537Examples:538  # Basic usage with default Gundam mode539  uv run deepseek-ocr.py my-images-dataset ocr-results540 541  # High quality processing542  uv run deepseek-ocr.py documents extracted-text --resolution-mode large543 544  # Fast processing for testing545  uv run deepseek-ocr.py dataset output --resolution-mode tiny --max-samples 100546 547  # Custom resolution settings548  uv run deepseek-ocr.py dataset output --base-size 1024 --image-size 640 --crop-mode549        """,550    )551 552    parser.add_argument("input_dataset", help="Input dataset ID from Hugging Face Hub")553    parser.add_argument("output_dataset", help="Output dataset ID for Hugging Face Hub")554    parser.add_argument(555        "--image-column",556        default="image",557        help="Column containing images (default: image)",558    )559    parser.add_argument(560        "--model",561        default="deepseek-ai/DeepSeek-OCR",562        help="Model to use (default: deepseek-ai/DeepSeek-OCR)",563    )564    parser.add_argument(565        "--resolution-mode",566        default="gundam",567        choices=list(RESOLUTION_MODES.keys()) + ["custom"],568        help="Resolution mode preset (default: gundam)",569    )570    parser.add_argument(571        "--base-size",572        type=int,573        help="Base resolution size (overrides resolution-mode)",574    )575    parser.add_argument(576        "--image-size",577        type=int,578        help="Image tile size (overrides resolution-mode)",579    )580    parser.add_argument(581        "--crop-mode",582        action="store_true",583        help="Enable dynamic multi-tile cropping (overrides resolution-mode)",584    )585    parser.add_argument(586        "--prompt",587        default="<image>\n<|grounding|>Convert the document to markdown.",588        help="Prompt for OCR (default: grounding markdown conversion)",589    )590    parser.add_argument("--hf-token", help="Hugging Face API token")591    parser.add_argument(592        "--split", default="train", help="Dataset split to use (default: train)"593    )594    parser.add_argument(595        "--max-samples",596        type=int,597        help="Maximum number of samples to process (for testing)",598    )599    parser.add_argument(600        "--private", action="store_true", help="Make output dataset private"601    )602    parser.add_argument(603        "--shuffle",604        action="store_true",605        help="Shuffle the dataset before processing (useful for random sampling)",606    )607    parser.add_argument(608        "--seed",609        type=int,610        default=42,611        help="Random seed for shuffling (default: 42)",612    )613    parser.add_argument(614        "--output-column",615        default="markdown",616        help="Column name for the OCR output text (default: markdown)",617    )618    parser.add_argument(619        "--overwrite",620        action="store_true",621        help="Replace the output column if it already exists in the input dataset "622        "(default: error out to avoid clobbering an existing column).",623    )624 625    args = parser.parse_args()626 627    main(628        input_dataset=args.input_dataset,629        output_dataset=args.output_dataset,630        image_column=args.image_column,631        model=args.model,632        resolution_mode=args.resolution_mode,633        base_size=args.base_size,634        image_size=args.image_size,635        crop_mode=args.crop_mode if args.crop_mode else None,636        prompt=args.prompt,637        hf_token=args.hf_token,638        split=args.split,639        max_samples=args.max_samples,640        private=args.private,641        shuffle=args.shuffle,642        seed=args.seed,643        output_column=args.output_column,644        overwrite=args.overwrite,645    )646