Team Ai
Datasetpublic

uv-scripts/ocr

OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.

sourceHugging Faceupdated 11d agoView on Hugging Face
163likes6.5kdownloads
nuextract3.py785 linesDownload Raw Back to root
1# /// script2# requires-python = ">=3.11"3# dependencies = [4#     "datasets>=3.1.0",5#     "huggingface-hub",6#     "pillow",7#     "toolz",8#     "numind",9# ]10#11# [tool.hf-jobs]12# image = "vllm/vllm-openai:v0.29.0"13# python = "/usr/bin/python3"14# env = { PYTHONPATH = "/usr/local/lib/python3.12/dist-packages" }15# flavor = "a10g-small"16# secrets = ["HF_TOKEN"]17# ///18 19"""20Convert document images to markdown OR extract structured JSON using NuExtract3 with vLLM.21 22NuExtract3 is a 4B Qwen3.5-based VLM for document understanding. It does two things:23 241. Document-to-Markdown OCR (default): images -> clean markdown with HTML tables,25   LaTeX math, and <figure> tags.262. Schema-guided structured extraction: images + a JSON template -> JSON output27   shaped exactly like the template. Useful for invoices, receipts, forms, contracts.28 29Modes are selected via flags:30- (no flags)         -> markdown OCR31- --mode content     -> plain-content extraction32- --template SOURCE  -> structured extraction with a NuExtract template33- --schema SOURCE    -> structured extraction with a JSON Schema34                        (auto-converted via numind.nuextract_utils)35- --instructions STR -> free-text guidance passed through to the model36                        (output-format rules, branch routing, etc.).37                        Combines with any of the modes above.38                        See https://huggingface.co/numind/NuExtract3#instructions39 40--template / --schema each accept inline JSON, a URL, or a local file path, so a41schema can be hosted (e.g. on an HF dataset's raw URL) and reused across jobs:42    --template https://huggingface.co/datasets/ORG/REPO/raw/main/card.json43 44HF Jobs invocation: the [tool.hf-jobs] header above sets the vllm/vllm-openai45image (vLLM + torch come from the image, with pre-built CUDA kernels), the46flavor and the HF_TOKEN secret. Needs `hf` CLI 1.32+.47 48    hf jobs uv run \\49        https://huggingface.co/datasets/uv-scripts/ocr/raw/main/nuextract3.py \\50        INPUT_DATASET OUTPUT_DATASET --max-samples 5 --shuffle --seed 4251 52On your own GPU: uv run --with vllm==0.29.0 nuextract3.py INPUT_DATASET OUTPUT_DATASET53 54Model: numind/NuExtract355License: Apache-2.056"""57 58import argparse59import base6460import io61import json62import logging63import os64import sys65import time66from datetime import datetime67from pathlib import Path68from typing import Any, Dict, List, Optional, Union69 70import torch71from datasets import load_dataset72from huggingface_hub import DatasetCard, login73from PIL import Image74from toolz import partition_all75# Disable vLLM's FlashInfer sampler: it JIT-compiles a CUDA kernel needing nvcc, which the76# default uv-script image lacks (engine init then crashes). Greedy OCR doesn't use it; this77# lets the plain default-image command work. On the vllm/vllm-openai image it's a harmless no-op.78os.environ.setdefault("VLLM_USE_FLASHINFER_SAMPLER", "0")79from vllm import LLM, SamplingParams80 81logging.basicConfig(level=logging.INFO)82logger = logging.getLogger(__name__)83 84MODEL_DEFAULT = "numind/NuExtract3"85MODEL_NAME = "NuExtract3"86 87 88def check_cuda_availability():89    """Check if CUDA is available and exit if not."""90    if not torch.cuda.is_available():91        logger.error("CUDA is not available. This script requires a GPU.")92        logger.error("Please run on a machine with a CUDA-capable GPU.")93        sys.exit(1)94    else:95        logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name(0)}")96 97 98def ensure_output_columns_free(dataset, columns, overwrite=False):99    """Fail fast if an output column would collide with an existing input column.100 101    Adding a column that already exists silently overwrites it (e.g. a ground-truth102    `text`/`markdown` column) or crashes on push with a duplicate-column error only103    *after* inference has run. Catch it up front. With overwrite=True, drop the clashing104    column(s) here instead (logged) so the later add_column is clean.105    """106    clash = [c for c in columns if c in dataset.column_names]107    if not clash:108        return dataset109    if overwrite:110        logger.warning(f"--overwrite: replacing existing column(s) {clash}")111        return dataset.remove_columns(clash)112    logger.error(113        f"Output column(s) {clash} already exist in the input dataset "114        f"(columns: {dataset.column_names})."115    )116    logger.error("Choose a different --output-column, or pass --overwrite to replace them.")117    sys.exit(1)118 119 120def load_template_arg(value: Optional[str]) -> Optional[Dict[str, Any]]:121    """Load a NuExtract template/JSON Schema from inline JSON, a URL, or a file path."""122    if value is None:123        return None124    text = value125    if value.startswith(("http://", "https://")):126        import urllib.request127 128        with urllib.request.urlopen(value) as resp:  # noqa: S310129            text = resp.read().decode("utf-8")130    elif "{" not in value:131        # Inline JSON often exceeds the OS filename limit, so only probe the132        # filesystem when the value doesn't look like JSON; treat OSError as133        # "not a path".134        try:135            candidate_path = Path(value)136            if candidate_path.is_file():137                text = candidate_path.read_text()138        except OSError:139            pass140    try:141        return json.loads(text)142    except json.JSONDecodeError as e:143        raise ValueError(144            f"Could not parse template/schema as JSON (tried URL/path/inline): {e}"145        ) from e146 147 148def resolve_template(149    template_arg: Optional[str],150    schema_arg: Optional[str],151) -> Optional[Dict[str, Any]]:152    """Resolve --template / --schema into a NuExtract template dict, or None."""153    if template_arg and schema_arg:154        raise ValueError("--template and --schema are mutually exclusive.")155 156    if template_arg is not None:157        return load_template_arg(template_arg)158 159    if schema_arg is not None:160        schema = load_template_arg(schema_arg)161        try:162            from numind.nuextract_utils import convert_json_schema_to_nuextract_template163        except ImportError as e:164            raise RuntimeError(165                "--schema requires the `numind` package. "166                "It should be listed in this script's PEP 723 dependencies."167            ) from e168        template, dropped = convert_json_schema_to_nuextract_template(schema)169        if dropped:170            logger.warning(171                f"numind dropped {len(dropped)} unsupported branches from the JSON Schema: "172                f"{dropped}"173            )174        return template175 176    return None177 178 179def image_to_data_uri(image: Union[Image.Image, Dict[str, Any], str]) -> str:180    """Normalize an HF dataset image cell to a PNG data URI."""181    if isinstance(image, Image.Image):182        pil_img = image183    elif isinstance(image, dict) and "bytes" in image:184        pil_img = Image.open(io.BytesIO(image["bytes"]))185    elif isinstance(image, str):186        pil_img = Image.open(image)187    else:188        raise ValueError(f"Unsupported image type: {type(image)}")189 190    pil_img = pil_img.convert("RGB")191    buf = io.BytesIO()192    pil_img.save(buf, format="PNG")193    return f"data:image/png;base64,{base64.b64encode(buf.getvalue()).decode()}"194 195 196def make_message(image: Union[Image.Image, Dict[str, Any], str]) -> List[Dict]:197    """Build an OpenAI-format chat message containing one image."""198    data_uri = image_to_data_uri(image)199    return [200        {201            "role": "user",202            "content": [203                {"type": "image_url", "image_url": {"url": data_uri}},204            ],205        }206    ]207 208 209def split_thinking(text: str) -> tuple[Optional[str], str]:210    """Return (reasoning, answer) if <think>...</think> is present, else (None, text)."""211    if "<think>" in text and "</think>" in text:212        reasoning = text.split("<think>", 1)[1].split("</think>", 1)[0].strip()213        answer = text.split("</think>", 1)[1].strip()214        return reasoning, answer215    return None, text.strip()216 217 218def parse_json_output(text: str) -> tuple[Optional[Any], bool]:219    """Parse an extraction output; strip ``` fences as the model card describes.220 221    Returns (parsed_value, parse_error). On failure, parsed_value is None.222    """223    stripped = text.strip()224    if stripped.startswith("```"):225        stripped = stripped.split("\n", 1)[-1] if "\n" in stripped else stripped[3:]226        if stripped.endswith("```"):227            stripped = stripped[:-3].rstrip()228    try:229        return json.loads(stripped), False230    except json.JSONDecodeError:231        return None, True232 233 234def create_dataset_card(235    source_dataset: str,236    model: str,237    num_samples: int,238    processing_time: str,239    mode_label: str,240    template: Optional[Dict[str, Any]],241    enable_thinking: bool,242    temperature: float,243    output_column: str,244    image_column: str,245    split: str,246) -> str:247    """Create a dataset card documenting the NuExtract3 run."""248    model_name = model.split("/")[-1]249    template_block = ""250    if template is not None:251        template_block = (252            "\n### Extraction Template\n\n```json\n"253            + json.dumps(template, indent=2)254            + "\n```\n"255        )256 257    return f"""---258tags:259- ocr260- structured-extraction261- document-processing262- nuextract3263- markdown264- uv-script265- generated266---267 268# {model_name} on {source_dataset}269 270This dataset contains outputs from [{source_dataset}](https://huggingface.co/datasets/{source_dataset}) processed with [NuExtract3](https://huggingface.co/{model}), a 4B vision-language model for document understanding.271 272## Processing Details273 274- **Source Dataset**: [{source_dataset}](https://huggingface.co/datasets/{source_dataset})275- **Model**: [{model}](https://huggingface.co/{model})276- **Mode**: {mode_label}277- **Number of Samples**: {num_samples:,}278- **Processing Time**: {processing_time}279- **Processing Date**: {datetime.now().strftime("%Y-%m-%d %H:%M UTC")}280 281### Configuration282 283- **Image Column**: `{image_column}`284- **Output Column**: `{output_column}`285- **Dataset Split**: `{split}`286- **Temperature**: {temperature}287- **Thinking Mode**: {"enabled" if enable_thinking else "disabled"}288{template_block}289## Dataset Structure290 291Original columns plus:292- `{output_column}`: NuExtract3 output ({"JSON string" if template else "markdown"})293- `inference_info`: JSON list tracking models applied to this dataset294{"- `" + output_column + "_reasoning`: model's thinking trace (when enabled)" if enable_thinking else ""}295 296Generated with [UV Scripts](https://huggingface.co/uv-scripts)297"""298 299 300def main(301    input_dataset: str,302    output_dataset: str,303    image_column: str = "image",304    batch_size: int = 16,305    max_model_len: int = 16384,306    max_tokens: int = 8192,307    gpu_memory_utilization: float = 0.8,308    mode: str = "markdown",309    template_arg: Optional[str] = None,310    schema_arg: Optional[str] = None,311    enable_thinking: bool = False,312    instructions: Optional[str] = None,313    temperature: Optional[float] = None,314    model: str = MODEL_DEFAULT,315    hf_token: str = None,316    split: str = "train",317    max_samples: int = None,318    private: bool = False,319    shuffle: bool = False,320    seed: int = 42,321    output_column: Optional[str] = None,322    overwrite: bool = False,323    verbose: bool = False,324    config: str = None,325    create_pr: bool = False,326):327    """Process images from an HF dataset through NuExtract3."""328 329    check_cuda_availability()330    start_time = datetime.now()331 332    HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")333    if HF_TOKEN:334        login(token=HF_TOKEN)335 336    template = resolve_template(template_arg, schema_arg)337    extraction_mode = template is not None338    mode_label = "structured-extraction" if extraction_mode else mode339 340    if output_column is None:341        output_column = "extraction" if extraction_mode else "markdown"342 343    if temperature is None:344        temperature = 0.6 if enable_thinking else 0.2345 346    logger.info(f"Using model: {model}")347    logger.info(f"Mode: {mode_label}")348    logger.info(f"Thinking: {enable_thinking}  Temperature: {temperature}")349    if extraction_mode:350        logger.info(f"Template: {json.dumps(template, indent=2)}")351 352    logger.info(f"Loading dataset: {input_dataset}")353    dataset = load_dataset(input_dataset, split=split)354 355    if image_column not in dataset.column_names:356        raise ValueError(357            f"Column '{image_column}' not found. Available: {dataset.column_names}"358        )359 360    # Fail fast if an output column would collide with an existing input column.361    # With --enable-thinking the script also writes "{output_column}_reasoning".362    out_cols = [output_column]363    if enable_thinking:364        out_cols.append(f"{output_column}_reasoning")365    dataset = ensure_output_columns_free(dataset, out_cols, overwrite=overwrite)366 367    if shuffle:368        logger.info(f"Shuffling dataset with seed {seed}")369        dataset = dataset.shuffle(seed=seed)370 371    if max_samples:372        dataset = dataset.select(range(min(max_samples, len(dataset))))373        logger.info(f"Limited to {len(dataset)} samples")374 375    logger.info("Initializing vLLM with NuExtract3")376    logger.info("This may take a few minutes on first run...")377    llm = LLM(378        model=model,379        trust_remote_code=True,380        max_model_len=max_model_len,381        gpu_memory_utilization=gpu_memory_utilization,382        limit_mm_per_prompt={"image": 1},383    )384 385    sampling_params = SamplingParams(386        temperature=temperature,387        max_tokens=max_tokens,388    )389 390    chat_template_kwargs: Dict[str, Any] = {"enable_thinking": enable_thinking}391    if extraction_mode:392        chat_template_kwargs["template"] = json.dumps(template, indent=4)393    else:394        chat_template_kwargs["mode"] = mode395    if instructions:396        chat_template_kwargs["instructions"] = instructions397 398    logger.info(f"Processing {len(dataset)} images in batches of {batch_size}")399    logger.info(f"Output will be written to column: {output_column}")400 401    all_outputs: List[str] = []402    all_reasoning: List[Optional[str]] = []403    all_parse_errors: List[bool] = []404    total_batches = (len(dataset) + batch_size - 1) // batch_size405    processed = 0406 407    for batch_num, batch_indices in enumerate(408        partition_all(batch_size, range(len(dataset))), 1409    ):410        batch_indices = list(batch_indices)411        batch_images = [dataset[i][image_column] for i in batch_indices]412 413        logger.info(414            f"Batch {batch_num}/{total_batches} "415            f"({processed}/{len(dataset)} images done)"416        )417 418        try:419            batch_messages = [make_message(img) for img in batch_images]420            outputs = llm.chat(421                batch_messages,422                sampling_params,423                chat_template_kwargs=chat_template_kwargs,424                chat_template_content_format="openai",425            )426 427            for output in outputs:428                raw_text = output.outputs[0].text429                reasoning, answer = split_thinking(raw_text)430 431                if extraction_mode:432                    parsed, parse_error = parse_json_output(answer)433                    stored = (434                        json.dumps(parsed, ensure_ascii=False)435                        if parsed is not None436                        else answer437                    )438                    all_outputs.append(stored)439                    all_parse_errors.append(parse_error)440                else:441                    all_outputs.append(answer)442                    all_parse_errors.append(False)443 444                all_reasoning.append(reasoning)445 446            processed += len(batch_images)447 448        except Exception as e:449            logger.error(f"Error processing batch: {e}")450            all_outputs.extend(["[NUEXTRACT3 ERROR]"] * len(batch_images))451            all_reasoning.extend([None] * len(batch_images))452            all_parse_errors.extend([True] * len(batch_images))453            processed += len(batch_images)454 455    processing_duration = datetime.now() - start_time456    processing_time_str = f"{processing_duration.total_seconds() / 60:.1f} min"457 458    logger.info(f"Adding '{output_column}' column to dataset")459    dataset = dataset.add_column(output_column, all_outputs)460 461    if enable_thinking and any(r is not None for r in all_reasoning):462        reasoning_col = f"{output_column}_reasoning"463        logger.info(f"Adding '{reasoning_col}' column to dataset")464        dataset = dataset.add_column(reasoning_col, all_reasoning)465 466    if extraction_mode:467        parse_error_count = sum(all_parse_errors)468        if parse_error_count:469            logger.warning(470                f"{parse_error_count}/{len(all_parse_errors)} extractions failed to parse as JSON"471            )472 473    inference_entry = {474        "model_id": model,475        "model_name": MODEL_NAME,476        "column_name": output_column,477        "timestamp": datetime.now().isoformat(),478        "mode": mode_label,479        "has_template": extraction_mode,480        "enable_thinking": enable_thinking,481        "temperature": temperature,482        "max_tokens": max_tokens,483    }484    if extraction_mode:485        inference_entry["parse_error_rate"] = (486            sum(all_parse_errors) / len(all_parse_errors) if all_parse_errors else 0.0487        )488 489    if "inference_info" in dataset.column_names:490        logger.info("Updating existing inference_info column")491 492        def update_inference_info(example):493            try:494                existing_info = (495                    json.loads(example["inference_info"])496                    if example["inference_info"]497                    else []498                )499            except (json.JSONDecodeError, TypeError):500                existing_info = []501            existing_info.append(inference_entry)502            return {"inference_info": json.dumps(existing_info)}503 504        dataset = dataset.map(update_inference_info)505    else:506        logger.info("Creating new inference_info column")507        inference_list = [json.dumps([inference_entry])] * len(dataset)508        dataset = dataset.add_column("inference_info", inference_list)509 510    logger.info(f"Pushing to {output_dataset}")511    max_retries = 3512    for attempt in range(1, max_retries + 1):513        try:514            if attempt > 1:515                logger.warning("Disabling XET (fallback to HTTP upload)")516                os.environ["HF_HUB_DISABLE_XET"] = "1"517            dataset.push_to_hub(518                output_dataset,519                private=private,520                token=HF_TOKEN,521                max_shard_size="500MB",522                **({"config_name": config} if config else {}),523                create_pr=create_pr,524                commit_message=f"Add {model} {mode_label} results ({len(dataset)} samples)"525                + (f" [{config}]" if config else ""),526            )527            break528        except Exception as e:529            logger.error(f"Upload attempt {attempt}/{max_retries} failed: {e}")530            if attempt < max_retries:531                delay = 30 * (2 ** (attempt - 1))532                logger.info(f"Retrying in {delay}s...")533                time.sleep(delay)534            else:535                logger.error("All upload attempts failed. Results are lost.")536                sys.exit(1)537 538    logger.info("Creating dataset card")539    card_content = create_dataset_card(540        source_dataset=input_dataset,541        model=model,542        num_samples=len(dataset),543        processing_time=processing_time_str,544        mode_label=mode_label,545        template=template,546        enable_thinking=enable_thinking,547        temperature=temperature,548        output_column=output_column,549        image_column=image_column,550        split=split,551    )552    card = DatasetCard(card_content)553    card.push_to_hub(output_dataset, token=HF_TOKEN)554 555    logger.info("Done! NuExtract3 processing complete.")556    logger.info(557        f"Dataset available at: https://huggingface.co/datasets/{output_dataset}"558    )559    logger.info(f"Processing time: {processing_time_str}")560    logger.info(561        f"Processing speed: {len(dataset) / processing_duration.total_seconds():.2f} images/sec"562    )563 564    if verbose:565        import importlib.metadata566 567        logger.info("--- Resolved package versions ---")568        for pkg in [569            "vllm",570            "transformers",571            "torch",572            "datasets",573            "pyarrow",574            "pillow",575            "numind",576        ]:577            try:578                logger.info(f"  {pkg}=={importlib.metadata.version(pkg)}")579            except importlib.metadata.PackageNotFoundError:580                logger.info(f"  {pkg}: not installed")581        logger.info("--- End versions ---")582 583 584if __name__ == "__main__":585    if len(sys.argv) == 1:586        print("=" * 70)587        print("NuExtract3 - Document-to-Markdown + Structured Extraction (4B)")588        print("=" * 70)589        print("\nModes:")590        print("  markdown          - Image -> markdown (default)")591        print("  content           - Image -> plain content")592        print("  --template / --schema  - Image -> JSON shaped like the template")593        print("\nExamples:")594        print("\n1. Markdown OCR:")595        print("   uv run --with vllm==0.29.0 nuextract3.py input-dataset output-dataset")596        print("\n2. Structured extraction with an inline template:")597        print("   uv run --with vllm==0.29.0 nuextract3.py input output \\")598        print('     --template \'{"title": "verbatim-string", "date": "date"}\'')599        print("\n3. Structured extraction from a JSON Schema (e.g. Pydantic):")600        print("   uv run --with vllm==0.29.0 nuextract3.py input output --schema schema.json")601        print("\n   (--template / --schema also accept a URL or a local file path)")602        print("\n4. Reasoning mode for harder documents:")603        print("   uv run --with vllm==0.29.0 nuextract3.py input output --enable-thinking")604        print("\n5. Test with 10 samples:")605        print("   uv run --with vllm==0.29.0 nuextract3.py large-ds test --max-samples 10 --shuffle")606        print("\n6. Running on HF Jobs (image/flavor/secrets from the script header, hf 1.32+):")607        print("   hf jobs uv run \\")608        print(609            "     https://huggingface.co/datasets/uv-scripts/ocr/raw/main/nuextract3.py \\"610        )611        print("       input-dataset output-dataset --batch-size 16")612        print("\nFor full help: uv run --with vllm==0.29.0 nuextract3.py --help")613        sys.exit(0)614 615    parser = argparse.ArgumentParser(616        description="NuExtract3: document-to-markdown + schema-guided JSON extraction (4B VLM)",617        formatter_class=argparse.RawDescriptionHelpFormatter,618        epilog="""619Modes:620  (default)     Markdown OCR (image -> clean markdown)621  --mode content622                Plain-content extraction (less structured than markdown)623  --template PATH_OR_JSON624                Structured extraction with a NuExtract template625  --schema PATH_OR_JSON626                Structured extraction from a JSON Schema627                (e.g. Pydantic Model.model_json_schema())628 629Examples:630  uv run --with vllm==0.29.0 nuextract3.py my-docs analyzed-docs631  uv run --with vllm==0.29.0 nuextract3.py receipts extracted \\632      --template '{"store": "verbatim-string", "total": "number"}'633  uv run --with vllm==0.29.0 nuextract3.py contracts extracted --schema contract_schema.json634  uv run --with vllm==0.29.0 nuextract3.py hard-docs out --enable-thinking635        """,636    )637 638    parser.add_argument("input_dataset", help="Input dataset ID from Hugging Face Hub")639    parser.add_argument("output_dataset", help="Output dataset ID for Hugging Face Hub")640    parser.add_argument(641        "--image-column",642        default="image",643        help="Column containing images (default: image)",644    )645    parser.add_argument(646        "--batch-size",647        type=int,648        default=16,649        help="Batch size for processing (default: 16)",650    )651    parser.add_argument(652        "--max-model-len",653        type=int,654        default=16384,655        help="Maximum model context length (default: 16384)",656    )657    parser.add_argument(658        "--max-tokens",659        type=int,660        default=8192,661        help="Maximum tokens to generate (default: 8192)",662    )663    parser.add_argument(664        "--gpu-memory-utilization",665        type=float,666        default=0.8,667        help="GPU memory utilization (default: 0.8)",668    )669    parser.add_argument(670        "--mode",671        choices=["markdown", "content"],672        default="markdown",673        help="OCR mode when no template/schema is given (default: markdown)",674    )675    parser.add_argument(676        "--template",677        help="NuExtract template: inline JSON, a URL, or a file path",678    )679    parser.add_argument(680        "--schema",681        help="JSON Schema to auto-convert: inline JSON, a URL, or a file path",682    )683    parser.add_argument(684        "--enable-thinking",685        action="store_true",686        help="Enable reasoning mode (slower, better on hard documents)",687    )688    parser.add_argument(689        "--instructions",690        default=None,691        help=(692            "Free-text instructions passed to NuExtract via "693            "chat_template_kwargs.instructions (e.g. routing guidance across "694            "optional schema branches, output-format rules). "695            "See https://huggingface.co/numind/NuExtract3#instructions"696        ),697    )698    parser.add_argument(699        "--temperature",700        type=float,701        default=None,702        help="Sampling temperature (default: 0.2 non-thinking, 0.6 thinking)",703    )704    parser.add_argument(705        "--model",706        default=MODEL_DEFAULT,707        help=f"Model ID (default: {MODEL_DEFAULT})",708    )709    parser.add_argument("--hf-token", help="Hugging Face API token")710    parser.add_argument(711        "--split", default="train", help="Dataset split to use (default: train)"712    )713    parser.add_argument(714        "--max-samples",715        type=int,716        help="Maximum number of samples to process (for testing)",717    )718    parser.add_argument(719        "--private", action="store_true", help="Make output dataset private"720    )721    parser.add_argument(722        "--config",723        help="Config/subset name when pushing to Hub (for benchmarking multiple models in one repo)",724    )725    parser.add_argument(726        "--create-pr",727        action="store_true",728        help="Create a pull request instead of pushing directly (for parallel benchmarking)",729    )730    parser.add_argument(731        "--shuffle", action="store_true", help="Shuffle dataset before processing"732    )733    parser.add_argument(734        "--seed",735        type=int,736        default=42,737        help="Random seed for shuffling (default: 42)",738    )739    parser.add_argument(740        "--output-column",741        default=None,742        help="Column name for output (default: 'markdown' in OCR mode, 'extraction' in template mode)",743    )744    parser.add_argument(745        "--overwrite",746        action="store_true",747        help="Replace the output column if it already exists in the input dataset "748        "(default: error out to avoid clobbering an existing column).",749    )750    parser.add_argument(751        "--verbose",752        action="store_true",753        help="Log resolved package versions after processing",754    )755 756    args = parser.parse_args()757 758    main(759        input_dataset=args.input_dataset,760        output_dataset=args.output_dataset,761        image_column=args.image_column,762        batch_size=args.batch_size,763        max_model_len=args.max_model_len,764        max_tokens=args.max_tokens,765        gpu_memory_utilization=args.gpu_memory_utilization,766        mode=args.mode,767        template_arg=args.template,768        schema_arg=args.schema,769        enable_thinking=args.enable_thinking,770        instructions=args.instructions,771        temperature=args.temperature,772        model=args.model,773        hf_token=args.hf_token,774        split=args.split,775        max_samples=args.max_samples,776        private=args.private,777        shuffle=args.shuffle,778        seed=args.seed,779        output_column=args.output_column,780        overwrite=args.overwrite,781        verbose=args.verbose,782        config=args.config,783        create_pr=args.create_pr,784    )785