uv-scripts/ocr
OCR UV Scripts Part of uv-scripts: self-contained UV scripts you run on Hugging Face Jobs in one command. One script per OCR model. Each script runs the model on a GPU with Hugging Face Jobs and writes the text as markdown: as a new column in a Hub dataset, as .md files in a Bucket, or as resumable parquet parts (the -saturate recipes). A few scripts return JSON from a schema, detect layout regions, or compare the output of two models. Quick Start First… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/ocr.
1636.5k
1# /// script2# requires-python = ">=3.11"3# dependencies = [4# "datasets>=3.1.0",5# "huggingface-hub",6# "pillow",7# "toolz",8# "numind",9# ]10#11# [tool.hf-jobs]12# image = "vllm/vllm-openai:v0.29.0"13# python = "/usr/bin/python3"14# env = { PYTHONPATH = "/usr/local/lib/python3.12/dist-packages" }15# flavor = "a10g-small"16# secrets = ["HF_TOKEN"]17# ///18 19"""20Convert document images to markdown OR extract structured JSON using NuExtract3 with vLLM.21 22NuExtract3 is a 4B Qwen3.5-based VLM for document understanding. It does two things:23 241. Document-to-Markdown OCR (default): images -> clean markdown with HTML tables,25 LaTeX math, and <figure> tags.262. Schema-guided structured extraction: images + a JSON template -> JSON output27 shaped exactly like the template. Useful for invoices, receipts, forms, contracts.28 29Modes are selected via flags:30- (no flags) -> markdown OCR31- --mode content -> plain-content extraction32- --template SOURCE -> structured extraction with a NuExtract template33- --schema SOURCE -> structured extraction with a JSON Schema34 (auto-converted via numind.nuextract_utils)35- --instructions STR -> free-text guidance passed through to the model36 (output-format rules, branch routing, etc.).37 Combines with any of the modes above.38 See https://huggingface.co/numind/NuExtract3#instructions39 40--template / --schema each accept inline JSON, a URL, or a local file path, so a41schema can be hosted (e.g. on an HF dataset's raw URL) and reused across jobs:42 --template https://huggingface.co/datasets/ORG/REPO/raw/main/card.json43 44HF Jobs invocation: the [tool.hf-jobs] header above sets the vllm/vllm-openai45image (vLLM + torch come from the image, with pre-built CUDA kernels), the46flavor and the HF_TOKEN secret. Needs `hf` CLI 1.32+.47 48 hf jobs uv run \\49 https://huggingface.co/datasets/uv-scripts/ocr/raw/main/nuextract3.py \\50 INPUT_DATASET OUTPUT_DATASET --max-samples 5 --shuffle --seed 4251 52On your own GPU: uv run --with vllm==0.29.0 nuextract3.py INPUT_DATASET OUTPUT_DATASET53 54Model: numind/NuExtract355License: Apache-2.056"""57 58import argparse59import base6460import io61import json62import logging63import os64import sys65import time66from datetime import datetime67from pathlib import Path68from typing import Any, Dict, List, Optional, Union69 70import torch71from datasets import load_dataset72from huggingface_hub import DatasetCard, login73from PIL import Image74from toolz import partition_all75# Disable vLLM's FlashInfer sampler: it JIT-compiles a CUDA kernel needing nvcc, which the76# default uv-script image lacks (engine init then crashes). Greedy OCR doesn't use it; this77# lets the plain default-image command work. On the vllm/vllm-openai image it's a harmless no-op.78os.environ.setdefault("VLLM_USE_FLASHINFER_SAMPLER", "0")79from vllm import LLM, SamplingParams80 81logging.basicConfig(level=logging.INFO)82logger = logging.getLogger(__name__)83 84MODEL_DEFAULT = "numind/NuExtract3"85MODEL_NAME = "NuExtract3"86 87 88def check_cuda_availability():89 """Check if CUDA is available and exit if not."""90 if not torch.cuda.is_available():91 logger.error("CUDA is not available. This script requires a GPU.")92 logger.error("Please run on a machine with a CUDA-capable GPU.")93 sys.exit(1)94 else:95 logger.info(f"CUDA is available. GPU: {torch.cuda.get_device_name(0)}")96 97 98def ensure_output_columns_free(dataset, columns, overwrite=False):99 """Fail fast if an output column would collide with an existing input column.100 101 Adding a column that already exists silently overwrites it (e.g. a ground-truth102 `text`/`markdown` column) or crashes on push with a duplicate-column error only103 *after* inference has run. Catch it up front. With overwrite=True, drop the clashing104 column(s) here instead (logged) so the later add_column is clean.105 """106 clash = [c for c in columns if c in dataset.column_names]107 if not clash:108 return dataset109 if overwrite:110 logger.warning(f"--overwrite: replacing existing column(s) {clash}")111 return dataset.remove_columns(clash)112 logger.error(113 f"Output column(s) {clash} already exist in the input dataset "114 f"(columns: {dataset.column_names})."115 )116 logger.error("Choose a different --output-column, or pass --overwrite to replace them.")117 sys.exit(1)118 119 120def load_template_arg(value: Optional[str]) -> Optional[Dict[str, Any]]:121 """Load a NuExtract template/JSON Schema from inline JSON, a URL, or a file path."""122 if value is None:123 return None124 text = value125 if value.startswith(("http://", "https://")):126 import urllib.request127 128 with urllib.request.urlopen(value) as resp: # noqa: S310129 text = resp.read().decode("utf-8")130 elif "{" not in value:131 # Inline JSON often exceeds the OS filename limit, so only probe the132 # filesystem when the value doesn't look like JSON; treat OSError as133 # "not a path".134 try:135 candidate_path = Path(value)136 if candidate_path.is_file():137 text = candidate_path.read_text()138 except OSError:139 pass140 try:141 return json.loads(text)142 except json.JSONDecodeError as e:143 raise ValueError(144 f"Could not parse template/schema as JSON (tried URL/path/inline): {e}"145 ) from e146 147 148def resolve_template(149 template_arg: Optional[str],150 schema_arg: Optional[str],151) -> Optional[Dict[str, Any]]:152 """Resolve --template / --schema into a NuExtract template dict, or None."""153 if template_arg and schema_arg:154 raise ValueError("--template and --schema are mutually exclusive.")155 156 if template_arg is not None:157 return load_template_arg(template_arg)158 159 if schema_arg is not None:160 schema = load_template_arg(schema_arg)161 try:162 from numind.nuextract_utils import convert_json_schema_to_nuextract_template163 except ImportError as e:164 raise RuntimeError(165 "--schema requires the `numind` package. "166 "It should be listed in this script's PEP 723 dependencies."167 ) from e168 template, dropped = convert_json_schema_to_nuextract_template(schema)169 if dropped:170 logger.warning(171 f"numind dropped {len(dropped)} unsupported branches from the JSON Schema: "172 f"{dropped}"173 )174 return template175 176 return None177 178 179def image_to_data_uri(image: Union[Image.Image, Dict[str, Any], str]) -> str:180 """Normalize an HF dataset image cell to a PNG data URI."""181 if isinstance(image, Image.Image):182 pil_img = image183 elif isinstance(image, dict) and "bytes" in image:184 pil_img = Image.open(io.BytesIO(image["bytes"]))185 elif isinstance(image, str):186 pil_img = Image.open(image)187 else:188 raise ValueError(f"Unsupported image type: {type(image)}")189 190 pil_img = pil_img.convert("RGB")191 buf = io.BytesIO()192 pil_img.save(buf, format="PNG")193 return f"data:image/png;base64,{base64.b64encode(buf.getvalue()).decode()}"194 195 196def make_message(image: Union[Image.Image, Dict[str, Any], str]) -> List[Dict]:197 """Build an OpenAI-format chat message containing one image."""198 data_uri = image_to_data_uri(image)199 return [200 {201 "role": "user",202 "content": [203 {"type": "image_url", "image_url": {"url": data_uri}},204 ],205 }206 ]207 208 209def split_thinking(text: str) -> tuple[Optional[str], str]:210 """Return (reasoning, answer) if <think>...</think> is present, else (None, text)."""211 if "<think>" in text and "</think>" in text:212 reasoning = text.split("<think>", 1)[1].split("</think>", 1)[0].strip()213 answer = text.split("</think>", 1)[1].strip()214 return reasoning, answer215 return None, text.strip()216 217 218def parse_json_output(text: str) -> tuple[Optional[Any], bool]:219 """Parse an extraction output; strip ``` fences as the model card describes.220 221 Returns (parsed_value, parse_error). On failure, parsed_value is None.222 """223 stripped = text.strip()224 if stripped.startswith("```"):225 stripped = stripped.split("\n", 1)[-1] if "\n" in stripped else stripped[3:]226 if stripped.endswith("```"):227 stripped = stripped[:-3].rstrip()228 try:229 return json.loads(stripped), False230 except json.JSONDecodeError:231 return None, True232 233 234def create_dataset_card(235 source_dataset: str,236 model: str,237 num_samples: int,238 processing_time: str,239 mode_label: str,240 template: Optional[Dict[str, Any]],241 enable_thinking: bool,242 temperature: float,243 output_column: str,244 image_column: str,245 split: str,246) -> str:247 """Create a dataset card documenting the NuExtract3 run."""248 model_name = model.split("/")[-1]249 template_block = ""250 if template is not None:251 template_block = (252 "\n### Extraction Template\n\n```json\n"253 + json.dumps(template, indent=2)254 + "\n```\n"255 )256 257 return f"""---258tags:259- ocr260- structured-extraction261- document-processing262- nuextract3263- markdown264- uv-script265- generated266---267 268# {model_name} on {source_dataset}269 270This dataset contains outputs from [{source_dataset}](https://huggingface.co/datasets/{source_dataset}) processed with [NuExtract3](https://huggingface.co/{model}), a 4B vision-language model for document understanding.271 272## Processing Details273 274- **Source Dataset**: [{source_dataset}](https://huggingface.co/datasets/{source_dataset})275- **Model**: [{model}](https://huggingface.co/{model})276- **Mode**: {mode_label}277- **Number of Samples**: {num_samples:,}278- **Processing Time**: {processing_time}279- **Processing Date**: {datetime.now().strftime("%Y-%m-%d %H:%M UTC")}280 281### Configuration282 283- **Image Column**: `{image_column}`284- **Output Column**: `{output_column}`285- **Dataset Split**: `{split}`286- **Temperature**: {temperature}287- **Thinking Mode**: {"enabled" if enable_thinking else "disabled"}288{template_block}289## Dataset Structure290 291Original columns plus:292- `{output_column}`: NuExtract3 output ({"JSON string" if template else "markdown"})293- `inference_info`: JSON list tracking models applied to this dataset294{"- `" + output_column + "_reasoning`: model's thinking trace (when enabled)" if enable_thinking else ""}295 296Generated with [UV Scripts](https://huggingface.co/uv-scripts)297"""298 299 300def main(301 input_dataset: str,302 output_dataset: str,303 image_column: str = "image",304 batch_size: int = 16,305 max_model_len: int = 16384,306 max_tokens: int = 8192,307 gpu_memory_utilization: float = 0.8,308 mode: str = "markdown",309 template_arg: Optional[str] = None,310 schema_arg: Optional[str] = None,311 enable_thinking: bool = False,312 instructions: Optional[str] = None,313 temperature: Optional[float] = None,314 model: str = MODEL_DEFAULT,315 hf_token: str = None,316 split: str = "train",317 max_samples: int = None,318 private: bool = False,319 shuffle: bool = False,320 seed: int = 42,321 output_column: Optional[str] = None,322 overwrite: bool = False,323 verbose: bool = False,324 config: str = None,325 create_pr: bool = False,326):327 """Process images from an HF dataset through NuExtract3."""328 329 check_cuda_availability()330 start_time = datetime.now()331 332 HF_TOKEN = hf_token or os.environ.get("HF_TOKEN")333 if HF_TOKEN:334 login(token=HF_TOKEN)335 336 template = resolve_template(template_arg, schema_arg)337 extraction_mode = template is not None338 mode_label = "structured-extraction" if extraction_mode else mode339 340 if output_column is None:341 output_column = "extraction" if extraction_mode else "markdown"342 343 if temperature is None:344 temperature = 0.6 if enable_thinking else 0.2345 346 logger.info(f"Using model: {model}")347 logger.info(f"Mode: {mode_label}")348 logger.info(f"Thinking: {enable_thinking} Temperature: {temperature}")349 if extraction_mode:350 logger.info(f"Template: {json.dumps(template, indent=2)}")351 352 logger.info(f"Loading dataset: {input_dataset}")353 dataset = load_dataset(input_dataset, split=split)354 355 if image_column not in dataset.column_names:356 raise ValueError(357 f"Column '{image_column}' not found. Available: {dataset.column_names}"358 )359 360 # Fail fast if an output column would collide with an existing input column.361 # With --enable-thinking the script also writes "{output_column}_reasoning".362 out_cols = [output_column]363 if enable_thinking:364 out_cols.append(f"{output_column}_reasoning")365 dataset = ensure_output_columns_free(dataset, out_cols, overwrite=overwrite)366 367 if shuffle:368 logger.info(f"Shuffling dataset with seed {seed}")369 dataset = dataset.shuffle(seed=seed)370 371 if max_samples:372 dataset = dataset.select(range(min(max_samples, len(dataset))))373 logger.info(f"Limited to {len(dataset)} samples")374 375 logger.info("Initializing vLLM with NuExtract3")376 logger.info("This may take a few minutes on first run...")377 llm = LLM(378 model=model,379 trust_remote_code=True,380 max_model_len=max_model_len,381 gpu_memory_utilization=gpu_memory_utilization,382 limit_mm_per_prompt={"image": 1},383 )384 385 sampling_params = SamplingParams(386 temperature=temperature,387 max_tokens=max_tokens,388 )389 390 chat_template_kwargs: Dict[str, Any] = {"enable_thinking": enable_thinking}391 if extraction_mode:392 chat_template_kwargs["template"] = json.dumps(template, indent=4)393 else:394 chat_template_kwargs["mode"] = mode395 if instructions:396 chat_template_kwargs["instructions"] = instructions397 398 logger.info(f"Processing {len(dataset)} images in batches of {batch_size}")399 logger.info(f"Output will be written to column: {output_column}")400 401 all_outputs: List[str] = []402 all_reasoning: List[Optional[str]] = []403 all_parse_errors: List[bool] = []404 total_batches = (len(dataset) + batch_size - 1) // batch_size405 processed = 0406 407 for batch_num, batch_indices in enumerate(408 partition_all(batch_size, range(len(dataset))), 1409 ):410 batch_indices = list(batch_indices)411 batch_images = [dataset[i][image_column] for i in batch_indices]412 413 logger.info(414 f"Batch {batch_num}/{total_batches} "415 f"({processed}/{len(dataset)} images done)"416 )417 418 try:419 batch_messages = [make_message(img) for img in batch_images]420 outputs = llm.chat(421 batch_messages,422 sampling_params,423 chat_template_kwargs=chat_template_kwargs,424 chat_template_content_format="openai",425 )426 427 for output in outputs:428 raw_text = output.outputs[0].text429 reasoning, answer = split_thinking(raw_text)430 431 if extraction_mode:432 parsed, parse_error = parse_json_output(answer)433 stored = (434 json.dumps(parsed, ensure_ascii=False)435 if parsed is not None436 else answer437 )438 all_outputs.append(stored)439 all_parse_errors.append(parse_error)440 else:441 all_outputs.append(answer)442 all_parse_errors.append(False)443 444 all_reasoning.append(reasoning)445 446 processed += len(batch_images)447 448 except Exception as e:449 logger.error(f"Error processing batch: {e}")450 all_outputs.extend(["[NUEXTRACT3 ERROR]"] * len(batch_images))451 all_reasoning.extend([None] * len(batch_images))452 all_parse_errors.extend([True] * len(batch_images))453 processed += len(batch_images)454 455 processing_duration = datetime.now() - start_time456 processing_time_str = f"{processing_duration.total_seconds() / 60:.1f} min"457 458 logger.info(f"Adding '{output_column}' column to dataset")459 dataset = dataset.add_column(output_column, all_outputs)460 461 if enable_thinking and any(r is not None for r in all_reasoning):462 reasoning_col = f"{output_column}_reasoning"463 logger.info(f"Adding '{reasoning_col}' column to dataset")464 dataset = dataset.add_column(reasoning_col, all_reasoning)465 466 if extraction_mode:467 parse_error_count = sum(all_parse_errors)468 if parse_error_count:469 logger.warning(470 f"{parse_error_count}/{len(all_parse_errors)} extractions failed to parse as JSON"471 )472 473 inference_entry = {474 "model_id": model,475 "model_name": MODEL_NAME,476 "column_name": output_column,477 "timestamp": datetime.now().isoformat(),478 "mode": mode_label,479 "has_template": extraction_mode,480 "enable_thinking": enable_thinking,481 "temperature": temperature,482 "max_tokens": max_tokens,483 }484 if extraction_mode:485 inference_entry["parse_error_rate"] = (486 sum(all_parse_errors) / len(all_parse_errors) if all_parse_errors else 0.0487 )488 489 if "inference_info" in dataset.column_names:490 logger.info("Updating existing inference_info column")491 492 def update_inference_info(example):493 try:494 existing_info = (495 json.loads(example["inference_info"])496 if example["inference_info"]497 else []498 )499 except (json.JSONDecodeError, TypeError):500 existing_info = []501 existing_info.append(inference_entry)502 return {"inference_info": json.dumps(existing_info)}503 504 dataset = dataset.map(update_inference_info)505 else:506 logger.info("Creating new inference_info column")507 inference_list = [json.dumps([inference_entry])] * len(dataset)508 dataset = dataset.add_column("inference_info", inference_list)509 510 logger.info(f"Pushing to {output_dataset}")511 max_retries = 3512 for attempt in range(1, max_retries + 1):513 try:514 if attempt > 1:515 logger.warning("Disabling XET (fallback to HTTP upload)")516 os.environ["HF_HUB_DISABLE_XET"] = "1"517 dataset.push_to_hub(518 output_dataset,519 private=private,520 token=HF_TOKEN,521 max_shard_size="500MB",522 **({"config_name": config} if config else {}),523 create_pr=create_pr,524 commit_message=f"Add {model} {mode_label} results ({len(dataset)} samples)"525 + (f" [{config}]" if config else ""),526 )527 break528 except Exception as e:529 logger.error(f"Upload attempt {attempt}/{max_retries} failed: {e}")530 if attempt < max_retries:531 delay = 30 * (2 ** (attempt - 1))532 logger.info(f"Retrying in {delay}s...")533 time.sleep(delay)534 else:535 logger.error("All upload attempts failed. Results are lost.")536 sys.exit(1)537 538 logger.info("Creating dataset card")539 card_content = create_dataset_card(540 source_dataset=input_dataset,541 model=model,542 num_samples=len(dataset),543 processing_time=processing_time_str,544 mode_label=mode_label,545 template=template,546 enable_thinking=enable_thinking,547 temperature=temperature,548 output_column=output_column,549 image_column=image_column,550 split=split,551 )552 card = DatasetCard(card_content)553 card.push_to_hub(output_dataset, token=HF_TOKEN)554 555 logger.info("Done! NuExtract3 processing complete.")556 logger.info(557 f"Dataset available at: https://huggingface.co/datasets/{output_dataset}"558 )559 logger.info(f"Processing time: {processing_time_str}")560 logger.info(561 f"Processing speed: {len(dataset) / processing_duration.total_seconds():.2f} images/sec"562 )563 564 if verbose:565 import importlib.metadata566 567 logger.info("--- Resolved package versions ---")568 for pkg in [569 "vllm",570 "transformers",571 "torch",572 "datasets",573 "pyarrow",574 "pillow",575 "numind",576 ]:577 try:578 logger.info(f" {pkg}=={importlib.metadata.version(pkg)}")579 except importlib.metadata.PackageNotFoundError:580 logger.info(f" {pkg}: not installed")581 logger.info("--- End versions ---")582 583 584if __name__ == "__main__":585 if len(sys.argv) == 1:586 print("=" * 70)587 print("NuExtract3 - Document-to-Markdown + Structured Extraction (4B)")588 print("=" * 70)589 print("\nModes:")590 print(" markdown - Image -> markdown (default)")591 print(" content - Image -> plain content")592 print(" --template / --schema - Image -> JSON shaped like the template")593 print("\nExamples:")594 print("\n1. Markdown OCR:")595 print(" uv run --with vllm==0.29.0 nuextract3.py input-dataset output-dataset")596 print("\n2. Structured extraction with an inline template:")597 print(" uv run --with vllm==0.29.0 nuextract3.py input output \\")598 print(' --template \'{"title": "verbatim-string", "date": "date"}\'')599 print("\n3. Structured extraction from a JSON Schema (e.g. Pydantic):")600 print(" uv run --with vllm==0.29.0 nuextract3.py input output --schema schema.json")601 print("\n (--template / --schema also accept a URL or a local file path)")602 print("\n4. Reasoning mode for harder documents:")603 print(" uv run --with vllm==0.29.0 nuextract3.py input output --enable-thinking")604 print("\n5. Test with 10 samples:")605 print(" uv run --with vllm==0.29.0 nuextract3.py large-ds test --max-samples 10 --shuffle")606 print("\n6. Running on HF Jobs (image/flavor/secrets from the script header, hf 1.32+):")607 print(" hf jobs uv run \\")608 print(609 " https://huggingface.co/datasets/uv-scripts/ocr/raw/main/nuextract3.py \\"610 )611 print(" input-dataset output-dataset --batch-size 16")612 print("\nFor full help: uv run --with vllm==0.29.0 nuextract3.py --help")613 sys.exit(0)614 615 parser = argparse.ArgumentParser(616 description="NuExtract3: document-to-markdown + schema-guided JSON extraction (4B VLM)",617 formatter_class=argparse.RawDescriptionHelpFormatter,618 epilog="""619Modes:620 (default) Markdown OCR (image -> clean markdown)621 --mode content622 Plain-content extraction (less structured than markdown)623 --template PATH_OR_JSON624 Structured extraction with a NuExtract template625 --schema PATH_OR_JSON626 Structured extraction from a JSON Schema627 (e.g. Pydantic Model.model_json_schema())628 629Examples:630 uv run --with vllm==0.29.0 nuextract3.py my-docs analyzed-docs631 uv run --with vllm==0.29.0 nuextract3.py receipts extracted \\632 --template '{"store": "verbatim-string", "total": "number"}'633 uv run --with vllm==0.29.0 nuextract3.py contracts extracted --schema contract_schema.json634 uv run --with vllm==0.29.0 nuextract3.py hard-docs out --enable-thinking635 """,636 )637 638 parser.add_argument("input_dataset", help="Input dataset ID from Hugging Face Hub")639 parser.add_argument("output_dataset", help="Output dataset ID for Hugging Face Hub")640 parser.add_argument(641 "--image-column",642 default="image",643 help="Column containing images (default: image)",644 )645 parser.add_argument(646 "--batch-size",647 type=int,648 default=16,649 help="Batch size for processing (default: 16)",650 )651 parser.add_argument(652 "--max-model-len",653 type=int,654 default=16384,655 help="Maximum model context length (default: 16384)",656 )657 parser.add_argument(658 "--max-tokens",659 type=int,660 default=8192,661 help="Maximum tokens to generate (default: 8192)",662 )663 parser.add_argument(664 "--gpu-memory-utilization",665 type=float,666 default=0.8,667 help="GPU memory utilization (default: 0.8)",668 )669 parser.add_argument(670 "--mode",671 choices=["markdown", "content"],672 default="markdown",673 help="OCR mode when no template/schema is given (default: markdown)",674 )675 parser.add_argument(676 "--template",677 help="NuExtract template: inline JSON, a URL, or a file path",678 )679 parser.add_argument(680 "--schema",681 help="JSON Schema to auto-convert: inline JSON, a URL, or a file path",682 )683 parser.add_argument(684 "--enable-thinking",685 action="store_true",686 help="Enable reasoning mode (slower, better on hard documents)",687 )688 parser.add_argument(689 "--instructions",690 default=None,691 help=(692 "Free-text instructions passed to NuExtract via "693 "chat_template_kwargs.instructions (e.g. routing guidance across "694 "optional schema branches, output-format rules). "695 "See https://huggingface.co/numind/NuExtract3#instructions"696 ),697 )698 parser.add_argument(699 "--temperature",700 type=float,701 default=None,702 help="Sampling temperature (default: 0.2 non-thinking, 0.6 thinking)",703 )704 parser.add_argument(705 "--model",706 default=MODEL_DEFAULT,707 help=f"Model ID (default: {MODEL_DEFAULT})",708 )709 parser.add_argument("--hf-token", help="Hugging Face API token")710 parser.add_argument(711 "--split", default="train", help="Dataset split to use (default: train)"712 )713 parser.add_argument(714 "--max-samples",715 type=int,716 help="Maximum number of samples to process (for testing)",717 )718 parser.add_argument(719 "--private", action="store_true", help="Make output dataset private"720 )721 parser.add_argument(722 "--config",723 help="Config/subset name when pushing to Hub (for benchmarking multiple models in one repo)",724 )725 parser.add_argument(726 "--create-pr",727 action="store_true",728 help="Create a pull request instead of pushing directly (for parallel benchmarking)",729 )730 parser.add_argument(731 "--shuffle", action="store_true", help="Shuffle dataset before processing"732 )733 parser.add_argument(734 "--seed",735 type=int,736 default=42,737 help="Random seed for shuffling (default: 42)",738 )739 parser.add_argument(740 "--output-column",741 default=None,742 help="Column name for output (default: 'markdown' in OCR mode, 'extraction' in template mode)",743 )744 parser.add_argument(745 "--overwrite",746 action="store_true",747 help="Replace the output column if it already exists in the input dataset "748 "(default: error out to avoid clobbering an existing column).",749 )750 parser.add_argument(751 "--verbose",752 action="store_true",753 help="Log resolved package versions after processing",754 )755 756 args = parser.parse_args()757 758 main(759 input_dataset=args.input_dataset,760 output_dataset=args.output_dataset,761 image_column=args.image_column,762 batch_size=args.batch_size,763 max_model_len=args.max_model_len,764 max_tokens=args.max_tokens,765 gpu_memory_utilization=args.gpu_memory_utilization,766 mode=args.mode,767 template_arg=args.template,768 schema_arg=args.schema,769 enable_thinking=args.enable_thinking,770 instructions=args.instructions,771 temperature=args.temperature,772 model=args.model,773 hf_token=args.hf_token,774 split=args.split,775 max_samples=args.max_samples,776 private=args.private,777 shuffle=args.shuffle,778 seed=args.seed,779 output_column=args.output_column,780 overwrite=args.overwrite,781 verbose=args.verbose,782 config=args.config,783 create_pr=args.create_pr,784 )785 