Team Ai
Datasetpublic

uv-scripts/marimo

Marimo UV Scripts Marimo notebooks that work as both interactive tutorials and batch scripts. What is this? Marimo notebooks are pure Python files that can be: Edited interactively with a reactive notebook interface Run as scripts with uv run - same as any UV script This makes them perfect for tutorials and educational content where you want users to explore step-by-step, but also run the whole thing as a batch job. Available Scripts Script… See the full description on the dataset page: https://huggingface.co/datasets/uv-scripts/marimo.

sourceHugging Faceupdated 8mo agoView on Hugging Face
0likes30downloads
getting-started.py157 linesDownload Raw Back to root
1# /// script2# requires-python = ">=3.10"3# dependencies = [4#     "marimo",5#     "datasets",6#     "huggingface-hub",7# ]8# ///9"""10Getting Started with Hugging Face Datasets11 12This marimo notebook works in two modes:13- Interactive: uvx marimo edit --sandbox getting-started.py14- Script: uv run getting-started.py --dataset squad15 16Same file, two experiences.17"""18 19import marimo20 21app = marimo.App(width="medium")22 23 24@app.cell25def _():26    import marimo as mo27    return (mo,)28 29 30@app.cell31def _(mo):32    mo.md(33        """34        # Getting Started with Hugging Face Datasets35 36        This notebook shows how to load and explore datasets from the Hugging Face Hub.37 38        **Run this notebook:**39        - Interactive: `uvx marimo edit --sandbox getting-started.py`40        - As a script: `uv run getting-started.py --dataset squad`41        """42    )43    return44 45 46@app.cell47def _(mo):48    mo.md(49        """50        ## Step 1: Configure51 52        Choose which dataset to load. In interactive mode, use the controls below.53        In script mode, pass `--dataset` argument.54        """55    )56    return57 58 59@app.cell60def _(mo):61    import argparse62 63    # Parse CLI args (works in both modes)64    parser = argparse.ArgumentParser()65    parser.add_argument("--dataset", default="stanfordnlp/imdb")66    parser.add_argument("--split", default="train")67    parser.add_argument("--samples", type=int, default=5)68    args, _ = parser.parse_known_args()69 70    # Interactive controls (only shown in notebook mode)71    dataset_input = mo.ui.text(value=args.dataset, label="Dataset")72    split_input = mo.ui.dropdown(["train", "test", "validation"], value=args.split, label="Split")73    samples_input = mo.ui.slider(1, 20, value=args.samples, label="Samples")74 75    mo.hstack([dataset_input, split_input, samples_input])76    return args, argparse, dataset_input, parser, samples_input, split_input77 78 79@app.cell80def _(args, dataset_input, mo, samples_input, split_input):81    # Use interactive values if available, otherwise CLI args82    dataset_name = dataset_input.value or args.dataset83    split_name = split_input.value or args.split84    num_samples = samples_input.value or args.samples85 86    print(f"Dataset: {dataset_name}, Split: {split_name}, Samples: {num_samples}")87    return dataset_name, num_samples, split_name88 89 90@app.cell91def _(mo):92    mo.md(93        """94        ## Step 2: Load Dataset95 96        We use the `datasets` library to stream data directly from the Hub.97        No need to download the entire dataset first!98        """99    )100    return101 102 103@app.cell104def _(dataset_name, split_name):105    from datasets import load_dataset106 107    print(f"Loading {dataset_name}...")108    dataset = load_dataset(dataset_name, split=split_name)109    print(f"Loaded {len(dataset):,} rows")110    print(f"Features: {list(dataset.features.keys())}")111    return dataset, load_dataset112 113 114@app.cell115def _(mo):116    mo.md(117        """118        ## Step 3: Explore the Data119 120        Let's look at a few samples from the dataset.121        """122    )123    return124 125 126@app.cell127def _(dataset, mo, num_samples):128    # Select samples and display129    samples = dataset.select(range(min(num_samples, len(dataset))))130    df = samples.to_pandas()131 132    # Truncate long text for display133    for col in df.select_dtypes(include=["object"]).columns:134        df[col] = df[col].apply(lambda x: str(x)[:200] + "..." if len(str(x)) > 200 else x)135 136    print(df.to_string())  # Shows in script mode137    mo.ui.table(df)  # Shows in interactive mode138    return df, samples139 140 141@app.cell142def _(mo):143    mo.md(144        """145        ## Next Steps146 147        - Try different datasets: `squad`, `emotion`, `wikitext`148        - Run on HF Jobs: `hf jobs uv run --flavor cpu-basic ... getting-started.py`149        - Check out more UV scripts at [uv-scripts](https://huggingface.co/uv-scripts)150        """151    )152    return153 154 155if __name__ == "__main__":156    app.run()157