Team Ai
Apppublic

launch-calcium/DatasetView.Typescript

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
app.py193 linesDownload Raw Back to root
1import gradio as gr2import pandas as pd3import requests4 5API_BASE = "https://datasets-server.huggingface.co"6 7 8def fetch(endpoint, params):9    try:10        response = requests.get(f"{API_BASE}/{endpoint}", params=params, timeout=30)11        response.raise_for_status()12        return response.json()13    except Exception as exc:14        return {"error": str(exc)}15 16 17def get_splits(dataset):18    dataset = (dataset or "").strip()19    if not dataset:20        return (21            pd.DataFrame(),22            gr.update(choices=[], value=None),23            "⚠️ Please enter a dataset name.",24        )25 26    data = fetch("splits", {"dataset": dataset})27    if "error" in data:28        return (29            pd.DataFrame(),30            gr.update(choices=[], value=None),31            f"❌ {data['error']}",32        )33 34    splits = data.get("splits", [])35    if not splits:36        return (37            pd.DataFrame(),38            gr.update(choices=[], value=None),39            "No splits found.",40        )41 42    df = pd.DataFrame(splits)43    columns = [c for c in ["dataset", "config", "split"] if c in df.columns]44    if columns:45        df = df[columns]46 47    choices = [f"{s.get('config', '')} :: {s.get('split', '')}" for s in splits]48 49    return (50        df,51        gr.update(choices=choices, value=choices[0] if choices else None),52        f"✅ Found {len(splits)} splits.",53    )54 55 56def get_rows(dataset, config_split, num_rows):57    dataset = (dataset or "").strip()58    if not dataset:59        return pd.DataFrame(), pd.DataFrame(), "⚠️ Please enter a dataset name."60 61    config = None62    split = None63 64    if config_split:65        if "::" in config_split:66            config, split = [x.strip() for x in config_split.split("::", 1)]67        else:68            split = config_split.strip()69 70    # Auto-fill missing config/split info using the splits endpoint71    if not config or not split:72        data = fetch("splits", {"dataset": dataset})73        if "error" in data:74            return pd.DataFrame(), pd.DataFrame(), f"❌ {data['error']}"75 76        splits = data.get("splits", [])77        if not splits:78            return pd.DataFrame(), pd.DataFrame(), "No splits found."79 80        if not split:81            if config:82                matching = [s for s in splits if s.get("config") == config]83                if not matching:84                    return pd.DataFrame(), pd.DataFrame(), f"❌ Config '{config}' not found."85                split = matching[0].get("split", "")86            else:87                config = splits[0].get("config", "")88                split = splits[0].get("split", "")89        else:90            for s in splits:91                if s.get("split") == split and (not config or s.get("config") == config):92                    config = s.get("config", "")93                    break94            else:95                return pd.DataFrame(), pd.DataFrame(), f"❌ Split '{split}' not found."96 97    data = fetch(98        "rows",99        {100            "dataset": dataset,101            "config": config,102            "split": split,103            "offset": 0,104            "length": int(num_rows),105        },106    )107 108    if "error" in data:109        return pd.DataFrame(), pd.DataFrame(), f"❌ {data['error']}"110 111    # Features / schema112    features = data.get("features", [])113    features_df = pd.DataFrame(features)114    if not features_df.empty and "type" in features_df.columns:115        features_df["type"] = features_df["type"].astype(str)116 117    # Rows118    rows = data.get("rows", [])119    if not rows:120        return features_df, pd.DataFrame(), "No rows returned."121 122    records = [r.get("row", {}) for r in rows]123    rows_df = pd.DataFrame(records)124 125    if rows and "row_idx" in rows[0]:126        rows_df.index = [r.get("row_idx", i) for i, r in enumerate(rows)]127        rows_df.index.name = "row_idx"128 129    return (130        features_df,131        rows_df,132        f"✅ Loaded {len(rows)} rows from `{config}` / `{split}`.",133    )134 135 136with gr.Blocks(title="HF Dataset Viewer") as demo:137    gr.Markdown(138        """139        # 🧐 Hugging Face Dataset Viewer140 141        Browse any public Hugging Face dataset through the142        [datasets-server API](https://datasets-server.huggingface.co).143        No dataset download needed.144        """145    )146 147    with gr.Row():148        dataset_text = gr.Textbox(149            label="Dataset name",150            value="imdb",151            placeholder="e.g. emotions, glue, squad",152        )153        splits_button = gr.Button("Get Splits")154 155    splits_table = gr.Dataframe(label="Splits", interactive=False)156    splits_status = gr.Markdown()157 158    with gr.Row():159        config_split_dropdown = gr.Dropdown(160            label="Config / Split",161            choices=[],162            interactive=True,163        )164        num_rows_slider = gr.Slider(165            minimum=1,166            maximum=100,167            value=10,168            step=1,169            label="Number of rows",170        )171        rows_button = gr.Button("Load Rows")172 173    with gr.Row():174        features_table = gr.Dataframe(label="Features", interactive=False)175        rows_table = gr.Dataframe(label="Rows", interactive=False)176 177    rows_status = gr.Markdown()178 179    splits_button.click(180        get_splits,181        inputs=dataset_text,182        outputs=[splits_table, config_split_dropdown, splits_status],183    )184 185    rows_button.click(186        get_rows,187        inputs=[dataset_text, config_split_dropdown, num_rows_slider],188        outputs=[features_table, rows_table, rows_status],189    )190 191 192if __name__ == "__main__":193    demo.launch()