launch-calcium/DatasetView.Typescript
0
1import gradio as gr2import pandas as pd3import requests4 5API_BASE = "https://datasets-server.huggingface.co"6 7 8def fetch(endpoint, params):9 try:10 response = requests.get(f"{API_BASE}/{endpoint}", params=params, timeout=30)11 response.raise_for_status()12 return response.json()13 except Exception as exc:14 return {"error": str(exc)}15 16 17def get_splits(dataset):18 dataset = (dataset or "").strip()19 if not dataset:20 return (21 pd.DataFrame(),22 gr.update(choices=[], value=None),23 "⚠️ Please enter a dataset name.",24 )25 26 data = fetch("splits", {"dataset": dataset})27 if "error" in data:28 return (29 pd.DataFrame(),30 gr.update(choices=[], value=None),31 f"❌ {data['error']}",32 )33 34 splits = data.get("splits", [])35 if not splits:36 return (37 pd.DataFrame(),38 gr.update(choices=[], value=None),39 "No splits found.",40 )41 42 df = pd.DataFrame(splits)43 columns = [c for c in ["dataset", "config", "split"] if c in df.columns]44 if columns:45 df = df[columns]46 47 choices = [f"{s.get('config', '')} :: {s.get('split', '')}" for s in splits]48 49 return (50 df,51 gr.update(choices=choices, value=choices[0] if choices else None),52 f"✅ Found {len(splits)} splits.",53 )54 55 56def get_rows(dataset, config_split, num_rows):57 dataset = (dataset or "").strip()58 if not dataset:59 return pd.DataFrame(), pd.DataFrame(), "⚠️ Please enter a dataset name."60 61 config = None62 split = None63 64 if config_split:65 if "::" in config_split:66 config, split = [x.strip() for x in config_split.split("::", 1)]67 else:68 split = config_split.strip()69 70 # Auto-fill missing config/split info using the splits endpoint71 if not config or not split:72 data = fetch("splits", {"dataset": dataset})73 if "error" in data:74 return pd.DataFrame(), pd.DataFrame(), f"❌ {data['error']}"75 76 splits = data.get("splits", [])77 if not splits:78 return pd.DataFrame(), pd.DataFrame(), "No splits found."79 80 if not split:81 if config:82 matching = [s for s in splits if s.get("config") == config]83 if not matching:84 return pd.DataFrame(), pd.DataFrame(), f"❌ Config '{config}' not found."85 split = matching[0].get("split", "")86 else:87 config = splits[0].get("config", "")88 split = splits[0].get("split", "")89 else:90 for s in splits:91 if s.get("split") == split and (not config or s.get("config") == config):92 config = s.get("config", "")93 break94 else:95 return pd.DataFrame(), pd.DataFrame(), f"❌ Split '{split}' not found."96 97 data = fetch(98 "rows",99 {100 "dataset": dataset,101 "config": config,102 "split": split,103 "offset": 0,104 "length": int(num_rows),105 },106 )107 108 if "error" in data:109 return pd.DataFrame(), pd.DataFrame(), f"❌ {data['error']}"110 111 # Features / schema112 features = data.get("features", [])113 features_df = pd.DataFrame(features)114 if not features_df.empty and "type" in features_df.columns:115 features_df["type"] = features_df["type"].astype(str)116 117 # Rows118 rows = data.get("rows", [])119 if not rows:120 return features_df, pd.DataFrame(), "No rows returned."121 122 records = [r.get("row", {}) for r in rows]123 rows_df = pd.DataFrame(records)124 125 if rows and "row_idx" in rows[0]:126 rows_df.index = [r.get("row_idx", i) for i, r in enumerate(rows)]127 rows_df.index.name = "row_idx"128 129 return (130 features_df,131 rows_df,132 f"✅ Loaded {len(rows)} rows from `{config}` / `{split}`.",133 )134 135 136with gr.Blocks(title="HF Dataset Viewer") as demo:137 gr.Markdown(138 """139 # 🧐 Hugging Face Dataset Viewer140 141 Browse any public Hugging Face dataset through the142 [datasets-server API](https://datasets-server.huggingface.co).143 No dataset download needed.144 """145 )146 147 with gr.Row():148 dataset_text = gr.Textbox(149 label="Dataset name",150 value="imdb",151 placeholder="e.g. emotions, glue, squad",152 )153 splits_button = gr.Button("Get Splits")154 155 splits_table = gr.Dataframe(label="Splits", interactive=False)156 splits_status = gr.Markdown()157 158 with gr.Row():159 config_split_dropdown = gr.Dropdown(160 label="Config / Split",161 choices=[],162 interactive=True,163 )164 num_rows_slider = gr.Slider(165 minimum=1,166 maximum=100,167 value=10,168 step=1,169 label="Number of rows",170 )171 rows_button = gr.Button("Load Rows")172 173 with gr.Row():174 features_table = gr.Dataframe(label="Features", interactive=False)175 rows_table = gr.Dataframe(label="Rows", interactive=False)176 177 rows_status = gr.Markdown()178 179 splits_button.click(180 get_splits,181 inputs=dataset_text,182 outputs=[splits_table, config_split_dropdown, splits_status],183 )184 185 rows_button.click(186 get_rows,187 inputs=[dataset_text, config_split_dropdown, num_rows_slider],188 outputs=[features_table, rows_table, rows_status],189 )190 191 192if __name__ == "__main__":193 demo.launch()