codekingpro/portable-devtools
114k
1"""CLI commands for Hugging Face Inference Endpoints."""2 3from typing import Annotated4 5import typer6 7from huggingface_hub._inference_endpoints import InferenceEndpointScalingMetric8from huggingface_hub.errors import HfHubHTTPError9 10from ._cli_utils import TokenOpt, get_hf_api, typer_factory11from ._output import out12 13 14ie_cli = typer_factory(help="Manage Hugging Face Inference Endpoints.")15 16catalog_app = typer_factory(help="Interact with the Inference Endpoints catalog.")17 18 19NameArg = Annotated[20 str,21 typer.Argument(help="Endpoint name."),22]23NameOpt = Annotated[24 str | None,25 typer.Option(help="Endpoint name."),26]27 28NamespaceOpt = Annotated[29 str | None,30 typer.Option(31 help="The namespace associated with the Inference Endpoint. Defaults to the current user's namespace.",32 ),33]34 35 36@ie_cli.command("list | ls", examples=["hf endpoints ls", "hf endpoints ls --namespace my-org"])37def ls(38 namespace: NamespaceOpt = None,39 token: TokenOpt = None,40) -> None:41 """Lists all Inference Endpoints for the given namespace."""42 api = get_hf_api(token=token)43 try:44 endpoints = api.list_inference_endpoints(namespace=namespace, token=token)45 except HfHubHTTPError as error:46 out.error(f"Listing failed: {error}")47 raise typer.Exit(code=error.response.status_code) from error48 49 results = []50 for endpoint in endpoints:51 raw = endpoint.raw52 status = raw.get("status", {})53 model = raw.get("model", {})54 compute = raw.get("compute", {})55 provider = raw.get("provider", {})56 results.append(57 {58 "name": raw.get("name", ""),59 "model": model.get("repository", "") if isinstance(model, dict) else "",60 "status": status.get("state", "") if isinstance(status, dict) else "",61 "task": model.get("task", "") if isinstance(model, dict) else "",62 "framework": model.get("framework", "") if isinstance(model, dict) else "",63 "instance": compute.get("instanceType", "") if isinstance(compute, dict) else "",64 "vendor": provider.get("vendor", "") if isinstance(provider, dict) else "",65 "region": provider.get("region", "") if isinstance(provider, dict) else "",66 }67 )68 out.table(results, id_key="name")69 70 71@ie_cli.command(name="deploy", examples=["hf endpoints deploy my-endpoint --repo gpt2 --framework pytorch ..."])72def deploy(73 name: NameArg,74 repo: Annotated[75 str,76 typer.Option(77 help="The name of the model repository associated with the Inference Endpoint (e.g. 'openai/gpt-oss-120b').",78 ),79 ],80 framework: Annotated[81 str,82 typer.Option(83 help="The machine learning framework used for the model (e.g. 'vllm').",84 ),85 ],86 accelerator: Annotated[87 str,88 typer.Option(89 help="The hardware accelerator to be used for inference (e.g. 'cpu').",90 ),91 ],92 instance_size: Annotated[93 str,94 typer.Option(95 help="The size or type of the instance to be used for hosting the model (e.g. 'x4').",96 ),97 ],98 instance_type: Annotated[99 str,100 typer.Option(101 help="The cloud instance type where the Inference Endpoint will be deployed (e.g. 'intel-icl').",102 ),103 ],104 region: Annotated[105 str,106 typer.Option(107 help="The cloud region in which the Inference Endpoint will be created (e.g. 'us-east-1').",108 ),109 ],110 vendor: Annotated[111 str,112 typer.Option(113 help="The cloud provider or vendor where the Inference Endpoint will be hosted (e.g. 'aws').",114 ),115 ],116 *,117 namespace: NamespaceOpt = None,118 task: Annotated[119 str | None,120 typer.Option(121 help="The task on which to deploy the model (e.g. 'text-classification').",122 ),123 ] = None,124 token: TokenOpt = None,125 min_replica: Annotated[126 int,127 typer.Option(128 help="The minimum number of replicas (instances) to keep running for the Inference Endpoint.",129 ),130 ] = 1,131 max_replica: Annotated[132 int,133 typer.Option(134 help="The maximum number of replicas (instances) to scale to for the Inference Endpoint.",135 ),136 ] = 1,137 scale_to_zero_timeout: Annotated[138 int | None,139 typer.Option(140 help="The duration in minutes before an inactive endpoint is scaled to zero.",141 ),142 ] = None,143 scaling_metric: Annotated[144 InferenceEndpointScalingMetric | None,145 typer.Option(146 help="The metric reference for scaling.",147 ),148 ] = None,149 scaling_threshold: Annotated[150 float | None,151 typer.Option(152 help="The scaling metric threshold used to trigger a scale up. Ignored when scaling metric is not provided.",153 ),154 ] = None,155) -> None:156 """Deploy an Inference Endpoint from a Hub repository."""157 api = get_hf_api(token=token)158 endpoint = api.create_inference_endpoint(159 name=name,160 repository=repo,161 framework=framework,162 accelerator=accelerator,163 instance_size=instance_size,164 instance_type=instance_type,165 region=region,166 vendor=vendor,167 namespace=namespace,168 task=task,169 token=token,170 min_replica=min_replica,171 max_replica=max_replica,172 scaling_metric=scaling_metric,173 scaling_threshold=scaling_threshold,174 scale_to_zero_timeout=scale_to_zero_timeout,175 )176 out.dict(endpoint.raw)177 178 179@catalog_app.command(name="deploy", examples=["hf endpoints catalog deploy --repo meta-llama/Llama-3.2-1B-Instruct"])180def deploy_from_catalog(181 repo: Annotated[182 str,183 typer.Option(184 help="The name of the model repository associated with the Inference Endpoint (e.g. 'openai/gpt-oss-120b').",185 ),186 ],187 name: NameOpt = None,188 accelerator: Annotated[189 str | None,190 typer.Option(191 help="The hardware accelerator to be used for inference (e.g. 'cpu', 'gpu', 'neuron').",192 ),193 ] = None,194 namespace: NamespaceOpt = None,195 token: TokenOpt = None,196) -> None:197 """Deploy an Inference Endpoint from the Model Catalog."""198 api = get_hf_api(token=token)199 try:200 endpoint = api.create_inference_endpoint_from_catalog(201 repo_id=repo,202 name=name,203 accelerator=accelerator,204 namespace=namespace,205 token=token,206 )207 except HfHubHTTPError as error:208 out.error(f"Deployment failed: {error}")209 raise typer.Exit(code=error.response.status_code) from error210 211 out.dict(endpoint.raw)212 213 214def list_catalog(215 token: TokenOpt = None,216) -> None:217 """List available Catalog models."""218 api = get_hf_api(token=token)219 try:220 models = api.list_inference_catalog(token=token)221 except HfHubHTTPError as error:222 out.error(f"Catalog fetch failed: {error}")223 raise typer.Exit(code=error.response.status_code) from error224 225 out.dict({"models": models})226 227 228catalog_app.command(name="list | ls", examples=["hf endpoints catalog ls"])(list_catalog)229ie_cli.command(name="list-catalog", hidden=True)(list_catalog)230 231 232ie_cli.add_typer(catalog_app, name="catalog")233 234 235@ie_cli.command(examples=["hf endpoints describe my-endpoint"])236def describe(237 name: NameArg,238 namespace: NamespaceOpt = None,239 token: TokenOpt = None,240) -> None:241 """Get information about an existing endpoint."""242 api = get_hf_api(token=token)243 try:244 endpoint = api.get_inference_endpoint(name=name, namespace=namespace, token=token)245 except HfHubHTTPError as error:246 out.error(f"Fetch failed: {error}")247 raise typer.Exit(code=error.response.status_code) from error248 249 out.dict(endpoint.raw)250 251 252@ie_cli.command(examples=["hf endpoints update my-endpoint --min-replica 2"])253def update(254 name: NameArg,255 namespace: NamespaceOpt = None,256 repo: Annotated[257 str | None,258 typer.Option(259 help="The name of the model repository associated with the Inference Endpoint (e.g. 'openai/gpt-oss-120b').",260 ),261 ] = None,262 accelerator: Annotated[263 str | None,264 typer.Option(265 help="The hardware accelerator to be used for inference (e.g. 'cpu').",266 ),267 ] = None,268 instance_size: Annotated[269 str | None,270 typer.Option(271 help="The size or type of the instance to be used for hosting the model (e.g. 'x4').",272 ),273 ] = None,274 instance_type: Annotated[275 str | None,276 typer.Option(277 help="The cloud instance type where the Inference Endpoint will be deployed (e.g. 'intel-icl').",278 ),279 ] = None,280 framework: Annotated[281 str | None,282 typer.Option(283 help="The machine learning framework used for the model (e.g. 'custom').",284 ),285 ] = None,286 revision: Annotated[287 str | None,288 typer.Option(289 help="The specific model revision to deploy on the Inference Endpoint (e.g. '6c0e6080953db56375760c0471a8c5f2929baf11').",290 ),291 ] = None,292 task: Annotated[293 str | None,294 typer.Option(295 help="The task on which to deploy the model (e.g. 'text-classification').",296 ),297 ] = None,298 min_replica: Annotated[299 int | None,300 typer.Option(301 help="The minimum number of replicas (instances) to keep running for the Inference Endpoint.",302 ),303 ] = None,304 max_replica: Annotated[305 int | None,306 typer.Option(307 help="The maximum number of replicas (instances) to scale to for the Inference Endpoint.",308 ),309 ] = None,310 scale_to_zero_timeout: Annotated[311 int | None,312 typer.Option(313 help="The duration in minutes before an inactive endpoint is scaled to zero.",314 ),315 ] = None,316 scaling_metric: Annotated[317 InferenceEndpointScalingMetric | None,318 typer.Option(319 help="The metric reference for scaling.",320 ),321 ] = None,322 scaling_threshold: Annotated[323 float | None,324 typer.Option(325 help="The scaling metric threshold used to trigger a scale up. Ignored when scaling metric is not provided.",326 ),327 ] = None,328 token: TokenOpt = None,329) -> None:330 """Update an existing endpoint."""331 api = get_hf_api(token=token)332 try:333 endpoint = api.update_inference_endpoint(334 name=name,335 namespace=namespace,336 repository=repo,337 framework=framework,338 revision=revision,339 task=task,340 accelerator=accelerator,341 instance_size=instance_size,342 instance_type=instance_type,343 min_replica=min_replica,344 max_replica=max_replica,345 scale_to_zero_timeout=scale_to_zero_timeout,346 scaling_metric=scaling_metric,347 scaling_threshold=scaling_threshold,348 token=token,349 )350 except HfHubHTTPError as error:351 out.error(f"Update failed: {error}")352 raise typer.Exit(code=error.response.status_code) from error353 out.dict(endpoint.raw)354 355 356@ie_cli.command(examples=["hf endpoints delete my-endpoint"])357def delete(358 name: NameArg,359 namespace: NamespaceOpt = None,360 yes: Annotated[361 bool,362 typer.Option("--yes", help="Skip confirmation prompts."),363 ] = False,364 token: TokenOpt = None,365) -> None:366 """Delete an Inference Endpoint permanently."""367 out.confirm(f"Delete endpoint '{name}'?", yes=yes)368 369 api = get_hf_api(token=token)370 try:371 api.delete_inference_endpoint(name=name, namespace=namespace, token=token)372 except HfHubHTTPError as error:373 out.error(f"Delete failed: {error}")374 raise typer.Exit(code=error.response.status_code) from error375 376 out.result(f"Deleted '{name}'.", name=name)377 378 379@ie_cli.command(examples=["hf endpoints pause my-endpoint"])380def pause(381 name: NameArg,382 namespace: NamespaceOpt = None,383 token: TokenOpt = None,384) -> None:385 """Pause an Inference Endpoint."""386 api = get_hf_api(token=token)387 try:388 endpoint = api.pause_inference_endpoint(name=name, namespace=namespace, token=token)389 except HfHubHTTPError as error:390 out.error(f"Pause failed: {error}")391 raise typer.Exit(code=error.response.status_code) from error392 393 out.dict(endpoint.raw)394 395 396@ie_cli.command(examples=["hf endpoints resume my-endpoint"])397def resume(398 name: NameArg,399 namespace: NamespaceOpt = None,400 fail_if_already_running: Annotated[401 bool,402 typer.Option(403 "--fail-if-already-running",404 help="If `True`, the method will raise an error if the Inference Endpoint is already running.",405 ),406 ] = False,407 token: TokenOpt = None,408) -> None:409 """Resume an Inference Endpoint."""410 api = get_hf_api(token=token)411 try:412 endpoint = api.resume_inference_endpoint(413 name=name,414 namespace=namespace,415 token=token,416 running_ok=not fail_if_already_running,417 )418 except HfHubHTTPError as error:419 out.error(f"Resume failed: {error}")420 raise typer.Exit(code=error.response.status_code) from error421 out.dict(endpoint.raw)422 423 424@ie_cli.command(examples=["hf endpoints scale-to-zero my-endpoint"])425def scale_to_zero(426 name: NameArg,427 namespace: NamespaceOpt = None,428 token: TokenOpt = None,429) -> None:430 """Scale an Inference Endpoint to zero."""431 api = get_hf_api(token=token)432 try:433 endpoint = api.scale_to_zero_inference_endpoint(name=name, namespace=namespace, token=token)434 except HfHubHTTPError as error:435 out.error(f"Scale To Zero failed: {error}")436 raise typer.Exit(code=error.response.status_code) from error437 438 out.dict(endpoint.raw)439 