Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
inference_endpoints.py439 linesDownload Raw Back to cli
1"""CLI commands for Hugging Face Inference Endpoints."""2 3from typing import Annotated4 5import typer6 7from huggingface_hub._inference_endpoints import InferenceEndpointScalingMetric8from huggingface_hub.errors import HfHubHTTPError9 10from ._cli_utils import TokenOpt, get_hf_api, typer_factory11from ._output import out12 13 14ie_cli = typer_factory(help="Manage Hugging Face Inference Endpoints.")15 16catalog_app = typer_factory(help="Interact with the Inference Endpoints catalog.")17 18 19NameArg = Annotated[20    str,21    typer.Argument(help="Endpoint name."),22]23NameOpt = Annotated[24    str | None,25    typer.Option(help="Endpoint name."),26]27 28NamespaceOpt = Annotated[29    str | None,30    typer.Option(31        help="The namespace associated with the Inference Endpoint. Defaults to the current user's namespace.",32    ),33]34 35 36@ie_cli.command("list | ls", examples=["hf endpoints ls", "hf endpoints ls --namespace my-org"])37def ls(38    namespace: NamespaceOpt = None,39    token: TokenOpt = None,40) -> None:41    """Lists all Inference Endpoints for the given namespace."""42    api = get_hf_api(token=token)43    try:44        endpoints = api.list_inference_endpoints(namespace=namespace, token=token)45    except HfHubHTTPError as error:46        out.error(f"Listing failed: {error}")47        raise typer.Exit(code=error.response.status_code) from error48 49    results = []50    for endpoint in endpoints:51        raw = endpoint.raw52        status = raw.get("status", {})53        model = raw.get("model", {})54        compute = raw.get("compute", {})55        provider = raw.get("provider", {})56        results.append(57            {58                "name": raw.get("name", ""),59                "model": model.get("repository", "") if isinstance(model, dict) else "",60                "status": status.get("state", "") if isinstance(status, dict) else "",61                "task": model.get("task", "") if isinstance(model, dict) else "",62                "framework": model.get("framework", "") if isinstance(model, dict) else "",63                "instance": compute.get("instanceType", "") if isinstance(compute, dict) else "",64                "vendor": provider.get("vendor", "") if isinstance(provider, dict) else "",65                "region": provider.get("region", "") if isinstance(provider, dict) else "",66            }67        )68    out.table(results, id_key="name")69 70 71@ie_cli.command(name="deploy", examples=["hf endpoints deploy my-endpoint --repo gpt2 --framework pytorch ..."])72def deploy(73    name: NameArg,74    repo: Annotated[75        str,76        typer.Option(77            help="The name of the model repository associated with the Inference Endpoint (e.g. 'openai/gpt-oss-120b').",78        ),79    ],80    framework: Annotated[81        str,82        typer.Option(83            help="The machine learning framework used for the model (e.g. 'vllm').",84        ),85    ],86    accelerator: Annotated[87        str,88        typer.Option(89            help="The hardware accelerator to be used for inference (e.g. 'cpu').",90        ),91    ],92    instance_size: Annotated[93        str,94        typer.Option(95            help="The size or type of the instance to be used for hosting the model (e.g. 'x4').",96        ),97    ],98    instance_type: Annotated[99        str,100        typer.Option(101            help="The cloud instance type where the Inference Endpoint will be deployed (e.g. 'intel-icl').",102        ),103    ],104    region: Annotated[105        str,106        typer.Option(107            help="The cloud region in which the Inference Endpoint will be created (e.g. 'us-east-1').",108        ),109    ],110    vendor: Annotated[111        str,112        typer.Option(113            help="The cloud provider or vendor where the Inference Endpoint will be hosted (e.g. 'aws').",114        ),115    ],116    *,117    namespace: NamespaceOpt = None,118    task: Annotated[119        str | None,120        typer.Option(121            help="The task on which to deploy the model (e.g. 'text-classification').",122        ),123    ] = None,124    token: TokenOpt = None,125    min_replica: Annotated[126        int,127        typer.Option(128            help="The minimum number of replicas (instances) to keep running for the Inference Endpoint.",129        ),130    ] = 1,131    max_replica: Annotated[132        int,133        typer.Option(134            help="The maximum number of replicas (instances) to scale to for the Inference Endpoint.",135        ),136    ] = 1,137    scale_to_zero_timeout: Annotated[138        int | None,139        typer.Option(140            help="The duration in minutes before an inactive endpoint is scaled to zero.",141        ),142    ] = None,143    scaling_metric: Annotated[144        InferenceEndpointScalingMetric | None,145        typer.Option(146            help="The metric reference for scaling.",147        ),148    ] = None,149    scaling_threshold: Annotated[150        float | None,151        typer.Option(152            help="The scaling metric threshold used to trigger a scale up. Ignored when scaling metric is not provided.",153        ),154    ] = None,155) -> None:156    """Deploy an Inference Endpoint from a Hub repository."""157    api = get_hf_api(token=token)158    endpoint = api.create_inference_endpoint(159        name=name,160        repository=repo,161        framework=framework,162        accelerator=accelerator,163        instance_size=instance_size,164        instance_type=instance_type,165        region=region,166        vendor=vendor,167        namespace=namespace,168        task=task,169        token=token,170        min_replica=min_replica,171        max_replica=max_replica,172        scaling_metric=scaling_metric,173        scaling_threshold=scaling_threshold,174        scale_to_zero_timeout=scale_to_zero_timeout,175    )176    out.dict(endpoint.raw)177 178 179@catalog_app.command(name="deploy", examples=["hf endpoints catalog deploy --repo meta-llama/Llama-3.2-1B-Instruct"])180def deploy_from_catalog(181    repo: Annotated[182        str,183        typer.Option(184            help="The name of the model repository associated with the Inference Endpoint (e.g. 'openai/gpt-oss-120b').",185        ),186    ],187    name: NameOpt = None,188    accelerator: Annotated[189        str | None,190        typer.Option(191            help="The hardware accelerator to be used for inference (e.g. 'cpu', 'gpu', 'neuron').",192        ),193    ] = None,194    namespace: NamespaceOpt = None,195    token: TokenOpt = None,196) -> None:197    """Deploy an Inference Endpoint from the Model Catalog."""198    api = get_hf_api(token=token)199    try:200        endpoint = api.create_inference_endpoint_from_catalog(201            repo_id=repo,202            name=name,203            accelerator=accelerator,204            namespace=namespace,205            token=token,206        )207    except HfHubHTTPError as error:208        out.error(f"Deployment failed: {error}")209        raise typer.Exit(code=error.response.status_code) from error210 211    out.dict(endpoint.raw)212 213 214def list_catalog(215    token: TokenOpt = None,216) -> None:217    """List available Catalog models."""218    api = get_hf_api(token=token)219    try:220        models = api.list_inference_catalog(token=token)221    except HfHubHTTPError as error:222        out.error(f"Catalog fetch failed: {error}")223        raise typer.Exit(code=error.response.status_code) from error224 225    out.dict({"models": models})226 227 228catalog_app.command(name="list | ls", examples=["hf endpoints catalog ls"])(list_catalog)229ie_cli.command(name="list-catalog", hidden=True)(list_catalog)230 231 232ie_cli.add_typer(catalog_app, name="catalog")233 234 235@ie_cli.command(examples=["hf endpoints describe my-endpoint"])236def describe(237    name: NameArg,238    namespace: NamespaceOpt = None,239    token: TokenOpt = None,240) -> None:241    """Get information about an existing endpoint."""242    api = get_hf_api(token=token)243    try:244        endpoint = api.get_inference_endpoint(name=name, namespace=namespace, token=token)245    except HfHubHTTPError as error:246        out.error(f"Fetch failed: {error}")247        raise typer.Exit(code=error.response.status_code) from error248 249    out.dict(endpoint.raw)250 251 252@ie_cli.command(examples=["hf endpoints update my-endpoint --min-replica 2"])253def update(254    name: NameArg,255    namespace: NamespaceOpt = None,256    repo: Annotated[257        str | None,258        typer.Option(259            help="The name of the model repository associated with the Inference Endpoint (e.g. 'openai/gpt-oss-120b').",260        ),261    ] = None,262    accelerator: Annotated[263        str | None,264        typer.Option(265            help="The hardware accelerator to be used for inference (e.g. 'cpu').",266        ),267    ] = None,268    instance_size: Annotated[269        str | None,270        typer.Option(271            help="The size or type of the instance to be used for hosting the model (e.g. 'x4').",272        ),273    ] = None,274    instance_type: Annotated[275        str | None,276        typer.Option(277            help="The cloud instance type where the Inference Endpoint will be deployed (e.g. 'intel-icl').",278        ),279    ] = None,280    framework: Annotated[281        str | None,282        typer.Option(283            help="The machine learning framework used for the model (e.g. 'custom').",284        ),285    ] = None,286    revision: Annotated[287        str | None,288        typer.Option(289            help="The specific model revision to deploy on the Inference Endpoint (e.g. '6c0e6080953db56375760c0471a8c5f2929baf11').",290        ),291    ] = None,292    task: Annotated[293        str | None,294        typer.Option(295            help="The task on which to deploy the model (e.g. 'text-classification').",296        ),297    ] = None,298    min_replica: Annotated[299        int | None,300        typer.Option(301            help="The minimum number of replicas (instances) to keep running for the Inference Endpoint.",302        ),303    ] = None,304    max_replica: Annotated[305        int | None,306        typer.Option(307            help="The maximum number of replicas (instances) to scale to for the Inference Endpoint.",308        ),309    ] = None,310    scale_to_zero_timeout: Annotated[311        int | None,312        typer.Option(313            help="The duration in minutes before an inactive endpoint is scaled to zero.",314        ),315    ] = None,316    scaling_metric: Annotated[317        InferenceEndpointScalingMetric | None,318        typer.Option(319            help="The metric reference for scaling.",320        ),321    ] = None,322    scaling_threshold: Annotated[323        float | None,324        typer.Option(325            help="The scaling metric threshold used to trigger a scale up. Ignored when scaling metric is not provided.",326        ),327    ] = None,328    token: TokenOpt = None,329) -> None:330    """Update an existing endpoint."""331    api = get_hf_api(token=token)332    try:333        endpoint = api.update_inference_endpoint(334            name=name,335            namespace=namespace,336            repository=repo,337            framework=framework,338            revision=revision,339            task=task,340            accelerator=accelerator,341            instance_size=instance_size,342            instance_type=instance_type,343            min_replica=min_replica,344            max_replica=max_replica,345            scale_to_zero_timeout=scale_to_zero_timeout,346            scaling_metric=scaling_metric,347            scaling_threshold=scaling_threshold,348            token=token,349        )350    except HfHubHTTPError as error:351        out.error(f"Update failed: {error}")352        raise typer.Exit(code=error.response.status_code) from error353    out.dict(endpoint.raw)354 355 356@ie_cli.command(examples=["hf endpoints delete my-endpoint"])357def delete(358    name: NameArg,359    namespace: NamespaceOpt = None,360    yes: Annotated[361        bool,362        typer.Option("--yes", help="Skip confirmation prompts."),363    ] = False,364    token: TokenOpt = None,365) -> None:366    """Delete an Inference Endpoint permanently."""367    out.confirm(f"Delete endpoint '{name}'?", yes=yes)368 369    api = get_hf_api(token=token)370    try:371        api.delete_inference_endpoint(name=name, namespace=namespace, token=token)372    except HfHubHTTPError as error:373        out.error(f"Delete failed: {error}")374        raise typer.Exit(code=error.response.status_code) from error375 376    out.result(f"Deleted '{name}'.", name=name)377 378 379@ie_cli.command(examples=["hf endpoints pause my-endpoint"])380def pause(381    name: NameArg,382    namespace: NamespaceOpt = None,383    token: TokenOpt = None,384) -> None:385    """Pause an Inference Endpoint."""386    api = get_hf_api(token=token)387    try:388        endpoint = api.pause_inference_endpoint(name=name, namespace=namespace, token=token)389    except HfHubHTTPError as error:390        out.error(f"Pause failed: {error}")391        raise typer.Exit(code=error.response.status_code) from error392 393    out.dict(endpoint.raw)394 395 396@ie_cli.command(examples=["hf endpoints resume my-endpoint"])397def resume(398    name: NameArg,399    namespace: NamespaceOpt = None,400    fail_if_already_running: Annotated[401        bool,402        typer.Option(403            "--fail-if-already-running",404            help="If `True`, the method will raise an error if the Inference Endpoint is already running.",405        ),406    ] = False,407    token: TokenOpt = None,408) -> None:409    """Resume an Inference Endpoint."""410    api = get_hf_api(token=token)411    try:412        endpoint = api.resume_inference_endpoint(413            name=name,414            namespace=namespace,415            token=token,416            running_ok=not fail_if_already_running,417        )418    except HfHubHTTPError as error:419        out.error(f"Resume failed: {error}")420        raise typer.Exit(code=error.response.status_code) from error421    out.dict(endpoint.raw)422 423 424@ie_cli.command(examples=["hf endpoints scale-to-zero my-endpoint"])425def scale_to_zero(426    name: NameArg,427    namespace: NamespaceOpt = None,428    token: TokenOpt = None,429) -> None:430    """Scale an Inference Endpoint to zero."""431    api = get_hf_api(token=token)432    try:433        endpoint = api.scale_to_zero_inference_endpoint(name=name, namespace=namespace, token=token)434    except HfHubHTTPError as error:435        out.error(f"Scale To Zero failed: {error}")436        raise typer.Exit(code=error.response.status_code) from error437 438    out.dict(endpoint.raw)439 
codekingpro/portable-devtools · Team Ai