Team Ai
Apppublic

evalstate/diffusers-pr-api

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
pr_search_api.py767 linesDownload Raw Back to app
1from __future__ import annotations2 3import os4from contextlib import asynccontextmanager5from dataclasses import dataclass6from pathlib import Path7from typing import Any, Literal8 9from fastapi import FastAPI, HTTPException, Request10from fastapi.responses import JSONResponse11 12from slop_farmer.config import PrSearchRefreshOptions13from slop_farmer.data.ghreplica_api import GhReplicaProbeUnavailableError, GhrProbeClient14from slop_farmer.data.snapshot_materialize import materialize_hf_dataset_snapshot15from slop_farmer.data.snapshot_paths import (16    CURRENT_ANALYSIS_MANIFEST_PATH,17    default_hf_materialize_dir,18)19from slop_farmer.reports.analysis_service import (20    get_analysis_best,21    get_analysis_meta_bug,22    get_analysis_status,23    get_pr_analysis,24    list_analysis_duplicate_prs,25    list_analysis_meta_bugs,26)27from slop_farmer.reports.pr_search_service import (28    get_pr_search_cluster,29    get_pr_search_clusters,30    get_pr_search_contributor_pulls,31    get_pr_search_pull_contributor,32    get_pr_search_similar_lookup,33    get_pr_search_status,34    list_pr_search_clusters,35    run_pr_search_refresh,36)37from slop_farmer.reports.read_views import (38    check_issue_cluster_membership,39    get_contributor,40    get_contributor_risk,41    get_contributor_status,42    get_issue_best,43    get_issue_cluster,44    get_issue_cluster_status,45    get_issue_clusters_for_pr,46    get_snapshot_surfaces,47    list_contributors,48    list_issue_clusters,49    list_issue_duplicate_prs,50)51 52 53@dataclass(slots=True)54class PrSearchApiSettings:55    default_repo: str | None56    index_path: Path57    output_dir: Path58    snapshot_dir: Path | None = None59    hf_repo_id: str | None = None60    hf_revision: str | None = None61    hf_materialize_dir: Path | None = None62    ghr_base_url: str | None = None63    http_timeout: int = 18064    http_max_retries: int = 565    refresh_if_missing: bool = False66    rebuild_on_start: bool = False67    include_drafts: bool = False68    include_closed: bool = False69    similar_limit_default: int = 1070    similar_limit_max: int = 5071    candidate_limit_default: int = 572    candidate_limit_max: int = 2073    cluster_list_limit_default: int = 5074    cluster_list_limit_max: int = 20075    issue_list_limit_default: int = 5076    issue_list_limit_max: int = 20077    contributor_list_limit_default: int = 5078    contributor_list_limit_max: int = 20079    probe_limit_default: int = 1080    probe_limit_max: int = 2581 82    @classmethod83    def from_env(cls) -> PrSearchApiSettings:84        output_dir = Path(os.environ.get("OUTPUT_DIR", "data")).resolve()85        index_path = Path(86            os.environ.get("INDEX_PATH", str(output_dir / "state" / "pr-search.duckdb"))87        ).resolve()88        snapshot_dir = _env_path("SNAPSHOT_DIR")89        hf_materialize_dir = _env_path("HF_MATERIALIZE_DIR")90        return cls(91            default_repo=os.environ.get("DEFAULT_REPO"),92            index_path=index_path,93            output_dir=output_dir,94            snapshot_dir=snapshot_dir,95            hf_repo_id=os.environ.get("HF_REPO_ID"),96            hf_revision=os.environ.get("HF_REVISION"),97            hf_materialize_dir=hf_materialize_dir,98            ghr_base_url=os.environ.get("GHR_BASE_URL"),99            http_timeout=_env_int("HTTP_TIMEOUT", 180),100            http_max_retries=_env_int("HTTP_MAX_RETRIES", 5),101            refresh_if_missing=_env_bool("REFRESH_IF_MISSING", False),102            rebuild_on_start=_env_bool("REBUILD_ON_START", False),103            include_drafts=_env_bool("INCLUDE_DRAFTS", False),104            include_closed=_env_bool("INCLUDE_CLOSED", False),105            similar_limit_default=_env_int("SIMILAR_LIMIT_DEFAULT", 10),106            similar_limit_max=_env_int("SIMILAR_LIMIT_MAX", 50),107            candidate_limit_default=_env_int("CANDIDATE_LIMIT_DEFAULT", 5),108            candidate_limit_max=_env_int("CANDIDATE_LIMIT_MAX", 20),109            cluster_list_limit_default=_env_int("CLUSTER_LIST_LIMIT_DEFAULT", 50),110            cluster_list_limit_max=_env_int("CLUSTER_LIST_LIMIT_MAX", 200),111            issue_list_limit_default=_env_int("ISSUE_LIST_LIMIT_DEFAULT", 50),112            issue_list_limit_max=_env_int("ISSUE_LIST_LIMIT_MAX", 200),113            contributor_list_limit_default=_env_int("CONTRIBUTOR_LIST_LIMIT_DEFAULT", 50),114            contributor_list_limit_max=_env_int("CONTRIBUTOR_LIST_LIMIT_MAX", 200),115            probe_limit_default=_env_int("PROBE_LIMIT_DEFAULT", 10),116            probe_limit_max=_env_int("PROBE_LIMIT_MAX", 25),117        )118 119 120def create_app(settings: PrSearchApiSettings | None = None) -> FastAPI:121    api_settings = settings or PrSearchApiSettings.from_env()122 123    @asynccontextmanager124    async def lifespan(app: FastAPI):125        app.state.settings = api_settings126        app.state.ready = False127        app.state.startup_error = None128        try:129            _bootstrap_snapshot_assets(api_settings)130            _bootstrap_index(api_settings)131            app.state.ready = _is_ready(api_settings)132        except Exception as exc:133            app.state.startup_error = str(exc)134        yield135 136    app = FastAPI(title="slop PR search API", version="0.1.1", lifespan=lifespan)137 138    @app.exception_handler(ValueError)139    async def handle_value_error(_request: Request, exc: ValueError) -> JSONResponse:140        status_code = 404 if _looks_not_found(exc) else 400141        return JSONResponse({"detail": str(exc)}, status_code=status_code)142 143    @app.exception_handler(GhReplicaProbeUnavailableError)144    async def handle_probe_unavailable(145        _request: Request, exc: GhReplicaProbeUnavailableError146    ) -> JSONResponse:147        return JSONResponse({"detail": str(exc)}, status_code=exc.status_code)148 149    @app.get("/healthz")150    async def healthz() -> dict[str, bool]:151        return {"ok": True}152 153    @app.get("/readyz")154    async def readyz(request: Request) -> JSONResponse:155        settings = request.app.state.settings156        error = request.app.state.startup_error157        ready = request.app.state.ready and _is_ready(settings)158        if ready:159            return JSONResponse({"ok": True})160        detail = error or _readiness_detail(settings)161        return JSONResponse({"ok": False, "detail": detail}, status_code=503)162 163    @app.get("/v1/repos/{owner}/{repo}/status")164    async def repo_status(owner: str, repo: str, request: Request) -> dict[str, Any]:165        settings = request.app.state.settings166        repo_slug = _repo_slug(settings, owner, repo)167        status = get_pr_search_status(settings.index_path, repo=repo_slug)168        issue_snapshot_dir = _surface_snapshot_dir(settings, repo_slug, surface="issues")169        contributor_snapshot_dir = _surface_snapshot_dir(170            settings, repo_slug, surface="contributors"171        )172        return {173            **status,174            "surfaces": {175                "issues": get_snapshot_surfaces(issue_snapshot_dir)["issues"],176                "contributors": get_snapshot_surfaces(contributor_snapshot_dir)["contributors"],177            },178        }179 180    @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/similar")181    async def pr_similar(182        owner: str,183        repo: str,184        number: int,185        request: Request,186        limit: int | None = None,187        mode: Literal["auto", "indexed", "live"] = "auto",188    ) -> dict[str, Any]:189        settings = request.app.state.settings190        repo_slug = _repo_slug(settings, owner, repo)191        return get_pr_search_similar_lookup(192            settings.index_path,193            repo=repo_slug,194            pr_number=number,195            limit=_limit(196                limit, default=settings.similar_limit_default, maximum=settings.similar_limit_max197            ),198            mode=mode,199            client=_probe_client(settings),200        )201 202    @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/clusters")203    async def pr_clusters(204        owner: str,205        repo: str,206        number: int,207        request: Request,208        limit: int | None = None,209        mode: Literal["auto", "indexed", "live"] = "auto",210    ) -> dict[str, Any]:211        settings = request.app.state.settings212        repo_slug = _repo_slug(settings, owner, repo)213        return get_pr_search_clusters(214            settings.index_path,215            repo=repo_slug,216            pr_number=number,217            limit=_limit(218                limit,219                default=settings.candidate_limit_default,220                maximum=settings.candidate_limit_max,221            ),222            mode=mode,223            client=_probe_client(settings),224        )225 226    @app.get("/v1/repos/{owner}/{repo}/clusters/{cluster_id}")227    async def cluster_view(228        owner: str,229        repo: str,230        cluster_id: str,231        request: Request,232    ) -> dict[str, Any]:233        settings = request.app.state.settings234        repo_slug = _repo_slug(settings, owner, repo)235        return get_pr_search_cluster(settings.index_path, repo=repo_slug, cluster_id=cluster_id)236 237    @app.get("/v1/repos/{owner}/{repo}/clusters")238    async def cluster_list(239        owner: str,240        repo: str,241        request: Request,242        limit: int | None = None,243    ) -> dict[str, Any]:244        settings = request.app.state.settings245        repo_slug = _repo_slug(settings, owner, repo)246        return list_pr_search_clusters(247            settings.index_path,248            repo=repo_slug,249            limit=_limit(250                limit,251                default=settings.cluster_list_limit_default,252                maximum=settings.cluster_list_limit_max,253            ),254        )255 256    @app.get("/v1/repos/{owner}/{repo}/contributors/{login}/pulls")257    async def contributor_pulls(258        owner: str,259        repo: str,260        login: str,261        request: Request,262        limit: int | None = None,263    ) -> dict[str, Any]:264        settings = request.app.state.settings265        repo_slug = _repo_slug(settings, owner, repo)266        return get_pr_search_contributor_pulls(267            settings.index_path,268            repo=repo_slug,269            author_login=login,270            limit=_limit(271                limit, default=settings.similar_limit_default, maximum=settings.similar_limit_max272            ),273        )274 275    @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/contributor")276    async def pull_contributor(277        owner: str,278        repo: str,279        number: int,280        request: Request,281    ) -> dict[str, Any]:282        settings = request.app.state.settings283        repo_slug = _repo_slug(settings, owner, repo)284        return get_pr_search_pull_contributor(settings.index_path, repo=repo_slug, pr_number=number)285 286    @app.get("/v1/repos/{owner}/{repo}/analysis/status")287    async def analysis_status(288        owner: str,289        repo: str,290        request: Request,291        variant: Literal["auto", "hybrid", "deterministic"] = "auto",292        snapshot_id: str | None = None,293        analysis_id: str | None = None,294    ) -> dict[str, Any]:295        settings = request.app.state.settings296        repo_slug = _repo_slug(settings, owner, repo)297        return get_analysis_status(298            settings.index_path,299            repo=repo_slug,300            variant=variant,301            snapshot_id=snapshot_id,302            analysis_id=analysis_id,303        )304 305    @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/analysis")306    async def pr_analysis(307        owner: str,308        repo: str,309        number: int,310        request: Request,311        variant: Literal["auto", "hybrid", "deterministic"] = "auto",312        snapshot_id: str | None = None,313        analysis_id: str | None = None,314    ) -> dict[str, Any]:315        settings = request.app.state.settings316        repo_slug = _repo_slug(settings, owner, repo)317        return get_pr_analysis(318            settings.index_path,319            repo=repo_slug,320            pr_number=number,321            variant=variant,322            snapshot_id=snapshot_id,323            analysis_id=analysis_id,324        )325 326    @app.get("/v1/repos/{owner}/{repo}/analysis/meta-bugs")327    async def analysis_meta_bugs(328        owner: str,329        repo: str,330        request: Request,331        limit: int | None = None,332        variant: Literal["auto", "hybrid", "deterministic"] = "auto",333        snapshot_id: str | None = None,334        analysis_id: str | None = None,335    ) -> dict[str, Any]:336        settings = request.app.state.settings337        repo_slug = _repo_slug(settings, owner, repo)338        return list_analysis_meta_bugs(339            settings.index_path,340            repo=repo_slug,341            variant=variant,342            limit=_limit(343                limit,344                default=settings.cluster_list_limit_default,345                maximum=settings.cluster_list_limit_max,346            ),347            snapshot_id=snapshot_id,348            analysis_id=analysis_id,349        )350 351    @app.get("/v1/repos/{owner}/{repo}/analysis/meta-bugs/{cluster_id}")352    async def analysis_meta_bug(353        owner: str,354        repo: str,355        cluster_id: str,356        request: Request,357        variant: Literal["auto", "hybrid", "deterministic"] = "auto",358        snapshot_id: str | None = None,359        analysis_id: str | None = None,360    ) -> dict[str, Any]:361        settings = request.app.state.settings362        repo_slug = _repo_slug(settings, owner, repo)363        return get_analysis_meta_bug(364            settings.index_path,365            repo=repo_slug,366            cluster_id=cluster_id,367            variant=variant,368            snapshot_id=snapshot_id,369            analysis_id=analysis_id,370        )371 372    @app.get("/v1/repos/{owner}/{repo}/analysis/duplicate-prs")373    async def analysis_duplicate_prs(374        owner: str,375        repo: str,376        request: Request,377        limit: int | None = None,378        variant: Literal["auto", "hybrid", "deterministic"] = "auto",379        snapshot_id: str | None = None,380        analysis_id: str | None = None,381    ) -> dict[str, Any]:382        settings = request.app.state.settings383        repo_slug = _repo_slug(settings, owner, repo)384        return list_analysis_duplicate_prs(385            settings.index_path,386            repo=repo_slug,387            variant=variant,388            limit=_limit(389                limit,390                default=settings.cluster_list_limit_default,391                maximum=settings.cluster_list_limit_max,392            ),393            snapshot_id=snapshot_id,394            analysis_id=analysis_id,395        )396 397    @app.get("/v1/repos/{owner}/{repo}/analysis/best")398    async def analysis_best(399        owner: str,400        repo: str,401        request: Request,402        variant: Literal["auto", "hybrid", "deterministic"] = "auto",403        snapshot_id: str | None = None,404        analysis_id: str | None = None,405    ) -> dict[str, Any]:406        settings = request.app.state.settings407        repo_slug = _repo_slug(settings, owner, repo)408        return get_analysis_best(409            settings.index_path,410            repo=repo_slug,411            variant=variant,412            snapshot_id=snapshot_id,413            analysis_id=analysis_id,414        )415 416    @app.get("/v1/repos/{owner}/{repo}/issues/status")417    async def issue_status(418        owner: str,419        repo: str,420        request: Request,421        variant: Literal["auto", "hybrid", "deterministic"] = "auto",422    ) -> dict[str, Any]:423        settings = request.app.state.settings424        repo_slug = _repo_slug(settings, owner, repo)425        return get_issue_cluster_status(426            _surface_snapshot_dir(settings, repo_slug, surface="issues"),427            variant=variant,428        )429 430    @app.get("/v1/repos/{owner}/{repo}/issues/clusters")431    async def issue_clusters(432        owner: str,433        repo: str,434        request: Request,435        limit: int | None = None,436        variant: Literal["auto", "hybrid", "deterministic"] = "auto",437    ) -> dict[str, Any]:438        settings = request.app.state.settings439        repo_slug = _repo_slug(settings, owner, repo)440        return list_issue_clusters(441            _surface_snapshot_dir(settings, repo_slug, surface="issues"),442            limit=_limit(443                limit,444                default=settings.issue_list_limit_default,445                maximum=settings.issue_list_limit_max,446            ),447            variant=variant,448        )449 450    @app.get("/v1/repos/{owner}/{repo}/issues/clusters/{cluster_id}")451    async def issue_cluster(452        owner: str,453        repo: str,454        cluster_id: str,455        request: Request,456        variant: Literal["auto", "hybrid", "deterministic"] = "auto",457    ) -> dict[str, Any]:458        settings = request.app.state.settings459        repo_slug = _repo_slug(settings, owner, repo)460        return get_issue_cluster(461            _surface_snapshot_dir(settings, repo_slug, surface="issues"),462            cluster_id=cluster_id,463            variant=variant,464        )465 466    @app.get("/v1/repos/{owner}/{repo}/issues/pulls/{number}")467    async def issue_clusters_for_pr(468        owner: str,469        repo: str,470        number: int,471        request: Request,472        variant: Literal["auto", "hybrid", "deterministic"] = "auto",473    ) -> dict[str, Any]:474        settings = request.app.state.settings475        repo_slug = _repo_slug(settings, owner, repo)476        return get_issue_clusters_for_pr(477            _surface_snapshot_dir(settings, repo_slug, surface="issues"),478            pr_number=number,479            variant=variant,480        )481 482    @app.get("/v1/repos/{owner}/{repo}/issues/pulls/{number}/membership")483    async def issue_membership_for_pr(484        owner: str,485        repo: str,486        number: int,487        request: Request,488        cluster_id: str | None = None,489        variant: Literal["auto", "hybrid", "deterministic"] = "auto",490    ) -> dict[str, Any]:491        settings = request.app.state.settings492        repo_slug = _repo_slug(settings, owner, repo)493        return check_issue_cluster_membership(494            _surface_snapshot_dir(settings, repo_slug, surface="issues"),495            pr_number=number,496            cluster_id=cluster_id,497            variant=variant,498        )499 500    @app.get("/v1/repos/{owner}/{repo}/issues/duplicate-prs")501    async def issue_duplicate_prs(502        owner: str,503        repo: str,504        request: Request,505        limit: int | None = None,506        variant: Literal["auto", "hybrid", "deterministic"] = "auto",507    ) -> dict[str, Any]:508        settings = request.app.state.settings509        repo_slug = _repo_slug(settings, owner, repo)510        return list_issue_duplicate_prs(511            _surface_snapshot_dir(settings, repo_slug, surface="issues"),512            limit=_limit(513                limit,514                default=settings.issue_list_limit_default,515                maximum=settings.issue_list_limit_max,516            ),517            variant=variant,518        )519 520    @app.get("/v1/repos/{owner}/{repo}/issues/best")521    async def issue_best(522        owner: str,523        repo: str,524        request: Request,525        variant: Literal["auto", "hybrid", "deterministic"] = "auto",526    ) -> dict[str, Any]:527        settings = request.app.state.settings528        repo_slug = _repo_slug(settings, owner, repo)529        return get_issue_best(530            _surface_snapshot_dir(settings, repo_slug, surface="issues"),531            variant=variant,532        )533 534    @app.get("/v1/repos/{owner}/{repo}/contributors/status")535    async def contributor_status(536        owner: str,537        repo: str,538        request: Request,539    ) -> dict[str, Any]:540        settings = request.app.state.settings541        repo_slug = _repo_slug(settings, owner, repo)542        return get_contributor_status(543            _surface_snapshot_dir(settings, repo_slug, surface="contributors")544        )545 546    @app.get("/v1/repos/{owner}/{repo}/contributors")547    async def contributors(548        owner: str,549        repo: str,550        request: Request,551        limit: int | None = None,552    ) -> dict[str, Any]:553        settings = request.app.state.settings554        repo_slug = _repo_slug(settings, owner, repo)555        return list_contributors(556            _surface_snapshot_dir(settings, repo_slug, surface="contributors"),557            limit=_limit(558                limit,559                default=settings.contributor_list_limit_default,560                maximum=settings.contributor_list_limit_max,561            ),562        )563 564    @app.get("/v1/repos/{owner}/{repo}/contributors/{login}")565    async def contributor(566        owner: str,567        repo: str,568        login: str,569        request: Request,570    ) -> dict[str, Any]:571        settings = request.app.state.settings572        repo_slug = _repo_slug(settings, owner, repo)573        return get_contributor(574            _surface_snapshot_dir(settings, repo_slug, surface="contributors"),575            author_login=login,576        )577 578    @app.get("/v1/repos/{owner}/{repo}/contributors/{login}/risk")579    async def contributor_risk(580        owner: str,581        repo: str,582        login: str,583        request: Request,584    ) -> dict[str, Any]:585        settings = request.app.state.settings586        repo_slug = _repo_slug(settings, owner, repo)587        return get_contributor_risk(588            _surface_snapshot_dir(settings, repo_slug, surface="contributors"),589            author_login=login,590        )591 592    return app593 594 595def _bootstrap_index(settings: PrSearchApiSettings) -> None:596    settings.output_dir.mkdir(parents=True, exist_ok=True)597    settings.index_path.parent.mkdir(parents=True, exist_ok=True)598    if not _needs_refresh(settings):599        return600    if settings.snapshot_dir is None and settings.hf_repo_id is None:601        return602    run_pr_search_refresh(603        PrSearchRefreshOptions(604            snapshot_dir=settings.snapshot_dir,605            output_dir=settings.output_dir,606            db=settings.index_path,607            hf_repo_id=settings.hf_repo_id,608            hf_revision=settings.hf_revision,609            hf_materialize_dir=settings.hf_materialize_dir,610            include_drafts=settings.include_drafts,611            include_closed=settings.include_closed,612        )613    )614 615 616def _bootstrap_snapshot_assets(settings: PrSearchApiSettings) -> None:617    if settings.snapshot_dir is not None or settings.hf_repo_id is None:618        return619    materialize_dir = settings.hf_materialize_dir or default_hf_materialize_dir(620        settings.output_dir,621        settings.hf_repo_id,622        settings.hf_revision,623    )624    materialize_hf_dataset_snapshot(625        repo_id=settings.hf_repo_id,626        local_dir=materialize_dir,627        revision=settings.hf_revision,628    )629 630 631def _needs_refresh(settings: PrSearchApiSettings) -> bool:632    if settings.rebuild_on_start:633        return True634    if not settings.refresh_if_missing:635        return False636    return not _is_ready(settings)637 638 639def _is_ready(settings: PrSearchApiSettings) -> bool:640    if not settings.index_path.exists():641        return False642    try:643        get_pr_search_status(settings.index_path, repo=settings.default_repo)644    except Exception:645        return False646    return True647 648 649def _readiness_detail(settings: PrSearchApiSettings) -> str:650    if not settings.index_path.exists():651        return f"index not found at {settings.index_path}"652    try:653        get_pr_search_status(settings.index_path, repo=settings.default_repo)654    except Exception as exc:655        return str(exc)656    return "ready"657 658 659def _repo_slug(settings: PrSearchApiSettings, owner: str, repo: str) -> str:660    repo_slug = f"{owner}/{repo}"661    if settings.default_repo and repo_slug != settings.default_repo:662        raise HTTPException(663            status_code=400,664            detail=f"repo {settings.default_repo} is the only configured repo for this deployment",665        )666    return repo_slug667 668 669def _active_snapshot_dir(settings: PrSearchApiSettings, repo_slug: str) -> Path:670    return _status_snapshot_dir(get_pr_search_status(settings.index_path, repo=repo_slug))671 672 673def _surface_snapshot_dir(674    settings: PrSearchApiSettings,675    repo_slug: str,676    *,677    surface: Literal["issues", "contributors"],678) -> Path:679    active_snapshot_dir = _active_snapshot_dir(settings, repo_slug)680    if _surface_available(active_snapshot_dir, surface=surface):681        return active_snapshot_dir682    materialized_snapshot_dir = _materialized_snapshot_dir(settings)683    if materialized_snapshot_dir is not None and _surface_available(684        materialized_snapshot_dir, surface=surface685    ):686        return materialized_snapshot_dir687    return active_snapshot_dir688 689 690def _status_snapshot_dir(status: dict[str, Any]) -> Path:691    snapshot_dir = status.get("snapshot_dir")692    if not snapshot_dir:693        raise HTTPException(status_code=503, detail="active snapshot directory is unavailable")694    return Path(str(snapshot_dir))695 696 697def _materialized_snapshot_dir(settings: PrSearchApiSettings) -> Path | None:698    if settings.hf_repo_id is None:699        return None700    return settings.hf_materialize_dir or default_hf_materialize_dir(701        settings.output_dir,702        settings.hf_repo_id,703        settings.hf_revision,704    )705 706 707def _surface_available(snapshot_dir: Path, *, surface: Literal["issues", "contributors"]) -> bool:708    if not snapshot_dir.exists():709        return False710    if surface == "issues":711        return (snapshot_dir / CURRENT_ANALYSIS_MANIFEST_PATH).exists() or any(712            snapshot_dir.glob("analysis-report*.json")713        )714    return (snapshot_dir / "new-contributors-report.json").exists()715 716 717def _limit(value: int | None, *, default: int, maximum: int) -> int:718    limit = default if value is None else value719    if limit < 1:720        raise HTTPException(status_code=400, detail="limit must be at least 1")721    if limit > maximum:722        raise HTTPException(status_code=400, detail=f"limit must be at most {maximum}")723    return limit724 725 726def _probe_client(settings: PrSearchApiSettings) -> Any:727    if not settings.ghr_base_url:728        return None729    return GhrProbeClient(730        base_url=settings.ghr_base_url,731        timeout=settings.http_timeout,732        max_retries=settings.http_max_retries,733    )734 735 736def _looks_not_found(exc: ValueError) -> bool:737    message = str(exc).lower()738    return (739        "not found" in message740        or "analysis report was not found" in message741        or "no analysis report was found" in message742        or "published analysis" in message743        or "materialized snapshot" in message744        or "no active pr search run" in message745        or "was not found in the active indexed universe" in message746    )747 748 749def _env_bool(name: str, default: bool) -> bool:750    raw = os.environ.get(name)751    if raw is None:752        return default753    return raw.strip().lower() in {"1", "true", "yes", "on"}754 755 756def _env_int(name: str, default: int) -> int:757    raw = os.environ.get(name)758    return default if raw is None else int(raw)759 760 761def _env_path(name: str) -> Path | None:762    raw = os.environ.get(name)763    return None if raw is None else Path(raw).resolve()764 765 766app = create_app()767