evalstate/diffusers-pr-api
0
1from __future__ import annotations2 3import os4from contextlib import asynccontextmanager5from dataclasses import dataclass6from pathlib import Path7from typing import Any, Literal8 9from fastapi import FastAPI, HTTPException, Request10from fastapi.responses import JSONResponse11 12from slop_farmer.config import PrSearchRefreshOptions13from slop_farmer.data.ghreplica_api import GhReplicaProbeUnavailableError, GhrProbeClient14from slop_farmer.data.snapshot_materialize import materialize_hf_dataset_snapshot15from slop_farmer.data.snapshot_paths import (16 CURRENT_ANALYSIS_MANIFEST_PATH,17 default_hf_materialize_dir,18)19from slop_farmer.reports.analysis_service import (20 get_analysis_best,21 get_analysis_meta_bug,22 get_analysis_status,23 get_pr_analysis,24 list_analysis_duplicate_prs,25 list_analysis_meta_bugs,26)27from slop_farmer.reports.pr_search_service import (28 get_pr_search_cluster,29 get_pr_search_clusters,30 get_pr_search_contributor_pulls,31 get_pr_search_pull_contributor,32 get_pr_search_similar_lookup,33 get_pr_search_status,34 list_pr_search_clusters,35 run_pr_search_refresh,36)37from slop_farmer.reports.read_views import (38 check_issue_cluster_membership,39 get_contributor,40 get_contributor_risk,41 get_contributor_status,42 get_issue_best,43 get_issue_cluster,44 get_issue_cluster_status,45 get_issue_clusters_for_pr,46 get_snapshot_surfaces,47 list_contributors,48 list_issue_clusters,49 list_issue_duplicate_prs,50)51 52 53@dataclass(slots=True)54class PrSearchApiSettings:55 default_repo: str | None56 index_path: Path57 output_dir: Path58 snapshot_dir: Path | None = None59 hf_repo_id: str | None = None60 hf_revision: str | None = None61 hf_materialize_dir: Path | None = None62 ghr_base_url: str | None = None63 http_timeout: int = 18064 http_max_retries: int = 565 refresh_if_missing: bool = False66 rebuild_on_start: bool = False67 include_drafts: bool = False68 include_closed: bool = False69 similar_limit_default: int = 1070 similar_limit_max: int = 5071 candidate_limit_default: int = 572 candidate_limit_max: int = 2073 cluster_list_limit_default: int = 5074 cluster_list_limit_max: int = 20075 issue_list_limit_default: int = 5076 issue_list_limit_max: int = 20077 contributor_list_limit_default: int = 5078 contributor_list_limit_max: int = 20079 probe_limit_default: int = 1080 probe_limit_max: int = 2581 82 @classmethod83 def from_env(cls) -> PrSearchApiSettings:84 output_dir = Path(os.environ.get("OUTPUT_DIR", "data")).resolve()85 index_path = Path(86 os.environ.get("INDEX_PATH", str(output_dir / "state" / "pr-search.duckdb"))87 ).resolve()88 snapshot_dir = _env_path("SNAPSHOT_DIR")89 hf_materialize_dir = _env_path("HF_MATERIALIZE_DIR")90 return cls(91 default_repo=os.environ.get("DEFAULT_REPO"),92 index_path=index_path,93 output_dir=output_dir,94 snapshot_dir=snapshot_dir,95 hf_repo_id=os.environ.get("HF_REPO_ID"),96 hf_revision=os.environ.get("HF_REVISION"),97 hf_materialize_dir=hf_materialize_dir,98 ghr_base_url=os.environ.get("GHR_BASE_URL"),99 http_timeout=_env_int("HTTP_TIMEOUT", 180),100 http_max_retries=_env_int("HTTP_MAX_RETRIES", 5),101 refresh_if_missing=_env_bool("REFRESH_IF_MISSING", False),102 rebuild_on_start=_env_bool("REBUILD_ON_START", False),103 include_drafts=_env_bool("INCLUDE_DRAFTS", False),104 include_closed=_env_bool("INCLUDE_CLOSED", False),105 similar_limit_default=_env_int("SIMILAR_LIMIT_DEFAULT", 10),106 similar_limit_max=_env_int("SIMILAR_LIMIT_MAX", 50),107 candidate_limit_default=_env_int("CANDIDATE_LIMIT_DEFAULT", 5),108 candidate_limit_max=_env_int("CANDIDATE_LIMIT_MAX", 20),109 cluster_list_limit_default=_env_int("CLUSTER_LIST_LIMIT_DEFAULT", 50),110 cluster_list_limit_max=_env_int("CLUSTER_LIST_LIMIT_MAX", 200),111 issue_list_limit_default=_env_int("ISSUE_LIST_LIMIT_DEFAULT", 50),112 issue_list_limit_max=_env_int("ISSUE_LIST_LIMIT_MAX", 200),113 contributor_list_limit_default=_env_int("CONTRIBUTOR_LIST_LIMIT_DEFAULT", 50),114 contributor_list_limit_max=_env_int("CONTRIBUTOR_LIST_LIMIT_MAX", 200),115 probe_limit_default=_env_int("PROBE_LIMIT_DEFAULT", 10),116 probe_limit_max=_env_int("PROBE_LIMIT_MAX", 25),117 )118 119 120def create_app(settings: PrSearchApiSettings | None = None) -> FastAPI:121 api_settings = settings or PrSearchApiSettings.from_env()122 123 @asynccontextmanager124 async def lifespan(app: FastAPI):125 app.state.settings = api_settings126 app.state.ready = False127 app.state.startup_error = None128 try:129 _bootstrap_snapshot_assets(api_settings)130 _bootstrap_index(api_settings)131 app.state.ready = _is_ready(api_settings)132 except Exception as exc:133 app.state.startup_error = str(exc)134 yield135 136 app = FastAPI(title="slop PR search API", version="0.1.1", lifespan=lifespan)137 138 @app.exception_handler(ValueError)139 async def handle_value_error(_request: Request, exc: ValueError) -> JSONResponse:140 status_code = 404 if _looks_not_found(exc) else 400141 return JSONResponse({"detail": str(exc)}, status_code=status_code)142 143 @app.exception_handler(GhReplicaProbeUnavailableError)144 async def handle_probe_unavailable(145 _request: Request, exc: GhReplicaProbeUnavailableError146 ) -> JSONResponse:147 return JSONResponse({"detail": str(exc)}, status_code=exc.status_code)148 149 @app.get("/healthz")150 async def healthz() -> dict[str, bool]:151 return {"ok": True}152 153 @app.get("/readyz")154 async def readyz(request: Request) -> JSONResponse:155 settings = request.app.state.settings156 error = request.app.state.startup_error157 ready = request.app.state.ready and _is_ready(settings)158 if ready:159 return JSONResponse({"ok": True})160 detail = error or _readiness_detail(settings)161 return JSONResponse({"ok": False, "detail": detail}, status_code=503)162 163 @app.get("/v1/repos/{owner}/{repo}/status")164 async def repo_status(owner: str, repo: str, request: Request) -> dict[str, Any]:165 settings = request.app.state.settings166 repo_slug = _repo_slug(settings, owner, repo)167 status = get_pr_search_status(settings.index_path, repo=repo_slug)168 issue_snapshot_dir = _surface_snapshot_dir(settings, repo_slug, surface="issues")169 contributor_snapshot_dir = _surface_snapshot_dir(170 settings, repo_slug, surface="contributors"171 )172 return {173 **status,174 "surfaces": {175 "issues": get_snapshot_surfaces(issue_snapshot_dir)["issues"],176 "contributors": get_snapshot_surfaces(contributor_snapshot_dir)["contributors"],177 },178 }179 180 @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/similar")181 async def pr_similar(182 owner: str,183 repo: str,184 number: int,185 request: Request,186 limit: int | None = None,187 mode: Literal["auto", "indexed", "live"] = "auto",188 ) -> dict[str, Any]:189 settings = request.app.state.settings190 repo_slug = _repo_slug(settings, owner, repo)191 return get_pr_search_similar_lookup(192 settings.index_path,193 repo=repo_slug,194 pr_number=number,195 limit=_limit(196 limit, default=settings.similar_limit_default, maximum=settings.similar_limit_max197 ),198 mode=mode,199 client=_probe_client(settings),200 )201 202 @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/clusters")203 async def pr_clusters(204 owner: str,205 repo: str,206 number: int,207 request: Request,208 limit: int | None = None,209 mode: Literal["auto", "indexed", "live"] = "auto",210 ) -> dict[str, Any]:211 settings = request.app.state.settings212 repo_slug = _repo_slug(settings, owner, repo)213 return get_pr_search_clusters(214 settings.index_path,215 repo=repo_slug,216 pr_number=number,217 limit=_limit(218 limit,219 default=settings.candidate_limit_default,220 maximum=settings.candidate_limit_max,221 ),222 mode=mode,223 client=_probe_client(settings),224 )225 226 @app.get("/v1/repos/{owner}/{repo}/clusters/{cluster_id}")227 async def cluster_view(228 owner: str,229 repo: str,230 cluster_id: str,231 request: Request,232 ) -> dict[str, Any]:233 settings = request.app.state.settings234 repo_slug = _repo_slug(settings, owner, repo)235 return get_pr_search_cluster(settings.index_path, repo=repo_slug, cluster_id=cluster_id)236 237 @app.get("/v1/repos/{owner}/{repo}/clusters")238 async def cluster_list(239 owner: str,240 repo: str,241 request: Request,242 limit: int | None = None,243 ) -> dict[str, Any]:244 settings = request.app.state.settings245 repo_slug = _repo_slug(settings, owner, repo)246 return list_pr_search_clusters(247 settings.index_path,248 repo=repo_slug,249 limit=_limit(250 limit,251 default=settings.cluster_list_limit_default,252 maximum=settings.cluster_list_limit_max,253 ),254 )255 256 @app.get("/v1/repos/{owner}/{repo}/contributors/{login}/pulls")257 async def contributor_pulls(258 owner: str,259 repo: str,260 login: str,261 request: Request,262 limit: int | None = None,263 ) -> dict[str, Any]:264 settings = request.app.state.settings265 repo_slug = _repo_slug(settings, owner, repo)266 return get_pr_search_contributor_pulls(267 settings.index_path,268 repo=repo_slug,269 author_login=login,270 limit=_limit(271 limit, default=settings.similar_limit_default, maximum=settings.similar_limit_max272 ),273 )274 275 @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/contributor")276 async def pull_contributor(277 owner: str,278 repo: str,279 number: int,280 request: Request,281 ) -> dict[str, Any]:282 settings = request.app.state.settings283 repo_slug = _repo_slug(settings, owner, repo)284 return get_pr_search_pull_contributor(settings.index_path, repo=repo_slug, pr_number=number)285 286 @app.get("/v1/repos/{owner}/{repo}/analysis/status")287 async def analysis_status(288 owner: str,289 repo: str,290 request: Request,291 variant: Literal["auto", "hybrid", "deterministic"] = "auto",292 snapshot_id: str | None = None,293 analysis_id: str | None = None,294 ) -> dict[str, Any]:295 settings = request.app.state.settings296 repo_slug = _repo_slug(settings, owner, repo)297 return get_analysis_status(298 settings.index_path,299 repo=repo_slug,300 variant=variant,301 snapshot_id=snapshot_id,302 analysis_id=analysis_id,303 )304 305 @app.get("/v1/repos/{owner}/{repo}/pulls/{number}/analysis")306 async def pr_analysis(307 owner: str,308 repo: str,309 number: int,310 request: Request,311 variant: Literal["auto", "hybrid", "deterministic"] = "auto",312 snapshot_id: str | None = None,313 analysis_id: str | None = None,314 ) -> dict[str, Any]:315 settings = request.app.state.settings316 repo_slug = _repo_slug(settings, owner, repo)317 return get_pr_analysis(318 settings.index_path,319 repo=repo_slug,320 pr_number=number,321 variant=variant,322 snapshot_id=snapshot_id,323 analysis_id=analysis_id,324 )325 326 @app.get("/v1/repos/{owner}/{repo}/analysis/meta-bugs")327 async def analysis_meta_bugs(328 owner: str,329 repo: str,330 request: Request,331 limit: int | None = None,332 variant: Literal["auto", "hybrid", "deterministic"] = "auto",333 snapshot_id: str | None = None,334 analysis_id: str | None = None,335 ) -> dict[str, Any]:336 settings = request.app.state.settings337 repo_slug = _repo_slug(settings, owner, repo)338 return list_analysis_meta_bugs(339 settings.index_path,340 repo=repo_slug,341 variant=variant,342 limit=_limit(343 limit,344 default=settings.cluster_list_limit_default,345 maximum=settings.cluster_list_limit_max,346 ),347 snapshot_id=snapshot_id,348 analysis_id=analysis_id,349 )350 351 @app.get("/v1/repos/{owner}/{repo}/analysis/meta-bugs/{cluster_id}")352 async def analysis_meta_bug(353 owner: str,354 repo: str,355 cluster_id: str,356 request: Request,357 variant: Literal["auto", "hybrid", "deterministic"] = "auto",358 snapshot_id: str | None = None,359 analysis_id: str | None = None,360 ) -> dict[str, Any]:361 settings = request.app.state.settings362 repo_slug = _repo_slug(settings, owner, repo)363 return get_analysis_meta_bug(364 settings.index_path,365 repo=repo_slug,366 cluster_id=cluster_id,367 variant=variant,368 snapshot_id=snapshot_id,369 analysis_id=analysis_id,370 )371 372 @app.get("/v1/repos/{owner}/{repo}/analysis/duplicate-prs")373 async def analysis_duplicate_prs(374 owner: str,375 repo: str,376 request: Request,377 limit: int | None = None,378 variant: Literal["auto", "hybrid", "deterministic"] = "auto",379 snapshot_id: str | None = None,380 analysis_id: str | None = None,381 ) -> dict[str, Any]:382 settings = request.app.state.settings383 repo_slug = _repo_slug(settings, owner, repo)384 return list_analysis_duplicate_prs(385 settings.index_path,386 repo=repo_slug,387 variant=variant,388 limit=_limit(389 limit,390 default=settings.cluster_list_limit_default,391 maximum=settings.cluster_list_limit_max,392 ),393 snapshot_id=snapshot_id,394 analysis_id=analysis_id,395 )396 397 @app.get("/v1/repos/{owner}/{repo}/analysis/best")398 async def analysis_best(399 owner: str,400 repo: str,401 request: Request,402 variant: Literal["auto", "hybrid", "deterministic"] = "auto",403 snapshot_id: str | None = None,404 analysis_id: str | None = None,405 ) -> dict[str, Any]:406 settings = request.app.state.settings407 repo_slug = _repo_slug(settings, owner, repo)408 return get_analysis_best(409 settings.index_path,410 repo=repo_slug,411 variant=variant,412 snapshot_id=snapshot_id,413 analysis_id=analysis_id,414 )415 416 @app.get("/v1/repos/{owner}/{repo}/issues/status")417 async def issue_status(418 owner: str,419 repo: str,420 request: Request,421 variant: Literal["auto", "hybrid", "deterministic"] = "auto",422 ) -> dict[str, Any]:423 settings = request.app.state.settings424 repo_slug = _repo_slug(settings, owner, repo)425 return get_issue_cluster_status(426 _surface_snapshot_dir(settings, repo_slug, surface="issues"),427 variant=variant,428 )429 430 @app.get("/v1/repos/{owner}/{repo}/issues/clusters")431 async def issue_clusters(432 owner: str,433 repo: str,434 request: Request,435 limit: int | None = None,436 variant: Literal["auto", "hybrid", "deterministic"] = "auto",437 ) -> dict[str, Any]:438 settings = request.app.state.settings439 repo_slug = _repo_slug(settings, owner, repo)440 return list_issue_clusters(441 _surface_snapshot_dir(settings, repo_slug, surface="issues"),442 limit=_limit(443 limit,444 default=settings.issue_list_limit_default,445 maximum=settings.issue_list_limit_max,446 ),447 variant=variant,448 )449 450 @app.get("/v1/repos/{owner}/{repo}/issues/clusters/{cluster_id}")451 async def issue_cluster(452 owner: str,453 repo: str,454 cluster_id: str,455 request: Request,456 variant: Literal["auto", "hybrid", "deterministic"] = "auto",457 ) -> dict[str, Any]:458 settings = request.app.state.settings459 repo_slug = _repo_slug(settings, owner, repo)460 return get_issue_cluster(461 _surface_snapshot_dir(settings, repo_slug, surface="issues"),462 cluster_id=cluster_id,463 variant=variant,464 )465 466 @app.get("/v1/repos/{owner}/{repo}/issues/pulls/{number}")467 async def issue_clusters_for_pr(468 owner: str,469 repo: str,470 number: int,471 request: Request,472 variant: Literal["auto", "hybrid", "deterministic"] = "auto",473 ) -> dict[str, Any]:474 settings = request.app.state.settings475 repo_slug = _repo_slug(settings, owner, repo)476 return get_issue_clusters_for_pr(477 _surface_snapshot_dir(settings, repo_slug, surface="issues"),478 pr_number=number,479 variant=variant,480 )481 482 @app.get("/v1/repos/{owner}/{repo}/issues/pulls/{number}/membership")483 async def issue_membership_for_pr(484 owner: str,485 repo: str,486 number: int,487 request: Request,488 cluster_id: str | None = None,489 variant: Literal["auto", "hybrid", "deterministic"] = "auto",490 ) -> dict[str, Any]:491 settings = request.app.state.settings492 repo_slug = _repo_slug(settings, owner, repo)493 return check_issue_cluster_membership(494 _surface_snapshot_dir(settings, repo_slug, surface="issues"),495 pr_number=number,496 cluster_id=cluster_id,497 variant=variant,498 )499 500 @app.get("/v1/repos/{owner}/{repo}/issues/duplicate-prs")501 async def issue_duplicate_prs(502 owner: str,503 repo: str,504 request: Request,505 limit: int | None = None,506 variant: Literal["auto", "hybrid", "deterministic"] = "auto",507 ) -> dict[str, Any]:508 settings = request.app.state.settings509 repo_slug = _repo_slug(settings, owner, repo)510 return list_issue_duplicate_prs(511 _surface_snapshot_dir(settings, repo_slug, surface="issues"),512 limit=_limit(513 limit,514 default=settings.issue_list_limit_default,515 maximum=settings.issue_list_limit_max,516 ),517 variant=variant,518 )519 520 @app.get("/v1/repos/{owner}/{repo}/issues/best")521 async def issue_best(522 owner: str,523 repo: str,524 request: Request,525 variant: Literal["auto", "hybrid", "deterministic"] = "auto",526 ) -> dict[str, Any]:527 settings = request.app.state.settings528 repo_slug = _repo_slug(settings, owner, repo)529 return get_issue_best(530 _surface_snapshot_dir(settings, repo_slug, surface="issues"),531 variant=variant,532 )533 534 @app.get("/v1/repos/{owner}/{repo}/contributors/status")535 async def contributor_status(536 owner: str,537 repo: str,538 request: Request,539 ) -> dict[str, Any]:540 settings = request.app.state.settings541 repo_slug = _repo_slug(settings, owner, repo)542 return get_contributor_status(543 _surface_snapshot_dir(settings, repo_slug, surface="contributors")544 )545 546 @app.get("/v1/repos/{owner}/{repo}/contributors")547 async def contributors(548 owner: str,549 repo: str,550 request: Request,551 limit: int | None = None,552 ) -> dict[str, Any]:553 settings = request.app.state.settings554 repo_slug = _repo_slug(settings, owner, repo)555 return list_contributors(556 _surface_snapshot_dir(settings, repo_slug, surface="contributors"),557 limit=_limit(558 limit,559 default=settings.contributor_list_limit_default,560 maximum=settings.contributor_list_limit_max,561 ),562 )563 564 @app.get("/v1/repos/{owner}/{repo}/contributors/{login}")565 async def contributor(566 owner: str,567 repo: str,568 login: str,569 request: Request,570 ) -> dict[str, Any]:571 settings = request.app.state.settings572 repo_slug = _repo_slug(settings, owner, repo)573 return get_contributor(574 _surface_snapshot_dir(settings, repo_slug, surface="contributors"),575 author_login=login,576 )577 578 @app.get("/v1/repos/{owner}/{repo}/contributors/{login}/risk")579 async def contributor_risk(580 owner: str,581 repo: str,582 login: str,583 request: Request,584 ) -> dict[str, Any]:585 settings = request.app.state.settings586 repo_slug = _repo_slug(settings, owner, repo)587 return get_contributor_risk(588 _surface_snapshot_dir(settings, repo_slug, surface="contributors"),589 author_login=login,590 )591 592 return app593 594 595def _bootstrap_index(settings: PrSearchApiSettings) -> None:596 settings.output_dir.mkdir(parents=True, exist_ok=True)597 settings.index_path.parent.mkdir(parents=True, exist_ok=True)598 if not _needs_refresh(settings):599 return600 if settings.snapshot_dir is None and settings.hf_repo_id is None:601 return602 run_pr_search_refresh(603 PrSearchRefreshOptions(604 snapshot_dir=settings.snapshot_dir,605 output_dir=settings.output_dir,606 db=settings.index_path,607 hf_repo_id=settings.hf_repo_id,608 hf_revision=settings.hf_revision,609 hf_materialize_dir=settings.hf_materialize_dir,610 include_drafts=settings.include_drafts,611 include_closed=settings.include_closed,612 )613 )614 615 616def _bootstrap_snapshot_assets(settings: PrSearchApiSettings) -> None:617 if settings.snapshot_dir is not None or settings.hf_repo_id is None:618 return619 materialize_dir = settings.hf_materialize_dir or default_hf_materialize_dir(620 settings.output_dir,621 settings.hf_repo_id,622 settings.hf_revision,623 )624 materialize_hf_dataset_snapshot(625 repo_id=settings.hf_repo_id,626 local_dir=materialize_dir,627 revision=settings.hf_revision,628 )629 630 631def _needs_refresh(settings: PrSearchApiSettings) -> bool:632 if settings.rebuild_on_start:633 return True634 if not settings.refresh_if_missing:635 return False636 return not _is_ready(settings)637 638 639def _is_ready(settings: PrSearchApiSettings) -> bool:640 if not settings.index_path.exists():641 return False642 try:643 get_pr_search_status(settings.index_path, repo=settings.default_repo)644 except Exception:645 return False646 return True647 648 649def _readiness_detail(settings: PrSearchApiSettings) -> str:650 if not settings.index_path.exists():651 return f"index not found at {settings.index_path}"652 try:653 get_pr_search_status(settings.index_path, repo=settings.default_repo)654 except Exception as exc:655 return str(exc)656 return "ready"657 658 659def _repo_slug(settings: PrSearchApiSettings, owner: str, repo: str) -> str:660 repo_slug = f"{owner}/{repo}"661 if settings.default_repo and repo_slug != settings.default_repo:662 raise HTTPException(663 status_code=400,664 detail=f"repo {settings.default_repo} is the only configured repo for this deployment",665 )666 return repo_slug667 668 669def _active_snapshot_dir(settings: PrSearchApiSettings, repo_slug: str) -> Path:670 return _status_snapshot_dir(get_pr_search_status(settings.index_path, repo=repo_slug))671 672 673def _surface_snapshot_dir(674 settings: PrSearchApiSettings,675 repo_slug: str,676 *,677 surface: Literal["issues", "contributors"],678) -> Path:679 active_snapshot_dir = _active_snapshot_dir(settings, repo_slug)680 if _surface_available(active_snapshot_dir, surface=surface):681 return active_snapshot_dir682 materialized_snapshot_dir = _materialized_snapshot_dir(settings)683 if materialized_snapshot_dir is not None and _surface_available(684 materialized_snapshot_dir, surface=surface685 ):686 return materialized_snapshot_dir687 return active_snapshot_dir688 689 690def _status_snapshot_dir(status: dict[str, Any]) -> Path:691 snapshot_dir = status.get("snapshot_dir")692 if not snapshot_dir:693 raise HTTPException(status_code=503, detail="active snapshot directory is unavailable")694 return Path(str(snapshot_dir))695 696 697def _materialized_snapshot_dir(settings: PrSearchApiSettings) -> Path | None:698 if settings.hf_repo_id is None:699 return None700 return settings.hf_materialize_dir or default_hf_materialize_dir(701 settings.output_dir,702 settings.hf_repo_id,703 settings.hf_revision,704 )705 706 707def _surface_available(snapshot_dir: Path, *, surface: Literal["issues", "contributors"]) -> bool:708 if not snapshot_dir.exists():709 return False710 if surface == "issues":711 return (snapshot_dir / CURRENT_ANALYSIS_MANIFEST_PATH).exists() or any(712 snapshot_dir.glob("analysis-report*.json")713 )714 return (snapshot_dir / "new-contributors-report.json").exists()715 716 717def _limit(value: int | None, *, default: int, maximum: int) -> int:718 limit = default if value is None else value719 if limit < 1:720 raise HTTPException(status_code=400, detail="limit must be at least 1")721 if limit > maximum:722 raise HTTPException(status_code=400, detail=f"limit must be at most {maximum}")723 return limit724 725 726def _probe_client(settings: PrSearchApiSettings) -> Any:727 if not settings.ghr_base_url:728 return None729 return GhrProbeClient(730 base_url=settings.ghr_base_url,731 timeout=settings.http_timeout,732 max_retries=settings.http_max_retries,733 )734 735 736def _looks_not_found(exc: ValueError) -> bool:737 message = str(exc).lower()738 return (739 "not found" in message740 or "analysis report was not found" in message741 or "no analysis report was found" in message742 or "published analysis" in message743 or "materialized snapshot" in message744 or "no active pr search run" in message745 or "was not found in the active indexed universe" in message746 )747 748 749def _env_bool(name: str, default: bool) -> bool:750 raw = os.environ.get(name)751 if raw is None:752 return default753 return raw.strip().lower() in {"1", "true", "yes", "on"}754 755 756def _env_int(name: str, default: int) -> int:757 raw = os.environ.get(name)758 return default if raw is None else int(raw)759 760 761def _env_path(name: str) -> Path | None:762 raw = os.environ.get(name)763 return None if raw is None else Path(raw).resolve()764 765 766app = create_app()767 