evalstate/diffusers-pr-api
0
1from __future__ import annotations2 3import os4import subprocess5from dataclasses import dataclass6from pathlib import Path7from typing import Any8 9 10def _read_gh_token() -> str | None:11 try:12 result = subprocess.run(13 ["gh", "auth", "token"],14 check=True,15 capture_output=True,16 text=True,17 )18 except (OSError, subprocess.CalledProcessError):19 return None20 token = result.stdout.strip()21 return token or None22 23 24def _read_dotenv_token() -> str | None:25 for directory in (Path.cwd(), *Path.cwd().parents):26 path = directory / ".env"27 if not path.exists():28 continue29 values: dict[str, str] = {}30 for line in path.read_text(encoding="utf-8").splitlines():31 line = line.strip()32 if not line or line.startswith("#") or "=" not in line:33 continue34 key, value = line.split("=", 1)35 values[key.strip()] = value.strip().strip("'").strip('"')36 for key in ("GITHUB_TOKEN", "GRAPHQL_TOKEN", "GH_TOKEN"):37 token = values.get(key)38 if token:39 return token40 return None41 42 43def resolve_github_token() -> str | None:44 return (45 os.environ.get("GITHUB_TOKEN")46 or os.environ.get("GRAPHQL_TOKEN")47 or os.environ.get("GH_TOKEN")48 or _read_dotenv_token()49 or _read_gh_token()50 )51 52 53@dataclass(slots=True)54class RepoRef:55 owner: str56 name: str57 58 @classmethod59 def parse(cls, raw: str) -> RepoRef:60 owner, sep, name = raw.partition("/")61 if not sep or not owner or not name:62 raise ValueError(f"Expected REPO in owner/name form, got: {raw!r}")63 return cls(owner=owner, name=name)64 65 @property66 def slug(self) -> str:67 return f"{self.owner}/{self.name}"68 69 70@dataclass(slots=True)71class PipelineOptions:72 repo: RepoRef73 output_dir: Path74 since: str | None75 resume: bool76 http_timeout: int77 http_max_retries: int78 max_issues: int | None79 max_prs: int | None80 max_issue_comments: int | None81 max_reviews_per_pr: int | None82 max_review_comments_per_pr: int | None83 fetch_timeline: bool84 new_contributor_report: bool85 new_contributor_window_days: int86 new_contributor_max_authors: int87 issue_max_age_days: int | None88 pr_max_age_days: int | None89 90 91@dataclass(slots=True)92class AnalysisOptions:93 snapshot_dir: Path | None94 output_dir: Path95 output: Path | None96 hf_repo_id: str | None97 hf_revision: str | None98 hf_materialize_dir: Path | None99 ranking_backend: str100 model: str101 max_clusters: int102 hybrid_llm_concurrency: int = 1103 open_prs_only: bool = False104 cached_analysis: bool = False105 pr_template_cleanup_mode: str = "merge_defaults"106 pr_template_strip_html_comments: bool = True107 pr_template_trim_closing_reference_prefix: bool = True108 pr_template_section_patterns: tuple[str, ...] = ()109 pr_template_line_patterns: tuple[str, ...] = ()110 cluster_suppression_rules: tuple[dict[str, Any], ...] = ()111 112 def __post_init__(self) -> None:113 if self.hybrid_llm_concurrency < 1:114 raise ValueError("hybrid_llm_concurrency must be >= 1")115 116 117@dataclass(slots=True)118class MarkdownReportOptions:119 input: Path120 output: Path | None121 snapshot_dir: Path | None122 123 124@dataclass(slots=True)125class NewContributorReportOptions:126 snapshot_dir: Path | None127 output_dir: Path128 output: Path | None129 json_output: Path | None130 window_days: int131 max_authors: int132 hf_repo_id: str | None = None133 hf_revision: str | None = None134 hf_materialize_dir: Path | None = None135 136 137@dataclass(slots=True)138class DashboardDataOptions:139 snapshot_dir: Path | None140 output_dir: Path141 analysis_input: Path | None142 contributors_input: Path | None143 pr_scope_input: Path | None144 window_days: int145 hf_repo_id: str | None = None146 hf_revision: str | None = None147 hf_materialize_dir: Path | None = None148 snapshot_root: Path | None = None149 150 151@dataclass(slots=True)152class DeployDashboardOptions:153 pipeline_data_dir: Path154 web_dir: Path155 snapshot_dir: Path | None156 analysis_input: Path | None157 contributors_input: Path | None158 pr_scope_input: Path | None159 hf_repo_id: str | None160 hf_revision: str | None161 hf_materialize_dir: Path | None162 refresh_contributors: bool163 dashboard_window_days: int164 contributor_window_days: int165 contributor_max_authors: int166 private_space: bool167 commit_message: str168 space_id: str169 space_title: str | None170 space_emoji: str171 space_color_from: str172 space_color_to: str173 space_short_description: str174 dataset_id: str | None175 space_tags: str | None176 177 178@dataclass(slots=True)179class PrScopeOptions:180 snapshot_dir: Path | None181 output_dir: Path182 output: Path | None183 hf_repo_id: str | None184 hf_revision: str | None185 hf_materialize_dir: Path | None186 cluster_suppression_rules: tuple[dict[str, Any], ...] = ()187 188 189@dataclass(slots=True)190class PrSearchRefreshOptions:191 snapshot_dir: Path | None192 output_dir: Path193 db: Path | None194 hf_repo_id: str | None195 hf_revision: str | None196 hf_materialize_dir: Path | None197 include_drafts: bool = False198 include_closed: bool = False199 limit_prs: int | None = None200 replace_active: bool = True201 cluster_suppression_rules: tuple[dict[str, Any], ...] = ()202 203 204@dataclass(slots=True)205class CheckpointImportOptions:206 source_repo_id: str207 output_dir: Path208 checkpoint_id: str | None209 checkpoint_root: str | None210 publish_repo_id: str | None211 private_hf_repo: bool212 force: bool213 214 215@dataclass(slots=True)216class SnapshotAdoptOptions:217 snapshot_dir: Path218 output_dir: Path219 next_since: str | None220 221 222@dataclass(slots=True)223class DatasetRefreshOptions:224 repo: RepoRef225 hf_repo_id: str226 private_hf_repo: bool227 max_issues: int | None228 max_prs: int | None229 max_issue_comments: int | None230 max_reviews_per_pr: int | None231 max_review_comments_per_pr: int | None232 fetch_timeline: bool233 new_contributor_report: bool234 new_contributor_window_days: int235 new_contributor_max_authors: int236 http_timeout: int237 http_max_retries: int238 checkpoint_every_comments: int239 checkpoint_every_prs: int240 cluster_suppression_rules: tuple[dict[str, Any], ...] = ()241 242 243@dataclass(slots=True)244class PublishAnalysisArtifactsOptions:245 output_dir: Path246 snapshot_dir: Path | None247 analysis_input: Path | None248 hf_repo_id: str249 analysis_id: str250 canonical: bool = False251 save_cache: bool = False252 private_hf_repo: bool = False253 254 255@dataclass(slots=True)256class SaveCacheOptions:257 output_dir: Path258 snapshot_dir: Path | None259 hf_repo_id: str260 private_hf_repo: bool = False261 262 263@dataclass(slots=True)264class DatasetStatusOptions:265 output_dir: Path266 hf_repo_id: str | None267 hf_revision: str | None268 repo: str | None = None269 json_output: bool = False270 