evalstate/diffusers-pr-api
0
1from __future__ import annotations2 3import argparse4import json5import sys6from collections.abc import Callable7from pathlib import Path8from typing import Any9 10from slop_farmer.app.duplicate_prs import DEFAULT_FILE_POLICY, FILE_POLICY_CHOICES11from slop_farmer.app_config import command_defaults, extract_cli_config_path12from slop_farmer.config import (13 AnalysisOptions,14 CheckpointImportOptions,15 DashboardDataOptions,16 DatasetRefreshOptions,17 DatasetStatusOptions,18 DeployDashboardOptions,19 MarkdownReportOptions,20 NewContributorReportOptions,21 PipelineOptions,22 PrScopeOptions,23 PrSearchRefreshOptions,24 PublishAnalysisArtifactsOptions,25 RepoRef,26 SaveCacheOptions,27 SnapshotAdoptOptions,28)29from slop_farmer.reports.duplicate_prs import DEFAULT_DUPLICATE_PR_MODEL30 31CommandHandler = Callable[[argparse.Namespace, Path | None], None]32 33 34def _int_at_least(minimum: int) -> Callable[[str], int]:35 def parse(raw: str) -> int:36 value = int(raw)37 if value < minimum:38 raise argparse.ArgumentTypeError(f"expected integer >= {minimum}")39 return value40 41 return parse42 43 44def build_parser(*, config_path: Path | None = None) -> argparse.ArgumentParser:45 defaults = _load_parser_defaults(config_path)46 47 parser = argparse.ArgumentParser(prog="slop-farmer")48 parser.add_argument(49 "--config",50 type=Path,51 help="YAML config file with shared repo/workspace/dashboard defaults.",52 )53 subparsers = parser.add_subparsers(dest="command", required=True)54 55 _add_scrape_parser(subparsers, defaults["scrape"])56 _add_refresh_dataset_parser(subparsers, defaults["refresh-dataset"])57 _add_analyze_parser(subparsers, defaults["analyze"])58 _add_pr_scope_parser(subparsers, defaults["pr-scope"])59 _add_checkpoint_import_parser(subparsers, defaults["import-hf-checkpoint"])60 _add_adopt_snapshot_parser(subparsers, defaults["adopt-snapshot"])61 _add_markdown_report_parser(subparsers)62 _add_duplicate_prs_parser(subparsers)63 _add_pr_search_parser(subparsers, defaults["pr-search"])64 _add_new_contributor_report_parser(subparsers, defaults["new-contributor-report"])65 _add_dashboard_data_parser(subparsers, defaults["dashboard-data"])66 _add_publish_analysis_artifacts_parser(subparsers, defaults["publish-analysis-artifacts"])67 _add_save_cache_parser(subparsers, defaults["save-cache"])68 _add_deploy_dashboard_parser(subparsers, defaults["deploy-dashboard"])69 _add_dataset_status_parser(subparsers, defaults["dataset-status"])70 return parser71 72 73def _load_parser_defaults(config_path: Path | None) -> dict[str, dict[str, Any]]:74 commands = (75 "scrape",76 "refresh-dataset",77 "analyze",78 "import-hf-checkpoint",79 "pr-scope",80 "pr-search",81 "adopt-snapshot",82 "new-contributor-report",83 "dashboard-data",84 "publish-analysis-artifacts",85 "save-cache",86 "deploy-dashboard",87 "dataset-status",88 )89 return {command: command_defaults(command, config_path=config_path) for command in commands}90 91 92# Parser builders93 94 95def _add_scrape_parser(subparsers: Any, defaults: dict[str, Any]) -> None:96 scrape = subparsers.add_parser("scrape", help="Scrape GitHub and write a snapshot dataset.")97 scrape.add_argument(98 "--repo",99 default=defaults.get("repo", "huggingface/transformers"),100 help="GitHub repository in owner/name form.",101 )102 scrape.add_argument("--output-dir", type=Path, default=Path(defaults.get("output-dir", "data")))103 scrape.add_argument("--since", help="Incremental sync lower bound in ISO 8601 format.")104 scrape.add_argument(105 "--resume",106 dest="resume",107 action="store_true",108 default=True,109 help="Resume from the last successful local watermark when --since is not provided.",110 )111 scrape.add_argument(112 "--no-resume",113 dest="resume",114 action="store_false",115 help="Ignore local watermark state and run from scratch unless --since is set.",116 )117 scrape.add_argument(118 "--http-timeout", type=int, default=180, help="Per-request timeout in seconds."119 )120 scrape.add_argument(121 "--http-max-retries", type=int, default=5, help="Retries for transient network failures."122 )123 scrape.add_argument(124 "--max-issues", type=int, default=None, help="Limit total issue endpoint items read."125 )126 scrape.add_argument(127 "--max-prs", type=int, default=None, help="Limit pull requests to hydrate in detail."128 )129 scrape.add_argument(130 "--issue-max-age-days",131 type=int,132 default=defaults.get("issue-max-age-days"),133 help="Optional created_at age cap for issues included in the snapshot.",134 )135 scrape.add_argument(136 "--pr-max-age-days",137 type=int,138 default=defaults.get("pr-max-age-days"),139 help="Optional created_at age cap for pull requests included in the snapshot.",140 )141 scrape.add_argument(142 "--max-issue-comments", type=int, default=None, help="Limit issue comment rows."143 )144 scrape.add_argument(145 "--max-reviews-per-pr", type=int, default=None, help="Limit review rows per PR."146 )147 scrape.add_argument(148 "--max-review-comments-per-pr",149 type=int,150 default=None,151 help="Limit inline review comment rows per PR.",152 )153 scrape.add_argument(154 "--fetch-timeline",155 action="store_true",156 default=bool(defaults.get("fetch-timeline", False)),157 help="Fetch issue timeline events for linkage rows.",158 )159 scrape.add_argument(160 "--new-contributor-report",161 dest="new_contributor_report",162 action="store_true",163 default=defaults.get("new-contributor-report"),164 help="Generate new contributor dataset/report artifacts for the local snapshot.",165 )166 scrape.add_argument(167 "--no-new-contributor-report",168 dest="new_contributor_report",169 action="store_false",170 help="Skip new contributor dataset/report generation.",171 )172 scrape.add_argument(173 "--new-contributor-window-days",174 type=int,175 default=int(defaults.get("new-contributor-window-days", 42)),176 help="Recent public activity window for contributor enrichment.",177 )178 scrape.add_argument(179 "--new-contributor-max-authors",180 type=int,181 default=int(defaults.get("new-contributor-max-authors", 25)),182 help="Maximum number of contributors to include in the new contributor report. Use 0 for no cap.",183 )184 185 186def _add_refresh_dataset_parser(subparsers: Any, defaults: dict[str, Any]) -> None:187 refresh = subparsers.add_parser(188 "refresh-dataset",189 help="Refresh the canonical Hugging Face dataset repo from remote watermark state.",190 )191 refresh.add_argument(192 "--repo",193 default=defaults.get("repo", "huggingface/transformers"),194 help="GitHub repository in owner/name form.",195 )196 refresh.add_argument(197 "--hf-repo-id",198 default=defaults.get("hf-repo-id"),199 required=defaults.get("hf-repo-id") is None,200 help="Canonical Hugging Face dataset repo id to refresh.",201 )202 refresh.add_argument("--max-issues", type=int, default=defaults.get("max-issues"))203 refresh.add_argument("--max-prs", type=int, default=defaults.get("max-prs"))204 refresh.add_argument(205 "--max-issue-comments", type=int, default=defaults.get("max-issue-comments")206 )207 refresh.add_argument(208 "--max-reviews-per-pr", type=int, default=defaults.get("max-reviews-per-pr")209 )210 refresh.add_argument(211 "--max-review-comments-per-pr",212 type=int,213 default=defaults.get("max-review-comments-per-pr"),214 )215 refresh.add_argument(216 "--fetch-timeline",217 action="store_true",218 default=bool(defaults.get("fetch-timeline", False)),219 )220 refresh.add_argument(221 "--new-contributor-report",222 dest="new_contributor_report",223 action="store_true",224 default=bool(defaults.get("new-contributor-report", True)),225 )226 refresh.add_argument(227 "--no-new-contributor-report",228 dest="new_contributor_report",229 action="store_false",230 )231 refresh.add_argument(232 "--new-contributor-window-days",233 type=int,234 default=int(defaults.get("new-contributor-window-days", 42)),235 )236 refresh.add_argument(237 "--new-contributor-max-authors",238 type=int,239 default=int(defaults.get("new-contributor-max-authors", 25)),240 )241 refresh.add_argument("--http-timeout", type=int, default=300)242 refresh.add_argument("--http-max-retries", type=int, default=8)243 refresh.add_argument("--checkpoint-every-comments", type=int, default=1000)244 refresh.add_argument("--checkpoint-every-prs", type=int, default=25)245 refresh.add_argument(246 "--private-hf-repo",247 dest="private_hf_repo",248 action="store_true",249 default=bool(defaults.get("private-hf-repo", False)),250 help="Create the target dataset repo as private if needed.",251 )252 refresh.add_argument(253 "--private",254 dest="private_hf_repo",255 action="store_true",256 help=argparse.SUPPRESS,257 )258 259 260def _add_analyze_parser(subparsers: Any, defaults: dict[str, Any]) -> None:261 analyze = subparsers.add_parser(262 "analyze",263 help="Analyze a snapshot and write a local JSON report. Canonical publication is separate.",264 )265 analyze.add_argument(266 "--snapshot-dir",267 type=Path,268 help="Snapshot directory to analyze. Defaults to the latest local snapshot.",269 )270 analyze.add_argument(271 "--output-dir", type=Path, default=Path(defaults.get("output-dir", "data"))272 )273 analyze.add_argument("--output", type=Path, help="Output path for the analysis JSON.")274 analyze.add_argument(275 "--hf-repo-id",276 default=defaults.get("hf-repo-id"),277 help="Analyze a canonical Hugging Face dataset repo by materializing a self-consistent published snapshot locally.",278 )279 analyze.add_argument(280 "--hf-revision",281 default=defaults.get("hf-revision"),282 help="Optional Hub revision for metadata and README download.",283 )284 analyze.add_argument(285 "--hf-materialize-dir",286 type=Path,287 default=Path(defaults["hf-materialize-dir"])288 if defaults.get("hf-materialize-dir")289 else None,290 help="Optional local directory used when materializing an HF dataset snapshot.",291 )292 analyze.add_argument(293 "--ranking-backend",294 choices=("hybrid", "deterministic"),295 default=defaults.get("ranking-backend", "hybrid"),296 help="Whether to use deterministic-only ranking or optional fast-agent enrichment.",297 )298 analyze.add_argument(299 "--model",300 default=defaults.get("model", "gpt-5.4-mini?service_tier=flex"),301 help="Model string used by fast-agent when enabled.",302 )303 analyze.add_argument(304 "--max-clusters",305 type=int,306 default=int(defaults.get("max-clusters", 10)),307 help="Maximum number of meta clusters to include in the report.",308 )309 analyze.add_argument(310 "--hybrid-llm-concurrency",311 type=_int_at_least(1),312 default=int(defaults.get("hybrid-llm-concurrency", 1)),313 help=(314 "Maximum number of hybrid LLM review units to run at once. "315 "Use 1 to minimize provider pressure."316 ),317 )318 analyze.add_argument(319 "--open-prs-only",320 action="store_true",321 default=bool(defaults.get("open-prs-only", False)),322 help="Restrict PR analysis/clustering to open PRs only. Draft PRs are still included.",323 )324 325 326def _add_pr_scope_parser(subparsers: Any, defaults: dict[str, Any]) -> None:327 pr_scope = subparsers.add_parser(328 "pr-scope", help="Cluster open PRs by holistic file/scope overlap."329 )330 pr_scope.add_argument(331 "--snapshot-dir",332 type=Path,333 help="Snapshot directory to analyze. Defaults to the latest local snapshot.",334 )335 pr_scope.add_argument(336 "--output-dir", type=Path, default=Path(defaults.get("output-dir", "data"))337 )338 pr_scope.add_argument(339 "--output",340 type=Path,341 help="Output path for the PR scope JSON. Defaults next to the snapshot.",342 )343 pr_scope.add_argument(344 "--hf-repo-id",345 default=defaults.get("hf-repo-id"),346 help="Analyze a Hugging Face dataset repo by materializing its parquet export locally.",347 )348 pr_scope.add_argument(349 "--hf-revision",350 default=defaults.get("hf-revision"),351 help="Optional Hub revision for metadata and README download.",352 )353 pr_scope.add_argument(354 "--hf-materialize-dir",355 type=Path,356 default=Path(defaults["hf-materialize-dir"])357 if defaults.get("hf-materialize-dir")358 else None,359 help="Optional local directory used when materializing an HF dataset snapshot.",360 )361 362 363def _add_checkpoint_import_parser(subparsers: Any, defaults: dict[str, Any]) -> None:364 checkpoint_import = subparsers.add_parser(365 "import-hf-checkpoint",366 help="Import a checkpoint snapshot from an HF dataset repo into a clean local snapshot.",367 )368 checkpoint_import.add_argument(369 "--source-repo-id",370 default=defaults.get("source-repo-id", "burtenshaw/transformers-pr-slop-dataset"),371 help="Source Hugging Face dataset repo id containing checkpoint folders.",372 )373 checkpoint_import.add_argument(374 "--output-dir",375 type=Path,376 default=Path(defaults.get("output-dir", "eval_data")),377 help="Local root directory where the imported snapshot should be written.",378 )379 checkpoint_import.add_argument(380 "--checkpoint-id",381 help="Optional checkpoint snapshot id. Defaults to the latest viable checkpoint.",382 )383 checkpoint_import.add_argument(384 "--checkpoint-root",385 choices=("checkpoints", "_checkpoints"),386 help="Optional checkpoint root directory. Defaults to auto-detect.",387 )388 checkpoint_import.add_argument(389 "--publish-repo-id",390 help="Optional HF dataset repo id to publish the imported clean snapshot to.",391 )392 checkpoint_import.add_argument(393 "--private-hf-repo",394 action="store_true",395 help="Create the publish target as private when --publish-repo-id is used.",396 )397 checkpoint_import.add_argument(398 "--force",399 action="store_true",400 help="Overwrite an existing imported snapshot directory if present.",401 )402 403 404def _add_adopt_snapshot_parser(subparsers: Any, defaults: dict[str, Any]) -> None:405 adopt_snapshot = subparsers.add_parser(406 "adopt-snapshot",407 help="Mark an existing snapshot as the current pipeline base so the next scrape resumes from it.",408 )409 adopt_snapshot.add_argument(410 "--snapshot-dir", type=Path, required=True, help="Existing local snapshot directory."411 )412 adopt_snapshot.add_argument(413 "--output-dir",414 type=Path,415 default=Path(defaults.get("output-dir", "data")),416 help="Pipeline workspace root where state/ and snapshots/latest.json should be written.",417 )418 adopt_snapshot.add_argument(419 "--next-since",420 help="Optional explicit watermark timestamp. Defaults to snapshot watermark.next_since, crawl_started_at, or extracted_at.",421 )422 423 424def _add_markdown_report_parser(subparsers: Any) -> None:425 markdown = subparsers.add_parser(426 "markdown-report", help="Render a markdown report from an analysis JSON file."427 )428 markdown.add_argument(429 "--input", type=Path, required=True, help="Path to an existing analysis JSON report."430 )431 markdown.add_argument(432 "--output",433 type=Path,434 help="Output path for the markdown report. Defaults next to the input JSON.",435 )436 markdown.add_argument(437 "--snapshot-dir",438 type=Path,439 help="Optional snapshot directory containing issues.parquet and pull_requests.parquet. Defaults to the input JSON parent directory.",440 )441 442 443def _add_duplicate_prs_parser(subparsers: Any) -> None:444 duplicate_prs = subparsers.add_parser(445 "duplicate-prs",446 help="List or merge mergeable duplicate PR clusters from hybrid-enriched analysis.",447 )448 duplicate_prs_subparsers = duplicate_prs.add_subparsers(449 dest="duplicate_prs_command", required=True450 )451 452 duplicate_list = duplicate_prs_subparsers.add_parser(453 "list",454 help="List mergeable duplicate PR clusters from a hybrid-enriched analysis report.",455 )456 duplicate_list_source = duplicate_list.add_mutually_exclusive_group(required=True)457 duplicate_list_source.add_argument(458 "--report", type=Path, help="Path to an analysis JSON report."459 )460 duplicate_list_source.add_argument(461 "--snapshot-dir", type=Path, help="Snapshot directory to analyze."462 )463 duplicate_list.add_argument(464 "--limit", type=int, default=10, help="Maximum number of mergeable clusters to print."465 )466 duplicate_list.add_argument(467 "--model",468 default=DEFAULT_DUPLICATE_PR_MODEL,469 help="Model string used for hybrid analysis and duplicate-PR mergeability gating.",470 )471 472 duplicate_merge = duplicate_prs_subparsers.add_parser(473 "merge",474 help="Use Codex to synthesize and publish a minimal upstream PR for a mergeable duplicate cluster.",475 )476 duplicate_merge_source = duplicate_merge.add_mutually_exclusive_group(required=True)477 duplicate_merge_source.add_argument(478 "--report", type=Path, help="Path to an analysis JSON report."479 )480 duplicate_merge_source.add_argument(481 "--snapshot-dir", type=Path, help="Snapshot directory to analyze."482 )483 duplicate_merge.add_argument(484 "--repo-dir",485 type=Path,486 required=True,487 help="Local upstream repository checkout used for the synthesis worktree.",488 )489 duplicate_merge.add_argument(490 "--upstream-repo",491 help="Optional owner/name override for the upstream target repository.",492 )493 duplicate_merge.add_argument(494 "--upstream-remote",495 default="origin",496 help="Remote in --repo-dir that points at the upstream repository. Defaults to origin.",497 )498 duplicate_merge.add_argument(499 "--fork-remote",500 default="fork",501 help="Remote in the synthesis worktree used for pushing the branch. Defaults to fork.",502 )503 duplicate_merge.add_argument("--cluster-id", help="Optional cluster override.")504 duplicate_merge.add_argument(505 "--fork-repo",506 help="Optional owner/name override for the fork push target. Overrides --fork-owner when both are set.",507 )508 duplicate_merge.add_argument(509 "--fork-owner",510 help="Optional GitHub fork owner override. Defaults to the authenticated user.",511 )512 duplicate_merge.add_argument(513 "--file-policy",514 choices=FILE_POLICY_CHOICES,515 default=DEFAULT_FILE_POLICY,516 help="Changed-file policy enforced on the synthesized branch.",517 )518 duplicate_merge.add_argument(519 "--model",520 default=DEFAULT_DUPLICATE_PR_MODEL,521 help="Model string used for hybrid analysis, mergeability gating, and Codex synthesis.",522 )523 524 525def _add_pr_search_parser(subparsers: Any, defaults: dict[str, Any]) -> None:526 pr_search = subparsers.add_parser(527 "pr-search",528 help="Refresh and query the DuckDB-backed PR code-similarity index.",529 )530 pr_search_subparsers = pr_search.add_subparsers(dest="pr_search_command", required=True)531 532 refresh = pr_search_subparsers.add_parser(533 "refresh",534 help="Refresh the PR code-similarity index from a local snapshot or HF dataset repo.",535 )536 refresh_source = refresh.add_mutually_exclusive_group()537 refresh_source.add_argument(538 "--snapshot-dir",539 type=Path,540 help="Snapshot directory to index. Defaults to the latest local snapshot.",541 )542 refresh_source.add_argument(543 "--hf-repo-id",544 default=defaults.get("hf-repo-id"),545 help="Hugging Face dataset repo id to materialize before indexing.",546 )547 refresh.add_argument(548 "--hf-revision",549 default=defaults.get("hf-revision"),550 help="Optional Hub revision for metadata and README download.",551 )552 refresh.add_argument(553 "--hf-materialize-dir",554 type=Path,555 default=Path(defaults["hf-materialize-dir"])556 if defaults.get("hf-materialize-dir")557 else None,558 help="Optional local directory used when materializing an HF dataset snapshot.",559 )560 refresh.add_argument(561 "--output-dir",562 type=Path,563 default=Path(defaults.get("output-dir", "data")),564 help="Workspace root used for latest snapshot resolution and default DB placement.",565 )566 refresh.add_argument(567 "--db",568 type=Path,569 default=Path(defaults["db"]) if defaults.get("db") else None,570 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",571 )572 refresh.add_argument("--limit-prs", type=int, help="Optional cap on indexed PRs.")573 refresh.add_argument(574 "--include-drafts",575 action="store_true",576 default=bool(defaults.get("include-drafts", False)),577 help="Include draft PRs in the indexed universe.",578 )579 refresh.add_argument(580 "--include-closed",581 action="store_true",582 default=bool(defaults.get("include-closed", False)),583 help="Include closed PRs in the indexed universe.",584 )585 refresh.add_argument(586 "--replace-active",587 dest="replace_active",588 action="store_true",589 default=True,590 help="Activate the new run on success. Enabled by default.",591 )592 refresh.add_argument(593 "--no-replace-active",594 dest="replace_active",595 action="store_false",596 help="Write the new run without switching the active run pointer.",597 )598 599 similar = pr_search_subparsers.add_parser(600 "similar", help="Show similar PRs for one indexed pull request."601 )602 similar.add_argument("pr_number", type=int, help="Pull request number to query.")603 similar.add_argument(604 "--db",605 type=Path,606 default=Path(defaults["db"]) if defaults.get("db") else None,607 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",608 )609 similar.add_argument(610 "--output-dir",611 type=Path,612 default=Path(defaults.get("output-dir", "data")),613 )614 similar.add_argument("--repo", help="Optional repo override when the DB holds multiple repos.")615 similar.add_argument("--limit", type=int, default=10, help="Maximum number of rows to show.")616 similar.add_argument("--json", action="store_true", help="Emit machine-readable JSON.")617 618 probe_github = pr_search_subparsers.add_parser(619 "probe-github",620 help="Fetch one live GitHub PR and compare it against the active indexed scope features.",621 )622 probe_github.add_argument("pr_number", type=int, help="Pull request number to probe.")623 probe_github.add_argument(624 "--repo",625 help="GitHub repository in owner/name form. Defaults to the active repo in the DB.",626 )627 probe_github.add_argument(628 "--db",629 type=Path,630 default=Path(defaults["db"]) if defaults.get("db") else None,631 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",632 )633 probe_github.add_argument(634 "--output-dir",635 type=Path,636 default=Path(defaults.get("output-dir", "data")),637 )638 probe_github.add_argument(639 "--limit",640 type=int,641 default=10,642 help="Maximum number of similar PR rows to show.",643 )644 probe_github.add_argument("--json", action="store_true", help="Emit machine-readable JSON.")645 646 candidate_clusters = pr_search_subparsers.add_parser(647 "candidate-clusters",648 help="Show candidate scope clusters for one indexed pull request.",649 )650 candidate_clusters.add_argument("pr_number", type=int, help="Pull request number to query.")651 candidate_clusters.add_argument(652 "--db",653 type=Path,654 default=Path(defaults["db"]) if defaults.get("db") else None,655 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",656 )657 candidate_clusters.add_argument(658 "--output-dir",659 type=Path,660 default=Path(defaults.get("output-dir", "data")),661 )662 candidate_clusters.add_argument(663 "--repo", help="Optional repo override when the DB holds multiple repos."664 )665 candidate_clusters.add_argument(666 "--limit", type=int, default=5, help="Maximum number of rows to show."667 )668 candidate_clusters.add_argument("--json", action="store_true", help="Emit JSON.")669 670 cluster = pr_search_subparsers.add_parser("cluster", help="Inspect one scope cluster.")671 cluster_subparsers = cluster.add_subparsers(dest="pr_search_cluster_command", required=True)672 cluster_show = cluster_subparsers.add_parser("show", help="Show cluster details.")673 cluster_show.add_argument("cluster_id", help="Cluster identifier.")674 cluster_show.add_argument(675 "--db",676 type=Path,677 default=Path(defaults["db"]) if defaults.get("db") else None,678 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",679 )680 cluster_show.add_argument(681 "--output-dir",682 type=Path,683 default=Path(defaults.get("output-dir", "data")),684 )685 cluster_show.add_argument("--repo", help="Optional repo override.")686 cluster_show.add_argument("--json", action="store_true", help="Emit JSON.")687 688 explain_pair = pr_search_subparsers.add_parser(689 "explain-pair",690 help="Explain one PR pair, falling back to on-demand scoring when needed.",691 )692 explain_pair.add_argument("left_pr_number", type=int)693 explain_pair.add_argument("right_pr_number", type=int)694 explain_pair.add_argument(695 "--db",696 type=Path,697 default=Path(defaults["db"]) if defaults.get("db") else None,698 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",699 )700 explain_pair.add_argument(701 "--output-dir",702 type=Path,703 default=Path(defaults.get("output-dir", "data")),704 )705 explain_pair.add_argument("--repo", help="Optional repo override.")706 explain_pair.add_argument("--json", action="store_true", help="Emit JSON.")707 708 status = pr_search_subparsers.add_parser("status", help="Show the active PR search run.")709 status.add_argument(710 "--db",711 type=Path,712 default=Path(defaults["db"]) if defaults.get("db") else None,713 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",714 )715 status.add_argument(716 "--output-dir",717 type=Path,718 default=Path(defaults.get("output-dir", "data")),719 )720 status.add_argument("--repo", help="Optional repo override.")721 status.add_argument("--json", action="store_true", help="Emit JSON.")722 723 contributor = pr_search_subparsers.add_parser(724 "contributor", help="Show indexed contributor summary for one author login."725 )726 contributor.add_argument("login", help="GitHub author login to query.")727 contributor.add_argument(728 "--db",729 type=Path,730 default=Path(defaults["db"]) if defaults.get("db") else None,731 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",732 )733 contributor.add_argument(734 "--output-dir",735 type=Path,736 default=Path(defaults.get("output-dir", "data")),737 )738 contributor.add_argument("--repo", help="Optional repo override.")739 contributor.add_argument("--json", action="store_true", help="Emit JSON.")740 741 contributor_prs = pr_search_subparsers.add_parser(742 "contributor-prs", help="List indexed PRs for one contributor login."743 )744 contributor_prs.add_argument("login", help="GitHub author login to query.")745 contributor_prs.add_argument(746 "--db",747 type=Path,748 default=Path(defaults["db"]) if defaults.get("db") else None,749 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",750 )751 contributor_prs.add_argument(752 "--output-dir",753 type=Path,754 default=Path(defaults.get("output-dir", "data")),755 )756 contributor_prs.add_argument("--repo", help="Optional repo override.")757 contributor_prs.add_argument("--limit", type=int, default=20, help="Maximum rows to show.")758 contributor_prs.add_argument("--json", action="store_true", help="Emit JSON.")759 760 pr_contributor = pr_search_subparsers.add_parser(761 "pr-contributor", help="Show contributor summary for the author of one indexed PR."762 )763 pr_contributor.add_argument("pr_number", type=int, help="Pull request number to query.")764 pr_contributor.add_argument(765 "--db",766 type=Path,767 default=Path(defaults["db"]) if defaults.get("db") else None,768 help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",769 )770 pr_contributor.add_argument(771 "--output-dir",772 type=Path,773 default=Path(defaults.get("output-dir", "data")),774 )775 pr_contributor.add_argument("--repo", help="Optional repo override.")776 pr_contributor.add_argument("--json", action="store_true", help="Emit JSON.")777 778 779def _add_new_contributor_report_parser(subparsers: Any, defaults: dict[str, Any]) -> None:780 new_contributor = subparsers.add_parser(781 "new-contributor-report",782 help="Render a markdown report for newly observed contributors in a snapshot.",783 )784 new_contributor.add_argument(785 "--snapshot-dir",786 type=Path,787 help="Snapshot directory to inspect. Defaults to the latest local snapshot.",788 )789 new_contributor.add_argument(790 "--output-dir", type=Path, default=Path(defaults.get("output-dir", "data"))791 )792 new_contributor.add_argument(793 "--output",794 type=Path,795 help="Output path for the markdown report. Defaults next to the snapshot.",796 )797 new_contributor.add_argument(798 "--json-output", type=Path, help="Optional JSON output path. Defaults next to the snapshot."799 )800 new_contributor.add_argument(801 "--hf-repo-id",802 default=defaults.get("hf-repo-id"),803 help="Analyze a Hugging Face dataset repo by materializing its parquet export locally.",804 )805 new_contributor.add_argument(806 "--hf-revision",807 default=defaults.get("hf-revision"),808 help="Optional Hub revision for metadata and README download.",809 )810 new_contributor.add_argument(811 "--hf-materialize-dir",812 type=Path,813 default=Path(defaults["hf-materialize-dir"])814 if defaults.get("hf-materialize-dir")815 else None,816 help="Optional local directory used when materializing an HF dataset snapshot.",817 )818 new_contributor.add_argument(819 "--window-days",820 type=int,821 default=int(defaults.get("window-days", 42)),822 help="Recent public activity window for contributor enrichment.",823 )824 new_contributor.add_argument(825 "--max-authors",826 type=int,827 default=int(defaults.get("max-authors", 25)),828 help="Maximum number of contributors to include. Use 0 for no cap.",829 )830 831 832def _add_dashboard_data_parser(subparsers: Any, defaults: dict[str, Any]) -> None:833 dashboard = subparsers.add_parser(834 "dashboard-data", help="Export frontend-ready JSON for the static dashboard."835 )836 dashboard.add_argument(837 "--snapshot-dir",838 type=Path,839 help="Snapshot directory to export. Defaults to the latest local snapshot.",840 )841 dashboard.add_argument(842 "--output-dir",843 type=Path,844 default=Path(defaults.get("output-dir", "web/public/data")),845 )846 dashboard.add_argument(847 "--analysis-input",848 type=Path,849 help="Optional analysis report JSON override. Defaults to canonical published current analysis when available, otherwise falls back to snapshot-local analysis files.",850 )851 dashboard.add_argument(852 "--contributors-input",853 type=Path,854 help="Optional contributor report JSON override. Defaults to the materialized snapshot's new-contributors-report.json.",855 )856 dashboard.add_argument(857 "--pr-scope-input",858 type=Path,859 help="Optional PR scope cluster JSON override. Defaults to the materialized snapshot's pr-scope-clusters.json.",860 )861 dashboard.add_argument(862 "--hf-repo-id",863 default=defaults.get("hf-repo-id"),864 help="Materialize the canonical Hugging Face dataset repo instead of using the latest local snapshot.",865 )866 dashboard.add_argument(867 "--hf-revision",868 default=defaults.get("hf-revision"),869 help="Optional Hub revision for metadata and README download.",870 )871 dashboard.add_argument(872 "--hf-materialize-dir",873 type=Path,874 default=Path(defaults["hf-materialize-dir"])875 if defaults.get("hf-materialize-dir")876 else None,877 help="Optional local directory used when materializing an HF dataset snapshot.",878 )879 dashboard.add_argument(880 "--window-days",881 type=int,882 default=int(defaults.get("window-days", 14)),883 help="Recent PR window to expose in the dashboard.",884 )885 886 887def _add_publish_analysis_artifacts_parser(subparsers: Any, defaults: dict[str, Any]) -> None:888 publish_analysis = subparsers.add_parser(889 "publish-analysis-artifacts",890 help="Publish archived and optional canonical hybrid analysis artifacts to a dataset repo.",891 )892 publish_analysis.add_argument(893 "--output-dir",894 type=Path,895 default=Path(defaults.get("output-dir", "data")),896 help="Pipeline workspace root containing snapshots/latest.json.",897 )898 publish_analysis.add_argument(899 "--snapshot-dir",900 type=Path,901 help="Optional explicit snapshot directory containing analysis-report-hybrid.json.",902 )903 publish_analysis.add_argument(904 "--analysis-input",905 type=Path,906 help="Optional explicit hybrid analysis report JSON to publish instead of snapshot-dir discovery.",907 )908 publish_analysis.add_argument(909 "--hf-repo-id",910 default=defaults.get("hf-repo-id"),911 required=defaults.get("hf-repo-id") is None,912 help="Target Hugging Face dataset repo id.",913 )914 publish_analysis.add_argument("--analysis-id", required=True, help="Immutable analysis run id.")915 publish_analysis.add_argument(916 "--canonical",917 action="store_true",918 default=bool(defaults.get("canonical", False)),919 help="Also update the stable analysis/current canonical alias.",920 )921 publish_analysis.add_argument(922 "--save-cache",923 action="store_true",924 default=bool(defaults.get("save-cache", False)),925 help="Also upload snapshot-local analysis-state/ as mutable operational cache at repo-root analysis-state/.",926 )927 publish_analysis.add_argument(928 "--private-hf-repo",929 action="store_true",930 default=bool(defaults.get("private-hf-repo", False)),931 help="Create the target dataset repo as private if needed.",932 )933 934 935def _add_save_cache_parser(subparsers: Any, defaults: dict[str, Any]) -> None:936 save_cache = subparsers.add_parser(937 "save-cache",938 help="Upload snapshot-local analysis-state/ as mutable operational cache to a dataset repo.",939 )940 save_cache.add_argument(941 "--output-dir",942 type=Path,943 default=Path(defaults.get("output-dir", "data")),944 help="Pipeline workspace root containing snapshots/latest.json.",945 )946 save_cache.add_argument(947 "--snapshot-dir",948 type=Path,949 help="Optional explicit snapshot directory containing analysis-state/.",950 )951 save_cache.add_argument(952 "--hf-repo-id",953 default=defaults.get("hf-repo-id"),954 required=defaults.get("hf-repo-id") is None,955 help="Target Hugging Face dataset repo id.",956 )957 save_cache.add_argument(958 "--private-hf-repo",959 action="store_true",960 default=bool(defaults.get("private-hf-repo", False)),961 help="Create the target dataset repo as private if needed.",962 )963 964 965def _add_deploy_dashboard_parser(subparsers: Any, defaults: dict[str, Any]) -> None:966 deploy_dashboard = subparsers.add_parser(967 "deploy-dashboard",968 help="Build and publish the static dashboard to a Hugging Face Space from a materialized dataset view.",969 )970 deploy_dashboard.add_argument(971 "--pipeline-data-dir",972 type=Path,973 default=Path(defaults.get("pipeline-data-dir", "data")),974 )975 deploy_dashboard.add_argument(976 "--web-dir", type=Path, default=Path(defaults.get("web-dir", "web"))977 )978 deploy_dashboard.add_argument(979 "--snapshot-dir",980 type=Path,981 help="Optional snapshot directory to publish. Defaults to the latest snapshot in --pipeline-data-dir.",982 )983 deploy_dashboard.add_argument(984 "--analysis-input",985 type=Path,986 help="Optional analysis report JSON override. Omit to prefer canonical published current analysis when available.",987 )988 deploy_dashboard.add_argument(989 "--contributors-input",990 type=Path,991 help="Optional contributor report JSON override.",992 )993 deploy_dashboard.add_argument(994 "--pr-scope-input",995 type=Path,996 help="Optional PR scope cluster JSON override.",997 )998 deploy_dashboard.add_argument(999 "--hf-repo-id",1000 default=defaults.get("hf-repo-id"),1001 help="Materialize the canonical Hugging Face dataset repo instead of using the latest local snapshot.",1002 )1003 deploy_dashboard.add_argument(1004 "--hf-revision",1005 default=defaults.get("hf-revision"),1006 help="Optional Hub revision for metadata and README download.",1007 )1008 deploy_dashboard.add_argument(1009 "--hf-materialize-dir",1010 type=Path,1011 default=Path(defaults["hf-materialize-dir"])1012 if defaults.get("hf-materialize-dir")1013 else None,1014 help="Optional local directory used when materializing an HF dataset snapshot.",1015 )1016 deploy_dashboard.add_argument(1017 "--refresh-contributors",1018 action="store_true",1019 default=bool(defaults.get("refresh-contributors", False)),1020 )1021 deploy_dashboard.add_argument(1022 "--dashboard-window-days",1023 type=int,1024 default=int(defaults.get("dashboard-window-days", 14)),1025 )1026 deploy_dashboard.add_argument(1027 "--contributor-window-days",1028 type=int,1029 default=int(1030 defaults.get("contributor-window-days", defaults.get("dashboard-window-days", 14))1031 ),1032 )1033 deploy_dashboard.add_argument(1034 "--contributor-max-authors",1035 type=int,1036 default=int(defaults.get("contributor-max-authors", 0)),1037 )1038 deploy_dashboard.add_argument(1039 "--private-space",1040 action="store_true",1041 default=bool(defaults.get("private-space", False)),1042 )1043 deploy_dashboard.add_argument(1044 "--commit-message",1045 default=defaults.get("commit-message", "Deploy dashboard"),1046 )1047 deploy_dashboard.add_argument(1048 "--space-id",1049 default=defaults.get("space-id"),1050 help="Hugging Face Space repo id.",1051 )1052 deploy_dashboard.add_argument("--space-title", default=defaults.get("space-title"))1053 deploy_dashboard.add_argument("--space-emoji", default=defaults.get("space-emoji", "๐"))1054 deploy_dashboard.add_argument(1055 "--space-color-from", default=defaults.get("space-color-from", "indigo")1056 )1057 deploy_dashboard.add_argument(1058 "--space-color-to", default=defaults.get("space-color-to", "blue")1059 )1060 deploy_dashboard.add_argument(1061 "--space-short-description",1062 default=defaults.get(1063 "space-short-description", "Static dashboard for the slop-farmer PR analysis pipeline."1064 ),1065 )1066 deploy_dashboard.add_argument("--dataset-id", default=defaults.get("dataset-id"))1067 deploy_dashboard.add_argument(1068 "--space-tags", default=defaults.get("space-tags", "dashboard,static")1069 )1070 1071 1072def _add_dataset_status_parser(subparsers: Any, defaults: dict[str, Any]) -> None:1073 dataset_status = subparsers.add_parser(1074 "dataset-status",1075 help="Inspect canonical dataset freshness and the local latest pointer.",1076 )1077 dataset_status.add_argument("--repo", default=defaults.get("repo"))1078 dataset_status.add_argument(1079 "--output-dir",1080 type=Path,1081 default=Path(defaults.get("output-dir", "data")),1082 help="Local workspace root containing snapshots/latest.json.",1083 )1084 dataset_status.add_argument(1085 "--hf-repo-id",1086 default=defaults.get("hf-repo-id"),1087 help="Canonical Hugging Face dataset repo id to inspect.",1088 )1089 dataset_status.add_argument(1090 "--hf-revision",1091 default=defaults.get("hf-revision"),1092 help="Optional Hub revision for metadata and README download.",1093 )1094 dataset_status.add_argument("--json", action="store_true", help="Emit machine-readable JSON.")1095 1096 1097# Dispatch helpers1098 1099 1100def _explicit_flag_present(flag: str) -> bool:1101 return any(arg == flag or arg.startswith(f"{flag}=") for arg in sys.argv[1:])1102 1103 1104def _resolve_hf_inputs(args: argparse.Namespace) -> tuple[str | None, str | None, Path | None]:1105 hf_repo_id = args.hf_repo_id1106 hf_revision = args.hf_revision1107 hf_materialize_dir = args.hf_materialize_dir1108 if args.snapshot_dir is not None and not _explicit_flag_present("--hf-repo-id"):1109 hf_repo_id = None1110 hf_revision = None1111 hf_materialize_dir = None1112 return hf_repo_id, hf_revision, hf_materialize_dir1113 1114 1115def _run_scrape(args: argparse.Namespace, config_path: Path | None) -> None:1116 from slop_farmer.app.pipeline import run_pipeline1117 1118 new_contributor_report = bool(args.new_contributor_report)1119 options = PipelineOptions(1120 repo=RepoRef.parse(args.repo),1121 output_dir=args.output_dir,1122 since=args.since,1123 resume=args.resume,1124 http_timeout=args.http_timeout,1125 http_max_retries=args.http_max_retries,1126 max_issues=args.max_issues,1127 max_prs=args.max_prs,1128 max_issue_comments=args.max_issue_comments,1129 max_reviews_per_pr=args.max_reviews_per_pr,1130 max_review_comments_per_pr=args.max_review_comments_per_pr,1131 fetch_timeline=args.fetch_timeline,1132 new_contributor_report=new_contributor_report,1133 new_contributor_window_days=args.new_contributor_window_days,1134 new_contributor_max_authors=args.new_contributor_max_authors,1135 issue_max_age_days=args.issue_max_age_days,1136 pr_max_age_days=args.pr_max_age_days,1137 )1138 print(run_pipeline(options))1139 1140 1141def _run_refresh_dataset(args: argparse.Namespace, config_path: Path | None) -> None:1142 from slop_farmer.app.dataset_refresh import run_dataset_refresh1143 1144 refresh_defaults = command_defaults("refresh-dataset", config_path=config_path)1145 result = run_dataset_refresh(1146 DatasetRefreshOptions(1147 repo=RepoRef.parse(args.repo),1148 hf_repo_id=args.hf_repo_id,1149 private_hf_repo=args.private_hf_repo,1150 max_issues=args.max_issues,1151 max_prs=args.max_prs,1152 max_issue_comments=args.max_issue_comments,1153 max_reviews_per_pr=args.max_reviews_per_pr,1154 max_review_comments_per_pr=args.max_review_comments_per_pr,1155 fetch_timeline=args.fetch_timeline,1156 new_contributor_report=args.new_contributor_report,1157 new_contributor_window_days=args.new_contributor_window_days,1158 new_contributor_max_authors=args.new_contributor_max_authors,1159 http_timeout=args.http_timeout,1160 http_max_retries=args.http_max_retries,1161 checkpoint_every_comments=args.checkpoint_every_comments,1162 checkpoint_every_prs=args.checkpoint_every_prs,1163 cluster_suppression_rules=tuple(refresh_defaults.get("cluster-suppression-rules", ())),1164 )1165 )1166 print(json.dumps(result, indent=2))1167 1168 1169def _run_analyze(args: argparse.Namespace, config_path: Path | None) -> None:1170 from slop_farmer.reports.analysis import run_analysis1171 1172 analyze_defaults = command_defaults("analyze", config_path=config_path)1173 hf_repo_id, hf_revision, hf_materialize_dir = _resolve_hf_inputs(args)1174 options = AnalysisOptions(1175 snapshot_dir=args.snapshot_dir,1176 output_dir=args.output_dir,1177 output=args.output,1178 hf_repo_id=hf_repo_id,1179 hf_revision=hf_revision,1180 hf_materialize_dir=hf_materialize_dir,1181 ranking_backend=args.ranking_backend,1182 model=args.model,1183 max_clusters=args.max_clusters,1184 hybrid_llm_concurrency=args.hybrid_llm_concurrency,1185 open_prs_only=args.open_prs_only,1186 cached_analysis=bool(analyze_defaults.get("cached_analysis", False)),1187 pr_template_cleanup_mode=str(1188 analyze_defaults.get("pr-template-cleanup-mode", "merge_defaults")1189 ),1190 pr_template_strip_html_comments=bool(1191 analyze_defaults.get("pr-template-strip-html-comments", True)1192 ),1193 pr_template_trim_closing_reference_prefix=bool(1194 analyze_defaults.get("pr-template-trim-closing-reference-prefix", True)1195 ),1196 pr_template_section_patterns=tuple(1197 analyze_defaults.get("pr-template-section-patterns", ())1198 ),1199 pr_template_line_patterns=tuple(analyze_defaults.get("pr-template-line-patterns", ())),1200 cluster_suppression_rules=tuple(analyze_defaults.get("cluster-suppression-rules", ())),