Team Ai
Apppublic

evalstate/diffusers-pr-api

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
cli.py1621 linesDownload Raw Back to app
1from __future__ import annotations2 3import argparse4import json5import sys6from collections.abc import Callable7from pathlib import Path8from typing import Any9 10from slop_farmer.app.duplicate_prs import DEFAULT_FILE_POLICY, FILE_POLICY_CHOICES11from slop_farmer.app_config import command_defaults, extract_cli_config_path12from slop_farmer.config import (13    AnalysisOptions,14    CheckpointImportOptions,15    DashboardDataOptions,16    DatasetRefreshOptions,17    DatasetStatusOptions,18    DeployDashboardOptions,19    MarkdownReportOptions,20    NewContributorReportOptions,21    PipelineOptions,22    PrScopeOptions,23    PrSearchRefreshOptions,24    PublishAnalysisArtifactsOptions,25    RepoRef,26    SaveCacheOptions,27    SnapshotAdoptOptions,28)29from slop_farmer.reports.duplicate_prs import DEFAULT_DUPLICATE_PR_MODEL30 31CommandHandler = Callable[[argparse.Namespace, Path | None], None]32 33 34def _int_at_least(minimum: int) -> Callable[[str], int]:35    def parse(raw: str) -> int:36        value = int(raw)37        if value < minimum:38            raise argparse.ArgumentTypeError(f"expected integer >= {minimum}")39        return value40 41    return parse42 43 44def build_parser(*, config_path: Path | None = None) -> argparse.ArgumentParser:45    defaults = _load_parser_defaults(config_path)46 47    parser = argparse.ArgumentParser(prog="slop-farmer")48    parser.add_argument(49        "--config",50        type=Path,51        help="YAML config file with shared repo/workspace/dashboard defaults.",52    )53    subparsers = parser.add_subparsers(dest="command", required=True)54 55    _add_scrape_parser(subparsers, defaults["scrape"])56    _add_refresh_dataset_parser(subparsers, defaults["refresh-dataset"])57    _add_analyze_parser(subparsers, defaults["analyze"])58    _add_pr_scope_parser(subparsers, defaults["pr-scope"])59    _add_checkpoint_import_parser(subparsers, defaults["import-hf-checkpoint"])60    _add_adopt_snapshot_parser(subparsers, defaults["adopt-snapshot"])61    _add_markdown_report_parser(subparsers)62    _add_duplicate_prs_parser(subparsers)63    _add_pr_search_parser(subparsers, defaults["pr-search"])64    _add_new_contributor_report_parser(subparsers, defaults["new-contributor-report"])65    _add_dashboard_data_parser(subparsers, defaults["dashboard-data"])66    _add_publish_analysis_artifacts_parser(subparsers, defaults["publish-analysis-artifacts"])67    _add_save_cache_parser(subparsers, defaults["save-cache"])68    _add_deploy_dashboard_parser(subparsers, defaults["deploy-dashboard"])69    _add_dataset_status_parser(subparsers, defaults["dataset-status"])70    return parser71 72 73def _load_parser_defaults(config_path: Path | None) -> dict[str, dict[str, Any]]:74    commands = (75        "scrape",76        "refresh-dataset",77        "analyze",78        "import-hf-checkpoint",79        "pr-scope",80        "pr-search",81        "adopt-snapshot",82        "new-contributor-report",83        "dashboard-data",84        "publish-analysis-artifacts",85        "save-cache",86        "deploy-dashboard",87        "dataset-status",88    )89    return {command: command_defaults(command, config_path=config_path) for command in commands}90 91 92# Parser builders93 94 95def _add_scrape_parser(subparsers: Any, defaults: dict[str, Any]) -> None:96    scrape = subparsers.add_parser("scrape", help="Scrape GitHub and write a snapshot dataset.")97    scrape.add_argument(98        "--repo",99        default=defaults.get("repo", "huggingface/transformers"),100        help="GitHub repository in owner/name form.",101    )102    scrape.add_argument("--output-dir", type=Path, default=Path(defaults.get("output-dir", "data")))103    scrape.add_argument("--since", help="Incremental sync lower bound in ISO 8601 format.")104    scrape.add_argument(105        "--resume",106        dest="resume",107        action="store_true",108        default=True,109        help="Resume from the last successful local watermark when --since is not provided.",110    )111    scrape.add_argument(112        "--no-resume",113        dest="resume",114        action="store_false",115        help="Ignore local watermark state and run from scratch unless --since is set.",116    )117    scrape.add_argument(118        "--http-timeout", type=int, default=180, help="Per-request timeout in seconds."119    )120    scrape.add_argument(121        "--http-max-retries", type=int, default=5, help="Retries for transient network failures."122    )123    scrape.add_argument(124        "--max-issues", type=int, default=None, help="Limit total issue endpoint items read."125    )126    scrape.add_argument(127        "--max-prs", type=int, default=None, help="Limit pull requests to hydrate in detail."128    )129    scrape.add_argument(130        "--issue-max-age-days",131        type=int,132        default=defaults.get("issue-max-age-days"),133        help="Optional created_at age cap for issues included in the snapshot.",134    )135    scrape.add_argument(136        "--pr-max-age-days",137        type=int,138        default=defaults.get("pr-max-age-days"),139        help="Optional created_at age cap for pull requests included in the snapshot.",140    )141    scrape.add_argument(142        "--max-issue-comments", type=int, default=None, help="Limit issue comment rows."143    )144    scrape.add_argument(145        "--max-reviews-per-pr", type=int, default=None, help="Limit review rows per PR."146    )147    scrape.add_argument(148        "--max-review-comments-per-pr",149        type=int,150        default=None,151        help="Limit inline review comment rows per PR.",152    )153    scrape.add_argument(154        "--fetch-timeline",155        action="store_true",156        default=bool(defaults.get("fetch-timeline", False)),157        help="Fetch issue timeline events for linkage rows.",158    )159    scrape.add_argument(160        "--new-contributor-report",161        dest="new_contributor_report",162        action="store_true",163        default=defaults.get("new-contributor-report"),164        help="Generate new contributor dataset/report artifacts for the local snapshot.",165    )166    scrape.add_argument(167        "--no-new-contributor-report",168        dest="new_contributor_report",169        action="store_false",170        help="Skip new contributor dataset/report generation.",171    )172    scrape.add_argument(173        "--new-contributor-window-days",174        type=int,175        default=int(defaults.get("new-contributor-window-days", 42)),176        help="Recent public activity window for contributor enrichment.",177    )178    scrape.add_argument(179        "--new-contributor-max-authors",180        type=int,181        default=int(defaults.get("new-contributor-max-authors", 25)),182        help="Maximum number of contributors to include in the new contributor report. Use 0 for no cap.",183    )184 185 186def _add_refresh_dataset_parser(subparsers: Any, defaults: dict[str, Any]) -> None:187    refresh = subparsers.add_parser(188        "refresh-dataset",189        help="Refresh the canonical Hugging Face dataset repo from remote watermark state.",190    )191    refresh.add_argument(192        "--repo",193        default=defaults.get("repo", "huggingface/transformers"),194        help="GitHub repository in owner/name form.",195    )196    refresh.add_argument(197        "--hf-repo-id",198        default=defaults.get("hf-repo-id"),199        required=defaults.get("hf-repo-id") is None,200        help="Canonical Hugging Face dataset repo id to refresh.",201    )202    refresh.add_argument("--max-issues", type=int, default=defaults.get("max-issues"))203    refresh.add_argument("--max-prs", type=int, default=defaults.get("max-prs"))204    refresh.add_argument(205        "--max-issue-comments", type=int, default=defaults.get("max-issue-comments")206    )207    refresh.add_argument(208        "--max-reviews-per-pr", type=int, default=defaults.get("max-reviews-per-pr")209    )210    refresh.add_argument(211        "--max-review-comments-per-pr",212        type=int,213        default=defaults.get("max-review-comments-per-pr"),214    )215    refresh.add_argument(216        "--fetch-timeline",217        action="store_true",218        default=bool(defaults.get("fetch-timeline", False)),219    )220    refresh.add_argument(221        "--new-contributor-report",222        dest="new_contributor_report",223        action="store_true",224        default=bool(defaults.get("new-contributor-report", True)),225    )226    refresh.add_argument(227        "--no-new-contributor-report",228        dest="new_contributor_report",229        action="store_false",230    )231    refresh.add_argument(232        "--new-contributor-window-days",233        type=int,234        default=int(defaults.get("new-contributor-window-days", 42)),235    )236    refresh.add_argument(237        "--new-contributor-max-authors",238        type=int,239        default=int(defaults.get("new-contributor-max-authors", 25)),240    )241    refresh.add_argument("--http-timeout", type=int, default=300)242    refresh.add_argument("--http-max-retries", type=int, default=8)243    refresh.add_argument("--checkpoint-every-comments", type=int, default=1000)244    refresh.add_argument("--checkpoint-every-prs", type=int, default=25)245    refresh.add_argument(246        "--private-hf-repo",247        dest="private_hf_repo",248        action="store_true",249        default=bool(defaults.get("private-hf-repo", False)),250        help="Create the target dataset repo as private if needed.",251    )252    refresh.add_argument(253        "--private",254        dest="private_hf_repo",255        action="store_true",256        help=argparse.SUPPRESS,257    )258 259 260def _add_analyze_parser(subparsers: Any, defaults: dict[str, Any]) -> None:261    analyze = subparsers.add_parser(262        "analyze",263        help="Analyze a snapshot and write a local JSON report. Canonical publication is separate.",264    )265    analyze.add_argument(266        "--snapshot-dir",267        type=Path,268        help="Snapshot directory to analyze. Defaults to the latest local snapshot.",269    )270    analyze.add_argument(271        "--output-dir", type=Path, default=Path(defaults.get("output-dir", "data"))272    )273    analyze.add_argument("--output", type=Path, help="Output path for the analysis JSON.")274    analyze.add_argument(275        "--hf-repo-id",276        default=defaults.get("hf-repo-id"),277        help="Analyze a canonical Hugging Face dataset repo by materializing a self-consistent published snapshot locally.",278    )279    analyze.add_argument(280        "--hf-revision",281        default=defaults.get("hf-revision"),282        help="Optional Hub revision for metadata and README download.",283    )284    analyze.add_argument(285        "--hf-materialize-dir",286        type=Path,287        default=Path(defaults["hf-materialize-dir"])288        if defaults.get("hf-materialize-dir")289        else None,290        help="Optional local directory used when materializing an HF dataset snapshot.",291    )292    analyze.add_argument(293        "--ranking-backend",294        choices=("hybrid", "deterministic"),295        default=defaults.get("ranking-backend", "hybrid"),296        help="Whether to use deterministic-only ranking or optional fast-agent enrichment.",297    )298    analyze.add_argument(299        "--model",300        default=defaults.get("model", "gpt-5.4-mini?service_tier=flex"),301        help="Model string used by fast-agent when enabled.",302    )303    analyze.add_argument(304        "--max-clusters",305        type=int,306        default=int(defaults.get("max-clusters", 10)),307        help="Maximum number of meta clusters to include in the report.",308    )309    analyze.add_argument(310        "--hybrid-llm-concurrency",311        type=_int_at_least(1),312        default=int(defaults.get("hybrid-llm-concurrency", 1)),313        help=(314            "Maximum number of hybrid LLM review units to run at once. "315            "Use 1 to minimize provider pressure."316        ),317    )318    analyze.add_argument(319        "--open-prs-only",320        action="store_true",321        default=bool(defaults.get("open-prs-only", False)),322        help="Restrict PR analysis/clustering to open PRs only. Draft PRs are still included.",323    )324 325 326def _add_pr_scope_parser(subparsers: Any, defaults: dict[str, Any]) -> None:327    pr_scope = subparsers.add_parser(328        "pr-scope", help="Cluster open PRs by holistic file/scope overlap."329    )330    pr_scope.add_argument(331        "--snapshot-dir",332        type=Path,333        help="Snapshot directory to analyze. Defaults to the latest local snapshot.",334    )335    pr_scope.add_argument(336        "--output-dir", type=Path, default=Path(defaults.get("output-dir", "data"))337    )338    pr_scope.add_argument(339        "--output",340        type=Path,341        help="Output path for the PR scope JSON. Defaults next to the snapshot.",342    )343    pr_scope.add_argument(344        "--hf-repo-id",345        default=defaults.get("hf-repo-id"),346        help="Analyze a Hugging Face dataset repo by materializing its parquet export locally.",347    )348    pr_scope.add_argument(349        "--hf-revision",350        default=defaults.get("hf-revision"),351        help="Optional Hub revision for metadata and README download.",352    )353    pr_scope.add_argument(354        "--hf-materialize-dir",355        type=Path,356        default=Path(defaults["hf-materialize-dir"])357        if defaults.get("hf-materialize-dir")358        else None,359        help="Optional local directory used when materializing an HF dataset snapshot.",360    )361 362 363def _add_checkpoint_import_parser(subparsers: Any, defaults: dict[str, Any]) -> None:364    checkpoint_import = subparsers.add_parser(365        "import-hf-checkpoint",366        help="Import a checkpoint snapshot from an HF dataset repo into a clean local snapshot.",367    )368    checkpoint_import.add_argument(369        "--source-repo-id",370        default=defaults.get("source-repo-id", "burtenshaw/transformers-pr-slop-dataset"),371        help="Source Hugging Face dataset repo id containing checkpoint folders.",372    )373    checkpoint_import.add_argument(374        "--output-dir",375        type=Path,376        default=Path(defaults.get("output-dir", "eval_data")),377        help="Local root directory where the imported snapshot should be written.",378    )379    checkpoint_import.add_argument(380        "--checkpoint-id",381        help="Optional checkpoint snapshot id. Defaults to the latest viable checkpoint.",382    )383    checkpoint_import.add_argument(384        "--checkpoint-root",385        choices=("checkpoints", "_checkpoints"),386        help="Optional checkpoint root directory. Defaults to auto-detect.",387    )388    checkpoint_import.add_argument(389        "--publish-repo-id",390        help="Optional HF dataset repo id to publish the imported clean snapshot to.",391    )392    checkpoint_import.add_argument(393        "--private-hf-repo",394        action="store_true",395        help="Create the publish target as private when --publish-repo-id is used.",396    )397    checkpoint_import.add_argument(398        "--force",399        action="store_true",400        help="Overwrite an existing imported snapshot directory if present.",401    )402 403 404def _add_adopt_snapshot_parser(subparsers: Any, defaults: dict[str, Any]) -> None:405    adopt_snapshot = subparsers.add_parser(406        "adopt-snapshot",407        help="Mark an existing snapshot as the current pipeline base so the next scrape resumes from it.",408    )409    adopt_snapshot.add_argument(410        "--snapshot-dir", type=Path, required=True, help="Existing local snapshot directory."411    )412    adopt_snapshot.add_argument(413        "--output-dir",414        type=Path,415        default=Path(defaults.get("output-dir", "data")),416        help="Pipeline workspace root where state/ and snapshots/latest.json should be written.",417    )418    adopt_snapshot.add_argument(419        "--next-since",420        help="Optional explicit watermark timestamp. Defaults to snapshot watermark.next_since, crawl_started_at, or extracted_at.",421    )422 423 424def _add_markdown_report_parser(subparsers: Any) -> None:425    markdown = subparsers.add_parser(426        "markdown-report", help="Render a markdown report from an analysis JSON file."427    )428    markdown.add_argument(429        "--input", type=Path, required=True, help="Path to an existing analysis JSON report."430    )431    markdown.add_argument(432        "--output",433        type=Path,434        help="Output path for the markdown report. Defaults next to the input JSON.",435    )436    markdown.add_argument(437        "--snapshot-dir",438        type=Path,439        help="Optional snapshot directory containing issues.parquet and pull_requests.parquet. Defaults to the input JSON parent directory.",440    )441 442 443def _add_duplicate_prs_parser(subparsers: Any) -> None:444    duplicate_prs = subparsers.add_parser(445        "duplicate-prs",446        help="List or merge mergeable duplicate PR clusters from hybrid-enriched analysis.",447    )448    duplicate_prs_subparsers = duplicate_prs.add_subparsers(449        dest="duplicate_prs_command", required=True450    )451 452    duplicate_list = duplicate_prs_subparsers.add_parser(453        "list",454        help="List mergeable duplicate PR clusters from a hybrid-enriched analysis report.",455    )456    duplicate_list_source = duplicate_list.add_mutually_exclusive_group(required=True)457    duplicate_list_source.add_argument(458        "--report", type=Path, help="Path to an analysis JSON report."459    )460    duplicate_list_source.add_argument(461        "--snapshot-dir", type=Path, help="Snapshot directory to analyze."462    )463    duplicate_list.add_argument(464        "--limit", type=int, default=10, help="Maximum number of mergeable clusters to print."465    )466    duplicate_list.add_argument(467        "--model",468        default=DEFAULT_DUPLICATE_PR_MODEL,469        help="Model string used for hybrid analysis and duplicate-PR mergeability gating.",470    )471 472    duplicate_merge = duplicate_prs_subparsers.add_parser(473        "merge",474        help="Use Codex to synthesize and publish a minimal upstream PR for a mergeable duplicate cluster.",475    )476    duplicate_merge_source = duplicate_merge.add_mutually_exclusive_group(required=True)477    duplicate_merge_source.add_argument(478        "--report", type=Path, help="Path to an analysis JSON report."479    )480    duplicate_merge_source.add_argument(481        "--snapshot-dir", type=Path, help="Snapshot directory to analyze."482    )483    duplicate_merge.add_argument(484        "--repo-dir",485        type=Path,486        required=True,487        help="Local upstream repository checkout used for the synthesis worktree.",488    )489    duplicate_merge.add_argument(490        "--upstream-repo",491        help="Optional owner/name override for the upstream target repository.",492    )493    duplicate_merge.add_argument(494        "--upstream-remote",495        default="origin",496        help="Remote in --repo-dir that points at the upstream repository. Defaults to origin.",497    )498    duplicate_merge.add_argument(499        "--fork-remote",500        default="fork",501        help="Remote in the synthesis worktree used for pushing the branch. Defaults to fork.",502    )503    duplicate_merge.add_argument("--cluster-id", help="Optional cluster override.")504    duplicate_merge.add_argument(505        "--fork-repo",506        help="Optional owner/name override for the fork push target. Overrides --fork-owner when both are set.",507    )508    duplicate_merge.add_argument(509        "--fork-owner",510        help="Optional GitHub fork owner override. Defaults to the authenticated user.",511    )512    duplicate_merge.add_argument(513        "--file-policy",514        choices=FILE_POLICY_CHOICES,515        default=DEFAULT_FILE_POLICY,516        help="Changed-file policy enforced on the synthesized branch.",517    )518    duplicate_merge.add_argument(519        "--model",520        default=DEFAULT_DUPLICATE_PR_MODEL,521        help="Model string used for hybrid analysis, mergeability gating, and Codex synthesis.",522    )523 524 525def _add_pr_search_parser(subparsers: Any, defaults: dict[str, Any]) -> None:526    pr_search = subparsers.add_parser(527        "pr-search",528        help="Refresh and query the DuckDB-backed PR code-similarity index.",529    )530    pr_search_subparsers = pr_search.add_subparsers(dest="pr_search_command", required=True)531 532    refresh = pr_search_subparsers.add_parser(533        "refresh",534        help="Refresh the PR code-similarity index from a local snapshot or HF dataset repo.",535    )536    refresh_source = refresh.add_mutually_exclusive_group()537    refresh_source.add_argument(538        "--snapshot-dir",539        type=Path,540        help="Snapshot directory to index. Defaults to the latest local snapshot.",541    )542    refresh_source.add_argument(543        "--hf-repo-id",544        default=defaults.get("hf-repo-id"),545        help="Hugging Face dataset repo id to materialize before indexing.",546    )547    refresh.add_argument(548        "--hf-revision",549        default=defaults.get("hf-revision"),550        help="Optional Hub revision for metadata and README download.",551    )552    refresh.add_argument(553        "--hf-materialize-dir",554        type=Path,555        default=Path(defaults["hf-materialize-dir"])556        if defaults.get("hf-materialize-dir")557        else None,558        help="Optional local directory used when materializing an HF dataset snapshot.",559    )560    refresh.add_argument(561        "--output-dir",562        type=Path,563        default=Path(defaults.get("output-dir", "data")),564        help="Workspace root used for latest snapshot resolution and default DB placement.",565    )566    refresh.add_argument(567        "--db",568        type=Path,569        default=Path(defaults["db"]) if defaults.get("db") else None,570        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",571    )572    refresh.add_argument("--limit-prs", type=int, help="Optional cap on indexed PRs.")573    refresh.add_argument(574        "--include-drafts",575        action="store_true",576        default=bool(defaults.get("include-drafts", False)),577        help="Include draft PRs in the indexed universe.",578    )579    refresh.add_argument(580        "--include-closed",581        action="store_true",582        default=bool(defaults.get("include-closed", False)),583        help="Include closed PRs in the indexed universe.",584    )585    refresh.add_argument(586        "--replace-active",587        dest="replace_active",588        action="store_true",589        default=True,590        help="Activate the new run on success. Enabled by default.",591    )592    refresh.add_argument(593        "--no-replace-active",594        dest="replace_active",595        action="store_false",596        help="Write the new run without switching the active run pointer.",597    )598 599    similar = pr_search_subparsers.add_parser(600        "similar", help="Show similar PRs for one indexed pull request."601    )602    similar.add_argument("pr_number", type=int, help="Pull request number to query.")603    similar.add_argument(604        "--db",605        type=Path,606        default=Path(defaults["db"]) if defaults.get("db") else None,607        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",608    )609    similar.add_argument(610        "--output-dir",611        type=Path,612        default=Path(defaults.get("output-dir", "data")),613    )614    similar.add_argument("--repo", help="Optional repo override when the DB holds multiple repos.")615    similar.add_argument("--limit", type=int, default=10, help="Maximum number of rows to show.")616    similar.add_argument("--json", action="store_true", help="Emit machine-readable JSON.")617 618    probe_github = pr_search_subparsers.add_parser(619        "probe-github",620        help="Fetch one live GitHub PR and compare it against the active indexed scope features.",621    )622    probe_github.add_argument("pr_number", type=int, help="Pull request number to probe.")623    probe_github.add_argument(624        "--repo",625        help="GitHub repository in owner/name form. Defaults to the active repo in the DB.",626    )627    probe_github.add_argument(628        "--db",629        type=Path,630        default=Path(defaults["db"]) if defaults.get("db") else None,631        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",632    )633    probe_github.add_argument(634        "--output-dir",635        type=Path,636        default=Path(defaults.get("output-dir", "data")),637    )638    probe_github.add_argument(639        "--limit",640        type=int,641        default=10,642        help="Maximum number of similar PR rows to show.",643    )644    probe_github.add_argument("--json", action="store_true", help="Emit machine-readable JSON.")645 646    candidate_clusters = pr_search_subparsers.add_parser(647        "candidate-clusters",648        help="Show candidate scope clusters for one indexed pull request.",649    )650    candidate_clusters.add_argument("pr_number", type=int, help="Pull request number to query.")651    candidate_clusters.add_argument(652        "--db",653        type=Path,654        default=Path(defaults["db"]) if defaults.get("db") else None,655        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",656    )657    candidate_clusters.add_argument(658        "--output-dir",659        type=Path,660        default=Path(defaults.get("output-dir", "data")),661    )662    candidate_clusters.add_argument(663        "--repo", help="Optional repo override when the DB holds multiple repos."664    )665    candidate_clusters.add_argument(666        "--limit", type=int, default=5, help="Maximum number of rows to show."667    )668    candidate_clusters.add_argument("--json", action="store_true", help="Emit JSON.")669 670    cluster = pr_search_subparsers.add_parser("cluster", help="Inspect one scope cluster.")671    cluster_subparsers = cluster.add_subparsers(dest="pr_search_cluster_command", required=True)672    cluster_show = cluster_subparsers.add_parser("show", help="Show cluster details.")673    cluster_show.add_argument("cluster_id", help="Cluster identifier.")674    cluster_show.add_argument(675        "--db",676        type=Path,677        default=Path(defaults["db"]) if defaults.get("db") else None,678        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",679    )680    cluster_show.add_argument(681        "--output-dir",682        type=Path,683        default=Path(defaults.get("output-dir", "data")),684    )685    cluster_show.add_argument("--repo", help="Optional repo override.")686    cluster_show.add_argument("--json", action="store_true", help="Emit JSON.")687 688    explain_pair = pr_search_subparsers.add_parser(689        "explain-pair",690        help="Explain one PR pair, falling back to on-demand scoring when needed.",691    )692    explain_pair.add_argument("left_pr_number", type=int)693    explain_pair.add_argument("right_pr_number", type=int)694    explain_pair.add_argument(695        "--db",696        type=Path,697        default=Path(defaults["db"]) if defaults.get("db") else None,698        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",699    )700    explain_pair.add_argument(701        "--output-dir",702        type=Path,703        default=Path(defaults.get("output-dir", "data")),704    )705    explain_pair.add_argument("--repo", help="Optional repo override.")706    explain_pair.add_argument("--json", action="store_true", help="Emit JSON.")707 708    status = pr_search_subparsers.add_parser("status", help="Show the active PR search run.")709    status.add_argument(710        "--db",711        type=Path,712        default=Path(defaults["db"]) if defaults.get("db") else None,713        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",714    )715    status.add_argument(716        "--output-dir",717        type=Path,718        default=Path(defaults.get("output-dir", "data")),719    )720    status.add_argument("--repo", help="Optional repo override.")721    status.add_argument("--json", action="store_true", help="Emit JSON.")722 723    contributor = pr_search_subparsers.add_parser(724        "contributor", help="Show indexed contributor summary for one author login."725    )726    contributor.add_argument("login", help="GitHub author login to query.")727    contributor.add_argument(728        "--db",729        type=Path,730        default=Path(defaults["db"]) if defaults.get("db") else None,731        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",732    )733    contributor.add_argument(734        "--output-dir",735        type=Path,736        default=Path(defaults.get("output-dir", "data")),737    )738    contributor.add_argument("--repo", help="Optional repo override.")739    contributor.add_argument("--json", action="store_true", help="Emit JSON.")740 741    contributor_prs = pr_search_subparsers.add_parser(742        "contributor-prs", help="List indexed PRs for one contributor login."743    )744    contributor_prs.add_argument("login", help="GitHub author login to query.")745    contributor_prs.add_argument(746        "--db",747        type=Path,748        default=Path(defaults["db"]) if defaults.get("db") else None,749        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",750    )751    contributor_prs.add_argument(752        "--output-dir",753        type=Path,754        default=Path(defaults.get("output-dir", "data")),755    )756    contributor_prs.add_argument("--repo", help="Optional repo override.")757    contributor_prs.add_argument("--limit", type=int, default=20, help="Maximum rows to show.")758    contributor_prs.add_argument("--json", action="store_true", help="Emit JSON.")759 760    pr_contributor = pr_search_subparsers.add_parser(761        "pr-contributor", help="Show contributor summary for the author of one indexed PR."762    )763    pr_contributor.add_argument("pr_number", type=int, help="Pull request number to query.")764    pr_contributor.add_argument(765        "--db",766        type=Path,767        default=Path(defaults["db"]) if defaults.get("db") else None,768        help="DuckDB file path. Defaults to <output-dir>/state/pr-search.duckdb.",769    )770    pr_contributor.add_argument(771        "--output-dir",772        type=Path,773        default=Path(defaults.get("output-dir", "data")),774    )775    pr_contributor.add_argument("--repo", help="Optional repo override.")776    pr_contributor.add_argument("--json", action="store_true", help="Emit JSON.")777 778 779def _add_new_contributor_report_parser(subparsers: Any, defaults: dict[str, Any]) -> None:780    new_contributor = subparsers.add_parser(781        "new-contributor-report",782        help="Render a markdown report for newly observed contributors in a snapshot.",783    )784    new_contributor.add_argument(785        "--snapshot-dir",786        type=Path,787        help="Snapshot directory to inspect. Defaults to the latest local snapshot.",788    )789    new_contributor.add_argument(790        "--output-dir", type=Path, default=Path(defaults.get("output-dir", "data"))791    )792    new_contributor.add_argument(793        "--output",794        type=Path,795        help="Output path for the markdown report. Defaults next to the snapshot.",796    )797    new_contributor.add_argument(798        "--json-output", type=Path, help="Optional JSON output path. Defaults next to the snapshot."799    )800    new_contributor.add_argument(801        "--hf-repo-id",802        default=defaults.get("hf-repo-id"),803        help="Analyze a Hugging Face dataset repo by materializing its parquet export locally.",804    )805    new_contributor.add_argument(806        "--hf-revision",807        default=defaults.get("hf-revision"),808        help="Optional Hub revision for metadata and README download.",809    )810    new_contributor.add_argument(811        "--hf-materialize-dir",812        type=Path,813        default=Path(defaults["hf-materialize-dir"])814        if defaults.get("hf-materialize-dir")815        else None,816        help="Optional local directory used when materializing an HF dataset snapshot.",817    )818    new_contributor.add_argument(819        "--window-days",820        type=int,821        default=int(defaults.get("window-days", 42)),822        help="Recent public activity window for contributor enrichment.",823    )824    new_contributor.add_argument(825        "--max-authors",826        type=int,827        default=int(defaults.get("max-authors", 25)),828        help="Maximum number of contributors to include. Use 0 for no cap.",829    )830 831 832def _add_dashboard_data_parser(subparsers: Any, defaults: dict[str, Any]) -> None:833    dashboard = subparsers.add_parser(834        "dashboard-data", help="Export frontend-ready JSON for the static dashboard."835    )836    dashboard.add_argument(837        "--snapshot-dir",838        type=Path,839        help="Snapshot directory to export. Defaults to the latest local snapshot.",840    )841    dashboard.add_argument(842        "--output-dir",843        type=Path,844        default=Path(defaults.get("output-dir", "web/public/data")),845    )846    dashboard.add_argument(847        "--analysis-input",848        type=Path,849        help="Optional analysis report JSON override. Defaults to canonical published current analysis when available, otherwise falls back to snapshot-local analysis files.",850    )851    dashboard.add_argument(852        "--contributors-input",853        type=Path,854        help="Optional contributor report JSON override. Defaults to the materialized snapshot's new-contributors-report.json.",855    )856    dashboard.add_argument(857        "--pr-scope-input",858        type=Path,859        help="Optional PR scope cluster JSON override. Defaults to the materialized snapshot's pr-scope-clusters.json.",860    )861    dashboard.add_argument(862        "--hf-repo-id",863        default=defaults.get("hf-repo-id"),864        help="Materialize the canonical Hugging Face dataset repo instead of using the latest local snapshot.",865    )866    dashboard.add_argument(867        "--hf-revision",868        default=defaults.get("hf-revision"),869        help="Optional Hub revision for metadata and README download.",870    )871    dashboard.add_argument(872        "--hf-materialize-dir",873        type=Path,874        default=Path(defaults["hf-materialize-dir"])875        if defaults.get("hf-materialize-dir")876        else None,877        help="Optional local directory used when materializing an HF dataset snapshot.",878    )879    dashboard.add_argument(880        "--window-days",881        type=int,882        default=int(defaults.get("window-days", 14)),883        help="Recent PR window to expose in the dashboard.",884    )885 886 887def _add_publish_analysis_artifacts_parser(subparsers: Any, defaults: dict[str, Any]) -> None:888    publish_analysis = subparsers.add_parser(889        "publish-analysis-artifacts",890        help="Publish archived and optional canonical hybrid analysis artifacts to a dataset repo.",891    )892    publish_analysis.add_argument(893        "--output-dir",894        type=Path,895        default=Path(defaults.get("output-dir", "data")),896        help="Pipeline workspace root containing snapshots/latest.json.",897    )898    publish_analysis.add_argument(899        "--snapshot-dir",900        type=Path,901        help="Optional explicit snapshot directory containing analysis-report-hybrid.json.",902    )903    publish_analysis.add_argument(904        "--analysis-input",905        type=Path,906        help="Optional explicit hybrid analysis report JSON to publish instead of snapshot-dir discovery.",907    )908    publish_analysis.add_argument(909        "--hf-repo-id",910        default=defaults.get("hf-repo-id"),911        required=defaults.get("hf-repo-id") is None,912        help="Target Hugging Face dataset repo id.",913    )914    publish_analysis.add_argument("--analysis-id", required=True, help="Immutable analysis run id.")915    publish_analysis.add_argument(916        "--canonical",917        action="store_true",918        default=bool(defaults.get("canonical", False)),919        help="Also update the stable analysis/current canonical alias.",920    )921    publish_analysis.add_argument(922        "--save-cache",923        action="store_true",924        default=bool(defaults.get("save-cache", False)),925        help="Also upload snapshot-local analysis-state/ as mutable operational cache at repo-root analysis-state/.",926    )927    publish_analysis.add_argument(928        "--private-hf-repo",929        action="store_true",930        default=bool(defaults.get("private-hf-repo", False)),931        help="Create the target dataset repo as private if needed.",932    )933 934 935def _add_save_cache_parser(subparsers: Any, defaults: dict[str, Any]) -> None:936    save_cache = subparsers.add_parser(937        "save-cache",938        help="Upload snapshot-local analysis-state/ as mutable operational cache to a dataset repo.",939    )940    save_cache.add_argument(941        "--output-dir",942        type=Path,943        default=Path(defaults.get("output-dir", "data")),944        help="Pipeline workspace root containing snapshots/latest.json.",945    )946    save_cache.add_argument(947        "--snapshot-dir",948        type=Path,949        help="Optional explicit snapshot directory containing analysis-state/.",950    )951    save_cache.add_argument(952        "--hf-repo-id",953        default=defaults.get("hf-repo-id"),954        required=defaults.get("hf-repo-id") is None,955        help="Target Hugging Face dataset repo id.",956    )957    save_cache.add_argument(958        "--private-hf-repo",959        action="store_true",960        default=bool(defaults.get("private-hf-repo", False)),961        help="Create the target dataset repo as private if needed.",962    )963 964 965def _add_deploy_dashboard_parser(subparsers: Any, defaults: dict[str, Any]) -> None:966    deploy_dashboard = subparsers.add_parser(967        "deploy-dashboard",968        help="Build and publish the static dashboard to a Hugging Face Space from a materialized dataset view.",969    )970    deploy_dashboard.add_argument(971        "--pipeline-data-dir",972        type=Path,973        default=Path(defaults.get("pipeline-data-dir", "data")),974    )975    deploy_dashboard.add_argument(976        "--web-dir", type=Path, default=Path(defaults.get("web-dir", "web"))977    )978    deploy_dashboard.add_argument(979        "--snapshot-dir",980        type=Path,981        help="Optional snapshot directory to publish. Defaults to the latest snapshot in --pipeline-data-dir.",982    )983    deploy_dashboard.add_argument(984        "--analysis-input",985        type=Path,986        help="Optional analysis report JSON override. Omit to prefer canonical published current analysis when available.",987    )988    deploy_dashboard.add_argument(989        "--contributors-input",990        type=Path,991        help="Optional contributor report JSON override.",992    )993    deploy_dashboard.add_argument(994        "--pr-scope-input",995        type=Path,996        help="Optional PR scope cluster JSON override.",997    )998    deploy_dashboard.add_argument(999        "--hf-repo-id",1000        default=defaults.get("hf-repo-id"),1001        help="Materialize the canonical Hugging Face dataset repo instead of using the latest local snapshot.",1002    )1003    deploy_dashboard.add_argument(1004        "--hf-revision",1005        default=defaults.get("hf-revision"),1006        help="Optional Hub revision for metadata and README download.",1007    )1008    deploy_dashboard.add_argument(1009        "--hf-materialize-dir",1010        type=Path,1011        default=Path(defaults["hf-materialize-dir"])1012        if defaults.get("hf-materialize-dir")1013        else None,1014        help="Optional local directory used when materializing an HF dataset snapshot.",1015    )1016    deploy_dashboard.add_argument(1017        "--refresh-contributors",1018        action="store_true",1019        default=bool(defaults.get("refresh-contributors", False)),1020    )1021    deploy_dashboard.add_argument(1022        "--dashboard-window-days",1023        type=int,1024        default=int(defaults.get("dashboard-window-days", 14)),1025    )1026    deploy_dashboard.add_argument(1027        "--contributor-window-days",1028        type=int,1029        default=int(1030            defaults.get("contributor-window-days", defaults.get("dashboard-window-days", 14))1031        ),1032    )1033    deploy_dashboard.add_argument(1034        "--contributor-max-authors",1035        type=int,1036        default=int(defaults.get("contributor-max-authors", 0)),1037    )1038    deploy_dashboard.add_argument(1039        "--private-space",1040        action="store_true",1041        default=bool(defaults.get("private-space", False)),1042    )1043    deploy_dashboard.add_argument(1044        "--commit-message",1045        default=defaults.get("commit-message", "Deploy dashboard"),1046    )1047    deploy_dashboard.add_argument(1048        "--space-id",1049        default=defaults.get("space-id"),1050        help="Hugging Face Space repo id.",1051    )1052    deploy_dashboard.add_argument("--space-title", default=defaults.get("space-title"))1053    deploy_dashboard.add_argument("--space-emoji", default=defaults.get("space-emoji", "๐Ÿ“Š"))1054    deploy_dashboard.add_argument(1055        "--space-color-from", default=defaults.get("space-color-from", "indigo")1056    )1057    deploy_dashboard.add_argument(1058        "--space-color-to", default=defaults.get("space-color-to", "blue")1059    )1060    deploy_dashboard.add_argument(1061        "--space-short-description",1062        default=defaults.get(1063            "space-short-description", "Static dashboard for the slop-farmer PR analysis pipeline."1064        ),1065    )1066    deploy_dashboard.add_argument("--dataset-id", default=defaults.get("dataset-id"))1067    deploy_dashboard.add_argument(1068        "--space-tags", default=defaults.get("space-tags", "dashboard,static")1069    )1070 1071 1072def _add_dataset_status_parser(subparsers: Any, defaults: dict[str, Any]) -> None:1073    dataset_status = subparsers.add_parser(1074        "dataset-status",1075        help="Inspect canonical dataset freshness and the local latest pointer.",1076    )1077    dataset_status.add_argument("--repo", default=defaults.get("repo"))1078    dataset_status.add_argument(1079        "--output-dir",1080        type=Path,1081        default=Path(defaults.get("output-dir", "data")),1082        help="Local workspace root containing snapshots/latest.json.",1083    )1084    dataset_status.add_argument(1085        "--hf-repo-id",1086        default=defaults.get("hf-repo-id"),1087        help="Canonical Hugging Face dataset repo id to inspect.",1088    )1089    dataset_status.add_argument(1090        "--hf-revision",1091        default=defaults.get("hf-revision"),1092        help="Optional Hub revision for metadata and README download.",1093    )1094    dataset_status.add_argument("--json", action="store_true", help="Emit machine-readable JSON.")1095 1096 1097# Dispatch helpers1098 1099 1100def _explicit_flag_present(flag: str) -> bool:1101    return any(arg == flag or arg.startswith(f"{flag}=") for arg in sys.argv[1:])1102 1103 1104def _resolve_hf_inputs(args: argparse.Namespace) -> tuple[str | None, str | None, Path | None]:1105    hf_repo_id = args.hf_repo_id1106    hf_revision = args.hf_revision1107    hf_materialize_dir = args.hf_materialize_dir1108    if args.snapshot_dir is not None and not _explicit_flag_present("--hf-repo-id"):1109        hf_repo_id = None1110        hf_revision = None1111        hf_materialize_dir = None1112    return hf_repo_id, hf_revision, hf_materialize_dir1113 1114 1115def _run_scrape(args: argparse.Namespace, config_path: Path | None) -> None:1116    from slop_farmer.app.pipeline import run_pipeline1117 1118    new_contributor_report = bool(args.new_contributor_report)1119    options = PipelineOptions(1120        repo=RepoRef.parse(args.repo),1121        output_dir=args.output_dir,1122        since=args.since,1123        resume=args.resume,1124        http_timeout=args.http_timeout,1125        http_max_retries=args.http_max_retries,1126        max_issues=args.max_issues,1127        max_prs=args.max_prs,1128        max_issue_comments=args.max_issue_comments,1129        max_reviews_per_pr=args.max_reviews_per_pr,1130        max_review_comments_per_pr=args.max_review_comments_per_pr,1131        fetch_timeline=args.fetch_timeline,1132        new_contributor_report=new_contributor_report,1133        new_contributor_window_days=args.new_contributor_window_days,1134        new_contributor_max_authors=args.new_contributor_max_authors,1135        issue_max_age_days=args.issue_max_age_days,1136        pr_max_age_days=args.pr_max_age_days,1137    )1138    print(run_pipeline(options))1139 1140 1141def _run_refresh_dataset(args: argparse.Namespace, config_path: Path | None) -> None:1142    from slop_farmer.app.dataset_refresh import run_dataset_refresh1143 1144    refresh_defaults = command_defaults("refresh-dataset", config_path=config_path)1145    result = run_dataset_refresh(1146        DatasetRefreshOptions(1147            repo=RepoRef.parse(args.repo),1148            hf_repo_id=args.hf_repo_id,1149            private_hf_repo=args.private_hf_repo,1150            max_issues=args.max_issues,1151            max_prs=args.max_prs,1152            max_issue_comments=args.max_issue_comments,1153            max_reviews_per_pr=args.max_reviews_per_pr,1154            max_review_comments_per_pr=args.max_review_comments_per_pr,1155            fetch_timeline=args.fetch_timeline,1156            new_contributor_report=args.new_contributor_report,1157            new_contributor_window_days=args.new_contributor_window_days,1158            new_contributor_max_authors=args.new_contributor_max_authors,1159            http_timeout=args.http_timeout,1160            http_max_retries=args.http_max_retries,1161            checkpoint_every_comments=args.checkpoint_every_comments,1162            checkpoint_every_prs=args.checkpoint_every_prs,1163            cluster_suppression_rules=tuple(refresh_defaults.get("cluster-suppression-rules", ())),1164        )1165    )1166    print(json.dumps(result, indent=2))1167 1168 1169def _run_analyze(args: argparse.Namespace, config_path: Path | None) -> None:1170    from slop_farmer.reports.analysis import run_analysis1171 1172    analyze_defaults = command_defaults("analyze", config_path=config_path)1173    hf_repo_id, hf_revision, hf_materialize_dir = _resolve_hf_inputs(args)1174    options = AnalysisOptions(1175        snapshot_dir=args.snapshot_dir,1176        output_dir=args.output_dir,1177        output=args.output,1178        hf_repo_id=hf_repo_id,1179        hf_revision=hf_revision,1180        hf_materialize_dir=hf_materialize_dir,1181        ranking_backend=args.ranking_backend,1182        model=args.model,1183        max_clusters=args.max_clusters,1184        hybrid_llm_concurrency=args.hybrid_llm_concurrency,1185        open_prs_only=args.open_prs_only,1186        cached_analysis=bool(analyze_defaults.get("cached_analysis", False)),1187        pr_template_cleanup_mode=str(1188            analyze_defaults.get("pr-template-cleanup-mode", "merge_defaults")1189        ),1190        pr_template_strip_html_comments=bool(1191            analyze_defaults.get("pr-template-strip-html-comments", True)1192        ),1193        pr_template_trim_closing_reference_prefix=bool(1194            analyze_defaults.get("pr-template-trim-closing-reference-prefix", True)1195        ),1196        pr_template_section_patterns=tuple(1197            analyze_defaults.get("pr-template-section-patterns", ())1198        ),1199        pr_template_line_patterns=tuple(analyze_defaults.get("pr-template-line-patterns", ())),1200        cluster_suppression_rules=tuple(analyze_defaults.get("cluster-suppression-rules", ())),

Showing the first 1,200 of 1621 lines. Download the file for the rest.