Team Ai
Apppublic

evalstate/diffusers-pr-api

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
normalize.py278 linesDownload Raw Back to data
1from __future__ import annotations2 3from typing import Any4from urllib.parse import urlparse5 6 7def _user_fields(user: dict[str, Any] | None) -> dict[str, Any]:8    user = user or {}9    return {10        "author_login": user.get("login"),11        "author_id": user.get("id"),12        "author_node_id": user.get("node_id"),13        "author_type": user.get("type"),14        "author_site_admin": user.get("site_admin"),15    }16 17 18def _labels(labels: list[dict[str, Any]] | None) -> list[str]:19    return [20        name21        for label in labels or []22        if isinstance(label, dict) and isinstance((name := label.get("name")), str) and name23    ]24 25 26def _assignees(users: list[dict[str, Any]] | None) -> list[str]:27    return [28        login29        for user in users or []30        if isinstance(user, dict) and isinstance((login := user.get("login")), str) and login31    ]32 33 34def issue_url_to_number(issue_url: str | None) -> int | None:35    if not issue_url:36        return None37    path = urlparse(issue_url).path.rstrip("/")38    tail = path.rsplit("/", 1)[-1]39    try:40        return int(tail)41    except ValueError:42        return None43 44 45def normalize_issue(46    repo: str, item: dict[str, Any], snapshot_id: str, extracted_at: str47) -> dict[str, Any]:48    return {49        "repo": repo,50        "github_id": item.get("id"),51        "github_node_id": item.get("node_id"),52        "number": item.get("number"),53        "html_url": item.get("html_url"),54        "api_url": item.get("url"),55        "title": item.get("title"),56        "body": item.get("body"),57        "state": item.get("state"),58        "state_reason": item.get("state_reason"),59        "locked": item.get("locked"),60        "comments_count": item.get("comments"),61        "labels": _labels(item.get("labels")),62        "assignees": _assignees(item.get("assignees")),63        "created_at": item.get("created_at"),64        "updated_at": item.get("updated_at"),65        "closed_at": item.get("closed_at"),66        "author_association": item.get("author_association"),67        "milestone_title": (item.get("milestone") or {}).get("title"),68        "snapshot_id": snapshot_id,69        "extracted_at": extracted_at,70        **_user_fields(item.get("user")),71    }72 73 74def normalize_pull_request(75    repo: str,76    issue_stub: dict[str, Any],77    pr_detail: dict[str, Any],78    snapshot_id: str,79    extracted_at: str,80) -> dict[str, Any]:81    head = pr_detail.get("head") or {}82    base = pr_detail.get("base") or {}83    return {84        "repo": repo,85        "github_id": pr_detail.get("id") or issue_stub.get("id"),86        "github_node_id": pr_detail.get("node_id") or issue_stub.get("node_id"),87        "number": issue_stub.get("number"),88        "html_url": issue_stub.get("html_url"),89        "api_url": issue_stub.get("url"),90        "title": issue_stub.get("title"),91        "body": issue_stub.get("body"),92        "state": issue_stub.get("state"),93        "state_reason": issue_stub.get("state_reason"),94        "locked": issue_stub.get("locked"),95        "comments_count": issue_stub.get("comments"),96        "labels": _labels(issue_stub.get("labels")),97        "assignees": _assignees(issue_stub.get("assignees")),98        "created_at": issue_stub.get("created_at"),99        "updated_at": issue_stub.get("updated_at"),100        "closed_at": issue_stub.get("closed_at"),101        "author_association": issue_stub.get("author_association")102        or pr_detail.get("author_association"),103        "merged_at": pr_detail.get("merged_at"),104        "merge_commit_sha": pr_detail.get("merge_commit_sha"),105        "merged": pr_detail.get("merged"),106        "mergeable": pr_detail.get("mergeable"),107        "mergeable_state": pr_detail.get("mergeable_state"),108        "draft": pr_detail.get("draft"),109        "additions": pr_detail.get("additions"),110        "deletions": pr_detail.get("deletions"),111        "changed_files": pr_detail.get("changed_files"),112        "commits": pr_detail.get("commits"),113        "review_comments_count": pr_detail.get("review_comments"),114        "maintainer_can_modify": pr_detail.get("maintainer_can_modify"),115        "head_ref": head.get("ref"),116        "head_sha": head.get("sha"),117        "head_repo_full_name": (head.get("repo") or {}).get("full_name"),118        "base_ref": base.get("ref"),119        "base_sha": base.get("sha"),120        "base_repo_full_name": (base.get("repo") or {}).get("full_name"),121        "snapshot_id": snapshot_id,122        "extracted_at": extracted_at,123        **_user_fields(issue_stub.get("user")),124    }125 126 127def normalize_comment(128    repo: str,129    item: dict[str, Any],130    parent_kind: str,131    parent_number: int | None,132    snapshot_id: str,133    extracted_at: str,134) -> dict[str, Any]:135    return {136        "repo": repo,137        "github_id": item.get("id"),138        "github_node_id": item.get("node_id"),139        "parent_kind": parent_kind,140        "parent_number": parent_number,141        "html_url": item.get("html_url"),142        "api_url": item.get("url"),143        "issue_api_url": item.get("issue_url"),144        "body": item.get("body"),145        "created_at": item.get("created_at"),146        "updated_at": item.get("updated_at"),147        "author_association": item.get("author_association"),148        "snapshot_id": snapshot_id,149        "extracted_at": extracted_at,150        **_user_fields(item.get("user")),151    }152 153 154def normalize_review(155    repo: str, pr_number: int, item: dict[str, Any], snapshot_id: str, extracted_at: str156) -> dict[str, Any]:157    return {158        "repo": repo,159        "github_id": item.get("id"),160        "github_node_id": item.get("node_id"),161        "pull_request_number": pr_number,162        "html_url": item.get("html_url"),163        "api_url": item.get("url"),164        "body": item.get("body"),165        "state": item.get("state"),166        "submitted_at": item.get("submitted_at"),167        "commit_id": item.get("commit_id"),168        "author_association": item.get("author_association"),169        "snapshot_id": snapshot_id,170        "extracted_at": extracted_at,171        **_user_fields(item.get("user")),172    }173 174 175def normalize_review_comment(176    repo: str, pr_number: int, item: dict[str, Any], snapshot_id: str, extracted_at: str177) -> dict[str, Any]:178    return {179        "repo": repo,180        "github_id": item.get("id"),181        "github_node_id": item.get("node_id"),182        "pull_request_number": pr_number,183        "review_id": item.get("pull_request_review_id"),184        "html_url": item.get("html_url"),185        "api_url": item.get("url"),186        "pull_request_api_url": item.get("pull_request_url"),187        "body": item.get("body"),188        "path": item.get("path"),189        "commit_id": item.get("commit_id"),190        "original_commit_id": item.get("original_commit_id"),191        "position": item.get("position"),192        "original_position": item.get("original_position"),193        "line": item.get("line"),194        "start_line": item.get("start_line"),195        "side": item.get("side"),196        "start_side": item.get("start_side"),197        "subject_type": item.get("subject_type"),198        "created_at": item.get("created_at"),199        "updated_at": item.get("updated_at"),200        "author_association": item.get("author_association"),201        "snapshot_id": snapshot_id,202        "extracted_at": extracted_at,203        **_user_fields(item.get("user")),204    }205 206 207def normalize_pr_file(208    repo: str,209    pr_number: int,210    item: dict[str, Any],211    snapshot_id: str,212    extracted_at: str,213) -> dict[str, Any]:214    return {215        "repo": repo,216        "pull_request_number": pr_number,217        "sha": item.get("sha"),218        "filename": item.get("filename"),219        "status": item.get("status"),220        "additions": item.get("additions"),221        "deletions": item.get("deletions"),222        "changes": item.get("changes"),223        "blob_url": item.get("blob_url"),224        "raw_url": item.get("raw_url"),225        "contents_url": item.get("contents_url"),226        "previous_filename": item.get("previous_filename"),227        "patch": item.get("patch"),228        "snapshot_id": snapshot_id,229        "extracted_at": extracted_at,230    }231 232 233def normalize_pr_diff(234    repo: str,235    pr_number: int,236    html_url: str | None,237    api_url: str | None,238    diff: str,239    snapshot_id: str,240    extracted_at: str,241) -> dict[str, Any]:242    return {243        "repo": repo,244        "pull_request_number": pr_number,245        "html_url": html_url,246        "api_url": api_url,247        "diff": diff,248        "snapshot_id": snapshot_id,249        "extracted_at": extracted_at,250    }251 252 253def normalize_timeline_event(254    repo: str,255    number: int,256    parent_kind: str,257    item: dict[str, Any],258    snapshot_id: str,259    extracted_at: str,260) -> dict[str, Any]:261    source = item.get("source") or {}262    issue = source.get("issue") or {}263    return {264        "repo": repo,265        "parent_kind": parent_kind,266        "parent_number": number,267        "event": item.get("event"),268        "created_at": item.get("created_at"),269        "actor_login": (item.get("actor") or {}).get("login"),270        "source_issue_number": issue.get("number"),271        "source_issue_title": issue.get("title"),272        "source_issue_url": issue.get("html_url"),273        "commit_id": item.get("commit_id"),274        "label_name": (item.get("label") or {}).get("name"),275        "snapshot_id": snapshot_id,276        "extracted_at": extracted_at,277    }278