mirror of
https://github.com/PrefectHQ/fastmcp.git
synced 2026-10-09 06:23:20 +02:00
297 lines
11 KiB
Python
297 lines
11 KiB
Python
"""Scan generated release notes for shared human contributor credit.
|
|
|
|
Usage: uv run scripts/release_contributor_credit.py /tmp/generated-notes.md
|
|
|
|
Prints attribution candidates as JSON. Reads GitHub metadata without modifying
|
|
PRs, issues, commits, or release notes. New-contributor history and opt-outs are
|
|
reviewed by the release agent only for the candidates returned here.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import subprocess
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
REPO = "PrefectHQ/fastmcp"
|
|
AGENT_LOGINS = {"claude", "codex", "copilot", "github-copilot", "openai-codex", "devin"}
|
|
AGENT_NAMES = re.compile(
|
|
r"^(?:claude (?:opus|sonnet|haiku)|codex|openai codex|github copilot|copilot|devin)(?:\b|$)",
|
|
re.I,
|
|
)
|
|
HANDLE = r"@[\w-]+(?:\[bot\])?"
|
|
PR_ENTRY = re.compile(
|
|
rf"^(?P<prefix>\* .+) by (?P<authors>{HANDLE}(?:, (?:and )?{HANDLE}| and {HANDLE})*) in "
|
|
r"https://github\.com/" + re.escape(REPO) + r"/pull/(?P<number>\d+)\s*$",
|
|
re.M,
|
|
)
|
|
|
|
|
|
def api(*args: str) -> dict[str, Any]:
|
|
result = json.loads(subprocess.check_output(["gh", "api", *args], text=True))
|
|
if result.get("errors"):
|
|
raise ValueError(f"GitHub query failed: {result['errors']}")
|
|
return result
|
|
|
|
|
|
def is_human(actor: dict[str, Any] | None) -> bool:
|
|
if not actor:
|
|
return False
|
|
login = actor["login"].lower()
|
|
return (
|
|
actor.get("__typename") != "Bot"
|
|
and not login.endswith("[bot]")
|
|
and login not in AGENT_LOGINS
|
|
)
|
|
|
|
|
|
def is_agent_author(author: dict[str, Any]) -> bool:
|
|
user = author.get("user")
|
|
return (
|
|
bool(user and not is_human(user))
|
|
or bool(AGENT_NAMES.match(author["name"]))
|
|
or author["email"].lower() in {"noreply@anthropic.com", "noreply@openai.com"}
|
|
)
|
|
|
|
|
|
def fetch_prs(numbers: list[int]) -> dict[int, dict[str, Any]]:
|
|
prs: dict[int, dict[str, Any]] = {}
|
|
for start in range(0, len(numbers), 30):
|
|
fields = "\n".join(
|
|
f"""p{number}: pullRequest(number:{number}) {{
|
|
number author {{ login __typename }}
|
|
closingIssuesReferences(first:100) {{
|
|
nodes {{ number url author {{ login __typename }} }}
|
|
pageInfo {{ hasNextPage }}
|
|
}}
|
|
commits(first:100) {{
|
|
nodes {{ commit {{ authors(first:100) {{
|
|
nodes {{ name email user {{ login __typename }} }}
|
|
pageInfo {{ hasNextPage }}
|
|
}} }} }}
|
|
pageInfo {{ hasNextPage }}
|
|
}}
|
|
mergeCommit {{ authors(first:100) {{
|
|
nodes {{ name email user {{ login __typename }} }}
|
|
pageInfo {{ hasNextPage }}
|
|
}} }}
|
|
}}"""
|
|
for number in numbers[start : start + 30]
|
|
)
|
|
query = (
|
|
'query { repository(owner:"PrefectHQ",name:"fastmcp") {' + fields + "} }"
|
|
)
|
|
result = api("graphql", "-f", f"query={query}")
|
|
for pr in result["data"]["repository"].values():
|
|
if pr is None or pr["mergeCommit"] is None:
|
|
raise ValueError("Release notes reference a missing or unmerged PR")
|
|
if any(
|
|
connection["pageInfo"]["hasNextPage"]
|
|
for connection in (
|
|
pr["closingIssuesReferences"],
|
|
pr["mergeCommit"]["authors"],
|
|
pr["commits"],
|
|
*(node["commit"]["authors"] for node in pr["commits"]["nodes"]),
|
|
)
|
|
):
|
|
raise ValueError(
|
|
f"PR #{pr['number']}: attribution metadata is truncated"
|
|
)
|
|
prs[pr["number"]] = pr
|
|
return prs
|
|
|
|
|
|
def release_prs(notes: str) -> dict[int, dict[str, Any]]:
|
|
entries = list(PR_ENTRY.finditer(notes))
|
|
# Fail visibly if the generated format changes; a silent empty scan loses credit.
|
|
linked = set(
|
|
re.findall(r"https://github\.com/" + re.escape(REPO) + r"/pull/(\d+)", notes)
|
|
)
|
|
parsed = {entry["number"] for entry in entries}
|
|
if linked - parsed:
|
|
raise ValueError(
|
|
f"Cannot parse release entries for PRs: {sorted(linked - parsed)}"
|
|
)
|
|
return fetch_prs(sorted(int(number) for number in parsed))
|
|
|
|
|
|
def commit_authors(pr: dict[str, Any]) -> list[dict[str, Any]]:
|
|
return [
|
|
author
|
|
for node in pr["commits"]["nodes"]
|
|
for author in node["commit"]["authors"]["nodes"]
|
|
] + pr["mergeCommit"]["authors"]["nodes"]
|
|
|
|
|
|
def note_actors(prs: dict[int, dict[str, Any]]) -> dict[str, dict[str, Any]]:
|
|
"""Keep GitHub actor types, including plain-handle aliases for app bots."""
|
|
actors = {}
|
|
for pr in prs.values():
|
|
sources = [pr["author"]]
|
|
sources.extend(
|
|
issue["author"] for issue in pr["closingIssuesReferences"]["nodes"]
|
|
)
|
|
sources.extend(author.get("user") for author in commit_authors(pr))
|
|
for actor in sources:
|
|
if actor:
|
|
key = actor["login"].lower()
|
|
actors[key] = actor
|
|
if key.endswith("[bot]"):
|
|
actors[key.removesuffix("[bot]")] = actor
|
|
return actors
|
|
|
|
|
|
def scan_notes(
|
|
notes: str, prs: dict[int, dict[str, Any]] | None = None
|
|
) -> list[dict[str, Any]]:
|
|
if prs is None:
|
|
prs = release_prs(notes)
|
|
actors = note_actors(prs)
|
|
entries = list(PR_ENTRY.finditer(notes))
|
|
permissions: dict[str, bool] = {}
|
|
|
|
def is_maintainer(login: str) -> bool:
|
|
key = login.lower()
|
|
if key not in permissions:
|
|
permission = api(f"repos/{REPO}/collaborators/{login}/permission")[
|
|
"permission"
|
|
]
|
|
permissions[key] = permission in {"admin", "maintain", "write"}
|
|
return permissions[key]
|
|
|
|
candidates = []
|
|
for entry in entries:
|
|
number = int(entry["number"])
|
|
pr = prs[number]
|
|
authors = commit_authors(pr)
|
|
humans = [author for author in authors if not is_agent_author(author)]
|
|
original_credit = {
|
|
mention.lstrip("@").lower(): mention.lstrip("@")
|
|
for mention in re.findall(HANDLE, entry["authors"])
|
|
if is_human(actors.get(mention.lstrip("@").lower()))
|
|
}
|
|
credited = dict(original_credit)
|
|
if is_human(pr["author"]):
|
|
login = pr["author"]["login"]
|
|
credited[login.lower()] = login
|
|
for author in humans:
|
|
if is_human(author.get("user")):
|
|
login = author["user"]["login"]
|
|
if login.lower() not in credited and not is_maintainer(login):
|
|
credited[login.lower()] = login
|
|
issues = [
|
|
issue
|
|
for issue in pr["closingIssuesReferences"]["nodes"]
|
|
if is_human(issue["author"])
|
|
]
|
|
missing_reporters = []
|
|
extra_reporters = [
|
|
issue
|
|
for issue in issues
|
|
if issue["author"]["login"].lower() not in credited
|
|
]
|
|
if extra_reporters and pr["author"]:
|
|
login = pr["author"]["login"]
|
|
if is_maintainer(login):
|
|
for issue in extra_reporters:
|
|
reporter = issue["author"]["login"]
|
|
if reporter.lower() not in credited and not is_maintainer(reporter):
|
|
credited[reporter.lower()] = reporter
|
|
missing_reporters.append(reporter)
|
|
unmapped = list(
|
|
dict.fromkeys(author["name"] for author in humans if not author.get("user"))
|
|
)
|
|
if (
|
|
set(credited) == set(original_credit)
|
|
and not missing_reporters
|
|
and not unmapped
|
|
):
|
|
continue
|
|
handles = [f"@{login}" for login in credited.values()]
|
|
shared = (
|
|
" and ".join(handles)
|
|
if len(handles) <= 2
|
|
else ", ".join(handles[:-1]) + ", and " + handles[-1]
|
|
)
|
|
candidates.append(
|
|
{
|
|
"pr": number,
|
|
"credit": list(credited.values()),
|
|
"closing_issues": [
|
|
{
|
|
"number": issue["number"],
|
|
"url": issue["url"],
|
|
"author": issue["author"]["login"],
|
|
}
|
|
for issue in issues
|
|
],
|
|
"missing_reporter_coauthors": missing_reporters,
|
|
"unmapped_commit_authors": unmapped,
|
|
"suggested_entry": (
|
|
f"{entry['prefix']} by {shared} in https://github.com/{REPO}/pull/{number}"
|
|
if handles
|
|
else None
|
|
),
|
|
}
|
|
)
|
|
return candidates
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("notes", type=Path)
|
|
args = parser.parse_args()
|
|
notes = args.notes.read_text()
|
|
prs = release_prs(notes)
|
|
candidates = scan_notes(notes, prs)
|
|
print(json.dumps(summarize_credit(notes, candidates, prs), indent=2))
|
|
|
|
|
|
def summarize_credit(
|
|
notes: str, candidates: list[dict[str, Any]], prs: dict[int, dict[str, Any]]
|
|
) -> dict[str, Any]:
|
|
"""Count unique human contributors and identify first-time history checks.
|
|
|
|
GitHub's first-time candidates and supplemental contributors need a history
|
|
check, including prior co-authorship, before finalizing that list and its count.
|
|
"""
|
|
actors = note_actors(prs)
|
|
original: dict[str, str] = {}
|
|
for entry in PR_ENTRY.finditer(notes):
|
|
for mention in re.findall(HANDLE, entry["authors"]):
|
|
login = mention.lstrip("@")
|
|
if is_human(actors.get(login.lower())):
|
|
original[login.lower()] = login
|
|
credited = dict(original)
|
|
for candidate in candidates:
|
|
for login in candidate["credit"]:
|
|
credited.setdefault(login.lower(), login)
|
|
first_time: dict[str, str] = {}
|
|
for login in re.findall(
|
|
rf"^\* ({HANDLE}) made their first contribution in ", notes, re.M
|
|
):
|
|
login = login.lstrip("@")
|
|
if is_human(actors.get(login.lower())):
|
|
first_time[login.lower()] = login
|
|
return {
|
|
"candidates": candidates,
|
|
"contributors": sorted(credited.values(), key=str.lower),
|
|
"total_contributors": len(credited),
|
|
"generated_first_time_contributors": sorted(first_time.values(), key=str.lower),
|
|
"generated_first_time_count": len(first_time),
|
|
"first_time_review": sorted(
|
|
(
|
|
login
|
|
for key, login in credited.items()
|
|
if key not in original or key in first_time
|
|
),
|
|
key=str.lower,
|
|
),
|
|
}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|