first pass of merging in warp (doesn't build)

This commit is contained in:
Ryan Ward
2026-07-01 16:08:58 -05:00
parent 2f64909469
commit 4770ac06b5
3662 changed files with 414574 additions and 89772 deletions
@@ -0,0 +1,175 @@
#!/usr/bin/env python3
"""Build a Slack Block Kit payload from release-pipeline changelog JSON."""
from __future__ import annotations
import argparse
import html
import json
import re
import sys
from pathlib import Path
from typing import Iterable
from urllib.parse import quote
MAX_SECTION_TEXT_LENGTH = 3000
MAX_MESSAGE_BLOCKS = 50
SECTION_ORDER = (
("newFeatures", "New Features"),
("improvements", "Improvements"),
("bugFixes", "Bug Fixes"),
("images", "Image"),
# Keep the existing label stable for compatibility with recent Slack posts.
("oz_updates", "oz_updates"),
)
MARKDOWN_LINK_RE = re.compile(r"\[([^\]]+)\]\((https?://[^)\s]+)\)")
SLACK_LINK_URL_SAFE_CHARS = "/:?&=#%+~@!$'()*;,[]"
def escape_slack_text(text: str) -> str:
"""Escape Slack control characters in ordinary mrkdwn text."""
return html.escape(text, quote=False)
def escape_slack_link_url(url: str) -> str:
"""Percent-encode Slack link delimiters before embedding a URL in mrkdwn."""
return quote(url, safe=SLACK_LINK_URL_SAFE_CHARS)
def slack_link(url: str, label: str) -> str:
"""Build a Slack mrkdwn link from an already-validated URL and label."""
return f"<{escape_slack_link_url(url)}|{escape_slack_text(label)}>"
def markdown_links_to_slack(text: str) -> str:
"""Convert standard Markdown links to Slack mrkdwn links."""
parts: list[str] = []
last_end = 0
for match in MARKDOWN_LINK_RE.finditer(text):
parts.append(escape_slack_text(text[last_end : match.start()]))
label, url = match.groups()
parts.append(slack_link(url, label))
last_end = match.end()
parts.append(escape_slack_text(text[last_end:]))
return "".join(parts)
def slack_lines(changelog: dict) -> list[str]:
"""Render non-empty changelog sections as Slack mrkdwn lines."""
lines: list[str] = []
for key, title in SECTION_ORDER:
values = changelog.get(key, [])
if not isinstance(values, list) or not values:
continue
lines.append(f"*{title}*")
for value in values:
lines.append(f"{markdown_links_to_slack(str(value))}")
return lines
def split_overlong_line(line: str) -> Iterable[str]:
"""Split a pathological line so every Slack section remains valid."""
while len(line) > MAX_SECTION_TEXT_LENGTH - 1:
yield line[: MAX_SECTION_TEXT_LENGTH - 1]
line = line[MAX_SECTION_TEXT_LENGTH - 1 :]
yield line
def chunk_lines(lines: list[str]) -> list[str]:
"""Split text into section-sized chunks while retaining copy boundaries."""
chunks: list[str] = []
buffer = ""
for raw_line in lines:
for line in split_overlong_line(raw_line):
candidate = line if not buffer else f"{buffer}\n{line}"
# Reserve a character for a trailing newline. Keeping it in each
# section preserves a separator when Slack copies adjacent blocks.
if len(candidate) + 1 > MAX_SECTION_TEXT_LENGTH:
chunks.append(f"{buffer}\n")
buffer = line
else:
buffer = candidate
if buffer:
chunks.append(f"{buffer}\n")
return chunks
def artifact_link_block(markdown_artifact_url: str) -> dict | None:
"""Build a Slack block linking to the downloadable Markdown artifact."""
if not markdown_artifact_url:
return None
return {
"type": "section",
"text": {
"type": "mrkdwn",
"text": (
"Raw Markdown changelog: "
f"{slack_link(markdown_artifact_url, 'Download raw Markdown changelog artifact')}"
),
},
}
def build_payload(
changelog: dict, release_tag: str, markdown_artifact_url: str = ""
) -> dict:
"""Build a Block Kit message and reject payloads Slack cannot accept."""
chunks = chunk_lines(slack_lines(changelog))
if not chunks:
return {"blocks": []}
blocks = [
{
"type": "header",
"text": {
"type": "plain_text",
"text": f"Changelog for {release_tag}",
},
}
]
artifact_block = artifact_link_block(markdown_artifact_url)
if artifact_block is not None:
blocks.append(artifact_block)
blocks.extend(
{
"type": "section",
"expand": True,
"text": {"type": "mrkdwn", "text": chunk},
}
for chunk in chunks
)
if len(blocks) > MAX_MESSAGE_BLOCKS:
raise ValueError(
"Slack payload would require "
f"{len(blocks)} blocks, exceeding Slack's {MAX_MESSAGE_BLOCKS}-block limit"
)
return {"blocks": blocks}
def main() -> None:
parser = argparse.ArgumentParser(
description="Build Slack payload JSON from changelog release JSON"
)
parser.add_argument("--input", required=True, help="Path to changelog JSON")
parser.add_argument("--release-tag", required=True, help="Release tag header text")
parser.add_argument(
"--markdown-artifact-url",
default="",
help="Download URL for the raw Markdown changelog artifact",
)
parser.add_argument("--output", required=True, help="Payload JSON output path")
args = parser.parse_args()
with open(args.input) as f:
changelog = json.load(f)
payload = build_payload(changelog, args.release_tag, args.markdown_artifact_url)
Path(args.output).write_text(json.dumps(payload, separators=(",", ":")) + "\n")
print(f"Built Slack payload with {len(payload['blocks'])} blocks", file=sys.stderr)
if __name__ == "__main__":
main()
@@ -0,0 +1,97 @@
#!/usr/bin/env python3
"""Classify GitHub usernames as internal, external, or bot.
Uses `gh api` to check org membership — stdlib only, no pip deps.
Usage:
python3 classify_contributors.py --org warpdotdev --authors user1,user2,user3
Outputs JSON to stdout.
"""
import argparse
import json
import subprocess
import sys
KNOWN_BOTS = frozenset(
{
"dependabot",
"dependabot[bot]",
"renovate",
"renovate[bot]",
"github-actions",
"github-actions[bot]",
"codecov",
"codecov[bot]",
"warp-bot",
"warp-bot[bot]",
}
)
def run(cmd: list[str], *, check: bool = True) -> subprocess.CompletedProcess:
return subprocess.run(cmd, capture_output=True, text=True, check=check)
def check_org_membership(org: str, username: str) -> str:
"""Check if a user is a member of the given GitHub org via gh api.
Returns:
'internal' if the user is an org member (HTTP 204),
'external' if the user is confirmed not a member (HTTP 404),
'unknown' if the check failed due to auth/permission issues.
"""
result = run(
["gh", "api", f"orgs/{org}/members/{username}", "--silent"],
check=False,
)
if result.returncode == 0:
return "internal"
# Distinguish auth failures from genuine "not a member" responses.
# gh api exits non-zero for both 404 (not a member) and 403/401 (no
# read:org scope). Only treat an explicit 404 as "external";
# everything else (network errors, rate limits, auth issues) is "unknown"
# to avoid publicly crediting internal or unverified users.
stderr = result.stderr.lower()
if "404" in stderr:
return "external"
return "unknown"
def main() -> None:
parser = argparse.ArgumentParser(description="Classify contributor types")
parser.add_argument("--org", required=True, help="GitHub org to check membership")
parser.add_argument(
"--authors",
required=True,
help="Comma-separated list of GitHub usernames",
)
args = parser.parse_args()
authors = [a.strip() for a in args.authors.split(",") if a.strip()]
internal: list[str] = []
external: list[str] = []
bot: list[str] = []
unknown: list[str] = []
for author in authors:
if author.lower() in KNOWN_BOTS or author.endswith("[bot]"):
bot.append(author)
else:
status = check_org_membership(args.org, author)
if status == "internal":
internal.append(author)
elif status == "unknown":
unknown.append(author)
else:
external.append(author)
output = {"internal": internal, "external": external, "bot": bot, "unknown": unknown}
json.dump(output, sys.stdout, indent=2)
print()
if __name__ == "__main__":
main()
@@ -0,0 +1,115 @@
#!/usr/bin/env python3
"""Convert changelog-draft.json to the release-pipeline-compatible changelog-release.json.
Reads the audit artifact produced by the changelog-draft skill and emits the
flat JSON structure consumed by the create_release workflow (Slack payload
builder + in-app changelog.json step).
Usage:
python3 convert_to_release_json.py --input <changelog-draft.json> --output <changelog-release.json>
The output schema:
{
"newFeatures": ["..."],
"improvements": ["..."],
"bugFixes": ["..."],
"images": ["..."],
"oz_updates": ["..."]
}
"""
import argparse
import json
import sys
# Map from changelog-draft.json category names to release JSON keys.
CATEGORY_MAP = {
"NEW-FEATURE": "newFeatures",
"IMPROVEMENT": "improvements",
"BUG-FIX": "bugFixes",
"OZ": "oz_updates",
"IMAGE": "images",
}
def github_profile_link(username: str) -> str:
"""Format a GitHub username as a markdown profile link."""
return f"[@{username}](https://github.com/{username})"
def format_entry(entry: dict) -> str:
"""Format a single changelog entry as a text line with a PR link.
Includes external contributor attribution when applicable.
"""
text = entry["text"]
pr_number = entry.get("pr_number") or entry.get("number")
url = entry.get("url") or entry.get("pr_url")
link = ""
if url and pr_number:
link = f" ([#{pr_number}]({url}))"
attribution = ""
if entry.get("is_external") and entry.get("author"):
attribution = f"{github_profile_link(entry['author'])}"
return f"{text}{link}{attribution}"
def convert(draft: dict) -> dict:
"""Convert a changelog-draft.json dict to changelog-release.json dict."""
release: dict[str, list[str]] = {
"newFeatures": [],
"improvements": [],
"bugFixes": [],
"images": [],
"oz_updates": [],
}
for entry in draft.get("entries", []):
category = entry.get("category", "")
release_key = CATEGORY_MAP.get(category)
if release_key is None:
continue
if category == "IMAGE":
# IMAGE entries store a URL in "text" — pass through directly.
release["images"].append(entry["text"])
else:
release[release_key].append(format_entry(entry))
return release
def main() -> None:
parser = argparse.ArgumentParser(
description="Convert changelog-draft.json to changelog-release.json"
)
parser.add_argument(
"--input",
required=True,
help="Path to changelog-draft.json",
)
parser.add_argument(
"--output",
required=True,
help="Path to write changelog-release.json",
)
args = parser.parse_args()
with open(args.input) as f:
draft = json.load(f)
release = convert(draft)
with open(args.output, "w") as f:
json.dump(release, f, indent=2)
f.write("\n")
# Summary to stdout for CI logs
for key, items in release.items():
print(f" {key}: {len(items)} entries")
if __name__ == "__main__":
main()
@@ -0,0 +1,55 @@
#!/usr/bin/env python3
"""Extract RELEASE_FLAGS, PREVIEW_FLAGS, and DOGFOOD_FLAGS from warp_features.
Parses crates/warp_features/src/lib.rs to find the const arrays and extracts
the FeatureFlag variant names. Stdlib only, no pip deps.
Usage:
python3 extract_feature_flags.py --file crates/warp_features/src/lib.rs
Outputs JSON to stdout.
"""
import argparse
import json
import re
import sys
def extract_flag_list(source: str, const_name: str) -> list[str]:
"""Extract FeatureFlag variant names from a const array definition."""
# Match: pub const CONST_NAME: &[FeatureFlag] = &[ ... ];
pattern = rf"pub\s+const\s+{re.escape(const_name)}\s*:\s*&\[FeatureFlag\]\s*=\s*&\[(.*?)\];"
m = re.search(pattern, source, re.DOTALL)
if not m:
return []
block = m.group(1)
# Extract FeatureFlag::VariantName entries, ignoring #[cfg(...)] attributes
variants = re.findall(r"FeatureFlag::(\w+)", block)
return variants
def main() -> None:
parser = argparse.ArgumentParser(description="Extract feature flag gate lists")
parser.add_argument(
"--file",
required=True,
help="Path to warp_features lib.rs",
)
args = parser.parse_args()
with open(args.file) as f:
source = f.read()
output = {
"release_flags": extract_flag_list(source, "RELEASE_FLAGS"),
"preview_flags": extract_flag_list(source, "PREVIEW_FLAGS"),
"dogfood_flags": extract_flag_list(source, "DOGFOOD_FLAGS"),
}
json.dump(output, sys.stdout, indent=2)
print()
if __name__ == "__main__":
main()
@@ -0,0 +1,127 @@
#!/usr/bin/env python3
"""Fetch the original reporters for GitHub issues linked to PRs in a release.
Uses `gh` CLI (must be authenticated) — stdlib only, no pip deps.
Usage:
python3 fetch_issue_reporters.py --repo warpdotdev/warp --issues 1234,5678,9012
Outputs JSON to stdout mapping issue numbers to reporter info.
"""
import argparse
import json
import subprocess
import sys
def run(cmd: list[str], *, check: bool = True) -> str:
result = subprocess.run(cmd, capture_output=True, text=True, check=check)
return result.stdout.strip()
def run_full(cmd: list[str], *, check: bool = True) -> subprocess.CompletedProcess:
return subprocess.run(cmd, capture_output=True, text=True, check=check)
def is_org_member(org: str, username: str) -> bool:
"""Check if a user is a member of the given GitHub org.
Returns True for members (HTTP 204), False for non-members (HTTP 404),
and True (conservative) for auth failures so internal users aren't
misattributed as external.
"""
result = run_full(
["gh", "api", f"orgs/{org}/members/{username}", "--silent"],
check=False,
)
if result.returncode == 0:
return True
stderr = result.stderr.lower()
# Auth failure — be conservative, treat as internal
if "403" in stderr or "401" in stderr or "saml" in stderr:
return True
return False
def fetch_issue_reporter(repo: str, issue_number: int) -> dict | None:
"""Fetch the reporter (author) of a GitHub issue via gh CLI."""
raw = run(
[
"gh",
"issue",
"view",
str(issue_number),
"--repo",
repo,
"--json",
"number,title,author,url",
],
check=False,
)
if not raw:
return None
try:
data = json.loads(raw)
except json.JSONDecodeError:
return None
author = ""
if isinstance(data.get("author"), dict):
author = data["author"].get("login", "")
elif isinstance(data.get("author"), str):
author = data["author"]
return {
"issue_number": data.get("number", issue_number),
"title": data.get("title", ""),
"reporter": author,
"reporter_url": f"https://github.com/{author}" if author else "",
"url": data.get("url", ""),
}
def main() -> None:
parser = argparse.ArgumentParser(
description="Fetch issue reporters for linked issues"
)
parser.add_argument("--repo", required=True, help="GitHub repo (owner/name)")
parser.add_argument(
"--org",
required=False,
default="",
help="GitHub org to filter out internal reporters (e.g. warpdotdev)",
)
parser.add_argument(
"--issues",
required=True,
help="Comma-separated issue numbers",
)
args = parser.parse_args()
issue_numbers = [
int(n.strip()) for n in args.issues.split(",") if n.strip().isdigit()
]
org = args.org
reporters: list[dict] = []
seen_reporters: set[str] = set()
for num in issue_numbers:
info = fetch_issue_reporter(args.repo, num)
if not info or not info["reporter"]:
continue
username = info["reporter"]
# Skip internal org members when --org is provided
if org and username not in seen_reporters and is_org_member(org, username):
seen_reporters.add(username)
continue
if username not in seen_reporters:
seen_reporters.add(username)
reporters.append(info)
json.dump({"issue_reporters": reporters}, sys.stdout, indent=2)
print() # trailing newline
if __name__ == "__main__":
main()
@@ -0,0 +1,358 @@
#!/usr/bin/env python3
"""Fetch PRs merged in a release range and extract explicit CHANGELOG markers.
Uses `gh` CLI (must be authenticated) and `git` — stdlib only, no pip deps.
Usage:
python3 fetch_prs.py --repo warpdotdev/warp --base-ref <prev_tag> --head-ref <release_tag>
Outputs JSON to stdout.
"""
import argparse
import json
import re
import subprocess
import sys
# Matches lines like: CHANGELOG-NEW-FEATURE: Added dark mode
MARKER_RE = re.compile(
r"^CHANGELOG-(NEW-FEATURE|IMPROVEMENT|BUG-FIX|IMAGE|OZ|NONE)\s*:?\s*(.*)$",
re.MULTILINE,
)
# Matches issue-closing keywords: Fixes #123, Closes #456, Resolves #789
LINKED_ISSUE_RE = re.compile(
r"(?:close[sd]?|fix(?:e[sd])?|resolve[sd]?)\s+#(\d+)",
re.IGNORECASE,
)
PUBLIC_REPO = "warpdotdev/warp"
INTERNAL_REPO = "warpdotdev/warp-internal"
REPO_SYNC_AUTHORS = frozenset(
{
"app/warp-repo-sync",
"warp-repo-sync",
"warp-repo-sync[bot]",
}
)
PUBLIC_PR_URL_RE = re.compile(r"https://github\.com/warpdotdev/warp/pull/(\d+)")
def run(cmd: list[str], *, check: bool = True) -> str:
result = subprocess.run(cmd, capture_output=True, text=True, check=check)
return result.stdout.strip()
def get_commits(base_ref: str, head_ref: str) -> list[str]:
"""Return SHAs of first-parent commits between base and head."""
log = run(
[
"git",
"log",
"--first-parent",
"--format=%H",
f"{base_ref}..{head_ref}",
]
)
if not log:
return []
return log.splitlines()
def extract_pr_number(sha: str) -> int | None:
"""Extract PR number from a squash-merge commit subject line.
Expects the GitHub squash format: 'feat: something (#1234)'.
Matches the trailing parenthesized (#N) to avoid grabbing issue
numbers from titles like 'Fixes #123 (#456)'.
"""
msg = run(["git", "log", "-1", "--format=%s", sha])
# Match the last (#N) in the subject — GitHub always appends the PR number
m = re.search(r"\(#(\d+)\)\s*$", msg)
if m:
return int(m.group(1))
# Fallback: first bare #N (for non-standard subjects)
m = re.search(r"#(\d+)", msg)
if m:
return int(m.group(1))
return None
def get_merged_commits(sha: str) -> list[str]:
"""For a merge commit, return the SHAs brought in by the merge.
A merge commit has two parents: the first parent is the mainline, the
second parent is the tip of the merged branch. The commits unique to
the merge are those reachable from the second parent but not the first.
Returns an empty list for non-merge commits.
"""
parents = run(["git", "log", "-1", "--format=%P", sha]).split()
if len(parents) < 2:
return []
log = run(
["git", "log", "--format=%H", f"{parents[0]}..{parents[1]}"],
check=False,
)
if not log:
return []
return log.splitlines()
def fetch_pr_data(repo: str, pr_number: int) -> dict | None:
"""Fetch PR metadata and changed file paths via gh CLI."""
fields = "number,title,author,body,labels,mergedAt,files,url"
raw = run(
["gh", "pr", "view", str(pr_number), "--repo", repo, "--json", fields],
check=False,
)
if not raw:
return None
try:
return json.loads(raw)
except json.JSONDecodeError:
return None
def fetch_pr_commit_messages(repo: str, pr_number: int) -> list[str]:
"""Fetch commit messages for a PR via the GitHub API."""
raw = run(
["gh", "api", f"repos/{repo}/pulls/{pr_number}/commits"],
check=False,
)
if not raw:
return []
try:
commits = json.loads(raw)
except json.JSONDecodeError:
return []
messages = []
for commit in commits:
if not isinstance(commit, dict):
continue
commit_data = commit.get("commit")
if isinstance(commit_data, dict):
message = commit_data.get("message")
if message:
messages.append(message)
return messages
def get_author_login(data: dict) -> str:
"""Extract a GitHub login from a gh PR JSON object."""
if isinstance(data.get("author"), dict):
return data["author"].get("login", "")
if isinstance(data.get("author"), str):
return data["author"]
return ""
def get_label_names(data: dict) -> list[str]:
"""Extract label names from a gh PR JSON object."""
label_names = []
for lbl in data.get("labels", []) or []:
if isinstance(lbl, dict):
label_names.append(lbl.get("name", ""))
else:
label_names.append(str(lbl))
return label_names
def get_file_paths(data: dict) -> list[str]:
"""Extract changed file paths from a gh PR JSON object."""
file_paths = []
for f in data.get("files", []) or []:
if isinstance(f, dict):
file_paths.append(f.get("path", ""))
return file_paths
def is_repo_sync_pr(data: dict) -> bool:
"""Return whether this PR was created by the public-to-internal repo sync bot."""
return get_author_login(data) in REPO_SYNC_AUTHORS
def should_include_pr(repo: str, data: dict) -> bool:
"""Return whether a PR should be exposed to changelog generation.
Releases are cut from warp-internal, but non-sync-bot PRs merged there are
private/internal changes. Do not expose them to the Oz changelog agent or to
generated artifacts.
"""
return repo != INTERNAL_REPO or is_repo_sync_pr(data)
def extract_public_pr_number(text: str) -> int | None:
"""Extract a public warpdotdev/warp PR number from text."""
if not text:
return None
m = PUBLIC_PR_URL_RE.search(text)
if m:
return int(m.group(1))
# Repo-sync commits commonly preserve the original public squash-merge
# subject, such as "feat: add thing (#1234)".
m = re.search(r"\(#(\d+)\)\s*$", text.splitlines()[0] if text else "")
if m:
return int(m.group(1))
return None
def resolve_public_pr_number(repo: str, pr_number: int, data: dict) -> int | None:
"""Resolve a repo-sync PR back to its original public warpdotdev/warp PR."""
public_pr_number = extract_public_pr_number(data.get("body", "") or "")
if public_pr_number is not None:
return public_pr_number
for message in fetch_pr_commit_messages(repo, pr_number):
public_pr_number = extract_public_pr_number(message)
if public_pr_number is not None:
return public_pr_number
return None
def pr_reference(repo: str, pr_number: int, data: dict) -> dict:
"""Build a compact audit reference to a PR."""
return {
"number": data.get("number", pr_number),
"url": data.get("url", ""),
"author": get_author_login(data),
"title": data.get("title", ""),
"repo": repo,
}
def normalize_pr_data(repo: str, pr_number: int, data: dict) -> tuple[str, dict, dict | None]:
"""Resolve repo-sync PRs to public PR metadata.
The release workflow runs from warp-internal, where public PRs are mirrored
as warp-repo-sync[bot] PRs with different PR numbers. For changelog output
and contributor attribution, use the original public PR metadata when it can
be resolved, and keep the internal PR under `internal_pr` for audit only.
"""
internal_pr = pr_reference(repo, pr_number, data) if repo != PUBLIC_REPO else None
if repo == PUBLIC_REPO or not is_repo_sync_pr(data):
return repo, data, internal_pr
public_pr_number = resolve_public_pr_number(repo, pr_number, data)
if public_pr_number is None:
return repo, data, internal_pr
public_data = fetch_pr_data(PUBLIC_REPO, public_pr_number)
if public_data is None:
return repo, data, internal_pr
return PUBLIC_REPO, public_data, internal_pr
def extract_linked_issues(body: str) -> list[int]:
"""Extract issue numbers from closing keywords in a PR body."""
if not body:
return []
return sorted(set(int(m.group(1)) for m in LINKED_ISSUE_RE.finditer(body)))
def strip_html_comments(text: str) -> str:
"""Remove HTML comment blocks (<!-- ... -->) from text.
This prevents template placeholders inside HTML comments from being
parsed as real CHANGELOG markers.
"""
return re.sub(r"<!--.*?-->", "", text, flags=re.DOTALL)
def extract_markers(body: str) -> list[dict]:
"""Extract CHANGELOG-* markers from a PR body."""
if not body:
return []
# Strip HTML comments so template placeholders aren't treated as real markers
cleaned = strip_html_comments(body)
entries = []
has_opt_out = False
for m in MARKER_RE.finditer(cleaned):
category = m.group(1)
text = m.group(2).strip()
# CHANGELOG-NONE is an explicit opt-out — skip all other markers
if category == "NONE":
has_opt_out = True
continue
# Skip template placeholders
if text.startswith("{{") or text.startswith("{text") or not text:
continue
entries.append({"category": category, "text": text})
# If the PR explicitly opted out, return a special marker
if has_opt_out:
return [{"category": "NONE", "text": ""}]
return entries
def main() -> None:
parser = argparse.ArgumentParser(description="Fetch PRs in a release range")
parser.add_argument("--repo", required=True, help="GitHub repo (owner/name)")
parser.add_argument("--base-ref", required=True, help="Previous release tag")
parser.add_argument("--head-ref", required=True, help="Current release tag")
args = parser.parse_args()
commit_shas = get_commits(args.base_ref, args.head_ref)
seen_prs: set[int] = set()
prs: list[dict] = []
def process_pr(pr_num: int) -> None:
"""Fetch and record a single PR by number."""
data = fetch_pr_data(args.repo, pr_num)
if data is None:
return
if not should_include_pr(args.repo, data):
return
source_repo, data, internal_pr = normalize_pr_data(args.repo, pr_num, data)
author_login = get_author_login(data)
label_names = get_label_names(data)
body = data.get("body", "") or ""
explicit_entries = extract_markers(body)
linked_issues = extract_linked_issues(body)
file_paths = get_file_paths(data)
pr = {
"number": data.get("number", pr_num),
"url": data.get("url", "") if source_repo == PUBLIC_REPO else "",
"title": data.get("title", ""),
"author": author_login,
"body": body,
"labels": label_names,
"merged_at": data.get("mergedAt", ""),
"explicit_entries": explicit_entries,
"linked_issues": linked_issues,
"changed_files": file_paths,
"source_repo": source_repo,
}
if internal_pr is not None:
pr["internal_pr"] = internal_pr
prs.append(pr)
for sha in commit_shas:
pr_num = extract_pr_number(sha)
if pr_num is not None and pr_num not in seen_prs:
# Normal squash-merge commit
seen_prs.add(pr_num)
process_pr(pr_num)
else:
# Merge commit fallback: walk the merged-in commits for PR numbers.
# This handles branches merged via merge commit (e.g. security-patches)
# rather than the usual squash merge.
for merged_sha in get_merged_commits(sha):
inner_pr = extract_pr_number(merged_sha)
if inner_pr is not None and inner_pr not in seen_prs:
seen_prs.add(inner_pr)
process_pr(inner_pr)
output = {
"range": {"base": args.base_ref, "head": args.head_ref},
"prs": prs,
}
json.dump(output, sys.stdout, indent=2)
print() # trailing newline
if __name__ == "__main__":
main()