From 0c0863654f2000c9c372a22c08ec0dd7f48fad50 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Fri, 2 Oct 2026 10:38:52 -0700 Subject: [PATCH] ci: add required-check migration tooling and cutover guide --- ci/tools/migrate_precommit_checks.py | 230 +++++++++++++ .../tests/test_migrate_precommit_checks.py | 309 ++++++++++++++++++ 2 files changed, 539 insertions(+) create mode 100644 ci/tools/migrate_precommit_checks.py create mode 100644 ci/tools/tests/test_migrate_precommit_checks.py diff --git a/ci/tools/migrate_precommit_checks.py b/ci/tools/migrate_precommit_checks.py new file mode 100644 index 00000000000..f4b9d720eb3 --- /dev/null +++ b/ci/tools/migrate_precommit_checks.py @@ -0,0 +1,230 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Preview the main ruleset cutover from pre-commit.ci to GitHub Actions. + +Writes require --apply, merged workflows, and successful representative checks. +This tool does not uninstall the pre-commit.ci GitHub App. +""" + +from __future__ import annotations + +import argparse +import copy +import json +import re +import subprocess +import sys +from urllib.parse import urlparse + +OLD_CONTEXT = "pre-commit.ci - pr" +OLD_INTEGRATION = 68672 +ACTIONS_INTEGRATION = 15368 +CHECK_WORKFLOWS = { + "Pre-commit (Linux)": ".github/workflows/pre-commit.yml", + "Pre-commit (Windows)": ".github/workflows/pre-commit.yml", + "Documentation links": ".github/workflows/lychee.yml", +} + + +def github_api(endpoint: str, *, method: str = "GET", payload: dict | None = None, paginate: bool = False): + """Call gh with structured JSON input, without invoking a shell.""" + command = ["gh", "api", "--method", method, endpoint, "-H", "Accept: application/vnd.github+json"] + if paginate: + command.extend(["--paginate", "--slurp"]) + if payload is not None: + command.extend(["--input", "-"]) + result = subprocess.run( # noqa: S603 - arguments are passed directly, without a shell. + command, + input=json.dumps(payload) if payload is not None else None, + capture_output=True, + text=True, + check=True, + ) + return json.loads(result.stdout) + + +def migrate_rules(rules: list[dict]) -> list[dict]: + """Replace only the known pre-commit.ci requirement, preserving all other rules.""" + migrated = copy.deepcopy(rules) + for rule in migrated: + if rule["type"] != "required_status_checks": + continue + checks = rule["parameters"]["required_status_checks"] + old_checks = [check for check in checks if check["context"] == OLD_CONTEXT] + if not old_checks: + continue + if any(check.get("integration_id") != OLD_INTEGRATION for check in old_checks): + raise RuntimeError(f"{OLD_CONTEXT!r} is not bound to expected integration {OLD_INTEGRATION}") + existing = set() + for check in checks: + context = check["context"] + if context in CHECK_WORKFLOWS: + if check.get("integration_id") != ACTIONS_INTEGRATION: + raise RuntimeError(f"{context!r} already exists with a different or unspecified integration") + existing.add(context) + replacement = [ + {"context": context, "integration_id": ACTIONS_INTEGRATION} + for context in CHECK_WORKFLOWS + if context not in existing + ] + updated = [] + for check in checks: + if check["context"] == OLD_CONTEXT: + updated.extend(replacement) + replacement = [] + else: + updated.append(check) + rule["parameters"]["required_status_checks"] = updated + return migrated + + +def discover_changes(repository: str) -> list[dict]: + """Find active requirements on main and fetch their complete repository rulesets.""" + pages = github_api(f"repos/{repository}/rules/branches/main?per_page=100", paginate=True) + ruleset_ids = set() + for rule in (rule for page in pages for rule in page): + if rule["type"] != "required_status_checks": + continue + checks = rule["parameters"]["required_status_checks"] + if not any(check["context"] == OLD_CONTEXT for check in checks): + continue + if rule["ruleset_source_type"] != "Repository" or rule["ruleset_source"].lower() != repository.lower(): + raise RuntimeError( + f"Requirement is inherited from {rule['ruleset_source_type']} {rule['ruleset_source']} " + f"ruleset {rule['ruleset_id']}; its owner must migrate it. No repository changes were made." + ) + ruleset_ids.add(rule["ruleset_id"]) + + changes = [] + for ruleset_id in sorted(ruleset_ids): + endpoint = f"repos/{repository}/rulesets/{ruleset_id}" + original = github_api(endpoint) + if original["source_type"] != "Repository" or original["source"].lower() != repository.lower(): + raise RuntimeError(f"Ruleset {ruleset_id} is not owned by {repository}") + if original["target"] != "branch" or original["enforcement"] != "active": + raise RuntimeError(f"Ruleset {ruleset_id} changed scope or enforcement; rerun the preview") + migrated = migrate_rules(original["rules"]) + if migrated != original["rules"]: + changes.append({"endpoint": endpoint, "original": original, "payload": {"rules": migrated}}) + return changes + + +def verify_readiness(repository: str, verified_sha: str | None) -> str: + """Require merged workflow files and green checks on main or an identified main PR.""" + metadata = github_api(f"repos/{repository}") + if metadata["default_branch"] != "main": + raise RuntimeError("This migration is scoped to repositories whose default branch is main") + main_sha = github_api(f"repos/{repository}/branches/main")["commit"]["sha"] + for path in sorted(set(CHECK_WORKFLOWS.values())): + content = github_api(f"repos/{repository}/contents/{path}?ref={main_sha}") + if not isinstance(content, dict) or content.get("type") != "file": + raise RuntimeError(f"{path} is not a workflow file on current main") + + sha = verified_sha or main_sha + if sha != main_sha: + pages = github_api(f"repos/{repository}/commits/{sha}/pulls?per_page=100", paginate=True) + matching_prs = [ + pr + for page in pages + for pr in page + if pr["state"] == "open" + and pr["base"]["ref"] == "main" + and pr["base"]["repo"]["full_name"].lower() == repository.lower() + and sha in (pr["head"]["sha"], pr.get("merge_commit_sha")) + ] + if not matching_prs: + raise RuntimeError("--verified-sha must be current main or the head/merge SHA of an open PR targeting main") + + pages = github_api(f"repos/{repository}/commits/{sha}/check-runs?filter=latest&per_page=100", paginate=True) + checks = [check for page in pages for check in page["check_runs"]] + runs = {} + for context, workflow in CHECK_WORKFLOWS.items(): + matching = [check for check in checks if check["name"] == context and check["app"]["id"] == ACTIONS_INTEGRATION] + if not matching: + raise RuntimeError(f"Missing GitHub Actions check {context!r} on {sha}") + check = max(matching, key=lambda check: check["id"]) + if check["status"] != "completed" or check["conclusion"] != "success": + raise RuntimeError(f"Check {context!r} on {sha} is not completed and successful") + url = urlparse(check["details_url"]) + match = re.fullmatch(rf"/{re.escape(repository)}/actions/runs/([0-9]+)/job/[0-9]+", url.path, re.IGNORECASE) + if url.hostname != "github.com" or match is None: + raise RuntimeError(f"Cannot identify the workflow run for {context!r}") + run_id = match.group(1) + if run_id not in runs: + runs[run_id] = github_api(f"repos/{repository}/actions/runs/{run_id}") + run = runs[run_id] + if ( + run["path"].split("@", 1)[0] != workflow + or run["check_suite_id"] != check["check_suite"]["id"] + or run["status"] != "completed" + or run["conclusion"] != "success" + ): + raise RuntimeError(f"Check {context!r} does not belong to a successful {workflow} run") + return sha + + +def apply_changes(repository: str, changes: list[dict], verified_sha: str | None) -> str: + """Preflight all changes before issuing narrowly scoped rules updates.""" + sha = verify_readiness(repository, verified_sha) + for change in changes: + if github_api(change["endpoint"]) != change["original"]: + raise RuntimeError(f"{change['endpoint']} changed after discovery; rerun the preview") + for change in changes: + updated = github_api(change["endpoint"], method="PUT", payload=change["payload"]) + original = change["original"] + settings = ("name", "target", "enforcement", "conditions", "bypass_actors") + if updated["rules"] != change["payload"]["rules"] or any( + updated.get(key) != original.get(key) for key in settings + ): + raise RuntimeError(f"Unexpected response after updating {change['endpoint']}; inspect that ruleset") + return sha + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repo", required=True, choices=("NVIDIA/cuda-python", "NVIDIA-dev/cuda-python-private")) + parser.add_argument("--verified-sha", help="Full SHA of current main or a representative open PR head/merge commit") + parser.add_argument( + "--apply", action="store_true", help="Update required checks after explicit cutover authorization" + ) + args = parser.parse_args(argv) + if args.verified_sha and re.fullmatch(r"[0-9a-f]{40}", args.verified_sha) is None: + parser.error("--verified-sha must be a full 40-character lowercase commit SHA") + + try: + changes = discover_changes(args.repo) + preview = { + "repository": args.repo, + "branch": "main", + "mode": "apply" if args.apply else "dry-run", + "changes": [ + { + "endpoint": change["endpoint"], + "before": change["original"], + "put_payload": change["payload"], + } + for change in changes + ], + } + print(json.dumps(preview, indent=2)) + if args.apply and changes: + sha = apply_changes(args.repo, changes, args.verified_sha) + print(f"Updated {len(changes)} ruleset(s); representative checks verified on {sha}", file=sys.stderr) + elif not changes: + print("No matching active pre-commit.ci requirement on main; no updates made.", file=sys.stderr) + else: + print( + "Dry run: no writes made. --apply checks merged workflows and successful checks before writing.", + file=sys.stderr, + ) + except (RuntimeError, subprocess.CalledProcessError, json.JSONDecodeError) as exc: + if isinstance(exc, subprocess.CalledProcessError) and exc.stderr: + print(exc.stderr.strip(), file=sys.stderr) + print(f"error: {exc}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/ci/tools/tests/test_migrate_precommit_checks.py b/ci/tools/tests/test_migrate_precommit_checks.py new file mode 100644 index 00000000000..816e299c09b --- /dev/null +++ b/ci/tools/tests/test_migrate_precommit_checks.py @@ -0,0 +1,309 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import copy +import json + +import pytest + +from ci.tools import migrate_precommit_checks as migration + +REPOSITORY = "NVIDIA/cuda-python" +MAIN_SHA = "a" * 40 +PR_SHA = "b" * 40 +RULESET_ENDPOINT = f"repos/{REPOSITORY}/rulesets/12" + + +@pytest.fixture +def ruleset(): + return { + "id": 12, + "name": "Main protection", + "source_type": "Repository", + "source": REPOSITORY, + "target": "branch", + "enforcement": "active", + "conditions": {"ref_name": {"include": ["~DEFAULT_BRANCH"], "exclude": []}}, + "bypass_actors": [{"actor_id": 9, "actor_type": "Team", "bypass_mode": "pull_request"}], + "rules": [ + {"type": "deletion"}, + { + "type": "required_status_checks", + "parameters": { + "strict_required_status_checks_policy": True, + "do_not_enforce_on_create": True, + "required_status_checks": [ + {"context": "CI", "integration_id": 15368}, + {"context": "pre-commit.ci - pr", "integration_id": 68672}, + {"context": "Other service", "integration_id": 777}, + ], + }, + }, + {"type": "pull_request", "parameters": {"required_approving_review_count": 2}}, + ], + } + + +@pytest.fixture +def github(monkeypatch, ruleset): + effective_rule = { + **copy.deepcopy(ruleset["rules"][1]), + "ruleset_id": ruleset["id"], + "ruleset_source_type": "Repository", + "ruleset_source": REPOSITORY, + } + checks = [ + { + "id": index, + "name": context, + "app": {"id": 15368}, + "status": "completed", + "conclusion": "success", + "details_url": f"https://github.com/{REPOSITORY}/actions/runs/{index}/job/100", + "check_suite": {"id": index + 100}, + } + for index, context in enumerate(migration.CHECK_WORKFLOWS, start=1) + ] + responses = { + f"repos/{REPOSITORY}/rules/branches/main?per_page=100": [[effective_rule]], + RULESET_ENDPOINT: copy.deepcopy(ruleset), + f"repos/{REPOSITORY}": {"default_branch": "main"}, + f"repos/{REPOSITORY}/branches/main": {"commit": {"sha": MAIN_SHA}}, + f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100": [{"check_runs": checks}], + f"repos/{REPOSITORY}/commits/{PR_SHA}/check-runs?filter=latest&per_page=100": [{"check_runs": checks}], + f"repos/{REPOSITORY}/commits/{PR_SHA}/pulls?per_page=100": [ + [ + { + "state": "open", + "base": {"ref": "main", "repo": {"full_name": REPOSITORY}}, + "head": {"sha": PR_SHA}, + "merge_commit_sha": "c" * 40, + } + ] + ], + } + for path in set(migration.CHECK_WORKFLOWS.values()): + responses[f"repos/{REPOSITORY}/contents/{path}?ref={MAIN_SHA}"] = {"path": path, "type": "file"} + for check in checks: + responses[f"repos/{REPOSITORY}/actions/runs/{check['id']}"] = { + "path": migration.CHECK_WORKFLOWS[check["name"]], + "check_suite_id": check["check_suite"]["id"], + "status": "completed", + "conclusion": "success", + } + calls = [] + + def api(endpoint, *, method="GET", payload=None, paginate=False): + calls.append((method, endpoint, copy.deepcopy(payload))) + if method == "PUT": + responses[endpoint]["rules"] = copy.deepcopy(payload["rules"]) + return copy.deepcopy(responses[endpoint]) + + monkeypatch.setattr(migration, "github_api", api) + return responses, calls + + +@pytest.mark.agent_authored(model="gpt-6") +def test_replaces_service_without_changing_unrelated_rules_or_parameters(ruleset): + original = copy.deepcopy(ruleset) + migrated = migration.migrate_rules(ruleset["rules"]) + + assert ruleset == original + assert migrated[0] == original["rules"][0] + assert migrated[2] == original["rules"][2] + parameters = migrated[1]["parameters"] + assert parameters["strict_required_status_checks_policy"] is True + assert parameters["do_not_enforce_on_create"] is True + checks = parameters["required_status_checks"] + assert checks == [ + {"context": "CI", "integration_id": 15368}, + {"context": "Pre-commit (Linux)", "integration_id": 15368}, + {"context": "Pre-commit (Windows)", "integration_id": 15368}, + {"context": "Documentation links", "integration_id": 15368}, + {"context": "Other service", "integration_id": 777}, + ] + assert migration.migrate_rules(migrated) == migrated + + +@pytest.mark.agent_authored(model="gpt-6") +def test_existing_actions_requirements_are_not_duplicated(ruleset): + checks = ruleset["rules"][1]["parameters"]["required_status_checks"] + existing = {"context": "Documentation links", "integration_id": 15368} + checks.append(existing) + + migrated = migration.migrate_rules(ruleset["rules"]) + + assert migrated[1]["parameters"]["required_status_checks"].count(existing) == 1 + + +@pytest.mark.parametrize("integration", [None, 777]) +@pytest.mark.agent_authored(model="gpt-6") +def test_refuses_unexpected_old_integration(ruleset, integration): + ruleset["rules"][1]["parameters"]["required_status_checks"][1]["integration_id"] = integration + + with pytest.raises(RuntimeError, match="expected integration"): + migration.migrate_rules(ruleset["rules"]) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_refuses_conflicting_replacement_integration(ruleset): + ruleset["rules"][1]["parameters"]["required_status_checks"].append( + {"context": "Documentation links", "integration_id": 777} + ) + + with pytest.raises(RuntimeError, match="different or unspecified integration"): + migration.migrate_rules(ruleset["rules"]) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_duplicate_valid_context_cannot_hide_conflicting_integration(ruleset): + ruleset["rules"][1]["parameters"]["required_status_checks"].extend( + [ + {"context": "Documentation links", "integration_id": 777}, + {"context": "Documentation links", "integration_id": 15368}, + ] + ) + + with pytest.raises(RuntimeError, match="different or unspecified integration"): + migration.migrate_rules(ruleset["rules"]) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_cli_defaults_to_read_only_preview(github, capsys): + responses, calls = github + + assert migration.main(["--repo", REPOSITORY]) == 0 + + preview = json.loads(capsys.readouterr().out) + assert preview["mode"] == "dry-run" + assert preview["changes"][0]["before"] == responses[RULESET_ENDPOINT] + assert set(preview["changes"][0]["put_payload"]) == {"rules"} + assert all(method == "GET" for method, endpoint, payload in calls) + assert not any("/contents/" in endpoint for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_inherited_requirement_is_rejected_before_any_write(github): + responses, calls = github + rule = responses[f"repos/{REPOSITORY}/rules/branches/main?per_page=100"][0][0] + rule["ruleset_source_type"] = "Organization" + rule["ruleset_source"] = "NVIDIA" + + with pytest.raises(RuntimeError, match="inherited from Organization NVIDIA ruleset 12"): + migration.discover_changes(REPOSITORY) + + assert len(calls) == 1 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_replacement_workflows_must_be_files_on_current_main(github): + responses, calls = github + responses[f"repos/{REPOSITORY}/contents/.github/workflows/lychee.yml?ref={MAIN_SHA}"] = [] + + with pytest.raises(RuntimeError, match="not a workflow file on current main"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.parametrize("verified_sha", [None, PR_SHA]) +@pytest.mark.agent_authored(model="gpt-6") +def test_apply_requires_green_representative_checks_and_preserves_settings(github, ruleset, verified_sha): + responses, calls = github + changes = migration.discover_changes(REPOSITORY) + + assert migration.apply_changes(REPOSITORY, changes, verified_sha) == (verified_sha or MAIN_SHA) + + writes = [call for call in calls if call[0] == "PUT"] + assert writes == [("PUT", RULESET_ENDPOINT, {"rules": migration.migrate_rules(ruleset["rules"])})] + assert {key: value for key, value in responses[RULESET_ENDPOINT].items() if key != "rules"} == { + key: value for key, value in ruleset.items() if key != "rules" + } + + +@pytest.mark.parametrize("conclusion", ["failure", "cancelled", "skipped", "neutral", None]) +@pytest.mark.agent_authored(model="gpt-6") +def test_failed_or_skipped_checks_prevent_all_writes(github, conclusion): + responses, calls = github + checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] + checks[-1]["conclusion"] = conclusion + changes = migration.discover_changes(REPOSITORY) + + with pytest.raises(RuntimeError, match="not completed and successful"): + migration.apply_changes(REPOSITORY, changes, None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_newer_pending_run_cannot_be_masked_by_older_success(github): + responses, calls = github + checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] + checks.append({**checks[0], "id": 99, "status": "in_progress", "conclusion": None}) + + with pytest.raises(RuntimeError, match="not completed and successful"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_other_apps_cannot_satisfy_replacement_checks(github): + responses, calls = github + checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] + checks[0]["app"]["id"] = 777 + + with pytest.raises(RuntimeError, match="Missing GitHub Actions check"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.parametrize( + ("field", "value"), + [("path", ".github/workflows/unrelated.yml"), ("check_suite_id", 777), ("conclusion", "failure")], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_check_must_belong_to_the_expected_successful_workflow(github, field, value): + responses, calls = github + responses[f"repos/{REPOSITORY}/actions/runs/1"][field] = value + + with pytest.raises(RuntimeError, match="does not belong to a successful"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_unassociated_sha_is_rejected(github): + responses, calls = github + responses[f"repos/{REPOSITORY}/commits/{PR_SHA}/pulls?per_page=100"] = [[]] + + with pytest.raises(RuntimeError, match="open PR targeting main"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), PR_SHA) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_ruleset_edit_during_preflight_prevents_writes(github): + responses, calls = github + changes = migration.discover_changes(REPOSITORY) + responses[RULESET_ENDPOINT]["conditions"]["ref_name"]["exclude"].append("refs/heads/release") + + with pytest.raises(RuntimeError, match="changed after discovery"): + migration.apply_changes(REPOSITORY, changes, None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_sha_must_be_explicit_full_commit(github): + responses, calls = github + + with pytest.raises(SystemExit, match="2"): + migration.main(["--repo", REPOSITORY, "--verified-sha", "main", "--apply"]) + + assert not calls