From fc9653cfee858ff29a6adbab4d1517dce39cb559 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:34:54 -0700 Subject: [PATCH 01/12] ci: check PR and nightly documentation links independently --- .github/workflows/build-docs.yml | 53 --- .github/workflows/ci-nightly.yml | 13 +- .github/workflows/ci.yml | 37 +-- .github/workflows/lychee.yml | 162 +++++++++ CONTRIBUTING.md | 18 +- ci/README-pre-commit-migration.md | 100 ++++++ ci/tools/build_docs_for_link_check.sh | 56 ++++ ci/tools/migrate_precommit_checks.py | 230 +++++++++++++ ci/tools/prepare_lychee_inputs.py | 70 ++++ .../tests/test_migrate_precommit_checks.py | 309 ++++++++++++++++++ ci/tools/tests/test_prepare_lychee_inputs.py | 70 ++++ cuda_core/pixi.toml | 4 + lychee.toml | 2 +- 13 files changed, 1031 insertions(+), 93 deletions(-) create mode 100644 .github/workflows/lychee.yml create mode 100644 ci/README-pre-commit-migration.md create mode 100644 ci/tools/build_docs_for_link_check.sh create mode 100644 ci/tools/migrate_precommit_checks.py create mode 100644 ci/tools/prepare_lychee_inputs.py create mode 100644 ci/tools/tests/test_migrate_precommit_checks.py create mode 100644 ci/tools/tests/test_prepare_lychee_inputs.py diff --git a/.github/workflows/build-docs.yml b/.github/workflows/build-docs.yml index 52ee9698ab0..f62d1ae5e9d 100644 --- a/.github/workflows/build-docs.yml +++ b/.github/workflows/build-docs.yml @@ -284,59 +284,6 @@ jobs: fi mv ${COMPONENT}/docs/build/html/* artifacts/docs/${TARGET} - - name: Write rendered docs file list - if: ${{ !inputs.is-release && startsWith(github.ref_name, 'pull-request/') }} - run: | - find "${GITHUB_WORKSPACE}/artifacts/docs" -type f -name '*.html' ! -path '*/_static/*' \ - | LC_ALL=C sort > lychee-rendered-html-files.txt - if [[ ! -s lychee-rendered-html-files.txt ]]; then - echo "error: no rendered HTML pages found for lychee" >&2 - exit 1 - fi - wc -l lychee-rendered-html-files.txt - - - name: Restore lychee cache - if: ${{ !inputs.is-release && startsWith(github.ref_name, 'pull-request/') }} - id: restore-lychee-cache - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 - with: - path: .lycheecache - key: docs-rendered-lychee-${{ env.PR_NUMBER }}-${{ github.sha }} - restore-keys: | - docs-rendered-lychee-${{ env.PR_NUMBER }}- - - - name: Check rendered docs links - if: ${{ !inputs.is-release && startsWith(github.ref_name, 'pull-request/') }} - uses: lycheeverse/lychee-action@6da1d14f3a43098a294b7696d93d938aa8d20fc0 # unreleased: supports v0.24.x archive layout - with: - args: >- - --files-from ${{ github.workspace }}/lychee-rendered-html-files.txt - --include-fragments=full - --cache - --max-cache-age 1d - --max-concurrency 16 - --host-concurrency 2 - --host-request-interval 250ms - --max-retries 3 - --retry-wait-time 5 - --timeout 30 - --no-progress - --config ${{ github.workspace }}/lychee.toml - fail: true - failIfEmpty: true - format: markdown - jobSummary: false - lycheeVersion: v0.24.2 - output: lychee-rendered-html.md - token: ${{ github.token }} - - - name: Save lychee cache - if: ${{ always() && !inputs.is-release && startsWith(github.ref_name, 'pull-request/') && steps.restore-lychee-cache.outputs.cache-hit != 'true' && steps.restore-lychee-cache.outputs.cache-primary-key != '' }} - uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 - with: - path: .lycheecache - key: ${{ steps.restore-lychee-cache.outputs.cache-primary-key }} - - name: Upload docs GitHub Pages artifact if: ${{ inputs.deploy-docs }} uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0 diff --git a/.github/workflows/ci-nightly.yml b/.github/workflows/ci-nightly.yml index a3179c1e155..3b75bdd2126 100644 --- a/.github/workflows/ci-nightly.yml +++ b/.github/workflows/ci-nightly.yml @@ -32,6 +32,15 @@ on: default: '' jobs: + documentation-links: + name: "Nightly: Documentation links" + if: ${{ github.repository_owner == 'nvidia' }} + permissions: + contents: read + uses: ./.github/workflows/lychee.yml + with: + refresh-cache: true + test-ci-tools-for-release: name: "Nightly: CI tools for release" if: ${{ github.repository_owner == 'nvidia' }} @@ -313,6 +322,7 @@ jobs: if: ${{ always() && github.repository_owner == 'nvidia' }} runs-on: ubuntu-latest needs: + - documentation-links - test-ci-tools-for-release - find-wheels - test-pytorch-linux @@ -334,7 +344,8 @@ jobs: # See ci.yml for the full rationale on why we must use always() # and explicitly check each result rather than relying on the # default behaviour. - if ${{ needs.test-ci-tools-for-release.result == 'cancelled' || + if ${{ needs.documentation-links.result != 'success' || + needs.test-ci-tools-for-release.result == 'cancelled' || needs.test-ci-tools-for-release.result == 'failure' || needs.find-wheels.result != 'success' }}; then exit 1 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 313be5ff77e..257aa480d2a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -622,38 +622,6 @@ jobs: with: is-release: ${{ github.ref_type == 'tag' }} - precommit-windows: - name: Pre-commit on Windows - runs-on: windows-latest - if: ${{ github.repository_owner == 'nvidia' && !fromJSON(needs.should-skip.outputs.skip) }} - needs: - - should-skip - permissions: - contents: read - steps: - - name: Checkout repository - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - fetch-depth: 1 - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: '3.13' - - - name: Install pre-commit - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip pre-commit - - - name: Run pre-commit - shell: bash - run: | - set -euxo pipefail - SKIP=lychee pre-commit run --all-files - checks: name: Check job status if: ${{ always() && github.repository_owner == 'nvidia' }} @@ -674,7 +642,6 @@ jobs: - api-check-core-vs-release - api-check-core-vs-base - doc - - precommit-windows steps: - name: Exit env: @@ -712,14 +679,12 @@ jobs: fi } - # Control jobs, the universal linux build, docs, and Windows - # pre-commit checks always run. + # Control jobs, the universal Linux build, and docs always run. check_result "ci-vars" "success" check_result "should-skip" "success" check_result "detect-changes" "success" check_result "build-linux-64" "success" check_result "doc" "success" - check_result "precommit-windows" "success" # Optional platform builds and wheel tests share the platform plan. linux_expected="skipped" diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml new file mode 100644 index 00000000000..4a622aaf591 --- /dev/null +++ b/.github/workflows/lychee.yml @@ -0,0 +1,162 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +name: "CI: Documentation links" + +on: + pull_request: + types: [opened, reopened, synchronize] + workflow_call: + inputs: + refresh-cache: + description: "Check links afresh and publish a new cache baseline" + type: boolean + default: false + workflow_dispatch: + inputs: + refresh-cache: + description: "Check links afresh and publish a new cache baseline" + type: boolean + default: false + push: + # Temporary bootstrap trigger; remove after manual dispatch is available. + branches: ["rwgk/maint/lychee-pr-nightly"] + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }}-${{ inputs.refresh-cache || false }} + cancel-in-progress: true + +defaults: + run: + shell: bash --noprofile --norc -euo pipefail {0} + +env: + PIXI_LOCKED: "true" + LYCHEE_VERSION: v0.24.2 + +jobs: + links: + name: Lychee (${{ matrix.kind }}) + runs-on: ubuntu-latest + timeout-minutes: 60 + strategy: + fail-fast: false + matrix: + kind: [authored, rendered] + steps: + - name: Checkout checked revision + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.event.pull_request.head.sha || github.sha }} + fetch-depth: 0 + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.13" + + - name: Setup pixi + if: ${{ matrix.kind == 'rendered' }} + uses: ./.github/actions/setup-pixi + + - name: Build all documentation from sources + if: ${{ matrix.kind == 'rendered' }} + run: | + export CUDA_PYTHON_DOCS_GITHUB_REF="$(git rev-parse HEAD)" + pixi run --manifest-path cuda_core/pixi.toml -e docs docs-build-all-latest + + - name: Prepare link-check inputs and cache policy + id: inputs + env: + KIND: ${{ matrix.kind }} + POLICY_HASH: ${{ hashFiles('lychee.toml', '.github/workflows/lychee.yml', 'ci/tools/prepare_lychee_inputs.py') }} + run: | + python ci/tools/prepare_lychee_inputs.py "${KIND}" \ + --output "${GITHUB_WORKSPACE}/lychee-files.txt" + prefix="lychee-v1-${LYCHEE_VERSION}-${KIND}-${POLICY_HASH}-baseline-" + echo "prefix=${prefix}" >> "${GITHUB_OUTPUT}" + echo "key=${prefix}${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" >> "${GITHUB_OUTPUT}" + if [[ "${KIND}" == "rendered" ]]; then + echo "fragment-args=--include-fragments=full" >> "${GITHUB_OUTPUT}" + fi + + - name: Restore successful checks from the nightly baseline + if: ${{ !inputs.refresh-cache }} + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: .lycheecache + key: ${{ steps.inputs.outputs.key }} + restore-keys: ${{ steps.inputs.outputs.prefix }} + + - name: Check documentation links + id: lychee + uses: lycheeverse/lychee-action@6da1d14f3a43098a294b7696d93d938aa8d20fc0 # supports v0.24.x archive layout + with: + args: >- + --files-from ${{ github.workspace }}/lychee-files.txt + ${{ steps.inputs.outputs.fragment-args }} + --cache + --max-cache-age 1d + --max-concurrency 16 + --host-concurrency 2 + --host-request-interval 250ms + --max-retries 3 + --retry-wait-time 5 + --timeout 30 + --no-progress + --config ${{ github.workspace }}/lychee.toml + fail: true + failIfEmpty: true + format: markdown + jobSummary: true + lycheeVersion: ${{ env.LYCHEE_VERSION }} + output: lychee-${{ matrix.kind }}.md + token: ${{ github.token }} + + - name: Require a successful checker result + if: ${{ steps.lychee.outcome == 'success' }} + env: + EXIT_CODE: ${{ steps.lychee.outputs.exit_code }} + run: | + if [[ "${EXIT_CODE}" != "0" ]]; then + echo "::error::Lychee did not report success (exit_code='${EXIT_CODE}')." + exit 1 + fi + + - name: Publish fresh successful checks + # Manual branch tests publish isolated branch caches. Nightly callers + # on the default branch publish the baseline that all PRs can restore. + if: ${{ always() && inputs.refresh-cache && steps.inputs.outputs.key != '' && hashFiles('.lycheecache') != '' }} + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: .lycheecache + key: ${{ steps.inputs.outputs.key }} + + - name: Upload link-check report + if: ${{ always() }} + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: lychee-${{ matrix.kind }}-${{ github.run_id }}-${{ github.run_attempt }} + path: lychee-${{ matrix.kind }}.md + if-no-files-found: ignore + retention-days: 7 + + checks: + name: Documentation links + if: ${{ always() }} + needs: links + runs-on: ubuntu-latest + steps: + - name: Require both link checks to succeed + env: + RESULT: ${{ needs.links.result }} + run: | + if [[ "${RESULT}" != "success" ]]; then + echo "::error::Documentation link checks did not succeed (${RESULT})." + exit 1 + fi diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 8b7b28d475e..b1fe8519f81 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -214,7 +214,21 @@ A few things to keep in mind: --config-file cuda_core/pyproject.toml ## Pre-commit -This project uses [pre-commit.ci](https://pre-commit.ci/) with GitHub Actions. All pull requests are automatically checked for pre-commit compliance, and any pre-commit failures will block merging until resolved. +GitHub Actions runs all pre-commit hooks on Linux and Windows for every pull +request update, including draft PRs. These jobs skip lychee and the local hook +installation reminder. A separate Linux workflow checks authored documentation +and freshly rendered HTML, including link fragments. It runs on PR updates and +from nightly CI; copied `pull-request/*` branches do not repeat these checks. + +Nightly link checks start with a fresh cache and publish successful checks for +PRs to reuse for up to one day. Quarterly pre-commit hook updates arrive as +draft PRs for review. Lychee version updates are maintained separately so its +local hook and CI binary stay aligned. The local lychee hook keeps its current +behavior until a stable release supports an explicit cache location. + +See [the pre-commit migration procedure](ci/README-pre-commit-migration.md) for +manual workflow testing and the required-check cutover. The existing +pre-commit.ci service remains required until that cutover is complete. To set yourself up for running pre-commit checks locally and to catch issues before pushing your changes, follow these steps: @@ -229,7 +243,7 @@ Installing the hook is required, not optional. Some of the automated checks keep the tree consistent if they run on *every* commit. Relying on manual `pre-commit run --all-files` invocations means these checks can be skipped between commits, leaving stale headers or out-of-date stubs in the history. -If the hook isn't installed, `pre-commit run` (and CI) will print a visible +If the hook isn't installed, local `pre-commit run` will print a visible warning reminding you to run `pre-commit install`. Windows contributors: see [Pre-commit lychee workaround](#pre-commit-lychee-workaround) under Development on Windows. diff --git a/ci/README-pre-commit-migration.md b/ci/README-pre-commit-migration.md new file mode 100644 index 00000000000..1cc8dcad67d --- /dev/null +++ b/ci/README-pre-commit-migration.md @@ -0,0 +1,100 @@ +# Migrating from pre-commit.ci + +The replacement checks are `Pre-commit (Linux)`, `Pre-commit (Windows)`, and +`Documentation links`, all from the GitHub Actions App (integration ID 15368). +They replace `pre-commit.ci - pr` from integration ID 68672. + +The workflow pull requests can remain draft while their runs are reviewed. +Keep pre-commit.ci installed and required throughout that review. Do not change +the live required checks until the replacement workflows are merged and green. + +## Test the draft workflows + +Use the PR branch as `REF` while reviewing the implementation. These commands +run hosted checks without copying a PR into the heavyweight CUDA CI: + +```sh +gh workflow run pre-commit.yml --repo NVIDIA/cuda-python --ref REF +gh workflow run pre-commit-autoupdate.yml --repo NVIDIA/cuda-python --ref REF \ + -f dry-run=true +gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ + -f refresh-cache=true +gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ + -f refresh-cache=false +gh run list --repo NVIDIA/cuda-python --branch REF +``` + +The updater dry run shows its diff without opening or updating a PR. Lychee's +refresh mode skips restoration and publishes a new snapshot even when some +links fail; only successful checks enter the persisted cache. Its normal mode +restores the newest matching snapshot and does not publish one. Branch tests +write branch-scoped caches. After merge, nightly refreshes run on `main` and +publish the baseline that all PRs can read, including fork PRs. Separate +authored/rendered namespaces include the checker version and checking policy; +documentation edits do not invalidate previously successful external checks. + +For the same source-based docs build locally: + +```sh +PIXI_LOCKED=true pixi run --manifest-path cuda_core/pixi.toml -e docs \ + docs-build-all-latest +``` + +The build verifies all three sibling libraries import from the checkout, +installs metapackage metadata without dependencies, and assembles all four +latest documentation trees under `artifacts/docs`. + +## Preview the ruleset change + +Run from the repository root with an authenticated GitHub CLI: + +```sh +pixi exec --spec python python ci/tools/migrate_precommit_checks.py \ + --repo NVIDIA/cuda-python > /tmp/cuda-python-precommit-cutover.json +``` + +The default mode only reads GitHub. It discovers active rules applying to +`main`, fetches complete repository rulesets, and prints their original +configuration and the exact proposed PUT payload. Review that JSON before +cutover. The payload changes only `rules`: unrelated rules and required checks, +strictness, branch conditions, bypass actors, and enforcement remain in place. +The tool rejects inherited organization rulesets and unexpected integration IDs; +an organization ruleset's owner must handle that migration separately. + +## Cut over after merge + +1. Merge `.github/workflows/pre-commit.yml` and `.github/workflows/lychee.yml`. + Check that they exist on the current default `main` branch. +2. Obtain successful, completed runs of all three replacement checks on current + `main` or a representative open PR targeting `main`. Record the full checked + commit SHA. This also verifies the PR event path when using a PR; a branch + push run alone does not validate PR-specific behavior. +3. After explicit authorization for the ruleset cutover, run the preview again, + then run the same command with `--apply --verified-sha FULL_COMMIT_SHA`. + Omitting `--verified-sha` requires checks on the current `main` commit. +4. Confirm the active rules now require all three replacement contexts from the + GitHub Actions App and still contain every unrelated requirement. Verify on + a representative PR that an intentional formatting or broken-link failure + is reported as a required failing check and blocks merging; repair it and + verify the same check passes. Check this independently of the PR's draft + status, which also prevents merging. +5. Only after that verification, remove this repository from the pre-commit.ci + GitHub App installation's repository access, or disable its service through + the repository/service settings. Preserve access for other repositories + using the same installation. The migration script never changes App access. + +`--apply` verifies the workflow files on `main`, the selected commit's checks, +their App IDs, and their successful owning workflow runs before writing. It +re-reads all affected rulesets and refuses changes made since discovery. +GitHub does not provide an atomic transaction across ruleset updates: if a +write fails partway through, inspect the emitted preview and actual rulesets +before retrying. Do not remove the service while any old required check remains. + +## Validate the helper + +```sh +pixi exec --spec python --spec pytest python -m pytest \ + --noconftest ci/tools/tests/test_migrate_precommit_checks.py +``` + +The tests mock GitHub; they cannot change repository settings. diff --git a/ci/tools/build_docs_for_link_check.sh b/ci/tools/build_docs_for_link_check.sh new file mode 100644 index 00000000000..26f482b2580 --- /dev/null +++ b/ci/tools/build_docs_for_link_check.sh @@ -0,0 +1,56 @@ +#!/usr/bin/env bash + +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Run through cuda_core's docs environment, which builds all three sibling +# packages from this checkout rather than installing their published releases. +set -euo pipefail + +REPO_ROOT=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd) +cd "${REPO_ROOT}" + +if [[ "${PIXI_ENVIRONMENT_NAME:-}" != "docs" ]]; then + echo 'Run with: pixi run --manifest-path cuda_core/pixi.toml -e docs docs-build-all-latest' >&2 + exit 1 +fi + +export CUDA_PYTHON_DOCS_GITHUB_REF="${CUDA_PYTHON_DOCS_GITHUB_REF:-$(git rev-parse HEAD)}" +export BUILD_LATEST=1 +export BUILD_PREVIEW=0 + +# The metapackage pins released cuda-core versions. Its dependencies are already +# installed from local Pixi source packages, so never resolve those pins on PyPI. +python -m pip install --no-deps "${REPO_ROOT}/cuda_python" +python - <<'PY' +from importlib import import_module +from importlib.metadata import version +from pathlib import Path + +for distribution, module_name in ( + ("cuda-bindings", "cuda.bindings"), + ("cuda-core", "cuda.core"), + ("cuda-pathfinder", "cuda.pathfinder"), +): + module = import_module(module_name) + source = Path.cwd() / distribution.replace("-", "_") + if not Path(module.__file__).resolve().is_relative_to(source): + raise RuntimeError(f"{distribution} must be imported from the checkout: {module.__file__}") + installed_version = version(distribution) + if module.__version__ != installed_version: + raise RuntimeError( + f"{distribution} import version {module.__version__} differs from metadata {installed_version}" + ) + print(f"{distribution}: {installed_version} ({module.__file__})") +print(f"cuda-python: {version('cuda-python')}") +PY + +pushd cuda_python/docs >/dev/null +# The docs Makefiles use O as an optional Sphinx argument. Ignore an unrelated +# host variable of that name so Sphinx does not treat it as an input filename. +env -u O bash ./build_all_docs.sh latest-only +popd >/dev/null + +rm -rf artifacts/docs +mkdir -p artifacts/docs +cp -a cuda_python/docs/build/html/. artifacts/docs/ diff --git a/ci/tools/migrate_precommit_checks.py b/ci/tools/migrate_precommit_checks.py new file mode 100644 index 00000000000..f4b9d720eb3 --- /dev/null +++ b/ci/tools/migrate_precommit_checks.py @@ -0,0 +1,230 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Preview the main ruleset cutover from pre-commit.ci to GitHub Actions. + +Writes require --apply, merged workflows, and successful representative checks. +This tool does not uninstall the pre-commit.ci GitHub App. +""" + +from __future__ import annotations + +import argparse +import copy +import json +import re +import subprocess +import sys +from urllib.parse import urlparse + +OLD_CONTEXT = "pre-commit.ci - pr" +OLD_INTEGRATION = 68672 +ACTIONS_INTEGRATION = 15368 +CHECK_WORKFLOWS = { + "Pre-commit (Linux)": ".github/workflows/pre-commit.yml", + "Pre-commit (Windows)": ".github/workflows/pre-commit.yml", + "Documentation links": ".github/workflows/lychee.yml", +} + + +def github_api(endpoint: str, *, method: str = "GET", payload: dict | None = None, paginate: bool = False): + """Call gh with structured JSON input, without invoking a shell.""" + command = ["gh", "api", "--method", method, endpoint, "-H", "Accept: application/vnd.github+json"] + if paginate: + command.extend(["--paginate", "--slurp"]) + if payload is not None: + command.extend(["--input", "-"]) + result = subprocess.run( # noqa: S603 - arguments are passed directly, without a shell. + command, + input=json.dumps(payload) if payload is not None else None, + capture_output=True, + text=True, + check=True, + ) + return json.loads(result.stdout) + + +def migrate_rules(rules: list[dict]) -> list[dict]: + """Replace only the known pre-commit.ci requirement, preserving all other rules.""" + migrated = copy.deepcopy(rules) + for rule in migrated: + if rule["type"] != "required_status_checks": + continue + checks = rule["parameters"]["required_status_checks"] + old_checks = [check for check in checks if check["context"] == OLD_CONTEXT] + if not old_checks: + continue + if any(check.get("integration_id") != OLD_INTEGRATION for check in old_checks): + raise RuntimeError(f"{OLD_CONTEXT!r} is not bound to expected integration {OLD_INTEGRATION}") + existing = set() + for check in checks: + context = check["context"] + if context in CHECK_WORKFLOWS: + if check.get("integration_id") != ACTIONS_INTEGRATION: + raise RuntimeError(f"{context!r} already exists with a different or unspecified integration") + existing.add(context) + replacement = [ + {"context": context, "integration_id": ACTIONS_INTEGRATION} + for context in CHECK_WORKFLOWS + if context not in existing + ] + updated = [] + for check in checks: + if check["context"] == OLD_CONTEXT: + updated.extend(replacement) + replacement = [] + else: + updated.append(check) + rule["parameters"]["required_status_checks"] = updated + return migrated + + +def discover_changes(repository: str) -> list[dict]: + """Find active requirements on main and fetch their complete repository rulesets.""" + pages = github_api(f"repos/{repository}/rules/branches/main?per_page=100", paginate=True) + ruleset_ids = set() + for rule in (rule for page in pages for rule in page): + if rule["type"] != "required_status_checks": + continue + checks = rule["parameters"]["required_status_checks"] + if not any(check["context"] == OLD_CONTEXT for check in checks): + continue + if rule["ruleset_source_type"] != "Repository" or rule["ruleset_source"].lower() != repository.lower(): + raise RuntimeError( + f"Requirement is inherited from {rule['ruleset_source_type']} {rule['ruleset_source']} " + f"ruleset {rule['ruleset_id']}; its owner must migrate it. No repository changes were made." + ) + ruleset_ids.add(rule["ruleset_id"]) + + changes = [] + for ruleset_id in sorted(ruleset_ids): + endpoint = f"repos/{repository}/rulesets/{ruleset_id}" + original = github_api(endpoint) + if original["source_type"] != "Repository" or original["source"].lower() != repository.lower(): + raise RuntimeError(f"Ruleset {ruleset_id} is not owned by {repository}") + if original["target"] != "branch" or original["enforcement"] != "active": + raise RuntimeError(f"Ruleset {ruleset_id} changed scope or enforcement; rerun the preview") + migrated = migrate_rules(original["rules"]) + if migrated != original["rules"]: + changes.append({"endpoint": endpoint, "original": original, "payload": {"rules": migrated}}) + return changes + + +def verify_readiness(repository: str, verified_sha: str | None) -> str: + """Require merged workflow files and green checks on main or an identified main PR.""" + metadata = github_api(f"repos/{repository}") + if metadata["default_branch"] != "main": + raise RuntimeError("This migration is scoped to repositories whose default branch is main") + main_sha = github_api(f"repos/{repository}/branches/main")["commit"]["sha"] + for path in sorted(set(CHECK_WORKFLOWS.values())): + content = github_api(f"repos/{repository}/contents/{path}?ref={main_sha}") + if not isinstance(content, dict) or content.get("type") != "file": + raise RuntimeError(f"{path} is not a workflow file on current main") + + sha = verified_sha or main_sha + if sha != main_sha: + pages = github_api(f"repos/{repository}/commits/{sha}/pulls?per_page=100", paginate=True) + matching_prs = [ + pr + for page in pages + for pr in page + if pr["state"] == "open" + and pr["base"]["ref"] == "main" + and pr["base"]["repo"]["full_name"].lower() == repository.lower() + and sha in (pr["head"]["sha"], pr.get("merge_commit_sha")) + ] + if not matching_prs: + raise RuntimeError("--verified-sha must be current main or the head/merge SHA of an open PR targeting main") + + pages = github_api(f"repos/{repository}/commits/{sha}/check-runs?filter=latest&per_page=100", paginate=True) + checks = [check for page in pages for check in page["check_runs"]] + runs = {} + for context, workflow in CHECK_WORKFLOWS.items(): + matching = [check for check in checks if check["name"] == context and check["app"]["id"] == ACTIONS_INTEGRATION] + if not matching: + raise RuntimeError(f"Missing GitHub Actions check {context!r} on {sha}") + check = max(matching, key=lambda check: check["id"]) + if check["status"] != "completed" or check["conclusion"] != "success": + raise RuntimeError(f"Check {context!r} on {sha} is not completed and successful") + url = urlparse(check["details_url"]) + match = re.fullmatch(rf"/{re.escape(repository)}/actions/runs/([0-9]+)/job/[0-9]+", url.path, re.IGNORECASE) + if url.hostname != "github.com" or match is None: + raise RuntimeError(f"Cannot identify the workflow run for {context!r}") + run_id = match.group(1) + if run_id not in runs: + runs[run_id] = github_api(f"repos/{repository}/actions/runs/{run_id}") + run = runs[run_id] + if ( + run["path"].split("@", 1)[0] != workflow + or run["check_suite_id"] != check["check_suite"]["id"] + or run["status"] != "completed" + or run["conclusion"] != "success" + ): + raise RuntimeError(f"Check {context!r} does not belong to a successful {workflow} run") + return sha + + +def apply_changes(repository: str, changes: list[dict], verified_sha: str | None) -> str: + """Preflight all changes before issuing narrowly scoped rules updates.""" + sha = verify_readiness(repository, verified_sha) + for change in changes: + if github_api(change["endpoint"]) != change["original"]: + raise RuntimeError(f"{change['endpoint']} changed after discovery; rerun the preview") + for change in changes: + updated = github_api(change["endpoint"], method="PUT", payload=change["payload"]) + original = change["original"] + settings = ("name", "target", "enforcement", "conditions", "bypass_actors") + if updated["rules"] != change["payload"]["rules"] or any( + updated.get(key) != original.get(key) for key in settings + ): + raise RuntimeError(f"Unexpected response after updating {change['endpoint']}; inspect that ruleset") + return sha + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repo", required=True, choices=("NVIDIA/cuda-python", "NVIDIA-dev/cuda-python-private")) + parser.add_argument("--verified-sha", help="Full SHA of current main or a representative open PR head/merge commit") + parser.add_argument( + "--apply", action="store_true", help="Update required checks after explicit cutover authorization" + ) + args = parser.parse_args(argv) + if args.verified_sha and re.fullmatch(r"[0-9a-f]{40}", args.verified_sha) is None: + parser.error("--verified-sha must be a full 40-character lowercase commit SHA") + + try: + changes = discover_changes(args.repo) + preview = { + "repository": args.repo, + "branch": "main", + "mode": "apply" if args.apply else "dry-run", + "changes": [ + { + "endpoint": change["endpoint"], + "before": change["original"], + "put_payload": change["payload"], + } + for change in changes + ], + } + print(json.dumps(preview, indent=2)) + if args.apply and changes: + sha = apply_changes(args.repo, changes, args.verified_sha) + print(f"Updated {len(changes)} ruleset(s); representative checks verified on {sha}", file=sys.stderr) + elif not changes: + print("No matching active pre-commit.ci requirement on main; no updates made.", file=sys.stderr) + else: + print( + "Dry run: no writes made. --apply checks merged workflows and successful checks before writing.", + file=sys.stderr, + ) + except (RuntimeError, subprocess.CalledProcessError, json.JSONDecodeError) as exc: + if isinstance(exc, subprocess.CalledProcessError) and exc.stderr: + print(exc.stderr.strip(), file=sys.stderr) + print(f"error: {exc}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/ci/tools/prepare_lychee_inputs.py b/ci/tools/prepare_lychee_inputs.py new file mode 100644 index 00000000000..d033110d3c3 --- /dev/null +++ b/ci/tools/prepare_lychee_inputs.py @@ -0,0 +1,70 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Write deterministic input lists for authored or rendered documentation links.""" + +from __future__ import annotations + +import argparse +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] + + +def authored_inputs(root: Path) -> list[Path]: + """Select existing tracked Markdown and reStructuredText outside qa/.""" + result = subprocess.run( + ["git", "ls-files", "-z", "--", "*.md", "*.rst"], # noqa: S607 + cwd=root, + check=True, + capture_output=True, + ) + paths = [] + for name in result.stdout.decode("utf-8").split("\0"): + if not name or name.startswith("qa/"): + continue + path = root / name + if path.is_file(): + paths.append(path) + return sorted(paths) + + +def rendered_inputs(docs_root: Path) -> list[Path]: + """Select built HTML, excluding theme assets under _static directories.""" + return sorted( + path + for path in docs_root.rglob("*.html") + if path.is_file() and "_static" not in path.relative_to(docs_root).parts + ) + + +def write_inputs(paths: list[Path], output: Path) -> None: + """Reject an empty or ambiguous file list before invoking lychee.""" + if not paths: + raise ValueError("No documentation inputs found for lychee") + lines = [str(path.absolute()) for path in paths] + if any("\n" in line or "\r" in line for line in lines): + raise ValueError("Lychee input paths must not contain newlines") + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text("\n".join(lines) + "\n", encoding="utf-8") + print(f"Wrote {len(lines)} lychee inputs to {output}") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("kind", choices=("authored", "rendered")) + parser.add_argument("--repo-root", type=Path, default=ROOT) + parser.add_argument("--docs-root", type=Path, default=Path("artifacts/docs")) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + root = args.repo_root.resolve() + if args.kind == "authored": + paths = authored_inputs(root) + else: + paths = rendered_inputs((root / args.docs_root).absolute()) + write_inputs(paths, args.output) + + +if __name__ == "__main__": + main() diff --git a/ci/tools/tests/test_migrate_precommit_checks.py b/ci/tools/tests/test_migrate_precommit_checks.py new file mode 100644 index 00000000000..816e299c09b --- /dev/null +++ b/ci/tools/tests/test_migrate_precommit_checks.py @@ -0,0 +1,309 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import copy +import json + +import pytest + +from ci.tools import migrate_precommit_checks as migration + +REPOSITORY = "NVIDIA/cuda-python" +MAIN_SHA = "a" * 40 +PR_SHA = "b" * 40 +RULESET_ENDPOINT = f"repos/{REPOSITORY}/rulesets/12" + + +@pytest.fixture +def ruleset(): + return { + "id": 12, + "name": "Main protection", + "source_type": "Repository", + "source": REPOSITORY, + "target": "branch", + "enforcement": "active", + "conditions": {"ref_name": {"include": ["~DEFAULT_BRANCH"], "exclude": []}}, + "bypass_actors": [{"actor_id": 9, "actor_type": "Team", "bypass_mode": "pull_request"}], + "rules": [ + {"type": "deletion"}, + { + "type": "required_status_checks", + "parameters": { + "strict_required_status_checks_policy": True, + "do_not_enforce_on_create": True, + "required_status_checks": [ + {"context": "CI", "integration_id": 15368}, + {"context": "pre-commit.ci - pr", "integration_id": 68672}, + {"context": "Other service", "integration_id": 777}, + ], + }, + }, + {"type": "pull_request", "parameters": {"required_approving_review_count": 2}}, + ], + } + + +@pytest.fixture +def github(monkeypatch, ruleset): + effective_rule = { + **copy.deepcopy(ruleset["rules"][1]), + "ruleset_id": ruleset["id"], + "ruleset_source_type": "Repository", + "ruleset_source": REPOSITORY, + } + checks = [ + { + "id": index, + "name": context, + "app": {"id": 15368}, + "status": "completed", + "conclusion": "success", + "details_url": f"https://github.com/{REPOSITORY}/actions/runs/{index}/job/100", + "check_suite": {"id": index + 100}, + } + for index, context in enumerate(migration.CHECK_WORKFLOWS, start=1) + ] + responses = { + f"repos/{REPOSITORY}/rules/branches/main?per_page=100": [[effective_rule]], + RULESET_ENDPOINT: copy.deepcopy(ruleset), + f"repos/{REPOSITORY}": {"default_branch": "main"}, + f"repos/{REPOSITORY}/branches/main": {"commit": {"sha": MAIN_SHA}}, + f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100": [{"check_runs": checks}], + f"repos/{REPOSITORY}/commits/{PR_SHA}/check-runs?filter=latest&per_page=100": [{"check_runs": checks}], + f"repos/{REPOSITORY}/commits/{PR_SHA}/pulls?per_page=100": [ + [ + { + "state": "open", + "base": {"ref": "main", "repo": {"full_name": REPOSITORY}}, + "head": {"sha": PR_SHA}, + "merge_commit_sha": "c" * 40, + } + ] + ], + } + for path in set(migration.CHECK_WORKFLOWS.values()): + responses[f"repos/{REPOSITORY}/contents/{path}?ref={MAIN_SHA}"] = {"path": path, "type": "file"} + for check in checks: + responses[f"repos/{REPOSITORY}/actions/runs/{check['id']}"] = { + "path": migration.CHECK_WORKFLOWS[check["name"]], + "check_suite_id": check["check_suite"]["id"], + "status": "completed", + "conclusion": "success", + } + calls = [] + + def api(endpoint, *, method="GET", payload=None, paginate=False): + calls.append((method, endpoint, copy.deepcopy(payload))) + if method == "PUT": + responses[endpoint]["rules"] = copy.deepcopy(payload["rules"]) + return copy.deepcopy(responses[endpoint]) + + monkeypatch.setattr(migration, "github_api", api) + return responses, calls + + +@pytest.mark.agent_authored(model="gpt-6") +def test_replaces_service_without_changing_unrelated_rules_or_parameters(ruleset): + original = copy.deepcopy(ruleset) + migrated = migration.migrate_rules(ruleset["rules"]) + + assert ruleset == original + assert migrated[0] == original["rules"][0] + assert migrated[2] == original["rules"][2] + parameters = migrated[1]["parameters"] + assert parameters["strict_required_status_checks_policy"] is True + assert parameters["do_not_enforce_on_create"] is True + checks = parameters["required_status_checks"] + assert checks == [ + {"context": "CI", "integration_id": 15368}, + {"context": "Pre-commit (Linux)", "integration_id": 15368}, + {"context": "Pre-commit (Windows)", "integration_id": 15368}, + {"context": "Documentation links", "integration_id": 15368}, + {"context": "Other service", "integration_id": 777}, + ] + assert migration.migrate_rules(migrated) == migrated + + +@pytest.mark.agent_authored(model="gpt-6") +def test_existing_actions_requirements_are_not_duplicated(ruleset): + checks = ruleset["rules"][1]["parameters"]["required_status_checks"] + existing = {"context": "Documentation links", "integration_id": 15368} + checks.append(existing) + + migrated = migration.migrate_rules(ruleset["rules"]) + + assert migrated[1]["parameters"]["required_status_checks"].count(existing) == 1 + + +@pytest.mark.parametrize("integration", [None, 777]) +@pytest.mark.agent_authored(model="gpt-6") +def test_refuses_unexpected_old_integration(ruleset, integration): + ruleset["rules"][1]["parameters"]["required_status_checks"][1]["integration_id"] = integration + + with pytest.raises(RuntimeError, match="expected integration"): + migration.migrate_rules(ruleset["rules"]) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_refuses_conflicting_replacement_integration(ruleset): + ruleset["rules"][1]["parameters"]["required_status_checks"].append( + {"context": "Documentation links", "integration_id": 777} + ) + + with pytest.raises(RuntimeError, match="different or unspecified integration"): + migration.migrate_rules(ruleset["rules"]) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_duplicate_valid_context_cannot_hide_conflicting_integration(ruleset): + ruleset["rules"][1]["parameters"]["required_status_checks"].extend( + [ + {"context": "Documentation links", "integration_id": 777}, + {"context": "Documentation links", "integration_id": 15368}, + ] + ) + + with pytest.raises(RuntimeError, match="different or unspecified integration"): + migration.migrate_rules(ruleset["rules"]) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_cli_defaults_to_read_only_preview(github, capsys): + responses, calls = github + + assert migration.main(["--repo", REPOSITORY]) == 0 + + preview = json.loads(capsys.readouterr().out) + assert preview["mode"] == "dry-run" + assert preview["changes"][0]["before"] == responses[RULESET_ENDPOINT] + assert set(preview["changes"][0]["put_payload"]) == {"rules"} + assert all(method == "GET" for method, endpoint, payload in calls) + assert not any("/contents/" in endpoint for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_inherited_requirement_is_rejected_before_any_write(github): + responses, calls = github + rule = responses[f"repos/{REPOSITORY}/rules/branches/main?per_page=100"][0][0] + rule["ruleset_source_type"] = "Organization" + rule["ruleset_source"] = "NVIDIA" + + with pytest.raises(RuntimeError, match="inherited from Organization NVIDIA ruleset 12"): + migration.discover_changes(REPOSITORY) + + assert len(calls) == 1 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_replacement_workflows_must_be_files_on_current_main(github): + responses, calls = github + responses[f"repos/{REPOSITORY}/contents/.github/workflows/lychee.yml?ref={MAIN_SHA}"] = [] + + with pytest.raises(RuntimeError, match="not a workflow file on current main"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.parametrize("verified_sha", [None, PR_SHA]) +@pytest.mark.agent_authored(model="gpt-6") +def test_apply_requires_green_representative_checks_and_preserves_settings(github, ruleset, verified_sha): + responses, calls = github + changes = migration.discover_changes(REPOSITORY) + + assert migration.apply_changes(REPOSITORY, changes, verified_sha) == (verified_sha or MAIN_SHA) + + writes = [call for call in calls if call[0] == "PUT"] + assert writes == [("PUT", RULESET_ENDPOINT, {"rules": migration.migrate_rules(ruleset["rules"])})] + assert {key: value for key, value in responses[RULESET_ENDPOINT].items() if key != "rules"} == { + key: value for key, value in ruleset.items() if key != "rules" + } + + +@pytest.mark.parametrize("conclusion", ["failure", "cancelled", "skipped", "neutral", None]) +@pytest.mark.agent_authored(model="gpt-6") +def test_failed_or_skipped_checks_prevent_all_writes(github, conclusion): + responses, calls = github + checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] + checks[-1]["conclusion"] = conclusion + changes = migration.discover_changes(REPOSITORY) + + with pytest.raises(RuntimeError, match="not completed and successful"): + migration.apply_changes(REPOSITORY, changes, None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_newer_pending_run_cannot_be_masked_by_older_success(github): + responses, calls = github + checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] + checks.append({**checks[0], "id": 99, "status": "in_progress", "conclusion": None}) + + with pytest.raises(RuntimeError, match="not completed and successful"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_other_apps_cannot_satisfy_replacement_checks(github): + responses, calls = github + checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] + checks[0]["app"]["id"] = 777 + + with pytest.raises(RuntimeError, match="Missing GitHub Actions check"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.parametrize( + ("field", "value"), + [("path", ".github/workflows/unrelated.yml"), ("check_suite_id", 777), ("conclusion", "failure")], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_check_must_belong_to_the_expected_successful_workflow(github, field, value): + responses, calls = github + responses[f"repos/{REPOSITORY}/actions/runs/1"][field] = value + + with pytest.raises(RuntimeError, match="does not belong to a successful"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_unassociated_sha_is_rejected(github): + responses, calls = github + responses[f"repos/{REPOSITORY}/commits/{PR_SHA}/pulls?per_page=100"] = [[]] + + with pytest.raises(RuntimeError, match="open PR targeting main"): + migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), PR_SHA) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_ruleset_edit_during_preflight_prevents_writes(github): + responses, calls = github + changes = migration.discover_changes(REPOSITORY) + responses[RULESET_ENDPOINT]["conditions"]["ref_name"]["exclude"].append("refs/heads/release") + + with pytest.raises(RuntimeError, match="changed after discovery"): + migration.apply_changes(REPOSITORY, changes, None) + + assert all(method == "GET" for method, endpoint, payload in calls) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_sha_must_be_explicit_full_commit(github): + responses, calls = github + + with pytest.raises(SystemExit, match="2"): + migration.main(["--repo", REPOSITORY, "--verified-sha", "main", "--apply"]) + + assert not calls diff --git a/ci/tools/tests/test_prepare_lychee_inputs.py b/ci/tools/tests/test_prepare_lychee_inputs.py new file mode 100644 index 00000000000..775c3a5d01e --- /dev/null +++ b/ci/tools/tests/test_prepare_lychee_inputs.py @@ -0,0 +1,70 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import subprocess +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).parent.parent)) +from prepare_lychee_inputs import authored_inputs, rendered_inputs, write_inputs + + +@pytest.mark.agent_authored(model="gpt-6") +def test_authored_inputs_use_tracked_documents_and_exclude_qa(tmp_path): + subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True) # noqa: S603, S607 + tracked = ["README.md", "docs/a guide.rst", "qa/README.md", "src/worker.py", "gone.md"] + for name in tracked: + path = tmp_path / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("content", encoding="utf-8") + subprocess.run(["git", "add", "--", *tracked], cwd=tmp_path, check=True) # noqa: S603, S607 + (tmp_path / "gone.md").unlink() + (tmp_path / "untracked.md").write_text("content", encoding="utf-8") + + assert authored_inputs(tmp_path) == [tmp_path / "README.md", tmp_path / "docs/a guide.rst"] + + +@pytest.mark.agent_authored(model="gpt-6") +def test_rendered_inputs_include_all_components_and_exclude_static_assets(tmp_path): + expected = ["cuda-bindings/latest/api.html", "cuda-core/latest/guide.html", "latest/index.html"] + for name in [*expected, "cuda-core/latest/_static/theme.html", "_static/vendor/embed.html", "versions.json"]: + path = tmp_path / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("content", encoding="utf-8") + (tmp_path / "directory.html").mkdir() + + assert rendered_inputs(tmp_path) == [tmp_path / name for name in expected] + + +@pytest.mark.agent_authored(model="gpt-6") +def test_write_inputs_keeps_spaces_and_writes_absolute_paths(tmp_path): + output = tmp_path / "lists/files.txt" + paths = [tmp_path / "docs/a guide.rst", tmp_path / "README.md"] + + write_inputs(paths, output) + + assert output.read_text(encoding="utf-8") == "".join(f"{path}\n" for path in paths) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_write_inputs_rejects_empty_inputs_without_publishing_list(tmp_path): + output = tmp_path / "files.txt" + + with pytest.raises(ValueError, match="No documentation inputs found"): + write_inputs([], output) + + assert not output.exists() + + +@pytest.mark.agent_authored(model="gpt-6") +def test_write_inputs_rejects_paths_that_would_split_into_multiple_inputs(tmp_path): + output = tmp_path / "files.txt" + + with pytest.raises(ValueError, match="must not contain newlines"): + write_inputs([tmp_path / "two\ninputs.md"], output) + + assert not output.exists() diff --git a/cuda_core/pixi.toml b/cuda_core/pixi.toml index cbeaf696607..02155197909 100644 --- a/cuda_core/pixi.toml +++ b/cuda_core/pixi.toml @@ -257,6 +257,10 @@ default-environment = "docs" cmd = ["$PIXI_PROJECT_ROOT/docs/build_docs.sh", "latest-only"] default-environment = "docs" +[target.linux.tasks.docs-build-all-latest] +cmd = ["bash", "$PIXI_PROJECT_ROOT/../ci/tools/build_docs_for_link_check.sh"] +default-environment = "docs" + [target.linux.tasks.docs-debug] cmd = ["$PIXI_PROJECT_ROOT/docs/build_docs.sh", "latest-only"] env = { SPHINXOPTS = "-v -j 1 -d build/.doctrees" } diff --git a/lychee.toml b/lychee.toml index d72bad9002f..cb2fd78f185 100644 --- a/lychee.toml +++ b/lychee.toml @@ -2,7 +2,7 @@ # SPDX-License-Identifier: Apache-2.0 # # Configuration for lychee, the doc link-checker invoked from -# .github/workflows/build-docs.yml. See https://lychee.cli.rs/usage/config/. +# .github/workflows/lychee.yml. See https://lychee.cli.rs/usage/config/. exclude = [ # PR-preview canonical URLs are checked by the preview deployment workflow. From c75e1a0f65496c1ca1c4c8976b0edeb3db25d83b Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:38:42 -0700 Subject: [PATCH 02/12] ci: skip symlinked link inputs and finish workflow bootstrap --- .github/workflows/lychee.yml | 3 --- ci/tools/prepare_lychee_inputs.py | 4 ++-- ci/tools/tests/test_prepare_lychee_inputs.py | 13 +++++++++++++ 3 files changed, 15 insertions(+), 5 deletions(-) diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index 4a622aaf591..f47805eb29f 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -19,9 +19,6 @@ on: description: "Check links afresh and publish a new cache baseline" type: boolean default: false - push: - # Temporary bootstrap trigger; remove after manual dispatch is available. - branches: ["rwgk/maint/lychee-pr-nightly"] permissions: contents: read diff --git a/ci/tools/prepare_lychee_inputs.py b/ci/tools/prepare_lychee_inputs.py index d033110d3c3..9c723398016 100644 --- a/ci/tools/prepare_lychee_inputs.py +++ b/ci/tools/prepare_lychee_inputs.py @@ -13,7 +13,7 @@ def authored_inputs(root: Path) -> list[Path]: - """Select existing tracked Markdown and reStructuredText outside qa/.""" + """Select tracked Markdown and reStructuredText outside qa/, skipping symlinks.""" result = subprocess.run( ["git", "ls-files", "-z", "--", "*.md", "*.rst"], # noqa: S607 cwd=root, @@ -25,7 +25,7 @@ def authored_inputs(root: Path) -> list[Path]: if not name or name.startswith("qa/"): continue path = root / name - if path.is_file(): + if path.is_file() and not path.is_symlink(): paths.append(path) return sorted(paths) diff --git a/ci/tools/tests/test_prepare_lychee_inputs.py b/ci/tools/tests/test_prepare_lychee_inputs.py index 775c3a5d01e..15bfb945dc2 100644 --- a/ci/tools/tests/test_prepare_lychee_inputs.py +++ b/ci/tools/tests/test_prepare_lychee_inputs.py @@ -28,6 +28,19 @@ def test_authored_inputs_use_tracked_documents_and_exclude_qa(tmp_path): assert authored_inputs(tmp_path) == [tmp_path / "README.md", tmp_path / "docs/a guide.rst"] +@pytest.mark.agent_authored(model="gpt-6") +def test_authored_inputs_skip_tracked_symlinks_and_keep_their_real_target(tmp_path): + subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True) # noqa: S603, S607 + target = tmp_path / "README.md" + target.write_text("[Guide](docs/guide.rst)\n", encoding="utf-8") + package = tmp_path / "cuda_python" + package.mkdir() + (package / "README.md").symlink_to("../README.md") + subprocess.run(["git", "add", "--", "README.md", "cuda_python/README.md"], cwd=tmp_path, check=True) # noqa: S607 + + assert authored_inputs(tmp_path) == [target] + + @pytest.mark.agent_authored(model="gpt-6") def test_rendered_inputs_include_all_components_and_exclude_static_assets(tmp_path): expected = ["cuda-bindings/latest/api.html", "cuda-core/latest/guide.html", "latest/index.html"] From 86fa4013f090c2f7d7ef8acd31bb3b80be0deaf8 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:40:32 -0700 Subject: [PATCH 03/12] ci: avoid historical blob downloads in link-check checkout --- .github/workflows/lychee.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index f47805eb29f..cc9ceeb9e3f 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -50,6 +50,8 @@ jobs: with: ref: ${{ github.event.pull_request.head.sha || github.sha }} fetch-depth: 0 + # Keep SCM history/tags without downloading historical source blobs. + filter: blob:none persist-credentials: false - name: Set up Python From 03cc8277552e0fc7070328ec4106fbeb1feff86b Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:41:00 -0700 Subject: [PATCH 04/12] ci: propagate checked-revision lookup failures --- .github/workflows/lychee.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index cc9ceeb9e3f..0410592a119 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -66,7 +66,8 @@ jobs: - name: Build all documentation from sources if: ${{ matrix.kind == 'rendered' }} run: | - export CUDA_PYTHON_DOCS_GITHUB_REF="$(git rev-parse HEAD)" + CUDA_PYTHON_DOCS_GITHUB_REF=$(git rev-parse HEAD) + export CUDA_PYTHON_DOCS_GITHUB_REF pixi run --manifest-path cuda_core/pixi.toml -e docs docs-build-all-latest - name: Prepare link-check inputs and cache policy From 12b067f8b844decd5959df50f2a4577228d82665 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Thu, 1 Oct 2026 16:47:14 -0700 Subject: [PATCH 05/12] ci: allow nightly link checks to be tested without GPU jobs --- .github/workflows/ci-nightly.yml | 22 ++++++++++++++++++++-- .github/workflows/lychee.yml | 6 +++++- ci/README-pre-commit-migration.md | 7 +++++++ 3 files changed, 32 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci-nightly.yml b/.github/workflows/ci-nightly.yml index 3b75bdd2126..d033ba1d251 100644 --- a/.github/workflows/ci-nightly.yml +++ b/.github/workflows/ci-nightly.yml @@ -13,7 +13,7 @@ name: "CI: Nightly optional-deps" concurrency: - group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name }} + group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name }}-${{ inputs.documentation-links-only || false }} cancel-in-progress: true on: @@ -24,6 +24,10 @@ on: - cron: "17 2 * * *" workflow_dispatch: inputs: + documentation-links-only: + description: "Test the nightly link checker without wheel or GPU jobs" + type: boolean + default: false run-id: description: > Override the CI run ID to download artifacts from. @@ -39,6 +43,7 @@ jobs: contents: read uses: ./.github/workflows/lychee.yml with: + concurrency-suffix: ${{ inputs.documentation-links-only && 'links-only' || 'full' }} refresh-cache: true test-ci-tools-for-release: @@ -58,7 +63,7 @@ jobs: python -m pytest -v --noconftest ci/tools/tests find-wheels: - if: ${{ github.repository_owner == 'nvidia' }} + if: ${{ github.repository_owner == 'nvidia' && !inputs.documentation-links-only }} runs-on: ubuntu-latest outputs: RUN_ID: ${{ steps.find.outputs.run_id }} @@ -338,7 +343,20 @@ jobs: - test-standard-linux-aarch64 steps: - name: Exit + env: + LINKS_ONLY: ${{ inputs.documentation-links-only || false }} + NEEDS_JSON: ${{ toJSON(needs) }} run: | + if [[ "${LINKS_ONLY}" == "true" ]]; then + # GPU jobs depend on find-wheels and must stay skipped in this mode. + jq -e 'all(to_entries[]; + if .key == "documentation-links" or .key == "test-ci-tools-for-release" + then .value.result == "success" + else .value.result == "skipped" + end)' <<< "${NEEDS_JSON}" + exit 0 + fi + # If any dependency was cancelled or failed, that's a failure. # # See ci.yml for the full rationale on why we must use always() diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index 0410592a119..edc2b11abd3 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -9,6 +9,10 @@ on: types: [opened, reopened, synchronize] workflow_call: inputs: + concurrency-suffix: + description: "Keep a caller's test mode separate from its full run" + type: string + default: full refresh-cache: description: "Check links afresh and publish a new cache baseline" type: boolean @@ -24,7 +28,7 @@ permissions: contents: read concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }}-${{ inputs.refresh-cache || false }} + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }}-${{ inputs.refresh-cache || false }}-${{ inputs.concurrency-suffix || 'full' }} cancel-in-progress: true defaults: diff --git a/ci/README-pre-commit-migration.md b/ci/README-pre-commit-migration.md index 1cc8dcad67d..87527d19813 100644 --- a/ci/README-pre-commit-migration.md +++ b/ci/README-pre-commit-migration.md @@ -21,6 +21,8 @@ gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ -f refresh-cache=true gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ -f refresh-cache=false +gh workflow run ci-nightly.yml --repo NVIDIA/cuda-python --ref REF \ + -f documentation-links-only=true gh run list --repo NVIDIA/cuda-python --branch REF ``` @@ -33,6 +35,11 @@ publish the baseline that all PRs can read, including fork PRs. Separate authored/rendered namespaces include the checker version and checking policy; documentation edits do not invalidate previously successful external checks. +The nightly documentation-only mode exercises the reusable-workflow call and +status gate, plus standalone CI-tool tests. It skips wheel lookup and requires +all wheel/GPU jobs to remain skipped. The scheduled/default nightly mode still +runs the complete existing suite. + For the same source-based docs build locally: ```sh From 375be61d8bc7256381c8fd41e0dc700f3a33460b Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Fri, 2 Oct 2026 10:36:02 -0700 Subject: [PATCH 06/12] ci: separate required-check migration from link checking --- CONTRIBUTING.md | 18 +- ci/README-pre-commit-migration.md | 60 +--- ci/tools/migrate_precommit_checks.py | 230 ------------- .../tests/test_migrate_precommit_checks.py | 309 ------------------ 4 files changed, 12 insertions(+), 605 deletions(-) delete mode 100644 ci/tools/migrate_precommit_checks.py delete mode 100644 ci/tools/tests/test_migrate_precommit_checks.py diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index b1fe8519f81..32830b4bb47 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -221,14 +221,16 @@ and freshly rendered HTML, including link fragments. It runs on PR updates and from nightly CI; copied `pull-request/*` branches do not repeat these checks. Nightly link checks start with a fresh cache and publish successful checks for -PRs to reuse for up to one day. Quarterly pre-commit hook updates arrive as -draft PRs for review. Lychee version updates are maintained separately so its -local hook and CI binary stay aligned. The local lychee hook keeps its current -behavior until a stable release supports an explicit cache location. - -See [the pre-commit migration procedure](ci/README-pre-commit-migration.md) for -manual workflow testing and the required-check cutover. The existing -pre-commit.ci service remains required until that cutover is complete. +PRs to reuse for up to one day. Dependabot checks pre-commit hook revisions +monthly and opens update PRs with the `CI/CD` and `dependencies` labels, without +an automatic assignee or milestone. Lychee version updates are maintained +separately so its local hook and CI binary stay aligned. The local lychee hook +keeps its current behavior until a stable release supports an explicit cache +location. + +See [the pre-commit workflow guide](ci/README-pre-commit-migration.md) for manual +workflow testing. The existing pre-commit.ci service remains required until the +required-check cutover is complete. To set yourself up for running pre-commit checks locally and to catch issues before pushing your changes, follow these steps: diff --git a/ci/README-pre-commit-migration.md b/ci/README-pre-commit-migration.md index 87527d19813..a7a39c4e81c 100644 --- a/ci/README-pre-commit-migration.md +++ b/ci/README-pre-commit-migration.md @@ -15,8 +15,6 @@ run hosted checks without copying a PR into the heavyweight CUDA CI: ```sh gh workflow run pre-commit.yml --repo NVIDIA/cuda-python --ref REF -gh workflow run pre-commit-autoupdate.yml --repo NVIDIA/cuda-python --ref REF \ - -f dry-run=true gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ -f refresh-cache=true gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ @@ -26,7 +24,8 @@ gh workflow run ci-nightly.yml --repo NVIDIA/cuda-python --ref REF \ gh run list --repo NVIDIA/cuda-python --branch REF ``` -The updater dry run shows its diff without opening or updating a PR. Lychee's +Dependabot checks hook revisions monthly and opens update PRs with the `CI/CD` +and `dependencies` labels, without an automatic assignee or milestone. Lychee's refresh mode skips restoration and publishes a new snapshot even when some links fail; only successful checks enter the persisted cache. Its normal mode restores the newest matching snapshot and does not publish one. Branch tests @@ -50,58 +49,3 @@ PIXI_LOCKED=true pixi run --manifest-path cuda_core/pixi.toml -e docs \ The build verifies all three sibling libraries import from the checkout, installs metapackage metadata without dependencies, and assembles all four latest documentation trees under `artifacts/docs`. - -## Preview the ruleset change - -Run from the repository root with an authenticated GitHub CLI: - -```sh -pixi exec --spec python python ci/tools/migrate_precommit_checks.py \ - --repo NVIDIA/cuda-python > /tmp/cuda-python-precommit-cutover.json -``` - -The default mode only reads GitHub. It discovers active rules applying to -`main`, fetches complete repository rulesets, and prints their original -configuration and the exact proposed PUT payload. Review that JSON before -cutover. The payload changes only `rules`: unrelated rules and required checks, -strictness, branch conditions, bypass actors, and enforcement remain in place. -The tool rejects inherited organization rulesets and unexpected integration IDs; -an organization ruleset's owner must handle that migration separately. - -## Cut over after merge - -1. Merge `.github/workflows/pre-commit.yml` and `.github/workflows/lychee.yml`. - Check that they exist on the current default `main` branch. -2. Obtain successful, completed runs of all three replacement checks on current - `main` or a representative open PR targeting `main`. Record the full checked - commit SHA. This also verifies the PR event path when using a PR; a branch - push run alone does not validate PR-specific behavior. -3. After explicit authorization for the ruleset cutover, run the preview again, - then run the same command with `--apply --verified-sha FULL_COMMIT_SHA`. - Omitting `--verified-sha` requires checks on the current `main` commit. -4. Confirm the active rules now require all three replacement contexts from the - GitHub Actions App and still contain every unrelated requirement. Verify on - a representative PR that an intentional formatting or broken-link failure - is reported as a required failing check and blocks merging; repair it and - verify the same check passes. Check this independently of the PR's draft - status, which also prevents merging. -5. Only after that verification, remove this repository from the pre-commit.ci - GitHub App installation's repository access, or disable its service through - the repository/service settings. Preserve access for other repositories - using the same installation. The migration script never changes App access. - -`--apply` verifies the workflow files on `main`, the selected commit's checks, -their App IDs, and their successful owning workflow runs before writing. It -re-reads all affected rulesets and refuses changes made since discovery. -GitHub does not provide an atomic transaction across ruleset updates: if a -write fails partway through, inspect the emitted preview and actual rulesets -before retrying. Do not remove the service while any old required check remains. - -## Validate the helper - -```sh -pixi exec --spec python --spec pytest python -m pytest \ - --noconftest ci/tools/tests/test_migrate_precommit_checks.py -``` - -The tests mock GitHub; they cannot change repository settings. diff --git a/ci/tools/migrate_precommit_checks.py b/ci/tools/migrate_precommit_checks.py deleted file mode 100644 index f4b9d720eb3..00000000000 --- a/ci/tools/migrate_precommit_checks.py +++ /dev/null @@ -1,230 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 - -"""Preview the main ruleset cutover from pre-commit.ci to GitHub Actions. - -Writes require --apply, merged workflows, and successful representative checks. -This tool does not uninstall the pre-commit.ci GitHub App. -""" - -from __future__ import annotations - -import argparse -import copy -import json -import re -import subprocess -import sys -from urllib.parse import urlparse - -OLD_CONTEXT = "pre-commit.ci - pr" -OLD_INTEGRATION = 68672 -ACTIONS_INTEGRATION = 15368 -CHECK_WORKFLOWS = { - "Pre-commit (Linux)": ".github/workflows/pre-commit.yml", - "Pre-commit (Windows)": ".github/workflows/pre-commit.yml", - "Documentation links": ".github/workflows/lychee.yml", -} - - -def github_api(endpoint: str, *, method: str = "GET", payload: dict | None = None, paginate: bool = False): - """Call gh with structured JSON input, without invoking a shell.""" - command = ["gh", "api", "--method", method, endpoint, "-H", "Accept: application/vnd.github+json"] - if paginate: - command.extend(["--paginate", "--slurp"]) - if payload is not None: - command.extend(["--input", "-"]) - result = subprocess.run( # noqa: S603 - arguments are passed directly, without a shell. - command, - input=json.dumps(payload) if payload is not None else None, - capture_output=True, - text=True, - check=True, - ) - return json.loads(result.stdout) - - -def migrate_rules(rules: list[dict]) -> list[dict]: - """Replace only the known pre-commit.ci requirement, preserving all other rules.""" - migrated = copy.deepcopy(rules) - for rule in migrated: - if rule["type"] != "required_status_checks": - continue - checks = rule["parameters"]["required_status_checks"] - old_checks = [check for check in checks if check["context"] == OLD_CONTEXT] - if not old_checks: - continue - if any(check.get("integration_id") != OLD_INTEGRATION for check in old_checks): - raise RuntimeError(f"{OLD_CONTEXT!r} is not bound to expected integration {OLD_INTEGRATION}") - existing = set() - for check in checks: - context = check["context"] - if context in CHECK_WORKFLOWS: - if check.get("integration_id") != ACTIONS_INTEGRATION: - raise RuntimeError(f"{context!r} already exists with a different or unspecified integration") - existing.add(context) - replacement = [ - {"context": context, "integration_id": ACTIONS_INTEGRATION} - for context in CHECK_WORKFLOWS - if context not in existing - ] - updated = [] - for check in checks: - if check["context"] == OLD_CONTEXT: - updated.extend(replacement) - replacement = [] - else: - updated.append(check) - rule["parameters"]["required_status_checks"] = updated - return migrated - - -def discover_changes(repository: str) -> list[dict]: - """Find active requirements on main and fetch their complete repository rulesets.""" - pages = github_api(f"repos/{repository}/rules/branches/main?per_page=100", paginate=True) - ruleset_ids = set() - for rule in (rule for page in pages for rule in page): - if rule["type"] != "required_status_checks": - continue - checks = rule["parameters"]["required_status_checks"] - if not any(check["context"] == OLD_CONTEXT for check in checks): - continue - if rule["ruleset_source_type"] != "Repository" or rule["ruleset_source"].lower() != repository.lower(): - raise RuntimeError( - f"Requirement is inherited from {rule['ruleset_source_type']} {rule['ruleset_source']} " - f"ruleset {rule['ruleset_id']}; its owner must migrate it. No repository changes were made." - ) - ruleset_ids.add(rule["ruleset_id"]) - - changes = [] - for ruleset_id in sorted(ruleset_ids): - endpoint = f"repos/{repository}/rulesets/{ruleset_id}" - original = github_api(endpoint) - if original["source_type"] != "Repository" or original["source"].lower() != repository.lower(): - raise RuntimeError(f"Ruleset {ruleset_id} is not owned by {repository}") - if original["target"] != "branch" or original["enforcement"] != "active": - raise RuntimeError(f"Ruleset {ruleset_id} changed scope or enforcement; rerun the preview") - migrated = migrate_rules(original["rules"]) - if migrated != original["rules"]: - changes.append({"endpoint": endpoint, "original": original, "payload": {"rules": migrated}}) - return changes - - -def verify_readiness(repository: str, verified_sha: str | None) -> str: - """Require merged workflow files and green checks on main or an identified main PR.""" - metadata = github_api(f"repos/{repository}") - if metadata["default_branch"] != "main": - raise RuntimeError("This migration is scoped to repositories whose default branch is main") - main_sha = github_api(f"repos/{repository}/branches/main")["commit"]["sha"] - for path in sorted(set(CHECK_WORKFLOWS.values())): - content = github_api(f"repos/{repository}/contents/{path}?ref={main_sha}") - if not isinstance(content, dict) or content.get("type") != "file": - raise RuntimeError(f"{path} is not a workflow file on current main") - - sha = verified_sha or main_sha - if sha != main_sha: - pages = github_api(f"repos/{repository}/commits/{sha}/pulls?per_page=100", paginate=True) - matching_prs = [ - pr - for page in pages - for pr in page - if pr["state"] == "open" - and pr["base"]["ref"] == "main" - and pr["base"]["repo"]["full_name"].lower() == repository.lower() - and sha in (pr["head"]["sha"], pr.get("merge_commit_sha")) - ] - if not matching_prs: - raise RuntimeError("--verified-sha must be current main or the head/merge SHA of an open PR targeting main") - - pages = github_api(f"repos/{repository}/commits/{sha}/check-runs?filter=latest&per_page=100", paginate=True) - checks = [check for page in pages for check in page["check_runs"]] - runs = {} - for context, workflow in CHECK_WORKFLOWS.items(): - matching = [check for check in checks if check["name"] == context and check["app"]["id"] == ACTIONS_INTEGRATION] - if not matching: - raise RuntimeError(f"Missing GitHub Actions check {context!r} on {sha}") - check = max(matching, key=lambda check: check["id"]) - if check["status"] != "completed" or check["conclusion"] != "success": - raise RuntimeError(f"Check {context!r} on {sha} is not completed and successful") - url = urlparse(check["details_url"]) - match = re.fullmatch(rf"/{re.escape(repository)}/actions/runs/([0-9]+)/job/[0-9]+", url.path, re.IGNORECASE) - if url.hostname != "github.com" or match is None: - raise RuntimeError(f"Cannot identify the workflow run for {context!r}") - run_id = match.group(1) - if run_id not in runs: - runs[run_id] = github_api(f"repos/{repository}/actions/runs/{run_id}") - run = runs[run_id] - if ( - run["path"].split("@", 1)[0] != workflow - or run["check_suite_id"] != check["check_suite"]["id"] - or run["status"] != "completed" - or run["conclusion"] != "success" - ): - raise RuntimeError(f"Check {context!r} does not belong to a successful {workflow} run") - return sha - - -def apply_changes(repository: str, changes: list[dict], verified_sha: str | None) -> str: - """Preflight all changes before issuing narrowly scoped rules updates.""" - sha = verify_readiness(repository, verified_sha) - for change in changes: - if github_api(change["endpoint"]) != change["original"]: - raise RuntimeError(f"{change['endpoint']} changed after discovery; rerun the preview") - for change in changes: - updated = github_api(change["endpoint"], method="PUT", payload=change["payload"]) - original = change["original"] - settings = ("name", "target", "enforcement", "conditions", "bypass_actors") - if updated["rules"] != change["payload"]["rules"] or any( - updated.get(key) != original.get(key) for key in settings - ): - raise RuntimeError(f"Unexpected response after updating {change['endpoint']}; inspect that ruleset") - return sha - - -def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--repo", required=True, choices=("NVIDIA/cuda-python", "NVIDIA-dev/cuda-python-private")) - parser.add_argument("--verified-sha", help="Full SHA of current main or a representative open PR head/merge commit") - parser.add_argument( - "--apply", action="store_true", help="Update required checks after explicit cutover authorization" - ) - args = parser.parse_args(argv) - if args.verified_sha and re.fullmatch(r"[0-9a-f]{40}", args.verified_sha) is None: - parser.error("--verified-sha must be a full 40-character lowercase commit SHA") - - try: - changes = discover_changes(args.repo) - preview = { - "repository": args.repo, - "branch": "main", - "mode": "apply" if args.apply else "dry-run", - "changes": [ - { - "endpoint": change["endpoint"], - "before": change["original"], - "put_payload": change["payload"], - } - for change in changes - ], - } - print(json.dumps(preview, indent=2)) - if args.apply and changes: - sha = apply_changes(args.repo, changes, args.verified_sha) - print(f"Updated {len(changes)} ruleset(s); representative checks verified on {sha}", file=sys.stderr) - elif not changes: - print("No matching active pre-commit.ci requirement on main; no updates made.", file=sys.stderr) - else: - print( - "Dry run: no writes made. --apply checks merged workflows and successful checks before writing.", - file=sys.stderr, - ) - except (RuntimeError, subprocess.CalledProcessError, json.JSONDecodeError) as exc: - if isinstance(exc, subprocess.CalledProcessError) and exc.stderr: - print(exc.stderr.strip(), file=sys.stderr) - print(f"error: {exc}", file=sys.stderr) - return 2 - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/ci/tools/tests/test_migrate_precommit_checks.py b/ci/tools/tests/test_migrate_precommit_checks.py deleted file mode 100644 index 816e299c09b..00000000000 --- a/ci/tools/tests/test_migrate_precommit_checks.py +++ /dev/null @@ -1,309 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 - -from __future__ import annotations - -import copy -import json - -import pytest - -from ci.tools import migrate_precommit_checks as migration - -REPOSITORY = "NVIDIA/cuda-python" -MAIN_SHA = "a" * 40 -PR_SHA = "b" * 40 -RULESET_ENDPOINT = f"repos/{REPOSITORY}/rulesets/12" - - -@pytest.fixture -def ruleset(): - return { - "id": 12, - "name": "Main protection", - "source_type": "Repository", - "source": REPOSITORY, - "target": "branch", - "enforcement": "active", - "conditions": {"ref_name": {"include": ["~DEFAULT_BRANCH"], "exclude": []}}, - "bypass_actors": [{"actor_id": 9, "actor_type": "Team", "bypass_mode": "pull_request"}], - "rules": [ - {"type": "deletion"}, - { - "type": "required_status_checks", - "parameters": { - "strict_required_status_checks_policy": True, - "do_not_enforce_on_create": True, - "required_status_checks": [ - {"context": "CI", "integration_id": 15368}, - {"context": "pre-commit.ci - pr", "integration_id": 68672}, - {"context": "Other service", "integration_id": 777}, - ], - }, - }, - {"type": "pull_request", "parameters": {"required_approving_review_count": 2}}, - ], - } - - -@pytest.fixture -def github(monkeypatch, ruleset): - effective_rule = { - **copy.deepcopy(ruleset["rules"][1]), - "ruleset_id": ruleset["id"], - "ruleset_source_type": "Repository", - "ruleset_source": REPOSITORY, - } - checks = [ - { - "id": index, - "name": context, - "app": {"id": 15368}, - "status": "completed", - "conclusion": "success", - "details_url": f"https://github.com/{REPOSITORY}/actions/runs/{index}/job/100", - "check_suite": {"id": index + 100}, - } - for index, context in enumerate(migration.CHECK_WORKFLOWS, start=1) - ] - responses = { - f"repos/{REPOSITORY}/rules/branches/main?per_page=100": [[effective_rule]], - RULESET_ENDPOINT: copy.deepcopy(ruleset), - f"repos/{REPOSITORY}": {"default_branch": "main"}, - f"repos/{REPOSITORY}/branches/main": {"commit": {"sha": MAIN_SHA}}, - f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100": [{"check_runs": checks}], - f"repos/{REPOSITORY}/commits/{PR_SHA}/check-runs?filter=latest&per_page=100": [{"check_runs": checks}], - f"repos/{REPOSITORY}/commits/{PR_SHA}/pulls?per_page=100": [ - [ - { - "state": "open", - "base": {"ref": "main", "repo": {"full_name": REPOSITORY}}, - "head": {"sha": PR_SHA}, - "merge_commit_sha": "c" * 40, - } - ] - ], - } - for path in set(migration.CHECK_WORKFLOWS.values()): - responses[f"repos/{REPOSITORY}/contents/{path}?ref={MAIN_SHA}"] = {"path": path, "type": "file"} - for check in checks: - responses[f"repos/{REPOSITORY}/actions/runs/{check['id']}"] = { - "path": migration.CHECK_WORKFLOWS[check["name"]], - "check_suite_id": check["check_suite"]["id"], - "status": "completed", - "conclusion": "success", - } - calls = [] - - def api(endpoint, *, method="GET", payload=None, paginate=False): - calls.append((method, endpoint, copy.deepcopy(payload))) - if method == "PUT": - responses[endpoint]["rules"] = copy.deepcopy(payload["rules"]) - return copy.deepcopy(responses[endpoint]) - - monkeypatch.setattr(migration, "github_api", api) - return responses, calls - - -@pytest.mark.agent_authored(model="gpt-6") -def test_replaces_service_without_changing_unrelated_rules_or_parameters(ruleset): - original = copy.deepcopy(ruleset) - migrated = migration.migrate_rules(ruleset["rules"]) - - assert ruleset == original - assert migrated[0] == original["rules"][0] - assert migrated[2] == original["rules"][2] - parameters = migrated[1]["parameters"] - assert parameters["strict_required_status_checks_policy"] is True - assert parameters["do_not_enforce_on_create"] is True - checks = parameters["required_status_checks"] - assert checks == [ - {"context": "CI", "integration_id": 15368}, - {"context": "Pre-commit (Linux)", "integration_id": 15368}, - {"context": "Pre-commit (Windows)", "integration_id": 15368}, - {"context": "Documentation links", "integration_id": 15368}, - {"context": "Other service", "integration_id": 777}, - ] - assert migration.migrate_rules(migrated) == migrated - - -@pytest.mark.agent_authored(model="gpt-6") -def test_existing_actions_requirements_are_not_duplicated(ruleset): - checks = ruleset["rules"][1]["parameters"]["required_status_checks"] - existing = {"context": "Documentation links", "integration_id": 15368} - checks.append(existing) - - migrated = migration.migrate_rules(ruleset["rules"]) - - assert migrated[1]["parameters"]["required_status_checks"].count(existing) == 1 - - -@pytest.mark.parametrize("integration", [None, 777]) -@pytest.mark.agent_authored(model="gpt-6") -def test_refuses_unexpected_old_integration(ruleset, integration): - ruleset["rules"][1]["parameters"]["required_status_checks"][1]["integration_id"] = integration - - with pytest.raises(RuntimeError, match="expected integration"): - migration.migrate_rules(ruleset["rules"]) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_refuses_conflicting_replacement_integration(ruleset): - ruleset["rules"][1]["parameters"]["required_status_checks"].append( - {"context": "Documentation links", "integration_id": 777} - ) - - with pytest.raises(RuntimeError, match="different or unspecified integration"): - migration.migrate_rules(ruleset["rules"]) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_duplicate_valid_context_cannot_hide_conflicting_integration(ruleset): - ruleset["rules"][1]["parameters"]["required_status_checks"].extend( - [ - {"context": "Documentation links", "integration_id": 777}, - {"context": "Documentation links", "integration_id": 15368}, - ] - ) - - with pytest.raises(RuntimeError, match="different or unspecified integration"): - migration.migrate_rules(ruleset["rules"]) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_cli_defaults_to_read_only_preview(github, capsys): - responses, calls = github - - assert migration.main(["--repo", REPOSITORY]) == 0 - - preview = json.loads(capsys.readouterr().out) - assert preview["mode"] == "dry-run" - assert preview["changes"][0]["before"] == responses[RULESET_ENDPOINT] - assert set(preview["changes"][0]["put_payload"]) == {"rules"} - assert all(method == "GET" for method, endpoint, payload in calls) - assert not any("/contents/" in endpoint for method, endpoint, payload in calls) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_inherited_requirement_is_rejected_before_any_write(github): - responses, calls = github - rule = responses[f"repos/{REPOSITORY}/rules/branches/main?per_page=100"][0][0] - rule["ruleset_source_type"] = "Organization" - rule["ruleset_source"] = "NVIDIA" - - with pytest.raises(RuntimeError, match="inherited from Organization NVIDIA ruleset 12"): - migration.discover_changes(REPOSITORY) - - assert len(calls) == 1 - - -@pytest.mark.agent_authored(model="gpt-6") -def test_replacement_workflows_must_be_files_on_current_main(github): - responses, calls = github - responses[f"repos/{REPOSITORY}/contents/.github/workflows/lychee.yml?ref={MAIN_SHA}"] = [] - - with pytest.raises(RuntimeError, match="not a workflow file on current main"): - migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.parametrize("verified_sha", [None, PR_SHA]) -@pytest.mark.agent_authored(model="gpt-6") -def test_apply_requires_green_representative_checks_and_preserves_settings(github, ruleset, verified_sha): - responses, calls = github - changes = migration.discover_changes(REPOSITORY) - - assert migration.apply_changes(REPOSITORY, changes, verified_sha) == (verified_sha or MAIN_SHA) - - writes = [call for call in calls if call[0] == "PUT"] - assert writes == [("PUT", RULESET_ENDPOINT, {"rules": migration.migrate_rules(ruleset["rules"])})] - assert {key: value for key, value in responses[RULESET_ENDPOINT].items() if key != "rules"} == { - key: value for key, value in ruleset.items() if key != "rules" - } - - -@pytest.mark.parametrize("conclusion", ["failure", "cancelled", "skipped", "neutral", None]) -@pytest.mark.agent_authored(model="gpt-6") -def test_failed_or_skipped_checks_prevent_all_writes(github, conclusion): - responses, calls = github - checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] - checks[-1]["conclusion"] = conclusion - changes = migration.discover_changes(REPOSITORY) - - with pytest.raises(RuntimeError, match="not completed and successful"): - migration.apply_changes(REPOSITORY, changes, None) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_newer_pending_run_cannot_be_masked_by_older_success(github): - responses, calls = github - checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] - checks.append({**checks[0], "id": 99, "status": "in_progress", "conclusion": None}) - - with pytest.raises(RuntimeError, match="not completed and successful"): - migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_other_apps_cannot_satisfy_replacement_checks(github): - responses, calls = github - checks = responses[f"repos/{REPOSITORY}/commits/{MAIN_SHA}/check-runs?filter=latest&per_page=100"][0]["check_runs"] - checks[0]["app"]["id"] = 777 - - with pytest.raises(RuntimeError, match="Missing GitHub Actions check"): - migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.parametrize( - ("field", "value"), - [("path", ".github/workflows/unrelated.yml"), ("check_suite_id", 777), ("conclusion", "failure")], -) -@pytest.mark.agent_authored(model="gpt-6") -def test_check_must_belong_to_the_expected_successful_workflow(github, field, value): - responses, calls = github - responses[f"repos/{REPOSITORY}/actions/runs/1"][field] = value - - with pytest.raises(RuntimeError, match="does not belong to a successful"): - migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), None) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_unassociated_sha_is_rejected(github): - responses, calls = github - responses[f"repos/{REPOSITORY}/commits/{PR_SHA}/pulls?per_page=100"] = [[]] - - with pytest.raises(RuntimeError, match="open PR targeting main"): - migration.apply_changes(REPOSITORY, migration.discover_changes(REPOSITORY), PR_SHA) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_ruleset_edit_during_preflight_prevents_writes(github): - responses, calls = github - changes = migration.discover_changes(REPOSITORY) - responses[RULESET_ENDPOINT]["conditions"]["ref_name"]["exclude"].append("refs/heads/release") - - with pytest.raises(RuntimeError, match="changed after discovery"): - migration.apply_changes(REPOSITORY, changes, None) - - assert all(method == "GET" for method, endpoint, payload in calls) - - -@pytest.mark.agent_authored(model="gpt-6") -def test_sha_must_be_explicit_full_commit(github): - responses, calls = github - - with pytest.raises(SystemExit, match="2"): - migration.main(["--repo", REPOSITORY, "--verified-sha", "main", "--apply"]) - - assert not calls From b2b8f087b20379d8ce5f59795cc19a6bfbe372b2 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Fri, 2 Oct 2026 17:09:10 -0700 Subject: [PATCH 07/12] LYCHEE_VERSION Must match comment; use Python 3.14 --- .github/workflows/lychee.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index edc2b11abd3..8d744fc63dd 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -37,7 +37,7 @@ defaults: env: PIXI_LOCKED: "true" - LYCHEE_VERSION: v0.24.2 + LYCHEE_VERSION: v0.24.2 # Must match the pinned rev in .pre-commit-config.yaml jobs: links: @@ -61,7 +61,7 @@ jobs: - name: Set up Python uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: - python-version: "3.13" + python-version: "3.14" - name: Setup pixi if: ${{ matrix.kind == 'rendered' }} From 27b3cf9fc09a450fef5241aad6fbdb66c32d11fe Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Fri, 2 Oct 2026 18:32:30 -0700 Subject: [PATCH 08/12] Add lychee.yml high-level comment. --- .github/workflows/lychee.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index 8d744fc63dd..ff4dabe9268 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -2,6 +2,12 @@ # # SPDX-License-Identifier: Apache-2.0 +# Check checked-in documentation and rendered HTML on PR updates and nightly. +# Build fresh HTML from sources to check Sphinx-generated links and anchors. +# This runs independently of ci.yml → build-docs.yml, which builds and deploys +# docs using CI-built wheels. Nightly caches successful link-check results for +# reuse by PR runs of this workflow. + name: "CI: Documentation links" on: From 3458740fb1e30219536c094c3f9c04556e4930b2 Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Sat, 3 Oct 2026 11:40:37 -0700 Subject: [PATCH 09/12] Trim back CONTRIBUTING.md changes, to keep the file focused on what external developers need to know. --- CONTRIBUTING.md | 24 +++++------------------- 1 file changed, 5 insertions(+), 19 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 32830b4bb47..bff02659125 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -214,23 +214,9 @@ A few things to keep in mind: --config-file cuda_core/pyproject.toml ## Pre-commit -GitHub Actions runs all pre-commit hooks on Linux and Windows for every pull -request update, including draft PRs. These jobs skip lychee and the local hook -installation reminder. A separate Linux workflow checks authored documentation -and freshly rendered HTML, including link fragments. It runs on PR updates and -from nightly CI; copied `pull-request/*` branches do not repeat these checks. - -Nightly link checks start with a fresh cache and publish successful checks for -PRs to reuse for up to one day. Dependabot checks pre-commit hook revisions -monthly and opens update PRs with the `CI/CD` and `dependencies` labels, without -an automatic assignee or milestone. Lychee version updates are maintained -separately so its local hook and CI binary stay aligned. The local lychee hook -keeps its current behavior until a stable release supports an explicit cache -location. - -See [the pre-commit workflow guide](ci/README-pre-commit-migration.md) for manual -workflow testing. The existing pre-commit.ci service remains required until the -required-check cutover is complete. +GitHub Actions runs pre-commit checks on Linux and Windows and checks +documentation links for pull requests. Check failures must be resolved +before merging. To set yourself up for running pre-commit checks locally and to catch issues before pushing your changes, follow these steps: @@ -245,8 +231,8 @@ Installing the hook is required, not optional. Some of the automated checks keep the tree consistent if they run on *every* commit. Relying on manual `pre-commit run --all-files` invocations means these checks can be skipped between commits, leaving stale headers or out-of-date stubs in the history. -If the hook isn't installed, local `pre-commit run` will print a visible -warning reminding you to run `pre-commit install`. +If the hook isn't installed, `pre-commit run` will print a visible warning +reminding you to run `pre-commit install`. Windows contributors: see [Pre-commit lychee workaround](#pre-commit-lychee-workaround) under Development on Windows. From d592efb851e3aacf5eecc68866d935fbc37fc49c Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Sat, 3 Oct 2026 12:00:20 -0700 Subject: [PATCH 10/12] ci: move pre-commit migration notes out of the repository --- ci/README-pre-commit-migration.md | 51 ------------------------------- 1 file changed, 51 deletions(-) delete mode 100644 ci/README-pre-commit-migration.md diff --git a/ci/README-pre-commit-migration.md b/ci/README-pre-commit-migration.md deleted file mode 100644 index a7a39c4e81c..00000000000 --- a/ci/README-pre-commit-migration.md +++ /dev/null @@ -1,51 +0,0 @@ -# Migrating from pre-commit.ci - -The replacement checks are `Pre-commit (Linux)`, `Pre-commit (Windows)`, and -`Documentation links`, all from the GitHub Actions App (integration ID 15368). -They replace `pre-commit.ci - pr` from integration ID 68672. - -The workflow pull requests can remain draft while their runs are reviewed. -Keep pre-commit.ci installed and required throughout that review. Do not change -the live required checks until the replacement workflows are merged and green. - -## Test the draft workflows - -Use the PR branch as `REF` while reviewing the implementation. These commands -run hosted checks without copying a PR into the heavyweight CUDA CI: - -```sh -gh workflow run pre-commit.yml --repo NVIDIA/cuda-python --ref REF -gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ - -f refresh-cache=true -gh workflow run lychee.yml --repo NVIDIA/cuda-python --ref REF \ - -f refresh-cache=false -gh workflow run ci-nightly.yml --repo NVIDIA/cuda-python --ref REF \ - -f documentation-links-only=true -gh run list --repo NVIDIA/cuda-python --branch REF -``` - -Dependabot checks hook revisions monthly and opens update PRs with the `CI/CD` -and `dependencies` labels, without an automatic assignee or milestone. Lychee's -refresh mode skips restoration and publishes a new snapshot even when some -links fail; only successful checks enter the persisted cache. Its normal mode -restores the newest matching snapshot and does not publish one. Branch tests -write branch-scoped caches. After merge, nightly refreshes run on `main` and -publish the baseline that all PRs can read, including fork PRs. Separate -authored/rendered namespaces include the checker version and checking policy; -documentation edits do not invalidate previously successful external checks. - -The nightly documentation-only mode exercises the reusable-workflow call and -status gate, plus standalone CI-tool tests. It skips wheel lookup and requires -all wheel/GPU jobs to remain skipped. The scheduled/default nightly mode still -runs the complete existing suite. - -For the same source-based docs build locally: - -```sh -PIXI_LOCKED=true pixi run --manifest-path cuda_core/pixi.toml -e docs \ - docs-build-all-latest -``` - -The build verifies all three sibling libraries import from the checkout, -installs metapackage metadata without dependencies, and assembles all four -latest documentation trees under `artifacts/docs`. From ed2159275e260c76f5b4600c9f60471c6e21f1bd Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Sat, 3 Oct 2026 14:14:43 -0700 Subject: [PATCH 11/12] ci: retry documentation link checks with an accumulating cache --- .github/workflows/lychee.yml | 51 +++---- ci/tools/retry_lychee.py | 121 +++++++++++++++++ ci/tools/tests/test_retry_lychee.py | 197 ++++++++++++++++++++++++++++ 3 files changed, 344 insertions(+), 25 deletions(-) create mode 100644 ci/tools/retry_lychee.py create mode 100644 ci/tools/tests/test_retry_lychee.py diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index ff4dabe9268..341a34eb54d 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -54,6 +54,21 @@ jobs: fail-fast: false matrix: kind: [authored, rendered] + env: + # The first pass and all retries use the same inputs and cache policy. + LYCHEE_ARGS: >- + --files-from "${{ github.workspace }}/lychee-files.txt" + ${{ matrix.kind == 'rendered' && '--include-fragments=full' || '' }} + --cache + --max-cache-age 1d + --max-concurrency 16 + --host-concurrency 2 + --host-request-interval 250ms + --max-retries 3 + --retry-wait-time 5 + --timeout 30 + --no-progress + --config "${{ github.workspace }}/lychee.toml" steps: - name: Checkout checked revision uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -84,16 +99,13 @@ jobs: id: inputs env: KIND: ${{ matrix.kind }} - POLICY_HASH: ${{ hashFiles('lychee.toml', '.github/workflows/lychee.yml', 'ci/tools/prepare_lychee_inputs.py') }} + POLICY_HASH: ${{ hashFiles('lychee.toml', '.github/workflows/lychee.yml', 'ci/tools/prepare_lychee_inputs.py', 'ci/tools/retry_lychee.py') }} run: | python ci/tools/prepare_lychee_inputs.py "${KIND}" \ --output "${GITHUB_WORKSPACE}/lychee-files.txt" prefix="lychee-v1-${LYCHEE_VERSION}-${KIND}-${POLICY_HASH}-baseline-" echo "prefix=${prefix}" >> "${GITHUB_OUTPUT}" echo "key=${prefix}${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" >> "${GITHUB_OUTPUT}" - if [[ "${KIND}" == "rendered" ]]; then - echo "fragment-args=--include-fragments=full" >> "${GITHUB_OUTPUT}" - fi - name: Restore successful checks from the nightly baseline if: ${{ !inputs.refresh-cache }} @@ -107,36 +119,25 @@ jobs: id: lychee uses: lycheeverse/lychee-action@6da1d14f3a43098a294b7696d93d938aa8d20fc0 # supports v0.24.x archive layout with: - args: >- - --files-from ${{ github.workspace }}/lychee-files.txt - ${{ steps.inputs.outputs.fragment-args }} - --cache - --max-cache-age 1d - --max-concurrency 16 - --host-concurrency 2 - --host-request-interval 250ms - --max-retries 3 - --retry-wait-time 5 - --timeout 30 - --no-progress - --config ${{ github.workspace }}/lychee.toml - fail: true + args: ${{ env.LYCHEE_ARGS }} + # The next step owns bounded retries and the final success gate. + fail: false failIfEmpty: true format: markdown - jobSummary: true + jobSummary: false lycheeVersion: ${{ env.LYCHEE_VERSION }} output: lychee-${{ matrix.kind }}.md token: ${{ github.token }} - - name: Require a successful checker result + - name: Retry failed links using successful checks from each pass if: ${{ steps.lychee.outcome == 'success' }} env: EXIT_CODE: ${{ steps.lychee.outputs.exit_code }} + REPORT: lychee-${{ matrix.kind }}.md + GITHUB_TOKEN: ${{ github.token }} run: | - if [[ "${EXIT_CODE}" != "0" ]]; then - echo "::error::Lychee did not report success (exit_code='${EXIT_CODE}')." - exit 1 - fi + python ci/tools/retry_lychee.py --initial-exit-code "${EXIT_CODE}" \ + --report "${REPORT}" --max-attempts 10 - name: Publish fresh successful checks # Manual branch tests publish isolated branch caches. Nightly callers @@ -152,7 +153,7 @@ jobs: uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: lychee-${{ matrix.kind }}-${{ github.run_id }}-${{ github.run_attempt }} - path: lychee-${{ matrix.kind }}.md + path: lychee-${{ matrix.kind }}*.md if-no-files-found: ignore retention-days: 7 diff --git a/ci/tools/retry_lychee.py b/ci/tools/retry_lychee.py new file mode 100644 index 00000000000..284323d05c2 --- /dev/null +++ b/ci/tools/retry_lychee.py @@ -0,0 +1,121 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Retry complete lychee passes while retaining successful checks in its cache.""" + +from __future__ import annotations + +import argparse +import os +import re +import shlex +import shutil +import subprocess +import time +from pathlib import Path + +LINK_CHECK_FAILURE = 2 + + +def read_report(path: Path) -> str: + """Reject missing, malformed, or empty-check reports instead of accepting them.""" + text = path.read_text(encoding="utf-8") + total = re.search(r"^\|[^|\n]*\bTotal\s*\|\s*(\d+)\s*\|", text, re.MULTILINE) + if total is None: + raise ValueError(f"Lychee report has no Total count: {path}") + if int(total.group(1)) == 0: + raise ValueError(f"Lychee checked no links: {path}") + return text + + +def attempt_report(report: Path, attempt: int) -> Path: + return report.with_name(f"{report.stem}-attempt-{attempt}{report.suffix}") + + +def finish(report_text: str | None, exit_codes: list[int], message: str) -> None: + """Publish the final result without presenting recovered failures as current.""" + summary = f"## Documentation links\n\n{message}\n\n| Attempt | Exit code |\n| --- | --- |\n" + summary += "".join(f"| {attempt} | {code} |\n" for attempt, code in enumerate(exit_codes, start=1)) + if report_text is not None: + summary += f"\n### Final attempt\n\n{report_text}\n" + else: + summary += ( + "\nThe final attempt did not produce a link-check report. Earlier reports are retained as artifacts.\n" + ) + print(message, flush=True) + if summary_path := os.environ.get("GITHUB_STEP_SUMMARY"): + with Path(summary_path).open("a", encoding="utf-8") as stream: + stream.write(summary) + + +def retry_lychee(initial_exit_code: int, report: Path, max_attempts: int = 10) -> int: + """Count the action's first pass toward the limit, and retry only link failures.""" + if initial_exit_code not in (0, 1, 2, 3): + raise ValueError(f"Unexpected initial lychee exit code: {initial_exit_code}") + if max_attempts < 1: + raise ValueError("max_attempts must be positive") + exit_codes = [initial_exit_code] + if initial_exit_code not in (0, LINK_CHECK_FAILURE): + finish(None, exit_codes, f"Lychee stopped on attempt 1 with exit code {initial_exit_code}; no retry.") + return initial_exit_code + + report_text = read_report(report) + shutil.copyfile(report, attempt_report(report, 1)) + if initial_exit_code == 0: + finish(report_text, exit_codes, "Lychee passed on attempt 1.") + return 0 + if max_attempts == 1: + finish(report_text, exit_codes, "Lychee still has broken links after 1 attempt.") + return LINK_CHECK_FAILURE + + raw_args = os.environ.get("LYCHEE_ARGS") + if not raw_args or not (lychee_args := shlex.split(raw_args)): + raise ValueError("LYCHEE_ARGS must contain the same arguments used by the first lychee pass") + + for attempt in range(2, max_attempts + 1): + cooldown = min(60 * (attempt - 1), 120) + print( + f"Lychee attempt {attempt - 1} failed; waiting {cooldown}s before attempt {attempt}/{max_attempts}.", + flush=True, + ) + time.sleep(cooldown) + current_report = attempt_report(report, attempt) + result = subprocess.run( # noqa: S603 + ["lychee", *lychee_args, "--mode", "task", "--format", "markdown", "--output", str(current_report)], # noqa: S607 + check=False, + ) + exit_codes.append(result.returncode) + if result.returncode not in (0, LINK_CHECK_FAILURE): + finish( + None, exit_codes, f"Lychee stopped on attempt {attempt} with exit code {result.returncode}; no retry." + ) + return result.returncode if result.returncode > 0 else 1 + report_text = read_report(current_report) + shutil.copyfile(current_report, report) + if result.returncode == 0: + finish( + report_text, + exit_codes, + f"Lychee passed on attempt {attempt}/{max_attempts}; successful checks were retained between passes.", + ) + return 0 + + finish(report_text, exit_codes, f"Lychee still has broken links after {max_attempts} attempts.") + return LINK_CHECK_FAILURE + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--initial-exit-code", type=int, required=True) + parser.add_argument("--report", type=Path, required=True) + parser.add_argument("--max-attempts", type=int, default=10) + args = parser.parse_args() + try: + return retry_lychee(args.initial_exit_code, args.report, args.max_attempts) + except (OSError, ValueError) as error: + print(f"::error::{error}", flush=True) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ci/tools/tests/test_retry_lychee.py b/ci/tools/tests/test_retry_lychee.py new file mode 100644 index 00000000000..b3e92adede2 --- /dev/null +++ b/ci/tools/tests/test_retry_lychee.py @@ -0,0 +1,197 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import subprocess +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).parent.parent)) +import retry_lychee + + +def markdown_report(total=2, errors=0): + return f"# Summary\n\n| Status | Count |\n| --- | --- |\n| \U0001f50d Total | {total} |\n| Error | {errors} |\n" + + +@pytest.mark.agent_authored(model="gpt-6") +def test_initial_success_never_runs_lychee_and_preserves_first_report(tmp_path, monkeypatch): + report = tmp_path / "lychee-authored.md" + report.write_text(markdown_report(), encoding="utf-8") + summary = tmp_path / "summary.md" + monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(0, report) == 0 + assert (tmp_path / "lychee-authored-attempt-1.md").read_text(encoding="utf-8") == markdown_report() + assert "passed on attempt 1" in summary.read_text(encoding="utf-8") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_retries_reuse_cache_and_identical_arguments_until_success(tmp_path, monkeypatch): + report = tmp_path / "lychee-rendered.md" + report.write_text(markdown_report(errors=1), encoding="utf-8") + cache = tmp_path / ".lycheecache" + cache.write_text("first-success\n", encoding="utf-8") + summary = tmp_path / "summary.md" + summary.write_text("Earlier step\n", encoding="utf-8") + monkeypatch.chdir(tmp_path) + monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) + monkeypatch.setenv("LYCHEE_ARGS", '--files-from "input files.txt" --cache --config "link config.toml"') + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + commands = [] + + def run(command, *, check): + assert check is False + commands.append(command) + assert command[:6] == ["lychee", "--files-from", "input files.txt", "--cache", "--config", "link config.toml"] + assert command[6:10] == ["--mode", "task", "--format", "markdown"] + assert cache.read_text(encoding="utf-8") == "first-success\n" + ( + "second-success\n" if len(commands) == 2 else "" + ) + code = 2 if len(commands) == 1 else 0 + Path(command[-1]).write_text(markdown_report(errors=int(code != 0)), encoding="utf-8") + if code == 2: + with cache.open("a", encoding="utf-8") as stream: + stream.write("second-success\n") + return subprocess.CompletedProcess(command, code) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == 0 + assert delays == [60, 120] + assert len(commands) == 2 + assert report.read_text(encoding="utf-8") == markdown_report() + assert (tmp_path / "lychee-rendered-attempt-1.md").read_text(encoding="utf-8") == markdown_report(errors=1) + assert (tmp_path / "lychee-rendered-attempt-2.md").read_text(encoding="utf-8") == markdown_report(errors=1) + assert (tmp_path / "lychee-rendered-attempt-3.md").read_text(encoding="utf-8") == markdown_report() + text = summary.read_text(encoding="utf-8") + assert text.startswith("Earlier step\n") + assert "passed on attempt 3/10" in text + assert "| 1 | 2 |\n| 2 | 2 |\n| 3 | 0 |" in text + assert text.endswith(markdown_report() + "\n") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_ten_attempts_exhausted_remains_a_failure(tmp_path, monkeypatch): + report = tmp_path / "lychee.md" + report.write_text(markdown_report(errors=1), encoding="utf-8") + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + calls = [] + + def run(command, *, check): + calls.append(command) + Path(command[-1]).write_text(markdown_report(errors=1), encoding="utf-8") + return subprocess.CompletedProcess(command, 2) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == 2 + assert len(calls) == 9 + assert delays == [60, *([120] * 8)] + assert len(list(tmp_path.glob("lychee-attempt-*.md"))) == 10 + + +@pytest.mark.parametrize("code", [1, 3]) +@pytest.mark.agent_authored(model="gpt-6") +def test_initial_non_link_failure_stops_without_retry(tmp_path, monkeypatch, code): + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(code, tmp_path / "missing.md") == code + + +@pytest.mark.parametrize("code", [1, 3, -15]) +@pytest.mark.agent_authored(model="gpt-6") +def test_retry_non_link_failure_stops_and_summary_does_not_claim_stale_report(tmp_path, monkeypatch, code): + report = tmp_path / "lychee.md" + report.write_text(markdown_report(errors=1), encoding="utf-8") + summary = tmp_path / "summary.md" + monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + calls = [] + + def run(command, *, check): + calls.append(command) + return subprocess.CompletedProcess(command, code) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == (code if code > 0 else 1) + assert len(calls) == 1 + assert delays == [60] + text = summary.read_text(encoding="utf-8") + assert "did not produce a link-check report" in text + assert "### Final attempt" not in text + assert "Total |" not in text + + +@pytest.mark.parametrize("contents", ["", "not a report", markdown_report(total=0)]) +@pytest.mark.agent_authored(model="gpt-6") +def test_invalid_or_empty_initial_report_cannot_pass(tmp_path, monkeypatch, contents): + report = tmp_path / "lychee.md" + report.write_text(contents, encoding="utf-8") + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + + with pytest.raises(ValueError, match="no Total count|checked no links"): + retry_lychee.retry_lychee(0, report) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_missing_initial_report_cannot_pass(tmp_path): + with pytest.raises(FileNotFoundError): + retry_lychee.retry_lychee(0, tmp_path / "missing.md") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_successful_retry_with_zero_links_still_fails(tmp_path, monkeypatch): + report = tmp_path / "lychee.md" + report.write_text(markdown_report(errors=1), encoding="utf-8") + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: None) + + def run(command, *, check): + Path(command[-1]).write_text(markdown_report(total=0), encoding="utf-8") + return subprocess.CompletedProcess(command, 0) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + with pytest.raises(ValueError, match="checked no links"): + retry_lychee.retry_lychee(2, report) + + +@pytest.mark.parametrize("code", [-1, 4, 255]) +@pytest.mark.agent_authored(model="gpt-6") +def test_unknown_initial_exit_code_is_rejected(tmp_path, code): + with pytest.raises(ValueError, match="Unexpected initial lychee exit code"): + retry_lychee.retry_lychee(code, tmp_path / "missing.md") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_retry_requires_shared_arguments_before_sleeping(tmp_path, monkeypatch): + report = tmp_path / "lychee.md" + report.write_text(markdown_report(errors=1), encoding="utf-8") + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + with pytest.raises(ValueError, match="LYCHEE_ARGS"): + retry_lychee.retry_lychee(2, report) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_one_attempt_limit_keeps_failure_without_requiring_retry_arguments(tmp_path, monkeypatch): + report = tmp_path / "lychee.md" + report.write_text(markdown_report(errors=1), encoding="utf-8") + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report, max_attempts=1) == 2 From d15a1516cbbc39583003f03d4970622ef9c5af3f Mon Sep 17 00:00:00 2001 From: "Ralf W. Grosse-Kunstleve" Date: Sat, 3 Oct 2026 15:41:09 -0700 Subject: [PATCH 12/12] ci: retry only transient documentation link failures --- .github/workflows/lychee.yml | 16 +- ci/tools/retry_lychee.py | 185 +++++++++++++---- ci/tools/tests/test_retry_lychee.py | 306 +++++++++++++++++++++++----- 3 files changed, 405 insertions(+), 102 deletions(-) diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml index 341a34eb54d..4ebd0473e1f 100644 --- a/.github/workflows/lychee.yml +++ b/.github/workflows/lychee.yml @@ -122,18 +122,20 @@ jobs: args: ${{ env.LYCHEE_ARGS }} # The next step owns bounded retries and the final success gate. fail: false - failIfEmpty: true - format: markdown + # The action's empty-report guard only understands Markdown. + # Our helper validates the JSON report and rejects zero checked links. + failIfEmpty: false + format: json jobSummary: false lycheeVersion: ${{ env.LYCHEE_VERSION }} - output: lychee-${{ matrix.kind }}.md + output: lychee-${{ matrix.kind }}.json token: ${{ github.token }} - - name: Retry failed links using successful checks from each pass + - name: Retry transient link failures using cached successful checks if: ${{ steps.lychee.outcome == 'success' }} env: EXIT_CODE: ${{ steps.lychee.outputs.exit_code }} - REPORT: lychee-${{ matrix.kind }}.md + REPORT: lychee-${{ matrix.kind }}.json GITHUB_TOKEN: ${{ github.token }} run: | python ci/tools/retry_lychee.py --initial-exit-code "${EXIT_CODE}" \ @@ -153,7 +155,9 @@ jobs: uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: lychee-${{ matrix.kind }}-${{ github.run_id }}-${{ github.run_attempt }} - path: lychee-${{ matrix.kind }}*.md + path: | + lychee-${{ matrix.kind }}*.json + lychee-${{ matrix.kind }}*.md if-no-files-found: ignore retention-days: 7 diff --git a/ci/tools/retry_lychee.py b/ci/tools/retry_lychee.py index 284323d05c2..b9cbf5f06c7 100644 --- a/ci/tools/retry_lychee.py +++ b/ci/tools/retry_lychee.py @@ -1,47 +1,143 @@ # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Retry complete lychee passes while retaining successful checks in its cache.""" +"""Retry transient lychee failures while retaining successful checks in its cache.""" from __future__ import annotations import argparse +import html +import json import os import re import shlex import shutil import subprocess import time +from dataclasses import dataclass from pathlib import Path +from urllib.parse import urlsplit LINK_CHECK_FAILURE = 2 -def read_report(path: Path) -> str: - """Reject missing, malformed, or empty-check reports instead of accepting them.""" - text = path.read_text(encoding="utf-8") - total = re.search(r"^\|[^|\n]*\bTotal\s*\|\s*(\d+)\s*\|", text, re.MULTILINE) - if total is None: - raise ValueError(f"Lychee report has no Total count: {path}") - if int(total.group(1)) == 0: +@dataclass(frozen=True) +class Failure: + source: str + url: str + text: str + details: str | None + code: int | None + timed_out: bool + + @property + def transient(self) -> bool: + try: + parsed = urlsplit(self.url) + is_http = parsed.scheme in ("http", "https") and bool(parsed.hostname) + except ValueError: + return False + return is_http and ( + self.timed_out or self.code in (408, 429) or (self.code is not None and 500 <= self.code <= 599) + ) + + +@dataclass(frozen=True) +class Report: + total: int + errors: int + timeouts: int + failures: tuple[Failure, ...] + + @property + def transient(self) -> bool: + return bool(self.failures) and all(failure.transient for failure in self.failures) + + +def read_report(path: Path, exit_code: int) -> Report: + """Validate the pinned lychee JSON schema and its agreement with the exit code.""" + data = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise ValueError(f"Lychee report must be a JSON object: {path}") + counts = {} + for name in ("total", "errors", "timeouts"): + value = data.get(name) + if type(value) is not int or value < 0: + raise ValueError(f"Lychee report {name} must be a nonnegative integer: {path}") + counts[name] = value + if counts["total"] == 0: raise ValueError(f"Lychee checked no links: {path}") - return text + if counts["errors"] + counts["timeouts"] > counts["total"]: + raise ValueError(f"Lychee failure counts exceed the total: {path}") + + failures = [] + for name, timed_out, count in (("error_map", False, counts["errors"]), ("timeout_map", True, counts["timeouts"])): + mapping = data.get(name) + if not isinstance(mapping, dict): + raise ValueError(f"Lychee report {name} must be an object: {path}") + entries = [] + for source, responses in mapping.items(): + if not isinstance(source, str) or not source or not isinstance(responses, list): + raise ValueError(f"Lychee report {name} must map source names to response lists: {path}") + for response in responses: + if not isinstance(response, dict) or not isinstance(response.get("url"), str) or not response["url"]: + raise ValueError(f"Lychee report has a malformed failure URL: {path}") + status = response.get("status") + if not isinstance(status, dict) or not isinstance(status.get("text"), str) or not status["text"]: + raise ValueError(f"Lychee report has a malformed failure status: {path}") + code = status.get("code") + details = status.get("details") + if "code" in status: + if type(code) is not int or not 100 <= code <= 999: + raise ValueError(f"Lychee report has a malformed HTTP status code: {path}") + elif not isinstance(details, str): + raise ValueError(f"Lychee report non-HTTP status must contain string details: {path}") + if "details" in status and not isinstance(details, str): + raise ValueError(f"Lychee report status details must be a string: {path}") + entries.append(Failure(source, response["url"], status["text"], details, code, timed_out)) + # Counters count occurrences; response maps deduplicate equivalent checks. + if bool(count) != bool(entries): + raise ValueError(f"Lychee report {name} disagrees with its failure count: {path}") + failures.extend(entries) + + if (exit_code == 0) != (not failures): + raise ValueError(f"Lychee report failures disagree with exit code {exit_code}: {path}") + return Report(counts["total"], counts["errors"], counts["timeouts"], tuple(failures)) + + +def escape_markdown(text: str) -> str: + text = html.escape(text).replace("\r", " ").replace("\n", " ") + return re.sub(r"([\\`*_~\[\]|])", r"\\\1", text) + + +def render_report(report: Report) -> str: + result = ( + "| Status | Count |\n| --- | --- |\n" + f"| Total | {report.total} |\n| Errors | {report.errors} |\n| Timeouts | {report.timeouts} |\n" + ) + if report.failures: + result += "\n| Source | URL | Status | Details |\n| --- | --- | --- | --- |\n" + for failure in report.failures: + cells = (failure.source, failure.url, failure.text, failure.details or "") + result += "| " + " | ".join(escape_markdown(cell) for cell in cells) + " |\n" + return result def attempt_report(report: Path, attempt: int) -> Path: return report.with_name(f"{report.stem}-attempt-{attempt}{report.suffix}") -def finish(report_text: str | None, exit_codes: list[int], message: str) -> None: +def finish(report: Path, data: Report | None, exit_codes: list[int], message: str) -> None: """Publish the final result without presenting recovered failures as current.""" summary = f"## Documentation links\n\n{message}\n\n| Attempt | Exit code |\n| --- | --- |\n" summary += "".join(f"| {attempt} | {code} |\n" for attempt, code in enumerate(exit_codes, start=1)) - if report_text is not None: - summary += f"\n### Final attempt\n\n{report_text}\n" + if data is not None: + summary += f"\n### Final attempt\n\n{render_report(data)}\n" else: summary += ( "\nThe final attempt did not produce a link-check report. Earlier reports are retained as artifacts.\n" ) + report.with_suffix(".md").write_text(summary, encoding="utf-8") print(message, flush=True) if summary_path := os.environ.get("GITHUB_STEP_SUMMARY"): with Path(summary_path).open("a", encoding="utf-8") as stream: @@ -49,59 +145,62 @@ def finish(report_text: str | None, exit_codes: list[int], message: str) -> None def retry_lychee(initial_exit_code: int, report: Path, max_attempts: int = 10) -> int: - """Count the action's first pass toward the limit, and retry only link failures.""" + """Retry only if every remaining failure is a recognized transient HTTP error.""" if initial_exit_code not in (0, 1, 2, 3): raise ValueError(f"Unexpected initial lychee exit code: {initial_exit_code}") if max_attempts < 1: raise ValueError("max_attempts must be positive") exit_codes = [initial_exit_code] if initial_exit_code not in (0, LINK_CHECK_FAILURE): - finish(None, exit_codes, f"Lychee stopped on attempt 1 with exit code {initial_exit_code}; no retry.") + finish(report, None, exit_codes, f"Lychee stopped on attempt 1 with exit code {initial_exit_code}; no retry.") return initial_exit_code - report_text = read_report(report) + data = read_report(report, initial_exit_code) shutil.copyfile(report, attempt_report(report, 1)) - if initial_exit_code == 0: - finish(report_text, exit_codes, "Lychee passed on attempt 1.") - return 0 - if max_attempts == 1: - finish(report_text, exit_codes, "Lychee still has broken links after 1 attempt.") - return LINK_CHECK_FAILURE - - raw_args = os.environ.get("LYCHEE_ARGS") - if not raw_args or not (lychee_args := shlex.split(raw_args)): - raise ValueError("LYCHEE_ARGS must contain the same arguments used by the first lychee pass") - - for attempt in range(2, max_attempts + 1): - cooldown = min(60 * (attempt - 1), 120) + lychee_args = None + while True: + attempt = len(exit_codes) + if exit_codes[-1] == 0: + finish(report, data, exit_codes, f"Lychee passed on attempt {attempt}/{max_attempts}.") + return 0 + if not data.transient: + finish( + report, + data, + exit_codes, + f"Lychee has permanent or unclassified link failures on attempt {attempt}; no retry.", + ) + return LINK_CHECK_FAILURE + if attempt >= max_attempts: + finish(report, data, exit_codes, f"Lychee still has transient link failures after {max_attempts} attempts.") + return LINK_CHECK_FAILURE + if lychee_args is None: + raw_args = os.environ.get("LYCHEE_ARGS") + if not raw_args or not (lychee_args := shlex.split(raw_args)): + raise ValueError("LYCHEE_ARGS must contain the same arguments used by the first lychee pass") + + cooldown = min(60 * attempt, 120) print( - f"Lychee attempt {attempt - 1} failed; waiting {cooldown}s before attempt {attempt}/{max_attempts}.", + f"Lychee attempt {attempt} had transient failures; waiting {cooldown}s before attempt {attempt + 1}/{max_attempts}.", flush=True, ) time.sleep(cooldown) - current_report = attempt_report(report, attempt) + current_report = attempt_report(report, attempt + 1) result = subprocess.run( # noqa: S603 - ["lychee", *lychee_args, "--mode", "task", "--format", "markdown", "--output", str(current_report)], # noqa: S607 + ["lychee", *lychee_args, "--mode", "task", "--format", "json", "--output", str(current_report)], # noqa: S607 check=False, ) exit_codes.append(result.returncode) if result.returncode not in (0, LINK_CHECK_FAILURE): finish( - None, exit_codes, f"Lychee stopped on attempt {attempt} with exit code {result.returncode}; no retry." + report, + None, + exit_codes, + f"Lychee stopped on attempt {attempt + 1} with exit code {result.returncode}; no retry.", ) return result.returncode if result.returncode > 0 else 1 - report_text = read_report(current_report) + data = read_report(current_report, result.returncode) shutil.copyfile(current_report, report) - if result.returncode == 0: - finish( - report_text, - exit_codes, - f"Lychee passed on attempt {attempt}/{max_attempts}; successful checks were retained between passes.", - ) - return 0 - - finish(report_text, exit_codes, f"Lychee still has broken links after {max_attempts} attempts.") - return LINK_CHECK_FAILURE def main() -> int: diff --git a/ci/tools/tests/test_retry_lychee.py b/ci/tools/tests/test_retry_lychee.py index b3e92adede2..2d00eeed80b 100644 --- a/ci/tools/tests/test_retry_lychee.py +++ b/ci/tools/tests/test_retry_lychee.py @@ -3,6 +3,7 @@ from __future__ import annotations +import json import subprocess import sys from pathlib import Path @@ -13,28 +14,52 @@ import retry_lychee -def markdown_report(total=2, errors=0): - return f"# Summary\n\n| Status | Count |\n| --- | --- |\n| \U0001f50d Total | {total} |\n| Error | {errors} |\n" +def response(url="https://example.test/guide", code=429, *, text="Request failed", details="Request timed out"): + status = {"text": text} + if code is None: + status["details"] = details + else: + status["code"] = code + return {"url": url, "status": status} + + +def report_data(*, errors=(), timeouts=(), total=5): + return { + "total": total, + "errors": len(errors), + "timeouts": len(timeouts), + "error_map": {"docs/index.html": list(errors)} if errors else {}, + "timeout_map": {"docs/index.html": list(timeouts)} if timeouts else {}, + } + + +def write_report(path, data): + path.write_text(json.dumps(data), encoding="utf-8") @pytest.mark.agent_authored(model="gpt-6") -def test_initial_success_never_runs_lychee_and_preserves_first_report(tmp_path, monkeypatch): - report = tmp_path / "lychee-authored.md" - report.write_text(markdown_report(), encoding="utf-8") +def test_initial_success_never_retries_and_preserves_json_and_final_markdown(tmp_path, monkeypatch): + report = tmp_path / "lychee-authored.json" + data = report_data() + write_report(report, data) summary = tmp_path / "summary.md" monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) assert retry_lychee.retry_lychee(0, report) == 0 - assert (tmp_path / "lychee-authored-attempt-1.md").read_text(encoding="utf-8") == markdown_report() - assert "passed on attempt 1" in summary.read_text(encoding="utf-8") + assert json.loads((tmp_path / "lychee-authored-attempt-1.json").read_text(encoding="utf-8")) == data + text = summary.read_text(encoding="utf-8") + assert "passed on attempt 1/10" in text + assert "| Total | 5 |" in text + assert report.with_suffix(".md").read_text(encoding="utf-8") == text @pytest.mark.agent_authored(model="gpt-6") def test_retries_reuse_cache_and_identical_arguments_until_success(tmp_path, monkeypatch): - report = tmp_path / "lychee-rendered.md" - report.write_text(markdown_report(errors=1), encoding="utf-8") + report = tmp_path / "lychee-rendered.json" + initial = report_data(errors=[response()]) + write_report(report, initial) cache = tmp_path / ".lycheecache" cache.write_text("first-success\n", encoding="utf-8") summary = tmp_path / "summary.md" @@ -50,12 +75,12 @@ def run(command, *, check): assert check is False commands.append(command) assert command[:6] == ["lychee", "--files-from", "input files.txt", "--cache", "--config", "link config.toml"] - assert command[6:10] == ["--mode", "task", "--format", "markdown"] + assert command[6:10] == ["--mode", "task", "--format", "json"] assert cache.read_text(encoding="utf-8") == "first-success\n" + ( "second-success\n" if len(commands) == 2 else "" ) code = 2 if len(commands) == 1 else 0 - Path(command[-1]).write_text(markdown_report(errors=int(code != 0)), encoding="utf-8") + write_report(Path(command[-1]), report_data(timeouts=[response(code=None)]) if code == 2 else report_data()) if code == 2: with cache.open("a", encoding="utf-8") as stream: stream.write("second-success\n") @@ -66,21 +91,88 @@ def run(command, *, check): assert retry_lychee.retry_lychee(2, report) == 0 assert delays == [60, 120] assert len(commands) == 2 - assert report.read_text(encoding="utf-8") == markdown_report() - assert (tmp_path / "lychee-rendered-attempt-1.md").read_text(encoding="utf-8") == markdown_report(errors=1) - assert (tmp_path / "lychee-rendered-attempt-2.md").read_text(encoding="utf-8") == markdown_report(errors=1) - assert (tmp_path / "lychee-rendered-attempt-3.md").read_text(encoding="utf-8") == markdown_report() + assert json.loads(report.read_text(encoding="utf-8")) == report_data() + assert json.loads((tmp_path / "lychee-rendered-attempt-1.json").read_text(encoding="utf-8")) == initial + assert json.loads((tmp_path / "lychee-rendered-attempt-2.json").read_text(encoding="utf-8"))["timeouts"] == 1 + assert json.loads((tmp_path / "lychee-rendered-attempt-3.json").read_text(encoding="utf-8")) == report_data() text = summary.read_text(encoding="utf-8") assert text.startswith("Earlier step\n") assert "passed on attempt 3/10" in text assert "| 1 | 2 |\n| 2 | 2 |\n| 3 | 0 |" in text - assert text.endswith(markdown_report() + "\n") + assert "example.test/guide" not in text + assert "| Errors | 0 |\n| Timeouts | 0 |" in text + + +@pytest.mark.parametrize("code", [408, 429, 500, 503, 599]) +@pytest.mark.agent_authored(model="gpt-6") +def test_known_transient_http_errors_are_retryable(tmp_path, code): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response(code=code)])) + + assert retry_lychee.read_report(report, 2).transient is True + + +@pytest.mark.parametrize("code", [None, 408, 200]) +@pytest.mark.agent_authored(model="gpt-6") +def test_http_timeout_map_entries_are_retryable(tmp_path, code): + report = tmp_path / "lychee.json" + write_report(report, report_data(timeouts=[response(code=code)])) + + assert retry_lychee.read_report(report, 2).transient is True + + +@pytest.mark.parametrize( + "failure", + [ + response(code=404), + response(code=403), + response(code=400), + response(code=600), + response(code=None, text="Network error", details="Connection refused"), + response("file:///docs/guide.html#missing", code=None, text="Missing fragment", details="Anchor not found"), + response("mailto:someone@example.test", code=429), + response("file:///docs/guide.html", code=503), + response("https:missing-host", code=503), + response("https://[malformed", code=503), + ], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_permanent_or_unknown_failures_stop_without_sleep_or_shared_arguments(tmp_path, monkeypatch, failure): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[failure])) + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report) == 2 + assert "permanent or unclassified" in report.with_suffix(".md").read_text(encoding="utf-8") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_local_timeouts_are_not_retryable(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + write_report(report, report_data(timeouts=[response("file:///docs/guide.html", code=None)])) + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report) == 2 @pytest.mark.agent_authored(model="gpt-6") -def test_ten_attempts_exhausted_remains_a_failure(tmp_path, monkeypatch): - report = tmp_path / "lychee.md" - report.write_text(markdown_report(errors=1), encoding="utf-8") +def test_mixed_permanent_and_transient_failures_stop_immediately(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response(), response(code=404)], timeouts=[response(code=None)])) + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report) == 2 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_transient_failure_turning_into_mixed_failures_stops_on_that_attempt(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response()])) + mixed = report_data(errors=[response(code=503), response(code=404)]) monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") delays = [] monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) @@ -88,7 +180,48 @@ def test_ten_attempts_exhausted_remains_a_failure(tmp_path, monkeypatch): def run(command, *, check): calls.append(command) - Path(command[-1]).write_text(markdown_report(errors=1), encoding="utf-8") + write_report(Path(command[-1]), mixed) + return subprocess.CompletedProcess(command, 2) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == 2 + assert len(calls) == 1 + assert delays == [60] + assert json.loads(report.read_text(encoding="utf-8")) == mixed + text = report.with_suffix(".md").read_text(encoding="utf-8") + assert "permanent or unclassified link failures on attempt 2" in text + assert "| 1 | 2 |\n| 2 | 2 |" in text + + +@pytest.mark.agent_authored(model="gpt-6") +def test_deduplicated_response_maps_do_not_require_counts_to_equal_entries(tmp_path): + report = tmp_path / "lychee.json" + data = report_data(errors=[response()], timeouts=[response(code=None)], total=100) + data.update(errors=27, timeouts=10) + write_report(report, data) + + result = retry_lychee.read_report(report, 2) + + assert result.transient is True + assert result.errors == 27 + assert result.timeouts == 10 + assert len(result.failures) == 2 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_ten_transient_attempts_exhausted_remains_a_failure(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + data = report_data(errors=[response()]) + write_report(report, data) + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + calls = [] + + def run(command, *, check): + calls.append(command) + write_report(Path(command[-1]), data) return subprocess.CompletedProcess(command, 2) monkeypatch.setattr(retry_lychee.subprocess, "run", run) @@ -96,7 +229,8 @@ def run(command, *, check): assert retry_lychee.retry_lychee(2, report) == 2 assert len(calls) == 9 assert delays == [60, *([120] * 8)] - assert len(list(tmp_path.glob("lychee-attempt-*.md"))) == 10 + assert len(list(tmp_path.glob("lychee-attempt-*.json"))) == 10 + assert "after 10 attempts" in report.with_suffix(".md").read_text(encoding="utf-8") @pytest.mark.parametrize("code", [1, 3]) @@ -105,16 +239,14 @@ def test_initial_non_link_failure_stops_without_retry(tmp_path, monkeypatch, cod monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) - assert retry_lychee.retry_lychee(code, tmp_path / "missing.md") == code + assert retry_lychee.retry_lychee(code, tmp_path / "missing.json") == code @pytest.mark.parametrize("code", [1, 3, -15]) @pytest.mark.agent_authored(model="gpt-6") -def test_retry_non_link_failure_stops_and_summary_does_not_claim_stale_report(tmp_path, monkeypatch, code): - report = tmp_path / "lychee.md" - report.write_text(markdown_report(errors=1), encoding="utf-8") - summary = tmp_path / "summary.md" - monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) +def test_retry_non_link_failure_does_not_claim_stale_report(tmp_path, monkeypatch, code): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response()])) monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") delays = [] monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) @@ -129,57 +261,125 @@ def run(command, *, check): assert retry_lychee.retry_lychee(2, report) == (code if code > 0 else 1) assert len(calls) == 1 assert delays == [60] - text = summary.read_text(encoding="utf-8") + text = report.with_suffix(".md").read_text(encoding="utf-8") assert "did not produce a link-check report" in text assert "### Final attempt" not in text - assert "Total |" not in text + assert "| Total |" not in text -@pytest.mark.parametrize("contents", ["", "not a report", markdown_report(total=0)]) +@pytest.mark.parametrize( + "name,value", [("total", True), ("errors", False), ("timeouts", -1), ("errors", 1.0), ("total", "5")] +) @pytest.mark.agent_authored(model="gpt-6") -def test_invalid_or_empty_initial_report_cannot_pass(tmp_path, monkeypatch, contents): - report = tmp_path / "lychee.md" +def test_invalid_counts_are_rejected(tmp_path, name, value): + report = tmp_path / "lychee.json" + data = report_data() + data[name] = value + write_report(report, data) + + with pytest.raises(ValueError, match="nonnegative integer"): + retry_lychee.read_report(report, 0) + + +@pytest.mark.parametrize( + "data", + [ + {}, + [], + report_data(total=0), + {**report_data(), "errors": 1}, + {**report_data(), "timeouts": 1}, + {**report_data(errors=[response()]), "errors": 0}, + {**report_data(timeouts=[response(code=None)]), "timeouts": 0}, + report_data(errors=[response()], timeouts=[response(code=None)], total=1), + {**report_data(errors=[response()]), "error_map": []}, + {**report_data(errors=[response()]), "error_map": {"source": response()}}, + {**report_data(errors=[response()]), "error_map": {"source": ["response"]}}, + { + **report_data(errors=[response()]), + "error_map": {"source": [{"url": 42, "status": {"text": "Error", "code": 429}}]}, + }, + report_data(errors=[{"url": "https://example.test/", "status": {"code": 429}}]), + report_data(errors=[response(code=True)]), + report_data(errors=[response(code="429")]), + report_data(errors=[response(code=None, details=None)]), + report_data(errors=[{"url": "https://example.test/", "status": {"text": "Unknown error"}}]), + ], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_malformed_or_contradictory_json_cannot_pass_or_trigger_retries(tmp_path, monkeypatch, data): + report = tmp_path / "lychee.json" + write_report(report, data) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + with pytest.raises(ValueError, match="Lychee"): + retry_lychee.retry_lychee(2, report) + + +@pytest.mark.parametrize( + "data,code", + [(report_data(), 2), (report_data(errors=[response()]), 0), (report_data(timeouts=[response(code=None)]), 0)], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_report_must_agree_with_exit_code(tmp_path, data, code): + report = tmp_path / "lychee.json" + write_report(report, data) + + with pytest.raises(ValueError, match="disagree with exit code"): + retry_lychee.retry_lychee(code, report) + + +@pytest.mark.parametrize("contents", ["", "not json", "{broken}"]) +@pytest.mark.agent_authored(model="gpt-6") +def test_invalid_json_cannot_pass(tmp_path, contents): + report = tmp_path / "lychee.json" report.write_text(contents, encoding="utf-8") - monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) - with pytest.raises(ValueError, match="no Total count|checked no links"): + with pytest.raises(json.JSONDecodeError): retry_lychee.retry_lychee(0, report) @pytest.mark.agent_authored(model="gpt-6") def test_missing_initial_report_cannot_pass(tmp_path): with pytest.raises(FileNotFoundError): - retry_lychee.retry_lychee(0, tmp_path / "missing.md") + retry_lychee.retry_lychee(0, tmp_path / "missing.json") @pytest.mark.agent_authored(model="gpt-6") -def test_successful_retry_with_zero_links_still_fails(tmp_path, monkeypatch): - report = tmp_path / "lychee.md" - report.write_text(markdown_report(errors=1), encoding="utf-8") - monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") - monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: None) +def test_final_markdown_escapes_untrusted_report_fields_and_retains_raw_json(tmp_path): + report = tmp_path / "lychee.json" + failure = response( + "file:///docs/x|[link].html", + code=None, + text="Missing ", + details="`a` *b* | c\n# injected", + ) + data = report_data(errors=[failure]) + data["error_map"] = {"docs/[source]|file\n# heading": [failure]} + write_report(report, data) - def run(command, *, check): - Path(command[-1]).write_text(markdown_report(total=0), encoding="utf-8") - return subprocess.CompletedProcess(command, 0) - - monkeypatch.setattr(retry_lychee.subprocess, "run", run) + assert retry_lychee.retry_lychee(2, report) == 2 - with pytest.raises(ValueError, match="checked no links"): - retry_lychee.retry_lychee(2, report) + text = report.with_suffix(".md").read_text(encoding="utf-8") + assert "