diff --git a/.github/workflows/build-docs.yml b/.github/workflows/build-docs.yml index 52ee9698ab0..f62d1ae5e9d 100644 --- a/.github/workflows/build-docs.yml +++ b/.github/workflows/build-docs.yml @@ -284,59 +284,6 @@ jobs: fi mv ${COMPONENT}/docs/build/html/* artifacts/docs/${TARGET} - - name: Write rendered docs file list - if: ${{ !inputs.is-release && startsWith(github.ref_name, 'pull-request/') }} - run: | - find "${GITHUB_WORKSPACE}/artifacts/docs" -type f -name '*.html' ! -path '*/_static/*' \ - | LC_ALL=C sort > lychee-rendered-html-files.txt - if [[ ! -s lychee-rendered-html-files.txt ]]; then - echo "error: no rendered HTML pages found for lychee" >&2 - exit 1 - fi - wc -l lychee-rendered-html-files.txt - - - name: Restore lychee cache - if: ${{ !inputs.is-release && startsWith(github.ref_name, 'pull-request/') }} - id: restore-lychee-cache - uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 - with: - path: .lycheecache - key: docs-rendered-lychee-${{ env.PR_NUMBER }}-${{ github.sha }} - restore-keys: | - docs-rendered-lychee-${{ env.PR_NUMBER }}- - - - name: Check rendered docs links - if: ${{ !inputs.is-release && startsWith(github.ref_name, 'pull-request/') }} - uses: lycheeverse/lychee-action@6da1d14f3a43098a294b7696d93d938aa8d20fc0 # unreleased: supports v0.24.x archive layout - with: - args: >- - --files-from ${{ github.workspace }}/lychee-rendered-html-files.txt - --include-fragments=full - --cache - --max-cache-age 1d - --max-concurrency 16 - --host-concurrency 2 - --host-request-interval 250ms - --max-retries 3 - --retry-wait-time 5 - --timeout 30 - --no-progress - --config ${{ github.workspace }}/lychee.toml - fail: true - failIfEmpty: true - format: markdown - jobSummary: false - lycheeVersion: v0.24.2 - output: lychee-rendered-html.md - token: ${{ github.token }} - - - name: Save lychee cache - if: ${{ always() && !inputs.is-release && startsWith(github.ref_name, 'pull-request/') && steps.restore-lychee-cache.outputs.cache-hit != 'true' && steps.restore-lychee-cache.outputs.cache-primary-key != '' }} - uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 - with: - path: .lycheecache - key: ${{ steps.restore-lychee-cache.outputs.cache-primary-key }} - - name: Upload docs GitHub Pages artifact if: ${{ inputs.deploy-docs }} uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5.0.0 diff --git a/.github/workflows/ci-nightly.yml b/.github/workflows/ci-nightly.yml index a3179c1e155..d033ba1d251 100644 --- a/.github/workflows/ci-nightly.yml +++ b/.github/workflows/ci-nightly.yml @@ -13,7 +13,7 @@ name: "CI: Nightly optional-deps" concurrency: - group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name }} + group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event_name }}-${{ inputs.documentation-links-only || false }} cancel-in-progress: true on: @@ -24,6 +24,10 @@ on: - cron: "17 2 * * *" workflow_dispatch: inputs: + documentation-links-only: + description: "Test the nightly link checker without wheel or GPU jobs" + type: boolean + default: false run-id: description: > Override the CI run ID to download artifacts from. @@ -32,6 +36,16 @@ on: default: '' jobs: + documentation-links: + name: "Nightly: Documentation links" + if: ${{ github.repository_owner == 'nvidia' }} + permissions: + contents: read + uses: ./.github/workflows/lychee.yml + with: + concurrency-suffix: ${{ inputs.documentation-links-only && 'links-only' || 'full' }} + refresh-cache: true + test-ci-tools-for-release: name: "Nightly: CI tools for release" if: ${{ github.repository_owner == 'nvidia' }} @@ -49,7 +63,7 @@ jobs: python -m pytest -v --noconftest ci/tools/tests find-wheels: - if: ${{ github.repository_owner == 'nvidia' }} + if: ${{ github.repository_owner == 'nvidia' && !inputs.documentation-links-only }} runs-on: ubuntu-latest outputs: RUN_ID: ${{ steps.find.outputs.run_id }} @@ -313,6 +327,7 @@ jobs: if: ${{ always() && github.repository_owner == 'nvidia' }} runs-on: ubuntu-latest needs: + - documentation-links - test-ci-tools-for-release - find-wheels - test-pytorch-linux @@ -328,13 +343,27 @@ jobs: - test-standard-linux-aarch64 steps: - name: Exit + env: + LINKS_ONLY: ${{ inputs.documentation-links-only || false }} + NEEDS_JSON: ${{ toJSON(needs) }} run: | + if [[ "${LINKS_ONLY}" == "true" ]]; then + # GPU jobs depend on find-wheels and must stay skipped in this mode. + jq -e 'all(to_entries[]; + if .key == "documentation-links" or .key == "test-ci-tools-for-release" + then .value.result == "success" + else .value.result == "skipped" + end)' <<< "${NEEDS_JSON}" + exit 0 + fi + # If any dependency was cancelled or failed, that's a failure. # # See ci.yml for the full rationale on why we must use always() # and explicitly check each result rather than relying on the # default behaviour. - if ${{ needs.test-ci-tools-for-release.result == 'cancelled' || + if ${{ needs.documentation-links.result != 'success' || + needs.test-ci-tools-for-release.result == 'cancelled' || needs.test-ci-tools-for-release.result == 'failure' || needs.find-wheels.result != 'success' }}; then exit 1 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 313be5ff77e..257aa480d2a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -622,38 +622,6 @@ jobs: with: is-release: ${{ github.ref_type == 'tag' }} - precommit-windows: - name: Pre-commit on Windows - runs-on: windows-latest - if: ${{ github.repository_owner == 'nvidia' && !fromJSON(needs.should-skip.outputs.skip) }} - needs: - - should-skip - permissions: - contents: read - steps: - - name: Checkout repository - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - fetch-depth: 1 - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - with: - python-version: '3.13' - - - name: Install pre-commit - shell: bash - run: | - set -euxo pipefail - python -m pip install --upgrade pip pre-commit - - - name: Run pre-commit - shell: bash - run: | - set -euxo pipefail - SKIP=lychee pre-commit run --all-files - checks: name: Check job status if: ${{ always() && github.repository_owner == 'nvidia' }} @@ -674,7 +642,6 @@ jobs: - api-check-core-vs-release - api-check-core-vs-base - doc - - precommit-windows steps: - name: Exit env: @@ -712,14 +679,12 @@ jobs: fi } - # Control jobs, the universal linux build, docs, and Windows - # pre-commit checks always run. + # Control jobs, the universal Linux build, and docs always run. check_result "ci-vars" "success" check_result "should-skip" "success" check_result "detect-changes" "success" check_result "build-linux-64" "success" check_result "doc" "success" - check_result "precommit-windows" "success" # Optional platform builds and wheel tests share the platform plan. linux_expected="skipped" diff --git a/.github/workflows/lychee.yml b/.github/workflows/lychee.yml new file mode 100644 index 00000000000..4ebd0473e1f --- /dev/null +++ b/.github/workflows/lychee.yml @@ -0,0 +1,177 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# SPDX-License-Identifier: Apache-2.0 + +# Check checked-in documentation and rendered HTML on PR updates and nightly. +# Build fresh HTML from sources to check Sphinx-generated links and anchors. +# This runs independently of ci.yml → build-docs.yml, which builds and deploys +# docs using CI-built wheels. Nightly caches successful link-check results for +# reuse by PR runs of this workflow. + +name: "CI: Documentation links" + +on: + pull_request: + types: [opened, reopened, synchronize] + workflow_call: + inputs: + concurrency-suffix: + description: "Keep a caller's test mode separate from its full run" + type: string + default: full + refresh-cache: + description: "Check links afresh and publish a new cache baseline" + type: boolean + default: false + workflow_dispatch: + inputs: + refresh-cache: + description: "Check links afresh and publish a new cache baseline" + type: boolean + default: false + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-${{ github.event_name }}-${{ inputs.refresh-cache || false }}-${{ inputs.concurrency-suffix || 'full' }} + cancel-in-progress: true + +defaults: + run: + shell: bash --noprofile --norc -euo pipefail {0} + +env: + PIXI_LOCKED: "true" + LYCHEE_VERSION: v0.24.2 # Must match the pinned rev in .pre-commit-config.yaml + +jobs: + links: + name: Lychee (${{ matrix.kind }}) + runs-on: ubuntu-latest + timeout-minutes: 60 + strategy: + fail-fast: false + matrix: + kind: [authored, rendered] + env: + # The first pass and all retries use the same inputs and cache policy. + LYCHEE_ARGS: >- + --files-from "${{ github.workspace }}/lychee-files.txt" + ${{ matrix.kind == 'rendered' && '--include-fragments=full' || '' }} + --cache + --max-cache-age 1d + --max-concurrency 16 + --host-concurrency 2 + --host-request-interval 250ms + --max-retries 3 + --retry-wait-time 5 + --timeout 30 + --no-progress + --config "${{ github.workspace }}/lychee.toml" + steps: + - name: Checkout checked revision + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.event.pull_request.head.sha || github.sha }} + fetch-depth: 0 + # Keep SCM history/tags without downloading historical source blobs. + filter: blob:none + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.14" + + - name: Setup pixi + if: ${{ matrix.kind == 'rendered' }} + uses: ./.github/actions/setup-pixi + + - name: Build all documentation from sources + if: ${{ matrix.kind == 'rendered' }} + run: | + CUDA_PYTHON_DOCS_GITHUB_REF=$(git rev-parse HEAD) + export CUDA_PYTHON_DOCS_GITHUB_REF + pixi run --manifest-path cuda_core/pixi.toml -e docs docs-build-all-latest + + - name: Prepare link-check inputs and cache policy + id: inputs + env: + KIND: ${{ matrix.kind }} + POLICY_HASH: ${{ hashFiles('lychee.toml', '.github/workflows/lychee.yml', 'ci/tools/prepare_lychee_inputs.py', 'ci/tools/retry_lychee.py') }} + run: | + python ci/tools/prepare_lychee_inputs.py "${KIND}" \ + --output "${GITHUB_WORKSPACE}/lychee-files.txt" + prefix="lychee-v1-${LYCHEE_VERSION}-${KIND}-${POLICY_HASH}-baseline-" + echo "prefix=${prefix}" >> "${GITHUB_OUTPUT}" + echo "key=${prefix}${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" >> "${GITHUB_OUTPUT}" + + - name: Restore successful checks from the nightly baseline + if: ${{ !inputs.refresh-cache }} + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: .lycheecache + key: ${{ steps.inputs.outputs.key }} + restore-keys: ${{ steps.inputs.outputs.prefix }} + + - name: Check documentation links + id: lychee + uses: lycheeverse/lychee-action@6da1d14f3a43098a294b7696d93d938aa8d20fc0 # supports v0.24.x archive layout + with: + args: ${{ env.LYCHEE_ARGS }} + # The next step owns bounded retries and the final success gate. + fail: false + # The action's empty-report guard only understands Markdown. + # Our helper validates the JSON report and rejects zero checked links. + failIfEmpty: false + format: json + jobSummary: false + lycheeVersion: ${{ env.LYCHEE_VERSION }} + output: lychee-${{ matrix.kind }}.json + token: ${{ github.token }} + + - name: Retry transient link failures using cached successful checks + if: ${{ steps.lychee.outcome == 'success' }} + env: + EXIT_CODE: ${{ steps.lychee.outputs.exit_code }} + REPORT: lychee-${{ matrix.kind }}.json + GITHUB_TOKEN: ${{ github.token }} + run: | + python ci/tools/retry_lychee.py --initial-exit-code "${EXIT_CODE}" \ + --report "${REPORT}" --max-attempts 10 + + - name: Publish fresh successful checks + # Manual branch tests publish isolated branch caches. Nightly callers + # on the default branch publish the baseline that all PRs can restore. + if: ${{ always() && inputs.refresh-cache && steps.inputs.outputs.key != '' && hashFiles('.lycheecache') != '' }} + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: .lycheecache + key: ${{ steps.inputs.outputs.key }} + + - name: Upload link-check report + if: ${{ always() }} + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: lychee-${{ matrix.kind }}-${{ github.run_id }}-${{ github.run_attempt }} + path: | + lychee-${{ matrix.kind }}*.json + lychee-${{ matrix.kind }}*.md + if-no-files-found: ignore + retention-days: 7 + + checks: + name: Documentation links + if: ${{ always() }} + needs: links + runs-on: ubuntu-latest + steps: + - name: Require both link checks to succeed + env: + RESULT: ${{ needs.links.result }} + run: | + if [[ "${RESULT}" != "success" ]]; then + echo "::error::Documentation link checks did not succeed (${RESULT})." + exit 1 + fi diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 8b7b28d475e..bff02659125 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -214,7 +214,9 @@ A few things to keep in mind: --config-file cuda_core/pyproject.toml ## Pre-commit -This project uses [pre-commit.ci](https://pre-commit.ci/) with GitHub Actions. All pull requests are automatically checked for pre-commit compliance, and any pre-commit failures will block merging until resolved. +GitHub Actions runs pre-commit checks on Linux and Windows and checks +documentation links for pull requests. Check failures must be resolved +before merging. To set yourself up for running pre-commit checks locally and to catch issues before pushing your changes, follow these steps: @@ -229,8 +231,8 @@ Installing the hook is required, not optional. Some of the automated checks keep the tree consistent if they run on *every* commit. Relying on manual `pre-commit run --all-files` invocations means these checks can be skipped between commits, leaving stale headers or out-of-date stubs in the history. -If the hook isn't installed, `pre-commit run` (and CI) will print a visible -warning reminding you to run `pre-commit install`. +If the hook isn't installed, `pre-commit run` will print a visible warning +reminding you to run `pre-commit install`. Windows contributors: see [Pre-commit lychee workaround](#pre-commit-lychee-workaround) under Development on Windows. diff --git a/ci/tools/build_docs_for_link_check.sh b/ci/tools/build_docs_for_link_check.sh new file mode 100644 index 00000000000..26f482b2580 --- /dev/null +++ b/ci/tools/build_docs_for_link_check.sh @@ -0,0 +1,56 @@ +#!/usr/bin/env bash + +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Run through cuda_core's docs environment, which builds all three sibling +# packages from this checkout rather than installing their published releases. +set -euo pipefail + +REPO_ROOT=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd) +cd "${REPO_ROOT}" + +if [[ "${PIXI_ENVIRONMENT_NAME:-}" != "docs" ]]; then + echo 'Run with: pixi run --manifest-path cuda_core/pixi.toml -e docs docs-build-all-latest' >&2 + exit 1 +fi + +export CUDA_PYTHON_DOCS_GITHUB_REF="${CUDA_PYTHON_DOCS_GITHUB_REF:-$(git rev-parse HEAD)}" +export BUILD_LATEST=1 +export BUILD_PREVIEW=0 + +# The metapackage pins released cuda-core versions. Its dependencies are already +# installed from local Pixi source packages, so never resolve those pins on PyPI. +python -m pip install --no-deps "${REPO_ROOT}/cuda_python" +python - <<'PY' +from importlib import import_module +from importlib.metadata import version +from pathlib import Path + +for distribution, module_name in ( + ("cuda-bindings", "cuda.bindings"), + ("cuda-core", "cuda.core"), + ("cuda-pathfinder", "cuda.pathfinder"), +): + module = import_module(module_name) + source = Path.cwd() / distribution.replace("-", "_") + if not Path(module.__file__).resolve().is_relative_to(source): + raise RuntimeError(f"{distribution} must be imported from the checkout: {module.__file__}") + installed_version = version(distribution) + if module.__version__ != installed_version: + raise RuntimeError( + f"{distribution} import version {module.__version__} differs from metadata {installed_version}" + ) + print(f"{distribution}: {installed_version} ({module.__file__})") +print(f"cuda-python: {version('cuda-python')}") +PY + +pushd cuda_python/docs >/dev/null +# The docs Makefiles use O as an optional Sphinx argument. Ignore an unrelated +# host variable of that name so Sphinx does not treat it as an input filename. +env -u O bash ./build_all_docs.sh latest-only +popd >/dev/null + +rm -rf artifacts/docs +mkdir -p artifacts/docs +cp -a cuda_python/docs/build/html/. artifacts/docs/ diff --git a/ci/tools/prepare_lychee_inputs.py b/ci/tools/prepare_lychee_inputs.py new file mode 100644 index 00000000000..9c723398016 --- /dev/null +++ b/ci/tools/prepare_lychee_inputs.py @@ -0,0 +1,70 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Write deterministic input lists for authored or rendered documentation links.""" + +from __future__ import annotations + +import argparse +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] + + +def authored_inputs(root: Path) -> list[Path]: + """Select tracked Markdown and reStructuredText outside qa/, skipping symlinks.""" + result = subprocess.run( + ["git", "ls-files", "-z", "--", "*.md", "*.rst"], # noqa: S607 + cwd=root, + check=True, + capture_output=True, + ) + paths = [] + for name in result.stdout.decode("utf-8").split("\0"): + if not name or name.startswith("qa/"): + continue + path = root / name + if path.is_file() and not path.is_symlink(): + paths.append(path) + return sorted(paths) + + +def rendered_inputs(docs_root: Path) -> list[Path]: + """Select built HTML, excluding theme assets under _static directories.""" + return sorted( + path + for path in docs_root.rglob("*.html") + if path.is_file() and "_static" not in path.relative_to(docs_root).parts + ) + + +def write_inputs(paths: list[Path], output: Path) -> None: + """Reject an empty or ambiguous file list before invoking lychee.""" + if not paths: + raise ValueError("No documentation inputs found for lychee") + lines = [str(path.absolute()) for path in paths] + if any("\n" in line or "\r" in line for line in lines): + raise ValueError("Lychee input paths must not contain newlines") + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text("\n".join(lines) + "\n", encoding="utf-8") + print(f"Wrote {len(lines)} lychee inputs to {output}") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("kind", choices=("authored", "rendered")) + parser.add_argument("--repo-root", type=Path, default=ROOT) + parser.add_argument("--docs-root", type=Path, default=Path("artifacts/docs")) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + root = args.repo_root.resolve() + if args.kind == "authored": + paths = authored_inputs(root) + else: + paths = rendered_inputs((root / args.docs_root).absolute()) + write_inputs(paths, args.output) + + +if __name__ == "__main__": + main() diff --git a/ci/tools/retry_lychee.py b/ci/tools/retry_lychee.py new file mode 100644 index 00000000000..b9cbf5f06c7 --- /dev/null +++ b/ci/tools/retry_lychee.py @@ -0,0 +1,220 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Retry transient lychee failures while retaining successful checks in its cache.""" + +from __future__ import annotations + +import argparse +import html +import json +import os +import re +import shlex +import shutil +import subprocess +import time +from dataclasses import dataclass +from pathlib import Path +from urllib.parse import urlsplit + +LINK_CHECK_FAILURE = 2 + + +@dataclass(frozen=True) +class Failure: + source: str + url: str + text: str + details: str | None + code: int | None + timed_out: bool + + @property + def transient(self) -> bool: + try: + parsed = urlsplit(self.url) + is_http = parsed.scheme in ("http", "https") and bool(parsed.hostname) + except ValueError: + return False + return is_http and ( + self.timed_out or self.code in (408, 429) or (self.code is not None and 500 <= self.code <= 599) + ) + + +@dataclass(frozen=True) +class Report: + total: int + errors: int + timeouts: int + failures: tuple[Failure, ...] + + @property + def transient(self) -> bool: + return bool(self.failures) and all(failure.transient for failure in self.failures) + + +def read_report(path: Path, exit_code: int) -> Report: + """Validate the pinned lychee JSON schema and its agreement with the exit code.""" + data = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise ValueError(f"Lychee report must be a JSON object: {path}") + counts = {} + for name in ("total", "errors", "timeouts"): + value = data.get(name) + if type(value) is not int or value < 0: + raise ValueError(f"Lychee report {name} must be a nonnegative integer: {path}") + counts[name] = value + if counts["total"] == 0: + raise ValueError(f"Lychee checked no links: {path}") + if counts["errors"] + counts["timeouts"] > counts["total"]: + raise ValueError(f"Lychee failure counts exceed the total: {path}") + + failures = [] + for name, timed_out, count in (("error_map", False, counts["errors"]), ("timeout_map", True, counts["timeouts"])): + mapping = data.get(name) + if not isinstance(mapping, dict): + raise ValueError(f"Lychee report {name} must be an object: {path}") + entries = [] + for source, responses in mapping.items(): + if not isinstance(source, str) or not source or not isinstance(responses, list): + raise ValueError(f"Lychee report {name} must map source names to response lists: {path}") + for response in responses: + if not isinstance(response, dict) or not isinstance(response.get("url"), str) or not response["url"]: + raise ValueError(f"Lychee report has a malformed failure URL: {path}") + status = response.get("status") + if not isinstance(status, dict) or not isinstance(status.get("text"), str) or not status["text"]: + raise ValueError(f"Lychee report has a malformed failure status: {path}") + code = status.get("code") + details = status.get("details") + if "code" in status: + if type(code) is not int or not 100 <= code <= 999: + raise ValueError(f"Lychee report has a malformed HTTP status code: {path}") + elif not isinstance(details, str): + raise ValueError(f"Lychee report non-HTTP status must contain string details: {path}") + if "details" in status and not isinstance(details, str): + raise ValueError(f"Lychee report status details must be a string: {path}") + entries.append(Failure(source, response["url"], status["text"], details, code, timed_out)) + # Counters count occurrences; response maps deduplicate equivalent checks. + if bool(count) != bool(entries): + raise ValueError(f"Lychee report {name} disagrees with its failure count: {path}") + failures.extend(entries) + + if (exit_code == 0) != (not failures): + raise ValueError(f"Lychee report failures disagree with exit code {exit_code}: {path}") + return Report(counts["total"], counts["errors"], counts["timeouts"], tuple(failures)) + + +def escape_markdown(text: str) -> str: + text = html.escape(text).replace("\r", " ").replace("\n", " ") + return re.sub(r"([\\`*_~\[\]|])", r"\\\1", text) + + +def render_report(report: Report) -> str: + result = ( + "| Status | Count |\n| --- | --- |\n" + f"| Total | {report.total} |\n| Errors | {report.errors} |\n| Timeouts | {report.timeouts} |\n" + ) + if report.failures: + result += "\n| Source | URL | Status | Details |\n| --- | --- | --- | --- |\n" + for failure in report.failures: + cells = (failure.source, failure.url, failure.text, failure.details or "") + result += "| " + " | ".join(escape_markdown(cell) for cell in cells) + " |\n" + return result + + +def attempt_report(report: Path, attempt: int) -> Path: + return report.with_name(f"{report.stem}-attempt-{attempt}{report.suffix}") + + +def finish(report: Path, data: Report | None, exit_codes: list[int], message: str) -> None: + """Publish the final result without presenting recovered failures as current.""" + summary = f"## Documentation links\n\n{message}\n\n| Attempt | Exit code |\n| --- | --- |\n" + summary += "".join(f"| {attempt} | {code} |\n" for attempt, code in enumerate(exit_codes, start=1)) + if data is not None: + summary += f"\n### Final attempt\n\n{render_report(data)}\n" + else: + summary += ( + "\nThe final attempt did not produce a link-check report. Earlier reports are retained as artifacts.\n" + ) + report.with_suffix(".md").write_text(summary, encoding="utf-8") + print(message, flush=True) + if summary_path := os.environ.get("GITHUB_STEP_SUMMARY"): + with Path(summary_path).open("a", encoding="utf-8") as stream: + stream.write(summary) + + +def retry_lychee(initial_exit_code: int, report: Path, max_attempts: int = 10) -> int: + """Retry only if every remaining failure is a recognized transient HTTP error.""" + if initial_exit_code not in (0, 1, 2, 3): + raise ValueError(f"Unexpected initial lychee exit code: {initial_exit_code}") + if max_attempts < 1: + raise ValueError("max_attempts must be positive") + exit_codes = [initial_exit_code] + if initial_exit_code not in (0, LINK_CHECK_FAILURE): + finish(report, None, exit_codes, f"Lychee stopped on attempt 1 with exit code {initial_exit_code}; no retry.") + return initial_exit_code + + data = read_report(report, initial_exit_code) + shutil.copyfile(report, attempt_report(report, 1)) + lychee_args = None + while True: + attempt = len(exit_codes) + if exit_codes[-1] == 0: + finish(report, data, exit_codes, f"Lychee passed on attempt {attempt}/{max_attempts}.") + return 0 + if not data.transient: + finish( + report, + data, + exit_codes, + f"Lychee has permanent or unclassified link failures on attempt {attempt}; no retry.", + ) + return LINK_CHECK_FAILURE + if attempt >= max_attempts: + finish(report, data, exit_codes, f"Lychee still has transient link failures after {max_attempts} attempts.") + return LINK_CHECK_FAILURE + if lychee_args is None: + raw_args = os.environ.get("LYCHEE_ARGS") + if not raw_args or not (lychee_args := shlex.split(raw_args)): + raise ValueError("LYCHEE_ARGS must contain the same arguments used by the first lychee pass") + + cooldown = min(60 * attempt, 120) + print( + f"Lychee attempt {attempt} had transient failures; waiting {cooldown}s before attempt {attempt + 1}/{max_attempts}.", + flush=True, + ) + time.sleep(cooldown) + current_report = attempt_report(report, attempt + 1) + result = subprocess.run( # noqa: S603 + ["lychee", *lychee_args, "--mode", "task", "--format", "json", "--output", str(current_report)], # noqa: S607 + check=False, + ) + exit_codes.append(result.returncode) + if result.returncode not in (0, LINK_CHECK_FAILURE): + finish( + report, + None, + exit_codes, + f"Lychee stopped on attempt {attempt + 1} with exit code {result.returncode}; no retry.", + ) + return result.returncode if result.returncode > 0 else 1 + data = read_report(current_report, result.returncode) + shutil.copyfile(current_report, report) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--initial-exit-code", type=int, required=True) + parser.add_argument("--report", type=Path, required=True) + parser.add_argument("--max-attempts", type=int, default=10) + args = parser.parse_args() + try: + return retry_lychee(args.initial_exit_code, args.report, args.max_attempts) + except (OSError, ValueError) as error: + print(f"::error::{error}", flush=True) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/ci/tools/tests/test_prepare_lychee_inputs.py b/ci/tools/tests/test_prepare_lychee_inputs.py new file mode 100644 index 00000000000..15bfb945dc2 --- /dev/null +++ b/ci/tools/tests/test_prepare_lychee_inputs.py @@ -0,0 +1,83 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import subprocess +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).parent.parent)) +from prepare_lychee_inputs import authored_inputs, rendered_inputs, write_inputs + + +@pytest.mark.agent_authored(model="gpt-6") +def test_authored_inputs_use_tracked_documents_and_exclude_qa(tmp_path): + subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True) # noqa: S603, S607 + tracked = ["README.md", "docs/a guide.rst", "qa/README.md", "src/worker.py", "gone.md"] + for name in tracked: + path = tmp_path / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("content", encoding="utf-8") + subprocess.run(["git", "add", "--", *tracked], cwd=tmp_path, check=True) # noqa: S603, S607 + (tmp_path / "gone.md").unlink() + (tmp_path / "untracked.md").write_text("content", encoding="utf-8") + + assert authored_inputs(tmp_path) == [tmp_path / "README.md", tmp_path / "docs/a guide.rst"] + + +@pytest.mark.agent_authored(model="gpt-6") +def test_authored_inputs_skip_tracked_symlinks_and_keep_their_real_target(tmp_path): + subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True) # noqa: S603, S607 + target = tmp_path / "README.md" + target.write_text("[Guide](docs/guide.rst)\n", encoding="utf-8") + package = tmp_path / "cuda_python" + package.mkdir() + (package / "README.md").symlink_to("../README.md") + subprocess.run(["git", "add", "--", "README.md", "cuda_python/README.md"], cwd=tmp_path, check=True) # noqa: S607 + + assert authored_inputs(tmp_path) == [target] + + +@pytest.mark.agent_authored(model="gpt-6") +def test_rendered_inputs_include_all_components_and_exclude_static_assets(tmp_path): + expected = ["cuda-bindings/latest/api.html", "cuda-core/latest/guide.html", "latest/index.html"] + for name in [*expected, "cuda-core/latest/_static/theme.html", "_static/vendor/embed.html", "versions.json"]: + path = tmp_path / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("content", encoding="utf-8") + (tmp_path / "directory.html").mkdir() + + assert rendered_inputs(tmp_path) == [tmp_path / name for name in expected] + + +@pytest.mark.agent_authored(model="gpt-6") +def test_write_inputs_keeps_spaces_and_writes_absolute_paths(tmp_path): + output = tmp_path / "lists/files.txt" + paths = [tmp_path / "docs/a guide.rst", tmp_path / "README.md"] + + write_inputs(paths, output) + + assert output.read_text(encoding="utf-8") == "".join(f"{path}\n" for path in paths) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_write_inputs_rejects_empty_inputs_without_publishing_list(tmp_path): + output = tmp_path / "files.txt" + + with pytest.raises(ValueError, match="No documentation inputs found"): + write_inputs([], output) + + assert not output.exists() + + +@pytest.mark.agent_authored(model="gpt-6") +def test_write_inputs_rejects_paths_that_would_split_into_multiple_inputs(tmp_path): + output = tmp_path / "files.txt" + + with pytest.raises(ValueError, match="must not contain newlines"): + write_inputs([tmp_path / "two\ninputs.md"], output) + + assert not output.exists() diff --git a/ci/tools/tests/test_retry_lychee.py b/ci/tools/tests/test_retry_lychee.py new file mode 100644 index 00000000000..2d00eeed80b --- /dev/null +++ b/ci/tools/tests/test_retry_lychee.py @@ -0,0 +1,397 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).parent.parent)) +import retry_lychee + + +def response(url="https://example.test/guide", code=429, *, text="Request failed", details="Request timed out"): + status = {"text": text} + if code is None: + status["details"] = details + else: + status["code"] = code + return {"url": url, "status": status} + + +def report_data(*, errors=(), timeouts=(), total=5): + return { + "total": total, + "errors": len(errors), + "timeouts": len(timeouts), + "error_map": {"docs/index.html": list(errors)} if errors else {}, + "timeout_map": {"docs/index.html": list(timeouts)} if timeouts else {}, + } + + +def write_report(path, data): + path.write_text(json.dumps(data), encoding="utf-8") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_initial_success_never_retries_and_preserves_json_and_final_markdown(tmp_path, monkeypatch): + report = tmp_path / "lychee-authored.json" + data = report_data() + write_report(report, data) + summary = tmp_path / "summary.md" + monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(0, report) == 0 + assert json.loads((tmp_path / "lychee-authored-attempt-1.json").read_text(encoding="utf-8")) == data + text = summary.read_text(encoding="utf-8") + assert "passed on attempt 1/10" in text + assert "| Total | 5 |" in text + assert report.with_suffix(".md").read_text(encoding="utf-8") == text + + +@pytest.mark.agent_authored(model="gpt-6") +def test_retries_reuse_cache_and_identical_arguments_until_success(tmp_path, monkeypatch): + report = tmp_path / "lychee-rendered.json" + initial = report_data(errors=[response()]) + write_report(report, initial) + cache = tmp_path / ".lycheecache" + cache.write_text("first-success\n", encoding="utf-8") + summary = tmp_path / "summary.md" + summary.write_text("Earlier step\n", encoding="utf-8") + monkeypatch.chdir(tmp_path) + monkeypatch.setenv("GITHUB_STEP_SUMMARY", str(summary)) + monkeypatch.setenv("LYCHEE_ARGS", '--files-from "input files.txt" --cache --config "link config.toml"') + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + commands = [] + + def run(command, *, check): + assert check is False + commands.append(command) + assert command[:6] == ["lychee", "--files-from", "input files.txt", "--cache", "--config", "link config.toml"] + assert command[6:10] == ["--mode", "task", "--format", "json"] + assert cache.read_text(encoding="utf-8") == "first-success\n" + ( + "second-success\n" if len(commands) == 2 else "" + ) + code = 2 if len(commands) == 1 else 0 + write_report(Path(command[-1]), report_data(timeouts=[response(code=None)]) if code == 2 else report_data()) + if code == 2: + with cache.open("a", encoding="utf-8") as stream: + stream.write("second-success\n") + return subprocess.CompletedProcess(command, code) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == 0 + assert delays == [60, 120] + assert len(commands) == 2 + assert json.loads(report.read_text(encoding="utf-8")) == report_data() + assert json.loads((tmp_path / "lychee-rendered-attempt-1.json").read_text(encoding="utf-8")) == initial + assert json.loads((tmp_path / "lychee-rendered-attempt-2.json").read_text(encoding="utf-8"))["timeouts"] == 1 + assert json.loads((tmp_path / "lychee-rendered-attempt-3.json").read_text(encoding="utf-8")) == report_data() + text = summary.read_text(encoding="utf-8") + assert text.startswith("Earlier step\n") + assert "passed on attempt 3/10" in text + assert "| 1 | 2 |\n| 2 | 2 |\n| 3 | 0 |" in text + assert "example.test/guide" not in text + assert "| Errors | 0 |\n| Timeouts | 0 |" in text + + +@pytest.mark.parametrize("code", [408, 429, 500, 503, 599]) +@pytest.mark.agent_authored(model="gpt-6") +def test_known_transient_http_errors_are_retryable(tmp_path, code): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response(code=code)])) + + assert retry_lychee.read_report(report, 2).transient is True + + +@pytest.mark.parametrize("code", [None, 408, 200]) +@pytest.mark.agent_authored(model="gpt-6") +def test_http_timeout_map_entries_are_retryable(tmp_path, code): + report = tmp_path / "lychee.json" + write_report(report, report_data(timeouts=[response(code=code)])) + + assert retry_lychee.read_report(report, 2).transient is True + + +@pytest.mark.parametrize( + "failure", + [ + response(code=404), + response(code=403), + response(code=400), + response(code=600), + response(code=None, text="Network error", details="Connection refused"), + response("file:///docs/guide.html#missing", code=None, text="Missing fragment", details="Anchor not found"), + response("mailto:someone@example.test", code=429), + response("file:///docs/guide.html", code=503), + response("https:missing-host", code=503), + response("https://[malformed", code=503), + ], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_permanent_or_unknown_failures_stop_without_sleep_or_shared_arguments(tmp_path, monkeypatch, failure): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[failure])) + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report) == 2 + assert "permanent or unclassified" in report.with_suffix(".md").read_text(encoding="utf-8") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_local_timeouts_are_not_retryable(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + write_report(report, report_data(timeouts=[response("file:///docs/guide.html", code=None)])) + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report) == 2 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_mixed_permanent_and_transient_failures_stop_immediately(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response(), response(code=404)], timeouts=[response(code=None)])) + monkeypatch.delenv("LYCHEE_ARGS", raising=False) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(2, report) == 2 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_transient_failure_turning_into_mixed_failures_stops_on_that_attempt(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response()])) + mixed = report_data(errors=[response(code=503), response(code=404)]) + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + calls = [] + + def run(command, *, check): + calls.append(command) + write_report(Path(command[-1]), mixed) + return subprocess.CompletedProcess(command, 2) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == 2 + assert len(calls) == 1 + assert delays == [60] + assert json.loads(report.read_text(encoding="utf-8")) == mixed + text = report.with_suffix(".md").read_text(encoding="utf-8") + assert "permanent or unclassified link failures on attempt 2" in text + assert "| 1 | 2 |\n| 2 | 2 |" in text + + +@pytest.mark.agent_authored(model="gpt-6") +def test_deduplicated_response_maps_do_not_require_counts_to_equal_entries(tmp_path): + report = tmp_path / "lychee.json" + data = report_data(errors=[response()], timeouts=[response(code=None)], total=100) + data.update(errors=27, timeouts=10) + write_report(report, data) + + result = retry_lychee.read_report(report, 2) + + assert result.transient is True + assert result.errors == 27 + assert result.timeouts == 10 + assert len(result.failures) == 2 + + +@pytest.mark.agent_authored(model="gpt-6") +def test_ten_transient_attempts_exhausted_remains_a_failure(tmp_path, monkeypatch): + report = tmp_path / "lychee.json" + data = report_data(errors=[response()]) + write_report(report, data) + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + calls = [] + + def run(command, *, check): + calls.append(command) + write_report(Path(command[-1]), data) + return subprocess.CompletedProcess(command, 2) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == 2 + assert len(calls) == 9 + assert delays == [60, *([120] * 8)] + assert len(list(tmp_path.glob("lychee-attempt-*.json"))) == 10 + assert "after 10 attempts" in report.with_suffix(".md").read_text(encoding="utf-8") + + +@pytest.mark.parametrize("code", [1, 3]) +@pytest.mark.agent_authored(model="gpt-6") +def test_initial_non_link_failure_stops_without_retry(tmp_path, monkeypatch, code): + monkeypatch.setattr(retry_lychee.subprocess, "run", lambda *_args, **_kwargs: pytest.fail("Unexpected retry")) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + assert retry_lychee.retry_lychee(code, tmp_path / "missing.json") == code + + +@pytest.mark.parametrize("code", [1, 3, -15]) +@pytest.mark.agent_authored(model="gpt-6") +def test_retry_non_link_failure_does_not_claim_stale_report(tmp_path, monkeypatch, code): + report = tmp_path / "lychee.json" + write_report(report, report_data(errors=[response()])) + monkeypatch.setenv("LYCHEE_ARGS", "--cache --files-from inputs.txt") + delays = [] + monkeypatch.setattr(retry_lychee.time, "sleep", delays.append) + calls = [] + + def run(command, *, check): + calls.append(command) + return subprocess.CompletedProcess(command, code) + + monkeypatch.setattr(retry_lychee.subprocess, "run", run) + + assert retry_lychee.retry_lychee(2, report) == (code if code > 0 else 1) + assert len(calls) == 1 + assert delays == [60] + text = report.with_suffix(".md").read_text(encoding="utf-8") + assert "did not produce a link-check report" in text + assert "### Final attempt" not in text + assert "| Total |" not in text + + +@pytest.mark.parametrize( + "name,value", [("total", True), ("errors", False), ("timeouts", -1), ("errors", 1.0), ("total", "5")] +) +@pytest.mark.agent_authored(model="gpt-6") +def test_invalid_counts_are_rejected(tmp_path, name, value): + report = tmp_path / "lychee.json" + data = report_data() + data[name] = value + write_report(report, data) + + with pytest.raises(ValueError, match="nonnegative integer"): + retry_lychee.read_report(report, 0) + + +@pytest.mark.parametrize( + "data", + [ + {}, + [], + report_data(total=0), + {**report_data(), "errors": 1}, + {**report_data(), "timeouts": 1}, + {**report_data(errors=[response()]), "errors": 0}, + {**report_data(timeouts=[response(code=None)]), "timeouts": 0}, + report_data(errors=[response()], timeouts=[response(code=None)], total=1), + {**report_data(errors=[response()]), "error_map": []}, + {**report_data(errors=[response()]), "error_map": {"source": response()}}, + {**report_data(errors=[response()]), "error_map": {"source": ["response"]}}, + { + **report_data(errors=[response()]), + "error_map": {"source": [{"url": 42, "status": {"text": "Error", "code": 429}}]}, + }, + report_data(errors=[{"url": "https://example.test/", "status": {"code": 429}}]), + report_data(errors=[response(code=True)]), + report_data(errors=[response(code="429")]), + report_data(errors=[response(code=None, details=None)]), + report_data(errors=[{"url": "https://example.test/", "status": {"text": "Unknown error"}}]), + ], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_malformed_or_contradictory_json_cannot_pass_or_trigger_retries(tmp_path, monkeypatch, data): + report = tmp_path / "lychee.json" + write_report(report, data) + monkeypatch.setattr(retry_lychee.time, "sleep", lambda _delay: pytest.fail("Unexpected sleep")) + + with pytest.raises(ValueError, match="Lychee"): + retry_lychee.retry_lychee(2, report) + + +@pytest.mark.parametrize( + "data,code", + [(report_data(), 2), (report_data(errors=[response()]), 0), (report_data(timeouts=[response(code=None)]), 0)], +) +@pytest.mark.agent_authored(model="gpt-6") +def test_report_must_agree_with_exit_code(tmp_path, data, code): + report = tmp_path / "lychee.json" + write_report(report, data) + + with pytest.raises(ValueError, match="disagree with exit code"): + retry_lychee.retry_lychee(code, report) + + +@pytest.mark.parametrize("contents", ["", "not json", "{broken}"]) +@pytest.mark.agent_authored(model="gpt-6") +def test_invalid_json_cannot_pass(tmp_path, contents): + report = tmp_path / "lychee.json" + report.write_text(contents, encoding="utf-8") + + with pytest.raises(json.JSONDecodeError): + retry_lychee.retry_lychee(0, report) + + +@pytest.mark.agent_authored(model="gpt-6") +def test_missing_initial_report_cannot_pass(tmp_path): + with pytest.raises(FileNotFoundError): + retry_lychee.retry_lychee(0, tmp_path / "missing.json") + + +@pytest.mark.agent_authored(model="gpt-6") +def test_final_markdown_escapes_untrusted_report_fields_and_retains_raw_json(tmp_path): + report = tmp_path / "lychee.json" + failure = response( + "file:///docs/x|[link].html", + code=None, + text="Missing ", + details="`a` *b* | c\n# injected", + ) + data = report_data(errors=[failure]) + data["error_map"] = {"docs/[source]|file\n# heading": [failure]} + write_report(report, data) + + assert retry_lychee.retry_lychee(2, report) == 2 + + text = report.with_suffix(".md").read_text(encoding="utf-8") + assert "