diff --git a/.devcontainer/Dockerfile b/.devcontainer/Dockerfile index 700148c0b..e5677288e 100644 --- a/.devcontainer/Dockerfile +++ b/.devcontainer/Dockerfile @@ -66,24 +66,24 @@ RUN apt-get update \ # TRIVY_VERSION -> https://github.com/aquasecurity/trivy/releases ENV PATH="/go/bin:/usr/local/go/bin:${PATH}" \ - KUBECTL_VERSION=v1.37.0 \ - KUSTOMIZE_VERSION=5.8.1 \ + KUBECTL_VERSION=v1.37.1 \ + KUSTOMIZE_VERSION=5.8.2 \ KUBEBUILDER_VERSION=4.16.0 \ - GOLANGCI_LINT_VERSION=v2.13.2 \ + GOLANGCI_LINT_VERSION=v2.14.0 \ HELM_VERSION=v4.3.0 \ K3D_VERSION=v5.9.0 \ - FLUX_VERSION=2.9.5 \ - FLUX_OPERATOR_VERSION=0.60.0 \ - TASK_VERSION=v3.53.1 \ + FLUX_VERSION=2.9.6 \ + FLUX_OPERATOR_VERSION=0.61.0 \ + TASK_VERSION=v3.54.0 \ ACTIONLINT_VERSION=1.7.12 \ HADOLINT_VERSION=2.15.1 \ VALKEY_VERSION=9.1.2 \ COSIGN_VERSION=v3.1.3 \ ORAS_VERSION=1.3.4 \ NODE_MAJOR=24 \ - MARKDOWNLINT_CLI2_VERSION=0.23.2 \ - VALE_VERSION=3.22.0 \ - TRIVY_VERSION=0.74.0 + MARKDOWNLINT_CLI2_VERSION=0.23.3 \ + VALE_VERSION=3.24.0 \ + TRIVY_VERSION=0.75.0 # Fail early on unsupported architectures instead of producing a partial image. RUN test "$(dpkg --print-architecture)" = "amd64" \ @@ -286,7 +286,7 @@ RUN chgrp -R godev /go && \ # same layer keeps those ~400MB out of the image instead of merely hiding them. RUN go install sigs.k8s.io/controller-tools/cmd/controller-gen@v0.22.0 \ && go install github.com/onsi/ginkgo/v2/ginkgo@v2.33.0 \ - && go install sigs.k8s.io/controller-runtime/tools/setup-envtest@v0.25.1 \ + && go install sigs.k8s.io/controller-runtime/tools/setup-envtest@v0.25.2 \ && go install golang.org/x/tools/cmd/goimports@v0.49.0 \ && go install github.com/boyter/scc/v4@v4.1.0 \ && go clean -cache diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a7e810840..7e210b89a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -853,7 +853,6 @@ jobs: # cross-target rule-change snapshot coupling, now fixed (per-target # effective watch plan — docs/spec/gittarget-isolation-on-rule-change.md). e2e_ginkgo_procs: "4" - k3d_agent_count: "0" - name: full-core script: "task test-e2e" # `bi-directional` and `source-cluster` moved out to their own legs (they @@ -864,7 +863,6 @@ jobs: needs_artifact: false coverage: "1" e2e_ginkgo_procs: "4" - k3d_agent_count: "0" # The bi-directional corner: the only leg that installs Argo CD, and the # only place gitops-reverser shares a path with a foreign GitOps engine. # Its own runner (hence its own cluster) is what keeps the Argo CD @@ -881,7 +879,6 @@ jobs: e2e_report_name: "bi-directional" needs_artifact: false coverage: "1" - k3d_agent_count: "0" # The source-cluster corner: the only leg that installs kcp, and the only # place gitops-reverser mirrors REMOTE clusters (GitTarget.spec.kubeConfig). # kcp workspaces are cheap logical clusters, so its own runner installs a @@ -898,7 +895,6 @@ jobs: e2e_report_name: "source-cluster" needs_artifact: false coverage: "1" - k3d_agent_count: "0" # Quickstart chain, sharded across two runners (was the single # `quickstart` lane). quickstart-install runs the two install-mode # validations on one cluster (cleanup-installs.sh resets the namespace @@ -918,7 +914,6 @@ jobs: needs_artifact: true coverage: "" e2e_ginkgo_procs: "2" - k3d_agent_count: "0" - name: image-refresh script: >- export INSTALL_MODE=helm HELM_CHART_SOURCE=./gitops-reverser.tgz @@ -926,7 +921,6 @@ jobs: needs_artifact: true coverage: "" e2e_ginkgo_procs: "2" - k3d_agent_count: "0" env: PROJECT_IMAGE: ${{ needs.build.outputs.image }} CI_CONTAINER: ${{ needs.ci-container.outputs.image }} @@ -1020,7 +1014,6 @@ jobs: -v /var/run/docker.sock:/var/run/docker.sock \ -w "${{ env.CI_WORKDIR }}" \ -e IMAGE_DELIVERY_MODE=${{ env.IMAGE_DELIVERY_MODE }} \ - -e K3D_AGENT_COUNT=${{ matrix.k3d_agent_count }} \ -e HOST_PROJECT_PATH=${{ github.workspace }} \ -e KUBECONFIG=${{ env.CI_WORKDIR }}/.stamps/cluster/k3d-gitops-reverser-test-e2e/.kube/config \ ${{ env.CI_CONTAINER }} \ @@ -1059,7 +1052,6 @@ jobs: -e E2E_GINKGO_PROCS=${{ matrix.e2e_ginkgo_procs }} \ -e E2E_LABEL_FILTER \ -e E2E_REPORT_NAME \ - -e K3D_AGENT_COUNT=${{ matrix.k3d_agent_count }} \ -e HOST_PROJECT_PATH=${{ github.workspace }} \ -e KUBECONFIG=${{ env.CI_WORKDIR }}/.stamps/cluster/k3d-gitops-reverser-test-e2e/.kube/config \ ${{ env.CI_CONTAINER }} \ diff --git a/README.md b/README.md index 392e9820a..831a7d64a 100644 --- a/README.md +++ b/README.md @@ -180,8 +180,8 @@ repository, so that two of them never write the same paths. - **Deletes:** the default mirrors observed delete events but retains documents absent from a reconnect snapshot. Choose a [deletion policy](docs/configuration.md#deletion-policy-specprunemode) that fits your repository. -- **Versions:** tested against Kubernetes `1.37` at the API level (envtest) and `1.36` end-to-end - (k3s, which has no stable `1.37` release yet). Other versions may work but are not in the matrix. +- **Versions:** the test suites run against Kubernetes `1.37`, at the API level with `envtest` + and end-to-end on `k3s`. Other versions may work but are not in the matrix. Importing this repo as a Go module carries its own version floor; see [`docs/UPGRADING.md`](docs/UPGRADING.md). diff --git a/docs/bi-directional.md b/docs/bi-directional.md index 5a9f83bb5..ed4f23d36 100644 --- a/docs/bi-directional.md +++ b/docs/bi-directional.md @@ -292,7 +292,7 @@ therefore still use the old pin. Flux's equivalent of split ownership is `Kustomization.spec.ignore`, added in kustomize-controller `v1.9.0`, with the same trade as Argo's: selected live fields survive later applies, Git seeds them at creation, later Git edits to them do not land, and Reverser still captures them back into Git. -Our e2e installs Flux `2.9.5` but no spec exercises `spec.ignore`, so it is upstream behavior we +Our e2e installs Flux `2.9.6` but no spec exercises `spec.ignore`, so it is upstream behavior we have read in the source and have no test of our own behind; the [source review](facts/gitops-apply-and-field-ignore.md) has the code paths. diff --git a/docs/ci/k3s-1.37-dns-after-node-restart.md b/docs/ci/k3s-1.37-dns-after-node-restart.md new file mode 100644 index 000000000..649d86be3 --- /dev/null +++ b/docs/ci/k3s-1.37-dns-after-node-restart.md @@ -0,0 +1,149 @@ +# K3s 1.37 e2e: cluster DNS dies after the webhook-TLS node restart + +Status report for PR #415 (`chore/dependency-upgrades-2026-10-05`), 2026-10-05. The PR moves +the e2e cluster from `rancher/k3s:v1.36.4-k3s1` to `rancher/k3s:v1.37.1-k3s1`. This report covers +the e2e failures that followed. The k3s bump stays in the PR while this is open. + +## Summary + +- On k3s 1.37.1, four of the seven CI bring-ups that restart the k3d server node failed. + The manager never turned Ready because it could not resolve + `valkey.valkey-e2e.svc.cluster.local`. +- The failure is intermittent. The same legs also pass: `full-core` and `full-manager` failed and + then passed on re-run, and `bi-directional` passed through the same restart. Only + `source-cluster` failed twice. +- Before this fix, CI ran every e2e leg on a single-node cluster (`k3d_agent_count: "0"`), so the + restart bounced every pod: the manager, CoreDNS and Valkey all came back on the same restarted + node. The local default was 3 agents, where most pods lived on agents and survived the restart. +- The fix gives every e2e cluster one agent and taints the server `NoSchedule`. With it, all four + restart legs recovered on their first attempt in CI, `source-cluster` included. +- A local bring-up on 1.37.1 (3 agents) recovered cleanly. A minimal single-node cluster and a + 1-server-3-agent cluster, on 1.37.1 and on 1.36.5, kept DNS working through the same restart. +- Nothing points at the Go module bumps: the failure is a DNS timeout before any spec runs. + +## The failure + +`hack/e2e/inject-webhook-tls.sh` writes the final audit webhook kubeconfig and then runs +`docker restart k3d--server-0` so kube-apiserver picks it up. The API returns within +seconds (`✅ Cluster API healthy`). The manager Deployment then never finishes its rollout: + +```text +❌ Manager did not recover after the node restart; dumping diagnostics. +gitops-reverser-57b99fc7c6-ffpvl 0/1 Running 0 5m15s 10.42.0.28 k3d-gitops-reverser-test-e2e-server-0 +... +"audit Redis not yet reachable; pod stays not-ready until first connection", +"error":"dial tcp: lookup valkey.valkey-e2e.svc.cluster.local: i/o timeout" +``` + +The readiness gate waits on its first Redis connection, so a pod without DNS stays `0/1` until +`MANAGER_ROLLOUT_TIMEOUT` (300s) expires. `prepare-e2e` fails, and Ginkgo reports `Ran 0 of 118 +Specs`. + +Only legs that install with attribution (audit TLS) do the restart. `quickstart-install` and +`image-refresh` skip it ("configured-author install; skipping audit webhook TLS injection"). + +## Evidence + +Every node restart in CI on this PR, plus the earlier 1.36.4 case. All are single-node clusters, +so `server-0` is the only node a pod can land on: + +| Run | Leg | k3s | Outcome | Manager pod placed on | +|---|---|---|---|---| +| 37326858093 attempt 1 | full-core | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.37) | +| 37326858093 attempt 1 | full-manager | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.36) | +| 37326858093 attempt 1 | source-cluster | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.37) | +| 37326858093 attempt 1 | bi-directional | 1.37.1 | recovered, passed | not dumped | +| 37326858093 attempt 2 | full-core | 1.37.1 | recovered, passed | not dumped | +| 37326858093 attempt 2 | source-cluster | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.28) | +| 37326858093 attempt 2 | full-manager | 1.37.1 | recovered, passed (78 of 118 specs) | not dumped | +| 36765899550 (2026-09-30) | source-cluster | 1.36.4 | DNS timeout, failed | `server-0` (10.42.0.35) | + +Before this PR, the merges of #410 and #413 ran all six legs green on 1.36.4. + +The placement column carries no signal on its own: with zero agents there is nowhere else to go. + +## Local reproduction + +All local runs used Docker 29.2.1 on the devcontainer host. + +1. **Minimal cluster, single node, 1.37.1.** DNS from a pod worked before and after + `docker restart` of the server. +2. **Minimal cluster, 1 server and 3 agents, 1.37.1 and 1.36.5 side by side.** A busybox pod pinned + to each node resolved `kubernetes.default.svc.cluster.local` before the restart, 45s after + and 105s after, on both versions. CoreDNS moved to an agent after the restart. +3. **The e2e bring-up on 1.37.1** (`task CTX=k3d-gitops-reverser-k137-e2e prepare-e2e`, same + arguments as CI). The restart recovered: `deployment "gitops-reverser" successfully rolled + out`, `✅ Audit pipeline live`, `✅ Webhook TLS injection complete`. The new manager pod landed on + `agent-1`; CoreDNS and Valkey were on `agent-0`. + +None of these reproduced the failure. Run 3 used the local default of 3 agents, so the restart +left most workloads untouched; it does not test the single-node case CI runs. Both e2e clusters +(`gitops-reverser-test-e2e`, `gitops-reverser-k137-e2e`) have since been deleted to free memory. +The host had about 25 GiB available during the runs, with several other k3d clusters up. + +## Working hypothesis + +On a single-node cluster, `docker restart` brings every pod back at once on the restarted node, +and the manager sometimes comes up without working cluster DNS. On 1.37.1 that happens far more +often than on 1.36.4 (4 of 7 versus one recorded case). + +What it does not explain yet: + +- The minimal single-node repro on 1.37.1 recovered, but it ran once and carried none of the e2e + load (audit webhook flags, Flux, cert-manager, Valkey, Gitea). +- Whether the break is pod networking after the container restart (flannel or kube-proxy state), + CoreDNS coming back late or wrong, or the manager's resolver caching an early failure. The + dumps hold no CoreDNS state, no node state and no in-pod probe. + +The 3-agent local runs never failed, which fits the hypothesis: there, the manager and Valkey +usually sit on agents that the restart does not touch. + +## Recommended next steps + +1. **Make the next failure explain itself.** Extend `wait_for_manager_rollout`'s failure dump in + `hack/e2e/inject-webhook-tls.sh` with: + - `kubectl get pods -n kube-system -o wide` (CoreDNS placement) + - `kubectl get nodes -o wide` plus the server's flannel and kube-proxy state + - an `nslookup` from a fresh pod + + One more CI failure then separates pod networking from CoreDNS. +2. **Reproduce the CI shape locally.** Bring up 1.37.1 with `K3D_AGENT_COUNT=0`, the way CI does, + and repeat the restart a few times. +3. **Mitigation: keep workloads off the node that restarts.** Implemented in PR #415: e2e + clusters, local and CI, now run one server and one agent, with the server tainted + `node-role.kubernetes.io/control-plane:NoSchedule`. The restart then bounces only the control + plane, and the manager, Valkey and Gitea keep running on the agent. This also matches how + production clusters keep workloads off control-plane nodes. The cost is one extra node at + bring-up, in CI as well, which ran with zero agents since 2026-06-01 to start faster. The local + default drops from three agents to one: all k3d nodes share the host CPU, so extra agents + bought memory use and startup time rather than capacity. Step 1's diagnostics ship with it. + +## Result + +PR #415 commit `5944cea8`, k3s 1.37.1, one agent and a tainted server. Every leg passed on its +first attempt: + +| Leg | Server tainted | Node restart | Recovered | DNS timeouts | Specs run | +|---|---|---|---|---|---| +| full-manager | yes | yes | yes | 0 | 78 of 118 | +| full-core | yes | yes | yes | 0 | 16 of 118 | +| source-cluster | yes | yes | yes | 0 | 11 of 118 | +| bi-directional | yes | yes | yes | 0 | 4 of 118 | +| quickstart-install | yes | skipped | n/a | 0 | 1 of 118 | +| image-refresh | yes | skipped | n/a | 0 | 6 of 118 | + +One green run does not prove the cause, since the failure was intermittent. It does show the +restart no longer touches the workloads the manager depends on. If a restart ever fails again, +the extended failure dump records why. + +## The rest of PR #415 + +- **Lint fix:** `hack/metricnames.sh` skips `CHANGELOG.md`, the file that made release PR #414 fail + lint. CI Lint is green on #415. #414 needs a re-run or rebase after #415 merges. +- **Go modules:** otel 1.47, controller-runtime 0.25.2, kustomize api/kyaml 0.21.2, go-git and + go-billy v6 beta.1. Unit tests green. Every leg that got past bring-up passed: `full-core` and + `full-manager` (re-runs), `bi-directional`, `quickstart-install`, `image-refresh`. So go-git + beta.1 has run every suite except `source-cluster` successfully. +- **Devcontainer tools:** CI rebuilt the image with them, and Lint passed on it. +- **CodeRabbit:** one minor note on the README versions line (active voice, backticks on `envtest` + and `k3s`). Addressed. diff --git a/go.mod b/go.mod index 71a117807..34befb005 100644 --- a/go.mod +++ b/go.mod @@ -7,8 +7,8 @@ require ( github.com/alicebob/miniredis/v2 v2.39.0 github.com/cespare/xxhash/v2 v2.3.0 github.com/fluxcd/pkg/apis/meta v1.32.0 - github.com/go-git/go-billy/v6 v6.0.0-alpha.2 - github.com/go-git/go-git/v6 v6.0.0-alpha.5 + github.com/go-git/go-billy/v6 v6.0.0-beta.1 + github.com/go-git/go-git/v6 v6.0.0-beta.1 github.com/go-logr/logr v1.4.4 github.com/onsi/ginkgo/v2 v2.33.0 github.com/onsi/gomega v1.44.0 @@ -17,10 +17,10 @@ require ( github.com/prometheus/common v0.72.0 github.com/redis/go-redis/v9 v9.22.0 github.com/stretchr/testify v1.12.1 - go.opentelemetry.io/otel v1.46.0 - go.opentelemetry.io/otel/exporters/prometheus v0.68.0 - go.opentelemetry.io/otel/metric v1.46.0 - go.opentelemetry.io/otel/sdk/metric v1.46.0 + go.opentelemetry.io/otel v1.47.0 + go.opentelemetry.io/otel/exporters/prometheus v0.69.0 + go.opentelemetry.io/otel/metric v1.47.0 + go.opentelemetry.io/otel/sdk/metric v1.47.0 go.uber.org/zap v1.28.0 golang.org/x/crypto v0.57.0 gopkg.in/yaml.v3 v3.0.1 @@ -31,9 +31,9 @@ require ( k8s.io/client-go v0.37.1 k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3 sigs.k8s.io/cli-utils v0.37.2 - sigs.k8s.io/controller-runtime v0.25.1 - sigs.k8s.io/kustomize/api v0.21.1 - sigs.k8s.io/kustomize/kyaml v0.21.1 + sigs.k8s.io/controller-runtime v0.25.2 + sigs.k8s.io/kustomize/api v0.21.2 + sigs.k8s.io/kustomize/kyaml v0.21.2 sigs.k8s.io/yaml v1.6.0 ) @@ -42,7 +42,7 @@ require ( filippo.io/hpke v0.4.0 // indirect github.com/Masterminds/semver/v3 v3.5.0 // indirect github.com/Microsoft/go-winio v0.6.2 // indirect - github.com/ProtonMail/go-crypto v1.4.1 // indirect + github.com/ProtonMail/go-crypto v1.5.2 // indirect github.com/antlr4-go/antlr/v4 v4.13.1 // indirect github.com/beorn7/perks v1.0.1 // indirect github.com/blang/semver/v4 v4.0.0 // indirect @@ -88,10 +88,10 @@ require ( github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee // indirect github.com/monochromegane/go-gitignore v0.0.0-20200626010858-205db1a8cc00 // indirect github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect - github.com/pjbgf/sha1cd v0.6.0 // indirect + github.com/pjbgf/sha1cd v0.7.0 // indirect github.com/prometheus/client_model v0.6.3 // indirect github.com/prometheus/otlptranslator v1.0.0 // indirect - github.com/prometheus/procfs v0.21.1 // indirect + github.com/prometheus/procfs v0.22.0 // indirect github.com/sergi/go-diff v1.4.0 // indirect github.com/spf13/cobra v1.10.2 // indirect github.com/spf13/pflag v1.0.10 // indirect @@ -102,8 +102,9 @@ require ( go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 // indirect go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 // indirect go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 // indirect - go.opentelemetry.io/otel/sdk v1.46.0 // indirect - go.opentelemetry.io/otel/trace v1.46.0 // indirect + go.opentelemetry.io/otel/log v1.47.0 // indirect + go.opentelemetry.io/otel/sdk v1.47.0 // indirect + go.opentelemetry.io/otel/trace v1.47.0 // indirect go.opentelemetry.io/proto/otlp v1.11.0 // indirect go.uber.org/atomic v1.11.0 // indirect go.uber.org/multierr v1.11.0 // indirect diff --git a/go.sum b/go.sum index d8c27697b..e1301f6b3 100644 --- a/go.sum +++ b/go.sum @@ -10,8 +10,8 @@ github.com/Masterminds/semver/v3 v3.5.0 h1:kQceYJfbupGfZOKZQg0kou0DgAKhzDg2NZPAw github.com/Masterminds/semver/v3 v3.5.0/go.mod h1:4V+yj/TJE1HU9XfppCwVMZq3I84lprf4nC11bSS5beM= github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERoyfY= github.com/Microsoft/go-winio v0.6.2/go.mod h1:yd8OoFMLzJbo9gZq8j5qaps8bJ9aShtEA8Ipt1oGCvU= -github.com/ProtonMail/go-crypto v1.4.1 h1:9RfcZHqEQUvP8RzecWEUafnZVtEvrBVL9BiF67IQOfM= -github.com/ProtonMail/go-crypto v1.4.1/go.mod h1:e1OaTyu5SYVrO9gKOEhTc+5UcXtTUa+P3uLudwcgPqo= +github.com/ProtonMail/go-crypto v1.5.2 h1:cucYnvqcY7UOXVD//mSyjeaPY0SSN3v5cDkYPxumINk= +github.com/ProtonMail/go-crypto v1.5.2/go.mod h1:/RaSu30DaKO4RY+XdV/ACcCcZkGr7AhUIduq5sjzzCo= github.com/alicebob/miniredis/v2 v2.39.0 h1:M7WbmV5BmV56L8KTG0rw6vEQ+woTOghpDgin2xv4A0g= github.com/alicebob/miniredis/v2 v2.39.0/go.mod h1:TcL7YfarKPGDAthEtl5NBeHZfeUQj6OXMm/+iu5cLMM= github.com/anmitsu/go-shlex v0.0.0-20200514113438-38f4b401e2be h1:9AeTilPcZAjCFIImctFaOjnTIavg87rW78vTPkQqLI8= @@ -67,12 +67,12 @@ github.com/go-errors/errors v1.5.1 h1:ZwEMSLRCapFLflTpT7NKaAc7ukJ8ZPEjzlxt8rPN8b github.com/go-errors/errors v1.5.1/go.mod h1:sIVyrIiJhuEF+Pj9Ebtd6P/rEYROXFi3BopGUQ5a5Og= github.com/go-git/gcfg/v2 v2.0.2 h1:MY5SIIfTGGEMhdA7d7JePuVVxtKL7Hp+ApGDJAJ7dpo= github.com/go-git/gcfg/v2 v2.0.2/go.mod h1:/lv2NsxvhepuMrldsFilrgct6pxzpGdSRC13ydTLSLs= -github.com/go-git/go-billy/v6 v6.0.0-alpha.2 h1:1Sv5WemXL8CxKrAx1gioJ+uHNb2bZJhiQLfwSZ4Et8c= -github.com/go-git/go-billy/v6 v6.0.0-alpha.2/go.mod h1:r/bsv9i/iDyyEU8/Z6mjC+YraOVwie1ddfUqBCElKXQ= +github.com/go-git/go-billy/v6 v6.0.0-beta.1 h1:OJ4r207zcEbeq/DJNajQUOMd+pph3r9A7hRq1Sjl1bA= +github.com/go-git/go-billy/v6 v6.0.0-beta.1/go.mod h1:QGOWN4CzmtRfpFeCw5paG3RJyKdTwsfGASkxBVkIaAU= github.com/go-git/go-git-fixtures/v6 v6.0.0-alpha.1 h1:gmqi2jvsreu0s8JMLylYDFq4sbjHwwlhktMw0DUg3mA= github.com/go-git/go-git-fixtures/v6 v6.0.0-alpha.1/go.mod h1:ECf1MqJlBdYpKggBrOXjo/0EnvRZx6D++I86UYjPgAQ= -github.com/go-git/go-git/v6 v6.0.0-alpha.5 h1:sE+OlkHgYWNMVmN1s9sR7uyFgsWLtxcNWse/vBYKxRE= -github.com/go-git/go-git/v6 v6.0.0-alpha.5/go.mod h1:3IjhiZnM+uBmUrOGSeqrJpsmi4Vd0H2NZO/uK2a7d0s= +github.com/go-git/go-git/v6 v6.0.0-beta.1 h1:1t+vJYGDBN6f9GRWskgWx5UjXPFUa7wfffZZZ2C5GtA= +github.com/go-git/go-git/v6 v6.0.0-beta.1/go.mod h1:ugDlr38Wwc+fOw85J+Oluh5TNXQtJgYMcriOljg3l4s= github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A= github.com/go-logr/logr v1.4.4 h1:tG4xh9yMsRCAiodLVTxyrkzSZ9+o0L1Kg/+cPVcbP/8= github.com/go-logr/logr v1.4.4/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY= @@ -180,8 +180,8 @@ github.com/onsi/ginkgo/v2 v2.33.0 h1:C8gBA6Uc2ZEubiV+SXiu5tZnMTwEmXHgkJwGozKtZf8 github.com/onsi/ginkgo/v2 v2.33.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= github.com/onsi/gomega v1.44.0 h1:eAiGl3Pw5jz5GQdDff0BcxYpAX1JxW8xD7mFUuwNfZQ= github.com/onsi/gomega v1.44.0/go.mod h1:e/C2HwaZ1DhvjzXXuFhcR7hY7Sh9pl7MmoWKEjzwcdA= -github.com/pjbgf/sha1cd v0.6.0 h1:3WJ8Wz8gvDz29quX1OcEmkAlUg9diU4GxJHqs0/XiwU= -github.com/pjbgf/sha1cd v0.6.0/go.mod h1:lhpGlyHLpQZoxMv8HcgXvZEhcGs0PG/vsZnEJ7H0iCM= +github.com/pjbgf/sha1cd v0.7.0 h1:ZRNPKHj+gfkLBf0KJv/p1Hmz+7bqCT5o311YX0nq+DA= +github.com/pjbgf/sha1cd v0.7.0/go.mod h1:pKR5Li+qTCo+ebqITZfwPlJGoCFCvSO00qxjbNtTUFQ= github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4= github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0= github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= @@ -195,8 +195,8 @@ github.com/prometheus/common v0.72.0 h1:tAYsE+sPJxIncDAobm4H5aQjmox9ZxEIIqPbiffa github.com/prometheus/common v0.72.0/go.mod h1:77NWqAQ2tXT7BIK40qjJdw5Acrsrg1TlHAnsQi3i6mk= github.com/prometheus/otlptranslator v1.0.0 h1:s0LJW/iN9dkIH+EnhiD3BlkkP5QVIUVEoIwkU+A6qos= github.com/prometheus/otlptranslator v1.0.0/go.mod h1:vRYWnXvI6aWGpsdY/mOT/cbeVRBlPWtBNDb7kGR3uKM= -github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI= -github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY= +github.com/prometheus/procfs v0.22.0 h1:6q9+/JL9IKAPbCmBrv9n5O5Ty3NKnciV5X7YGw0oics= +github.com/prometheus/procfs v0.22.0/go.mod h1:CvmFr/GVhIjIvWJZW3tgkODBQMRIf0EyWMQLHCHab58= github.com/redis/go-redis/v9 v9.22.0 h1:laDvpYXTJtZLloinw1fA5Kqd6HAEH2XKxOkG/PDq2F0= github.com/redis/go-redis/v9 v9.22.0/go.mod h1:y2g0Wj8rQvuK0ELM+oxSudcLtC09JScs98I/X9gRWY4= github.com/rogpeppe/go-internal v1.16.0 h1:O9DK+vNMDVGLr2BeZqmpLeMjiMNkuXfcqntWbZV6S5g= @@ -237,24 +237,26 @@ go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 h1:LMuyCAyfalSjDyjdC65nK6N0zoTT63+E/u95X0JovZI= go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0/go.mod h1:085m8qbm4hgc8rZWGDEa4vmyyo2c3nPxUslYUKUIU04= -go.opentelemetry.io/otel v1.46.0 h1:FHt5/CDyVxi/8IM1CH7VE/rRgq3kLHa2mSTVMO8AWyc= -go.opentelemetry.io/otel v1.46.0/go.mod h1:Gj3SEScelsNC45tp4nSxRYlS+f5iez7W8XPMCt905kE= +go.opentelemetry.io/otel v1.47.0 h1:j7ALJ/zgkS7Z6aeJW09p8VC9804bC+PpeTfCD4XPnOM= +go.opentelemetry.io/otel v1.47.0/go.mod h1:8wS9O2qfXrYrzp6hIF/HOYJJf/wIhFPhR2xLuP+iXQU= go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 h1:QRefszxJmfPdjXUUm3j6iDzY03mTPXMjqErFqQ67vUg= go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0/go.mod h1:Tiz03lTBVBrm7eWZBOidzEaYaJa8tjwGUGv6d8mlTyk= go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 h1:fG5MCxGz8+2VtrN/WgqSpJFctVz24gpxj8CxkKmc8Ww= go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0/go.mod h1:BmAYTn+3ysbRe+IU2msxmf5Rx3g6DHvex+tWI3LdhYI= -go.opentelemetry.io/otel/exporters/prometheus v0.68.0 h1:QOf2IftqQwITVRJpnn0M7M9ZCbgWfxz4P7i9C9yc2N4= -go.opentelemetry.io/otel/exporters/prometheus v0.68.0/go.mod h1:bgSvqu2TWGXiz7yr5UTMfObH8oqxJWHTnubQ3ef9BO4= -go.opentelemetry.io/otel/metric v1.46.0 h1:yBnkXvgV7AXFILZc5K6IZe/CBFF3OS7BJ8ov6/lj0K8= -go.opentelemetry.io/otel/metric v1.46.0/go.mod h1:iPmdWqifKUdzziPkvvzIJXITl56fQx2mGM/DHLB3/2o= -go.opentelemetry.io/otel/metric/x v0.68.0 h1:TA/cBT23D3MnxYPwHL7YFOdYGdx0A0v+s7Mzotpd1dU= -go.opentelemetry.io/otel/metric/x v0.68.0/go.mod h1:agudOmvWhwUTjgibWDzxD2PoWYnpw5Ht5jISYOD2Hd4= -go.opentelemetry.io/otel/sdk v1.46.0 h1:h5CNQQjEbuQXY/JfZtgt3i7HVFV3aHPO2OAwO2eTYPI= -go.opentelemetry.io/otel/sdk v1.46.0/go.mod h1:GAERFXFt5SYCEB+YiKUbMBeza6UaDH7GmGOZEfh2gSM= -go.opentelemetry.io/otel/sdk/metric v1.46.0 h1:0piZ26EG4RBfebb2jhDH6ERCYHoVWduc3kLgPCwSnSE= -go.opentelemetry.io/otel/sdk/metric v1.46.0/go.mod h1:I1PbKrdVc8Qu8HYVDNtqVIwLwjNrhsV/uFuxfwg8mO4= -go.opentelemetry.io/otel/trace v1.46.0 h1:OULy7ccdJnZtJ0UDYFOIGaCmiWzJ8Vi2G/Rsu60qs1c= -go.opentelemetry.io/otel/trace v1.46.0/go.mod h1:J7GAXweO77XSFkB/rmAqk9D6ihszhFjLU+d9WuUxDLI= +go.opentelemetry.io/otel/exporters/prometheus v0.69.0 h1:LQVBTGnHvzrfGJpDoCSVTY7eaLZArvaTSq5l84Skxew= +go.opentelemetry.io/otel/exporters/prometheus v0.69.0/go.mod h1:moif0nlgj/M16NRT8lDZVt8MC52+EOAJfR8EKRJ8LyU= +go.opentelemetry.io/otel/log v1.47.0 h1:cOTS1CcLbSQeZKanGJ+0JpF/+t4PELi3O3bbl2lqCcI= +go.opentelemetry.io/otel/log v1.47.0/go.mod h1:9byitSQ5pLC6PpqwGXjqdMKya6ZTswHRZh2vvXT33nw= +go.opentelemetry.io/otel/metric v1.47.0 h1:4PptaldXx3Eat1XjMZ68pPJEs5wrhlemctZE9a3UdWY= +go.opentelemetry.io/otel/metric v1.47.0/go.mod h1:ADGSXxRrXM6bjbvLo535EstVFlPpPYZm4LBKixjDHwU= +go.opentelemetry.io/otel/metric/x v0.69.0 h1:DjRLr15H83v+hCW7JA9NoJvOkYTtmq5YoDRbe9deYpM= +go.opentelemetry.io/otel/metric/x v0.69.0/go.mod h1:uVvsMPMFFyj/HUQfrUnH3JjnOQ1dwFDorgFLRBasM0k= +go.opentelemetry.io/otel/sdk v1.47.0 h1:zWXEr4j2lFefG87TU6Yg8a7ngfohIKFZHKp0Hf5hC6I= +go.opentelemetry.io/otel/sdk v1.47.0/go.mod h1:VUc24kiOeoGsxG8G9ULx3fWKvB7jMhnGE8Oi607lgR0= +go.opentelemetry.io/otel/sdk/metric v1.47.0 h1:lfISg2j93VT6yqdk9OfUaZmw/GfcZqCCV3jdXtsPnKw= +go.opentelemetry.io/otel/sdk/metric v1.47.0/go.mod h1:ypLp+mW1Nt2x+Szt3b5/i1syodyts49lMOwxpDI3VGw= +go.opentelemetry.io/otel/trace v1.47.0 h1:JOjX/Oci8K94QHddo+bbfya/Ai/nf6/dt9ZfrFNWSrM= +go.opentelemetry.io/otel/trace v1.47.0/go.mod h1:jNaSLa2PZEYFG6fRjJABAu+bw4FS08uDmPg28lTghu0= go.opentelemetry.io/proto/otlp v1.11.0 h1:5rrYs0Ykyj50sdU/JU0x8etU+LubXWb+gED6TbEdMIk= go.opentelemetry.io/proto/otlp v1.11.0/go.mod h1:SmVizdCOAm3XBtG1g1NnOdhW6jtddT72hLMhv8VwA8E= go.uber.org/atomic v1.11.0 h1:ZvwS0R+56ePWxUNi+Atn9dWONBPp/AUETXlHW0DxSjE= @@ -341,14 +343,14 @@ sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0 h1:/YpDJ4vReG7Zm sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0/go.mod h1:tJo1aepTXyR+8Xs3sUsGBDk4Ub2AM5dPAPKJx0mpm5c= sigs.k8s.io/cli-utils v0.37.2 h1:GOfKw5RV2HDQZDJlru5KkfLO1tbxqMoyn1IYUxqBpNg= sigs.k8s.io/cli-utils v0.37.2/go.mod h1:V+IZZr4UoGj7gMJXklWBg6t5xbdThFBcpj4MrZuCYco= -sigs.k8s.io/controller-runtime v0.25.1 h1:BKgU9OeE8xv8EbbM8cY0NVzTQs35rokkdq1jh12fMb4= -sigs.k8s.io/controller-runtime v0.25.1/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= +sigs.k8s.io/controller-runtime v0.25.2 h1:bEkK3PVOIVK9X8QWLGVhJgmFc++47vfT6wakSzAOLsQ= +sigs.k8s.io/controller-runtime v0.25.2/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg= -sigs.k8s.io/kustomize/api v0.21.1 h1:lzqbzvz2CSvsjIUZUBNFKtIMsEw7hVLJp0JeSIVmuJs= -sigs.k8s.io/kustomize/api v0.21.1/go.mod h1:f3wkKByTrgpgltLgySCntrYoq5d3q7aaxveSagwTlwI= -sigs.k8s.io/kustomize/kyaml v0.21.1 h1:IVlbmhC076nf6foyL6Taw4BkrLuEsXUXNpsE+ScX7fI= -sigs.k8s.io/kustomize/kyaml v0.21.1/go.mod h1:hmxADesM3yUN2vbA5z1/YTBnzLJ1dajdqpQonwBL1FQ= +sigs.k8s.io/kustomize/api v0.21.2 h1:MRyw+zLnFBP+G40gZJoKZErAuRiOPEPao+ddS9L6xt4= +sigs.k8s.io/kustomize/api v0.21.2/go.mod h1:inubcVvQjJR/BjUti22YVBWr4EX+XlurEWhB81v2JV4= +sigs.k8s.io/kustomize/kyaml v0.21.2 h1:1javwStFk7cgOeLU7yJtPmXcgMEhQgC2X0WjFT6U0p0= +sigs.k8s.io/kustomize/kyaml v0.21.2/go.mod h1:zX3qwtuouXd2K1fMiCV0VSFReX06a+CY1rhyf5Dy7hQ= sigs.k8s.io/randfill v1.0.0 h1:JfjMILfT8A6RbawdsK2JXGBR5AQVfd+9TbzrlneTyrU= sigs.k8s.io/randfill v1.0.0/go.mod h1:XeLlZ/jmk4i1HRopwe7/aU3H5n1zNUcX6TM94b3QxOY= sigs.k8s.io/structured-merge-diff/v6 v6.4.2 h1:qdOxHwrl2Kaag1aQEarlYcOA9vSyGCp3CIki3aW8c4Q= diff --git a/hack/e2e/inject-webhook-tls.sh b/hack/e2e/inject-webhook-tls.sh index b0dfe8b40..a5be09ec6 100755 --- a/hack/e2e/inject-webhook-tls.sh +++ b/hack/e2e/inject-webhook-tls.sh @@ -77,9 +77,37 @@ wait_for_manager_rollout() { kubectl --context "${CTX}" -n "${NAMESPACE}" logs "${MANAGER_DEPLOY}" --tail=80 >&2 || true kubectl --context "${CTX}" get events -n "${NAMESPACE}" \ --sort-by=.lastTimestamp 2>/dev/null | tail -30 >&2 || true + dump_cluster_dns_diagnostics return 1 } +# dump_cluster_dns_diagnostics shows whether the cluster can resolve names at all after the +# restart. The recorded failure mode is a manager stuck 0/1 on "lookup valkey...: i/o timeout" +# (docs/ci/k3s-1.37-dns-after-node-restart.md); this separates CoreDNS being down from pod +# networking on the manager's node being broken. +dump_cluster_dns_diagnostics() { + echo "--- nodes" >&2 + kubectl --context "${CTX}" get nodes -o wide >&2 || true + echo "--- kube-system pods" >&2 + kubectl --context "${CTX}" -n kube-system get pods -o wide >&2 || true + echo "--- coredns logs" >&2 + kubectl --context "${CTX}" -n kube-system logs -l k8s-app=kube-dns --tail=30 >&2 || true + local selector node + selector="$(kubectl --context "${CTX}" -n "${NAMESPACE}" get "${MANAGER_DEPLOY}" \ + -o go-template='{{range $k, $v := .spec.selector.matchLabels}}{{$k}}={{$v}},{{end}}' 2>/dev/null || true)" + node="$(kubectl --context "${CTX}" -n "${NAMESPACE}" get pods -l "${selector%,}" \ + -o jsonpath='{.items[0].spec.nodeName}' 2>/dev/null || true)" + # kube-system, not the manager's namespace: that one enforces the restricted PodSecurity + # profile, which rejects a bare busybox pod. nodeName still puts the probe beside the manager. + echo "--- DNS probe from a fresh pod on the manager's node (${node:-unknown})" >&2 + kubectl --context "${CTX}" -n kube-system delete pod dns-probe --ignore-not-found >/dev/null 2>&1 || true + timeout 90 kubectl --context "${CTX}" -n kube-system run dns-probe --image=busybox:1.36 \ + --restart=Never --rm -i --quiet \ + --overrides="{\"spec\":{\"nodeName\":\"${node}\"}}" --command -- \ + sh -c 'cat /etc/resolv.conf; nslookup kubernetes.default.svc.cluster.local; nslookup valkey.valkey-e2e.svc.cluster.local' \ + >&2 || true +} + # warmup_audit_path drives a throwaway audited write on every iteration and waits # until the manager logs that it received an audit request. The kube-apiserver audit # webhook reconnects to the freshly restarted manager with backoff, so the first diff --git a/hack/metricnames.sh b/hack/metricnames.sh index 26c7b3eea..b979342ff 100755 --- a/hack/metricnames.sh +++ b/hack/metricnames.sh @@ -18,9 +18,11 @@ cd "$(dirname "$0")/.." # Files that legitimately name metrics which no longer exist: # UPGRADING.md the old -> new migration tables; naming the old name is the point +# CHANGELOG.md release-please copies each BREAKING CHANGE footer in verbatim, and a +# rename's footer names the old metric; the changelog is history # docs/design/ the plan's "Deleted" and "Renamed" sections argue about past names # docs/finished/ shipped-and-archived plans, frozen at their own moment -EXCLUDED_PATHS=(':!external-sources' ':!docs/UPGRADING.md' ':!docs/design' ':!docs/finished') +EXCLUDED_PATHS=(':!external-sources' ':!CHANGELOG.md' ':!docs/UPGRADING.md' ':!docs/design' ':!docs/finished') # Tokens that look like metric names but are not references to one. Keep this list short and # justified; anything added here is a check that stopped checking something. diff --git a/internal/git/commit_window_timers_test.go b/internal/git/commit_window_timers_test.go index 77ef5323a..b5922b619 100644 --- a/internal/git/commit_window_timers_test.go +++ b/internal/git/commit_window_timers_test.go @@ -40,12 +40,17 @@ func TestWindowTimers_ContinuousActivityStopsAtTheTargetsMaxDuration(t *testing. createPlainGitTarget(t, worker, "team-a", "team-a") loop := newBranchWorkerEventLoop(worker, time.Hour) // the idle timer never fires here defer loop.stopTimers() - loop.defaultWindow.maxDuration = 20 * time.Millisecond + loop.defaultWindow.maxDuration = time.Hour loop.lastPushAt = time.Now() // hold the push, so the local commit stays inspectable writeTo(loop, "first") require.NotNil(t, loop.openWindow) - time.Sleep(30 * time.Millisecond) + // Rewind the deadline rather than sleeping past a short one: a 20ms maxDuration raced the + // opening write itself, which under a loaded test run took longer than 20ms and closed the + // window before the assertion above could see it. + require.WithinDuration(t, time.Now().Add(time.Hour), loop.openWindow.timers.maxAt, time.Minute, + "the window takes the target's maxDuration when it opens") + loop.openWindow.timers.maxAt = time.Now().Add(-time.Millisecond) writeTo(loop, "second") // restarts idle, but maxDuration has already passed assert.Nil(t, loop.openWindow, "a window closes at maxDuration however much keeps arriving") diff --git a/internal/manifestanalyzer/kustomize_render.go b/internal/manifestanalyzer/kustomize_render.go index 195031f32..f3e0cc34b 100644 --- a/internal/manifestanalyzer/kustomize_render.go +++ b/internal/manifestanalyzer/kustomize_render.go @@ -63,19 +63,20 @@ var errRemoteBase = errors.New("kustomization reaches a remote base; the operato // errInvalidImageName refuses a build whose images: entry carries a name kustomize // cannot compile. // -// An images: entry's name: is a REGULAR EXPRESSION, not a literal, and kustomize -// compiles it while DISCARDING the compile error (api/internal/image/image.go): +// An images: entry's name: is a REGULAR EXPRESSION, not a literal. Up to api v0.21.1 +// kustomize compiled it while DISCARDING the compile error (api/internal/image/image.go): // // pattern, _ := regexp.Compile("^" + name + "(:[a-zA-Z0-9_.{}-]*)?(@sha256:...)?$") // -// It then dereferences the nil *Regexp. So `- name: "ngin["` does not fail the build — -// it PANICS inside it, on content that came straight from a user's repository. Like the -// remote-base check, this one must run before krusty, and for the same reason: it is not -// a modelling question, it is what keeps a hostile kustomization.yaml from taking the -// process somewhere it cannot come back from. +// and then dereferenced the nil *Regexp, so `- name: "ngin["` PANICKED inside the build. +// v0.21.2 treats an uncompilable name as matching nothing instead. We still refuse it, +// for two reasons. Flux's kustomize-controller (v1.9.6) pins api v0.21.1, so the folder +// does not deploy in production; refusing keeps us in step with what deploys. And an +// entry that silently matches nothing is never what its author meant. Like the +// remote-base check, this one runs before krusty. var errInvalidImageName = errors.New("images: entry name is not a valid regular expression") -// errBuildPanicked is the net under krusty. errInvalidImageName covers the one panic we +// errBuildPanicked is the net under krusty. errInvalidImageName covered the one panic we // found; this covers the ones we have not. A build runs library code we do not own over // bytes we do not control, so a panic there has to become a refused folder — never a // crashed CLI, and never a GitTarget that panics, requeues and panics again for as long @@ -87,7 +88,7 @@ var errBuildPanicked = errors.New("kustomize build panicked") // (api/internal/image/image.go). We validate the WHOLE pattern, not the name alone, so // that what we accept is exactly what kustomize can compile. func imageNamePattern(name string) string { - return "^" + name + "(:[a-zA-Z0-9_.{}-]*)?(@sha256:[a-zA-Z0-9_.{}-]*)?$" + return "^" + name + "(:[a-zA-Z0-9_.{}-]*)?(@[a-zA-Z0-9]+([.+_-][a-zA-Z0-9]+)*:[a-zA-Z0-9_.{}-]*)?$" } // renderedObject is one object kustomize produced, with the provenance saying @@ -209,7 +210,7 @@ func build(fSys filesys.FileSystem, target string) (_ resmap.ResMap, err error) // refuseBeforeBuild refuses the build when any kustomization THIS ROOT REACHES is one we // must not hand to krusty: it declares a remote base (kustomize would fetch it), or an -// images: entry name kustomize would nil-deref on (see errInvalidImageName). +// images: entry name that is not a valid regular expression (see errInvalidImageName). // // Scoping it to the reachable graph is deliberate, and it is both safer and more // accurate than a scan-wide check: kustomize only loads what it actually reaches, diff --git a/internal/manifestanalyzer/kustomize_render_hostile_test.go b/internal/manifestanalyzer/kustomize_render_hostile_test.go index 5c23a3b9e..ddea42219 100644 --- a/internal/manifestanalyzer/kustomize_render_hostile_test.go +++ b/internal/manifestanalyzer/kustomize_render_hostile_test.go @@ -16,9 +16,9 @@ import ( // does with content nobody would write on purpose — and for the one shape that makes a // folder INVISIBLE to the refusal path rather than merely broken. -// An images: entry's name: is a regular expression, and kustomize compiles it while -// discarding the compile error, then dereferences the nil *Regexp. `- name: "ngin["` does -// not fail the build; it panics inside it. We must refuse before krusty ever sees it. +// An images: entry's name: is a regular expression. Kustomize up to api v0.21.1 (what +// Flux builds with) panics on `- name: "ngin["`; newer ones silently match nothing. Either +// way the folder does not do what it says, so we refuse before krusty ever sees it. func TestRenderRoot_InvalidImageNameIsRefusedBeforeTheBuild(t *testing.T) { files := imageFixture("nginx:v1", " - name: \"ngin[\"\n newTag: \"2.0\"\n") @@ -92,18 +92,20 @@ func TestRenderRoot_KustomizationYMLCarriesProvenance(t *testing.T) { require.NotEmpty(t, rendered[0].TransformedBy, "the override chain must be readable") } +// panickingFS panics on every ReadFile, standing in for a panic anywhere inside krusty. +type panickingFS struct{ filesys.FileSystem } + +func (panickingFS) ReadFile(string) ([]byte, error) { panic("read exploded") } + // The net under krusty: whatever panics in there, the caller gets an error and the process -// keeps its footing. Driven straight at build(), because the refusal above means the panic -// we know about can no longer reach it. +// keeps its footing. Driven straight at build() with a filesystem that panics, because the +// one panic we found in kustomize itself was fixed upstream (api v0.21.2) and could not +// reach build() past the refusal above anyway. func TestBuild_PanicBecomesAnError(t *testing.T) { fSys := filesys.MakeFsInMemory() - require.NoError(t, fSys.WriteFile("/scan/kustomization.yaml", - []byte("resources:\n - deployment.yaml\nimages:\n - name: \"ngin[\"\n newTag: \"2.0\"\n"))) - require.NoError(t, fSys.WriteFile("/scan/deployment.yaml", []byte( - "apiVersion: apps/v1\nkind: Deployment\nmetadata:\n name: web\nspec:\n template:\n spec:\n"+ - " containers:\n - name: web\n image: nginx:v1\n"))) + require.NoError(t, fSys.WriteFile("/scan/kustomization.yaml", []byte("resources:\n - deployment.yaml\n"))) - resMap, err := build(fSys, "/scan") // must not panic + resMap, err := build(panickingFS{fSys}, "/scan") // must not panic require.Nil(t, resMap) require.ErrorIs(t, err, errBuildPanicked) diff --git a/test/e2e/cluster/README.md b/test/e2e/cluster/README.md index 3cabbc744..6f7360c73 100644 --- a/test/e2e/cluster/README.md +++ b/test/e2e/cluster/README.md @@ -22,9 +22,15 @@ The Task-driven e2e prep flow first copies the tracked audit assets into `.stamp - `max-requests-inflight=800` (override with `KUBE_APISERVER_MAX_REQUESTS_INFLIGHT`) - `max-mutating-requests-inflight=400` (override with `KUBE_APISERVER_MAX_MUTATING_REQUESTS_INFLIGHT`) -The k3s node image defaults to `rancher/k3s:v1.36.1-k3s1`, matching the current k3s `latest` +The k3s node image defaults to `rancher/k3s:v1.37.1-k3s1`, matching the current k3s `latest` channel. Override it with `K3S_IMAGE` when intentionally testing a different k3s release. +The cluster has one server and one agent (`K3D_AGENT_COUNT`, default `1`). The server node is +tainted `node-role.kubernetes.io/control-plane:NoSchedule`, so workloads run on the agent. The +audit webhook setup restarts the server container to load the final webhook config, and the taint +keeps that restart from taking the manager, Valkey and Gitea down with it. `K3D_AGENT_COUNT=0` +still builds a single-node cluster, with no taint. + It also disables these packaged k3s components by default: - `traefik` diff --git a/test/e2e/cluster/start-cluster.sh b/test/e2e/cluster/start-cluster.sh index 0ec8540e1..e62ddb410 100644 --- a/test/e2e/cluster/start-cluster.sh +++ b/test/e2e/cluster/start-cluster.sh @@ -11,8 +11,11 @@ DISABLE_K3S_TRAEFIK="${DISABLE_K3S_TRAEFIK:-true}" DISABLE_K3S_SERVICELB="${DISABLE_K3S_SERVICELB:-true}" KUBE_APISERVER_MAX_REQUESTS_INFLIGHT="${KUBE_APISERVER_MAX_REQUESTS_INFLIGHT:-800}" KUBE_APISERVER_MAX_MUTATING_REQUESTS_INFLIGHT="${KUBE_APISERVER_MAX_MUTATING_REQUESTS_INFLIGHT:-400}" -K3D_AGENT_COUNT="${K3D_AGENT_COUNT:-3}" -K3S_IMAGE="${K3S_IMAGE:-rancher/k3s:v1.36.4-k3s1}" +# One agent, with the server tainted NoSchedule (see create_cluster), so every workload runs on +# a node that _webhook-tls-ready's `docker restart` of the server never touches. 0 is still +# accepted for a single-node cluster, which leaves the server untainted. +K3D_AGENT_COUNT="${K3D_AGENT_COUNT:-1}" +K3S_IMAGE="${K3S_IMAGE:-rancher/k3s:v1.37.1-k3s1}" AUDIT_DIR_REL="${AUDIT_DIR_REL:-test/e2e/cluster/audit}" K3D_CREATE_LOG_FILE="${TMPDIR:-/tmp}/k3d-create-${CLUSTER_NAME}.log" REPO_PWD="$(pwd -P)" @@ -238,6 +241,17 @@ create_cluster() { k3s_args+=("--disable=traefik@server:0") fi + # The audit webhook setup restarts the server container (hack/e2e/inject-webhook-tls.sh), + # which on a single-node cluster brings every pod back at once on a freshly restarted node. + # The manager then sometimes came up with cluster DNS dead and never went Ready (4 of 7 + # restarts on k3s 1.37.1, docs/ci/k3s-1.37-dns-after-node-restart.md). With the server + # tainted, the manager, Valkey and Gitea keep running on the agent and only the control + # plane restarts. k3s's own add-ons (CoreDNS, metrics-server, local-path) tolerate the taint. + if [ "${K3D_AGENT_COUNT}" -gt 0 ]; then + echo "🔧 Tainting the server node NoSchedule so workloads run on the ${K3D_AGENT_COUNT} agent(s)" + k3s_args+=("--node-taint=node-role.kubernetes.io/control-plane:NoSchedule@server:0") + fi + if [ "${DISABLE_K3S_SERVICELB}" = "true" ]; then echo "🔧 Disabling packaged k3s ServiceLB to avoid svclb pods for local services" k3s_args+=("--disable=servicelb@server:0")