Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 10 additions & 10 deletions .devcontainer/Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -66,24 +66,24 @@ RUN apt-get update \
# TRIVY_VERSION -> https://github.com/aquasecurity/trivy/releases

ENV PATH="/go/bin:/usr/local/go/bin:${PATH}" \
KUBECTL_VERSION=v1.37.0 \
KUSTOMIZE_VERSION=5.8.1 \
KUBECTL_VERSION=v1.37.1 \
KUSTOMIZE_VERSION=5.8.2 \
KUBEBUILDER_VERSION=4.16.0 \
GOLANGCI_LINT_VERSION=v2.13.2 \
GOLANGCI_LINT_VERSION=v2.14.0 \
HELM_VERSION=v4.3.0 \
K3D_VERSION=v5.9.0 \
FLUX_VERSION=2.9.5 \
FLUX_OPERATOR_VERSION=0.60.0 \
TASK_VERSION=v3.53.1 \
FLUX_VERSION=2.9.6 \
FLUX_OPERATOR_VERSION=0.61.0 \
TASK_VERSION=v3.54.0 \
ACTIONLINT_VERSION=1.7.12 \
HADOLINT_VERSION=2.15.1 \
VALKEY_VERSION=9.1.2 \
COSIGN_VERSION=v3.1.3 \
ORAS_VERSION=1.3.4 \
NODE_MAJOR=24 \
MARKDOWNLINT_CLI2_VERSION=0.23.2 \
VALE_VERSION=3.22.0 \
TRIVY_VERSION=0.74.0
MARKDOWNLINT_CLI2_VERSION=0.23.3 \
VALE_VERSION=3.24.0 \
TRIVY_VERSION=0.75.0

# Fail early on unsupported architectures instead of producing a partial image.
RUN test "$(dpkg --print-architecture)" = "amd64" \
Expand Down Expand Up @@ -286,7 +286,7 @@ RUN chgrp -R godev /go && \
# same layer keeps those ~400MB out of the image instead of merely hiding them.
RUN go install sigs.k8s.io/controller-tools/cmd/controller-gen@v0.22.0 \
&& go install github.com/onsi/ginkgo/v2/ginkgo@v2.33.0 \
&& go install sigs.k8s.io/controller-runtime/tools/setup-envtest@v0.25.1 \
&& go install sigs.k8s.io/controller-runtime/tools/setup-envtest@v0.25.2 \
&& go install golang.org/x/tools/cmd/goimports@v0.49.0 \
&& go install github.com/boyter/scc/v4@v4.1.0 \
&& go clean -cache
Expand Down
8 changes: 0 additions & 8 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -853,7 +853,6 @@ jobs:
# cross-target rule-change snapshot coupling, now fixed (per-target
# effective watch plan — docs/spec/gittarget-isolation-on-rule-change.md).
e2e_ginkgo_procs: "4"
k3d_agent_count: "0"
- name: full-core
script: "task test-e2e"
# `bi-directional` and `source-cluster` moved out to their own legs (they
Expand All @@ -864,7 +863,6 @@ jobs:
needs_artifact: false
coverage: "1"
e2e_ginkgo_procs: "4"
k3d_agent_count: "0"
# The bi-directional corner: the only leg that installs Argo CD, and the
# only place gitops-reverser shares a path with a foreign GitOps engine.
# Its own runner (hence its own cluster) is what keeps the Argo CD
Expand All @@ -881,7 +879,6 @@ jobs:
e2e_report_name: "bi-directional"
needs_artifact: false
coverage: "1"
k3d_agent_count: "0"
# The source-cluster corner: the only leg that installs kcp, and the only
# place gitops-reverser mirrors REMOTE clusters (GitTarget.spec.kubeConfig).
# kcp workspaces are cheap logical clusters, so its own runner installs a
Expand All @@ -898,7 +895,6 @@ jobs:
e2e_report_name: "source-cluster"
needs_artifact: false
coverage: "1"
k3d_agent_count: "0"
# Quickstart chain, sharded across two runners (was the single
# `quickstart` lane). quickstart-install runs the two install-mode
# validations on one cluster (cleanup-installs.sh resets the namespace
Expand All @@ -918,15 +914,13 @@ jobs:
needs_artifact: true
coverage: ""
e2e_ginkgo_procs: "2"
k3d_agent_count: "0"
- name: image-refresh
script: >-
export INSTALL_MODE=helm HELM_CHART_SOURCE=./gitops-reverser.tgz
&& env -u PROJECT_IMAGE IMAGE_DELIVERY_MODE=load task test-image-refresh
needs_artifact: true
coverage: ""
e2e_ginkgo_procs: "2"
k3d_agent_count: "0"
env:
PROJECT_IMAGE: ${{ needs.build.outputs.image }}
CI_CONTAINER: ${{ needs.ci-container.outputs.image }}
Expand Down Expand Up @@ -1020,7 +1014,6 @@ jobs:
-v /var/run/docker.sock:/var/run/docker.sock \
-w "${{ env.CI_WORKDIR }}" \
-e IMAGE_DELIVERY_MODE=${{ env.IMAGE_DELIVERY_MODE }} \
-e K3D_AGENT_COUNT=${{ matrix.k3d_agent_count }} \
-e HOST_PROJECT_PATH=${{ github.workspace }} \
-e KUBECONFIG=${{ env.CI_WORKDIR }}/.stamps/cluster/k3d-gitops-reverser-test-e2e/.kube/config \
${{ env.CI_CONTAINER }} \
Expand Down Expand Up @@ -1059,7 +1052,6 @@ jobs:
-e E2E_GINKGO_PROCS=${{ matrix.e2e_ginkgo_procs }} \
-e E2E_LABEL_FILTER \
-e E2E_REPORT_NAME \
-e K3D_AGENT_COUNT=${{ matrix.k3d_agent_count }} \
-e HOST_PROJECT_PATH=${{ github.workspace }} \
-e KUBECONFIG=${{ env.CI_WORKDIR }}/.stamps/cluster/k3d-gitops-reverser-test-e2e/.kube/config \
${{ env.CI_CONTAINER }} \
Expand Down
4 changes: 2 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -180,8 +180,8 @@ repository, so that two of them never write the same paths.
- **Deletes:** the default mirrors observed delete events but retains documents absent from a
reconnect snapshot. Choose a [deletion policy](docs/configuration.md#deletion-policy-specprunemode)
that fits your repository.
- **Versions:** tested against Kubernetes `1.37` at the API level (envtest) and `1.36` end-to-end
(k3s, which has no stable `1.37` release yet). Other versions may work but are not in the matrix.
- **Versions:** the test suites run against Kubernetes `1.37`, at the API level with `envtest`
and end-to-end on `k3s`. Other versions may work but are not in the matrix.
Importing this repo as a Go module carries its own version floor; see
[`docs/UPGRADING.md`](docs/UPGRADING.md).

Expand Down
2 changes: 1 addition & 1 deletion docs/bi-directional.md
Original file line number Diff line number Diff line change
Expand Up @@ -292,7 +292,7 @@ therefore still use the old pin.
Flux's equivalent of split ownership is `Kustomization.spec.ignore`, added in kustomize-controller
`v1.9.0`, with the same trade as Argo's: selected live fields survive later applies, Git seeds them
at creation, later Git edits to them do not land, and Reverser still captures them back into Git.
Our e2e installs Flux `2.9.5` but no spec exercises `spec.ignore`, so it is upstream behavior we
Our e2e installs Flux `2.9.6` but no spec exercises `spec.ignore`, so it is upstream behavior we
have read in the source and have no test of our own behind; the
[source review](facts/gitops-apply-and-field-ignore.md) has the code paths.

Expand Down
149 changes: 149 additions & 0 deletions docs/ci/k3s-1.37-dns-after-node-restart.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,149 @@
# K3s 1.37 e2e: cluster DNS dies after the webhook-TLS node restart

Status report for PR #415 (`chore/dependency-upgrades-2026-10-05`), 2026-10-05. The PR moves
the e2e cluster from `rancher/k3s:v1.36.4-k3s1` to `rancher/k3s:v1.37.1-k3s1`. This report covers
the e2e failures that followed. The k3s bump stays in the PR while this is open.

## Summary

- On k3s 1.37.1, four of the seven CI bring-ups that restart the k3d server node failed.
The manager never turned Ready because it could not resolve
`valkey.valkey-e2e.svc.cluster.local`.
- The failure is intermittent. The same legs also pass: `full-core` and `full-manager` failed and
then passed on re-run, and `bi-directional` passed through the same restart. Only
`source-cluster` failed twice.
- Before this fix, CI ran every e2e leg on a single-node cluster (`k3d_agent_count: "0"`), so the
restart bounced every pod: the manager, CoreDNS and Valkey all came back on the same restarted
node. The local default was 3 agents, where most pods lived on agents and survived the restart.
- The fix gives every e2e cluster one agent and taints the server `NoSchedule`. With it, all four
restart legs recovered on their first attempt in CI, `source-cluster` included.
- A local bring-up on 1.37.1 (3 agents) recovered cleanly. A minimal single-node cluster and a
1-server-3-agent cluster, on 1.37.1 and on 1.36.5, kept DNS working through the same restart.
- Nothing points at the Go module bumps: the failure is a DNS timeout before any spec runs.

## The failure

`hack/e2e/inject-webhook-tls.sh` writes the final audit webhook kubeconfig and then runs
`docker restart k3d-<cluster>-server-0` so kube-apiserver picks it up. The API returns within
seconds (`✅ Cluster API healthy`). The manager Deployment then never finishes its rollout:

```text
❌ Manager did not recover after the node restart; dumping diagnostics.
gitops-reverser-57b99fc7c6-ffpvl 0/1 Running 0 5m15s 10.42.0.28 k3d-gitops-reverser-test-e2e-server-0
...
"audit Redis not yet reachable; pod stays not-ready until first connection",
"error":"dial tcp: lookup valkey.valkey-e2e.svc.cluster.local: i/o timeout"
```

The readiness gate waits on its first Redis connection, so a pod without DNS stays `0/1` until
`MANAGER_ROLLOUT_TIMEOUT` (300s) expires. `prepare-e2e` fails, and Ginkgo reports `Ran 0 of 118
Specs`.

Only legs that install with attribution (audit TLS) do the restart. `quickstart-install` and
`image-refresh` skip it ("configured-author install; skipping audit webhook TLS injection").

## Evidence

Every node restart in CI on this PR, plus the earlier 1.36.4 case. All are single-node clusters,
so `server-0` is the only node a pod can land on:

| Run | Leg | k3s | Outcome | Manager pod placed on |
|---|---|---|---|---|
| 37326858093 attempt 1 | full-core | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.37) |
| 37326858093 attempt 1 | full-manager | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.36) |
| 37326858093 attempt 1 | source-cluster | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.37) |
| 37326858093 attempt 1 | bi-directional | 1.37.1 | recovered, passed | not dumped |
| 37326858093 attempt 2 | full-core | 1.37.1 | recovered, passed | not dumped |
| 37326858093 attempt 2 | source-cluster | 1.37.1 | DNS timeout, failed | `server-0` (10.42.0.28) |
| 37326858093 attempt 2 | full-manager | 1.37.1 | recovered, passed (78 of 118 specs) | not dumped |
| 36765899550 (2026-09-30) | source-cluster | 1.36.4 | DNS timeout, failed | `server-0` (10.42.0.35) |

Before this PR, the merges of #410 and #413 ran all six legs green on 1.36.4.

The placement column carries no signal on its own: with zero agents there is nowhere else to go.

## Local reproduction

All local runs used Docker 29.2.1 on the devcontainer host.

1. **Minimal cluster, single node, 1.37.1.** DNS from a pod worked before and after
`docker restart` of the server.
2. **Minimal cluster, 1 server and 3 agents, 1.37.1 and 1.36.5 side by side.** A busybox pod pinned
to each node resolved `kubernetes.default.svc.cluster.local` before the restart, 45s after
and 105s after, on both versions. CoreDNS moved to an agent after the restart.
3. **The e2e bring-up on 1.37.1** (`task CTX=k3d-gitops-reverser-k137-e2e prepare-e2e`, same
arguments as CI). The restart recovered: `deployment "gitops-reverser" successfully rolled
out`, `✅ Audit pipeline live`, `✅ Webhook TLS injection complete`. The new manager pod landed on
`agent-1`; CoreDNS and Valkey were on `agent-0`.

None of these reproduced the failure. Run 3 used the local default of 3 agents, so the restart
left most workloads untouched; it does not test the single-node case CI runs. Both e2e clusters
(`gitops-reverser-test-e2e`, `gitops-reverser-k137-e2e`) have since been deleted to free memory.
The host had about 25 GiB available during the runs, with several other k3d clusters up.

## Working hypothesis

On a single-node cluster, `docker restart` brings every pod back at once on the restarted node,
and the manager sometimes comes up without working cluster DNS. On 1.37.1 that happens far more
often than on 1.36.4 (4 of 7 versus one recorded case).

What it does not explain yet:

- The minimal single-node repro on 1.37.1 recovered, but it ran once and carried none of the e2e
load (audit webhook flags, Flux, cert-manager, Valkey, Gitea).
- Whether the break is pod networking after the container restart (flannel or kube-proxy state),
CoreDNS coming back late or wrong, or the manager's resolver caching an early failure. The
dumps hold no CoreDNS state, no node state and no in-pod probe.

The 3-agent local runs never failed, which fits the hypothesis: there, the manager and Valkey
usually sit on agents that the restart does not touch.

## Recommended next steps

1. **Make the next failure explain itself.** Extend `wait_for_manager_rollout`'s failure dump in
`hack/e2e/inject-webhook-tls.sh` with:
- `kubectl get pods -n kube-system -o wide` (CoreDNS placement)
- `kubectl get nodes -o wide` plus the server's flannel and kube-proxy state
- an `nslookup` from a fresh pod

One more CI failure then separates pod networking from CoreDNS.
2. **Reproduce the CI shape locally.** Bring up 1.37.1 with `K3D_AGENT_COUNT=0`, the way CI does,
and repeat the restart a few times.
3. **Mitigation: keep workloads off the node that restarts.** Implemented in PR #415: e2e
clusters, local and CI, now run one server and one agent, with the server tainted
`node-role.kubernetes.io/control-plane:NoSchedule`. The restart then bounces only the control
plane, and the manager, Valkey and Gitea keep running on the agent. This also matches how
production clusters keep workloads off control-plane nodes. The cost is one extra node at
bring-up, in CI as well, which ran with zero agents since 2026-06-01 to start faster. The local
default drops from three agents to one: all k3d nodes share the host CPU, so extra agents
bought memory use and startup time rather than capacity. Step 1's diagnostics ship with it.

## Result

PR #415 commit `5944cea8`, k3s 1.37.1, one agent and a tainted server. Every leg passed on its
first attempt:

| Leg | Server tainted | Node restart | Recovered | DNS timeouts | Specs run |
|---|---|---|---|---|---|
| full-manager | yes | yes | yes | 0 | 78 of 118 |
| full-core | yes | yes | yes | 0 | 16 of 118 |
| source-cluster | yes | yes | yes | 0 | 11 of 118 |
| bi-directional | yes | yes | yes | 0 | 4 of 118 |
| quickstart-install | yes | skipped | n/a | 0 | 1 of 118 |
| image-refresh | yes | skipped | n/a | 0 | 6 of 118 |

One green run does not prove the cause, since the failure was intermittent. It does show the
restart no longer touches the workloads the manager depends on. If a restart ever fails again,
the extended failure dump records why.

## The rest of PR #415

- **Lint fix:** `hack/metricnames.sh` skips `CHANGELOG.md`, the file that made release PR #414 fail
lint. CI Lint is green on #415. #414 needs a re-run or rebase after #415 merges.
- **Go modules:** otel 1.47, controller-runtime 0.25.2, kustomize api/kyaml 0.21.2, go-git and
go-billy v6 beta.1. Unit tests green. Every leg that got past bring-up passed: `full-core` and
`full-manager` (re-runs), `bi-directional`, `quickstart-install`, `image-refresh`. So go-git
beta.1 has run every suite except `source-cluster` successfully.
- **Devcontainer tools:** CI rebuilt the image with them, and Lint passed on it.
- **CodeRabbit:** one minor note on the README versions line (active voice, backticks on `envtest`
and `k3s`). Addressed.
29 changes: 15 additions & 14 deletions go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -7,8 +7,8 @@ require (
github.com/alicebob/miniredis/v2 v2.39.0
github.com/cespare/xxhash/v2 v2.3.0
github.com/fluxcd/pkg/apis/meta v1.32.0
github.com/go-git/go-billy/v6 v6.0.0-alpha.2
github.com/go-git/go-git/v6 v6.0.0-alpha.5
github.com/go-git/go-billy/v6 v6.0.0-beta.1
github.com/go-git/go-git/v6 v6.0.0-beta.1
github.com/go-logr/logr v1.4.4
github.com/onsi/ginkgo/v2 v2.33.0
github.com/onsi/gomega v1.44.0
Expand All @@ -17,10 +17,10 @@ require (
github.com/prometheus/common v0.72.0
github.com/redis/go-redis/v9 v9.22.0
github.com/stretchr/testify v1.12.1
go.opentelemetry.io/otel v1.46.0
go.opentelemetry.io/otel/exporters/prometheus v0.68.0
go.opentelemetry.io/otel/metric v1.46.0
go.opentelemetry.io/otel/sdk/metric v1.46.0
go.opentelemetry.io/otel v1.47.0
go.opentelemetry.io/otel/exporters/prometheus v0.69.0
go.opentelemetry.io/otel/metric v1.47.0
go.opentelemetry.io/otel/sdk/metric v1.47.0
go.uber.org/zap v1.28.0
golang.org/x/crypto v0.57.0
gopkg.in/yaml.v3 v3.0.1
Expand All @@ -31,9 +31,9 @@ require (
k8s.io/client-go v0.37.1
k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3
sigs.k8s.io/cli-utils v0.37.2
sigs.k8s.io/controller-runtime v0.25.1
sigs.k8s.io/kustomize/api v0.21.1
sigs.k8s.io/kustomize/kyaml v0.21.1
sigs.k8s.io/controller-runtime v0.25.2
sigs.k8s.io/kustomize/api v0.21.2
sigs.k8s.io/kustomize/kyaml v0.21.2
sigs.k8s.io/yaml v1.6.0
)

Expand All @@ -42,7 +42,7 @@ require (
filippo.io/hpke v0.4.0 // indirect
github.com/Masterminds/semver/v3 v3.5.0 // indirect
github.com/Microsoft/go-winio v0.6.2 // indirect
github.com/ProtonMail/go-crypto v1.4.1 // indirect
github.com/ProtonMail/go-crypto v1.5.2 // indirect
github.com/antlr4-go/antlr/v4 v4.13.1 // indirect
github.com/beorn7/perks v1.0.1 // indirect
github.com/blang/semver/v4 v4.0.0 // indirect
Expand Down Expand Up @@ -88,10 +88,10 @@ require (
github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee // indirect
github.com/monochromegane/go-gitignore v0.0.0-20200626010858-205db1a8cc00 // indirect
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
github.com/pjbgf/sha1cd v0.6.0 // indirect
github.com/pjbgf/sha1cd v0.7.0 // indirect
github.com/prometheus/client_model v0.6.3 // indirect
github.com/prometheus/otlptranslator v1.0.0 // indirect
github.com/prometheus/procfs v0.21.1 // indirect
github.com/prometheus/procfs v0.22.0 // indirect
github.com/sergi/go-diff v1.4.0 // indirect
github.com/spf13/cobra v1.10.2 // indirect
github.com/spf13/pflag v1.0.10 // indirect
Expand All @@ -102,8 +102,9 @@ require (
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 // indirect
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 // indirect
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 // indirect
go.opentelemetry.io/otel/sdk v1.46.0 // indirect
go.opentelemetry.io/otel/trace v1.46.0 // indirect
go.opentelemetry.io/otel/log v1.47.0 // indirect
go.opentelemetry.io/otel/sdk v1.47.0 // indirect
go.opentelemetry.io/otel/trace v1.47.0 // indirect
go.opentelemetry.io/proto/otlp v1.11.0 // indirect
go.uber.org/atomic v1.11.0 // indirect
go.uber.org/multierr v1.11.0 // indirect
Expand Down
Loading
Loading