From fc036d53d57af8d4d5e17dd4dff780eabd4a8bff Mon Sep 17 00:00:00 2001 From: Niek Otten Date: Fri, 18 Sep 2026 15:40:42 +0200 Subject: [PATCH] ci(deploy): deploy the stack to Talos on merge to main (refs #175) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The chart has been deployable by hand since #25 and linted in CI since #168; this makes a merged PR actually ship it to the lab server's Talos VM. Neither the Kubernetes API nor the in-cluster registry is publicly reachable, so the job forwards 6443, 30500 and 30141 over the same SSH hop into the Fedora host that the Gitea-runner pipeline uses. That splits the registry into two names for one store: images are pushed through the tunnel to localhost:30500, and the node pulls them from its own NodePort — the address its registry-mirror patch trusts over plain HTTP. It deploys with `make k8s-reseed` rather than `make k8s-up`: the bootstrap Jobs are idempotent, and deleting them first is what stops a changed Job template from wedging `helm upgrade`. The nine deployments are then rolled explicitly, because `dev` is a mutable tag and helm sees an unchanged pod template. PR CI is the merge gate, so this workflow does not re-run the checks. Deploys queue instead of cancelling: a `helm upgrade` killed half-way leaves the release in `pending-upgrade` and needs unwedging by hand. Co-Authored-By: Claude Opus 5 (1M context) --- .gitea/workflows/deploy.yaml | 110 ++++++++++++++++++++++++++++++ docs/runbooks/ci.md | 4 ++ docs/runbooks/kubernetes-talos.md | 41 +++++++++++ 3 files changed, 155 insertions(+) create mode 100644 .gitea/workflows/deploy.yaml diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml new file mode 100644 index 0000000..8e9d5f6 --- /dev/null +++ b/.gitea/workflows/deploy.yaml @@ -0,0 +1,110 @@ +name: Deploy to Talos + +# A merge to main ships the stack to the Talos cluster on the lab server +# (docs/runbooks/kubernetes-talos.md §9). PR CI is the merge gate, so main is +# green by construction — this workflow only deploys. +on: + push: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +# Queue deploys, never cancel one: a helm upgrade killed half-way leaves the +# release in `pending-upgrade` and the next run has to be unwedged by hand. +concurrency: + group: deploy-talos + cancel-in-progress: false + +jobs: + deploy: + runs-on: ubuntu-latest + env: + # The Talos VM as seen from the Fedora host (libvirt guest IP), and the + # address a browser uses to reach the cluster. `localhost` is deliberate: + # the portals' PKCE needs a secure context, so they are reached over + # `kubectl port-forward` — runbook §5. Override with repo variables. + TALOS_VM_IP: ${{ vars.TALOS_VM_IP }} + TALOS_HOST: ${{ vars.TALOS_HOST }} + steps: + - uses: https://github.com/actions/checkout@v4 + + # Pinned static binaries, the same URLs the Talos runbook §0 gives a + # developer and the same helm the `k8s` CI job uses — no action to vet. + - name: Install kubectl, helm and crane + run: | + set -euo pipefail + bin="$HOME/.local/bin"; mkdir -p "$bin" + curl -sSLo "$bin/kubectl" https://dl.k8s.io/release/v1.37.0/bin/linux/amd64/kubectl + curl -sSL https://get.helm.sh/helm-v3.16.4-linux-amd64.tar.gz | tar xz -O linux-amd64/helm > "$bin/helm" + curl -sSL https://github.com/google/go-containerregistry/releases/download/v0.20.2/go-containerregistry_Linux_x86_64.tar.gz | tar xz -O crane > "$bin/crane" + chmod +x "$bin"/{kubectl,helm,crane} + echo "$bin" >> "$GITHUB_PATH" + + # The cluster's API and its registry are only reachable through the Fedora + # host, so forward both to the runner. 30141 is the openbaar portal, for + # the smoke at the end. + - name: Tunnel the Talos API + registry through the Fedora host + env: + SSH_KEY: ${{ secrets.TALOS_SSH_KEY }} + run: | + set -euo pipefail + : "${TALOS_VM_IP:=192.168.122.173}" + umask 077 + printf '%s\n' "$SSH_KEY" > ~/.ssh_talos + ssh -i ~/.ssh_talos -o StrictHostKeyChecking=no -o IdentitiesOnly=yes \ + -o ExitOnForwardFailure=yes -p 6667 -f -N \ + -L 6443:$TALOS_VM_IP:6443 \ + -L 30500:$TALOS_VM_IP:30500 \ + -L 30141:$TALOS_VM_IP:30141 \ + user@labs.respellion.tech + + # The kubeconfig's server must be https://127.0.0.1:6443 — Talos puts + # 127.0.0.1 in the apiserver cert SANs, so TLS verification still holds + # through the tunnel. + - name: Write the kubeconfig + env: + KUBECONFIG_B64: ${{ secrets.TALOS_KUBECONFIG }} + run: | + set -euo pipefail + umask 077 + base64 -d <<< "$KUBECONFIG_B64" > "$RUNNER_TEMP/kubeconfig" + echo "KUBECONFIG=$RUNNER_TEMP/kubeconfig" >> "$GITHUB_ENV" + kubectl --kubeconfig "$RUNNER_TEMP/kubeconfig" get nodes + + # Idempotent; also makes a first deploy onto a bare cluster work. The + # registry's storage is an emptyDir, so a replaced pod loses the images — + # which the push in the next step puts back anyway. + - name: Ensure the in-cluster registry + run: make k8s-registry + + # Push through the tunnel (localhost), pull from the node's own NodePort + # (the address in the Talos registry-mirror patch) — same registry, two + # names, so the two `make` calls get different K8S_REGISTRY values. + - name: Build and push the images + run: make k8s-images K8S_REGISTRY=localhost:30500 + + # k8s-reseed = seed configmaps + helm upgrade + re-run the bootstrap jobs. + # The jobs are idempotent, and deleting them first is what keeps a changed + # Job template from wedging the upgrade (`cannot patch … with kind Job`). + - name: Deploy the chart + run: make k8s-reseed TALOS_HOST=${TALOS_HOST:-localhost} K8S_REGISTRY=${TALOS_VM_IP:-192.168.122.173}:30500 + + # `dev` is a mutable tag and helm sees an unchanged pod template, so the + # new images only land on a restart (pullPolicy is already Always). + - name: Roll the services onto the new images + run: | + set -euo pipefail + svcs="acl domain bff event-subscriber projection-api self-service openbaar behandel beheer" + kubectl -n big rollout restart deploy $svcs + kubectl -n big rollout status --timeout=300s deploy $svcs + + # Proves portal → Caddy → BFF → projection end to end. An empty register is + # a pass; a 502 or a timeout is not. + - name: Smoke the public register + run: curl -fsS --retry 10 --retry-delay 6 --retry-all-errors http://localhost:30141/openbaar/register + + - name: Pods on failure + if: failure() + run: kubectl -n big get pods,jobs || true diff --git a/docs/runbooks/ci.md b/docs/runbooks/ci.md index 0a22a95..76dd9c1 100644 --- a/docs/runbooks/ci.md +++ b/docs/runbooks/ci.md @@ -31,6 +31,10 @@ and CI cannot drift: > services by **container IP** (the runner can't reach published ports — see > [gitea-actions-gotchas.md §5/§6](gitea-actions-gotchas.md)). +A second workflow, `.gitea/workflows/deploy.yaml`, deploys the stack to the Talos +cluster on the lab server when a PR is merged to `main` — see +[kubernetes-talos.md §9](kubernetes-talos.md) for its secrets and the SSH tunnel it needs. + All `uses:` references are absolute, tag-pinned URLs (`https://github.com/actions/checkout@v4`, `https://github.com/actions/setup-dotnet@v4`) per CLAUDE.md §8.7 and §15 — Gitea Actions resolves them from GitHub. diff --git a/docs/runbooks/kubernetes-talos.md b/docs/runbooks/kubernetes-talos.md index 52271eb..cf73ea9 100644 --- a/docs/runbooks/kubernetes-talos.md +++ b/docs/runbooks/kubernetes-talos.md @@ -360,6 +360,47 @@ immutable, so `helm upgrade` is rejected with `cannot patch "…" with kind Job` | Pods `Evicted` / `OOMKilled` | the VM is too small (§0) | | A Job shows `BackoffLimitExceeded` | read it: `kubectl -n big logs job/` | +## 9. Deploying on merge to main + +`.gitea/workflows/deploy.yaml` runs the §3–§4 steps against the **lab server's** Talos VM +every time a PR is squash-merged to `main` (and on demand via *Run workflow*). PR CI is the +merge gate, so the workflow deploys without re-running the checks. + +The cluster's API and registry are not exposed publicly, so the job forwards them over the +same SSH hop the Gitea-runner pipeline uses: + +``` +ssh -p 6667 user@labs.respellion.tech -L 6443 -L 30500 -L 30141 → +``` + +Consequences worth knowing: + +- Images are **pushed** to `localhost:30500` (the tunnel) and **pulled** by the node from + `:30500` (its own NodePort, the address in the Talos registry-mirror patch). + Same registry, two names — hence the two `K8S_REGISTRY` values in the workflow. +- It calls `make k8s-reseed`, not `make k8s-up`: the bootstrap Jobs are idempotent, and + deleting them first is what stops a changed Job template from wedging `helm upgrade` (§7). +- `dev` is a mutable tag, so a `rollout restart` of the nine repo deployments is what + actually puts the new images in the pods. +- Deploys **queue** (`cancel-in-progress: false`): a helm upgrade killed half-way leaves the + release in `pending-upgrade`, which has to be unwedged by hand. + +Settings, all on the repository in Gitea: + +| Kind | Name | What | +|---|---|---| +| Secret | `TALOS_SSH_KEY` | private key for `user@labs.respellion.tech` (the Fedora host) | +| Secret | `TALOS_KUBECONFIG` | base64 of the kubeconfig, **`server: https://127.0.0.1:6443`** — Talos puts `127.0.0.1` in the apiserver cert SANs, so TLS still verifies through the tunnel | +| Variable | `TALOS_VM_IP` | the VM's libvirt address (default `192.168.122.173`) | +| Variable | `TALOS_HOST` | the browser-facing host baked into Keycloak's issuer (default `localhost`, see §5) | + +The last step smokes `GET /openbaar/register` through the openbaar portal, which exercises +portal → Caddy → BFF → projection. An empty register passes; a 502 does not. + +Not covered: the portals still need `make k8s-portals` (or an SSH forward) to be usable in a +browser, because PKCE needs a secure context (§5). Giving the server a hostname + TLS is the +upgrade path. + ## What is not ported - **Observability** (Tempo, Prometheus, Grafana) is defined but disabled — those are built -- 2.54.0