Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
dfa1a370ac | ||
|
|
e8cb1ec7e9 | ||
|
|
de6db7d35b | ||
|
|
d17e79959b | ||
|
|
fc036d53d5 |
@@ -0,0 +1,126 @@
|
||||
name: Deploy to Talos
|
||||
|
||||
# A merge to main ships the stack to the Talos cluster on the lab server
|
||||
# (docs/runbooks/kubernetes-talos.md §9). PR CI is the merge gate, so main is
|
||||
# green by construction — this workflow only deploys.
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Queue deploys, never cancel one: a helm upgrade killed half-way leaves the
|
||||
# release in `pending-upgrade` and the next run has to be unwedged by hand.
|
||||
concurrency:
|
||||
group: deploy-talos
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
# The Talos VM as seen from the Fedora host (libvirt guest IP), and the
|
||||
# address a browser uses to reach the cluster. `localhost` is deliberate:
|
||||
# the portals' PKCE needs a secure context, so they are reached over
|
||||
# `kubectl port-forward` — runbook §5. Override with repo variables.
|
||||
TALOS_VM_IP: ${{ vars.TALOS_VM_IP }}
|
||||
TALOS_HOST: ${{ vars.TALOS_HOST }}
|
||||
# Set it and the stack is published over TLS on <sub>.<domain> by the
|
||||
# in-cluster edge (ADR-0035, runbook §10). Empty = NodePorts, as before.
|
||||
PUBLIC_DOMAIN: ${{ vars.PUBLIC_DOMAIN }}
|
||||
PUBLIC_EMAIL: ${{ vars.PUBLIC_EMAIL }}
|
||||
steps:
|
||||
- uses: https://github.com/actions/checkout@v4
|
||||
|
||||
# Pinned static binaries, the same URLs the Talos runbook §0 gives a
|
||||
# developer and the same helm the `k8s` CI job uses — no action to vet.
|
||||
- name: Install kubectl, helm and crane
|
||||
run: |
|
||||
set -euo pipefail
|
||||
bin="$HOME/.local/bin"; mkdir -p "$bin"
|
||||
curl -sSLo "$bin/kubectl" https://dl.k8s.io/release/v1.37.0/bin/linux/amd64/kubectl
|
||||
curl -sSL https://get.helm.sh/helm-v3.16.4-linux-amd64.tar.gz | tar xz -O linux-amd64/helm > "$bin/helm"
|
||||
curl -sSL https://github.com/google/go-containerregistry/releases/download/v0.20.2/go-containerregistry_Linux_x86_64.tar.gz | tar xz -O crane > "$bin/crane"
|
||||
chmod +x "$bin"/{kubectl,helm,crane}
|
||||
echo "$bin" >> "$GITHUB_PATH"
|
||||
|
||||
# The cluster's API and its registry are only reachable through the Fedora
|
||||
# host, so forward both to the runner. 30141 is the openbaar portal, for
|
||||
# the smoke at the end.
|
||||
- name: Tunnel the Talos API + registry through the Fedora host
|
||||
env:
|
||||
SSH_KEY: ${{ secrets.TALOS_SSH_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
: "${TALOS_VM_IP:=192.168.122.173}"
|
||||
umask 077
|
||||
printf '%s\n' "$SSH_KEY" > ~/.ssh_talos
|
||||
ssh -i ~/.ssh_talos -o StrictHostKeyChecking=no -o IdentitiesOnly=yes \
|
||||
-o ExitOnForwardFailure=yes -p 6667 -f -N \
|
||||
-L 6443:$TALOS_VM_IP:6443 \
|
||||
-L 30500:$TALOS_VM_IP:30500 \
|
||||
-L 30141:$TALOS_VM_IP:30141 \
|
||||
user@labs.respellion.tech
|
||||
|
||||
# The kubeconfig's server must be https://127.0.0.1:6443 — Talos puts
|
||||
# 127.0.0.1 in the apiserver cert SANs, so TLS verification still holds
|
||||
# through the tunnel.
|
||||
- name: Write the kubeconfig
|
||||
env:
|
||||
KUBECONFIG_B64: ${{ secrets.TALOS_KUBECONFIG }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
umask 077
|
||||
base64 -d <<< "$KUBECONFIG_B64" > "$RUNNER_TEMP/kubeconfig"
|
||||
echo "KUBECONFIG=$RUNNER_TEMP/kubeconfig" >> "$GITHUB_ENV"
|
||||
kubectl --kubeconfig "$RUNNER_TEMP/kubeconfig" get nodes
|
||||
|
||||
# Idempotent; also makes a first deploy onto a bare cluster work. The
|
||||
# registry's storage is an emptyDir, so a replaced pod loses the images —
|
||||
# which the push in the next step puts back anyway.
|
||||
- name: Ensure the in-cluster registry
|
||||
run: make k8s-registry
|
||||
|
||||
# Push through the tunnel (localhost), pull from the node's own NodePort
|
||||
# (the address in the Talos registry-mirror patch) — same registry, two
|
||||
# names, so the two `make` calls get different K8S_REGISTRY values.
|
||||
- name: Build and push the images
|
||||
run: make k8s-images K8S_REGISTRY=localhost:30500
|
||||
|
||||
# k8s-reseed = seed configmaps + helm upgrade + re-run the bootstrap jobs.
|
||||
# The jobs are idempotent, and deleting them first is what keeps a changed
|
||||
# Job template from wedging the upgrade (`cannot patch … with kind Job`).
|
||||
- name: Deploy the chart
|
||||
run: |
|
||||
set -euo pipefail
|
||||
publish="${PUBLIC_DOMAIN:+--set public.domain=$PUBLIC_DOMAIN --set public.email=${PUBLIC_EMAIL:-}}"
|
||||
make k8s-reseed \
|
||||
TALOS_HOST=${TALOS_HOST:-localhost} \
|
||||
K8S_REGISTRY=${TALOS_VM_IP:-192.168.122.173}:30500 \
|
||||
K8S_SET="$publish"
|
||||
|
||||
# `dev` is a mutable tag and helm sees an unchanged pod template, so the
|
||||
# new images only land on a restart (pullPolicy is already Always).
|
||||
- name: Roll the services onto the new images
|
||||
run: |
|
||||
set -euo pipefail
|
||||
svcs="acl domain bff event-subscriber projection-api self-service openbaar behandel beheer"
|
||||
kubectl -n big rollout restart deploy $svcs
|
||||
kubectl -n big rollout status --timeout=300s deploy $svcs
|
||||
|
||||
# Proves portal → Caddy → BFF → projection end to end. An empty register is
|
||||
# a pass; a 502 or a timeout is not.
|
||||
- name: Smoke the public register
|
||||
run: curl -fsS --retry 10 --retry-delay 6 --retry-all-errors http://localhost:30141/openbaar/register
|
||||
|
||||
# Cluster-wide, not just `big`: the first thing that can fail is the registry
|
||||
# in its own namespace, and a scheduling problem shows up in the events, not
|
||||
# in `rollout status` — which only ever says "timed out waiting".
|
||||
- name: Pods and events on failure
|
||||
if: failure()
|
||||
run: |
|
||||
kubectl get pods -A -o wide || true
|
||||
kubectl -n big get jobs || true
|
||||
kubectl get events -A --sort-by=.lastTimestamp | tail -30 || true
|
||||
@@ -352,7 +352,6 @@ K8S_IMAGES := acl domain bff event-subscriber projection-api self-service open
|
||||
k8s-lint:
|
||||
helm lint $(K8S_CHART)
|
||||
helm template big $(K8S_CHART) -n $(K8S_NS) --set images.registry=registry.invalid:5000 >/dev/null
|
||||
python3 infra/helm/check-issuer.py
|
||||
|
||||
## k8s-drift: fail if compose and the Helm chart describe different stacks
|
||||
# Compose is CI-canonical (ADR-0033) and the chart is a transcription of it; this
|
||||
|
||||
@@ -1,102 +0,0 @@
|
||||
# ADR-0035: The public TLS edge is a Caddy deployment in the cluster
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-09-18
|
||||
- **Deciders:** Respellion engineering
|
||||
- **Slice:** [#177](https://git.labs.respellion.tech/eho/register-referentie/issues/177)
|
||||
|
||||
## Context
|
||||
|
||||
The stack deploys to a Talos VM on the lab server (ADR-0033, issue #175). Until now it was
|
||||
only usable through five SSH port-forwards: the portals' OIDC flow uses PKCE, PKCE needs
|
||||
`crypto.subtle`, and browsers expose that only in a **secure context** — HTTPS or an origin
|
||||
on `localhost`. A NodePort on the VM's address is neither, so the deployment was pinned to
|
||||
`host: localhost` and every viewer had to forward all five browser-facing ports (a portal
|
||||
without Keycloak on the same `localhost:30180` fails on the discovery document).
|
||||
|
||||
That is not a demo anyone can be sent a link to. We want public hostnames with real
|
||||
certificates — and we want the routing and the certificates to be cluster state, not
|
||||
host-side configuration that no `helm upgrade` can see.
|
||||
|
||||
The public IP is on the Fedora host (`46.224.220.37`); the cluster is a libvirt guest
|
||||
behind it.
|
||||
|
||||
## Decision
|
||||
|
||||
**Terminate TLS in the cluster, with a Caddy deployment rendered by the chart
|
||||
(`templates/edge.yaml`), and give the Fedora host nothing but a layer-4 forward.**
|
||||
|
||||
- `public.domain` is the single switch. Empty — the default, and what compose and CI use —
|
||||
renders nothing: the stack is reached on its NodePorts and `host` pins the OIDC origin
|
||||
exactly as before. Set it, and the edge appears.
|
||||
- `public.routes` maps a subdomain to an in-cluster `service:port`. Caddy proxies to the
|
||||
**ClusterIP** services, so a public deployment does not use the browser-facing NodePorts
|
||||
at all.
|
||||
- Caddy obtains and renews certificates itself (ACME HTTP-01). There is no cert-manager.
|
||||
- The host forwards `:80`/`:443` to two NodePorts with two `firewall-cmd
|
||||
--add-forward-port` rules. No TLS, no routing, no per-service knowledge there — adding a
|
||||
portal is a chart change, not a host change.
|
||||
- `KC_HOSTNAME` and the portals' `config.json` stop being `host` + NodePort. Both now come
|
||||
from one helper, `big.keycloakUrl`, so the issuer Keycloak pins and the authority the
|
||||
portals are configured with cannot drift apart (ADR-0010).
|
||||
|
||||
### Alternatives considered
|
||||
|
||||
- **Caddy on the Fedora host.** Fewest moving parts — but the routing table and the
|
||||
certificates would live outside the cluster, in a file no deployment touches, and adding
|
||||
a portal would mean editing a host we deploy to over SSH. Rejected on exactly the ground
|
||||
this ADR exists to record.
|
||||
- **Traefik or ingress-nginx, plus cert-manager.** The conventional answer, and the right
|
||||
one for a cluster with many teams and changing hostnames. Here it buys a controller, a
|
||||
set of CRDs and Ingress objects to describe five hostnames that never change — and
|
||||
cert-manager to do what Caddy already does unprompted.
|
||||
- **A `LoadBalancer` service (MetalLB).** Solves address allocation, which is not the
|
||||
problem; the node has exactly one address and it still is not the public one.
|
||||
- **Keep the SSH forwards.** Free, and genuinely fine for one developer. It is not a demo
|
||||
you can send to someone.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive**
|
||||
|
||||
- No new dependency: the four portals already run `caddy:2-alpine` (ADR-0034), whose
|
||||
ceiling note called this out — *"a real hostname makes TLS a one-line `Caddyfile`
|
||||
change"*. This is that change.
|
||||
- Routing is cluster state: `kubectl -n big get cm caddy-edge-config -o yaml` is the whole
|
||||
truth about what is published, and `helm upgrade` is how it changes.
|
||||
- The secure context is real, so `TALOS_HOST=localhost` and the five forwards disappear —
|
||||
and with them the class of failure where a mismatched issuer logs the user out silently.
|
||||
- Nothing changes for compose, CI or a laptop cluster: with `public.domain` empty the
|
||||
rendered manifests are byte-identical to before.
|
||||
|
||||
**Negative / costs**
|
||||
|
||||
- The host forward is irreducible. Two firewalld rules, applied by hand once, with `sudo`
|
||||
on a machine our pipeline reaches only over SSH. If someone rebuilds that host, the stack
|
||||
is unreachable until they are re-applied, and nothing in the cluster can tell them so.
|
||||
- **Certificates need a volume.** On the default `emptyDir` every pod restart asks Let's
|
||||
Encrypt again, and its duplicate-certificate limit is five per week — a handful of
|
||||
restarts and the edge serves an untrusted certificate for a week. `persistence.storageClass`
|
||||
stops being optional for anything public (runbook §6).
|
||||
- **All five hostnames are published, including `behandel` and `beheer`**, which approve
|
||||
registrations and administer the register. They are protected by synthetic accounts with
|
||||
well-known passwords, and by MFA on the medewerker realm (ADR-0031). That is a deliberate
|
||||
choice for a demonstration environment holding synthetic data only, and it is the reason
|
||||
this bullet is in the ADR rather than in a comment: if this stack ever holds anything
|
||||
real, this decision is the first one to revisit.
|
||||
- One more workload in the chart with no counterpart in compose — compose has no edge
|
||||
because it has no hostname. The drift check (`make k8s-drift`) renders the defaults, so
|
||||
it does not see it.
|
||||
- `auth` is load-bearing: `big.keycloakUrl` builds the issuer from that subdomain, so
|
||||
renaming the key in `public.routes` without the helper breaks every login. Both carry a
|
||||
comment saying so.
|
||||
|
||||
- ponytail ceiling: one replica, no HSTS, no security headers beyond Caddy's defaults, no
|
||||
rate limiting, and HTTP-01 rather than DNS-01 (so a wildcard certificate is not
|
||||
available). Upgrade path in that order; DNS-01 first if the subdomain list ever grows.
|
||||
|
||||
## Coupling rules touched (CLAUDE.md §8)
|
||||
|
||||
None. §8.3 holds — the browser reaches a portal, the portal reverse-proxies its own BFF
|
||||
group, and the edge is in front of all of it. The edge terminates TLS and routes by
|
||||
hostname; it does not know what any service does.
|
||||
@@ -31,6 +31,10 @@ and CI cannot drift:
|
||||
> services by **container IP** (the runner can't reach published ports — see
|
||||
> [gitea-actions-gotchas.md §5/§6](gitea-actions-gotchas.md)).
|
||||
|
||||
A second workflow, `.gitea/workflows/deploy.yaml`, deploys the stack to the Talos
|
||||
cluster on the lab server when a PR is merged to `main` — see
|
||||
[kubernetes-talos.md §9](kubernetes-talos.md) for its secrets and the SSH tunnel it needs.
|
||||
|
||||
All `uses:` references are absolute, tag-pinned URLs (`https://github.com/actions/checkout@v4`,
|
||||
`https://github.com/actions/setup-dotnet@v4`) per CLAUDE.md §8.7 and §15 — Gitea
|
||||
Actions resolves them from GitHub.
|
||||
|
||||
@@ -222,9 +222,6 @@ string, so the port the browser uses has to match the one baked into `config.jso
|
||||
This is the same mechanism `infra/host-browser.yml` uses for the compose stack (which pins
|
||||
`localhost:8180`); only the addresses differ.
|
||||
|
||||
All of this is what §10 removes: with a public domain the portals have real certificates,
|
||||
so the browser gets its secure context and no forwarding is involved.
|
||||
|
||||
### The admin UIs work straight off the NodePorts
|
||||
|
||||
These are server-rendered and need no secure context, so they are reachable at the VM's
|
||||
@@ -363,67 +360,96 @@ immutable, so `helm upgrade` is rejected with `cannot patch "…" with kind Job`
|
||||
| Pods `Evicted` / `OOMKilled` | the VM is too small (§0) |
|
||||
| A Job shows `BackoffLimitExceeded` | read it: `kubectl -n big logs job/<name>` |
|
||||
|
||||
## 10. Publishing it on a public domain
|
||||
## 9. Deploying on merge to main
|
||||
|
||||
By default the stack has no hostname: it is reached on NodePorts, and §5's secure-context
|
||||
problem forces `TALOS_HOST=localhost` plus five SSH forwards. Setting `public.domain` puts a
|
||||
Caddy deployment in front of it that terminates TLS for real hostnames (ADR-0035), and the
|
||||
forwards go away.
|
||||
`.gitea/workflows/deploy.yaml` runs the §3–§4 steps against the **lab server's** Talos VM
|
||||
every time a PR is squash-merged to `main` (and on demand via *Run workflow*). PR CI is the
|
||||
merge gate, so the workflow deploys without re-running the checks.
|
||||
|
||||
### Once, outside the cluster
|
||||
|
||||
**DNS** — five A records to the *host's* public address (the cluster is behind it):
|
||||
|
||||
```
|
||||
register.<domain> mijn.<domain> behandel.<domain> beheer.<domain> auth.<domain> → 46.224.220.37
|
||||
```
|
||||
|
||||
**The host's forward** — the public IP is on the Fedora host, so it has to hand 80/443 to
|
||||
the node. This is the only host-side configuration, and it is dumb layer 4:
|
||||
**Prerequisite: the VM must have been installed with the §1 patch.** A stock Talos config
|
||||
gives you a node that still carries the control-plane taint and knows nothing about the
|
||||
plain-HTTP registry, and the deploy hits those in that order: the `registry` pod sits
|
||||
`Pending` until `rollout status` times out, and once that is fixed every repo image fails to
|
||||
pull. Two separate fixes:
|
||||
|
||||
```bash
|
||||
sudo firewall-cmd --permanent --zone=public --add-forward-port=port=80:proto=tcp:toaddr=<TALOS_VM_IP>:toport=32080
|
||||
sudo firewall-cmd --permanent --zone=public --add-forward-port=port=443:proto=tcp:toaddr=<TALOS_VM_IP>:toport=32443
|
||||
sudo firewall-cmd --permanent --zone=public --add-masquerade
|
||||
sudo firewall-cmd --reload
|
||||
# on the Fedora host — 1. let workloads onto the only node (§1)
|
||||
export KUBECONFIG=~/talos-kubeconfig-local
|
||||
kubectl taint node --all node-role.kubernetes.io/control-plane-
|
||||
|
||||
# 2. trust the in-cluster registry over plain HTTP (§2)
|
||||
cat > /tmp/registry-patch.yaml <<'YAML'
|
||||
machine:
|
||||
registries:
|
||||
mirrors:
|
||||
"<TALOS_VM_IP>:30500":
|
||||
endpoints:
|
||||
- http://<TALOS_VM_IP>:30500
|
||||
YAML
|
||||
talosctl -n <TALOS_VM_IP> -e <TALOS_VM_IP> patch mc --patch @/tmp/registry-patch.yaml
|
||||
```
|
||||
|
||||
`--add-masquerade` is what makes the return path work: without it the node answers the
|
||||
client's address directly and the reply never goes back through the host.
|
||||
Keep those two apart. On Talos 1.14 a patch that also sets
|
||||
`cluster.allowSchedulingOnControlPlanes` is rejected with *".cluster.allowSchedulingOnControlPlanes
|
||||
is already set in v1alpha1 config"* — the field moved out of the v1alpha1 schema, the same way
|
||||
`machine.install` did (§1) — and the rejection takes the whole patch with it, so the mirror
|
||||
silently doesn't land either. `kubectl taint` is the documented way (§1); it is undone if the
|
||||
node ever re-registers, which is a reboot, not a deploy.
|
||||
|
||||
**A StorageClass.** Caddy's certificates live in `/data`, which is an `emptyDir` unless
|
||||
`persistence.storageClass` is set (§6). Let's Encrypt allows five duplicate certificates per
|
||||
week, so on an `emptyDir` a handful of pod restarts leaves the edge serving an untrusted
|
||||
certificate until the limit resets. Install local-path first (§6).
|
||||
The cluster's API and registry are not exposed publicly, so the job forwards them over the
|
||||
same SSH hop the Gitea-runner pipeline uses:
|
||||
|
||||
### Deploy
|
||||
```
|
||||
ssh -p 6667 user@labs.respellion.tech -L 6443 -L 30500 -L 30141 → <TALOS_VM_IP>
|
||||
```
|
||||
|
||||
Consequences worth knowing:
|
||||
|
||||
- Images are **pushed** to `localhost:30500` (the tunnel) and **pulled** by the node from
|
||||
`<TALOS_VM_IP>:30500` (its own NodePort, the address in the Talos registry-mirror patch).
|
||||
Same registry, two names — hence the two `K8S_REGISTRY` values in the workflow.
|
||||
- It calls `make k8s-reseed`, not `make k8s-up`: the bootstrap Jobs are idempotent, and
|
||||
deleting them first is what stops a changed Job template from wedging `helm upgrade` (§7).
|
||||
- `dev` is a mutable tag, so a `rollout restart` of the nine repo deployments is what
|
||||
actually puts the new images in the pods.
|
||||
- Deploys **queue** (`cancel-in-progress: false`): a helm upgrade killed half-way leaves the
|
||||
release in `pending-upgrade`, which has to be unwedged by hand.
|
||||
|
||||
Settings, all on the repository in Gitea:
|
||||
|
||||
| Kind | Name | What |
|
||||
|---|---|---|
|
||||
| Secret | `TALOS_SSH_KEY` | private key for `user@labs.respellion.tech` (the Fedora host) |
|
||||
| Secret | `TALOS_KUBECONFIG` | base64 of the kubeconfig, **`server: https://127.0.0.1:6443`** — Talos puts `127.0.0.1` in the apiserver cert SANs, so TLS still verifies through the tunnel |
|
||||
| Variable | `TALOS_VM_IP` | the VM's libvirt address (default `192.168.122.173`) |
|
||||
| Variable | `TALOS_HOST` | the browser-facing host baked into Keycloak's issuer (default `localhost`, see §5) |
|
||||
|
||||
The last step smokes `GET /openbaar/register` through the openbaar portal, which exercises
|
||||
portal → Caddy → BFF → projection. An empty register passes; a 502 does not.
|
||||
|
||||
### Reaching the portals from a laptop
|
||||
|
||||
The deployed portals are pinned to `http://localhost:30180` for Keycloak (§5), so a browser
|
||||
needs **all five** browser-facing ports on its own localhost — the portal alone is not
|
||||
enough, and a missing Keycloak shows up as `ERR_CONNECTION_REFUSED` on
|
||||
`/realms/*/.well-known/openid-configuration` followed by an opaque `ERROR Error: [object Object]`.
|
||||
`make k8s-portals` does this when kubectl can reach the cluster; through the lab server one
|
||||
SSH does it without a kubeconfig at all:
|
||||
|
||||
```bash
|
||||
make k8s-up TALOS_HOST=<domain-facing name> K8S_REGISTRY=<TALOS_VM_IP>:30500 \
|
||||
K8S_SET='--set public.domain=<domain> --set public.email=<ops address> --set persistence.storageClass=local-path'
|
||||
ssh -N -p 6667 \
|
||||
-L 30140:<TALOS_VM_IP>:30140 -L 30141:<TALOS_VM_IP>:30141 \
|
||||
-L 30142:<TALOS_VM_IP>:30142 -L 30143:<TALOS_VM_IP>:30143 \
|
||||
-L 30180:<TALOS_VM_IP>:30180 \
|
||||
user@labs.respellion.tech
|
||||
```
|
||||
|
||||
`public.domain` is the only switch: with it empty nothing in `templates/edge.yaml` renders
|
||||
and the stack behaves exactly as §4 describes. With it set, `KC_HOSTNAME` and the portals'
|
||||
`config.json` both become `https://auth.<domain>` — one helper builds both, so the issuer
|
||||
and the authority cannot drift (ADR-0010).
|
||||
Then the §5 table's URLs work as written. The admin UIs (OpenZaak, Flowable, …) need no
|
||||
forward — they are server-rendered, so the VM's address is fine.
|
||||
|
||||
Watch the first certificate being issued:
|
||||
|
||||
```bash
|
||||
kubectl -n big logs deploy/caddy-edge -f # "certificate obtained successfully"
|
||||
curl -sSI https://register.<domain>/openbaar/register | head -1
|
||||
```
|
||||
|
||||
### When it doesn't work
|
||||
|
||||
| Symptom | Cause |
|
||||
|---|---|
|
||||
| ACME fails with `connection refused` or a timeout on the HTTP-01 challenge | the host's 80 → 32080 forward is missing, or `--add-masquerade` is |
|
||||
| ACME fails with `NXDOMAIN` / `no such host` | the A record isn't there yet. Caddy retries with backoff; fix DNS and it recovers |
|
||||
| An untrusted certificate after several restarts | the Let's Encrypt duplicate limit, from certificates on an `emptyDir` — see above |
|
||||
| The portal loads but login bounces back logged out | `public.domain` changed without the portals rolling. The chart hashes the issuer into their pod template, so `helm upgrade` should do it — check `kubectl -n big describe deploy/self-service` |
|
||||
| `404` from the edge on a name that should work | the name isn't in `public.routes`; Caddy answers 404 for a Host it has no site block for |
|
||||
Not covered: the portals still need `make k8s-portals` (or an SSH forward) to be usable in a
|
||||
browser, because PKCE needs a secure context (§5). Giving the server a hostname + TLS is the
|
||||
upgrade path.
|
||||
|
||||
## What is not ported
|
||||
|
||||
|
||||
@@ -135,20 +135,6 @@ cluster-internal hosts ({{ .Release.Namespace }}) and the node address
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
The origin a browser reaches Keycloak on, and so the issuer its tokens carry and
|
||||
the authority the portals are configured with (ADR-0010). With a public edge that
|
||||
is the `auth` hostname on `public.domain` — which must stay in step with the `auth`
|
||||
key in `public.routes`; without one it is the node address plus Keycloak's NodePort.
|
||||
*/}}
|
||||
{{- define "big.keycloakUrl" -}}
|
||||
{{- if .Values.public.domain -}}
|
||||
https://auth.{{ .Values.public.domain }}
|
||||
{{- else -}}
|
||||
http://{{ .Values.host }}:{{ index .Values.nodePorts "keycloak" }}
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{- define "big.labels" -}}
|
||||
app.kubernetes.io/name: {{ .name }}
|
||||
app.kubernetes.io/instance: {{ .root.Release.Name }}
|
||||
|
||||
@@ -40,5 +40,5 @@ metadata:
|
||||
{{- include "big.labels" (dict "root" $ "name" (printf "portal-config-%s" $realm)) | nindent 4 }}
|
||||
data:
|
||||
config.json: |
|
||||
{ "authority": "{{ include "big.keycloakUrl" $ }}/realms/{{ $realm }}" }
|
||||
{ "authority": "{{ printf "http://%s:%v" $.Values.host (index $.Values.nodePorts "keycloak") }}/realms/{{ $realm }}" }
|
||||
{{- end }}
|
||||
|
||||
@@ -28,7 +28,7 @@ spec:
|
||||
{{- range $w.files }}
|
||||
{{- if hasPrefix "portal-config-" .configMap }}
|
||||
annotations:
|
||||
checksum/portal-config: {{ include "big.keycloakUrl" $ | sha256sum }}
|
||||
checksum/portal-config: {{ printf "%s|%v" $.Values.host (index $.Values.nodePorts "keycloak") | sha256sum }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
labels:
|
||||
|
||||
@@ -1,136 +0,0 @@
|
||||
{{- /*
|
||||
The public TLS edge (ADR-0035). Rendered only when `public.domain` is set; with it
|
||||
empty the stack is reached on the NodePorts below and nothing here exists.
|
||||
|
||||
Caddy rather than an ingress controller: the four portals already run caddy:2-alpine,
|
||||
so this adds no dependency, and it does ACME itself — no cert-manager, no CRDs, no
|
||||
Ingress objects for five hostnames that never change. It proxies to the ClusterIP
|
||||
services, so the browser-facing NodePorts are not involved in a public deployment.
|
||||
|
||||
The public IP lives on the Fedora host, which forwards 80/443 to the two NodePorts
|
||||
below. That forward is dumb L4 — no TLS, no routing — see the runbook.
|
||||
*/}}
|
||||
{{- if .Values.public.domain }}
|
||||
{{- $pub := .Values.public }}
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: caddy-edge-config
|
||||
labels:
|
||||
{{- include "big.labels" (dict "root" $ "name" "caddy-edge") | nindent 4 }}
|
||||
data:
|
||||
Caddyfile: |
|
||||
{
|
||||
{{- with $pub.email }}
|
||||
email {{ . }}
|
||||
{{- end }}
|
||||
}
|
||||
{{- range $sub, $target := $pub.routes }}
|
||||
|
||||
{{ $sub }}.{{ $pub.domain }} {
|
||||
reverse_proxy {{ $target }}
|
||||
}
|
||||
{{- end }}
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: caddy-edge
|
||||
labels:
|
||||
{{- include "big.labels" (dict "root" $ "name" "caddy-edge") | nindent 4 }}
|
||||
spec:
|
||||
replicas: 1
|
||||
strategy:
|
||||
type: Recreate
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: caddy-edge
|
||||
app.kubernetes.io/instance: {{ .Release.Name }}
|
||||
template:
|
||||
metadata:
|
||||
annotations:
|
||||
# A ConfigMap mounted with subPath never updates in place, so a changed
|
||||
# Caddyfile has to roll the pod.
|
||||
checksum/caddyfile: {{ printf "%s|%v|%v" $pub.domain $pub.email $pub.routes | sha256sum }}
|
||||
labels:
|
||||
{{- include "big.labels" (dict "root" $ "name" "caddy-edge") | nindent 8 }}
|
||||
spec:
|
||||
containers:
|
||||
- name: caddy-edge
|
||||
image: {{ $pub.image }}
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 80
|
||||
- name: https
|
||||
containerPort: 443
|
||||
# TCP, not HTTP: a GET with no matching Host gets a 404 from Caddy, which
|
||||
# would fail an httpGet probe for a perfectly healthy edge.
|
||||
readinessProbe:
|
||||
tcpSocket: { port: 443 }
|
||||
volumeMounts:
|
||||
- name: config
|
||||
mountPath: /etc/caddy/Caddyfile
|
||||
subPath: Caddyfile
|
||||
readOnly: true
|
||||
- name: data
|
||||
mountPath: /data
|
||||
- name: run
|
||||
mountPath: /config
|
||||
volumes:
|
||||
- name: config
|
||||
configMap:
|
||||
name: caddy-edge-config
|
||||
- name: run
|
||||
emptyDir: {}
|
||||
- name: data
|
||||
{{- if .Values.persistence.storageClass }}
|
||||
persistentVolumeClaim:
|
||||
claimName: caddy-edge-data
|
||||
{{- else }}
|
||||
# Certificates live here. On an emptyDir every pod restart asks Let's
|
||||
# Encrypt again, and its duplicate-certificate limit is five per week —
|
||||
# set persistence.storageClass for anything that stays up.
|
||||
emptyDir: {}
|
||||
{{- end }}
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: caddy-edge
|
||||
labels:
|
||||
{{- include "big.labels" (dict "root" $ "name" "caddy-edge") | nindent 4 }}
|
||||
spec:
|
||||
type: NodePort
|
||||
selector:
|
||||
app.kubernetes.io/name: caddy-edge
|
||||
app.kubernetes.io/instance: {{ .Release.Name }}
|
||||
ports:
|
||||
- name: http
|
||||
port: 80
|
||||
targetPort: 80
|
||||
nodePort: {{ $pub.nodePorts.http }}
|
||||
- name: https
|
||||
port: 443
|
||||
targetPort: 443
|
||||
nodePort: {{ $pub.nodePorts.https }}
|
||||
{{- if .Values.persistence.storageClass }}
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: caddy-edge-data
|
||||
labels:
|
||||
{{- include "big.labels" (dict "root" $ "name" "caddy-edge") | nindent 4 }}
|
||||
# Keep the certificates when the release is uninstalled — re-issuing them on
|
||||
# every reinstall is what burns the rate limit.
|
||||
annotations:
|
||||
helm.sh/resource-policy: keep
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
storageClassName: {{ .Values.persistence.storageClass }}
|
||||
resources:
|
||||
requests:
|
||||
storage: 128Mi
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
@@ -49,32 +49,6 @@ persistence:
|
||||
# the data across pod restarts.
|
||||
storageClass: ""
|
||||
|
||||
# The public TLS edge (ADR-0035). Empty `domain` = no edge at all: nothing in
|
||||
# templates/edge.yaml is rendered and the stack is reached on the NodePorts below,
|
||||
# with `host` above pinning the OIDC origin.
|
||||
#
|
||||
# Set it and an in-cluster Caddy terminates TLS for `<sub>.<domain>`, gets its own
|
||||
# certificates from Let's Encrypt and proxies to the ClusterIP services. The node
|
||||
# only has to be reachable on the two NodePorts here — the Fedora host forwards
|
||||
# 80/443 to them (see docs/runbooks/kubernetes-talos.md).
|
||||
public:
|
||||
domain: ""
|
||||
# ACME registration address; Let's Encrypt uses it for expiry warnings.
|
||||
email: ""
|
||||
image: docker.io/library/caddy:2-alpine
|
||||
# <subdomain>: <in-cluster service:port>. `auth` is not free-form — big.keycloakUrl
|
||||
# builds the pinned issuer from it.
|
||||
routes:
|
||||
register: openbaar:80
|
||||
mijn: self-service:80
|
||||
behandel: behandel:80
|
||||
beheer: beheer:80
|
||||
auth: keycloak:8080
|
||||
# Where the host's 80/443 forward lands. Not 30080/30443: 30080 is the BFF.
|
||||
nodePorts:
|
||||
http: 32080
|
||||
https: 32443
|
||||
|
||||
# The only place a port is published outside the cluster. A workload listed here
|
||||
# gets a NodePort on its single port; everything else stays ClusterIP.
|
||||
nodePorts:
|
||||
@@ -294,7 +268,7 @@ workloads:
|
||||
# Pin the issuer to the address the browser uses, and let backchannel calls
|
||||
# keep using keycloak:8080 — the BFF discovers metadata in-cluster and gets
|
||||
# this issuer back, which is what browser tokens carry (infra/host-browser.yml).
|
||||
KC_HOSTNAME: '{{ include "big.keycloakUrl" . }}'
|
||||
KC_HOSTNAME: "http://{{ .Values.host }}:{{ index .Values.nodePorts \"keycloak\" }}"
|
||||
KC_HOSTNAME_BACKCHANNEL_DYNAMIC: "true"
|
||||
ports: [{ name: http, port: 8080 }]
|
||||
# TCP, not /health/ready on the management port: nothing here gates on realm
|
||||
|
||||
@@ -1,80 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fail when the pinned issuer and the portals' OIDC authority stop agreeing.
|
||||
|
||||
Keycloak pins one issuer (`KC_HOSTNAME`) and each portal is configured with one
|
||||
authority (`config.json`). A browser token carries the first; the BFF validates
|
||||
against what it discovers from the second (ADR-0010). When the two drift the
|
||||
symptom is three services away — a login that bounces back logged out, or a 401
|
||||
from the BFF — so the chart builds both from one helper and this asserts it.
|
||||
|
||||
It also pins the two halves of the public edge (ADR-0035): that setting
|
||||
`public.domain` actually publishes the hostnames, and that leaving it empty
|
||||
renders no edge at all, which is what compose, CI and a laptop cluster rely on.
|
||||
|
||||
Run it with `make k8s-lint`. No cluster needed.
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
CHART = Path(__file__).resolve().parent / "big-reference"
|
||||
DOMAIN = "example.test"
|
||||
|
||||
|
||||
def render(*sets: str) -> str:
|
||||
argv = ["helm", "template", "big", str(CHART), "-n", "big"]
|
||||
for s in sets:
|
||||
argv += ["--set", s]
|
||||
proc = subprocess.run(argv, capture_output=True, text=True)
|
||||
if proc.returncode != 0:
|
||||
sys.exit(f"helm template failed:\n{proc.stderr}")
|
||||
return proc.stdout
|
||||
|
||||
|
||||
def issuer(out: str) -> str:
|
||||
"""The value of KC_HOSTNAME in the rendered manifests."""
|
||||
m = re.search(r"name: KC_HOSTNAME\n\s+value: \"(\S+)\"", out)
|
||||
return m[1] if m else ""
|
||||
|
||||
|
||||
def authorities(out: str) -> set[str]:
|
||||
"""Every portal's OIDC authority, with the realm path stripped."""
|
||||
found = set()
|
||||
for line in re.findall(r'\{ "authority": .* \}', out):
|
||||
url = json.loads(line)["authority"]
|
||||
found.add(url.rsplit("/realms/", 1)[0])
|
||||
return found
|
||||
|
||||
|
||||
def main() -> int:
|
||||
problems = []
|
||||
|
||||
public = render(f"public.domain={DOMAIN}")
|
||||
if issuer(public) != f"https://auth.{DOMAIN}":
|
||||
problems.append(f" with public.domain set, KC_HOSTNAME is {issuer(public)!r}, not https://auth.{DOMAIN}")
|
||||
if authorities(public) != {f"https://auth.{DOMAIN}"}:
|
||||
problems.append(f" with public.domain set, the portals point at {sorted(authorities(public))}")
|
||||
for host in (f"register.{DOMAIN}", f"mijn.{DOMAIN}", f"behandel.{DOMAIN}", f"beheer.{DOMAIN}", f"auth.{DOMAIN}"):
|
||||
if host not in public:
|
||||
problems.append(f" {host} is not published by the edge")
|
||||
|
||||
private = render()
|
||||
if authorities(private) != {issuer(private)}:
|
||||
problems.append(f" by default the portals point at {sorted(authorities(private))}, the issuer is {issuer(private)!r}")
|
||||
if "caddy-edge" in private:
|
||||
problems.append(" the edge renders with no public.domain — compose, CI and a laptop cluster expect nothing")
|
||||
|
||||
if problems:
|
||||
print("the chart's OIDC origin is inconsistent:\n" + "\n".join(problems))
|
||||
print("\nBoth halves come from the `big.keycloakUrl` helper — change it, not one caller.")
|
||||
return 1
|
||||
|
||||
print(f"issuer + portal authority agree, with and without a public domain")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -56,7 +56,6 @@ nav:
|
||||
- "ADR-0032: Werkbak live refresh": architecture/adr-0032-werkbak-live-refresh.md
|
||||
- "ADR-0033: Kubernetes via one Helm chart": architecture/adr-0033-kubernetes-via-one-helm-chart.md
|
||||
- "ADR-0034: Caddy serves the portals": architecture/adr-0034-caddy-serves-the-portals.md
|
||||
- "ADR-0035: Public TLS edge in the cluster": architecture/adr-0035-public-tls-edge-in-cluster.md
|
||||
- FDS-architectuur:
|
||||
- Overzicht: architecture/fds/README.md
|
||||
- Componentview (L3): architecture/fds/c4-component-view.md
|
||||
|
||||
Reference in New Issue
Block a user