Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4568b682a6 | ||
|
|
5494363221 | ||
|
|
0074a1bff3 | ||
|
|
594fdde227 | ||
|
|
804031eeb8 | ||
|
|
6cfcc4cf83 | ||
|
|
9d7e8e5b65 | ||
|
|
17f1f2f809 |
@@ -41,6 +41,27 @@ jobs:
|
||||
nuget-${{ runner.os }}-
|
||||
- run: make lint
|
||||
|
||||
# The Helm chart's only automated gate: it renders and schema-checks the whole
|
||||
# stack, and checks it still describes the same stack as the compose file
|
||||
# (ADR-0033). No cluster involved — see docs/runbooks/kubernetes-talos.md.
|
||||
k8s:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: https://github.com/actions/checkout@v4
|
||||
# helm as its pinned static binary rather than a marketplace action: one URL,
|
||||
# the same one the Talos runbook §0 gives a developer, and no third-party
|
||||
# action to vet (CLAUDE.md §13). The drift check also needs `docker compose`,
|
||||
# which the runner already has (see docs/runbooks/ci.md).
|
||||
- name: Install helm
|
||||
run: |
|
||||
mkdir -p "$HOME/.local/bin"
|
||||
curl -sSL https://get.helm.sh/helm-v3.16.4-linux-amd64.tar.gz \
|
||||
| tar xz -O linux-amd64/helm > "$HOME/.local/bin/helm"
|
||||
chmod +x "$HOME/.local/bin/helm"
|
||||
echo "$HOME/.local/bin" >> "$GITHUB_PATH"
|
||||
- run: make k8s-lint
|
||||
- run: make k8s-drift
|
||||
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
@@ -186,8 +207,14 @@ jobs:
|
||||
# dispatched (gitea-actions-gotchas.md §7). Default `if: success()` dispatches normally. Cost: a
|
||||
# failing mutation ratchet now skips verify-stack instead of running it anyway; the fix-and-re-push
|
||||
# re-run exercises verify-stack, so we still get the signal.
|
||||
#
|
||||
# Main only, not on PRs: the runner shares the lab node with the deployed stack, and a second
|
||||
# full stack per PR was what got the runner OOM-killed (#182). PRs still gate on every job above;
|
||||
# the live-stack check runs once per merge. A plain event `if` keeps the implicit success(), so it
|
||||
# is not the status-function case from gotchas §7.
|
||||
verify-stack:
|
||||
needs: [mutation]
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: https://github.com/actions/checkout@v4
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
name: Deploy to Talos
|
||||
|
||||
# A merge to main ships the stack to the Talos cluster on the lab server
|
||||
# (docs/runbooks/kubernetes-talos.md §9). PR CI is the merge gate, so main is
|
||||
# green by construction — this workflow only deploys.
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Queue deploys, never cancel one: a helm upgrade killed half-way leaves the
|
||||
# release in `pending-upgrade` and the next run has to be unwedged by hand.
|
||||
concurrency:
|
||||
group: deploy-talos
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
# The Talos VM as seen from the Fedora host (libvirt guest IP), and the
|
||||
# address a browser uses to reach the cluster. `localhost` is deliberate:
|
||||
# the portals' PKCE needs a secure context, so they are reached over
|
||||
# `kubectl port-forward` — runbook §5. Override with repo variables.
|
||||
TALOS_VM_IP: ${{ vars.TALOS_VM_IP }}
|
||||
TALOS_HOST: ${{ vars.TALOS_HOST }}
|
||||
# Set it when the labs Caddy publishes the portals: Keycloak's public https
|
||||
# origin, e.g. https://big-auth.labs.respellion.tech (runbook, "Publishing
|
||||
# through the labs Caddy").
|
||||
KEYCLOAK_URL: ${{ vars.KEYCLOAK_URL }}
|
||||
# `true` fills in the medewerker OTP step for the public demo (chart value
|
||||
# demo.otpAutofill). The fixture secret is committed: demo only.
|
||||
OTP_AUTOFILL: ${{ vars.OTP_AUTOFILL }}
|
||||
steps:
|
||||
- uses: https://github.com/actions/checkout@v4
|
||||
|
||||
# Pinned static binaries, the same URLs the Talos runbook §0 gives a
|
||||
# developer and the same helm the `k8s` CI job uses — no action to vet.
|
||||
- name: Install kubectl, helm and crane
|
||||
run: |
|
||||
set -euo pipefail
|
||||
bin="$HOME/.local/bin"; mkdir -p "$bin"
|
||||
curl -sSLo "$bin/kubectl" https://dl.k8s.io/release/v1.37.0/bin/linux/amd64/kubectl
|
||||
curl -sSL https://get.helm.sh/helm-v3.16.4-linux-amd64.tar.gz | tar xz -O linux-amd64/helm > "$bin/helm"
|
||||
curl -sSL https://github.com/google/go-containerregistry/releases/download/v0.20.2/go-containerregistry_Linux_x86_64.tar.gz | tar xz -O crane > "$bin/crane"
|
||||
chmod +x "$bin"/{kubectl,helm,crane}
|
||||
echo "$bin" >> "$GITHUB_PATH"
|
||||
|
||||
# The cluster's API and its registry are only reachable through the Fedora
|
||||
# host, so forward both to the runner. 30141 is the openbaar portal, for
|
||||
# the smoke at the end.
|
||||
- name: Tunnel the Talos API + registry through the Fedora host
|
||||
env:
|
||||
SSH_KEY: ${{ secrets.TALOS_SSH_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
: "${TALOS_VM_IP:=192.168.122.173}"
|
||||
umask 077
|
||||
printf '%s\n' "$SSH_KEY" > ~/.ssh_talos
|
||||
ssh -i ~/.ssh_talos -o StrictHostKeyChecking=no -o IdentitiesOnly=yes \
|
||||
-o ExitOnForwardFailure=yes -p 6667 -f -N \
|
||||
-L 6443:$TALOS_VM_IP:6443 \
|
||||
-L 30500:$TALOS_VM_IP:30500 \
|
||||
-L 30141:$TALOS_VM_IP:30141 \
|
||||
user@labs.respellion.tech
|
||||
|
||||
# The kubeconfig's server must be https://127.0.0.1:6443 — Talos puts
|
||||
# 127.0.0.1 in the apiserver cert SANs, so TLS verification still holds
|
||||
# through the tunnel.
|
||||
- name: Write the kubeconfig
|
||||
env:
|
||||
KUBECONFIG_B64: ${{ secrets.TALOS_KUBECONFIG }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
umask 077
|
||||
base64 -d <<< "$KUBECONFIG_B64" > "$RUNNER_TEMP/kubeconfig"
|
||||
echo "KUBECONFIG=$RUNNER_TEMP/kubeconfig" >> "$GITHUB_ENV"
|
||||
kubectl --kubeconfig "$RUNNER_TEMP/kubeconfig" get nodes
|
||||
|
||||
# Idempotent; also makes a first deploy onto a bare cluster work. The
|
||||
# registry's storage is an emptyDir, so a replaced pod loses the images —
|
||||
# which the push in the next step puts back anyway.
|
||||
- name: Ensure the in-cluster registry
|
||||
run: make k8s-registry
|
||||
|
||||
# Push through the tunnel (localhost), pull from the node's own NodePort
|
||||
# (the address in the Talos registry-mirror patch) — same registry, two
|
||||
# names, so the two `make` calls get different K8S_REGISTRY values.
|
||||
- name: Build and push the images
|
||||
run: make k8s-images K8S_REGISTRY=localhost:30500
|
||||
|
||||
# k8s-reseed = seed configmaps + helm upgrade + re-run the bootstrap jobs.
|
||||
# The jobs are idempotent, and deleting them first is what keeps a changed
|
||||
# Job template from wedging the upgrade (`cannot patch … with kind Job`).
|
||||
- name: Deploy the chart
|
||||
run: |
|
||||
make k8s-reseed \
|
||||
TALOS_HOST=${TALOS_HOST:-localhost} \
|
||||
K8S_REGISTRY=${TALOS_VM_IP:-192.168.122.173}:30500 \
|
||||
K8S_SET="${KEYCLOAK_URL:+--set keycloakUrl=$KEYCLOAK_URL} --set demo.otpAutofill=${OTP_AUTOFILL:-false}"
|
||||
|
||||
# `dev` is a mutable tag and helm sees an unchanged pod template, so the
|
||||
# new images only land on a restart (pullPolicy is already Always).
|
||||
- name: Roll the services onto the new images
|
||||
run: |
|
||||
set -euo pipefail
|
||||
svcs="acl domain bff event-subscriber projection-api self-service openbaar behandel beheer"
|
||||
kubectl -n big rollout restart deploy $svcs
|
||||
kubectl -n big rollout status --timeout=300s deploy $svcs
|
||||
|
||||
# Proves portal → Caddy → BFF → projection end to end. An empty register is
|
||||
# a pass; a 502 or a timeout is not.
|
||||
- name: Smoke the public register
|
||||
run: curl -fsS --retry 10 --retry-delay 6 --retry-all-errors http://localhost:30141/openbaar/register
|
||||
|
||||
- name: Pods on failure
|
||||
if: failure()
|
||||
run: kubectl -n big get pods,jobs || true
|
||||
@@ -43,7 +43,7 @@ export DOCKER_HOST := unix://$(PODMAN_SOCK)
|
||||
endif
|
||||
endif
|
||||
|
||||
.PHONY: ci lint build unit mutation frontend integration verify verify-up verify-acl verify-nrc verify-projection verify-bff verify-domain verify-observability verify-tracing verify-metrics verify-objecttypen verify-objecten verify-registerrecord verify-objecten-notifications verify-notifications smoke up down local verify-local local-down changelog openzaak-up openzaak-smoke openzaak-seed openzaak-down stack-up stack-smoke stack-down keycloak-up keycloak-smoke keycloak-down flowable-up flowable-smoke flowable-down k8s-lint k8s-registry k8s-images k8s-seed k8s-up k8s-reseed k8s-portals k8s-down k8s-purge help
|
||||
.PHONY: ci lint build unit mutation frontend integration verify verify-up verify-acl verify-nrc verify-projection verify-bff verify-domain verify-observability verify-tracing verify-metrics verify-objecttypen verify-objecten verify-registerrecord verify-objecten-notifications verify-notifications smoke up down local verify-local local-down changelog openzaak-up openzaak-smoke openzaak-seed openzaak-down stack-up stack-smoke stack-down keycloak-up keycloak-smoke keycloak-down flowable-up flowable-smoke flowable-down k8s-lint k8s-drift k8s-registry k8s-images k8s-seed k8s-up k8s-reseed k8s-portals k8s-down k8s-purge help
|
||||
|
||||
## ci: run the full pipeline — lint, build, unit, mutation, frontend, verify (mirrors Gitea Actions)
|
||||
## `verify` is the live-stack stage (full stack up once → ACL + notification checks).
|
||||
@@ -353,6 +353,13 @@ k8s-lint:
|
||||
helm lint $(K8S_CHART)
|
||||
helm template big $(K8S_CHART) -n $(K8S_NS) --set images.registry=registry.invalid:5000 >/dev/null
|
||||
|
||||
## k8s-drift: fail if compose and the Helm chart describe different stacks
|
||||
# Compose is CI-canonical (ADR-0033) and the chart is a transcription of it; this
|
||||
# compares what each one deploys — workload names and resolved images. Needs
|
||||
# `docker compose` and `helm`, no cluster.
|
||||
k8s-drift:
|
||||
python3 infra/helm/check-drift.py
|
||||
|
||||
## k8s-registry: deploy the in-cluster image registry (NodePort 30500)
|
||||
k8s-registry:
|
||||
kubectl apply -f infra/helm/registry.yaml
|
||||
|
||||
@@ -126,8 +126,9 @@ Consequences of that shape, each chosen deliberately:
|
||||
|
||||
**Negative / costs**
|
||||
|
||||
- A second deployment description to keep in step with compose. Nothing enforces that
|
||||
today; a drift check belongs in CI (follow-up).
|
||||
- A second deployment description to keep in step with compose. `make k8s-drift` (#168)
|
||||
now enforces the part that bites — the workload set and the resolved images, with the
|
||||
four deviations below declared — but not per-workload env, ports or volumes.
|
||||
- `helm install` alone is not enough — the ConfigMaps must be seeded first, and a missing
|
||||
one surfaces as `ContainerCreating`, not as a clear error.
|
||||
- Generic templates mean a values typo can render valid-but-wrong YAML; `k8s-lint` catches
|
||||
|
||||
+11
-3
@@ -2,8 +2,10 @@
|
||||
|
||||
> **Status: active.** The workflow `.gitea/workflows/ci.yaml` runs on Gitea's
|
||||
> hosted `ubuntu-latest` runner — no self-hosted runner required.
|
||||
> **`make ci` is still the local gate** — it runs the exact same checks
|
||||
> (the workflow calls the same `make` targets).
|
||||
> **`make ci` is still the local gate** — it runs the same checks via the same
|
||||
> `make` targets, with one exception: the `k8s` job's targets are not in `make ci`,
|
||||
> because `helm` is optional for everyone not deploying to Kubernetes. Run
|
||||
> `make k8s-lint k8s-drift` by hand after touching the chart or the compose file.
|
||||
|
||||
## The pipeline
|
||||
|
||||
@@ -16,8 +18,10 @@ and CI cannot drift:
|
||||
| `lint` | `make lint` → `dotnet format … --verify-no-changes` | .NET 10 SDK |
|
||||
| `build` | `make build` → `dotnet build … -c Release` | .NET 10 SDK |
|
||||
| `unit` | `make unit` → `dotnet test … -c Release --filter "Category!=Integration"` | .NET 10 SDK |
|
||||
| `frontend` | `make frontend` → Nx lint/test/build for the four portals | pnpm + Node |
|
||||
| `k8s` | `make k8s-lint` (render + schema-check the Helm chart) → `make k8s-drift` (chart still describes the same stack as `infra/docker-compose.yml`) | pinned `helm` binary + `docker compose` |
|
||||
| `mutation` | `make mutation` → `dotnet tool restore` → `dotnet stryker` (ACL); uploads the HTML report as an artifact | .NET 10 SDK |
|
||||
| `verify-stack` | the single live-stack stage — steps: `make verify-up` (full stack up + health, the DoD smoke) → `make verify-acl` (ACL ↔ OpenZaak) → `make verify-nrc` (OpenZaak → NRC delivery) → `make down` | container engine + egress (base images, nuget, `selectielijst.openzaak.nl`) |
|
||||
| `verify-stack` | **push to `main` only, skipped on PRs** (#182) — the single live-stack stage — steps: `make verify-up` (full stack up + health, the DoD smoke) → `make verify-acl` (ACL ↔ OpenZaak) → `make verify-nrc` (OpenZaak → NRC delivery) → `make down` | container engine + egress (base images, nuget, `selectielijst.openzaak.nl`) |
|
||||
|
||||
> **Why one `verify-stack` job, not three.** The single self-hosted runner runs jobs
|
||||
> **sequentially**, so booting OpenZaak once (instead of once per check) is the
|
||||
@@ -27,6 +31,10 @@ and CI cannot drift:
|
||||
> services by **container IP** (the runner can't reach published ports — see
|
||||
> [gitea-actions-gotchas.md §5/§6](gitea-actions-gotchas.md)).
|
||||
|
||||
A second workflow, `.gitea/workflows/deploy.yaml`, deploys the stack to the Talos
|
||||
cluster on the lab server when a PR is merged to `main` — see
|
||||
[kubernetes-talos.md §9](kubernetes-talos.md) for its secrets and the SSH tunnel it needs.
|
||||
|
||||
All `uses:` references are absolute, tag-pinned URLs (`https://github.com/actions/checkout@v4`,
|
||||
`https://github.com/actions/setup-dotnet@v4`) per CLAUDE.md §8.7 and §15 — Gitea
|
||||
Actions resolves them from GitHub.
|
||||
|
||||
@@ -322,6 +322,7 @@ The PVCs carry `helm.sh/resource-policy: keep`, so `make k8s-down` leaves the da
|
||||
|
||||
```bash
|
||||
make k8s-lint # render + schema-check the chart, no cluster needed
|
||||
make k8s-drift # fail if compose and the chart describe different stacks
|
||||
make k8s-portals # forward the portals + Keycloak to localhost (browser access)
|
||||
make k8s-images K8S_REGISTRY=... # after changing a service or a portal
|
||||
make k8s-up TALOS_HOST=... K8S_REGISTRY=...
|
||||
@@ -359,6 +360,95 @@ immutable, so `helm upgrade` is rejected with `cannot patch "…" with kind Job`
|
||||
| Pods `Evicted` / `OOMKilled` | the VM is too small (§0) |
|
||||
| A Job shows `BackoffLimitExceeded` | read it: `kubectl -n big logs job/<name>` |
|
||||
|
||||
## 9. Deploying on merge to main
|
||||
|
||||
`.gitea/workflows/deploy.yaml` runs the §3–§4 steps against the **lab server's** Talos VM
|
||||
every time a PR is squash-merged to `main` (and on demand via *Run workflow*). PR CI is the
|
||||
merge gate, so the workflow deploys without re-running the checks.
|
||||
|
||||
The cluster's API and registry are not exposed publicly, so the job forwards them over the
|
||||
same SSH hop the Gitea-runner pipeline uses:
|
||||
|
||||
```
|
||||
ssh -p 6667 user@labs.respellion.tech -L 6443 -L 30500 -L 30141 → <TALOS_VM_IP>
|
||||
```
|
||||
|
||||
Consequences worth knowing:
|
||||
|
||||
- Images are **pushed** to `localhost:30500` (the tunnel) and **pulled** by the node from
|
||||
`<TALOS_VM_IP>:30500` (its own NodePort, the address in the Talos registry-mirror patch).
|
||||
Same registry, two names — hence the two `K8S_REGISTRY` values in the workflow.
|
||||
- It calls `make k8s-reseed`, not `make k8s-up`: the bootstrap Jobs are idempotent, and
|
||||
deleting them first is what stops a changed Job template from wedging `helm upgrade` (§7).
|
||||
- `dev` is a mutable tag, so a `rollout restart` of the nine repo deployments is what
|
||||
actually puts the new images in the pods.
|
||||
- Deploys **queue** (`cancel-in-progress: false`): a helm upgrade killed half-way leaves the
|
||||
release in `pending-upgrade`, which has to be unwedged by hand.
|
||||
|
||||
Settings, all on the repository in Gitea:
|
||||
|
||||
| Kind | Name | What |
|
||||
|---|---|---|
|
||||
| Secret | `TALOS_SSH_KEY` | private key for `user@labs.respellion.tech` (the Fedora host) |
|
||||
| Secret | `TALOS_KUBECONFIG` | base64 of the kubeconfig, **`server: https://127.0.0.1:6443`** — Talos puts `127.0.0.1` in the apiserver cert SANs, so TLS still verifies through the tunnel |
|
||||
| Variable | `TALOS_VM_IP` | the VM's libvirt address (default `192.168.122.173`) |
|
||||
| Variable | `TALOS_HOST` | the browser-facing host baked into Keycloak's issuer (default `localhost`, see §5) |
|
||||
|
||||
The last step smokes `GET /openbaar/register` through the openbaar portal, which exercises
|
||||
portal → Caddy → BFF → projection. An empty register passes; a 502 does not.
|
||||
|
||||
Not covered: the portals still need `make k8s-portals` (or an SSH forward) to be usable in a
|
||||
browser, because PKCE needs a secure context (§5). Giving the server a hostname + TLS is the
|
||||
upgrade path.
|
||||
|
||||
## Publishing through the labs Caddy
|
||||
|
||||
The portals can be reached on real hostnames through the Caddy that already fronts
|
||||
`*.labs.respellion.tech` (repo `Infra`, `infra/development/`). The chain:
|
||||
|
||||
```
|
||||
browser → Caddy (labs server, TLS) → openssh-server:3014x/30180
|
||||
→ reverse SSH tunnel → Fedora host → <TALOS_VM_IP>:3014x/30180 (NodePorts)
|
||||
```
|
||||
|
||||
| URL | NodePort |
|
||||
|---|---|
|
||||
| `https://big-register.labs.respellion.tech` | 30141 openbaar |
|
||||
| `https://big-mijn.labs.respellion.tech` | 30140 self-service |
|
||||
| `https://big-behandel.labs.respellion.tech` | 30142 behandel |
|
||||
| `https://big-beheer.labs.respellion.tech` | 30143 beheer |
|
||||
| `https://big-auth.labs.respellion.tech` | 30180 Keycloak (`/admin` blocked) |
|
||||
|
||||
HTTPS makes the portals a secure context, so PKCE works without port-forwards — but
|
||||
Keycloak's issuer must be the public origin. Deploy with it:
|
||||
|
||||
```bash
|
||||
make k8s-up TALOS_HOST=localhost K8S_REGISTRY=<TALOS_HOST>:30500 \
|
||||
K8S_SET="--set keycloakUrl=https://big-auth.labs.respellion.tech"
|
||||
```
|
||||
|
||||
For deploy-on-merge, set the repository variable `KEYCLOAK_URL` to the same value.
|
||||
With it set, the `localhost` port-forwards (§5) no longer log in: the issuer is one string.
|
||||
|
||||
Staff logins still hit the enforced OTP step. For a demo, set the repository variable
|
||||
`OTP_AUTOFILL=true` (chart value `demo.otpAutofill`): Keycloak then uses the `big-demo`
|
||||
theme, which fills in and submits the code from the fixture secret, so the step is visible
|
||||
but needs no authenticator. Keycloak restarts when the value flips. Demo only — the secret
|
||||
is committed.
|
||||
|
||||
The theme lives in `infra/keycloak/themes/big-demo/` and is seeded as the `rr-kc-theme`
|
||||
ConfigMap by `infra/helm/seed-configmaps.sh` on every deploy. Keycloak runs `start-dev`,
|
||||
which doesn't cache themes, so an edit shows up about a minute after the ConfigMap changes.
|
||||
A *new* theme file also needs a key in the seed script and a path in the keycloak `files`
|
||||
in `values.yaml`.
|
||||
|
||||
One-time setup:
|
||||
|
||||
1. Fedora host: install `infra/development/big-portals-tunnel.service` from the Infra repo
|
||||
(instructions in the file).
|
||||
2. Labs server: deploy the Infra `Caddyfile` + `compose.yml` (Caddy joins the
|
||||
`openssh_default` network to reach the tunnel ends).
|
||||
|
||||
## What is not ported
|
||||
|
||||
- **Observability** (Tempo, Prometheus, Grafana) is defined but disabled — those are built
|
||||
@@ -366,5 +456,7 @@ immutable, so `helm upgrade` is rejected with `cannot patch "…" with kind Job`
|
||||
`K8S_SET='--set workloads.tempo.enabled=true --set workloads.prometheus.enabled=true --set workloads.grafana.enabled=true'`.
|
||||
The .NET services still export OTLP; the exporter fails harmlessly when Tempo is absent.
|
||||
- **The verify/e2e lanes.** `make verify*` and the Playwright e2e drive compose, not the
|
||||
chart. The Kubernetes path is verified with §5's smoke test.
|
||||
chart. The Kubernetes path is verified with §5's smoke test. CI's `k8s` job runs the two
|
||||
clusterless checks (`k8s-lint`, `k8s-drift`) on every PR — a values typo or a compose
|
||||
image bump that skipped the chart fails there, but nothing deploys the chart in CI.
|
||||
- **Ingress, TLS, and resource requests.** See the ponytail ceiling in ADR-0033.
|
||||
|
||||
@@ -57,6 +57,9 @@ services:
|
||||
# share this anchor and ignore it — they don't run uwsgi.
|
||||
UWSGI_PROCESSES: "1"
|
||||
UWSGI_THREADS: "2"
|
||||
# Same lever for oz-celery: unset, the worker forks one process per CPU (22 on the lab node,
|
||||
# ~225 MB each), which OOM-killed the shared runner mid-verify-stack. Only celery reads it.
|
||||
CELERY_WORKER_CONCURRENCY: "2"
|
||||
DJANGO_SETTINGS_MODULE: openzaak.conf.docker
|
||||
SECRET_KEY: ${OZ_SECRET_KEY:-dev-only-not-for-production}
|
||||
DB_HOST: oz-db
|
||||
@@ -144,6 +147,8 @@ services:
|
||||
# 1 uWSGI worker, not the image default of 4×4 (#147) — see the oz-env note above.
|
||||
UWSGI_PROCESSES: "1"
|
||||
UWSGI_THREADS: "2"
|
||||
# Two celery workers, not one per CPU — see the oz-env note above.
|
||||
CELERY_WORKER_CONCURRENCY: "2"
|
||||
DJANGO_SETTINGS_MODULE: nrc.conf.docker
|
||||
SECRET_KEY: ${NRC_SECRET_KEY:-dev-only-not-for-production}
|
||||
DB_HOST: nrc-db
|
||||
|
||||
@@ -94,6 +94,10 @@ volumes:
|
||||
{{- with .defaultMode }}
|
||||
defaultMode: {{ . }}
|
||||
{{- end }}
|
||||
{{- with .items }}
|
||||
items:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- with $w.data }}
|
||||
- name: data
|
||||
@@ -125,14 +129,26 @@ volumes:
|
||||
{{/*
|
||||
Env list from a map. Every value is run through `tpl`, so values.yaml can name
|
||||
cluster-internal hosts ({{ .Release.Namespace }}) and the node address
|
||||
({{ .Values.host }}) without the chart hard-coding either.
|
||||
({{ .Values.host }}) without the chart hard-coding either. A value that renders
|
||||
empty is left out, which is how a setting is made conditional on a chart value.
|
||||
*/}}
|
||||
{{- define "big.env" -}}
|
||||
{{- $root := index . 0 -}}
|
||||
{{- range $k, $v := index . 1 }}
|
||||
{{- $val := tpl (toString $v) $root }}
|
||||
{{- if $val }}
|
||||
- name: {{ $k }}
|
||||
value: {{ tpl (toString $v) $root | quote }}
|
||||
value: {{ $val | quote }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
The origin a browser reaches Keycloak on: the issuer Keycloak pins and the
|
||||
authority the portals use, from one place so they cannot drift (ADR-0010).
|
||||
*/}}
|
||||
{{- define "big.keycloakUrl" -}}
|
||||
{{- .Values.keycloakUrl | default (printf "http://%s:%v" .Values.host (index .Values.nodePorts "keycloak")) -}}
|
||||
{{- end -}}
|
||||
|
||||
{{- define "big.labels" -}}
|
||||
|
||||
@@ -40,5 +40,5 @@ metadata:
|
||||
{{- include "big.labels" (dict "root" $ "name" (printf "portal-config-%s" $realm)) | nindent 4 }}
|
||||
data:
|
||||
config.json: |
|
||||
{ "authority": "{{ printf "http://%s:%v" $.Values.host (index $.Values.nodePorts "keycloak") }}/realms/{{ $realm }}" }
|
||||
{ "authority": "{{ include "big.keycloakUrl" $ }}/realms/{{ $realm }}" }
|
||||
{{- end }}
|
||||
|
||||
@@ -28,7 +28,7 @@ spec:
|
||||
{{- range $w.files }}
|
||||
{{- if hasPrefix "portal-config-" .configMap }}
|
||||
annotations:
|
||||
checksum/portal-config: {{ printf "%s|%v" $.Values.host (index $.Values.nodePorts "keycloak") | sha256sum }}
|
||||
checksum/portal-config: {{ include "big.keycloakUrl" $ | sha256sum }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
labels:
|
||||
|
||||
@@ -25,6 +25,17 @@
|
||||
# string, so browser tokens and the BFF's discovered issuer agree.
|
||||
host: 192.168.122.100
|
||||
|
||||
# Set when a TLS proxy outside the cluster publishes Keycloak: the full origin, no
|
||||
# trailing slash. It replaces `host` + Keycloak's NodePort as the issuer and the
|
||||
# portals' authority (runbook, "Publishing through the labs Caddy").
|
||||
keycloakUrl: ""
|
||||
|
||||
demo:
|
||||
# Fill in and submit the medewerker OTP step from the fixture secret, so a public
|
||||
# demo shows MFA enforced without an authenticator: makes the big-demo theme
|
||||
# (infra/keycloak/themes/big-demo) Keycloak's default. Demo only: the secret is committed.
|
||||
otpAutofill: false
|
||||
|
||||
# Set when pulling from a private registry (e.g. the Gitea Container Registry).
|
||||
imagePullSecrets: []
|
||||
|
||||
@@ -71,6 +82,7 @@ envGroups:
|
||||
oz:
|
||||
UWSGI_PROCESSES: "1"
|
||||
UWSGI_THREADS: "2"
|
||||
CELERY_WORKER_CONCURRENCY: "2"
|
||||
DJANGO_SETTINGS_MODULE: openzaak.conf.docker
|
||||
SECRET_KEY: dev-only-not-for-production
|
||||
DB_HOST: oz-db
|
||||
@@ -93,6 +105,7 @@ envGroups:
|
||||
nrc:
|
||||
UWSGI_PROCESSES: "1"
|
||||
UWSGI_THREADS: "2"
|
||||
CELERY_WORKER_CONCURRENCY: "2"
|
||||
DJANGO_SETTINGS_MODULE: nrc.conf.docker
|
||||
SECRET_KEY: dev-only-not-for-production
|
||||
DB_HOST: nrc-db
|
||||
@@ -268,13 +281,31 @@ workloads:
|
||||
# Pin the issuer to the address the browser uses, and let backchannel calls
|
||||
# keep using keycloak:8080 — the BFF discovers metadata in-cluster and gets
|
||||
# this issuer back, which is what browser tokens carry (infra/host-browser.yml).
|
||||
KC_HOSTNAME: "http://{{ .Values.host }}:{{ index .Values.nodePorts \"keycloak\" }}"
|
||||
KC_HOSTNAME: '{{ include "big.keycloakUrl" . }}'
|
||||
KC_HOSTNAME_BACKCHANNEL_DYNAMIC: "true"
|
||||
# Only rendered with demo.otpAutofill (big.env skips empty values); off, Keycloak
|
||||
# keeps its stock theme and the mounted big-demo theme is unused.
|
||||
KC_SPI_THEME_DEFAULT: '{{ if .Values.demo.otpAutofill }}big-demo{{ end }}'
|
||||
# Behind a TLS proxy (keycloakUrl) the dynamic backchannel URLs — token,
|
||||
# userinfo, certs — take their scheme from the request, which reaches Keycloak
|
||||
# as plain http; trusting X-Forwarded-Proto keeps them https so the browser
|
||||
# doesn't block them as mixed content. In-cluster calls send no such header.
|
||||
KC_PROXY_HEADERS: xforwarded
|
||||
ports: [{ name: http, port: 8080 }]
|
||||
# TCP, not /health/ready on the management port: nothing here gates on realm
|
||||
# import, and a wrong health path would leave the Service with no endpoints.
|
||||
probe: { tcpSocket: { port: 8080 }, initialDelaySeconds: 15 }
|
||||
files: [{ configMap: rr-kc-realms, mountPath: /opt/keycloak/data/import }]
|
||||
files:
|
||||
- { configMap: rr-kc-realms, mountPath: /opt/keycloak/data/import }
|
||||
# infra/keycloak/themes/big-demo, seeded by infra/helm/seed-configmaps.sh.
|
||||
- configMap: rr-kc-theme
|
||||
mountPath: /opt/keycloak/themes/big-demo
|
||||
items:
|
||||
- { key: login.properties, path: login/theme.properties }
|
||||
- { key: otp-autofill.js, path: login/resources/js/otp-autofill.js }
|
||||
- { key: account.properties, path: account/theme.properties }
|
||||
- { key: admin.properties, path: admin/theme.properties }
|
||||
- { key: email.properties, path: email/theme.properties }
|
||||
|
||||
# ── Flowable (S-03) ─────────────────────────────────────────────────────────
|
||||
flowable-db:
|
||||
|
||||
Executable
+116
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fail when the compose stack and the Helm chart stop describing the same stack.
|
||||
|
||||
`infra/docker-compose.yml` is CI-canonical; `infra/helm/big-reference` is a
|
||||
transcription of it (ADR-0033), and until now nothing kept the two in step — an
|
||||
upstream image bump or a new service applied to only one of them landed
|
||||
unnoticed. This compares what each side actually *deploys*, not the two files:
|
||||
the rendered chart against `docker compose config`. Both tools are already
|
||||
prerequisites of the `k8s-*` make targets.
|
||||
|
||||
Run it with `make k8s-drift`. No cluster needed.
|
||||
|
||||
ponytail: names and images only, as sets — no per-workload env/ports/volumes.
|
||||
Those differ by design in four documented places (ADR-0033), so comparing them
|
||||
would mean re-encoding every deviation field by field; a tag bump and a missing
|
||||
service are the drift that actually bites.
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
COMPOSE = ROOT / "infra/docker-compose.yml"
|
||||
CHART = ROOT / "infra/helm/big-reference"
|
||||
|
||||
# The busybox init container that every `waitFor` workload gets exists only in
|
||||
# the chart (compose has `depends_on`). Rendering it under a sentinel makes it
|
||||
# filterable without teaching the check what busybox is.
|
||||
BUSYBOX = "drift-check-ignored-init-image"
|
||||
|
||||
# Differences that Kubernetes forces, not drift (ADR-0033). A name listed here is
|
||||
# expected to be on exactly one side; anything else fails.
|
||||
DEVIATIONS = {
|
||||
# The four Django services apply their own setup_configuration in the web pod
|
||||
# (`args: [sh, -c, "/setup_configuration.sh && exec /start.sh"]`) rather than in a
|
||||
# separate init Job. Both that script and /start.sh run `manage.py migrate`, and
|
||||
# Kubernetes has no `depends_on: service_completed_successfully` to serialise them,
|
||||
# so the Job and its web pod migrated the same database concurrently.
|
||||
"oz-init": "folded into the openzaak pod",
|
||||
"nrc-init": "folded into the nrc-web pod",
|
||||
"objecttypen-init": "folded into the objecttypen pod",
|
||||
"objecten-init": "folded into the objecten pod",
|
||||
# Compose seeds these from the host — the verify scripts `docker cp` the two
|
||||
# scripts into a running container, and docker-compose.local.yml carries
|
||||
# `local-seed` + `nrc-subscribe` for `make local`. A cluster has no host to seed
|
||||
# from, so both became Jobs in the chart.
|
||||
"seed-zaaktype": "compose seeds the catalogus from the host (infra/openzaak/seed_catalogus.py)",
|
||||
"nrc-subscribe": "compose registers the abonnement from the host (infra/local/register-abonnement.py)",
|
||||
}
|
||||
|
||||
# Workloads the observability backplane adds. Off by default in both stacks'
|
||||
# defaults, so they are rendered on purpose here — otherwise their images drift
|
||||
# unwatched.
|
||||
OBSERVABILITY = ["tempo", "prometheus", "grafana"]
|
||||
|
||||
|
||||
def compose_services() -> dict[str, str]:
|
||||
"""Service name -> image, with ${TAG:-default} interpolation already applied."""
|
||||
out = run(["docker", "compose", "-f", str(COMPOSE), "config", "--format", "json"])
|
||||
return {name: svc.get("image", "") for name, svc in json.loads(out)["services"].items()}
|
||||
|
||||
|
||||
def chart_workloads() -> dict[str, str]:
|
||||
"""Workload name -> image, read back out of the rendered manifests."""
|
||||
out = run(
|
||||
["helm", "template", "big", str(CHART), "-n", "big", "--set", f"images.busybox={BUSYBOX}"]
|
||||
+ [f"--set=workloads.{w}.enabled=true" for w in OBSERVABILITY]
|
||||
)
|
||||
workloads = {}
|
||||
for doc in out.split("\n---"):
|
||||
if not re.search(r"^kind: (Deployment|Job)$", doc, re.M):
|
||||
continue
|
||||
name = re.search(r"^ name: (\S+)$", doc, re.M)[1]
|
||||
images = [i for i in re.findall(r"^\s+image: (\S+)$", doc, re.M) if i != BUSYBOX]
|
||||
workloads[name] = images[0]
|
||||
return workloads
|
||||
|
||||
|
||||
def run(argv: list[str]) -> str:
|
||||
proc = subprocess.run(argv, capture_output=True, text=True)
|
||||
if proc.returncode != 0:
|
||||
sys.exit(f"{argv[0]} failed:\n{proc.stderr}")
|
||||
return proc.stdout
|
||||
|
||||
|
||||
def main() -> int:
|
||||
compose, chart = compose_services(), chart_workloads()
|
||||
problems = []
|
||||
|
||||
for name in sorted(set(compose) - set(chart) - set(DEVIATIONS)):
|
||||
problems.append(f" {name}: in docker-compose.yml, not in the chart")
|
||||
for name in sorted(set(chart) - set(compose) - set(DEVIATIONS)):
|
||||
problems.append(f" {name}: in the chart, not in docker-compose.yml")
|
||||
for name in sorted(set(compose) & set(chart)):
|
||||
if compose[name] != chart[name]:
|
||||
problems.append(f" {name}: compose runs {compose[name]}, the chart runs {chart[name]}")
|
||||
|
||||
if problems:
|
||||
print("compose and the Helm chart describe different stacks:\n" + "\n".join(problems))
|
||||
print(
|
||||
"\nPort the change to the other stack, or — if the difference is forced by\n"
|
||||
"Kubernetes — declare it in DEVIATIONS in this file, with the reason."
|
||||
)
|
||||
return 1
|
||||
|
||||
print(f"no drift: {len(chart)} workloads, images identical on both stacks")
|
||||
for name, why in sorted(DEVIATIONS.items()):
|
||||
print(f" deviation (declared): {name} — {why}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -30,6 +30,15 @@ seed() { # name <kubectl --from-file args...>
|
||||
seed rr-oz-config --from-file="$repo/infra/openzaak/setup_configuration/"
|
||||
seed rr-nrc-config --from-file="$repo/infra/opennotificaties/setup_configuration/"
|
||||
seed rr-kc-realms --from-file="$repo/infra/keycloak/realms/"
|
||||
# The big-demo login theme (demo.otpAutofill). ConfigMap keys are flat, so each
|
||||
# file gets a key here and its path back in the keycloak `files` in values.yaml.
|
||||
theme="$repo/infra/keycloak/themes/big-demo"
|
||||
seed rr-kc-theme \
|
||||
--from-file=login.properties="$theme/login/theme.properties" \
|
||||
--from-file=otp-autofill.js="$theme/login/resources/js/otp-autofill.js" \
|
||||
--from-file=account.properties="$theme/account/theme.properties" \
|
||||
--from-file=admin.properties="$theme/admin/theme.properties" \
|
||||
--from-file=email.properties="$theme/email/theme.properties"
|
||||
seed rr-objecttypen-config --from-file="$repo/infra/objecttypen/setup_configuration/"
|
||||
seed rr-objecten-config --from-file="$repo/infra/objecten/setup_configuration/"
|
||||
# register.py + the RegisterRecord JSON schema (the __pycache__ dir is skipped:
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
# The chart makes big-demo the default for every theme type, and Keycloak does not
|
||||
# fall back for a type a theme lacks (the account page then fails), so each type is
|
||||
# declared as a plain child of Keycloak 26's own default.
|
||||
parent=keycloak.v3
|
||||
@@ -0,0 +1,4 @@
|
||||
# The chart makes big-demo the default for every theme type, and Keycloak does not
|
||||
# fall back for a type a theme lacks (the admin page then fails), so each type is
|
||||
# declared as a plain child of Keycloak 26's own default.
|
||||
parent=keycloak.v2
|
||||
@@ -0,0 +1,4 @@
|
||||
# The chart makes big-demo the default for every theme type, and Keycloak does not
|
||||
# fall back for a type a theme lacks (the email page then fails), so each type is
|
||||
# declared as a plain child of Keycloak 26's own default.
|
||||
parent=keycloak
|
||||
@@ -0,0 +1,24 @@
|
||||
// RFC 6238 with Keycloak's default policy (HmacSHA1, 6 digits, 30 s) over the
|
||||
// raw bytes of the medewerker fixture secret — same as tests/e2e/keycloak-login.ts.
|
||||
document.addEventListener('DOMContentLoaded', async () => {
|
||||
const input = document.querySelector('input[name="otp"]');
|
||||
if (!input || !input.form) return;
|
||||
const key = await crypto.subtle.importKey('raw',
|
||||
new TextEncoder().encode('BIGMEDEWERKEROTPSEED'), { name: 'HMAC', hash: 'SHA-1' }, false, ['sign']);
|
||||
// A code is single-use, so a second login in the same window spends the next
|
||||
// counter (Keycloak's look-ahead accepts it). Past that, fill but don't submit,
|
||||
// so a rejected code can't turn into a submit loop.
|
||||
const now = Math.floor(Date.now() / 30000);
|
||||
let last = -1;
|
||||
try { last = Number(sessionStorage.getItem('big-otp-counter')) || -1; } catch {}
|
||||
const counter = Math.max(now, last + 1);
|
||||
const msg = new DataView(new ArrayBuffer(8));
|
||||
msg.setBigUint64(0, BigInt(counter));
|
||||
const mac = new Uint8Array(await crypto.subtle.sign('HMAC', key, msg.buffer));
|
||||
const o = mac[19] & 0x0f;
|
||||
const n = ((mac[o] & 0x7f) << 24 | mac[o + 1] << 16 | mac[o + 2] << 8 | mac[o + 3]) % 1e6;
|
||||
input.value = String(n).padStart(6, '0');
|
||||
if (counter > now + 1) return;
|
||||
try { sessionStorage.setItem('big-otp-counter', String(counter)); } catch {}
|
||||
input.form.requestSubmit();
|
||||
});
|
||||
@@ -0,0 +1,11 @@
|
||||
# Demo login theme for the public Talos deployment: keycloak.v2 plus a script that
|
||||
# fills in and submits the medewerker OTP step from the committed fixture secret
|
||||
# (docs/runbooks/keycloak.md). Only used when the chart's demo.otpAutofill is on —
|
||||
# it then becomes Keycloak's default theme. Never enable it anywhere real.
|
||||
#
|
||||
# Add styles, messages or template overrides here as in any Keycloak theme
|
||||
# (https://www.keycloak.org/ui-customization/themes); new files must also be
|
||||
# listed in infra/helm/seed-configmaps.sh and the keycloak `files` in values.yaml.
|
||||
parent=keycloak.v2
|
||||
import=common/keycloak
|
||||
scripts=js/otp-autofill.js
|
||||
Reference in New Issue
Block a user