Compare commits
9
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
dfa1a370ac | ||
|
|
e8cb1ec7e9 | ||
|
|
de6db7d35b | ||
|
|
d17e79959b | ||
|
|
fc036d53d5 | ||
|
|
9d7e8e5b65 | ||
|
|
17f1f2f809 | ||
|
|
1dd8bd4e1b | ||
|
|
d6b3f9764f |
@@ -41,6 +41,27 @@ jobs:
|
||||
nuget-${{ runner.os }}-
|
||||
- run: make lint
|
||||
|
||||
# The Helm chart's only automated gate: it renders and schema-checks the whole
|
||||
# stack, and checks it still describes the same stack as the compose file
|
||||
# (ADR-0033). No cluster involved — see docs/runbooks/kubernetes-talos.md.
|
||||
k8s:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: https://github.com/actions/checkout@v4
|
||||
# helm as its pinned static binary rather than a marketplace action: one URL,
|
||||
# the same one the Talos runbook §0 gives a developer, and no third-party
|
||||
# action to vet (CLAUDE.md §13). The drift check also needs `docker compose`,
|
||||
# which the runner already has (see docs/runbooks/ci.md).
|
||||
- name: Install helm
|
||||
run: |
|
||||
mkdir -p "$HOME/.local/bin"
|
||||
curl -sSL https://get.helm.sh/helm-v3.16.4-linux-amd64.tar.gz \
|
||||
| tar xz -O linux-amd64/helm > "$HOME/.local/bin/helm"
|
||||
chmod +x "$HOME/.local/bin/helm"
|
||||
echo "$HOME/.local/bin" >> "$GITHUB_PATH"
|
||||
- run: make k8s-lint
|
||||
- run: make k8s-drift
|
||||
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
name: Deploy to Talos
|
||||
|
||||
# A merge to main ships the stack to the Talos cluster on the lab server
|
||||
# (docs/runbooks/kubernetes-talos.md §9). PR CI is the merge gate, so main is
|
||||
# green by construction — this workflow only deploys.
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Queue deploys, never cancel one: a helm upgrade killed half-way leaves the
|
||||
# release in `pending-upgrade` and the next run has to be unwedged by hand.
|
||||
concurrency:
|
||||
group: deploy-talos
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
# The Talos VM as seen from the Fedora host (libvirt guest IP), and the
|
||||
# address a browser uses to reach the cluster. `localhost` is deliberate:
|
||||
# the portals' PKCE needs a secure context, so they are reached over
|
||||
# `kubectl port-forward` — runbook §5. Override with repo variables.
|
||||
TALOS_VM_IP: ${{ vars.TALOS_VM_IP }}
|
||||
TALOS_HOST: ${{ vars.TALOS_HOST }}
|
||||
# Set it and the stack is published over TLS on <sub>.<domain> by the
|
||||
# in-cluster edge (ADR-0035, runbook §10). Empty = NodePorts, as before.
|
||||
PUBLIC_DOMAIN: ${{ vars.PUBLIC_DOMAIN }}
|
||||
PUBLIC_EMAIL: ${{ vars.PUBLIC_EMAIL }}
|
||||
steps:
|
||||
- uses: https://github.com/actions/checkout@v4
|
||||
|
||||
# Pinned static binaries, the same URLs the Talos runbook §0 gives a
|
||||
# developer and the same helm the `k8s` CI job uses — no action to vet.
|
||||
- name: Install kubectl, helm and crane
|
||||
run: |
|
||||
set -euo pipefail
|
||||
bin="$HOME/.local/bin"; mkdir -p "$bin"
|
||||
curl -sSLo "$bin/kubectl" https://dl.k8s.io/release/v1.37.0/bin/linux/amd64/kubectl
|
||||
curl -sSL https://get.helm.sh/helm-v3.16.4-linux-amd64.tar.gz | tar xz -O linux-amd64/helm > "$bin/helm"
|
||||
curl -sSL https://github.com/google/go-containerregistry/releases/download/v0.20.2/go-containerregistry_Linux_x86_64.tar.gz | tar xz -O crane > "$bin/crane"
|
||||
chmod +x "$bin"/{kubectl,helm,crane}
|
||||
echo "$bin" >> "$GITHUB_PATH"
|
||||
|
||||
# The cluster's API and its registry are only reachable through the Fedora
|
||||
# host, so forward both to the runner. 30141 is the openbaar portal, for
|
||||
# the smoke at the end.
|
||||
- name: Tunnel the Talos API + registry through the Fedora host
|
||||
env:
|
||||
SSH_KEY: ${{ secrets.TALOS_SSH_KEY }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
: "${TALOS_VM_IP:=192.168.122.173}"
|
||||
umask 077
|
||||
printf '%s\n' "$SSH_KEY" > ~/.ssh_talos
|
||||
ssh -i ~/.ssh_talos -o StrictHostKeyChecking=no -o IdentitiesOnly=yes \
|
||||
-o ExitOnForwardFailure=yes -p 6667 -f -N \
|
||||
-L 6443:$TALOS_VM_IP:6443 \
|
||||
-L 30500:$TALOS_VM_IP:30500 \
|
||||
-L 30141:$TALOS_VM_IP:30141 \
|
||||
user@labs.respellion.tech
|
||||
|
||||
# The kubeconfig's server must be https://127.0.0.1:6443 — Talos puts
|
||||
# 127.0.0.1 in the apiserver cert SANs, so TLS verification still holds
|
||||
# through the tunnel.
|
||||
- name: Write the kubeconfig
|
||||
env:
|
||||
KUBECONFIG_B64: ${{ secrets.TALOS_KUBECONFIG }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
umask 077
|
||||
base64 -d <<< "$KUBECONFIG_B64" > "$RUNNER_TEMP/kubeconfig"
|
||||
echo "KUBECONFIG=$RUNNER_TEMP/kubeconfig" >> "$GITHUB_ENV"
|
||||
kubectl --kubeconfig "$RUNNER_TEMP/kubeconfig" get nodes
|
||||
|
||||
# Idempotent; also makes a first deploy onto a bare cluster work. The
|
||||
# registry's storage is an emptyDir, so a replaced pod loses the images —
|
||||
# which the push in the next step puts back anyway.
|
||||
- name: Ensure the in-cluster registry
|
||||
run: make k8s-registry
|
||||
|
||||
# Push through the tunnel (localhost), pull from the node's own NodePort
|
||||
# (the address in the Talos registry-mirror patch) — same registry, two
|
||||
# names, so the two `make` calls get different K8S_REGISTRY values.
|
||||
- name: Build and push the images
|
||||
run: make k8s-images K8S_REGISTRY=localhost:30500
|
||||
|
||||
# k8s-reseed = seed configmaps + helm upgrade + re-run the bootstrap jobs.
|
||||
# The jobs are idempotent, and deleting them first is what keeps a changed
|
||||
# Job template from wedging the upgrade (`cannot patch … with kind Job`).
|
||||
- name: Deploy the chart
|
||||
run: |
|
||||
set -euo pipefail
|
||||
publish="${PUBLIC_DOMAIN:+--set public.domain=$PUBLIC_DOMAIN --set public.email=${PUBLIC_EMAIL:-}}"
|
||||
make k8s-reseed \
|
||||
TALOS_HOST=${TALOS_HOST:-localhost} \
|
||||
K8S_REGISTRY=${TALOS_VM_IP:-192.168.122.173}:30500 \
|
||||
K8S_SET="$publish"
|
||||
|
||||
# `dev` is a mutable tag and helm sees an unchanged pod template, so the
|
||||
# new images only land on a restart (pullPolicy is already Always).
|
||||
- name: Roll the services onto the new images
|
||||
run: |
|
||||
set -euo pipefail
|
||||
svcs="acl domain bff event-subscriber projection-api self-service openbaar behandel beheer"
|
||||
kubectl -n big rollout restart deploy $svcs
|
||||
kubectl -n big rollout status --timeout=300s deploy $svcs
|
||||
|
||||
# Proves portal → Caddy → BFF → projection end to end. An empty register is
|
||||
# a pass; a 502 or a timeout is not.
|
||||
- name: Smoke the public register
|
||||
run: curl -fsS --retry 10 --retry-delay 6 --retry-all-errors http://localhost:30141/openbaar/register
|
||||
|
||||
# Cluster-wide, not just `big`: the first thing that can fail is the registry
|
||||
# in its own namespace, and a scheduling problem shows up in the events, not
|
||||
# in `rollout status` — which only ever says "timed out waiting".
|
||||
- name: Pods and events on failure
|
||||
if: failure()
|
||||
run: |
|
||||
kubectl get pods -A -o wide || true
|
||||
kubectl -n big get jobs || true
|
||||
kubectl get events -A --sort-by=.lastTimestamp | tail -30 || true
|
||||
@@ -43,7 +43,7 @@ export DOCKER_HOST := unix://$(PODMAN_SOCK)
|
||||
endif
|
||||
endif
|
||||
|
||||
.PHONY: ci lint build unit mutation frontend integration verify verify-up verify-acl verify-nrc verify-projection verify-bff verify-domain verify-observability verify-tracing verify-metrics verify-objecttypen verify-objecten verify-registerrecord verify-objecten-notifications verify-notifications smoke up down local verify-local local-down changelog openzaak-up openzaak-smoke openzaak-seed openzaak-down stack-up stack-smoke stack-down keycloak-up keycloak-smoke keycloak-down flowable-up flowable-smoke flowable-down k8s-lint k8s-registry k8s-images k8s-seed k8s-up k8s-reseed k8s-portals k8s-down k8s-purge help
|
||||
.PHONY: ci lint build unit mutation frontend integration verify verify-up verify-acl verify-nrc verify-projection verify-bff verify-domain verify-observability verify-tracing verify-metrics verify-objecttypen verify-objecten verify-registerrecord verify-objecten-notifications verify-notifications smoke up down local verify-local local-down changelog openzaak-up openzaak-smoke openzaak-seed openzaak-down stack-up stack-smoke stack-down keycloak-up keycloak-smoke keycloak-down flowable-up flowable-smoke flowable-down k8s-lint k8s-drift k8s-registry k8s-images k8s-seed k8s-up k8s-reseed k8s-portals k8s-down k8s-purge help
|
||||
|
||||
## ci: run the full pipeline — lint, build, unit, mutation, frontend, verify (mirrors Gitea Actions)
|
||||
## `verify` is the live-stack stage (full stack up once → ACL + notification checks).
|
||||
@@ -64,6 +64,9 @@ frontend:
|
||||
## lint: verify formatting (no changes)
|
||||
lint:
|
||||
dotnet format $(SLN) --verify-no-changes
|
||||
# Only pages in mkdocs.yml's nav are published, and mkdocs keeps a build green
|
||||
# when one is missing — so the nav is checked here rather than not at all.
|
||||
python3 infra/check-docs-nav.py
|
||||
|
||||
## build: release build
|
||||
build:
|
||||
@@ -71,8 +74,11 @@ build:
|
||||
|
||||
## unit: run unit tests (excludes the container-backed Integration lane)
|
||||
# TRX per test project (→ TestResults/) feeds the CI per-service summary (#136); harmless locally.
|
||||
# The CI reporting scripts are stdlib Python with their own assert-based self-checks (#161) — they
|
||||
# ride this lane so a broken job summary is caught by CI rather than by the next red pipeline.
|
||||
unit:
|
||||
dotnet test $(SLN) -c Release --filter "Category!=Integration" --logger trx --results-directory TestResults
|
||||
python3 infra/test_playwright_summary.py
|
||||
python3 infra/test_portal_caddyfiles.py
|
||||
|
||||
## mutation: run the Stryker.NET ratchet on each service with branching logic (fails below baseline)
|
||||
@@ -347,6 +353,13 @@ k8s-lint:
|
||||
helm lint $(K8S_CHART)
|
||||
helm template big $(K8S_CHART) -n $(K8S_NS) --set images.registry=registry.invalid:5000 >/dev/null
|
||||
|
||||
## k8s-drift: fail if compose and the Helm chart describe different stacks
|
||||
# Compose is CI-canonical (ADR-0033) and the chart is a transcription of it; this
|
||||
# compares what each one deploys — workload names and resolved images. Needs
|
||||
# `docker compose` and `helm`, no cluster.
|
||||
k8s-drift:
|
||||
python3 infra/helm/check-drift.py
|
||||
|
||||
## k8s-registry: deploy the in-cluster image registry (NodePort 30500)
|
||||
k8s-registry:
|
||||
kubectl apply -f infra/helm/registry.yaml
|
||||
|
||||
@@ -3,8 +3,8 @@
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-09-04
|
||||
- **Deciders:** Respellion engineering
|
||||
- **Slice:** _(none yet — raised directly as a deployment-target request; see
|
||||
"Process note" at the end)_
|
||||
- **Slice:** #25 (S-24) — raised directly as a deployment-target request and matched to
|
||||
that issue afterwards; see the "Process note" at the end
|
||||
|
||||
## Context
|
||||
|
||||
@@ -126,8 +126,9 @@ Consequences of that shape, each chosen deliberately:
|
||||
|
||||
**Negative / costs**
|
||||
|
||||
- A second deployment description to keep in step with compose. Nothing enforces that
|
||||
today; a drift check belongs in CI (follow-up).
|
||||
- A second deployment description to keep in step with compose. `make k8s-drift` (#168)
|
||||
now enforces the part that bites — the workload set and the resolved images, with the
|
||||
four deviations below declared — but not per-workload env, ports or volumes.
|
||||
- `helm install` alone is not enough — the ConfigMaps must be seeded first, and a missing
|
||||
one surfaces as `ContainerCreating`, not as a clear error.
|
||||
- Generic templates mean a values typo can render valid-but-wrong YAML; `k8s-lint` catches
|
||||
|
||||
@@ -14,6 +14,8 @@ should teach.
|
||||
In Dutch; the strategic framing lives in `Respellion/innovation-lab`.
|
||||
- **[Working in Gitea](gitea-workflow.md)** — issues, milestones, branches, PRs.
|
||||
- **[CI runbook](runbooks/ci.md)** — the pipeline and the `make ci` local gate.
|
||||
- **[Kubernetes on Talos](runbooks/kubernetes-talos.md)** — the second deployment target:
|
||||
one Helm chart, a single-node cluster, and the parts that bite (ADR-0033).
|
||||
|
||||
## Quickstart
|
||||
|
||||
|
||||
+10
-2
@@ -2,8 +2,10 @@
|
||||
|
||||
> **Status: active.** The workflow `.gitea/workflows/ci.yaml` runs on Gitea's
|
||||
> hosted `ubuntu-latest` runner — no self-hosted runner required.
|
||||
> **`make ci` is still the local gate** — it runs the exact same checks
|
||||
> (the workflow calls the same `make` targets).
|
||||
> **`make ci` is still the local gate** — it runs the same checks via the same
|
||||
> `make` targets, with one exception: the `k8s` job's targets are not in `make ci`,
|
||||
> because `helm` is optional for everyone not deploying to Kubernetes. Run
|
||||
> `make k8s-lint k8s-drift` by hand after touching the chart or the compose file.
|
||||
|
||||
## The pipeline
|
||||
|
||||
@@ -16,6 +18,8 @@ and CI cannot drift:
|
||||
| `lint` | `make lint` → `dotnet format … --verify-no-changes` | .NET 10 SDK |
|
||||
| `build` | `make build` → `dotnet build … -c Release` | .NET 10 SDK |
|
||||
| `unit` | `make unit` → `dotnet test … -c Release --filter "Category!=Integration"` | .NET 10 SDK |
|
||||
| `frontend` | `make frontend` → Nx lint/test/build for the four portals | pnpm + Node |
|
||||
| `k8s` | `make k8s-lint` (render + schema-check the Helm chart) → `make k8s-drift` (chart still describes the same stack as `infra/docker-compose.yml`) | pinned `helm` binary + `docker compose` |
|
||||
| `mutation` | `make mutation` → `dotnet tool restore` → `dotnet stryker` (ACL); uploads the HTML report as an artifact | .NET 10 SDK |
|
||||
| `verify-stack` | the single live-stack stage — steps: `make verify-up` (full stack up + health, the DoD smoke) → `make verify-acl` (ACL ↔ OpenZaak) → `make verify-nrc` (OpenZaak → NRC delivery) → `make down` | container engine + egress (base images, nuget, `selectielijst.openzaak.nl`) |
|
||||
|
||||
@@ -27,6 +31,10 @@ and CI cannot drift:
|
||||
> services by **container IP** (the runner can't reach published ports — see
|
||||
> [gitea-actions-gotchas.md §5/§6](gitea-actions-gotchas.md)).
|
||||
|
||||
A second workflow, `.gitea/workflows/deploy.yaml`, deploys the stack to the Talos
|
||||
cluster on the lab server when a PR is merged to `main` — see
|
||||
[kubernetes-talos.md §9](kubernetes-talos.md) for its secrets and the SSH tunnel it needs.
|
||||
|
||||
All `uses:` references are absolute, tag-pinned URLs (`https://github.com/actions/checkout@v4`,
|
||||
`https://github.com/actions/setup-dotnet@v4`) per CLAUDE.md §8.7 and §15 — Gitea
|
||||
Actions resolves them from GitHub.
|
||||
|
||||
@@ -245,3 +245,47 @@ the verify-stack check table, and per-spec e2e results (`infra/playwright-summar
|
||||
- Getting a report out of the e2e container: Playwright writes `playwright-report.json`
|
||||
inside the container; `infra/run-e2e-check.sh` `docker cp`s it back to the host
|
||||
(capturing the test exit code first) so the summary step can read it.
|
||||
|
||||
---
|
||||
|
||||
## 9. `if: always()` does not survive the job being killed — bound the work itself
|
||||
|
||||
`if: always()` makes a step run when an *earlier step failed*. It does **not** help when
|
||||
the job as a whole is stopped: the run's remaining steps are simply never dispatched.
|
||||
|
||||
That is how #161 lost its diagnosis. `verify-stack` entered `make verify-e2e` at 09:48:17
|
||||
and the job ended at 10:14:54 — 26½ minutes later, mid-suite. Every step after the e2e
|
||||
shows a **0-second `failure`** stamped at that same instant:
|
||||
|
||||
```
|
||||
14 failure 09:48:17 -> 10:14:54 Self-service e2e (Playwright, login → submit → success)
|
||||
15 failure 10:14:54 -> 10:14:54 verify-stack check summary ← if: always()
|
||||
16 failure 10:14:54 -> 10:14:54 e2e spec summary ← if: always()
|
||||
17 failure 10:14:54 -> 10:14:54 Dump container logs on failure ← if: failure()
|
||||
18 failure 10:14:54 -> 10:14:54 Tear down ← if: always()
|
||||
```
|
||||
|
||||
So the per-spec summary, the container-log dump and the teardown never ran, and the job
|
||||
log — which also loses whatever the killed process had buffered — ended at a single `✘`
|
||||
line. A job that dies takes its own post-mortem with it.
|
||||
|
||||
**Read the step timings, not just the log.** `GET /api/v1/repos/{owner}/{repo}/actions/jobs/{id}`
|
||||
returns every step with `started_at`/`completed_at`; a row of identical zero-length
|
||||
steps at the end means *killed*, not *silent*. (Job ids come from
|
||||
`…/actions/runs/{run}/jobs`, and that route returns only the **latest attempt** — a
|
||||
re-run hides the failed one, so keep the failing job id from the original report. Logs:
|
||||
`…/actions/jobs/{id}/logs`, see also `gitea-ci-logs`.)
|
||||
|
||||
**Conventions that follow:**
|
||||
|
||||
- **Bound long-running work inside the tool**, where it can still report. Playwright's
|
||||
`globalTimeout` (`tests/e2e/playwright.config.ts`) ends the run, writes the JSON
|
||||
report and exits, so the summary and log-dump steps still get their turn. A
|
||||
`timeout-minutes` on the job would reproduce the very failure above.
|
||||
- **Never let an auto-waiting action be the timeout.** Playwright actions (`fill`,
|
||||
`click`) inherit the *test* timeout, not `expect.timeout`, so a missing element costs
|
||||
the full 90 s and reports `locator.fill: Test timeout …` — the symptom. Assert the
|
||||
element visible first with its own budget and a message (`tests/e2e/keycloak-login.ts`).
|
||||
- Remember `concurrency.cancel-in-progress: true` in `ci.yaml`: a new push to the same
|
||||
ref, or a re-run, kills the in-flight run the same way. Check `run_attempt` before
|
||||
concluding a job hung.
|
||||
|
||||
@@ -322,6 +322,7 @@ The PVCs carry `helm.sh/resource-policy: keep`, so `make k8s-down` leaves the da
|
||||
|
||||
```bash
|
||||
make k8s-lint # render + schema-check the chart, no cluster needed
|
||||
make k8s-drift # fail if compose and the chart describe different stacks
|
||||
make k8s-portals # forward the portals + Keycloak to localhost (browser access)
|
||||
make k8s-images K8S_REGISTRY=... # after changing a service or a portal
|
||||
make k8s-up TALOS_HOST=... K8S_REGISTRY=...
|
||||
@@ -359,6 +360,97 @@ immutable, so `helm upgrade` is rejected with `cannot patch "…" with kind Job`
|
||||
| Pods `Evicted` / `OOMKilled` | the VM is too small (§0) |
|
||||
| A Job shows `BackoffLimitExceeded` | read it: `kubectl -n big logs job/<name>` |
|
||||
|
||||
## 9. Deploying on merge to main
|
||||
|
||||
`.gitea/workflows/deploy.yaml` runs the §3–§4 steps against the **lab server's** Talos VM
|
||||
every time a PR is squash-merged to `main` (and on demand via *Run workflow*). PR CI is the
|
||||
merge gate, so the workflow deploys without re-running the checks.
|
||||
|
||||
**Prerequisite: the VM must have been installed with the §1 patch.** A stock Talos config
|
||||
gives you a node that still carries the control-plane taint and knows nothing about the
|
||||
plain-HTTP registry, and the deploy hits those in that order: the `registry` pod sits
|
||||
`Pending` until `rollout status` times out, and once that is fixed every repo image fails to
|
||||
pull. Two separate fixes:
|
||||
|
||||
```bash
|
||||
# on the Fedora host — 1. let workloads onto the only node (§1)
|
||||
export KUBECONFIG=~/talos-kubeconfig-local
|
||||
kubectl taint node --all node-role.kubernetes.io/control-plane-
|
||||
|
||||
# 2. trust the in-cluster registry over plain HTTP (§2)
|
||||
cat > /tmp/registry-patch.yaml <<'YAML'
|
||||
machine:
|
||||
registries:
|
||||
mirrors:
|
||||
"<TALOS_VM_IP>:30500":
|
||||
endpoints:
|
||||
- http://<TALOS_VM_IP>:30500
|
||||
YAML
|
||||
talosctl -n <TALOS_VM_IP> -e <TALOS_VM_IP> patch mc --patch @/tmp/registry-patch.yaml
|
||||
```
|
||||
|
||||
Keep those two apart. On Talos 1.14 a patch that also sets
|
||||
`cluster.allowSchedulingOnControlPlanes` is rejected with *".cluster.allowSchedulingOnControlPlanes
|
||||
is already set in v1alpha1 config"* — the field moved out of the v1alpha1 schema, the same way
|
||||
`machine.install` did (§1) — and the rejection takes the whole patch with it, so the mirror
|
||||
silently doesn't land either. `kubectl taint` is the documented way (§1); it is undone if the
|
||||
node ever re-registers, which is a reboot, not a deploy.
|
||||
|
||||
The cluster's API and registry are not exposed publicly, so the job forwards them over the
|
||||
same SSH hop the Gitea-runner pipeline uses:
|
||||
|
||||
```
|
||||
ssh -p 6667 user@labs.respellion.tech -L 6443 -L 30500 -L 30141 → <TALOS_VM_IP>
|
||||
```
|
||||
|
||||
Consequences worth knowing:
|
||||
|
||||
- Images are **pushed** to `localhost:30500` (the tunnel) and **pulled** by the node from
|
||||
`<TALOS_VM_IP>:30500` (its own NodePort, the address in the Talos registry-mirror patch).
|
||||
Same registry, two names — hence the two `K8S_REGISTRY` values in the workflow.
|
||||
- It calls `make k8s-reseed`, not `make k8s-up`: the bootstrap Jobs are idempotent, and
|
||||
deleting them first is what stops a changed Job template from wedging `helm upgrade` (§7).
|
||||
- `dev` is a mutable tag, so a `rollout restart` of the nine repo deployments is what
|
||||
actually puts the new images in the pods.
|
||||
- Deploys **queue** (`cancel-in-progress: false`): a helm upgrade killed half-way leaves the
|
||||
release in `pending-upgrade`, which has to be unwedged by hand.
|
||||
|
||||
Settings, all on the repository in Gitea:
|
||||
|
||||
| Kind | Name | What |
|
||||
|---|---|---|
|
||||
| Secret | `TALOS_SSH_KEY` | private key for `user@labs.respellion.tech` (the Fedora host) |
|
||||
| Secret | `TALOS_KUBECONFIG` | base64 of the kubeconfig, **`server: https://127.0.0.1:6443`** — Talos puts `127.0.0.1` in the apiserver cert SANs, so TLS still verifies through the tunnel |
|
||||
| Variable | `TALOS_VM_IP` | the VM's libvirt address (default `192.168.122.173`) |
|
||||
| Variable | `TALOS_HOST` | the browser-facing host baked into Keycloak's issuer (default `localhost`, see §5) |
|
||||
|
||||
The last step smokes `GET /openbaar/register` through the openbaar portal, which exercises
|
||||
portal → Caddy → BFF → projection. An empty register passes; a 502 does not.
|
||||
|
||||
### Reaching the portals from a laptop
|
||||
|
||||
The deployed portals are pinned to `http://localhost:30180` for Keycloak (§5), so a browser
|
||||
needs **all five** browser-facing ports on its own localhost — the portal alone is not
|
||||
enough, and a missing Keycloak shows up as `ERR_CONNECTION_REFUSED` on
|
||||
`/realms/*/.well-known/openid-configuration` followed by an opaque `ERROR Error: [object Object]`.
|
||||
`make k8s-portals` does this when kubectl can reach the cluster; through the lab server one
|
||||
SSH does it without a kubeconfig at all:
|
||||
|
||||
```bash
|
||||
ssh -N -p 6667 \
|
||||
-L 30140:<TALOS_VM_IP>:30140 -L 30141:<TALOS_VM_IP>:30141 \
|
||||
-L 30142:<TALOS_VM_IP>:30142 -L 30143:<TALOS_VM_IP>:30143 \
|
||||
-L 30180:<TALOS_VM_IP>:30180 \
|
||||
user@labs.respellion.tech
|
||||
```
|
||||
|
||||
Then the §5 table's URLs work as written. The admin UIs (OpenZaak, Flowable, …) need no
|
||||
forward — they are server-rendered, so the VM's address is fine.
|
||||
|
||||
Not covered: the portals still need `make k8s-portals` (or an SSH forward) to be usable in a
|
||||
browser, because PKCE needs a secure context (§5). Giving the server a hostname + TLS is the
|
||||
upgrade path.
|
||||
|
||||
## What is not ported
|
||||
|
||||
- **Observability** (Tempo, Prometheus, Grafana) is defined but disabled — those are built
|
||||
@@ -366,5 +458,7 @@ immutable, so `helm upgrade` is rejected with `cannot patch "…" with kind Job`
|
||||
`K8S_SET='--set workloads.tempo.enabled=true --set workloads.prometheus.enabled=true --set workloads.grafana.enabled=true'`.
|
||||
The .NET services still export OTLP; the exporter fails harmlessly when Tempo is absent.
|
||||
- **The verify/e2e lanes.** `make verify*` and the Playwright e2e drive compose, not the
|
||||
chart. The Kubernetes path is verified with §5's smoke test.
|
||||
chart. The Kubernetes path is verified with §5's smoke test. CI's `k8s` job runs the two
|
||||
clusterless checks (`k8s-lint`, `k8s-drift`) on every PR — a values typo or a compose
|
||||
image bump that skipped the chart fails there, but nothing deploys the chart in CI.
|
||||
- **Ingress, TLS, and resource requests.** See the ponytail ceiling in ADR-0033.
|
||||
|
||||
Executable
+30
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fail when a page under docs/ is missing from mkdocs.yml's nav.
|
||||
|
||||
docs/ is the source of truth (CLAUDE.md §12), but only the pages listed in the nav
|
||||
are published — and mkdocs' own `omitted_files: warn` keeps a build green while
|
||||
silently dropping them, which is how every ADR after 0010 and every runbook but
|
||||
ci.md fell off the site.
|
||||
|
||||
ponytail: a substring test, not a YAML parse — a page's path either appears in
|
||||
mkdocs.yml or it doesn't, and that needs no dependency.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
nav = (ROOT / "mkdocs.yml").read_text()
|
||||
|
||||
missing = sorted(
|
||||
str(page.relative_to(ROOT / "docs"))
|
||||
for page in (ROOT / "docs").rglob("*.md")
|
||||
if str(page.relative_to(ROOT / "docs")) not in nav
|
||||
)
|
||||
|
||||
if missing:
|
||||
print(f"{len(missing)} page(s) under docs/ are not in mkdocs.yml's nav:")
|
||||
print("\n".join(f" {m}" for m in missing))
|
||||
sys.exit(1)
|
||||
|
||||
print("docs nav complete: every page under docs/ is published")
|
||||
Executable
+116
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fail when the compose stack and the Helm chart stop describing the same stack.
|
||||
|
||||
`infra/docker-compose.yml` is CI-canonical; `infra/helm/big-reference` is a
|
||||
transcription of it (ADR-0033), and until now nothing kept the two in step — an
|
||||
upstream image bump or a new service applied to only one of them landed
|
||||
unnoticed. This compares what each side actually *deploys*, not the two files:
|
||||
the rendered chart against `docker compose config`. Both tools are already
|
||||
prerequisites of the `k8s-*` make targets.
|
||||
|
||||
Run it with `make k8s-drift`. No cluster needed.
|
||||
|
||||
ponytail: names and images only, as sets — no per-workload env/ports/volumes.
|
||||
Those differ by design in four documented places (ADR-0033), so comparing them
|
||||
would mean re-encoding every deviation field by field; a tag bump and a missing
|
||||
service are the drift that actually bites.
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
COMPOSE = ROOT / "infra/docker-compose.yml"
|
||||
CHART = ROOT / "infra/helm/big-reference"
|
||||
|
||||
# The busybox init container that every `waitFor` workload gets exists only in
|
||||
# the chart (compose has `depends_on`). Rendering it under a sentinel makes it
|
||||
# filterable without teaching the check what busybox is.
|
||||
BUSYBOX = "drift-check-ignored-init-image"
|
||||
|
||||
# Differences that Kubernetes forces, not drift (ADR-0033). A name listed here is
|
||||
# expected to be on exactly one side; anything else fails.
|
||||
DEVIATIONS = {
|
||||
# The four Django services apply their own setup_configuration in the web pod
|
||||
# (`args: [sh, -c, "/setup_configuration.sh && exec /start.sh"]`) rather than in a
|
||||
# separate init Job. Both that script and /start.sh run `manage.py migrate`, and
|
||||
# Kubernetes has no `depends_on: service_completed_successfully` to serialise them,
|
||||
# so the Job and its web pod migrated the same database concurrently.
|
||||
"oz-init": "folded into the openzaak pod",
|
||||
"nrc-init": "folded into the nrc-web pod",
|
||||
"objecttypen-init": "folded into the objecttypen pod",
|
||||
"objecten-init": "folded into the objecten pod",
|
||||
# Compose seeds these from the host — the verify scripts `docker cp` the two
|
||||
# scripts into a running container, and docker-compose.local.yml carries
|
||||
# `local-seed` + `nrc-subscribe` for `make local`. A cluster has no host to seed
|
||||
# from, so both became Jobs in the chart.
|
||||
"seed-zaaktype": "compose seeds the catalogus from the host (infra/openzaak/seed_catalogus.py)",
|
||||
"nrc-subscribe": "compose registers the abonnement from the host (infra/local/register-abonnement.py)",
|
||||
}
|
||||
|
||||
# Workloads the observability backplane adds. Off by default in both stacks'
|
||||
# defaults, so they are rendered on purpose here — otherwise their images drift
|
||||
# unwatched.
|
||||
OBSERVABILITY = ["tempo", "prometheus", "grafana"]
|
||||
|
||||
|
||||
def compose_services() -> dict[str, str]:
|
||||
"""Service name -> image, with ${TAG:-default} interpolation already applied."""
|
||||
out = run(["docker", "compose", "-f", str(COMPOSE), "config", "--format", "json"])
|
||||
return {name: svc.get("image", "") for name, svc in json.loads(out)["services"].items()}
|
||||
|
||||
|
||||
def chart_workloads() -> dict[str, str]:
|
||||
"""Workload name -> image, read back out of the rendered manifests."""
|
||||
out = run(
|
||||
["helm", "template", "big", str(CHART), "-n", "big", "--set", f"images.busybox={BUSYBOX}"]
|
||||
+ [f"--set=workloads.{w}.enabled=true" for w in OBSERVABILITY]
|
||||
)
|
||||
workloads = {}
|
||||
for doc in out.split("\n---"):
|
||||
if not re.search(r"^kind: (Deployment|Job)$", doc, re.M):
|
||||
continue
|
||||
name = re.search(r"^ name: (\S+)$", doc, re.M)[1]
|
||||
images = [i for i in re.findall(r"^\s+image: (\S+)$", doc, re.M) if i != BUSYBOX]
|
||||
workloads[name] = images[0]
|
||||
return workloads
|
||||
|
||||
|
||||
def run(argv: list[str]) -> str:
|
||||
proc = subprocess.run(argv, capture_output=True, text=True)
|
||||
if proc.returncode != 0:
|
||||
sys.exit(f"{argv[0]} failed:\n{proc.stderr}")
|
||||
return proc.stdout
|
||||
|
||||
|
||||
def main() -> int:
|
||||
compose, chart = compose_services(), chart_workloads()
|
||||
problems = []
|
||||
|
||||
for name in sorted(set(compose) - set(chart) - set(DEVIATIONS)):
|
||||
problems.append(f" {name}: in docker-compose.yml, not in the chart")
|
||||
for name in sorted(set(chart) - set(compose) - set(DEVIATIONS)):
|
||||
problems.append(f" {name}: in the chart, not in docker-compose.yml")
|
||||
for name in sorted(set(compose) & set(chart)):
|
||||
if compose[name] != chart[name]:
|
||||
problems.append(f" {name}: compose runs {compose[name]}, the chart runs {chart[name]}")
|
||||
|
||||
if problems:
|
||||
print("compose and the Helm chart describe different stacks:\n" + "\n".join(problems))
|
||||
print(
|
||||
"\nPort the change to the other stack, or — if the difference is forced by\n"
|
||||
"Kubernetes — declare it in DEVIATIONS in this file, with the reason."
|
||||
)
|
||||
return 1
|
||||
|
||||
print(f"no drift: {len(chart)} workloads, images identical on both stacks")
|
||||
for name, why in sorted(DEVIATIONS.items()):
|
||||
print(f" deviation (declared): {name} — {why}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,20 @@
|
||||
# Overlay: make the CI compose stack usable from a HOST browser.
|
||||
# Same two mechanisms infra/docker-compose.local.yml already uses — pin Keycloak's issuer to the
|
||||
# host-published address, and point each portal's runtime config.json at it. The BFF needs no
|
||||
# change: it discovers metadata over keycloak:8080 and the discovered issuer is the pinned
|
||||
# localhost:8180, which is what browser tokens carry.
|
||||
services:
|
||||
keycloak:
|
||||
environment:
|
||||
KC_HOSTNAME: http://localhost:8180
|
||||
KC_HOSTNAME_BACKCHANNEL_DYNAMIC: "true"
|
||||
self-service:
|
||||
volumes:
|
||||
- ./local-config/self-service.config.json:/usr/share/caddy/config.json:ro,z
|
||||
behandel:
|
||||
volumes:
|
||||
- ./local-config/behandel.config.json:/usr/share/caddy/config.json:ro,z
|
||||
# beheer is the same medewerker realm as behandel, so it reuses behandel's config verbatim.
|
||||
beheer:
|
||||
volumes:
|
||||
- ./local-config/behandel.config.json:/usr/share/caddy/config.json:ro,z
|
||||
@@ -7,10 +7,34 @@ redirects it into $GITHUB_STEP_SUMMARY. Stdlib only.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
STATUS_ICON = {"expected": "✅", "unexpected": "❌", "skipped": "⏭️", "flaky": "⚠️"}
|
||||
|
||||
# A verdict alone still costs a log dive, and a killed or truncated job leaves no log to dive into
|
||||
# (#161) — so a failing spec carries its first error into the table. Playwright errors are multi-line
|
||||
# with a "Call log:", which a markdown table cell cannot hold, so they are flattened and clipped.
|
||||
ERROR_CLIP = 300
|
||||
|
||||
|
||||
def first_error(spec):
|
||||
"""The first error message across a spec's test results, flattened for one table cell."""
|
||||
for test in spec.get("tests", []):
|
||||
for result in test.get("results", []):
|
||||
for error in result.get("errors", []):
|
||||
message = (error.get("message") or "").strip()
|
||||
if not message:
|
||||
continue
|
||||
# Strip ANSI colour, collapse to one line, and keep it inside the cell.
|
||||
message = re.sub(r"\x1b\[[0-9;]*m", "", message)
|
||||
message = " ".join(message.split())
|
||||
if len(message) > ERROR_CLIP:
|
||||
message = message[:ERROR_CLIP - 1].rstrip() + "…"
|
||||
# `|` would end the cell early.
|
||||
return message.replace("|", "\\|")
|
||||
return ""
|
||||
|
||||
|
||||
def walk(suite, out):
|
||||
for spec in suite.get("specs", []):
|
||||
@@ -22,7 +46,8 @@ def walk(suite, out):
|
||||
else "expected" if spec.get("ok", False)
|
||||
else "unexpected")
|
||||
out.append({"file": spec.get("file") or suite.get("file") or suite.get("title", ""),
|
||||
"title": spec.get("title", ""), "status": status})
|
||||
"title": spec.get("title", ""), "status": status,
|
||||
"error": first_error(spec) if status in ("unexpected", "flaky") else ""})
|
||||
for child in suite.get("suites", []):
|
||||
walk(child, out)
|
||||
|
||||
@@ -46,10 +71,17 @@ def main(path):
|
||||
if not specs:
|
||||
print("_No specs ran._")
|
||||
return 0
|
||||
print("| Spec | Result |")
|
||||
print("| ---- | :----: |")
|
||||
for s in specs:
|
||||
print(f"| {s['file']} › {s['title']} | {STATUS_ICON.get(s['status'], '❔')} |")
|
||||
# The failure column only earns its width when something failed.
|
||||
if any(s["error"] for s in specs):
|
||||
print("| Spec | Result | Why |")
|
||||
print("| ---- | :----: | --- |")
|
||||
for s in specs:
|
||||
print(f"| {s['file']} › {s['title']} | {STATUS_ICON.get(s['status'], '❔')} | {s['error']} |")
|
||||
else:
|
||||
print("| Spec | Result |")
|
||||
print("| ---- | :----: |")
|
||||
for s in specs:
|
||||
print(f"| {s['file']} › {s['title']} | {STATUS_ICON.get(s['status'], '❔')} |")
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Self-check for infra/playwright-summary.py — stdlib asserts, no framework.
|
||||
|
||||
Run: python3 infra/test_playwright_summary.py (also runs in `make unit`).
|
||||
|
||||
A red e2e is only useful if the job summary says WHY it failed: #161 lost a 36-minute
|
||||
verify-stack job whose only surviving output was one ✘ line with no assertion detail.
|
||||
"""
|
||||
import importlib.util
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from contextlib import redirect_stdout
|
||||
|
||||
# The script's filename is not a valid module name, so load it by path.
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"playwright_summary",
|
||||
os.path.join(os.path.dirname(os.path.abspath(__file__)), "playwright-summary.py"),
|
||||
)
|
||||
summary = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(summary)
|
||||
|
||||
|
||||
def render(report):
|
||||
"""Run the renderer over a report dict and return its markdown."""
|
||||
with tempfile.NamedTemporaryFile("w", suffix=".json", delete=False) as fh:
|
||||
json.dump(report, fh)
|
||||
path = fh.name
|
||||
try:
|
||||
out = io.StringIO()
|
||||
with redirect_stdout(out):
|
||||
summary.main(path)
|
||||
return out.getvalue()
|
||||
finally:
|
||||
os.unlink(path)
|
||||
|
||||
|
||||
def spec_entry(title, status, errors=()):
|
||||
return {
|
||||
"title": title,
|
||||
"file": "catalogus.spec.ts",
|
||||
"ok": status == "expected",
|
||||
"tests": [{"status": status, "results": [{"errors": [{"message": m} for m in errors]}]}],
|
||||
}
|
||||
|
||||
|
||||
def test_failing_spec_reports_why():
|
||||
md = render({
|
||||
"stats": {"expected": 4, "unexpected": 1, "flaky": 0, "skipped": 0, "duration": 108_000},
|
||||
"suites": [{"file": "catalogus.spec.ts", "specs": [
|
||||
spec_entry("a beheerder sees the published zaaktypen in the catalogus", "unexpected",
|
||||
["locator.fill: Test timeout of 90000ms exceeded.\n"
|
||||
"Call log:\n - waiting for locator('#username')\n"]),
|
||||
]}],
|
||||
})
|
||||
assert "❌" in md, md
|
||||
# The point of the slice: the summary names the cause, not just the verdict.
|
||||
assert "Test timeout of 90000ms exceeded" in md, md
|
||||
assert "waiting for locator('#username')" in md, md
|
||||
# A multi-line Playwright error must not break out of its table row.
|
||||
assert not any(line.startswith("Call log:") for line in md.splitlines()), md
|
||||
|
||||
|
||||
def test_real_playwright_error_is_flattened():
|
||||
# A real report's message is multi-line and ANSI-coloured, and embeds the source snippet with
|
||||
# `|` gutters — all three would break the table cell. Shape verified against an actual
|
||||
# @playwright/test 1.61 JSON report.
|
||||
md = render({
|
||||
"stats": {"expected": 0, "unexpected": 1, "flaky": 0, "skipped": 0, "duration": 1_000},
|
||||
"suites": [{"file": "catalogus.spec.ts", "specs": [
|
||||
spec_entry("a beheerder sees the catalogus", "unexpected",
|
||||
["Error: expect(locator).toBeVisible() failed\n\n"
|
||||
"\x1b[2mLocator: \x1b[22mgetByRole('heading')\n"
|
||||
" 12 | await login(page);\n> 13 | await expect(heading).toBeVisible();\n"]),
|
||||
]}],
|
||||
})
|
||||
row = [line for line in md.splitlines() if line.startswith("| catalogus.spec.ts")][0]
|
||||
assert "\x1b" not in row, row
|
||||
assert "Locator: getByRole('heading')" in row, row
|
||||
# Every literal `|` from the snippet gutters is escaped, so the row keeps exactly 3 cells.
|
||||
assert row.count("|") - row.count("\\|") == 4, row
|
||||
|
||||
|
||||
def test_passing_run_stays_quiet():
|
||||
md = render({
|
||||
"stats": {"expected": 1, "unexpected": 0, "flaky": 0, "skipped": 0, "duration": 5_000},
|
||||
"suites": [{"file": "catalogus.spec.ts",
|
||||
"specs": [spec_entry("a beheerder sees the catalogus", "expected")]}],
|
||||
})
|
||||
assert "✅" in md, md
|
||||
assert "timeout" not in md.lower(), md
|
||||
|
||||
|
||||
def test_missing_report_is_not_a_crash():
|
||||
out = io.StringIO()
|
||||
with redirect_stdout(out):
|
||||
rc = summary.main("/nonexistent/playwright-report.json")
|
||||
assert rc == 0
|
||||
assert "did not reach the e2e step" in out.getvalue()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
for name, fn in sorted(globals().items()):
|
||||
if name.startswith("test_") and callable(fn):
|
||||
fn()
|
||||
print(f" ok {name}")
|
||||
print("playwright-summary self-check passed")
|
||||
+31
@@ -32,6 +32,30 @@ nav:
|
||||
- "ADR-0008: Read projection store": architecture/adr-0008-read-projection-store.md
|
||||
- "ADR-0009: External-task job worker": architecture/adr-0009-external-task-job-worker.md
|
||||
- "ADR-0010: BFF OIDC validation": architecture/adr-0010-bff-oidc.md
|
||||
- "ADR-0011: Approval status flow": architecture/adr-0011-approval-status-flow.md
|
||||
- "ADR-0012: Citizen reference correlation": architecture/adr-0012-citizen-reference-correlation.md
|
||||
- "ADR-0013: Behandel-portal wiring": architecture/adr-0013-behandel-portal-wiring.md
|
||||
- "ADR-0014: Withdrawal cancels the process": architecture/adr-0014-withdrawal-cancels-the-process.md
|
||||
- "ADR-0015: Beoordeling escalation": architecture/adr-0015-beoordeling-escalation.md
|
||||
- "ADR-0016: Diploma eligibility DMN": architecture/adr-0016-diploma-eligibility-dmn.md
|
||||
- "ADR-0017: Document-wait timeout": architecture/adr-0017-document-wait-timeout-cancellation.md
|
||||
- "ADR-0018: Diploma upload via the ACL": architecture/adr-0018-diploma-upload-via-acl-documenten.md
|
||||
- "ADR-0019: Zaak cancellation on timeout": architecture/adr-0019-zaak-cancellation-on-timeout.md
|
||||
- "ADR-0020: Local stack self-seeds": architecture/adr-0020-local-stack-self-seeds.md
|
||||
- "ADR-0021: Zaaktype by identificatie": architecture/adr-0021-acl-resolves-zaaktype-by-identificatie.md
|
||||
- "ADR-0022: Quartz scheduler": architecture/adr-0022-quartz-scheduler.md
|
||||
- "ADR-0023: Observability stack": architecture/adr-0023-observability-stack.md
|
||||
- "ADR-0024: Prometheus AspNetCore exporter": architecture/adr-0024-prometheus-aspnetcore-exporter.md
|
||||
- "ADR-0025: BFF reads catalogus via the ACL": architecture/adr-0025-bff-reads-catalogus-via-acl.md
|
||||
- "ADR-0026: Mutable default-fill store": architecture/adr-0026-mutable-default-fill-store.md
|
||||
- "ADR-0027: RegisterRecord objecttype": architecture/adr-0027-registerrecord-objecttype-schema.md
|
||||
- "ADR-0028: Objecten holds the register": architecture/adr-0028-objecten-holds-the-register.md
|
||||
- "ADR-0029: Objecten publishes to NRC": architecture/adr-0029-objecten-publishes-to-nrc.md
|
||||
- "ADR-0030: Projection sourced from the register": architecture/adr-0030-projection-sourced-from-the-register.md
|
||||
- "ADR-0031: MFA on the medewerker realm": architecture/adr-0031-mfa-on-the-medewerker-realm.md
|
||||
- "ADR-0032: Werkbak live refresh": architecture/adr-0032-werkbak-live-refresh.md
|
||||
- "ADR-0033: Kubernetes via one Helm chart": architecture/adr-0033-kubernetes-via-one-helm-chart.md
|
||||
- "ADR-0034: Caddy serves the portals": architecture/adr-0034-caddy-serves-the-portals.md
|
||||
- FDS-architectuur:
|
||||
- Overzicht: architecture/fds/README.md
|
||||
- Componentview (L3): architecture/fds/c4-component-view.md
|
||||
@@ -46,8 +70,15 @@ nav:
|
||||
- Working in Gitea: gitea-workflow.md
|
||||
- Frontend decisions: frontend-decisions.md
|
||||
- Demo script: demo-script.md
|
||||
- Synthetic data: synthetic-data.md
|
||||
- Runbooks:
|
||||
- CI: runbooks/ci.md
|
||||
- OpenZaak: runbooks/openzaak.md
|
||||
- Open Notificaties (NRC): runbooks/opennotificaties.md
|
||||
- Keycloak: runbooks/keycloak.md
|
||||
- Flowable: runbooks/flowable.md
|
||||
- Kubernetes on Talos: runbooks/kubernetes-talos.md
|
||||
- Gitea Actions gotchas: runbooks/gitea-actions-gotchas.md
|
||||
|
||||
markdown_extensions:
|
||||
- admonition
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
import { loginMedewerker } from './medewerker-login';
|
||||
import { loginMedewerker } from './keycloak-login';
|
||||
|
||||
// S-15a walking skeleton: a beheerder logs in to the beheer portal (medewerker realm) and sees the
|
||||
// read-only ZTC catalogus. The verify stack seeds and publishes the BIG-REGISTRATIE zaaktype (the
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
import { loginMedewerker } from './medewerker-login';
|
||||
import { loginMedewerker } from './keycloak-login';
|
||||
|
||||
// S-15b: a beheerder edits the ACL default-fill in the beheer portal and gets a saved confirmation.
|
||||
// Runs against the shared verify stack; it edits + saves (the ACL store is in-memory, ADR-0026) and
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
import { OTP_PERIOD_MS, nextUnusedCounter } from './medewerker-login';
|
||||
import { OTP_PERIOD_MS, nextUnusedCounter } from './keycloak-login';
|
||||
|
||||
// Pure check of the TOTP counter guard in loginMedewerker — no browser, no stack. Keycloak refuses
|
||||
// a code it has already accepted (its otpPolicyCodeReusable defaults to false), so two logins as
|
||||
@@ -0,0 +1,99 @@
|
||||
import { createHmac } from 'node:crypto';
|
||||
import { readFileSync, writeFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { expect, type Page } from '@playwright/test';
|
||||
|
||||
// Every portal login in the suite goes through this module — citizen realms (mock DigiD) and the
|
||||
// medewerker realm alike — so the shared Keycloak form handling lives in exactly one place.
|
||||
|
||||
// The medewerker realm enforces MFA (S-15c), so a staff login is two steps: password, then a TOTP
|
||||
// code. The realm export seeds every medewerker with this fixture secret — Keycloak HMACs the raw
|
||||
// secret bytes — so the e2e can compute a valid code instead of enrolling an authenticator.
|
||||
const OTP_SECRET = 'BIGMEDEWERKEROTPSEED';
|
||||
|
||||
export const OTP_PERIOD_MS = 30_000;
|
||||
|
||||
/**
|
||||
* How long a Keycloak form gets to appear. Generous enough for a cold first browser launch and a
|
||||
* loaded stack, far short of the 90-second test timeout an auto-waiting action would otherwise eat.
|
||||
*/
|
||||
const FORM_TIMEOUT_MS = 20_000;
|
||||
const FORM_NEVER_APPEARED =
|
||||
'the Keycloak login form never appeared — the portal did not reach Keycloak (check its ' +
|
||||
'config.json fetch and the OIDC discovery on the authority it was built with)';
|
||||
const OTP_NEVER_APPEARED =
|
||||
'the Keycloak OTP form never appeared — the password step did not complete (check the ' +
|
||||
'medewerker realm seeded this user with both a password and a TOTP credential)';
|
||||
|
||||
// RFC 6238 TOTP: HMAC-SHA1 over the 30-second counter, dynamically truncated to 6 digits.
|
||||
export function totp(secret = OTP_SECRET, at = Date.now()): string {
|
||||
const counter = Buffer.alloc(8);
|
||||
counter.writeBigUInt64BE(BigInt(Math.floor(at / OTP_PERIOD_MS)));
|
||||
const mac = createHmac('sha1', secret).update(counter).digest();
|
||||
const offset = mac[mac.length - 1] & 0x0f;
|
||||
return String((mac.readUInt32BE(offset) & 0x7fffffff) % 1_000_000).padStart(6, '0');
|
||||
}
|
||||
|
||||
// Keycloak refuses a TOTP code it has already accepted (its otpPolicyCodeReusable defaults to
|
||||
// false), so two logins as the same medewerker inside one 30-second window would both submit the
|
||||
// same code and the second is rejected. Spend the first counter this medewerker has left.
|
||||
export function nextUnusedCounter(now: number, spent: number): number {
|
||||
return Math.max(Math.floor(now / OTP_PERIOD_MS), spent + 1);
|
||||
}
|
||||
|
||||
// The spent counter lives on disk rather than in module state: Playwright starts a fresh worker
|
||||
// process for a retry, which would otherwise forget it and resubmit the rejected code.
|
||||
function spendCounter(username: string): number {
|
||||
const file = join(tmpdir(), `otp-counter-${username}`);
|
||||
let spent = -1;
|
||||
try {
|
||||
spent = Number(readFileSync(file, 'utf8')) || -1;
|
||||
} catch {
|
||||
// first login as this medewerker in this run
|
||||
}
|
||||
const counter = nextUnusedCounter(Date.now(), spent);
|
||||
writeFileSync(file, String(counter));
|
||||
return counter;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fill Keycloak's login form. Every portal is guarded, so the first navigation redirects here; the
|
||||
* form ids are stable across themes.
|
||||
*
|
||||
* The form is asserted visible *before* it is filled. A portal that never reaches Keycloak — its
|
||||
* runtime `config.json` fetch or the OIDC discovery behind `authorize()` failed, so it never
|
||||
* bootstrapped and shows a blank page (main.ts only logs to the console) — would otherwise leave
|
||||
* `fill()` auto-waiting until the whole test times out: 90 seconds spent to report
|
||||
* `locator.fill: Test timeout of 90000ms exceeded`, naming the symptom and not the cause. That is
|
||||
* how #161's catalogus.spec burned 1.8 minutes. This fails in a quarter of the time and says which
|
||||
* step never happened.
|
||||
*/
|
||||
async function submitPassword(page: Page, username: string): Promise<void> {
|
||||
await expect(page.locator('#username'), FORM_NEVER_APPEARED).toBeVisible({ timeout: FORM_TIMEOUT_MS });
|
||||
await page.locator('#username').fill(username);
|
||||
await page.locator('#password').fill('test123');
|
||||
await page.locator('#kc-login').click();
|
||||
}
|
||||
|
||||
/** A citizen login on a mock-DigiD realm — no second factor (ADR-0031). */
|
||||
export async function loginBurger(page: Page, username: string): Promise<void> {
|
||||
await submitPassword(page, username);
|
||||
}
|
||||
|
||||
/** A staff login on the medewerker realm: password, then the enforced TOTP second factor. */
|
||||
export async function loginMedewerker(page: Page, username: string): Promise<void> {
|
||||
await submitPassword(page, username);
|
||||
|
||||
// Keycloak's conditional-OTP step. Same reasoning as the password form above: assert it arrived
|
||||
// rather than letting `fill()` swallow the test timeout.
|
||||
await expect(page.locator('#otp'), OTP_NEVER_APPEARED).toBeVisible({ timeout: FORM_TIMEOUT_MS });
|
||||
|
||||
// Wait out the rest of the window if the counter we may spend is still in the future; Keycloak's
|
||||
// lookAheadWindow would accept the code a moment early, but only by one counter — waiting keeps a
|
||||
// third login in the same window valid too.
|
||||
const counter = spendCounter(username);
|
||||
await page.waitForTimeout(Math.max(0, counter * OTP_PERIOD_MS - Date.now()));
|
||||
await page.locator('#otp').fill(totp(OTP_SECRET, counter * OTP_PERIOD_MS));
|
||||
await page.locator('#kc-login').click();
|
||||
}
|
||||
@@ -1,57 +0,0 @@
|
||||
import { createHmac } from 'node:crypto';
|
||||
import { readFileSync, writeFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import type { Page } from '@playwright/test';
|
||||
|
||||
// The medewerker realm enforces MFA (S-15c), so a staff login is two steps: password, then a TOTP
|
||||
// code. The realm export seeds every medewerker with this fixture secret — Keycloak HMACs the raw
|
||||
// secret bytes — so the e2e can compute a valid code instead of enrolling an authenticator.
|
||||
const OTP_SECRET = 'BIGMEDEWERKEROTPSEED';
|
||||
|
||||
export const OTP_PERIOD_MS = 30_000;
|
||||
|
||||
// RFC 6238 TOTP: HMAC-SHA1 over the 30-second counter, dynamically truncated to 6 digits.
|
||||
export function totp(secret = OTP_SECRET, at = Date.now()): string {
|
||||
const counter = Buffer.alloc(8);
|
||||
counter.writeBigUInt64BE(BigInt(Math.floor(at / OTP_PERIOD_MS)));
|
||||
const mac = createHmac('sha1', secret).update(counter).digest();
|
||||
const offset = mac[mac.length - 1] & 0x0f;
|
||||
return String((mac.readUInt32BE(offset) & 0x7fffffff) % 1_000_000).padStart(6, '0');
|
||||
}
|
||||
|
||||
// Keycloak refuses a TOTP code it has already accepted (its otpPolicyCodeReusable defaults to
|
||||
// false), so two logins as the same medewerker inside one 30-second window would both submit the
|
||||
// same code and the second is rejected. Spend the first counter this medewerker has left.
|
||||
export function nextUnusedCounter(now: number, spent: number): number {
|
||||
return Math.max(Math.floor(now / OTP_PERIOD_MS), spent + 1);
|
||||
}
|
||||
|
||||
// The spent counter lives on disk rather than in module state: Playwright starts a fresh worker
|
||||
// process for a retry, which would otherwise forget it and resubmit the rejected code.
|
||||
function spendCounter(username: string): number {
|
||||
const file = join(tmpdir(), `otp-counter-${username}`);
|
||||
let spent = -1;
|
||||
try {
|
||||
spent = Number(readFileSync(file, 'utf8')) || -1;
|
||||
} catch {
|
||||
// first login as this medewerker in this run
|
||||
}
|
||||
const counter = nextUnusedCounter(Date.now(), spent);
|
||||
writeFileSync(file, String(counter));
|
||||
return counter;
|
||||
}
|
||||
|
||||
export async function loginMedewerker(page: Page, username: string): Promise<void> {
|
||||
await page.locator('#username').fill(username);
|
||||
await page.locator('#password').fill('test123');
|
||||
await page.locator('#kc-login').click();
|
||||
|
||||
// Keycloak's conditional-OTP step. Wait out the rest of the window if the counter we may spend is
|
||||
// still in the future; its lookAheadWindow would accept the code a moment early, but only by one
|
||||
// counter — waiting keeps a third login in the same window valid too.
|
||||
const counter = spendCounter(username);
|
||||
await page.waitForTimeout(Math.max(0, counter * OTP_PERIOD_MS - Date.now()));
|
||||
await page.locator('#otp').fill(totp(OTP_SECRET, counter * OTP_PERIOD_MS));
|
||||
await page.locator('#kc-login').click();
|
||||
}
|
||||
@@ -15,6 +15,12 @@ export default defineConfig({
|
||||
timeout: 90_000,
|
||||
expect: { timeout: 15_000 },
|
||||
retries: 1,
|
||||
// Bound the whole run, not just each test (#161). A wedged suite used to run until CI killed the
|
||||
// job — which also killed the `if: always()` steps that would have said why: the per-spec summary
|
||||
// and the container-log dump never ran, leaving a 36-minute job whose entire surviving output was
|
||||
// one ✘ line. On `globalTimeout` Playwright stops and *reports*, so the JSON report is written and
|
||||
// those steps still run. Generous over the ~1-minute suite: this is a backstop, not a budget.
|
||||
globalTimeout: 12 * 60_000,
|
||||
// Run the specs serially. Each spec drives a full `channel: 'chromium'` browser, and the e2e
|
||||
// shares an 8 GB runner with the entire compose stack (OpenZaak, NRC, Keycloak, Flowable, 4×
|
||||
// Postgres, every service + 3 portals). Two parallel browsers exhaust memory and the renderer is
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { expect, request, test } from '@playwright/test';
|
||||
import { loginMedewerker } from './medewerker-login';
|
||||
import { loginBurger, loginMedewerker } from './keycloak-login';
|
||||
|
||||
// Walking-skeleton happy path (S-08d + S-09 + S-09b + S-12 + S-10a + S-19b-2): a zorgprofessional
|
||||
// logs in via mock DigiD and submits through the self-service portal → BFF → domain; the entry
|
||||
@@ -23,9 +23,7 @@ test('DigiD submit → public INGEDIEND → documenten → behandelaar goedkeurt
|
||||
// checks submit as jan-burger (bsn 123456782) before the e2e runs on the shared stack, and
|
||||
// resume-on-load (S-26) would otherwise restore one of those on login — so each self-service spec
|
||||
// uses a dedicated citizen no other actor touches.
|
||||
await page.locator('#username').fill('emma-burger');
|
||||
await page.locator('#password').fill('test123');
|
||||
await page.locator('#kc-login').click();
|
||||
await loginBurger(page, 'emma-burger');
|
||||
|
||||
// Back on the portal, authenticated.
|
||||
await expect(page.getByRole('heading', { name: /Zelfservice/i })).toBeVisible();
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
import { loginBurger } from './keycloak-login';
|
||||
|
||||
// S-26: a zorgprofessional submits, then reloads the self-service portal. On load the portal asks the
|
||||
// BFF for the caller's current open registration (owner-scoped by the DigiD token's bsn) and restores
|
||||
@@ -9,9 +10,7 @@ test('DigiD submit → reload → self-service restores the existing registratio
|
||||
// Its own DigiD user (like every self-service spec): on the shared verify stack, resume-on-load
|
||||
// (S-26) restores any open registration for the bsn, so each spec uses a dedicated citizen that no
|
||||
// other spec or verify-* check touches. This one in particular leaves an open registration.
|
||||
await page.locator('#username').fill('sanne-burger');
|
||||
await page.locator('#password').fill('test123');
|
||||
await page.locator('#kc-login').click();
|
||||
await loginBurger(page, 'sanne-burger');
|
||||
|
||||
await expect(page.getByRole('heading', { name: /Zelfservice/i })).toBeVisible();
|
||||
await page.getByRole('button', { name: /indienen/i }).click();
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
import { loginBurger } from './keycloak-login';
|
||||
|
||||
// S-11 (Flow 3): a zorgprofessional logs in via mock DigiD, submits a registration, then withdraws
|
||||
// it ("trek aanvraag in") from the self-service portal. The withdrawal goes portal → BFF (owner-
|
||||
@@ -10,9 +11,7 @@ test('DigiD submit → trek aanvraag in → self-service confirms ingetrokken',
|
||||
|
||||
// Its own DigiD user — isolated from the verify-* checks (jan-burger/123456782) so resume-on-load
|
||||
// (S-26) can't restore someone else's registration on the shared stack.
|
||||
await page.locator('#username').fill('lars-burger');
|
||||
await page.locator('#password').fill('test123');
|
||||
await page.locator('#kc-login').click();
|
||||
await loginBurger(page, 'lars-burger');
|
||||
|
||||
await expect(page.getByRole('heading', { name: /Zelfservice/i })).toBeVisible();
|
||||
|
||||
|
||||
Reference in New Issue
Block a user