feat(obs): golden-signal metrics on /metrics + Prometheus scrape + Grafana dashboard (S-16c, refs #124)
CI / lint (pull_request) Successful in 5m7s
CI / unit (pull_request) Successful in 2m2s
CI / frontend (pull_request) Successful in 3m58s
CI / mutation (pull_request) Successful in 6m30s
CI / verify-stack (pull_request) Successful in 9m34s
CI / build (pull_request) Successful in 4m57s

Wire OTel metrics into the four remaining .NET services (acl, domain, event-subscriber,
projection-api) exactly as the BFF: ASP.NET Core + HttpClient instrumentation + the built-in
System.Runtime meter, exposed at /metrics via the Prometheus AspNetCore exporter (ADR-0024).
Prometheus scrapes one job per service; Grafana ships a pre-built 'Request path — golden
signals' dashboard (traffic/errors/latency/saturation). A verify-metrics CI step proves the
endpoints are scraped end to end.
This commit is contained in:
not
2026-07-24 10:06:34 +02:00
parent 965782dd95
commit 4ac2c3ff6c
19 changed files with 375 additions and 10 deletions
+2 -2
View File
@@ -1,4 +1,4 @@
# Grafana with datasources baked in via provisioning (S-16a, ADR-0023).
# Dashboards (S-16c, #124) are added under provisioning/dashboards later.
# Grafana with datasources + the golden-signals dashboard baked in via provisioning
# (S-16a/S-16c, ADR-0023). Everything under provisioning/ is copied in below.
FROM grafana/grafana:11.3.0
COPY provisioning/ /etc/grafana/provisioning/
@@ -0,0 +1,13 @@
# Dashboard provider (S-16c, ADR-0023): Grafana loads every *.json in this folder as a
# read-only, code-owned dashboard. The golden-signals board is versioned here, not
# clicked together in the UI.
apiVersion: 1
providers:
- name: register-referentie
type: file
disableDeletion: true
allowUiUpdates: false
options:
path: /etc/grafana/provisioning/dashboards
foldersFromFilesStructure: false
@@ -0,0 +1,87 @@
{
"uid": "golden-signals",
"title": "Request path — golden signals",
"tags": ["s-16c", "golden-signals"],
"timezone": "browser",
"schemaVersion": 39,
"version": 1,
"editable": true,
"refresh": "10s",
"time": { "from": "now-15m", "to": "now" },
"templating": {
"list": [
{
"name": "job",
"type": "query",
"datasource": { "type": "prometheus", "uid": "prometheus" },
"query": "label_values(http_server_request_duration_seconds_count, job)",
"includeAll": true,
"multi": true,
"current": { "text": "All", "value": "$__all" },
"refresh": 2
}
]
},
"panels": [
{
"id": 1,
"title": "Traffic — requests/sec",
"type": "timeseries",
"datasource": { "type": "prometheus", "uid": "prometheus" },
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 0 },
"fieldConfig": { "defaults": { "unit": "reqps", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] },
"targets": [
{
"refId": "A",
"expr": "sum by (job) (rate(http_server_request_duration_seconds_count{job=~\"$job\"}[$__rate_interval]))",
"legendFormat": "{{job}}"
}
]
},
{
"id": 2,
"title": "Errors — 5xx responses/sec",
"type": "timeseries",
"datasource": { "type": "prometheus", "uid": "prometheus" },
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 0 },
"fieldConfig": { "defaults": { "unit": "reqps", "custom": { "drawStyle": "line", "fillOpacity": 10 }, "color": { "mode": "fixed", "fixedColor": "red" } }, "overrides": [] },
"targets": [
{
"refId": "A",
"expr": "sum by (job) (rate(http_server_request_duration_seconds_count{job=~\"$job\",http_response_status_code=~\"5..\"}[$__rate_interval]))",
"legendFormat": "{{job}}"
}
]
},
{
"id": 3,
"title": "Latency — p95 request duration",
"type": "timeseries",
"datasource": { "type": "prometheus", "uid": "prometheus" },
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 },
"fieldConfig": { "defaults": { "unit": "s", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] },
"targets": [
{
"refId": "A",
"expr": "histogram_quantile(0.95, sum by (job, le) (rate(http_server_request_duration_seconds_bucket{job=~\"$job\"}[$__rate_interval])))",
"legendFormat": "{{job}} p95"
}
]
},
{
"id": 4,
"title": "Saturation — CPU cores in use",
"type": "timeseries",
"datasource": { "type": "prometheus", "uid": "prometheus" },
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 },
"fieldConfig": { "defaults": { "unit": "none", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] },
"targets": [
{
"refId": "A",
"expr": "sum by (job) (rate(dotnet_process_cpu_time_seconds_total{job=~\"$job\"}[$__rate_interval]))",
"legendFormat": "{{job}}"
}
]
}
]
}
+20 -3
View File
@@ -1,6 +1,7 @@
# Prometheus scrape config (S-16a, ADR-0023). For the backplane slice it scrapes
# only itself; the .NET services' /metrics scrape targets are added in S-16c
# (#124) when the services expose metrics.
# Prometheus scrape config (S-16c, ADR-0023). Each .NET service exposes OTel metrics
# at /metrics (Prometheus text format); one scrape job per service, so the service is
# identified by the `job` label in the golden-signal dashboard. Targets are reached by
# compose service name on the shared `cg` network (internal port 8080).
global:
scrape_interval: 15s
@@ -8,3 +9,19 @@ scrape_configs:
- job_name: prometheus
static_configs:
- targets: ['localhost:9090']
- job_name: acl
static_configs:
- targets: ['acl:8080']
- job_name: domain
static_configs:
- targets: ['domain:8080']
- job_name: bff
static_configs:
- targets: ['bff:8080']
- job_name: event-subscriber
static_configs:
- targets: ['event-subscriber:8080']
- job_name: projection-api
static_configs:
- targets: ['projection-api:8080']