CI / lint (pull_request) Successful in 5m7s
CI / unit (pull_request) Successful in 2m2s
CI / frontend (pull_request) Successful in 3m58s
CI / mutation (pull_request) Successful in 6m30s
CI / verify-stack (pull_request) Successful in 9m34s
CI / build (pull_request) Successful in 4m57s
Wire OTel metrics into the four remaining .NET services (acl, domain, event-subscriber, projection-api) exactly as the BFF: ASP.NET Core + HttpClient instrumentation + the built-in System.Runtime meter, exposed at /metrics via the Prometheus AspNetCore exporter (ADR-0024). Prometheus scrapes one job per service; Grafana ships a pre-built 'Request path — golden signals' dashboard (traffic/errors/latency/saturation). A verify-metrics CI step proves the endpoints are scraped end to end.
88 lines
3.0 KiB
JSON
88 lines
3.0 KiB
JSON
{
|
|
"uid": "golden-signals",
|
|
"title": "Request path — golden signals",
|
|
"tags": ["s-16c", "golden-signals"],
|
|
"timezone": "browser",
|
|
"schemaVersion": 39,
|
|
"version": 1,
|
|
"editable": true,
|
|
"refresh": "10s",
|
|
"time": { "from": "now-15m", "to": "now" },
|
|
"templating": {
|
|
"list": [
|
|
{
|
|
"name": "job",
|
|
"type": "query",
|
|
"datasource": { "type": "prometheus", "uid": "prometheus" },
|
|
"query": "label_values(http_server_request_duration_seconds_count, job)",
|
|
"includeAll": true,
|
|
"multi": true,
|
|
"current": { "text": "All", "value": "$__all" },
|
|
"refresh": 2
|
|
}
|
|
]
|
|
},
|
|
"panels": [
|
|
{
|
|
"id": 1,
|
|
"title": "Traffic — requests/sec",
|
|
"type": "timeseries",
|
|
"datasource": { "type": "prometheus", "uid": "prometheus" },
|
|
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 0 },
|
|
"fieldConfig": { "defaults": { "unit": "reqps", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] },
|
|
"targets": [
|
|
{
|
|
"refId": "A",
|
|
"expr": "sum by (job) (rate(http_server_request_duration_seconds_count{job=~\"$job\"}[$__rate_interval]))",
|
|
"legendFormat": "{{job}}"
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"id": 2,
|
|
"title": "Errors — 5xx responses/sec",
|
|
"type": "timeseries",
|
|
"datasource": { "type": "prometheus", "uid": "prometheus" },
|
|
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 0 },
|
|
"fieldConfig": { "defaults": { "unit": "reqps", "custom": { "drawStyle": "line", "fillOpacity": 10 }, "color": { "mode": "fixed", "fixedColor": "red" } }, "overrides": [] },
|
|
"targets": [
|
|
{
|
|
"refId": "A",
|
|
"expr": "sum by (job) (rate(http_server_request_duration_seconds_count{job=~\"$job\",http_response_status_code=~\"5..\"}[$__rate_interval]))",
|
|
"legendFormat": "{{job}}"
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"id": 3,
|
|
"title": "Latency — p95 request duration",
|
|
"type": "timeseries",
|
|
"datasource": { "type": "prometheus", "uid": "prometheus" },
|
|
"gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 },
|
|
"fieldConfig": { "defaults": { "unit": "s", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] },
|
|
"targets": [
|
|
{
|
|
"refId": "A",
|
|
"expr": "histogram_quantile(0.95, sum by (job, le) (rate(http_server_request_duration_seconds_bucket{job=~\"$job\"}[$__rate_interval])))",
|
|
"legendFormat": "{{job}} p95"
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"id": 4,
|
|
"title": "Saturation — CPU cores in use",
|
|
"type": "timeseries",
|
|
"datasource": { "type": "prometheus", "uid": "prometheus" },
|
|
"gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 },
|
|
"fieldConfig": { "defaults": { "unit": "none", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] },
|
|
"targets": [
|
|
{
|
|
"refId": "A",
|
|
"expr": "sum by (job) (rate(dotnet_process_cpu_time_seconds_total{job=~\"$job\"}[$__rate_interval]))",
|
|
"legendFormat": "{{job}}"
|
|
}
|
|
]
|
|
}
|
|
]
|
|
}
|