From 6623c9d3d9953ca0560137794ced18e66f37be0e Mon Sep 17 00:00:00 2001 From: Simon van Lierde Date: Mon, 28 Sep 2026 13:38:31 +0200 Subject: [PATCH] feat(alerting): add p99 latency and host memory alerts, db pool panel - HighLatencyP99: p99 above 2s for 5m per (job, project, env), with a traffic floor - HostMemoryHigh: MemAvailable-based usage above 90% for 15m - Service Health: DB connection pool panel from the SQLAlchemy instrumentation --- CHANGELOG.md | 12 ++++++++ README.md | 6 ++-- config/grafana/alerting/rules.yaml | 44 ++++++++++++++++++++++++++++++ dashboards/service-health.json | 39 +++++++++++++++++++++++++- 4 files changed, 97 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a923e2a..d16471e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,18 @@ Notable changes to this stack. Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); versions follow [SemVer](https://semver.org/). +## [Unreleased] + +### Added + +- **`HighLatencyP99`**: p99 request latency above 2s for 5 minutes, per + (job, project, env), once a service handles more than 30 requests in 5 + minutes. +- **`HostMemoryHigh`**: host memory above 90% used (by `MemAvailable`) for 15 + minutes. +- Service Health shows a **DB connection pool** panel (used and idle) from the + OpenTelemetry SQLAlchemy instrumentation. + ## [0.3.1] - 2026-09-07 ### Upgrade diff --git a/README.md b/README.md index 1a36466..e0f9765 100644 --- a/README.md +++ b/README.md @@ -140,9 +140,9 @@ agent that ships container logs and host metrics. Grafana evaluates and delivers the rules in `config/grafana/alerting/`: telemetry silent per project, a project sending telemetry with no rule file, container crash-looping or OOM-killed, scrape target down, OTel export -failures, alert delivery failing, error rate above 5%, disk above 80%, disk -projected full within 3 days, Prometheus active series above 30k, and 5,000 -new series in 30 minutes. There is no Alertmanager. Grafana rules can query +failures, alert delivery failing, error rate above 5%, p99 latency above 2s, +host memory above 90%, disk above 80%, disk projected full within 3 days, +Prometheus active series above 30k, and 5,000 new series in 30 minutes. There is no Alertmanager. Grafana rules can query Loki as well as Prometheus, and one engine means one answer to "who gets told". diff --git a/config/grafana/alerting/rules.yaml b/config/grafana/alerting/rules.yaml index 4f1247c..016d8d1 100644 --- a/config/grafana/alerting/rules.yaml +++ b/config/grafana/alerting/rules.yaml @@ -171,6 +171,27 @@ groups: Extrapolated from the last 6 hours. Find what is growing (a spoke shipping more than before, a log loop, a backup that stopped rotating) before HostDiskSpaceLow makes it urgent. + # MemAvailable, not MemFree: page cache counts as free here, so this fires on + # real pressure, not on a warm cache. 15m rides out a build or a backup. + - uid: host-memory-high + title: HostMemoryHigh + !!merge <<: *rule_defaults + for: 15m + data: + - !!merge <<: *query_node + model: + !!merge <<: *query_model + expr: > + (1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) > bool 0.9 + - *firing + labels: + severity: warning + annotations: + summary: "Memory on {{ or $labels.host_name $labels.instance }} is over 90% used" + description: >- + Sustained for 15 minutes. The OOM killer picks by RSS, so the next + allocation spike can take a database or the application. Find the + consumer on the host-containers dashboard. # ~4x the baseline of ~7k series with one spoke (measured after the scrape # trimming in config/prometheus.yaml and the agent config); raise it as spokes # are onboarded. A spoke costs ~1,400 series, so this is ~18 spokes of room, @@ -284,3 +305,26 @@ groups: description: >- More than 5% of requests to {{ $labels.job }} ({{ $labels.project }}/{{ $labels.env }}) returned 5xx over the last 5 minutes. + # The traffic floor (30 requests in 5m) keeps a single slow request on a + # quiet service from being its own p99. Same (job, project, env) key as above. + - uid: high-latency-p99 + title: HighLatencyP99 + !!merge <<: *rule_defaults + for: 5m + data: + - !!merge <<: *query_node + model: + !!merge <<: *query_model + expr: > + histogram_quantile(0.99, sum by (le, job, project, env) + (rate(http_server_request_duration_seconds_bucket[5m]))) > 2 + and sum by (job, project, env) (rate(http_server_request_duration_seconds_count[5m])) > 0.1 + - *firing + labels: + severity: warning + annotations: + summary: "{{ $labels.project }}/{{ $labels.env }} {{ $labels.job }} p99 latency above 2s" + description: >- + The slowest 1% of requests to {{ $labels.job }} + ({{ $labels.project }}/{{ $labels.env }}) took over 2 seconds for 5 minutes. + The Latency p95 by route panel on Service Health shows which route. diff --git a/dashboards/service-health.json b/dashboards/service-health.json index 3dd9390..63bcd17 100644 --- a/dashboards/service-health.json +++ b/dashboards/service-health.json @@ -237,6 +237,43 @@ "overrides": [] } }, + { + "type": "timeseries", + "title": "DB connection pool", + "description": "Client-side pool state from the OpenTelemetry SQLAlchemy instrumentation. Used near the pool size means requests queue for a connection. Empty for services that do not instrument SQLAlchemy.", + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 16 + }, + "id": 6, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (state) ({__name__=~\"db_client_connections_usage(_connections)?\", project=\"$project\", env=\"$env\", job=\"$service\"})", + "legendFormat": "{{state}}", + "refId": "A" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "decimals": 0, + "custom": { + "lineWidth": 2, + "fillOpacity": 10, + "stacking": { + "mode": "normal" + } + } + }, + "overrides": [] + } + }, { "type": "logs", "title": "Logs", @@ -245,7 +282,7 @@ "h": 10, "w": 24, "x": 0, - "y": 16 + "y": 24 }, "id": 5, "datasource": {