From 12686b7253e1dd85fa0e2e37cfdc9fdbba4452ac Mon Sep 17 00:00:00 2001 From: Stefan Martinov Date: Tue, 8 Sep 2026 17:16:53 +0200 Subject: [PATCH] docs: add update failure alert and rule tests --- README.md | 5 ++ docs/monitoring/README.md | 73 ++++++++++++++++ docs/monitoring/alerts.test.yml | 146 ++++++++++++++++++++++++++++++++ docs/monitoring/alerts.yml | 13 +++ 4 files changed, 237 insertions(+) create mode 100644 docs/monitoring/README.md create mode 100644 docs/monitoring/alerts.test.yml create mode 100644 docs/monitoring/alerts.yml diff --git a/README.md b/README.md index ff680fb..9ba2dc7 100644 --- a/README.md +++ b/README.md @@ -147,6 +147,11 @@ Uses Azure SDK default credential chain (environment variables, managed identity See [gocloud.dev/blob](https://gocloud.dev/howto/blob/) for detailed authentication configuration. +## Monitoring + +See [update-failure monitoring](docs/monitoring/README.md) for timestamp metrics, +an example five-minute alert rule, and rule validation commands. + ## How to Contribute Contributions are welcome! Please read the [contributing guide](docs/CONTRIBUTING.md). diff --git a/docs/monitoring/README.md b/docs/monitoring/README.md new file mode 100644 index 0000000..1899bb3 --- /dev/null +++ b/docs/monitoring/README.md @@ -0,0 +1,73 @@ +# Update-failure monitoring + +Enable the HTTP Prometheus service with `--service-prometheus-enabled` (disabled +by default), then configure your monitoring system to scrape each instance. + +| Metric | Meaning | +| --- | --- | +| `contentserver_last_successful_update_timestamp_seconds` | Completion time of the latest successful update or unchanged-content check | +| `contentserver_last_failed_update_timestamp_seconds` | Completion time of the latest failed update attempt | + +Both gauges export Unix seconds with fractional precision and start at `0`. +They belong to the process-wide metrics package and assume one repository per +process. They add no application labels or persisted state. Prometheus attaches +the target labels when scraping. + +A successful load, HTTP 304, or unchanged-version response advances success. +Fetch, parsing, and dimension-load errors advance failure. Restoring cached +content does not establish upstream success. Busy-request rejections are not +update attempts. History persistence errors use the separate +`contentserver_history_persist_failed_count` counter and do not turn an otherwise +successful update into a failure. Existing update counters remain available. + +## Alert rule + +Load [alerts.yml](alerts.yml) through the owning Prometheus configuration's +`rule_files` setting. This repository supplies an example; it does not install +the rule or configure scraping or notification routing. + +The rule compares the last failure with the last success on each target and uses +`for: 5m` to alert when a failure remains unrecovered for five minutes of rule +evaluations. An initial failure can alert before any successful load. Repeated +failures keep the condition active without restarting the delay. A successful +retry clears the condition at the next evaluation. Scrape and evaluation +intervals affect when changes become visible. + +Preserve each target's labels. Aggregating success across replicas could mask a +failing instance. The gauges detect completed update failures; stalled polling, +missing targets, and failures to generate fresh upstream content need separate +signals. + +## Restarts + +An application restart resets both gauges to `0`; snapshot restoration does not +restore them. Prometheus clears pending or firing state when an evaluation +observes a false condition or an absent series. If the process fails again before +Prometheus evaluates the reset or absence, the condition can remain continuously +active and retain its previous pending time. A restart alone does not guarantee +a fresh five-minute delay. This describes application restarts, not restarts of +the Prometheus server itself. + +## Validation + +From the repository root, with `promtool` installed: + +```sh +promtool check rules docs/monitoring/alerts.yml +promtool test rules docs/monitoring/alerts.test.yml +``` + +The fixtures cover startup zeros, delayed firing, repeated failures, recovery, +independent replicas, and evaluated versus unobserved restart resets. + +The metric model follows Prometheus guidance on +[timestamps](https://prometheus.io/docs/practices/instrumentation/#timestamps-not-time-since), +and the delay uses its +[alerting rule semantics](https://prometheus.io/docs/prometheus/latest/configuration/alerting_rules/). + +## Rollback / Reverse Plan + +If validation fails, revert the instrumentation and example-rule change before +release. No content migration or data loss is involved. If the example is later +installed, revert the monitoring configuration change separately; rule removal +takes effect after configuration reload and evaluation. diff --git a/docs/monitoring/alerts.test.yml b/docs/monitoring/alerts.test.yml new file mode 100644 index 0000000..cec418c --- /dev/null +++ b/docs/monitoring/alerts.test.yml @@ -0,0 +1,146 @@ +rule_files: + - alerts.yml + +evaluation_interval: 1m + +tests: + - name: startup zeros and equal timestamps stay inactive + interval: 1m + input_series: + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x5 360x5' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x5 360x5' + alert_rule_test: + - eval_time: 5m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 11m + alertname: ContentserverUpdateFailed + exp_alerts: [] + + - name: initial and repeated failures fire after five minutes only on the failing replica and clear on recovery + interval: 1m + input_series: + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0 60 60 180 180 300 300 300' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x6 420' + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-b", cluster="test"}' + values: '0x7' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-b", cluster="test"}' + values: '0+60x7' + alert_rule_test: + - eval_time: 0m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 5m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 6m + alertname: ContentserverUpdateFailed + exp_alerts: + - exp_labels: + job: contentserver + instance: replica-a + cluster: test + severity: warning + exp_annotations: + summary: Contentserver update failure remains unrecovered + description: contentserver / replica-a has not completed a successful update/check since its failure. + - eval_time: 7m + alertname: ContentserverUpdateFailed + exp_alerts: [] + + - name: recovery during pending prevents firing + interval: 1m + input_series: + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0 60x9' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x2 180x6' + alert_rule_test: + - eval_time: 2m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 6m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 9m + alertname: ContentserverUpdateFailed + exp_alerts: [] + + - name: evaluated restart zeros reset pending time + interval: 1m + input_series: + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '60x2 0 240x5' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x9' + alert_rule_test: + - eval_time: 5m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 8m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 9m + alertname: ContentserverUpdateFailed + exp_alerts: + - exp_labels: + job: contentserver + instance: replica-a + cluster: test + severity: warning + exp_annotations: + summary: Contentserver update failure remains unrecovered + description: contentserver / replica-a has not completed a successful update/check since its failure. + + - name: restart without an evaluated reset preserves pending time + interval: 1m + input_series: + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '60x2 180+60x3' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x6' + alert_rule_test: + - eval_time: 4m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 5m + alertname: ContentserverUpdateFailed + exp_alerts: + - exp_labels: + job: contentserver + instance: replica-a + cluster: test + severity: warning + exp_annotations: + summary: Contentserver update failure remains unrecovered + description: contentserver / replica-a has not completed a successful update/check since its failure. + + - name: evaluated absent series reset pending time + interval: 1m + input_series: + - series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '60x2 stale 240x5' + - series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}' + values: '0x2 stale 0x5' + alert_rule_test: + - eval_time: 5m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 8m + alertname: ContentserverUpdateFailed + exp_alerts: [] + - eval_time: 9m + alertname: ContentserverUpdateFailed + exp_alerts: + - exp_labels: + job: contentserver + instance: replica-a + cluster: test + severity: warning + exp_annotations: + summary: Contentserver update failure remains unrecovered + description: contentserver / replica-a has not completed a successful update/check since its failure. diff --git a/docs/monitoring/alerts.yml b/docs/monitoring/alerts.yml new file mode 100644 index 0000000..9915ce2 --- /dev/null +++ b/docs/monitoring/alerts.yml @@ -0,0 +1,13 @@ +groups: + - name: contentserver + rules: + - alert: ContentserverUpdateFailed + expr: | + contentserver_last_failed_update_timestamp_seconds + > contentserver_last_successful_update_timestamp_seconds + for: 5m + labels: + severity: warning + annotations: + summary: "Contentserver update failure remains unrecovered" + description: "{{ $labels.job }} / {{ $labels.instance }} has not completed a successful update/check since its failure."