mirror of
https://github.com/foomo/contentserver.git
synced 2026-10-06 06:56:59 +00:00
docs: add update failure alert and rule tests
This commit is contained in:
@@ -147,6 +147,11 @@ Uses Azure SDK default credential chain (environment variables, managed identity
|
||||
|
||||
See [gocloud.dev/blob](https://gocloud.dev/howto/blob/) for detailed authentication configuration.
|
||||
|
||||
## Monitoring
|
||||
|
||||
See [update-failure monitoring](docs/monitoring/README.md) for timestamp metrics,
|
||||
an example five-minute alert rule, and rule validation commands.
|
||||
|
||||
## How to Contribute
|
||||
|
||||
Contributions are welcome! Please read the [contributing guide](docs/CONTRIBUTING.md).
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
# Update-failure monitoring
|
||||
|
||||
Enable the HTTP Prometheus service with `--service-prometheus-enabled` (disabled
|
||||
by default), then configure your monitoring system to scrape each instance.
|
||||
|
||||
| Metric | Meaning |
|
||||
| --- | --- |
|
||||
| `contentserver_last_successful_update_timestamp_seconds` | Completion time of the latest successful update or unchanged-content check |
|
||||
| `contentserver_last_failed_update_timestamp_seconds` | Completion time of the latest failed update attempt |
|
||||
|
||||
Both gauges export Unix seconds with fractional precision and start at `0`.
|
||||
They belong to the process-wide metrics package and assume one repository per
|
||||
process. They add no application labels or persisted state. Prometheus attaches
|
||||
the target labels when scraping.
|
||||
|
||||
A successful load, HTTP 304, or unchanged-version response advances success.
|
||||
Fetch, parsing, and dimension-load errors advance failure. Restoring cached
|
||||
content does not establish upstream success. Busy-request rejections are not
|
||||
update attempts. History persistence errors use the separate
|
||||
`contentserver_history_persist_failed_count` counter and do not turn an otherwise
|
||||
successful update into a failure. Existing update counters remain available.
|
||||
|
||||
## Alert rule
|
||||
|
||||
Load [alerts.yml](alerts.yml) through the owning Prometheus configuration's
|
||||
`rule_files` setting. This repository supplies an example; it does not install
|
||||
the rule or configure scraping or notification routing.
|
||||
|
||||
The rule compares the last failure with the last success on each target and uses
|
||||
`for: 5m` to alert when a failure remains unrecovered for five minutes of rule
|
||||
evaluations. An initial failure can alert before any successful load. Repeated
|
||||
failures keep the condition active without restarting the delay. A successful
|
||||
retry clears the condition at the next evaluation. Scrape and evaluation
|
||||
intervals affect when changes become visible.
|
||||
|
||||
Preserve each target's labels. Aggregating success across replicas could mask a
|
||||
failing instance. The gauges detect completed update failures; stalled polling,
|
||||
missing targets, and failures to generate fresh upstream content need separate
|
||||
signals.
|
||||
|
||||
## Restarts
|
||||
|
||||
An application restart resets both gauges to `0`; snapshot restoration does not
|
||||
restore them. Prometheus clears pending or firing state when an evaluation
|
||||
observes a false condition or an absent series. If the process fails again before
|
||||
Prometheus evaluates the reset or absence, the condition can remain continuously
|
||||
active and retain its previous pending time. A restart alone does not guarantee
|
||||
a fresh five-minute delay. This describes application restarts, not restarts of
|
||||
the Prometheus server itself.
|
||||
|
||||
## Validation
|
||||
|
||||
From the repository root, with `promtool` installed:
|
||||
|
||||
```sh
|
||||
promtool check rules docs/monitoring/alerts.yml
|
||||
promtool test rules docs/monitoring/alerts.test.yml
|
||||
```
|
||||
|
||||
The fixtures cover startup zeros, delayed firing, repeated failures, recovery,
|
||||
independent replicas, and evaluated versus unobserved restart resets.
|
||||
|
||||
The metric model follows Prometheus guidance on
|
||||
[timestamps](https://prometheus.io/docs/practices/instrumentation/#timestamps-not-time-since),
|
||||
and the delay uses its
|
||||
[alerting rule semantics](https://prometheus.io/docs/prometheus/latest/configuration/alerting_rules/).
|
||||
|
||||
## Rollback / Reverse Plan
|
||||
|
||||
If validation fails, revert the instrumentation and example-rule change before
|
||||
release. No content migration or data loss is involved. If the example is later
|
||||
installed, revert the monitoring configuration change separately; rule removal
|
||||
takes effect after configuration reload and evaluation.
|
||||
@@ -0,0 +1,146 @@
|
||||
rule_files:
|
||||
- alerts.yml
|
||||
|
||||
evaluation_interval: 1m
|
||||
|
||||
tests:
|
||||
- name: startup zeros and equal timestamps stay inactive
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x5 360x5'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x5 360x5'
|
||||
alert_rule_test:
|
||||
- eval_time: 5m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 11m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
|
||||
- name: initial and repeated failures fire after five minutes only on the failing replica and clear on recovery
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0 60 60 180 180 300 300 300'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x6 420'
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-b", cluster="test"}'
|
||||
values: '0x7'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-b", cluster="test"}'
|
||||
values: '0+60x7'
|
||||
alert_rule_test:
|
||||
- eval_time: 0m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 5m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 6m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
job: contentserver
|
||||
instance: replica-a
|
||||
cluster: test
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: Contentserver update failure remains unrecovered
|
||||
description: contentserver / replica-a has not completed a successful update/check since its failure.
|
||||
- eval_time: 7m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
|
||||
- name: recovery during pending prevents firing
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0 60x9'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x2 180x6'
|
||||
alert_rule_test:
|
||||
- eval_time: 2m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 6m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 9m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
|
||||
- name: evaluated restart zeros reset pending time
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '60x2 0 240x5'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x9'
|
||||
alert_rule_test:
|
||||
- eval_time: 5m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 8m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 9m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
job: contentserver
|
||||
instance: replica-a
|
||||
cluster: test
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: Contentserver update failure remains unrecovered
|
||||
description: contentserver / replica-a has not completed a successful update/check since its failure.
|
||||
|
||||
- name: restart without an evaluated reset preserves pending time
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '60x2 180+60x3'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x6'
|
||||
alert_rule_test:
|
||||
- eval_time: 4m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 5m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
job: contentserver
|
||||
instance: replica-a
|
||||
cluster: test
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: Contentserver update failure remains unrecovered
|
||||
description: contentserver / replica-a has not completed a successful update/check since its failure.
|
||||
|
||||
- name: evaluated absent series reset pending time
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'contentserver_last_failed_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '60x2 stale 240x5'
|
||||
- series: 'contentserver_last_successful_update_timestamp_seconds{job="contentserver", instance="replica-a", cluster="test"}'
|
||||
values: '0x2 stale 0x5'
|
||||
alert_rule_test:
|
||||
- eval_time: 5m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 8m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts: []
|
||||
- eval_time: 9m
|
||||
alertname: ContentserverUpdateFailed
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
job: contentserver
|
||||
instance: replica-a
|
||||
cluster: test
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: Contentserver update failure remains unrecovered
|
||||
description: contentserver / replica-a has not completed a successful update/check since its failure.
|
||||
@@ -0,0 +1,13 @@
|
||||
groups:
|
||||
- name: contentserver
|
||||
rules:
|
||||
- alert: ContentserverUpdateFailed
|
||||
expr: |
|
||||
contentserver_last_failed_update_timestamp_seconds
|
||||
> contentserver_last_successful_update_timestamp_seconds
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Contentserver update failure remains unrecovered"
|
||||
description: "{{ $labels.job }} / {{ $labels.instance }} has not completed a successful update/check since its failure."
|
||||
Reference in New Issue
Block a user