grafana-drift-alert.stub.yaml
# Grafana alert-rule STUB — measurement drift (measurement-rails)
# claim_label: built, pre-benchmark.
#
# STUB status: this file is an operator-completed provisioning template, NOT
# an active alert. It lives under docs/measurement-rails/ (not monitoring/)
# on purpose — nothing in the live monitoring tree references it until the
# operator copies it into their Grafana provisioning path and replaces
# <PROMETHEUS_DATASOURCE_UID>. The metrics it reads are served by the
# measurement-rails /metrics endpoint once the operator deploys the service
# and creates the daily drift feed (RUNBOOK §5) — neither exists until then.
#
# Semantics mirror drift_monitor.py exactly: the service itself computes the
# 3-consecutive-day streak (gap days reset it) and exposes
# mizoki_measurement_drift_alert as 0/1 — so the rule fires on the service's
# own verdict rather than re-deriving streak logic in PromQL.
apiVersion: 1
groups:
- orgId: 1
name: measurement-rails-drift
folder: MIZ OKI Measurement
interval: 5m
rules:
- uid: mizoki-measurement-drift-alert
title: "Measurement drift: platform vs house >20% for 3 consecutive days"
condition: drift_alert
for: 15m
data:
- refId: drift_alert
relativeTimeRange:
from: 3600
to: 0
datasourceUid: <PROMETHEUS_DATASOURCE_UID>
model:
expr: max by (tenant, source) (mizoki_measurement_drift_alert) >= 1
instant: true
refId: drift_alert
noDataState: NoData # missing data is a gap, never an alert
execErrState: Error
annotations:
summary: >-
Platform-reported revenue for {{ $labels.source }}
(tenant {{ $labels.tenant }}) has diverged from house-attributed
revenue by more than 20% for 3 consecutive observed days.
runbook_url: docs/measurement-rails/RUNBOOK.md
detail: >-
Current divergence gauge: mizoki_measurement_drift_ratio
{tenant="{{ $labels.tenant }}",source="{{ $labels.source }}"}.
The JSON alert artifact from drift_monitor.write_alerts carries
the three offending days. claim_label: built, pre-benchmark.
labels:
severity: warning
system: measurement_rails
adr: mr-001