# audit-chain_test.yaml — promtool unit tests for the audit hash-chain rules.
#
# Run:  bash scripts/check-alert-rules.sh      (also run by CI)
# CI:   .github/workflows/alert-rules.yml
# Rules: deploy/alerts/audit-chain.yaml · Issues #2873, #2316 · Task #2915
#
# ── WHAT THIS FIXTURE IS FOR ────────────────────────────────────────────────
#
# #2873's acceptance names one mutation specifically:
#
#     emit 0 on the session grain and 1 on chain_scope="model_gateway" for one
#     org, and show the rule fires.
#
# That is T1 below, and it is the case every rejected predicate gets wrong in a
# DIFFERENT way, which is why it is the one worth pinning:
#
#   * `max(...) == 0`         — never fires; the healthy scope hides the break.
#   * `min(...) == 0`         — fires, but with NO labels, so the page names
#                               neither the org nor the grain.
#   * `min by (org_id) == 0`  — fires and names the org, but collapses the two
#                               grains, sending the operator to whichever
#                               service they guess.
#   * `... == 0` (this rule)  — fires once, per series, fully labelled.
#
# T3 is the same mutation the other way round (session healthy, scope broken).
# Both halves matter: a rule keyed only on the pre-#2814 label set passes T1 and
# fails T3, and that is exactly the regression #2859's fix exists to prevent.
#
# ── THE ASSERTION THAT IS NOT ABOUT FIRING ─────────────────────────────────
#
# T7 asserts a NON-event, and it is the load-bearing one. A verifier that has
# stopped running writes no status series, so AuditChainBroken is silent — and
# silence from a detection control reads as "nothing is wrong" to every human
# who sees it. T7 pins that the ruleset is NOT silent in that world: the
# absent() arm fires instead. Without that arm the whole file is satisfiable by
# a cluster where the verifier has been suspended for a year.
#
# ── HOW ASSERTIONS ARE WRITTEN ─────────────────────────────────────────────
#
# Via the `ALERTS` series and `promql_expr_test`, not `exp_alerts` — promtool
# compares annotations byte-for-byte and these rules carry multi-paragraph
# runbooks. Same choice as wf-reconcile-drift_test.yaml and memory-embed_test.yaml,
# documented in both. `alertstate="firing"` is always spelled out: ALERTS also
# carries `pending`, and a rule that only ever reaches pending has not fired.

rule_files:
  # The CRD's own .spec, freshly rendered by scripts/check-alert-rules.sh
  # immediately before this runs. NOT mirrored into dev/prometheus/rules/:
  # cmd/audit-verify is not a compose service, and arming an alert on beta-dev
  # for a job that does not run there produces pure absent-series noise, which
  # is how a real alert gets ignored.
  - ../.rendered/audit-chain.rules.yml

# 1m, matching the rule group's own interval.
evaluation_interval: 1m

tests:
  # ══════════════════════════════════════════════════════════════════════════
  # T1. #2873'S NAMED MUTATION. Session grain 0 + model_gateway grain 1, one
  #     org. Exactly one alert, on the session series, fully labelled.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "1+0x30"
      # Present and fresh, so the liveness arms stay quiet and T1 measures only
      # the thing it is about.
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      # Exactly ONE firing alert. Counting rather than asserting existence:
      # `min(...)` would also produce a firing alert here, and a presence-only
      # assertion cannot tell the two predicates apart (ADR-0034 condition 2).
      - eval_time: 10m
        expr: 'count(ALERTS{alertname="AuditChainBroken",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 1
      # …and it is the SESSION series: org named, epoch named, grain named,
      # no chain_scope label.
      - eval_time: 10m
        expr: 'ALERTS{alertname="AuditChainBroken",alertstate="firing"}'
        exp_samples:
          - labels: 'ALERTS{alertname="AuditChainBroken", alertstate="firing", component="audit", epoch="1", grain="session", org_id="org-a", severity="critical"}'
            value: 1
      # Before `for: 5m` elapses it is pending, not firing. Pins that the `for`
      # is real — a rule with `for` accidentally dropped passes every assertion
      # above and fails this one.
      - eval_time: 2m
        expr: 'count(ALERTS{alertname="AuditChainBroken",alertstate="firing"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0
      - eval_time: 2m
        expr: 'count(ALERTS{alertname="AuditChainBroken",alertstate="pending"})'
        exp_samples:
          - labels: "{}"
            value: 1

  # ══════════════════════════════════════════════════════════════════════════
  # T2. Both grains healthy for two orgs — silence. The negative control for
  #     T1: without it, `vector(1)` would pass T1.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "1+0x30"
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "1+0x30"
      - series: 'audit_chain_verifier_status{org_id="org-b",epoch="1"}'
        values: "1+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      - eval_time: 20m
        expr: 'count(ALERTS{alertname="AuditChainBroken"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T3. THE MIRROR OF T1 — session healthy, model_gateway scope broken.
  #
  #     This is the half that a rule written before #2814 misses entirely. The
  #     Model Gateway's governance-denial rows are org-grain and session-less
  #     by construction; if only the session series is watched, the tamper
  #     evidence on exactly the rows that exist to prove a refusal happened is
  #     unobserved.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "1+0x30"
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "0+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      - eval_time: 10m
        expr: 'ALERTS{alertname="AuditChainBroken",alertstate="firing"}'
        exp_samples:
          - labels: 'ALERTS{alertname="AuditChainBroken", alertstate="firing", chain_scope="model_gateway", component="audit", epoch="1", grain="scope", org_id="org-a", severity="critical"}'
            value: 1

  # ══════════════════════════════════════════════════════════════════════════
  # T4. Two orgs broken on different grains — two alerts, not one.
  #
  #     Any aggregation (`min`, `min by (org_id)`, `count`) collapses this to
  #     one sample or drops labels. Per-org-per-grain, counted, is LLD #2901
  #     §6 condition 2 applied to this task verbatim.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_status{org_id="org-b",epoch="1",chain_scope="model_gateway"}'
        values: "0+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      - eval_time: 10m
        expr: 'count(ALERTS{alertname="AuditChainBroken",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 2
      - eval_time: 10m
        expr: 'count by (org_id) (ALERTS{alertname="AuditChainBroken",alertstate="firing"})'
        exp_samples:
          - labels: '{org_id="org-a"}'
            value: 1
          - labels: '{org_id="org-b"}'
            value: 1

  # ══════════════════════════════════════════════════════════════════════════
  # T5. STALENESS. The verifier ran once and then stopped.
  #
  #     `last_run` is negative so that `time() - last_run` clears 26h inside
  #     promtool's epoch-zero timeline; the arithmetic under test is the
  #     subtraction and the threshold, not the absolute timestamp.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      # Every chain reads HEALTHY throughout. This is the trap: the status
      # gauge says "intact" and it is telling the truth about a reading that
      # is a day and a half old.
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "1+0x60"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "-100000+0x60"
    promql_expr_test:
      - eval_time: 40m
        expr: 'count(ALERTS{alertname="AuditChainVerifierStale",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 1
      # And AuditChainBroken is correctly silent — the two signals are not
      # redundant, they are orthogonal.
      - eval_time: 40m
        expr: 'count(ALERTS{alertname="AuditChainBroken"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T6. A FRESH pass does not trigger staleness. Negative control for T5.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "1+0x60"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x60"
    promql_expr_test:
      - eval_time: 40m
        expr: 'count(ALERTS{alertname="AuditChainVerifierStale"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T7. THE ONE THAT MATTERS: a verifier that never ran is not a clean bill of
  #     health.
  #
  #     No status series and no last_run series — the state this platform was
  #     ACTUALLY in when #2873 was filed, and the state it would still be in if
  #     T13 had shipped only the `== 0` rule. AuditChainBroken is silent, and
  #     must be: there is nothing to be broken. The ruleset as a whole must not
  #     be silent, and that is the assertion.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      # An unrelated series purely to give the test a timeline. Nothing in the
      # audit-chain namespace exists.
      - series: "up{job=\"placeholder\"}"
        values: "1+0x420"
    promql_expr_test:
      - eval_time: 400m
        expr: 'count(ALERTS{alertname="AuditChainBroken"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0
      - eval_time: 400m
        expr: 'count(ALERTS{alertname="AuditChainVerifierMetricsAbsent",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 1
      # Not before 6h. A cluster rebuild or a fresh Prometheus must not page.
      - eval_time: 200m
        expr: 'count(ALERTS{alertname="AuditChainVerifierMetricsAbsent",alertstate="firing"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T8. Metrics ARRIVING suppresses the absent arm. Negative control for T7 —
  #     without it, `absent()` mistyped as `vector(1)` passes T7.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x420"
    promql_expr_test:
      - eval_time: 400m
        expr: 'count(ALERTS{alertname="AuditChainVerifierMetricsAbsent"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T9. WALK ERRORS are their own state — neither verified nor broken.
  #
  #     Also pins that the predicate is a bare `> 0` and not
  #     `increase(...[26h]) > 0`. With one sample per day an `increase()` range
  #     vector routinely holds a single point, over which increase() is EMPTY:
  #     the rule would be structurally incapable of firing on the ordinary
  #     schedule while looking entirely reasonable in review.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_errors_total{org_id="org-a",stage="session"}'
        values: "1+0x60"
      - series: 'audit_chain_verifier_errors_total{org_id="org-a",stage="scope"}'
        values: "0+0x60"
      - series: 'audit_chain_verifier_status{org_id="org-b",epoch="1"}'
        values: "1+0x60"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x60"
    promql_expr_test:
      - eval_time: 40m
        expr: 'ALERTS{alertname="AuditChainVerifierWalkErrors",alertstate="firing"}'
        exp_samples:
          - labels: 'ALERTS{alertname="AuditChainVerifierWalkErrors", alertstate="firing", component="audit", org_id="org-a", severity="warning", stage="session"}'
            value: 1
      # org-a's scope stage emitted an explicit 0 — a clean walk overwrites a
      # previous run's count rather than leaving it standing. Exactly one alert,
      # not two.
      - eval_time: 40m
        expr: 'count(ALERTS{alertname="AuditChainVerifierWalkErrors",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 1

  # ══════════════════════════════════════════════════════════════════════════
  # T10..T13 — THE #2968 MIXED-VERSION DISCRIMINATOR, PROVED BOTH WAYS.
  #
  # A rollout of migration 224 breaks every chain a pre-224 binary appends to,
  # correctly, and `AuditChainBroken` used to page that as tampering. The route
  # added in #2968 is only worth having if BOTH directions hold, so both are
  # asserted and neither is inferred from the other:
  #
  #   T10  a REAL break while a rollout is in progress  => still pages critical
  #   T11  a rollout artefact on its own                => tamper page suppressed,
  #                                                        skew page fires instead
  #   T12  a clean chain                                => neither fires
  #   T13  discriminator series ABSENT                  => tamper page fires
  #
  # T11 is the case the issue was filed for. T10 is the one that decides whether
  # the fix is safe, and T13 is the one that decides whether it is safe while
  # HALF-DEPLOYED — the verifier binary is itself subject to the mixed-version
  # problem it now reports on.
  # ══════════════════════════════════════════════════════════════════════════

  # ══════════════════════════════════════════════════════════════════════════
  # T10. A REAL BREAK DURING A DEPLOY WINDOW.
  #
  #      org-a's session chain is a genuine content tamper (broken, and the
  #      discriminator says 0). org-b's session chain is the rollout artefact
  #      (broken, discriminator 1) and org-a's model_gateway scope is a SECOND
  #      artefact — so skew is loudly present in the environment, on the same
  #      org and on another one, in the same evaluation.
  #
  #      The whole fix is worthless if that context can excuse org-a's break.
  #      Assert the exact firing series, not just a count: `unless on (org_id)`
  #      would pass a count-only assertion here by removing org-a's session
  #      series because org-a's SCOPE series is skewed, which is precisely the
  #      over-broad join this test exists to forbid.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      # org-a session: GENUINE content tamper during the window.
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_unsequenced_break{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      # org-a model_gateway scope: rollout artefact, same org, other grain.
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_unsequenced_break{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "1+0x30"
      # org-b session: rollout artefact, other org.
      - series: 'audit_chain_verifier_status{org_id="org-b",epoch="1"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_unsequenced_break{org_id="org-b",epoch="1"}'
        values: "1+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      # THE LOAD-BEARING ASSERTION. Exactly one critical tamper page, and it is
      # org-a's SESSION series — labels intact, `chain_scope` absent, grain
      # "session". Three chains are broken; one of them is tampering.
      - eval_time: 10m
        expr: 'ALERTS{alertname="AuditChainBroken",alertstate="firing"}'
        exp_samples:
          - labels: 'ALERTS{alertname="AuditChainBroken", alertstate="firing", component="audit", epoch="1", grain="session", org_id="org-a", severity="critical"}'
            value: 1
      # The two artefacts route to the skew alert, one per series — the org-a
      # scope grain and org-b's session grain.
      - eval_time: 10m
        expr: 'count(ALERTS{alertname="AuditChainRolloutSkew",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 2
      # And every broken chain is covered exactly once: three status==0 series,
      # three alerts, no gap and no double-page.
      - eval_time: 10m
        expr: 'count(ALERTS{alertname=~"AuditChainBroken|AuditChainRolloutSkew",alertstate="firing"})'
        exp_samples:
          - labels: "{}"
            value: 3

  # ══════════════════════════════════════════════════════════════════════════
  # T11. THE FILED CASE — a pure mixed-version window, nothing else wrong.
  #
  #      Both grains of the one org break at chain_seq = 0. The tamper page,
  #      the one whose runbook says "escalate to the founder, a confirmed break
  #      is a disclosure question", must not fire. Something must.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_unsequenced_break{org_id="org-a",epoch="1"}'
        values: "1+0x30"
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "0+0x30"
      - series: 'audit_chain_verifier_unsequenced_break{org_id="org-a",epoch="1",chain_scope="model_gateway"}'
        values: "1+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      # SUPPRESSED. This is the assertion #2968 asked for.
      - eval_time: 10m
        expr: 'count(ALERTS{alertname="AuditChainBroken"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0
      # NOT SILENT. Two chains broke, two warnings, fully labelled — the
      # negative above is only acceptable because of this one.
      - eval_time: 10m
        expr: 'ALERTS{alertname="AuditChainRolloutSkew",alertstate="firing"}'
        exp_samples:
          - labels: 'ALERTS{alertname="AuditChainRolloutSkew", alertstate="firing", component="audit", epoch="1", grain="session", org_id="org-a", severity="warning"}'
            value: 1
          - labels: 'ALERTS{alertname="AuditChainRolloutSkew", alertstate="firing", chain_scope="model_gateway", component="audit", epoch="1", grain="scope", org_id="org-a", severity="warning"}'
            value: 1
      # The severity DOWNGRADE is the user-visible half of the fix, so pin it
      # rather than trusting the label block: a warning routes to the queue and
      # a critical wakes somebody up.
      - eval_time: 10m
        expr: 'count(ALERTS{alertstate="firing",severity="critical"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0
      # `for: 5m` on the skew arm too — the two rules must reach the on-call at
      # the same moment or a broken chain is routed away from one before it
      # arrives at the other.
      - eval_time: 2m
        expr: 'count(ALERTS{alertstate="firing"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T12. NEGATIVE CONTROL. Chains intact, discriminator explicitly 0 — the
  #      shape cmd/audit-verify emits on every clean pass. Without this,
  #      `vector(1)` as the skew predicate passes T11.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "1+0x30"
      - series: 'audit_chain_verifier_unsequenced_break{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      - eval_time: 20m
        expr: 'count(ALERTS{alertname=~"AuditChainBroken|AuditChainRolloutSkew"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0

  # ══════════════════════════════════════════════════════════════════════════
  # T13. THE DISCRIMINATOR ITSELF IS ABSENT — and the tamper page still fires.
  #
  #      The verifier binary is subject to the same mixed-version problem it
  #      now reports on: a replica predating this change emits `status` and no
  #      `unsequenced_break`. `unless` then matches nothing and removes
  #      nothing, so the rule degrades to exactly what it was before #2968.
  #
  #      This is the fail-open a route like this invites — an absent right-hand
  #      side that quietly excuses everything — and asserting it is how we know
  #      we did not build one. T11 and T13 differ ONLY in whether the
  #      discriminator series exists, and they produce opposite verdicts.
  # ══════════════════════════════════════════════════════════════════════════
  - interval: 1m
    input_series:
      - series: 'audit_chain_verifier_status{org_id="org-a",epoch="1"}'
        values: "0+0x30"
      - series: "audit_chain_verifier_last_run_timestamp_seconds"
        values: "0+0x30"
    promql_expr_test:
      - eval_time: 10m
        expr: 'ALERTS{alertname="AuditChainBroken",alertstate="firing"}'
        exp_samples:
          - labels: 'ALERTS{alertname="AuditChainBroken", alertstate="firing", component="audit", epoch="1", grain="session", org_id="org-a", severity="critical"}'
            value: 1
      - eval_time: 10m
        expr: 'count(ALERTS{alertname="AuditChainRolloutSkew"}) or vector(0)'
        exp_samples:
          - labels: "{}"
            value: 0
