Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
86 changes: 86 additions & 0 deletions frontend/src/test/triageBand.volume.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
/**
* The frontend half of the volume oracle (issue #50, ADR-0068 D4).
*
* Feeds the Python harness's committed derived artifact
* (`tests/volume/fixtures/derived_threats.json` — regenerated via
* `uv run python scripts/regen_volume_fixtures.py`) through
* `deriveTriageActors` and asserts IDENTICAL membership and ordering to what
* the Python decision slice (`tests/volume/harness.py`) computed — closing
* the JS-side channel independently: in JavaScript `null <= 2` is `true`, so
* an unguarded frontend against a `tier: null` backend would re-create the
* triage flood by coercion even when every Python test in
* `tests/volume/test_triage_volume.py` is green (ADR-0067 D2).
*
* This file does NOT re-derive scoring — it trusts the committed fixture
* (drift-checked on the Python side by
* `TestDeterminism::test_committed_derived_threats_fixture_matches_current_generation`)
* and asserts only what `deriveTriageActors` itself is responsible for:
* the same set, in the same order, with every `tier: null` actor excluded.
*/
/// <reference types="node" />
import { readFileSync } from 'node:fs'
import { fileURLToPath } from 'node:url'
import path from 'node:path'
import { describe, it, expect } from 'vitest'
import { deriveTriageActors, isHighTierEscalation } from '../lib/triageBand'
import type { ThreatScore } from '../api/types'

const __dirname = path.dirname(fileURLToPath(import.meta.url))
const FIXTURE_PATH = path.resolve(
__dirname,
'../../../tests/volume/fixtures/derived_threats.json',
)

const threats: ThreatScore[] = JSON.parse(readFileSync(FIXTURE_PATH, 'utf-8'))

// The two planted breach-overlay actors (mirrors
// tests/volume/manifests/ambient_night.json's breach_overlay — the manifest
// is the single source of truth; these IPs are asserted against it below,
// not hand-picked).
const TIER1_ACTOR_IP = '203.0.113.129'
const BAND_HIGH_ACTOR_IP = '203.0.113.130'

describe('triageBand.volume — the ADR-0068 D4 frontend sibling', () => {
it('loads a realistic-scale fixture (>100 actors, matching the Python harness)', () => {
expect(threats.length).toBeGreaterThan(100)
})

it('derives EXACTLY the two planted actors as the triage queue', () => {
const queue = deriveTriageActors(threats, 'HIGH')
const ips = queue.map((t) => t.source_ip).sort()
expect(ips).toEqual([TIER1_ACTOR_IP, BAND_HIGH_ACTOR_IP].sort())
})

it('sorts the Tier-1 actor first', () => {
const queue = deriveTriageActors(threats, 'HIGH')
expect(queue[0].source_ip).toBe(TIER1_ACTOR_IP)
expect(queue[0].escalation?.tier).toBe(1)
})

it('never admits a tier:null actor via the coercion channel (null <= 2)', () => {
const queue = deriveTriageActors(threats, 'HIGH')
for (const actor of queue) {
expect(actor.escalation?.tier).not.toBeNull()
}
})

it('the ambient-noise mass (tier: null, sub-HIGH band) is excluded from the queue', () => {
const observedOnly = threats.filter(
(t) => t.escalation?.disposition === 'observed' && t.escalation?.tier === null,
)
expect(observedOnly.length).toBeGreaterThan(100)
for (const actor of observedOnly) {
expect(isHighTierEscalation(actor)).toBe(false)
}
const queue = deriveTriageActors(threats, 'HIGH')
const queueIps = new Set(queue.map((t) => t.source_ip))
for (const actor of observedOnly) {
expect(queueIps.has(actor.source_ip)).toBe(false)
}
})

it('flood tripwire: the derived queue stays at or under 10 actors', () => {
const queue = deriveTriageActors(threats, 'HIGH')
expect(queue.length).toBeLessThanOrEqual(10)
})
})
46 changes: 46 additions & 0 deletions scripts/regen_volume_fixtures.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
#!/usr/bin/env python3
"""Single regeneration entrypoint for the volume oracle's committed derived
fixture (ADR-0068 D3/D4, issue #50).

Regenerates ``tests/volume/fixtures/derived_threats.json`` from the ambient
night manifest (breach variant — the richer artifact the frontend sibling
test consumes, since it exercises both a queue-worthy and a null-tier actor
population) via the SAME seeded generator + real-normalizer harness the
pytest suite calls. Run this after a deliberate manifest edit (with the
distribution justification recorded in the PR — README.md's discipline
section); ``test_triage_volume.py``'s determinism test fails loudly if this
file drifts from what the generator/harness actually produce.

Usage::

uv run python scripts/regen_volume_fixtures.py
"""
from __future__ import annotations

import json
import sys
from pathlib import Path

_REPO_ROOT = Path(__file__).parent.parent
_VOLUME_DIR = _REPO_ROOT / "tests" / "volume"
sys.path.insert(0, str(_VOLUME_DIR))

import generator # noqa: E402 # pyright: ignore[reportMissingImports]
import harness # noqa: E402 # pyright: ignore[reportMissingImports]

SEED = 20260202


def main() -> None:
manifest = generator.load_manifest()
scenario = generator.build_ambient_scenario(manifest, seed=SEED, breach=True)
scores = harness.score_all(scenario.raw_events, scenario.now)

out_path = _VOLUME_DIR / "fixtures" / "derived_threats.json"
payload = [t.model_dump(mode="json") for t in scores]
out_path.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
print(f"wrote {len(payload)} actors to {out_path}")


if __name__ == "__main__":
main()
118 changes: 118 additions & 0 deletions tests/volume/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,118 @@
# `tests/volume/` — the volume oracle

**Question this oracle answers:** *does realistic input produce a usable
result?* (Sibling of `tests/golden/`, which answers *"does the same input
produce the same exact score?"* — deliberately separate disciplines, ADR-0068
D1. Neither inherits the other's change ritual.)

## What lives here

```
manifests/ambient_night.json — the reviewable manifest: personas + expected
classification, justified against Suricata's
shipped classification.config / ADR-0069's
syslog recalibration (ADR-0068 D2/D3)
generator.py — pure: (manifest, seed) -> list[RawEvent],
expanding recorded templates
(tests/golden/fixtures/eve_*.json, the
"Failed password" line shape from
packages/sources/syslog/tests) — no scoring
imports, no hand-built SecurityEvents
harness.py — RawEvents -> the REAL normalizers
(firewatch_suricata/firewatch_syslog/
firewatch_syslog_cef) -> per-actor
ThreatScore + EscalationVerdict, by mirroring
(not reimplementing) Pipeline.analyze_ip's
decision slice. No DB, no API server, no AI.
test_triage_volume.py — the invariants (ADR-0068 D2) + the ADR-0070
distribution-table personas as named,
individually-failing assertions
fixtures/derived_threats.json — committed, regenerated via
scripts/regen_volume_fixtures.py; consumed
by the frontend sibling test
```

The frontend sibling lives at `frontend/src/test/triageBand.volume.test.ts` —
it feeds the committed `derived_threats.json` through `deriveTriageActors`
(`frontend/src/lib/triageBand.ts`) and asserts the SAME membership/ordering
the Python harness computed, closing the JS-side `tier: null` <= 2 coercion
channel independently (ADR-0068 D4).

## The two scenario variants (ADR-0068 D2)

- **Ambient-only** (`generator.build_ambient_scenario(manifest, seed, breach=False)`)
— pure noise: ~127 actors built from Suricata priority-2 scanners, syslog
"Failed password" ambient scanners, a leaving 5-in-10-min sshd burst, and
the Maintainer's isolated 2-attempts/30-min INFORM case. Expected queue:
**empty** — the calm state is a machine-checked precondition, not a hope.
- **Breach variant** (`breach=True`) — the SAME ambient noise plus two
overlay actors: a Tier-1 actor (ALLOW + a corroborating detection) and a
band-HIGH accumulator (a pure BLOCK/port-scan actor crossing the HIGH band
via `run_rules` alone, independent of the tier axis). Expected queue:
**exactly those two actors**, Tier-1 sorted first — a gate that only
rewards silence fails this test.

## The ADR-0070 (+ Amendment 1) persona ledger

`test_triage_volume.py`'s `TestPersonaFiftyPerMinuteAttacker` through
`TestPersonaAmbientSuricataPriority2NoTicket` are the ledger of record for
`H` / `theta_press` / `theta_high` / `theta_quiet` / `D_endure`
(`firewatch_core.attempts`, `firewatch_core.detector`). Each persona is run
through the REAL syslog normalizer (never a hand-built `SecurityEvent`),
mirroring — not duplicating — the unit-level pins in
`packages/firewatch-core/tests/test_issue_54_attack_in_progress_campaign.py`.
A constants change that breaks a persona fails a NAMED test here with the
persona's own name in the traceback, not a silent drift.

## Manifest-change discipline (ADR-0068 D1)

Unlike `tests/golden/`'s one-time architect-signed re-bless, a manifest edit
here (adding/resizing a persona, changing a severity) requires:

1. **A stated distribution justification in the PR** — "what a real
deployment produces these numbers" (a citation to Suricata's shipped
`classification.config`, a Sigma `level` definition, an ADR, or a live
capture — see the calibration procedure below). No bless ceremony.
2. **Regenerate the derived fixture**: `uv run python
scripts/regen_volume_fixtures.py`. `test_triage_volume.py`'s
`TestDeterminism::test_committed_derived_threats_fixture_matches_current_generation`
fails loudly if you forget — the fixture is drift-checked, not
hand-maintained.
3. **Never touch `tests/golden/`** from this discipline — the two oracles
stay independent (ADR-0068 D1).

## Live-data calibration procedure (ADR-0068 D3)

Live infrastructure (a Pi running Suricata, an internet-exposed `sshd`
capture, a Terraform Azure WAF deployment) is **calibration for this
manifest, never a per-PR gate** — it must never block CI. After a real
collection night:

1. Export the actor-level persona distribution (event counts per IP, time
spans, severities) from the real capture.
2. Compare against `manifests/ambient_night.json`'s declared personas.
3. If the live distribution disagrees with a persona's declared shape (e.g.
more than the flood tripwire's worth of real ambient actors reach
`theta_press`/`theta_high`), adjust the manifest's counts/severities OR
file a `contract-change`-style finding against the relevant ADR's D5
falsifier (`ADR-0070` D5/D9) if a CONSTANT (not just the manifest) looks
miscalibrated.
4. Regenerate (`scripts/regen_volume_fixtures.py`) and state the live-data
justification in the PR.

## Adding a new volume surface (ADR-0068 D5)

This ADR builds ONE scenario (the triage surface — the one with a live,
maintainer-hit failure). Future surfaces (Network Logs at 50k events,
Analytics aggregation, Settings at N instances, the entity graph at 500
nodes) arrive **on demand** — when an ADR asserts a rarity assumption or a
walkthrough finds a scale defect — each as its own small issue, reusing
`generator.py`'s schedule helpers and `harness.py`'s normalize-dispatch
pattern rather than a new framework. Do not add a speculative scenario
without a concrete trigger (gold-plating).

## Running

No opt-in marker — this scenario runs in the default `uv run pytest`
invocation, alongside `tests/golden/`. Both variants together score in well
under a second (`TestCiBudget`, ADR-0068's <=5s budget).
11 changes: 11 additions & 0 deletions tests/volume/conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
"""Conftest for tests/volume — adds this directory to sys.path so sibling
modules (``generator``, ``harness``) are importable by bare name, mirroring
``tests/golden/conftest.py``'s convention."""
from __future__ import annotations

import sys
from pathlib import Path

_this_dir = Path(__file__).parent
if str(_this_dir) not in sys.path:
sys.path.insert(0, str(_this_dir))
Loading
Loading