From 521dcca5b5c5e5c95f4338862a9d307120516662 Mon Sep 17 00:00:00 2001 From: Overcuriousity Date: Wed, 16 Sep 2026 11:40:34 +0000 Subject: [PATCH 1/5] =?UTF-8?q?feat(detectors):=20transition=20speed=20?= =?UTF-8?q?=E2=80=94=20value=20pairs=20reached=20faster=20than=20their=20l?= =?UTF-8?q?earned=20floor=20(D15)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A thirteenth statistical detector, transition_time, adapted from AMiner's MinimalTransitionTimeDetector. A transition is one step a → b between consecutive events of one stream whose series-field values differ; the stream is the source, split by a new partition_field (the identifier whose moves are timed), and rows without it are left out rather than pooled. Built on _ngram_inner_sql with n = 2, which gains an optional partition_col and now emits pkey and the arriving event's id. Two frames from the start: min-transition learns each pair's fastest baseline transition (over at least stat_transition_min_transitions of them) and flags a suspect transition that undercuts it by stat_transition_min_ratio; self-min-transition takes the pair's next-fastest transition anywhere in the scope as the leave-one-out floor. A zero floor is skipped and counted in a warning. Score = 1 − observed / reference. Gate entry, params model, /anomalies and agent knobs, persisted-run snapshot (partition_field, min_transitions), three settings with registry specs, the method card with a Stream knob, TransitionTimeFinding, a two-bar evidence figure labelled by reference_kind, and ANOMALY_DETECTION.md §15. CACHE_VERSION moves to 5 for the shared subquery's new columns. The demo case gains an administrator's routine jump-host hop (JUMP-01, then FILE-01 20–90 s later) as the floor and the contractor landing on FILE-01 two seconds after each wmic call as the signal, asserted in both frames. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 27 + CLAUDE.md | 4 +- README.md | 4 +- docs/ANOMALY_DETECTION.md | 119 +++- docs/PROGRESS.md | 60 +- docs/ROADMAP.md | 10 +- frontend/src/api/anomalies.ts | 6 +- frontend/src/api/types.ts | 56 +- .../components/analysis/FindingEvidence.tsx | 20 + .../src/components/analysis/FindingGroup.tsx | 6 +- .../components/analysis/MethodKnobForm.tsx | 2 + .../components/analysis/detector-registry.ts | 3 + .../components/analysis/method-registry.ts | 31 +- frontend/src/lib/finding-frame.ts | 7 +- frontend/src/lib/finding-normalize.ts | 7 + frontend/src/lib/finding-subject.ts | 9 + frontend/src/lib/finding-verdict.ts | 22 + frontend/src/test/methodRegistry.test.ts | 4 +- src/vestigo/agent/tools.py | 11 +- src/vestigo/api/routers/analysis.py | 21 + src/vestigo/api/routers/events.py | 78 ++- src/vestigo/core/config.py | 11 + src/vestigo/core/settings_registry.py | 18 + src/vestigo/db/analysis_cache.py | 8 +- src/vestigo/db/analysis_plan.py | 27 +- src/vestigo/db/anomaly_stats.py | 572 +++++++++++++++++- src/vestigo/demo/sources/windows.py | 61 +- tests/test_analysis_plan.py | 6 + tests/test_anomaly_stats.py | 324 ++++++++++ .../test_demo_detector_coverage_clickhouse.py | 5 + 30 files changed, 1488 insertions(+), 51 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 97114856..a18e1485 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,33 @@ All notable changes to this project are documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +### Added + +- **Transition speed (D15).** A thirteenth statistical detector, `transition_time`, adapted + from AMiner's `MinimalTransitionTimeDetector`: per ordered value pair of a series field + it learns the fastest a stream ever moved from one value to the next and reports a + transition that undercuts that floor by at least `stat_transition_min_ratio` (default + 2×) — one account on two hosts seconds apart, a session skipping states. Transitions are + one step of the sequence detectors' n-gram assembly, per source and per value of a new + `partition_field` knob (the identifier whose moves are timed; rows without it are left + out rather than pooled), so two users' interleaved logons never read as one actor. Both + frames from day one: with a baseline the floor is the baseline window's fastest + transition of the pair (`min-transition`); without one it is the pair's next-fastest + transition anywhere on the timeline (`self-min-transition`, leave-one-out by + construction). A floor is learned from at least `stat_transition_min_transitions` (3) + transitions, a zero floor is skipped and counted in a warning, the per-source candidate + cap `stat_transition_max_candidates` (2000) keeps the fastest pairs and discloses itself, + and score is `1 − observed / reference`. The finding carries the pair, the stream that + made the move, both durations, which floor was used and the speed-up; the allowlist key + is `(series_field, "a → b")` in both frames. Gate, wizard card, evidence figure, agent + tool, persisted-run snapshot (`partition_field`, `min_transitions`) and + `docs/ANOMALY_DETECTION.md` §15 ship with it; the analysis cache moves to version 5. The + demo case gains an administrator's routine jump-host hop as the floor and the + contractor's wmic call landing on `FILE-01` two seconds later as the signal, asserted in + both frames. + ## [1.19.7] — 2026-09-15 ### Added diff --git a/CLAUDE.md b/CLAUDE.md index ef989a91..86cd93a2 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -26,7 +26,7 @@ trail (`api/routers/auth.py`, `admin.py`, `deps.py`). - `CONCEPT.md` / `MODEL_REFINEMENT.md` — product vision and the Case/Source/Timeline/Event/ Artifact data model. Read before touching the model; rarely changes. - `TECH_STACK.md` — backing-service decision record (*why*, not *what's shipped*). -- `ANOMALY_DETECTION.md` — reference for all fourteen analysis tools actually running +- `ANOMALY_DETECTION.md` — reference for all fifteen analysis tools actually running (statistical detectors, Sigma runner, log templates, semantic similarity), plus the baseline/disposition model. Update alongside any detector change in the same commit. - `AGENT.md` — the optional AI investigation agent (design invariants, MCP tools, provider @@ -253,7 +253,7 @@ instead of rebuilding. does not strand its verdict bar a screen below the claim; `ToolsSheet` is four tabs (Scope, Methods, Signatures, Explore) rather than one scroll, so a thousand-row template list cannot bury the baseline picker; `method-registry.ts` is the - single description of all twelve methods, including the prose that used to live in a + single description of all thirteen methods, including the prose that used to live in a Method tab and each method's optional `railFloor`, a presentation-only bar on the ranked feed whose held-back count is always disclosed; the sheet's method mode runs a method with the analyst's own knob values, which is what keeps the analysis diff --git a/README.md b/README.md index 6de7c371..58f1400d 100644 --- a/README.md +++ b/README.md @@ -69,7 +69,7 @@ resolved on the machine you deploy to. ClickHouse, time histogram with anomaly overlays, keyset pagination with jump-to-time, tag/comment annotations with bulk apply, saved views, and streaming CSV/JSONL export that keeps the forensic columns. -- **Anomaly detection** — fourteen analysis tools: twelve statistical detectors over +- **Anomaly detection** — fifteen analysis tools: thirteen statistical detectors over ClickHouse needing no embeddings, a Sigma rule runner, and semantic similarity search over local embeddings. Each is documented method by method, scores against explicit baseline-vs-suspect windows, and yields findings whose confirm/dismiss disposition @@ -109,7 +109,7 @@ feel like, and the Case/Timeline model here is descended from it. That is the co invite, and three axes are where we think we are already the better place to run an investigation: -- **Detection is the workflow, not an add-on** — fourteen analysis tools in the box, each +- **Detection is the workflow, not an add-on** — fifteen analysis tools in the box, each scoring against an analyst-declared baseline and carrying a verdict that survives re-scans, so triage accumulates instead of being redone. - **Provenance goes all the way down** — not just "this file was imported": a finding is diff --git a/docs/ANOMALY_DETECTION.md b/docs/ANOMALY_DETECTION.md index ef29c5a1..460016e0 100644 --- a/docs/ANOMALY_DETECTION.md +++ b/docs/ANOMALY_DETECTION.md @@ -7,7 +7,7 @@ This document covers every detector actually running in the codebase today. If a detector described here changes (formula, default, field name), update this file and the "Method" tab copy in the same commit. -There are fourteen independent analysis tools in Vestigo: +There are fifteen independent analysis tools in Vestigo: 1. [Value novelty](#1-value-novelty-rare--first-seen-values) — rare/new field values, single field or [combinations](#value-combinations-the-value_combo-variant) (ClickHouse, no ML) 2. [Frequency anomalies](#2-frequency-anomalies-volume-spikes--silences) — volume spikes/silences (ClickHouse, no ML) @@ -23,13 +23,14 @@ There are fourteen independent analysis tools in Vestigo: 12. [Repeating sequences](#12-repeating-sequences-motif-mining) — recurring time-ordered n-grams of a field's values, ranked by support and cadence regularity; the discovery/mining complement of detector 9 (ClickHouse, no ML) 13. [Sigma rule runner](#13-sigma-rule-runner-signature-matching) — signature matching: community/custom Sigma rules compiled to ClickHouse predicates (ClickHouse, no ML) 14. [Log templates](#14-log-templates-structural-line-clustering) — structural clustering of raw lines into templates, so rare *shapes* surface without naming a field (ClickHouse, no ML) +15. [Transition speed](#15-transition-speed-value-to-value-moves-faster-than-ever-seen) — a stream reaching the next value of a field faster than that pair was ever reached before (ClickHouse, no ML) All but the eleventh are **statistical or rule-based**: pure counting, arithmetic and predicate matching over already-ingested events — no machine learning, no network calls, working the instant ingestion finishes. The eleventh needs an explicit embedding step first. -Code: `src/vestigo/db/anomaly_stats.py` (detectors 1–10, 12, 14), +Code: `src/vestigo/db/anomaly_stats.py` (detectors 1–10, 12, 14, 15), `src/vestigo/db/similarity.py` (detector 11), `src/vestigo/sigma/` (detector 13). UI: `frontend/src/components/analysis/`. @@ -61,6 +62,7 @@ teaches the tooling instead of the work. What each tool has to find: | 12 | Repeating sequences | The beacon motif (proxy request + matching firewall allow in the same second), alongside a benign nightly-backup motif. | | 13 | Sigma | Four case-scoped rules ship inside the case: encoded PowerShell, suspicious service install, wmic remote process creation, failed-logon burst. | | 14 | Log templates | ~40 syslog templates, with an `unattended-upgrade` shape that appears only in the suspect window. | +| 15 | Transition speed | The contractor account on `FILE-01` two seconds after its wmic call from `JUMP-01`, timed per account over `attr:computer_name`, against an administrator whose routine jump-host hop takes the same pair 20–90 s. | `tests/test_demo_detector_coverage_clickhouse.py` asserts that each of these actually returns findings. If a retuned threshold silences one of them, that @@ -357,9 +359,9 @@ checked, and the UI must never present the two the same way. | `charset` | ≥1 field above the enum-like ceiling | `analysis_gate_max_enum_distinct` | | `frequency` | span of at least the minimum number of seconds | `analysis_gate_min_frequency_buckets` | | `interval_periodicity` | enough events per series value to fit a cadence | `analysis_gate_min_interval_periods` | -| `sequence_novelty` | a series field with ≥2 distinct values | `analysis_gate_min_series_distinct` | +| `sequence_novelty`, `transition_time` | a series field with ≥2 distinct values (one value yields no ordering and no transition) | `analysis_gate_min_series_distinct` | | `proportion_shift`, `value_distribution_drift` | a span of more than one instant (self frame: it is cut into slices) | `stat_self_slices` | -| any of those four in the **baseline frame** | an active baseline definition | — (reported as `needs_setup`) | +| any of those five in the **baseline frame** | an active baseline definition | — (reported as `needs_setup`) | Three rows encode a distinction worth stating, because each was once drawn wrong: @@ -2538,6 +2540,112 @@ entry if that need grows past a browser flag. --- +## 15. Transition speed (value-to-value moves faster than ever seen) + +**What it answers:** "Did something get from *here* to *there* faster than it ever +has?" The AMiner `MinimalTransitionTimeDetector` analog (roadmap D15). One account +logging on to two hosts three seconds apart, a session skipping from `created` to +`closed` with nothing in between, a workflow reaching its last state in a fraction of +its usual time — each value is ordinary, each ordering may even be ordinary, but the +*speed* is not. Event sequences (§9) own the ordering axis; this detector owns the +time between two consecutive values. + +**How it works.** A **transition** is one step `a → b` between two consecutive events +of one *stream* whose `series_field` values differ (same-value repeats are not +transitions). The stream is the source, further split by the **partition field** when +one is set — the identifier whose movements are being timed, `attr:user` for host +moves, a session id for state moves. Without it every source is a single stream, and +two users' interleaved logons read as one actor moving between their hosts, which is +rarely the question. Transitions are one step of the sequence detectors' n-gram +assembly (`_ngram_inner_sql`, n = 2), so they inherit its deterministic ordering +(effective timestamp, then record order), the once-per-source scan discipline, and the +guarantee that a pair never spans a window boundary. The duration is +`dateDiff('millisecond', previous, current)`. Per ordered pair the detector learns a +**floor** — the fastest that pair was ever reached in the reference — and flags a +transition that undercuts it by at least the speed-up factor (`min_ratio`, default 2): +a floor learned from a handful of transitions is not precise to the second, so +"slightly faster" is not a finding. + +**Two frames.** + +| | Self (`self-min-transition`) | Baseline (`min-transition`) | +|---|---|---| +| Reference | the pair's **next-fastest** transition anywhere in the scope | the pair's fastest transition in the baseline window | +| Learned from | at least `stat_transition_min_transitions` transitions of the pair in the scope (default 3) | at least that many in the baseline window | +| Flags | the pair's single fastest transition, when it undercuts the next-fastest by `min_ratio`× | each suspect window's fastest transition of the pair, when it undercuts the baseline floor by `min_ratio`× | +| Zero floor | two instantaneous transitions: skipped, counted in a warning | a baseline floor of zero: skipped, counted in a warning | + +The self frame is leave-one-out by construction: every other transition of the pair is +at least as slow as the second-fastest, so the fastest is judged against everything +else that pair ever did. Two equally fast transitions vouch for each other and nothing +is flagged. A zero floor cannot be undercut — with second-resolution timestamps, +zero-length transitions are routine — so such pairs are skipped rather than scored, and +the run says how many. + +Per source: the `stat_transition_max_candidates` fastest pairs (baseline frame: per +suspect window) are fetched fastest-first, with a warning when the cap is hit — the cap +keeps the fastest, which are the ones the question is about; in the baseline frame the +floor is then learned for exactly those candidate pairs. On a multi-source scope the +floor is the minimum over every source's reference and counts are summed, so a pair +that is slow in one source and fast in another is judged against the fast one. + +**Score = 1 − observed / reference**, in `[0, 1]`: 1.0 is an instantaneous transition, +0.5 is exactly twice as fast as the floor, and nothing under `1 − 1/min_ratio` is +reported. The representative event is the **arriving** event of the fastest transition +(the `b` side), and `first_seen` is its timestamp. Findings carry `observed_seconds`, +`reference_seconds`, `reference_kind` (`next-fastest` / `baseline-min`), `speedup` +(reference ÷ observed; `null` when the observation is instant), `count` (the pair's +transitions in the window, or in the scope), `baseline_count` (the transitions the +floor was learned from) and, when a partition field was set, `partition_field` / +`partition_value` — the stream that made the move, which is how the analyst finds the +actor. The baseline frame adds `window_label`/`window_start`/`window_end`; the self +frame adds `scope_transitions` and no window keys. + +**Parameters.** + +- `series_field` (request, default `artifact`) — the field whose consecutive values + form transitions. Shared with the frequency and sequence detectors' group-by; any + field token works, including `attr:` and mapped canonical fields. As with event + sequences, a source whose every row carries one artifact value has no transitions + under the default — pick the field that moves (`attr:computer_name`, a state field). +- `partition_field` (request, default unset) — the stream key. Rows without a value + for it are left out rather than lumped into one anonymous stream. Snapshotted into + the persisted `DetectorRun` as `partition_field`. +- `min_ratio` (request) / `VESTIGO_STAT_TRANSITION_MIN_RATIO` (server default 2.0, + must exceed 1) — the speed-up floor. Snapshotted as `min_ratio`. +- `VESTIGO_STAT_TRANSITION_MIN_TRANSITIONS` (default 3) — the learning floor. + Snapshotted as `min_transitions`. +- `VESTIGO_STAT_TRANSITION_MAX_CANDIDATES` (default 2000) — the per-source candidate + cap; hitting it attaches a warning. + +**Allowlist key:** `(series_field, "a → b")` — the same shape as event sequences, in +both frames, so one **Normal** verdict on a pair covers it wherever it recurs. + +### Caveats + +- **The floor is only as good as the reference.** A baseline in which a pair occurred + three times has a floor that is the fastest of three draws; the learning floor keeps + one-off pairs out, and `min_ratio` keeps "a bit faster" out, but on a thin baseline + the flagged speed-ups are as much about the baseline's poverty as about the suspect + window. Prefer longer baselines, or the self frame, which learns from everything. +- **Interleaved streams manufacture speed.** Without a partition field, a multi-writer + source (one Windows log carrying every account) times the gap between *any* two + consecutive events with different values, whoever produced them — and two accounts + on two hosts in the same second reads as an impossible move. Set the partition field + to the identifier whose moves are meant, as the demo case does with `attr:user`. +- **Timestamp resolution bounds the question.** Second-resolution sources produce + zero-length transitions routinely; a pair whose floor is zero is skipped with a + warning rather than scored, and a floor of one second is compared against + observations that are themselves rounded. Millisecond sources make the detector far + sharper. +- **Only the fastest transition per pair is reported** (per window in the baseline + frame). `count` says how many transitions of the pair the window held; it does not + say how many undercut the floor. The Explorer drill on the pair shows all of them. +- A fast transition is **not malicious by itself** — a scripted deployment legitimately + touches ten hosts in ten seconds. Rank for triage; mark the routine pair Normal once. + +--- + ## Dispositions and normality (implementation notes) Analyst verdicts on findings live in one `finding_dispositions` table with the @@ -2567,7 +2675,8 @@ Every successful scan (`GET .../anomalies` with the default `persist=true`, and always for `tag_anomalies`) writes a `DetectorRun` row: the request params it ran with — fields, `series_field`, thresholds, `baseline_id`, resolved windows, `windows_hash`, `dispositions_hash`, the per-source clock-skew offsets in -effect, entropy's `variant`, and for a self-frame run of the slice methods the +effect, entropy's `variant`, transition speed's `partition_field` and +`min_transitions`, and for a self-frame run of the slice methods the `slices` payload, its `slices_hash` and the resolved self settings (`self_slices`, `pause_ratio`, `min_span_seconds`, `sequence_rarity_floor`) — plus the serialized result, and returns its id as `run_id`. Rows diff --git a/docs/PROGRESS.md b/docs/PROGRESS.md index f61ba5a9..63c0dfca 100644 --- a/docs/PROGRESS.md +++ b/docs/PROGRESS.md @@ -4,8 +4,64 @@ Append-only session log — what changed and why, newest first. This file keeps sessions only; older ones live in git history, and every release is summarized in `CHANGELOG.md`. Plans belong in `ROADMAP.md`, not here. -Last updated: 2026-09-15 (v1.19.7; session 238 — review findings on the D18/D19/D11 -branch: the self frame's complement, the per-slice scan budget, three disclosure gaps). +Last updated: 2026-09-16 (1.20 in progress; session 239 — the transition-speed detector, +D15, first of the 1.20 detector cluster). + +## Session 239 — 2026-09-16: transition speed (D15), the first 1.20 detector + +The 1.20 cluster is the three cheap AMiner analogs left on the roadmap — D15, D12, D13 — +landed one commit each on `feat/1.20`, in that order because each reuses machinery the +previous fortnight touched. This session is D15, `transition_time`. + +**What it is.** A transition is one step `a → b` between consecutive events of one stream +whose series-field values differ; the detector learns each pair's fastest transition and +reports one that undercuts it by `min_ratio`×. The stream is the source, split by a new +`partition_field` (the identifier whose moves are timed), which is the knob that makes +the question meaningful: without it a Windows log carrying every account times the gap +between *any* two consecutive logons, and two users on two hosts in the same second reads +as an impossible move. Rows without a partition value are left out rather than pooled +into one anonymous stream, for the same reason. + +**Built on `_ngram_inner_sql`, not beside it.** A transition is an n-gram of length two +with `gram[1] != gram[2]`, so the assembly, the per-source scan discipline, the +record-order tie-breaks and the window-boundary guarantee are the sequence detectors'. +The helper gained an optional `partition_col` (added to every `PARTITION BY`, `None` +keeps the sequence detectors' SQL shape) and now emits `pkey` and the arriving event's +`last_eid` — the representative event is the `b` side, the one that arrived too soon. +Adding columns to a shared subquery is what moved `CACHE_VERSION` to 5, not the new +method id, which could not collide with an old key on its own. + +**Both frames from day one (D18 is a rule now, not a migration).** Baseline: +`min-transition`, the floor is the pair's fastest baseline-window transition over at +least `min_transitions` of them, learned only for the candidate pairs the suspect scan +surfaced (Query B is bound to Query A's grams). Self: `self-min-transition`, the floor is +the pair's *next-fastest* transition anywhere in the scope — `groupArraySorted(2)` per +source, the two smallest merged across sources — which is leave-one-out by construction, +and two equally fast transitions vouch for each other. A zero floor is skipped in both +frames and counted in a warning: with second-resolution timestamps zero-length +transitions are routine, and "faster than instant" is not a claim. The self mode was +first named `loo-min-transition`; the demo coverage test's `self-`/`rare-` prefix +convention is the right one, so it is `self-min-transition` like its siblings. + +**The demo had no such signal.** The baseline frame found nothing on the demo case: the +contractor's lateral moves are minutes apart while every account's random alternation +between its own home hosts produces baseline floors of seconds. Per this file's own rule +the fabricated signal was strengthened rather than the assertion: one administrator now +has a routine jump-host hop (JUMP-01 by RDP, FILE-01 by network logon 20–90 s later, +twice a working day, kept tight so their own workstation logons rarely fall between), and +each wmic remote process creation during lateral movement now logs the contractor on to +FILE-01 two seconds later — which from JUMP-01 is the administrator's pair at a tenth of +the time. Both frames assert on it. + +**Surface.** Gate entry (same `series_distinct ≥ 2` floor as sequences: one value has no +transition; `needs_setup` in the baseline frame without a baseline), `_TransitionTimeParams` +(`series_field`, `partition_field`, `min_ratio`), the `/anomalies` and tag endpoints and +the agent tool gain `partition_field`, the persisted run snapshots `partition_field` and +`min_transitions`, three settings with registry specs. Frontend: the thirteenth method +card (`Gauge` icon, a "Stream" field knob), `TransitionTimeFinding`, a two-bar evidence +figure labelled by `reference_kind`, verdict/normalize/subject cases, and the mode sets in +`finding-frame.ts`. Docs: `ANOMALY_DETECTION.md` §15, the tool list, demo table and gate +table; README and CLAUDE.md counts. ## Session 238 — 2026-09-15: review of the D18/D19/D11 branch (PR #377) diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index a9efd763..928153e0 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -12,7 +12,7 @@ from that day. **Milestone 10 — AI agent log investigation — is the 2.0 thru everything below**; the numbered list orders the remaining 1.x work by payoff-per-effort: 1. **A12** local transform tools — no design round, no OPSEC gate. -2. **D12** / **D13** / **D15** — cheap detectors reusing existing SQL machinery. +2. **D12** / **D13** — cheap detectors reusing existing SQL machinery (D15 shipped in 1.20). 3. **W8** query-time field extraction — makes bespoke unstructured logs first-class. 4. **A8** external MCP toolsets — needs its own design round (policy, not plumbing). 5. **D10** / **D16** — heaviest lifts, last of the detector line. @@ -141,7 +141,7 @@ burns its numbers out of that file**; the migration is done when the file is `{} Detectors adapted from [ait-aecid/logdata-anomaly-miner](https://github.com/ait-aecid/logdata-anomaly-miner), constrained to be **field-agnostic** and SQL-explainable per the forensic-reproducibility -requirement. D1–D9, `proportion_shift` and `sequence_motif` shipped — `ANOMALY_DETECTION.md` +requirement. D1–D9, D15, `proportion_shift` and `sequence_motif` shipped — `ANOMALY_DETECTION.md` is each detector's contract, updated in the same commit as any detector change. Every item below is incomplete until the frontend half lands with it: a plain-language @@ -160,12 +160,6 @@ A detector whose reasoning an analyst cannot read does not count as shipped. D10. Reuses `GROUP BY a, b` plus the G-test and Benjamini–Hochberg pool that `proportion_shift` has. Field-pair explosion is the design problem: needs a preselection rule and a candidate cap in the `HEAVY_SCAN_SETTINGS` family, honestly reported. -- [ ] **D15 — Impossible-speed transitions** (`MinimalTransitionTimeDetector`): learn the - minimum observed time between consecutive values of a field per identifier, flag a - suspect-window transition faster than the baseline ever saw. `find_sequence_novelty`'s - `lagInFrame` partitions already produce the pairs; this is a `min(dateDiff)` over the - same shape. Score = `1 − (observed / learned_min)`. - **High effort, high value:** - [ ] **D10 — Event correlation rules** (`EventCorrelationDetector`): mine baseline diff --git a/frontend/src/api/anomalies.ts b/frontend/src/api/anomalies.ts index f4b5a42b..e153b062 100644 --- a/frontend/src/api/anomalies.ts +++ b/frontend/src/api/anomalies.ts @@ -21,7 +21,7 @@ export interface LogTemplatesParams { } export interface AnomalyParams { - detector?: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift"; + detector?: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time"; /** Comma-separated field tokens for value_novelty, e.g. "artifact,display_name,attr:user_agent" */ fields?: string; /** Field to group frequency series / build event sequences by */ @@ -46,6 +46,8 @@ export interface AnomalyParams { group_field?: string; /** sequence_novelty / sequence_motif only: break an n-gram when consecutive events are more than this many seconds apart. Omit for no gap bound. */ max_gap_seconds?: number; + /** transition_time only: the stream whose transitions are timed (e.g. "attr:user"). Omit for one stream per source. */ + partition_field?: string; /** ID of a saved baseline definition (baseline range + suspect windows). Omit for self-baseline. */ baseline_id?: string; limit?: number; @@ -102,7 +104,7 @@ export const anomaliesApi = { sourceId: string, eventId: string, body: { - detector: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift"; + detector: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time"; content: string; details: Record; /** diff --git a/frontend/src/api/types.ts b/frontend/src/api/types.ts index f32edd20..4681f6df 100644 --- a/frontend/src/api/types.ts +++ b/frontend/src/api/types.ts @@ -786,6 +786,56 @@ export interface SequenceMotifFinding { confirmed_other_scope?: boolean; } +/** + * One value-to-value transition faster than its learned floor, from the + * transition_time detector (D15). `details.method` is `min-transition` + * (the floor is the baseline window's fastest transition of the pair, and + * `details` carries `window_*` keys) or `self-min-transition` (the floor is + * the pair's next-fastest transition anywhere in the timeline, with + * `scope_transitions` and no window keys). `reference_kind` names which. + */ +export interface TransitionTimeFinding { + type: "transition_time"; + /** Field token whose consecutive values form the transition (e.g. "attr:computer_name"). */ + field: string; + /** [from, to]. */ + values: string[]; + /** "from → to" — display form and the allowlist key. */ + value: string; + /** The stream the transition was timed within (e.g. "attr:user"); null = per source. */ + partition_field: string | null; + /** That stream key's value on the flagged transition; null when unpartitioned. */ + partition_value: string | null; + /** The fastest transition of this pair in the window (or the timeline), in seconds. */ + observed_seconds: number; + /** The floor it undercut, in seconds. */ + reference_seconds: number; + reference_kind: "baseline-min" | "next-fastest"; + /** reference_seconds ÷ observed_seconds; null when the observation is instant. */ + speedup: number | null; + /** Transitions of this pair in the window (baseline frame) or the timeline (self). */ + count: number; + /** Transitions the floor was learned from. */ + baseline_count: number; + /** 1 − observed ÷ reference; 1.0 = instantaneous. */ + score: number; + /** Timestamp of the arriving event of the fastest transition. */ + first_seen: string | null; + event_id: string | null; + event: Event | null; + details: Record; + /** Present (true) only when the request passed `include_dismissed`. */ + dismissed?: boolean; + /** Present (true) when a confirmed disposition covers this finding's event. */ + confirmed?: boolean; + /** + * Present (true) when the only confirmed verdict on this event was reached + * under a *different* comparison. The claim stands, but not for this scope — + * so the row is marked rather than badged, and Confirm stays live. + */ + confirmed_other_scope?: boolean; +} + export type AnomalyFinding = | ValueNoveltyFinding | ValueComboFinding @@ -798,7 +848,8 @@ export type AnomalyFinding = | IntervalPeriodicityFinding | SequenceNoveltyFinding | SequenceMotifFinding - | DistributionDriftFinding; + | DistributionDriftFinding + | TransitionTimeFinding; export interface AnomaliesResponse { status: "ok" | "no_data" | "insufficient_data"; @@ -1004,7 +1055,8 @@ export interface AnomalyMarker { | "proportion_shift" | "interval_periodicity" | "sequence_novelty" - | "value_distribution_drift"; + | "value_distribution_drift" + | "transition_time"; /** Raw structured finding data — stored verbatim on the persisted annotation. */ rawDetails: Record; /** End of the anomalous window, for frequency findings — enables a range highlight. */ diff --git a/frontend/src/components/analysis/FindingEvidence.tsx b/frontend/src/components/analysis/FindingEvidence.tsx index 434ca45e..620e4aa4 100644 --- a/frontend/src/components/analysis/FindingEvidence.tsx +++ b/frontend/src/components/analysis/FindingEvidence.tsx @@ -326,6 +326,26 @@ export function FindingEvidence({ finding }: { finding: MethodResult }) { case "sequence_novelty": case "sequence_motif": return ; + case "transition_time": + // The claim is one duration against one floor, both measured; which + // floor is in the label, since the two answer different questions. + return ( + + ); case "timestamp_order": return (
diff --git a/frontend/src/components/analysis/FindingGroup.tsx b/frontend/src/components/analysis/FindingGroup.tsx index 534d7161..bea565b0 100644 --- a/frontend/src/components/analysis/FindingGroup.tsx +++ b/frontend/src/components/analysis/FindingGroup.tsx @@ -50,9 +50,9 @@ interface Props { * Rows this group holds that no sweep method produces — today, Sigma hits in * the Named-techniques group. * - * A slot rather than a thirteenth entry in `METHODS`: that registry is pinned - * by tests to exactly the twelve ids `db/analysis_plan.py` plans for and the - * twelve param sets `api/routers/analysis.py` accepts, and Sigma is neither + * A slot rather than a fourteenth entry in `METHODS`: that registry is pinned + * by tests to exactly the thirteen ids `db/analysis_plan.py` plans for and the + * thirteen param sets `api/routers/analysis.py` accepts, and Sigma is neither * planned nor run through the findings endpoint. */ extraRows?: React.ReactNode; diff --git a/frontend/src/components/analysis/MethodKnobForm.tsx b/frontend/src/components/analysis/MethodKnobForm.tsx index 92a5c541..f892993f 100644 --- a/frontend/src/components/analysis/MethodKnobForm.tsx +++ b/frontend/src/components/analysis/MethodKnobForm.tsx @@ -129,6 +129,8 @@ export function knobHelp(knob: MethodKnob): string { return "How many consecutive events form one sequence. Three is a good default."; case "max_gap_seconds": return "Break a sequence when consecutive events are farther apart than this."; + case "partition_field": + return "Whose moves are timed: transitions are measured within one value of this field, such as one account. Without it every source is a single stream."; case "field": return "The text field to cluster into templates. Usually the message."; case "order": diff --git a/frontend/src/components/analysis/detector-registry.ts b/frontend/src/components/analysis/detector-registry.ts index c69d412a..3aec234a 100644 --- a/frontend/src/components/analysis/detector-registry.ts +++ b/frontend/src/components/analysis/detector-registry.ts @@ -11,6 +11,7 @@ */ import { Activity, + Gauge, Hash, Layers, ListOrdered, @@ -32,6 +33,7 @@ export type DetectorId = | "interval" | "drift" | "sequence" + | "transition" | "order" | "range" | "charset" @@ -68,6 +70,7 @@ export const DETECTORS: DetectorMeta[] = [ { id: "drift", detector: "value_distribution_drift", icon: Replace, label: "Distribution drift", hint: "Whole-field value-mix changes between windows", category: "volume", scoreUnit: "−log₁₀ p" }, { id: "order", detector: "timestamp_order", icon: Rewind, label: "Timestamp order", hint: "Timestamps running backwards", category: "volume", scoreUnit: "s skew" }, { id: "sequence", detector: "sequence_novelty", icon: ListOrdered, label: "Event sequences", hint: "Never-seen or rare event orderings (n-grams)", category: "sequences", scoreUnit: "surprise" }, + { id: "transition", detector: "transition_time", icon: Gauge, label: "Transition speed", hint: "Value-to-value moves faster than ever seen", category: "sequences", scoreUnit: "1 − obs/ref" }, ]; export const DETECTORS_BY_ID = Object.fromEntries(DETECTORS.map((d) => [d.id, d])) as Record< diff --git a/frontend/src/components/analysis/method-registry.ts b/frontend/src/components/analysis/method-registry.ts index e8ffa36f..f82f7d9a 100644 --- a/frontend/src/components/analysis/method-registry.ts +++ b/frontend/src/components/analysis/method-registry.ts @@ -22,6 +22,7 @@ import { Activity, FileText, + Gauge, Hash, Layers, ListOrdered, @@ -46,6 +47,7 @@ export type MethodId = | "interval_periodicity" | "timestamp_order" | "sequence_novelty" + | "transition_time" | "log_template"; export type EvidenceClass = "named" | "statistical" | "exploration"; @@ -110,7 +112,7 @@ export interface MethodMeta { hint: string; /** * When to configure it, in one sentence for the wizard's card. Starts with - * "Use this when" — a test enforces it — so the twelve cards read as one list. + * "Use this when" — a test enforces it — so the thirteen cards read as one list. */ useWhen: string; icon: React.ElementType; @@ -384,6 +386,33 @@ export const METHODS: MethodMeta[] = [ { param: "max_gap_seconds", label: "Max gap", kind: "number", placeholder: "300" }, ], }, + { + id: "transition_time", + label: "Transition speed", + hint: "Value-to-value moves faster than ever seen", + useWhen: + "Use this when one actor should not be able to reach the next value that fast — an account on two hosts seconds apart, a session skipping states. Pick the stream field.", + icon: Gauge, + evidenceClass: "statistical", + costClass: "heavy", + scoreUnit: "1 − obs/ref", + what: "Times each step between consecutive different values of the series field within one stream (per source, and per value of the stream field when set), learns the fastest each ordered pair was ever reached, and reports a transition that undercuts that floor by the speed-up factor. With a baseline the floor is the baseline window's fastest transition of the pair; without one it is the pair's next-fastest transition anywhere on the timeline, so a single outlier is judged against everything else that pair ever did.", + querySketch: `SELECT [prev, val] AS pair, min(dur) AS fastest FROM (\n SELECT AS val,\n lagInFrame(val) OVER w AS prev,\n dateDiff('millisecond', lagInFrame() OVER w, ) AS dur\n FROM events WHERE case_id = {case}\n WINDOW w AS (PARTITION BY source_id, ORDER BY )\n) WHERE prev != val\nGROUP BY pair\n-- baseline: reported when fastest < baseline min(dur) / {min_ratio}\n-- self: reported when fastest < the pair's next-fastest dur / {min_ratio}`, + knobs: [ + SERIES_KNOB, + { + param: "partition_field", + label: "Stream", + kind: "field", + placeholder: "(per source)", + // Which identifier's moves are timed. Without it every source is one + // stream, and two users' interleaved logons read as one actor moving. + fieldOptions: SERIES_FIELD_OPTIONS, + noneLabel: "Per source", + }, + RATIO_KNOB, + ], + }, { id: "log_template", label: "Log templates", diff --git a/frontend/src/lib/finding-frame.ts b/frontend/src/lib/finding-frame.ts index c7098f19..6a72e09c 100644 --- a/frontend/src/lib/finding-frame.ts +++ b/frontend/src/lib/finding-frame.ts @@ -23,14 +23,19 @@ export const TEMPORAL_MODES: ReadonlySet = new Set([ "cadence", "ngram", "drift", + "min-transition", ]); -/** The self modes of the four formerly baseline-only methods. */ +/** + * The self modes of the four formerly baseline-only methods, plus the + * transition detector's, which was born with both frames. + */ export const SELF_MODES: ReadonlySet = new Set([ "self-g-test", "self-drift", "self-cadence", "rare-ngram", + "self-min-transition", ]); /** The two self modes that compare leave-one-out time slices with the rest of the scope. */ diff --git a/frontend/src/lib/finding-normalize.ts b/frontend/src/lib/finding-normalize.ts index a5ccd6fd..088835ae 100644 --- a/frontend/src/lib/finding-normalize.ts +++ b/frontend/src/lib/finding-normalize.ts @@ -134,6 +134,13 @@ export function normalizeFinding(meta: DetectorMeta, f: AnomalyFinding, rank: nu subtitle = `×${f.support}${f.period_seconds !== null ? ` every ~${f.period_seconds}s` : ""}`; ts = ts ?? f.first_seen; break; + case "transition_time": + title = `${fieldLabel(f.field)}: ${truncate(f.value, 70)}`; + subtitle = `${f.observed_seconds}s against a floor of ${f.reference_seconds}s (${ + f.reference_kind === "next-fastest" ? "next-fastest on the timeline" : "baseline minimum" + })${f.partition_value ? ` · ${truncate(f.partition_value, 30)}` : ""}`; + ts = ts ?? f.first_seen; + break; } return { detectorId: meta.id, diff --git a/frontend/src/lib/finding-subject.ts b/frontend/src/lib/finding-subject.ts index 33fe59fa..0c3625a9 100644 --- a/frontend/src/lib/finding-subject.ts +++ b/frontend/src/lib/finding-subject.ts @@ -61,6 +61,15 @@ function scoredSubject(f: AnomalyFinding): SubjectPair[] { // The claim is about the whole value mix, so naming any one value would // misstate what was compared. return [{ label: "field", value: fieldLabel(f.field) }]; + case "transition_time": + // The pair is the subject; the stream that made the move is how the + // analyst finds the actor, so it is offered when the run had one. + return f.partition_field && f.partition_value + ? [ + { label: fieldLabel(f.field), value: String(f.value) }, + { label: fieldLabel(f.partition_field), value: f.partition_value }, + ] + : [{ label: fieldLabel(f.field), value: String(f.value) }]; } } diff --git a/frontend/src/lib/finding-verdict.ts b/frontend/src/lib/finding-verdict.ts index 645c52f5..c00b0878 100644 --- a/frontend/src/lib/finding-verdict.ts +++ b/frontend/src/lib/finding-verdict.ts @@ -165,9 +165,31 @@ function scoredVerdict(f: AnomalyFinding): Verdict { highlight: `${f.support} times`, tail: `across ${f.sources_count} source${f.sources_count === 1 ? "" : "s"} — a routine pattern, not a finding.`, }; + case "transition_time": { + const stream = f.partition_value + ? `${fieldLabel(f.partition_field ?? "")} = ${truncate(f.partition_value, 40)} moved ` + : "A stream moved "; + const floor = + f.reference_kind === "next-fastest" + ? `its next-fastest such move anywhere on the timeline took ${fmtSeconds(f.reference_seconds)} (${f.count} transitions)` + : `the baseline never saw it under ${fmtSeconds(f.reference_seconds)} across ${f.baseline_count} transitions`; + return { + lead: `${stream}${truncate(f.values[0] ?? "", 30)} → ${truncate(f.values[1] ?? "", 30)} in`, + highlight: fmtSeconds(f.observed_seconds), + tail: `— ${floor}${f.speedup === null ? "" : `, ${f.speedup.toFixed(1)}× faster`}.`, + }; + } } } +/** Seconds as a short human duration; every figure comes from the finding. */ +function fmtSeconds(s: number): string { + if (s < 60) return `${s % 1 === 0 ? s.toFixed(0) : s.toFixed(1)} s`; + if (s < 3600) return `${(s / 60).toFixed(1)} min`; + if (s < 86400) return `${(s / 3600).toFixed(1)} h`; + return `${(s / 86400).toFixed(1)} d`; +} + export function findingVerdict(finding: MethodResult): Verdict { if (isTemplateRow(finding)) { return { diff --git a/frontend/src/test/methodRegistry.test.ts b/frontend/src/test/methodRegistry.test.ts index 51329ef3..2c68ea55 100644 --- a/frontend/src/test/methodRegistry.test.ts +++ b/frontend/src/test/methodRegistry.test.ts @@ -47,6 +47,7 @@ describe("method registry", () => { interval_periodicity: ["series_field", "fdr_q", "min_ratio"], timestamp_order: ["min_skew_seconds"], sequence_novelty: ["series_field", "ngram_size", "max_gap_seconds"], + transition_time: ["series_field", "partition_field", "min_ratio"], log_template: ["field", "order", "only_new"], }; for (const m of METHODS) { @@ -56,7 +57,7 @@ describe("method registry", () => { } }); - it("covers exactly the twelve methods the gate plans for", () => { + it("covers exactly the thirteen methods the gate plans for", () => { // METHOD_IDS in db/analysis_plan.py. A method here that the plan never // reports would render with no status; one there that is missing here // would never be shown at all. @@ -71,6 +72,7 @@ describe("method registry", () => { "proportion_shift", "sequence_novelty", "timestamp_order", + "transition_time", "value_combo", "value_distribution_drift", "value_novelty", diff --git a/src/vestigo/agent/tools.py b/src/vestigo/agent/tools.py index 63fa6dc8..d28a6733 100644 --- a/src/vestigo/agent/tools.py +++ b/src/vestigo/agent/tools.py @@ -1967,15 +1967,20 @@ async def run_anomaly_detector( group_field: str | None = None, max_gap_seconds: int | None = Field(default=None, ge=1), variant: Literal["shannon", "bigram"] | None = None, + partition_field: str | None = None, ) -> dict[str, Any]: """Run a statistical anomaly detector over the timeline. Detectors: value_novelty (rare/first-seen values), value_combo, frequency (volume spikes/silences), timestamp_order, numeric_range, charset, entropy, proportion_shift, interval_periodicity, - sequence_novelty, sequence_motif, value_distribution_drift. + sequence_novelty, sequence_motif, value_distribution_drift, + transition_time (a value pair reached faster than the pair's learned + floor, e.g. one account on two hosts seconds apart). `fields` is a comma-separated field list for value detectors (omit to - auto-recommend); `series_field` groups frequency/sequence detectors. + auto-recommend); `series_field` groups frequency/sequence/transition + detectors; `partition_field` (transition_time) is the stream whose + transitions are timed, e.g. attr:user. Every detector runs without a `baseline_id` (the timeline is its own reference); pass one from list_baselines to score suspect windows against a baseline instead. Optional knobs (server defaults @@ -1995,6 +2000,7 @@ async def run_anomaly_detector( """ _reject_time_fields(fields, "fields") _reject_time_fields(series_field, "series_field") + _reject_time_fields(partition_field, "partition_field") result, resolution = await _run_stat_detector( scope.case_id, scope.timeline_id, @@ -2015,6 +2021,7 @@ async def run_anomaly_detector( group_field=group_field, max_gap_seconds=max_gap_seconds, variant=variant, + partition_field=partition_field, field_mappings=scope.field_mappings, source_offsets=scope.source_offsets, ) diff --git a/src/vestigo/api/routers/analysis.py b/src/vestigo/api/routers/analysis.py index c6e277c3..2f3a2e21 100644 --- a/src/vestigo/api/routers/analysis.py +++ b/src/vestigo/api/routers/analysis.py @@ -357,6 +357,25 @@ class _SequenceNoveltyParams(_Params): max_gap_seconds: int | None = Field(default=None, ge=1) +class _TransitionTimeParams(_Params): + series_field: str = DEFAULT_SERIES_FIELD + #: The stream whose transitions are timed (a user, a session); None = + #: one stream per source. + partition_field: str | None = None + min_ratio: float | None = Field(default=None, gt=1) + + @field_validator("partition_field", mode="before") + @classmethod + def _empty_is_none(cls, v: Any) -> Any: + """A cleared field select and an omitted knob ask the same question. + + The form spells "per source" as ``""``; the runner spells it ``None``. + Normalizing here keeps the two from fingerprinting as different cache + keys for one answer. + """ + return None if v == "" else v + + class _LogTemplateParams(_Params): #: Not a `_run_stat_detector` detector — log templating is a browser with #: its own service call (see :func:`_run_log_templates`). Routing it through @@ -378,6 +397,7 @@ class _LogTemplateParams(_Params): "interval_periodicity": _IntervalPeriodicityParams, "timestamp_order": _TimestampOrderParams, "sequence_novelty": _SequenceNoveltyParams, + "transition_time": _TransitionTimeParams, "log_template": _LogTemplateParams, } @@ -708,6 +728,7 @@ async def get_analysis_findings( group_field=kwargs.get("group_field"), max_gap_seconds=kwargs.get("max_gap_seconds"), variant=kwargs.get("variant"), + partition_field=kwargs.get("partition_field"), # Both come from _resolve_timeline_scope and are not optional # niceties: without field_mappings a canonical field alias is # ignored, and without source_offsets a declared per-source diff --git a/src/vestigo/api/routers/events.py b/src/vestigo/api/routers/events.py index 74617795..2dde06d7 100644 --- a/src/vestigo/api/routers/events.py +++ b/src/vestigo/api/routers/events.py @@ -49,6 +49,7 @@ ShiftFinding, StatisticalAnomalyService, TimeWindow, + TransitionFinding, ValueFinding, _canonical_hash, ) @@ -2252,6 +2253,7 @@ async def _run_stat_detector( group_field: str | None = None, max_gap_seconds: int | None = None, variant: str | None = None, + partition_field: str | None = None, field_mappings: dict[str, list[str]] | None = None, source_offsets: dict[str, int] | None = None, field_overrides: dict[str, bool] | None = _RESOLVE_OVERRIDES, @@ -2397,6 +2399,37 @@ async def _run_stat_detector( except ValueError as exc: raise HTTPException(status_code=422, detail=str(exc)) from exc + if detector == "transition_time": + # Transitions are steps between consecutive values of the single + # series_field, timed per source and per partition_field stream (D15). + # Snapshot the effective floors so the persisted run stays + # self-describing after a default changes. + resolution["transition_min_ratio"] = ( + min_ratio if min_ratio is not None else cfg.stat_transition_min_ratio + ) + resolution["transition_min_transitions"] = cfg.stat_transition_min_transitions + resolution["transition_partition_field"] = partition_field or None + try: + result = await run_scan( + svc.find_transition_times, + case_id=case_id, + source_ids=source_ids, + source_offsets=source_offsets, + series_field=series_field, + partition_field=partition_field or None, + limit=limit, + windows=windows, + min_ratio=resolution["transition_min_ratio"], + min_transitions=cfg.stat_transition_min_transitions, + max_candidates=cfg.stat_transition_max_candidates, + exclude_event_ids=exclude_ids, + allowlist=allowlist, + field_mappings=field_mappings, + ) + return result, resolution + except ValueError as exc: + raise HTTPException(status_code=422, detail=str(exc)) from exc + if detector == "sequence_motif": # Mode-less mining over the single series_field: recurring n-grams, # no baseline needed — `windows` is deliberately ignored. Optional @@ -2763,13 +2796,16 @@ def _serialize_finding( | IntervalFinding | SequenceFinding | MotifFinding - | DistributionDriftFinding, + | DistributionDriftFinding + | TransitionFinding, ) -> dict[str, Any]: - """Serialise a Value/Freq/Order/Combo/Range/Charset/Entropy/Shift/Interval/Sequence/Motif/Drift finding to a JSON-safe dict.""" - # Charset/Entropy/Shift/Interval/Sequence/Motif/Drift finding dataclass - # fields are exactly the wire keys, so asdict() avoids a hand-maintained - # field-by-field transcription that would silently drop any newly added - # field. + """Serialise a Value/Freq/Order/Combo/Range/Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition finding to a JSON-safe dict.""" + # Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition finding + # dataclass fields are exactly the wire keys, so asdict() avoids a + # hand-maintained field-by-field transcription that would silently drop + # any newly added field. + if isinstance(r, TransitionFinding): + return {"type": "transition_time", **asdict(r)} if isinstance(r, DistributionDriftFinding): return {"type": "value_distribution_drift", **asdict(r)} if isinstance(r, MotifFinding): @@ -3150,7 +3186,12 @@ async def _persist_detector_run( or resolution.get("interval_fdr_q") or resolution.get("drift_fdr_q"), "min_ratio": resolution.get("shift_min_ratio") - or resolution.get("interval_min_rate_ratio"), + or resolution.get("interval_min_rate_ratio") + or resolution.get("transition_min_ratio"), + # transition_time: the stream key transitions were timed within + # (None = per source) and the learning floor the run used (D15). + "partition_field": resolution.get("transition_partition_field"), + "min_transitions": resolution.get("transition_min_transitions"), # sequence_novelty: effective (request-or-default) n-gram length. "ngram_size": resolution.get("sequence_ngram"), # charset: per-identifier scoping (None = one alphabet per field). @@ -3225,7 +3266,7 @@ async def list_anomalies( timeline_id: str, detector: str = Query( default="value_novelty", - description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', or 'value_distribution_drift'.", + description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', or 'transition_time'.", ), fields: str | None = Query( default=None, @@ -3315,6 +3356,14 @@ async def list_anomalies( "the reference values; catches ordinary characters in an unusual order)." ), ), + partition_field: str | None = Query( + default=None, + description=( + "transition_time only: the stream whose transitions are timed (e.g. " + "'attr:user' to time each account's moves between hosts). Omit for " + "one stream per source." + ), + ), start: datetime | None = Query( default=None, description="sequence_motif only: scope mining to events at/after this time (ISO, UTC).", @@ -3392,6 +3441,11 @@ async def list_anomalies( spacing test, beaconing). Without: whole-scope Greenwood beaconing with pauses excluded, and a robust-Gamma silence test. BH-FDR across the run. + **transition_time**: per ordered value pair of `series_field`, learns the + fastest a stream (per source, per `partition_field` value) ever moved from + one value to the next and flags transitions faster than that floor by + `min_ratio`x — the impossible-speed question (D15). + **sequence_novelty**: per source, builds time-ordered n-grams of `series_field` values and flags orderings that occur in a suspect window but never in the baseline window (AMiner EventSequenceDetector analog), @@ -3433,6 +3487,7 @@ async def list_anomalies( group_field=group_field, max_gap_seconds=max_gap_seconds, variant=variant, + partition_field=partition_field, field_mappings=field_mappings, source_offsets=source_offsets, ) @@ -3565,7 +3620,7 @@ class TagAnomaliesRequest(BaseModel): detector: str = Field( default="value_novelty", - description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', or 'value_distribution_drift'.", + description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', or 'transition_time'.", ) fields: str | None = Field( default=None, @@ -3623,6 +3678,10 @@ class TagAnomaliesRequest(BaseModel): default=None, description="entropy only: 'shannon' (default) or 'bigram' — the statistic the band is learned over.", ) + partition_field: str | None = Field( + default=None, + description="transition_time only: the stream whose transitions are timed (e.g. 'attr:user').", + ) start: datetime | None = Field( default=None, description="sequence_motif only: scope mining to events at/after this time.", @@ -3690,6 +3749,7 @@ async def tag_anomalies( group_field=body.group_field, max_gap_seconds=body.max_gap_seconds, variant=body.variant, + partition_field=body.partition_field, field_mappings=field_mappings, source_offsets=source_offsets, ) diff --git a/src/vestigo/core/config.py b/src/vestigo/core/config.py index 19d0e0db..0322d70a 100644 --- a/src/vestigo/core/config.py +++ b/src/vestigo/core/config.py @@ -185,6 +185,17 @@ class Settings(BaseSettings): # occurrences, not value occurrences — the charset floor is separate for # the same reason. stat_sequence_rarity_floor: int = Field(default=3, ge=1) + # Transition-time detector (D15): a suspect transition must undercut the + # pair's learned floor by at least this factor to be reported — a floor + # learned from a handful of transitions is not precise to the second. + stat_transition_min_ratio: float = Field(default=2.0, gt=1) + # A pair's floor is learned from at least this many transitions (baseline + # window, or the whole scope in the self frame); fewer and the pair is + # skipped rather than scored against one or two observations. + stat_transition_min_transitions: int = Field(default=3, ge=2) + # Cap on candidate pairs fetched per source, fastest first; hitting it + # carries a warning. + stat_transition_max_candidates: int = 2000 # ── Self frame for the slice-based detectors (D18) ────────────────────── # How many equal-width time slices proportion_shift and # value_distribution_drift cut the scope into when no baseline is declared; diff --git a/src/vestigo/core/settings_registry.py b/src/vestigo/core/settings_registry.py index 0929ee6d..26002d2c 100644 --- a/src/vestigo/core/settings_registry.py +++ b/src/vestigo/core/settings_registry.py @@ -536,6 +536,24 @@ class SettingGroup: "Sequence rarity floor (self frame)", "Without a baseline, an n-gram occurring at most this many times in the scope is a rare ordering.", ), + SettingSpec( + "stat_transition_min_ratio", + "detectors", + "Transition speed-up floor", + "A transition is reported only when it is at least this many times faster than the pair's learned floor.", + ), + SettingSpec( + "stat_transition_min_transitions", + "detectors", + "Transition learning floor", + "A value pair's fastest transition is learned from at least this many transitions; fewer and the pair is skipped.", + ), + SettingSpec( + "stat_transition_max_candidates", + "detectors", + "Transition candidate cap", + "Candidate value pairs fetched per source, fastest first.", + ), SettingSpec( "stat_self_slices", "detectors", diff --git a/src/vestigo/db/analysis_cache.py b/src/vestigo/db/analysis_cache.py index 211b6546..0099309e 100644 --- a/src/vestigo/db/analysis_cache.py +++ b/src/vestigo/db/analysis_cache.py @@ -57,7 +57,13 @@ #: spends a bounded scan budget. Both are the runner answering differently #: for the same inputs: a v3 row holds "up" findings for every short-lived #: value on the timeline, which is the shape this bump exists to retire. -CACHE_VERSION = 4 +#: +#: 5 — the 1.20 detectors (transition_time, D15). New method ids cannot +#: collide with an older row's key on their own, but the shared n-gram +#: assembly now emits two more columns, and a key that names the runner's +#: inputs while the runner's SQL changed underneath it is the case this +#: constant exists for. +CACHE_VERSION = 5 def fingerprint( diff --git a/src/vestigo/db/analysis_plan.py b/src/vestigo/db/analysis_plan.py index 853a1c00..900c68a4 100644 --- a/src/vestigo/db/analysis_plan.py +++ b/src/vestigo/db/analysis_plan.py @@ -48,13 +48,15 @@ "interval_periodicity", "timestamp_order", "sequence_novelty", + "transition_time", "log_template", ) #: The methods that select fields for themselves, and so the only ones a -#: timeline's ``field_overrides`` can steer. The other four take no field -#: selection to steer: ``frequency`` and ``sequence_novelty`` take a single -#: ``series_field`` the analyst names outright, ``timestamp_order`` reads no +#: timeline's ``field_overrides`` can steer. The other five take no field +#: selection to steer: ``frequency``, ``sequence_novelty`` and +#: ``transition_time`` take a single ``series_field`` the analyst names +#: outright, ``timestamp_order`` reads no #: field at all, and ``log_template`` clusters the message text. A declaration #: stored against one of those would be audited, rendered as "declared" and #: then quietly apply to nothing — which is the same lie an unknown method id @@ -88,6 +90,7 @@ "value_distribution_drift": "heavy", "interval_periodicity": "heavy", "sequence_novelty": "heavy", + "transition_time": "heavy", "log_template": "heavy", } @@ -340,6 +343,7 @@ def build_plan(inputs: PlanInputs, cfg: Settings) -> list[MethodPlan]: "value_distribution_drift", "interval_periodicity", "sequence_novelty", + "transition_time", ): if frame_needs_baseline: plans[method] = _setup( @@ -397,6 +401,23 @@ def build_plan(inputs: PlanInputs, cfg: Settings) -> list[MethodPlan]: ), ) + # A transition is a step between two *different* values of the series + # field, so the same floor applies: one distinct value yields no transition + # at all. Two values yield two ordered pairs, each with a floor to learn. + plans.setdefault( + "transition_time", + _ok("transition_time") + if inputs.series_distinct >= cfg.analysis_gate_min_series_distinct + else _no( + "transition_time", + "the series field holds one value, so no event moves between two", + { + "series_distinct": inputs.series_distinct, + "required": cfg.analysis_gate_min_series_distinct, + }, + ), + ) + # Log templating clusters the `message` materialized column, which is part # of the events schema and therefore always present. There is no data shape # that makes it structurally unable to produce a template, so gating it diff --git a/src/vestigo/db/anomaly_stats.py b/src/vestigo/db/anomaly_stats.py index f24a0660..90e6beea 100644 --- a/src/vestigo/db/anomaly_stats.py +++ b/src/vestigo/db/anomaly_stats.py @@ -182,6 +182,25 @@ window occurrence of the most-shifted category (categorical). Score = ``-log10(p)`` so the two statistics rank on one scale. +**transition_time** (``detector="transition_time"``) + Per ordered value pair ``a → b`` of a series field, learn the fastest a + stream ever moved from ``a`` to ``b`` and flag a transition faster than + that floor by at least ``min_ratio``×. Adapted from AMiner's + ``MinimalTransitionTimeDetector`` (roadmap D15): the "impossible speed" + question — one account on two hosts three seconds apart when the + baseline never saw it under forty minutes. Transitions are one step of + the sequence detectors' n-gram assembly (:func:`_ngram_inner_sql`, + ``n = 2``), per source and optionally per *partition field* (the + identifier whose stream is timed — a user, a session), so consecutive + events of different identifiers never form one transition; same-value + repeats are not transitions. Two frames: *baseline* + (``method="min-transition"``) learns each pair's minimum over the baseline + window (at least ``min_transitions`` of them, floor > 0) and scores each + suspect window's fastest transition of the pair against it; *self* + (``method="self-min-transition"``, D18) takes the pair's **next-fastest** + transition anywhere in the scope as the leave-one-out floor. Score = + ``1 − observed / reference``. + **timestamp_order** (``detector="timestamp_order"``) Flag events whose parsed timestamp jumps *backwards* relative to the previous record in the source file (record order = ``byte_offset``, then @@ -1240,6 +1259,44 @@ class SequenceFinding: details: dict[str, Any] +@dataclass +class TransitionFinding: + """One value-to-value transition faster than its learned floor (transition_time, D15).""" + + # Field token whose consecutive values form the transition (e.g. "attr:computer_name"). + field: str + # [from, to]. + values: list[str] + # "from → to" — display form and the allowlist key. + value: str + # The stream key the transition was timed within (e.g. "attr:user"); None = per source. + partition_field: str | None + # That key's value on the flagged transition; None when unpartitioned. + partition_value: str | None + # The fastest transition of this pair in the window (baseline frame) or + # the scope (self frame), in seconds. + observed_seconds: float + # The floor it undercut: the baseline's fastest (baseline frame) or the + # pair's next-fastest transition anywhere in the scope (self frame). + reference_seconds: float + reference_kind: str # "baseline-min" | "next-fastest" + # reference_seconds / observed_seconds; None when the observation is instant. + speedup: float | None + # Transitions of this pair in the window (baseline frame) or the scope (self). + count: int + # Transitions the floor was learned from: the baseline's (baseline frame) + # or, in the self frame, the same scope count as `count`. + baseline_count: int + # 1 − observed / reference; 1.0 = instantaneous. + score: float + # Timestamp of the arriving event of the fastest transition. + first_seen: str | None + # The arriving event. + event_id: str | None + event: dict[str, Any] | None + details: dict[str, Any] + + @dataclass class MotifFinding: """One recurring event-order n-gram surfaced by the sequence-motif miner.""" @@ -1282,10 +1339,12 @@ class StatAnomalyResult: # "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" # | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" # | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" + # | "transition_time" detector: str # "self-baseline" | "temporal" | "z-score" | "temporal-z-score" | "sequential" # | "iqr" | "temporal-range" | "rare-chars" | "temporal-charset" | "temporal-iqr" - # | "g-test" | "cadence" | "ngram" | "motif" | "drift" + # | "g-test" | "cadence" | "ngram" | "motif" | "drift" | "min-transition" + # | "self-min-transition" method: str baseline_size: int # total events (value_novelty) or event-count used for z-score results: list[ @@ -1301,6 +1360,7 @@ class StatAnomalyResult: | SequenceFinding | MotifFinding | DistributionDriftFinding + | TransitionFinding ] = field(default_factory=list) # Effective |z| cutoff used by the frequency detector; None for value_novelty. z_threshold: float | None = None @@ -1746,6 +1806,7 @@ def _ngram_inner_sql( w_idx_expr: str, scope_pred: str, max_gap: int | None = None, + partition_col: str | None = None, ) -> str: """Two-level n-gram assembly subquery shared by the sequence detectors. @@ -1765,6 +1826,13 @@ def _ngram_inner_sql( form. ``None`` keeps the pre-D14 shape bit-identical. The value is an int inlined as a literal, same convention as the ``ngram - 1`` frame. + *partition_col* (D15) names a second stream key — an identifier such as + the user whose host transitions are being timed. Rows are additionally + partitioned by it (``pkey``), so consecutive events of *different* + identifiers never form one n-gram, and rows without a value for it are + left out rather than lumped into one anonymous stream. ``None`` keeps the + per-source partition and emits an empty ``pkey``. + Callers must run the enclosing query **once per source** (it bounds the window sort — see the query-cost discipline in docs/ANOMALY_DETECTION.md) and filter on ``guard IS NOT NULL`` to keep only complete n-grams. @@ -1772,10 +1840,13 @@ def _ngram_inner_sql( gram_lags = ", ".join( [f"lagInFrame(val, {ngram - 1 - j}) OVER w" for j in range(ngram - 1)] + ["val"] ) + pkey_expr = f"toString({partition_col})" if partition_col is not None else "''" + pkey_pred = f"AND {partition_col} != ''" if partition_col is not None else "" level1 = f""" SELECT source_id, {col} AS val, + {pkey_expr} AS pkey, toString(event_id) AS eid, {eff} AS ets, byte_offset, @@ -1786,9 +1857,11 @@ def _ngram_inner_sql( WHERE case_id = {{cid:String}} AND has({{src:Array(String)}}, source_id) AND {col} != '' + {pkey_pred} AND {VESTIGO_NOT_SENTINEL_SQL} AND ({scope_pred}) """ + stream = "source_id, w_idx, pkey" if partition_col is not None else "source_id, w_idx" if max_gap is not None: gap = int(max_gap) # gap_s is NULL on each partition's first row; ClickHouse's `if` takes @@ -1814,19 +1887,21 @@ def _ngram_inner_sql( age('second', lagInFrame(ets, 1) OVER ord, ets) AS gap_s FROM ({level1}) WINDOW ord AS ( - PARTITION BY source_id, w_idx + PARTITION BY {stream} ORDER BY ets, byte_offset, line_number, event_id ) ) WINDOW ord AS ( - PARTITION BY source_id, w_idx + PARTITION BY {stream} ORDER BY ets, byte_offset, line_number, event_id ) """ - partition = "source_id, w_idx, seg" if max_gap is not None else "source_id, w_idx" + partition = f"{stream}, seg" if max_gap is not None else stream return f""" SELECT val, + pkey, + eid AS last_eid, ets, w_idx, [{gram_lags}] AS gram, @@ -8460,6 +8535,495 @@ def _find_sequence_novelty_self( warnings=run_warnings, ) + # ------------------------------------------------------------------ + # Transition time (D15) + # ------------------------------------------------------------------ + + @gated_heavy_scan + def find_transition_times( + self, + case_id: str, + source_ids: list[str], + series_field: str = "artifact", + partition_field: str | None = None, + limit: int = 50, + windows: AnalysisWindows | None = None, + min_ratio: float = 2.0, + min_transitions: int = 3, + max_candidates: int = 2000, + exclude_event_ids: set[str] | None = None, + allowlist: set[tuple[str, str]] | None = None, + field_mappings: dict[str, list[str]] | None = None, + source_offsets: dict[str, int] | None = None, + ) -> StatAnomalyResult: + """Return value-to-value transitions faster than the pair's learned floor. + + AMiner ``MinimalTransitionTimeDetector`` analog (D15). A *transition* + is one step ``a → b`` between consecutive events of one stream whose + *series_field* values differ; the stream is the source, further split + by *partition_field* when given (the identifier whose movements are + timed — a user across hosts, a session across states). Transitions are + assembled by :func:`_ngram_inner_sql` with ``n = 2``, so they inherit + the sequence detectors' ordering, the per-source scan discipline and + the guarantee that a pair never spans a window boundary. The duration + is ``dateDiff('millisecond', first_ts, ets)``. + + *Baseline frame* (``method="min-transition"``, *windows* given): per + pair the floor is the minimum baseline-window transition, learned from + at least *min_transitions* of them; each suspect window's fastest + transition of the pair is flagged when it undercuts the floor by at + least *min_ratio*×. A learned floor of zero cannot be undercut — with + second-resolution timestamps zero-length transitions are routine — so + such pairs are skipped and counted in a warning. *Self frame* + (``method="self-min-transition"``, no *windows*, D18): the floor is the + pair's **next-fastest** transition anywhere in the scope + (``groupArraySorted(2)``, the leave-one-out minimum), over at least + *min_transitions* transitions; two equally fastest transitions vouch + for each other and nothing is flagged. + + Candidates are the *max_candidates* fastest pairs per source (cap → + warning). On a multi-source scope the floor is the minimum over every + source's reference and counts are summed. Score = + ``1 − observed / reference``; the representative event is the arriving + event of the fastest transition. The allowlist key is + ``(series_field, "a → b")`` in both frames. + """ + detector = "transition_time" + if min_ratio <= 1.0: + raise ValueError("min_ratio must be greater than 1") + if windows is None: + return self._find_transition_times_self( + case_id, + source_ids, + series_field=series_field, + partition_field=partition_field, + limit=limit, + min_ratio=min_ratio, + min_transitions=min_transitions, + max_candidates=max_candidates, + exclude_event_ids=exclude_event_ids, + allowlist=allowlist, + field_mappings=field_mappings, + source_offsets=source_offsets, + ) + method = "min-transition" + self.ch.init_schema() + db = self.ch.database + + total_events = self._count_events(case_id, source_ids) + if total_events == 0: + return StatAnomalyResult( + status="no_data", + detector=detector, + method=method, + baseline_size=0, + windows=windows.payload(), + ) + baseline_size, suspect_totals = self._window_totals( + case_id, source_ids, windows, source_offsets + ) + run_warnings = _window_size_warnings(windows, suspect_totals) + if baseline_size == 0: + return StatAnomalyResult( + status="insufficient_data", + detector=detector, + method=method, + baseline_size=0, + warnings=[*run_warnings, "The baseline window contains no events."], + windows=windows.payload(), + ) + + params: dict[str, Any] = {"cid": case_id, "src": source_ids} + col = _col_expr(series_field, params, field_mappings) + pcol = ( + _col_expr(partition_field, params, field_mappings, prefix="pk") + if partition_field + else None + ) + bp, sps = _window_preds(windows, params, source_offsets) + eff = effective_ts_sql(source_offsets) + w_branches = ", ".join(f"{sp}, {i}" for i, sp in enumerate(sps)) + inner = _ngram_inner_sql( + db=db, + col=col, + eff=eff, + ngram=2, + w_idx_expr=f"multiIf({bp}, -1, {w_branches}, -2)", + scope_pred=" OR ".join([bp, *sps]), + partition_col=pcol, + ) + dur = "toInt64(dateDiff('millisecond', first_ts, ets))" + + # Query A (once per source): each suspect window's fastest transition + # per pair, fastest first — the candidates. The cap keeps the fastest, + # which are the ones the question is about; hitting it is disclosed. + params["cap"] = max_candidates + cand_sql = f""" + SELECT + gram, + w_idx, + min({dur}) AS fastest_ms, + argMin(last_eid, {dur}) AS evt, + argMin(ets, {dur}) AS at, + argMin(pkey, {dur}) AS pval, + count() AS n + FROM ({inner}) + WHERE guard IS NOT NULL AND gram[1] != gram[2] AND w_idx >= 0 + GROUP BY gram, w_idx + ORDER BY fastest_ms ASC, gram ASC, w_idx ASC + LIMIT {{cap:UInt32}} + {heavy_scan_settings()} + """ + # (gram, w_idx) -> [fastest_ms, evt, at, pval, n] + cands: dict[tuple[tuple[str, ...], int], list[Any]] = {} + for sid in source_ids: + rows = self.ch.client.query(cand_sql, parameters={**params, "src": [sid]}).result_rows + if len(rows) >= max_candidates: + run_warnings.append( + f"Source {sid}: hit the {max_candidates}-pair candidate cap — only its " + f"{max_candidates} fastest suspect transitions were fetched; slower " + f"pairs that still undercut their floor may be missing." + ) + for row in rows: + key = (tuple(str(v) for v in row[0]), int(row[1])) + fastest = int(row[2]) + slot = cands.get(key) + if slot is None: + cands[key] = [fastest, row[3], row[4], row[5], int(row[6])] + continue + slot[4] += int(row[6]) + if fastest < slot[0]: + slot[0], slot[1], slot[2], slot[3] = fastest, row[3], row[4], row[5] + if not cands: + return self._finalize_findings( + [], + detector=detector, + method=method, + total_events=baseline_size, + evaluated_fields=1, + exclude_event_ids=exclude_event_ids, + limit=limit, + case_id=case_id, + source_ids=source_ids, + allowlist=allowlist, + warnings=run_warnings, + windows=windows, + ) + + # Query B (once per source): the baseline floor for the candidate pairs. + pairs = sorted({g for g, _w in cands}) + learn_sql = f""" + SELECT gram, min({dur}) AS min_ms, count() AS n + FROM ({inner}) + WHERE guard IS NOT NULL AND gram[1] != gram[2] AND w_idx = -1 + AND has({{cands:Array(Array(String))}}, gram) + GROUP BY gram + {heavy_scan_settings()} + """ + learned: dict[tuple[str, ...], list[int]] = {} # gram -> [min_ms, n] + for sid in source_ids: + rows = self.ch.client.query( + learn_sql, parameters={**params, "src": [sid], "cands": [list(g) for g in pairs]} + ).result_rows + for row in rows: + gram = tuple(str(v) for v in row[0]) + slot = learned.setdefault(gram, [int(row[1]), 0]) + slot[0] = min(slot[0], int(row[1])) + slot[1] += int(row[2]) + if not learned: + return StatAnomalyResult( + status="insufficient_data", + detector=detector, + method=method, + baseline_size=baseline_size, + warnings=[ + *run_warnings, + "No suspect-window transition has a counterpart in the baseline window, " + "so there is no floor to compare against.", + ], + windows=windows.payload(), + ) + + zero_floors = 0 + findings: list[TransitionFinding] = [] + for (gram, wi), (fastest, evt, at, pval, n) in cands.items(): + ref = learned.get(gram) + if ref is None or ref[1] < min_transitions: + continue + floor_ms, n_baseline = ref + if floor_ms <= 0: + zero_floors += 1 + continue + if fastest * min_ratio >= floor_ms: + continue + window = windows.suspects[wi] + findings.append( + self._transition_finding( + case_id=case_id, + detector=detector, + method=method, + series_field=series_field, + partition_field=partition_field, + gram=gram, + fastest_ms=fastest, + reference_ms=floor_ms, + reference_kind="baseline-min", + evt=evt, + at=at, + pval=pval, + count=n, + baseline_count=n_baseline, + min_ratio=min_ratio, + min_transitions=min_transitions, + extra={ + "window_label": window.label, + "window_start": ensure_utc(window.start).isoformat(), + "window_end": ensure_utc(window.end).isoformat(), + "window_transitions": n, + "baseline_transitions": n_baseline, + "baseline_size": baseline_size, + }, + ) + ) + if zero_floors: + run_warnings.append( + f"{zero_floors} pair{'s' if zero_floors != 1 else ''} skipped: the baseline's " + f"fastest transition is zero seconds (same-timestamp records), and nothing " + f"can be faster than instant." + ) + return self._finalize_findings( + findings, + detector=detector, + method=method, + total_events=baseline_size, + evaluated_fields=1, + exclude_event_ids=exclude_event_ids, + limit=limit, + case_id=case_id, + source_ids=source_ids, + allowlist=allowlist, + warnings=run_warnings, + windows=windows, + ) + + def _find_transition_times_self( + self, + case_id: str, + source_ids: list[str], + *, + series_field: str, + partition_field: str | None, + limit: int, + min_ratio: float, + min_transitions: int, + max_candidates: int, + exclude_event_ids: set[str] | None, + allowlist: set[tuple[str, str]] | None, + field_mappings: dict[str, list[str]] | None, + source_offsets: dict[str, int] | None, + ) -> StatAnomalyResult: + """Self frame of :meth:`find_transition_times` (``method="self-min-transition"``). + + One pseudo-window over every scope event; per source and pair the two + fastest transitions (``groupArraySorted(2)``), the arriving event of + the fastest, and the pair's transition count. Merged across sources + by taking the two smallest durations overall. The floor is the second + of them — every other transition of the pair is at least that slow — + and the fastest is flagged when it undercuts the floor by *min_ratio*× + over at least *min_transitions* transitions. A zero floor (two + instantaneous transitions) is skipped and counted in a warning, as in + the baseline frame. + """ + detector = "transition_time" + method = "self-min-transition" + self.ch.init_schema() + db = self.ch.database + + total_events = self._count_events(case_id, source_ids) + if total_events == 0: + return StatAnomalyResult( + status="no_data", detector=detector, method=method, baseline_size=0 + ) + params: dict[str, Any] = {"cid": case_id, "src": source_ids} + bind_offset_params(source_offsets, params) + col = _col_expr(series_field, params, field_mappings) + pcol = ( + _col_expr(partition_field, params, field_mappings, prefix="pk") + if partition_field + else None + ) + eff = effective_ts_sql(source_offsets) + inner = _ngram_inner_sql( + db=db, + col=col, + eff=eff, + ngram=2, + w_idx_expr="0", + scope_pred="1", + partition_col=pcol, + ) + dur = "toInt64(dateDiff('millisecond', first_ts, ets))" + params["cap"] = max_candidates + sql = f""" + SELECT + gram, + groupArraySorted(2)({dur}) AS two_fastest_ms, + argMin(last_eid, {dur}) AS evt, + argMin(ets, {dur}) AS at, + argMin(pkey, {dur}) AS pval, + count() AS n + FROM ({inner}) + WHERE guard IS NOT NULL AND gram[1] != gram[2] + GROUP BY gram + ORDER BY two_fastest_ms[1] ASC, gram ASC + LIMIT {{cap:UInt32}} + {heavy_scan_settings()} + """ + run_warnings: list[str] = [] + # gram -> [two_fastest_ms (sorted, ≤ 2), evt, at, pval, n] + merged: dict[tuple[str, ...], list[Any]] = {} + for sid in source_ids: + rows = self.ch.client.query(sql, parameters={**params, "src": [sid]}).result_rows + if len(rows) >= max_candidates: + run_warnings.append( + f"Source {sid}: hit the {max_candidates}-pair candidate cap — only its " + f"{max_candidates} fastest pairs were fetched; slower pairs that still " + f"undercut their floor may be missing." + ) + for row in rows: + gram = tuple(str(v) for v in row[0]) + fastest = [int(v) for v in row[1]] + slot = merged.get(gram) + if slot is None: + merged[gram] = [fastest[:2], row[2], row[3], row[4], int(row[5])] + continue + if fastest and fastest[0] < slot[0][0]: + slot[1], slot[2], slot[3] = row[2], row[3], row[4] + slot[0] = sorted(slot[0] + fastest)[:2] + slot[4] += int(row[5]) + + zero_floors = 0 + findings: list[TransitionFinding] = [] + for gram, (two, evt, at, pval, n) in merged.items(): + if n < min_transitions or len(two) < 2: + continue + fastest, floor_ms = two[0], two[1] + if floor_ms <= 0: + zero_floors += 1 + continue + if fastest * min_ratio >= floor_ms: + continue + findings.append( + self._transition_finding( + case_id=case_id, + detector=detector, + method=method, + series_field=series_field, + partition_field=partition_field, + gram=gram, + fastest_ms=fastest, + reference_ms=floor_ms, + reference_kind="next-fastest", + evt=evt, + at=at, + pval=pval, + count=n, + baseline_count=n, + min_ratio=min_ratio, + min_transitions=min_transitions, + extra={"scope_transitions": n}, + ) + ) + if zero_floors: + run_warnings.append( + f"{zero_floors} pair{'s' if zero_floors != 1 else ''} skipped: the pair's two " + f"fastest transitions are both zero seconds (same-timestamp records), and " + f"nothing can be faster than instant." + ) + return self._finalize_findings( + findings, + detector=detector, + method=method, + total_events=total_events, + evaluated_fields=1, + exclude_event_ids=exclude_event_ids, + limit=limit, + case_id=case_id, + source_ids=source_ids, + allowlist=allowlist, + warnings=run_warnings, + ) + + @staticmethod + def _transition_finding( + *, + case_id: str, + detector: str, + method: str, + series_field: str, + partition_field: str | None, + gram: tuple[str, ...], + fastest_ms: int, + reference_ms: int, + reference_kind: str, + evt: Any, + at: Any, + pval: Any, + count: int, + baseline_count: int, + min_ratio: float, + min_transitions: int, + extra: dict[str, Any], + ) -> TransitionFinding: + """Build one :class:`TransitionFinding`; shared by both frames.""" + values = list(gram) + joined = " → ".join(values) + observed = fastest_ms / 1000.0 + reference = reference_ms / 1000.0 + score = 1.0 - fastest_ms / reference_ms + speedup = reference_ms / fastest_ms if fastest_ms > 0 else None + first_seen = _present_ts(at) + evt_id = str(evt) if evt else None + partition_value = str(pval) if partition_field and pval else None + details: dict[str, Any] = { + "detector": detector, + "method": method, + "field": series_field, + "values": values, + "value": joined, + "partition_field": partition_field, + "partition_value": partition_value, + "observed_seconds": observed, + "reference_seconds": reference, + "reference_kind": reference_kind, + "speedup": round(speedup, 4) if speedup is not None else None, + "count": count, + "min_ratio": min_ratio, + "min_transitions": min_transitions, + "first_seen": first_seen, + **extra, + "allowlist_field": series_field, + "allowlist_value": joined, + } + return TransitionFinding( + field=series_field, + values=values, + value=joined, + partition_field=partition_field, + partition_value=partition_value, + observed_seconds=observed, + reference_seconds=reference, + reference_kind=reference_kind, + speedup=round(speedup, 4) if speedup is not None else None, + count=count, + baseline_count=baseline_count, + score=round(score, 6), + first_seen=first_seen, + event_id=evt_id, + event=_stub_event(evt_id, case_id, first_seen), + details=details, + ) + @gated_heavy_scan def find_sequence_motifs( self, diff --git a/src/vestigo/demo/sources/windows.py b/src/vestigo/demo/sources/windows.py index 5aa57f11..0c864b10 100644 --- a/src/vestigo/demo/sources/windows.py +++ b/src/vestigo/demo/sources/windows.py @@ -1,9 +1,11 @@ """Windows Security channel, shaped like an EVTX-derived Plaso CSV export. Baseline: ordinary logon churn, process creation from a small software -vocabulary, and a stable set of service installs. The intrusion adds a -credential spray, an encoded-PowerShell process creation, a never-before-seen -service install for persistence, and wmic lateral movement. +vocabulary, a stable set of service installs, and one administrator's routine +jump-host hop (JUMP-01, then FILE-01 tens of seconds later). The intrusion adds +a credential spray, an encoded-PowerShell process creation, a never-before-seen +service install for persistence, and wmic lateral movement whose remote +process creation lands the account on FILE-01 two seconds after the call. Only ``datetime``, ``timestamp_desc``, ``message`` and ``source`` are consumed as event fields by the CSV parser; every other column lands in the event's @@ -165,6 +167,47 @@ def _baseline_process_creation() -> Iterator[dict[str, str]]: ) +#: The administrator whose routine is the jump-host hop below. +ADMIN_USER = "a.lindqvist" + + +def _admin_hops() -> Iterator[dict[str, str]]: + """One administrator's routine: JUMP-01, then FILE-01 tens of seconds later. + + Twice a working day across the whole window, an RDP logon to the jump host + followed by a network logon to the file server 20–90 s later — the time a + person takes to open the next session. This is the floor the transition + detector (§15) learns for the pair JUMP-01 → FILE-01, which the contractor's + wmic call then undercuts by an order of magnitude; the hop is kept tight so + the administrator's own home-workstation logons rarely fall between the two. + """ + r = scenario.rng("windows-admin-hops") + day = scenario.SCENARIO_START + while day < scenario.SCENARIO_END: + if day.weekday() < 5: + for hour in (r.randint(8, 11), r.randint(13, 17)): + start = day + timedelta( + hours=hour, minutes=r.randrange(60), seconds=r.randrange(60) + ) + yield _row( + start, + "4624", + f"An account was successfully logged on. Account Name: {ADMIN_USER}", + computer_name=scenario.JUMP_HOST, + user=ADMIN_USER, + logon_type="10", + ) + yield _row( + start + timedelta(seconds=r.uniform(20, 90)), + "4624", + f"An account was successfully logged on. Account Name: {ADMIN_USER}", + computer_name=scenario.FILE_SERVER, + user=ADMIN_USER, + logon_type="3", + ) + day += timedelta(days=1) + + def _baseline_service_installs() -> Iterator[dict[str, str]]: """Roughly two service installs a week, drawn only from the known set.""" r = scenario.rng("windows-services") @@ -285,6 +328,17 @@ def _lateral() -> Iterator[dict[str, str]]: process_name="wmic.exe", command_line=f"wmic /node:{scenario.FILE_SERVER} process call create cmd.exe /c hostname", ) + # The remote process creation logs the account on to the target at + # once — a network logon two seconds after the call. From JUMP-01 that + # is the pair the administrator's hop takes 20–90 s over (§15). + yield _row( + moment + timedelta(seconds=r.uniform(1.5, 2.5)), + "4624", + f"An account was successfully logged on. Account Name: {scenario.COMPROMISED_USER}", + computer_name=scenario.FILE_SERVER, + user=scenario.COMPROMISED_USER, + logon_type="3", + ) moment += timedelta(minutes=r.uniform(20, 90)) yield _row( @@ -305,6 +359,7 @@ def windows_rows() -> Iterator[dict[str, str]]: _baseline_logons, _baseline_process_creation, _baseline_service_installs, + _admin_hops, _spray, _foothold, _lateral, diff --git a/tests/test_analysis_plan.py b/tests/test_analysis_plan.py index 8aa201fd..f7c36418 100644 --- a/tests/test_analysis_plan.py +++ b/tests/test_analysis_plan.py @@ -74,6 +74,9 @@ def test_numeric_range_gated_off_without_numeric_fields(cfg): "value_distribution_drift", "interval_periodicity", "sequence_novelty", + # Born with both frames (D15), gated the same way: a baseline comparison + # with no baseline is the one thing an analyst action repairs. + "transition_time", ) @@ -122,6 +125,9 @@ def test_self_frame_shape_gates_speak_for_cadence_and_sequences(cfg): ) assert plans["sequence_novelty"].status == "not_applicable" assert plans["interval_periodicity"].status == "not_applicable" + # One value: nothing ever moves between two, so there is no transition. + assert plans["transition_time"].status == "not_applicable" + assert plans["transition_time"].reason_facts == {"series_distinct": 1, "required": 2} # The baseline frame still outranks the shape reason: it is the one the # analyst can act on, and it must not hide the "Set a baseline" affordance. plans = _by_id( diff --git a/tests/test_anomaly_stats.py b/tests/test_anomaly_stats.py index c37b737a..d6bd9a50 100644 --- a/tests/test_anomaly_stats.py +++ b/tests/test_anomaly_stats.py @@ -6501,3 +6501,327 @@ def test_gamma_from_median_recovers_the_median(): else: hi = mid assert abs(lo * theta - median) / median < 0.01 + + +# --------------------------------------------------------------------------- +# transition_time — detector (D15) +# --------------------------------------------------------------------------- + +# Baseline frame query order: count, window totals, then per source one +# suspect-candidate scan (Query A) and, when candidates exist, one baseline +# learn scan (Query B). Query A row layout: gram ([from, to]), w_idx, +# fastest_ms, evt, at, pval, n. Query B row layout: gram, min_ms, n. +# Self frame: count, then per source one scan whose rows are gram, +# two_fastest_ms ([m1, m2]), evt, at, pval, n. + +_TRANS_CAND_COLS = ["gram", "w_idx", "fastest_ms", "evt", "at", "pval", "n"] +_TRANS_LEARN_COLS = ["gram", "min_ms", "n"] +_TRANS_SELF_COLS = ["gram", "two_fastest_ms", "evt", "at", "pval", "n"] + + +def _trans_responses( + total: int, + window_totals: tuple[int, int], + cand_rows: list[tuple], + learn_rows: list[tuple] | None, +) -> list[FakeQueryResult]: + out = [ + FakeQueryResult(result_rows=[(total,)], column_names=["count()"]), + FakeQueryResult(result_rows=[window_totals], column_names=["bl_total", "w0_total"]), + FakeQueryResult(result_rows=cand_rows, column_names=_TRANS_CAND_COLS), + ] + if learn_rows is not None: + out.append(FakeQueryResult(result_rows=learn_rows, column_names=_TRANS_LEARN_COLS)) + return out + + +def test_transition_min_ratio_validation(): + svc = _svc([]) + with pytest.raises(ValueError, match="min_ratio"): + svc.find_transition_times("c1", ["s1"], min_ratio=1.0, windows=_seq_windows()) + assert svc.ch.client._calls == [] + + +def test_transition_no_data(): + svc = _svc([FakeQueryResult(result_rows=[(0,)], column_names=["count()"])]) + result = svc.find_transition_times("c1", ["s1"], windows=_seq_windows()) + assert result.status == "no_data" + assert result.detector == "transition_time" + assert result.windows is not None + + +def test_transition_baseline_flags_a_transition_faster_than_the_learned_floor(): + at = datetime(2024, 1, 17, 12, 0, tzinfo=UTC) + svc = _svc( + _trans_responses( + total=10_000, + window_totals=(8000, 2000), + # JUMP-01 → FILE-01 in 3 s during the incident; 4 such transitions. + cand_rows=[(["JUMP-01", "FILE-01"], 0, 3_000, "evt-1", at, "m.okonkwo", 4)], + # The baseline never saw it faster than 41 minutes, over 12 transitions. + learn_rows=[(["JUMP-01", "FILE-01"], 2_460_000, 12)], + ) + ) + result = svc.find_transition_times( + "c1", + ["s1"], + series_field="attr:computer_name", + partition_field="attr:user", + windows=_seq_windows(), + ) + assert result.status == "ok" + assert result.method == "min-transition" + assert result.baseline_size == 8000 + assert len(result.results) == 1 + r = result.results[0] + assert r.field == "attr:computer_name" + assert r.values == ["JUMP-01", "FILE-01"] + assert r.value == "JUMP-01 → FILE-01" + assert r.partition_field == "attr:user" + assert r.partition_value == "m.okonkwo" + assert r.observed_seconds == 3.0 + assert r.reference_seconds == 2460.0 + assert r.reference_kind == "baseline-min" + assert r.count == 4 + assert r.baseline_count == 12 + assert abs(r.score - (1 - 3 / 2460)) < 1e-5 + assert abs(r.speedup - 820.0) < 1e-9 + assert r.event_id == "evt-1" + assert r.first_seen is not None and r.first_seen.startswith("2024-01-17T12:00") + assert r.details["window_label"] == "incident" + assert r.details["min_ratio"] == 2.0 + assert r.details["min_transitions"] == 3 + assert r.details["allowlist_field"] == "attr:computer_name" + assert r.details["allowlist_value"] == "JUMP-01 → FILE-01" + # Both scans ran once for the single source: count, totals, A, B. + assert len(svc.ch.client._calls) == 4 + # The learn scan is bound to the candidate pairs. + assert svc.ch.client._all_parameters[3]["cands"] == [["JUMP-01", "FILE-01"]] + + +def test_transition_baseline_effect_floor_and_learning_floor(): + """A transition merely faster than the floor is not flagged; neither is one + whose pair the baseline saw too few times, nor one whose learned floor is + zero (second-resolution logs make zero-length transitions routine).""" + at = datetime(2024, 1, 17, tzinfo=UTC) + svc = _svc( + _trans_responses( + total=10_000, + window_totals=(8000, 2000), + cand_rows=[ + # 1.5x faster than the learned floor — under min_ratio 2. + (["a", "b"], 0, 20_000, "e1", at, "", 5), + # 100x faster, but the baseline holds only 2 transitions. + (["b", "c"], 0, 100, "e2", at, "", 5), + # Learned floor is 0 — nothing can be faster than instant. + (["c", "d"], 0, 0, "e3", at, "", 5), + # 10x faster over a well-learned pair: the one finding. + (["d", "e"], 0, 1_000, "e4", at, "", 2), + ], + learn_rows=[ + (["a", "b"], 30_000, 40), + (["b", "c"], 10_000, 2), + (["c", "d"], 0, 40), + (["d", "e"], 10_000, 40), + ], + ) + ) + result = svc.find_transition_times("c1", ["s1"], windows=_seq_windows()) + assert result.status == "ok" + assert [r.value for r in result.results] == ["d → e"] + assert result.results[0].partition_field is None + assert result.results[0].partition_value is None + assert any("zero" in w for w in result.warnings) + + +def test_transition_baseline_without_learned_pairs_is_insufficient(): + """No candidate pair has a baseline floor: nothing to compare against.""" + at = datetime(2024, 1, 17, tzinfo=UTC) + svc = _svc( + _trans_responses( + total=10_000, + window_totals=(8000, 2000), + cand_rows=[(["x", "y"], 0, 100, "e1", at, "", 3)], + learn_rows=[], + ) + ) + result = svc.find_transition_times("c1", ["s1"], windows=_seq_windows()) + assert result.status == "insufficient_data" + assert any("baseline" in w for w in result.warnings) + + +def test_transition_baseline_multi_source_merges_floor_and_fastest(): + """Per-source scans: the learned floor is the minimum over every source's + baseline, counts are summed, and the fastest suspect transition across + sources supplies the representative event.""" + at_fast = datetime(2024, 1, 16, 8, 0, tzinfo=UTC) + at_slow = datetime(2024, 1, 17, 8, 0, tzinfo=UTC) + responses = [ + FakeQueryResult(result_rows=[(10_000,)], column_names=["count()"]), + FakeQueryResult(result_rows=[(8000, 2000)], column_names=["bl_total", "w0_total"]), + FakeQueryResult( + result_rows=[(["a", "b"], 0, 5_000, "e-s1", at_slow, "", 2)], + column_names=_TRANS_CAND_COLS, + ), + FakeQueryResult( + result_rows=[(["a", "b"], 0, 2_000, "e-s2", at_fast, "", 3)], + column_names=_TRANS_CAND_COLS, + ), + FakeQueryResult(result_rows=[(["a", "b"], 60_000, 4)], column_names=_TRANS_LEARN_COLS), + FakeQueryResult(result_rows=[(["a", "b"], 30_000, 6)], column_names=_TRANS_LEARN_COLS), + ] + svc = _svc(responses) + result = svc.find_transition_times("c1", ["s1", "s2"], windows=_seq_windows()) + assert result.status == "ok" + assert len(result.results) == 1 + r = result.results[0] + assert r.observed_seconds == 2.0 + assert r.reference_seconds == 30.0 + assert r.count == 5 + assert r.baseline_count == 10 + assert r.event_id == "e-s2" + assert len(svc.ch.client._calls) == 6 + for p in svc.ch.client._all_parameters[2:]: + assert p["src"] in (["s1"], ["s2"]) + + +def test_transition_sql_shape_and_partition_field(): + at = datetime(2024, 1, 17, tzinfo=UTC) + client = RecordingClient( + _trans_responses( + total=10_000, + window_totals=(8000, 2000), + cand_rows=[(["a", "b"], 0, 1_000, "e1", at, "u1", 3)], + learn_rows=[(["a", "b"], 10_000, 9)], + ) + ) + svc = StatisticalAnomalyService.__new__(StatisticalAnomalyService) + svc.ch = FakeClickHouseStore(client) + result = svc.find_transition_times( + "c1", + ["s1"], + series_field="attr:computer_name", + partition_field="attr:user", + windows=_seq_windows(), + ) + assert result.status == "ok" + cand_sql, learn_sql = client.full_queries[2], client.full_queries[3] + for sql in (cand_sql, learn_sql): + # A transition is one step of the shared n-gram assembly, per stream. + assert "PARTITION BY source_id, w_idx, pkey" in sql + assert "ROWS BETWEEN 1 PRECEDING AND CURRENT ROW" in sql + assert "guard IS NOT NULL" in sql + # Same-value repeats are not transitions. + assert "gram[1] != gram[2]" in sql + assert "dateDiff('millisecond', first_ts, ets)" in sql + assert "w_idx >= 0" in cand_sql + assert "w_idx = -1" in learn_sql + assert "has({cands:Array(Array(String))}, gram)" in learn_sql + params = client._all_parameters[2] + assert params["b0"] == "2024-01-01 00:00:00.000" + assert params["w0s"] == "2024-01-16 00:00:00.000" + # Both fields are bound as attribute keys, never inlined. + assert "attributes[{fk:String}]" in cand_sql + assert "attributes[{pk:String}]" in cand_sql + assert params["fk"] == "computer_name" + assert params["pk"] == "user" + + +def test_transition_self_frame_uses_the_next_fastest_transition(): + """Without a baseline the reference is leave-one-out: the pair's second + fastest transition anywhere in the scope.""" + at = datetime(2024, 1, 17, tzinfo=UTC) + svc = _svc( + [ + FakeQueryResult(result_rows=[(10_000,)], column_names=["count()"]), + FakeQueryResult( + result_rows=[ + # 2 s against a next-fastest of 600 s: flagged. + (["JUMP-01", "FILE-01"], [2_000, 600_000], "e1", at, "m.okonkwo", 7), + # Two equally fast transitions: leave-one-out floor equals + # the observation, nothing is faster than its own twin. + (["a", "b"], [1_000, 1_000], "e2", at, "", 9), + # Below the transition floor of 3. + (["b", "c"], [1, 900_000], "e3", at, "", 2), + # 1.5x — under min_ratio. + (["c", "d"], [4_000, 6_000], "e4", at, "", 30), + ], + column_names=_TRANS_SELF_COLS, + ), + ] + ) + result = svc.find_transition_times( + "c1", ["s1"], series_field="attr:computer_name", partition_field="attr:user" + ) + assert result.status == "ok" + assert result.method == "self-min-transition" + assert result.windows is None + assert [r.value for r in result.results] == ["JUMP-01 → FILE-01"] + r = result.results[0] + assert r.observed_seconds == 2.0 + assert r.reference_seconds == 600.0 + assert r.reference_kind == "next-fastest" + assert r.count == 7 + assert r.baseline_count == 7 + assert r.partition_value == "m.okonkwo" + assert abs(r.score - (1 - 2 / 600)) < 1e-5 + assert r.details["scope_transitions"] == 7 + assert "window_label" not in r.details + assert len(svc.ch.client._calls) == 2 + + +def test_transition_self_frame_multi_source_takes_the_two_fastest_overall(): + at = datetime(2024, 1, 17, tzinfo=UTC) + svc = _svc( + [ + FakeQueryResult(result_rows=[(10_000,)], column_names=["count()"]), + FakeQueryResult( + result_rows=[(["a", "b"], [50_000, 70_000], "e-s1", at, "", 4)], + column_names=_TRANS_SELF_COLS, + ), + FakeQueryResult( + result_rows=[(["a", "b"], [1_000, 90_000], "e-s2", at, "", 3)], + column_names=_TRANS_SELF_COLS, + ), + ] + ) + result = svc.find_transition_times("c1", ["s1", "s2"]) + assert result.status == "ok" + r = result.results[0] + # Next-fastest is s1's 50 s, not s2's own 90 s. + assert r.observed_seconds == 1.0 + assert r.reference_seconds == 50.0 + assert r.count == 7 + assert r.event_id == "e-s2" + + +def test_transition_self_frame_candidate_cap_warns(): + at = datetime(2024, 1, 17, tzinfo=UTC) + rows = [([f"h{i}", f"h{i + 1}"], [1_000, 90_000], f"e{i}", at, "", 5) for i in range(3)] + svc = _svc( + [ + FakeQueryResult(result_rows=[(10_000,)], column_names=["count()"]), + FakeQueryResult(result_rows=rows, column_names=_TRANS_SELF_COLS), + ] + ) + result = svc.find_transition_times("c1", ["s1"], max_candidates=3) + assert result.status == "ok" + assert any("candidate cap" in w for w in result.warnings) + assert svc.ch.client._all_parameters[1]["cap"] == 3 + + +def test_transition_allowlist_suppresses_the_pair_in_both_frames(): + at = datetime(2024, 1, 17, tzinfo=UTC) + svc = _svc( + [ + FakeQueryResult(result_rows=[(10_000,)], column_names=["count()"]), + FakeQueryResult( + result_rows=[(["a", "b"], [1_000, 90_000], "e1", at, "", 5)], + column_names=_TRANS_SELF_COLS, + ), + ] + ) + result = svc.find_transition_times("c1", ["s1"], allowlist={("artifact", "a → b")}) + assert result.status == "ok" + assert result.results == [] + assert result.total_findings == 0 diff --git a/tests/test_demo_detector_coverage_clickhouse.py b/tests/test_demo_detector_coverage_clickhouse.py index e78bc2dd..d0aeb872 100644 --- a/tests/test_demo_detector_coverage_clickhouse.py +++ b/tests/test_demo_detector_coverage_clickhouse.py @@ -31,6 +31,9 @@ _SERIES_FIELD = { "find_sequence_novelty": {"series_field": "attr:computer_name"}, "find_sequence_motifs": {"series_field": "attr:event_id"}, + # Host-to-host moves timed per account: the contractor reaches hosts in + # minutes that the baseline population moved between over hours. + "find_transition_times": {"series_field": "attr:computer_name", "partition_field": "attr:user"}, } #: Detectors that score a baseline against suspect windows. @@ -45,6 +48,7 @@ "find_interval_periodicity", "find_distribution_drift", "find_sequence_novelty", + "find_transition_times", ) #: Detectors with no baseline/suspect split at all. @@ -110,6 +114,7 @@ def test_windowed_detector_finds_something(demo, ch_store, method): "find_interval_periodicity": {"fields": ["attr:host"]}, "find_distribution_drift": {"fields": ["attr:bytes_out"]}, "find_sequence_novelty": {"series_field": "attr:computer_name"}, + "find_transition_times": {"series_field": "attr:computer_name", "partition_field": "attr:user"}, } From 221dc7a29a3ed3af73e0e61d4bb9fe5aceb67232 Mon Sep 17 00:00:00 2001 From: Overcuriousity Date: Wed, 16 Sep 2026 12:07:27 +0000 Subject: [PATCH 2/5] =?UTF-8?q?feat(detectors):=20time-of-day=20habit=20?= =?UTF-8?q?=E2=80=94=20values=20at=20an=20hour=20they=20never=20keep=20(D1?= =?UTF-8?q?2)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A fourteenth statistical detector, time_of_day, adapted from AMiner's PathValueTimeIntervalDetector. Per (field, value) the day is cut into bucket_minutes-wide wall-clock buckets read in an explicit IANA timezone; the value's habit is the buckets holding at least stat_habit_min_bucket_count reference occurrences (for values with at least stat_habit_min_baseline of them), and an occurrence in any other bucket is reported, scored by the circular distance in hours to the nearest habitual bucket. Cadence measures the gap between arrivals; this reads the hour on the wall. Both frames: habit learns from the baseline window and scores each suspect window; self-habit takes the value's busy buckets across the scope and scores its thin ones. The zone is a server setting (stat_habit_timezone, UTC by default) overridable per run, validated against zoneinfo and a strict token pattern before it is inlined, and snapshotted with bucket_minutes into the persisted run and every finding — the same instant is a different hour elsewhere. Gate entry (always offered in the self frame, needs_setup in the baseline frame without a baseline), _TimeOfDayParams, /anomalies and agent knobs, five settings with registry specs, the method card with a bucket choice and a zone box, TimeOfDayFinding, a day-strip evidence figure, and ANOMALY_DETECTION.md §16. The agent tool's docstring is rewritten compact to keep the tool schema under its budget (43,832 over 35 tools). The demo case gains a one-off manual afternoon backup run (the self-frame signal) and moves the contractor's lateral movement onto the jump host at 03:00, a host whose baseline logons are an administrator's office hours; the nightly backup's move to 03:40 is the benign hit. Asserted in both frames. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 21 + CLAUDE.md | 4 +- README.md | 4 +- docs/AGENT.md | 6 + docs/ANOMALY_DETECTION.md | 123 +++++- docs/PROGRESS.md | 53 ++- docs/ROADMAP.md | 9 +- frontend/src/api/anomalies.ts | 8 +- frontend/src/api/types.ts | 53 ++- .../components/analysis/FindingEvidence.tsx | 61 +++ .../src/components/analysis/FindingGroup.tsx | 6 +- .../components/analysis/MethodKnobForm.tsx | 4 + .../components/analysis/detector-registry.ts | 3 + .../components/analysis/method-registry.ts | 35 +- frontend/src/lib/finding-frame.ts | 6 +- frontend/src/lib/finding-normalize.ts | 7 + frontend/src/lib/finding-subject.ts | 1 + frontend/src/lib/finding-verdict.ts | 15 + frontend/src/test/methodRegistry.test.ts | 4 +- src/vestigo/agent/tools.py | 49 ++- src/vestigo/api/routers/analysis.py | 16 + src/vestigo/api/routers/events.py | 90 ++++- src/vestigo/core/config.py | 15 + src/vestigo/core/settings_registry.py | 30 ++ src/vestigo/db/analysis_cache.py | 2 +- src/vestigo/db/analysis_plan.py | 9 + src/vestigo/db/anomaly_stats.py | 379 +++++++++++++++++- src/vestigo/demo/sources/linux.py | 26 +- src/vestigo/demo/sources/windows.py | 4 +- tests/test_analysis_plan.py | 5 +- tests/test_anomaly_stats.py | 275 +++++++++++++ .../test_demo_detector_coverage_clickhouse.py | 7 + tests/test_demo_generator.py | 11 +- 33 files changed, 1272 insertions(+), 69 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a18e1485..ba8e91ee 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,27 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **Time-of-day habit (D12).** A fourteenth statistical detector, `time_of_day`, adapted + from AMiner's `PathValueTimeIntervalDetector`: per (field, value) it cuts the day into + `bucket_minutes`-wide wall-clock buckets (15–240 minutes, default 60) read in an explicit + IANA `timezone` (default `UTC`, `stat_habit_timezone` for the site), learns the value's + habit — the buckets holding at least `stat_habit_min_bucket_count` reference occurrences, + for values with at least `stat_habit_min_baseline` of them — and reports an occurrence in + any other bucket, scored by the circular distance in hours to the nearest habitual one. + Interval cadence measures the gap between arrivals; this reads the hour on the wall, and + the two are independent (a nightly job that moves keeps its cadence and breaks its + habit). Both frames: `habit` learns from the baseline window and scores each suspect + window; `self-habit` takes the value's own busy buckets across the timeline and scores + its thin ones. The zone and resolution are snapshotted into the persisted run and carried + on every finding, since the same instant is a different hour elsewhere. Auto field + selection follows the novelty recommender and the timeline's field overrides; the + allowlist key is `(field, value)`. Gate entry (always offered in the self frame), wizard + card with a bucket choice and a zone box, a day-strip evidence figure, agent knobs, five + settings with registry specs and `docs/ANOMALY_DETECTION.md` §16 ship with it. The demo + case gains a one-off manual afternoon backup run (the self-frame signal) and moves the + contractor's lateral movement onto the jump host at 03:00, a host whose baseline logons + are an administrator's office hours; the nightly backup's move to 03:40 is the benign hit. + - **Transition speed (D15).** A thirteenth statistical detector, `transition_time`, adapted from AMiner's `MinimalTransitionTimeDetector`: per ordered value pair of a series field it learns the fastest a stream ever moved from one value to the next and reports a diff --git a/CLAUDE.md b/CLAUDE.md index 86cd93a2..52e65ce1 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -26,7 +26,7 @@ trail (`api/routers/auth.py`, `admin.py`, `deps.py`). - `CONCEPT.md` / `MODEL_REFINEMENT.md` — product vision and the Case/Source/Timeline/Event/ Artifact data model. Read before touching the model; rarely changes. - `TECH_STACK.md` — backing-service decision record (*why*, not *what's shipped*). -- `ANOMALY_DETECTION.md` — reference for all fifteen analysis tools actually running +- `ANOMALY_DETECTION.md` — reference for all sixteen analysis tools actually running (statistical detectors, Sigma runner, log templates, semantic similarity), plus the baseline/disposition model. Update alongside any detector change in the same commit. - `AGENT.md` — the optional AI investigation agent (design invariants, MCP tools, provider @@ -253,7 +253,7 @@ instead of rebuilding. does not strand its verdict bar a screen below the claim; `ToolsSheet` is four tabs (Scope, Methods, Signatures, Explore) rather than one scroll, so a thousand-row template list cannot bury the baseline picker; `method-registry.ts` is the - single description of all thirteen methods, including the prose that used to live in a + single description of all fourteen methods, including the prose that used to live in a Method tab and each method's optional `railFloor`, a presentation-only bar on the ranked feed whose held-back count is always disclosed; the sheet's method mode runs a method with the analyst's own knob values, which is what keeps the analysis diff --git a/README.md b/README.md index 58f1400d..2461fdd8 100644 --- a/README.md +++ b/README.md @@ -69,7 +69,7 @@ resolved on the machine you deploy to. ClickHouse, time histogram with anomaly overlays, keyset pagination with jump-to-time, tag/comment annotations with bulk apply, saved views, and streaming CSV/JSONL export that keeps the forensic columns. -- **Anomaly detection** — fifteen analysis tools: thirteen statistical detectors over +- **Anomaly detection** — sixteen analysis tools: fourteen statistical detectors over ClickHouse needing no embeddings, a Sigma rule runner, and semantic similarity search over local embeddings. Each is documented method by method, scores against explicit baseline-vs-suspect windows, and yields findings whose confirm/dismiss disposition @@ -109,7 +109,7 @@ feel like, and the Case/Timeline model here is descended from it. That is the co invite, and three axes are where we think we are already the better place to run an investigation: -- **Detection is the workflow, not an add-on** — fifteen analysis tools in the box, each +- **Detection is the workflow, not an add-on** — sixteen analysis tools in the box, each scoring against an analyst-declared baseline and carrying a verdict that survives re-scans, so triage accumulates instead of being redone. - **Provenance goes all the way down** — not just "this file was imported": a finding is diff --git a/docs/AGENT.md b/docs/AGENT.md index 615f0623..a59a4c1e 100644 --- a/docs/AGENT.md +++ b/docs/AGENT.md @@ -716,6 +716,12 @@ detector runs without a baseline (D18) took it to **43,891 over 35 tools**, ceil unchanged — the first draft landed at 44,197 and the rule was applied: the tool's docstring was rewritten compact (−306 chars net) rather than the ceiling moved. +The 1.20 detectors (2026-09-16: `transition_time`'s `partition_field`, `time_of_day`'s +`bucket_minutes` and `timezone`) took the first draft to 44,421 and the rule was applied +again: the docstring now names each detector once and each knob in a few words, and the +`timezone` length constraints came off the schema (the runner validates the zone anyway) — +**43,832 over 35 tools**, ceiling unchanged, below the D11 figure. + Detector findings additionally reduce their inline example event in the **model's copy** to `event_id` + truncated `message` (`_deflate_findings` — on the turn that motivated it: 33.7k → 15.7k tokens); diff --git a/docs/ANOMALY_DETECTION.md b/docs/ANOMALY_DETECTION.md index 460016e0..58723528 100644 --- a/docs/ANOMALY_DETECTION.md +++ b/docs/ANOMALY_DETECTION.md @@ -7,7 +7,7 @@ This document covers every detector actually running in the codebase today. If a detector described here changes (formula, default, field name), update this file and the "Method" tab copy in the same commit. -There are fifteen independent analysis tools in Vestigo: +There are sixteen independent analysis tools in Vestigo: 1. [Value novelty](#1-value-novelty-rare--first-seen-values) — rare/new field values, single field or [combinations](#value-combinations-the-value_combo-variant) (ClickHouse, no ML) 2. [Frequency anomalies](#2-frequency-anomalies-volume-spikes--silences) — volume spikes/silences (ClickHouse, no ML) @@ -24,13 +24,14 @@ There are fifteen independent analysis tools in Vestigo: 13. [Sigma rule runner](#13-sigma-rule-runner-signature-matching) — signature matching: community/custom Sigma rules compiled to ClickHouse predicates (ClickHouse, no ML) 14. [Log templates](#14-log-templates-structural-line-clustering) — structural clustering of raw lines into templates, so rare *shapes* surface without naming a field (ClickHouse, no ML) 15. [Transition speed](#15-transition-speed-value-to-value-moves-faster-than-ever-seen) — a stream reaching the next value of a field faster than that pair was ever reached before (ClickHouse, no ML) +16. [Time-of-day habit](#16-time-of-day-habit-values-at-an-hour-they-never-keep) — a value occurring at a wall-clock hour it has no habit of, in an explicit timezone (ClickHouse, no ML) All but the eleventh are **statistical or rule-based**: pure counting, arithmetic and predicate matching over already-ingested events — no machine learning, no network calls, working the instant ingestion finishes. The eleventh needs an explicit embedding step first. -Code: `src/vestigo/db/anomaly_stats.py` (detectors 1–10, 12, 14, 15), +Code: `src/vestigo/db/anomaly_stats.py` (detectors 1–10, 12, 14–16), `src/vestigo/db/similarity.py` (detector 11), `src/vestigo/sigma/` (detector 13). UI: `frontend/src/components/analysis/`. @@ -63,6 +64,7 @@ teaches the tooling instead of the work. What each tool has to find: | 13 | Sigma | Four case-scoped rules ship inside the case: encoded PowerShell, suspicious service install, wmic remote process creation, failed-logon burst. | | 14 | Log templates | ~40 syslog templates, with an `unattended-upgrade` shape that appears only in the suspect window. | | 15 | Transition speed | The contractor account on `FILE-01` two seconds after its wmic call from `JUMP-01`, timed per account over `attr:computer_name`, against an administrator whose routine jump-host hop takes the same pair 20–90 s. | +| 16 | Time-of-day habit | The nightly backup program at 03:xx and 04:xx after three weeks of 02:xx runs (benign, the same move interval cadence sees), and the contractor on `JUMP-01` at 03:00, a host whose baseline logons are an administrator's office hours. Without a baseline, the one manual afternoon backup run twelve hours from the program's nightly slot. | `tests/test_demo_detector_coverage_clickhouse.py` asserts that each of these actually returns findings. If a retuned threshold silences one of them, that @@ -353,7 +355,7 @@ checked, and the UI must never present the two the same way. | Method | Precondition | Setting | |---|---|---| -| `value_novelty`, `timestamp_order`, `log_template`, `entropy` | none — always offered | — | +| `value_novelty`, `timestamp_order`, `log_template`, `entropy`, `time_of_day` | none — always offered | — | | `value_combo` | ≥2 categorical fields | — | | `numeric_range` | ≥1 field whose sampled values are ≥90 % numeric | `analysis_gate_min_numeric_ratio` | | `charset` | ≥1 field above the enum-like ceiling | `analysis_gate_max_enum_distinct` | @@ -361,7 +363,7 @@ checked, and the UI must never present the two the same way. | `interval_periodicity` | enough events per series value to fit a cadence | `analysis_gate_min_interval_periods` | | `sequence_novelty`, `transition_time` | a series field with ≥2 distinct values (one value yields no ordering and no transition) | `analysis_gate_min_series_distinct` | | `proportion_shift`, `value_distribution_drift` | a span of more than one instant (self frame: it is cut into slices) | `stat_self_slices` | -| any of those five in the **baseline frame** | an active baseline definition | — (reported as `needs_setup`) | +| any of those five, or `time_of_day`, in the **baseline frame** | an active baseline definition | — (reported as `needs_setup`) | Three rows encode a distinction worth stating, because each was once drawn wrong: @@ -2644,6 +2646,116 @@ both frames, so one **Normal** verdict on a pair covers it wherever it recurs. - A fast transition is **not malicious by itself** — a scripted deployment legitimately touches ten hosts in ten seconds. Rank for triage; mark the routine pair Normal once. +## 16. Time-of-day habit (values at an hour they never keep) + +**What it answers:** "Does this value keep hours — and did it break them?" The AMiner +`PathValueTimeIntervalDetector` analog (roadmap D12). A backup job that runs at 02:15, +a service account that works office hours, a user who has never logged on after +eight: each has a daily *habit*, and an occurrence outside it is worth a look whatever +the value itself is. Interval cadence (§8) measures the gap between arrivals; this +detector reads the hour on the wall. The two are independent: the demo's backup keeps +its once-a-day cadence perfectly when it moves from 02:15 to 03:40, and breaks its +habit. + +**How it works.** The day is cut into `1440 / bucket_minutes` wall-clock buckets +(default 60, so 24 hourly buckets) read in an explicit **IANA timezone** (default +`UTC`; `stat_habit_timezone` sets a site's zone once). The zone is not a +presentation choice: the same instant is 03:40 in Berlin and 01:40 in London, so the +run records it in `DetectorRun.params` and every finding carries it, or the run could +not be reproduced. Per (field, value) the detector counts occurrences per bucket in +the reference, and the value's **habit** is the set of buckets holding at least +`stat_habit_min_bucket_count` (3) of them — learned only for values with at least +`stat_habit_min_baseline` (20) reference occurrences, since a handful of events say +nothing about hours kept. An occurrence in any other bucket is flagged, one finding +per (value, bucket) — per suspect window in the baseline frame — with the bucket's +count, and scored by the **circular distance in hours** to the nearest habitual +bucket: 23:xx is one hour from a 00:xx habit, not twenty-three. + +**Two frames.** + +| | Self (`self-habit`) | Baseline (`habit`) | +|---|---|---| +| Reference | the value's own occurrences across the whole scope | the value's occurrences in the baseline window | +| Habit | its buckets holding ≥ `min_bucket_count` scope occurrences | its buckets holding ≥ `min_bucket_count` baseline occurrences | +| Flags | every *thin* bucket — under the floor, hence not habitual — against the busy ones | each suspect window's occurrences outside the baseline habit, habitual or not | +| Catches | one manual afternoon run of a nightly job | a nightly job that moved; an account active at 03:00 for the first time | + +The self frame is leave-one-out in the only sense that matters here: a value that +occurs once at 15:00 and fifty times at 02:00 is judged by the fifty. What it cannot +see is a *repeated* new hour — a bucket that fills past the floor becomes habitual by +definition — which is what a baseline is for: the same fifty-plus-eight becomes "eight +occurrences at 03:xx against a baseline habit of 02:xx" once the analyst declares the +window. + +One scan per field: per (value, bucket, window) counts with the first occurrence, +restricted to the `stat_habit_max_candidates_per_field` (500) highest-volume values +(cap → warning; lower than the other per-field caps because each value returns one +row per occupied bucket and window). Auto field selection is the novelty +recommender's categorical set, steered by the timeline's +[field overrides](#declaring-which-fields-a-method-reads). Values below the learning +floor are counted in a warning rather than silently skipped. + +**Score = hours to the nearest habitual bucket**, in whole multiples of the bucket +width: a 60-minute resolution scores 1, 2 … 12; a 15-minute one 0.25 upward. The +representative event is the first occurrence in the bucket (within the window, or the +scope), and `first_seen` is its timestamp. Findings carry `bucket`, `bucket_label` +(`"03:00–04:00"`), `bucket_minutes`, `timezone`, `count`, `baseline_count` (the +reference occurrences the habit was learned from), `habit_buckets` with their labels, +`nearest_habit` / `nearest_habit_label` and `distance_hours`; the baseline frame adds +`window_label`/`window_start`/`window_end`, the self frame `scope_occurrences`. + +**Parameters.** + +- `fields` (request, default auto) — the recommender's categorical fields, as for + value novelty; steered by field overrides, bypassed by an explicit list. +- `bucket_minutes` (request) / `VESTIGO_STAT_HABIT_BUCKET_MINUTES` (server default + 60) — one of 15, 30, 60, 120, 180, 240; each divides the day so the last bucket ends + at midnight. Snapshotted as `bucket_minutes`. +- `timezone` (request) / `VESTIGO_STAT_HABIT_TIMEZONE` (server default `UTC`) — an + IANA zone name, validated against the host's zoneinfo database and a strict token + pattern before it is inlined into SQL (ClickHouse takes a zone as a constant). + Snapshotted as `timezone`. +- `VESTIGO_STAT_HABIT_MIN_BASELINE` (default 20) — reference occurrences a value + needs before it has a habit. +- `VESTIGO_STAT_HABIT_MIN_BUCKET_COUNT` (default 3) — reference occurrences a bucket + needs to be habitual. +- `VESTIGO_STAT_HABIT_MAX_CANDIDATES_PER_FIELD` (default 500) — the per-field + candidate cap; hitting it attaches a warning. + +**Allowlist key:** `(field, value)` — a value declared Normal is normal at any hour. +There is no per-bucket key on purpose: "the backup may run at 03:40 now" is a change +to the baseline definition, not a verdict on a finding. + +### Caveats + +- **The zone is the claim.** A run in `UTC` over a site that works in `Asia/Tokyo` + reports office hours as a night-time habit and a 03:00 logon as ordinary. Set + `stat_habit_timezone` for the site, or the knob per run; the finding says which zone + it read, so a reader can tell. +- **Habits need volume.** Twenty occurrences over a three-week baseline is the floor, + not a comfortable sample; a value with thirty occurrences spread thinly across the + day has a patchy habit, and the empty buckets between its busy ones will flag. The + bucket floor (3) is what keeps one stray reference occurrence from becoming a habit, + and it also means a bucket with two reference occurrences is *not* habit — read + `habit_buckets` before reading the distance. +- **Round-the-clock values have no habit to break.** A value present in every bucket + is never flagged, which is correct: an all-hours process has no hour it does not + keep. High-volume fields (a busy user, a chatty host) mostly look like this, and the + detector earns its keep on the low-volume, scheduled and role-bound values around + them. +- **Weekends and holidays are not modelled.** A weekday-only habit is still a + time-of-day habit, and a Saturday occurrence at 10:00 is inside it. Day-of-week is + a different question (`time:day_of_week` in Visualize answers it), deliberately not + folded in here. +- **Clock skew moves the hour.** Occurrences are bucketed on the corrected timestamp + (W2), so a source with a declared offset is read at its corrected wall-clock hour, + as every other time-derived query reads it. +- An off-hours occurrence is **not malicious by itself** — a manual run, a time-zone + traveller, a shifted maintenance window. Rank for triage; mark the value Normal, or + move the baseline, once it is explained. + +--- + --- ## Dispositions and normality (implementation notes) @@ -2676,7 +2788,8 @@ always for `tag_anomalies`) writes a `DetectorRun` row: the request params it ran with — fields, `series_field`, thresholds, `baseline_id`, resolved windows, `windows_hash`, `dispositions_hash`, the per-source clock-skew offsets in effect, entropy's `variant`, transition speed's `partition_field` and -`min_transitions`, and for a self-frame run of the slice methods the +`min_transitions`, time-of-day's `bucket_minutes` and `timezone`, and for a +self-frame run of the slice methods the `slices` payload, its `slices_hash` and the resolved self settings (`self_slices`, `pause_ratio`, `min_span_seconds`, `sequence_rarity_floor`) — plus the serialized result, and returns its id as `run_id`. Rows diff --git a/docs/PROGRESS.md b/docs/PROGRESS.md index 63c0dfca..65067989 100644 --- a/docs/PROGRESS.md +++ b/docs/PROGRESS.md @@ -4,8 +4,57 @@ Append-only session log — what changed and why, newest first. This file keeps sessions only; older ones live in git history, and every release is summarized in `CHANGELOG.md`. Plans belong in `ROADMAP.md`, not here. -Last updated: 2026-09-16 (1.20 in progress; session 239 — the transition-speed detector, -D15, first of the 1.20 detector cluster). +Last updated: 2026-09-16 (1.20 in progress; sessions 239–240 — transition speed D15 and +time-of-day habit D12, the first two of the 1.20 detector cluster). + +## Session 240 — 2026-09-16: time-of-day habit (D12) + +The second 1.20 detector, `time_of_day`, in the same shape as D15: one commit with its +gate entry, params model, agent knobs, run snapshot, settings, method card, finding type, +evidence figure, reference section and demo signal. + +**The timezone is the design decision, and it is a knob plus a setting, not a timeline +attribute.** The roadmap's one requirement was an explicit zone stamped into +`DetectorRun.params`. Inventing a per-timeline zone would be a data-model change +(`MODEL_REFINEMENT.md` territory) for a single detector, so the zone is `stat_habit_timezone` +(server default `UTC`, set once for a site) overridable per run, validated against zoneinfo +and a strict token pattern, then inlined into SQL — ClickHouse takes a zone as a constant, +so it cannot be bound — and recorded on the run and on every finding. `_time_fields.py` +already pins its `toHour` to `'UTC'` for the same reason this detector cannot leave the +zone implicit: the server's zone can change under a stored run. + +**Habit = buckets with enough reference mass; distance is circular.** A bucket is habitual +when it holds at least `stat_habit_min_bucket_count` (3) reference occurrences, and only a +value with at least `stat_habit_min_baseline` (20) of them has a habit at all. Score is +the circular distance in hours to the nearest habitual bucket, in multiples of the bucket +width, so 23:xx sits one hour from a 00:xx habit. The self frame is the same rule over the +whole scope: every thin bucket is, by definition, not habitual, and is scored against the +busy ones — which catches a one-off manual run and cannot catch a *repeated* new hour, and +the reference section says so rather than pretending otherwise; that is what the baseline +frame is for. + +**One scan per field, bounded by value.** Rows are per (value, bucket, window), so the +per-field cap is 500 values rather than the usual 2000: at a 15-minute resolution with +four suspect windows a value can return 480 rows. The candidate set is the highest-volume +values, selected in a subquery over the same predicate. + +**The demo's habits.** `walk()` gives every hour of the day a non-zero weight, so every +high-volume value (any human account, any home workstation) is habitual round the clock — +correct behaviour, and it means the demo signals had to come from low-volume scheduled or +role-bound streams. Baseline frame: the nightly backup program at 03:xx/04:xx against +three weeks of 02:xx (benign, deliberately the same move interval cadence already sees), +and the contractor on `JUMP-01` at 03:00 — the lateral-movement leg now visits the jump +host first, at night, and the jump host's baseline logons are the administrator's hop from +session 239, all office hours. Self frame: one manual backup run on a baseline afternoon, +twelve hours from the program's nightly slot; without it the self frame's only hit was two +runs that happened to spill past 04:00, which is the kind of RNG accident this file's +demo-coverage rule exists to replace with a fabricated signal. + +**Environment note.** This machine cannot run rootless podman, so PostgreSQL 16 and the +pinned ClickHouse 26.6.1.1193 run user-space from `~/.local/share/vestigo-devstack/`; the +`embeddings` extra was installed to match CI's `--all-extras`. Qdrant is still absent, so +five case-delete tests (`test_stories_api`, `test_rbac_api`, `test_demo_api`) 502 here and +only here; they are unrelated to this work and pass in CI. ## Session 239 — 2026-09-16: transition speed (D15), the first 1.20 detector diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 928153e0..e614d7d2 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -12,7 +12,7 @@ from that day. **Milestone 10 — AI agent log investigation — is the 2.0 thru everything below**; the numbered list orders the remaining 1.x work by payoff-per-effort: 1. **A12** local transform tools — no design round, no OPSEC gate. -2. **D12** / **D13** — cheap detectors reusing existing SQL machinery (D15 shipped in 1.20). +2. **D13** — the last cheap detector reusing existing SQL machinery (D15 and D12 shipped in 1.20). 3. **W8** query-time field extraction — makes bespoke unstructured logs first-class. 4. **A8** external MCP toolsets — needs its own design round (policy, not plumbing). 5. **D10** / **D16** — heaviest lifts, last of the detector line. @@ -141,7 +141,7 @@ burns its numbers out of that file**; the migration is done when the file is `{} Detectors adapted from [ait-aecid/logdata-anomaly-miner](https://github.com/ait-aecid/logdata-anomaly-miner), constrained to be **field-agnostic** and SQL-explainable per the forensic-reproducibility -requirement. D1–D9, D15, `proportion_shift` and `sequence_motif` shipped — `ANOMALY_DETECTION.md` +requirement. D1–D9, D12, D15, `proportion_shift` and `sequence_motif` shipped — `ANOMALY_DETECTION.md` is each detector's contract, updated in the same commit as any detector change. Every item below is incomplete until the frontend half lands with it: a plain-language @@ -150,11 +150,6 @@ A detector whose reasoning an analyst cannot read does not count as shipped. **Low effort, high value:** -- [ ] **D12 — Time-of-day habit** (`PathValueTimeIntervalDetector`): per value, learn which - times of day it occurs at in the baseline, flag suspect-window occurrences outside that - habit. Distinct from `interval_periodicity`, which measures inter-arrival gaps. Bucket by - `toHour`/`toMinute`, score by distance to the nearest occupied bucket. Needs an explicit - **timezone** decision stamped into `DetectorRun.params`, or the run is not reproducible. - [ ] **D13 — Cross-field value correlation** (`VariableCorrelationDetector`): learn which field-value pairs co-occur *within the same event*, flag violations. Intra-record, unlike D10. Reuses `GROUP BY a, b` plus the G-test and Benjamini–Hochberg pool that diff --git a/frontend/src/api/anomalies.ts b/frontend/src/api/anomalies.ts index e153b062..4de251fc 100644 --- a/frontend/src/api/anomalies.ts +++ b/frontend/src/api/anomalies.ts @@ -21,7 +21,7 @@ export interface LogTemplatesParams { } export interface AnomalyParams { - detector?: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time"; + detector?: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time" | "time_of_day"; /** Comma-separated field tokens for value_novelty, e.g. "artifact,display_name,attr:user_agent" */ fields?: string; /** Field to group frequency series / build event sequences by */ @@ -48,6 +48,10 @@ export interface AnomalyParams { max_gap_seconds?: number; /** transition_time only: the stream whose transitions are timed (e.g. "attr:user"). Omit for one stream per source. */ partition_field?: string; + /** time_of_day only: width of the wall-clock buckets (15, 30, 60, 120, 180 or 240 minutes). Omit for the server default. */ + bucket_minutes?: number; + /** time_of_day only: IANA zone the clock is read in (e.g. "Europe/Berlin"). Omit for the server default. */ + timezone?: string; /** ID of a saved baseline definition (baseline range + suspect windows). Omit for self-baseline. */ baseline_id?: string; limit?: number; @@ -104,7 +108,7 @@ export const anomaliesApi = { sourceId: string, eventId: string, body: { - detector: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time"; + detector: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time" | "time_of_day"; content: string; details: Record; /** diff --git a/frontend/src/api/types.ts b/frontend/src/api/types.ts index 4681f6df..6db08061 100644 --- a/frontend/src/api/types.ts +++ b/frontend/src/api/types.ts @@ -836,6 +836,53 @@ export interface TransitionTimeFinding { confirmed_other_scope?: boolean; } +/** + * One value occurring at a time of day it has no habit of, from the + * time_of_day detector (D12). `details.method` is `habit` (the habit was + * learned from the baseline window; `details` carries `window_*` keys) or + * `self-habit` (the value's own busy buckets across the timeline, with + * `scope_occurrences` and no window keys). The bucket resolution and the + * IANA zone the clock was read in are on every finding — the same wall-clock + * hour in two zones is two different claims. + */ +export interface TimeOfDayFinding { + type: "time_of_day"; + field: string; + value: string; + /** The offending wall-clock bucket: index, "HH:MM–HH:MM" label, width and zone. */ + bucket: number; + bucket_label: string; + bucket_minutes: number; + timezone: string; + /** Occurrences of the value in this bucket (in the suspect window, or the timeline). */ + count: number; + /** Reference occurrences the habit was learned from. */ + baseline_count: number; + /** The habitual bucket indexes, ascending; the nearest one and its label. */ + habit_buckets: number[]; + nearest_habit: number; + nearest_habit_label: string; + /** Circular distance to the nearest habitual bucket, in hours. */ + distance_hours: number; + /** = distance_hours — used for ranking. */ + score: number; + /** First occurrence in the bucket. */ + first_seen: string | null; + event_id: string | null; + event: Event | null; + details: Record; + /** Present (true) only when the request passed `include_dismissed`. */ + dismissed?: boolean; + /** Present (true) when a confirmed disposition covers this finding's event. */ + confirmed?: boolean; + /** + * Present (true) when the only confirmed verdict on this event was reached + * under a *different* comparison. The claim stands, but not for this scope — + * so the row is marked rather than badged, and Confirm stays live. + */ + confirmed_other_scope?: boolean; +} + export type AnomalyFinding = | ValueNoveltyFinding | ValueComboFinding @@ -849,7 +896,8 @@ export type AnomalyFinding = | SequenceNoveltyFinding | SequenceMotifFinding | DistributionDriftFinding - | TransitionTimeFinding; + | TransitionTimeFinding + | TimeOfDayFinding; export interface AnomaliesResponse { status: "ok" | "no_data" | "insufficient_data"; @@ -1056,7 +1104,8 @@ export interface AnomalyMarker { | "interval_periodicity" | "sequence_novelty" | "value_distribution_drift" - | "transition_time"; + | "transition_time" + | "time_of_day"; /** Raw structured finding data — stored verbatim on the persisted annotation. */ rawDetails: Record; /** End of the anomalous window, for frequency findings — enables a range highlight. */ diff --git a/frontend/src/components/analysis/FindingEvidence.tsx b/frontend/src/components/analysis/FindingEvidence.tsx index 620e4aa4..f8912cf4 100644 --- a/frontend/src/components/analysis/FindingEvidence.tsx +++ b/frontend/src/components/analysis/FindingEvidence.tsx @@ -172,6 +172,57 @@ function NovelChars({ value, novel }: { value: string; novel: string[] }) { ); } +/** + * The day as a strip of buckets: the habitual ones in the reference neutral, + * the offending one in the anomaly accent, the rest empty. Every cell comes + * from the finding (`habit_buckets`, `bucket`, `bucket_minutes`); nothing is + * drawn for buckets the payload says nothing about. + */ +function DayStrip({ + bucket, + habit, + bucketMinutes, + bucketLabel, + timezone, +}: { + bucket: number; + habit: number[]; + bucketMinutes: number; + bucketLabel: string; + timezone: string; +}) { + const n = Math.max(1, Math.floor(1440 / bucketMinutes)); + const habitual = new Set(habit); + const caption = `${bucketLabel} ${timezone}; habitual buckets ${habit.length}`; + return ( +
+
+ {Array.from({ length: n }, (_, i) => ( + + ))} +
+
+ 00:00 + + {bucketLabel} {timezone} + + 24:00 +
+
+ ); +} + /** The n-gram, oldest → newest, so the *order* is what the eye reads. */ function Ngram({ values }: { values: string[] }) { return ( @@ -326,6 +377,16 @@ export function FindingEvidence({ finding }: { finding: MethodResult }) { case "sequence_novelty": case "sequence_motif": return ; + case "time_of_day": + return ( + + ); case "transition_time": // The claim is one duration against one floor, both measured; which // floor is in the label, since the two answer different questions. diff --git a/frontend/src/components/analysis/FindingGroup.tsx b/frontend/src/components/analysis/FindingGroup.tsx index bea565b0..d8167483 100644 --- a/frontend/src/components/analysis/FindingGroup.tsx +++ b/frontend/src/components/analysis/FindingGroup.tsx @@ -50,9 +50,9 @@ interface Props { * Rows this group holds that no sweep method produces — today, Sigma hits in * the Named-techniques group. * - * A slot rather than a fourteenth entry in `METHODS`: that registry is pinned - * by tests to exactly the thirteen ids `db/analysis_plan.py` plans for and the - * thirteen param sets `api/routers/analysis.py` accepts, and Sigma is neither + * A slot rather than a fifteenth entry in `METHODS`: that registry is pinned + * by tests to exactly the fourteen ids `db/analysis_plan.py` plans for and the + * fourteen param sets `api/routers/analysis.py` accepts, and Sigma is neither * planned nor run through the findings endpoint. */ extraRows?: React.ReactNode; diff --git a/frontend/src/components/analysis/MethodKnobForm.tsx b/frontend/src/components/analysis/MethodKnobForm.tsx index f892993f..4cdebe1a 100644 --- a/frontend/src/components/analysis/MethodKnobForm.tsx +++ b/frontend/src/components/analysis/MethodKnobForm.tsx @@ -131,6 +131,10 @@ export function knobHelp(knob: MethodKnob): string { return "Break a sequence when consecutive events are farther apart than this."; case "partition_field": return "Whose moves are timed: transitions are measured within one value of this field, such as one account. Without it every source is a single stream."; + case "bucket_minutes": + return "How finely the day is cut. One hour tells a nightly job from a daytime one; fifteen minutes tells 02:15 from 02:45."; + case "timezone": + return "The zone the clock is read in, as an IANA name such as Europe/Berlin. Recorded on the run, since the same instant is a different hour elsewhere."; case "field": return "The text field to cluster into templates. Usually the message."; case "order": diff --git a/frontend/src/components/analysis/detector-registry.ts b/frontend/src/components/analysis/detector-registry.ts index 3aec234a..e339f388 100644 --- a/frontend/src/components/analysis/detector-registry.ts +++ b/frontend/src/components/analysis/detector-registry.ts @@ -11,6 +11,7 @@ */ import { Activity, + Clock, Gauge, Hash, Layers, @@ -34,6 +35,7 @@ export type DetectorId = | "drift" | "sequence" | "transition" + | "habit" | "order" | "range" | "charset" @@ -68,6 +70,7 @@ export const DETECTORS: DetectorMeta[] = [ { id: "shift", detector: "proportion_shift", icon: Percent, label: "Proportion shift", hint: "Value shares that change between windows", category: "volume", scoreUnit: "G" }, { id: "interval", detector: "interval_periodicity", icon: Timer, label: "Interval cadence", hint: "Broken heartbeats and new beaconing", category: "volume", scoreUnit: "−log₁₀ p" }, { id: "drift", detector: "value_distribution_drift", icon: Replace, label: "Distribution drift", hint: "Whole-field value-mix changes between windows", category: "volume", scoreUnit: "−log₁₀ p" }, + { id: "habit", detector: "time_of_day", icon: Clock, label: "Time-of-day habit", hint: "Values at an hour they never keep", category: "volume", scoreUnit: "h off habit" }, { id: "order", detector: "timestamp_order", icon: Rewind, label: "Timestamp order", hint: "Timestamps running backwards", category: "volume", scoreUnit: "s skew" }, { id: "sequence", detector: "sequence_novelty", icon: ListOrdered, label: "Event sequences", hint: "Never-seen or rare event orderings (n-grams)", category: "sequences", scoreUnit: "surprise" }, { id: "transition", detector: "transition_time", icon: Gauge, label: "Transition speed", hint: "Value-to-value moves faster than ever seen", category: "sequences", scoreUnit: "1 − obs/ref" }, diff --git a/frontend/src/components/analysis/method-registry.ts b/frontend/src/components/analysis/method-registry.ts index f82f7d9a..45caf50e 100644 --- a/frontend/src/components/analysis/method-registry.ts +++ b/frontend/src/components/analysis/method-registry.ts @@ -21,6 +21,7 @@ */ import { Activity, + Clock, FileText, Gauge, Hash, @@ -48,6 +49,7 @@ export type MethodId = | "timestamp_order" | "sequence_novelty" | "transition_time" + | "time_of_day" | "log_template"; export type EvidenceClass = "named" | "statistical" | "exploration"; @@ -112,7 +114,7 @@ export interface MethodMeta { hint: string; /** * When to configure it, in one sentence for the wizard's card. Starts with - * "Use this when" — a test enforces it — so the thirteen cards read as one list. + * "Use this when" — a test enforces it — so the fourteen cards read as one list. */ useWhen: string; icon: React.ElementType; @@ -354,6 +356,37 @@ export const METHODS: MethodMeta[] = [ querySketch: `SELECT series, gap FROM (\n SELECT AS series,\n dateDiff('second', lagInFrame() OVER w, ) AS gap\n FROM events WHERE case_id = {case}\n WINDOW w AS (PARTITION BY series ORDER BY )\n)\n-- baseline: Poisson-rate G (cadence break) or Greenwood G (new regularity)\n-- self: Greenwood G over retained gaps; Gamma tail over the longest gap`, knobs: [SERIES_KNOB, FDR_KNOB, RATIO_KNOB], }, + { + id: "time_of_day", + label: "Time-of-day habit", + hint: "Values at an hour they never keep", + useWhen: + "Use this when a value keeps office hours or a nightly slot and showing up at another hour would matter — a job that moved, an account active at 03:00.", + icon: Clock, + evidenceClass: "statistical", + costClass: "heavy", + scoreUnit: "h off habit", + what: "Cuts the day into wall-clock buckets in the zone you name and learns, per value, which buckets it habitually occurs in. An occurrence in any other bucket is reported, scored by how many hours it sits from the nearest habitual bucket, around the clock. With a baseline the habit is the baseline window's; without one it is the value's own busy buckets across the timeline, and its thin buckets are judged against them. Cadence measures the gap between arrivals; this measures the hour on the wall.", + querySketch: `SELECT AS value,\n intDiv(toHour(, '') * 60 + toMinute(, ''), {bucket_minutes}) AS bucket,\n count() AS n\nFROM events\nWHERE case_id = {case}\nGROUP BY value, bucket\n-- habit = buckets with n >= {min_bucket_count} in the reference\n-- reported when an occurrence falls outside it; score = hours to the nearest habitual bucket`, + knobs: [ + FIELDS_KNOB, + { + param: "bucket_minutes", + label: "Bucket", + kind: "choice", + placeholder: "60", + options: [ + { value: "60", label: "1 hour" }, + { value: "15", label: "15 minutes" }, + { value: "30", label: "30 minutes" }, + { value: "120", label: "2 hours" }, + { value: "180", label: "3 hours" }, + { value: "240", label: "4 hours" }, + ], + }, + { param: "timezone", label: "Zone", kind: "text", placeholder: "UTC" }, + ], + }, { id: "timestamp_order", label: "Timestamp order", diff --git a/frontend/src/lib/finding-frame.ts b/frontend/src/lib/finding-frame.ts index 6a72e09c..e7be00ab 100644 --- a/frontend/src/lib/finding-frame.ts +++ b/frontend/src/lib/finding-frame.ts @@ -24,11 +24,12 @@ export const TEMPORAL_MODES: ReadonlySet = new Set([ "ngram", "drift", "min-transition", + "habit", ]); /** - * The self modes of the four formerly baseline-only methods, plus the - * transition detector's, which was born with both frames. + * The self modes of the four formerly baseline-only methods, plus those of + * the transition and time-of-day detectors, which were born with both frames. */ export const SELF_MODES: ReadonlySet = new Set([ "self-g-test", @@ -36,6 +37,7 @@ export const SELF_MODES: ReadonlySet = new Set([ "self-cadence", "rare-ngram", "self-min-transition", + "self-habit", ]); /** The two self modes that compare leave-one-out time slices with the rest of the scope. */ diff --git a/frontend/src/lib/finding-normalize.ts b/frontend/src/lib/finding-normalize.ts index 088835ae..5fc093cd 100644 --- a/frontend/src/lib/finding-normalize.ts +++ b/frontend/src/lib/finding-normalize.ts @@ -141,6 +141,13 @@ export function normalizeFinding(meta: DetectorMeta, f: AnomalyFinding, rank: nu })${f.partition_value ? ` · ${truncate(f.partition_value, 30)}` : ""}`; ts = ts ?? f.first_seen; break; + case "time_of_day": + title = pair(f.field, f.value); + subtitle = `×${f.count} at ${f.bucket_label} ${f.timezone} · habit ${f.nearest_habit_label}${ + f.habit_buckets.length > 1 ? ` +${f.habit_buckets.length - 1}` : "" + }${findingMode(f) === "self-habit" ? " across the timeline" : ""}`; + ts = ts ?? f.first_seen; + break; } return { detectorId: meta.id, diff --git a/frontend/src/lib/finding-subject.ts b/frontend/src/lib/finding-subject.ts index 0c3625a9..60e46f82 100644 --- a/frontend/src/lib/finding-subject.ts +++ b/frontend/src/lib/finding-subject.ts @@ -41,6 +41,7 @@ function scoredSubject(f: AnomalyFinding): SubjectPair[] { case "interval_periodicity": case "sequence_novelty": case "sequence_motif": + case "time_of_day": return [{ label: fieldLabel(f.field), value: String(f.value) }]; case "value_combo": return f.fields.map((field, i) => ({ diff --git a/frontend/src/lib/finding-verdict.ts b/frontend/src/lib/finding-verdict.ts index c00b0878..8df2628b 100644 --- a/frontend/src/lib/finding-verdict.ts +++ b/frontend/src/lib/finding-verdict.ts @@ -179,6 +179,21 @@ function scoredVerdict(f: AnomalyFinding): Verdict { tail: `— ${floor}${f.speedup === null ? "" : `, ${f.speedup.toFixed(1)}× faster`}.`, }; } + case "time_of_day": { + const reference = + findingMode(f) === "self-habit" + ? `across the timeline its ${f.baseline_count} occurrences keep to` + : `in the baseline its ${f.baseline_count} occurrences keep to`; + const habit = + f.habit_buckets.length === 1 + ? f.nearest_habit_label + : `${f.habit_buckets.length} buckets, the nearest ${f.nearest_habit_label}`; + return { + lead: `${fieldLabel(f.field)} = ${truncate(String(f.value), 40)} occurs ${f.count} time${f.count === 1 ? "" : "s"} at ${f.bucket_label} (${f.timezone}),`, + highlight: `${f.distance_hours % 1 === 0 ? f.distance_hours.toFixed(0) : f.distance_hours.toFixed(1)} h off its habit`, + tail: `— ${reference} ${habit}.`, + }; + } } } diff --git a/frontend/src/test/methodRegistry.test.ts b/frontend/src/test/methodRegistry.test.ts index 2c68ea55..aac1a90d 100644 --- a/frontend/src/test/methodRegistry.test.ts +++ b/frontend/src/test/methodRegistry.test.ts @@ -48,6 +48,7 @@ describe("method registry", () => { timestamp_order: ["min_skew_seconds"], sequence_novelty: ["series_field", "ngram_size", "max_gap_seconds"], transition_time: ["series_field", "partition_field", "min_ratio"], + time_of_day: ["fields", "bucket_minutes", "timezone"], log_template: ["field", "order", "only_new"], }; for (const m of METHODS) { @@ -57,7 +58,7 @@ describe("method registry", () => { } }); - it("covers exactly the thirteen methods the gate plans for", () => { + it("covers exactly the fourteen methods the gate plans for", () => { // METHOD_IDS in db/analysis_plan.py. A method here that the plan never // reports would render with no status; one there that is missing here // would never be shown at all. @@ -71,6 +72,7 @@ describe("method registry", () => { "numeric_range", "proportion_shift", "sequence_novelty", + "time_of_day", "timestamp_order", "transition_time", "value_combo", diff --git a/src/vestigo/agent/tools.py b/src/vestigo/agent/tools.py index d28a6733..38be7913 100644 --- a/src/vestigo/agent/tools.py +++ b/src/vestigo/agent/tools.py @@ -1968,35 +1968,30 @@ async def run_anomaly_detector( max_gap_seconds: int | None = Field(default=None, ge=1), variant: Literal["shannon", "bigram"] | None = None, partition_field: str | None = None, + bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = None, + timezone: str | None = None, ) -> dict[str, Any]: """Run a statistical anomaly detector over the timeline. - Detectors: value_novelty (rare/first-seen values), value_combo, - frequency (volume spikes/silences), timestamp_order, numeric_range, - charset, entropy, proportion_shift, interval_periodicity, - sequence_novelty, sequence_motif, value_distribution_drift, - transition_time (a value pair reached faster than the pair's learned - floor, e.g. one account on two hosts seconds apart). - `fields` is a comma-separated field list for value detectors (omit to - auto-recommend); `series_field` groups frequency/sequence/transition - detectors; `partition_field` (transition_time) is the stream whose - transitions are timed, e.g. attr:user. - Every detector runs without a `baseline_id` (the timeline is its own - reference); pass one from list_baselines to score suspect windows - against a baseline instead. Optional knobs (server defaults - otherwise): z_threshold (frequency), min_skew_seconds - (timestamp_order), fdr_q (BH false-discovery ceiling), min_ratio - (effect floor), ngram_size (2-5), min_support and start/end - (sequence_motif), group_field (charset: one alphabet per value of - this field, e.g. per host), max_gap_seconds (sequence_novelty/ - sequence_motif: break a sequence at longer gaps), variant (entropy: - "shannon" = character entropy, default; "bigram" = character-pair - surprisal, catches ordinary letters in an unusual order, e.g. a DGA - domain). Returns findings plus a persisted run_id. Each finding - carries an example `event_id`; the result's `fidelity`/`note` say how - much of that event came with it — call get_event for the full record. - The virtual `time:` fields from list_fields are rejected here; they - are for charting and filtering only. + Detectors: value_novelty, value_combo, frequency, timestamp_order, + numeric_range, charset, entropy, proportion_shift, + interval_periodicity, sequence_novelty, sequence_motif, + value_distribution_drift, transition_time (a value pair reached + faster than its learned floor), time_of_day (a value at an hour it + has no habit of). `fields`: comma-separated list for value detectors + (omit to auto-recommend); `series_field`: the field frequency, + sequence and transition detectors group by; `partition_field` + (transition_time): the stream timed, e.g. attr:user. Every detector + runs without `baseline_id` (the timeline is its own reference); pass + one from list_baselines to score suspect windows instead. Knobs + (server defaults otherwise): z_threshold (frequency), + min_skew_seconds (timestamp_order), fdr_q, min_ratio (effect floor), + ngram_size, min_support and start/end (sequence_motif), group_field + (charset: one alphabet per value), max_gap_seconds (sequences), + variant (entropy: shannon or bigram), bucket_minutes and IANA + timezone (time_of_day). Returns findings plus a persisted run_id; + each finding carries an example event_id — call get_event for the + full record. Virtual `time:` fields are rejected here. """ _reject_time_fields(fields, "fields") _reject_time_fields(series_field, "series_field") @@ -2022,6 +2017,8 @@ async def run_anomaly_detector( max_gap_seconds=max_gap_seconds, variant=variant, partition_field=partition_field, + bucket_minutes=bucket_minutes, + timezone=timezone, field_mappings=scope.field_mappings, source_offsets=scope.source_offsets, ) diff --git a/src/vestigo/api/routers/analysis.py b/src/vestigo/api/routers/analysis.py index 2f3a2e21..47368fd5 100644 --- a/src/vestigo/api/routers/analysis.py +++ b/src/vestigo/api/routers/analysis.py @@ -376,6 +376,19 @@ def _empty_is_none(cls, v: Any) -> Any: return None if v == "" else v +class _TimeOfDayParams(_FieldsParams): + #: Wall-clock resolution; each option divides the day. None = server default. + bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = None + #: IANA zone the clock is read in; validated by the runner. None = server default. + timezone: str | None = Field(default=None, min_length=1, max_length=64) + + @field_validator("timezone", mode="before") + @classmethod + def _empty_is_none(cls, v: Any) -> Any: + """A cleared text box and an omitted knob ask the same question.""" + return None if v == "" else v + + class _LogTemplateParams(_Params): #: Not a `_run_stat_detector` detector — log templating is a browser with #: its own service call (see :func:`_run_log_templates`). Routing it through @@ -398,6 +411,7 @@ class _LogTemplateParams(_Params): "timestamp_order": _TimestampOrderParams, "sequence_novelty": _SequenceNoveltyParams, "transition_time": _TransitionTimeParams, + "time_of_day": _TimeOfDayParams, "log_template": _LogTemplateParams, } @@ -729,6 +743,8 @@ async def get_analysis_findings( max_gap_seconds=kwargs.get("max_gap_seconds"), variant=kwargs.get("variant"), partition_field=kwargs.get("partition_field"), + bucket_minutes=kwargs.get("bucket_minutes"), + timezone=kwargs.get("timezone"), # Both come from _resolve_timeline_scope and are not optional # niceties: without field_mappings a canonical field alias is # ignored, and without source_offsets a declared per-source diff --git a/src/vestigo/api/routers/events.py b/src/vestigo/api/routers/events.py index 2dde06d7..4a0f3bde 100644 --- a/src/vestigo/api/routers/events.py +++ b/src/vestigo/api/routers/events.py @@ -40,6 +40,7 @@ DistributionDriftFinding, EntropyFinding, FreqFinding, + HabitFinding, IntervalFinding, MotifFinding, NoveltyFieldInfo, @@ -2254,6 +2255,8 @@ async def _run_stat_detector( max_gap_seconds: int | None = None, variant: str | None = None, partition_field: str | None = None, + bucket_minutes: int | None = None, + timezone: str | None = None, field_mappings: dict[str, list[str]] | None = None, source_offsets: dict[str, int] | None = None, field_overrides: dict[str, bool] | None = _RESOLVE_OVERRIDES, @@ -2583,6 +2586,38 @@ async def _run_stat_detector( raise HTTPException(status_code=422, detail=str(exc)) from exc return result, resolution + if detector == "time_of_day": + # The resolution and the zone decide what "03:40" means, so both are + # resolved here (request, else server default) and snapshotted (D12). + resolution["habit_bucket_minutes"] = ( + bucket_minutes if bucket_minutes is not None else cfg.stat_habit_bucket_minutes + ) + resolution["habit_timezone"] = timezone or cfg.stat_habit_timezone + try: + result = await run_scan( + svc.find_time_of_day_habits, + case_id=case_id, + source_ids=source_ids, + source_offsets=source_offsets, + fields=parsed_fields, + limit=limit, + windows=windows, + bucket_minutes=resolution["habit_bucket_minutes"], + timezone=resolution["habit_timezone"], + min_baseline=cfg.stat_habit_min_baseline, + min_bucket_count=cfg.stat_habit_min_bucket_count, + max_candidates_per_field=cfg.stat_habit_max_candidates_per_field, + exclude_event_ids=exclude_ids, + allowlist=allowlist, + field_mappings=field_mappings, + inventory=inventory, + inventory_total=inventory_total, + field_overrides=field_overrides, + ) + return result, resolution + except ValueError as exc: + raise HTTPException(status_code=422, detail=str(exc)) from exc + if detector == "proportion_shift": # Snapshot the *effective* thresholds (request override or server # default) so the persisted run stays self-describing. @@ -2797,13 +2832,16 @@ def _serialize_finding( | SequenceFinding | MotifFinding | DistributionDriftFinding - | TransitionFinding, + | TransitionFinding + | HabitFinding, ) -> dict[str, Any]: - """Serialise a Value/Freq/Order/Combo/Range/Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition finding to a JSON-safe dict.""" - # Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition finding - # dataclass fields are exactly the wire keys, so asdict() avoids a + """Serialise a Value/Freq/Order/Combo/Range/Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition/Habit finding to a JSON-safe dict.""" + # Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition/Habit + # finding dataclass fields are exactly the wire keys, so asdict() avoids a # hand-maintained field-by-field transcription that would silently drop # any newly added field. + if isinstance(r, HabitFinding): + return {"type": "time_of_day", **asdict(r)} if isinstance(r, TransitionFinding): return {"type": "transition_time", **asdict(r)} if isinstance(r, DistributionDriftFinding): @@ -3192,6 +3230,11 @@ async def _persist_detector_run( # (None = per source) and the learning floor the run used (D15). "partition_field": resolution.get("transition_partition_field"), "min_transitions": resolution.get("transition_min_transitions"), + # time_of_day: the wall-clock resolution and the IANA zone the run + # read the clock in (D12) — without the zone the run is not + # reproducible, which is why it is recorded rather than implied. + "bucket_minutes": resolution.get("habit_bucket_minutes"), + "timezone": resolution.get("habit_timezone"), # sequence_novelty: effective (request-or-default) n-gram length. "ngram_size": resolution.get("sequence_ngram"), # charset: per-identifier scoping (None = one alphabet per field). @@ -3266,7 +3309,7 @@ async def list_anomalies( timeline_id: str, detector: str = Query( default="value_novelty", - description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', or 'transition_time'.", + description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', 'transition_time', or 'time_of_day'.", ), fields: str | None = Query( default=None, @@ -3364,6 +3407,22 @@ async def list_anomalies( "one stream per source." ), ), + bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = Query( + default=None, + description=( + "time_of_day only: width of the wall-clock buckets the day is cut into. " + "Omit to use the server default." + ), + ), + timezone: str | None = Query( + default=None, + min_length=1, + max_length=64, + description=( + "time_of_day only: IANA zone the clock is read in (e.g. 'Europe/Berlin'); " + "stamped into the persisted run. Omit to use the server default." + ), + ), start: datetime | None = Query( default=None, description="sequence_motif only: scope mining to events at/after this time (ISO, UTC).", @@ -3441,6 +3500,11 @@ async def list_anomalies( spacing test, beaconing). Without: whole-scope Greenwood beaconing with pauses excluded, and a robust-Gamma silence test. BH-FDR across the run. + **time_of_day**: per (field, value), learns which wall-clock buckets of + the day (in an explicit IANA `timezone`) the value habitually occurs in + and flags occurrences outside that habit, scored by the hours to the + nearest habitual bucket (D12). + **transition_time**: per ordered value pair of `series_field`, learns the fastest a stream (per source, per `partition_field` value) ever moved from one value to the next and flags transitions faster than that floor by @@ -3488,6 +3552,8 @@ async def list_anomalies( max_gap_seconds=max_gap_seconds, variant=variant, partition_field=partition_field, + bucket_minutes=bucket_minutes, + timezone=timezone, field_mappings=field_mappings, source_offsets=source_offsets, ) @@ -3620,7 +3686,7 @@ class TagAnomaliesRequest(BaseModel): detector: str = Field( default="value_novelty", - description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', or 'transition_time'.", + description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', 'transition_time', or 'time_of_day'.", ) fields: str | None = Field( default=None, @@ -3682,6 +3748,16 @@ class TagAnomaliesRequest(BaseModel): default=None, description="transition_time only: the stream whose transitions are timed (e.g. 'attr:user').", ) + bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = Field( + default=None, + description="time_of_day only: width of the wall-clock buckets the day is cut into.", + ) + timezone: str | None = Field( + default=None, + min_length=1, + max_length=64, + description="time_of_day only: IANA zone the clock is read in; stamped into the run.", + ) start: datetime | None = Field( default=None, description="sequence_motif only: scope mining to events at/after this time.", @@ -3750,6 +3826,8 @@ async def tag_anomalies( max_gap_seconds=body.max_gap_seconds, variant=body.variant, partition_field=body.partition_field, + bucket_minutes=body.bucket_minutes, + timezone=body.timezone, field_mappings=field_mappings, source_offsets=source_offsets, ) diff --git a/src/vestigo/core/config.py b/src/vestigo/core/config.py index 0322d70a..e9522dc2 100644 --- a/src/vestigo/core/config.py +++ b/src/vestigo/core/config.py @@ -196,6 +196,21 @@ class Settings(BaseSettings): # Cap on candidate pairs fetched per source, fastest first; hitting it # carries a warning. stat_transition_max_candidates: int = 2000 + # Time-of-day habit detector (D12): the day is cut into buckets this many + # minutes wide (15, 30, 60, 120, 180 or 240 — each divides the day) in + # stat_habit_timezone, an IANA zone stamped into every run. UTC by default + # because that is the only zone every source agrees on; set the site's + # zone here once so "03:40" means what the analyst reads on the wall. + stat_habit_bucket_minutes: int = Field(default=60, ge=15, le=240) + stat_habit_timezone: str = "UTC" + # A value needs at least this many reference occurrences to have a habit, + # and a bucket needs at least this many of them to be habitual. + stat_habit_min_baseline: int = Field(default=20, ge=2) + stat_habit_min_bucket_count: int = Field(default=3, ge=1) + # Per-field cap on candidate values (highest volume first). Lower than the + # other per-field caps because each value returns one row per occupied + # bucket and window. + stat_habit_max_candidates_per_field: int = 500 # ── Self frame for the slice-based detectors (D18) ────────────────────── # How many equal-width time slices proportion_shift and # value_distribution_drift cut the scope into when no baseline is declared; diff --git a/src/vestigo/core/settings_registry.py b/src/vestigo/core/settings_registry.py index 26002d2c..e2b7fd22 100644 --- a/src/vestigo/core/settings_registry.py +++ b/src/vestigo/core/settings_registry.py @@ -554,6 +554,36 @@ class SettingGroup: "Transition candidate cap", "Candidate value pairs fetched per source, fastest first.", ), + SettingSpec( + "stat_habit_bucket_minutes", + "detectors", + "Time-of-day bucket width", + "Minutes per wall-clock bucket for the time-of-day habit detector: 15, 30, 60, 120, 180 or 240.", + ), + SettingSpec( + "stat_habit_timezone", + "detectors", + "Time-of-day zone", + "IANA zone the time-of-day habit detector reads the clock in (e.g. Europe/Berlin); stamped into every run.", + ), + SettingSpec( + "stat_habit_min_baseline", + "detectors", + "Habit learning floor", + "A value needs at least this many reference occurrences before it has a time-of-day habit.", + ), + SettingSpec( + "stat_habit_min_bucket_count", + "detectors", + "Habitual bucket floor", + "A wall-clock bucket counts as habitual for a value when it holds at least this many reference occurrences.", + ), + SettingSpec( + "stat_habit_max_candidates_per_field", + "detectors", + "Time-of-day candidate cap", + "Candidate values scanned per field for a habit, highest volume first.", + ), SettingSpec( "stat_self_slices", "detectors", diff --git a/src/vestigo/db/analysis_cache.py b/src/vestigo/db/analysis_cache.py index 0099309e..0c56e45c 100644 --- a/src/vestigo/db/analysis_cache.py +++ b/src/vestigo/db/analysis_cache.py @@ -58,7 +58,7 @@ #: for the same inputs: a v3 row holds "up" findings for every short-lived #: value on the timeline, which is the shape this bump exists to retire. #: -#: 5 — the 1.20 detectors (transition_time, D15). New method ids cannot +#: 5 — the 1.20 detectors (transition_time D15, time_of_day D12). New method ids cannot #: collide with an older row's key on their own, but the shared n-gram #: assembly now emits two more columns, and a key that names the runner's #: inputs while the runner's SQL changed underneath it is the case this diff --git a/src/vestigo/db/analysis_plan.py b/src/vestigo/db/analysis_plan.py index 900c68a4..616ed169 100644 --- a/src/vestigo/db/analysis_plan.py +++ b/src/vestigo/db/analysis_plan.py @@ -49,6 +49,7 @@ "timestamp_order", "sequence_novelty", "transition_time", + "time_of_day", "log_template", ) @@ -71,6 +72,7 @@ "proportion_shift", "value_distribution_drift", "interval_periodicity", + "time_of_day", } ) @@ -91,6 +93,7 @@ "interval_periodicity": "heavy", "sequence_novelty": "heavy", "transition_time": "heavy", + "time_of_day": "heavy", "log_template": "heavy", } @@ -344,6 +347,7 @@ def build_plan(inputs: PlanInputs, cfg: Settings) -> list[MethodPlan]: "interval_periodicity", "sequence_novelty", "transition_time", + "time_of_day", ): if frame_needs_baseline: plans[method] = _setup( @@ -418,6 +422,11 @@ def build_plan(inputs: PlanInputs, cfg: Settings) -> list[MethodPlan]: ), ) + # A time-of-day habit needs nothing structural beyond dated events: a + # value's busy buckets are learned from whatever the scope holds, and too + # few occurrences is a per-value floor the run reports, not a gate. + plans.setdefault("time_of_day", _ok("time_of_day")) + # Log templating clusters the `message` materialized column, which is part # of the events schema and therefore always present. There is no data shape # that makes it structurally unable to produce a template, so gating it diff --git a/src/vestigo/db/anomaly_stats.py b/src/vestigo/db/anomaly_stats.py index 90e6beea..5530a18b 100644 --- a/src/vestigo/db/anomaly_stats.py +++ b/src/vestigo/db/anomaly_stats.py @@ -201,6 +201,23 @@ transition anywhere in the scope as the leave-one-out floor. Score = ``1 − observed / reference``. +**time_of_day** (``detector="time_of_day"``) + Per (field, value), learn which wall-clock buckets of the day the value + habitually occurs in and flag occurrences outside that habit, scored by + the circular distance in hours to the nearest habitual bucket. Adapted + from AMiner's ``PathValueTimeIntervalDetector`` (roadmap D12); distinct + from ``interval_periodicity``, which measures inter-arrival gaps — a + backup moved from 02:15 to 03:40 keeps its cadence and breaks its habit. + Buckets are ``bucket_minutes`` wide in an explicit IANA ``timezone`` + (validated, inlined, stamped into every finding — the run is not + reproducible without it). A bucket is habitual when it holds at least + ``min_bucket_count`` reference occurrences; a value needs + ``min_baseline`` of them to have a habit at all. Two frames: *baseline* + (``method="habit"``) learns from the baseline window and scores each + suspect window's occurrences; *self* (``method="self-habit"``, D18) takes + the value's busy buckets across the scope as its habit and scores every + thin bucket against them. + **timestamp_order** (``detector="timestamp_order"``) Flag events whose parsed timestamp jumps *backwards* relative to the previous record in the source file (record order = ``byte_offset``, then @@ -217,6 +234,8 @@ from __future__ import annotations import math +import re +import zoneinfo from collections import defaultdict from collections.abc import Iterable, Mapping, Sequence from concurrent.futures import ThreadPoolExecutor @@ -473,6 +492,40 @@ class _CharsetLearn(NamedTuple): # interval detector's beaconing gate). _MOTIF_GREENWOOD_MIN_INTERVALS = 10 +# Time-of-day habit (D12): the bucket resolutions the day can be cut into. +# Each divides 1440, so the last bucket ends exactly at midnight and the +# circular distance is well defined. +_HABIT_BUCKET_MINUTES = (15, 30, 60, 120, 180, 240) +# What an IANA zone name may look like before zoneinfo is asked about it. The +# name is inlined into SQL (ClickHouse takes a timezone as a constant), so the +# token pattern is the first line of defence and zoneinfo the second. +_TZ_TOKEN = re.compile(r"^[A-Za-z0-9_+\-/]{1,64}$") + + +def _validate_timezone(name: str) -> str: + """Return *name* if it is an IANA zone this host knows; raise ``ValueError`` otherwise.""" + if not isinstance(name, str) or not _TZ_TOKEN.match(name): + raise ValueError("timezone must be an IANA zone name such as 'UTC' or 'Europe/Berlin'") + try: + zoneinfo.ZoneInfo(name) + except (zoneinfo.ZoneInfoNotFoundError, ValueError) as exc: + raise ValueError(f"timezone {name!r} is not a known IANA zone name") from exc + return name + + +def _circular_bucket_distance(a: int, b: int, n: int) -> int: + """Distance between two of *n* buckets around the clock (23:xx is one hour from 00:xx).""" + d = abs(a - b) % n + return min(d, n - d) + + +def _habit_bucket_label(bucket: int, bucket_minutes: int) -> str: + """``"HH:MM–HH:MM"`` for one bucket; the last one ends at ``00:00``.""" + start = bucket * bucket_minutes + end = start + bucket_minutes + return f"{start // 60:02d}:{start % 60:02d}–{(end // 60) % 24:02d}:{end % 60:02d}" + + # Separators used to flatten a value_combo finding's field/value tuples into # the single (field, value) key the detector allowlist stores. The fields are # comma-joined (tokens never contain commas); the values are joined with the @@ -1297,6 +1350,37 @@ class TransitionFinding: details: dict[str, Any] +@dataclass +class HabitFinding: + """One value occurring at a time of day it has no habit of (time_of_day, D12).""" + + field: str + value: str + # The offending wall-clock bucket: index, "HH:MM–HH:MM" label, width and zone. + bucket: int + bucket_label: str + bucket_minutes: int + timezone: str + # Occurrences of the value in this bucket (in the suspect window, or the scope). + count: int + # Reference occurrences of the value the habit was learned from: the + # baseline window's (baseline frame) or the whole scope's (self frame). + baseline_count: int + # The habitual buckets, ascending; the nearest one and its label. + habit_buckets: list[int] + nearest_habit: int + nearest_habit_label: str + # Circular distance to the nearest habitual bucket, in hours. + distance_hours: float + # = distance_hours; used for ranking. + score: float + # First occurrence in the bucket (within the window, or the scope). + first_seen: str | None + event_id: str | None + event: dict[str, Any] | None + details: dict[str, Any] + + @dataclass class MotifFinding: """One recurring event-order n-gram surfaced by the sequence-motif miner.""" @@ -1339,12 +1423,12 @@ class StatAnomalyResult: # "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" # | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" # | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" - # | "transition_time" + # | "transition_time" | "time_of_day" detector: str # "self-baseline" | "temporal" | "z-score" | "temporal-z-score" | "sequential" # | "iqr" | "temporal-range" | "rare-chars" | "temporal-charset" | "temporal-iqr" # | "g-test" | "cadence" | "ngram" | "motif" | "drift" | "min-transition" - # | "self-min-transition" + # | "self-min-transition" | "habit" | "self-habit" method: str baseline_size: int # total events (value_novelty) or event-count used for z-score results: list[ @@ -1361,6 +1445,7 @@ class StatAnomalyResult: | MotifFinding | DistributionDriftFinding | TransitionFinding + | HabitFinding ] = field(default_factory=list) # Effective |z| cutoff used by the frequency detector; None for value_novelty. z_threshold: float | None = None @@ -9024,6 +9109,296 @@ def _transition_finding( details=details, ) + # ------------------------------------------------------------------ + # Time-of-day habit (D12) + # ------------------------------------------------------------------ + + @gated_heavy_scan + def find_time_of_day_habits( + self, + case_id: str, + source_ids: list[str], + fields: list[str] | None = None, + limit: int = 50, + windows: AnalysisWindows | None = None, + bucket_minutes: int = 60, + timezone: str = "UTC", + min_baseline: int = 20, + min_bucket_count: int = 3, + max_candidates_per_field: int = 500, + exclude_event_ids: set[str] | None = None, + allowlist: set[tuple[str, str]] | None = None, + field_mappings: dict[str, list[str]] | None = None, + inventory: list[tuple[str, int, int]] | None = None, + inventory_total: int | None = None, + source_offsets: dict[str, int] | None = None, + field_overrides: dict[str, bool] | None = None, + ) -> StatAnomalyResult: + """Return value occurrences at a time of day the value has no habit of. + + AMiner ``PathValueTimeIntervalDetector`` analog (D12). Per (field, + value) the day is cut into ``1440 / bucket_minutes`` wall-clock + buckets in *timezone* (an IANA name, validated here and stamped into + every finding — without it the run is not reproducible, since the + server's zone can change under it). The value's **habit** is the set + of buckets holding at least *min_bucket_count* reference occurrences, + learned only for values with at least *min_baseline* of them; an + occurrence in any other bucket is flagged, scored by the circular + distance to the nearest habitual bucket in hours. Distinct from + :meth:`find_interval_periodicity`, which measures inter-arrival gaps: + a backup that runs at 02:15 and then at 03:40 keeps its cadence and + breaks its habit. + + Two frames. *Baseline* (``method="habit"``, *windows* given): the + habit is learned from the baseline window and each suspect window's + occurrences are scored against it, one finding per (value, window, + bucket) with the bucket's count. *Self* (``method="self-habit"``, D18): + the habit is the value's own busy buckets across the scope, and every + thin bucket — under *min_bucket_count*, so not habitual by definition — + is scored against them; a value that occurs in one thin bucket is + judged by its busy ones, which is what leave-one-out means here. + + One scan per field: per (value, bucket, window) counts with the first + occurrence, restricted to the *max_candidates_per_field* highest-volume + values (cap → warning). Auto field selection is the novelty + recommender's categorical set, steered by *field_overrides*. The + allowlist key is ``(field, value)``: a value declared Normal is normal + at any hour. + """ + detector = "time_of_day" + if bucket_minutes not in _HABIT_BUCKET_MINUTES: + raise ValueError( + f"bucket_minutes must be one of {', '.join(str(b) for b in _HABIT_BUCKET_MINUTES)}" + ) + tz = _validate_timezone(timezone) + method = "habit" if windows is not None else "self-habit" + self.ch.init_schema() + db = self.ch.database + base_params: dict[str, Any] = {"cid": case_id, "src": source_ids} + eff = effective_ts_sql(source_offsets) + n_buckets = 1440 // bucket_minutes + + total_events = self._count_events(case_id, source_ids) + if total_events == 0: + return StatAnomalyResult( + status="no_data", + detector=detector, + method=method, + baseline_size=0, + windows=windows.payload() if windows is not None else None, + ) + run_warnings: list[str] = [] + if windows is not None: + baseline_size, suspect_totals = self._window_totals( + case_id, source_ids, windows, source_offsets + ) + run_warnings += _window_size_warnings(windows, suspect_totals) + if baseline_size == 0: + return StatAnomalyResult( + status="insufficient_data", + detector=detector, + method=method, + baseline_size=0, + warnings=[*run_warnings, "The baseline window contains no events."], + windows=windows.payload(), + ) + reference_size = baseline_size + else: + reference_size = total_events + + if fields is not None: + scan_fields = fields + else: + rec = self.recommend_novelty_fields( + case_id, + source_ids, + total=inventory_total if inventory is not None else total_events, + field_mappings=field_mappings, + inventory=inventory, + ) + scan_fields = [f.token for f in rec if f.recommended] or _DEFAULT_NOVELTY_FIELDS + scan_fields, override_notes = apply_field_overrides( + scan_fields, field_overrides, [f.token for f in rec] + ) + run_warnings += override_notes + scan_fields = scan_fields[:_MAX_AUTO_SCAN_FIELDS] + + findings: list[HabitFinding] = [] + evaluated_fields = 0 + thin_values = 0 + for field_token in scan_fields: + params: dict[str, Any] = {**base_params} + bind_offset_params(source_offsets, params) + col = _col_expr(field_token, params, field_mappings) + if windows is not None: + bp, sps = _window_preds(windows, params, source_offsets) + w_branches = ", ".join(f"{sp}, {i}" for i, sp in enumerate(sps)) + w_idx_expr = f"multiIf({bp}, -1, {w_branches}, -2)" + scope_pred = " OR ".join([bp, *sps]) + else: + w_idx_expr, scope_pred = "0", "1" + params["bm"] = bucket_minutes + params["cap"] = max_candidates_per_field + where = f""" + WHERE case_id = {{cid:String}} + AND has({{src:Array(String)}}, source_id) + AND {col} != '' + AND {VESTIGO_NOT_SENTINEL_SQL} + AND ({scope_pred}) + """ + # The zone is inlined, not bound: ClickHouse takes a timezone as a + # constant expression, and the name was validated against the + # zoneinfo database and a strict token pattern above. + sql = f""" + SELECT + val, + bucket, + w_idx, + count() AS cnt, + min(ts) AS first_ts, + toString(argMin(event_id, ts)) AS first_evt + FROM ( + SELECT + {col} AS val, + {eff} AS ts, + event_id, + toUInt16(intDiv(toHour(ts, '{tz}') * 60 + toMinute(ts, '{tz}'), {{bm:UInt16}})) AS bucket, + {w_idx_expr} AS w_idx + FROM {db}.events + {where} + ) + WHERE val IN ( + SELECT {col} AS val + FROM {db}.events + {where} + GROUP BY val + ORDER BY count() DESC, val ASC + LIMIT {{cap:UInt32}} + ) + GROUP BY val, bucket, w_idx + {heavy_scan_settings()} + """ + rows = self.ch.client.query(sql, parameters=params).result_rows + if not rows: + continue + # value -> {bucket: reference count}; value -> [(w_idx, bucket, cnt, first_ts, evt)] + reference: dict[str, dict[int, int]] = defaultdict(dict) + observed: dict[str, list[tuple[int, int, int, Any, Any]]] = defaultdict(list) + for val, bucket, w_idx, cnt, first_ts, first_evt in rows: + key, b, w, n = str(val), int(bucket), int(w_idx), int(cnt) + if not key or not 0 <= b < n_buckets: + continue + if windows is None: + reference[key][b] = reference[key].get(b, 0) + n + observed[key].append((0, b, n, first_ts, first_evt)) + elif w == -1: + reference[key][b] = reference[key].get(b, 0) + n + elif w >= 0: + observed[key].append((w, b, n, first_ts, first_evt)) + if len({r[0] for r in rows}) >= max_candidates_per_field: + run_warnings.append( + f"Field {field_token!r} hit the {max_candidates_per_field}-value " + f"candidate cap — only its {max_candidates_per_field} highest-volume " + f"values were scanned for a habit." + ) + learned_any = False + for val, occupancy in reference.items(): + ref_total = sum(occupancy.values()) + habit = sorted(b for b, n in occupancy.items() if n >= min_bucket_count) + if ref_total < min_baseline or not habit: + thin_values += 1 + continue + learned_any = True + for w, b, n, first_ts, evt in observed.get(val, []): + if b in habit: + continue + nearest = min( + habit, + key=lambda h: (_circular_bucket_distance(b, h, n_buckets), h), + ) + distance_hours = ( + _circular_bucket_distance(b, nearest, n_buckets) * bucket_minutes / 60 + ) + first_seen = _present_ts(first_ts) + evt_id = str(evt) if evt else None + details: dict[str, Any] = { + "detector": detector, + "method": method, + "field": field_token, + "value": val, + "bucket": b, + "bucket_label": _habit_bucket_label(b, bucket_minutes), + "bucket_minutes": bucket_minutes, + "timezone": tz, + "count": n, + "habit_buckets": habit, + "habit_labels": [_habit_bucket_label(h, bucket_minutes) for h in habit], + "nearest_habit": nearest, + "nearest_habit_label": _habit_bucket_label(nearest, bucket_minutes), + "distance_hours": distance_hours, + "min_baseline": min_baseline, + "min_bucket_count": min_bucket_count, + "first_seen": first_seen, + "allowlist_field": field_token, + "allowlist_value": val, + } + if windows is not None: + window = windows.suspects[w] + details.update( + { + "baseline_count": ref_total, + "baseline_size": reference_size, + "window_label": window.label, + "window_start": ensure_utc(window.start).isoformat(), + "window_end": ensure_utc(window.end).isoformat(), + } + ) + else: + details["scope_occurrences"] = ref_total + findings.append( + HabitFinding( + field=field_token, + value=val, + bucket=b, + bucket_label=details["bucket_label"], + bucket_minutes=bucket_minutes, + timezone=tz, + count=n, + baseline_count=ref_total, + habit_buckets=habit, + nearest_habit=nearest, + nearest_habit_label=details["nearest_habit_label"], + distance_hours=distance_hours, + score=distance_hours, + first_seen=first_seen, + event_id=evt_id, + event=_stub_event(evt_id, case_id, first_seen), + details=details, + ) + ) + if learned_any: + evaluated_fields += 1 + if thin_values: + run_warnings.append( + f"{thin_values} value{'s' if thin_values != 1 else ''} skipped: fewer than " + f"{min_baseline} reference occurrences, or none in any bucket at least " + f"{min_bucket_count} times — too thin to learn a daily habit from." + ) + return self._finalize_findings( + findings, + detector=detector, + method=method, + total_events=reference_size, + evaluated_fields=evaluated_fields, + exclude_event_ids=exclude_event_ids, + limit=limit, + case_id=case_id, + source_ids=source_ids, + allowlist=allowlist, + warnings=run_warnings, + windows=windows, + ) + @gated_heavy_scan def find_sequence_motifs( self, diff --git a/src/vestigo/demo/sources/linux.py b/src/vestigo/demo/sources/linux.py index db141514..661a80db 100644 --- a/src/vestigo/demo/sources/linux.py +++ b/src/vestigo/demo/sources/linux.py @@ -7,7 +7,8 @@ * ``APP-01`` drifts against NTP, emitting records slightly out of order. It is benign, and being able to see that quickly is the point. * the nightly backup on ``BACKUP-01`` moves from 02:15 to 03:40 mid-month — - also benign, and a useful counterweight to the malicious beacon. + also benign, and a useful counterweight to the malicious beacon — and runs + once by hand on a baseline afternoon, the program's one daytime appearance. * sudo on the file server shifts toward archiving commands as the intruder stages data. @@ -183,6 +184,29 @@ def _backup_job() -> Iterator[dict[str, str]]: f"bytes={r.randrange(10**9, 9 * 10**9)}", ) day += timedelta(days=1) + # One manual run in the middle of a baseline afternoon — an administrator + # taking a copy before a migration. Benign, and the one time the backup + # program is seen twelve hours from its habit (§15's time-of-day detector + # in the self frame, where the nightly slot is the program's own habit). + manual = scenario.SCENARIO_START + timedelta(days=9, hours=15, minutes=2, seconds=17) + pid = r.randrange(400, 65_000) + yield _row( + manual, + scenario.BACKUP_HOST, + "backup", + pid, + "svc_backup", + "manual backup run started target=/srv/shares retention=7d requested_by=a.lindqvist", + ) + yield _row( + manual + timedelta(minutes=19), + scenario.BACKUP_HOST, + "backup", + pid, + "svc_backup", + f"manual backup run completed files={r.randrange(9000, 41000)} " + f"bytes={r.randrange(10**9, 9 * 10**9)}", + ) def _intrusion() -> Iterator[dict[str, str]]: diff --git a/src/vestigo/demo/sources/windows.py b/src/vestigo/demo/sources/windows.py index 0c864b10..a4f1ce01 100644 --- a/src/vestigo/demo/sources/windows.py +++ b/src/vestigo/demo/sources/windows.py @@ -306,7 +306,9 @@ def _lateral() -> Iterator[dict[str, str]]: """Movement onto hosts the contractor has never touched, then staging.""" r = scenario.rng("windows-lateral") phase = scenario.PHASES[2] - new_hosts = (scenario.FILE_SERVER, "WKS-007", "WKS-009", scenario.JUMP_HOST) + # The jump host first, at three in the morning: its baseline logons are an + # administrator's office hours, so the hour itself is a finding (§16). + new_hosts = (scenario.JUMP_HOST, scenario.FILE_SERVER, "WKS-007", "WKS-009") moment = phase.start + timedelta(hours=3) for host in new_hosts: for _ in range(r.randint(3, 7)): diff --git a/tests/test_analysis_plan.py b/tests/test_analysis_plan.py index f7c36418..71ab7ead 100644 --- a/tests/test_analysis_plan.py +++ b/tests/test_analysis_plan.py @@ -74,9 +74,10 @@ def test_numeric_range_gated_off_without_numeric_fields(cfg): "value_distribution_drift", "interval_periodicity", "sequence_novelty", - # Born with both frames (D15), gated the same way: a baseline comparison - # with no baseline is the one thing an analyst action repairs. + # Born with both frames (D15, D12), gated the same way: a baseline + # comparison with no baseline is the one thing an analyst action repairs. "transition_time", + "time_of_day", ) diff --git a/tests/test_anomaly_stats.py b/tests/test_anomaly_stats.py index d6bd9a50..cc7b42f7 100644 --- a/tests/test_anomaly_stats.py +++ b/tests/test_anomaly_stats.py @@ -6825,3 +6825,278 @@ def test_transition_allowlist_suppresses_the_pair_in_both_frames(): assert result.status == "ok" assert result.results == [] assert result.total_findings == 0 + + +# --------------------------------------------------------------------------- +# time_of_day — detector (D12) +# --------------------------------------------------------------------------- + +# Baseline frame query order: count, window totals, then one occupancy scan +# per field. Occupancy row layout: val, bucket, w_idx, cnt, first_ts, +# first_evt — one row per (value, time-of-day bucket, window), baseline rows +# under w_idx = -1. Self frame: count, then one scan per field with every row +# under w_idx = 0. + +_HABIT_COLS = ["val", "bucket", "w_idx", "cnt", "first_ts", "first_evt"] + + +def _habit_responses( + total: int, window_totals: tuple[int, int], rows: list[tuple] +) -> list[FakeQueryResult]: + return [ + FakeQueryResult(result_rows=[(total,)], column_names=["count()"]), + FakeQueryResult(result_rows=[window_totals], column_names=["bl_total", "w0_total"]), + FakeQueryResult(result_rows=rows, column_names=_HABIT_COLS), + ] + + +def _habit_baseline_rows(val: str, buckets: dict[int, int]) -> list[tuple]: + return [(val, b, -1, n, None, None) for b, n in buckets.items()] + + +def test_habit_parameter_validation(): + svc = _svc([]) + with pytest.raises(ValueError, match="bucket_minutes"): + svc.find_time_of_day_habits("c1", ["s1"], fields=["attr:program"], bucket_minutes=7) + with pytest.raises(ValueError, match="timezone"): + svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], timezone="Mars/Olympus_Mons" + ) + with pytest.raises(ValueError, match="timezone"): + svc.find_time_of_day_habits("c1", ["s1"], fields=["attr:program"], timezone="UTC'; --") + assert svc.ch.client._calls == [] + + +def test_habit_no_data(): + svc = _svc([FakeQueryResult(result_rows=[(0,)], column_names=["count()"])]) + result = svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], windows=_seq_windows() + ) + assert result.status == "no_data" + assert result.detector == "time_of_day" + + +def test_habit_baseline_flags_an_occurrence_outside_the_learned_hours(): + at = datetime(2024, 1, 17, 3, 41, tzinfo=UTC) + rows = [ + # The nightly backup: 48 baseline runs, all in the 02:00 bucket. + *_habit_baseline_rows("backup", {2: 48}), + # Suspect window: eight runs at 03:xx and two at 04:xx. + ("backup", 3, 0, 8, at, "evt-3"), + ("backup", 4, 0, 2, at + timedelta(hours=1), "evt-4"), + # Still at 02:xx in the suspect window — inside the habit, no finding. + ("backup", 2, 0, 2, at, "evt-2"), + ] + svc = _svc(_habit_responses(10_000, (8000, 2000), rows)) + result = svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], windows=_seq_windows() + ) + assert result.status == "ok" + assert result.method == "habit" + assert result.baseline_size == 8000 + by_bucket = {r.bucket: r for r in result.results} + assert sorted(by_bucket) == [3, 4] + r = by_bucket[4] + assert r.field == "attr:program" + assert r.value == "backup" + assert r.count == 2 + assert r.baseline_count == 48 + assert r.bucket_minutes == 60 + assert r.timezone == "UTC" + assert r.bucket_label == "04:00–05:00" + assert r.habit_buckets == [2] + assert r.nearest_habit_label == "02:00–03:00" + assert r.distance_hours == 2.0 + assert r.score == 2.0 + assert r.event_id == "evt-4" + assert r.first_seen is not None and r.first_seen.startswith("2024-01-17T04:41") + assert r.details["window_label"] == "incident" + assert r.details["allowlist_field"] == "attr:program" + assert r.details["allowlist_value"] == "backup" + # Ranked farthest-from-habit first. + assert [f.bucket for f in result.results] == [4, 3] + assert by_bucket[3].distance_hours == 1.0 + assert len(svc.ch.client._calls) == 3 + + +def test_habit_distance_is_circular_and_buckets_follow_the_resolution(): + """23:xx is one hour from a 00:xx habit, not twenty-three; and a 30-minute + resolution labels half-hour buckets and measures in half hours.""" + at = datetime(2024, 1, 17, 23, 10, tzinfo=UTC) + rows = [ + *_habit_baseline_rows("cron", {0: 30, 1: 25}), + ("cron", 47, 0, 1, at, "e1"), # 23:30–00:00 at 30-min buckets + ("cron", 24, 0, 1, at, "e2"), # 12:00–12:30 + ] + svc = _svc(_habit_responses(10_000, (8000, 2000), rows)) + result = svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], windows=_seq_windows(), bucket_minutes=30 + ) + assert result.status == "ok" + by_bucket = {r.bucket: r for r in result.results} + assert by_bucket[47].bucket_label == "23:30–00:00" + assert by_bucket[47].distance_hours == 0.5 + assert by_bucket[47].nearest_habit_label == "00:00–00:30" + assert by_bucket[24].distance_hours == 11.5 + assert by_bucket[24].habit_buckets == [0, 1] + + +def test_habit_learning_floors_skip_thin_values_and_thin_buckets(): + at = datetime(2024, 1, 17, tzinfo=UTC) + rows = [ + # Too few baseline occurrences to call anything a habit (floor 20). + *_habit_baseline_rows("rare", {9: 10}), + ("rare", 22, 0, 1, at, "e1"), + # A bucket with fewer than min_bucket_count baseline hits is not habit: + # 09:xx is habitual, 21:xx (2 hits) is not, so a 21:xx occurrence is a + # finding measured from 09:xx — and so is the 22:xx one. + *_habit_baseline_rows("job", {9: 40, 21: 2}), + ("job", 21, 0, 3, at, "e2"), + ("job", 22, 0, 1, at, "e3"), + # Every bucket habitual: nothing can be outside. + *_habit_baseline_rows("chatty", dict.fromkeys(range(24), 5)), + ("chatty", 3, 0, 1, at, "e4"), + ] + svc = _svc(_habit_responses(10_000, (8000, 2000), rows)) + result = svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], windows=_seq_windows() + ) + assert result.status == "ok" + assert sorted((r.value, r.bucket) for r in result.results) == [("job", 21), ("job", 22)] + assert {r.distance_hours for r in result.results} == {12.0, 11.0} + assert any("fewer than 20" in w for w in result.warnings) + + +def test_habit_baseline_without_learnable_values_is_insufficient(): + at = datetime(2024, 1, 17, tzinfo=UTC) + rows = [*_habit_baseline_rows("rare", {9: 3}), ("rare", 22, 0, 1, at, "e1")] + svc = _svc(_habit_responses(10_000, (8000, 2000), rows)) + result = svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], windows=_seq_windows() + ) + assert result.status == "insufficient_data" + + +def test_habit_sql_shape_binds_resolution_and_inlines_a_validated_timezone(): + at = datetime(2024, 1, 17, tzinfo=UTC) + client = RecordingClient( + _habit_responses( + 10_000, + (8000, 2000), + # Two-hour buckets: 02:00–04:00 is the habit, 04:00–06:00 the finding. + [*_habit_baseline_rows("backup", {1: 48}), ("backup", 2, 0, 1, at, "e1")], + ) + ) + svc = StatisticalAnomalyService.__new__(StatisticalAnomalyService) + svc.ch = FakeClickHouseStore(client) + result = svc.find_time_of_day_habits( + "c1", + ["s1"], + fields=["attr:program"], + windows=_seq_windows(), + timezone="Europe/Berlin", + bucket_minutes=120, + max_candidates_per_field=7, + ) + assert result.status == "ok" + sql = client.full_queries[2] + # The wall-clock minute in the analyst's zone, cut into the resolution. + assert "toHour(ts, 'Europe/Berlin') * 60 + toMinute(ts, 'Europe/Berlin')" in sql + assert "intDiv(" in sql and "{bm:UInt16}" in sql + assert "GROUP BY val, bucket, w_idx" in sql + # Candidate values are the highest-volume ones, capped. + assert "LIMIT {cap:UInt32}" in sql + params = client._all_parameters[2] + assert params["bm"] == 120 + assert params["cap"] == 7 + assert params["fk"] == "program" + assert params["b0"] == "2024-01-01 00:00:00.000" + r = result.results[0] + assert r.timezone == "Europe/Berlin" + assert r.bucket_minutes == 120 + assert r.bucket_label == "04:00–06:00" + assert r.details["timezone"] == "Europe/Berlin" + + +def test_habit_self_frame_measures_each_value_against_its_own_hours(): + """Without a baseline the habit is the value's own busy buckets across the + scope, and a thin bucket away from them is the finding.""" + at = datetime(2024, 1, 17, 15, 2, tzinfo=UTC) + rows = [ + ("backup", 2, 0, 50, None, None), + ("backup", 3, 0, 8, None, None), + # One manual run mid-afternoon: 12 h from the nearest habitual bucket. + ("backup", 15, 0, 1, at, "e1"), + # Two runs spilling past 04:00 — thin, one hour from 03:xx. + ("backup", 4, 0, 2, at, "e2"), + # A value below the learning floor is skipped. + ("rare", 9, 0, 10, None, None), + ("rare", 22, 0, 1, at, "e3"), + ] + svc = _svc( + [ + FakeQueryResult(result_rows=[(10_000,)], column_names=["count()"]), + FakeQueryResult(result_rows=rows, column_names=_HABIT_COLS), + ] + ) + result = svc.find_time_of_day_habits("c1", ["s1"], fields=["attr:program"]) + assert result.status == "ok" + assert result.method == "self-habit" + assert result.windows is None + assert [(r.value, r.bucket, r.distance_hours) for r in result.results] == [ + ("backup", 15, 11.0), + ("backup", 4, 1.0), + ] + r = result.results[0] + assert r.habit_buckets == [2, 3] + assert r.count == 1 + assert r.baseline_count == 61 + assert r.details["scope_occurrences"] == 61 + assert "window_label" not in r.details + assert len(svc.ch.client._calls) == 2 + + +def test_habit_auto_fields_run_through_the_recommender_and_overrides(): + """Auto mode scans the recommended categorical fields, minus any the + timeline declared off, and says so.""" + at = datetime(2024, 1, 17, tzinfo=UTC) + inventory = [("attr:host", 8, 950), ("attr:program", 12, 900), ("attr:pid", 950, 1000)] + svc = _svc( + [ + FakeQueryResult(result_rows=[(1000,)], column_names=["count()"]), + FakeQueryResult( + result_rows=[("backup", 2, 0, 48, None, None), ("backup", 14, 0, 1, at, "e1")], + column_names=_HABIT_COLS, + ), + ] + ) + result = svc.find_time_of_day_habits( + "c1", + ["s1"], + inventory=inventory, + inventory_total=1000, + field_overrides={"attr:host": False}, + ) + assert result.status == "ok" + # One scan for the one field left after the override; pid is an identifier. + assert len(svc.ch.client._calls) == 2 + assert svc.ch.client._all_parameters[1]["fk"] == "program" + assert any("attr:host" in w for w in result.warnings) + + +def test_habit_allowlist_suppresses_the_value(): + at = datetime(2024, 1, 17, tzinfo=UTC) + svc = _svc( + [ + FakeQueryResult(result_rows=[(1000,)], column_names=["count()"]), + FakeQueryResult( + result_rows=[("backup", 2, 0, 48, None, None), ("backup", 14, 0, 1, at, "e1")], + column_names=_HABIT_COLS, + ), + ] + ) + result = svc.find_time_of_day_habits( + "c1", ["s1"], fields=["attr:program"], allowlist={("attr:program", "backup")} + ) + assert result.status == "ok" + assert result.results == [] diff --git a/tests/test_demo_detector_coverage_clickhouse.py b/tests/test_demo_detector_coverage_clickhouse.py index d0aeb872..894944da 100644 --- a/tests/test_demo_detector_coverage_clickhouse.py +++ b/tests/test_demo_detector_coverage_clickhouse.py @@ -34,6 +34,11 @@ # Host-to-host moves timed per account: the contractor reaches hosts in # minutes that the baseline population moved between over hours. "find_transition_times": {"series_field": "attr:computer_name", "partition_field": "attr:user"}, + # The nightly backup program keeps a 02:xx slot and moves to 03:40; the + # jump host keeps an administrator's office hours and sees the contractor + # at 03:00. Named rather than auto-picked so the assertion is about the + # detector, not the recommender's field order on this corpus. + "find_time_of_day_habits": {"fields": ["attr:program", "attr:computer_name"]}, } #: Detectors that score a baseline against suspect windows. @@ -49,6 +54,7 @@ "find_distribution_drift", "find_sequence_novelty", "find_transition_times", + "find_time_of_day_habits", ) #: Detectors with no baseline/suspect split at all. @@ -115,6 +121,7 @@ def test_windowed_detector_finds_something(demo, ch_store, method): "find_distribution_drift": {"fields": ["attr:bytes_out"]}, "find_sequence_novelty": {"series_field": "attr:computer_name"}, "find_transition_times": {"series_field": "attr:computer_name", "partition_field": "attr:user"}, + "find_time_of_day_habits": {"fields": ["attr:program"]}, } diff --git a/tests/test_demo_generator.py b/tests/test_demo_generator.py index e12e30eb..c7870207 100644 --- a/tests/test_demo_generator.py +++ b/tests/test_demo_generator.py @@ -131,13 +131,22 @@ def test_backup_cadence_shifts_after_the_foothold(): runs = [ _dt.fromisoformat(r["timestamp"]) for r in linux.linux_rows() - if r["hostname"] == scenario.BACKUP_HOST and "backup run started" in r["message"] + if r["hostname"] == scenario.BACKUP_HOST and "nightly backup run started" in r["message"] ] early = [r for r in runs if r < scenario.PHASES[1].start] late = [r for r in runs if r >= scenario.PHASES[1].start] assert early and late assert {r.hour for r in early} == {2} assert {r.hour for r in late} == {3} + # The one manual run is a baseline-afternoon event, twelve hours from the + # nightly slot: the time-of-day detector's self-frame signal. + manual = [ + _dt.fromisoformat(r["timestamp"]) + for r in linux.linux_rows() + if r["hostname"] == scenario.BACKUP_HOST and "manual backup run started" in r["message"] + ] + assert len(manual) == 1 + assert manual[0] < scenario.BASELINE_END and manual[0].hour == 15 def test_file_server_sudo_mix_shifts_during_the_intrusion(): From 1d113290e91820fe417785bcaaa0f24d2d5f2619 Mon Sep 17 00:00:00 2001 From: Overcuriousity Date: Wed, 16 Sep 2026 14:09:49 +0000 Subject: [PATCH 3/5] =?UTF-8?q?feat(detectors):=20value=20correlation=20?= =?UTF-8?q?=E2=80=94=20field-to-field=20rules=20that=20break=20(D13)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A fifteenth statistical detector, value_correlation, adapted from AMiner's VariableCorrelationDetector and intra-record: for a field pair it mines implication rules A = x ⇒ B = y — an antecedent value with at least stat_correlation_min_support reference events whose dominant consequent accounts for at least stat_correlation_rule_confidence of them, both directions — and reports a window in which the rule's violation rate rises: a 2×2 G-test of conforming against violating events between the reference and the window, one Benjamini–Hochberg pool per run, an effect floor of stat_correlation_min_ratio on the violation-rate ratio. Only rises are reported; a rule that appears is a proportion shift. Both frames: rule-g-test mines from the baseline window and tests each suspect window; self-rule-g-test mines from the scope and tests each leave-one-out slice against the rest. Pairs come from an explicit field list or the recommender's top stat_correlation_auto_fields categorical fields, capped at stat_correlation_max_pairs with a warning; each pair is one GROUP BY a, b scan capped at stat_correlation_max_rows_per_pair rows. Findings carry the rule as mined, both sides' counts and violation rates, and the consequent value that most often took the rule's place; the allowlist key is the combo one. Gate entry (two categorical fields; a sliceable span in the self frame; needs_setup in the baseline frame without a baseline), params model, /anomalies and agent knobs (rule_confidence), run snapshot, seven settings with registry specs, the method card, ValueCorrelationFinding, a two-bar evidence figure, and ANOMALY_DETECTION.md §17. The agent tool docstring is trimmed again to keep the schema under budget (43,845 over 35 tools). The demo case needed no new signal: the contractor's user ⇒ home workstation rule holds over three weeks and breaks on every host the intrusion visits, asserted in both frames. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 21 + CLAUDE.md | 4 +- README.md | 4 +- docs/AGENT.md | 4 +- docs/ANOMALY_DETECTION.md | 116 ++++- docs/PROGRESS.md | 40 +- docs/ROADMAP.md | 19 +- frontend/src/api/anomalies.ts | 6 +- frontend/src/api/types.ts | 62 ++- .../components/analysis/FindingEvidence.tsx | 18 + .../src/components/analysis/FindingGroup.tsx | 6 +- .../components/analysis/MethodKnobForm.tsx | 4 + .../components/analysis/detector-registry.ts | 3 + .../components/analysis/method-registry.ts | 24 +- frontend/src/lib/finding-frame.ts | 10 +- frontend/src/lib/finding-normalize.ts | 7 + frontend/src/lib/finding-subject.ts | 1 + frontend/src/lib/finding-verdict.ts | 11 + frontend/src/test/methodRegistry.test.ts | 4 +- src/vestigo/agent/tools.py | 32 +- src/vestigo/api/routers/analysis.py | 10 + src/vestigo/api/routers/events.py | 91 +++- src/vestigo/core/config.py | 14 + src/vestigo/core/settings_registry.py | 42 ++ src/vestigo/db/analysis_cache.py | 2 +- src/vestigo/db/analysis_plan.py | 18 +- src/vestigo/db/anomaly_stats.py | 474 +++++++++++++++++- tests/test_analysis_plan.py | 3 +- tests/test_anomaly_stats.py | 251 ++++++++++ .../test_demo_detector_coverage_clickhouse.py | 6 + 30 files changed, 1244 insertions(+), 63 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ba8e91ee..012b62db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,27 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **Value correlation (D13).** A fifteenth statistical detector, `value_correlation`, + adapted from AMiner's `VariableCorrelationDetector` and intra-record: for a field pair it + mines implication rules `A = x ⇒ B = y` — an antecedent value with at least + `stat_correlation_min_support` reference events whose dominant consequent accounts for at + least `stat_correlation_rule_confidence` of them, both directions — and reports a window in + which the rule's violation rate rises: a 2×2 G-test of conforming against violating events + between the reference and the window, one Benjamini–Hochberg pool per run, an effect floor + of `stat_correlation_min_ratio` on the violation-rate ratio. Only rises are reported; a rule + that appears is a proportion shift. Both frames: `rule-g-test` mines from the baseline + window and tests each suspect window; `self-rule-g-test` mines from the timeline and tests + each leave-one-out slice against the rest. Pairs come from an explicit field list or the + recommender's top `stat_correlation_auto_fields` categorical fields, capped at + `stat_correlation_max_pairs` with a warning; each pair is one `GROUP BY a, b` scan capped + at `stat_correlation_max_rows_per_pair` rows. Findings carry the rule as mined, both + sides' counts and violation rates, and the consequent value that most often took the + rule's place; the allowlist key is the combo one. Gate entry (two categorical fields, a + sliceable span in the self frame), wizard card, evidence figure, agent knob + (`rule_confidence`), seven settings with registry specs, run snapshot and + `docs/ANOMALY_DETECTION.md` §17. The demo case asserts the contractor's + `user ⇒ home workstation` rule breaking on every host the intrusion visits, in both frames. + - **Time-of-day habit (D12).** A fourteenth statistical detector, `time_of_day`, adapted from AMiner's `PathValueTimeIntervalDetector`: per (field, value) it cuts the day into `bucket_minutes`-wide wall-clock buckets (15–240 minutes, default 60) read in an explicit diff --git a/CLAUDE.md b/CLAUDE.md index 52e65ce1..7b9465cc 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -26,7 +26,7 @@ trail (`api/routers/auth.py`, `admin.py`, `deps.py`). - `CONCEPT.md` / `MODEL_REFINEMENT.md` — product vision and the Case/Source/Timeline/Event/ Artifact data model. Read before touching the model; rarely changes. - `TECH_STACK.md` — backing-service decision record (*why*, not *what's shipped*). -- `ANOMALY_DETECTION.md` — reference for all sixteen analysis tools actually running +- `ANOMALY_DETECTION.md` — reference for all seventeen analysis tools actually running (statistical detectors, Sigma runner, log templates, semantic similarity), plus the baseline/disposition model. Update alongside any detector change in the same commit. - `AGENT.md` — the optional AI investigation agent (design invariants, MCP tools, provider @@ -253,7 +253,7 @@ instead of rebuilding. does not strand its verdict bar a screen below the claim; `ToolsSheet` is four tabs (Scope, Methods, Signatures, Explore) rather than one scroll, so a thousand-row template list cannot bury the baseline picker; `method-registry.ts` is the - single description of all fourteen methods, including the prose that used to live in a + single description of all fifteen methods, including the prose that used to live in a Method tab and each method's optional `railFloor`, a presentation-only bar on the ranked feed whose held-back count is always disclosed; the sheet's method mode runs a method with the analyst's own knob values, which is what keeps the analysis diff --git a/README.md b/README.md index 2461fdd8..f4963ede 100644 --- a/README.md +++ b/README.md @@ -69,7 +69,7 @@ resolved on the machine you deploy to. ClickHouse, time histogram with anomaly overlays, keyset pagination with jump-to-time, tag/comment annotations with bulk apply, saved views, and streaming CSV/JSONL export that keeps the forensic columns. -- **Anomaly detection** — sixteen analysis tools: fourteen statistical detectors over +- **Anomaly detection** — seventeen analysis tools: fifteen statistical detectors over ClickHouse needing no embeddings, a Sigma rule runner, and semantic similarity search over local embeddings. Each is documented method by method, scores against explicit baseline-vs-suspect windows, and yields findings whose confirm/dismiss disposition @@ -109,7 +109,7 @@ feel like, and the Case/Timeline model here is descended from it. That is the co invite, and three axes are where we think we are already the better place to run an investigation: -- **Detection is the workflow, not an add-on** — sixteen analysis tools in the box, each +- **Detection is the workflow, not an add-on** — seventeen analysis tools in the box, each scoring against an analyst-declared baseline and carrying a verdict that survives re-scans, so triage accumulates instead of being redone. - **Provenance goes all the way down** — not just "this file was imported": a finding is diff --git a/docs/AGENT.md b/docs/AGENT.md index a59a4c1e..7f35e432 100644 --- a/docs/AGENT.md +++ b/docs/AGENT.md @@ -720,7 +720,9 @@ The 1.20 detectors (2026-09-16: `transition_time`'s `partition_field`, `time_of_ `bucket_minutes` and `timezone`) took the first draft to 44,421 and the rule was applied again: the docstring now names each detector once and each knob in a few words, and the `timezone` length constraints came off the schema (the runner validates the zone anyway) — -**43,832 over 35 tools**, ceiling unchanged, below the D11 figure. +**43,832 over 35 tools**, ceiling unchanged, below the D11 figure. `value_correlation`'s +`rule_confidence` (D13) pushed it over once more; the same treatment (no schema-level +bounds, a terser docstring) lands it at **43,845 over 35 tools**. Detector findings additionally reduce their inline example event in the **model's copy** to `event_id` + truncated `message` diff --git a/docs/ANOMALY_DETECTION.md b/docs/ANOMALY_DETECTION.md index 58723528..84432aba 100644 --- a/docs/ANOMALY_DETECTION.md +++ b/docs/ANOMALY_DETECTION.md @@ -7,7 +7,7 @@ This document covers every detector actually running in the codebase today. If a detector described here changes (formula, default, field name), update this file and the "Method" tab copy in the same commit. -There are sixteen independent analysis tools in Vestigo: +There are seventeen independent analysis tools in Vestigo: 1. [Value novelty](#1-value-novelty-rare--first-seen-values) — rare/new field values, single field or [combinations](#value-combinations-the-value_combo-variant) (ClickHouse, no ML) 2. [Frequency anomalies](#2-frequency-anomalies-volume-spikes--silences) — volume spikes/silences (ClickHouse, no ML) @@ -25,13 +25,14 @@ There are sixteen independent analysis tools in Vestigo: 14. [Log templates](#14-log-templates-structural-line-clustering) — structural clustering of raw lines into templates, so rare *shapes* surface without naming a field (ClickHouse, no ML) 15. [Transition speed](#15-transition-speed-value-to-value-moves-faster-than-ever-seen) — a stream reaching the next value of a field faster than that pair was ever reached before (ClickHouse, no ML) 16. [Time-of-day habit](#16-time-of-day-habit-values-at-an-hour-they-never-keep) — a value occurring at a wall-clock hour it has no habit of, in an explicit timezone (ClickHouse, no ML) +17. [Value correlation](#17-value-correlation-field-to-field-rules-that-break) — an implication rule between two fields of the same event (`A = x ⇒ B = y`) whose violation rate rises in a window (ClickHouse + a real significance test, no ML) All but the eleventh are **statistical or rule-based**: pure counting, arithmetic and predicate matching over already-ingested events — no machine learning, no network calls, working the instant ingestion finishes. The eleventh needs an explicit embedding step first. -Code: `src/vestigo/db/anomaly_stats.py` (detectors 1–10, 12, 14–16), +Code: `src/vestigo/db/anomaly_stats.py` (detectors 1–10, 12, 14–17), `src/vestigo/db/similarity.py` (detector 11), `src/vestigo/sigma/` (detector 13). UI: `frontend/src/components/analysis/`. @@ -65,6 +66,7 @@ teaches the tooling instead of the work. What each tool has to find: | 14 | Log templates | ~40 syslog templates, with an `unattended-upgrade` shape that appears only in the suspect window. | | 15 | Transition speed | The contractor account on `FILE-01` two seconds after its wmic call from `JUMP-01`, timed per account over `attr:computer_name`, against an administrator whose routine jump-host hop takes the same pair 20–90 s. | | 16 | Time-of-day habit | The nightly backup program at 03:xx and 04:xx after three weeks of 02:xx runs (benign, the same move interval cadence sees), and the contractor on `JUMP-01` at 03:00, a host whose baseline logons are an administrator's office hours. Without a baseline, the one manual afternoon backup run twelve hours from the program's nightly slot. | +| 17 | Value correlation | The rule `user = m.okonkwo ⇒ computer_name = WKS-004`, which holds over three weeks of the contractor's logons and breaks on every host the intrusion takes the account to — the jump host, the file server, two finance workstations. Without a baseline, the same rule against the rest of the timeline, slice by slice. | `tests/test_demo_detector_coverage_clickhouse.py` asserts that each of these actually returns findings. If a retuned threshold silences one of them, that @@ -356,14 +358,14 @@ checked, and the UI must never present the two the same way. | Method | Precondition | Setting | |---|---|---| | `value_novelty`, `timestamp_order`, `log_template`, `entropy`, `time_of_day` | none — always offered | — | -| `value_combo` | ≥2 categorical fields | — | +| `value_combo`, `value_correlation` | ≥2 categorical fields | — | | `numeric_range` | ≥1 field whose sampled values are ≥90 % numeric | `analysis_gate_min_numeric_ratio` | | `charset` | ≥1 field above the enum-like ceiling | `analysis_gate_max_enum_distinct` | | `frequency` | span of at least the minimum number of seconds | `analysis_gate_min_frequency_buckets` | | `interval_periodicity` | enough events per series value to fit a cadence | `analysis_gate_min_interval_periods` | | `sequence_novelty`, `transition_time` | a series field with ≥2 distinct values (one value yields no ordering and no transition) | `analysis_gate_min_series_distinct` | -| `proportion_shift`, `value_distribution_drift` | a span of more than one instant (self frame: it is cut into slices) | `stat_self_slices` | -| any of those five, or `time_of_day`, in the **baseline frame** | an active baseline definition | — (reported as `needs_setup`) | +| `proportion_shift`, `value_distribution_drift`, `value_correlation` | a span of more than one instant (self frame: it is cut into slices) | `stat_self_slices` | +| any of those six, or `time_of_day`, in the **baseline frame** | an active baseline definition | — (reported as `needs_setup`) | Three rows encode a distinction worth stating, because each was once drawn wrong: @@ -2756,6 +2758,105 @@ to the baseline definition, not a verdict on a finding. --- +## 17. Value correlation (field-to-field rules that break) + +**What it answers:** "One field used to decide another — when did that stop being true?" +The AMiner `VariableCorrelationDetector` analog (roadmap D13), intra-record: two fields of +the *same event*, unlike the sequence detectors, which relate consecutive events. An +account that always logs on to its own workstation, a status code that always follows a +given action, a service name that always runs from one path — each is an **implication +rule** `A = x ⇒ B = y`, and the event where it fails is often the event that matters even +when `x` and `y` are each ordinary. Value combos (§1) find a *pair* that is rare; this +finds a *rule* that broke. + +**How it works.** For a field pair `(A, B)` the detector counts events per `(a, b)` value +pair in the reference and per window. In each direction, an antecedent value `x` with at +least `stat_correlation_min_support` (20) reference events forms a rule with its dominant +consequent `y` when `y` accounts for at least `stat_correlation_rule_confidence` (0.95) +of them — nineteen in twenty, so a user with two home workstations has no rule and a +user with one has. A rule is **broken** in a window when the share of `x` events whose +`B` is not `y` rises: a 2×2 G-test of conforming against violating events between the +reference and the window (the same log-likelihood ratio proportion shift uses), one +Benjamini–Hochberg pool over every (rule, window) test in the run, and an effect floor +of `stat_correlation_min_ratio` (2) on the violation-rate ratio — 4 % to 6 % is not a +break however significant. Only rises are reported: a rule that *appears* in a window is +a value whose share changed, which proportion shift already owns, and a rule that +tightens is not a finding. + +**Two frames.** + +| | Self (`self-rule-g-test`) | Baseline (`rule-g-test`) | +|---|---|---| +| Rules mined from | the whole scope | the baseline window | +| Window | each of `stat_self_slices` equal time slices | each suspect window | +| Reference | the other slices (leave-one-out) | the baseline window | +| Catches | a rule that breaks in one stretch of the timeline | a rule that breaks after the incident start | + +In the self frame a rule mined from the whole scope already contains its own violations, +so a rule broken everywhere is not a rule and is never tested — which is right, since +nothing in the scope says it should have held. A rule broken in one stretch keeps its +confidence over the scope and fails against the complement of that stretch. + +**Which pairs.** An explicit `fields` list is scanned as every pair among them; auto mode +takes the recommender's top `stat_correlation_auto_fields` (6) categorical fields, steered +by the timeline's [field overrides](#declaring-which-fields-a-method-reads), and pairs +them (15 pairs). More than `stat_correlation_max_pairs` (20) pairs are truncated with a +warning naming the count — name fewer fields to choose which. Each pair is one +`GROUP BY a, b` scan capped at `stat_correlation_max_rows_per_pair` (5000) +highest-volume value pairs (cap → warning: a rule whose antecedent lives in the tail is +not tested). Fields with many distinct values (identifiers) are not recommended in the +first place, and pairing two of them is the one way to make this detector expensive. + +**Score = the G statistic**, like proportion shift. The representative event is the +**first violating occurrence** in the window, and `first_seen` is its timestamp. Findings +carry `fields` (`[antecedent, consequent]`) and `values` (`[x, y]`), `confidence` and +`support` (the rule as mined), `count` and `violations` (the window), `baseline_count` and +`baseline_violations` (the reference side, whichever frame), both violation rates and +their `rate_ratio`, `top_violator` with its count — the consequent value that most often +took `y`'s place, usually the answer to "where did it go instead?" — and `g_statistic`, +`p_value`, `q_value`. The baseline frame adds `window_label`/`window_start`/`window_end`; +the self frame the slice keys (`slice_index`, `rest_slices`). + +**Parameters.** + +- `fields` (request, default auto) — two or more; every pair among them is scanned. +- `fdr_q` (request) / `VESTIGO_STAT_CORRELATION_FDR_Q` (0.05) — the BH ceiling. +- `min_ratio` (request) / `VESTIGO_STAT_CORRELATION_MIN_RATIO` (2.0) — the violation-rate + ratio floor. +- `rule_confidence` (request) / `VESTIGO_STAT_CORRELATION_RULE_CONFIDENCE` (0.95) — the + consequent share that forms a rule. Snapshotted into the persisted `DetectorRun`. +- `min_support` (request) / `VESTIGO_STAT_CORRELATION_MIN_SUPPORT` (20) — reference + events an antecedent value needs. Snapshotted. +- `VESTIGO_STAT_CORRELATION_AUTO_FIELDS` (6), `VESTIGO_STAT_CORRELATION_MAX_PAIRS` (20), + `VESTIGO_STAT_CORRELATION_MAX_ROWS_PER_PAIR` (5000) — the pair and row caps above. + +**Allowlist key:** the combo one — `(A,B)` joined with `,` and `x␟y` joined with the +combo value separator — so **Mark normal** on a broken rule suppresses that rule in both +frames and wherever it recurs, and a rule declared normal from a combo row is the same +key. + +### Caveats + +- **Confidence is not causation, and the direction is mined both ways.** `user ⇒ host` + and `host ⇒ user` are different rules with different supports; a shared workstation + has no `host ⇒ user` rule while each of its users may keep a `user ⇒ host` one. Read + `fields` to see which direction broke. +- **A rule the reference already breaks a little is judged on the rise.** The G-test + compares rates, so a 2 % baseline violation rate that becomes 40 % is a strong finding + and one that becomes 3 % is not, whatever the counts. +- **Thin antecedents have no rules.** Twenty reference events is the floor; a value seen + a dozen times cannot form a rule however consistent, and appears in no finding. Lower + `min_support` for a short baseline, knowing that a rule mined from twenty events has a + 95 % confidence that is one violation wide. +- **The self frame cannot see a rule broken from the start.** Mined over the scope, a + rule that never held is not a rule; the baseline frame, mined over a declared normal + period, is where "it held before the incident and not after" lives. +- **Identifier pairs are the cost.** Two high-cardinality fields produce a pair table the + row cap truncates; the warning says so, and the recommender keeps identifiers out of + auto mode for this reason. +- A broken rule is **not malicious by itself** — a new laptop, a reassigned service, a + migration. Rank for triage; mark the rule Normal once it is explained. + --- ## Dispositions and normality (implementation notes) @@ -2788,8 +2889,9 @@ always for `tag_anomalies`) writes a `DetectorRun` row: the request params it ran with — fields, `series_field`, thresholds, `baseline_id`, resolved windows, `windows_hash`, `dispositions_hash`, the per-source clock-skew offsets in effect, entropy's `variant`, transition speed's `partition_field` and -`min_transitions`, time-of-day's `bucket_minutes` and `timezone`, and for a -self-frame run of the slice methods the +`min_transitions`, time-of-day's `bucket_minutes` and `timezone`, value +correlation's `rule_confidence` and `min_support`, and for a self-frame run of +the slice methods the `slices` payload, its `slices_hash` and the resolved self settings (`self_slices`, `pause_ratio`, `min_span_seconds`, `sequence_rarity_floor`) — plus the serialized result, and returns its id as `run_id`. Rows diff --git a/docs/PROGRESS.md b/docs/PROGRESS.md index 65067989..4871dbbf 100644 --- a/docs/PROGRESS.md +++ b/docs/PROGRESS.md @@ -4,8 +4,44 @@ Append-only session log — what changed and why, newest first. This file keeps sessions only; older ones live in git history, and every release is summarized in `CHANGELOG.md`. Plans belong in `ROADMAP.md`, not here. -Last updated: 2026-09-16 (1.20 in progress; sessions 239–240 — transition speed D15 and -time-of-day habit D12, the first two of the 1.20 detector cluster). +Last updated: 2026-09-16 (1.20 in progress; sessions 239–241 — transition speed D15, +time-of-day habit D12 and value correlation D13, the 1.20 detector cluster). + +## Session 241 — 2026-09-16: value correlation (D13) + +The third and last cheap AMiner analog, `value_correlation`, in the same one-commit shape. + +**Rules, not pairs.** Value combos already find a rare `(x, y)`. What this detector adds is +the *rule*: an antecedent value `x` with at least 20 reference events whose dominant +consequent `y` covers at least 95 % of them, mined in both directions per field pair, and +tested per window with the 2×2 G-test proportion shift already has — conforming against +violating events, reference against window, one BH pool, a 2× floor on the violation-rate +ratio. The roadmap's design problem was field-pair explosion, and the answer is three caps +that each disclose themselves: auto mode pairs the recommender's top six categorical +fields (15 pairs), more than 20 pairs are truncated with a warning naming the count, and +each pair's `GROUP BY a, b` is capped at 5000 highest-volume rows with a warning that a +tail antecedent was not tested. Identifiers never enter auto mode, which is what keeps the +pair table small in the common case. + +**Only rises, only breaks.** A rule that appears in a window is a value whose share rose, +which proportion shift owns; a rule that tightens is not a finding. And the self frame +mines its rules over the whole scope, so a rule broken from the start is not a rule and is +never tested — stated in the reference section as the frame's limit rather than hidden. +The finding carries `top_violator`, the consequent value that most often took `y`'s +place, because "where did it go instead?" is the first question an analyst asks of a +broken rule and the data to answer it was already in the scan. + +**The demo needed nothing.** Every human account has one or two home workstations +(`_home_hosts`); the contractor has one, so `m.okonkwo ⇒ WKS-004` holds over 6,500 +baseline logons at confidence ~1.0 and breaks on the jump host, the file server and two +finance workstations. Two-home users never form a rule (confidence ~0.5), the +administrator's hop is two events a day against 280 and keeps their rule intact, and in +the self frame the contractor's violations sit in the last five of 24 slices. Asserted in +both frames with `fields=["attr:user", "attr:computer_name"]`. + +**The agent schema budget, again.** Adding `rule_confidence` to `run_anomaly_detector` was +absorbed by the compact docstring from session 240; the measured total is recorded in +`AGENT.md`. ## Session 240 — 2026-09-16: time-of-day habit (D12) diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index e614d7d2..84b63c73 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -12,11 +12,11 @@ from that day. **Milestone 10 — AI agent log investigation — is the 2.0 thru everything below**; the numbered list orders the remaining 1.x work by payoff-per-effort: 1. **A12** local transform tools — no design round, no OPSEC gate. -2. **D13** — the last cheap detector reusing existing SQL machinery (D15 and D12 shipped in 1.20). -3. **W8** query-time field extraction — makes bespoke unstructured logs first-class. -4. **A8** external MCP toolsets — needs its own design round (policy, not plumbing). -5. **D10** / **D16** — heaviest lifts, last of the detector line. -6. **Milestone 11** external processors — P1 (the protocol doc) gates the rest; the +2. **W8** query-time field extraction — makes bespoke unstructured logs first-class. +3. **A8** external MCP toolsets — needs its own design round (policy, not plumbing). +4. **D10** / **D16** — heaviest lifts, last of the detector line (the cheap three, D15, D12 + and D13, shipped in 1.20). +5. **Milestone 11** external processors — P1 (the protocol doc) gates the rest; the Hayabusa engine half lives in `overcuriousity/hayabusa-processor`. Milestones 2–3 are polish, picked up opportunistically. Milestone 9 is additive work on @@ -141,20 +141,13 @@ burns its numbers out of that file**; the migration is done when the file is `{} Detectors adapted from [ait-aecid/logdata-anomaly-miner](https://github.com/ait-aecid/logdata-anomaly-miner), constrained to be **field-agnostic** and SQL-explainable per the forensic-reproducibility -requirement. D1–D9, D12, D15, `proportion_shift` and `sequence_motif` shipped — `ANOMALY_DETECTION.md` +requirement. D1–D9, D12, D13, D15, `proportion_shift` and `sequence_motif` shipped — `ANOMALY_DETECTION.md` is each detector's contract, updated in the same commit as any detector change. Every item below is incomplete until the frontend half lands with it: a plain-language method explanation, the SQL/params visible on the finding, disposition + allowlist wiring. A detector whose reasoning an analyst cannot read does not count as shipped. -**Low effort, high value:** - -- [ ] **D13 — Cross-field value correlation** (`VariableCorrelationDetector`): learn which - field-value pairs co-occur *within the same event*, flag violations. Intra-record, unlike - D10. Reuses `GROUP BY a, b` plus the G-test and Benjamini–Hochberg pool that - `proportion_shift` has. Field-pair explosion is the design problem: needs a preselection - rule and a candidate cap in the `HEAVY_SCAN_SETTINGS` family, honestly reported. **High effort, high value:** - [ ] **D10 — Event correlation rules** (`EventCorrelationDetector`): mine baseline diff --git a/frontend/src/api/anomalies.ts b/frontend/src/api/anomalies.ts index 4de251fc..205c4ff5 100644 --- a/frontend/src/api/anomalies.ts +++ b/frontend/src/api/anomalies.ts @@ -21,7 +21,7 @@ export interface LogTemplatesParams { } export interface AnomalyParams { - detector?: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time" | "time_of_day"; + detector?: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time" | "time_of_day" | "value_correlation"; /** Comma-separated field tokens for value_novelty, e.g. "artifact,display_name,attr:user_agent" */ fields?: string; /** Field to group frequency series / build event sequences by */ @@ -52,6 +52,8 @@ export interface AnomalyParams { bucket_minutes?: number; /** time_of_day only: IANA zone the clock is read in (e.g. "Europe/Berlin"). Omit for the server default. */ timezone?: string; + /** value_correlation only: share of an antecedent's reference events one consequent value must account for to form a rule. */ + rule_confidence?: number; /** ID of a saved baseline definition (baseline range + suspect windows). Omit for self-baseline. */ baseline_id?: string; limit?: number; @@ -108,7 +110,7 @@ export const anomaliesApi = { sourceId: string, eventId: string, body: { - detector: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time" | "time_of_day"; + detector: "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" | "transition_time" | "time_of_day" | "value_correlation"; content: string; details: Record; /** diff --git a/frontend/src/api/types.ts b/frontend/src/api/types.ts index 6db08061..c5ef5056 100644 --- a/frontend/src/api/types.ts +++ b/frontend/src/api/types.ts @@ -883,6 +883,62 @@ export interface TimeOfDayFinding { confirmed_other_scope?: boolean; } +/** + * One implication rule `A = x ⇒ B = y` broken in a window, from the + * value_correlation detector (D13). `fields` is [antecedent, consequent] and + * `values` is [x, y], the combo shape, so the allowlist key is the combo one. + * `details.method` is `rule-g-test` (mined from the baseline window, tested + * per suspect window; `window_*` keys) or `self-rule-g-test` (mined from the + * timeline, each leave-one-out slice tested against the rest; `slice_index`, + * `rest_slices`). The `baseline_*` fields hold the reference side either way. + */ +export interface ValueCorrelationFinding { + type: "value_correlation"; + fields: string[]; + values: string[]; + /** "x ⇒ y" — display form. */ + value: string; + /** Share of the antecedent's reference events that carried y. */ + confidence: number; + /** Antecedent events the rule was mined from. */ + support: number; + /** Antecedent events in the window or slice under test. */ + count: number; + /** Of those, the ones whose consequent was not y. */ + violations: number; + /** The reference side: antecedent events and violations in the baseline window, or the other slices. */ + baseline_count: number; + baseline_violations: number; + violation_rate: number; + baseline_violation_rate: number; + /** violation_rate ÷ baseline_violation_rate (0.5-smoothed when the reference has none). */ + rate_ratio: number; + /** The most common violating consequent value in the window, and its count. */ + top_violator: string; + top_violator_count: number; + g_statistic: number; + p_value: number; + /** Benjamini–Hochberg adjusted p-value across every test in the run. */ + q_value: number; + /** = g_statistic — used for ranking. */ + score: number; + /** First violating occurrence in the window. */ + first_seen: string | null; + event_id: string | null; + event: Event | null; + details: Record; + /** Present (true) only when the request passed `include_dismissed`. */ + dismissed?: boolean; + /** Present (true) when a confirmed disposition covers this finding's event. */ + confirmed?: boolean; + /** + * Present (true) when the only confirmed verdict on this event was reached + * under a *different* comparison. The claim stands, but not for this scope — + * so the row is marked rather than badged, and Confirm stays live. + */ + confirmed_other_scope?: boolean; +} + export type AnomalyFinding = | ValueNoveltyFinding | ValueComboFinding @@ -897,7 +953,8 @@ export type AnomalyFinding = | SequenceMotifFinding | DistributionDriftFinding | TransitionTimeFinding - | TimeOfDayFinding; + | TimeOfDayFinding + | ValueCorrelationFinding; export interface AnomaliesResponse { status: "ok" | "no_data" | "insufficient_data"; @@ -1105,7 +1162,8 @@ export interface AnomalyMarker { | "sequence_novelty" | "value_distribution_drift" | "transition_time" - | "time_of_day"; + | "time_of_day" + | "value_correlation"; /** Raw structured finding data — stored verbatim on the persisted annotation. */ rawDetails: Record; /** End of the anomalous window, for frequency findings — enables a range highlight. */ diff --git a/frontend/src/components/analysis/FindingEvidence.tsx b/frontend/src/components/analysis/FindingEvidence.tsx index f8912cf4..4751dcbf 100644 --- a/frontend/src/components/analysis/FindingEvidence.tsx +++ b/frontend/src/components/analysis/FindingEvidence.tsx @@ -377,6 +377,24 @@ export function FindingEvidence({ finding }: { finding: MethodResult }) { case "sequence_novelty": case "sequence_motif": return ; + case "value_correlation": { + // The claim is a violation rate against a reference rate, both counted. + const self = findingMode(finding) === "self-rule-g-test"; + return ( + + ); + } case "time_of_day": return ( AS a, AS b,\n countIf() AS ref_n,\n countIf() AS win_n\nFROM events\nWHERE case_id = {case}\nGROUP BY a, b\n-- rule a=x => b=y when y holds >= {rule_confidence} of x's reference events (support >= {min_support})\n-- G-test of conforming vs violating events, reference vs window; Benjamini-Hochberg across the run`, + knobs: [ + fieldsKnob({ minSelected: 2, autoCount: 6, autoLabel: "top 6, all pairs" }), + FDR_KNOB, + RATIO_KNOB, + { param: "rule_confidence", label: "Rule confidence", kind: "number", placeholder: "0.95" }, + { param: "min_support", label: "Min support", kind: "number", placeholder: "20" }, + ], + }, { id: "numeric_range", label: "Numeric range", diff --git a/frontend/src/lib/finding-frame.ts b/frontend/src/lib/finding-frame.ts index e7be00ab..109e0407 100644 --- a/frontend/src/lib/finding-frame.ts +++ b/frontend/src/lib/finding-frame.ts @@ -25,6 +25,7 @@ export const TEMPORAL_MODES: ReadonlySet = new Set([ "drift", "min-transition", "habit", + "rule-g-test", ]); /** @@ -38,10 +39,15 @@ export const SELF_MODES: ReadonlySet = new Set([ "rare-ngram", "self-min-transition", "self-habit", + "self-rule-g-test", ]); -/** The two self modes that compare leave-one-out time slices with the rest of the scope. */ -export const SLICE_MODES: ReadonlySet = new Set(["self-g-test", "self-drift"]); +/** The self modes that compare leave-one-out time slices with the rest of the scope. */ +export const SLICE_MODES: ReadonlySet = new Set([ + "self-g-test", + "self-drift", + "self-rule-g-test", +]); export function isTemporalMode(method: string): boolean { return TEMPORAL_MODES.has(method); diff --git a/frontend/src/lib/finding-normalize.ts b/frontend/src/lib/finding-normalize.ts index 5fc093cd..a4c83fe1 100644 --- a/frontend/src/lib/finding-normalize.ts +++ b/frontend/src/lib/finding-normalize.ts @@ -148,6 +148,13 @@ export function normalizeFinding(meta: DetectorMeta, f: AnomalyFinding, rank: nu }${findingMode(f) === "self-habit" ? " across the timeline" : ""}`; ts = ts ?? f.first_seen; break; + case "value_correlation": + title = `${pair(f.fields[0] ?? "", f.values[0] ?? "")} ⇒ ${pair(f.fields[1] ?? "", f.values[1] ?? "")}`; + subtitle = `broken ${f.violations}× of ${f.count} in ${String(f.details["window_label"] ?? "the suspect window")}${ + findingMode(f) === "self-rule-g-test" ? " vs the rest" : "" + } · mostly ${truncate(f.top_violator, 30)} (q=${f.q_value.toExponential(1)})`; + ts = ts ?? f.first_seen; + break; } return { detectorId: meta.id, diff --git a/frontend/src/lib/finding-subject.ts b/frontend/src/lib/finding-subject.ts index 60e46f82..5a00c01e 100644 --- a/frontend/src/lib/finding-subject.ts +++ b/frontend/src/lib/finding-subject.ts @@ -44,6 +44,7 @@ function scoredSubject(f: AnomalyFinding): SubjectPair[] { case "time_of_day": return [{ label: fieldLabel(f.field), value: String(f.value) }]; case "value_combo": + case "value_correlation": return f.fields.map((field, i) => ({ label: fieldLabel(field), value: String(f.values[i] ?? ""), diff --git a/frontend/src/lib/finding-verdict.ts b/frontend/src/lib/finding-verdict.ts index 8df2628b..67852fcc 100644 --- a/frontend/src/lib/finding-verdict.ts +++ b/frontend/src/lib/finding-verdict.ts @@ -194,6 +194,17 @@ function scoredVerdict(f: AnomalyFinding): Verdict { tail: `— ${reference} ${habit}.`, }; } + case "value_correlation": { + const where = + findingMode(f) === "self-rule-g-test" + ? `in ${detailString(f.details, "window_label") ?? "this slice"} against the rest of the timeline` + : `in ${detailString(f.details, "window_label") ?? "the suspect window"}`; + return { + lead: `${fieldLabel(f.fields[0] ?? "")} = ${truncate(f.values[0] ?? "", 40)} normally means ${fieldLabel(f.fields[1] ?? "")} = ${truncate(f.values[1] ?? "", 40)} (${pct(f.confidence)} of ${f.support} reference events). ${where} it did not hold in`, + highlight: `${f.violations} of ${f.count} events`, + tail: `— most often ${fieldLabel(f.fields[1] ?? "")} = ${truncate(f.top_violator, 40)} (×${f.top_violator_count}), against ${f.baseline_violations} of ${f.baseline_count} in the reference (q=${f.q_value.toExponential(1)}).`, + }; + } } } diff --git a/frontend/src/test/methodRegistry.test.ts b/frontend/src/test/methodRegistry.test.ts index aac1a90d..e39b4979 100644 --- a/frontend/src/test/methodRegistry.test.ts +++ b/frontend/src/test/methodRegistry.test.ts @@ -49,6 +49,7 @@ describe("method registry", () => { sequence_novelty: ["series_field", "ngram_size", "max_gap_seconds"], transition_time: ["series_field", "partition_field", "min_ratio"], time_of_day: ["fields", "bucket_minutes", "timezone"], + value_correlation: ["fields", "fdr_q", "min_ratio", "rule_confidence", "min_support"], log_template: ["field", "order", "only_new"], }; for (const m of METHODS) { @@ -58,7 +59,7 @@ describe("method registry", () => { } }); - it("covers exactly the fourteen methods the gate plans for", () => { + it("covers exactly the fifteen methods the gate plans for", () => { // METHOD_IDS in db/analysis_plan.py. A method here that the plan never // reports would render with no status; one there that is missing here // would never be shown at all. @@ -76,6 +77,7 @@ describe("method registry", () => { "timestamp_order", "transition_time", "value_combo", + "value_correlation", "value_distribution_drift", "value_novelty", ].sort(), diff --git a/src/vestigo/agent/tools.py b/src/vestigo/agent/tools.py index 38be7913..74d3a5f0 100644 --- a/src/vestigo/agent/tools.py +++ b/src/vestigo/agent/tools.py @@ -1970,6 +1970,7 @@ async def run_anomaly_detector( partition_field: str | None = None, bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = None, timezone: str | None = None, + rule_confidence: float | None = None, ) -> dict[str, Any]: """Run a statistical anomaly detector over the timeline. @@ -1977,21 +1978,21 @@ async def run_anomaly_detector( numeric_range, charset, entropy, proportion_shift, interval_periodicity, sequence_novelty, sequence_motif, value_distribution_drift, transition_time (a value pair reached - faster than its learned floor), time_of_day (a value at an hour it - has no habit of). `fields`: comma-separated list for value detectors - (omit to auto-recommend); `series_field`: the field frequency, - sequence and transition detectors group by; `partition_field` - (transition_time): the stream timed, e.g. attr:user. Every detector - runs without `baseline_id` (the timeline is its own reference); pass - one from list_baselines to score suspect windows instead. Knobs - (server defaults otherwise): z_threshold (frequency), - min_skew_seconds (timestamp_order), fdr_q, min_ratio (effect floor), - ngram_size, min_support and start/end (sequence_motif), group_field - (charset: one alphabet per value), max_gap_seconds (sequences), - variant (entropy: shannon or bigram), bucket_minutes and IANA - timezone (time_of_day). Returns findings plus a persisted run_id; - each finding carries an example event_id — call get_event for the - full record. Virtual `time:` fields are rejected here. + faster than ever), time_of_day (a value at an unusual hour), + value_correlation (a rule A=x ⇒ B=y that breaks). `fields`: + comma-separated, value detectors, omit to auto-recommend; + `series_field`: frequency/sequence/transition group-by; + `partition_field`: the stream transition_time times, e.g. attr:user. + Every detector runs without `baseline_id` (timeline as its own + reference); pass one from list_baselines to score suspect windows. + Knobs (server defaults otherwise): z_threshold, min_skew_seconds, + fdr_q, min_ratio, ngram_size, start/end (sequence_motif), + min_support (motif or rule support), rule_confidence (0-1), + group_field (charset: one alphabet per value), max_gap_seconds + (sequences), variant (entropy: shannon|bigram), bucket_minutes and + IANA timezone (time_of_day). Returns findings plus a persisted + run_id; each finding carries an example event_id — call get_event for + the full record. Virtual `time:` fields are rejected. """ _reject_time_fields(fields, "fields") _reject_time_fields(series_field, "series_field") @@ -2019,6 +2020,7 @@ async def run_anomaly_detector( partition_field=partition_field, bucket_minutes=bucket_minutes, timezone=timezone, + rule_confidence=rule_confidence, field_mappings=scope.field_mappings, source_offsets=scope.source_offsets, ) diff --git a/src/vestigo/api/routers/analysis.py b/src/vestigo/api/routers/analysis.py index 47368fd5..9e2af21b 100644 --- a/src/vestigo/api/routers/analysis.py +++ b/src/vestigo/api/routers/analysis.py @@ -389,6 +389,13 @@ def _empty_is_none(cls, v: Any) -> Any: return None if v == "" else v +class _ValueCorrelationParams(_FieldsParams): + fdr_q: float | None = Field(default=None, gt=0, le=1) + min_ratio: float | None = Field(default=None, gt=1) + rule_confidence: float | None = Field(default=None, gt=0, le=1) + min_support: int | None = Field(default=None, ge=2) + + class _LogTemplateParams(_Params): #: Not a `_run_stat_detector` detector — log templating is a browser with #: its own service call (see :func:`_run_log_templates`). Routing it through @@ -412,6 +419,7 @@ class _LogTemplateParams(_Params): "sequence_novelty": _SequenceNoveltyParams, "transition_time": _TransitionTimeParams, "time_of_day": _TimeOfDayParams, + "value_correlation": _ValueCorrelationParams, "log_template": _LogTemplateParams, } @@ -745,6 +753,8 @@ async def get_analysis_findings( partition_field=kwargs.get("partition_field"), bucket_minutes=kwargs.get("bucket_minutes"), timezone=kwargs.get("timezone"), + min_support=kwargs.get("min_support"), + rule_confidence=kwargs.get("rule_confidence"), # Both come from _resolve_timeline_scope and are not optional # niceties: without field_mappings a canonical field alias is # ignored, and without source_offsets a declared per-source diff --git a/src/vestigo/api/routers/events.py b/src/vestigo/api/routers/events.py index 4a0f3bde..b4665561 100644 --- a/src/vestigo/api/routers/events.py +++ b/src/vestigo/api/routers/events.py @@ -37,6 +37,7 @@ AnalysisWindows, CharsetFinding, ComboFinding, + CorrelationFinding, DistributionDriftFinding, EntropyFinding, FreqFinding, @@ -2257,6 +2258,7 @@ async def _run_stat_detector( partition_field: str | None = None, bucket_minutes: int | None = None, timezone: str | None = None, + rule_confidence: float | None = None, field_mappings: dict[str, list[str]] | None = None, source_offsets: dict[str, int] | None = None, field_overrides: dict[str, bool] | None = _RESOLVE_OVERRIDES, @@ -2586,6 +2588,50 @@ async def _run_stat_detector( raise HTTPException(status_code=422, detail=str(exc)) from exc return result, resolution + if detector == "value_correlation": + # The rule floors and the test thresholds all shape what a "rule" and + # a "break" are, so every effective value is snapshotted (D13). + resolution["correlation_fdr_q"] = fdr_q if fdr_q is not None else cfg.stat_correlation_fdr_q + resolution["correlation_min_ratio"] = ( + min_ratio if min_ratio is not None else cfg.stat_correlation_min_ratio + ) + resolution["correlation_rule_confidence"] = ( + rule_confidence if rule_confidence is not None else cfg.stat_correlation_rule_confidence + ) + resolution["correlation_min_support"] = ( + min_support if min_support is not None else cfg.stat_correlation_min_support + ) + if windows is None: + resolution["self_slices"] = cfg.stat_self_slices + try: + result = await run_scan( + svc.find_value_correlations, + case_id=case_id, + source_ids=source_ids, + source_offsets=source_offsets, + fields=parsed_fields, + limit=limit, + windows=windows, + fdr_q=resolution["correlation_fdr_q"], + min_ratio=resolution["correlation_min_ratio"], + rule_confidence=resolution["correlation_rule_confidence"], + min_support=resolution["correlation_min_support"], + max_pairs=cfg.stat_correlation_max_pairs, + max_rows_per_pair=cfg.stat_correlation_max_rows_per_pair, + auto_fields=cfg.stat_correlation_auto_fields, + exclude_event_ids=exclude_ids, + allowlist=allowlist, + field_mappings=field_mappings, + inventory=inventory, + inventory_total=inventory_total, + field_overrides=field_overrides, + self_slices=cfg.stat_self_slices, + ) + _snapshot_slices(result, resolution) + return result, resolution + except ValueError as exc: + raise HTTPException(status_code=422, detail=str(exc)) from exc + if detector == "time_of_day": # The resolution and the zone decide what "03:40" means, so both are # resolved here (request, else server default) and snapshotted (D12). @@ -2833,13 +2879,16 @@ def _serialize_finding( | MotifFinding | DistributionDriftFinding | TransitionFinding - | HabitFinding, + | HabitFinding + | CorrelationFinding, ) -> dict[str, Any]: """Serialise a Value/Freq/Order/Combo/Range/Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition/Habit finding to a JSON-safe dict.""" # Charset/Entropy/Shift/Interval/Sequence/Motif/Drift/Transition/Habit # finding dataclass fields are exactly the wire keys, so asdict() avoids a # hand-maintained field-by-field transcription that would silently drop # any newly added field. + if isinstance(r, CorrelationFinding): + return {"type": "value_correlation", **asdict(r)} if isinstance(r, HabitFinding): return {"type": "time_of_day", **asdict(r)} if isinstance(r, TransitionFinding): @@ -3222,10 +3271,15 @@ async def _persist_detector_run( # per detector, None for every other one. "fdr_q": resolution.get("shift_fdr_q") or resolution.get("interval_fdr_q") - or resolution.get("drift_fdr_q"), + or resolution.get("drift_fdr_q") + or resolution.get("correlation_fdr_q"), "min_ratio": resolution.get("shift_min_ratio") or resolution.get("interval_min_rate_ratio") - or resolution.get("transition_min_ratio"), + or resolution.get("transition_min_ratio") + or resolution.get("correlation_min_ratio"), + # value_correlation: what counted as a rule (D13). `min_support` is + # shared with sequence_motif's key by the same disjointness rule. + "rule_confidence": resolution.get("correlation_rule_confidence"), # transition_time: the stream key transitions were timed within # (None = per source) and the learning floor the run used (D15). "partition_field": resolution.get("transition_partition_field"), @@ -3235,6 +3289,11 @@ async def _persist_detector_run( # reproducible, which is why it is recorded rather than implied. "bucket_minutes": resolution.get("habit_bucket_minutes"), "timezone": resolution.get("habit_timezone"), + "min_support": ( + resolution["correlation_min_support"] + if "correlation_min_support" in resolution + else resolution.get("motif_min_support") + ), # sequence_novelty: effective (request-or-default) n-gram length. "ngram_size": resolution.get("sequence_ngram"), # charset: per-identifier scoping (None = one alphabet per field). @@ -3309,7 +3368,7 @@ async def list_anomalies( timeline_id: str, detector: str = Query( default="value_novelty", - description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', 'transition_time', or 'time_of_day'.", + description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', 'transition_time', 'time_of_day', or 'value_correlation'.", ), fields: str | None = Query( default=None, @@ -3423,6 +3482,15 @@ async def list_anomalies( "stamped into the persisted run. Omit to use the server default." ), ), + rule_confidence: float | None = Query( + default=None, + gt=0, + le=1, + description=( + "value_correlation only: share of an antecedent's reference events one " + "consequent value must account for to form a rule. Omit to use the server default." + ), + ), start: datetime | None = Query( default=None, description="sequence_motif only: scope mining to events at/after this time (ISO, UTC).", @@ -3500,6 +3568,11 @@ async def list_anomalies( spacing test, beaconing). Without: whole-scope Greenwood beaconing with pauses excluded, and a robust-Gamma silence test. BH-FDR across the run. + **value_correlation**: per field pair, mines implication rules + `A = x ⇒ B = y` within the same event from the reference (support and + confidence floors, both directions) and flags a window in which a rule's + violation rate rises — G-test, BH pool, effect floor (D13). + **time_of_day**: per (field, value), learns which wall-clock buckets of the day (in an explicit IANA `timezone`) the value habitually occurs in and flags occurrences outside that habit, scored by the hours to the @@ -3554,6 +3627,7 @@ async def list_anomalies( partition_field=partition_field, bucket_minutes=bucket_minutes, timezone=timezone, + rule_confidence=rule_confidence, field_mappings=field_mappings, source_offsets=source_offsets, ) @@ -3686,7 +3760,7 @@ class TagAnomaliesRequest(BaseModel): detector: str = Field( default="value_novelty", - description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', 'transition_time', or 'time_of_day'.", + description="Detector to run: 'value_novelty', 'value_combo', 'frequency', 'timestamp_order', 'numeric_range', 'charset', 'entropy', 'proportion_shift', 'interval_periodicity', 'sequence_novelty', 'sequence_motif', 'value_distribution_drift', 'transition_time', 'time_of_day', or 'value_correlation'.", ) fields: str | None = Field( default=None, @@ -3758,6 +3832,12 @@ class TagAnomaliesRequest(BaseModel): max_length=64, description="time_of_day only: IANA zone the clock is read in; stamped into the run.", ) + rule_confidence: float | None = Field( + default=None, + gt=0, + le=1, + description="value_correlation only: consequent share that forms a rule.", + ) start: datetime | None = Field( default=None, description="sequence_motif only: scope mining to events at/after this time.", @@ -3828,6 +3908,7 @@ async def tag_anomalies( partition_field=body.partition_field, bucket_minutes=body.bucket_minutes, timezone=body.timezone, + rule_confidence=body.rule_confidence, field_mappings=field_mappings, source_offsets=source_offsets, ) diff --git a/src/vestigo/core/config.py b/src/vestigo/core/config.py index e9522dc2..40091b17 100644 --- a/src/vestigo/core/config.py +++ b/src/vestigo/core/config.py @@ -211,6 +211,20 @@ class Settings(BaseSettings): # other per-field caps because each value returns one row per occupied # bucket and window. stat_habit_max_candidates_per_field: int = 500 + # Value-correlation detector (D13): an antecedent value needs this many + # reference events, and one consequent value must account for at least + # this share of them, before "A = x ⇒ B = y" counts as a rule. + stat_correlation_min_support: int = Field(default=20, ge=2) + stat_correlation_rule_confidence: float = Field(default=0.95, gt=0, le=1) + # BH false-discovery ceiling and the violation-rate ratio floor for a + # broken rule — same meaning as the proportion-shift pair. + stat_correlation_fdr_q: float = 0.05 + stat_correlation_min_ratio: float = 2.0 + # How many recommended fields auto mode pairs up (6 → 15 pairs), the cap + # on pairs scanned per run, and the per-pair cap on (a, b) value rows. + stat_correlation_auto_fields: int = Field(default=6, ge=2) + stat_correlation_max_pairs: int = Field(default=20, ge=1) + stat_correlation_max_rows_per_pair: int = 5000 # ── Self frame for the slice-based detectors (D18) ────────────────────── # How many equal-width time slices proportion_shift and # value_distribution_drift cut the scope into when no baseline is declared; diff --git a/src/vestigo/core/settings_registry.py b/src/vestigo/core/settings_registry.py index e2b7fd22..5f5d1a6d 100644 --- a/src/vestigo/core/settings_registry.py +++ b/src/vestigo/core/settings_registry.py @@ -584,6 +584,48 @@ class SettingGroup: "Time-of-day candidate cap", "Candidate values scanned per field for a habit, highest volume first.", ), + SettingSpec( + "stat_correlation_min_support", + "detectors", + "Correlation rule support", + "An antecedent value needs at least this many reference events before a rule is mined from it.", + ), + SettingSpec( + "stat_correlation_rule_confidence", + "detectors", + "Correlation rule confidence", + "Share of an antecedent's reference events one consequent value must account for to form a rule (0.95 = 19 in 20).", + ), + SettingSpec( + "stat_correlation_fdr_q", + "detectors", + "Correlation FDR q", + "Benjamini–Hochberg false-discovery ceiling for broken-rule tests.", + ), + SettingSpec( + "stat_correlation_min_ratio", + "detectors", + "Correlation effect floor", + "A rule's violation rate must rise by at least this factor to be reported.", + ), + SettingSpec( + "stat_correlation_auto_fields", + "detectors", + "Correlation auto fields", + "How many recommended fields auto mode pairs up (6 fields make 15 pairs).", + ), + SettingSpec( + "stat_correlation_max_pairs", + "detectors", + "Correlation pair cap", + "Field pairs scanned per run; the rest are dropped with a warning.", + ), + SettingSpec( + "stat_correlation_max_rows_per_pair", + "detectors", + "Correlation candidate rows", + "Value-pair rows fetched per field pair, highest volume first.", + ), SettingSpec( "stat_self_slices", "detectors", diff --git a/src/vestigo/db/analysis_cache.py b/src/vestigo/db/analysis_cache.py index 0c56e45c..e51cf688 100644 --- a/src/vestigo/db/analysis_cache.py +++ b/src/vestigo/db/analysis_cache.py @@ -58,7 +58,7 @@ #: for the same inputs: a v3 row holds "up" findings for every short-lived #: value on the timeline, which is the shape this bump exists to retire. #: -#: 5 — the 1.20 detectors (transition_time D15, time_of_day D12). New method ids cannot +#: 5 — the 1.20 detectors (transition_time D15, time_of_day D12, value_correlation D13). New method ids cannot #: collide with an older row's key on their own, but the shared n-gram #: assembly now emits two more columns, and a key that names the runner's #: inputs while the runner's SQL changed underneath it is the case this diff --git a/src/vestigo/db/analysis_plan.py b/src/vestigo/db/analysis_plan.py index 616ed169..97ca1076 100644 --- a/src/vestigo/db/analysis_plan.py +++ b/src/vestigo/db/analysis_plan.py @@ -50,6 +50,7 @@ "sequence_novelty", "transition_time", "time_of_day", + "value_correlation", "log_template", ) @@ -73,6 +74,7 @@ "value_distribution_drift", "interval_periodicity", "time_of_day", + "value_correlation", } ) @@ -94,6 +96,7 @@ "sequence_novelty": "heavy", "transition_time": "heavy", "time_of_day": "heavy", + "value_correlation": "heavy", "log_template": "heavy", } @@ -348,6 +351,7 @@ def build_plan(inputs: PlanInputs, cfg: Settings) -> list[MethodPlan]: "sequence_novelty", "transition_time", "time_of_day", + "value_correlation", ): if frame_needs_baseline: plans[method] = _setup( @@ -359,7 +363,19 @@ def build_plan(inputs: PlanInputs, cfg: Settings) -> list[MethodPlan]: # slices, and a span of one instant cannot be sliced. That is the one # structural impossibility; a short span merely yields thin slices, which # the run warns about rather than the gate withholding it. - for method in ("proportion_shift", "value_distribution_drift"): + # A rule needs two fields to relate, exactly as a combo does. Checked + # before the slice rule so a one-field timeline says why it is gated in + # either frame; the baseline frame's needs_setup above still outranks it. + if len(cats) < 2: + plans.setdefault( + "value_correlation", + _no( + "value_correlation", + "only one usable categorical field — a rule needs two", + {"categorical_fields": len(cats), "required": 2}, + ), + ) + for method in ("proportion_shift", "value_distribution_drift", "value_correlation"): if inputs.frame == "self" and inputs.span_seconds <= 0.0: plans.setdefault( method, diff --git a/src/vestigo/db/anomaly_stats.py b/src/vestigo/db/anomaly_stats.py index 5530a18b..d2eac7a0 100644 --- a/src/vestigo/db/anomaly_stats.py +++ b/src/vestigo/db/anomaly_stats.py @@ -218,6 +218,22 @@ the value's busy buckets across the scope as its habit and scores every thin bucket against them. +**value_correlation** (``detector="value_correlation"``) + Per field pair, mine implication rules ``A = x ⇒ B = y`` within the same + event — an antecedent value with at least ``min_support`` reference + events whose consequent is one value at least ``rule_confidence`` of the + time, both directions — and flag a window in which the rule's violation + rate rises: a 2×2 G-test of conforming against violating events between + the reference and the window, one Benjamini–Hochberg pool per run, an + effect floor of ``min_ratio`` on the violation-rate ratio. Adapted from + AMiner's ``VariableCorrelationDetector`` (roadmap D13); intra-record, + unlike the sequence detectors. Two frames: *baseline* + (``method="rule-g-test"``) mines from the baseline window and tests each + suspect window; *self* (``method="self-rule-g-test"``, D18) mines from the + scope and tests each leave-one-out slice against its complement. Pairs + come from an explicit field list or the recommender's top categorical + fields, capped at ``max_pairs``. Score = the G statistic. + **timestamp_order** (``detector="timestamp_order"``) Flag events whose parsed timestamp jumps *backwards* relative to the previous record in the source file (record order = ``byte_offset``, then @@ -1381,6 +1397,47 @@ class HabitFinding: details: dict[str, Any] +@dataclass +class CorrelationFinding: + """One implication rule ``A = x ⇒ B = y`` broken in a window (value_correlation, D13).""" + + # [antecedent field, consequent field] and [x, y] — the combo shape. + fields: list[str] + values: list[str] + # "x ⇒ y" — display form. + value: str + # Share of the antecedent's reference events that carried y (≥ rule_confidence). + confidence: float + # Antecedent events the rule was mined from (baseline window, or the scope). + support: int + # Antecedent events in the window (or slice) under test. + count: int + # Of those, the ones whose consequent was not y. + violations: int + # The reference side of the test: antecedent events and violations in the + # baseline window (baseline frame) or in the other slices (self frame). + baseline_count: int + baseline_violations: int + violation_rate: float + baseline_violation_rate: float + # violation_rate / baseline_violation_rate (0.5-smoothed when the reference has none). + rate_ratio: float + # The most common violating consequent value in the window, and its count. + top_violator: str + top_violator_count: int + g_statistic: float + p_value: float + # Benjamini–Hochberg adjusted p-value across every test in this run. + q_value: float + # = g_statistic; used for ranking. + score: float + # First violating occurrence in the window. + first_seen: str | None + event_id: str | None + event: dict[str, Any] | None + details: dict[str, Any] + + @dataclass class MotifFinding: """One recurring event-order n-gram surfaced by the sequence-motif miner.""" @@ -1423,12 +1480,13 @@ class StatAnomalyResult: # "value_novelty" | "value_combo" | "frequency" | "timestamp_order" | "numeric_range" # | "charset" | "entropy" | "proportion_shift" | "interval_periodicity" # | "sequence_novelty" | "sequence_motif" | "value_distribution_drift" - # | "transition_time" | "time_of_day" + # | "transition_time" | "time_of_day" | "value_correlation" detector: str # "self-baseline" | "temporal" | "z-score" | "temporal-z-score" | "sequential" # | "iqr" | "temporal-range" | "rare-chars" | "temporal-charset" | "temporal-iqr" # | "g-test" | "cadence" | "ngram" | "motif" | "drift" | "min-transition" - # | "self-min-transition" | "habit" | "self-habit" + # | "self-min-transition" | "habit" | "self-habit" | "rule-g-test" + # | "self-rule-g-test" method: str baseline_size: int # total events (value_novelty) or event-count used for z-score results: list[ @@ -1446,6 +1504,7 @@ class StatAnomalyResult: | DistributionDriftFinding | TransitionFinding | HabitFinding + | CorrelationFinding ] = field(default_factory=list) # Effective |z| cutoff used by the frequency detector; None for value_novelty. z_threshold: float | None = None @@ -9399,6 +9458,417 @@ def find_time_of_day_habits( windows=windows, ) + # ------------------------------------------------------------------ + # Value correlation (D13) + # ------------------------------------------------------------------ + + @gated_heavy_scan + def find_value_correlations( + self, + case_id: str, + source_ids: list[str], + fields: list[str] | None = None, + limit: int = 50, + windows: AnalysisWindows | None = None, + fdr_q: float = 0.05, + min_ratio: float = 2.0, + rule_confidence: float = 0.95, + min_support: int = 20, + max_pairs: int = 20, + max_rows_per_pair: int = 5000, + auto_fields: int = 6, + exclude_event_ids: set[str] | None = None, + allowlist: set[tuple[str, str]] | None = None, + field_mappings: dict[str, list[str]] | None = None, + inventory: list[tuple[str, int, int]] | None = None, + inventory_total: int | None = None, + source_offsets: dict[str, int] | None = None, + field_overrides: dict[str, bool] | None = None, + self_slices: int = 24, + ) -> StatAnomalyResult: + """Return implication rules between two fields that break in a window. + + AMiner ``VariableCorrelationDetector`` analog (D13), intra-record: for + a field pair ``(A, B)`` and an antecedent value ``x`` of ``A`` with at + least *min_support* reference events, the rule ``A = x ⇒ B = y`` holds + when ``y`` accounts for at least *rule_confidence* of those events. + Both directions are mined. A rule is **broken** in a window when the + share of ``x`` events whose ``B`` is not ``y`` rises: a 2×2 G-test of + conforming against violating events between the reference and the + window, one Benjamini–Hochberg pool per run, and an effect floor of + *min_ratio* on the violation-rate ratio. Only rises are reported — a + rule that appears in a window is a proportion shift, which owns it. + + Two frames. *Baseline* (``method="rule-g-test"``): rules are mined from + the baseline window and each suspect window is tested against it. + *Self* (``method="self-rule-g-test"``, D18): rules are mined from the + whole scope and each of *self_slices* leave-one-out time slices is + tested against its complement — a slice of violations against every + other slice. + + Pairs come from an explicit *fields* list (every pair among them) or + from the recommender's top *auto_fields* categorical fields, steered by + *field_overrides*; more than *max_pairs* pairs are truncated with a + warning. One ``GROUP BY a, b`` scan per pair, capped at + *max_rows_per_pair* highest-volume value pairs (cap → warning). Score + = the G statistic, like proportion shift. The allowlist key is the + combo one: ``(A,B)`` and ``x␟y`` joined with the combo separators, so + one Normal verdict covers the rule in both frames. + """ + detector = "value_correlation" + if not 0.0 < rule_confidence <= 1.0: + raise ValueError("rule_confidence must be in (0, 1]") + if min_ratio <= 1.0: + raise ValueError("min_ratio must be greater than 1") + if fields is not None and len(fields) < 2: + raise ValueError("value_correlation requires at least two fields") + method = "rule-g-test" if windows is not None else "self-rule-g-test" + self.ch.init_schema() + db = self.ch.database + base_params: dict[str, Any] = {"cid": case_id, "src": source_ids} + eff = effective_ts_sql(source_offsets) + + total_events = self._count_events(case_id, source_ids) + if total_events == 0: + return StatAnomalyResult( + status="no_data", + detector=detector, + method=method, + baseline_size=0, + windows=windows.payload() if windows is not None else None, + ) + + run_warnings: list[str] = [] + slices: SelfSlices | None = None + slice_totals: list[int] = [] + if windows is not None: + reference_size, suspect_totals = self._window_totals( + case_id, source_ids, windows, source_offsets + ) + run_warnings += _window_size_warnings(windows, suspect_totals) + if reference_size == 0: + return StatAnomalyResult( + status="insufficient_data", + detector=detector, + method=method, + baseline_size=0, + warnings=[*run_warnings, "The baseline window contains no events."], + windows=windows.payload(), + ) + n_frames = len(windows.suspects) + else: + slices = self._self_slices(case_id, source_ids, self_slices, source_offsets) + if slices is None: + return StatAnomalyResult( + status="insufficient_data", + detector=detector, + method=method, + baseline_size=total_events, + warnings=["The scope holds a single timestamp, so it cannot be sliced."], + ) + slice_totals = self._slice_totals(case_id, source_ids, slices, source_offsets) + reference_size = sum(slice_totals) + run_warnings += self._small_slice_warning(slices, slice_totals) + n_frames = slices.k + + # Pair selection. + if fields is not None: + pool = list(dict.fromkeys(fields)) + else: + rec = self.recommend_novelty_fields( + case_id, + source_ids, + total=inventory_total if inventory is not None else total_events, + field_mappings=field_mappings, + inventory=inventory, + ) + pool, override_notes = apply_field_overrides( + [f.token for f in rec if f.recommended], field_overrides, [f.token for f in rec] + ) + run_warnings += override_notes + pool = pool[:auto_fields] + if len(pool) < 2: + return StatAnomalyResult( + status="insufficient_data", + detector=detector, + method=method, + baseline_size=reference_size, + warnings=[ + *run_warnings, + "Fewer than two categorical fields to correlate — name two with `fields`.", + ], + windows=windows.payload() if windows is not None else None, + slices=slices.payload() if slices is not None else None, + ) + pairs = [(pool[i], pool[j]) for i in range(len(pool)) for j in range(i + 1, len(pool))] + if len(pairs) > max_pairs: + run_warnings.append( + f"{len(pairs)} field pairs were eligible; only the first {max_pairs} were " + f"scanned (the pair cap). Name fewer fields to choose which." + ) + pairs = pairs[:max_pairs] + + # Phase 1: one scan per pair. Row layout: a, b, reference count (the + # baseline window's, or the scope total), then per frame + # (count, first_ts, first_evt). + # cand = (a_field, b_field, a, b, ref, [(cnt, first, evt) per frame]) + cands: list[tuple[str, str, str, str, int, list[tuple[int, Any, Any]]]] = [] + evaluated_pairs = 0 + for a_field, b_field in pairs: + params: dict[str, Any] = {**base_params} + bind_offset_params(source_offsets, params) + a_expr = _col_expr(a_field, params, field_mappings, prefix="fk0") + b_expr = _col_expr(b_field, params, field_mappings, prefix="fk1") + params["cap"] = max_rows_per_pair + if windows is not None: + bp, sps = _window_preds(windows, params, source_offsets) + frame_blocks = ",\n ".join( + f"countIf({sp}) AS f{i}_cnt," + f" minIf({eff}, {sp}) AS f{i}_first," + f" toString(argMinIf(event_id, {eff}, {sp})) AS f{i}_evt" + for i, sp in enumerate(sps) + ) + f_sum = " + ".join(f"f{i}_cnt" for i in range(n_frames)) + sql = f""" + SELECT + {a_expr} AS a, + {b_expr} AS b, + countIf({bp}) AS ref, + {frame_blocks} + FROM {db}.events + WHERE case_id = {{cid:String}} + AND has({{src:Array(String)}}, source_id) + AND {a_expr} != '' AND {b_expr} != '' + AND {VESTIGO_NOT_SENTINEL_SQL} + AND ({" OR ".join([bp, *sps])}) + GROUP BY a, b + ORDER BY (ref + {f_sum}) DESC, a ASC, b ASC + LIMIT {{cap:UInt32}} + {heavy_scan_settings()} + """ + elif slices is not None: + slice_expr = slices.index_sql(eff, params) + frame_blocks = ",\n ".join( + f"countIf(slice = {i}) AS f{i}_cnt," + f" minIf(ts, slice = {i}) AS f{i}_first," + f" toString(argMinIf(event_id, ts, slice = {i})) AS f{i}_evt" + for i in range(n_frames) + ) + sql = f""" + SELECT + a, + b, + count() AS ref, + {frame_blocks} + FROM ( + SELECT {a_expr} AS a, {b_expr} AS b, {eff} AS ts, event_id, + {slice_expr} AS slice + FROM {db}.events + WHERE case_id = {{cid:String}} + AND has({{src:Array(String)}}, source_id) + AND {a_expr} != '' AND {b_expr} != '' + AND {VESTIGO_NOT_SENTINEL_SQL} + ) + GROUP BY a, b + ORDER BY ref DESC, a ASC, b ASC + LIMIT {{cap:UInt32}} + {heavy_scan_settings()} + """ + rows = self.ch.client.query(sql, parameters=params).result_rows + if not rows: + continue + evaluated_pairs += 1 + if len(rows) >= max_rows_per_pair: + run_warnings.append( + f"Pair ({a_field}, {b_field}) hit the {max_rows_per_pair}-row candidate " + f"cap — rules are mined over its {max_rows_per_pair} highest-volume value " + f"pairs only; a rule whose antecedent lives in the tail is not tested." + ) + for row in rows: + a, b = row[0], row[1] + if not a or not b: + continue + per_frame = [ + (int(row[3 + i * 3]), row[4 + i * 3], row[5 + i * 3]) for i in range(n_frames) + ] + cands.append((a_field, b_field, str(a), str(b), int(row[2]), per_frame)) + + # Phase 2: mine rules in both directions and test each (rule, frame). + # tests: (rule, frame, g, p, n_x, v, n_ref, v_ref, top_violator, top_count, first_ts, evt) + tests: list[tuple[Any, ...]] = [] + by_pair: dict[tuple[str, str], list[tuple[str, str, int, list[tuple[int, Any, Any]]]]] = ( + defaultdict(list) + ) + for a_field, b_field, a, b, ref, per_frame in cands: + by_pair[(a_field, b_field)].append((a, b, ref, per_frame)) + for (a_field, b_field), rows in by_pair.items(): + for ante_field, cons_field, oriented in ( + (a_field, b_field, [(a, b, ref, pf) for a, b, ref, pf in rows]), + (b_field, a_field, [(b, a, ref, pf) for a, b, ref, pf in rows]), + ): + table: dict[str, dict[str, tuple[int, list[tuple[int, Any, Any]]]]] = defaultdict( + dict + ) + for x, y, ref, pf in oriented: + table[x][y] = (ref, pf) + for x, outcomes in table.items(): + support = sum(ref for ref, _pf in outcomes.values()) + if support < min_support: + continue + y_star, (conforming, _pf) = max( + outcomes.items(), key=lambda kv: (kv[1][0], kv[0]) + ) + confidence = conforming / support + if confidence < rule_confidence: + continue + v_ref_total = support - conforming + for fi in range(n_frames): + n_x = sum(pf[fi][0] for _ref, pf in outcomes.values()) + if n_x <= 0: + continue + conf_f = outcomes[y_star][1][fi][0] + v_f = n_x - conf_f + if windows is not None: + n_ref, v_ref = support, v_ref_total + else: + # Leave-one-out: the complement is every other slice. + n_ref, v_ref = support - n_x, v_ref_total - v_f + if n_ref <= 0: + continue + if v_f <= 0: + continue + violators = [ + (y, pf[fi]) + for y, (_ref, pf) in outcomes.items() + if y != y_star and pf[fi][0] > 0 + ] + top_y, (top_cnt, _t, _e) = max(violators, key=lambda kv: (kv[1][0], kv[0])) + first_ts, evt = min( + ((pf[1], pf[2]) for _y, pf in violators if pf[1] is not None), + default=(None, None), + ) + g = _g_statistic(v_ref, n_ref - v_ref, v_f, n_x - v_f) + tests.append( + ( + (ante_field, cons_field, x, y_star, support, confidence), + fi, + g, + _chi2_sf_df1(g), + n_x, + v_f, + n_ref, + v_ref, + top_y, + top_cnt, + first_ts, + evt, + ) + ) + qvals = _bh_qvalues([t[3] for t in tests]) + m_tests = len(tests) + + # Phase 3: FDR + effect floor, build findings. + findings: list[CorrelationFinding] = [] + for test, q in zip(tests, qvals, strict=True): + (rule, fi, g, p, n_x, v_f, n_ref, v_ref, top_y, top_cnt, first_ts, evt) = test + if q > fdr_q: + continue + ante_field, cons_field, x, y_star, support, confidence = rule + rate_f = v_f / n_x + rate_ref = v_ref / n_ref if v_ref > 0 else 0.5 / n_ref + ratio = rate_f / rate_ref + if ratio < min_ratio: + continue + first_seen = _present_ts(first_ts) + evt_id = str(evt) if evt else None + details: dict[str, Any] = { + "detector": detector, + "method": method, + "fields": [ante_field, cons_field], + "values": [x, y_star], + "value": f"{x} ⇒ {y_star}", + "antecedent_field": ante_field, + "antecedent_value": x, + "consequent_field": cons_field, + "consequent_value": y_star, + "confidence": round(confidence, 4), + "support": support, + "count": n_x, + "violations": v_f, + "violation_rate": round(rate_f, 6), + "reference_count": n_ref, + "reference_violations": v_ref, + "reference_violation_rate": round(v_ref / n_ref, 6), + "rate_ratio": round(ratio, 4), + "top_violator": top_y, + "top_violator_count": top_cnt, + "g_statistic": round(g, 4), + "p_value": round(p, 6), + "q_value": round(q, 6), + "m_tests": m_tests, + "q_threshold": fdr_q, + "min_ratio": min_ratio, + "rule_confidence": rule_confidence, + "min_support": min_support, + "first_seen": first_seen, + "allowlist_field": COMBO_FIELD_SEP.join([ante_field, cons_field]), + "allowlist_value": COMBO_VALUE_SEP.join([x, y_star]), + } + if windows is not None: + window = windows.suspects[fi] + details.update( + { + "baseline_size": reference_size, + "window_label": window.label, + "window_start": ensure_utc(window.start).isoformat(), + "window_end": ensure_utc(window.end).isoformat(), + } + ) + elif slices is not None: + details.update(self._slice_details(slices, fi)) + details["rest_slices"] = slices.k - 1 + findings.append( + CorrelationFinding( + fields=[ante_field, cons_field], + values=[x, y_star], + value=f"{x} ⇒ {y_star}", + confidence=round(confidence, 4), + support=support, + count=n_x, + violations=v_f, + baseline_count=n_ref, + baseline_violations=v_ref, + violation_rate=round(rate_f, 6), + baseline_violation_rate=round(v_ref / n_ref, 6), + rate_ratio=round(ratio, 4), + top_violator=top_y, + top_violator_count=top_cnt, + g_statistic=round(g, 4), + p_value=round(p, 6), + q_value=round(q, 6), + score=round(g, 4), + first_seen=first_seen, + event_id=evt_id, + event=_stub_event(evt_id, case_id, first_seen), + details=details, + ) + ) + return self._finalize_findings( + findings, + detector=detector, + method=method, + total_events=reference_size, + evaluated_fields=evaluated_pairs, + exclude_event_ids=exclude_event_ids, + limit=limit, + case_id=case_id, + source_ids=source_ids, + allowlist=allowlist, + warnings=run_warnings, + windows=windows, + slices=slices, + ) + @gated_heavy_scan def find_sequence_motifs( self, diff --git a/tests/test_analysis_plan.py b/tests/test_analysis_plan.py index 71ab7ead..c07b889c 100644 --- a/tests/test_analysis_plan.py +++ b/tests/test_analysis_plan.py @@ -78,6 +78,7 @@ def test_numeric_range_gated_off_without_numeric_fields(cfg): # comparison with no baseline is the one thing an analyst action repairs. "transition_time", "time_of_day", + "value_correlation", ) @@ -106,7 +107,7 @@ def test_self_frame_slice_methods_need_more_than_one_instant(cfg): plans = _by_id( build_plan(_inputs(frame="self", has_active_baseline=False, span_seconds=0.0), cfg) ) - for method in ("proportion_shift", "value_distribution_drift"): + for method in ("proportion_shift", "value_distribution_drift", "value_correlation"): assert plans[method].status == "not_applicable" assert plans[method].reason_facts == {"span_seconds": 0.0, "slices": cfg.stat_self_slices} # A short span merely yields thin slices; the run warns, the gate offers. diff --git a/tests/test_anomaly_stats.py b/tests/test_anomaly_stats.py index cc7b42f7..6a7c8a51 100644 --- a/tests/test_anomaly_stats.py +++ b/tests/test_anomaly_stats.py @@ -7100,3 +7100,254 @@ def test_habit_allowlist_suppresses_the_value(): ) assert result.status == "ok" assert result.results == [] + + +# --------------------------------------------------------------------------- +# value_correlation — detector (D13) +# --------------------------------------------------------------------------- + +# Baseline frame query order: count, window totals, then one scan per field +# pair. Row layout: a, b, ref (baseline count), then per suspect window +# (cnt, first_ts, first_evt). Self frame: count, timestamp range, slice +# totals, then one scan per pair whose rows are a, b, ref (scope total), then +# per slice (cnt, first_ts, first_evt). + +_CORR_COLS = ["a", "b", "ref", "f0_cnt", "f0_first", "f0_evt"] + + +def _corr_responses( + total: int, window_totals: tuple[int, int], pair_rows: list[list[tuple]] +) -> list[FakeQueryResult]: + out = [ + FakeQueryResult(result_rows=[(total,)], column_names=["count()"]), + FakeQueryResult(result_rows=[window_totals], column_names=["bl_total", "w0_total"]), + ] + out += [FakeQueryResult(result_rows=rows, column_names=_CORR_COLS) for rows in pair_rows] + return out + + +_USER_HOST = ["attr:user", "attr:computer_name"] + + +def test_correlation_parameter_validation(): + svc = _svc([]) + with pytest.raises(ValueError, match="two fields"): + svc.find_value_correlations("c1", ["s1"], fields=["attr:user"], windows=_seq_windows()) + with pytest.raises(ValueError, match="rule_confidence"): + svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, rule_confidence=1.5) + with pytest.raises(ValueError, match="min_ratio"): + svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, min_ratio=1.0) + assert svc.ch.client._calls == [] + + +def test_correlation_no_data(): + svc = _svc([FakeQueryResult(result_rows=[(0,)], column_names=["count()"])]) + result = svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, windows=_seq_windows()) + assert result.status == "no_data" + assert result.detector == "value_correlation" + + +def test_correlation_baseline_reports_a_broken_rule(): + """m.okonkwo ⇒ WKS-004 holds over 6000 baseline logons and breaks in the + suspect window; b.moreau splits two homes and never forms a rule.""" + t_jump = datetime(2024, 1, 17, 6, 0, tzinfo=UTC) + t_file = datetime(2024, 1, 17, 3, 0, tzinfo=UTC) + rows = [ + ("m.okonkwo", "WKS-004", 6000, 5, datetime(2024, 1, 16, tzinfo=UTC), "e-home"), + ("m.okonkwo", "JUMP-01", 0, 8, t_jump, "e-jump"), + ("m.okonkwo", "FILE-01", 0, 4, t_file, "e-file"), + ("b.moreau", "WKS-002", 3000, 200, datetime(2024, 1, 16, tzinfo=UTC), "e-b2"), + ("b.moreau", "WKS-003", 2900, 190, datetime(2024, 1, 16, tzinfo=UTC), "e-b3"), + ] + svc = _svc(_corr_responses(20_000, (12_000, 8_000), [rows])) + result = svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, windows=_seq_windows()) + assert result.status == "ok" + assert result.method == "rule-g-test" + assert result.baseline_size == 12_000 + assert len(result.results) == 1 + r = result.results[0] + assert r.fields == ["attr:user", "attr:computer_name"] + assert r.values == ["m.okonkwo", "WKS-004"] + assert r.value == "m.okonkwo ⇒ WKS-004" + assert r.confidence == 1.0 + assert r.support == 6000 + assert r.count == 17 + assert r.violations == 12 + assert r.baseline_violations == 0 + assert r.baseline_count == 6000 + assert abs(r.violation_rate - 12 / 17) < 1e-6 + assert r.top_violator == "JUMP-01" + assert r.top_violator_count == 8 + assert r.q_value <= 0.05 + assert r.score == r.g_statistic > 0 + # The earliest violating occurrence is the representative event. + assert r.event_id == "e-file" + assert r.first_seen is not None and r.first_seen.startswith("2024-01-17T03:00") + assert r.details["window_label"] == "incident" + assert r.details["allowlist_field"] == "attr:user,attr:computer_name" + assert r.details["allowlist_value"] == "m.okonkwo\x1fWKS-004" + # count, window totals, one pair scan. + assert len(svc.ch.client._calls) == 3 + params = svc.ch.client._all_parameters[2] + assert params["fk0"] == "user" and params["fk1"] == "computer_name" + assert params["cap"] == 5000 + + +def test_correlation_mines_both_directions(): + """The reverse rule WKS-004 ⇒ m.okonkwo also holds and also breaks when a + second account appears on that host in the window.""" + t = datetime(2024, 1, 17, tzinfo=UTC) + rows = [ + ("m.okonkwo", "WKS-004", 6000, 300, t, "e1"), + ("c.nakamura", "WKS-004", 0, 40, t, "e2"), + ("c.nakamura", "WKS-009", 5000, 250, t, "e3"), + ] + svc = _svc(_corr_responses(20_000, (12_000, 8_000), [rows])) + result = svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, windows=_seq_windows()) + assert result.status == "ok" + rules = {(tuple(r.fields), r.value) for r in result.results} + assert (("attr:computer_name", "attr:user"), "WKS-004 ⇒ m.okonkwo") in rules + # c.nakamura ⇒ WKS-009 is broken too (40 of 290 window events elsewhere). + assert (("attr:user", "attr:computer_name"), "c.nakamura ⇒ WKS-009") in rules + # m.okonkwo ⇒ WKS-004 is intact: no violations in the window. + assert (("attr:user", "attr:computer_name"), "m.okonkwo ⇒ WKS-004") not in rules + + +def test_correlation_floors_support_confidence_and_effect(): + t = datetime(2024, 1, 17, tzinfo=UTC) + rows = [ + # Support 10 < 20: no rule however clean. + ("thin", "H1", 10, 2, t, "e1"), + ("thin", "H2", 0, 5, t, "e2"), + # Confidence 0.9 < 0.95: no rule. + ("split", "H1", 900, 50, t, "e3"), + ("split", "H2", 100, 50, t, "e4"), + # A real rule whose violation rate merely creeps from 4% to 6%: under + # the 2x effect floor even where the G-test would pass. + ("creep", "H1", 9600, 940, t, "e5"), + ("creep", "H2", 400, 60, t, "e6"), + # A real rule with no window violations: nothing to report. + ("steady", "H1", 5000, 400, t, "e7"), + ] + svc = _svc(_corr_responses(50_000, (30_000, 20_000), [rows])) + result = svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, windows=_seq_windows()) + assert result.status == "ok" + assert [r.value for r in result.results if r.fields[0] == "attr:user"] == [] + + +def test_correlation_pair_cap_and_auto_fields(): + """Six auto-picked fields make fifteen pairs; the cap scans the first + three and says so.""" + t = datetime(2024, 1, 17, tzinfo=UTC) + inventory = [(f"attr:f{i}", 10, 1000) for i in range(6)] + [("attr:pid", 990, 1000)] + pair_rows = [[("x", "y", 100, 5, t, "e")] for _ in range(3)] + svc = _svc(_corr_responses(1000, (700, 300), pair_rows)) + result = svc.find_value_correlations( + "c1", + ["s1"], + windows=_seq_windows(), + inventory=inventory, + inventory_total=1000, + max_pairs=3, + ) + assert result.status == "ok" + assert len(svc.ch.client._calls) == 5 + assert any("15 field pairs" in w and "first 3" in w for w in result.warnings) + # An explicit two-field list is exactly one pair, no cap warning. + svc = _svc(_corr_responses(1000, (700, 300), [[("x", "y", 100, 5, t, "e")]])) + result = svc.find_value_correlations( + "c1", ["s1"], fields=_USER_HOST, windows=_seq_windows(), max_pairs=3 + ) + assert not any("pair cap" in w for w in result.warnings) + + +def test_correlation_sql_shape(): + t = datetime(2024, 1, 17, tzinfo=UTC) + client = RecordingClient( + _corr_responses(20_000, (12_000, 8_000), [[("u", "h", 6000, 5, t, "e")]]) + ) + svc = StatisticalAnomalyService.__new__(StatisticalAnomalyService) + svc.ch = FakeClickHouseStore(client) + svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, windows=_seq_windows()) + sql = client.full_queries[2] + assert "attributes[{fk0:String}]" in sql and "attributes[{fk1:String}]" in sql + assert "GROUP BY a, b" in sql + assert "LIMIT {cap:UInt32}" in sql + assert "countIf(" in sql and "argMinIf(event_id" in sql + params = client._all_parameters[2] + assert params["b0"] == "2024-01-01 00:00:00.000" + assert params["w0s"] == "2024-01-16 00:00:00.000" + + +def test_correlation_self_frame_tests_each_slice_against_the_rest(): + """Without a baseline: rules over the whole scope, one leave-one-out + G-test per (rule, slice).""" + k = 4 + span = (datetime(2024, 1, 1, tzinfo=UTC), datetime(2024, 1, 5, tzinfo=UTC)) + t = datetime(2024, 1, 4, 3, 0, tzinfo=UTC) + cols = ["a", "b", "ref"] + [f"f{i}_{c}" for i in range(k) for c in ("cnt", "first", "evt")] + # m.okonkwo: 6000 home logons spread over four slices; 12 stray logons, + # all in slice 3 → the rule breaks in slice 3 against slices 0–2. + rows = [ + ( + "m.okonkwo", + "WKS-004", + 6000, + 1500, + None, + None, + 1500, + None, + None, + 1500, + None, + None, + 1500, + None, + None, + ), + ("m.okonkwo", "JUMP-01", 12, 0, None, None, 0, None, None, 0, None, None, 12, t, "e-j"), + ] + svc = _svc( + [ + FakeQueryResult(result_rows=[(6012,)], column_names=["count()"]), + FakeQueryResult(result_rows=[span], column_names=["min_ts", "max_ts"]), + FakeQueryResult(result_rows=[(i, 1503) for i in range(k)], column_names=["slice", "n"]), + FakeQueryResult(result_rows=rows, column_names=cols), + ] + ) + result = svc.find_value_correlations("c1", ["s1"], fields=_USER_HOST, self_slices=k) + assert result.status == "ok", result.warnings + assert result.method == "self-rule-g-test" + assert result.windows is None and result.slices is not None + assert result.slices["k"] == k + assert [r.value for r in result.results] == ["m.okonkwo ⇒ WKS-004"] + r = result.results[0] + assert r.count == 1512 + assert r.violations == 12 + # The complement: the other three slices, with no violations at all. + assert r.baseline_count == 4500 + assert r.baseline_violations == 0 + assert r.details["slice_index"] == 3 + assert r.details["rest_slices"] == k - 1 + assert r.event_id == "e-j" + assert "baseline_size" not in r.details + assert len(svc.ch.client._calls) == 4 + + +def test_correlation_allowlist_suppresses_the_rule(): + t = datetime(2024, 1, 17, tzinfo=UTC) + rows = [ + ("m.okonkwo", "WKS-004", 6000, 5, t, "e1"), + ("m.okonkwo", "JUMP-01", 0, 8, t, "e2"), + ] + svc = _svc(_corr_responses(20_000, (12_000, 8_000), [rows])) + result = svc.find_value_correlations( + "c1", + ["s1"], + fields=_USER_HOST, + windows=_seq_windows(), + allowlist={("attr:user,attr:computer_name", "m.okonkwo\x1fWKS-004")}, + ) + assert result.status == "ok" + assert result.results == [] diff --git a/tests/test_demo_detector_coverage_clickhouse.py b/tests/test_demo_detector_coverage_clickhouse.py index 894944da..6ed66d6e 100644 --- a/tests/test_demo_detector_coverage_clickhouse.py +++ b/tests/test_demo_detector_coverage_clickhouse.py @@ -39,6 +39,10 @@ # at 03:00. Named rather than auto-picked so the assertion is about the # detector, not the recommender's field order on this corpus. "find_time_of_day_habits": {"fields": ["attr:program", "attr:computer_name"]}, + # Every human account has its home workstations; the contractor has one, + # and the rule "m.okonkwo ⇒ WKS-004" breaks on every host the intrusion + # takes the account to. + "find_value_correlations": {"fields": ["attr:user", "attr:computer_name"]}, } #: Detectors that score a baseline against suspect windows. @@ -55,6 +59,7 @@ "find_sequence_novelty", "find_transition_times", "find_time_of_day_habits", + "find_value_correlations", ) #: Detectors with no baseline/suspect split at all. @@ -122,6 +127,7 @@ def test_windowed_detector_finds_something(demo, ch_store, method): "find_sequence_novelty": {"series_field": "attr:computer_name"}, "find_transition_times": {"series_field": "attr:computer_name", "partition_field": "attr:user"}, "find_time_of_day_habits": {"fields": ["attr:program"]}, + "find_value_correlations": {"fields": ["attr:user", "attr:computer_name"]}, } From 8441bc1773079bb95dd73c5aab3a9f299731c52f Mon Sep 17 00:00:00 2001 From: Overcuriousity Date: Wed, 16 Sep 2026 14:16:02 +0000 Subject: [PATCH 4/5] fix(api): the anomalies query endpoint accepts bucket_minutes as a plain int GET /anomalies?bucket_minutes=30 answered 422 "Input should be 15, 30, 60, 120, 180 or 240": a query string cannot carry an int literal, so the Literal-typed Query param rejected every value. Both the query param and the tag request body take an int now and the runner's own membership check answers 422 for anything that does not divide the day. Found by the end-to-end verification of the 1.20 detectors; the storable-params test gains the three new methods' out-of-bounds cases. Co-Authored-By: Claude Fable 5.1 --- src/vestigo/api/routers/events.py | 11 ++++++----- tests/test_timeline_detectors_api.py | 3 +++ 2 files changed, 9 insertions(+), 5 deletions(-) diff --git a/src/vestigo/api/routers/events.py b/src/vestigo/api/routers/events.py index b4665561..27404e80 100644 --- a/src/vestigo/api/routers/events.py +++ b/src/vestigo/api/routers/events.py @@ -3466,11 +3466,12 @@ async def list_anomalies( "one stream per source." ), ), - bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = Query( + bucket_minutes: int | None = Query( default=None, description=( - "time_of_day only: width of the wall-clock buckets the day is cut into. " - "Omit to use the server default." + "time_of_day only: width of the wall-clock buckets the day is cut into — " + "15, 30, 60, 120, 180 or 240 (the runner rejects anything else with 422; a " + "query string cannot carry an int literal). Omit to use the server default." ), ), timezone: str | None = Query( @@ -3822,9 +3823,9 @@ class TagAnomaliesRequest(BaseModel): default=None, description="transition_time only: the stream whose transitions are timed (e.g. 'attr:user').", ) - bucket_minutes: Literal[15, 30, 60, 120, 180, 240] | None = Field( + bucket_minutes: int | None = Field( default=None, - description="time_of_day only: width of the wall-clock buckets the day is cut into.", + description="time_of_day only: width of the wall-clock buckets (15, 30, 60, 120, 180 or 240).", ) timezone: str | None = Field( default=None, diff --git a/tests/test_timeline_detectors_api.py b/tests/test_timeline_detectors_api.py index 96a65172..6bd78247 100644 --- a/tests/test_timeline_detectors_api.py +++ b/tests/test_timeline_detectors_api.py @@ -114,6 +114,9 @@ def test_unknown_method_is_rejected(client, admin_bootstrap): ("frequency", {"z_threshold": -1}), # gt=0 ("sequence_novelty", {"ngram_size": 9}), # le=5 ("log_template", {"order": "sideways"}), # not in the Literal + ("time_of_day", {"bucket_minutes": 7}), # does not divide the day + ("value_correlation", {"rule_confidence": 1.5}), # le=1 + ("transition_time", {"min_ratio": 1.0}), # gt=1 ], ) def test_params_the_findings_endpoint_rejects_are_not_storable( From b2ef552d3d588f22707e41ae90d0c7fc593cd82d Mon Sep 17 00:00:00 2001 From: overcuriousity Date: Mon, 21 Sep 2026 16:10:06 +0200 Subject: [PATCH 5/5] fix(detectors): four review findings on the 1.20 detector cluster (#378) - Tagging transition_time / time_of_day / value_correlation findings no longer 500s: tag_anomalies writes a reason per new finding type instead of falling through to the frequency-spike branch, whose fields they lack. - The Bucket knob works for every option: the form sends numeric choices as numbers, and the params model also accepts "15" for 15, since a stored detector keeps the client's shape. - transition_time judges a zero-length observation by its source's timestamp resolution. In a source with sub-second timestamps it stays instant; in a whole-second source it means "under a second" and is judged and scored at that one-second bound, so it no longer undercuts any floor at score 1.0. Held-back pairs are disclosed in a warning; the finding records timestamp_resolution and observed_upper_bound_seconds. - Settings refuse detector defaults the detectors refuse: habit bucket widths that do not divide the day, unknown IANA zones, a correlation min_ratio <= 1 or an fdr_q outside (0, 1]. The bucket list and zone check live in core/time_of_day.py, shared by the settings and the detector. Co-Authored-By: Claude Opus 5 (1M context) --- docs/ANOMALY_DETECTION.md | 16 +- frontend/src/api/types.ts | 9 +- .../components/analysis/FindingEvidence.tsx | 22 ++- .../components/analysis/MethodKnobForm.tsx | 3 +- .../components/analysis/method-registry.ts | 6 + frontend/src/test/methodRegistry.test.ts | 17 ++ src/vestigo/api/routers/analysis.py | 13 ++ src/vestigo/api/routers/events.py | 46 ++++++ src/vestigo/core/config.py | 19 ++- src/vestigo/core/time_of_day.py | 41 +++++ src/vestigo/db/anomaly_stats.py | 156 +++++++++++++----- tests/test_analysis_findings_api.py | 10 ++ tests/test_anomaly_stats.py | 78 +++++++++ tests/test_events_router.py | 149 +++++++++++++++++ tests/test_settings_api.py | 42 +++++ 15 files changed, 573 insertions(+), 54 deletions(-) create mode 100644 src/vestigo/core/time_of_day.py diff --git a/docs/ANOMALY_DETECTION.md b/docs/ANOMALY_DETECTION.md index 84432aba..25ebda37 100644 --- a/docs/ANOMALY_DETECTION.md +++ b/docs/ANOMALY_DETECTION.md @@ -2586,6 +2586,20 @@ is flagged. A zero floor cannot be undercut — with second-resolution timestamp zero-length transitions are routine — so such pairs are skipped rather than scored, and the run says how many. +A zero-length *observation* is only as precise as the timestamps it is the difference +of. In a source that records sub-second timestamps (any event with a non-zero +millisecond part) a 0 ms gap is genuinely instant and judged as such. In a source whose +every timestamp is a whole second — Plaso CSV, syslog — `12:00:05 → 12:00:05` says only +that the move took *under a second*: judged as instant it would undercut any positive +floor (`0 × min_ratio` beats everything) and rank first, though it may have taken +0.99 s against a 1 s floor. So there it is judged and scored at its **one-second +bound**: flagged only when one second still undercuts the floor by `min_ratio`×, scored +`1 − 1 / reference_seconds`, with `speedup` a lower bound. Pairs the bound holds back +are counted in a warning. The resolution is probed per source, only for a source that +produced a zero-length candidate, and the probe stops at its first sub-second +timestamp. A zero-length finding carries `timestamp_resolution` (`second` / +`sub-second`) and `observed_upper_bound_seconds` (`1.0`, or `null` when instant). + Per source: the `stat_transition_max_candidates` fastest pairs (baseline frame: per suspect window) are fetched fastest-first, with a warning when the cap is hit — the cap keeps the fastest, which are the ones the question is about; in the baseline frame the @@ -2593,7 +2607,7 @@ floor is then learned for exactly those candidate pairs. On a multi-source scope floor is the minimum over every source's reference and counts are summed, so a pair that is slow in one source and fast in another is judged against the fast one. -**Score = 1 − observed / reference**, in `[0, 1]`: 1.0 is an instantaneous transition, +**Score = 1 − observed / reference** (observed at its bound, above), in `[0, 1]`: 1.0 is an instantaneous transition, 0.5 is exactly twice as fast as the floor, and nothing under `1 − 1/min_ratio` is reported. The representative event is the **arriving** event of the fastest transition (the `b` side), and `first_seen` is its timestamp. Findings carry `observed_seconds`, diff --git a/frontend/src/api/types.ts b/frontend/src/api/types.ts index c5ef5056..22da2934 100644 --- a/frontend/src/api/types.ts +++ b/frontend/src/api/types.ts @@ -811,13 +811,18 @@ export interface TransitionTimeFinding { /** The floor it undercut, in seconds. */ reference_seconds: number; reference_kind: "baseline-min" | "next-fastest"; - /** reference_seconds ÷ observed_seconds; null when the observation is instant. */ + /** + * reference_seconds ÷ observed_seconds; null when the observation is instant. + * A zero gap between whole-second timestamps is judged at its one-second + * bound (`details.timestamp_resolution === "second"`), so this is then a + * lower bound. + */ speedup: number | null; /** Transitions of this pair in the window (baseline frame) or the timeline (self). */ count: number; /** Transitions the floor was learned from. */ baseline_count: number; - /** 1 − observed ÷ reference; 1.0 = instantaneous. */ + /** 1 − observed ÷ reference (observed at its bound, see `speedup`); 1.0 = instantaneous. */ score: number; /** Timestamp of the arriving event of the fastest transition. */ first_seen: string | null; diff --git a/frontend/src/components/analysis/FindingEvidence.tsx b/frontend/src/components/analysis/FindingEvidence.tsx index 4751dcbf..9b6a47f3 100644 --- a/frontend/src/components/analysis/FindingEvidence.tsx +++ b/frontend/src/components/analysis/FindingEvidence.tsx @@ -405,9 +405,12 @@ export function FindingEvidence({ finding }: { finding: MethodResult }) { timezone={finding.timezone} /> ); - case "transition_time": + case "transition_time": { // The claim is one duration against one floor, both measured; which - // floor is in the label, since the two answer different questions. + // floor is in the label, since the two answer different questions. A + // zero gap between whole-second timestamps means "under a second", and + // was judged at that bound — the figure shows the bound, not an instant. + const bounded = finding.details.timestamp_resolution === "second"; return ( ); + } case "timestamp_order": return (
diff --git a/frontend/src/components/analysis/MethodKnobForm.tsx b/frontend/src/components/analysis/MethodKnobForm.tsx index 512e1b94..b636e2d4 100644 --- a/frontend/src/components/analysis/MethodKnobForm.tsx +++ b/frontend/src/components/analysis/MethodKnobForm.tsx @@ -44,7 +44,8 @@ export function buildParams( } const value = (raw[knob.param] ?? "").trim(); if (!value) continue; - out[knob.param] = knob.kind === "number" ? Number(value) : value; + const numeric = knob.kind === "number" || (knob.kind === "choice" && knob.numeric); + out[knob.param] = numeric ? Number(value) : value; } return out; } diff --git a/frontend/src/components/analysis/method-registry.ts b/frontend/src/components/analysis/method-registry.ts index 231066ef..90a49d6d 100644 --- a/frontend/src/components/analysis/method-registry.ts +++ b/frontend/src/components/analysis/method-registry.ts @@ -102,6 +102,11 @@ export interface MethodKnob { placeholder: string; /** `kind: "choice"` only — the options, first one the default. */ options?: { value: string; label: string }[]; + /** + * `kind: "choice"` only — the option values are numbers, sent as such. The + * API's int `Literal` for them refuses the string a `