diff --git a/src/incident-evidence.test.ts b/src/incident-evidence.test.ts new file mode 100644 index 0000000..de66717 --- /dev/null +++ b/src/incident-evidence.test.ts @@ -0,0 +1,126 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { buildIncidentEvidence, type IncidentEvidenceInput } from "./incident-evidence.js"; + +const SVC = "api"; +const unhealthy = (o: object = {}) => ({ service: SVC, at: "2026-10-04T10:05:00Z", status: "unhealthy" as const, release: "r2", samples: 40, ...o }); +const deploy = (o: object = {}) => ({ service: SVC, at: "2026-10-04T10:00:00Z", revision: "r2", outcome: "succeeded" as const, ...o }); +const errLog = (o: object = {}) => ({ service: SVC, at: "2026-10-04T10:06:00Z", release: "r2", excerpt: "ok\nERROR handler failed\nok", ...o }); +const run = (o: Partial = {}) => buildIncidentEvidence({ service: SVC, ...o }); +const NONE = { service: SVC, state: "none" as const }; + +test("deploy before onset with matching release and error logs: medium hypothesis, correlation caveat, still escalates", () => { + const e = run({ health: [unhealthy()], deploys: [deploy()], logs: [errLog()], migration: NONE }); + assert.equal(e.assessment, "unhealthy-observed"); + const h = e.hypotheses[0]!; + assert.equal(h.confidence, "medium"); + assert.match(h.caveat, /not a cause/); + assert.equal(e.proposal.status, "candidate-only"); + assert.equal(e.escalation.required, true); + assert.ok(h.basis.every((id) => e.facts.some((f) => f.id === id)), "every basis id is a recorded fact"); +}); + +test("deploy correlation without release attribution or log corroboration stays low and says what would confirm", () => { + const e = run({ health: [unhealthy({ release: undefined })], deploys: [deploy()], logs: [], migration: NONE }); + assert.equal(e.hypotheses[0]!.confidence, "low"); + assert.ok(e.hypotheses[0]!.wouldConfirm.length >= 2); + assert.ok(e.unknowns.some((u) => u.about === "log evidence")); + assert.ok(e.escalation.actions.length > 0); +}); + +test("NEGATIVE: no hypothesis is ever high confidence, however much evidence agrees", () => { + const e = run({ + health: [unhealthy(), unhealthy({ at: "2026-10-04T10:10:00Z" })], + deploys: [deploy()], + logs: [errLog({ excerpt: "ERROR column foo does not exist" })], + migration: { service: SVC, state: "applied", at: "2026-10-04T09:59:00Z", ids: ["003"] }, + }); + assert.ok(e.hypotheses.length >= 2); + for (const h of e.hypotheses) assert.notEqual(h.confidence as string, "high"); + assert.equal(e.proposal.status, "candidate-only"); +}); + +test("NEGATIVE: another service's unhealthy metrics, logs, deploys and migrations are set aside, not evidence", () => { + const e = run({ + health: [unhealthy({ service: "billing" })], + deploys: [deploy({ service: "billing" })], + logs: [errLog({ service: "billing" })], + migration: { service: "billing", state: "applied", at: "2026-10-04T09:00:00Z", ids: ["9"] }, + }); + assert.equal(e.assessment, "unknown"); + assert.equal(e.setAside.length, 4); + assert.equal(e.hypotheses.length, 0); + assert.ok(!e.facts.some((f) => /billing/.test(f.statement))); +}); + +test("NEGATIVE: unavailable telemetry, no readings, and thin traffic are unknown, never healthy", () => { + const cases = [ + [{ service: SVC, at: "2026-10-04T10:00:00Z", status: "unavailable" as const, detail: "token rejected" }], + [], + [{ service: SVC, at: "2026-10-04T10:00:00Z", status: "healthy" as const, samples: 0 }], + [{ service: SVC, at: "2026-10-04T10:00:00Z", status: "healthy" as const, samples: 2 }], + ]; + for (const health of cases) { + const e = run({ health }); + assert.equal(e.assessment, "unknown", JSON.stringify(health)); + assert.equal(e.escalation.required, true); + assert.ok(e.unknowns.some((u) => u.about === "current health")); + } +}); + +test("healthy with enough samples is observed healthy and needs no escalation, with the window caveat in the reason", () => { + const e = run({ health: [{ service: SVC, at: "2026-10-04T10:00:00Z", status: "healthy", samples: 50 }], deploys: [], migration: NONE, logs: [] }); + assert.equal(e.assessment, "healthy-observed"); + assert.equal(e.escalation.required, false); + assert.match(e.escalation.reason, /window only/); +}); + +test("unhealthy with no correlating change: no hypothesis, an unknown cause, no proposal, escalation", () => { + const e = run({ health: [unhealthy()], deploys: [deploy({ at: "2026-09-01T00:00:00Z" })], migration: NONE, logs: [] }); + assert.equal(e.hypotheses.length, 0); + assert.ok(e.unknowns.some((u) => u.about === "cause")); + assert.match(e.proposal.note, /no proposal/); + assert.equal(e.escalation.required, true); +}); + +test("NEGATIVE: a deploy AFTER onset, or a failed deploy, is not offered as the cause", () => { + const after = run({ health: [unhealthy()], deploys: [deploy({ at: "2026-10-04T10:30:00Z" })], migration: NONE }); + assert.equal(after.hypotheses.length, 0); + const failed = run({ health: [unhealthy()], deploys: [deploy({ outcome: "failed" })], migration: NONE }); + assert.equal(failed.hypotheses.length, 0); +}); + +test("unread migration state and unread deploy history are unknowns, not 'none'", () => { + const e = run({ health: [unhealthy()] }); + assert.ok(e.unknowns.some((u) => u.about === "migration state")); + assert.ok(e.unknowns.some((u) => u.about === "recent deploys")); + assert.ok(!e.facts.some((f) => f.kind === "migration")); +}); + +test("schema-looking log lines raise a migration hypothesis to medium only with the log", () => { + const mig = { service: SVC, state: "applied" as const, at: "2026-10-04T09:59:00Z", ids: ["003"] }; + const without = run({ health: [unhealthy()], migration: mig, deploys: [] }); + assert.equal(without.hypotheses.find((h) => /schema/.test(h.statement))!.confidence, "low"); + const withLog = run({ health: [unhealthy()], migration: mig, deploys: [], logs: [errLog({ excerpt: "relation users_v2 does not exist" })] }); + assert.equal(withLog.hypotheses.find((h) => /schema/.test(h.statement))!.confidence, "medium"); +}); + +test("timeline: ordered by time, unparseable timestamps kept last and flagged, nothing dropped", () => { + const e = run({ + health: [unhealthy({ at: "not-a-time" }), unhealthy({ at: "2026-10-04T10:20:00Z" })], + deploys: [deploy({ at: "2026-10-04T10:00:00Z" })], + logs: [errLog({ at: "2026-10-04T10:06:00Z" })], + migration: NONE, + }); + const dated = e.timeline.map((t) => t.at).filter((a): a is string => a !== null); + assert.deepEqual(dated, [...dated].sort()); + const bad = e.timeline.find((t) => t.orderUnknown)!; + assert.equal(bad.at, null); + assert.ok(e.timeline.indexOf(bad) >= dated.length, "unordered entry sorts after ordered ones"); + assert.equal(e.timeline.length, e.facts.length); +}); + +test("log excerpts are clamped in facts", () => { + const e = run({ logs: [errLog({ excerpt: "ERROR " + "x".repeat(5000) })] }); + assert.ok(e.facts.find((f) => f.kind === "log")!.statement.length < 900); +}); diff --git a/src/incident-evidence.ts b/src/incident-evidence.ts new file mode 100644 index 0000000..1a9ffc6 --- /dev/null +++ b/src/incident-evidence.ts @@ -0,0 +1,353 @@ +/** + * S17 part: incident evidence model and read-only timeline. PURE and UNWIRED. + * + * WHY. incidents.ts grades a scan run's write-up. Before (or beside) a model + * reads anything, the observations themselves — health, a log excerpt, recent + * deploys, migration state — should be turned into a record that keeps three + * things apart that diagnoses habitually blur: what was OBSERVED, what is only + * a HYPOTHESIS, and what is UNKNOWN. A page that shows "error rate up after + * deploy" as a cause has already done the damage. + * + * RULES (each has a negative control in the test file). + * - Only observations attributed to the incident's own service become facts. + * Another service's numbers are listed as set aside, never evidence (the + * wrong-service story at the head of incidents.ts). + * - Unavailable telemetry, absent traffic or too few samples is UNKNOWN, never + * healthy. Healthy needs enough samples in a window. + * - A hypothesis never reaches "high" confidence here (the type forbids it), + * states what would confirm it, and is correlation by construction: a deploy + * before onset is a candidate, not a cause. "medium" needs two independent + * kinds of evidence pointing the same way. + * - A proposal is always `candidate-only`; nothing here can mark a fix certain + * or authorised. Low or missing confidence sets an actionable escalation. + * - Nothing observed is dropped: unparseable timestamps keep their entry, + * listed after the ordered ones, with an unknown-order flag. + * + * No I/O, no clock (the caller passes `now` inputs as observation times), no + * model. Log text is CLAMPED here but must be redacted by the caller, as with + * every other excerpt in incidents.ts. + */ + +export type HealthStatus = "healthy" | "unhealthy" | "unavailable"; + +export interface HealthObservation { + service: string; + at: string; + status: HealthStatus; + /** Release the reading was attributed to, when the source knows. */ + release?: string; + /** Requests/samples behind the reading; 0 or missing = no traffic evidence. */ + samples?: number; + /** Why unavailable. */ + detail?: string; +} + +export interface LogObservation { + service: string; + at: string; + release?: string; + excerpt: string; +} + +export interface DeployObservation { + service: string; + at: string; + revision: string; + outcome: "succeeded" | "failed" | "unknown"; +} + +export interface MigrationObservation { + service: string; + /** "applied": ids were applied at/after `at`; "none": read and nothing applied; "unknown": not read. */ + state: "applied" | "none" | "unknown"; + at?: string; + ids?: string[]; +} + +export interface IncidentEvidenceInput { + /** The service the incident was attributed to. */ + service: string; + health?: HealthObservation[]; + logs?: LogObservation[]; + deploys?: DeployObservation[]; + migration?: MigrationObservation; + /** Minimum samples for a healthy reading to count. Default 5. */ + minSamples?: number; + /** A deploy older than this before onset is not offered as a candidate. Default 24h. */ + deployWindowMs?: number; +} + +export type EvidenceKind = "health" | "log" | "deploy" | "migration"; + +export interface Fact { + id: string; + kind: EvidenceKind; + at?: string; + /** What was observed, stated without interpretation. */ + statement: string; +} + +export interface Hypothesis { + id: string; + statement: string; + /** Never "high". */ + confidence: "low" | "medium"; + basis: string[]; + caveat: string; + wouldConfirm: string[]; +} + +export interface Unknown { + id: string; + about: string; + /** What to do to find out. */ + toResolve: string; +} + +export interface SetAside { + kind: EvidenceKind; + service: string; + reason: string; +} + +export interface TimelineEntry { + at: string | null; + /** True when the timestamp could not be parsed; the entry sorts last. */ + orderUnknown: boolean; + kind: EvidenceKind; + factId: string; + text: string; +} + +export type Assessment = "unhealthy-observed" | "healthy-observed" | "unknown"; + +export interface IncidentEvidence { + service: string; + assessment: Assessment; + facts: Fact[]; + hypotheses: Hypothesis[]; + unknowns: Unknown[]; + setAside: SetAside[]; + timeline: TimelineEntry[]; + /** A proposal can only ever be a candidate for a human to authorise. */ + proposal: { status: "candidate-only"; basis: string[]; note: string }; + escalation: { required: boolean; reason: string; actions: string[] }; +} + +export const LOG_EXCERPT_MAX = 600; +const ERROR_LINE = /\b(error|exception|fatal|panic|traceback|unhandled|5\d\d)\b/i; +const SCHEMA_LINE = /(column|relation|table)\b.*\b(does not exist|not found|missing)|\bmigration\b.*\b(fail|error|pending|lock)/i; + +const clamp = (s: string, n: number): string => (s.length <= n ? s : `${s.slice(0, n)}…[+${s.length - n} chars]`); +const ts = (s: string | undefined): number | null => { + if (s === undefined) return null; + const t = Date.parse(s); + return Number.isNaN(t) ? null : t; +}; +const sameService = (a: string, b: string): boolean => a.trim() !== "" && a.trim().toLowerCase() === b.trim().toLowerCase(); + +export function buildIncidentEvidence(input: IncidentEvidenceInput): IncidentEvidence { + const minSamples = input.minSamples ?? 5; + const deployWindowMs = input.deployWindowMs ?? 24 * 3600_000; + const facts: Fact[] = []; + const unknowns: Unknown[] = []; + const setAside: SetAside[] = []; + const hypotheses: Hypothesis[] = []; + const add = (kind: EvidenceKind, at: string | undefined, statement: string): Fact => { + const f: Fact = { id: `f${facts.length + 1}`, kind, ...(at !== undefined ? { at } : {}), statement }; + facts.push(f); + return f; + }; + const unk = (about: string, toResolve: string): void => { + unknowns.push({ id: `u${unknowns.length + 1}`, about, toResolve }); + }; + + // ---- attribution: only the incident's own service becomes evidence + const mine = (kind: EvidenceKind, xs: T[] | undefined): T[] => { + const out: T[] = []; + for (const x of xs ?? []) { + if (sameService(x.service, input.service)) out.push(x); + else setAside.push({ kind, service: x.service, reason: `attributed to "${x.service}", not the incident's service "${input.service}"` }); + } + return out; + }; + const health = mine("health", input.health); + const logs = mine("log", input.logs); + const deploys = mine("deploy", input.deploys); + let migration = input.migration; + if (migration !== undefined && !sameService(migration.service, input.service)) { + setAside.push({ kind: "migration", service: migration.service, reason: `attributed to "${migration.service}", not the incident's service "${input.service}"` }); + migration = undefined; + } + + // ---- health + const unhealthy: { f: Fact; h: HealthObservation }[] = []; + let healthyCounted = 0; + let healthyThin = 0; + let unavailable = 0; + for (const h of health) { + if (h.status === "unavailable") { + unavailable++; + add("health", h.at, `health read of ${h.service} was unavailable${h.detail ? ` (${clamp(h.detail, 120)})` : ""}`); + } else if (h.status === "unhealthy") { + const f = add("health", h.at, `${h.service} reported unhealthy${h.release ? ` on release ${h.release}` : ""} (${h.samples ?? 0} samples)`); + unhealthy.push({ f, h }); + } else if ((h.samples ?? 0) >= minSamples) { + healthyCounted++; + add("health", h.at, `${h.service} reported healthy over ${h.samples} samples`); + } else { + healthyThin++; + add("health", h.at, `${h.service} reported healthy but over only ${h.samples ?? 0} samples (< ${minSamples}); not counted as healthy`); + } + } + let assessment: Assessment; + if (unhealthy.length > 0) assessment = "unhealthy-observed"; + else if (healthyCounted > 0) assessment = "healthy-observed"; + else assessment = "unknown"; + if (assessment === "unknown") { + unk( + "current health", + health.length === 0 + ? "no health reading for this service: read Observe (or the target's health endpoint) for it" + : unavailable > 0 && healthyThin === 0 + ? "health reads were unavailable: fix telemetry access, then re-read" + : `healthy readings had fewer than ${minSamples} samples: wait for traffic or read a longer window`, + ); + } + if (assessment === "unhealthy-observed" && healthyCounted > 0) { + unk("whether the fault is ongoing", "healthy and unhealthy readings both exist: order them by time and re-read the latest window"); + } + + // ---- logs (observed lines only; no interpretation beyond a pattern count) + const errorLogs: { f: Fact; l: LogObservation }[] = []; + const schemaLogs: { f: Fact; l: LogObservation }[] = []; + for (const l of logs) { + const lines = l.excerpt.split("\n"); + const errs = lines.filter((x) => ERROR_LINE.test(x)); + const schema = lines.filter((x) => SCHEMA_LINE.test(x)); + const f = add("log", l.at, `log excerpt${l.release ? ` (release ${l.release})` : ""} has ${errs.length} of ${lines.length} line(s) matching an error pattern: ${clamp(errs[0] ?? lines[0] ?? "", LOG_EXCERPT_MAX)}`); + if (errs.length > 0) errorLogs.push({ f, l }); + if (schema.length > 0) schemaLogs.push({ f, l }); + } + if (logs.length === 0) unk("log evidence", "no log excerpt for this service: read the service logs around the first unhealthy reading"); + + // ---- deploys + const deployFacts = new Map(); + for (const d of deploys) { + deployFacts.set(d, add("deploy", d.at, `deploy of revision ${d.revision} ${d.outcome}`)); + } + if (input.deploys === undefined) unk("recent deploys", "deploy history was not read: read the delivery records for this service"); + + // ---- migration + let migFact: Fact | undefined; + if (migration === undefined) { + unk("migration state", "migration state was not read for this service: read the applied-migration list from the target"); + } else if (migration.state === "unknown") { + unk("migration state", "migration state could not be read: read the applied-migration list from the target"); + } else if (migration.state === "none") { + add("migration", migration.at, "no migration was applied in the observed period"); + } else { + migFact = add("migration", migration.at, `migration(s) applied: ${(migration.ids ?? []).join(", ") || "(ids not recorded)"}`); + } + + // ---- hypotheses: onset = the earliest unhealthy reading we can order + const onsets = unhealthy.map((u) => ({ u, t: ts(u.h.at) })).filter((x): x is { u: typeof x.u; t: number } => x.t !== null).sort((a, b) => a.t - b.t); + const onset = onsets[0]; + if (assessment === "unhealthy-observed") { + if (onset === undefined) { + unk("onset time", "no unhealthy reading has a parseable timestamp, so nothing can be correlated with it"); + } else { + const cands = deploys + .map((d) => ({ d, t: ts(d.at) })) + .filter((x): x is { d: DeployObservation; t: number } => x.t !== null && x.d.outcome === "succeeded" && x.t <= onset.t && onset.t - x.t <= deployWindowMs) + .sort((a, b) => b.t - a.t); + const deploy = cands[0]; + if (deploy !== undefined) { + const rel = deploy.d.revision; + const attributed = onset.u.h.release === rel; + const corroborated = errorLogs.some((e) => e.l.release === rel); + const strong = attributed && corroborated; + hypotheses.push({ + id: `h${hypotheses.length + 1}`, + statement: `the fault may have been introduced by the deploy of ${rel}`, + confidence: strong ? "medium" : "low", + basis: [deployFacts.get(deploy.d)!.id, onset.u.f.id, ...(strong ? errorLogs.filter((e) => e.l.release === rel).map((e) => e.f.id) : [])], + caveat: "timing correlation only; a deploy before onset is a candidate, not a cause", + wouldConfirm: [ + ...(attributed ? [] : [`a health reading attributed to release ${rel} (the first unhealthy reading names ${onset.u.h.release ?? "no release"})`]), + ...(corroborated ? [] : [`error log lines from release ${rel}`]), + "the unhealthy reading clearing after recovery to the previous release", + ], + }); + } else if (deploys.length > 0 || input.deploys !== undefined) { + add("deploy", undefined, "no succeeded deploy of this service precedes the first unhealthy reading within the window"); + } + if (migFact !== undefined && migration?.state === "applied") { + const mt = ts(migration.at); + const before = mt !== null && mt <= onset.t; + if (before) { + const corroborated = schemaLogs.length > 0; + hypotheses.push({ + id: `h${hypotheses.length + 1}`, + statement: "the fault may be related to a schema change applied before onset", + confidence: corroborated ? "medium" : "low", + basis: [migFact.id, onset.u.f.id, ...schemaLogs.map((s) => s.f.id)], + caveat: "a migration before onset is a candidate, not a cause", + wouldConfirm: [ + ...(corroborated ? [] : ["log lines naming a missing column/relation or a failed migration"]), + "the failure reproducing against the pre-migration schema", + ], + }); + } else if (mt === null) { + unk("migration timing", "the migration observation has no parseable time, so it cannot be ordered against onset"); + } + } + if (hypotheses.length === 0) { + unk("cause", "no deploy or migration correlates with the first unhealthy reading: widen the window or investigate dependencies and load"); + } + } + } + + // ---- timeline: ordered where we can, never dropping + const timeline: TimelineEntry[] = facts.map((f) => { + const t = ts(f.at); + return { at: t === null ? null : new Date(t).toISOString(), orderUnknown: f.at !== undefined && t === null, kind: f.kind, factId: f.id, text: f.statement }; + }); + const dated = timeline.filter((e) => e.at !== null).sort((a, b) => Date.parse(a.at!) - Date.parse(b.at!) || a.factId.localeCompare(b.factId, "en", { numeric: true })); + const undated = timeline.filter((e) => e.at === null); + + // ---- escalation: required unless healthy evidence stands, or the best hypothesis is medium and nothing is unresolved + const best = hypotheses.some((h) => h.confidence === "medium") ? "medium" : hypotheses.length > 0 ? "low" : "none"; + const actions = [...unknowns.map((u) => u.toResolve), ...hypotheses.filter((h) => h.confidence === "low").flatMap((h) => h.wouldConfirm)]; + let reason: string; + let required: boolean; + if (assessment === "healthy-observed" && unknowns.every((u) => u.about !== "current health")) { + required = false; + reason = "healthy readings with enough samples; no fault observed (observation window only)"; + } else if (assessment === "unknown") { + required = true; + reason = "health is unknown; unknown is not healthy"; + } else if (best === "medium") { + required = true; + reason = "best hypothesis is medium confidence and correlational: a human must confirm before any change"; + } else { + required = true; + reason = best === "low" ? "only low-confidence hypotheses" : "no supported hypothesis"; + } + + return { + service: input.service, + assessment, + facts, + hypotheses, + unknowns, + setAside, + timeline: [...dated, ...undated], + proposal: { + status: "candidate-only", + basis: hypotheses.map((h) => h.id), + note: hypotheses.length === 0 ? "no proposal: nothing supports one" : "any change is a candidate to investigate read-only first; it is not certain and is not authorised", + }, + escalation: { required, reason, actions: [...new Set(actions)] }, + }; +} diff --git a/src/recovery-compat.test.ts b/src/recovery-compat.test.ts new file mode 100644 index 0000000..0ac380d --- /dev/null +++ b/src/recovery-compat.test.ts @@ -0,0 +1,62 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { checkRecoveryCompat, planIsSpecific, type RecoveryCompatInput } from "./recovery-compat.js"; + +const base = (o: Partial = {}): RecoveryCompatInput => ({ + retainedRevision: "v1", + retainedKnownMigrations: ["001", "002"], + appliedMigrations: [{ id: "001", class: "expand" }, { id: "002", class: "expand" }], + ...o, +}); + +test("no newer migration: compatible, and never claims the artifact rollback itself", () => { + const r = checkRecoveryCompat(base()); + assert.equal(r.verdict, "compatible"); + assert.equal(r.provesArtifactRollback, false); +}); + +test("newer additive migrations stay compatible", () => { + const r = checkRecoveryCompat(base({ appliedMigrations: [{ id: "001", class: "expand" }, { id: "003", class: "expand" }] })); + assert.equal(r.verdict, "compatible"); + assert.deepEqual(r.newerMigrations, ["003"]); +}); + +test("NEGATIVE: a newer contract or irreversible migration is incompatible, not compatible", () => { + for (const cls of ["contract", "irreversible"] as const) { + const r = checkRecoveryCompat(base({ appliedMigrations: [{ id: "003", class: cls }] })); + assert.equal(r.verdict, "incompatible", cls); + assert.deepEqual(r.blocking, ["003"]); + assert.equal(r.recoveryPlan, "missing"); + } +}); + +test("a specific recovery plan is recorded but does not turn the verdict into a pass; a vague one is missing", () => { + const ok = checkRecoveryCompat(base({ appliedMigrations: [{ id: "003", class: "irreversible" }], recoveryPlan: { strategy: "backup-restore", reference: "backup-2026-10-04" } })); + assert.equal(ok.verdict, "incompatible"); + assert.equal(ok.recoveryPlan, "present"); + const vague = checkRecoveryCompat(base({ appliedMigrations: [{ id: "003", class: "irreversible" }], recoveryPlan: { strategy: "forward-fix", reference: " " } })); + assert.equal(vague.recoveryPlan, "missing"); + assert.equal(planIsSpecific(undefined), false); +}); + +test("NEGATIVE: unknown is hold, never compatible (each unread side, missing class, no retained revision)", () => { + assert.equal(checkRecoveryCompat(base({ appliedMigrations: undefined })).verdict, "hold"); + assert.equal(checkRecoveryCompat(base({ retainedKnownMigrations: undefined })).verdict, "hold"); + assert.equal(checkRecoveryCompat(base({ retainedRevision: " " })).verdict, "hold"); + const unclassified = checkRecoveryCompat(base({ appliedMigrations: [{ id: "003" }] })); + assert.equal(unclassified.verdict, "hold"); + assert.deepEqual(unclassified.unclassified, ["003"]); + assert.equal(checkRecoveryCompat(base({ appliedMigrations: [{ id: "003", class: "unknown" }] })).verdict, "hold"); +}); + +test("a known-destructive migration is not hidden behind an unclassified one", () => { + const r = checkRecoveryCompat(base({ appliedMigrations: [{ id: "003" }, { id: "004", class: "contract" }] })); + assert.equal(r.verdict, "incompatible"); + assert.deepEqual(r.unclassified, ["003"]); +}); + +test("repeated ids count once; an empty applied list is read-and-none, not unknown", () => { + assert.equal(checkRecoveryCompat(base({ appliedMigrations: [] })).verdict, "compatible"); + const r = checkRecoveryCompat(base({ appliedMigrations: [{ id: "003", class: "expand" }, { id: "003", class: "expand" }] })); + assert.deepEqual(r.newerMigrations, ["003"]); +}); diff --git a/src/recovery-compat.ts b/src/recovery-compat.ts new file mode 100644 index 0000000..9da147a --- /dev/null +++ b/src/recovery-compat.ts @@ -0,0 +1,148 @@ +/** + * S15 part: migration-aware recovery check. PURE and UNWIRED. + * + * WHY. delivery.ts rolls an ARTIFACT back to the retained version and reads + * the target back. That proves the old code is serving; it says nothing about + * whether the old code can still work against the database the newer release + * has already migrated. Rolling back across a dropped column turns an + * incident into a data incident, and the read-back still says "retained + * version serving". Artifact rollback and data recovery are separate questions + * (S15), so this module answers only the second one and never claims the first. + * + * THE RULE. Compatible only when we KNOW both sides: the migrations the + * retained release understands, and the migrations actually applied to the + * target. Anything unread, unclassified or unrecorded is `hold` — unknown is + * never compatible. A migration the retained release does not know about is + * safe only when it is classified additive ("expand"); a "contract" or + * "irreversible" one makes the rollback `incompatible` unless the operator has + * recorded a specific recovery plan, and even then the verdict stays blocked: + * a plan is a precondition to a human decision, not a pass. + * + * This module decides nothing about authority and executes nothing. Callers + * (a future delivery gate, behind a flag, default off) feed it observations. + */ + +export type MigrationClass = "expand" | "contract" | "irreversible" | "unknown"; + +export interface AppliedMigration { + id: string; + /** How the migration was classified when authored. Missing = unknown, not expand. */ + class?: MigrationClass; +} + +export type RecoveryStrategy = "backup-restore" | "forward-fix" | "expand-contract"; + +export interface RecoveryPlan { + strategy: RecoveryStrategy; + /** Where the plan lives (runbook, ticket, backup id). A plan with no reference is not specific. */ + reference: string; +} + +export interface RecoveryCompatInput { + /** The retained release rollback would restore. Empty = nothing to check against. */ + retainedRevision: string; + /** Migration ids the retained release ships/understands. undefined = not read. */ + retainedKnownMigrations?: readonly string[]; + /** Migrations applied to the target database, in order. undefined = not read. */ + appliedMigrations?: readonly AppliedMigration[]; + /** Operator-recorded plan for data that artifact rollback cannot restore. */ + recoveryPlan?: RecoveryPlan; +} + +export type RecoveryCompatVerdict = "compatible" | "incompatible" | "hold"; + +export interface RecoveryCompatResult { + verdict: RecoveryCompatVerdict; + /** One sentence for the operator. */ + reason: string; + /** Applied migrations the retained release does not know. */ + newerMigrations: string[]; + /** The subset that blocks rollback (contract/irreversible). */ + blocking: string[]; + /** The subset whose class is missing or unrecognised. */ + unclassified: string[]; + /** What is not known, for `hold`. */ + unknowns: string[]; + /** Only meaningful when blocking is non-empty. */ + recoveryPlan: "present" | "missing" | "not-needed"; + /** Always false: this check never proves the artifact rollback itself. */ + provesArtifactRollback: false; +} + +const CLASSES: ReadonlySet = new Set(["expand", "contract", "irreversible"]); + +function result( + verdict: RecoveryCompatVerdict, + reason: string, + parts: Partial> = {}, +): RecoveryCompatResult { + return { + verdict, + reason, + newerMigrations: [], + blocking: [], + unclassified: [], + unknowns: [], + recoveryPlan: "not-needed", + ...parts, + provesArtifactRollback: false, + }; +} + +export function planIsSpecific(plan: RecoveryPlan | undefined): boolean { + if (plan === undefined) return false; + const ok: RecoveryStrategy[] = ["backup-restore", "forward-fix", "expand-contract"]; + return ok.includes(plan.strategy) && typeof plan.reference === "string" && plan.reference.trim() !== ""; +} + +export function checkRecoveryCompat(input: RecoveryCompatInput): RecoveryCompatResult { + const unknowns: string[] = []; + if (input.retainedRevision.trim() === "") unknowns.push("no retained revision recorded"); + if (input.retainedKnownMigrations === undefined) unknowns.push("the migrations the retained release understands were not read"); + if (input.appliedMigrations === undefined) unknowns.push("the migrations applied to the target were not read"); + if (unknowns.length > 0) { + return result("hold", `held: ${unknowns.join("; ")} — unknown is not compatible`, { unknowns }); + } + + const known = new Set(input.retainedKnownMigrations); + const applied = input.appliedMigrations ?? []; + const seen = new Set(); + const newer: AppliedMigration[] = []; + for (const m of applied) { + if (seen.has(m.id)) continue; // a repeated id is one migration + seen.add(m.id); + if (!known.has(m.id)) newer.push(m); + } + const newerMigrations = newer.map((m) => m.id); + const unclassified = newer.filter((m) => m.class === undefined || !CLASSES.has(m.class)).map((m) => m.id); + const blocking = newer.filter((m) => m.class === "contract" || m.class === "irreversible").map((m) => m.id); + + // Blocking beats unclassified: a known-destructive migration is a definite + // answer, and holding would hide it behind "unknown". + if (blocking.length > 0) { + const plan = planIsSpecific(input.recoveryPlan) ? "present" : "missing"; + return result( + "incompatible", + `rollback to ${input.retainedRevision} crosses ${blocking.length} contract/irreversible migration(s) it does not know (${blocking.join(", ")}); ` + + (plan === "present" + ? `a ${input.recoveryPlan?.strategy} plan is recorded (${input.recoveryPlan?.reference}) but a plan is a precondition for a human decision, not a pass` + : "no specific recovery plan is recorded"), + { newerMigrations, blocking, unclassified, recoveryPlan: plan }, + ); + } + if (unclassified.length > 0) { + return result( + "hold", + `held: ${unclassified.length} migration(s) newer than ${input.retainedRevision} have no recognised class (${unclassified.join(", ")}) — unclassified is not additive`, + { newerMigrations, unclassified, unknowns: unclassified.map((id) => `class of migration ${id}`) }, + ); + } + if (newerMigrations.length === 0) { + return result("compatible", `no applied migration is newer than ${input.retainedRevision}; data compatibility holds (artifact rollback is separately unproven)`); + } + return result( + "compatible", + `${newerMigrations.length} newer migration(s) are all additive (expand); ${input.retainedRevision} can run against them (artifact rollback is separately unproven)`, + { newerMigrations }, + ); +}