From 03678e0045440848582bfa9a826c7c632c01615e Mon Sep 17 00:00:00 2001 From: Christopher Nelson Date: Sun, 20 Sep 2026 01:58:27 -0400 Subject: [PATCH] feat: add reproducible zero-swarm comparison harness --- ROADMAP.md | 10 +- apps/game-api/package.json | 1 + apps/game-api/src/simulation-service.ts | 142 +++-- apps/game-api/src/swarm-comparison-cli.ts | 89 +++ apps/game-api/src/swarm-comparison.test.ts | 64 +++ apps/game-api/src/swarm-comparison.ts | 597 +++++++++++++++++++++ docs/ARCHITECTURE.md | 9 + docs/EXPERIMENT_ARCHIVE.md | 8 + docs/TESTING.md | 7 + docs/ZERO_SWARM_COMPARISON.md | 97 ++++ package.json | 1 + 11 files changed, 975 insertions(+), 50 deletions(-) create mode 100644 apps/game-api/src/swarm-comparison-cli.ts create mode 100644 apps/game-api/src/swarm-comparison.test.ts create mode 100644 apps/game-api/src/swarm-comparison.ts create mode 100644 docs/ZERO_SWARM_COMPARISON.md diff --git a/ROADMAP.md b/ROADMAP.md index 48368d2..d84040b 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -6,10 +6,12 @@ The active zero-swarm sequence supersedes the earlier social-agent direction for new experiments while retaining `legacy-multi-agent` as a comparison mode. PRs A-C established the Jev reflex seam, Agent Zero planner/tick, and World Lab presentation. PR D makes directives persistent across ticks and wakes Zero on -fixed review and material events. PR E will add a stronger optional deterministic -human-pressure simulator. PR F will compare zero-swarm, legacy, and deterministic -worker baselines before deciding which legacy systems to retire. See ADRs -0028-0031. Historical milestones below remain as implementation history. +fixed review and material events. PR E added optional deterministic +`trail-hunter-v1` capture pressure while retaining `casual-cleaner`. PR F adds +seeded offline comparisons of zero-swarm, legacy, and deterministic worker +policies. The first comparison retains legacy mode pending real-provider trials; +see `docs/ZERO_SWARM_COMPARISON.md`. See ADRs 0028-0032. Historical milestones +below remain as implementation history. ## Experimental Patient Zero coordinator diff --git a/apps/game-api/package.json b/apps/game-api/package.json index 40fa5ae..a163349 100644 --- a/apps/game-api/package.json +++ b/apps/game-api/package.json @@ -5,6 +5,7 @@ "type": "module", "scripts": { "build": "esbuild src/server.ts --bundle --platform=node --format=esm --outfile=dist/server.js", + "compare:offline": "node --import tsx src/swarm-comparison-cli.ts", "dev": "tsx watch src/server.ts", "lint": "eslint .", "start": "node dist/server.js", diff --git a/apps/game-api/src/simulation-service.ts b/apps/game-api/src/simulation-service.ts index 860d723..804101c 100644 --- a/apps/game-api/src/simulation-service.ts +++ b/apps/game-api/src/simulation-service.ts @@ -115,7 +115,10 @@ import { ObservationHistory } from './observation-history'; import { AttemptAccounting } from './attempt-accounting'; import { chooseReflexWorldAction, + compileReflexObservation, ReflexSelectionCancelledError, + type CompiledReflexObservation, + type ReflexSelection, } from './reflex-execution'; function attemptAccountingForScenario( @@ -135,6 +138,19 @@ const MAX_TURN_HISTORY = 120; const MAX_WORLD_EVENT_HISTORY = 120; const DEFAULT_EXPERIMENT_RETENTION = 5_000; +function chooseDeterministicWorkerAction( + compiled: CompiledReflexObservation, + selectCandidateId: (compiled: CompiledReflexObservation) => string, +): ReflexSelection { + const action = compiled.actions.get(selectCandidateId(compiled)); + return { + action: action ?? { type: 'wait' }, + observation: compiled.observation, + decision: null, + cognitionSource: 'deterministic-fallback', + }; +} + export function selectMostRecentPatientZeroThreats< T extends { eventId: string; occurredAt: string }, >(events: readonly T[]): T[] { @@ -292,6 +308,13 @@ export interface SimulationServiceOptions { /** Separate strategic and reflex cognition used only by zero-swarm-v1. */ swarmPlanner?: SwarmPlanner; reflexProvider?: ReflexProvider; + /** + * Offline comparison seam: choose one opaque, already legal candidate without + * calling a reflex provider. Production zero-swarm execution leaves this unset. + */ + deterministicWorkerCandidateSelector?: ( + compiled: CompiledReflexObservation, + ) => string; now?: () => string; createEventId?: () => string; createExperimentId?: () => string; @@ -304,6 +327,8 @@ export class SimulationService { readonly #provider: AgentProvider; readonly #swarmPlanner: SwarmPlanner | undefined; readonly #reflexProvider: ReflexProvider | undefined; + readonly #deterministicWorkerCandidateSelector: + ((compiled: CompiledReflexObservation) => string) | undefined; readonly #now: () => string; readonly #createEventId: () => string; readonly #createExperimentId: () => string; @@ -354,6 +379,7 @@ export class SimulationService { provider, swarmPlanner, reflexProvider, + deterministicWorkerCandidateSelector, now = () => new Date().toISOString(), createEventId = () => crypto.randomUUID(), createExperimentId = () => crypto.randomUUID(), @@ -369,6 +395,8 @@ export class SimulationService { this.#provider = provider; this.#swarmPlanner = swarmPlanner; this.#reflexProvider = reflexProvider; + this.#deterministicWorkerCandidateSelector = + deterministicWorkerCandidateSelector; this.#now = now; this.#createEventId = createEventId; this.#createExperimentId = createExperimentId; @@ -1333,7 +1361,7 @@ export class SimulationService { if (result.outcome === 'lost-tick') continue; const applied = applyCommunication( state, - preTickState, + playerAdvance.state, agentId, result.decision.decision.communication, context, @@ -1587,8 +1615,18 @@ export class SimulationService { ) replanReasons.push('roster-changed'); const replan = replanReasons.length > 0; - // Planning ticks reserve Zero plus workers; directive reuse only reserves workers. - if (!this.#attemptAccounting.reserve(agents.length - (replan ? 0 : 1))) { + // Planning ticks reserve Zero plus Jev workers. The explicit deterministic + // comparison seam only reserves Zero's planner call; it never dispatches a + // reflex provider attempt. + const requiredAttempts = this.#deterministicWorkerCandidateSelector + ? replan + ? 1 + : 0 + : agents.length - (replan ? 0 : 1); + if ( + requiredAttempts > 0 && + !this.#attemptAccounting.reserve(requiredAttempts) + ) { this.#status = 'budget-exhausted'; throw new SimulationValidationError( 'experiment_budget_exhausted', @@ -1696,43 +1734,48 @@ export class SimulationService { ({ agentId }) => agentId === worker.id, )!; this.#activeAgentId = worker.id; - const choice = await chooseReflexWorldAction( - candidate, - directive, - this.#reflexProvider, - { - history: { - previousCell: this.#lastSwarmPositions.get(worker.id), - recentCleanedCells: this.#simulatedPlayerEvents - .filter( - ( - event, - ): event is Extract< - SimulatedPlayerEvent, - { type: 'hex-disinfected' } - > => event.type === 'hex-disinfected', - ) - .slice(-6) - .map(({ cell }) => cell), - captureAlerts: captureAlertsFrom(playerAdvance.events), - territoryDelta: - this.#lastSwarmTerritoryDeltas.get(worker.id) ?? 0, - recentActionOutcome: this.#swarmTicks - .at(-1) - ?.workers.find(({ agentId }) => agentId === worker.id) - ?.actionResult?.accepted - ? 'success' - : 'unknown', - }, - signal: controller.signal, - deadlineAtMs, - accounting: this.#attemptAccounting, - initialPermitReserved: true, - intendedTickNumber: tickNumber, - intendedTurnNumber: tickTurnBase + order.indexOf(worker.id) + 1, - now: this.#now, - }, - ); + const history = { + previousCell: this.#lastSwarmPositions.get(worker.id), + recentCleanedCells: this.#simulatedPlayerEvents + .filter( + ( + event, + ): event is Extract< + SimulatedPlayerEvent, + { type: 'hex-disinfected' } + > => event.type === 'hex-disinfected', + ) + .slice(-6) + .map(({ cell }) => cell), + captureAlerts: captureAlertsFrom(playerAdvance.events), + territoryDelta: this.#lastSwarmTerritoryDeltas.get(worker.id) ?? 0, + recentActionOutcome: this.#swarmTicks + .at(-1) + ?.workers.find(({ agentId }) => agentId === worker.id)?.actionResult + ?.accepted + ? ('success' as const) + : ('unknown' as const), + }; + const choice = this.#deterministicWorkerCandidateSelector + ? chooseDeterministicWorkerAction( + compileReflexObservation(candidate, directive, history), + this.#deterministicWorkerCandidateSelector, + ) + : await chooseReflexWorldAction( + candidate, + directive, + this.#reflexProvider, + { + history, + signal: controller.signal, + deadlineAtMs, + accounting: this.#attemptAccounting, + initialPermitReserved: true, + intendedTickNumber: tickNumber, + intendedTurnNumber: tickTurnBase + order.indexOf(worker.id) + 1, + now: this.#now, + }, + ); selected.set(worker.id, choice); } if (controller.signal.aborted) throw new SimulationTurnCancelledError(); @@ -3001,6 +3044,13 @@ export class SimulationService { }; } + #historicalAgent(agentId: AgentId): Agent | undefined { + return ( + this.#state.agents.get(agentId) ?? + this.#initialExperimentAgents.find(({ id }) => id === agentId) + ); + } + #buildObservation( agentId: AgentId, currentCandidatePlayerEvents: readonly SimulatedPlayerEvent[] = [], @@ -3109,7 +3159,7 @@ export class SimulationService { const recentPublicMessages = this.#observationHistory .publicMessages() .map((event) => { - const sender = this.#state.agents.get(event.agentId); + const sender = this.#historicalAgent(event.agentId); if (!sender) throw new Error('A public-message sender does not exist.'); return { eventId: event.id, @@ -3122,8 +3172,8 @@ export class SimulationService { const recentDirectMessages = this.#observationHistory .directMessages(agent.id) .map((event) => { - const sender = this.#state.agents.get(event.agentId); - const recipient = this.#state.agents.get(event.recipientId); + const sender = this.#historicalAgent(event.agentId); + const recipient = this.#historicalAgent(event.recipientId); if (!sender || !recipient) throw new Error('A communication participant does not exist.'); return { @@ -3141,7 +3191,7 @@ export class SimulationService { const recentAllianceMessages = this.#observationHistory .allianceMessages(agent.id) .map((event) => { - const sender = this.#state.agents.get(event.agentId); + const sender = this.#historicalAgent(event.agentId); if (!sender) throw new Error('An alliance-message sender does not exist.'); return { @@ -3156,7 +3206,7 @@ export class SimulationService { const recentZeroMessages = this.#observationHistory .zeroMessages(agent.id) .map((event) => { - const sender = this.#state.agents.get(event.agentId); + const sender = this.#historicalAgent(event.agentId); if (!sender) throw new Error('A Zero-message sender does not exist.'); return { eventId: event.id, @@ -3175,7 +3225,7 @@ export class SimulationService { ? event.previousControllerAgentId : event.controllerAgentId; if (otherAgentId === null) return []; - const otherAgent = this.#state.agents.get(otherAgentId); + const otherAgent = this.#historicalAgent(otherAgentId); if (!otherAgent) throw new Error('A control-change participant does not exist.'); return [ diff --git a/apps/game-api/src/swarm-comparison-cli.ts b/apps/game-api/src/swarm-comparison-cli.ts new file mode 100644 index 0000000..c75e7b2 --- /dev/null +++ b/apps/game-api/src/swarm-comparison-cli.ts @@ -0,0 +1,89 @@ +#!/usr/bin/env node +import { runOfflineComparison } from './swarm-comparison.js'; + +const MAX_TICKS = 60; +const MAX_SEEDS = 16; +const MAX_SEED_LENGTH = 80; + +interface Arguments { + ticks?: number; + seeds?: string[]; +} + +function usage(): string { + return `Usage: pnpm compare:offline [--ticks <1-${MAX_TICKS}>] [--seeds ] + +Runs the fixed, offline zero-swarm comparison fixtures and writes one JSON report +to stdout. It does not load provider credentials, contact provider services, or +write an experiment archive.`; +} + +function parsePositiveInteger(value: string, option: string): number { + if (!/^[0-9]+$/.test(value)) throw new Error(`${option} must be an integer.`); + const parsed = Number(value); + if (parsed < 1 || parsed > MAX_TICKS) + throw new Error(`${option} must be between 1 and ${MAX_TICKS}.`); + return parsed; +} + +function parseSeeds(value: string): string[] { + const seeds = value.split(',').map((seed) => seed.trim()); + if ( + seeds.length === 0 || + seeds.length > MAX_SEEDS || + seeds.some( + (seed) => + seed.length === 0 || + seed.length > MAX_SEED_LENGTH || + !/^[a-zA-Z0-9][a-zA-Z0-9._-]*$/.test(seed), + ) + ) { + throw new Error( + `--seeds must contain 1 to ${MAX_SEEDS} comma-separated identifiers of at most ${MAX_SEED_LENGTH} characters.`, + ); + } + return [...new Set(seeds)].sort((left, right) => + left < right ? -1 : left > right ? 1 : 0, + ); +} + +function parseArguments(input: readonly string[]): Arguments { + const parsed: Arguments = {}; + for (let index = 0; index < input.length; index += 1) { + const option = input[index]; + if (option === '--help' || option === '-h') { + process.stdout.write(`${usage()}\n`); + process.exit(0); + } + const value = input[index + 1]; + if (!value || value.startsWith('--')) + throw new Error(`Option ${option} requires a value.`); + if (option === '--ticks') { + if (parsed.ticks !== undefined) + throw new Error('--ticks was supplied twice.'); + parsed.ticks = parsePositiveInteger(value, '--ticks'); + } else if (option === '--seeds') { + if (parsed.seeds !== undefined) + throw new Error('--seeds was supplied twice.'); + parsed.seeds = parseSeeds(value); + } else throw new Error(`Unknown option: ${option}`); + index += 1; + } + return parsed; +} + +async function main(): Promise { + const options = parseArguments(process.argv.slice(2)); + const report = await runOfflineComparison({ + ...(options.ticks === undefined ? {} : { tickCap: options.ticks }), + ...(options.seeds === undefined ? {} : { seeds: options.seeds }), + }); + process.stdout.write(`${JSON.stringify(report, null, 2)}\n`); +} + +main().catch((error: unknown) => { + process.stderr.write( + `${error instanceof Error ? error.message : String(error)}\n${usage()}\n`, + ); + process.exitCode = 1; +}); diff --git a/apps/game-api/src/swarm-comparison.test.ts b/apps/game-api/src/swarm-comparison.test.ts new file mode 100644 index 0000000..5f9199b --- /dev/null +++ b/apps/game-api/src/swarm-comparison.test.ts @@ -0,0 +1,64 @@ +import { describe, expect, it } from 'vitest'; +import { runOfflineComparison } from './swarm-comparison'; + +describe('runOfflineComparison', () => { + it('is byte-for-byte reproducible for the same deterministic inputs', async () => { + const options = { seeds: ['worker-capture-spawn'], tickCap: 3 }; + await expect(runOfflineComparison(options)).resolves.toEqual( + await runOfflineComparison(options), + ); + }); + + it('covers every comparison mode and retains engine capture telemetry', async () => { + const report = await runOfflineComparison({ + seeds: ['worker-capture-spawn'], + tickCap: 3, + }); + expect(report.variants.map(({ variant }) => variant)).toEqual([ + 'legacy-multi-agent', + 'zero-swarm-v1', + 'deterministic-worker-baseline', + ]); + for (const variant of report.variants) { + const run = variant.runs[0]!; + expect(run.providerAttempts.started).toBeGreaterThanOrEqual(0); + expect(run.providerAttempts.finalized).toBe(run.providerAttempts.started); + expect(run.final.captures).toBeGreaterThan(0); + } + const swarm = report.variants.find( + ({ variant }) => variant === 'zero-swarm-v1', + )!; + expect(swarm.aggregate.totalGenerativeAttempts).toBeGreaterThan(0); + expect(swarm.aggregate.totalReflexAttempts).toBeGreaterThan(0); + const baseline = report.variants.find( + ({ variant }) => variant === 'deterministic-worker-baseline', + )!; + expect(baseline.aggregate.totalReflexAttempts).toBe(0); + expect(baseline.aggregate.totalProviderAttempts).toBe( + baseline.aggregate.totalGenerativeAttempts, + ); + expect( + baseline.runs.flatMap(({ samples }) => + samples.flatMap(({ reflexConfidence }) => reflexConfidence), + ), + ).toEqual([]); + expect( + swarm.runs[0]!.samples.flatMap( + ({ reflexProbabilityDistributions }) => reflexProbabilityDistributions, + ), + ).toEqual(expect.arrayContaining([expect.any(Object)])); + expect(report.costDisclaimer).toContain('no authoritative billed'); + }); + + it('continues legacy observations after captured message participants leave the roster', async () => { + const report = await runOfflineComparison({ + seeds: ['worker-capture-spawn'], + tickCap: 12, + }); + const legacy = report.variants.find( + ({ variant }) => variant === 'legacy-multi-agent', + )!.runs[0]!; + expect(legacy.samples).toHaveLength(12); + expect(legacy.final.captures).toBeGreaterThan(0); + }); +}); diff --git a/apps/game-api/src/swarm-comparison.ts b/apps/game-api/src/swarm-comparison.ts new file mode 100644 index 0000000..dda94e8 --- /dev/null +++ b/apps/game-api/src/swarm-comparison.ts @@ -0,0 +1,597 @@ +import { + BrowserTestAgentProvider, + ReflexProviderError, + type PlannerOptions, + type ReflexDecisionOptions, + type ReflexProvider, + type SwarmPlanner, +} from '@hexzero/agent-runtime'; +import { + assignBehavior, + reflexDecisionSchema, + type CompatibleModel, + type ProviderMetadata, + type ReflexObservation, + type SwarmPlan, + type ZeroStrategicObservation, +} from '@hexzero/shared'; +import { gridDistance } from 'h3-js'; +import { generateDeterministicRoster } from '@hexzero/world-engine'; +import { SimulationService } from './simulation-service'; +import type { CompiledReflexObservation } from './reflex-execution'; + +export type OfflineComparisonVariant = + 'legacy-multi-agent' | 'zero-swarm-v1' | 'deterministic-worker-baseline'; + +export interface OfflineComparisonOptions { + /** Defaults include a known deterministic capture case. */ + seeds?: readonly string[]; + /** Bounded to protect the offline experiment surface from accidental large runs. */ + tickCap?: number; +} + +export interface OfflineComparisonTickSample { + tick: number; + infectedCells: number; + controlledCells: number; + abandonedCells: number; + activeAgents: number; + capturesThisTick: number; + disinfectionsThisTick: number; + terminalStatus: string | null; + attemptsStarted: number; + attemptsFinalized: number; + syntheticInputTokens: number; + syntheticOutputTokens: number; + syntheticLatencyMs: number; + generativeAttempts: number; + zeroPlans: number; + reflexDecisions: number; + workerStalls: number; + replanSignals: number; + reflexConfidence: readonly number[]; + reflexProbabilityDistributions: readonly Readonly>[]; +} + +export interface OfflineComparisonRun { + seed: string; + samples: readonly OfflineComparisonTickSample[]; + final: { + tick: number; + infectedCells: number; + controlledCells: number; + abandonedCells: number; + activeAgents: number; + captures: number; + disinfections: number; + terminalStatus: string | null; + }; + providerAttempts: { started: number; finalized: number }; +} + +export interface OfflineComparisonAggregate { + runCount: number; + totalProviderAttempts: number; + totalGenerativeAttempts: number; + totalReflexAttempts: number; + totalSyntheticInputTokens: number; + totalSyntheticOutputTokens: number; + totalSyntheticLatencyMs: number; + totalCaptures: number; + totalDisinfections: number; + totalWorkerStalls: number; + totalReplanSignals: number; + terminalRuns: number; +} + +export interface OfflineComparisonVariantReport { + variant: OfflineComparisonVariant; + providerKind: string; + runs: readonly OfflineComparisonRun[]; + aggregate: OfflineComparisonAggregate; +} + +export interface OfflineComparisonReport { + formatVersion: 1; + kind: 'offline-deterministic-comparison'; + seeds: readonly string[]; + tickCap: number; + providerDisclaimer: string; + costDisclaimer: string; + variants: readonly OfflineComparisonVariantReport[]; +} + +const DEFAULT_SEEDS = [ + 'worker-capture-spawn', + 'comparison-seed-b', + 'comparison-seed-c', +]; +const DEFAULT_TICK_CAP = 12; +const MAX_TICK_CAP = 60; +const model: CompatibleModel = { + id: 'offline/comparison-model', + name: 'Offline comparison model', + author: 'Hex Zero', + contextLength: 4_096, + inputPricePerToken: '0', + outputPricePerToken: '0', + supportedParameters: [], + isFree: true, + reasoning: { mandatory: false, supportedEfforts: ['low'] }, +}; + +const metadata = (modelId: string): ProviderMetadata => ({ + provider: 'scripted-test', + model: modelId, + latencyMs: 0, + promptTokens: 0, + completionTokens: 0, + totalTokens: 0, + reasoningTokens: 0, + cachedReadTokens: 0, + cacheWriteTokens: 0, + costCredits: 0, +}); + +class OfflinePlanner implements SwarmPlanner { + readonly mode = 'scripted-swarm-test' as const; + readonly configured = true; + async plan( + observation: ZeroStrategicObservation, + selectedModel: string, + options: PlannerOptions = {}, + ) { + const finalize = options.beginAttempt?.('initial'); + if (finalize === null) + throw new Error('Offline planner could not obtain an attempt permit.'); + const zeroAction = + observation.legalZeroActions.find( + ({ action }) => action.type === 'infect', + ) ?? + observation.legalZeroActions.find( + ({ action }) => action.type === 'move', + ) ?? + observation.legalZeroActions.find( + ({ action }) => action.type === 'wait', + ) ?? + observation.legalZeroActions[0]!; + const openTargets = observation.strategicTargetCells.filter((cell) => + observation.cells.some( + ({ cell: knownCell, state }) => knownCell === cell && state === 'open', + ), + ); + const plan: SwarmPlan = { + strategySummary: 'Deterministic offline perimeter expansion.', + zeroActionCandidateId: zeroAction.id, + directives: observation.agents + .filter(({ agentId }) => agentId !== observation.zeroAgentId) + .map((agent, index) => ({ + id: `offline-${observation.tickNumber}-${index}`, + agentId: agent.agentId, + mission: 'expand' as const, + targetCell: + nearestTarget(agent.position, openTargets) ?? agent.position, + priority: 'normal' as const, + riskTolerance: 'medium' as const, + issuedAtTick: observation.tickNumber, + // PR D cadence: directives normally cover five ticks, with events + // still able to bring Zero back sooner. + expiresAtTick: observation.tickNumber + 4, + })), + }; + finalize?.({ + outcome: 'completed', + provider: metadata(selectedModel), + swarmPlan: plan, + }); + return { plan, metadata: metadata(selectedModel) }; + } +} + +function nearestTarget( + position: ZeroStrategicObservation['agents'][number]['position'], + targets: readonly ZeroStrategicObservation['strategicTargetCells'][number][], +) { + return [...targets].sort((left, right) => { + const leftDistance = safeDistance(position, left); + const rightDistance = safeDistance(position, right); + return leftDistance - rightDistance || left.localeCompare(right); + })[0]; +} + +function safeDistance(left: string, right: string): number { + try { + return gridDistance(left, right); + } catch { + return Number.MAX_SAFE_INTEGER; + } +} + +/** Select from the server's opaque legal candidate map without a provider call. */ +function selectGreedyCandidate(compiled: CompiledReflexObservation): string { + return (compiled.observation.candidates.find(({ description }) => + description.startsWith('Infect'), + ) ?? + compiled.observation.candidates.find(({ description }) => + description.includes('open territory'), + ) ?? + compiled.observation.candidates[0])!.id; +} + +class OfflineReflex implements ReflexProvider { + readonly mode = 'scripted-reflex-test' as const; + readonly configured = true; + readonly model: string; + constructor(private readonly strategy: 'semantic' | 'greedy') { + this.model = `offline-${strategy}-reflex`; + } + + async decide( + observation: ReflexObservation, + options: ReflexDecisionOptions = {}, + ) { + const finalize = options.beginAttempt?.('initial'); + if (finalize === null) + throw new ReflexProviderError({ + code: 'budget-exhausted', + message: 'Offline reflex could not obtain an attempt permit.', + retryable: false, + }); + const candidates = observation.candidates; + const selected = + this.strategy === 'semantic' + ? (candidates.find(({ description }) => + description.startsWith('Infect'), + ) ?? + candidates.find(({ description }) => + description.includes('advances toward'), + ) ?? + candidates.find(({ description }) => + description.startsWith('Remain'), + ) ?? + candidates[0]) + : (candidates.find(({ description }) => + description.startsWith('Infect'), + ) ?? + candidates.find(({ description }) => + description.includes('open territory'), + ) ?? + candidates[0]); + if (!selected) + throw new Error('Offline reflex received no legal candidates.'); + const remainder = + candidates.length > 1 ? 0.25 / (candidates.length - 1) : 0; + const probabilities = Object.fromEntries( + candidates.map(({ id }) => [id, id === selected.id ? 0.75 : remainder]), + ); + // Correct the single-candidate case while retaining an exact distribution. + if (candidates.length === 1) probabilities[selected.id] = 1; + const decision = reflexDecisionSchema.parse({ + chosenCandidateId: selected.id, + confidence: candidates.length === 1 ? 1 : 0.75, + probabilities, + replanProbability: + observation.currentSituation.directiveProgress === 'blocked' + ? 0.85 + : 0.05, + model: this.model, + latencyMs: 0, + inputTokens: 0, + outputTokens: 0, + directiveId: observation.directive.id, + cognitionSource: 'jev-reflex', + }); + finalize?.({ + outcome: 'completed', + provider: metadata(this.model), + reflexDecision: decision, + }); + return decision; + } +} + +function deterministicIds(prefix: string) { + let sequence = 0; + return () => + `00000000-0000-4000-8000-${prefix}${String(++sequence).padStart(11, '0')}`; +} + +function territory(snapshot: ReturnType) { + const infected = snapshot.world.hexes.filter( + ({ state }) => state === 'infected', + ); + return { + infectedCells: infected.length, + controlledCells: infected.filter( + ({ controllerAgentId }) => controllerAgentId !== null, + ).length, + abandonedCells: infected.filter( + ({ controllerAgentId }) => controllerAgentId === null, + ).length, + }; +} + +function sample( + snapshot: ReturnType, +): OfflineComparisonTickSample { + const playerEvents = snapshot.world.events.filter( + ( + event, + ): event is Extract< + (typeof snapshot.world.events)[number], + { + type: + | 'simulated-player-moved' + | 'hex-disinfected' + | 'simulated-player-clean-blocked' + | 'simulated-player-agent-captured'; + } + > => + (event.type === 'simulated-player-moved' || + event.type === 'hex-disinfected' || + event.type === 'simulated-player-clean-blocked' || + event.type === 'simulated-player-agent-captured') && + event.originatingTick === snapshot.tickNumber, + ); + const latestSwarm = snapshot.swarmTicks?.at(-1); + const swarm = + latestSwarm?.tickNumber === snapshot.tickNumber ? latestSwarm : undefined; + const reflex = + swarm?.workers.flatMap(({ reflexDecision }) => + reflexDecision ? [reflexDecision] : [], + ) ?? []; + const providerMetadata = [ + ...(swarm?.plannerMetadata ? [swarm.plannerMetadata] : []), + ...reflex.map(({ latencyMs, inputTokens, outputTokens, model }) => ({ + ...metadata(model), + latencyMs, + promptTokens: inputTokens, + completionTokens: outputTokens, + totalTokens: inputTokens + outputTokens, + })), + ...snapshot.turns + .filter(({ tickNumber }) => tickNumber === snapshot.tickNumber) + .flatMap(({ provider }) => (provider ? [provider] : [])), + ]; + return { + tick: snapshot.tickNumber, + ...territory(snapshot), + activeAgents: snapshot.world.agents.length, + capturesThisTick: playerEvents.filter( + ({ type }) => type === 'simulated-player-agent-captured', + ).length, + disinfectionsThisTick: playerEvents.filter( + ({ type }) => type === 'hex-disinfected', + ).length, + terminalStatus: + snapshot.status === 'patient-zero-captured' || + snapshot.status === 'infection-eliminated' + ? snapshot.status + : null, + attemptsStarted: snapshot.experiment.attemptAccounting.attemptsStarted, + attemptsFinalized: snapshot.experiment.attemptAccounting.attemptsFinalized, + syntheticInputTokens: providerMetadata.reduce( + (total, item) => total + (item.promptTokens ?? 0), + 0, + ), + syntheticOutputTokens: providerMetadata.reduce( + (total, item) => total + (item.completionTokens ?? 0), + 0, + ), + syntheticLatencyMs: providerMetadata.reduce( + (total, item) => total + item.latencyMs, + 0, + ), + generativeAttempts: + swarm?.planSource === 'zero-llm' + ? 1 + : snapshot.turns.filter( + ({ tickNumber }) => tickNumber === snapshot.tickNumber, + ).length, + zeroPlans: swarm?.planSource === 'zero-llm' ? 1 : 0, + reflexDecisions: reflex.length, + workerStalls: + swarm?.workers.filter( + (worker) => + worker.actionResult?.accepted === false || + (worker.action?.type === 'wait' && + worker.directive.mission !== 'hold'), + ).length ?? 0, + replanSignals: swarm?.signals?.length ?? 0, + reflexConfidence: reflex.map(({ confidence }) => confidence), + reflexProbabilityDistributions: reflex.map( + ({ probabilities }) => probabilities, + ), + }; +} + +function createService(variant: OfflineComparisonVariant, seed: string) { + const planner = new OfflinePlanner(); + const service = new SimulationService({ + provider: new BrowserTestAgentProvider(), + swarmPlanner: planner, + // Kept for the ordinary swarm variant and snapshot provider status. The + // deterministic baseline uses the explicit server-side selector below and + // never invokes this provider. + reflexProvider: new OfflineReflex('semantic'), + ...(variant === 'deterministic-worker-baseline' + ? { deterministicWorkerCandidateSelector: selectGreedyCandidate } + : {}), + now: () => '2026-08-13T12:00:00.000Z', + createEventId: deterministicIds('1'), + createExperimentId: deterministicIds('2'), + createAllianceId: deterministicIds('3'), + createProposalId: deterministicIds('4'), + }); + service.setCompatibleModels([model]); + const request = service.getDefaultWorldSetup(); + // Preserve the original comparison shape: one Zero and seven workers. + const roster = generateDeterministicRoster(8, 'worker-capture-roster'); + service.applyWorldSetup({ + ...request, + cognitionMode: + variant === 'legacy-multi-agent' ? 'legacy-multi-agent' : 'zero-swarm-v1', + roster, + patientZeroAgentId: roster[1]!.id, + worldSeed: `offline-world-${seed}`, + spawnSeed: seed, + objectiveVersion: 'durable-influence-v3', + capabilities: { ...request.capabilities, simulatedPlayerPressure: true }, + simulatedPlayer: { enabled: true, profile: 'trail-hunter-v1', seed }, + modelConfiguration: { + globalModelId: model.id, + globalReasoningProfile: 'low', + overrides: [], + locked: false, + }, + behaviorConfiguration: { + ...request.behaviorConfiguration, + assignments: assignBehavior( + roster.map(({ id }) => id), + `offline-behavior-${seed}`, + 'balanced-random', + ), + }, + }); + return service; +} + +async function runVariant( + variant: OfflineComparisonVariant, + seed: string, + tickCap: number, +): Promise { + const service = createService(variant, seed); + const samples: OfflineComparisonTickSample[] = []; + for (let tick = 0; tick < tickCap; tick += 1) { + const before = service.getSnapshot(); + if ( + before.status === 'patient-zero-captured' || + before.status === 'infection-eliminated' + ) + break; + await service.executeNextTick(); + samples.push(sample(service.getSnapshot())); + } + const snapshot = service.getSnapshot(); + const latest = samples.at(-1); + const captures = samples.reduce( + (total, entry) => total + entry.capturesThisTick, + 0, + ); + const disinfections = samples.reduce( + (total, entry) => total + entry.disinfectionsThisTick, + 0, + ); + return { + seed, + samples, + final: { + tick: snapshot.tickNumber, + ...territory(snapshot), + activeAgents: snapshot.world.agents.length, + captures, + disinfections, + terminalStatus: latest?.terminalStatus ?? null, + }, + providerAttempts: { + started: snapshot.experiment.attemptAccounting.attemptsStarted, + finalized: snapshot.experiment.attemptAccounting.attemptsFinalized, + }, + }; +} + +function aggregate( + runs: readonly OfflineComparisonRun[], +): OfflineComparisonAggregate { + const samples = runs.flatMap(({ samples: entries }) => entries); + return { + runCount: runs.length, + totalProviderAttempts: runs.reduce( + (total, run) => total + run.providerAttempts.started, + 0, + ), + totalGenerativeAttempts: samples.reduce( + (total, entry) => total + entry.generativeAttempts, + 0, + ), + totalReflexAttempts: samples.reduce( + (total, entry) => total + entry.reflexDecisions, + 0, + ), + totalSyntheticInputTokens: samples.reduce( + (total, entry) => total + entry.syntheticInputTokens, + 0, + ), + totalSyntheticOutputTokens: samples.reduce( + (total, entry) => total + entry.syntheticOutputTokens, + 0, + ), + totalSyntheticLatencyMs: samples.reduce( + (total, entry) => total + entry.syntheticLatencyMs, + 0, + ), + totalCaptures: runs.reduce((total, run) => total + run.final.captures, 0), + totalDisinfections: runs.reduce( + (total, run) => total + run.final.disinfections, + 0, + ), + totalWorkerStalls: samples.reduce( + (total, entry) => total + entry.workerStalls, + 0, + ), + totalReplanSignals: samples.reduce( + (total, entry) => total + entry.replanSignals, + 0, + ), + terminalRuns: runs.filter(({ final }) => final.terminalStatus !== null) + .length, + }; +} + +/** Run the three comparison modes through SimulationService without a network call. */ +export async function runOfflineComparison( + options: OfflineComparisonOptions = {}, +): Promise { + const seeds = [...(options.seeds ?? DEFAULT_SEEDS)]; + const tickCap = options.tickCap ?? DEFAULT_TICK_CAP; + if (!seeds.length || seeds.some((seed) => !seed.trim())) + throw new Error('Offline comparison requires at least one non-empty seed.'); + if (!Number.isInteger(tickCap) || tickCap < 1 || tickCap > MAX_TICK_CAP) + throw new Error( + `tickCap must be an integer between 1 and ${MAX_TICK_CAP}.`, + ); + const variants: OfflineComparisonVariant[] = [ + 'legacy-multi-agent', + 'zero-swarm-v1', + 'deterministic-worker-baseline', + ]; + return { + formatVersion: 1, + kind: 'offline-deterministic-comparison', + seeds, + tickCap, + providerDisclaimer: + 'All providers in this report are deterministic offline fakes. Legacy mode uses BrowserTestAgentProvider, whose scripted social policy is not a behavioral substitute for a live legacy model. Token and latency fields describe only fake-provider telemetry, never live inference usage.', + costDisclaimer: + 'Fake-provider costCredits are accounting fixtures only. This report contains no authoritative billed monetary cost.', + variants: await Promise.all( + variants.map(async (variant) => { + const runs = await Promise.all( + seeds.map((seed) => runVariant(variant, seed, tickCap)), + ); + return { + variant, + providerKind: + variant === 'legacy-multi-agent' + ? 'BrowserTestAgentProvider' + : variant === 'zero-swarm-v1' + ? 'OfflinePlanner + semantic OfflineReflex' + : 'OfflinePlanner + deterministic legal-candidate selector', + runs, + aggregate: aggregate(runs), + }; + }), + ), + }; +} diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 98ec985..3b9db03 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -31,6 +31,15 @@ unknown or cancelled calls retain exposure, and a reported reservation overage stops future admission when a credit ceiling is enabled. These decimal-string totals are server authority, but they do not enforce the upstream account balance. See ADRs 0025 and 0026. +PR F supplies a separate offline comparison harness. It runs legacy, +zero-swarm, and deterministic-worker fixtures against the same seeded scenario +and player pressure, then emits JSON per-tick samples and aggregates. It is a +read-only research runner: it creates no live providers, reads no provider key, +and does not write to the experiment archive. Archive comparison remains useful +for safe completed exports, but is legacy-turn-centric and does not substitute +for the same-scenario swarm harness. See +[Zero-swarm offline comparison](ZERO_SWARM_COMPARISON.md). + ## Simultaneous tick authority Before the frozen agent snapshot, the optional seeded simulated-player profile diff --git a/docs/EXPERIMENT_ARCHIVE.md b/docs/EXPERIMENT_ARCHIVE.md index 71afd28..2a1f12d 100644 --- a/docs/EXPERIMENT_ARCHIVE.md +++ b/docs/EXPERIMENT_ARCHIVE.md @@ -119,3 +119,11 @@ directives, physical action results, and worker choice telemetry. Full all-agent exports carry these records separately from legacy `turns`. Provider attempts remain in the independent v4 ledger, including attempts from cancelled or rolled-back swarm ticks. + +## Zero-swarm comparisons + +The archive can preserve safe schema-v11 swarm tick records and independent +provider attempts, but its `compare` command remains centered on normalized +legacy turns. Use `pnpm compare:offline` for PR F's reproducible three-variant +same-scenario fixture report. The runner does not import, write, or modify this +archive. See [Zero-swarm offline comparison](ZERO_SWARM_COMPARISON.md). diff --git a/docs/TESTING.md b/docs/TESTING.md index a561f65..1103446 100644 --- a/docs/TESTING.md +++ b/docs/TESTING.md @@ -22,6 +22,13 @@ capture, abandoned territory, active-roster dispatch after player pressure, terminal capture outcomes, cancellation rollback, and safe export/UI handling. The casual-cleaner baseline remains covered by its existing offline tests. +PR F adds a deterministic comparison harness for `legacy-multi-agent`, +`zero-swarm-v1`, and `deterministic-worker-baseline`. It uses the same seeded +scenario and `trail-hunter-v1` pressure for all variants, retains per-tick and +aggregate metrics, and has no provider, credential, or archive-write path. +Run `pnpm compare:offline` (or bounded `--ticks` and `--seeds` options) to +produce its JSON report. See [Zero-swarm offline comparison](ZERO_SWARM_COMPARISON.md). + Attempt-budget tests use deterministic providers and cover whole-roster tick admission, retry permits, cancellation finalization, and the distinction between known zero cost and missing/unknown cost. Provider catalog probes are diff --git a/docs/ZERO_SWARM_COMPARISON.md b/docs/ZERO_SWARM_COMPARISON.md new file mode 100644 index 0000000..df4a18e --- /dev/null +++ b/docs/ZERO_SWARM_COMPARISON.md @@ -0,0 +1,97 @@ +# Zero-swarm offline comparison + +PR F compares three cognition variants with the same seeded scenario and the +same optional `trail-hunter-v1` pressure profile: + +- `legacy-multi-agent` +- `zero-swarm-v1` +- `deterministic-worker-baseline` + +Run the deterministic fixture suite with: + +```bash +pnpm --silent compare:offline +pnpm --silent compare:offline --ticks 40 --seeds pressure-a,pressure-b,pressure-c +``` + +The command writes one JSON report to standard output. Redirect it only when a +saved artifact is useful for review: + +```bash +pnpm --silent compare:offline --ticks 40 --seeds pressure-a,pressure-b > /tmp/hexzero-swarm-comparison.json +``` + +`--ticks` accepts an integer from 1 through 60. `--seeds` accepts up to 16 +comma-separated stable seed identifiers and is sorted before execution. Omit +both options to use the harness defaults. The command has no network or archive +write path and never loads provider credentials. + +## Report contents + +Each variant retains per-tick samples and an aggregate. The report covers: + +- provider attempts, generative attempts, and reflex input/output tokens; +- territory over time and retained territory under pressure; +- worker stalls, replan requests, Zero replans, and Jev confidence/distribution + telemetry where the mode produces it; +- surviving/captured agents and terminal outcomes; +- deterministic virtual tick time and reported provider latency; and +- a provider-cost disclaimer. + +Token fields report only factual values supplied by the deterministic fixture. +They are not billed usage. The report intentionally omits monetary cost because +neither fake providers nor TypeSafe usage metadata establish an authoritative +billed amount. Any future displayed estimate must name the model version and +pricing configuration and remain explicitly non-authoritative. + +The fixture reports synthetic provider latency as zero. It does not measure +live provider or end-to-end production latency. + +## Interpretation + +The runner uses scripted legacy decisions, a scripted Zero planner, scripted +Jev choices, and a deterministic worker policy that selects from the same legal +candidate map without a reflex provider call. It proves experiment +reproducibility and exposes comparable telemetry; it does not demonstrate +real-model quality, production latency, or provider cost. Real-provider studies +remain explicitly opted in and should archive their safe exports separately. + +The existing archive comparison command is currently legacy-turn-centric. It +can retain schema-v11 swarm tick records and independent provider attempts, but +it does not replace this same-scenario, per-tick swarm harness. + +## Results + +The fixed three-seed run uses eight agents and at most 12 ticks per variant: + +```bash +pnpm --silent compare:offline --ticks 12 --seeds worker-capture-spawn,comparison-seed-b,comparison-seed-c +``` + +The JSON report was rerun byte-for-byte and has SHA-256 +`43ee6680e2b88e679856bbf6b7d7577919da0fa5fce73b253da0bf5dd737965a`. +Its aggregate values are: + +| Scripted variant | Committed ticks | Provider attempts | Generative-equivalent attempts | Reflex decisions | Final controlled cells | Captures | Worker stalls | Terminal runs | +| ----------------------------- | --------------: | ----------------: | -----------------------------: | ---------------: | ---------------------: | -------: | ------------: | ------------: | +| Legacy multi-agent | 36 | 215 | 215 | 0 | 13 | 9 | 0 | 0 | +| Zero-swarm v1 | 28 | 157 | 16 | 141 | 53 | 8 | 38 | 2 | +| Deterministic worker baseline | 32 | 23 | 23 | 0 | 74 | 7 | 7 | 1 | + +All synthetic input/output tokens and provider latency are zero by fixture +design. Zero-swarm ended early in two of three seeds because Patient Zero was +captured; the deterministic worker baseline ended early in one seed. The +zero-swarm scripted reflex produced more stalled worker actions than the greedy +baseline. The legacy fixture made every agent a +scripted generative-equivalent call, but its social policy differs from both +swarm worker policies; this table cannot rank real model quality. Per-tick +territory, confidence, probability distributions, capture events, and replan +counts remain in the JSON report. Across the three seeds, Zero made 13 reviews +after its initial plans in the scripted swarm and 20 in the greedy baseline; +the fake reflex policy raised no worker replan signal. + +**Retirement decision:** retain `legacy-multi-agent` for now. The offline run +proves the accounting and comparison path, but it does not establish better +survival or cost for live Zero/Jev cognition. Revisit retirement after +reproducible real-provider trials, credible pressure outcomes, and authoritative +cost or clearly labeled pricing estimates. diff --git a/package.json b/package.json index 56475fc..2d142f2 100644 --- a/package.json +++ b/package.json @@ -9,6 +9,7 @@ }, "scripts": { "build": "pnpm -r --if-present build", + "compare:offline": "pnpm --filter @hexzero/game-api compare:offline", "dev": "pnpm --parallel --filter @hexzero/game-api --filter @hexzero/world-lab dev", "dev:test-provider": "HEXZERO_PROVIDER=scripted pnpm dev", "dev:api": "pnpm --filter @hexzero/game-api dev",