Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 14 additions & 17 deletions apps/game-api/src/reflex-execution.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -81,26 +81,23 @@ describe('zero-swarm reflex execution seam', () => {
expect(applied.state.agents.get(agent.id)?.currentCell).toBe(targetCell);
});

it('passes bounded authoritative capture facts to Jev without player route data', () => {
it('retains only the four most recent authoritative capture alerts', () => {
const { state, directive } = fixture();
const capturedAgentId = [...state.agents.keys()][0]!;
const compiled = compileReflexObservation(state, directive, {
captureAlerts: [
{
capturedAgentId: [...state.agents.keys()][0]!,
cell: directive.targetCell!,
originatingTick: 4,
abandonedCellCount: 2,
},
],
captureAlerts: Array.from({ length: 5 }, (_, index) => ({
capturedAgentId,
cell: directive.targetCell!,
originatingTick: index,
abandonedCellCount: index,
})),
});
expect(compiled.observation.captureAlerts).toEqual([
{
capturedAgentId: [...state.agents.keys()][0]!,
cell: directive.targetCell,
originatingTick: 4,
abandonedCellCount: 2,
},
]);
expect(compiled.observation.captureAlerts).toHaveLength(4);
expect(
compiled.observation.captureAlerts?.map(
({ abandonedCellCount }) => abandonedCellCount,
),
).toEqual([1, 2, 3, 4]);
expect(compiled.observation).not.toHaveProperty('simulatedPlayer');
});

Expand Down
1 change: 0 additions & 1 deletion apps/game-api/src/reflex-execution.ts
Original file line number Diff line number Diff line change
Expand Up @@ -166,7 +166,6 @@ export function compileReflexObservation(
recentTerritoryTrend: territoryTrend,
recentActionOutcome: history.recentActionOutcome ?? 'unknown',
},
relevantRecentFacts: [],
...(history.captureAlerts?.length
? { captureAlerts: [...history.captureAlerts].slice(-4) }
: {}),
Expand Down
9 changes: 9 additions & 0 deletions docs/ARCHITECTURE.md
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,15 @@ service maps to a world action for seeded engine resolution. One complete tick
commits atomically. Planner and worker failures use explicit deterministic
fallbacks; provider attempts survive world rollback. See ADRs 0028, 0029, and 0031.

The worker observation includes bounded, current capture alerts. The TypeSafe
request projects them as structured capture pressure without cell IDs or tick
numbers. The unused prose `relevantRecentFacts` field is removed; current
situation and legal candidate descriptions supply the relevant local facts.
Zero planning uses the same provider-reported OpenRouter usage normalization as
legacy turns, including actual `usage.cost` when returned, and preserves that
metadata when a returned plan is rejected or a bounded non-success response
contains usage. Missing provider cost remains unknown.

Swarm tick records are separate from legacy agent turn records. Full all-agent
exports and the archive retain safe plans, directives, action choices, and
factual provider usage; selective legacy turn filters do not export partial
Expand Down
8 changes: 5 additions & 3 deletions docs/SECURITY.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,9 +40,11 @@ their nested rollups.
## Secrets and deployment

`TYPESAFE_API_KEY` is server-only for the zero-swarm reflex provider. Jev sees
compact semantic state and opaque candidate IDs. The outbound state omits H3
cell IDs and tick numbers; the server retains directive targets and the action
mapping. The key, raw
compact semantic state and opaque candidate IDs. When current player pressure
captures a worker, Jev also receives a bounded structured capture-pressure
count and abandoned-cell counts from the worker's authorized capture alerts.
The outbound state omits H3 cell IDs, tick numbers, and hidden hunter routing;
the server retains directive targets and the action mapping. The key, raw
TypeSafe requests and responses, and provider error bodies never enter safe
telemetry, World Lab, archives, or exports. TypeSafe token usage is factual;
no monetary cost is inferred from it.
Expand Down
6 changes: 6 additions & 0 deletions docs/TESTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,12 @@ aggregate metrics, and has no provider, credential, or archive-write path.
Run `pnpm compare:offline` (or bounded `--ticks` and `--seeds` options) to
produce its JSON report. See [Zero-swarm offline comparison](ZERO_SWARM_COMPARISON.md).

The pre-live provider audit tests the Jev request's bounded capture-pressure
projection, omission of hidden hunter state and empty capture noise, and the
removal of the inert recent-facts field. Planner tests cover provider-reported
OpenRouter cost and detailed tokens for valid, rejected, and non-success Zero
responses, plus sanitized response IDs/models; missing cost stays unknown.

Attempt-budget tests use deterministic providers and cover whole-roster tick
admission, retry permits, cancellation finalization, and the distinction
between known zero cost and missing/unknown cost. Provider catalog probes are
Expand Down
49 changes: 2 additions & 47 deletions packages/agent-runtime/src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@ import {
type ProviderMetadata,
type ReasoningProfile,
} from '@hexzero/shared';
import { normalizeOpenRouterUsage } from './openrouter-usage';

export { applyProviderEnvironmentFile } from './provider-environment';
export * from './model-catalog';
Expand Down Expand Up @@ -203,24 +204,6 @@ const openRouterResponseSchema = z.object({
usage: z.unknown().optional(),
});

const openRouterUsageSchema = z.object({
prompt_tokens: z.number().int().nonnegative().optional(),
completion_tokens: z.number().int().nonnegative().optional(),
total_tokens: z.number().int().nonnegative().optional(),
cost: z.number().nonnegative().finite().optional(),
completion_tokens_details: z
.object({ reasoning_tokens: z.number().int().nonnegative().optional() })
.passthrough()
.optional(),
prompt_tokens_details: z
.object({
cached_tokens: z.number().int().nonnegative().optional(),
cache_write_tokens: z.number().int().nonnegative().optional(),
})
.passthrough()
.optional(),
});

export function normalizeFlatDecision(input: unknown) {
const parsed = wireDecisionSchema.safeParse(input);
if (!parsed.success) return providerDecisionEnvelopeSchema.safeParse({});
Expand Down Expand Up @@ -855,8 +838,6 @@ function providerMetadataFromResponse(
sensitiveValues: string[],
): ProviderMetadata {
const root = asRecord(raw);
const usage = openRouterUsageSchema.safeParse(root?.usage);
const parsedUsage = usage.success ? usage.data : undefined;
const requestId =
sanitizeDiagnosticCode(root?.id, sensitiveValues, 160) ??
sanitizeDiagnosticCode(
Expand All @@ -876,33 +857,7 @@ function providerMetadataFromResponse(
latencyMs,
httpStatus: response.status,
...(requestId === undefined ? {} : { requestId }),
...(parsedUsage?.prompt_tokens === undefined
? {}
: { promptTokens: parsedUsage.prompt_tokens }),
...(parsedUsage?.completion_tokens === undefined
? {}
: { completionTokens: parsedUsage.completion_tokens }),
...(parsedUsage?.total_tokens === undefined
? {}
: { totalTokens: parsedUsage.total_tokens }),
...(parsedUsage?.completion_tokens_details?.reasoning_tokens === undefined
? {}
: {
reasoningTokens:
parsedUsage.completion_tokens_details.reasoning_tokens,
}),
...(parsedUsage?.prompt_tokens_details?.cached_tokens === undefined
? {}
: { cachedReadTokens: parsedUsage.prompt_tokens_details.cached_tokens }),
...(parsedUsage?.prompt_tokens_details?.cache_write_tokens === undefined
? {}
: {
cacheWriteTokens:
parsedUsage.prompt_tokens_details.cache_write_tokens,
}),
...(parsedUsage?.cost === undefined
? {}
: { costCredits: parsedUsage.cost }),
...normalizeOpenRouterUsage(root?.usage),
};
}

Expand Down
68 changes: 68 additions & 0 deletions packages/agent-runtime/src/openrouter-usage.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,68 @@
import { z } from 'zod';
import type { ProviderMetadata } from '@hexzero/shared';

const nonnegativeInteger = z.number().int().nonnegative();
const nonnegativeFinite = z.number().nonnegative().finite();

/** Preserves provider-reported OpenRouter accounting without price estimates. */
export function normalizeOpenRouterUsage(
usage: unknown,
): Partial<
Pick<
ProviderMetadata,
| 'promptTokens'
| 'completionTokens'
| 'totalTokens'
| 'reasoningTokens'
| 'cachedReadTokens'
| 'cacheWriteTokens'
| 'costCredits'
>
> {
if (!usage || typeof usage !== 'object' || Array.isArray(usage)) return {};
const record = usage as Record<string, unknown>;
const completionDetails = asRecord(record.completion_tokens_details);
const promptDetails = asRecord(record.prompt_tokens_details);
return {
...numberMetadata('promptTokens', record.prompt_tokens, nonnegativeInteger),
...numberMetadata(
'completionTokens',
record.completion_tokens,
nonnegativeInteger,
),
...numberMetadata('totalTokens', record.total_tokens, nonnegativeInteger),
...numberMetadata(
'reasoningTokens',
completionDetails?.reasoning_tokens,
nonnegativeInteger,
),
...numberMetadata(
'cachedReadTokens',
promptDetails?.cached_tokens,
nonnegativeInteger,
),
...numberMetadata(
'cacheWriteTokens',
promptDetails?.cache_write_tokens,
nonnegativeInteger,
),
...numberMetadata('costCredits', record.cost, nonnegativeFinite),
};
}

function asRecord(value: unknown): Record<string, unknown> | undefined {
return value && typeof value === 'object' && !Array.isArray(value)
? (value as Record<string, unknown>)
: undefined;
}

function numberMetadata<Key extends keyof ProviderMetadata>(
key: Key,
value: unknown,
schema: z.ZodType<NonNullable<ProviderMetadata[Key]>>,
): Partial<Pick<ProviderMetadata, Key>> {
const parsed = schema.safeParse(value);
return parsed.success
? ({ [key]: parsed.data } as Partial<Pick<ProviderMetadata, Key>>)
: {};
}
Loading
Loading