Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 22 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,28 @@ Pi displays a device code, opens the Meta authorization flow, and mints a Model

The access key is re-minted daily.

## Prompt caching and encrypted reasoning

Muse Spark on `api.meta.ai` returns no useful cache hits on
`/v1/chat/completions`, so this provider drives the Responses API and sets
`prompt_cache_retention: "24h"` on every request (unless the payload already
sets one) for 93–99% cache hits.

Cross-turn reasoning continuity needs `include: ["reasoning.encrypted_content"]`,
but not every key is entitled to it: keys minted through `/muse-code/key`
can answer HTTP 400 `reasoning \`encrypted_content\` was not issued to this
caller`, which fails the whole request. So the provider probes each key once
per process — a ~16-token `/v1/responses` call carrying the include with
`max_output_tokens: 16` — and:

- **200** → the key is entitled: the include is kept, reasoning carries across turns;
- **400 mentioning `encrypted_content`** → the include is stripped on every request, so calls never 400;
- **anything else** (transient error) → treated as not entitled for safety, re-probed after a 5-minute cooldown.

The probe runs in the background on the first request for a given key and is
recached when the daily key rotation delivers a new key. If Meta changes
entitlement policy, a restart (or daily key rotation) picks it up.

## Models

Fallback models use a 1,048,576-token context window, up to 256K output tokens, image input, and reasoning levels `minimal`, `low`, `medium`, `high`, and `xhigh` (`muse-spark-1.3` additionally supports `max`).
Expand Down
115 changes: 113 additions & 2 deletions extensions/meta.ts
Original file line number Diff line number Diff line change
Expand Up @@ -601,6 +601,95 @@ export function metaFallbackCost(
/** Meta prompt-cache opt-in. Measured 0% hits on /chat/completions vs 93–99% on /responses with 24h. */
export const META_PROMPT_CACHE_RETENTION = "24h";

const ENCRYPTED_REASONING_INCLUDE = "reasoning.encrypted_content";
const PROBE_RETRY_MS = 5 * 60 * 1000;

/**
* Keys minted through /muse-code/key are not always entitled to encrypted
* reasoning replay (Meta answers HTTP 400 "reasoning `encrypted_content`
* was not issued to this caller" when they aren't). Entitlement can change
* between minted keys, so probe once per key per process instead of
* hard-coding a decision: requests never 400 and reasoning continuity is
* kept whenever the key allows it.
*/
interface EntitlementCache {
keyHash?: string;
known?: boolean;
lastAttemptAt: number;
}
const entitlementCache: EntitlementCache = { lastAttemptAt: 0 };

function apiKeyHash(key: string): string {
// Non-cryptographic FNV-1a; only used to key the in-process probe cache.
let hash = 2166136261;
for (let i = 0; i < key.length; i++) {
hash ^= key.charCodeAt(i);
hash = Math.imul(hash, 16777619);
}
return (hash >>> 0).toString(16);
}

/**
* Probe whether the given API key can request reasoning.encrypted_content.
* Returns true (200), false (Meta rejects the include), or undefined when
* the probe was inconclusive (transient error) and must be retried later.
*/
export async function probeEncryptedReasoningEntitlement(
apiKey: string,
fetchImpl: Fetch = fetch,
): Promise<boolean | undefined> {
try {
const response = await fetchImpl(`${META_API_BASE_URL}/responses`, {
method: "POST",
headers: {
Accept: "application/json",
Authorization: `Bearer ${apiKey}`,
"Content-Type": "application/json",
"x-api-version": "1.0.0",
},
body: JSON.stringify({
model: "muse-spark-1.3",
input: "Answer with the single letter: a",
include: [ENCRYPTED_REASONING_INCLUDE],
max_output_tokens: 16,
store: false,
}),
});
if (response.status === 200) return true;
if (response.status === 400) {
const text = await response.text();
if (text.includes("encrypted_content")) return false;
}
return undefined;
} catch {
return undefined;
}
}

function scheduleEntitlementProbe(apiKey: string): void {
const hash = apiKeyHash(apiKey);
if (
entitlementCache.keyHash === hash &&
entitlementCache.known !== undefined
) {
return;
}
if (Date.now() - entitlementCache.lastAttemptAt < PROBE_RETRY_MS) return;
entitlementCache.lastAttemptAt = Date.now();
void probeEncryptedReasoningEntitlement(apiKey).then((known) => {
if (known !== undefined) {
entitlementCache.keyHash = hash;
entitlementCache.known = known;
}
});
}

function keepEncryptedReasoningFor(apiKey: string | undefined): boolean {
if (!apiKey) return false;
if (entitlementCache.keyHash !== apiKeyHash(apiKey)) return false;
return entitlementCache.known === true;
}

function asRecord(value: unknown): Record<string, unknown> | undefined {
return value !== null && typeof value === "object" && !Array.isArray(value)
? (value as Record<string, unknown>)
Expand All @@ -614,12 +703,23 @@ function asRecord(value: unknown): Record<string, unknown> | undefined {
*/
export function applyMetaResponsesCacheHints(
payload: unknown,
keepEncryptedReasoning = false,
): Record<string, unknown> | undefined {
const body = asRecord(payload);
if (!body) return undefined;
if (body.prompt_cache_retention === undefined) {
body.prompt_cache_retention = META_PROMPT_CACHE_RETENTION;
}
// Drop the encrypted-reasoning include unless the key was probed and
// found entitled: unentitled keys get a fatal HTTP 400 for it (see
// probeEncryptedReasoningEntitlement). Other include entries survive.
if (!keepEncryptedReasoning && Array.isArray(body.include)) {
const include = body.include.filter(
(item) => item !== "reasoning.encrypted_content",
);
if (include.length === 0) delete body.include;
else body.include = include;
}
const reasoning = asRecord(body.reasoning);
if (
reasoning &&
Expand Down Expand Up @@ -664,8 +764,19 @@ export default function metaOAuthProvider(pi: ExtensionAPI): void {
process.env["MODEL_API_KEY"] = process.env[META_ENV_VAR];
}
pi.registerProvider(META_PROVIDER_ID, createMetaProviderConfig());
pi.on("before_provider_request", (event, ctx) => {
pi.on("before_provider_request", async (event, ctx) => {
if (ctx.model?.provider !== META_PROVIDER_ID) return undefined;
return applyMetaResponsesCacheHints(event.payload);
let apiKey: string | undefined;
try {
apiKey = (await ctx.modelRegistry?.getProviderAuth(META_PROVIDER_ID))
?.auth?.apiKey;
} catch {
apiKey = undefined;
}
if (apiKey) scheduleEntitlementProbe(apiKey);
return applyMetaResponsesCacheHints(
event.payload,
keepEncryptedReasoningFor(apiKey),
);
});
}
66 changes: 64 additions & 2 deletions tests/meta-cache.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@ import {
META_API_BASE_URL,
META_PROMPT_CACHE_RETENTION,
META_PROVIDER_ID,
probeEncryptedReasoningEntitlement,
toProviderModels,
} from "../extensions/meta.ts";
import metaOAuthProvider from "../extensions/meta.ts";
Expand Down Expand Up @@ -225,6 +226,67 @@ describe("Meta Responses cache and reasoning contracts", () => {
});
});

test("strips reasoning.encrypted_content include when the key is not entitled", () => {
expect(
applyMetaResponsesCacheHints({
include: ["reasoning.encrypted_content"],
reasoning: { effort: "high", summary: "auto" },
}),
).toEqual({
prompt_cache_retention: "24h",
reasoning: { effort: "high", summary: "auto" },
});
expect(
applyMetaResponsesCacheHints({
include: ["reasoning.encrypted_content", "something.else"],
}),
).toMatchObject({ include: ["something.else"] });
});

test("keeps reasoning.encrypted_content include when the key is entitled", () => {
expect(
applyMetaResponsesCacheHints(
{
include: ["reasoning.encrypted_content"],
reasoning: { effort: "high", summary: "auto" },
},
true,
),
).toEqual({
prompt_cache_retention: "24h",
include: ["reasoning.encrypted_content"],
reasoning: { effort: "high", summary: "auto" },
});
});

test("probes encrypted-reasoning entitlement: 200 means entitled", async () => {
const known = await probeEncryptedReasoningEntitlement(
"test-key",
(async () => new Response("{}", { status: 200 })) as unknown as typeof fetch,
);
expect(known).toBe(true);
});

test("probe reports not entitled when Meta rejects encrypted_content", async () => {
const known = await probeEncryptedReasoningEntitlement(
"test-key",
(async () =>
new Response(
'{"type":"invalid_request_error","message":"reasoning `encrypted_content` was not issued to this caller"}',
{ status: 400 },
)) as unknown as typeof fetch,
);
expect(known).toBe(false);
});

test("probe is inconclusive on transient errors", async () => {
const known = await probeEncryptedReasoningEntitlement(
"test-key",
(async () => new Response("{}", { status: 502 })) as unknown as typeof fetch,
);
expect(known).toBeUndefined();
});

test("pi-ai hits /v1/responses, not /chat/completions", async () => {
const { url, payload } = await captureResponsesRequest();
expect(url).toContain("https://api.meta.ai/v1/responses");
Expand Down Expand Up @@ -293,13 +355,13 @@ describe("Meta Responses cache and reasoning contracts", () => {
} as unknown as ExtensionAPI);
expect(handler).toBeDefined();

const other = handler?.(
const other = await handler?.(
{ payload: { model: "gpt" } },
{ model: { provider: "openai" } },
);
expect(other).toBeUndefined();

const meta = handler?.(
const meta = await handler?.(
{ payload: { model: "muse-spark-1.2" } },
{ model: { provider: META_PROVIDER_ID } },
);
Expand Down
Loading