Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion client/src/locales/en/translation.json
Original file line number Diff line number Diff line change
Expand Up @@ -362,9 +362,10 @@
"com_endpoint_anthropic_prompt_cache_ttl": "How long a cached prompt prefix stays warm between requests. Leave unset to use the provider default (1 hour). The 1-hour cache keeps the prefix warm across longer gaps at a higher one-time write cost; choose 5 minutes for the legacy behavior.",
"com_endpoint_anthropic_temp": "Ranges from 0 to 1. Use temp closer to 0 for analytical / multiple choice, and closer to 1 for creative and generative tasks. We recommend altering this or Top P but not both.",
"com_endpoint_anthropic_thinking": "Enables internal reasoning for supported Claude models. For newer models (Opus 4.6+), uses adaptive thinking controlled by the Effort parameter. For legacy models, requires \"Thinking Budget\" to be set and lower than \"Max Output Tokens\".",
"com_endpoint_anthropic_thinking_between_tools": "Enables adaptive thinking controlled by the Effort parameter. This model cannot turn thinking off entirely: switching this off uses its lowest setting, which skips extended thinking and keeps only brief notes between tool calls. Effort is capped at High while off.",
"com_endpoint_anthropic_thinking_budget": "Determines the max number of tokens Claude is allowed to use for its internal reasoning process. Larger budgets can improve response quality by enabling more thorough analysis for complex problems, although Claude may not use the entire budget allocated, especially at ranges above 32K. This setting must be lower than \"Max Output Tokens.\"",
"com_endpoint_anthropic_thinking_display": "Thought Visibility",
"com_endpoint_anthropic_thinking_display_desc": "Controls whether Claude's reasoning is returned. 'Auto' opts in to summarized thoughts for models that hide them by default (Opus 4.7+); 'Summarized' always shows them; 'Omitted' always hides them for slightly lower latency.",
"com_endpoint_anthropic_thinking_display_desc": "Controls whether Claude's reasoning is returned. 'Auto' opts in to summarized thoughts for models that hide them by default (Opus 4.7+); 'Summarized' always shows them; 'Omitted' always hides them for slightly lower latency; 'Updates' shows only the progress notes Claude writes between tool calls and hides the reasoning itself.",
"com_endpoint_anthropic_topk": "Top-k changes how the model selects tokens for output. A top-k of 1 means the selected token is the most probable among all tokens in the model's vocabulary (also called greedy decoding), while a top-k of 3 means that the next token is selected from among the 3 most probable tokens (using temperature).",
"com_endpoint_anthropic_topp": "Top-p changes how the model selects tokens for output. Tokens are selected from most K (see topK parameter) probable to least until the sum of their probabilities equals the top-p value.",
"com_endpoint_anthropic_use_web_search": "Enable web search functionality using Anthropic's built-in search capabilities. This allows the model to search the web for up-to-date information and provide more accurate, current responses.",
Expand Down Expand Up @@ -2894,6 +2895,7 @@
"com_ui_update_shared_link_confirm_description": "This publishes the latest messages and your current file-sharing choice to the existing link. The URL stays the same, and anyone with access can see the updated snapshot.",
"com_ui_update_shared_link_confirm_title": "Update shared link?",
"com_ui_updated_file": "Updated {{0}}",
"com_ui_updates": "Updates",
"com_ui_updating": "Updating...",
"com_ui_upload": "Upload",
"com_ui_upload_agent_avatar": "Successfully updated agent avatar",
Expand Down
27 changes: 16 additions & 11 deletions packages/api/src/endpoints/anthropic/helpers.ts
Original file line number Diff line number Diff line change
@@ -1,15 +1,16 @@
import { logger } from '@librechat/data-schemas';
import { AnthropicClientOptions } from '@librechat/agents';
import {
isOpus55Model,
OPUS_55_BLOCK_BINDING,
ThinkingDisplay,
AnthropicEffort,
anthropicSettings,
bindsThinkingBlocks,
supportsPromptCache,
hasAlwaysOnThinking,
THINKING_BLOCK_BINDING,
resolveThinkingDisplay,
supportsAdaptiveThinking,
supportsPromptCache,
requiresExplicitThinkingDisabled,
resolveThinkingOffConfig,
} from 'librechat-data-provider';

const FINE_GRAINED_TOOL_STREAMING_BETA = 'fine-grained-tool-streaming-2025-05-14';
Expand Down Expand Up @@ -87,17 +88,21 @@ function configureReasoning(
/**
* Sonnet 5 and Opus 5 run adaptive thinking by default when the `thinking`
* field is omitted, so honoring a user who turns thinking off requires
* sending an explicit disabled config rather than leaving the field unset.
* This returns before effort is applied, which is why the Opus 5 effort cap
* is enforced by the caller.
* sending an explicit disabled config rather than leaving the field unset;
* Sonnet 5.5+ rejects `disabled` and takes `between_tools` instead. This
* returns before effort is applied, which is why the effort cap for these
* configs is enforced by the caller. Always-on models (Opus 5.5+, Fable)
* ignore a stored "off" and always send the adaptive config.
*/
if (!extendedOptions.thinking && modelName && requiresExplicitThinkingDisabled(modelName)) {
updatedOptions.thinking = { type: 'disabled' } as AnthropicClientOptions['thinking'];
const thinkingOffConfig =
!extendedOptions.thinking && modelName ? resolveThinkingOffConfig(modelName) : undefined;
if (thinkingOffConfig) {
updatedOptions.thinking = thinkingOffConfig as AnthropicClientOptions['thinking'];
return updatedOptions;
}

if (
(extendedOptions.thinking || isOpus55Model(modelName)) &&
(extendedOptions.thinking || hasAlwaysOnThinking(modelName)) &&
modelName &&
supportsAdaptiveThinking(modelName)
) {
Expand All @@ -113,7 +118,7 @@ function configureReasoning(
const adaptive = {
type: 'adaptive' as const,
...(display ? { display } : {}),
...(isOpus55Model(modelName) ? { block_binding: { ...OPUS_55_BLOCK_BINDING } } : {}),
...(bindsThinkingBlocks(modelName) ? { block_binding: { ...THINKING_BLOCK_BINDING } } : {}),
};
/**
* TODO: Remove the cast once `@librechat/agents` updates its
Expand Down
166 changes: 162 additions & 4 deletions packages/api/src/endpoints/anthropic/llm.spec.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,11 @@
import { Providers, getChatModelClass } from '@librechat/agents';
import { AuthKeys, AnthropicEffort, ThinkingDisplay } from 'librechat-data-provider';
import {
AuthKeys,
AnthropicEffort,
ThinkingDisplay,
bindsThinkingBlocks,
THINKING_BINDING_BETA,
} from 'librechat-data-provider';
import type * as t from '~/types';
import { FINE_GRAINED_TOOL_STREAMING_BETA } from './helpers';
import { getLLMConfig } from './llm';
Expand Down Expand Up @@ -1367,6 +1373,154 @@ describe('getLLMConfig', () => {
expect(result.llmConfig.thinking).toEqual({ type: 'adaptive', display: 'summarized' });
});

describe('Sonnet 5.5', () => {
const SONNET_55_IDS = ['claude-sonnet-5-5', 'claude-sonnet-5.5'];

it.each(SONNET_55_IDS)(
'binds adaptive thinking blocks, omits sampling and caches prompts for %s',
(model) => {
const result = getLLMConfig('test-key', {
modelOptions: {
model,
thinking: true,
effort: AnthropicEffort.max,
temperature: 0.7,
topP: 0.9,
topK: 40,
},
});

expect(result.llmConfig.model).toBe(model);
expect(result.llmConfig.thinking).toEqual({
type: 'adaptive',
display: 'summarized',
block_binding: { prefix_mismatch_behavior: 'drop_block' },
});
expect(result.llmConfig.outputConfig).toEqual({ effort: AnthropicEffort.max });
expect(result.llmConfig).not.toHaveProperty('temperature');
expect(result.llmConfig).not.toHaveProperty('topP');
expect(result.llmConfig).not.toHaveProperty('topK');
expect(result.llmConfig.maxTokens).toBe(128000);
expect(result.llmConfig).toHaveProperty('promptCache', true);
const beta = (result.llmConfig.clientOptions?.defaultHeaders as Record<string, string>)[
'anthropic-beta'
];
expect(beta).toContain('thinking-binding-controls-2026-08-01');
expect(beta).not.toContain('fine-grained-tool-streaming-2025-05-14');
},
);

it.each(SONNET_55_IDS)(
'maps thinking off to between_tools without display, binding or budget for %s',
(model) => {
const result = getLLMConfig('test-key', {
modelOptions: {
model,
thinking: false,
thinkingBudget: 5000,
thinkingDisplay: ThinkingDisplay.summarized,
effort: AnthropicEffort.high,
},
});

expect(result.llmConfig.thinking).toEqual({ type: 'between_tools' });
expect(result.llmConfig.outputConfig).toEqual({ effort: AnthropicEffort.high });
},
);

it.each([AnthropicEffort.xhigh, AnthropicEffort.max])(
'clamps effort %s to high under between_tools',
(effort) => {
const { llmConfig } = getLLMConfig('test-key', {
modelOptions: { model: 'claude-sonnet-5-5', thinking: false, effort },
});
const Anthropic = getChatModelClass(Providers.ANTHROPIC);
const payload = new Anthropic(llmConfig).invocationParams();

expect(payload.thinking).toEqual({ type: 'between_tools' });
expect(payload.output_config).toEqual({ effort: AnthropicEffort.high });
expect(payload).not.toHaveProperty('temperature');
},
);

it('keeps low and medium effort under between_tools', () => {
for (const effort of [AnthropicEffort.low, AnthropicEffort.medium]) {
const { llmConfig } = getLLMConfig('test-key', {
modelOptions: { model: 'claude-sonnet-5-5', thinking: false, effort },
});
expect(llmConfig.outputConfig).toEqual({ effort });
}
});

it('treats a persisted between_tools config as thinking off', () => {
const result = getLLMConfig('test-key', {
modelOptions: {
model: 'claude-sonnet-5-5',
thinking: { type: 'between_tools' } as unknown as boolean,
},
});

expect(result.llmConfig.thinking).toEqual({ type: 'between_tools' });
});

it('never sends disabled thinking for Sonnet 5.5', () => {
const result = getLLMConfig('test-key', {
modelOptions: {
model: 'claude-sonnet-5-5',
thinking: { type: 'disabled' } as unknown as boolean,
},
});

expect(result.llmConfig.thinking).toEqual({ type: 'between_tools' });
});

it('requests display updates with its beta header', () => {
const result = getLLMConfig('test-key', {
modelOptions: {
model: 'claude-sonnet-5-5',
thinking: true,
thinkingDisplay: ThinkingDisplay.updates,
},
});

expect(result.llmConfig.thinking).toMatchObject({
type: 'adaptive',
display: 'updates',
});
const beta = (result.llmConfig.clientOptions?.defaultHeaders as Record<string, string>)[
'anthropic-beta'
];
expect(beta.split(',')).toEqual(
expect.arrayContaining([
'thinking-binding-controls-2026-08-01',
'thinking-display-updates-2026-08-18',
]),
);
});

it('demotes display updates to summarized when client options are dropped', () => {
const result = getLLMConfig('test-key', {
modelOptions: {
model: 'claude-sonnet-5-5',
thinking: true,
thinkingDisplay: ThinkingDisplay.updates,
},
dropParams: ['clientOptions'],
});

expect(result.llmConfig).not.toHaveProperty('clientOptions');
expect(result.llmConfig.thinking).toEqual({ type: 'adaptive', display: 'summarized' });
});

it('leaves Sonnet 5 on explicit disabled thinking', () => {
const result = getLLMConfig('test-key', {
modelOptions: { model: 'claude-sonnet-5', thinking: false },
});

expect(result.llmConfig.thinking).toEqual({ type: 'disabled' });
});
});

it('should omit sampling parameters for Opus 5', () => {
const result = getLLMConfig('test-key', {
modelOptions: {
Expand Down Expand Up @@ -2000,12 +2154,16 @@ describe('getLLMConfig', () => {
const headers = result.llmConfig.clientOptions?.defaultHeaders;
expect(headers).toBeDefined();
const betaHeader = (headers as Record<string, string>)['anthropic-beta'];
expect(betaHeader).toContain(FINE_GRAINED_TOOL_STREAMING_BETA);
/** Sonnet 5.5+ (here `claude-sonnet-6`) swaps the streaming beta for the binding beta. */
const baseBeta = bindsThinkingBlocks(model)
? THINKING_BINDING_BETA
: FINE_GRAINED_TOOL_STREAMING_BETA;
expect(betaHeader).toContain(baseBeta);

if (shouldHaveHeaders) {
expect(betaHeader).not.toBe(FINE_GRAINED_TOOL_STREAMING_BETA);
expect(betaHeader).not.toBe(baseBeta);
} else {
expect(betaHeader).toBe(FINE_GRAINED_TOOL_STREAMING_BETA);
expect(betaHeader).toBe(baseBeta);
}

if (shouldHavePromptCache) {
Expand Down
39 changes: 26 additions & 13 deletions packages/api/src/endpoints/anthropic/llm.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,11 +2,13 @@ import { Agent } from 'undici';
import { logger } from '@librechat/data-schemas';
import { AnthropicClientOptions } from '@librechat/agents';
import {
isOpus55Model,
THINKING_BINDING_BETA,
THINKING_DISPLAY_UPDATES_BETA,
requestsThinkingDisplayUpdates,
clampOutputConfigEffort,
omitsSamplingParameters,
isThinkingDisabled,
bindsThinkingBlocks,
isThinkingOffConfig,
anthropicSettings,
removeNullishValues,
ThinkingDisplay,
Expand Down Expand Up @@ -152,12 +154,13 @@ function getLLMConfig(
/**
* `thinking` may round-trip as the full Anthropic object rather than a
* boolean. Normalize to a flag so a persisted `{ type: 'disabled' }` (e.g. a
* Sonnet 5 "thinking off" config stored back into `model_parameters`) is
* treated as off β€” a truthy object would otherwise flip thinking back on.
* Sonnet 5 "thinking off" config stored back into `model_parameters`) or
* `{ type: 'between_tools' }` (Sonnet 5.5) is treated as off β€” a truthy
* object would otherwise flip thinking back on.
*/
const thinkingFlag =
typeof persistedThinking === 'object' && persistedThinking != null
? (persistedThinking as { type?: string }).type !== 'disabled'
? !isThinkingOffConfig(persistedThinking)
: (persistedThinking ?? anthropicSettings.thinking.default);

const systemOptions = {
Expand Down Expand Up @@ -261,12 +264,12 @@ function getLLMConfig(
}

/**
* Opus 5 rejects `xhigh`/`max` effort while thinking is disabled (400).
* `configureReasoning` returns before setting effort on the disabled path, so
* the value applied just above is the one that would ship β€” clamp it to the
* highest level the model accepts in that combination.
* Opus 5 rejects `xhigh`/`max` effort while thinking is disabled, and Sonnet
* 5.5 does the same under `between_tools` (400). `configureReasoning` returns
* before setting effort on that path, so the value applied just above is the
* one that would ship β€” clamp it to the highest level the model accepts.
*/
if (isThinkingDisabled(requestOptions.thinking)) {
if (isThinkingOffConfig(requestOptions.thinking)) {
clampOutputConfigEffort(resolvedModel, requestOptions.invocationKwargs?.output_config);
}

Expand Down Expand Up @@ -381,15 +384,19 @@ function getLLMConfig(
requestOptions.outputConfig = requestOptions.invocationKwargs.output_config;
}

/** block_binding is invalid without its beta header. Honor an administrator
* dropping clientOptions without leaving a beta-only field in the body. */
/** block_binding and display `updates` are invalid without their beta headers.
* Honor an administrator dropping clientOptions without leaving a beta-only
* field in the body. */
if (
shouldDropClientOptions &&
requestOptions.thinking &&
'block_binding' in requestOptions.thinking
) {
delete requestOptions.thinking.block_binding;
}
if (shouldDropClientOptions && requestsThinkingDisplayUpdates(requestOptions.thinking)) {
(requestOptions.thinking as { display?: string }).display = ThinkingDisplay.summarized;
}

if (shouldOmitSamplingParameters) {
delete requestOptions.temperature;
Expand Down Expand Up @@ -423,8 +430,14 @@ function getLLMConfig(
}
requestOptions.clientOptions.defaultHeaders = appendAnthropicBetaHeader(
requestOptions.clientOptions.defaultHeaders as Record<string, string> | undefined,
isOpus55Model(resolvedModel) ? THINKING_BINDING_BETA : FINE_GRAINED_TOOL_STREAMING_BETA,
bindsThinkingBlocks(resolvedModel) ? THINKING_BINDING_BETA : FINE_GRAINED_TOOL_STREAMING_BETA,
);
if (requestsThinkingDisplayUpdates(requestOptions.thinking)) {
requestOptions.clientOptions.defaultHeaders = appendAnthropicBetaHeader(
requestOptions.clientOptions.defaultHeaders as Record<string, string> | undefined,
THINKING_DISPLAY_UPDATES_BETA,
);
}
}

/**
Expand Down
14 changes: 14 additions & 0 deletions packages/api/src/utils/tokens.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -297,6 +297,20 @@ describe('Opus 5.5 token limits', () => {
});
});

describe('Sonnet 5.5 token limits', () => {
it.each([
'claude-sonnet-5-5',
'claude-sonnet-5.5',
'anthropic/claude-sonnet-5-5',
'global.anthropic.claude-sonnet-5-5',
])('resolves %s to the modern Claude profile', (model) => {
expect(getModelMaxTokens(model)).toBe(1000000);
expect(getModelMaxOutputTokens(model)).toBe(128000);
expect(getModelMaxTokens(model, EModelEndpoint.anthropic)).toBe(1000000);
expect(getModelMaxOutputTokens(model, EModelEndpoint.anthropic)).toBe(128000);
});
});

describe.each(['gpt-6-sol', 'gpt-6-luna'])('%s token limits', (model) => {
it('resolves exact, snapshot, and provider-prefixed IDs', () => {
for (const name of [model, `${model}-2026-09-22`, `openai/${model}`]) {
Expand Down
Loading
Loading