Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,7 @@ console.log(answer);

Notes:

- `mode` defaults to `'system1'`; passing `'system1'` currently throws
- `mode` defaults to `'system1'`; passing `'system2'` currently throws
(`Not implemented yet`).
- `confidence` is derived from the distribution shape. See
[under the hood](#how-it-works-under-the-hood) for the exact formula.
Expand Down
40 changes: 0 additions & 40 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

8 changes: 3 additions & 5 deletions package.json
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,12 @@
"lint": "eslint .",
"format": "prettier --write .",
"format:check": "prettier --check .",
"build": "tsc",
"build": "tsc -p tsconfig.build.json",
"typecheck": "tsc",
"test": "vitest run",
"test:coverage": "vitest run --coverage",
"example": "node --env-file=.env --import tsx examples/basic.ts",
"prepublishOnly": "rm -rf dist && npm run lint && npm run format:check && npm run build && npm test"
"prepublishOnly": "rm -rf dist && npm run lint && npm run format:check && npm run typecheck && npm run build && npm test"
},
"repository": {
"type": "git",
Expand Down Expand Up @@ -53,8 +54,5 @@
"typescript": "^6.0.3",
"typescript-eslint": "^8.70.1",
"vitest": "^5.0.1"
},
"dependencies": {
"openai": "^7.23.0"
}
}
41 changes: 26 additions & 15 deletions src/choice/system1-choice.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
import OpenAI from 'openai';
import { generateText } from '../utils/llms/generate-text.js';
import { normalizeEntropy } from '../utils/math/normalize-entropy.js';
import type { ChoiceAnswer } from '../types/choice-answer.js';
import type { Question } from '../types/question.js';
Expand All @@ -22,6 +22,18 @@ const LETTERS = 'ABCDEFGHIJKLMNOPQRSTUVWXYZ'.split('');
* distribution over every option (sums to 1), and a 0..1 confidence
* based on the distribution's entropy (flat → low, single peak → high).
* @throws If there are fewer than 2 or more than 26 options (one per letter).
* @example
* ```ts
* const answer = await system1Choice({
* apiBaseUrl: 'http://localhost:8000/v1',
* apiKey: process.env.API_KEY!,
* model: '/models/Qwen3.5-4B-Q4_K_M.gguf',
* state: "It is raining and I am at home. I'm bored.",
* instructions: 'Give me a good plan to do now',
* criteria: { walk: 'Go for a walk', movie: 'Watch a movie' },
* });
* console.log(answer.choice); // 'movie'
* ```
*/
export async function system1Choice(question: Question): Promise<ChoiceAnswer> {
const entries = Object.entries(question.criteria);
Expand All @@ -46,23 +58,22 @@ export async function system1Choice(question: Question): Promise<ChoiceAnswer> {
`Answer with exactly one letter (${LETTERS.slice(0, names.length).join(', ')}). ` +
`Reply with that single letter and nothing else.`;

const client = new OpenAI({
baseURL: question.apiBaseUrl,
apiKey: question.apiKey,
});

// The whole decision is one forward pass generating one token. Each param below
// nudges the model towards emitting just the chosen option's letter.
// TODO: probably better to move instructions to system prompt for KV cache reuse
const res = await client.chat.completions.create({
model: question.model,
messages: [{ role: 'user', content: prompt }],
max_tokens: 1, // the answer is a single letter
temperature: 0, // greedy: always the most likely letter
logprobs: true,
top_logprobs: 50, // llama.cpp server max; margin so every declared letter (and its token variants) lands in the report
chat_template_kwargs: { enable_thinking: false },
} as OpenAI.Chat.ChatCompletionCreateParamsNonStreaming);
const res = await generateText(
{ apiBaseUrl: question.apiBaseUrl, apiKey: question.apiKey },
{
model: question.model,
messages: [{ role: 'user', content: prompt }],
max_tokens: 1, // the answer is a single letter
temperature: 0, // greedy: always the most likely letter
logprobs: true,
top_logprobs: 50, // llama.cpp server max; margin so every declared letter (and its token variants) lands in the report
chat_template_kwargs: { enable_thinking: false },
},
{ maxRetries: question.maxRetries, timeoutMs: question.timeoutMs },
);

// The first generated token is the answer letter; its candidate tokens carry
// the logprobs we turn into the option distribution. Missing logprobs means we
Expand Down
18 changes: 18 additions & 0 deletions src/types/chat-completion-choice.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
import type { ChatCompletionTokenLogprob } from './chat-completion-token-logprob.js';

/**
* One completion alternative as returned by the API.
*
* @example
* ```ts
* const choice: ChatCompletionChoice = {
* logprobs: {
* content: [{ token: 'A', logprob: -0.5, top_logprobs: [{ token: 'A', logprob: -0.5 }] }],
* },
* };
* ```
*/
export interface ChatCompletionChoice {
/** Log probability information for the generated content, when requested */
logprobs?: { content?: ChatCompletionTokenLogprob[] | null } | null;
}
37 changes: 37 additions & 0 deletions src/types/chat-completion-request.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
import type { ChatMessage } from './chat-message.js';

/**
* Chat completions request body, in the wire format of the OpenAI-compatible
* `/v1/chat/completions` API.
*
* Keys are kept verbatim from the API spec, so no mapping layer is needed
* between this type and the request body.
*
* @example
* ```ts
* const request: ChatCompletionRequest = {
* model: '/models/Qwen3.5-4B-Q4_K_M.gguf',
* messages: [{ role: 'user', content: 'Reply with a single letter: A or B' }],
* max_tokens: 1,
* temperature: 0,
* logprobs: true,
* top_logprobs: 20,
* };
* ```
*/
export interface ChatCompletionRequest {
/** The model identifier as required by the API provider */
model: string;
/** The conversation so far */
messages: ChatMessage[];
/** Maximum number of tokens to generate */
max_tokens?: number;
/** Sampling temperature; 0 makes generation (near-)greedy */
temperature?: number;
/** Whether to return log probabilities of the generated tokens */
logprobs?: boolean;
/** How many of the most likely tokens to report at each position (requires `logprobs`) */
top_logprobs?: number;
/** Extra parameters forwarded to the model's chat template, i.e. `{ enable_thinking: false }` for llama.cpp */
chat_template_kwargs?: Record<string, unknown>;
}
25 changes: 25 additions & 0 deletions src/types/chat-completion-token-logprob.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
import type { TopLogprob } from './top-logprob.js';

/**
* Log probability information for one generated content token.
*
* @example
* ```ts
* const info: ChatCompletionTokenLogprob = {
* token: 'A',
* logprob: -0.5,
* top_logprobs: [
* { token: 'A', logprob: -0.5 },
* { token: 'B', logprob: -1.2 },
* ],
* };
* ```
*/
export interface ChatCompletionTokenLogprob {
/** The generated token */
token: string;
/** Natural log of the probability the model assigned to this token */
logprob: number;
/** The most likely tokens at this position; may be fewer than requested */
top_logprobs?: TopLogprob[];
}
21 changes: 21 additions & 0 deletions src/types/chat-completion.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
import type { ChatCompletionChoice } from './chat-completion-choice.js';

/**
* A chat completions response, narrowed to the fields this library reads.
*
* Response fields we don't consume (ids, usage, timestamps, ...) are passed
* through untouched at runtime; they are just not typed.
*
* @example
* ```ts
* const completion: ChatCompletion = {
* choices: [
* { logprobs: { content: [{ token: 'A', logprob: -0.5, top_logprobs: [] }] } },
* ],
* };
* ```
*/
export interface ChatCompletion {
/** The generated choices; the first one is the answer */
choices: ChatCompletionChoice[];
}
15 changes: 15 additions & 0 deletions src/types/chat-message.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
/**
* A message of the conversation, as accepted by OpenAI-compatible chat
* completions APIs.
*
* @example
* ```ts
* const message: ChatMessage = { role: 'user', content: 'Hello!' };
* ```
*/
export interface ChatMessage {
/** Who is speaking: the prompt author (`user`), a previous model reply (`assistant`) or standing instructions (`system`) */
role: 'system' | 'user' | 'assistant';
/** Message text */
content: string;
}
13 changes: 13 additions & 0 deletions src/types/choice-answer.ts
Original file line number Diff line number Diff line change
@@ -1,3 +1,16 @@
/**
* The outcome of a decision: which option won, how likely every option was and
* how decisive the win was.
*
* @example
* ```ts
* const answer: ChoiceAnswer = {
* choice: 'movie',
* probabilities: { walk: 0.00081, movie: 0.99913, beach: 0.00006 },
* confidence: 0.9935,
* };
* ```
*/
export interface ChoiceAnswer {
/** Option name with the highest probability */
choice: string;
Expand Down
13 changes: 12 additions & 1 deletion src/types/choice-mode.ts
Original file line number Diff line number Diff line change
@@ -1,2 +1,13 @@
/** Which mode to answer with. Follows Kahneman's dual-process theory (Thinking, Fast and Slow): System 1 is fast, automatic and instinctive, and costs little in inference (one forward pass, one generated token); System 2 is slow, effortful and deliberate, and more costly (the LLM may reason and has to generate a full structured response). */
/**
* Which mode to answer with, following Kahneman's dual-process theory (Thinking,
* Fast and Slow): System 1 is fast, automatic and instinctive, and costs little
* in inference (one forward pass, one generated token); System 2 is slow,
* effortful and deliberate, and more costly (the LLM may reason and has to
* generate a full structured response).
*
* @example
* ```ts
* const mode: ChoiceMode = 'system1';
* ```
*/
export type ChoiceMode = 'system1' | 'system2';
14 changes: 14 additions & 0 deletions src/types/generate-text-options.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
/**
* Retry and timeout settings for `generateText`.
*
* @example
* ```ts
* const options: GenerateTextOptions = { maxRetries: 3, timeoutMs: 30_000 };
* ```
*/
export interface GenerateTextOptions {
/** Retries after a failed attempt: network errors and 408/409/429/5xx responses (not timeouts). Defaults to 2 */
maxRetries?: number | undefined;
/** Per-attempt timeout in milliseconds. Defaults to 600000 */
timeoutMs?: number | undefined;
}
19 changes: 19 additions & 0 deletions src/types/question.ts
Original file line number Diff line number Diff line change
@@ -1,12 +1,31 @@
import type { ChoiceMode } from './choice-mode.js';

/**
* The decision to make: options, state, instructions and provider settings.
*
* @example
* ```ts
* const question: Question = {
* apiBaseUrl: 'http://localhost:8000/v1',
* apiKey: 'a-super-secret-api-key',
* model: '/models/Qwen3.5-4B-Q4_K_M.gguf',
* state: "It is raining and I am at home. I'm bored.",
* instructions: 'Give me a good plan to do now',
* criteria: { walk: 'Go for a walk', movie: 'Watch a movie' },
* };
* ```
*/
export interface Question {
/** Base url of an OpenAI compatible v1 API, i.e. `'http://localhost:8000/v1'` */
apiBaseUrl: string;
/** The API key as required or not by your provider */
apiKey: string;
/** The model identifier as required by your API provider. i.e. `'/models/Qwen3.5-4B-Q4_K_M.gguf'` for llama.cpp */
model: string;
/** Maximum number of retries after a failed attempt (network errors, 408/409/429/5xx). Defaults to 2 */
maxRetries?: number;
/** Per-attempt timeout in milliseconds. Defaults to 600000 */
timeoutMs?: number;
/** Choose between System 1 or System 2 mode. Defaults to `'system1'` */
mode?: ChoiceMode;
/** The state to evaluate */
Expand Down
15 changes: 15 additions & 0 deletions src/types/top-logprob.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
/**
* A candidate token and the log probability the model assigned to it at a
* generation position.
*
* @example
* ```ts
* const candidate: TopLogprob = { token: 'A', logprob: -0.5 };
* ```
*/
export interface TopLogprob {
/** The candidate token */
token: string;
/** Natural log of the token probability */
logprob: number;
}
22 changes: 22 additions & 0 deletions src/utils/error/message-of.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
/**
* Human-readable message of any thrown value, `Error` or not.
*
* JavaScript can throw anything. `Error` instances carry a `message`, but values
* thrown by other code paths (strings, numbers, response bodies, third-party
* classes) do not. This reads the `message` property of `Error` instances and
* stringifies everything else, so the result is always a usable string for
* logging, error wrapping or user-facing messages.
*
* @param err - The thrown value, of any type.
* @returns The `message` of `Error` instances, or the string representation of
* anything else.
* @example
* ```ts
* messageOf(new Error('connection refused')); // 'connection refused'
* messageOf(new DOMException('aborted', 'AbortError')); // 'aborted'
* messageOf('boom'); // 'boom'
* messageOf(42); // '42'
* ```
*/
export const messageOf = (err: unknown): string =>
err instanceof Error ? err.message : String(err);
Loading
Loading