diff --git a/.env.example b/.env.example
index eee9bea9f21..19e0e6155af 100644
--- a/.env.example
+++ b/.env.example
@@ -595,7 +595,7 @@ GOOGLE_KEY=user_provided
#============#
OPENAI_API_KEY=user_provided
-# OPENAI_MODELS=gpt-6-astra,gpt-6-sol,gpt-6-luna,gpt-5.6,gpt-5.6-terra,gpt-5.6-luna,gpt-5.5,gpt-5.5-pro,chat-latest,gpt-5.4,gpt-5.4-pro,gpt-5.4-mini,gpt-5.4-nano,gpt-5.3-codex,gpt-5.2,gpt-5,gpt-5-codex,gpt-5-mini,gpt-5-nano,o3-pro,o3,o4-mini,gpt-4.1,gpt-4.1-mini,gpt-4.1-nano,o3-mini,o1-pro,o1,gpt-4o,gpt-4o-mini
+# OPENAI_MODELS=gpt-6-astra,gpt-6.1-sol,gpt-6-sol,gpt-6-luna,gpt-5.6,gpt-5.6-terra,gpt-5.6-luna,gpt-5.5,gpt-5.5-pro,chat-latest,gpt-5.4,gpt-5.4-pro,gpt-5.4-mini,gpt-5.4-nano,gpt-5.3-codex,gpt-5.2,gpt-5,gpt-5-codex,gpt-5-mini,gpt-5-nano,o3-pro,o3,o4-mini,gpt-4.1,gpt-4.1-mini,gpt-4.1-nano,o3-mini,o1-pro,o1,gpt-4o,gpt-4o-mini
DEBUG_OPENAI=false
diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS
new file mode 100644
index 00000000000..a39dace8506
--- /dev/null
+++ b/.github/CODEOWNERS
@@ -0,0 +1,5 @@
+# Enable required code-owner reviews in dev/main branch rules to enforce this.
+/.github/CODEOWNERS @danny-avila
+/.github/workflows/ @danny-avila
+/.github/scripts/ @danny-avila
+/.github/MAIN_PROMOTION.md @danny-avila
diff --git a/.github/MAIN_PROMOTION.md b/.github/MAIN_PROMOTION.md
new file mode 100644
index 00000000000..c5627578928
--- /dev/null
+++ b/.github/MAIN_PROMOTION.md
@@ -0,0 +1,138 @@
+# Manually Approved Main Promotion
+
+`main` is a release pointer into `dev` history. The **Promote Dev to Main** workflow
+moves that pointer to one approved, fully tested `dev` commit. It creates no merge,
+rebase, squash, or new commit and never force-pushes. It is not an automated merge bot.
+
+## Trust boundary
+
+```text
+maintainer dispatches workflow on main with two exact SHAs
+ -> different environment reviewer approves
+ -> read-only validation of platform protections, actors, refs and full CI
+ -> short-lived App token for this repository, Contents: write only
+ -> repeat all validation, fetch into a bare object store, non-forced push
+ -> verify the new main SHA and revoke the App token
+```
+
+The privileged job runs inline code from the workflow revision on `main`, not a
+script from the candidate branch. It does not check out application source, restore
+caches, download PR artifacts, install packages, or execute repository-local actions.
+The sole action used there is `actions/create-github-app-token`, pinned to a full
+upstream commit SHA. The separate policy-test job has read-only permissions, no App
+credential, no protected environment, and does not persist checkout credentials.
+
+Inputs travel through environment variables and must be complete lowercase SHA-1
+commit IDs before they are used as API or Git arguments. Repository identity, GitHub
+hosts, dispatch event, workflow path and `main` ref are fixed. Both the original actor
+and any rerun actor must have GitHub's `admin` or `maintain` role. API failures,
+incomplete responses, missing CI runs, duplicate/missing jobs and unexpected states
+refuse promotion. There is no bypass or "ignore failures" input.
+
+The CI gate accepts only the latest full `push` or `workflow_dispatch` run on `dev`
+for the exact input SHA, from the repository's existing backend and frontend workflow
+IDs and paths. All build, typecheck, test-shard and circular-dependency jobs must be
+present and successful. Only the PR-only **Codegraph select** job may be skipped.
+An older successful run cannot conceal a newer failed or pending full run. PR checks
+and checks from a different commit do not qualify. This gate does not imply that
+Lighthouse, Playwright or integration lanes that only run on PRs tested the resulting
+post-merge SHA; maintainers still assess those PR results and release readiness.
+
+The candidate must be the current `dev` tip, and the reviewed baseline must still be
+the current `main` tip. `main` must be an ancestor of the candidate. Ref and CI
+validation is repeated after approval and immediately before the push. Concurrency
+serializes promotion runs without cancelling a job mid-push. Git's non-forced server
+update rejects a concurrent change that cannot fast-forward to the candidate. The
+cross-ref check is not an atomic lock on `dev`: if it advances after the final read,
+`main` still receives only the pinned, previously approved SHA, never the new tip.
+
+## Required repository setup (administrator)
+
+This PR adds code, **not** platform protection settings or credentials. Do not place
+the promotion key in repository-wide or organization-wide secrets. Keep it exclusively
+in the protected environment below. A missing environment can be auto-created by
+GitHub without protection; the inline gate refuses that state before minting a token.
+
+1. Protect **both `dev` and `main`** with rulesets/branch protection:
+ - Disallow force pushes and deletions. Require code-owner reviews for workflow,
+ script and CODEOWNERS changes, with stale approvals dismissed and approval of
+ the most recent reviewable push by someone other than its author.
+ - Restrict updates to `main` to release maintainers and the dedicated promotion
+ App. A fast-forward App update is not a PR merge: if a PR-only rule prevents
+ it, decide the narrowly scoped release-App exception explicitly. Never give
+ the App unrestricted bypass of force-push, deletion or other safety rules.
+ - `.github/CODEOWNERS` requests Danny's review of the release control plane;
+ it enforces nothing until the platform requires code-owner review. Add another
+ trusted code owner before Danny-authored control-plane changes can satisfy
+ mandatory code-owner review; the author cannot approve their own PR.
+2. Create an environment named **`main-promotion`**:
+ - Configure trusted **Required reviewers** and enable **Prevent self-review**.
+ Use enough trusted reviewers that someone other than the initiator can approve.
+ - Disable **Allow administrators to bypass configured protection rules**.
+ - Select **Selected branches and tags**. Add exactly one rule: type **Branch**,
+ name **`main`**. No tag rule, wildcard, `dev`, or other branch.
+3. Create a **dedicated GitHub App**, installed only on `LibreChat-AI/LibreChat`, with
+ repository **Contents: read and write** (and GitHub's required metadata access).
+ Do not reuse a PAT or a broad existing App. Do not grant Workflows, Actions,
+ Administration, Secrets or organization permissions.
+4. In that environment only, store:
+ - Variable **`MAIN_PROMOTION_APP_ID`**: the dedicated App's ID.
+ - Secret **`MAIN_PROMOTION_APP_PRIVATE_KEY`**: its private key.
+ The workflow requests a repository-scoped token with only Contents: write. The
+ pinned token action attempts revocation in its post-job step, including failures.
+ An interrupted runner may not execute cleanup; the installation token's short
+ lifetime bounds that residual risk. Do not disable token revocation.
+5. Merge this implementation into **`dev`** after review. Bootstrap it onto `main`
+ once with a separately approved, ordinary exact-SHA fast-forward; `workflow_dispatch`
+ cannot run a new workflow until its definition reaches the default branch.
+ Never bypass failed checks or protections as part of bootstrap. Changes to any
+ `.github/workflows/`, `.github/scripts/` or `.github/CODEOWNERS` file are deliberately
+ excluded from automated promotion, including this workflow's own updates. Those
+ use the separately reviewed manual path, so the App needs no Workflows permission.
+
+A repository administrator and the trusted `main` workflow are the root of trust.
+This is not a defense against a malicious repository administrator or a compromised
+reviewer who knowingly authorizes a malicious release. Nor does it harden the other
+existing publish workflows: an App push to `main` intentionally triggers the existing
+main-push automation, with its own permissions and dependencies. Review that downstream
+publishing surface separately before relying on the release process end to end.
+
+## Promote a normal tested dev commit
+
+1. Run `git ls-remote --heads origin main dev` in your own upstream clone. Review the
+ candidate and both **full** CI runs at the exact `dev` SHA.
+2. If path filters omitted a full run for that SHA, run the existing **Backend Unit
+ Tests** and **Frontend Unit Tests** workflows manually on `dev`, then wait for
+ success. Their new `workflow_dispatch` trigger uses full suites, not PR selection.
+3. Dispatch **Promote Dev to Main** using branch **`main`**. Supply the full `dev` SHA
+ as `dev_sha` and the full current `main` SHA as `expected_main_sha`. Example:
+
+ ```sh
+ gh workflow run promote-main.yml --repo LibreChat-AI/LibreChat --ref main \
+ -f dev_sha= \
+ -f expected_main_sha=
+ ```
+
+4. A different configured reviewer examines the input SHAs and approves the
+ `main-promotion` environment deployment. The job still revalidates everything.
+5. Read the run summary and confirm `git ls-remote --heads origin main`. Monitor
+ downstream main-push publishing workflows separately. If refs or CI changed,
+ review a fresh dispatch. If the final read failed after a push, inspect the live
+ `main` tip before retrying: the push may have succeeded. Do not force-push.
+
+Rollback of code is a new reviewed revert on `dev`, followed by another approved
+fast-forward. Never move `main` backwards. Promotion adds no merge commit, but any
+merge commits already present in `dev` are preserved as part of its history.
+
+## Local validation
+
+```sh
+python3 -I .github/scripts/test_main_promotion.py
+```
+
+Tests extract and execute the workflow's actual inline policy against controlled
+API responses and exercise real local bare Git repositories. They never use live
+GitHub credentials or update a remote repository. Run the policy-test workflow and
+YAML/action lint before delivery. A production promotion cannot be verified until
+administrator setup and bootstrap are complete; do not report it as deployed merely
+because local tests pass.
diff --git a/.github/scripts/test_main_promotion.py b/.github/scripts/test_main_promotion.py
new file mode 100644
index 00000000000..c9bce8b2585
--- /dev/null
+++ b/.github/scripts/test_main_promotion.py
@@ -0,0 +1,450 @@
+"""Exercise the policy embedded in the trusted workflow, using only the stdlib."""
+import copy
+import json
+import os
+import re
+import subprocess
+import sys
+import tempfile
+import types
+import unittest
+from pathlib import Path
+from unittest.mock import patch
+
+ROOT = Path(__file__).resolve().parents[2]
+WORKFLOW = ROOT / '.github/workflows/promote-main.yml'
+TEXT = WORKFLOW.read_text()
+MATCH = re.search(r"^ cat > .* <<'PY'\n(.*?)^ PY$", TEXT, re.M | re.S)
+if MATCH is None:
+ raise RuntimeError('Inline production policy was not found')
+SOURCE = '\n'.join(line[10:] for line in MATCH[1].splitlines())
+POLICY = types.ModuleType('promotion_policy')
+exec(compile(SOURCE, str(WORKFLOW), 'exec'), POLICY.__dict__)
+DEV = 'd' * 40
+MAIN = 'a' * 40
+ENV = {
+ 'DEV_SHA': DEV,
+ 'EXPECTED_MAIN_SHA': MAIN,
+ 'GITHUB_REPOSITORY': POLICY.REPO,
+ 'GITHUB_EVENT_NAME': 'workflow_dispatch',
+ 'GITHUB_REF': 'refs/heads/main',
+ 'GITHUB_SERVER_URL': 'https://github.com',
+ 'GITHUB_API_URL': 'https://api.github.com',
+ 'WORKFLOW_REF': POLICY.WORKFLOW,
+ 'WORKFLOW_SHA': MAIN,
+ 'GITHUB_SHA': MAIN,
+ 'GITHUB_ACTOR': 'danny-avila',
+ 'GITHUB_TRIGGERING_ACTOR': 'release-maintainer',
+ 'GH_TOKEN': 'read-only-fixture',
+ 'PROMOTION_TOKEN': 'write-fixture',
+}
+
+
+def fixtures():
+ result = {
+ 'environments/main-promotion': {
+ 'can_admins_bypass': False,
+ 'deployment_branch_policy': {
+ 'protected_branches': False, 'custom_branch_policies': True,
+ },
+ 'protection_rules': [{
+ 'type': 'required_reviewers', 'prevent_self_review': True,
+ 'reviewers': [{'type': 'User', 'reviewer': {'login': 'reviewer'}}],
+ }],
+ },
+ 'environments/main-promotion/deployment-branch-policies?per_page=100': {
+ 'total_count': 1, 'branch_policies': [{'name': 'main', 'type': 'branch'}],
+ },
+ }
+ for actor in ('danny-avila', 'release-maintainer'):
+ result[f'collaborators/{actor}/permission'] = {
+ 'permission': 'admin', 'role_name': 'admin',
+ }
+ for branch, sha in [('main', MAIN), ('dev', DEV)]:
+ result[f'git/ref/heads/{branch}'] = {
+ 'ref': f'refs/heads/{branch}', 'object': {'type': 'commit', 'sha': sha},
+ }
+ for index, (filename, (workflow_id, required)) in enumerate(POLICY.CI.items(), 1):
+ result[f'actions/workflows/{filename}'] = {
+ 'id': workflow_id, 'path': f'.github/workflows/{filename}', 'state': 'active',
+ }
+ run = {
+ 'id': index, 'run_attempt': 2, 'workflow_id': workflow_id,
+ 'path': f'.github/workflows/{filename}', 'event': 'push',
+ 'head_branch': 'dev', 'head_sha': DEV,
+ 'repository': {'full_name': POLICY.REPO},
+ 'status': 'completed', 'conclusion': 'success',
+ }
+ result[f'actions/workflows/{workflow_id}/runs?head_sha={DEV}&per_page=100'] = {
+ 'total_count': 1, 'workflow_runs': [run],
+ }
+ result[f'actions/runs/{index}'] = copy.deepcopy(run)
+ jobs = [{'name': name, 'head_sha': DEV, 'status': 'completed', 'conclusion': 'success'}
+ for name in sorted(required)]
+ jobs.append({'name': 'Codegraph select', 'head_sha': DEV,
+ 'status': 'completed', 'conclusion': 'skipped'})
+ result[f'actions/runs/{index}/attempts/2/jobs?per_page=100'] = {
+ 'total_count': len(jobs), 'jobs': jobs,
+ }
+ return result
+
+
+class PolicyTests(unittest.TestCase):
+ def setUp(self):
+ self.data = fixtures()
+ self.calls = []
+ self.git_calls = []
+ self.scratch = tempfile.TemporaryDirectory(dir=ROOT)
+ self.addCleanup(self.scratch.cleanup)
+ self.addCleanup(patch.stopall)
+ self.env = dict(ENV, RUNNER_TEMP=self.scratch.name,
+ GITHUB_STEP_SUMMARY=str(Path(self.scratch.name) / 'summary'))
+ patch.dict(os.environ, self.env, clear=True).start()
+ patch.object(POLICY, 'api', self.api).start()
+ patch.object(POLICY, 'git', self.git).start()
+
+ def api(self, path):
+ self.calls.append(path)
+ if path not in self.data:
+ raise POLICY.Refused('Missing fixture')
+ return copy.deepcopy(self.data[path])
+
+ def git(self, args, token):
+ self.git_calls.append((args, token))
+ if 'rev-parse' in args:
+ return MAIN if args[-1] == 'refs/heads/main' else DEV
+ return ''
+
+ def refuse(self, operation=POLICY.verify):
+ with self.assertRaises(POLICY.Refused):
+ operation()
+ self.assertFalse(any('push' in args for args, _ in self.git_calls))
+
+ def ci_run(self, index=1):
+ workflow_id = list(POLICY.CI.values())[index - 1][0]
+ return self.data[f'actions/workflows/{workflow_id}/runs?head_sha={DEV}&per_page=100']
+
+ def ci_jobs(self, index=1):
+ return self.data[f'actions/runs/{index}/attempts/2/jobs?per_page=100']
+
+ def test_verified_full_ci_fetches_objects_but_never_pushes(self):
+ self.assertEqual(POLICY.verify()[:2], (DEV, MAIN))
+ self.assertTrue(any('merge-base' in args for args, _ in self.git_calls))
+ self.assertFalse(any('checkout' in args or 'push' in args for args, _ in self.git_calls))
+ self.assertTrue(all(token == ENV['GH_TOKEN'] for _, token in self.git_calls))
+
+ def test_rejects_shell_injection_and_nonexact_hashes(self):
+ for candidate in ('dev', DEV[:7], DEV.upper(), DEV + '\n', '--all',
+ '$(touch release)', DEV + ';echo unsafe', ''):
+ with self.subTest(candidate=candidate), patch.dict(os.environ, DEV_SHA=candidate):
+ self.refuse()
+ self.assertEqual(self.git_calls, [])
+
+ def test_requires_main_dispatch_and_fixed_repository_and_hosts(self):
+ for key, value in {
+ 'GITHUB_REPOSITORY': 'attacker/LibreChat', 'GITHUB_EVENT_NAME': 'pull_request_target',
+ 'GITHUB_REF': 'refs/heads/dev', 'GITHUB_SERVER_URL': 'https://evil.example',
+ 'GITHUB_API_URL': 'https://evil.example', 'WORKFLOW_REF': POLICY.WORKFLOW + '-evil',
+ 'WORKFLOW_SHA': DEV, 'GITHUB_SHA': DEV,
+ }.items():
+ with self.subTest(key=key), patch.dict(os.environ, {key: value}):
+ self.refuse()
+
+ def test_requires_original_and_rerun_actor_authorization(self):
+ for actor in ('danny-avila', 'release-maintainer'):
+ path = f'collaborators/{actor}/permission'
+ original = self.data[path]
+ self.data[path] = {'permission': 'write', 'role_name': 'write'}
+ self.refuse()
+ self.data[path] = original
+
+ def test_maintain_role_is_allowed_but_custom_writer_role_is_not(self):
+ path = 'collaborators/release-maintainer/permission'
+ self.data[path] = {'permission': 'write', 'role_name': 'maintain'}
+ self.assertEqual(POLICY.context(), (DEV, MAIN))
+ self.data[path] = {'permission': 'write', 'role_name': 'publisher'}
+ self.refuse()
+
+ def test_actor_cannot_inject_an_api_path(self):
+ with patch.dict(os.environ, GITHUB_ACTOR='../git/refs'):
+ self.refuse()
+
+ def test_missing_or_unprotected_environment_refuses(self):
+ value = self.data['environments/main-promotion']
+ for key, replacement in [
+ ('can_admins_bypass', True), ('can_admins_bypass', None),
+ ('protection_rules', []), ('deployment_branch_policy', None),
+ ('deployment_branch_policy', {'protected_branches': True, 'custom_branch_policies': False}),
+ ]:
+ with self.subTest(key=key, replacement=replacement):
+ old = value[key]
+ value[key] = replacement
+ self.refuse()
+ value[key] = old
+ del self.data['environments/main-promotion']
+ self.refuse()
+
+ def test_requires_nonself_approval_and_reviewers(self):
+ rule = self.data['environments/main-promotion']['protection_rules'][0]
+ for key, value in [('prevent_self_review', False), ('prevent_self_review', None), ('reviewers', [])]:
+ old = rule[key]
+ rule[key] = value
+ self.refuse()
+ rule[key] = old
+
+ def test_environment_rejects_wildcards_tags_and_extra_branches(self):
+ path = 'environments/main-promotion/deployment-branch-policies?per_page=100'
+ for rows in [[{'name': '*', 'type': 'branch'}], [{'name': 'main', 'type': 'tag'}],
+ [{'name': 'main', 'type': 'branch'}, {'name': 'dev', 'type': 'branch'}], []]:
+ with self.subTest(rows=rows):
+ self.data[path] = {'total_count': len(rows), 'branch_policies': rows}
+ self.refuse()
+
+ def test_drifted_main_or_dev_refuses(self):
+ for branch in ('main', 'dev'):
+ value = self.data[f'git/ref/heads/{branch}']['object']
+ old = value['sha']
+ value['sha'] = 'b' * 40
+ self.refuse()
+ value['sha'] = old
+
+ def test_incomplete_or_oversized_responses_refuse(self):
+ for value in ({'total_count': 2, 'rows': [1]}, {'rows': []},
+ {'total_count': 100, 'rows': list(range(100))}):
+ with self.subTest(value=value.get('total_count')):
+ self.refuse(lambda: POLICY.complete_list(value, 'rows'))
+
+ def test_requires_real_workflow_ids_paths_and_active_state(self):
+ for filename, (workflow_id, _) in POLICY.CI.items():
+ workflow = self.data[f'actions/workflows/{filename}']
+ for key, replacement in [('id', workflow_id + 1), ('path', '.github/workflows/fake.yml'),
+ ('state', 'disabled_manually')]:
+ old = workflow[key]
+ workflow[key] = replacement
+ self.refuse()
+ workflow[key] = old
+
+ def test_only_exact_dev_full_runs_count_not_prs_forks_or_old_commits(self):
+ run = self.ci_run()['workflow_runs'][0]
+ for key, replacement in [('event', 'pull_request'), ('event', 'workflow_run'),
+ ('head_branch', 'topic'), ('head_sha', MAIN),
+ ('repository', {'full_name': 'attacker/LibreChat'})]:
+ old = run[key]
+ run[key] = replacement
+ self.refuse()
+ run[key] = old
+ run['event'] = 'workflow_dispatch'
+ self.data['actions/runs/1']['event'] = 'workflow_dispatch'
+ self.assertEqual(POLICY.verify()[:2], (DEV, MAIN))
+
+ def test_no_workflow_run_means_not_tested(self):
+ self.ci_run().update(total_count=0, workflow_runs=[])
+ self.refuse()
+
+ def test_newer_failed_or_pending_run_cannot_hide_behind_old_success(self):
+ value = self.ci_run()
+ old = copy.deepcopy(value['workflow_runs'][0])
+ for status, conclusion in [('completed', 'failure'), ('in_progress', None),
+ ('completed', 'cancelled'), ('completed', 'skipped')]:
+ value['workflow_runs'] = [old, dict(old, id=100, status=status, conclusion=conclusion)]
+ value['total_count'] = 2
+ self.refuse()
+
+ def test_ci_rerun_during_job_verification_refuses(self):
+ for key, value in [('run_attempt', 3), ('status', 'in_progress'), ('conclusion', 'failure')]:
+ old = self.data['actions/runs/1'][key]
+ self.data['actions/runs/1'][key] = value
+ self.refuse()
+ self.data['actions/runs/1'][key] = old
+
+ def test_every_required_shard_must_exist_and_succeed(self):
+ for index in (1, 2):
+ jobs = self.ci_jobs(index)
+ original = copy.deepcopy(jobs)
+ for name in POLICY.CI[list(POLICY.CI)[index - 1]][1]:
+ with self.subTest(job=name):
+ jobs['jobs'] = [job for job in original['jobs'] if job['name'] != name]
+ jobs['total_count'] = len(jobs['jobs'])
+ self.refuse()
+ jobs.update(original)
+ for conclusion in ('skipped', 'cancelled', 'neutral', 'failure', None):
+ jobs['jobs'][0]['conclusion'] = conclusion
+ self.refuse()
+ jobs.update(original)
+
+ def test_duplicate_incomplete_wrong_sha_and_failed_extra_jobs_refuse(self):
+ jobs = self.ci_jobs()
+ old = copy.deepcopy(jobs)
+ jobs['jobs'].append(copy.deepcopy(jobs['jobs'][0]))
+ jobs['total_count'] += 1
+ self.refuse()
+ jobs.update(copy.deepcopy(old))
+ jobs['jobs'][0]['head_sha'] = MAIN
+ self.refuse()
+ jobs.update(copy.deepcopy(old))
+ jobs['jobs'][0]['status'] = 'in_progress'
+ self.refuse()
+ jobs.update(copy.deepcopy(old))
+ jobs['jobs'].append({'name': 'Additional security test', 'head_sha': DEV,
+ 'status': 'completed', 'conclusion': 'failure'})
+ jobs['total_count'] += 1
+ self.refuse()
+
+ def test_control_plane_changes_refuse_automated_promotion(self):
+ original = self.git
+ def changed(args, token):
+ return '.github/workflows/changed.yml' if 'diff' in args else original(args, token)
+ with patch.object(POLICY, 'git', changed):
+ self.refuse()
+
+ def test_main_or_dev_can_change_during_fetch_without_any_push(self):
+ original = self.git
+ for branch in ('main', 'dev'):
+ def move(args, token):
+ if 'fetch' in args:
+ self.data[f'git/ref/heads/{branch}']['object']['sha'] = 'b' * 40
+ return original(args, token)
+ self.data.update(fixtures())
+ with patch.object(POLICY, 'git', move):
+ self.refuse()
+
+ def test_revalidation_refuses_before_using_write_token(self):
+ self.data['git/ref/heads/dev']['object']['sha'] = MAIN
+ with patch.object(sys, 'argv', ['policy', 'promote']):
+ self.refuse(POLICY.run)
+ self.assertFalse(any(token == ENV['PROMOTION_TOKEN'] for _, token in self.git_calls))
+
+ def test_success_pushes_only_pinned_sha_to_main_without_force(self):
+ original = self.git
+ def push(args, token):
+ if 'push' in args:
+ self.data['git/ref/heads/main']['object']['sha'] = DEV
+ return original(args, token)
+ with patch.object(POLICY, 'git', push), patch.object(sys, 'argv', ['policy', 'promote']):
+ POLICY.run()
+ pushes = [(args, token) for args, token in self.git_calls if 'push' in args]
+ self.assertEqual(len(pushes), 1)
+ self.assertEqual(pushes[0][0][-2:], [POLICY.REMOTE, f'{DEV}:refs/heads/main'])
+ self.assertEqual(pushes[0][1], ENV['PROMOTION_TOKEN'])
+ self.assertFalse(any('force' in arg for arg in pushes[0][0]))
+
+ def test_api_failures_and_malformed_data_fail_closed(self):
+ with patch.object(POLICY, 'api', side_effect=POLICY.Refused('API unavailable')):
+ self.refuse()
+
+ def test_already_current_does_not_push(self):
+ with patch.object(POLICY, 'verify', return_value=(DEV, DEV, [])), \
+ patch.object(POLICY, 'ref', return_value=DEV), \
+ patch.object(sys, 'argv', ['policy', 'promote']):
+ POLICY.run()
+ self.assertEqual(self.git_calls, [])
+ self.assertIn('Already current', Path(self.env['GITHUB_STEP_SUMMARY']).read_text())
+
+ def test_failed_post_push_read_does_not_repeat_the_push(self):
+ original = self.git
+ def push(args, token):
+ result = original(args, token)
+ if 'push' in args:
+ del self.data['git/ref/heads/main']
+ return result
+ with patch.object(POLICY, 'git', push), patch.object(sys, 'argv', ['policy', 'promote']):
+ with self.assertRaises(POLICY.Refused):
+ POLICY.run()
+ self.assertEqual(sum('push' in args for args, _ in self.git_calls), 1)
+
+
+class AdapterTests(unittest.TestCase):
+ def test_api_uses_fixed_host_and_never_prints_raw_errors(self):
+ failed = subprocess.CompletedProcess([], 1, stdout='credential-like diagnostic',
+ stderr='untrusted server error')
+ with patch.object(subprocess, 'run', return_value=failed) as run:
+ with self.assertRaisesRegex(POLICY.Refused, '^GitHub policy read failed'):
+ POLICY.api('git/ref/heads/main')
+ self.assertIn('github.com', run.call_args.args[0])
+ self.assertNotIn('shell', run.call_args.kwargs)
+ for output in ('not-json', ''):
+ with patch.object(subprocess, 'run', return_value=subprocess.CompletedProcess([], 0, output)):
+ with self.assertRaisesRegex(POLICY.Refused, 'Invalid GitHub response'):
+ POLICY.api('git/ref/heads/main')
+
+ def test_git_credentials_are_not_arguments_or_persisted_config(self):
+ result = subprocess.CompletedProcess([], 0, stdout='ok', stderr='')
+ token = 'synthetic-token'
+ with patch.object(subprocess, 'run', return_value=result) as run, patch('builtins.print'):
+ self.assertEqual(POLICY.git(['push', POLICY.REMOTE, DEV + ':refs/heads/main'], token), 'ok')
+ argv = run.call_args.args[0]
+ self.assertFalse(any(token in arg for arg in argv))
+ self.assertEqual(argv[:3], ['git', '-c', 'core.hooksPath=/dev/null'])
+ env = run.call_args.kwargs['env']
+ self.assertEqual(env['GIT_CONFIG_GLOBAL'], '/dev/null')
+ self.assertEqual(env['GIT_CONFIG_NOSYSTEM'], '1')
+ self.assertEqual(env['GIT_CONFIG_KEY_0'], 'http.https://github.com/.extraheader')
+ self.assertNotIn('shell', run.call_args.kwargs)
+
+
+class GitIntegrationTests(unittest.TestCase):
+ def test_bare_git_fast_forward_preserves_hashes_and_refuses_divergence(self):
+ with tempfile.TemporaryDirectory(dir=ROOT) as directory:
+ env = dict(os.environ, GIT_CONFIG_GLOBAL='/dev/null', GIT_CONFIG_NOSYSTEM='1',
+ GIT_AUTHOR_NAME='test', GIT_AUTHOR_EMAIL='test@example.invalid',
+ GIT_COMMITTER_NAME='test', GIT_COMMITTER_EMAIL='test@example.invalid')
+ origin = str(Path(directory) / 'origin.git')
+ local = str(Path(directory) / 'local.git')
+ def git(*args, stdin=None, check=True):
+ return subprocess.run(['git', *args], env=env, input=stdin, capture_output=True,
+ text=True, check=check).stdout.strip()
+ git('init', '--bare', origin)
+ tree = git('--git-dir', origin, 'mktree', stdin='')
+ main = git('--git-dir', origin, 'commit-tree', tree, '-m', 'main')
+ dev = git('--git-dir', origin, 'commit-tree', tree, '-p', main, '-m', 'dev')
+ divergent = git('--git-dir', origin, 'commit-tree', tree, '-p', main, '-m', 'other')
+ git('--git-dir', origin, 'update-ref', 'refs/heads/main', main)
+ git('--git-dir', origin, 'update-ref', 'refs/heads/dev', dev)
+ git('init', '--bare', local)
+ git('--git-dir', local, 'fetch', '--no-tags', origin,
+ '+refs/heads/main:refs/heads/main', '+refs/heads/dev:refs/heads/dev')
+ git('--git-dir', local, 'merge-base', '--is-ancestor', main, dev)
+ git('--git-dir', local, 'push', origin, f'{dev}:refs/heads/main')
+ self.assertEqual(git('--git-dir', origin, 'rev-parse', 'refs/heads/main'), dev)
+ git('--git-dir', origin, 'update-ref', 'refs/heads/main', divergent)
+ refused = subprocess.run(['git', '--git-dir', local, 'push', origin,
+ f'{dev}:refs/heads/main'], env=env, capture_output=True)
+ self.assertNotEqual(refused.returncode, 0)
+ self.assertEqual(git('--git-dir', origin, 'rev-parse', 'refs/heads/main'), divergent)
+
+
+class WorkflowBoundaryTests(unittest.TestCase):
+ def test_write_job_has_no_untrusted_execution_sources(self):
+ self.assertNotRegex(TEXT, r'pull_request_target:|workflow_run:|repository_dispatch:|schedule:')
+ self.assertNotRegex(TEXT, r'uses: (?:\./|actions/(?:checkout|cache|download-artifact|setup-node))')
+ self.assertNotRegex(TEXT, r'\bnpm\s|\bnpx\s|\bpip\s|--force|force-with-lease|shell:\s*python')
+ self.assertIn('permissions: {}', TEXT)
+ self.assertIn('environment: main-promotion', TEXT)
+ self.assertIn('cancel-in-progress: false', TEXT)
+ self.assertIn('permission-contents: write', TEXT)
+ self.assertNotIn('permission-workflows:', TEXT)
+ actions = re.findall(r'uses: ([^\s]+)', TEXT)
+ self.assertEqual(actions, ['actions/create-github-app-token@fee1f7d63c2ff003460e3d139729b119787bc349'])
+ self.assertNotRegex(SOURCE, r'\beval\(|\bexec\(|shell=True')
+
+ def test_inputs_and_credentials_are_not_interpolated_into_shell_source(self):
+ for block in re.findall(r' run: \|\n(.*?)(?=\n -|\Z)', TEXT, re.S):
+ self.assertNotIn('${{', block)
+ self.assertLess(TEXT.index('python3 -I "$RUNNER_TEMP/main-promotion.py" verify'),
+ TEXT.index('uses: actions/create-github-app-token@'))
+ self.assertNotIn('skip-token-revoke: true', TEXT)
+
+ def test_selected_pr_tests_cannot_masquerade_as_full_ci(self):
+ for filename in POLICY.CI:
+ text = (ROOT / '.github/workflows' / filename).read_text()
+ self.assertIn(' workflow_dispatch:', text)
+ self.assertIn("github.event_name == 'pull_request'", text)
+
+ def test_control_plane_is_code_owned(self):
+ owners = (ROOT / '.github/CODEOWNERS').read_text()
+ for path in ('/.github/CODEOWNERS', '/.github/workflows/', '/.github/scripts/'):
+ self.assertIn(f'{path} @danny-avila', owners)
+
+
+if __name__ == '__main__':
+ unittest.main(verbosity=2)
diff --git a/.github/workflows/backend-review.yml b/.github/workflows/backend-review.yml
index 82d0088cdc9..2845f4c4b29 100644
--- a/.github/workflows/backend-review.yml
+++ b/.github/workflows/backend-review.yml
@@ -24,6 +24,8 @@ on:
- 'config/circular-deps.mjs'
- '.github/workflows/backend-review.yml'
- '!**.md'
+ # Promotion requires full CI at the exact dev SHA, including path-filtered commits.
+ workflow_dispatch:
permissions:
contents: read
diff --git a/.github/workflows/frontend-review.yml b/.github/workflows/frontend-review.yml
index 40ebe5b7221..f6b88df7a09 100644
--- a/.github/workflows/frontend-review.yml
+++ b/.github/workflows/frontend-review.yml
@@ -24,6 +24,8 @@ on:
- 'package.json'
- 'package-lock.json'
- '.github/workflows/frontend-review.yml'
+ # Promotion requires full CI at the exact dev SHA, including path-filtered commits.
+ workflow_dispatch:
permissions:
contents: read
diff --git a/.github/workflows/promote-main.yml b/.github/workflows/promote-main.yml
new file mode 100644
index 00000000000..311828b220c
--- /dev/null
+++ b/.github/workflows/promote-main.yml
@@ -0,0 +1,275 @@
+name: Promote Dev to Main
+run-name: Promote dev ${{ inputs.dev_sha }} to main
+
+on:
+ workflow_dispatch:
+ inputs:
+ dev_sha:
+ description: 'Approved current dev commit, full lowercase 40-character SHA'
+ type: string
+ required: true
+ expected_main_sha:
+ description: 'Current main commit reviewed as the promotion baseline, full SHA'
+ type: string
+ required: true
+
+permissions: {}
+
+concurrency:
+ group: main-promotion
+ cancel-in-progress: false
+
+jobs:
+ promote:
+ if: >-
+ github.repository == 'LibreChat-AI/LibreChat' &&
+ github.event_name == 'workflow_dispatch' &&
+ github.ref == 'refs/heads/main'
+ runs-on: ubuntu-24.04
+ timeout-minutes: 10
+ environment: main-promotion
+ permissions:
+ contents: read
+ actions: read
+ env:
+ GH_HOST: github.com
+ GH_PROMPT_DISABLED: '1'
+ DEV_SHA: ${{ inputs.dev_sha }}
+ EXPECTED_MAIN_SHA: ${{ inputs.expected_main_sha }}
+ WORKFLOW_REF: ${{ github.workflow_ref }}
+ WORKFLOW_SHA: ${{ github.workflow_sha }}
+ steps:
+ # Inline policy comes from the workflow on main. No candidate code, checkout,
+ # artifacts, caches, dependency installs, or repository-local actions run here.
+ - name: Verify protected approval, commit ancestry, and full dev CI
+ env:
+ GH_TOKEN: ${{ github.token }}
+ run: |
+ set -euo pipefail
+ cat > "$RUNNER_TEMP/main-promotion.py" <<'PY'
+ import base64
+ import json
+ import os
+ import re
+ import subprocess
+ import sys
+ from pathlib import Path
+
+ REPO = 'LibreChat-AI/LibreChat'
+ REMOTE = 'https://github.com/LibreChat-AI/LibreChat.git'
+ WORKFLOW = REPO + '/.github/workflows/promote-main.yml@refs/heads/main'
+ SHA = re.compile(r'[0-9a-f]{40}')
+ LOGIN = re.compile(r'[A-Za-z0-9][A-Za-z0-9-]{0,38}')
+ CI = {
+ 'backend-review.yml': (59857785, {
+ 'Build packages', 'Circular dependency checks', 'TypeScript type checks',
+ 'Tests: data-provider', 'Tests: data-schemas',
+ *[f'Tests: api (shard {n}/3)' for n in range(1, 4)],
+ *[f'Tests: @librechat/api (shard {n}/4)' for n in range(1, 5)],
+ }),
+ 'frontend-review.yml': (59789527, {
+ 'Build packages', 'Client build recovery regression',
+ 'TypeScript type checks (client)', 'Vite build verification',
+ 'Tests: @librechat/client',
+ *[f'Tests: Ubuntu (shard {n}/2)' for n in range(1, 3)],
+ }),
+ }
+
+ class Refused(Exception):
+ pass
+
+ def require(condition, message):
+ if not condition:
+ raise Refused(message)
+
+ def api(path):
+ result = subprocess.run(
+ ['gh', 'api', '--hostname', 'github.com', '--method', 'GET',
+ '-H', 'Accept: application/vnd.github+json',
+ '-H', 'X-GitHub-Api-Version: 2022-11-28', f'repos/{REPO}/{path}'],
+ capture_output=True, text=True, check=False, timeout=30,
+ )
+ require(result.returncode == 0, 'GitHub policy read failed; refusing promotion')
+ try:
+ return json.loads(result.stdout)
+ except ValueError:
+ raise Refused('Invalid GitHub response; refusing promotion') from None
+
+ def complete_list(value, key):
+ items = value.get(key)
+ require(isinstance(items, list) and value.get('total_count') == len(items)
+ and len(items) < 100, 'Incomplete or oversized policy response')
+ return items
+
+ def ref(branch):
+ value = api(f'git/ref/heads/{branch}')
+ obj = value.get('object', {})
+ require(value.get('ref') == f'refs/heads/{branch}' and obj.get('type') == 'commit'
+ and SHA.fullmatch(obj.get('sha', '')), 'Invalid branch response')
+ return obj['sha']
+
+ def context():
+ dev = os.environ.get('DEV_SHA', '')
+ main = os.environ.get('EXPECTED_MAIN_SHA', '')
+ require(SHA.fullmatch(dev) and SHA.fullmatch(main), 'Supply two full lowercase commit SHAs')
+ require(os.environ.get('GITHUB_REPOSITORY') == REPO
+ and os.environ.get('GITHUB_EVENT_NAME') == 'workflow_dispatch'
+ and os.environ.get('GITHUB_REF') == 'refs/heads/main'
+ and os.environ.get('GITHUB_SERVER_URL') == 'https://github.com'
+ and os.environ.get('GITHUB_API_URL') == 'https://api.github.com'
+ and os.environ.get('WORKFLOW_REF') == WORKFLOW
+ and os.environ.get('WORKFLOW_SHA') == main
+ and os.environ.get('GITHUB_SHA') == main,
+ 'Dispatch the trusted workflow from the current main branch')
+ actors = {os.environ.get('GITHUB_ACTOR', ''),
+ os.environ.get('GITHUB_TRIGGERING_ACTOR', '')}
+ for actor in actors:
+ require(LOGIN.fullmatch(actor), 'Invalid workflow actor')
+ permission = api(f'collaborators/{actor}/permission')
+ require(permission.get('role_name') in ('admin', 'maintain')
+ and permission.get('permission') in ('admin', 'write'),
+ 'Both the initiator and rerun actor must be maintainers')
+ return dev, main
+
+ def approval():
+ value = api('environments/main-promotion')
+ require(value.get('can_admins_bypass') is False, 'Disable administrator approval bypass')
+ require(value.get('deployment_branch_policy') == {
+ 'protected_branches': False, 'custom_branch_policies': True,
+ }, 'Use a main-only environment branch policy')
+ rules = value.get('protection_rules', [])
+ require(any(rule.get('type') == 'required_reviewers'
+ and rule.get('prevent_self_review') is True
+ and bool(rule.get('reviewers')) for rule in rules),
+ 'Configure required reviewers and prevent self-review')
+ policies = complete_list(
+ api('environments/main-promotion/deployment-branch-policies?per_page=100'),
+ 'branch_policies',
+ )
+ require(len(policies) == 1 and policies[0].get('name') == 'main'
+ and policies[0].get('type') == 'branch',
+ 'Only the main branch may access promotion credentials')
+
+ def full_ci(dev):
+ for filename, (workflow_id, required) in CI.items():
+ workflow = api(f'actions/workflows/{filename}')
+ require(workflow.get('id') == workflow_id
+ and workflow.get('path') == f'.github/workflows/{filename}'
+ and workflow.get('state') == 'active', 'Unexpected CI workflow identity')
+ runs = complete_list(api(
+ f'actions/workflows/{workflow_id}/runs?head_sha={dev}&per_page=100'
+ ), 'workflow_runs')
+ eligible = [run for run in runs
+ if run.get('event') in ('push', 'workflow_dispatch')
+ and run.get('head_branch') == 'dev'
+ and run.get('head_sha') == dev
+ and run.get('repository', {}).get('full_name') == REPO]
+ require(bool(eligible), f'Missing full dev CI: {filename}')
+ run = max(eligible, key=lambda item: item['id'])
+ require(run.get('workflow_id') == workflow_id
+ and run.get('path') == f'.github/workflows/{filename}'
+ and run.get('status') == 'completed' and run.get('conclusion') == 'success',
+ f'Latest full dev CI is not successful: {filename}')
+ run_id, attempt = run['id'], run['run_attempt']
+ require(type(run_id) is int and type(attempt) is int and run_id > 0 and attempt > 0,
+ 'Invalid CI run identity')
+ jobs = complete_list(api(
+ f'actions/runs/{run_id}/attempts/{attempt}/jobs?per_page=100'
+ ), 'jobs')
+ names = [job.get('name') for job in jobs]
+ require(len(names) == len(set(names)) and required.issubset(names),
+ f'Missing or duplicate full-suite jobs: {filename}')
+ require(all(job.get('status') == 'completed' and job.get('head_sha') == dev
+ and (job.get('conclusion') == 'success'
+ or (job.get('name') == 'Codegraph select'
+ and job.get('conclusion') == 'skipped')) for job in jobs),
+ f'Failed, skipped, or incomplete full-suite job: {filename}')
+ latest = api(f'actions/runs/{run_id}')
+ require(all(latest.get(key) == run.get(key) for key in (
+ 'id', 'run_attempt', 'head_sha', 'head_branch', 'event',
+ 'workflow_id', 'path', 'status', 'conclusion',
+ )), 'CI changed during verification; wait and dispatch again')
+
+ def git(args, token):
+ # No token in arguments, remote config, files, or output. Ignore global
+ # git config and hooks; only a disposable bare object database is used.
+ header = base64.b64encode(f'x-access-token:{token}'.encode()).decode()
+ print(f'::add-mask::{header}', flush=True)
+ env = dict(os.environ, GIT_CONFIG_NOSYSTEM='1', GIT_CONFIG_GLOBAL='/dev/null',
+ GIT_TERMINAL_PROMPT='0', GIT_CONFIG_COUNT='1',
+ GIT_CONFIG_KEY_0='http.https://github.com/.extraheader',
+ GIT_CONFIG_VALUE_0=f'AUTHORIZATION: basic {header}')
+ result = subprocess.run(
+ ['git', '-c', 'core.hooksPath=/dev/null', *args],
+ env=env, capture_output=True, text=True, check=False, timeout=90,
+ )
+ require(result.returncode == 0, 'Git operation refused; do not force or bypass protection')
+ return result.stdout.strip()
+
+ def verify():
+ dev, main = context()
+ approval()
+ require(ref('dev') == dev and ref('main') == main, 'Branch tips changed; dispatch again')
+ full_ci(dev)
+ directory = Path(os.environ['RUNNER_TEMP']) / 'main-promotion.git'
+ token = os.environ['GH_TOKEN']
+ git(['init', '--bare', str(directory)], token)
+ prefix = ['--git-dir', str(directory)]
+ git([*prefix, 'fetch', '--no-tags', REMOTE,
+ '+refs/heads/main:refs/heads/main', '+refs/heads/dev:refs/heads/dev'], token)
+ require(git([*prefix, 'rev-parse', 'refs/heads/main'], token) == main
+ and git([*prefix, 'rev-parse', 'refs/heads/dev'], token) == dev,
+ 'Branch tips changed during fetch; dispatch again')
+ git([*prefix, 'merge-base', '--is-ancestor', main, dev], token)
+ # The App deliberately has no Workflows permission. Updating the
+ # release control plane requires a separately reviewed manual FF.
+ changed = git([*prefix, 'diff', '--name-only', main, dev, '--',
+ '.github/workflows', '.github/scripts', '.github/CODEOWNERS'], token)
+ require(not changed, 'Workflow/control-plane changes require a maintainer-reviewed manual fast-forward')
+ require(ref('dev') == dev and ref('main') == main, 'Branch tips changed after CI verification')
+ return dev, main, prefix
+
+ def run():
+ dev, main, prefix = verify()
+ if sys.argv[1] == 'promote':
+ require(bool(os.environ.get('PROMOTION_TOKEN')), 'No promotion credential')
+ # No force flag. A concurrent non-ancestor main update is rejected
+ # by Git receive-pack; dev can never change the pinned target SHA.
+ if dev != main:
+ git([*prefix, 'push', '--porcelain', REMOTE, f'{dev}:refs/heads/main'],
+ os.environ['PROMOTION_TOKEN'])
+ require(ref('main') == dev, 'Post-push verification failed; inspect main before retrying')
+ outcome = 'Promoted' if dev != main else 'Already current'
+ else:
+ require(sys.argv[1] == 'verify', 'Invalid operation')
+ outcome = 'Verified; main has not been changed'
+ with open(os.environ['GITHUB_STEP_SUMMARY'], 'a') as summary:
+ summary.write(f'### {outcome}\n\nMain baseline: `{main}`\n\nApproved dev: `{dev}`\n')
+
+ if __name__ == '__main__':
+ try:
+ run()
+ except Refused as error:
+ print(f'::error::Promotion refused: {error}', file=sys.stderr)
+ sys.exit(1)
+ except (KeyError, TypeError, ValueError, subprocess.TimeoutExpired):
+ print('::error::Invalid or unavailable policy data; refusing promotion.', file=sys.stderr)
+ sys.exit(1)
+ PY
+ python3 -I "$RUNNER_TEMP/main-promotion.py" verify
+
+ - name: Mint a repository-scoped short-lived promotion token
+ id: app
+ uses: actions/create-github-app-token@fee1f7d63c2ff003460e3d139729b119787bc349 # v2
+ with:
+ app-id: ${{ vars.MAIN_PROMOTION_APP_ID }}
+ private-key: ${{ secrets.MAIN_PROMOTION_APP_PRIVATE_KEY }}
+ owner: LibreChat-AI
+ repositories: LibreChat
+ permission-contents: write
+
+ - name: Revalidate and fast-forward only main
+ env:
+ GH_TOKEN: ${{ github.token }}
+ PROMOTION_TOKEN: ${{ steps.app.outputs.token }}
+ run: python3 -I "$RUNNER_TEMP/main-promotion.py" promote
diff --git a/.github/workflows/promotion-tests.yml b/.github/workflows/promotion-tests.yml
new file mode 100644
index 00000000000..5c340c19ac0
--- /dev/null
+++ b/.github/workflows/promotion-tests.yml
@@ -0,0 +1,35 @@
+name: Main Promotion Policy Tests
+
+on:
+ pull_request:
+ branches: [dev]
+ paths:
+ - '.github/workflows/promote-main.yml'
+ - '.github/workflows/promotion-tests.yml'
+ - '.github/workflows/backend-review.yml'
+ - '.github/workflows/frontend-review.yml'
+ - '.github/scripts/test_main_promotion.py'
+ - '.github/CODEOWNERS'
+ push:
+ branches: [dev]
+ paths:
+ - '.github/workflows/promote-main.yml'
+ - '.github/workflows/promotion-tests.yml'
+ - '.github/workflows/backend-review.yml'
+ - '.github/workflows/frontend-review.yml'
+ - '.github/scripts/test_main_promotion.py'
+ - '.github/CODEOWNERS'
+
+permissions:
+ contents: read
+
+jobs:
+ policy:
+ name: Main promotion denial and fast-forward tests
+ runs-on: ubuntu-24.04
+ timeout-minutes: 5
+ steps:
+ - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
+ with:
+ persist-credentials: false
+ - run: python3 -I .github/scripts/test_main_promotion.py
diff --git a/AGENTS.md b/AGENTS.md
index d35d3d392d5..6a486ba134e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -1,4 +1,9 @@
-See CLAUDE.md.
+# LibreChat contributor and agent guidance
+
+`AGENTS.md` is the repository's contributor guidance. Backend code lives in `packages/api`
+(TypeScript) and `packages/data-schemas` (database methods); `api` is legacy Express wiring.
+Shared API types and services live in `packages/data-provider`, and the React app lives in
+`client` with shared primitives in `packages/client`.
## Branching and pull requests
@@ -18,8 +23,9 @@ explicitly based on `canary` stays there; the main-to-dev retarget workflow does
Still link related issues in the PR (for example, `Related to #N`) so the work remains
traceable. `Fixes #N` does not close an issue on a `dev` or `canary` merge — GitHub honors closing
keywords only on the default branch. Close resolved issues by hand after merging. Worktrees share
-one stash stack, so never use a bare `git stash pop`. See the detailed policy in `CLAUDE.md` under
-"Branching and Pull Requests".
+one stash stack, so never use a bare `git stash pop`. Prefer a WIP commit; if a stash is necessary,
+apply the specific tagged entry. The `target: main` label and release-bound upstream branches are
+exempt from the main-to-dev retarget workflow; do not use either to backport ordinary work.
Write the description for a reader who has not followed the branch: what breaks, what triggers it,
how it behaves after the change, then one or two views of the mechanism — a focused diff, a call
@@ -34,17 +40,18 @@ Read the inline review threads themselves — a summary comment or notification
Audit each one against the current code, fix what is valid, and reject what is obsolete in a reply
that says why. After each round: focused tests, `npx tsc --noEmit` in every workspace you changed,
push, then request the next review naming the pull request's exact remote head — a clean review of
-an earlier head says nothing about what you just pushed, and CI runs on its own clock. After two
-actionable rounds, stop patching thread by thread and read the subsystem by invariant instead.
-Which reviewer and what phrase triggers it will change; that the review must cover the exact pushed
-head will not.
+an earlier head says nothing about what you just pushed, and CI runs on its own clock. Reply on
+each resolved thread with the resolving commit and evidence. After two actionable rounds, stop
+patching thread by thread and read the subsystem by invariant instead: identity, authorization,
+persistence, retry, cancellation, cleanup and mixed-version behavior where relevant. Which reviewer
+and what phrase triggers it will change; that the review must cover the exact pushed head will not.
A clean review is one completion signal, not the definition of done. Ship the observable experience
— loading, empty, success, failure, cancellation, retry, restored session — with strings localized,
accessibility intact, defaults and stored data preserved, and no backend capability left without a
-frontend entry point. Report the pushed head, what you ran locally, CI state, the review result at
-that head, and any finding you rejected with the reasoning. See `CLAUDE.md` under "Review and
-Completion".
+frontend entry point. Keep fixes small and test the missed behavior. Report the pushed head, what you
+ran locally, CI state, the review result at that head, any finding you rejected with the reasoning,
+and checks you could not run.
## Verification
@@ -56,11 +63,17 @@ See [budgets, reproduction and failure diagnosis](e2e/lighthouse/README.md).
A green build is not a typecheck: `packages/api`, `packages/client` and `packages/data-schemas` build
with `tsdown`, which emits without checking types. Run `npx tsc --noEmit` in the workspace you
changed. `packages/client` excludes `*.spec.ts(x)` and `*.test.ts(x)` from typechecking entirely.
-`npm run sort-imports` with no arguments rewrites every source root — pass the paths you touched. See
-`CLAUDE.md` under "Typechecking" and "Formatting".
+`npm run sort-imports` with no arguments rewrites every source root — pass the paths you touched.
+Run `npm run static-checks -- --against origin/dev` to reproduce the PR's static checks; use
+`npm run static-checks:full` for the slower gates. Fix all formatting, lint, and TypeScript
+warnings/errors in the code you change.
## Module boundaries and configuration
+All new backend behavior belongs in TypeScript under `packages/api`; database-specific shared logic
+belongs in `packages/data-schemas`, and frontend/backend shared API logic belongs in
+`packages/data-provider`. Build data-provider from the project root with `npm run build:data-provider`.
+
`/api` holds wiring, not behavior. When a change would add logic to a CJS file there — a branch, a
helper, a validation step, a service call — the logic belongs in `packages/api`, and the JS file
keeps requires, route registration and the call into the TS module (`MCPRequestContext.js` is the
@@ -85,7 +98,46 @@ supplies, so a second implementation is a new argument instead of a new branch.
under `packages/api/src/mcp` are the shape to stop extending, not a pattern to copy. This is the
backend half of client state ownership: pass it in, do not reach for it.
-See `CLAUDE.md` under "Workspace Boundaries".
+## Service failures and user-facing errors
+
+This section applies to backend services (`packages/api`) and database methods/repositories
+(`packages/data-schemas`), not to React components. For new or substantially changed code in
+those layers, choose failure behavior at the contract boundary instead of treating every
+unsuccessful outcome as an exception:
+
+- Return a plain value on success. Use `null` only for a documented absence (for example, a
+ successful lookup with no record), never to hide a query failure.
+- For an expected failure the caller can handle without aborting the operation, prefer a typed
+ discriminated result such as `{ ok: true; value: T } | { ok: false; error: { code: string;
+message?: string } }`. Keep an existing domain-specific result shape when changing it would
+ break callers; do not introduce interchangeable `ok`, `valid`, and bare `{ message }` contracts
+ in the same service. Codes should be stable, machine-readable identifiers when the caller
+ needs to distinguish failures. Internal validation helpers may use the existing local pattern.
+- Throw for unexpected operational failures (database, provider, network) and violated invariants.
+ Use a typed/coded error when a boundary must recognize a particular failure. Catch where you can
+ recover or translate it; otherwise let it reach the owning request or job boundary. Do not catch
+ an exception and return a truthy record, `null`, or `{ message }` in its place: callers can then
+ mistake an outage for data or send it as an HTTP 200. Keep separate migrations for existing
+ contracts, with their callers and tests, rather than changing wire shapes incidentally.
+- At the HTTP, stream, or job boundary, map failures to the existing status and response contract.
+ Only expose intentionally approved codes and helpful user-safe information. Error disclosure is
+ a security boundary: never forward raw exceptions, stacks, query text, credentials, request
+ bodies, headers, provider payloads, or arbitrary `error.message` to a client or persisted
+ user-visible status. Sanitize diagnostics and log only what the path can safely retain. For
+ errors that may echo submitted content, prefer `getSafeErrorMetadata` over error text;
+ `packages/api/src/utils/errors.ts` documents when redacted error text is appropriate.
+- Localization is a UI responsibility, not a database-method return shape. When a failure reaches
+ a user-facing surface, expose a safe, stable code the UI can map to localized, actionable copy
+ and a fallback. See `client/src/components/Messages/Content/Error/registry.ts` and
+ `client/src/components/Chat/Trace/Viewer.tsx`. Keep any intentionally disclosed provider detail
+ behind its existing protection policy rather than treating arbitrary upstream text as safe.
+
+Examples of local contracts, not a mandate to copy their shapes: `packages/api/src/admin/auditLog.ts`
+uses a parsing result, `packages/api/src/agents/openai/service.ts` validates before mapping to an
+OpenAI-compatible response, and `packages/api/src/skills/sync/errors.ts` names a sync failure by
+stable code. `packages/data-schemas/src/methods/prompt.ts` currently returns `{ message }` on query
+failure; do not extend that pattern. Test absence versus failure, the boundary's non-success response,
+and that sensitive diagnostics cannot reach a user-visible surface.
## Frontend theming and styling
@@ -93,22 +145,16 @@ For frontend work, compose existing `@librechat/client` primitives and variants
feature-local styles. Use semantic theme/Tailwind roles for color and shared appearance; do not
introduce raw palette utilities, hard-coded colors, or arbitrary theme CSS. If the system cannot
express a reusable design need, deepen the shared primitive or versioned theme-token registry
-instead of copying classes into a feature. Keep genuine layout and behavior local, and document
-why any new custom CSS cannot be expressed by the shared system.
-
-`npm run lint` enforces this: `@shadcn/lint` reads each primitive's `cva` variants and reports a
-`className` that overrides what the primitive owns, naming the variant, size or file to use
-instead. Do not reach for `eslint-disable`, and do not widen a file's recorded count in
-`eslint-suppressions.json` to land a restyle — that file holds the backlog the rules inherited, so
-raising an entry is a reviewable claim that the override is right. After fixing violations, run
-`npm run lint:design:prune`; after moving a file that carries suppressions, run
-`npm run lint:design:suppress`. See the detailed policy in `CLAUDE.md` under “Theming and styling.”
+instead of copying classes into a feature. Theme definitions select semantic colors and appearance,
+not selectors, app behavior, or alternate layouts. Keep genuine layout and behavior local, and
+explain why any new custom CSS cannot use shared primitives. Support light/dark and reduced motion;
+preserve defaults and test a deliberately different reference theme for reusable new variants.
## Backend auth cache
When adding or changing code that mutates user documents, invalidate the auth user document cache
-for affected users, including bulk role and user mutations. See the detailed policy in `CLAUDE.md`
-under “Auth cache invalidation”.
+for affected users, including bulk role and user mutations. Otherwise burst-cached JWT requests can
+serve a stale `req.user` until the cache expires.
## Client state ownership
@@ -122,5 +168,44 @@ and keep it inside the feature; app-global preferences and shell state a feature
through props or a small host-supplied context rather than reached for through `~/store`; when a
consumer sits outside the feature you are changing, leave that atom on Recoil and pass it in. Passing
them in is what lets a feature move to its own workspace later without a rewrite, and it keeps the
-Jotai conversion scoped to the state a feature owns. See the detailed policy in `CLAUDE.md` under
-“Client State Ownership”.
+Jotai conversion scoped to the state a feature owns. For persisted atoms, use the helpers in
+`client/src/store/jotai-utils.ts`.
+
+## Code style and performance
+
+Use short, single-word file names when possible; group related modules in a single-word directory.
+Prefer flat early returns, pure functions, and explicit types. Avoid `any`, broad `unknown`/casts,
+duplicated types, and dynamic imports. Extract genuinely repeated logic instead of copying it;
+write comments only for non-obvious behavior or public API contracts.
+
+Imports have three sections: package values (shortest line first, with `react` first), `import type`
+(longest line first, package types before local types), and local/project values (longest line first).
+Use standalone `import type { ... }`, not inline `type` within a value import. Run the scoped
+import sorter on files you change.
+
+Avoid extra passes over shared message arrays and unnecessary allocations. For startup and
+request paths, reuse already-loaded user/config data, avoid serial database reads, and start
+independent reads in parallel. Do not weaken authorization or tenant checks or write a response
+before those checks succeed.
+
+## Frontend rules
+
+Use the chat Share/Export action-menu pattern: `DropdownPopup` from `@librechat/client` with
+`Ariakit.MenuButton` (see `HeaderMenu.tsx` and `useExportShare.tsx`). Never introduce or reintroduce
+the Radix `DropdownMenu` family for app action or sort menus. Preserve the Share/Export
+dialog-item contract (`hideOnClick: false`, item ref, button render, and dialog `triggerRef`)
+when a menu action opens a dialog.
+
+Use `useLocalize()` for all visible copy and update only English keys in
+`client/src/locales/en/translation.json`. Use semantic HTML, keyboard behavior, and ARIA labels.
+Use React Query for API interactions and invalidate related queries after mutations; define query
+and mutation keys in `packages/data-provider/src/keys.ts`. Put shared endpoint definitions, types,
+and data services in `packages/data-provider`, and encode dynamic URL parameters. Cover loading,
+success, and failure states in focused component tests.
+
+## Testing
+
+Run focused Jest tests from the owning workspace, not the entire monorepo. Prefer real logic and
+spies over mocked internals; use `mongodb-memory-server` for database queries and the real MCP SDK
+for MCP behavior. Mock only external HTTP APIs or uncontrollable services. For frontend tests use
+the component's colocated `__tests__` and `test/layout-test-utils` where applicable.
diff --git a/CLAUDE.md b/CLAUDE.md
deleted file mode 100644
index 226b6f72506..00000000000
--- a/CLAUDE.md
+++ /dev/null
@@ -1,474 +0,0 @@
-# LibreChat
-
-## Project Overview
-
-LibreChat is a monorepo with the following key workspaces:
-
-| Workspace | Language | Side | Dependency | Purpose |
-|---|---|---|---|---|
-| `/api` | JS (legacy) | Backend | `packages/api`, `packages/data-schemas`, `packages/data-provider`, `@librechat/agents` | Express server — minimize changes here |
-| `/packages/api` | **TypeScript** | Backend | `packages/data-schemas`, `packages/data-provider` | New backend code lives here (TS only, consumed by `/api`) |
-| `/packages/data-schemas` | TypeScript | Backend | `packages/data-provider` | Database models/schemas, shareable across backend projects |
-| `/packages/data-provider` | TypeScript | Shared | — | Shared API types, endpoints, data-service — used by both frontend and backend |
-| `/client` | TypeScript/React | Frontend | `packages/data-provider`, `packages/client` | Frontend SPA |
-| `/packages/client` | TypeScript | Frontend | `packages/data-provider` | Shared frontend utilities |
-
-The source code for `@librechat/agents` (major backend dependency, same team) lives at
-.
-
----
-
-## Workspace Boundaries
-
-- **All new backend code must be TypeScript** in `/packages/api`.
-- **`/api` holds wiring, not behavior.** When a change would add logic to a CJS file under `/api` —
- a branch, a helper, a validation step, a new service call — that logic goes in `/packages/api`,
- and the JS file keeps only what wires it up: requires, route registration, request plumbing, and
- the call into the TS module. `api/server/services/MCPRequestContext.js` is the shape, at thirteen
- lines of re-export. "Minimum" describes how much behavior `/api` gains, not how small the diff is:
- lifting a function into `/packages/api` and calling it is the larger diff and the correct one.
- Editing an existing CJS file is the common case and the rule applies there, not only to new files.
-- Database-specific shared logic goes in `/packages/data-schemas`.
-- Frontend/backend shared API logic (endpoints, types, data-service) goes in `/packages/data-provider`.
-- Build data-provider from project root: `npm run build:data-provider`.
-- **Database contracts stay inside `/packages/data-schemas`.** A Mongoose type in an exported
- signature — `FilterQuery`, `Types.ObjectId`, `Document`, `HydratedDocument` — makes the storage
- engine part of that module's public API, and every consumer then depends on Mongo instead of on
- the data it needs. Take and return plain typed objects, and express the query behind a
- data-schemas method. The boundary already leaks across dozens of files in `/packages/api`, so the
- rule is to stop widening it rather than to rewrite what exists; `/client` carries none of it and
- must stay that way.
-- **New levers ship configurable.** A limit, timeout, toggle or capability introduced in code earns
- a field on `configSchema` (`packages/data-provider/src/config.ts`) so an operator can set it in
- `librechat.yaml`, with a default that reproduces today's behavior. Hard-coded constants and
- env-only switches need a reason. The schema is also what keeps one definition of the value instead
- of a constant, a fallback and a doc line that drift apart.
-- **A backend module takes its dependencies, it does not reach for them.** Code in `/packages/api`
- should receive its config, database methods and clients from the caller the way
- `createModels(mongoose)` receives the app's connection, rather than importing app singletons or
- reading global state. A module the caller constructs can be tested without a running app and moved
- to another workspace without a rewrite; one that calls `getInstance()` can do neither. This is the
- backend half of "Client State Ownership" — pass it in, do not reach for it. The static singletons
- under `packages/api/src/mcp` are the shape to stop extending, not a pattern to copy.
-- **Integrations arrive through an interface the caller supplies.** A provider SDK, storage backend,
- vector store or OAuth server is injected, so a second implementation is a new argument instead of
- a new branch in shared code, and a test can exercise the real logic against a substitute at the
- boundary rather than mocking the module that holds it.
-
----
-
-## Branching and Pull Requests
-
-- **Normally branch off `dev` and target `dev`.** This is the default for work ready to follow the
- regular release path. A maintainer-directed canary exception is described below.
-- **`main` is the released branch.** It is kept as a fast-forward of `dev` and synced as-is, so it
- is always an ancestor of `dev` — equal to it right after a sync, behind it otherwise. It never
- carries a commit that `dev` does not have.
-- **Never open a backport pull request to `main`.** Anything merged to `dev` reaches `main` at the
- next sync; a second pull request for the same change is redundant.
-- **The repository's default branch is `main`**, so `gh pr create` and the GitHub UI target it
- unless told otherwise — pass `--base dev` for normal work or `--base canary` for an explicitly
- maintainer-directed canary PR.
-- Pull requests opened against `main` are retargeted to `dev` automatically by
- `.github/workflows/pr-retarget-dev.yml`. The `target: main` label exempts one, as do release-bound
- upstream branches (`dev`, `release/*`, `hotfix/*`). PRs deliberately based on `canary` are not
- retargeted, including manual sweeps; the script skips them explicitly. Backport branches are
- deliberately not exempt — a backport merged straight to `main` breaks the fast-forward invariant.
-- **Canary is an explicit, maintainer-selected integration target, not a path that automatically
- flows to `dev` or `main`.** Experimental changes and PRs that are ready to merge following review
- but must not enter the next `main` sync can go to `canary`. The maintainer may decide this when
- assigning the work or while reviewing an older/stale PR. For a new canary PR, branch from the
- current `origin/canary` and set `--base canary`. Never infer that a canary-targeted PR should move
- to `dev` just because `dev` is the default; do not copy canary features into `dev` without an
- explicit maintainer decision. If a stale PR is reassigned to canary, check the PR's base, diff and
- exact reviewed head against current canary before changing its base or proposing a new canary
- branch/PR; if a rebase changes the head, verify and review that new head before merging.
-- **Link related issues in the PR even when they will not auto-close.** Reference each relevant
- issue in the description (for example, `Related to #N`) so reviewers can find the context and
- track the work. GitHub honors `Fixes #N` and other closing keywords only when a pull request
- merges into the default branch (`main`). Merging to `dev` or `canary` does not close an issue,
- and the later fast-forward of `main` is not a merge event either — close resolved issues by hand
- after merging.
-- **Git worktrees share one stash stack.** `refs/stash` lives in the common `.git` directory, so a
- bare `git stash pop` in one worktree can take work stashed in another. Prefer a throwaway WIP
- commit; if you must stash, `git stash push -m ` and `apply` that specific entry.
-- **Write the description for a reader who has not followed the branch.** Say what breaks, what
- triggers it, and how it behaves after the change, then show the mechanism with whichever one or
- two views make it reviewable — a focused diff, a call tree, a shallow file tree, or a Mermaid
- sequence — keeping only the calls, files and state the change actually carries. Describe the code
- as it stands: do not narrate what earlier commits tried or what a review round changed. Naming the
- merged pull request that caused the bug is different — that is history the reader needs.
- `.github/pull_request_template.md` carries the formats and examples.
-
----
-
-## Review and Completion
-
-### AI review cycles
-
-The reviewer, its trigger phrase and its cadence all change; this subsection is the fluid one, so
-rewrite it when they do. What survives a change of tool: a review counts only for the exact commit
-it ran on, its findings are judged against the code rather than accepted or dismissed wholesale, and
-findings that keep arriving mean the subsystem needs a sweep, not another patch.
-
-- **Inline review threads are the source of truth.** A summary comment, a check name or a
- notification list omits findings — read the threads on the pull request itself.
-- Audit every finding against the current code. Fix the valid ones; reject the obsolete or wrong
- ones in a reply that says why.
-- After each round of fixes, run the focused tests and `npx tsc --noEmit` for every workspace you
- changed, push, read the pull request's remote head (`gh pr view --json headRefOid`), and
- request the next review naming that exact SHA. **A clean review of an earlier head says nothing
- about what you just pushed.** Do not wait for CI before asking — review and CI run on their own
- clocks.
-- Reply on each thread you resolved with the commit that resolved it and the coverage that proves
- it.
-- **After two actionable rounds** — or sooner, when each fix uncovers an adjacent defect — stop
- answering threads one at a time and read the subsystem by invariant: identity, ownership,
- authorization, persistence, retry, replay, abort, cleanup, expiry, rollout. Follow producers,
- consumers, adapters, alternate write paths, and the final consumer of every limit; check
- mixed-version behavior in both directions; read the whole base-to-head diff with the callers and
- tests around it; then add transition or failure-injection coverage at the deepest boundary that
- owns the behavior.
-- The cycle ends when the exact pushed head draws no major findings, or only repeats ones already
- resolved. A clean review is one completion signal, not the definition of done.
-
-### Definition of done
-
-- **Ship the observable experience, not the reported path.** Where they apply, cover loading, empty,
- success, failure, cancellation, retry and restored-session behavior.
-- **A backend capability with no frontend entry point is unfinished**, and so is a control with no
- validation, persistence, error handling or authorization behind it.
-- Localize every visible string through `useLocalize()`, keep semantic HTML, keyboard behavior and
- ARIA intact, and compose shared primitives and semantic theme roles before adding local styling
- (see "Frontend Rules"). Custom styling that proves unavoidable still supports light/dark and
- reduced motion.
-- Preserve existing defaults, configuration compatibility, stored data, and mixed-version behavior,
- and expose any new lever through `configSchema` rather than a constant (see "Workspace
- Boundaries").
-- Make the fix the smallest one consistent with the patterns already in the file, and test the
- behavior that was missed rather than the line a reviewer pointed at.
-- **Report what you actually ran**: the pushed head, the local checks from "Testing" and
- "Typechecking", CI state, the review result at that head, and any finding you rejected with the
- reasoning. Name the checks you could not run instead of implying coverage.
-
----
-
-## Code Style
-
-### Naming and File Organization
-
-- **Single-word file names** whenever possible (e.g., `permissions.ts`, `capabilities.ts`, `service.ts`).
-- When multiple words are needed, prefer grouping related modules under a **single-word directory** rather than using multi-word file names (e.g., `admin/capabilities.ts` not `adminCapabilities.ts`).
-- The directory already provides context — `app/service.ts` not `app/appConfigService.ts`.
-
-### Structure and Clarity
-
-- **Never-nesting**: early returns, flat code, minimal indentation. Break complex operations into well-named helpers.
-- **Functional first**: pure functions, immutable data, `map`/`filter`/`reduce` over imperative loops. Only reach for OOP when it clearly improves domain modeling or state encapsulation.
-- **No dynamic imports** unless absolutely necessary.
-
-### DRY
-
-- Extract repeated logic into utility functions.
-- Reusable hooks / higher-order components for UI patterns.
-- Parameterized helpers instead of near-duplicate functions.
-- Constants for repeated values; configuration objects over duplicated init code.
-- Shared validators, centralized error handling, single source of truth for business rules.
-- Shared typing system with interfaces/types extending common base definitions.
-- Abstraction layers for external API interactions.
-
-### Iteration and Performance
-
-- **Minimize looping** — especially over shared data structures like message arrays, which are iterated frequently throughout the codebase. Every additional pass adds up at scale.
-- Consolidate sequential O(n) operations into a single pass whenever possible; never loop over the same collection twice if the work can be combined.
-- Choose data structures that reduce the need to iterate (e.g., `Map`/`Set` for lookups instead of `Array.find`/`Array.includes`).
-- Avoid unnecessary object creation; consider space-time tradeoffs.
-- Prevent memory leaks: careful with closures, dispose resources/event listeners, no circular references.
-
-### Backend Database Performance
-
-- On request startup and first page load paths, watch for serial database reads.
- Multiple round trips to MongoDB can add significant latency when the database
- is far from the app server.
-- Prefer passing already-loaded request/user/config data through helper
- functions instead of re-reading the same user, role, tenant, or principal data.
-- When two reads are independent, start them in parallel and gate the response
- on the authorization or validation result before returning data.
-- Keep authorization, permission, and tenant checks semantically identical when
- parallelizing reads. Speculative reads must remain scoped to the authenticated
- user or tenant and must not write to the response before validation succeeds.
-
-### Type Safety
-
-- **Never use `any`**. Explicit types for all parameters, return values, and variables.
-- **Limit `unknown`** — avoid `unknown`, `Record`, and `as unknown as T` assertions. A `Record` almost always signals a missing explicit type definition.
-- **Don't duplicate types** — before defining a new type, check whether it already exists in the project (especially `packages/data-provider`). Reuse and extend existing types rather than creating redundant definitions.
-- Use union types, generics, and interfaces appropriately.
-- All TypeScript and ESLint warnings/errors must be addressed — do not leave unresolved diagnostics.
-
-### Comments and Documentation
-
-- Write self-documenting code; no inline comments narrating what code does.
-- JSDoc only for complex/non-obvious logic or intellisense on public APIs.
-- Single-line JSDoc for brief docs, multi-line for complex cases.
-- Avoid standalone `//` comments unless absolutely necessary.
-
-### Import Order
-
-Imports are organized into three sections:
-
-1. **Package imports** — sorted shortest to longest line length (`react` always first).
-2. **`import type` imports** — sorted longest to shortest (package types first, then local types; length resets between sub-groups).
-3. **Local/project imports** — sorted longest to shortest.
-
-Multi-line imports count total character length across all lines. Consolidate value imports from the same module. Always use standalone `import type { ... }` — never inline `type` inside value imports.
-
-### JS/TS Loop Preferences
-
-- **Limit looping as much as possible.** Prefer single-pass transformations and avoid re-iterating the same data.
-- `for (let i = 0; ...)` for performance-critical or index-dependent operations.
-- `for...of` for simple array iteration.
-- `for...in` only for object property enumeration.
-
----
-
-## Frontend Rules (`client/src/**/*`)
-
-### Localization
-
-- All user-facing text must use `useLocalize()`.
-- Only update English keys in `client/src/locales/en/translation.json` (other languages are automated externally).
-- Semantic key prefixes: `com_ui_`, `com_assistants_`, etc.
-
-### Components
-
-- TypeScript for all React components with proper type imports.
-- Semantic HTML with ARIA labels (`role`, `aria-label`) for accessibility.
-- Group related components in feature directories (e.g., `SidePanel/Memories/`).
-- Use index files for clean exports.
-- Use the chat Share/Export action-menu pattern: `DropdownPopup` from `@librechat/client`
- with `Ariakit.MenuButton` (see `HeaderMenu.tsx` and `useExportShare.tsx`). Never introduce
- or reintroduce the Radix `DropdownMenu` family for app action or sort menus. Preserve
- the Share/Export dialog-item contract (`hideOnClick: false`, item ref, button render,
- and dialog `triggerRef`) when a menu action opens a dialog.
-
-### Theming and styling
-
-- **Compose before styling.** Search `@librechat/client` for an existing primitive, semantic
- variant, or composition before adding feature-local classes or CSS.
-- **Use semantic roles.** Colors and shared appearance values must come from the semantic
- Tailwind/theme roles. Do not add raw palette utilities, hard-coded hex/RGB/HSL colors, or
- light/dark-specific values in feature components.
-- **Deepen the system when the need is reusable.** Add a focused variant to a shared primitive or
- extend the canonical, versioned theme-token registry when multiple screens should share the
- same design decision. Do not create shallow local wrappers that merely relocate class strings.
-- **Themes are data, not arbitrary CSS.** Theme definitions may select semantic colors and shared
- appearance roles. They must not contain selectors, arbitrary CSS, application behavior, or
- alternate feature layouts. Preserve existing environment and stored-theme compatibility when
- changing the theme engine.
-- **Keep layout and behavior local.** Feature structure, responsive layout, state-driven
- transitions, and specialized visualization may remain feature-owned. Expose a theme role only
- when it represents a stable, reusable appearance decision; do not turn every measurement into a
- global token.
-- **Treat custom CSS as an exception.** Use it only when shared primitives and semantic utilities
- cannot express the requirement. Keep it narrowly scoped, consume theme variables where
- applicable, support light/dark and reduced motion, and add a brief code or PR explanation of why
- the exception is necessary.
-- **Preserve defaults and prove variability.** New theme-aware variants must reproduce the current
- default appearance unless a redesign is explicitly requested. Test semantic-token use and, when
- extending theme capabilities, include a deliberately different reference theme to prove that
- components adapt without feature-specific overrides.
-- **The design rules are enforced by `npm run lint`.** `@shadcn/lint` is registered in
- `eslint.config.mjs` for `client/src/**` and `packages/client/src/**`. It resolves
- `@librechat/client` (and the `~/components/ui` re-exports) as the design system, reads each
- primitive's `cva` variants, and reports a `className` that overrides what the primitive owns —
- naming the variants, sizes and defining file to use instead. Six rules run at `error`:
- `no-restyle` (layout and `icon-*` sizing allowed), `no-raw-colors`, `no-arbitrary-values`
- (layout allowed), `no-inline-styles` (measured geometry allowed: width, height, inset,
- transform, z-index), `require-static-classes` and `no-unknown-classes`. The last one asks the
- installed Tailwind whether a class generates any CSS, which needs v4 and the color tokens in
- CSS (`packages/client/src/theme/tokens.css`) for the rule to resolve a theme; under 3.4 its
- fallback reported every preset utility (`duration-theme-fast`, `rounded-theme-control`) as a
- typo. What it cannot see is a class defined in a stylesheet Tailwind does not read, or one a
- dependency puts in the DOM, so those live in its `allow` list in `eslint.config.mjs`, add a
- name there when a plain selector or a third party owns it, and fix the class when nothing
- defines it. `no-raw-colors` additionally reports a `bg-`/`text-`/`border-` name whose token is
- not declared in the theme, which is the same failure wearing a color's clothes.
-- **Contracts say what a caller owns; suppressions say what is owed.** A `no-restyle` contract in
- `eslint.config.mjs` opens a category for one component because the caller legitimately owns it:
- `Label`, `Description`, `DialogTitle` and `DialogDescription` take the caller's typography, and
- `Skeleton` takes the silhouette and footprint of the content it stands in for. Color stays with
- the theme in every case. Widen a contract when a whole category belongs to callers; do not widen
- one to clear a single call site.
-- **The existing backlog is recorded, not exempted.** `eslint-suppressions.json` holds the
- violations the tree carried when the rules landed, as a per-file, per-rule count.
- Adding a violation to a file reports every violation of that rule in it, so raising a file's
- count is a visible diff in that file, review it like any other change, and expect a raise to
- be justified by what the diff does. Strengthening a rule is the one case where counts rise in
- files nobody edited: when a rule starts seeing a class it could not classify before, the
- violations it reports are the tree's existing styling, and recording them is the same backlog
- arriving later. Say which change did that and keep the two apart in review.
- `npm run lint:design:prune` drops entries whose violations are gone (do this after fixing
- some); `npm run lint:design:suppress` re-records the design rules and then prunes, which is
- what a file move needs, since suppressions are keyed by path, the re-record adds the new
- path and only the prune removes the old one. It re-baselines everything under `client/src`
- and `packages/client/src`, so for a single move prefer a scoped re-record of just that file:
- the same six `--suppress-rule` flags `lint:design:suppress` passes (`shadcn/no-restyle`,
- `no-raw-colors`, `no-arbitrary-values`, `no-inline-styles`, `require-static-classes`,
- `no-unknown-classes`) with
- `` in place of the directories, since a moved file usually carries entries for
- more than one rule, followed by `npm run lint:design:prune`;
- `npm run lint:design:suppress -- ` does not scope, because npm appends the argument to
- the script's own directory arguments. Neither command changes which rules run; both only
- rewrite the recorded counts. The re-record's own exit status is deliberately ignored: it lints
- the roots under every rule, those roots carry a pre-existing error backlog, and the baseline is
- written before ESLint reports it — so `lint:design:suppress` runs `lint:design:record` inside
- `|| node -e ""` (`true` is not a command under `cmd.exe`, where npm runs scripts on Windows)
- and lets the prune be the step that can fail. Expect the same when running a scoped re-record
- by hand, and build the primitives first: `shadcn/no-restyle` reads each variant out of
- `packages/client/dist`, so a missing or stale bundle records counts that describe a library
- nobody is compiling against. `lint`, `lint:fix`, `lint:design:record` and `lint:design:prune`
- each run `npm run build:client-package` before they lint, which is why `lint:design:suppress`
- no longer builds on its own account; a bare `eslint --suppress-rule ...` does not, so run that
- build ahead of it.
-- **The ratchet is on the count, and on what the count stands for.** ESLint compares a file's
- current violation count for a rule against the recorded one and suppresses when it is not higher,
- so replacing one suppressed violation with a different violation of the same rule in the same
- file keeps the count equal and ESLint reports nothing. The backlog is a budget per file, but it is
- not a budget you may respend: for every file a diff touches, the runner lints that file's version
- at the merge base and reports a design-rule message the head has that the base did not, beyond
- whatever that entry's count grew by. So a swap arrives as a failure naming the new violation, a
- violation that merely moved does not, adding one with the count raised to cover it is the
- documented `npm run lint:design:suppress` path and stays green — the raised count is the line a
- reviewer reads — and a file the change adds brings no allowance with it, while a renamed one
- keeps what its violations came with. Two kinds of swap still pass: one whose diagnostic message
- is the same as the message it replaced, because some rules name the property and not the value
- (`Inline style sets display.` is one message for `none` and for `flex`), and one caused by the
- diff's own change to a primitive or to the config, because both versions are linted under the
- head's design inputs. `npm run static-checks` and the Static Checks lane also
- validate the baseline itself (shape, positive counts, rules the plugin defines, paths that still
- exist, paths a design-rule lint actually reports on, and counts that match the file's violations).
-- **The design rules read JSX, not CSS.** They are AST rules over `className`, `cva` and
- `style` in `{ts,tsx,js,jsx}`, so a `.css` file under either root — `client/src/style.css`,
- `packages/client/src/components/Field.css` — is outside every one of them, and so is the
- changed-file runner's selection. A raw hex in a stylesheet is caught by review, not by
- `npm run lint`. The theme tokens are the exception: they live in CSS and carry their own
- check.
-
-### Data Management
-
-- Feature hooks: `client/src/data-provider/[Feature]/queries.ts` → `[Feature]/index.ts` → `client/src/data-provider/index.ts`.
-- React Query (`@tanstack/react-query`) for all API interactions; proper query invalidation on mutations.
-- QueryKeys and MutationKeys in `packages/data-provider/src/keys.ts`.
-
-### Client State Ownership
-
-The client is migrating from Recoil to Jotai. **New state is always Jotai**, including inside a file
-that already imports Recoil. For existing state, the unit of conversion is one atom together with
-every file that reads or writes it: the two libraries hold different atom objects, so an atom cannot
-be half converted, and many files already import both — mixed imports are not a signal that either
-choice is fine here. Convert the areas you touch rather than migrating wholesale, and split the work
-by who owns the state:
-
-- **Feature-owned state** — atoms a single feature both writes and reads. Convert these to
- Jotai as you touch them, with all of their consumers, and keep them inside the feature.
- `client/src/store/jotai-utils.ts` carries the equivalents for persisted atoms
- (`createStorageAtom`, `createStorageAtomWithEffect`, `createTabIsolatedAtom`), so a Recoil atom
- with a localStorage effect has a direct port.
-- **App-global state** — preferences and shell state a feature merely consumes
- (`maximizeChatSpace`, `showScrollButton`, `enterToSend`, artifact visibility). A feature
- that could plausibly be extracted must not reach into `~/store` for these; accept them
- through props or a small context the host supplies. When a consumer sits outside the feature you
- are changing, leave the atom on Recoil and pass it in — do not convert the shell to make one
- feature tidy.
-
-Passing app-global state in — rather than reaching for it — is what lets a feature move to
-its own workspace later without a rewrite, and it keeps the Jotai conversion scoped to the
-state a feature actually owns instead of dragging the global migration forward early.
-
-### Data-Provider Integration
-
-- Endpoints: `packages/data-provider/src/api-endpoints.ts`
-- Data service: `packages/data-provider/src/data-service.ts`
-- Types: `packages/data-provider/src/types/queries.ts`
-- Use `encodeURIComponent` for dynamic URL parameters.
-
-### Performance
-
-- Prioritize memory and speed efficiency at scale.
-- Cursor pagination for large datasets.
-- Proper dependency arrays to avoid unnecessary re-renders.
-- Leverage React Query caching and background refetching.
-
----
-
-## Backend Rules (`api/**`, `packages/api/**`)
-
-### Auth cache invalidation
-
-When adding or changing code that mutates user documents, invalidate the auth user document cache
-for the affected users. This covers single-user updates as well as bulk role and user mutations.
-Without it, OpenID JWT request burst caching can serve a stale `req.user` until its TTL expires.
-
----
-
-## Development Commands
-
-| Command | Purpose |
-|---|---|
-| `npm run smart-reinstall` | Install deps (if lockfile changed) + build via Turborepo |
-| `npm run reinstall` | Clean install — wipe `node_modules` and reinstall from scratch |
-| `npm run backend` | Start the backend server |
-| `npm run backend:dev` | Start backend with file watching (development) |
-| `npm run build` | Build all compiled code via Turborepo (parallel, cached) |
-| `npm run frontend` | Build all compiled code sequentially (legacy fallback) |
-| `npm run frontend:dev` | Start frontend dev server with HMR (port 3090, requires backend running) |
-| `npm run build:data-provider` | Rebuild `packages/data-provider` after changes |
-| `npm run lint:design:prune` | Drop `eslint-suppressions.json` entries whose violations are fixed |
-| `npm run lint:design:suppress` | Re-record the design-rule backlog, then prune (needed after a file move) |
-
-- Node.js: v24.16.0
-- Database: MongoDB
-- Backend runs on `http://localhost:3080/`; frontend dev server on `http://localhost:3090/`
-
----
-
-## Testing
-
-- Framework: **Jest**, run per-workspace.
-- Run tests from their workspace directory: `cd api && npx jest `, `cd packages/api && npx jest `, etc.
-- Frontend tests: `__tests__` directories alongside components; use `test/layout-test-utils` for rendering.
-- Cover loading, success, and error states for UI/data flows.
-
-### Typechecking
-
-- **A green build is not a typecheck.** `packages/api`, `packages/client` and `packages/data-schemas`
- build with `tsdown` alone, which emits without checking types. Only `packages/data-provider` runs
- `tsc` as part of its build.
-- Run `npx tsc --noEmit` in the workspace you changed before calling it done. `client` also exposes
- it as `npm run typecheck`.
-- `packages/client/tsconfig.json` excludes `*.spec.ts(x)` and `*.test.ts(x)`, so test files there are
- never typechecked — a type error in a spec surfaces only when the test runs.
-- `npm run static-checks` runs the Static Checks CI job locally against your staged files;
- `npm run static-checks -- --against origin/dev` reproduces what CI sees for a pull request, and
- `npm run static-checks:full` adds the slow gates (TypeScript, config migration tests, unused i18n
- keys, unused npm packages).
-
-### Philosophy
-
-- **Real logic over mocks.** Exercise actual code paths with real dependencies. Mocking is a last resort.
-- **Spies over mocks.** Assert that real functions are called with expected arguments and frequency without replacing underlying logic.
-- **MongoDB**: use `mongodb-memory-server` for a real in-memory MongoDB instance. Test actual queries and schema validation, not mocked DB calls.
-- **MCP**: use real `@modelcontextprotocol/sdk` exports for servers, transports, and tool definitions. Mirror real scenarios, don't stub SDK internals.
-- Only mock what you cannot control: external HTTP APIs, rate-limited services, non-deterministic system calls.
-- Heavy mocking is a code smell, not a testing strategy.
-
----
-
-## Formatting
-
-Fix all formatting lint errors (trailing spaces, tabs, newlines, indentation) using auto-fix when available. All TypeScript/ESLint warnings and errors **must** be resolved.
-
-`npm run sort-imports` with no arguments rewrites every file under `api/`, `client/src` and the four
-`packages/*/src` roots — far beyond what you touched. Always pass explicit paths:
-`npm run sort-imports -- path/to/file.ts`.
diff --git a/Dockerfile b/Dockerfile
index 70804f7259b..d89e6c60f57 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -1,4 +1,4 @@
-# v0.8.8-rc4
+# v0.8.8
# Base node image
FROM node:24.16.0-alpine AS node
diff --git a/Dockerfile.multi b/Dockerfile.multi
index 771754afbc2..2e837734549 100644
--- a/Dockerfile.multi
+++ b/Dockerfile.multi
@@ -1,5 +1,5 @@
# Dockerfile.multi
-# v0.8.8-rc4
+# v0.8.8
# Set configurable max-old-space-size with default
ARG NODE_MAX_OLD_SPACE_SIZE=6144
diff --git a/README.md b/README.md
index 279604df75e..2caf0102964 100644
--- a/README.md
+++ b/README.md
@@ -51,17 +51,21 @@
-## 🚀 What's New in v0.8.8-rc4
-
-- **Public Agents API docs:** Serve an OpenAPI specification and interactive Swagger UI for inference, events, Agent management, and Skill management.
-- **Attached workspaces (highly experimental):** Isolate workspaces by conversation, load repository instructions, and use bounded queue waits and command timeouts.
-- **Trace Viewer:** Inspect model conversations as ordered steps with roles, Agent identity, tool rounds, previews, and cost.
-- **Skills:** Author or import a Skill and invoke it in the same Agent run, with safer rollback for failed imports.
-- **Agent activity:** Render system events as distinct turns and hold live activity to one stable row.
-- **MCP reliability:** Send per-request headers without hiding tools, coordinate OAuth refresh across replicas, and preserve credentials through provider outages.
-- **Performance:** Stream Markdown incrementally, virtualize model search, and reduce completed Agent message rendering work.
-
-Read the [full v0.8.8-rc4 changelog](https://www.librechat.ai/changelog/v0.8.8-rc4).
+## 🚀 What's New in v0.8.8
+
+- **Agent Management API (beta):** Create, discover, update, and delete Agents; manage Agent files and Skills; and authenticate machine clients through deployment-bound OIDC identities while preserving existing role and Agent access controls. Public OpenAPI and Swagger UI cover inference, events, Agent management, and Skill management.
+- **Attached workspaces (highly experimental):** Select or save a per-Agent default workspace for each managed or personal code worker, then let Agents inspect trees, read and search files, author changes, and run Bash with bounded timeouts. Personal workers support bounded self-service enrollment, readiness status, and per-Agent Git identity.
+- **Background tool controls:** Optionally cancel ordinary background tools, including attached Bash, while keeping detached Subagent execution independent.
+- **Code approval controls:** Choose **Ask**, **Allow**, or **Deny** for file writes and command execution where administrators permit it, including a **Full access** mode for trusted attached environments. File Search and Run Code also honor role grants.
+- **Manual context compaction:** Start a summarize-only turn before the context window fills while preserving recent conversation content according to the deployment's summarization policy.
+- **Context Usage:** Inspect dialogue, retained tool traffic, Agent instructions, cache, cost, and runway pressure without double-counting category subsets.
+- **Unified attachments:** Upload once and let LibreChat route content to the model or extracted text, then provision File Search and Code tools only when needed.
+- **Models:** Added GPT-6 Astra and GPT-6.1 Sol, with Responses API routing and tool-call support.
+- **Agent and chat UI:** Unified tool activity, reasoning, search, and Agent workflows; added one draggable Pinned section for chats and favorites, morphing state icons, high-contrast themes, rich-text message copying, clearer sidebar titles, and refined live phase layouts.
+- **Observability:** Inspect ordered model conversations, tool rounds, and costs in the Trace Viewer; export correlated application logs through OpenTelemetry, configure allowlisted Langfuse trace identity and metadata, tag browser diagnostics with client build IDs, and scope Insights to authorized Agents.
+- **Reliability and security:** Strengthened Agent continuation and checkpoint recovery, Redis liveness detection, DocumentDB coordination, OpenID and MCP OAuth sessions, shared-link throttling, tenant isolation, attachment bounds, and upload error handling.
+
+Read the [full v0.8.8 changelog](https://www.librechat.ai/changelog/v0.8.8).
# ✨ Features
diff --git a/api/app/clients/BaseClient.js b/api/app/clients/BaseClient.js
index 12bb1046908..dc37997ffe9 100644
--- a/api/app/clients/BaseClient.js
+++ b/api/app/clients/BaseClient.js
@@ -25,6 +25,7 @@ const {
withBalanceReservations,
findCheckpointSummaryPart,
getSummaryPartText,
+ resolveCheckpointMessage,
runAfterSeed,
saveTurnConversation,
seedTurnConversation,
@@ -1463,6 +1464,7 @@ class BaseClient {
* - The message's 'role' is set to 'system'.
* - The message's 'text' is set to its 'summary'.
* - If the message has a 'summaryTokenCount', the message's 'tokenCount' is set to 'summaryTokenCount'.
+ * - A message with a summary content block keeps its content from that block on, for the SDK formatter to promote.
* The traversal stops at the message with the 'summary' property.
*
* Each message object should have an 'id' or 'messageId' property and may have a 'parentMessageId' property.
@@ -1512,35 +1514,13 @@ class BaseClient {
break;
}
- let resolved = message;
- let hasSummary = false;
- if (summary) {
- const summaryBlock = findCheckpointSummaryPart(message.content);
- if (summaryBlock) {
- const summaryText = getSummaryPartText(summaryBlock);
- resolved = {
- ...message,
- role: 'system',
- content: [{ type: ContentTypes.TEXT, text: summaryText }],
- tokenCount: summaryBlock.tokenCount,
- };
- hasSummary = true;
- } else if (message.summary) {
- resolved = {
- ...message,
- role: 'system',
- content: [{ type: ContentTypes.TEXT, text: message.summary }],
- tokenCount: message.summaryTokenCount ?? message.tokenCount,
- };
- hasSummary = true;
- }
- }
-
+ const checkpoint = summary ? resolveCheckpointMessage(message) : null;
+ const resolved = checkpoint ?? message;
const shouldMap = mapMethod != null && (mapCondition != null ? mapCondition(resolved) : true);
const processedMessage = shouldMap ? mapMethod(resolved) : resolved;
orderedMessages.push(processedMessage);
- if (hasSummary) {
+ if (checkpoint) {
break;
}
diff --git a/api/app/clients/specs/BaseClient.test.js b/api/app/clients/specs/BaseClient.test.js
index 1a1059c2db8..d6e1549adec 100644
--- a/api/app/clients/specs/BaseClient.test.js
+++ b/api/app/clients/specs/BaseClient.test.js
@@ -550,9 +550,10 @@ describe('BaseClient', () => {
summary: true,
});
expect(result).toHaveLength(2);
- expect(result[0].role).toBe('system');
- expect(result[0].content).toEqual([{ type: 'text', text: 'Content block summary' }]);
- expect(result[0].tokenCount).toBe(42);
+ expect(result[0].role).toBeUndefined();
+ expect(result[0].content).toEqual([
+ { type: 'summary', text: 'Content block summary', tokenCount: 42 },
+ ]);
});
it('should prefer content block summary over legacy summary field', () => {
@@ -573,8 +574,9 @@ describe('BaseClient', () => {
summary: true,
});
expect(result).toHaveLength(2);
- expect(result[0].content).toEqual([{ type: 'text', text: 'Content block summary' }]);
- expect(result[0].tokenCount).toBe(20);
+ expect(result[0].content).toEqual([
+ { type: 'summary', text: 'Content block summary', tokenCount: 20 },
+ ]);
});
it('should fallback to legacy summary when no content block exists', () => {
@@ -598,6 +600,34 @@ describe('BaseClient', () => {
expect(result[0].tokenCount).toBe(15);
});
+ it('keeps the parts a response produced after its summary (summary mode)', () => {
+ const trailingText = { type: 'text', text: 'Answer after summarizing' };
+ const messagesWithTrailingParts = [
+ { id: '1', parentMessageId: null, text: 'Message 1' },
+ {
+ id: '2',
+ parentMessageId: '1',
+ text: 'Before summarizing',
+ content: [
+ { type: 'text', text: 'Before summarizing' },
+ { type: 'summary', text: 'Earlier context', tokenCount: 6 },
+ trailingText,
+ ],
+ },
+ { id: '3', parentMessageId: '2', text: 'Message 3' },
+ ];
+ const result = TestClient.constructor.getMessagesForConversation({
+ messages: messagesWithTrailingParts,
+ parentMessageId: '3',
+ summary: true,
+ });
+ expect(result.map((message) => message.id)).toEqual(['2', '3']);
+ expect(result[0].content).toEqual([
+ { type: 'summary', text: 'Earlier context', tokenCount: 6 },
+ trailingText,
+ ]);
+ });
+
it('should not stop traversal at a failed summary, keeping the prior history', () => {
/** A summarize round that errored keeps the deltas it streamed, so its
* text is a truncated prefix; treating it as the checkpoint would send
diff --git a/api/cache/banViolation.js b/api/cache/banViolation.js
index 36945ca4207..acae73b91ef 100644
--- a/api/cache/banViolation.js
+++ b/api/cache/banViolation.js
@@ -1,6 +1,6 @@
const { logger } = require('@librechat/data-schemas');
const { ViolationTypes } = require('librechat-data-provider');
-const { isEnabled, math, removePorts } = require('@librechat/api');
+const { isEnabled, math, removePorts, getBanIp } = require('@librechat/api');
const { deleteAllUserSessions } = require('~/models');
const getLogStores = require('./getLogStores');
@@ -65,6 +65,7 @@ const banViolation = async (req, res, errorMessage) => {
}
req.ip = removePorts(req);
+ const banIp = getBanIp(req);
logger.info(
`[BAN] Banning user ${user_id} ${req.ip ? `@ ${req.ip} ` : ''}for ${
duration / 1000 / 60
@@ -73,8 +74,8 @@ const banViolation = async (req, res, errorMessage) => {
const expiresAt = Date.now() + duration;
await banLogs.set(user_id, { type, violation_count, duration, expiresAt });
- if (req.ip) {
- await banLogs.set(req.ip, { type, user_id, violation_count, duration, expiresAt });
+ if (banIp) {
+ await banLogs.set(banIp, { type, user_id, violation_count, duration, expiresAt });
}
errorMessage.ban = true;
diff --git a/api/cache/banViolation.spec.js b/api/cache/banViolation.spec.js
index df987534986..dbcbddd4e43 100644
--- a/api/cache/banViolation.spec.js
+++ b/api/cache/banViolation.spec.js
@@ -1,5 +1,8 @@
const mongoose = require('mongoose');
const { MongoMemoryServer } = require('mongodb-memory-server');
+const { ViolationTypes } = require('librechat-data-provider');
+const { deleteAllUserSessions } = require('~/models');
+const getLogStores = require('./getLogStores');
const banViolation = require('./banViolation');
// Mock deleteAllUserSessions since we're testing ban logic, not session deletion
@@ -81,6 +84,40 @@ describe('banViolation', () => {
expect(errorMessage.ban).toBeTruthy();
});
+ it.each(['127.0.0.1', '::1', '::ffff:127.0.0.1'])(
+ 'bans only the user for a verified trigger request from %s',
+ async (ip) => {
+ const banLogs = getLogStores(ViolationTypes.BAN);
+ await banLogs.delete(ip);
+ req.ip = ip;
+ req._isAgentTrigger = true;
+ errorMessage.violation_count = 20;
+
+ await banViolation(req, res, errorMessage);
+
+ expect(await banLogs.get(errorMessage.user_id)).toEqual(
+ expect.objectContaining({ violation_count: 20 }),
+ );
+ expect(await banLogs.get(ip)).toBeUndefined();
+ expect(deleteAllUserSessions).toHaveBeenCalledWith({ userId: errorMessage.user_id });
+ expect(res.clearCookie).toHaveBeenCalledWith('refreshToken');
+ expect(errorMessage.ban).toBe(true);
+ expect(req.ip).toBe(ip);
+ },
+ );
+
+ it('still bans the IP when an ordinary request sends a trigger header', async () => {
+ req.headers = { 'x-lc-agent-trigger': '1' };
+ errorMessage.violation_count = 20;
+
+ await banViolation(req, res, errorMessage);
+
+ const banLogs = getLogStores(ViolationTypes.BAN);
+ expect(await banLogs.get(req.ip)).toEqual(
+ expect.objectContaining({ user_id: errorMessage.user_id, violation_count: 20 }),
+ );
+ });
+
it('should handle invalid BAN_INTERVAL and default to 20', async () => {
process.env.BAN_INTERVAL = 'invalid';
errorMessage.prev_count = 19;
diff --git a/api/package.json b/api/package.json
index 3cdb8cb3b31..6e9c002a699 100644
--- a/api/package.json
+++ b/api/package.json
@@ -1,6 +1,6 @@
{
"name": "@librechat/backend",
- "version": "v0.8.8-rc4",
+ "version": "v0.8.8",
"description": "",
"scripts": {
"start": "echo 'please run this from the root directory'",
@@ -46,7 +46,7 @@
"@azure/storage-blob": "^12.30.0",
"@google/genai": "^2.8.0",
"@keyv/redis": "5.1.6",
- "@librechat/agents": "^3.9.7",
+ "@librechat/agents": "^4.0.1",
"@librechat/api": "*",
"@librechat/data-schemas": "*",
"@microsoft/microsoft-graph-client": "^3.0.7",
@@ -74,7 +74,7 @@
"cookie-parser": "^1.4.7",
"cors": "^2.8.5",
"dedent": "^1.5.3",
- "dompurify": "^3.4.12",
+ "dompurify": "^3.4.16",
"dotenv": "^16.0.3",
"eventsource": "^3.0.2",
"express": "^5.2.1",
@@ -107,10 +107,10 @@
"module-alias": "^2.2.3",
"mongodb": "^6.14.2",
"mongoose": "^8.24.1",
- "multer": "^2.3.0",
+ "multer": "^2.4.0",
"nanoid": "^3.3.18",
"node-fetch": "^2.7.0",
- "nodemailer": "^10.0.1",
+ "nodemailer": "^10.0.2",
"ollama": "^0.5.0",
"openai": "5.8.2",
"openid-client": "^6.5.0",
@@ -131,7 +131,7 @@
"sharp": "^0.35.4",
"swagger-ui-dist": "^5.32.15",
"ua-parser-js": "^1.0.36",
- "undici": "^7.29.0",
+ "undici": "^7.29.1",
"winston": "^3.11.0",
"winston-daily-rotate-file": "^5.0.0",
"xlsx": "https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz",
diff --git a/api/server/controllers/agents/__tests__/callbacks.spec.js b/api/server/controllers/agents/__tests__/callbacks.spec.js
index 2ffcc79952d..19f4be4c425 100644
--- a/api/server/controllers/agents/__tests__/callbacks.spec.js
+++ b/api/server/controllers/agents/__tests__/callbacks.spec.js
@@ -33,6 +33,8 @@ jest.mock('@librechat/api', () => ({
isCodeArtifactToolOutput: jest.requireActual('@librechat/api').isCodeArtifactToolOutput,
isCodeSessionToolName: jest.requireActual('@librechat/api').isCodeSessionToolName,
collectToolCallIds: jest.requireActual('@librechat/api').collectToolCallIds,
+ captureSubagentIdentity: jest.requireActual('@librechat/api').captureSubagentIdentity,
+ createToolTimingAdapter: jest.requireActual('@librechat/api').createToolTimingAdapter,
stampCommandExecutor: jest.requireActual('@librechat/api').stampCommandExecutor,
}));
@@ -41,6 +43,7 @@ jest.mock('@librechat/data-schemas', () => ({
debug: jest.fn(),
info: jest.fn(),
error: jest.fn(),
+ warn: jest.fn(),
},
}));
@@ -371,6 +374,139 @@ describe('resumable event generation fencing', () => {
expect(resumedPublish.mock.calls[0][0].activityEventId).not.toBe(firstUpdate.activityEventId);
});
+ it('publishes tool preparation and handoff into event-child activity', async () => {
+ const { GraphEvents, createContentAggregator } = jest.requireActual('@librechat/agents');
+ const { getDefaultHandlers } = require('../callbacks');
+ const publish = jest.fn().mockResolvedValue(undefined);
+ const { contentParts, stepMap, aggregateContent } = createContentAggregator();
+ const handlers = getDefaultHandlers({
+ res: { write: jest.fn() },
+ aggregateContent,
+ contentParts,
+ stepMap,
+ toolEndCallback: jest.fn(),
+ collectedUsage: [],
+ streamId: 'event-thread',
+ eventChildActivity: {
+ runId: 'event-thread',
+ parentRunId: 'parent-conversation',
+ subagentRunId: 'child-1',
+ subagentType: 'researcher',
+ subagentAgentId: 'agent-1',
+ parentAgentId: 'director',
+ publish,
+ },
+ });
+ const step = {
+ id: 'step-child',
+ index: 0,
+ type: 'tool_calls',
+ stepDetails: {
+ type: 'tool_calls',
+ tool_calls: [{ id: 'call-child', name: 'query', args: '{}' }],
+ },
+ };
+ await handlers[GraphEvents.ON_RUN_STEP].handle(GraphEvents.ON_RUN_STEP, step);
+ await handlers[GraphEvents.ON_RUN_STEP_DELTA].handle(GraphEvents.ON_RUN_STEP_DELTA, {
+ id: 'step-child',
+ observed_at: 100,
+ delta: { type: 'tool_calls', tool_calls: [{ id: 'call-child', index: 0, args: '{' }] },
+ });
+ await handlers[StepEvents.ON_TOOL_CALLS_DISPATCHED].handle(
+ StepEvents.ON_TOOL_CALLS_DISPATCHED,
+ {
+ dispatched_at: 500,
+ toolCalls: [{ id: 'call-child', name: 'query', stepId: 'step-child' }],
+ },
+ );
+ await new Promise((resolve) => setImmediate(resolve));
+ expect(publish.mock.calls.map(([value]) => value.phase)).toEqual([
+ 'run_step',
+ 'tool_preparation',
+ 'run_step_delta',
+ 'tool_calls_dispatched',
+ ]);
+ expect(publish.mock.calls[1][0].data).toEqual({
+ id: 'step-child',
+ index: 0,
+ toolCallId: 'call-child',
+ observed_at: 100,
+ });
+ expect(publish.mock.calls[3][0].data.toolCalls[0]).not.toHaveProperty('args');
+ });
+
+ it('folds child dispatch and result into the parent-owned subagent tool part', async () => {
+ const { GraphEvents } = jest.requireActual('@librechat/agents');
+ const { getDefaultHandlers } = require('../callbacks');
+ const aggregators = new Map();
+ const handlers = getDefaultHandlers({
+ res: { write: jest.fn() },
+ aggregateContent: jest.fn(),
+ toolEndCallback: jest.fn(),
+ collectedUsage: [],
+ subagentAggregatorsByToolCallId: aggregators,
+ });
+ const base = {
+ parentToolCallId: 'parent-call',
+ parentRunId: 'parent-run',
+ subagentRunId: 'child-run',
+ subagentType: 'researcher',
+ subagentAgentId: 'child-agent',
+ runId: 'parent-run',
+ };
+ for (const event of [
+ {
+ phase: 'run_step',
+ data: {
+ id: 'child-step',
+ index: 0,
+ type: 'tool_calls',
+ stepDetails: {
+ type: 'tool_calls',
+ tool_calls: [{ id: 'child-call', name: 'query', args: '{}' }],
+ },
+ },
+ },
+ {
+ phase: 'tool_preparation',
+ data: { id: 'child-step', toolCallId: 'child-call', observed_at: 100 },
+ },
+ {
+ phase: 'tool_calls_dispatched',
+ data: {
+ dispatched_at: 500,
+ toolCalls: [{ id: 'child-call', stepId: 'child-step', name: 'query' }],
+ },
+ },
+ {
+ phase: 'run_step_completed',
+ data: {
+ result: {
+ id: 'child-step',
+ index: 0,
+ type: 'tool_call',
+ completed_at: 540,
+ tool_call: { id: 'child-call', name: 'query', args: '{}', output: 'ok', progress: 1 },
+ },
+ },
+ },
+ ]) {
+ await handlers[GraphEvents.ON_SUBAGENT_UPDATE].handle(GraphEvents.ON_SUBAGENT_UPDATE, {
+ ...base,
+ ...event,
+ });
+ }
+ expect(jest.requireMock('@librechat/data-schemas').logger.warn).not.toHaveBeenCalled();
+ expect(aggregators.get('parent-call')?.contentParts[0]?.tool_call).toMatchObject({
+ id: 'child-call',
+ toolPreparationStartedAt: 100,
+ toolPreparationDurationMs: 400,
+ toolDispatchedAt: 500,
+ toolExecutionDurationMs: 40,
+ output: 'ok',
+ });
+ });
+
it('forwards the originating job epoch with deferred attachments', () => {
const { GenerationJobManager } = require('@librechat/api');
const { createAttachmentEmitter } = require('../callbacks');
@@ -1381,6 +1517,145 @@ describe('createToolEndCallback', () => {
});
});
+describe('tool dispatch timing', () => {
+ it('forwards the SDK handoff and stores preparation and result intervals independently', async () => {
+ const { GraphEvents, createContentAggregator } = jest.requireActual('@librechat/agents');
+ const { GenerationJobManager } = require('@librechat/api');
+ const { getDefaultHandlers } = require('../callbacks');
+ const { contentParts, stepMap, aggregateContent } = createContentAggregator();
+ const handlers = getDefaultHandlers({
+ res: { write: jest.fn() },
+ contentParts,
+ stepMap,
+ aggregateContent,
+ toolEndCallback: jest.fn(),
+ collectedUsage: [],
+ streamId: 'run',
+ });
+ const step = {
+ id: 'step-1',
+ index: 0,
+ type: 'tool_calls',
+ stepDetails: {
+ type: 'tool_calls',
+ tool_calls: [{ id: 'call-1', name: 'query', args: '{}' }],
+ },
+ };
+ await handlers[GraphEvents.ON_RUN_STEP].handle(GraphEvents.ON_RUN_STEP, step);
+ await handlers[GraphEvents.ON_RUN_STEP_DELTA].handle(GraphEvents.ON_RUN_STEP_DELTA, {
+ id: 'step-1',
+ observed_at: 1_000,
+ delta: { type: 'tool_calls', tool_calls: [{ id: 'call-1', index: 0, args: '{' }] },
+ });
+ const dispatched = {
+ dispatched_at: 248_000,
+ toolCalls: [{ id: 'call-1', name: 'query', stepId: 'step-1' }],
+ };
+ await handlers[StepEvents.ON_TOOL_CALLS_DISPATCHED].handle(
+ StepEvents.ON_TOOL_CALLS_DISPATCHED,
+ dispatched,
+ );
+ await handlers[GraphEvents.ON_RUN_STEP_COMPLETED].handle(GraphEvents.ON_RUN_STEP_COMPLETED, {
+ result: {
+ id: 'step-1',
+ completed_at: 248_340,
+ index: 0,
+ tool_call: { id: 'call-1', name: 'query', args: '{}', output: 'ok' },
+ },
+ });
+ await handlers[GraphEvents.ON_RUN_STEP_CLOSED].handle(GraphEvents.ON_RUN_STEP_CLOSED, {
+ id: 'step-1',
+ index: 0,
+ type: 'tool_calls',
+ status: 'completed',
+ created_at: 1_000,
+ closed_at: 248_340,
+ });
+ expect(GenerationJobManager.emitChunk).toHaveBeenCalledWith(
+ 'run',
+ {
+ event: StepEvents.ON_TOOL_PREPARATION,
+ data: {
+ id: 'step-1',
+ index: 0,
+ toolCallId: 'call-1',
+ observed_at: 1_000,
+ },
+ },
+ expect.anything(),
+ );
+ expect(GenerationJobManager.emitChunk).toHaveBeenCalledWith(
+ 'run',
+ { event: StepEvents.ON_TOOL_CALLS_DISPATCHED, data: dispatched },
+ expect.anything(),
+ );
+ expect(contentParts[0].tool_call).toMatchObject({
+ runStepDurationMs: 247_340,
+ runStepClosedAt: 248_340,
+ toolPreparationDurationMs: 247_000,
+ toolExecutionDurationMs: 340,
+ });
+ });
+
+ it('keeps preparation across a new handler created after HITL approval', async () => {
+ const { GraphEvents, createContentAggregator } = jest.requireActual('@librechat/agents');
+ const { getDefaultHandlers } = require('../callbacks');
+ const { contentParts, stepMap, aggregateContent } = createContentAggregator();
+ const handlers = getDefaultHandlers({
+ res: { write: jest.fn() },
+ contentParts,
+ stepMap,
+ aggregateContent,
+ toolEndCallback: jest.fn(),
+ collectedUsage: [],
+ toolTimingReplayEvents: [
+ {
+ event: StepEvents.ON_TOOL_PREPARATION,
+ data: {
+ id: 'step-1',
+ index: 0,
+ toolCallId: 'call-1',
+ observed_at: 1_000,
+ },
+ },
+ ],
+ });
+ await handlers[GraphEvents.ON_RUN_STEP].handle(GraphEvents.ON_RUN_STEP, {
+ id: 'step-1',
+ index: 0,
+ type: 'tool_calls',
+ stepDetails: {
+ type: 'tool_calls',
+ tool_calls: [{ id: 'call-1', name: 'query', args: '{}' }],
+ },
+ });
+ await handlers[StepEvents.ON_TOOL_CALLS_DISPATCHED].handle(
+ StepEvents.ON_TOOL_CALLS_DISPATCHED,
+ { dispatched_at: 51_000, toolCalls: [{ id: 'call-1', name: 'query', stepId: 'step-1' }] },
+ );
+ await handlers[GraphEvents.ON_RUN_STEP_COMPLETED].handle(GraphEvents.ON_RUN_STEP_COMPLETED, {
+ result: {
+ id: 'step-1',
+ index: 0,
+ completed_at: 51_200,
+ tool_call: { id: 'call-1', name: 'query', args: '{}', output: 'ok' },
+ },
+ });
+ await handlers[GraphEvents.ON_RUN_STEP_CLOSED].handle(GraphEvents.ON_RUN_STEP_CLOSED, {
+ id: 'step-1',
+ index: 0,
+ type: 'tool_calls',
+ status: 'completed',
+ created_at: 1_000,
+ closed_at: 51_200,
+ });
+ expect(contentParts[0].tool_call).toMatchObject({
+ toolPreparationDurationMs: 50_000,
+ toolExecutionDurationMs: 200,
+ });
+ });
+});
+
describe('tool input validation marker', () => {
it('marks the streamed result and persisted content part out of band', async () => {
const { GraphEvents, createContentAggregator } = jest.requireActual('@librechat/agents');
diff --git a/api/server/controllers/agents/__tests__/client.retainedAnswers.spec.js b/api/server/controllers/agents/__tests__/client.retainedAnswers.spec.js
index aa0f035dbf4..27a308a3038 100644
--- a/api/server/controllers/agents/__tests__/client.retainedAnswers.spec.js
+++ b/api/server/controllers/agents/__tests__/client.retainedAnswers.spec.js
@@ -222,6 +222,11 @@ describe('AgentClient retained answers', () => {
client.user = 'user-123';
client.processMemory = jest.fn();
const rows = compactedBranch();
+ /** The stored count covers the whole response, including what preceded its summary. */
+ const storedResponseTokens = 5000;
+ const response = rows.find((row) => row.messageId === 'a2');
+ response.tokenCount = storedResponseTokens;
+ response.content.unshift({ type: ContentTypes.TEXT, text: 'Checking the pipeline.' });
getMessages.mockResolvedValue(rows);
/** The real loader: one read of the conversation, then the summary-bounded walk. */
const cut = await client.loadHistory('convo-123', 'u3');
@@ -239,6 +244,10 @@ describe('AgentClient retained answers', () => {
expect(text.indexOf(ANSWER_LINE)).toBeLessThan(text.indexOf(LATEST_TEXT));
expect(text.endsWith(LATEST_TEXT)).toBe(true);
expect(client.options.agent.additional_instructions ?? '').not.toContain(ANSWER_LINE);
+ expect(text).toContain('Deployed.');
+ expect(text).not.toContain('Checking the pipeline.');
+ expect(tokenCountMap.a2).toBeGreaterThan(0);
+ expect(tokenCountMap.a2).toBeLessThan(storedResponseTokens);
expect(cut[1].text).toBe(LATEST_TEXT);
expect(cut[1].content).toBeUndefined();
expect(counts[prompt.length - 1]).toBe(tokenCountMap.u3);
diff --git a/api/server/controllers/agents/callbacks.js b/api/server/controllers/agents/callbacks.js
index 1651661d8f5..767c5ebc0af 100644
--- a/api/server/controllers/agents/callbacks.js
+++ b/api/server/controllers/agents/callbacks.js
@@ -9,6 +9,7 @@ const {
ErrorTypes,
UsageEvents,
getRunStepDurationMs,
+ getRunStepCloseMetadata,
} = require('librechat-data-provider');
const {
GraphEvents,
@@ -34,6 +35,7 @@ const {
captureSubagentIdentity,
getAttachmentOwnership,
collectToolCallIds,
+ createToolTimingAdapter,
stampCommandExecutor,
} = require('@librechat/api');
const { processFileCitations } = require('~/server/services/Files/Citations');
@@ -334,10 +336,11 @@ function subagentPhaseToGraphEvent(event) {
* @param {{ aggregateContent: Function, contentParts?: Array, stepMap?: Map }} aggregator
* @param {SubagentUpdateEvent} event
*/
-function feedSubagentAggregator(aggregator, event) {
+function feedSubagentAggregator(aggregator, event, applyChildTiming) {
const graphEvent = subagentPhaseToGraphEvent(event);
+ if (graphEvent) aggregator.aggregateContent({ event: graphEvent, data: event.data });
+ applyChildTiming(aggregator, event);
if (!graphEvent) return;
- aggregator.aggregateContent({ event: graphEvent, data: event.data });
/** The SDK aggregator intentionally projects run-step tool calls onto its
* public content shape, so host-only routing metadata is not copied. Restore
@@ -408,6 +411,7 @@ function getDefaultHandlers({
usageEmitSink = null,
eventChildActivity = null,
resolveMcpServerName = null,
+ toolTimingReplayEvents = [],
}) {
if (!res || !aggregateContent) {
throw new Error(
@@ -417,6 +421,8 @@ function getDefaultHandlers({
const eventActivityPhases = {
[GraphEvents.ON_RUN_STEP]: 'run_step',
[GraphEvents.ON_RUN_STEP_DELTA]: 'run_step_delta',
+ [StepEvents.ON_TOOL_PREPARATION]: 'tool_preparation',
+ [StepEvents.ON_TOOL_CALLS_DISPATCHED]: 'tool_calls_dispatched',
[GraphEvents.ON_RUN_STEP_COMPLETED]: 'run_step_completed',
[GraphEvents.ON_RUN_STEP_CLOSED]: 'run_step_closed',
[GraphEvents.ON_MESSAGE_DELTA]: 'message_delta',
@@ -506,7 +512,12 @@ function getDefaultHandlers({
}
return emitForJob({ event: UsageEvents.ON_TOKEN_USAGE, data: payload });
};
+ const toolTiming = createToolTimingAdapter({
+ replayEvents: toolTimingReplayEvents,
+ emit: emitForJob,
+ });
const handlers = {
+ [StepEvents.ON_TOOL_CALLS_DISPATCHED]: toolTiming.dispatch,
[GraphEvents.CHAT_MODEL_END]: new ModelEndHandler(
collectedUsage,
collectedThoughtSignatures,
@@ -587,7 +598,9 @@ function getDefaultHandlers({
const index = stepMap?.get(stepId)?.index;
const part = typeof index === 'number' ? contentParts[index] : undefined;
if (part?.type === ContentTypes.TOOL_CALL && part.tool_call) {
+ toolTiming.close(part.tool_call, stepId);
part.tool_call.runStepStatus = data.status;
+ Object.assign(part.tool_call, getRunStepCloseMetadata(data));
/**
* The raw derivable duration, left unset rather than zeroed when
* the event cannot support a trustworthy one — no `created_at`,
@@ -613,6 +626,7 @@ function getDefaultHandlers({
*/
handle: async (event, data, metadata) => {
aggregateContent({ event, data });
+ await toolTiming.delta(data);
if (data?.delta.type === StepTypes.TOOL_CALLS) {
await emitForJob({ event, data });
} else if (checkIfLastAgent(metadata?.last_agent_id, metadata?.langgraph_node)) {
@@ -648,6 +662,7 @@ function getDefaultHandlers({
agentId: metadata?.agent_id,
});
}
+ toolTiming.completed(data);
aggregateContent({ event, data });
const stepId = data?.result?.id;
const runStep = stepMap?.get(stepId);
@@ -761,7 +776,7 @@ function getDefaultHandlers({
}
try {
captureSubagentIdentity(aggregator, data);
- feedSubagentAggregator(aggregator, data);
+ feedSubagentAggregator(aggregator, data, toolTiming.child);
} catch (err) {
logger.warn(
`[ON_SUBAGENT_UPDATE] Failed to aggregate phase "${data?.phase}" for tool_call ${key}: ${err?.message ?? err}`,
diff --git a/api/server/controllers/agents/resume.js b/api/server/controllers/agents/resume.js
index c162b05216a..6608e8beb76 100644
--- a/api/server/controllers/agents/resume.js
+++ b/api/server/controllers/agents/resume.js
@@ -1961,6 +1961,7 @@ const ResumeAgentController = async (req, res, next, initializeClient, addTitle)
checkpointNamespace,
foregroundRunId: mcpRequestBody.messageId,
requestBody: mcpRequestBody,
+ toolTimingReplayEvents: resumeState?.replayEvents,
});
client = result.client;
diff --git a/api/server/experimental.js b/api/server/experimental.js
index 0877482f66c..52870f2e4cf 100644
--- a/api/server/experimental.js
+++ b/api/server/experimental.js
@@ -85,7 +85,6 @@ const { getAppConfig } = require('./services/Config');
const staticCache = require('./utils/staticCache');
const optionalJwtAuth = require('./middleware/optionalJwtAuth');
const noIndex = require('./middleware/noIndex');
-const routes = require('./routes');
const agentEventMethods = require('~/models');
/** Route admin file-config MIME patterns through a linear-time engine (ReDoS-safe) on upload. */
@@ -517,6 +516,10 @@ if (cluster.isMaster) {
await updateInterfacePerms({ appConfig, getRoleByName, updateAccessPermissions });
});
+ /* Route modules build their rate limiters as they load, so they load only after the
+ * startup checks have applied `rateLimits` from librechat.yaml. */
+ const routes = require('./routes');
+
/** Load index.html for SPA serving */
const indexPath = path.join(appConfig.paths.dist, 'index.html');
let indexHTML = fs.readFileSync(indexPath, 'utf8');
diff --git a/api/server/index.js b/api/server/index.js
index 33b14a1db57..134cc5b7a3a 100644
--- a/api/server/index.js
+++ b/api/server/index.js
@@ -87,7 +87,6 @@ const createSpaFallback = require('./utils/fallback');
const { getAppConfig } = require('./services/Config');
const staticCache = require('./utils/staticCache');
const noIndex = require('./middleware/noIndex');
-const routes = require('./routes');
const agentEventMethods = require('~/models');
/** Route admin file-config MIME patterns through a linear-time engine (ReDoS-safe) on upload. */
@@ -273,6 +272,10 @@ const startServer = async () => {
await updateInterfacePermissions({ appConfig, getRoleByName, updateAccessPermissions });
});
+ /* Route modules build their rate limiters as they load, so they load only after the
+ * startup checks have applied `rateLimits` from librechat.yaml. */
+ const routes = require('./routes');
+
const indexPath = path.join(appConfig.paths.dist, 'index.html');
let indexHTML = fs.readFileSync(indexPath, 'utf8');
diff --git a/api/server/middleware/checkBan.js b/api/server/middleware/checkBan.js
index 0871e51c047..305c87bfe42 100644
--- a/api/server/middleware/checkBan.js
+++ b/api/server/middleware/checkBan.js
@@ -2,7 +2,7 @@ const { Keyv } = require('keyv');
const uap = require('ua-parser-js');
const { logger } = require('@librechat/data-schemas');
const { ErrorTypes, ViolationTypes } = require('librechat-data-provider');
-const { isEnabled, keyvMongo, removePorts } = require('@librechat/api');
+const { isEnabled, keyvMongo, removePorts, getBanIp } = require('@librechat/api');
const { getLogStores } = require('~/cache');
const { isOAuthNavigation, redirectOAuthFailure } = require('./oauthNavigation');
const denyRequest = require('./denyRequest');
@@ -88,6 +88,7 @@ const checkBan = async (req, res, next = () => {}) => {
}
req.ip = removePorts(req);
+ const banIp = getBanIp(req);
let userId = req.user?.id ?? req.user?._id?.toString() ?? null;
if (!userId && req?.body?.email) {
@@ -95,12 +96,12 @@ const checkBan = async (req, res, next = () => {}) => {
userId = user?._id ? user._id.toString() : userId;
}
- if (!userId && !req.ip) {
+ if (!userId && !banIp) {
return next();
}
const useRedis = isEnabled(process.env.USE_REDIS);
- const ipKey = getBanCacheKey('ip', req.ip, useRedis);
+ const ipKey = getBanCacheKey('ip', banIp, useRedis);
const userKey = getBanCacheKey('user', userId, useRedis);
const [cachedIPBan, cachedUserBan] = await Promise.all([
@@ -121,7 +122,7 @@ const checkBan = async (req, res, next = () => {}) => {
}
const [ipBan, userBan] = await Promise.all([
- req.ip ? banLogs.get(req.ip) : undefined,
+ banIp ? banLogs.get(banIp) : undefined,
userId ? banLogs.get(userId) : undefined,
]);
@@ -142,7 +143,7 @@ const checkBan = async (req, res, next = () => {}) => {
if (timeLeft <= 0) {
const cleanups = [];
if (ipBan) {
- cleanups.push(banLogs.delete(req.ip));
+ cleanups.push(banLogs.delete(banIp));
}
if (userBan) {
cleanups.push(banLogs.delete(userId));
diff --git a/api/server/middleware/checkBan.spec.js b/api/server/middleware/checkBan.spec.js
index ee83f0bef8f..8e52719cc0b 100644
--- a/api/server/middleware/checkBan.spec.js
+++ b/api/server/middleware/checkBan.spec.js
@@ -75,6 +75,38 @@ describe('checkBan with namespaced Keyv stores and ObjectId user ids', () => {
expect(req.banned).toBeUndefined();
});
+ it('isolates concurrent trigger violations to their user on the shared loopback transport', async () => {
+ const userId = new mongoose.Types.ObjectId();
+ const req = {
+ ip: '::1',
+ user: { id: userId.toString() },
+ headers: {},
+ _isAgentTrigger: true,
+ };
+ const errorMessage = { type: ViolationTypes.CONCURRENT };
+
+ await logViolation(req, createRes(), ViolationTypes.CONCURRENT, errorMessage, 20);
+ expect(errorMessage.ban).toBe(true);
+
+ const otherUserReq = {
+ ...req,
+ user: { id: new mongoose.Types.ObjectId().toString() },
+ };
+ const next = jest.fn();
+ await checkBan(otherUserReq, createRes(), next);
+
+ expect(next).toHaveBeenCalledWith();
+ expect(otherUserReq.banned).toBeUndefined();
+
+ const bannedUserRes = createRes();
+ const bannedUserNext = jest.fn();
+ await checkBan(req, bannedUserRes, bannedUserNext);
+
+ expect(bannedUserNext).not.toHaveBeenCalled();
+ expect(bannedUserRes.status).toHaveBeenCalledWith(403);
+ expect(req.banned).toBe(true);
+ });
+
it('enforces a ban recorded for the same user from another address', async () => {
const userId = new mongoose.Types.ObjectId();
const violationReq = createOAuthCallbackReq(userId, '10.0.0.2');
diff --git a/api/server/routes/auth.2fa-ratelimit.test.js b/api/server/routes/auth.2fa-ratelimit.test.js
index 7dba77869f5..cd496cce335 100644
--- a/api/server/routes/auth.2fa-ratelimit.test.js
+++ b/api/server/routes/auth.2fa-ratelimit.test.js
@@ -7,6 +7,11 @@ const mockCheckBan = jest.fn((req, res, next) => next());
const mockVerify2FAWithTempToken = jest.fn((req, res) => res.status(204).end());
jest.mock('@librechat/api', () => ({
+ limiterCache: jest.fn(() => undefined),
+ createTwoFactorManagementLimiter: (...args) =>
+ require('../../../packages/api/src/middleware/twoFactor').createTwoFactorManagementLimiter(
+ ...args,
+ ),
createSetBalanceConfig: jest.fn(() => (req, res, next) => next()),
forceRefreshCloudFrontAuthCookies: jest.fn(),
}));
@@ -55,7 +60,7 @@ jest.mock('~/models', () => ({
}));
jest.mock('~/server/services/Config', () => ({
- getAppConfig: jest.fn(),
+ getAppConfig: jest.fn(async () => ({ config: { rateLimits: {} } })),
}));
jest.mock('~/server/middleware', () => {
@@ -78,10 +83,108 @@ jest.mock('~/server/middleware', () => {
resetPasswordLimiter: pass,
resetPasswordSubmissionLimiter: pass,
validatePasswordReset: pass,
- requireJwtAuth: pass,
+ requireJwtAuth: (req, res, next) => {
+ if (!req.headers['x-user']) return res.sendStatus(401);
+ req.user = { id: req.headers['x-user'], tenantId: req.headers['x-tenant'] };
+ next();
+ },
};
});
+describe('authenticated 2FA management budget', () => {
+ let app;
+ const paths = ['enable', 'verify', 'confirm', 'disable', 'backup/regenerate'];
+ beforeEach(() => {
+ require('~/server/services/Config').getAppConfig.mockResolvedValue({
+ config: { rateLimits: {} },
+ });
+ app = express();
+ app.use(express.json());
+ app.use('/api/auth', authRouter);
+ });
+
+ it('shares one budget across routes and rejects before invoking the controller', async () => {
+ for (let i = 0; i < 7; i++) {
+ await request(app)
+ .post(`/api/auth/2fa/${paths[i % paths.length]}`)
+ .set('x-user', 'cross-route-user')
+ .send({ token: '111111' })
+ .expect(204);
+ }
+ const controllers = require('~/server/controllers/TwoFactorController');
+ const calls = controllers.regenerateBackupCodes.mock.calls.length;
+ const blocked = await request(app)
+ .post('/api/auth/2fa/backup/regenerate')
+ .set('x-user', 'cross-route-user')
+ .send({ token: '111111' })
+ .expect(429);
+ expect(blocked.headers['retry-after']).toBeDefined();
+ expect(controllers.regenerateBackupCodes.mock.calls).toHaveLength(calls);
+ await request(app).post('/api/auth/2fa/disable').set('x-user', 'cross-route-user').expect(429);
+ });
+
+ it('isolates accounts and tenants and authenticates before counting', async () => {
+ const controllers = require('~/server/controllers/TwoFactorController');
+ const calls = controllers.regenerateBackupCodes.mock.calls.length;
+ for (let i = 0; i < 8; i++) {
+ await request(app).post('/api/auth/2fa/backup/regenerate').expect(401);
+ }
+ expect(controllers.regenerateBackupCodes.mock.calls).toHaveLength(calls);
+ for (let i = 0; i < 7; i++) {
+ await request(app)
+ .post('/api/auth/2fa/verify')
+ .set('x-user', 'tenant-user')
+ .set('x-tenant', 'one')
+ .expect(204);
+ }
+ await request(app)
+ .post('/api/auth/2fa/verify')
+ .set('x-user', 'tenant-user')
+ .set('x-tenant', 'one')
+ .expect(429);
+ await request(app)
+ .post('/api/auth/2fa/verify')
+ .set('x-user', 'tenant-user')
+ .set('x-tenant', 'two')
+ .expect(204);
+ await request(app)
+ .post('/api/auth/2fa/verify')
+ .set('x-user', 'other-user')
+ .set('x-tenant', 'one')
+ .expect(204);
+ });
+
+ it('admits only seven concurrent requests across operations', async () => {
+ const responses = await Promise.all(
+ Array.from({ length: 20 }, (_, i) =>
+ request(app)
+ .post(`/api/auth/2fa/${paths[i % paths.length]}`)
+ .set('x-user', 'concurrent-user')
+ .send({ token: '111111' }),
+ ),
+ );
+ expect(responses.filter(({ status }) => status === 204)).toHaveLength(7);
+ expect(responses.filter(({ status }) => status === 429)).toHaveLength(13);
+ });
+
+ it('admits the entire setup sequence with the minimum configured budget', async () => {
+ require('~/server/services/Config').getAppConfig.mockResolvedValue({
+ config: { rateLimits: { twoFactorManagement: { requestsPerFiveMinutes: 3 } } },
+ });
+ for (const operation of ['enable', 'verify', 'confirm']) {
+ await request(app)
+ .post(`/api/auth/2fa/${operation}`)
+ .set('x-user', 'minimum-budget-user')
+ .expect(204);
+ }
+ const blocked = await request(app)
+ .post('/api/auth/2fa/verify')
+ .set('x-user', 'minimum-budget-user')
+ .expect(429);
+ expect(blocked.headers['ratelimit-limit']).toBe('3');
+ });
+});
+
const authRouter = require('./auth');
describe('POST /api/auth/2fa/verify-temp rate limiting', () => {
diff --git a/api/server/routes/auth.cloudfront.test.js b/api/server/routes/auth.cloudfront.test.js
index 47421947f2c..764cb169aa0 100644
--- a/api/server/routes/auth.cloudfront.test.js
+++ b/api/server/routes/auth.cloudfront.test.js
@@ -4,6 +4,8 @@ const request = require('supertest');
const mockForceRefreshCloudFrontAuthCookies = jest.fn();
jest.mock('@librechat/api', () => ({
+ limiterCache: jest.fn(),
+ createTwoFactorManagementLimiter: jest.fn(() => (req, res, next) => next()),
createSetBalanceConfig: jest.fn(() => (req, res, next) => next()),
forceRefreshCloudFrontAuthCookies: (...args) => mockForceRefreshCloudFrontAuthCookies(...args),
}));
diff --git a/api/server/routes/auth.js b/api/server/routes/auth.js
index 55de3f365db..c8539b33eb2 100644
--- a/api/server/routes/auth.js
+++ b/api/server/routes/auth.js
@@ -1,5 +1,10 @@
const express = require('express');
-const { createSetBalanceConfig, forceRefreshCloudFrontAuthCookies } = require('@librechat/api');
+const {
+ limiterCache,
+ createSetBalanceConfig,
+ createTwoFactorManagementLimiter,
+ forceRefreshCloudFrontAuthCookies,
+} = require('@librechat/api');
const {
resetPasswordRequestController,
resetPasswordController,
@@ -37,6 +42,10 @@ const setBalanceConfig = createSetBalanceConfig({
});
const router = express.Router();
+const twoFactorManagementLimiter = createTwoFactorManagementLimiter({
+ getAppConfig: () => getAppConfig({ baseOnly: true }),
+ store: limiterCache('two_factor_management_user_limiter'),
+});
const getCloudFrontAuthCookieRefreshResult = (req, res) => {
const warmedResult = req.cloudFrontAuthCookieRefreshResult;
if (warmedResult && (warmedResult.attempted || !warmedResult.enabled)) {
@@ -97,8 +106,8 @@ router.post(
resetPasswordController,
);
-router.post('/2fa/enable', middleware.requireJwtAuth, enable2FA);
-router.post('/2fa/verify', middleware.requireJwtAuth, verify2FA);
+router.post('/2fa/enable', middleware.requireJwtAuth, twoFactorManagementLimiter, enable2FA);
+router.post('/2fa/verify', middleware.requireJwtAuth, twoFactorManagementLimiter, verify2FA);
router.post(
'/2fa/verify-temp',
middleware.requireSameOrigin,
@@ -107,9 +116,14 @@ router.post(
middleware.checkBan,
verify2FAWithTempToken,
);
-router.post('/2fa/confirm', middleware.requireJwtAuth, confirm2FA);
-router.post('/2fa/disable', middleware.requireJwtAuth, disable2FA);
-router.post('/2fa/backup/regenerate', middleware.requireJwtAuth, regenerateBackupCodes);
+router.post('/2fa/confirm', middleware.requireJwtAuth, twoFactorManagementLimiter, confirm2FA);
+router.post('/2fa/disable', middleware.requireJwtAuth, twoFactorManagementLimiter, disable2FA);
+router.post(
+ '/2fa/backup/regenerate',
+ middleware.requireJwtAuth,
+ twoFactorManagementLimiter,
+ regenerateBackupCodes,
+);
/* Passkeys (WebAuthn) */
router.post(
diff --git a/api/server/routes/auth.reset-password-ratelimit.test.js b/api/server/routes/auth.reset-password-ratelimit.test.js
index 907d7f29fc6..f8f8ead7d8b 100644
--- a/api/server/routes/auth.reset-password-ratelimit.test.js
+++ b/api/server/routes/auth.reset-password-ratelimit.test.js
@@ -7,6 +7,8 @@ const mockValidatePasswordReset = jest.fn((req, res, next) => next());
const mockResetPasswordController = jest.fn((req, res) => res.status(204).end());
jest.mock('@librechat/api', () => ({
+ limiterCache: jest.fn(),
+ createTwoFactorManagementLimiter: jest.fn(() => (req, res, next) => next()),
createSetBalanceConfig: jest.fn(() => (req, res, next) => next()),
forceRefreshCloudFrontAuthCookies: jest.fn(),
}));
diff --git a/api/server/services/Endpoints/agents/initialize.js b/api/server/services/Endpoints/agents/initialize.js
index e62e955b9ab..df95bd051dc 100644
--- a/api/server/services/Endpoints/agents/initialize.js
+++ b/api/server/services/Endpoints/agents/initialize.js
@@ -205,6 +205,7 @@ const initializeClientWithProvider = async ({
checkpointNamespace,
foregroundRunId,
requestBody,
+ toolTimingReplayEvents,
upstreamTokenProvider,
upstreamTokenProviderResolver,
}) => {
@@ -1841,6 +1842,7 @@ const initializeClientWithProvider = async ({
usageEmitSink,
eventChildActivity,
resolveMcpServerName,
+ toolTimingReplayEvents,
});
const client = new AgentClient({
diff --git a/api/server/services/Files/Code/process.js b/api/server/services/Files/Code/process.js
index 6371b1c40fe..2f1afdbf2f7 100644
--- a/api/server/services/Files/Code/process.js
+++ b/api/server/services/Files/Code/process.js
@@ -1068,6 +1068,7 @@ async function readSandboxFile({
* @param {string} params.file_path
* @param {string} params.workspace_id
* @param {string} [params.workspace_instance_id]
+ * @param {boolean} [params.linked_worktrees]
* @param {number} params.start_line
* @param {number} params.max_lines
* @param {string} params.codeApiBaseUrl
@@ -1080,6 +1081,7 @@ async function readWorkspaceFile({
file_path,
workspace_id,
workspace_instance_id,
+ linked_worktrees,
start_line,
max_lines,
codeApiBaseUrl,
@@ -1093,7 +1095,9 @@ async function readWorkspaceFile({
}) {
return executeWorkspaceTool({
baseURL: codeApiBaseUrl,
+ linkedWorktrees: linked_worktrees,
maxQueueWaitMs,
+ codeApiMaxRetryWaitMs: req?.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs,
deadlineAtMs,
/** Minted per admission attempt: a queued call outlives one token TTL. */
@@ -1121,6 +1125,7 @@ async function readWorkspaceFile({
* @param {string} params.query
* @param {string} params.workspace_id
* @param {string} [params.workspace_instance_id]
+ * @param {boolean} [params.linked_worktrees]
* @param {string} [params.path]
* @param {number} params.max_results
* @param {string} params.codeApiBaseUrl
@@ -1133,6 +1138,7 @@ async function searchWorkspace({
query,
workspace_id,
workspace_instance_id,
+ linked_worktrees,
path,
max_results,
codeApiBaseUrl,
@@ -1146,7 +1152,9 @@ async function searchWorkspace({
}) {
return executeWorkspaceTool({
baseURL: codeApiBaseUrl,
+ linkedWorktrees: linked_worktrees,
maxQueueWaitMs,
+ codeApiMaxRetryWaitMs: req?.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs,
deadlineAtMs,
/** Minted per admission attempt: a queued call outlives one token TTL. */
@@ -1173,6 +1181,7 @@ async function searchWorkspace({
* @param {Object} params
* @param {string} params.workspace_id
* @param {string} [params.workspace_instance_id]
+ * @param {boolean} [params.linked_worktrees]
* @param {string} [params.path]
* @param {string} [params.after_path]
* @param {number} params.max_results
@@ -1185,6 +1194,7 @@ async function searchWorkspace({
async function listWorkspaceFiles({
workspace_id,
workspace_instance_id,
+ linked_worktrees,
path,
after_path,
max_results,
@@ -1199,7 +1209,9 @@ async function listWorkspaceFiles({
}) {
return executeWorkspaceTool({
baseURL: codeApiBaseUrl,
+ linkedWorktrees: linked_worktrees,
maxQueueWaitMs,
+ codeApiMaxRetryWaitMs: req?.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs,
deadlineAtMs,
/** Minted per admission attempt: a queued call outlives one token TTL. */
@@ -1227,6 +1239,7 @@ async function writeWorkspaceFile({
overwrite,
workspace_id,
workspace_instance_id,
+ linked_worktrees,
codeApiBaseUrl,
executionProfile,
bridgeWorkerId,
@@ -1238,7 +1251,9 @@ async function writeWorkspaceFile({
}) {
return executeWorkspaceTool({
baseURL: codeApiBaseUrl,
+ linkedWorktrees: linked_worktrees,
maxQueueWaitMs,
+ codeApiMaxRetryWaitMs: req?.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs,
deadlineAtMs,
/** Minted per admission attempt: a queued call outlives one token TTL. */
@@ -1264,8 +1279,10 @@ async function editWorkspaceFile({
file_path,
edits,
expected_base_sha256,
+ matching,
workspace_id,
workspace_instance_id,
+ linked_worktrees,
codeApiBaseUrl,
executionProfile,
bridgeWorkerId,
@@ -1277,7 +1294,9 @@ async function editWorkspaceFile({
}) {
return executeWorkspaceTool({
baseURL: codeApiBaseUrl,
+ linkedWorktrees: linked_worktrees,
maxQueueWaitMs,
+ codeApiMaxRetryWaitMs: req?.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs,
deadlineAtMs,
/** Minted per admission attempt: a queued call outlives one token TTL. */
@@ -1293,6 +1312,7 @@ async function editWorkspaceFile({
path: file_path,
edits,
...(expected_base_sha256 ? { expectedBaseSha256: expected_base_sha256 } : {}),
+ matching,
},
...(signal ? { signal } : {}),
});
@@ -1302,8 +1322,10 @@ async function editWorkspaceFile({
async function previewWorkspaceEdit({
file_path,
edits,
+ matching,
workspace_id,
workspace_instance_id,
+ linked_worktrees,
codeApiBaseUrl,
executionProfile,
bridgeWorkerId,
@@ -1315,7 +1337,9 @@ async function previewWorkspaceEdit({
}) {
return executeWorkspaceTool({
baseURL: codeApiBaseUrl,
+ linkedWorktrees: linked_worktrees,
maxQueueWaitMs,
+ codeApiMaxRetryWaitMs: req?.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs,
deadlineAtMs,
/** Minted per admission attempt: a queued call outlives one token TTL. */
@@ -1330,6 +1354,7 @@ async function previewWorkspaceEdit({
...(workspace_instance_id ? { workspaceInstanceId: workspace_instance_id } : {}),
path: file_path,
edits,
+ matching,
},
...(signal ? { signal } : {}),
});
diff --git a/api/server/services/Files/Code/process.spec.js b/api/server/services/Files/Code/process.spec.js
index 875368196fd..c6969d7dc66 100644
--- a/api/server/services/Files/Code/process.spec.js
+++ b/api/server/services/Files/Code/process.spec.js
@@ -2033,6 +2033,28 @@ describe('Code Process', () => {
});
describe('readWorkspaceFile', () => {
+ it('forwards the configured rate-limit budget to the workspace transport', async () => {
+ const req = {
+ ...mockReq,
+ config: { ...mockReq.config, endpoints: { agents: { codeApiMaxRetryWaitMs: 0 } } },
+ };
+ mockExecuteWorkspaceTool.mockResolvedValueOnce({ operation: 'read_file' });
+
+ await readWorkspaceFile({
+ file_path: 'src/app.ts',
+ workspace_id: 'primary',
+ start_line: 1,
+ max_lines: 10,
+ codeApiBaseUrl: 'https://attached-code.example.com/v1',
+ executionProfile: 'stateful',
+ req,
+ });
+
+ expect(mockExecuteWorkspaceTool).toHaveBeenCalledWith(
+ expect.objectContaining({ codeApiMaxRetryWaitMs: 0 }),
+ );
+ });
+
it('forwards authenticated reads to the selected attached worker', async () => {
const controller = new AbortController();
const result = {
@@ -2083,6 +2105,7 @@ describe('Code Process', () => {
baseURL: 'https://attached-code.example.com/v1',
authHeaders: expect.any(Function),
maxQueueWaitMs: 0,
+ codeApiMaxRetryWaitMs: undefined,
maxRequestTimeoutMs: 125_000,
deadlineAtMs: 160_000,
request: {
@@ -2144,6 +2167,7 @@ describe('Code Process', () => {
baseURL: 'https://attached-code.example.com/v1',
authHeaders: expect.any(Function),
maxQueueWaitMs: 0,
+ codeApiMaxRetryWaitMs: undefined,
maxRequestTimeoutMs: undefined,
deadlineAtMs: undefined,
request: {
@@ -2175,6 +2199,7 @@ describe('Code Process', () => {
await expect(
listWorkspaceFiles({
workspace_id: 'primary',
+ linked_worktrees: true,
path: 'src',
after_path: 'src/app.ts',
max_results: 20,
@@ -2202,8 +2227,10 @@ describe('Code Process', () => {
expect(getCodeApiAuthHeaders).toHaveBeenNthCalledWith(2, mockReq, 'worker-user-1');
expect(mockExecuteWorkspaceTool).toHaveBeenCalledWith({
baseURL: 'https://attached-code.example.com/v1',
+ linkedWorktrees: true,
authHeaders: expect.any(Function),
maxQueueWaitMs: 0,
+ codeApiMaxRetryWaitMs: undefined,
maxRequestTimeoutMs: undefined,
deadlineAtMs: undefined,
request: {
@@ -2266,6 +2293,7 @@ describe('Code Process', () => {
baseURL: 'https://attached-code.example.com/v1',
authHeaders: expect.any(Function),
maxQueueWaitMs: 0,
+ codeApiMaxRetryWaitMs: undefined,
maxRequestTimeoutMs: undefined,
deadlineAtMs: undefined,
request: {
@@ -2330,6 +2358,32 @@ describe('Code Process', () => {
);
});
+ it('forwards negotiated matching and replaceAll on edits and previews', async () => {
+ const edits = [{ oldText: 'false', newText: 'true', replaceAll: true }];
+ mockExecuteWorkspaceTool.mockResolvedValue({});
+ const shared = {
+ file_path: 'src/app.ts',
+ edits,
+ matching: 'tolerant',
+ workspace_id: 'primary',
+ codeApiBaseUrl: 'https://attached-code.example.com/v1',
+ executionProfile: 'stateful',
+ bridgeWorkerId: 'worker-user-1',
+ req: mockReq,
+ };
+
+ await editWorkspaceFile(shared);
+ await previewWorkspaceEdit(shared);
+
+ for (const operation of ['edit_file', 'preview_edit']) {
+ expect(mockExecuteWorkspaceTool).toHaveBeenCalledWith(
+ expect.objectContaining({
+ request: expect.objectContaining({ operation, edits, matching: 'tolerant' }),
+ }),
+ );
+ }
+ });
+
it('forwards a non-mutating attached-workspace edit preview', async () => {
const result = {
protocolVersion: 1,
diff --git a/api/server/services/ToolService.js b/api/server/services/ToolService.js
index 53c36562b6d..5a85e7ff220 100644
--- a/api/server/services/ToolService.js
+++ b/api/server/services/ToolService.js
@@ -1614,6 +1614,7 @@ async function loadToolDefinitionsWrapper({
enabled: codeExecutionEnabled,
context: resolvedCodeExecutionContext,
principalId: JSON.stringify([getTenantId(), req.user.id]),
+ codeApiMaxRetryWaitMs: req.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
getAuthHeaders: (workerId) => getCodeApiAuthHeaders(req, workerId),
}),
};
@@ -1817,6 +1818,7 @@ async function loadAgentTools({
enabled: codeExecutionEnabled,
context: codeExecutionContext,
principalId: JSON.stringify([getTenantId(), req.user.id]),
+ codeApiMaxRetryWaitMs: req.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
getAuthHeaders: (workerId) => getCodeApiAuthHeaders(req, workerId),
});
const { loadedTools, toolContextMap, dynamicToolContextMap, primedCodeFiles } = await loadTools({
@@ -2342,6 +2344,7 @@ async function loadToolsForExecution({
baseUrl: codeExecutionContext.baseUrl,
workspaceId: codeExecutionContext.codeWorkspace.workspaceId,
workspaceInstanceId: codeExecutionContext.codeWorkspace.workspaceInstanceId,
+ linkedWorktrees: codeExecutionContext.codeWorkspace.linkedWorktrees,
environment: codeExecutionContext.codeWorkspace.environment,
gitIdentity: agent?.git_identity,
maxTimeoutMs: resolveAttachedWorkspaceCommandTimeoutMax(
@@ -2351,9 +2354,12 @@ async function loadToolsForExecution({
maxQueueWaitMs: resolveAttachedWorkspaceQueueWaitMs(
codeExecutionContext.codeEnvironmentConfigSchema,
),
+ codeApiMaxRetryWaitMs: req.config?.endpoints?.agents?.codeApiMaxRetryWaitMs,
maxRequestTimeoutMs: resolveAttachedWorkspaceRequestTimeoutMs(
codeExecutionContext.codeEnvironmentConfigSchema,
),
+ minCommandAdmissionMs:
+ codeExecutionContext.codeEnvironmentConfigSchema?.limits?.minCommandAdmissionMs,
})
: createBashExecutionTool({
authHeaders,
diff --git a/api/server/services/__tests__/ToolService.spec.js b/api/server/services/__tests__/ToolService.spec.js
index 3f323d98b49..aa7cdef494c 100644
--- a/api/server/services/__tests__/ToolService.spec.js
+++ b/api/server/services/__tests__/ToolService.spec.js
@@ -784,6 +784,43 @@ describe('ToolService - Action Capability Gating', () => {
});
});
+ it.each([true, false])(
+ 'passes the Code API retry limit to repository instructions (definitionsOnly=%s)',
+ async (definitionsOnly) => {
+ const capabilities = [
+ AgentCapabilities.tools,
+ AgentCapabilities.execute_code,
+ AgentCapabilities.stateful_code_sessions,
+ ];
+ const req = createMockReq(capabilities);
+ req.config.endpoints[EModelEndpoint.agents].codeApiMaxRetryWaitMs = 0;
+ mockGetEndpointsConfig.mockResolvedValue(createEndpointsConfig(capabilities));
+ mockResolveCodeExecutionContext.mockReturnValueOnce({
+ baseUrl: 'https://attached-code.example.com/v1',
+ codeSessionKey: 'attached-session',
+ executionProfile: 'stateful',
+ statefulSessions: true,
+ environmentType: 'attached',
+ environmentId: 'personal-machine',
+ });
+
+ const result = await loadAgentTools({
+ req,
+ res: {},
+ agent: {
+ id: 'attached-agent',
+ tools: [Tools.execute_code],
+ stateful_code_sessions: true,
+ },
+ definitionsOnly,
+ });
+
+ expect(result.repositoryInstructionSource).toEqual(
+ expect.objectContaining({ codeApiMaxRetryWaitMs: 0 }),
+ );
+ },
+ );
+
describe('isActionTool — cross-delimiter collision guard', () => {
it('should identify real action tools', () => {
expect(isActionTool(`get_weather${actionDelimiter}api_example_com`)).toBe(true);
@@ -3006,6 +3043,7 @@ describe('ToolService - Action Capability Gating', () => {
AgentCapabilities.stateful_code_sessions,
];
const req = createMockReq(capabilities);
+ req.config.endpoints[EModelEndpoint.agents].codeApiMaxRetryWaitMs = 0;
req.body = {
codeWorkspaces: [{ environmentId: 'personal-machine', workspaceId: 'project-a' }],
};
@@ -3018,7 +3056,14 @@ describe('ToolService - Action Capability Gating', () => {
environmentType: 'attached',
environmentId: 'personal-machine',
bridgeWorkerId: 'worker-abc',
- codeEnvironmentConfigSchema: { limits: { maxCommandTimeoutMs: 120000, maxQueueWaitMs: 0 } },
+ codeEnvironmentConfigSchema: {
+ limits: {
+ maxCommandTimeoutMs: 80_000,
+ maxQueueWaitMs: 0,
+ maxRequestTimeoutMs: 90_000,
+ minCommandAdmissionMs: 15_000,
+ },
+ },
});
const toolRegistry = new Map([
[AgentConstants.BASH_TOOL, { name: AgentConstants.BASH_TOOL }],
@@ -3044,8 +3089,11 @@ describe('ToolService - Action Capability Gating', () => {
baseUrl: 'http://attached-code.test/v1',
workspaceId: 'project-a',
gitIdentity: { name: 'LibreChat Agent', email: 'agent@example.com' },
- maxTimeoutMs: 120000,
+ maxTimeoutMs: 65_000,
maxQueueWaitMs: 0,
+ codeApiMaxRetryWaitMs: 0,
+ maxRequestTimeoutMs: 90_000,
+ minCommandAdmissionMs: 15_000,
});
expect(mockResolveCodeExecutionWorkspaceContext).toHaveBeenCalledWith(
expect.objectContaining({ requestedSelections: req.body.codeWorkspaces }),
diff --git a/api/test/server/middleware/checkBan.test.js b/api/test/server/middleware/checkBan.test.js
index 39775c389a4..86024bd82d9 100644
--- a/api/test/server/middleware/checkBan.test.js
+++ b/api/test/server/middleware/checkBan.test.js
@@ -40,6 +40,7 @@ jest.mock('@librechat/api', () => ({
return false;
},
keyvMongo: {},
+ getBanIp: jest.requireActual('../../../../packages/api/src/middleware/ban').getBanIp,
removePorts: jest.fn((req) => req.ip),
redirectToAuthFailure: (res, { clientDomain, authFailedError }) =>
res.redirect(`${clientDomain}/login?redirect=false&error=${authFailedError}`),
@@ -141,6 +142,93 @@ describe('checkBan middleware', () => {
});
});
+ describe.each([false, true])('verified trigger requests (USE_REDIS=%s)', (useRedis) => {
+ beforeEach(() => {
+ process.env.USE_REDIS = String(useRedis);
+ });
+
+ const userKey = useRedis ? 'ban_cache:user:user123' : 'user123';
+ const ipKey = useRedis ? 'ban_cache:ip:127.0.0.1' : '127.0.0.1';
+ const createTriggerReq = () => createReq({ ip: '127.0.0.1', _isAgentTrigger: true });
+
+ it('ignores cached and persistent loopback bans without querying either IP key', async () => {
+ const ban = { expiresAt: Date.now() + 60000 };
+ mockBanCacheGet.mockImplementation(async (key) => (key === ipKey ? ban : undefined));
+ mockBanLogsGet.mockImplementation(async (key) => (key === '127.0.0.1' ? ban : undefined));
+ const req = createTriggerReq();
+ const next = jest.fn();
+
+ await checkBan(req, createRes(), next);
+
+ expect(next).toHaveBeenCalledWith();
+ expect(req.banned).toBeUndefined();
+ expect(req.ip).toBe('127.0.0.1');
+ expect(mockBanCacheGet.mock.calls).toEqual([[userKey]]);
+ expect(mockBanLogsGet.mock.calls).toEqual([['user123']]);
+ expect(mockBanCacheSet).not.toHaveBeenCalled();
+ });
+
+ it('still rejects a cached user ban', async () => {
+ mockBanCacheGet.mockResolvedValueOnce({ expiresAt: Date.now() + 60000 });
+ const res = createRes();
+ const next = jest.fn();
+
+ await checkBan(createTriggerReq(), res, next);
+
+ expect(next).not.toHaveBeenCalled();
+ expect(res.status).toHaveBeenCalledWith(403);
+ expect(mockBanCacheGet.mock.calls).toEqual([[userKey]]);
+ expect(mockBanLogsGet).not.toHaveBeenCalled();
+ });
+
+ it('caches a persistent user ban only under the user key', async () => {
+ const ban = { expiresAt: Date.now() + 60000 };
+ mockBanLogsGet.mockResolvedValueOnce(ban);
+ const res = createRes();
+ const next = jest.fn();
+
+ await checkBan(createTriggerReq(), res, next);
+
+ expect(next).not.toHaveBeenCalled();
+ expect(res.status).toHaveBeenCalledWith(403);
+ expect(mockBanCacheSet).toHaveBeenCalledTimes(1);
+ expect(mockBanCacheSet).toHaveBeenCalledWith(userKey, ban, expect.any(Number));
+ });
+
+ it('cleans up an expired user ban without touching the transport IP', async () => {
+ mockBanLogsGet.mockResolvedValueOnce({ expiresAt: Date.now() - 1000 });
+ const next = jest.fn();
+
+ await checkBan(createTriggerReq(), createRes(), next);
+
+ expect(next).toHaveBeenCalledWith();
+ expect(mockBanLogsDelete.mock.calls).toEqual([['user123']]);
+ expect(mockBanCacheSet).not.toHaveBeenCalled();
+ });
+
+ it('does not treat a trigger header on an ordinary request as an exemption', async () => {
+ mockBanCacheGet.mockResolvedValueOnce({ expiresAt: Date.now() + 60000 });
+ const req = createReq({
+ ip: '127.0.0.1',
+ headers: { 'x-lc-agent-trigger': '1' },
+ _isAgentTrigger: false,
+ });
+ const res = createRes();
+ const next = jest.fn();
+
+ await checkBan(req, res, next);
+
+ expect(next).not.toHaveBeenCalled();
+ expect(res.status).toHaveBeenCalledWith(403);
+ expect(mockBanCacheGet).toHaveBeenCalledWith(ipKey);
+ });
+
+ afterEach(() => {
+ mockBanCacheGet.mockReset().mockResolvedValue(undefined);
+ mockBanLogsGet.mockReset().mockResolvedValue(undefined);
+ });
+ });
+
describe('cache hit path', () => {
it('returns 403 when IP ban is cached', async () => {
mockBanCacheGet.mockResolvedValueOnce({ expiresAt: Date.now() + 60000 });
diff --git a/bun.lock b/bun.lock
index 125c855c98d..8eb42917e3d 100644
--- a/bun.lock
+++ b/bun.lock
@@ -39,7 +39,7 @@
},
"api": {
"name": "@librechat/backend",
- "version": "0.8.8-rc4",
+ "version": "0.8.8",
"dependencies": {
"@anthropic-ai/vertex-sdk": "^0.16.0",
"@aws-sdk/client-bedrock-runtime": "^3.1013.0",
@@ -151,7 +151,7 @@
},
"client": {
"name": "@librechat/frontend",
- "version": "0.8.8-rc4",
+ "version": "0.8.8",
"dependencies": {
"@ariakit/react": "^0.4.29",
"@ariakit/react-components": "^0.1.2",
@@ -294,7 +294,7 @@
},
"packages/api": {
"name": "@librechat/api",
- "version": "1.7.49",
+ "version": "1.7.52",
"dependencies": {
"@langchain/langgraph-checkpoint": "^1.1.2",
"@langchain/langgraph-checkpoint-mongodb": "^1.4.0",
@@ -407,7 +407,7 @@
},
"packages/client": {
"name": "@librechat/client",
- "version": "0.4.79",
+ "version": "0.4.82",
"devDependencies": {
"@babel/core": "^7.28.5",
"@babel/preset-env": "^7.29.5",
@@ -490,7 +490,7 @@
},
"packages/data-provider": {
"name": "librechat-data-provider",
- "version": "0.8.524",
+ "version": "0.8.527",
"dependencies": {
"axios": "^1.16.0",
"dayjs": "^1.11.13",
@@ -525,7 +525,7 @@
},
"packages/data-schemas": {
"name": "@librechat/data-schemas",
- "version": "0.0.71",
+ "version": "0.0.74",
"dependencies": {
"mdast-util-directive": "^3.0.0",
"mdast-util-from-markdown": "^2.0.1",
diff --git a/client/index.html b/client/index.html
index dc96c252d30..9fe3086ed9f 100644
--- a/client/index.html
+++ b/client/index.html
@@ -28,7 +28,12 @@
}