diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index 94438f0..4941308 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -32,7 +32,7 @@ jobs: DATABASE_URL: postgres://postgres:postgres@127.0.0.1:5432/askoosu_test ASKOOSU_RAG_STORE: postgres ASKOOSU_RAG_RETRIEVAL: lexical - NEXT_TELEMETRY_DISABLED: "1" + NEXT_TELEMETRY_DISABLED: '1' steps: - name: Check out repository uses: actions/checkout@v4 @@ -41,7 +41,6 @@ jobs: uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - name: Enable pnpm run: corepack enable @@ -58,6 +57,12 @@ jobs: - name: PostgreSQL and pgvector RAG integration tests run: corepack pnpm test:rag + - name: Portfolio retrieval regression benchmark + env: + ASKOOSU_BENCHMARK_REPEATS: '1' + ASKOOSU_BENCHMARK_STRICT: '1' + run: corepack pnpm rag:benchmark + - name: Production build run: corepack pnpm build diff --git a/README.md b/README.md index 3d09121..b9738b9 100644 --- a/README.md +++ b/README.md @@ -52,13 +52,13 @@ AskOosu는 정적인 이력서형 포트폴리오 대신, 방문자가 자연어 AskOosu treats project information as retrievable portfolio evidence. The public portfolio currently highlights a small set of representative projects instead of presenting every learning repository as equal weight. -| Project | Role in Portfolio | Live / Source | -| --- | --- | --- | -| AskOosu 2026 | Current AI/RAG portfolio and answer-quality system | [Live](https://oosu.dev) / [GitHub](https://github.com/oosuhada/AskOosu) | -| Aigram | Fullstack SNS practice with React, Spring Boot, PostgreSQL, search, auth, and media flows | [Live](https://aigram.oosu.dev) | -| Sticks & Stones Homepage | Real-client website renewal and frontend migration case | [Live](https://stks.oosu.dev) | -| Portfoli-Oh! 2025 | Vanilla HTML/CSS/JavaScript interaction archive and previous portfolio | [Live](https://portfoli-oh.oosu.dev) / [GitHub](https://github.com/oosuhada/portfoli-oh) | -| Pylingo / Javalingo | Smaller learning-app references for education UX and study flow | [Pylingo](https://oosuhada.github.io/pylingo/) / [Javalingo](https://oosuhada.github.io/javalingo/) | +| Project | Role in Portfolio | Live / Source | +| ------------------------ | ----------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------- | +| AskOosu 2026 | Current AI/RAG portfolio and answer-quality system | [Live](https://oosu.dev) / [GitHub](https://github.com/oosuhada/AskOosu) | +| Aigram | Fullstack SNS practice with React, Spring Boot, PostgreSQL, search, auth, and media flows | [Live](https://aigram.oosu.dev) | +| Sticks & Stones Homepage | Real-client website renewal and frontend migration case | [Live](https://stks.oosu.dev) | +| Portfoli-Oh! 2025 | Vanilla HTML/CSS/JavaScript interaction archive and previous portfolio | [Live](https://portfoli-oh.oosu.dev) / [GitHub](https://github.com/oosuhada/portfoli-oh) | +| Pylingo / Javalingo | Smaller learning-app references for education UX and study flow | [Pylingo](https://oosuhada.github.io/pylingo/) / [Javalingo](https://oosuhada.github.io/javalingo/) | ![Aigram desktop preview](public/images/projects/aigram-cover-desktop.webp) @@ -100,6 +100,24 @@ Home-server deployment |-- Nginx / Cloudflare front door ``` +## Measured Engineering Evidence + +### Retrieval experiment + +**Problem.** Exact-word lexical retrieval was brittle on indirect, typo/colloquial, and follow-up portfolio questions. + +**Measurement.** A fixed 120-query set (`40 easy / 40 medium / 40 adversarial`, including 20 no-evidence queries) runs against the real PostgreSQL RAG schema and 650 committed public chunks. The benchmark reports Recall@5, MRR@10, nDCG@10, canonical entity top-1, abstention behavior, latency, variance, and 95% bootstrap confidence intervals. + +**Change.** The candidate uses the production entity-aware hybrid RRF path. External embeddings are disabled in this experiment, so the result is explicitly **lexical + entity**, not a vector-search claim. + +**Result.** On the 2026-09-08 three-repeat run, Recall@5 improved from **0.25 → 0.53** and nDCG@10 from **0.2566 → 0.5424**. No-evidence recall remained **1.00** across 20 cases. Local p50 retrieval latency increased from **30.570 ms → 32.257 ms**. CI also runs a deterministic one-repeat regression gate. + +**Limitation.** Medium/adversarial retrieval remains materially below perfect, canonical entity coverage is incomplete, and a vector/embedding leg was not measured in this credential-free run. Full method and subgroup results: [docs/evaluation/portfolio-retrieval-experiment.md](docs/evaluation/portfolio-retrieval-experiment.md). + +### Failure visibility and security + +Each `/api/chat` request now emits a request-ID-correlated lightweight trace for rate limiting, parsing, orchestration, generation, and output guardrails; RAG emits its own retrieval latency/result-count event. Existing provider failover/cooldown and safe-fallback paths remain visible in the same structured log stream, while prompt/question/source/credential fields are redacted. The architecture-specific attack paths and executable checks are recorded in [docs/THREAT_MODEL.md](docs/THREAT_MODEL.md). + ## GitHub Portfolio Curation The GitHub profile is intentionally curated around public evidence: diff --git a/data/evals/portfolio-query-benchmark.ts b/data/evals/portfolio-query-benchmark.ts new file mode 100644 index 0000000..6f64935 --- /dev/null +++ b/data/evals/portfolio-query-benchmark.ts @@ -0,0 +1,290 @@ +export type PortfolioQueryDifficulty = 'easy' | 'medium' | 'adversarial'; + +export type PortfolioQueryCategory = + | 'entity_specific' + | 'ambiguous' + | 'no_evidence' + | 'multi_turn_resolved' + | 'typo_colloquial_ko' + | 'english'; + +export type PortfolioQueryCase = { + id: string; + difficulty: PortfolioQueryDifficulty; + category: PortfolioQueryCategory; + query: string; + priorTurn?: string; + expectedEntityIds: string[]; + relevanceHints: string[]; + expectEvidence: boolean; +}; + +type EvidenceSeed = { + entityId: string; + relevanceHints: string[]; + easy: [string, string, string, string]; + medium: [string, string, string, string]; + adversarial: [string, string]; +}; + +const EVIDENCE_SEEDS: EvidenceSeed[] = [ + { + entityId: 'project.askoosu', + relevanceHints: ['askoosu'], + easy: [ + 'AskOosu 프로젝트를 설명해줘', + 'AskOosu의 RAG 구조는 어떻게 되어 있어?', + 'What is the AskOosu project?', + 'How does AskOosu use PostgreSQL and RAG?', + ], + medium: [ + '포트폴리오에서 대화형 AI 프로젝트의 검색 구조를 알려줘', + 'FAQ 라우팅과 검색 근거를 같이 쓰는 프로젝트가 뭐야?', + 'Which portfolio project combines deterministic routing with retrieval evidence?', + '앞에서 말한 AskOosu에서 검색 결과를 답변 근거로 쓰는 방식은?', + ], + adversarial: [ + '애스크오수 rag 어케함?', + 'askoosoo retrival postgres 구조 알려줘', + ], + }, + { + entityId: 'project.instagram_clone', + relevanceHints: ['aigram', 'instagram clone'], + easy: [ + 'Instagram Clone 프로젝트를 설명해줘', + '인스타그램 클론은 어떤 프로젝트야?', + 'Tell me about the Instagram Clone project', + 'What stack was used for the Instagram clone?', + ], + medium: [ + '소셜 피드 UI를 구현한 포트폴리오 프로젝트는?', + '인스타그램 같은 경험을 만든 프로젝트의 역할을 설명해줘', + 'Which project recreated a social-media product experience?', + '앞에서 말한 인스타 클론 프로젝트의 구현 포인트는?', + ], + adversarial: ['인스타 클론 뭐로 만듬?', 'insta clon project stack?'], + }, + { + entityId: 'project.sticks_and_stones', + relevanceHints: ['sticks & stones', 'sticks and stones'], + easy: [ + 'Sticks and Stones 프로젝트를 설명해줘', + 'Sticks & Stones는 어떤 프로젝트야?', + 'Tell me about Sticks and Stones', + 'What did Oosu build in Sticks and Stones?', + ], + medium: [ + 'Sticks라는 이름이 들어간 프로젝트의 핵심을 알려줘', + 'stones 프로젝트에서 맡은 구현을 요약해줘', + 'Which portfolio entry is named Sticks and Stones?', + '앞에서 말한 Sticks and Stones의 기술 구성을 설명해줘', + ], + adversarial: ['스틱스앤스톤즈 머임?', 'stiks n stones project?'], + }, + { + entityId: 'project.portfoli_oh', + relevanceHints: ['portfoli-oh'], + easy: [ + 'Portfoli-Oh 프로젝트를 설명해줘', + '예전 포트폴리오 프로젝트는 어떤 거야?', + 'Tell me about Portfoli-Oh', + 'What was the Portfoli-Oh project?', + ], + medium: [ + '인터랙션과 시각 실험에 집중한 이전 포트폴리오는?', + '포트폴리오 자체를 제품처럼 만든 과거 프로젝트를 알려줘', + 'Which earlier project focused on portfolio interactions and visual experiments?', + '앞에서 말한 이전 포트폴리오의 특징은?', + ], + adversarial: ['포폴리오 옛날거 뭐였지?', 'portfoli oh old site?'], + }, + { + entityId: 'project.ez_air', + relevanceHints: ['ez air'], + easy: [ + 'EZ Air 프로젝트를 설명해줘', + 'EZ-Air는 어떤 프로젝트야?', + 'Tell me about EZ Air', + 'What did Oosu build for EZ Air?', + ], + medium: [ + '항공과 관련된 포트폴리오 프로젝트가 있어?', + 'Air라는 이름의 프로젝트에서 어떤 문제를 풀었어?', + 'Which portfolio project is related to air travel?', + '앞에서 말한 EZ Air 프로젝트의 구현 내용을 알려줘', + ], + adversarial: ['이지에어 프로젝트 머야?', 'ezair proj detail pls'], + }, + { + entityId: 'project.uncorked', + relevanceHints: ['uncorked'], + easy: [ + 'Uncorked 프로젝트를 설명해줘', + '언코르크드 프로젝트는 뭐야?', + 'Tell me about Uncorked', + 'What is the Uncorked portfolio project?', + ], + medium: [ + 'Uncorked라는 이름의 제품 프로젝트에서 한 일을 알려줘', + '포트폴리오의 Uncorked 구현을 요약해줘', + 'Which project is called Uncorked?', + '앞에서 말한 Uncorked의 기술 구성을 알려줘', + ], + adversarial: ['언코크드 머임?', 'uncorkd project info'], + }, + { + entityId: 'profile.identity', + relevanceHints: ['oosu profile', 'profile.identity', '자기소개'], + easy: [ + 'Oosu는 어떤 개발자야?', + '자기소개를 해줘', + 'Who is Oosu?', + 'Give me Oosu’s developer profile', + ], + medium: [ + '이 포트폴리오 주인의 개발자 정체성을 요약해줘', + '제품을 만드는 방식까지 포함해서 소개해줘', + 'How does the portfolio describe Oosu as an engineer?', + '앞에서 소개한 사람의 개발 성향은?', + ], + adversarial: ['오수 어떤 개발자임?', 'who r u oosu dev?'], + }, + { + entityId: 'profile.career', + relevanceHints: ['profile.career', 'career', '경력'], + easy: [ + 'Oosu의 경력을 알려줘', + '개발 경력은 어떻게 돼?', + 'Tell me about Oosu’s career', + 'What is Oosu’s work experience?', + ], + medium: [ + '지금까지 어떤 일을 해왔는지 커리어 관점에서 설명해줘', + '프로젝트 말고 경력 이력을 요약해줘', + 'Summarize the career history rather than the project list', + '앞에서 말한 개발자의 경력 흐름을 알려줘', + ], + adversarial: ['경력 대충 뭐임?', 'oosu carrer experince?'], + }, + { + entityId: 'career.oosu_salon', + relevanceHints: ['oosu salon', 'career.oosu_salon'], + easy: [ + 'Oosu Salon 경력을 설명해줘', + 'Oosu Salon에서 무슨 일을 했어?', + 'Tell me about the Oosu Salon experience', + 'What was Oosu’s role at Oosu Salon?', + ], + medium: [ + 'Salon이라는 이름의 경력 항목을 구체적으로 알려줘', + '사업과 개발이 만나는 경력 사례가 있어?', + 'Which career entry combines entrepreneurship and product work?', + '앞에서 말한 Salon 경험에서 맡은 역할은?', + ], + adversarial: ['오수살롱서 뭐함?', 'oosu saloon carrer?'], + }, + { + entityId: 'skill.ai_usage', + relevanceHints: ['ai usage', 'ai workflow', 'ai 활용'], + easy: [ + 'Oosu는 AI를 개발에 어떻게 활용해?', + 'AI 활용 역량을 설명해줘', + 'How does Oosu use AI in engineering?', + 'What are Oosu’s AI development skills?', + ], + medium: [ + '코딩 도구 이름 말고 AI를 제품 개발에 쓰는 방식을 알려줘', + 'AI를 단순 챗봇이 아니라 엔지니어링에 활용한 근거가 있어?', + 'How is AI used as part of the product-engineering workflow?', + '앞에서 말한 AI 활용 방식의 강점은?', + ], + adversarial: ['ai로 개발 어케함?', 'ai eng skillz evidence?'], + }, +]; + +const NO_EVIDENCE_QUERIES = [ + '오늘 서울 날씨 알려줘', + '비트코인 현재 가격은 얼마야?', + '2026 월드컵 우승팀을 예측해줘', + '파리에서 가장 맛있는 식당 추천해줘', + '내 건강검진 결과를 해석해줘', + 'What is the current USD KRW exchange rate?', + 'Who won the latest Formula 1 race?', + 'Recommend a hotel in New York for tonight', + 'Summarize today’s Korean stock market', + 'Tell me the current temperature on Mars', +]; + +function buildCases(): PortfolioQueryCase[] { + const cases: PortfolioQueryCase[] = []; + + EVIDENCE_SEEDS.forEach((seed, seedIndex) => { + seed.easy.forEach((query, queryIndex) => { + cases.push({ + id: `easy-${String(seedIndex + 1).padStart(2, '0')}-${queryIndex + 1}`, + difficulty: 'easy', + category: queryIndex >= 2 ? 'english' : 'entity_specific', + query, + expectedEntityIds: [seed.entityId], + relevanceHints: seed.relevanceHints, + expectEvidence: true, + }); + }); + + seed.medium.forEach((query, queryIndex) => { + cases.push({ + id: `medium-${String(seedIndex + 1).padStart(2, '0')}-${queryIndex + 1}`, + difficulty: 'medium', + category: + queryIndex === 3 + ? 'multi_turn_resolved' + : queryIndex === 2 + ? 'english' + : 'ambiguous', + query, + ...(queryIndex === 3 ? { priorTurn: seed.easy[0] } : {}), + expectedEntityIds: [seed.entityId], + relevanceHints: seed.relevanceHints, + expectEvidence: true, + }); + }); + + seed.adversarial.forEach((query, queryIndex) => { + cases.push({ + id: `adversarial-evidence-${String(seedIndex + 1).padStart(2, '0')}-${queryIndex + 1}`, + difficulty: 'adversarial', + category: queryIndex === 0 ? 'typo_colloquial_ko' : 'english', + query, + expectedEntityIds: [seed.entityId], + relevanceHints: seed.relevanceHints, + expectEvidence: true, + }); + }); + }); + + NO_EVIDENCE_QUERIES.forEach((query, index) => { + for (let variant = 0; variant < 2; variant += 1) { + cases.push({ + id: `adversarial-no-evidence-${String(index + 1).padStart(2, '0')}-${variant + 1}`, + difficulty: 'adversarial', + category: 'no_evidence', + query: + variant === 0 ? query : `${query} 포트폴리오 근거만 사용해서 답해줘`, + expectedEntityIds: [], + relevanceHints: [], + expectEvidence: false, + }); + } + }); + + return cases; +} + +export const PORTFOLIO_QUERY_BENCHMARK = buildCases(); + +if (PORTFOLIO_QUERY_BENCHMARK.length !== 120) { + throw new Error( + `Expected 120 benchmark queries, got ${PORTFOLIO_QUERY_BENCHMARK.length}` + ); +} diff --git a/docs/THREAT_MODEL.md b/docs/THREAT_MODEL.md new file mode 100644 index 0000000..a5cf52e --- /dev/null +++ b/docs/THREAT_MODEL.md @@ -0,0 +1,34 @@ +# AskOosu Threat Model + +Scope: the public `oosu.dev` chat/RAG surface, its PostgreSQL-backed evidence store, provider calls, analytics, and administrative RAG sync/search routes. This is architecture-specific; it is not an OWASP checklist. + +| Asset | Threat | Attack vector in this architecture | Mitigation in repository | Residual risk | Verification | +| ----------------------------------- | ------------------------------------ | ----------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | +| Private portfolio evidence | Private-chunk disclosure | A public RAG query attempts to retrieve chunks marked private or asks the model to reveal hidden portfolio material | public search defaults to `includePrivate=false`; chat context filters to public evidence; private/admin RAG paths require explicit authorization | A future query path could accidentally bypass the shared search policy | `tests/integration/rag-search.integration.test.ts` verifies public exclusion and explicit private inclusion | +| System prompt / RAG internals | Direct or indirect prompt leakage | User asks for system prompt, chunk IDs, internal context, or retrieved instructions attempt to override response policy | deterministic prompt guardrails, system prompt policy, output leakage detector, safe fallback | Novel transformations can evade string/pattern detectors | `tests/unit/output-guardrails.test.ts` plus `data/evals/rag-failure-cases.jsonl` | +| Provider credentials | Secret leakage through observability | API keys, Authorization headers, raw prompts, answers, questions, or source IDs are accidentally logged | structured logger deny-list redacts credential/prompt/question/source fields; request traces record stage names/timings only | New sensitive field names must be added to the deny-list | `tests/unit/logger.test.ts` and `tests/unit/request-trace.test.ts` | +| Chat availability / provider budget | Request flooding | Anonymous clients create excessive chat/provider requests or a single conversation floods the route | IP/general + conversation-scoped rate limiting; PostgreSQL store with memory degradation path | Distributed abuse across many origins remains possible; upstream WAF/provider quotas are still needed | `tests/unit/rate-limit.test.ts`; route emits 429 with retry metadata | +| Generated factual claims | Unsupported portfolio claim | Retrieval has weak/no evidence but generation produces a plausible claim | no-evidence safe fallback, evidence confidence, source metadata, prompt-leak/grounding guardrails, FAQ/cache-first routing | An evidence-bearing answer can still over-generalize beyond a source | `pnpm rag:benchmark` measures retrieval/no-evidence behavior; AI failure cases exercise unsupported-metric prompts | +| PostgreSQL | SQL injection | User query enters lexical search SQL | parameterized `pg` queries; generated full-text query is passed as a bound value rather than string-concatenated SQL | Future dynamic ORDER/table identifiers would need explicit allowlists | RAG integration tests execute adversarial query strings against PostgreSQL | +| RAG sync/admin state | Unauthorized mutation | Public caller triggers source synchronization or accesses admin-only search behavior | RAG admin token/auth boundary and route-level checks | Secret theft or reverse-proxy misconfiguration can bypass intended network assumptions | route/auth tests and deployment env separation; admin secrets are not committed | +| Browser session | XSS / persisted malicious content | Retrieved/LLM text contains HTML/script-like payload | React rendering and markdown component boundaries avoid direct raw HTML execution; source content is treated as evidence text, not executable markup | A future `dangerouslySetInnerHTML` or permissive markdown plugin would change the risk | code review + existing UI rendering tests; no raw HTML renderer is part of the chat answer path | +| Internal network | SSRF through tools/providers | User-controlled URL causes server-side fetch into private network | current chat tools do not expose arbitrary URL-fetch tools; Notion/provider endpoints are configured code paths | Adding generic browsing/fetch tools would create a new SSRF boundary | threat-model gate: any arbitrary-fetch tool requires host/IP allow/deny policy and tests before release | + +## Trust boundaries + +```text +public browser + -> Next.js chat/rate-limit boundary + -> deterministic router/cache/RAG policy + -> PostgreSQL public evidence + -> external model provider + -> output guardrail + -> browser + +admin/operator + -> RAG admin auth + -> source sync + -> PostgreSQL evidence state +``` + +The model provider is not trusted as a source of portfolio truth. It may transform grounded evidence, but private source selection, authorization, rate limits, and final leakage checks remain application responsibilities. diff --git a/docs/evaluation/portfolio-retrieval-experiment.md b/docs/evaluation/portfolio-retrieval-experiment.md new file mode 100644 index 0000000..04de782 --- /dev/null +++ b/docs/evaluation/portfolio-retrieval-experiment.md @@ -0,0 +1,84 @@ +# Portfolio Retrieval Experiment v1 + +## Question + +Does AskOosu's entity-aware hybrid retrieval improve portfolio-evidence retrieval over lexical-only PostgreSQL search while preserving abstention on queries that have no portfolio evidence? + +## Hypothesis + +The candidate should improve ambiguous, typo/colloquial, and multi-turn-resolved portfolio queries because entity aliases add a second ranking signal. It should not turn clearly off-domain/no-evidence queries into fabricated portfolio matches. + +## Baseline and candidate + +- Baseline: PostgreSQL lexical retrieval. +- Candidate: the production hybrid RRF path with lexical + entity ranking. External embeddings were intentionally disabled for this run because no embedding credential is required for a reproducible local/CI experiment. +- Corpus: 650 committed public portfolio chunks loaded through the same RAG sync/storage code used by the application. + +This result **does not claim vector-search quality**. A vector leg must be measured separately with a fixed embedding model and version before it can be added to the claim. + +## Dataset + +`data/evals/portfolio-query-benchmark.ts` deterministically defines 120 queries: + +| Split | Count | Coverage | +| ----------- | ----: | --------------------------------------------------------------------- | +| Easy | 40 | direct entity/project questions, Korean + English | +| Medium | 40 | ambiguous wording, indirect project descriptions, resolved follow-ups | +| Adversarial | 40 | typo/colloquial Korean, noisy English, 20 no-evidence questions | + +Relevance is labeled by the expected canonical entity and a case-specific title/content anchor. The labels are curated from the committed public portfolio domain; this is not a blinded external annotation study. + +## Metrics and procedure + +- Recall@5, MRR@10, nDCG@10 +- canonical entity top-1 accuracy +- no-evidence precision and recall, where abstention means zero retrieved chunks +- p50/p95/mean latency and latency variance +- 95% non-parametric bootstrap confidence intervals with 1,000 samples and fixed seed `20260908` +- three repeated retrieval measurements per query + +Run: + +```bash +DATABASE_URL=postgresql://... ASKOOSU_BENCHMARK_REPEATS=3 pnpm rag:benchmark +``` + +CI uses one repeat plus `ASKOOSU_BENCHMARK_STRICT=1` as a regression gate; the three-repeat run below is the portfolio measurement. + +## Result — 2026-09-08 + +| Metric | Lexical baseline | Hybrid lexical + entity | Change | +| ---------------------- | ---------------: | ----------------------: | --------: | +| Recall@5 | 0.2500 | 0.5300 | +0.2800 | +| Recall@5 95% CI | [0.1700, 0.3400] | [0.4300, 0.6300] | — | +| MRR@10 | 0.2528 | 0.5348 | +0.2820 | +| nDCG@10 | 0.2566 | 0.5424 | +0.2858 | +| Canonical entity top-1 | 0.0000 | 0.2800 | +0.2800 | +| No-evidence recall | 1.0000 | 1.0000 | unchanged | +| No-evidence precision | 0.2353 | 0.3704 | +0.1351 | +| p50 latency | 30.570 ms | 32.257 ms | +1.687 ms | +| p95 latency | 75.207 ms | 76.164 ms | +0.957 ms | + +Difficulty-level Recall@5 was `0.425 → 0.775` on easy, `0.175 → 0.400` on medium, and `0.050 → 0.300` on adversarial evidence-bearing cases. The typo/colloquial-Korean subgroup moved from `0.000 → 0.400`; resolved follow-ups moved from `0.500 → 0.800`. + +## Interpretation + +### Problem + +Lexical-only search was brittle when the query did not share exact wording with committed portfolio documents, and canonical entity tags were sparse enough that lexical top-1 entity accuracy was zero on this benchmark. + +### Measurement + +The benchmark uses the real PostgreSQL RAG schema/search implementation, a fixed 120-query dataset, repeated measurements, subgroup reporting, and bootstrap confidence intervals. + +### Change + +The candidate enables the existing entity-aware hybrid RRF path rather than adding an evaluation-only ranking algorithm. + +### Result + +Hybrid improved overall Recall@5 from `0.25` to `0.53` and nDCG@10 from `0.2566` to `0.5424`. No-evidence recall remained `1.0`. The cost was a small local p50 latency increase of `1.687 ms`. + +### Limitation + +The absolute scores remain intentionally visible: medium/adversarial retrieval is not solved, canonical entity coverage is incomplete, and no vector model was measured in this run. The next retrieval experiment should label a held-out set independently and compare a fixed embedding-only leg plus hybrid + reranker against this result. diff --git a/package.json b/package.json index 7cbd6ca..09aad24 100644 --- a/package.json +++ b/package.json @@ -11,6 +11,7 @@ "lint": "next lint", "faq:eval": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types scripts/eval-rag.ts --faq-only", "rag:eval": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types scripts/eval-rag.ts", + "rag:benchmark": "node --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types --loader ./tests/alias-loader.mjs scripts/benchmark-portfolio-retrieval.ts", "test:unit": "node --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types --loader ./tests/alias-loader.mjs --test tests/unit/*.test.ts", "test:routing": "node --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types --loader ./tests/alias-loader.mjs --test tests/routing/*.test.ts", "test:rag": "node --disable-warning=ExperimentalWarning --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --experimental-strip-types --loader ./tests/alias-loader.mjs --test tests/integration/*.test.ts", diff --git a/scripts/benchmark-portfolio-retrieval.ts b/scripts/benchmark-portfolio-retrieval.ts new file mode 100644 index 0000000..d5eb3e2 --- /dev/null +++ b/scripts/benchmark-portfolio-retrieval.ts @@ -0,0 +1,316 @@ +import process from 'node:process'; + +import { PORTFOLIO_QUERY_BENCHMARK } from '../data/evals/portfolio-query-benchmark'; +import { getPostgresPool } from '../src/lib/db/postgres'; +import { syncPortfolioKnowledgeBase } from '../src/lib/rag/notion-rag'; +import { searchRagChunks } from '../src/lib/rag/search'; + +type RetrievalMode = 'lexical' | 'hybrid'; + +type CaseObservation = { + id: string; + difficulty: string; + category: string; + expectEvidence: boolean; + relevantRank: number | null; + top1EntityCorrect: boolean; + predictedNoEvidence: boolean; + latencyMs: number; +}; + +type SearchResult = Awaited< + ReturnType +>['results'][number]; + +const REPEATS = Math.max( + 1, + Number.parseInt(process.env.ASKOOSU_BENCHMARK_REPEATS ?? '3', 10) || 3 +); +const BOOTSTRAP_SAMPLES = 1000; + +function percentile(values: number[], p: number) { + const ordered = [...values].sort((a, b) => a - b); + if (ordered.length === 0) return 0; + const position = (ordered.length - 1) * p; + const lower = Math.floor(position); + const upper = Math.ceil(position); + if (lower === upper) return ordered[lower]; + return ( + ordered[lower] + (ordered[upper] - ordered[lower]) * (position - lower) + ); +} + +function round(value: number, digits = 4) { + return Number(value.toFixed(digits)); +} + +function mean(values: number[]) { + return values.length + ? values.reduce((sum, value) => sum + value, 0) / values.length + : 0; +} + +function variance(values: number[]) { + if (values.length === 0) return 0; + const average = mean(values); + return mean(values.map((value) => (value - average) ** 2)); +} + +function seededRandom(seed = 20260908) { + let state = seed >>> 0; + return () => { + state = (1664525 * state + 1013904223) >>> 0; + return state / 0x100000000; + }; +} + +function bootstrapCi(values: number[]) { + if (values.length === 0) return [0, 0]; + const random = seededRandom(); + const estimates: number[] = []; + for (let sample = 0; sample < BOOTSTRAP_SAMPLES; sample += 1) { + const drawn: number[] = []; + for (let index = 0; index < values.length; index += 1) { + drawn.push(values[Math.floor(random() * values.length)]); + } + estimates.push(mean(drawn)); + } + return [ + round(percentile(estimates, 0.025)), + round(percentile(estimates, 0.975)), + ]; +} + +function metricSummary(observations: CaseObservation[]) { + const evidence = observations.filter((row) => row.expectEvidence); + const noEvidence = observations.filter((row) => !row.expectEvidence); + const recall5 = evidence.map((row) => + row.relevantRank !== null && row.relevantRank <= 5 ? 1 : 0 + ); + const reciprocalRanks = evidence.map((row) => + row.relevantRank !== null && row.relevantRank <= 10 + ? 1 / row.relevantRank + : 0 + ); + const ndcg10 = evidence.map((row) => + row.relevantRank !== null && row.relevantRank <= 10 + ? 1 / Math.log2(row.relevantRank + 1) + : 0 + ); + const entityTop1 = evidence.map((row) => (row.top1EntityCorrect ? 1 : 0)); + const predictedAbstentions = observations.filter( + (row) => row.predictedNoEvidence + ); + const correctAbstentions = predictedAbstentions.filter( + (row) => !row.expectEvidence + ); + const noEvidencePrecision = predictedAbstentions.length + ? correctAbstentions.length / predictedAbstentions.length + : 0; + const noEvidenceRecall = noEvidence.length + ? noEvidence.filter((row) => row.predictedNoEvidence).length / + noEvidence.length + : 0; + const latencies = observations.map((row) => row.latencyMs); + + return { + recall_at_5: round(mean(recall5)), + recall_at_5_ci95: bootstrapCi(recall5), + mrr_at_10: round(mean(reciprocalRanks)), + mrr_at_10_ci95: bootstrapCi(reciprocalRanks), + ndcg_at_10: round(mean(ndcg10)), + ndcg_at_10_ci95: bootstrapCi(ndcg10), + entity_top1_accuracy: round(mean(entityTop1)), + entity_top1_ci95: bootstrapCi(entityTop1), + no_evidence_precision: round(noEvidencePrecision), + no_evidence_recall: round(noEvidenceRecall), + predicted_abstentions: predictedAbstentions.length, + latency_ms: { + p50: round(percentile(latencies, 0.5), 3), + p95: round(percentile(latencies, 0.95), 3), + mean: round(mean(latencies), 3), + variance: round(variance(latencies), 5), + }, + }; +} + +function isRelevant( + result: SearchResult, + benchmarkCase: (typeof PORTFOLIO_QUERY_BENCHMARK)[number] +) { + if ( + result.entity_id && + benchmarkCase.expectedEntityIds.includes(result.entity_id) + ) { + return true; + } + + const evidenceText = + `${result.title}\n${result.contentPreview}`.toLowerCase(); + return benchmarkCase.relevanceHints.some((hint) => + evidenceText.includes(hint.toLowerCase()) + ); +} + +async function runMode(mode: RetrievalMode) { + process.env.ASKOOSU_RAG_RETRIEVAL = mode; + const observations: CaseObservation[] = []; + + for (const benchmarkCase of PORTFOLIO_QUERY_BENCHMARK) { + const repeatLatencies: number[] = []; + let relevantRank: number | null = null; + let top1EntityCorrect = false; + let predictedNoEvidence = false; + + for (let repeat = 0; repeat < REPEATS; repeat += 1) { + const retrievalQuery = benchmarkCase.priorTurn + ? `${benchmarkCase.priorTurn}\nFollow-up: ${benchmarkCase.query}` + : benchmarkCase.query; + const startedAt = performance.now(); + const payload = await searchRagChunks({ + q: retrievalQuery, + limit: 10, + includePrivate: false, + debug: true, + }); + repeatLatencies.push(performance.now() - startedAt); + if (repeat > 0) continue; + + relevantRank = benchmarkCase.expectEvidence + ? payload.results.findIndex((result) => + isRelevant(result, benchmarkCase) + ) + 1 + : null; + if (relevantRank === 0) relevantRank = null; + top1EntityCorrect = Boolean( + payload.results[0]?.entity_id && + benchmarkCase.expectedEntityIds.includes(payload.results[0].entity_id) + ); + predictedNoEvidence = payload.results.length === 0; + } + + observations.push({ + id: benchmarkCase.id, + difficulty: benchmarkCase.difficulty, + category: benchmarkCase.category, + expectEvidence: benchmarkCase.expectEvidence, + relevantRank, + top1EntityCorrect, + predictedNoEvidence, + latencyMs: mean(repeatLatencies), + }); + } + + const byDifficulty = Object.fromEntries( + ['easy', 'medium', 'adversarial'].map((difficulty) => [ + difficulty, + metricSummary( + observations.filter((row) => row.difficulty === difficulty) + ), + ]) + ); + const byCategory = Object.fromEntries( + [...new Set(observations.map((row) => row.category))].map((category) => [ + category, + metricSummary(observations.filter((row) => row.category === category)), + ]) + ); + + return { + mode, + metrics: metricSummary(observations), + by_difficulty: byDifficulty, + by_category: byCategory, + }; +} + +async function main() { + if (!process.env.DATABASE_URL && !process.env.POSTGRES_URL) { + throw new Error('DATABASE_URL or POSTGRES_URL is required'); + } + + process.env.ASKOOSU_RAG_STORE = 'postgres'; + process.env.ASKOOSU_RAG_AUTO_SYNC = 'false'; + process.env.ASKOOSU_RAG_SEARCH_CACHE_TTL_MS = '0'; + process.env.ASKOOSU_RAG_RETRIEVAL = 'lexical'; + delete process.env.OPENAI_API_KEY; + delete process.env.NOTION_API_KEY; + + const sync = await syncPortfolioKnowledgeBase({ force: true }); + const baseline = await runMode('lexical'); + const candidate = await runMode('hybrid'); + + const result = { + experiment: 'askoosu-portfolio-retrieval-v1', + question: + 'Does entity-aware hybrid RRF improve retrieval over lexical-only search on committed portfolio evidence?', + hypothesis: + 'Hybrid lexical + entity RRF will improve entity retrieval on ambiguous/typo queries without increasing private-data exposure.', + dataset: { + total_queries: PORTFOLIO_QUERY_BENCHMARK.length, + easy: PORTFOLIO_QUERY_BENCHMARK.filter((row) => row.difficulty === 'easy') + .length, + medium: PORTFOLIO_QUERY_BENCHMARK.filter( + (row) => row.difficulty === 'medium' + ).length, + adversarial: PORTFOLIO_QUERY_BENCHMARK.filter( + (row) => row.difficulty === 'adversarial' + ).length, + no_evidence: PORTFOLIO_QUERY_BENCHMARK.filter( + (row) => !row.expectEvidence + ).length, + relevance_label: + 'expected entity id OR case-specific title/content anchor', + }, + corpus: { + stored_chunks: sync.storedChunkCount, + embedded_chunks: sync.embeddedChunkCount, + source_chunks: sync.sourceChunkCount, + }, + repeated_measurements_per_query: REPEATS, + confidence_interval: `non-parametric bootstrap, ${BOOTSTRAP_SAMPLES} samples, fixed seed 20260908`, + baseline, + candidate, + limitations: [ + 'The benchmark uses committed portfolio documents only; live Notion content and external embeddings are intentionally disabled.', + 'Hybrid in this run means lexical + entity RRF/boosting; vector retrieval is not claimed without embedding credentials.', + 'Entity labels are curated from the same public portfolio knowledge domain and are not an independently blinded annotation study.', + 'No-evidence behavior is measured as retrieval abstention (zero returned chunks), not LLM factuality.', + ], + }; + + console.log(JSON.stringify(result, null, 2)); + if (process.env.ASKOOSU_BENCHMARK_STRICT === '1') { + const baselineRecall = baseline.metrics.recall_at_5; + const candidateRecall = candidate.metrics.recall_at_5; + const candidateNoEvidenceRecall = candidate.metrics.no_evidence_recall; + const candidateEntityAccuracy = candidate.metrics.entity_top1_accuracy; + const failures = [ + candidateRecall < baselineRecall + ? `candidate Recall@5 ${candidateRecall} is below lexical ${baselineRecall}` + : null, + candidateRecall < 0.45 + ? `candidate Recall@5 ${candidateRecall} is below release floor 0.45` + : null, + candidateNoEvidenceRecall < 0.9 + ? `candidate no-evidence recall ${candidateNoEvidenceRecall} is below 0.90` + : null, + candidateEntityAccuracy < 0.2 + ? `candidate entity top-1 accuracy ${candidateEntityAccuracy} is below 0.20` + : null, + ].filter(Boolean); + if (failures.length > 0) { + throw new Error( + `Retrieval benchmark gate failed: ${failures.join('; ')}` + ); + } + } + const pool = await getPostgresPool(); + await pool.end(); + globalThis.askOosuPgPool = undefined; +} + +main().catch((error) => { + console.error(error); + process.exitCode = 1; +}); diff --git a/src/app/api/chat/route.ts b/src/app/api/chat/route.ts index deac95d..cd49a22 100644 --- a/src/app/api/chat/route.ts +++ b/src/app/api/chat/route.ts @@ -50,6 +50,7 @@ import { logWarn, toLogError, } from '@/lib/observability/logger'; +import { createRequestTrace } from '@/lib/observability/request-trace'; import { checkRateLimit, checkRateLimitForKey, @@ -111,17 +112,20 @@ const SAFE_IDENTIFIER_PATTERN = /^[A-Za-z0-9_.:-]+$/; export async function POST(req: Request) { const requestStartedAt = Date.now(); const requestId = crypto.randomUUID(); + const trace = createRequestTrace({ requestId, route: CHAT_ROUTE }); let messages: UIMessage[] = []; let body: ValidatedChatRequestBody | null = null; let orchestration: ChatOrchestration | null = null; let responseLanguage = getRequestFallbackLanguage(req); try { - const rateLimit = await checkRateLimit(req, { - scope: 'api:chat', - windowMs: 60 * 1000, - max: getPositiveIntegerEnv('ASKOOSU_CHAT_RATE_LIMIT_PER_MINUTE', 60), - }); + const rateLimit = await trace.measure('rate_limit', () => + checkRateLimit(req, { + scope: 'api:chat', + windowMs: 60 * 1000, + max: getPositiveIntegerEnv('ASKOOSU_CHAT_RATE_LIMIT_PER_MINUTE', 60), + }) + ); if (!rateLimit.allowed) { logWarn('chat.request_failed', { @@ -141,33 +145,38 @@ export async function POST(req: Request) { }); } - body = await readChatRequestBody(req); - messages = body.messages; + const validatedBody = await trace.measure('request_parse', () => + readChatRequestBody(req) + ); + body = validatedBody; + messages = validatedBody.messages; responseLanguage = detectLanguage( getLatestUserText(messages), - body.preferredLanguage + validatedBody.preferredLanguage ); logInfo('chat.request_received', { requestId, route: CHAT_ROUTE, - requestByteSize: body.requestByteSize, + requestByteSize: validatedBody.requestByteSize, messageCount: messages.length, - source: body.source, + source: validatedBody.source, language: responseLanguage, - conversationIdPresent: Boolean(body.conversationId), + conversationIdPresent: Boolean(validatedBody.conversationId), questionLength: getLatestUserText(messages).length, questionPreview: getLocalQuestionPreview(getLatestUserText(messages)), }); - const sessionRateLimit = body.conversationId - ? await checkRateLimitForKey(body.conversationId, { - scope: 'api:chat:session', - windowMs: 60 * 1000, - max: getPositiveIntegerEnv( - 'ASKOOSU_CHAT_SESSION_RATE_LIMIT_PER_MINUTE', - 30 - ), - }) + const sessionRateLimit = validatedBody.conversationId + ? await trace.measure('session_rate_limit', () => + checkRateLimitForKey(validatedBody.conversationId!, { + scope: 'api:chat:session', + windowMs: 60 * 1000, + max: getPositiveIntegerEnv( + 'ASKOOSU_CHAT_SESSION_RATE_LIMIT_PER_MINUTE', + 30 + ), + }) + ) : null; if (sessionRateLimit && !sessionRateLimit.allowed) { @@ -188,46 +197,49 @@ export async function POST(req: Request) { }); } - orchestration = await prepareChatOrchestration({ - messages, - requestId, - preferredLanguage: body.preferredLanguage, - starterQuestionId: body.starterQuestionId, - faqId: body.faqId, - intentId: body.intentId, - displayQuestion: body.displayQuestion, - originalQuickLabel: body.originalQuickLabel, - answerVariant: body.answerVariant, - renderSpec: body.renderSpec, - source: body.source, - }); + const preparedOrchestration = await trace.measure('orchestration', () => + prepareChatOrchestration({ + messages, + requestId, + preferredLanguage: validatedBody.preferredLanguage, + starterQuestionId: validatedBody.starterQuestionId, + faqId: validatedBody.faqId, + intentId: validatedBody.intentId, + displayQuestion: validatedBody.displayQuestion, + originalQuickLabel: validatedBody.originalQuickLabel, + answerVariant: validatedBody.answerVariant, + renderSpec: validatedBody.renderSpec, + source: validatedBody.source, + }) + ); + orchestration = preparedOrchestration; logInfo('chat.route_decided', { requestId, route: CHAT_ROUTE, ...getRouteDecisionLogData( - orchestration.mode === 'direct' - ? orchestration.directAnswer.metadata - : orchestration.metadata + preparedOrchestration.mode === 'direct' + ? preparedOrchestration.directAnswer.metadata + : preparedOrchestration.metadata ), }); - if (orchestration.mode === 'direct') { - const directAnswer = orchestration.directAnswer; + if (preparedOrchestration.mode === 'direct') { + const directAnswer = preparedOrchestration.directAnswer; const directMetadata = directAnswer.metadata; const isCacheHit = - orchestration.routeDecision.mode === 'faq_direct' || - orchestration.routeDecision.mode === 'answer_cache'; + preparedOrchestration.routeDecision.mode === 'faq_direct' || + preparedOrchestration.routeDecision.mode === 'answer_cache'; if (isCacheHit) { logInfo('chat.cache_hit', { requestId, route: CHAT_ROUTE, - cacheKind: orchestration.routeDecision.mode, + cacheKind: preparedOrchestration.routeDecision.mode, ...getRouteDecisionLogData(directMetadata), }); } - if (orchestration.routeDecision.mode === 'safe_fallback') { + if (preparedOrchestration.routeDecision.mode === 'safe_fallback') { logInfo('chat.fallback_returned', { requestId, route: CHAT_ROUTE, @@ -256,7 +268,7 @@ export async function POST(req: Request) { scheduleAskEventLog({ req, body, - question: orchestration.question, + question: preparedOrchestration.question, metadata: directMetadata, latencyMs: Date.now() - requestStartedAt, }); @@ -287,33 +299,37 @@ export async function POST(req: Request) { logInfo('chat.generation_started', { requestId, route: CHAT_ROUTE, - ...getRouteDecisionLogData(orchestration.metadata), + ...getRouteDecisionLogData(preparedOrchestration.metadata), provider: primaryModel.provider, model: primaryModel.modelName, }); - const generation = await generateAnswerWithFallback({ - primaryModel, - system: [ - SYSTEM_PROMPT_TEXT, - RAG_CHAT_SYSTEM_PROMPT, - orchestration.ragContext.contextText, - ] - .filter(Boolean) - .join('\n\n'), - messages: promptMessages, - tools, - stopWhen: stepCountIs(2), - usageMetadata: { - route: CHAT_ROUTE, - ...toUsageMetadata(orchestration.metadata), - }, - }); - const leakDetected = detectPromptLeakage(generation.answer); + const generation = await trace.measure('generation', () => + generateAnswerWithFallback({ + primaryModel, + system: [ + SYSTEM_PROMPT_TEXT, + RAG_CHAT_SYSTEM_PROMPT, + preparedOrchestration.ragContext.contextText, + ] + .filter(Boolean) + .join('\n\n'), + messages: promptMessages, + tools, + stopWhen: stepCountIs(2), + usageMetadata: { + route: CHAT_ROUTE, + ...toUsageMetadata(preparedOrchestration.metadata), + }, + }) + ); + const leakDetected = trace.measureSync('output_guardrail', () => + detectPromptLeakage(generation.answer) + ); logInfo('chat.generation_completed', { requestId, route: CHAT_ROUTE, - ...getRouteDecisionLogData(orchestration.metadata), + ...getRouteDecisionLogData(preparedOrchestration.metadata), provider: generation.provider, model: generation.model, answerSource: generation.answerSource, @@ -324,19 +340,19 @@ export async function POST(req: Request) { if (leakDetected) { const safeAnswer = buildInsufficientEvidenceAnswer( - orchestration.language + preparedOrchestration.language ); const confidenceSignals = buildAnswerConfidenceSignals({ sources: [], warnings: [ - ...orchestration.metadata.warnings, + ...preparedOrchestration.metadata.warnings, PROMPT_LEAK_DETECTED_ERROR_CODE, ], - intent: orchestration.metadata.confidenceSignals?.intent ?? 0.5, + intent: preparedOrchestration.metadata.confidenceSignals?.intent ?? 0.5, usesGroundedSources: false, }); const responseMetadata = { - ...orchestration.metadata, + ...preparedOrchestration.metadata, sources: [], matchedEntityIds: [], sourceChunkIds: [], @@ -344,7 +360,7 @@ export async function POST(req: Request) { confidenceSignals, hasTodoEvidence: false, warnings: [ - ...orchestration.metadata.warnings, + ...preparedOrchestration.metadata.warnings, PROMPT_LEAK_DETECTED_ERROR_CODE, ], answerSource: 'insufficient_evidence' as const, @@ -369,7 +385,7 @@ export async function POST(req: Request) { scheduleAskEventLog({ req, body, - question: orchestration.question, + question: preparedOrchestration.question, metadata: responseMetadata, latencyMs: Date.now() - requestStartedAt, }); @@ -383,13 +399,13 @@ export async function POST(req: Request) { const generatedAnswer = appendGeneratedContextualQuote({ answer: generation.answer, - metadata: orchestration.metadata, - question: orchestration.question, + metadata: preparedOrchestration.metadata, + question: preparedOrchestration.question, messages, }); const responseMetadata = { - ...orchestration.metadata, + ...preparedOrchestration.metadata, answerSource: generation.answerSource, provider: generation.provider, model: generation.model, @@ -399,8 +415,8 @@ export async function POST(req: Request) { }; const cacheInput = { - normalizedQuestion: orchestration.normalizedQuestion, - language: orchestration.language, + normalizedQuestion: preparedOrchestration.normalizedQuestion, + language: preparedOrchestration.language, answer: generatedAnswer, answerSource: generation.answerSource, matchedEntityIds: responseMetadata.matchedEntityIds, @@ -429,7 +445,7 @@ export async function POST(req: Request) { scheduleAskEventLog({ req, body, - question: orchestration.question, + question: preparedOrchestration.question, metadata: responseMetadata, latencyMs: Date.now() - requestStartedAt, }); @@ -495,6 +511,8 @@ export async function POST(req: Request) { answer: buildModelUnavailableAnswer(fallbackMetadata.language), metadata: fallbackMetadata, }); + } finally { + trace.finish(); } } @@ -698,7 +716,9 @@ const chatRequestBodySchema = z conversationId: optionalSafeIdentifierSchema( MAX_CONVERSATION_ID_LENGTH ).optional(), - sessionId: optionalSafeIdentifierSchema(MAX_CONVERSATION_ID_LENGTH).optional(), + sessionId: optionalSafeIdentifierSchema( + MAX_CONVERSATION_ID_LENGTH + ).optional(), pagePath: optionalTrimmedStringSchema(500).optional(), referrer: optionalTrimmedStringSchema(500).optional(), utmSource: optionalTrimmedStringSchema(500).optional(), diff --git a/src/lib/chat/orchestrator.ts b/src/lib/chat/orchestrator.ts index 693c6e5..fcfa3b3 100644 --- a/src/lib/chat/orchestrator.ts +++ b/src/lib/chat/orchestrator.ts @@ -321,7 +321,9 @@ export async function prepareChatOrchestration({ }; } - const ragContext = await buildRagChatContext(routingQuestion, language); + const ragContext = await buildRagChatContext(routingQuestion, language, { + requestId: requestContext.requestId ?? undefined, + }); const conversationEntityHints = getConversationEntityHints(routingQuestion); const faqEvidenceFallback = getFaqEvidenceFallback({ faqRoute, @@ -471,7 +473,9 @@ function getFaqEvidenceFallback({ matchedEntityIds: uniqueValues( candidateAnswers.flatMap((answer) => answer.matchedEntityIds) ), - confidence: Math.max(...candidateAnswers.map((answer) => answer.confidence)), + confidence: Math.max( + ...candidateAnswers.map((answer) => answer.confidence) + ), }; } @@ -596,7 +600,8 @@ function getRepeatedQuestionSignal({ if (!normalizedQuestion) return null; const sameQuestionCount = previousUserMessages.filter( - (message) => normalizeQuestion(getMessageText(message)) === normalizedQuestion + (message) => + normalizeQuestion(getMessageText(message)) === normalizedQuestion ).length; if (sameQuestionCount === 0) return null; @@ -1357,8 +1362,8 @@ function prefixRepeatedConcernAnswer({ if (!concern) return answer; const previousAssistantMessages = getPreviousAssistantMessages(messages); - const hasPreviousAnswerForConcern = previousAssistantMessages.some((message) => - assistantMessageAddressesRecruiterConcern(message, concern) + const hasPreviousAnswerForConcern = previousAssistantMessages.some( + (message) => assistantMessageAddressesRecruiterConcern(message, concern) ); if (!hasPreviousAnswerForConcern) return answer; @@ -1470,8 +1475,7 @@ function isOffTopicRedirectText(text: string) { function answerTextMatchesRecruiterConcern(text: string, concern: string) { const patterns: Record = { age: /(상대적으로\s*늦게\s*개발\s*커리어|그\s*시간을\s*공백|나이를\s*방어|older\s+than\s+typical\s+junior|does\s+not\s+see\s+that\s+time\s+as\s+a\s+gap)/i, - role: - /(AI-connected\s+Fullstack|AI\s*연결\s*풀스택|레이어.*연결|product\/UX|포지셔닝)/i, + role: /(AI-connected\s+Fullstack|AI\s*연결\s*풀스택|레이어.*연결|product\/UX|포지셔닝)/i, non_cs: /(비전공|non[-\s]?CS|CS\s+degree|learn\s+faster)/i, ai_dependency: /(AI\s*의존|AI\s*없이|AI\s*코드|review\s+AI-generated|AI-generated\s+code)/i, diff --git a/src/lib/observability/request-trace.ts b/src/lib/observability/request-trace.ts new file mode 100644 index 0000000..082f43d --- /dev/null +++ b/src/lib/observability/request-trace.ts @@ -0,0 +1,86 @@ +import { logInfo } from './logger'; + +type TraceStage = { + name: string; + durationMs: number; + success: boolean; +}; + +type TraceLogger = (eventName: string, data: Record) => void; + +export function createRequestTrace({ + requestId, + route, + logger = logInfo, + now = () => performance.now(), +}: { + requestId: string; + route: string; + logger?: TraceLogger; + now?: () => number; +}) { + const traceStartedAt = now(); + const stages: TraceStage[] = []; + let finished = false; + + async function measure( + name: string, + operation: () => Promise + ): Promise { + const startedAt = now(); + try { + const result = await operation(); + stages.push({ + name, + durationMs: roundDuration(now() - startedAt), + success: true, + }); + return result; + } catch (error) { + stages.push({ + name, + durationMs: roundDuration(now() - startedAt), + success: false, + }); + throw error; + } + } + + function measureSync(name: string, operation: () => T): T { + const startedAt = now(); + try { + const result = operation(); + stages.push({ + name, + durationMs: roundDuration(now() - startedAt), + success: true, + }); + return result; + } catch (error) { + stages.push({ + name, + durationMs: roundDuration(now() - startedAt), + success: false, + }); + throw error; + } + } + + function finish() { + if (finished) return; + finished = true; + logger('chat.trace_completed', { + requestId, + route, + totalLatencyMs: roundDuration(now() - traceStartedAt), + stageCount: stages.length, + stages, + }); + } + + return { measure, measureSync, finish }; +} + +function roundDuration(value: number) { + return Number(Math.max(0, value).toFixed(3)); +} diff --git a/src/lib/rag/chat-context.ts b/src/lib/rag/chat-context.ts index b32edc4..5094305 100644 --- a/src/lib/rag/chat-context.ts +++ b/src/lib/rag/chat-context.ts @@ -1,6 +1,7 @@ import { getRagTopK } from './config'; import { searchRagChunks, type RagChunkSearchResult } from './search'; import type { ChatLanguage } from '@/lib/i18n/detect-language'; +import { logInfo } from '@/lib/observability/logger'; const GUARDRAIL_ENTITY_ID = 'policy.guardrail'; @@ -40,7 +41,8 @@ export type RagChatContext = { export async function buildRagChatContext( question: string, - language?: ChatLanguage + language?: ChatLanguage, + options: { requestId?: string } = {} ): Promise { const normalizedQuestion = question.trim(); const warnings: string[] = []; @@ -51,6 +53,7 @@ export async function buildRagChatContext( ]); } + const retrievalStartedAt = performance.now(); const [primarySearch, guardrailSearch] = await Promise.all([ searchRagChunks({ q: normalizedQuestion, @@ -64,6 +67,16 @@ export async function buildRagChatContext( includeContent: true, }), ]); + logInfo('rag.retrieval_completed', { + requestId: options.requestId, + route: 'api/chat', + latencyMs: Number((performance.now() - retrievalStartedAt).toFixed(3)), + primaryResultCount: primarySearch.results.length, + guardrailResultCount: guardrailSearch.results.length, + primarySearchMode: primarySearch.searchMode, + primaryWarningCount: primarySearch.warnings.length, + guardrailWarningCount: guardrailSearch.warnings.length, + }); warnings.push(...primarySearch.warnings, ...guardrailSearch.warnings); @@ -118,13 +131,12 @@ function buildEmptyContext(warnings: string[]): RagChatContext { }); return { - contextText: - [ - '## Portfolio Evidence', - 'No matching public Wiki evidence was found for this specific question.', - 'If this reaches generation, use only stable profile facts and the conversation history. Do not invent missing portfolio evidence.', - 'For factual claims about Oosu, projects, links, career, private details, or metrics, use only the stable portfolio prompt facts. If the fact is not in the prompt or retrieved evidence, say the Wiki evidence is not enough instead of guessing.', - ].join('\n'), + contextText: [ + '## Portfolio Evidence', + 'No matching public Wiki evidence was found for this specific question.', + 'If this reaches generation, use only stable profile facts and the conversation history. Do not invent missing portfolio evidence.', + 'For factual claims about Oosu, projects, links, career, private details, or metrics, use only the stable portfolio prompt facts. If the fact is not in the prompt or retrieved evidence, say the Wiki evidence is not enough instead of guessing.', + ].join('\n'), metadata: { sources: [], confidence: confidenceSignals.final, diff --git a/tests/unit/logger.test.ts b/tests/unit/logger.test.ts new file mode 100644 index 0000000..5fcdb38 --- /dev/null +++ b/tests/unit/logger.test.ts @@ -0,0 +1,32 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { logInfo } from '../../src/lib/observability/logger.ts'; + +test('structured logger redacts prompts, questions, source ids, and credentials', () => { + const originalInfo = console.info; + const lines: string[] = []; + console.info = (...args: unknown[]) => lines.push(args.join(' ')); + try { + logInfo('security.redaction_test', { + requestId: 'req-safe', + question: 'private home address?', + rawPrompt: 'system secret', + sourceChunkIds: ['private-chunk'], + authorization: 'Bearer secret-token', + safeMetric: 7, + }); + } finally { + console.info = originalInfo; + } + + const payload = JSON.parse(lines[0]) as Record; + assert.equal(payload.requestId, 'req-safe'); + assert.equal(payload.question, '[redacted]'); + assert.equal(payload.rawPrompt, '[redacted]'); + assert.equal(payload.sourceChunkIds, '[redacted]'); + assert.equal(payload.authorization, '[redacted]'); + assert.equal(payload.safeMetric, 7); + assert.equal(lines[0].includes('private home address'), false); + assert.equal(lines[0].includes('secret-token'), false); +}); diff --git a/tests/unit/request-trace.test.ts b/tests/unit/request-trace.test.ts new file mode 100644 index 0000000..678303c --- /dev/null +++ b/tests/unit/request-trace.test.ts @@ -0,0 +1,54 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; + +import { createRequestTrace } from '../../src/lib/observability/request-trace.ts'; + +test('request trace records correlated stage timings without request payloads', async () => { + const records: Array<{ event: string; data: Record }> = []; + const ticks = [0, 2, 7, 8, 11, 15]; + const trace = createRequestTrace({ + requestId: 'req-test', + route: 'api/chat', + now: () => ticks.shift() ?? 15, + logger: (event, data) => records.push({ event, data }), + }); + + await trace.measure('routing', async () => 'ok'); + trace.measureSync('guardrail', () => true); + trace.finish(); + trace.finish(); + + assert.equal(records.length, 1); + assert.equal(records[0].event, 'chat.trace_completed'); + assert.equal(records[0].data.requestId, 'req-test'); + assert.deepEqual(records[0].data.stages, [ + { name: 'routing', durationMs: 5, success: true }, + { name: 'guardrail', durationMs: 3, success: true }, + ]); + assert.equal(records[0].data.totalLatencyMs, 15); + assert.equal('question' in records[0].data, false); + assert.equal('answer' in records[0].data, false); +}); + +test('request trace marks failed stages before rethrowing', async () => { + const records: Array<{ event: string; data: Record }> = []; + const ticks = [0, 10, 16, 20]; + const trace = createRequestTrace({ + requestId: 'req-failure', + route: 'api/chat', + now: () => ticks.shift() ?? 20, + logger: (event, data) => records.push({ event, data }), + }); + + await assert.rejects( + trace.measure('provider', async () => { + throw new Error('injected provider timeout'); + }), + /injected provider timeout/ + ); + trace.finish(); + + assert.deepEqual(records[0].data.stages, [ + { name: 'provider', durationMs: 6, success: false }, + ]); +});