diff --git a/.claude/hooks/filter-test-output.sh b/.claude/hooks/filter-test-output.sh new file mode 100755 index 0000000..90d5683 --- /dev/null +++ b/.claude/hooks/filter-test-output.sh @@ -0,0 +1,130 @@ +#!/usr/bin/env bash +# ============================================================================= +# Test-output filter – fluory-system (PreToolUse hook, matcher: Bash) +# Template: Entwicklungsplan/templates/base/.claude/hooks/filter-test-output.sh +# +# Routes UNAMBIGUOUS test/verify commands (npm|pnpm|yarn|bun test/verify*, vitest, jest, mocha, +# playwright test, cypress run, pytest -- also behind the Python runners uv|poetry|pdm|hatch|rye, +# go test, cargo test, make/just test|verify) through +# scripts/quiet-run.sh, which trims success output, keeps failures and preserves the exit code +# (SYSTEM.md §11). The command itself is executed unchanged inside the wrapper. +# Suite policy (§11): an UNSCOPED suite run (npm test without a path, vitest/jest/pytest without +# file or filter, playwright test without a spec, go test ./..., cargo test) is +# - P0 (stage experiment or unknown): allowed, rewritten, with a hint as additional context +# - P1/P2 (stage internal|production): BLOCKED (exit 2) – run the affected tests, verify:changed, +# verify or verify:full; deliberate full run once: prefix FLUORY_FULL_SUITE=1, reason in the PR. +# The policy also applies to commands already wrapped in quiet-run.sh (no bypass via the wrapper). +# Never rewritten: commands with FLUORY_FULL_OUTPUT=1, chains (&&, ||, ;, |), redirections, +# subshells, multi-line commands, anything already wrapped, or when quiet-run.sh is missing. +# Output: JSON with updatedInput (docs: hooks reference, PreToolUse). Fail open: exit 0. +# Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .tool_input.command + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +have_json_tool() { + command -v jq >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1 || command -v node >/dev/null 2>&1 +} +emit() { # $1 = wrapped command, $2 = additional context (optional) → JSON with tool_input + {command} + local reason="fluory-system: test output filtered by scripts/quiet-run.sh (exit code preserved; prefix FLUORY_FULL_OUTPUT=1 for the full output)" + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq --arg c "$1" --arg r "$reason" --arg x "${2:-}" '{hookSpecificOutput: ({hookEventName: "PreToolUse", permissionDecision: "allow", permissionDecisionReason: $r, updatedInput: (.tool_input + {command: $c})} + (if $x == "" then {} else {additionalContext: $x} end))}' + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +d = json.load(sys.stdin); ti = dict(d.get("tool_input") or {}); ti["command"] = sys.argv[1] +o = {"hookEventName": "PreToolUse", "permissionDecision": "allow", "permissionDecisionReason": sys.argv[2], "updatedInput": ti} +if sys.argv[3]: o["additionalContext"] = sys.argv[3] +print(json.dumps({"hookSpecificOutput": o}, indent=2)) +' "$1" "$reason" "${2:-}" + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + const d = JSON.parse(s); const ti = Object.assign({}, d.tool_input || {}); ti.command = process.argv[1]; + const o = {hookEventName: "PreToolUse", permissionDecision: "allow", permissionDecisionReason: process.argv[2], updatedInput: ti}; + if (process.argv[3]) o.additionalContext = process.argv[3]; + console.log(JSON.stringify({hookSpecificOutput: o}, null, 2)); });' "$1" "$reason" "${2:-}" + fi +} +block_suite() { + { + echo "[fluory-system test filter] BLOCKED: unscoped suite run in a P1/P2 project: $1" + echo "Rule (SYSTEM.md §11): locally the agent works focused – run the affected tests (path, -k or spec), verify:changed, or the canonical verify / verify:full. Deliberate full run once: prefix FLUORY_FULL_SUITE=1 and name the reason in the PR." + } >&2 + exit 2 +} + +have_json_tool || exit 0 +cmd="$(json_get .tool_input.command)" +[ -n "$cmd" ] || exit 0 +root="${CLAUDE_PROJECT_DIR:-}" +{ [ -n "$root" ] && [ -d "$root" ]; } || root="$(json_get .cwd)" +{ [ -n "$root" ] && [ -d "$root" ]; } || root="$PWD" +qr="" +for c in "$root/scripts/quiet-run.sh" "$root/templates/base/scripts/quiet-run.sh"; do + [ -f "$c" ] && { qr="$c"; break; } +done +[ -n "$qr" ] || exit 0 + +# exceptions: full output requested, chains, pipes, redirections, subshells, newlines +case "$cmd" in *FLUORY_FULL_OUTPUT=*) exit 0 ;; esac +case "$cmd" in *'&&'*|*'||'*|*';'*|*'|'*|*'>'*|*'<'*|*'`'*|*'$('*|*$'\n'*) exit 0 ;; esac + +# already wrapped: never rewrite, but the suite policy still applies to the inner command +wrapped=0; m="$cmd" +case "$cmd" in + *quiet-run.sh*) wrapped=1; m="$(printf '%s' "$cmd" | sed -E "s/^.*quiet-run\.sh[\"']?[[:space:]]+//; s/^[\"']//; s/[\"']\$//")" ;; +esac +m="$(printf '%s' "$m" | sed -E 's/^([A-Za-z_][A-Za-z0-9_]*=[^ ]* +)+//')" # env prefixes stay in the executed command + +runner='^((pnpm|npm|yarn|bun)( run)? (test|verify)(:[a-z-]+)?( --)?( [^ ]+)*|(npx |pnpm exec |pnpm dlx |yarn )?(vitest|jest|mocha|playwright test|cypress run)( [^ ]+)*|(uv run |poetry run |pdm run |hatch run |rye run )?(python3? -m )?pytest( [^ ]+)*|go test( [^ ]+)*|cargo test( [^ ]+)*|(make|just) (test|verify)([:-][a-z-]+)?( [^ ]+)*)$' +printf '%s' "$m" | grep -qE "$runner" || exit 0 + +flags='( --| --?[A-Za-z-]+(=[^ ]+)?)*$' +broad=0 +case "$m" in + *verify*) ;; # verify:changed / verify / verify:full are the canonical, scoped-by-design commands + *) + if printf '%s' "$m" | grep -qE "^(pnpm|npm|yarn|bun)( run)? test$flags"; then broad=1 + elif printf '%s' "$m" | grep -qE "^(npx |pnpm exec |pnpm dlx |yarn )?(vitest|jest|mocha)( run)?$flags"; then broad=1 + elif printf '%s' "$m" | grep -qE "^(npx |pnpm exec |pnpm dlx |yarn )?(playwright test|cypress run)$flags"; then broad=1 + elif printf '%s' "$m" | grep -qE "^(uv run |poetry run |pdm run |hatch run |rye run )?(python3? -m )?pytest$flags"; then broad=1 + elif printf '%s' "$m" | grep -qE '^go test( -[a-z]+)*( \./\.\.\.)?$'; then broad=1 + elif printf '%s' "$m" | grep -qE "^cargo test$flags"; then broad=1 + fi ;; +esac +stage="$(sed -n 's/^[[:space:]]*stage:[[:space:]]*\([a-z]*\).*/\1/p' "$root/project-profile.yml" 2>/dev/null | head -1)" +if [ "$broad" -eq 1 ] && { [ "$stage" = "internal" ] || [ "$stage" = "production" ]; }; then + case "$cmd" in *FLUORY_FULL_SUITE=1*) ;; *) block_suite "$cmd" ;; esac +fi +[ "$wrapped" -eq 1 ] && exit 0 + +ctx="" +[ "$broad" -eq 1 ] && ctx="fluory-system: unscoped suite run (allowed in P0) – prefer verify:changed or the affected test file; in P1/P2 this is blocked (SYSTEM.md §11)." +quoted="$(printf '%s' "$cmd" | sed "s/'/'\\\\''/g")" +emit "bash \"$qr\" '$quoted'" "$ctx" +exit 0 diff --git a/.claude/hooks/guard-checkpoint.sh b/.claude/hooks/guard-checkpoint.sh new file mode 100755 index 0000000..4e43b52 --- /dev/null +++ b/.claude/hooks/guard-checkpoint.sh @@ -0,0 +1,110 @@ +#!/usr/bin/env bash +# ============================================================================= +# Checkpoint guard – fluory-system (PreToolUse hook, matcher: Bash) +# Template: Entwicklungsplan/templates/base/.claude/hooks/guard-checkpoint.sh +# +# Risky operations need a git reset point first (SYSTEM.md §4): data migrations, generator +# runs, dependency upgrades, mass file operations (git mv, find -exec, sed -i over globs, rm -r, +# codemods, formatters over directories or the whole tree – a single file is not a mass +# operation) and commands that discard changes (git reset --hard, checkout or restore of the +# tree, clean, stash drop). Chains, pipes, subshells, env prefixes, wrappers (bash -c, eval, xargs) +# and multi-line commands are unwrapped; xargs feeding sed -i, rm, mv or a formatter is a mass +# operation. With uncommitted changes in the working tree such a call is blocked (exit 2) until +# a checkpoint exists: +# git add -A && git commit -m "chore: checkpoint before " (push if the work matters) +# /rewind is no substitute: it restores neither Bash, generator, database nor subagent changes. +# Deliberate exception for one call: prefix FLUORY_NO_CHECKPOINT=1 and name the reason in the PR. +# Fail open without jq/python3/node. Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +set -f +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .tool_input.command + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +have_json_tool() { + command -v jq >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1 || command -v node >/dev/null 2>&1 +} +strip_heredocs() { # heredoc bodies (<\" (push if the work matters), then run the command again. Deliberate exception for this one call: prefix FLUORY_NO_CHECKPOINT=1 and name the reason in the PR. /rewind is no substitute for git (SYSTEM.md §4)." + } >&2 + exit 2 +} + +have_json_tool || exit 0 +cmd="$(json_get .tool_input.command)" +[ -n "$cmd" ] || exit 0 +case "$cmd" in *FLUORY_NO_CHECKPOINT=1*) exit 0 ;; esac + +cwd="$(json_get .cwd)" +{ [ -n "$cwd" ] && [ -d "$cwd" ]; } || cwd="${CLAUDE_PROJECT_DIR:-$PWD}" + +segments="$(printf '%s\n' "$cmd" | strip_heredocs | awk '{ gsub(/&&|\|\||;|\||\$\(|`/, "\n"); print }')" +risky=""; what="" +while IFS= read -r seg; do + seg="${seg#"${seg%%[![:space:]]*}"}" + [ -n "$seg" ] || continue + s="$seg"; via_xargs=0 + for _ in 1 2; do # two passes: bash -c "npx prisma …", eval "…", $(…), env prefixes, wrappers + s="${s#\$(}"; s="${s#\`}"; s="${s#(}"; s="${s#\"}"; s="${s#\'}"; s="${s%\"}"; s="${s%\'}"; s="${s%)}"; s="${s%\`}" + s="$(printf '%s' "$s" | sed -E 's/^([A-Za-z_][A-Za-z0-9_]*=[^ ]* +)+//')" + case "$s" in xargs\ *) via_xargs=1 ;; esac + s="$(printf '%s' "$s" | sed -E 's/^((command|sudo|exec|time|nohup|env|builtin|eval|xargs|bash|sh|zsh|dash|ksh|npx|bunx|pnpm exec|pnpm dlx|yarn dlx|yarn exec|poetry run|pipenv run|uv run|bundle exec|python3? -m)( -[A-Za-z0-9=.,_-]+)* +)+//')" + done + if [ "$via_xargs" -eq 1 ] && printf '%s' "$s" | grep -qE '^(sed -i|perl -[a-zA-Z]*i|rm |mv |git mv|prettier --write|eslint --fix|black|ruff format|gofmt -w)'; then + risky=1; what="mass file operation via xargs" + elif printf '%s' "$s" | grep -qE '^(prisma (migrate|db push|db execute|generate)|alembic (upgrade|downgrade)|knex migrate|typeorm migration:(run|revert)|drizzle-kit (push|migrate|generate)|supabase db (push|reset)|rails (db:migrate|db:reset|generate|g) |flyway (migrate|clean)|liquibase (update|rollback)|dotnet ef database update|php artisan migrate|sequelize db:migrate|django-admin migrate|manage\.py migrate)'; then + risky=1; what="migration or generator run" + elif printf '%s' "$s" | grep -qE '^(openapi-generator|graphql-codegen|swagger-codegen|protoc |codegen|orval|kubb|nx g |ng (generate|g) |yo |hygen |plop)'; then + risky=1; what="code generator run" + elif printf '%s' "$s" | grep -qE '^(npm (update|upgrade)|pnpm (up|update|upgrade)|yarn (upgrade|up)|pip install .*(-U|--upgrade)|poetry update|cargo update|bundle update|go get -u|ncu -u|npm-check-updates -u|composer update)'; then + risky=1; what="dependency upgrade" + elif printf '%s' "$s" | grep -qE '^(git mv |find .* -exec |find .* -delete|sed -i[^ ]* .*(\*|\$\(|`)|rm -[A-Za-z]*[rR]|jscodeshift|codemod|(prettier --write|eslint --fix)( -[^ ]+)* (\.|[^ ]*\*[^ ]*|[^ .]+)( |$)|black( -[^ ]+)* \.|ruff format( -[^ ]+)*( \.| [^ .]+|$)|gofmt -w( \.| [^ .]+)|cargo fmt|rustfmt$)'; then + risky=1; what="mass file operation" + elif printf '%s' "$s" | grep -qE '^git (reset --hard|checkout -- |checkout \.|restore( --staged)? \.|restore( --staged)? -- |clean -[a-zA-Z]*f|stash (drop|clear))'; then + risky=1; what="discarding changes" + fi + [ -n "$risky" ] && break +done <<< "$segments" +[ -n "$risky" ] || exit 0 + +git -C "$cwd" rev-parse --is-inside-work-tree >/dev/null 2>&1 || exit 0 +dirty="$(git -C "$cwd" status --porcelain 2>/dev/null | wc -l | tr -d ' ')" +[ "${dirty:-0}" -gt 0 ] || exit 0 +block "$what with $dirty uncommitted file(s): $cmd" \ + "Rule (SYSTEM.md §4): before a data migration, generator run, dependency upgrade, mass file operation or discarding changes a git checkpoint must exist – a clean commit, a clear stash or a documented baseline commit." diff --git a/.claude/hooks/guard-git.sh b/.claude/hooks/guard-git.sh new file mode 100755 index 0000000..f36ee6b --- /dev/null +++ b/.claude/hooks/guard-git.sh @@ -0,0 +1,198 @@ +#!/usr/bin/env bash +# ============================================================================= +# Push guard – fluory-system (PreToolUse hook, matcher: Bash) +# Template: Entwicklungsplan/templates/base/.claude/hooks/guard-git.sh +# +# Blocks in EVERY Claude Code session of this repo (SYSTEM.md §5, rules 1 + 8): +# - git push to main/master – explicit, via refspec (HEAD:main, x:main, +main), +# via upstream default (bare `git push` on main), --all/--mirror +# - force push (--force, -f, +refspec); --force-with-lease only on non-main branches +# - deleting main/master on the remote +# - git commit/merge/rebase/cherry-pick/revert/am ON main/master (once commits exist) +# - --no-verify (commit -n, push --no-verify) +# +# Behaviour: exit 2 + reason on stderr → Claude sees the reason and takes the rule path. +# Without jq/python3/node: fail open (exit 0) – the session card warns visibly. +# Protected branches: FLUORY_PROTECTED_BRANCHES="main master" (space separated). +# Chains, pipes, subshells, env prefixes and wrappers (bash -c, sh -c, eval, xargs) are unwrapped; +# heredoc bodies are never scanned. Known limit: variable indirection (c="git push …"; $c). +# Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +set -f # no glob expansion while tokenising the command + +PROTECTED="${FLUORY_PROTECTED_BRANCHES:-main master}" +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .tool_input.command + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +have_json_tool() { + command -v jq >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1 || command -v node >/dev/null 2>&1 +} +is_protected() { local b; for b in $PROTECTED; do [ "$1" = "$b" ] && return 0; done; return 1; } +current_branch() { git -C "$1" rev-parse --abbrev-ref HEAD 2>/dev/null || true; } +has_commits() { git -C "$1" rev-parse --verify -q HEAD >/dev/null 2>&1; } +strip_heredocs() { # heredoc bodies (<&2 + exit 2 +} + +have_json_tool || exit 0 +cmd="$(json_get .tool_input.command)" +[ -n "$cmd" ] || exit 0 +case "$cmd" in *git*) ;; *) exit 0 ;; esac + +cwd="$(json_get .cwd)" +if [ -z "$cwd" ] || [ ! -d "$cwd" ]; then cwd="${CLAUDE_PROJECT_DIR:-$PWD}"; fi + +# split command chains into single commands (&&, ||, ;, |, newline) – without heredoc bodies +segments="$(printf '%s\n' "$cmd" | strip_heredocs | awk '{ gsub(/&&|\|\||;|\||\$\(|`/, "\n"); print }')" + +check_push() { # $1 = index of the first token after "push"; uses toks, branch + local j t force=0 lease=0 noverify=0 delete=0 all=0 remote="" dst src + local refspecs=() targets=() + for (( j=$1; j<${#toks[@]}; j++ )); do + t="${toks[$j]}" + case "$t" in + --) ;; + --force) force=1 ;; + --force-with-lease|--force-with-lease=*|--force-if-includes) lease=1 ;; + --no-verify) noverify=1 ;; + --delete) delete=1 ;; + --all|--mirror|--branches) all=1 ;; + --repo=*|--receive-pack=*|--exec=*|--push-option=*|--signed=*|--recurse-submodules=*) ;; + --repo|--receive-pack|--exec|-o|--push-option) j=$((j+1)) ;; + --*) ;; + -*) case "$t" in *f*) force=1 ;; esac + case "$t" in *d*) delete=1 ;; esac ;; + *) if [ -z "$remote" ]; then remote="$t"; else refspecs+=("$t"); fi ;; + esac + done + if [ "$all" -eq 1 ]; then + block "git push --all/--mirror (would push $PROTECTED too)." \ + "Rule 1: main only via PR – push only your own branch: git push -u origin ." + fi + if [ ${#refspecs[@]} -eq 0 ]; then + if [ -n "$branch" ] && [ "$branch" != "HEAD" ]; then targets+=("$branch"); fi + else + for t in "${refspecs[@]}"; do + case "$t" in +*) force=1; t="${t#+}" ;; esac + case "$t" in + *:*) src="${t%%:*}"; dst="${t#*:}"; [ -n "$dst" ] || dst="$src"; [ -n "$src" ] || delete=1 ;; + *) dst="$t" ;; + esac + [ "$dst" = "HEAD" ] && dst="$branch" + dst="${dst#refs/heads/}" + targets+=("$dst") + done + fi + if [ "$noverify" -eq 1 ]; then + block "git push --no-verify." "Checks are never bypassed (SYSTEM.md §11, rule 16)." + fi + if [ "$force" -eq 1 ]; then + block "force push (git ${toks[*]})." \ + "Force push is blocked – the history of a pushed branch is never rewritten (§6: handover runs through commits). After a rebase on your own claude/ branch: --force-with-lease, noted in the PR." + fi + if [ ${#targets[@]} -gt 0 ]; then + for dst in "${targets[@]}"; do + if is_protected "$dst"; then + if [ "$delete" -eq 1 ]; then + block "deleting '$dst' on the remote." "Rule 1: $dst is the only verified product state." + fi + block "git push to '$dst'." \ + "Rule 1 (AGENTS.md): $dst only via PR. Way: git switch -c claude/-- → git push -u origin → draft PR → review by the reviewer role." + fi + done + fi +} + +while IFS= read -r seg; do + seg="${seg#"${seg%%[![:space:]]*}"}" # strip leading whitespace + [ -n "$seg" ] || continue + # shellcheck disable=SC2206 # deliberate: naive tokenisation is enough for git commands + toks=( $seg ) + for k in "${!toks[@]}"; do # strip subshell/quote remnants around tokens + t="${toks[$k]}"; t="${t%)}"; t="${t%\`}"; t="${t%\"}"; t="${t%\'}"; t="${t#\"}"; t="${t#\'}"; toks[$k]="$t" + done + n=${#toks[@]} + i=0 + while [ "$i" -lt "$n" ]; do # skip prefixes: env assignments, wrappers, shells (bash -c), eval, xargs and their options + case "${toks[$i]}" in + [A-Za-z_]*=*) ;; + command|sudo|exec|time|nohup|env|builtin|eval|xargs|bash|sh|zsh|dash|ksh|*/bash|*/sh|*/zsh|*/dash) ;; + -*) ;; + *) break ;; + esac + i=$((i+1)) + done + [ "$i" -lt "$n" ] || continue + case "${toks[$i]}" in git|*/git) ;; *) continue ;; esac + i=$((i+1)) + gitdir="$cwd" + while [ "$i" -lt "$n" ]; do # global git options + case "${toks[$i]}" in + -C) i=$((i+1)); d="${toks[$i]}" + case "$d" in /*) gitdir="$d" ;; *) gitdir="$cwd/$d" ;; esac ;; + -c|--git-dir|--work-tree|--namespace|--exec-path) i=$((i+1)) ;; + -*) ;; + *) break ;; + esac + i=$((i+1)) + done + [ "$i" -lt "$n" ] || continue + sub="${toks[$i]}"; i=$((i+1)) + branch="$(current_branch "$gitdir")" + case "$sub" in + push) check_push "$i" ;; + commit|merge|rebase|cherry-pick|revert|am) + for (( j=i; j--, commit there, push, open a draft PR." + fi ;; + esac +done <<< "$segments" + +exit 0 diff --git a/.claude/hooks/guard-read.sh b/.claude/hooks/guard-read.sh new file mode 100755 index 0000000..b22a7cf --- /dev/null +++ b/.claude/hooks/guard-read.sh @@ -0,0 +1,175 @@ +#!/usr/bin/env bash +# ============================================================================= +# Read guard – fluory-system (PreToolUse hook, matchers: Read and Bash) +# Template: Entwicklungsplan/templates/base/.claude/hooks/guard-read.sh +# +# Two classes of paths are not context (SYSTEM.md §10): +# ALWAYS blocked – secrets: .env and environment files (.env.example/.sample/.template stay +# readable), credentials/, secrets/, private-keys/, production-dumps/, key and access files +# (*.pem, *.key, *.p12, *.pfx, *.jks, *.keystore, *.ppk, id_rsa*/id_dsa*/id_ecdsa*/ +# id_ed25519*, .netrc, .pgpass, .git-credentials, .htpasswd, secrets.*/credentials.* config +# files, *service-account*.json, *.tfstate), customer data exports (*kundendaten*, +# *customer-data*, *.dump, *.sql.gz). permissions.deny in .claude/settings.json blocks them +# as well; this hook adds the reason and covers more programs. No exception. +# NORMALLY blocked – build artifacts: node_modules/, dist/, build/, coverage/, .next/, +# .turbo/, .cache/. Exception for one task, explicitly: FLUORY_ALLOW_ARTIFACTS=1 as command +# prefix (Bash) or in the session environment (Read tool) – name the reason in the PR. +# Checked: reading programs (cat, head, tail, sed, awk, grep, rg, source, editors …: secrets +# and artifacts), carriers that copy, encode, move or transmit a file (cp, base64, gzip, tar, +# dd, openssl, curl, mv, ln, interpreters with a path argument …: secrets only) and git +# subcommands that print or stage file contents (show, cat-file, diff, log, blame, grep, +# restore, checkout: both classes; add: secrets) – through symlinks, globs (cat .env*), +# subshells, bash -c / eval, xargs pipelines and name=value arguments (dd if=, --file=, +# -F f=@). Test and build commands (npm test, npm run build, vitest, node …) are never touched. +# Known limits – a text guard cannot see them; from P1 the sandbox option (filesystem.denyRead) +# is the enforcement: interpreter one-liners (python3 -c, node -e), variable indirection +# (c="cat .env"; $c) and searches without a file name (rg SECRET . – rg honours .gitignore, +# so keep .env ignored). +# Fail open without jq/python3/node. Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +set -f +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .tool_input.command + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +have_json_tool() { + command -v jq >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1 || command -v node >/dev/null 2>&1 +} +strip_heredocs() { # heredoc bodies (<&2; exit 2; } +SECRET_MSG="Secrets are never read, printed, copied or committed (SYSTEM.md §10, rule 7). No exception – variable names live in .env.example; runtime values come from the secret manager." +ARTIFACT_MSG="node_modules, dist, build, coverage, .next, .turbo and .cache are not context (SYSTEM.md §10). Needed for this task? Bash: prefix FLUORY_ALLOW_ARTIFACTS=1 · Read tool: start the session with FLUORY_ALLOW_ARTIFACTS=1 (or set it under env in .claude/settings.local.json) – and name the reason in the PR." + +is_secret() { # $1 = path in any form + local p="$1" b l; b="${p##*/}"; l="$(printf '%s' "$p" | tr '[:upper:]' '[:lower:]')" + case "$b" in .env.example|.env.sample|.env.template|.env.dist|.env.schema) return 1 ;; esac + case "$b" in .env|.env.*) return 0 ;; esac + case "$p" in */credentials/*|credentials/*|*/secrets/*|secrets/*|*/private-keys/*|private-keys/*|*/production-dumps/*|production-dumps/*) return 0 ;; esac + case "$b" in *.pem|*.key|*.p12|*.pfx|*.jks|*.keystore|*.ppk|id_rsa*|id_dsa*|id_ecdsa*|id_ed25519*|*.dump|*.sql.gz|*.tfstate|*.tfstate.backup) return 0 ;; esac + case "$b" in .netrc|_netrc|.pgpass|.git-credentials|.htpasswd|credentials|secrets.json|secrets.yml|secrets.yaml|secrets.toml|secrets.ini|secrets.properties|secret.json|secret.yml|secret.yaml|credentials.json|credentials.yml|credentials.yaml|credentials.toml|credentials.ini|credentials.xml) return 0 ;; esac + case "$l" in *kundendaten*|*customer-data*|*customerdata*|*service-account*.json|*service_account*.json|*serviceaccount*.json) return 0 ;; esac + return 1 +} +is_artifact() { + case "$1" in + */node_modules/*|node_modules/*|*/dist/*|dist/*|*/build/*|build/*|*/coverage/*|coverage/*|*/.next/*|.next/*|*/.turbo/*|.turbo/*|*/.cache/*|.cache/*) return 0 ;; + esac + return 1 +} +is_reader() { # programs that print or open file contents – secrets and artifacts are checked + case "$1" in cat|head|tail|less|more|bat|sed|awk|grep|rg|ag|egrep|fgrep|strings|xxd|od|hexdump|cut|sort|uniq|nl|tac|diff|source|.|vim|vi|nvim|nano|emacs|view) return 0 ;; esac + return 1 +} +is_carrier() { # programs that copy, encode, move or transmit a file – secrets are checked, artifacts may be handled + case "$1" in cp|scp|rsync|sftp|mv|ln|install|base64|base32|gzip|gunzip|zcat|bzip2|bzcat|xz|xzcat|zstd|zip|unzip|tar|7z|7za|dd|openssl|split|iconv|paste|join|comm|curl|wget|nc|ncat|socat|python|python2|python3|node|deno|bun|ruby|perl|php) return 0 ;; esac + return 1 +} +is_git_reader() { # git subcommands that print file contents from history or the tree + case "$1" in show|cat-file|diff|log|blame|grep|archive|restore|checkout) return 0 ;; esac + return 1 +} +clean() { # strip subshell, backtick, parenthesis and quote characters around a token + local c="$1" + c="${c#\$(}"; c="${c#\`}"; c="${c#(}"; c="${c#\"}"; c="${c#\'}" + c="${c%\`}"; c="${c%)}"; c="${c%\"}"; c="${c%\'}" + printf '%s' "$c" +} +resolve() { # symlinks: the real target counts too + local p="$1" r="" + case "$p" in /*) ;; *) p="$cwd/$p" ;; esac + [ -e "$p" ] || { printf '%s' "$1"; return; } + r="$(readlink -f "$p" 2>/dev/null || realpath "$p" 2>/dev/null || true)" + printf '%s' "${r:-$1}" +} +secret_path() { is_secret "$1" || is_secret "$(resolve "$1")"; } +expand() { # a glob reads the files it matches – check those, not only the pattern text + case "$1" in + *\**|*\?*|*\[*) ( cd "$cwd" 2>/dev/null || exit 0; set +f; for m in $1; do [ -e "$m" ] && printf '%s\n' "$m"; done; printf '%s\n' "$1" ) ;; + *) printf '%s\n' "$1" ;; + esac +} + +have_json_tool || exit 0 +tool="$(json_get .tool_name)" +allow_art=0; [ "${FLUORY_ALLOW_ARTIFACTS:-0}" = "1" ] && allow_art=1 +cwd="$(json_get .cwd)" +{ [ -n "$cwd" ] && [ -d "$cwd" ]; } || cwd="${CLAUDE_PROJECT_DIR:-$PWD}" + +if [ "$tool" = "Read" ]; then + f="$(json_get .tool_input.file_path)"; [ -n "$f" ] || exit 0 + secret_path "$f" && block "reading a secret file: $f" "$SECRET_MSG" + if [ "$allow_art" -eq 0 ] && is_artifact "$f"; then block "reading a build artifact: $f" "$ARTIFACT_MSG"; fi + exit 0 +fi + +cmd="$(json_get .tool_input.command)" +[ -n "$cmd" ] || exit 0 +case "$cmd" in *FLUORY_ALLOW_ARTIFACTS=1*) allow_art=1 ;; esac +segments="$(printf '%s\n' "$cmd" | strip_heredocs | awk '{ gsub(/&&|\|\||;|\||\$\(|`/, "\n"); print }')" +# shellcheck disable=SC2206 # deliberate: naive tokenisation is enough for path arguments +all_toks=( $(printf '%s' "$segments" | tr '\n' ' ') ) # xargs: the file names arrive from other segments +while IFS= read -r seg; do + seg="${seg#"${seg%%[![:space:]]*}"}" + [ -n "$seg" ] || continue + # shellcheck disable=SC2206 + toks=( $seg ) + reader=0; carrier=0; prog=""; in_git=0; via_xargs=0 + for t in "${toks[@]}"; do + c="$(clean "$t")" + case "$c" in xargs) via_xargs=1; continue ;; esac + case "$c" in [A-Za-z_]*=*|command|sudo|exec|time|nohup|env|builtin|timeout|nice) continue ;; esac + if is_reader "$c"; then reader=1; [ -n "$prog" ] || prog="$c"; fi + if is_carrier "$c"; then carrier=1; [ -n "$prog" ] || prog="$c"; fi + [ "$c" = "git" ] && in_git=1 + if [ "$in_git" -eq 1 ]; then + if is_git_reader "$c"; then reader=1; prog="git $c"; fi + case "$c" in add|stage) carrier=1; prog="git $c" ;; esac + fi + done + [ "$reader" -eq 1 ] || [ "$carrier" -eq 1 ] || continue + scan=("${toks[@]}"); [ "$via_xargs" -eq 1 ] && scan=("${all_toks[@]}") + for t in "${scan[@]}"; do + c="$(clean "$t")" + case "$c" in -*=*|[A-Za-z_]*=*) c="${c#*=}" ;; -*|"") continue ;; esac # dd if=.env, --file=.env, -F f=@.env + c="${c#@}" + case "$c" in *:*) c="${c#*:}" ;; esac # git show :, host:path + [ -n "$c" ] || continue + for p in $(expand "$c"); do + secret_path "$p" && block "secret file as argument of $prog: $p" "$SECRET_MSG" + if [ "$reader" -eq 1 ] && [ "$allow_art" -eq 0 ] && is_artifact "$p"; then block "reading a build artifact with $prog: $p" "$ARTIFACT_MSG"; fi + done + done +done <<< "$segments" +exit 0 diff --git a/.claude/hooks/pre-compact.sh b/.claude/hooks/pre-compact.sh new file mode 100755 index 0000000..33ee9f0 --- /dev/null +++ b/.claude/hooks/pre-compact.sh @@ -0,0 +1,174 @@ +#!/usr/bin/env bash +# ============================================================================= +# Compaction snapshot – fluory-system (PreCompact hook: manual and auto) +# Template: Entwicklungsplan/templates/base/.claude/hooks/pre-compact.sh +# +# Before every context compaction this hook writes a small, local, ephemeral snapshot of the +# work state OUTSIDE the repository (SYSTEM.md §6): session id, branch, draft PR, issue, changed +# files, affected tests, last verify result, the PR section "Arbeitsstand" and the agent's last +# message – all verbatim from git, gh and the transcript; missing information is marked unknown, +# nothing is inferred. No file contents; secrets are redacted. session-start.sh replays the snapshot after the +# compaction. The snapshot is neither a handover document nor project knowledge and is never +# committed – handover lives in the draft PR (section "Arbeitsstand"). +# +# Location: ${FLUORY_STATE_DIR:-$HOME/.claude/fluory-state}//snapshot-.md +# Snapshots older than 14 days are removed. Never blocks a compaction: always exit 0. +# Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .trigger + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +have_json_tool() { + command -v jq >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1 || command -v node >/dev/null 2>&1 +} +last_note() { # $1 = transcript (jsonl) → last assistant text, max. 700 characters + [ -f "$1" ] || return 0 + if command -v python3 >/dev/null 2>&1; then + python3 - "$1" <<'PY' 2>/dev/null +import json, sys +last = '' +for line in open(sys.argv[1], encoding='utf-8', errors='replace'): + try: + o = json.loads(line) + except Exception: + continue + if o.get('type') != 'assistant': + continue + c = (o.get('message') or {}).get('content') + if isinstance(c, str): + t = c + elif isinstance(c, list): + t = '\n'.join(b.get('text', '') for b in c if isinstance(b, dict) and b.get('type') == 'text') + else: + continue + if t.strip(): + last = t +print(last.strip()[-700:]) +PY + elif command -v jq >/dev/null 2>&1; then + jq -R -r 'fromjson? | select(.type=="assistant") | (.message // {}).content | if type=="string" then . elif type=="array" then (map(select(.type=="text") | .text) | join("\n")) else empty end' "$1" 2>/dev/null | awk 'NF' | tail -n 8 | tail -c 700 + elif command -v node >/dev/null 2>&1; then + node -e ' +const fs = require("fs"); let last = ""; +for (const line of fs.readFileSync(process.argv[1], "utf8").split("\n")) { + let o; try { o = JSON.parse(line); } catch (e) { continue; } + if (o.type !== "assistant") continue; + const c = (o.message || {}).content; let t = ""; + if (typeof c === "string") t = c; + else if (Array.isArray(c)) t = c.filter(b => b && b.type === "text").map(b => b.text || "").join("\n"); + else continue; + if (t.trim()) last = t; +} +process.stdout.write(last.trim().slice(-700));' "$1" 2>/dev/null + fi +} +redact() { # secrets never enter the snapshot – not even through the agent's own notes + local kw='([Aa][Pp][Ii][_-]?[Kk][Ee][Yy]|[Ss][Ee][Cc][Rr][Ee][Tt]|[Tt][Oo][Kk][Ee][Nn]|[Pp][Aa][Ss][Ss][Ww]([Oo][Rr])?[Dd]|[Cc][Rr][Ee][Dd][Ee][Nn][Tt][Ii][Aa][Ll]|[Bb][Ee][Aa][Rr][Ee][Rr])' + sed -E \ + -e 's/(-----BEGIN [A-Z ]*PRIVATE KEY-----).*/\1 [redacted]/' \ + -e 's/(^|[^A-Za-z0-9])(AKIA|ASIA)[0-9A-Z]{16}/\1[redacted]/g' \ + -e 's/(^|[^A-Za-z0-9])(ghp|gho|ghu|ghs|github_pat)_[A-Za-z0-9_]{20,}/\1[redacted]/g' \ + -e 's/(^|[^A-Za-z0-9])sk-[A-Za-z0-9_-]{16,}/\1[redacted]/g' \ + -e 's/(^|[^A-Za-z0-9])xox[baprs]-[A-Za-z0-9-]{10,}/\1[redacted]/g' \ + -e 's/eyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}/[redacted]/g' \ + -e "s/(${kw}[A-Za-z0-9_-]*[[:space:]]*[=:][[:space:]]*)[^[:space:]]+/\\1[redacted]/g" +} +gh_cmd() { if command -v timeout >/dev/null 2>&1; then timeout 8 gh "$@"; else gh "$@"; fi; } + +have_json_tool || exit 0 +sid="$(json_get .session_id)"; [ -n "$sid" ] || sid="unknown" +trigger="$(json_get .trigger)"; [ -n "$trigger" ] || trigger="unknown" +custom="$(json_get .custom_instructions)" +transcript="$(json_get .transcript_path)" +cwd="$(json_get .cwd)" +{ [ -n "$cwd" ] && [ -d "$cwd" ]; } || cwd="${CLAUDE_PROJECT_DIR:-$PWD}" +root="$(git -C "$cwd" rev-parse --show-toplevel 2>/dev/null || true)" +[ -n "$root" ] || exit 0 + +STATE="${FLUORY_STATE_DIR:-$HOME/.claude/fluory-state}" +key="$(printf '%s' "$root" | sed 's#[^A-Za-z0-9._-]#_#g')" +dir="$STATE/$key" +mkdir -p "$dir" 2>/dev/null || exit 0 +find "$dir" -type f -mtime +14 -delete 2>/dev/null +snap="$dir/snapshot-$sid.md" +count_file="$dir/compactions-$sid" +n=0; [ -f "$count_file" ] && n="$(tr -dc '0-9' < "$count_file")"; n=$(( ${n:-0} + 1 )); printf '%s\n' "$n" > "$count_file" + +branch="$(git -C "$root" rev-parse --abbrev-ref HEAD 2>/dev/null || echo '?')" +issue="$(printf '%s' "$branch" | sed -nE 's/.*-([0-9]+)$/#\1/p')"; [ -n "$issue" ] || issue="(no issue number in the branch name)" +up="$(git -C "$root" rev-parse --abbrev-ref '@{u}' 2>/dev/null || true)" +if [ -n "$up" ]; then + ahead="$(git -C "$root" rev-list --count '@{u}..HEAD' 2>/dev/null || echo 0)" + behind="$(git -C "$root" rev-list --count 'HEAD..@{u}' 2>/dev/null || echo 0)" + sync="upstream $up · $ahead ahead · $behind behind" +else + sync="no upstream – never pushed" +fi +last="$(git -C "$root" log -1 --format='%h %s' 2>/dev/null || echo '(no commits)')" +dirty="$(git -C "$root" status --porcelain 2>/dev/null | head -40)" +ndirty="$(git -C "$root" status --porcelain 2>/dev/null | wc -l | tr -d ' ')" +changed="$( { git -C "$root" diff --name-only HEAD 2>/dev/null; git -C "$root" ls-files --others --exclude-standard 2>/dev/null; [ -n "$up" ] && git -C "$root" diff --name-only "$up...HEAD" 2>/dev/null; } | awk 'NF' | sort -u | head -60)" +tests="$(printf '%s\n' "$changed" | grep -E '(^|/)(tests?|__tests__|e2e|spec)/|\.(test|spec)\.[A-Za-z]+$|_test\.(go|py|rs)$' || true)" +pr="unknown (gh not available)" +if command -v gh >/dev/null 2>&1; then + pr="$(gh_cmd pr list --head "$branch" --state open --json number,isDraft,url --jq '.[] | "#\(.number)\(if .isDraft then " (draft)" else "" end) \(.url)"' 2>/dev/null | head -1)" + [ -n "$pr" ] || pr="none open for $branch – open a draft PR (SYSTEM.md §6)" +fi +verify="(none recorded – run verify:changed or verify through scripts/quiet-run.sh)" +[ -f "$dir/last-verify.txt" ] && verify="$(head -6 "$dir/last-verify.txt")" +note="$(last_note "$transcript")"; [ -n "$note" ] || note="unknown (no transcript available)" +prstate="unknown (gh not available)" +if command -v gh >/dev/null 2>&1; then + prstate="$(gh_cmd pr view --json body -q .body 2>/dev/null | awk '/^## Arbeitsstand/{f=1; next} /^## /{f=0} f' | awk 'NF' | head -12)" + [ -n "$prstate" ] || prstate="unknown (no open PR for this branch or no Arbeitsstand section)" +fi + +{ + echo "# Work snapshot · fluory-system" + echo "- Written: $(date '+%Y-%m-%d %H:%M:%S') · trigger: $trigger · compaction no. $n in this session" + echo "- Session-ID: $sid" + echo "- Repo: $root" + echo "- Branch: $branch · $sync · last commit: $last" + echo "- Draft-PR: $pr" + echo "- Issue: $issue" + echo "- Uncommitted: $ndirty file(s)" + [ -n "$dirty" ] && printf '%s\n' "$dirty" | sed 's/^/ /' + echo "- Changed files (working tree and commits ahead of upstream):" + if [ -n "$changed" ]; then printf '%s\n' "$changed" | sed 's/^/ /'; else echo " (none)"; fi + echo "- Affected tests (by path):" + if [ -n "$tests" ]; then printf '%s\n' "$tests" | sed 's/^/ /'; else echo " (none among the changed files – pick the focused test for the touched module)"; fi + echo "- Last verify result (scripts/quiet-run.sh):" + printf '%s\n' "$verify" | sed 's/^/ /' + echo "- Compaction instructions given: ${custom:-(none)}" + echo "- Draft PR section \"Arbeitsstand\" (verbatim – the only durable source for goal, open items and next step):" + printf '%s\n' "$prstate" | sed 's/^/ /' + echo "- Last message of the agent before the compaction (verbatim from the transcript, not interpreted):" + printf '%s\n' "$note" | sed 's/^/ /' + echo "- Hypothesis, risk and next smallest step: nothing is inferred here. If the two sources above do not state them, they are unknown – re-orient from issue → PR → diff → tests before editing (SYSTEM.md §6)." +} | redact > "$snap" +echo "[fluory-system] compaction snapshot written: $snap" +exit 0 diff --git a/.claude/hooks/session-start.sh b/.claude/hooks/session-start.sh new file mode 100755 index 0000000..fedee01 --- /dev/null +++ b/.claude/hooks/session-start.sh @@ -0,0 +1,195 @@ +#!/usr/bin/env bash +# ============================================================================= +# Session card – fluory-system (SessionStart hook: startup, resume, clear, compact) +# Template: Entwicklungsplan/templates/base/.claude/hooks/session-start.sh +# +# Prints the rule context into the session at every start, resume and AFTER EVERY CONTEXT +# COMPACTION (stdout → context): repo/branch, safety state, profile version, size of the +# mandatory reading, guard status, open PRs and the short rules from AGENTS.md (section +# "## Rules (short form)"; "## Regeln (Kurzfassung)" is still accepted). After a compaction or +# resume it replays the local work snapshot written by pre-compact.sh and, from the second +# compaction on, the checkpoint questions (SYSTEM.md §6). In the control center additionally: +# drift-check summary and age of the portfolio state. Runs in seconds, changes nothing. +# ============================================================================= + +PROTECTED="${FLUORY_PROTECTED_BRANCHES:-main master}" +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .source + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +have_json_tool() { + command -v jq >/dev/null 2>&1 || command -v python3 >/dev/null 2>&1 || command -v node >/dev/null 2>&1 +} +is_protected() { local b; for b in $PROTECTED; do [ "$1" = "$b" ] && return 0; done; return 1; } +lines() { wc -l < "$1" | tr -d ' '; } +days_since() { # $1 = YYYY-MM-DD + local t + t=$(date -d "$1" +%s 2>/dev/null || date -j -f '%Y-%m-%d' "$1" +%s 2>/dev/null) || return 1 + echo $(( ( $(date +%s) - t ) / 86400 )) +} +gh_cmd() { if command -v timeout >/dev/null 2>&1; then timeout 8 gh "$@"; else gh "$@"; fi; } +W() { printf ' ⚠ %s' "$1"; } # append a warning to a line +WL() { printf -- '- ⚠ %s\n' "$1"; } # warning as its own line + +src="$(json_get .source)"; [ -n "$src" ] || src="startup" +sid="$(json_get .session_id)"; [ -n "$sid" ] || sid="unknown" +cwd="$(json_get .cwd)" +root="${CLAUDE_PROJECT_DIR:-}" +{ [ -n "$root" ] && [ -d "$root" ]; } || root="$cwd" +{ [ -n "$root" ] && [ -d "$root" ]; } || root="$PWD" +cd "$root" 2>/dev/null || exit 0 + +echo "## Session card · fluory-system · $src" +name="$(basename "$root")" + +# --- Repo & safety ------------------------------------------------------------------------ +if git rev-parse --is-inside-work-tree >/dev/null 2>&1; then + branch="$(git rev-parse --abbrev-ref HEAD 2>/dev/null || echo '?')" + l="- Repo: $name · branch: $branch" + is_protected "$branch" && l="$l$(W "on $branch – never commit here: git switch -c claude/-- (rule 1)")" + echo "$l" + dirty="$(git status --porcelain 2>/dev/null | wc -l | tr -d ' ')" + if git rev-parse --abbrev-ref '@{u}' >/dev/null 2>&1; then + ahead="$(git rev-list --count '@{u}..HEAD' 2>/dev/null || echo 0)"; up="yes" + else + ahead=0; up="no" + fi + wt="$(git worktree list 2>/dev/null | wc -l | tr -d ' ')" + l="- Safety: $dirty uncommitted · $ahead unpushed · upstream: $up · worktrees: $wt" + [ "${dirty:-0}" -gt 0 ] && l="$l$(W "uncommitted work – commit and push early (rule 4)")" + [ "${ahead:-0}" -gt 0 ] && l="$l$(W "unpushed commits")" + echo "$l" +else + echo "- Repo: $name (not a git working tree)" +fi + +# --- Profile ------------------------------------------------------------------------------ +if [ -f project-profile.yml ]; then + stage="$(sed -n 's/^[[:space:]]*stage:[[:space:]]*\([a-z]*\).*/\1/p' project-profile.yml | head -1)" + ver="$(sed -n 's/^[[:space:]]*fluory_system_version:[[:space:]]*"\{0,1\}\([0-9][0-9.]*\)"\{0,1\}.*/\1/p' project-profile.yml | head -1)" + rev="$(sed -n 's/^[[:space:]]*last_system_review:[[:space:]]*\([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\).*/\1/p' project-profile.yml | head -1)" + l="- Profile: stage ${stage:-?} · fluory-system ${ver:-?}" + [ -n "$ver" ] || l="$l$(W "system: version missing in the profile (§3)")" + if [ -n "$rev" ]; then + d="$(days_since "$rev" 2>/dev/null || true)" + if [ -n "$d" ]; then + l="$l · last system review $rev ($d days)" + [ "$d" -gt 90 ] && l="$l$(W "over 90 days – reconcile with Entwicklungsplan/templates/VERSION (§12)")" + fi + else + l="$l$(W "last_system_review not set – set it at the next reconciliation (§12)")" + fi + echo "$l" +fi + +# --- Mandatory reading -------------------------------------------------------------------- +if [ -f AGENTS.md ]; then + n="$(lines AGENTS.md)" + l="- Mandatory reading: AGENTS.md $n lines" + [ "$n" -gt 150 ] && l="$l$(W "over 150 – condense instead of extending (§9)")" + for a in docs/ARCHITEKTUR.md docs/technical/architecture.md; do + if [ -f "$a" ]; then + m="$(lines "$a")"; l="$l · $a $m lines" + [ "$m" -gt 150 ] && l="$l$(W "over ~3 screens – condense (§9)")" + fi + done + echo "$l" +fi + +# --- Guards ------------------------------------------------------------------------------- +if have_json_tool; then g="push guard active"; else g="$(W "push guard INACTIVE – install jq, python3 or node")"; fi +if [ -f scripts/doku-check.sh ]; then dc="doku-check present"; else dc="doku-check missing"; fi +[ -f scripts/check-drift.sh ] && dc="check-drift present" +if [ -f .claude/hooks/pre-compact.sh ] || [ -f templates/base/.claude/hooks/pre-compact.sh ]; then sn="compaction snapshot active" +else sn="$(W "pre-compact.sh missing – no snapshot before compaction (§6)")"; fi +if [ -f .claude/settings.json ] && grep -qE '"autoMemoryEnabled":[[:space:]]*false' .claude/settings.json; then am="auto memory off" +else am="$(W "auto memory ON – set autoMemoryEnabled: false in .claude/settings.json (§10)")"; fi +echo "- Guards: $g · stop check active · $dc · $sn · $am" + +# --- Control-center extras ---------------------------------------------------------------- +if [ -f scripts/check-drift.sh ]; then + echo "- $(bash scripts/check-drift.sh --summary 2>/dev/null || echo 'drift check not executable')" +fi +if [ -f PROJEKTE.md ]; then + st="$(sed -n 's/.*Stand: \([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\).*/\1/p' PROJEKTE.md | head -1)" + if [ -n "$st" ]; then + d="$(days_since "$st" 2>/dev/null || true)" + if [ -n "$d" ]; then + l="- Portfolio: PROJEKTE.md state $st ($d days)" + [ "$d" -gt 45 ] && l="$l$(W "older than 45 days – verify the rows at the source, orchestration level only (§3)")" + echo "$l" + fi + fi +fi + +# --- Open PRs (best effort) --------------------------------------------------------------- +if command -v gh >/dev/null 2>&1 && git rev-parse --is-inside-work-tree >/dev/null 2>&1; then + prs="$(gh_cmd pr list --state open --limit 5 --json number,title,isDraft \ + --jq '.[] | "#\(.number)\(if .isDraft then " (draft)" else "" end) \(.title)"' 2>/dev/null \ + | paste -sd '|' - | sed 's/|/ · /g')" + [ -n "$prs" ] && echo "- Open PRs: $prs" +fi + +# --- Short rules from AGENTS.md ----------------------------------------------------------- +if [ -f AGENTS.md ]; then + rules="$(awk '/^## (Regeln|Rules)/{f=1; next} /^## /{f=0} f' AGENTS.md | sed -e '//d' -e '//d' | cat -s | sed '/./,$!d')" + if [ -n "$(printf '%s' "$rules" | tr -d '[:space:]')" ]; then + echo + echo "### Rules (short form) – from AGENTS.md" + printf '%s\n' "$rules" + else + WL "section \"## Rules (short form)\" (or \"## Regeln (Kurzfassung)\") not found in AGENTS.md – restore the heading (template: templates/base/AGENTS.md)" + fi +else + WL "AGENTS.md missing – adopt the project card from Entwicklungsplan/templates/base/ (§3)" +fi + +# --- Work snapshot after compaction or resume (§6) ---------------------------------------- +STATE="${FLUORY_STATE_DIR:-$HOME/.claude/fluory-state}" +key="$(git rev-parse --show-toplevel 2>/dev/null || printf '%s' "$root")" +key="$(printf '%s' "$key" | sed 's#[^A-Za-z0-9._-]#_#g')" +dir="$STATE/$key" +snap="$dir/snapshot-$sid.md" +cnt=0; [ -f "$dir/compactions-$sid" ] && cnt="$(tr -dc '0-9' < "$dir/compactions-$sid")"; cnt="${cnt:-0}" +if [ "$src" = "compact" ] || [ "$src" = "resume" ]; then + echo + if [ -f "$snap" ]; then + echo "### Work snapshot (written by pre-compact.sh before the compaction – ephemeral, not a handover)" + head -60 "$snap" + else + echo "- No work snapshot for this session found ($src) – re-orient from issue, draft PR and git diff." + fi + if [ "$cnt" -ge 2 ]; then + echo + echo "### Checkpoint after the second compaction (SYSTEM.md §6)" + echo "Answer before continuing: 1. Is the next smallest step clear? 2. Is the draft PR (Arbeitsstand) current? 3. Is there a green focused proof (verify:changed)? 4. Is an architecture or risk decision open?" + echo "If any answer is no or unclear: update the draft PR and start a new session. Otherwise continue focused." + fi +fi + +echo +echo "Re-orientation before editing: issue → draft PR (Arbeitsstand) → git diff → affected tests → area rule (SYSTEM.md §6). Before the end: /finish-work – verify + scripts/doku-check.sh, plain-language section in the PR, commit + push, handover comment when handing off." +exit 0 diff --git a/.claude/hooks/stop-check.sh b/.claude/hooks/stop-check.sh new file mode 100755 index 0000000..15be911 --- /dev/null +++ b/.claude/hooks/stop-check.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +# ============================================================================= +# Stop check – fluory-system (Stop hook) +# Template: Entwicklungsplan/templates/base/.claude/hooks/stop-check.sh +# +# Stops a session exactly ONCE when unsaved work exists at the end (SYSTEM.md §6, rules 4 + 5): +# uncommitted files, unpushed commits, a branch without upstream, a branch without an open PR +# (only checkable when `gh` is installed and logged in). Claude then has to save – or tell the +# human in the chat why not. On the second stop (stop_hook_active = true) the hook lets the +# session end: no endless loop. Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= + +PROTECTED="${FLUORY_PROTECTED_BRANCHES:-main master}" +INPUT="$(cat 2>/dev/null || true)" + +json_get() { # $1 = path like .cwd + if command -v jq >/dev/null 2>&1; then + printf '%s' "$INPUT" | jq -r "$1 // empty" 2>/dev/null + elif command -v python3 >/dev/null 2>&1; then + printf '%s' "$INPUT" | python3 -c ' +import json, sys +keys = [k for k in sys.argv[1].split(".") if k] +try: + v = json.load(sys.stdin) + for k in keys: + v = v[k] +except Exception: + sys.exit(0) +if isinstance(v, bool): + print("true" if v else "false") +elif v is not None: + print(v) +' "$1" 2>/dev/null + elif command -v node >/dev/null 2>&1; then + printf '%s' "$INPUT" | node -e ' +let s = ""; process.stdin.on("data", d => s += d).on("end", () => { + try { let v = JSON.parse(s); for (const k of process.argv[1].split(".").filter(Boolean)) v = v[k]; + if (v !== undefined && v !== null) console.log(String(v)); } catch (e) {} });' "$1" 2>/dev/null + fi +} +is_protected() { local b; for b in $PROTECTED; do [ "$1" = "$b" ] && return 0; done; return 1; } +gh_cmd() { if command -v timeout >/dev/null 2>&1; then timeout 8 gh "$@"; else gh "$@"; fi; } + +[ "$(json_get .stop_hook_active)" = "true" ] && exit 0 # second stop: let the session end + +cwd="$(json_get .cwd)" +if [ -z "$cwd" ] || [ ! -d "$cwd" ]; then cwd="${CLAUDE_PROJECT_DIR:-$PWD}"; fi +cd "$cwd" 2>/dev/null || exit 0 +git rev-parse --is-inside-work-tree >/dev/null 2>&1 || exit 0 + +branch="$(git rev-parse --abbrev-ref HEAD 2>/dev/null || true)" +findings="" +add() { findings="${findings}- $1"$'\n'; } + +dirty="$(git status --porcelain 2>/dev/null | wc -l | tr -d ' ')" +[ "${dirty:-0}" -gt 0 ] && add "$dirty uncommitted/untracked file(s) – git status" + +if [ -n "$branch" ] && [ "$branch" != "HEAD" ]; then + if git rev-parse --abbrev-ref '@{u}' >/dev/null 2>&1; then + ahead="$(git rev-list --count '@{u}..HEAD' 2>/dev/null || echo 0)" + [ "${ahead:-0}" -gt 0 ] && add "$ahead unpushed commit(s) on '$branch'" + if ! is_protected "$branch" && command -v gh >/dev/null 2>&1; then + n="$(gh_cmd pr list --head "$branch" --state open --json number --jq 'length' 2>/dev/null || true)" + [ "$n" = "0" ] && add "no open PR for '$branch' – open a draft PR (rule 5)" + fi + elif ! is_protected "$branch" && git rev-parse --verify -q HEAD >/dev/null 2>&1; then + add "branch '$branch' has no upstream – never pushed (git push -u origin $branch)" + fi +fi + +[ -z "$findings" ] && exit 0 + +{ + echo "[fluory-system stop check] Unsaved work – SYSTEM.md rules 4/5:" + printf '%s' "$findings" + if is_protected "$branch"; then + echo "You are on '$branch': create the branch claude/-- first, then save there." + fi + echo "Now: commit + push + update the PR description (Arbeitsstand, plain-language section) – /finish-work." + echo "If this deliberately should NOT happen (read-only session, someone else's local changes), tell the human in the chat in one sentence what stays unsaved and why – then end." +} >&2 +exit 2 diff --git a/.claude/rules/ai-rag.md b/.claude/rules/ai-rag.md new file mode 100644 index 0000000..75d4eaf --- /dev/null +++ b/.claude/rules/ai-rag.md @@ -0,0 +1,24 @@ +--- +paths: + - "prompts/**" + - "evals/**" + - "src/**/prompts/**" + - "src/**/rag/**" + - "src/**/llm/**" + - "src/**/agents/**" + - "**/*prompt*" + - "**/*embedding*" + - "services/ai/**/prompts/**" + - "services/ai/evals/**" + - "services/ai/src/**/extraction/**" + - "services/ai/src/**/grounding/**" + - "src/features/extraction/**" +--- + +# AI / RAG rules (loaded when prompt, retrieval or agent code is read) + +- Prompts are versioned in the repo; a change to prompt, retrieval or model runs the eval set (`evals/`) before the merge. +- The eval set is weighted by known weaknesses; every computed quality verdict counts in PASS/FAIL; prompt-injection cases are part of the set. +- Fake or mock paths are fail-closed: a failed client init aborts with an error, never silently switches to a mock. +- Retrieval respects permissions: documents only for users allowed to see them; no confidential data to external providers without approval; prompt and output logging without sensitive content. +- High-impact actions (send, delete, order, change rights): agent proposes → system shows target and effect → human confirms → execution with audit log. diff --git a/.claude/rules/api.md b/.claude/rules/api.md new file mode 100644 index 0000000..4085156 --- /dev/null +++ b/.claude/rules/api.md @@ -0,0 +1,22 @@ +--- +paths: + - "src/**/api/**" + - "src/**/routes/**" + - "src/**/controllers/**" + - "src/**/handlers/**" + - "**/openapi*.{yaml,yml,json}" + - "**/*.contract.*" + - "contracts/**" + - "src/app/api/**" + - "services/ai/src/**/api/**" + - "src/features/export/**" + - "src/features/erp-mock/**" +--- + +# API rules (loaded when API code is read) + +- The interface is a promise: change the contract (`docs/technical/api.md` or the OpenAPI file) in the same PR as the code. +- Breaking changes only with a new version (`/v2/`); additive changes stay backwards compatible. Check every schema change for backwards compatibility. +- Every endpoint validates input, uses the project's standard error shape, and has timeouts, rate limits and payload size limits where it is public or expensive. +- Proof at the boundary: a contract or integration test for the touched endpoint runs in `verify`; a public API change needs human approval (SYSTEM.md §5) and triggers the plan gate (§4). +- Never log request bodies with personal data or secrets. diff --git a/.claude/rules/database-migrations.md b/.claude/rules/database-migrations.md new file mode 100644 index 0000000..7bcd00b --- /dev/null +++ b/.claude/rules/database-migrations.md @@ -0,0 +1,19 @@ +--- +paths: + - "src/**/migrations/**" + - "prisma/migrations/**" + - "prisma/schema.prisma" + - "db/migrations/**" + - "supabase/migrations/**" + - "**/alembic/**" + - "src/db/**" + - "drizzle.config.*" +--- + +# Database migration rules (loaded when migration code is read) + +- Test every migration locally on a fresh database before the merge; a migration is an explicit deploy step, never a side effect of the app start. +- Every destructive change (drop, rename, type change, new NOT NULL) needs a backup and rollback decision in the PR and runs as expand → migrate → contract so old app and new schema work in parallel. +- Additive, compatible changes may ride in the feature PR (P0/P1); data transfers are their own step. +- Two parallel branches never make competing migrations merge-ready at the same time: merge the first, then rebase the second. +- Commit a git checkpoint before running a migration or generator locally (`/rewind` does not undo them). Migrations are hard to reverse: human approval (SYSTEM.md §5), plan gate (§4). diff --git a/.claude/rules/frontend-e2e.md b/.claude/rules/frontend-e2e.md new file mode 100644 index 0000000..817131f --- /dev/null +++ b/.claude/rules/frontend-e2e.md @@ -0,0 +1,16 @@ +--- +paths: + - "e2e/**" + - "tests/e2e/**" + - "**/*.e2e.*" + - "playwright.config.*" + - "src/**/*.{tsx,jsx,vue,svelte}" +--- + +# E2E and UI rules (loaded when UI components or browser tests are read) + +- Browser tests are outer-loop proof: run only the affected spec or the defined smoke flows, targeted before the PR or in CI – never after every edit. +- Locators by role, label or test id (`getByRole`, `getByLabel`, `getByTestId`); no CSS classes, no XPath, no copy text that changes with wording. +- Recurring flows (login, navigation, form submit) live in page objects or component objects; a spec reads like the user story. +- A failing E2E run links trace or screenshot in the PR; reproduce the one failing flow first instead of starting new broad runs. +- UI changes from P1: keyboard operation, understandable labels, loading/empty/error states checked (review checklist); styling and layout get a proof of check, not an artificial test. diff --git a/.claude/rules/infrastructure.md b/.claude/rules/infrastructure.md new file mode 100644 index 0000000..df75265 --- /dev/null +++ b/.claude/rules/infrastructure.md @@ -0,0 +1,22 @@ +--- +paths: + - "infra/**" + - "terraform/**" + - "deploy/**" + - "k8s/**" + - "**/Dockerfile*" + - "**/docker-compose*" + - ".github/workflows/**" + - "**/*.tf" + - "**/compose*.y*ml" + - "vercel.ts" + - "vercel.json" +--- + +# Infrastructure rules (loaded when infra or CI code is read) + +- Infrastructure only as code with a mirror in the repo – no click configuration. +- Separate secrets per environment, only in the secret manager or CI secrets; never in git, chat, logs or prompts. +- Least privilege: CI and agents get only the rights for exactly their job; development agents never touch production. +- Changes to infrastructure, CI, hooks or security checks are named explicitly in the PR plain-language section and need human approval (SYSTEM.md §5); larger changes go to staging first; the rollback path is documented in `docs/technical/operations.md`. +- Keep the base CI cheap: PR-only, `cancel-in-progress`, `timeout-minutes`, Linux only, no workflow-wide `paths-ignore` on the mandatory check (SYSTEM.md §11). diff --git a/.claude/rules/security.md b/.claude/rules/security.md new file mode 100644 index 0000000..721adce --- /dev/null +++ b/.claude/rules/security.md @@ -0,0 +1,26 @@ +--- +paths: + - "src/**/auth/**" + - "src/**/security/**" + - "src/**/middleware/**" + - "**/*auth*" + - "**/*permission*" + - "**/*session*" + - "**/*password*" + - "**/*token*" + - "src/features/identity/**" + - "src/features/tenancy/**" + - "src/features/storage/**" + - "**/*rls*" + - "**/*polic*" + - "src/features/intake/**" + - "services/ai/src/**/parsing/**" +--- + +# Security rules (loaded when auth, permission or session code is read) + +- Authorization is decided server-side, never only by hiding UI. Every protected action checks the role or ownership close to the data. +- Rate limits at least for login, password reset, registration and expensive endpoints; account recovery is designed deliberately – it is usually the weakest spot. +- Never log secrets, tokens or personal data; error responses carry no internal stack traces. +- Tests use fakes and synthetic data; never real credentials or production data. +- Any change here touches the risk matrix: human approval (SYSTEM.md §5), plan gate (§4), test-first for permission logic (§11), `/security-review` when installed. diff --git a/.claude/rules/testing.md b/.claude/rules/testing.md new file mode 100644 index 0000000..97de556 --- /dev/null +++ b/.claude/rules/testing.md @@ -0,0 +1,21 @@ +--- +paths: + - "**/*.test.*" + - "**/*.spec.*" + - "**/*_test.*" + - "tests/**" + - "test/**" + - "**/__tests__/**" + - "**/test_*.py" + - "services/ai/tests/**" + - "services/ai/evals/**" +--- + +# Testing rules (loaded when tests are read) + +- Local work is focused: `verify:changed` after every small change (format, typecheck, tests of the touched files); `verify` before ready-for-review; `verify:full` before a release, for P2 or after risky refactors. Never run the whole suite reflexively after every edit. +- Pyramid: unit for business logic, calculations, permissions and parsers; integration only where the change touches a real system boundary (API, DB, adapter, queue); architecture checks (imports, cycles, layers) in `verify` from P1; E2E only for affected user flows (see frontend-e2e.md); manual proof with step and result in the PR where automation makes no sense. +- Test-first is mandatory for business logic, permission logic, calculations and parsers, reproducible bugfixes and stable API/service contracts; preferred for adapters, DB access, API handlers and providers; a proof of check suffices for UI text, styling, docs, infra/provider config and prototypes. +- No fake tests: a test that cannot fail proves nothing. Mocks only at external I/O boundaries (network, filesystem, time, third parties), never around the logic under test. +- Never weaken, skip or delete a test to get green; a switch that disables a production safeguard for the whole suite is an ADR (SYSTEM.md §11). A failing test follows the stuck protocol (§4), not more attempts with the same hypothesis. +- Test output is trimmed by `scripts/quiet-run.sh` (exit code unchanged, full log path printed); prefix `FLUORY_FULL_OUTPUT=1` once when the cause is unclear. diff --git a/.claude/settings.json b/.claude/settings.json new file mode 100644 index 0000000..84fba73 --- /dev/null +++ b/.claude/settings.json @@ -0,0 +1,120 @@ +{ + "permissions": { + "deny": [ + "Read(.env)", + "Read(.env.local)", + "Read(.env.development)", + "Read(.env.production)", + "Read(.env.staging)", + "Read(.env.test)", + "Read(.env.*.local)", + "Read(credentials/**)", + "Read(secrets/**)", + "Read(private-keys/**)", + "Read(production-dumps/**)", + "Read(**/*.pem)", + "Read(**/*.key)", + "Read(**/*.p12)", + "Read(**/*.pfx)", + "Read(**/id_rsa*)", + "Read(**/id_ed25519*)", + "Read(**/*kundendaten*)", + "Read(**/*customer-data*)", + "Read(**/*.dump)", + "Read(secrets.json)", + "Read(secrets.yml)", + "Read(secrets.yaml)", + "Read(credentials.json)", + "Read(credentials.yml)", + "Read(credentials.yaml)", + "Read(.netrc)", + "Read(.pgpass)", + "Read(.git-credentials)", + "Read(.htpasswd)", + "Read(**/*.tfstate)", + "Read(**/*.tfstate.backup)", + "Read(**/id_dsa*)", + "Read(**/id_ecdsa*)", + "Read(**/*.jks)", + "Read(**/*.keystore)", + "Read(**/*.ppk)", + "Read(**/*service-account*.json)" + ] + }, + "autoMemoryEnabled": false, + "env": { + "CLAUDE_CODE_DISABLE_AUTO_MEMORY": "1" + }, + "hooks": { + "SessionStart": [ + { + "hooks": [ + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/session-start.sh\"", + "timeout": 20 + } + ] + } + ], + "PreToolUse": [ + { + "matcher": "Bash", + "hooks": [ + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/guard-git.sh\"", + "timeout": 10 + }, + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/guard-checkpoint.sh\"", + "timeout": 10 + }, + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/guard-read.sh\"", + "timeout": 10 + }, + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/filter-test-output.sh\"", + "timeout": 10 + } + ] + }, + { + "matcher": "Read", + "hooks": [ + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/guard-read.sh\"", + "timeout": 10 + } + ] + } + ], + "Stop": [ + { + "hooks": [ + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/stop-check.sh\"", + "timeout": 15 + } + ] + } + ], + "PreCompact": [ + { + "hooks": [ + { + "type": "command", + "command": "bash \"${CLAUDE_PROJECT_DIR:-.}/.claude/hooks/pre-compact.sh\"", + "timeout": 20 + } + ] + } + ] + } +} diff --git a/.claude/skills/finish-work/SKILL.md b/.claude/skills/finish-work/SKILL.md new file mode 100644 index 0000000..59bd095 --- /dev/null +++ b/.claude/skills/finish-work/SKILL.md @@ -0,0 +1,28 @@ +--- +name: finish-work +description: Beendet eine Arbeitssession nach Systemregeln. Verwenden bei "Fertig", "Beende die Session", "Ich wechsle den Account", "Bereite den PR vor" oder vor jedem Sessionende mit offener Arbeit. Verhindert ungepushte Arbeit, zurückgelassene Worktrees, fehlende Tests und unklare Übergaben. +--- + +# /finish-work – Session sauber abschließen + +## Ablauf + +1. `git status` prüfen; uncommitted und ungepushte Änderungen **sichtbar auflisten**. +2. Kanonisches verify-Kommando (AGENTS.md „Prüfen") **und** `scripts/doku-check.sh` ausführen; Ergebnis ehrlich berichten – bei Fix-Schleifen gilt das Stuck-Protokoll (SYSTEM.md §4): zweiter Versuch nur mit neuer Hypothese, nach zwei Fehlschlägen zurücksetzen, dokumentieren, `blocked`. +3. Akzeptanzkriterien des Issues gegen den Stand prüfen (Testplan-Abgleich). +4. **Doku-Entscheidung** treffen (genau eine): keine langlebige Doku betroffen – oder Produktdoku, technische Doku, Architekturkarte, ADR, CHANGELOG im selben PR aktualisiert. +5. PR-Beschreibung aktualisieren: **Arbeitsstand** (Erledigt, Offen, Annahmen, nächster kleinster Schritt), **Nachweis** (verify-Stufen, letzter fokussierter Test, E2E falls betroffen) und **„Was ist passiert (Klartext)"** – `scripts/pr-check.sh` prüft die Struktur in der CI. +6. Bei WIP/Abgabe: **Übergabe-Kommentar** (append-only) in den PR: Erledigt / Offen / Nächster Schritt / Risiko. +7. Alles committen und pushen. +8. Status eindeutig setzen: `ready-for-review` | `WIP` | `blocked` | `decision-needed` – und im Project-Board spiegeln. +9. Cleanup **nur wenn** alles gepusht ist und ein PR existiert: Worktree entfernen. Bei offenem Draft-PR darf er bleiben. + +## Grenzen + +- **Nie** unfertige Arbeit stillschweigend als fertig markieren – rote Tests ⇒ Status bleibt `WIP`, klar benannt. +- Kein Merge (das ist die Reviewer-Rolle), kein Force-Push, kein Verwerfen von Änderungen ohne ausdrückliche Bestätigung. +- Fehlt das Issue oder der PR: erst nachziehen, dann abschließen. + +## Ergebnis + +Kurze Meldung: Status, PR-Link, Testergebnis, was offen ist – plus Übergabe-Kommentar bei WIP. Braucht der Mensch eine Entscheidung: Hintergrund, Optionen mit Folgen und Empfehlung im Chat (SYSTEM.md §6), nie nur die nackte Frage. diff --git a/.claude/skills/plan-issue/SKILL.md b/.claude/skills/plan-issue/SKILL.md new file mode 100644 index 0000000..27afed6 --- /dev/null +++ b/.claude/skills/plan-issue/SKILL.md @@ -0,0 +1,28 @@ +--- +name: plan-issue +description: Übersetzt eine Idee des Orchestrators in einen umsetzbaren Issue-Entwurf mit Scope und Testplan. Verwenden bei "Mach daraus ein Issue", "Plane dieses Feature" oder wenn eine vage Idee strukturiert werden soll. Erfindet nie Aufgaben und entscheidet nie Priorität. +--- + +# /plan-issue – Idee → Arbeitsvertrag + +## Ablauf + +1. Idee entgegennehmen (ein Satz reicht als Input). +2. Betroffene Module über die Architekturkarte identifizieren (nur relevante Zeilen lesen); ähnliche offene oder gemergte Issues suchen – Doppelarbeit ist ein Befund, kein neues Issue. +3. Issue-Entwurf nach `.github/ISSUE_TEMPLATE/feature.md` erstellen. + +## Output + +- Problem/Nutzen (1–3 Sätze) +- Akzeptanzkriterien (testbar formuliert) +- Nicht-Ziele +- Betroffene Module + Abhängigkeiten +- Testplan (Kriterium → Prüfart) +- **Offene Entscheidungen für den Orchestrator** (z. B. „Jeder Nutzer oder nur Admin?") +- Einschätzung: Security-/Profil-Auswirkung? Architekturfrage (→ /architecture-decision)? + +## Grenzen + +- Der Entwurf bleibt Entwurf: **Der Mensch bestätigt Priorität, Scope und setzt `ready`** – nie der Skill. +- Keine Implementierung, keine Branch-Erstellung (das macht /start-work nach `ready`). +- Bei unklarem Problem: Rückfragen stellen statt Annahmen erfinden. diff --git a/.claude/skills/review-pr/SKILL.md b/.claude/skills/review-pr/SKILL.md new file mode 100644 index 0000000..76617d9 --- /dev/null +++ b/.claude/skills/review-pr/SKILL.md @@ -0,0 +1,36 @@ +--- +name: review-pr +description: Reviews a pull request against issue, tests, module boundaries and docs as a fresh, independent session. Use on "Review PR #N", "Is this mergeable?", "Check against issue #N" or whenever the reviewer role is taken. Delivers at most 8 prioritised findings, never changes code. +--- + +# /review-pr – fresh review against the contract + +## Reads only + +The issue with its acceptance criteria · the PR description (Arbeitsstand, Impact Manifest, Nachweis, Doku-Entscheidung) and the diff · the affected tests · the area rule in `.claude/rules/` for the touched paths · relevant ADRs and the relevant lines of the architecture map. Not the whole repo, no exploration. + +## Checks + +- Acceptance criteria met? Non-goals violated? Hidden scope creep? +- Proof matches the pyramid: focused tests for the change, `verify` green (CI), E2E only where UI, browser or system integration is affected, manual proof with step and result where automation makes no sense; no weakened, skipped or fake tests. +- Plan gate answered honestly: a trigger (module boundary, public API, migration, auth/rights, payment/data/infra path, more than two modules, architecture variants, hard to reverse) means an Impact Manifest that matches the diff. +- Module boundaries kept, new files assigned to a module of the architecture map, no deep imports. +- New dependency justified? Security or privacy impact handled? No secrets in the diff? +- Docs decision: exactly one, and the chosen docs really updated in this PR; removed or renamed things grepped. +- Plain-language section understandable for the human. +- Full checklist: `Entwicklungsplan/templates/review-checkliste.md`. + +## Result (max. 8 findings, prioritised) + +``` +[Blocker] – location – problem – impact – recommendation +[Important] – location – problem – impact – recommendation +[Note] – location – problem – impact – recommendation +``` + +Close with a verdict: mergeable per risk matrix · changes requested · needs human approval. Post it as a PR comment and link it in the PR section "Nachweis". + +## Limits + +- No code changes, no unasked refactors, no change to the requirement – reviewer, not second implementer. +- Merge only per risk matrix (SYSTEM.md §5); hard to reverse or P2 → human approval. diff --git a/.claude/skills/start-work/SKILL.md b/.claude/skills/start-work/SKILL.md new file mode 100644 index 0000000..5d58dc5 --- /dev/null +++ b/.claude/skills/start-work/SKILL.md @@ -0,0 +1,29 @@ +--- +name: start-work +description: Startet die Arbeit an einem GitHub-Issue nach Systemregeln. Verwenden bei "Arbeite an Issue #N", "Starte das Feature", "Übernimm diesen Bug" oder jedem Beginn echter Implementierungsarbeit. Verhindert Arbeit ohne ready-Issue, Claim, Branch, Worktree und Draft-PR. +--- + +# /start-work – Arbeit regelkonform beginnen + +## Ablauf + +1. `AGENTS.md` des Projekts lesen (nur diese, nicht die ganze Doku). +2. Issue prüfen: Existiert es? Trägt es `ready` (Definition of Ready erfüllt)? Ist es `blocked` oder hat unerledigte Abhängigkeiten? → Wenn nein: **stoppen** und dem Orchestrator melden, nicht improvisieren. +3. Kollisionscheck: Gibt es bereits einen offenen Branch oder PR zu diesem Issue? Ist bereits jemand zugewiesen (Claim aktiv, < 48 h)? → Dann übernehmen statt neu starten, oder stoppen. +4. WIP-Limit prüfen: Sind schon 2 Issues `In Progress`? → Stoppen und melden. +5. Issue sichtbar claimen: Assignee setzen, Project-Status `In Progress`, Kommentar: „Claimed by @account on Branch claude/--". +6. Branch `claude/--` + eigenen Worktree erstellen. +7. Nach dem ersten Commit früh einen **Draft-PR** eröffnen (PR-Template). +8. Nur relevanten Kontext laden: Issue + betroffenes Modul + zugehörige Tests (Kontextladen Stufe 0–1). +9. Abschnitt **Arbeitsstand** im PR ausfüllen (Ziel, Nicht-Ziele, nächster kleinster Schritt) und die **Plan-Pflicht** beantworten – trifft ein Auslöser zu (SYSTEM.md §4), erst das Impact Manifest, dann Code. + +## Grenzen + +- Keine Umsetzung ohne `ready`-Issue – der Skill erfindet keine Aufgaben. +- Keine vollständige Architekturrecherche, keine langen Zusammenfassungen. +- Keine Cloud-/Infrastrukturänderungen. +- Bei sicherheits-, daten- oder architekturrelevanten Überraschungen: `decision-needed` an den Orchestrator. + +## Ergebnis + +Kurze Meldung: Issue, Branch, Worktree-Pfad, Draft-PR-Link, 5-Punkte-Plan. diff --git a/.gitattributes b/.gitattributes index 71fb4d7..eed2faf 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,6 +1,9 @@ -# Normalise line endings: shell hooks and CI scripts break on CRLF (Windows checkouts with core.autocrlf=true) +# Normalise line endings to LF. Git Bash on Windows tolerates CRLF, Linux shells (WSL2, containers +# mounting a Windows checkout with core.autocrlf=true) are not guaranteed to – keep scripts portable. * text=auto eol=lf *.sh text eol=lf +# Raw e-mails keep their CRLF bytes (RFC 5322) – duplicate fingerprints hash the raw file +*.eml -text *.png binary *.jpg binary *.pdf binary diff --git a/.github/ISSUE_TEMPLATE/bug.md b/.github/ISSUE_TEMPLATE/bug.md new file mode 100644 index 0000000..29d50b3 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug.md @@ -0,0 +1,22 @@ +--- +name: Bug +about: Fehlerreport mit Reproduktion +title: "" +labels: bug +--- + +## Beobachtet + + + +## Erwartet + + + +## Reproduktion + +1. + +## Akzeptanzkriterien + +- [ ] Fehler behoben, Regressionstest existiert diff --git a/.github/ISSUE_TEMPLATE/feature.md b/.github/ISSUE_TEMPLATE/feature.md new file mode 100644 index 0000000..c2c57f2 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature.md @@ -0,0 +1,41 @@ +--- +name: Feature / Aufgabe +about: Der Arbeitsvertrag zwischen Mensch und KI – ein Issue = ein Branch = ein PR +title: "" +labels: feature +--- + +## Ziel + +<1–3 Sätze: Was soll danach möglich sein, für wen?> + +## Akzeptanzkriterien + + +- [ ] + +## Nicht Teil dieser Aufgabe + +- + +## Betroffene Bereiche + +- `src/features/…` + +## Testplan + + +| Kriterium | Prüfart | +|---|---| +| <…> | Unit-Test / Integrationstest / manueller Smoke-Test | + +## Security/Privacy betroffen? + + + + diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md new file mode 100644 index 0000000..edcb784 --- /dev/null +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -0,0 +1,99 @@ + + +## Warum + +Fixes # + +## Arbeitsstand + + +- **Ziel:** +- **Nicht-Ziele:** <…> +- **Erledigt:** <…> +- **Offen:** <…> +- **Annahmen:** <…> +- **Nächster kleinster Schritt:** <…> + +## Was ist passiert (Klartext) + + + +## Plan-Pflicht (SYSTEM.md §4) + +- [ ] Kein Auslöser – keine Modulgrenze, öffentliche API, Migration, Auth/Rechte, kein Zahlungs-/Daten-/Infrapfad, höchstens zwei Module, keine Architekturvarianten, umkehrbar +- [ ] Auslöser zutreffend – Impact Manifest ausgefüllt (Plan vor Code) + +### Impact Manifest + + +- **Betroffene Module:** +- **Schnittstellen / Datenänderungen:** +- **Akzeptanzkriterien:** +- **Testplan:** +- **Verifizierte Fakten:** +- **Offene Annahmen:** +- **Nicht-Ziele:** +- **Risiken und Rollback:** + +## Geändert + +- : + +## Nachweis (SYSTEM.md §11) + +- `verify:changed`: +- `verify`: +- `verify:full` / E2E-Spec: +- Manueller Prüfnachweis: +- Frischer Review (P1 vor Ready-for-review; Architektur/API/DB immer): + +## Doku-Entscheidung (genau eine) + +- [ ] Keine langlebige Doku betroffen – Begründung: <…> +- [ ] Doku betroffen und im selben PR aktualisiert: + - [ ] Produktdoku (P0: README): + - [ ] Technische Doku: + - [ ] Architekturkarte: + - [ ] ADR: + - [ ] CHANGELOG `[Unreleased]` (sichtbares Feature oder Verhalten – im selben PR, nie „später") + +Entferntes oder Umbenanntes: `docs/` + README gegrept, Treffer bereinigt: + +## Dateigrößen und neue Bausteine (SYSTEM.md §7) + +Dateien über 500 Zeilen im Diff (Ausnahmen: generierter Code, Lockfiles, Fixtures, Migrationen, Schemas, Ressourcen, Doku, Konfiguration): +- [ ] keine +- [ ] bewusst belassen – Begründung: <…> +- [ ] im selben PR nach fachlicher Verantwortung geteilt +- [ ] Folge-Issue # + +Über 800 Zeilen mit neuer Fachlogik oder über 1000 Zeilen (P1/P2): + +Neue Shared-Komponente, Utility-Datei, Adapter oder fachlicher Service: +- [ ] nein +- [ ] ja – gesucht nach: ; gefunden: + +## Subagent-Einsätze + + + +## Risiken / offene Punkte + +- Keine + + diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..d05a958 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,110 @@ +# Base CI – minute-saving without breaking required gates (SYSTEM.md §11) +# Adapted from Entwicklungsplan/templates/base/ci.yml: pnpm + Node 24 (TS app), uv (Python AI service), +# action majors as of 2026-09-22 (template still pins v4). +# +# No workflow-wide paths-ignore: a skipped required workflow stays "Pending" and blocks doc PRs. +# The required job `check` always runs, detects doc-only changes internally and is green at once – +# expensive steps run only for code changes with a manifest. + +name: CI + +# No branches filter: stacked PRs on slice branches must be checked too. +# Explicit types: the defaults (opened, synchronize, reopened) never re-run the PR guard when a draft is +# marked ready (verify-green rule) or when the description is edited. +on: + pull_request: + types: [opened, synchronize, reopened, ready_for_review, edited] + +concurrency: + group: ci-${{ github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + check: + runs-on: ubuntu-latest # Linux only: Windows counts 2x, macOS 10x minutes. + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + + - name: Detect doc-only changes + id: diff + run: | + if git diff --name-only "origin/${{ github.base_ref }}...HEAD" | grep -qvE '^docs/|\.md$'; then + echo "code=true" >> "$GITHUB_OUTPUT" + else + echo "code=false" >> "$GITHUB_OUTPUT" + fi + if git diff --name-only "origin/${{ github.base_ref }}...HEAD" | grep -qE '^(services/ai|contracts)/'; then + echo "ai=true" >> "$GITHUB_OUTPUT" + else + echo "ai=false" >> "$GITHUB_OUTPUT" + fi + + # Docs guard: runs on EVERY PR. Seconds, no dependencies. (SYSTEM.md §2, §9) + - name: Docs guard + run: bash scripts/doku-check.sh + + # PR guard: plain language, plan gate, exactly one docs decision, file-size gate, + # before-creating proof, next step for drafts, green verify for ready-for-review. + - name: PR guard + env: + PR_BODY: ${{ github.event.pull_request.body }} + PR_DRAFT: ${{ github.event.pull_request.draft }} + PR_TITLE: ${{ github.event.pull_request.title }} + BASE_REF: origin/${{ github.base_ref }} + run: bash scripts/pr-check.sh + + - name: Doc-only result + if: steps.diff.outputs.code == 'false' + run: echo "Docs only – docs guard green, base gate green." + + # Foundation phase: no manifest yet. Report honestly instead of faking green tests. + - name: Foundation phase (TS) + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') == '' + run: echo "No package.json – foundation phase, TS verify not applicable yet. Disappears with the first code issue." + + # --- TS app: verify (lint, types, tests, architecture check, build, audit) --------------- + - uses: pnpm/action-setup@v6 + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') != '' + - uses: actions/setup-node@v7 + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') != '' + with: + node-version: 24 + cache: pnpm + - run: pnpm install --frozen-lockfile + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') != '' + - run: pnpm verify + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') != '' + + # verify:full (integration + E2E smoke + eval gate): P2 always, otherwise only with label verify-full. + - name: Read stage from profile + id: profile + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') != '' + run: echo "stage=$(sed -n 's/^[[:space:]]*stage:[[:space:]]*\([a-z]*\).*/\1/p' project-profile.yml | head -1)" >> "$GITHUB_OUTPUT" + - run: pnpm verify:full + if: steps.diff.outputs.code == 'true' && hashFiles('package.json') != '' && (steps.profile.outputs.stage == 'production' || contains(github.event.pull_request.labels.*.name, 'verify-full')) + + # --- Python AI service: path-targeted (services/ai, contracts) --------------------------- + # The AI eval gate (ADR-0001 D8) joins this block with Epic 2 (#17): it runs on every change + # under services/ai/ and needs Vertex credentials via Workload Identity Federation. + - name: Foundation phase (AI service) + if: steps.diff.outputs.ai == 'true' && hashFiles('services/ai/pyproject.toml') == '' + run: echo "No services/ai/pyproject.toml – AI service not created yet." + - uses: astral-sh/setup-uv@v10.2.0 # no floating v10 tag exists (verified 2026-09-22) + if: steps.diff.outputs.ai == 'true' && hashFiles('services/ai/pyproject.toml') != '' + with: + enable-cache: true + - name: AI service verify (ruff, pyright, pytest) + if: steps.diff.outputs.ai == 'true' && hashFiles('services/ai/pyproject.toml') != '' + working-directory: services/ai + run: | + uv sync --frozen + uv run ruff check . + uv run ruff format --check . + uv run pyright + uv run pytest -q diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..23d69e4 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,90 @@ +# AGENTS.md – RequestFlow + +RequestFlow turns quote requests (e-mail + PDF/Excel/Word attachments) into reviewed, source-linked +structured data and exports each approved request exactly once to an ERP. Reference project, treated +as a real customer engagement for a mid-sized machine-building company. **All data is synthetic.** + +> Project card: keep it under ~100 lines (hard limit 200 – `scripts/doku-check.sh`). Full rulebook: +> https://github.com/fluory/entwicklungsplan/blob/main/SYSTEM.md · Profile: `project-profile.yml` +> (stage P1 + add-ons) · Area rules: `.claude/rules/` (load only when matching files are read). + +## Commands & proof + +> **Foundation phase:** there is no product code yet. Until the first code issue lands, `verify` is +> `bash scripts/doku-check.sh`. The commands below are the agreed targets; the first code issue +> creates them and removes this note. + +- Setup: `pnpm install && uv sync --project services/ai` · Start: `docker compose up` +- `verify:changed` – inner loop: format, typecheck, focused tests of the touched files. `pnpm verify:changed -- ` · AI service: `uv run --project services/ai pytest ` +- `verify` – canonical PR proof: lint, types, unit tests **and integration tests against real Postgres + S3 storage** (they prove tenant isolation and exactly-once export on every PR), architecture check (dependency-cruiser), build, `pnpm audit`, plus ruff/pyright/pytest for `services/ai`. `pnpm verify` (needs `docker compose up -d postgres storage`). A PR is not `ready-for-review` while verify fails, cannot run, or the exception is not justified in the PR. +- `verify:full` – Playwright smoke flow + full AI eval run. `pnpm verify:full` – before a release, after risky refactors or with PR label `verify-full`. +- **AI eval gate** (ADR-0001 D8): additionally runs path-targeted in CI on every change under `services/ai/` (prompts, parsing, extraction, model config) – a regression on a key field fails the PR. +- Test and verify output is trimmed automatically (`scripts/quiet-run.sh` via the hook `filter-test-output.sh`): exit code unchanged, full log path printed; prefix `FLUORY_FULL_OUTPUT=1` once when the cause is unclear. +- **Docs guard:** `scripts/doku-check.sh` – runs in CI and in `/finish-work`. + +## Technical defaults + +System-wide decisions: `Entwicklungsplan/STACK-DEFAULTS.md`. Project choice (ADR-0001, accepted +2026-09-22): TypeScript modular monolith (Next.js App Router, Node 24, Drizzle, PostgreSQL 17, +pg-boss, Better Auth, S3 API) + stateless Python 3.13 AI service (FastAPI, docling, Vertex AI `eu`). +Deviations from the orchestrator's personal defaults (Supabase, Vercel as primary runtime): see +`docs/decisions/ADR-0001-pilot-architecture.md` D3, D6, D11. + +## Rules (short form) + + + +1. `main` only via PR. Work only with an issue + branch `claude/--` + draft PR. +2. Claim the issue: assign + comment "Claimed by @account on branch …" + draft PR within ~1 h. Stale claims (48 h without push) go back to Ready. Max. 2 issues `In Progress` per project. +3. Acceptance criteria or spec first, then test, then code. Verify green before ready-for-review. + **Stuck protocol:** hypothesis → focused check; failed → document result and cause; a second attempt only with a changed hypothesis; after two failures reset (`/rewind` and/or git), update the draft PR, set `blocked`, ask a precise question. Never weaken, delete or bypass tests to get green. +4. Docs only for long-lived knowledge – but in the same PR. Every PR has the plain-language section „Was ist passiert (Klartext)". +5. Never merge your own PR – reviewer role, preferably the other account (checklist: Entwicklungsplan/templates/review-checkliste.md). +6. New files belong to a module of the architecture map – otherwise update the map first. +7. No secrets in code, logs or repo. Remove the worktree once everything is pushed and a PR exists (or the work was consciously discarded); with an open draft PR it may stay – it is always replaceable. +8. Guards (`.claude/settings.json`): push to `main`/`master`, force-push and `--no-verify` are blocked; the stop check demands a safe state before the end. False alarm → issue in the control center, never disable or bypass a hook. +9. Load context in stages and navigate from the precise signal to the broad one (see below); never read whole folders, logs or history without a concrete reason. +10. Git checkpoint before risky operations (migration, generator, dependency upgrade, file moves, mass edits, discarding changes): commit or stash first – `/rewind` is no substitute for git. Before a new shared utility, adapter, validator or service: search for existing functionality and state it in the PR. +11. Language: technical artefacts are English by default (code, identifiers, this card, rules, ADRs, technical docs, commit messages, PR titles; issue titles and acceptance criteria in public projects). User-facing, customer and legal texts follow their audience. Never maintain a technical rule in two languages; existing German text is translated at its next substantive edit, never in a bulk refactor. + +## Context routing (read first, not in advance) + +| Topic | Read first | +|---|---| +| Architecture / modules / exceptions register | `docs/technical/architecture.md` | +| Decisions | `docs/decisions/INDEX.md` → the one ADR you need | +| Product scope, milestones, customer proposal | `docs/product/project-brief.md`, `docs/product/roadmap.md` | +| Discovery (why this project exists, approvals) | `PROJECT-START.md` – only for scope questions | +| Current work | open draft PRs + issues | +| Files provided by the human (customer request, samples) | `docs/input/` – only those linked from the issue | +| Rules for one area (API, migrations, infra, security, AI/RAG, tests, E2E) | `.claude/rules/.md` – loaded automatically when you read matching files; do not copy them here | + +## Navigation ladder + +1. Issue and current PR diff → 2. affected test file → 3. directly imported implementation → 4. LSP: definition, references, type, call hierarchy → 5. targeted text search → 6. public interface of the neighbouring module → 7. ADR, architecture map, old PRs or logs → 8. broad repository exploration, only last. +Where a symbol is defined or used: LSP before text search. What a feature changed: `git diff` or the PR diff. Which boundary applies: architecture map, area rule, ADR. Unknown failure: a read-only scout with a narrow brief. + +## Reading rule + +Stages: this card + issue → PR diff + module + tests → neighbouring interface → history only on concrete occasion. Delegate mass reading to a read-only scout: state what to find, max. 12 files, per finding path + line range + role in one sentence, no whole files, no summary of the whole project. Subagents use the smallest sufficient model for mechanical work (SYSTEM.md §8); the top model only for architecture, review and hard debugging. + +## Skill register (load on demand – not everything up front) + +Installed in `.claude/skills/` (core, always): **start-work · finish-work · review-pr · plan-issue**. +Others exist as templates in `Entwicklungsplan/templates/skills/` and are installed once their trigger occurs: +`project-start`/`project-plan`/`setup-project` (founding: discovery → plan → setup after approval) · `architecture-decision` · `legacy-audit` · `refactor-module` · `security-review` (from P1) · +`ai-eval` (AI/RAG) · `database-migration` (DB) · `infra-change` (infra) · `observability`/`incident` (operations) · +`cost-review` · `performance-check` · `release`/`release-notes` (releases) · `ui-feature`/`accessibility-review` (UI) · +`api-contract` (API) · `reuse-scan`/`package-extract`/`make-universal` (reuse: find → decide → build, rule of three). Catalogue: `Entwicklungsplan/templates/skills/README.md`. +Expected early triggers here: `security-review`, `database-migration`, `ai-eval`, `api-contract`. + +## Project specifics + +- **Synthetic data only.** Public repo: never commit real mails, attachments, names or company data – not in fixtures, evals, issues or screenshots. +- **Tenant context is mandatory.** Every data access runs through a module repository inside `withTenant(companyId, …)`; `companyId` comes from the session, never from client input. No raw DB client outside `src/db` and `src/features/tenancy` (ADR-0001 D7). +- **The AI service is stateless.** It never gets DB or storage credentials or tenant logic; the TS worker sends bytes and persists results (D8). pg-boss is the only queue (D4). +- **"Found" needs proof.** A field is `found` only if the grounding verifier confirmed its quote in the cited segment; never relax this to make evals pass (D8). +- **Gemini free tier:** local development with synthetic data only – never in the showcase or with customer data (D8). +- **Exactly-once export** relies on the idempotency key + unique export row + row lock – keep all three (D9). +- Line endings: `.gitattributes` forces LF. Git Bash on Windows tolerates CRLF (tested 2026-09-22); LF keeps scripts portable to Linux shells (CI, WSL2, containers). diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..99ecee2 --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,6 @@ +# Changelog – RequestFlow + +Format: [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). Newest entry on top. +This file records what changes **in the product** – process and session state live in the PR plain-language section. + +## [Unreleased] diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..037d67a --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,17 @@ +# CLAUDE.md – RequestFlow + +@AGENTS.md + +The imported project card (AGENTS.md) is binding – for every agent, not only Claude. +The `@AGENTS.md` import is mandatory: Claude Code does not read AGENTS.md on its own. + +## Claude Code + +- Load context in stages; never read whole directories or history; delegate mass reading to read-only scouts. +- Open a draft PR early and push. Finish a session with `/finish-work`: verify + `scripts/doku-check.sh`, plain-language section in the PR, handover comment when handing off, remove the worktree. +- The guards in `.claude/settings.json` (session-start card, push guard on `main`, stop check) are part of the system: on a false alarm open an issue in the control center, never disable or bypass a hook. +- Area rules live in `.claude/rules/` and load only when you read matching files; do not copy them here. + +# Compact instructions + +When compacting, preserve: the goal and non-goals of the current issue; the affected files and interfaces; the current diff state (committed vs. uncommitted); the last test and verify command with its result; open risks, assumptions, the open hypothesis and the next smallest step. Drop exploration output, tool logs and superseded attempts. The rules come back through AGENTS.md and the session card. diff --git a/PROJECT-START.md b/PROJECT-START.md new file mode 100644 index 0000000..b23749d --- /dev/null +++ b/PROJECT-START.md @@ -0,0 +1,317 @@ +# Project Start + +> **Zweck:** Diese Datei führt Orchestrator und KI-Session durch die Gründung eines neuen Projekts. +> Sie wird zu Beginn gemeinsam ausgefüllt, im Initialisierungs-PR committed und danach nur bei grundlegenden Richtungsänderungen aktualisiert. +> +> **Regel:** Die KI stellt offene Fragen, macht Vorschläge und dokumentiert Entscheidungen. Der Orchestrator entscheidet Ziel, Scope, Risiko, Budget und Priorität. + +## 0. Arbeitsmodus + +**Session-Auftrag:** +1. Lies diese Datei vollständig. +2. Stelle nur die noch offenen Fragen aus Abschnitt 1–6. +3. Mache keine Implementierung, keine Cloud-Anlage und keinen Projekt-Setup-Commit, bevor Abschnitt 7 vom Orchestrator freigegeben wurde. +4. Halte Antworten kurz, strukturiert und entscheidungsorientiert. +5. Wenn eine Entscheidung sicherheits-, kosten-, daten- oder architekturrelevant ist: Optionen mit Folgen nennen und `decision-needed` markieren. + +**Aktueller Status:** `setup-in-progress` (Freigabe 2026-09-22) +Mögliche Werte: `discovery` | `approved-for-setup` | `setup-in-progress` | `foundation-ready` | `paused` + +**Orchestrator:** Fluory +**Projektverantwortung:** Fluory +**Datum gestartet:** 2026-09-22 + +--- + +## 1. Problem und Ziel + +> Quelle: `docs/input/2026-09-22-kundenanfrage.md` (Kundenauftrag, Eingang 2026-09-22). Was dort nicht steht, ist hier +> als **Vorschlag** oder **offen** markiert – nichts davon ist mit dem Kunden abgestimmt. +> +> **Projektcharakter (entschieden 2026-09-22, Orchestrator):** Referenz-/Testprojekt – der Kunde +> ist ein Übungsfall, wird aber **wie ein echter Kundenauftrag** behandelt: Vorschlag, Discovery, +> Pilot und Betrieb in Kundenqualität. Es gibt **keine echten Kunden- oder Personendaten** – +> alle Anfragen, Anhänge und Eval-Fälle sind synthetisch. Ergebnis: Kunden-Vorschlag + Pilot. + +### Problem +Der Vertrieb eines mittelständischen Anlagen- und Maschinenbauers erhält täglich **ca. 20–50 +Angebotsanfragen per E-Mail**. Die relevanten Angaben stehen im Mailtext oder in Anhängen +(PDF, Excel, Word). Mitarbeiter öffnen jede Anfrage, suchen die Angaben manuell heraus und +übertragen sie in die interne Auftragsübersicht. + +### Zielgruppe +Vertriebsmitarbeiter des Kunden (mehrere Nutzer); perspektivisch mehrere Gesellschaften der +Unternehmensgruppe, die strikt nur ihre eigenen Daten sehen dürfen. + +### Nutzenversprechen +Anfrage rein → strukturierte, vom Menschen geprüfte Daten mit **Quellenbeleg** raus – ohne +Abtippen. Die KI erfindet nichts; Unsicheres wird markiert. Freigegebene Anfragen gehen genau +einmal ans ERP; keine Anfrage geht bei Ausfällen verloren. + +### Erster Meilenstein +**Pilot:** vollständig nutzbare End-to-End-Version, mit einigen Mitarbeitern testbar: +Anfrage erhalten/hochladen → Dokumente verarbeiten → Informationen extrahieren → prüfen und +korrigieren → freigeben → Export über (simulierte) REST-Schnittstelle. + +### Erfolgskriterien +> **Vorschlag – vom Kunden nicht beziffert, Zielwerte mit ihm festlegen.** +- [ ] Pilotnutzer bearbeiten reale Anfragen vollständig in der App – vom Eingang bis zum Export (kein Medienbruch) +- [ ] Extraktionsqualität je Kernfeld ist auf einem Eval-Set gemessen und reproduzierbar; Zielwert je Feld mit Kunde vereinbart +- [ ] Kein doppelter ERP-Export und keine verlorene Anfrage bei Ausfall externer Dienste – durch automatisierte Tests belegt +- [ ] Bearbeitungszeit je Anfrage messbar kürzer als heute (Baseline vom Kunden nötig) + +### Nicht-Ziele +- Echte ERP-Anbindung (Pilot: simulierte REST-Schnittstelle) – *laut Anfrage* +- Anbindung des echten Produktiv-Postfachs und produktiver Rollout – *laut Anfrage erst nach dem Pilot* +- „Perfekte Enterprise-Lösung" – *laut Anfrage* +- *Vorschlag:* SSO/Active-Directory-Login, Mehrgesellschafts-Betrieb aktiv (nur im Datenmodell vorbereitet), Angebotskalkulation/Preise, mobile Nutzung + +### Offene Produktfragen +> An den Kunden – Grundlage für den Vorschlag. Vollständige Liste: `docs/product/pilot-vorschlag.md` §9. +- [ ] Welches ERP, welche Zielfelder – gibt es eine Feldliste/ein Schema der Auftragsübersicht? +- [ ] Wie sehen reale Anfragen aus (Beispiele, anonymisiert)? Anteil gescannter PDFs/Bilder, Sprachen? +- [ ] Welche KI-Anbieter und Regionen sind zulässig (AVV, EU-Datenresidenz, kein Training)? +- [ ] Wer entscheidet fachlich, wer nimmt den Pilot ab – gegen welche Kriterien? + +--- + +## 2. Scope und Nutzerablauf + +### Nutzerrollen +| Rolle | Darf / braucht | +|---|---| +| Sachbearbeiter (Vertrieb) | Anfragen hochladen, Extraktion prüfen, Werte korrigieren, freigeben oder ablehnen, Fehler sehen und Neuverarbeitung anstoßen – nur Daten der eigenen Gesellschaft | +| Admin (je Gesellschaft) | zusätzlich Benutzer anlegen/sperren, Rollen vergeben – *genaues Rechtemodell offen* | +| System (Postfach-Import, Worker) | Mails abholen, Dokumente verarbeiten, extrahieren, exportieren – jede Aktion im Audit-Log | + +**Anfrage-Status (laut Kunde):** Neu → Verarbeitung → Prüfung → Freigegeben → Exportiert; plus Fehler +(und *Vorschlag:* Abgelehnt als Endstatus, da „ablehnen" gefordert, aber kein Status dafür genannt). + +### Erster Vertical Slice +```text +Sachbearbeiter lädt eine Anfrage hoch (.eml/.msg oder PDF/Excel/Word) → Status Neu + ↓ Duplikat-Prüfung (Inhalts-Hash); Treffer wird angezeigt, nicht still verworfen +System extrahiert Text je Dokument, KI liefert Felder mit Quellenbeleg → Verarbeitung + ↓ (Dokument + Stelle); Unsicheres/Fehlendes wird markiert, nie geraten +Sachbearbeiter prüft Feld für Feld neben der Quelle, korrigiert → Prüfung + ↓ jede Änderung mit Wer/Wann/Alt/Neu in der Historie +Sachbearbeiter gibt frei (oder lehnt ab) → Freigegeben + ↓ +System exportiert genau einmal an die ERP-Simulation (Idempotenz-Schlüssel) → Exportiert + ↓ +Fehlerfall: KI/ERP nicht erreichbar → Job wird wiederholt, Status Fehler mit + sichtbarer Ursache, manuelle Neuverarbeitung – Anfrage geht nie verloren +``` + +### Grobe Epics +- [ ] Eingang: Upload + Postfach-Import (Pilot: Test-Postfach – *offen*) +- [ ] Dokumentverarbeitung: Mail, PDF, Excel, Word → Text mit Positionsangaben (OCR *offen*) +- [ ] KI-Extraktion mit Quellenbelegen und Unsicherheitsmarkierung +- [ ] Prüf-Oberfläche: Feld ↔ Quelle, Korrektur, Freigabe/Ablehnung +- [ ] Export: ERP-Simulation (REST), Idempotenz, Retry +- [ ] Identität: Login, Rollen, Mandantentrennung je Gesellschaft +- [ ] Nachvollziehbarkeit: Audit-Historie, Statusmaschine, Duplikaterkennung +- [ ] Zuverlässigkeit: Job-Queue, Retry, sichtbare Fehler +- [ ] KI-Qualität: Eval-Set der Kernfelder, Regressionsvergleich bei Prompt-/Modellwechsel +- [ ] Betrieb: Deployment, Logs, Backups – *Hosting offen* + +--- + +## 3. Reifegrad und Risiko + +### Projektstufe +- [ ] P0 – Experiment / lokaler Proof of Concept +- [x] P1 – internes Tool / Beta / Unternehmensdaten +- [ ] P2 – produktives SaaS / Kunden / kritischer Prozess + +> **Entschieden 2026-09-22 (Orchestrator):** P1 für Discovery bis Pilot (M0–M2) → +> **Reifegradwechsel auf P2 als eigener PR vor dem produktiven Rollout** (M4). +> Add-on SaaS/Auftrag ab Start, weil externer Auftraggeber (SYSTEM.md §14). + +### Repository +- Sichtbarkeit: **`public`** (entschieden 2026-09-22, Orchestrator) +- Begründung: Referenzprojekt als öffentlicher Nachweis (Portfolio). Kein realer Kunde, keine + Unternehmensinterna; Code, Issues und Testdaten sind durchgehend synthetisch. Bedingung: Wird + daraus je ein echter Auftrag, laufen Kundendaten nie durch dieses Repo (eigenes privates Repo/Fork). +- Sprache: englisch für Code, technische Doku, ADRs, Commits, PR-Titel (bei öffentlichen Projekten auch Issue-Titel und Akzeptanzkriterien); nutzernahe, Kunden- und Rechtstexte nach Zielgruppe (SYSTEM.md „Sprache und Portfolio") + +### Risikoprofil +| Frage | Ja/Nein | Konsequenz / offene Frage | +|---|---|---| +| Öffentliche Nutzer? | Nein | Interne Anwendung; Login-Seite ggf. aus dem Internet erreichbar (Hosting offen) | +| Login / Rollen? | **Ja** | Login + einfache Benutzer-/Rechteverwaltung laut Anfrage; serverseitige Autorisierung | +| Personenbezogene Daten? | **Ja** | Ansprechpartner + Kontaktdaten der Anfragenden; Mitarbeiterkonten im Audit-Log | +| Vertrauliche Unternehmensdaten? | **Ja** | Anfragen, Spezifikationen, Zeichnungen der Endkunden des Kunden | +| Persistente Datenbank? | **Ja** | Anfragen, Extraktionen, Historie; Backup/Restore nötig | +| Datei-Upload oder Download? | **Ja** | PDF/Excel/Word/Mail – Größen-/Typlimits, Malware-Risiko, Parser-Härtung | +| Externe APIs? | **Ja** | KI-Anbieter, Postfach (IMAP/Graph), ERP (Pilot: Simulation) – Timeouts, Retry | +| LLM, RAG oder Agentenfunktion? | **Ja** | Extraktion (kein RAG); „nichts erfinden" + Eval-Set laut Anfrage; Prompt-Injection über Anhänge | +| Zahlungen / Rechnungen? | Nein | | +| Öffentliche API? | Nein | Export ist ausgehend; ERP-Schnittstelle als interner Vertrag | +| Öffentliche Website (Impressum/Datenschutz nötig)? | Nein | Internes Tool – Datenschutzinformation für Nutzer trotzdem nötig | +| Externer Auftraggeber? | **Ja** | Scope, Abnahme, AVV (du als Auftragsverarbeiter, falls du hostest), fachlicher Entscheider | + +### Aktivierte Add-ons +> Aus dem Risikoprofil abgeleitet – **bestätigt 2026-09-22 (Orchestrator)**. +> Infrastruktur wird mit dem Showcase-Deploy aktiviert, Releases mit der ersten Auslieferung. +- [x] Security-Basis – Login, Mandanten, Datei-Upload +- [x] Datenschutz – Personendaten, AVV-Kette inkl. LLM-Anbieter +- [x] Datenbank – persistente Daten, Migrationen, Backups +- [ ] Infrastruktur – erst mit Hosting-Entscheidung +- [x] Betrieb / Monitoring – Laut Anfrage „nachvollziehbarer Betrieb", sichtbare Fehler +- [x] API-Contracts – ERP-Schnittstelle als stabiler Vertrag (Simulation → echt) +- [x] KI / RAG Governance – Prompts versioniert, Eval-Set, fail-closed, Injection-Tests +- [ ] Releases – erst ab Auslieferung an den Kunden +- [ ] Recht – keine öffentliche Website; AVV/Vertrag laufen über SaaS/Auftrag + Datenschutz +- [x] SaaS / Auftrag – externer Auftraggeber (SYSTEM.md §14) + +--- + +## 4. Technikentscheidungen + +> Die KI schlägt Optionen vor. Entscheidungen mit langfristiger Wirkung werden hier festgehalten und bei Annahme später als ADR ausgearbeitet. + +### Stack-Defaults + +Referenz: `Entwicklungsplan/STACK-DEFAULTS.md`. Dort steht, was systemweit festgelegt ist +(GitHub, Actions, Feature-Module) und was **bewusst frei** bleibt (u. a. Programmiersprache – +pro Projekt hier entscheiden und dokumentieren). Abweichungen von Festlegungen brauchen +Begründung, Auswirkungsanalyse und bei langfristiger Wirkung einen ADR. + +### Plattform und Tech Stack +> Die Session schlägt hier **mehrere gleichwertige Optionen mit Folgen** vor, hergeleitet aus +> den Anforderungen dieses Projekts – keine Hausnorm, keine Lieblingsstacks (STACK-DEFAULTS.md). + +| Bereich | Entscheidung | Warum | Status | +|---|---|---|---| +| Frontend/UI-Rahmen | Next.js (App Router) + Tailwind + shadcn/ui | Modularer Monolith, ein Codebase für UI + Server (ADR-0001 D1) | entschieden 2026-09-22 | +| Backend/Sprache | TypeScript (Node 24) für App/Worker + zustandsloser Python-3.13-AI-Service (FastAPI) | Dokumentenqualität ab Tag 1 via docling; Security/Tenancy bleiben in TS (D1, D2, D8) | entschieden 2026-09-22 | +| Datenbank | PostgreSQL 17 + Drizzle; Queue pg-boss 12 im selben Postgres | Transaktionaler Job-Eintrag = keine verlorene/doppelte Anfrage (D3, D4) | entschieden 2026-09-22 | +| Auth | Better Auth ≥1.7.5, nur Einladung, organization + admin; RLS + Repository-Scoping | EU-resident, offline lauffähig, Weg zu Entra-SSO (D6, D7) | entschieden 2026-09-22 | +| Hosting | Lokal Docker Compose; Showcase nach Abnahme auf Vercel + Neon (FRA) + R2 EU; AI-Service-Host per Spike | Reproduzierbar, gleicher Image-Pfad in Produktion (D5, D11) | entschieden 2026-09-22 | +| CI | GitHub Actions (systemweit festgelegt) | SYSTEM.md §11 | gesetzt | +| Observability | JSON-Logs ohne PII, Audit in derselben Transaktion, Health-Endpoint | „Nachvollziehbarer Betrieb" laut Anfrage (D10) | entschieden 2026-09-22 | +| KI/RAG | Vertex AI `eu`, gemini-3.5-flash; Grounding-Check im Code; Eval-Set mit CI-Gate; Gemini Free Tier nur lokal + synthetisch | Kundenanforderungen „nichts erfinden" + „Qualität nachvollziehbar" (D8) | entschieden 2026-09-22 | + +> Vollständige Begründung, Alternativen, Trade-offs und Neubewertungs-Trigger: `docs/decisions/ADR-0001-pilot-architecture.md` (Accepted 2026-09-22). + +### Nicht verhandelbare technische Regeln +- Fachlogik wird feature-orientiert unter `src/features/` organisiert. +- Jede echte Arbeit nutzt Issue → Branch/Worktree → Draft-PR → Review → Merge. +- `main` ist der geprüfte Produktstand. +- Secrets gehören nie in Git, Chat, Logs oder Prompts. +- Neue Dependencies brauchen eine PR-Begründung. + +### Architektur-Skizze +Siehe Abschnitt „Overview" in `docs/decisions/ADR-0001-pilot-architecture.md` (Browser → web/worker (TS) → PostgreSQL + S3; worker → stateless AI-Service (Python) → Vertex AI `eu`; Export → ERP-Mock). + +### Entscheidungsbedarf +Keine offene Architekturentscheidung – ADR-0001 ist angenommen. Offene Prüfpunkte der Umsetzung und Kundenfragen: ADR-0001 „Open points". + +--- + +## 5. Sicherheit, Daten und Betrieb + +### Datenklassifikation +| Datenart | Klasse | Speicherort | Zugriff | Aufbewahrung | +|---|---|---|---|---| +| Kontaktdaten der Anfragenden (Name, E-Mail, Telefon, Firma) | personenbezogen | DB (Hosting offen) | Sachbearbeiter der eigenen Gesellschaft | offen – mit Kunde festlegen | +| Original-Mails und Anhänge | vertraulich (+ personenbezogen) | Objektspeicher (offen) | wie oben | offen | +| Extrahierte Felder + Quellenbelege | vertraulich | DB | wie oben | offen | +| Audit-Historie (wer/wann/was) | personenbezogen (Mitarbeiter) | DB, nur anhängend | Admin der Gesellschaft | offen – Nachweispflichten vs. Löschung | +| Texte in KI-Prompts | vertraulich (+ personenbezogen) | beim KI-Anbieter (transient) | – | nur mit AVV, EU-Region, Trainingsausschluss | +| Eval-Set | synthetisch oder vom Kunden freigegeben | Repo nur wenn synthetisch/anonymisiert | Entwickler | Projektdauer | + +### Minimale Sicherheitsmaßnahmen +- [ ] MFA für GitHub, Cloud, Domain und E-Mail aktiviert +- [ ] `.env` und lokale Credentials sind ignoriert +- [ ] Secret Scan und Dependency Scan vorgesehen +- [ ] TLS/HTTPS erforderlich, falls öffentlich +- [ ] Authentifizierung und serverseitige Autorisierung geklärt +- [ ] Rate Limits / Timeouts / Größenlimits bei öffentlichen oder teuren Endpunkten geplant +- [ ] Backup und Wiederherstellung bei persistenten Daten geplant +- [ ] Logging ohne Secrets oder unnötige personenbezogene Daten +- [ ] `.claude/settings.json` aus `templates/base/`: Read-Sperren für Secrets, Artefakt-Wächter, Auto Memory aus (SYSTEM.md §10) + +### Betrieb +| Thema | Entscheidung | +|---|---| +| Dev-Umgebung | Lokal `docker compose up`: postgres:17, SeaweedFS, web, worker, ai – nur synthetische Daten (ADR-0001 D11) | +| Staging | Keine im Pilot (P1); der Showcase nach Abnahme dient als Demo-Umgebung | +| Produktion | Nach dem Pilot, in Kundenumgebung oder EU-Cloud – offen (Kundenfrage 10) | +| Health Check | `GET /api/health`: DB, Storage, Queue-Rückstand, AI-Service erreichbar (D10) | +| Logs | JSON-Logs (pino / Python), Korrelations-IDs, keine Dokumentinhalte/PII (D10) | +| Backup / Restore | Pilot lokal: entfällt (synthetisch, reproduzierbar per Seed); Showcase: Neon-Standard; Produktion: PITR + geprobter Restore (Add-on Datenbank) | +| Rollback | Pilot: Revert-PR + Re-Deploy; Migrationen nur vorwärtskompatibel; Produktion: Rollback per Image-Tag (vor P2 ausarbeiten) | +| Kostenlimit | Lokal 0 € + KI-Nutzung (Gemini Free Tier nur lokal/synthetisch); Showcase auf Free Tiers (Vercel Hobby, Neon Free, R2 Free) + Vertex mit GCP-Budget-Alert bei **10 €/Monat** (entschieden 2026-09-22) | +| Agent-Sandbox (ab P1 risikobasiert, SYSTEM.md §10) | **`false` (entschieden 2026-09-22):** nur synthetische Daten, keine Produktions-Credentials; Windows-Host ohne WSL2 (Sandbox nur macOS/Linux/WSL2). Neu bewerten, sobald echte Credentials oder Kundendaten ins Spiel kommen | + +--- + +## 6. Planung und Arbeitsfluss + +### GitHub Project +`Inbox → Ready → In Progress → In Review → Done` + +### Labels +`idea`, `feature`, `bug`, `tech-debt`, `security`, `blocked`, `decision-needed`, `ready`, `ai`, `verify-full` + +### Merge-Regeln +Es gilt die Risikomatrix aus SYSTEM.md §5 (P0 klein: Autor nach Checkliste · P1 Feature: Zweit-Account/Review-Session · Architektur/API/DB und alles ab P2: menschliche Freigabe). + +### Erstes Epic +**#2 – Epic 1: Vertical Slice – eine Anfrage Ende-zu-Ende** (dünn durch alle Schichten, dann verbreitern) + +### Erste umsetzbare Issues +> Angelegt beim Setup (2026-09-22) mit Ziel, Akzeptanzkriterien, Nicht-Zielen, Testplan. +> #3 ergänzt gegenüber dem Discovery-Entwurf: Ohne App-Gerüst (Compose, verify-Befehle) hätte kein Slice-Issue einen Startpunkt. +- [ ] #3 `chore(app)`: TS app skeleton, docker compose and verify commands – **ready** +- [ ] #4 `feat(identity,tenancy)`: invite-only login, companies and forced RLS +- [ ] #5 `feat(intake)`: upload a request and enqueue processing atomically +- [ ] #6 `feat(ai-service)`: stateless extraction service with grounding verifier +- [ ] #7 `feat(jobs,extraction)`: process requests in the worker with retries and visible errors +- [ ] #8 `feat(review)`: review fields beside their source, correct, approve or reject +- [ ] #9 `feat(export)`: export approved requests exactly once to the ERP mock + +**Add-on-Checklisten:** #10 Security · #11 Datenschutz · #12 Datenbank · #13 Betrieb · #14 API · #15 KI/RAG · #16 SaaS/Auftrag +**Folge-Epics:** #17 Vollständige Extraktion & Qualität · #18 Robustheit & Betrieb · #19 Showcase (nach Abnahme) + +--- + +## 7. Setup-Freigabe + +> Erst nach dieser Freigabe darf eine KI-Session das Repository und die Basisstruktur erzeugen. + +- [x] Problem, Ziel und Nicht-Ziele verstanden +- [x] Erster Vertical Slice festgelegt +- [x] Projektstufe und Risikoprofil entschieden +- [x] Repository-Sichtbarkeit entschieden +- [x] Tech-Stack entschieden oder offene Entscheidung als Issue angelegt +- [x] Aktivierte Add-ons bestätigt +- [x] Budget-/Kostenrahmen bekannt +- [x] Orchestrator gibt Setup frei + +**Freigabe durch:** Fluory · **Datum:** 2026-09-22 + +--- + +## 8. Setup-Auftrag an die KI + +Nach Freigabe führt die Setup-Session ausschließlich diese Schritte aus: + +1. Repository erstellen bzw. klonen. +2. `templates/base/` passend zur Projektstufe übernehmen (P0-Minimalstruktur, ANLEITUNG.md A.2). +3. `project-profile.yml` aus diesem Dokument ableiten. +4. `AGENTS.md` als projektbezogene Agentenkarte (englisch), `CLAUDE.md` mit `@AGENTS.md`-Import und `# Compact instructions`, `.claude/` mit Wächtern, Rules und Kern-Skills erstellen. +5. Architektur-Kurzfassung erstellen (leer erlaubt – nur Entschiedenes). +6. Issue-/PR-Templates, Labels und GitHub Project anlegen. +7. Schlanke PR-CI einrichten; bei öffentlichem Repo Branch Protection. +8. Initialisierungs-PR öffnen: `chore: initialize project foundation` – mit erstellten Dateien und Warum, Test-/Lint-/Build-Befehlen, offenen Entscheidungen, nächstem `ready`-Issue. + +### Setup-Abschluss +- [ ] Repository erreichbar · Initialisierungs-PR existiert · CI läuft (oder Einschränkung dokumentiert) +- [ ] Board, Labels und Vorlagen vorhanden +- [ ] Vertical-Slice-Epic existiert, mindestens ein Issue `ready` +- [ ] Kein Secret committed + +**Status nach Setup:** `foundation-ready` diff --git a/README.md b/README.md index 96701c5..b9d4fef 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,57 @@ # RequestFlow -AI-assisted intake of quote requests: extract structured data from e-mails and attachments, -review it beside its source, and export it to an ERP exactly once. +AI-assisted intake of quote requests for industrial sales teams: extract structured data from +e-mails and PDF/Excel/Word attachments, review every value **beside its source**, and export each +approved request **exactly once** to an ERP. -> Status: project foundation in progress. +> **Status:** foundation phase – architecture decided, no product code yet. +> **Reference project:** built like a real customer engagement for a mid-sized machine-building +> company; the customer is fictional and **all data in this repository is synthetic**. + +## What it does (pilot scope) + +```text +upload (.eml/.msg/PDF/XLSX/DOCX) → parse → extract with evidence → verify grounding + → review & correct beside the source → approve / reject → idempotent export to the ERP (mock) +``` + +- **No invented data:** the model must quote its source; a deterministic verifier checks that the + quote exists in the cited segment – otherwise the value is flagged for review. +- **Nothing lost, nothing doubled:** jobs are enqueued in the same database transaction as the + status change; the export uses an idempotency key, a unique export record and a row lock. +- **Tenant isolation:** repository scoping plus forced PostgreSQL row-level security per company. +- **Measurable AI quality:** an eval set with per-field metrics gates prompt and model changes. + +## Architecture + +TypeScript modular monolith (Next.js, Node 24, Drizzle, PostgreSQL 17, pg-boss, Better Auth, +S3 API) plus a stateless Python AI service (FastAPI, docling, Gemini on Vertex AI in the EU). +Rationale, alternatives and trade-offs: [ADR-0001](docs/decisions/ADR-0001-pilot-architecture.md). + +## Documentation map + +| Topic | Where | +|---|---| +| Architecture map and exceptions register | [docs/technical/architecture.md](docs/technical/architecture.md) | +| Decisions | [docs/decisions/](docs/decisions/INDEX.md) | +| Project brief, roadmap | [docs/product/](docs/product/project-brief.md) | +| Customer proposal (German) | [docs/product/pilot-vorschlag.md](docs/product/pilot-vorschlag.md) | +| Discovery and approvals (German) | [PROJECT-START.md](PROJECT-START.md) | +| Working rules for humans and agents | [AGENTS.md](AGENTS.md) | + +## Getting started + +Not runnable yet. The first code issue adds `docker compose up` (PostgreSQL, S3-compatible storage, +web, worker, AI service) and the `verify` commands listed in [AGENTS.md](AGENTS.md). + +## How this repository is run + +Every change starts as an issue, lands through a pull request with a plain-language summary and a +green CI, and is never merged by its author. The process follows the fluory-system rulebook +(guard hooks in `.claude/`, docs and PR guards in `scripts/`). + +## Kurzfassung (Deutsch) + +KI-gestützte Erfassung von Angebotsanfragen: Daten aus E-Mails und Anhängen extrahieren, jede +Angabe neben ihrer Fundstelle prüfen und freigegebene Anfragen genau einmal ans ERP übergeben. +Referenzprojekt mit fiktivem Kunden – alle Daten sind synthetisch. diff --git a/docs/decisions/ADR-0001-pilot-architecture.md b/docs/decisions/ADR-0001-pilot-architecture.md new file mode 100644 index 0000000..23b2d87 --- /dev/null +++ b/docs/decisions/ADR-0001-pilot-architecture.md @@ -0,0 +1,704 @@ +# ADR-0001 · Pilot architecture baseline + +- **Status:** Accepted – 2026-09-22 by Fluory (orchestrator); every decision D1–D11 was confirmed individually +- **Deviation from the draft:** D8 – the orchestrator chose a Python AI service from day 1 instead of + the drafted recommendation (full TypeScript); the draft recommendation is kept as alternative 1 in D8 +- **Deciders:** Fluory (orchestrator) · drafted by a Claude session +- **Inputs:** `docs/input/2026-09-22-kundenanfrage.md` (customer request), `PROJECT-START.md` (discovery) +- **Facts verified:** 2026-09-22 against official docs, registries and provider terms (sources at the end). + Statements without a source are marked *heuristic* or *to verify*. + +| Decision | Topic | Confirmed | +|---|---|---| +| D1 | Application architecture: TS modular monolith + stateless Python AI service | ✔ 2026-09-22 | +| D2 | Runtime: Node.js 24 + Python 3.13; `web`/`worker` from one codebase; `drain()` | ✔ 2026-09-22 | +| D3 | Database: PostgreSQL 17 + Drizzle; Neon Frankfurt for the showcase | ✔ 2026-09-22 | +| D4 | Queue: pg-boss 12 with transactional enqueue | ✔ 2026-09-22 (challenged) | +| D5 | Object storage: S3 API – SeaweedFS / R2 EU / customer's choice | ✔ 2026-09-22 (challenged) | +| D6 | Authentication: Better Auth, pinned, minimal plugins | ✔ 2026-09-22 (challenged) | +| D7 | Tenant isolation: repository scoping + forced RLS | ✔ 2026-09-22 (challenged) | +| D8 | AI & documents: Python AI service (FastAPI + docling + Vertex `eu`) | ✔ 2026-09-22 (challenged, deviates from draft) | +| D9 | Integrations: upload in the pilot, ERP mock with idempotency contract | ✔ 2026-09-22 | +| D10 | Observability: JSON logs without PII, transactional audit, health | ✔ 2026-09-22 | +| D11 | Deployment: Docker Compose; Vercel showcase after acceptance | ✔ 2026-09-22 | + +## Context + +A mid-sized machine-building company (reference case, treated as a real customer) receives +20–50 quote requests per day by e-mail. The relevant data sits in the mail body or in PDF, +Excel and Word attachments. Staff copy it by hand into an internal order overview. + +The pilot must cover, end to end: **receive/upload → process documents → extract → review and +correct → approve → export via a (simulated) REST interface** – with a realistic path to +production. Load is tiny (~1,000 requests/month); correctness, traceability and data protection +dominate, not throughput. Scope confirmed by the orchestrator: **~13 working days, complete core** +(see the budget check). + +Runtime targets: + +| Target | Purpose | Data | +|---|---|---| +| Local Docker Compose | Reference runtime, production-like, reproducible by any reviewer | synthetic | +| Vercel showcase (Hobby) | Public demo, after pilot acceptance | synthetic only | +| Customer environment (later) | Production: EU cloud or on-prem – unknown today | real | + +## Decision drivers (from the customer request) + +| # | Driver | +|---|---| +| DR1 | The AI must not invent data; unclear values are flagged; each value shows its source (document + location) | +| DR2 | Staff review, correct, approve or reject | +| DR3 | A request is never exported to the ERP twice | +| DR4 | No request is lost when the AI or another external service is down; retry later; error visible | +| DR5 | Users see only data of their own company (several subsidiaries later) | +| DR6 | Traceable history of who changed or approved what | +| DR7 | Duplicate requests are detected | +| DR8 | Extraction quality is measurable; regressions after prompt/model changes are caught | +| DR9 | Short pilot; clean implementation, extensibility, traceable operation | +| DR10 | GDPR: EU processing, few processors, DPA chain including the LLM provider (add-on `datenschutz`) | + +## Overview + +```text +Browser ──► web (Next.js, Node 24) ──► PostgreSQL 17 ─┬─ app schema (RLS per company) + │ upload, review, approve ├─ auth schema (Better Auth) + │ enqueue job IN SAME TRANSACTION ├─ pgboss schema (queue) + ▼ └─ audit_events (append-only) + Object storage (S3 API, private) + ▲ +worker (same TS codebase) ── pg-boss fetch ──► POST /v1/extract (bytes) ──► ai-service (Python, stateless) + │ persists result in one transaction ◄── segments + fields + evidence ──┘ docling → Gemini (Vertex "eu") → grounding check + │ + └── on approval ──► ERP adapter ──HTTP + Idempotency-Key──► ERP mock (REST) +``` + +Request states: `NEW → PROCESSING → REVIEW → APPROVED → EXPORTED`, plus `REJECTED` (end state after +review) and `ERROR` (with the failed stage `processing | export`, the cause and the next retry). + +--- + +## D1 · Application architecture + +**Problem.** One developer, a short pilot, and several features that must stay separable +(intake, parsing, extraction, review, export). A production rollout comes later. + +**Decision.** A **modular monolith in TypeScript** on Next.js (App Router), plus **one stateless +Python AI service** (D8). + +**Repository layout.** +- `src/` – the TS app. +- `services/ai/` – the Python service. +- `contracts/` – OpenAPI contracts for the AI service and the ERP. + +**How the TS app is organised.** +- Feature modules live under `src/features//`, each with one public entry file + (`index.ts`). +- dependency-cruiser checks the boundaries in `verify`. +- The domain core is pure TypeScript and built test-first: status machine, duplicate fingerprint, + field normalisation for display and export. +- UI components and route handlers call module APIs and never touch the database directly. + +| Module | Responsibility | +|---|---| +| `intake` | upload, later mailbox adapters; creates request + documents; duplicate fingerprint | +| `documents` | document records, storage references, hashes | +| `extraction` | AI-service client (generated from the OpenAPI contract); persists runs, fields and evidence | +| `requests` | request aggregate, status machine, transitions | +| `review` | review UI, corrections, approve/reject | +| `export` | ERP port, REST adapter, idempotency | +| `erp-mock` | simulated ERP REST API (dev/showcase only, behind a flag) | +| `identity` | Better Auth setup, users, companies (organisations), roles | +| `tenancy` | tenant context, `withTenant()` transaction wrapper, RLS policies | +| `audit` | append-only audit events | +| `jobs` | pg-boss setup, job definitions, `drain()`, worker entrypoint | +| `storage` | `BlobStore` port + S3 adapter | +| `observability` | logger, health check, ops data for the request list | + +The Python service has its own packages: `parsing`, `extraction`, `grounding`, `api`, plus `evals/`. +It is one deployable with one purpose, not a second domain. + +**Alternatives.** +1. *Separate API (Hono/NestJS) + SPA* – explicit API contract and an independent frontend, but two + deployables, duplicated types and a slower first slice. +2. *Microservices (intake / extraction / export)* – independent scaling is irrelevant at 50 + requests/day. Distributed transactions would work against DR3/DR4. + +**Trade-offs.** A monolith can be misused as a "big ball of mud"; the module boundaries and the +architecture check prevent that. Scaling happens per process type (`web`, `worker`, `ai`), not per +module. + +**Rationale.** It is the fastest structure that still keeps modules replaceable (SYSTEM.md §14: +modular monolith first). The only split, the AI service, follows a real difference in technology. + +**Pilot cost.** Foundation: about 1 day (repo, Docker Compose, TS CI, module skeleton). *Heuristic.* +**Revisit when** a module needs its own scaling, release cycle or security boundary, or other +clients need a public API. + +--- + +## D2 · Runtime + +**Problem.** Jobs such as parsing, LLM calls and export take seconds to minutes and must survive +failures. The app runs as long-lived containers (Docker/on-prem) *and* on serverless Vercel. + +**Decision.** **Node.js 24 LTS** for the TS app, with **one codebase and two entrypoints**: +- `web` – the Next.js server. +- `worker` – a long-running pg-boss consumer. + +**Python 3.13** runs the AI service (FastAPI/uvicorn, dependencies managed with `uv`). + +Job handlers are plain functions. A shared `drain({ maxMs })` processes available jobs until its +time budget is used up. Who calls it depends on the runtime: + +| Runtime | Who calls `drain()` | +|---|---| +| Docker / production | the `worker` process in a loop | +| Vercel showcase | `after()` right after enqueueing (bounded by the 300 s function limit on Hobby), an authenticated "retry now" action, and one daily cron sweep (Hobby cron: at most once per day) | + +**Alternatives.** +1. *Vercel Workflow* (GA since 2026-04-16, durable steps) – elegant on Vercel, but it ties the + durability layer to a platform runtime and adds a second queue model next to Postgres. +2. *Bun* – faster start-up, but a compatibility risk with pg-boss, and no need for it. + +**Trade-offs.** On the showcase, retries with backoff are only picked up when something triggers +`drain()`. Timely unattended retries exist only where the worker runs (Docker/production). This is +recorded in the exceptions register (`docs/technical/architecture.md`), not hidden. + +**Rationale.** The same job code runs in both environments, and production behaviour (a real worker) +is the default. + +**Pilot cost.** About 0.5 day for `drain()` (the Vercel triggers come with the showcase). *Heuristic.* +**Revisit when** the showcase needs unattended retries → Vercel Queues/Workflow as a push trigger, +or a small always-on worker host. + +--- + +## D3 · Database + +**Problem.** Requests, extraction results, corrections, audit, auth and jobs need consistent, +transactional storage with tenant isolation, both locally and on the showcase. + +**Decision.** **PostgreSQL 17** (pg-boss needs ≥ 13) with **Drizzle ORM** and drizzle-kit SQL +migrations that are checked in and reviewed. One database holds everything, in separate schemas: +`app`, `auth`, `pgboss` and the audit table. **Only the TS app connects to the database**; the AI +service has no database access. + +| Environment | Database | +|---|---| +| Local | `postgres:17` container | +| Showcase | **Neon Free** in `aws-eu-central-1` (Frankfurt) via the Vercel Marketplace. 0.5 GB is enough because files are not stored in the database. | +| Production | managed PostgreSQL in the EU or on-prem, with PITR and a rehearsed restore (add-on `datenbank`) | + +**Alternatives.** +1. *Supabase* (Postgres + Auth + Storage) – bundled services, but we would only use Postgres, and + its API surface adds exposure we don't need. +2. *Prisma instead of Drizzle* – mature developer experience, but Drizzle stays closer to SQL, + declares RLS policies in the schema (`pgPolicy`) and has a verified pg-boss helper + (`fromDrizzle(tx, sql)`). + +**Trade-offs.** Neon's pooler runs PgBouncer in transaction mode, which drops session `SET` and +LISTEN/NOTIFY. We therefore only use transaction-local settings (`set_config(..., true)`), and +pg-boss polls instead of using LISTEN/NOTIFY – both compatible. Neon scales to zero after 5 +minutes, so the first showcase request after idle is slower. + +**Rationale.** One transactional store is the precondition for DR3/DR4 (D4) and DR5 (D7). + +**Pilot cost.** About 0.5 day on top of the foundation (schema, migrations, database roles). *Heuristic.* +**Revisit when** reporting or read load grows (add a read replica), or the customer mandates a +specific database platform. + +--- + +## D4 · Queue / background processing — *challenged: pg-boss vs. Redis* + +**Problem.** DR4 (never lose a request, retry, visible error) and DR3 (never export twice) require +durable jobs. The critical failure is a **dual write**: the status change is committed but the +job is lost, or the reverse. + +**Decision.** **pg-boss 12** (PostgreSQL ≥ 13, Node ≥ 22.12): + +- **Transactional enqueue.** The job is inserted in the *same transaction* as the status change, + via the `db` option with `fromDrizzle(tx, sql)`. That gives a transactional outbox without a + separate outbox table. +- **Retries.** `retryLimit`, `retryBackoff`, `retryDelayMax`. Exhausted jobs go to a `deadLetter` + queue, and the request moves to `ERROR` with a visible cause and a manual "reprocess" action. +- **One job per request at a time.** `singletonKey = requestId`. +- **At-least-once delivery.** Every handler is idempotent (D9 covers the export). +- **Polling, not LISTEN/NOTIFY** (default 2 s). This works through the transaction-mode pooler. +- **Serverless mode.** On Vercel, pg-boss runs with `supervise/schedule/migrate: false`; `drain()` + runs the maintenance explicitly. *Exact maintenance API to verify during implementation.* +- **Calls to the AI service** happen inside the job with a timeout. A timeout or a 5xx is a normal + job failure and is retried with backoff. + +**Alternatives.** +1. *BullMQ + Redis* – mature, fast, good dashboards. But the enqueue cannot join the Postgres + transaction, so the dual-write risk returns and needs an outbox table plus a relay anyway. It + adds a Redis service in Docker and a hosted Redis for Vercel, and BullMQ workers are + long-running, which does not match serverless. Its throughput headroom is irrelevant at 50/day. +2. *Vercel Queues* (public beta since 2026-02-27, at-least-once) – managed and push-based, but the + same dual-write issue, beta status, and not available on-prem. + +**Trade-offs.** Polling adds up to about 2 s latency and a little database load (irrelevant here). +There is no BullBoard-style dashboard; request states plus pg-boss tables feed the status, attempts +and error shown in the request list, which DR4 requires anyway. + +**Rationale.** It is the only option where "saved" and "queued" are one atomic fact, with zero +extra infrastructure. Because pg-boss is Node-only, only the TS worker consumes jobs; the Python +service is called and never polls. + +**Pilot cost.** About 1 day including retry and dead-letter tests. *Heuristic.* +**Revisit when** load is sustained above roughly 100 jobs/s, cross-service fan-out is needed, or +the customer platform standardises on a broker. *Heuristic threshold.* + +--- + +## D5 · Object storage — *challenged: S3-compatible provider* + +**Problem.** Original mails and attachments are confidential (they contain personal data in real +use). They must be stored privately, served only to authorised users of the right company, +portable to on-prem, and runnable locally and on Vercel. + +**Decision.** **S3 API** via `@aws-sdk/client-s3` behind a `BlobStore` port: + +- The bucket is private only. +- Object keys are `{companyId}/{requestId}/{documentId}`. +- Files are served only through authenticated app routes that check tenant and role. There are + never public URLs. +- A SHA-256 hash is stored per file (duplicates, integrity). +- The AI service never receives storage credentials: the worker streams the bytes to it. + +| Environment | Provider | Why | +|---|---|---| +| Local | **SeaweedFS** (Apache-2.0, S3 gateway, very active) | MinIO is out – see below | +| Showcase | **Cloudflare R2**, bucket in the **EU jurisdiction** | 10 GB free, no egress cost, S3 core API; *EU jurisdiction on the free plan: medium confidence → verify at setup* | +| Production | customer's choice via config: Hetzner Object Storage (DE, from €6.49/month), AWS S3 `eu-central-1`, or on-prem S3 | same adapter | + +**Alternatives.** +1. *Vercel Blob, private* (GA 2026-06-30, EU region selectable, signed URLs) – zero configuration + on Vercel, but it has its own SDK and no S3 API. That is not portable to on-prem and would need a + second adapter. Kept as the fallback if R2's EU jurisdiction is unavailable on the free plan. +2. *Postgres `bytea`* – simplest and transactional, but Neon Free has 0.5 GB, backups bloat, and + there is no production path. + +*Rejected:* **MinIO**. The community edition went source-only in October 2025 and into maintenance +mode in December 2025; the repo is archived, and `minio/minio` was removed from Docker Hub +(~2026-09-12). Garage (AGPL-3.0, S3 subset) is a viable local alternative. RustFS reached 1.0 only on +2026-09-16, which is too young. + +**Trade-offs.** R2 has no versioning or object lock. The pilot does not need them; production +retention or legal hold may, in which case choose an S3 backend with versioning. + +**Rationale.** One API everywhere; the provider is configuration, not code. + +**Pilot cost.** About 0.5 day. *Heuristic.* +**Revisit when** the customer requires WORM/retention, on-prem storage, or a specific provider. + +--- + +## D6 · Authentication — *challenged: Better Auth vs. managed auth* + +**Problem.** Several staff need login and simple user management. Companies (subsidiaries) must +map to tenants. Data should stay in the EU, and the local Docker demo should run without external +accounts. The production target is most likely the customer's own identity provider – *assumption: +Microsoft 365 / Entra ID, to confirm with the customer*. + +**Decision.** **Better Auth** (MIT), pinned to **≥ 1.7.5**, with `@better-auth/drizzle-adapter`. +Sessions are stored in our database, in the `auth` schema. + +- **Pilot scope:** e-mail + password, invite-only (no public sign-up). +- **Plugins:** `organization` plugin (organisation = company = tenant) and `admin` plugin (user + management, used via an invite form and a seed script – no role-admin UI in the pilot). +- **Rate limiting:** the built-in rate limit, with database storage so it works on serverless. +- **Minimal plugin surface:** no magic link, SSO or SCIM in the pilot. These are the areas with + 2025–2026 advisories. +- **CI:** `pnpm audit` fails the build on high/critical findings. +- **Production path:** Entra ID through `@better-auth/sso` (OIDC), which brings MFA and user + lifecycle from the customer's identity provider. + +**Alternatives.** +1. *Clerk* – fastest, with organisations and SSO built in. But **no regional data residency** (US + infrastructure, EU transfers via the Data Privacy Framework). Priced per user/org/SSO + connection. It cannot run on-prem, and the local demo would depend on an external service. +2. *WorkOS AuthKit* – strong enterprise SSO, but its DPA names the USA as the storage location, and + SSO costs $125 per connection. +3. *Self-hosted identity provider (Keycloak/Zitadel)* – standards-based and EU/on-prem capable, but + one more service to run and harden. Too heavy for the pilot. + +**Trade-offs.** We own the auth configuration: password policy, session lifetime, cookies/CSRF and +rate limits. That becomes a security-review item. The library had several advisories in 2025–2026, +including in the organisation and SSO plugins, which we mitigate by pinning, running audits, using +a minimal plugin set and watching advisories. Better Auth joined Vercel on 2026-07-07 and stays +MIT; watch for platform coupling. + +**Rationale.** It is the only option that is EU-resident by construction, runs offline in Docker, +maps tenants natively (organisations) and has a documented path to Entra ID. + +**Pilot cost.** About 0.5 day: invite form, seed script, roles. *Heuristic.* +**Revisit when** real rollout happens (make Entra SSO mandatory), or if advisories keep hitting the +plugins we use. The fallback is OIDC-only against the customer's identity provider, or Keycloak. + +--- + +## D7 · Authorization and tenant isolation — *challenged: application-level vs. RLS* + +**Problem.** DR5 requires that users see only their company's data. The classic failure is one +forgotten `WHERE company_id = …` → a cross-tenant data leak. + +**Decision.** **Two layers, one mechanism.** + +1. **Application layer.** Every data access goes through module repositories that require a + `TenantContext`. The `companyId` comes from the session's active organisation, never from + client input. Role checks run through one `authorize(actor, action, resource)` function per + module. Pilot roles: `admin` (user management plus everything) and `clerk` (process requests). +2. **Database layer.** + - Rows: every company-owned table has a `company_id` column, with **RLS enabled and forced** and + the policy `company_id = current_setting('app.company_id')::uuid`. + - Roles: the app connects as `app_rw`, which is neither table owner nor able to bypass RLS + (`NOBYPASSRLS`). Migrations run as the owner role. + - `withTenant(companyId, fn)`: opens a transaction, calls + `set_config('app.company_id', $1, true)` (transaction-local, so it is safe with the pooler), + then runs `fn`. + - Jobs: the payload carries `companyId`, and the handler runs inside `withTenant` and re-checks + the request row. + - The AI service receives no tenant data beyond opaque IDs. +3. **Exception.** Better Auth tables (`auth`) and the queue (`pgboss`) are not company-owned + business data. They live in separate schemas that only server code reaches. This deviates from + the global rule "RLS on every table"; the rule's intent (database-enforced tenant isolation) is + met for all business data. → exceptions register in `docs/technical/architecture.md`. + +**Alternatives.** +1. *Application-level filtering only* – simplest and works with any pooler, but a single missing + filter leaks data, and only tests would catch it. +2. *Schema or database per tenant* – strongest isolation, but migrations run N times. That is + overkill for "possibly several subsidiaries". + +**Trade-offs.** RLS costs about 0.5–1 day (roles, policies, wrapper, tests). "Why is this list +empty?" is harder to debug. Every query must run inside `withTenant`, which the repository API +enforces (no raw database client is exported). **Proof:** integration tests with two tenants that +try cross-tenant reads and writes, both through the repositories and as raw SQL under `app_rw`. + +**Rationale.** Defence in depth for the customer's most explicit security requirement, at a cost +the pilot can carry. + +**Pilot cost.** About 1 day. *Heuristic.* +**Revisit when** there are many tenants with differing retention or residency needs → schema or +database per tenant. + +--- + +## D8 · AI and document processing — *challenged: full TypeScript vs. Python AI service* + +**Problem.** DR1 (nothing invented, source visible) and DR8 (measurable quality) are the core of +the product. Inputs are the mail body plus PDF, XLSX and DOCX attachments, `.msg` files and +possibly scans and table-heavy PDFs. + +**Decision.** A separate, **stateless Python AI service** (`services/ai/`, Python 3.13, FastAPI) +owns *parse → extract → verify* and the evals. Its pipeline has four steps. + +1. **Parse to segments with stable locators** using **docling** (MIT, LF AI & Data). + - Formats: PDF (layout, reading order, tables, OCR for scans), DOCX, XLSX, and mail formats as + listed in the docling docs. + - *To verify at implementation:* EML/MSG coverage. The fallback for EML is the Python standard + library `email` package. + - Locators come from docling's provenance data: page + bounding box, table cell, sheet cell, + paragraph, mail body line. +2. **Extract** via the Google Gen AI SDK against **Vertex AI** (renamed "Gemini Enterprise Agent + Platform" on 2026-04-22), multi-region endpoint **`eu`**, model **`gemini-3.5-flash`**. + - It is available in `eu` and `europe-west3`; the Gemini 2.5 models retire on 2026-10-20. + - Structured output (`responseSchema`) is derived from pydantic models. For each field it + returns `value | null`, `status` (`found | uncertain | missing`) and evidence + `{ segmentId, quote }`. + - *To verify at implementation:* the exact SDK configuration for the `eu` endpoint. +3. **Verify grounding (deterministic, test-first).** The quote must occur (normalised) in the cited + segment, and the value must be consistent with the quote (numbers, units and dates normalised). + Otherwise the status becomes **`unverified`**, which forces human attention. **The model never + has the final say on "found"** – DR1 is enforced by code, not by prompt wording. +4. **Return** the segments, fields and run metadata (model ID, prompt version, schema version, + tokens, latency). + +**Service contract and boundaries:** +- The contract is **OpenAPI 3.1** in `contracts/ai-service.openapi.yaml`; the TS client and types + are generated from it. +- The TS worker sends the document bytes plus opaque IDs and persists the response in *its* + transaction. +- The AI service has **no database, no storage credentials and no tenant logic** – only model + credentials. +- Internal authentication: a bearer token in the pilot; IAM/OIDC in production. +- Prompts are versioned files inside the service. + +**Model and provider:** +- **Data use:** Google Cloud terms say no training on customer data. For production, request the + zero-data-retention exemption. +- **Authentication:** locally, application default credentials; on a hosted showcase, Workload + Identity Federation (Vercel OIDC or the Cloud Run service identity, depending on where the + service runs). +- **Gemini API free tier – local development with synthetic data only.** Its terms allow human + review and product improvement of free-tier content, and state: *"You may use only Paid Services + when making API Clients available to users in the EEA, Switzerland, or the UK."* The showcase + therefore uses Vertex (paid). Cost is capped by rate limits, upload limits and a GCP budget alert. +- **Avoid:** PyMuPDF / pymupdf4llm (AGPL-3.0 – a licensing problem for a closed customer product). + +**Prompt injection.** Document text is data inside delimiters. The model has no tools, and its +output is constrained by the schema. The only high-impact action (the ERP export) requires human +approval (add-on `ki-rag`). The eval set contains injection cases. + +**Evals** (DR8, add-on `ki-rag`): +- **Cases:** `services/ai/evals/` holds synthetic cases (mail + attachments + `expected.json`), + starting with 15 cases weighted toward known weaknesses (tables, scans, missing values, + injection). +- **Runner:** a pytest-driven runner executes the **production pipeline** (parse → extract → + verify). +- **Metrics per field:** + - accuracy of found values + - precision/recall of "missing" + - grounding pass rate + - false-found rate (the hallucination indicator) +- **Gate:** a committed baseline file. The CI gate fails when a key field drops by more than *x* + points; *x* is set with the customer. It runs on changes under `services/ai/` or with the + `verify-full` label. + +**Alternatives.** +1. *Full TypeScript pipeline* (the drafted recommendation: unpdf, SheetJS from its CDN, mammoth, + postal-mime, AI SDK v7). One language, and about 2 days cheaper. **Not chosen by the + orchestrator:** document quality – tables, scans, `.msg` – is the product's core from day 1, + and a later migration to docling would re-do the parsing layer. +2. *A Python service that also consumes jobs and reads the database* – pg-boss is Node-only, so this + needs a second queue library and a second place for tenancy and security. Rejected in favour of + the stateless service. +3. *Native PDF input to Gemini, no own parsing* (supported, 15 MB inline) – the source is only a page + the model claims, not verifiable against text, so it breaks DR1 enforcement. Kept as a fallback + for pages where docling finds no text; visual evidence there is always `uncertain`. +4. *promptfoo for evals* (MIT, OpenAI-owned since 2026-03, supports Vertex) – it evaluates prompt + + model, not our parser and verifier code path. Not used in the pilot. + +**Trade-offs.** +- Two languages and toolchains: pnpm/Vitest and uv/ruff/pyright/pytest. +- A service contract that must stay in sync, mitigated by generated TS types and contract tests. +- A larger container image because of docling's models. *Size unverified.* +- CPU latency of OCR and table models. *Unverified – measure in the pilot.* +- The showcase host of the AI service is open (see D11). + +**Rationale.** The strongest document handling from day 1. The security and tenancy surface stays +in one place because the service is stateless. + +**Pilot cost.** About 5.25 days: service skeleton + CI chain + contract 1.5, docling parsing to +segments 1, extraction + verifier 1.5, eval runner + 15 cases 0.75, worker integration (client, +timeouts, error mapping) 0.5. *Heuristic.* +**Revisit when** docling's latency or footprint is unacceptable → a lighter parser per format +(e.g. pdfplumber, MIT), or when GPU inference becomes necessary. + +--- + +## D9 · External integrations + +**Problem.** Requests arrive from a mailbox or an upload. Approved data must reach the ERP exactly +once (DR3), and the pilot uses a simulated REST ERP. + +**Decision.** + +**Intake:** +- **Pilot:** web upload (`.eml`/`.msg` or loose PDF/XLSX/DOCX files) through an `IntakeSource` port. +- **After the pilot:** a mailbox adapter. For production that is most likely Microsoft Graph for an + M365 mailbox. *Depends on the customer's mail system; open question.* + +**Duplicates (DR7):** +- Exact duplicates are detected via the `Message-ID` header and the SHA-256 of the raw mail/files. +- Near-duplicate hints (same sender, similar subject) come after the pilot. +- Nothing is silently discarded; staff decide. + +**ERP:** +- An `ErpExporter` port with an **OpenAPI 3.1 contract** in `contracts/erp-export.openapi.yaml`: + `POST /v1/quote-requests` with header `Idempotency-Key: `. +- **The ERP mock** stores idempotency keys: a repeated call returns the *same* ERP reference, + never a duplicate. Failure injection (5xx, timeout) is configurable to demonstrate retry. +- **Exactly once** comes from three things together: + - at-least-once jobs (D4) + - `unique(request_id)` on the export record plus the `APPROVED → EXPORTED` transition under a row + lock + - an idempotent receiver +- **Timeouts** on every outbound call (AI service, ERP). +- **Pilot:** the mock is a module (`erp-mock`) inside the same app behind a flag, called over HTTP + via a configured base URL, so switching to the real ERP is a configuration change. + +**Alternatives.** +1. *File-based export (CSV/XML drop)* – often available for ERP systems that lack a REST API + (*heuristic*). It would be a second adapter behind the same port. +2. *Writing directly into the ERP database* – rejected, because it bypasses the ERP's business + logic. +3. *ERP mock as a separate service/container* – a more realistic boundary, but one more deployable. + Not worth it for the pilot. + +**Trade-offs.** A mock proves the contract and the idempotency, not the real ERP's semantics. The +real mapping is a post-pilot work package, together with the customer. + +**Pilot cost.** About 1.25 days: upload + exact-duplicate check 0.25 (0.75 together with D5 in the +budget table), export + idempotency + mock + contract tests 1. *Heuristic.* +**Revisit when** the ERP is known (REST / file / middleware) and the mail system is known (Graph / +IMAP / Exchange on-prem). + +--- + +## D10 · Observability + +**Problem.** "Traceable operation" (DR9) and "error visible for the employee" (DR4), without +leaking personal data into logs. + +**Decision.** +- **Logs:** structured JSON – `pino` in TS, JSON logging in the AI service – with correlation IDs + (`requestId`, `jobId`, `companyId`) propagated to the AI service. **No document content or + personal data**, only IDs. +- **Business audit:** the `audit_events` table (DR6) is written in the **same transaction** as the + change and holds who, when, what, old and new values. The app role has only `INSERT`/`SELECT` on + it, so append-only is enforced by the database. +- **Operator visibility:** + - status, attempts, last error and next retry per request, shown in the request list + - `GET /api/health` (database, storage, queue backlog, AI service reachability) +- **LLM usage per run:** model, prompt version, tokens, latency. +- **Showcase:** platform logs. +- **Production:** OpenTelemetry export to the customer's stack, or error tracking in an EU region. + +**Alternatives.** +1. *A full OpenTelemetry stack now* – too heavy for the pilot. +2. *Langfuse* (LLM tracing, self-hostable) – later, if prompt debugging needs traces. + +**Trade-offs.** There is no central error tracking and no separate ops page during the pilot; +errors surface in the request list, the health endpoint and the logs. + +**Rationale.** It covers what the customer asked for – visible errors, history and quality – with +the data we store anyway. + +**Pilot cost.** About 0.25 day (no separate ops page in the pilot). *Heuristic.* +**Revisit when** the production rollout happens (alerting and on-call are needed; add-on `betrieb`). + +--- + +## D11 · Deployment + +**Problem.** The same system must run locally in a reproducible way, as a public showcase, and +later in an unknown customer environment. + +**Decision.** + +| Environment | Setup | +|---|---| +| **Local** | `docker compose up` starts `postgres:17`, SeaweedFS, `web`, `worker` and `ai`. There are two images: the TS app (Next.js standalone output, two commands) and the Python AI service. `.env.example` documents every variable; secrets never go into the repo. | +| **Showcase** (after pilot acceptance) | TS app on Vercel (Hobby) + Neon (Frankfurt, via the Marketplace) + R2 (EU) + Vertex `eu`. Demo accounts are invite-only, data is synthetic only, a banner says "Demo – synthetic data only", and upload limits plus rate limits cap costs. **The AI service host is decided by a spike:** a Vercel Python Function (5 GB package limit, 300 s max on Hobby; whether docling fits is unverified) or Google Cloud Run in the EU (same GCP project as Vertex). | +| **CI** | GitHub Actions from the framework template. TS `verify` = lint + typecheck + unit and integration tests (Postgres service container) + architecture check + build + `pnpm audit`. A Python job = ruff + pyright + pytest, targeted at `services/ai/` and `contracts/`. Evals and the contract check run on relevant paths or the `verify-full` label. Docker images are built only when the Dockerfiles or the compose file change. | +| **Production path** | The same images in the customer's EU cloud or on-prem (Compose or Kubernetes); managed PostgreSQL with PITR; versioned object storage; worker and AI service as their own deployments; Entra SSO; Vertex `eu` with zero data retention. | + +**Alternatives.** +1. *Showcase on Fly.io/Railway/Hetzner with the real containers* – closest to production, but it + costs money every month and adds ops effort, and the orchestrator chose Vercel. +2. *Vercel only, no Docker* – loses the on-prem path and the offline reproducibility. + +**Trade-offs.** There are three build artefacts (the Vercel build and two images); CI covers them +path-targeted. The showcase deviates from production on retries (D2); that is recorded in the +exceptions register (`docs/technical/architecture.md`). + +**Rationale.** Reviewers can run everything with one command; production uses the same images. + +**Pilot cost.** Local setup is part of the foundation. The showcase is about 1 day after +acceptance, including the AI-service hosting spike. *Heuristic.* +**Revisit when** the customer environment is known. + +--- + +## Summary of the challenged decisions + +| # | Question | Verdict | Decisive argument | +|---|---|---|---| +| 1 | Full TS vs. Python AI service | **Python AI service from day 1**, stateless, called by the TS worker (orchestrator decision; the draft recommended full TS) | docling handles tables, scans and `.msg` from day 1; security, tenancy and the queue stay in TS because the service holds no state | +| 2 | pg-boss vs. Redis queue | **pg-boss** | A transactional enqueue removes the dual-write risk behind DR3/DR4; no extra infrastructure | +| 3 | Better Auth vs. managed auth | **Better Auth** (pinned, minimal plugins) | Clerk and WorkOS offer no EU residency; runs offline; organisations = tenants; path to Entra SSO | +| 4 | App-level vs. RLS | **Both** – repository scoping + forced RLS | One forgotten filter must not become a leak; the cost is about 1 day | +| 5 | S3-compatible provider | **S3 API**: SeaweedFS locally, R2 EU for the showcase, the customer's choice in production | MinIO is dead; Vercel Blob is not S3 | + +## Consequences + +**Positive** +- DR1, DR3, DR4 and DR5 are enforced by mechanisms (verifier, transaction, idempotency, RLS), not + by conventions. +- Every external dependency sits behind a port or contract: model, storage, ERP, intake, AI + service. +- The local demo needs nothing but Docker plus model credentials. + +**Negative / accepted** +- There are two toolchains and a service contract to maintain. +- On the showcase, retries are only picked up when something triggers `drain()`. +- We own the auth configuration and have to watch the Better Auth advisories. +- The AI service's showcase host and container footprint are unverified; a spike is due before the + showcase. +- The pilot takes ~13 days, not 10 (confirmed by the orchestrator). + +## Pilot budget check (heuristic, solo developer) + +| Work package | Decision | Days | +|---|---|---| +| Foundation: repo, Docker Compose, TS CI, module skeleton | D1, D11 | 1 | +| AI service foundation: FastAPI skeleton, Python CI, OpenAPI contract, generated TS client | D8 | 1.5 | +| Database schema, migrations, database roles | D3 | 0.5 | +| Auth, companies, roles (invite form + seed, no role-admin UI) | D6 | 0.5 | +| Tenant isolation: RLS, `withTenant`, cross-tenant tests | D7 | 1 | +| Intake upload, object storage, exact-duplicate check | D5, D9 | 0.75 | +| Jobs: status machine, pg-boss, retry, dead-letter, `drain()` | D2, D4 | 1.5 | +| Parsing to segments with docling | D8 | 1 | +| Extraction + grounding verifier | D8 | 1.5 | +| Worker ↔ AI service integration (client, timeouts, error mapping) | D8 | 0.5 | +| Review UI: value beside its highlighted source, corrections, approve/reject, history | DR2, DR6 | 1.5 | +| Export + ERP mock + idempotency + contract tests | D9 | 1 | +| Eval runner + 15 synthetic cases + CI gate | D8 | 0.75 | +| Observability: logs, health, status/errors in the request list | D10 | 0.25 | +| **Pilot total (confirmed scope)** | | **≈ 13** (13.25) | +| Showcase deployment incl. AI-service hosting spike – after acceptance | D11 | 1 | + +**Out of the pilot, planned afterwards:** the showcase, near-duplicate hints, the role-admin UI, a +separate ops page, the IMAP/Graph mailbox import and the real ERP mapping. Security (D6/D7), the +grounding verifier (D8) and exactly-once export (D9) are **not** cut: they are the customer's +explicit requirements. + +## Exceptions register entries (recorded in `docs/technical/architecture.md`) + +| Exception | Why accepted | Expires | +|---|---|---| +| Showcase: no unattended retries (Hobby cron once/day) | Showcase only; production runs a worker | when a production-like demo is needed | +| No RLS on the `auth`/`pgboss` schemas | Not tenant business data; only server code has access | on review at M3 | +| Gemini free tier for local development | Synthetic data only; never in showcase or production | when the Vertex budget is set up for development | + +## Open points + +**Customer input needed** +- The real mail system (M365/Graph, Exchange on-prem, IMAP) and the ERP (product, interface type, + target fields). +- A sample set of real, anonymised requests: share of scans, languages, table-heavy PDFs. +- The permitted AI provider and region; whether a zero-data-retention exemption is required. +- Retention periods for original documents and audit data. +- Identity provider for production (Entra ID?), MFA requirement. +- Acceptance thresholds per key field for the eval gate. + +**To verify during implementation** +- pg-boss maintenance API for serverless `drain()`. +- docling EML/MSG coverage and its provenance granularity for XLSX cells. +- Google Gen AI SDK configuration for the Vertex `eu` endpoint. +- R2 EU jurisdiction on the free plan. +- The AI-service showcase host (spike). + +## Sources (verified 2026-09-22) + +- pg-boss: github.com/timgit/pg-boss · pgboss.io/api/jobs · pgboss.io/api/adapters · pgboss.io/api/constructor +- Vercel: vercel.com/docs/cron-jobs/usage-and-pricing · vercel.com/docs/functions/limitations · + nextjs.org/docs/app/api-reference/functions/after · vercel.com/changelog/vercel-queues-now-in-public-beta · + vercel.com/docs/workflows/pricing · vercel.com/docs/vercel-blob · vercel.com/docs/oidc/gcp +- Neon: neon.com/docs/introduction/plans · neon.com/docs/introduction/regions · neon.com/docs/connect/connection-pooling +- MinIO: github.com/minio/minio · stablebuild.com/blog/minio-images-disappeared-from-docker-hub +- Storage: developers.cloudflare.com/r2/pricing · developers.cloudflare.com/r2/reference/data-location · + docs.hetzner.com/storage/object-storage/overview +- Better Auth: better-auth.com/docs/plugins/organization · /docs/plugins/admin · /docs/plugins/sso · + /docs/adapters/drizzle · /docs/concepts/rate-limit · github.com/better-auth/better-auth/security/advisories · + better-auth.com/blog/authjs-joins-better-auth · better-auth.com/blog/better-auth-joins-vercel +- Managed auth: clerk.com/security · clerk.com/pricing · workos.com/legal/data-processing-addendum · workos.com/pricing +- Gemini/Vertex: ai.google.dev/gemini-api/terms · docs.cloud.google.com/vertex-ai/generative-ai/docs/learn/data-residency · + …/docs/learn/model-versions · …/docs/data-governance · …/multimodal/document-understanding +- AI SDK (alternative 1 in D8): ai-sdk.dev/docs/migration-guides/migration-guide-6-0 · migration-guide-7-0 +- Parsing: github.com/docling-project/docling · pypi.org/project/PyMuPDF · pypi.org/project/pdfplumber · + npm registry (alternative 1: unpdf, mammoth, postal-mime, @kenjiuno/msgreader) · docs.sheetjs.com +- promptfoo: github.com/promptfoo/promptfoo · promptfoo.dev/docs/providers/vertex diff --git a/docs/decisions/INDEX.md b/docs/decisions/INDEX.md new file mode 100644 index 0000000..17ef4db --- /dev/null +++ b/docs/decisions/INDEX.md @@ -0,0 +1,7 @@ +# Decision index + +ADRs are immutable once accepted; a new ADR supersedes an old one with a note ("supersedes ADR-x"). + +| ADR | Date | Title | Status | +|---|---|---|---| +| [ADR-0001](ADR-0001-pilot-architecture.md) | 2026-09-22 | Pilot architecture baseline (D1–D11) | Accepted | diff --git a/docs/input/2026-09-22-kundenanfrage.md b/docs/input/2026-09-22-kundenanfrage.md new file mode 100644 index 0000000..90b49d1 --- /dev/null +++ b/docs/input/2026-09-22-kundenanfrage.md @@ -0,0 +1,61 @@ +# Kundenanfrage (Original, wörtlich) + +> Eingang: 2026-09-22 · Absender: fiktiver Kunde (mittelständisches Unternehmen, Anlagen- und Maschinenbau) +> Nur Absätze wiederhergestellt, Wortlaut unverändert. Referenz-/Testprojekt: Die Anfrage ist ein +> Übungsfall ohne realen Absender und darf ins öffentliche Repo (PROJECT-START.md §3). + +--- + +Hallo Florian, + +wir sind ein mittelständisches Unternehmen im Bereich Anlagen- und Maschinenbau und möchten einen aktuell größtenteils manuellen Prozess digitalisieren. + +Unser Vertrieb erhält täglich etwa 20–50 Angebotsanfragen per E-Mail. Die relevanten Informationen befinden sich entweder direkt im E-Mail-Text oder in Anhängen wie PDFs, Excel-Dateien und Word-Dokumenten. Aktuell müssen unsere Mitarbeiter jede Anfrage öffnen, die Informationen manuell heraussuchen und anschließend in unsere interne Auftragsübersicht übertragen. + +Wir würden diesen Prozess gerne mit einer internen Webanwendung und KI-Unterstützung vereinfachen. + +## Gewünschter Ablauf + +Eine neue Anfrage soll entweder automatisch aus einem Postfach eingelesen oder von einem Mitarbeiter manuell hochgeladen werden können. + +Das System soll anschließend möglichst automatisch Informationen wie Unternehmen, Ansprechpartner, Kontaktdaten, angefragte Produkte/Positionen, Mengen, Materialien, Abmessungen, gewünschten Liefertermin und zusätzliche Anforderungen aus den vorhandenen Dokumenten erkennen. + +Uns ist dabei wichtig, dass die KI keine Informationen erfindet. Wenn etwas nicht eindeutig erkannt werden kann, soll das entsprechend gekennzeichnet werden. + +Ein Mitarbeiter soll die Ergebnisse anschließend in einer übersichtlichen Oberfläche prüfen können. Idealerweise möchten wir neben dem erkannten Wert auch sehen können, aus welchem Dokument bzw. welcher Stelle die Information stammt. + +Der Mitarbeiter soll Werte korrigieren und die Anfrage anschließend freigeben oder ablehnen können. + +Nach der Freigabe sollen die strukturierten Daten gespeichert und an unser bestehendes ERP-System übertragen werden. Für die erste Version reicht hier eine simulierte REST-Schnittstelle, da wir die echte ERP-Anbindung später gemeinsam umsetzen würden. + +## Weitere Anforderungen + +Die Anwendung soll von mehreren Mitarbeitern genutzt werden können und benötigt daher einen Login sowie eine einfache Benutzer- und Rechteverwaltung. Anfragen sollten beispielsweise die Status Neu, Verarbeitung, Prüfung, Freigegeben, Fehler und Exportiert besitzen. + +Außerdem benötigen wir eine nachvollziehbare Historie darüber, wer welche Daten geändert oder freigegeben hat. + +Doppelt eingegangene bzw. hochgeladene Anfragen sollten nach Möglichkeit erkannt werden. Insbesondere darf eine Anfrage nicht versehentlich zweimal an unser ERP übertragen werden. + +Falls die KI-Schnittstelle oder ein anderer externer Dienst vorübergehend nicht verfügbar ist, darf die Anfrage nicht verloren gehen. Sie sollte später erneut verarbeitet werden können und der Fehler sollte für den Mitarbeiter sichtbar sein. + +Da zukünftig möglicherweise mehrere unserer Gesellschaften das System verwenden sollen, sollte bei der technischen Umsetzung berücksichtigt werden, dass Benutzer ausschließlich die Daten ihres jeweiligen Unternehmens sehen dürfen. + +## Qualität der KI + +Für uns wäre außerdem wichtig, nachvollziehen zu können, wie zuverlässig die automatische Extraktion funktioniert. + +Wir möchten deshalb für die wichtigsten Felder eine kleine Testsammlung aufbauen, mit der beispielsweise nach Änderungen an Prompts oder Modellen überprüft werden kann, ob sich die Extraktionsqualität verschlechtert hat. + +## Erste Version + +Für den ersten Pilot benötigen wir noch keine perfekte Enterprise-Lösung. Uns ist wichtiger, innerhalb kurzer Zeit eine vollständig nutzbare End-to-End-Version zu erhalten, die wir mit einigen Mitarbeitern testen können. + +Der Pilot sollte mindestens folgenden Prozess vollständig abdecken: + +Anfrage erhalten/hochladen → Dokumente verarbeiten → Informationen extrahieren → Ergebnisse prüfen und korrigieren → Anfrage freigeben → Daten über Schnittstelle exportieren. + +Die konkrete technische Architektur würden wir gerne dir überlassen. Wichtig sind uns eine saubere technische Umsetzung, Erweiterbarkeit und ein nachvollziehbarer Betrieb. + +Wenn der Pilot funktioniert, würden wir im nächsten Schritt gerne über die Anbindung unseres tatsächlichen E-Mail-Postfachs und ERP-Systems sowie einen produktiven Rollout sprechen. + +Kannst du uns bitte einen Vorschlag schicken, wie du das Projekt technisch und organisatorisch angehen würdest, welche offenen Fragen du vor Projektstart noch hast und welchen Umfang du für einen ersten Pilot empfehlen würdest? diff --git a/docs/product/pilot-vorschlag.md b/docs/product/pilot-vorschlag.md new file mode 100644 index 0000000..774ccd5 --- /dev/null +++ b/docs/product/pilot-vorschlag.md @@ -0,0 +1,158 @@ +# RequestFlow – Vorschlag für den Pilot + +> Antwort auf eure Anfrage vom 22.09.2026 · Stand: Entwurf 2026-09-22 +> Grundlage: `docs/input/2026-09-22-kundenanfrage.md`, Architekturentscheidung `docs/decisions/ADR-0001-pilot-architecture.md` + +Hallo zusammen, + +danke für die ausführliche Beschreibung – sie macht es leicht, konkret zu werden. Im Folgenden +beschreibe ich, wie ich das Projekt technisch und organisatorisch angehen würde, welchen +Pilot-Umfang ich empfehle und welche Fragen wir vor dem Start klären sollten. + +--- + +## 1. Worum es geht – mein Verständnis + +Euer Vertrieb erhält täglich 20–50 Angebotsanfragen per E-Mail. Die Angaben stecken im Mailtext +oder in PDF-, Excel- und Word-Anhängen und werden heute von Hand in die Auftragsübersicht +übertragen. Ziel ist eine interne Webanwendung, die diese Angaben automatisch vorbereitet, +**ohne dass die KI etwas erfindet** – und in der eure Mitarbeiter jede Angabe mit einem Blick +auf die Quelle prüfen, korrigieren und freigeben, bevor sie genau einmal ans ERP gehen. + +## 2. So würde der Pilot aus Sicht eurer Mitarbeiter funktionieren + +1. **Anfrage hochladen** – als gespeicherte E-Mail (`.eml`/`.msg`) oder als einzelne Dokumente. + Doppelt hochgeladene Anfragen erkennt das System und weist darauf hin. +2. **Automatische Verarbeitung** – das System liest Mailtext und Anhänge (auch Tabellen und + gescannte PDFs) und erkennt: Unternehmen, Ansprechpartner, Kontaktdaten, Positionen mit Menge, + Material und Abmessungen, gewünschten Liefertermin und Zusatzanforderungen. +3. **Prüfen** – jede erkannte Angabe steht neben ihrer **Fundstelle** (Dokument, Seite bzw. + Tabellenzelle, markierte Textstelle). Jede Angabe trägt einen Status: + *erkannt* · *unsicher* · *nicht gefunden* · *nicht belegt*. +4. **Korrigieren und freigeben oder ablehnen** – jede Änderung und jede Freigabe wird mit Person + und Zeitpunkt in der Historie festgehalten. +5. **Export** – freigegebene Anfragen gehen über eine REST-Schnittstelle an das ERP (im Pilot: + eine simulierte Schnittstelle, die sich wie ein echtes ERP verhält, inklusive Ausfällen). + +Status einer Anfrage: **Neu → Verarbeitung → Prüfung → Freigegeben → Exportiert**, außerdem +**Abgelehnt** und **Fehler** (mit sichtbarer Ursache und automatischer Wiederholung). + +## 3. Wie ich eure wichtigsten Anforderungen absichere + +| Eure Anforderung | Wie sie umgesetzt wird | +|---|---| +| **Die KI darf nichts erfinden** | Die KI muss zu jedem Wert die wörtliche Textstelle liefern. Das System **prüft selbst**, ob diese Stelle im Dokument wirklich steht und zum Wert passt. Ist das nicht der Fall, wird der Wert als *nicht belegt* markiert und muss geprüft werden. Die KI hat nie das letzte Wort. | +| **Quelle sichtbar** | Jeder Wert verweist auf Dokument und Stelle; die Prüfansicht zeigt beides nebeneinander. | +| **Nie doppelt ans ERP** | Jede Anfrage trägt einen eindeutigen Schlüssel, den die Schnittstelle bei Wiederholungen erkennt; zusätzlich schließt die Datenbank einen zweiten Export technisch aus. Das wird automatisiert getestet. | +| **Nichts geht verloren** | Eine Anfrage wird erst als angenommen bestätigt, wenn sie samt Verarbeitungsauftrag gespeichert ist. Fällt die KI oder das ERP aus, wird automatisch später wiederholt; der Fehler ist in der Übersicht sichtbar, eine manuelle Neuverarbeitung ist möglich. | +| **Jede Gesellschaft sieht nur ihre Daten** | Doppelt abgesichert: in der Anwendung und zusätzlich direkt in der Datenbank (Row-Level Security). Auch ein Programmierfehler in einer Abfrage kann keine fremden Daten liefern. Getestet mit zwei Gesellschaften. | +| **Nachvollziehbare Historie** | Jede Änderung wird zusammen mit der Änderung selbst gespeichert (wer, wann, alter/neuer Wert); die Historie ist nachträglich nicht veränderbar. | +| **Qualität der KI messbar** | Eine Testsammlung mit Beispielanfragen und erwarteten Ergebnissen misst je Feld: Trefferquote, erkannte Lücken, belegte Werte und „erfundene" Werte. Nach jeder Änderung an Prompt oder Modell läuft sie automatisch; verschlechtert sich ein Kernfeld, fällt der Check rot aus. | + +## 4. Technischer Ansatz (Kurzfassung) + +- **Webanwendung** (TypeScript, Next.js) als modularer Monolith: ein System mit klar getrennten + Bausteinen (Eingang, Verarbeitung, Prüfung, Export, Benutzer, Historie) – einfach zu betreiben, + gut erweiterbar. +- **KI-Dienst** (Python) für Dokumentenverarbeitung und Extraktion – mit docling, einer + Open-Source-Bibliothek, die auch Tabellen und gescannte Dokumente zuverlässig liest. Der Dienst + speichert selbst nichts und hat keinen Zugriff auf die Datenbank. +- **KI-Modell:** Google Gemini über Vertex AI mit **Verarbeitung in der EU**; laut + Google-Cloud-Bedingungen werden eure Daten nicht zum Training verwendet. Das Modell ist + austauschbar – ein Wechsel wird über die Testsammlung abgesichert. +- **Datenhaltung:** PostgreSQL für Anfragen, Ergebnisse und Historie; Dokumente in einem privaten, + S3-kompatiblen Speicher. Hintergrundverarbeitung über eine Warteschlange in derselben Datenbank. +- **Anmeldung:** eigene Benutzerverwaltung (Einladung durch einen Admin, Rollen *Admin* und + *Sachbearbeitung*), vorbereitet für die spätere Anmeldung über euer Microsoft-Konto (Entra ID). +- **Betrieb:** alles läuft als Container und kann in eurer Umgebung oder in einer EU-Cloud + betrieben werden. Fehler, Wiederholungen und KI-Nutzung sind nachvollziehbar protokolliert – + ohne Dokumentinhalte oder personenbezogene Daten in den Logs. + +Details mit Begründungen und verworfenen Alternativen: Architekturentscheidung ADR-0001. + +## 5. Empfohlener Pilot-Umfang + +**Im Pilot enthalten** +- Upload von E-Mails (`.eml`, `.msg`) und Dokumenten (PDF inkl. Scans, Excel, Word) +- Automatische Extraktion der oben genannten Felder mit Fundstellen und Prüfung auf Belege +- Prüfansicht mit Korrektur, Freigabe und Ablehnung; Historie +- Status inkl. Fehlerstatus, automatische Wiederholung, Erkennung exakter Duplikate +- Export über die simulierte ERP-Schnittstelle (dokumentierter Schnittstellenvertrag) +- Anmeldung mit Einladung, zwei Rollen, Datenmodell für mehrere Gesellschaften +- Testsammlung (Start: 15 Fälle) mit automatischer Qualitätsmessung + +**Bewusst nach dem Pilot** +- Anbindung eures echten Postfachs und eures echten ERP-Systems +- Anmeldung über Microsoft (Entra ID / SSO) +- Hinweise auf *ähnliche* (nicht identische) Anfragen, Rollenverwaltung per Oberfläche, + eigene Betriebsübersicht +- Produktiver Rollout (Monitoring, Backups mit geprobter Wiederherstellung, Rollback) + +## 6. Vorgehen und Zeitplan + +| Phase | Inhalt | Dauer (Vorschlag) | +|---|---|---| +| **0 · Kickoff** | Ziele, Feldliste und Abnahmekriterien festlegen; 20–30 anonymisierte Beispielanfragen; Ansprechpartner | 1 Termin + Übergabe der Beispiele | +| **1 · Pilot-Entwicklung** | Umsetzung in kleinen, prüfbaren Schritten; **wöchentliche Demo** mit lauffähigem Stand | ca. 13 Arbeitstage (≈ 3 Wochen) | +| **2 · Pilotbetrieb** | 3–5 eurer Mitarbeiter arbeiten mit echten Anfragen; Feedback und Qualitätsmessung | 2 Wochen (Vorschlag) | +| **3 · Abnahme & Ausblick** | Abnahme gegen die vereinbarten Kriterien; Entscheidung über Postfach-/ERP-Anbindung und Rollout | 1 Termin | + +**Zusammenarbeit:** ein fester fachlicher Ansprechpartner auf eurer Seite, der Fragen entscheidet; +offene Punkte und Fortschritt in einem gemeinsamen Board; jede Entscheidung mit Tragweite wird +schriftlich festgehalten. Neue oder unklare Anforderungen setze ich nicht stillschweigend um, +sondern kläre sie vorher mit euch. + +## 7. Abnahmekriterien (Vorschlag – Zielwerte legen wir im Kickoff fest) + +- [ ] Eine Anfrage mit Anhängen läuft ohne Medienbruch vom Upload bis zum Export. +- [ ] Kein Wert steht als *erkannt* ohne überprüften Beleg im Dokument. +- [ ] Extraktionsqualität je Kernfeld ist gemessen und erreicht die vereinbarten Zielwerte. +- [ ] Doppelter Export und Verlust bei simuliertem Ausfall sind per Test ausgeschlossen. +- [ ] Mitarbeiter einer Gesellschaft sehen nachweislich keine Daten einer anderen. +- [ ] Die Historie zeigt jede Änderung und Freigabe mit Person und Zeitpunkt. +- [ ] Die Bearbeitungszeit pro Anfrage ist im Pilotbetrieb gegenüber heute gemessen. + +## 8. Datenschutz + +- Verarbeitung in der EU (Anwendung, Datenbank, Dokumentenspeicher, KI-Modell). +- Auftragsverarbeitungsvertrag (AVV) zwischen uns, sofern ich den Pilot betreibe; die + Unterauftragsverarbeiter (Hosting, KI-Anbieter) lege ich vollständig offen. +- Datenminimierung: Logs enthalten keine Dokumentinhalte; Aufbewahrungsfristen legen wir gemeinsam fest. +- Für Entwicklung und Tests verwende ich ausschließlich synthetische oder von euch anonymisierte Daten. + +## 9. Offene Fragen vor dem Start + +**Fachlich** +1. Wer entscheidet fachlich und nimmt den Pilot ab? +2. Welche Felder braucht eure Auftragsübersicht genau (Feldliste/Beispiel-Datensatz)? +3. Wie sieht heute eine „gute" Anfrage vs. eine schwierige aus – gibt es typische Sonderfälle? +4. Wie lange dauert die manuelle Erfassung heute pro Anfrage (für den Vorher-Nachher-Vergleich)? + +**Daten und Dokumente** +5. Können wir 20–30 anonymisierte Beispielanfragen bekommen (inkl. Anhänge)? +6. Wie hoch ist der Anteil gescannter PDFs, Zeichnungen oder Bilder? Welche Sprachen kommen vor? +7. Wie lange sollen Original-Mails, Dokumente und Historie aufbewahrt werden? + +**IT und Integration** +8. Welches ERP-System nutzt ihr, und welche Schnittstelle bietet es (REST, Datei-Import, Middleware)? +9. Welches Mailsystem (Microsoft 365, Exchange on-prem, anderes)? +10. Wo soll der Pilot laufen – in eurer Umgebung (Container) oder von mir in einer EU-Cloud betrieben? +11. Nutzt ihr Microsoft Entra ID für die Anmeldung, und ist MFA Pflicht? + +**Datenschutz und Organisation** +12. Welche KI-Anbieter/Regionen sind bei euch zulässig? Ist eine Zusage ohne Datenspeicherung beim KI-Anbieter (Zero Data Retention) erforderlich? +13. Wer ist bei euch Ansprechpartner für Datenschutz (AVV)? +14. Wie viele Mitarbeiter sollen im Pilot testen, und aus wie vielen Gesellschaften? + +## 10. Aufwand und nächste Schritte + +- **Pilot-Entwicklung:** ca. 13 Arbeitstage; Pilotbetrieb-Begleitung nach Aufwand. +- **Konditionen:** [Tagessatz / Festpreis eintragen – kommerzielles Angebot] +- **Laufende Kosten im Pilot:** KI-Nutzung und Hosting im niedrigen Bereich; genaue Zahlen nach der + ersten Messung mit euren Beispielanfragen. + +**Nächste Schritte:** Kickoff-Termin vereinbaren, Beispielanfragen und Feldliste bereitstellen, +fachlichen Ansprechpartner benennen. Danach starte ich mit der Umsetzung. + +Viele Grüße +Florian Hein diff --git a/docs/product/project-brief.md b/docs/product/project-brief.md new file mode 100644 index 0000000..a7caf4f --- /dev/null +++ b/docs/product/project-brief.md @@ -0,0 +1,49 @@ +# Project brief – RequestFlow + +> Add-on `saas-auftrag`, phase 1. Source: [customer request](../input/2026-09-22-kundenanfrage.md), +> discovery in [`PROJECT-START.md`](../../PROJECT-START.md). Reference project: the customer is +> fictional and treated as real; all data is synthetic. + +## Problem + +The sales team of a mid-sized machine-building company receives about 20–50 quote requests per +day by e-mail. The relevant information sits in the mail body or in PDF, Excel and Word +attachments; staff read every request and copy the data by hand into an internal order overview. + +## Users and roles + +- **Clerk (sales):** uploads requests, reviews extracted fields beside their source, corrects, + approves or rejects, sees errors and triggers reprocessing – own company only. +- **Admin (per company):** additionally invites users and assigns roles. +- **System (worker, AI service):** processes, extracts and exports; every action is audited. + +## Success (measurable) + +Proposed – target values are agreed with the customer at kickoff. +- Pilot users process a request from upload to export without leaving the app. +- No field is shown as *found* without verified evidence in the source document. +- Extraction quality per key field is measured on the eval set and meets the agreed targets. +- Double export and loss on simulated outages are excluded by automated tests. +- Handling time per request is measured against today's baseline. + +## Non-goals (pilot) + +Real ERP and real mailbox integration, SSO (Entra ID), near-duplicate hints, role-admin UI, +separate ops page, productive rollout (monitoring, rehearsed restore, rollback). + +## Risks + +- **Data protection:** personal and confidential data reach an LLM – EU endpoint, DPA chain, + no training, zero data retention for production (ADR-0001 D8). +- **External dependencies:** AI provider outages, model retirement (Gemini 2.5 retires + 2026-10-20) – retries, model ID as configuration, eval gate on every model change. +- **Unclear domain rules:** real field list and ERP semantics are unknown – mock + contract now, + mapping with the customer after the pilot. +- **Document variety:** scans and table-heavy PDFs may underperform – measured by weighted eval + cases; docling OCR/table models. +- **AI cost:** capped by rate/upload limits and a 10 €/month GCP budget alert. + +## Open decisions + +Customer questions: [pilot proposal §9](pilot-vorschlag.md#9-offene-fragen-vor-dem-start). +Implementation verifications: [ADR-0001 → Open points](../decisions/ADR-0001-pilot-architecture.md#open-points). diff --git a/docs/product/roadmap.md b/docs/product/roadmap.md new file mode 100644 index 0000000..5b6b144 --- /dev/null +++ b/docs/product/roadmap.md @@ -0,0 +1,15 @@ +# Roadmap – RequestFlow + +> Add-on `saas-auftrag`, phase 3. Milestones have a verifiable acceptance – never "80 % done". +> Work items live in GitHub issues; this file only names the milestones. + +| Milestone | Content | Acceptance | Status | +|---|---|---|---| +| **M0 – Problem confirmed** | Discovery, proposal, architecture (ADR-0001), foundation | Proposal sent; §7 approval; foundation PR merged | in progress | +| **M1 – Secure core** | Epic 1 vertical slice: one request end to end (login, tenant isolation, upload, AI extraction with grounding, review, idempotent export) | Slice demo on local Docker; cross-tenant and idempotency tests green | open | +| **M2 – Internal beta (pilot)** | Epic 2 full extraction + eval gate, Epic 3 robustness | Pilot acceptance criteria (proposal §7) met with 3–5 test users | open | +| **M3 – Acceptance-ready** | Findings from the pilot, security review, stage change P1 → P2 prepared | Customer acceptance against agreed thresholds | open | +| **M4 – Production** | Real mailbox + ERP, Entra SSO, monitoring, backup + rehearsed restore, rollback | Release approval by the orchestrator | open | +| **M5 – Expansion** | Further subsidiaries, near-duplicate hints, ops page | Per new brief | open | + +The public showcase (Vercel) follows M2 acceptance (ADR-0001 D11). diff --git a/docs/technical/architecture.md b/docs/technical/architecture.md new file mode 100644 index 0000000..721d8be --- /dev/null +++ b/docs/technical/architecture.md @@ -0,0 +1,70 @@ +# Architecture – RequestFlow + +> Living document: whoever changes the structure changes this document **in the same PR**. +> Altitude: modules/folders, not single files. Current state – plans belong in specs and ADRs. +> Guard: the reviewer role. Beyond ~3 screens it gets condensed, not extended. + +## Summary + +RequestFlow receives quote requests (e-mail + PDF/Excel/Word attachments), extracts structured +fields with source evidence via a stateless Python AI service, lets staff review, correct and +approve them, and exports each approved request exactly once to an ERP (mock in the pilot). +It is a TypeScript modular monolith (`web` + `worker` from one codebase) on PostgreSQL, plus the +AI service. Decisions and rationale: [ADR-0001](../decisions/ADR-0001-pilot-architecture.md). + +**Current state (2026-09-22): foundation only – no module is built yet.** Status per module below. + +## Modules + +Every new file belongs to one of these modules – otherwise add the module here first (SYSTEM.md §7). + +| Module | Location | Task | Exposure | Data class | Protection | Status | +|---|---|---|---|---|---|---| +| `intake` | `src/features/intake/` | upload, duplicate fingerprint, creates request + documents | authenticated UI/route | confidential + personal | session, tenant context, size/type limits | planned | +| `documents` | `src/features/documents/` | document records, storage references, hashes | internal | confidential | tenant context | planned | +| `extraction` | `src/features/extraction/` | AI-service client, persists runs/fields/evidence | internal | confidential + personal | tenant context, contract validation | planned | +| `requests` | `src/features/requests/` | request aggregate, status machine | internal | confidential | tenant context | planned | +| `review` | `src/features/review/` | review UI, corrections, approve/reject | authenticated UI | confidential + personal | session, role check, audit | planned | +| `export` | `src/features/export/` | ERP port + REST adapter, idempotency | outbound HTTP | confidential | idempotency key, unique export, timeout | planned | +| `erp-mock` | `src/features/erp-mock/` | simulated ERP REST API | route behind flag | synthetic | disabled unless `ERP_MOCK_ENABLED` | planned | +| `identity` | `src/features/identity/` | Better Auth, users, companies, roles | public login route | personal (staff) | rate limit, invite-only | planned | +| `tenancy` | `src/features/tenancy/` | `withTenant()`, RLS policies | internal | – | forced RLS, `app_rw` without BYPASSRLS | planned | +| `audit` | `src/features/audit/` | append-only audit events | internal | personal (staff) | INSERT/SELECT only | planned | +| `jobs` | `src/features/jobs/` | pg-boss, job handlers, `drain()`, worker entrypoint | internal | IDs only | transactional enqueue | planned | +| `storage` | `src/features/storage/` | `BlobStore` port + S3 adapter | internal | confidential | private bucket, access via app routes | planned | +| `observability` | `src/features/observability/` | logger, health, request-list ops data | `/api/health` | IDs only | no PII in logs | planned | +| `db` | `src/db/` | Drizzle schema, migrations, DB roles | internal | – | migrations as owner role | planned | +| AI service | `services/ai/` | docling parsing, extraction, grounding, evals | internal HTTP | confidential + personal (transient) | bearer token, stateless, no DB/storage access | planned | +| Contracts | `contracts/` | OpenAPI: AI service, ERP export | – | – | contract tests | planned | + +## Exceptions register + +Deliberately accepted risks – without an entry here a deviation counts as a defect. + +| Exception | Why accepted | Owner | Expires | +|---|---|---|---| +| No RLS on the `auth` and `pgboss` schemas | Not company-owned business data; reachable only by server code (ADR-0001 D7) | Fluory | 2026-12-31 (review at M3) | +| Showcase without unattended retries (Vercel Hobby cron once/day) | Showcase only; production runs a worker (D2) | Fluory | when a production-like demo is needed | +| Gemini API free tier for local development | Synthetic data only; never showcase or customer data (D8) | Fluory | when a Vertex development budget exists | + +## Data flow + +```text +upload ─► intake ─► storage (S3) + requests(NEW) + job ── one transaction +worker ─► jobs.drain ─► extraction ─► AI service (bytes in, segments + fields + evidence out) + ─► requests(REVIEW) + fields + audit ── one transaction +review ─► corrections + approve ─► requests(APPROVED) + export job + audit +worker ─► export ─► ERP (Idempotency-Key) ─► requests(EXPORTED) +failure at any step ─► retry with backoff ─► dead letter ─► requests(ERROR, visible cause) +``` + +## External services & interfaces + +| Service | Purpose | Environments | +|---|---|---| +| PostgreSQL 17 | all state incl. queue and auth | local container · showcase Neon (aws-eu-central-1) | +| S3-compatible storage | original mails and attachments | local SeaweedFS · showcase Cloudflare R2 (EU jurisdiction) | +| Vertex AI (`eu` endpoint, gemini-3.5-flash) | extraction | showcase + customer; local dev may use the Gemini free tier with synthetic data | +| ERP | export target | pilot: `erp-mock`; contract `contracts/erp-export.openapi.yaml` (planned) | + +No secrets in this document; configuration lives in `.env.example` (created with the first code issue). diff --git a/project-profile.yml b/project-profile.yml new file mode 100644 index 0000000..0877b20 --- /dev/null +++ b/project-profile.yml @@ -0,0 +1,48 @@ +# Project profile – determines which controls and add-ons apply (SYSTEM.md §10). +# Update in its own PR on a maturity change (P1 → P2 before the production rollout). + +system: + fluory_system_version: "2.1" # = Entwicklungsplan/templates/VERSION at adoption + last_system_review: 2026-09-22 + +project: + name: requestflow + type: saas # customer engagement: web app + AI service + stage: internal # P1 until pilot acceptance; P2 (production) before the productive rollout (M4) + +repository: + visibility: public + public_reason: "Reference project for the portfolio; fictional customer, synthetic data only" + sensitive_context: none # would be customer-data in a real engagement – then a private repo (PROJECT-START §3) + +governance: + p0_self_merge_exception: false + approved_by: Fluory + approved_at: 2026-09-22 + +agent: + sandbox: false # synthetic data only, no production credentials, Windows host without WSL2 – re-evaluate with real credentials or customer data + +budget: + frame: "Local Docker 0 € + AI usage; showcase on free tiers (Vercel Hobby, Neon Free, R2 Free) + Vertex AI with a GCP budget alert at 10 €/month" + +risk: + public_users: false + personal_data: true # contact data of requesters (synthetic in this repo) + company_data: true # quote requests, specifications + authentication: true + persistent_data: true + external_api: true # Vertex AI, ERP (mock), later mailbox + ai_or_rag: true + +addons: + security: true + datenschutz: true + datenbank: true + infrastruktur: false # activated with the showcase deployment + betrieb: true + api: true # ERP export contract, AI service contract + ki_rag: true + releases: false # activated with the first delivery to the customer + saas_auftrag: true + recht: false # activated with the public showcase (Impressum/Datenschutz/Lizenz); DPA/contract via saas_auftrag + datenschutz diff --git a/scripts/doku-check.sh b/scripts/doku-check.sh new file mode 100755 index 0000000..4d7cac8 --- /dev/null +++ b/scripts/doku-check.sh @@ -0,0 +1,171 @@ +#!/usr/bin/env bash +# ============================================================================= +# Docs guard – fluory-system (SYSTEM.md §2, §3, §6, §8, §9, §10, §12) +# Template: Entwicklungsplan/templates/base/scripts/doku-check.sh +# +# Checks the documentation truth of a project repo mechanically – seconds, no dependencies. +# Runs in the CI docs check on every PR (ci.yml) and in /finish-work. +# FAIL → exit 1 (blocks) WARN → hint OK → one line +# Usage: scripts/doku-check.sh [--warn-only] +# Budgets: DOKU_AGENTS_WARN=150 DOKU_AGENTS_MAX=200 DOKU_ARCH_WARN=150 DOKU_UNRELEASED_WARN=100 +# DOKU_RULE_WARN=60 +# Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +WARN_ONLY=0; [ "${1:-}" = "--warn-only" ] && WARN_ONLY=1 +cd "$(git rev-parse --show-toplevel 2>/dev/null || pwd)" || exit 1 + +AGENTS_WARN="${DOKU_AGENTS_WARN:-150}"; AGENTS_MAX="${DOKU_AGENTS_MAX:-200}" +ARCH_WARN="${DOKU_ARCH_WARN:-150}"; UNRELEASED_WARN="${DOKU_UNRELEASED_WARN:-100}" +RULE_WARN="${DOKU_RULE_WARN:-60}" +fails=0; warns=0 +ok() { printf 'OK %s\n' "$*"; } +warn() { printf 'WARN %s\n' "$*"; warns=$((warns+1)); } +fail() { printf 'FAIL %s\n' "$*"; fails=$((fails+1)); } +lines() { wc -l < "$1" | tr -d ' '; } +all_files() { + if git rev-parse --is-inside-work-tree >/dev/null 2>&1; then + git ls-files --cached --others --exclude-standard + else + find . -type f -not -path './.git/*' | sed 's#^\./##' + fi +} + +# --- 1 AGENTS.md – mandatory reading (§8, §9) -------------------------------------------- +if [ -f AGENTS.md ]; then + n="$(lines AGENTS.md)" + if [ "$n" -gt "$AGENTS_MAX" ]; then + fail "AGENTS.md has $n lines (> $AGENTS_MAX) – condense instead of extending (§9 mandatory-reading guard)" + elif [ "$n" -gt "$AGENTS_WARN" ]; then + warn "AGENTS.md has $n lines (guideline ≤ $AGENTS_WARN) – open a condensing issue" + else + ok "AGENTS.md $n lines" + fi + grep -qE '^## (Rules|Regeln)' AGENTS.md || fail "AGENTS.md has no section \"## Rules (short form)\" – the session-start card re-injects exactly this section" + grep -qE '<(Project name|command|verify command|Projektname|Befehl|verify-Kommando)' AGENTS.md && warn "AGENTS.md still contains template placeholders (, , )" +else + fail "AGENTS.md missing – adopt it from Entwicklungsplan/templates/base/ (§3)" +fi + +# --- 2 CLAUDE.md imports AGENTS.md ------------------------------------------------------- +if [ -f CLAUDE.md ]; then + if grep -qE '^@AGENTS\.md[[:space:]]*$' CLAUDE.md; then ok "CLAUDE.md imports AGENTS.md" + else fail "CLAUDE.md does not import AGENTS.md (line \"@AGENTS.md\" missing) – Claude Code does not read AGENTS.md otherwise"; fi + grep -q '^# Compact instructions' CLAUDE.md || fail "CLAUDE.md has no \"# Compact instructions\" section – a compaction would drop the work state (§6; template: templates/base/CLAUDE.md)" +else + fail "CLAUDE.md missing (template: templates/base/CLAUDE.md – contains the @AGENTS.md import)" +fi + +# --- 3 Profile (§3, §10, §12) ------------------------------------------------------------- +if [ -f project-profile.yml ]; then + ver="$(sed -n 's/^[[:space:]]*fluory_system_version:[[:space:]]*"\{0,1\}\([0-9][0-9.]*\)"\{0,1\}.*/\1/p' project-profile.yml | head -1)" + if [ -n "$ver" ]; then ok "project-profile.yml: fluory-system $ver" + else fail "project-profile.yml without a valid fluory_system_version (system: block, §3)"; fi + rev="$(sed -n 's/^[[:space:]]*last_system_review:[[:space:]]*\([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\).*/\1/p' project-profile.yml | head -1)" + [ -n "$rev" ] || warn "project-profile.yml: last_system_review not set (YYYY-MM-DD) – set it at the next version review (§12)" + grep -qE '^[[:space:]]*name:[[:space:]]*<' project-profile.yml && warn "project-profile.yml: placeholder not replaced" +else + fail "project-profile.yml missing (§10)" +fi + +# --- 4 Forbidden status/handover files (§2: duplicates drift) ----------------------------- +forbidden="$(all_files | grep -Ei '(^|/)(todo|status|notes|handoff[^/]*|handover[^/]*|session[-_]?log[^/]*)\.md$|(^|/)docs/(sessions|archive|handoff|handover)(/|$)' || true)" +if [ -n "$forbidden" ]; then + fail "forbidden status/handover files (§2, §9 – state belongs in issue/PR, history comes from git):" + printf '%s\n' "$forbidden" | sed 's/^/ /' +else + ok "no status/handover files" +fi + +# --- 5 CHANGELOG (§9, §12) ---------------------------------------------------------------- +if [ -f CHANGELOG.md ]; then + if grep -q '^## \[Unreleased\]' CHANGELOG.md; then + un="$(awk '/^## \[Unreleased\]/{f=1; next} /^## /{f=0} f' CHANGELOG.md | wc -l | tr -d ' ')" + if [ "$un" -gt "$UNRELEASED_WARN" ]; then + warn "CHANGELOG [Unreleased] has $un lines (> $UNRELEASED_WARN) – release due or condense (§12 CHANGELOG guard)" + else + ok "CHANGELOG [Unreleased] $un lines" + fi + else + fail "CHANGELOG.md without a section \"## [Unreleased]\" (Keep a Changelog, §9/§12)" + fi +else + fail "CHANGELOG.md missing (§3)" +fi + +# --- 6 Architecture map (§9, §10) --------------------------------------------------------- +arch="" +for a in docs/ARCHITEKTUR.md docs/technical/architecture.md; do + if [ -f "$a" ]; then arch="$a"; break; fi +done +if [ -n "$arch" ]; then + m="$(lines "$arch")" + if [ "$m" -gt "$ARCH_WARN" ]; then warn "$arch has $m lines (> ~3 screens) – condense (§9)" + else ok "$arch $m lines"; fi +else + fail "architecture map missing: docs/ARCHITEKTUR.md (P0) or docs/technical/architecture.md (from P1) – §10" +fi + +# --- 7 Guards installed (§5, §6) ---------------------------------------------------------- +if [ -f .claude/settings.json ]; then + ok ".claude/settings.json present" + grep -qE '"autoMemoryEnabled":[[:space:]]*false' .claude/settings.json || fail ".claude/settings.json: autoMemoryEnabled is not false – auto memory would carry project knowledge past GitHub (§10)" + for r in 'Read(.env)' 'Read(secrets/**)' 'Read(credentials/**)' 'Read(private-keys/**)' 'Read(production-dumps/**)'; do + grep -qF "\"$r\"" .claude/settings.json || fail ".claude/settings.json: permissions.deny lacks \"$r\" (§10; template: templates/base/.claude/settings.json)" + done +else warn ".claude/settings.json missing – adopt the guard hooks, read locks and auto-memory switch from templates/base/.claude/ (§5, §6, §10)"; fi +for h in session-start.sh guard-git.sh stop-check.sh pre-compact.sh filter-test-output.sh guard-checkpoint.sh guard-read.sh; do + if [ -f ".claude/hooks/$h" ]; then + [ -x ".claude/hooks/$h" ] || warn ".claude/hooks/$h is not executable (chmod +x)" + elif [ -f .claude/settings.json ]; then + warn ".claude/hooks/$h missing although settings.json references it" + fi +done + +if [ -f scripts/quiet-run.sh ]; then [ -x scripts/quiet-run.sh ] || warn "scripts/quiet-run.sh is not executable (chmod +x)" +else warn "scripts/quiet-run.sh missing – test output filter and verify recording inactive (§11; template: templates/base/scripts/)"; fi + +if [ -f scripts/pr-check.sh ]; then [ -x scripts/pr-check.sh ] || warn "scripts/pr-check.sh is not executable (chmod +x)" +else warn "scripts/pr-check.sh missing – PR guard (plan gate, docs decision, proof) inactive (§6; template: templates/base/scripts/)"; fi +if [ -f .github/PULL_REQUEST_TEMPLATE.md ]; then + for k in "Arbeitsstand" "Klartext" "Plan-Pflicht" "Nachweis" "Doku-Entscheidung" "Dateigrößen"; do + grep -q "$k" .github/PULL_REQUEST_TEMPLATE.md || warn ".github/PULL_REQUEST_TEMPLATE.md has no \"$k\" section – outdated (template: templates/base/PULL_REQUEST_TEMPLATE.md)" + done +else + warn ".github/PULL_REQUEST_TEMPLATE.md missing (§6)" +fi + +# --- 8 Dated folders (§3, §9) ------------------------------------------------------------- +for d in docs/input docs/reports; do + [ -d "$d" ] || continue + bad="$(find "$d" -type f ! -name 'README.md' ! -name '.gitkeep' 2>/dev/null | grep -vE "^$d/[0-9]{4}-[0-9]{2}-[0-9]{2}-" || true)" + if [ -n "$bad" ]; then + warn "$d/: files without date prefix YYYY-MM-DD- (§9):" + printf '%s\n' "$bad" | sed 's/^/ /' + fi +done + +# --- 9 Path-scoped rules (§8): scoped, small, not a second project card ------------------ +if [ -d .claude/rules ]; then + nr=0; unscoped=0 + while IFS= read -r r; do + [ -n "$r" ] || continue + nr=$((nr+1)) + if [ "$(head -1 "$r")" = "---" ] && sed -n '2,/^---$/p' "$r" | grep -q '^paths:'; then : + else unscoped=$((unscoped+1)); warn "$r has no paths: frontmatter – it loads in every session; scope it to the files it is about (§8)"; fi + rl="$(lines "$r")" + [ "$rl" -gt "$RULE_WARN" ] && warn "$r has $rl lines (guideline ≤ $RULE_WARN) – a rule file is a checklist, not a handbook (§8)" + done < <(find .claude/rules -type f -name '*.md' | sort) + [ "$nr" -gt 0 ] && ok ".claude/rules: $nr rule files, $((nr-unscoped)) path-scoped" || warn ".claude/rules/ is empty – area rules from templates/base/.claude/rules/ (§8)" +else + warn ".claude/rules/ missing – area rules from templates/base/.claude/rules/ load only when matching files are read (§8)" +fi + +# --- Summary ------------------------------------------------------------------------------ +echo "----" +if [ "$fails" -gt 0 ]; then + echo "Docs guard: $fails FAIL, $warns WARN" + [ "$WARN_ONLY" -eq 1 ] && exit 0 + exit 1 +fi +echo "Docs guard: green ($warns WARN)" +exit 0 diff --git a/scripts/pr-check.sh b/scripts/pr-check.sh new file mode 100755 index 0000000..b73f8c5 --- /dev/null +++ b/scripts/pr-check.sh @@ -0,0 +1,219 @@ +#!/usr/bin/env bash +# ============================================================================= +# PR guard – fluory-system (SYSTEM.md §4, §6, §7, §9, §11) +# Template: Entwicklungsplan/templates/base/scripts/pr-check.sh +# +# Checks the PR DESCRIPTION mechanically. ci.yml passes the body as PR_BODY on every PR: +# - plain-language section "Was ist passiert (Klartext)" filled, no template placeholder +# - "Doku-Entscheidung": exactly one top-level decision; "docs affected" needs ≥ 1 kind +# - "Plan-Pflicht": exactly one box; "trigger applies" needs a filled Impact Manifest +# - draft PR: "Arbeitsstand" names the next smallest step +# - non-draft PR: "Nachweis" reports verify green (no ready-for-review without it) +# - file size gate (§7): >300 hint, >500 decision, >800 justified exception, >1000 architecture +# finding in P1/P2 (issue or ADR); test files are hint-only; exemptions for generated code, +# lockfiles, fixtures, migrations, schemas, resources, docs, build output, config +# (extra globs: SIZE_IGNORE="a|b") +# - before creating: new shared/utility/adapter/service files need a documented search +# - PR_EXACTLY_ONE / PR_AT_LEAST_ONE: extra sections with checkbox rules (control center) +# It cannot judge whether a decision is right – that stays a review task. +# Env: PR_BODY (required) PR_DRAFT=true|false PR_TITLE BASE_REF=origin/main PR_REQUIRED_SECTIONS="A|B" +# PR_EXACTLY_ONE="A|B" PR_AT_LEAST_ONE="A|B" FAIL → exit 1, --warn-only → exit 0 +# Local: PR_BODY="$(gh pr view --json body -q .body)" PR_DRAFT=true scripts/pr-check.sh +# Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +WARN_ONLY=0; [ "${1:-}" = "--warn-only" ] && WARN_ONLY=1 +REQUIRED="${PR_REQUIRED_SECTIONS-Was ist passiert (Klartext)|Doku-Entscheidung}" +EXACTLY_ONE="${PR_EXACTLY_ONE-}" +AT_LEAST_ONE="${PR_AT_LEAST_ONE-}" +fails=0; warns=0 +ok() { printf 'OK %s\n' "$*"; } +warn() { printf 'WARN %s\n' "$*"; warns=$((warns+1)); } +fail() { printf 'FAIL %s\n' "$*"; fails=$((fails+1)); } +finish() { + echo "----" + if [ "$fails" -gt 0 ]; then echo "PR guard: $fails FAIL, $warns WARN"; [ "$WARN_ONLY" -eq 1 ] && exit 0; exit 1; fi + echo "PR guard: green ($warns WARN)"; exit 0 +} + +body="$(printf '%s\n' "${PR_BODY:-}" | sed 's/\r$//')" +if [ -z "$(printf '%s' "$body" | tr -d '[:space:]')" ]; then + fail "PR_BODY is empty – the PR has no description (template: .github/PULL_REQUEST_TEMPLATE.md)"; finish +fi +has_section() { printf '%s\n' "$body" | awk -v h="$1" 'index($0, "## " h) == 1 { f = 1 } END { exit !f }'; } +section() { # text of "## …" up to the next "## " heading, HTML comments removed + printf '%s\n' "$body" | awk -v h="$1" ' + /^## / { if (f) exit; if (index($0, "## " h) == 1) { f = 1; next } } + f' | sed -e '//d' -e '//d' +} +checked_top() { section "$1" | grep -cE '^- \[[xX]\]'; } +checked_top_text() { section "$1" | grep -E '^- \[[xX]\]' | head -1; } +checked_sub() { section "$1" | grep -cE '^ {2,}- \[[xX]\]'; } +filled() { # $1 = text: at least 20 visible characters and no obvious template placeholder + local t; t="$(printf '%s' "$1" | grep -vE '^[[:space:]]*$' || true)" + [ "$(printf '%s' "$t" | tr -d '[:space:]' | wc -c | tr -d ' ')" -ge 20 ] || return 1 + printf '%s' "$t" | grep -qE '|<\.\.\.>||$'; } + +# --- language: PR titles are English (SYSTEM.md "Sprache und Portfolio") ------------------------ +if [ -n "${PR_TITLE:-}" ] && printf '%s' "$PR_TITLE" | grep -qE '[äöüÄÖÜß]| (und|oder|für|mit|nicht|neue[rs]?|wird) '; then + warn "PR title looks German – technical artefacts, commits and PR titles are English: $PR_TITLE" +fi + +# --- required sections -------------------------------------------------------------------- +IFS='|' read -r -a req <<< "$REQUIRED" +for h in "${req[@]}"; do + [ -n "$h" ] || continue + has_section "$h" || fail "section \"## $h\" missing – PR template outdated? (templates/base/PULL_REQUEST_TEMPLATE.md)" +done + +# --- plain language ----------------------------------------------------------------------- +if has_section "Was ist passiert (Klartext)"; then + if filled "$(section 'Was ist passiert (Klartext)')"; then ok "plain-language section filled" + else fail "\"Was ist passiert (Klartext)\" is empty or still the template placeholder (SYSTEM.md §6)"; fi +fi + +# --- docs decision: exactly one ------------------------------------------------------------ +if has_section "Doku-Entscheidung"; then + n="$(checked_top 'Doku-Entscheidung')" + if [ "$n" -ne 1 ]; then fail "Doku-Entscheidung: exactly one decision required, found $n (SYSTEM.md §9)" + else + t="$(checked_top_text 'Doku-Entscheidung')"; s="$(checked_sub 'Doku-Entscheidung')" + if printf '%s' "$t" | grep -qiE 'betroffen und|docs affected'; then + [ "$s" -ge 1 ] && ok "docs decision: docs affected, $s kind(s) named" || fail "Doku-Entscheidung: \"docs affected\" but no kind of documentation checked below it" + else + [ "$s" -eq 0 ] && ok "docs decision: no long-lived docs affected" || fail "Doku-Entscheidung: \"no docs affected\" contradicts $s checked kind(s) below" + fi + fi +fi + +# --- plan gate + impact manifest ----------------------------------------------------------- +if has_section "Plan-Pflicht"; then + n="$(checked_top 'Plan-Pflicht')" + if [ "$n" -ne 1 ]; then fail "Plan-Pflicht: exactly one answer required, found $n (SYSTEM.md §4)" + elif checked_top_text 'Plan-Pflicht' | grep -qiE 'zutreffend|applies'; then + miss="" + for f in "Betroffene Module" "Schnittstellen / Datenänderungen" "Akzeptanzkriterien" "Testplan" "Risiken und Rollback"; do + field_filled 'Plan-Pflicht' "$f" || miss="$miss \"$f\"" + done + [ -z "$miss" ] && ok "plan gate: trigger applies, impact manifest filled" || fail "Plan-Pflicht: trigger applies but the Impact Manifest is incomplete –$miss (SYSTEM.md §4)" + else + ok "plan gate: no trigger" + fi +fi + +# --- draft: next smallest step ------------------------------------------------------------- +if [ "${PR_DRAFT:-false}" = "true" ]; then + if ! has_section "Arbeitsstand"; then fail "draft PR without \"## Arbeitsstand\" – the next session reads it first (SYSTEM.md §6)" + elif field_filled 'Arbeitsstand' 'Nächster kleinster Schritt'; then ok "draft: next smallest step named" + else fail "draft PR: \"Nächster kleinster Schritt\" in Arbeitsstand is empty (SYSTEM.md §6)"; fi +fi + +# --- ready for review: verify green -------------------------------------------------------- +if [ "${PR_DRAFT:-false}" != "true" ] && has_section "Nachweis"; then + v="$(section 'Nachweis' | grep -E '^- `verify`:' | head -1)" + if [ -z "$v" ]; then fail "Nachweis: line \"- \`verify\`: …\" missing (SYSTEM.md §11)" + elif printf '%s' "$v" | grep -qiE 'gr(ü|ue)n|green|pass'; then ok "proof: verify green" + else fail "ready for review without green verify: $v (SYSTEM.md §11)"; fi +fi + +# --- file size gate and before-creating (§7) ------------------------------------------------ +base="${BASE_REF:-}" +if [ -z "$base" ]; then + for c in origin/main origin/master main master; do git rev-parse --verify -q "$c" >/dev/null 2>&1 && { base="$c"; break; }; done +fi +exempt() { # generated code, lockfiles, fixtures, migrations, schemas, resources, docs, build output, config + case "$1" in + */generated/*|*.generated.*|*/__generated__/*|*.g.ts|*.pb.go|*_pb2.py|*.d.ts) return 0 ;; + package-lock.json|pnpm-lock.yaml|yarn.lock|Cargo.lock|poetry.lock|Pipfile.lock|composer.lock|Gemfile.lock|go.sum|*.lock) return 0 ;; + */fixtures/*|*/__fixtures__/*|*/testdata/*|*/__snapshots__/*|*.snap) return 0 ;; + */migrations/*|*.sql) return 0 ;; + *.schema.*|*/schema/*|openapi*|*.graphql|*.proto|*.prisma) return 0 ;; + */locales/*|*/i18n/*|*.json|*.csv|*.xml|*.yaml|*.yml|*.toml|*.ini) return 0 ;; + *.md|*.mdx|*.rst|*.txt|docs/*) return 0 ;; + dist/*|build/*|vendor/*|node_modules/*|*.min.*|public/*|*.config.*) return 0 ;; + esac + if [ -n "${SIZE_IGNORE:-}" ]; then + local g; IFS='|' read -r -a ig <<< "$SIZE_IGNORE" + for g in "${ig[@]}"; do [ -n "$g" ] || continue; case "$1" in $g) return 0 ;; esac; done + fi + return 1 +} +big500=""; big800=""; big1000=""; hint=""; newshared="" +if [ -n "$base" ] && git rev-parse --verify -q "$base" >/dev/null 2>&1 && git rev-parse --verify -q HEAD >/dev/null 2>&1; then + stage="$(sed -n 's/^[[:space:]]*stage:[[:space:]]*\([a-z]*\).*/\1/p' project-profile.yml 2>/dev/null | head -1)" + while IFS= read -r f; do + [ -f "$f" ] || continue + exempt "$f" && continue + n="$(wc -l < "$f" | tr -d ' ')" + case "$f" in *.test.*|*.spec.*|*/tests/*|*/test/*|*/__tests__/*|tests/*|test/*) # test files: hint only + [ "$n" -gt 300 ] && hint="$hint $f($n,test)"; continue ;; + esac + if [ "$n" -gt 1000 ]; then big1000="$big1000 $f($n)" + elif [ "$n" -gt 800 ]; then big800="$big800 $f($n)" + elif [ "$n" -gt 500 ]; then big500="$big500 $f($n)" + elif [ "$n" -gt 300 ]; then hint="$hint $f($n)"; fi + done < <(git diff --name-only "$base...HEAD" 2>/dev/null) + while IFS= read -r f; do + [ -n "$f" ] || continue + case "$f" in *.test.*|*.spec.*|*/tests/*|*/test/*|*/__tests__/*) continue ;; esac + exempt "$f" && continue + if printf '%s' "$f" | grep -qiE '(^|/)(shared|utils?|helpers?|lib|adapters?|services?|validators?)/|(util|helper|adapter|service|validator)s?\.[a-z]+$'; then newshared="$newshared $f"; fi + done < <(git diff --diff-filter=A --name-only "$base...HEAD" 2>/dev/null) + [ -n "$hint" ] && warn "files over 300 lines touched – check cohesion before adding logic (§7):$hint" + [ -n "$big500$big800$big1000" ] && ok "files over 500 lines in the diff:$big500$big800$big1000" +else + warn "size gate skipped: no base ref (set BASE_REF=origin/)" +fi +if has_section "Dateigrößen"; then + sec="$(section 'Dateigrößen')" + A="$(printf '%s\n' "$sec" | awk '/^Über 800/{exit} {print}')" + B="$(printf '%s\n' "$sec" | awk '/^Neue Shared/{f=1} f')" + over="$(printf '%s\n' "$sec" | grep -E '^Über 800' | head -1 | sed -E 's/^Über 800.*\(P1\/P2\):[[:space:]]*//')" + na="$(printf '%s\n' "$A" | grep -cE '^- \[[xX]\]')"; at="$(printf '%s\n' "$A" | grep -E '^- \[[xX]\]' | head -1)" + nb="$(printf '%s\n' "$B" | grep -cE '^- \[[xX]\]')"; bt="$(printf '%s\n' "$B" | grep -E '^- \[[xX]\]' | head -1)" + [ "$na" -eq 1 ] || fail "Dateigrößen: exactly one decision for files over 500 lines required, found $na (§7)" + [ "$nb" -eq 1 ] || fail "Dateigrößen: exactly one answer for new shared component / utility / adapter / service required, found $nb (§7)" + if [ -n "$big500$big800$big1000" ] && printf '%s' "$at" | grep -qiE '\] *keine'; then + fail "Dateigrößen: the diff touches files over 500 lines but \"keine\" is checked – decide: keep deliberately (reason), split in this PR, or follow-up issue (§7):$big500$big800$big1000" + fi + if [ -n "$big800$big1000" ]; then + if [ -z "$(printf '%s' "$over" | tr -d '[:space:]')" ] || printf '%s' "$over" | grep -qiE '^(nicht betroffen|<)'; then + fail "Dateigrößen: files over 800 lines touched –$big800$big1000 – a justified exception or an accompanying split is required (§7)" + elif [ -n "$big1000" ] && { [ "$stage" = "internal" ] || [ "$stage" = "production" ]; } && ! printf '%s' "$over" | grep -qE '#[0-9]+|ADR'; then + fail "Dateigrößen: files over 1000 lines in a P1/P2 project are an architecture finding – name the issue (#nr) or ADR:$big1000 (§7)" + else + ok "size gate: files over 800 lines justified: $over" + fi + elif [ -n "$big1000" ]; then + warn "files over 1000 lines touched (P0 – no finding required yet):$big1000" + fi + if [ -n "$newshared" ]; then + if printf '%s' "$bt" | grep -qiE '\] *nein'; then + fail "before creating: new shared/utility/adapter/service file(s) added –$newshared – but \"nein\" is checked; search for existing functionality and state what you searched and found (§7)" + else + v="$(printf '%s' "$bt" | sed -E 's/.*gesucht nach:[[:space:]]*//')" + if [ -z "$(printf '%s' "$v" | tr -d '[:space:]')" ] || printf '%s' "$v" | grep -q '^<'; then fail "before creating: \"ja\" needs what was searched and what was found (§7)" + else ok "before creating: search documented for$newshared"; fi + fi + fi +elif [ -n "$big500$big800$big1000$newshared" ]; then + fail "section \"## Dateigrößen\" missing although the diff touches large or new shared files – PR template outdated? (§7)" +fi + +# --- extra checkbox rules (control center) -------------------------------------------------- +IFS='|' read -r -a ex1 <<< "$EXACTLY_ONE" +for h in "${ex1[@]}"; do + [ -n "$h" ] && has_section "$h" || continue + n="$(checked_top "$h")"; [ "$n" -eq 1 ] && ok "$h: exactly one" || fail "$h: exactly one box must be checked, found $n" +done +IFS='|' read -r -a al1 <<< "$AT_LEAST_ONE" +for h in "${al1[@]}"; do + [ -n "$h" ] && has_section "$h" || continue + n="$(checked_top "$h")"; [ "$n" -ge 1 ] && ok "$h: $n checked" || fail "$h: at least one box must be checked" +done +finish diff --git a/scripts/quiet-run.sh b/scripts/quiet-run.sh new file mode 100755 index 0000000..4424d0e --- /dev/null +++ b/scripts/quiet-run.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +# ============================================================================= +# quiet-run – fluory-system output filter for test and verify runs (SYSTEM.md §11) +# Template: Entwicklungsplan/templates/base/scripts/quiet-run.sh +# +# Runs a command, trims SUCCESS output to a header and the summary, keeps FAILURE output +# (test names, error, diff, context, last lines), never changes the exit code, and records the +# result for the compaction snapshot (pre-compact.sh). The full log is always kept as a file. +# scripts/quiet-run.sh pnpm verify:changed +# FLUORY_FULL_OUTPUT=1 scripts/quiet-run.sh pnpm verify # once, when the cause is unclear +# The hook .claude/hooks/filter-test-output.sh routes unambiguous test commands through this +# script automatically. Knobs: FLUORY_QUIET_TAIL=15 FLUORY_QUIET_MAX=200 FLUORY_STATE_DIR +# Self-test: Entwicklungsplan/scripts/test-hooks.sh +# ============================================================================= +cmd="$*" +[ -n "$cmd" ] || { echo "usage: scripts/quiet-run.sh " >&2; exit 64; } +TAIL="${FLUORY_QUIET_TAIL:-15}"; MAX="${FLUORY_QUIET_MAX:-200}" + +STATE="${FLUORY_STATE_DIR:-$HOME/.claude/fluory-state}" +root="$(git rev-parse --show-toplevel 2>/dev/null || pwd)" +key="$(printf '%s' "$root" | sed 's#[^A-Za-z0-9._-]#_#g')" +dir="$STATE/$key" +mkdir -p "$dir" 2>/dev/null || dir="$(mktemp -d)" +find "$dir" -type f -name 'quiet-run-*.log' -mtime +2 -delete 2>/dev/null +log="$dir/quiet-run-$(date +%Y%m%d-%H%M%S)-$$.log" + +record() { # $1 = exit code – last result for the compaction snapshot + { + echo "$(date '+%Y-%m-%d %H:%M:%S') · exit $1 · $cmd" + [ -f "$log" ] && awk 'NF' "$log" | tail -n 3 + } > "$dir/last-verify.txt" 2>/dev/null +} + +if [ "${FLUORY_FULL_OUTPUT:-0}" = "1" ]; then + bash -c "$cmd" 2>&1 | tee "$log" + rc=${PIPESTATUS[0]} + record "$rc" + exit "$rc" +fi + +bash -c "$cmd" > "$log" 2>&1 +rc=$? +total="$(wc -l < "$log" | tr -d ' ')" +if [ "$rc" -eq 0 ]; then + echo "[quiet-run] OK · exit 0 · $cmd · $total lines (full log: $log)" + if [ "$total" -le $((TAIL * 2)) ]; then cat "$log"; else echo "…"; tail -n "$TAIL" "$log"; fi +else + echo "[quiet-run] FAILED · exit $rc · $cmd · $total lines (full log: $log)" + if [ "$total" -le "$MAX" ]; then + cat "$log" + else + echo "[quiet-run] relevant lines (pattern match with context, capped at $MAX):" + grep -n -E -B2 -A8 '(FAIL|✗|✘|×|Error|error:|Exception|Assertion|expected|Expected|received|Received|Traceback|panic:|failed|not ok|Cannot |Unhandled|TypeError|ReferenceError)' "$log" | head -n "$MAX" + echo "[quiet-run] last 30 lines:" + tail -n 30 "$log" + fi +fi +record "$rc" +exit "$rc"