-
Notifications
You must be signed in to change notification settings - Fork 2
Expand file tree
/
Copy pathconfig.example.yaml
More file actions
161 lines (150 loc) · 6.38 KB
/
Copy pathconfig.example.yaml
File metadata and controls
161 lines (150 loc) · 6.38 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
# LLMRed engagement configuration
# Secrets are NEVER stored here — reference environment variables with ${VAR_NAME}.
# Copy to config.yaml and edit per engagement.
engagement:
client_name: "Example AI Assistant"
auditor: "Authorized Security Team"
scope_reference: "SOW-2026-XXX" # link to signed scope / rules of engagement
language: "en" # fr | en
# Drives Phase 1 (Governance & Scoping) test prioritization. One of:
# finance | healthcare | government | ecommerce | technology | general
# Unrecognized values fall back to "general" - every technique still runs
# regardless of sector, this only changes order and report narrative.
sector: "general"
# Machine-enforced rules of engagement. Enable this for every real
# assessment after replacing the example scope and budgets below.
engagement_policy:
enabled: false
policy_id: "SOW-2026-XXX"
authorized_targets: [] # exact hostnames/IPs or CIDRs
authorized_ports: [] # empty means any port on an authorized target
valid_from: null # ISO-8601 timestamp, preferably with timezone
valid_until: null
max_run_seconds: 3600
max_requests: 1000
max_total_tokens: 2000000
allow_write_tests: false # separate explicit consent
allow_dos_tests: false # separate explicit consent
target:
provider: "openai" # openai | ollama | custom_chat_api
base_url: "${OPENAI_BASE_URL}"
api_key: "${OPENAI_API_KEY}"
model: "gpt-4o-mini"
timeout_seconds: 60
throttle_rps: null # e.g. 5 - caps requests/second from THIS client across
# every phase. Leave null only if you've coordinated
# rate limits with the client beforehand; otherwise
# set this before pointing at a production endpoint.
audit:
concurrency: 5
nist_questionnaire: true
# Optional. If null, cli.py auto-loads nist-answers.yaml beside this config
# when that file exists. The normal one-command workflow does not change.
nist_answers_file: null
output_dir: "./reports"
output_report_html: "audit_report.html"
output_report_pdf: "audit_report.pdf"
output_report_json: "audit_report.json"
export_json: false # machine-parseable output for CI/CD gating (e.g.
# fail a build on summary.has_critical == true)
audit_log: false # writes every chat() request/response to a JSONL
audit_log_filename: "audit_log.jsonl" # append-only local trail, separate from
# the findings report - secrets are redacted the
# same way they are in report evidence.
cleanup_ledger_filename: "cleanup_recovery.jsonl"
forensic_bundle: true
forensic_bundle_dir: "evidence"
encrypt_forensic_bundle: false
# When encryption is true, this environment variable must contain a
# URL-safe base64 encoding of exactly 32 random bytes. The key is never
# stored in config or in the evidence bundle.
forensic_encryption_key_env: "LLM_PENTEST_EVIDENCE_KEY"
phases:
recon: true
access: true
execution: true
impact: true
specialized: true
# Optional constrained plan/act/observe mode. Fixed-sequence execution remains
# the default. Agentic mode requires engagement_policy.enabled=true and an
# explicit allowlist; the planner cannot supply URLs, shell commands, payloads,
# or arbitrary tool arguments.
agentic:
enabled: false
allowed_techniques: [] # e.g. [model_fingerprint, indirect_injection]
allowed_risk_levels: [low, medium]
max_steps: 12
max_planner_errors: 2
max_consecutive_tool_errors: 2
max_repeat_per_tool: 1
stop_on_critical_finding: true
execution:
dos_test:
enabled: true
concurrency: 10
duration_seconds: 15
max_requests: 200
mode: "both" # output_amplification | input_flood | both - these are
# genuinely different load shapes; a target can be
# resilient to one and open to the other.
input_flood_word_count: 6000
# Non-deterministic model behavior is evaluated with repeated attack/control
# pairs. Empty or failed responses are inconclusive, never counted as safe.
evaluation:
enabled: true
trials_per_case: 3
min_valid_trials: 2
attack_success_threshold: 0.67
max_control_positive_rate: 0.20
min_attack_control_delta: 0.50
confidence_level: 0.95
require_human_review_for_borderline: true
# Target capability profile. null means unknown, not false. Explicit
# declarations override safe runtime inference and remain labeled "declared"
# in reports. Do not claim a capability is absent just because an adapter or
# test credential was not supplied to this run.
capabilities:
# Integration boundary, independent of features the underlying model may
# theoretically support.
surface_type: "unknown" # unknown | raw_model_api | llm_application |
# rag_application | agentic_application | local_model_lab
chat: true
client_supplied_history: true
server_side_state: null
system_prompt_control: null
rag: null
memory: null
tools: null
mcp: null
agent_actions: null
multimodal: null
streaming: null
external_content_ingestion: null
authentication: null
role_based_access: null
local_model_artifacts: null
notes: {}
# STRIDE uses its six standard categories. AI-specific manifestations are
# stored separately as AI/ML domains; they are not presented as an official
# extension to the STRIDE mnemonic.
threat_model:
enabled: true
assets: [] # e.g. ["customer PII", "system prompt"]
trust_boundaries: [] # e.g. ["browser -> API", "API -> model provider"]
sensitive_data_types: []
high_impact_actions: []
external_dependencies: []
notes: []
# Phase 0 - network-level discovery of AI infrastructure (MLflow, vector DBs,
# Jupyter, inference servers) rather than chat-level testing of the model.
# Disabled by default. targets MUST match the signed engagement scope exactly
# - this scans hosts beyond target.base_url above, which is a materially
# different authorization than "we gave you an API key to this one endpoint."
infrastructure_recon:
enabled: false
targets: [] # e.g. ["10.10.45.0/24"] or ["ml-internal.client.com"]
# ports: defaults to the full known AI-stack port list if omitted
timeout_seconds: 2.0
max_concurrent: 50
max_hosts: 256 # refuses to scan a range larger than this - raise
# explicitly if a bigger range is genuinely in scope