From 9d1af77a9415525e416c774ec197150ec4f068dd Mon Sep 17 00:00:00 2001
From: Miles <187469313+miles2542@users.noreply.github.com>
Date: Sat, 18 Apr 2026 02:17:05 +0700
Subject: [PATCH] docs: public README and other docs, my gosh
---
Justfile | 57 +++
README.md | 379 ++++++++++++++++++
docker-compose.yml | 39 ++
docs/MLOPS_ARCHITECTURE.md | 603 +++++++++++++++++++++++++++++
docs/ML_PIPELINE.md | 435 +++++++++++++++++++++
docs/OBSERVABILITY_AND_SECURITY.md | 508 ++++++++++++++++++++++++
docs/TEAM_CHEAT_SHEET.md | 360 +++++++++++++++++
docs/assets/gha_pipeline.png | Bin 0 -> 214063 bytes
docs/assets/grafana_demo.png | Bin 0 -> 442782 bytes
docs/assets/k8s_status.png | Bin 0 -> 332581 bytes
docs/assets/mlflow_ui.png | Bin 0 -> 230219 bytes
docs/assets/project_structure.png | Bin 0 -> 1028667 bytes
docs/assets/shap_summary.png | Bin 0 -> 78922 bytes
docs/assets/streamlit_ui.png | Bin 0 -> 158640 bytes
docs/assets/swagger_ui.png | Bin 0 -> 162705 bytes
scripts/install_k8s_tools.py | 124 ++++++
16 files changed, 2505 insertions(+)
create mode 100644 README.md
create mode 100644 docker-compose.yml
create mode 100644 docs/MLOPS_ARCHITECTURE.md
create mode 100644 docs/ML_PIPELINE.md
create mode 100644 docs/OBSERVABILITY_AND_SECURITY.md
create mode 100644 docs/TEAM_CHEAT_SHEET.md
create mode 100644 docs/assets/gha_pipeline.png
create mode 100644 docs/assets/grafana_demo.png
create mode 100644 docs/assets/k8s_status.png
create mode 100644 docs/assets/mlflow_ui.png
create mode 100644 docs/assets/project_structure.png
create mode 100644 docs/assets/shap_summary.png
create mode 100644 docs/assets/streamlit_ui.png
create mode 100644 docs/assets/swagger_ui.png
create mode 100644 scripts/install_k8s_tools.py
diff --git a/Justfile b/Justfile
index 54cfe98..8981c6c 100644
--- a/Justfile
+++ b/Justfile
@@ -90,6 +90,16 @@ deploy-all: \
setup-prod \
k8s-monitoring-setup \
k8s-apply-monitors
+ @echo ""
+ @echo " Full stack deployed. Allow 30-60s for pods to stabilize."
+ @echo ""
+ @echo " Open in browser:"
+ @echo " Streamlit UI -> http://localhost:30000"
+ @echo " API Docs -> http://localhost:30100/docs"
+ @echo " API Health -> http://localhost:30100/health"
+ @echo " Grafana -> http://localhost:30200 (admin / prom-operator)"
+ @echo " Prometheus -> http://localhost:30300"
+ @echo ""
mlflow-ui:
uv run mlflow ui --backend-store-uri sqlite:///mlruns/mlflow.db
@@ -99,3 +109,50 @@ k8s-update:
just build-ui
just k8s-load
-kubectl rollout restart deployment rossmann-api rossmann-ui
+
+# ── Local development servers (no Docker, no K8s) ──────────────────────────
+
+serve-api:
+ @echo ""
+ @echo " Starting FastAPI inference server ..."
+ @echo " Once running, open: http://localhost:8000/docs"
+ @echo ""
+ uv run uvicorn rossmann_ops.api.main:app --host 0.0.0.0 --port 8000 --reload
+
+serve-ui:
+ @echo ""
+ @echo " Starting Streamlit dashboard ..."
+ @echo " Once running, open: http://localhost:8501"
+ @echo ""
+ uv run streamlit run ui/app.py
+
+demo:
+ @echo ""
+ @echo " Running observability demo (3 phases: normal / schema errors / poisoning attack)."
+ @echo " Requires inference API running on port 30100 (K8s or Docker Compose)."
+ @echo " Watch live Grafana metrics at: http://localhost:30200 (K8s only)"
+ @echo ""
+ uv run python scripts/observability_demo.py
+
+# ── Docker Compose — published images, no K8s required ─────────────────────
+
+docker-up: check-docker
+ docker compose pull
+ docker compose up -d
+ @echo ""
+ @echo " Services started from DockerHub images. Open:"
+ @echo " Streamlit UI -> http://localhost:30000"
+ @echo " API Docs -> http://localhost:30100/docs"
+ @echo " API Health -> http://localhost:30100/health"
+ @echo ""
+ @echo " Note: Grafana/Prometheus not available in Docker Compose mode."
+ @echo " Run 'just deploy-all' for the full K8s stack with observability."
+ @echo ""
+
+docker-down:
+ docker compose down
+
+# ── Tool installation ───────────────────────────────────────────────────────
+
+install-k8s-tools:
+ uv run python scripts/install_k8s_tools.py
diff --git a/README.md b/README.md
new file mode 100644
index 0000000..9778b67
--- /dev/null
+++ b/README.md
@@ -0,0 +1,379 @@
+
+
+# Rossmann Store Sales Demand Forecasting (MLOps Pipeline)
+
+[](https://python.org)
+[](LICENSE)
+[](https://github.com/miles2542/rossmann-ops/actions/workflows/mlops_pipeline.yaml)
+[](https://github.com/miles2542/rossmann-ops/tree/main/tests)
+
+[](https://dvc.org)
+[](https://mlflow.org)
+[](https://docker.com)
+[](https://kubernetes.io)
+
+
+
+*Predicts daily store sales for 1,115 Rossmann stores using a Random Forest model trained on 2+ years of transactional data. Deployed as a multi-replica microservice on a local Kubernetes cluster, with end-to-end automation covering CI, CD, and CT.*
+
+
+
+---
+
+## System Highlights
+
+| Capability | Implementation |
+| :---------------------- | :-------------------------------------------------------------------------------------------------------------------------------- |
+| **Serving** | FastAPI + 2-replica K8s deployment with `RollingUpdate` zero-downtime strategy |
+| **Security** | Layered defense: Pandera schema validation → Pydantic bounds → CompetitionDistance guard |
+| **Telemetry** | Prometheus custom metrics (`sales_inference_total`, `inference_anomalies_blocked`) scraped at 1s intervals, visualized in Grafana |
+| **Explainability** | SHAP feature importance served via API, rendered live in the Streamlit dashboard |
+| **Reproducibility** | `uv.lock` + DVC-pinned artifacts; `just setup` installs everything deterministically |
+| **Continuous Training** | KS-Test drift detection triggers `repository_dispatch` → auto-retrain pipeline |
+
+---
+
+## Architecture
+
+### Diagram 1 — Deployment Topology
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b", "clusterBkg": "#111827", "clusterBorder": "#1f2937", "titleColor": "#f1f5f9"}}}%%
+flowchart TB
+ classDef user fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef extStore fill:#2e1065,stroke:#a78bfa,color:#ede9fe,stroke-width:2px
+ classDef port fill:#1c1917,stroke:#44403c,color:#a8a29e,stroke-width:1px
+ classDef svc fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px,font-weight:bold
+ classDef pod fill:#022c22,stroke:#16a34a,color:#86efac,stroke-width:1px
+ classDef mon fill:#431407,stroke:#f97316,color:#fdba74,stroke-width:2px
+ classDef svcmon fill:#1c1917,stroke:#fbbf24,color:#fde68a,stroke-width:1.5px
+
+ User(["User / API Client"]):::user
+
+ subgraph External [" External Services "]
+ direction LR
+ DagsHub[("DagsHub\n─────────────────\nDVC Remote Storage\nMLflow Experiment Tracking")]:::extStore
+ DockerHub[("DockerHub\n─────────────────\nImage Registry\nmiles25420/rossmann-api\nmiles25420/rossmann-ui")]:::extStore
+ end
+
+ subgraph Cluster [" Kubernetes Cluster — rossmann-cluster | KinD | 1 Control Plane + 2 Workers "]
+ direction TB
+
+ subgraph Ports [" Host → Cluster Port Mappings "]
+ direction LR
+ P30000>":30000 → Streamlit UI"]:::port
+ P30100>":30100 → Inference API"]:::port
+ P30200>":30200 → Grafana"]:::port
+ P30300>":30300 → Prometheus"]:::port
+ end
+
+ subgraph App [" Application Layer — default namespace "]
+ direction TB
+ UISvc["Service: ui-service\nNodePort 30000"]:::svc
+ APISvc["Service: api-service\nNodePort 30100"]:::svc
+
+ subgraph UIPods [" Streamlit UI Pods (×2 replicas · RollingUpdate · maxUnavailable=0) "]
+ direction LR
+ UI1[["Pod 1 · Streamlit :8501\nReadiness → GET /_stcore/health"]]:::pod
+ UI2[["Pod 2 · Streamlit :8501\nReadiness → GET /_stcore/health"]]:::pod
+ end
+
+ subgraph APIPods [" Inference API Pods (×2 replicas · RollingUpdate · maxUnavailable=0) "]
+ direction LR
+ API1[["Pod 1 · FastAPI :8000\nReadiness → GET /health\n/predict /health /store/{id}\n/health/shap /metrics /docs"]]:::pod
+ API2[["Pod 2 · FastAPI :8000\nReadiness → GET /health\n/predict /health /store/{id}\n/health/shap /metrics /docs"]]:::pod
+ end
+
+ UISvc --> UI1 & UI2
+ APISvc --> API1 & API2
+ end
+
+ subgraph Mon [" Monitoring Stack — monitoring namespace | kube-prometheus-stack via Helm "]
+ direction LR
+ SvcMon{{"ServiceMonitor\nlabel: app=rossmann-api\nscrapeInterval: 1s"}}:::svcmon
+ Prom[("Prometheus\n:9090")]:::mon
+ Graf["Grafana :3000\nDashboard-as-Code\nConfigMap auto-provisioned"]:::mon
+ end
+ end
+
+ User --> P30000 & P30100 & P30200 & P30300
+
+ P30000 --> UISvc
+ P30100 --> APISvc
+ P30200 --> Graf
+ P30300 --> Prom
+
+ UI1 & UI2 -->|"K8s internal DNS\nhttp://api-service:8000"| APISvc
+ SvcMon -->|"label-select: app=rossmann-api"| APISvc
+ Prom -->|"scrape /metrics via ServiceMonitor"| SvcMon
+ Graf -->|"PromQL queries"| Prom
+
+ API1 & API2 -->|"load artifacts on startup\nlocal path or MLflow URI"| DagsHub
+ DockerHub -->|"image pulled on pod scheduling"| API1 & API2 & UI1 & UI2
+```
+
+### Active Cluster State
+
+
+
+### Diagram 2 — Automated Pipeline (CI / CD / CT)
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b", "clusterBkg": "#111827", "clusterBorder": "#1f2937"}}}%%
+flowchart TD
+ classDef trigger fill:#1e1b4b,stroke:#818cf8,color:#e0e7ff,stroke-width:2px
+ classDef job fill:#0c2340,stroke:#3b82f6,color:#bfdbfe,stroke-width:2px,font-weight:bold
+ classDef ctNode fill:#2d0a47,stroke:#c084fc,color:#f3e8ff,stroke-width:2px
+ classDef artifact fill:#042f2e,stroke:#14b8a6,color:#99f6e4,stroke-width:2px
+
+ subgraph Triggers [" Pipeline Triggers "]
+ direction LR
+ T1{{"Push\nany branch"}}:::trigger
+ T2{{"Push\nmain or tag"}}:::trigger
+ T3{{"Push\nversion tag v*.*.*"}}:::trigger
+ T4{{"repository_dispatch\nevent: drift_detected"}}:::trigger
+ T5{{"workflow_dispatch\nmanual trigger"}}:::trigger
+ end
+
+ subgraph Jobs [" GitHub Actions Jobs — .github/workflows/mlops_pipeline.yaml "]
+ direction TB
+
+ CI["① CI — Lint & Test\n────────────────────\nuv sync --frozen\nnbstripout check lenient\nruff check lenient\npytest strict gate"]:::job
+
+ Train["② Train Model\n────────────────────\nDVC credentials via Secrets\ndvc pull -r dagshub\npython -m rossmann_ops.train_model\nMLflow logs → DagsHub\nupload-artifact: models/"]:::job
+
+ Simulate["③ Simulate (manual only)\n────────────────────\nMode: schema | attack | drift\nattack: continue-on-error=true\n exit 1 = detection confirmed"]:::job
+
+ Build["④ Build & Push Images\n────────────────────\ndownload-artifact: models/\ndownload: data/raw/store.csv\nDockerfile.api → miles25420/rossmann-api\nDockerfile.ui → miles25420/rossmann-ui\npush branch → :latest\npush tag → :vX.Y.Z + :latest"]:::job
+
+ Release["⑤ GitHub Release (tags only)\n────────────────────\nRewrite k8s manifests\n :latest → :vX.Y.Z\nBuild deployment-bundle.zip\n src/ k8s/ configs/ Justfile\n uv.lock pyproject.toml\n store_target_means.json\nAttach to GitHub Release"]:::job
+ end
+
+ subgraph CT [" Continuous Training Loop "]
+ direction LR
+ Drift(["simulate_production.py --mode drift\n────────────────────\nKS-Test on CompetitionDistance\np-value < 0.05 → drift detected"]):::ctNode
+ Webhook(["GitHub API\nPOST /repos/{owner}/{repo}/dispatches\n{ event_type: drift_detected }\nRequires: GITHUB_PAT (repo scope)"]):::ctNode
+ end
+
+ subgraph Artifacts [" Artifact Destinations "]
+ direction LR
+ DH[("DagsHub\nMLflow run logged\nExperiment tracked")]:::artifact
+ DHu[("DockerHub\nmiles25420/rossmann-api\nmiles25420/rossmann-ui")]:::artifact
+ GHR[("GitHub Release\ndeployment-bundle.zip\napi.yaml · ui.yaml")]:::artifact
+ end
+
+ T1 -->|"every push"| CI
+ T2 -->|"main / tag"| CI
+ T3 -->|"tag only"| CI
+ T4 -->|"drift event"| CI
+ T5 -->|"manual"| CI
+
+ CI -->|"tests pass — main / tag / dispatch"| Train
+ CI -->|"tests pass — workflow_dispatch"| Simulate
+
+ Train -->|"models/ uploaded to artifact store"| Build
+ Train --> DH
+
+ Build --> DHu
+ Build -->|"tag push only"| Release
+ Release --> GHR
+
+ Drift -->|"p < 0.05"| Webhook
+ Webhook -->|"fires repository_dispatch"| T4
+```
+
+---
+
+## Getting Started
+
+```bash
+git clone https://github.com/miles2542/rossmann-ops.git
+cd rossmann-ops
+```
+
+---
+
+## Prerequisites
+
+This project uses [`just`](https://github.com/casey/just) as its unified task runner and [`uv`](https://docs.astral.sh/uv/) for Python dependency management. Install **only these two tools** manually — `just setup` / `just deploy-all` handle everything else automatically (`.venv` creation, dependency install, DVC pull, model training, Docker builds, K8s deploy).
+
+### Step 0 — Install `just`
+
+`just` is a lightweight, cross-platform command runner. Install it system-wide, then **restart your shell or reopen VS Code** before continuing.
+
+| Platform | Command |
+| :------------------- | :----------------------------------------------------------------------------------------------------- |
+| **Windows** (winget) | `winget install --id Casey.Just` |
+| **macOS** (Homebrew) | `brew install just` |
+| **Linux** | `curl --proto '=https' --tlsv1.2 -sSf https://just.systems/install.sh \| bash -s -- --to ~/.local/bin` |
+
+### Step 1 — Install `uv`
+
+`uv` is a fast Python package and environment manager. Install it **into your system Python** (not inside a project venv) so it is available as a global command:
+
+```bash
+pip install uv
+
+# Or via the official installer:
+# Windows (PowerShell): powershell -c "irm https://astral.sh/uv/install.ps1 | iex"
+# macOS / Linux: curl -LsSf https://astral.sh/uv/install.sh | sh
+```
+
+> Once `uv` is installed globally, `just setup` will automatically create the project `.venv` and install all Python dependencies. No manual `pip install` or `venv` commands needed.
+
+### Step 2 — Install Docker Desktop and Helm
+
+Required for Docker Compose (Option B) and K8s deployment (Option C). Not needed for local-only development (Option A).
+
+| Tool | Installation |
+| :----------------- | :------------------------------------------------------------------------------------ |
+| **Docker Desktop** | [docker.com/products/docker-desktop](https://www.docker.com/products/docker-desktop/) |
+| **Helm** | [helm.sh/docs/intro/install](https://helm.sh/docs/intro/install/) |
+
+### Step 3 — Install `kind` and `kubectl` (K8s path only)
+
+```bash
+just install-k8s-tools
+```
+
+Detects your OS and uses `winget` (Windows), `brew` (macOS), or downloads official binaries (Linux). Restart your shell after completion.
+
+### Required Credentials
+
+Copy `.env.example` → `.env` and populate:
+
+```bash
+# DagsHub — remote DVC storage + MLflow tracking
+# Repository: https://dagshub.com/miles2542/rossmann-ops
+DAGSHUB_USERNAME=miles2542
+DAGSHUB_PAT=
+
+# MLflow remote tracking server (points to DagsHub)
+MLFLOW_TRACKING_URI=https://dagshub.com/miles2542/rossmann-ops.mlflow
+MLFLOW_TRACKING_USERNAME=miles2542
+MLFLOW_TRACKING_PASSWORD=
+```
+
+> [!NOTE]
+> **GitHub Actions CI/CD Secrets:** In repository Settings → Secrets and Variables → Actions, configure:
+> `DAGSHUB_USERNAME`, `DAGSHUB_PAT`, `DOCKERHUB_USERNAME` (`miles25420`), `DOCKERHUB_TOKEN`, and optionally `GITHUB_PAT` (needed for the CT retraining webhook from a local machine).
+
+> **DockerHub credentials** are only needed as GitHub Actions secrets for the CD stage — not for local deployments.
+
+---
+
+## Quickstart
+
+> [!NOTE]
+> Recommend **option C** for full testing (e.g. for professor)
+
+
+Option A — Local Development (no Docker, no K8s — fastest path)
+
+```bash
+# 1. Install deps and pull DVC-tracked data artifacts
+just setup
+
+# 2. Train the production model (logs to MLflow, ~5-10 min)
+just train-prod
+
+# 3. Start the inference API (terminal 1) and dashboard (terminal 2)
+just serve-api # FastAPI -> http://localhost:8000/docs
+just serve-ui # Streamlit -> http://localhost:8501
+
+# 4. (Optional) Inspect experiment runs in the MLflow UI
+just mlflow-ui # -> http://localhost:5000
+```
+
+
+
+
+Option B — Docker Compose (no K8s required)
+
+Pulls the published images from DockerHub and runs them with a single command. Requires **Docker Desktop only** — no `kind`, `kubectl`, or `helm` needed.
+
+```bash
+just docker-up # pull latest images and start (detached)
+just docker-down # stop and remove containers
+```
+
+| Service | URL |
+| :----------- | :------------------------------ |
+| Streamlit UI | `http://localhost:30000` |
+| API Docs | `http://localhost:30100/docs` |
+| API Health | `http://localhost:30100/health` |
+
+> [!NOTE]
+> This path uses the latest published image from DockerHub (`miles25420/rossmann-api:latest`). Prometheus/Grafana monitoring is **not** included. For the full observability stack, use Option C.
+
+
+
+
+Option C — Full K8s Production Deployment ✦ Recommended for graders
+
+Trains the model locally, builds Docker images, deploys to a local KinD cluster, and provisions Prometheus + Grafana monitoring. Requires Steps 0–3 of Prerequisites.
+
+```bash
+just deploy-all
+```
+
+> Estimated runtime: 10–20 minutes on a modern machine (mainly due to Docker build and loading image into KinD).
+
+Once running, all services are available at `localhost`:
+
+| Service | URL | Credentials |
+| :----------- | :------------------------------ | :------------------------ |
+| Streamlit UI | `http://localhost:30000` | — |
+| API Docs | `http://localhost:30100/docs` | — |
+| API Health | `http://localhost:30100/health` | — |
+| Grafana | `http://localhost:30200` | `admin` / `prom-operator` |
+| Prometheus | `http://localhost:30300` | — |
+
+
+
+### Useful Maintenance Commands
+
+```bash
+just k8s-status # Show all pods + services across namespaces
+just k8s-down # Delete the KinD cluster
+just k8s-update # Rebuild images + rolling restart deployments
+just lint # Run Ruff linter, can add --fix to auto-fix
+just format # Auto-format code (Ruff)
+just test # Run full pytest suite with coverage
+```
+
+---
+
+## Running Observability Demos
+
+With the K8s stack or Docker Compose running:
+
+```bash
+just demo
+```
+
+Sends 3 sequential traffic phases — normal predictions, malformed schema requests (422s), and data-poisoned payloads (`CompetitionDistance > 100,000m`) — to create visible spikes in the Grafana dashboard at `http://localhost:30200`.
+
+
+
+---
+
+## Repository Structure
+
+
+
+---
+
+## Extra Documentation
+
+| Document | Description |
+| :------------------------------------------------------------- | :------------------------------------------------------------------- |
+| [ML Pipeline](docs/ML_PIPELINE.md) | Feature engineering, modeling strategy, metrics, SHAP explainability |
+| [MLOps Architecture](docs/MLOPS_ARCHITECTURE.md) | K8s topology, deployment strategy, CI/CD/CT flow |
+| [Observability & Security](docs/OBSERVABILITY_AND_SECURITY.md) | Defensive layers, telemetry, drift detection, Grafana dashboard |
+
+---
+
+## License
+
+Apache License 2.0. See [LICENSE](LICENSE) for full terms.
diff --git a/docker-compose.yml b/docker-compose.yml
new file mode 100644
index 0000000..8242a46
--- /dev/null
+++ b/docker-compose.yml
@@ -0,0 +1,39 @@
+# docker-compose.yml
+#
+# Runs the published DockerHub images without requiring a Kubernetes cluster.
+# Requires only Docker Desktop — no kind, kubectl, or helm needed.
+#
+# Usage:
+# just docker-up # Pull latest images and start services
+# just docker-down # Stop and remove containers
+#
+# Services will be available at:
+# Streamlit UI -> http://localhost:30000
+# API Docs -> http://localhost:30100/docs
+# API Health -> http://localhost:30100/health
+#
+# Note: Prometheus/Grafana monitoring is NOT included in this mode.
+# Use 'just deploy-all' for the full K8s stack with observability.
+
+services:
+ api:
+ image: miles25420/rossmann-api:latest
+ ports:
+ - "30100:8000"
+ healthcheck:
+ test: ["CMD-SHELL", "curl -f http://localhost:8000/health || exit 1"]
+ interval: 15s
+ timeout: 5s
+ retries: 5
+ start_period: 15s
+
+ ui:
+ image: miles25420/rossmann-ui:latest
+ ports:
+ - "30000:8501"
+ environment:
+ # Docker Compose default network allows services to resolve each other by name.
+ - API_URL=http://api:8000
+ depends_on:
+ api:
+ condition: service_healthy
diff --git a/docs/MLOPS_ARCHITECTURE.md b/docs/MLOPS_ARCHITECTURE.md
new file mode 100644
index 0000000..3f166b2
--- /dev/null
+++ b/docs/MLOPS_ARCHITECTURE.md
@@ -0,0 +1,603 @@
+# MLOps Architecture
+
+**Infrastructure topology, deployment strategy, and automated pipeline design.**
+
+---
+
+## Table of Contents
+
+1. [Kubernetes Cluster Topology](#1-kubernetes-cluster-topology)
+2. [Application Deployments](#2-application-deployments)
+3. [Deployment Strategy — RollingUpdate](#3-deployment-strategy--rollingupdate)
+4. [Health Checks & Readiness Gates](#4-health-checks--readiness-gates)
+5. [Monitoring Stack](#5-monitoring-stack)
+6. [CI/CD/CT Pipeline](#6-cicdct-pipeline)
+7. [Docker Image Strategy](#7-docker-image-strategy)
+8. [Artifact Bridging Between Jobs](#8-artifact-bridging-between-jobs)
+
+---
+
+## 1. Kubernetes Cluster Topology
+
+**Implementation**: `k8s/kind-config.yaml`
+
+The local cluster is provisioned using **kind (Kubernetes IN Docker)** v0.31.0 — a single-host Kubernetes environment that runs nodes as Docker containers.
+
+### Cluster Definition
+
+```yaml
+# k8s/kind-config.yaml
+apiVersion: kind.x-k8s.io/v1alpha4
+kind: Cluster
+name: rossmann-cluster
+nodes:
+ - role: control-plane
+ extraPortMappings:
+ - containerPort: 30000 # Streamlit UI
+ hostPort: 30000
+ - containerPort: 30100 # FastAPI Inference API
+ hostPort: 30100
+ - containerPort: 30200 # Grafana Dashboard
+ hostPort: 30200
+ - containerPort: 30300 # Prometheus UI
+ hostPort: 30300
+ - role: worker
+ - role: worker
+```
+
+**NodePort mappings** forward `localhost:{hostPort}` directly to the cluster's transport layer — no additional ingress controller is required for local operation.
+
+### Cluster Topology
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b", "clusterBkg": "#111827", "clusterBorder": "#1f2937"}}}%%
+flowchart TB
+ classDef host fill:#1c1917,stroke:#78716c,color:#d6d3d1,stroke-width:1.5px
+ classDef cp fill:#1e3a5f,stroke:#3b82f6,color:#bfdbfe,stroke-width:2px
+ classDef svc fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px,font-weight:bold
+ classDef apiPod fill:#022c22,stroke:#16a34a,color:#86efac,stroke-width:1px
+ classDef uiPod fill:#162032,stroke:#60a5fa,color:#bfdbfe,stroke-width:1px
+ classDef monPod fill:#431407,stroke:#f97316,color:#fdba74,stroke-width:1.5px
+
+ subgraph Host [" Host Machine (localhost) "]
+ direction LR
+ H1>":30000 Streamlit UI"]:::host
+ H2>":30100 Inference API"]:::host
+ H3>":30200 Grafana"]:::host
+ H4>":30300 Prometheus"]:::host
+ end
+
+ subgraph KinD [" rossmann-cluster — KinD (Kubernetes IN Docker) | 3 nodes "]
+ direction TB
+
+ subgraph CP [" Control Plane Node "]
+ direction LR
+ KAPI["kube-apiserver"]:::cp
+ KSched["kube-scheduler"]:::cp
+ KCM["kube-controller-manager"]:::cp
+ etcd[("etcd")]:::cp
+ end
+
+ APISvc["Service: api-service\nType: NodePort → 30100"]:::svc
+ UISvc["Service: ui-service\nType: NodePort → 30000"]:::svc
+
+ subgraph W1 [" Worker Node 1 "]
+ direction TB
+ API_R1[["rossmann-api Pod 1\nFastAPI :8000\ncpu: 100m–500m mem: 256–512Mi"]]:::apiPod
+ UI_R1[["rossmann-ui Pod 1\nStreamlit :8501\ncpu: 100m–500m mem: 256–512Mi"]]:::uiPod
+ MON["Prometheus Pod\nGrafana Pod"]:::monPod
+ end
+
+ subgraph W2 [" Worker Node 2 "]
+ direction TB
+ API_R2[["rossmann-api Pod 2\nFastAPI :8000\ncpu: 100m–500m mem: 256–512Mi"]]:::apiPod
+ UI_R2[["rossmann-ui Pod 2\nStreamlit :8501\ncpu: 100m–500m mem: 256–512Mi"]]:::uiPod
+ end
+
+ APISvc --> API_R1 & API_R2
+ UISvc --> UI_R1 & UI_R2
+ MON -->|"ServiceMonitor\nscrape /metrics"| APISvc
+ end
+
+ H1 --> UISvc
+ H2 --> APISvc
+ H3 & H4 --> MON
+```
+
+### Active Cluster State
+
+
+
+---
+
+## 2. Application Deployments
+
+Both application deployments share the same structural template: `apps/v1` Deployment + `v1` NodePort Service, defined in a single YAML file each.
+
+### FastAPI Inference API (`k8s/api.yaml`)
+
+```yaml
+spec:
+ replicas: 2
+ strategy:
+ type: RollingUpdate
+ rollingUpdate:
+ maxSurge: 1 # 1 extra pod created during update
+ maxUnavailable: 0 # no pod taken down before replacement is ready
+
+ containers:
+ - name: rossmann-api
+ image: rossmann-api:latest # rewritten to DockerHub tag on release
+ ports:
+ - containerPort: 8000
+ resources:
+ requests:
+ cpu: "100m"
+ memory: "256Mi"
+ limits:
+ cpu: "500m"
+ memory: "512Mi"
+```
+
+**Service**:
+- Type: `NodePort`
+- `port: 8000` → `targetPort: 8000` → `nodePort: 30100`
+
+### Streamlit UI (`k8s/ui.yaml`)
+
+```yaml
+spec:
+ replicas: 2
+ strategy:
+ type: RollingUpdate
+ rollingUpdate:
+ maxSurge: 1
+ maxUnavailable: 0
+
+ containers:
+ - name: rossmann-ui
+ image: rossmann-ui:latest
+ ports:
+ - containerPort: 8501
+ env:
+ - name: API_URL
+ value: "http://api-service:8000" # K8s internal DNS — stable across pod restarts
+ resources:
+ requests:
+ cpu: "100m"
+ memory: "256Mi"
+ limits:
+ cpu: "500m"
+ memory: "512Mi"
+```
+
+**Service**:
+- Type: `NodePort`
+- `port: 8501` → `targetPort: 8501` → `nodePort: 30000`
+
+**UI-to-API communication**: Uses Kubernetes internal DNS `http://api-service:8000` rather than `localhost` or `NodePort`. This resolves correctly within the cluster regardless of which node the UI pod is scheduled on, and avoids the extra network hop through the host machine.
+
+---
+
+## 3. Deployment Strategy — RollingUpdate
+
+Both deployments explicitly define `strategy: type: RollingUpdate` (`k8s/api.yaml`, line 11; `k8s/ui.yaml`, line 11).
+
+### Why RollingUpdate?
+
+| Strategy | Behavior | Downtime | Use Case |
+| :--- | :--- | :--- | :--- |
+| **RollingUpdate** | Incrementally replaces old pods with new ones | **Zero** | Stateless services where any replica can serve any request |
+| `Recreate` | Terminates all old pods, then starts new ones | **Full** | Stateful services requiring exclusive resource access (e.g., database migrations) |
+
+This system's inference API is **fully stateless** — each pod loads the same model artifact independently. Any replica can handle any request without coordination. `RollingUpdate` is therefore the correct and safe strategy.
+
+### Rolling Update Sequence
+
+With `replicas: 2`, `maxSurge: 1`, `maxUnavailable: 0`:
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "signalColor": "#94a3b8", "actorBkg": "#1e293b", "actorBorder": "#475569", "actorTextColor": "#f1f5f9", "labelBoxBkgColor": "#1e293b", "labelBoxBorderColor": "#334155", "labelTextColor": "#94a3b8", "noteBkgColor": "#1e3a5f", "noteBorderColor": "#3b82f6", "noteTextColor": "#bfdbfe", "activationBkgColor": "#052e16", "activationBorderColor": "#22c55e"}}}%%
+sequenceDiagram
+ participant K8s as Kubernetes Controller
+ participant V1A as API v1 — Pod A
+ participant V1B as API v1 — Pod B
+ participant V2A as API v2 — Pod A (new)
+ participant V2B as API v2 — Pod B (new)
+ participant Lb as Live Traffic
+
+ Note over V1A,V1B: Initial state — 2 pods serving
+ Lb->>V1A: requests
+ Lb->>V1B: requests
+
+ Note over K8s: Rollout triggered (image update detected)
+ K8s->>+V2A: Create pod (maxSurge=1 → 3 pods total)
+ V2A-->>K8s: Starting...
+ K8s->>V2A: Readiness probe: GET /health
+ V2A-->>-K8s: HTTP 200 — model loaded, all artifacts ready
+
+ Note over K8s: v2 Pod A is ready. Drain v1 Pod A.
+ K8s--xV1A: Graceful termination
+ Lb->>V1B: requests (V1A removed from endpoint pool)
+ Lb->>V2A: requests (V2A added to endpoint pool)
+ Note over V1B,V2A: Still 2 serving pods — zero downtime
+
+ K8s->>+V2B: Create pod
+ V2B-->>-K8s: Readiness probe passes
+ K8s--xV1B: Graceful termination
+
+ Note over V2A,V2B: Final state — 2 v2 pods serving. Rollout complete.
+ Lb->>V2A: requests
+ Lb->>V2B: requests
+```
+
+At no point during this sequence does the number of serving pods drop below 2 (`maxUnavailable: 0`). Live traffic continues without interruption.
+
+---
+
+## 4. Health Checks & Readiness Gates
+
+Two distinct probe types are configured per deployment, targeting different API endpoints.
+
+### FastAPI Probes
+
+| Probe | Endpoint | Delay | Period | Behaviour |
+| :--- | :--- | :--- | :--- | :--- |
+| **Readiness** | `GET /health` | 15s | 10s | Returns `200` only if model + store data + target means are all loaded. K8s routes traffic to this pod only when ready. |
+| **Liveness** | `GET /health/live` | 20s | 20s | Returns `200` always (process is alive). K8s restarts pod if this fails. |
+
+**Readiness probe logic** (`api/main.py`, lines 142–160):
+```python
+@app.get("/health")
+def readiness_check(response: Response):
+ is_ready = MODEL is not None and STORE_DF is not None and STORE_MEANS is not None
+ if not is_ready:
+ response.status_code = 503
+ return {
+ "status": "healthy" if is_ready else "degraded",
+ "model_loaded": MODEL is not None,
+ "store_data_loaded": STORE_DF is not None,
+ "store_means_loaded": STORE_MEANS is not None,
+ "model_version": MODEL_VERSION,
+ "environment": os.getenv("ENV", "development"),
+ }
+```
+
+This is an **"honest" readiness check** — it verifies that the actual inference prerequisites are loaded, not just that the HTTP server is accepting connections. A pod that is alive but missing its model artifact will receive `503` and be excluded from the Service's endpoint pool until artifacts load successfully.
+
+**Liveness probe logic** (`api/main.py`, lines 136–139):
+```python
+@app.get("/health/live")
+def liveness_check():
+ return {"status": "alive"}
+```
+
+The liveness probe is deliberately minimal — only the process crash case should trigger a pod restart. A model loading failure is a readiness concern, not a liveness concern.
+
+### Streamlit Probes
+
+Both readiness and liveness probes target Streamlit's built-in health endpoint:
+```
+GET /_stcore/health (port 8501)
+initialDelaySeconds: 10 (readiness) / 20 (liveness)
+periodSeconds: 10 (readiness) / 20 (liveness)
+```
+
+---
+
+## 5. Monitoring Stack
+
+**Installation**: `kube-prometheus-stack` Helm chart (Prometheus Community).
+
+**Recipe** (`Justfile`, lines 72–83):
+```bash
+helm upgrade --install prom -n monitoring --create-namespace \
+ prometheus-community/kube-prometheus-stack \
+ --set grafana.service.type=NodePort \
+ --set grafana.service.nodePort=30200 \
+ --set grafana.adminPassword=prom-operator \
+ --set prometheus.prometheusSpec.serviceMonitorSelectorNilUsesHelmValues=false \
+ --set prometheus.prometheusSpec.scrapeInterval=1s \
+ --set prometheus.service.type=NodePort \
+ --set prometheus.service.nodePort=30300
+```
+
+Key settings:
+- `scrapeInterval=1s` — 1-second resolution for near-realtime responsiveness during live demos.
+- `serviceMonitorSelectorNilUsesHelmValues=false` — allows Prometheus to discover `ServiceMonitor` resources in all namespaces, not just those labelled by the Helm release.
+
+### ServiceMonitor (`k8s/servicemonitor.yaml`)
+
+```yaml
+apiVersion: monitoring.coreos.com/v1
+kind: ServiceMonitor
+metadata:
+ name: rossmann-api-monitor
+ namespace: monitoring
+spec:
+ selector:
+ matchLabels:
+ app: rossmann-api
+ endpoints:
+ - port: http
+ path: /metrics
+```
+
+Prometheus automatically discovers and scrapes the `/metrics` endpoint exposed by `prometheus-fastapi-instrumentator` on all pods matching the `rossmann-api` label.
+
+### Grafana Dashboard (`k8s/grafana-dashboard.yaml`)
+
+Deployed as a Kubernetes `ConfigMap` with the label `grafana_dashboard: "1"`, which the Grafana sidecar picks up and provisions automatically (Dashboard-as-Code pattern).
+
+Dashboard panels:
+
+| Panel | Type | PromQL | Threshold |
+| :--- | :--- | :--- | :--- |
+| Global RPS (1m) | Stat | `sum(irate(http_requests_total[1m]))` | > 100 req/s → red |
+| Error Rate % (5m) | Gauge | `sum(rate(http_requests_total{status=~"4..\|5.."}[5m])) / sum(rate(http_requests_total[5m])) * 100` | > 5% yellow, > 20% red |
+| Total Predictions | Stat | `sum(sales_inference_total)` | — |
+| Anomalies Blocked | Stat | `sum(inference_anomalies_blocked_total)` | ≥ 1 → orange |
+| p95 / p50 Latency | Time series | `histogram_quantile(0.95, ...)` | > 0.5s → red |
+| HTTP Status Dist. | Donut chart | `sum by (status)(rate(http_requests_total[5m]))` | 4xx orange, 5xx red |
+
+Uses `irate` (instantaneous rate) rather than `rate` for the RPS panel, giving 1-scrape-interval sensitivity — critical for making burst spikes visible in a short demo window (`configs/observability.yaml`: `time_window: "3m"`).
+
+---
+
+## 6. CI/CD/CT Pipeline
+
+**Implementation**: `.github/workflows/mlops_pipeline.yaml` (279 lines, 5 jobs)
+
+### Trigger Matrix
+
+| `workflow_dispatch` | CI → **Simulate** |
+
+### Automated Pipeline Workflow (GHA)
+
+
+
+### Job 1: CI — Lint & Test
+
+```yaml
+steps:
+ - uses: astral-sh/setup-uv@v5 # official uv installer
+ - run: uv sync --frozen # exact lockfile restore
+ - name: Check notebook outputs # lenient — annotation only
+ continue-on-error: true
+ run: uv run nbstripout --check notebooks/*.ipynb
+ - name: Lint — ruff # lenient — annotation only
+ continue-on-error: true
+ run: uv run ruff check .
+ - name: Run tests # STRICT — blocks downstream
+ run: uv run pytest tests/ -v --cov=src --cov-report=term-missing
+```
+
+**Strict vs lenient gates**: Lint and notebook output checks are `continue-on-error: true` — they annotate the PR but do not block training or deployment. Tests are strict: a test failure prevents the `train` job from starting.
+
+### Job 2: Train
+
+Runs on: push to `main`, version tag push, or `repository_dispatch`.
+
+```yaml
+env:
+ MLFLOW_TRACKING_URI: https://dagshub.com/${{ secrets.DAGSHUB_USERNAME }}/rossmann-ops.mlflow
+ MLFLOW_TRACKING_USERNAME: ${{ secrets.DAGSHUB_USERNAME }}
+ MLFLOW_TRACKING_PASSWORD: ${{ secrets.DAGSHUB_PAT }}
+steps:
+ - name: Configure DVC credentials
+ run: |
+ uv run dvc remote modify dagshub --local auth basic
+ uv run dvc remote modify dagshub --local user ${{ secrets.DAGSHUB_USERNAME }}
+ uv run dvc remote modify dagshub --local password ${{ secrets.DAGSHUB_PAT }}
+ - run: uv run dvc pull -r dagshub # pulls raw data from DagsHub
+ - run: uv run python -m rossmann_ops.train_model
+```
+
+### Experiment Tracking Workspace (MLflow)
+
+
+
+DVC credentials are set via `--local` flag at runtime (ephemeral, not committed to `.dvc/config`). This is the correct CI pattern: credentials live in GitHub Secrets only, never in configuration files.
+
+### Job 3: Simulate (Manual Only)
+
+Triggered exclusively via `workflow_dispatch`. Accepts a `mode` parameter (`schema`, `attack`, `drift`). The `attack` mode is marked `continue-on-error: true` because it is **designed to exit with code 1** when poisoning is detected — a non-zero exit code would otherwise fail the job.
+
+### Job 4: Build & Push Docker Images
+
+```yaml
+needs: train
+steps:
+ - uses: actions/download-artifact@v4 # pulls models/ from Job 2
+ with:
+ name: production-model
+ path: models/
+ - name: Resolve image tags
+ run: |
+ if [[ "$GITHUB_REF" == refs/tags/* ]]; then
+ VER="${GITHUB_REF_NAME}"
+ echo "tags_api=user/rossmann-api:${VER},user/rossmann-api:latest" >> $GITHUB_OUTPUT
+ else
+ echo "tags_api=user/rossmann-api:latest" >> $GITHUB_OUTPUT
+ fi
+ - uses: docker/build-push-action@v5
+ with:
+ cache-from: type=gha
+ cache-to: type=gha,mode=max
+```
+
+**Tagging strategy**: On version tag pushes (`v*.*.*`), images receive **both** a semantic version tag and `latest`. Branch pushes get only `latest`. This enables immutable version pinning for graders while keeping `latest` convenient for development.
+
+GitHub Actions layer cache (`type=gha`) is used for Docker build caching — dramatically speeds up consecutive builds when `uv.lock` and source code haven't changed.
+
+### Job 5: GitHub Release (Tags Only)
+
+```yaml
+needs: build-and-push
+if: startsWith(github.ref, 'refs/tags/')
+steps:
+ - name: Rewrite manifests for DockerHub
+ run: |
+ sed -i "s|image: rossmann-api:latest|image: user/rossmann-api:${{ github.ref_name }}|g" k8s/api.yaml
+ sed -i "s|image: rossmann-ui:latest|image: user/rossmann-ui:${{ github.ref_name }}|g" k8s/ui.yaml
+ - name: Create Deployment Bundle (ZIP)
+ run: |
+ zip -r deployment-bundle-${{ github.ref_name }}.zip \
+ src/ scripts/ k8s/ configs/ Justfile pyproject.toml uv.lock \
+ models/store_target_means.json
+ - uses: softprops/action-gh-release@v2
+ with:
+ files: |
+ deployment-bundle-${{ github.ref_name }}.zip
+ k8s/api.yaml
+ k8s/ui.yaml
+```
+
+The release provides **two deployment paths**:
+1. **Direct container deploy** — download `api.yaml` + `ui.yaml` (already rewritten with DockerHub image tags), apply immediately to any K8s cluster. No local build or source required.
+2. **Full system replication** — extract `deployment-bundle.zip`, run `just setup`, complete local training and deployment.
+
+### Continuous Training Flow
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart LR
+ classDef script fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef logic fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+ classDef decision fill:#1c1917,stroke:#78716c,color:#d6d3d1,stroke-width:1.5px
+ classDef webhook fill:#2d0a47,stroke:#c084fc,color:#f3e8ff,stroke-width:2px
+ classDef pipeline fill:#0c2340,stroke:#3b82f6,color:#bfdbfe,stroke-width:2px,font-weight:bold
+ classDef output fill:#042f2e,stroke:#14b8a6,color:#99f6e4,stroke-width:2px
+
+ Prod(["Production Traffic\nor Periodic Job"]):::script
+ Script["simulate_production.py\n--mode drift"]:::script
+ KS["KS-Test\nscipy.stats.ks_2samp\n───────────────\nBaseline vs Shifted\nCompetitionDistance"]:::logic
+ Dec{"p-value\n< 0.05?"}:::decision
+ Log["Log: No Drift Detected\nNo action taken"]:::logic
+ Webhook["GitHub API\nPOST /dispatches\n{ event_type: drift_detected }\nRequires: GITHUB_PAT"]:::webhook
+ GHA["GitHub Actions\nrepository_dispatch received"]:::pipeline
+ Retrain["Full Pipeline Executes\n① CI → ② Train\n④ Build & Push new images\n⑤ GitHub Release (if tag)"]:::pipeline
+ NewModel[("Updated Model\nDockerHub :vX.Y.Z\nMLflow run logged")]:::output
+
+ Prod --> Script --> KS --> Dec
+ Dec -->|"No"| Log
+ Dec -->|"Yes"| Webhook
+ Webhook -->|"GITHUB_PAT (repo scope)"| GHA
+ GHA --> Retrain --> NewModel
+```
+
+**Required secret**: `GITHUB_PAT` with `repo` scope.
+
+---
+
+## 7. Docker Image Strategy
+
+Both images use the same pattern: `python:3.12-slim` base + `uv` binary injection from the official `ghcr.io/astral-sh/uv:latest` image.
+
+### `Dockerfile.api`
+
+```dockerfile
+FROM python:3.12-slim
+
+COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /bin/
+
+WORKDIR /app
+COPY pyproject.toml uv.lock ./
+RUN uv sync --frozen --no-dev --no-install-project # prod deps only, from exact lockfile
+
+COPY src/ ./src/
+COPY configs/ ./configs/
+COPY models/ ./models/ # includes serialized RF + store_target_means.json
+COPY data/raw/store.csv ./data/raw/ # store metadata for CompetitionDistance enrichment
+
+ENV PYTHONPATH=/app/src
+ENV MODEL_VERSION=v1.0.0
+EXPOSE 8000
+CMD ["uv", "run", "uvicorn", "rossmann_ops.api.main:app", "--host", "0.0.0.0", "--port", "8000"]
+```
+
+### `Dockerfile.ui`
+
+```dockerfile
+FROM python:3.12-slim
+
+COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /bin/
+
+WORKDIR /app
+COPY pyproject.toml uv.lock ./
+RUN uv sync --frozen --no-dev --no-install-project
+
+COPY ui/ ./ui/
+COPY src/ ./src/ # features.py (if UI ever needs local feature logic)
+COPY configs/ ./configs/
+
+ENV PYTHONPATH=/app/src
+ENV API_URL=http://localhost:8000 # overridden in K8s to api-service:8000
+EXPOSE 8501
+CMD ["uv", "run", "streamlit", "run", "ui/app.py", "--server.port", "8501", "--server.address", "0.0.0.0"]
+```
+
+**Key design choices**:
+- `--frozen`: Respects `uv.lock` exactly — deterministic builds across environments.
+- `--no-dev`: Excludes `pytest`, `ruff`, `optuna`, etc. from production images.
+- `--no-install-project`: Installs dependencies only; the project's own source code is COPY'd in the next step — avoids double-installing.
+- `uv run` in CMD: Uses the project's venv managed by uv, avoiding PATH configuration issues.
+
+---
+
+## 8. Artifact Bridging Between Jobs
+
+A key CI design challenge: GitHub Actions jobs run on **separate runners** with no shared filesystem. The `train` job produces model artifacts; the `build-and-push` job needs them to build the Docker image.
+
+**Solution — GitHub Actions Artifact upload / download**:
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b", "clusterBkg": "#111827", "clusterBorder": "#1f2937"}}}%%
+flowchart TD
+ classDef step fill:#0c2340,stroke:#3b82f6,color:#bfdbfe,stroke-width:2px
+ classDef artifact fill:#042f2e,stroke:#14b8a6,color:#99f6e4,stroke-width:2px
+ classDef store fill:#1e1b4b,stroke:#818cf8,color:#e0e7ff,stroke-width:2px,font-weight:bold
+ classDef release fill:#2d0a47,stroke:#c084fc,color:#f3e8ff,stroke-width:2px
+
+ subgraph Job2 [" Job ②: Train Model (runner A) "]
+ direction TB
+ T1["dvc pull -r dagshub\n→ data/raw/train.csv\n→ data/raw/store.csv"]:::step
+ T2["python -m rossmann_ops.train_model\n→ models/production_model/\n→ models/store_target_means.json\n→ models/shap_summary.png"]:::step
+ T3["upload-artifact: production-model\n path: models/ | retention: 7 days"]:::artifact
+ T4["upload-artifact: store-data\n path: data/raw/store.csv"]:::artifact
+ T1 --> T2 --> T3
+ T2 --> T4
+ end
+
+ GHStore[("GitHub Artifact Store\n(ephemeral · cross-runner bridge)")]:::store
+
+ subgraph Job4 [" Job ④: Build & Push Images (runner B) "]
+ direction TB
+ B1["download-artifact: production-model\n→ models/"]:::step
+ B2["download-artifact: store-data\n→ data/raw/store.csv"]:::step
+ B3["docker build -f Dockerfile.api .\n COPY models/ → /app/models/\n COPY store.csv → /app/data/raw/"]:::step
+ B4["docker build -f Dockerfile.ui ."]:::step
+ B5["docker push rossmann-api:tag\ndocker push rossmann-ui:tag"]:::step
+ B1 & B2 --> B3
+ B2 --> B4
+ B3 & B4 --> B5
+ end
+
+ subgraph Job5 [" Job ⑤: GitHub Release (tags only) "]
+ direction TB
+ R1["download-artifact: production-model"]:::step
+ R2["Rewrite k8s manifests\n rossmann-api:latest → :vX.Y.Z\n rossmann-ui:latest → :vX.Y.Z"]:::step
+ R3["zip deployment-bundle-vX.Y.Z.zip\n src/ k8s/ configs/ Justfile\n uv.lock pyproject.toml\n store_target_means.json"]:::step
+ R4["softprops/action-gh-release\nAttach: bundle.zip + api.yaml + ui.yaml"]:::release
+ R1 --> R2 --> R3 --> R4
+ end
+
+ T3 --> GHStore
+ T4 --> GHStore
+ GHStore -->|"cross-runner download"| B1 & B2
+ GHStore -->|"cross-runner download"| R1
+```
+
+This bridges the two jobs without requiring DVC push to DagsHub during CI, while ensuring the **freshly trained model** is the one baked into the published image.
+
+The same `download-artifact` pattern is used in the `release` job to generate the deployment bundle ZIP.
diff --git a/docs/ML_PIPELINE.md b/docs/ML_PIPELINE.md
new file mode 100644
index 0000000..344dd48
--- /dev/null
+++ b/docs/ML_PIPELINE.md
@@ -0,0 +1,435 @@
+# ML Pipeline
+
+**End-to-end walkthrough of the Rossmann forecasting model: from raw data to production predictions.**
+
+---
+
+## Table of Contents
+
+1. [Dataset Overview](#1-dataset-overview)
+2. [Data Validation](#2-data-validation)
+3. [Feature Engineering](#3-feature-engineering)
+4. [Modeling Strategy](#4-modeling-strategy)
+5. [Evaluation Metrics](#5-evaluation-metrics)
+6. [Experiment Tracking](#6-experiment-tracking)
+7. [Model Explainability (SHAP)](#7-model-explainability-shap)
+8. [Artifact Management](#8-artifact-management)
+
+---
+
+## 1. Dataset Overview
+
+**Source**: Kaggle Rossmann Store Sales (managed via DVC, stored on DagsHub).
+
+| File | Rows (approx.) | Key Columns |
+| :--- | :--- | :--- |
+| `data/raw/train.csv` | 1,017,209 | `Store`, `Date`, `Sales`, `Customers`, `Open`, `Promo`, `StateHoliday`, `SchoolHoliday` |
+| `data/raw/store.csv` | 1,115 | `Store`, `StoreType`, `Assortment`, `CompetitionDistance`, `Promo2`, `PromoInterval` |
+
+**Pre-training filtering** (in `train_model.py`, lines 55–66):
+- Only rows where `Open == 1` AND `Sales > 0` are retained.
+- Zero-sales days (store closures) are excluded because RMSPE is undefined when the true value is zero.
+
+---
+
+## 2. Data Validation
+
+**Implementation**: `src/rossmann_ops/data_validation.py` — Pandera `DataFrameModel`.
+
+Validation runs **before any feature engineering** in `train_model.py` (line 52: `df = validate_data(df)`). This is the first line of defense against data poisoning entering the training pipeline.
+
+### Schema Contract
+
+| Column | Expected Type | Constraint | Nullable |
+| :--- | :--- | :--- | :--- |
+| `Store` | `int` | ≥ 1 | No |
+| `DayOfWeek` | `int` | 1–7 | No |
+| `Sales` | `int` | ≥ 0 | No |
+| `Customers` | `int` | ≥ 0 | Yes |
+| `Open` | `int` | `∈ {0, 1}` | No |
+| `Promo` | `int` | `∈ {0, 1}` | No |
+| `StateHoliday` | `str` | — | Yes |
+| `SchoolHoliday` | `int` | `∈ {0, 1}` | Yes |
+| `StoreType` | `str` | — | Yes |
+| `Assortment` | `str` | — | Yes |
+
+**Config**: `strict=False` (extra columns from the `store.csv` merge are allowed), `coerce=True` (type coercion before validation).
+
+**On failure**: raises `pa.errors.SchemaError` — halts training immediately, preventing corrupted data from producing a silently wrong model.
+
+---
+
+## 3. Feature Engineering
+
+**Implementation**: `src/rossmann_ops/features.py`
+
+Two functions with a strict separation of concerns:
+
+| Function | Role |
+| :--- | :--- |
+| `build_features(df, train_comp_median, expected_ohe_cols)` | Stateless transforms applied identically at train and inference time |
+| `apply_target_encoding(df, store_means, global_mean)` | Stateful: reads from a pre-computed artifact |
+
+### 3.1 Stateless Transforms (`build_features`)
+
+**Date decomposition** (`features.py`, lines 44–51):
+```python
+df["Year"] = df["Date"].dt.year
+df["Month"] = df["Date"].dt.month
+df["WeekOfYear"] = df["Date"].dt.isocalendar().week.astype(int)
+df["DayOfWeek"] = df["Date"].dt.dayofweek + 1 # 1=Monday, 7=Sunday
+```
+DayOfWeek is standardized to the 1–7 business convention (not Python's 0–6 default).
+
+**Categorical One-Hot Encoding** (`features.py`, lines 53–61):
+`StoreType`, `Assortment`, and `StateHoliday` are passed to `pd.get_dummies`. The resulting columns follow the pattern `StoreType_a`, `Assortment_c`, `StateHoliday_0`, etc.
+
+**Competition Distance — Log transform with median imputation** (`features.py`, lines 63–66):
+```python
+dist = pd.to_numeric(df["CompetitionDistance"], errors="coerce")
+df["LogCompDist"] = np.log1p(dist.fillna(train_comp_median))
+```
+- Missing `CompetitionDistance` (rural stores with no recorded competitor) is filled with the **training-set median**, not the global mean. This prevents train-serve skew.
+- `log1p` compresses the extreme right-skew of distance values (ranges from 20m to 75,860m in the training data).
+- `train_comp_median` is computed once in `train_model.py` (line 77) and must be reused identically at inference. The API currently derives it from the enriched request row as a pragmatic fallback since `STORE_MEANS` already captures the store-level distance signal (see `api/main.py`, line 238).
+
+**OHE Column Alignment — Inference guard** (`features.py`, lines 68–84):
+```python
+if expected_ohe_cols is not None:
+ for col in expected_ohe_cols:
+ if col not in df.columns:
+ df[col] = 0 # fill missing OHE columns with 0 (not that category)
+ # Drop any unexpected OHE columns
+ extra = [c for c in df.columns if any(c.startswith(p) for p in
+ ("StoreType_", "Assortment_", "StateHoliday_")) and c not in expected_ohe_cols]
+ df.drop(columns=extra, inplace=False)
+```
+The **OHE column contract** is defined in `configs/params.yaml` (lines 14–25):
+```yaml
+features:
+ ohe_expected_columns:
+ - StoreType_a
+ - StoreType_b
+ - StoreType_c
+ - StoreType_d
+ - Assortment_a
+ - Assortment_b
+ - Assortment_c
+ - StateHoliday_0
+ - StateHoliday_a
+ - StateHoliday_b
+ - StateHoliday_c
+```
+This contract ensures that a single-store inference payload that only contains one `StoreType` value does not produce a 1-column OHE matrix — instead, all 11 OHE columns are always present, matching the training schema exactly.
+
+### 3.2 Target Encoding — Store Mean Sales (`apply_target_encoding`)
+
+**Implementation**: `features.py`, lines 90–118; artifact: `models/store_target_means.json`
+
+**Training-time computation** (leak-free, `train_model.py`, lines 90–100):
+```python
+df_cv["Store_TargetMean"] = df_cv.groupby("Store")["Sales"].transform(
+ lambda x: x.shift().expanding().mean() # expanding mean of past values only
+)
+global_mean = float(df_cv["Sales"].mean())
+df_cv["Store_TargetMean"].fillna(global_mean) # handles first row per store
+
+# Final per-store average — written to the inference artifact
+final_store_means = df_cv.groupby("Store")["Sales"].mean().to_dict()
+```
+Uses an **expanding mean with a one-row shift** to prevent data leakage: when computing the target mean for a given `(store, date)` row, only all *prior* rows for that store are used — never the current value.
+
+**Inference-time lookup** (`apply_target_encoding`, lines 112–113):
+```python
+df["Store_TargetMean"] = df["Store"].map(store_means).fillna(global_mean)
+```
+Unseen store IDs (new stores not in the training set) fall back to the global mean — preventing `NaN` from propagating to the model.
+
+**Artifact**: `models/store_target_means.json` — JSON with two keys:
+```json
+{
+ "store_means": {"1": 5263.72, "2": 4821.13, ...},
+ "global_mean": 5773.82
+}
+```
+JSON keys are strings (JSON limitation); the API converts them back to `int` on load (`api/main.py`, line 100).
+
+### 3.3 Final Feature Set
+
+The model receives the following 18 features, in this exact order:
+
+| Feature | Type | Source |
+| :--- | :--- | :--- |
+| `DayOfWeek` | int (1–7) | Date extraction |
+| `Promo` | int (0/1) | Raw |
+| `Year` | int | Date extraction |
+| `Month` | int (1–12) | Date extraction |
+| `WeekOfYear` | int (1–53) | Date extraction |
+| `LogCompDist` | float | Log-transformed CompetitionDistance |
+| `Store_TargetMean` | float | `store_target_means.json` lookup |
+| `StoreType_a` … `StoreType_d` | int (0/1) | OHE |
+| `Assortment_a` … `Assortment_c` | int (0/1) | OHE |
+| `StateHoliday_0` … `StateHoliday_c` | int (0/1) | OHE |
+
+---
+
+## 4. Modeling Strategy
+
+### 4.1 Data Split
+
+**Implementation**: `train_model.py`, lines 57–73
+
+Strictly **chronological** — no random shuffling to avoid temporal leakage.
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart LR
+ classDef phase fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef hold fill:#2d0a47,stroke:#c084fc,color:#f3e8ff,stroke-width:2px
+ classDef sim fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+ classDef marker fill:#1c1917,stroke:#78716c,color:#d6d3d1,stroke-width:1.5px
+
+ Start("Full Chronological Dataset"):::marker
+
+ subgraph Timeline [" Time Axis ──────────────────────────────────────────────────────────────────► "]
+ direction LR
+ CV["② CV Set\n(Train + Validation)\n────────────────────\nAll history up to\nholdout_cutoff"]:::phase
+ HO["③ Holdout Set\n────────────────────\nLast 28 days of\nCV-eligible data"]:::hold
+ Sim["④ Simulation Set\n────────────────────\nLast 42 days of\nentire dataset"]:::sim
+ end
+
+ Start --> CV --> HO --> Sim
+
+ C1{{"holdout_cutoff\n(28d from sim_cutoff)"}}:::marker
+ C2{{"sim_cutoff\n(42d from max_date)"}}:::marker
+
+ CV -.- C1
+ HO -.- C2
+```
+
+| Split | Duration | Purpose |
+| :--- | :--- | :--- |
+| **CV set** | All data before holdout period | Model training |
+| **Holdout** | Last 28 days of CV-eligible data | Unbiased evaluation |
+| **Simulation set** | Last 42 days of entire dataset | Never seen during training; used by `simulate_production.py` for drift detection |
+
+Both the holdout and simulation cutoffs are computed from `configs/params.yaml`:
+```yaml
+data_split:
+ simulation_days: 42
+ holdout_days: 28
+```
+
+### 4.2 Target Transformation
+
+The target `Sales` is modeled in **log-space** (`train_model.py`, line 117):
+```python
+y_cv_log = np.log1p(df_cv["Sales"])
+```
+
+**Why**: Raw sales values have a heavy right tail (range: 1 – 41,551). Log-transformation compresses this, making residuals more homoskedastic and reducing the impact of large-value outliers on the loss function.
+
+**At inference**: predictions are restored to the original sales scale via:
+```python
+prediction = float(np.expm1(prediction_log[0])) # api/main.py, line 271
+```
+The holdout evaluation (`train_model.py`, lines 165–173) also applies `np.expm1` before computing metrics, ensuring all reported metrics are in **actual EUR sales units** — not log-space.
+
+### 4.3 Production Model — Random Forest
+
+**Config** (`configs/params.yaml`, lines 32–39):
+```yaml
+model:
+ type: "RandomForest"
+ production:
+ n_estimators: 50
+ max_depth: 10
+ min_samples_split: 6
+ random_state: 105
+```
+
+**Training call** (`train_model.py`, lines 152–162):
+```python
+model = RandomForestRegressor(
+ n_estimators=50,
+ max_depth=10,
+ min_samples_split=6,
+ random_state=105,
+ n_jobs=-1, # uses all available CPU cores
+)
+model.fit(X_cv.astype(np.float64), y_cv_log)
+```
+
+The explicit `astype(np.float64)` cast ensures consistent MLflow schema logging and robustness to integer feature inputs at inference time.
+
+
+Why Random Forest over XGBoost?
+
+XGBoost was evaluated during hyperparameter search (see `notebooks/03_optuna.ipynb`). Random Forest was selected for the production model because:
+- Comparable RMSPE on the holdout set in this feature regime.
+- Native parallelism via `n_jobs=-1` without needing GPU or specialized builds.
+- SHAP TreeExplainer is directly compatible with `sklearn` ensemble models.
+- Simpler dependency footprint in production containers.
+
+
+
+---
+
+## 5. Evaluation Metrics
+
+**Primary metric**: **RMSPE** (Root Mean Squared Percentage Error) — the official Kaggle competition metric.
+
+$$\text{RMSPE} = \sqrt{\frac{1}{n} \sum_{i=1}^{n} \left(\frac{y_i - \hat{y}_i}{y_i}\right)^2}$$
+
+**Implementation** (`train_model.py`, lines 27–32):
+```python
+def rmspe(y_true: np.ndarray, y_pred: np.ndarray) -> float:
+ y_pred = np.maximum(y_pred, 1.0) # floor predictions to avoid /0
+ mask = y_true != 0 # exclude zero-sales rows
+ return float(np.sqrt(np.mean(((y_true[mask] - y_pred[mask]) / y_true[mask]) ** 2)))
+```
+
+**Key design choices**:
+- Zero-sales rows are **excluded** from RMSPE computation (`mask = y_true != 0`). Including them would produce `inf` errors.
+- Predictions are floored at 1.0 (`np.maximum(y_pred, 1.0)`) to prevent division-by-zero on near-zero predictions.
+
+All four metrics are logged to MLflow automatically:
+
+| Metric | Key | Description |
+| :--- | :--- | :--- |
+| RMSPE | `holdout_rmspe` | Primary competition metric (lower is better) |
+| RMSE | `holdout_rmse` | EUR, sensitive to large errors |
+| MAE | `holdout_mae` | EUR, robust to outliers |
+| R² | `holdout_r2` | Variance explained (1.0 = perfect) |
+
+---
+
+## 6. Experiment Tracking
+
+**Implementation**: `train_model.py`, lines 128–221 — MLflow integrated throughout.
+
+**Tracking URI**: Defaults to DagsHub remote (`MLFLOW_TRACKING_URI` env var) with a local SQLite fallback:
+```python
+mlflow.set_tracking_uri(
+ os.getenv("MLFLOW_TRACKING_URI", f"sqlite:///{project_root}/mlruns/mlflow.db")
+)
+```
+
+**Logged artifacts**:
+- **Parameters**: `model_type`, `n_estimators`, `max_depth`, `min_samples_split`, `random_state`, `holdout_days`, `simulation_days`, `n_cv_rows`, `train_comp_median`
+- **Metrics**: `holdout_rmspe`, `holdout_rmse`, `holdout_mae`, `holdout_r2`
+- **Artifacts**: `shap_summary.png` (global feature importance plot)
+- **Model**: Logged via `mlflow.sklearn.log_model(model, name="production_model")` and also saved locally to `models/production_model/` (MLflow serialization format)
+
+`mlflow.sklearn.autolog(log_models=False)` is also active, capturing sklearn-specific metadata automatically. `log_models=False` prevents autolog from saving a second copy of the model — the explicit `log_model` call is used instead for fine-grained control.
+
+**View locally**:
+```bash
+just mlflow-ui
+# → http://localhost:5000
+```
+
+---
+
+## 7. Model Explainability (SHAP)
+
+**Implementation**: `train_model.py`, lines 191–209
+
+```python
+explainer = shap.TreeExplainer(model)
+sample_X = X_cv.sample(min(500, len(X_cv)), random_state=105)
+shap_values = explainer.shap_values(sample_X)
+
+plt.figure(figsize=(10, 6))
+shap.summary_plot(shap_values, sample_X, show=False)
+plt.savefig("models/shap_summary.png")
+```
+
+SHAP values are computed on **a 500-row sample** from the CV set for speed. `TreeExplainer` uses the exact fast-path algorithm specific to tree ensembles (not the generic permutation approximation).
+
+The resulting `shap_summary.png` is:
+1. **Saved to** `models/shap_summary.png` (DVC-tracked artifact)
+2. **Logged to** MLflow as a run artifact
+3. **Served live** by the API at `GET /health/shap`
+4. **Rendered in** the Streamlit dashboard via the "Model Diagnostics & Explainability" expander
+
+**In the dashboard**:
+```python
+# ui/app.py, lines 188–198
+if st.button("Fetch SHAP Summary"):
+ r = requests.get(f"{api_url}/health/shap")
+ if r.status_code == 200:
+ st.image(r.content, use_container_width=True)
+```
+
+### SHAP Round-Trip Flow
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart TD
+ classDef ui fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef api fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+ classDef disk fill:#2e1065,stroke:#a78bfa,color:#ede9fe,stroke-width:2px
+
+ UI["Streamlit Dashboard\n'Fetch SHAP Summary'"]:::ui
+ Endpoint["FastAPI Endpoint\nGET /health/shap"]:::api
+ Artifact[("Artifact Storage\nmodels/shap_summary.png")]:::disk
+ Render["Display SHAP Plot\n(Current Model's Global Importance)"]:::ui
+
+ UI -->|"HTTP GET"| Endpoint
+ Endpoint -->|"Read file-like object"| Artifact
+ Artifact -->|"PNG bytes"| Endpoint
+ Endpoint -->|"Application/PNG"| UI
+ UI --> Render
+```
+
+This round-trip ensures the SHAP plot always corresponds to the **currently loaded model version**, not a stale cached image.
+
+---
+
+## 8. Artifact Management
+
+All production artifacts are DVC-tracked and committed as `.dvc` pointer files:
+
+| Artifact | Path | DVC File |
+| :--- | :--- | :--- |
+| Serialized model | `models/production_model/` | `models/production_model.dvc` |
+| SHAP plot | `models/shap_summary.png` | `models/shap_summary.png.dvc` |
+| Target means | `models/store_target_means.json` | `models/store_target_means.json.dvc` |
+
+**Remote storage**: DagsHub (configured in `.dvc/config`).
+
+**Pull artifacts locally**:
+```bash
+dvc pull # or: just pull
+```
+
+### Artifact Pipeline Bridge
+
+After training, model artifacts are uploaded as a **GitHub Actions Artifact** (`production-model`) and then **downloaded** in the subsequent `build-and-push` job, bridging the two jobs without requiring DVC push in CI.
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart LR
+ classDef job fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef artifact fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+ classDef target fill:#0c2d48,stroke:#2e9cca,color:#d1e8e2,stroke-width:2px
+
+ J1["① Train Model Job\n(Runner A)"]:::job
+ AM["Artifact Store\n(GHA Temporary)"]:::artifact
+ J2["② Build Image Job\n(Runner B)"]:::job
+ Docker["Production Image\n(rossmann-api:latest)"]:::target
+
+ J1 -->|"actions/upload-artifact@v4"| AM
+ AM -->|"actions/download-artifact@v4"| J2
+ J2 -->|"COPY models/ /app/models/"| Docker
+```
+
+```yaml
+- name: Upload model artifact
+ uses: actions/upload-artifact@v4
+ with:
+ name: production-model
+ path: models/
+ retention-days: 7
+```
diff --git a/docs/OBSERVABILITY_AND_SECURITY.md b/docs/OBSERVABILITY_AND_SECURITY.md
new file mode 100644
index 0000000..1cbe4c9
--- /dev/null
+++ b/docs/OBSERVABILITY_AND_SECURITY.md
@@ -0,0 +1,508 @@
+# Observability & Security
+
+**Telemetry stack, defensive architecture, and drift detection mechanisms.**
+
+---
+
+## Table of Contents
+
+- [Observability \& Security](#observability--security)
+ - [Table of Contents](#table-of-contents)
+ - [1. Security Architecture — Layered Defense](#1-security-architecture--layered-defense)
+ - [Request Lifecycle \& Defense Layers](#request-lifecycle--defense-layers)
+ - [2. Layer 1: Schema Validation (Pandera)](#2-layer-1-schema-validation-pandera)
+ - [Schema Definition](#schema-definition)
+ - [3. Layer 2: API Input Bounds (Pydantic)](#3-layer-2-api-input-bounds-pydantic)
+ - [4. Layer 3: Functional Guard (Inference-time)](#4-layer-3-functional-guard-inference-time)
+ - [5. Layer 4: Z-Score Batch Poisoning Detection](#5-layer-4-z-score-batch-poisoning-detection)
+ - [6. Telemetry — Prometheus Metrics](#6-telemetry--prometheus-metrics)
+ - [Auto-instrumented Metrics](#auto-instrumented-metrics)
+ - [Custom Metrics](#custom-metrics)
+ - [7. Grafana Dashboard](#7-grafana-dashboard)
+ - [Panel Specifications](#panel-specifications)
+ - [8. Drift Detection — KS-Test](#8-drift-detection--ks-test)
+ - [Mechanics](#mechanics)
+ - [KS Statistic Interpretation](#ks-statistic-interpretation)
+ - [9. Continuous Training Trigger](#9-continuous-training-trigger)
+ - [Demo Phase Sequences](#demo-phase-sequences)
+ - [Phase 1: Normal Traffic (750 requests)](#phase-1-normal-traffic-750-requests)
+ - [Phase 2: Schema Errors (350 requests)](#phase-2-schema-errors-350-requests)
+ - [Phase 3: Poisoning Attack (350 requests)](#phase-3-poisoning-attack-350-requests)
+ - [Manual Variants (`simulate_production.py --mode`)](#manual-variants-simulate_productionpy---mode)
+
+---
+
+## 1. Security Architecture — Layered Defense
+
+The system implements **four independent defensive layers** against malicious or malformed inputs, operating at different stages of the request lifecycle:
+
+### Request Lifecycle & Defense Layers
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart TD
+ classDef req fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef sync fill:#0c2340,stroke:#3b82f6,color:#bfdbfe,stroke-width:2px
+ classDef async fill:#1c1917,stroke:#78716c,color:#d6d3d1,stroke-width:1.5px
+ classDef fail fill:#431407,stroke:#f97316,color:#fdba74,stroke-width:2px
+ classDef pass fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+
+ Start(["Incoming Request"]):::req
+
+ subgraph Sync [" Synchronous Real-Time Defense (FastAPI) "]
+ direction TB
+ L2["Layer 2: Pydantic Schema\nType checks, bounds, presence"]:::sync
+ L3["Layer 3: Functional Guard\nDomain-specific sanity checks"]:::sync
+ end
+
+ subgraph Async [" Asynchronous / Pipeline Defense "]
+ direction TB
+ L1["Layer 1: Pandera (Training)\nDataFrame contract enforcement"]:::async
+ L4["Layer 4: Z-Score (Batch)\nDetection of poisoned datasets"]:::async
+ end
+
+ Model[["Model Inference & Response"]]:::pass
+
+ Start --> L2
+ L2 -->|"Passed"| L3
+ L3 -->|"Passed"| Model
+
+ L2 -->|"Failed"| F2["HTTP 422\nUnprocessable Entity"]:::fail
+ L3 -->|"Blocked"| F3["HTTP 422\ninference_anomalies_blocked++"]:::fail
+
+ L1 -->|"SchemaError"| F1["Training Halted"]:::fail
+ L4 -->|"Z > 3.0"| F4["Batch Aborted\nsys.exit(1)"]:::fail
+```
+
+**Synchronous defenses** (Layers 2 & 3) block individual requests in real-time and generate observable signals in Prometheus.
+**Asynchronous/pipeline defenses** (Layers 1 & 4) protect batch data integrity before it can influence the model.
+
+---
+
+## 2. Layer 1: Schema Validation (Pandera)
+
+**Implementation**: `src/rossmann_ops/data_validation.py`
+**Called at**: `train_model.py`, line 52 — before any feature engineering.
+
+### Schema Definition
+
+```python
+class RossmannSchema(pa.DataFrameModel):
+ Store: Series[int] = pa.Field(ge=1)
+ DayOfWeek: Series[int] = pa.Field(ge=1, le=7)
+ Sales: Series[int] = pa.Field(ge=0) # no negative sales
+ Customers: Optional[Series[int]]= pa.Field(ge=0) # no negative customers
+ Open: Series[int] = pa.Field(isin=[0, 1])
+ Promo: Series[int] = pa.Field(isin=[0, 1])
+ StateHoliday: Optional[Series[str]]= pa.Field(nullable=True)
+ SchoolHoliday: Optional[Series[int]]= pa.Field(isin=[0, 1], nullable=True)
+ StoreType: Optional[Series[str]]= pa.Field(nullable=True)
+ Assortment: Optional[Series[str]]= pa.Field(nullable=True)
+
+ class Config:
+ strict = False # extra columns (from store.csv merge) are allowed
+ coerce = True # type coercion before validation
+```
+
+**Protections**:
+- `Sales >= 0`: Prevents injected negative sales values from distorting the model.
+- `DayOfWeek in [1,7]`: Rejects impossible day encodings.
+- `Open in {0, 1}`: Rejects multi-class open-status injections.
+- `Store >= 1`: Rejects ID=0 or negative IDs that could poison target encoding.
+
+**On failure**: `pa.errors.SchemaError` is raised (`data_validation.py`, line 56), caught by the training script, and logged. Training halts. A poisoned dataset cannot produce a silently wrong model.
+
+---
+
+## 3. Layer 2: API Input Bounds (Pydantic)
+
+**Implementation**: `src/rossmann_ops/api/schemas.py`
+**Applied at**: Every `POST /predict` request.
+
+```python
+class PredictRequest(BaseModel):
+ Store: int = Field(..., gt=0)
+ DayOfWeek: int = Field(..., ge=1, le=7)
+ Date: date = Field(...)
+ Promo: int = Field(..., ge=0, le=1)
+ StateHoliday: str = Field(...) # validated by downstream OHE
+ StoreType: str = Field(..., pattern="^[abcd]$")
+ Assortment: str = Field(..., pattern="^[abc]$")
+ CompetitionDistance: Optional[float] = Field(None, ge=0)
+```
+
+**Type coercion**: FastAPI + Pydantic v2 automatically handles JSON type coercion (e.g., `"1"` → `1` for integers). However, a completely wrong type (e.g., `"NOT_AN_INTEGER"` for `DayOfWeek`) is rejected immediately.
+
+**Validation failures** produce `HTTP 422 Unprocessable Entity` with a structured error body detailing exactly which field failed and why. These 422 responses are automatically counted by the Prometheus instrumentation middleware and appear in the "Error Rate" and "HTTP Status Distribution" Grafana panels.
+
+**Regex guards**:
+- `StoreType: pattern="^[abcd]$"` — only accepts single lowercase letter a–d.
+- `Assortment: pattern="^[abc]$"` — only accepts single lowercase letter a–c.
+
+These prevent injections of arbitrary string values that could create unexpected OHE columns and break the feature contract.
+
+**`CompetitionDistance` is `Optional[float]`** — `None` triggers an automatic lookup from `store.csv` at inference time. This accommodates rural stores with no recorded competitor without requiring the caller to guess a value.
+
+### Interactive API Documentation (Swagger)
+
+
+
+---
+
+## 4. Layer 3: Functional Guard (Inference-time)
+
+**Implementation**: `src/rossmann_ops/api/main.py`, lines 204–219
+**Applied at**: Inside `predict()`, after Pydantic validation passes.
+
+```python
+# api/main.py, lines 204-219
+if (
+ request.CompetitionDistance is not None
+ and request.CompetitionDistance > 100_000
+):
+ INFERENCE_ANOMALIES_BLOCKED.inc()
+ logger.warning(
+ "Anomalous CompetitionDistance=%.2f blocked for Store=%d.",
+ request.CompetitionDistance,
+ request.Store,
+ )
+ raise HTTPException(
+ status_code=422,
+ detail="CompetitionDistance exceeds plausible range (>100 000 m). Request blocked.",
+ )
+```
+
+**Why this exists alongside Pydantic**: The Pydantic schema (`schemas.py`, line 36) defines `CompetitionDistance: Optional[float] = Field(None, ge=0)` — a lower bound of 0, but **no upper bound**. This is intentional: rural stores with very large but legitimate competitor distances must be accepted. The maximum recorded value in the training dataset is ~75,860m. The 100,000m threshold was chosen to reject geometrically absurd values (>100km) while accommodating realistic edge cases.
+
+**Observable signal**: Every blocked request increments `inference_anomalies_blocked_total`, the custom Prometheus counter defined at:
+```python
+# api/main.py, lines 72-75
+INFERENCE_ANOMALIES_BLOCKED = Counter(
+ "inference_anomalies_blocked",
+ "Total prediction requests blocked due to anomalous/poisoned input.",
+)
+```
+
+This counter is visualized in the Grafana "Anomalies Blocked" stat panel.
+
+---
+
+## 5. Layer 4: Z-Score Batch Poisoning Detection
+
+**Implementation**: `scripts/simulate_production.py`, function `simulate_attack()`, lines 116–139
+**Applied at**: Pre-ingestion of batch data in the simulation pipeline.
+
+```python
+def simulate_attack():
+ # Inject astronomically high Sales values into a batch
+ malicious_batch = y_test.copy().values
+ malicious_batch[0:10] = 999_999_999 # 10 poisoned rows
+
+ # Compute Z-scores on the entire batch
+ mean_val = np.mean(malicious_batch)
+ std_val = np.std(malicious_batch)
+ z_scores = (malicious_batch - mean_val) / std_val
+ max_z = np.max(z_scores)
+
+ if max_z > Z_SCORE_THRESHOLD: # threshold: 3.0 (from configs/params.yaml)
+ logger.error(f"Anomaly/Poisoning detected! Z-Score {max_z:.2f} exceeds threshold.")
+ sys.exit(1) # non-zero exit → CI job marks as "detected"
+```
+
+**Threshold**: `z_score_threshold: 3.0` (`configs/params.yaml`, line 47). This is the standard statistical threshold for outlier detection — a value 3 standard deviations from the mean is present in ~0.27% of a normal distribution.
+
+**In CI**: The `simulate` job marks `continue-on-error: true` for `attack` mode because `sys.exit(1)` is the **expected success state** — it proves the detection logic fired. Without `continue-on-error`, CI would incorrectly report the job as failed.
+
+---
+
+## 6. Telemetry — Prometheus Metrics
+
+**Instrumentation library**: `prometheus-fastapi-instrumentator` v7.1.0+
+**Initialization** (`api/main.py`, lines 40–58):
+
+```python
+Instrumentator().instrument(
+ app,
+ latency_highr_buckets=(
+ 0.005, 0.01, 0.025, 0.05, 0.075, 0.1,
+ 0.25, 0.5, 0.75, 1.0, 2.5, 5.0, 7.5, 10.0,
+ ),
+).expose(app)
+```
+
+`latency_highr_buckets` configures the histogram bucket boundaries for `http_request_duration_seconds`. Custom sub-100ms granularity (5ms, 10ms, 25ms, 50ms, 75ms, 100ms) enables accurate p95/p50 quantile computation for fast inference responses — the "bucket blindness" that would occur with default buckets (which start at 250ms) is avoided.
+
+### Auto-instrumented Metrics
+
+| Metric | Type | Labels |
+| :------------------------------ | :-------- | :---------------------------- |
+| `http_requests_total` | Counter | `method`, `handler`, `status` |
+| `http_request_duration_seconds` | Histogram | `method`, `handler` |
+
+### Custom Metrics
+
+| Metric | Type | Description | Defined |
+| :---------------------------------- | :------ | :---------------------------- | :--------------------- |
+| `sales_inference_total` | Counter | Successful predictions served | `api/main.py`, line 68 |
+| `inference_anomalies_blocked_total` | Counter | Poisoned requests blocked | `api/main.py`, line 72 |
+
+**`sales_inference_total` incremented at** (`api/main.py`, line 273):
+```python
+SALES_INFERENCE_TOTAL.inc()
+return {"Store": ..., "PredictedSales": prediction, ...}
+```
+Incremented only after a **successful prediction** — not on errors — providing an accurate count of served forecasts.
+
+**Metrics endpoint**: `GET /metrics` (exposed automatically by `.expose(app)`, Prometheus scrapes this).
+
+---
+
+## 7. Grafana Dashboard
+
+**Deployment**: ConfigMap (`k8s/grafana-dashboard.yaml`), dashboard ID `rossmann-perf`.
+**Access**: `http://localhost:30200` (credentials: `admin / prom-operator`)
+**Auto-refresh**: 1s (dashboard `"refresh": "1s"`)
+**Default time window**: Last 15 minutes (`"time": {"from": "now-15m", "to": "now"}`)
+
+### Panel Specifications
+
+
+Global RPS (1m) — Stat panel (top-left)
+
+```promql
+sum(irate(http_requests_total[1m]))
+```
+- Type: `stat` with area graph
+- Color: fixed green, turns red at 100 req/s
+- Unit: `reqps`
+
+Uses `irate` (instantaneous derivative) not `rate` (average over window) — gives per-scrape-interval sensitivity (1s), making burst spikes immediately visible rather than smoothed over the rate window.
+
+
+
+
+Error Rate % (5m) — Gauge panel (top, second)
+
+```promql
+sum(rate(http_requests_total{status=~"4..|5.."}[5m])) / sum(rate(http_requests_total[5m])) * 100
+```
+- Type: `gauge`
+- Thresholds: < 5% green, 5–20% yellow, > 20% red
+- Unit: `percent`
+
+Counts both 4xx (client errors, including 422 schema blocks) and 5xx (server errors) in the numerator. During the observability demo Phase 2 and Phase 3, this gauge will climb into the yellow/red zones.
+
+
+
+
+Total Predictions Served — Stat panel (top, third)
+
+```promql
+sum(sales_inference_total)
+```
+- Type: `stat`
+- Color: fixed blue
+- Unit: integer (no decimals)
+
+Cumulative count across all replicas. Sums across pods — the `sum()` aggregation is critical because each pod maintains its own counter independently.
+
+
+
+
+Anomalies Blocked — Stat panel (top-right)
+
+```promql
+sum(inference_anomalies_blocked_total)
+```
+- Type: `stat` with background color mode
+- Thresholds: 0 = green background, ≥ 1 = orange background
+
+Immediately highlights any poisoning attempt. The color change from green → orange provides a clear visual signal during the demo.
+
+
+
+
+p95 / p50 Latency (5m) — Time series (middle-left)
+
+```promql
+# p95
+histogram_quantile(0.95, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))
+
+# p50 (median)
+histogram_quantile(0.50, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))
+```
+- Type: `timeseries`
+- Both quantiles plotted simultaneously; legend shows mean + max
+- Red threshold line at 0.5s
+- Smooth line interpolation, 10% fill opacity
+
+
+
+
+HTTP Status Distribution (5m) — Donut chart (middle-right)
+
+```promql
+sum by (status) (rate(http_requests_total[5m]))
+```
+- Type: `piechart` (donut mode)
+- Color overrides: 4xx → orange, 5xx → red, 2xx → default green
+- Legend in table mode (right placement)
+
+During normal operation, this will be nearly entirely 2xx. During Phase 2/3 of the observability demo, 422 slices become clearly visible.
+
+
+
+---
+
+## 8. Drift Detection — KS-Test
+
+**Implementation**: `scripts/simulate_production.py`, function `simulate_drift()`, lines 166–187
+
+### Mechanics
+
+The **Kolmogorov-Smirnov two-sample test** (`scipy.stats.ks_2samp`) measures whether two samples were drawn from the same underlying distribution. It is non-parametric — no distribution assumption is made.
+
+```python
+def simulate_drift():
+ X_test, _ = get_test_data()
+
+ # Baseline: real CompetitionDistance values from the simulation set
+ baseline_feature = X_test[DRIFT_FEATURE].fillna(0).values
+
+ # Shifted: simulated production data with introduced distributional shift
+ shifted_feature = baseline_feature + DRIFT_SHIFT # +50,000m shift (from params.yaml)
+
+ stat, p_value = stats.ks_2samp(baseline_feature, shifted_feature)
+
+ if p_value < P_VALUE_THRESHOLD: # 0.05 (from params.yaml)
+ trigger_github_workflow()
+```
+
+**Parameters** (all in `configs/params.yaml`, lines 46–50):
+```yaml
+pipeline:
+ simulation:
+ z_score_threshold: 3.0
+ p_value_threshold: 0.05
+ drift_feature: "CompetitionDistance"
+ drift_shift: 50000.0
+```
+
+**Drift feature**: `CompetitionDistance` is the monitored feature. A 50,000m shift is applied to simulate a scenario where incoming data's competitor distance distribution has significantly changed from the training distribution (e.g., mass market changes, new competitor dataset, geographical relocation of stores).
+
+### KS Statistic Interpretation
+
+The KS statistic $D$ is the maximum absolute difference between the empirical CDFs of the two distributions:
+$$D = \sup_x |F_1(x) - F_2(x)|$$
+
+A high $D$ value (→ 1.0) and low p-value (→ 0) indicates the distributions are significantly different. At `drift_shift = 50,000`, the two distributions are completely non-overlapping → KS statistic will be 1.0 and p-value effectively 0.
+
+---
+
+## 9. Continuous Training Trigger
+
+**Implementation**: `scripts/simulate_production.py`, function `trigger_github_workflow()`, lines 142–163
+
+```python
+def trigger_github_workflow():
+ url = f"https://api.github.com/repos/{REPO_OWNER}/{REPO_NAME}/dispatches"
+ headers = {
+ "Accept": "application/vnd.github.v3+json",
+ "Authorization": f"token {GITHUB_PAT}",
+ }
+ payload = {"event_type": "drift_detected"}
+ res = requests.post(url, headers=headers, json=payload)
+ # Success: HTTP 204 No Content
+```
+
+**Required environment variable**: `GITHUB_PAT` — a GitHub Personal Access Token with `repo` scope. Without it, drift is logged but no retraining is triggered.
+
+**Receiving side** (`.github/workflows/mlops_pipeline.yaml`, lines 9–10):
+```yaml
+on:
+ repository_dispatch:
+ types: [drift_detected]
+```
+
+When this event is received, GitHub Actions runs the full CI → Train → Build → Push pipeline, producing a new model artifact that reflects the shifted data distribution.
+
+---
+
+**Run**:
+```bash
+just demo
+```
+
+### Demo Phase Sequences
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart LR
+ classDef phase fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef obs fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+ classDef warn fill:#431407,stroke:#f97316,color:#fdba74,stroke-width:2px
+
+ subgraph P1 [" Phase 1: Normal "]
+ direction TB
+ N1["750 valid requests"]:::phase
+ O1["Dashboard all green\nPredictions increment"]:::obs
+ end
+
+ subgraph P2 [" Phase 2: Schema Errors "]
+ direction TB
+ N2["350 malformed JSON"]:::phase
+ O2["Error Rate % gauge climbs\nDonut chart turns orange"]:::warn
+ end
+
+ subgraph P3 [" Phase 3: Poisoning Attack "]
+ direction TB
+ N3["350 extreme values"]:::phase
+ O3["Anomalies Blocked stat panel\nBackground turns orange"]:::warn
+ end
+
+ P1 ==> P2 ==> P3
+```
+
+### Demo Observations (Grafana)
+
+
+
+### Phase 1: Normal Traffic (750 requests)
+- Sends valid prediction requests with randomized store IDs (1–1000).
+- **Observe**: "Global RPS" climbs, "Total Predictions" increments steadily. Dashboard stays green.
+
+### Phase 2: Schema Errors (350 requests)
+- Sends `{"Store": "INVALID_TYPE", "Missing": "Fields"}` — wrong types, missing required fields.
+- FastAPI/Pydantic rejects each request with `HTTP 422 Unprocessable Entity`.
+- **Observe**: "Error Rate %" gauge climbs into yellow. Donut chart shows 4xx slice growing orange. "Anomalies Blocked" stays at 0 (these are schema errors, not poisoning attempts).
+
+### Phase 3: Poisoning Attack (350 requests)
+- Sends `CompetitionDistance: 999999.0` — geometrically impossible value (>100km).
+- Both Pydantic (`ge=0` passes) and the functional guard (`> 100,000`) evaluate. The guard blocks each request and increments `inference_anomalies_blocked_total`.
+- **Observe**: "Anomalies Blocked" increments. "Error Rate %" stays elevated. Dashboard background turns orange for the blocked counter.
+
+### Manual Variants (`simulate_production.py --mode`)
+
+| Mode | What Happens |
+| :------- | :-------------------------------------------------------------------------------------------------- |
+| `schema` | Single malformed JSON request; logs the 422 response with details |
+| `attack` | Generates a batch with 10 poisoned rows (Sales=999,999,999); Z-score check fires; exits with code 1 |
+| `drift` | KS-Test on CompetitionDistance; fires repository_dispatch if p < 0.05 |
+
+---
+
+## 11. Model Interpretability (SHAP)
+
+**Implementation**: `train_model.py`, lines 191–209
+**Artifact**: `models/shap_summary.png`
+
+The system uses **SHAP (SHapley Additive exPlanations)** to provide globally interpretable insights into model behavior. A `TreeExplainer` is trained on a representative sample of the training data, producing a summary plot that ranks features by their impact on forecasted sales.
+
+### Global Feature Importance
+
+
+
+**Key Insights**:
+- **`Store_TargetMean`**: Dominant predictor; captures individual store potential.
+- **`Promo`**: Strongest dynamic catalyst for sales spikes.
+- **`DayOfWeek`**: Captures weekly cyclical patterns (e.g., Saturday surges).
diff --git a/docs/TEAM_CHEAT_SHEET.md b/docs/TEAM_CHEAT_SHEET.md
new file mode 100644
index 0000000..0389bf4
--- /dev/null
+++ b/docs/TEAM_CHEAT_SHEET.md
@@ -0,0 +1,360 @@
+# Team Cheat Sheet
+
+**Anticipated Q&A mapped directly to source code and implementation details.**
+
+---
+
+## How to Use This Document
+
+Each entry follows:
+- **Q**: Exact or paraphrased question likely to be asked
+- **A**: Precise technical answer with file path + line number references
+- **Demo**: How to show it live or in a screenshot
+
+---
+
+## Section 1: Data & Feature Engineering
+
+---
+
+**Q: How do you handle missing `CompetitionDistance` values in production?**
+
+**A**: `CompetitionDistance` is optional in the API schema (`schemas.py`, line 36: `Optional[float] = Field(None, ge=0)`). When `None`, the API merges the request row against `store.csv` (`api/main.py`, lines 226–233) and uses the store's recorded value. If both are missing (truly unknown), `build_features()` fills it with the training-set median before log-transforming (`features.py`, line 66: `dist.fillna(train_comp_median)`). This prevents the `NaN → log1p(NaN) = NaN` failure path that would corrupt the inference feature vector.
+
+**Demo**: Call `GET /store/{store_id}` with a real store ID — returns CompetitionDistance from metadata.
+
+---
+
+**Q: How do you prevent data leakage in your feature engineering?**
+
+**A**: Two mechanisms:
+1. **Target encoding** uses an **expanding mean with a 1-row shift** (`train_model.py`, lines 92–93): `x.shift().expanding().mean()`. This means at row $t$, only rows $\{0, 1, ..., t-1\}$ contribute. The current row's sales value never informs its own encoding.
+2. **Train/holdout split is strictly chronological** (`train_model.py`, lines 57–73). No random shuffling. The holdout set (last 28 days) is never seen during training. The simulation set (last 42 days) is never seen during either training or evaluation.
+
+---
+
+**Q: Why did you use `log1p` transformation on the target?**
+
+**A**: Raw sales values are right-skewed (range: 1–41,551 EUR). Log transformation compresses this, making the residual distribution more homoskedastic — which is an assumption of many loss functions and benefits tree model split quality. `np.log1p` is used instead of `np.log` to handle any edge-case zeros (though zero-sales rows are pre-filtered). Predictions are restored to the original scale via `np.expm1` at inference time (`api/main.py`, line 271). All reported metrics are computed in the original EUR space, not log-space (`train_model.py`, lines 167–168: `preds_log` → `preds = np.expm1(preds_log)`).
+
+---
+
+**Q: What is OHE column alignment and why does it matter?**
+
+**A**: At training time, `pd.get_dummies` produces all categories present in the training data (e.g., all 4 store types → 4 columns). At inference, a single request row has only one store type — `pd.get_dummies` on that row produces only 1 OHE column. Without alignment, the model would receive a feature vector with the wrong number of columns and fail. The `expected_ohe_cols` contract in `configs/params.yaml` (lines 14–25) and the alignment logic in `build_features()` (`features.py`, lines 68–84) ensure all 11 OHE columns are always present, with missing ones filled as 0.
+
+---
+
+**Q: How do you ensure the same feature transformations are applied at training and inference?**
+
+**A**: The shared `build_features(df, train_comp_median, expected_ohe_cols)` function in `src/rossmann_ops/features.py` is called identically during both training (`train_model.py`, lines 83–88) and at inference (`api/main.py`, lines 245–249). The `train_comp_median` is the only stateful parameter — it is computed once from the training set and stored in the target means artifact for reuse. The OHE contract (`ohe_expected_columns` in `params.yaml`) acts as the schema contract between training and serving.
+
+---
+
+## Section 2: Modeling
+
+---
+
+**Q: What model did you use? Why not XGBoost?**
+
+**A**: Production model is a **Random Forest Regressor** (`sklearn.ensemble.RandomForestRegressor`). XGBoost was evaluated during Optuna hyperparameter search (`notebooks/03_optuna.ipynb`). Random Forest was selected because it delivered comparable RMSPE on the holdout set with: (a) native parallelism via `n_jobs=-1` without GPU/specialized builds; (b) direct compatibility with SHAP's `TreeExplainer` fast path; (c) simpler production dependency footprint. Hyperparameters: `n_estimators=50`, `max_depth=10`, `min_samples_split=6`, `random_state=105` (defined in `configs/params.yaml`, lines 36–39).
+
+---
+
+**Q: What metric did you use to evaluate the model and why?**
+
+**A**: **RMSPE** (Root Mean Squared Percentage Error) — the official Kaggle Rossmann competition metric. It penalizes large percentage errors more than absolute errors, which is appropriate for sales forecasting where a 20% error on a high-volume store is more costly than a 20% error on a low-volume store. Implementation at `train_model.py`, lines 27–32. Zero-sales rows are excluded from computation (`mask = y_true != 0`) because RMSPE is undefined for true=0. All metrics are logged to MLflow: `holdout_rmspe`, `holdout_rmse`, `holdout_mae`, `holdout_r2`.
+
+---
+
+**Q: How large is your holdout set?**
+
+**A**: 28 calendar days (`configs/params.yaml`, line 8: `holdout_days: 28`). The dataset spans 2013–2015. The last 28 days of training-eligible data form the holdout; the last 42 days are the simulation set excluded from all training and evaluation. Exact sizes are logged to MLflow on each run (`train_model.py`, line 71: `"n_cv_rows": len(X_cv)`).
+
+---
+
+**Q: How do you track experiments?**
+
+**A**: MLflow with local SQLite tracking (`mlruns/mlflow.db`). Both `mlflow.sklearn.autolog()` (sklearn-specific metadata) and explicit `mlflow.log_params()` / `mlflow.log_metrics()` calls are used (`train_model.py`, lines 128–221). Each training run logs 8 parameters and 4 metrics, plus the SHAP plot artifact and the serialized model.
+
+**Demo**: Run `just mlflow-ui` to view the local experiment dashboard with all historical runs and model artifacts.
+
+---
+
+## Section 3: API & Serving
+
+---
+
+**Q: How does the API load the model?**
+
+**A**: The API first checks for `MLFLOW_MODEL_URI` + `MLFLOW_TRACKING_URI` environment variables. If both are set, it loads from the MLflow registry (`api/main.py`, lines 112–122). If that fails or vars are unset, it falls back to the local filesystem path `models/production_model/` (`api/main.py`, lines 124–130). This allows the same Docker image to work with either a live remote registry or a baked-in local model.
+
+**Demo**: `GET /health` returns `"model_loaded": true` + `model_version` when the model is successfully loaded.
+
+---
+
+**Q: What are the API endpoints?**
+
+| Endpoint | Method | Purpose |
+| :------------------ | :----- | :------------------------------------------------------------------------------------ |
+| `/predict` | `POST` | Main inference endpoint. Returns `PredictedSales` as a float in original EUR scale. |
+| `/health` | `GET` | Readiness check. Returns `503` if model/data not loaded. Used by K8s Readiness Probe. |
+| `/health/live` | `GET` | Liveness check. Always `200` if the process is running. Used by K8s Liveness Probe. |
+| `/health/shap` | `GET` | Serves `models/shap_summary.png` for UI rendering. Returns `404` if not found. |
+| `/store/{store_id}` | `GET` | Returns store metadata (StoreType, Assortment, CompetitionDistance) for UI pre-fill. |
+| `/drift-trigger` | `POST` | Stub endpoint for monitoring integration. Logs event, returns drift status. |
+| `/metrics` | `GET` | Prometheus scrape endpoint (auto-exposed by instrumentator). |
+| `/docs` | `GET` | Auto-generated Swagger/OpenAPI documentation. |
+
+---
+
+**Q: Walk me through what happens when I POST to `/predict`.**
+
+**A**:
+1. **Pydantic validation** (`schemas.py`): Type checks, field presence, value bounds. Returns `422` on failure.
+2. **Functional guard** (`api/main.py`, lines 204–219): `CompetitionDistance > 100,000m` → blocks, increments `inference_anomalies_blocked_total`, returns `422`.
+3. **DataFrame reconstruction**: Request is converted to a single-row DataFrame.
+4. **Store metadata enrichment** (`api/main.py`, lines 226–233): Merges with `STORE_DF` to fill missing CompetitionDistance from `store.csv`.
+5. **Feature transforms** (`build_features`): Date extraction, OHE, log-transform CompDist, OHE alignment.
+6. **Target encoding** (`apply_target_encoding`): Adds `Store_TargetMean` from the pre-loaded artifact.
+7. **Column alignment**: Ensures all 18 model features are present in the correct order.
+8. **Inference**: `MODEL.predict(X)` returns a log-space value.
+9. **Inverse transform**: `np.expm1(prediction_log[0])` restores to EUR scale.
+10. **Counter**: `SALES_INFERENCE_TOTAL.inc()`.
+11. **Response**: `{"Store": ..., "Date": ..., "PredictedSales": ..., "ModelVersion": ...}`.
+
+---
+
+`CompetitionDistance` is optional — omit it and the API fills it from `store.csv`.
+
+### Inference Request Lifecycle
+
+```mermaid
+%%{init: {"theme": "base", "themeVariables": {"background": "#0f172a", "primaryColor": "#1e293b", "primaryTextColor": "#f1f5f9", "primaryBorderColor": "#334155", "lineColor": "#64748b", "edgeLabelBackground": "#1e293b"}}}%%
+flowchart TD
+ classDef req fill:#1e3a5f,stroke:#60a5fa,color:#bfdbfe,stroke-width:2px
+ classDef guard fill:#0c2340,stroke:#3b82f6,color:#bfdbfe,stroke-width:2px
+ classDef logic fill:#052e16,stroke:#22c55e,color:#bbf7d0,stroke-width:2px
+ classDef model fill:#2d0a47,stroke:#c084fc,color:#f3e8ff,stroke-width:2px,font-weight:bold
+ classDef stat fill:#431407,stroke:#f97316,color:#fdba74,stroke-width:1.5px
+ classDef fail fill:#7f1d1d,stroke:#ef4444,color:#fee2e2,stroke-width:2px
+
+ Start(["POST /predict\nJSON Payload"]):::req
+
+ subgraph Defense [" Defensive Stack "]
+ direction TB
+ L1["Layer 2: Pydantic Validation\nType & Bounds check"]:::guard
+ L2["Layer 3: Functional Guard\nCompDist > 100km?"]:::guard
+ end
+
+ subgraph Pipe [" Feature Enrichment & Scaling "]
+ direction TB
+ E1["Store metadata merge\n(data/raw/store.csv)"]:::logic
+ E2["Stateless Transforms\n(build_features)"]:::logic
+ E3["Target Mean Encoding\n(store_target_means.json)"]:::logic
+ end
+
+ Inference{{"Model Inference\nRandom Forest\n(log-space)"}}:::model
+ Inverse["Inverse Transform\nnp.expm1(y)"]:::logic
+ Metric["SALES_INFERENCE_TOTAL++"]:::stat
+ Res(["JSON Response\n{ PredictedSales: float }"]):::req
+
+ Start --> L1
+ L1 -->|"Pass"| L2
+ L1 -->|"Fail"| F1["HTTP 422\nSchema Error"]:::fail
+
+ L2 -->|"Pass"| E1
+ L2 -->|"Block"| F2["HTTP 422\nAnomalous Input"]:::fail
+
+ E1 --> E2 --> E3 --> Inference
+ Inference --> Inverse --> Metric --> Res
+```
+
+---
+
+## Section 4: Security & Robustness
+
+---
+
+**Q: How do you defend against data poisoning?**
+
+**A**: Four independent layers:
+1. **Pandera** (`data_validation.py`): Validates training data schema before any feature engineering. Negative sales, impossible day numbers, invalid open-status values → `SchemaError` → training halted.
+2. **Pydantic** (`schemas.py`): Validates every API request. Wrong types, missing required fields, values outside bounds → `HTTP 422`, request rejected in <1ms.
+3. **Functional guard** (`api/main.py`, lines 204–219): Explicitly blocks `CompetitionDistance > 100,000m` — a geometrically impossible value used as a poisoning attack vector. Increments `inference_anomalies_blocked_total` counter.
+4. **Distribution Defense**: For live production, Z-Score batch detection or KS-Test drift detection triggers automated retraining.
+ - **Demo**: Use `just demo` (Phase 3). It sends a batch of extreme values that trigger Layer 3 guards and create visible spikes in the "Anomalies Blocked" Grafana panel.
+
+---
+
+**Q: What happens if the model artifacts are not loaded at startup?**
+
+**A**: The API application still starts (does not crash). However:
+- `GET /health` returns `HTTP 503` with `"status": "degraded"` and per-artifact flags (`model_loaded: false`, etc.).
+- `POST /predict` returns `HTTP 503` with `"Model or target means not loaded"`.
+- K8s Readiness Probe (`/health`) receives `503` → pod is **excluded from the Service endpoint pool** → no traffic is routed to the unready pod.
+- Only when all three artifacts (model, store data, target means) are loaded does `/health` return `200` → pod is added back to rotation.
+
+---
+
+**Q: How do you ensure `StoreType` and `Assortment` can only be valid values?**
+
+**A**: Pydantic regex patterns in `schemas.py`:
+```python
+StoreType: str = Field(..., pattern="^[abcd]$") # only a, b, c, d
+Assortment: str = Field(..., pattern="^[abc]$") # only a, b, c
+```
+Any other value returns `422` immediately. This prevents novel OHE categories from being injected at inference that would break the model's feature contract.
+
+---
+
+## Section 5: Kubernetes & Infrastructure
+
+---
+
+**Q: Why did you choose `RollingUpdate` over `Recreate` for the deployment strategy?**
+
+**A**: The inference API is **fully stateless** — each pod independently loads the same model artifact, and any replica can serve any request without coordination. `RollingUpdate` with `maxUnavailable: 0` guarantees **zero-downtime updates**: old pods are only terminated after the new pod's readiness probe confirms it is healthy and serving. `Recreate` would cause complete downtime during updates and is appropriate only for stateful services requiring exclusive resource access (e.g., single-process database migrations). Both `api.yaml` (line 11) and `ui.yaml` (line 11) explicitly declare this strategy.
+
+---
+
+**Q: How does the Streamlit UI communicate with the API inside Kubernetes?**
+
+**A**: Via Kubernetes internal DNS. The UI deployment sets `API_URL=http://api-service:8000` (`k8s/ui.yaml`, line 33). `api-service` is the name of the K8s Service object defined in `k8s/api.yaml`. Kubernetes resolves this hostname to the Service's ClusterIP, which load-balances across all healthy API pod replicas. This is more robust than `localhost` (which would only reach the same pod) or `NodePort` (which adds a network hop through the host machine).
+
+---
+
+**Q: What resources are allocated to each container?**
+
+| Container | CPU Request | CPU Limit | Memory Request | Memory Limit |
+| :------------- | :---------- | :-------- | :------------- | :----------- |
+| `rossmann-api` | 100m | 500m | 256Mi | 512Mi |
+| `rossmann-ui` | 100m | 500m | 256Mi | 512Mi |
+
+Requests define the guaranteed allocation for scheduling; limits cap total consumption. Both containers are identical — appropriate since both perform lightweight I/O-bound work (the model inference itself is CPU-bound but fast for a 50-tree RF predicting a single row).
+
+---
+
+**Q: How do you check cluster status?**
+
+```bash
+just k8s-status
+# equivalent: kubectl get pods,services,servicemonitors -A
+```
+
+---
+
+## Section 6: CI/CD & Continuous Training
+
+---
+
+**Q: What triggers the CI/CD pipeline?**
+
+**A**: Multiple triggers (`.github/workflows/mlops_pipeline.yaml`, lines 3–18):
+- **Push to any branch**: CI only (lint + test).
+- **Push to `main`** or any version tag: CI → Train → Build & Push Docker images.
+- **Version tag (`v*.*.*`)**: CI → Train → Build & Push → **GitHub Release** (with deployment bundle ZIP and versioned K8s manifests).
+- **`repository_dispatch` (event: `drift_detected`)**: Full CI → Train → Build & Push. Triggered by the monitoring layer or manual dispatch when feature/prediction drift exceeds thresholds.
+- **`workflow_dispatch` (manual)**: CI + Optional Deployment verification.
+
+---
+
+**Q: How does the pipeline get training data in CI — there's no data in the repo?**
+
+**A**: Data is fetched from DagsHub via DVC pull (`mlops_pipeline.yaml`, line 85: `uv run dvc pull -r dagshub`). DVC credentials are set at runtime using GitHub Secrets (lines 78–82):
+```yaml
+uv run dvc remote modify dagshub --local auth basic
+uv run dvc remote modify dagshub --local user ${{ secrets.DAGSHUB_USERNAME }}
+uv run dvc remote modify dagshub --local password ${{ secrets.DAGSHUB_PAT }}
+```
+The `--local` flag writes credentials to `.dvc/config.local` (not tracked by Git) — credentials never touch the repository.
+
+---
+
+**Q: How are Docker images tagged?**
+
+**A**: On version tag pushes (`refs/tags/v*.*.*`), images receive both the semantic version tag AND `latest`:
+```
+user/rossmann-api:v1.2.3
+user/rossmann-api:latest
+```
+On `main` branch pushes (no tag), only `latest` is updated. This enables **immutable version pinning**: a grader can deploy `v1.2.3` exactly, and that image will never change regardless of future `latest` updates.
+
+---
+
+**Q: How does continuous training work end-to-end?**
+
+**A**:
+1. Feature/Prediction drift is monitored (conceptualized in `scripts/simulate_production.py`).
+2. KS-Test is performed: baseline distributions vs. incoming production data.
+3. If drift is detected (p-value < 0.05), a `repository_dispatch` is fired to GitHub Actions.
+4. GitHub Actions runs the full pipeline: CI → Train → Build → Push.
+5. A new Docker image with the retrained model is published and deployed via `RollingUpdate`.
+
+**Demo**: Since we lack a live incoming stream for the grader demo, we use `just demo`. This script simulates three phases of traffic against the cluster, creating the baseline for monitoring and then triggering the defensive "Blocked" spikes in Grafana, demonstrating the system acts on anomalous data.
+
+---
+
+**Q: Where are API tests? What do they test?**
+
+**A**: `tests/test_api.py`. Tests use `TestClient` with mocked model artifacts (deterministic mock objects). Key test cases:
+- `test_health_endpoint_with_model`: Asserts `GET /health` returns `200` + `model_loaded: true`.
+- `test_predict_valid_payload`: Asserts `POST /predict` returns `200` + `PredictedSales` as a float.
+- `test_predict_missing_required_fields`: Asserts missing required fields → `422`.
+- `test_predict_invalid_store_type`: Asserts invalid `StoreType` (e.g., `"z"`) → `422`.
+- `test_predict_invalid_competition_distance`: Asserts `CompetitionDistance > 100,000m` → `422`.
+- API latency test: Asserts p99 latency on 50 consecutive requests stays under a threshold.
+
+Run: `just test` or `uv run pytest tests/ -v`.
+
+---
+
+## Section 7: Explainability & Interpretability
+
+---
+
+**Q: How do you explain your model's predictions?**
+
+**A**: SHAP (SHapley Additive exPlanations) via `shap.TreeExplainer` (`train_model.py`, lines 191–209). The TreeExplainer uses an algorithm specific to tree ensembles — exact and orders of magnitude faster than the generic permutation-based approach. SHAP values are computed on a 500-row sample of the training set. The global summary plot (`models/shap_summary.png`) shows:
+- Each feature on the Y-axis, ordered by mean absolute SHAP value (importance).
+- Each point is one training sample, coloured by feature value (red = high, blue = low).
+- X-axis shows impact on model output (log-space sales prediction).
+
+**Demo**: In the Streamlit dashboard, open "Model Diagnostics & Explainability (SHAP)" → click "Fetch SHAP Summary". The UI calls `GET /health/shap`, receives the PNG directly, and renders it in-browser.
+
+---
+
+**Q: What feature was most important to the model?**
+
+**A**: Based on the SHAP analysis, `Store_TargetMean` (the per-store mean sales target encoding) is typically the dominant feature — it captures each store's baseline sales level. `Promo` is the second most important, and `DayOfWeek` captures the weekly sales cycle. This is interpretable: a store's historical average is the strongest predictor of tomorrow's sales, promotions significantly lift sales, and day-of-week captures regular shopping patterns.
+
+---
+
+## Section 8: Reproducibility
+
+---
+
+**Q: How do you ensure reproducibility?**
+
+**A**: Several mechanisms:
+- `uv.lock` pins all direct and transitive Python dependencies to exact versions (`python requires-python = "~=3.12.0"`).
+- `random_state=105` is fixed in `configs/params.yaml` and passed to both `RandomForestRegressor` and `sample()` calls in training.
+- DVC pins data artifacts by content hash (SHA256 in `.dvc` files) — `dvc pull` always retrieves the exact same data.
+- Docker images built from the same `Dockerfile.api/ui` + `uv.lock` + model artifacts are byte-for-byte identical.
+- GitHub Actions uses `uv sync --frozen` — refuses to proceed if `uv.lock` is out of sync with `pyproject.toml`.
+
+---
+
+**Q: How does a grader deploy and test the system from scratch?**
+
+**A**: Three primary options, ordered by complexity:
+1. **Option A: Local Development** (Fastest): `just setup` → `just train-prod` → `just serve-api` / `just serve-ui`. Runs directly on host machine.
+2. **Option B: Docker Compose**: `just docker-up`. Pulls pre-built images from DockerHub and runs the stack in containers without K8s.
+3. **Option C: Full K8s Production** (**Recommended**): `just deploy-all`. Trains, builds, and deploys to a local KinD cluster with Prometheus/Grafana monitors as a complete MLOps environment.
+
+**Prerequisites**: `just` and `uv` (global), Docker Desktop, Helm (for K8s path).
+## Section 9: Simulation Scripts vs. Observability Demo
+
+- **`simulate_production.py`**: A conceptual script implementing KS-Test drift detection and Z-Score poisoning logic. It is intended for architecture reference and CI/CD integration.
+- **`observability_demo.py` (`just demo`)**: The primary tool for live demonstration. It hits the production endpoints (K8s or Docker) sequentially with Normal, Malformed, and Attack traffic.
+- **Why use `just demo`?** In a local grader environment, we cannot wait for natural drift. `just demo` creates instant metric spikes in Grafana, proving the Prometheus instrumentation and defensive guards are functional.
diff --git a/docs/assets/gha_pipeline.png b/docs/assets/gha_pipeline.png
new file mode 100644
index 0000000000000000000000000000000000000000..36c4bbe1f092282ebbd4bdf8d55ef1ae46046dc6
GIT binary patch
literal 214063
zcmeFZg;!f&w*^W|3xy&DTHK1e7Pkt;o#O6L+?`Nb+=>QIixv$Y+#Nznp+JD31xg{f
z>pT73aqstCd1Jgk;GHotNY2PPIeYK5*P3gtxro+KlgG!U#KpkCz*kg|)xyBQfns1h
zggn9m&J^cnTmpaYd1}eO#;6*n-U2>6uzRKY3In4i@$t3AL*O%xn}UHS1_nXz-`BlC
z*KgJs7L*)w!nl#pX^K7RBuO!g#&YtC05H@-4B=@iDQpachCI%*~p#hrKKe
z|NBs-Cyh-4XAl1QS$?q&hFqK%5bD?2;tM~7Mo0b6L&5X%ydd#e+y8MNIKsvG3DN&L
z9Ny!alKii;Nwy`+LjUV5k%Q0w|K)$p%KzJJH26)t{y!Jz=rZKe_y3&14IA=^uo>C^
zdHez;I%H)OJ2+S63tJW5KK(z|_0H2tSVA}(K4%;bx9zpnj?&Sd7>W6xb}@AKIY#Au
z1N@ur%z>6sW39Py*W
zIQ8?^=bwp)_pdy7>F2ya7B=HEoiS+VD)wYNws`v8pg+j?@6u{l{RskB+bkb
z4b5`g#55Hh(1M0$&D7LX4Gj&27nDU6_7YTJg!%A}Yt1~(gO9Hc4{;xT%4cF>TJdN-
zxF8uG9c5-^#dZ?53`N=YZXd1s`14z05i!QluzuBe1d$-zeL
z7kC`Q$ern?Hv4|mAHsk
z&%&2#0#>~8C)g&wedyiFFDy)To^M=JROoZ?)zl@We?Fm3|D?~hZsjBZN1mZ6N6oT^
zyjz1jyZ6=m_Z&7h&+mW0;7QC@=%at~=t*opx!|oIfyCw0g~bQgAu?NA4x-cq+r5q+
zfy{oD-|FMSR<*=!>T7+_BEq)LtU|*}Ok2cMmZv$*ZKrdfi_2`+N|O^|%X7DJsZFN@
zsZ)DL@V^|KLW<~P<|Wzt$dK^N+Rj
zZdb!T&V+T-i-|SM7HE11b6HKO(>=i@#bl9t5o3|nr$Y5M{4HSlNf|kEnifQIQ%sdM
zvF9OTVxqZfT3T8PN=iL0usSIbusNTs;d{!JZ?9gwUsm2Hilzh>i1^V1Hg0Zav36#V
zXdo_TWIV9gKWc3rx0ymuEQCu3>4GzCc^R^^&1B2#tsRd-igXMGnlIi?Z?xOqMBwLm
zF4ASbxag6?H!f)V_1jxyVQ6BI7Ln8JS&91BDl^3IzHCaTXV6NyH2v$bj=t^vFnkaD(U;eDqXTVeErK^Q45q60{)YrmUcj@(Y`%ckfsZYs_
z=sZ&11B3oiOCOB`&o-?pWn%GT=YDLMoekfB8;RZ}rZjIZUR9c?{>ik1tbHO;x1Ct+
zpXT}3fRnH!KL3Y7SItnK%+0p_2?#nR%_YO1{Ja+QrSk9oeZCz!o&V#uo?w&i9c;(V
zo5n2n#}6$JkE2J}xS3}cvDVJ|;gc&}E-(^7Jiyd#$NE;>-eFaS_gVmp;{v9F&8sT9
zc?yDfsaztG~cDF^HKLjL_1JE1wYk%ryoZYzgjQB
zkB2$=B2&mp9D272O?|qaoRU3LK{1;p>hwc1h$q7Ic|h9z1)aNE_X9#
z#$nn<-R9K-Yjf8KGbRy7l8h&qDuNLfJ)#U3IZBpFDcVy!;8k&c1!~=y>
zSk#|Cd8FT(h2n-S4oK{^IXeFv`Y45AZW)F}DX9JlsMLhU!^^9H7&YNP!G|w4jy*T$
z$h}Hg`;XhA+cq!1-l`Ea+TjmHSFF&|O4RkMaZ8*+xqJDBhKC=T_b2XYBzNLR_I`l7
z#Deh}kk!Z?(H#N|z%5rZz8pupuepg1jlz4;?vo5sxjj9v8(l}SCi|k~*g1Ie>+3%y
zCu`leu`$PljmBG$e|r6jnm_^v2WQsl$)mIt4>7<~%sL6@@URv;B)3YXFw1}*yi`-;
zi69~@bX2;!L^4Pi)cf&+Z!bJvfBoFx2yH&v4ZJw6LsY{5&2Is3?^hr{4`<M~W8^N3fVAMwdd36~`(1o26i%-{b4@Yd7g
z7oRFlI1l_JXR9s9u>gmyNz3*7AizyLC*viTEpcqWn|Q9@g3If#Pj~rlv(csU@-!_MQ;Hh)h}CjRT3m+Iq|$ZkIM
zt^nztj*Bf<1RP7a6+e=ZR?<#z?Vg8lHn6q{0494%nw*uc#6`;Z1n)Z5^p
zs=@PvNYCG2*T_Y`$e>!hi|i12$2-VuQ>nloCBAf*XM5_fU6i7-YDw^w4<3zp>hbXl
zai2e9`zwqTqV`XMuTF}zH^=K8CL>9mu?DjO-L{)kJDzfe6qdd$c8JpK0k9&I8vm@`idZdy^4UTN&l!measy)xxp=!X;CNp24R)47ISI>nBw_R%h7Hm|vW&iwSLHiTn(rI^w^D
zmhDka`*upPb8*=2hP~x7s>wpnd$O_dJb#Lp{9B!gnK_y#S6smFrP<}lR`@)+c6X8f
zBv(85Y+g^uS{fheYX=EFp^-u4z#GxcCj{h&7i|lmLDj-yCZpE(l^sW=dm(|!{giuE
zc=l=`MBG{_J`%+&4<9~!`t)h7%OY8Y*`Z(vh#|s4LsvHf<-$rFZRtDfE@nbrL%w1k
zr|=>SwcF@v)r(&xsJn`*S7{Uk|tIfdBpaY(1ZHcSIpL{Vq47Zz1=l=C;
zT~ps!#luodvRAn|`G?lD^Q^3zDeXmg__?I1an%4$`YNT&~VPr>qJk^YqY(c#XGWpX3nPLZGK|E?T
zY}3R<1*Ry60X80XEX&l?^d|(7QT=!-MxXMe0)h96`z#v9?_~33q7_6{d~}7G`)!d?c&QRuyZ3sFXKD9+;0(*P)aFl0#gp&s*jJZEOzxTaF
z-4%g*^8N>No7P^75}jC5JSu_DDGAxL_V=jn%5C8$UB`5G`}1=rC*$Cis
z&~I|NMDq7&QY7!xKikIuB{Sk#E``szrZ>9Yf&rR
zw2)9>kvx8_MDkhAe!((nb174?jxVj5&SA;@uJK|R3$Ndo!z(wbQ1Jc_qsj#ytM8Z&^qdmqzBr?6_#T(c)J^y%;H=)~
zrvYHP@APF&L!_xTg1awm1P?BP0MNmZ^r%K(QiB^XC?Fmqol?^uM#rnOg*
z?TPLDsgs9?hk~kd@7jUJyLYmdmI6ovByg!=y@K8Mu6xH@li1+Vw*bN`Dk}Q*?AMxC
zD?1z47qg{~Ysob-VLO*@3*YZmR@ig(h%fbAFVCt_Y@GOE7K|dS;Qq&?#BA)(iQ~;F
zH7^sX2fq~{7^S%5e;HyuoSt4wlx_n(fIhtcDlhoLJ51tcVpjk1bZaRU
zAt2Q~T9nw)6iLM!*;9XS|Mv3H*(_^Y=e4hcw~C4mr48jVzX}UY?$1|Xo$igMf%4*d
zSE_w}DrrCT6e=FjKf+ZNS{ew76Dn=eh5R}@Dv`;vK`-w8^u^z9h5Ve~3myY6-C~5I
zTE{~Y)dUaeDTVZ#1MumBE}6HdUGH~o9=XHUwSP}k#(MkZPWh^xpa|)o
z;(BhH(Xp_KEY0;pcqJj4y^-&3@5=W=mMb0TFaeXM=gIcyRx6-Vf3q?9Ir!e<#j4ZLA%AoLv5|
zb+T#z8U@|{YHJsGabbO2X5Gd;mf(Ps>cyO$fr{qG
z8B(s#VM~5O6wZc1O3Rd?a4g`R`|pg8es?K|g8)e{`(KY87*s2HpuXM|Mkyt?C-;%_
zFwE}_o~#3O0epir!4D$iC=|g7G_zT`ss{85Y}8oh`&}V9A%_RlA-`YY(TIkorZNGc
z3Mafbt>~u3d&{~RJm*j*kQHCbKi&3P8xG^AvrLs_$fW$nhff_#DEwl
zc0?**tpQ5*t=r4{534l4&1b%TVI_>g3UPY@&@)V(HysJJFGJx*0%C{zZ!mzd^7`Es
z@N%4Sa&TA-|9Cc*)f$SQ4f|fKS}rFcLiIP8KHi^cMAq0T-QnpTTA?!(={`^Fn%NKg#o+ewlO5or76z#prxuB+^Pw
z`90lGce1fvDD8zhw!V$B|>mAiy?iw65{!f3I6*`o_id-rI-aq9&$Yat`z_
z*~nQa1r=S9F~@Og=RwjrY9%&}CuiGZ6gy-SZ`%XB?11$J46SkkW>uyr)XaMxj1~{c
zZE)mR{PT-!hcbNO>0w#4Pt|z(m`cZ$OG|@6p(Z+~!|cE8wy-h1(inck+xd!#vVZ%)
z?Pn30$7BOoFyNQQT%$wy=oS;VPndlL}%k-;m!aM
zZQXvZdRy>G3Te>@TOgr!}kD)x+d5Mvv
zAX%+R=`x-3FzEEBFGl^SJ-l+p?3LWM?zBa#_v~|QcVz10YNXSr9*f&OGlaITp)F0L
z`sgU{x9f{poXE*eb>0X9TJpWo*t2KP$d0zimUI&^+1R;le!4muGnsg=-jb3Tlhi^V!XYP;hTxW4x7yqUp`o0@%%-RxpL4b_-3ee40?SA!e|=6DpCF(@XpLMycpw_KKyZ1VWkfaJ3
zZ2#tqWAr@4TvW+jk%GO{fy=P|)23NaFDa#(BvwCPQ<}ikgf&pR`oP4RaaC*yh>&O=9$cOVT1C&EXpuR
z;nSvR0Ki)TB*sF(%`-gOS%i2+%yIoYm-bs+031E)!iEGXPB;2(Fd*xermO6}hoYeF
zZ}2I*mN#fx3?xXTN?Iu9@~glvx#RTZnQ#kwIH&e1}k*Z0>3;_lOX
zI_lb_3lo*L*ZvSN4Y+Z}05){J21ts8qD9r|y#`x(G*-Wn6_$%VZ!zlMQxHljsVQ0h=}+PO7{tc=5ZNj(GG|*KaoERt7jmm}zguIS9N8Is
z;VjAd6^OyBL~B5~%eoGk;n(*S!wCVmI{5Ky`nf94lSdYEZ=PC12iFeX>GFEDqS6dz
zFjOX)NMkLsZ2jc~Nd!p|LmSl+)$=={(Qqs^dC5mS-J88DUaKB~UA6_^3R5-M;w-Y9
z=IX;$vqc2`X8By2*KKbvyD*L2`_pD=)M8>{C?x&(XvBS($RBANYmM}iK@cZ1X8rx(
z>rB#0!#XKg=xVao$jJTKTmmN_)>-Q7#sKytjnJ204j^_{Z};c!^3|np#^&A?)QyWz
z&P_3?Nu4|eX!3?;Lz(M+%zkF(Y1`5|@@z0-Ij@{mu%m@|3O*hc$0>4FQtXY`Xlm
zjhv+Y>UuG20#kHq*wu)ycnk43cNrLr!w{4S4u7mJS$ypyFSwQOc>W9gtk29edvWfN
za@Ry+hL@({ztZ&Uy^D=22s=jtNX-9-zuot*!RQf{fCZPTCTjs{vunXaLTRLCV_6Mg
zS#K&~W0eOUC08H)bM3b<^bwX&-*N<0vIOf~)t=J+afnsM%?j&S6p5hn}!{1gIpBSx6*WPh(
z@Yqhbg=55#i+@{0B)u7k>)DRRwggGtusZst2&kr*^C>u(vV@0+r>3U^$^08rQ`!g%
zAX4CjzZRuV@Z&Rlec!>s0q_wEJlF);_>+8m@be81ae(lY%_mAu(Rvh#9}}z0SCiD>
zETIhMDJm@dlv@8TW+Y}9jwqs+fn0~($2~l9_*wSsO$5G8pbJ5{9+W{)FaD
z6!@})VafjyEhhGh?CJ8+rGkPA1GOL<2RpNE)~5oCk&%(1g>x!^znxeLCCZny7jmyNI=gdAlgPE{lWHk
zlhlJ1kXb=7>;->8C~K8C=g9vwsZs!-9vO}KXp)rKJSFagP#?(
zd7|}_OTG-jS*CQco%)W#)ff_arhY{lIeO4SFy*k)<2QDw#L7T~HlDhbNa
zXXVqMGaKOIg~OjlFneM>8vo)6Sq1?2hK`&ynnKCIqdEp!Q?_*$b;ueddKSIxHR~q6
z)09JCtVZ@XA44SotP#L9pZd8sOE#fQ=X`%FKuz22CtL7`6-z^DfP_s>&In+ISfBGp
z3p)$2@$gt2tQI*TlcO!jkN1`!slS*EVJ@ncukSzfzer6_{hXSr03>b%C77+F;e*?~
zYezLw60{|n{G9{&%MHQbcaU|JszNPphfI#zDdIeZ|U885Rs!j1KD`R5KA}n46Y<-Q~>ns>K9rRSg
zT>i#LI9~)+-AS}8I>B8!V$&I}k_-F#Mj)Pthh@k2
zo{~O0+TLl_nUlV$y~N1aaTZIHVP)>ebbedtv%zaYy8{zp#)&mjKFCkzv8
zOE;pJp)Mxk6E#-7_azgh535H)2cexe-E%1G{G@X-6Q#qt{ebqr1p+lzX-J|zm)r}=
zN-duH!_ABjs0StR2o#o&k)-{?me+Fqpf+!2aWsV>&Vx4(UO$(l(
z-7E`;2TC48XXFD$(`RW2UVr6_r1cTSiB<&iGd|0{N5Zai3f1+gg?$eU8x&aC`IGwk
zWCM0f9Jz2!=droXOj~9Af~D!?X$|h!CNC9K
zjy-2B8p>JQ-mkxROqtTWD}XMi%g`mx)EsI5gb#yFhBcP2_CU;e%OR29i0(@=F*R{X
z1(FKZUTn7C$-@L$-FH6IbncfQx?RXXR%gHFBE^bPxl`e#A}aS91p-4RX|KGuMKtx_
z$7~|1{
zunGe}1Q6Wy7w@vth)m3JVMm#5j+Io@jFxLVGfk$_cP41M0b4tR?bzF
z7`h=*M1r#&&Jsp-b!-3!!=({3xkX>#OeOB~J9KcfsJbbtNEw`*o!$GxW^}gEnF^SN
zlEAZFKPlQ;0a%!WxuB|~8d>RZv>!@M0On-b+p;u_Y>Xd+XV11-U)c!DEr5vEUhgi>XHdXrhN;lQkCo{`e
z5oeSx(8jRv?DB=vBog(_gH4%ywzh4m=nNwBjS{Haa49Q)%Rk_yyR<}YC6^Ao`+_6d
zN?juP&mlxCxa0l&b0r6Nas?)CxFdSuBY9nM`xqF>of$CUFm+ra=W*3d&0ibaK!IV_
zli-SeQ8beL)kNH|l@h+=koX<;_r@|GnvFaz{O1I7#9dwhfVeMg=lnwUq9*4K9cb)3
z(Hw`Ci2{bk@3g?b;5N$0rJvm}+WgS`)8h-5xyKSTcdv5XRw*RU_Y!fY{f${tMz6*L
zcckLnp6}hP-5wsfJ+L!#qY>2YFMwFd;n2z4J+f1kY&)b54w*yzO!wy=F@$rfOASGM
zTK2LKTV;t~8_%@8<71I6fAn^n13WIob%S@O^)#p7n+E-IQR4=2Jwr5rOM;Eyl0LyI
zZw)xl9oMZWyOCtZ253jDz9GZbPXJ;DS_nR<;L_5ht=cQn7_ZGH|{$!
zMb)%B9e4rUmM^Zwq>z(~W7Q1Nsnt9UHue@glRoa_Im7v!)}9+l8yU)p4F{9|>>guSTFa{?diSYin6(+6eLXOkbE;j+SwY3H{
zmdr*1f(F&*_gGlk1I}1DC8#NmV+&K7a)1Bm%A
zuP`aTcghS$AH)s)4KfMyE}d&Gj#k=0J*W8X`{V}yL}90ev35eK?l`ke;mbz*ErEmz
zJMr<$aXao8k9t+Ck4TP!UsAL7;Hb{+%@LVMMc4YOQlZ@NyCl-%Joo<1b0CmYAzdWs)
zTy}9YgqzcLzx(*@OI(vl5C8q{O(b#D{o(7Cwn=;N&U53o=qsMwv0wyf$Tb^Abf=A<
z|G{#)!nChmWU+tIf{*dC9y+E)#$$?bCR5os@a?d*a=G^`?aR@T=diE$@-fNaa%rz9
z+s4NdGponnZI|hV+i%ziJFiGCKCqd8ZujOk3lfO6Adx#7rvV>)rsG%B|!j(_&8RC%6R4*WI1x$TnoqX;iFz)XMK@<^A4ZI)8@djrhB&
z{&7diCe{U?E$1=g)`lRO0w=@PI9`z9;3twq`ky3s!j9IPpJRz&xPq6#kPP$@)`w))sc(c03QeD{AGV!%qg3c%iQJ
z`3@&gbunr6vK;F2BaJNnoS6n+xQP8dS&rQv=mT|foosowq;i+UZB!@rC(6CaSf|_#
z&?Owq8SPCJ80jU53*=SiuQmpF?+wiFfyVwUO$&HwfW&?5<=qw8JE5p9UcpU|p^T&^
zlthQPiI(BW$HbcyfPS86@m9t<-d0u8A}$A+jUGd6`MkbkB&Ya-Dx@!WIMt+Y*&L5t
zDwnFG31x0#{*u`VH1G8|-us?u>yk#*+5?zjWk?{ch;rBC_X
zjY@$6a~ifqz5}4*@@z|SkN`bfD-`=yIlazp*kd5R)Wq@w+^EsxDwr?@aLS8PEt3UC
zc5N<%K$*2z58+bpIA83sqb_K7G^>r!FgEO2{tyhX{ER>%K9<`Vwc|Wb!C(I;d>)%N
zB)NZ7_j{4He85=@!V0Lhjpe~UA0NLH_g!!|DQGH8{afP;Zp9K^X!b%f0A1dNO;Ev2W=Ka`?YwH%`-^a9>??ULB&Cueq;$G4zaoZCFu>Iqcvi7
z8|HY5o0QfbfA1z>&1<*GP>0Gt{J^~Ba%^kVwPSOAW(l?4;_5vQt=kj~8hZIdYn!%y
z;j<9z{ZNo^2|w*xeWLVc8meLwh$ngpCe`*^JU7qg5rUhaCt8GP=3tc`S0@cny3Nbo
zv)e%kNj&B8_IxbxQ%0?v;jb-jPk+lq$*s-gDvjO;O1c6;L-gC
zwR}XysX=E+?5D3xd7|_cZs?*^>tu-@q2rm0khNKP#8}DmtNUY_ol%tr^9oAGo_!Vv
zOOxI@##%kchqc*^5%|S7YRtOZW&a2uICbv^^wzug0cPL2puNN%$&f2{LjYY*T$u5i1_e`&T3tlT~0dyYX2W0qD<-#8Av=&Cy5wzkXReYS$^K
z$Aii%gbxpoDpXbftjp6C-PsTx^nLVoL#FBNcTMuTPo^>s8C~Ivs5d8}Z^l>B^dPHc
zoY3VKiMw%51WJwS(0><>D6?)Wi?ugrBsBJBHJXdts8p9;xs`Bb>)e|;?+j$EyS)PY
z9eiv~TbqHOr@}hSdM`v|dPdb{hHUgqyLCiR-Be~l<1~YWrM>@b(t^0#S!MN$ar
z*P08Pb=+|co;qVc@cY9na(Gfpvy`8Ga!?-B#z3O5c5An2vsK$)q2HJxCKEi~Fi2P$
zq*;Y-%Hyn=5vvcHLpXV>M>j#M&JsW+NCW^sjHWgQTrY(hi=Hb{sOvq@Avy
z5FIwdtIayQn-`IJ%#$86?o`cJ<6zOF#gYsmSDweV!K0O%eFaNL(;4cTY24!aC#+53eBA=G{RXW9Jd_P>Dn@b94-EGrEI}^CLm|$ZC=Voz1d8>>ce4
z4APt;z3%CM3Qeqi!FsETvNxx>yWHmdb?+kqq%ZMkBOQt~UIc9si8JX!ST+AEBFu~<
zryF%sF3N;)V`Df*$r~Jg_ROK22&hD`LYp6h^;z&Gzad)=QQ*ZBSG@|^TaUZk+@D_(ton{!z
z8;-o2Z8BE`)%bTvR*6&sFZh(IL}zm$wREju|&)Q#b#96qu9xEdf_uBlMFSuUmub^#RUi`$J$8
z+9HgG81K#?a`2`G){@*H>c|8NXXnN6$*twAQZfL(dK#V6DA}6S`uGL&8DflOB-KLZsL4AET2&t+KzCAgXy6pfYY8w}N
ziwibO=Kc;l0GG9)cSPG#dQL0}_`E045pK2V?L5;gY*u
z2J0pG&}!6bB7lUkH)^>Wsj%PlC^a2$!mrt$b`kVh@*h6-KsNbfhQGbAwywAQ4wMl8
zEJi;7$IC)u8!HVzubxeCaL1FDI?SrLiRJXYCgQUM`VAy`$aqR<1WAkR&Z?+E*Avv9
z#}5n}73_L1Pn;H;GVKd}4ph8#Os;#SOROS&7Q9Cr0AjQ#z60e2Uk55p+xc>m0Riet
zN~dZZi!Z;OTA2Db!>QV)eSQz;xGyZ{hCvtNWVQ&s?x$BaxKGBDef~Y`KIbdrC)q2`
z1b9IIDOmN=HDYli8~%!?bA|Gewv=}M0$u`tSQpb{p=4d0q8
zVq_0Imor;?G`W2q+x)AZzsc7yyW;Jn|JtMthH=mgiLuDKs9_1ys#oCFl;exDyDN)`
z3)D+#*Y-PAAncD`R$iB=#Yum|ekRsQJUQ<Ya4nl
zq*68Clv!b9oNWbieQT_+%UCHf8AaQF!9zB=V`qc)^#4>;Snt^{HR9iXI=2JIrdL(_%#8H(k8*-`7lRRy
zP2n`6^;i0yoBfzSeNMM?9s)&`_n4iME@Wh?;$)Ep?~6Xhvg9gCU*B9c#!(360BE|L`vdaHhnJxgb(`+9P5iBr~)+SfyM}EZ_GI_z2(YIxg
z>%7}Y5{wKz+RyqcJCkY*{mIK;-Ud)oR2tP5MXc4h+{hIhyOE;jb>JO`@G}{plGmdt
zPun=^UZv=v#ZUsFzFUCxblnC!0K>QWU|s+r`N6l>b_ethTnjp3OukT5Q_2THOjRUT
zH3s4-Q~_IepK=Qx+B}*7>e~sKqIqu!XcfTna9+3Z^CtH9i49P*^jf57
zg)9B?Xl^=fSUqj-2jd
zLk>@l>B@KN0L7yJB@@3_lzrJFj4rC-k!lG`cPx5Ixw2}9H}qt8v8#xIzM+K0-e
zgeDSyN7!w4N7UJ8K^U8bz;d<*Ndt5N0F_i*@>ypV+7gmT{2ZS^NvrJMdn|n_DN#f$
zq>m>%-cVuT|9OMh&9$Fri62J`?u$A5cGg+g5@@P&`=lT$t7$>oJP*@M
zg+vnJYiqqz_*M}I`1?87_DdL#l>MYQCd^7cyl3FuP3z}ijgkcP)?}T0f$OuSZ%8oH
z`FpbS6foIcZ%OldDoC(E*s=yC-!9ykv?gb%ij_UrKvnsHpg?}o2K@fav=V6T9{Kn?s`0l%7Oyt%SCck?_NYzND6G$-SvoIb&
zBlCerQG&J