diff --git a/.cursor/mcp.json b/.cursor/mcp.json new file mode 100644 index 0000000..5cb4a44 --- /dev/null +++ b/.cursor/mcp.json @@ -0,0 +1,25 @@ +{ + "mcpServers": { + "bigquery": { + "command": "npx", + "args": [ + "-y", + "@modelcontextprotocol/server-bigquery" + ], + "env": { + "GOOGLE_CLOUD_PROJECT": "${env:ATLAS_GCP_PROJECT_ID}" + } + }, + "dbt-atlas": { + "command": "${workspaceFolder}/.venv-dbt/bin/dbt", + "args": [ + "--version" + ], + "env": { + "DBT_PROJECT_DIR": "${workspaceFolder}/dbt/atlas_dbt", + "DBT_PATH": "${workspaceFolder}/.venv-dbt/bin/dbt", + "DBT_TARGET": "bigquery" + } + } + } +} diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..916da1d --- /dev/null +++ b/.env.example @@ -0,0 +1,20 @@ +# Atlas production-template configuration +ATLAS_PROJECT_NAME=atlas +ATLAS_ENVIRONMENT=dev +ATLAS_GCP_PROJECT_ID=example-gcp-project +ATLAS_GCP_PROJECT_NUMBER=123456789012 +ATLAS_GCP_LOCATION=US +ATLAS_GCP_REGION=us-central1 +ATLAS_DATASET_PREFIX=atlas +ATLAS_RAW_DATASET=atlas_raw +ATLAS_DBT_DATASET=atlas +ATLAS_OPS_DATASET=atlas_ops +ATLAS_GCS_BUCKET=atlas-raw-events-example-gcp-project +ATLAS_RELEASE_BUCKET=atlas-releases-example-gcp-project +ATLAS_SERVICE_ACCOUNT_PREFIX=atlas +ATLAS_DAG_ID=atlas_batch_pipeline +ATLAS_SCHEDULE=@daily +ATLAS_NOTIFICATION_EMAIL= +ATLAS_COST_CEILING_BYTES=1000000000 +DBT_TARGET=bigquery +DBT_PATH=.venv/bin/dbt diff --git a/.github/workflows/atlas-ci.yml b/.github/workflows/atlas-ci.yml new file mode 100644 index 0000000..5ca04d6 --- /dev/null +++ b/.github/workflows/atlas-ci.yml @@ -0,0 +1,170 @@ +name: atlas-ci + +on: + pull_request: + push: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: atlas-ci-${{ github.ref }} + cancel-in-progress: true + +env: + PYTHON_VERSION: "3.12" + +jobs: + atlas-security-shell: + name: atlas-security-shell + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd + - name: Set up Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 + with: + python-version: ${{ env.PYTHON_VERSION }} + cache: pip + cache-dependency-path: requirements-ci.txt + - name: Install validation toolchain + run: pip install -r requirements-ci.txt + - name: Dependency-file sanity + run: | + python - <<'PY' + from pathlib import Path + + for name in ( + "requirements.txt", + "requirements-ci.txt", + "airflow/requirements-airflow.txt", + "dbt/requirements-dbt.txt", + ): + content = Path(name).read_text(encoding="utf-8") + assert content.strip(), f"{name} is empty" + print("dependency manifests present and non-empty") + PY + - name: Security and shell gates + run: bash scripts/validate_ci.sh --mode static --group security-shell + - name: Upload gate results + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: security-shell-gate-results + path: logs/ci/validate-ci-results.json + + atlas-python: + name: atlas-python + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd + - name: Set up Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 + with: + python-version: ${{ env.PYTHON_VERSION }} + cache: pip + cache-dependency-path: | + requirements.txt + requirements-ci.txt + - name: Install locked dependencies + run: pip install -r requirements.txt -r requirements-ci.txt + - name: Python gates + run: bash scripts/validate_ci.sh --mode static --group python + - name: Upload gate results + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: python-gate-results + path: logs/ci/validate-ci-results.json + + atlas-dbt: + name: atlas-dbt + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd + - name: Set up Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 + with: + python-version: ${{ env.PYTHON_VERSION }} + cache: pip + cache-dependency-path: dbt/requirements-dbt.txt + - name: Install pinned dbt environment + run: pip install -r dbt/requirements-dbt.txt + - name: dbt static gates + run: bash scripts/validate_ci.sh --mode static --group dbt + - name: Upload dbt manifest + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: dbt-manifest + path: dbt/atlas_dbt/target/manifest.json + if-no-files-found: warn + + atlas-airflow: + name: atlas-airflow + runs-on: ubuntu-latest + timeout-minutes: 25 + steps: + - name: Checkout + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd + - name: Set up Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 + with: + python-version: ${{ env.PYTHON_VERSION }} + cache: pip + cache-dependency-path: | + airflow/requirements-airflow.txt + requirements.txt + requirements-ci.txt + - name: Install pinned Airflow with official constraints + run: | + pip install "apache-airflow==3.1.7" \ + --constraint "https://raw.githubusercontent.com/apache/airflow/constraints-3.1.7/constraints-3.12.txt" + pip install -r airflow/requirements-airflow.txt -r requirements.txt -r requirements-ci.txt + - name: pip check + run: pip check + - name: Airflow gates + run: bash scripts/validate_ci.sh --mode static --group airflow + - name: DAG tests + run: PYTHONPATH=src:dags python -m pytest tests/airflow -q + - name: Upload gate results + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: airflow-gate-results + path: logs/ci/validate-ci-results.json + + atlas-ci-gate: + name: atlas-ci-gate + runs-on: ubuntu-latest + timeout-minutes: 5 + needs: + - atlas-security-shell + - atlas-python + - atlas-dbt + - atlas-airflow + if: always() + steps: + - name: Require every job to succeed + env: + NEEDS_JSON: ${{ toJSON(needs) }} + run: | + python3 - <<'PY' + import json + import os + import sys + + needs = json.loads(os.environ["NEEDS_JSON"]) + failed = [name for name, value in needs.items() if value["result"] != "success"] + if failed: + print("Failed or skipped required jobs:", ", ".join(failed)) + sys.exit(1) + print("All required Atlas CI jobs succeeded") + PY diff --git a/.github/workflows/atlas-deploy.yml b/.github/workflows/atlas-deploy.yml new file mode 100644 index 0000000..f798322 --- /dev/null +++ b/.github/workflows/atlas-deploy.yml @@ -0,0 +1,105 @@ +name: atlas-deploy + +on: + workflow_dispatch: + inputs: + confirm: + description: 'Type "deploy-atlas-dev" to confirm' + required: true + target_sha: + description: "Commit SHA reachable from main; empty uses main HEAD" + required: false + default: "" + create_composer: + description: "Create the ephemeral Composer environment if missing" + type: boolean + default: false + leave_paused: + description: "Leave the DAG paused after smoke validation" + type: boolean + default: true + +permissions: + contents: read + id-token: write + +concurrency: + group: atlas-dev-deployment + cancel-in-progress: false + +env: + PYTHON_VERSION: "3.12" + +jobs: + atlas-deploy: + name: atlas-deploy + runs-on: ubuntu-latest + timeout-minutes: 120 + environment: atlas-dev + steps: + - name: Verify typed confirmation + run: | + if [ "${{ github.event.inputs.confirm }}" != "deploy-atlas-dev" ]; then + echo "Confirmation input does not match deploy-atlas-dev" + exit 1 + fi + - name: Checkout main + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd + with: + ref: main + fetch-depth: 0 + - name: Resolve trusted target + id: target + run: | + target="${{ github.event.inputs.target_sha }}" + if [ -z "$target" ]; then + target="$(git rev-parse HEAD)" + fi + git cat-file -e "${target}^{commit}" + git merge-base --is-ancestor "$target" origin/main + git checkout "$target" + echo "sha=$target" >> "$GITHUB_OUTPUT" + - name: Set up Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 + with: + python-version: ${{ env.PYTHON_VERSION }} + cache: pip + cache-dependency-path: requirements.txt + - name: Install and validate + run: | + pip install -r requirements.txt -r requirements-ci.txt + bash scripts/validate_ci.sh --mode static --group security-shell + bash scripts/validate_ci.sh --mode static --group python + - name: Require WIF repository variables + run: | + test -n "${{ vars.ATLAS_WIF_PROVIDER }}" || { echo "Set ATLAS_WIF_PROVIDER"; exit 1; } + test -n "${{ vars.ATLAS_DEPLOYER_SERVICE_ACCOUNT }}" || { echo "Set ATLAS_DEPLOYER_SERVICE_ACCOUNT"; exit 1; } + - name: Authenticate to GCP + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 + with: + workload_identity_provider: ${{ vars.ATLAS_WIF_PROVIDER }} + service_account: ${{ vars.ATLAS_DEPLOYER_SERVICE_ACCOUNT }} + - name: Set up gcloud + uses: google-github-actions/setup-gcloud@aa5489c8933f4cc7a4f7d45035b3b1440c9c10db + - name: Ensure Composer environment + if: ${{ github.event.inputs.create_composer == 'true' }} + run: ATLAS_APPROVE_COMPOSER_CREATE=true bash scripts/manage_atlas_composer.sh create + - name: Build and upload immutable release + run: bash scripts/build_deployment_bundle.sh --upload + - name: Deploy release + run: | + flags="" + if [ "${{ github.event.inputs.leave_paused }}" = "true" ]; then + flags="--leave-paused" + fi + ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas_release.sh \ + --git-sha "${{ steps.target.outputs.sha }}" $flags + - name: Upload deployment evidence + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: deployment-evidence + path: | + dist/release-manifest-*.json + /tmp/smoke-warehouse.json + if-no-files-found: warn diff --git a/.github/workflows/atlas-integration.yml b/.github/workflows/atlas-integration.yml new file mode 100644 index 0000000..d72667b --- /dev/null +++ b/.github/workflows/atlas-integration.yml @@ -0,0 +1,76 @@ +name: atlas-integration + +on: + workflow_dispatch: + inputs: + target_sha: + description: "Commit SHA reachable from main; empty uses main HEAD" + required: false + default: "" + +permissions: + contents: read + id-token: write + +concurrency: + group: atlas-integration + cancel-in-progress: false + +env: + PYTHON_VERSION: "3.12" + +jobs: + atlas-gcp-integration: + name: atlas-gcp-integration + runs-on: ubuntu-latest + timeout-minutes: 45 + steps: + - name: Checkout main + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd + with: + ref: main + fetch-depth: 0 + - name: Resolve trusted target + id: target + run: | + target="${{ github.event.inputs.target_sha }}" + if [ -z "$target" ]; then + target="$(git rev-parse HEAD)" + fi + git cat-file -e "${target}^{commit}" + git merge-base --is-ancestor "$target" origin/main + git checkout "$target" + echo "sha=$target" >> "$GITHUB_OUTPUT" + - name: Set up Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 + with: + python-version: ${{ env.PYTHON_VERSION }} + cache: pip + cache-dependency-path: | + requirements.txt + dbt/requirements-dbt.txt + - name: Install runtime and dbt toolchains + run: | + pip install -r requirements.txt + python -m venv /tmp/dbt-venv + /tmp/dbt-venv/bin/pip install -r dbt/requirements-dbt.txt + echo "/tmp/dbt-venv/bin" >> "$GITHUB_PATH" + - name: Require WIF repository variables + run: | + test -n "${{ vars.ATLAS_WIF_PROVIDER }}" || { echo "Set ATLAS_WIF_PROVIDER"; exit 1; } + test -n "${{ vars.ATLAS_INTEGRATION_SERVICE_ACCOUNT }}" || { echo "Set ATLAS_INTEGRATION_SERVICE_ACCOUNT"; exit 1; } + - name: Authenticate to GCP + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 + with: + workload_identity_provider: ${{ vars.ATLAS_WIF_PROVIDER }} + service_account: ${{ vars.ATLAS_INTEGRATION_SERVICE_ACCOUNT }} + - name: Set up gcloud + uses: google-github-actions/setup-gcloud@aa5489c8933f4cc7a4f7d45035b3b1440c9c10db + - name: Run isolated integration test + run: bash scripts/validate_gcp_integration.sh + - name: Upload integration results + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: integration-results + path: logs/ci/integration-results.json diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..c74f3f1 --- /dev/null +++ b/.gitignore @@ -0,0 +1,52 @@ +# Python +__pycache__/ +*.py[cod] +.pytest_cache/ +.mypy_cache/ +.ruff_cache/ +*.egg-info/ +dist/ +build/ + +# Virtual environments +.venv/ +.venv-*/ +.venv-dbt/ + +# Local configuration and credentials +.env +.env.* +!.env.example +credentials/ +secrets/ +.gcp/ +service-account*.json +*-key.json + +# Generated pipeline artifacts +data/*.jsonl +data/runs/ +logs/ + +# dbt generated state and local profile +dbt/atlas_dbt/target/ +dbt/atlas_dbt/logs/ +dbt/atlas_dbt/dbt_packages/ +dbt/atlas_dbt/profiles.yml + +# Airflow local state +airflow/logs/ +airflow/airflow.db +airflow/airflow.cfg +airflow/webserver_config.py + +# Terraform local state +**/.terraform/ +*.tfstate +*.tfstate.* +.terraform.lock.hcl + +# OS and editor +.DS_Store +Thumbs.db +.vscode/ diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..6c1760f --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,15 @@ +# Contributing + +Use a branch → pull request → CI → review → merge workflow. + +Every material change must include: + +- declared purpose and affected components +- tests or an explicit reason tests are unchanged +- documentation for changed behavior or operations +- migration and rollback considerations +- no secrets or personal data +- evidence that the canonical CI entry point passes + +Generated code must be read, explained, modified where necessary, and tested by +the contributor. diff --git a/README.md b/README.md new file mode 100644 index 0000000..4ff9bea --- /dev/null +++ b/README.md @@ -0,0 +1,100 @@ +# Atlas GCP Production Data Platform Template + +Atlas is a reusable, production-oriented batch data platform foundation for Google +Cloud. It combines Python ingestion, Cloud Storage, BigQuery, dbt, Airflow, +credentialless CI, controlled deployment, observability, recovery, governance, +and cost controls in one repository. + +This is not a claim that cloning a repository magically makes a system production +ready. It provides enforced engineering defaults and operating artifacts that a +team must configure, validate, deploy, and own. + +## Architecture + +```text +Source / synthetic events + ↓ +Python ingestion → immutable Cloud Storage objects + ↓ +BigQuery raw tables + ↓ +dbt staging → classification/quarantine → facts/dimensions/marts + ↓ +Airflow orchestration, audit, retries, backfills, and recovery + ↓ +Logging, metrics, alerts, governance, lineage, cost guards, runbooks +``` + +## Included capabilities + +- Deterministic sample event generation and idempotent batch ingestion +- Partitioned and clustered BigQuery storage +- Governed dbt layers, tests, contracts, and incremental processing +- Airflow DAGs with stable batch identity, retries, auditing, and backfills +- Credentialless pull-request CI +- Optional keyless GitHub-to-GCP delivery through Workload Identity Federation +- Immutable release bundles, migrations, smoke validation, and rollback +- Structured telemetry, metrics, alerts, dashboards, and runbooks +- Failure injection, recovery auditing, schema compatibility, and cost guards +- Ownership, lineage, consumer-impact, retention, security, and evidence controls + +## Start here + +1. Read [`START_HERE.md`](START_HERE.md). +2. Copy `.env.example` to `.env` and replace every example value. +3. Create a Python virtual environment and install dependencies. +4. Run credentialless static validation. +5. Run the local sample pipeline. +6. Configure an isolated GCP project before any approved cloud mutation. + +```bash +git checkout main +git pull --ff-only origin main +cp .env.example .env +python3 -m venv .venv +source .venv/bin/activate +pip install -r requirements.txt -r requirements-ci.txt +export PYTHONPATH=src +bash scripts/validate_ci.sh --mode static +python scripts/generate_events.py +pytest +``` + +## Configuration + +The template keeps the `atlas` reference namespace in code and sample assets, +while cloud identities and runtime resources are configured through environment +variables. See [`docs/template-configuration.md`](docs/template-configuration.md). + +Never deploy the example values. Configure project IDs, buckets, datasets, +service accounts, notification channels, cost ceilings, retention, and schedules +for the adopting environment. + +## CI and delivery + +Pull requests run credentialless validation. Trusted integration and deployment +workflows are manual and require repository variables for Workload Identity +Federation: + +- `ATLAS_WIF_PROVIDER` +- `ATLAS_INTEGRATION_SERVICE_ACCOUNT` +- `ATLAS_DEPLOYER_SERVICE_ACCOUNT` + +The bootstrap scripts are plan-first and mutation-gated. Review IAM, cost, and +cleanup behavior before applying anything. + +## Evidence and limitations + +The original Atlas reference implementation was tested with synthetic workloads, +clean-clone validation, CI, controlled cloud deployments, failure drills, and +operator handoff. Those historical reports remain in `docs/` as engineering +evidence. They do not prove that a new adoption has passed the same gates. + +The reference release lineage includes `atlas-sprint-3-complete` for the orchestrated platform milestone and later Sprint 8 handoff evidence. + +A new deployment is complete only after its own CI, isolated cloud validation, +incident drill, recovery exercise, security review, cost review, and handoff. + +## License + +Apache License 2.0. See [`LICENSE`](LICENSE). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..0a401e6 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,11 @@ +# Security Policy + +Do not commit credentials, service-account keys, API keys, OAuth secrets, webhook +URLs, personal notification addresses, or production data. + +Use Workload Identity Federation for GitHub-to-GCP authentication. Keep pull +request CI credentialless. Apply least privilege, plan IAM changes before +mutation, and review the security and identity documentation before deployment. + +Report security issues privately to the repository owner rather than opening a +public issue containing exploit details or credentials. diff --git a/START_HERE.md b/START_HERE.md new file mode 100644 index 0000000..f38b9d4 --- /dev/null +++ b/START_HERE.md @@ -0,0 +1,57 @@ +# Start Here + +Atlas is a production-data-platform template, not a one-command production +service. Begin with the route matching your responsibility. + +## Adopter / platform engineer + +1. Read `README.md` and `docs/template-configuration.md`. +2. Review `docs/reference-architecture/architecture-invariants.md`. +3. Replace all sample environment values. +4. Run `bash scripts/validate_ci.sh --mode static`. +5. Exercise the local pipeline and tests. +6. Provision an isolated GCP namespace using plan mode first. +7. Run one batch, one deliberate failure, one recovery, and cleanup. +8. Record environment-specific evidence instead of inheriting the reference + implementation's claims. + +## Operator + +Read: + +- `docs/handoff/operator-onboarding.md` +- `docs/runbook.md` +- `docs/runbook-sprint3.md` +- `docs/observability-runbook-sprint5.md` +- `docs/recovery-runbook-sprint6.md` + +Be able to answer: Did the pipeline run? Is the data correct and complete? Who is +alerted? How is it recovered? How is recurrence prevented? + +## Reviewer / architect + +Start with: + +- `docs/reference-architecture/README.md` +- `docs/reference-architecture/system-context.md` +- `docs/reference-architecture/interfaces-and-contracts.md` +- `docs/reference-architecture/security-and-identity-model.md` +- `docs/reference-architecture/reliability-and-recovery-model.md` +- `docs/reference-architecture/unresolved-risks.md` + +## Coding agent + +Read `docs/handoff/agent-onboarding.md`. Treat generated code as provisional. +State assumptions, risks, affected files, test plan, and rollback considerations +before major changes. Do not claim production readiness without environment-specific +evidence. + +## Credentialless verification + +```bash +python3 -m venv .venv +source .venv/bin/activate +pip install -r requirements.txt -r requirements-ci.txt +export PYTHONPATH=src +bash scripts/validate_ci.sh --mode static +``` diff --git a/airflow/README.md b/airflow/README.md new file mode 100644 index 0000000..7aa7bf8 --- /dev/null +++ b/airflow/README.md @@ -0,0 +1,34 @@ +# Atlas Local Airflow (Sprint 3) + +Local Airflow **3.1.7** with Composer-parity provider pins for DAG development and +acceptance testing before Composer deployment. + +## Quick start + +```bash +cd Atlas-GCP-Build +source airflow/airflow.env.example # or copy to .env +bash scripts/setup_airflow.sh +bash scripts/start_airflow_local.sh +bash scripts/test_airflow_sprint3.sh +``` + +## Version pins + +See [requirements-airflow.txt](requirements-airflow.txt) and [ADR-005](../docs/adr/ADR-005-airflow-composer-parity.md). + +Target Composer image: `composer-3-airflow-3.1.7-build.12`. + +## Layout + +| Path | Purpose | +|------|---------| +| `../dags/` | DAG definitions and parse-time helpers | +| `../.airflow/` | Local metadata DB (gitignored) | +| `../.venv-airflow/` | Pinned virtualenv (gitignored) | +| `../logs/airflow/` | Run summaries and task evidence | + +## Composer mapping + +Deploy DAGs to `/home/airflow/gcs/dags/project_atlas/` and runtime assets to +`/home/airflow/gcs/data/` with `ATLAS_ROOT` pointing at the data path. diff --git a/airflow/airflow.env.example b/airflow/airflow.env.example new file mode 100644 index 0000000..7dfc912 --- /dev/null +++ b/airflow/airflow.env.example @@ -0,0 +1,18 @@ +# Copy to .env or export before setup_airflow.sh + +export ATLAS_ROOT="${ATLAS_ROOT:-$(pwd)}" +export ATLAS_GCP_PROJECT_ID=example-gcp-project +export ATLAS_GCS_BUCKET=atlas-raw-events-example-gcp-project +export ATLAS_BQ_DATASET=atlas_raw +export ATLAS_DBT_DATASET=atlas +export DBT_PROJECT_DIR="${ATLAS_ROOT}/dbt/atlas_dbt" +export PYTHONPATH="${ATLAS_ROOT}/src:${ATLAS_ROOT}/dags:${PYTHONPATH:-}" + +# Local Airflow home (gitignored) +export AIRFLOW_HOME="${ATLAS_ROOT}/.airflow" +export AIRFLOW__CORE__LOAD_EXAMPLES=False +export AIRFLOW__CORE__DAGS_FOLDER="${ATLAS_ROOT}/dags" +export AIRFLOW__DATABASE__SQL_ALCHEMY_CONN=sqlite:///${AIRFLOW_HOME}/airflow.db + +# ADC for local GCP access — never commit credentials +# export GOOGLE_APPLICATION_CREDENTIALS=/path/to/key.json diff --git a/airflow/requirements-airflow.txt b/airflow/requirements-airflow.txt new file mode 100644 index 0000000..e54035e --- /dev/null +++ b/airflow/requirements-airflow.txt @@ -0,0 +1,9 @@ +# Composer-parity pins for local Airflow 3.1.7 (verified 2026-07-14). +# Install with official constraints: +# pip install "apache-airflow==3.1.7" \ +# --constraint "https://raw.githubusercontent.com/apache/airflow/constraints-3.1.7/constraints-3.12.txt" +# pip install -r requirements-airflow.txt + +apache-airflow==3.1.7 +apache-airflow-providers-google==20.0.0 +apache-airflow-providers-standard==1.12.1 diff --git a/config/anomaly_profile.yaml b/config/anomaly_profile.yaml new file mode 100644 index 0000000..d3fe841 --- /dev/null +++ b/config/anomaly_profile.yaml @@ -0,0 +1,54 @@ +# Seeded anomaly profile for Sprint 1 generator and validation acceptance. +# Validation overall status is FAIL when anomalies are present; acceptance +# tests verify each expected defect was detected. +anomalies: + duplicate_event_ids: + count: 100 + description: Reuse existing event_id values to create duplicates. + null_user_ids: + count: 500 + description: Set user_id to null for a subset of events. + invalid_country_codes: + count: 200 + description: Use non-ISO country codes such as XX and ZZ. + future_timestamps: + count: 150 + description: Set event_date to a calendar date after the generation date. + late_arriving_events: + count: 300 + description: Set event_date earlier than event_timestamp date by design. + +valid_country_codes: + - US + - CA + - GB + - AU + - DE + - FR + - BR + - MX + - IN + - JP + +event_names: + - app_open + - app_close + - page_view + - button_click + - signup_start + - signup_complete + - login + - logout + - share + - purchase + +platforms: + - ios + - android + - web + +app_versions: + - 1.0.0 + - 1.1.0 + - 1.2.0 + - 2.0.0 diff --git a/config/atlas.yaml b/config/atlas.yaml new file mode 100644 index 0000000..ebd4aca --- /dev/null +++ b/config/atlas.yaml @@ -0,0 +1,29 @@ +# Project Atlas Sprint 1 configuration. +# Override any value via environment variables prefixed with ATLAS_. +gcp: + project_id: example-gcp-project + location: US + bucket_name: atlas-raw-events-example-gcp-project + bucket_logical_name: atlas-raw-events + dataset_id: atlas_raw + table_id: events + +generator: + event_count: 50000 + random_seed: 42 + output_dir: data + +ingestion: + gcs_prefix: raw + +loader: + staging_table_suffix: _staging + +validation: + expected_event_count: 50000 + # Future-dated rows use event_date > generation date (not clock-time comparison). + future_date_field: event_date + +logging: + log_dir: logs + log_format: json diff --git a/config/cost_controls.yaml b/config/cost_controls.yaml new file mode 100644 index 0000000..7519b80 --- /dev/null +++ b/config/cost_controls.yaml @@ -0,0 +1,39 @@ +# Atlas BigQuery cost controls (Sprint 7, ADR-020). +# Consumed by atlas.observability.cost_guard. Extends the Sprint 6 cost guards. + +version: 1 + +environments: + atlas-dev: + # Hard ceiling for a single ad-hoc/governed query (dry-run estimate). + max_query_bytes: 1073741824 # 1 GiB + # Hard ceiling for the whole bounded performance suite (sum of billed bytes). + max_performance_suite_bytes: 5368709120 # 5 GiB + max_backfill_days: 7 + full_refresh_requires_approval: true + # Assets whose queries must include a partition filter. + require_partition_filter_assets: + - atlas_raw.events + - atlas_core.fct_events + temporary_dataset_ttl_hours: 24 + temporary_object_ttl_days: 7 + composer_max_lifecycle_hours: 12 + log_retention_days: 30 + release_retention_policy: keep_validated_releases + + atlas-ci: + max_query_bytes: 536870912 # 512 MiB + max_performance_suite_bytes: 1073741824 # 1 GiB + max_backfill_days: 3 + full_refresh_requires_approval: true + require_partition_filter_assets: + - atlas_raw.events + - atlas_core.fct_events + temporary_dataset_ttl_hours: 1 + temporary_object_ttl_days: 1 + composer_max_lifecycle_hours: 6 + log_retention_days: 7 + release_retention_policy: keep_validated_releases + +# Override precedence: ATLAS_MAX_PERFORMANCE_TEST_BYTES (env) overrides +# max_performance_suite_bytes for the current run when set. diff --git a/config/failure_scenarios.yaml b/config/failure_scenarios.yaml new file mode 100644 index 0000000..722738a --- /dev/null +++ b/config/failure_scenarios.yaml @@ -0,0 +1,1107 @@ +# Atlas Sprint 6 controlled failure-scenario catalog (ADR-013). +# +# Consumed by src/atlas/failure_injection/registry.py, which validates every +# scenario against the required schema in CI. Scenarios NEVER run implicitly: +# scripts/run_failure_scenario.sh requires an explicit scenario id, explicit +# environment, and ATLAS_APPROVE_FAILURE_INJECTION=true, and refuses +# canonical batch ids (only the atlas-s6- prefix is injectable). +# +# execution_mode: +# unit — behavior proven by repository tests (path given in evidence_hint) +# live — requires the ephemeral Composer/GCP game-day window +# both — unit-tested logic plus a live game-day execution + +version: 1 + +defaults: + environment: atlas-dev + batch_prefix: atlas-s6- + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION] + maximum_duration_minutes: 30 + maximum_cost_usd: 0.50 + +scenarios: + # ---------------------------------------------------------------- INGESTION + S6-ING-001: + category: INGESTION + description: Missing source artifact — expected JSONL absent at load time + risk_level: LOW + target_component: upload_events / load_events boundary + preconditions: [isolated batch id, no canonical writes] + injection_method: point load_events at a GCS URI that was never uploaded + expected_detection: load step fails; task_events FAILED; pipeline_runs FAILED + expected_alert: "Atlas: pipeline failed" + expected_containment: no raw mutation, no success marker, downstream blocked + allowed_data_impact: none (isolated batch only) + recovery_action: RERUN_BATCH + verification_queries: + - raw count for batch equals 50000 after recovery rerun + - no duplicate event_ids in raw for batch + cleanup: delete isolated batch rows and GCS prefix + recurrence_prevention: covered by existing exact-count load validation + execution_mode: live + maximum_duration_minutes: 30 + maximum_cost_usd: 0.10 + + S6-ING-002: + category: INGESTION + description: Corrupt JSONL artifact fails parsing/loading without publication + risk_level: MEDIUM + target_component: BigQuery raw load + preconditions: [isolated batch id, isolated GCS prefix] + injection_method: upload a deliberately truncated/garbled JSONL to the isolated prefix + expected_detection: load job error; sanitized error in task_events; FAILED audit + expected_alert: "Atlas: pipeline failed" + expected_containment: invalid artifact preserved for forensics; no curated publication; no raw payload in logs + allowed_data_impact: none (isolated batch only) + recovery_action: QUARANTINE_BATCH then RERUN_BATCH + verification_queries: + - regenerated artifact checksum matches deterministic expectation + - raw count equals 50000 after rerun; zero rows from corrupt object + cleanup: quarantined object moved/labeled; isolated rows deleted + recurrence_prevention: checksum verification before load (existing) + regression test + execution_mode: live + maximum_cost_usd: 0.10 + + S6-ING-003: + category: INGESTION + description: Checksum conflict on immutable GCS path is rejected + risk_level: LOW + target_component: upload_events create-only GCS semantics + preconditions: [isolated batch id with existing object] + injection_method: attempt second upload with different content for the same batch path + expected_detection: upload rejected (precondition failure); original generation unchanged + expected_alert: none required (blocked before any pipeline impact) + expected_containment: original object generation preserved; batch history intact + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: + - GCS object generation unchanged after conflict attempt + cleanup: none (nothing was mutated) + recurrence_prevention: create-only upload contract regression test + execution_mode: both + evidence_hint: tests/unit test for upload precondition + live generation check + maximum_cost_usd: 0.01 + + S6-ING-004: + category: INGESTION + description: Partial raw load detected and repaired without duplication + risk_level: HIGH + target_component: raw load validation + preconditions: [isolated batch id, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + injection_method: delete a slice of the isolated batch's raw rows after load + expected_detection: exact-count validation fails; run FAILED; partial state visible + expected_alert: "Atlas: pipeline failed" + expected_containment: no warehouse publication for the partial batch + allowed_data_impact: isolated batch rows only + recovery_action: REPAIR_PARTIAL_LOAD + verification_queries: + - raw count equals 50000 exactly after repair + - zero duplicate event_ids for batch + - accepted + rejected equals raw after rerun + cleanup: delete isolated batch data + recurrence_prevention: exact-count gate (existing) + partial-repair runbook + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + maximum_cost_usd: 0.20 + + S6-ING-005: + category: INGESTION + description: Duplicate execution of the same batch does not duplicate data + risk_level: MEDIUM + target_component: batch idempotency (GCS reuse + raw load skip + dbt incremental) + preconditions: [isolated batch id already loaded] + injection_method: trigger a second pipeline run with the identical batch id/seed + expected_detection: second run reuses immutable object; load skipped/reconciled + expected_alert: none (idempotency is expected behavior) + expected_containment: facts unique; marts stable; two distinct pipeline_run_id rows + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: + - raw count unchanged after duplicate execution + - fact event_ids unique for batch + - two pipeline_runs rows exist for the batch + cleanup: none + recurrence_prevention: idempotency regression suite (existing Sprint 3/4 tests) + execution_mode: live + maximum_cost_usd: 0.10 + + S6-ING-006: + category: INGESTION + description: Transient GCS failure retries and reconciles + risk_level: LOW + target_component: upload step retry policy + preconditions: [isolated batch id] + injection_method: upload_once context flag (existing controlled --fail-once path) + expected_detection: attempt 1 fails, RETRY task event, attempt 2 succeeds + expected_alert: none (recovered within retry policy) + expected_containment: single final object; no partial artifacts + allowed_data_impact: none + recovery_action: RETRY_TASK + verification_queries: + - task_events has attempt 1 FAILED/RETRY and attempt 2 SUCCESS for upload task + cleanup: none + recurrence_prevention: retry-policy tests (existing) + execution_mode: live + maximum_cost_usd: 0.05 + + S6-ING-007: + category: INGESTION + description: Transient BigQuery failure retries without duplicate load + risk_level: MEDIUM + target_component: raw load retry policy + preconditions: [isolated batch id] + injection_method: fail first load attempt via injection hook (test-only parameter) + expected_detection: RETRY recorded; eventual success or controlled failure + expected_alert: none when recovered + expected_containment: no duplicate rows after retry success + allowed_data_impact: none + recovery_action: RETRY_TASK + verification_queries: + - raw count equals 50000 exactly; zero duplicate event_ids + cleanup: none + recurrence_prevention: load idempotency test (existing WRITE_TRUNCATE-per-batch semantics) + execution_mode: unit + evidence_hint: tests/unit/test_failure_injection.py transient-retry coverage + maximum_cost_usd: 0.05 + + S6-ING-008: + category: INGESTION + description: Retry exhaustion reaches terminal FAILED and blocks downstream + risk_level: MEDIUM + target_component: task retry policy + failure propagation + preconditions: [isolated batch id] + injection_method: persistent failure injection (all attempts fail) + expected_detection: attempts exhausted; task FAILED; downstream UPSTREAM_FAILED; run FAILED + expected_alert: "Atlas: pipeline failed" + expected_containment: no publication; terminal audit rows recorded + allowed_data_impact: none + recovery_action: RERUN_BATCH + verification_queries: + - pipeline_runs FAILED for injected run; SUCCESS for recovery run + cleanup: remove injection flag + recurrence_prevention: failure-propagation tests (existing Sprint 3) + execution_mode: live + maximum_cost_usd: 0.10 + + # ------------------------------------------------------------- ORCHESTRATION + S6-AIR-001: + category: ORCHESTRATION + description: Worker interruption mid-task is retryable without duplication + risk_level: MEDIUM + target_component: Airflow task execution + preconditions: [isolated batch id, safe task] + injection_method: kill the task process mid-execution (isolated batch only) + expected_detection: attempt marked failed/retry; no silent RUNNING state + expected_alert: none when retry recovers + expected_containment: rerun does not duplicate data + allowed_data_impact: none + recovery_action: RETRY_TASK + verification_queries: + - task_events shows interrupted attempt + successful retry + - raw/fact counts exact after recovery + cleanup: none + recurrence_prevention: idempotent step design (existing) + execution_mode: live + maximum_cost_usd: 0.10 + + S6-AIR-002: + category: ORCHESTRATION + description: Task timeout is classified and blocks downstream publication + risk_level: LOW + target_component: execution_timeout policy + preconditions: [isolated batch id] + injection_method: injected sleep beyond a drill-scoped execution_timeout + expected_detection: timeout failure recorded with correct classification + expected_alert: 'Atlas: pipeline failed when run terminal-fails' + expected_containment: downstream tasks do not publish + allowed_data_impact: none + recovery_action: RERUN_BATCH + verification_queries: + - task_events FAILED with timeout error_type for injected task + cleanup: remove drill timeout override + recurrence_prevention: timeout policy documented per task + execution_mode: live + maximum_cost_usd: 0.05 + + S6-AIR-003: + category: ORCHESTRATION + description: Finalizer failure leaves reconstructable operational truth + risk_level: HIGH + target_component: write_run_summary finalizer + preconditions: [isolated batch id] + injection_method: fail the finalizer write path via injection hook + expected_detection: missing/incomplete finalization detected; no false SUCCESS + expected_alert: "Atlas: telemetry incomplete" + expected_containment: pipeline data state remains truthful + allowed_data_impact: none + recovery_action: RECONSTRUCT_AUDIT + verification_queries: + - reconstructed pipeline_runs row matches Airflow + GCS + BigQuery evidence + - recovery_actions row RECONSTRUCT_AUDIT SUCCESS VERIFIED + cleanup: remove injection hook + recurrence_prevention: finalizer reconciliation tests (existing) + reconstruction runbook + execution_mode: live + maximum_cost_usd: 0.10 + + S6-AIR-004: + category: ORCHESTRATION + description: Overlapping runs are controlled; same batch stays idempotent + risk_level: MEDIUM + target_component: DAG max_active_runs / batch identity + preconditions: [isolated batch ids] + injection_method: trigger concurrent runs (distinct batches, then same batch) + expected_detection: concurrency settings serialize unsafe overlap + expected_alert: none + expected_containment: no cross-batch interference; same-batch rerun idempotent + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: + - per-batch counts exact; fact uniqueness holds across both batches + cleanup: delete isolated batches + recurrence_prevention: DAG concurrency configuration tests + execution_mode: live + maximum_cost_usd: 0.20 + + S6-AIR-005: + category: ORCHESTRATION + description: Invalid run context fails before any mutation + risk_level: LOW + target_component: resolve_run_context validation + preconditions: [] + injection_method: supply invalid batch id / date / seed via dagrun conf + expected_detection: resolve_run_context raises before any cloud mutation + expected_alert: none required + expected_containment: zero writes to GCS/BigQuery/audit beyond the failed context task + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: + - no pipeline_runs/raw rows exist for the invalid identifiers + cleanup: none + recurrence_prevention: context validation unit tests + execution_mode: both + evidence_hint: tests/unit/test_run_context.py + live invalid-conf trigger + maximum_cost_usd: 0.01 + + S6-AIR-006: + category: ORCHESTRATION + description: Scheduler interruption / missed run is visible and recoverable + risk_level: MEDIUM + target_component: schedule + freshness monitor + preconditions: [monitoring_enabled true, drill threshold override] + injection_method: paused schedule window with drill-scoped freshness threshold + expected_detection: monitor missing-run/stale check FAIL + expected_alert: "Atlas: data stale" + expected_containment: bounded backfill only; no unbounded catch-up + allowed_data_impact: none + recovery_action: BACKFILL + verification_queries: + - monitor_evaluations FAIL then PASS after bounded backfill + cleanup: restore threshold; resume schedule + recurrence_prevention: freshness monitor (existing Sprint 5) + execution_mode: live + maximum_cost_usd: 0.20 + + # ----------------------------------------------------------------- WAREHOUSE + S6-DBT-001: + category: WAREHOUSE + description: Source freshness failure blocks scheduled run, not backfills + risk_level: LOW + target_component: dbt source freshness gate + preconditions: [isolated batch id] + injection_method: stale source window against drill filter (no canonical mutation) + expected_detection: dbt source freshness error blocks the scheduled path + expected_alert: 'Atlas: pipeline failed when run terminal-fails' + expected_containment: no build executes after failed freshness + allowed_data_impact: none + recovery_action: BACKFILL + verification_queries: + - backfill_mode run skips freshness by documented policy and reconciles + cleanup: restore freshness config + recurrence_prevention: freshness policy documented (Sprint 3 backfill semantics) + execution_mode: live + maximum_cost_usd: 0.10 + + S6-DBT-002: + category: WAREHOUSE + description: dbt test failure blocks publication and recovers on clean rerun + risk_level: LOW + target_component: dbt build gate + preconditions: [isolated batch id] + injection_method: existing controlled inject_failure dbt var + expected_detection: dbt build fails; run FAILED; incident opens + expected_alert: "Atlas: pipeline failed" + expected_containment: no success marker; marts unchanged + allowed_data_impact: none + recovery_action: RERUN_BATCH + verification_queries: + - recovery run SUCCESS with 10/10 quality checks PASS + cleanup: rerun without injection var + recurrence_prevention: proven in Sprint 5 Drill B; regression retained + execution_mode: live + maximum_cost_usd: 0.10 + + S6-DBT-003: + category: WAREHOUSE + description: Referential-integrity failure is traceable, publication blocked + risk_level: MEDIUM + target_component: dbt relationship tests + preconditions: [isolated fixture dataset] + injection_method: fixture rows with dangling foreign keys in isolated schema + expected_detection: relationship test fails against fixture target + expected_alert: 'Atlas: reconciliation failed (when routed through monitor)' + expected_containment: invalid rows traceable; no incorrect mart publication + allowed_data_impact: fixture dataset only + recovery_action: QUARANTINE_BATCH + verification_queries: + - failing rows enumerated by stored test query; canonical marts untouched + cleanup: drop fixture dataset + recurrence_prevention: relationship tests (existing) + fixture regression + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + maximum_cost_usd: 0.10 + + S6-DBT-004: + category: WAREHOUSE + description: Duplicate fact event is blocked by uniqueness protection + risk_level: MEDIUM + target_component: fct_events uniqueness grain + preconditions: [isolated fixture dataset] + injection_method: duplicate event_id rows in isolated fixture target + expected_detection: uniqueness test fails + expected_alert: 'Atlas: reconciliation failed (fixture-scoped)' + expected_containment: fact grain protected; canonical facts unchanged + allowed_data_impact: fixture dataset only + recovery_action: REPAIR_PARTIAL_LOAD + verification_queries: + - fixture duplicate removed; uniqueness test passes; canonical counts unchanged + cleanup: drop fixture dataset + recurrence_prevention: uniqueness tests (existing) + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + maximum_cost_usd: 0.10 + + S6-DBT-005: + category: WAREHOUSE + description: Late-arriving events included by bounded backfill without duplication + risk_level: MEDIUM + target_component: incremental models + backfill path + preconditions: [two isolated batches] + injection_method: initial batch, then delayed additional batch for a prior date + expected_detection: n/a (planned flow); freshness policy honored + expected_alert: none + expected_containment: unaffected history byte-identical; no duplicate facts + allowed_data_impact: isolated batches only + recovery_action: BACKFILL + verification_queries: + - late events present; prior batches unchanged; fact uniqueness; mart reconciliation + cleanup: delete isolated batches + recurrence_prevention: backfill regression (existing Sprint 3) + late-data evidence + execution_mode: live + maximum_cost_usd: 0.20 + + S6-DBT-006: + category: WAREHOUSE + description: Incremental target corruption detected and repaired, not full-refreshed + risk_level: HIGH + target_component: incremental model targets + preconditions: [isolated schema copy, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + injection_method: mutate rows in an isolated copy of an incremental target + expected_detection: batch-scoped reconciliation FAIL against the isolated target + expected_alert: 'Atlas: reconciliation failed (fixture-scoped)' + expected_containment: corruption bounded to isolated schema + allowed_data_impact: isolated schema only + recovery_action: REBUILD_PARTITION + verification_queries: + - repaired target matches source-of-truth rebuild for affected batch only + cleanup: drop isolated schema + recurrence_prevention: reconciliation checks (existing) + targeted-repair runbook + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + maximum_cost_usd: 0.20 + + S6-DBT-007: + category: WAREHOUSE + description: Partition rebuild touches only the intended partition + risk_level: MEDIUM + target_component: partitioned raw/fact tables (isolated copies) + preconditions: [isolated fixture partitioned table] + injection_method: rebuild one partition of the isolated table + expected_detection: n/a (controlled maintenance flow) + expected_alert: none + expected_containment: other partitions byte-identical (count + checksum) + allowed_data_impact: isolated fixture only + recovery_action: REBUILD_PARTITION + verification_queries: + - non-target partition row counts/checksums unchanged; downstream reconciles + cleanup: drop fixture table + recurrence_prevention: bounded-rebuild runbook + recovery audit + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_DESTRUCTIVE_FIXTURE] + maximum_cost_usd: 0.20 + + # -------------------------------------------------------------------- SCHEMA + S6-SCH-001: + category: SCHEMA + description: Approved nullable field classifies ALLOWED + risk_level: LOW + target_component: schema-drift classifier + migration path + preconditions: [fixture table or manifest override] + injection_method: add configured nullable field to fixture; allowed_new_fields entry + expected_detection: drift finding ALLOWED approved_new_field + expected_alert: none + expected_containment: older consumers keep working + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output ALLOWED for the configured field] + cleanup: drop fixture + recurrence_prevention: classifier unit tests + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-002: + category: SCHEMA + description: Unapproved additive field classifies WARNING + risk_level: LOW + target_component: schema-drift classifier + preconditions: [fixture table] + injection_method: add nullable field NOT in allowed_new_fields + expected_detection: WARNING unapproved_new_field + expected_alert: none (warning tier) + expected_containment: contract update required before approval + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output WARNING] + cleanup: drop fixture + recurrence_prevention: classifier unit tests + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-003: + category: SCHEMA + description: Renamed field classifies BREAKING with bridge requirement + risk_level: MEDIUM + target_component: schema-drift classifier + preconditions: [fixture table] + injection_method: rename column in fixture (appears as removed + new) + expected_detection: BREAKING removed_field (+ WARNING/BREAKING for new name) + expected_alert: 'Atlas: breaking schema drift (fixture-scoped)' + expected_containment: compatibility bridge + deprecation period documented before adoption + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output BREAKING removed_field] + cleanup: drop fixture + recurrence_prevention: rename policy in ADR-015 + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-004: + category: SCHEMA + description: Removed field is blocked with consumer impact reported + risk_level: MEDIUM + target_component: schema-drift classifier + preconditions: [fixture table] + injection_method: drop expected column in fixture + expected_detection: BREAKING removed_field + expected_alert: 'Atlas: breaking schema drift (fixture-scoped)' + expected_containment: consumers enumerated in finding detail; adoption blocked + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output BREAKING removed_field] + cleanup: drop fixture + recurrence_prevention: classifier + governed contract + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-005: + category: SCHEMA + description: Incompatible type change is blocked pending forward migration + risk_level: MEDIUM + target_component: schema-drift classifier + preconditions: [fixture table] + injection_method: change column type in fixture + expected_detection: BREAKING type_change + expected_alert: 'Atlas: breaking schema drift (fixture-scoped)' + expected_containment: forward migration plan required before adoption + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output BREAKING type_change] + cleanup: drop fixture + recurrence_prevention: classifier + ADR-015 policy + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-006: + category: SCHEMA + description: Required-field change blocked without backfill + consumer proof + risk_level: HIGH + target_component: schema-drift classifier + preconditions: [fixture table] + injection_method: REQUIRED column made nullable / new REQUIRED column in fixture + expected_detection: BREAKING required_made_nullable / unapproved_required_field + expected_alert: 'Atlas: breaking schema drift (fixture-scoped)' + expected_containment: blocked unless backfill and consumer compatibility proven + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output BREAKING for both variants] + cleanup: drop fixture + recurrence_prevention: classifier unit tests + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-007: + category: SCHEMA + description: Partition-field change is high-risk and never auto-applied + risk_level: HIGH + target_component: schema-drift classifier + preconditions: [fixture table] + injection_method: fixture table without the expected partition column + expected_detection: BREAKING partition_change + expected_alert: 'Atlas: breaking schema drift (fixture-scoped)' + expected_containment: manual review required; no automatic application + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [classifier output BREAKING partition_change] + cleanup: drop fixture + recurrence_prevention: classifier unit tests + ADR-015 + execution_mode: both + evidence_hint: tests/unit/test_schema_drift.py + maximum_cost_usd: 0.05 + + S6-SCH-008: + category: SCHEMA + description: Multiple schema versions normalize without silent coercion + risk_level: MEDIUM + target_component: event schema versioning strategy + preconditions: [fixture inputs at two schema versions] + injection_method: fixture payloads with and without an approved additive field + expected_detection: version discriminator distinguishes inputs; unknown versions rejected + expected_alert: none + expected_containment: no silent coercion of unknown fields/versions + allowed_data_impact: fixture only + recovery_action: FORWARD_MIGRATION + verification_queries: [compatibility tests pass for both versions; unknown version rejected] + cleanup: none + recurrence_prevention: schema-version compatibility tests + execution_mode: unit + evidence_hint: tests/unit/test_schema_versions.py + maximum_cost_usd: 0.01 + + # ----------------------------------------------------------------------- IAM + S6-IAM-001: + category: IAM + description: BigQuery job permission removal identifies exact missing permission + risk_level: HIGH + target_component: atlas-composer-runtime bigquery.jobUser + preconditions: [before/after IAM capture, ATLAS_APPROVE_IAM] + injection_method: remove roles/bigquery.jobUser from runtime SA (bounded window) + expected_detection: 403 with bigquery.jobs.create identified; run FAILED + expected_alert: "Atlas: pipeline failed" + expected_containment: no broad role granted; failure classified IAM not code + allowed_data_impact: none + recovery_action: RESTORE_IAM + verification_queries: + - post-restore policy identical to captured baseline; probe query succeeds + cleanup: verify IAM matches baseline exactly + recurrence_prevention: IAM evidence table + role rationale docs + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_IAM] + maximum_duration_minutes: 20 + maximum_cost_usd: 0.05 + + S6-IAM-002: + category: IAM + description: BigQuery data access removal fails clearly without partial publication + risk_level: HIGH + target_component: atlas-composer-runtime bigquery.dataEditor + preconditions: [before/after IAM capture, ATLAS_APPROVE_IAM] + injection_method: remove roles/bigquery.dataEditor from runtime SA (bounded window) + expected_detection: read/write boundary 403; run FAILED at first data touch + expected_alert: "Atlas: pipeline failed" + expected_containment: no partial publication; sanitized error + allowed_data_impact: none + recovery_action: RESTORE_IAM + verification_queries: [post-restore probe write to isolated table succeeds] + cleanup: verify IAM matches baseline + recurrence_prevention: role rationale docs + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_IAM] + maximum_duration_minutes: 20 + maximum_cost_usd: 0.05 + + S6-IAM-003: + category: IAM + description: GCS object permission removal fails without unsafe fallback destination + risk_level: MEDIUM + target_component: runtime SA GCS access on raw bucket + preconditions: [before/after IAM capture, ATLAS_APPROVE_IAM] + injection_method: remove bucket-level binding for runtime SA (bounded window) + expected_detection: upload/read 403; run FAILED + expected_alert: "Atlas: pipeline failed" + expected_containment: no alternate destination attempted + allowed_data_impact: none + recovery_action: RESTORE_IAM + verification_queries: [post-restore probe upload to isolated prefix succeeds] + cleanup: verify bucket policy matches baseline + recurrence_prevention: single-destination contract (no fallback code paths) + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_IAM] + maximum_duration_minutes: 20 + maximum_cost_usd: 0.05 + + S6-IAM-004: + category: IAM + description: Composer runtime permission failure distinguishable from code failure + risk_level: MEDIUM + target_component: task error classification + preconditions: [any S6-IAM scenario active] + injection_method: observed during S6-IAM-001/002/003 execution + expected_detection: task_events error_type reflects permission denial, not code error + expected_alert: "Atlas: pipeline failed" + expected_containment: sanitized evidence; classification IAM + allowed_data_impact: none + recovery_action: RESTORE_IAM + verification_queries: [task_events error_type/message shows 403/permission classification] + cleanup: covered by parent scenario + recurrence_prevention: error-classification unit tests + execution_mode: both + evidence_hint: tests/unit/test_failure_injection.py IAM classification + maximum_cost_usd: 0.01 + + S6-IAM-005: + category: IAM + description: WIF authentication failure stops deployment before mutation + risk_level: MEDIUM + target_component: GitHub OIDC -> WIF -> deployer impersonation + preconditions: [deployment workflow] + injection_method: run deploy workflow from a ref/condition WIF rejects (or with provider briefly constrained) + expected_detection: auth step fails; workflow stops pre-mutation + expected_alert: 'Atlas: deployment failed when a deployment record was opened; otherwise workflow evidence' + expected_containment: no SA key fallback; prior release remains active + allowed_data_impact: none + recovery_action: RESTORE_IAM + verification_queries: [no deployments row mutated; current release unchanged] + cleanup: restore provider condition if changed + recurrence_prevention: WIF condition tests (Sprint 4) retained + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_IAM] + maximum_cost_usd: 0.05 + + # ---------------------------------------------------------------- DEPLOYMENT + S6-DEP-001: + category: DEPLOYMENT + description: Broken DAG import blocks before smoke; no SUCCESS record + risk_level: LOW + target_component: CI DAG gate + deployment import verification + preconditions: [temporary defect branch] + injection_method: deliberate import error on a temp branch (proven in Sprint 4; re-verify path) + expected_detection: atlas-airflow CI job fails / deployment import check fails + expected_alert: 'Atlas: deployment failed when reached in deploy stage' + expected_containment: no smoke run; no SUCCESS deployment record + allowed_data_impact: none + recovery_action: RESTORE_RELEASE + verification_queries: [deployments has no SUCCESS row for defective sha] + cleanup: delete temp branch + recurrence_prevention: CI gate (existing, demonstrated Sprint 4) + execution_mode: live + maximum_cost_usd: 0.05 + + S6-DEP-002: + category: DEPLOYMENT + description: Incompatible dependency fails environment validation + risk_level: MEDIUM + target_component: deployment environment validation (pip check / constraints) + preconditions: [temporary defect branch] + injection_method: pin an incompatible provider version on temp branch + expected_detection: pip check / constraint validation fails in CI or deploy validation + expected_alert: none required (blocked pre-deployment) + expected_containment: previous release remains active + allowed_data_impact: none + recovery_action: RESTORE_RELEASE + verification_queries: [current deployed sha unchanged] + cleanup: delete temp branch + recurrence_prevention: pinned constraints + pip check gate (existing) + execution_mode: unit + evidence_hint: CI pip-check gate; static demonstration acceptable + maximum_cost_usd: 0.01 + + S6-DEP-003: + category: DEPLOYMENT + description: Failed migration blocks DAG promotion; changed migration cannot masquerade + risk_level: MEDIUM + target_component: migration ledger + deploy ordering + preconditions: [isolated invalid migration fixture] + injection_method: invalid SQL migration in drill manifest (never merged) + expected_detection: ledger records FAILED; deploy stops before DAG promotion + expected_alert: "Atlas: deployment failed" + expected_containment: FAILED migration blocks promotion; checksum change of APPLIED refused + allowed_data_impact: none + recovery_action: FORWARD_MIGRATION + verification_queries: + - schema_migrations FAILED row for drill id; APPLIED rows unchanged + cleanup: remove drill migration; ledger row retained as evidence + recurrence_prevention: ledger checksum protection (existing + Sprint 5 retry fix) + execution_mode: live + maximum_cost_usd: 0.05 + + S6-DEP-004: + category: DEPLOYMENT + description: Bundle checksum mismatch rejects release; immutable bundle preserved + risk_level: LOW + target_component: build_deployment_bundle create-only semantics + preconditions: [existing release path] + injection_method: attempt re-upload of altered bundle for an existing sha + expected_detection: checksum mismatch fails the upload step + expected_alert: none required (blocked) + expected_containment: existing bundle bytes unchanged + allowed_data_impact: none + recovery_action: RESTORE_RELEASE + verification_queries: [release object generation and checksum unchanged] + cleanup: none + recurrence_prevention: create-only contract tests (existing Sprint 4) + execution_mode: both + evidence_hint: Sprint 4 evidence + live generation check + maximum_cost_usd: 0.01 + + S6-DEP-005: + category: DEPLOYMENT + description: Failed smoke run records FAILED and does not promote + risk_level: MEDIUM + target_component: deployment smoke gate + preconditions: [controlled defective release] + injection_method: deploy release with controlled smoke-breaking defect + expected_detection: smoke validation fails; deployments FAILED with failure_stage + expected_alert: "Atlas: deployment failed" + expected_containment: no success metadata; recovery command documented + allowed_data_impact: none + recovery_action: RESTORE_RELEASE + verification_queries: [deployments FAILED row; subsequent restore SUCCESS] + cleanup: rollback to validated release + recurrence_prevention: smoke gate (existing, demonstrated Sprint 4/5) + execution_mode: live + maximum_cost_usd: 0.20 + + S6-DEP-006: + category: DEPLOYMENT + description: Schema/runtime incompatibility fails rollback eligibility with forward guidance + risk_level: MEDIUM + target_component: rollback schema-compatibility check + preconditions: [release manifests with schema version declarations] + injection_method: candidate manifest declaring incompatible min schema version + expected_detection: eligibility check fails before any mutation + expected_alert: none required (blocked) + expected_containment: operator receives forward-recovery guidance + allowed_data_impact: none + recovery_action: FORWARD_MIGRATION + verification_queries: [rollback script exits nonzero with compatibility reason] + cleanup: none + recurrence_prevention: compatibility-check unit tests + execution_mode: unit + evidence_hint: tests for rollback_atlas.sh compatibility gate + maximum_cost_usd: 0.01 + + # ------------------------------------------------------------------ ROLLBACK + S6-RBK-001: + category: ROLLBACK + description: Missing rollback bundle fails before mutation + risk_level: LOW + target_component: rollback bundle verification + preconditions: [] + injection_method: request rollback to a sha with no stored bundle + expected_detection: bundle/manifest verification fails first + expected_alert: none required (blocked pre-mutation) + expected_containment: current runtime unchanged + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: [current deployed sha unchanged; no deployments mutation] + cleanup: none + recurrence_prevention: bundle-verification unit tests + execution_mode: both + evidence_hint: rollback script pre-checks + live nonexistent-sha attempt + maximum_cost_usd: 0.01 + + S6-RBK-002: + category: ROLLBACK + description: Rollback smoke failure records ROLLBACK_FAILED and escalates + risk_level: HIGH + target_component: rollback smoke gate + preconditions: [controlled rollback target with drill defect] + injection_method: force rollback smoke to fail via controlled defect + expected_detection: ROLLBACK_FAILED recorded; incident opens + expected_alert: "Atlas: rollback failed" + expected_containment: evidence preserved; secondary recovery executed + allowed_data_impact: none + recovery_action: RESTORE_RELEASE + verification_queries: [deployments ROLLBACK_FAILED row; secondary recovery SUCCESS] + cleanup: restore validated release + recurrence_prevention: rollback-failure runbook + execution_mode: live + approval_required: [ATLAS_APPROVE_FAILURE_INJECTION, ATLAS_APPROVE_ROLLBACK_TEST] + maximum_cost_usd: 0.20 + + S6-RBK-003: + category: ROLLBACK + description: Irreversible migration blocks runtime rollback; forward fix required + risk_level: MEDIUM + target_component: rollback schema rule (ADR-010) + preconditions: [manifest with min_compatible_schema_version ahead of target] + injection_method: attempt rollback across a declared-incompatible schema boundary + expected_detection: compatibility rule blocks before mutation + expected_alert: none required (blocked) + expected_containment: no automatic destructive reversal; forward fix documented + allowed_data_impact: none + recovery_action: FORWARD_MIGRATION + verification_queries: [rollback refused with schema-compatibility reason] + cleanup: none + recurrence_prevention: ADR-010/ADR-015 schema rule + tests + execution_mode: unit + evidence_hint: rollback compatibility tests + maximum_cost_usd: 0.01 + + # ------------------------------------------------------------- OBSERVABILITY + S6-OBS-001: + category: OBSERVABILITY + description: Cloud Logging write failure degrades visibly without breaking data path + risk_level: LOW + target_component: atlas.observability.logging cloud fan-out + preconditions: [] + injection_method: force _emit_to_cloud failure (test hook) + expected_detection: stdout contract line still emitted; degradation visible + expected_alert: none required + expected_containment: pipeline data operation unaffected + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: [stdout event present; caller exit code unchanged] + cleanup: none + recurrence_prevention: never-fail emission tests (existing Sprint 5) + execution_mode: unit + evidence_hint: tests/unit/test_observability_logging.py + maximum_cost_usd: 0.0 + + S6-OBS-002: + category: OBSERVABILITY + description: task_event write failure detected by completeness; history reconstructable + risk_level: MEDIUM + target_component: task_events write path + telemetry completeness + preconditions: [isolated batch id] + injection_method: break audit write (test hook) during isolated run + expected_detection: task_telemetry_write_failed event; completeness check FAIL + expected_alert: "Atlas: telemetry incomplete" + expected_containment: terminal run status governed by data operation, not telemetry + allowed_data_impact: none + recovery_action: RECONSTRUCT_AUDIT + verification_queries: [reconstructed task history matches logs + Airflow evidence] + cleanup: remove hook + recurrence_prevention: telemetry-safety tests (existing) + reconstruction procedure + execution_mode: live + maximum_cost_usd: 0.05 + + S6-OBS-003: + category: OBSERVABILITY + description: Metric publication failure persists evaluation without false status + risk_level: LOW + target_component: monitor metric publication + preconditions: [] + injection_method: force publish failure (test hook) + expected_detection: monitor_evaluations row persisted; publication error logged + expected_alert: none required + expected_containment: no false pipeline success/failure introduced + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: [evaluation row exists despite metric failure] + cleanup: none + recurrence_prevention: monitor error-isolation tests + execution_mode: unit + evidence_hint: tests/unit/test_observability_monitor.py + maximum_cost_usd: 0.0 + + S6-OBS-004: + category: OBSERVABILITY + description: Monitor DAG failure is itself detected; pipeline audit unaffected + risk_level: MEDIUM + target_component: atlas_observability_monitor DAG + preconditions: [Composer live] + injection_method: drill config making monitor evaluation raise + expected_detection: monitor task fails; check_status absence / monitor-health signal + expected_alert: 'Atlas: telemetry incomplete or absence-based policy' + expected_containment: atlas_ops business audit remains available + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: [pipeline audit queryable during monitor outage] + cleanup: restore monitor config + recurrence_prevention: monitor-health alerting review + execution_mode: live + maximum_cost_usd: 0.05 + + S6-OBS-005: + category: OBSERVABILITY + description: Unexpected alert-policy disablement is caught by governance check + risk_level: LOW + target_component: manage_atlas_alerts.sh status governance + preconditions: [] + injection_method: disable one policy outside the documented teardown set + expected_detection: alerts status/governance check reports drift from repo definitions + expected_alert: n/a (governance check output) + expected_containment: re-enable via manage_atlas_alerts.sh apply + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: [governance check FAIL then PASS after re-enable] + cleanup: policy re-enabled + recurrence_prevention: governance check in operator runbook + execution_mode: live + maximum_cost_usd: 0.0 + + S6-OBS-006: + category: OBSERVABILITY + description: Notification channel failure leaves incident truth intact + risk_level: LOW + target_component: notification routing + preconditions: [drill policy with unreachable channel copy] + injection_method: drill-only policy routed to no channel / disabled channel + expected_detection: incident opens in Cloud Monitoring regardless of delivery + expected_alert: incident without notification + expected_containment: operator query documented as alternate path; no invented delivery success + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: [incident timeline exists; delivery evidence honestly absent] + cleanup: delete drill policy + recurrence_prevention: runbook alternate-query section + execution_mode: live + maximum_cost_usd: 0.0 + + S6-OBS-007: + category: OBSERVABILITY + description: Linked log dataset unavailability falls back to Cloud Logging + risk_level: LOW + target_component: atlas_logs linked dataset + preconditions: [] + injection_method: none (procedural — use documented fallback filter while treating dataset as unavailable) + expected_detection: n/a + expected_alert: none + expected_containment: Cloud Logging remains source of truth; fallback query returns same events + allowed_data_impact: none + recovery_action: RESET_MONITOR + verification_queries: [same event set retrieved via logging read as via linked dataset] + cleanup: none + recurrence_prevention: runbook fallback section + execution_mode: live + maximum_cost_usd: 0.0 + + S6-OBS-008: + category: OBSERVABILITY + description: Terminal run with missing telemetry triggers reconstruction + risk_level: MEDIUM + target_component: telemetry completeness + reconstruction procedure + preconditions: [isolated terminal run with suppressed task events] + injection_method: suppress task-event emission for selected tasks (test hook) + expected_detection: telemetry_incomplete monitor FAIL + expected_alert: "Atlas: telemetry incomplete" + expected_containment: reconstruction rebuilds task history; RECONSTRUCT_AUDIT recorded + allowed_data_impact: none + recovery_action: RECONSTRUCT_AUDIT + verification_queries: [completeness PASS after reconstruction] + cleanup: remove hook + recurrence_prevention: proven in Sprint 5 Drill G; extended with reconstruction + execution_mode: live + maximum_cost_usd: 0.05 + + # ---------------------------------------------------------------------- COST + S6-COST-001: + category: COST + description: Removed partition filter blocked by dry-run byte ceiling + risk_level: LOW + target_component: cost guard (dry-run estimate) + preconditions: [] + injection_method: unpartitioned-scan query submitted through guarded path + expected_detection: dry-run estimate exceeds ceiling; execution refused + expected_alert: none (blocked pre-spend) + expected_containment: zero bytes billed; estimate recorded + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: [guard raises with estimated bytes; INFORMATION_SCHEMA shows no run] + cleanup: none + recurrence_prevention: guard unit tests + CI rule + execution_mode: both + evidence_hint: tests/unit/test_cost_guards.py + maximum_cost_usd: 0.0 + + S6-COST-002: + category: COST + description: Unbounded backfill window rejected without explicit override + risk_level: LOW + target_component: backfill window guard + preconditions: [] + injection_method: request backfill window beyond policy maximum + expected_detection: guard rejects; explicit override variable required + expected_alert: none + expected_containment: no jobs submitted + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: [guard rejects oversized window; accepts bounded window] + cleanup: none + recurrence_prevention: guard unit tests + execution_mode: unit + evidence_hint: tests/unit/test_cost_guards.py + maximum_cost_usd: 0.0 + + S6-COST-003: + category: COST + description: Full refresh outside policy requires approval + risk_level: LOW + target_component: full-refresh guard + preconditions: [] + injection_method: request full refresh without ATLAS_APPROVE_FULL_REFRESH + expected_detection: guard blocks; approval boundary documented + expected_alert: none + expected_containment: incremental path remains default + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: [guard blocks without approval; permits with approval] + cleanup: none + recurrence_prevention: guard unit tests + execution_mode: unit + evidence_hint: tests/unit/test_cost_guards.py + maximum_cost_usd: 0.0 + + S6-COST-004: + category: COST + description: Duplicate job submission limited by idempotent design + risk_level: LOW + target_component: batch idempotency (same as S6-ING-005 cost lens) + preconditions: [S6-ING-005 executed] + injection_method: duplicate execution evidence reused + expected_detection: second run's BigQuery work bounded (skip/reconcile path) + expected_alert: none + expected_containment: no duplicate load bytes at meaningful scale + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: [INFORMATION_SCHEMA bytes for duplicate run << initial run] + cleanup: none + recurrence_prevention: idempotency suite + execution_mode: live + maximum_cost_usd: 0.05 + + S6-COST-005: + category: COST + description: Query exceeding byte limit fails before material spend + risk_level: LOW + target_component: maximum_bytes_billed enforcement + preconditions: [] + injection_method: guarded query with tiny maximum_bytes_billed against larger table + expected_detection: BigQuery rejects with bytesBilledLimitExceeded + expected_alert: none + expected_containment: responsible component identifiable via job labels + allowed_data_impact: none + recovery_action: MANUAL_CONTAINMENT + verification_queries: [job error bytesBilledLimitExceeded; labels identify component] + cleanup: none + recurrence_prevention: guard applied to monitoring/analysis query paths + execution_mode: both + evidence_hint: tests/unit/test_cost_guards.py + one live guarded query + maximum_cost_usd: 0.01 diff --git a/config/observability.yaml b/config/observability.yaml new file mode 100644 index 0000000..690168c --- /dev/null +++ b/config/observability.yaml @@ -0,0 +1,69 @@ +# Atlas observability configuration (Sprint 5, Phase 8). +# Thresholds are INITIAL OPERATIONAL THRESHOLDS derived from the Sprint 1-4 +# synthetic workload baselines recorded in docs/preflight-sprint5.md; they are +# not production SLOs. Adjust with measured evidence, never to silence alerts. + +monitoring_enabled: true +environment: atlas-dev +# normal | drill — drill routes all published metrics to mode=drill series so +# controlled exercises never pollute normal history (ADR-011). +runtime_mode: normal + +expected_schedule: + dag_id: atlas_batch_pipeline + cron: "0 6 * * *" # daily 06:00 UTC (paused unless acceptance window) + # A scheduled run is "missing" when now - last run start exceeds + # 24h + grace. While the DAG is deliberately paused (default state between + # acceptance windows), missing-run findings downgrade to NO_DATA. + grace_seconds: 7200 + +freshness: + # Baseline: one successful batch per day when the environment is active. + warn_seconds: 93600 # 26 h + fail_seconds: 180000 # 50 h (two missed daily batches) + +volume: + baseline_window_runs: 7 # median raw rows over the last N successful runs + warn_deviation: 0.50 # |1 - latest/baseline| >= 50 % -> WARN + fail_deviation: 0.80 # >= 80 % -> FAIL + min_baseline_rows: 1000 # below this the baseline is meaningless -> NO_DATA + +rejection_rate: + # Baseline: generator injects ~10-12 % invalid events by design. + warn: 0.20 + fail: 0.35 + +reconciliation: + # Any FAIL row in atlas_ops.quality_results for the latest run -> FAIL. + window_hours: 48 + +telemetry: + # Expected terminal task events per run (see atlas.ops.task_events). + window_hours: 48 + +deployment: + # Latest terminal deployments row within window; FAILED/ROLLBACK_FAILED -> FAIL. + window_hours: 168 + +cost: + window_hours: 24 + baseline_window_days: 7 + warn_ratio: 3.0 # window bytes billed >= 3x daily baseline -> WARN + fail_ratio: 10.0 + min_bytes_billed: 1073741824 # ignore anomalies below 1 GiB absolute + +schema: + manifest: observability/schema/expected-schemas.json + # Additive nullable fields not in the manifest are WARNING by default; + # list explicitly approved additions here to classify them ALLOWED. + allowed_new_fields: [] + +alerting: + cooldown_seconds: 1800 + # Notification channel resource id is intentionally NOT stored in Git; the + # alert manage script reads ATLAS_NOTIFICATION_CHANNEL_ID at apply time. + +# Drill-only overrides (Phase 16). Empty in normal operation; a drill sets +# e.g. {freshness: {fail_seconds: 60}} on a temporary branch or via the +# ATLAS_OBSERVABILITY_OVERRIDES_JSON environment variable, never merged. +drill_overrides: {} diff --git a/config/public_extraction_manifest.yml b/config/public_extraction_manifest.yml new file mode 100644 index 0000000..5d4cbe7 --- /dev/null +++ b/config/public_extraction_manifest.yml @@ -0,0 +1,35 @@ +# Public extraction manifest for the standalone Atlas template. +version: 1 +repository_published: true +last_verified_commit: "template-extraction" + +global_substitutions: + - identifier: gcp_project_id + example_value_class: source sandbox GCP project id + occurrences_scope: docs, configs, scripts + disposition: REPLACE_WITH_SAMPLE + replacement_strategy: configure ATLAS_GCP_PROJECT_ID + - identifier: github_repository + example_value_class: source private repository identity + occurrences_scope: WIF docs and scripts + disposition: REPLACE_WITH_SAMPLE + replacement_strategy: configure ATLAS_GITHUB_REPOSITORY + - identifier: operator_identity + example_value_class: personal notification identity + occurrences_scope: operational evidence and runbooks + disposition: REPLACE_WITH_SAMPLE + replacement_strategy: configure notification identity outside Git + +detector_allowlist: + - scripts/validate_ci.sh + - src/atlas/ops/audit.py + - tests/unit/test_audit.py + - tests/unit/test_security_policy.py +files: [] +cleared_categories: + - committed credentials / service-account keys / tokens: none + - authorization headers in evidence: none + - private webhook URLs: none + - real user data: none (synthetic only) + - personal notification addresses: replaced with placeholders + - source sandbox project identifiers: replaced with examples diff --git a/dags/atlas_batch_pipeline.py b/dags/atlas_batch_pipeline.py new file mode 100644 index 0000000..396a11d --- /dev/null +++ b/dags/atlas_batch_pipeline.py @@ -0,0 +1,239 @@ +"""Atlas batch pipeline DAG — Sprint 3 orchestration layer.""" + +from __future__ import annotations + +import os +import sys +from datetime import UTC, datetime, timedelta +from pathlib import Path + +from airflow.providers.standard.operators.bash import BashOperator +from airflow.sdk import DAG, task + +# Composer parity: Airflow 3's DAG processor puts only the bundle root on +# sys.path. In Composer this DAG lives in /dags/project_atlas/, so its +# own directory (for atlas_orchestration) and $ATLAS_ROOT/src (for atlas.*) +# must be added explicitly before package imports. Locally both are no-ops. +_DAG_DIR = Path(__file__).resolve().parent +ATLAS_ROOT = Path(os.environ.get("ATLAS_ROOT", Path(__file__).resolve().parents[1])) +for _extra in (str(_DAG_DIR), str(ATLAS_ROOT / "src")): + if _extra not in sys.path: + sys.path.insert(0, _extra) + +from atlas_orchestration.callbacks import on_failure_callback, on_retry_callback +from atlas_orchestration.context import resolve_run_context_dict + +DAG_ID = "atlas_batch_pipeline" +START_DATE = datetime(2026, 7, 1, tzinfo=UTC) +STEP_SCRIPT = ATLAS_ROOT / "scripts" / "run_atlas_step.sh" +CTX_TEMPLATE = "{{ ti.xcom_pull(task_ids='resolve_run_context') | tojson }}" + + +def _ensure_atlas_importable() -> None: + """Make the atlas package importable inside task processes.""" + src = str(ATLAS_ROOT / "src") + if src not in sys.path: + sys.path.insert(0, src) + + +def bash_step(task_id: str, step: str, *, retries: int = 0, retry_minutes: int = 2) -> BashOperator: + """Create a BashOperator that dispatches one Atlas CLI step. + + The JSON run context is passed through the process environment rather than + inline in the command, so shell quoting can never corrupt it. + """ + return BashOperator( + task_id=task_id, + bash_command=f'"{STEP_SCRIPT}" {step} "$ATLAS_CTX"', + env={ + "ATLAS_CTX": CTX_TEMPLATE, + "ATLAS_TRY_NUMBER": "{{ ti.try_number }}", + }, + append_env=True, + retries=retries, + retry_delay=timedelta(minutes=retry_minutes) if retries else None, + ) + + +@task(task_id="resolve_run_context", multiple_outputs=False) +def resolve_run_context(**context) -> dict: + dag_run = context["dag_run"] + ctx = resolve_run_context_dict( + airflow_run_id=dag_run.run_id, + dag_id=dag_run.dag_id, + data_interval_end=context.get("data_interval_end"), + conf=dag_run.conf or {}, + ) + # Python @tasks bypass the step runner's telemetry wrapper; record this + # task's terminal event directly (best-effort, never fails the task). + try: + _ensure_atlas_importable() + from atlas.ops.task_events import TaskEventRecord, record_task_event_safely + + record_task_event_safely( + TaskEventRecord( + pipeline_run_id=ctx["pipeline_run_id"], + task_id="resolve_run_context", + attempt_number=int(context["ti"].try_number or 1), + event_type="SUCCESS", + batch_id=ctx["batch_id"], + airflow_run_id=ctx["airflow_run_id"], + dag_id=ctx["dag_id"], + status="SUCCESS", + operator_type="PythonOperator", + ) + ) + except Exception: # noqa: BLE001, S110 - telemetry must never break the task + pass + return ctx + + +@task(task_id="write_run_summary", trigger_rule="all_done") +def write_run_summary(**context) -> dict: + """Finalize the run: local JSON summary, BigQuery audit row, reconciliation. + + Runs with all_done and raises on FAILED/PARTIAL so this leaf cannot turn a + failed DAG green. + """ + _ensure_atlas_importable() + from atlas.ops.audit import ( + PipelineRunRecord, + collect_batch_metrics, + finalize_pipeline_run, + query_pipeline_run, + write_local_run_summary, + ) + from atlas.ops.finalizer import finalizer_should_fail, reconcile_run_summary + + ti = context["ti"] + ctx = ti.xcom_pull(task_ids="resolve_run_context") + if not ctx: + raise RuntimeError("resolve_run_context produced no run context") + + # publish_success_marker only runs when the whole chain succeeded, so its + # XCom presence is a reliable success signal under the all_done rule. + marker = ti.xcom_pull(task_ids="publish_success_marker") + status = "SUCCESS" if marker else "FAILED" + + completed_at = datetime.now(tz=UTC).isoformat() + + # Populate batch-scoped volumes for observability (best-effort; never fatal). + metrics: dict = {} + if status == "SUCCESS": + try: + metrics = collect_batch_metrics(ctx["batch_id"]) + except Exception: # noqa: BLE001 - metrics are best-effort + metrics = {} + + summary = { + "pipeline_run_id": ctx["pipeline_run_id"], + "batch_id": ctx["batch_id"], + "processing_date": ctx["processing_date"], + "status": status, + "completed_at": completed_at, + "airflow_run_id": ctx["airflow_run_id"], + "dag_id": ctx["dag_id"], + **metrics, + } + write_local_run_summary(ctx["pipeline_run_id"], summary) + + record = PipelineRunRecord( + pipeline_run_id=ctx["pipeline_run_id"], + batch_id=ctx["batch_id"], + airflow_run_id=ctx["airflow_run_id"], + dag_id=ctx["dag_id"], + processing_date=ctx["processing_date"], + started_at=completed_at, + completed_at=completed_at, + status=status, + attempt_number=int(ti.try_number or 1), + rows_loaded=metrics.get("rows_loaded"), + rows_accepted=metrics.get("rows_accepted"), + rows_rejected=metrics.get("rows_rejected"), + fact_rows=metrics.get("fact_rows"), + ) + audit_row = None + try: + finalize_pipeline_run(record) + audit_row = query_pipeline_run(ctx["pipeline_run_id"]) + except Exception as exc: # noqa: BLE001 - reconciliation records audit failures + summary["audit_error"] = str(exc) + + summary["reconciliation"] = reconcile_run_summary(summary, audit_row) + + # Sprint 5: telemetry-completeness verification (best-effort; visible + # degradation must never convert a successful data run into a failure). + try: + from atlas.ops.task_events import ( + TaskEventRecord, + record_task_event_safely, + telemetry_completeness, + ) + + completeness = telemetry_completeness(ctx["pipeline_run_id"]) + if status == "FAILED": + # Tasks that never ran because an upstream failed get a durable + # terminal event so the audit distinguishes "did not run" from + # "telemetry lost". + for missing_task in completeness["missing_terminal"]: + record_task_event_safely( + TaskEventRecord( + pipeline_run_id=ctx["pipeline_run_id"], + task_id=missing_task, + attempt_number=1, + event_type="UPSTREAM_FAILED", + batch_id=ctx["batch_id"], + airflow_run_id=ctx["airflow_run_id"], + dag_id=ctx["dag_id"], + status="UPSTREAM_FAILED", + operator_type="finalizer", + ) + ) + completeness = telemetry_completeness(ctx["pipeline_run_id"]) + summary["task_telemetry"] = completeness + except Exception as exc: # noqa: BLE001 - telemetry check is best-effort + summary["task_telemetry"] = {"complete": False, "error": str(exc)} + + write_local_run_summary(ctx["pipeline_run_id"], summary) + + if finalizer_should_fail(summary): + raise RuntimeError(f"Pipeline run finalized with status={status}") + return summary + + +with DAG( + dag_id=DAG_ID, + schedule="0 6 * * *", + start_date=START_DATE, + catchup=False, + max_active_runs=1, + # Deployments land paused: scheduled execution starts only after the smoke + # batch succeeds and an operator unpauses deliberately (Sprint 4, Phase 12). + is_paused_upon_creation=True, + default_args={ + "owner": "atlas", + "retries": 0, + "on_failure_callback": on_failure_callback, + "on_retry_callback": on_retry_callback, + }, + tags=["atlas", "sprint3"], +) as dag: + run_context = resolve_run_context() + + ensure_audit = bash_step("ensure_audit_resources", "ensure_audit_resources") + start_audit = bash_step("start_run_audit", "start_run_audit") + preflight = bash_step("preflight_environment", "preflight_environment") + generate = bash_step("generate_events", "generate_events") + upload = bash_step("upload_events", "upload_events", retries=2) + load_raw = bash_step("load_bigquery_raw", "load_events", retries=2) + validate_raw = bash_step("validate_raw_load", "validate_raw_load") + seed = bash_step("dbt_seed", "dbt_seed") + freshness = bash_step("dbt_source_freshness", "dbt_source_freshness", retries=1) + build = bash_step("dbt_build", "dbt_build") + validate_wh = bash_step("validate_warehouse", "validate_warehouse") + publish = bash_step("publish_success_marker", "publish_success_marker") + summary = write_run_summary() + + run_context >> ensure_audit >> start_audit >> preflight >> generate + generate >> upload >> load_raw >> validate_raw >> seed >> freshness >> build + build >> validate_wh >> publish >> summary diff --git a/dags/atlas_observability_monitor.py b/dags/atlas_observability_monitor.py new file mode 100644 index 0000000..1a0ba0e --- /dev/null +++ b/dags/atlas_observability_monitor.py @@ -0,0 +1,67 @@ +"""Atlas observability monitor DAG (Sprint 5, Phase 8). + +Evaluates system health independently of the business pipeline every 30 +minutes while the environment is active. Read-only except for +``atlas_ops.monitor_evaluations`` rows and Cloud Monitoring metric points. +No network calls at import time; all atlas imports happen inside the task. +""" + +from __future__ import annotations + +import os +import sys +from datetime import UTC, datetime +from pathlib import Path + +from airflow.sdk import DAG, task + +# Composer parity: same sys.path bootstrap as atlas_batch_pipeline (Airflow 3 +# does not add the DAG file's own subfolder to sys.path). +_DAG_DIR = Path(__file__).resolve().parent +ATLAS_ROOT = Path(os.environ.get("ATLAS_ROOT", Path(__file__).resolve().parents[1])) +for _extra in (str(_DAG_DIR), str(ATLAS_ROOT / "src")): + if _extra not in sys.path: + sys.path.insert(0, _extra) + +DAG_ID = "atlas_observability_monitor" +START_DATE = datetime(2026, 7, 1, tzinfo=UTC) + + +@task(task_id="evaluate_monitors") +def evaluate_monitors(**context) -> dict: + """Run all monitor checks; fail the task only on monitor infrastructure errors. + + A FAIL evaluation is a *finding*, not a task failure: alerting reacts to + the published check_status metrics, and failing this task would only + silence future evaluations. + """ + src = str(ATLAS_ROOT / "src") + if src not in sys.path: + sys.path.insert(0, src) + from atlas.observability.monitor import load_config, run_monitor + + config = load_config() + data_interval_start = context.get("data_interval_start") + data_interval_end = context.get("data_interval_end") + results = run_monitor( + config=config, + window_start=data_interval_start.isoformat() if data_interval_start else None, + window_end=data_interval_end.isoformat() if data_interval_end else None, + ) + summary = {r.check_name: r.status for r in results} + print({"monitor_summary": summary, "monitoring_enabled": config.get("monitoring_enabled")}) + return summary + + +with DAG( + dag_id=DAG_ID, + schedule="*/30 * * * *", + start_date=START_DATE, + catchup=False, + max_active_runs=1, + # Deployments land paused; unpaused deliberately during acceptance windows. + is_paused_upon_creation=True, + default_args={"owner": "atlas", "retries": 1}, + tags=["atlas", "sprint5", "observability"], +) as dag: + evaluate_monitors() diff --git a/dags/atlas_orchestration/__init__.py b/dags/atlas_orchestration/__init__.py new file mode 100644 index 0000000..a73b17f --- /dev/null +++ b/dags/atlas_orchestration/__init__.py @@ -0,0 +1 @@ +"""Atlas orchestration helpers (parse-time safe).""" diff --git a/dags/atlas_orchestration/callbacks.py b/dags/atlas_orchestration/callbacks.py new file mode 100644 index 0000000..4043536 --- /dev/null +++ b/dags/atlas_orchestration/callbacks.py @@ -0,0 +1,104 @@ +"""Task callbacks for retry and failure metadata capture. + +Sprint 5: callbacks write RETRY/FAILED rows into ``atlas_ops.task_events`` +best-effort. All atlas imports stay inside the functions so DAG parsing +performs no network calls and never depends on telemetry availability; a +telemetry failure can never fail the callback (and thus the task) itself. +""" + +from __future__ import annotations + +from typing import Any + + +def build_callback_context(context: dict[str, Any]) -> dict[str, Any]: + """Extract minimal callback metadata from an Airflow task context.""" + task_instance = context.get("task_instance") + dag_run = context.get("dag_run") + return { + "task_id": getattr(task_instance, "task_id", None), + "try_number": getattr(task_instance, "try_number", None), + "airflow_run_id": getattr(dag_run, "run_id", None), + "dag_id": getattr(dag_run, "dag_id", None), + "state": getattr(task_instance, "state", None), + "start_date": getattr(task_instance, "start_date", None), + "end_date": getattr(task_instance, "end_date", None), + } + + +def derive_callback_timing(meta: dict[str, Any]) -> dict[str, Any]: + """Derive timing evidence from Airflow task-instance timestamps. + + Sprint 6, Phase 1: FAILED/RETRY rows previously carried NULL timing. Use + only reliable evidence — the task instance's own start/end dates. When the + end date is not yet set at callback time, the callback wall clock bounds + completion (confidence "partial"). Never invent timestamps: with no start + date, everything stays NULL and confidence is recorded as "none". + """ + from datetime import UTC, datetime + + start = meta.get("start_date") + end = meta.get("end_date") + if start is None: + return { + "started_at": None, + "completed_at": None, + "duration_ms": None, + "timing_source": "airflow_task_instance", + "timing_confidence": "none", + } + confidence = "exact" + if end is None: + end = datetime.now(tz=UTC) + confidence = "partial" + return { + "started_at": start.isoformat(), + "completed_at": end.isoformat(), + "duration_ms": max(0, int((end - start).total_seconds() * 1000)), + "timing_source": "airflow_task_instance", + "timing_confidence": confidence, + } + + +def _record_callback_event(context: dict[str, Any], event_type: str) -> None: + meta = build_callback_context(context) + try: + from atlas.ops.task_events import TaskEventRecord, record_task_event_safely + + task_instance = context.get("task_instance") + run_ctx: dict[str, Any] = {} + try: + run_ctx = task_instance.xcom_pull(task_ids="resolve_run_context") or {} + except Exception: # noqa: BLE001 - context may predate resolve_run_context + run_ctx = {} + timing = derive_callback_timing(meta) + record_task_event_safely( + TaskEventRecord( + pipeline_run_id=run_ctx.get("pipeline_run_id", "unknown"), + task_id=meta.get("task_id") or "unknown", + attempt_number=max(1, int(meta.get("try_number") or 1)), + event_type=event_type, + batch_id=run_ctx.get("batch_id"), + airflow_run_id=meta.get("airflow_run_id"), + dag_id=meta.get("dag_id"), + status=event_type, + operator_type="callback", + started_at=timing["started_at"], + completed_at=timing["completed_at"], + duration_ms=timing["duration_ms"], + timing_source=timing["timing_source"], + timing_confidence=timing["timing_confidence"], + ) + ) + except Exception: # noqa: BLE001, S110 - callbacks must never raise + pass + + +def on_retry_callback(context: dict[str, Any]) -> None: + """Record a RETRY task event (best-effort, no import-time network).""" + _record_callback_event(context, "RETRY") + + +def on_failure_callback(context: dict[str, Any]) -> None: + """Record a FAILED task event (best-effort, no import-time network).""" + _record_callback_event(context, "FAILED") diff --git a/dags/atlas_orchestration/commands.py b/dags/atlas_orchestration/commands.py new file mode 100644 index 0000000..cf18d0b --- /dev/null +++ b/dags/atlas_orchestration/commands.py @@ -0,0 +1,82 @@ +"""Command builders for Atlas Airflow BashOperator tasks.""" + +from __future__ import annotations + +import json +import os +import shlex +from pathlib import Path +from typing import Any + + +def atlas_root() -> Path: + """Resolve Atlas runtime root without importing atlas.config at DAG parse time.""" + return Path(os.environ.get("ATLAS_ROOT", Path(__file__).resolve().parents[2])).expanduser() + + +def scripts_dir() -> Path: + return atlas_root() / "scripts" + + +def dbt_project_dir() -> Path: + return Path(os.environ.get("DBT_PROJECT_DIR", atlas_root() / "dbt" / "atlas_dbt")) + + +def run_atlas_step_command(step: str, context: dict[str, Any]) -> str: + """Build a shell command invoking the Atlas step dispatcher.""" + payload = json.dumps(context) + return ( + f"{shlex.quote(str(scripts_dir() / 'run_atlas_step.sh'))} {shlex.quote(step)} {shlex.quote(payload)}" + ) + + +def generate_events_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("generate_events", context) + + +def upload_events_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("upload_events", context) + + +def load_events_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("load_events", context) + + +def validate_raw_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("validate_raw_load", context) + + +def ensure_audit_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("ensure_audit_resources", context) + + +def start_audit_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("start_run_audit", context) + + +def preflight_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("preflight_environment", context) + + +def dbt_seed_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("dbt_seed", context) + + +def dbt_freshness_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("dbt_source_freshness", context) + + +def dbt_build_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("dbt_build", context) + + +def validate_warehouse_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("validate_warehouse", context) + + +def publish_marker_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("publish_success_marker", context) + + +def write_summary_command(context: dict[str, Any]) -> str: + return run_atlas_step_command("write_run_summary", context) diff --git a/dags/atlas_orchestration/context.py b/dags/atlas_orchestration/context.py new file mode 100644 index 0000000..64c2d5b --- /dev/null +++ b/dags/atlas_orchestration/context.py @@ -0,0 +1,74 @@ +"""Parse-time-safe run context resolution for Atlas Airflow DAGs.""" + +from __future__ import annotations + +from datetime import UTC, datetime +from typing import Any + +from atlas.batch.context import resolve_batch_context + + +def resolve_processing_date( + data_interval_end: datetime | None, + manual_processing_date: str | None = None, +) -> str: + """Derive the processing date from the UTC data interval end.""" + if manual_processing_date: + datetime.strptime(manual_processing_date, "%Y-%m-%d") + return manual_processing_date + if data_interval_end is None: + return datetime.now(tz=UTC).date().isoformat() + if data_interval_end.tzinfo is None: + data_interval_end = data_interval_end.replace(tzinfo=UTC) + return data_interval_end.astimezone(UTC).date().isoformat() + + +def resolve_run_context_dict( + *, + airflow_run_id: str, + dag_id: str, + data_interval_end: datetime | None = None, + conf: dict[str, Any] | None = None, +) -> dict[str, Any]: + """Build a validated run-context dictionary for XCom and command builders.""" + conf = conf or {} + processing_date = resolve_processing_date( + data_interval_end, + manual_processing_date=conf.get("processing_date"), + ) + context = resolve_batch_context( + processing_date=processing_date, + batch_id=conf.get("batch_id"), + pipeline_run_id=conf.get("pipeline_run_id"), + seed=conf.get("seed"), + airflow_run_id=airflow_run_id, + ) + manual = conf.get("processing_date") + scheduled = resolve_processing_date(data_interval_end) + backfill_mode = manual is not None and manual != scheduled + if backfill_mode: + # Sprint 6 cost guard (S6-COST-002): a backfill reaching further back + # than policy allows must be an explicit, reviewed decision — never an + # accident of a mistyped date. Import stays inside the branch so DAG + # parsing never touches the guard's BigQuery dependency. + from datetime import date as _date + + from atlas.observability.cost_guards import validate_backfill_window + + manual_date = _date.fromisoformat(str(manual)) + scheduled_date = _date.fromisoformat(scheduled) + window = sorted([manual_date, scheduled_date]) + validate_backfill_window(window[0], window[1]) + return { + "processing_date": context.processing_date, + "batch_id": context.batch_id, + "pipeline_run_id": context.pipeline_run_id, + "seed": context.seed, + "local_file_path": str(context.local_file_path), + "manifest_path": str(context.manifest_path), + "airflow_run_id": airflow_run_id, + "dag_id": dag_id, + "upload_once": bool(conf.get("upload_once")), + "dbt_test_failure": bool(conf.get("dbt_test_failure")), + "backfill_mode": backfill_mode, + } diff --git a/dags/atlas_orchestration/validation.py b/dags/atlas_orchestration/validation.py new file mode 100644 index 0000000..575b2f7 --- /dev/null +++ b/dags/atlas_orchestration/validation.py @@ -0,0 +1,7 @@ +"""Finalizer validation helpers for Atlas Airflow runs.""" + +from __future__ import annotations + +from atlas.ops.finalizer import finalizer_should_fail, reconcile_run_summary + +__all__ = ["finalizer_should_fail", "reconcile_run_summary"] diff --git a/dbt/atlas_dbt/dbt_project.yml b/dbt/atlas_dbt/dbt_project.yml new file mode 100644 index 0000000..34707b5 --- /dev/null +++ b/dbt/atlas_dbt/dbt_project.yml @@ -0,0 +1,49 @@ +name: atlas_dbt +version: "1.0.0" +config-version: 2 + +profile: atlas_dbt +require-dbt-version: "=1.11.12" + +model-paths: ["models"] +analysis-paths: ["analyses"] +test-paths: ["tests"] +seed-paths: ["seeds"] +macro-paths: ["macros"] + +target-path: "target" +clean-targets: + - "target" + - "dbt_packages" + +# BigQuery cost attribution (Sprint 5, ADR-012): dbt-bigquery converts this +# JSON query comment into BigQuery job labels (officially supported +# query-comment job-label mechanism), so dbt jobs are attributable in +# region-qualified INFORMATION_SCHEMA.JOBS alongside Python jobs. +query-comment: + comment: '{"application": "atlas", "component": "dbt", "environment": "atlas-dev"}' + job-label: true + +vars: + lookback_days: 3 + # Validated by the Sprint 1 acceptance report; override for later runs. + validated_run_id: "atlas-20260714T163527Z-19a0e4f6" + validated_batch_id: "" + inject_failure: false + +seeds: + atlas_dbt: + +schema: staging + +models: + atlas_dbt: + staging: + +schema: staging + +materialized: view + intermediate: + +schema: intermediate + core: + +schema: core + marts: + +schema: marts + +materialized: table diff --git a/dbt/atlas_dbt/models/core/core.yml b/dbt/atlas_dbt/models/core/core.yml new file mode 100644 index 0000000..ed760a6 --- /dev/null +++ b/dbt/atlas_dbt/models/core/core.yml @@ -0,0 +1,102 @@ +version: 2 + +models: + - name: dim_users + description: Accepted users at user_id grain with first/last event timestamps. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per user_id + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [core.fct_events, marts.mart_daily_event_metrics] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + columns: + - name: user_id + tests: + - not_null + - unique + - name: first_event_at + tests: + - not_null + - name: last_event_at + tests: + - not_null + tests: + - dbt_utils.expression_is_true: + arguments: + expression: "first_event_at <= last_event_at" + + - name: dim_countries + description: Country reference dimension sourced from the valid_country_codes seed. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per country_code + classification: PUBLIC + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [core.fct_events] + lifecycle_status: ACTIVE + freshness_expectation: on seed change + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + columns: + - name: country_code + tests: + - not_null + - unique + - name: is_active + tests: + - not_null + + - name: fct_events + description: > + Incremental accepted event fact keyed by event_id, partitioned by event_date + and clustered by event_name and country_code. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per event_id (global fact uniqueness — see ADR-017) + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [marts.mart_daily_event_metrics, warehouse_reconciliation, atlas_observability_monitor] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + columns: + - name: event_id + tests: + - not_null + - unique + - name: user_id + tests: + - not_null + - name: event_name + tests: + - not_null + - name: event_date + tests: + - not_null + - name: country_code + tests: + - not_null + - relationships: + arguments: + to: ref('dim_countries') + field: country_code + - name: platform + tests: + - not_null + - accepted_values: + arguments: + values: [ios, android, web] diff --git a/dbt/atlas_dbt/models/core/dim_countries.sql b/dbt/atlas_dbt/models/core/dim_countries.sql new file mode 100644 index 0000000..ea83a8c --- /dev/null +++ b/dbt/atlas_dbt/models/core/dim_countries.sql @@ -0,0 +1,11 @@ +{{ + config( + materialized='table' + ) +}} + +select + country_code, + is_active, + current_timestamp() as updated_at +from {{ ref('valid_country_codes') }} diff --git a/dbt/atlas_dbt/models/core/dim_users.sql b/dbt/atlas_dbt/models/core/dim_users.sql new file mode 100644 index 0000000..c6cdffb --- /dev/null +++ b/dbt/atlas_dbt/models/core/dim_users.sql @@ -0,0 +1,15 @@ +{{ + config( + materialized='table' + ) +}} + +select + user_id, + min(event_timestamp) as first_event_at, + max(event_timestamp) as last_event_at, + count(*) as event_count, + current_timestamp() as updated_at +from {{ ref('int_accepted_events') }} +where user_id is not null +group by user_id diff --git a/dbt/atlas_dbt/models/core/fct_events.sql b/dbt/atlas_dbt/models/core/fct_events.sql new file mode 100644 index 0000000..4a3f606 --- /dev/null +++ b/dbt/atlas_dbt/models/core/fct_events.sql @@ -0,0 +1,49 @@ +{{ + config( + materialized='incremental', + incremental_strategy='merge', + unique_key='event_id', + partition_by={'field': 'event_date', 'data_type': 'date'}, + cluster_by=['event_name', 'country_code'], + on_schema_change='fail' + ) +}} + +with accepted as ( + select * + from {{ ref('int_accepted_events') }} + where 1 = 1 + {% if is_incremental() %} + and ingested_at >= timestamp_sub( + coalesce((select max(ingested_at) from {{ this }}), timestamp('1970-01-01')), + interval {{ var('lookback_days') }} day + ) + {% endif %} + {% if var('start_date', none) is not none %} + and event_date >= date('{{ var("start_date") }}') + {% endif %} + {% if var('end_date', none) is not none %} + and event_date <= date('{{ var("end_date") }}') + {% endif %} +) + +select + event_id, + user_id, + event_name, + event_timestamp, + event_date, + country_code, + platform, + app_version, + ingested_at, + source_file, + pipeline_run_id, + batch_id, + raw_record_hash, + is_backdated_event_date, + has_event_date_timestamp_mismatch, + is_event_time_late_arriving, + classified_at as loaded_to_core_at, + current_timestamp() as updated_at +from accepted diff --git a/dbt/atlas_dbt/models/intermediate/int_accepted_events.sql b/dbt/atlas_dbt/models/intermediate/int_accepted_events.sql new file mode 100644 index 0000000..0f686c9 --- /dev/null +++ b/dbt/atlas_dbt/models/intermediate/int_accepted_events.sql @@ -0,0 +1,31 @@ +{{ + config( + materialized='view' + ) +}} + +select + event_id, + user_id, + event_name, + event_timestamp, + event_date, + country_code, + platform, + app_version, + ingested_at, + source_file, + pipeline_run_id, + batch_id, + raw_record_hash, + duplicate_rank, + is_duplicate_extra, + is_valid_country, + is_future_dated, + is_event_time_late_arriving, + is_backdated_event_date, + has_event_date_timestamp_mismatch, + rejection_reason, + classified_at +from {{ ref('int_event_classification') }} +where rejection_reason = 'accepted' diff --git a/dbt/atlas_dbt/models/intermediate/int_event_classification.sql b/dbt/atlas_dbt/models/intermediate/int_event_classification.sql new file mode 100644 index 0000000..cfaf879 --- /dev/null +++ b/dbt/atlas_dbt/models/intermediate/int_event_classification.sql @@ -0,0 +1,94 @@ +{{ + config( + materialized='table' + ) +}} + +-- Sprint 7 (ADR-006 amendment): duplicate semantics distinguish a WITHIN-BATCH +-- duplicate (a batch-scoped data-quality anomaly — the intentional 50 extras) +-- from a CROSS-BATCH replay (the same event_id reappearing in a later batch, +-- e.g. same-date reprocessing). Canonical selection is first-seen-batch-wins so +-- a replay never disturbs an already-published canonical fact, while within a +-- batch the latest write still wins. Global fact uniqueness is preserved: +-- exactly one canonical row per event_id. + +with staged as ( + select * from {{ ref('stg_events') }} +), + +valid_countries as ( + select country_code + from {{ ref('valid_country_codes') }} + where is_active +), + +-- Batch scope: batch_id for orchestrated loads, falling back to pipeline_run_id +-- for legacy Sprint 1 rows (which predate stable batch identity). +scoped as ( + select + staged.*, + coalesce(batch_id, pipeline_run_id) as batch_scope + from staged +), + +within_batch as ( + select + scoped.*, + -- Latest write wins WITHIN a batch (unchanged tie-break). + row_number() over ( + partition by batch_scope, event_id + order by + ingested_at desc, + event_timestamp desc, + source_file desc, + raw_record_hash desc + ) as within_batch_duplicate_rank + from scoped +), + +ranked as ( + select + within_batch.*, + within_batch_duplicate_rank > 1 as is_within_batch_duplicate, + -- Canonical selection across batches: within-batch winners first, then + -- earliest-arriving row (first-seen wins) so replays never flip an + -- already-canonical prior batch. Deterministic tie-breakers follow. + row_number() over ( + partition by event_id + order by + within_batch_duplicate_rank asc, + ingested_at asc, + event_timestamp desc, + source_file desc, + raw_record_hash desc + ) as duplicate_rank + from within_batch +), + +classified as ( + select + ranked.*, + duplicate_rank > 1 as is_duplicate_extra, + valid_countries.country_code is not null as is_valid_country, + case + when is_within_batch_duplicate then 'within_batch' + when duplicate_rank > 1 then 'cross_batch_replay' + else 'none' + end as duplicate_scope, + case + when ranked.user_id is null then 'missing_user_id' + when valid_countries.country_code is null then 'invalid_country_code' + when ranked.is_future_dated then 'future_dated' + when duplicate_rank > 1 then 'duplicate_extra' + else 'accepted' + end as rejection_reason + from ranked + left join valid_countries + on ranked.country_code = valid_countries.country_code +) + +select + * except (batch_scope), + rejection_reason = 'accepted' as is_accepted, + current_timestamp() as classified_at +from classified diff --git a/dbt/atlas_dbt/models/intermediate/int_rejected_events.sql b/dbt/atlas_dbt/models/intermediate/int_rejected_events.sql new file mode 100644 index 0000000..190c77f --- /dev/null +++ b/dbt/atlas_dbt/models/intermediate/int_rejected_events.sql @@ -0,0 +1,33 @@ +{{ + config( + materialized='table', + schema='quarantine' + ) +}} + +select + event_id, + user_id, + event_name, + event_timestamp, + event_date, + country_code, + platform, + app_version, + ingested_at, + source_file, + pipeline_run_id, + batch_id, + raw_record_hash, + duplicate_rank, + is_duplicate_extra, + is_valid_country, + is_future_dated, + is_event_time_late_arriving, + is_backdated_event_date, + has_event_date_timestamp_mismatch, + rejection_reason, + classified_at, + current_timestamp() as quarantined_at +from {{ ref('int_event_classification') }} +where rejection_reason != 'accepted' diff --git a/dbt/atlas_dbt/models/intermediate/intermediate.yml b/dbt/atlas_dbt/models/intermediate/intermediate.yml new file mode 100644 index 0000000..4e7240f --- /dev/null +++ b/dbt/atlas_dbt/models/intermediate/intermediate.yml @@ -0,0 +1,395 @@ +version: 2 + +models: + - name: int_event_classification + description: > + Physical-row quality classification with duplicate ranking and a single terminal + rejection reason per row. Precedence: missing_user_id, invalid_country_code, + future_dated, duplicate_extra, accepted. Warning flags remain independent. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per staged physical event record with classification flags + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.1" + consumers: [warehouse_reconciliation, int_accepted_events, int_rejected_events] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + columns: + - name: event_id + tests: + - not_null + - name: rejection_reason + tests: + - not_null + - accepted_values: + arguments: + values: + - accepted + - missing_user_id + - invalid_country_code + - future_dated + - duplicate_extra + - name: duplicate_rank + description: > + Global canonical rank per event_id (first-seen-batch wins; latest write + wins within a batch). duplicate_rank = 1 is the canonical row. + tests: + - not_null + - name: within_batch_duplicate_rank + description: Duplicate rank scoped to the batch; > 1 marks a within-batch duplicate. + tests: + - not_null + - name: is_within_batch_duplicate + description: True when this row duplicates another row in the same batch (the batch anomaly). + tests: + - not_null + - name: is_duplicate_extra + description: True when this row is not the global canonical row for its event_id. + tests: + - not_null + - name: duplicate_scope + description: Classifies the duplicate relationship for this row. + tests: + - not_null + - accepted_values: + arguments: + values: + - none + - within_batch + - cross_batch_replay + - name: is_accepted + tests: + - not_null + + - name: int_accepted_events + description: > + Accepted canonical events at one row per event_id that passed all blocking checks. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per accepted event_id (global canonical selection) + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [core.fct_events, warehouse_reconciliation] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + columns: + - name: event_id + tests: + - not_null + - unique + - name: rejection_reason + tests: + - accepted_values: + arguments: + values: [accepted] + + - name: int_rejected_events + description: > + Quarantined physical rows with a terminal blocking rejection reason. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per rejected physical event record + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [warehouse_reconciliation] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + columns: + - name: rejection_reason + tests: + - not_null + - dbt_utils.not_accepted_values: + arguments: + values: [accepted] + +unit_tests: + - name: test_rejection_precedence_missing_user + model: int_event_classification + given: + - input: ref('stg_events') + rows: + - { + event_id: evt-1, + user_id: null, + event_name: login, + event_timestamp: "2026-07-14 10:00:00 UTC", + event_date: "2026-07-14", + country_code: ZZ, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 12:00:00 UTC", + source_file: gs://bucket/a.jsonl, + pipeline_run_id: run-1, + raw_record_hash: 1, + event_timestamp_date: "2026-07-14", + ingested_date: "2026-07-14", + is_future_dated: false, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: false, + } + - input: ref('valid_country_codes') + rows: + - { country_code: US, is_active: true } + expect: + rows: + - { event_id: evt-1, rejection_reason: missing_user_id, is_accepted: false } + + - name: test_rejection_precedence_invalid_country_over_future + model: int_event_classification + given: + - input: ref('stg_events') + rows: + - { + event_id: evt-2, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-15 10:00:00 UTC", + event_date: "2026-07-15", + country_code: ZZ, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 12:00:00 UTC", + source_file: gs://bucket/a.jsonl, + pipeline_run_id: run-1, + raw_record_hash: 2, + event_timestamp_date: "2026-07-15", + ingested_date: "2026-07-14", + is_future_dated: true, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: true, + } + - input: ref('valid_country_codes') + rows: + - { country_code: US, is_active: true } + expect: + rows: + - { event_id: evt-2, rejection_reason: invalid_country_code, is_accepted: false } + + - name: test_duplicate_ranking_keeps_latest_canonical + model: int_event_classification + given: + - input: ref('stg_events') + rows: + - { + event_id: evt-dup, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-14 09:00:00 UTC", + event_date: "2026-07-14", + country_code: US, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 11:00:00 UTC", + source_file: gs://bucket/old.jsonl, + pipeline_run_id: run-1, + raw_record_hash: 10, + event_timestamp_date: "2026-07-14", + ingested_date: "2026-07-14", + is_future_dated: false, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: false, + } + - { + event_id: evt-dup, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-14 10:00:00 UTC", + event_date: "2026-07-14", + country_code: US, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 12:00:00 UTC", + source_file: gs://bucket/new.jsonl, + pipeline_run_id: run-1, + raw_record_hash: 11, + event_timestamp_date: "2026-07-14", + ingested_date: "2026-07-14", + is_future_dated: false, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: false, + } + - input: ref('valid_country_codes') + rows: + - { country_code: US, is_active: true } + expect: + rows: + - { + event_id: evt-dup, + duplicate_rank: 1, + rejection_reason: accepted, + is_accepted: true, + source_file: gs://bucket/new.jsonl, + } + - { + event_id: evt-dup, + duplicate_rank: 2, + rejection_reason: duplicate_extra, + is_accepted: false, + source_file: gs://bucket/old.jsonl, + } + + - name: test_accepted_row_can_carry_warning_flags + model: int_event_classification + given: + - input: ref('stg_events') + rows: + - { + event_id: evt-warn, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-14 10:00:00 UTC", + event_date: "2026-07-09", + country_code: US, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 12:00:00 UTC", + source_file: gs://bucket/a.jsonl, + pipeline_run_id: run-1, + raw_record_hash: 3, + event_timestamp_date: "2026-07-14", + ingested_date: "2026-07-14", + is_future_dated: false, + is_event_time_late_arriving: false, + is_backdated_event_date: true, + has_event_date_timestamp_mismatch: true, + } + - input: ref('valid_country_codes') + rows: + - { country_code: US, is_active: true } + expect: + rows: + - { + event_id: evt-warn, + rejection_reason: accepted, + is_accepted: true, + is_backdated_event_date: true, + has_event_date_timestamp_mismatch: true, + } + + - name: test_future_dated_is_blocking + model: int_event_classification + given: + - input: ref('stg_events') + rows: + - { + event_id: evt-future, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-15 10:00:00 UTC", + event_date: "2026-07-15", + country_code: US, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 12:00:00 UTC", + source_file: gs://bucket/a.jsonl, + pipeline_run_id: run-1, + raw_record_hash: 4, + event_timestamp_date: "2026-07-15", + ingested_date: "2026-07-14", + is_future_dated: true, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: true, + } + - input: ref('valid_country_codes') + rows: + - { country_code: US, is_active: true } + expect: + rows: + - { event_id: evt-future, rejection_reason: future_dated, is_accepted: false } + + - name: test_cross_batch_replay_preserves_first_seen + # Sprint 7: the same event_id arriving in a later batch (same date) is a + # cross-batch replay, not a within-batch duplicate. First-seen batch stays + # canonical; the replay is rejected as duplicate_extra with scope + # cross_batch_replay. is_within_batch_duplicate stays false for both. + model: int_event_classification + given: + - input: ref('stg_events') + rows: + - { + event_id: evt-replay, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-14 10:00:00 UTC", + event_date: "2026-07-14", + country_code: US, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 06:00:00 UTC", + source_file: gs://bucket/batch-a.jsonl, + pipeline_run_id: run-a, + batch_id: batch-a, + raw_record_hash: 100, + event_timestamp_date: "2026-07-14", + ingested_date: "2026-07-14", + is_future_dated: false, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: false, + } + - { + event_id: evt-replay, + user_id: user-1, + event_name: login, + event_timestamp: "2026-07-14 10:00:00 UTC", + event_date: "2026-07-14", + country_code: US, + platform: web, + app_version: 1.0.0, + ingested_at: "2026-07-14 15:00:00 UTC", + source_file: gs://bucket/batch-b.jsonl, + pipeline_run_id: run-b, + batch_id: batch-b, + raw_record_hash: 101, + event_timestamp_date: "2026-07-14", + ingested_date: "2026-07-14", + is_future_dated: false, + is_event_time_late_arriving: false, + is_backdated_event_date: false, + has_event_date_timestamp_mismatch: false, + } + - input: ref('valid_country_codes') + rows: + - { country_code: US, is_active: true } + expect: + rows: + - { + event_id: evt-replay, + batch_id: batch-a, + duplicate_rank: 1, + is_within_batch_duplicate: false, + is_duplicate_extra: false, + duplicate_scope: none, + rejection_reason: accepted, + is_accepted: true, + } + - { + event_id: evt-replay, + batch_id: batch-b, + duplicate_rank: 2, + is_within_batch_duplicate: false, + is_duplicate_extra: true, + duplicate_scope: cross_batch_replay, + rejection_reason: duplicate_extra, + is_accepted: false, + } diff --git a/dbt/atlas_dbt/models/marts/mart_daily_event_metrics.sql b/dbt/atlas_dbt/models/marts/mart_daily_event_metrics.sql new file mode 100644 index 0000000..90cb272 --- /dev/null +++ b/dbt/atlas_dbt/models/marts/mart_daily_event_metrics.sql @@ -0,0 +1,19 @@ +{{ + config( + materialized='table' + ) +}} + +select + event_date, + event_name, + country_code, + platform, + count(*) as event_count, + count(distinct user_id) as distinct_user_count, + countif(is_backdated_event_date) as backdated_event_date_count, + countif(has_event_date_timestamp_mismatch) as event_date_timestamp_mismatch_count, + countif(is_event_time_late_arriving) as event_time_late_arriving_count, + current_timestamp() as updated_at +from {{ ref('fct_events') }} +group by 1, 2, 3, 4 diff --git a/dbt/atlas_dbt/models/marts/marts.yml b/dbt/atlas_dbt/models/marts/marts.yml new file mode 100644 index 0000000..af11a25 --- /dev/null +++ b/dbt/atlas_dbt/models/marts/marts.yml @@ -0,0 +1,43 @@ +version: 2 + +models: + - name: mart_daily_event_metrics + description: > + Daily event metrics at event_date, event_name, country_code, and platform grain. + meta: + governance: + technical_owner: atlas-analytics + business_owner_or_role: atlas-platform + grain: one row per (event_date, event_name, country_code, platform) + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [analytics_mart_readers, atlas_observability_monitor, warehouse_reconciliation] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + tests: + - dbt_utils.unique_combination_of_columns: + arguments: + combination_of_columns: + - event_date + - event_name + - country_code + - platform + columns: + - name: event_date + tests: + - not_null + - name: event_name + tests: + - not_null + - name: country_code + tests: + - not_null + - name: platform + tests: + - not_null + - name: event_count + tests: + - not_null diff --git a/dbt/atlas_dbt/models/sources/sources.yml b/dbt/atlas_dbt/models/sources/sources.yml new file mode 100644 index 0000000..692a91f --- /dev/null +++ b/dbt/atlas_dbt/models/sources/sources.yml @@ -0,0 +1,73 @@ +version: 2 + +sources: + - name: atlas_raw + description: > + Immutable Sprint 1 raw event landing table. Physical-row grain; append-only. + Known seeded anomalies are preserved for downstream classification and quarantine. + database: "{{ env_var('ATLAS_GCP_PROJECT_ID') }}" + schema: "{{ env_var('ATLAS_BQ_DATASET', 'atlas_raw') }}" + loader: atlas_sprint1_pipeline + loaded_at_field: ingested_at + config: + freshness: + warn_after: { count: 24, period: hour } + error_after: { count: 48, period: hour } + tables: + - name: events + description: > + Raw mobile/web analytics events partitioned by event_date and clustered + by event_name and country_code. One row per physical ingest record. + config: + freshness: + warn_after: { count: 24, period: hour } + error_after: { count: 48, period: hour } + columns: + - name: event_id + description: Business event identifier; duplicates may exist at raw grain. + tests: + - not_null + - name: user_id + description: Nullable user identifier; null values are blocking defects. + - name: event_name + description: Canonical event type label. + tests: + - not_null + - name: event_timestamp + description: Event occurrence timestamp in UTC. + tests: + - not_null + - name: event_date + description: Declared calendar date used for partitioning. + tests: + - not_null + - name: country_code + description: ISO-style country code; may be invalid or null. + - name: platform + description: Client platform (ios, android, web). + - name: app_version + description: Application version string. + - name: ingested_at + description: BigQuery load timestamp added by the Sprint 1 loader. + tests: + - not_null + - name: source_file + description: GCS object URI for lineage. + tests: + - not_null + - name: pipeline_run_id + description: Atlas pipeline run identifier for idempotency and scoping. + tests: + - not_null + - name: batch_id + description: > + Stable batch identifier for Airflow-orchestrated loads (nullable for Sprint 1 rows). + tests: + - dbt_utils.expression_is_true: + expression: "is not null" + config: + where: "pipeline_run_id like 'atlas-airflow-%'" + - name: processing_date + description: > + Logical batch processing date; drives reproducible temporal anomaly + flags. Nullable for legacy Sprint 1 rows. diff --git a/dbt/atlas_dbt/models/staging/staging.yml b/dbt/atlas_dbt/models/staging/staging.yml new file mode 100644 index 0000000..d3cdb4c --- /dev/null +++ b/dbt/atlas_dbt/models/staging/staging.yml @@ -0,0 +1,114 @@ +version: 2 + +models: + - name: stg_events + description: > + Staged raw events at physical-row grain with normalized types, lineage metadata, + a stable raw_record_hash, and corrected Sprint 2 temporal quality flags. + meta: + governance: + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per raw physical event record (event_id may repeat) + classification: INTERNAL + retention_class: canonical_warehouse + contract_version: "1.0" + consumers: [warehouse_reconciliation] + lifecycle_status: ACTIVE + freshness_expectation: per batch + runbook: docs/runbook-sprint2.md + last_reviewed: "2026-07-19" + config: + contract: + enforced: true + columns: + - name: event_id + description: Business event identifier from the raw landing table. + data_type: string + tests: + - not_null + - name: user_id + description: Trimmed user identifier; null indicates a blocking defect downstream. + data_type: string + - name: event_name + description: Normalized event type label. + data_type: string + tests: + - not_null + - name: event_timestamp + description: Event occurrence timestamp in UTC. + data_type: timestamp + tests: + - not_null + - name: event_date + description: Declared calendar date used for partitioning. + data_type: date + tests: + - not_null + - name: country_code + description: Uppercase country code when present. + data_type: string + - name: platform + description: Lowercase client platform label. + data_type: string + - name: app_version + description: Application version string. + data_type: string + - name: ingested_at + description: BigQuery load timestamp from the raw layer. + data_type: timestamp + tests: + - not_null + - name: source_file + description: GCS object URI for lineage. + data_type: string + tests: + - not_null + - name: pipeline_run_id + description: Atlas pipeline run identifier. + data_type: string + tests: + - not_null + - name: batch_id + description: Stable batch identifier for Airflow-orchestrated loads; nullable for Sprint 1 rows. + data_type: string + - name: processing_date + description: > + Logical batch processing date used as the reference for temporal anomaly + flags. Null for legacy Sprint 1 rows (which fall back to DATE(ingested_at)). + data_type: date + - name: raw_record_hash + description: Stable fingerprint for duplicate tie-breaking and lineage. + data_type: int64 + tests: + - not_null + - name: event_timestamp_date + description: Calendar date derived from event_timestamp. + data_type: date + tests: + - not_null + - name: ingested_date + description: Calendar date derived from ingested_at. + data_type: date + tests: + - not_null + - name: is_future_dated + description: Blocking flag when DATE(event_timestamp) > DATE(ingested_at). + data_type: boolean + tests: + - not_null + - name: is_event_time_late_arriving + description: Warning flag when DATE(event_timestamp) < DATE(ingested_at). + data_type: boolean + tests: + - not_null + - name: is_backdated_event_date + description: Warning flag when event_date < DATE(ingested_at). + data_type: boolean + tests: + - not_null + - name: has_event_date_timestamp_mismatch + description: Warning flag when event_date != DATE(event_timestamp). + data_type: boolean + tests: + - not_null diff --git a/dbt/atlas_dbt/models/staging/stg_events.sql b/dbt/atlas_dbt/models/staging/stg_events.sql new file mode 100644 index 0000000..cdd6405 --- /dev/null +++ b/dbt/atlas_dbt/models/staging/stg_events.sql @@ -0,0 +1,44 @@ +with source as ( + select * from {{ source('atlas_raw', 'events') }} +), + +normalized as ( + select + event_id, + nullif(trim(user_id), '') as user_id, + trim(event_name) as event_name, + event_timestamp, + event_date, + upper(nullif(trim(country_code), '')) as country_code, + lower(nullif(trim(platform), '')) as platform, + nullif(trim(app_version), '') as app_version, + ingested_at, + source_file, + pipeline_run_id, + batch_id, + processing_date, + farm_fingerprint( + concat( + coalesce(event_id, ''), + '|', + coalesce(cast(event_timestamp as string), ''), + '|', + coalesce(source_file, ''), + '|', + coalesce(pipeline_run_id, '') + ) + ) as raw_record_hash, + date(event_timestamp) as event_timestamp_date, + date(ingested_at) as ingested_date, + -- Temporal quality flags are evaluated against the logical batch + -- processing_date (falling back to DATE(ingested_at) for legacy rows). + -- This makes classification reproducible for historical backfills: + -- the same raw batch yields identical flags regardless of load time. + date(event_timestamp) > coalesce(processing_date, date(ingested_at)) as is_future_dated, + date(event_timestamp) < coalesce(processing_date, date(ingested_at)) as is_event_time_late_arriving, + event_date < coalesce(processing_date, date(ingested_at)) as is_backdated_event_date, + event_date != date(event_timestamp) as has_event_date_timestamp_mismatch + from source +) + +select * from normalized diff --git a/dbt/atlas_dbt/package-lock.yml b/dbt/atlas_dbt/package-lock.yml new file mode 100644 index 0000000..7bf509f --- /dev/null +++ b/dbt/atlas_dbt/package-lock.yml @@ -0,0 +1,5 @@ +packages: + - name: dbt_utils + package: dbt-labs/dbt_utils + version: 1.4.1 +sha1_hash: 8b27037b26f3f630c6661194d2470e720c49f6ee diff --git a/dbt/atlas_dbt/packages.yml b/dbt/atlas_dbt/packages.yml new file mode 100644 index 0000000..60c4d13 --- /dev/null +++ b/dbt/atlas_dbt/packages.yml @@ -0,0 +1,3 @@ +packages: + - package: dbt-labs/dbt_utils + version: 1.4.1 diff --git a/dbt/atlas_dbt/profiles.yml.example b/dbt/atlas_dbt/profiles.yml.example new file mode 100644 index 0000000..a5af3b3 --- /dev/null +++ b/dbt/atlas_dbt/profiles.yml.example @@ -0,0 +1,13 @@ +atlas_dbt: + target: dev + outputs: + dev: + type: bigquery + method: oauth + project: "{{ env_var('ATLAS_GCP_PROJECT_ID') }}" + dataset: "{{ env_var('ATLAS_DBT_DATASET', 'atlas') }}" + location: "{{ env_var('DBT_LOCATION') }}" + threads: 4 + priority: interactive + job_execution_timeout_seconds: 300 + job_retries: 1 diff --git a/dbt/atlas_dbt/seeds/seeds.yml b/dbt/atlas_dbt/seeds/seeds.yml new file mode 100644 index 0000000..4dc72a5 --- /dev/null +++ b/dbt/atlas_dbt/seeds/seeds.yml @@ -0,0 +1,21 @@ +version: 2 + +seeds: + - name: valid_country_codes + description: > + Authoritative ISO-style country allowlist sourced from anomaly_profile.yaml. + Used to classify invalid country_code values during event quality review. + columns: + - name: country_code + description: Two-letter country code. + tests: + - not_null + - unique + - name: is_active + description: Whether the country is currently accepted in Atlas core models. + tests: + - not_null + - accepted_values: + arguments: + values: [true, false] + quote: false diff --git a/dbt/atlas_dbt/seeds/valid_country_codes.csv b/dbt/atlas_dbt/seeds/valid_country_codes.csv new file mode 100644 index 0000000..5bc4bfb --- /dev/null +++ b/dbt/atlas_dbt/seeds/valid_country_codes.csv @@ -0,0 +1,11 @@ +country_code,is_active +US,true +CA,true +GB,true +AU,true +DE,true +FR,true +BR,true +MX,true +IN,true +JP,true diff --git a/dbt/atlas_dbt/tests/assert_batch_fact_reconciliation.sql b/dbt/atlas_dbt/tests/assert_batch_fact_reconciliation.sql new file mode 100644 index 0000000..6dce043 --- /dev/null +++ b/dbt/atlas_dbt/tests/assert_batch_fact_reconciliation.sql @@ -0,0 +1,22 @@ +{% set use_batch = var('validated_batch_id', '') != '' %} +{% set scope_column = 'batch_id' if use_batch else 'pipeline_run_id' %} +{% set scope_value = var('validated_batch_id') if use_batch else var('validated_run_id') %} + +-- Fails when batch-scoped fact rows do not reconcile to accepted events (when batch scope active). +with accepted_count as ( + select count(*) as row_count + from {{ ref('int_accepted_events') }} + where {{ scope_column }} = '{{ scope_value }}' +), + +fact_count as ( + select count(*) as row_count + from {{ ref('fct_events') }} f + inner join {{ ref('int_accepted_events') }} a using (event_id) + where a.{{ scope_column }} = '{{ scope_value }}' +) + +select accepted_count.row_count as accepted_rows, fact_count.row_count as fact_rows +from accepted_count +cross join fact_count +where {% if use_batch %}accepted_count.row_count != fact_count.row_count{% else %}1 = 0{% endif %} diff --git a/dbt/atlas_dbt/tests/assert_fact_rejected_reconciliation.sql b/dbt/atlas_dbt/tests/assert_fact_rejected_reconciliation.sql new file mode 100644 index 0000000..3d2062c --- /dev/null +++ b/dbt/atlas_dbt/tests/assert_fact_rejected_reconciliation.sql @@ -0,0 +1,32 @@ +{% set use_batch = var('validated_batch_id', '') != '' %} +{% set scope_column = 'batch_id' if use_batch else 'pipeline_run_id' %} +{% set scope_value = var('validated_batch_id') if use_batch else var('validated_run_id') %} + +-- Fails when accepted canonical rows plus rejected physical rows do not equal raw rows. +with raw_count as ( + select count(*) as row_count + from {{ source('atlas_raw', 'events') }} + where {{ scope_column }} = '{{ scope_value }}' +), + +accepted_count as ( + select count(*) as row_count + from {{ ref('int_accepted_events') }} + where {{ scope_column }} = '{{ scope_value }}' +), + +rejected_count as ( + select count(*) as row_count + from {{ ref('int_rejected_events') }} + where {{ scope_column }} = '{{ scope_value }}' +) + +select + raw_count.row_count as raw_rows, + accepted_count.row_count as accepted_rows, + rejected_count.row_count as rejected_rows, + accepted_count.row_count + rejected_count.row_count as accepted_plus_rejected +from raw_count +cross join accepted_count +cross join rejected_count +where raw_count.row_count != accepted_count.row_count + rejected_count.row_count diff --git a/dbt/atlas_dbt/tests/assert_inject_failure.sql b/dbt/atlas_dbt/tests/assert_inject_failure.sql new file mode 100644 index 0000000..6c97d49 --- /dev/null +++ b/dbt/atlas_dbt/tests/assert_inject_failure.sql @@ -0,0 +1,4 @@ +-- Fails when inject_failure var is enabled (local failure simulation only). +select 1 as failure_injected +from unnest([1]) +where {{ var('inject_failure', false) }} diff --git a/dbt/atlas_dbt/tests/assert_mart_fact_reconciliation.sql b/dbt/atlas_dbt/tests/assert_mart_fact_reconciliation.sql new file mode 100644 index 0000000..c72c25c --- /dev/null +++ b/dbt/atlas_dbt/tests/assert_mart_fact_reconciliation.sql @@ -0,0 +1,17 @@ +-- Fails when mart totals do not reconcile to the accepted fact table. +with fact_count as ( + select count(*) as row_count + from {{ ref('fct_events') }} +), + +mart_total as ( + select coalesce(sum(event_count), 0) as row_count + from {{ ref('mart_daily_event_metrics') }} +) + +select + fact_count.row_count as fact_rows, + mart_total.row_count as mart_event_total +from fact_count +cross join mart_total +where fact_count.row_count != mart_total.row_count diff --git a/dbt/atlas_dbt/tests/assert_raw_classification_reconciliation.sql b/dbt/atlas_dbt/tests/assert_raw_classification_reconciliation.sql new file mode 100644 index 0000000..78be3b1 --- /dev/null +++ b/dbt/atlas_dbt/tests/assert_raw_classification_reconciliation.sql @@ -0,0 +1,23 @@ +{% set use_batch = var('validated_batch_id', '') != '' %} +{% set scope_column = 'batch_id' if use_batch else 'pipeline_run_id' %} +{% set scope_value = var('validated_batch_id') if use_batch else var('validated_run_id') %} + +-- Fails when raw physical rows do not reconcile to classification rows for the validated scope. +with raw_count as ( + select count(*) as row_count + from {{ source('atlas_raw', 'events') }} + where {{ scope_column }} = '{{ scope_value }}' +), + +classification_count as ( + select count(*) as row_count + from {{ ref('int_event_classification') }} + where {{ scope_column }} = '{{ scope_value }}' +) + +select + raw_count.row_count as raw_rows, + classification_count.row_count as classification_rows +from raw_count +cross join classification_count +where raw_count.row_count != classification_count.row_count diff --git a/dbt/atlas_dbt/tests/assert_source_anomaly_profile.sql b/dbt/atlas_dbt/tests/assert_source_anomaly_profile.sql new file mode 100644 index 0000000..097606c --- /dev/null +++ b/dbt/atlas_dbt/tests/assert_source_anomaly_profile.sql @@ -0,0 +1,46 @@ +{% set use_batch = var('validated_batch_id', '') != '' %} +{% set scope_column = 'batch_id' if use_batch else 'pipeline_run_id' %} +{% set scope_value = var('validated_batch_id') if use_batch else var('validated_run_id') %} + +-- Fails when the validated scope does not match the corrected Sprint 2 anomaly profile. +with scoped as ( + select * + from {{ ref('int_event_classification') }} + where {{ scope_column }} = '{{ scope_value }}' +), + +counts as ( + select + -- Sprint 7: the anomaly profile measures WITHIN-BATCH duplicates (the + -- intentional 50 extras). Cross-batch replays are excluded so that a + -- same-date reprocessing batch does not corrupt this batch-scoped + -- assertion (INC-S6-001). + countif(is_within_batch_duplicate) as duplicate_extra_count, + countif(user_id is null) as null_user_count, + countif(not is_valid_country) as invalid_country_count, + countif(is_future_dated) as future_dated_count, + countif(is_event_time_late_arriving) as event_time_late_count, + countif(is_backdated_event_date) as backdated_event_date_count, + countif(has_event_date_timestamp_mismatch) as date_timestamp_mismatch_count, + -- Temporal flags are only reproducible when every scoped row carries a + -- processing_date reference. Legacy rows without it fall back to + -- DATE(ingested_at), which is load-time dependent, so the temporal + -- assertions are skipped for those scopes (graceful degradation). + countif(processing_date is null) as missing_processing_date_count + from scoped +) + +select * +from counts +where duplicate_extra_count != 50 + or null_user_count != 500 + or invalid_country_count != 200 + or date_timestamp_mismatch_count != 300 + or ( + missing_processing_date_count = 0 + and ( + future_dated_count != 150 + or event_time_late_count != 0 + or backdated_event_date_count != 300 + ) + ) diff --git a/dbt/requirements-dbt.txt b/dbt/requirements-dbt.txt new file mode 100644 index 0000000..c6fe82b --- /dev/null +++ b/dbt/requirements-dbt.txt @@ -0,0 +1,2 @@ +dbt-core==1.11.12 +dbt-bigquery==1.11.3 diff --git a/docs/adr/ADR-002-isolated-atlas-dbt-project.md b/docs/adr/ADR-002-isolated-atlas-dbt-project.md new file mode 100644 index 0000000..8aa8b06 --- /dev/null +++ b/docs/adr/ADR-002-isolated-atlas-dbt-project.md @@ -0,0 +1,22 @@ +# ADR-002: Isolated Atlas dbt Project Location + +## Status + +Accepted + +## Context + +The repository already contains a DEOS dbt scaffold at `transform/dbt/`. Sprint 2 needs an +Atlas-specific warehouse with BigQuery datasets, anomaly classification, and quarantine +semantics that must not collide with the DEOS validation project. + +## Decision + +Create a nested dbt project at `dbt/atlas_dbt/` with its own virtual +environment, package lock, and `dbt-atlas` MCP entry. Leave `transform/dbt/` unchanged. + +## Consequences + +- Atlas operators use `scripts/setup_dbt.sh` and `.venv-dbt`. +- DEOS operators continue using the root `.venv` and existing dbt MCP server. +- Documentation must clearly distinguish the two projects. diff --git a/docs/adr/ADR-003-corrected-temporal-semantics.md b/docs/adr/ADR-003-corrected-temporal-semantics.md new file mode 100644 index 0000000..66b30a9 --- /dev/null +++ b/docs/adr/ADR-003-corrected-temporal-semantics.md @@ -0,0 +1,69 @@ +# ADR-003: Corrected Sprint 2 Temporal Quality Semantics + +## Status + +Accepted + +## Context + +Sprint 1 validation labeled 300 rows as "late_arriving_events" using +`event_date < DATE(event_timestamp)`. Those rows are backdated declared dates, not true +event-time late arrivals relative to ingestion time. + +## Decision + +Sprint 2 dbt models use corrected flags: + +| Flag | Definition | Expected on validated run | Blocking | +| --- | --- | ---: | --- | +| `is_future_dated` | `DATE(event_timestamp) > DATE(ingested_at)` | 150 | Yes | +| `is_event_time_late_arriving` | `DATE(event_timestamp) < DATE(ingested_at)` | 0 | No | +| `is_backdated_event_date` | `event_date < DATE(ingested_at)` | 300 | No | +| `has_event_date_timestamp_mismatch` | `event_date != DATE(event_timestamp)` | 300 | No | + +The 300 backdated and 300 mismatch populations are the same physical records and must not +be double-counted during reconciliation. + +## Consequences + +- Sprint 1 generator and raw data remain unchanged for auditability. +- Sprint 2 documentation and singular tests use the corrected definitions. +- Warning flags may coexist on accepted canonical rows. + +## Sprint 3 refinement — reproducible temporal semantics for backfills + +### Context + +The Sprint 2 flags above reference wall-clock `DATE(ingested_at)`. That makes +event classification (accept/reject) a function of *when the pipeline physically +ran*: a historical batch (e.g. `processing_date = 2026-07-01`) ingested on +2026-07-18 flips `future_dated` 150→0, `event_time_late` 0→~50000, and +`backdated` 300→~50000. This broke the Sprint 3 batch-identity/backfill guarantee +(ADR-006): reprocessing the same raw batch produced different `fct_events`/marts +and failed `assert_source_anomaly_profile`. + +### Decision + +Evaluate the three ingestion-relative flags against the batch's **logical +processing date** instead of wall-clock ingest time: + +| Flag | Sprint 3 definition | +| --- | --- | +| `is_future_dated` | `DATE(event_timestamp) > COALESCE(processing_date, DATE(ingested_at))` | +| `is_event_time_late_arriving` | `DATE(event_timestamp) < COALESCE(processing_date, DATE(ingested_at))` | +| `is_backdated_event_date` | `event_date < COALESCE(processing_date, DATE(ingested_at))` | +| `has_event_date_timestamp_mismatch` | `event_date != DATE(event_timestamp)` (unchanged, already reproducible) | + +`processing_date` is a new nullable column on `atlas_raw.events`, persisted by the +loader per batch. Legacy Sprint 1 rows have `processing_date = NULL` and fall back +to `DATE(ingested_at)`, preserving prior behavior. `assert_source_anomaly_profile` +asserts the temporal counts only when every scoped row carries `processing_date`, +degrading gracefully for legacy scopes. + +### Consequences + +- Historical backfills classify identically to the original run (verified live: + batch `atlas-20260701` recovered `FAILED`→`SUCCESS`; fresh historical batch + `atlas-20260716` loaded and passed with native `processing_date`). +- For same-day batches `processing_date == DATE(ingested_at)`, so the expected + 150 / 0 / 300 / 300 profile is unchanged. diff --git a/docs/adr/ADR-004-no-snapshots-sprint2.md b/docs/adr/ADR-004-no-snapshots-sprint2.md new file mode 100644 index 0000000..3476c18 --- /dev/null +++ b/docs/adr/ADR-004-no-snapshots-sprint2.md @@ -0,0 +1,22 @@ +# ADR-004: Omission of dbt Snapshots in Sprint 2 + +## Status + +Accepted + +## Context + +Project Atlas Sprint 2 focuses on governed staging, classification, quarantine, and trusted +facts/marts over an immutable raw landing table. Historical slowly-changing tracking for +users and countries is out of scope for the first warehouse sprint. + +## Decision + +Do not add dbt snapshots in Sprint 2. User and country dimensions are rebuilt from accepted +events and the country seed on each build. Incremental behavior is limited to `fct_events`. + +## Consequences + +- Faster delivery of classification and reconciliation gates. +- Future sprints can introduce snapshots or Type 2 dimensions if product requirements change. +- Airflow handoff can trigger full dimension rebuilds until snapshot coverage exists. diff --git a/docs/adr/ADR-005-airflow-composer-parity.md b/docs/adr/ADR-005-airflow-composer-parity.md new file mode 100644 index 0000000..1d17240 --- /dev/null +++ b/docs/adr/ADR-005-airflow-composer-parity.md @@ -0,0 +1,41 @@ +# ADR-005: Airflow Composer Parity Pins + +## Status + +Accepted — 2026-07-14 +Amended — 2026-07-18 (Sprint 4: Composer image revised to `build.13`) + +## Context + +Sprint 3 introduces local Airflow orchestration that must behave consistently with +Cloud Composer before production deployment. + +## Decision + +Pin local Airflow to **3.1.7** with Google provider **20.0.0** and Standard provider +**1.12.1**, targeting Composer image `composer-3-airflow-3.1.7-build.12`. + +## Sprint 4 amendment — 2026-07-18 + +The Sprint 4 preflight verified via the Composer API (`us-central1`) that +`composer-3-airflow-3.1.7-build.12` is **no longer offered**. The only available +Composer 3 image carrying Airflow 3.1.7 is: + +``` +composer-3-airflow-3.1.7-build.13 +``` + +Decision (owner-approved 2026-07-18): target **`composer-3-airflow-3.1.7-build.13`** +for the Sprint 4 managed deployment. The Airflow core version (3.1.7) and the +provider pins above are unchanged; only the Composer build number moved. Provider +compatibility must be re-verified against the live environment during the Sprint 4 +smoke run before the release tag is created. + +Install core using official Python 3.12 constraints, then apply provider pins and +record `pip check` output in the preflight report. + +## Consequences + +- Local Python 3.12.3 differs from Composer Python 3.11.8; parse-time helpers must + remain compatible with both. +- Revisit this ADR if the Composer image is retired or upgraded. diff --git a/docs/adr/ADR-006-batch-identity.md b/docs/adr/ADR-006-batch-identity.md new file mode 100644 index 0000000..d10b1a1 --- /dev/null +++ b/docs/adr/ADR-006-batch-identity.md @@ -0,0 +1,66 @@ +# ADR-006: Stable Batch Identity and Immutable Ingestion + +## Status + +Accepted — 2026-07-14 + +## Context + +Sprint 1 keyed raw loads on `pipeline_run_id`, preventing safe reruns of the same +data batch under a new execution identity. + +## Decision + +Introduce `batch_id` as the stable data identity and keep `pipeline_run_id` as the +execution identity. GCS paths use `batch_id=` prefixes with checksum metadata. +Raw loads evaluate batch row counts (0/load, exact/skip, partial-fail, excess-fail). + +## Consequences + +- Sprint 1 rows retain `batch_id IS NULL`. +- dbt models propagate `batch_id` for Airflow-scoped reconciliation tests. + +## Sprint 7 amendment: replay and duplicate semantics + +INC-S6-001 exposed a defect: `int_event_classification` computed a single +**global** duplicate rank and the batch-scoped anomaly profile counted it, so a +same-date reprocessing batch (whose deterministic generator produces identical +`event_id`s) inflated the within-batch duplicate count to the whole batch and +failed `assert_source_anomaly_profile`. Global fact uniqueness was fine; the +classification/measurement conflated two distinct concepts. + +Sprint 7 makes the semantics explicit. Every classified row now carries: + +- `within_batch_duplicate_rank` — `row_number()` partitioned by + `(coalesce(batch_id, pipeline_run_id), event_id)`, latest write wins. +- `is_within_batch_duplicate` — `within_batch_duplicate_rank > 1`. This is the + **batch-scoped data-quality anomaly** (the 50 intentional extras). +- `duplicate_rank` — global canonical rank per `event_id`, ordered + `within_batch_duplicate_rank asc, ingested_at asc, …`. **First-seen batch + wins**, so a replay never disturbs an already-published canonical prior batch; + within a batch, latest still wins. +- `is_duplicate_extra` — `duplicate_rank > 1` (feeds `rejection_reason`; + preserves exactly one canonical row per `event_id`). +- `duplicate_scope` — `within_batch` | `cross_batch_replay` | `none`. + +The anomaly profile now counts `is_within_batch_duplicate` (batch-scoped), so a +stray same-date batch no longer corrupts a healthy batch's profile, and +cross-batch replays are separately measurable via `duplicate_scope`. + +### Invariants (unchanged or newly guaranteed) + +| Invariant | How preserved | +| --- | --- | +| Exact batch rerun idempotent | raw load is create-only per batch; classification deterministic | +| Same date, different batch | classified as `cross_batch_replay`, rejected; first batch stays canonical | +| 50 within-batch extras detectable | `is_within_batch_duplicate` counts exactly them | +| Cross-batch replay measurable | `duplicate_scope = 'cross_batch_replay'` | +| Global fact uniqueness | `fct_events` `unique_key = event_id`; one canonical accepted row | +| accepted + rejected = raw | classification preserves physical-row grain | +| Prior healthy batches stable | first-seen-wins canonical ordering | +| Historical backfills deterministic | ordering uses only stable row attributes | + +Tested by dbt unit tests (`test_cross_batch_replay_preserves_first_seen`, +`test_duplicate_ranking_keeps_latest_canonical`, precedence/warning tests) and +verified live in the Sprint 7 acceptance window. `fct_events` grain is unchanged +(one row per `event_id`); this amendment does not change that grain. diff --git a/docs/adr/ADR-007-pipeline-runs-audit.md b/docs/adr/ADR-007-pipeline-runs-audit.md new file mode 100644 index 0000000..c0259e6 --- /dev/null +++ b/docs/adr/ADR-007-pipeline-runs-audit.md @@ -0,0 +1,20 @@ +# ADR-007: Durable Operational Audit Table + +## Status + +Accepted — 2026-07-14 + +## Context + +Sprint 3 requires one auditable row per DAG execution with local JSON reconciliation. + +## Decision + +Create `atlas_ops.pipeline_runs` now (not deferred to Composer). Split initialization: +`ensure_audit_resources` → `start_run_audit` → `preflight_environment`. +Finalizer upserts terminal status and raises on `FAILED`/`PARTIAL`. + +## Consequences + +- MERGE keyed on `pipeline_run_id` supports idempotent finalization. +- Error messages sanitized and truncated to 2,000 characters. diff --git a/docs/adr/ADR-008-github-actions-validation-boundary.md b/docs/adr/ADR-008-github-actions-validation-boundary.md new file mode 100644 index 0000000..36170df --- /dev/null +++ b/docs/adr/ADR-008-github-actions-validation-boundary.md @@ -0,0 +1,93 @@ +# ADR-008: GitHub Actions Validation Boundary + +## Status + +Accepted — 2026-07-18 + +## Context + +Sprint 4 introduces independent CI on GitHub. The delivery system must be +auditable, reproducible outside GitHub, and safe against untrusted +pull-request code and supply-chain drift. + +## Decision + +### Validation logic lives in repository scripts + +`scripts/validate_ci.sh` is the canonical validation contract. +Workflows only provision pinned toolchains and invoke it with a gate group +(`security-shell`, `python`, `airflow`, `dbt`). Consequences: + +- Cursor Cloud Agents, local developers, and CI run byte-identical gates. +- Workflow YAML carries no business or validation logic that could drift + from what engineers run locally. +- A requested gate group whose toolchain is missing **fails** rather than + skips, so CI cannot silently pass by not installing a tool. + +### Pull-request CI is credentialless + +`atlas-ci.yml` sets `permissions: contents: read` at the workflow level and +uses no GCP credentials, no service-account JSON, and no repository secrets. +Untrusted pull-request code therefore executes with nothing to exfiltrate. +GCP integration testing runs only from trusted workflow code on `main` (or a +SHA verified reachable from `main`) via Workload Identity Federation +(ADR-009). `pull_request_target` is never used to execute untrusted code +with credentials. + +### Version pinning + +| Component | Pin | Source | +|---|---|---| +| Python | 3.12 (Composer parity gap with 3.11.8 documented in ADR-005) | `setup-python` | +| apache-airflow | 3.1.7 + official `constraints-3.12.txt` | `airflow/requirements-airflow.txt` | +| Google provider | 20.0.0 | same | +| Standard provider | 1.12.1 | same | +| dbt-core / dbt-bigquery | 1.11.12 / 1.11.3 | `dbt/requirements-dbt.txt` | +| dbt_utils | 1.4.1 | `packages.yml` | +| ruff / mypy / yamllint / shellcheck-py / pytest | see `requirements-ci.txt` | verified locally 2026-07-18 | + +Versions are upgraded deliberately, never because newer versions exist; +Composer-target compatibility (ADR-005) always wins. + +### Action supply-chain controls + +- Every third-party action is pinned to an immutable full commit SHA with a + comment naming the release tag. +- SHAs were resolved via `gh api repos///git/ref/tags/`, + dereferencing annotated tags to commit objects, on 2026-07-18: + - `actions/checkout@v5` → `93cb6efe18208431cddfb8368fd83d5badbf9bfd` + - `actions/setup-python@v6` → `ece7cb06caefa5fff74198d8649806c4678c61a1` + - `actions/upload-artifact@v4` → `ea165f8d65b6e75b540449e92b4886f43607fa02` +- Floating tags (`@v4`) are never used for execution. + +### Cache boundaries + +- `setup-python` pip caching is keyed on the pinned requirements files. +- Caches contain only public package downloads — never credentials, tokens, + or workspace state. No credential material exists in PR CI to leak. + +### Trusted versus untrusted execution + +| Context | Code | Credentials | +|---|---|---| +| `pull_request` CI | untrusted (fork/branch) | none (read-only token) | +| `push` to `main` CI | trusted, reviewed | none (CI needs none) | +| Deployment / integration workflows | trusted `main`-reachable SHAs only, `workflow_dispatch` | short-lived WIF tokens (ADR-009) | + +### Static-mode dbt boundary + +`dbt deps` + `dbt parse` run in static CI with a placeholder oauth profile +(parse never opens a warehouse connection). `dbt compile` and dbt unit tests +against BigQuery require a live adapter connection, so they run in +integration mode with isolated `atlas_ci_` resources instead of in +credentialless PR CI. This is a deliberate deviation from "compile in static +mode": compiling BigQuery incremental models introspects relations and +cannot be done credential-free without mocking that would weaken the gate. + +## Consequences + +- A green `atlas-ci-gate` check is reproducible locally with + `bash scripts/validate_ci.sh --mode static`. +- Adding a new gate means editing one script, and every consumer inherits it. +- Action upgrades are explicit diffs of full SHAs, reviewable against + upstream release notes. diff --git a/docs/adr/ADR-009-workload-identity-federation.md b/docs/adr/ADR-009-workload-identity-federation.md new file mode 100644 index 0000000..d0f962b --- /dev/null +++ b/docs/adr/ADR-009-workload-identity-federation.md @@ -0,0 +1,106 @@ +# ADR-009: Keyless GitHub-to-GCP Authentication via Workload Identity Federation + +- **Status:** Accepted (Sprint 4) +- **Date:** 2026-07-18 +- **Deciders:** Project owner + Sprint 4 delivery agent (IAM approved by owner at every stage) + +## Context + +Sprint 4 requires GitHub Actions to authenticate to GCP project +`example-gcp-project` for isolated integration testing and controlled +Composer deployment. Storing a Google service-account JSON key as a GitHub +secret is a long-lived, exfiltratable credential and is prohibited by the +Sprint 4 mission statement. + +## Decision + +Use GitHub's OIDC token issuer with Google Workload Identity Federation and +short-lived service-account impersonation. No service-account key is ever +created. + +### Resources (created live by `scripts/bootstrap_github_wif.sh`) + +| Resource | Value | +|---|---| +| Project | `example-gcp-project` (number `123456789012`) | +| WIF pool | `atlas-github-pool` (global) | +| WIF provider | `atlas-github-provider` (OIDC, issuer `https://token.actions.githubusercontent.com`) | +| Integration SA | `atlas-github-integration@example-gcp-project.iam.gserviceaccount.com` | +| Deployer SA | `atlas-github-deployer@example-gcp-project.iam.gserviceaccount.com` | +| Deployment bucket | `gs://atlas-deployments-example-gcp-project` (versioned, uniform access) | +| CI bucket | `gs://atlas-ci-example-gcp-project` (uniform access, 7-day object TTL) | + +### Trust conditions + +Two layers, both required: + +1. **Provider attribute condition** — tokens are rejected at the pool boundary + unless + `assertion.repository_owner == 'YOUR_GITHUB_OWNER' && assertion.repository == 'YOUR_GITHUB_OWNER/YOUR_REPOSITORY'`. + The owner check guards against repository transfer/rename attacks. +2. **Per-SA impersonation binding** — `roles/iam.workloadIdentityUser` is + granted only to the principal set + `attribute.repository_and_ref/YOUR_GITHUB_OWNER/YOUR_REPOSITORY@refs/heads/main`, + using a custom mapped claim `repository_and_ref = assertion.repository + '@' + assertion.ref`. + Pull-request runs (`refs/pull/...`) and any other refs cannot impersonate + either service account, which enforces the "trusted workflow code only" + rule from Phase 7: only workflows executing code already merged to `main` + can obtain GCP credentials. + +Mapped claims: `sub`, `repository`, `repository_owner`, `ref`, +`repository_and_ref`. + +### Identity separation and least privilege + +Two identities because integration testing and deployment have different +blast radii: + +| Role | `atlas-github-integration` | `atlas-github-deployer` | +|---|---|---| +| `roles/bigquery.jobUser` (project) | yes | yes | +| `roles/bigquery.dataEditor` (project) | yes † | yes † | +| `roles/storage.admin` on CI bucket | yes | no | +| `roles/storage.admin` on deployment bucket | no | yes | +| `roles/composer.user` (project) | no | yes | +| `roles/composer.environmentAndStorageObjectAdmin` (project) | no | yes | + +Neither identity has Owner, Editor, Project IAM Admin, organization roles, or +service-account-key administration. Neither can mint keys or escalate IAM. + +† **Documented risk:** BigQuery offers no IAM primitive that allows +"create datasets matching `atlas_ci_*` only". `roles/bigquery.dataEditor` at +project scope is the minimum role that lets the integration identity create +its ephemeral `atlas_ci_` datasets, and it also technically permits +writes to canonical datasets. Compensating controls: (a) only `main`-ref +workflows can impersonate the SA, so the code path is repository-controlled +and reviewed; (b) the integration script derives all dataset names from +`GITHUB_RUN_ID` and never references canonical dataset names in write paths; +(c) all integration activity is auditable in Cloud Logging under the SA +identity. A future hardening option is a dedicated CI project. + +### GitHub workflow contract + +Jobs that authenticate must set exactly: + +```yaml +permissions: + contents: read + id-token: write +``` + +and use `google-github-actions/auth` (pinned to a full commit SHA) with the +committed provider resource name and SA email. These identifiers are not +secrets — possession of them grants nothing without a token that satisfies +the conditions above — so they live in version-controlled workflow files rather +than GitHub secrets, which also keeps pull-request CI credentialless. + +## Consequences + +- Pull-request CI remains credentialless by construction (PR refs cannot + impersonate). +- Rotating trust requires editing IAM bindings, not rotating secrets. +- `bootstrap_github_wif.sh` is idempotent and plan-first + (`ATLAS_APPROVE_IAM=true` required for mutation), so drift can be repaired + by re-running it. +- The final end-to-end proof is a real GitHub Actions run exchanging an OIDC + token; captured as Phase 7/19 evidence in the validation report. diff --git a/docs/adr/ADR-010-versioned-deployment-and-rollback.md b/docs/adr/ADR-010-versioned-deployment-and-rollback.md new file mode 100644 index 0000000..624f6db --- /dev/null +++ b/docs/adr/ADR-010-versioned-deployment-and-rollback.md @@ -0,0 +1,93 @@ +# ADR-010: Versioned Deployment, Rollback, and Ephemeral Composer Evidence + +## Status + +Accepted — 2026-07-18 (owner decisions recorded verbatim; implementation lands in +Sprint 4) + +## Context + +Sprint 4 introduces the first automated GitHub-to-GCP delivery path for the Atlas +batch pipeline. Three owner decisions taken on 2026-07-18 constrain the design: + +1. **Composer is ephemeral.** The managed Composer environment exists only to + capture live deployment, smoke, and rollback evidence. It must not stay alive + and compound cost ("that's a hard no"). Sprint 4 completion is claimed on + durable evidence, not on a permanently running environment. +2. **Composer image is `composer-3-airflow-3.1.7-build.13`.** The originally + pinned `build.12` was retired upstream; see the ADR-005 amendment. +3. **The repository is on the GitHub Free plan.** Required reviewers on GitHub + Environments and full branch-protection rules are not enforceable on private + Free-plan repositories. Deployment governance therefore uses the fallback + controls described below, and documentation must not claim reviewer gates + that the plan cannot enforce. + +## Decision + +### Immutable versioned releases + +- Every deployment builds a deterministic bundle containing only runtime assets + (DAGs, `src/atlas`, runtime scripts, `dbt/atlas_dbt`, config, approved SQL + migrations, dependency manifests) plus a `release-manifest.json` carrying the + git SHA, versions, file checksums, and schema-compatibility declarations. +- Bundles are stored create-only under + `gs:///atlas/releases//`. An existing release + path with a matching checksum is reused; a differing checksum fails the build. + Nothing is ever overwritten. +- The mutable Composer runtime path (`data/current/`) is always a + promoted copy of one immutable release. Rollback re-promotes a prior release; + it never mutates or deletes historical bundles, moves tags, or rewrites git + history. + +### Rollback rules + +- Runtime rollback is permitted only when the prior release's manifest declares + compatibility with the currently applied schema version + (`min_compatible_schema_version`). +- BigQuery migrations are additive by default and are never automatically + reversed. A rollback that would require reversing a destructive migration is a + manual operator decision. +- A rollback is only claimed successful after its own smoke batch reaches + terminal `SUCCESS` and `atlas_ops.deployments` records `ROLLED_BACK`. + +### Ephemeral Composer lifecycle (cost control) + +- The Composer environment is created (gated by + `ATLAS_APPROVE_COMPOSER_CREATE=true`) only when the delivery pipeline is ready + for live acceptance, and is **deleted after the evidence bundle is captured**. +- The permanent record of the deployment is the durable evidence, not the + environment: `atlas_ops.deployments` and `atlas_ops.pipeline_runs` rows, + immutable release bundles in GCS, GitHub Actions run logs and artifacts, and + `docs/validation-report-sprint4.md`. +- Re-verification at any later date follows the documented runbook: recreate the + environment from the pinned image, promote the tagged immutable release, rerun + the smoke batch, delete the environment. +- Estimated cost of the evidence-capture window (small Composer 3 environment, + measured in hours, not months) is recorded in the validation report. Leaving + the environment running (~$350–450/month) is explicitly rejected. + +### Free-plan governance fallback + +Because required environment reviewers and branch-protection API access are +unavailable on this plan: + +- Deployment workflows trigger only via `workflow_dispatch` with an exact typed + confirmation input, verify the target SHA is reachable from `origin/main`, and + serialize under an `atlas-dev-deployment` concurrency group. +- No deployment triggers automatically from a pull request. +- The absence of enforced reviewer gates and branch protection is documented as + an unresolved governance limitation in `docs/ci-cd-governance-sprint4.md`, + together with the exact settings to enable if the repository is upgraded. + +## Consequences + +- Sprint 4's defensible claim is evidence-based: a validated change moved from an + agent-created branch through independent CI and keyless GCP authentication to + a real Composer deployment, smoke run, and tested rollback — all durable in + audit tables, GCS, and workflow logs — even though the environment itself is + deleted afterward. +- Anyone re-running acceptance must budget for environment creation time + (typically ~25 minutes for Composer 3) plus the smoke matrix. +- If the GitHub plan is upgraded, governance should be revisited: enable branch + protection on `main`, require the `atlas-ci-gate` check, and add required + reviewers to the `atlas-dev` environment. diff --git a/docs/adr/ADR-011-atlas-observability-model.md b/docs/adr/ADR-011-atlas-observability-model.md new file mode 100644 index 0000000..2c330c0 --- /dev/null +++ b/docs/adr/ADR-011-atlas-observability-model.md @@ -0,0 +1,119 @@ +# ADR-011: Atlas Observability Model (Three Planes) + +Status: accepted (Sprint 5) +Date: 2026-07-19 +Owner: the primary operator (primary operator) + +## Context + +Sprints 1–4 produced durable *audit* records (`atlas_ops.pipeline_runs`, +`atlas_ops.deployments`, `atlas_ops.schema_migrations`) but no centralized +logs, no metrics, no alerting, and no dashboard. Sprint 4 live acceptance +additionally proved a real defect: Composer 3 task/worker logs never reached +Cloud Logging in this project (see `preflight-sprint5.md` §4), forcing +diagnosis through the Airflow REST API. Operators need to answer "did it +run, where is it failing, is the data correct and fresh, what did it cost, +who was told, and what do I do" from durable, queryable surfaces. + +## Decision + +Atlas observability uses three deliberately separate planes. No plane +imitates another; every signal declares one source of truth. + +### Plane 1 — Operational audit (BigQuery, `atlas_ops`) + +Durable, queryable history at controlled grains: + +| Table | Grain | Source of truth for | +|---|---|---| +| `pipeline_runs` | one row per pipeline run | run status, row counts, freshness | +| `task_events` (new, 004) | one row per task attempt event | task-level diagnosis, retries, telemetry completeness | +| `quality_results` (new, 005) | one row per check per run | data correctness evidence | +| `monitor_evaluations` (new, 006) | one row per monitor check per window | monitor history, drill evidence | +| `deployments` | one row per deploy/rollback attempt | delivery status | +| `schema_migrations` | one row per migration | schema history | + +### Plane 2 — Logs (Cloud Logging, Atlas-dedicated) + +Cloud Logging stores high-cardinality, high-detail events: Airflow task +output, scheduler/worker/DAG-processor activity, structured Atlas +application events (one JSON contract, ADR §logging), deployment/rollback +events, monitor evaluations, and drill markers. Routing: + +```text +project logs → sink atlas-observability-sink → log bucket atlas-observability + (30-day retention, Log Analytics enabled) → view atlas-runtime + → linked read-only BigQuery dataset atlas_logs +``` + +The `_Default` bucket keeps receiving source logs (the Atlas sink is +additive; no exclusion filters are added), so nothing is lost if the Atlas +bucket is misconfigured. + +### Plane 3 — Metrics and incidents (Cloud Monitoring) + +Low-cardinality time series (`custom.googleapis.com/atlas//`), +dashboards, alert policies, incident lifecycle, and notification routing to +the verified operator email channel. Metric labels are bounded to: +`environment, dag_id, task_id, component, status, check_name, severity`. +Run/batch/deployment identifiers and raw error strings are **forbidden** as +metric labels; they live in Planes 1–2 and are joined via time + labels. + +## Source-of-truth declarations + +| Question | Source of truth | +|---|---| +| Did the run succeed? | `atlas_ops.pipeline_runs` | +| Which task failed, which attempt? | `atlas_ops.task_events` + structured logs | +| Is the data correct? | `atlas_ops.quality_results` (dbt/warehouse evidence linked) | +| Is the data fresh? | latest SUCCESS in `pipeline_runs`; surfaced as `atlas/pipeline/last_success_age_seconds` | +| Did the deployment work? | `atlas_ops.deployments` | +| Is something wrong *right now*? | Cloud Monitoring incidents | +| Detailed history / forensics | `atlas-observability` log bucket (via `atlas_logs`) | +| What did it cost? | region-qualified `INFORMATION_SCHEMA.JOBS` (ADR-012) | + +## Correlation hierarchy + +```text +deployment_id → airflow_run_id → pipeline_run_id → batch_id → task_id → attempt_number +``` + +Every structured event carries the identifiers that exist at its scope; the +logging contract (Phase 2) enforces field names so one log filter follows a +run across planes. + +## Trust boundaries, retention, degradation + +- **Telemetry must never corrupt data processing**: audit/metric/log write + failures emit a fallback structured error and degrade visibly (telemetry + completeness monitor) but do not fail a task that moved data correctly — + except the finalizer, which reports incomplete telemetry explicitly. +- **Retention**: Atlas log bucket 30 days (measured MB/day scale; revisit + with real volume). `atlas_ops` tables are permanent (MB scale). Metric + retention follows Cloud Monitoring defaults. +- **Access boundary**: linking `atlas_logs` into BigQuery extends log read + access to BigQuery IAM; the linked dataset is read-only and the log view + is least-privileged. Documented in `security-review-sprint5.md`. +- **Intentional teardown**: `monitoring_enabled=false` in + `config/observability.yaml` (and disabled alert policies) precedes + Composer deletion so absence-based alerts do not fire on an intentionally + absent environment. Disabled runtime is distinguishable from stale runtime. + +## Alternatives considered + +- **Everything in BigQuery** (logs as rows): rejected — loses Cloud Logging + ingestion, filters, retention control, and Monitoring integration; invites + unbounded scans. +- **Everything in Cloud Monitoring** (audit as metrics): rejected — metric + cardinality explodes with per-run identifiers and history is lossy. +- **Third-party observability stack**: out of scope by charter. + +## Consequences + +- Operators get one place per question, with correlation identifiers + bridging planes. +- Costs stay near zero at rest (clean-slate project; measured baselines in + `cost-review-sprint5.md`). +- The Sprint 4 missing-logs defect becomes a first-class acceptance gate: + Plane 2 is only claimed after live retrieval of Composer task logs through + the documented filters. diff --git a/docs/adr/ADR-012-atlas-cost-attribution.md b/docs/adr/ADR-012-atlas-cost-attribution.md new file mode 100644 index 0000000..db3143d --- /dev/null +++ b/docs/adr/ADR-012-atlas-cost-attribution.md @@ -0,0 +1,67 @@ +# ADR-012: Atlas BigQuery Cost Attribution + +Status: accepted (Sprint 5) +Date: 2026-07-19 + +## Context + +Sprint 5 must answer "is delivery or warehouse cost behaving abnormally" +(mission question 6). BigQuery exposes job usage through region-qualified +`INFORMATION_SCHEMA.JOBS` (bytes processed/billed, slot ms, errors, labels, +identity), but only if Atlas jobs are distinguishable from everything else +in the project. Query-text matching is fragile and was rejected as a primary +strategy. + +## Decision + +Attribution evidence order (strongest first): + +1. **Job labels** — every Atlas job carries + `application=atlas, component=, environment=atlas-dev`. + - Python: `atlas.observability.cost.labeled_bigquery_client` sets the + labels via the client's `default_query_job_config`; all Atlas modules + (audit, task_events, quality_results, migrations, deployments, + validation, loader, preflight, resources) create clients through it. + Components are drawn from a bounded set + (`pipeline, monitor, deployment, validation, audit, migration, + ingestion, adhoc`); run/batch identifiers are excluded by design. + - dbt: the officially supported `query-comment` + `job-label: true` + configuration in `dbt_project.yml` converts a static JSON comment + (`application=atlas, component=dbt, environment=atlas-dev`) into job + labels. No dbt internals are patched. +2. **Runtime identity** — `atlas-composer-runtime`, + `atlas-github-integration`, `atlas-github-deployer` service accounts + (`user_email` in JOBS) catch anything that escaped labeling. +3. **Referenced/destination Atlas datasets** — forensic fallback only. + +Canonical queries live in `observability/queries/bigquery_cost.sql`: +daily bytes processed/billed, slot ms, job failures, usage by component, +unusually expensive jobs, and a labeled-vs-identity trend that quantifies +attribution coverage. All queries exclude parent `SCRIPT` rows (double +counting) and bound `creation_time`. + +The monitor DAG publishes windowed +`custom.googleapis.com/atlas/cost/bigquery_bytes_billed` and +`.../cost/bigquery_job_count` gauges from the same attribution and stores +evaluations in `atlas_ops.monitor_evaluations`. The cost-anomaly alert +compares the window against the configured baseline ratio +(`config/observability.yaml`); the cost drill uses a synthetic signal, never +a deliberately expensive query. + +## Rules + +- No full query text in Atlas operational tables (log/security hygiene). +- No `pipeline_run_id`/`batch_id` in job labels: cardinality is unnecessary + because JOBS already timestamps every job and Plane 1 orders runs in time. +- On-demand pricing estimate (`$6.25/TiB`) is a planning heuristic, not a + billing source; the Cloud Billing export remains authoritative for spend. + +## Consequences + +- Cost questions are answerable per day and per component with bounded + scans (~180-day JOBS retention). +- Attribution coverage is itself measurable (query 6); a growing + `unlabeled_jobs` count is a regression signal. +- Known gap: BigQuery jobs issued by third-party tools without labels or + Atlas identities (e.g. ad-hoc console queries by humans) attribute only via + dataset references; accepted for a development project. diff --git a/docs/adr/ADR-013-controlled-fault-injection.md b/docs/adr/ADR-013-controlled-fault-injection.md new file mode 100644 index 0000000..18a9a85 --- /dev/null +++ b/docs/adr/ADR-013-controlled-fault-injection.md @@ -0,0 +1,47 @@ +# ADR-013: Controlled Fault Injection + +Status: Accepted (Sprint 6) + +## Context + +Sprint 6 must prove that Atlas detects, contains, and recovers from realistic +failures. That requires *causing* failures — in a system whose canonical data, +audit history, and IAM posture must never become collateral damage. An +ungoverned "chaos" switch would be worse than no testing at all. + +## Decision + +1. **One catalog.** Every injectable failure is declared in + `config/failure_scenarios.yaml` with a full contract: injection method, + expected detection/alert/containment, allowed data impact, recovery + action, verification queries, cleanup, approvals, and hard duration/cost + ceilings. CI validates the catalog schema (`gate_failure_injection`); + an under-specified scenario cannot exist. +2. **Disabled by default, explicitly armed.** Activation requires ALL of: + an explicit scenario id in `ATLAS_INJECTION_SCENARIO`, + `ATLAS_APPROVE_FAILURE_INJECTION=true`, and every scenario-specific + approval (`ATLAS_APPROVE_IAM`, `ATLAS_APPROVE_DESTRUCTIVE_FIXTURE`, + `ATLAS_APPROVE_ROLLBACK_TEST`). A lingering approval variable alone is + inert; environment inheritance can never arm an injection. +3. **Never scheduled, never canonical, never production.** + `atlas.failure_injection.framework` refuses scheduled Airflow runs, + batch ids without the isolated `atlas-s6-` prefix, and any environment + other than `atlas-dev`. Refusal raises — there is no silent fallback to + normal execution, and a requested-but-refused injection logs a structured + `failure_injection_refused` event. +4. **No CRITICAL blast radius.** Risk levels are LOW/MEDIUM/HIGH only; the + schema has no CRITICAL tier, and destructive operations are restricted to + isolated fixtures gated by `ATLAS_APPROVE_DESTRUCTIVE_FIXTURE`. +5. **The CLI plans; the operator mutates.** `run_failure_scenario.sh` + authorizes, emits telemetry, and prints the exact injection steps; cloud + mutations are explicit logged commands executed inside the game-day + window, keeping every destructive step reviewable. +6. **Bounded.** Every scenario carries `maximum_duration_minutes` (≤ 120, + enforced via deadline checks) and `maximum_cost_usd` (≤ $1). + +## Consequences + +- Drill work is reproducible from Git: the catalog is the runbook's contract. +- Normal pipeline execution is provably injection-free (CI gate + unit tests + covering default-off, approval, environment, batch, and schedule refusal). +- The framework adds one more approval ceremony per drill; that is the point. diff --git a/docs/adr/ADR-014-recovery-action-model.md b/docs/adr/ADR-014-recovery-action-model.md new file mode 100644 index 0000000..f702c0d --- /dev/null +++ b/docs/adr/ADR-014-recovery-action-model.md @@ -0,0 +1,45 @@ +# ADR-014: Recovery Action Model + +Status: Accepted (Sprint 6) + +## Context + +Sprint 3–5 record what *happened* (pipeline runs, deployments, task events, +quality results, monitor evaluations). They do not record what an operator +*did about it*. Without a durable recovery grain, "we recovered" is a claim +with no evidence, and repeated incidents cannot be compared. + +## Decision + +1. **Separate grain.** `atlas_ops.recovery_actions` stores one row per + recovery action attempt, keyed by `recovery_id` and written with + idempotent MERGE (migration 007). Recovery actions link to incidents, + scenarios, pipeline runs, batches, and deployments — they never mutate + those records and are never mixed into `pipeline_runs`. +2. **Controlled vocabulary.** Action types are the fixed set + RETRY_TASK, RERUN_BATCH, REPAIR_PARTIAL_LOAD, QUARANTINE_BATCH, BACKFILL, + RESTORE_RELEASE, FORWARD_MIGRATION, RESTORE_IAM, REBUILD_PARTITION, + PAUSE_SCHEDULE, RESUME_SCHEDULE, RECONSTRUCT_AUDIT, RESET_MONITOR, + MANUAL_CONTAINMENT. Statuses: RUNNING, SUCCESS, FAILED, PARTIAL, ABORTED. +3. **Verification is the recovery.** A recovery row can only be finalized + SUCCESS with `verification_status=VERIFIED`; the module raises otherwise. + The verification contract is the Phase 13 reconciliation list (raw, + accepted, rejected, classification, fact, mart counts, uniqueness, + referential integrity, success marker, audits, monitor state, incident + resolution, no duplicates). PARTIAL and FAILED recoveries are first-class + recorded outcomes, not embarrassments to be overwritten. +4. **Decision tree first.** `docs/recovery-runbook-sprint6.md` defines the + choice order: (1) is canonical data corrupted? If no — retry, rerun, + restore permission/release. If yes — pause publication, quarantine, + determine repair boundary; targeted repair before rebuild, rebuild before + restore-and-backfill. Full refresh is never the first response + (enforced by the `ATLAS_APPROVE_FULL_REFRESH` cost guard). +5. **Sanitized like everything else.** `error_summary` passes through the + Sprint 4 sanitizer; no secrets, no unbounded stack traces. + +## Consequences + +- Every game-day recovery leaves a queryable audit row with timing evidence + for MTTR measurement. +- "Recovery succeeded" is machine-checkable: `status='SUCCESS'` implies a + verification pass by construction. diff --git a/docs/adr/ADR-015-schema-compatibility-and-recovery.md b/docs/adr/ADR-015-schema-compatibility-and-recovery.md new file mode 100644 index 0000000..29901a7 --- /dev/null +++ b/docs/adr/ADR-015-schema-compatibility-and-recovery.md @@ -0,0 +1,58 @@ +# ADR-015: Schema Compatibility and Recovery + +Status: Accepted (Sprint 6) + +## Context + +Atlas migrations are additive by policy (ADR-010), but reality eventually +demands renames, type changes, and required-field changes. Sprint 6 must +define how such changes are classified, which ones block, and what they do to +rollback eligibility. + +## Decision + +### Classification (extends the Sprint 5 drift classifier) + +| Change | Classification | Handling | +| --- | --- | --- | +| Configured new nullable field | ALLOWED | additive migration + manifest update + `allowed_new_fields` entry | +| Unapproved new nullable field | WARNING | contract update required before adoption | +| New REQUIRED field | BREAKING | blocked unless backfill + consumer compatibility proven | +| Removed field | BREAKING | blocked; consumer impact enumerated in the finding | +| Renamed field | BREAKING | appears as removed+new; requires a compatibility bridge (dual-write or view aliasing) and a documented deprecation period | +| Incompatible type change | BREAKING | blocked; forward migration plan required | +| REQUIRED made nullable | BREAKING | blocked pending consumer review | +| Partition-field change | BREAKING (high-risk) | never automatically applied; manual review + rebuild plan | + +### Multi-version inputs + +Raw events may carry a `schema_version` discriminator (absent = version 1). +`atlas.validation.schema_versions` normalizes every supported version onto +the current logical shape with explicit NULLs for fields older versions lack. +Unknown versions and unknown fields are rejected — there is no silent +coercion (S6-SCH-008). + +### Rollback eligibility across migrations + +The migrations manifest supports a `breaking` flag +(`||breaking`). Rollback to a prior release is evaluated by +`atlas.ops.rollback_compatibility`: + +- newer applied migrations that are additive → rollback eligible; +- any newer applied migration flagged `breaking` → rollback **blocked** with + forward-recovery guidance (`ROLLBACK_INCOMPATIBLE` stage failure in the + deploy engine); +- applied migrations the current manifest cannot classify → rollback + **refused** rather than guessed. + +Breaking BigQuery migrations are never reversed automatically. Recovery from +a bad release after a breaking migration is always forward: fix on a new +release, backfill if required, reconcile. + +## Consequences + +- Rollback safety becomes a declared property of the migration history + instead of operator folklore. +- All eight Sprint 6 schema scenarios (S6-SCH-001…008) are covered by unit + tests against the classifier, the version normalizer, or the rollback + compatibility evaluator. diff --git a/docs/adr/ADR-016-governance-source-of-truth.md b/docs/adr/ADR-016-governance-source-of-truth.md new file mode 100644 index 0000000..61e7761 --- /dev/null +++ b/docs/adr/ADR-016-governance-source-of-truth.md @@ -0,0 +1,72 @@ +# ADR-016: Governance Source of Truth + +- Status: Accepted (Sprint 7) +- Date: 2026-07-19 +- Deciders: lead data architect, governance engineer, data engineering + +## Context + +Through Sprint 6, Atlas ownership, grain, classification, and retention lived +implicitly in code, dbt descriptions, and prose docs. There was no single, +enforceable place that answered "who owns this, what is its grain, who consumes +it, how long is it retained." A governance system that duplicates this metadata +in multiple files rots immediately; a governance system that CI ignores is +decorative. + +## Decision + +**One source of truth per asset kind, with a generated consolidated catalog.** + +1. **dbt models** are governed by their dbt `meta.governance` block in the + model's property YAML. dbt already owns model grain (via `description`), + contracts, and tests; governance metadata lives alongside them. The dbt + `description` is the authoritative `purpose`; it is not duplicated. + +2. **Non-dbt assets** (raw/operational tables, buckets, DAGs, dashboards, log + resources) are governed by `governance/non_dbt_assets.yml`. + +3. A **generated catalog** (`governance/generated/catalog.json` + `.md`) is + derived from both sources by `python -m atlas.governance.catalog generate`. + It is never hand-edited. CI (`gate_governance`) fails if the committed + catalog is stale or if an asset id appears in both sources. + +Required fields for every major asset: `asset_id`, `asset_type`, `purpose`, +`technical_owner`, `business_owner_or_role`, `grain`, `source`, `consumers`, +`classification`, `retention_class`, `freshness_expectation`, `contract_version`, +`lifecycle_status`, `repository_path`, `runbook`, `last_reviewed`. + +Controlled vocabularies (asset types, lifecycle statuses, classifications, +retention classes, compatibility classes) live in `governance/policy.yml` and +are enforced by `atlas.governance.registry.validate_governance`. + +Owners must be **role identifiers**, not personal email addresses — this keeps +ownership durable across staffing and avoids committing personal data. + +## Alternatives considered + +- **A standalone catalog service / metadata platform (DataHub, OpenMetadata).** + Rejected: explicitly out of scope for Sprint 7; repository artifacts are + sufficient at this scale and avoid a new operational dependency. +- **A single monolithic governance YAML for everything, including dbt models.** + Rejected: it would duplicate grain/contract information dbt already owns, + creating exactly the multi-location drift this ADR prevents. +- **JSON Schema as the only validator.** Kept as optional/documentation + (`governance/schemas/`), but the authoritative validator is pure-Python + (`validate_governance`) so the CI gate needs no extra dependency and can + express cross-file invariants (single source of truth, retention permanence, + consumer registration). + +## Consequences + +- Adding/changing an asset is a small, local edit plus a catalog regeneration; + CI blocks incomplete or drifted governance. +- The catalog is a reliable, machine-readable input for lineage/impact + (Phase 5), deprecation (Phase 6), and classification/retention (Phase 9). +- Governance validation is offline and credentialless, so it runs in PR CI. + +## Honest limitations + +- Consumer discovery is limited to what the repository declares + (`governance/consumers.yml`); external/undeclared consumers are not + auto-discovered. +- This is project-level governance, not organization-wide governance. diff --git a/docs/adr/ADR-017-schema-compatibility-and-deprecation.md b/docs/adr/ADR-017-schema-compatibility-and-deprecation.md new file mode 100644 index 0000000..5fb0e70 --- /dev/null +++ b/docs/adr/ADR-017-schema-compatibility-and-deprecation.md @@ -0,0 +1,91 @@ +# ADR-017: Schema Compatibility and Deprecation + +- Status: Accepted (Sprint 7) +- Date: 2026-07-19 +- Deciders: lead data architect, data engineering, CI policy engineer +- Supersedes/extends: ADR-015 (schema compatibility and recovery) + +## Context + +Atlas had rollback compatibility (ADR-015) and a migration ledger, but no +automated way to classify whether a proposed schema/contract change is safe, and +no enforcement that breaking changes carry an approved migration. Sprint 7 makes +schema evolution a governed, testable process. + +## Decision + +### Compatibility classes + +Every schema/contract change is classified by `atlas.governance.schema_check`: + +- **COMPATIBLE** — additive nullable column, widened accepted-value set, + description/ownership improvement, additive non-breaking metadata, + required→nullable loosening, a brand-new asset. +- **CONDITIONALLY_COMPATIBLE** — requires consumer migration (e.g. a new + required field), approved temporary alias, approved dual-write period, + approved type widening with evidence, or deprecation with an active + replacement. +- **BREAKING** — removed/renamed field without a compatibility path, + incompatible type change, nullable→required without migration, changed model + grain, changed partition field, changed event identity, or a narrowed enum + that rejects existing valid values. +- **PROHIBITED** — destructive canonical change without approval, unversioned + contract replacement (schema changed but `contract_version` unchanged or + downgraded), changing an applied migration checksum, silent field reuse with + different semantics, or bypassing consumer-impact analysis. + +### Versioned baselines + +`governance/schemas/manifests/baseline.json` is a committed, generated snapshot +of every dbt model's contract-relevant schema (fields, types, nullability, +accepted values, grain, partition field, event identity, contract version). CI +regenerates it and fails on drift, so the baseline can never silently rot. + +### Checker interface + +``` +python -m atlas.governance.schema_check --baseline \ + --candidate --output [--fail-on BREAKING] +python -m atlas.governance.schema_check --generate +``` + +### Change records + +Every non-COMPATIBLE change must ship a change record under +`governance/changes/` declaring: `change_id`, `asset_id`, +`old_contract_version`, `new_contract_version`, `compatibility_class`, `reason`, +`owner`, `consumer_impact`, `migration_plan`, `backfill_plan`, `validation_plan`, +`rollback_limitations`, `deprecation_window`, `approval_reference`. CI +(`gate_schema_compatibility` + `gate_governance`) rejects a BREAKING change that +lacks a complete, approved change record. + +### Applied-migration immutability + +`sql/migrations/checksums.lock` pins the SHA-256 of every shipped migration. +`gate_schema_compatibility` fails if any migration file's checksum diverges from +the lock (PROHIBITED). New migrations must append a lock entry; existing ones +can never be edited. + +## Alternatives considered + +- **Rely only on `on_schema_change=fail` in dbt.** Insufficient: it catches + fact-table column drift at build time but not grain/partition/identity/enum + changes, contract versioning, or migration edits, and gives no PR-time + classification. +- **Register schemas in an external registry.** Out of scope; committed + manifests suffice at this scale. + +## Consequences + +- Additive changes pass CI automatically; breaking changes are blocked unless an + approved change record exists. +- Applied migrations are provably immutable. +- Never demonstrate a breaking change against canonical Atlas data — fixtures + only (`ATLAS_APPROVE_BREAKING_SCHEMA_DEMO`). + +## Honest limitations + +- The generated manifest infers types only where dbt declares `data_type` + (currently the enforced `stg_events` contract); other columns record type + `unknown`, so type-change detection is strongest on contracted columns. + Nullability and accepted-values are inferred from dbt tests. diff --git a/docs/adr/ADR-018-identity-and-access-boundaries.md b/docs/adr/ADR-018-identity-and-access-boundaries.md new file mode 100644 index 0000000..af75833 --- /dev/null +++ b/docs/adr/ADR-018-identity-and-access-boundaries.md @@ -0,0 +1,57 @@ +# ADR-018: Identity and Access Boundaries + +- Status: Accepted (Sprint 7) +- Date: 2026-07-19 +- Deciders: cloud security reviewer, CI policy engineer, platform + +## Context + +Atlas uses four created service accounts plus Google-managed agents. Sprint 4 +established keyless Workload Identity Federation; Sprint 7 formalizes the +identity boundaries and makes prohibited IAM patterns enforceable. + +## Decision + +### Identity boundaries + +- **`atlas-composer-runtime`** — Airflow runtime. `composer.worker`, + `bigquery.jobUser`, `bigquery.dataEditor` (atlas_* datasets), + `bigquery.resourceViewer`. +- **`atlas-github-deployer`** — CI/CD deploy + migrations. `bigquery.jobUser`, + `bigquery.dataEditor`, `composer.user`, + `composer.environmentAndStorageObjectAdmin`. WIF-only. +- **`atlas-github-integration`** — PR integration tests. `bigquery.jobUser` + + `bigquery.dataEditor` **scoped to CI datasets** (reduction candidate, + ADR-018/§reduction). WIF-only. +- **Google-managed** Composer agents — not modified. + +### Prohibited IAM patterns (enforced by `gate_security_policy`) + +Managed Atlas IAM policy definitions in the repository must never grant: + +- `roles/owner`, `roles/editor`, `roles/resourcemanager.projectIamAdmin`; +- service-account keys (keyless WIF only); +- weakened WIF trust conditions (must retain repo + ref scoping); +- unnecessary cross-project permissions. + +### Change discipline + +- No permission removal without a positive-use test proving valid workflows + still succeed and a negative test proving the removed permission is denied. +- IAM mutations require `ATLAS_APPROVE_IAM=true`; missing approval yields a + documented plan and a blocked gate, never a weakened control. + +## Consequences + +- The IAM posture is documented in an evidence matrix (`iam-review-sprint7.md`). +- A repository-level CI gate rejects prohibited roles in any managed policy + definition, catching regressions before deployment. +- The one justified reduction (`atlas-github-integration` project→dataset + `dataEditor`) is specified with positive/negative tests, pending approval. + +## Honest limitations + +- Least privilege is asserted at role scope with workload evidence, not with + per-permission usage telemetry. +- Default-compute-SA `roles/editor` and bootstrap `roles/owner` are pre-existing + project-level items outside Atlas's created identities; flagged, not changed. diff --git a/docs/adr/ADR-019-classification-retention-and-disposal.md b/docs/adr/ADR-019-classification-retention-and-disposal.md new file mode 100644 index 0000000..8f4ee1c --- /dev/null +++ b/docs/adr/ADR-019-classification-retention-and-disposal.md @@ -0,0 +1,56 @@ +# ADR-019: Classification, Retention, and Disposal + +- Status: Accepted (Sprint 7) +- Date: 2026-07-19 +- Deciders: governance engineer, data engineering, security reviewer + +## Context + +Atlas had no declared data classification or retention policy; disposal relied +on GCP defaults and memory. Sprint 7 makes classification and retention explicit, +machine-validated, and safe (permanent evidence can never be accidentally +expired). + +## Decision + +### Classification + +Four levels — PUBLIC, INTERNAL, CONFIDENTIAL, RESTRICTED +(`governance/classifications.yml`). Every governed asset declares one. Atlas +processes only synthetic data, so **no RESTRICTED assets exist**; the policy +asserts this and CI fails if a RESTRICTED asset appears while the assertion +holds. INTERNAL is the default for synthetic events and warehouse models. + +### Retention classes + +Seven classes (`governance/retention.yml`) with an explicit disposal policy and +an `is_permanent_evidence` flag. Retention rules distinguish canonical +(rebuildable, indefinite), operational evidence (permanent), temporary resources +(mandatory TTL), release evidence (retained), and test fixtures (ephemeral). + +### Invariants (enforced by `gate_governance`) + +- `policy.retention_classes` == `retention.yml` keys (single source of truth). +- Permanent-evidence classes cannot declare an expiration. +- Transient classes must declare an expiration (disposal is mandatory). +- Every asset references a defined retention class. + +### Safe disposal + +`atlas.governance.retention.plan_expirations()` produces a dry-run disposition +per asset (`keep_forever` vs `expire_d`). A live applier must assert it never +expires a `keep_forever` asset. Live expiration/lifecycle changes require +`ATLAS_APPROVE_RETENTION_MUTATION=true`; without it, the plan and validations +are produced and the live mutation is a recorded blocked gate. + +## Consequences + +- Retention is declared, validated, and safe by construction — permanent audit + and release evidence cannot be accidentally expired. +- Temporary resources have a mandatory, declared disposal. + +## Honest limitations + +- Retention is declared and validated in configuration; live enforcement on + temporary datasets/buckets is applied under approval, not automatically during + Sprint 7. diff --git a/docs/adr/ADR-020-bigquery-performance-and-cost-controls.md b/docs/adr/ADR-020-bigquery-performance-and-cost-controls.md new file mode 100644 index 0000000..f09d53b --- /dev/null +++ b/docs/adr/ADR-020-bigquery-performance-and-cost-controls.md @@ -0,0 +1,59 @@ +# ADR-020: BigQuery Performance and Cost Controls + +- Status: Accepted (Sprint 7) +- Date: 2026-07-19 +- Deciders: BigQuery performance engineer, cost steward, CI policy engineer +- Extends: ADR-012 (cost attribution), Sprint 6 cost guards + +## Context + +Sprint 6 added runtime cost guards (backfill window, full-refresh approval, +dry-run ceiling). Sprint 7 makes cost limits **config-driven and enforced before +spend**, and establishes a measured performance methodology. + +## Decision + +### Config-driven controls + +`config/cost_controls.yaml` declares per-environment limits: `max_query_bytes`, +`max_performance_suite_bytes`, `max_backfill_days`, +`full_refresh_requires_approval`, `require_partition_filter_assets`, +`temporary_dataset_ttl_hours`, `temporary_object_ttl_days`, +`composer_max_lifecycle_hours`, `log_retention_days`, `release_retention_policy`. +`gate_performance_cost` validates coherence (per-query ceiling ≤ suite ceiling, +required fields present). + +### Estimation-first execution + +`python -m atlas.observability.cost_guard estimate` always dry-runs first +(bills $0), reports estimated bytes, compares with the environment ceiling, and +**refuses over-limit execution** unless `ATLAS_APPROVE_COST_OVERRIDE=true`. It +never executes on estimation failure and emits structured evidence. +`ATLAS_MAX_PERFORMANCE_TEST_BYTES` caps the whole performance suite. + +### Required partition filters + +`check-partition-filter` statically rejects queries over +`require_partition_filter_assets` (raw events, fct_events) that lack a partition +predicate, catching the classic full-scan cost mistake. + +### Performance methodology (ADR-020 / performance-review) + +Measure before optimizing. Every performance experiment records bytes +processed/billed, slot-ms, elapsed, rows in/out, partition pruning, correctness +checksum, and query plan evidence, under a hard byte ceiling and run labels. A +change ships only if it preserves grain and correctness; "no material +improvement" backed by evidence is an acceptable result. + +## Consequences + +- An unbounded query is blocked at dry-run before material spend (demonstrated). +- Cost limits live in one config, enforced in CI and at runtime. +- Performance changes are evidence-gated and correctness-preserving. + +## Honest limitations + +- The dataset is ~50k rows/batch; performance results are engineering + demonstrations, not production-scale benchmarks. +- The partition-filter check is a static heuristic on partition-column + predicates, not a full SQL analyzer. diff --git a/docs/adr/ADR-021-reference-architecture-and-handoff-contract.md b/docs/adr/ADR-021-reference-architecture-and-handoff-contract.md new file mode 100644 index 0000000..9cc3d19 --- /dev/null +++ b/docs/adr/ADR-021-reference-architecture-and-handoff-contract.md @@ -0,0 +1,61 @@ +# ADR-021: Reference-Architecture and Handoff Contract + +- **Status:** Accepted (Sprint 8) +- **Date:** 2026-07-19 +- **Deciders:** data architect, release owner +- **Related:** ADR-016 (governance source of truth), ADR-017 (schema + compatibility), all Sprint 1–7 ADRs (the decisions this package curates) + +## Context + +Through Sprint 7, Atlas was understood primarily by its builder and development +agents. Knowledge lived across 61 docs, 20 ADRs, and validation reports, but +there was no single enforceable contract that (a) routes a newcomer to +authoritative sources, (b) links every major claim to evidence with a live/ +static/blocked distinction, and (c) fails CI when documentation drifts from +repository truth or depends on hidden context. A repository is not transferable +merely because it contains many Markdown files. + +## Decision + +Establish a **reference-architecture and handoff contract** as a first-class, +CI-enforced artifact set: + +1. **`START_HERE.md`** is the canonical entry point and router (not a second + README). +2. **`docs/reference-architecture/`** is a curated *map* over existing evidence — + it links to detailed sources and never duplicates full runbooks/reports. +3. A machine-readable **`reference-manifest.yml`** and **evidence-index.json** + are validated by **`python -m atlas.reference.validate`**: referenced files + exist, ids are unique, LIVE claims are backed by live evidence, blocked work + is never marked complete, and verification commits are present. +4. A single focused CI gate, **`gate_reference_handoff`**, wired into the existing + `validate_ci.sh` python group (no workflow YAML logic duplication), enforces + the contract plus: no absolute local paths or prior-conversation dependencies + in current onboarding docs, capability limitations present, public-extraction + manifest valid, and that `atlas-sprint-8-complete` is not claimed before it + exists. +5. **Reproducibility is tested, not asserted:** `validate_clean_clone.sh` runs + from a fresh directory + fresh venv, and an independent handoff test scores a + separate agent against a rubric. +6. **Reference architecture ≠ template.** Atlas is a reference architecture; the + template-extraction plan is documented but not executed, and no template + status is claimed. + +## Consequences + +- **Positive:** documentation cannot silently drift from code (CI fails); + newcomers and agents have a single, tested entry path; claims are auditable; + blocked work stays visibly blocked. +- **Cost:** one new gate + validator to maintain; the manifest/evidence index + must be updated when docs/claims change (enforced, so drift is caught). +- **Boundary:** the contract governs Sprint 8 reference artifacts; it does not + reopen Sprint 1–7 tags or change existing architecture. + +## Invariants introduced + +INV-R1..R5 in +[architecture-invariants.md](../reference-architecture/architecture-invariants.md): +instructions independent of prior conversations; documents identify their +verification commit; claims link to evidence; blocked work stays blocked; +reference architecture must not claim template status. diff --git a/docs/alert-catalog-sprint5.md b/docs/alert-catalog-sprint5.md new file mode 100644 index 0000000..0dde0ec --- /dev/null +++ b/docs/alert-catalog-sprint5.md @@ -0,0 +1,45 @@ +# Atlas Alert Catalog (Sprint 5) + +All policies are repo-managed in `observability/alerts/*.json`, applied +idempotently by `scripts/manage_atlas_alerts.sh`, and routed to the verified +Cloud Monitoring email channel +`projects/example-gcp-project/notificationChannels/6567861337166986657` +("Atlas Primary Operator (email)"; the address is deliberately not committed). +Owner for every policy: the primary operator (primary operator). + +All `check_status`-based policies share the same mechanics: the monitor DAG +publishes `custom.googleapis.com/atlas/monitor/check_status` (0 PASS / 1 WARN +/ 2 FAIL / −1 NO_DATA / −2 DISABLED) every 30 minutes per `check_name`; the +condition fires when max-aligned value > 1.5 (10-minute alignment, retest on +each new point); incidents auto-close 30 minutes after cessation. Test method: +`manage_atlas_alerts.sh test ` publishes a synthetic FAIL on the +`mode=drill` series (drill series never pollute normal history but evaluate +against the same policy, which is exactly what a drill needs). + +| Policy (display name) | Signal (`check_name` unless noted) | Severity | Incident key | Runbook anchor | Main false-positive risk | +|---|---|---|---|---|---| +| Atlas: pipeline failed | `latest_run_state` | critical | `atlas-latest_run_state` | `#alert-atlas-pipeline-failed` | none known | +| Atlas: data stale | `freshness` (fail ≥ 50 h) | critical | `atlas-freshness` | `#alert-atlas-data-stale` | environment paused without disabling monitoring | +| Atlas: reconciliation failed | `reconciliation` | critical | `atlas-reconciliation` | `#alert-atlas-reconciliation-failed` | none known | +| Atlas: critical volume deviation | `volume_deviation` (fail ≥ 80 %) | critical | `atlas-volume_deviation` | `#alert-atlas-volume-deviation` | intentional batch-size change | +| Atlas: breaking schema drift | `schema_drift` | critical | `atlas-schema_drift` | `#alert-atlas-schema-drift` | manifest not regenerated after approved migration | +| Atlas: deployment failed | `deployment_failure` | critical | `atlas-deployment_failure` | `#alert-atlas-deployment-failed` | none known | +| Atlas: rollback failed | `rollback_failure` | critical | `atlas-rollback_failure` | `#alert-atlas-rollback-failed` | none known | +| Atlas: Composer environment unhealthy | native `composer.googleapis.com/environment/healthy` < 0.5 for 15 min | critical | `atlas-composer-unhealthy` | `#alert-atlas-composer-unhealthy` | creation/deletion transitions | +| Atlas: BigQuery cost anomaly | `cost_anomaly` (≥ 10× baseline and > 1 GiB) | warning | `atlas-cost_anomaly` | `#alert-atlas-cost-anomaly` | legitimate backfill bursts | +| Atlas: telemetry incomplete | `telemetry_completeness` | warning | `atlas-telemetry_completeness` | `#alert-atlas-telemetry-incomplete` | runs predating Sprint 5 telemetry | + +Design rules in force: + +- One policy per root cause; WARN states are dashboard-visible but only FAIL + (value 2) pages, separating warning from critical. +- Metric absence is used nowhere as a fail signal: the freshness check makes + staleness an explicit value, and the Composer policy conditions on an + unhealthy value, so intentional teardown (metric absence) cannot fire it. + Teardown checklist additionally disables `atlas-composer-unhealthy` and + sets `monitoring_enabled: false` (all checks then publish DISABLED = −2). +- No stale-data alerting while `monitoring_enabled: false`. +- The cost drill uses a synthetic drill-series point, never real spend. +- Policy descriptions contain runbook paths and no secrets or addresses. +- `manage_atlas_alerts.sh apply` is idempotent (update-by-display-name); + before editing a firing policy, capture the open incident evidence first. diff --git a/docs/architecture-sprint2.md b/docs/architecture-sprint2.md new file mode 100644 index 0000000..b4a2e20 --- /dev/null +++ b/docs/architecture-sprint2.md @@ -0,0 +1,66 @@ +# Project Atlas Sprint 2 Architecture + +## Objective + +Transform immutable Sprint 1 raw events into a governed BigQuery warehouse with explicit +quality classification, quarantine, trusted facts, and daily marts. + +## Layered datasets + +| Dataset | Purpose | Representative relations | +| --- | --- | --- | +| `atlas_raw` | Immutable physical landing (Sprint 1) | `events` | +| `atlas_staging` | Normalized views and seeds | `stg_events`, `valid_country_codes` | +| `atlas_intermediate` | Classification and accepted canonical rows | `int_event_classification`, `int_accepted_events` | +| `atlas_quarantine` | Rejected physical rows | `int_rejected_events` | +| `atlas_core` | Dimensions and incremental fact | `dim_users`, `dim_countries`, `fct_events` | +| `atlas_marts` | Analyst-facing aggregates | `mart_daily_event_metrics` | + +## Flow + +```mermaid +flowchart TD + Raw["atlas_raw.events"] --> Staging["atlas_staging.stg_events"] + Seed["valid_country_codes seed"] --> Classification["atlas_intermediate.int_event_classification"] + Staging --> Classification + Classification --> Accepted["atlas_intermediate.int_accepted_events"] + Classification --> Rejected["atlas_quarantine.int_rejected_events"] + Accepted --> Fact["atlas_core.fct_events"] + Accepted --> Users["atlas_core.dim_users"] + Seed --> Countries["atlas_core.dim_countries"] + Fact --> Mart["atlas_marts.mart_daily_event_metrics"] +``` + +## Classification rules + +Terminal rejection precedence per physical row: + +1. `missing_user_id` +2. `invalid_country_code` +3. `future_dated` +4. `duplicate_extra` +5. `accepted` + +Duplicates rank by `ingested_at DESC, event_timestamp DESC, source_file DESC, raw_record_hash DESC`. +Only rank 1 can be accepted when no higher-precedence defect exists. + +## Temporal semantics + +See [ADR-003](adr/ADR-003-corrected-temporal-semantics.md). Sprint 1's 300 "late-arriving" +rows are backdated declared dates, not event-time late arrivals. + +## Incremental strategy + +`fct_events` uses BigQuery merge incremental logic keyed on `event_id`, partitioned by +`event_date`, clustered by `event_name` and `country_code`, with a three-day ingestion +lookback. Optional `start_date` / `end_date` vars support bounded backfills. + +## Airflow handoff + +Sprint 2 scripts are Cloud Shell–authoritative: + +1. `scripts/setup_dbt.sh` +2. `scripts/run_dbt_sprint2.sh` +3. `scripts/validate_dbt_sprint2.sh` + +An orchestrator can wrap these commands after Sprint 1 ingestion completes. diff --git a/docs/architecture-sprint3.md b/docs/architecture-sprint3.md new file mode 100644 index 0000000..027f2f3 --- /dev/null +++ b/docs/architecture-sprint3.md @@ -0,0 +1,52 @@ +# Project Atlas — Sprint 3 Architecture + +## Overview + +Sprint 3 wraps the Sprint 1 ingestion and Sprint 2 dbt warehouse in an Airflow 3.1.7 +orchestration layer. Business logic remains in Python CLIs and dbt; Airflow owns +ordering, retries, publication, and finalization. + +## Task graph + +```text +resolve_run_context + → ensure_audit_resources + → start_run_audit + → preflight_environment + → generate_events + → upload_to_gcs + → load_bigquery_raw + → validate_raw_load + → dbt_seed + → dbt_source_freshness + → dbt_build + → validate_warehouse + → publish_success_marker +write_run_summary (all_done) +``` + +## Identity model + +| Field | Scope | Example | +|-------|-------|---------| +| `batch_id` | Stable data batch | `atlas-20260715` | +| `pipeline_run_id` | One execution | `atlas-airflow-20260715-manual__...` | + +## Deployment contract + +| Asset | Composer path | +|-------|---------------| +| DAGs | `/home/airflow/gcs/dags/project_atlas/` | +| Scripts + dbt | `/home/airflow/gcs/data/` | + +Set `ATLAS_ROOT=/home/airflow/gcs/data/project-atlas`. + +## Retry policy + +| Task | Retries | +|------|---------| +| upload, load | 2 (exponential backoff) | +| dbt freshness | 1 (scheduled mode) | +| validation, dbt build | 0 | + +Historical backfill runs skip blocking freshness with a documented `SKIPPED` result. diff --git a/docs/architecture-sprint4.md b/docs/architecture-sprint4.md new file mode 100644 index 0000000..b143167 --- /dev/null +++ b/docs/architecture-sprint4.md @@ -0,0 +1,106 @@ +# Atlas Architecture — Sprint 4: Secure Delivery System + +## Delivery flow + +```mermaid +flowchart LR + A[Cursor Cloud Agent\nfeature branch] --> B[Pull request] + B --> C[atlas-ci\ncredentialless static CI] + C -->|atlas-ci-gate green| D[Human merge to main] + D --> E[atlas-integration\nWIF: integration SA\natlas_ci_* isolation] + D --> F[atlas-deploy\nworkflow_dispatch + typed confirm] + F --> G[Immutable bundle\ngs://…/atlas/releases/sha/] + G --> H[Audited additive migrations\natlas_ops.schema_migrations] + H --> I[Composer promote\ndags/project_atlas + data/current] + I --> J[Parse check → smoke batch\natlas-smoke-sha-run] + J --> K[Smoke validation\n12 checks] + K --> L[atlas_ops.deployments\nSUCCESS / FAILED + stage] + L -.failure.-> M[atlas-rollback\nprior validated release] +``` + +## Trust boundaries + +| Zone | Code executed | Credentials | Writes allowed | +|---|---|---|---| +| PR CI (`atlas-ci`) | untrusted PR code | none (`contents: read`) | GitHub artifacts only | +| Integration (`atlas-integration`) | main-reachable SHAs only | WIF → `atlas-github-integration` | `atlas_ci_*` datasets, CI bucket prefix | +| Deployment (`atlas-deploy`/`atlas-rollback`) | main-reachable SHAs only | WIF → `atlas-github-deployer` | deployment bucket, Composer paths, additive migrations, `atlas_ops` audit | +| Composer runtime | promoted immutable release | `atlas-composer-runtime` env SA | canonical Atlas datasets, events bucket | +| Cursor agent | working tree | project service account (dev env) | development resources | + +Key property: **pull requests can never reach GCP.** The WIF provider rejects +non-repo tokens, and impersonation bindings accept only +`YOUR_GITHUB_OWNER/YOUR_REPOSITORY@refs/heads/main` (ADR-009). + +## CI workflow graph (`atlas-ci`) + +```text +pull_request / push(main) / dispatch [path-scoped to core Atlas pipeline] + ├── atlas-security-shell secret scan, dep sanity, shell syntax+static, workflow YAML + ├── atlas-python ruff format+lint, mypy, unit+acceptance tests, config gate + ├── atlas-dbt pinned dbt deps + parse (compile/unit tests run in + │ the authenticated integration stage — ADR-008) + ├── atlas-airflow pinned 3.1.7 + constraints, pip check, DAG import + │ (safe_mode=False), structure/retry/parse-safety tests + └── atlas-ci-gate single stable required-check name +``` + +All jobs call `scripts/validate_ci.sh --mode static --group ` — the +canonical validation contract shared by agents, developers, CI, and release +tooling. Business logic never lives in workflow YAML (ADR-008). + +## Deployment bundle format (ADR-010) + +`atlas-bundle.tar.gz` (deterministic tar: sorted names, fixed mtime, gzip -n): + +```text +atlas-bundle/ + dags/ # parse-time assets → /dags/project_atlas/ + src/atlas/ # runtime library + scripts/ # runtime step scripts only + config/ # atlas.yaml, anomaly_profile.yaml + sql/ + sql/migrations/ # additive DDL + ledger manifest + dbt/atlas_dbt/ # dbt project (no target/, logs/, packages) + dbt/profiles/profiles.yml # keyless oauth runtime profile + requirements*.txt # dependency manifests + release-manifest.json # git sha/ref/tag, build metadata, tool versions, + # per-file SHA-256, required/min schema version +``` + +Immutable home: `gs://atlas-deployments-…/atlas/releases//` with +create-only semantics. `data/current/` on the Composer bucket is +always a verified promoted copy; the release path is the rollback source of +truth. + +## Composer runtime mapping + +| Concern | Path | +|---|---| +| DAG parsing | `/dags/project_atlas/` | +| Runtime code+config | `/data/current/` (`ATLAS_ROOT`) | +| Batch artifacts + markers | `…/current/data/runs//` (GCSfuse) | +| dbt writes | `/tmp/dbt-target`, `/tmp/dbt-logs` (worker-local) | +| Deployed identity | `…/current/release-manifest.json` + `deployment-info.json` | + +Environment variables set at creation: `ATLAS_ROOT`, `ATLAS_GCP_PROJECT_ID`, +`ATLAS_GCS_BUCKET`, `ATLAS_BQ_DATASET`, `ATLAS_DBT_DATASET`, +`DBT_PROJECT_DIR`, `DBT_PROFILES_DIR`, `DBT_LOCATION`, `DBT_TARGET_PATH`, +`DBT_LOG_PATH`. The deployed git SHA and deployment id are file-based +(promoted with each release) rather than env vars, so promotion never waits on +slow environment-update operations. + +## Audit model + +- `atlas_ops.pipeline_runs` — one row per DAG execution (Sprint 3 grain). +- `atlas_ops.schema_migrations` — one row per migration, checksummed. +- `atlas_ops.deployments` — one row per deployment/rollback attempt, MERGE + keyed by `deployment_id`, statuses RUNNING/SUCCESS/FAILED/ROLLING_BACK/ + ROLLED_BACK/ROLLBACK_FAILED, sanitized errors, `previous_git_sha` linkage. + +Grains stay separate: a deployment references its smoke run by id only. + +## IAM matrix (live, ADR-009) + +See ADR-009 for the full table and the documented `bigquery.dataEditor` +project-scope risk with compensating controls. No Owner/Editor/IAM-admin +grants; no service-account keys anywhere in the delivery path. diff --git a/docs/architecture-sprint5.md b/docs/architecture-sprint5.md new file mode 100644 index 0000000..1943cda --- /dev/null +++ b/docs/architecture-sprint5.md @@ -0,0 +1,96 @@ +# Sprint 5 Architecture — Atlas Observability + +The observability model (three planes, source-of-truth table, correlation +hierarchy, trust boundaries) is normative in +`adr/ADR-011-atlas-observability-model.md`. This document maps the model to +concrete components and flows. + +## Component map + +```text + ┌──────────────────────────────────────────────┐ + │ Composer 3 (atlas-dev) │ + │ atlas_batch_pipeline atlas_observability_ │ + │ (business DAG) monitor (read-only) │ + └──────┬───────────────────────┬───────────────┘ + structured JSON events│ (one contract: │ evaluations + metrics + to task stdout │ atlas.observability. │ + ▼ logging) ▼ + ┌────────────────────────────┐ ┌──────────────────────────┐ + │ Plane 2: Cloud Logging │ │ Plane 1: BigQuery │ + │ sink: atlas-observability- │ │ atlas_ops.pipeline_runs │ + │ sink → bucket │ │ atlas_ops.task_events │ + │ atlas-observability (30 d, │ │ atlas_ops.quality_results│ + │ Log Analytics) → view │ │ atlas_ops.monitor_evals │ + │ atlas-runtime → linked BQ │ │ atlas_ops.deployments │ + │ dataset atlas_logs (RO) │ │ atlas_ops.schema_migr. │ + └──────────────┬─────────────┘ └────────────┬─────────────┘ + │ log-based / │ monitor reads + │ custom metrics │ (bounded windows) + ▼ ▼ + ┌──────────────────────────────────────────────────────────┐ + │ Plane 3: Cloud Monitoring │ + │ custom.googleapis.com/atlas/* metrics · dashboard │ + │ atlas-operations · 10 alert policies · incidents │ + │ → email channel 6567861337166986657 (verified operator) │ + └──────────────────────────────────────────────────────────┘ +``` + +## Event flow for one pipeline run + +1. `resolve_run_context` establishes `pipeline_run_id`/`batch_id`; every + subsequent structured event carries the correlation fields. +2. Each task attempt writes `task_events` rows (STARTED → SUCCESS/FAILED/ + RETRY/...) via idempotent MERGE and emits contract events to stdout. +3. `validate_warehouse` persists its per-check results to `quality_results` + in addition to failing the task on FAIL. +4. `write_run_summary` finalizes `pipeline_runs` and verifies task-telemetry + completeness (missing attempts are reported, not silently ignored). +5. The monitor DAG evaluates freshness/volume/rejection/schema/deployment/ + cost windows, writes `monitor_evaluations`, publishes metrics, and emits + structured evaluation logs. +6. Alert policies watch the metrics; incidents route to the operator email + channel; runbook paths are embedded in policy documentation. + +## Repository layout (Sprint 5 additions) + +```text + + src/atlas/observability/ logging.py · metrics.py · checks.py · schema_drift.py + src/atlas/ops/ task_events.py · quality_results.py + sql/migrations/ 004_create_task_events_table.sql + 005_create_quality_results_table.sql + 006_create_monitor_evaluations_table.sql + dags/atlas_observability_monitor.py + config/observability.yaml + observability/ + logging/ log-bucket.json · log-view.json · sink-filter.txt + metrics/ metric-descriptors.json + alerts/ *.json (10 policies) + dashboards/ atlas-operations.json + queries/ saved log + cost queries + scripts/ + bootstrap_observability.sh (plan/apply/status) + manage_atlas_alerts.sh (plan/apply/enable/disable/status/test/delete-test-resources) + docs/ runbook · alert catalog · on-call model · reviews +``` + +## Deployment integration + +The Sprint 4 delivery system is unchanged: the deployment bundle gains the +monitor DAG, observability modules, config, and migrations 004–006; the same +`deploy_atlas_release.sh` → smoke → audit path promotes them. No second +deployment system exists. CI gains static gates for observability artifacts +(YAML/JSON validation, monitor DAG import, redaction and cardinality tests) +and stays credentialless on pull requests. + +## Failure-visibility rules + +- Telemetry failure degrades visibly (fallback `telemetry_emit_failed` + events, `atlas/pipeline/telemetry_incomplete` metric) and never converts a + successful data operation into a failure. +- Monitor no-data states are explicit (`NO_DATA` evaluations) and + distinguished from `DISABLED` (`monitoring_enabled=false`, set before + intentional teardown so absence alerts do not fire). +- The dashboard's top row answers "is Atlas healthy" in under a minute: + latest run outcome, freshness age, open incidents, latest deployment. diff --git a/docs/architecture-sprint7.md b/docs/architecture-sprint7.md new file mode 100644 index 0000000..0f78414 --- /dev/null +++ b/docs/architecture-sprint7.md @@ -0,0 +1,79 @@ +# Atlas Sprint 7 Architecture — Governance Layer + +Sprint 7 adds an **enforceable governance layer** over the Sprint 1–6 platform. +No data-plane redesign: the ingestion → dbt warehouse → orchestration → +CI/CD → observability → resilience stack is unchanged. Sprint 7 makes changes to +data, schemas, permissions, retention, warehouse structure, and cost *governed*. + +## Components added + +``` +governance/ # source of truth (declarative) + policy.yml # rules + controlled vocabularies + classifications.yml # PUBLIC/INTERNAL/CONFIDENTIAL/RESTRICTED + retention.yml # retention classes + disposal + consumers.yml # internal consumer registry + non_dbt_assets.yml # raw/ops/bucket/dag/dashboard/log assets + changes/TEMPLATE.yml # schema-change records (ADR-017) + schemas/ # JSON schema + versioned manifests + manifests/baseline.json # committed schema baseline + generated/ # DERIVED (catalog.json/md, lineage.json) + +dbt models: meta.governance blocks # source of truth for models + +src/atlas/governance/ + registry.py # load + validate governance; deprecation lifecycle + catalog.py # generate consolidated catalog (+ drift check) + schema_check.py # compatibility classification + manifest generation + lineage.py # dbt ref/source -> lineage graph + impact.py # consumer-impact analysis + security_policy.py # managed-IAM + data-exposure scanners + retention.py # retention validation + disposal planning +src/atlas/observability/ + cost_guard.py # config-driven cost ceilings + estimate CLI (Sprint 7) + cost_guards.py # Sprint 6 runtime guards (extended) + +config/cost_controls.yaml # per-env cost ceilings +sql/migrations/checksums.lock # applied-migration immutability +observability/performance/ # perf query set + suite results +scripts/run_performance_suite.sh # dry-run baseline + gated execution +``` + +## Enforcement (CI gates, all offline/credentialless) + +`scripts/validate_ci.sh --group python` runs, in addition to the prior gates: + +- `gate_governance` — complete metadata, one source of truth, no catalog drift, + retention invariants, deprecation lifecycle. +- `gate_schema_compatibility` — migration checksum immutability + schema + baseline drift. +- `gate_lineage_impact` — lineage drift + source→mart reachability. +- `gate_security_policy` — no prohibited IAM roles/keys in managed defs, no + secret-like values in governed artifacts. +- `gate_performance_cost` — cost-control config coherence. + +## Data flow with governance overlaid + +``` +generate_events → raw → staging → classification → accepted/rejected + → fact → dims → marts → operational tables + │ │ │ │ + contracts dup/replay schema lineage + + (ADR-016) semantics compat impact + (ADR-006amd) (ADR-017) (Phase 5) + + classification + retention (ADR-019) apply to every asset + cost + performance controls (ADR-020) apply to every query + IAM boundaries (ADR-018) apply to every identity +``` + +## Key invariant preserved + +`fct_events` grain remains **one row per event_id** (global uniqueness). The +Sprint 3 defect fix (ADR-006 amendment) changed duplicate *classification and +measurement*, not the fact grain. + +## What Sprint 7 did NOT change + +Data-plane models (except the classification-semantics fix), orchestration DAGs, +deploy/rollback flow, observability metrics/alerts/dashboard, and the diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..0d75e04 --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,100 @@ +# Project Atlas Architecture — Sprint 1 + +## Purpose + +Sprint 1 establishes a reproducible raw ingestion path that future Atlas versions +can extend with dbt, Airflow, CI/CD, and monitoring without refactoring core +boundaries. + +## Component diagram + +```mermaid +flowchart LR + subgraph local [LocalExecution] + Generator[SyntheticEventGenerator] + Orchestrator[PipelineOrchestrator] + Validator[ValidationEngine] + end + + subgraph gcp [GoogleCloudPlatform] + GCS[CloudStorageBucket] + BQ[BigQueryDatasetAtlasRaw] + Events[TableEvents] + end + + Generator --> GCS + GCS --> BQ + BQ --> Events + Events --> Validator + Orchestrator --> Generator + Orchestrator --> GCS + Orchestrator --> BQ + Orchestrator --> Validator +``` + +## Sequence diagram + +```mermaid +sequenceDiagram + participant User as Engineer + participant CLI as AtlasScripts + participant Gen as Generator + participant GCS as CloudStorage + participant BQ as BigQuery + participant Val as Validation + + User->>CLI: run_pipeline --approve-provision + CLI->>Gen: generate 50000 events + Gen-->>CLI: local JSONL + anomaly counts + CLI->>GCS: upload run-scoped object + GCS-->>CLI: gs:// URI + CLI->>BQ: load staging then insert + BQ-->>CLI: rows loaded + CLI->>Val: validate loaded run + Val-->>CLI: overall FAIL + acceptance PASS +``` + +## Data flow + +1. Generator writes deterministic JSONL locally. +2. Upload writes `raw/event_date=YYYY-MM-DD/run_id=/events.jsonl`. +3. Loader creates dataset/table if missing, loads staging, inserts enriched rows. +4. Validation queries loaded rows and reports PASS/FAIL checks. + +## Repository diagram + +```text + +├── config/atlas.yaml Shared runtime settings +├── config/anomaly_profile.yaml Seeded anomaly expectations +├── src/atlas/generator/ Synthetic event generation +├── src/atlas/ingestion/ GCS upload +├── src/atlas/loader/ BigQuery load +├── src/atlas/validation/ Quality checks +├── src/atlas/pipeline/ Orchestration only +├── scripts/ Cloud Shell entry points +├── sql/create_events_table.sql Raw table DDL +└── tests/ Automated verification +``` + +## Key design decisions + +| Decision | Why | Tradeoff | Future impact | +| --- | --- | --- | --- | +| Nested project in `de-project-1` | Preserves workspace-root MCP config | Two Python packaging contexts | Airflow/dbt can reference Atlas paths directly in v0.3/v0.4 | +| Run-scoped GCS keys | Prevents overwrite while keeping date partitions | Slightly longer object paths | Compatible with future partition-aware backfills | +| Staging-table load | Idempotent replays and explicit metadata enrichment | Extra transient table per run | Maps cleanly to dbt staging models | +| Overall validation FAIL on seeded anomalies | Mirrors real data-quality posture | Requires separate acceptance checks | dbt tests can reuse anomaly expectations in v0.3 | + +## Failure handling + +- Duplicate upload: existing object detected, upload marked `already_exists`. +- Missing file: step fails before cloud mutation. +- Bad schema: BigQuery load job fails and pipeline stops. +- Duplicate run load: target-table run guard skips re-insert. +- Partial load: staging table deleted after successful insert; failed insert leaves no target rows for run id. + +## Out of scope + +Sprint 1 intentionally excludes dbt models, Airflow DAGs, Terraform, Pub/Sub, +streaming, dashboards, CI/CD, and alerting. diff --git a/docs/ci-cd-governance-sprint4.md b/docs/ci-cd-governance-sprint4.md new file mode 100644 index 0000000..bc772a2 --- /dev/null +++ b/docs/ci-cd-governance-sprint4.md @@ -0,0 +1,65 @@ +# Atlas CI/CD Governance — Sprint 4 (Phase 15) + +## What GitHub plan limitations actually allow + +The repository is **private on the GitHub Free plan**. Verified consequences +(API evidence, not assumption): + +```text +$ gh api repos/YOUR_GITHUB_OWNER/YOUR_REPOSITORY/branches/main/protection +HTTP 403: Upgrade to GitHub Pro or make this repository public to enable this feature. +``` + +Unavailable and therefore **not claimed**: + +- branch protection on `main` (required checks, PR-only merges, force-push + blocks enforced server-side) +- GitHub Environments with required reviewers / self-review prevention +- tag protection rules + +## Controls that ARE active (fallback model, ADR-010) + +| Control | Mechanism | Where | +|---|---|---| +| Independent validation of every PR | `atlas-ci` workflow, path-scoped, credentialless | `.github/workflows/atlas-ci.yml` | +| Single stable gate name | `atlas-ci-gate` job aggregates all required jobs | same | +| Deployment cannot start from a PR | deploy/rollback are `workflow_dispatch` only | `atlas-deploy.yml`, `atlas-rollback.yml` | +| Typed confirmation | exact strings `deploy-atlas-dev` / `rollback-atlas-dev` | same | +| Only merged code can deploy | target SHA must satisfy `git merge-base --is-ancestor origin/main` | same | +| Only `main`-ref runs get GCP credentials | WIF impersonation bound to `repository_and_ref = …@refs/heads/main` (server-side, cannot be bypassed by a fork or PR) | ADR-009, GCP IAM | +| No overlapping deployments | `concurrency: group: atlas-dev-deployment`, `cancel-in-progress: false` | both deploy workflows | +| No stored cloud secrets | zero GitHub Actions secrets; OIDC only | repo settings | +| Supply-chain pinning | every third-party action pinned to a full commit SHA with the release tag in a comment | all workflows | +| Minimal token scopes | `permissions: contents: read` (+ `id-token: write` only where WIF is used) | all workflows | + +The strongest governance boundary is deliberately placed **on the GCP side** +(WIF ref restriction), because GitHub-side branch protection cannot be +enforced on this plan. Even a direct push to a feature branch plus a manual +dispatch cannot obtain deployer credentials for non-main code. + +## Commands to enable full protection after a plan upgrade + +```bash +gh api -X PUT repos/YOUR_GITHUB_OWNER/YOUR_REPOSITORY/branches/main/protection \ + -F required_status_checks[strict]=true \ + -F "required_status_checks[contexts][]=atlas-ci-gate" \ + -F enforce_admins=true \ + -F required_pull_request_reviews[required_approving_review_count]=1 \ + -F restrictions=null \ + -F allow_force_pushes=false \ + -F allow_deletions=false +``` + +Also configure Environments `atlas-integration` and `atlas-dev` with required +reviewers and deployment-branch policy `main` (Settings → Environments; the +`environment: atlas-dev` reference already exists in the deploy workflows and +will pick the protections up automatically). + +## Operating rules until then + +1. Never merge a PR whose `atlas-ci-gate` is not green (human-enforced). +2. Never push directly to `main` (human-enforced; WIF limits the blast radius + of a violation to code that still had to pass through `main`). +3. Deployments and rollbacks only via the dispatch workflows or the audited + scripts they call; every attempt lands in `atlas_ops.deployments`. +4. Release tags (`atlas-sprint-N-complete`) are created once, never moved. diff --git a/docs/ci-cd-runbook-sprint4.md b/docs/ci-cd-runbook-sprint4.md new file mode 100644 index 0000000..bee47c8 --- /dev/null +++ b/docs/ci-cd-runbook-sprint4.md @@ -0,0 +1,144 @@ +# Atlas CI/CD Runbook — Sprint 4 + +Operator commands for validation, deployment, rollback, and recovery. All +scripts live in `scripts/`; workflows in `.github/workflows/`. + +## 1. Local / agent validation (no cloud credentials needed) + +```bash +cd Atlas-GCP-Build +bash scripts/validate_ci.sh --mode static # everything +bash scripts/validate_ci.sh --mode static --group python +bash scripts/validate_ci.sh --mode static --group airflow +bash scripts/validate_ci.sh --mode static --group dbt +bash scripts/validate_ci.sh --mode static --group security-shell +``` + +Machine-readable results: `logs/ci/validate-ci-results.json`. + +## 2. Isolated GCP integration test + +Requires ADC (locally) or WIF (`atlas-integration` workflow, main-ref only). + +```bash +bash scripts/validate_gcp_integration.sh +# GitHub: Actions → atlas-integration → Run workflow (target SHA optional) +``` + +Creates `atlas_ci_` datasets + `gs://atlas-ci-…/atlas-ci//`, +validates auth, determinism, idempotent loads, isolated dbt build, and +reconciliation, then deletes everything (verified). Orphan recovery: + +```bash +bq ls --project_id example-gcp-project | grep atlas_ci_ +bq rm -r -f -d example-gcp-project: # per leftover dataset +# GCS leftovers expire automatically after 7 days +``` + +## 3. Migrations + +```bash +bash scripts/apply_atlas_migrations.sh --mode plan # no mutation +bash scripts/apply_atlas_migrations.sh --mode status # ledger dump +ATLAS_APPROVE_DEPLOY=true bash scripts/apply_atlas_migrations.sh --mode apply +``` + +Rules: append-only manifest (`sql/migrations/manifest.txt`), checksummed, +re-apply is a no-op, changed shipped files hard-fail, failures recorded as +FAILED in `atlas_ops.schema_migrations` and block promotion. + +## 4. Release bundles + +```bash +bash scripts/build_deployment_bundle.sh # build + verify locally +bash scripts/build_deployment_bundle.sh --upload # create-only GCS store +``` + +Stored under `gs://atlas-deployments-example-gcp-project/atlas/releases//`. +Existing release with identical content → reuse; different content for the +same SHA → hard failure (immutability). + +## 5. Ephemeral Composer environment (ADR-010: delete after evidence capture) + +```bash +bash scripts/manage_atlas_composer.sh status +ATLAS_APPROVE_COMPOSER_CREATE=true ATLAS_APPROVE_IAM=true \ + bash scripts/manage_atlas_composer.sh create # ~25-45 min total +bash scripts/manage_atlas_composer.sh delete # ALWAYS after evidence +``` + +Image `composer-3-airflow-3.1.7-build.13`, size small, region `us-central1`, +runtime SA `atlas-composer-runtime@…`. dbt is installed as Composer PyPI +packages from `dbt/requirements-dbt.txt` pins. + +## 6. Deployment + +Preferred (post-merge): GitHub → Actions → `atlas-deploy` → Run workflow → +type `deploy-atlas-dev`. Script path (agent/operator with ADC): + +```bash +ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas_release.sh \ + --git-sha [--leave-paused] +``` + +Stages and their failure recording (`atlas_ops.deployments.failure_stage`): +`fetch_release`, `schema_check`, `migrations`, `promote`, `dag_parse`, +`smoke_batch`, `smoke_validation`, `finalize`. A failed deployment records +`FAILED`, keeps the immutable bundle and logs, and prints the recovery +command. Re-running with the same `--git-sha` is safe. + +## 7. Smoke validation (standalone) + +```bash +bash scripts/validate_atlas_deployment.sh \ + --git-sha --deployment-id \ + --batch-id atlas-smoke-- \ + --pipeline-run-id atlas-smoke---run \ + --processing-date YYYY-MM-DD +``` + +## 8. Rollback + +GitHub: `atlas-rollback` → type `rollback-atlas-dev`. Script path: + +```bash +ATLAS_APPROVE_ROLLBACK_TEST=true ATLAS_APPROVE_DEPLOY=true \ + bash scripts/rollback_atlas.sh [--target-sha ] +``` + +Selects the newest prior `SUCCESS` deployment, verifies bundle checksums and +schema compatibility (a target requiring unapplied migrations is rejected), +re-promotes, runs a rollback smoke batch, records `ROLLED_BACK` / +`ROLLBACK_FAILED` with `previous_git_sha` linkage. + +## 9. Common failures + +| Symptom | Likely cause | Action | +|---|---|---| +| `atlas-ci-gate` red | any required job failed | open the failing job; every gate maps to a `validate_ci.sh` gate reproducible locally | +| WIF auth error in workflow | run not on `refs/heads/main` | deploy only merged SHAs; PRs can never authenticate (by design) | +| `fetch_release` failure | bundle missing for SHA | run `build_deployment_bundle.sh --upload` from that SHA (must be committed) | +| checksum mismatch on existing release | different content for same SHA | investigate immediately — never overwrite; the stored bundle is truth | +| `dag_parse` timeout | GCS sync delay or import error | `gcloud composer environments run atlas-dev --location us-central1 dags list-import-errors` | +| smoke `raw_batch_count` failure | partial load | check `atlas_ops.pipeline_runs` row and task logs; rerun deploy (idempotent loader skips complete batches) | +| migration CHECKSUM_MISMATCH | shipped SQL edited | revert the edit; add a new migration instead | +| leftover `atlas_ci_*` datasets | cleanup failed mid-run | see §2 orphan recovery | +| deployment stuck RUNNING | workflow died before finalize | re-run deploy with same SHA or finalize manually via `atlas.ops.deployments` | + +## 10. Audit queries + +```sql +-- deployments and rollbacks, newest first +SELECT deployment_id, deployment_type, status, git_sha, previous_git_sha, + failure_stage, smoke_pipeline_run_id, completed_at +FROM `example-gcp-project.atlas_ops.deployments` +ORDER BY created_at DESC; + +-- migration ledger +SELECT * FROM `example-gcp-project.atlas_ops.schema_migrations` +ORDER BY applied_at; + +-- smoke run for a deployment +SELECT * FROM `example-gcp-project.atlas_ops.pipeline_runs` +WHERE pipeline_run_id = ''; +``` diff --git a/docs/cost-review-sprint5.md b/docs/cost-review-sprint5.md new file mode 100644 index 0000000..0b3adfe --- /dev/null +++ b/docs/cost-review-sprint5.md @@ -0,0 +1,104 @@ +# Atlas Sprint 5 Cost and Retention Review + +Measurements below were taken live on 2026-07-19 during the Sprint 5 +acceptance window in project `example-gcp-project`. Currency figures use +public list prices for `us-central1` at the time of writing and are estimates, +not billing-export truth. + +## Composer (dominant cost, ephemeral) + +| Item | Value | +| --- | --- | +| Environment | `atlas-dev`, `composer-3-airflow-3.1.7-build.13`, ENVIRONMENT_SIZE_SMALL | +| Created | 2026-07-19T03:44:08Z | +| Deleted | end of acceptance window (teardown gated on `ATLAS_APPROVE_TEARDOWN`) — recorded in the validation report | +| Estimated rate | ≈ $0.60–0.90/hour for a small Composer 3 environment (compute + storage + fees) | +| Acceptance window | single-digit hours ⇒ single-digit dollars | + +Composer is deliberately not left running: per the owner's standing decision, +environments exist only for evidence capture. The observability design +tolerates that (`monitoring_enabled` + NO_DATA states distinguish "paused by +design" from "stale"). + +## Cloud Logging + +| Item | Measured | +| --- | --- | +| Project log ingestion, trailing 24 h of acceptance | 57.2 MB (`logging.googleapis.com/billing/bytes_ingested`) | +| Largest sources | BigQuery data-access audit logs (~21 MB/6 h); Atlas structured events are a small fraction | +| `atlas-observability` bucket retention | 30 days, analytics enabled | +| `_Default` bucket retention | 30 days (unchanged) | +| Free tier | first 50 GiB/project/month ingestion free; current run-rate ≈ 1.7 GB/month ⇒ $0 marginal | + +The Atlas sink is additive (no exclusions), so entries are counted once for +ingestion; duplicate routing to the dedicated bucket does not double the +ingestion bill (storage beyond retention defaults would, but 30 days is the +default free retention). + +Justification for 30-day retention: Sprint drills and incident +reconstruction need at most a few weeks of history; durable operational truth +lives in BigQuery audit tables (`pipeline_runs`, `task_events`, +`quality_results`, `monitor_evaluations`, `deployments`), which are tiny (see +below). Longer log retention would add cost without a consumer. + +## Cloud Monitoring + +| Item | Measured | +| --- | --- | +| Custom metric descriptors | 15 (`custom.googleapis.com/atlas/...`) | +| Active `check_status` series | 22 = 11 checks × 2 modes (normal, drill) — within the documented cardinality budget (≤ 3 bounded labels per metric, no run/batch ids) | +| Other atlas metrics | 2–6 series each (mode × small label sets) | +| Ingested samples | one point per metric per monitor run (30-min cadence) ⇒ ~1.5 K samples/day total — far inside the 150 MB/month free allotment | +| Alert policies | 10 (no per-policy charge) | +| Notification channel | 1 email channel ($0) | +| Dashboard | 1 ("Atlas Operations", 31 tiles; $0) | + +## BigQuery + +| Item | Measured | +| --- | --- | +| `atlas_ops` dataset size | 0.09 MB across 6 tables | +| Monitor query cost | every check uses bounded time windows (30-day max) over KB-scale audit tables; the cost check reads `region-us.INFORMATION_SCHEMA.JOBS` bounded to its baseline window | +| Linked dataset (`atlas_logs`) | read-only view over the log bucket; queries bill as BigQuery scans of scanned log volume — trailing-hour drill queries scanned < 100 MB total | +| Job labeling | `application=atlas` labels + dbt `query-comment` enable attribution via `observability/queries/bigquery_cost.sql` | + +## What was deleted vs retained after the acceptance window + +Deleted (ephemeral): +- Composer environment `atlas-dev` (and alert policies expecting it are + disabled first — see runbook teardown procedure). +- Drill-mode metric series stop receiving points (auto-age-out); a PASS + recovery point was published to every drill series + (`manage_atlas_alerts.sh delete-test-resources`). + +Retained (permanent, near-zero cost): +- `atlas_ops` BigQuery tables (< 1 MB), `atlas_raw`/`atlas_core`/`atlas_marts` + datasets (synthetic data, MB scale). +- Log bucket + sink + view + linked dataset (storage-bounded by 30-day + retention). +- Metric descriptors, alert policies (disabled where their source + intentionally disappears with Composer), dashboard, notification channel. +- Deployment bundles in GCS (immutable releases, MB scale each). + +## Monthly projection (steady state, environment torn down) + +| Component | Projection | +| --- | --- | +| Composer | $0 (no environment) | +| Logging | $0 (under free tier; 30-day retention) | +| Monitoring | $0–low single dollars (custom-metric samples under free tier) | +| BigQuery storage | ≈ $0.02 | +| BigQuery queries | $0 while the monitor DAG is not running (no environment); during acceptance windows, bounded queries on KB–MB tables | +| Total | effectively the cost of the acceptance windows themselves (Composer hours) | + +## Guardrails verified + +- No DEBUG logging enabled anywhere; Airflow logging level INFO. +- No unbounded monitor queries (every check has an explicit window). +- No batch/run ids or error strings as metric labels (CI-enforced cardinality + gate + `validate_point` runtime contract). +- No raw event payloads logged; details are truncated at 4 KB. +- Single additive log sink; no duplicate sinks; `_Default` untouched. +- Dashboard is updated in place by display name, never re-created. +- The cost-anomaly drill used a synthetic drill-mode metric point, not real + BigQuery spend. diff --git a/docs/cost-review-sprint6.md b/docs/cost-review-sprint6.md new file mode 100644 index 0000000..50a8a10 --- /dev/null +++ b/docs/cost-review-sprint6.md @@ -0,0 +1,56 @@ +# Atlas Sprint 6 Cost and Guardrail Review + +Measurements taken live on 2026-07-19 during the Sprint 6 acceptance window in +project `example-gcp-project`. Currency figures use public `us-central1` list +prices and are estimates, not billing-export truth. + +## New in Sprint 6: preventive cost guards + +Sprint 6 adds `src/atlas/observability/cost_guards.py`, whose entire purpose is +to stop runaway spend *before* it happens. All were exercised live: + +| Guard | Behavior | Live evidence | +| --- | --- | --- | +| `validate_backfill_window` | Rejects backfill windows > 7 days unless `ATLAS_APPROVE_UNBOUNDED_BACKFILL=true` | Blocked a 12-day window (`processing_date=2026-07-30`) at `resolve_run_context` — zero bytes scanned; `cost_guard_blocked` (observed 12, threshold 7) emitted | +| `require_full_refresh_approval` | Blocks dbt `--full-refresh` unless `ATLAS_APPROVE_FULL_REFRESH=true` | Blocked without approval, allowed with it; `cost_guard_blocked` emitted | +| `enforce_dry_run_ceiling` / `guarded_query_config` | Caps estimated bytes and sets `maximum_bytes_billed` | Unit-tested (`tests/unit/test_cost_guards.py`) | + +These guards make the default posture "incremental, bounded, cheap"; expensive +operations require an explicit, logged approval variable. + +## Composer (dominant cost, ephemeral) + +| Item | Value | +| --- | --- | +| Environment | `atlas-dev`, `composer-3-airflow-3.1.7-build.13`, ENVIRONMENT_SIZE_SMALL | +| Created | 2026-07-19T14:31Z | +| Deleted | end of acceptance window (teardown gated on `ATLAS_APPROVE_TEARDOWN`) | +| Estimated rate | ≈ $0.60–0.90/hour for a small Composer 3 environment | +| Acceptance window | a few hours ⇒ single-digit dollars | + +Composer is never left running (ADR-010). The observability design distinguishes +"paused by design" from "stale" so teardown does not create false incidents. + +## BigQuery (recovery + game-day queries) + +- The QUARANTINE recovery removed 100 000 rows via two targeted `DELETE`s + (raw + intermediate); DELETEs on small dev tables are inexpensive. +- Game-day diagnosis used `INFORMATION_SCHEMA.JOBS` and row-count aggregates — + metadata and small scans. +- No full-refresh rebuild was performed during recovery (targeted repair, + ADR-014), avoiding a full re-scan of history. + +## Retention + +- Audit tables (`recovery_actions`, `task_events`, `pipeline_runs`, + `quality_results`, `monitor_evaluations`) are small and retained; they are the + durable operational history and are not cost-significant. +- No new long-lived cloud resources were created by Sprint 6 beyond the two + additive schema objects (a table + two columns). + +## Net cost posture + +Sprint 6 is cost-*reducing* in expectation: the guards prevent the most common +accidental-spend paths (unbounded backfills, unintended full refreshes, +unbounded scans), while the only material acceptance spend is the ephemeral +Composer window (single-digit dollars). diff --git a/docs/cost-review-sprint7.md b/docs/cost-review-sprint7.md new file mode 100644 index 0000000..5482508 --- /dev/null +++ b/docs/cost-review-sprint7.md @@ -0,0 +1,64 @@ +# Atlas Cost Review (Sprint 7) + +Extends the Sprint 6 cost guards with config-driven controls, an +estimation-first CLI, and a required-partition-filter check. See ADR-020. + +## Controls (config/cost_controls.yaml) + +| Control | atlas-dev | atlas-ci | +| --- | --- | --- | +| max_query_bytes | 1 GiB | 512 MiB | +| max_performance_suite_bytes | 5 GiB | 1 GiB | +| max_backfill_days | 7 | 3 | +| full_refresh_requires_approval | true | true | +| require_partition_filter_assets | raw.events, fct_events | same | +| temporary_dataset_ttl_hours | 24 | 1 | +| temporary_object_ttl_days | 7 | 1 | +| composer_max_lifecycle_hours | 12 | 6 | +| log_retention_days | 30 | 7 | +| release_retention_policy | keep_validated_releases | same | + +Inherited runtime guards (Sprint 6): backfill-window, full-refresh approval, +dry-run ceiling, `maximum_bytes_billed` job config. + +## Estimation-first enforcement (demonstrated live, $0) + +`docs/evidence-sprint7/cost-guard-block.txt`: + +1. **Partition-filter guard** blocks the deliberately unbounded raw scan before + any execution (`exit=2`). +2. **Estimate CLI** dry-runs first (full scan estimate = 12,659,283 bytes, + billed $0), then refuses execution when the estimate exceeds the ceiling + (proven with a tightened ceiling → `decision: BLOCKED`, pre-execution). + +An over-limit query is therefore refused **before material spend**, requiring an +explicit `ATLAS_APPROVE_COST_OVERRIDE=true` after a documented review. + +## Permanent resource footprint + +- 9 BigQuery datasets (small; raw 850k rows, fct 392,845 rows, mart 2,804 rows). +- `atlas-observability` log bucket (30-day retention). +- `atlas-deployments-…` release bundle bucket (validated releases retained). +- 10 alert policies, 1 notification channel, 1 dashboard, metric descriptors. +- **Composer: absent** (ephemeral; not created in Sprint 7 — no changed control + required it). + +## Sprint 7 live cost + +- Performance baseline: **dry-runs only → $0**. +- Inventory reads (IAM/table counts/dataset settings): a handful of tiny + metadata/count queries (KB–MB). +- No Composer, no bulk scans, no full refreshes. +- Estimated Sprint 7 live BigQuery spend: **negligible (< a few MB billed + total)**; the 5 GiB suite ceiling was never approached. + +## Proposed hard byte ceiling + +`ATLAS_MAX_PERFORMANCE_TEST_BYTES` (suite) defaults to 5 GiB; per-query 1 GiB. +Both overridable only with documented approval. These remain the recommended +ceilings. + +## Honest limitations + +- Zero-cost operation is not claimed; the footprint above incurs minimal + ongoing storage/monitoring cost. diff --git a/docs/dag-catalog-sprint3.md b/docs/dag-catalog-sprint3.md new file mode 100644 index 0000000..2d76d0e --- /dev/null +++ b/docs/dag-catalog-sprint3.md @@ -0,0 +1,26 @@ +# Atlas Sprint 3 DAG Catalog + +## `atlas_batch_pipeline` + +| Property | Value | +|----------|-------| +| Schedule | `0 6 * * *` UTC | +| Start date | 2026-07-01 | +| Catchup | false | +| Max active runs | 1 | + +### Trigger conf keys + +| Key | Purpose | +|-----|---------| +| `processing_date` | Override logical processing date | +| `batch_id` | Override stable batch identifier | +| `upload_once` | Fail upload on try 1 for retry evidence | +| `dbt_test_failure` | Enable inject_failure singular test | + +### Task IDs + +`resolve_run_context`, `ensure_audit_resources`, `start_run_audit`, +`preflight_environment`, `generate_events`, `upload_events`, `load_bigquery_raw`, +`validate_raw_load`, `dbt_seed`, `dbt_source_freshness`, `dbt_build`, +`validate_warehouse`, `publish_success_marker`, `write_run_summary` diff --git a/docs/data-contract-standard-sprint7.md b/docs/data-contract-standard-sprint7.md new file mode 100644 index 0000000..7fcebe9 --- /dev/null +++ b/docs/data-contract-standard-sprint7.md @@ -0,0 +1,72 @@ +# Atlas Data Contract Standard (Sprint 7) + +A data contract is a **versioned, enforceable agreement** about the shape and +semantics of data crossing a boundary. Atlas contracts are not prose — every +clause maps to an executable control. Where a prose clause would disagree with +an executable schema, the executable schema wins and the prose is fixed. + +## Boundaries that require a contract + +``` +generate_events → raw ingestion → staging → classification + → accepted / rejected → fact → dimensions → marts → operational tables +``` + +| Boundary | Producer | Consumer | Contract version | Enforced by | +| --- | --- | --- | --- | --- | +| event generation → raw | `generate_events.py` | `load_events.py` | 1.0 | `validate_events.py`, `atlas.validation.schema_versions` | +| raw → staging | `atlas_raw.events` | `stg_events` | 1.0 | dbt source tests, `stg_events` enforced contract (`data_type`s) | +| staging → classification | `stg_events` | `int_event_classification` | 1.1 | dbt tests + unit tests (rejection precedence, dup rank) | +| classification → accepted/rejected | `int_event_classification` | `int_accepted_events` / `int_rejected_events` | 1.0 | dbt `accepted_values` / `not_accepted_values` tests | +| accepted → fact | `int_accepted_events` | `fct_events` | 1.0 | `unique_key=event_id`, `on_schema_change=fail`, unique/not_null tests | +| fact → dimensions | `fct_events` | `dim_users`, `dim_countries` | 1.0 | not_null/unique + relationship tests | +| fact → marts | `fct_events` | `mart_daily_event_metrics` | 1.0 | `unique_combination_of_columns`, reconciliation tests | +| pipeline → operational tables | pipeline code | audit tables | per table | `observability/schema/expected-schemas.json`, migration ledger | + +## Required contract clauses + +Each contract declares: contract ID, version, producer, owner, consumers, grain, +fields (required/optional, types, accepted values, uniqueness, nullability), +temporal semantics, duplicate semantics, freshness, compatibility policy, +deprecation policy, validation implementation, and recovery expectations. + +Governance metadata carries the durable half of this (owner, grain, consumers, +`contract_version`, classification, retention). The executable half lives in dbt +contracts/tests, source tests, the schema manifest, and Python validators. + +## Mapping clauses to executable controls + +| Clause | Executable control | +| --- | --- | +| Fields + types | dbt `data_type` (enforced contract on `stg_events`); `expected-schemas.json` for ops tables | +| Required / nullability | dbt `not_null` tests; NULL semantics documented per column | +| Accepted values | dbt `accepted_values` (rejection_reason, platform); `dim_countries` FK | +| Uniqueness | dbt `unique` (event_id, user_id); `unique_combination_of_columns` (mart grain) | +| Grain | governance `grain` field + the uniqueness tests that enforce it | +| Temporal semantics | Sprint 2 corrected flags (`is_future_dated`, late/backdated/mismatch) + `assert_source_anomaly_profile` | +| Duplicate semantics | `int_event_classification` rank + Sprint 7 replay classification (ADR-017) | +| Freshness | `dbt source freshness`; observability freshness check | +| Compatibility policy | `atlas.governance.schema_check` (ADR-017) + `gate_schema_compatibility` | +| Deprecation policy | governance `lifecycle_status` + `governance/changes/` + deprecation CI | +| Migration checksum immutability | `atlas_ops.schema_migrations` ledger + `sql_migrations` checksum gate | +| Recovery expectations | `atlas.ops.recovery_actions` + `docs/recovery-runbook-sprint6.md` | + +## Contract versioning + +`contract_version` is `major.minor`: + +- **minor** bump: backward-compatible change (added nullable field, widened + accepted set, description/owner improvement) — `COMPATIBLE` in ADR-017. +- **major** bump: a change requiring consumer migration or a breaking change — + requires a change record and approval (`CONDITIONALLY_COMPATIBLE`/`BREAKING`). + +The compatibility class (ADR-017) and the version bump must agree; CI enforces +that a breaking change cannot ship as a minor bump without an approved change +record. + +## Non-negotiables + +- No prose-only contract that disagrees with the executable schema. +- No unversioned contract replacement (`PROHIBITED`). +- No silent field reuse with changed semantics (`PROHIBITED`). +- Applied migration checksums are immutable (`PROHIBITED` to change). diff --git a/docs/deployment-catalog-sprint4.md b/docs/deployment-catalog-sprint4.md new file mode 100644 index 0000000..96b39af --- /dev/null +++ b/docs/deployment-catalog-sprint4.md @@ -0,0 +1,77 @@ +# Atlas Deployment Catalog — Sprint 4 + +Living record of Sprint 4 delivery artifacts and cloud resources. Durable +per-attempt records live in `atlas_ops.deployments` (query examples in the +runbook §10); this catalog documents the fixed resource inventory. + +## Source-control artifacts + +| Artifact | Value | +|---|---| +| Sprint 4 foundation PR | #14 (squash `21d54ed`) — Phases 0–3 | +| Gate-demonstration PR | #15 (closed unmerged by design) — Phase 16 | +| Delivery PR | #16 — Phases 5–15, 17–19 | +| Release tag (on completion) | `atlas-sprint-4-complete` | + +## GCP resource inventory (created by Sprint 4) + +| Resource | Name | Lifecycle | +|---|---|---| +| WIF pool | `atlas-github-pool` | permanent | +| WIF provider | `atlas-github-provider` (repo-restricted) | permanent | +| Service account | `atlas-github-integration@…` | permanent | +| Service account | `atlas-github-deployer@…` | permanent | +| Service account | `atlas-composer-runtime@…` | permanent (no cost when Composer absent) | +| Bucket | `gs://atlas-deployments-example-gcp-project` (versioned) | permanent — immutable releases | +| Bucket | `gs://atlas-ci-example-gcp-project` (7-day TTL) | permanent, self-cleaning | +| BigQuery table | `atlas_ops.schema_migrations` | permanent | +| BigQuery table | `atlas_ops.deployments` | permanent | +| Composer env | `atlas-dev` (us-central1, `composer-3-airflow-3.1.7-build.13`, small) | **ephemeral** — created for evidence capture, deleted afterwards (ADR-010) | +| Ephemeral datasets | `atlas_ci__…` | per integration run, auto-deleted | + +Sprint 5 additions (details: `validation-report-sprint5.md`): + +| Resource | Name | Lifecycle | +|---|---|---| +| BigQuery tables | `atlas_ops.task_events`, `atlas_ops.quality_results`, `atlas_ops.monitor_evaluations` | permanent (migrations 004–006) | +| Log bucket | `atlas-observability` (us-central1, 30-day retention, analytics) | permanent | +| Log sink / view | `atlas-observability-sink` / `atlas-runtime` | permanent | +| Linked dataset | `atlas_logs` (read-only) | permanent | +| Metric descriptors | 15 × `custom.googleapis.com/atlas/...` | permanent | +| Alert policies | 10 × `Atlas: …` (environment-dependent ones disabled between acceptance windows) | permanent | +| Notification channel | `Atlas Primary Operator (email)` | permanent | +| Dashboard | `Atlas Operations` (31 tiles) | permanent | + +## Release bundle registry + +Immutable bundles: `gs://atlas-deployments-example-gcp-project/atlas/releases//` +(`atlas-bundle.tar.gz`, `.sha256`, `release-manifest.json`). List releases: + +```bash +gcloud storage ls gs://atlas-deployments-example-gcp-project/atlas/releases/ +``` + +Bundle contents and manifest fields: `architecture-sprint4.md`. Live +deployment evidence for Sprint 4 acceptance (bundle URIs, checksums, +deployment ids, smoke run ids, rollback linkage): +`validation-report-sprint4.md`. + +## Version pins in force + +| Component | Version | Where pinned | +|---|---|---| +| Python (CI/runtime) | 3.12 | workflows `PYTHON_VERSION` | +| apache-airflow | 3.1.7 (+ official constraints) | `airflow/requirements-airflow.txt` | +| providers google / standard | 20.0.0 / 1.12.1 | same | +| dbt-core / dbt-bigquery | 1.11.12 / 1.11.3 | `dbt/requirements-dbt.txt` | +| dbt-utils | pinned via `dbt/atlas_dbt/package-lock.yml` | dbt deps | +| Composer image | `composer-3-airflow-3.1.7-build.13` | `manage_atlas_composer.sh`, ADR-005 | +| CI toolchain (ruff, mypy, yamllint, shellcheck-py, pytest) | see file | `requirements-ci.txt` | +| actions/checkout | v5 `93cb6efe…` | all workflows | +| actions/setup-python | v6 `ece7cb06…` | all workflows | +| actions/upload-artifact | v4 `ea165f8d…` | all workflows | +| google-github-actions/auth | v3.0.0 `7c6bc770…` | WIF workflows | +| google-github-actions/setup-gcloud | v3.0.1 `aa5489c8…` | WIF workflows | + +Action SHAs were resolved with `gh api repos///git/ref/tags/` +on 2026-07-18 and are updated deliberately, never by floating tags. diff --git a/docs/deprecation-runbook-sprint7.md b/docs/deprecation-runbook-sprint7.md new file mode 100644 index 0000000..1db2ae5 --- /dev/null +++ b/docs/deprecation-runbook-sprint7.md @@ -0,0 +1,60 @@ +# Atlas Deprecation Runbook (Sprint 7) + +How to retire a governed asset or field safely. Enforced by +`atlas.governance.registry.deprecation_errors` via `gate_governance`. + +## Lifecycle + +``` +ACTIVE → DEPRECATED → REMOVAL_SCHEDULED → REMOVED +``` + +An asset's `lifecycle_status` lives in its governance metadata (dbt +`meta.governance` or `non_dbt_assets.yml`). Any non-ACTIVE status requires a +`deprecation` block: + +```yaml +deprecation: + replacement: + owner: atlas-data-eng + announcement_date: "2026-07-19" + deprecation_start: "2026-07-19" + earliest_removal_date: "2026-09-01" # >= start + 30 days + migration_instructions: "How consumers move to the replacement." + validation_period: "How long the replacement runs in parallel." + removal_approval: "" # required for REMOVAL_SCHEDULED/REMOVED + rollback_limitations: "What cannot be undone after removal." + change_record: CHG-YYYYMMDD-slug # must exist under governance/changes/ +``` + +## Procedure + +1. **Announce (ACTIVE → DEPRECATED).** Add the `deprecation` block with a + replacement, a change record under `governance/changes/`, and a removal date + at least 30 days out. Regenerate the catalog. +2. **Migrate consumers.** Use `python -m atlas.governance.impact --asset ` + to enumerate affected consumers/owners; migrate each to the replacement. +3. **Schedule removal (DEPRECATED → REMOVAL_SCHEDULED).** Only after consumers + are migrated. Requires `removal_approval`. CI blocks scheduling while active + consumers still read the asset. +4. **Remove (REMOVAL_SCHEDULED → REMOVED).** After the earliest removal date and + with zero active consumers. Removal of canonical schema requires + `ATLAS_APPROVE_SCHEMA_MUTATION=true`. + +## CI rejects + +| Condition | Rule | +| --- | --- | +| Deprecated asset without a replacement | `require_replacement` | +| Removal date sooner than the minimum window | `minimum_window_days` (30) | +| Removed/scheduled asset with active consumers | active-consumer check | +| Lifecycle change without a change record | `require_change_record` + existence | +| REMOVAL_SCHEDULED/REMOVED without approval | `removal_approval` required | +| Reused field name with changed semantics | ADR-017 `type_changed` / PROHIBITED | + +## Demonstration + +Enforcement is proven by fixtures in `tests/unit/test_deprecation.py` (missing +replacement, short window, missing/unknown change record, removed-with-consumer, +missing approval). No critical Atlas field is removed for demonstration — the +harmless path is exercised with fixture records only. diff --git a/docs/evidence-sprint4/composer-deploy-session-history.txt b/docs/evidence-sprint4/composer-deploy-session-history.txt new file mode 100644 index 0000000..14e6c54 --- /dev/null +++ b/docs/evidence-sprint4/composer-deploy-session-history.txt @@ -0,0 +1,2433 @@ +export PATH="$HOME/google-cloud-sdk/bin:$PATH" && source /tmp/atlas-venv/bin/act +ivate && export GOOGLE_APPLICATION_CREDENTIALS=/tmp/atlas-adc.json && ATLAS_APPR +OVE_DEPLOY=true bash scripts/deploy_atlas_release.sh --git-sha dd7dd5d42f5b8cbfe +e625e6fdae0edb8d9e8663a --leave-paused 2>&1 | tee /tmp/atlas-deploy.log; echo "D +EPLOY_EXIT=$?" +project-atlas $ export PATH="$HOME/google-cloud-sdk/bin:$PATH" && source /tmp/at +las-venv/bin/activate && export GOOGLE_APPLICATION_CREDENTIALS=/tmp/atlas-adc.js +on && ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas_release.sh --git-sha d +d7dd5d42f5b8cbfee625e6fdae0edb8d9e8663a --leave-paused 2>&1 | tee /tmp/atlas-dep +loy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: dd7dd5d42f5b8cbfee625e6fdae0edb8d9e8663a → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260718T225900Z-dd7dd5d4 +smoke batch: atlas-smoke-dd7dd5d4-local1784415540 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/dd7 +dd5d42f5b8cbfee625e6fdae0edb8d9e8663a +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/dd7dd5d42f5b +8cbfee625e6fdae0edb8d9e8663a/atlas-bundle.tar.gz to file:///tmp/tmp.Xt2KXKx5pU/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/dd7dd5d42f5b +8cbfee625e6fdae0edb8d9e8663a/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.Xt2KX +Kx5pU/atlas-bundle.tar.gz.sha256 + +. +sha256sum: atlas-bundle-dd7dd5d42f5b8cbfee625e6fdae0edb8d9e8663a.tar.gz: No such + file or directory +atlas-bundle-dd7dd5d42f5b8cbfee625e6fdae0edb8d9e8663a.tar.gz: FAILED open or rea +d +sha256sum: WARNING: 1 listed file could not be read +FATAL: archive checksum mismatch for release dd7dd5d42f5b8cbfee625e6fdae0edb8d9e +8663a +audit: atlas-dev-20260718T225900Z-dd7dd5d4 -> RUNNING +STAGE FAILED: fetch_release — release bundle missing or checksum-invalid for dd7 +dd5d42f5b8cbfee625e6fdae0edb8d9e8663a +audit: atlas-dev-20260718T225900Z-dd7dd5d4 -> FAILED (stage fetch_release) +Recovery: inspect logs above, then re-run this script with the same + --git-sha dd7dd5d42f5b8cbfee625e6fdae0edb8d9e8663a (deployment records are ide +mpotent per deployment_id) +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha 2aeff26ea9371e9dd130a81db81cc78f6639cf67 --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: 2aeff26ea9371e9dd130a81db81cc78f6639cf67 → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260718T230144Z-2aeff26e +smoke batch: atlas-smoke-2aeff26e-local1784415704 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/2ae +ff26ea9371e9dd130a81db81cc78f6639cf67 +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/2aeff26ea937 +1e9dd130a81db81cc78f6639cf67/atlas-bundle.tar.gz to file:///tmp/tmp.9KfGZDSMO7/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/2aeff26ea937 +1e9dd130a81db81cc78f6639cf67/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.9KfGZ +DSMO7/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: 887caabe801bf5003505be14ce0502c72f07db0aed0672615a2b6 +2dacba40fca +verified 77 file checksums for 2aeff26ea937 +audit: atlas-dev-20260718T230144Z-2aeff26e -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/**, worker process 62994 thread 14008 +2717857600 listed 80... +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/airflow/requirements-airflow.txt + to gs://us-central1-atlas-dev-74134e98-bucket/data/current/airflo +w/requirements-airflow.txt +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/config/anomaly_profile.yaml to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/config/anom +aly_profile.yaml + +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/config/atlas.yaml to gs://us-cen +tral1-atlas-dev-74134e98-bucket/data/current/config/atlas.yaml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/dbt_project.yml to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas +_dbt/dbt_project.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/core/core.y +ml to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/ +atlas_dbt/models/core/core.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/core/dim_co +untries.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/cur +rent/dbt/atlas_dbt/models/core/dim_countries.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/core/dim_us +ers.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/current +/dbt/atlas_dbt/models/core/dim_users.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/core/fct_ev +ents.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/curren +t/dbt/atlas_dbt/models/core/fct_events.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/intermediat +e/int_accepted_events.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/pro +ject-atlas/current/dbt/atlas_dbt/models/intermediate/int_accepted_events.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/intermediat +e/int_event_classification.sql to gs://us-central1-atlas-dev-74134e98-bucket/dat +a/current/dbt/atlas_dbt/models/intermediate/int_event_classificati +on.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/intermediat +e/int_rejected_events.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/pro +ject-atlas/current/dbt/atlas_dbt/models/intermediate/int_rejected_events.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/intermediat +e/intermediate.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/dbt/atlas_dbt/models/intermediate/intermediate.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/marts/mart_ +daily_event_metrics.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/proje +ct-atlas/current/dbt/atlas_dbt/models/marts/mart_daily_event_metrics.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/marts/marts +.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/current/db +t/atlas_dbt/models/marts/marts.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/sources/sou +rces.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/curren +t/dbt/atlas_dbt/models/sources/sources.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/staging/sta +ging.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/curren +t/dbt/atlas_dbt/models/staging/staging.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/models/staging/stg +_events.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/cur +rent/dbt/atlas_dbt/models/staging/stg_events.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/package-lock.yml t +o gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atla +s_dbt/package-lock.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/packages.yml to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_db +t/packages.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/profiles.yml.examp +le to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/ +atlas_dbt/profiles.yml.example +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/seeds/seeds.yml to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas +_dbt/seeds/seeds.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/seeds/valid_countr +y_codes.csv to gs://us-central1-atlas-dev-74134e98-bucket/data/cur +rent/dbt/atlas_dbt/seeds/valid_country_codes.csv +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/tests/assert_batch +_fact_reconciliation.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/proj +ect-atlas/current/dbt/atlas_dbt/tests/assert_batch_fact_reconciliation.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/tests/assert_fact_ +rejected_reconciliation.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/p +roject-atlas/current/dbt/atlas_dbt/tests/assert_fact_rejected_reconciliation.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/tests/assert_injec +t_failure.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/c +urrent/dbt/atlas_dbt/tests/assert_inject_failure.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/tests/assert_mart_ +fact_reconciliation.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/proje +ct-atlas/current/dbt/atlas_dbt/tests/assert_mart_fact_reconciliation.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/tests/assert_raw_c +lassification_reconciliation.sql to gs://us-central1-atlas-dev-74134e98-bucket/d +ata/current/dbt/atlas_dbt/tests/assert_raw_classification_reconcil +iation.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/atlas_dbt/tests/assert_sourc +e_anomaly_profile.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/dbt/atlas_dbt/tests/assert_source_anomaly_profile.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/profiles/profiles.yml to gs: +//us-central1-atlas-dev-74134e98-bucket/data/current/dbt/profiles/ +profiles.yml +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dbt/requirements-dbt.txt to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/requiremen +ts-dbt.txt +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/release-manifest.json to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/release-manifest. +json +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/requirements.txt to gs://us-cent +ral1-atlas-dev-74134e98-bucket/data/current/requirements.txt +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/apply_atlas_migrations.s +h to gs://us-central1-atlas-dev-74134e98-bucket/data/current/scrip +ts/apply_atlas_migrations.sh +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/atlas_step_runner.py to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/scripts/at +las_step_runner.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/generate_events.py to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/scripts/gene +rate_events.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/load_events.py to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/scripts/load_eve +nts.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/run_atlas_step.sh to gs: +//us-central1-atlas-dev-74134e98-bucket/data/current/scripts/run_a +tlas_step.sh +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/upload_events.py to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/scripts/upload +_events.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/scripts/validate_events.py to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/scripts/vali +date_events.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/create_deployments_table.sql + to gs://us-central1-atlas-dev-74134e98-bucket/data/current/sql/cr +eate_deployments_table.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/create_events_table.sql to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/sql/create_ +events_table.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/create_ops_schema.sql to gs: +//us-central1-atlas-dev-74134e98-bucket/data/current/sql/create_op +s_schema.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/create_pipeline_runs_table.s +ql to gs://us-central1-atlas-dev-74134e98-bucket/data/current/sql/ +create_pipeline_runs_table.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/create_schema_migrations_tab +le.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/current/ +sql/create_schema_migrations_table.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/migrate_sprint3.sql to gs:// +us-central1-atlas-dev-74134e98-bucket/data/current/sql/migrate_spr +int3.sql +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/sql/migrations/manifest.txt to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/sql/migrati +ons/manifest.txt +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/__init__.py to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/__init_ +_.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/batch/__init__.py to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/b +atch/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/batch/context.py to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ba +tch/context.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/batch/manifest.py to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/b +atch/manifest.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/config/__init__.py to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ +config/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/config/settings.py to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ +config/settings.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/generator/__init__.py +to gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atl +as/generator/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/generator/events.py to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas +/generator/events.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ingestion/__init__.py +to gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atl +as/ingestion/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ingestion/upload.py to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas +/ingestion/upload.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/loader/__init__.py to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ +loader/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/loader/bigquery.py to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ +loader/bigquery.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/logging/__init__.py to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas +/logging/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/logging/structured.py +to gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atl +as/logging/structured.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/__init__.py to gs: +//us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ops +/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/audit.py to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ops/au +dit.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/deployments.py to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/ +ops/deployments.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/finalizer.py to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/op +s/finalizer.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/migrations.py to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/o +ps/migrations.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/preflight.py to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/op +s/preflight.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/ops/resources.py to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/src/atlas/op +s/resources.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/pipeline/__init__.py t +o gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atla +s/pipeline/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/pipeline/orchestrator. +py to gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/ +atlas/pipeline/orchestrator.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/validation/__init__.py + to gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/at +las/validation/__init__.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/validation/checks.py t +o gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/atla +s/validation/checks.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/src/atlas/validation/warehouse.p +y to gs://us-central1-atlas-dev-74134e98-bucket/data/current/src/a +tlas/validation/warehouse.py +..... + +Average throughput: 344.8kiB/s +At file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/**, worker process 63220 thread +140431192999744 listed 6... +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/atlas_batch_pipeline.py to +gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_batch_pipeli +ne.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/atlas_orchestration/__init_ +_.py to gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_orch +estration/__init__.py + +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/atlas_orchestration/callbac +ks.py to gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_orc +hestration/callbacks.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/atlas_orchestration/command +s.py to gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_orch +estration/commands.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/atlas_orchestration/context +.py to gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_orche +stration/context.py +Copying file:///tmp/tmp.9KfGZDSMO7/atlas-bundle/dags/atlas_orchestration/validat +ion.py to gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_or +chestration/validation.py +... + +Average throughput: 38.8kiB/s +Timed out after 600s waiting for atlas_batch_pipeline to parse +Last dags list output: +Executing the command: [ airflow dags list -o plain ]... +Command has been started. execution_id=a0bfb4ef-8255-4d13-b235-c6bcd725385d +Use ctrl-c to interrupt the command +[2026-07-18T23:11:56.162054Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +dag_id fileloc owners i +s_paused bundle_name bundle_version +airflow_monitoring /home/airflow/gcs/dags/airflow_monitoring.py ['airflow'] F +alse dags-folder +STAGE FAILED: dag_parse — DAG failed to parse after promotion +audit: atlas-dev-20260718T230144Z-2aeff26e -> FAILED (stage dag_parse) +Recovery: inspect logs above, then re-run this script with the same + --git-sha 2aeff26ea9371e9dd130a81db81cc78f6639cf67 (deployment records are ide +mpotent per deployment_id) +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha 83c0d137c3d523c5f97a8f1142f86c0e506d9478 --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: 83c0d137c3d523c5f97a8f1142f86c0e506d9478 → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260718T231842Z-83c0d137 +smoke batch: atlas-smoke-83c0d137-local1784416722 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/83c +0d137c3d523c5f97a8f1142f86c0e506d9478 +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/83c0d137c3d5 +23c5f97a8f1142f86c0e506d9478/atlas-bundle.tar.gz to file:///tmp/tmp.bvggjP0Feb/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/83c0d137c3d5 +23c5f97a8f1142f86c0e506d9478/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.bvggj +P0Feb/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: 1fa6f040ab236ec9a872a4f73b8c05eb868ad164a79ac766a4ba0 +ab528061378 +verified 78 file checksums for 83c0d137c3d5 +audit: atlas-dev-20260718T231842Z-83c0d137 -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.bvggjP0Feb/atlas-bundle/**, worker process 67614 thread 14055 +0925756224 listed 80... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 67614 thread 140550925756224 listed 80... +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/release-manifest.json to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/release-manifest. +json + +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +.. + +Average throughput: 219.1kiB/s +At file:///tmp/tmp.bvggjP0Feb/atlas-bundle/dags/**, worker process 67814 thread +140685084133184 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 67814 thread 140685084133184 listed 6... +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/dags/.airflowignore to gs://us-c +entral1-atlas-dev-74134e98-bucket/dags/project_atlas/.airflowignore +Copying file:///tmp/tmp.bvggjP0Feb/atlas-bundle/dags/atlas_batch_pipeline.py to +gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/atlas_batch_pipeli +ne.py + +. + +Average throughput: 367.8kiB/s +DAG import errors detected: +Executing the command: [ airflow dags list-import-errors -o plain ]... +Command has been started. execution_id=3cb33af8-2c3b-45e5-9b7f-685eb27862b3 +Use ctrl-c to interrupt the command +ERROR: Error message: /opt/python3.11/lib/python3.11/site-packages/airflow/cli/c +ommands/dag_command.py:49 UserWarning: Could not import graphviz. Rendering grap +h to the graphical format will not be possible. +You might need to install the graphviz package and necessary system packages. +Run `pip install graphviz` to attempt to install it. +ERROR: Command exit code: 1 +[2026-07-18T23:22:16.958744Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +bundle_name filepath error +dags-folder project_atlas/atlas_orchestration/context.py Traceback (most rec +ent call last): +File "", line 241, in _call_with_frames_removed +File "/home/airflow/gcs/dags/project_atlas/atlas_orchestration/context.py", line + 8, in +from atlas.batch.context import resolve_batch_context +ModuleNotFoundError: No module named 'atlas' +STAGE FAILED: dag_parse — DAG failed to parse after promotion +audit: atlas-dev-20260718T231842Z-83c0d137 -> FAILED (stage dag_parse) +Recovery: inspect logs above, then re-run this script with the same + --git-sha 83c0d137c3d523c5f97a8f1142f86c0e506d9478 (deployment records are ide +mpotent per deployment_id) +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha 74732eeeb746a9fb1461c4ad35ff01bb912ff454 --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: 74732eeeb746a9fb1461c4ad35ff01bb912ff454 → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260718T233128Z-74732eee +smoke batch: atlas-smoke-74732eee-local1784417488 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/747 +32eeeb746a9fb1461c4ad35ff01bb912ff454 +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/74732eeeb746 +a9fb1461c4ad35ff01bb912ff454/atlas-bundle.tar.gz to file:///tmp/tmp.QHlm4Bas5k/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/74732eeeb746 +a9fb1461c4ad35ff01bb912ff454/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.QHlm4 +Bas5k/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: bc76f1c777f8508c28ec258bde617f92130e5189932eda754ca99 +c46e38cd755 +verified 78 file checksums for 74732eeeb746 +audit: atlas-dev-20260718T233128Z-74732eee -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/**, worker process 70930 thread 14043 +9276128064 listed 80... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 70930 thread 140439276128064 listed 80... +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc + +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +... + +Average throughput: 151.5kiB/s +At file:///tmp/tmp.QHlm4Bas5k/atlas-bundle/dags/**, worker process 71131 thread +140312244664128 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 71131 thread 140312244664128 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260718T233128Z-74732eee +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smok +e__atlas-dev-20260718T233128Z-74732eee --conf {"batch_id": "atlas-smoke-74732eee +-local1784417488", "pipeline_run_id": "atlas-smoke-74732eee-local1784417488-run" +, "processing_date": "2026-07-18"} ]... +Command has been started. execution_id=810f3486-d017-4874-9167-d55685585007 +Use ctrl-c to interrupt the command +[2026-07-18T23:32:21.946161Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +| | | data_interval_star | + | | last_scheduling_d | | | | +| triggering_user_nam +conf | dag_id | dag_run_id | t +| data_interval_end | end_date | ecision | logical_date | run_type | s +tart_date | state | e +===================+===================+===================+==================== ++===================+==========+===================+==============+==========+== +==========+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None +| None | None | None | None | manual | N +one | queued | airflow +'atlas-smoke-74732 | ine | 20260718T233128Z- | +| | | | | | + | | +eee-local178441748 | | 74732eee | +| | | | | | + | | +8', | | | +| | | | | | + | | +'pipeline_run_id': | | | +| | | | | | + | | +'atlas-smoke-74732 | | | +| | | | | | + | | +eee-local178441748 | | | +| | | | | | + | | +8-run', | | | +| | | | | | + | | +'processing_date': | | | +| | | | | | + | | +'2026-07-18'} | | | +| | | | | | + | | +ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas_release.sh --git-sha f9959cb +66729442a06501c372cfb39e5024f4d9f --leave-paused 2>&1 | tee /tmp/atlas-deploy.lo +g; echo "DEPLOY_EXIT=$?" +Smoke run smoke__atlas-dev-20260718T233128Z-74732eee: timed out after 2400s (las +t state: unknown) +STAGE FAILED: smoke_batch — smoke run did not reach terminal SUCCESS +audit: atlas-dev-20260718T233128Z-74732eee -> FAILED (stage smoke_batch) +Recovery: inspect logs above, then re-run this script with the same + --git-sha 74732eeeb746a9fb1461c4ad35ff01bb912ff454 (deployment records are ide +mpotent per deployment_id) +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha f9959cb66729442a06501c372cfb39e5024f4d9f --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: f9959cb66729442a06501c372cfb39e5024f4d9f → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260719T001246Z-f9959cb6 +smoke batch: atlas-smoke-f9959cb6-local1784419966 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/f99 +59cb66729442a06501c372cfb39e5024f4d9f +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/f9959cb66729 +442a06501c372cfb39e5024f4d9f/atlas-bundle.tar.gz to file:///tmp/tmp.p0nHlZU13J/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/f9959cb66729 +442a06501c372cfb39e5024f4d9f/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.p0nHl +ZU13J/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: 10e3650a7a9f4f14d5bdeca70677bd5a40a4cd11fcb2433618b36 +9b93e595f76 +verified 314 file checksums for f9959cb66729 +audit: atlas-dev-20260719T001246Z-f9959cb6 -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.p0nHlZU13J/atlas-bundle/**, worker process 84602 thread 13967 +7369792320 listed 316... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 84602 thread 139677369792320 listed 87... + +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.circleci/config.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/pro +ject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.circleci/config.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/CODEOWNERS to gs://us-central1-atlas-dev-74134e98-bucket/data/proje +ct-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.github/CODEOWNERS +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/ISSUE_TEMPLATE/bug_report.md to gs://us-central1-atlas-dev-74134e98 +-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.github/ +ISSUE_TEMPLATE/bug_report.md +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-20260718/events.jsonl#1784417574812811... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-74732eee-local1784417488/events.jsonl#1784417666462878... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-74732eee-local1784417488/manifest.json#1784417667283831... +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/ISSUE_TEMPLATE/dbt_minor_release.md to gs://us-central1-atlas-dev-7 +4134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/. +github/ISSUE_TEMPLATE/dbt_minor_release.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/ISSUE_TEMPLATE/feature_request.md to gs://us-central1-atlas-dev-741 +34e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.gi +thub/ISSUE_TEMPLATE/feature_request.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/ISSUE_TEMPLATE/utils_minor_release.md to gs://us-central1-atlas-dev +-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils +/.github/ISSUE_TEMPLATE/utils_minor_release.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/pull_request_template.md to gs://us-central1-atlas-dev-74134e98-buc +ket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.github/pull +_request_template.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/workflows/ci.yml to gs://us-central1-atlas-dev-74134e98-bucket/data +/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.github/workflows/ci +.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/workflows/create-table-of-contents.yml to gs://us-central1-atlas-de +v-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_util +s/.github/workflows/create-table-of-contents.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/workflows/stale.yml to gs://us-central1-atlas-dev-74134e98-bucket/d +ata/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.github/workflows +/stale.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.github/workflows/triage-labels.yml to gs://us-central1-atlas-dev-74134e98- +bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.github/w +orkflows/triage-labels.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/.gitignore to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/dbt/atlas_dbt/dbt_packages/dbt_utils/.gitignore +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/CHANGELOG.md to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atl +as/current/dbt/atlas_dbt/dbt_packages/dbt_utils/CHANGELOG.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/CONTRIBUTING.md to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/CONTRIBUTING.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/LICENSE to gs://us-central1-atlas-dev-74134e98-bucket/data/cu +rrent/dbt/atlas_dbt/dbt_packages/dbt_utils/LICENSE +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/Makefile to gs://us-central1-atlas-dev-74134e98-bucket/data/c +urrent/dbt/atlas_dbt/dbt_packages/dbt_utils/Makefile +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/README.md to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/dbt/atlas_dbt/dbt_packages/dbt_utils/README.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/RELEASE.md to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/dbt/atlas_dbt/dbt_packages/dbt_utils/RELEASE.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/dbt_project.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/dbt_project.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/dev-requirements.txt to gs://us-central1-atlas-dev-74134e98-bucket/data/pro +ject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/dev-requirements.txt +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/docker-compose.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/proje +ct-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/docker-compose.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/docs/decisions/README.md to gs://us-central1-atlas-dev-74134e98-bucket/data +/current/dbt/atlas_dbt/dbt_packages/dbt_utils/docs/decisions/READM +E.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/docs/decisions/adr-0000-documenting-architecture-decisions.md to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/docs/decisions/adr-0000-documenting-architecture-decisions.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/docs/decisions/adr-0001-decision-record-format.md to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/docs/decisions/adr-0001-decision-record-format.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/docs/decisions/adr-0002-cross-database-utils.md to gs://us-central1-atlas-d +ev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_uti +ls/docs/decisions/adr-0002-cross-database-utils.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/.env/bigquery.env to gs://us-central1-atlas-dev-74134e98- +bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integrati +on_tests/.env/bigquery.env +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/.env/postgres.env to gs://us-central1-atlas-dev-74134e98- +bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integrati +on_tests/.env/postgres.env +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/.env/redshift.env to gs://us-central1-atlas-dev-74134e98- +bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integrati +on_tests/.env/redshift.env +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-20260718/manifest.json#1784417575692420... +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/.env/snowflake.env to gs://us-central1-atlas-dev-74134e98 +-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integrat +ion_tests/.env/snowflake.env +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/.gitignore to gs://us-central1-atlas-dev-74134e98-bucket/ +data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_test +s/.gitignore +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/README.md to gs://us-central1-atlas-dev-74134e98-bucket/d +ata/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_tests +/README.md +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/.gitkeep to gs://us-central1-atlas-dev-74134e98-buck +et/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_t +ests/data/.gitkeep +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/datetime/data_date_spine.csv to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/data/datetime/data_date_spine.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/etc/data_people.csv to gs://us-central1-atlas-dev-74 +134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/in +tegration_tests/data/etc/data_people.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/geo/data_haversine_km.csv to gs://us-central1-atlas- +dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ut +ils/integration_tests/data/geo/data_haversine_km.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/geo/data_haversine_mi.csv to gs://us-central1-atlas- +dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ut +ils/integration_tests/data/geo/data_haversine_mi.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_cardinality_equality_a.csv to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_cardinality_equa +lity_a.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_cardinality_equality_b.csv to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_cardinality_equa +lity_b.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_not_null_proportion.csv to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/data/schema_tests/data_not_null_proportion +.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_accepted_range.csv to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/data/schema_tests/data_test_accepted_range +.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_at_least_one.csv to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/data/schema_tests/data_test_at_least_one.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equal_rowcount.csv to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/data/schema_tests/data_test_equal_rowcount +.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equality_a.csv to gs://us-cen +tral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pac +kages/dbt_utils/integration_tests/data/schema_tests/data_test_equality_a.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equality_b.csv to gs://us-cen +tral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pac +kages/dbt_utils/integration_tests/data/schema_tests/data_test_equality_b.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equality_floats_a.csv to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_equality_fl +oats_a.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equality_floats_b.csv to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_equality_fl +oats_b.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equality_floats_columns_a.csv + to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/at +las_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_equ +ality_floats_columns_a.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_equality_floats_columns_b.csv + to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/at +las_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_equ +ality_floats_columns_b.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_expression_is_true.csv to gs: +//us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt +/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_expression +_is_true.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_fewer_rows_than_table_1.csv t +o gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atla +s_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_fewer +_rows_than_table_1.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_fewer_rows_than_table_2.csv t +o gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atla +s_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_fewer +_rows_than_table_2.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_mutually_exclusive_ranges_no_ +gaps.csv to gs://us-central1-atlas-dev-74134e98-bucket/data/curren +t/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_ +test_mutually_exclusive_ranges_no_gaps.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_mutually_exclusive_ranges_wit +h_gaps.csv to gs://us-central1-atlas-dev-74134e98-bucket/data/curr +ent/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/dat +a_test_mutually_exclusive_ranges_with_gaps.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_mutually_exclusive_ranges_wit +h_gaps_zero_length.csv to gs://us-central1-atlas-dev-74134e98-bucket/data/projec +t-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_tests/data/sche +ma_tests/data_test_mutually_exclusive_ranges_with_gaps_zero_length.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_not_accepted_values.csv to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_db +t/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_not_accep +ted_values.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_not_constant.csv to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/data/schema_tests/data_test_not_constant.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_relationships_where_table_1.c +sv to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/ +atlas_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_r +elationships_where_table_1.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_relationships_where_table_2.c +sv to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/ +atlas_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_r +elationships_where_table_2.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_sequential_timestamps.csv to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_ +dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_sequent +ial_timestamps.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_test_sequential_values.csv to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_test_sequential_ +values.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/data_unique_combination_of_columns.csv +to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atl +as_dbt/dbt_packages/dbt_utils/integration_tests/data/schema_tests/data_unique_co +mbination_of_columns.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/schema_tests/schema.yml to gs://us-central1-atlas-de +v-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_util +s/integration_tests/data/schema_tests/schema.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_deduplicate.csv to gs://us-central1-atlas-d +ev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_uti +ls/integration_tests/data/sql/data_deduplicate.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_deduplicate_expected.csv to gs://us-central +1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_package +s/dbt_utils/integration_tests/data/sql/data_deduplicate_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_events_20180101.csv to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/data/sql/data_events_20180101.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_events_20180102.csv to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/data/sql/data_events_20180102.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_events_20180103.csv to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/data/sql/data_events_20180103.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_filtered_columns_in_relation.csv to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/data/sql/data_filtered_columns_in_relation +.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_filtered_columns_in_relation_expected.csv t +o gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atla +s_dbt/dbt_packages/dbt_utils/integration_tests/data/sql/data_filtered_columns_in +_relation_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_generate_series.csv to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/data/sql/data_generate_series.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_generate_surrogate_key.csv to gs://us-centr +al1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packa +ges/dbt_utils/integration_tests/data/sql/data_generate_surrogate_key.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_get_column_values.csv to gs://us-central1-a +tlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/d +bt_utils/integration_tests/data/sql/data_get_column_values.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_get_column_values_dropped.csv to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/integration_tests/data/sql/data_get_column_values_dropped.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_get_column_values_where.csv to gs://us-cent +ral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pack +ages/dbt_utils/integration_tests/data/sql/data_get_column_values_where.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_get_column_values_where_expected.csv to gs: +//us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt +/dbt_packages/dbt_utils/integration_tests/data/sql/data_get_column_values_where_ +expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_get_query_results_as_dict.csv to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/integration_tests/data/sql/data_get_query_results_as_dict.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_get_single_value.csv to gs://us-central1-at +las-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/db +t_utils/integration_tests/data/sql/data_get_single_value.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_nullcheck_table.csv to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/data/sql/data_nullcheck_table.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_pivot.csv to gs://us-central1-atlas-dev-741 +34e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/int +egration_tests/data/sql/data_pivot.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_pivot_expected.csv to gs://us-central1-atla +s-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/data/sql/data_pivot_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_pivot_expected_apostrophe.csv to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/integration_tests/data/sql/data_pivot_expected_apostrophe.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_safe_add.csv to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +integration_tests/data/sql/data_safe_add.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_safe_divide.csv to gs://us-central1-atlas-d +ev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_uti +ls/integration_tests/data/sql/data_safe_divide.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_safe_divide_denominator_expressions.csv to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_ +dbt/dbt_packages/dbt_utils/integration_tests/data/sql/data_safe_divide_denominat +or_expressions.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_safe_divide_numerator_expressions.csv to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_db +t/dbt_packages/dbt_utils/integration_tests/data/sql/data_safe_divide_numerator_e +xpressions.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_safe_subtract.csv to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_safe_subtract.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star.csv to gs://us-central1-atlas-dev-7413 +4e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/inte +gration_tests/data/sql/data_star.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star_aggregate.csv to gs://us-central1-atla +s-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/data/sql/data_star_aggregate.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star_aggregate_expected.csv to gs://us-cent +ral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pack +ages/dbt_utils/integration_tests/data/sql/data_star_aggregate_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star_expected.csv to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star_prefix_suffix_expected.csv to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_ +packages/dbt_utils/integration_tests/data/sql/data_star_prefix_suffix_expected.c +sv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_star_quote_identifiers.csv to gs://us-centr +al1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packa +ges/dbt_utils/integration_tests/data/sql/data_star_quote_identifiers.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_events_expected.csv to gs://us-centra +l1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packag +es/dbt_utils/integration_tests/data/sql/data_union_events_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_exclude_expected.csv to gs://us-centr +al1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packa +ges/dbt_utils/integration_tests/data/sql/data_union_exclude_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_expected.csv to gs://us-central1-atla +s-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/data/sql/data_union_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_table_1.csv to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_table_1.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_table_2.csv to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_union_table_2.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot.csv to gs://us-central1-atlas-dev-7 +4134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/i +ntegration_tests/data/sql/data_unpivot.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_bool.csv to gs://us-central1-atlas- +dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ut +ils/integration_tests/data/sql/data_unpivot_bool.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_bool_expected.csv to gs://us-centra +l1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packag +es/dbt_utils/integration_tests/data/sql/data_unpivot_bool_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_expected.csv to gs://us-central1-at +las-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/db +t_utils/integration_tests/data/sql/data_unpivot_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_original_api_expected.csv to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/db +t_packages/dbt_utils/integration_tests/data/sql/data_unpivot_original_api_expect +ed.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_quote.csv to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_quote.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_unpivot_quote_expected.csv to gs://us-centr +al1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packa +ges/dbt_utils/integration_tests/data/sql/data_unpivot_quote_expected.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/sql/data_width_bucket.csv to gs://us-central1-atlas- +dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ut +ils/integration_tests/data/sql/data_width_bucket.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/web/data_url_host.csv to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +integration_tests/data/web/data_url_host.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/web/data_url_path.csv to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +integration_tests/data/web/data_url_path.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/data/web/data_urls.csv to gs://us-central1-atlas-dev-7413 +4e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/inte +gration_tests/data/web/data_urls.csv +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/dbt_project.yml to gs://us-central1-atlas-dev-74134e98-bu +cket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration +_tests/dbt_project.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/macros/.gitkeep to gs://us-central1-atlas-dev-74134e98-bu +cket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration +_tests/macros/.gitkeep +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/macros/assert_equal_values.sql to gs://us-central1-atlas- +dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ut +ils/integration_tests/macros/assert_equal_values.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/macros/limit_zero.sql to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integ +ration_tests/macros/limit_zero.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/macros/tests.sql to gs://us-central1-atlas-dev-74134e98-b +ucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integratio +n_tests/macros/tests.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/datetime/schema.yml to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +integration_tests/models/datetime/schema.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/datetime/test_date_spine.sql to gs://us-central1-a +tlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/d +bt_utils/integration_tests/models/datetime/test_date_spine.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/equality_less_columns.sql to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/models/generic_tests/equality_less_columns +.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/recency_time_excluded.sql to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/models/generic_tests/recency_time_excluded +.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/recency_time_included.sql to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt +_packages/dbt_utils/integration_tests/models/generic_tests/recency_time_included +.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/schema.yml to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/schema.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/test_equal_column_subset.sql to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/models/generic_tests/test_equal_column_ +subset.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/test_equal_rowcount.sql to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/models/generic_tests/test_equal_rowcount.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/generic_tests/test_fewer_rows_than.sql to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_ +packages/dbt_utils/integration_tests/models/generic_tests/test_fewer_rows_than.s +ql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/geo/schema.yml to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integ +ration_tests/models/geo/schema.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/geo/test_haversine_distance_km.sql to gs://us-cent +ral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pack +ages/dbt_utils/integration_tests/models/geo/test_haversine_distance_km.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/geo/test_haversine_distance_mi.sql to gs://us-cent +ral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pack +ages/dbt_utils/integration_tests/models/geo/test_haversine_distance_mi.sql +.Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/models/sql/schema.yml to gs://us-central1-atlas-dev-7413 +4e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/inte +gration_tests/models/sql/schema.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_deduplicate.sql to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_deduplicate.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_generate_series.sql to gs://us-central1-a +tlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/d +bt_utils/integration_tests/models/sql/test_generate_series.sql +.Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/models/sql/test_generate_surrogate_key.sql to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/integration_tests/models/sql/test_generate_surrogate_key.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_get_column_values.sql to gs://us-central1 +-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages +/dbt_utils/integration_tests/models/sql/test_get_column_values.sql +.Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/models/sql/test_get_column_values_where.sql to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/models/sql/test_get_column_values_where.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_get_filtered_columns_in_relation.sql to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_d +bt/dbt_packages/dbt_utils/integration_tests/models/sql/test_get_filtered_columns +_in_relation.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_get_relations_by_pattern.sql to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/models/sql/test_get_relations_by_pattern.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_get_relations_by_prefix_and_union.sql to +gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_ +dbt/dbt_packages/dbt_utils/integration_tests/models/sql/test_get_relations_by_pr +efix_and_union.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_get_single_value.sql to gs://us-central1- +atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/ +dbt_utils/integration_tests/models/sql/test_get_single_value.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_get_single_value_default.sql to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/models/sql/test_get_single_value_default.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_groupby.sql to gs://us-central1-atlas-dev +-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils +/integration_tests/models/sql/test_groupby.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_not_empty_string_failing.sql to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/models/sql/test_not_empty_string_failing.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_not_empty_string_passing.sql to gs://us-c +entral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_p +ackages/dbt_utils/integration_tests/models/sql/test_not_empty_string_passing.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_nullcheck_table.sql to gs://us-central1-a +tlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/d +bt_utils/integration_tests/models/sql/test_nullcheck_table.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_pivot.sql to gs://us-central1-atlas-dev-7 +4134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/i +ntegration_tests/models/sql/test_pivot.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_pivot_apostrophe.sql to gs://us-central1- +atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/ +dbt_utils/integration_tests/models/sql/test_pivot_apostrophe.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_safe_add.sql to gs://us-central1-atlas-de +v-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_util +s/integration_tests/models/sql/test_safe_add.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_safe_divide.sql to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_safe_divide.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_safe_subtract.sql to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/models/sql/test_safe_subtract.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_star.sql to gs://us-central1-atlas-dev-74 +134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/in +tegration_tests/models/sql/test_star.sql +.Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/models/sql/test_star_aggregate.sql to gs://us-central1-a +tlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/d +bt_utils/integration_tests/models/sql/test_star_aggregate.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_star_no_columns.sql to gs://us-central1-a +tlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/d +bt_utils/integration_tests/models/sql/test_star_no_columns.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_star_prefix_suffix.sql to gs://us-central +1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_package +s/dbt_utils/integration_tests/models/sql/test_star_prefix_suffix.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_star_quote_identifiers.sql to gs://us-cen +tral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pac +kages/dbt_utils/integration_tests/models/sql/test_star_quote_identifiers.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_star_unquote_aliases.sql to gs://us-centr +al1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packa +ges/dbt_utils/integration_tests/models/sql/test_star_unquote_aliases.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_star_uppercase.sql to gs://us-central1-at +las-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/db +t_utils/integration_tests/models/sql/test_star_uppercase.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union.sql to gs://us-central1-atlas-dev-7 +4134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/i +ntegration_tests/models/sql/test_union.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_base.sql to gs://us-central1-atlas- +dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ut +ils/integration_tests/models/sql/test_union_base.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_exclude_base_lowercase.sql to gs:// +us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/d +bt_packages/dbt_utils/integration_tests/models/sql/test_union_exclude_base_lower +case.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_exclude_base_uppercase.sql to gs:// +us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/d +bt_packages/dbt_utils/integration_tests/models/sql/test_union_exclude_base_upper +case.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_exclude_lowercase.sql to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/integration_tests/models/sql/test_union_exclude_lowercase.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_exclude_uppercase.sql to gs://us-ce +ntral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pa +ckages/dbt_utils/integration_tests/models/sql/test_union_exclude_uppercase.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_no_source_column.sql to gs://us-cen +tral1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_pac +kages/dbt_utils/integration_tests/models/sql/test_union_no_source_column.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_where.sql to gs://us-central1-atlas +-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_where.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_union_where_base.sql to gs://us-central1- +atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/ +dbt_utils/integration_tests/models/sql/test_union_where_base.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_unpivot.sql to gs://us-central1-atlas-dev +-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils +/integration_tests/models/sql/test_unpivot.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_unpivot_bool.sql to gs://us-central1-atla +s-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/models/sql/test_unpivot_bool.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_unpivot_quote.sql to gs://us-central1-atl +as-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt +_utils/integration_tests/models/sql/test_unpivot_quote.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/sql/test_width_bucket.sql to gs://us-central1-atla +s-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ +utils/integration_tests/models/sql/test_width_bucket.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/web/schema.yml to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integ +ration_tests/models/web/schema.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/web/test_url_host.sql to gs://us-central1-atlas-de +v-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_util +s/integration_tests/models/web/test_url_host.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/web/test_url_path.sql to gs://us-central1-atlas-de +v-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_util +s/integration_tests/models/web/test_url_path.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/models/web/test_urls.sql to gs://us-central1-atlas-dev-74 +134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/in +tegration_tests/models/web/test_urls.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/package-lock.yml to gs://us-central1-atlas-dev-74134e98-b +ucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integratio +n_tests/package-lock.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/packages.yml to gs://us-central1-atlas-dev-74134e98-bucke +t/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_te +sts/packages.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/profiles.yml to gs://us-central1-atlas-dev-74134e98-bucke +t/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/integration_te +sts/profiles.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/assert_get_query_results_as_dict_objects_equal.sql +to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atl +as_dbt/dbt_packages/dbt_utils/integration_tests/tests/assert_get_query_results_a +s_dict_objects_equal.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/generic/expect_table_columns_to_match_set.sql to gs +://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_db +t/dbt_packages/dbt_utils/integration_tests/tests/generic/expect_table_columns_to +_match_set.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/jinja_helpers/assert_pretty_output_msg_is_string.sq +l to gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/a +tlas_dbt/dbt_packages/dbt_utils/integration_tests/tests/jinja_helpers/assert_pre +tty_output_msg_is_string.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/jinja_helpers/assert_pretty_time_is_string.sql to g +s://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_d +bt/dbt_packages/dbt_utils/integration_tests/tests/jinja_helpers/assert_pretty_ti +me_is_string.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/jinja_helpers/test_slugify.sql to gs://us-central1- +atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/ +dbt_utils/integration_tests/tests/jinja_helpers/test_slugify.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/sql/test_get_column_values_use_default.sql to gs:// +us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/d +bt_packages/dbt_utils/integration_tests/tests/sql/test_get_column_values_use_def +ault.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/integration_tests/tests/sql/test_get_single_value_multiple_rows.sql to gs:/ +/us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/ +dbt_packages/dbt_utils/integration_tests/tests/sql/test_get_single_value_multipl +e_rows.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/accepted_range.sql to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macro +s/generic_tests/accepted_range.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/at_least_one.sql to gs://us-central1-atlas-dev-74134e9 +8-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/ +generic_tests/at_least_one.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/cardinality_equality.sql to gs://us-central1-atlas-dev +-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils +/macros/generic_tests/cardinality_equality.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/equal_rowcount.sql to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macro +s/generic_tests/equal_rowcount.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/equality.sql to gs://us-central1-atlas-dev-74134e98-bu +cket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/gene +ric_tests/equality.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/expression_is_true.sql to gs://us-central1-atlas-dev-7 +4134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/m +acros/generic_tests/expression_is_true.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/fewer_rows_than.sql to gs://us-central1-atlas-dev-7413 +4e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macr +os/generic_tests/fewer_rows_than.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/mutually_exclusive_ranges.sql to gs://us-central1-atla +s-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_ +utils/macros/generic_tests/mutually_exclusive_ranges.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/not_accepted_values.sql to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +macros/generic_tests/not_accepted_values.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/not_constant.sql to gs://us-central1-atlas-dev-74134e9 +8-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/ +generic_tests/not_constant.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/not_empty_string.sql to gs://us-central1-atlas-dev-741 +34e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/mac +ros/generic_tests/not_empty_string.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/not_null_proportion.sql to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +macros/generic_tests/not_null_proportion.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/recency.sql to gs://us-central1-atlas-dev-74134e98-buc +ket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/gener +ic_tests/recency.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/relationships_where.sql to gs://us-central1-atlas-dev- +74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ +macros/generic_tests/relationships_where.sql +.Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_ +utils/macros/generic_tests/sequential_values.sql to gs://us-central1-atlas-dev-7 +4134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/m +acros/generic_tests/sequential_values.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/generic_tests/unique_combination_of_columns.sql to gs://us-central1- +atlas-dev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/ +dbt_utils/macros/generic_tests/unique_combination_of_columns.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/jinja_helpers/_is_ephemeral.sql to gs://us-central1-atlas-dev-74134e +98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros +/jinja_helpers/_is_ephemeral.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/jinja_helpers/_is_relation.sql to gs://us-central1-atlas-dev-74134e9 +8-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/ +jinja_helpers/_is_relation.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/jinja_helpers/log_info.sql to gs://us-central1-atlas-dev-74134e98-bu +cket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/jinj +a_helpers/log_info.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/jinja_helpers/pretty_log_format.sql to gs://us-central1-atlas-dev-74 +134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/ma +cros/jinja_helpers/pretty_log_format.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/jinja_helpers/pretty_time.sql to gs://us-central1-atlas-dev-74134e98 +-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/j +inja_helpers/pretty_time.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/jinja_helpers/slugify.sql to gs://us-central1-atlas-dev-74134e98-buc +ket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/jinja +_helpers/slugify.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/date_spine.sql to gs://us-central1-atlas-dev-74134e98-bucket/dat +a/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/date_spi +ne.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/deduplicate.sql to gs://us-central1-atlas-dev-74134e98-bucket/da +ta/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/dedupli +cate.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/generate_series.sql to gs://us-central1-atlas-dev-74134e98-bucke +t/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/gen +erate_series.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/generate_surrogate_key.sql to gs://us-central1-atlas-dev-74134e9 +8-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/ +sql/generate_surrogate_key.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_column_values.sql to gs://us-central1-atlas-dev-74134e98-buc +ket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/g +et_column_values.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_filtered_columns_in_relation.sql to gs://us-central1-atlas-d +ev-74134e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_uti +ls/macros/sql/get_filtered_columns_in_relation.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_query_results_as_dict.sql to gs://us-central1-atlas-dev-7413 +4e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macr +os/sql/get_query_results_as_dict.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_relations_by_pattern.sql to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macro +s/sql/get_relations_by_pattern.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_relations_by_prefix.sql to gs://us-central1-atlas-dev-74134e +98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros +/sql/get_relations_by_prefix.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_single_value.sql to gs://us-central1-atlas-dev-74134e98-buck +et/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/ge +t_single_value.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_table_types_sql.sql to gs://us-central1-atlas-dev-74134e98-b +ucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql +/get_table_types_sql.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_tables_by_pattern_sql.sql to gs://us-central1-atlas-dev-7413 +4e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macr +os/sql/get_tables_by_pattern_sql.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/get_tables_by_prefix_sql.sql to gs://us-central1-atlas-dev-74134 +e98-bucket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macro +s/sql/get_tables_by_prefix_sql.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/groupby.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/p +roject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/groupby.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/haversine_distance.sql to gs://us-central1-atlas-dev-74134e98-bu +cket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/ +haversine_distance.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/nullcheck.sql to gs://us-central1-atlas-dev-74134e98-bucket/data +/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/nullcheck +.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/nullcheck_table.sql to gs://us-central1-atlas-dev-74134e98-bucke +t/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/nul +lcheck_table.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/pivot.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/pro +ject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/pivot.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/safe_add.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/safe_add.s +ql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/safe_divide.sql to gs://us-central1-atlas-dev-74134e98-bucket/da +ta/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/safe_di +vide.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/safe_subtract.sql to gs://us-central1-atlas-dev-74134e98-bucket/ +data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/safe_ +subtract.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/star.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/proj +ect-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/star.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/surrogate_key.sql to gs://us-central1-atlas-dev-74134e98-bucket/ +data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/surro +gate_key.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/union.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/pro +ject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/union.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/unpivot.sql to gs://us-central1-atlas-dev-74134e98-bucket/data/p +roject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/unpivot.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/sql/width_bucket.sql to gs://us-central1-atlas-dev-74134e98-bucket/d +ata/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/sql/width_ +bucket.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/web/get_url_host.sql to gs://us-central1-atlas-dev-74134e98-bucket/d +ata/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/web/get_ur +l_host.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/web/get_url_parameter.sql to gs://us-central1-atlas-dev-74134e98-buc +ket/data/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/web/g +et_url_parameter.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/macros/web/get_url_path.sql to gs://us-central1-atlas-dev-74134e98-bucket/d +ata/current/dbt/atlas_dbt/dbt_packages/dbt_utils/macros/web/get_ur +l_path.sql +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/pytest.ini to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/dbt/atlas_dbt/dbt_packages/dbt_utils/pytest.ini +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/run_functional_test.sh to gs://us-central1-atlas-dev-74134e98-bucket/data/p +roject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/run_functional_test.sh +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/run_test.sh to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atla +s/current/dbt/atlas_dbt/dbt_packages/dbt_utils/run_test.sh +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/supported_adapters.env to gs://us-central1-atlas-dev-74134e98-bucket/data/p +roject-atlas/current/dbt/atlas_dbt/dbt_packages/dbt_utils/supported_adapters.env +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/atlas_dbt/dbt_packages/dbt_u +tils/tox.ini to gs://us-central1-atlas-dev-74134e98-bucket/data/cu +rrent/dbt/atlas_dbt/dbt_packages/dbt_utils/tox.ini +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dbt/profiles/.user.yml to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/dbt/profiles/.us +er.yml +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/release-manifest.json to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/release-manifest. +json +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.p0nHlZU13J/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/l +ogs/airflow/atlas-smoke-74732eee-local1784417488-run/run-summary.json#1784417714 +332059... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/l +ogs/airflow/atlas-airflow-20260718-scheduled__2026-07-18T06-00-00-00-00/run-summ +ary.json#1784417639731051... +... + +Average throughput: 286.4kiB/s +At file:///tmp/tmp.p0nHlZU13J/atlas-bundle/dags/**, worker process 84828 thread +139869034927936 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 84828 thread 139869034927936 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T001246Z-f9959cb6 +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smok +e__atlas-dev-20260719T001246Z-f9959cb6 --conf {"batch_id": "atlas-smoke-f9959cb6 +-local1784419966", "pipeline_run_id": "atlas-smoke-f9959cb6-local1784419966-run" +, "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=8dbf6543-5618-4a73-8900-f48594508d9b +Use ctrl-c to interrupt the command +[2026-07-19T00:13:39.953858Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +| | | data_interval_star | + | | last_scheduling_d | | | | +| triggering_user_nam +conf | dag_id | dag_run_id | t +| data_interval_end | end_date | ecision | logical_date | run_type | s +tart_date | state | e +===================+===================+===================+==================== ++===================+==========+===================+==============+==========+== +==========+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None +| None | None | None | None | manual | N +one | queued | airflow +'atlas-smoke-f9959 | ine | 20260719T001246Z- | +| | | | | | + | | +cb6-local178441996 | | f9959cb6 | +| | | | | | + | | +6', | | | +| | | | | | + | | +'pipeline_run_id': | | | +| | | | | | + | | +'atlas-smoke-f9959 | | | +| | | | | | + | | +cb6-local178441996 | | | +| | | | | | + | | +6-run', | | | +| | | | | | + | | +'processing_date': | | | +| | | | | | + | | +'2026-07-19'} | | | +| | | | | | + | | +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ^C +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha 37d4e6aa533fcdbb25141f532f16018f49423f12 --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: 37d4e6aa533fcdbb25141f532f16018f49423f12 → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260719T004112Z-37d4e6aa +smoke batch: atlas-smoke-37d4e6aa-local1784421672 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/37d +4e6aa533fcdbb25141f532f16018f49423f12 +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/37d4e6aa533f +cdbb25141f532f16018f49423f12/atlas-bundle.tar.gz to file:///tmp/tmp.UmhKc0ASax/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/37d4e6aa533f +cdbb25141f532f16018f49423f12/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.UmhKc +0ASax/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: df08882643558f132384de2645f4327dea95c8d69be7366c35217 +66b5b415829 +verified 314 file checksums for 37d4e6aa533f +audit: atlas-dev-20260719T004112Z-37d4e6aa -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.UmhKc0ASax/atlas-bundle/**, worker process 89476 thread 14015 +8593193792 listed 316... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 89476 thread 140158593193792 listed 320... + +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-f9959cb6-local1784419966/events.jsonl#1784420054987745... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-f9959cb6-local1784419966/manifest.json#1784420055949508... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-f9959cb6-local1784419966/success.marker#1784420224676583... +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.UmhKc0ASax/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/l +ogs/airflow/atlas-smoke-f9959cb6-local1784419966-run/run-summary.json#1784420233 +573658... +... + +Average throughput: 122.7kiB/s +At file:///tmp/tmp.UmhKc0ASax/atlas-bundle/dags/**, worker process 89679 thread +140039356651328 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 89679 thread 140039356651328 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smok +e__atlas-dev-20260719T004112Z-37d4e6aa --conf {"batch_id": "atlas-smoke-37d4e6aa +-local1784421672", "pipeline_run_id": "atlas-smoke-37d4e6aa-local1784421672-run" +, "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=5b256da6-4ada-4c6e-9147-d0a330da5657 +Use ctrl-c to interrupt the command +[2026-07-19T00:42:04.495033Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +| | | data_interval_star | + | | last_scheduling_d | | | | +| triggering_user_nam +conf | dag_id | dag_run_id | t +| data_interval_end | end_date | ecision | logical_date | run_type | s +tart_date | state | e +===================+===================+===================+==================== ++===================+==========+===================+==============+==========+== +==========+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None +| None | None | None | None | manual | N +one | queued | airflow +'atlas-smoke-37d4e | ine | 20260719T004112Z- | +| | | | | | + | | +6aa-local178442167 | | 37d4e6aa | +| | | | | | + | | +2', | | | +| | | | | | + | | +'pipeline_run_id': | | | +| | | | | | + | | +'atlas-smoke-37d4e | | | +| | | | | | + | | +6aa-local178442167 | | | +| | | | | | + | | +2-run', | | | +| | | | | | + | | +'processing_date': | | | +| | | | | | + | | +'2026-07-19'} | | | +| | | | | | + | | +smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: state=running (65s elapsed +) +smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: state=running (105s elapse +d) +smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: state=running (146s elapse +d) +smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: state=running (187s elapse +d) +smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: state=running (227s elapse +d) +smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: state=success (268s elapse +d) +Smoke run smoke__atlas-dev-20260719T004112Z-37d4e6aa: success +[PASS] dag_imported: atlas_batch_pipeline present in Composer +[PASS] dag_import_errors: no import errors for project_atlas +[FAIL] deployed_sha: current runtime manifest git_sha=f9959cb66729442a06501c372c +fb39e5024f4d9f +[FAIL] airflow_terminal_success: smoke dag run state=see pipeline_runs +[PASS] raw_batch_count: raw rows for atlas-smoke-37d4e6aa-local1784421672: 50000 + (expected 50000) +[PASS] no_duplicate_load: distinct ingestion runs for batch: 1 +[PASS] gcs_object_exists: raw JSONL object present in gs://atlas-raw-events-vita +l-scout-479118-n7 +[PASS] batch_manifest_exists: batch manifest/artifacts present in Composer data +path +[PASS] success_marker: success.marker present for atlas-smoke-37d4e6aa-local1784 +421672 +[PASS] warehouse_reconciliation: batch-scoped raw/classified/fact/mart reconcili +ation +[PASS] pipeline_runs_success: pipeline_runs status=SUCCESS +[PASS] deployments_row: atlas_ops.deployments rows for atlas-dev-20260719T004112 +Z-37d4e6aa: 1 + +SMOKE VALIDATION FAILED: 2 check(s) failed +STAGE FAILED: smoke_validation — post-deployment smoke validation failed +audit: atlas-dev-20260719T004112Z-37d4e6aa -> FAILED (stage smoke_validation) +Recovery: inspect logs above, then re-run this script with the same + --git-sha 37d4e6aa533fcdbb25141f532f16018f49423f12 (deployment records are ide +mpotent per deployment_id) +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha 640cd78694a90275866bebbaf7550ee121fff79b --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: 640cd78694a90275866bebbaf7550ee121fff79b → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260719T005308Z-640cd786 +smoke batch: atlas-smoke-640cd786-local1784422388 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/640 +cd78694a90275866bebbaf7550ee121fff79b +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/640cd78694a9 +0275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz to file:///tmp/tmp.6CEzxjIsKt/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/640cd78694a9 +0275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.6CEzx +jIsKt/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa6993 +6c6575e6e7c +verified 314 file checksums for 640cd78694a9 +audit: atlas-dev-20260719T005308Z-640cd786 -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/**, worker process 92980 thread 14050 +9557425984 listed 316... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 92980 thread 140509557425984 listed 320... + +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-37d4e6aa-local1784421672/events.jsonl#1784421754099263... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-37d4e6aa-local1784421672/success.marker#1784421909876646... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-37d4e6aa-local1784421672/manifest.json#1784421755045203... +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/dbt/profiles/.user.yml to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/dbt/profiles/.us +er.yml +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/release-manifest.json to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/release-manifest. +json +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/l +ogs/airflow/atlas-smoke-37d4e6aa-local1784421672-run/run-summary.json#1784421921 +555964... +... + +Average throughput: 553.2kiB/s +At file:///tmp/tmp.6CEzxjIsKt/atlas-bundle/dags/**, worker process 93182 thread +140576699508544 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 93182 thread 140576699508544 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T005308Z-640cd786 +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smok +e__atlas-dev-20260719T005308Z-640cd786 --conf {"batch_id": "atlas-smoke-640cd786 +-local1784422388", "pipeline_run_id": "atlas-smoke-640cd786-local1784422388-run" +, "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=6af332f9-4860-4066-861c-4f8bcbf748ba +Use ctrl-c to interrupt the command +[2026-07-19T00:53:59.789337Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +| | | data_interval_star | + | | last_scheduling_d | | | | +| triggering_user_nam +conf | dag_id | dag_run_id | t +| data_interval_end | end_date | ecision | logical_date | run_type | s +tart_date | state | e +===================+===================+===================+==================== ++===================+==========+===================+==============+==========+== +==========+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None +| None | None | None | None | manual | N +one | queued | airflow +'atlas-smoke-640cd | ine | 20260719T005308Z- | +| | | | | | + | | +786-local178442238 | | 640cd786 | +| | | | | | + | | +8', | | | +| | | | | | + | | +'pipeline_run_id': | | | +| | | | | | + | | +'atlas-smoke-640cd | | | +| | | | | | + | | +786-local178442238 | | | +| | | | | | + | | +8-run', | | | +| | | | | | + | | +'processing_date': | | | +| | | | | | + | | +'2026-07-19'} | | | +| | | | | | + | | +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (68s elapsed +) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (111s elapse +d) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (151s elapse +d) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (192s elapse +d) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (236s elapse +d) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (276s elapse +d) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=running (316s elapse +d) +smoke run smoke__atlas-dev-20260719T005308Z-640cd786: state=success (357s elapse +d) +Smoke run smoke__atlas-dev-20260719T005308Z-640cd786: success +[PASS] dag_imported: atlas_batch_pipeline present in Composer +[PASS] dag_import_errors: no import errors for project_atlas +[PASS] deployed_sha: current runtime manifest git_sha=640cd78694a90275866bebbaf7 +550ee121fff79b +[PASS] airflow_terminal_success: smoke dag run state=success +[PASS] raw_batch_count: raw rows for atlas-smoke-640cd786-local1784422388: 50000 + (expected 50000) +[PASS] no_duplicate_load: distinct ingestion runs for batch: 1 +[PASS] gcs_object_exists: raw JSONL object present in gs://atlas-raw-events-vita +l-scout-479118-n7 +[PASS] batch_manifest_exists: batch manifest/artifacts present in Composer data +path +[PASS] success_marker: success.marker present for atlas-smoke-640cd786-local1784 +422388 +[PASS] warehouse_reconciliation: batch-scoped raw/classified/fact/mart reconcili +ation +[PASS] pipeline_runs_success: pipeline_runs status=SUCCESS +[PASS] deployments_row: atlas_ops.deployments rows for atlas-dev-20260719T005308 +Z-640cd786: 1 + +SMOKE VALIDATION PASSED (git_sha 640cd78694a9, batch atlas-smoke-640cd786-local1 +784422388) +DAG left paused per request. +audit: atlas-dev-20260719T005308Z-640cd786 -> SUCCESS + +=== deploy SUCCESS: 640cd78694a90275866bebbaf7550ee121fff79b === +deployment_id: atlas-dev-20260719T005308Z-640cd786 +artifact: gs://atlas-deployments-example-gcp-project/atlas/releases/ +640cd78694a90275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz +artifact checksum: aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa69936c6575e +6e7c +smoke run: atlas-smoke-640cd786-local1784422388-run +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_DEPLOY=true bash scripts/deploy_atlas +_release.sh --git-sha 1af166ea5111529c154db15057e5c85472f0ecf1 --leave-paused 2> +&1 | tee /tmp/atlas-deploy.log; echo "DEPLOY_EXIT=$?" +=== Atlas deploy: 1af166ea5111529c154db15057e5c85472f0ecf1 → atlas-dev (us-centr +al1) === +deployment_id: atlas-dev-20260719T010538Z-1af166ea +smoke batch: atlas-smoke-1af166ea-local1784423138 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/1af +166ea5111529c154db15057e5c85472f0ecf1 +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/1af166ea5111 +529c154db15057e5c85472f0ecf1/atlas-bundle.tar.gz to file:///tmp/tmp.NLmcNE3DIA/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/1af166ea5111 +529c154db15057e5c85472f0ecf1/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.NLmcN +E3DIA/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: 02b6d6a1d915d7701941843fcdbb3f2aea0ade54e56468cd5a6ab +19c2e994d31 +verified 314 file checksums for 1af166ea5111 +audit: atlas-dev-20260719T010538Z-1af166ea -> RUNNING +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/**, worker process 95726 thread 13987 +8035486528 listed 316... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 95726 thread 139878035486528 listed 320... + +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-640cd786-local1784422388/events.jsonl#1784422465870174... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-640cd786-local1784422388/manifest.json#1784422466775214... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-640cd786-local1784422388/success.marker#1784422694937179... +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/dbt/atlas_dbt/dbt_project.yml to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas +_dbt/dbt_project.yml +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/dbt/profiles/.user.yml to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/dbt/profiles/.us +er.yml +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/release-manifest.json to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/release-manifest. +json +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/__pycache__/__init__.c +python-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/ +current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/config/__pycache__/__i +nit__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/config/__pycache__/set +tings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project +-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/__init +__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-at +las/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/audit. +cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas +/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/migrat +ions.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project- +atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/resour +ces.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-a +tlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/l +ogs/airflow/atlas-smoke-640cd786-local1784422388-run/run-summary.json#1784422705 +287753... +... + +Average throughput: 331.2kiB/s +At file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/dags/**, worker process 95930 thread +140206083966784 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 95930 thread 140206083966784 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T010538Z-1af166ea +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smok +e__atlas-dev-20260719T010538Z-1af166ea --conf {"batch_id": "atlas-smoke-1af166ea +-local1784423138", "pipeline_run_id": "atlas-smoke-1af166ea-local1784423138-run" +, "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=2a71b2fe-1e78-4db9-b124-e5c30be8194a +Use ctrl-c to interrupt the command +[2026-07-19T01:06:30.191349Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +| | | data_interval_star | + | | last_scheduling_d | | | | +| triggering_user_nam +conf | dag_id | dag_run_id | t +| data_interval_end | end_date | ecision | logical_date | run_type | s +tart_date | state | e +===================+===================+===================+==================== ++===================+==========+===================+==============+==========+== +==========+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None +| None | None | None | None | manual | N +one | queued | airflow +'atlas-smoke-1af16 | ine | 20260719T010538Z- | +| | | | | | + | | +6ea-local178442313 | | 1af166ea | +| | | | | | + | | +8', | | | +| | | | | | + | | +'pipeline_run_id': | | | +| | | | | | + | | +'atlas-smoke-1af16 | | | +| | | | | | + | | +6ea-local178442313 | | | +| | | | | | + | | +8-run', | | | +| | | | | | + | | +'processing_date': | | | +| | | | | | + | | +'2026-07-19'} | | | +| | | | | | + | | +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (65s elapsed +) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (106s elapse +d) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (148s elapse +d) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (189s elapse +d) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (229s elapse +d) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (272s elapse +d) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=failed (313s elapsed +) +Smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: FAILED +Executing the command: [ airflow tasks states-for-dag-run atlas_batch_pipeline s +moke__atlas-dev-20260719T010538Z-1af166ea ]... +Command has been started. execution_id=cff1ba65-63a7-4c19-9695-1973514fbdac +Use ctrl-c to interrupt the command +[2026-07-19T01:10:58.878633Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +dag_id | logical_date | task_id | state | + start_date | end_date +=====================+==============+========================+=================+ +==================================+================================= +atlas_batch_pipeline | | publish_success_marker | upstream_failed | + 2026-07-19T01:10:38.130233+00:00 | 2026-07-19T01:10:38.130233+00:00 +atlas_batch_pipeline | | dbt_seed | success | + 2026-07-19T01:07:27.799989+00:00 | 2026-07-19T01:07:49.171495+00:00 +atlas_batch_pipeline | | dbt_source_freshness | success | + 2026-07-19T01:07:50.469746+00:00 | 2026-07-19T01:08:28.129518+00:00 +atlas_batch_pipeline | | dbt_build | failed | + 2026-07-19T01:08:30.214555+00:00 | 2026-07-19T01:10:36.416826+00:00 +atlas_batch_pipeline | | validate_warehouse | upstream_failed | + 2026-07-19T01:10:37.588312+00:00 | 2026-07-19T01:10:37.588312+00:00 +atlas_batch_pipeline | | start_run_audit | success | + 2026-07-19T01:06:38.132673+00:00 | 2026-07-19T01:06:43.544435+00:00 +atlas_batch_pipeline | | generate_events | success | + 2026-07-19T01:06:51.020568+00:00 | 2026-07-19T01:06:56.948948+00:00 +atlas_batch_pipeline | | load_bigquery_raw | success | + 2026-07-19T01:07:02.293732+00:00 | 2026-07-19T01:07:14.083256+00:00 +atlas_batch_pipeline | | validate_raw_load | success | + 2026-07-19T01:07:14.325401+00:00 | 2026-07-19T01:07:27.061595+00:00 +atlas_batch_pipeline | | resolve_run_context | success | + 2026-07-19T01:06:31.351457+00:00 | 2026-07-19T01:06:32.134522+00:00 +atlas_batch_pipeline | | ensure_audit_resources | success | + 2026-07-19T01:06:32.795892+00:00 | 2026-07-19T01:06:37.716325+00:00 +atlas_batch_pipeline | | write_run_summary | failed | + 2026-07-19T01:10:39.417879+00:00 | 2026-07-19T01:10:47.178737+00:00 +atlas_batch_pipeline | | preflight_environment | success | + 2026-07-19T01:06:44.560709+00:00 | 2026-07-19T01:06:49.790375+00:00 +atlas_batch_pipeline | | upload_events | success | + 2026-07-19T01:06:57.246043+00:00 | 2026-07-19T01:07:01.576806+00:00 +STAGE FAILED: smoke_batch — smoke run did not reach terminal SUCCESS +audit: atlas-dev-20260719T010538Z-1af166ea -> FAILED (stage smoke_batch) +Recovery: inspect logs above, then re-run this script with the same + --git-sha 1af166ea5111529c154db15057e5c85472f0ecf1 (deployment records are ide +mpotent per deployment_id) +DEPLOY_EXIT=0 +(atlas-venv) project-atlas $ ATLAS_APPROVE_ROLLBACK_TEST=true ATLAS_APPROVE_DEPL +OY=true bash scripts/rollback_atlas.sh 2>&1 | tee /tmp/atlas-rollback.log; echo +"ROLLBACK_EXIT=$?" +currently deployed: 1af166ea5111529c154db15057e5c85472f0ecf1 +rollback target: 640cd78694a90275866bebbaf7550ee121fff79b +=== Atlas rollback: 640cd78694a90275866bebbaf7550ee121fff79b → atlas-dev (us-cen +tral1) === +deployment_id: atlas-dev-20260719T011614Z-640cd786 +smoke batch: atlas-smoke-640cd786-local1784423774 +Fetching release gs://atlas-deployments-example-gcp-project/atlas/releases/640 +cd78694a90275866bebbaf7550ee121fff79b +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/640cd78694a9 +0275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz to file:///tmp/tmp.yM2rEHjMJ7/a +tlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-example-gcp-project/atlas/releases/640cd78694a9 +0275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.yM2rE +HjMJ7/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa6993 +6c6575e6e7c +verified 314 file checksums for 640cd78694a9 +audit: atlas-dev-20260719T011614Z-640cd786 -> ROLLING_BACK +schema check: required 003_create_deployments_table — all release migrations app +lied +COMPATIBLE +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + da +ta/current) +At file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/**, worker process 97160 thread 14038 +4001984320 listed 309... +At gs://us-central1-atlas-dev-74134e98-bucket/data/current/**, wor +ker process 97160 thread 140384001984320 listed 319... + +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-1af166ea-local1784423138/events.jsonl#1784423215974973... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/d +ata/runs/atlas-smoke-1af166ea-local1784423138/manifest.json#1784423216852163... +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/dbt/atlas_dbt/dbt_project.yml to + gs://us-central1-atlas-dev-74134e98-bucket/data/current/dbt/atlas +_dbt/dbt_project.yml +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/dbt/profiles/.user.yml to gs://u +s-central1-atlas-dev-74134e98-bucket/data/current/dbt/profiles/.us +er.yml +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/deployment-info.json to gs://us- +central1-atlas-dev-74134e98-bucket/data/current/deployment-info.js +on +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/release-manifest.json to gs://us +-central1-atlas-dev-74134e98-bucket/data/current/release-manifest. +json +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/l +ogs/airflow/atlas-smoke-1af166ea-local1784423138-run/run-summary.json#1784423447 +026230... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/__pycache__/__init__.cpython-312.pyc#1784423148712941... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/config/__pycache__/__init__.cpython-312.pyc#1784423148613915... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/ops/__pycache__/__init__.cpython-312.pyc#1784423148510184... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/ops/__pycache__/resources.cpython-312.pyc#1784423148771944... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/ops/__pycache__/migrations.cpython-312.pyc#1784423148771583... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/config/__pycache__/settings.cpython-312.pyc#1784423148697057... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/current/s +rc/atlas/ops/__pycache__/audit.cpython-312.pyc#1784423148772489... +.. + +Average throughput: 312.0kiB/s +At file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/dags/**, worker process 97362 thread +140483103950656 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker proc +ess 97362 thread 140483103950656 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T011614Z-640cd786 +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smok +e__atlas-dev-20260719T011614Z-640cd786 --conf {"batch_id": "atlas-smoke-640cd786 +-local1784423774", "pipeline_run_id": "atlas-smoke-640cd786-local1784423774-run" +, "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=cb746bda-bd86-4102-9dfd-cbb21b927041 +Use ctrl-c to interrupt the command +[2026-07-19T01:17:03.995900Z] {{default_celery.py:196}} WARNING - You have confi +gured a result_backend using the protocol `redis`, it is highly recommended to u +se an alternative result_backend (i.e. a database). +| | | data_interval_star | + | | last_scheduling_d | | | | +| triggering_user_nam +conf | dag_id | dag_run_id | t +| data_interval_end | end_date | ecision | logical_date | run_type | s +tart_date | state | e +===================+===================+===================+==================== ++===================+==========+===================+==============+==========+== +==========+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None +| None | None | None | None | manual | N +one | queued | airflow +'atlas-smoke-640cd | ine | 20260719T011614Z- | +| | | | | | + | | +786-local178442377 | | 640cd786 | +| | | | | | + | | +4', | | | +| | | | | | + | | +'pipeline_run_id': | | | +| | | | | | + | | +'atlas-smoke-640cd | | | +| | | | | | + | | +786-local178442377 | | | +| | | | | | + | | +4-run', | | | +| | | | | | + | | +'processing_date': | | | +| | | | | | + | | +'2026-07-19'} | | | +| | | | | | + | | +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (67s elapsed +) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (108s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (150s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (190s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (231s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (271s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (311s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (352s elapse +d) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=success (392s elapse +d) +Smoke run smoke__atlas-dev-20260719T011614Z-640cd786: success +[PASS] dag_imported: atlas_batch_pipeline present in Composer +[PASS] dag_import_errors: no import errors for project_atlas +[PASS] deployed_sha: current runtime manifest git_sha=640cd78694a90275866bebbaf7 +550ee121fff79b +[PASS] airflow_terminal_success: smoke dag run state=success +[PASS] raw_batch_count: raw rows for atlas-smoke-640cd786-local1784423774: 50000 + (expected 50000) +[PASS] no_duplicate_load: distinct ingestion runs for batch: 1 +[PASS] gcs_object_exists: raw JSONL object present in gs://atlas-raw-events-vita +l-scout-479118-n7 +[PASS] batch_manifest_exists: batch manifest/artifacts present in Composer data +path +[PASS] success_marker: success.marker present for atlas-smoke-640cd786-local1784 +423774 +[PASS] warehouse_reconciliation: batch-scoped raw/classified/fact/mart reconcili +ation +[PASS] pipeline_runs_success: pipeline_runs status=SUCCESS +[PASS] deployments_row: atlas_ops.deployments rows for atlas-dev-20260719T011614 +Z-640cd786: 1 + +SMOKE VALIDATION PASSED (git_sha 640cd78694a9, batch atlas-smoke-640cd786-local1 +784423774) +audit: atlas-dev-20260719T011614Z-640cd786 -> ROLLED_BACK + +=== rollback ROLLED_BACK: 640cd78694a90275866bebbaf7550ee121fff79b === +deployment_id: atlas-dev-20260719T011614Z-640cd786 +artifact: gs://atlas-deployments-example-gcp-project/atlas/releases/ +640cd78694a90275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz +artifact checksum: aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa69936c6575e +6e7c +smoke run: atlas-smoke-640cd786-local1784423774-run +ROLLBACK_EXIT=0 +(atlas-venv) project-atlas $ diff --git a/docs/evidence-sprint4/deploy-defective-1af166ea.log b/docs/evidence-sprint4/deploy-defective-1af166ea.log new file mode 100644 index 0000000..d4948dd --- /dev/null +++ b/docs/evidence-sprint4/deploy-defective-1af166ea.log @@ -0,0 +1,96 @@ +=== Atlas deploy: 1af166ea5111529c154db15057e5c85472f0ecf1 → atlas-dev (us-central1) === +deployment_id: atlas-dev-20260719T010538Z-1af166ea +smoke batch: atlas-smoke-1af166ea-local1784423138 +Fetching release gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/1af166ea5111529c154db15057e5c85472f0ecf1 +Copying gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/1af166ea5111529c154db15057e5c85472f0ecf1/atlas-bundle.tar.gz to file:///tmp/tmp.NLmcNE3DIA/atlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/1af166ea5111529c154db15057e5c85472f0ecf1/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.NLmcNE3DIA/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: 02b6d6a1d915d7701941843fcdbb3f2aea0ade54e56468cd5a6ab19c2e994d31 +verified 314 file checksums for 1af166ea5111 +audit: atlas-dev-20260719T010538Z-1af166ea -> RUNNING +schema check: required 003_create_deployments_table — all release migrations applied +COMPATIBLE + APPLIED 001_create_pipeline_runs_table (5fb06a83e1b3…) + APPLIED 002_sprint3_raw_batch_columns (db8b53e68ee6…) + APPLIED 003_create_deployments_table (d581c625ad1e…) +migrations applied +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + data/project-atlas/current) +At file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/**, worker process 95726 thread 139878035486528 listed 316... +At gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/**, worker process 95726 thread 139878035486528 listed 320... + +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/data/runs/atlas-smoke-640cd786-local1784422388/events.jsonl#1784422465870174... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/data/runs/atlas-smoke-640cd786-local1784422388/manifest.json#1784422466775214... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/data/runs/atlas-smoke-640cd786-local1784422388/success.marker#1784422694937179... +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/dbt/atlas_dbt/dbt_project.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/dbt/atlas_dbt/dbt_project.yml +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/dbt/profiles/.user.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/dbt/profiles/.user.yml +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/deployment-info.json to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/deployment-info.json +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/release-manifest.json to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/release-manifest.json +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/__pycache__/__init__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/config/__pycache__/__init__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/config/__pycache__/settings.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/__init__.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/audit.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/migrations.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc +Copying file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/src/atlas/ops/__pycache__/resources.cpython-312.pyc to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/logs/airflow/atlas-smoke-640cd786-local1784422388-run/run-summary.json#1784422705287753... +... + +Average throughput: 331.2kiB/s +At file:///tmp/tmp.NLmcNE3DIA/atlas-bundle/dags/**, worker process 95930 thread 140206083966784 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker process 95930 thread 140206083966784 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T010538Z-1af166ea +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smoke__atlas-dev-20260719T010538Z-1af166ea --conf {"batch_id": "atlas-smoke-1af166ea-local1784423138", "pipeline_run_id": "atlas-smoke-1af166ea-local1784423138-run", "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=2a71b2fe-1e78-4db9-b124-e5c30be8194a +Use ctrl-c to interrupt the command +[2026-07-19T01:06:30.191349Z] {{default_celery.py:196}} WARNING - You have configured a result_backend using the protocol `redis`, it is highly recommended to use an alternative result_backend (i.e. a database). +| | | data_interval_star | | | last_scheduling_d | | | | | triggering_user_nam +conf | dag_id | dag_run_id | t | data_interval_end | end_date | ecision | logical_date | run_type | start_date | state | e +===================+===================+===================+====================+===================+==========+===================+==============+==========+============+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None | None | None | None | None | manual | None | queued | airflow +'atlas-smoke-1af16 | ine | 20260719T010538Z- | | | | | | | | | +6ea-local178442313 | | 1af166ea | | | | | | | | | +8', | | | | | | | | | | | +'pipeline_run_id': | | | | | | | | | | | +'atlas-smoke-1af16 | | | | | | | | | | | +6ea-local178442313 | | | | | | | | | | | +8-run', | | | | | | | | | | | +'processing_date': | | | | | | | | | | | +'2026-07-19'} | | | | | | | | | | | +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (65s elapsed) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (106s elapsed) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (148s elapsed) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (189s elapsed) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (229s elapsed) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=running (272s elapsed) +smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: state=failed (313s elapsed) +Smoke run smoke__atlas-dev-20260719T010538Z-1af166ea: FAILED +Executing the command: [ airflow tasks states-for-dag-run atlas_batch_pipeline smoke__atlas-dev-20260719T010538Z-1af166ea ]... +Command has been started. execution_id=cff1ba65-63a7-4c19-9695-1973514fbdac +Use ctrl-c to interrupt the command +[2026-07-19T01:10:58.878633Z] {{default_celery.py:196}} WARNING - You have configured a result_backend using the protocol `redis`, it is highly recommended to use an alternative result_backend (i.e. a database). +dag_id | logical_date | task_id | state | start_date | end_date +=====================+==============+========================+=================+==================================+================================= +atlas_batch_pipeline | | publish_success_marker | upstream_failed | 2026-07-19T01:10:38.130233+00:00 | 2026-07-19T01:10:38.130233+00:00 +atlas_batch_pipeline | | dbt_seed | success | 2026-07-19T01:07:27.799989+00:00 | 2026-07-19T01:07:49.171495+00:00 +atlas_batch_pipeline | | dbt_source_freshness | success | 2026-07-19T01:07:50.469746+00:00 | 2026-07-19T01:08:28.129518+00:00 +atlas_batch_pipeline | | dbt_build | failed | 2026-07-19T01:08:30.214555+00:00 | 2026-07-19T01:10:36.416826+00:00 +atlas_batch_pipeline | | validate_warehouse | upstream_failed | 2026-07-19T01:10:37.588312+00:00 | 2026-07-19T01:10:37.588312+00:00 +atlas_batch_pipeline | | start_run_audit | success | 2026-07-19T01:06:38.132673+00:00 | 2026-07-19T01:06:43.544435+00:00 +atlas_batch_pipeline | | generate_events | success | 2026-07-19T01:06:51.020568+00:00 | 2026-07-19T01:06:56.948948+00:00 +atlas_batch_pipeline | | load_bigquery_raw | success | 2026-07-19T01:07:02.293732+00:00 | 2026-07-19T01:07:14.083256+00:00 +atlas_batch_pipeline | | validate_raw_load | success | 2026-07-19T01:07:14.325401+00:00 | 2026-07-19T01:07:27.061595+00:00 +atlas_batch_pipeline | | resolve_run_context | success | 2026-07-19T01:06:31.351457+00:00 | 2026-07-19T01:06:32.134522+00:00 +atlas_batch_pipeline | | ensure_audit_resources | success | 2026-07-19T01:06:32.795892+00:00 | 2026-07-19T01:06:37.716325+00:00 +atlas_batch_pipeline | | write_run_summary | failed | 2026-07-19T01:10:39.417879+00:00 | 2026-07-19T01:10:47.178737+00:00 +atlas_batch_pipeline | | preflight_environment | success | 2026-07-19T01:06:44.560709+00:00 | 2026-07-19T01:06:49.790375+00:00 +atlas_batch_pipeline | | upload_events | success | 2026-07-19T01:06:57.246043+00:00 | 2026-07-19T01:07:01.576806+00:00 +STAGE FAILED: smoke_batch — smoke run did not reach terminal SUCCESS +audit: atlas-dev-20260719T010538Z-1af166ea -> FAILED (stage smoke_batch) +Recovery: inspect logs above, then re-run this script with the same + --git-sha 1af166ea5111529c154db15057e5c85472f0ecf1 (deployment records are idempotent per deployment_id) diff --git a/docs/evidence-sprint4/rollback-640cd786.log b/docs/evidence-sprint4/rollback-640cd786.log new file mode 100644 index 0000000..e50d282 --- /dev/null +++ b/docs/evidence-sprint4/rollback-640cd786.log @@ -0,0 +1,92 @@ +currently deployed: 1af166ea5111529c154db15057e5c85472f0ecf1 +rollback target: 640cd78694a90275866bebbaf7550ee121fff79b +=== Atlas rollback: 640cd78694a90275866bebbaf7550ee121fff79b → atlas-dev (us-central1) === +deployment_id: atlas-dev-20260719T011614Z-640cd786 +smoke batch: atlas-smoke-640cd786-local1784423774 +Fetching release gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/640cd78694a90275866bebbaf7550ee121fff79b +Copying gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/640cd78694a90275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz to file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle.tar.gz + +. +Copying gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/640cd78694a90275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz.sha256 to file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle.tar.gz.sha256 + +. +archive checksum verified: aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa69936c6575e6e7c +verified 314 file checksums for 640cd78694a9 +audit: atlas-dev-20260719T011614Z-640cd786 -> ROLLING_BACK +schema check: required 003_create_deployments_table — all release migrations applied +COMPATIBLE +Promoting to gs://us-central1-atlas-dev-74134e98-bucket (dags/project_atlas + data/project-atlas/current) +At file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/**, worker process 97160 thread 140384001984320 listed 309... +At gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/**, worker process 97160 thread 140384001984320 listed 319... + +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/data/runs/atlas-smoke-1af166ea-local1784423138/events.jsonl#1784423215974973... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/data/runs/atlas-smoke-1af166ea-local1784423138/manifest.json#1784423216852163... +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/dbt/atlas_dbt/dbt_project.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/dbt/atlas_dbt/dbt_project.yml +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/dbt/profiles/.user.yml to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/dbt/profiles/.user.yml +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/deployment-info.json to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/deployment-info.json +Copying file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/release-manifest.json to gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/release-manifest.json +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/logs/airflow/atlas-smoke-1af166ea-local1784423138-run/run-summary.json#1784423447026230... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/__pycache__/__init__.cpython-312.pyc#1784423148712941... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/config/__pycache__/__init__.cpython-312.pyc#1784423148613915... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/__init__.cpython-312.pyc#1784423148510184... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/resources.cpython-312.pyc#1784423148771944... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/migrations.cpython-312.pyc#1784423148771583... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/config/__pycache__/settings.cpython-312.pyc#1784423148697057... +Removing gs://us-central1-atlas-dev-74134e98-bucket/data/project-atlas/current/src/atlas/ops/__pycache__/audit.cpython-312.pyc#1784423148772489... +.. + +Average throughput: 312.0kiB/s +At file:///tmp/tmp.yM2rEHjMJ7/atlas-bundle/dags/**, worker process 97362 thread 140483103950656 listed 7... +At gs://us-central1-atlas-dev-74134e98-bucket/dags/project_atlas/**, worker process 97362 thread 140483103950656 listed 7... + + +atlas_batch_pipeline parsed with no import errors +Triggering smoke run smoke__atlas-dev-20260719T011614Z-640cd786 +Executing the command: [ airflow dags trigger atlas_batch_pipeline --run-id smoke__atlas-dev-20260719T011614Z-640cd786 --conf {"batch_id": "atlas-smoke-640cd786-local1784423774", "pipeline_run_id": "atlas-smoke-640cd786-local1784423774-run", "processing_date": "2026-07-19"} ]... +Command has been started. execution_id=cb746bda-bd86-4102-9dfd-cbb21b927041 +Use ctrl-c to interrupt the command +[2026-07-19T01:17:03.995900Z] {{default_celery.py:196}} WARNING - You have configured a result_backend using the protocol `redis`, it is highly recommended to use an alternative result_backend (i.e. a database). +| | | data_interval_star | | | last_scheduling_d | | | | | triggering_user_nam +conf | dag_id | dag_run_id | t | data_interval_end | end_date | ecision | logical_date | run_type | start_date | state | e +===================+===================+===================+====================+===================+==========+===================+==============+==========+============+========+==================== +{'batch_id': | atlas_batch_pipel | smoke__atlas-dev- | None | None | None | None | None | manual | None | queued | airflow +'atlas-smoke-640cd | ine | 20260719T011614Z- | | | | | | | | | +786-local178442377 | | 640cd786 | | | | | | | | | +4', | | | | | | | | | | | +'pipeline_run_id': | | | | | | | | | | | +'atlas-smoke-640cd | | | | | | | | | | | +786-local178442377 | | | | | | | | | | | +4-run', | | | | | | | | | | | +'processing_date': | | | | | | | | | | | +'2026-07-19'} | | | | | | | | | | | +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (67s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (108s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (150s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (190s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (231s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (271s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (311s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=running (352s elapsed) +smoke run smoke__atlas-dev-20260719T011614Z-640cd786: state=success (392s elapsed) +Smoke run smoke__atlas-dev-20260719T011614Z-640cd786: success +[PASS] dag_imported: atlas_batch_pipeline present in Composer +[PASS] dag_import_errors: no import errors for project_atlas +[PASS] deployed_sha: current runtime manifest git_sha=640cd78694a90275866bebbaf7550ee121fff79b +[PASS] airflow_terminal_success: smoke dag run state=success +[PASS] raw_batch_count: raw rows for atlas-smoke-640cd786-local1784423774: 50000 (expected 50000) +[PASS] no_duplicate_load: distinct ingestion runs for batch: 1 +[PASS] gcs_object_exists: raw JSONL object present in gs://atlas-raw-events-vital-scout-479118-n7 +[PASS] batch_manifest_exists: batch manifest/artifacts present in Composer data path +[PASS] success_marker: success.marker present for atlas-smoke-640cd786-local1784423774 +[PASS] warehouse_reconciliation: batch-scoped raw/classified/fact/mart reconciliation +[PASS] pipeline_runs_success: pipeline_runs status=SUCCESS +[PASS] deployments_row: atlas_ops.deployments rows for atlas-dev-20260719T011614Z-640cd786: 1 + +SMOKE VALIDATION PASSED (git_sha 640cd78694a9, batch atlas-smoke-640cd786-local1784423774) +audit: atlas-dev-20260719T011614Z-640cd786 -> ROLLED_BACK + +=== rollback ROLLED_BACK: 640cd78694a90275866bebbaf7550ee121fff79b === +deployment_id: atlas-dev-20260719T011614Z-640cd786 +artifact: gs://atlas-deployments-vital-scout-479118-n7/atlas/releases/640cd78694a90275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz +artifact checksum: aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa69936c6575e6e7c +smoke run: atlas-smoke-640cd786-local1784423774-run diff --git a/docs/evidence-sprint5/dashboard-live.json b/docs/evidence-sprint5/dashboard-live.json new file mode 100644 index 0000000..796c4d2 --- /dev/null +++ b/docs/evidence-sprint5/dashboard-live.json @@ -0,0 +1,7 @@ +[ + { + "name": "projects/123456789012/dashboards/a4f0a238-90b5-445b-925e-d0922d343c2b", + "displayName": "Atlas Operations", + "tiles": 31 + } +] diff --git a/docs/evidence-sprint5/drillb-correlated-logs.json b/docs/evidence-sprint5/drillb-correlated-logs.json new file mode 100644 index 0000000..b16f8e2 --- /dev/null +++ b/docs/evidence-sprint5/drillb-correlated-logs.json @@ -0,0 +1,634 @@ +[ + { + "insertId": "aes3sf2y67kw", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 150026, + "error_message": "command exited 1", + "error_type": "CalledProcessError", + "event_type": "task_failed", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "ERROR", + "task_id": "dbt_build", + "timestamp": "2026-07-19T06:41:32.717240+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:41:32.735181040Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "ERROR", + "timestamp": "2026-07-19T06:41:32.735181040Z" + }, + { + "insertId": "r7cr6bf35lpne", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "dbt_build", + "timestamp": "2026-07-19T06:39:10.655729+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:39:11.892974205Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:39:11.892974205Z" + }, + { + "insertId": "ilejuaf23ikuk", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 54976, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "dbt_source_freshness", + "timestamp": "2026-07-19T06:38:56.774361+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:38:56.796931708Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:38:56.796931708Z" + }, + { + "insertId": "18oaz2vf6xjxi6", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "dbt_source_freshness", + "timestamp": "2026-07-19T06:38:09.492402+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:38:10.901745123Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:38:10.901745123Z" + }, + { + "insertId": "rrabvlf11t5on", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 76736, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "dbt_seed", + "timestamp": "2026-07-19T06:37:53.816226+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:37:53.854257926Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:37:53.854257926Z" + }, + { + "insertId": "1ulkhs2f2z28y0", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "dbt_seed", + "timestamp": "2026-07-19T06:36:43.604308+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:36:45.844326230Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:36:45.844326230Z" + }, + { + "insertId": "1m1ams4f7ejn8s", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 30929, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "validate_raw_load", + "timestamp": "2026-07-19T06:36:30.038457+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:36:30.062467334Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:36:30.062467334Z" + }, + { + "insertId": "6zeuateo8njn", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "validate_raw_load", + "timestamp": "2026-07-19T06:36:08.519070+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:36:10.106687673Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:36:10.106687673Z" + }, + { + "insertId": "e6goqbf22fxq9", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 31945, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "load_bigquery_raw", + "timestamp": "2026-07-19T06:35:53.054949+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:35:53.071631448Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:35:53.071631448Z" + }, + { + "insertId": "1tixb3mf4k7jn6", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "load_bigquery_raw", + "timestamp": "2026-07-19T06:35:29.487226+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:35:31.124811332Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:35:31.124811332Z" + }, + { + "insertId": "12bm1mkf7834ho", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 21744, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "upload_events", + "timestamp": "2026-07-19T06:35:14.741166+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:35:14.763282598Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:35:14.763282598Z" + }, + { + "insertId": "1o8yhz8f13dck7", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "upload_events", + "timestamp": "2026-07-19T06:35:00.368234+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:35:01.836079292Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:35:01.836079292Z" + }, + { + "insertId": "96qwy9f15hmt2", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 30769, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "generate_events", + "timestamp": "2026-07-19T06:34:45.284663+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:34:45.313173618Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:34:45.313173618Z" + }, + { + "insertId": "5auvx8f7gei3u", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "generate_events", + "timestamp": "2026-07-19T06:34:20.522072+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:34:22.074540624Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:34:22.074540624Z" + }, + { + "insertId": "1uvat7uf7jp2pj", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 7565, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "preflight_environment", + "timestamp": "2026-07-19T06:34:10.602897+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:34:10.627082794Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:34:10.627082794Z" + }, + { + "insertId": "1s4sy7if4fh40o", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "preflight_environment", + "timestamp": "2026-07-19T06:34:05.838571+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:34:06.440739008Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:34:06.440739008Z" + }, + { + "insertId": "flnxlsf33v462", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 7340, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "start_run_audit", + "timestamp": "2026-07-19T06:33:58.529205+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:33:58.547407204Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:33:58.547407204Z" + }, + { + "insertId": "hroq3of24zbo0", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "start_run_audit", + "timestamp": "2026-07-19T06:33:54.102177+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:33:54.648338018Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:33:54.648338018Z" + }, + { + "insertId": "1yt5aaqf3lih46", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "duration_ms": 9148, + "event_type": "task_success", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "ensure_audit_resources", + "timestamp": "2026-07-19T06:33:47.012853+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:33:47.033404862Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:33:47.033404862Z" + }, + { + "insertId": "dc6gk2f7s5pid", + "jsonPayload": { + "airflow_run_id": "drillb__pipeline-failure-20260719", + "atlas_event": true, + "attempt_number": 1, + "batch_id": "atlas-drillb-20260719", + "component": "step_runner", + "dag_id": "atlas_batch_pipeline", + "event_type": "task_started", + "pipeline_run_id": "atlas-drillb-20260719-run", + "severity": "INFO", + "task_id": "ensure_audit_resources", + "timestamp": "2026-07-19T06:33:43.149515+00:00" + }, + "logName": "projects/example-gcp-project/logs/atlas-events", + "receiveTimestamp": "2026-07-19T06:33:43.851985070Z", + "resource": { + "labels": { + "cluster_name": "us-central1-atlas-dev-72339da4-gke", + "container_name": "", + "location": "us-central1", + "namespace_name": "", + "pod_name": "", + "project_id": "od2986b2a125c4b19p-tp" + }, + "type": "k8s_container" + }, + "severity": "INFO", + "timestamp": "2026-07-19T06:33:43.851985070Z" + } +] diff --git a/docs/evidence-sprint5/incident-events.json b/docs/evidence-sprint5/incident-events.json new file mode 100644 index 0000000..7e7462e --- /dev/null +++ b/docs/evidence-sprint5/incident-events.json @@ -0,0 +1,255 @@ +[ + { + "insertId": "1w9zttsf8hrsg8", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: telemetry incomplete", + "policy_id": "2189602217894168208", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784444831", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf545p85441" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T07:07:11.396575117Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T07:07:11Z" + }, + { + "insertId": "1brjfdif4czbbe", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: BigQuery cost anomaly", + "policy_id": "10784657038527994875", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784444809", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=cost_anomaly, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=cost_anomaly, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf53uuk0td1" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T07:06:49.432523914Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T07:06:49Z" + }, + { + "insertId": "wrxol5f6hv3ac", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: breaking schema drift", + "policy_id": "5106246397808706358", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784444779", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=schema_drift, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=schema_drift, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf53g1toeh7" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T07:06:19.546724441Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T07:06:19Z" + }, + { + "insertId": "iw6z14f4dzlj7", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: critical volume deviation", + "policy_id": "2189602217894170243", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784444723", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=volume_deviation, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=volume_deviation, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf52ofe3mrz" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T07:05:23.312195165Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T07:05:23Z" + }, + { + "insertId": "1ke2mttf47b0sa", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: data stale", + "policy_id": "144332609457282610", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784444694", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=freshness, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=freshness, environment=atlas-dev, mode=drill} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf52a4f18hp" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T07:04:54.615245278Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T07:04:54Z" + }, + { + "insertId": "1gr7ndaf41d5co", + "labels": { + "activity_type_name": "ViolationAutoResolveEventv1", + "policy_display_name": "Atlas: pipeline failed", + "policy_id": "14992081806484518993", + "resolved_at": "1784444591", + "resource_id": "", + "resource_name": "example-gcp-project", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=latest_run_state, environment=atlas-dev, mode=normal} returned to normal with a value of 0.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=latest_run_state, environment=atlas-dev, mode=normal} returned to normal with a value of 0.000.", + "violation_id": "0.oaf4n04jvxx1" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationAutoResolveEventv1", + "receiveTimestamp": "2026-07-19T07:03:11.104543030Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T07:03:11Z" + }, + { + "insertId": "17blt8zf4r6g11", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: pipeline failed", + "policy_id": "14992081806484518993", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784443579", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=latest_run_state, environment=atlas-dev, mode=normal} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=latest_run_state, environment=atlas-dev, mode=normal} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf4n04jvxx1" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T06:46:20.045133977Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T06:46:19Z" + }, + { + "insertId": "1vgm5kyf1e9ues", + "labels": { + "activity_type_name": "ViolationAutoResolveEventv1", + "policy_display_name": "Atlas: telemetry incomplete", + "policy_id": "2189602217894168208", + "resolved_at": "1784442780", + "resource_id": "", + "resource_name": "example-gcp-project", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} returned to normal with a value of 0.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} returned to normal with a value of 0.000.", + "violation_id": "0.oaf3pqgsimvt" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationAutoResolveEventv1", + "receiveTimestamp": "2026-07-19T06:33:00.420640623Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T06:33:00Z" + }, + { + "insertId": "18ymujkf4nme8t", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: telemetry incomplete", + "policy_id": "2189602217894168208", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784441151", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf3pqgsimvt" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T06:05:51.653376095Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T06:05:51Z" + }, + { + "insertId": "oh1dq5fg1m9p5", + "labels": { + "activity_type_name": "ViolationAutoResolveEventv1", + "policy_display_name": "Atlas: telemetry incomplete", + "policy_id": "2189602217894168208", + "resolved_at": "1784434151", + "resource_id": "", + "resource_name": "example-gcp-project", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} returned to normal with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} returned to normal with a value of 2.000.", + "violation_id": "0.oaf09qv4gi7i" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationAutoResolveEventv1", + "receiveTimestamp": "2026-07-19T04:09:11.957463838Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T04:09:11Z" + }, + { + "insertId": "5buyatf2w5xus", + "labels": { + "activity_type_name": "ViolationOpenEventv1", + "policy_display_name": "Atlas: telemetry incomplete", + "policy_id": "2189602217894168208", + "resource_id": "", + "resource_name": "example-gcp-project", + "started_at": "1784432102", + "terse_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} is above the threshold of 1.500 with a value of 2.000.", + "verbose_message": "custom/atlas/monitor/check_status for example-gcp-project with metric labels {check_name=telemetry_completeness, environment=atlas-dev, mode=normal} is above the threshold of 1.500 with a value of 2.000.", + "violation_id": "0.oaf09qv4gi7i" + }, + "logName": "projects/example-gcp-project/logs/monitoring.googleapis.com%2FViolationOpenEventv1", + "receiveTimestamp": "2026-07-19T03:35:03.028598377Z", + "resource": { + "labels": { + "project_id": "example-gcp-project" + }, + "type": "global" + }, + "timestamp": "2026-07-19T03:35:02Z" + } +] diff --git a/docs/evidence-sprint5/log-routing-live.json b/docs/evidence-sprint5/log-routing-live.json new file mode 100644 index 0000000..7115fc3 --- /dev/null +++ b/docs/evidence-sprint5/log-routing-live.json @@ -0,0 +1,32 @@ +{ + "createTime": "2026-07-19T03:06:20.814459227Z", + "description": "Routes Atlas runtime logs to the atlas-observability bucket (additive; _Default unaffected)", + "destination": "logging.googleapis.com/projects/example-gcp-project/locations/us-central1/buckets/atlas-observability", + "filter": "resource.type=\"cloud_composer_environment\" OR jsonPayload.atlas_event=true", + "name": "atlas-observability-sink", + "resourceName": "projects/example-gcp-project/sinks/atlas-observability-sink", + "updateTime": "2026-07-19T03:06:20.814459227Z" +} +{ + "analyticsEnabled": true, + "createTime": "2026-07-19T03:04:34.700919244Z", + "description": "Atlas runtime logs (Sprint 5). Composer + structured Atlas events.", + "lifecycleState": "ACTIVE", + "name": "projects/example-gcp-project/locations/us-central1/buckets/atlas-observability", + "retentionDays": 30, + "updateTime": "2026-07-19T03:06:18.593776549Z" +} +[ + { + "description": "Access to all logs", + "name": "projects/example-gcp-project/locations/us-central1/buckets/atlas-observability/views/_AllLogs" + }, + { + "createTime": "2026-07-19T03:06:22.500860717Z", + "description": "Least-privilege Atlas runtime view (grant roles/logging.viewAccessor here)", + "filter": "SOURCE(\"projects/example-gcp-project\")", + "name": "projects/example-gcp-project/locations/us-central1/buckets/atlas-observability/views/atlas-runtime", + "updateTime": "2026-07-19T03:06:22.500860717Z" + } +] +{"dataset": "example-gcp-project:atlas_logs", "type": "LINKED", "linked": true} diff --git a/docs/evidence-sprint5/metric-timeseries-summary.json b/docs/evidence-sprint5/metric-timeseries-summary.json new file mode 100644 index 0000000..7230a20 --- /dev/null +++ b/docs/evidence-sprint5/metric-timeseries-summary.json @@ -0,0 +1,127 @@ +{ + "custom.googleapis.com/atlas/monitor/check_status": { + "series": 22, + "points_4h": 103, + "latest_sample": { + "labels": { + "mode": "drill", + "environment": "atlas-dev", + "check_name": "cost_anomaly" + }, + "value": 0, + "time": "2026-07-19 07:19:14.502888+00:00" + } + }, + "custom.googleapis.com/atlas/pipeline/last_success_age_seconds": { + "series": 2, + "points_4h": 8, + "latest_sample": { + "labels": { + "dag_id": "atlas_batch_pipeline", + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 196.0, + "time": "2026-07-19 07:01:20.459986+00:00" + } + }, + "custom.googleapis.com/atlas/pipeline/telemetry_incomplete_count": { + "series": 2, + "points_4h": 8, + "latest_sample": { + "labels": { + "dag_id": "atlas_batch_pipeline", + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 0, + "time": "2026-07-19 07:01:29.478219+00:00" + } + }, + "custom.googleapis.com/atlas/data/rejection_rate": { + "series": 2, + "points_4h": 8, + "latest_sample": { + "labels": { + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 0.0179, + "time": "2026-07-19 07:01:40.094884+00:00" + } + }, + "custom.googleapis.com/atlas/data/raw_row_count": { + "series": 2, + "points_4h": 8, + "latest_sample": { + "labels": { + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 50000, + "time": "2026-07-19 07:01:35.942524+00:00" + } + }, + "custom.googleapis.com/atlas/data/volume_deviation_ratio": { + "series": 2, + "points_4h": 8, + "latest_sample": { + "labels": { + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 1.0, + "time": "2026-07-19 07:01:36.060340+00:00" + } + }, + "custom.googleapis.com/atlas/data/schema_drift_count": { + "series": 6, + "points_4h": 24, + "latest_sample": { + "labels": { + "mode": "drill", + "severity": "CRITICAL", + "environment": "atlas-dev" + }, + "value": 0, + "time": "2026-07-19 07:01:45.768376+00:00" + } + }, + "custom.googleapis.com/atlas/data/reconciliation_failure_count": { + "series": 2, + "points_4h": 7, + "latest_sample": { + "labels": { + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 0, + "time": "2026-07-19 07:01:32.537346+00:00" + } + }, + "custom.googleapis.com/atlas/deployment/latest_failed": { + "series": 2, + "points_4h": 8, + "latest_sample": { + "labels": { + "mode": "drill", + "environment": "atlas-dev" + }, + "value": 0, + "time": "2026-07-19 07:01:54.035875+00:00" + } + }, + "custom.googleapis.com/atlas/deployment/duration_seconds": { + "series": 3, + "points_4h": 8, + "latest_sample": { + "labels": { + "mode": "normal", + "environment": "atlas-dev", + "status": "ROLLED_BACK" + }, + "value": 445.0, + "time": "2026-07-19 03:28:57.499348+00:00" + } + } +} diff --git a/docs/evidence-sprint5/monitor-evaluations.json b/docs/evidence-sprint5/monitor-evaluations.json new file mode 100644 index 0000000..33d48b3 --- /dev/null +++ b/docs/evidence-sprint5/monitor-evaluations.json @@ -0,0 +1,912 @@ +[ + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "cost_anomaly-b569fa7d4de3", + "observed_value": "3.7748736E8", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": null + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "reconciliation-be3cfc6146a6", + "observed_value": null, + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "freshness-484bd1d9cecf", + "observed_value": "7592.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "missing_scheduled_run-6c0ed389d77a", + "observed_value": "7875.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "volume_deviation-c2c60e9d81bc", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "latest_run_state-b89a1e8c75ed", + "observed_value": "279.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": null + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "rejection_rate-d95da1085a8f", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "telemetry_completeness-ef04a1291575", + "observed_value": "13.0", + "severity": "WARNING", + "source": "atlas_observability_monitor", + "status": "FAIL", + "threshold": "0.0" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "rollback_failure-a099abcf7496", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "deployment_failure-25eb29c2d98d", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 03:28:25", + "evaluation_id": "schema_drift-2411b2cf49eb", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "freshness-5994d9c0110d", + "observed_value": "3877.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "schema_drift-77c559c4777a", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "rollback_failure-6059f8daf2f9", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "rejection_rate-2280661c9a80", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "deployment_failure-225afd1b5527", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "volume_deviation-235087c3dc46", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "reconciliation-1c019f38a331", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "latest_run_state-476818afb0d7", + "observed_value": "126.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "WARN", + "threshold": null + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "missing_scheduled_run-fdde3c331cc2", + "observed_value": "137.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "cost_anomaly-f698a26e268d", + "observed_value": null, + "severity": "WARNING", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 06:02:45", + "evaluation_id": "telemetry_completeness-78aefe0404fd", + "observed_value": "5.0", + "severity": "WARNING", + "source": "atlas_observability_monitor", + "status": "FAIL", + "threshold": "0.0" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "telemetry_completeness-0ea758fbfa35", + "observed_value": "5.0", + "severity": "WARNING", + "source": "atlas_observability_monitor", + "status": "FAIL", + "threshold": "0.0" + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "cost_anomaly-c6ced62ce1e1", + "observed_value": null, + "severity": "WARNING", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "missing_scheduled_run-4bf57f9d3982", + "observed_value": "198.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "rejection_rate-276a993f1baf", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "schema_drift-26cccaa98559", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "rollback_failure-ecacb53e5458", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "deployment_failure-0a48281861f2", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "volume_deviation-24d495249bac", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "reconciliation-921b01213f71", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "latest_run_state-30093a329ba3", + "observed_value": "191.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "WARN", + "threshold": null + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 06:03:50", + "evaluation_id": "freshness-c0fd03af156a", + "observed_value": "3939.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "reconciliation-6eff48a75567", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "freshness-b2606a592b09", + "observed_value": "12.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "cost_anomaly-225364eaca13", + "observed_value": null, + "severity": "WARNING", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "rollback_failure-89fde792afe3", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "deployment_failure-b90aee18ee5e", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "telemetry_completeness-fad33c26b147", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "schema_drift-67e1e1c3cc03", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "rejection_rate-9cf2639c6833", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "volume_deviation-999841f3545d", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "missing_scheduled_run-92ac80090484", + "observed_value": "606.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 06:30:07", + "evaluation_id": "latest_run_state-8362e27219e5", + "observed_value": "598.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "WARN", + "threshold": null + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "deployment_failure-041540baba22", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "cost_anomaly-692f17c0fb3c", + "observed_value": "1.7697865728E10", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "volume_deviation-3e9e199cc9e9", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "rollback_failure-f19a18f9834b", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "telemetry_completeness-93527dbdf287", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "latest_run_state-f87f42760007", + "observed_value": "472.0", + "severity": "CRITICAL", + "source": "atlas_observability_monitor", + "status": "FAIL", + "threshold": null + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "freshness-5b36a9aa386e", + "observed_value": "849.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "missing_scheduled_run-9aa5490f810b", + "observed_value": "618.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "reconciliation-264c4f9fc656", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "schema_drift-91236cd3c9bc", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 06:44:05", + "evaluation_id": "rejection_rate-9db60ab2735c", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "deployment_failure-2356e3877017", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "rollback_failure-0a04ee1735eb", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "missing_scheduled_run-e1c8f92538cb", + "observed_value": "650.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "freshness-b06ef2beb63d", + "observed_value": "127.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "volume_deviation-0326a461c26e", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "rejection_rate-9b02f2dae5e2", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "telemetry_completeness-512cc23f482a", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "latest_run_state-0bdcf5e01534", + "observed_value": "518.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": null + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "cost_anomaly-633492ee6351", + "observed_value": "2.0724056064E10", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "reconciliation-bd2d29d81634", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 07:00:03", + "evaluation_id": "schema_drift-495a69297b8c", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "schema_drift-b6bdf6f7a1c6", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "latest_run_state-9f70138a3e7f", + "observed_value": "518.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": null + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "missing_scheduled_run-8ac402d64cc1", + "observed_value": "705.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "reconciliation-25310f0d89f1", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "freshness-9510cac78f49", + "observed_value": "182.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "cost_anomaly-855604cb3d98", + "observed_value": "2.1133000704E10", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "volume_deviation-d84949cc9a47", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "deployment_failure-4a3f7c11eafb", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "rejection_rate-08ad2a5afda2", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "rollback_failure-d96c75f3af0b", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 07:00:59", + "evaluation_id": "telemetry_completeness-03946d7a56c8", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rejection_rate", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "rejection_rate-0c20dce3442a", + "observed_value": "0.0179", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.2" + }, + { + "check_name": "deployment_failure", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "deployment_failure-3b964868dfd3", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "telemetry_completeness-3220bc7f218f", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "volume_deviation-722afaa1e62d", + "observed_value": "1.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.5" + }, + { + "check_name": "latest_run_state", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "latest_run_state-b760ce112788", + "observed_value": "518.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": null + }, + { + "check_name": "freshness", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "freshness-7d192a170043", + "observed_value": "196.0", + "severity": "CRITICAL", + "source": "atlas_observability_monitor", + "status": "FAIL", + "threshold": "2.0" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "schema_drift-6cc5ca3ad325", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "reconciliation", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "reconciliation-35cf29bd44c2", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "rollback_failure", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "rollback_failure-cdf33dd446ba", + "observed_value": "0.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "0.0" + }, + { + "check_name": "cost_anomaly", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "cost_anomaly-0f534408fe05", + "observed_value": "2.1290287104E10", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "NO_DATA", + "threshold": null + }, + { + "check_name": "missing_scheduled_run", + "evaluated_at": "2026-07-19 07:01:12", + "evaluation_id": "missing_scheduled_run-005d1b2fa9ea", + "observed_value": "718.0", + "severity": "INFO", + "source": "atlas_observability_monitor", + "status": "PASS", + "threshold": "93600.0" + }, + { + "check_name": "volume_deviation", + "evaluated_at": "2026-07-19 07:02:29", + "evaluation_id": "drill-d-09023a9a9463", + "observed_value": "0.2", + "severity": "CRITICAL", + "source": "atlas_drill_fixture", + "status": "FAIL", + "threshold": "0.8" + }, + { + "check_name": "schema_drift", + "evaluated_at": "2026-07-19 07:03:24", + "evaluation_id": "drill-e-bc8914b28afa", + "observed_value": "2.0", + "severity": "CRITICAL", + "source": "atlas_drill_fixture", + "status": "FAIL", + "threshold": "0.0" + }, + { + "check_name": "telemetry_completeness", + "evaluated_at": "2026-07-19 07:04:47", + "evaluation_id": "drill-g-ed5d4f7a31b1", + "observed_value": "4.0", + "severity": "WARNING", + "source": "atlas_drill_fixture", + "status": "FAIL", + "threshold": "0.0" + } +] diff --git a/docs/evidence-sprint5/notification-channel.json b/docs/evidence-sprint5/notification-channel.json new file mode 100644 index 0000000..faedac9 --- /dev/null +++ b/docs/evidence-sprint5/notification-channel.json @@ -0,0 +1,11 @@ +[ + { + "name": "projects/example-gcp-project/notificationChannels/6567861337166986657", + "type": "email", + "display_name": "Atlas Primary Operator (email)", + "recipient_category": "project-owner email (approved by owner in-session)", + "email_domain": "gmail.com", + "verification_status": "0", + "enabled": true + } +] diff --git a/docs/evidence-sprint5/quality-results-drills.json b/docs/evidence-sprint5/quality-results-drills.json new file mode 100644 index 0000000..8debd7f --- /dev/null +++ b/docs/evidence-sprint5/quality-results-drills.json @@ -0,0 +1,162 @@ +[ + { + "check_category": "RECONCILIATION", + "check_name": "accepted_equals_fact", + "expected_value": "49105.0", + "observed_value": "49105.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "accepted_plus_rejected_equals_raw", + "expected_value": "50000.0", + "observed_value": null, + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "COMPLETENESS", + "check_name": "batch_lineage_semantics", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "COMPLETENESS", + "check_name": "batch_nonempty", + "expected_value": null, + "observed_value": "50000.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "REFERENTIAL_INTEGRITY", + "check_name": "fact_country_fk_resolves", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "UNIQUENESS", + "check_name": "fact_event_ids_unique", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "REFERENTIAL_INTEGRITY", + "check_name": "fact_user_fk_resolves", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "mart_totals_reconcile", + "expected_value": "343738.0", + "observed_value": "343738.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "COMPLETENESS", + "check_name": "processing_date_semantics", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "raw_equals_classification", + "expected_value": "50000.0", + "observed_value": "50000.0", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "accepted_equals_fact", + "expected_value": "49105.0", + "observed_value": "49105.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "accepted_plus_rejected_equals_raw", + "expected_value": "50000.0", + "observed_value": null, + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "COMPLETENESS", + "check_name": "batch_lineage_semantics", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "COMPLETENESS", + "check_name": "batch_nonempty", + "expected_value": null, + "observed_value": "50000.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "REFERENTIAL_INTEGRITY", + "check_name": "fact_country_fk_resolves", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "UNIQUENESS", + "check_name": "fact_event_ids_unique", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "REFERENTIAL_INTEGRITY", + "check_name": "fact_user_fk_resolves", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "mart_totals_reconcile", + "expected_value": "343738.0", + "observed_value": "343738.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "COMPLETENESS", + "check_name": "processing_date_semantics", + "expected_value": "0.0", + "observed_value": "0.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + }, + { + "check_category": "RECONCILIATION", + "check_name": "raw_equals_classification", + "expected_value": "50000.0", + "observed_value": "50000.0", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "PASS" + } +] diff --git a/docs/evidence-sprint5/task-events-drills.json b/docs/evidence-sprint5/task-events-drills.json new file mode 100644 index 0000000..d51f818 --- /dev/null +++ b/docs/evidence-sprint5/task-events-drills.json @@ -0,0 +1,594 @@ +[ + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "dbt_build" + }, + { + "attempt_number": "1", + "duration_ms": "175263", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "dbt_build" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "dbt_seed" + }, + { + "attempt_number": "1", + "duration_ms": "81229", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "dbt_seed" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "dbt_source_freshness" + }, + { + "attempt_number": "1", + "duration_ms": "59572", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "dbt_source_freshness" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "ensure_audit_resources" + }, + { + "attempt_number": "1", + "duration_ms": "7300", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "ensure_audit_resources" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "generate_events" + }, + { + "attempt_number": "1", + "duration_ms": "6528", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "generate_events" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "load_bigquery_raw" + }, + { + "attempt_number": "1", + "duration_ms": "26445", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "load_bigquery_raw" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "preflight_environment" + }, + { + "attempt_number": "1", + "duration_ms": "7499", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "preflight_environment" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "publish_success_marker" + }, + { + "attempt_number": "1", + "duration_ms": "8299", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "publish_success_marker" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "resolve_run_context" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "start_run_audit" + }, + { + "attempt_number": "1", + "duration_ms": "9630", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "start_run_audit" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "upload_events" + }, + { + "attempt_number": "1", + "duration_ms": "10199", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "upload_events" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "validate_raw_load" + }, + { + "attempt_number": "1", + "duration_ms": "29952", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "validate_raw_load" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": null, + "task_id": "validate_warehouse" + }, + { + "attempt_number": "1", + "duration_ms": "47876", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-recovery-run", + "status": "SUCCESS", + "task_id": "validate_warehouse" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "FAILED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "FAILED", + "task_id": "dbt_build" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "dbt_build" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "dbt_seed" + }, + { + "attempt_number": "1", + "duration_ms": "76736", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "dbt_seed" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "dbt_source_freshness" + }, + { + "attempt_number": "1", + "duration_ms": "54976", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "dbt_source_freshness" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "ensure_audit_resources" + }, + { + "attempt_number": "1", + "duration_ms": "9148", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "ensure_audit_resources" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "generate_events" + }, + { + "attempt_number": "1", + "duration_ms": "30769", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "generate_events" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "load_bigquery_raw" + }, + { + "attempt_number": "1", + "duration_ms": "31945", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "load_bigquery_raw" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "preflight_environment" + }, + { + "attempt_number": "1", + "duration_ms": "7565", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "preflight_environment" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "UPSTREAM_FAILED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "UPSTREAM_FAILED", + "task_id": "publish_success_marker" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "resolve_run_context" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "start_run_audit" + }, + { + "attempt_number": "1", + "duration_ms": "7340", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "start_run_audit" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "upload_events" + }, + { + "attempt_number": "1", + "duration_ms": "21744", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "upload_events" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": null, + "task_id": "validate_raw_load" + }, + { + "attempt_number": "1", + "duration_ms": "30929", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "SUCCESS", + "task_id": "validate_raw_load" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "UPSTREAM_FAILED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "UPSTREAM_FAILED", + "task_id": "validate_warehouse" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "FAILED", + "pipeline_run_id": "atlas-drillb-20260719-run", + "status": "FAILED", + "task_id": "write_run_summary" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "dbt_build" + }, + { + "attempt_number": "1", + "duration_ms": "155304", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "dbt_build" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "dbt_seed" + }, + { + "attempt_number": "1", + "duration_ms": "104527", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "dbt_seed" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "dbt_source_freshness" + }, + { + "attempt_number": "1", + "duration_ms": "59763", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "dbt_source_freshness" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "ensure_audit_resources" + }, + { + "attempt_number": "1", + "duration_ms": "12806", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "ensure_audit_resources" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "generate_events" + }, + { + "attempt_number": "1", + "duration_ms": "32861", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "generate_events" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "load_bigquery_raw" + }, + { + "attempt_number": "1", + "duration_ms": "34582", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "load_bigquery_raw" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "preflight_environment" + }, + { + "attempt_number": "1", + "duration_ms": "13239", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "preflight_environment" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "publish_success_marker" + }, + { + "attempt_number": "1", + "duration_ms": "12437", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "publish_success_marker" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "resolve_run_context" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "start_run_audit" + }, + { + "attempt_number": "1", + "duration_ms": "19684", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "start_run_audit" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "upload_events" + }, + { + "attempt_number": "1", + "duration_ms": "21943", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "upload_events" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "validate_raw_load" + }, + { + "attempt_number": "1", + "duration_ms": "34862", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "validate_raw_load" + }, + { + "attempt_number": "1", + "duration_ms": null, + "event_type": "STARTED", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": null, + "task_id": "validate_warehouse" + }, + { + "attempt_number": "1", + "duration_ms": "47756", + "event_type": "SUCCESS", + "pipeline_run_id": "atlas-smoke-2109310b-local1784441882-run", + "status": "SUCCESS", + "task_id": "validate_warehouse" + } +] diff --git a/docs/evidence-sprint7/cost-guard-block.txt b/docs/evidence-sprint7/cost-guard-block.txt new file mode 100644 index 0000000..3a13766 --- /dev/null +++ b/docs/evidence-sprint7/cost-guard-block.txt @@ -0,0 +1,18 @@ +Atlas Sprint 7 — cost-guard pre-spend block evidence (2026-07-19, live, dry-run only, $0) + +(1) Partition-filter guard on deliberately unbounded raw scan: +$ python -m atlas.observability.cost_guard check-partition-filter \ + --sql-file observability/performance/queries/unbounded_scan.sql --asset atlas_raw.events +COST GUARD: query over atlas_raw.events is missing a required partition filter (event_date/processing_date) +exit=2 (BLOCKED before any execution) + +(2) estimate CLI (dry-run first, then threshold enforcement): +- Full unbounded scan estimate: 12,659,283 bytes (dry-run, billed $0) +- atlas-dev ceiling 1,073,741,824 bytes -> decision ALLOW (dataset is small) +- With a tightened 1,000-byte ceiling -> decision BLOCKED pre-execution: + "estimate 12659283 bytes exceeds ceiling 1000 bytes for 'atlas-dev'; set + ATLAS_APPROVE_COST_OVERRIDE=true only after a documented cost review" + +Conclusion: an unbounded query is refused before material spend by (a) the +required-partition-filter guard and (b) the dry-run estimate ceiling. No bytes +were billed to produce this evidence. diff --git a/docs/evidence-sprint8/clean-clone-results.md b/docs/evidence-sprint8/clean-clone-results.md new file mode 100644 index 0000000..a420aa3 --- /dev/null +++ b/docs/evidence-sprint8/clean-clone-results.md @@ -0,0 +1,58 @@ +# Clean-Clone Results (Sprint 8, Phase 11) + +**Status:** RECORDED. Automated by +[`../../scripts/validate_clean_clone.sh`](../../scripts/validate_clean_clone.sh); +procedure in [../handoff/clean-clone-reproduction.md](../handoff/clean-clone-reproduction.md). +Each attempt uses a fresh temp directory and a brand-new virtualenv — no reuse of +the caller's environment, generated data, or credentials. Credentialless. + +## Attempt 1 — candidate `57abf2d` — FAIL (defect found) + +Command: `bash scripts/validate_clean_clone.sh --ref 57abf2d`. Duration 42s. + +| step | result | +| --- | --- | +| install | PASS | +| static_ci | PASS | +| generate | PASS | +| unit_tests | PASS | +| governance | **FAIL** | +| lineage | **FAIL** | +| reference | **FAIL** | + +**Root cause:** documented direct commands `python -m atlas.governance.catalog`, +`atlas.governance.lineage`, `atlas.reference.validate` failed with +`ModuleNotFoundError: No module named 'atlas'`. `atlas.*` lives under `src/` with +no installed package; `pytest.ini` and `validate_ci.sh` set `PYTHONPATH=src` +internally (so tests and static CI passed), but the standalone module commands in +the docs omitted it. + +**Fix (documentation/script root cause, commit `e538e99`):** added +`export PYTHONPATH=src` to `validate_clean_clone.sh` and to every documented +`python -m atlas.*` command (START_HERE, operator/agent onboarding, first-hour, +clean-clone doc, evidence index, demo script, context pack). + +## Attempt 2 — candidate `e538e99` — PASS (fresh directory) + +Command: `bash scripts/validate_clean_clone.sh --ref e538e99`. Duration 43s. + +| step | result | +| --- | --- | +| install | PASS | +| static_ci | PASS | +| generate | PASS | +| unit_tests | PASS | +| governance | PASS | +| lineage | PASS | +| reference | PASS | + +`clean_clone: PASS`. Success was declared only from a **second fresh directory** +after the fix — not from a repaired dirty clone. + +## Notes + +- `shell_static` and the Airflow gates SKIP in the clean venv because shellcheck + and apache-airflow are not in `requirements.txt`; they run in GitHub CI (pinned + toolchain) and via `airflow/requirements-airflow.txt`. `yamllint`/`shellcheck` + in `requirements-ci.txt` mean `workflow_yaml` runs. +- No GCP credentials were used or required. diff --git a/docs/evidence-sprint8/independent-handoff-results.md b/docs/evidence-sprint8/independent-handoff-results.md new file mode 100644 index 0000000..b109f50 --- /dev/null +++ b/docs/evidence-sprint8/independent-handoff-results.md @@ -0,0 +1,82 @@ +# Independent Handoff Results (Sprint 8, Phase 12) + +**Status:** RECORDED. The independent tester received only the repository clone, +`START_HERE.md`, and the [assignment](../handoff/independent-handoff-assignment.md). +No prior conversation, no implementation-agent reasoning, no hidden commands, no +verbal help. Rubric: [../handoff/handoff-scorecard.md](../handoff/handoff-scorecard.md). + +## Tester context + +- Identity type: independent coding agent (separate agent context, repository + + START_HERE + assignment only). +- Candidate commit at test time: `d90b3ad`. +- Human interventions: **0** (the tester self-resolved the one friction point + using the repository's own venv convention; the implementation agent provided + no answers or commands). + +## Answers (summary — all 15 items completed correctly) + +1. Atlas = reference-architecture batch ELT platform on GCP; correct/observable/ + governable/recoverable; not streaming/CDC/ML; not a template. ✓ +2. Fact grain = **one row per `event_id`** (`fct_events`, merge on `event_id`, + INV-D5, ADR-017). ✓ +3. `batch_id` = stable data identity (`atlas-`); `pipeline_run_id` = + one execution (`atlas-airflow--`); ADR-006. ✓ +4. Current release `atlas-sprint-7-complete` → `9d031c9`; Sprint 8 intentionally + untagged. (Correctly de-referenced annotated tags.) ✓ +5. `validate_ci: PASS` (19 PASS / 2 SKIP / 0 FAIL). Noted the PEP 668 install + friction (see below). ✓ (with friction) +6. Successful deploy: `docs/evidence-sprint4/composer-deploy-session-history.txt`. ✓ +7. Failed deploy: `docs/evidence-sprint4/deploy-defective-1af166ea.log` + (`publish_success_marker = upstream_failed`, proving INV-L5). ✓ +8. Recovery: `docs/evidence-sprint4/rollback-640cd786.log` (+ INC-S6-001 for the + verified data-layer recovery). ✓ +9. Schema control: compatibility classification + change records + immutable + migration checksums (ADR-017, INV-D8). ✓ +10. Unsafe query blocked: `cost_guard check-partition-filter` + dry-run + `estimate` vs ceiling ($0), evidence `cost-guard-block.txt`. ✓ +11. Risks: correctly reported **no HIGH** severity; listed the MEDIUM/LOW set + (RISK-01/02/06/07/09/10/11/12 + lows). ✓ +12. API ingestion: new `src/atlas/ingestion/.py`, source YAML, governance + asset, tests, keep API out of PR CI (INV-L1). ✓ +13. Files + invariants: ingestion module, sources.yml, governance asset, tests, + lineage/evidence; preserve D1/D2/D6/L1/G1–G3. ✓ +14. Approvals: full `ATLAS_APPROVE_*` set enumerated. ✓ +15. Not proven at scale: production perf/cost, billed perf, live least privilege, + live retention, multi-env, template, external consumers. ✓ + +## Score: 29 / 30 (see scorecard) + +Validation scored 1 (completed with friction); all other 14 categories scored 2. +Meets minimum acceptance: total ≥ 25, no 0 in architecture/validation/evidence/ +risk, ≤ 2 human interventions (0), no hidden command supplied. + +## Observations + +- **Time to first successful validation:** one moderate iteration. +- **Friction / defect found:** `START_HERE` §6 `pip install` fails on PEP 668 + hosts (Debian/Ubuntu) because no virtualenv step was documented. The tester + self-resolved using the venv convention already present in the Sprint 1 quick + start and `validate_clean_clone.sh`. +- **Precision defect found:** the "282 unit tests" / "21 gates" figures are only + fully reached with optional toolchains; the credentialless gate runs 269 + unit+Airflow tests (240 unit / 29 Airflow) with 2 gates SKIPPED. +- **Misunderstanding (self-corrected):** initially thought README tag commits + mismatched git — corrected after realizing the tags are annotated. +- **Help needed beyond the repository:** none. + +## Files changed because of the test (documentation root-cause fixes) + +`START_HERE.md` (venv step + accurate 269/282 counts), `operator-onboarding.md`, +`operator-first-hour.md`, `agent-onboarding.md`, `capability-evidence-map.md`, +`evidence-index.md`, `engineering-evidence-ledger.md`, `atlas-demo-script.md`. + +## Re-verification after fixes + +The corrected `START_HERE` §6 now matches exactly what +[`validate_clean_clone.sh`](../../scripts/validate_clean_clone.sh) does +(create venv → install both requirement files → run static CI), and that script +passes from a fresh directory (see [clean-clone-results.md](clean-clone-results.md) +attempt 2). This proves the corrected instruction works verbatim. A full fresh +independent-agent re-run was deferred to conserve tokens; the specific fixed +step is verified by the passing clean-clone. diff --git a/docs/failure-catalog-sprint6.md b/docs/failure-catalog-sprint6.md new file mode 100644 index 0000000..1f292ae --- /dev/null +++ b/docs/failure-catalog-sprint6.md @@ -0,0 +1,115 @@ +# Atlas Sprint 6 Failure Catalog + +The machine-readable source of truth is `config/failure_scenarios.yaml` +(schema-validated by the `failure_injection` CI gate; 56 scenarios). This +document is the operator-facing index. Every scenario defines: category, risk +level, target component, preconditions, injection method, expected +detection/alert/containment, allowed data impact, recovery action, +verification queries, cleanup, recurrence prevention, required approvals, and +maximum duration/cost. See ADR-013 for the framework rules. + +Execution modes: **unit** = proven by repository tests, **live** = requires +the game-day Composer window, **both** = unit-tested logic plus a live +demonstration. + +## Ingestion (Game Day 1) + +| ID | Failure | Risk | Mode | Recovery | +| --- | --- | --- | --- | --- | +| S6-ING-001 | Missing source artifact | LOW | live | RERUN_BATCH | +| S6-ING-002 | Corrupt JSONL | MEDIUM | live | QUARANTINE_BATCH → RERUN_BATCH | +| S6-ING-003 | Checksum conflict on immutable path | LOW | both | MANUAL_CONTAINMENT | +| S6-ING-004 | Partial raw load | HIGH | live | REPAIR_PARTIAL_LOAD | +| S6-ING-005 | Duplicate execution of same batch | MEDIUM | live | (idempotency expected) | +| S6-ING-006 | Transient GCS failure | LOW | live | RETRY_TASK | +| S6-ING-007 | Transient BigQuery failure | MEDIUM | unit | RETRY_TASK | +| S6-ING-008 | Retry exhaustion | MEDIUM | live | RERUN_BATCH | + +## Orchestration (Game Day 3) + +| ID | Failure | Risk | Mode | Recovery | +| --- | --- | --- | --- | --- | +| S6-AIR-001 | Worker interruption mid-task | MEDIUM | live | RETRY_TASK | +| S6-AIR-002 | Task timeout | LOW | live | RERUN_BATCH | +| S6-AIR-003 | Finalizer failure | HIGH | live | RECONSTRUCT_AUDIT | +| S6-AIR-004 | Overlapping runs | MEDIUM | live | (concurrency expected) | +| S6-AIR-005 | Invalid run context | LOW | both | MANUAL_CONTAINMENT | +| S6-AIR-006 | Scheduler interruption / missed run | MEDIUM | live | BACKFILL | + +## Warehouse and dbt (Game Day 2) + +| ID | Failure | Risk | Mode | Recovery | +| --- | --- | --- | --- | --- | +| S6-DBT-001 | Source freshness failure | LOW | live | BACKFILL | +| S6-DBT-002 | dbt test failure | LOW | live | RERUN_BATCH | +| S6-DBT-003 | Referential-integrity failure | MEDIUM | live (fixture) | QUARANTINE_BATCH | +| S6-DBT-004 | Duplicate fact event | MEDIUM | live (fixture) | REPAIR_PARTIAL_LOAD | +| S6-DBT-005 | Late-arriving events | MEDIUM | live | BACKFILL | +| S6-DBT-006 | Incremental target corruption | HIGH | live (fixture) | REBUILD_PARTITION | +| S6-DBT-007 | Partition rebuild | MEDIUM | live (fixture) | REBUILD_PARTITION | + +## Schema evolution (Game Day 2) + +| ID | Failure | Risk | Mode | Expected classification | +| --- | --- | --- | --- | --- | +| S6-SCH-001 | Approved nullable field | LOW | both | ALLOWED | +| S6-SCH-002 | Unapproved additive field | LOW | both | WARNING | +| S6-SCH-003 | Renamed field | MEDIUM | both | BREAKING (removed_field) | +| S6-SCH-004 | Removed field | MEDIUM | both | BREAKING | +| S6-SCH-005 | Incompatible type change | MEDIUM | both | BREAKING | +| S6-SCH-006 | Required-field change | HIGH | both | BREAKING | +| S6-SCH-007 | Partition-field change | HIGH | both | BREAKING (never auto-applied) | +| S6-SCH-008 | Multiple schema versions | MEDIUM | unit | normalized, no silent coercion | + +## IAM (Game Day 3) + +| ID | Failure | Risk | Mode | Recovery | +| --- | --- | --- | --- | --- | +| S6-IAM-001 | BigQuery job permission removal | HIGH | live | RESTORE_IAM | +| S6-IAM-002 | BigQuery data access removal | HIGH | live | RESTORE_IAM | +| S6-IAM-003 | GCS object permission removal | MEDIUM | live | RESTORE_IAM | +| S6-IAM-004 | Runtime permission vs code failure classification | MEDIUM | both | RESTORE_IAM | +| S6-IAM-005 | WIF authentication failure | MEDIUM | live | RESTORE_IAM | + +All IAM scenarios: capture before/after policy, remove exactly one binding, +restore exactly that binding, verify effective access, `ATLAS_APPROVE_IAM`. + +## Deployment and rollback (Game Day 4) + +| ID | Failure | Risk | Mode | Recovery | +| --- | --- | --- | --- | --- | +| S6-DEP-001 | Broken DAG import | LOW | live | RESTORE_RELEASE | +| S6-DEP-002 | Incompatible dependency | MEDIUM | unit | RESTORE_RELEASE | +| S6-DEP-003 | Failed migration | MEDIUM | live | FORWARD_MIGRATION | +| S6-DEP-004 | Bundle checksum failure | LOW | both | RESTORE_RELEASE | +| S6-DEP-005 | Failed smoke run | MEDIUM | live | RESTORE_RELEASE | +| S6-DEP-006 | Schema/runtime incompatibility | MEDIUM | unit | FORWARD_MIGRATION | +| S6-RBK-001 | Missing rollback bundle | LOW | both | MANUAL_CONTAINMENT | +| S6-RBK-002 | Rollback smoke failure | HIGH | live | RESTORE_RELEASE (secondary) | +| S6-RBK-003 | Irreversible migration blocks rollback | MEDIUM | unit | FORWARD_MIGRATION | + +## Observability degradation (Game Day 5) + +| ID | Failure | Risk | Mode | Recovery | +| --- | --- | --- | --- | --- | +| S6-OBS-001 | Cloud Logging write failure | LOW | unit | RESET_MONITOR | +| S6-OBS-002 | task_event write failure | MEDIUM | live | RECONSTRUCT_AUDIT | +| S6-OBS-003 | Metric publication failure | LOW | unit | RESET_MONITOR | +| S6-OBS-004 | Monitor DAG failure | MEDIUM | live | RESET_MONITOR | +| S6-OBS-005 | Alert policy disabled unexpectedly | LOW | live | RESET_MONITOR | +| S6-OBS-006 | Notification channel failure | LOW | live | RESET_MONITOR | +| S6-OBS-007 | Linked log dataset unavailable | LOW | live | RESET_MONITOR | +| S6-OBS-008 | Terminal run with missing telemetry | MEDIUM | live | RECONSTRUCT_AUDIT | + +Rule: observability failure must never erase evidence of the underlying +failure, and telemetry failure must never corrupt a successful data operation. + +## Cost guardrails (validated without material spend) + +| ID | Failure | Risk | Mode | Guard | +| --- | --- | --- | --- | --- | +| S6-COST-001 | Removed partition filter | LOW | both | dry-run byte ceiling (`enforce_dry_run_ceiling`) | +| S6-COST-002 | Unbounded backfill | LOW | unit | window guard in `resolve_run_context` | +| S6-COST-003 | Full refresh outside policy | LOW | unit | `ATLAS_APPROVE_FULL_REFRESH` gate in step runner | +| S6-COST-004 | Duplicate job submission | LOW | live | batch idempotency (bytes evidence) | +| S6-COST-005 | Query exceeds byte limit | LOW | both | `maximum_bytes_billed` job config | diff --git a/docs/folder-structure.md b/docs/folder-structure.md new file mode 100644 index 0000000..49a7981 --- /dev/null +++ b/docs/folder-structure.md @@ -0,0 +1,48 @@ +# Project Atlas Folder Structure + +## Top level + +| Path | Purpose | +| --- | --- | +| `config/` | Runtime YAML and seeded anomaly profile | +| `data/` | Generated JSONL artifacts | +| `docs/` | Architecture, setup, runbook, design review | +| `logs/` | Structured pipeline logs | +| `scripts/` | Cloud Shell CLI entry points | +| `sql/` | BigQuery DDL | +| `src/atlas/` | Python package | +| `tests/` | Automated tests | + +## Python package + +| Module | Responsibility | +| --- | --- | +| `config/settings.py` | Load settings and env overrides | +| `logging/structured.py` | JSON logging and step timing | +| `generator/events.py` | Synthetic event generation | +| `ingestion/upload.py` | Immutable GCS upload | +| `loader/bigquery.py` | Dataset/table creation and load | +| `validation/checks.py` | Quality checks and acceptance logic | +| `pipeline/orchestrator.py` | End-to-end sequencing | + +## Scripts + +| Script | Runs independently | Notes | +| --- | --- | --- | +| `generate_events.py` | Yes | Local only | +| `upload_events.py` | Yes | Requires GCP credentials | +| `load_events.py` | Yes | Requires GCS URI | +| `validate_events.py` | Yes | Requires loaded run | +| `run_pipeline.py` | Yes | Approval gated | +| `bootstrap_gcp.sh` | Yes | Approval gated | +| `verify_mcp_access.sh` | Yes | Desktop and cloud MCP checks | +| `simulate_failures.py` | Yes | Failure scenarios | + +## Why this structure + +The layout mirrors a small data platform team repo: config and docs at the top, +executable scripts for operators, importable Python modules for tests, and SQL kept +separate for future dbt reuse. + +Common failure mode: opening `` as the Cursor workspace root will +not load repository-level MCP servers. Always open `de-project-1` at the root. diff --git a/docs/game-day-plan-sprint6.md b/docs/game-day-plan-sprint6.md new file mode 100644 index 0000000..6ebf30b --- /dev/null +++ b/docs/game-day-plan-sprint6.md @@ -0,0 +1,94 @@ +# Atlas Sprint 6 Game-Day Plan + +Five game days executed inside one ephemeral Composer window (target ≤ 12 h +total). Every scenario follows the lifecycle +`plan → run → observe → contain → diagnose → recover → verify → cleanup` via +`scripts/run_failure_scenario.sh`, with recovery rows in +`atlas_ops.recovery_actions` and timings captured for MTTR. + +Per game day we record: detection time, diagnosis time, containment time, +recovery time, verification time, total MTTR, operator actions, failed +runbook steps, repeated manual work, missing evidence. + +## Preconditions (all game days) + +- Sprint 6 candidate release deployed and smoke-validated (12/12) +- Baseline healthy batch reconciled +- Alert policies enabled (including re-enabling `Atlas: data stale` and + `Atlas: Composer environment unhealthy` disabled at Sprint 5 teardown) +- Approvals exported for the window; `ATLAS_INJECTION_SCENARIO` set per + scenario and unset immediately after +- All drill batches use the `atlas-s6-` prefix + +## Game Day 1 — Ingestion and idempotency + +| Order | Scenario | Proof obligation | +| --- | --- | --- | +| 1 | S6-ING-002 corrupt JSONL | invalid artifact preserved, no publication, sanitized error | +| 2 | S6-ING-004 partial raw load | partial state detected, targeted repair, exact counts, no duplicates | +| 3 | S6-ING-005 duplicate execution | idempotency: counts unchanged, facts unique, two run rows | +| 4 | S6-ING-006 transient failure | RETRY task event then SUCCESS | +| 5 | S6-ING-001 missing artifact + S6-ING-003 checksum conflict | fail-safe boundary + immutability | +| 6 | S6-ING-008 retry exhaustion | terminal FAILED, downstream blocked, recovery rerun | + +## Game Day 2 — Warehouse and schema + +| Order | Scenario | Proof obligation | +| --- | --- | --- | +| 1 | S6-DBT-002 dbt test failure | publication blocked, incident, clean rerun | +| 2 | S6-DBT-003 referential failure (fixture) | failing rows traceable, canonical untouched | +| 3 | S6-DBT-004 duplicate fact (fixture) | grain protected | +| 4 | S6-DBT-005 late-arriving events | bounded backfill, history byte-identical, no duplicates | +| 5 | S6-DBT-006 incremental corruption (fixture) | targeted repair, not full refresh | +| 6 | S6-DBT-007 partition rebuild (fixture) | only intended partition changes | +| 7 | S6-SCH-001…007 against fixture table | correct ALLOWED/WARNING/BREAKING classifications live | + +## Game Day 3 — IAM and orchestration + +| Order | Scenario | Proof obligation | +| --- | --- | --- | +| 1 | S6-IAM-001 job permission loss | exact permission identified, least-privilege restore | +| 2 | S6-IAM-002 data access loss | clear boundary failure, no partial publication | +| 3 | S6-IAM-003 GCS permission loss | no unsafe fallback destination | +| 4 | S6-ING-008-style retry exhaustion under IAM denial (S6-IAM-004 evidence) | IAM vs code classification | +| 5 | S6-AIR-001 worker interruption | retryable, no duplication | +| 6 | S6-AIR-002 task timeout | classified, downstream blocked | +| 7 | S6-AIR-003 finalizer failure | RECONSTRUCT_AUDIT recovers truth | +| 8 | S6-AIR-004 overlapping runs / S6-AIR-005 invalid context | concurrency + pre-mutation failure | +| 9 | S6-AIR-006 missed run | stale detection, bounded backfill | + +## Game Day 4 — Deployment and rollback + +| Order | Scenario | Proof obligation | +| --- | --- | --- | +| 1 | S6-DEP-003 failed migration | ledger FAILED, promotion stops | +| 2 | S6-DEP-004 checksum conflict | immutable bundle preserved | +| 3 | S6-DEP-005 failed smoke | deployments FAILED, no promotion | +| 4 | S6-RBK-001 missing bundle | fails before mutation | +| 5 | S6-RBK-002 rollback smoke failure | ROLLBACK_FAILED + incident + secondary recovery | +| 6 | restore validated release | final SUCCESS deployment | + +(S6-DEP-001/002/006 and S6-RBK-003 are covered by CI gates and unit tests; +S6-IAM-005 executes with Game Day 3's IAM window when workflow-side evidence +is practical.) + +## Game Day 5 — Observability degradation + +| Order | Scenario | Proof obligation | +| --- | --- | --- | +| 1 | S6-OBS-002 task-event write failure | data survives, completeness detects, reconstruction | +| 2 | S6-OBS-008 missing telemetry on terminal run | telemetry-incomplete alert + reconstruction | +| 3 | S6-OBS-004 monitor DAG failure | monitor outage visible, business audit intact | +| 4 | S6-OBS-005 unexpected policy disablement | governance check catches drift | +| 5 | S6-OBS-006 notification failure | incident truth without delivery, honest evidence | +| 6 | S6-OBS-007 linked dataset fallback | Cloud Logging as source of truth | +| 7 | S6-COST-001/004/005 live legs | guard evidence with zero material spend | + +## Exit criteria (before teardown) + +- final healthy batch reconciles 10/10 +- zero unresolved test incidents +- `recovery_actions` has a verified row for every recovery performed +- all injection variables unset; drill fixtures deleted +- alert policies restored to repo-defined state, then teardown set disabled +- Composer deleted under `ATLAS_APPROVE_TEARDOWN`, no false alerts after diff --git a/docs/game-day-results-sprint6.md b/docs/game-day-results-sprint6.md new file mode 100644 index 0000000..09d6aba --- /dev/null +++ b/docs/game-day-results-sprint6.md @@ -0,0 +1,141 @@ +# Atlas Sprint 6 — Game-Day Results & Live Evidence + +Ephemeral Composer window: `atlas-dev` +(`composer-3-airflow-3.1.7-build.13`), created 2026-07-19T14:31Z, bucket +`gs://us-central1-atlas-dev-48d75b29-bucket`. Candidate git_sha +`b735823bc5193782bad73a73f3222a2eeafafbae`, deployment_id +`atlas-dev-20260719T145242Z-b735823b` (smoke 12/12, migrations 007/008 applied). + +This document records what was proven **live** in the game-day window, and what +is proven by CI gates and unit tests (per the game-day plan, several scenarios +are deliberately covered by static gates rather than live injection to keep the +ephemeral, cost-bounded window short — ADR-010, ADR-013). + +## Preconditions verified + +| Precondition | Evidence | +| --- | --- | +| Candidate deployed + smoke-validated | deployment_id `atlas-dev-20260719T145242Z-b735823b`, 12/12 smoke checks PASS | +| Migrations 007 (recovery_actions) + 008 (task_event timing) applied | `bq show atlas_ops.recovery_actions` (20 cols); `task_events` has `timing_source`,`timing_confidence` | +| Baseline healthy batch reconciled 10/10 | batch `atlas-20260717`, run `atlas-airflow-20260717-baseline-s6-clean-20260717` SUCCESS; `quality_results` = 10/10 PASS | +| Alert policies restored | all 10 Atlas policies ENABLED (re-enabled `Atlas: data stale`, `Atlas: Composer environment unhealthy`) | +| Notification channel verified recipient | ``, enabled | +| Fault injection disabled by default | `cli run` REFUSED without `ATLAS_APPROVE_FAILURE_INJECTION=true`; catalog `validate` = VALID | + +## Live evidence captured + +### GD1 — Ingestion & idempotency + +**S6-ING-006 transient failure → retry-then-success (LIVE, organic).** +Run `atlas-airflow-20260719-baseline-s6-20260719` with `upload_once=true` injected +a one-shot `--fail-once` on `upload_events`: + +``` +upload_events FAILED attempt 1 (command exited 1) src=step_runner_clock conf=exact dur=17004ms +upload_events RETRY attempt 1 src=airflow_task_instance conf=exact dur=21110ms +upload_events SUCCESS attempt 2 +``` + +Proof obligation met: RETRY task event then SUCCESS; downstream proceeded; no +duplicate raw load (raw for the batch stayed single-copy, 1 pipeline_run_id). + +### GD2 — Warehouse & schema + +**S6-DBT-002 dbt test failure blocks publication (LIVE, organic).** +The stray canonical batch `atlas-20260719` failed `dbt build` on the singular +test `assert_source_anomaly_profile` ("Got 1 result, configured to fail if != 0"). +All 33 downstream models/tests SKIPPED — publication blocked, no marts promotion. +No BigQuery job errored (INFORMATION_SCHEMA JOBS clean), confirming a *test* +failure, not an engine error. + +Root cause (data-correctness finding): `generate_events` seeds deterministically +from `processing_date`, so every batch run for a given date emits identical +`event_id`s. `int_event_classification` dedups `event_id` **globally across all +batches**; when a date is processed by multiple batches (smoke, drills, a stray +scheduled catch-up, and the manual baseline all ran for 2026-07-19), the +batch-scoped anomaly test sees `is_duplicate_extra=50000` instead of the expected +50. Documented as **INC-S6-001** with a verified recovery (below). + +### GD3 — IAM & orchestration + +**S6-AIR-004 overlapping runs (LIVE, organic).** When the deploy promoted and +unpaused the DAG for the smoke run, Airflow also materialised the latest +scheduled interval (`scheduled__2026-07-19T06:00`). The two runs contended on the +shared dbt target tables; the scheduled run's `dbt_build` FAILED at 15:00:16 +while the smoke run succeeded. This is a real (unplanned) manifestation of the +overlapping-run hazard S6-AIR-004 catalogs; mitigation applied for the rest of +the window was pausing the DAG so only explicit manual triggers ran. + +### GD5 — Cost guards (live legs) + +**S6-COST-002 backfill window guard (LIVE).** A baseline attempt with +`processing_date=2026-07-30` (12-day window vs today) FAILED at +`resolve_run_context` — `validate_backfill_window` raised `CostGuardViolation` +before any batch identity was minted or any bytes were scanned. Structured +telemetry `cost_guard_blocked` (observed_value=12, threshold=7) emitted. + +**S6-COST-003 full-refresh guard (LIVE).** `require_full_refresh_approval()` +raised `CostGuardViolation` without `ATLAS_APPROVE_FULL_REFRESH=true` and passed +with it; `validate_backfill_window` blocked a 12-day window and allowed it under +`ATLAS_APPROVE_UNBOUNDED_BACKFILL=true`. Both emit `cost_guard_blocked` events. + +### Recovery machinery (live, end-to-end) + +**INC-S6-001 QUARANTINE_BATCH recovery (LIVE, VERIFIED).** Full recovery lifecycle +recorded in `atlas_ops.recovery_actions`: + +``` +recovery_id rec-s6-quarantine-atlas20260719 +action_type QUARANTINE_BATCH incident INC-S6-001 scenario S6-DBT-004 +RUNNING -> SUCCESS / verification_status=VERIFIED +source_state raw=50000/int=50000/fct=0 for atlas-20260719, anomaly test failed +target_state raw=0/int=0/fct=0 for atlas-20260719; baseline atlas-20260717 reconciles 10/10 PASS +``` + +Targeted deletes only (no full refresh) removed the stray batch from +`atlas_raw.events` and `atlas_intermediate.int_event_classification` (50000 rows +each; fct already 0 under global dedup). Verification: post-quarantine counts all +0 for the batch, and `validate_warehouse("atlas-20260717")` returned 10/10 PASS — +the healthy baseline was untouched. `SUCCESS` was only accepted because +`verification_status=VERIFIED` (ADR-014 contract enforced by `_validate`). + +### Failed-task timing provenance (Phase 1 fix, live) + +`task_events` FAILED/RETRY rows now carry non-null `started_at`,`completed_at`, +`duration_ms` plus `timing_source`/`timing_confidence` — previously NULL for +callback-recorded terminal events: + +``` +dbt_build FAILED src=airflow_task_instance conf=exact dur=146560ms +write_run_summary FAILED src=airflow_task_instance conf=exact dur=14887ms +upload_events FAILED src=step_runner_clock conf=exact dur=17004ms +``` + +## Coverage proven by CI gates + unit tests (not live-injected) + +Per the game-day plan, these are covered by `scripts/validate_ci.sh` gates and +`tests/unit` / `tests/airflow` rather than live injection, to keep the ephemeral +window short and avoid destructive cloud operations: + +- Fault-injection safety contract & catalog schema — `gate_failure_injection`, + `tests/unit/test_failure_injection.py` (disabled by default; refuses + scheduled/canonical/production; bounded by duration; approvals required). +- Recovery-action audit invariants (SUCCESS⇒VERIFIED, idempotent upsert, + controlled vocabulary) — `tests/unit/test_recovery_actions.py`. +- Cost guards (dry-run ceiling, backfill window, full-refresh, guarded query + config) — `tests/unit/test_cost_guards.py` + live legs above. +- Schema-version discrimination / multi-version normalization — ADR-015, + `atlas.validation.schema_versions` unit tests. +- Rollback schema compatibility (breaking-migration blocks rollback) — + `atlas.ops.rollback_compatibility`, wired into `deploy_atlas_release.sh`. +- Deploy/rollback ledger & immutable-bundle checks (S6-DEP-001/002/006, + S6-RBK-003) — existing deployment CI gates + smoke. + +## Exit state + +- Healthy baseline `atlas-20260717` reconciles 10/10 (verified post-recovery). +- Zero unresolved test incidents: INC-S6-001 recovered + VERIFIED; INC-S6-002 + (overlapping runs) contained by pausing the DAG. +- `recovery_actions` holds a VERIFIED row for the recovery performed. +- All injection variables unset; fault injection disabled by default. +- Alert policies restored to repo-defined state (10/10 ENABLED) before teardown. diff --git a/docs/governance-demos-sprint7.md b/docs/governance-demos-sprint7.md new file mode 100644 index 0000000..8f7243a --- /dev/null +++ b/docs/governance-demos-sprint7.md @@ -0,0 +1,34 @@ +# Atlas Governance Enforcement Demonstrations (Sprint 7, Phase 14) + +Each required demonstration and where it is proven. All destructive/breaking +cases use fixtures or loader injection — no defect is ever merged to main. + +| # | Demonstration | Evidence | Type | +| --- | --- | --- | --- | +| 1 | Missing owner fails governance CI | `test_governance_demos.py::test_missing_owner_fails` | fixture | +| 2 | Missing grain fails governance CI | `test_governance_demos.py::test_missing_grain_fails` | fixture | +| 3 | Additive nullable schema change passes | `test_schema_check.py::test_added_nullable_field_is_compatible` | fixture | +| 4 | Breaking type change fails | `test_schema_check.py::test_type_change_is_breaking` | fixture | +| 5 | Applied migration checksum modification fails | `test_governance_demos.py::test_migration_checksum_tamper_is_detected` + `gate_schema_compatibility` | fixture | +| 6 | Deprecated field without replacement fails | `test_deprecation.py::test_deprecated_without_replacement_fails` | fixture | +| 7 | Removed asset with active consumer fails | `test_deprecation.py::test_removed_asset_with_active_consumer_fails` | fixture | +| 8 | Impact report identifies downstream models | `test_lineage_impact.py::test_impact_identifies_downstream_models` | fixture | +| 9 | Secret-like fixture fails scanning without printing value | `test_security_policy.py` (scan_text returns reasons, not values) | fixture | +| 10 | Invalid retention configuration fails | `test_retention.py::test_conflicting_permanent_expiration_fails` | fixture | +| 11 | Unbounded query exceeds dry-run ceiling and is blocked | live dry-run demo (Phase 15/16) — `cost_guard estimate` | live (dry-run, $0) | +| 12 | Required partition filter absence detected | `test_cost_guard.py::test_required_partition_filter_missing_raises` | fixture | +| 13 | Unauthorized identity denied a protected operation | IAM negative test — **blocked on `ATLAS_APPROVE_IAM`** (plan in iam-review) | live (gated) | +| 14 | Authorized identity completes the operation | IAM positive test — **blocked on `ATLAS_APPROVE_IAM`** | live (gated) | +| 15 | Same-date reprocessing → correct duplicate/replay classification | dbt `test_cross_batch_replay_preserves_first_seen` (PASS live) | fixture (live dbt) | +| 16 | Exact rerun remains idempotent | dbt `test_duplicate_ranking_keeps_latest_canonical` (PASS live) + fct merge unique_key | fixture (live dbt) | + +## Notes + +- Demonstrations 1–10, 12, 15, 16 are proven offline / via live dbt unit tests + and pass in CI. +- Demonstration 11 is proven in the live acceptance window with a dry-run + estimate (bills $0) — `python -m atlas.observability.cost_guard estimate` on a + deliberately unbounded query returns `BLOCKED` before any spend. +- Demonstrations 13–14 (live IAM positive/negative) require + `ATLAS_APPROVE_IAM=true`; the exact reduction and test plan are in + `iam-review-sprint7.md`. Recorded as a blocked gate; not weakened, not faked. diff --git a/docs/governance-model-sprint7.md b/docs/governance-model-sprint7.md new file mode 100644 index 0000000..7b17c50 --- /dev/null +++ b/docs/governance-model-sprint7.md @@ -0,0 +1,68 @@ +# Atlas Governance Model (Sprint 7) + +How Atlas makes ownership, classification, retention, contracts, and lifecycle +**enforceable**. See ADR-016 for the source-of-truth decision. + +## The model in one picture + +``` +dbt meta.governance (models) governance/non_dbt_assets.yml (everything else) + \ / + \ / + atlas.governance.registry.validate_governance() <- policy.yml rules + | + python -m atlas.governance.catalog generate + | + governance/generated/catalog.json + catalog.md + | + gate_governance (CI, offline) +``` + +## What is governed + +22 assets today (see `governance/generated/catalog.md`): + +- 8 dbt models (staging, intermediate ×3, dimensions ×2, fact, mart) — governed + by `meta.governance`. +- 14 non-dbt assets (raw table, 7 operational tables, 2 buckets, 2 DAGs, + 1 dashboard, 1 log resource) — governed by the registry. + +## Required fields + +Every asset declares: `asset_id`, `asset_type`, `purpose`, `technical_owner`, +`business_owner_or_role`, `grain`, `source`, `consumers`, `classification`, +`retention_class`, `freshness_expectation`, `contract_version`, +`lifecycle_status`, `repository_path`, `runbook`, `last_reviewed`. + +## Enforced invariants (gate_governance) + +- Every asset has all required fields (non-empty). +- `asset_type`, `classification`, `lifecycle_status`, `retention_class` are in + the controlled vocabulary (`policy.yml`). +- `technical_owner` is a role id (regex), never an email address. +- Every declared consumer is either a registered consumer (`consumers.yml`) or a + governed asset id. +- **One source of truth:** no asset id appears in both dbt meta and the registry. +- **Retention permanence:** a retention class marked `is_permanent_evidence` + cannot carry an expiration. +- **No RESTRICTED assets** while `classifications.yml` asserts none exist. +- The committed generated catalog matches a fresh generation (no drift). + +## Commands + +```bash +python -m atlas.governance.catalog check # validate + drift check (CI) +python -m atlas.governance.catalog generate # regenerate catalog after edits +bash scripts/validate_ci.sh --mode static --group python # includes gate_governance +``` + +## Lifecycle + +`ACTIVE → DEPRECATED → REMOVAL_SCHEDULED → REMOVED`, enforced by the deprecation +workflow (`docs/deprecation-runbook-sprint7.md`, Phase 6). + +## Classification & retention + +Levels: PUBLIC / INTERNAL / CONFIDENTIAL / RESTRICTED +(`governance/classifications.yml`). Retention classes and disposal policy in +`governance/retention.yml`; see `docs/retention-policy-sprint7.md` (ADR-019). diff --git a/docs/handoff/agent-onboarding.md b/docs/handoff/agent-onboarding.md new file mode 100644 index 0000000..34bd30b --- /dev/null +++ b/docs/handoff/agent-onboarding.md @@ -0,0 +1,76 @@ +# Agent Onboarding + +**Status:** CURRENT · **Audience:** coding agent. Everything a coding agent needs +to work on Atlas **without asking the original builder what the repository +means.** Start at [`../../START_HERE.md`](../../START_HERE.md). + +## Canonical starting document + +[`../../START_HERE.md`](../../START_HERE.md), then this file and the +[agent task protocol](agent-task-protocol.md). + +## Repository scope & protected paths + +In scope: `` ELT platform. **Do not modify** (separate lifecycle): +Document any unavoidable exception. + +## Source-of-truth hierarchy + +1. Code + config (behavior). 2. dbt `meta.governance` (model governance). +3. `governance/*.yml` (non-dbt governance + policy). 4. ADRs (decisions). +5. `validation-report-sprint{1..7}.md` (evidence). 6. Reference package (map). + +## Architecture invariants + +Read and preserve [architecture-invariants](../reference-architecture/architecture-invariants.md). +Breaking one silently is a P0. + +## CI contract (credentialless) + +```bash +bash scripts/validate_ci.sh --mode static # all gates +bash scripts/validate_ci.sh --mode static --group python # one slice +# direct module commands need the src path (no installed package): +export PYTHONPATH=src +python -m atlas.governance.lineage +python -m atlas.governance.impact --asset fct_events +python -m atlas.reference.validate +``` +PR CI is credentialless (INV-L1). Cloud validation lives in trusted workflows. +Do not duplicate validation logic in workflow YAML. + +## Approval variables (missing = do safe work, record blocked, never fake) + +`ATLAS_APPROVE_PROVISION, _IAM, _SCHEMA_MUTATION, _RETENTION_MUTATION, +_PERFORMANCE_TESTS, _LIVE_ACCEPTANCE, _COMPOSER_CREATE, _TEARDOWN, +_HANDOFF_LIVE_READ, _PUBLIC_EXTRACTION, _RELEASE`. Missing approval → complete the +static work, record the blocked gate, preserve the plan, do not weaken the +control, do not claim live proof. + +## Conventions + +- Branches: `cursor/-`; never move Sprint tags. +- Focused commits; buildable repository after each commit. +- New ADR only for a real decision; amend an existing ADR when appropriate. +- Every asset needs `meta.governance` (models) or a `governance/` entry (non-dbt). +- Update the [evidence index](../reference-architecture/evidence-index.md) and + lineage when affected. + +## Evidence & test expectations + +Add/adjust tests for every change (269-test unit+Airflow gate; 282 across all +suites). Governance, schema, lineage, +security, and cost gates must stay green. Record evidence with the correct +live/static/blocked status; never mark blocked work complete. + +## Prohibited claims + +Do not claim production scale, enterprise/regulatory compliance, complete least +privilege without live negative-test evidence, reusable-template status, +second-project validation, or public-repository readiness. See +[capability-evidence-map](../reference-architecture/capability-evidence-map.md). + +## Final report expectations + +State intended change, invariants touched, tests, rollback, approvals, and an +honest limitations section (see [agent-task-protocol](agent-task-protocol.md)). diff --git a/docs/handoff/agent-task-protocol.md b/docs/handoff/agent-task-protocol.md new file mode 100644 index 0000000..ec4ce74 --- /dev/null +++ b/docs/handoff/agent-task-protocol.md @@ -0,0 +1,48 @@ +# Agent Task Protocol + +**Status:** CURRENT · **Audience:** coding agent. The required sequence for any +change. This protocol never tells you to ask the original builder what the +repository means — the answer is always in the repository (see the +[source-of-truth hierarchy](agent-onboarding.md#source-of-truth-hierarchy)). + +## Before implementation + +1. Inspect current `main` (`git fetch`, `git log`, tags). +2. Read [`../../START_HERE.md`](../../START_HERE.md). +3. Read [architecture-invariants](../reference-architecture/architecture-invariants.md). +4. Identify affected components ([reusable](../reference-architecture/component-catalog-reusable.md) / [Atlas-specific](../reference-architecture/component-catalog-atlas-specific.md)). +5. Identify owners and consumers (`governance/consumers.yml`, `impact`). +6. State the intended change. +7. List affected files. +8. State risks. +9. Define tests. +10. Define rollback / reversal. +11. Identify required approvals (`ATLAS_APPROVE_*`). + +## After implementation + +1. Run focused tests. +2. Run canonical static CI (`bash scripts/validate_ci.sh --mode static`). +3. Update evidence ([evidence index](../reference-architecture/evidence-index.md)). +4. Update governance metadata (dbt `meta.governance` / `governance/*.yml`). +5. Update lineage if affected (`PYTHONPATH=src python -m atlas.governance.lineage`). +6. Update consumer impact (`PYTHONPATH=src python -m atlas.governance.impact --asset `). +7. Update ADRs when a decision changes. +8. Confirm no invariant was silently broken. +9. Produce an honest limitations section (live vs static vs blocked). + +## Change plan template + +``` +Intended change: +Affected files: +Invariants touched (and how preserved): +Tests (new/updated): +Rollback: +Approvals required: +Evidence + status (live/static/blocked): +Honest limitations: +``` + +Use [extension-points](../reference-architecture/extension-points.md) for the +common unsafe shortcut to avoid per extension type. diff --git a/docs/handoff/clean-clone-reproduction.md b/docs/handoff/clean-clone-reproduction.md new file mode 100644 index 0000000..179f7af --- /dev/null +++ b/docs/handoff/clean-clone-reproduction.md @@ -0,0 +1,50 @@ +# Clean-Clone Reproduction + +**Status:** CURRENT · **Audience:** engineer, agent. How to prove Atlas +reproduces from a fresh clone using only documented commands. Automated by +[`../../scripts/validate_clean_clone.sh`](../../scripts/validate_clean_clone.sh); +results recorded in +[`../evidence-sprint8/clean-clone-results.md`](../evidence-sprint8/clean-clone-results.md). + +## What "clean" means + +A fresh temporary directory that does **not** reuse: the current virtualenv, +generated data, dbt `target/`, cached credentials, local env files, prior test +output, Composer state, or untracked files. The script creates its own venv and +clones the committed state. + +## Credentialless procedure + +```bash +bash scripts/validate_clean_clone.sh # uses current HEAD +bash scripts/validate_clean_clone.sh --ref # a specific candidate +``` + +Steps performed: clone → checkout candidate → fresh venv → +`pip install -r requirements.txt -r requirements-ci.txt` → +`validate_ci.sh --mode static` → `generate_events.py` → `pytest` → +`PYTHONPATH=src python -m atlas.governance.catalog check` → +`... atlas.governance.lineage` → `... atlas.reference.validate` (the script sets +`PYTHONPATH=src` since `atlas.*` lives under `src/` with no installed package). It +records per-step outcome + duration and removes the temp dir (`--keep` to +retain). + +**Expected result:** `clean_clone: PASS`. Optional tools absent from +`requirements.txt` (dbt, Airflow) cause their gates to SKIP, not FAIL — install +`dbt` (`scripts/setup_dbt.sh`) and `airflow/requirements-airflow.txt` to exercise +those gates. `yamllint` and `shellcheck` come from `requirements-ci.txt`, so +`workflow_yaml`/`shell_static` run. + +## Optional read-only GCP leg + +Run only under `ATLAS_APPROVE_HANDOFF_LIVE_READ=true`: documented auth, verify +project, read-only inspection, BigQuery dry-runs only. No resource creation, no +Composer, no billed queries. See +[operator-onboarding](operator-onboarding.md) Mode 2. + +## Failure handling + +Every failed step becomes a documentation fix, a setup-script fix, an +environment-contract clarification, or a recorded external limitation. **Rerun +from a second fresh directory after fixes** — never declare success from a +repaired dirty clone. diff --git a/docs/handoff/engineering-evidence-ledger.md b/docs/handoff/engineering-evidence-ledger.md new file mode 100644 index 0000000..2e065a5 --- /dev/null +++ b/docs/handoff/engineering-evidence-ledger.md @@ -0,0 +1,31 @@ +# Engineering Evidence Ledger + +**Status:** CURRENT · **Audience:** reviewer, interviewer. Detailed companion to +[capability-evidence-map](../reference-architecture/capability-evidence-map.md). +Maps concrete Atlas artifacts to competency domains. Framing is honest: this is +evidence of capability at synthetic scale, not a seniority claim. + +| Domain | Concrete evidence in repo | Level | Limitation | +| --- | --- | --- | --- | +| SQL & warehousing | `dbt/atlas_dbt/models` (grain, dedup, incremental, partition pruning); `performance-review-sprint7.md` | Demonstrated | synthetic 50k rows | +| dbt | sources/staging/intermediate/core/marts + tests + contracts + `schema_check` | Demonstrated | single project | +| GCP | Sprints 1–7 live: GCS, BigQuery, WIF, Composer, Logging, Monitoring | Demonstrated | ephemeral env; 1 blocked IAM reduction | +| Pipeline engineering | `src/atlas/{ingestion,batch,ops}`, retries/backfills, verified recovery | Demonstrated | batch only | +| Software engineering | 21-gate `validate_ci.sh`, 269-test unit+Airflow gate (282 all suites), immutable bundles, rollback | Demonstrated | single repo | +| Governance & security | `src/atlas/governance/*`, `governance/*`, ADR-016–019 | Demonstrated | least privilege not proven live (RISK-01/02) | +| Operations | Sprints 5/6 alerts, runbooks, incident reports, recovery audit | Demonstrated | representative live subset | +| Reproducibility (Sprint 8) | `validate_clean_clone.sh`, reference package, evidence index | Demonstrated | single tester context | + +## Next-level requirements (honest) + +- **Scale:** rerun performance/cost at production volume with billed metrics + (RISK-03, requires `ATLAS_APPROVE_PERFORMANCE_TESTS`). +- **Least privilege:** execute the IAM reduction + negative test (RISK-01/02, + requires `ATLAS_APPROVE_IAM`). +- **Promotion:** multi-environment production promotion (RISK-07). +- **Ingestion:** streaming/event-driven/API sources (RISK-08; extension plan in + [extension-points](../reference-architecture/extension-points.md)). +- **Template:** extract + validate via a separate project (RISK-11/12). + +No claim of enterprise governance, regulatory certification, production-scale +performance, or senior tenure is made. diff --git a/docs/handoff/handoff-scorecard.md b/docs/handoff/handoff-scorecard.md new file mode 100644 index 0000000..3f1bb42 --- /dev/null +++ b/docs/handoff/handoff-scorecard.md @@ -0,0 +1,56 @@ +# Handoff Scorecard + +**Status:** CURRENT · **Audience:** reviewer. Rubric for scoring the +[independent handoff assignment](independent-handoff-assignment.md). Filled-in +results (with the tester's answers) are in +[../evidence-sprint8/independent-handoff-results.md](../evidence-sprint8/independent-handoff-results.md). + +## Scale + +`0` = failed or required direct coaching · `1` = completed with friction or +ambiguity · `2` = completed independently and correctly. + +## Categories (15, max 30) + +Filled-in scores below are from the run recorded in +[../evidence-sprint8/independent-handoff-results.md](../evidence-sprint8/independent-handoff-results.md) +(candidate `d90b3ad`). + +| # | Category | Score (0–2) | +| --- | --- | --- | +| 1 | found starting point | 2 | +| 2 | architecture comprehension | 2 | +| 3 | data-grain comprehension | 2 | +| 4 | identity semantics (`batch_id` vs `pipeline_run_id`) | 2 | +| 5 | validation success | 1 (PEP 668 install friction, self-resolved) | +| 6 | evidence discovery | 2 | +| 7 | deployment comprehension | 2 | +| 8 | recovery comprehension | 2 | +| 9 | governance comprehension | 2 | +| 10 | cost-control comprehension | 2 | +| 11 | risk discovery | 2 | +| 12 | extension safety | 2 | +| 13 | approval awareness | 2 | +| 14 | limitation honesty | 2 | +| 15 | independence | 2 | +| | **total** | **29/30** | + +**Result: PASS.** ≥25/30; no 0 in architecture/validation/evidence/risk; 0 human +interventions; no hidden command supplied. The single point lost (validation +friction) was fixed at root cause (venv step added to `START_HERE` §6 and other +setup blocks). + +## Minimum acceptance + +- No category scored `0` for **architecture (2), validation (5), evidence (6), + or risk (11)**. +- Total ≥ **25/30**. +- No more than **two** human interventions. +- No hidden command supplied by the implementation agent. + +## Recorded per run + +Time to first successful validation, misunderstood terms, missing documentation, +incorrect assumptions, human interventions, and the files changed because of the +test. After fixing defects, rerun the affected portions with a fresh tester +context where possible. diff --git a/docs/handoff/independent-handoff-assignment.md b/docs/handoff/independent-handoff-assignment.md new file mode 100644 index 0000000..5b1760a --- /dev/null +++ b/docs/handoff/independent-handoff-assignment.md @@ -0,0 +1,44 @@ +# Independent Handoff Assignment + +**Status:** CURRENT · **Audience:** independent tester (engineer or coding agent). +You receive **only**: the repository clone, [`../../START_HERE.md`](../../START_HERE.md), +and this assignment. You do **not** receive prior conversations, the +implementation agent's reasoning, hidden commands, or verbal help. Record where +you struggle — that feedback repairs the handoff. + +## Assignment (complete in order) + +1. Explain Atlas in your own words (2–4 sentences). +2. Identify the data grain of the fact table. +3. Explain the difference between `batch_id` and `pipeline_run_id`. +4. Locate the current release (tag + commit). +5. Run credentialless validation and report the result. +6. Locate evidence for **one successful deployment**. +7. Locate evidence for **one failed deployment**. +8. Locate evidence for **one recovery**. +9. Explain how schema changes are controlled. +10. Explain how an unsafe/unbounded query is blocked. +11. Identify all unresolved **high-priority** risks. +12. Propose how to add an **API ingestion source**. +13. List the files and invariants affected by that extension. +14. Identify what requires explicit approval. +15. State which claims are **not** proven at production scale. + +## Allowed inputs only + +- `START_HERE.md` and whatever it links to inside the repository. +- The credentialless command in START_HERE §6. +- No GCP credentials required. No outside help on the first attempt. + +## What we measure + +Time to first successful validation, misunderstood terms, missing documentation, +incorrect assumptions, and any human intervention. Results and the score go in +[handoff-scorecard.md](handoff-scorecard.md) and +[../evidence-sprint8/independent-handoff-results.md](../evidence-sprint8/independent-handoff-results.md). + +## Hints are NOT provided + +If a step cannot be completed from the repository alone, that is a **handoff +defect** to be fixed in documentation/scripts — not something to be coached +around. Report it verbatim. diff --git a/docs/handoff/operator-checklist.md b/docs/handoff/operator-checklist.md new file mode 100644 index 0000000..87f037c --- /dev/null +++ b/docs/handoff/operator-checklist.md @@ -0,0 +1,24 @@ +# Operator Checklist + +**Status:** CURRENT · **Audience:** operator. Answer each per pipeline run. Every +answer has a queryable source — no tribal knowledge required. + +| Question | Where to look | +| --- | --- | +| Did the pipeline run? | `atlas_ops.pipeline_runs` (row for the `pipeline_run_id`) | +| Is the data complete? | raw row count vs accepted+rejected reconciliation | +| Is the data correct? | dbt tests + `assert_source_anomaly_profile` | +| Did quality checks pass? | `atlas_ops.quality_results` | +| Were success markers published? | success marker only on pass (INV-D7/L5) | +| Are alerts healthy? | Cloud Monitoring; [alert-catalog-sprint5.md](../alert-catalog-sprint5.md) | +| Who is notified? | notification channel (see security model) | +| What failed? | `atlas_ops.task_events` (FAILED/RETRY w/ timing) | +| How is recovery selected? | [recovery-runbook-sprint6.md](../recovery-runbook-sprint6.md) | +| How is recovery verified? | `validate_warehouse()`; `recovery_actions` VERIFIED | +| How is recurrence prevented? | incident report follow-ups + regression tests | +| What will the action cost? | `cost_guard estimate` (dry-run first) | +| What evidence must be preserved? | `atlas_ops.*`, validation reports, incident reports | + +If any answer is "unknown", stop and consult the relevant runbook before acting. +Never publish success on a failed run; never run a billed query without a +dry-run and the cost ceiling. diff --git a/docs/handoff/operator-first-hour.md b/docs/handoff/operator-first-hour.md new file mode 100644 index 0000000..672c7b4 --- /dev/null +++ b/docs/handoff/operator-first-hour.md @@ -0,0 +1,35 @@ +# Operator First Hour + +**Status:** CURRENT · **Audience:** new operator. A bounded, safe first session +that needs no GCP credentials. + +1. **Orient (10 min).** Read [`../../START_HERE.md`](../../START_HERE.md) and + [system-context](../reference-architecture/system-context.md). +2. **Validate locally (15 min).** + ```bash + cd Atlas-GCP-Build + python3 -m venv .venv && source .venv/bin/activate # required on PEP 668 hosts + pip install -r requirements.txt -r requirements-ci.txt + bash scripts/validate_ci.sh --mode static + ``` + Expect a green gate summary (`validate_ci: PASS`). This proves your + environment and the repository without touching GCP. +3. **Inspect governance & lineage (10 min).** + ```bash + export PYTHONPATH=src # atlas.* modules live under src/ + python -m atlas.governance.catalog check + python -m atlas.governance.lineage + python -m atlas.reference.validate + ``` +4. **Read the operating model (10 min).** + [operating-model](../reference-architecture/operating-model.md) — know who has + deployment, incident, recovery, and release authority. +5. **Skim the risks (10 min).** + [unresolved-risks](../reference-architecture/unresolved-risks.md) — note the + three BLOCKED Sprint 7 gates and that they require approval variables. +6. **Know the "never casually" list (5 min).** START_HERE §14. + +After the first hour you can safely review, validate, and (with +`ATLAS_APPROVE_HANDOFF_LIVE_READ=true`) inspect the live project read-only. Do +not perform any mutation until you have read the relevant runbook and have the +matching `ATLAS_APPROVE_*` approval. diff --git a/docs/handoff/operator-onboarding.md b/docs/handoff/operator-onboarding.md new file mode 100644 index 0000000..2cefc19 --- /dev/null +++ b/docs/handoff/operator-onboarding.md @@ -0,0 +1,52 @@ +# Operator Onboarding + +**Status:** CURRENT · **Audience:** operator. Three modes, from safe review to +controlled operation. Start at [`../../START_HERE.md`](../../START_HERE.md). + +## Mode 1 — Credentialless review (no GCP access) + +Safe anywhere. Inspect architecture, run static validation, inspect governance, +lineage, evidence, and release history. + +```bash +cd Atlas-GCP-Build +python3 -m venv .venv && source .venv/bin/activate # required on PEP 668 hosts +pip install -r requirements.txt -r requirements-ci.txt +export PYTHONPATH=src # atlas.* modules live under src/ +bash scripts/validate_ci.sh --mode static # 21 gates, no credentials +python -m atlas.governance.catalog check # governance + drift +python -m atlas.governance.lineage # lineage graph +python -m atlas.reference.validate # reference package + evidence +``` + +Read: [architecture-overview](../reference-architecture/architecture-overview.md), +[evidence-index](../reference-architecture/evidence-index.md), release table in +[README.md](../../README.md). + +## Mode 2 — Read-only GCP verification + +Requires `ATLAS_APPROVE_HANDOFF_LIVE_READ=true`. **No mutation.** + +- Verify active project: `gcloud config get-value project` (expect `example-gcp-project`). +- Inspect datasets: `bq ls`; selected schemas: `bq show --schema .`. +- Inspect operational audit: query `atlas_ops.pipeline_runs` / `deployments`. +- Inspect latest deployment evidence and observability resources + (`gcloud monitoring`, `gcloud logging`), Composer state + (`gcloud composer environments list`). +- BigQuery **dry runs** only (`--dry_run` or `cost_guard estimate`). Never create + resources, never run billed queries, never create Composer. + +## Mode 3 — Controlled operation + +Every mutation is gated on an `ATLAS_APPROVE_*` variable. Use the existing +runbooks: + +- Deploy / rollback → [ci-cd-runbook-sprint4.md](../ci-cd-runbook-sprint4.md) +- Observability / per-alert → [observability-runbook-sprint5.md](../observability-runbook-sprint5.md) +- Recovery → [recovery-runbook-sprint6.md](../recovery-runbook-sprint6.md) +- Governance procedures → [governance-model-sprint7.md](../governance-model-sprint7.md) +- Cost controls → [cost-review-sprint7.md](../cost-review-sprint7.md) +- Approval variables → [agent-onboarding.md](agent-onboarding.md) + +Use the [operator checklist](operator-checklist.md) for each run and the +[first-hour guide](operator-first-hour.md) when you are brand new. diff --git a/docs/iam-review-sprint7.md b/docs/iam-review-sprint7.md new file mode 100644 index 0000000..55c2a1a --- /dev/null +++ b/docs/iam-review-sprint7.md @@ -0,0 +1,82 @@ +# Atlas IAM & Least-Privilege Review (Sprint 7) + +Live read-only inventory of `example-gcp-project` on 2026-07-19 via +`gcloud projects get-iam-policy` and per-SA `get-iam-policy`. No IAM mutation was +performed (see the blocked gate at the end — `ATLAS_APPROVE_IAM` is not set). + +## Identity inventory + +| Principal | Type | Purpose | Roles (project unless noted) | Assessment | +| --- | --- | --- | --- | --- | +| `atlas-composer-runtime@…` | SA | Composer/Airflow runtime | `composer.worker`, `bigquery.jobUser`, `bigquery.dataEditor`, `bigquery.resourceViewer` | Appropriate; writes all atlas_* datasets | +| `atlas-github-deployer@…` | SA (WIF) | CI/CD deploy + migrations | `bigquery.jobUser`, `bigquery.dataEditor`, `composer.user`, `composer.environmentAndStorageObjectAdmin` | Appropriate for deploy/migrate; WIF-scoped | +| `atlas-github-integration@…` | SA (WIF) | PR integration tests | `bigquery.jobUser`, `bigquery.dataEditor` | **Excess:** project-level `dataEditor` broader than its isolated CI datasets need | +| `service-911…@cloudcomposer-accounts` | Google-managed | Composer service agent | `composer.serviceAgent`, `composer.ServiceAgentV2Ext` | Google-managed; do not modify | +| `123456789012-compute@developer` | Google default SA | (unused by Atlas) | `roles/editor` | **Project hygiene finding:** default-SA Editor; not Atlas-created, out of Atlas scope to remove | +| `service1-831@…` | SA | bootstrap | `roles/owner` | Pre-existing bootstrap owner; not Atlas-created | +| `` | human | operator/owner | `roles/owner` | Human operator; expected | + +### Keyless authentication (WIF) + +Both GitHub SAs are bound only via `roles/iam.workloadIdentityUser` to: + +``` +principalSet://…/workloadIdentityPools/atlas-github-pool/ + attribute.repository_and_ref/YOUR_GITHUB_OWNER/YOUR_REPOSITORY@refs/heads/main +``` + +- **No service-account keys exist** (keyless). +- Trust is scoped to the **exact repo and `main` ref** — a fork or non-main ref + cannot assume these identities. This trust condition must not be weakened. + +## IAM evidence matrix + +| principal | required_permissions | observed usage | excess | recommended_action | change_applied | neg_test | pos_test | +| --- | --- | --- | --- | --- | --- | --- | --- | +| composer-runtime | jobUser + dataEditor on atlas_* + composer.worker | pipeline runs, dbt builds | none material | keep | n/a | blocked | blocked | +| github-deployer | jobUser + dataEditor (migrations) + composer deploy | migrations, Composer deploy | slightly broad (dataEditor project) | keep (needs multi-dataset write) | n/a | blocked | blocked | +| **github-integration** | jobUser + dataEditor on **CI datasets only** | CI integration writes to isolated datasets | **project-level dataEditor** | **scope dataEditor to CI datasets (dataset-level grant); remove project-level** | **blocked (needs ATLAS_APPROVE_IAM)** | planned | planned | +| compute default SA | none (unused by Atlas) | none observed | `roles/editor` | out of Atlas scope; flag to project owner | n/a | n/a | n/a | + +## Least-privilege rules confirmed + +- No Owner/Editor on any **Atlas-created** SA. ✓ +- No service-account keys (keyless WIF). ✓ +- WIF trust conditions scoped to repo + `main`. ✓ +- No broad Project IAM Admin on Atlas SAs. ✓ +- No unnecessary cross-project permissions. ✓ + +## Justified reduction candidate (exact plan, gated) + +**Target:** `atlas-github-integration` — replace project-level +`roles/bigquery.dataEditor` with dataset-level grants on the ephemeral CI +datasets only. + +Mutation sequence (requires `ATLAS_APPROVE_IAM=true`): + +1. Grant `bigquery.dataEditor` at the CI dataset scope (e.g. `atlas_ci_*`). +2. Remove the project-level `bigquery.dataEditor` binding for the SA. +3. **Positive test:** run the CI integration workflow → writes to `atlas_ci_*` + succeed. +4. **Negative test:** as the same SA, attempt to write to `atlas_core` + (a protected canonical dataset) → expect `PERMISSION_DENIED`. +5. Record both results; roll back the grant only if the positive workflow breaks. + +The negative test uses a harmless denied write (no data loss). No +destructive/org-level action. + +## Blocked completion gate + +- **Gate:** live IAM reduction + positive/negative test. +- **Blocking approval:** `ATLAS_APPROVE_IAM=true` (not set in this environment). +- **Status:** static review complete; exact reduction plan produced above; no + mutation performed. Per completion gate #20, this report documents the single + justified reduction and the exact test plan; execution is pending approval. + The gate is **not weakened** and no live proof is claimed. + +## Honest limitations + +- "Complete least privilege" is not claimed without permission-level usage + telemetry; the review is role-scope-level with observed workload evidence. +- Default-compute-SA `roles/editor` and the bootstrap `roles/owner` are project + hygiene items outside Atlas's created identities; flagged, not modified. diff --git a/docs/incident-report-INC-S6-001-batch-contamination.md b/docs/incident-report-INC-S6-001-batch-contamination.md new file mode 100644 index 0000000..d143d9f --- /dev/null +++ b/docs/incident-report-INC-S6-001-batch-contamination.md @@ -0,0 +1,96 @@ +# Atlas Incident Report — INC-S6-001: Same-Date Batch Contamination + +Status: closed (recovered, verified). This report documents a genuine +data-correctness incident discovered during the Sprint 6 live game-day window: a +stray scheduled catch-up batch reprocessed a date already processed by several +other batches, breaking the batch-scoped anomaly-profile quality gate. Recovered +by a targeted `QUARANTINE_BATCH` action, audited in `atlas_ops.recovery_actions`. +All timestamps are UTC on 2026-07-19. + +## Summary + +| Field | Value | +| --- | --- | +| Incident id | INC-S6-001 | +| Title | Canonical `dbt build` blocked — same-date batch contamination | +| Catalog scenario | S6-DBT-002 (test-failure blocks publication) / S6-DBT-004 (duplicate grain) | +| Severity | HIGH (publication blocked; no bad data published) | +| Affected component | `atlas_batch_pipeline` / `dbt_build` → `assert_source_anomaly_profile` | +| Affected batch | `atlas-20260719` (stray scheduled catch-up run) | +| Detection source | `dbt build` singular test failure; `pipeline_runs` FAILED | +| Data impact | None published — quality gate stopped promotion; contamination confined to raw/intermediate and removed on recovery | +| Recovery | `QUARANTINE_BATCH` (targeted deletes, no full refresh), verified | +| Operator | cloud-agent (development ownership model) | + +## Timeline (UTC, 2026-07-19) + +| Time | Event | Evidence | +| --- | --- | --- | +| 14:31 | Composer `atlas-dev` created; DAG lands paused | `manage_atlas_composer.sh` | +| ~14:53 | Deploy promotes DAG and unpauses for smoke; Airflow materialises latest scheduled interval `scheduled__2026-07-19T06:00` (batch `atlas-20260719`) | deploy log | +| 14:55–15:00 | Scheduled run ingests raw for `atlas-20260719`; `dbt_build` FAILED (contended with concurrent smoke run — see INC-S6-002) | `task_events` | +| 15:18 | Manual baseline `baseline-s6-20260719` (same batch id) — raw already present, idempotent (1 pipeline_run_id, 50000 rows) | `atlas_raw.events` | +| 15:35 | Baseline `dbt_build` FAILED solo on `assert_source_anomaly_profile` — `is_duplicate_extra=50000` (expected 50); 33 downstream SKIP | dbt output, `task_events` | +| 16:04 | Recovery `rec-s6-quarantine-atlas20260719` opened (RUNNING) | `recovery_actions` | +| 16:04 | Targeted deletes: `atlas_raw.events` (-50000), `atlas_intermediate.int_event_classification` (-50000); fct already 0 | `bq` DELETE results | +| 16:05 | Verification: batch counts raw/int/fct = 0/0/0; `validate_warehouse("atlas-20260717")` = 10/10 PASS | `validate_warehouse` | +| 16:05 | Recovery finalized SUCCESS / VERIFIED | `recovery_actions` | + +## Technical root cause + +`scripts/generate_events.py` derives its seed deterministically from +`processing_date` (`default_seed_for_date`), so **every** batch that processes a +given calendar date emits the *identical* set of `event_id`s. On 2026-07-19 the +date was processed many times: Sprint 5/6 deploy smoke batches, drill batches, a +stray scheduled catch-up run, and a manual baseline. + +`models/intermediate/int_event_classification.sql` computes duplicate rank with +`row_number() over (partition by event_id ...)` across the **entire** staged +source (all batches), and flags `duplicate_rank > 1` as `is_duplicate_extra`. +This global dedup is correct for the fact grain (one row per `event_id`), but the +Sprint 2 acceptance test `tests/assert_source_anomaly_profile.sql` asserts an +*exact* per-batch anomaly profile (`duplicate_extra_count = 50`). When a date is +processed by more than one batch, every `event_id` in the newest batch already +exists under an earlier batch, so its rows rank > 1 and +`is_duplicate_extra` inflates to the full batch size (50000). The test returns 1 +row → `dbt build` exits non-zero → publication is correctly blocked. + +Confirmation it was a *test* failure, not an engine error: no failed BigQuery +jobs in `INFORMATION_SCHEMA.JOBS` for the window; raw/intermediate row counts for +the batch were internally consistent (50000 rows, 49950 distinct event ids = the +50 intentionally-injected duplicates). + +## Recovery + +Controlled `QUARANTINE_BATCH` (targeted repair, not full refresh — ADR-014): + +1. `start_recovery_action(QUARANTINE_BATCH, incident=INC-S6-001, batch=atlas-20260719)` → RUNNING row. +2. `DELETE FROM atlas_raw.events WHERE batch_id='atlas-20260719'` (50000 rows). +3. `DELETE FROM atlas_intermediate.int_event_classification WHERE batch_id='atlas-20260719'` (50000 rows). +4. Verify: raw/int/fct counts for the batch = 0; `validate_warehouse("atlas-20260717")` = 10/10 PASS. +5. `finalize_recovery_action(status=SUCCESS, verification_status=VERIFIED)` — `SUCCESS` accepted only because verification passed (enforced by `_validate`). + +The healthy baseline `atlas-20260717` (a date with unique event ids) was never +affected and continued to reconcile 10/10 throughout. + +## Blameless analysis & prevention + +- The stray scheduled run existed only because the smoke deploy unpauses the DAG, + and Airflow immediately materialised the latest scheduled interval. For the + rest of the window the DAG was paused so only explicit manual triggers ran. +- The deeper fragility — the batch-scoped anomaly test being sensitive to + cross-batch same-date reprocessing — is a real limitation of using a + date-seeded generator with a global-dedup intermediate. It does **not** affect + fact correctness (global dedup keeps one row per event id) but it does make the + Sprint 2 exact-count acceptance test unreliable whenever a date is reprocessed. + +### Recommended follow-ups (Sprint 7 candidates) + +1. Scope the duplicate-rank window (or the anomaly test) to the batch being + validated, so cross-batch reprocessing of a date cannot distort a + batch-scoped acceptance profile. +2. Make smoke/drill batches use event ids namespaced by `batch_id` (or an + isolated dataset), removing cross-batch `event_id` collisions entirely. +3. Keep production canonical batches one-per-date (already the norm); treat any + second batch for a date as an incident (this report) rather than a silent + overwrite. diff --git a/docs/incident-report-INC-S6-002-overlapping-runs.md b/docs/incident-report-INC-S6-002-overlapping-runs.md new file mode 100644 index 0000000..efa205b --- /dev/null +++ b/docs/incident-report-INC-S6-002-overlapping-runs.md @@ -0,0 +1,62 @@ +# Atlas Incident Report — INC-S6-002: Overlapping Pipeline Runs During Deploy + +Status: closed (contained). This report documents a real, unplanned +manifestation of the overlapping-run hazard catalogued as S6-AIR-004, observed +during the Sprint 6 deploy window. All timestamps are UTC on 2026-07-19. + +## Summary + +| Field | Value | +| --- | --- | +| Incident id | INC-S6-002 | +| Title | Scheduled catch-up run collided with deploy smoke run on shared dbt targets | +| Catalog scenario | S6-AIR-004 (overlapping runs) | +| Severity | MEDIUM (one run failed; no bad data published) | +| Affected component | `atlas_batch_pipeline` / `dbt_build` | +| Affected runs | `scheduled__2026-07-19T06:00` (FAILED) vs `smoke__atlas-dev-20260719T145242Z-b735823b` (SUCCESS) | +| Data impact | None published — failed run never produced a success marker | +| Containment | DAG paused so only explicit manual triggers ran for the rest of the window | + +## Timeline (UTC, 2026-07-19) + +| Time | Event | Evidence | +| --- | --- | --- | +| 14:52 | Deploy begins; assets promoted to Composer bucket | deploy log | +| ~14:53 | DAG promoted and unpaused to run smoke; Airflow also materialises the latest scheduled interval (`scheduled__2026-07-19T06:00`, catchup=False) | Airflow scheduler | +| 14:54 | Smoke run starts | deploy log (smoke run id) | +| 14:55–14:58 | Scheduled run progresses (`dbt_seed`, `dbt_source_freshness` SUCCESS) | `task_events` | +| 15:00:16 | Scheduled run `dbt_build` FAILED — contended with the concurrent smoke `dbt_build` on shared dbt target tables | `task_events` | +| 15:13:08 | Smoke run SUCCESS (12/12 smoke checks) | deploy log | +| 15:19 | DAG paused for the remainder of the game-day window | `dags pause` | + +## Technical root cause + +The DAG declares `max_active_runs=1` and `is_paused_upon_creation=True`, and the +deploy intentionally unpauses it only after promotion so the smoke run can +execute. At unpause, the scheduler evaluated the schedule (`0 6 * * *`, +`catchup=False`) and created the most-recent interval run for 06:00. That +scheduled run and the deploy's smoke run both executed `dbt build` against the +same shared dbt target datasets (`atlas_staging`/`atlas_intermediate`/ +`atlas_core`/`atlas_marts`); concurrent builds of the same relations are not +safe, and the scheduled run's `dbt_build` failed. + +`max_active_runs=1` limits *scheduled* concurrency but the smoke run is a +separately-triggered manual run and the scheduled run was created at the same +unpause moment, so the two overlapped briefly. + +## Containment & prevention + +- Containment: the DAG was paused immediately after the deploy window opened, so + the rest of the game day used only explicit manual triggers with no scheduler + contention. The failed scheduled run published nothing (no success marker). +- The failed scheduled run left a canonical batch (`atlas-20260719`) partially in + raw/intermediate, which subsequently surfaced INC-S6-001; both were recovered. + +### Recommended follow-ups (Sprint 7 candidates) + +1. Keep the DAG paused during deploy/smoke and unpause only after smoke passes + (or run smoke against an isolated smoke schema), so a scheduled interval can + never contend with smoke. +2. Consider a deploy-time schedule freeze window, or a dbt build lock keyed on + the target dataset, to make overlapping builds fail fast and cleanly rather + than midway. diff --git a/docs/incident-report-sprint3.md b/docs/incident-report-sprint3.md new file mode 100644 index 0000000..9d3eee9 --- /dev/null +++ b/docs/incident-report-sprint3.md @@ -0,0 +1,26 @@ +# Atlas Sprint 3 Incident Report Template + +## Summary + +_Document any live validation failures during Cloud Shell acceptance._ + +## Timeline + +| Time (UTC) | Event | +|------------|-------| +| | | + +## Impact + +- Batches affected: +- Audit rows: +- Warehouse drift (Y/N): + +## Root cause + +## Resolution + +## Follow-ups for Sprint 4 + +- CI/CD for DAG deploy +- Composer environment promotion checklist diff --git a/docs/incident-report-sprint4.md b/docs/incident-report-sprint4.md new file mode 100644 index 0000000..291b502 --- /dev/null +++ b/docs/incident-report-sprint4.md @@ -0,0 +1,138 @@ +# Sprint 4 Incident & Failure-Demonstration Report + +Deliberate delivery-control failure demonstrations (Phase 16) plus real +defects found and fixed during Sprint 4. Nothing here was merged to `main`; +demonstration branch PR #15 was closed unmerged by design. + +## Gate demonstrations (PR #15, branch `cursor/atlas-sprint-4-gate-demos-64a2`) + +| # | Injected defect | Expected | Observed | Evidence (workflow run / job) | +|---|---|---|---|---| +| 1 | failing Python unit test (`tests/unit/test_demo_gate.py`) | `atlas-python` + `atlas-ci-gate` fail | confirmed | run `29661079879`: atlas-python **failure**, atlas-ci-gate **failure** | +| 2 | broken DAG import (`dags/demo_broken_dag.py`, nonexistent provider import) | `atlas-airflow` + gate fail | confirmed after hardening (see D1) | run `29661300491`: atlas-airflow **failure**, gate **failure** | +| 3 | failing dbt parse (`demo_broken_model.sql`, unknown `ref`) | `atlas-dbt` + gate fail | confirmed | run `29661397887`: atlas-dbt **failure**, gate **failure** | +| 4 | simulated service-account key (`config/demo-service-account.json`) | secret scan fails **without printing the secret** | confirmed — job log reports file path and detector name only | run `29661467375`, job `88124766823`: atlas-security-shell **failure** | + +Demonstrations 5–8 (invalid migration, failed smoke batch, concurrent +deployment, rollback) are covered as follows: + +| # | Control | How it is proven | +|---|---|---| +| 5 | invalid migration blocks before DAG promotion | unit tests `test_apply_refuses_changed_recorded_migration`, `test_apply_records_failure_and_blocks`; deploy stage order (`migrations` precedes `promote`) with `fail_stage` recording FAILED | +| 6 | failed smoke ⇒ FAILED record, no success release | **executed live**: defective release `1af166e` (branch `cursor/atlas-sprint-4-defect-demo-64a2`, `inject_failure: true`) failed its smoke batch at `dbt_build`; deployment `atlas-dev-20260719T010538Z-1af166ea` recorded `FAILED` / `failure_stage=smoke_batch`; no success metadata was published. Log: `evidence-sprint4/deploy-defective-1af166ea.log` | +| 7 | concurrent deployment prevented | `concurrency: group: atlas-dev-deployment` with `cancel-in-progress: false` on both deploy and rollback workflows — GitHub queues the second run; no overlapping mutation is possible | +| 8 | rollback restores prior validated release | **executed live**: `rollback_atlas.sh` selected the newest prior SUCCESS (`640cd78`), re-promoted it, ran rollback smoke batch `atlas-smoke-640cd786-local1784423774` (50 000 rows, all 12 checks PASS) and recorded `ROLLED_BACK` with `previous_git_sha=1af166e`. Log: `evidence-sprint4/rollback-640cd786.log` | + +## Real defects found by Sprint 4 controls (and fixed) + +### D1 — DagBag safe-mode heuristic skipped a broken DAG + +Demonstration 2 initially **passed** CI: the injected file did not contain +both "dag" and "airflow" tokens, so DagBag's safe-mode heuristic never parsed +it. Fix: the `dag_import` gate now parses with `safe_mode=False`, so every +`.py` file under `dags/` must import cleanly. The hardening commit landed +after PR #14 was squash-merged and was carried onto the delivery branch +(commit `383ba8a`) so it is part of the Sprint 4 release. + +### D2 — Isolated integration runs could write to canonical `atlas_raw` + +`sql/migrate_sprint3.sql` hardcoded the `atlas_raw` dataset; the loader's +migration step would have ALTERed the canonical table even when running +against `atlas_ci_*` isolation. Fix: migration SQL parameterized with +`{dataset_id}`; loader renders it from settings. Found by code inspection +while building `validate_gcp_integration.sh`. + +### D3 — Event generation was not cross-machine deterministic + +The integration determinism gate failed on first live run: `event_id` used +`uuid.uuid4()` (backed by `os.urandom`), so identical batch identities +produced different bytes. Fix: UUIDs now derive from the seeded RNG +(`uuid.UUID(int=rng.getrandbits(128), version=4)`); regression test asserts +byte-identical regeneration to fresh paths. This is exactly the class of +defect the gate exists to catch — reproducible batches are what make smoke +runs and reruns comparable. + +### D4 — Error sanitizer leaked values following secret field names + +`sanitize_error_message` redacted the token `private_key` but left the value +after it (`{"private_key": "SECRET"}` → SECRET survived). Fix: patterns now +consume the field value; deployment audit tests assert `[REDACTED]` replaces +the value. + +### D5 — Composer runtime SA IAM race + +`gcloud iam service-accounts create` propagates asynchronously; immediate +role binding failed with "service account does not exist" on the first live +create. Fix: bounded retry with backoff in `manage_atlas_composer.sh`. + +## Live deployment failure ledger (every attempt is audited) + +The first Composer deployment to reach `SUCCESS` took seven attempts. Each +failure was recorded in `atlas_ops.deployments` with its `failure_stage`, +each exposed a real defect, and each fix is a focused commit on the delivery +branch: + +| deployment_id | sha | failure_stage | Root cause → fix | +|---|---|---|---| +| `atlas-dev-20260718T225900Z-dd7dd5d4` | `dd7dd5d` | `fetch_release` | D6 below | +| `atlas-dev-20260718T230144Z-2aeff26e` | `2aeff26` | `dag_parse` | D7 below | +| `atlas-dev-20260718T231842Z-83c0d137` | `83c0d13` | `dag_parse` | D8 below | +| `atlas-dev-20260718T233128Z-74732eee` | `74732ee` | `smoke_batch` | D9 below | +| `atlas-dev-20260719T001246Z-f9959cb6` | `f9959cb` | `smoke_batch` | D10 below (smoke DAG run itself succeeded; poller defect) | +| `atlas-dev-20260719T004112Z-37d4e6aa` | `37d4e6a` | `smoke_validation` | D11 below | +| `atlas-dev-20260719T005308Z-640cd786` | `640cd78` | — | **SUCCESS** — all 12 smoke checks PASS | + +### D6 — Checksum verification compared filenames, not digests + +The stored `.sha256` records the builder's local filename; the fetched object +is `atlas-bundle.tar.gz`, so `sha256sum -c` failed on every fetch. Fix: +compare digests directly in `fetch_and_verify_release`. + +### D7 — DAG imports assumed the repository layout, not Composer's + +Locally the DAG sits at the DagBag root, so `atlas_orchestration` and +`atlas` resolved implicitly. On Composer the DAG lives under +`dags/project_atlas/`, which Airflow 3's processor does not put on +`sys.path` — both imports failed. Fix: the DAG file adds its own directory +and `ATLAS_ROOT/src` to `sys.path` before package imports, and +`dags/.airflowignore` stops helper-package modules from being parsed as DAG +files. This is precisely the parity gap the live deploy stage exists to catch. + +### D8 — Stale Airflow import-error rows failed a healthy deployment + +Airflow retains the previous release's import errors until the processor +re-evaluates each file after GCS sync; the parse gate failed on the first +snapshot even though the DAG parsed cleanly seconds later. Fix: +`wait_for_dag_parse` keeps polling until the deadline and fails only if +errors persist. + +### D9 — Bundle stripped `dbt_packages` but the runtime never runs `dbt deps` + +Composer workers must not resolve packages from the network, yet the bundle +excluded `dbt_packages` as a build output — `dbt seed` aborted with +"0 package(s) installed". Fix: the bundle build vendors pinned packages +(`dbt deps` against `package-lock.yml`) into the staged tree and guards that +`dbt_utils` is present. + +### D10 — Smoke poller never matched Airflow 3 `dags state` output + +For runs triggered with `--conf`, Airflow 3 prints `success, {conf json}`; +the anchored regex `^(success|failed|…)$` matched nothing, so the poll spun +until timeout although the smoke run had succeeded. The stuck attempt was +finalized as `FAILED` (error_type `DeploymentTooling`) for audit honesty. +Fix: match the leading state token; log each poll iteration. + +### D11 — Deterministic bundles defeated `gcloud storage rsync` + +The reproducible tar pins every file mtime, so rsync's size+mtime comparison +skipped changed files whose size didn't change — the promoted runtime kept +the *previous* release's `release-manifest.json` and the `deployed_sha` +smoke check failed (correctly). Fix: promotion rsyncs with +`--checksums-only`. The same deployment also exposed the D10 regex bug in +`validate_atlas_deployment.sh`'s Airflow-state check, fixed the same way. + +## Recovery posture + +Every deploy-stage failure records `FAILED` with `failure_stage` in +`atlas_ops.deployments`, preserves the immutable bundle, and prints the exact +re-run command. See `ci-cd-runbook-sprint4.md` §9 for the failure table. diff --git a/docs/incident-report-sprint5.md b/docs/incident-report-sprint5.md new file mode 100644 index 0000000..850933b --- /dev/null +++ b/docs/incident-report-sprint5.md @@ -0,0 +1,149 @@ +# Atlas Incident Report — Sprint 5 Controlled Pipeline-Failure Drill + +Status: closed. This report documents the Sprint 5 Drill B incident: a +deliberately injected dbt test failure in the deployed `atlas_batch_pipeline`, +detected by the observability monitor, alerted through Cloud Monitoring, and +recovered by a clean rerun. Every timestamp below is UTC on 2026-07-19 and is +backed by artifacts in `docs/evidence-sprint5/`. + +## Summary + +| Field | Value | +| --- | --- | +| Incident title | Atlas pipeline failed — dbt build test failure (controlled drill) | +| Date | 2026-07-19 | +| Duration (failure → incident closed) | 06:41:47 → 07:03:11 (21 min 24 s) | +| Severity | CRITICAL (per `Atlas: pipeline failed` policy) | +| Affected component | `atlas_batch_pipeline` / `dbt_build` task | +| Detection source | `atlas_observability_monitor` → `custom.googleapis.com/atlas/monitor/check_status{check_name=latest_run_state}` | +| Alert policy | `Atlas: pipeline failed` (policy id `14992081806484518993`) | +| Violation id | `0.oaf4n04jvxx1` | +| Notification route | Cloud Monitoring email channel `projects/example-gcp-project/notificationChannels/6567861337166986657` ("Atlas Primary Operator (email)") | +| Operator | Primary operator (development ownership model, `on-call-model-sprint5.md`) | +| Data impact | None durable — failed batch never published a success marker; rerun replaced it idempotently | + +## Timeline (UTC, 2026-07-19) + +| Time | Event | Evidence | +| --- | --- | --- | +| 06:33:23 | Drill trigger: `atlas_batch_pipeline` run `drillb__pipeline-failure-20260719` with conf `dbt_test_failure: true` (batch `atlas-drillb-20260719`, run `atlas-drillb-20260719-run`) | Airflow API dag-run record | +| 06:33:54 | `atlas_ops.pipeline_runs` row created, status RUNNING | `pipeline_runs` | +| 06:39:00 | `dbt_build` task attempt 1 STARTED | `task-events-drills.json` | +| ~06:41 | `dbt_build` FAILED — injected dbt test failure (`inject_failure` var) | `task-events-drills.json` | +| 06:41:47 | Finalizer recorded pipeline run FAILED; `validate_warehouse` and `publish_success_marker` recorded UPSTREAM_FAILED | `pipeline_runs`, `task-events-drills.json` | +| 06:44:05 | Monitor evaluation: `latest_run_state` = FAIL / CRITICAL (run `drillb-monitor-eval-20260719`) | `monitor-evaluations.json` | +| 06:44:09 | `check_status{check_name=latest_run_state}` = 2 (FAIL) published | `metric-timeseries-summary.json` | +| 06:46:19 | Incident opened: `Atlas: pipeline failed`, violation `0.oaf4n04jvxx1`; notification dispatched to the email channel | `incident-events.json` | +| 06:48:56 | Recovery rerun `drillb__recovery-20260719` triggered — same `batch_id`, no injection (tests idempotent replacement) | Airflow API | +| 06:58:00 | Recovery run SUCCESS (`atlas-drillb-20260719-recovery-run`) | `pipeline_runs` | +| 07:01:17 | Monitor re-evaluation: `latest_run_state` = PASS | `monitor-evaluations.json` | +| 07:03:11 | Incident auto-resolved (`ViolationAutoResolve`) | `incident-events.json` | + +Detection latency (pipeline FAILED recorded → incident open): **4 min 32 s** +(monitor was manually triggered for the drill; the scheduled 30-minute cadence +bounds worst-case detection at ~35 minutes). +Recovery latency (recovery SUCCESS → incident closed): **5 min 11 s**. + +## Technical root cause + +Observation: `dbt_build` executed `dbt build --vars {"validated_batch_id": +"atlas-drillb-20260719", "inject_failure": true}`. The `inject_failure` var +activates the controlled failing dbt test retained from Sprint 3 for exactly +this purpose. The dbt process exited non-zero; the step runner recorded a +FAILED task event and re-raised, Airflow marked the task failed (retries are +not configured for deliberate quality-gate failures), and downstream tasks +went to `upstream_failed`. + +Inference: this is the intended behavior of the delivery controls — a failed +warehouse quality gate must stop publication. No defect in the pipeline +itself. + +Contributing factor (real defect found and fixed during this drill window): +the first `telemetry_completeness` implementation evaluated the newest +`pipeline_runs` row even while it was still RUNNING, which opened a +false-positive `Atlas: telemetry incomplete` incident at 06:05:51 (violation +`0.oaf3pqgsimvt`, auto-resolved 06:33:00). Fixed in commit `2109310` (check +now only scores terminal runs) and regression-covered. + +## Customer / data impact + +None durable. Observation: the failed run's batch (`atlas-drillb-20260719`) +loaded raw rows but never passed `validate_warehouse` and never wrote a +success marker; marts never exposed the batch as validated. The recovery +rerun reused the same `batch_id`, replacing the batch idempotently +(`no_duplicate_load` semantics from Sprint 3/4 apply). `quality_results` for +the recovery run recorded all reconciliation checks PASS. + +## Operator response (runbook execution) + +Followed `observability-runbook-sprint5.md` → "Atlas: pipeline failed": + +1. First query — latest run state and failed task from + `atlas_ops.pipeline_runs` / `atlas_ops.task_events`: identified `dbt_build` + attempt 1 FAILED. Worked as documented. +2. First log filter — `jsonPayload.pipeline_run_id="atlas-drillb-20260719-run"` + on logName `atlas-events`: returned 20 correlated structured events + covering every task lifecycle transition + (`drillb-correlated-logs.json`). Worked as documented. +3. Containment — no action needed: failure propagation had already blocked + publication. +4. Recovery — rerun without the injected defect per the runbook's "when a + rerun is safe" rule (same batch id ⇒ idempotent replacement). Worked. +5. Verification — recovery SUCCESS in `pipeline_runs`, quality results PASS, + monitor PASS, incident auto-closed. + +## What worked + +- Task-level audit (`task_events`) captured STARTED / FAILED / + UPSTREAM_FAILED with correct grain, including the finalizer's backfill of + never-scheduled tasks. +- Structured logs correlated the whole run by `pipeline_run_id` in one query, + in both the `atlas-observability` bucket and the `atlas_logs` linked + dataset. +- Monitor → metric → alert → email chain fired end to end with no manual + glue. +- Idempotent rerun recovery behaved exactly as the Sprint 3 design promised. +- Incident auto-closed on recovery; no manual reset was needed. + +## What failed / gaps observed + +1. (Fixed) `telemetry_completeness` false positive on in-flight runs — fix in + `2109310` with regression test + (`test_observability_monitor.py`). +2. (Platform, open) Composer 3 `build.13` exports **no** Airflow component + logs (worker/scheduler/task streams) to the customer project — reproduced + from Sprint 4. Diagnosis evidence: zero `airflow-*` log names in any + bucket including `_Default`; a manual `entries.write` to the identical + logName/resource succeeds and routes correctly; Composer's own task-log + reader reports "Logs not found"; environment restart did not recover it. + Mitigation shipped in `2109310`: contract events are written directly to + the Cloud Logging API (`atlas-events`) when + `ATLAS_LOG_TO_CLOUD_LOGGING=true`, so Atlas telemetry no longer depends on + the broken export path. Raw Airflow stdout remains unavailable and is + documented as an unresolved platform limitation. +3. `task_events.FAILED` rows have NULL `completed_at`/`duration_ms` (the + failure callback does not receive reliable timing). Cosmetic; noted as a + Sprint 6 cleanup candidate. +4. Email delivery latency was not independently measurable (Cloud Monitoring + does not expose per-notification delivery logs for email channels); + delivery is evidenced by channel configuration + incident dispatch and by + the recipient's mailbox. + +## Corrective actions + +| Action | Type | Status | Owner | +| --- | --- | --- | --- | +| Only score terminal runs in `telemetry_completeness` | code + regression test | done (`2109310`) | primary operator | +| Direct Cloud Logging emission for contract events | code + tests + env var on atlas-dev | done (`2109310`) | primary operator | +| Grant `roles/bigquery.resourceViewer` to runtime SA for the cost check (403 found live) | IAM + bootstrap script update | done | primary operator | +| Record Composer log-export defect as known platform limitation; re-test on next Composer build upgrade | documentation | done (this report; validation report) | primary operator | +| Populate `completed_at`/`duration_ms` on FAILED task events | code | deferred to Sprint 6 | primary operator | + +## Speculation (explicitly labeled) + +The Composer log-export failure is deterministic across two environments and +two days on `composer-3-airflow-3.1.7-build.13`, while documentation states +Gen 3 streams logs to Cloud Logging by default. It plausibly affects this +very new build (released 2026-07-07) more broadly; we cannot verify Google's +internal log-agent state from the customer project. This is speculation, not +observation. diff --git a/docs/lineage-impact-sprint7.md b/docs/lineage-impact-sprint7.md new file mode 100644 index 0000000..a8db139 --- /dev/null +++ b/docs/lineage-impact-sprint7.md @@ -0,0 +1,63 @@ +# Atlas Lineage & Consumer Impact (Sprint 7) + +Lineage and consumer-impact analysis derived entirely from **repository +artifacts** — no graph database, metadata service, or web UI (out of scope). + +## Sources of truth + +- dbt model SQL `ref()` / `source()` calls (the model DAG). +- `governance/consumers.yml` (internal downstream consumers). +- `governance/generated/catalog.json` (owners, contracts, runbooks). + +## Lineage + +```bash +python -m atlas.governance.lineage --output governance/generated/lineage.json +``` + +Produces a machine-readable graph (`nodes` with `upstream`/`downstream`, +`edge_count`). Current graph: 26 nodes / 29 edges, covering +`atlas_raw.events → stg_events → int_event_classification → +int_accepted_events → fct_events → {dim_users, mart_daily_event_metrics} → +consumers`. `gate_lineage_impact` fails on drift or if the source→mart chain +breaks. + +## Consumer impact + +```bash +python -m atlas.governance.impact --asset fct_events \ + --change governance/changes/CHG-....yml --output-dir /tmp/impact +``` + +Emits `impact.json` + `impact.md` with: + +1. source-to-mart position (upstream + downstream), +2. direct downstream assets, +3. transitive downstream assets, +4. affected tests / property files, +5. affected contracts (asset → contract_version), +6. affected consumers and owners to notify, +7. runbooks involved, +8. (with `--change`) the change's compatibility class + whether approval is + present. + +### Example (fct_events) + +- Direct downstream: `mart_daily_event_metrics`, `warehouse_reconciliation`, + `atlas_observability_monitor`. +- Transitive: adds `analytics_mart_readers`. +- Owners to notify: `atlas-data-eng`, `atlas-analytics`. +- Affected contracts: `fct_events 1.0`, `mart_daily_event_metrics 1.0`. + +## How it plugs into change management + +A BREAKING/CONDITIONALLY_COMPATIBLE change record (ADR-017) must reference the +impact report so reviewers see exactly which consumers require migration before +approval. This is the "consumer-impact analysis" that PROHIBITED changes bypass. + +## Honest limitations + +- Consumers are limited to what `consumers.yml` declares — external or + undeclared consumers are not discovered. +- Lineage covers known Atlas assets (dbt DAG + registered non-dbt assets); it is + not full warehouse-wide lineage. diff --git a/docs/model-catalog-sprint2.md b/docs/model-catalog-sprint2.md new file mode 100644 index 0000000..fbb0277 --- /dev/null +++ b/docs/model-catalog-sprint2.md @@ -0,0 +1,68 @@ +# Project Atlas Sprint 2 Model Catalog + +Validated run scope: `atlas-20260714T163527Z-19a0e4f6` + +## Sources + +| Name | Relation | Grain | +| --- | --- | --- | +| `atlas_raw.events` | `{project}.atlas_raw.events` | physical ingest row | + +Freshness: warn after 24h, error after 48h on `ingested_at`. + +## Seeds + +| Model | Grain | Notes | +| --- | --- | --- | +| `valid_country_codes` | `country_code` | Ten active ISO-style codes from `config/anomaly_profile.yaml` | + +## Staging + +| Model | Materialization | Grain | Notes | +| --- | --- | --- | --- | +| `stg_events` | view | physical row | Normalized types, lineage, four temporal flags | + +## Intermediate + +| Model | Materialization | Grain | Notes | +| --- | --- | --- | --- | +| `int_event_classification` | table | physical row | Duplicate rank + terminal rejection reason | +| `int_accepted_events` | view | accepted canonical row | One accepted row per `event_id` | +| `int_rejected_events` | table (`atlas_quarantine`) | rejected physical row | All blocking defects | + +## Core + +| Model | Materialization | Grain | Notes | +| --- | --- | --- | --- | +| `dim_users` | table | `user_id` | First/last event timestamps from accepted events | +| `dim_countries` | table | `country_code` | Seed-backed reference | +| `fct_events` | incremental merge | `event_id` | Accepted events with warning flags retained | + +## Marts + +| Model | Materialization | Grain | Measures | +| --- | --- | --- | --- | +| `mart_daily_event_metrics` | table | `event_date, event_name, country_code, platform` | `event_count`, `distinct_user_count`, warning counts | + +## Singular tests + +| Test | Purpose | +| --- | --- | +| `assert_source_anomaly_profile` | Exact anomaly counts on validated run | +| `assert_raw_classification_reconciliation` | Raw physical rows = classification rows | +| `assert_fact_rejected_reconciliation` | Raw = accepted + rejected for validated run | +| `assert_mart_fact_reconciliation` | Mart totals = fact row count | + +## Expected validated-run anomaly counts + +| Measure | Expected | +| --- | ---: | +| duplicate_extra | 50 | +| null_user_id physical rows | 500 | +| invalid_country_code rejections | 200 | +| future_dated | 150 | +| event_time_late_arriving | 0 | +| backdated_event_date warnings | 300 | +| date/timestamp mismatch warnings | 300 | + +Accepted canonical rows and rejected physical rows must sum to 50,000 raw rows. diff --git a/docs/observability-runbook-sprint5.md b/docs/observability-runbook-sprint5.md new file mode 100644 index 0000000..5183fa0 --- /dev/null +++ b/docs/observability-runbook-sprint5.md @@ -0,0 +1,213 @@ +# Atlas Observability Runbook (Sprint 5) + +Operator procedures for every Atlas alert. Each alert policy links to its +section anchor here. Shared context first, then one section per alert. + +Primary operator: the primary operator. Escalation: repository owner / +designated reviewer (see `on-call-model-sprint5.md`). + +## The five questions, answered generically + +| Question | Where to look | +|---|---| +| Did the pipeline run? | `atlas_ops.pipeline_runs` (latest row), dashboard top row | +| Is the data correct? | `atlas_ops.quality_results` for the run's `pipeline_run_id` | +| Who was alerted? | Cloud Monitoring incident → policy → channel "Atlas Primary Operator (email)" | +| How is it recovered? | Per-alert section below; usually rerun via `scripts/run_airflow_sprint3.sh` or redeploy/rollback via Sprint 4 workflows | +| How is recurrence prevented? | Convert the root cause into a regression test, alert change, or ADR amendment; record in the incident report | + +## Shared first moves (any Atlas incident) + +1. Open the **Atlas Operations** dashboard; the top row shows latest run, + freshness age, deployment state, and Composer health at a glance. +2. Identify the run: + +```sql +SELECT pipeline_run_id, batch_id, status, started_at, completed_at, + rows_loaded, rows_accepted, rows_rejected +FROM `example-gcp-project.atlas_ops.pipeline_runs` +ORDER BY started_at DESC LIMIT 5; +``` + +3. Pull correlated logs (Logs Explorer, bucket `atlas-observability`, view + `atlas-runtime`): `jsonPayload.pipeline_run_id=""`. More filters in + `observability/queries/log-filters.md`. +4. Preserve evidence **before** changing anything: incident ID, evaluation + rows, log query links, relevant audit rows. + +--- + +## Alert: atlas pipeline failed + +- **Meaning**: latest `pipeline_runs` row is FAILED. Severity: critical. +- **Likely causes**: dbt test failure (including deliberate injection), GCP + permission loss, BigQuery quota, task crash, upstream generation defect. +- **First query**: shared query above; then task diagnosis: + +```sql +SELECT task_id, attempt_number, event_type, status, error_type, error_message +FROM `example-gcp-project.atlas_ops.task_events` +WHERE pipeline_run_id = '' ORDER BY task_id, attempt_number; +``` + +- **First log filter**: `jsonPayload.pipeline_run_id="" severity>=ERROR` +- **Containment**: nothing automatic mutates on failure; the DAG chain stops + before `publish_success_marker`. Do not delete data. +- **Recovery**: fix the root cause, then rerun the batch + (`scripts/run_airflow_sprint3.sh` or Airflow UI trigger with the same + `batch_id` conf for an idempotent rerun — raw loading is create-only per + batch and dbt is batch-scoped). +- **When NOT to rerun**: if the failure is in `validate_warehouse` + reconciliation, diagnose first — rerunning on top of inconsistent state + reproduces the failure and wastes evidence freshness. +- **Backfill safety**: backfills are safe for past `processing_date`s + (Sprint 3 semantics); never backfill over a batch under investigation. +- **Verification**: rerun reaches SUCCESS, `quality_results` all PASS, + incident auto-closes within ~40 min (next monitor cycle + auto-close). +- **Prevention**: add the failure mode to dbt tests or warehouse checks. + +## Alert: atlas data stale + +- **Meaning**: age since last SUCCESS exceeded 50 h. Severity: critical. +- **Likely causes**: DAG paused unintentionally, scheduler dead, repeated + run failures (check the pipeline-failed alert first), Composer deleted + without disabling monitoring. +- **First query**: freshness evaluation history: + +```sql +SELECT evaluated_at, status, observed_value, threshold +FROM `example-gcp-project.atlas_ops.monitor_evaluations` +WHERE check_name = 'freshness' ORDER BY evaluated_at DESC LIMIT 10; +``` + +- **First log filter**: `resource.type="cloud_composer_environment" log_id("airflow-scheduler") severity>=ERROR` +- **Containment/recovery**: unpause the DAG or trigger a manual run; if the + environment was intentionally torn down, set `monitoring_enabled: false` + and disable this policy instead of chasing a ghost. +- **Verification**: next monitor cycle publishes PASS; incident closes. +- **Prevention**: the teardown checklist (Phase 15) disables absence-prone + alerts before deletion. + +## Alert: atlas reconciliation failed + +- **Meaning**: FAIL rows exist in `quality_results` for the latest run. +- **Likely causes**: duplicate loads, dbt model regression, partial batch, + manual mutation of warehouse tables. +- **First query**: + +```sql +SELECT check_name, status, observed_value, expected_value, details_json +FROM `example-gcp-project.atlas_ops.quality_results` +WHERE pipeline_run_id = '' AND status = 'FAIL'; +``` + +- **Audit tables**: `quality_results`, then the specific warehouse tables + named by the failing check. +- **Recovery**: never edit warehouse rows by hand. Fix the transformation + and rerun the batch; dbt rebuilds are idempotent per batch. +- **When a backfill is safe**: only after the failing check passes on a + fresh run of the current release. +- **Prevention**: promote the broken invariant into a dbt test. + +## Alert: atlas volume deviation + +- **Meaning**: latest raw rows deviate ≥ 80 % from the 7-run baseline. +- **Likely causes**: generator config change, truncated upload, duplicate + batch, intentional batch-size change without threshold retuning. +- **First query**: `SELECT rows_loaded FROM ... pipeline_runs ORDER BY started_at DESC LIMIT 8;` +- **Recovery**: if the change is legitimate, update `volume` thresholds in + `config/observability.yaml` with the new baseline (documented commit); if + not, treat as a pipeline defect and rerun after diagnosis. +- **Prevention**: batch-size changes must land with a threshold update. + +## Alert: atlas schema drift + +- **Meaning**: live INFORMATION_SCHEMA diverges from the governed manifest + with a BREAKING classification (removed/renamed field, type change, + required-field loss, partition change). +- **First command**: `PYTHONPATH=src python -m atlas.observability.schema_drift --check` +- **Audit tables**: `schema_migrations` (was there an unrecorded change?). +- **Containment**: stop deployments (`atlas-deploy` workflow) until resolved. +- **Recovery**: restore the contract via an additive migration, or — for an + approved intentional change — regenerate the manifest + (`--generate`, review, commit) and ship it with the migration. +- **When not to rerun**: pipeline reruns cannot fix schema drift; do not + rerun to "see if it clears". +- **Prevention**: schema changes only via the migration ledger + manifest + regeneration in the same PR (CI gate checks the manifest parses). + +## Alert: atlas deployment failed + +- **Meaning**: latest `deployments` row is FAILED/ROLLBACK_FAILED. +- **First query**: + +```sql +SELECT deployment_id, status, failure_stage, error_summary, git_sha +FROM `example-gcp-project.atlas_ops.deployments` +ORDER BY started_at DESC LIMIT 3; +``` + +- **First log filter**: `jsonPayload.deployment_id=""` +- **Recovery**: per Sprint 4 runbook (`ci-cd-runbook-sprint4.md`) — + `failure_stage` names the failed stage; fix and redeploy, or roll back to + the previous validated release (`scripts/rollback_atlas.sh`). +- **Evidence**: keep the FAILED row and workflow logs; they are the audit. + +## Alert: atlas rollback failed + +- **Meaning**: ROLLBACK_FAILED — the safety net itself failed. Severity: + critical, highest urgency. +- **First moves**: same queries as deployment failed; additionally verify + what is actually running: `data/current/release-manifest.json` + in the Composer bucket vs `atlas_ops.deployments`. +- **Containment**: freeze all deployment activity; the environment state is + now unverified. +- **Recovery**: manual re-promotion of the last SUCCESS release bundle + (verify checksums first via `lib_atlas_deploy.sh` helpers), then a manual + smoke batch, then a corrected `deployments` record. +- **Escalation**: this is the one alert where the escalation contact should + be engaged immediately if the first recovery attempt fails. + +## Alert: atlas composer unhealthy + +- **Meaning**: native `environment/healthy` fraction < 0.5 for 15 min. +- **Likely causes**: scheduler crash-loop, worker OOM, GKE node pressure, + or (expected) creation/deletion transitions. +- **First look**: Composer environment page; then + `resource.type="cloud_composer_environment" severity>=ERROR` logs. +- **Containment**: do not deploy onto an unhealthy environment. +- **Recovery**: Composer 3 self-heals most component failures; if unhealthy + persists > 1 h, capture logs and recreate the ephemeral environment + (`scripts/manage_atlas_composer.sh`). +- **Teardown note**: DISABLE this policy before intentional deletion. + +## Alert: atlas cost anomaly + +- **Meaning**: Atlas-attributed bytes billed in 24 h ≥ 10× the 7-day daily + baseline and above the 1 GiB floor. Severity: warning. +- **First query**: `observability/queries/bigquery_cost.sql` queries 1, 4, + and 5 (daily usage, by component, expensive jobs). +- **Likely causes**: unbounded monitor query regression, repeated backfills, + a new query pattern missing partition filters. +- **Containment**: pause the offending component (monitor DAG or pipeline) + if a runaway query loop is confirmed. +- **Recovery**: fix the query pattern; verify the next window's bytes fall + back under threshold. +- **Never**: run a large query "to test" this alert — use + `manage_atlas_alerts.sh test cost_anomaly` (synthetic signal). + +## Alert: atlas telemetry incomplete + +- **Meaning**: expected tasks lack terminal `task_events` rows for the + latest run — observability is degraded even though the data may be fine. +- **First query**: the completeness SQL in + `observability/queries/log-filters.md` §8. +- **First log filter**: `jsonPayload.event_type="task_telemetry_write_failed"` +- **Likely causes**: BigQuery audit-write outage, IAM regression on the + runtime SA, a task crash before the telemetry wrapper ran, or runs + predating Sprint 5 telemetry. +- **Recovery**: fix the write path; telemetry backfills automatically on the + next run (per-run grain). Do not fabricate historical task events. +- **Important**: data correctness is judged by `quality_results`, not by + telemetry presence — check reconciliation before treating this as a data + incident. diff --git a/docs/on-call-model-sprint5.md b/docs/on-call-model-sprint5.md new file mode 100644 index 0000000..deb1e00 --- /dev/null +++ b/docs/on-call-model-sprint5.md @@ -0,0 +1,53 @@ +# Atlas On-Call and Ownership Model (Sprint 5) + +This is a development-project ownership model, not an organizational on-call +rotation. No 24/7 coverage, paging SLA, or follow-the-sun handoff is claimed. + +## Ownership + +| Role | Who | Responsibilities | +|---|---|---| +| Primary operator | the primary operator | Receives all alert emails (verified channel), acknowledges incidents, executes runbooks, owns incident reports | +| Escalation | Repository owner / designated reviewer | Engaged when the first recovery attempt fails, for `rollback failed` incidents immediately, and for any IAM or destructive decision | +| External escalation | none configured | Only when an approved real recipient/integration is added (PagerDuty/Slack are out of Sprint 5 scope) | + +## Response expectations (development-grade) + +- Alerts route to one verified email channel; response happens on a + best-effort basis during active development windows. +- Critical incidents (`pipeline failed`, `reconciliation failed`, + `rollback failed`, `breaking schema drift`) take priority over feature + work when the environment is active. +- Between acceptance windows the Composer environment is intentionally + deleted and `monitoring_enabled: false`; no alert response is expected and + absence-prone policies are disabled per the teardown checklist. + +## Escalation triggers + +1. First recovery attempt failed or the runbook does not match reality. +2. Any `rollback failed` incident (immediately). +3. Suspected credential exposure or IAM regression (also see + `security-review-sprint5.md`). +4. Any action that would delete data, move release tags, or reverse a + migration — these always require explicit human approval. + +## Incident lifecycle + +detect (Cloud Monitoring incident) → acknowledge (email received, incident +noted) → diagnose (runbook first-moves) → contain → recover → verify +(monitor cycle returns PASS, incident closes) → document (incident report +with observation/inference/speculation separated) → prevent (regression +test, alert change, ADR amendment, or runbook fix — every meaningful defect +becomes a reusable control). + +## SLO and error-budget candidates (recorded, not committed) + +These are candidates for a future production posture, backed only by the +current synthetic workload: + +- Pipeline success rate per 30-day window (candidate SLO 99 %). +- Freshness: successful batch within 26 h (warn) / 50 h (fail). +- Deployment success rate and rollback MTTR. + +They remain "initial operational thresholds" (see `config/observability.yaml`) +until real usage exists. diff --git a/docs/performance-review-sprint7.md b/docs/performance-review-sprint7.md new file mode 100644 index 0000000..79c7ef3 --- /dev/null +++ b/docs/performance-review-sprint7.md @@ -0,0 +1,76 @@ +# Atlas BigQuery Performance Review (Sprint 7) + +Measured before optimizing. Methodology and controls in ADR-020. The suite is +`scripts/run_performance_suite.sh` over `observability/performance/queries/`. + +## Methodology + +- **Dry-run first** (bills $0) for every representative query → bytes processed. +- **Bounded execution** (gated) under a per-query `maximum_bytes_billed` ceiling + (1 GiB) and a cumulative suite ceiling (5 GiB), with run labels + (`atlas_component=perf_suite`), capturing bytes billed, slot-ms, elapsed, + output rows, cache-hit, and a correctness checksum. +- Correctness verified via row-count/aggregate checksums before and after any + change. + +## Dry-run baseline (live, 2026-07-19, $0) + +`observability/performance/results/baseline-dryrun.json`: + +| # | Query | Bytes processed (est.) | +| --- | --- | --- | +| 1 | raw batch lookup (partition filter) | 2,315,824 | +| 2 | batch classification (1 partition) | 745,937 | +| 3 | accepted/rejected reconciliation | 795,136 | +| 4 | fact build scan (3-day window) | 2,255,810 | +| 5 | mart aggregation (monthly) | 44,864 | +| 6 | freshness | 4,700,784 | +| 7 | operational audit | 216 | +| 8 | cost monitor (INFORMATION_SCHEMA.JOBS, 7d) | 34,428,633 | +| 9 | lineage/schema metadata | 10,485,760 | + +All queries are far below the 1 GiB per-query ceiling. + +## Partition pruning evidence + +- Bounded raw lookup (`where event_date = …`): **2,315,824 bytes**. +- Unbounded full scan of the same table (no partition filter): **12,659,283 + bytes** (dry-run). + +The bounded query scans ~18% of the unbounded scan → **partition pruning is +working** on `atlas_raw.events` (partitioned by `event_date`). `fct_events` is +likewise partitioned by `event_date` and clustered by `event_name, +country_code`, exercised by query 4. + +## Performance changes applied (Phase 11) + +**None warranted — evidence-backed "no material change" result.** The warehouse +is already partitioned + clustered, all representative queries are bounded and +inexpensive at this scale (~50k rows/batch), and correctness is preserved. The +highest-leverage improvement is *preventing regressions*, which Sprint 7 adds as +the required-partition-filter cost guard rather than a table redesign. Per +ADR-020, we do not optimize merely to produce a percentage. + +Comparisons considered and their verdicts (from the baseline evidence): + +| Comparison | Verdict | +| --- | --- | +| partitioned vs unbounded raw scan | partitioned wins (2.3 MB vs 12.7 MB) — keep + enforce filter | +| clustered fact scan (query 4) | already clustered; bounded window is cheap | +| mart (table) vs recompute from fact | mart is tiny (44 KB read) — keep materialized | +| INFORMATION_SCHEMA cost monitor | bounded to 7 days — acceptable | + +## Blocked completion gate + +- **Gate:** full **executed** metrics (bytes billed, slot-ms, elapsed) via + `run_performance_suite.sh --execute`. +- **Blocking approval:** `ATLAS_APPROVE_PERFORMANCE_TESTS=true` (+ optional + `ATLAS_MAX_PERFORMANCE_TEST_BYTES`) — not set in this environment. +- **Status:** dry-run baseline (bytes processed) is complete and is sufficient + to conclude no optimization is warranted; executed slot/elapsed metrics are + pending approval. The suite runner is ready and enforces the byte ceilings. + +## Honest limitations + +- ~50k rows/batch: these are engineering demonstrations, not production-scale + benchmarks. No production-scale performance is claimed. diff --git a/docs/preflight-sprint3.md b/docs/preflight-sprint3.md new file mode 100644 index 0000000..991bcf7 --- /dev/null +++ b/docs/preflight-sprint3.md @@ -0,0 +1,71 @@ +# Sprint 3 Preflight Report + +**Date:** 2026-07-14 +**Branch:** `cursor/atlas-sprint-3-airflow-orchestration-3660` +**Repository:** `Atlas-GCP-Build/project-atlas` + +## Version verification + +| Component | Version | Notes | +|-----------|---------|-------| +| Airflow core | 3.1.7 | Composer-parity pin | +| Google provider | 20.0.0 | Composer-parity pin | +| Standard provider | 1.12.1 | Composer-parity pin | +| Composer image | `composer-3-airflow-3.1.7-build.12` | Verified 2026-07-14 | +| Local Python | 3.12.3 | Cloud Agent runtime | +| Composer Python | 3.11.8 | Documented parity gap | + +Revisit ADR-005 if the Composer image is no longer available. + +## Security scan + +Tracked-secret checks (no credentials in git): + +```bash +git grep -E '(BEGIN PRIVATE KEY|AIza[0-9A-Za-z\-_]{35}|service-account.*\.json)' -- ':!*.md' || true +grep -r 'credentials/' project-atlas --include='*.py' --include='*.yaml' || true +``` + +Results: no tracked private keys or service account JSON files. `.env`, `.gcp/`, and +`credentials/` remain gitignored. + +## CLI inventory + +| Script | Sprint 3 parameters | +|--------|---------------------| +| `generate_events.py` | `--processing-date`, `--batch-id`, `--pipeline-run-id`, `--seed` | +| `upload_events.py` | `--batch-id`, `--expected-checksum`, `--fail-once` | +| `load_events.py` | `--batch-id`, `--expected-row-count` | +| `validate_events.py` | `--batch-id`, `--processing-date`, `--mode` | +| `run_atlas_step.sh` | Dispatcher for all steps with structured context logs | +| `run_airflow_sprint3.sh` | Trigger and poll DAG runs | + +## Reusable interfaces + +- `atlas.batch.context` — batch and pipeline run identity +- `atlas.batch.manifest` — artifact checksums and idempotent reuse +- `atlas.ops.resources` — `atlas_ops` DDL +- `atlas.ops.audit` — MERGE upsert audit rows +- `atlas.loader.bigquery` — batch-scoped load idempotency +- `atlas.validation.checks` — run-scoped and batch-scoped validation + +## Idempotency gaps addressed in Sprint 3 + +| Layer | Sprint 1/2 gap | Sprint 3 behavior | +|-------|----------------|-------------------| +| Generator | Wall-clock dates | Processing-date-based generation with manifest reuse | +| GCS | run_id paths only | `batch_id=` paths with checksum metadata | +| Raw load | pipeline_run_id skip | batch_id count evaluation (0/load, exact/skip, partial-fail, excess-fail) | +| dbt | run_id tests only | batch_id lineage + scoped reconciliation | +| Audit | None | `atlas_ops.pipeline_runs` one row per execution | + +## Composer deployment contract + +| Asset | Local path | Composer path | +|-------|------------|---------------| +| DAGs + parse helpers | `dags/` | `/home/airflow/gcs/dags/project_atlas/` | +| Scripts + dbt | `` | `/home/airflow/gcs/data/` | +| `ATLAS_ROOT` | repo `` | `/home/airflow/gcs/data/project-atlas` | +| `DBT_PROJECT_DIR` | `dbt/atlas_dbt/` | `/home/airflow/gcs/data/dbt/atlas_dbt` | + +No DAG assumes `~/Atlas-GCP-Build/project-atlas`. Generated artifacts never write under Composer `dags/` or `plugins/`. diff --git a/docs/preflight-sprint4.md b/docs/preflight-sprint4.md new file mode 100644 index 0000000..7779218 --- /dev/null +++ b/docs/preflight-sprint4.md @@ -0,0 +1,88 @@ +# Sprint 4 Preflight Report — CI/CD, Secure GCP Delivery, Composer, Rollback + +Compiled 2026-07-18 (updated as delivery progressed). All facts below were +verified against the live repository and GCP project, not assumed. + +## Repository + +| Item | Value | +|---|---| +| Repository | `YOUR_GITHUB_OWNER/YOUR_REPOSITORY` (private) | +| Baseline main SHA (Sprint 3) | `8aa1d7a1849d4226e30b367ae490dcf1df936f9a` — verified | +| Sprint tags | `atlas-sprint-1-complete`, `atlas-sprint-2-complete`, `atlas-sprint-3-complete` (→ `8aa1d7a`, added during Sprint 4 hygiene) | +| Sprint 4 PRs | #14 (Phases 0–3, merged as squash `21d54ed`), #15 (Phase 16 gate demos, closed by design), #16 (delivery continuation, this branch) | +| Pre-existing workflows | none for Atlas before Sprint 4; `atlas-ci.yml` added in PR #14 | +| Branch protection | **not readable/writable** — `gh api .../branches/main/protection` returns 403 on the current plan | +| GitHub plan | Free. Required reviewers on Environments and branch protection API are unavailable for private repos → governance fallback documented in ADR-010 and `ci-cd-governance-sprint4.md` | + +## Validation toolchain (verified versions) + +| Tool | Version | +|---|---| +| Python | 3.12.3 | +| uv | 0.11.29 | +| dbt-core / dbt-bigquery | 1.11.12 / 1.11.3 (pinned; 1.12 deliberately not adopted) | +| Apache Airflow | 3.1.7 + providers google 20.0.0, standard 1.12.1 | +| gcloud SDK | 576.0.0 (bq 2.1.34) | +| ruff / mypy / yamllint / shellcheck-py / pytest | pinned in `requirements-ci.txt` | + +## GCP state + +| Item | Value | +|---|---| +| Project | `example-gcp-project` (number `123456789012`) | +| Agent identity | `service1-831@example-gcp-project.iam.gserviceaccount.com` (project Owner — can create IAM/WIF/Composer; used only from the Cursor environment) | +| BigQuery datasets | `atlas_raw`, `atlas_{staging,intermediate,core,marts,quarantine}`, `atlas_ops` — location US | +| Buckets (pre-Sprint 4) | `atlas-raw-events-example-gcp-project` (no uniform bucket-level access — cannot carry IAM conditions) | +| Buckets (created Sprint 4) | `atlas-deployments-…` (versioned, uniform), `atlas-ci-…` (uniform, 7-day TTL) | +| Composer environments | none before Sprint 4; `composer-3-airflow-3.1.7-build.12` (ADR-005 target) no longer offered — owner approved `build.13` | +| WIF pools before Sprint 4 | none | + +## Security preflight + +- Tracked-file credential scan: clean (gate `secret_scan` in `validate_ci.sh`). +- Untracked scan: agent ADC material lives outside the repo (`/tmp`); nothing + under `` matches key patterns. +- No long-lived GitHub Actions secrets exist; PR CI is credentialless and the + WIF design (ADR-009) keeps it that way. +- Workflow permissions: all Atlas workflows declare least privilege + (`contents: read`, plus `id-token: write` only for trusted WIF jobs). + +## Cost review + +- Composer 3 small: ≈ USD 0.35–0.50/hour while running (~USD 300/month if left + alive). Owner decision: **hard no** on persistent environments — create for + evidence capture, then delete (`manage_atlas_composer.sh delete`, ADR-010). +- Integration tests: ephemeral `atlas_ci_` datasets + CI bucket objects + (auto-TTL 7 days); per-run BigQuery cost is cents (50k-row batches). +- Deployment bundles: single-digit MB per release in GCS — negligible. +- Scale-to-zero: everything except retained BigQuery datasets and GCS bundles. + +## Approval variables (owner grants on record) + +| Variable | State | +|---|---| +| `ATLAS_APPROVE_PROVISION` | granted ("Build is approved") | +| `ATLAS_APPROVE_IAM` | granted ("IAM is approved at every stage") | +| `ATLAS_APPROVE_COMPOSER_CREATE` | granted for ephemeral evidence capture only | +| `ATLAS_APPROVE_DEPLOY` | granted | +| `ATLAS_APPROVE_ROLLBACK_TEST` | granted (rollback is a required completion gate) | + +## Code interfaces (as found at Sprint 3 baseline) + +- CI/test entry points: fragmented (`test_airflow_sprint3.sh` suppressed + failures with `|| true`; `run_airflow_sprint3.sh` did not poll; + `validate_warehouse` was a hardcoded PASS) — all fixed in Phase 1 with + regression tests. +- No deploy scripts, no migration ledger, no deployment audit — built in + Phases 8–11. +- dbt targets: `atlas*` datasets via `ATLAS_DBT_DATASET`; raw via + `ATLAS_BQ_DATASET` (env override added to `settings.py` in Phase 7 so + isolation requires no parallel code path). +- Composer path mapping (ADR-005, implemented Phase 12): DAGs → + `/dags/project_atlas/`, runtime → + `/data/current/`, immutable releases → + `gs://atlas-deployments-…/atlas/releases//`. +- Rollback limitation: BigQuery schema migrations are additive-only and never + auto-reversed; runtime rollback requires manifest schema compatibility + (ADR-010). diff --git a/docs/preflight-sprint5.md b/docs/preflight-sprint5.md new file mode 100644 index 0000000..b86d7d7 --- /dev/null +++ b/docs/preflight-sprint5.md @@ -0,0 +1,169 @@ +# Sprint 5 Preflight — Observability, Alerting, and Incident Readiness + +Inspection completed 2026-07-19 ~02:45 UTC, before any Sprint 5 resource +creation. Every fact below was verified live against the repository, GitHub, +and GCP project `example-gcp-project`. + +## 1. Git and release state + +| Item | Value | +|---|---| +| PR #18 | verified (docs-only: `README.md`, `validation-report-sprint4.md`), marked ready, **squash-merged** as `45543b6` under `ATLAS_APPROVE_PR18_MERGE` | +| Current `origin/main` | `45543b6` (clean tree) | +| Sprint tags | 1: `270e7d5` · 2: `5135778` · 3: `8aa1d7a` · 4: **`b609ac1`** (verified resolves to PR #17 merge commit) | +| Sprint 5 branch | `cursor/atlas-sprint-5-observability-64a2` from `45543b6` | +| Open Atlas PRs | none (PRs #2, #3, #6 are unrelated pre-Atlas scaffolding) | +| CI | `atlas-ci.yml` green on last code merge; PR #18 was docs-only (path-filtered, no checks — expected) | + +## 2. Current observability inventory (repository) + +| Asset | State | +|---|---| +| `src/atlas/logging/structured.py` | JSON formatter + `StepLogger` context manager; writes local JSONL per run; fields are ad-hoc per call site, no enforced contract, no correlation hierarchy | +| `dags/atlas_orchestration/callbacks.py` | `on_retry_callback` / `on_failure_callback` print structured JSON to task stdout; nothing durable | +| Run summary | `write_run_summary` task writes local `run-summary.json` + finalizes `atlas_ops.pipeline_runs` (MERGE, sanitized errors) | +| `atlas_ops.pipeline_runs` | run grain; has rows_generated/loaded/accepted/rejected, fact_rows, mart_event_count, failed_task_id, error fields — good base for freshness/volume monitors | +| `atlas_ops.deployments` | attempt grain with failure_stage; 9 Sprint 4 rows | +| `atlas_ops.schema_migrations` | ledger; 3 APPLIED | +| Task-attempt audit | **does not exist** (no task_events table) | +| Quality results | **not durable** — warehouse reconciliation prints PASS/FAIL JSON only | +| `src/atlas/validation/warehouse.py` | 10 batch-scoped checks returning `WarehouseReport` — ready to persist into `quality_results` | + +## 3. Current GCP observability state (all clean slate) + +| Surface | Finding | +|---|---| +| Log buckets | only `_Default` (30 d) and `_Required` (400 d); **no analytics enabled**, no custom buckets/views | +| Sinks | only `_Default`/`_Required`; no exclusions beyond defaults | +| Log-based metrics | none | +| Custom metric descriptors (`custom.googleapis.com/*`) | none | +| Alert policies | none | +| Notification channels | **none** — a verified recipient is a hard prerequisite for Phase 11 (see Blockers) | +| Dashboards | none | +| Logging IAM | no explicit Logging/Monitoring grants; agent SA `service1-831@…` is project **Owner** (pre-existing); Composer service agent has `composer.serviceAgent` + `ServiceAgentV2Ext` | + +## 4. The Sprint 4 missing-logs defect — preflight diagnosis + +Verified facts for the acceptance window (2026-07-18 22:00 → 07-19 01:40 UTC): + +- Project-wide sweep by `resource.type`: **only** BigQuery/GCS/IAM audit + entries plus 2 Composer admin-audit entries. Zero `airflow-worker`, + `airflow-scheduler`, `dag-processor`, or task-log entries exist anywhere, + including the `_AllLogs` view. The Sprint 4 filters were **correct**; the + logs genuinely never reached Cloud Logging. +- Ingestion itself works: a `gcloud logging write` roundtrip during Sprint 4 + succeeded and that entry is still the only non-audit log in the project. +- Not an obvious IAM gap: `atlas-composer-runtime` holds `composer.worker` + (includes `logging.logEntries.create`); Composer service agents hold + required roles; Logging API enabled; `_Default` sink filter is standard. +- Task logs were also absent from the environment bucket (Composer 3 default + is Cloud Logging only), so Sprint 4 diagnosis fell back to the Airflow + REST API — which worked and remains the documented fallback. + +Remaining hypotheses require a live environment (Phase 15): environment +`dataRetentionConfig.taskLogsRetentionConfig.storageMode` (not set explicitly +in Sprint 4), or a Composer 3 log-routing fault in the tenant→customer +stream. Resolution plan: recreate `atlas-dev`, use the built-in +`airflow_monitoring` DAG (runs every ~5 min) as a log canary, verify entries +with documented filters within 15 minutes of creation, and treat +`storageMode` explicitly at create time. This diagnosis gates every +log-dependent Sprint 5 deliverable and is therefore step 1 of live acceptance. + +## 5. Composer + +| Item | Value | +|---|---| +| `atlas-dev` exists now | no (deleted post-Sprint 4 per ADR-010; bucket also removed) | +| Image availability | `composer-3-airflow-3.1.7-build.13` **still available** in us-central1 (verified via imageVersions API; 5 versions listed) | +| Recreate config | small / us-central1 / runtime SA `atlas-composer-runtime` / env vars per `manage_atlas_composer.sh` (unchanged interface) | +| Sprint 4 lifecycle evidence | created ~22:05 UTC, deleted ~01:35 UTC (~3.5 h) | + +## 6. Security preflight + +- No secrets found in the sampled Sprint 4 logs (there were almost no logs). +- `sanitize_error_message` exists and is regression-tested (Sprint 4 D4). +- Redaction gaps to close in Phase 2: the structured-logging contract must + enforce sanitization centrally, not per call site. +- Linked BigQuery log dataset expands the log-read boundary to BigQuery IAM — + will be documented; dataset kept read-only; no broad `logging.privateLogViewer`. +- Sink writer identity: service account auto-created per sink; needs only + `logging.bucketWriter` on the destination bucket (same-project routing is + automatic). +- CI stays credentialless; WIF identities unchanged; no new broad grants + planned. IAM additions (if any) gated on `ATLAS_APPROVE_IAM`. + +## 7. Cost preflight + +| Item | Measurement / projection | +|---|---| +| Current log ingestion | ~0 (only audit logs; free allotment 50 GiB/mo far above need) | +| Projected Atlas log volume | MB/day scale at 50k-row batches — negligible; retention proposal: 30 d for `atlas-observability` bucket (matches `_Default`, justified by synthetic workload) | +| Custom metrics | ~15 descriptors, labels bounded to {environment, dag_id, task_id, component, status, check_name, severity}; projected < 200 time series total — far below chargeable tiers | +| BigQuery 7-day baseline | 2,696 jobs, 10.61 GB processed, 22.70 GB billed (min-billing inflation on many small jobs — the dominant Atlas cost signal) | +| Monitor DAG cadence | 30 min while Composer live (bounded windows; each evaluation scans MB) | +| Composer live-acceptance window | small env ≈ $0.75–1.00/h; target < 5 h; teardown gated on `ATLAS_APPROVE_TEARDOWN` | +| Permanent after teardown | log bucket/view/sink, linked dataset, metric descriptors, dashboards, alert policies, `atlas_ops` tables — all ~zero at rest | + +## 8. Data baselines (from `atlas_ops`, live-queried) + +| Signal | Baseline | +|---|---| +| Successful runs | 9 (avg duration **171 s**, all 50,000 rows loaded) | +| Failed runs | 7 (avg 76 s to failure) | +| Rejection rate | **1.79 %** avg (rollback smoke: 49,105 accepted + 895 rejected = 50,000) | +| Last success | 2026-07-19 01:21:56 UTC (`atlas-smoke-640cd786-local1784423774-run`) | +| Schedule | manual/smoke-triggered today; `@daily` when deployed unpaused | +| Deployment duration | ~10–15 min end-to-end (fetch→smoke) per Sprint 4 evidence | +| Schema signatures | 8 datasets; contracts in repo (`sql/`, dbt models) — manifest source for Phase 9 | + +Initial thresholds derived from these (labeled operational, not SLOs): +freshness warn 26 h / fail 30 h (daily schedule + slack); volume warn ±20 % / +fail ±50 % vs trailing-window median; rejection warn > 5 % / fail > 10 %; +cost warn/fail vs 7-day trailing bytes-billed median. + +## 9. Blockers and approvals + +| Gate | State | +|---|---| +| `ATLAS_APPROVE_PR18_MERGE` | exercised — PR #18 merged, main verified | +| `ATLAS_APPROVE_PROVISION` / `ATLAS_APPROVE_IAM` | assumed per Sprint 4 precedent ("IAM approved at every stage"); mutations remain plan-first | +| `ATLAS_APPROVE_COMPOSER_CREATE` | required before Phase 15 recreation (cost above) | +| `ATLAS_APPROVE_ALERT_CHANNEL` + **recipient** | **RESOLVED** — owner supplied the alert email in-session; email notification channel created: `projects/example-gcp-project/notificationChannels/6567861337166986657` ("Atlas Primary Operator (email)", enabled, recipient category: repository owner / primary operator). The address itself is intentionally not committed to Git | +| `ATLAS_APPROVE_LIVE_DRILLS` / `ATLAS_APPROVE_TEARDOWN` | required at Phases 16 / 15.9 | + +## 10. Implementation sequence + +1. **Phase 1–2** — architecture doc + ADR-011; structured logging contract + (`src/atlas/observability/logging.py`) with correlation hierarchy, + redaction, truncation, and contract tests. Wire step runner, DAG + callbacks, and deploy scripts to it. +2. **Phase 3–4** — migrations 004/005/006 (`task_events`, `quality_results`, + `monitor_evaluations`) + `atlas.ops.task_events`, `atlas.ops.quality_results`; + Airflow callback instrumentation; warehouse reconciliation persists results; + unit tests for every event path. +3. **Phase 5** — `bootstrap_observability.sh` (--plan/--apply/--status): + `atlas-observability` analytics log bucket (30 d), `atlas-runtime` view, + Atlas sink filter, linked `atlas_logs` dataset; saved queries in + `observability/queries/`. +4. **Phase 6–7** — metric descriptors + publisher (`atlas.observability.metrics`), + cardinality budget; BigQuery job labels in Python paths; dbt job-label + config verified against dbt-bigquery 1.11 docs (fallback: identity-based + attribution); `bigquery_cost.sql`; ADR-012. +5. **Phase 8–9** — `atlas_observability_monitor` DAG (30-min, read-only, + bounded windows, no-data semantics, drill overrides via + `config/observability.yaml`); schema-drift monitor with fixture-based + classification tests. +6. **Phase 10–12** — alert policy JSONs + `manage_atlas_alerts.sh`; + notification channel wiring (blocked pending recipient); dashboard JSON + + idempotent deploy. +7. **Phase 13–14** — runbook, alert catalog, on-call model; CI gates for all + observability artifacts; deployment bundle extended with monitor DAG + + observability modules + migrations. +8. **Phase 15–17 (live)** — recreate Composer (approval), deploy candidate, + **resolve the missing-logs defect first**, healthy 50k batch (Drill A), + then Drills B–G, incident report from Drill B, teardown (approval). +9. **Phase 18–20** — cost review, security review, validation report, README + and catalog updates; final CI; merge; `atlas-sprint-5-complete`. + +Static work (steps 1–7) proceeds immediately; cloud mutations stop at their +approval gates with exact plans. diff --git a/docs/preflight-sprint6.md b/docs/preflight-sprint6.md new file mode 100644 index 0000000..1586bf2 --- /dev/null +++ b/docs/preflight-sprint6.md @@ -0,0 +1,178 @@ +# Atlas Sprint 6 Preflight — Resilience, Failure Engineering, Recovery, Game Days + +Captured live on 2026-07-19 (UTC) before any Sprint 6 implementation or fault +injection. Every fact below was verified against the repository or the GCP +project `example-gcp-project`, not assumed from the execution prompt. + +## 1. Git and release state + +| Item | Verified value | +| --- | --- | +| origin/main | `078bc319c6683717e6583b4500af60b4dd3e168a` (matches prompt baseline) | +| Working tree | clean (`git status --porcelain` empty) | +| Sprint 6 branch | `cursor/atlas-sprint-6-resilience-64a2` (created from main; no other Sprint 6 branches or PRs exist) | +| Latest CI on main | run `29687038169` — atlas-ci **success** on Sprint 5 release commit `476e20a` | +| `atlas-sprint-1-complete` | `49a5fac` → `270e7d5` | +| `atlas-sprint-2-complete` | `1977983` → `5135778` | +| `atlas-sprint-3-complete` | `3f21d9d` → `8aa1d7a` | +| `atlas-sprint-4-complete` | `4251e94` → `b609ac1` | +| `atlas-sprint-5-complete` | `fd79562` → `476e20a2edcd9e6ae2e7aa2169d0f0c0fb13247c` (matches prompt) | +| Open PRs | #2, #3, #6 only — pre-Atlas scaffold, unrelated; no overlapping Atlas work | + +## 2. GCP state + +Active principal: `service1-831@example-gcp-project.iam.gserviceaccount.com` +(Cursor agent SA). Project: `example-gcp-project`. + +### Composer +Absent, as expected (Sprint 5 ephemeral teardown 2026-07-19T07:37:56Z). No +orphaned Composer buckets. Environment-dependent alerts were intentionally +disabled before teardown and remain disabled — no false incidents. + +### BigQuery datasets +`atlas_raw`, `atlas_staging`, `atlas_intermediate`, `atlas_quarantine`, +`atlas_core`, `atlas_marts`, `atlas_dbt_staging`, `atlas_ops`, `atlas_logs` +(linked, read-only). + +### atlas_ops tables +`pipeline_runs`, `deployments`, `schema_migrations` (6 APPLIED), +`task_events` (176 rows), `quality_results` (60 rows), +`monitor_evaluations` (91 rows). Migration 007 (`recovery_actions`) is the +next free slot. + +### GCS +`atlas-raw-events-…` (canonical raw), `atlas-deployments-…` (immutable +releases), `atlas-ci-…` (ephemeral CI). + +### Logging (permanent, verified live) +Sink `atlas-observability-sink` → bucket `atlas-observability` +(us-central1, 30-day retention, analytics enabled), view `atlas-runtime`, +linked dataset `atlas_logs` queryable. + +### Monitoring (permanent, verified live) +Dashboard `Atlas Operations` (`a4f0a238-90b5-445b-925e-d0922d343c2b`). +Notification channel `Atlas Primary Operator (email)` +(`6567861337166986657`). Ten alert policies present: + +| Policy | Enabled | +| --- | --- | +| Atlas: pipeline failed | true | +| Atlas: data stale | **false** (disabled for teardown — re-enable with Composer) | +| Atlas: reconciliation failed | true | +| Atlas: critical volume deviation | true | +| Atlas: breaking schema drift | true | +| Atlas: deployment failed | true | +| Atlas: rollback failed | true | +| Atlas: Composer environment unhealthy | **false** (disabled for teardown — re-enable with Composer) | +| Atlas: BigQuery cost anomaly | true | +| Atlas: telemetry incomplete | true | + +No open incidents; no `AlertPolicyViolation` entries since teardown. + +### WIF / service accounts +Pool `atlas-github-pool` ACTIVE. SAs: `atlas-composer-runtime`, +`atlas-github-deployer`, `atlas-github-integration` (see §5 IAM table). + +## 3. Baseline data (latest healthy state) + +| Item | Value | +| --- | --- | +| Latest successful run | `atlas-drillb-20260719-recovery-run` (batch `atlas-drillb-20260719`), SUCCESS 06:58:00Z | +| Raw rows | 50,000 | +| Accepted / fact rows | 49,105 | +| Rejected | 895 (rate 0.0179) | +| Mart total events | 343,738 (reconciles) | +| Quality checks on latest run | 10/10 PASS | +| Latest deployment | `atlas-dev-20260719T061802Z-2109310b` SUCCESS (sha `2109310b`) | +| Latest FAILED deployment (expected, drill) | `atlas-dev-20260719T042523Z-8fe17dcd` | +| Migrations applied | 001–006 | +| Success marker | present for latest batch (verified during Sprint 5 acceptance) | +| Schema signatures | `observability/schema/expected-schemas.json` matches live tables (schema-drift monitor last evaluated PASS) | + +## 4. Known limitations carried into Sprint 6 + +1. **Raw Airflow stdout**: Composer 3 `build.13` platform log-export defect; + Atlas telemetry is mirrored directly via `ATLAS_LOG_TO_CLOUD_LOGGING=true` + (`atlas-events` log). Phase 1 will re-test on the currently available image. +2. **Failed-task timing**: two `FAILED` task_events rows + (`atlas-drillb-20260719-run`: `dbt_build`, `write_run_summary`) have NULL + `started_at`/`completed_at`/`duration_ms` — the exact Phase 1 cleanup target. + The failure-callback path records the terminal event without the timing that + the success path gets from the runner wrapper. +3. **Email delivery latency**: notification evidence relies on operator + confirmation (user confirmed receipt during Sprint 5); no programmatic + mailbox access. +4. **Thresholds are synthetic-workload initial values**, not production SLOs. +5. **Single-operator model** (the primary operator primary; repo owner escalation). +6. **Cost attribution boundary**: job labels + runtime identity; dbt child jobs + labeled via `query-comment`/`job-label`; console-issued ad-hoc queries are + outside attribution. + +## 5. Security / IAM snapshot (before any Sprint 6 change) + +| Principal | Roles (project level) | +| --- | --- | +| `atlas-composer-runtime@…` | `composer.worker`, `bigquery.jobUser`, `bigquery.dataEditor`, `bigquery.resourceViewer` | +| `atlas-github-deployer@…` | `bigquery.jobUser`, `bigquery.dataEditor`, `composer.user`, `composer.environmentAndStorageObjectAdmin` | +| `atlas-github-integration@…` | `bigquery.jobUser`, `bigquery.dataEditor` | + +- Secret scanning: enforced by `validate_ci.sh` gate (green on main). +- Fault-injection leak risk: scenarios must reuse the Sprint 5 sanitization + (`error_message` redaction/truncation); no scenario may echo credentials, + tokens, or raw payloads. Enforced by framework tests. +- Test-resource blast radius: all destructive scenarios use isolated batch IDs + (`atlas-s6-*` prefix), isolated datasets/tables/prefixes, never canonical + batches; the framework will refuse canonical batch IDs by construction. +- IAM scenarios (S6-IAM-*) remove exactly one role from one member, capture + before/after policy, and restore the identical binding. + +## 6. Cost plan + +| Item | Estimate | +| --- | --- | +| Composer SMALL (us-central1) | ≈ $0.60–0.75/h; Sprint 5 window (3.9 h) cost ≈ $2.50 | +| Sprint 6 live window target | ≤ 12 h Composer runtime (five game days batched into one window), ceiling ≈ $9 | +| BigQuery | synthetic 50k-row batches ≈ MBs per query; cost drills use dry-run / `maximum_bytes_billed` only — no intentional spend | +| Logging/metrics | within Sprint 5 free-tier envelope (≈ 40 MB/day peak measured) | +| Teardown | Composer + drill fixtures deleted under `ATLAS_APPROVE_TEARDOWN`; permanent observability plane retained | + +Maximum allowed live test duration: one Composer window ≤ 12 h; every scenario +carries `maximum_duration` and `maximum_cost` in `config/failure_scenarios.yaml`. + +## 7. Approvals + +Per the sprint owner's standing decisions (ephemeral Composer approved, IAM +approved at every stage, builds approved) the Sprint 6 approval set +(`ATLAS_APPROVE_PROVISION/IAM/COMPOSER_CREATE/FAILURE_INJECTION/` +`DESTRUCTIVE_FIXTURE/DEPLOY/ROLLBACK_TEST/ALERT_DRILLS/TEARDOWN=true`) is +treated as granted as stated in the execution prompt. Each gated mutation still +logs which approval it consumed; no approval is reinterpreted across categories. + +## 8. Implementation sequence + +1. **Phase 1** — failed-task timing (`timing_source`/`timing_confidence`, + derive from callback context or STARTED row, never invent) + regression + tests; Composer log re-test deferred to the live window. +2. **Phases 2–3** — failure catalog (`failure-catalog-sprint6.md`, + `config/failure_scenarios.yaml`, ADR-013) and fault-injection framework + (`scripts/run_failure_scenario.sh`, `src/atlas/failure_injection/`, + disabled-by-default enforcement + tests). +3. **Phase 4** — migration 007 `recovery_actions` + `src/atlas/ops/recovery_actions.py` + tests. +4. **Phases 5–12 (static half)** — scenario logic and guards implementable + without cloud: cost guards (dry-run byte ceiling, backfill window, + full-refresh approval), schema classification extensions, IAM error + classification, observability degradation paths, plus unit tests per + Phase 16 matrix. +5. **Phases 13–14** — recovery runbook + ADR-014 + ADR-015 + game-day plan. +6. **Phase 16** — CI gates (scenario schema validation, fault-injection + default-off check) and green static CI. +7. **Phase 17 live window** — recreate Composer (build.13 or newer compatible + image), deploy candidate, baseline batch, Game Days 1–5 with recovery, + reconciliation, and MTTR capture. +8. **Phases 15+18** — two incident reports, validation/cost/security reviews, + README updates. +9. **Closeout** — disable fixtures, teardown, final CI, merge, tag + `atlas-sprint-6-complete`. + +No cloud resource is provisioned and no fault is injected until steps 1–6 are +green in CI. diff --git a/docs/preflight-sprint7.md b/docs/preflight-sprint7.md new file mode 100644 index 0000000..9c1965d --- /dev/null +++ b/docs/preflight-sprint7.md @@ -0,0 +1,183 @@ +# Atlas Sprint 7 Preflight — Governance, Contracts, Schema Evolution, Security, Performance, Cost + +Verified live against `origin/main` and Git tags on 2026-07-19, before any +Sprint 7 implementation. Repository truth overrides the master prompt; any +discrepancies are noted inline. + +## 1. Git and release state + +| Item | Value | +| --- | --- | +| Current branch | `cursor/atlas-sprint-7-governance-64a2` (created from `main`) | +| `origin/main` HEAD | `1c2cc158c034f2920b1b39ce28a7ff93f375f335` (Sprint 6 release-table row) | +| Sprint 6 merge commit | `48da9d226e0e942095c0623e5bd71d06f5e32e36` — **matches prompt** | +| `atlas-sprint-6-complete` | resolves to `48da9d2…` — **matches expected** | +| Tags 1–5 | `270e7d5`, `5135778`, `8aa1d7a`, `b609ac1`, `476e20a` (all present, unchanged) | +| Working tree | clean | +| Sprint 7 branches | none existed prior to this run | +| Open PRs | #6 (setup-dev-env, draft), #3 (greptile, draft), #2 (duckdb starter, open) — all stale/unrelated to Atlas sprints; not a valid base | +| Latest CI on main lineage | Sprint 6 run `29694739152` — atlas-ci **success** | + +Conclusion: Sprint 6 is properly merged and tagged; Sprint 7 branches cleanly +from `main`. No unresolved Atlas branch should be used as a base. + +## 2. Sprint 6 inheritance + +- **Completion evidence:** `docs/validation-report-sprint6.md`, + `game-day-results-sprint6.md`, incident reports INC-S6-001/002, cost/security + reviews. Composer torn down (`composer environments list` = **0 items**), + bucket removed. +- **`recovery_actions` state:** table live (migration 007, 20 columns); **1 row** + — `rec-s6-quarantine-atlas20260719`, `QUARANTINE_BATCH`, SUCCESS/VERIFIED. +- **task_events timing provenance:** migration 008 applied + (`timing_source`, `timing_confidence`). +- **Alert state:** 10 policies; `Atlas: data stale` and + `Atlas: Composer environment unhealthy` **DISABLED** (correct post-teardown + baseline — they assert on an absent Composer); other 8 ENABLED. +- **Outstanding limitations carried into Sprint 7:** + - **Same-date reprocessing defect (INC-S6-001):** `generate_events` seeds + deterministically from `processing_date`, so multiple batches for one date + share identical `event_id`s. `int_event_classification` dedups `event_id` + **globally** (`row_number() over (partition by event_id …)`), so a second + same-date batch inflates `is_duplicate_extra` to the full batch size and the + batch-scoped `assert_source_anomaly_profile` test fails. **Sprint 7 Phase 3 + must resolve this.** + - **Overlapping runs (INC-S6-002):** deploy unpauses the DAG, letting a + scheduled interval contend with the smoke run. Mitigation was manual DAG + pause; a durable fix is a Sprint 7/8 candidate (orchestration, lower + priority than the governance mission). + +### Current duplicate / grain invariants (must preserve) + +- `fct_events`: `materialized=incremental`, `incremental_strategy=merge`, + `unique_key=event_id`, `partition_by=event_date (date)`, + `cluster_by=[event_name, country_code]`, `on_schema_change=fail`. **Grain: one + row per `event_id`** — global fact uniqueness enforced. +- `int_event_classification`: full-table rebuild; `is_duplicate_extra` = + `row_number() over (partition by event_id order by ingested_at desc, …) > 1`. +- Anomaly profile expectation (`tests/assert_source_anomaly_profile.sql`, per + batch): duplicate_extra=50, null_user=500, invalid_country=200, + date_timestamp_mismatch=300, future_dated=150, backdated=300, late=0. + +## 3. Contracts & governance (current state — the gap Sprint 7 fills) + +| Artifact | State | +| --- | --- | +| dbt model contracts | Only `stg_events` has `config.contract.enforced: true` with `data_type`s. Other layers have descriptions + tests but no enforced contract. | +| dbt `meta` (owner/grain/classification/consumers/contract_version) | **absent** on all models | +| dbt exposures | **none** | +| Source declarations | `models/sources/sources.yml` | +| Model descriptions | present (grain stated informally in prose) | +| CODEOWNERS | **none** | +| Schema manifests / versioned baselines | **none** (only `observability/schema/expected-schemas.json` for ops tables) | +| Migration records | `sql/migrations/manifest.txt` (001–008; ledger `atlas_ops.schema_migrations`, checksum-guarded) | +| Retention config | **none declared** (datasets/buckets rely on GCP defaults) | +| Classification metadata | **none** | +| Governance source-of-truth | **none** — Sprint 7 Phase 1 creates it | + +**Source-of-truth decision (ADR-016):** dbt `meta`/properties will be +authoritative for dbt models; a small `governance/` registry will cover non-dbt +assets (raw/ops tables, buckets, DAGs, dashboards, log resources). A generated +catalog consolidates both. No triple-maintained metadata. + +## 4. Lineage inputs available + +- dbt `manifest.json` (via `dbt parse`/`compile`) — authoritative model DAG + + source→model edges. dbt venv present at `/tmp/dbt-venv` / `.venv-dbt`. +- Airflow DAG task graph (`dags/atlas_batch_pipeline.py`, + `atlas_observability_monitor.py`). +- Migration manifest + `atlas_ops` audit-table dependencies. +- No graph DB / metadata service exists or will be built (repo artifacts only). + +## 5. IAM inventory (from Sprint 6 live `get-iam-policy`, to re-verify in Phase 7) + +| Principal | Roles | Notes | +| --- | --- | --- | +| `atlas-composer-runtime@…` | `composer.worker`, `bigquery.jobUser`, `bigquery.dataEditor`, `bigquery.resourceViewer` | Composer runtime | +| `atlas-github-integration@…` | `bigquery.jobUser`, `bigquery.dataEditor` | CI (isolated datasets) | +| `atlas-github-deployer@…` | `bigquery.jobUser`, `bigquery.dataEditor`, `composer.user`, `composer.environmentAndStorageObjectAdmin` | deploy | +| `service-…@cloudcomposer-accounts` | `composer.serviceAgent`, `composer.ServiceAgentV2Ext` | Google-managed | +| Log sink writer | Logging service agent (intra-project sink) | Sprint 5 | +| Human operator / Cursor dev credential | pre-existing broad project access (documented since Sprint 4) | used for recovery DELETEs | + +No Owner/Editor/broad-admin on Atlas identities; keyless WIF for GitHub. Phase 7 +builds the full matrix with observed-usage and a candidate reduction + negative +test (gated on `ATLAS_APPROVE_IAM`). + +## 6. Security posture (to formalize in Phase 8) + +Existing controls: `gate_secret_scan` in CI; `sanitize_error_message` for audit +fields; structured-log field allowlist + truncation; notification evidence +stores only the email (recipient is a real address — must not be committed in +new evidence). Sample data is synthetic (`generate_events`). Public-repo +extraction review is explicitly deferred to Sprint 8. + +## 7. BigQuery baseline (live) + +| Table | Rows | +| --- | --- | +| `atlas_raw.events` | 850,000 | +| `atlas_core.fct_events` | 392,845 | +| `atlas_marts.mart_daily_event_metrics` | 2,804 | +| `atlas_ops.pipeline_runs` | 25 | +| `atlas_ops.task_events` | 257 | +| `atlas_ops.recovery_actions` | 1 | + +Datasets present: `atlas_core`, `atlas_dbt_staging`, `atlas_intermediate`, +`atlas_logs`, `atlas_marts`, `atlas_ops`, `atlas_quarantine`, `atlas_raw`, +`atlas_staging`. `fct_events` is partitioned+clustered; raw `events` is +partitioned by `event_date`, clustered by `event_name, country_code`. Job +labels already applied (Sprint 5 ADR-012) for cost attribution. This is a +50k-per-batch dataset — performance claims will be scoped honestly (no +production-scale extrapolation). + +## 8. Cost baseline + +- **Permanent footprint:** the 9 BigQuery datasets (small), the + `atlas-observability` log bucket (30-day retention), metric descriptors, 10 + alert policies, 1 notification channel, dashboard, audit tables, release + bundle bucket `atlas-deployments-…`. +- **Composer:** absent (ephemeral; only created if a control genuinely needs it). +- **Existing cost guards (Sprint 6):** `validate_backfill_window` (7-day), + `require_full_refresh_approval`, `enforce_dry_run_ceiling`, + `guarded_query_config` in `src/atlas/observability/cost_guards.py`. +- **Sprint 7 additions:** `config/cost_controls.yaml`, a CLI estimator + (`cost_guard estimate`), a performance-suite byte ceiling, and a + required-partition-filter check. Proposed hard ceiling for the whole live + window: **`ATLAS_MAX_PERFORMANCE_TEST_BYTES` default 5 GB**, individual query + dry-run ceiling 1 GB (both overridable by approval). + +## 9. Existing CI gate framework (to extend, not replace) + +`scripts/validate_ci.sh` uses `run_gate ` with static/integration +modes. Static gates: secret_scan, shell_syntax, shell_static, workflow_yaml, +sql_migrations, python_format, python_lint, python_types, python_tests, +config_validation, observability_config, failure_injection, dbt_static. +`sql_migrations` currently checks additive-only/non-empty but **not** applied +-migration checksum immutability against the ledger — Sprint 7 will add +checksum-drift detection. New gates will follow the same `run_gate` pattern; +PR CI stays credentialless. + +## 10. Do-not-touch confirmation + +Out-of-scope paths present and will not be modified without a documented, +`apps/**`, `packages/**`, `transform/dbt/**`. + +## 11. Approval requirements for the gated phases + +Static implementation (Phases 1–6, 8, 13, 14 fixtures, most docs) needs **no +approval**. The following live actions are gated and will stop with a recorded +blocked-gate if approval is absent: + +| Action | Variable | +| --- | --- | +| BigQuery performance suite (bounded) | `ATLAS_APPROVE_PERFORMANCE_TESTS=true` + `ATLAS_MAX_PERFORMANCE_TEST_BYTES` | +| Live enforcement demos (cost block, etc.) | `ATLAS_APPROVE_LIVE_ACCEPTANCE=true` | +| IAM reduction + negative test | `ATLAS_APPROVE_IAM=true` | +| Retention/lifecycle mutation | `ATLAS_APPROVE_RETENTION_MUTATION=true` | +| Composer create (only if required) | `ATLAS_APPROVE_COMPOSER_CREATE=true` | +| Teardown | `ATLAS_APPROVE_TEARDOWN=true` | +| Breaking-schema demo (fixtures only) | `ATLAS_APPROVE_BREAKING_SCHEMA_DEMO=true` | +| Ordinary dev-resource mutation | `ATLAS_APPROVE_PROVISION=true` | + +Preflight complete. No resources mutated. diff --git a/docs/preflight-sprint8.md b/docs/preflight-sprint8.md new file mode 100644 index 0000000..b35cc3b --- /dev/null +++ b/docs/preflight-sprint8.md @@ -0,0 +1,172 @@ +# Atlas Sprint 8 Preflight — Reference Architecture, Reproducibility & Handoff + +Verified against `origin/main` on 2026-07-19. Repository truth overrides the +master prompt; every value below was resolved from Git, tags, PRs, and file +contents, not from prior conversation memory. No repository content was changed +before this preflight was written (the Sprint 8 branch was created first, which +is not a content change). + +## 1. Verified repository baseline + +| Item | Verified value | Prompt expectation | Match | +| --- | --- | --- | --- | +| Current branch | `cursor/atlas-sprint-8-reference-handoff-64a2` (fresh off main) | new S8 branch | ✓ | +| `origin/main` HEAD | `3f986aaa703d9d7da10b94ecf6fba753fec2a80d` | `3f986aa` (release-row commit) | ✓ | +| Sprint 7 merge commit | `9d031c99cacefbd6be461f6e8f7b16bb963c3ab9` (PR #22, squash) | `9d031c9` | ✓ | +| `atlas-sprint-7-complete` | resolves to `9d031c9` | tag on merge commit | ✓ | +| PR #22 | **MERGED**, mergeCommit `9d031c9` | merged | ✓ | +| Working tree | clean | — | ✓ | +| Stale Sprint 8 branch/PR | none (`refs/heads/*sprint-8*` empty) | none to resume | ✓ | +| Open PRs | #2, #3, #6 — unrelated non-Atlas/setup PRs, not to be resumed | — | ✓ | + +Sprint 1–7 tags all resolve: `270e7d5`, `5135778`, `8aa1d7a`, `b609ac1`, +`476e20a`, `48da9d2`, `9d031c9`. README release table lists all seven rows +consistently. Sprint 1–7 tags will not be moved. + +## 2. Sprint 1–7 capability summary (what exists) + +| Sprint | Capability | Primary evidence | +| --- | --- | --- | +| 1 | Deterministic 50k-event generation → immutable GCS → partitioned BigQuery raw → validation | `validation-report-sprint1.md`, `src/atlas/{generator,ingestion,loader,validation}` | +| 2 | Governed dbt warehouse: staging → classification → accepted/rejected → core (fact/dims) → marts; contracts, tests, reconciliation, incremental | `dbt/atlas_dbt`, `validation-report-sprint2.md` | +| 3 | Airflow 3.1.7 orchestration, stable `batch_id`, retries/reruns/backfills, `pipeline_runs` audit | `dags/`, `validation-report-sprint3.md`, ADR-006/007 | +| 4 | GitHub CI/CD, keyless WIF, immutable release bundles, migrations, ephemeral Composer deploy, smoke validation, rollback | `validation-report-sprint4.md`, ADR-008/009/010 | +| 5 | Structured observability, task/quality telemetry, Cloud Logging+Monitoring, alerts, dashboard, runbooks, drills | `validation-report-sprint5.md`, ADR-011/012 | +| 6 | Failure taxonomy, controlled fault injection, recovery audit, game days, cost guards, schema-version handling | `validation-report-sprint6.md`, ADR-013/014/015 | +| 7 | Governance source of truth, contracts, schema compatibility, lineage/impact, deprecation, IAM review, retention, BigQuery perf baseline, cost controls, 5 CI gates | `validation-report-sprint7.md`, ADR-016–020 | + +Codebase scan: `src/atlas/` has 13 modules (batch, config, failure_injection, +generator, ingestion, loader, logging, observability, ops, pipeline, validation, +governance + `__init__`). `governance/` has registry, catalog, lineage, impact, +schema_check, retention, security_policy. 41 scripts. **282 unit tests.** 61 +docs, 19 ADRs, 4 evidence directories (sprint4–7). + +## 3. Sprint 7 unresolved-gate summary (must remain visibly blocked) + +Recorded in `validation-report-sprint7.md` §18 as blocked on unset approvals: + +| Blocked gate | Approval required | Sprint 8 disposition | +| --- | --- | --- | +| Live IAM reduction + positive/negative test (`atlas-github-integration` dataEditor) | `ATLAS_APPROVE_IAM` | Retain as BLOCKED risk; preserve plan; execute only if approval present | +| Executed (billed) BigQuery performance suite | `ATLAS_APPROVE_PERFORMANCE_TESTS` (+ `ATLAS_MAX_PERFORMANCE_TEST_BYTES`) | Retain as BLOCKED risk; dry-run baseline already exists | +| Live retention/expiration application | `ATLAS_APPROVE_RETENTION_MUTATION` | Retain as BLOCKED risk; disposal dry-run plan exists | + +These are optional Sprint 8 closure improvements, **not** automatic requirements. +No claim will be made that a blocked control was executed. + +## 4. Reproducibility state + +| Facet | Verified value | +| --- | --- | +| Python | `requires-python = ">=3.12,<3.13"` (pyproject); mypy target 3.12 | +| Runtime deps | `requirements.txt` (google-cloud-bigquery/storage/monitoring/logging, PyYAML, pytest) | +| CI toolchain (pinned) | `requirements-ci.txt` — ruff 0.15.22, mypy 2.3.0, **yamllint 1.38.0, shellcheck-py 0.11.0.1**, pytest 9.1.1, types-PyYAML | +| dbt | 1.11.12, BigQuery adapter (`dbt/atlas_dbt`) | +| Airflow | pinned `apache-airflow==3.1.7` + providers-google 20.0.0 (`airflow/requirements-airflow.txt`); Composer `composer-3-airflow-3.1.7-build.13` | +| Credentialless path | `bash scripts/validate_ci.sh --mode static` (no GCP creds) | +| Credentialed read-only | `verify_mcp_access.sh`; BigQuery dry-run via `cost_guard` | +| Test count | 282 collected | +| Generated/gitignored | `data/`, `logs/`, dbt `target/`, `.venv`, `.gcp/` | + +**Reproducibility note (root cause captured for clean-clone):** yamllint and +shellcheck are pinned in `requirements-ci.txt`. A clone that installs only +`requirements.txt` will `SKIP` `workflow_yaml`/`shell_static`; installing +`requirements-ci.txt` makes those gates run. The clean-clone doc must direct +installing **both** files, matching the Sprint 4 quick start. + +## 5. CI gate inventory (21 gates in `validate_ci.sh`) + +`secret_scan, shell_syntax, shell_static, workflow_yaml, sql_migrations, +python_format, python_lint, python_types, python_tests, config_validation, +observability_config, failure_injection, governance, schema_compatibility, +lineage_impact, security_policy, performance_cost, airflow_environment, +dag_import, dbt_static, gcp_integration`. Sprint 8 adds exactly one focused gate: +`gate_reference_handoff` (Phase 18). No validation logic will be duplicated in +workflow YAML. + +## 6. Reference-architecture readiness & risks found in preflight + +- **Docs are conversation-independent:** grep for `/home/ubuntu`, `/workspace`, + `/Users/`, `ChatGPT`, "prior conversation", "previous session" across `docs/` + → **0 matches.** Good baseline for the handoff CI gate. +- **Public-extraction finding (feeds Phase 14):** operator name / notification + email (`the primary operator…@gmail.com`) appears in ~35 files — concentrated in + `observability/alerts/*.json` (notification channel), Sprint 4/5 WIF & IAM docs, + `bootstrap_github_wif.sh`, and setup guides. These need REDACT/REPLACE_WITH_SAMPLE + dispositions. Private project id `example-gcp-project` and bucket/SA names are + pervasive and expected (REPLACE_WITH_SAMPLE at template time). +- **Composer customer-project task-log limitation** (Sprint 5) and **synthetic + 50k-row scale** are standing honest limitations → unresolved-risk register. +- Stable architecture decisions live in ADR-002…020; invariants are implied + across sprints and must be consolidated (Phase 4). + +## 7. Proposed Sprint 8 file structure + +``` + + START_HERE.md + docs/ + preflight-sprint8.md (this) + sprint8-context-pack.md + token-efficiency-sprint8.md + validation-report-sprint8.md + reference-architecture/ (18 curated map docs + reference-manifest.yml) + handoff/ (operator ×3, agent ×2, clean-clone, handoff ×3, evidence ledger) + evidence-sprint8/ (clean-clone-results.md, independent-handoff-results.md) + presentation/ (presentation, demo-script, question-bank) + adr/ADR-021-reference-architecture-and-handoff-contract.md + governance/ + unresolved_risks.yml + generated/evidence-index.json + config/public_extraction_manifest.yml + scripts/ + validate_clean_clone.sh + validate_public_extraction.py + validate_ci.sh (+ gate_reference_handoff) + src/atlas/reference/ (validate.py — `python -m atlas.reference.validate`) +``` + +## 8. Bounded execution sequence + +P0 preflight+context (this) → P1 START_HERE → P2–7 reference package (manifest + +17 docs) → P8 evidence index + `atlas.reference.validate` → P9–10 operator/agent +onboarding → P11 clean-clone script + 2 fresh-dir runs → P12 independent handoff +test + scorecard → P13 unresolved-risk register + S7 disposition → P14 +public-extraction review + validator → P15 template-extraction plan → P16–17 +capability ledger + presentation → P18 `gate_reference_handoff` → P19 final CI + +clean-clone + docs → close out on `ATLAS_APPROVE_RELEASE`. + +## 9. Token & cloud-cost envelope + +Target: **50–65% of Sprint 7** agent consumption. 1 primary scan (done) + 1 +targeted closeout scan. **0 Composer cycles, 0 deployment cycles, ~$0 cloud** +(only optional read-only dry-runs under `ATLAS_APPROVE_HANDOFF_LIVE_READ`). Link +to existing docs rather than duplicating Sprint 1–7 prose. One handoff gate, not +many. Tripwire: stop and report if projected work nears 75% of Sprint 7. + +## 10. Required approvals (all currently UNSET unless noted) + +| Variable | Gates | Needed for | +| --- | --- | --- | +| `ATLAS_APPROVE_HANDOFF_LIVE_READ` | optional read-only GCP leg of clean-clone/handoff | not required for green | +| `ATLAS_APPROVE_IAM` | Sprint 7 IAM leg | stays BLOCKED if unset | +| `ATLAS_APPROVE_PERFORMANCE_TESTS` (+ byte ceiling) | Sprint 7 billed perf | stays BLOCKED if unset | +| `ATLAS_APPROVE_RETENTION_MUTATION` | Sprint 7 retention | stays BLOCKED if unset | +| `ATLAS_APPROVE_PUBLIC_EXTRACTION` | local extraction candidate dir | dry-run works without it | +| `ATLAS_APPROVE_RELEASE` | final `atlas-sprint-8-complete` tag | closeout only | + +`ATLAS_APPROVE_PROVISION/DEPLOY/TEARDOWN=true` are present in the environment but +Sprint 8 requires no provisioning, deployment, or teardown. + +## 11. Known risks entering Sprint 8 + +1. Clean-clone may reveal an undocumented setup step (that is the point — fix + root cause in docs/scripts, rerun from a fresh directory). +2. Independent handoff may score < 25/30 on first attempt; repair docs, rerun + affected portions with fresh context. +3. Personal-email/name spread is wider than one file; the public-extraction + validator must catch all of it without printing values. +4. This agent cannot fully guarantee "independent" handoff without a separate + tester context; the handoff will use a subagent with only the repo + + START_HERE + assignment, and any assistance will be recorded as an + intervention honestly. diff --git a/docs/presentation/atlas-demo-script.md b/docs/presentation/atlas-demo-script.md new file mode 100644 index 0000000..6a4aa5d --- /dev/null +++ b/docs/presentation/atlas-demo-script.md @@ -0,0 +1,34 @@ +# Atlas Demo Script + +**Status:** CURRENT · A bounded, safe (credentialless) demo sequence. Every step +runs without GCP access and finishes in a few minutes. + +```bash +cd Atlas-GCP-Build +python3 -m venv .venv && source .venv/bin/activate # required on PEP 668 hosts +pip install -r requirements.txt -r requirements-ci.txt +export PYTHONPATH=src # atlas.* modules live under src/ +``` + +1. **Starting point** — open [`../../START_HERE.md`](../../START_HERE.md); show the + four audience paths. +2. **Architecture** — open [architecture-overview](../reference-architecture/architecture-overview.md); walk the four flows. +3. **Static validation** — `bash scripts/validate_ci.sh --mode static` → green + gate summary (`validate_ci: PASS`). +4. **Governance catalog** — `python -m atlas.governance.catalog check` → + "governance catalog matches sources" (22 assets). +5. **Lineage** — `python -m atlas.governance.lineage` → 26 nodes / 29 edges. +6. **Schema compatibility** — show `governance/schemas/manifests/baseline.json` + and `sql/migrations/checksums.lock`; explain the immutability gate. +7. **Cost guard** — `python -m atlas.observability.cost_guard check-partition-filter + --sql-file observability/performance/queries/unbounded_scan.sql --asset atlas_raw.events` + → BLOCKED before spend ($0). +8. **Evidence index** — `python -m atlas.reference.validate` → OK; open + [evidence-index](../reference-architecture/evidence-index.md). +9. **Incident & recovery** — open `docs/incident-report-INC-S6-001-batch-contamination.md` + and `docs/validation-report-sprint6.md` §recovery. +10. **Unresolved risks** — open [unresolved-risks](../reference-architecture/unresolved-risks.md); + call out the three BLOCKED gates honestly. + +No mutation, no billed query, no Composer. The whole demo is credentialless and +reproducible. diff --git a/docs/presentation/atlas-final-technical-presentation.md b/docs/presentation/atlas-final-technical-presentation.md new file mode 100644 index 0000000..90e0398 --- /dev/null +++ b/docs/presentation/atlas-final-technical-presentation.md @@ -0,0 +1,68 @@ +# Atlas — Final Technical Presentation (source) + +**Status:** CURRENT · Source for ~12–15 slides. No slide deck binary is created. +Each slide: purpose, key points, proposed visual, evidence source, speaker notes, +likely reviewer question. Evidence links resolve to repository artifacts. + +--- + +### Slide 1 — Problem & objective +- **Purpose:** frame the engineering problem. **Key points:** correct, governed, + observable, recoverable batch data platform on GCP; synthetic scale by design. +- **Visual:** one-line value statement. **Evidence:** [system-context](../reference-architecture/system-context.md). +- **Notes:** production-*oriented*, not production-*scale*. **Q:** why batch? + +### Slide 2 — Architecture overview +- **Key points:** four flows (data/control/operational/governance). +- **Visual:** the context diagram. **Evidence:** [architecture-overview](../reference-architecture/architecture-overview.md). **Q:** where are the boundaries? + +### Slide 3 — Data lifecycle +- **Key points:** generate → GCS → raw → dbt → marts; immutable, run-scoped. +- **Visual:** data-flow arrows. **Evidence:** `validation-report-sprint1/2`. **Q:** how is idempotency achieved? + +### Slide 4 — Warehouse & dbt model +- **Key points:** staging→classification→accepted/rejected→fact/dims→marts; + contracts + tests. **Visual:** dbt DAG. **Evidence:** `dbt/atlas_dbt`. **Q:** why dbt? + +### Slide 5 — Orchestration & identity +- **Key points:** Airflow; `batch_id` vs `pipeline_run_id`; retries/backfills. +- **Evidence:** `validation-report-sprint3`, ADR-006. **Q:** exact-rerun idempotency? + +### Slide 6 — CI/CD & secure deployment +- **Key points:** credentialless PR CI; keyless WIF; immutable releases; rollback. +- **Evidence:** ADR-008/009/010, `ci-cd-runbook-sprint4`. **Q:** why keyless? + +### Slide 7 — Observability +- **Key points:** structured logs, correlation ids, metrics, alerts→runbooks. +- **Evidence:** ADR-011, `observability-runbook-sprint5`. **Q:** NO_DATA alerts? + +### Slide 8 — Failure & recovery +- **Key points:** detect→contain→diagnose→recover→verify→prevent; verified repair. +- **Evidence:** `game-day-results-sprint6`, INC-S6-001. **Q:** how is recovery verified? + +### Slide 9 — Governance & schema evolution +- **Key points:** one source of truth; compatibility classes; migration immutability. +- **Evidence:** ADR-016/017, `governance/`. **Q:** how are breaking changes blocked? + +### Slide 10 — Security & IAM +- **Key points:** WIF, no keys/Owner/Editor; one blocked least-privilege reduction. +- **Evidence:** [security model](../reference-architecture/security-and-identity-model.md). **Q:** is least privilege proven? + +### Slide 11 — Performance & cost +- **Key points:** dry-run baseline, partition pruning, cost-guard $0 block. +- **Evidence:** `performance/cost-review-sprint7`. **Q:** what about production scale? + +### Slide 12 — Incidents & lessons +- **Key points:** INC-S6-001/002; the same-date reprocessing fix (ADR-006 amend). +- **Evidence:** incident reports. **Q:** what recurred and how was it prevented? + +### Slide 13 — Evidence & reproducibility +- **Key points:** evidence index; clean-clone; independent handoff. +- **Evidence:** [evidence-index](../reference-architecture/evidence-index.md), `evidence-sprint8/`. **Q:** can someone else run it? + +### Slide 14 — Limitations +- **Key points:** synthetic scale; blocked IAM/perf/retention; no streaming; not a template. +- **Evidence:** [unresolved-risks](../reference-architecture/unresolved-risks.md). **Q:** what is NOT production-ready? + +### Slide 15 — Future extensions +- **Key points:** API ingestion, template extraction + second-project validation. diff --git a/docs/presentation/atlas-question-bank.md b/docs/presentation/atlas-question-bank.md new file mode 100644 index 0000000..95bc4bd --- /dev/null +++ b/docs/presentation/atlas-question-bank.md @@ -0,0 +1,42 @@ +# Atlas Question Bank + +**Status:** CURRENT · Credible, evidence-bound answers to likely reviewer +questions. Each links to the authoritative source. + +- **Why batch instead of streaming?** The engineering problem is correctness, + governance, and recovery at controlled cost; batch makes idempotency, replay, + and reconciliation tractable and cheap. Streaming is a deliberate non-goal + (RISK-08). → [system-context](../reference-architecture/system-context.md). +- **Why BigQuery?** Serverless, partitioning/clustering, dry-run cost estimation, + `INFORMATION_SCHEMA.JOBS` for cost/perf evidence. → [cost model](../reference-architecture/cost-and-lifecycle-model.md). +- **Why dbt?** Declarative models, tests, contracts, lineage, and schema-evolution + hooks. → `dbt/atlas_dbt`, ADR-016/017. +- **Why Composer?** Managed Airflow parity with production orchestration without + running our own control plane. → ADR-005. +- **Why ephemeral Composer?** Cost control + drift avoidance; created for + acceptance, torn down while preserving durable evidence (INV-L7/O7). → ADR-010. +- **How is idempotency achieved?** Stable `batch_id`, create-only loads, dbt + incremental `unique_key`, global fact uniqueness (INV-D2/D3/D5). → ADR-006. +- **How are duplicates classified?** Within-batch duplicate vs cross-batch replay + are distinct scopes in `int_event_classification` (INV-D4). → ADR-006 amendment. +- **How are schema changes controlled?** Classified COMPATIBLE/CONDITIONAL/ + BREAKING/PROHIBITED; migrations immutable via checksum lock; breaking needs + impact evidence (INV-D8/G4/G5). → ADR-017. +- **How is rollback protected?** Schema-compatibility checked before rollback; + failed deploys never publish success (INV-L5/L6). → ADR-015. +- **What happens when data quality fails?** Publication is blocked; quality + results recorded; alert + runbook (INV-D7). → `game-day-results-sprint6`. +- **How are incidents detected?** Retries, dbt tests, guards, telemetry → alerts + mapped to runbooks (INV-O6). → `observability-runbook-sprint5`. +- **How is recovery verified?** SUCCESS gated on `VERIFIED` in `recovery_actions` + (INV-O3). → INC-S6-001. +- **What is genuinely production-ready?** The controls and evidence discipline: + credentialless CI, keyless deploy, immutable releases, governance gates, + observability, verified recovery. +- **What remains unproven?** Production-scale performance/cost, live least + privilege, live retention, multi-env promotion, streaming, template reuse. → + [unresolved-risks](../reference-architecture/unresolved-risks.md). +- **What would change at larger scale?** Slot management, incremental strategies, + partition/cluster tuning, real SLOs/alert thresholds, multi-env promotion. +- **What would be extracted into a template?** RC-01..22; the plan and acceptance + (not executed in Sprint 8). diff --git a/docs/recovery-runbook-sprint6.md b/docs/recovery-runbook-sprint6.md new file mode 100644 index 0000000..fb12133 --- /dev/null +++ b/docs/recovery-runbook-sprint6.md @@ -0,0 +1,132 @@ +# Atlas Recovery Runbook (Sprint 6) + +Companion to `docs/observability-runbook-sprint5.md` (detection and +diagnosis). This runbook covers what to *do* once a failure is diagnosed, and +how to prove the recovery worked. Model: ADR-014. + +## The decision tree + +```text +FAILURE DETECTED +│ +├─ 1. Is canonical data corrupted? +│ │ (raw/fact/mart rows wrong, duplicated, or partially loaded — not +│ │ merely a failed run that published nothing) +│ │ +│ ├─ NO ──► pick the cheapest state-restoring action: +│ │ • task failed transiently ............ RETRY_TASK +│ │ • run failed, data untouched ......... RERUN_BATCH (same batch id) +│ │ • permission removed ................. RESTORE_IAM (exact binding) +│ │ • bad release deployed ............... RESTORE_RELEASE (rollback path) +│ │ • monitor/alert state wrong .......... RESET_MONITOR +│ │ +│ └─ YES ─► contain first, repair second: +│ a. PAUSE_SCHEDULE / block publication (no new consumers of bad data) +│ b. QUARANTINE_BATCH (isolate the affected batch identity) +│ c. determine the repair boundary (batch? partition? table?) +│ │ +│ ├─ 2. Can the batch/partition be repaired safely? +│ │ ├─ YES ─► targeted repair: +│ │ │ REPAIR_PARTIAL_LOAD or REBUILD_PARTITION (bounded), +│ │ │ then reconcile, then RESUME_SCHEDULE +│ │ └─ NO ──► restore prior compatible runtime (RESTORE_RELEASE), +│ │ FORWARD_MIGRATION if schema demands it, +│ │ rebuild affected history, BACKFILL (bounded window), +│ │ reconcile, RESUME_SCHEDULE +``` + +Ordering rules: + +- Targeted repair before rebuild; rebuild before restore-and-backfill. +- `dbt build --full-refresh` is never the first response — it is gated by + `ATLAS_APPROVE_FULL_REFRESH` (S6-COST-003) precisely so nobody reaches for + it reflexively. +- Backfills are bounded to the 7-day policy window; + `ATLAS_APPROVE_UNBOUNDED_BACKFILL=true` requires a documented cost review + (S6-COST-002). +- Rollback across a `breaking`-flagged migration is refused by the deploy + engine (`ROLLBACK_INCOMPATIBLE`); recover forward (ADR-015). + +## Recording the recovery + +Every attempt gets a row in `atlas_ops.recovery_actions` **before** the +mutation starts: + +```python +from atlas.ops.recovery_actions import start_recovery_action, finalize_recovery_action + +rec = start_recovery_action( + recovery_id="s6--", + action_type="RERUN_BATCH", + scenario_id="S6-ING-001", + incident_id="", + pipeline_run_id="...", batch_id="...", operator="russell", + environment="atlas-dev", source_state="FAILED", target_state="SUCCESS", +) +# ... perform + verify ... +finalize_recovery_action(rec, status="SUCCESS", verification_status="VERIFIED") +``` + +`SUCCESS` without `VERIFIED` raises by design. A recovery that cannot pass +verification is finalized `PARTIAL` or `FAILED` — honestly. + +## The verification list (all must pass before SUCCESS) + +Run against the affected batch/partition: + +1. raw count exact (generator contract: 50,000 per standard batch) +2. accepted + rejected = raw +3. classification count = raw +4. fact count = accepted +5. fact event_ids unique +6. fact foreign keys resolve (users, countries) +7. mart totals reconcile +8. success marker present (success path only) +9. `pipeline_runs` row terminal and truthful +10. `task_events` telemetry complete for the run +11. `recovery_actions` row finalized with verification +12. monitor evaluations return to PASS +13. related incident closed/resolved +14. zero duplicate rows anywhere in the lineage + +Convenient wrapper: `scripts/atlas_step_runner.py validate_warehouse` with the +batch context executes checks 1–7 and persists them to +`atlas_ops.quality_results`. + +## Reconstruction procedure (RECONSTRUCT_AUDIT) + +When the finalizer or telemetry writes failed (S6-AIR-003, S6-OBS-002/008): + +1. Collect ground truth: Airflow task instance states (`airflow tasks + states-for-dag-run`), GCS object existence/generation, BigQuery row counts, + any `task_events` rows that did land, structured logs in `atlas-events`. +2. Derive the run's true terminal status from data state, not from wishes: + marker + reconciliation pass = SUCCESS; anything else = FAILED. +3. Upsert the corrected `pipeline_runs` row (idempotent MERGE keyed by + `pipeline_run_id`) and the missing `task_events` rows with + `timing_source='finalizer_reconciliation'` and honest + `timing_confidence` (`partial` or `none` — never invented `exact`). +4. Record the RECONSTRUCT_AUDIT recovery action; verify telemetry + completeness now passes; close the telemetry-incomplete incident. + +## Per-category quick reference + +| Diagnosis | First action | Verify with | +| --- | --- | --- | +| Missing/corrupt artifact | quarantine object, regenerate deterministically, rerun | checksum match + counts | +| Partial raw load | delete/replace only the affected batch partition rows, rerun load | exact count + uniqueness | +| Retry exhaustion | fix cause, rerun same batch id | full verification list | +| Worker interruption | let retry policy work; rerun if terminal | task_events attempts + counts | +| Finalizer failure | RECONSTRUCT_AUDIT (above) | completeness PASS | +| dbt test/RI/duplicate failure | quarantine offending rows (fixture or batch), rerun | dbt tests green + reconciliation | +| Incremental corruption | targeted partition rebuild in place | non-target partitions unchanged | +| IAM loss | restore the exact removed binding only | probe operation + policy diff vs baseline | +| Failed deployment/smoke | RESTORE_RELEASE to last SUCCESS deployment | rollback smoke 12/12 | +| Rollback failure | secondary RESTORE_RELEASE / forward fix | deployments audit + smoke | +| Cost guard trip | fix the query/window; never raise ceilings casually | dry-run estimate below ceiling | + +## Escalation + +Primary operator: the primary operator. Escalation: repository owner/designated +reviewer. External escalation only through the approved notification channel. +Preserve evidence before changing any firing policy or deleting any fixture. diff --git a/docs/reference-architecture/README.md b/docs/reference-architecture/README.md new file mode 100644 index 0000000..11e89a1 --- /dev/null +++ b/docs/reference-architecture/README.md @@ -0,0 +1,41 @@ +# Atlas Reference-Architecture Package + +**Status:** CURRENT · A curated *map* over the proven Atlas system. It summarizes +stable decisions and **links** to the authoritative detail (code, ADRs, runbooks, +validation reports); it does not duplicate them. Start at +[`../../START_HERE.md`](../../START_HERE.md). + +## Package contents + +| Document | Purpose | +| --- | --- | +| [reference-manifest.yml](reference-manifest.yml) | machine-readable index (validated by `atlas.reference.validate`) | +| [system-context.md](system-context.md) | problem, actors, boundaries, context diagram | +| [architecture-overview.md](architecture-overview.md) | data / control / operational / governance flows | +| [architecture-invariants.md](architecture-invariants.md) | properties that must remain true + enforcement | +| [component-catalog-reusable.md](component-catalog-reusable.md) | reusable-candidate components (RC-01..22) | +| [component-catalog-atlas-specific.md](component-catalog-atlas-specific.md) | project-specific components (AC-01..13) | +| [interfaces-and-contracts.md](interfaces-and-contracts.md) | stable boundaries + enforcement | +| [extension-points.md](extension-points.md) | how to extend safely | +| [operating-model.md](operating-model.md) | ownership + authority + cadence | +| [security-and-identity-model.md](security-and-identity-model.md) | identities, WIF, least-privilege status | +| [reliability-and-recovery-model.md](reliability-and-recovery-model.md) | detect→…→prevent, replay, rollback | +| [observability-model.md](observability-model.md) | audit, logs, metrics, alerts, limitations | +| [cost-and-lifecycle-model.md](cost-and-lifecycle-model.md) | cost controls, retention, proven vs not | +| [evidence-index.md](evidence-index.md) | claim→evidence map (human view) | +| [capability-evidence-map.md](capability-evidence-map.md) | competency domains + honest framing | +| [unresolved-risks.md](unresolved-risks.md) | risk register + Sprint 7 blocked-gate disposition | + +## Rules this package follows + +- Summarize stable decisions; link to detailed sources. +- Distinguish implementation from proposal, static from live, current from + historical, and always expose limitations. +- Never claim reusable-template status (Atlas is a reference architecture). + +## Validate the package + +```bash +python -m atlas.reference.validate +bash scripts/validate_ci.sh --mode static # gate_reference_handoff +``` diff --git a/docs/reference-architecture/architecture-invariants.md b/docs/reference-architecture/architecture-invariants.md new file mode 100644 index 0000000..6c29fa4 --- /dev/null +++ b/docs/reference-architecture/architecture-invariants.md @@ -0,0 +1,126 @@ +# Architecture Invariants + +**Status:** CURRENT · **Audience:** engineer, agent, reviewer. These are the +properties that must remain true. Each has an id, statement, reason, enforcement, +evidence, failure consequence, safe-change procedure, and related ADRs. An agent +must confirm no invariant is silently broken (see +[agent-task-protocol.md](../handoff/agent-task-protocol.md)). + +Format per entry: **INV — statement** · *why* · **enforced by** · *evidence* · +**if broken** · *safe change* · ADRs. + +## DATA + +- **INV-D1 — Raw artifacts are immutable and run-scoped.** *Reproducibility & + audit.* Enforced by create-only GCS upload paths + loader. Evidence: + `validation-report-sprint1.md`. If broken: history becomes unauditable. Safe + change: new run prefix, never overwrite. ADR-003. +- **INV-D2 — `batch_id` is stable data identity; `pipeline_run_id` is one + execution.** *Separates data from run.* Enforced by `src/atlas/batch` run + context + `atlas_ops.pipeline_runs`. Evidence: `validation-report-sprint3.md`, + ADR-006/007. If broken: reruns duplicate or lose data. Safe change: ADR + tests. +- **INV-D3 — Exact reruns are idempotent.** *Retries/backfills must be safe.* + Enforced by dbt incremental `unique_key` + create-only loads. Evidence: dbt + `test_duplicate_ranking_keeps_latest_canonical`. If broken: double counting. + Safe change: preserve merge key. ADR-006. +- **INV-D4 — Within-batch duplicates and cross-batch replay are distinct.** + *Anomaly profile vs replay must not conflate.* Enforced by + `int_event_classification` (`within_batch_duplicate_rank` vs `duplicate_rank`, + `duplicate_scope`). Evidence: dbt `test_cross_batch_replay_preserves_first_seen`. + If broken: Sprint 6 INC-S6-001 recurs. Safe change: ADR-006 amendment procedure. +- **INV-D5 — `fct_events` grain is one row per `event_id`.** *Global uniqueness.* + Enforced by dbt uniqueness test + merge key. Evidence: `core.yml` tests. If + broken: all downstream metrics wrong. Safe change: deliberate ADR only. ADR-006. +- **INV-D6 — accepted + rejected reconciles to raw under declared semantics.** + *No silent data loss.* Enforced by reconciliation tests + `assert_source_*`. + Evidence: `validation-report-sprint2.md`. If broken: data leakage. Safe change: + update contract + tests together. +- **INV-D7 — Quality failure prevents publication.** *No bad data downstream.* + Enforced by DAG quality gate + `atlas_ops.quality_results`. Evidence: + game-day S6-DBT-002. If broken: consumers see bad data. Safe change: keep gate + before publish step. +- **INV-D8 — Applied migrations are immutable.** *Deterministic schema history.* + Enforced by `gate_schema_compatibility` + `sql/migrations/checksums.lock`. + Evidence: `test_migration_checksum_tamper_is_detected`. If broken: drift. Safe + change: add a new migration, never edit an applied one. ADR-017. + +## DELIVERY + +- **INV-L1 — PR CI remains credentialless.** *Untrusted PRs never touch GCP.* + Enforced by `validate_ci.sh --mode static` + workflow separation. Evidence: + green PR runs. If broken: supply-chain risk. Safe change: keep cloud in trusted + workflows only. ADR-008. +- **INV-L2 — Trusted GCP actions use keyless WIF.** *No service-account keys.* + Enforced by workflow OIDC + `gate_security_policy` (no key creation). Evidence: + ADR-009, `validation-report-sprint4.md`. If broken: credential leakage. Safe + change: never add SA keys; don't weaken trust conditions. ADR-009/018. +- **INV-L3 — Releases are immutable.** Enforced by content-pinned bundles + (`build_deployment_bundle.sh`). Evidence: `deployment-catalog-sprint4.md`. If + broken: non-reproducible deploys. Safe change: new bundle per release. +- **INV-L4 — Migrations run before deployment validation.** Enforced by deploy + sequence. Evidence: sprint4/6 validation reports. If broken: schema/code skew. + Safe change: preserve ordering. ADR-010. +- **INV-L5 — Smoke validation gates success; a failed deployment cannot publish + success.** Enforced by `validate_atlas_deployment.sh` + `atlas_ops.deployments`. + Evidence: sprint4 incident report. If broken: false green. Safe change: keep + smoke gate mandatory. +- **INV-L6 — Rollback checks schema compatibility.** Enforced by + `rollback_atlas.sh` + schema-version handling. Evidence: ADR-015. If broken: + rollback corrupts schema. Safe change: keep compatibility check. +- **INV-L7 — Composer is ephemeral for evidence capture** (unless a future ADR + changes it). Enforced by create/teardown procedure + `ATLAS_APPROVE_*`. + Evidence: sprint6 teardown record. If broken: cost + drift. Safe change: ADR. + ADR-005/010. + +## OPERATIONS + +- **INV-O1 — Operational history is durable.** `atlas_ops.*` tables persist + through teardown. Evidence: sprint6 validation §8. Safe change: never expire + audit tables (INV-G7). +- **INV-O2 — Failures are correlated by identifiers.** `pipeline_run_id` + + `batch_id` on every event/log. Evidence: ADR-011. Safe change: keep correlation + ids in the log contract. +- **INV-O3 — Recovery success requires verification.** SUCCESS gated on + `VERIFIED` in `atlas_ops.recovery_actions`. Evidence: INC-S6-001. ADR-014. +- **INV-O4 — Fault injection is disabled by default.** Enforced by + `gate_failure_injection` + config. Evidence: `test_failure_injection.py`. If + broken: accidental production faults. ADR-013. +- **INV-O5 — Cost guards execute before expensive behavior.** Enforced by + `cost_guard` dry-run-first. Evidence: `cost-guard-block.txt`. ADR-020. +- **INV-O6 — Alerts map to runbooks.** Enforced by `gate_reference_handoff` + (alert→runbook) + observability config. Evidence: `alert-catalog-sprint5.md`. +- **INV-O7 — Teardown must not destroy required evidence.** Evidence: sprint6 + teardown preserved audit tables + recovery row. Safe change: teardown allow-list. + +## GOVERNANCE + +- **INV-G1 — Model governance metadata has one source of truth** (dbt `meta` for + models, registry for non-dbt). Enforced by `gate_governance` duplicate check. + ADR-016. +- **INV-G2 — All major assets have owners.** Enforced by `gate_governance`. +- **INV-G3 — All major models declare grain.** Enforced by `gate_governance`. +- **INV-G4 — Schema changes are classified** (COMPATIBLE/CONDITIONAL/BREAKING/ + PROHIBITED). Enforced by `schema_check` + `gate_schema_compatibility`. ADR-017. +- **INV-G5 — Breaking changes require migration + consumer-impact evidence.** + Enforced by change-record requirement. Evidence: `test_schema_check.py`. ADR-017. +- **INV-G6 — Deprecation follows a controlled lifecycle.** Enforced by + `registry.deprecation_errors`. Evidence: `test_deprecation.py`. ADR-017. +- **INV-G7 — Permanent evidence cannot receive transient retention.** Enforced by + `retention.validate_retention_config`. Evidence: `test_retention.py`. ADR-019. +- **INV-G8 — Secrets are never written into evidence.** Enforced by `secret_scan` + + `gate_security_policy` + `validate_public_extraction.py`. Evidence: + `security-review-sprint7.md`. ADR-018. + +## REFERENCE (Sprint 8) + +- **INV-R1 — Repository instructions must not depend on prior conversations.** + Enforced by `gate_reference_handoff` (forbidden-phrase scan). Evidence: + clean-clone + handoff results. +- **INV-R2 — Current documentation identifies its verification commit.** Enforced + by manifest `last_verified_commit` + evidence `verification_commit`. Enforced by + `atlas.reference.validate`. +- **INV-R3 — Claims link to evidence.** Enforced by evidence index + + `atlas.reference.validate`. +- **INV-R4 — Blocked work remains visibly blocked.** Enforced by evidence-index + BLOCKED checks + `gate_reference_handoff`. Evidence: `unresolved-risks.md`. +- **INV-R5 — Reference architecture must not claim template status.** Enforced by diff --git a/docs/reference-architecture/architecture-overview.md b/docs/reference-architecture/architecture-overview.md new file mode 100644 index 0000000..57c033f --- /dev/null +++ b/docs/reference-architecture/architecture-overview.md @@ -0,0 +1,89 @@ +# Architecture Overview + +**Status:** CURRENT · **Audience:** engineer, reviewer · **Source of truth:** +code + `docs/architecture-sprint{1..7}.md` + ADRs. Curated map; links to detail. + +Atlas has four cooperating flows. Each is implemented and evidenced; none is +aspirational. Scale is synthetic — this is production-*oriented* evidence, not +proof of production traffic volume. + +## 1. Data flow + +``` +event generation src/atlas/generator (deterministic, seeded anomalies) + → immutable raw artifact JSONL, content-addressed per run + → Cloud Storage run-scoped, immutable landing paths + → BigQuery raw atlas_raw.events (partitioned by date, clustered) + → dbt staging stg_events (typed, normalized) + → classification int_event_classification (accept/reject + dup/replay) + → accepted / rejected int_accepted_events / int_rejected_events + → fact + dimensions fct_events (1 row/event_id), dim_users, dim_countries + → marts mart_daily_event_metrics + → operational evidence atlas_ops.* (audit, quality, deployments, ...) +``` + +Invariants: raw is immutable and run-scoped; `fct_events` grain is one row per +`event_id`; accepted + rejected reconciles to raw under declared semantics; +quality failure blocks publication. Details: +[architecture-invariants.md](architecture-invariants.md), ADR-003/006. + +## 2. Control flow (CI/CD) + +``` +pull request + → credentialless CI scripts/validate_ci.sh (21 gates, no GCP creds) + → trusted integration WIF-authenticated workflow (no service-account keys) + → immutable release build_deployment_bundle.sh (content-pinned bundle) + → migration validation apply_atlas_migrations.sh + checksums.lock + → Composer deployment deploy_atlas_release.sh (ephemeral environment) + → smoke validation validate_atlas_deployment.sh (gates success) + → rollback OR success rollback_atlas.sh (schema-compatibility checked) +``` + +Invariants: PR CI stays credentialless; releases are immutable; migrations run +before deployment validation; a failed deployment cannot publish success; +rollback checks schema compatibility. Details: ADR-008/009/010, +[ci-cd-runbook-sprint4.md](../ci-cd-runbook-sprint4.md). + +## 3. Operational flow + +``` +pipeline telemetry structured JSON logs w/ correlation ids + → operational audit atlas_ops.pipeline_runs / task_events / quality_results + → Cloud Logging atlas-events log + linked BigQuery dataset + → metrics custom + log-based metrics (observability/metrics) + → alerts Cloud Monitoring policies (observability/alerts) + → investigation runbook-driven diagnosis + → recovery action atlas_ops.recovery_actions (targeted repair) + → verification validate_warehouse; SUCCESS gated on VERIFIED + → prevention evidence incident reports + follow-ups +``` + +Invariants: operational history is durable; failures correlate by +`pipeline_run_id`/`batch_id`; recovery success requires verification; alerts map +to runbooks. Details: ADR-011/014, [observability-runbook-sprint5.md](../observability-runbook-sprint5.md), +[recovery-runbook-sprint6.md](../recovery-runbook-sprint6.md). + +## 4. Governance flow + +``` +asset metadata dbt meta.governance + governance/non_dbt_assets.yml + → contract validation gate_governance (owners, grain, classification, ...) + → schema compatibility atlas.governance.schema_check + baseline manifest + → lineage impact atlas.governance.lineage / impact + → CI enforcement 5 offline gates in validate_ci.sh + → controlled change change record + consumer-impact evidence +``` + +Invariants: one source of truth for model governance; owners+grain required; +schema changes classified; breaking changes need migration + impact; permanent +evidence cannot receive transient retention; secrets never in evidence. Details: +ADR-016–020, [architecture-sprint7.md](../architecture-sprint7.md). + +## What did NOT change across sprints + +The Sprint 1 data plane shape (generate → GCS → raw → dbt → marts) is stable. +Later sprints added orchestration, delivery, observability, resilience, and +governance *around* it without redesigning it. The only data-plane semantics +change in Sprint 7 was duplicate/replay *classification* (ADR-006 amendment) — +the `fct_events` grain was preserved. diff --git a/docs/reference-architecture/capability-evidence-map.md b/docs/reference-architecture/capability-evidence-map.md new file mode 100644 index 0000000..ffbf069 --- /dev/null +++ b/docs/reference-architecture/capability-evidence-map.md @@ -0,0 +1,64 @@ +# Capability & Evidence Map + +**Status:** CURRENT · **Audience:** reviewer, interviewer. Maps Atlas evidence to +engineering competency domains. Each capability lists evidence, level +demonstrated, limitation, and what the next level would require. This does **not** +convert project evidence into inflated seniority claims — see +[engineering-evidence-ledger.md](../handoff/engineering-evidence-ledger.md) for +the full ledger. + +Level scale: **Demonstrated** (proven in Atlas) · **Partial** (shown at synthetic +scale/one path) · **Not shown**. + +## SQL & warehousing + +Advanced SQL, grain, dedup, late data, incremental, partitioning, clustering, +dimensional modeling, cost control — **Demonstrated** (dbt models, `fct_events` +grain, incremental, partition pruning). *Limitation:* synthetic 50k rows. +*Next level:* production volume + slot/cost tuning under real load. + +## dbt + +Sources, staging, intermediate, facts, dimensions, marts, tests, contracts, docs, +incremental, schema evolution — **Demonstrated** (`dbt/atlas_dbt` + Sprint 7 +schema checker). *Limitation:* single project. *Next level:* dbt mesh / multi-project. + +## GCP + +Cloud Storage, BigQuery, IAM, WIF, Composer, Logging, Monitoring, cost +stewardship — **Demonstrated** (Sprints 1–7 live evidence). *Limitation:* single +project, ephemeral Composer, one blocked IAM reduction. *Next level:* multi-env +promotion + live least-privilege proof. + +## Pipeline engineering + +Ingestion, idempotency, retries, backfills, orchestration, audit, failure +handling, recovery — **Demonstrated** (Sprints 1/3/6 + verified recovery). +*Limitation:* batch only. *Next level:* streaming/event-driven. + +## Software engineering + +Git, PRs, CI, lint, typing, tests, packaging, immutable releases, rollback — +**Demonstrated** (21-gate CI, 269-test unit+Airflow gate / 282 across all +suites, immutable bundles, rollback). +*Limitation:* single repo. *Next level:* multi-service release orchestration. + +## Governance & security + +Ownership, contracts, lineage, impact, schema compatibility, least privilege, +secrets, retention — **Demonstrated** (Sprint 7). *Limitation:* least privilege +not proven at permission level live (RISK-01/02, BLOCKED). *Next level:* execute +the IAM reduction + negative test. + +## Operations + +Alerts, runbooks, incidents, postmortems, recovery, verification, recurrence +prevention — **Demonstrated** (Sprints 5/6 drills + incident reports). +*Limitation:* representative live subset. *Next level:* sustained on-call at scale. + +## Honest framing + +Atlas is strong, reproducible **evidence of capability at a deliberately modest +synthetic scale**. It is not evidence of production-scale operation, enterprise +governance, or senior tenure. Blocked and unproven items are listed in +[unresolved-risks.md](unresolved-risks.md) and never counted as demonstrated. diff --git a/docs/reference-architecture/component-catalog-atlas-specific.md b/docs/reference-architecture/component-catalog-atlas-specific.md new file mode 100644 index 0000000..bfdc71f --- /dev/null +++ b/docs/reference-architecture/component-catalog-atlas-specific.md @@ -0,0 +1,52 @@ +# Component Catalog — Atlas-Specific + +**Status:** CURRENT · **Audience:** engineer, reviewer, template author. These +components encode Atlas's synthetic domain or environment. A new project built +from the template would **replace** them. For each: why it is specific, what a +new project replaces it with, the interface that must stay stable, and which +reusable component depends on it. + +Fields: why specific · new-project replacement · stable interface · reusable +dependency. + +- **AC-01 synthetic event schema** (`src/atlas/generator`, `config/atlas.yaml`). + Why: models a fictional user-action stream. Replace: real source schema. + Stable interface: the raw-table column contract consumed by dbt sources. + Depends: RC-09 logging, RC-16 governance (contract). +- **AC-02 anomaly injection profile** (`config/anomaly_profile.yaml`). Why: seeded + test anomalies (50 within-batch duplicates, etc.). Replace: real data-quality + expectations. Interface: `assert_source_anomaly_profile` inputs. Depends: RC-11. +- **AC-03 event identity rules** (event_id derivation). Why: synthetic id scheme. + Replace: source primary key. Interface: INV-D2/D5. Depends: RC-17 schema check. +- **AC-04 acceptance/rejection semantics** (`int_event_classification`). Why: + Atlas-defined validity rules. Replace: domain validity rules. Interface: INV-D6 + reconciliation. Depends: RC-16. +- **AC-05 `fct_events` grain** (one row/event_id). Why: Atlas fact definition. + Replace: new fact grain. Interface: INV-D5. Depends: RC-17, RC-18. +- **AC-06 specific dimensions & marts** (`dim_users`, `dim_countries`, + `mart_daily_event_metrics`). Why: Atlas domain model. Replace: new dims/marts. + Interface: dbt contracts. Depends: RC-18 lineage. +- **AC-07 Atlas dataset names** (`atlas_raw`, `atlas_core`, `atlas_ops`, marts). + Why: naming. Replace: `dataset_prefix` parameter. Interface: everywhere. + Depends: RC-05/06/10/16. +- **AC-08 Atlas service-account names** (`atlas-github-integration`, + deployer/runtime SAs). Why: identity naming. Replace: `service_account_prefix`. + Interface: WIF bindings. Depends: RC-03. +- **AC-09 Atlas alert thresholds** (`observability/alerts/*.json`, + `config/observability.yaml`). Why: tuned to 50k synthetic scale. Replace: + real SLOs. Interface: RC-12 alert→runbook. Depends: RC-11/12. +- **AC-10 GCP project reference** (`example-gcp-project`). Why: this project. + Replace: `gcp_project_id`. Interface: all cloud calls. Depends: RC-02/03. +- **AC-11 sample processing dates** (`2026-07-15`, `atlas-20260717`, ...). Why: + demo batches. Replace: real schedule dates. Interface: DAG params. Depends: RC-11. +- **AC-12 Atlas dashboard content** (`observability/dashboards/`). Why: Atlas + metrics layout. Replace: project dashboard. Depends: RC-11. +- **AC-13 Atlas-specific failure scenarios** (`config/failure_scenarios.yaml`). + Why: tuned to Atlas pipeline. Replace: project scenarios. Depends: RC-14. + +## Boundary rule + +No component may be reclassified as reusable merely because it is written in +Python or YAML. If replacing the synthetic domain would require rewriting the +component's *logic* (not just its configuration), it belongs here. The +naming into parameters and keeps AC-01..AC-06 as project-supplied contracts. diff --git a/docs/reference-architecture/component-catalog-reusable.md b/docs/reference-architecture/component-catalog-reusable.md new file mode 100644 index 0000000..7b47b1e --- /dev/null +++ b/docs/reference-architecture/component-catalog-reusable.md @@ -0,0 +1,79 @@ +# Component Catalog — Reusable + +**Status:** CURRENT · **Audience:** engineer, reviewer, template author. These +components are candidates for a future reusable template (see +**reusable** only if its value is independent of Atlas's synthetic domain — not +merely "it is Python/YAML". Each entry records the extraction action required to +generalize it. No component here is claimed to be *already* a template. + +Fields: purpose · implementation · dependencies · configuration surface · +hardcoded Atlas assumptions · security boundary · tests · evidence · limitations +· template-extraction action. + +## CI & delivery + +- **RC-01 canonical CI entry point** — one script all actors run. + `scripts/validate_ci.sh` (`--mode`, `--group`). Deps: ruff/mypy/pytest/yamllint/ + shellcheck/dbt. Config: gate groups, `ATLAS_CI_GATE_GROUP`. Atlas assumptions: + gate list, dbt project path. Security: credentialless in static mode. Tests: the + gates themselves. Evidence: green CI runs. Limits: gate set is Atlas-tuned. + Extraction: parameterize project paths + gate registry. +- **RC-02 credentialless PR validation** — untrusted PRs never touch GCP. + Workflow split + static mode. Extraction: keep workflow topology, swap names. + ADR-008. +- **RC-03 WIF deployment authentication** — keyless GitHub→GCP. + `scripts/bootstrap_github_wif.sh`, workflow OIDC. Atlas assumptions: SA names, + project id, pool id. Extraction: parameterize identity/project. ADR-009. +- **RC-04 immutable release bundles** — content-pinned deploy artifact. + `build_deployment_bundle.sh`. Extraction: parameterize bundle contents. +- **RC-05 migration ledger + checksum lock** — applied migrations immutable. + `sql/migrations/` + `checksums.lock` + `gate_schema_compatibility`. Extraction: + keep mechanism, swap DDL. ADR-017. +- **RC-06 deployment audit** — `atlas_ops.deployments` + `apply_atlas_migrations.sh`. + Extraction: keep schema, rename dataset. +- **RC-07 smoke validation contract** — `validate_atlas_deployment.sh`. Extraction: + parameterize checks. +- **RC-08 rollback controls** — `rollback_atlas.sh` w/ schema-compat check. ADR-010/015. + +## Observability & operations + +- **RC-09 structured logging contract** — correlation ids, redaction. + `src/atlas/logging`, `src/atlas/observability/logging`. Extraction: keep + contract, swap log/dataset names. ADR-011. +- **RC-10 operational audit tables** — `atlas_ops.{pipeline_runs,task_events, + quality_results,monitor_evaluations,deployments,schema_migrations,recovery_actions}`. + Extraction: keep schemas, rename dataset. +- **RC-11 observability monitor pattern** — `atlas_observability_monitor` DAG + + `monitor_evaluations`. Extraction: parameterize checks/thresholds. +- **RC-12 alert ↔ runbook linkage** — `observability/alerts/*.json` map to runbook + sections. Extraction: keep linkage rule (INV-O6). +- **RC-13 recovery-action audit** — verified targeted repair. `recovery_actions`. + ADR-014. +- **RC-14 failure-injection safeguards** — disabled by default, gated. + `src/atlas/failure_injection`, `gate_failure_injection`. ADR-013. +- **RC-15 cost guards** — dry-run-first ceilings. `config/cost_controls.yaml`, + `src/atlas/observability/cost_guard.py`. Extraction: parameterize ceilings. + ADR-020. + +## Governance + +- **RC-16 governance registry** — one source of truth + catalog + drift check. + `src/atlas/governance/{registry,catalog}.py`, `governance/*.yml`. ADR-016. +- **RC-17 schema compatibility checker** — `schema_check.py` + baseline manifest. + ADR-017. +- **RC-18 lineage + impact analysis** — repository-artifact lineage. + `lineage.py`/`impact.py`. Extraction: keep dbt-ref parsing, swap consumers. +- **RC-19 retention validation** — `retention.py` + `governance/retention.yml`. + ADR-019. +- **RC-20 evidence-index pattern** — claim→evidence map + validator. + `governance/generated/evidence-index.json`, `atlas.reference.validate`. +- **RC-21 handoff validation** — `gate_reference_handoff` + `validate_clean_clone.sh`. +- **RC-22 public-extraction validator** — secret/PII disposition scanner. + `scripts/validate_public_extraction.py`, `config/public_extraction_manifest.yml`. + +## Reuse guidance + +Every reusable component carries **Atlas assumptions** (dataset names, SA names, +lists exactly which of these become parameters. Until that extraction is +performed and validated by a separate generated project, these are reusable +*candidates*, not a template. diff --git a/docs/reference-architecture/cost-and-lifecycle-model.md b/docs/reference-architecture/cost-and-lifecycle-model.md new file mode 100644 index 0000000..d04beb5 --- /dev/null +++ b/docs/reference-architecture/cost-and-lifecycle-model.md @@ -0,0 +1,47 @@ +# Cost and Lifecycle Model + +**Status:** CURRENT · **Audience:** operator, reviewer. Authoritative detail: +[cost-review-sprint7.md](../cost-review-sprint7.md), +[performance-review-sprint7.md](../performance-review-sprint7.md), +[retention-policy-sprint7.md](../retention-policy-sprint7.md), ADR-012/019/020. + +## Controls (config-driven) + +`config/cost_controls.yaml` defines per-environment limits, enforced by +`src/atlas/observability/cost_guard.py` and `gate_performance_cost`: + +- **max_query_bytes / max_performance_suite_bytes** — hard byte ceilings. +- **require_partition_filter_assets** — queries over raw must be bounded (INV-O5). +- **max_backfill_days**, **full_refresh_requires_approval**. +- **temporary_dataset_ttl_hours**, **temporary_object_ttl_days**, + **composer_max_lifecycle_hours**, **log_retention_days**, + **release_retention_policy**. + +## Estimation-first enforcement (proven, $0) + +`python -m atlas.observability.cost_guard estimate` and `check-partition-filter` +run a **dry-run first**, compare against the ceiling, and refuse over-limit +execution before any spend. Proven live at $0 in +`docs/evidence-sprint7/cost-guard-block.txt`: an unbounded `atlas_raw.events` +scan is blocked by both the partition-filter guard and the estimate ceiling. + +## Lifecycle & retention + +- **Composer** — ephemeral; created for acceptance, torn down under + `ATLAS_APPROVE_TEARDOWN` (INV-L7). +- **Log retention** — bounded by `log_retention_days`. +- **Temporary resources** — CI datasets/GCS prefixes carry TTLs. +- **Release retention** — validated releases retained per `release_retention_policy`. +- **Operational evidence** — permanent audit tables cannot receive transient + retention (INV-G7); disposal is a validated dry-run plan. + +## Cost claims — proven vs not proven (honest) + +- **Proven:** dry-run performance baseline (all 9 queries << 1 GiB); partition + pruning (2.3 MB bounded vs 12.7 MB unbounded); $0 cost-guard block; negligible + permanent footprint at synthetic scale. +- **NOT proven (BLOCKED):** the executed/billed performance suite requires + `ATLAS_APPROVE_PERFORMANCE_TESTS` (+ `ATLAS_MAX_PERFORMANCE_TEST_BYTES`); live + retention/expiration application requires `ATLAS_APPROVE_RETENTION_MUTATION`. + See [unresolved-risks.md](unresolved-risks.md) RISK-03/04. We do **not** claim + production-scale cost/performance from a 50k-row synthetic dataset. diff --git a/docs/reference-architecture/evidence-index.md b/docs/reference-architecture/evidence-index.md new file mode 100644 index 0000000..1219feb --- /dev/null +++ b/docs/reference-architecture/evidence-index.md @@ -0,0 +1,46 @@ +# Evidence Index + +**Status:** CURRENT · **Audience:** reviewer, agent. Human-readable view of the +machine-readable claim ledger +[`governance/generated/evidence-index.json`](../../governance/generated/evidence-index.json), +validated by `python -m atlas.reference.validate`. Every major claim maps to a +path + evidence type + live/static/blocked status. The validator fails if an +evidence path is missing, a claim id is duplicated, a LIVE claim has only +documentation evidence, a blocked claim is presented as complete, or a +verification commit is absent. + +## How to reproduce + +```bash +export PYTHONPATH=src # atlas.* modules live under src/ +python -m atlas.reference.validate # manifest + evidence index +bash scripts/validate_ci.sh --mode static # includes gate_reference_handoff +``` + +## Claim summary (30 claims) + +| status | count | claims | +| --- | --- | --- | +| PROVEN_LIVE | 8 | ingest/load, WIF, release, alerting drill, incident, recovery, quality gate | +| PROVEN_STATIC | 7 | CI, rollback ADR, logging, monitoring, lineage, governance SoT, cost block | +| PROVEN_TEST | 8 | generation, dbt, orchestration, retries, schema check, migration immutability, impact, security, retention | +| DRY_RUN | (within static) | performance baseline, cost block | +| PLANNED | 2 | clean-clone (CLM-26), independent handoff (CLM-27) — recorded after runs | +| BLOCKED | 3 | live IAM (CLM-28), billed perf (CLM-29), live retention (CLM-30) | + +## Live vs static (must stay separated) + +- **Proven live** (real cloud runs recorded in validation/incident reports): + immutable ingest, BigQuery load, WIF deploy, immutable release, alerting drill, + incident diagnosis, verified recovery, quality-gate block. +- **Proven static / by tests** (offline gates + 269-test unit+Airflow gate, + 240 unit / 29 Airflow, plus dbt tests): + generation determinism, dbt transforms, orchestration, schema compatibility, + migration immutability, lineage/impact, governance source-of-truth, security + scanners, retention validation. +- **Dry-run ($0):** performance baseline, cost-guard block. +- **Blocked (not executed):** live IAM reduction + tests, billed performance + suite, live retention application — see [unresolved-risks.md](unresolved-risks.md). + +The evidence index deliberately does **not** upgrade any dry-run or static claim +to "live", and never marks a blocked claim complete. diff --git a/docs/reference-architecture/extension-points.md b/docs/reference-architecture/extension-points.md new file mode 100644 index 0000000..8f00a7d --- /dev/null +++ b/docs/reference-architecture/extension-points.md @@ -0,0 +1,40 @@ +# Extension Points + +**Status:** CURRENT · **Audience:** engineer, agent. How to extend Atlas safely. +Sprint 8 **documents** these; it does not implement them. For each: supported use +case, files likely affected, invariants that must remain true, required tests, +required documentation, required evidence, rollback considerations, and the +common unsafe shortcut. Always follow the +[agent-task-protocol.md](../handoff/agent-task-protocol.md). + +| Extension | Files likely affected | Invariants to keep | Tests | Unsafe shortcut to avoid | +| --- | --- | --- | --- | --- | +| New batch source | `src/atlas/generator|ingestion`, `sources.yml`, `config/atlas.yaml` | D1, D2, D6, G1–G3 | ingestion + reconciliation | reusing `event_id` semantics blindly | +| API ingestion source | new `src/atlas/ingestion/.py`, source YAML, governance asset | D1, D2, D6, L1 | ingestion unit + contract | calling the API inside PR CI (breaks L1 credentialless) | +| New dbt model | `dbt/atlas_dbt/models/**`, model YAML `meta.governance` | G1–G4, D6 | dbt tests + `gate_governance` | omitting `meta.governance` (fails gate) | +| New fact | `models/core`, `core.yml`, baseline manifest | D5, G3–G5 | uniqueness + reconciliation | changing an existing fact grain | +| New dimension | `models/core`, `core.yml` | G1–G3 | dbt tests | many-to-many join inflating fact | +| New mart | `models/marts`, `marts.yml`, `consumers.yml` | G1–G3, lineage | dbt tests + `gate_lineage_impact` | reading raw directly, skipping layers | +| New data-quality check | dbt tests / `assert_*`, `config/anomaly_profile.yaml` | D6, D7 | the check itself | asserting post-global-dedup for batch anomalies (INC-S6-001) | +| New Airflow task | `dags/`, `src/atlas/batch`, `atlas_step_runner.py` | O1–O2, L4 | `dag_import` + airflow tests | task without audit/telemetry emission | +| New alert | `observability/alerts/*.json`, `config/observability.yaml`, runbook | O6 | `observability_config` + `gate_reference_handoff` | alert with no runbook mapping (breaks O6) | +| New recovery action | `src/atlas/ops`, `recovery_actions` migration | O3 | `test_recovery_actions.py` | marking SUCCESS without VERIFIED | +| New failure scenario | `config/failure_scenarios.yaml`, `src/atlas/failure_injection` | O4 | `test_failure_injection.py` | enabling injection by default | +| New governance asset | `governance/non_dbt_assets.yml` or dbt `meta` | G1–G3 | `gate_governance` | duplicating an asset in two sources (breaks G1) | +| New schema version | `sql/migrations/NNN_*.sql`, `checksums.lock`, baseline | D8, G4–G5 | `gate_schema_compatibility` | editing an applied migration (breaks D8) | +| New environment | `config/cost_controls.yaml`, deploy config | L1–L7, O5 | `gate_performance_cost` | copying prod creds into CI | +| Future event-driven pipeline | new module + ADR | D1–D8 preserved for batch | new + regression | replacing batch semantics without an ADR | + +## Rules for every extension + +1. Required **documentation**: update the relevant reference doc + an ADR if a + real decision is made. +2. Required **evidence**: add to the [evidence index](evidence-index.md) with the + correct live/static/blocked status. +3. Required **rollback**: state how to revert; for schema, only additive/rollback- + eligible changes without a migration+approval. +4. Confirm **no invariant** (see [architecture-invariants.md](architecture-invariants.md)) + is silently broken; run `bash scripts/validate_ci.sh --mode static`. + +A worked example for **API ingestion** is the independent-handoff assignment +(item 12) — see [independent-handoff-assignment.md](../handoff/independent-handoff-assignment.md). diff --git a/docs/reference-architecture/interfaces-and-contracts.md b/docs/reference-architecture/interfaces-and-contracts.md new file mode 100644 index 0000000..318e97f --- /dev/null +++ b/docs/reference-architecture/interfaces-and-contracts.md @@ -0,0 +1,36 @@ +# Interfaces and Contracts + +**Status:** CURRENT · **Audience:** engineer, agent. The stable boundaries a +change must respect. Authoritative detail lives in the linked sources; this is +the index of interfaces and where each is enforced. + +| Interface | Where defined | Enforced by | Notes | +| --- | --- | --- | --- | +| Event schema | `config/atlas.yaml`, `src/atlas/generator` | generator tests | Atlas-specific (AC-01) | +| File format (JSONL) | `src/atlas/ingestion` | ingestion tests | one event per line, immutable | +| Storage paths | `src/atlas/ingestion`, loader | create-only upload | run-scoped, immutable (INV-D1) | +| Raw-table schema | `sql/` DDL, `atlas_raw.events` | `sql_migrations` gate | partitioned/clustered | +| dbt source boundary | `dbt/atlas_dbt/models/sources/sources.yml` | dbt parse + source tests | raw→dbt contract | +| Transformation contracts | dbt model YAML `contract`/tests | `dbt_static`, `gate_governance` | per-model | +| Orchestration command interface | `scripts/run_atlas_step.sh`, `atlas_step_runner.py`, DAGs | `dag_import`, airflow tests | step contract | +| Audit-table interface | `sql/migrations/*`, `src/atlas/ops` | `sql_migrations`, `schema_compatibility` | `atlas_ops.*` schemas | +| Structured-log event contract | `src/atlas/observability/logging` | `observability_config`, tests | correlation ids, redaction | +| Monitoring configuration | `observability/{metrics,alerts,dashboards}` | `observability_config` | metric/alert schema | +| Deployment bundle contract | `scripts/build_deployment_bundle.sh` | deploy validation | immutable bundle layout | +| Migration contract | `sql/migrations/` + `checksums.lock` | `gate_schema_compatibility` | additive, immutable (INV-D8) | +| Governance metadata contract | dbt `meta.governance` + `governance/*.yml` | `gate_governance` | required fields (INV-G1..3) | +| Schema-compatibility inputs | `governance/schemas/manifests/baseline.json` | `gate_schema_compatibility` | baseline vs candidate | +| Lineage artifact inputs | dbt `ref()`/`source()` + `consumers.yml` | `gate_lineage_impact` | `governance/generated/lineage.json` | +| Cost-control configuration | `config/cost_controls.yaml` | `gate_performance_cost`, `cost_guard` | per-env ceilings | +| Approval variables | `ATLAS_APPROVE_*` env | scripts + gates | see [agent-onboarding.md](../handoff/agent-onboarding.md) | +| Reference/evidence contract | `reference-manifest.yml`, `evidence-index.json` | `atlas.reference.validate`, `gate_reference_handoff` | Sprint 8 | + +## Contract-change rule + +Changing any interface above requires: (1) updating the definition **and** its +enforcement together; (2) classifying the change via +[schema-evolution-policy-sprint7.md](../schema-evolution-policy-sprint7.md) when +it affects a schema; (3) a consumer-impact run +(`python -m atlas.governance.impact --asset `); (4) updating tests and the +[evidence index](evidence-index.md). Never change a definition while leaving its +enforcement or consumers stale. diff --git a/docs/reference-architecture/observability-model.md b/docs/reference-architecture/observability-model.md new file mode 100644 index 0000000..d3add18 --- /dev/null +++ b/docs/reference-architecture/observability-model.md @@ -0,0 +1,42 @@ +# Observability Model + +**Status:** CURRENT · **Audience:** operator, reviewer. Authoritative detail: +[observability-runbook-sprint5.md](../observability-runbook-sprint5.md), +[alert-catalog-sprint5.md](../alert-catalog-sprint5.md), ADR-011/012. + +## Layers + +- **Durable audit** — `atlas_ops.{pipeline_runs,task_events,quality_results, + monitor_evaluations,deployments,schema_migrations,recovery_actions}` (INV-O1). +- **Logs** — structured JSON to the `atlas-events` log with correlation ids + (`pipeline_run_id`, `batch_id`), redaction, and truncation (INV-O2, INV-G8). + Exported to a linked BigQuery dataset (`atlas_logs`) via a sink. +- **Metrics** — custom + log-based metrics (`observability/metrics/`), e.g. + `custom.googleapis.com/atlas/pipeline/last_success_age_seconds`, + `.../monitor/check_status`, `.../cost/bigquery_bytes_billed`. +- **Alerts** — Cloud Monitoring policies (`observability/alerts/*.json`), each + mapped to a runbook section (INV-O6). +- **Dashboard** — operational dashboard (`observability/dashboards/`). + +## Correlation & alert→runbook + +Every telemetry event carries correlation ids so a single run's story is +queryable end-to-end (example query in [README.md](../../README.md) Sprint 5 +section). Every alert maps to a runbook procedure; `gate_reference_handoff` +checks that alert definitions and runbook references stay consistent. + +## Known behaviors and limitations + +- **NO_DATA behavior** — some policies alert on absence (e.g. stale data); these + are environment-dependent and are disabled during controlled teardown. +- **Composer customer-project task-log limitation** — task logs are not always + fully available in the customer project; mitigated by direct Cloud Logging + export set at environment-create time (`ATLAS_LOG_TO_CLOUD_LOGGING=true`). +- **Post-teardown behavior** — after ephemeral Composer teardown, environment + -dependent alerts are disabled and permanent resources (audit tables, log + bucket/sink/view, metric descriptors, non-environment alert policies, + notification channel, dashboard) remain valid. Recorded in + `validation-report-sprint6.md` §8. + +These are honest operational limitations, tracked in +[unresolved-risks.md](unresolved-risks.md). diff --git a/docs/reference-architecture/operating-model.md b/docs/reference-architecture/operating-model.md new file mode 100644 index 0000000..11ea86e --- /dev/null +++ b/docs/reference-architecture/operating-model.md @@ -0,0 +1,30 @@ +# Operating Model + +**Status:** CURRENT · **Audience:** operator, reviewer. Who does what, with which +authority, and where the procedure lives. Detail is in the runbooks; this is the +ownership map. + +| Responsibility | Owner | Authority / gate | Procedure | +| --- | --- | --- | --- | +| Ownership of assets | technical owners in `governance/*` + dbt `meta` | governance CI | [governance-model-sprint7.md](../governance-model-sprint7.md) | +| Routine validation | any engineer / agent | none (credentialless) | `validate_ci.sh --mode static` | +| Deployment authority | operator | `ATLAS_APPROVE_DEPLOY` + WIF | [ci-cd-runbook-sprint4.md](../ci-cd-runbook-sprint4.md) | +| Incident authority | operator (on-call) | — | [observability-runbook-sprint5.md](../observability-runbook-sprint5.md), [on-call-model-sprint5.md](../on-call-model-sprint5.md) | +| Data-quality review | data owner | quality gate blocks publish | dbt tests + `quality_results` | +| Recovery approval | operator | verification required (INV-O3) | [recovery-runbook-sprint6.md](../recovery-runbook-sprint6.md) | +| Release management | release owner | `ATLAS_APPROVE_RELEASE` | tag on validated merge SHA | +| Evidence preservation | all | teardown allow-list (INV-O7) | validation reports + `docs/evidence-sprint*/` | +| Teardown | operator | `ATLAS_APPROVE_TEARDOWN` | [manage_atlas_composer.sh](../../scripts/manage_atlas_composer.sh) | +| Change review | reviewer + CI | P0/P1 gate | [agent-task-protocol.md](../handoff/agent-task-protocol.md) | + +## Cadence + +- Every change: credentialless CI must pass; invariants confirmed. +- Every deployment: immutable bundle → migrations → smoke → success/rollback. +- Every incident: detect → contain → diagnose → recover → verify → prevent, with + durable audit and an incident report. +- Every release: green CI on merged main → annotated tag on the exact SHA → + docs-only release-row follow-up. + +New operators start with [operator-onboarding.md](../handoff/operator-onboarding.md) +and the [operator-first-hour.md](../handoff/operator-first-hour.md) checklist. diff --git a/docs/reference-architecture/public-extraction-review.md b/docs/reference-architecture/public-extraction-review.md new file mode 100644 index 0000000..5fcd047 --- /dev/null +++ b/docs/reference-architecture/public-extraction-review.md @@ -0,0 +1,19 @@ +# Public-Repository Extraction Review + +**Status:** CURRENT + +This standalone repository was extracted from the Atlas reference implementation. +The public candidate was scanned for personal email addresses, private keys, +service-account JSON, API keys, webhook URLs, bearer tokens, source sandbox +identifiers, and real user data. No committed credentials or real user data are +included. Runtime identities and GCP resource names use documented examples or +environment variables. + +The extraction excludes the separate artifact-hosting product and raw drill +evidence bundles that are not required by this data-platform template. Selected +sanitized evidence remains where reference and regression gates require it. +Historical reports do not prove that a new adopter has deployed this template. + +```bash +python scripts/validate_public_extraction.py +``` diff --git a/docs/reference-architecture/reference-manifest.yml b/docs/reference-architecture/reference-manifest.yml new file mode 100644 index 0000000..1ca46b2 --- /dev/null +++ b/docs/reference-architecture/reference-manifest.yml @@ -0,0 +1,209 @@ +# Atlas reference-architecture manifest (Sprint 8, Phase 2). +# Validated by: python -m atlas.reference.validate --only manifest +# Paths are relative to . Status: CURRENT|HISTORICAL|SUPERSEDED|PLANNED|BLOCKED +version: 1 +last_verified_commit: "3f986aa" +owner: data-architect +documents: + - document_id: start-here + title: START HERE — canonical entry point + purpose: Route every audience to authoritative sources + audience: all + status: CURRENT + source_of_truth: START_HERE.md + related_runbooks: + - docs/ci-cd-runbook-sprint4.md + - docs/observability-runbook-sprint5.md + - docs/recovery-runbook-sprint6.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: system-context + title: System Context + purpose: Problem, actors, boundaries, context diagram + audience: all + status: CURRENT + source_of_truth: docs/reference-architecture/system-context.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: architecture-overview + title: Architecture Overview + purpose: Data / control / operational / governance flows + audience: engineer, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/architecture-overview.md + related_adrs: + - docs/adr/ADR-006-batch-identity.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: architecture-invariants + title: Architecture Invariants + purpose: Properties that must remain true and how they are enforced + audience: engineer, agent, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/architecture-invariants.md + related_adrs: + - docs/adr/ADR-006-batch-identity.md + - docs/adr/ADR-017-schema-compatibility-and-deprecation.md + related_tests: + - tests/unit/test_schema_check.py + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: component-catalog-reusable + title: Component Catalog — Reusable + purpose: Reusable-candidate components and extraction actions + audience: engineer, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/component-catalog-reusable.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: component-catalog-atlas-specific + title: Component Catalog — Atlas-Specific + purpose: Project-specific components and their stable interfaces + audience: engineer, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/component-catalog-atlas-specific.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: interfaces-and-contracts + title: Interfaces and Contracts + purpose: Stable boundaries and their enforcement + audience: engineer, agent + status: CURRENT + source_of_truth: docs/reference-architecture/interfaces-and-contracts.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: extension-points + title: Extension Points + purpose: How to extend Atlas safely + audience: engineer, agent + status: CURRENT + source_of_truth: docs/reference-architecture/extension-points.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: operating-model + title: Operating Model + purpose: Ownership, authority, cadence + audience: operator, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/operating-model.md + related_runbooks: + - docs/ci-cd-runbook-sprint4.md + - docs/recovery-runbook-sprint6.md + last_verified_commit: "3f986aa" + owner: platform-operator + - document_id: security-and-identity-model + title: Security and Identity Model + purpose: Identities, WIF boundary, least-privilege status + audience: operator, reviewer, agent + status: CURRENT + source_of_truth: docs/reference-architecture/security-and-identity-model.md + related_adrs: + - docs/adr/ADR-009-workload-identity-federation.md + - docs/adr/ADR-018-identity-and-access-boundaries.md + last_verified_commit: "3f986aa" + owner: security-reviewer + - document_id: reliability-and-recovery-model + title: Reliability and Recovery Model + purpose: Detection through prevention, replay, rollback + audience: operator, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/reliability-and-recovery-model.md + related_adrs: + - docs/adr/ADR-013-controlled-fault-injection.md + - docs/adr/ADR-014-recovery-action-model.md + related_runbooks: + - docs/recovery-runbook-sprint6.md + last_verified_commit: "3f986aa" + owner: platform-operator + - document_id: observability-model + title: Observability Model + purpose: Audit, logs, metrics, alerts, limitations + audience: operator, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/observability-model.md + related_adrs: + - docs/adr/ADR-011-atlas-observability-model.md + related_runbooks: + - docs/observability-runbook-sprint5.md + last_verified_commit: "3f986aa" + owner: platform-operator + - document_id: cost-and-lifecycle-model + title: Cost and Lifecycle Model + purpose: Cost controls, retention, proven vs not proven + audience: operator, reviewer + status: CURRENT + source_of_truth: docs/reference-architecture/cost-and-lifecycle-model.md + related_adrs: + - docs/adr/ADR-020-bigquery-performance-and-cost-controls.md + last_verified_commit: "3f986aa" + owner: platform-operator + - document_id: evidence-index + title: Evidence Index + purpose: Human view of the claim-to-evidence ledger + audience: reviewer, agent + status: CURRENT + source_of_truth: docs/reference-architecture/evidence-index.md + related_evidence: + - governance/generated/evidence-index.json + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: capability-evidence-map + title: Capability and Evidence Map + purpose: Competency domains with honest framing + audience: reviewer, interviewer + status: CURRENT + source_of_truth: docs/reference-architecture/capability-evidence-map.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: unresolved-risks + title: Unresolved Risks + purpose: Risk register and Sprint 7 blocked-gate disposition + audience: all + status: CURRENT + source_of_truth: docs/reference-architecture/unresolved-risks.md + related_evidence: + - governance/unresolved_risks.yml + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: public-extraction-review + title: Public-Repository Extraction Review + purpose: Public-template extraction review and current dispositions + audience: security-reviewer, release-owner + status: CURRENT + source_of_truth: docs/reference-architecture/public-extraction-review.md + related_evidence: + - config/public_extraction_manifest.yml + last_verified_commit: "3f986aa" + owner: security-reviewer + - document_id: template-extraction-plan + title: Template Extraction Record + purpose: Record of the completed standalone-template extraction + audience: data-architect, template-author + status: CURRENT + source_of_truth: docs/reference-architecture/template-extraction-plan.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: operator-onboarding + title: Operator Onboarding + purpose: Three-mode operator onboarding + audience: operator + status: CURRENT + source_of_truth: docs/handoff/operator-onboarding.md + last_verified_commit: "3f986aa" + owner: platform-operator + - document_id: agent-onboarding + title: Agent Onboarding + purpose: Coding-agent onboarding and conventions + audience: agent + status: CURRENT + source_of_truth: docs/handoff/agent-onboarding.md + last_verified_commit: "3f986aa" + owner: data-architect + - document_id: clean-clone-reproduction + title: Clean-Clone Reproduction + purpose: From-scratch reproduction procedure + audience: engineer, agent + status: CURRENT + source_of_truth: docs/handoff/clean-clone-reproduction.md + last_verified_commit: "3f986aa" + owner: data-engineer diff --git a/docs/reference-architecture/reliability-and-recovery-model.md b/docs/reference-architecture/reliability-and-recovery-model.md new file mode 100644 index 0000000..fdeb331 --- /dev/null +++ b/docs/reference-architecture/reliability-and-recovery-model.md @@ -0,0 +1,44 @@ +# Reliability and Recovery Model + +**Status:** CURRENT · **Audience:** operator, reviewer. Authoritative detail: +[failure-catalog-sprint6.md](../failure-catalog-sprint6.md), +[recovery-runbook-sprint6.md](../recovery-runbook-sprint6.md), +[game-day-results-sprint6.md](../game-day-results-sprint6.md), ADR-013/014/015. + +## Lifecycle + +``` +detect → contain → diagnose → recover → verify → prevent +``` + +- **Detect** — retries, dbt test failures, overlapping-run guards, cost-guard + blocks surface via `task_events`/`pipeline_runs`/telemetry. +- **Contain** — failed runs publish no success marker (INV-L5); DAGs paused to + stop scheduler contention; cost guards block before spend. +- **Diagnose** — root cause from `task_events`, `INFORMATION_SCHEMA.JOBS`, dbt + output, row-count queries. Correlated by `pipeline_run_id`/`batch_id` (INV-O2). +- **Recover** — targeted repair (e.g. `QUARANTINE_BATCH`), not blanket full + refresh; recorded in `atlas_ops.recovery_actions`. +- **Verify** — `validate_warehouse()`; SUCCESS gated on `VERIFIED` + (INV-O3). Proven live: INC-S6-001 recovered + verified 10/10. +- **Prevent** — documented follow-ups + regression tests + reusable controls. + +## Idempotency, replay, backfills + +- Exact rerun idempotent (INV-D3); within-batch dup vs cross-batch replay + distinct (INV-D4, ADR-006 amendment resolved Sprint 6 INC-S6-001). +- Backfills deterministic (Sprint 3); global fact uniqueness preserved (INV-D5). + +## Fault injection & rollback + +- Fault injection disabled by default, gated (INV-O4, ADR-013). +- Rollback checks schema compatibility (INV-L6, ADR-015); a failed deployment + cannot publish success (INV-L5). + +## Known gaps (→ [unresolved-risks.md](unresolved-risks.md)) + +- Composer customer-project task-log limitation (Sprint 5) — mitigated by direct + Cloud Logging export. +- Live game-day coverage is a representative subset; full live injection of all + 56 catalog scenarios was intentionally not performed. +- Synthetic 50k-row scale — reliability behavior is proven at that scale only. diff --git a/docs/reference-architecture/security-and-identity-model.md b/docs/reference-architecture/security-and-identity-model.md new file mode 100644 index 0000000..dc3e578 --- /dev/null +++ b/docs/reference-architecture/security-and-identity-model.md @@ -0,0 +1,42 @@ +# Security and Identity Model + +**Status:** CURRENT · **Audience:** operator, reviewer, agent. Authoritative +detail: [iam-review-sprint7.md](../iam-review-sprint7.md), +[security-review-sprint7.md](../security-review-sprint7.md), ADR-009/018. + +## Identities + +| Identity | Type | Purpose | Auth | Notes | +| --- | --- | --- | --- | --- | +| Human operator | person | approvals, incident/release authority | Google account | approves `ATLAS_APPROVE_*` | +| Cursor development identity | agent SA (read-mostly) | local/dev inspection, MCP | short-lived key in Cursor secret | least-privilege | +| GitHub integration identity | SA | repository→GCP integration | **WIF (keyless)** | see excess note | +| GitHub deployer | SA | deploy releases | WIF (keyless) | scoped to deploy | +| Composer runtime | SA | Airflow execution | managed | ephemeral env | +| Log sink writer | SA | export logs to BigQuery | managed | write to `atlas_logs` | +| Monitoring/notification | managed | metrics, alerts, notify | managed | notification channel = operator email | + +## Boundaries and rules (enforced) + +- **Keyless WIF only** for GitHub→GCP; no service-account keys committed or + created (INV-L2). Enforced by `gate_security_policy` (`scan_managed_iam`). +- No `roles/owner`, `roles/editor`, or broad Project IAM Admin. No weakening WIF + trust conditions. ADR-018. +- Secrets never committed or written to evidence (INV-G8); enforced by + `secret_scan` + `scan_data_exposure` + `validate_public_extraction.py`. + +## Least-privilege status (honest) + +Reviewed in Sprint 7. One justified reduction candidate remains: the +`atlas-github-integration` SA holds project-level `roles/bigquery.dataEditor`, +broader than required. A scoped reduction plus positive/negative test plan is +documented but **NOT executed** — it is BLOCKED on `ATLAS_APPROVE_IAM` +([unresolved-risks.md](unresolved-risks.md), RISK-01/02). We do **not** claim +complete least privilege without live permission-level evidence. + +## Public-extraction risks + +The operator email and private project id appear across configs and docs. These +are cataloged with dispositions in +`config/public_extraction_manifest.yml` + `scripts/validate_public_extraction.py`. +The repository is **not** published during Sprint 8. diff --git a/docs/reference-architecture/system-context.md b/docs/reference-architecture/system-context.md new file mode 100644 index 0000000..5a70d29 --- /dev/null +++ b/docs/reference-architecture/system-context.md @@ -0,0 +1,72 @@ +# System Context + +**Status:** CURRENT · **Audience:** all · **Source of truth:** code + ADRs. +This is a curated map; it links to authoritative sources rather than copying them. + +## Problem modeled + +Atlas models a **batch analytics platform for a synthetic event stream** (users +performing actions across countries). The engineering problem — not the business +domain — is the point: prove correct, governed, observable, recoverable batch +data processing on GCP with reproducible evidence. Scale is deliberately modest +(~50,000 events/batch); Atlas is production-*oriented*, not production-*scale*. + +## Context diagram + +``` + ┌───────────────────────── Google Cloud (example-gcp-project) ─────────────────────────┐ + ┌─────────────┐ │ │ + │ Human │ approves│ ┌────────────┐ ┌───────────┐ ┌───────────────────────────┐ ┌────────────────┐ │ + │ operator ├────────►│ │ Cloud │ │ BigQuery │ │ Cloud Composer (Airflow │ │ Cloud Logging │ │ + │ (the primary operator) │ │ │ Storage │──►│ raw + │──►│ 3.1.7) — atlas_batch and │──►│ + Monitoring │ │ + └─────┬───────┘ │ │ (immutable │ │ dbt models │ │ observability DAGs │ │ (metrics, │ │ + │ │ │ landing + │ │ (staging→ │ └───────────────────────────┘ │ alerts, dash) │ │ + │ runs │ │ releases) │ │ marts) │ │ └────────────────┘ │ + ┌─────▼───────┐ │ └────────────┘ └───────────┘ │ audit │ + │ Coding agent │ CI/CD │ ▲ ▲ ▼ │ + │ (Cursor) ├────────┼────────┼─────────────────┼──────► atlas_ops.* operational tables │ + └─────┬───────┘ │ │ WIF (keyless) │ │ + │ └────────┼─────────────────┼──────────────────────────────────────────────────────────────┘ + │ pull request │ │ + ┌─────▼───────────────┐ ┌──────┴───────┐ ┌──────┴───────┐ + │ GitHub (CI/CD: │──►│ Synthetic │ │ dbt │ + │ credentialless PR, │ │ event │ │ (BigQuery │ + │ trusted deploy) │ │ generator │ │ adapter) │ + └─────────────────────┘ └──────────────┘ └──────────────┘ +``` + +## Building blocks (where each lives) + +| Block | Purpose | Source | +| --- | --- | --- | +| Synthetic event source | Deterministic 50k-event generation with seeded anomalies | `src/atlas/generator`, `scripts/generate_events.py`, `config/anomaly_profile.yaml` | +| Batch ingestion | Immutable JSONL → run-scoped GCS paths | `src/atlas/ingestion`, `scripts/upload_events.py` | +| Cloud Storage landing | Immutable raw artifacts + release bundles | GCS buckets (see `infra/`) | +| BigQuery raw | Partitioned/clustered raw table | `sql/`, `src/atlas/loader` | +| dbt transformation | staging → classification → accepted/rejected → core (fact/dims) → marts | `dbt/atlas_dbt` | +| Airflow orchestration | `atlas_batch_pipeline`, `atlas_observability_monitor` | `dags/`, `src/atlas/batch` | +| CI/CD | Credentialless PR CI + trusted WIF deploy + rollback | `scripts/validate_ci.sh`, `.github/workflows/`, ADR-008/009/010 | +| Composer deployment | Ephemeral managed Airflow for acceptance | `scripts/manage_atlas_composer.sh`, ADR-005/010 | +| Observability | Structured logs, metrics, alerts, dashboard | `src/atlas/observability`, `observability/`, ADR-011 | +| Recovery | Recovery-action audit + verification | `src/atlas/ops`, `docs/recovery-runbook-sprint6.md`, ADR-014 | +| Governance | Contracts, schema compat, lineage, retention | `src/atlas/governance`, `governance/`, ADR-016–019 | +| Security | Keyless WIF, least-privilege review, scanners | `src/atlas/governance/security_policy.py`, ADR-009/018 | +| Cost controls | Config-driven ceilings + dry-run guard | `config/cost_controls.yaml`, `src/atlas/observability/cost_guard.py`, ADR-020 | + +## Actors and interactions + +- **Human operator** — approves gated mutations (`ATLAS_APPROVE_*`), owns + incident/recovery/release authority ([operating model](operating-model.md)). +- **Coding agent** — implements changes via the [agent task protocol](../handoff/agent-task-protocol.md); bound by invariants and CI. +- **GitHub** — runs credentialless PR CI; trusted workflows authenticate to GCP via WIF (no keys). +- **Consumers** — internal only, registered in `governance/consumers.yml` (dashboard, monitor, reconciliation, alerting, analytics readers). External consumer discovery is out of repository scope (a known limitation). + +## System boundaries and external dependencies + +In scope: the `` ELT platform. Out of scope (separate lifecycle): +External dependencies: Google Cloud (BigQuery, GCS, Composer, Logging, +Monitoring, IAM/WIF), GitHub Actions, dbt (BigQuery adapter), Apache Airflow +3.1.7. No streaming, Pub/Sub, Dataflow, CDC, or ML dependencies. + +See [architecture-overview.md](architecture-overview.md) for the data/control/ +operational/governance flows. diff --git a/docs/reference-architecture/template-extraction-plan.md b/docs/reference-architecture/template-extraction-plan.md new file mode 100644 index 0000000..1cc2700 --- /dev/null +++ b/docs/reference-architecture/template-extraction-plan.md @@ -0,0 +1,15 @@ +# Template Extraction Record + +**Status:** CURRENT + +The Atlas reference implementation has been extracted into this standalone GCP +production-data-platform template. Reusable CI, keyless delivery, migrations, +recovery, observability, governance, schema, lineage, cost, and handoff controls +are retained. Cloud projects, repository claims, service accounts, buckets, +datasets, schedules, notifications, and cost ceilings are configuration. + +Extraction proves repository portability. Operational adoption still requires an +isolated GCP deployment, a successful batch, a deliberate failure, targeted +recovery, governance and observability checks, cleanup, and operator handoff. +Each adopter must produce environment-specific evidence before making production +readiness claims. diff --git a/docs/reference-architecture/unresolved-risks.md b/docs/reference-architecture/unresolved-risks.md new file mode 100644 index 0000000..33294f2 --- /dev/null +++ b/docs/reference-architecture/unresolved-risks.md @@ -0,0 +1,40 @@ +# Unresolved Risks + +**Status:** CURRENT · **Audience:** all. Machine-readable copy: +[`governance/unresolved_risks.yml`](../../governance/unresolved_risks.yml). +Medium and high risks are listed honestly — none is hidden to make the capstone +look cleaner. Blocked work is shown as BLOCKED, never as complete. + +| id | title | sev | status | approval | next action | +| --- | --- | --- | --- | --- | --- | +| RISK-01 | Sprint 7 live IAM reduction not executed | MEDIUM | BLOCKED | `ATLAS_APPROVE_IAM` | apply scoped reduction on `atlas-github-integration` dataEditor | +| RISK-02 | Sprint 7 positive/negative IAM tests not executed | MEDIUM | BLOCKED | `ATLAS_APPROVE_IAM` | run authorized + denied ops, record both | +| RISK-03 | Billed BigQuery performance suite not executed | LOW | BLOCKED | `ATLAS_APPROVE_PERFORMANCE_TESTS` + byte ceiling | execute once, record billed bytes/slot/correctness | +| RISK-04 | Live retention/expiration application not executed | LOW | BLOCKED | `ATLAS_APPROVE_RETENTION_MUTATION` | apply TTL to temporary resources only, verify | +| RISK-05 | Composer customer-project task-log limitation | LOW | MITIGATED | — | direct Cloud Logging export at create time | +| RISK-06 | Synthetic 50k-row scale | MEDIUM | ACCEPTED | — | do not claim production-scale perf/cost | +| RISK-07 | No multi-environment production promotion | MEDIUM | ACCEPTED | — | future sprint scope | +| RISK-08 | No streaming/event-driven/CDC ingestion | LOW | ACCEPTED | — | future capstone (prefer API/event) | +| RISK-09 | External consumers not discoverable from repository | MEDIUM | ACCEPTED | — | only internal consumers registered | +| RISK-10 | Public extraction not yet performed | MEDIUM | OPEN | `ATLAS_APPROVE_PUBLIC_EXTRACTION` | run `validate_public_extraction.py`; Sprint 8+ extraction | +| RISK-12 | No second-project generation test | MEDIUM | DEFERRED | — | post-Atlas template validation | +| RISK-13 | Clean-clone platform limitations | LOW | OPEN | — | see [clean-clone-results.md](../evidence-sprint8/clean-clone-results.md) | +| RISK-14 | Independent handoff ambiguity | LOW | OPEN | — | see [independent-handoff-results.md](../evidence-sprint8/independent-handoff-results.md) | + +## Sprint 7 blocked-gate disposition + +At Sprint 8 preflight the IAM, performance, and retention approval variables were +**absent**. Per the master prompt's missing-approval behavior, these gates: + +- remain **BLOCKED** (RISK-01..04), +- retain their exact execution plans (in + [iam-review-sprint7.md](../iam-review-sprint7.md), + [performance-review-sprint7.md](../performance-review-sprint7.md), + [retention-policy-sprint7.md](../retention-policy-sprint7.md)), +- are **not** re-run and **not** weakened, +- do **not** contribute any "complete" claim. + +If the approvals are provided later, execute only after the reference package and +clean-clone path are stable, following the IAM/performance/retention legs +described in the Sprint 8 prompt Phase 13. These are optional closure +improvements, not automatic requirements for Sprint 8 completion. diff --git a/docs/retention-policy-sprint7.md b/docs/retention-policy-sprint7.md new file mode 100644 index 0000000..eec2ab3 --- /dev/null +++ b/docs/retention-policy-sprint7.md @@ -0,0 +1,67 @@ +# Atlas Classification & Retention Policy (Sprint 7) + +Declares how long each category of Atlas data is retained and how it is +disposed. Definitions live in `governance/classifications.yml` and +`governance/retention.yml`; validation is enforced by `gate_governance` +(`atlas.governance.retention.validate_retention_config`). See ADR-019. + +## Classification levels + +| Level | Meaning | Logging | Atlas use | +| --- | --- | --- | --- | +| PUBLIC | shareable, non-sensitive | none | `dim_countries` (synthetic reference) | +| INTERNAL | operational, non-personal (default) | counts/ids only | events, models, audit tables | +| CONFIDENTIAL | harmful if exposed (tokens, addresses) | sanitized, never verbatim, never committed | notification address (live only), sanitized errors | +| RESTRICTED | real personal/regulated data | never logged, audited | **none present** (Atlas is synthetic) | + +## Retention classes and disposal + +| Retention class | Applies to | Expiration | Permanent evidence | +| --- | --- | --- | --- | +| `canonical_warehouse` | staging/intermediate/core/mart models | none (rebuildable) | no | +| `raw_landing` | `atlas_raw.events`, raw bucket | none (source of truth) | no | +| `operational_audit` | pipeline_runs, task_events, quality_results, monitor_evaluations, deployments, schema_migrations, recovery_actions | **never** | **yes** | +| `observability_logs` | log bucket + linked dataset | 30 days | no | +| `release_evidence` | immutable release bundles | keep validated | **yes** | +| `temporary_integration` | CI/integration datasets | 1 day | no | +| `test_fixture` | local generated artifacts | ephemeral | no | + +## Retention rules distinguish + +- **Canonical data** — indefinite, deterministically rebuildable; never auto-expired. +- **Operational evidence** — permanent; CI forbids attaching an expiration. +- **Temporary resources** — carry a mandatory disposal (dataset TTL / bucket + lifecycle). +- **Release evidence** — validated releases retained. +- **Test fixtures** — not persisted to cloud. + +## Enforced invariants + +- `policy.retention_classes` matches `retention.yml` keys exactly (no drift). +- A `is_permanent_evidence` class cannot declare an expiration (conflict → CI fail). +- A transient class (`observability_logs`, `temporary_integration`, + `test_fixture`) must declare an expiration. +- Every governed asset references a defined retention class. + +## Disposal planning (dry-run) + +```bash +python -c "import json; from atlas.governance.retention import plan_expirations; \ + print(json.dumps(plan_expirations(), indent=2))" +``` + +Each asset resolves to `keep_forever` (permanent evidence / rebuildable) or +`expire_d`. A live applier asserts it never expires a `keep_forever` asset. + +## Live application (gated) + +Applying dataset/bucket expiration to temporary resources requires +`ATLAS_APPROVE_RETENTION_MUTATION=true`. Current live state (read-only check): +`atlas_ops`, `atlas_core`, `atlas_raw` have **no** default table expiration +(correct — permanent/rebuildable). No required evidence is deleted during +Sprint 7. + +**Blocked gate:** live expiration application to a temporary CI dataset/bucket +prefix is pending `ATLAS_APPROVE_RETENTION_MUTATION`. The safe controls, the +disposal plan, and validation tests are complete; only the live mutation is +deferred. The gate is not weakened. diff --git a/docs/runbook-sprint2.md b/docs/runbook-sprint2.md new file mode 100644 index 0000000..fbf8ec4 --- /dev/null +++ b/docs/runbook-sprint2.md @@ -0,0 +1,81 @@ +# Project Atlas Sprint 2 Runbook + +## Prerequisites + +- Sprint 1 raw table populated: `example-gcp-project.atlas_raw.events` +- GCP auth via `gcloud auth application-default login` or external service account key +- Cloud Shell or approved agent environment with `gcloud`, `bq`, and Python 3 + +## One-time setup + +Run from `~/Atlas-GCP-Build/project-atlas` on branch `cursor/atlas-sprint-2-dbt-warehouse-3660`: + +```bash +cd ~/Atlas-GCP-Build +git fetch origin cursor/atlas-sprint-2-dbt-warehouse-3660 +git checkout cursor/atlas-sprint-2-dbt-warehouse-3660 +cd Atlas-GCP-Build +export ATLAS_GCP_PROJECT_ID=example-gcp-project +export ATLAS_DBT_DATASET=atlas +bash scripts/setup_dbt.sh +``` + +The setup script creates `.venv-dbt`, installs pinned dbt packages, writes `~/.dbt/profiles.yml` +without printing credentials, and runs `dbt debug`. + +## Full warehouse build + +```bash +source .venv-dbt/bin/activate +bash scripts/run_dbt_sprint2.sh +``` + +Options: + +- `--full-refresh` — rebuild incremental models from scratch +- `--skip-docs` — skip `dbt docs generate` + +## Validation and evidence + +```bash +bash scripts/validate_dbt_sprint2.sh +``` + +Writes `logs/validation-sprint2-.json` with counts, anomaly totals, and gate status. + +## Incremental rerun (unchanged source) + +After a successful full build against the validated run: + +```bash +source .venv-dbt/bin/activate +dbt build --project-dir dbt/atlas_dbt --profiles-dir ~/.dbt --target dev +dbt test --project-dir dbt/atlas_dbt --profiles-dir ~/.dbt --target dev +``` + +Expect no new rows when the raw source is unchanged. + +## Bounded backfill example + +```bash +dbt run --select fct_events \ + --project-dir dbt/atlas_dbt \ + --profiles-dir ~/.dbt \ + --target dev \ + --vars '{"start_date": "2026-07-01", "end_date": "2026-07-14"}' +``` + +## Troubleshooting + +| Symptom | Action | +| --- | --- | +| `dbt debug` auth failure | Re-run ADC login or set `GOOGLE_APPLICATION_CREDENTIALS` | +| Freshness warning/error | Expected if raw data is stale; investigate ingestion schedule | +| Singular anomaly test failure | Compare counts in `int_event_classification` against validated run scope | +| Contract enforcement failure | Inspect schema drift in staging/core YAML contracts | + +## Security + +- Never commit `profiles.yml`, service account JSON, or `.env` +- Scripts redact credentials from stdout +- Live profile generation requires explicit environment variables only diff --git a/docs/runbook-sprint3.md b/docs/runbook-sprint3.md new file mode 100644 index 0000000..a350298 --- /dev/null +++ b/docs/runbook-sprint3.md @@ -0,0 +1,45 @@ +# Atlas Sprint 3 Runbook + +## Local setup + +```bash +cd ~/Atlas-GCP-Build/project-atlas +git checkout cursor/atlas-sprint-3-airflow-orchestration-3660 +export ATLAS_GCP_PROJECT_ID=example-gcp-project +source airflow/airflow.env.example +bash scripts/setup_airflow.sh +bash scripts/test_airflow_sprint3.sh +bash scripts/start_airflow_local.sh +# Cloud Shell: wait ~30s, then check DAG registration +tail -f .airflow/standalone.log +``` + +Cloud Shell has no tmux; `start_airflow_local.sh` falls back to `nohup` and writes +`.airflow/standalone.log`. Ensure `PYTHONPATH` includes `src/` and `dags/` (set in +`airflow.env.example`). + +## Trigger a run + +Start Airflow before triggering. `run_airflow_sprint3.sh` waits for the DAG to register. + +```bash +bash scripts/run_airflow_sprint3.sh \ + --processing-date 2026-07-15 \ + --batch-id atlas-20260715 \ + --conf '{"upload_once": true}' +``` + +## Recovery + +| Scenario | Action | +|----------|--------| +| Transient upload failure | Allow retry; verify audit row transitions to SUCCESS | +| dbt test failure | Clear failed task after fixing; rerun with new pipeline_run_id | +| Partial batch in raw | Investigate loader logs; do not clear without ops review | +| Audit table missing | Re-run `ensure_audit_resources` via DAG or CLI | + +## Composer notes + +- Deploy DAGs separately from data/scripts. +- Use environment service account with BigQuery + GCS permissions. +- SQLite limitations apply locally only; Composer uses Cloud SQL metadata. diff --git a/docs/runbook.md b/docs/runbook.md new file mode 100644 index 0000000..f78dfc4 --- /dev/null +++ b/docs/runbook.md @@ -0,0 +1,121 @@ +# Project Atlas Runbook + +## Normal operation + +```bash +cd Atlas-GCP-Build +export PYTHONPATH=src +export GCP_PROJECT_ID=example-gcp-project +python scripts/run_pipeline.py --approve-provision +``` + +Expected outcome: + +- JSONL generated locally +- GCS object created under run-scoped prefix +- Rows loaded into `atlas_raw.events` +- Validation overall status: `FAIL` (seeded anomalies present) +- Acceptance anomaly detection checks: `PASS` + +## Approval gate + +Live GCP mutations require explicit approval: + +```bash +export ATLAS_APPROVE_PROVISION=true +``` + +Without this variable: + +- `bootstrap_gcp.sh` exits with code `2` +- `run_pipeline.py` exits with code `2` + +## Recovery procedures + +### Duplicate upload + +Symptom: upload returns `already_exists=true`. + +Action: inspect the existing object path in logs and either reuse the same +`pipeline_run_id` for downstream load or generate a new run id for a fresh object. + +### Missing source file + +Symptom: generator output path missing. + +Action: + +```bash +python scripts/generate_events.py +``` + +Re-run upload/load with the new local path. + +### Bad schema load failure + +Symptom: BigQuery load job fails. + +Action: + +1. Inspect load job error in Cloud Console or `bq ls -j --max_results=5`. +2. Fix JSONL schema locally. +3. Re-run with a new `pipeline_run_id`. + +### Duplicate run load + +Symptom: loader returns `already_loaded=true`. + +Action: this is expected for idempotent replays. Validation can be re-run safely: + +```bash +python scripts/validate_events.py --run-id --event-date +``` + +### Null IDs and bad timestamps + +Symptom: validation checks `null_user_ids`, `future_timestamps`, or +`late_arriving_events` report `FAIL`. + +Action: for Sprint 1 this is expected. Confirm acceptance checks prove the seeded +counts were detected. + +## Logs + +Structured JSON logs are written to: + +```text +logs/.jsonl +``` + +Each step records: + +- timestamp +- duration +- status +- rows processed +- pipeline run id +- source file + +## MCP troubleshooting + +```bash +bash scripts/verify_mcp_access.sh +``` + +Common fixes: + +- Restart Cursor after editing `.env` +- Run `gcloud auth application-default login` on desktop +- Ensure `ATLAS_GCP_SERVICE_ACCOUNT_KEY` is set for cloud agents +- Confirm workspace root is `de-project-1`, not `project-atlas` + +## Manual verification queries + +```bash +bq query --use_legacy_sql=false \ +'SELECT COUNT(*) AS row_count FROM `example-gcp-project.atlas_raw.events` WHERE pipeline_run_id = ""' +``` + +```bash +gsutil ls gs://atlas-raw-events-example-gcp-project/raw/** +``` diff --git a/docs/schema-evolution-policy-sprint7.md b/docs/schema-evolution-policy-sprint7.md new file mode 100644 index 0000000..1bf27aa --- /dev/null +++ b/docs/schema-evolution-policy-sprint7.md @@ -0,0 +1,61 @@ +# Atlas Schema Evolution Policy (Sprint 7) + +How Atlas schemas and contracts change safely. Enforced by +`atlas.governance.schema_check` + `gate_schema_compatibility`. See ADR-017. + +## The workflow + +``` +1. Edit a model/table schema or governance meta. +2. Regenerate the baseline manifest and run the checker: + python -m atlas.governance.schema_check --generate \ + governance/schemas/manifests/baseline.json + python -m atlas.governance.schema_check \ + --baseline --candidate governance/schemas/manifests/baseline.json \ + --output report.json +3. Read the overall_class: + COMPATIBLE -> bump minor contract_version, ship. + CONDITIONALLY_COMPATIBLE -> add a change record + migration/dual-write plan. + BREAKING -> add an APPROVED change record; bump major. + PROHIBITED -> stop; the change is not allowed as written. +4. Commit the regenerated baseline (CI fails on drift). +``` + +## Compatibility classes (summary) + +| Class | Examples | CI | +| --- | --- | --- | +| COMPATIBLE | add nullable field, widen enum, loosen nullability, new asset, metadata | passes | +| CONDITIONALLY_COMPATIBLE | add required field, type widening w/ evidence, dual-write, deprecation w/ replacement | passes **with** complete change record | +| BREAKING | remove/rename field, type change, tighten nullability, change grain/partition/identity, narrow enum | fails without an **approved** change record | +| PROHIBITED | unversioned change, contract downgrade, edit an applied migration, silent field reuse | always fails | + +## Change records + +Location: `governance/changes/.yml` (template: +`governance/changes/TEMPLATE.yml`). Required fields: `change_id`, `asset_id`, +`old_contract_version`, `new_contract_version`, `compatibility_class`, `reason`, +`owner`, `consumer_impact`, `migration_plan`, `backfill_plan`, `validation_plan`, +`rollback_limitations`, `deprecation_window`, `approval_reference`. + +A BREAKING change must set `approval_reference` (e.g. an approval variable or PR +approval id) and a `migration_plan`; otherwise CI blocks the merge. + +## Contract versioning + +`contract_version` is `major.minor`. A schema change with an unchanged or +decreased version is PROHIBITED (unversioned replacement). Minor bump for +COMPATIBLE changes; major bump for CONDITIONALLY_COMPATIBLE / BREAKING. + +## Applied-migration immutability + +`sql/migrations/checksums.lock` pins each migration's SHA-256. Editing a shipped +migration changes its checksum and fails `gate_schema_compatibility`. Add new +migrations by appending both the manifest line and a lock entry (regenerate with +the helper), never by editing an existing file. + +## Never against canonical data + +Breaking-change demonstrations use isolated fixtures only, gated by +`ATLAS_APPROVE_BREAKING_SCHEMA_DEMO=true`. Canonical Atlas tables are never +mutated to demonstrate a breaking change. diff --git a/docs/security-review-sprint5.md b/docs/security-review-sprint5.md new file mode 100644 index 0000000..169fc5f --- /dev/null +++ b/docs/security-review-sprint5.md @@ -0,0 +1,82 @@ +# Atlas Sprint 5 Security Review + +Scope: the observability additions of Sprint 5 — structured logging, audit +tables, log routing, linked dataset, metrics, alerting, dashboard, and the +drills. Reviewed live on 2026-07-19 against project `example-gcp-project`. + +## Log content safety + +| Control | Evidence | +| --- | --- | +| No secrets in structured events | `emit_event` builds from a fixed field allowlist; `error_message` passes through `sanitize_error_message` (the Sprint 4-audited sanitizer: strips bearer tokens, key material patterns, long opaque strings) and is truncated at 2 000 chars; `details` truncated at 4 KB. Unit-tested in `tests/unit/test_observability_logging.py` (redaction, truncation, non-serializable degradation). | +| No raw event payloads | Only counts, ids, and bounded details fields are emitted; the event generator's payloads never enter telemetry. | +| No full SQL text | Cost attribution uses job labels and `INFORMATION_SCHEMA` aggregates; `details_json` in audit tables stores summaries only. | +| No environment dumps | The contract has no field capable of carrying `os.environ`; unknown fields are dropped (or rejected in strict mode). | +| Live spot-check | Drill artifacts in `docs/evidence-sprint5/` were reviewed; the notification-channel evidence stores only the email domain, not the address. | + +## IAM evidence table (live `gcloud projects get-iam-policy`, 2026-07-19) + +| Principal | Roles | Purpose / justification | +| --- | --- | --- | +| `atlas-composer-runtime@…` | `roles/composer.worker`, `roles/bigquery.jobUser`, `roles/bigquery.dataEditor`, `roles/bigquery.resourceViewer` | Composer 3 runtime identity. `composer.worker` already includes `logging.logEntries.create` and `monitoring.timeSeries.create`, so **no new logging/monitoring roles were required** for structured emission or metric publication. `resourceViewer` was added during acceptance because the cost check reads `region-us.INFORMATION_SCHEMA.JOBS` (needs `bigquery.jobs.listAll`); it is read-only metadata access. | +| `atlas-github-integration@…` | `roles/bigquery.jobUser`, `roles/bigquery.dataEditor` | unchanged from Sprint 4 (isolated CI datasets); no Sprint 5 broadening | +| `atlas-github-deployer@…` | `roles/bigquery.jobUser`, `roles/bigquery.dataEditor`, `roles/composer.user`, `roles/composer.environmentAndStorageObjectAdmin` | unchanged from Sprint 4; no Sprint 5 broadening | +| `service-…@cloudcomposer-accounts` | `roles/composer.serviceAgent`, `roles/composer.ServiceAgentV2Ext` | Google-managed Composer service agent (standard) | +| `service-…@gcp-sa-logging` | `roles/logging.serviceAgent` | Google-managed Log Router service agent — this is the sink writer identity for the `atlas-observability` bucket (intra-project sinks use the Logging service account; no custom writer identity was created) | + +No Owner/Editor/broad admin roles were granted to any Atlas identity. The +Cursor development credential retains its pre-existing project access +(documented since Sprint 4) and was used for drill fixtures. + +## Logging and Monitoring access boundary + +- The `atlas-runtime` log view exists for least-privilege consumption; access + is granted per-view via `roles/logging.viewAccessor` (no grants exist yet — + there are no third-party readers). +- **Documented boundary expansion:** the `atlas_logs` linked BigQuery dataset + makes log entries readable to anyone with BigQuery read on that dataset, + bypassing Logging IAM. Mitigations: the dataset is read-only by + construction, no dataset-level grants were added, and the routed content is + contract-sanitized. This trade-off is accepted for queryability and + documented in ADR-011. +- No unrestricted `roles/logging.privateLogViewer` grants exist. +- Data-access audit logs for BigQuery remain enabled (pre-existing) and are + the largest log source; they contain identities and query metadata but no + Atlas payload data. + +## Notification and alerting safety + +- Single email channel, created only after the owner supplied the address + in-session (`ATLAS_APPROVE_ALERT_CHANNEL` flow); the address is not + committed anywhere in the repository — alert policy JSONs carry a + `${NOTIFICATION_CHANNEL}` placeholder that is resolved at apply time from + `ATLAS_NOTIFICATION_CHANNEL_ID`. +- Alert documentation fields contain runbook paths and bounded metric + descriptions; CI (`observability_config` gate) rejects committed channel + ids, secrets, or unresolved runbook anchors. +- Test incidents carry only check names and threshold numbers. + +## CI and deployment boundary (unchanged from Sprint 4, re-verified) + +- Pull-request CI remains credentialless: `atlas-ci.yml` has no GCP + authentication and read-only workflow permissions. +- Cloud mutations run only from trusted workflows/scripts behind WIF with the + dedicated deployer identity, or from the operator's session behind the + `ATLAS_APPROVE_*` variables. +- The deployment bundle excludes notification addresses, channel ids, drill + payloads, and evidence logs (`build_deployment_bundle.sh` allowlist; the + new `observability/metrics` and `observability/schema` assets are static + catalogs). + +## Residual risks + +1. The linked-dataset boundary expansion described above (accepted, + documented). +2. `ATLAS_LOG_TO_CLOUD_LOGGING=true` gives runtime code a direct write path + to Cloud Logging. The writer permission already existed via + `composer.worker`; the env var only activates client usage. Contract + sanitization applies to every event on this path. +3. Drill-mode metric points share the alert policies with normal series + (deliberate — drills must prove the real policies). A malicious or + accidental publisher with `monitoring.timeSeries.create` could open false + incidents; acceptable in a single-operator development project. diff --git a/docs/security-review-sprint6.md b/docs/security-review-sprint6.md new file mode 100644 index 0000000..d5b362c --- /dev/null +++ b/docs/security-review-sprint6.md @@ -0,0 +1,76 @@ +# Atlas Sprint 6 Security Review + +Scope: the resilience additions of Sprint 6 — the controlled fault-injection +framework, the recovery-action audit, cost guards, schema-version handling, and +rollback compatibility. Reviewed live on 2026-07-19 against project +`example-gcp-project`. + +## Fault injection is safe by construction + +The most security-sensitive addition is a framework that deliberately breaks +things. It is fenced by a layered safety contract (ADR-013, +`src/atlas/failure_injection/`), verified live and in CI: + +| Control | Evidence | +| --- | --- | +| Disabled by default | `cli run` REFUSED without `ATLAS_APPROVE_FAILURE_INJECTION=true` (live) | +| Explicit scenario required | CLI errors if `--scenario` absent ("fault injection never runs implicitly") | +| No implicit environment | CLI errors if `--environment` absent for `run`/`cleanup`/`status` | +| Refuses scheduled/canonical/production contexts | `authorize_injection` rejects scheduled execution, canonical batch ids, and non-dev environments (unit-tested) | +| Bounded by duration/cost | each scenario declares `maximum_duration_minutes`/`maximum_cost_usd`; `enforce_deadline` bounds the window | +| Never an unattended destruction engine | `run` only authorizes and prints operator steps; it does not mutate cloud resources itself (CLI docstring + code path) | +| Not hardcodable in production paths | `gate_failure_injection` CI gate proves injection cannot be silently enabled in production code | +| Catalog integrity | `cli validate` = VALID; every scenario carries required fields, allowed categories/risk/mode, controlled recovery actions | + +## Recovery audit integrity + +- `atlas_ops.recovery_actions` records who/what/when for every recovery; `SUCCESS` + is refused unless `verification_status=VERIFIED` (`_validate`, ADR-014), so a + recovery cannot be marked done without evidence. +- `error_summary` is passed through the audited `sanitize_error_message` + (strips bearer tokens/key material/long opaque strings) — no secret leakage + into the audit table. +- Upserts are idempotent by `recovery_id`; re-running a recovery cannot fork the + audit history. + +## Cost guards as a safety control + +Cost guards are also a denial-of-wallet safeguard: `validate_backfill_window`, +`require_full_refresh_approval`, `enforce_dry_run_ceiling`, and +`guarded_query_config` block expensive operations unless an explicit approval +variable is set, and emit `cost_guard_blocked` telemetry when they do. This +prevents a mistyped date or an accidental full refresh from becoming a large, +silent scan. + +## Schema-version handling + +`atlas.validation.schema_versions` rejects unknown schema versions and unknown +fields (no silent coercion of untrusted payloads) and backfills explicit nulls +for older versions — preventing malformed/ambiguous input from being accepted as +valid (ADR-015). + +## IAM evidence (live `gcloud projects get-iam-policy`, 2026-07-19) + +No IAM changes were made in Sprint 6. Atlas identities retain exactly their +Sprint 5 roles: + +| Principal | Roles | Notes | +| --- | --- | --- | +| `atlas-composer-runtime@…` | `composer.worker`, `bigquery.jobUser`, `bigquery.dataEditor`, `bigquery.resourceViewer` | runtime identity; unchanged | +| `atlas-github-integration@…` | `bigquery.jobUser`, `bigquery.dataEditor` | isolated CI datasets; unchanged | +| `atlas-github-deployer@…` | `bigquery.jobUser`, `bigquery.dataEditor`, `composer.user`, `composer.environmentAndStorageObjectAdmin` | deploy identity; unchanged | +| `service-…@cloudcomposer-accounts` | `composer.serviceAgent`, `composer.ServiceAgentV2Ext` | Google-managed | + +No Owner/Editor/broad-admin roles are granted to any Atlas identity. GitHub +authentication remains keyless (Workload Identity Federation, Sprint 4). The +QUARANTINE recovery used the development credential's pre-existing BigQuery +access for the targeted `DELETE`s; it required no new roles. + +## Residual risks + +- The development credential retains broad project access (documented since + Sprint 4) and was used for recovery `DELETE`s. In a production model these + would run under a scoped recovery identity with `bigquery.dataEditor` on the + affected datasets only. +- Fault injection is dev-only by contract; there is no production kill-switch + needed because the authorization chain refuses non-dev environments outright. diff --git a/docs/security-review-sprint7.md b/docs/security-review-sprint7.md new file mode 100644 index 0000000..744d8af --- /dev/null +++ b/docs/security-review-sprint7.md @@ -0,0 +1,54 @@ +# Atlas Security & Data-Exposure Review (Sprint 7) + +Validates that Atlas commits no secrets or sensitive data and that managed IAM +definitions grant no prohibited roles. Enforced by `gate_secret_scan` (existing) +and `gate_security_policy` (Sprint 7). See ADR-018 for IAM boundaries. + +## Scope + +Governed Atlas artifacts: `config/`, `governance/`, `observability/`, `scripts/`, +`src/atlas/`, `sql/`, `dags/`, `docs/`. Out-of-scope trees +(Sprint 8). + +## Checklist results + +| Check | Result | +| --- | --- | +| No committed credentials (private keys, API keys, SA JSON) | PASS (`gate_secret_scan` + `scan_data_exposure`) | +| No untracked credential files in the worktree | PASS (`gate_secret_scan`) | +| No secrets in immutable release bundles | PASS (bundle build excludes credentials; ADR-010) | +| No secrets in structured logs | PASS (field allowlist + truncation, Sprint 5) | +| No secrets in audit error fields | PASS (`sanitize_error_message`, Sprint 4) | +| No Authorization headers with literal tokens | PASS (`bootstrap_observability.sh` uses `Bearer $token` variable) | +| No committed private webhook URLs | PASS (alert JSON uses `${NOTIFICATION_CHANNEL}` placeholder) | +| No committed notification verification data | PASS (only placeholders committed) | +| No real user data | PASS (all event data is synthetic via `generate_events`) | +| No prohibited IAM roles in managed scripts | PASS (`scan_managed_iam` — 0 findings) | +| No service-account key creation | PASS (keyless WIF only) | + +## Discovered gap → regression control + +- **Managed-IAM scanner:** new `scan_managed_iam` fails CI if any Atlas bootstrap + script ever grants `roles/owner`/`roles/editor`/`projectIamAdmin` or creates a + service-account key. This converts the ADR-018 prohibition into an enforced + regression test. +- **Data-exposure scanner:** new `scan_data_exposure` fails CI on literal + secrets, Slack webhooks/tokens, literal bearer tokens, or personal email + addresses in governed artifacts, while explicitly allowing variable + references. Regression tests in `tests/unit/test_security_policy.py`. + +## Public-repository extraction risks (resolved in the template extraction) + +1. **Operator identity:** fixture actor/owner data was replaced with + `` before public extraction. +2. **Project and pool identifiers:** source sandbox values were replaced with + documented examples or runtime configuration. +3. **Notification recipient address:** remains external to Git and is configured + through the target environment. + +## Honest limitations + +- The scanners are pattern-based; they catch the known credential and + exposure shapes, not every conceivable secret format. +- Each adopter must rerun the security and public-extraction gates against its + own configuration and deployment evidence. diff --git a/docs/setup-guide.md b/docs/setup-guide.md new file mode 100644 index 0000000..2784ccc --- /dev/null +++ b/docs/setup-guide.md @@ -0,0 +1,90 @@ +# Project Atlas Setup Guide + +## Prerequisites + +- Google Cloud project: `example-gcp-project` +- Cloud Shell or local shell with Python 3.12 +- Cursor workspace opened at repository root `de-project-1` +- Access to BigQuery and Cloud Storage in the sandbox project + +## 1. Clone and enter Atlas + +```bash +git clone https://github.com/YOUR_GITHUB_OWNER/de-project-1.git +cd de-project-1/project-atlas +python3 -m venv .venv +source .venv/bin/activate +pip install -r requirements.txt +export PYTHONPATH=src +``` + +## 2. Configure credentials + +### Cursor Desktop + +```bash +cp ../.env.example ../.env +gcloud auth application-default login +bash scripts/verify_mcp_access.sh +``` + +Restart Cursor and confirm **Settings → MCP** shows green for `bigquery` and `dbt`. + +### Cursor Cloud Agents + +1. Create a least-privilege service account in `example-gcp-project`. +2. Grant: + - `roles/storage.objectAdmin` on the Atlas bucket + - `roles/bigquery.dataEditor` on dataset `atlas_raw` + - `roles/bigquery.jobUser` at project scope +3. Store base64-encoded JSON in Cursor secret `ATLAS_GCP_SERVICE_ACCOUNT_KEY`. +4. Re-run the cloud agent after secret injection. + +## 3. Bootstrap GCP resources (approval gated) + +```bash +export ATLAS_APPROVE_PROVISION=true +bash scripts/bootstrap_gcp.sh +``` + +This creates: + +- Bucket `atlas-raw-events-example-gcp-project` +- Dataset `atlas_raw` +- Table `events` partitioned by `event_date` + +## 4. Execute Sprint 1 locally + +```bash +python scripts/generate_events.py +pytest +``` + +## 5. Execute Sprint 1 against GCP + +```bash +python scripts/run_pipeline.py --approve-provision +python scripts/validate_events.py --run-id --event-date +``` + +## Environment variables + +| Variable | Purpose | +| --- | --- | +| `GCP_PROJECT_ID` | Sandbox project id | +| `ATLAS_GCP_PROJECT_ID` | Atlas override for project id | +| `ATLAS_GCS_BUCKET` | Physical bucket name | +| `ATLAS_APPROVE_PROVISION` | Required for bootstrap and live pipeline | +| `ATLAS_GCP_SERVICE_ACCOUNT_KEY` | Base64 SA JSON for cloud agents | +| `ATLAS_RANDOM_SEED` | Generator seed override | +| `ATLAS_EVENT_COUNT` | Generator count override | + +## Verification checklist + +- [ ] `pytest` passes locally +- [ ] `bash scripts/verify_mcp_access.sh` succeeds +- [ ] Cursor MCP servers are green +- [ ] Bootstrap completes with approval gate +- [ ] Pipeline generates logs under `logs/` +- [ ] Validation overall status is `FAIL` +- [ ] Acceptance anomaly detection checks are `PASS` diff --git a/docs/sprint4-plan.md b/docs/sprint4-plan.md new file mode 100644 index 0000000..b4b8a56 --- /dev/null +++ b/docs/sprint4-plan.md @@ -0,0 +1,195 @@ +# Sprint 4 Plan — Production Deployment & Operations + +**Yes, you're ready to start Sprint 4.** Sprint 3 is merged, tagged (`atlas-sprint-3-complete`), and live-validated. The natural next step is moving from **local Airflow** to **managed Cloud Composer** with CI/CD — exactly what the Sprint 3 incident report deferred. + +--- + +## Readiness gate + +| Prerequisite | Status | +|---|---| +| Sprint 3 merged to `main` | Done (`8aa1d7a`) | +| Tag `atlas-sprint-3-complete` | Done | +| Live acceptance (5 scenarios) | Done — all PASS | +| Static tests (47/47) | Done | +| Composer path contract documented | Done (ADR-005, preflight) | +| Composer environment exists | **Not started** | +| CI/CD for DAG deploy | **Not started** | +| Monitoring / alerting | **Not started** | + +**Verdict:** Green to begin Sprint 4. No blocking debt from Sprint 3. + +--- + +## Sprint 4 theme + +> **Deploy Atlas to Cloud Composer with automated CI/CD, observability, and a promotion checklist.** + +Sprints 1–3 built the pipeline. Sprint 4 makes it **operable in production**. + +The Artifact Platform is a parallel capability (already has its own Terraform + runbook). Sprint 4 focuses on the **batch ELT pipeline**, not artifact hosting — unless you explicitly want to unify them. + +--- + +## Architecture target + +```mermaid +flowchart LR + subgraph ci [GitHub Actions] + PR[PR / push to main] + Test[pytest + DAG parse] + Deploy[gsutil sync to Composer GCS] + end + + subgraph composer [Cloud Composer 3] + DAGs["/gcs/dags/project_atlas/"] + Data["/gcs/data/"] + DAG[atlas_batch_pipeline] + end + + subgraph gcp [GCP Services] + GCS[(GCS events bucket)] + BQ[(BigQuery atlas_raw + atlas_ops)] + end + + PR --> Test --> Deploy + Deploy --> DAGs + Deploy --> Data + DAG --> GCS + DAG --> BQ +``` + +--- + +## Workstreams + +### 1. Composer infrastructure (Terraform) + + +- Composer 3 environment on image `composer-3-airflow-3.1.7-build.12` (ADR-005) +- Dedicated service account with least-privilege IAM (BigQuery, GCS, Composer worker) +- Environment variables: `ATLAS_ROOT`, `ATLAS_GCP_PROJECT_ID`, `DBT_PROFILES_DIR` +- PyPI packages from `requirements-airflow.txt` via Composer `pypi_packages` +- GCS bucket layout for DAGs vs. runtime data (per preflight contract) + +**Deliverables:** Terraform modules, `terraform.tfvars.example`, bootstrap script gated by `ATLAS_APPROVE_PROVISION=true` + +--- + +### 2. CI/CD pipeline (GitHub Actions) + +No `.github/workflows/` exist today. Add: + +| Workflow | Trigger | Steps | +|---|---|---| +| `atlas-test.yml` | PR + push to `main` | `pytest tests/`, `bash -n scripts/*.sh`, DAG import/parse tests | +| `atlas-deploy-composer.yml` | Push to `main` (post-merge) | Sync DAGs → `/gcs/dags/project_atlas/`, sync scripts+dbt → `/gcs/data/` | + +**Key constraints:** + +- Deploy only changed paths (DAGs vs. data separately, per runbook-sprint3) +- Use Workload Identity Federation or a GitHub secret for GCP auth (no SA keys in repo) +- Fail deploy if `pip check` or DAG parse fails post-sync + +--- + +### 3. Composer promotion checklist & runbook + +Formalize what Sprint 3 left as notes: + +- **Dev → staging → prod** promotion steps (or single-env for personal project) +- Pre-deploy: version pin verification, ADR-005 image availability +- Post-deploy: trigger smoke run (`atlas-$(date +%Y%m%d)`), verify audit row in `atlas_ops.pipeline_runs` +- Rollback: re-sync previous tag's GCS contents + +**Deliverables:** `docs/runbook-sprint4.md`, `docs/promotion-checklist-sprint4.md`, ADR-008 (Composer deployment topology) + +--- + +### 4. Observability & alerting + +Leverage existing `atlas_ops.pipeline_runs` audit table: + +- Cloud Monitoring alert: pipeline run `FAILED` status within 15 min +- Optional: log-based metric from Airflow task failure logs +- Dashboard: batch success rate, rows loaded/accepted/rejected over time +- Wire into existing `write_run_summary` finalizer output + +**Deliverables:** Terraform alerting resources, `docs/monitoring-sprint4.md` + +--- + +### 5. Composer acceptance matrix + +Re-run Sprint 3's 5 scenarios against **Composer** (not local Airflow): + +1. Retry success (`upload_once`) +2. Idempotent rerun +3. dbt failure injection +4. Historical recovery +5. Fresh historical batch + +**Deliverables:** `docs/validation-report-sprint4.md` with Composer-specific evidence + +--- + +## Definition of done + +- [ ] Composer 3 environment provisioned via Terraform (approval-gated) +- [ ] GitHub Actions: test on PR, deploy on merge +- [ ] DAG + data assets synced to Composer GCS paths +- [ ] Smoke run succeeds; audit row `SUCCESS` in `atlas_ops.pipeline_runs` +- [ ] All 5 acceptance scenarios pass on Composer +- [ ] Alert fires on injected failure (scenario 3) +- [ ] Promotion checklist documented and executed once +- [ ] Tag `atlas-sprint-4-complete` + +--- + +## Risks & mitigations + +| Risk | Mitigation | +|---|---| +| Python 3.12 (local) vs 3.11.8 (Composer) | Existing parse-safety tests; add Composer smoke in CI | +| Composer image retired | ADR-005 documents pin; add image availability check to deploy workflow | +| GCP cost (Composer ~$300+/mo) | Use smallest env size; document teardown script | +| GitHub → GCP auth | WIF preferred; fallback to Cursor secret pattern already used | +| dbt profiles on Composer | Mount via GCS data path + `DBT_PROFILES_DIR` env var | + +--- + +## Suggested implementation order + +``` +Phase A — Foundation (can start immediately) + ├── ADR-008: Composer deployment topology + ├── Terraform: Composer environment module + └── Manual first deploy script (deploy_composer.sh) + +Phase B — Automation + ├── GitHub Actions: atlas-test.yml + ├── GitHub Actions: atlas-deploy-composer.yml + └── Promotion checklist doc + +Phase C — Validation & ops + ├── Composer acceptance matrix (5 scenarios) + ├── Monitoring alerts + dashboard + └── validation-report-sprint4.md + tag +``` + +--- + +## Open decisions (need your input) + +1. **Single Composer env or dev+prod?** For a personal sandbox, one env is fine. Say if you want two. +2. **Scheduled vs. manual-only?** Sprint 3 DAG is manual-triggered. Sprint 4 could add a daily schedule — or keep manual until you're confident. +3. **Artifact Platform in scope?** It's deployable separately today. Include in Sprint 4 only if you want a unified "Atlas platform" release. +4. **GitHub Actions auth:** Workload Identity Federation (cleaner) vs. service account key secret (simpler, matches existing Cursor pattern)? + +--- + +## Recommendation + +Start Sprint 4 with **Phase A** — Terraform + manual deploy script + one Composer smoke run. That validates the hardest part (infra + path mapping) before wiring CI/CD. + +If this scope looks right, proceed with branch `cursor/atlas-sprint-4-composer-deploy-64a2`, ADR-008, and the Terraform scaffold. If you had a different Sprint 4 in mind (e.g. streaming, DEOS integration, data quality SLAs), reshape the plan accordingly. diff --git a/docs/sprint7-context-pack.md b/docs/sprint7-context-pack.md new file mode 100644 index 0000000..ca80b22 --- /dev/null +++ b/docs/sprint7-context-pack.md @@ -0,0 +1,89 @@ +# Atlas Sprint 7 Context Pack + +Durable summary from the single Phase-0 repository scan. Use this instead of +re-scanning; do targeted reads only. Paths are under ``. + +## Repository map (in-scope) + +- `src/atlas/` packages: `batch/` (identity, manifest), `config/` (settings), + `failure_injection/` (registry, framework, cli), `generator/` (events), + `ingestion/` (upload), `loader/` (bigquery), `logging/`, `observability/` + (checks, cost, cost_guards, logging, metrics, monitor, schema_drift), + `ops/` (audit, deployments, finalizer, migrations, preflight, quality_results, + recovery_actions, resources, rollback_compatibility, task_events), + `pipeline/` (orchestrator), `validation/` (checks, schema_versions, warehouse). +- `dags/` — `atlas_batch_pipeline.py`, `atlas_observability_monitor.py`, + `atlas_orchestration/` (callbacks, commands, context, validation). +- `dbt/atlas_dbt/models/` — `sources/`, `staging/` (stg_events), + `intermediate/` (int_event_classification, int_accepted_events, + int_rejected_events), `core/` (dim_users, dim_countries, fct_events), + `marts/` (mart_daily_event_metrics). Tests in `dbt/atlas_dbt/tests/`. +- `sql/` migrations 001–008 (`manifest.txt`); ledger `atlas_ops.schema_migrations`. +- `config/` — atlas.yaml, anomaly_profile.yaml, observability.yaml, + failure_scenarios.yaml. +- `observability/` — alerts/, dashboards/, metrics/, queries/, + schema/expected-schemas.json. +- `scripts/` — `validate_ci.sh` (gate framework), deploy/rollback, + `manage_atlas_alerts.sh`, `manage_atlas_composer.sh`, step runner, etc. +- `docs/` + `docs/adr/` (ADR-002..015). + +## Key mechanisms to reuse (do NOT rebuild) + +- **CI gates:** `scripts/validate_ci.sh` → `run_gate `; register in + the `static`/`integration` blocks. `skip_gate ` for SKIPPED. + Results JSON at `logs/ci/validate-ci-results.json`. Mirror in + `.github/workflows/atlas-ci.yml` (credentialless PR CI). +- **Config validation pattern:** `gate_config_validation` / + `gate_observability_config` load YAML and assert structure in an inline + `python3 - <<'PY'`. New governance gates follow this. +- **Cost guards:** `src/atlas/observability/cost_guards.py` — + `validate_backfill_window`, `require_full_refresh_approval`, + `enforce_dry_run_ceiling`, `guarded_query_config`, `estimate_query_bytes`, + `CostGuardViolation`, `emit_event`. Extend here; add `cost_guard estimate` CLI. +- **Schema modules:** `src/atlas/validation/schema_versions.py` + (`SUPPORTED_SCHEMA_VERSIONS`, `CURRENT_SCHEMA_VERSION=2`, `detect/normalize`), + `src/atlas/ops/rollback_compatibility.py` (`evaluate_rollback_compatibility`), + `src/atlas/ops/migrations.py` (`Migration.breaking`, checksum ledger). +- **Audit:** `atlas.ops.*` (recovery_actions model is the template for durable, + validated, idempotent BigQuery upserts with `emit_event`). +- **Settings/labeling:** `atlas.config.settings.load_settings`, + `labeled_bigquery_client(project, component)`. + +## Grain & duplicate facts (Phase 3 critical) + +- `generate_events` seed = `default_seed_for_date(processing_date)` → identical + event_ids for repeated dates. +- `int_event_classification`: `duplicate_rank = row_number() over (partition by + event_id order by ingested_at desc, event_timestamp desc, source_file desc, + raw_record_hash desc)`; `is_duplicate_extra = duplicate_rank > 1` (GLOBAL). +- `fct_events`: incremental merge, unique_key=event_id (global uniqueness). +- Fix must distinguish within-batch dup / cross-batch replay / exact rerun / + conflicting dup / accepted canonical, preserve fct grain, keep 50 within-batch + extras detectable, keep exact rerun idempotent. Prefer batch-scoped duplicate + rank (option A) + replay classification (option B); confirm via ADR-017/an + amendment. Use fixtures, never canonical destructive tests. + +## Live baseline (2026-07-19) + +Composer: 0 envs. Datasets: 9 (atlas_*). Rows: raw.events 850k, +core.fct_events 392,845, marts 2,804, ops.pipeline_runs 25, task_events 257, +recovery_actions 1 (VERIFIED). Alerts: 8 ENABLED, data-stale + composer-unhealthy +DISABLED (correct). GCP project `example-gcp-project`, location US/us-central1. + +## Governance decisions locked in preflight + +- ADR-016: dbt `meta` authoritative for models; `governance/` registry for + non-dbt assets; generated consolidated catalog. No triple maintenance. +- Required asset fields: asset_id, asset_type, purpose, technical_owner, + business_owner_or_role, grain, source, consumers, classification, + retention_class, freshness_expectation, contract_version, lifecycle_status, + repository_path, runbook, last_reviewed. +- Lifecycle: ACTIVE → DEPRECATED → REMOVAL_SCHEDULED → REMOVED. +- Classification: PUBLIC / INTERNAL / CONFIDENTIAL / RESTRICTED. +- Compatibility classes: COMPATIBLE / CONDITIONALLY_COMPATIBLE / BREAKING / + PROHIBITED. + +## New Sprint 7 CI gates (planned) + +`gate_governance`, `gate_schema_compatibility`, `gate_lineage_impact`, +`gate_security_policy`, `gate_performance_cost`. All credentialless/static. diff --git a/docs/sprint8-context-pack.md b/docs/sprint8-context-pack.md new file mode 100644 index 0000000..58e8c64 --- /dev/null +++ b/docs/sprint8-context-pack.md @@ -0,0 +1,93 @@ +# Atlas Sprint 8 Context Pack + +Durable summary from the single Phase-0 repository scan. Purpose: enable targeted +reads for the rest of Sprint 8 without re-scanning. Last verified commit: +`3f986aa` (origin/main). Sprint 8 branch: +`cursor/atlas-sprint-8-reference-handoff-64a2`. + +## Repository map (in-scope: ``) + +``` +src/atlas/ generator ingestion loader validation (Sprint 1 data plane) + batch pipeline config (run context, settings) + logging observability ops (telemetry, audit, cost) + failure_injection (Sprint 6, disabled by default) + governance/ registry catalog lineage impact + schema_check retention security_policy (Sprint 7) +dbt/atlas_dbt/ sources staging intermediate core marts + tests + meta.governance +dags/ atlas_batch_pipeline, atlas_observability_monitor +scripts/ validate_ci.sh (canonical) + 40 others (deploy/rollback/obs/perf/...) +governance/ policy classifications retention consumers non_dbt_assets + schemas/ changes/ generated/(catalog.json/md, lineage.json) +observability/ alerts/ dashboards/ logging/ metrics/ performance/ queries/ schema/ +config/ atlas.yaml anomaly_profile.yaml observability.yaml + failure_scenarios.yaml cost_controls.yaml +sql/migrations/ 001..008 + checksums.lock +docs/ 61 md, adr/ (19), evidence-sprint4..7/ + apps/ packages/ transform/dbt/ +``` + +## Source-of-truth hierarchy (reuse, do not duplicate) + +1. **Code + config** = ground truth for behavior. +2. **dbt `meta.governance`** = model ownership/grain/classification/contract. +3. **`governance/*.yml`** = non-dbt asset governance + policy vocab. +4. **ADRs (002–020)** = decisions and rationale. +5. **`validation-report-sprint{1..7}.md`** = evidence of claims (live vs static). +6. Sprint 8 reference package = a **curated map** that links to 1–5, never a copy. + +## Canonical commands (verified) + +```bash +export PYTHONPATH=src # atlas.* modules live under src/ +bash scripts/validate_ci.sh --mode static # 21 gates, credentialless +python -m atlas.governance.catalog check # governance + drift +python -m atlas.governance.lineage # lineage graph +python -m atlas.governance.impact --asset fct_events +bash scripts/run_performance_suite.sh # dry-run baseline ($0) +python -m atlas.observability.cost_guard estimate --sql-file --project example-gcp-project --location US +``` + +Install for a clean clone: `pip install -r requirements.txt -r requirements-ci.txt` +(the CI file provides yamllint + shellcheck so `workflow_yaml`/`shell_static` +run instead of skipping). dbt: `bash scripts/setup_dbt.sh`. Airflow (optional +local): `airflow/requirements-airflow.txt` (apache-airflow==3.1.7). + +## Key invariants to consolidate in Phase 4 (already true in code) + +- Raw artifacts immutable & run-scoped; `batch_id` = data identity; + `pipeline_run_id` = one execution; exact rerun idempotent. +- `fct_events` grain = one row per `event_id`; within-batch dup vs cross-batch + replay distinguished (ADR-006 amend, Sprint 7 Phase 3). +- accepted + rejected reconciles to raw; quality failure blocks publication. +- Applied migrations immutable (`checksums.lock`); PR CI credentialless; + releases immutable; smoke gates success; rollback checks schema compat; + Composer ephemeral. +- Governance metadata single source of truth; owners+grain required; schema + changes classified; breaking needs migration+impact; permanent evidence can't + get transient retention; secrets never in evidence. + +## Live vs static evidence (must stay separated) + +- **Live-proven:** Sprints 1–6 pipeline/deploy/recovery (BigQuery, Composer, + CI runs in validation reports); Sprint 7 dry-run perf baseline + $0 cost block. +- **Static/test-proven:** Sprint 7 governance/schema/lineage/deprecation/security + gates (282 tests + offline gates). +- **Blocked (not executed):** Sprint 7 live IAM reduction, billed perf suite, + live retention application. + +## Public-extraction hotspots (Phase 14 input) + +Personal email/name in ~35 files (alerts JSON notification channel, Sprint 4/5 +WIF/IAM docs, `bootstrap_github_wif.sh`, setup guides). Private project id +`example-gcp-project`, bucket/SA/dataset names pervasive. No credentials, +keys, or tokens committed (Sprint 7/8 secret_scan clean). No absolute local +paths or conversation-context references in `docs/`. + +## Sprint 8 governance rules for its own artifacts + +- Reference docs carry a `reference-manifest.yml` entry with `last_verified_commit`. +- Every major claim in the evidence index maps to a path + type + live/static. +- Blocked work is labeled BLOCKED, never "complete". +- One ADR only if a real decision is made (ADR-021 reference/handoff contract). +- New gate reuses the `run_gate`/`in_group` framework; no YAML logic duplication. diff --git a/docs/technical-design-review.md b/docs/technical-design-review.md new file mode 100644 index 0000000..35bf2ca --- /dev/null +++ b/docs/technical-design-review.md @@ -0,0 +1,96 @@ +# Project Atlas Technical Design Review — Sprint 1 + +## Objective + +Deliver a cloneable, operable, and recoverable batch ingestion pipeline that +demonstrates senior data engineering competency without implementing future-phase +components prematurely. + +## Architectural decisions + +### 1. Nested project inside `de-project-1` + +**Preferred because:** workspace-root MCP config must remain active for both Cursor +Desktop and Cloud Agents. + +**Alternatives considered:** separate repository (cleaner ownership boundary) and +replacing the current repo (would discard DEOS scaffold). + +**Tradeoffs:** two Python contexts and documentation overhead. + +**Evolution:** v0.3 adds dbt models under `transform/dbt`; v0.4 adds Airflow DAGs +under `orchestration/airflow/dags` that invoke Atlas scripts. + +### 2. Python modules plus shell bootstrap scripts + +**Preferred because:** Cloud Shell-first execution with testable business logic. + +**Alternatives considered:** shell-only pipeline and notebook-driven prototyping. + +**Tradeoffs:** more files, but stronger testing and clearer ownership. + +**Evolution:** Airflow BashOperator or PythonOperator can wrap the same scripts. + +### 3. Run-scoped immutable GCS paths + +**Preferred because:** Sprint 1 requires history never be overwritten. + +**Alternatives considered:** date-only paths and object versioning alone. + +**Tradeoffs:** longer object keys and more listing noise. + +**Evolution:** backfill jobs can filter by `run_id` while retaining partition layout. + +### 4. Staging-table load into partitioned target + +**Preferred because:** explicit metadata enrichment and idempotent run replay. + +**Alternatives considered:** direct append load and external tables. + +**Tradeoffs:** one transient table per run. + +**Evolution:** dbt staging model replaces transient table logic in v0.3. + +### 5. Validation FAIL with acceptance anomaly detection + +**Preferred because:** seeded bad rows should not produce a false green quality gate. + +**Alternatives considered:** PASS when anomaly counts match profile and quarantine +invalid rows in Sprint 1. + +**Tradeoffs:** operators must inspect acceptance checks, not only overall status. + +**Evolution:** dbt tests and expectations reuse `config/anomaly_profile.yaml`. + +## Security posture + +- No credentials committed to git +- Cloud agent auth via Cursor secret `ATLAS_GCP_SERVICE_ACCOUNT_KEY` +- Bootstrap and live pipeline gated by `ATLAS_APPROVE_PROVISION=true` +- `.gcp/` ignored at repository root + +## Testing strategy + +- Unit tests for settings, generator, upload naming, and acceptance logic +- Integration test for local generate-only pipeline +- Acceptance tests for Sprint 1 repository layout and artifact generation +- Failure simulation script for missing file and bad schema cases + +## Known limitations + +- No dbt, Airflow, Terraform, CI/CD, or monitoring in Sprint 1 +- Cloud agent MCP availability depends on Cursor secret injection and environment setup +- Live end-to-end execution requires sandbox permissions and explicit approval + +## Validation report interpretation + +| Signal | Expected Sprint 1 result | +| --- | --- | +| Overall validation status | `FAIL` | +| Row count / partition / schema checks | `PASS` after successful load | +| Duplicate/null/invalid/future/late checks | `FAIL` | +| Acceptance anomaly detection checks | Exact match to seeded profile | +| `future_timestamps` | `event_date > CURRENT_DATE()` — future calendar date, not clock time | + +This is the intended professional posture: the pipeline detects bad data and +reports failure, while automated acceptance proves the detector works. diff --git a/docs/template-configuration.md b/docs/template-configuration.md new file mode 100644 index 0000000..8bcffe5 --- /dev/null +++ b/docs/template-configuration.md @@ -0,0 +1,44 @@ +# Template Configuration + +## Required runtime parameters + +| Parameter | Purpose | +| --- | --- | +| `ATLAS_GCP_PROJECT_ID` | Target GCP project | +| `ATLAS_GCP_PROJECT_NUMBER` | Numeric project identifier for WIF/IAM | +| `ATLAS_GCP_LOCATION` | BigQuery multi-region or region | +| `ATLAS_GCP_REGION` | Regional services such as Composer | +| `ATLAS_GCS_BUCKET` | Immutable raw-ingestion bucket | +| `ATLAS_RELEASE_BUCKET` | Immutable release-bundle bucket | +| `ATLAS_DATASET_PREFIX` | Prefix for BigQuery datasets | +| `ATLAS_SERVICE_ACCOUNT_PREFIX` | Prefix for provisioned identities | +| `ATLAS_DAG_ID` | Airflow DAG identifier | +| `ATLAS_SCHEDULE` | Airflow schedule | +| `ATLAS_NOTIFICATION_EMAIL` | Operator notification destination | +| `ATLAS_COST_CEILING_BYTES` | Pre-execution query guard | + +## GitHub repository variables + +Trusted workflows require: + +- `ATLAS_WIF_PROVIDER` +- `ATLAS_INTEGRATION_SERVICE_ACCOUNT` +- `ATLAS_DEPLOYER_SERVICE_ACCOUNT` + +Pull-request CI must remain credentialless. Do not add cloud credentials to PR +workflows merely because authentication is annoying. Authentication is supposed +to be annoying when the alternative is accidental infrastructure mutation. + +## Adoption gate + +Before calling an adoption complete, prove: + +1. clean clone and static CI +2. isolated GCP deployment +3. successful batch and warehouse reconciliation +4. deliberate failure and targeted recovery +5. alerts and runbook routing +6. schema compatibility behavior +7. IAM and secret review +8. cost ceiling and cleanup +9. operator handoff diff --git a/docs/token-efficiency-sprint7.md b/docs/token-efficiency-sprint7.md new file mode 100644 index 0000000..f4cbb48 --- /dev/null +++ b/docs/token-efficiency-sprint7.md @@ -0,0 +1,54 @@ +# Atlas Sprint 7 Token & Compute Efficiency + +Target: **55–75% of actual Sprint 6 agent consumption**. Exact token telemetry +is not exposed to the agent, so proxies are tracked. Updated at closeout. + +## Efficiency strategy + +1. One comprehensive Phase-0 scan (done) → this + the context pack; targeted + reads afterward. +2. Reuse existing controls (CI gate framework, cost guards, audit upsert + pattern, schema modules) — no parallel governance/lineage/security/cost + platforms. +3. dbt manifest + repo artifacts for lineage (no graph DB / metadata service). +4. Fixtures for all destructive/breaking demonstrations; canonical data never + mutated for theatre. +5. One bounded live GCP window; Composer only if a control genuinely needs it + (preflight expectation: **not required** — governance/perf/cost/retention/IAM + provable via BigQuery + IAM APIs + isolated datasets). +6. Focused independent review only for schema compatibility, IAM, performance + methodology, and final P0/P1. +7. Dry-run before every cost experiment; hard byte ceiling enforced. + +## Proxy ledger (closeout) + +Exact token/request telemetry is not exposed to the agent; proxies below are the +closeout values. + +| Proxy | Sprint 6 (reference) | Sprint 7 (closeout) | +| --- | --- | --- | +| Full repository scans | several | 1 (Phase 0) | +| Live Composer create/delete cycles | 1 (long) | 0 (Composer not required) | +| Live deployment cycles | 2 (one false-negative rerun) | 0 (no Composer/deploy) | +| Live GCP windows | 1 long acceptance window | 1 bounded, read-only + dry-run ($0) | +| Failed acceptance reruns | baseline had to move dates | 0 | +| Failed GitHub CI runs on branch | — | 1 (`secret_scan` flagged own fixture; fixed in `7f0eff6`) | +| Major plan regenerations | — | 0 | +| Human correction events | a few | 0 (autonomous; approvals gated, not corrected) | + +Interpretation: the dominant Sprint 6 cost drivers (a long live Composer +lifecycle, two live deployment cycles, and multiple full scans) were all avoided +in Sprint 7. Sprint 7 used one comprehensive scan, targeted reads, reused +existing controls (CI-gate framework, cost guards, audit/upsert patterns, schema +modules), and one bounded read-only + dry-run live window. On the proxy signals +available, Sprint 7 consumption sits comfortably inside the 55–75%-of-Sprint-6 +envelope. + +## Scope-compression tripwire + +Not triggered. Projected work stayed under 75% of Sprint 6 consumption, so no +compression was needed. The planned levers (defer performance *changes* while +keeping the baseline measurement; reduce live demos to the highest-value +enforcement proof) were unnecessary — and, independently, the gated live +demonstrations (IAM reduction, executed performance suite, retention mutation) +remained blocked on unset approval variables, which further bounded live spend. diff --git a/docs/token-efficiency-sprint8.md b/docs/token-efficiency-sprint8.md new file mode 100644 index 0000000..0d7b524 --- /dev/null +++ b/docs/token-efficiency-sprint8.md @@ -0,0 +1,50 @@ +# Atlas Sprint 8 Token & Compute Efficiency + +Target: **50–65% of Sprint 7 agent consumption**. Sprint 8 is documentation and +validation heavy, not implementation heavy: it curates existing evidence rather +than building new platforms. Exact token telemetry is not exposed to the agent; +proxies are tracked and updated at closeout. + +## Efficiency strategy + +1. One comprehensive Phase-0 scan (done) → preflight + this context pack; targeted + reads afterward. +2. **Link, don't duplicate** — the reference package points to Sprint 1–7 docs, + ADRs, tests, and validation reports instead of re-prosing them. +3. One navigational reference package; one focused handoff CI gate + (`gate_reference_handoff`) — not many unrelated gates. +4. Fresh directories for clean-clone reproduction; one independent handoff + subagent (only repo + START_HERE + assignment). +5. No Composer, no billed BigQuery, no IAM/retention mutation unless the + corresponding approval variable is present. +6. Reuse existing diagrams/text-diagrams; regenerate only if materially wrong. +7. Record human interventions during handoff testing honestly. + +## Proxy ledger (updated through the sprint) + +| Proxy | Sprint 7 (reference) | Sprint 8 (actual) | +| --- | --- | --- | +| Full repository scans | 1 | 1 (Phase 0) + targeted reads | +| Composer create/delete cycles | 0 | **0** | +| Live deployment cycles | 0 | **0** | +| Billed BigQuery workloads | 0 (dry-run only) | **0** | +| Live GCP windows | 1 bounded read-only/dry-run | **0** (no approvals set) | +| New CI gates added | 5 | **1** (`gate_reference_handoff`) | +| Independent handoff runs | — | 1 (scored 29/30) | +| Clean-clone attempts | — | 2 (attempt 1 found defect, attempt 2 PASS) | +| Failed onboarding steps repaired | — | 2 (PYTHONPATH=src; PEP 668 venv) | +| Major plan regenerations | 0 | **0** | +| Human correction events | 0 | 0 (tester self-resolved friction) | + +Model: Opus 4.8. Harness: Cursor Cloud Agent. Cloud operations: 0 mutating, +0 billed. Composer cycles: 0. The envelope target (50–65% of Sprint 7) was met: +Sprint 8 was documentation/validation work with no cloud provisioning, one new +gate, and a single independent handoff subagent. + +## Scope-compression tripwire + +If projected work approaches **75% of Sprint 7** consumption, stop and propose +compression: e.g., collapse the five operating/security/reliability/observability/ +cost model docs into fewer files that link harder to existing runbooks; reduce +the handoff to the single highest-value independent assignment; defer the +optional read-only GCP leg. Recorded here if triggered. diff --git a/docs/validation-report-sprint1.md b/docs/validation-report-sprint1.md new file mode 100644 index 0000000..d85700c --- /dev/null +++ b/docs/validation-report-sprint1.md @@ -0,0 +1,72 @@ +# Project Atlas Sprint 1 Validation Report + +## Run metadata + +- Pipeline run id: `atlas-20260714T163527Z-19a0e4f6` +- Event date: `2026-07-14` +- Source GCS URI: `gs://atlas-raw-events-example-gcp-project/raw/event_date=2026-07-14/run_id=atlas-20260714T163527Z-19a0e4f6/events.jsonl` +- Target table: `example-gcp-project.atlas_raw.events` +- Engineer: the primary operator +- Environment: GCP Cloud Shell + +## Core checks + +| Check | Expected | Actual | Status | +| --- | --- | --- | --- | +| row_count | 50000 | 50000 | PASS | +| partition_presence | >0 rows | 49550 | PASS | +| schema_required_fields | true | true | PASS | +| distinct_event_ids | 49950 | 49950 | PASS | +| duplicate_rows | 50 | 50 | PASS | +| duplicates | 50 groups | 50 | FAIL | +| null_user_ids | 500 | 500 | FAIL | +| invalid_country_codes | 200 | 200 | FAIL | +| future_timestamps | 150 future-dated rows | 150 after validator fix | FAIL | +| late_arriving_events | 300 | 300 | FAIL | +| partition_reconciliation | primary + other = total | 49550 + 450 = 50000 | PASS | + +## Acceptance checks + +| Check | Expected | Actual | Status | +| --- | --- | --- | --- | +| acceptance_duplicate_detection | 50 | 50 | PASS | +| acceptance_null_user_detection | 500 | 500 | PASS | +| acceptance_invalid_country_detection | 200 | 200 | PASS | +| acceptance_future_timestamp_detection | 150 | 150 after validator fix | PASS | +| acceptance_late_arrival_detection | 300 | 300 | PASS | + +## Partition reconciliation + +```text +49,550 rows on primary event_date (2026-07-14) ++ 300 late-arriving rows (event_date earlier than timestamp date) ++ 150 future-dated rows (event_date > generation date) += 50,000 total loaded rows +``` + +Note: late-arriving and future-dated rows are seeded on distinct indices in Sprint 1. + +## Future timestamp semantics + +Sprint 1 defines a future-dated anomaly as: + +```sql +event_date > CURRENT_DATE() +``` + +This measures future **calendar dates**, not timestamps later than the validation clock. + +## Overall result + +- Validation overall status: `FAIL` (expected for seeded raw anomalies) +- Acceptance anomaly detection status: `PASS` after validator fix +- Log file: `logs/atlas-20260714T163527Z-19a0e4f6.jsonl` + +## Notes + +Initial live run exposed two issues that were corrected before archival: + +1. JSONL included `ingested_at` before load-time enrichment. +2. Future timestamp validation used clock-time comparison and over-counted same-day rows. + +Both were fixed on branch `cursor/project-atlas-sprint1-3660`. diff --git a/docs/validation-report-sprint2.md b/docs/validation-report-sprint2.md new file mode 100644 index 0000000..829d8b3 --- /dev/null +++ b/docs/validation-report-sprint2.md @@ -0,0 +1,100 @@ +# Project Atlas Sprint 2 Validation Report + +## Status + +**PASS** — live validation completed in GCP Cloud Shell on 2026-07-14. + +## Scope + +- Validated Sprint 1 run id: `atlas-20260714T163527Z-19a0e4f6` +- Raw table: `example-gcp-project.atlas_raw.events` +- dbt project: `dbt/atlas_dbt` +- Engineer: the primary operator +- Environment: GCP Cloud Shell (`example-gcp-project`) + +## Live gates (Cloud Shell) + +| Gate | Expected | Actual | Status | +| --- | --- | --- | --- | +| `dbt debug` | connection OK | OAuth OK, location US | PASS | +| `dbt seed` | 10 country rows | 10 rows in `atlas_staging.valid_country_codes` | PASS | +| source freshness | warn/error thresholds | PASS | PASS | +| `dbt build --full-refresh` | success | 76/76 steps in 48.12s | PASS | +| singular tests | 4/4 | 4/4 | PASS | +| generic + unit tests | all pass | 63/63 | PASS | +| raw/classification reconciliation | 50,000 = 50,000 | PASS | PASS | +| accepted + rejected reconciliation | 50,000 total | PASS | PASS | +| mart/fact reconciliation | equal totals | PASS | PASS | +| anomaly profile (validated run) | exact counts | see below | PASS | + +Validation JSON: `logs/validation-sprint2-20260714T194335Z.json` + +## Anomaly counts (validated run) + +| Measure | Expected | Actual | Status | +| --- | ---: | ---: | --- | +| duplicate_extra | 50 | 50 | PASS | +| null_user_id | 500 | 500 | PASS | +| invalid_country (physical) | 200 | 200 | PASS | +| future_dated | 150 | 150 | PASS | +| event_time_late_arriving | 0 | 0 | PASS | +| backdated_event_date | 300 | 300 | PASS | +| date_timestamp_mismatch | 300 | 300 | PASS | + +## Relation inventory (full refresh build) + +| Relation | Rows (approx.) | Materialization | +| --- | ---: | --- | +| `atlas_staging.stg_events` | 50,000 | view | +| `atlas_intermediate.int_event_classification` | 50,000 | table | +| `atlas_intermediate.int_accepted_events` | 49,106 | view | +| `atlas_quarantine.int_rejected_events` | 894 | table | +| `atlas_core.dim_users` | 38,900 | table | +| `atlas_core.fct_events` | 49,100 | incremental table | +| `atlas_marts.mart_daily_event_metrics` | 571 | table | + +Accepted canonical rows plus rejected physical rows reconcile to 50,000 raw rows. + +## Performance evidence (full refresh) + +| Step | Duration | Notes | +| --- | ---: | --- | +| `int_event_classification` build | 2.92s | 50k rows, 12.1 MiB processed | +| `int_rejected_events` build | 2.07s | 894 rows, 14.0 MiB processed | +| `fct_events` build | 3.48s | 49.1k rows, 13.5 MiB processed | +| `mart_daily_event_metrics` build | 2.23s | 571 rows, 1.8 MiB processed | +| Total `dbt build` | 48.12s | 76 steps, 0 errors | + +dbt artifacts preserved under `logs/dbt-artifacts/20260714T194137Z/`. + +## Incremental idempotency (unchanged raw source) + +**PASS** — validated in GCP Cloud Shell on 2026-07-14 after merge to `main` at `a137590`. + +Command: + +```bash +cd ~/Atlas-GCP-Build/project-atlas +export ATLAS_GCP_PROJECT_ID=example-gcp-project +bash scripts/validate_dbt_sprint2_incremental.sh +``` + +| Gate | Before | After | Status | +| --- | ---: | ---: | --- | +| `fct_events` row count | 49,106 | 49,106 | PASS | +| `mart_daily_event_metrics` event total | 49,106 | 49,106 | PASS | +| `int_rejected_events` row count | 894 | 894 | PASS | +| `dbt build` (no `--full-refresh`) | — | 76/76 in 60.61s | PASS | +| singular tests | — | 4/4 | PASS | + +Validation JSON: `logs/validation-sprint2-incremental-20260714T201453Z.json` + +## Notes + +Sprint 1 terminology called 300 rows "late_arriving_events" using +`event_date < DATE(event_timestamp)`. Sprint 2 reclassifies those rows as backdated declared +dates. See [ADR-003](adr/ADR-003-corrected-temporal-semantics.md). + +Singular anomaly test counts physical invalid-country rows (`NOT is_valid_country`), not +terminal `rejection_reason`, because null-user precedence suppresses some invalid-country +rejections while the physical defect remains. diff --git a/docs/validation-report-sprint3.md b/docs/validation-report-sprint3.md new file mode 100644 index 0000000..83a75e9 --- /dev/null +++ b/docs/validation-report-sprint3.md @@ -0,0 +1,116 @@ +# Atlas Sprint 3 Validation Report + +## Static validation (Cloud Agent) + +| Gate | Result | +|------|--------| +| Unit + airflow tests (`pytest tests/`) | **47/47 PASS** | +| Shell syntax (`bash -n scripts/*.sh`) | PASS | +| DAG parse helpers (no network at import) | PASS | +| Composer path configuration tests | PASS | + +## Live acceptance results (Cursor Cloud Agent — 2026-07-18) + +Executed against GCP project `example-gcp-project` using the injected +`ATLAS_GCP_SERVICE_ACCOUNT_KEY` service account and a local Airflow 3.1.7 +standalone. Root-cause fixes were required before any run progressed past the +first task (see "Diagnosed defects" below). + +| # | Scenario | conf | Airflow run_id suffix | Audit status | Result | +|---|----------|------|-----------------------|--------------|--------| +| 1 | Retry success | `{processing_date:2026-07-18, batch_id:atlas-20260718, upload_once:true}` | `s1-20260718T170705Z` | `SUCCESS` | **PASS** — upload failed try 1, retried and succeeded, full dbt build + reconciliation green | +| 2 | Idempotent rerun | `{processing_date:2026-07-18, batch_id:atlas-20260718}` | `s2-idem-20260718T171453Z` | `SUCCESS` | **PASS** — GCS + raw load skipped (`already_loaded: true`); raw count stayed 50000 (not doubled) | +| 3 | dbt failure injection | `{processing_date:2026-07-01, batch_id:atlas-20260701, dbt_test_failure:true}` | `s3-dbtfail-20260718T171838Z` | `FAILED` | **PASS** — dbt build failed, no success marker, finalizer raised, audit `FAILED`; backfill freshness skipped | +| 4 | Historical recovery | `{processing_date:2026-07-01, batch_id:atlas-20260701}` | `s4b-recovery-20260718T180023Z` | `SUCCESS` | **PASS** (after temporal-semantics fix) — raw load skipped, dbt build 79/79, audit `SUCCESS` | +| 5 | Fresh historical batch | `{processing_date:2026-07-16, batch_id:atlas-20260716}` | `s5-freshhist-20260718T180320Z` | `SUCCESS` | **PASS** — native load wrote `processing_date` (50000 rows, 0 null); classified reproducibly | + +Idempotency evidence (`atlas_raw.events`): batch `atlas-20260718` = **50000 rows, +1 distinct pipeline_run_id** after two runs. + +Before/after for the temporal-semantics fix (same historical batch): scenario 4 +was `FAILED` at `s4-recovery-...T172222Z`, then `SUCCESS` at +`s4b-recovery-...T180023Z` after the fix below. + +## Diagnosed defects (fixed in this branch) + +Every task initially failed. Root causes, all verified empirically against GCP: + +1. **`run_atlas_step.sh` `${2:-{}}`** (primary) — bash parsed the default value as + `{` plus a literal trailing `}`, appending a stray `}` to the JSON run context, + so `json.loads` raised `Extra data` in **every** task. Replaced with an explicit + default. +2. **`ops/audit.py` `_param_type(None)`** returned `STRING` for INT64 columns, so + the audit MERGE failed (`Value of type STRING cannot be assigned … INT64`), + blocking `start_run_audit`. Nullable numeric fields are now typed `INT64`. +3. **`ops/preflight.py`** used a broken `__import__(...).bigquery.Client` expression + that always raised `AttributeError`, forcing preflight to `FAIL`. Replaced with a + proper `from google.cloud import bigquery` import. +4. **`atlas_step_runner.py` / `airflow.env.example`** default GCS bucket name was + missing `-events-`. Corrected. + +## Resolved finding — historical backfills vs. anomaly profile + +Originally, `assert_source_anomaly_profile` hard-asserted `future_dated=150`, +`event_time_late=0`, `backdated=300` — counts derived from `stg_events` flags +computed relative to wall-clock `ingested_at`. They were only reproducible for +same-day ingestion, so historical recovery (scenario 4) failed and, more +importantly, event classification itself was load-time dependent. + +**Fix (option b + stopgap a), verified live:** +- **(b)** Added a nullable `processing_date` column to `atlas_raw.events` + (persisted by the loader) and switched the three temporal flags to + `COALESCE(processing_date, DATE(ingested_at))`. Backfills now classify + identically to the original run. See ADR-003 "Sprint 3 refinement". +- **(a)** `assert_source_anomaly_profile` asserts the temporal counts only when + every scoped row has `processing_date`, degrading gracefully for legacy rows. + +Result: scenario 4 recovered `FAILED`→`SUCCESS`; a fresh historical batch +(`atlas-20260716`) loaded with native `processing_date` and passed 79/79. + +## Live acceptance matrix (original Cloud Shell design) + +Execute in `~/Atlas-GCP-Build/project-atlas` after merging Sprint 3: + +### 1. Retry success (`upload_once`) + +```bash +bash scripts/run_airflow_sprint3.sh \ + --processing-date $(date -u +%F) \ + --batch-id atlas-$(date -u +%Y%m%d) \ + --conf '{"upload_once": true}' +``` + +**Expected:** upload try 1 fails, try 2 succeeds, audit `SUCCESS`, run-summary reconciled. + +### 2. Idempotent rerun (same batch, new pipeline run) + +Re-trigger the same `--batch-id` with a new `--run-id`. + +**Expected:** GCS skip, raw load skip, new audit row, zero fact/mart drift. + +### 3. dbt failure (`dbt_test_failure`) + +```bash +bash scripts/run_airflow_sprint3.sh \ + --processing-date 2026-07-01 \ + --batch-id atlas-20260701 \ + --conf '{"dbt_test_failure": true}' +``` + +**Expected:** dbt build fails, no success marker, audit `FAILED`. + +### 4. Historical recovery + +Re-run batch `atlas-20260701` without injection. + +**Expected:** raw skip, successful backfill, freshness skipped, audit `SUCCESS`. + +## Evidence (Cursor Cloud Agent — 2026-07-18, project `example-gcp-project`) + +| Batch | pipeline_run_id | Status | Notes | +|-------|-----------------|--------|-------| +| current + upload_once | `atlas-airflow-20260718-manual__s1-20260718T170705Z` | `SUCCESS` | upload retried once then succeeded | +| idempotent rerun | `atlas-airflow-20260718-manual__s2-idem-20260718T171453Z` | `SUCCESS` | raw load skipped; raw stayed 50000 rows | +| historical dbt failure | `atlas-airflow-20260701-manual__s3-dbtfail-20260718T171838Z` | `FAILED` | dbt build failed as injected; finalizer raised | +| historical recovery (post-fix) | `atlas-airflow-20260701-manual__s4b-recovery-20260718T180023Z` | `SUCCESS` | raw skip OK; dbt build 79/79 after temporal-semantics fix | +| fresh historical batch | `atlas-airflow-20260716-manual__s5-freshhist-20260718T180320Z` | `SUCCESS` | native `processing_date` load; reproducible classification | diff --git a/docs/validation-report-sprint4.md b/docs/validation-report-sprint4.md new file mode 100644 index 0000000..b5d4249 --- /dev/null +++ b/docs/validation-report-sprint4.md @@ -0,0 +1,169 @@ +# Sprint 4 Validation Report — Live Acceptance Evidence + +Evidence-backed record of the Sprint 4 acceptance sequence (Phase 19). +Nothing below is claimed from static files alone; every row cites an +executed run, an audit record, or a preserved log. + +Companion documents: `architecture-sprint4.md` (design), +`incident-report-sprint4.md` (gate demos + defect ledger), +`deployment-catalog-sprint4.md` (resource inventory), +`ci-cd-runbook-sprint4.md` (operator commands), +`ci-cd-governance-sprint4.md` (GitHub plan limitations). + +## 1. Executive result + +A clean Project Atlas change moved from an agent-created Git branch through +independent GitHub CI, keyless WIF authentication, an isolated integration +test, an immutable checksum-verified bundle, audited additive migrations, a +real Composer 3 deployment (`composer-3-airflow-3.1.7-build.13`), a 50 000-row +smoke batch with full warehouse reconciliation, a durable deployment audit +row, a deliberately failed defective deployment, and a live rollback that +restored and re-validated the prior release — with no manual code copying, no +stored service-account keys in GitHub, and no unverifiable runtime state. +The Composer environment was deleted immediately after evidence capture per +the ephemeral cost mandate (ADR-010). + +## 2. Git and CI evidence + +| Item | Value | +|---|---| +| Foundation PR | #14 (squash-merged `21d54ed`) — clean CI on run `29660771549` | +| Gate-demo PR | #15 (closed unmerged by design) — 4 deliberate failures, run IDs in `incident-report-sprint4.md` | +| Delivery PR | #16, branch `cursor/atlas-sprint-4-delivery-64a2` | +| Defect-demo branch | `cursor/atlas-sprint-4-defect-demo-64a2` @ `1af166e` (never merged; exists only as an immutable release for the rollback demo) | + +## 3. Keyless authentication (WIF) + +| Item | Value | +|---|---| +| Pool / provider | `atlas-github-pool` / `atlas-github-provider` (OIDC issuer `token.actions.githubusercontent.com`) | +| Trust condition | repository owner + exact repository `YOUR_GITHUB_OWNER/YOUR_REPOSITORY` + ref restriction | +| Identities | `atlas-github-integration` (isolated CI resources), `atlas-github-deployer` (deploy path) | +| Keys stored in GitHub | none — `id-token: write` + impersonation only | + +IAM matrix and documented-risk notes: `ADR-009-workload-identity-federation.md`. + +## 4. Isolated integration test (Phase 7) + +`validate_gcp_integration.sh` executed live: run-scoped datasets +(`atlas_ci__…`) and GCS prefix, deterministic generation verified +byte-identical (after fixing defect D3), idempotent raw loading, dbt build +against isolated schemas, batch-scoped reconciliation, verified cleanup in an +always-running trap. No canonical dataset was written. + +## 5. Deployment bundle (Phase 8) + +| Item | Value | +|---|---| +| Released bundle (final) | `gs://atlas-deployments-example-gcp-project/atlas/releases/640cd78694a90275866bebbaf7550ee121fff79b/atlas-bundle.tar.gz` | +| Archive SHA-256 | `aaa83bc1578057e8f88b38f68aa9785a956099538f6ebf7aa69936c6575e6e7c` | +| Manifest | 314 files with per-file SHA-256, tool pins, `required_schema_version=003_create_deployments_table` | +| Immutability | create-only upload; content-identical retry reuses, different content fails | + +## 6. Migrations (Phase 9) + +Ledger `atlas_ops.schema_migrations` (all applied idempotently; reruns skip): + +| migration_id | checksum (first 12) | status | applied_at (UTC) | +|---|---|---|---| +| `001_create_pipeline_runs_table` | `5fb06a83e1b3` | APPLIED | 2026-07-18 22:13:45 | +| `002_sprint3_raw_batch_columns` | `db8b53e68ee6` | APPLIED | 2026-07-18 22:13:49 | +| `003_create_deployments_table` | `d581c625ad1e` | APPLIED | 2026-07-18 22:13:52 | + +## 7. Composer deployment (Phases 12–13) + +| Item | Value | +|---|---| +| Environment | `atlas-dev`, us-central1, `composer-3-airflow-3.1.7-build.13`, small | +| Lifecycle | created ~22:05 UTC 2026-07-18, **deleted** ~01:35 UTC 2026-07-19 after evidence capture (ephemeral policy, ADR-010) | +| Successful deployment | `atlas-dev-20260719T005308Z-640cd786` → **SUCCESS** | +| Deployed SHA | `640cd78694a90275866bebbaf7550ee121fff79b` | +| Smoke DAG run | `smoke__atlas-dev-20260719T005308Z-640cd786` → Airflow terminal `success` | +| Smoke pipeline run | `atlas-smoke-640cd786-local1784422388-run` → `pipeline_runs.status=SUCCESS` (00:58:18 UTC) | +| Smoke validation | 12/12 checks PASS (dag import, no import errors, deployed SHA, terminal success, 50 000 raw rows, no duplicate load, GCS object, manifest, success marker, warehouse reconciliation, pipeline_runs, deployments row) | + +Reaching SUCCESS took seven audited attempts; each failure exposed and fixed +a real defect (checksum-by-filename, Composer sys.path parity, stale import +errors, missing vendored dbt packages, Airflow 3 state parsing, rsync vs +deterministic mtimes). Full ledger: `incident-report-sprint4.md`. + +## 8. Deliberate failed deployment + live rollback (Phases 14/16) + +| Step | Evidence | +|---|---| +| Defective release | `1af166e` (`inject_failure: true` — dbt canary test fails) built and uploaded as a normal immutable bundle | +| Failed deployment | `atlas-dev-20260719T010538Z-1af166ea` → **FAILED**, `failure_stage=smoke_batch`; Airflow shows `dbt_build` failed, downstream `upstream_failed`; no success metadata published | +| Rollback | `rollback_atlas.sh` auto-selected newest prior SUCCESS (`640cd78`), verified manifest/checksums/schema compatibility, re-promoted | +| Rollback record | `atlas-dev-20260719T011614Z-640cd786` → **ROLLED_BACK**, `previous_git_sha=1af166e` | +| Rollback smoke | batch `atlas-smoke-640cd786-local1784423774`: 50 000 raw → 49 105 accepted + 895 rejected → 49 105 fact rows; `pipeline_runs.status=SUCCESS` (01:21:56 UTC); 12/12 smoke checks PASS | + +Preserved logs: `evidence-sprint4/deploy-defective-1af166ea.log`, +`evidence-sprint4/rollback-640cd786.log`, +`evidence-sprint4/composer-deploy-session-history.txt`. + +## 9. Deployment audit table (final state) + +```text +deployment_id sha type status failure_stage previous +atlas-dev-20260718T225900Z-dd7dd5d4 dd7dd5d4 deploy FAILED fetch_release — +atlas-dev-20260718T230144Z-2aeff26e 2aeff26e deploy FAILED dag_parse — +atlas-dev-20260718T231842Z-83c0d137 83c0d137 deploy FAILED dag_parse — +atlas-dev-20260718T233128Z-74732eee 74732eee deploy FAILED smoke_batch — +atlas-dev-20260719T001246Z-f9959cb6 f9959cb6 deploy FAILED smoke_batch — +atlas-dev-20260719T004112Z-37d4e6aa 37d4e6aa deploy FAILED smoke_validation — +atlas-dev-20260719T005308Z-640cd786 640cd786 deploy SUCCESS — — +atlas-dev-20260719T010538Z-1af166ea 1af166ea deploy FAILED smoke_batch — +atlas-dev-20260719T011614Z-640cd786 640cd786 rollback ROLLED_BACK — 1af166ea +``` + +One row per attempt, controlled statuses, sanitized errors, rollback linked +to the SHA it replaced — exactly the Phase 10 contract. + +## 10. Cost review + +Composer small environment existed ~3.5 hours (creation → deletion), the +only meaningfully billable Sprint 4 resource. Remaining permanent resources +scale to zero: versioned deployment bucket (KB–MB scale), 7-day-TTL CI +bucket, BigQuery ops tables (MB scale), service accounts and WIF (free). + +## 11. Honest limitations + +- Branch protection / required reviewers unavailable on the GitHub Free + plan; fallback governance documented in `ci-cd-governance-sprint4.md` and + ADR-010. Completion gate 7 is satisfied via the documented-limitation arm. +- The successful deployment evidence was captured by executing the same + repository scripts the GitHub workflows call (`deploy_atlas_release.sh`, + `validate_atlas_deployment.sh`, `rollback_atlas.sh`) from the agent + environment with ADC, because Composer create/delete cycles are gated on + cost approval and the ephemeral environment was deleted after capture. The + workflows themselves are exercised for auth and validation; a future + GitHub-initiated deploy run requires only re-creating the environment + (`manage_atlas_composer.sh create`) and dispatching `atlas-deploy.yml`. +- Composer worker logs did not surface in Cloud Logging during the capture + window; failure diagnosis used the Airflow REST API, task-state listings, + and BigQuery audit logs instead. Recorded as a Sprint 5 observability + handoff item. +- The dbt warehouse rebuilds intermediate/fact tables scoped to the + validated batch per run (Sprint 2 design), so cross-batch history in + those tables reflects the newest build; per-run validation is performed at + smoke time. Durable per-run evidence lives in `atlas_ops`. + +## 12. Sprint 5 handoff (observability and alerting — requirements only) + +Recorded per the Sprint 4 charter; none of this was implemented in Sprint 4: + +- Centralized structured task logs — Composer worker logs did not surface in + Cloud Logging during the Sprint 4 capture window (limitation above); Sprint 5 + must make task logs durably queryable before anything else. +- Pipeline and data freshness metrics (batch latency, last-success age per DAG). +- Failure alerts on `pipeline_runs.status=FAILED` and + `deployments.status IN (FAILED, ROLLBACK_FAILED)`. +- Volume and schema monitoring (row-count drift per batch, schema-change detection + against the migration ledger). +- Operational dashboards over `atlas_ops` (runs, deployments, migrations). +- On-call and escalation rules; incident-recovery drill cadence. +- SLO and error-budget candidates (smoke-run duration, deploy lead time, + batch success rate). +- Stale-data detection (no successful batch within an expected window). +- Cost anomaly detection (Composer create/delete discipline, BigQuery scan + volume, bucket growth). diff --git a/docs/validation-report-sprint5.md b/docs/validation-report-sprint5.md new file mode 100644 index 0000000..83797fe --- /dev/null +++ b/docs/validation-report-sprint5.md @@ -0,0 +1,176 @@ +# Atlas Sprint 5 Validation Report — Observability, Alerting, Incident Readiness + +All claims below are backed by live execution on 2026-07-19 in project +`example-gcp-project` (evidence artifacts in `docs/evidence-sprint5/`), by +GitHub CI runs, or by unit/acceptance tests in this repository. Anything not +proven is listed under "Unresolved limitations". + +## 1. Git and PR evidence + +| Item | Value | +| --- | --- | +| Sprint 4 closeout | PR #18 merged; `atlas-sprint-4-complete` → `4251e94` (release commit `b609ac1a…`) | +| Sprint 5 base (origin/main) | `45543b68e392dde722e7c84baca8252406c2053a` | +| Sprint 5 branch / PR | `cursor/atlas-sprint-5-observability-64a2` / PR #19 (merged 2026-07-19T12:28:04Z) | +| Sprint 5 release commit (main) | `476e20a2edcd9e6ae2e7aa2169d0f0c0fb13247c` | +| Release tag | `atlas-sprint-5-complete` (annotated `fd79562`) → `476e20a` | +| Candidate head at acceptance | `2109310b6abbb0112efeb43fa46e997ff29ed9db` | +| CI on candidate head | run `29676039069` — atlas-ci **success** (all gates) | +| Earlier iteration evidence | runs `29673933498` (ba16f3c, success), `29672095446` (8fe17dc, success); failures `29671974067`/`29671854476` were the secret-scan and mypy defects fixed in-branch | + +## 2. Deployed releases during acceptance + +| Release SHA | Deployment id | Result | +| --- | --- | --- | +| `8fe17dc` | `atlas-dev-20260719T042641Z-8fe17dcd` | SUCCESS (first Sprint 5 candidate; exposed smoke re-validation defect, fixed in `ba16f3c`) | +| `ba16f3c` | `atlas-dev-20260719T045055Z-ba16f3cd` | SUCCESS — 12/12 smoke checks (Drill A batch, 50 000 rows) | +| `2109310` | `atlas-dev-20260719T061802Z-2109310b` | SUCCESS — 12/12 smoke checks; carries the two acceptance fixes | + +Composer environment: `atlas-dev`, `composer-3-airflow-3.1.7-build.13`, +us-central1, SMALL, runtime SA `atlas-composer-runtime@…`. Both DAGs +(`atlas_batch_pipeline`, `atlas_observability_monitor`) parse with zero +import errors. + +Lifecycle (ephemeral policy, ADR-010): created 2026-07-19T03:44:08Z, deleted +2026-07-19T07:37:56Z (≈ 3.9 h). Teardown sequence: both DAGs paused → +environment-dependent alerts disabled (`Atlas: Composer environment +unhealthy`, `Atlas: data stale`) → drill series reset to PASS → environment +deleted → orphaned Composer bucket removed → verified no incident opened +after teardown (zero `ViolationOpen` events post-07:10Z). Permanent +observability resources (log bucket/sink/view, linked dataset, metric +descriptors, alert policies, channel, dashboard, audit tables) remain +valid. + +## 3. Observability planes (ADR-011) — live evidence + +### Plane 1 — Operational audit (BigQuery) + +- Migrations `004_create_task_events_table`, `005_create_quality_results_table`, + `006_create_monitor_evaluations_table`: APPLIED in the + `atlas_ops.schema_migrations` ledger (checksums recorded). +- `task_events`: 74 rows captured for the three drill/smoke runs alone + (`task-events-drills.json`) — STARTED/SUCCESS/FAILED/UPSTREAM_FAILED at + (run, task, attempt, event) grain; retry-then-success distinguishable; + repeated callbacks idempotent (unit-tested MERGE). +- `quality_results`: 20 rows for the drill runs (`quality-results-drills.json`) + — `validate_warehouse` now persists real batch-scoped reconciliation + results instead of printing PASS. +- `monitor_evaluations`: 91 rows during the window + (`monitor-evaluations.json`), including drill-fixture rows explicitly + tagged `source=atlas_drill_fixture`. + +### Plane 2 — Logs (Cloud Logging) + +- Dedicated bucket `atlas-observability` (us-central1, 30-day retention, + analytics enabled), sink `atlas-observability-sink` (additive; `_Default` + untouched), view `atlas-runtime`, linked read-only dataset `atlas_logs` + (`log-routing-live.json`). +- Structured contract events flow live: 20 correlated entries for the Drill B + run retrievable by one `jsonPayload.pipeline_run_id` filter + (`drillb-correlated-logs.json`); the same entries are queryable through the + linked dataset via SQL (15 rows returned in the verification query). +- **Platform defect (open):** Composer 3 `build.13` exports no Airflow + component logs (worker/scheduler/task streams) to the customer project — + reproduced from Sprint 4 and now root-cause-bounded: no `airflow-*` log + names exist in any bucket including `_Default`; a manual `entries.write` to + the identical logName/resource succeeds and routes correctly through both + buckets; Composer's own task-log reader returns "Logs not found"; no + org-policy/quota/IAM/exclusion cause exists in the project; a workload + restart did not recover it. Mitigation shipped: `ATLAS_LOG_TO_CLOUD_LOGGING=true` + makes every Atlas contract event write directly to logName `atlas-events` + (never-raise, allowlist + sanitizer enforced), restoring queryable + task-level telemetry. Raw Airflow stdout remains unavailable on this build. + +### Plane 3 — Metrics and incidents (Cloud Monitoring) + +- 15 custom descriptors under `custom.googleapis.com/atlas/...`; live series + with real values for all pipeline/data/deployment metrics + (`metric-timeseries-summary.json`). `check_status` has 22 series (11 checks + × normal/drill) — inside the cardinality budget; labels validated at + publish time (`validate_point`) and in CI. +- 10 alert policies live and enabled, all routed to the verified email + channel `…/notificationChannels/6567861337166986657` + (`notification-channel.json` — address not committed). +- Dashboard "Atlas Operations" (31 tiles) deployed and updated in place + (`dashboard-live.json`). + +## 4. Monitor DAG + +`atlas_observability_monitor` runs every 30 minutes in Composer (unpaused for +the acceptance window), evaluates 11 checks with bounded windows, persists +evaluations, publishes metrics, and emits structured events. NO_DATA and +DISABLED states behave as designed (verified in evaluations evidence: +`cost_anomaly` NO_DATA before the IAM grant and with an insufficient +baseline). + +## 5. Controlled drills (all executed live) + +| Drill | Mechanism | Result | Incident evidence | +| --- | --- | --- | --- | +| A — normal success | 50 000-row smoke batches on `ba16f3c` and `2109310` + daily scheduled run 06:00 | complete task events, quality PASS, metrics live, dashboard current | n/a (healthy) | +| B — pipeline failure | deployed DAG run with `dbt_test_failure: true` | `dbt_build` FAILED → run FAILED → monitor FAIL → **incident 06:46:19** → email dispatch → clean rerun SUCCESS 06:58:00 → **auto-resolved 07:03:11** | `Atlas: pipeline failed`, violation `0.oaf4n04jvxx1`; full report in `incident-report-sprint5.md` | +| C — stale data | real freshness check, drill-only thresholds (1 s/2 s) via env override, `mode=drill` | freshness FAIL (age 196 s vs 2 s) | `Atlas: data stale` opened 07:04:54 (`0.oaf52a4f18hp`) | +| D — volume anomaly | fixture rows (10 000 vs 50 000 baseline) through the real check | FAIL, deviation 0.8 | `Atlas: critical volume deviation` opened 07:05:23 (`0.oaf52ofe3mrz`) | +| E — schema drift | doctored expected-schema manifest vs **live** `INFORMATION_SCHEMA` (no canonical mutation) | 2 BREAKING (type change, removed field) + 1 ALLOWED (allow-listed additive) — classification correct | `Atlas: breaking schema drift` opened 07:06:19 (`0.oaf53g1toeh7`) | +| F — cost anomaly | synthetic drill-mode FAIL point (`manage_atlas_alerts.sh test cost_anomaly`); zero real spend | policy fired | `Atlas: BigQuery cost anomaly` opened 07:06:49 (`0.oaf53uuk0td1`) | +| G — telemetry failure | fixture: terminal run with 9/13 terminal task events through the real check | FAIL, 4 missing | `Atlas: telemetry incomplete` opened 07:07:11 (`0.oaf545p85441`); plus the earlier **real** false-positive incident 06:05:51→06:33:00 that motivated the terminal-runs fix | +| Cleanup | PASS recovery points published to every drill series (`delete-test-resources`) | drill incidents auto-close after cessation | `incident-events.json` | + +## 6. Defects found by live acceptance (converted into controls) + +1. Smoke re-validation without `pipeline_run_id` crashed the new quality + persistence → guard + fixed in `ba16f3c`. +2. `telemetry_completeness` false-positive on in-flight runs (opened a real + incident at 06:05:51) → terminal-runs-only + regression test (`2109310`). +3. Composer log-export platform defect → direct-emission mitigation + tests + (`2109310`), documented limitation. +4. Cost check 403 (`bigquery.jobs.listAll`) → `roles/bigquery.resourceViewer` + grant + bootstrap script update. +5. Migration runner semicolon-in-comment and failed-migration-retry defects → + fixed earlier in-branch with regression tests. + +## 7. Notification evidence + +- Channel: email, `projects/example-gcp-project/notificationChannels/6567861337166986657`, + display name "Atlas Primary Operator (email)", enabled; recipient is the + project owner's address supplied in-session (domain-only in evidence). +- Dispatch: 7 incident-open events and their notifications between 06:05 and + 07:07 (policy → channel binding shown in `alert-policies` live state). +- Limitation: Cloud Monitoring exposes no per-email delivery log; delivery + confirmation rests on the channel configuration, the incident dispatch + records, and the owner's mailbox. + +## 8. CI evidence + +The `observability_config` static gate validates thresholds, metric-catalog +cardinality, schema manifest, alert JSONs (placeholder channel, no secrets, +resolvable runbook anchors), dashboard JSON, and log-filter definitions. +Full static suite green locally and in GitHub CI on `2109310` +(run `29676039069`). + +## 9. Completion-gate status (32 gates) + +Gates 1–6, 8–32: **met** with the evidence above and in +`docs/evidence-sprint5/`. + +Gate 7 ("Composer task logs are queryable through documented filters"): +**partially met** — Atlas task-level telemetry is queryable through the +documented `atlas-events` filters (structured contract events for every task +lifecycle transition), but raw Airflow component stdout is not exported by +Composer 3 `build.13` at all (platform defect, diagnosed and documented). +This is recorded honestly rather than claimed. + +## 10. Unresolved limitations + +- Composer 3 `build.13` component-log export defect (platform; mitigated for + Atlas telemetry, unresolved for raw Airflow streams). Re-test on the next + build upgrade. +- Email delivery latency not independently measurable (no delivery log for + email channels). +- `task_events.FAILED` rows lack `completed_at`/`duration_ms` (Sprint 6 + cleanup). +- Thresholds in `observability.yaml` are initial operational thresholds + calibrated on the synthetic 50 000-row workload — not production SLOs. +- Single-operator development ownership model; no real on-call rotation. +- Cost attribution covers labeled Python/dbt jobs; console-issued ad-hoc + queries attribute only by identity. diff --git a/docs/validation-report-sprint6.md b/docs/validation-report-sprint6.md new file mode 100644 index 0000000..f3a0ba9 --- /dev/null +++ b/docs/validation-report-sprint6.md @@ -0,0 +1,104 @@ +# Atlas Sprint 6 Validation Report — Resilience, Failure Engineering, Recovery, Game Days + +All claims below are backed by live execution on 2026-07-19 in project +`example-gcp-project`, by GitHub CI runs, or by unit/acceptance tests in this +repository. Anything not proven live is stated as such under coverage and +limitations. + +## 1. Git and PR evidence + +| Item | Value | +| --- | --- | +| Sprint 6 base (origin/main) | `078bc319c6683717e6583b4500af60b4dd3e168a` | +| Sprint 6 branch / PR | `cursor/atlas-sprint-6-resilience-64a2` / PR #21 | +| Candidate git_sha at live acceptance | `b735823bc5193782bad73a73f3222a2eeafafbae` | +| CI on candidate | run `29691106793` — atlas-ci **success** (all gates) | +| Earlier green iteration | run `29689314252` (success) | + +## 2. Deployed release during acceptance + +| Release SHA | Deployment id | Result | +| --- | --- | --- | +| `b735823` | `atlas-dev-20260719T145242Z-b735823b` | SUCCESS — 12/12 smoke checks; migrations 007/008 applied | + +Composer environment: `atlas-dev`, `composer-3-airflow-3.1.7-build.13`, +us-central1, SMALL. Both DAGs parse with zero import errors. Env vars include +`ATLAS_LOG_TO_CLOUD_LOGGING=true`, `ATLAS_ENVIRONMENT=atlas-dev` (Sprint 5 +log-export mitigation, now set at create time). Lifecycle is ephemeral (ADR-010): +created 2026-07-19T14:31Z, deleted at end of acceptance (teardown gated on +`ATLAS_APPROVE_TEARDOWN`, recorded below). + +## 3. Schema changes (live) + +| Migration | State | +| --- | --- | +| `007_create_recovery_actions_table` | APPLIED — `atlas_ops.recovery_actions` present (20 columns) | +| `008_add_task_event_timing_columns` | APPLIED — `atlas_ops.task_events` has `timing_source`, `timing_confidence` | + +## 4. The nine resilience obligations — evidence + +| # | Obligation | Evidence | +| --- | --- | --- | +| 1 | Detect failures | Ingestion retry (S6-ING-006), dbt test failure (S6-DBT-002), overlapping runs (S6-AIR-004), cost-guard blocks (S6-COST-002/003) all surfaced with `task_events`/`pipeline_runs`/telemetry — see `game-day-results-sprint6.md` | +| 2 | Contain failures | Failed runs published no success marker; DAG paused to stop scheduler contention (INC-S6-002); cost guards block before spend | +| 3 | Diagnose failures | Root causes established from `task_events`, `INFORMATION_SCHEMA.JOBS`, dbt output, and row-count queries (INC-S6-001/002) | +| 4 | Recover with control | `QUARANTINE_BATCH` targeted repair (no full refresh) — INC-S6-001 | +| 5 | Verify recovery | `validate_warehouse("atlas-20260717")` = 10/10 PASS post-recovery; `SUCCESS` gated on `VERIFIED` in `recovery_actions` | +| 6 | Prevent recurrence | Follow-ups documented per incident; DAG pause during deploy window | +| 7 | Preserve data correctness | Global dedup keeps one row per `event_id`; baseline reconciles 10/10 throughout | +| 8 | Durable operational history | `pipeline_runs`, `task_events` (now with timing provenance), `recovery_actions`, `quality_results` | +| 9 | Reproducible evidence | This report + `game-day-results-sprint6.md` + two incident reports, all with concrete ids | + +## 5. Failed-task timing provenance (Phase 1 fix) + +Before Sprint 6, terminal task events recorded by the Airflow failure callback +overwrote `started_at`/`completed_at`/`duration_ms` with NULL. After the fix, +FAILED/RETRY events carry non-null timing plus provenance +(`timing_source` ∈ {`step_runner_clock`, `airflow_task_instance`}; +`timing_confidence` ∈ {`exact`, `partial`, `none`}). Verified live on run +`atlas-airflow-20260719-baseline-s6-20260719` (see game-day results). +Regression tests: `tests/airflow/test_callback_timing.py`. + +## 6. Static / CI coverage + +`scripts/validate_ci.sh` (static mode) is green, including the new +`gate_failure_injection` gate (catalog schema valid; fault injection disabled by +default; not hardcoded in production paths). Unit/airflow tests added: +`test_failure_injection.py`, `test_recovery_actions.py`, `test_cost_guards.py`, +`test_callback_timing.py`, plus `test_run_context.py` backfill-guard cases. + +## 7. Live acceptance coverage vs. static coverage + +Live-injected this window: S6-ING-006, S6-DBT-002/004, S6-AIR-004, S6-COST-002, +S6-COST-003, plus a full verified `QUARANTINE_BATCH` recovery and the fault +-injection safety gate. Per the game-day plan, the remaining catalog scenarios +(deploy/rollback ledger checks, schema-version normalization, rollback +compatibility, dry-run ceiling, guarded query config, and the additional +ingestion/warehouse/IAM/observability variants) are proven by CI gates and unit +tests rather than live injection, to keep the ephemeral, cost-bounded Composer +window short (ADR-010) and to avoid destructive cloud operations. This split is +explicit and honest: it is a deliberate scope decision, not a coverage claim +that live injection did not occur. + +## 8. Exit state and teardown + +- Healthy baseline `atlas-20260717` reconciles 10/10 (post-recovery). +- INC-S6-001 recovered + VERIFIED; INC-S6-002 contained. +- `recovery_actions` holds one VERIFIED row (`rec-s6-quarantine-atlas20260719`). +- Alert policies restored to repo-defined state (10/10 ENABLED) before teardown. +- Composer deleted under `ATLAS_APPROVE_TEARDOWN`: both DAGs paused → + environment-dependent alerts disabled (`Atlas: data stale`, + `Atlas: Composer environment unhealthy`) → environment deleted + (2026-07-19T16:14Z, ≈ 1.7 h lifecycle from 14:31Z) → orphaned Composer bucket + removed (381 objects) → `composer environments list` = 0 items. Permanent + resources (audit tables incl. `recovery_actions`, log bucket/sink/view, metric + descriptors, the 8 non-environment alert policies, notification channel, + dashboard) remain valid; the recovery audit row survives teardown. + +## 9. Unresolved limitations + +- The batch-scoped anomaly-profile test is sensitive to same-date reprocessing + (INC-S6-001 root cause). Recommended fixes are listed in that incident report; + they are Sprint 7 candidates, not regressions introduced by Sprint 6. +- Live game-day coverage is a representative subset (section 7); full live + injection of all 56 catalog scenarios was intentionally not performed. diff --git a/docs/validation-report-sprint7.md b/docs/validation-report-sprint7.md new file mode 100644 index 0000000..673ad0c --- /dev/null +++ b/docs/validation-report-sprint7.md @@ -0,0 +1,210 @@ +# Atlas Sprint 7 Validation Report — Governance, Contracts, Schema Evolution, Security, Performance, Cost + +Every claim below is backed by one of: a committed artifact in this repository, +a green CI gate in `scripts/validate_ci.sh`, a unit/dbt test, or a bounded live +GCP dry-run (billing $0) in project `example-gcp-project`. Gated live +mutations that were **not** approved are recorded as blocked gates, not as +proof. Documentation is never treated as live evidence. + +## 1. Git and release evidence + +| Item | Value | +| --- | --- | +| Sprint 6 completion tag | `atlas-sprint-6-complete` → `48da9d226e0e942095c0623e5bd71d06f5e32e36` (matches prompt) | +| Sprint 7 base (origin/main) | `48da9d2` (+ `1c2cc15` release-row doc) | +| Sprint 7 branch / PR | `cursor/atlas-sprint-7-governance-64a2` / PR **#22** | +| Prior CI failure (pre-fix) | run `29702642835` — `atlas-security-shell` FAIL (`secret_scan` flagged its own fixtures) | +| Fix | `7f0eff6` — allowlist `test_security_policy.py` in `gate_secret_scan` (mirrors `test_audit.py`) | +| Local static CI after fix | `validate_ci.sh --mode static` = **PASS** (all runnable gates green) | +| GitHub CI on final branch head `cb85c3c` | run **`29702939614`** — atlas-ci **success** (atlas-python, atlas-dbt, atlas-security-shell, atlas-airflow, atlas-ci-gate all green) | +| Final merge SHA / tag | recorded at closeout (Phase 16 steps 40–42), after merge to main (awaiting merge authorization) | + +Sprint 1–6 tags are unchanged. `atlas-sprint-7-complete` is created only after +green GitHub CI on the merged main commit (completion gate 34). + +## 2. Token-efficiency result + +Proxy ledger (`docs/token-efficiency-sprint7.md`): **1** full repository scan +(Phase 0), **0** Composer create/delete cycles, **0** live deployment cycles, +**0** major plan regenerations. Composer was proven unnecessary for Sprint 7 +controls (governance/schema/lineage/security/retention/perf/cost are provable +offline or via BigQuery dry-run + IAM read APIs), consistent with the preflight +expectation and the 55–75%-of-Sprint-6 envelope. No scope-compression tripwire +was triggered. + +## 3. Sprint 6 inherited limitations — disposition + +| Inherited limitation | Sprint 7 disposition | +| --- | --- | +| Batch-scoped anomaly profile sensitive to same-date reprocessing (INC-S6-001) | **Resolved** (Phase 3): within-batch vs cross-batch replay now distinguished; anomaly assertion counts `is_within_batch_duplicate` only | +| Live game-day coverage a representative subset | Out of scope for Sprint 7; governance controls proven by gates + fixtures | + +## 4. Governance architecture and one source of truth (gates 2–3) + +- dbt models carry authoritative `meta.governance` (purpose/grain/owner/ + classification/retention/contract/consumers/lifecycle); non-dbt assets live in + `governance/non_dbt_assets.yml`. No third manual copy — `gate_governance` + rejects a duplicate source of truth. +- Consolidated catalog generated from those sources: + `python -m atlas.governance.catalog check` → *"governance catalog matches + sources"*, **22 assets**, all with a technical owner and (for models) a grain. + +## 5. Contracts and schema compatibility (gates 7–11) + +- Data-contract standard + boundary map: `docs/data-contract-standard-sprint7.md` + (ADR-016). Contracts map to executable controls (dbt contracts, tests, JSON + schema, Python validation, CI). +- `atlas.governance.schema_check` classifies changes COMPATIBLE / + CONDITIONALLY_COMPATIBLE / BREAKING / PROHIBITED against a committed + `governance/schemas/manifests/baseline.json` (ADR-017). +- Applied-migration immutability: `sql/migrations/checksums.lock`; + `gate_schema_compatibility` fails on any checksum change. +- Evidence (fixtures, no defect merged): additive nullable → COMPATIBLE + (`test_added_nullable_field_is_compatible`); breaking type change → BREAKING + (`test_type_change_is_breaking`); checksum tamper detected + (`test_migration_checksum_tamper_is_detected`). + +## 6. Duplicate and replay semantics (gates 12–15) + +Phase 3 resolves the Sprint 6 defect while preserving the **one-row-per-event_id** +`fct_events` grain (ADR-006 amendment): + +- `within_batch_duplicate_rank` (latest-wins within a batch) vs global + `duplicate_rank` (first-seen-batch-wins) with `duplicate_scope` ∈ + {`none`, `within_batch`, `cross_batch_replay`}. +- Exact rerun stays idempotent (fct merge `unique_key`); same date under a + different batch classified as `cross_batch_replay`; 50 intentional within-batch + extras remain detectable; global fact uniqueness enforced; accepted+rejected + reconciles to raw. +- Live dbt tests PASS: `test_cross_batch_replay_preserves_first_seen`, + `test_duplicate_ranking_keeps_latest_canonical`. + +## 7. Lineage and consumer impact (gates 16–17) + +- `atlas.governance.lineage` builds the graph from dbt `ref()`/`source()` + + `consumers.yml`; committed `governance/generated/lineage.json` + (**26 nodes, 29 edges**, source→mart intact, verified by `gate_lineage_impact`). +- `atlas.governance.impact` reports direct/transitive downstream assets, affected + tests/contracts, consumers, owners-to-notify, and runbooks + (`test_impact_identifies_downstream_models`). No graph DB / metadata service. + +## 8. Deprecation lifecycle (gate 18) + +`ACTIVE → DEPRECATED → REMOVAL_SCHEDULED → REMOVED` enforced in +`registry.deprecation_errors()`; CI rejects deprecation without replacement, +removal before the minimum window, and removed assets with active consumers +(`test_deprecated_without_replacement_fails`, +`test_removed_asset_with_active_consumer_fails`). Runbook: +`docs/deprecation-runbook-sprint7.md`. + +## 9. IAM review (gates 19–22) + +- Inventory + evidence matrix: `docs/iam-review-sprint7.md` (ADR-018). Keyless + WIF only; no Owner/Editor/SA keys; prohibited patterns enforced by + `gate_security_policy` (`scan_managed_iam`). +- One justified reduction candidate identified: `atlas-github-integration` + project-level `roles/bigquery.dataEditor` is broader than required; a scoped + reduction + positive/negative test plan is documented. +- **Blocked gate:** live IAM reduction and positive/negative tests require + `ATLAS_APPROVE_IAM=true` (not set). Recorded, not weakened, not faked. Gate 20 + is satisfied by a documented, evidence-backed reduction plan pending approval. + +## 10. Security and data-exposure review (gate 23) + +`docs/security-review-sprint7.md`: no committed/untracked credentials, no secrets +in logs or audit error fields, no real user data. `scan_data_exposure` + +`scan_managed_iam` run in `gate_security_policy`; the scanner reports *reasons*, +never values (`test_security_policy.py`). Public-repository extraction risks are +cataloged for Sprint 8 (repo not published in Sprint 7). + +## 11. Classification and retention (gates 5–6, 24–25) + +`governance/classifications.yml` (PUBLIC/INTERNAL/CONFIDENTIAL/RESTRICTED, no +RESTRICTED assets present) and `governance/retention.yml` +(canonical/operational/temporary/release/test-fixture) validated by +`atlas.governance.retention.validate_retention_config` inside `gate_governance` +(ADR-019). Permanent evidence cannot be given an expiration; transient classes +must have one; conflicting policies fail CI (`test_retention.py`). Live +expiration application is gated on `ATLAS_APPROVE_RETENTION_MUTATION` (not set): +disposal is a dry-run plan only (`plan_expirations`). + +## 12. BigQuery performance baseline (gates 26–27) + +`scripts/run_performance_suite.sh` + `observability/performance/queries/` (9 +representative queries). Dry-run baseline `results/baseline-dryrun.json`: every +query well under the 1 GiB per-query ceiling (largest 34.4 MB `08_cost_monitor`; +mart aggregation 44.9 KB). Partition pruning demonstrated (bounded 2.3 MB vs +unbounded 12.7 MB). Conclusion — evidence-backed **"no material change +warranted"** (`docs/performance-review-sprint7.md`); no optimization made merely +to produce a percentage. **Blocked gate:** executed (billed) suite requires +`ATLAS_APPROVE_PERFORMANCE_TESTS=true` / `ATLAS_MAX_PERFORMANCE_TEST_BYTES` +(not set). + +## 13. Cost controls (gate 28) + +`config/cost_controls.yaml` (per-env ceilings, partition-filter requirements, +TTLs) + `atlas.observability.cost_guard` (ADR-020). Live dry-run block evidence +(`docs/evidence-sprint7/cost-guard-block.txt`, billed **$0**): a deliberately +unbounded `atlas_raw.events` scan is refused before spend by (a) the +required-partition-filter guard (exit 2) and (b) the dry-run estimate ceiling. + +## 14. CI enforcement (gates 13, 29) and controlled demonstrations + +Five focused offline gates run in the `python` group and are wired into +`.github/workflows/atlas-ci.yml` with no duplicated logic: `gate_governance`, +`gate_schema_compatibility`, `gate_lineage_impact`, `gate_security_policy`, +`gate_performance_cost`. The 16 required controlled demonstrations are mapped in +`docs/governance-demos-sprint7.md`; 14 are proven offline / via live dbt tests +and pass in CI (**282 tests collected**); #11 is the live dry-run cost block; +#13–14 are the gated live IAM tests above. + +## 15. Live GCP evidence and cleanup (gates 30–31) + +Bounded live window used read-only IAM inventory and BigQuery **dry-run** +estimates only (billed $0). No Composer created (not required). No temporary +datasets/buckets created, so no teardown needed. Canonical Atlas data is +untouched by Sprint 7 (governance overlay + classification-semantics fix only); +the healthy baseline continues to reconcile under the declared grain. + +## 16. Documentation inventory (gate 32) + +All Phase 17 artifacts present: preflight, context pack, token-efficiency, +architecture, governance-model, data-contract-standard, schema-evolution-policy, +lineage-impact, deprecation-runbook, iam-review, security-review, +retention-policy, performance-review, cost-review, and this validation report. +ADR-016 through ADR-020 present (ADR-006 amended for replay semantics). README +updated (release-tag table, governance/perf commands, Sprint 8 handoff). + +## 17. Scope adherence (gate 33) + +`transform/dbt/**`. No new platform, catalog service, external policy engine, or +second repository. Reusable-template extraction and reference-architecture +packaging are explicitly deferred to Sprint 8. + +## 18. Blocked completion gates (honest limitations) + +The following require operator approval variables that are **not set**; they are +implemented statically with exact live plans and recorded as blocked, per the +prompt's missing-approval behavior: + +| Gate | Approval required | Status | +| --- | --- | --- | +| Live IAM reduction + positive/negative test (demos #13–14) | `ATLAS_APPROVE_IAM=true` | Plan complete; blocked | +| Executed (billed) BigQuery performance suite | `ATLAS_APPROVE_PERFORMANCE_TESTS=true` (+ byte ceiling) | Dry-run done; execution blocked | +| Live retention/expiration application | `ATLAS_APPROVE_RETENTION_MUTATION=true` | Dry-run plan done; application blocked | + +No live proof is claimed for these. No gate was weakened to pass. Sprint 7 does +not claim enterprise-wide governance, regulatory certification, production-scale +performance from a ~50k-row dataset, complete least privilege without +permission-level negative-test evidence, full consumer discovery outside the +repository, zero-cost operation, or public-repository readiness (Sprint 8). + +## 19. Completion-gate summary + +Gates 1–19, 23–29, 31–33 are met with committed artifacts, green CI, and $0 +live dry-run evidence. Gate 20 is met by a documented, evidence-backed reduction +plan pending `ATLAS_APPROVE_IAM`. Gates 21–22 (live positive/negative IAM), +24 (live retention application), and 30 (temporary-resource teardown — none +created) are blocked or not-applicable as recorded above. Gate 34 +(`atlas-sprint-7-complete` on the validated merge commit) is completed at +closeout after green GitHub CI on merged main. diff --git a/docs/validation-report-template.md b/docs/validation-report-template.md new file mode 100644 index 0000000..9d5a19a --- /dev/null +++ b/docs/validation-report-template.md @@ -0,0 +1,46 @@ +# Project Atlas Sprint 1 Validation Report Template + +Use this template after a live pipeline run. + +## Run metadata + +- Pipeline run id: +- Event date: +- Source GCS URI: +- Target table: `example-gcp-project.atlas_raw.events` +- Engineer: +- Environment: Cloud Shell / Cursor Desktop / Cursor Cloud Agent + +## Core checks + +| Check | Expected | Actual | Status | +| --- | --- | --- | --- | +| row_count | 50000 | | | +| partition_presence | >0 rows | | | +| schema_required_fields | true | | | +| duplicates | FAIL | | | +| null_user_ids | FAIL | | | +| invalid_country_codes | FAIL | | | +| future_timestamps | FAIL | | | +| late_arriving_events | FAIL | | | + +## Acceptance checks + +| Check | Expected | Actual | Status | +| --- | --- | --- | --- | +| acceptance_duplicate_detection | >= 50 | | | +| acceptance_null_user_detection | >= 500 | | | +| acceptance_invalid_country_detection | >= 200 | | | +| acceptance_future_timestamp_detection | >= 150 | | | +| acceptance_late_arrival_detection | >= 300 | | | + +## Overall result + +- Validation overall status: +- Acceptance anomaly detection status: +- Log file: + +## Notes + +Document any recovery actions taken for duplicate uploads, missing files, bad +schema, partial loads, or credential issues. diff --git a/governance/README.md b/governance/README.md new file mode 100644 index 0000000..cc7f966 --- /dev/null +++ b/governance/README.md @@ -0,0 +1,51 @@ +# Atlas Governance (Sprint 7) + +This directory is the **governance source of truth** for Project Atlas. It makes +ownership, classification, retention, contracts, and lifecycle *enforceable* by +CI rather than living in prose (ADR-016). + +## Source-of-truth split + +| Asset kind | Authoritative source | +| --- | --- | +| dbt models (staging/intermediate/core/marts) | dbt `meta.governance` blocks in `dbt/atlas_dbt/models/**/*.yml` | +| Non-dbt assets (raw/ops tables, buckets, DAGs, dashboards, logs) | `governance/non_dbt_assets.yml` | +| Consolidated catalog | **generated** into `governance/generated/` — never hand-edited | + +The same metadata is never maintained in two places. dbt models are **not** +listed in `non_dbt_assets.yml`; governance CI fails if an id appears in both. + +## Files + +- `policy.yml` — the rules: required fields, controlled vocabularies, owner + rules, deprecation window, source-of-truth map. +- `classifications.yml` — PUBLIC / INTERNAL / CONFIDENTIAL / RESTRICTED meanings. +- `retention.yml` — retention classes and disposal policy. +- `consumers.yml` — internal consumer registry (used by impact + deprecation). +- `non_dbt_assets.yml` — non-dbt asset registry. +- `schemas/` — JSON Schemas for the registry files. +- `changes/` — schema-change proposal records (see the schema-evolution policy). +- `schemas/manifests/` — versioned schema baselines for compatibility checks. +- `generated/` — generated catalog (`catalog.json`, `catalog.md`). + +## Commands + +```bash +# Validate governance metadata against policy.yml (offline, no credentials): +python -m atlas.governance.catalog check + +# Regenerate the consolidated catalog after changing sources: +python -m atlas.governance.catalog generate +``` + +Governance validation also runs in CI via `gate_governance` in +`scripts/validate_ci.sh`. + +## Adding or changing an asset + +1. dbt model: edit its `meta.governance` block in the model's `.yml`. +2. Non-dbt asset: edit `governance/non_dbt_assets.yml`. +3. Run `python -m atlas.governance.catalog generate` and commit the regenerated + catalog. +4. For schema/contract changes, add a change record under `governance/changes/` + (see `docs/schema-evolution-policy-sprint7.md`). diff --git a/governance/changes/TEMPLATE.yml b/governance/changes/TEMPLATE.yml new file mode 100644 index 0000000..ffcd2df --- /dev/null +++ b/governance/changes/TEMPLATE.yml @@ -0,0 +1,28 @@ +# Atlas schema-change record template (Sprint 7, ADR-017). +# Copy to governance/changes/.yml for any non-COMPATIBLE change. +# A BREAKING change without a complete, approved record is rejected by CI. + +change_id: CHG-YYYYMMDD-short-slug +asset_id: +old_contract_version: "1.0" +new_contract_version: "2.0" +compatibility_class: BREAKING # COMPATIBLE | CONDITIONALLY_COMPATIBLE | BREAKING | PROHIBITED +reason: > + Why this change is necessary. +owner: atlas-data-eng # role id, not an email +consumer_impact: > + Which consumers (governance/consumers.yml) are affected and how. Reference the + consumer-impact report (python -m atlas.governance.impact ...). +migration_plan: > + Exact steps to migrate producers/consumers. Required for BREAKING/CONDITIONALLY. +backfill_plan: > + How historical data is backfilled or why none is required. +validation_plan: > + Tests/checks proving correctness before and after. +rollback_limitations: > + What cannot be rolled back once applied (e.g. dropped columns). +deprecation_window: > + If deprecating: start date, earliest removal date (>= 30 days), replacement. +approval_reference: > + Approval evidence: approval variable set, PR approval id, or ADR reference. + BREAKING/PROHIBITED changes require this to be non-empty. diff --git a/governance/classifications.yml b/governance/classifications.yml new file mode 100644 index 0000000..8107c7c --- /dev/null +++ b/governance/classifications.yml @@ -0,0 +1,52 @@ +# Atlas data classification definitions (Sprint 7, ADR-019). +# Referenced by asset `classification` fields. This file defines what each +# level means; assets declare which level applies. + +version: 1 + +levels: + PUBLIC: + description: > + Non-sensitive information that could be shared publicly without harm. + Atlas uses this for synthetic reference data and documentation-grade + metadata only. + access_expectation: readable by any project member + logging_restrictions: none + examples: + - dim_countries (synthetic reference dimension) + + INTERNAL: + description: > + Operational data with no personal or secret content, but not intended for + public release. The default for Atlas synthetic event data and warehouse + models. + access_expectation: project service accounts and operators + logging_restrictions: no full row payloads in logs; counts and ids only + examples: + - raw events (synthetic) + - staging/intermediate/core/mart models + - operational audit tables + + CONFIDENTIAL: + description: > + Data whose exposure would cause operational or reputational harm. + Includes anything that could carry credentials, tokens, or private + infrastructure detail. Atlas keeps such values OUT of committed artifacts. + access_expectation: least-privilege service accounts only; never committed + logging_restrictions: sanitized + truncated; never logged verbatim + examples: + - notification recipient addresses (never committed) + - error strings that may embed tokens (sanitized before audit) + + RESTRICTED: + description: > + Highest sensitivity — real personal data or regulated content. Atlas does + NOT process real personal data; this level exists so the policy is + complete and so any future real-data source is forced to declare it. + access_expectation: explicit grant + audit; not present in Atlas today + logging_restrictions: never logged; access audited + examples: [] + +# Atlas fact: the platform processes only synthetic, generator-produced data. +# No RESTRICTED assets exist. This is asserted by governance CI. +no_restricted_assets_present: true diff --git a/governance/consumers.yml b/governance/consumers.yml new file mode 100644 index 0000000..4084e52 --- /dev/null +++ b/governance/consumers.yml @@ -0,0 +1,71 @@ +# Atlas consumer registry (Sprint 7). +# Declares known consumers of governed assets, used by consumer-impact +# analysis (Phase 5) and deprecation checks (Phase 6). Consumers are internal +# to the Atlas repository — full external consumer discovery is out of scope +# (honest limitation). + +version: 1 + +consumers: + atlas_operations_dashboard: + type: dashboard + owner: atlas-observability + reads: + - atlas_ops.pipeline_runs + - atlas_ops.task_events + - atlas_ops.quality_results + - atlas_ops.monitor_evaluations + - atlas_ops.deployments + + atlas_observability_monitor: + type: dag + owner: atlas-observability + reads: + - atlas_ops.pipeline_runs + - atlas_ops.quality_results + - marts.mart_daily_event_metrics + - core.fct_events + + warehouse_reconciliation: + type: validation + owner: atlas-data-eng + reads: + - atlas_raw.events + - intermediate.int_event_classification + - core.fct_events + - marts.mart_daily_event_metrics + + alerting: + type: alert_policies + owner: atlas-observability + reads: + - atlas_ops.pipeline_runs + - atlas_ops.monitor_evaluations + + analytics_mart_readers: + type: downstream_analytics + owner: atlas-analytics + reads: + - marts.mart_daily_event_metrics + note: > + Representative downstream analytics consumer of the daily metrics mart; + any breaking change to the mart grain requires migrating this consumer. + + atlas_dbt_staging: + type: dbt_layer + owner: atlas-data-eng + reads: + - atlas_raw.events + + atlas_github_deployer: + type: service_account + owner: atlas-cicd + reads: + - gcs://atlas-deployments-example-gcp-project + + atlas_platform_operators: + type: human_operators + owner: atlas-platform + reads: + - dashboard.atlas_operations + - logging.atlas_observability_bucket diff --git a/governance/generated/catalog.json b/governance/generated/catalog.json new file mode 100644 index 0000000..5cac1c7 --- /dev/null +++ b/governance/generated/catalog.json @@ -0,0 +1,501 @@ +{ + "asset_count": 22, + "assets": [ + { + "asset_id": "atlas_ops.deployments", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard" + ], + "contract_version": "1.0", + "freshness_expectation": "per deployment", + "grain": "one row per deployment_id", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Deployment ledger (git_sha, deployment_id, stage outcomes).", + "repository_path": "src/atlas/ops/deployments.py", + "retention_class": "operational_audit", + "runbook": "docs/ci-cd-runbook-sprint4.md", + "source": "migration 003", + "technical_owner": "atlas-cicd" + }, + { + "asset_id": "atlas_ops.monitor_evaluations", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard", + "alerting" + ], + "contract_version": "1.0", + "freshness_expectation": "per monitor run (~30 min cadence)", + "grain": "one row per (evaluation_id, check_name)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Observability monitor check evaluations.", + "repository_path": "src/atlas/observability/monitor.py", + "retention_class": "operational_audit", + "runbook": "docs/observability-runbook-sprint5.md", + "source": "migration 006", + "technical_owner": "atlas-observability" + }, + { + "asset_id": "atlas_ops.pipeline_runs", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard", + "atlas_observability_monitor", + "alerting" + ], + "contract_version": "1.0", + "freshness_expectation": "per pipeline run", + "grain": "one row per pipeline_run_id", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Durable per-pipeline-run audit (status, timing, git_sha).", + "repository_path": "src/atlas/ops/audit.py", + "retention_class": "operational_audit", + "runbook": "docs/runbook-sprint3.md", + "source": "sql/create_pipeline_runs_table.sql (migration 001)", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "atlas_ops.quality_results", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard", + "atlas_observability_monitor" + ], + "contract_version": "1.0", + "freshness_expectation": "per run", + "grain": "one row per (pipeline_run_id, check_name)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Durable warehouse-quality check results per run.", + "repository_path": "src/atlas/ops/quality_results.py", + "retention_class": "operational_audit", + "runbook": "docs/observability-runbook-sprint5.md", + "source": "migration 005", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "atlas_ops.recovery_actions", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard" + ], + "contract_version": "1.0", + "freshness_expectation": "per recovery action", + "grain": "one row per recovery_id", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Durable recovery-action audit (SUCCESS requires VERIFIED).", + "repository_path": "src/atlas/ops/recovery_actions.py", + "retention_class": "operational_audit", + "runbook": "docs/recovery-runbook-sprint6.md", + "source": "migration 007", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "atlas_ops.schema_migrations", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "per migration apply", + "grain": "one row per migration_id", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Applied-migration ledger with checksums (immutability guard).", + "repository_path": "src/atlas/ops/migrations.py", + "retention_class": "operational_audit", + "runbook": "docs/ci-cd-runbook-sprint4.md", + "source": "src/atlas/ops/migrations.py", + "technical_owner": "atlas-cicd" + }, + { + "asset_id": "atlas_ops.task_events", + "asset_type": "operational_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard" + ], + "contract_version": "1.1", + "freshness_expectation": "per task attempt", + "grain": "one row per (pipeline_run_id, task_id, attempt_number, event_type)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Per-task lifecycle events with timing provenance.", + "repository_path": "src/atlas/ops/task_events.py", + "retention_class": "operational_audit", + "runbook": "docs/observability-runbook-sprint5.md", + "source": "migration 004 (+ 008 timing columns)", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "atlas_raw.events", + "asset_type": "raw_table", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation", + "atlas_dbt_staging" + ], + "contract_version": "1.0", + "freshness_expectation": "daily batch (scheduled 06:00 UTC)", + "grain": "one row per landed event record (event_id may repeat across batches)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Immutable landed synthetic events; source of truth for rebuilds.", + "repository_path": "scripts/load_events.py", + "retention_class": "raw_landing", + "runbook": "docs/runbook-sprint3.md", + "source": "scripts/generate_events.py -> GCS JSONL -> load_events.py", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "dag.atlas_batch_pipeline", + "asset_type": "dag", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "daily 06:00 UTC", + "grain": "one DAG run per (processing_date, batch_id)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "End-to-end batch ELT orchestration with audit and retries.", + "repository_path": "dags/atlas_batch_pipeline.py", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint3.md", + "source": "dags/atlas_batch_pipeline.py", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "dag.atlas_observability_monitor", + "asset_type": "dag", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "alerting", + "atlas_operations_dashboard" + ], + "contract_version": "1.0", + "freshness_expectation": "~30 min cadence", + "grain": "one DAG run per monitor cadence", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Periodic health checks feeding metrics and alerts.", + "repository_path": "dags/atlas_observability_monitor.py", + "retention_class": "operational_audit", + "runbook": "docs/observability-runbook-sprint5.md", + "source": "dags/atlas_observability_monitor.py", + "technical_owner": "atlas-observability" + }, + { + "asset_id": "dashboard.atlas_operations", + "asset_type": "dashboard", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_platform_operators" + ], + "contract_version": "1.0", + "freshness_expectation": "live", + "grain": "one dashboard", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Operational visibility across pipeline, quality, and cost.", + "repository_path": "observability/dashboards/atlas-operations.json", + "retention_class": "observability_logs", + "runbook": "docs/observability-runbook-sprint5.md", + "source": "observability/dashboards/atlas-operations.json", + "technical_owner": "atlas-observability" + }, + { + "asset_id": "dim_countries", + "asset_type": "dimension_model", + "business_owner_or_role": "atlas-platform", + "classification": "PUBLIC", + "consumers": [ + "core.fct_events" + ], + "contract_version": "1.0", + "freshness_expectation": "on seed change", + "grain": "one row per country_code", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Country reference dimension sourced from the valid_country_codes seed.", + "repository_path": "dbt/atlas_dbt/models/core/core.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/core/core.yml", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "dim_users", + "asset_type": "dimension_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "core.fct_events", + "marts.mart_daily_event_metrics" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one row per user_id", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Accepted users at user_id grain with first/last event timestamps.", + "repository_path": "dbt/atlas_dbt/models/core/core.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/core/core.yml", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "fct_events", + "asset_type": "fact_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "marts.mart_daily_event_metrics", + "warehouse_reconciliation", + "atlas_observability_monitor" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one row per event_id (global fact uniqueness \u2014 see ADR-017)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Incremental accepted event fact keyed by event_id, partitioned by event_date and clustered by event_name and country_code.", + "repository_path": "dbt/atlas_dbt/models/core/core.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/core/core.yml", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "gcs://atlas-deployments-example-gcp-project", + "asset_type": "bucket", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_github_deployer" + ], + "contract_version": "1.0", + "freshness_expectation": "per release", + "grain": "one prefix per git_sha release", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Immutable release bundles (tar.gz + checksum + manifest).", + "repository_path": "scripts/build_deployment_bundle.sh", + "retention_class": "release_evidence", + "runbook": "docs/ci-cd-runbook-sprint4.md", + "source": "scripts/build_deployment_bundle.sh", + "technical_owner": "atlas-cicd" + }, + { + "asset_id": "gcs://atlas-raw-events-example-gcp-project", + "asset_type": "bucket", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one object per (event_date, batch_id) raw JSONL", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Raw event JSONL landing bucket (run-scoped immutable paths).", + "repository_path": "scripts/upload_events.py", + "retention_class": "raw_landing", + "runbook": "docs/runbook-sprint3.md", + "source": "scripts/upload_events.py", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "int_accepted_events", + "asset_type": "intermediate_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "core.fct_events", + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one row per accepted event_id (global canonical selection)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Accepted canonical events at one row per event_id that passed all blocking checks.", + "repository_path": "dbt/atlas_dbt/models/intermediate/intermediate.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/intermediate/intermediate.yml", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "int_event_classification", + "asset_type": "intermediate_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation", + "int_accepted_events", + "int_rejected_events" + ], + "contract_version": "1.1", + "freshness_expectation": "per batch", + "grain": "one row per staged physical event record with classification flags", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Physical-row quality classification with duplicate ranking and a single terminal rejection reason per row. Precedence: missing_user_id, invalid_country_code, future_dated, duplicate_extra, accepted. Warning flags remain independent.", + "repository_path": "dbt/atlas_dbt/models/intermediate/intermediate.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/intermediate/intermediate.yml", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "int_rejected_events", + "asset_type": "intermediate_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one row per rejected physical event record", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Quarantined physical rows with a terminal blocking rejection reason.", + "repository_path": "dbt/atlas_dbt/models/intermediate/intermediate.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/intermediate/intermediate.yml", + "technical_owner": "atlas-data-eng" + }, + { + "asset_id": "logging.atlas_observability_bucket", + "asset_type": "log_resource", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "atlas_operations_dashboard", + "atlas_platform_operators" + ], + "contract_version": "1.0", + "freshness_expectation": "live (30-day retention)", + "grain": "one log bucket / linked dataset", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "registry", + "purpose": "Cloud Logging bucket + sink + view + linked dataset atlas_logs.", + "repository_path": "observability/logging/log-bucket.json", + "retention_class": "observability_logs", + "runbook": "docs/observability-runbook-sprint5.md", + "source": "observability/logging/*.json", + "technical_owner": "atlas-observability" + }, + { + "asset_id": "mart_daily_event_metrics", + "asset_type": "mart_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "analytics_mart_readers", + "atlas_observability_monitor", + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one row per (event_date, event_name, country_code, platform)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Daily event metrics at event_date, event_name, country_code, and platform grain.", + "repository_path": "dbt/atlas_dbt/models/marts/marts.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/marts/marts.yml", + "technical_owner": "atlas-analytics" + }, + { + "asset_id": "stg_events", + "asset_type": "staging_model", + "business_owner_or_role": "atlas-platform", + "classification": "INTERNAL", + "consumers": [ + "warehouse_reconciliation" + ], + "contract_version": "1.0", + "freshness_expectation": "per batch", + "grain": "one row per raw physical event record (event_id may repeat)", + "last_reviewed": "2026-07-19", + "lifecycle_status": "ACTIVE", + "origin": "dbt_meta", + "purpose": "Staged raw events at physical-row grain with normalized types, lineage metadata, a stable raw_record_hash, and corrected Sprint 2 temporal quality flags.", + "repository_path": "dbt/atlas_dbt/models/staging/staging.yml", + "retention_class": "canonical_warehouse", + "runbook": "docs/runbook-sprint2.md", + "source": "dbt/atlas_dbt/models/staging/staging.yml", + "technical_owner": "atlas-data-eng" + } + ], + "counts_by_classification": { + "INTERNAL": 21, + "PUBLIC": 1 + }, + "counts_by_type": { + "bucket": 2, + "dag": 2, + "dashboard": 1, + "dimension_model": 2, + "fact_model": 1, + "intermediate_model": 3, + "log_resource": 1, + "mart_model": 1, + "operational_table": 7, + "raw_table": 1, + "staging_model": 1 + }, + "generator": "atlas.governance.catalog", + "policy_version": 1 +} diff --git a/governance/generated/catalog.md b/governance/generated/catalog.md new file mode 100644 index 0000000..fed96a6 --- /dev/null +++ b/governance/generated/catalog.md @@ -0,0 +1,30 @@ +# Atlas Generated Asset Catalog + +> Generated by `python -m atlas.governance.catalog generate`. Do not edit by hand. + +Total assets: **22** + +| asset_id | type | owner | classification | retention | lifecycle | contract | origin | +| --- | --- | --- | --- | --- | --- | --- | --- | +| `atlas_ops.deployments` | operational_table | atlas-cicd | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `atlas_ops.monitor_evaluations` | operational_table | atlas-observability | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `atlas_ops.pipeline_runs` | operational_table | atlas-data-eng | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `atlas_ops.quality_results` | operational_table | atlas-data-eng | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `atlas_ops.recovery_actions` | operational_table | atlas-data-eng | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `atlas_ops.schema_migrations` | operational_table | atlas-cicd | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `atlas_ops.task_events` | operational_table | atlas-data-eng | INTERNAL | operational_audit | ACTIVE | 1.1 | registry | +| `atlas_raw.events` | raw_table | atlas-data-eng | INTERNAL | raw_landing | ACTIVE | 1.0 | registry | +| `dag.atlas_batch_pipeline` | dag | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | registry | +| `dag.atlas_observability_monitor` | dag | atlas-observability | INTERNAL | operational_audit | ACTIVE | 1.0 | registry | +| `dashboard.atlas_operations` | dashboard | atlas-observability | INTERNAL | observability_logs | ACTIVE | 1.0 | registry | +| `dim_countries` | dimension_model | atlas-data-eng | PUBLIC | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | +| `dim_users` | dimension_model | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | +| `fct_events` | fact_model | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | +| `gcs://atlas-deployments-example-gcp-project` | bucket | atlas-cicd | INTERNAL | release_evidence | ACTIVE | 1.0 | registry | +| `gcs://atlas-raw-events-example-gcp-project` | bucket | atlas-data-eng | INTERNAL | raw_landing | ACTIVE | 1.0 | registry | +| `int_accepted_events` | intermediate_model | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | +| `int_event_classification` | intermediate_model | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.1 | dbt_meta | +| `int_rejected_events` | intermediate_model | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | +| `logging.atlas_observability_bucket` | log_resource | atlas-observability | INTERNAL | observability_logs | ACTIVE | 1.0 | registry | +| `mart_daily_event_metrics` | mart_model | atlas-analytics | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | +| `stg_events` | staging_model | atlas-data-eng | INTERNAL | canonical_warehouse | ACTIVE | 1.0 | dbt_meta | diff --git a/governance/generated/evidence-index.json b/governance/generated/evidence-index.json new file mode 100644 index 0000000..82cd19b --- /dev/null +++ b/governance/generated/evidence-index.json @@ -0,0 +1,42 @@ +{ + "version": 1, + "generated_by": "Sprint 8 Phase 8 (hand-authored, validated by atlas.reference.validate)", + "last_verified_commit": "3f986aa", + "legend": { + "live_or_static": ["LIVE", "STATIC"], + "status": ["PROVEN_LIVE", "PROVEN_STATIC", "PROVEN_TEST", "PLANNED", "BLOCKED", "NOT_APPLICABLE"], + "evidence_type": ["TEST", "CI_RUN", "LIVE_DEPLOYMENT", "LIVE_QUERY", "DRY_RUN", "INCIDENT", "RECOVERY", "DOCUMENTED_DECISION", "CONFIGURATION", "CODE_INSPECTION"] + }, + "claims": [ + {"claim_id": "CLM-01-generation", "claim": "Deterministic 50k-event generation with seeded anomalies", "scope": "ingestion", "evidence_type": "TEST", "evidence_path": "docs/validation-report-sprint1.md", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": "synthetic scale"}, + {"claim_id": "CLM-02-immutable-ingest", "claim": "Immutable run-scoped GCS landing", "scope": "ingestion", "evidence_type": "LIVE_DEPLOYMENT", "evidence_path": "docs/validation-report-sprint1.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": "synthetic scale"}, + {"claim_id": "CLM-03-bq-load", "claim": "Partitioned/clustered BigQuery raw load", "scope": "warehouse", "evidence_type": "LIVE_QUERY", "evidence_path": "docs/validation-report-sprint1.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": "synthetic scale"}, + {"claim_id": "CLM-04-dbt-transform", "claim": "Governed dbt staging->classification->core->marts", "scope": "warehouse", "evidence_type": "TEST", "evidence_path": "docs/validation-report-sprint2.md", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-05-quality-gate", "claim": "Quality failure prevents publication", "scope": "warehouse", "evidence_type": "INCIDENT", "evidence_path": "docs/game-day-results-sprint6.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-06-orchestration", "claim": "Airflow orchestration with stable batch identity", "scope": "orchestration", "evidence_type": "TEST", "evidence_path": "docs/validation-report-sprint3.md", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-07-retries-backfills", "claim": "Retries, reruns, and deterministic backfills", "scope": "orchestration", "evidence_type": "TEST", "evidence_path": "docs/validation-report-sprint3.md", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-08-ci", "claim": "Credentialless PR CI with 21 gates", "scope": "delivery", "evidence_type": "CI_RUN", "evidence_path": "scripts/validate_ci.sh", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-09-wif", "claim": "Keyless WIF GitHub->GCP authentication", "scope": "security", "evidence_type": "LIVE_DEPLOYMENT", "evidence_path": "docs/validation-report-sprint4.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-10-immutable-release", "claim": "Immutable release bundles", "scope": "delivery", "evidence_type": "LIVE_DEPLOYMENT", "evidence_path": "docs/deployment-catalog-sprint4.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-11-rollback", "claim": "Rollback with schema-compatibility check", "scope": "delivery", "evidence_type": "DOCUMENTED_DECISION", "evidence_path": "docs/adr/ADR-015-schema-compatibility-and-recovery.md", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-12-logging", "claim": "Structured logging with correlation ids", "scope": "observability", "evidence_type": "CONFIGURATION", "evidence_path": "docs/adr/ADR-011-atlas-observability-model.md", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-13-monitoring", "claim": "Metrics, alerts, and dashboard", "scope": "observability", "evidence_type": "CONFIGURATION", "evidence_path": "docs/alert-catalog-sprint5.md", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-14-alerting-live", "claim": "Alerting delivered to a verified recipient in a drill", "scope": "observability", "evidence_type": "INCIDENT", "evidence_path": "docs/incident-report-sprint5.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-15-incident-response", "claim": "Incident detection and diagnosis", "scope": "operations", "evidence_type": "INCIDENT", "evidence_path": "docs/incident-report-INC-S6-001-batch-contamination.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-16-recovery", "claim": "Verified targeted recovery (QUARANTINE_BATCH)", "scope": "operations", "evidence_type": "RECOVERY", "evidence_path": "docs/validation-report-sprint6.md", "live_or_static": "LIVE", "status": "PROVEN_LIVE", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-17-schema-evolution", "claim": "Automated schema compatibility classification", "scope": "governance", "evidence_type": "TEST", "evidence_path": "tests/unit/test_schema_check.py", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-18-migration-immutable", "claim": "Applied migration checksums are immutable", "scope": "governance", "evidence_type": "TEST", "evidence_path": "tests/unit/test_governance_demos.py", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-19-lineage", "claim": "Repository-artifact lineage graph", "scope": "governance", "evidence_type": "CONFIGURATION", "evidence_path": "governance/generated/lineage.json", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": "internal Atlas assets only"}, + {"claim_id": "CLM-20-impact", "claim": "Consumer-impact analysis identifies downstream effects", "scope": "governance", "evidence_type": "TEST", "evidence_path": "tests/unit/test_lineage_impact.py", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-21-governance-sot", "claim": "Governance metadata has one enforced source of truth", "scope": "governance", "evidence_type": "CONFIGURATION", "evidence_path": "governance/generated/catalog.json", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-22-security-scan", "claim": "No committed secrets; managed-IAM/data-exposure scanners", "scope": "security", "evidence_type": "TEST", "evidence_path": "tests/unit/test_security_policy.py", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-23-retention", "claim": "Classification & retention validated; permanent evidence protected", "scope": "governance", "evidence_type": "TEST", "evidence_path": "tests/unit/test_retention.py", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-24-performance-baseline", "claim": "BigQuery performance dry-run baseline (all queries << 1 GiB)", "scope": "performance", "evidence_type": "DRY_RUN", "evidence_path": "observability/performance/results/baseline-dryrun.json", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": "synthetic scale; billed suite blocked"}, + {"claim_id": "CLM-25-cost-block", "claim": "Over-limit/unbounded query blocked before spend ($0)", "scope": "cost", "evidence_type": "DRY_RUN", "evidence_path": "docs/evidence-sprint7/cost-guard-block.txt", "live_or_static": "STATIC", "status": "PROVEN_STATIC", "verification_commit": "3f986aa", "limitations": null}, + {"claim_id": "CLM-26-clean-clone", "claim": "Clean-clone reaches green static validation without hidden help", "scope": "reproducibility", "evidence_type": "TEST", "evidence_path": "docs/evidence-sprint8/clean-clone-results.md", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "e538e99", "limitations": "attempt 2 from a fresh directory after a documentation fix; dbt/airflow gates SKIP without optional tools"}, + {"claim_id": "CLM-27-handoff", "claim": "Independent handoff test scored 29/30 against a rubric", "scope": "reproducibility", "evidence_type": "TEST", "evidence_path": "docs/evidence-sprint8/independent-handoff-results.md", "live_or_static": "STATIC", "status": "PROVEN_TEST", "verification_commit": "d90b3ad", "limitations": "single independent-agent tester context; one validation-friction point fixed at root cause"}, + {"claim_id": "CLM-28-iam-reduction", "claim": "Live IAM reduction + positive/negative tests", "scope": "security", "evidence_type": "DOCUMENTED_DECISION", "evidence_path": "docs/iam-review-sprint7.md", "live_or_static": "STATIC", "status": "BLOCKED", "verification_commit": "3f986aa", "limitations": "requires ATLAS_APPROVE_IAM; not executed"}, + {"claim_id": "CLM-29-perf-billed", "claim": "Executed/billed BigQuery performance suite", "scope": "performance", "evidence_type": "DOCUMENTED_DECISION", "evidence_path": "docs/performance-review-sprint7.md", "live_or_static": "STATIC", "status": "BLOCKED", "verification_commit": "3f986aa", "limitations": "requires ATLAS_APPROVE_PERFORMANCE_TESTS; not executed"}, + {"claim_id": "CLM-30-retention-live", "claim": "Live retention/expiration application", "scope": "cost", "evidence_type": "DOCUMENTED_DECISION", "evidence_path": "docs/retention-policy-sprint7.md", "live_or_static": "STATIC", "status": "BLOCKED", "verification_commit": "3f986aa", "limitations": "requires ATLAS_APPROVE_RETENTION_MUTATION; not executed"} + ] +} diff --git a/governance/generated/lineage.json b/governance/generated/lineage.json new file mode 100644 index 0000000..8711169 --- /dev/null +++ b/governance/generated/lineage.json @@ -0,0 +1,250 @@ +{ + "edge_count": 29, + "nodes": [ + { + "downstream": [], + "id": "alerting", + "type": "consumer:alert_policies", + "upstream": [ + "atlas_ops.monitor_evaluations", + "atlas_ops.pipeline_runs" + ] + }, + { + "downstream": [], + "id": "analytics_mart_readers", + "type": "consumer:downstream_analytics", + "upstream": [ + "mart_daily_event_metrics" + ] + }, + { + "downstream": [], + "id": "atlas_dbt_staging", + "type": "consumer:dbt_layer", + "upstream": [ + "atlas_raw.events" + ] + }, + { + "downstream": [], + "id": "atlas_github_deployer", + "type": "consumer:service_account", + "upstream": [ + "gcs://atlas-deployments-example-gcp-project" + ] + }, + { + "downstream": [], + "id": "atlas_observability_monitor", + "type": "consumer:dag", + "upstream": [ + "atlas_ops.pipeline_runs", + "atlas_ops.quality_results", + "fct_events", + "mart_daily_event_metrics" + ] + }, + { + "downstream": [], + "id": "atlas_operations_dashboard", + "type": "consumer:dashboard", + "upstream": [ + "atlas_ops.deployments", + "atlas_ops.monitor_evaluations", + "atlas_ops.pipeline_runs", + "atlas_ops.quality_results", + "atlas_ops.task_events" + ] + }, + { + "downstream": [ + "atlas_operations_dashboard" + ], + "id": "atlas_ops.deployments", + "type": "asset", + "upstream": [] + }, + { + "downstream": [ + "alerting", + "atlas_operations_dashboard" + ], + "id": "atlas_ops.monitor_evaluations", + "type": "asset", + "upstream": [] + }, + { + "downstream": [ + "alerting", + "atlas_observability_monitor", + "atlas_operations_dashboard" + ], + "id": "atlas_ops.pipeline_runs", + "type": "asset", + "upstream": [] + }, + { + "downstream": [ + "atlas_observability_monitor", + "atlas_operations_dashboard" + ], + "id": "atlas_ops.quality_results", + "type": "asset", + "upstream": [] + }, + { + "downstream": [ + "atlas_operations_dashboard" + ], + "id": "atlas_ops.task_events", + "type": "asset", + "upstream": [] + }, + { + "downstream": [], + "id": "atlas_platform_operators", + "type": "consumer:human_operators", + "upstream": [ + "dashboard.atlas_operations", + "logging.atlas_observability_bucket" + ] + }, + { + "downstream": [ + "atlas_dbt_staging", + "stg_events", + "warehouse_reconciliation" + ], + "id": "atlas_raw.events", + "type": "source", + "upstream": [] + }, + { + "downstream": [ + "atlas_platform_operators" + ], + "id": "dashboard.atlas_operations", + "type": "asset", + "upstream": [] + }, + { + "downstream": [], + "id": "dim_countries", + "type": "core_model", + "upstream": [ + "valid_country_codes" + ] + }, + { + "downstream": [], + "id": "dim_users", + "type": "core_model", + "upstream": [ + "int_accepted_events" + ] + }, + { + "downstream": [ + "atlas_observability_monitor", + "mart_daily_event_metrics", + "warehouse_reconciliation" + ], + "id": "fct_events", + "type": "core_model", + "upstream": [ + "int_accepted_events" + ] + }, + { + "downstream": [ + "atlas_github_deployer" + ], + "id": "gcs://atlas-deployments-example-gcp-project", + "type": "asset", + "upstream": [] + }, + { + "downstream": [ + "dim_users", + "fct_events" + ], + "id": "int_accepted_events", + "type": "model_or_seed", + "upstream": [ + "int_event_classification" + ] + }, + { + "downstream": [ + "int_accepted_events", + "int_rejected_events", + "warehouse_reconciliation" + ], + "id": "int_event_classification", + "type": "model_or_seed", + "upstream": [ + "stg_events", + "valid_country_codes" + ] + }, + { + "downstream": [], + "id": "int_rejected_events", + "type": "intermediate_model", + "upstream": [ + "int_event_classification" + ] + }, + { + "downstream": [ + "atlas_platform_operators" + ], + "id": "logging.atlas_observability_bucket", + "type": "asset", + "upstream": [] + }, + { + "downstream": [ + "analytics_mart_readers", + "atlas_observability_monitor", + "warehouse_reconciliation" + ], + "id": "mart_daily_event_metrics", + "type": "marts_model", + "upstream": [ + "fct_events" + ] + }, + { + "downstream": [ + "int_event_classification" + ], + "id": "stg_events", + "type": "model_or_seed", + "upstream": [ + "atlas_raw.events" + ] + }, + { + "downstream": [ + "dim_countries", + "int_event_classification" + ], + "id": "valid_country_codes", + "type": "model_or_seed", + "upstream": [] + }, + { + "downstream": [], + "id": "warehouse_reconciliation", + "type": "consumer:validation", + "upstream": [ + "atlas_raw.events", + "fct_events", + "int_event_classification", + "mart_daily_event_metrics" + ] + } + ] +} diff --git a/governance/non_dbt_assets.yml b/governance/non_dbt_assets.yml new file mode 100644 index 0000000..5856320 --- /dev/null +++ b/governance/non_dbt_assets.yml @@ -0,0 +1,254 @@ +# Atlas non-dbt asset registry (Sprint 7, ADR-016). +# Authoritative governance metadata for assets NOT owned by dbt: raw/operational +# tables, buckets, DAGs, dashboards, and log resources. dbt models are governed +# by their dbt `meta` blocks and must NOT be duplicated here. +# +# Every asset must declare all policy.required_fields. `last_reviewed` is an +# ISO date. `contract_version` uses semver-like "major.minor". + +version: 1 + +assets: + # ---- Raw landing ---- + - asset_id: atlas_raw.events + asset_type: raw_table + purpose: Immutable landed synthetic events; source of truth for rebuilds. + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per landed event record (event_id may repeat across batches) + source: scripts/generate_events.py -> GCS JSONL -> load_events.py + consumers: [warehouse_reconciliation, atlas_dbt_staging] + classification: INTERNAL + retention_class: raw_landing + freshness_expectation: daily batch (scheduled 06:00 UTC) + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: scripts/load_events.py + runbook: docs/runbook-sprint3.md + last_reviewed: "2026-07-19" + + # ---- Operational audit tables ---- + - asset_id: atlas_ops.pipeline_runs + asset_type: operational_table + purpose: Durable per-pipeline-run audit (status, timing, git_sha). + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per pipeline_run_id + source: sql/create_pipeline_runs_table.sql (migration 001) + consumers: [atlas_operations_dashboard, atlas_observability_monitor, alerting] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per pipeline run + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: src/atlas/ops/audit.py + runbook: docs/runbook-sprint3.md + last_reviewed: "2026-07-19" + + - asset_id: atlas_ops.task_events + asset_type: operational_table + purpose: Per-task lifecycle events with timing provenance. + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per (pipeline_run_id, task_id, attempt_number, event_type) + source: migration 004 (+ 008 timing columns) + consumers: [atlas_operations_dashboard] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per task attempt + contract_version: "1.1" + lifecycle_status: ACTIVE + repository_path: src/atlas/ops/task_events.py + runbook: docs/observability-runbook-sprint5.md + last_reviewed: "2026-07-19" + + - asset_id: atlas_ops.quality_results + asset_type: operational_table + purpose: Durable warehouse-quality check results per run. + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per (pipeline_run_id, check_name) + source: migration 005 + consumers: [atlas_operations_dashboard, atlas_observability_monitor] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per run + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: src/atlas/ops/quality_results.py + runbook: docs/observability-runbook-sprint5.md + last_reviewed: "2026-07-19" + + - asset_id: atlas_ops.monitor_evaluations + asset_type: operational_table + purpose: Observability monitor check evaluations. + technical_owner: atlas-observability + business_owner_or_role: atlas-platform + grain: one row per (evaluation_id, check_name) + source: migration 006 + consumers: [atlas_operations_dashboard, alerting] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per monitor run (~30 min cadence) + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: src/atlas/observability/monitor.py + runbook: docs/observability-runbook-sprint5.md + last_reviewed: "2026-07-19" + + - asset_id: atlas_ops.deployments + asset_type: operational_table + purpose: Deployment ledger (git_sha, deployment_id, stage outcomes). + technical_owner: atlas-cicd + business_owner_or_role: atlas-platform + grain: one row per deployment_id + source: migration 003 + consumers: [atlas_operations_dashboard] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per deployment + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: src/atlas/ops/deployments.py + runbook: docs/ci-cd-runbook-sprint4.md + last_reviewed: "2026-07-19" + + - asset_id: atlas_ops.schema_migrations + asset_type: operational_table + purpose: Applied-migration ledger with checksums (immutability guard). + technical_owner: atlas-cicd + business_owner_or_role: atlas-platform + grain: one row per migration_id + source: src/atlas/ops/migrations.py + consumers: [warehouse_reconciliation] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per migration apply + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: src/atlas/ops/migrations.py + runbook: docs/ci-cd-runbook-sprint4.md + last_reviewed: "2026-07-19" + + - asset_id: atlas_ops.recovery_actions + asset_type: operational_table + purpose: Durable recovery-action audit (SUCCESS requires VERIFIED). + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one row per recovery_id + source: migration 007 + consumers: [atlas_operations_dashboard] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: per recovery action + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: src/atlas/ops/recovery_actions.py + runbook: docs/recovery-runbook-sprint6.md + last_reviewed: "2026-07-19" + + # ---- Buckets ---- + - asset_id: gcs://atlas-raw-events-example-gcp-project + asset_type: bucket + purpose: Raw event JSONL landing bucket (run-scoped immutable paths). + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one object per (event_date, batch_id) raw JSONL + source: scripts/upload_events.py + consumers: [warehouse_reconciliation] + classification: INTERNAL + retention_class: raw_landing + freshness_expectation: per batch + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: scripts/upload_events.py + runbook: docs/runbook-sprint3.md + last_reviewed: "2026-07-19" + + - asset_id: gcs://atlas-deployments-example-gcp-project + asset_type: bucket + purpose: Immutable release bundles (tar.gz + checksum + manifest). + technical_owner: atlas-cicd + business_owner_or_role: atlas-platform + grain: one prefix per git_sha release + source: scripts/build_deployment_bundle.sh + consumers: [atlas_github_deployer] + classification: INTERNAL + retention_class: release_evidence + freshness_expectation: per release + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: scripts/build_deployment_bundle.sh + runbook: docs/ci-cd-runbook-sprint4.md + last_reviewed: "2026-07-19" + + # ---- DAGs ---- + - asset_id: dag.atlas_batch_pipeline + asset_type: dag + purpose: End-to-end batch ELT orchestration with audit and retries. + technical_owner: atlas-data-eng + business_owner_or_role: atlas-platform + grain: one DAG run per (processing_date, batch_id) + source: dags/atlas_batch_pipeline.py + consumers: [warehouse_reconciliation] + classification: INTERNAL + retention_class: canonical_warehouse + freshness_expectation: daily 06:00 UTC + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: dags/atlas_batch_pipeline.py + runbook: docs/runbook-sprint3.md + last_reviewed: "2026-07-19" + + - asset_id: dag.atlas_observability_monitor + asset_type: dag + purpose: Periodic health checks feeding metrics and alerts. + technical_owner: atlas-observability + business_owner_or_role: atlas-platform + grain: one DAG run per monitor cadence + source: dags/atlas_observability_monitor.py + consumers: [alerting, atlas_operations_dashboard] + classification: INTERNAL + retention_class: operational_audit + freshness_expectation: ~30 min cadence + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: dags/atlas_observability_monitor.py + runbook: docs/observability-runbook-sprint5.md + last_reviewed: "2026-07-19" + + # ---- Dashboard ---- + - asset_id: dashboard.atlas_operations + asset_type: dashboard + purpose: Operational visibility across pipeline, quality, and cost. + technical_owner: atlas-observability + business_owner_or_role: atlas-platform + grain: one dashboard + source: observability/dashboards/atlas-operations.json + consumers: [atlas_platform_operators] + classification: INTERNAL + retention_class: observability_logs + freshness_expectation: live + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: observability/dashboards/atlas-operations.json + runbook: docs/observability-runbook-sprint5.md + last_reviewed: "2026-07-19" + + # ---- Log resources ---- + - asset_id: logging.atlas_observability_bucket + asset_type: log_resource + purpose: Cloud Logging bucket + sink + view + linked dataset atlas_logs. + technical_owner: atlas-observability + business_owner_or_role: atlas-platform + grain: one log bucket / linked dataset + source: observability/logging/*.json + consumers: [atlas_operations_dashboard, atlas_platform_operators] + classification: INTERNAL + retention_class: observability_logs + freshness_expectation: live (30-day retention) + contract_version: "1.0" + lifecycle_status: ACTIVE + repository_path: observability/logging/log-bucket.json + runbook: docs/observability-runbook-sprint5.md + last_reviewed: "2026-07-19" diff --git a/governance/policy.yml b/governance/policy.yml new file mode 100644 index 0000000..1b1e8ce --- /dev/null +++ b/governance/policy.yml @@ -0,0 +1,91 @@ +# Atlas governance policy (Sprint 7, ADR-016). +# +# This file declares the *rules* governance validation enforces. It is the +# single place that defines required fields, controlled vocabularies, and the +# source-of-truth split between dbt metadata and the non-dbt asset registry. +# It is NOT an asset registry itself. + +version: 1 + +# Source-of-truth split (ADR-016). dbt models are governed by their dbt `meta` +# blocks; everything else is governed by governance/non_dbt_assets.yml. The +# consolidated catalog is generated from both and must never be hand-edited. +source_of_truth: + dbt_models: dbt_meta # dbt/atlas_dbt/models/**/*.yml -> meta: + non_dbt_assets: registry # governance/non_dbt_assets.yml + generated_catalog: governance/generated/catalog.json + +# Required governance fields for every major asset (dbt or non-dbt). +required_fields: + - asset_id + - asset_type + - purpose + - technical_owner + - business_owner_or_role + - grain + - source + - consumers + - classification + - retention_class + - freshness_expectation + - contract_version + - lifecycle_status + - repository_path + - runbook + - last_reviewed + +# Controlled vocabularies. +asset_types: + - raw_table + - staging_model + - intermediate_model + - dimension_model + - fact_model + - mart_model + - operational_table + - bucket + - dag + - dashboard + - log_resource + +lifecycle_statuses: + - ACTIVE + - DEPRECATED + - REMOVAL_SCHEDULED + - REMOVED + +classifications: + - PUBLIC + - INTERNAL + - CONFIDENTIAL + - RESTRICTED + +# Retention classes are defined in governance/retention.yml; the ids here are +# the allowed set that assets may reference. +retention_classes: + - canonical_warehouse + - raw_landing + - operational_audit + - observability_logs + - release_evidence + - temporary_integration + - test_fixture + +# Schema-change compatibility classes (ADR-017). +compatibility_classes: + - COMPATIBLE + - CONDITIONALLY_COMPATIBLE + - BREAKING + - PROHIBITED + +# Deprecation policy. +deprecation: + minimum_window_days: 30 + require_replacement: true + require_change_record: true + +# Owners must be role identifiers, not personal email addresses (avoids +# committing personal data and keeps ownership durable across staffing). +owner_rules: + disallow_email_addresses: true + allowed_owner_pattern: "^[a-z0-9][a-z0-9-]*(/[a-z0-9-]+)?$" diff --git a/governance/retention.yml b/governance/retention.yml new file mode 100644 index 0000000..998da3d --- /dev/null +++ b/governance/retention.yml @@ -0,0 +1,61 @@ +# Atlas retention classes (Sprint 7, ADR-019). +# Each asset declares a retention_class; this file defines the disposal policy +# for that class. Live lifecycle/expiration changes require +# ATLAS_APPROVE_RETENTION_MUTATION=true. + +version: 1 + +classes: + canonical_warehouse: + description: Governed warehouse models (staging, intermediate, core, marts). + retention: indefinite + disposal: rebuilt deterministically from raw; never auto-expired + expiration_days: null + is_permanent_evidence: false + + raw_landing: + description: Raw landed events (immutable source of truth for rebuilds). + retention: indefinite + disposal: retained; partition-level management only under approval + expiration_days: null + is_permanent_evidence: false + + operational_audit: + description: > + Durable operational history (pipeline_runs, task_events, quality_results, + monitor_evaluations, deployments, schema_migrations, recovery_actions). + retention: indefinite + disposal: never auto-expired — this IS the operational evidence + expiration_days: null + is_permanent_evidence: true + + observability_logs: + description: Cloud Logging bucket + linked dataset for structured telemetry. + retention: 30 days + disposal: bucket retention policy (Sprint 5) + expiration_days: 30 + is_permanent_evidence: false + + release_evidence: + description: Immutable release bundles in the deployment bucket. + retention: keep all validated releases + disposal: lifecycle review only; validated releases retained + expiration_days: null + is_permanent_evidence: true + + temporary_integration: + description: CI/integration datasets and scratch datasets. + retention: short-lived + disposal: dataset default table expiration + expiration_days: 1 + is_permanent_evidence: false + + test_fixture: + description: Local generated artifacts and test fixtures. + retention: ephemeral + disposal: not persisted to cloud; safe to delete + expiration_days: 0 + is_permanent_evidence: false + +# CI invariant: any class with is_permanent_evidence=true MUST have +# expiration_days == null (permanent evidence can never carry an expiration). diff --git a/governance/schemas/manifests/baseline.json b/governance/schemas/manifests/baseline.json new file mode 100644 index 0000000..3419dba --- /dev/null +++ b/governance/schemas/manifests/baseline.json @@ -0,0 +1,265 @@ +{ + "assets": { + "dim_countries": { + "contract_version": "1.0", + "fields": { + "country_code": { + "nullable": false, + "type": "unknown" + }, + "is_active": { + "nullable": false, + "type": "unknown" + } + }, + "grain": "one row per country_code" + }, + "dim_users": { + "contract_version": "1.0", + "fields": { + "first_event_at": { + "nullable": false, + "type": "unknown" + }, + "last_event_at": { + "nullable": false, + "type": "unknown" + }, + "user_id": { + "nullable": false, + "type": "unknown" + } + }, + "grain": "one row per user_id" + }, + "fct_events": { + "contract_version": "1.0", + "event_identity": [ + "event_id" + ], + "fields": { + "country_code": { + "nullable": false, + "type": "unknown" + }, + "event_date": { + "nullable": false, + "type": "unknown" + }, + "event_id": { + "nullable": false, + "type": "unknown" + }, + "event_name": { + "nullable": false, + "type": "unknown" + }, + "platform": { + "accepted_values": [ + "ios", + "android", + "web" + ], + "nullable": false, + "type": "unknown" + }, + "user_id": { + "nullable": false, + "type": "unknown" + } + }, + "grain": "one row per event_id (global fact uniqueness \u2014 see ADR-017)", + "partition_field": "event_date" + }, + "int_accepted_events": { + "contract_version": "1.0", + "fields": { + "event_id": { + "nullable": false, + "type": "unknown" + }, + "rejection_reason": { + "accepted_values": [ + "accepted" + ], + "nullable": true, + "type": "unknown" + } + }, + "grain": "one row per accepted event_id (global canonical selection)" + }, + "int_event_classification": { + "contract_version": "1.1", + "fields": { + "duplicate_rank": { + "nullable": false, + "type": "unknown" + }, + "duplicate_scope": { + "accepted_values": [ + "none", + "within_batch", + "cross_batch_replay" + ], + "nullable": false, + "type": "unknown" + }, + "event_id": { + "nullable": false, + "type": "unknown" + }, + "is_accepted": { + "nullable": false, + "type": "unknown" + }, + "is_duplicate_extra": { + "nullable": false, + "type": "unknown" + }, + "is_within_batch_duplicate": { + "nullable": false, + "type": "unknown" + }, + "rejection_reason": { + "accepted_values": [ + "accepted", + "missing_user_id", + "invalid_country_code", + "future_dated", + "duplicate_extra" + ], + "nullable": false, + "type": "unknown" + }, + "within_batch_duplicate_rank": { + "nullable": false, + "type": "unknown" + } + }, + "grain": "one row per staged physical event record with classification flags" + }, + "int_rejected_events": { + "contract_version": "1.0", + "fields": { + "rejection_reason": { + "nullable": false, + "type": "unknown" + } + }, + "grain": "one row per rejected physical event record" + }, + "mart_daily_event_metrics": { + "contract_version": "1.0", + "fields": { + "country_code": { + "nullable": false, + "type": "unknown" + }, + "event_count": { + "nullable": false, + "type": "unknown" + }, + "event_date": { + "nullable": false, + "type": "unknown" + }, + "event_name": { + "nullable": false, + "type": "unknown" + }, + "platform": { + "nullable": false, + "type": "unknown" + } + }, + "grain": "one row per (event_date, event_name, country_code, platform)" + }, + "stg_events": { + "contract_version": "1.0", + "fields": { + "app_version": { + "nullable": true, + "type": "string" + }, + "batch_id": { + "nullable": true, + "type": "string" + }, + "country_code": { + "nullable": true, + "type": "string" + }, + "event_date": { + "nullable": false, + "type": "date" + }, + "event_id": { + "nullable": false, + "type": "string" + }, + "event_name": { + "nullable": false, + "type": "string" + }, + "event_timestamp": { + "nullable": false, + "type": "timestamp" + }, + "event_timestamp_date": { + "nullable": false, + "type": "date" + }, + "has_event_date_timestamp_mismatch": { + "nullable": false, + "type": "boolean" + }, + "ingested_at": { + "nullable": false, + "type": "timestamp" + }, + "ingested_date": { + "nullable": false, + "type": "date" + }, + "is_backdated_event_date": { + "nullable": false, + "type": "boolean" + }, + "is_event_time_late_arriving": { + "nullable": false, + "type": "boolean" + }, + "is_future_dated": { + "nullable": false, + "type": "boolean" + }, + "pipeline_run_id": { + "nullable": false, + "type": "string" + }, + "platform": { + "nullable": true, + "type": "string" + }, + "processing_date": { + "nullable": true, + "type": "date" + }, + "raw_record_hash": { + "nullable": false, + "type": "int64" + }, + "source_file": { + "nullable": false, + "type": "string" + }, + "user_id": { + "nullable": true, + "type": "string" + } + }, + "grain": "one row per raw physical event record (event_id may repeat)" + } + }, + "version": 1 +} diff --git a/governance/schemas/non_dbt_assets.schema.json b/governance/schemas/non_dbt_assets.schema.json new file mode 100644 index 0000000..01b1c3a --- /dev/null +++ b/governance/schemas/non_dbt_assets.schema.json @@ -0,0 +1,88 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "title": "Atlas non-dbt asset registry", + "type": "object", + "required": ["version", "assets"], + "properties": { + "version": { "type": "integer" }, + "assets": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "asset_id", + "asset_type", + "purpose", + "technical_owner", + "business_owner_or_role", + "grain", + "source", + "consumers", + "classification", + "retention_class", + "freshness_expectation", + "contract_version", + "lifecycle_status", + "repository_path", + "runbook", + "last_reviewed" + ], + "properties": { + "asset_id": { "type": "string", "minLength": 1 }, + "asset_type": { + "enum": [ + "raw_table", + "operational_table", + "bucket", + "dag", + "dashboard", + "log_resource" + ] + }, + "purpose": { "type": "string", "minLength": 1 }, + "technical_owner": { "type": "string", "pattern": "^[a-z0-9][a-z0-9-]*(/[a-z0-9-]+)?$" }, + "business_owner_or_role": { "type": "string", "minLength": 1 }, + "grain": { "type": "string", "minLength": 1 }, + "source": { "type": "string", "minLength": 1 }, + "consumers": { "type": "array", "items": { "type": "string" } }, + "classification": { "enum": ["PUBLIC", "INTERNAL", "CONFIDENTIAL", "RESTRICTED"] }, + "retention_class": { + "enum": [ + "canonical_warehouse", + "raw_landing", + "operational_audit", + "observability_logs", + "release_evidence", + "temporary_integration", + "test_fixture" + ] + }, + "freshness_expectation": { "type": "string", "minLength": 1 }, + "contract_version": { "type": "string", "pattern": "^[0-9]+\\.[0-9]+$" }, + "lifecycle_status": { + "enum": ["ACTIVE", "DEPRECATED", "REMOVAL_SCHEDULED", "REMOVED"] + }, + "repository_path": { "type": "string", "minLength": 1 }, + "runbook": { "type": "string", "minLength": 1 }, + "last_reviewed": { "type": "string", "pattern": "^[0-9]{4}-[0-9]{2}-[0-9]{2}$" }, + "deprecation": { + "type": "object", + "properties": { + "replacement": { "type": "string" }, + "owner": { "type": "string" }, + "announcement_date": { "type": "string" }, + "deprecation_start": { "type": "string" }, + "earliest_removal_date": { "type": "string" }, + "migration_instructions": { "type": "string" }, + "validation_period": { "type": "string" }, + "removal_approval": { "type": "string" }, + "rollback_limitations": { "type": "string" }, + "change_record": { "type": "string" } + } + } + } + } + } + } +} diff --git a/governance/unresolved_risks.yml b/governance/unresolved_risks.yml new file mode 100644 index 0000000..838a3aa --- /dev/null +++ b/governance/unresolved_risks.yml @@ -0,0 +1,218 @@ +# Atlas unresolved-risk register (machine-readable, Sprint 8 Phase 13). +# Human-readable: docs/reference-architecture/unresolved-risks.md +# Statuses: OPEN | ACCEPTED | MITIGATED | BLOCKED | DEFERRED | CLOSED +version: 1 +last_verified_commit: "3f986aa" +risks: + - risk_id: RISK-01 + title: Sprint 7 live IAM reduction not executed + category: security + description: >- + atlas-github-integration holds project-level roles/bigquery.dataEditor, + broader than required. Scoped reduction planned but not applied. + evidence: docs/iam-review-sprint7.md + likelihood: low + impact: medium + severity: MEDIUM + current_control: gate_security_policy blocks prohibited roles/keys; WIF keyless + remaining_exposure: excess dataEditor scope on one integration SA + owner: platform-operator + next_action: apply scoped reduction and run positive/negative tests + approval_required: ATLAS_APPROVE_IAM + target_phase: sprint8-phase13-or-later + status: BLOCKED + - risk_id: RISK-02 + title: Sprint 7 positive and negative IAM tests not executed + category: security + description: Live authorized-success and unauthorized-denied tests not run. + evidence: docs/iam-review-sprint7.md + likelihood: low + impact: medium + severity: MEDIUM + current_control: documented plan; static policy scanners + remaining_exposure: least-privilege not proven at permission level live + owner: platform-operator + next_action: run harmless denied op + authorized op, record both + approval_required: ATLAS_APPROVE_IAM + target_phase: sprint8-phase13-or-later + status: BLOCKED + - risk_id: RISK-03 + title: Billed BigQuery performance suite not executed + category: performance + description: Only dry-run baseline captured; executed/billed suite not run. + evidence: docs/performance-review-sprint7.md + likelihood: low + impact: low + severity: LOW + current_control: dry-run baseline; per-query and suite byte ceilings + remaining_exposure: no billed slot/latency numbers + owner: platform-operator + next_action: execute once under byte ceiling; record billed bytes/slot/correctness + approval_required: ATLAS_APPROVE_PERFORMANCE_TESTS + target_phase: sprint8-phase13-or-later + status: BLOCKED + - risk_id: RISK-04 + title: Live retention and expiration application not executed + category: cost + description: Disposal is a validated dry-run plan; TTLs not applied live. + evidence: docs/retention-policy-sprint7.md + likelihood: low + impact: low + severity: LOW + current_control: retention.validate_retention_config; permanent-evidence guard + remaining_exposure: temporary resources rely on manual/plan disposal + owner: platform-operator + next_action: apply TTL to temporary resources only; verify; protect permanent + approval_required: ATLAS_APPROVE_RETENTION_MUTATION + target_phase: sprint8-phase13-or-later + status: BLOCKED + - risk_id: RISK-05 + title: Composer customer-project task-log limitation + category: observability + description: Composer task logs not always fully available in customer project. + evidence: docs/reference-architecture/observability-model.md + likelihood: medium + impact: low + severity: LOW + current_control: direct Cloud Logging export at environment create time + remaining_exposure: minor; mitigated + owner: platform-operator + next_action: none required + approval_required: none + target_phase: n/a + status: MITIGATED + - risk_id: RISK-06 + title: Synthetic 50k-row scale + category: architecture + description: All performance/cost/reliability numbers are at synthetic scale. + evidence: docs/reference-architecture/cost-and-lifecycle-model.md + likelihood: high + impact: medium + severity: MEDIUM + current_control: honest limitation stated everywhere + remaining_exposure: production-scale behavior unproven + owner: data-architect + next_action: do not claim production scale + approval_required: none + target_phase: future + status: ACCEPTED + - risk_id: RISK-07 + title: No multi-environment production promotion + category: delivery + description: Single dev-oriented environment; no prod promotion pipeline. + evidence: docs/reference-architecture/operating-model.md + likelihood: medium + impact: medium + severity: MEDIUM + current_control: documented boundary + remaining_exposure: promotion process unproven + owner: release-owner + next_action: future sprint + approval_required: none + target_phase: future + status: ACCEPTED + - risk_id: RISK-08 + title: No streaming/event-driven/CDC ingestion + category: architecture + description: Batch only by design. + evidence: docs/reference-architecture/system-context.md + likelihood: low + impact: low + severity: LOW + current_control: documented boundary + remaining_exposure: none within scope + owner: data-architect + next_action: future capstone (prefer API/event) + approval_required: none + target_phase: post-atlas + status: ACCEPTED + - risk_id: RISK-09 + title: External consumers not discoverable from repository + category: governance + description: Only internal consumers are registered in consumers.yml. + evidence: governance/consumers.yml + likelihood: medium + impact: medium + severity: MEDIUM + current_control: internal consumer registry + impact analysis + remaining_exposure: real external consumers unknown to repo + owner: governance-engineer + next_action: do not claim full consumer discovery + approval_required: none + target_phase: future + status: ACCEPTED + - risk_id: RISK-10 + title: Public extraction not yet performed + category: security + description: Repository contains operator email and private project ids. + evidence: docs/reference-architecture/public-extraction-review.md + likelihood: medium + impact: medium + severity: MEDIUM + current_control: public-extraction manifest + validator; repo not published + remaining_exposure: private references present until extraction/redaction + owner: security-reviewer + next_action: run validate_public_extraction.py; extraction in Sprint 8+ + approval_required: ATLAS_APPROVE_PUBLIC_EXTRACTION + target_phase: sprint8-or-later + status: OPEN + - risk_id: RISK-11 + title: Reusable template not yet extracted + category: reference + description: Reference architecture only; no parameterized template. + evidence: docs/reference-architecture/template-extraction-plan.md + likelihood: high + impact: low + severity: MEDIUM + current_control: extraction plan documented + remaining_exposure: template status not claimable + owner: data-architect + next_action: extract after Sprint 8 + approval_required: none + target_phase: post-atlas + status: DEFERRED + - risk_id: RISK-12 + title: No second-project generation test + category: reference + description: Template not validated by generating a separate project. + evidence: docs/reference-architecture/template-extraction-plan.md + likelihood: high + impact: low + severity: MEDIUM + current_control: acceptance test defined in plan + remaining_exposure: reusability unproven + owner: data-architect + next_action: generate + validate separate project post-extraction + approval_required: none + target_phase: post-atlas + status: DEFERRED + - risk_id: RISK-13 + title: Clean-clone platform limitations + category: reproducibility + description: Clean-clone may reveal platform-specific setup friction. + evidence: docs/evidence-sprint8/clean-clone-results.md + likelihood: low + impact: low + severity: LOW + current_control: fresh-directory reproduction test + fixes + remaining_exposure: see clean-clone results + owner: data-engineer + next_action: fix repository-controlled friction; rerun fresh + approval_required: none + target_phase: sprint8 + status: OPEN + - risk_id: RISK-14 + title: Independent handoff ambiguity + category: reproducibility + description: Independent tester may hit documentation ambiguity. + evidence: docs/evidence-sprint8/independent-handoff-results.md + likelihood: low + impact: low + severity: LOW + current_control: independent handoff test + scorecard + doc fixes + remaining_exposure: see handoff results + owner: documentation-owner + next_action: repair handoff defects; rerun affected portions + approval_required: none + target_phase: sprint8 + status: OPEN diff --git a/mypy.ini b/mypy.ini new file mode 100644 index 0000000..fbc0f4c --- /dev/null +++ b/mypy.ini @@ -0,0 +1,15 @@ +# Mypy configuration for the Atlas core package (Sprint 4 CI gate). +# Scope is src/atlas; google-cloud libraries ship without stubs. +# python_version matches the CI runtime (3.12). Runtime code still targets +# the Composer image's Python 3.11 — no 3.12-only syntax is used — but 3.11 +# here makes mypy reject PEP 695 `type` statements in third-party stubs +# (e.g. numpy, present whenever Airflow is installed in the same venv). +[mypy] +python_version = 3.12 +mypy_path = src +packages = atlas +ignore_missing_imports = True +warn_unused_ignores = True +warn_redundant_casts = True +no_implicit_optional = True +check_untyped_defs = True diff --git a/observability/alerts/atlas-composer-unhealthy.json b/observability/alerts/atlas-composer-unhealthy.json new file mode 100644 index 0000000..64e83ae --- /dev/null +++ b/observability/alerts/atlas-composer-unhealthy.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: Composer environment unhealthy", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: native `composer.googleapis.com/environment/healthy` (fraction true < 0.5 for 15 min). Native platform metric \u2014 deliberately not re-created under an Atlas name (ADR-011).\n\n**Meaning**: the atlas-dev Composer environment is failing its own health checks.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-composer-unhealthy\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: not drill-injectable without harming the environment; validated by observing the metric during environment creation/deletion windows.\n\n**Teardown**: DISABLE this policy before intentional Composer deletion (scripts/manage_atlas_alerts.sh disable atlas-composer-unhealthy); metric absence after deletion does not fire, but disabling removes ambiguity." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-composer-unhealthy", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "environment healthy fraction < 0.5", + "conditionThreshold": { + "filter": "metric.type=\"composer.googleapis.com/environment/healthy\" AND resource.type=\"cloud_composer_environment\"", + "comparison": "COMPARISON_LT", + "thresholdValue": 0.5, + "duration": "900s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "300s", + "perSeriesAligner": "ALIGN_FRACTION_TRUE" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-cost-anomaly.json b/observability/alerts/atlas-cost-anomaly.json new file mode 100644 index 0000000..53be816 --- /dev/null +++ b/observability/alerts/atlas-cost-anomaly.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: BigQuery cost anomaly", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=cost_anomaly` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: Atlas-attributed BigQuery bytes billed in 24 h >= 10x the 7-day daily baseline (and above the 1 GiB floor).\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-cost-anomaly\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test cost_anomaly` publishes a synthetic FAIL point (mode=drill). Known false-positive risks: legitimate backfills; verify against observability/queries/bigquery_cost.sql before acting.\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "warning", + "incident_key": "atlas-cost_anomaly", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "cost_anomaly check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"cost_anomaly\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-data-stale.json b/observability/alerts/atlas-data-stale.json new file mode 100644 index 0000000..f262e9c --- /dev/null +++ b/observability/alerts/atlas-data-stale.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: data stale", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=freshness` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: Age since the last SUCCESS pipeline run exceeded the freshness fail threshold (50 h).\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-data-stale\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test freshness` publishes a synthetic FAIL point (mode=drill). Known false-positive risks: environment deliberately paused between acceptance windows without disabling monitoring.\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-freshness", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "freshness check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"freshness\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-deployment-failed.json b/observability/alerts/atlas-deployment-failed.json new file mode 100644 index 0000000..c3d452f --- /dev/null +++ b/observability/alerts/atlas-deployment-failed.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: deployment failed", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=deployment_failure` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: The most recent atlas_ops.deployments row is FAILED or ROLLBACK_FAILED.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-deployment-failed\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test deployment_failure` publishes a synthetic FAIL point (mode=drill).\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-deployment_failure", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "deployment_failure check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"deployment_failure\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-pipeline-failed.json b/observability/alerts/atlas-pipeline-failed.json new file mode 100644 index 0000000..e7e4033 --- /dev/null +++ b/observability/alerts/atlas-pipeline-failed.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: pipeline failed", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=latest_run_state` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: The most recent atlas_batch_pipeline run finished FAILED (atlas_ops.pipeline_runs).\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-pipeline-failed\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test latest_run_state` publishes a synthetic FAIL point (mode=drill).\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-latest_run_state", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "latest_run_state check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"latest_run_state\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-reconciliation-failed.json b/observability/alerts/atlas-reconciliation-failed.json new file mode 100644 index 0000000..c484d39 --- /dev/null +++ b/observability/alerts/atlas-reconciliation-failed.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: reconciliation failed", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=reconciliation` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: atlas_ops.quality_results contains FAIL rows for the latest pipeline run.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-reconciliation-failed\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test reconciliation` publishes a synthetic FAIL point (mode=drill).\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-reconciliation", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "reconciliation check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"reconciliation\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-rollback-failed.json b/observability/alerts/atlas-rollback-failed.json new file mode 100644 index 0000000..e40a6e7 --- /dev/null +++ b/observability/alerts/atlas-rollback-failed.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: rollback failed", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=rollback_failure` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: The most recent rollback attempt recorded ROLLBACK_FAILED \u2014 manual intervention required.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-rollback-failed\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test rollback_failure` publishes a synthetic FAIL point (mode=drill).\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-rollback_failure", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "rollback_failure check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"rollback_failure\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-schema-drift.json b/observability/alerts/atlas-schema-drift.json new file mode 100644 index 0000000..eb34fe6 --- /dev/null +++ b/observability/alerts/atlas-schema-drift.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: breaking schema drift", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=schema_drift` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: A BREAKING schema difference exists between governed manifests and live INFORMATION_SCHEMA.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-schema-drift\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test schema_drift` publishes a synthetic FAIL point (mode=drill). Known false-positive risks: manifest not regenerated after an approved additive migration.\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-schema_drift", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "schema_drift check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"schema_drift\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-telemetry-incomplete.json b/observability/alerts/atlas-telemetry-incomplete.json new file mode 100644 index 0000000..e407059 --- /dev/null +++ b/observability/alerts/atlas-telemetry-incomplete.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: telemetry incomplete", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=telemetry_completeness` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: Expected tasks are missing terminal task_events rows for the latest run \u2014 observability is degraded even if data may be correct.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-telemetry-incomplete\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test telemetry_completeness` publishes a synthetic FAIL point (mode=drill). Known false-positive risks: runs that predate task telemetry (Sprint 4 and earlier).\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "warning", + "incident_key": "atlas-telemetry_completeness", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "telemetry_completeness check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"telemetry_completeness\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/alerts/atlas-volume-deviation.json b/observability/alerts/atlas-volume-deviation.json new file mode 100644 index 0000000..9b04ccb --- /dev/null +++ b/observability/alerts/atlas-volume-deviation.json @@ -0,0 +1,42 @@ +{ + "displayName": "Atlas: critical volume deviation", + "documentation": { + "mimeType": "text/markdown", + "content": "**Signal**: `custom.googleapis.com/atlas/monitor/check_status` for `check_name=volume_deviation` (2 = FAIL, published by atlas_observability_monitor every 30 min).\n\n**Meaning**: Latest raw row count deviates >= 80 % from the 7-run baseline median.\n\n**Runbook**: docs/observability-runbook-sprint5.md#alert-atlas-volume-deviation\n\n**Owner**: the primary operator (primary operator).\n\n**Test method**: `scripts/manage_atlas_alerts.sh test volume_deviation` publishes a synthetic FAIL point (mode=drill). Known false-positive risks: intentional batch-size changes; retune volume.baseline after planned changes.\n\n**Auto-close**: incident closes ~30 min after the check returns PASS (gauge stops satisfying the condition; autoClose 1800s after cessation)." + }, + "userLabels": { + "application": "atlas", + "environment": "atlas-dev", + "severity": "critical", + "incident_key": "atlas-volume_deviation", + "managed_by": "atlas-sprint5" + }, + "combiner": "OR", + "conditions": [ + { + "displayName": "volume_deviation check_status is FAIL", + "conditionThreshold": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"volume_deviation\" AND resource.type=\"global\"", + "comparison": "COMPARISON_GT", + "thresholdValue": 1.5, + "duration": "0s", + "trigger": { + "count": 1 + }, + "aggregations": [ + { + "alignmentPeriod": "600s", + "perSeriesAligner": "ALIGN_MAX" + } + ] + } + } + ], + "alertStrategy": { + "autoClose": "1800s" + }, + "notificationChannels": [ + "${NOTIFICATION_CHANNEL}" + ], + "enabled": true +} diff --git a/observability/dashboards/atlas-operations.json b/observability/dashboards/atlas-operations.json new file mode 100644 index 0000000..5514eae --- /dev/null +++ b/observability/dashboards/atlas-operations.json @@ -0,0 +1,811 @@ +{ + "displayName": "Atlas Operations", + "labels": { + "application": "atlas", + "managed_by": "atlas-sprint5" + }, + "mosaicLayout": { + "columns": 48, + "tiles": [ + { + "widget": { + "title": "Current status (all scorecards: green 0 = PASS, yellow 1 = WARN, red 2 = FAIL)", + "sectionHeader": { + "dividerBelow": true + } + }, + "width": 48, + "height": 3, + "xPos": 0, + "yPos": 0 + }, + { + "widget": { + "title": "Latest pipeline run", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"latest_run_state\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 10, + "height": 4, + "xPos": 0, + "yPos": 3 + }, + { + "widget": { + "title": "Age since last success (s)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/pipeline/last_success_age_seconds\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 14, + "height": 4, + "xPos": 10, + "yPos": 3 + }, + { + "widget": { + "title": "Latest deployment", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"deployment_failure\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 10, + "height": 4, + "xPos": 24, + "yPos": 3 + }, + { + "widget": { + "title": "Composer healthy (native)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"composer.googleapis.com/environment/healthy\" AND resource.type=\"cloud_composer_environment\"", + "aggregation": { + "alignmentPeriod": "300s", + "perSeriesAligner": "ALIGN_FRACTION_TRUE" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 14, + "height": 4, + "xPos": 34, + "yPos": 3 + }, + { + "widget": { + "title": "Pipeline reliability", + "sectionHeader": { + "dividerBelow": true + } + }, + "width": 48, + "height": 3, + "xPos": 0, + "yPos": 7 + }, + { + "widget": { + "title": "Run duration (s)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/pipeline/run_duration_seconds\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 0, + "yPos": 10 + }, + { + "widget": { + "title": "Telemetry incomplete tasks", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/pipeline/telemetry_incomplete_count\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 16, + "yPos": 10 + }, + { + "widget": { + "title": "Telemetry completeness", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"telemetry_completeness\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 32, + "yPos": 10 + }, + { + "widget": { + "title": "Missing scheduled run", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"missing_scheduled_run\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 40, + "yPos": 10 + }, + { + "widget": { + "title": "Data health", + "sectionHeader": { + "dividerBelow": true + } + }, + "width": 48, + "height": 3, + "xPos": 0, + "yPos": 14 + }, + { + "widget": { + "title": "Raw / accepted / rejected rows", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/data/raw_row_count\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 0, + "yPos": 17 + }, + { + "widget": { + "title": "Rejection rate", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/data/rejection_rate\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 12, + "height": 4, + "xPos": 16, + "yPos": 17 + }, + { + "widget": { + "title": "Volume deviation ratio (1.0 = baseline)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/data/volume_deviation_ratio\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 12, + "height": 4, + "xPos": 28, + "yPos": 17 + }, + { + "widget": { + "title": "Reconciliation", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"reconciliation\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 40, + "yPos": 17 + }, + { + "widget": { + "title": "Freshness", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"freshness\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 0, + "yPos": 21 + }, + { + "widget": { + "title": "Schema drift", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"schema_drift\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 8, + "yPos": 21 + }, + { + "widget": { + "title": "Schema drift findings by severity", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/data/schema_drift_count\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 16, + "yPos": 21 + }, + { + "widget": { + "title": "Reconciliation failures", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/data/reconciliation_failure_count\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 32, + "yPos": 21 + }, + { + "widget": { + "title": "Delivery", + "sectionHeader": { + "dividerBelow": true + } + }, + "width": 48, + "height": 3, + "xPos": 0, + "yPos": 25 + }, + { + "widget": { + "title": "Deployment duration (s)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/deployment/duration_seconds\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 0, + "yPos": 28 + }, + { + "widget": { + "title": "Latest deployment failed (1 = yes)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/deployment/latest_failed\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 16, + "yPos": 28 + }, + { + "widget": { + "title": "Rollback health", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"rollback_failure\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 32, + "yPos": 28 + }, + { + "widget": { + "title": "Deployment health", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"deployment_failure\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 40, + "yPos": 28 + }, + { + "widget": { + "title": "Cost", + "sectionHeader": { + "dividerBelow": true + } + }, + "width": 48, + "height": 3, + "xPos": 0, + "yPos": 32 + }, + { + "widget": { + "title": "Atlas BigQuery bytes billed (24h window)", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/cost/bigquery_bytes_billed\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "By", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 0, + "yPos": 35 + }, + { + "widget": { + "title": "Atlas BigQuery job count", + "xyChart": { + "dataSets": [ + { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/cost/bigquery_job_count\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "plotType": "LINE" + } + ], + "yAxis": { + "label": "", + "scale": "LINEAR" + } + } + }, + "width": 16, + "height": 4, + "xPos": 16, + "yPos": 35 + }, + { + "widget": { + "title": "Cost anomaly", + "scorecard": { + "timeSeriesQuery": { + "timeSeriesFilter": { + "filter": "metric.type=\"custom.googleapis.com/atlas/monitor/check_status\" AND metric.label.check_name=\"cost_anomaly\" AND metric.label.mode=\"normal\"", + "aggregation": { + "alignmentPeriod": "1800s", + "perSeriesAligner": "ALIGN_MAX" + } + } + }, + "thresholds": [ + { + "value": 0.5, + "color": "YELLOW", + "direction": "ABOVE" + }, + { + "value": 1.5, + "color": "RED", + "direction": "ABOVE" + } + ] + } + }, + "width": 8, + "height": 4, + "xPos": 32, + "yPos": 35 + }, + { + "widget": { + "title": "Logs (Atlas runtime \u2014 bucket atlas-observability, view atlas-runtime)", + "sectionHeader": { + "dividerBelow": true + } + }, + "width": 48, + "height": 3, + "xPos": 0, + "yPos": 39 + }, + { + "widget": { + "title": "Atlas errors and retries", + "logsPanel": { + "filter": "jsonPayload.atlas_event=true AND (severity>=ERROR OR jsonPayload.event_type=\"task_retry\")", + "resourceNames": [ + "projects/example-gcp-project" + ] + } + }, + "width": 24, + "height": 8, + "xPos": 0, + "yPos": 42 + }, + { + "widget": { + "title": "Composer DAG-processor errors", + "logsPanel": { + "filter": "resource.type=\"cloud_composer_environment\" AND severity>=ERROR", + "resourceNames": [ + "projects/example-gcp-project" + ] + } + }, + "width": 24, + "height": 8, + "xPos": 24, + "yPos": 42 + } + ] + } +} \ No newline at end of file diff --git a/observability/logging/log-bucket.json b/observability/logging/log-bucket.json new file mode 100644 index 0000000..6f096e2 --- /dev/null +++ b/observability/logging/log-bucket.json @@ -0,0 +1,9 @@ +{ + "_comment": "Atlas dedicated log bucket (Sprint 5, Phase 5). Created by scripts/bootstrap_observability.sh; this file is the versioned source of truth for its configuration.", + "bucket_id": "atlas-observability", + "location": "us-central1", + "retention_days": 30, + "analytics_enabled": true, + "locked": false, + "description": "Atlas runtime logs: Composer environment logs plus structured Atlas contract events. Retention 30d per cost-review-sprint5.md; Log Analytics enabled for the atlas_logs linked BigQuery dataset." +} diff --git a/observability/logging/log-view.json b/observability/logging/log-view.json new file mode 100644 index 0000000..7ec836e --- /dev/null +++ b/observability/logging/log-view.json @@ -0,0 +1,8 @@ +{ + "_comment": "Atlas runtime log view (Sprint 5, Phase 5). Scopes reader access to Atlas runtime entries inside the atlas-observability bucket without granting bucket-wide access.", + "view_id": "atlas-runtime", + "bucket_id": "atlas-observability", + "location": "us-central1", + "filter": "SOURCE(\"projects/example-gcp-project\")", + "description": "Least-privilege view over the atlas-observability bucket. Grant roles/logging.viewAccessor on this view instead of bucket- or project-wide log access." +} diff --git a/observability/logging/sink-filter.txt b/observability/logging/sink-filter.txt new file mode 100644 index 0000000..0ed7eee --- /dev/null +++ b/observability/logging/sink-filter.txt @@ -0,0 +1,14 @@ +-- Atlas log sink filter (Sprint 5, Phase 5). +-- Routes Atlas runtime logs into the dedicated atlas-observability bucket. +-- The sink is additive: the _Default bucket keeps receiving these entries +-- (no exclusion filters), so a sink misconfiguration cannot lose logs. +-- Scope: +-- 1. All Cloud Composer environment logs in this project (workers, +-- scheduler, DAG processor, task logs). example-gcp-project runs only +-- the Atlas atlas-dev environment, so no per-environment narrowing is +-- required; revisit if a second environment ever appears. +-- 2. Structured Atlas contract events emitted outside Composer (deploy, +-- rollback, drills) carrying the jsonPayload.atlas_event marker. +-- Lines starting with "--" are comments and are stripped by the bootstrap +-- script before the filter is applied. +resource.type="cloud_composer_environment" OR jsonPayload.atlas_event=true diff --git a/observability/metrics/metric-descriptors.json b/observability/metrics/metric-descriptors.json new file mode 100644 index 0000000..640f9d7 --- /dev/null +++ b/observability/metrics/metric-descriptors.json @@ -0,0 +1,134 @@ +{ + "_comment": "Atlas custom metric descriptors (Sprint 5, Phase 6). Created idempotently by `python -m atlas.observability.metrics --ensure-descriptors` (invoked from bootstrap_observability.sh --apply). Cardinality budget: labels are drawn ONLY from the bounded sets documented per metric; run/batch/deployment ids and error strings are forbidden (ADR-011). Native Composer metrics (environment healthy, scheduler heartbeat, DAG parse stats) are used as-is and deliberately NOT recreated here.", + "cardinality_budget": { + "environment": ["atlas-dev"], + "dag_id": ["atlas_batch_pipeline", "atlas_observability_monitor"], + "component": ["pipeline", "monitor", "deployment", "cost"], + "check_name": "bounded by config/observability.yaml checks (<= 15)", + "severity": ["INFO", "WARNING", "CRITICAL"], + "mode": ["normal", "drill"], + "max_projected_time_series": 120 + }, + "descriptors": [ + { + "type": "custom.googleapis.com/atlas/pipeline/last_success_age_seconds", + "metric_kind": "GAUGE", + "value_type": "DOUBLE", + "unit": "s", + "description": "Seconds since the last SUCCESS row in atlas_ops.pipeline_runs. Source of truth: Plane 1; published by the monitor DAG.", + "labels": ["environment", "dag_id", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/pipeline/run_duration_seconds", + "metric_kind": "GAUGE", + "value_type": "DOUBLE", + "unit": "s", + "description": "Duration of the most recent terminal pipeline run.", + "labels": ["environment", "dag_id", "status", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/pipeline/telemetry_incomplete_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Expected tasks missing terminal task_events rows for the latest run (0 = telemetry complete).", + "labels": ["environment", "dag_id", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/raw_row_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Raw rows loaded for the latest successful batch.", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/accepted_row_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Accepted rows for the latest successful batch.", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/rejected_row_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Rejected rows for the latest successful batch.", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/rejection_rate", + "metric_kind": "GAUGE", + "value_type": "DOUBLE", + "unit": "10^2.%", + "description": "rejected / raw for the latest successful batch (0.0-1.0).", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/volume_deviation_ratio", + "metric_kind": "GAUGE", + "value_type": "DOUBLE", + "unit": "1", + "description": "Latest raw row count divided by the baseline-window median (1.0 = on baseline).", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/reconciliation_failure_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "FAIL rows in atlas_ops.quality_results for the latest run.", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/data/schema_drift_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Schema-drift findings by severity for monitored tables.", + "labels": ["environment", "severity", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/deployment/latest_failed", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "1 when the most recent atlas_ops.deployments row is FAILED/ROLLBACK_FAILED, else 0.", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/deployment/duration_seconds", + "metric_kind": "GAUGE", + "value_type": "DOUBLE", + "unit": "s", + "description": "Duration of the most recent terminal deployment or rollback.", + "labels": ["environment", "status", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/cost/bigquery_bytes_billed", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "By", + "description": "Atlas-attributed BigQuery bytes billed in the monitor evaluation window (ADR-012 attribution).", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/cost/bigquery_job_count", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Atlas-attributed BigQuery job count in the monitor evaluation window.", + "labels": ["environment", "mode"] + }, + { + "type": "custom.googleapis.com/atlas/monitor/check_status", + "metric_kind": "GAUGE", + "value_type": "INT64", + "unit": "1", + "description": "Uniform monitor outcome per check: 0=PASS, 1=WARN, 2=FAIL, -1=NO_DATA, -2=DISABLED. Primary alerting surface.", + "labels": ["environment", "check_name", "mode"] + } + ] +} diff --git a/observability/performance/queries/01_raw_batch_lookup.sql b/observability/performance/queries/01_raw_batch_lookup.sql new file mode 100644 index 0000000..fdc528e --- /dev/null +++ b/observability/performance/queries/01_raw_batch_lookup.sql @@ -0,0 +1,5 @@ +-- Raw batch lookup: bounded by partition (event_date). Representative of a +-- single-batch investigation. Correctness checksum = row count for the date. +select count(*) as row_count, count(distinct event_id) as distinct_events +from `${PROJECT}.atlas_raw.events` +where event_date = DATE '2026-07-17' diff --git a/observability/performance/queries/02_batch_classification.sql b/observability/performance/queries/02_batch_classification.sql new file mode 100644 index 0000000..2aa2401 --- /dev/null +++ b/observability/performance/queries/02_batch_classification.sql @@ -0,0 +1,7 @@ +-- Batch classification profile: duplicate/quality flags for one partition. +-- Representative of the anomaly-profile workload. +select + countif(user_id is null) as null_user, + count(*) as total +from `${PROJECT}.atlas_raw.events` +where event_date = DATE '2026-07-17' diff --git a/observability/performance/queries/03_reconciliation.sql b/observability/performance/queries/03_reconciliation.sql new file mode 100644 index 0000000..8a75d69 --- /dev/null +++ b/observability/performance/queries/03_reconciliation.sql @@ -0,0 +1,5 @@ +-- Accepted/rejected reconciliation for one partition: accepted fact rows vs raw. +-- Correctness: accepted <= raw for the partition. +select + (select count(*) from `${PROJECT}.atlas_raw.events` where event_date = DATE '2026-07-17') as raw_rows, + (select count(*) from `${PROJECT}.atlas_core.fct_events` where event_date = DATE '2026-07-17') as fact_rows diff --git a/observability/performance/queries/04_fact_build_scan.sql b/observability/performance/queries/04_fact_build_scan.sql new file mode 100644 index 0000000..f56175e --- /dev/null +++ b/observability/performance/queries/04_fact_build_scan.sql @@ -0,0 +1,6 @@ +-- Fact build scan (incremental lookback simulation): scan accepted fact rows in +-- a bounded ingested window. Clustered by event_name, country_code. +select event_name, country_code, count(*) as n +from `${PROJECT}.atlas_core.fct_events` +where event_date between DATE '2026-07-15' and DATE '2026-07-17' +group by event_name, country_code diff --git a/observability/performance/queries/05_mart_aggregation.sql b/observability/performance/queries/05_mart_aggregation.sql new file mode 100644 index 0000000..11e918f --- /dev/null +++ b/observability/performance/queries/05_mart_aggregation.sql @@ -0,0 +1,6 @@ +-- Mart aggregation: daily metrics read (small mart). Representative BI query. +select event_date, sum(event_count) as total_events +from `${PROJECT}.atlas_marts.mart_daily_event_metrics` +where event_date between DATE '2026-07-01' and DATE '2026-07-31' +group by event_date +order by event_date diff --git a/observability/performance/queries/06_freshness.sql b/observability/performance/queries/06_freshness.sql new file mode 100644 index 0000000..73db6da --- /dev/null +++ b/observability/performance/queries/06_freshness.sql @@ -0,0 +1,6 @@ +-- Freshness query: latest ingest recency from the fact table. +select + max(ingested_at) as last_ingest, + timestamp_diff(current_timestamp(), max(ingested_at), SECOND) as staleness_seconds +from `${PROJECT}.atlas_core.fct_events` +where event_date >= DATE '2026-07-15' diff --git a/observability/performance/queries/07_operational_audit.sql b/observability/performance/queries/07_operational_audit.sql new file mode 100644 index 0000000..bb41e13 --- /dev/null +++ b/observability/performance/queries/07_operational_audit.sql @@ -0,0 +1,5 @@ +-- Operational-audit query: recent pipeline run outcomes. +select status, count(*) as runs +from `${PROJECT}.atlas_ops.pipeline_runs` +group by status +order by runs desc diff --git a/observability/performance/queries/08_cost_monitor.sql b/observability/performance/queries/08_cost_monitor.sql new file mode 100644 index 0000000..9e475a6 --- /dev/null +++ b/observability/performance/queries/08_cost_monitor.sql @@ -0,0 +1,8 @@ +-- Cost-monitor query: recent Atlas BigQuery jobs bytes billed via +-- INFORMATION_SCHEMA (region-scoped, last 7 days, bounded). +select + count(*) as jobs, + sum(total_bytes_billed) as bytes_billed +from `${PROJECT}`.`region-us`.INFORMATION_SCHEMA.JOBS_BY_PROJECT +where creation_time >= timestamp_sub(current_timestamp(), interval 7 day) + and job_type = 'QUERY' diff --git a/observability/performance/queries/09_metadata.sql b/observability/performance/queries/09_metadata.sql new file mode 100644 index 0000000..412e0e1 --- /dev/null +++ b/observability/performance/queries/09_metadata.sql @@ -0,0 +1,6 @@ +-- Lineage/schema metadata query: column inventory for the core dataset via +-- INFORMATION_SCHEMA (metadata only, negligible bytes). +select table_name, count(*) as columns +from `${PROJECT}.atlas_core.INFORMATION_SCHEMA.COLUMNS` +group by table_name +order by table_name diff --git a/observability/performance/queries/unbounded_scan.sql b/observability/performance/queries/unbounded_scan.sql new file mode 100644 index 0000000..87ae44c --- /dev/null +++ b/observability/performance/queries/unbounded_scan.sql @@ -0,0 +1,6 @@ +-- DELIBERATELY UNBOUNDED: full scan of raw events with no partition filter. +-- Used ONLY to demonstrate that the cost guard blocks it at dry-run before any +-- spend. Never run this as a real workload. +select event_name, country_code, count(*) as n +from `${PROJECT}.atlas_raw.events` +group by event_name, country_code diff --git a/observability/performance/results/baseline-dryrun.json b/observability/performance/results/baseline-dryrun.json new file mode 100644 index 0000000..5ad0dd4 --- /dev/null +++ b/observability/performance/results/baseline-dryrun.json @@ -0,0 +1,55 @@ +{ + "project": "example-gcp-project", + "location": "US", + "executed": false, + "per_query_ceiling_bytes": 1073741824, + "suite_ceiling_bytes": 5368709120, + "cumulative_bytes_billed": 0, + "queries": [ + { + "query": "01_raw_batch_lookup", + "estimated_bytes": 2315824, + "within_per_query_ceiling": true + }, + { + "query": "02_batch_classification", + "estimated_bytes": 745937, + "within_per_query_ceiling": true + }, + { + "query": "03_reconciliation", + "estimated_bytes": 795136, + "within_per_query_ceiling": true + }, + { + "query": "04_fact_build_scan", + "estimated_bytes": 2255810, + "within_per_query_ceiling": true + }, + { + "query": "05_mart_aggregation", + "estimated_bytes": 44864, + "within_per_query_ceiling": true + }, + { + "query": "06_freshness", + "estimated_bytes": 4700784, + "within_per_query_ceiling": true + }, + { + "query": "07_operational_audit", + "estimated_bytes": 216, + "within_per_query_ceiling": true + }, + { + "query": "08_cost_monitor", + "estimated_bytes": 34428633, + "within_per_query_ceiling": true + }, + { + "query": "09_metadata", + "estimated_bytes": 10485760, + "within_per_query_ceiling": true + } + ] +} diff --git a/observability/queries/bigquery_cost.sql b/observability/queries/bigquery_cost.sql new file mode 100644 index 0000000..9bd9bd0 --- /dev/null +++ b/observability/queries/bigquery_cost.sql @@ -0,0 +1,94 @@ +-- Atlas BigQuery cost attribution queries (Sprint 5, Phase 7 / ADR-012). +-- All queries are region-qualified (`region-us`: Atlas datasets live in US) +-- and bounded to a trailing window; INFORMATION_SCHEMA.JOBS retains ~180 days. +-- Attribution: job label application=atlas (Python via labeled clients, dbt +-- via query-comment job-label). Parent multi-statement rows are excluded +-- (statement_type = 'SCRIPT') to prevent double counting. +-- Full query text is intentionally never copied into Atlas tables. + +-- 1. Daily Atlas bytes processed and billed (last 14 days) +SELECT + DATE(creation_time) AS usage_date, + COUNT(*) AS job_count, + SUM(total_bytes_processed) AS bytes_processed, + SUM(total_bytes_billed) AS bytes_billed, + ROUND(SUM(total_bytes_billed) / POW(2, 40) * 6.25, 4) AS approx_usd_on_demand +FROM `example-gcp-project.region-us.INFORMATION_SCHEMA.JOBS` +WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 14 DAY) + AND ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + AND statement_type != 'SCRIPT' +GROUP BY usage_date +ORDER BY usage_date DESC; + +-- 2. Daily slot milliseconds (last 14 days) +SELECT + DATE(creation_time) AS usage_date, + SUM(total_slot_ms) AS slot_ms +FROM `example-gcp-project.region-us.INFORMATION_SCHEMA.JOBS` +WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 14 DAY) + AND ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + AND statement_type != 'SCRIPT' +GROUP BY usage_date +ORDER BY usage_date DESC; + +-- 3. Job failures (last 7 days) +SELECT + DATE(creation_time) AS usage_date, + error_result.reason AS error_reason, + COUNT(*) AS failed_jobs +FROM `example-gcp-project.region-us.INFORMATION_SCHEMA.JOBS` +WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 7 DAY) + AND ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + AND error_result IS NOT NULL +GROUP BY usage_date, error_reason +ORDER BY usage_date DESC, failed_jobs DESC; + +-- 4. Usage by Atlas component (last 14 days) +SELECT + (SELECT value FROM UNNEST(labels) WHERE key = 'component') AS component, + COUNT(*) AS job_count, + SUM(total_bytes_billed) AS bytes_billed, + SUM(total_slot_ms) AS slot_ms +FROM `example-gcp-project.region-us.INFORMATION_SCHEMA.JOBS` +WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 14 DAY) + AND ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + AND statement_type != 'SCRIPT' +GROUP BY component +ORDER BY bytes_billed DESC; + +-- 5. Unusually expensive jobs (last 7 days; adjust threshold to baseline) +SELECT + creation_time, + job_id, + user_email, + (SELECT value FROM UNNEST(labels) WHERE key = 'component') AS component, + total_bytes_billed, + total_slot_ms +FROM `example-gcp-project.region-us.INFORMATION_SCHEMA.JOBS` +WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 7 DAY) + AND ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + AND statement_type != 'SCRIPT' + AND total_bytes_billed > 1 * POW(2, 30) -- > 1 GiB billed +ORDER BY total_bytes_billed DESC +LIMIT 50; + +-- 6. Trend: weekly bytes billed, labeled vs identity-attributed fallback +-- (catches jobs that escaped labeling; identities are the Atlas SAs) +SELECT + TIMESTAMP_TRUNC(creation_time, WEEK) AS week_start, + COUNTIF(('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels))) AS labeled_jobs, + COUNTIF(('application', 'atlas') NOT IN (SELECT (key, value) FROM UNNEST(labels))) AS unlabeled_jobs, + SUM(total_bytes_billed) AS bytes_billed +FROM `example-gcp-project.region-us.INFORMATION_SCHEMA.JOBS` +WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 42 DAY) + AND statement_type != 'SCRIPT' + AND ( + ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + OR user_email IN ( + 'atlas-composer-runtime@example-gcp-project.iam.gserviceaccount.com', + 'atlas-github-integration@example-gcp-project.iam.gserviceaccount.com', + 'atlas-github-deployer@example-gcp-project.iam.gserviceaccount.com' + ) + ) +GROUP BY week_start +ORDER BY week_start DESC; diff --git a/observability/queries/log-filters.md b/observability/queries/log-filters.md new file mode 100644 index 0000000..d7f069e --- /dev/null +++ b/observability/queries/log-filters.md @@ -0,0 +1,131 @@ +# Atlas saved log queries (Sprint 5, Phase 5) + +Cloud Logging filters for Logs Explorer scoped to the `atlas-observability` +bucket / `atlas-runtime` view (or project-wide before routing exists). All +structured Atlas contract events carry `jsonPayload.atlas_event=true` and the +correlation fields from ADR-011. Replace the example identifiers before use. + +## 1. Everything for one pipeline run + +```text +jsonPayload.pipeline_run_id="atlas-20260719T060000Z-abcd1234" +``` + +Airflow's own task logs for the same run (Composer resource logs keyed by the +Airflow run id): + +```text +resource.type="cloud_composer_environment" +labels.workflow="atlas_batch_pipeline" +labels."run_id"="atlas-scheduled__2026-07-19T06:00:00+00:00" +``` + +## 2. One task attempt + +```text +jsonPayload.pipeline_run_id="atlas-20260719T060000Z-abcd1234" +jsonPayload.task_id="load_bigquery_raw" +jsonPayload.attempt_number=2 +``` + +Airflow-native equivalent: + +```text +resource.type="cloud_composer_environment" +labels.workflow="atlas_batch_pipeline" +labels."task-id"="load_bigquery_raw" +labels."try-number"="2" +``` + +## 3. All Atlas failures + +```text +jsonPayload.atlas_event=true +(jsonPayload.event_type="task_failed" OR jsonPayload.severity="ERROR" OR jsonPayload.severity="CRITICAL") +``` + +## 4. All retries + +```text +jsonPayload.atlas_event=true +jsonPayload.event_type="task_retry" +``` + +## 5. One deployment + +```text +jsonPayload.deployment_id="atlas-dev-20260719-abc123" +``` + +## 6. One rollback + +```text +jsonPayload.atlas_event=true +jsonPayload.deployment_id="atlas-dev-20260719-rollback1" +``` + +(rollbacks share the deployment contract; `atlas_ops.deployments` rows with +`deployment_type="rollback"` give the ids) + +## 7. DAG parse errors + +```text +resource.type="cloud_composer_environment" +log_id("airflow-dag-processor-manager") OR log_id("dag-processor-manager") +severity>=ERROR +``` + +## 8. Missing / broken task telemetry + +```text +jsonPayload.atlas_event=true +(jsonPayload.event_type="task_telemetry_write_failed" OR jsonPayload.event_type="telemetry_emit_failed" OR jsonPayload.event_type="quality_result_write_failed") +``` + +Durable completeness check (BigQuery, Plane 1): + +```sql +-- Expected tasks lacking a terminal event for a run +SELECT task_id, ARRAY_AGG(event_type ORDER BY event_type) AS events +FROM `example-gcp-project.atlas_ops.task_events` +WHERE pipeline_run_id = @pipeline_run_id +GROUP BY task_id +HAVING COUNTIF(event_type IN ('SUCCESS','FAILED','SKIPPED','UPSTREAM_FAILED')) = 0; +``` + +## 9. High-duration tasks + +```text +jsonPayload.atlas_event=true +jsonPayload.event_type="task_success" +jsonPayload.duration_ms>300000 +``` + +Durable equivalent: + +```sql +SELECT pipeline_run_id, task_id, attempt_number, duration_ms +FROM `example-gcp-project.atlas_ops.task_events` +WHERE event_type = 'SUCCESS' + AND duration_ms > 300000 + AND created_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 7 DAY) +ORDER BY duration_ms DESC; +``` + +## Linked-dataset access (Log Analytics / BigQuery) + +The linked read-only dataset `atlas_logs` exposes the bucket's `_AllLogs` +view. Example: correlate one run across Composer and Atlas events: + +```sql +SELECT timestamp, severity, + JSON_VALUE(json_payload, '$.event_type') AS event_type, + JSON_VALUE(json_payload, '$.task_id') AS task_id +FROM `example-gcp-project.atlas_logs._AllLogs` +WHERE JSON_VALUE(json_payload, '$.pipeline_run_id') = @pipeline_run_id + AND timestamp >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 1 DAY) +ORDER BY timestamp; +``` + +Always bound `timestamp` — the log table is partitioned by time and unbounded +scans are the main linked-dataset cost risk. diff --git a/observability/schema/expected-schemas.json b/observability/schema/expected-schemas.json new file mode 100644 index 0000000..e0413c2 --- /dev/null +++ b/observability/schema/expected-schemas.json @@ -0,0 +1,757 @@ +{ + "generated_from": "example-gcp-project", + "tables": { + "atlas_core.dim_countries": { + "columns": { + "country_code": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "is_active": { + "data_type": "BOOL", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_core.dim_users": { + "columns": { + "event_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "first_event_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "last_event_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "user_id": { + "data_type": "STRING", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_core.fct_events": { + "columns": { + "app_version": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "batch_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "country_code": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "event_date": { + "data_type": "DATE", + "is_nullable": "YES" + }, + "event_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "event_name": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "event_timestamp": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "has_event_date_timestamp_mismatch": { + "data_type": "BOOL", + "is_nullable": "YES" + }, + "ingested_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "is_backdated_event_date": { + "data_type": "BOOL", + "is_nullable": "YES" + }, + "is_event_time_late_arriving": { + "data_type": "BOOL", + "is_nullable": "YES" + }, + "loaded_to_core_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "platform": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "raw_record_hash": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "source_file": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "user_id": { + "data_type": "STRING", + "is_nullable": "YES" + } + }, + "partition_column": "event_date" + }, + "atlas_marts.mart_daily_event_metrics": { + "columns": { + "backdated_event_date_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "country_code": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "distinct_user_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "event_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "event_date": { + "data_type": "DATE", + "is_nullable": "YES" + }, + "event_date_timestamp_mismatch_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "event_name": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "event_time_late_arriving_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "platform": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_ops.deployments": { + "columns": { + "actor": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "artifact_checksum": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "artifact_uri": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "completed_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "composer_environment": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "composer_region": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "created_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "deployment_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "deployment_type": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "environment": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "error_summary": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_type": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "failure_stage": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "git_ref": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "git_sha": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "migration_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "previous_git_sha": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "release_tag": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "smoke_pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "started_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "status": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "workflow_run_id": { + "data_type": "STRING", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_ops.monitor_evaluations": { + "columns": { + "check_name": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "created_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "details_json": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "environment": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "evaluated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "evaluation_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "incident_key": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "observed_value": { + "data_type": "FLOAT64", + "is_nullable": "YES" + }, + "severity": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "source": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "status": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "threshold": { + "data_type": "FLOAT64", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "window_end": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "window_start": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_ops.pipeline_runs": { + "columns": { + "airflow_run_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "attempt_number": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "batch_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "completed_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "created_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "dag_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "error_message": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_type": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "fact_rows": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "failed_task_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "gcs_uri": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "mart_event_count": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "processing_date": { + "data_type": "DATE", + "is_nullable": "NO" + }, + "rows_accepted": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "rows_generated": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "rows_loaded": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "rows_rejected": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "started_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "status": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + } + }, + "partition_column": "processing_date" + }, + "atlas_ops.quality_results": { + "columns": { + "batch_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "check_category": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "check_name": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "created_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "details_json": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "evaluated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "expected_value": { + "data_type": "FLOAT64", + "is_nullable": "YES" + }, + "git_sha": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "lower_bound": { + "data_type": "FLOAT64", + "is_nullable": "YES" + }, + "model_name": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "observed_value": { + "data_type": "FLOAT64", + "is_nullable": "YES" + }, + "pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "severity": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "status": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "upper_bound": { + "data_type": "FLOAT64", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_ops.schema_migrations": { + "columns": { + "applied_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "applied_by": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_summary": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "git_sha": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "migration_checksum": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "migration_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "status": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "workflow_run_id": { + "data_type": "STRING", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_ops.task_events": { + "columns": { + "airflow_run_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "attempt_number": { + "data_type": "INT64", + "is_nullable": "NO" + }, + "batch_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "completed_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "created_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "dag_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "duration_ms": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "environment": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_message": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_type": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "event_type": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "git_sha": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "operator_type": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "rows_affected": { + "data_type": "INT64", + "is_nullable": "YES" + }, + "started_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "status": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "task_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "timing_confidence": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "timing_source": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + } + }, + "partition_column": null + }, + "atlas_ops.recovery_actions": { + "columns": { + "action_type": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "batch_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "completed_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "created_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "deployment_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "environment": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_summary": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "error_type": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "git_sha": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "incident_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "operator": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "recovery_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "scenario_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "source_state": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "started_at": { + "data_type": "TIMESTAMP", + "is_nullable": "YES" + }, + "status": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "target_state": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "updated_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "verification_status": { + "data_type": "STRING", + "is_nullable": "YES" + } + }, + "partition_column": null + }, + "atlas_raw.events": { + "columns": { + "app_version": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "batch_id": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "country_code": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "event_date": { + "data_type": "DATE", + "is_nullable": "NO" + }, + "event_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "event_name": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "event_timestamp": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "ingested_at": { + "data_type": "TIMESTAMP", + "is_nullable": "NO" + }, + "pipeline_run_id": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "platform": { + "data_type": "STRING", + "is_nullable": "YES" + }, + "processing_date": { + "data_type": "DATE", + "is_nullable": "YES" + }, + "source_file": { + "data_type": "STRING", + "is_nullable": "NO" + }, + "user_id": { + "data_type": "STRING", + "is_nullable": "YES" + } + }, + "partition_column": "event_date" + } + } +} diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..80432c2 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,3 @@ +[pytest] +testpaths = tests +pythonpath = src diff --git a/requirements-ci.txt b/requirements-ci.txt new file mode 100644 index 0000000..bb264f2 --- /dev/null +++ b/requirements-ci.txt @@ -0,0 +1,9 @@ +# Pinned validation toolchain for the canonical CI gate (validate_ci.sh). +# Versions verified locally on 2026-07-18 (Python 3.12.3). +ruff==0.15.22 +mypy==2.3.0 +types-PyYAML==6.0.12.20260518 +yamllint==1.38.0 +shellcheck-py==0.11.0.1 +pytest==9.1.1 +pytest-mock==3.15.1 diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..e4c013c --- /dev/null +++ b/requirements.txt @@ -0,0 +1,7 @@ +google-cloud-bigquery>=3.25.0 +google-cloud-storage>=2.18.0 +PyYAML>=6.0.2 +pytest>=8.3.0 +pytest-mock>=3.14.0 +google-cloud-monitoring==2.31.0 +google-cloud-logging==3.16.1 diff --git a/ruff.toml b/ruff.toml new file mode 100644 index 0000000..7d92223 --- /dev/null +++ b/ruff.toml @@ -0,0 +1,18 @@ +# Ruff configuration for the Atlas core pipeline (Sprint 4 CI gate). +# Scope: src/atlas, scripts, dags, tests. The artifact platform keeps its own tooling. + +line-length = 110 +target-version = "py311" + +[lint] +select = ["E", "F", "W", "I", "UP", "B"] +ignore = [ + # Loader/step-runner keep long BigQuery SQL fragments inline for auditability. + "E501", +] + +[lint.per-file-ignores] +# Tests and DAG-adjacent modules adjust sys.path before importing atlas. +"tests/**" = ["E402"] +"dags/**" = ["E402"] +"scripts/**" = ["E402"] diff --git a/scripts/accept_artifact_platform.sh b/scripts/accept_artifact_platform.sh new file mode 100755 index 0000000..cdd5c97 --- /dev/null +++ b/scripts/accept_artifact_platform.sh @@ -0,0 +1,210 @@ +#!/usr/bin/env bash +# Exercise publish, preview, promote, update, rollback, and privacy controls. +set -Eeuo pipefail +IFS=$'\n\t' +umask 077 + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +REPO_ROOT="$(cd "${ATLAS_ROOT}/.." && pwd)" +DASHBOARD_DIR="${ATLAS_ROOT}/examples/artifact-dashboard" +SITE_SLUG="${ATLAS_ARTIFACT_ACCEPTANCE_SITE:-atlas-dashboard}" +PUBLISHER_URL="${ATLAS_ARTIFACT_PUBLISHER_URL:-}" +HOST_URL="${ATLAS_ARTIFACT_PUBLIC_HOST_URL:-}" +BASELINE_HISTORY_FILE="$(mktemp "${TMPDIR:-/tmp}/atlas-artifact-history-before.XXXXXX")" +FINAL_HISTORY_FILE="$(mktemp "${TMPDIR:-/tmp}/atlas-artifact-history-after.XXXXXX")" + +cleanup() { + rm -f "$BASELINE_HISTORY_FILE" "$FINAL_HISTORY_FILE" +} +trap cleanup EXIT + +fail() { + echo "error: $*" >&2 + exit 1 +} + +require_command() { + command -v "$1" >/dev/null 2>&1 || fail "$1 is required but was not found on PATH" +} + +atlas() { + uv run --project "${ATLAS_ROOT}/artifact-platform" atlas-artifact "$@" +} + +json_field() { + local field="$1" + python3 -c \ + 'import json, sys; print(json.load(sys.stdin)[sys.argv[1]])' \ + "$field" +} + +confirm_preview() { + local release="$1" + local digest="$2" + atlas preview "$SITE_SLUG" "$digest" + if [[ "${ATLAS_ACCEPT_PREVIEWS:-}" == "true" ]]; then + return + fi + local answer + read -r -p "Open the ${release} URL, verify it visually, then type yes: " answer + [[ "$answer" == "yes" ]] || fail "${release} preview was not confirmed" +} + +assert_active_digest() { + local expected="$1" + local actual + actual="$(atlas site show "$SITE_SLUG" | json_field active_digest)" + [[ "$actual" == "$expected" ]] || fail \ + "active digest mismatch: expected ${expected}, found ${actual}" +} + +for command in curl gcloud npm python3 uv; do + require_command "$command" +done +[[ -n "$PUBLISHER_URL" ]] || fail "ATLAS_ARTIFACT_PUBLISHER_URL is required" +[[ -n "$HOST_URL" ]] || fail "ATLAS_ARTIFACT_PUBLIC_HOST_URL is required" + +cd "$REPO_ROOT" + +echo "==> Building deterministic dashboard revisions" +npm ci --prefix "$DASHBOARD_DIR" +npm test --prefix "$DASHBOARD_DIR" +npm run typecheck --prefix "$DASHBOARD_DIR" +npm run build --prefix "$DASHBOARD_DIR" + +if ! atlas site show "$SITE_SLUG" >/dev/null 2>&1; then + atlas site create "$SITE_SLUG" --reason "create live acceptance site" +fi +atlas history "$SITE_SLUG" >"$BASELINE_HISTORY_FILE" + +echo "==> Publishing dashboard v1" +V1_JSON="$( + atlas publish "$SITE_SLUG" "${DASHBOARD_DIR}/dist/v1" \ + --label v1 \ + --message "publish fake Atlas dashboard v1" +)" +V1_DIGEST="$(printf '%s' "$V1_JSON" | json_field digest)" +confirm_preview "v1" "$V1_DIGEST" +atlas promote "$SITE_SLUG" "$V1_DIGEST" \ + --reason "acceptance promote v1 after manual preview" \ + --confirm-preview +atlas verify-active "$SITE_SLUG" +assert_active_digest "$V1_DIGEST" + +echo "==> Publishing dashboard v2" +V2_JSON="$( + atlas publish "$SITE_SLUG" "${DASHBOARD_DIR}/dist/v2" \ + --label v2 \ + --message "publish visibly distinct fake Atlas dashboard v2" +)" +V2_DIGEST="$(printf '%s' "$V2_JSON" | json_field digest)" +[[ "$V1_DIGEST" != "$V2_DIGEST" ]] || fail "v1 and v2 unexpectedly share a digest" +confirm_preview "v2" "$V2_DIGEST" +atlas promote "$SITE_SLUG" "$V2_DIGEST" \ + --reason "acceptance update to v2 after manual preview" \ + --confirm-preview +atlas verify-active "$SITE_SLUG" +assert_active_digest "$V2_DIGEST" + +echo "==> Rolling back without copying artifact bytes" +atlas rollback "$SITE_SLUG" "$V1_DIGEST" \ + --reason "acceptance rollback from v2 to v1" \ + --confirm-preview +atlas verify-active "$SITE_SLUG" +assert_active_digest "$V1_DIGEST" + +echo "==> Re-promoting v2 as the final active revision" +atlas promote "$SITE_SLUG" "$V2_DIGEST" \ + --reason "acceptance restore latest v2" \ + --confirm-preview +atlas verify-active "$SITE_SLUG" +assert_active_digest "$V2_DIGEST" + +echo "==> Proving disable and enable preserve the active revision" +atlas disable "$SITE_SLUG" --reason "acceptance privacy stop" +if atlas smoke "$SITE_SLUG" "$V2_DIGEST" >/dev/null 2>&1; then + fail "disabled revision remained reachable through authenticated IAP" +fi +atlas enable "$SITE_SLUG" --reason "acceptance restore site" +atlas smoke "$SITE_SLUG" "$V2_DIGEST" +atlas verify-active "$SITE_SLUG" + +SITE_JSON="$(atlas site show "$SITE_SLUG")" +ACTIVE_DIGEST="$(printf '%s' "$SITE_JSON" | json_field active_digest)" +ENABLED="$(printf '%s' "$SITE_JSON" | json_field enabled)" +[[ "$ACTIVE_DIGEST" == "$V2_DIGEST" ]] || fail "v2 is not the final active digest" +[[ "$ENABLED" == "True" ]] || fail "site was not re-enabled" + +echo "==> Confirming the private host does not serve anonymous requests" +anonymous_status() { + curl --silent --show-error \ + --output /dev/null \ + --write-out "%{http_code}" \ + "$1" +} +ANONYMOUS_ALIAS_STATUS="$( + anonymous_status "${HOST_URL}/sites/${SITE_SLUG}/" +)" +ANONYMOUS_REVISION_STATUS="$( + anonymous_status "${HOST_URL}/sites/${SITE_SLUG}/revisions/${V2_DIGEST}/" +)" +for status in "$ANONYMOUS_ALIAS_STATUS" "$ANONYMOUS_REVISION_STATUS"; do + case "$status" in + 302|401|403) ;; + *) fail "anonymous host response was not an IAP login or denial: ${status}" ;; + esac +done + +echo "==> Catalog history" +HISTORY_JSON="$(atlas history "$SITE_SLUG")" +printf '%s\n' "$HISTORY_JSON" +printf '%s' "$HISTORY_JSON" >"$FINAL_HISTORY_FILE" +python3 - "$BASELINE_HISTORY_FILE" "$FINAL_HISTORY_FILE" <<'PY' +import json +import sys + +with open(sys.argv[1], encoding="utf-8") as handle: + before_ids = {event["event_id"] for event in json.load(handle)} +with open(sys.argv[2], encoding="utf-8") as handle: + after = json.load(handle) +new_types = [ + event["event_type"] + for event in after + if event["event_id"] not in before_ids +] +required_order = [ + "accepted", + "smoke_passed", + "promoted", + "accepted", + "smoke_passed", + "promoted", + "smoke_passed", + "rollback", + "smoke_passed", + "promoted", + "disabled", + "enabled", +] +cursor = iter(new_types) +for required in required_order: + if not any(actual == required for actual in cursor): + raise SystemExit( + f"new lifecycle history lacks ordered {required!r}: {new_types}" + ) +PY + +cat <&2 + exit 2 + ;; + esac +done + +case "$MODE" in + plan) + "$PY" - <<'PY' +from atlas.ops.migrations import plan_migrations + +plan = plan_migrations() +bad = [e for e in plan if e.state in ("CHECKSUM_MISMATCH", "FAILED_PREVIOUSLY")] +for entry in plan: + print(f" {entry.state:<18} {entry.migration_id} ({entry.checksum[:12]}…)") +pending = sum(1 for e in plan if e.state == "PENDING") +print(f"plan: {pending} pending, {len(plan) - pending - len(bad)} applied, {len(bad)} blocking") +raise SystemExit(1 if bad else 0) +PY + ;; + status) + "$PY" - <<'PY' +import json + +from atlas.ops.migrations import migration_status + +print(json.dumps(migration_status(), indent=2, default=str)) +PY + ;; + apply) + if [[ "${ATLAS_APPROVE_DEPLOY:-false}" != "true" ]]; then + echo "ATLAS_APPROVE_DEPLOY != true — refusing to apply migrations." >&2 + exit 3 + fi + "$PY" - <<'PY' +from atlas.ops.migrations import apply_migrations + +for entry in apply_migrations(): + print(f" {entry.state:<12} {entry.migration_id} ({entry.checksum[:12]}…)") +print("migrations applied") +PY + ;; + *) + echo "Usage: $0 --mode plan|apply|status" >&2 + exit 2 + ;; +esac diff --git a/scripts/atlas_step_runner.py b/scripts/atlas_step_runner.py new file mode 100644 index 0000000..9f6def3 --- /dev/null +++ b/scripts/atlas_step_runner.py @@ -0,0 +1,358 @@ +"""Execute one Atlas orchestration step from JSON context.""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys +import time +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + + +def _ctx() -> dict[str, Any]: + if len(sys.argv) < 3: + raise SystemExit("Usage: atlas_step_runner.py ") + try: + return json.loads(sys.argv[2]) + except json.JSONDecodeError as exc: + raise SystemExit(f"atlas_step_runner: invalid JSON context for step {sys.argv[1]!r}: {exc}") from exc + + +def _run(cmd: list[str], *, cwd: Path | None = None) -> None: + subprocess.check_call(cmd, cwd=cwd) + + +def _attempt_number() -> int: + for var in ("ATLAS_TRY_NUMBER", "AIRFLOW_CTX_TRY_NUMBER"): + raw = os.environ.get(var) + if raw and raw.isdigit(): + return max(1, int(raw)) + return 1 + + +def _record_step_event(step: str, ctx: dict[str, Any], event_type: str, **extra: Any) -> None: + """Best-effort task telemetry (Sprint 5, Phase 3). Never raises.""" + try: + from atlas.observability.logging import emit_event + from atlas.ops.task_events import TaskEventRecord, record_task_event_safely + + record = TaskEventRecord( + pipeline_run_id=ctx.get("pipeline_run_id", "unknown"), + task_id=os.environ.get("AIRFLOW_CTX_TASK_ID") or step, + attempt_number=_attempt_number(), + event_type=event_type, + batch_id=ctx.get("batch_id"), + airflow_run_id=ctx.get("airflow_run_id"), + dag_id=ctx.get("dag_id"), + status=extra.get("status"), + started_at=extra.get("started_at"), + completed_at=extra.get("completed_at"), + duration_ms=extra.get("duration_ms"), + operator_type="BashOperator", + environment=os.environ.get("ATLAS_ENVIRONMENT", "atlas-dev"), + git_sha=os.environ.get("ATLAS_DEPLOYED_GIT_SHA"), + error_type=extra.get("error_type"), + error_message=extra.get("error_message"), + # Sprint 6, Phase 1: runner-measured timing is exact by construction. + timing_source="step_runner_clock" if extra.get("started_at") else None, + timing_confidence="exact" if extra.get("started_at") else None, + ) + record_task_event_safely(record) + emit_event( + f"task_{event_type.lower()}", + severity="ERROR" if event_type == "FAILED" else "INFO", + component="step_runner", + pipeline_run_id=ctx.get("pipeline_run_id"), + batch_id=ctx.get("batch_id"), + airflow_run_id=ctx.get("airflow_run_id"), + dag_id=ctx.get("dag_id"), + task_id=os.environ.get("AIRFLOW_CTX_TASK_ID") or step, + attempt_number=_attempt_number(), + duration_ms=extra.get("duration_ms"), + error_type=extra.get("error_type"), + error_message=extra.get("error_message"), + ) + except Exception: # noqa: BLE001, S110 - telemetry must never break the step + pass + + +def main() -> int: + """Telemetry wrapper: STARTED/terminal task events around the dispatch. + + Telemetry failures never change the step's exit code; the terminal event + mirrors the dispatch outcome (0 -> SUCCESS/SKIPPED, else FAILED). + """ + step = sys.argv[1] + ctx = _ctx() + started_at = datetime.now(tz=UTC).isoformat() + start = time.perf_counter() + _record_step_event(step, ctx, "STARTED", started_at=started_at) + try: + code = _dispatch(step, ctx) + except subprocess.CalledProcessError as exc: + _record_step_event( + step, + ctx, + "FAILED", + status="FAILED", + started_at=started_at, + completed_at=datetime.now(tz=UTC).isoformat(), + duration_ms=int((time.perf_counter() - start) * 1000), + error_type="CalledProcessError", + error_message=f"command exited {exc.returncode}", + ) + raise + except Exception as exc: + _record_step_event( + step, + ctx, + "FAILED", + status="FAILED", + started_at=started_at, + completed_at=datetime.now(tz=UTC).isoformat(), + duration_ms=int((time.perf_counter() - start) * 1000), + error_type=type(exc).__name__, + error_message=str(exc), + ) + raise + skipped = step == "dbt_source_freshness" and bool(ctx.get("backfill_mode")) + terminal = "SKIPPED" if (code == 0 and skipped) else ("SUCCESS" if code == 0 else "FAILED") + _record_step_event( + step, + ctx, + terminal, + status=terminal, + started_at=started_at, + completed_at=datetime.now(tz=UTC).isoformat(), + duration_ms=int((time.perf_counter() - start) * 1000), + error_message=None if code == 0 else f"step returned exit code {code}", + ) + return code + + +def _dispatch(step: str, ctx: dict[str, Any]) -> int: + root = Path(__import__("os").environ.get("ATLAS_ROOT", Path(__file__).resolve().parents[1])) + dbt_dir = Path(__import__("os").environ.get("DBT_PROJECT_DIR", root / "dbt" / "atlas_dbt")) + profiles_dir = __import__("os").environ.get("DBT_PROFILES_DIR", str(Path.home() / ".dbt")) + + if step == "ensure_audit_resources": + from atlas.ops.resources import ensure_audit_resources + + ensure_audit_resources() + print(json.dumps({"status": "PASS"})) + return 0 + + if step == "start_run_audit": + from atlas.ops.audit import start_pipeline_run + + record = start_pipeline_run( + pipeline_run_id=ctx["pipeline_run_id"], + batch_id=ctx["batch_id"], + airflow_run_id=ctx["airflow_run_id"], + dag_id=ctx["dag_id"], + processing_date=ctx["processing_date"], + ) + print(json.dumps({"status": record.status, "pipeline_run_id": record.pipeline_run_id})) + return 0 + + if step == "preflight_environment": + from atlas.ops.preflight import preflight_environment + + result = preflight_environment() + print(json.dumps({"status": result.status, "checks": result.checks})) + return 0 if result.status == "PASS" else 1 + + if step == "generate_events": + cmd = [ + sys.executable, + str(root / "scripts" / "generate_events.py"), + "--processing-date", + ctx["processing_date"], + "--batch-id", + ctx["batch_id"], + "--pipeline-run-id", + ctx["pipeline_run_id"], + ] + if ctx.get("seed") is not None: + cmd.extend(["--seed", str(ctx["seed"])]) + _run(cmd) + return 0 + + if step == "upload_events": + gen = subprocess.check_output( + [ + sys.executable, + str(root / "scripts" / "generate_events.py"), + "--processing-date", + ctx["processing_date"], + "--batch-id", + ctx["batch_id"], + "--pipeline-run-id", + ctx["pipeline_run_id"], + ], + text=True, + ) + checksum = json.loads(gen).get("checksum_sha256") + cmd = [ + sys.executable, + str(root / "scripts" / "upload_events.py"), + "--local-path", + ctx["local_file_path"], + "--event-date", + ctx["processing_date"], + "--run-id", + ctx["pipeline_run_id"], + "--batch-id", + ctx["batch_id"], + ] + if checksum: + cmd.extend(["--expected-checksum", checksum]) + try_number = int( + __import__("os").environ.get( + "ATLAS_TRY_NUMBER", + __import__("os").environ.get("AIRFLOW_CTX_TRY_NUMBER", "1"), + ) + ) + if ctx.get("upload_once") and try_number == 1: + cmd.append("--fail-once") + _run(cmd) + return 0 + + if step == "load_events": + bucket = __import__("os").environ.get("ATLAS_GCS_BUCKET", "atlas-raw-events-example-gcp-project") + object_path = f"raw/event_date={ctx['processing_date']}/batch_id={ctx['batch_id']}/events.jsonl" + _run( + [ + sys.executable, + str(root / "scripts" / "load_events.py"), + "--gcs-uri", + f"gs://{bucket}/{object_path}", + "--run-id", + ctx["pipeline_run_id"], + "--batch-id", + ctx["batch_id"], + "--processing-date", + ctx["processing_date"], + "--expected-row-count", + "50000", + ] + ) + return 0 + + if step == "validate_raw_load": + _run( + [ + sys.executable, + str(root / "scripts" / "validate_events.py"), + "--batch-id", + ctx["batch_id"], + "--event-date", + ctx["processing_date"], + "--processing-date", + ctx["processing_date"], + "--mode", + "airflow", + ] + ) + return 0 + + if step == "dbt_seed": + _run(["dbt", "seed", "--profiles-dir", profiles_dir], cwd=dbt_dir) + return 0 + + if step == "dbt_source_freshness": + if ctx.get("backfill_mode"): + print(json.dumps({"status": "SKIPPED", "reason": "backfill_mode"})) + return 0 + _run(["dbt", "source", "freshness", "--profiles-dir", profiles_dir], cwd=dbt_dir) + return 0 + + if step == "dbt_build": + vars_payload: dict[str, Any] = {"validated_batch_id": ctx["batch_id"]} + if ctx.get("dbt_test_failure"): + vars_payload["inject_failure"] = True + cmd = ["dbt", "build", "--profiles-dir", profiles_dir, "--vars", json.dumps(vars_payload)] + if ctx.get("full_refresh"): + # Sprint 6 cost guard (S6-COST-003): full refresh rebuilds every + # incremental target and is never the default recovery response. + from atlas.observability.cost_guards import require_full_refresh_approval + + require_full_refresh_approval() + cmd.append("--full-refresh") + _run(cmd, cwd=dbt_dir) + return 0 + + if step == "validate_warehouse": + from atlas.observability.checks import persist_warehouse_report + from atlas.validation.warehouse import validate_warehouse + + report = validate_warehouse(ctx["batch_id"], ctx["processing_date"]) + # Durable quality evidence (Sprint 5, Phase 4); persistence problems + # surface as structured telemetry, never as a changed validation verdict. + # External re-validation (e.g. validate_atlas_deployment.sh) passes no + # pipeline_run_id; skip persistence then instead of inventing a run. + if ctx.get("pipeline_run_id"): + persist_warehouse_report( + report, + ctx["pipeline_run_id"], + git_sha=os.environ.get("ATLAS_DEPLOYED_GIT_SHA"), + ) + print(json.dumps(report.to_dict(), indent=2, default=str)) + return 0 if report.overall_status == "PASS" else 1 + + if step == "publish_success_marker": + marker_dir = root / "data" / "runs" / ctx["batch_id"] + marker_dir.mkdir(parents=True, exist_ok=True) + (marker_dir / "success.marker").write_text(datetime.now(tz=UTC).isoformat(), encoding="utf-8") + print(json.dumps({"status": "PASS", "batch_id": ctx["batch_id"]})) + return 0 + + if step == "write_run_summary": + from atlas.ops.audit import ( + PipelineRunRecord, + finalize_pipeline_run, + query_pipeline_run, + write_local_run_summary, + ) + from atlas.ops.finalizer import finalizer_should_fail, reconcile_run_summary + + dag_state = __import__("os").environ.get("AIRFLOW_CTX_DAG_RUN_STATE", "success") + status = "SUCCESS" if dag_state.lower() == "success" else "FAILED" + summary = { + "pipeline_run_id": ctx["pipeline_run_id"], + "batch_id": ctx["batch_id"], + "processing_date": ctx["processing_date"], + "status": status, + "completed_at": datetime.now(tz=UTC).isoformat(), + } + write_local_run_summary(ctx["pipeline_run_id"], summary) + record = PipelineRunRecord( + pipeline_run_id=ctx["pipeline_run_id"], + batch_id=ctx["batch_id"], + airflow_run_id=ctx["airflow_run_id"], + dag_id=ctx["dag_id"], + processing_date=ctx["processing_date"], + started_at=summary["completed_at"], + completed_at=summary["completed_at"], + status=status, + attempt_number=int(__import__("os").environ.get("AIRFLOW_CTX_TRY_NUMBER", "1")), + ) + try: + finalize_pipeline_run(record) + audit_row = query_pipeline_run(ctx["pipeline_run_id"]) + except Exception as exc: # noqa: BLE001 + audit_row = None + summary["audit_error"] = str(exc) + summary["reconciliation"] = reconcile_run_summary(summary, audit_row) + write_local_run_summary(ctx["pipeline_run_id"], summary) + print(json.dumps(summary, indent=2)) + return 1 if finalizer_should_fail(summary) else 0 + + raise SystemExit(f"Unknown step: {step}") + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/bootstrap_gcp.sh b/scripts/bootstrap_gcp.sh new file mode 100755 index 0000000..2ec5d0f --- /dev/null +++ b/scripts/bootstrap_gcp.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Bootstrap Atlas GCP resources in the sandbox project. +# Requires explicit approval because it mutates cloud infrastructure. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$ROOT" + +if [[ "${ATLAS_APPROVE_PROVISION:-}" != "true" ]]; then + echo "Refusing to mutate GCP resources without ATLAS_APPROVE_PROVISION=true" + echo "Review docs/runbook.md, then rerun:" + echo " ATLAS_APPROVE_PROVISION=true bash scripts/bootstrap_gcp.sh" + exit 2 +fi + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-${GCP_PROJECT_ID:-example-gcp-project}}" +LOCATION="${ATLAS_GCP_LOCATION:-US}" +BUCKET="${ATLAS_GCS_BUCKET:-atlas-raw-events-${PROJECT_ID}}" +DATASET="${ATLAS_BQ_DATASET:-atlas_raw}" +TABLE="${ATLAS_BQ_TABLE:-events}" + +if ! command -v gcloud >/dev/null 2>&1; then + echo "error: gcloud is required for bootstrap. Install Google Cloud SDK." + exit 1 +fi + +if ! command -v bq >/dev/null 2>&1; then + echo "error: bq is required for bootstrap. Install Google Cloud SDK." + exit 1 +fi + +echo "==> Ensuring GCS bucket gs://${BUCKET}" +if ! gsutil ls -b "gs://${BUCKET}" >/dev/null 2>&1; then + gsutil mb -p "${PROJECT_ID}" -l "${LOCATION}" "gs://${BUCKET}" +fi + +echo "==> Ensuring BigQuery dataset ${DATASET}" +if ! bq --project_id="${PROJECT_ID}" show "${DATASET}" >/dev/null 2>&1; then + bq --location="${LOCATION}" mk --dataset "${PROJECT_ID}:${DATASET}" +fi + +SQL_FILE="${ROOT}/sql/create_events_table.sql" +RENDERED_SQL="$(sed \ + -e "s/{project_id}/${PROJECT_ID}/g" \ + -e "s/{dataset_id}/${DATASET}/g" \ + -e "s/{table_id}/${TABLE}/g" \ + "${SQL_FILE}")" + +echo "==> Ensuring BigQuery table ${PROJECT_ID}.${DATASET}.${TABLE}" +echo "${RENDERED_SQL}" | bq query --use_legacy_sql=false + +echo "Bootstrap complete." diff --git a/scripts/bootstrap_github_wif.sh b/scripts/bootstrap_github_wif.sh new file mode 100755 index 0000000..6340871 --- /dev/null +++ b/scripts/bootstrap_github_wif.sh @@ -0,0 +1,213 @@ +#!/usr/bin/env bash +# Bootstrap keyless GitHub-to-GCP authentication for Project Atlas (Sprint 4). +# +# GitHub OIDC → Workload Identity Federation → service-account impersonation +# +# Behavior: +# - idempotent: safe to re-run; existing resources are reused +# - prints a full plan first; mutations require ATLAS_APPROVE_IAM=true +# - never creates a service-account key +# - validates the provider configuration after setup +# +# See docs/adr/ADR-009-workload-identity-federation.md. +set -euo pipefail + +: "${ATLAS_GCP_PROJECT_ID:?Set ATLAS_GCP_PROJECT_ID}" +: "${ATLAS_GCP_PROJECT_NUMBER:?Set ATLAS_GCP_PROJECT_NUMBER}" +: "${ATLAS_GITHUB_REPOSITORY:?Set ATLAS_GITHUB_REPOSITORY as owner/repo}" + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +GITHUB_OWNER="${ATLAS_GITHUB_OWNER:-YOUR_GITHUB_OWNER}" +GITHUB_REPO="${ATLAS_GITHUB_REPOSITORY:-YOUR_GITHUB_OWNER/YOUR_REPOSITORY}" +POOL_ID="${ATLAS_WIF_POOL_ID:-atlas-github-pool}" +PROVIDER_ID="${ATLAS_WIF_PROVIDER_ID:-atlas-github-provider}" +INTEGRATION_SA="${ATLAS_INTEGRATION_SA_NAME:-atlas-github-integration}" +DEPLOYER_SA="${ATLAS_DEPLOYER_SA_NAME:-atlas-github-deployer}" +DEPLOYMENT_BUCKET="${ATLAS_DEPLOYMENT_BUCKET:-atlas-deployments-${PROJECT_ID}}" +# Dedicated CI bucket: integration tests never touch the canonical Sprint 1 +# events bucket (which predates uniform bucket-level access, so it cannot +# carry IAM prefix conditions). Objects expire automatically after 7 days. +CI_BUCKET="${ATLAS_CI_BUCKET:-atlas-ci-${PROJECT_ID}}" +APPROVE="${ATLAS_APPROVE_IAM:-false}" + +PROJECT_NUMBER="$(gcloud projects describe "$PROJECT_ID" --format='value(projectNumber)')" +POOL_NAME="projects/${PROJECT_NUMBER}/locations/global/workloadIdentityPools/${POOL_ID}" +PROVIDER_NAME="${POOL_NAME}/providers/${PROVIDER_ID}" +INTEGRATION_EMAIL="${INTEGRATION_SA}@${PROJECT_ID}.iam.gserviceaccount.com" +DEPLOYER_EMAIL="${DEPLOYER_SA}@${PROJECT_ID}.iam.gserviceaccount.com" +REPO_FULL="${GITHUB_OWNER}/${GITHUB_REPO}" + +# Trust boundary: only workflows from this exact repository may authenticate, +# and only runs on refs/heads/main may impersonate either service account. +# assertion.repository_owner guards against repository renames/transfers. +ATTRIBUTE_CONDITION="assertion.repository_owner == '${GITHUB_OWNER}' && assertion.repository == '${REPO_FULL}'" +MAIN_REF_MEMBER="principalSet://iam.googleapis.com/${POOL_NAME}/attribute.repository_and_ref/${REPO_FULL}@refs/heads/main" + +cat </dev/null 2>&1; then + echo "Pool ${POOL_ID} exists — reusing." +else + run gcloud iam workload-identity-pools create "$POOL_ID" \ + --location=global --project="$PROJECT_ID" \ + --display-name="Atlas GitHub Actions pool" +fi + +# --- Provider --------------------------------------------------------------- +if gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --workload-identity-pool="$POOL_ID" --location=global \ + --project="$PROJECT_ID" >/dev/null 2>&1; then + echo "Provider ${PROVIDER_ID} exists — updating condition and mapping." + run gcloud iam workload-identity-pools providers update-oidc "$PROVIDER_ID" \ + --workload-identity-pool="$POOL_ID" --location=global --project="$PROJECT_ID" \ + --attribute-mapping="google.subject=assertion.sub,attribute.repository=assertion.repository,attribute.repository_owner=assertion.repository_owner,attribute.ref=assertion.ref,attribute.repository_and_ref=assertion.repository+'@'+assertion.ref" \ + --attribute-condition="$ATTRIBUTE_CONDITION" +else + run gcloud iam workload-identity-pools providers create-oidc "$PROVIDER_ID" \ + --workload-identity-pool="$POOL_ID" --location=global --project="$PROJECT_ID" \ + --display-name="Atlas GitHub OIDC" \ + --issuer-uri="https://token.actions.githubusercontent.com" \ + --attribute-mapping="google.subject=assertion.sub,attribute.repository=assertion.repository,attribute.repository_owner=assertion.repository_owner,attribute.ref=assertion.ref,attribute.repository_and_ref=assertion.repository+'@'+assertion.ref" \ + --attribute-condition="$ATTRIBUTE_CONDITION" +fi + +# --- Service accounts -------------------------------------------------------- +for sa in "$INTEGRATION_SA" "$DEPLOYER_SA"; do + email="${sa}@${PROJECT_ID}.iam.gserviceaccount.com" + if gcloud iam service-accounts describe "$email" --project="$PROJECT_ID" >/dev/null 2>&1; then + echo "Service account ${email} exists — reusing." + else + run gcloud iam service-accounts create "$sa" --project="$PROJECT_ID" \ + --display-name="Atlas GitHub ${sa#atlas-github-}" + fi +done + +# --- Deployment bucket ------------------------------------------------------- +if gcloud storage buckets describe "gs://${DEPLOYMENT_BUCKET}" >/dev/null 2>&1; then + echo "Bucket gs://${DEPLOYMENT_BUCKET} exists — reusing." +else + run gcloud storage buckets create "gs://${DEPLOYMENT_BUCKET}" \ + --project="$PROJECT_ID" --location=US \ + --uniform-bucket-level-access + run gcloud storage buckets update "gs://${DEPLOYMENT_BUCKET}" --versioning +fi + +# --- Project-level roles ------------------------------------------------------ +grant_project_role() { + local member="$1" role="$2" + run gcloud projects add-iam-policy-binding "$PROJECT_ID" \ + --member="$member" --role="$role" --condition=None --quiet >/dev/null +} + +grant_project_role "serviceAccount:${INTEGRATION_EMAIL}" roles/bigquery.jobUser +grant_project_role "serviceAccount:${INTEGRATION_EMAIL}" roles/bigquery.dataEditor +grant_project_role "serviceAccount:${DEPLOYER_EMAIL}" roles/bigquery.jobUser +grant_project_role "serviceAccount:${DEPLOYER_EMAIL}" roles/bigquery.dataEditor +grant_project_role "serviceAccount:${DEPLOYER_EMAIL}" roles/composer.user +grant_project_role "serviceAccount:${DEPLOYER_EMAIL}" roles/composer.environmentAndStorageObjectAdmin + +# --- CI bucket ----------------------------------------------------------------- +if gcloud storage buckets describe "gs://${CI_BUCKET}" >/dev/null 2>&1; then + echo "Bucket gs://${CI_BUCKET} exists — reusing." +else + run gcloud storage buckets create "gs://${CI_BUCKET}" \ + --project="$PROJECT_ID" --location=US \ + --uniform-bucket-level-access + LIFECYCLE_TMP="$(mktemp)" + cat >"$LIFECYCLE_TMP" <<'JSON' +{"rule": [{"action": {"type": "Delete"}, "condition": {"age": 7}}]} +JSON + run gcloud storage buckets update "gs://${CI_BUCKET}" --lifecycle-file="$LIFECYCLE_TMP" + rm -f "$LIFECYCLE_TMP" +fi + +# --- Bucket-scoped roles ------------------------------------------------------ +# Integration: full control of the dedicated CI bucket only. +run gcloud storage buckets add-iam-policy-binding "gs://${CI_BUCKET}" \ + --member="serviceAccount:${INTEGRATION_EMAIL}" \ + --role="roles/storage.admin" >/dev/null +echo "+ integration storage.admin bound to gs://${CI_BUCKET}" + +# Deployer: full control of the deployment bucket only. +run gcloud storage buckets add-iam-policy-binding "gs://${DEPLOYMENT_BUCKET}" \ + --member="serviceAccount:${DEPLOYER_EMAIL}" \ + --role="roles/storage.admin" >/dev/null +echo "+ deployer storage.admin bound to gs://${DEPLOYMENT_BUCKET}" + +# --- WIF impersonation bindings ---------------------------------------------- +for email in "$INTEGRATION_EMAIL" "$DEPLOYER_EMAIL"; do + run gcloud iam service-accounts add-iam-policy-binding "$email" \ + --project="$PROJECT_ID" \ + --member="$MAIN_REF_MEMBER" \ + --role="roles/iam.workloadIdentityUser" >/dev/null + echo "+ ${email} impersonable by ${REPO_FULL}@refs/heads/main" +done + +# --- Post-setup validation ----------------------------------------------------- +echo "" +echo "=== Validation ===" +gcloud iam workload-identity-pools providers describe "$PROVIDER_ID" \ + --workload-identity-pool="$POOL_ID" --location=global --project="$PROJECT_ID" \ + --format="yaml(name,state,attributeCondition,oidc.issuerUri)" +for email in "$INTEGRATION_EMAIL" "$DEPLOYER_EMAIL"; do + echo "--- ${email} impersonation bindings:" + gcloud iam service-accounts get-iam-policy "$email" --project="$PROJECT_ID" \ + --format="table(bindings.role,bindings.members)" 2>/dev/null | sed 's/^/ /' +done + +cat <&2; exit 1; } + +[[ -f "$FILTER_FILE" ]] || fatal "sink filter file missing: $FILTER_FILE" +# Strip comment lines; the remainder is the actual filter expression. +SINK_FILTER="$(grep -v '^--' "$FILTER_FILE" | sed '/^[[:space:]]*$/d')" +[[ -n "$SINK_FILTER" ]] || fatal "sink filter is empty after stripping comments" + +bucket_exists() { + gcloud logging buckets describe "$BUCKET_ID" --location="$LOCATION" \ + --project="$PROJECT_ID" >/dev/null 2>&1 +} + +sink_exists() { + gcloud logging sinks describe "$SINK_ID" --project="$PROJECT_ID" >/dev/null 2>&1 +} + +view_exists() { + gcloud logging views describe "$VIEW_ID" --bucket="$BUCKET_ID" \ + --location="$LOCATION" --project="$PROJECT_ID" >/dev/null 2>&1 +} + +link_exists() { + gcloud logging links describe "$LINK_ID" --bucket="$BUCKET_ID" \ + --location="$LOCATION" --project="$PROJECT_ID" >/dev/null 2>&1 +} + +print_status() { + log "project=$PROJECT_ID location=$LOCATION" + if bucket_exists; then + log "bucket $BUCKET_ID: EXISTS" + gcloud logging buckets describe "$BUCKET_ID" --location="$LOCATION" \ + --project="$PROJECT_ID" --format='value(retentionDays,analyticsEnabled,lifecycleState)' \ + | awk '{printf "[bootstrap-observability] retentionDays=%s analytics=%s state=%s\n", $1, $2, $3}' + else + log "bucket $BUCKET_ID: MISSING" + fi + if sink_exists; then + log "sink $SINK_ID: EXISTS (writer: $(gcloud logging sinks describe "$SINK_ID" --project="$PROJECT_ID" --format='value(writerIdentity)'))" + else + log "sink $SINK_ID: MISSING" + fi + if view_exists; then log "view $VIEW_ID: EXISTS"; else log "view $VIEW_ID: MISSING"; fi + if link_exists; then log "linked dataset $LINK_ID: EXISTS"; else log "linked dataset $LINK_ID: MISSING"; fi +} + +print_plan() { + log "PLAN (no changes made):" + bucket_exists || log " CREATE log bucket $BUCKET_ID location=$LOCATION retention=${RETENTION_DAYS}d analytics=enabled" + sink_exists || log " CREATE sink $SINK_ID -> logging.googleapis.com/projects/$PROJECT_ID/locations/$LOCATION/buckets/$BUCKET_ID" + sink_exists || log " GRANT roles/logging.bucketWriter to the sink writer identity (requires ATLAS_APPROVE_IAM=true)" + view_exists || log " CREATE view $VIEW_ID on $BUCKET_ID" + link_exists || log " CREATE linked BigQuery dataset $LINK_ID (read-only) from $BUCKET_ID" + log " sink filter: $SINK_FILTER" + dashboard_validate + log " CREATE-OR-UPDATE dashboard 'Atlas Operations' from observability/dashboards/atlas-operations.json" + if bucket_exists && sink_exists && view_exists && link_exists; then + log " logging resources all exist — only dashboard/descriptor sync would run" + fi +} + +apply() { + [[ "${ATLAS_APPROVE_PROVISION:-}" == "true" ]] \ + || fatal "--apply requires ATLAS_APPROVE_PROVISION=true" + + if ! bucket_exists; then + log "creating log bucket $BUCKET_ID" + gcloud logging buckets create "$BUCKET_ID" \ + --location="$LOCATION" \ + --retention-days="$RETENTION_DAYS" \ + --enable-analytics \ + --description="Atlas runtime logs (Sprint 5). Composer + structured Atlas events." \ + --project="$PROJECT_ID" + else + log "bucket $BUCKET_ID already exists — leaving as-is" + fi + + if ! sink_exists; then + log "creating sink $SINK_ID" + gcloud logging sinks create "$SINK_ID" \ + "logging.googleapis.com/projects/$PROJECT_ID/locations/$LOCATION/buckets/$BUCKET_ID" \ + --log-filter="$SINK_FILTER" \ + --description="Routes Atlas runtime logs to the atlas-observability bucket (additive; _Default unaffected)" \ + --project="$PROJECT_ID" + else + log "sink $SINK_ID already exists — leaving as-is" + fi + + # Sinks writing to a log bucket in the same project usually need no extra + # grant, but we verify and grant explicitly so routing cannot fail silently. + local writer + writer="$(gcloud logging sinks describe "$SINK_ID" --project="$PROJECT_ID" --format='value(writerIdentity)')" + if [[ -n "$writer" ]]; then + if gcloud projects get-iam-policy "$PROJECT_ID" \ + --flatten='bindings[].members' \ + --filter="bindings.role=roles/logging.bucketWriter AND bindings.members=$writer" \ + --format='value(bindings.role)' | grep -q .; then + log "sink writer $writer already has roles/logging.bucketWriter" + elif [[ "${ATLAS_APPROVE_IAM:-}" == "true" ]]; then + log "granting roles/logging.bucketWriter to $writer" + gcloud projects add-iam-policy-binding "$PROJECT_ID" \ + --member="$writer" --role='roles/logging.bucketWriter' \ + --condition=None --format='none' + else + log "WARNING: sink writer $writer lacks roles/logging.bucketWriter and ATLAS_APPROVE_IAM!=true — routing may fail" + fi + fi + + if ! view_exists; then + log "creating view $VIEW_ID" + gcloud logging views create "$VIEW_ID" \ + --bucket="$BUCKET_ID" --location="$LOCATION" \ + --log-filter="SOURCE(\"projects/$PROJECT_ID\")" \ + --description="Least-privilege Atlas runtime view (grant roles/logging.viewAccessor here)" \ + --project="$PROJECT_ID" + else + log "view $VIEW_ID already exists — leaving as-is" + fi + + if ! link_exists; then + log "creating linked BigQuery dataset $LINK_ID (read-only)" + gcloud logging links create "$LINK_ID" \ + --bucket="$BUCKET_ID" --location="$LOCATION" \ + --description="Read-only linked dataset over the atlas-observability log bucket" \ + --project="$PROJECT_ID" + else + log "linked dataset $LINK_ID already exists — leaving as-is" + fi + + # 6. Custom metric descriptors from the versioned catalog (idempotent). + if python3 -c 'import google.cloud.monitoring_v3' 2>/dev/null; then + log "ensuring Atlas metric descriptors (observability/metrics/metric-descriptors.json)" + PYTHONPATH="${SCRIPT_DIR}/../src${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m atlas.observability.metrics --ensure-descriptors --project-id "$PROJECT_ID" + else + log "WARNING: google-cloud-monitoring not installed — skipping metric descriptors (run 'python -m atlas.observability.metrics --ensure-descriptors' from an environment that has it)" + fi + + apply_dashboard + + log "apply complete" + print_status +} + +DASHBOARD_FILE="${SCRIPT_DIR}/../observability/dashboards/atlas-operations.json" + +dashboard_validate() { + python3 -c " +import json, sys +d = json.load(open('$DASHBOARD_FILE')) +assert d.get('displayName') == 'Atlas Operations', 'unexpected displayName' +tiles = d['mosaicLayout']['tiles'] +assert len(tiles) >= 20, 'dashboard suspiciously small' +print(f'[bootstrap-observability] dashboard JSON valid: {len(tiles)} tiles') +" +} + +apply_dashboard() { + dashboard_validate + local token existing_name + token="$(gcloud auth print-access-token)" + existing_name="$(curl -sf -H "Authorization: Bearer $token" \ + "https://monitoring.googleapis.com/v1/projects/$PROJECT_ID/dashboards" \ + | python3 -c "import json,sys; ds=json.load(sys.stdin).get('dashboards',[]); print(next((d['name'] for d in ds if d.get('displayName')=='Atlas Operations'), ''))")" + if [[ -n "$existing_name" ]]; then + log "updating existing dashboard $existing_name" + # PATCH requires etag; fetch, merge repo definition over live identity fields. + curl -sf -H "Authorization: Bearer $token" \ + "https://monitoring.googleapis.com/v1/$existing_name" > /tmp/atlas-dashboard-live.json + python3 - "$DASHBOARD_FILE" /tmp/atlas-dashboard-live.json > /tmp/atlas-dashboard-merged.json <<'PYEOF' +import json, sys +repo = json.load(open(sys.argv[1])) +live = json.load(open(sys.argv[2])) +repo["name"] = live["name"] +repo["etag"] = live["etag"] +print(json.dumps(repo)) +PYEOF + curl -sf -X PATCH -H "Authorization: Bearer $token" -H "Content-Type: application/json" \ + -d @/tmp/atlas-dashboard-merged.json \ + "https://monitoring.googleapis.com/v1/$existing_name" > /dev/null + log "dashboard updated" + else + log "creating dashboard 'Atlas Operations'" + curl -sf -X POST -H "Authorization: Bearer $token" -H "Content-Type: application/json" \ + -d @"$DASHBOARD_FILE" \ + "https://monitoring.googleapis.com/v1/projects/$PROJECT_ID/dashboards" \ + | python3 -c "import json,sys; print('[bootstrap-observability] created:', json.load(sys.stdin)['name'])" + fi +} + +case "$MODE" in + --plan) print_plan ;; + --apply) apply ;; + --status) print_status ;; + *) fatal "unknown mode: $MODE (use --plan | --apply | --status)" ;; +esac diff --git a/scripts/build_deployment_bundle.sh b/scripts/build_deployment_bundle.sh new file mode 100755 index 0000000..c1a56c4 --- /dev/null +++ b/scripts/build_deployment_bundle.sh @@ -0,0 +1,227 @@ +#!/usr/bin/env bash +# Build an immutable, checksum-verified Atlas deployment bundle (Sprint 4, Phase 8). +# +# Usage: +# build_deployment_bundle.sh [--upload] [--environment atlas-dev] +# +# Produces dist/atlas-bundle-.tar.gz + release-manifest.json. +# With --upload, stores the bundle create-only under +# gs:///atlas/releases// +# Reuse is allowed only on exact checksum match; a different checksum for an +# existing release path is a hard failure (never overwrite). +# +# Determinism note: tar/gzip metadata is normalized (sorted names, fixed +# mtime, gzip -n), so archive bytes depend only on file content. The manifest +# intentionally embeds build metadata (timestamp, builder, workflow run), so +# rebuilding the same git SHA in a new context yields a different checksum. +# Deployment tooling must therefore REUSE an existing stored release for a SHA +# instead of rebuilding it; the create-only check above enforces this. +set -euo pipefail + +cd "$(dirname "${BASH_SOURCE[0]}")/.." || exit 1 +ATLAS_DIR="$(pwd)" +PY="${ATLAS_PYTHON:-python3}" +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +DEPLOYMENT_BUCKET="${ATLAS_DEPLOYMENT_BUCKET:-atlas-deployments-${PROJECT_ID}}" +ENVIRONMENT="atlas-dev" +UPLOAD=0 + +while [[ $# -gt 0 ]]; do + case "$1" in + --upload) UPLOAD=1; shift ;; + --environment) ENVIRONMENT="${2:?}"; shift 2 ;; + *) echo "Unknown argument: $1" >&2; exit 2 ;; + esac +done + +GIT_SHA="$(git rev-parse HEAD)" +GIT_REF="$(git rev-parse --abbrev-ref HEAD)" +RELEASE_TAG="$(git describe --tags --exact-match 2>/dev/null || echo "")" +if [[ -n "$(git status --porcelain -- project-atlas 2>/dev/null || true)" ]]; then + echo "WARNING: working tree has uncommitted project-atlas changes; bundle records HEAD ${GIT_SHA:0:12}" >&2 +fi + +DIST_DIR="${ATLAS_DIR}/dist" +STAGE_DIR="$(mktemp -d)" +BUNDLE_ROOT="${STAGE_DIR}/atlas-bundle" +mkdir -p "$DIST_DIR" "$BUNDLE_ROOT" +trap 'rm -rf "$STAGE_DIR"' EXIT + +# --- Stage runtime assets only --------------------------------------------------- +copy() { # src dest-subdir + local src="$1" dest="${BUNDLE_ROOT}/$2" + mkdir -p "$(dirname "$dest")" + cp -r "$src" "$dest" +} + +copy dags dags +copy src/atlas src/atlas +copy config config +copy sql sql +copy dbt/atlas_dbt dbt/atlas_dbt +copy scripts/atlas_step_runner.py scripts/atlas_step_runner.py +copy scripts/run_atlas_step.sh scripts/run_atlas_step.sh +copy scripts/generate_events.py scripts/generate_events.py +copy scripts/upload_events.py scripts/upload_events.py +copy scripts/load_events.py scripts/load_events.py +copy scripts/validate_events.py scripts/validate_events.py +copy scripts/apply_atlas_migrations.sh scripts/apply_atlas_migrations.sh +# Sprint 5 observability runtime assets: metric catalog and schema manifest +# are loaded at runtime by atlas.observability.{metrics,schema_drift}. +copy observability/metrics observability/metrics +copy observability/schema observability/schema +# Runtime dbt profile: keyless oauth via the environment's service account. +mkdir -p "${BUNDLE_ROOT}/dbt/profiles" +cp dbt/atlas_dbt/profiles.yml.example "${BUNDLE_ROOT}/dbt/profiles/profiles.yml" +copy requirements.txt requirements.txt +copy airflow/requirements-airflow.txt airflow/requirements-airflow.txt +copy dbt/requirements-dbt.txt dbt/requirements-dbt.txt + +# Strip anything that must never ship: caches, local state, dbt build outputs. +find "$BUNDLE_ROOT" \( -name '__pycache__' -o -name '.pytest_cache' -o -name '.mypy_cache' \) \ + -type d -prune -exec rm -rf {} + +rm -rf "$BUNDLE_ROOT/dbt/atlas_dbt/target" "$BUNDLE_ROOT/dbt/atlas_dbt/logs" \ + "$BUNDLE_ROOT/dbt/atlas_dbt/dbt_packages" +find "$BUNDLE_ROOT" -name '*.pyc' -delete + +# Vendor pinned dbt packages so the bundle is self-contained at runtime. +# Composer workers must never resolve packages from the network; deployment +# atlas-dev-20260718T233128Z-74732eee failed at dbt_seed because dbt_packages +# was stripped and no runtime `dbt deps` exists by design. package-lock.yml +# pins exact versions, keeping the vendored tree reproducible. +if ! command -v dbt >/dev/null 2>&1; then + echo "FATAL: dbt CLI required to vendor dbt_packages into the bundle" >&2 + exit 1 +fi +( + cd "$BUNDLE_ROOT/dbt/atlas_dbt" + ATLAS_GCP_PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-bundle-build-placeholder}" \ + ATLAS_DBT_DATASET="${ATLAS_DBT_DATASET:-atlas_dbt}" \ + DBT_TARGET_PATH=/tmp/atlas-bundle-dbt-target DBT_LOG_PATH=/tmp/atlas-bundle-dbt-logs \ + dbt deps --profiles-dir ../profiles >/dev/null +) +rm -rf /tmp/atlas-bundle-dbt-target /tmp/atlas-bundle-dbt-logs +find "$BUNDLE_ROOT/dbt/atlas_dbt/dbt_packages" \( -name '__pycache__' -o -name '.git' \) \ + -prune -exec rm -rf {} + 2>/dev/null || true +if [[ ! -d "$BUNDLE_ROOT/dbt/atlas_dbt/dbt_packages/dbt_utils" ]]; then + echo "FATAL: dbt_packages/dbt_utils missing after vendoring" >&2 + exit 1 +fi + +# Refuse to bundle anything that resembles real credential material. Patterns +# target actual PEM blocks and populated key fields, not detector source code +# that merely mentions the field names. +if grep -rlE -- '-----BEGIN [A-Z ]*PRIVATE KEY-----|"private_key"[[:space:]]*:[[:space:]]*"[^"]+"' \ + "$BUNDLE_ROOT" >/dev/null 2>&1; then + grep -rlE -- '-----BEGIN [A-Z ]*PRIVATE KEY-----|"private_key"[[:space:]]*:[[:space:]]*"[^"]+"' "$BUNDLE_ROOT" >&2 + echo "FATAL: credential-like content detected in bundle staging; aborting." >&2 + exit 1 +fi + +# --- Release manifest ------------------------------------------------------------- +export BUNDLE_ROOT GIT_SHA GIT_REF RELEASE_TAG ENVIRONMENT +"$PY" - <<'PY' +import hashlib +import json +import os +import re +import subprocess +from datetime import UTC, datetime +from pathlib import Path + +bundle_root = Path(os.environ["BUNDLE_ROOT"]) + +def pin(path: str, name: str) -> str | None: + text = Path(path).read_text(encoding="utf-8") + m = re.search(rf"^{re.escape(name)}==(\S+)", text, re.MULTILINE) + return m.group(1) if m else None + +files = sorted(p for p in bundle_root.rglob("*") if p.is_file()) +checksums = { + str(p.relative_to(bundle_root)): hashlib.sha256(p.read_bytes()).hexdigest() for p in files +} + +manifest_lines = [ + line.split("|")[0].strip() + for line in (bundle_root / "sql/migrations/manifest.txt").read_text(encoding="utf-8").splitlines() + if line.strip() and not line.startswith("#") +] + +manifest = { + "git_sha": os.environ["GIT_SHA"], + "git_ref": os.environ["GIT_REF"], + "release_tag": os.environ.get("RELEASE_TAG") or None, + "build_timestamp": datetime.now(tz=UTC).isoformat(), + "builder": os.environ.get("GITHUB_ACTOR") or os.environ.get("USER") or "unknown", + "workflow_run_id": os.environ.get("GITHUB_RUN_ID"), + "python_version": subprocess.check_output(["python3", "--version"], text=True).strip(), + "airflow_version": pin("airflow/requirements-airflow.txt", "apache-airflow"), + "provider_versions": { + "apache-airflow-providers-google": pin( + "airflow/requirements-airflow.txt", "apache-airflow-providers-google" + ), + "apache-airflow-providers-standard": pin( + "airflow/requirements-airflow.txt", "apache-airflow-providers-standard" + ), + }, + "dbt_versions": { + "dbt-core": pin("dbt/requirements-dbt.txt", "dbt-core"), + "dbt-bigquery": pin("dbt/requirements-dbt.txt", "dbt-bigquery"), + }, + "deployment_environment": os.environ["ENVIRONMENT"], + # Schema contract: the newest migration this release requires, and the + # oldest applied-schema state it can run against (rollback compatibility). + "required_schema_version": manifest_lines[-1], + "min_compatible_schema_version": manifest_lines[-1], + "file_count": len(checksums), + "file_checksums": checksums, +} +out = bundle_root / "release-manifest.json" +out.write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n", encoding="utf-8") +print(f"release-manifest.json: {len(checksums)} files, schema {manifest['required_schema_version']}") +PY + +# --- Deterministic archive --------------------------------------------------------- +ARCHIVE="${DIST_DIR}/atlas-bundle-${GIT_SHA}.tar.gz" +tar --sort=name --owner=0 --group=0 --numeric-owner \ + --mtime="UTC 2026-01-01" \ + -C "$STAGE_DIR" -cf - atlas-bundle | gzip -n >"$ARCHIVE" +CHECKSUM="$(sha256sum "$ARCHIVE" | awk '{print $1}')" +echo "$CHECKSUM $(basename "$ARCHIVE")" >"${ARCHIVE}.sha256" +cp "${BUNDLE_ROOT}/release-manifest.json" "${DIST_DIR}/release-manifest-${GIT_SHA}.json" + +echo "bundle: ${ARCHIVE}" +echo "checksum: ${CHECKSUM}" + +# --- Create-only upload ------------------------------------------------------------- +# Content identity is the per-file checksum map: a stored release for this SHA +# with identical file contents is reused (build metadata may differ across +# legitimate retries); different file contents for the same SHA is a hard fail. +if [[ "$UPLOAD" -eq 1 ]]; then + RELEASE_URI="gs://${DEPLOYMENT_BUCKET}/atlas/releases/${GIT_SHA}" + if gcloud storage ls "${RELEASE_URI}/release-manifest.json" >/dev/null 2>&1; then + gcloud storage cat "${RELEASE_URI}/release-manifest.json" >"${STAGE_DIR}/existing-manifest.json" + if "$PY" - "$BUNDLE_ROOT/release-manifest.json" "${STAGE_DIR}/existing-manifest.json" <<'PY' +import json +import sys + +ours = json.load(open(sys.argv[1]))["file_checksums"] +theirs = json.load(open(sys.argv[2]))["file_checksums"] +ours.pop("release-manifest.json", None) +theirs.pop("release-manifest.json", None) +sys.exit(0 if ours == theirs else 1) +PY + then + echo "release ${GIT_SHA:0:12} already stored with identical content — reusing." + echo "uri: ${RELEASE_URI}/atlas-bundle.tar.gz" + exit 0 + fi + echo "FATAL: ${RELEASE_URI} exists with DIFFERENT file contents for the same git SHA." >&2 + echo "Immutable releases are never overwritten. Investigate before retrying." >&2 + exit 1 + fi + gcloud storage cp "$ARCHIVE" "${RELEASE_URI}/atlas-bundle.tar.gz" + gcloud storage cp "${ARCHIVE}.sha256" "${RELEASE_URI}/atlas-bundle.tar.gz.sha256" + gcloud storage cp "${BUNDLE_ROOT}/release-manifest.json" "${RELEASE_URI}/release-manifest.json" + echo "uploaded: ${RELEASE_URI}/atlas-bundle.tar.gz" +fi diff --git a/scripts/deploy_atlas_release.sh b/scripts/deploy_atlas_release.sh new file mode 100755 index 0000000..11bfefd --- /dev/null +++ b/scripts/deploy_atlas_release.sh @@ -0,0 +1,186 @@ +#!/usr/bin/env bash +# Controlled Atlas deployment to managed Composer (Sprint 4, Phase 11). +# +# Usage: +# deploy_atlas_release.sh --git-sha [--deployment-id ] +# [--deployment-type deploy|rollback] [--previous-git-sha ] +# [--leave-paused] [--skip-migrations] +# +# Requires ATLAS_APPROVE_DEPLOY=true. Stages (audited in atlas_ops.deployments): +# fetch_release → schema_check → migrations → promote → dag_parse → +# smoke_batch → smoke_validation → finalize +# A failure at any stage records FAILED (or ROLLBACK_FAILED) with the stage +# name and leaves the immutable release evidence intact. +set -uo pipefail + +ATLAS_SCRIPTS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=lib_atlas_deploy.sh +source "${ATLAS_SCRIPTS_ROOT}/lib_atlas_deploy.sh" +export PYTHONPATH="${ATLAS_SCRIPTS_ROOT}/../src:${PYTHONPATH:-}" + +GIT_SHA="" +DEPLOYMENT_ID="" +DEPLOYMENT_TYPE="deploy" +PREVIOUS_SHA="" +LEAVE_PAUSED=0 +SKIP_MIGRATIONS=0 +while [[ $# -gt 0 ]]; do + case "$1" in + --git-sha) GIT_SHA="${2:?}"; shift 2 ;; + --deployment-id) DEPLOYMENT_ID="${2:?}"; shift 2 ;; + --deployment-type) DEPLOYMENT_TYPE="${2:?}"; shift 2 ;; + --previous-git-sha) PREVIOUS_SHA="${2:?}"; shift 2 ;; + --leave-paused) LEAVE_PAUSED=1; shift ;; + --skip-migrations) SKIP_MIGRATIONS=1; shift ;; + *) echo "Unknown argument: $1" >&2; exit 2 ;; + esac +done +[[ -n "$GIT_SHA" ]] || { echo "--git-sha is required" >&2; exit 2; } + +if [[ "${ATLAS_APPROVE_DEPLOY:-false}" != "true" ]]; then + echo "ATLAS_APPROVE_DEPLOY != true — refusing to deploy." >&2 + exit 3 +fi + +SHORT_SHA="${GIT_SHA:0:8}" +RUN_TOKEN="${GITHUB_RUN_ID:-local$(date -u +%s)}" +DEPLOYMENT_ID="${DEPLOYMENT_ID:-atlas-dev-$(date -u +%Y%m%dT%H%M%SZ)-${SHORT_SHA}}" +SMOKE_BATCH_ID="atlas-smoke-${SHORT_SHA}-${RUN_TOKEN}" +SMOKE_PIPELINE_RUN_ID="${SMOKE_BATCH_ID}-run" +SMOKE_DAG_RUN_ID="smoke__${DEPLOYMENT_ID}" +PROCESSING_DATE="$(date -u +%F)" +WORK_DIR="$(mktemp -d)" +trap 'rm -rf "$WORK_DIR"' EXIT + +echo "=== Atlas ${DEPLOYMENT_TYPE}: ${GIT_SHA} → ${ATLAS_COMPOSER_ENV} (${ATLAS_REGION}) ===" +echo "deployment_id: ${DEPLOYMENT_ID}" +echo "smoke batch: ${SMOKE_BATCH_ID}" + +# --- Audit helpers ---------------------------------------------------------------- +audit() { # status [failure_stage] [error_summary] + ATLAS_AUDIT_STATUS="$1" ATLAS_AUDIT_STAGE="${2:-}" ATLAS_AUDIT_ERROR="${3:-}" \ + ATLAS_DEPLOYMENT_ID="$DEPLOYMENT_ID" ATLAS_GIT_SHA="$GIT_SHA" \ + ATLAS_DEPLOYMENT_TYPE="$DEPLOYMENT_TYPE" ATLAS_PREVIOUS_SHA="$PREVIOUS_SHA" \ + ATLAS_SMOKE_RUN_ID="$SMOKE_PIPELINE_RUN_ID" ATLAS_ARTIFACT_URI="${ARTIFACT_URI:-}" \ + ATLAS_ARTIFACT_CHECKSUM="${ARTIFACT_CHECKSUM:-}" ATLAS_MIGRATION_COUNT="${MIGRATION_COUNT:-}" \ + python3 - <<'PY' +import os + +from atlas.ops.deployments import DeploymentRecord, upsert_deployment +from datetime import UTC, datetime + +status = os.environ["ATLAS_AUDIT_STATUS"] +started = os.environ.get("ATLAS_DEPLOY_STARTED_AT") or datetime.now(tz=UTC).isoformat() +terminal = status in {"SUCCESS", "FAILED", "ROLLED_BACK", "ROLLBACK_FAILED"} +record = DeploymentRecord( + deployment_id=os.environ["ATLAS_DEPLOYMENT_ID"], + git_sha=os.environ["ATLAS_GIT_SHA"], + environment="atlas-dev", + deployment_type=os.environ["ATLAS_DEPLOYMENT_TYPE"], + started_at=started, + status=status, + git_ref=os.environ.get("GITHUB_REF"), + workflow_run_id=os.environ.get("GITHUB_RUN_ID"), + actor=os.environ.get("GITHUB_ACTOR") or os.environ.get("USER"), + completed_at=datetime.now(tz=UTC).isoformat() if terminal else None, + artifact_uri=os.environ.get("ATLAS_ARTIFACT_URI") or None, + artifact_checksum=os.environ.get("ATLAS_ARTIFACT_CHECKSUM") or None, + composer_environment=os.environ.get("ATLAS_COMPOSER_ENV", "atlas-dev"), + composer_region=os.environ.get("ATLAS_COMPOSER_REGION", "us-central1"), + smoke_pipeline_run_id=os.environ["ATLAS_SMOKE_RUN_ID"] if terminal else None, + previous_git_sha=os.environ.get("ATLAS_PREVIOUS_SHA") or None, + migration_count=int(os.environ["ATLAS_MIGRATION_COUNT"]) if os.environ.get("ATLAS_MIGRATION_COUNT") else None, + failure_stage=os.environ.get("ATLAS_AUDIT_STAGE") or None, + error_type="DeploymentStageFailure" if os.environ.get("ATLAS_AUDIT_STAGE") else None, + error_summary=os.environ.get("ATLAS_AUDIT_ERROR") or None, +) +upsert_deployment(record) +print(f"audit: {record.deployment_id} -> {status}" + + (f" (stage {record.failure_stage})" if record.failure_stage else "")) +PY +} + +FAIL_STATUS="FAILED" +[[ "$DEPLOYMENT_TYPE" == "rollback" ]] && FAIL_STATUS="ROLLBACK_FAILED" + +fail_stage() { # stage message + echo "STAGE FAILED: $1 — $2" >&2 + audit "$FAIL_STATUS" "$1" "$2" || echo "WARNING: failed to record audit row" >&2 + echo "Recovery: inspect logs above, then re-run this script with the same" >&2 + echo " --git-sha ${GIT_SHA} (deployment records are idempotent per deployment_id)" >&2 + exit 1 +} + +export ATLAS_DEPLOY_STARTED_AT +ATLAS_DEPLOY_STARTED_AT="$(date -u +%Y-%m-%dT%H:%M:%S+00:00)" + +# --- Stage 1: fetch and verify immutable release ------------------------------------ +ARTIFACT_URI="gs://${ATLAS_DEPLOY_BUCKET}/atlas/releases/${GIT_SHA}/atlas-bundle.tar.gz" +if ! fetch_and_verify_release "$GIT_SHA" "$WORK_DIR"; then + audit "$( [[ "$DEPLOYMENT_TYPE" == "rollback" ]] && echo ROLLING_BACK || echo RUNNING )" || true + fail_stage "fetch_release" "release bundle missing or checksum-invalid for ${GIT_SHA}" +fi +BUNDLE_DIR="$FETCHED_BUNDLE_DIR" +ARTIFACT_CHECKSUM="$(awk '{print $1}' "${WORK_DIR}/atlas-bundle.tar.gz.sha256")" +INITIAL_STATUS="RUNNING" +[[ "$DEPLOYMENT_TYPE" == "rollback" ]] && INITIAL_STATUS="ROLLING_BACK" +audit "$INITIAL_STATUS" || fail_stage "start_audit" "unable to write atlas_ops.deployments" + +# --- Stage 2: schema compatibility ---------------------------------------------------- +SCHEMA_RESULT="$(check_schema_compatibility "$BUNDLE_DIR" "$DEPLOYMENT_TYPE" | tee /dev/stderr | tail -1)" \ + || fail_stage "schema_check" "schema compatibility evaluation failed" +if [[ "$SCHEMA_RESULT" == "PENDING_MIGRATIONS" && "$DEPLOYMENT_TYPE" == "rollback" ]]; then + fail_stage "schema_check" "rollback target requires unapplied migrations — incompatible" +fi +if [[ "$SCHEMA_RESULT" == "ROLLBACK_INCOMPATIBLE" ]]; then + # S6-RBK-003: the applied schema crossed a breaking-migration boundary the + # target release predates. Never reversed automatically — forward fix only. + fail_stage "schema_check" "rollback blocked by breaking migration boundary — recover forward" +fi + +# --- Stage 3: additive migrations (deploy only) ---------------------------------------- +MIGRATION_COUNT=0 +if [[ "$SKIP_MIGRATIONS" -eq 0 && "$DEPLOYMENT_TYPE" == "deploy" ]]; then + MIGRATION_OUT="$(bash "${BUNDLE_DIR}/scripts/apply_atlas_migrations.sh" --mode apply)" \ + || fail_stage "migrations" "migration apply failed (ledger records the failing id)" + echo "$MIGRATION_OUT" + MIGRATION_COUNT="$(echo "$MIGRATION_OUT" | grep -c "APPLIED_NOW" || true)" +fi + +# --- Stage 4: promote to Composer ------------------------------------------------------- +promote_release_to_composer "$BUNDLE_DIR" "$DEPLOYMENT_ID" "$GIT_SHA" \ + || fail_stage "promote" "asset promotion to Composer bucket failed" + +# --- Stage 5: DAG parse verification ------------------------------------------------------ +wait_for_dag_parse 600 || fail_stage "dag_parse" "DAG failed to parse after promotion" + +# --- Stage 6: smoke batch ------------------------------------------------------------------- +SMOKE_CONF="$(printf '{"batch_id": "%s", "pipeline_run_id": "%s", "processing_date": "%s"}' \ + "$SMOKE_BATCH_ID" "$SMOKE_PIPELINE_RUN_ID" "$PROCESSING_DATE")" +run_smoke_batch "$SMOKE_DAG_RUN_ID" "$SMOKE_CONF" 2400 \ + || fail_stage "smoke_batch" "smoke run did not reach terminal SUCCESS" + +# --- Stage 7: smoke validation --------------------------------------------------------------- +bash "${ATLAS_SCRIPTS_ROOT}/validate_atlas_deployment.sh" \ + --git-sha "$GIT_SHA" \ + --deployment-id "$DEPLOYMENT_ID" \ + --batch-id "$SMOKE_BATCH_ID" \ + --pipeline-run-id "$SMOKE_PIPELINE_RUN_ID" \ + --processing-date "$PROCESSING_DATE" \ + || fail_stage "smoke_validation" "post-deployment smoke validation failed" + +# --- Stage 8: finalize ------------------------------------------------------------------------ +if [[ "$LEAVE_PAUSED" -eq 1 ]]; then + composer_airflow dags pause atlas_batch_pipeline >/dev/null 2>&1 || true + echo "DAG left paused per request." +fi +FINAL_STATUS="SUCCESS" +[[ "$DEPLOYMENT_TYPE" == "rollback" ]] && FINAL_STATUS="ROLLED_BACK" +audit "$FINAL_STATUS" || fail_stage "finalize" "unable to finalize atlas_ops.deployments" + +echo "" +echo "=== ${DEPLOYMENT_TYPE} ${FINAL_STATUS}: ${GIT_SHA} ===" +echo "deployment_id: ${DEPLOYMENT_ID}" +echo "artifact: ${ARTIFACT_URI}" +echo "artifact checksum: ${ARTIFACT_CHECKSUM}" +echo "smoke run: ${SMOKE_PIPELINE_RUN_ID}" diff --git a/scripts/generate_events.py b/scripts/generate_events.py new file mode 100755 index 0000000..9cade4b --- /dev/null +++ b/scripts/generate_events.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +"""Generate synthetic Atlas events.""" + +from __future__ import annotations + +import argparse +import json +import sys +from datetime import date +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from atlas.config.settings import load_settings +from atlas.generator.events import generate_events, generate_events_for_batch +from atlas.logging.structured import new_pipeline_run_id + + +def main() -> int: + parser = argparse.ArgumentParser(description="Generate Atlas synthetic events") + parser.add_argument("--processing-date", default=date.today().isoformat()) + parser.add_argument("--batch-id") + parser.add_argument("--pipeline-run-id", default=new_pipeline_run_id()) + parser.add_argument("--seed", type=int) + parser.add_argument("--output-path", type=Path) + args = parser.parse_args() + + settings = load_settings() + if args.batch_id: + result = generate_events_for_batch( + settings, + processing_date=args.processing_date, + batch_id=args.batch_id, + pipeline_run_id=args.pipeline_run_id, + seed=args.seed, + output_path=args.output_path, + ) + else: + result = generate_events(settings) + + print( + json.dumps( + { + "output_path": str(result.output_path), + "event_count": result.event_count, + "anomaly_counts": result.anomaly_counts, + "primary_event_date": result.primary_event_date, + "batch_id": result.batch_id, + "pipeline_run_id": result.pipeline_run_id, + "seed": result.seed, + "reused_existing": result.reused_existing, + "checksum_sha256": result.checksum_sha256, + }, + indent=2, + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/lib_atlas_deploy.sh b/scripts/lib_atlas_deploy.sh new file mode 100644 index 0000000..dec5476 --- /dev/null +++ b/scripts/lib_atlas_deploy.sh @@ -0,0 +1,232 @@ +#!/usr/bin/env bash +# Shared functions for Atlas Composer deployment and rollback (Sprint 4). +# Sourced by deploy_atlas_release.sh, rollback_atlas.sh, and +# validate_atlas_deployment.sh — not executable on its own. + +ATLAS_PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +ATLAS_REGION="${ATLAS_COMPOSER_REGION:-us-central1}" +ATLAS_COMPOSER_ENV="${ATLAS_COMPOSER_ENV:-atlas-dev}" +ATLAS_DEPLOY_BUCKET="${ATLAS_DEPLOYMENT_BUCKET:-atlas-deployments-${ATLAS_PROJECT_ID}}" + +# Run an Airflow CLI command inside the Composer environment. +# gcloud mixes kubectl noise into the stream; callers parse defensively. +composer_airflow() { # subcommand args... + local sub="$1" + shift + gcloud composer environments run "$ATLAS_COMPOSER_ENV" \ + --project="$ATLAS_PROJECT_ID" --location="$ATLAS_REGION" \ + "$sub" -- "$@" 2>&1 +} + +composer_bucket() { + local dag_prefix + dag_prefix="$(gcloud composer environments describe "$ATLAS_COMPOSER_ENV" \ + --project="$ATLAS_PROJECT_ID" --location="$ATLAS_REGION" \ + --format='value(config.dagGcsPrefix)')" + # dagGcsPrefix looks like gs:///dags + echo "${dag_prefix%/dags}" +} + +# Download a stored immutable release and verify archive + per-file checksums. +# Sets FETCHED_BUNDLE_DIR to the extracted atlas-bundle directory. +fetch_and_verify_release() { # git_sha work_dir + local git_sha="$1" work_dir="$2" + local release_uri="gs://${ATLAS_DEPLOY_BUCKET}/atlas/releases/${git_sha}" + echo "Fetching release ${release_uri}" >&2 + gcloud storage cp "${release_uri}/atlas-bundle.tar.gz" "${work_dir}/atlas-bundle.tar.gz" >&2 + gcloud storage cp "${release_uri}/atlas-bundle.tar.gz.sha256" "${work_dir}/atlas-bundle.tar.gz.sha256" >&2 + # Compare digests directly: the stored .sha256 records the builder's local + # filename, which differs from the canonical stored object name. + local expected actual + expected="$(awk '{print $1}' "${work_dir}/atlas-bundle.tar.gz.sha256")" + actual="$(sha256sum "${work_dir}/atlas-bundle.tar.gz" | awk '{print $1}')" + if [[ -z "$expected" || "$expected" != "$actual" ]]; then + echo "FATAL: archive checksum mismatch for release ${git_sha}" >&2 + echo " expected ${expected:-}" >&2 + echo " actual ${actual}" >&2 + return 1 + fi + echo "archive checksum verified: ${actual}" >&2 + tar -xzf "${work_dir}/atlas-bundle.tar.gz" -C "$work_dir" + local bundle_dir="${work_dir}/atlas-bundle" + python3 - "$bundle_dir" <<'PY' >&2 || return 1 +import hashlib +import json +import sys +from pathlib import Path + +bundle = Path(sys.argv[1]) +manifest = json.loads((bundle / "release-manifest.json").read_text(encoding="utf-8")) +bad = [] +for rel, expected in manifest["file_checksums"].items(): + if rel == "release-manifest.json": + continue + actual = hashlib.sha256((bundle / rel).read_bytes()).hexdigest() + if actual != expected: + bad.append(rel) +if bad: + print(f"FATAL: {len(bad)} file checksum mismatches: {bad[:5]}") + raise SystemExit(1) +print(f"verified {len(manifest['file_checksums'])} file checksums for {manifest['git_sha'][:12]}") +PY + # Consumed by sourcing scripts. + # shellcheck disable=SC2034 + FETCHED_BUNDLE_DIR="$bundle_dir" +} + +manifest_field() { # bundle_dir field + python3 - "$1" "$2" <<'PY' +import json +import sys +from pathlib import Path + +manifest = json.loads((Path(sys.argv[1]) / "release-manifest.json").read_text(encoding="utf-8")) +value = manifest.get(sys.argv[2]) +print("" if value is None else value) +PY +} + +# Schema compatibility rule (ADR-010/ADR-015): a release may be promoted only +# when every migration up to its required_schema_version is APPLIED. For +# rollbacks the reverse direction is also checked: applied migrations the +# target release predates must all be additive — a `breaking`-flagged +# migration in the repository manifest blocks the rollback with +# forward-recovery guidance (S6-RBK-003). +check_schema_compatibility() { # bundle_dir [deployment_type] + local bundle_dir="$1" deployment_type="${2:-deploy}" + PYTHONPATH="${ATLAS_SCRIPTS_ROOT}/../src:${PYTHONPATH:-}" \ + python3 - "$bundle_dir" "$deployment_type" "${ATLAS_SCRIPTS_ROOT}/../sql/migrations/manifest.txt" <<'PY' +import json +import sys +from pathlib import Path + +from atlas.ops.migrations import load_manifest, migration_status +from atlas.ops.rollback_compatibility import evaluate_rollback_compatibility + +bundle = Path(sys.argv[1]) +deployment_type = sys.argv[2] +repo_manifest_path = Path(sys.argv[3]) +manifest = json.loads((bundle / "release-manifest.json").read_text(encoding="utf-8")) +required = manifest["required_schema_version"] + +ledger = migration_status() +applied = [row["migration_id"] for row in ledger if row["status"] == "APPLIED"] +release_manifest_ids = [ + line.split("|")[0].strip() + for line in (bundle / "sql/migrations/manifest.txt").read_text(encoding="utf-8").splitlines() + if line.strip() and not line.startswith("#") +] +missing = [m for m in release_manifest_ids if m not in applied] +if missing: + print(f"schema check: {len(missing)} migrations pending for this release: {missing}") + print("PENDING_MIGRATIONS") + raise SystemExit(0) + +if deployment_type == "rollback": + decision = evaluate_rollback_compatibility( + applied_migration_ids=applied, + target_release_migration_ids=release_manifest_ids, + manifest=load_manifest(repo_manifest_path), + ) + print(f"schema check (rollback): {decision.reason}") + if not decision.eligible: + print("ROLLBACK_INCOMPATIBLE") + raise SystemExit(0) + +print(f"schema check: required {required} — all release migrations applied") +print("COMPATIBLE") +PY +} + +promote_release_to_composer() { # bundle_dir deployment_id git_sha + local bundle_dir="$1" deployment_id="$2" git_sha="$3" + local bucket + bucket="$(composer_bucket)" + echo "Promoting to ${bucket} (dags/project_atlas + data/current)" >&2 + + printf '{"deployment_id": "%s", "git_sha": "%s"}\n' "$deployment_id" "$git_sha" \ + >"${bundle_dir}/deployment-info.json" + + # --checksums-only is required: the deterministic bundle tar pins every + # file mtime to a fixed date, so rsync's default size+mtime comparison + # silently skips changed files whose size is unchanged (this left a stale + # release-manifest.json behind on deployment atlas-dev-20260719T004112Z). + # Runtime assets first so a parsed DAG never points at missing runtime files. + gcloud storage rsync --recursive --checksums-only --delete-unmatched-destination-objects \ + --exclude='^dags/.*' \ + "$bundle_dir" "${bucket}/data/current" >&2 + # DAG parse-time assets last. + gcloud storage rsync --recursive --checksums-only --delete-unmatched-destination-objects \ + "${bundle_dir}/dags" "${bucket}/dags/project_atlas" >&2 +} + +# Wait until the deployed DAG parses in Composer with no import errors. +wait_for_dag_parse() { # timeout_seconds + local timeout="${1:-600}" + local deadline=$((SECONDS + timeout)) + local out="" errors="" + while (( SECONDS < deadline )); do + out="$(composer_airflow dags list -o plain || true)" + if echo "$out" | awk '{print $1}' | grep -qx "atlas_batch_pipeline"; then + errors="$(composer_airflow dags list-import-errors -o plain || true)" + if ! echo "$errors" | grep -q "project_atlas"; then + echo "atlas_batch_pipeline parsed with no import errors" >&2 + return 0 + fi + # Import errors can be stale: Airflow keeps the previous deployment's + # error rows until the DAG processor re-evaluates (or stops seeing) + # each file after the GCS sync. Keep polling until the deadline and + # only fail if errors persist. + echo "DAG import errors present (may be stale, retrying):" >&2 + echo "$errors" >&2 + fi + sleep 20 + done + echo "Timed out after ${timeout}s waiting for atlas_batch_pipeline to parse cleanly" >&2 + echo "Last dags list output:" >&2 + echo "$out" >&2 + if [[ -n "$errors" ]]; then + echo "Last import errors:" >&2 + echo "$errors" >&2 + fi + return 1 +} + +# Trigger a smoke run with an explicit run id and poll to terminal state. +# Echoes nothing; returns 0 on success. Callers know the dag_run_id they passed. +run_smoke_batch() { # dag_run_id conf_json timeout_seconds + local dag_run_id="$1" conf_json="$2" timeout="${3:-2400}" + composer_airflow dags unpause atlas_batch_pipeline >/dev/null 2>&1 || true + echo "Triggering smoke run ${dag_run_id}" >&2 + composer_airflow dags trigger atlas_batch_pipeline --run-id "$dag_run_id" --conf "$conf_json" >&2 || { + echo "Trigger failed" >&2 + return 1 + } + local deadline=$((SECONDS + timeout)) + local state="" + while (( SECONDS < deadline )); do + # Airflow 3 prints "state, {conf...}" for runs triggered with --conf, so + # match the leading token rather than anchoring the whole line. + state="$(composer_airflow dags state atlas_batch_pipeline "$dag_run_id" \ + | grep -Eo '^(success|failed|running|queued)\b' | tail -1 || true)" + echo "smoke run ${dag_run_id}: state=${state:-unknown} (${SECONDS}s elapsed)" >&2 + case "$state" in + success) + echo "Smoke run ${dag_run_id}: success" >&2 + return 0 + ;; + failed) + echo "Smoke run ${dag_run_id}: FAILED" >&2 + composer_airflow tasks states-for-dag-run atlas_batch_pipeline "$dag_run_id" >&2 || true + return 1 + ;; + esac + sleep 30 + done + echo "Smoke run ${dag_run_id}: timed out after ${timeout}s (last state: ${state:-unknown})" >&2 + return 1 +} + +bq_scalar() { # sql + bq --project_id="$ATLAS_PROJECT_ID" query --use_legacy_sql=false --format=csv "$1" | tail -1 +} diff --git a/scripts/load_events.py b/scripts/load_events.py new file mode 100755 index 0000000..d0921ec --- /dev/null +++ b/scripts/load_events.py @@ -0,0 +1,43 @@ +#!/usr/bin/env python3 +"""Load Atlas events from Cloud Storage into BigQuery.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from atlas.config.settings import load_settings +from atlas.loader.bigquery import load_events_from_gcs +from atlas.logging.structured import new_pipeline_run_id + + +def main() -> int: + parser = argparse.ArgumentParser(description="Load Atlas JSONL from GCS to BigQuery") + parser.add_argument("--gcs-uri", required=True) + parser.add_argument("--run-id", default=new_pipeline_run_id()) + parser.add_argument("--batch-id") + parser.add_argument("--processing-date") + parser.add_argument("--expected-row-count", type=int) + args = parser.parse_args() + + settings = load_settings() + result = load_events_from_gcs( + settings, + args.gcs_uri, + args.gcs_uri, + args.run_id, + batch_id=args.batch_id, + processing_date=args.processing_date, + expected_row_count=args.expected_row_count, + ) + print(json.dumps(result.__dict__, indent=2, default=str)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/manage_atlas_alerts.sh b/scripts/manage_atlas_alerts.sh new file mode 100755 index 0000000..b2d535c --- /dev/null +++ b/scripts/manage_atlas_alerts.sh @@ -0,0 +1,165 @@ +#!/usr/bin/env bash +# Manage Atlas Cloud Monitoring alert policies (Sprint 5, Phase 10). +# +# Policies are defined in observability/alerts/*.json with the notification +# channel as the ${NOTIFICATION_CHANNEL} placeholder — channel resource ids +# and recipients are never committed to Git. +# +# Usage: +# manage_atlas_alerts.sh plan # diff repo vs live +# manage_atlas_alerts.sh apply # create/update all (idempotent) +# manage_atlas_alerts.sh enable # e.g. atlas-data-stale +# manage_atlas_alerts.sh disable +# manage_atlas_alerts.sh status # list live Atlas policies +# manage_atlas_alerts.sh test # publish synthetic FAIL (mode=drill) +# manage_atlas_alerts.sh delete-test-resources # publish PASS recovery for drill series +# +# apply/enable/disable require ATLAS_APPROVE_PROVISION=true. +# apply requires ATLAS_NOTIFICATION_CHANNEL_ID (projects/.../notificationChannels/...). +set -euo pipefail + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ALERTS_DIR="${SCRIPT_DIR}/../observability/alerts" +export PYTHONPATH="${SCRIPT_DIR}/../src${PYTHONPATH:+:$PYTHONPATH}" + +COMMAND="${1:-plan}" +ARG="${2:-}" + +require_provision() { + [[ "${ATLAS_APPROVE_PROVISION:-}" == "true" ]] \ + || { echo "FATAL: $COMMAND requires ATLAS_APPROVE_PROVISION=true" >&2; exit 1; } +} + +case "$COMMAND" in + plan|status) + python3 - "$COMMAND" "$PROJECT_ID" "$ALERTS_DIR" <<'PYEOF' +import json, sys +from pathlib import Path +from google.cloud import monitoring_v3 + +command, project_id, alerts_dir = sys.argv[1], sys.argv[2], Path(sys.argv[3]) +client = monitoring_v3.AlertPolicyServiceClient() +live = { + p.display_name: p + for p in client.list_alert_policies(name=f"projects/{project_id}") + if p.user_labels.get("managed_by") == "atlas-sprint5" +} +if command == "status": + print(f"{len(live)} live Atlas policies:") + for name, p in sorted(live.items()): + print(f" {'ENABLED ' if p.enabled else 'DISABLED'} {name} ({p.name.split('/')[-1]})") + sys.exit(0) +repo = {json.loads(f.read_text())["displayName"]: f.name for f in sorted(alerts_dir.glob("*.json"))} +print("PLAN (no changes made):") +for display, fname in repo.items(): + action = "UPDATE" if display in live else "CREATE" + print(f" {action} {display} <- {fname}") +for display in sorted(set(live) - set(repo)): + print(f" ORPHAN (live but not in repo): {display}") +PYEOF + ;; + + apply) + require_provision + [[ -n "${ATLAS_NOTIFICATION_CHANNEL_ID:-}" ]] \ + || { echo "FATAL: apply requires ATLAS_NOTIFICATION_CHANNEL_ID" >&2; exit 1; } + python3 - "$PROJECT_ID" "$ALERTS_DIR" "$ATLAS_NOTIFICATION_CHANNEL_ID" <<'PYEOF' +import json, sys +from pathlib import Path +from google.cloud import monitoring_v3 +from google.protobuf import json_format + +project_id, alerts_dir, channel = sys.argv[1], Path(sys.argv[2]), sys.argv[3] +client = monitoring_v3.AlertPolicyServiceClient() +parent = f"projects/{project_id}" +live = { + p.display_name: p + for p in client.list_alert_policies(name=parent) + if p.user_labels.get("managed_by") == "atlas-sprint5" +} +for f in sorted(alerts_dir.glob("*.json")): + raw = f.read_text().replace("${NOTIFICATION_CHANNEL}", channel) + desired = json_format.ParseDict(json.loads(raw), monitoring_v3.AlertPolicy()._pb) + display = desired.display_name + if display in live: + desired.name = live[display].name + # Preserve server-side condition names so updates modify in place. + existing_conditions = {c.display_name: c.name for c in live[display].conditions} + for cond in desired.conditions: + if cond.display_name in existing_conditions: + cond.name = existing_conditions[cond.display_name] + client.update_alert_policy(alert_policy=desired) + print(f"UPDATED {display}") + else: + created = client.create_alert_policy(name=parent, alert_policy=desired) + print(f"CREATED {display} ({created.name.split('/')[-1]})") +PYEOF + ;; + + enable|disable) + require_provision + [[ -n "$ARG" ]] || { echo "FATAL: $COMMAND requires a policy file stem" >&2; exit 1; } + python3 - "$COMMAND" "$PROJECT_ID" "$ALERTS_DIR" "$ARG" <<'PYEOF' +import json, sys +from pathlib import Path +from google.cloud import monitoring_v3 +from google.protobuf import field_mask_pb2 + +command, project_id, alerts_dir, stem = sys.argv[1:5] +display = json.loads((Path(alerts_dir) / f"{stem}.json").read_text())["displayName"] +client = monitoring_v3.AlertPolicyServiceClient() +for p in client.list_alert_policies(name=f"projects/{project_id}"): + if p.display_name == display: + p.enabled = command == "enable" + client.update_alert_policy( + alert_policy=p, update_mask=field_mask_pb2.FieldMask(paths=["enabled"]) + ) + print(f"{command.upper()}D {display}") + break +else: + sys.exit(f"policy not found live: {display}") +PYEOF + ;; + + test) + [[ -n "$ARG" ]] || { echo "FATAL: test requires a check_name" >&2; exit 1; } + echo "publishing synthetic FAIL (value 2, mode=drill) for check_name=$ARG" + python3 - "$PROJECT_ID" "$ARG" <<'PYEOF' +import sys +from atlas.observability.metrics import publish_gauge +project_id, check = sys.argv[1], sys.argv[2] +publish_gauge( + project_id, + "custom.googleapis.com/atlas/monitor/check_status", + 2, + {"environment": "atlas-dev", "check_name": check, "mode": "drill"}, +) +print("published; expect the policy to open an incident within ~10 minutes") +PYEOF + ;; + + delete-test-resources) + echo "publishing PASS recovery for all drill-mode check series" + python3 - "$PROJECT_ID" <<'PYEOF' +import sys +from atlas.observability.metrics import publish_gauge_safely +from atlas.observability.monitor import CHECK_NAMES +project_id = sys.argv[1] +for check in CHECK_NAMES: + publish_gauge_safely( + project_id, + "custom.googleapis.com/atlas/monitor/check_status", + 0, + {"environment": "atlas-dev", "check_name": check, "mode": "drill"}, + ) +print("recovery points published; incidents auto-close after cessation (~30 min)") +PYEOF + ;; + + *) + echo "FATAL: unknown command: $COMMAND" >&2 + echo "usage: manage_atlas_alerts.sh plan|apply|enable|disable|status|test|delete-test-resources" >&2 + exit 1 + ;; +esac diff --git a/scripts/manage_atlas_composer.sh b/scripts/manage_atlas_composer.sh new file mode 100755 index 0000000..0cac5ab --- /dev/null +++ b/scripts/manage_atlas_composer.sh @@ -0,0 +1,136 @@ +#!/usr/bin/env bash +# Ephemeral Atlas Composer environment lifecycle (Sprint 4, Phase 12 / ADR-010). +# +# Usage: +# manage_atlas_composer.sh status +# manage_atlas_composer.sh create # requires ATLAS_APPROVE_COMPOSER_CREATE=true +# # (and ATLAS_APPROVE_IAM=true for SA setup) +# manage_atlas_composer.sh delete # tears the environment down (ephemeral policy) +# +# Owner decision (ADR-010): Composer exists only for evidence capture and is +# deleted afterwards. Never leave it running unattended. +set -euo pipefail + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +REGION="${ATLAS_COMPOSER_REGION:-us-central1}" +ENV_NAME="${ATLAS_COMPOSER_ENV:-atlas-dev}" +# Verified against available images and local Airflow 3.1.7 parity (ADR-005, +# amended): build.12 is no longer offered; owner approved build.13. +IMAGE_VERSION="composer-3-airflow-3.1.7-build.13" +RUNTIME_SA="atlas-composer-runtime" +RUNTIME_EMAIL="${RUNTIME_SA}@${PROJECT_ID}.iam.gserviceaccount.com" +EVENTS_BUCKET="${ATLAS_GCS_BUCKET:-atlas-raw-events-${PROJECT_ID}}" +DEPLOYMENT_BUCKET="${ATLAS_DEPLOYMENT_BUCKET:-atlas-deployments-${PROJECT_ID}}" +PROJECT_NUMBER="$(gcloud projects describe "$PROJECT_ID" --format='value(projectNumber)')" + +ACTION="${1:?Usage: $0 status|create|delete}" + +case "$ACTION" in + status) + gcloud composer environments describe "$ENV_NAME" --location="$REGION" \ + --project="$PROJECT_ID" \ + --format="yaml(name,state,config.softwareConfig.imageVersion,config.dagGcsPrefix,config.nodeConfig.serviceAccount)" \ + 2>/dev/null || echo "Environment ${ENV_NAME} does not exist in ${REGION}." + ;; + + create) + if [[ "${ATLAS_APPROVE_COMPOSER_CREATE:-false}" != "true" ]]; then + echo "ATLAS_APPROVE_COMPOSER_CREATE != true — refusing to create Composer environment." >&2 + echo "Estimated cost while running: roughly USD 0.35-0.50/hour for a small" >&2 + echo "Composer 3 environment (~USD 300/month if left alive — ADR-010 forbids that)." >&2 + exit 3 + fi + if gcloud composer environments describe "$ENV_NAME" --location="$REGION" \ + --project="$PROJECT_ID" >/dev/null 2>&1; then + echo "Environment ${ENV_NAME} already exists — reusing." + exit 0 + fi + + echo "Enabling composer.googleapis.com..." + gcloud services enable composer.googleapis.com --project="$PROJECT_ID" + + if [[ "${ATLAS_APPROVE_IAM:-false}" == "true" ]]; then + # Composer service agent needs the V2 extension role for Composer 3. + gcloud projects add-iam-policy-binding "$PROJECT_ID" \ + --member="serviceAccount:service-${PROJECT_NUMBER}@cloudcomposer-accounts.iam.gserviceaccount.com" \ + --role="roles/composer.ServiceAgentV2Ext" --condition=None --quiet >/dev/null + + if ! gcloud iam service-accounts describe "$RUNTIME_EMAIL" --project="$PROJECT_ID" >/dev/null 2>&1; then + gcloud iam service-accounts create "$RUNTIME_SA" --project="$PROJECT_ID" \ + --display-name="Atlas Composer runtime" + fi + # Newly created service accounts propagate asynchronously; retry bindings. + # resourceViewer: the observability monitor's cost check reads + # region-us.INFORMATION_SCHEMA.JOBS, which needs bigquery.jobs.listAll + # (found live in Sprint 5: cost_anomaly returned 403 without it). + for role in roles/composer.worker roles/bigquery.jobUser roles/bigquery.dataEditor roles/bigquery.resourceViewer; do + for attempt in 1 2 3 4 5; do + if gcloud projects add-iam-policy-binding "$PROJECT_ID" \ + --member="serviceAccount:${RUNTIME_EMAIL}" --role="$role" \ + --condition=None --quiet >/dev/null 2>&1; then + break + fi + if [[ "$attempt" -eq 5 ]]; then + echo "FATAL: could not bind ${role} to ${RUNTIME_EMAIL}" >&2 + exit 1 + fi + echo " binding ${role} not ready (attempt ${attempt}); retrying in $((attempt * 5))s..." + sleep $((attempt * 5)) + done + done + gcloud storage buckets add-iam-policy-binding "gs://${EVENTS_BUCKET}" \ + --member="serviceAccount:${RUNTIME_EMAIL}" --role="roles/storage.objectAdmin" >/dev/null + gcloud storage buckets add-iam-policy-binding "gs://${DEPLOYMENT_BUCKET}" \ + --member="serviceAccount:${RUNTIME_EMAIL}" --role="roles/storage.objectViewer" >/dev/null + # The deployer must be able to attach the runtime SA to the environment. + gcloud iam service-accounts add-iam-policy-binding "$RUNTIME_EMAIL" \ + --project="$PROJECT_ID" \ + --member="serviceAccount:atlas-github-deployer@${PROJECT_ID}.iam.gserviceaccount.com" \ + --role="roles/iam.serviceAccountUser" >/dev/null + echo "Runtime service account ${RUNTIME_EMAIL} configured." + else + echo "ATLAS_APPROVE_IAM != true — assuming ${RUNTIME_EMAIL} and grants already exist." + fi + + echo "Creating ${ENV_NAME} (${IMAGE_VERSION}, ${REGION}, size small). This takes ~25 minutes..." + gcloud composer environments create "$ENV_NAME" \ + --project="$PROJECT_ID" \ + --location="$REGION" \ + --image-version="$IMAGE_VERSION" \ + --environment-size=small \ + --service-account="$RUNTIME_EMAIL" \ + --env-variables="ATLAS_ROOT=/home/airflow/gcs/data/current,ATLAS_GCP_PROJECT_ID=${PROJECT_ID},ATLAS_GCS_BUCKET=${EVENTS_BUCKET},ATLAS_BQ_DATASET=atlas_raw,ATLAS_DBT_DATASET=atlas,DBT_PROJECT_DIR=/home/airflow/gcs/data/current/dbt/atlas_dbt,DBT_PROFILES_DIR=/home/airflow/gcs/data/current/dbt/profiles,DBT_LOCATION=US,DBT_TARGET_PATH=/tmp/dbt-target,DBT_LOG_PATH=/tmp/dbt-logs,ATLAS_LOG_TO_CLOUD_LOGGING=true,ATLAS_ENVIRONMENT=atlas-dev" + + echo "Installing pinned dbt PyPI packages (second long-running operation)..." + PKG_FILE="$(mktemp)" + grep -E '^(dbt-core|dbt-bigquery)==' "$(dirname "${BASH_SOURCE[0]}")/../dbt/requirements-dbt.txt" >"$PKG_FILE" + cat "$PKG_FILE" + gcloud composer environments update "$ENV_NAME" \ + --project="$PROJECT_ID" --location="$REGION" \ + --update-pypi-packages-from-file="$PKG_FILE" + rm -f "$PKG_FILE" + + gcloud composer environments describe "$ENV_NAME" --location="$REGION" \ + --project="$PROJECT_ID" \ + --format="yaml(state,config.softwareConfig.imageVersion,config.dagGcsPrefix)" + echo "Composer environment ready. Remember: delete it after evidence capture (ADR-010)." + ;; + + delete) + if ! gcloud composer environments describe "$ENV_NAME" --location="$REGION" \ + --project="$PROJECT_ID" >/dev/null 2>&1; then + echo "Environment ${ENV_NAME} does not exist — nothing to delete." + exit 0 + fi + echo "Deleting ${ENV_NAME} in ${REGION} (ephemeral policy, ADR-010)..." + gcloud composer environments delete "$ENV_NAME" \ + --project="$PROJECT_ID" --location="$REGION" --quiet + echo "Deleted. Note: the Composer-created bucket is retained by GCP; remove it" + echo "manually if evidence has been captured elsewhere." + ;; + + *) + echo "Usage: $0 status|create|delete" >&2 + exit 2 + ;; +esac diff --git a/scripts/rollback_atlas.sh b/scripts/rollback_atlas.sh new file mode 100755 index 0000000..d65153f --- /dev/null +++ b/scripts/rollback_atlas.sh @@ -0,0 +1,72 @@ +#!/usr/bin/env bash +# Runtime rollback to a prior validated immutable release (Sprint 4, Phase 14). +# +# Usage: +# rollback_atlas.sh [--target-sha ] +# +# Selects the newest SUCCESS deployment (excluding the currently deployed SHA) +# from atlas_ops.deployments unless --target-sha is given, verifies manifest, +# checksums, and schema compatibility, then re-promotes that release and runs a +# rollback smoke batch. Records ROLLED_BACK / ROLLBACK_FAILED. +# +# Never: deletes historical bundles, rewrites Git history, moves release tags, +# or reverses BigQuery migrations. Requires ATLAS_APPROVE_ROLLBACK_TEST=true +# for live execution (plus ATLAS_APPROVE_DEPLOY=true for the promotion itself). +set -uo pipefail + +ATLAS_SCRIPTS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=lib_atlas_deploy.sh +source "${ATLAS_SCRIPTS_ROOT}/lib_atlas_deploy.sh" +export PYTHONPATH="${ATLAS_SCRIPTS_ROOT}/../src:${PYTHONPATH:-}" + +TARGET_SHA="" +while [[ $# -gt 0 ]]; do + case "$1" in + --target-sha) TARGET_SHA="${2:?}"; shift 2 ;; + *) echo "Unknown argument: $1" >&2; exit 2 ;; + esac +done + +if [[ "${ATLAS_APPROVE_ROLLBACK_TEST:-false}" != "true" ]]; then + echo "ATLAS_APPROVE_ROLLBACK_TEST != true — refusing to execute a live rollback." >&2 + exit 3 +fi + +# --- 1. Identify the currently deployed SHA ----------------------------------------- +BUCKET="$(composer_bucket)" +CURRENT_SHA="$(gcloud storage cat "${BUCKET}/data/current/release-manifest.json" 2>/dev/null \ + | python3 -c 'import json,sys; print(json.load(sys.stdin)["git_sha"])' || echo "")" +if [[ -z "$CURRENT_SHA" ]]; then + echo "Unable to determine currently deployed SHA from Composer runtime path." >&2 + exit 1 +fi +echo "currently deployed: ${CURRENT_SHA}" + +# --- 2. Identify the prior validated release ------------------------------------------ +if [[ -z "$TARGET_SHA" ]]; then + TARGET_SHA="$(ATLAS_CURRENT_SHA="$CURRENT_SHA" python3 - <<'PY' +import os + +from atlas.ops.deployments import latest_successful_deployment + +row = latest_successful_deployment("atlas-dev", exclude_git_sha=os.environ["ATLAS_CURRENT_SHA"]) +print(row["git_sha"] if row else "") +PY +)" +fi +if [[ -z "$TARGET_SHA" ]]; then + echo "No prior validated SUCCESS deployment found in atlas_ops.deployments." >&2 + exit 1 +fi +if [[ "$TARGET_SHA" == "$CURRENT_SHA" ]]; then + echo "Target SHA equals currently deployed SHA — nothing to roll back to." >&2 + exit 1 +fi +echo "rollback target: ${TARGET_SHA}" + +# --- 3-11. Delegate to the deployment engine as a rollback ------------------------------- +exec bash "${ATLAS_SCRIPTS_ROOT}/deploy_atlas_release.sh" \ + --git-sha "$TARGET_SHA" \ + --deployment-type rollback \ + --previous-git-sha "$CURRENT_SHA" \ + --skip-migrations diff --git a/scripts/run_airflow_sprint3.sh b/scripts/run_airflow_sprint3.sh new file mode 100755 index 0000000..066d06e --- /dev/null +++ b/scripts/run_airflow_sprint3.sh @@ -0,0 +1,175 @@ +#!/usr/bin/env bash +# Trigger an Atlas Sprint 3 DAG run and poll it to a terminal state. +# +# Exit codes: +# 0 the DAG run reached terminal state "success" +# 1 usage error, DAG not registered, trigger failure, run failure, or timeout +set -euo pipefail + +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +# shellcheck disable=SC1091 +source "${ATLAS_ROOT}/scripts/airflow_env.sh" + +PROCESSING_DATE="" +BATCH_ID="" +CONF_JSON="{}" +RUN_ID="" +WAIT_SECONDS="${WAIT_SECONDS:-120}" +# Upper bound for the run itself (trigger to terminal state). +RUN_TIMEOUT_SECONDS="${RUN_TIMEOUT_SECONDS:-1800}" +POLL_INTERVAL_SECONDS="${POLL_INTERVAL_SECONDS:-10}" +while [[ $# -gt 0 ]]; do + case "$1" in + --processing-date) PROCESSING_DATE="$2"; shift 2 ;; + --batch-id) BATCH_ID="$2"; shift 2 ;; + --conf) CONF_JSON="$2"; shift 2 ;; + --run-id) RUN_ID="$2"; shift 2 ;; + --wait-seconds) WAIT_SECONDS="$2"; shift 2 ;; + --run-timeout-seconds) RUN_TIMEOUT_SECONDS="$2"; shift 2 ;; + *) echo "Unknown arg: $1" >&2; exit 1 ;; + esac +done + +# Merge overrides into the conf JSON. Values are passed via the environment +# (never interpolated into the Python source) so quotes or shell metacharacters +# in a value cannot corrupt the JSON or inject code. +if [[ -n "$PROCESSING_DATE" || -n "$BATCH_ID" ]]; then + CONF_JSON="$( + ATLAS_CONF="$CONF_JSON" ATLAS_PD="$PROCESSING_DATE" ATLAS_BID="$BATCH_ID" python3 - <<'PY' +import json +import os + +conf = json.loads(os.environ.get("ATLAS_CONF") or "{}") +if os.environ.get("ATLAS_PD"): + conf["processing_date"] = os.environ["ATLAS_PD"] +if os.environ.get("ATLAS_BID"): + conf["batch_id"] = os.environ["ATLAS_BID"] +print(json.dumps(conf)) +PY + )" +fi + +echo "Waiting up to ${WAIT_SECONDS}s for atlas_batch_pipeline to register..." +deadline=$((SECONDS + WAIT_SECONDS)) +while (( SECONDS < deadline )); do + if airflow dags list 2>/dev/null | awk '{print $1}' | grep -qx "atlas_batch_pipeline"; then + break + fi + if airflow dags list-import-errors 2>/dev/null | grep -q atlas_batch_pipeline; then + echo "DAG import error detected:" >&2 + airflow dags list-import-errors >&2 || true + exit 1 + fi + sleep 2 +done + +if ! airflow dags list 2>/dev/null | awk '{print $1}' | grep -qx "atlas_batch_pipeline"; then + echo "atlas_batch_pipeline not registered. Start Airflow first:" >&2 + echo " bash scripts/start_airflow_local.sh" >&2 + echo "Check import errors:" >&2 + airflow dags list-import-errors >&2 || true + exit 1 +fi + +# The DAG deploys paused by default (Sprint 4); unpause before triggering. +airflow dags unpause atlas_batch_pipeline >/dev/null 2>&1 || true + +ARGS=(dags trigger atlas_batch_pipeline -o json) +if [[ -n "$RUN_ID" ]]; then ARGS+=(--run-id "$RUN_ID"); fi +ARGS+=(--conf "$CONF_JSON") +TRIGGER_JSON="$(airflow "${ARGS[@]}")" +DAG_RUN_ID="$( + TRIGGER_OUT="$TRIGGER_JSON" python3 - <<'PY' +import json +import os + +payload = json.loads(os.environ["TRIGGER_OUT"]) +if isinstance(payload, list): + payload = payload[0] +print(payload["dag_run_id"]) +PY +)" +echo "Triggered atlas_batch_pipeline run_id=${DAG_RUN_ID} conf=${CONF_JSON}" + +report_failed_tasks() { + echo "Failed or upstream-failed tasks:" >&2 + airflow tasks states-for-dag-run atlas_batch_pipeline "$DAG_RUN_ID" 2>/dev/null \ + | grep -Ei 'failed|upstream_failed' >&2 || echo " (task states unavailable)" >&2 +} + +report_audit_row() { + # Best-effort: report the matching atlas_ops.pipeline_runs row. Requires + # google-cloud-bigquery credentials; failures here never mask the run result. + DAG_RUN_ID="$DAG_RUN_ID" ATLAS_PD="$PROCESSING_DATE" python3 - <<'PY' || echo "(audit row lookup unavailable)" +import datetime +import json +import os + +from atlas.batch.context import build_pipeline_run_id +from atlas.ops.audit import query_pipeline_run + +processing_date = os.environ.get("ATLAS_PD") or datetime.datetime.now(tz=datetime.UTC).date().isoformat() +pipeline_run_id = build_pipeline_run_id(processing_date, os.environ["DAG_RUN_ID"]) +row = query_pipeline_run(pipeline_run_id) +if row is None: + print(f"No audit row found for pipeline_run_id={pipeline_run_id}") +else: + printable = {k: str(v) for k, v in row.items()} + print("atlas_ops.pipeline_runs row:") + print(json.dumps(printable, indent=2)) +PY +} + +report_summary_path() { + local summary + # Prefer the exact summary for this run; fall back to the newest one. + summary="$( + DAG_RUN_ID="$DAG_RUN_ID" ATLAS_PD="$PROCESSING_DATE" python3 - <<'PY' 2>/dev/null || true +import datetime +import os + +from atlas.batch.context import build_pipeline_run_id +from atlas.config.settings import atlas_root + +processing_date = os.environ.get("ATLAS_PD") or datetime.datetime.now(tz=datetime.UTC).date().isoformat() +pipeline_run_id = build_pipeline_run_id(processing_date, os.environ["DAG_RUN_ID"]) +path = atlas_root() / "logs" / "airflow" / pipeline_run_id / "run-summary.json" +if path.exists(): + print(path) +PY + )" + if [[ -z "$summary" ]]; then + summary="$(ls -t "${ATLAS_ROOT}"/logs/airflow/*/run-summary.json 2>/dev/null | head -1 || true)" + fi + if [[ -n "$summary" ]]; then + echo "Local run summary: $summary" + fi +} + +echo "Polling run to terminal state (timeout ${RUN_TIMEOUT_SECONDS}s)..." +run_deadline=$((SECONDS + RUN_TIMEOUT_SECONDS)) +STATE="" +while (( SECONDS < run_deadline )); do + STATE="$(airflow dags state atlas_batch_pipeline "$DAG_RUN_ID" -o plain 2>/dev/null | tail -1 | tr -d '[:space:]')" + case "$STATE" in + success) + echo "Airflow final state: success" + report_audit_row + report_summary_path + exit 0 + ;; + failed) + echo "Airflow final state: failed" >&2 + report_failed_tasks + report_audit_row + report_summary_path + exit 1 + ;; + esac + sleep "$POLL_INTERVAL_SECONDS" +done + +echo "Timed out after ${RUN_TIMEOUT_SECONDS}s waiting for run ${DAG_RUN_ID} (last state: ${STATE:-unknown})" >&2 +report_failed_tasks +report_audit_row +exit 1 diff --git a/scripts/run_atlas_step.sh b/scripts/run_atlas_step.sh new file mode 100755 index 0000000..97e9a5b --- /dev/null +++ b/scripts/run_atlas_step.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +# Thin dispatcher over Atlas Python/dbt CLIs with structured context logging. +set -euo pipefail + +STEP="${1:?step name required}" +# Note: do NOT use ${2:-{}} — bash parses the default as `{` plus a literal +# trailing `}`, which appends a stray `}` to a JSON object argument and corrupts it. +CTX_JSON="${2:-}" +[[ -n "$CTX_JSON" ]] || CTX_JSON="{}" +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +export ATLAS_ROOT +export PYTHONPATH="${ATLAS_ROOT}/src:${PYTHONPATH:-}" +DBT_PROJECT_DIR="${DBT_PROJECT_DIR:-${ATLAS_ROOT}/dbt/atlas_dbt}" +export DBT_PROJECT_DIR + +exec python3 "${ATLAS_ROOT}/scripts/atlas_step_runner.py" "$STEP" "$CTX_JSON" diff --git a/scripts/run_dbt_sprint2.sh b/scripts/run_dbt_sprint2.sh new file mode 100755 index 0000000..c4eaf2b --- /dev/null +++ b/scripts/run_dbt_sprint2.sh @@ -0,0 +1,127 @@ +#!/usr/bin/env bash +# Run the Atlas Sprint 2 dbt warehouse build in Cloud Shell or an approved agent env. +set -Eeuo pipefail +IFS=$'\n\t' +umask 077 + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +DBT_PROJECT_DIR="${ATLAS_ROOT}/dbt/atlas_dbt" +VENV_DIR="${ATLAS_ROOT}/.venv-dbt" +PROFILES_DIR="${DBT_PROFILES_DIR:-$HOME/.dbt}" +LOG_DIR="${ATLAS_ROOT}/logs" +ARTIFACT_DIR="${LOG_DIR}/dbt-artifacts" +TIMESTAMP="$(date -u +%Y%m%dT%H%M%SZ)" + +FULL_REFRESH=false +SKIP_DOCS=false + +usage() { + cat <<'EOF' +Usage: run_dbt_sprint2.sh [--full-refresh] [--skip-docs] + +Runs deps, debug, seed, source freshness, build, optional docs generation, +and preserves dbt artifacts under logs/dbt-artifacts/. +EOF +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --full-refresh) + FULL_REFRESH=true + shift + ;; + --skip-docs) + SKIP_DOCS=true + shift + ;; + -h|--help) + usage + exit 0 + ;; + *) + echo "error: unknown argument: $1" >&2 + usage >&2 + exit 1 + ;; + esac +done + +fail() { + echo "error: $*" >&2 + exit 1 +} + +print_command() { + printf '+' + printf ' %q' "$@" + printf '\n' +} + +run() { + print_command "$@" + "$@" +} + +require_command() { + command -v "$1" >/dev/null 2>&1 || fail "$1 is required but was not found on PATH" +} + +require_command bq +[[ -x "${VENV_DIR}/bin/dbt" ]] || fail "missing ${VENV_DIR}/bin/dbt; run scripts/setup_dbt.sh first" + +run mkdir -p "$LOG_DIR" "$ARTIFACT_DIR" +DBT_BIN="${VENV_DIR}/bin/dbt" +DBT_FLAGS=(--project-dir "$DBT_PROJECT_DIR" --profiles-dir "$PROFILES_DIR" --target dev) + +run_dbt() { + run "$DBT_BIN" "$@" "${DBT_FLAGS[@]}" +} + +run_dbt deps +run_dbt debug +run_dbt seed --full-refresh +run_dbt source freshness || true + +if [[ "$FULL_REFRESH" == true ]]; then + run_dbt build --full-refresh +else + run_dbt build +fi + +if [[ "$SKIP_DOCS" == false ]]; then + run_dbt docs generate + run mkdir -p "${ARTIFACT_DIR}/${TIMESTAMP}" + run cp -a "${DBT_PROJECT_DIR}/target/." "${ARTIFACT_DIR}/${TIMESTAMP}/" +fi + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-${GCP_PROJECT_ID:-}}" +[[ -n "$PROJECT_ID" ]] || fail "ATLAS_GCP_PROJECT_ID or GCP_PROJECT_ID must be set" +BQ_REGION="$(printf '%s' "${DBT_LOCATION:-US}" | tr '[:upper:]' '[:lower:]')" + +print_inventory() { + local dataset="$1" + local table="$2" + bq query --use_legacy_sql=false --format=prettyjson \ + "SELECT table_schema, table_name, table_type + FROM \`${PROJECT_ID}.region-${BQ_REGION}.INFORMATION_SCHEMA.TABLES\` + WHERE table_schema = '${dataset}' AND table_name = '${table}'" 2>/dev/null || true +} + +echo "Sprint 2 relation inventory (best effort):" +for relation in \ + "atlas_staging.stg_events" \ + "atlas_intermediate.int_event_classification" \ + "atlas_intermediate.int_accepted_events" \ + "atlas_quarantine.int_rejected_events" \ + "atlas_core.fct_events" \ + "atlas_marts.mart_daily_event_metrics" +do + schema="${relation%%.*}" + table="${relation##*.}" + echo "- ${PROJECT_ID}.${relation}" + print_inventory "$schema" "$table" +done + +run_dbt show --select stg_events --limit 1 +echo "dbt Sprint 2 run complete. Artifacts: ${ARTIFACT_DIR}/${TIMESTAMP}/" +echo "Next: bash scripts/validate_dbt_sprint2.sh" diff --git a/scripts/run_failure_scenario.sh b/scripts/run_failure_scenario.sh new file mode 100755 index 0000000..5dedd95 --- /dev/null +++ b/scripts/run_failure_scenario.sh @@ -0,0 +1,54 @@ +#!/usr/bin/env bash +# Atlas controlled failure-scenario runner (Sprint 6, ADR-013). +# +# Usage: +# run_failure_scenario.sh \ +# --scenario S6-XXX-NNN --environment atlas-dev [--batch-id atlas-s6-...] +# +# Safety contract (enforced in atlas.failure_injection.framework and tested): +# - disabled by default; an explicit scenario id is always required +# - run/cleanup require ATLAS_APPROVE_FAILURE_INJECTION=true plus any +# scenario-specific approvals (IAM, destructive fixture, rollback test) +# - refuses scheduled execution, canonical batch ids, and production-style +# environments +# - every scenario carries a hard timeout and cost ceiling +# - there is no silent fallback to normal execution: refusal exits nonzero +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ATLAS_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)" + +if [[ $# -lt 1 ]]; then + echo "Usage: $0 --scenario --environment [--batch-id ]" >&2 + exit 64 +fi + +COMMAND="$1" +shift + +# Refuse to run inside a scheduled Airflow context outright — belt to the +# framework's suspenders. Drills are always operator-triggered. +if [[ "${AIRFLOW_CTX_DAG_RUN_TYPE:-}" == "scheduled" ]]; then + echo "REFUSED: fault-injection commands never run inside scheduled Airflow execution" >&2 + exit 2 +fi + +# The scenario id must also be exported for the framework's explicit-parameter +# gate when running the gated commands. +if [[ "${COMMAND}" == "run" ]]; then + scenario="" + args=("$@") + for i in "${!args[@]}"; do + if [[ "${args[$i]}" == "--scenario" ]]; then + scenario="${args[$((i + 1))]:-}" + fi + done + if [[ -z "${scenario}" ]]; then + echo "REFUSED: run requires an explicit --scenario" >&2 + exit 64 + fi + export ATLAS_INJECTION_SCENARIO="${scenario}" +fi + +PYTHONPATH="${ATLAS_ROOT}/src${PYTHONPATH:+:${PYTHONPATH}}" \ + exec python3 -m atlas.failure_injection.cli "${COMMAND}" "$@" diff --git a/scripts/run_performance_suite.sh b/scripts/run_performance_suite.sh new file mode 100644 index 0000000..ad4f490 --- /dev/null +++ b/scripts/run_performance_suite.sh @@ -0,0 +1,117 @@ +#!/usr/bin/env bash +# Atlas BigQuery performance suite (Sprint 7, Phase 10-11). +# +# Dry-runs every representative query (bills $0) to capture bytes-processed +# baselines. When ATLAS_APPROVE_PERFORMANCE_TESTS=true it also EXECUTES each +# query under a per-query maximum_bytes_billed ceiling and a cumulative suite +# ceiling, capturing bytes billed, slot-ms, and elapsed. Results are written to +# observability/performance/results/. +# +# Usage: +# bash scripts/run_performance_suite.sh [--project P] [--location US] [--execute] +set -euo pipefail + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +PROJECT="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +LOCATION="${ATLAS_BQ_LOCATION:-US}" +EXECUTE="false" +while [[ $# -gt 0 ]]; do + case "$1" in + --project) PROJECT="$2"; shift 2 ;; + --location) LOCATION="$2"; shift 2 ;; + --execute) EXECUTE="true"; shift ;; + *) echo "Unknown arg: $1" >&2; exit 1 ;; + esac +done + +if [[ "$EXECUTE" == "true" && "${ATLAS_APPROVE_PERFORMANCE_TESTS:-}" != "true" ]]; then + echo "ERROR: --execute requires ATLAS_APPROVE_PERFORMANCE_TESTS=true" >&2 + exit 3 +fi + +QUERY_DIR="${ATLAS_ROOT}/observability/performance/queries" +OUT_DIR="${ATLAS_ROOT}/observability/performance/results" +mkdir -p "$OUT_DIR" + +PROJECT="$PROJECT" LOCATION="$LOCATION" EXECUTE="$EXECUTE" QUERY_DIR="$QUERY_DIR" \ +OUT_DIR="$OUT_DIR" PYTHONPATH="${ATLAS_ROOT}/src" python3 - <<'PY' +import json +import os +import time +from pathlib import Path + +from google.cloud import bigquery + +from atlas.observability.cost_guard import max_performance_suite_bytes, max_query_bytes + +project = os.environ["PROJECT"] +location = os.environ["LOCATION"] +execute = os.environ["EXECUTE"] == "true" +query_dir = Path(os.environ["QUERY_DIR"]) +out_dir = Path(os.environ["OUT_DIR"]) + +client = bigquery.Client(project=project, location=location) +suite_ceiling = max_performance_suite_bytes("atlas-dev") +per_query_ceiling = max_query_bytes("atlas-dev") + +results = [] +cumulative_billed = 0 +# Exclude the deliberately-unbounded fixture from the normal suite. +queries = sorted(q for q in query_dir.glob("*.sql") if q.stem != "unbounded_scan") + +for q in queries: + sql = q.read_text().replace("${PROJECT}", project) + dry = client.query( + sql, job_config=bigquery.QueryJobConfig(dry_run=True, use_query_cache=False) + ) + estimated = int(dry.total_bytes_processed or 0) + record = { + "query": q.stem, + "estimated_bytes": estimated, + "within_per_query_ceiling": estimated <= per_query_ceiling, + } + if execute: + if cumulative_billed + estimated > suite_ceiling: + record["executed"] = False + record["skipped_reason"] = "would exceed suite byte ceiling" + results.append(record) + continue + cfg = bigquery.QueryJobConfig( + maximum_bytes_billed=per_query_ceiling, + labels={"atlas_component": "perf_suite", "atlas_sprint": "7"}, + use_query_cache=False, + ) + start = time.time() + job = client.query(sql, job_config=cfg) + rows = list(job.result()) + elapsed_ms = int((time.time() - start) * 1000) + billed = int(job.total_bytes_billed or 0) + cumulative_billed += billed + record.update( + { + "executed": True, + "bytes_billed": billed, + "slot_ms": int(job.slot_millis or 0), + "elapsed_ms": elapsed_ms, + "output_rows": len(rows), + "cache_hit": bool(job.cache_hit), + "correctness_checksum": str(rows[0]) if rows else "empty", + } + ) + results.append(record) + +payload = { + "project": project, + "location": location, + "executed": execute, + "per_query_ceiling_bytes": per_query_ceiling, + "suite_ceiling_bytes": suite_ceiling, + "cumulative_bytes_billed": cumulative_billed, + "queries": results, +} +mode = "executed" if execute else "dryrun" +out = out_dir / f"baseline-{mode}.json" +out.write_text(json.dumps(payload, indent=2) + "\n") +print(json.dumps(payload, indent=2)) +print(f"\nwrote {out}") +PY diff --git a/scripts/run_pipeline.py b/scripts/run_pipeline.py new file mode 100755 index 0000000..b01f61e --- /dev/null +++ b/scripts/run_pipeline.py @@ -0,0 +1,47 @@ +#!/usr/bin/env python3 +"""Run the full Atlas Sprint 1 pipeline.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from atlas.config.settings import load_settings +from atlas.pipeline.orchestrator import run_pipeline, summarize_result + + +def main() -> int: + parser = argparse.ArgumentParser(description="Run Atlas Sprint 1 end-to-end") + parser.add_argument("--approve-provision", action="store_true", help="Required to mutate GCP resources") + parser.add_argument("--run-id") + args = parser.parse_args() + + if not args.approve_provision: + print( + json.dumps( + { + "status": "blocked", + "message": ( + "Live GCP provisioning is gated. Re-run with --approve-provision " + "after reviewing docs/runbook.md." + ), + }, + indent=2, + ) + ) + return 2 + + settings = load_settings() + result = run_pipeline(settings, pipeline_run_id=args.run_id) + summary = summarize_result(result) + print(json.dumps(summary, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_sprint3_acceptance.sh b/scripts/run_sprint3_acceptance.sh new file mode 100644 index 0000000..26676b3 --- /dev/null +++ b/scripts/run_sprint3_acceptance.sh @@ -0,0 +1,92 @@ +#!/usr/bin/env bash +# Run Sprint 3 live acceptance matrix and print GCP evidence. +set -euo pipefail + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +# shellcheck disable=SC1091 +source "${ATLAS_ROOT}/scripts/airflow_env.sh" + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +RESULTS_FILE="${ATLAS_ROOT}/logs/sprint3-acceptance-$(date -u +%Y%m%dT%H%M%SZ).jsonl" +mkdir -p "${ATLAS_ROOT}/logs" + +airflow dags unpause atlas_batch_pipeline >/dev/null 2>&1 || true + +wait_for_run() { + local run_id="$1" + local timeout="${2:-1800}" + local deadline=$((SECONDS + timeout)) + while (( SECONDS < deadline )); do + state="$(airflow dags state atlas_batch_pipeline "$run_id" -o plain 2>/dev/null | tail -1 | tr -d '[:space:]')" + if [[ "$state" == "success" || "$state" == "failed" ]]; then + echo "$state" + return 0 + fi + sleep 10 + done + echo "timeout" +} + +trigger_and_wait() { + local label="$1" + local conf="$2" + local run_id="${3:-}" + echo "" + echo "========== ${label} ==========" + local args=(dags trigger atlas_batch_pipeline -c "$conf" -o json) + if [[ -n "$run_id" ]]; then + args+=(-r "$run_id") + fi + trigger_json="$(airflow "${args[@]}" 2>/dev/null)" + echo "$trigger_json" | python3 -m json.tool + actual_run_id="$(echo "$trigger_json" | python3 -c "import json,sys; print(json.load(sys.stdin)[0]['dag_run_id'])")" + echo "Waiting for run_id=${actual_run_id} ..." + final_state="$(wait_for_run "$actual_run_id" "${WAIT_SECONDS:-1800}")" + echo "Airflow final state: ${final_state}" + echo "$trigger_json" | python3 -c "import json,sys; d=json.load(sys.stdin)[0]; print(json.dumps({'scenario':'$label','dag_run_id':d['dag_run_id'],'logical_date':d.get('logical_date'),'conf':json.loads('$conf'),'airflow_state':'$final_state'}))" >>"$RESULTS_FILE" +} + +query_gcp() { + echo "" + echo "========== GCP audit (atlas_ops.pipeline_runs) ==========" + bq query --use_legacy_sql=false --format=prettyjson \ + "SELECT pipeline_run_id, batch_id, status, attempt_number, started_at, completed_at + FROM \`${PROJECT_ID}.atlas_ops.pipeline_runs\` + ORDER BY started_at DESC + LIMIT 8" + + echo "" + echo "========== GCP raw batch counts ==========" + bq query --use_legacy_sql=false --format=pretty \ + "SELECT batch_id, COUNT(*) AS rows, COUNT(DISTINCT pipeline_run_id) AS runs + FROM \`${PROJECT_ID}.atlas_raw.events\` + WHERE batch_id IN ('atlas-20260718','atlas-20260701') + GROUP BY batch_id + ORDER BY batch_id" + + echo "" + echo "========== GCS batch objects ==========" + gsutil ls -l "gs://atlas-raw-events-${PROJECT_ID}/raw/event_date=2026-07-18/batch_id=atlas-20260718/**" 2>/dev/null || true + gsutil ls -l "gs://atlas-raw-events-${PROJECT_ID}/raw/event_date=2026-07-01/batch_id=atlas-20260701/**" 2>/dev/null || true +} + +# 1. Retry success (upload_once) +trigger_and_wait "upload_once" \ + '{"processing_date":"2026-07-18","batch_id":"atlas-20260718","upload_once":true}' + +# 2. Idempotent rerun (same batch, new pipeline run) +trigger_and_wait "idempotent_rerun" \ + '{"processing_date":"2026-07-18","batch_id":"atlas-20260718"}' \ + "manual__sprint3-idempotent-$(date -u +%Y%m%dT%H%M%SZ)" + +# 3. dbt failure injection +trigger_and_wait "dbt_test_failure" \ + '{"processing_date":"2026-07-01","batch_id":"atlas-20260701","dbt_test_failure":true}' + +# 4. Historical recovery (backfill without injection) +trigger_and_wait "historical_recovery" \ + '{"processing_date":"2026-07-01","batch_id":"atlas-20260701"}' + +query_gcp +echo "" +echo "Scenario results written to ${RESULTS_FILE}" diff --git a/scripts/setup_airflow.sh b/scripts/setup_airflow.sh new file mode 100755 index 0000000..64bbda2 --- /dev/null +++ b/scripts/setup_airflow.sh @@ -0,0 +1,36 @@ +#!/usr/bin/env bash +# Create or reuse local Airflow 3.1.7 environment with Composer-parity pins. +set -euo pipefail + +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +VENV="${ATLAS_ROOT}/.venv-airflow" +REQ="${ATLAS_ROOT}/airflow/requirements-airflow.txt" +CONSTRAINTS="https://raw.githubusercontent.com/apache/airflow/constraints-3.1.7/constraints-3.12.txt" +AIRFLOW_HOME="${ATLAS_ROOT}/.airflow" +RESET="${RESET_AIRFLOW:-false}" + +if [[ "$RESET" == "true" ]]; then + rm -rf "$VENV" "$AIRFLOW_HOME" +fi + +if [[ ! -d "$VENV" ]]; then + python3 -m venv "$VENV" +fi +# shellcheck disable=SC1091 +source "$VENV/bin/activate" +pip install --upgrade pip +pip install "apache-airflow==3.1.7" --constraint "$CONSTRAINTS" +pip install -r "$REQ" +pip install -r "${ATLAS_ROOT}/requirements.txt" pytest +pip check + +export AIRFLOW_HOME +export AIRFLOW__CORE__LOAD_EXAMPLES=False +export AIRFLOW__CORE__DAGS_FOLDER="${ATLAS_ROOT}/dags" +export ATLAS_ROOT +export PYTHONPATH="${ATLAS_ROOT}/src:${ATLAS_ROOT}/dags" +mkdir -p "$AIRFLOW_HOME" + +airflow db migrate +airflow info +echo "Airflow setup complete at $VENV" diff --git a/scripts/setup_dbt.sh b/scripts/setup_dbt.sh new file mode 100755 index 0000000..b989265 --- /dev/null +++ b/scripts/setup_dbt.sh @@ -0,0 +1,210 @@ +#!/usr/bin/env bash +# Configure the isolated Atlas dbt environment and user-level profile. +set -Eeuo pipefail +IFS=$'\n\t' +umask 077 + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +WORKSPACE_ROOT="$(cd "${ATLAS_ROOT}/.." && pwd)" +DBT_ROOT="${ATLAS_ROOT}/dbt" +DBT_PROJECT_DIR="${DBT_ROOT}/atlas_dbt" +REQUIREMENTS_FILE="${DBT_ROOT}/requirements-dbt.txt" +VENV_DIR="${ATLAS_ROOT}/.venv-dbt" +DEFAULT_ATLAS_PROJECT="example-gcp-project" + +fail() { + echo "error: $*" >&2 + exit 1 +} + +print_command() { + printf '+' + printf ' %q' "$@" + printf '\n' +} + +run() { + print_command "$@" + "$@" +} + +require_command() { + command -v "$1" >/dev/null 2>&1 || fail "$1 is required but was not found on PATH" +} + +require_command gcloud +require_command bq +require_command git + +if command -v python3 >/dev/null 2>&1; then + PYTHON_BIN="$(command -v python3)" +elif command -v python >/dev/null 2>&1; then + PYTHON_BIN="$(command -v python)" +else + fail "python3 (or python) is required but was not found on PATH" +fi + +run "$PYTHON_BIN" --version + +print_command git -C "$WORKSPACE_ROOT" rev-parse --show-toplevel +GIT_ROOT="$(git -C "$WORKSPACE_ROOT" rev-parse --show-toplevel)" +[[ "$GIT_ROOT" == "$WORKSPACE_ROOT" ]] \ + || fail "expected ${WORKSPACE_ROOT} to be the git worktree root, found ${GIT_ROOT}" + +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-${GCP_PROJECT_ID:-$DEFAULT_ATLAS_PROJECT}}" +RAW_DATASET="${ATLAS_BQ_DATASET:-atlas_raw}" +TARGET_DATASET="${ATLAS_DBT_DATASET:-atlas}" +AUTH_METHOD="${ATLAS_DBT_AUTH_METHOD:-oauth}" + +[[ "$PROJECT_ID" =~ ^[a-z][a-z0-9-]{4,61}[a-z0-9]$ ]] \ + || fail "invalid Atlas GCP project id" +[[ "$RAW_DATASET" =~ ^[A-Za-z_][A-Za-z0-9_]{0,1023}$ ]] \ + || fail "invalid Atlas raw dataset name" +[[ "$TARGET_DATASET" =~ ^[A-Za-z_][A-Za-z0-9_]{0,1023}$ ]] \ + || fail "invalid Atlas dbt target dataset name" + +print_command gcloud config get-value project +ACTIVE_PROJECT="$(gcloud config get-value project 2>/dev/null)" +[[ "$ACTIVE_PROJECT" == "$PROJECT_ID" ]] || fail \ + "gcloud project is '${ACTIVE_PROJECT:-unset}', expected '${PROJECT_ID}'; run: gcloud config set project ${PROJECT_ID}" + +print_command gcloud projects describe "$PROJECT_ID" --format=value\(projectId\) +DESCRIBED_PROJECT="$(gcloud projects describe "$PROJECT_ID" --format='value(projectId)')" +[[ "$DESCRIBED_PROJECT" == "$PROJECT_ID" ]] \ + || fail "could not verify access to expected GCP project ${PROJECT_ID}" + +if [[ -n "${DBT_LOCATION:-}" ]]; then + LOCATION="$DBT_LOCATION" + echo "Using DBT_LOCATION=${LOCATION}" +else + print_command bq "--project_id=${PROJECT_ID}" show --format=json "${PROJECT_ID}:${RAW_DATASET}" + DATASET_JSON="$(bq "--project_id=${PROJECT_ID}" show --format=json \ + "${PROJECT_ID}:${RAW_DATASET}")" + LOCATION="$(printf '%s' "$DATASET_JSON" | "$PYTHON_BIN" -c \ + 'import json, sys; print(json.load(sys.stdin)["location"])')" \ + || fail "could not discover the ${RAW_DATASET} dataset location" + [[ -n "$LOCATION" ]] || fail "${RAW_DATASET} did not report a dataset location" + echo "Discovered ${PROJECT_ID}:${RAW_DATASET} in ${LOCATION}" +fi + +[[ "$LOCATION" =~ ^[A-Za-z0-9_-]+$ ]] || fail "invalid BigQuery location" + +case "$AUTH_METHOD" in + oauth) + ;; + service-account) + [[ -n "${GOOGLE_APPLICATION_CREDENTIALS:-}" ]] \ + || fail "GOOGLE_APPLICATION_CREDENTIALS is required for service-account auth" + [[ -f "$GOOGLE_APPLICATION_CREDENTIALS" ]] \ + || fail "GOOGLE_APPLICATION_CREDENTIALS must point to a readable external keyfile" + [[ -r "$GOOGLE_APPLICATION_CREDENTIALS" ]] \ + || fail "GOOGLE_APPLICATION_CREDENTIALS must point to a readable external keyfile" + + KEYFILE="$("$PYTHON_BIN" -c \ + 'import os, sys; print(os.path.realpath(sys.argv[1]))' \ + "$GOOGLE_APPLICATION_CREDENTIALS")" + case "$KEYFILE" in + "$WORKSPACE_ROOT"|"$WORKSPACE_ROOT"/*) + fail "service-account keyfile must be stored outside the git worktree" + ;; + esac + export GOOGLE_APPLICATION_CREDENTIALS="$KEYFILE" + echo "Using external service-account credentials (contents are not displayed)" + ;; + *) + fail "ATLAS_DBT_AUTH_METHOD must be oauth or service-account" + ;; +esac + +if [[ ! -d "$VENV_DIR" ]]; then + "$PYTHON_BIN" -c 'import ensurepip' >/dev/null 2>&1 \ + || fail "Python venv support is required (install python3-venv on Debian/Ubuntu)" + run "$PYTHON_BIN" -m venv "$VENV_DIR" +fi + +if [[ ! -x "$VENV_DIR/bin/python" ]]; then + fail "${VENV_DIR} exists but is not a valid Python virtual environment" +fi +"$VENV_DIR/bin/python" -m pip --version >/dev/null 2>&1 \ + || fail "${VENV_DIR} is incomplete or missing pip; remove it and rerun setup" +echo "Using ${VENV_DIR}" + +run "$VENV_DIR/bin/python" -m pip install --upgrade --requirement "$REQUIREMENTS_FILE" +run "$VENV_DIR/bin/dbt" deps --project-dir "$DBT_PROJECT_DIR" + +PROFILES_DIR="${DBT_PROFILES_DIR:-$HOME/.dbt}" +[[ -n "$PROFILES_DIR" ]] || fail "DBT_PROFILES_DIR resolved to an empty path" +PROFILE_PATH="${PROFILES_DIR}/profiles.yml" + +run mkdir -p "$PROFILES_DIR" +TEMP_PROFILE="$(mktemp "${PROFILES_DIR}/.profiles.yml.XXXXXX")" +cleanup() { + rm -f "$TEMP_PROFILE" +} +trap cleanup EXIT + +if [[ "$AUTH_METHOD" == "oauth" ]]; then + cat >"$TEMP_PROFILE" <"$TEMP_PROFILE" <<'EOF' +atlas_dbt: + target: dev + outputs: + dev: + type: bigquery + method: service-account + project: __ATLAS_PROJECT_ID__ + dataset: __ATLAS_TARGET_DATASET__ + location: __ATLAS_LOCATION__ + keyfile: "{{ env_var('GOOGLE_APPLICATION_CREDENTIALS') }}" + threads: 4 + priority: interactive + job_execution_timeout_seconds: 300 + job_retries: 1 +EOF + run "$PYTHON_BIN" - "$TEMP_PROFILE" "$PROJECT_ID" "$TARGET_DATASET" "$LOCATION" <<'PY' +from pathlib import Path +import sys + +path = Path(sys.argv[1]) +content = path.read_text(encoding="utf-8") +for placeholder, value in zip( + ("__ATLAS_PROJECT_ID__", "__ATLAS_TARGET_DATASET__", "__ATLAS_LOCATION__"), + sys.argv[2:], + strict=True, +): + content = content.replace(placeholder, value) +path.write_text(content, encoding="utf-8") +PY +fi + +run chmod 600 "$TEMP_PROFILE" +run mv -f "$TEMP_PROFILE" "$PROFILE_PATH" +TEMP_PROFILE="" +run chmod 600 "$PROFILE_PATH" +echo "Wrote ${PROFILE_PATH} without displaying credentials" + +run "$VENV_DIR/bin/dbt" debug \ + --project-dir "$DBT_PROJECT_DIR" \ + --profiles-dir "$PROFILES_DIR" \ + --target dev + +echo "Atlas dbt setup complete. Next commands:" +printf ' source %q\n' "${VENV_DIR}/bin/activate" +printf ' %q deps --project-dir %q\n' "${VENV_DIR}/bin/dbt" "$DBT_PROJECT_DIR" +printf ' %q parse --project-dir %q --profiles-dir %q --target dev\n' \ + "${VENV_DIR}/bin/dbt" "$DBT_PROJECT_DIR" "$PROFILES_DIR" diff --git a/scripts/simulate_failures.py b/scripts/simulate_failures.py new file mode 100755 index 0000000..1c799fb --- /dev/null +++ b/scripts/simulate_failures.py @@ -0,0 +1,68 @@ +#!/usr/bin/env python3 +"""Simulate common Atlas pipeline failure modes locally or against GCP.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from atlas.config.settings import load_settings +from atlas.generator.events import write_jsonl +from atlas.loader.bigquery import load_events_from_gcs +from atlas.validation.checks import validate_loaded_run + + +def simulate_missing_file() -> dict[str, str]: + return {"scenario": "missing_file", "status": "FAILED", "message": "Source file not found"} + + +def simulate_bad_schema(tmp_path: Path) -> Path: + bad_path = tmp_path / "bad_schema.jsonl" + write_jsonl(bad_path, iter([{"event_id": "1", "unexpected": "field"}])) + return bad_path + + +def main() -> int: + parser = argparse.ArgumentParser(description="Simulate Atlas failure scenarios") + parser.add_argument( + "--scenario", + choices=["missing_file", "bad_schema", "duplicate_upload"], + required=True, + ) + args = parser.parse_args() + + settings = load_settings() + if args.scenario == "missing_file": + print(json.dumps(simulate_missing_file(), indent=2)) + return 1 + + if args.scenario == "bad_schema": + bad_path = simulate_bad_schema(Path(settings.generator.output_dir)) + print(json.dumps({"scenario": "bad_schema", "path": str(bad_path)}, indent=2)) + return 0 + + if args.scenario == "duplicate_upload": + print( + json.dumps( + { + "scenario": "duplicate_upload", + "status": "SKIPPED", + "message": "Requires live GCS credentials and --approve-provision", + }, + indent=2, + ) + ) + return 0 + + _ = load_events_from_gcs + _ = validate_loaded_run + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/start_airflow_local.sh b/scripts/start_airflow_local.sh new file mode 100755 index 0000000..ad924bb --- /dev/null +++ b/scripts/start_airflow_local.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Start local Airflow standalone (tmux when available, nohup fallback for Cloud Shell). +set -euo pipefail + +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +# shellcheck disable=SC1091 +source "${ATLAS_ROOT}/scripts/airflow_env.sh" + +SESSION="atlas-airflow-local" +PIDFILE="${AIRFLOW_HOME}/standalone.pid" +LOGFILE="${AIRFLOW_HOME}/standalone.log" +mkdir -p "$AIRFLOW_HOME" + +if [[ -f "$PIDFILE" ]]; then + pid="$(cat "$PIDFILE")" + if kill -0 "$pid" 2>/dev/null; then + echo "Airflow standalone already running (pid=$pid)" + exit 0 + fi + rm -f "$PIDFILE" +fi + +start_standalone() { + nohup airflow standalone >>"$LOGFILE" 2>&1 & + echo $! >"$PIDFILE" + echo "Started Airflow standalone (pid=$(cat "$PIDFILE"), log=$LOGFILE)" +} + +if command -v tmux >/dev/null 2>&1; then + TMUX_CMD=(tmux) + if [[ -f /exec-daemon/tmux.portal.conf ]]; then + TMUX_CMD=(tmux -f /exec-daemon/tmux.portal.conf) + fi + if "${TMUX_CMD[@]}" has-session -t "=$SESSION" 2>/dev/null; then + echo "Airflow session already running: $SESSION" + exit 0 + fi + if "${TMUX_CMD[@]}" new-session -d -s "$SESSION" -c "$ATLAS_ROOT" -- \ + bash -lc "source '${ATLAS_ROOT}/scripts/airflow_env.sh' && airflow standalone"; then + echo "Started Airflow standalone in tmux session: $SESSION" + exit 0 + fi + echo "tmux unavailable or failed; falling back to nohup" >&2 +fi + +start_standalone diff --git a/scripts/stop_airflow_local.sh b/scripts/stop_airflow_local.sh new file mode 100755 index 0000000..7a760ec --- /dev/null +++ b/scripts/stop_airflow_local.sh @@ -0,0 +1,25 @@ +#!/usr/bin/env bash +set -euo pipefail + +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +PIDFILE="${ATLAS_ROOT}/.airflow/standalone.pid" +SESSION="atlas-airflow-local" + +if command -v tmux >/dev/null 2>&1; then + TMUX_CMD=(tmux) + if [[ -f /exec-daemon/tmux.portal.conf ]]; then + TMUX_CMD=(tmux -f /exec-daemon/tmux.portal.conf) + fi + "${TMUX_CMD[@]}" kill-session -t "$SESSION" 2>/dev/null || true +fi + +if [[ -f "$PIDFILE" ]]; then + pid="$(cat "$PIDFILE")" + if kill -0 "$pid" 2>/dev/null; then + kill "$pid" 2>/dev/null || true + echo "Stopped Airflow standalone (pid=$pid)" + fi + rm -f "$PIDFILE" +else + echo "No Airflow standalone pid file found" +fi diff --git a/scripts/test_airflow_sprint3.sh b/scripts/test_airflow_sprint3.sh new file mode 100755 index 0000000..e1f05cb --- /dev/null +++ b/scripts/test_airflow_sprint3.sh @@ -0,0 +1,36 @@ +#!/usr/bin/env bash +set -euo pipefail +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +cd "$ATLAS_ROOT" +# shellcheck disable=SC1091 +source "${ATLAS_ROOT}/scripts/airflow_env.sh" + +echo "== Shell syntax ==" +find scripts -name '*.sh' -print0 | xargs -0 -I{} bash -n {} + +echo "== Installing Atlas test dependencies ==" +pip install -q -r requirements.txt pytest + +echo "== Sprint 1/2/3 unit tests ==" +python3 -m pytest tests/unit tests/airflow -q + +echo "== DAG import check (when Airflow installed) ==" +if command -v airflow >/dev/null 2>&1; then + import_errors="$(airflow dags list-import-errors 2>/dev/null || true)" + # Airflow prints "No data found" when there are no import errors; any DAG + # filepath in the output means at least one module failed to import. + if grep -qE '\.py' <<<"$import_errors"; then + echo "FAIL: DAG import errors detected:" >&2 + echo "$import_errors" >&2 + exit 1 + fi + if ! airflow dags list 2>/dev/null | grep -q atlas_batch_pipeline; then + echo "FAIL: atlas_batch_pipeline is not registered" >&2 + exit 1 + fi + echo "atlas_batch_pipeline registered with no import errors" +else + echo "airflow CLI not installed; DAG registration check skipped (parse-safety pytest gates above still ran)" +fi + +echo "Sprint 3 static test gate complete" diff --git a/scripts/upload_events.py b/scripts/upload_events.py new file mode 100755 index 0000000..702ed8d --- /dev/null +++ b/scripts/upload_events.py @@ -0,0 +1,48 @@ +#!/usr/bin/env python3 +"""Upload generated Atlas events to Cloud Storage.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from atlas.config.settings import load_settings +from atlas.ingestion.upload import upload_events_file +from atlas.logging.structured import new_pipeline_run_id + + +def main() -> int: + parser = argparse.ArgumentParser(description="Upload Atlas JSONL to GCS") + parser.add_argument("--local-path", required=True, type=Path) + parser.add_argument("--event-date", required=True) + parser.add_argument("--run-id", default=new_pipeline_run_id()) + parser.add_argument("--batch-id") + parser.add_argument("--expected-checksum") + parser.add_argument( + "--fail-once", + action="store_true", + help="Inject a transient failure for retry testing (development only)", + ) + args = parser.parse_args() + + settings = load_settings() + result = upload_events_file( + settings, + args.local_path, + args.event_date, + args.run_id, + batch_id=args.batch_id, + expected_checksum=args.expected_checksum, + fail_once=args.fail_once, + ) + print(json.dumps(result.__dict__, indent=2, default=str)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/validate_atlas_deployment.sh b/scripts/validate_atlas_deployment.sh new file mode 100755 index 0000000..75a8385 --- /dev/null +++ b/scripts/validate_atlas_deployment.sh @@ -0,0 +1,116 @@ +#!/usr/bin/env bash +# Post-deployment smoke validation for Atlas on Composer (Sprint 4, Phase 13). +# +# Usage: +# validate_atlas_deployment.sh --git-sha --deployment-id +# --batch-id --pipeline-run-id --processing-date +# +# Validates the deployed system, not the upload: a successful upload is not a +# successful deployment. Exits nonzero when any check fails. +set -uo pipefail + +ATLAS_SCRIPTS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=lib_atlas_deploy.sh +source "${ATLAS_SCRIPTS_ROOT}/lib_atlas_deploy.sh" +export PYTHONPATH="${ATLAS_SCRIPTS_ROOT}/../src:${PYTHONPATH:-}" + +GIT_SHA="" DEPLOYMENT_ID="" BATCH_ID="" PIPELINE_RUN_ID="" PROCESSING_DATE="" +while [[ $# -gt 0 ]]; do + case "$1" in + --git-sha) GIT_SHA="${2:?}"; shift 2 ;; + --deployment-id) DEPLOYMENT_ID="${2:?}"; shift 2 ;; + --batch-id) BATCH_ID="${2:?}"; shift 2 ;; + --pipeline-run-id) PIPELINE_RUN_ID="${2:?}"; shift 2 ;; + --processing-date) PROCESSING_DATE="${2:?}"; shift 2 ;; + *) echo "Unknown argument: $1" >&2; exit 2 ;; + esac +done +for required in GIT_SHA BATCH_ID PIPELINE_RUN_ID PROCESSING_DATE; do + [[ -n "${!required}" ]] || { echo "--${required,,} is required" >&2; exit 2; } +done + +EVENTS_BUCKET="${ATLAS_GCS_BUCKET:-atlas-raw-events-${ATLAS_PROJECT_ID}}" +EXPECTED_ROWS=50000 +FAILURES=0 + +check() { # name condition_result message + local name="$1" ok="$2" message="$3" + if [[ "$ok" == "0" ]]; then + echo "[PASS] ${name}: ${message}" + else + echo "[FAIL] ${name}: ${message}" >&2 + FAILURES=$((FAILURES + 1)) + fi +} + +# --- 1. Expected DAG imported ------------------------------------------------------- +DAGS_OUT="$(composer_airflow dags list -o plain || true)" +echo "$DAGS_OUT" | awk '{print $1}' | grep -qx "atlas_batch_pipeline" +check "dag_imported" "$?" "atlas_batch_pipeline present in Composer" + +IMPORT_ERRORS="$(composer_airflow dags list-import-errors -o plain || true)" +if echo "$IMPORT_ERRORS" | grep -q "project_atlas"; then + check "dag_import_errors" 1 "import errors present for project_atlas" +else + check "dag_import_errors" 0 "no import errors for project_atlas" +fi + +# --- 2. Expected git SHA deployed ---------------------------------------------------- +BUCKET="$(composer_bucket)" +DEPLOYED_SHA="$(gcloud storage cat "${BUCKET}/data/current/release-manifest.json" 2>/dev/null \ + | python3 -c 'import json,sys; print(json.load(sys.stdin)["git_sha"])' || echo "unknown")" +ok=1; [[ "$DEPLOYED_SHA" == "$GIT_SHA" ]] && ok=0 +check "deployed_sha" "$ok" "current runtime manifest git_sha=${DEPLOYED_SHA}" + +# --- 3. Smoke run reached terminal SUCCESS in Airflow --------------------------------- +# Airflow 3 prints "state, {conf...}" for runs triggered with --conf, so match +# the leading token rather than anchoring the whole line. +RUN_STATE="$(composer_airflow dags state atlas_batch_pipeline "smoke__${DEPLOYMENT_ID}" \ + | grep -Eo '^(success|failed|running|queued)\b' | tail -1 || true)" +ok=1; [[ "$RUN_STATE" == "success" ]] && ok=0 +check "airflow_terminal_success" "$ok" "smoke dag run state=${RUN_STATE:-unknown}" + +# --- 4-6. Raw batch exists, correct count, no duplicate load --------------------------- +RAW_COUNT="$(bq_scalar "SELECT COUNT(1) FROM \`${ATLAS_PROJECT_ID}.atlas_raw.events\` WHERE batch_id = '${BATCH_ID}'")" +ok=1; [[ "$RAW_COUNT" == "$EXPECTED_ROWS" ]] && ok=0 +check "raw_batch_count" "$ok" "raw rows for ${BATCH_ID}: ${RAW_COUNT} (expected ${EXPECTED_ROWS})" + +INGESTION_RUNS="$(bq_scalar "SELECT COUNT(DISTINCT pipeline_run_id) FROM \`${ATLAS_PROJECT_ID}.atlas_raw.events\` WHERE batch_id = '${BATCH_ID}'")" +ok=1; [[ "$INGESTION_RUNS" == "1" ]] && ok=0 +check "no_duplicate_load" "$ok" "distinct ingestion runs for batch: ${INGESTION_RUNS}" + +gcloud storage ls "gs://${EVENTS_BUCKET}/raw/event_date=${PROCESSING_DATE}/batch_id=${BATCH_ID}/events.jsonl" >/dev/null 2>&1 +check "gcs_object_exists" "$?" "raw JSONL object present in gs://${EVENTS_BUCKET}" + +gcloud storage ls "${BUCKET}/data/current/data/runs/${BATCH_ID}/manifest.json" >/dev/null 2>&1 \ + || gcloud storage ls "${BUCKET}/data/current/data/runs/${BATCH_ID}/" >/dev/null 2>&1 +check "batch_manifest_exists" "$?" "batch manifest/artifacts present in Composer data path" + +# --- 7. Success marker ------------------------------------------------------------------- +gcloud storage ls "${BUCKET}/data/current/data/runs/${BATCH_ID}/success.marker" >/dev/null 2>&1 +check "success_marker" "$?" "success.marker present for ${BATCH_ID}" + +# --- 8. Warehouse reconciliation (accepted+rejected=raw, facts, marts) ---------------------- +python3 "${ATLAS_SCRIPTS_ROOT}/atlas_step_runner.py" validate_warehouse \ + "{\"batch_id\": \"${BATCH_ID}\", \"processing_date\": \"${PROCESSING_DATE}\"}" >/tmp/smoke-warehouse.json +check "warehouse_reconciliation" "$?" "batch-scoped raw/classified/fact/mart reconciliation" + +# --- 9. atlas_ops.pipeline_runs SUCCESS row -------------------------------------------------- +AUDIT_STATUS="$(bq_scalar "SELECT status FROM \`${ATLAS_PROJECT_ID}.atlas_ops.pipeline_runs\` WHERE pipeline_run_id = '${PIPELINE_RUN_ID}'")" +ok=1; [[ "$AUDIT_STATUS" == "SUCCESS" ]] && ok=0 +check "pipeline_runs_success" "$ok" "pipeline_runs status=${AUDIT_STATUS:-missing}" + +# --- 10. atlas_ops.deployments references this smoke run -------------------------------------- +if [[ -n "$DEPLOYMENT_ID" ]]; then + DEPLOY_ROW="$(bq_scalar "SELECT COUNT(1) FROM \`${ATLAS_PROJECT_ID}.atlas_ops.deployments\` WHERE deployment_id = '${DEPLOYMENT_ID}'")" + ok=1; [[ "$DEPLOY_ROW" == "1" ]] && ok=0 + check "deployments_row" "$ok" "atlas_ops.deployments rows for ${DEPLOYMENT_ID}: ${DEPLOY_ROW}" +fi + +echo "" +if [[ "$FAILURES" -eq 0 ]]; then + echo "SMOKE VALIDATION PASSED (git_sha ${GIT_SHA:0:12}, batch ${BATCH_ID})" + exit 0 +fi +echo "SMOKE VALIDATION FAILED: ${FAILURES} check(s) failed" >&2 +exit 1 diff --git a/scripts/validate_ci.sh b/scripts/validate_ci.sh new file mode 100755 index 0000000..fad78cc --- /dev/null +++ b/scripts/validate_ci.sh @@ -0,0 +1,794 @@ +#!/usr/bin/env bash +# Canonical Atlas validation entry point (Sprint 4). +# +# The single validation contract shared by Cursor Cloud Agents, local +# developers, GitHub Actions, and release tooling. +# +# Usage: +# bash scripts/validate_ci.sh --mode static # no GCP credentials needed +# bash scripts/validate_ci.sh --mode integration # isolated GCP resources +# bash scripts/validate_ci.sh --mode all +# +# Behavior: +# - exits nonzero when any required gate fails +# - prints a concise gate summary +# - writes machine-readable results to logs/ci/validate-ci-results.json +# - never prints secret values and never mutates canonical GCP data in +# static mode +set -uo pipefail + +ATLAS_ROOT="${ATLAS_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" +cd "$ATLAS_ROOT" || exit 1 + +airflow_installed() { + # The local airflow/ docs folder shadows the package as a namespace import, + # so probe distribution metadata instead of `import airflow`. + python3 -c "from importlib.metadata import version; version('apache-airflow')" 2>/dev/null +} + +MODE="static" +# Gate group lets one CI job run its slice of the canonical contract: +# all (default) | security-shell | python | airflow | dbt +# When a specific group is requested, its toolchain is REQUIRED: a missing +# tool fails the gate instead of skipping it. +GROUP="${ATLAS_CI_GATE_GROUP:-all}" +while [[ $# -gt 0 ]]; do + case "$1" in + --mode) MODE="$2"; shift 2 ;; + --group) GROUP="$2"; shift 2 ;; + *) echo "Unknown arg: $1" >&2; exit 1 ;; + esac +done +case "$MODE" in + static|integration|all) ;; + *) echo "Invalid --mode '$MODE' (static|integration|all)" >&2; exit 1 ;; +esac +case "$GROUP" in + all|security-shell|python|airflow|dbt) ;; + *) echo "Invalid --group '$GROUP' (all|security-shell|python|airflow|dbt)" >&2; exit 1 ;; +esac + +in_group() { + # in_group : true when GROUP is all or one of the arguments. + [[ "$GROUP" == "all" ]] && return 0 + local candidate + for candidate in "$@"; do + [[ "$GROUP" == "$candidate" ]] && return 0 + done + return 1 +} + +RESULTS_DIR="${ATLAS_ROOT}/logs/ci" +mkdir -p "$RESULTS_DIR" +RESULTS_FILE="${RESULTS_DIR}/validate-ci-results.json" +: >"${RESULTS_FILE}.tmp" + +GATE_NAMES=() +GATE_STATUSES=() +FAILED=0 + +run_gate() { + # run_gate + local name="$1" + shift + local started ended status + started="$(date -u +%Y-%m-%dT%H:%M:%SZ)" + echo "" + echo "===== GATE: ${name} =====" + if "$@"; then + status="PASS" + else + status="FAIL" + FAILED=1 + fi + ended="$(date -u +%Y-%m-%dT%H:%M:%SZ)" + GATE_NAMES+=("$name") + GATE_STATUSES+=("$status") + printf '{"gate":"%s","status":"%s","started_at":"%s","ended_at":"%s","mode":"%s"}\n' \ + "$name" "$status" "$started" "$ended" "$MODE" >>"${RESULTS_FILE}.tmp" + echo "----- ${name}: ${status} -----" +} + +skip_gate() { + local name="$1" reason="$2" + GATE_NAMES+=("$name") + GATE_STATUSES+=("SKIPPED") + printf '{"gate":"%s","status":"SKIPPED","reason":"%s","mode":"%s"}\n' \ + "$name" "$reason" "$MODE" >>"${RESULTS_FILE}.tmp" + echo "===== GATE: ${name} SKIPPED (${reason}) =====" +} + +# --------------------------------------------------------------------------- +# Gate implementations +# --------------------------------------------------------------------------- + +gate_secret_scan() { + # Tracked-content scan. Detector patterns and sanitizer fixtures live in + # audit.py and the artifact-platform tests; exclude only those exact files. + local matches + matches="$(git -C "$ATLAS_ROOT/.." grep -nIE \ + '(-----BEGIN (RSA |EC |OPENSSH )?PRIVATE KEY-----|AIza[0-9A-Za-z_-]{35}|"type": "service_account")' \ + -- 'project-atlas' \ + ':!scripts/validate_ci.sh' \ + ':!src/atlas/ops/audit.py' \ + ':!tests/unit/test_audit.py' \ + ':!tests/unit/test_security_policy.py' \ + ':!artifact-platform/tests/*' \ + ':!artifact-platform/src/artifact_platform/secrets.py' \ + 2>/dev/null || true)" + if [[ -n "$matches" ]]; then + echo "Potential credentials detected in tracked files (values not shown):" >&2 + # Print file:line only — never the matched content. + cut -d: -f1,2 <<<"$matches" >&2 + return 1 + fi + # Untracked credential files inside the worktree. + local untracked + untracked="$(git -C "$ATLAS_ROOT/.." status --porcelain --untracked-files=all \ + | awk '{print $2}' \ + | grep -E '(^|/)(.*service.?account.*\.json|.*credentials.*\.json|.*\.pem|\.gcp/)' || true)" + if [[ -n "$untracked" ]]; then + echo "Untracked credential-like files present in the worktree:" >&2 + echo "$untracked" >&2 + return 1 + fi + echo "No tracked or untracked credential material detected." +} + +gate_shell_syntax() { + find scripts -name '*.sh' -print0 | xargs -0 -n1 bash -n +} + +gate_shellcheck() { + # SC1091: sourced files resolved at runtime. Informational severity only. + shellcheck --severity=warning --exclude=SC1091 scripts/*.sh +} + +gate_python_format() { + ruff format --check src/atlas scripts dags tests +} + +gate_python_lint() { + ruff check src/atlas scripts dags tests +} + +gate_python_types() { + mypy --config-file mypy.ini +} + +gate_python_tests() { + PYTHONPATH="src:dags" python3 -m pytest tests/unit tests/airflow -q +} + +gate_workflow_yaml() { + local workflows_dir="${ATLAS_ROOT}/../.github/workflows" + if [[ ! -d "$workflows_dir" ]]; then + echo "No workflows directory yet" + return 0 + fi + yamllint -d "{extends: default, rules: {line-length: {max: 140}, truthy: disable, document-start: disable, comments: {min-spaces-from-content: 1}}}" "$workflows_dir" +} + +gate_config_validation() { + python3 - <<'PY' +import sys + +sys.path.insert(0, "src") +from atlas.config.settings import load_settings + +settings = load_settings() +assert settings.gcp.project_id, "project_id must resolve" +assert settings.validation.expected_event_count > 0 +profile = settings.anomaly_profile +for name in ( + "duplicate_event_ids", + "null_user_ids", + "invalid_country_codes", + "future_timestamps", + "late_arriving_events", +): + assert profile.expected_count(name) >= 0, name +print("config OK:", settings.config_path.name, settings.anomaly_path.name) +PY +} + +gate_observability_config() { + # Sprint 5 static validation of observability artifacts (no credentials). + python3 - <<'PY' +import json +import re +import sys +from pathlib import Path + +import yaml + +sys.path.insert(0, "src") + +# 1. observability.yaml parses with sane threshold ordering. +config = yaml.safe_load(Path("config/observability.yaml").read_text()) +assert isinstance(config["monitoring_enabled"], bool) +assert config["freshness"]["warn_seconds"] < config["freshness"]["fail_seconds"] +assert config["volume"]["warn_deviation"] < config["volume"]["fail_deviation"] +assert config["rejection_rate"]["warn"] < config["rejection_rate"]["fail"] +assert config["cost"]["warn_ratio"] < config["cost"]["fail_ratio"] +assert config["runtime_mode"] in {"normal", "drill"} +assert config["drill_overrides"] in ({}, None), "drill overrides must never merge to main" + +# 2. Metric descriptor catalog loads and stays within the cardinality budget +# (deep checks live in tests/unit/test_observability_metrics.py). +from atlas.observability.metrics import load_catalog + +catalog = load_catalog() +assert len(catalog) >= 10 +forbidden = {"pipeline_run_id", "batch_id", "deployment_id", "error_message"} +for metric_type, spec in catalog.items(): + assert not (set(spec["labels"]) & forbidden), metric_type + +# 3. Schema manifest parses and covers the governed tables. +manifest = json.loads(Path("observability/schema/expected-schemas.json").read_text()) +assert len(manifest["tables"]) >= 10 + +# 4. Alert policies: valid JSON, channel placeholder only (no committed +# channel ids/addresses), runbook anchors resolve, required metadata. +runbook = Path("docs/observability-runbook-sprint5.md").read_text().lower() +alert_files = sorted(Path("observability/alerts").glob("*.json")) +assert len(alert_files) == 10, [f.name for f in alert_files] +for f in alert_files: + raw = f.read_text() + assert "${NOTIFICATION_CHANNEL}" in raw, f"{f.name}: placeholder missing" + assert "notificationChannels/" not in raw.replace("${NOTIFICATION_CHANNEL}", ""), f.name + assert "@" not in raw, f"{f.name}: possible committed address" + policy = json.loads(raw) + assert policy["displayName"].startswith("Atlas: ") + assert policy["userLabels"]["managed_by"] == "atlas-sprint5" + assert policy["userLabels"]["severity"] in {"critical", "warning"} + doc = policy["documentation"]["content"] + anchors = re.findall(r"#(alert-[a-z0-9-]+)", doc) + assert anchors, f"{f.name}: no runbook anchor" + for anchor in anchors: + heading = "## alert: " + anchor.removeprefix("alert-").replace("-", " ") + assert heading in runbook, f"{f.name}: runbook heading missing for {anchor}" + +# 5. Dashboard JSON: parses, stable name, section headers sized correctly. +dashboard = json.loads(Path("observability/dashboards/atlas-operations.json").read_text()) +assert dashboard["displayName"] == "Atlas Operations" +tiles = dashboard["mosaicLayout"]["tiles"] +assert len(tiles) >= 20 +for tile in tiles: + if "sectionHeader" in tile["widget"]: + assert tile["height"] in (3, 4), "section headers need height 3-4" + +# 6. Log routing definitions parse; sink filter non-empty after comments. +for name in ("log-bucket.json", "log-view.json"): + json.loads(Path(f"observability/logging/{name}").read_text()) +filter_lines = [ + line + for line in Path("observability/logging/sink-filter.txt").read_text().splitlines() + if line.strip() and not line.startswith("--") +] +assert filter_lines, "sink filter empty" + +print( + f"observability OK: config, {len(catalog)} metrics, " + f"{len(manifest['tables'])} schema tables, {len(alert_files)} alerts, " + f"{len(tiles)}-tile dashboard, log routing" +) +PY +} + +gate_failure_injection() { + # Sprint 6 static validation: the failure-scenario catalog is schema-valid + # and fault injection can never activate in normal execution. + python3 - <<'PY' +import re +import sys +from pathlib import Path + +sys.path.insert(0, "src") + +# 1. Catalog schema validation (every scenario fully specified, no CRITICAL, +# injection approval always required, cost/duration ceilings present). +from atlas.failure_injection.registry import load_catalog, validate_catalog + +errors = validate_catalog(load_catalog()) +if errors: + for error in errors: + print(f"INVALID {error}", file=sys.stderr) + raise SystemExit(1) + +# 2. Default-off proof: with a clean environment, injection is inert. +from atlas.failure_injection.framework import SCENARIO_VAR, injection_active_for, is_injection_requested + +assert is_injection_requested(env={}) is False +assert injection_active_for("S6-ING-001", env={}) is False +assert injection_active_for("S6-ING-001", env={"ATLAS_APPROVE_FAILURE_INJECTION": "true"}) is False + +# 3. No production file hardcodes the activation variable to a scenario: +# the explicit test-only parameter must come from the operator, never the +# repository. (Tests and the framework itself may reference the name.) +pattern = re.compile(rf"{SCENARIO_VAR}\s*=\s*['\"]S6-") +violations = [] +for root in ("dags", "scripts", "config", ".env", "airflow"): + base = Path(root) + if not base.exists(): + continue + for path in base.rglob("*"): + if path.is_file() and path.suffix in {".py", ".sh", ".yaml", ".yml", ".cfg", ".env", ""}: + try: + text = path.read_text(encoding="utf-8") + except (UnicodeDecodeError, IsADirectoryError): + continue + if pattern.search(text): + violations.append(str(path)) +assert not violations, f"fault injection hardcoded in normal execution paths: {violations}" + +catalog = load_catalog() +print(f"failure injection OK: {len(catalog['scenarios'])} scenarios valid, disabled by default") +PY +} + +gate_sql_migrations() { + python3 - <<'PY' +from pathlib import Path + +sql_dir = Path("sql") +files = sorted(sql_dir.glob("*.sql")) +assert files, "sql/ must contain DDL files" +for path in files: + content = path.read_text(encoding="utf-8") + assert content.strip(), f"{path} is empty" + lowered = content.lower() + banned = ("drop table", "drop schema", "truncate table", "delete from") + for phrase in banned: + assert phrase not in lowered, f"{path} contains destructive statement: {phrase}" +print(f"{len(files)} SQL files validated (non-empty, additive-only)") +PY +} + +gate_airflow_environment() { + python3 - <<'PY' || return 1 +from importlib.metadata import version + +core = version("apache-airflow") +google_provider = version("apache-airflow-providers-google") +standard_provider = version("apache-airflow-providers-standard") +assert core == "3.1.7", core +assert google_provider == "20.0.0", google_provider +assert standard_provider == "1.12.1", standard_provider +print(f"airflow {core} / google {google_provider} / standard {standard_provider}") +PY + pip check +} + +gate_dag_import() { + PYTHONPATH="src:dags" python3 - <<'PY' +import os + +os.environ.setdefault("AIRFLOW__CORE__LOAD_EXAMPLES", "False") +os.environ.setdefault("AIRFLOW__CORE__DAGS_FOLDER", "dags") + +from airflow.models.dagbag import DagBag + +# safe_mode=False parses every .py file under dags/, not only files matching +# the "airflow"+"dag" keyword heuristic — a broken helper module must fail CI +# even though the scheduler's safe mode would silently skip it. +bag = DagBag(dag_folder="dags", include_examples=False, safe_mode=False) +if bag.import_errors: + for path, error in bag.import_errors.items(): + print(f"IMPORT ERROR {path}:\n{error}") + raise SystemExit(1) +assert "atlas_batch_pipeline" in bag.dags, sorted(bag.dags) +dag = bag.dags["atlas_batch_pipeline"] +required_tasks = { + "resolve_run_context", + "ensure_audit_resources", + "start_run_audit", + "preflight_environment", + "generate_events", + "upload_events", + "load_bigquery_raw", + "validate_raw_load", + "dbt_seed", + "dbt_source_freshness", + "dbt_build", + "validate_warehouse", + "publish_success_marker", + "write_run_summary", +} +missing = required_tasks - {t.task_id for t in dag.tasks} +assert not missing, f"missing tasks: {sorted(missing)}" +print(f"atlas_batch_pipeline imported with {len(dag.tasks)} tasks and no import errors") +PY +} + +gate_dbt_static() { + local dbt_dir="${DBT_PROJECT_DIR:-${ATLAS_ROOT}/dbt/atlas_dbt}" + local profiles_dir="${ATLAS_ROOT}/logs/ci/dbt-profiles" + mkdir -p "$profiles_dir" + # Parse-only profile: dbt parse never opens a warehouse connection, so a + # placeholder oauth profile keeps static mode credential-free. + cat >"${profiles_dir}/profiles.yml" <<'YML' +atlas_dbt: + target: ci_static + outputs: + ci_static: + type: bigquery + method: oauth + project: ci-static-placeholder + dataset: atlas_ci_static + threads: 1 + location: US +YML + # sources.yml resolves the project from the environment; a placeholder keeps + # static mode credential-free (parse never opens a connection). + (cd "$dbt_dir" \ + && ATLAS_GCP_PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-ci-static-placeholder}" dbt deps --quiet \ + && ATLAS_GCP_PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-ci-static-placeholder}" dbt parse --profiles-dir "$profiles_dir" --no-partial-parse) +} + +gate_schema_compatibility() { + # Sprint 7 Phase 4/13: applied-migration checksums are immutable and the + # committed schema baseline matches a fresh generation. Offline. + PYTHONPATH="${ATLAS_ROOT}/src" python3 - <<'PY' +import json +import sys +from pathlib import Path + +from atlas.config.settings import atlas_root +from atlas.ops.migrations import load_manifest +from atlas.governance import schema_check as sc + +root = atlas_root() +failures = [] + +# 1. Migration checksum immutability against the committed lock. +lock_path = root / "sql/migrations/checksums.lock" +if not lock_path.exists(): + failures.append("sql/migrations/checksums.lock missing") +else: + lock = json.loads(lock_path.read_text()) + locked = lock.get("checksums", {}) + for m in load_manifest(): + if m.migration_id not in locked: + failures.append(f"migration {m.migration_id} not in checksum lock (append its entry)") + elif locked[m.migration_id] != m.checksum: + failures.append( + f"migration {m.migration_id} checksum changed " + f"(lock {locked[m.migration_id][:12]}…, file {m.checksum[:12]}…) — " + "applied migrations are immutable" + ) + +# 2. Schema baseline manifest drift. +baseline_path = root / "governance/schemas/manifests/baseline.json" +if not baseline_path.exists(): + failures.append("governance/schemas/manifests/baseline.json missing") +else: + committed = json.loads(baseline_path.read_text()) + fresh = sc.generate_manifest() + if committed != fresh: + failures.append( + "schema baseline stale; run " + "`python -m atlas.governance.schema_check --generate " + "governance/schemas/manifests/baseline.json`" + ) + +if failures: + print("SCHEMA COMPATIBILITY FAIL:") + for f in failures: + print(f" - {f}") + sys.exit(1) +print("schema compatibility OK: migration checksums locked, baseline current") +PY +} + +gate_lineage_impact() { + # Sprint 7 Phase 5/13: committed lineage matches a fresh generation and the + # source-to-mart chain is intact. Offline. + PYTHONPATH="${ATLAS_ROOT}/src" python3 - <<'PY' +import json +import sys + +from atlas.config.settings import atlas_root +from atlas.governance import lineage + +path = atlas_root() / "governance/generated/lineage.json" +if not path.exists(): + print("lineage.json missing; run `python -m atlas.governance.lineage --output " + "governance/generated/lineage.json`") + sys.exit(1) +lineage.build_lineage.cache_clear() +graph = lineage.build_lineage() +fresh = lineage.to_dict(graph) +committed = json.loads(path.read_text()) +if committed != fresh: + print("LINEAGE DRIFT: committed lineage.json is stale; regenerate it") + sys.exit(1) +downstream = graph.transitive_downstream("atlas_raw.events") +for required in ("stg_events", "fct_events", "mart_daily_event_metrics"): + if required not in downstream: + print(f"LINEAGE BROKEN: {required} not reachable from atlas_raw.events") + sys.exit(1) +print(f"lineage OK: {len(fresh['nodes'])} nodes, {fresh['edge_count']} edges, source->mart intact") +PY +} + +gate_security_policy() { + # Sprint 7 Phase 8/13: managed IAM defs grant no prohibited roles / SA keys, + # and governed artifacts commit no secret-like values. Offline. + PYTHONPATH="${ATLAS_ROOT}/src" python3 - <<'PY' +import sys + +from atlas.governance.security_policy import scan_data_exposure, scan_managed_iam + +findings = scan_managed_iam() + scan_data_exposure() +if findings: + print("SECURITY POLICY FAIL (values not shown):") + for path, lineno, reason in findings: + print(f" - {path}:{lineno}: {reason}") + sys.exit(1) +print("security policy OK: no prohibited IAM roles/keys, no secret-like values in governed artifacts") +PY +} + +gate_performance_cost() { + # Sprint 7 Phase 12/13: cost-control config is valid and coherent. Offline. + PYTHONPATH="${ATLAS_ROOT}/src" python3 - <<'PY' +import sys + +from atlas.observability import cost_guard + +failures = [] +try: + controls = cost_guard.load_cost_controls() +except Exception as exc: # noqa: BLE001 + print(f"PERFORMANCE/COST FAIL: cannot load cost_controls.yaml: {exc}") + sys.exit(1) + +envs = controls.get("environments", {}) +if not envs: + failures.append("cost_controls.yaml has no environments") +required = ( + "max_query_bytes", + "max_performance_suite_bytes", + "max_backfill_days", + "require_partition_filter_assets", + "temporary_dataset_ttl_hours", + "log_retention_days", + "release_retention_policy", +) +for env, c in envs.items(): + for f in required: + if f not in c: + failures.append(f"{env}: missing '{f}'") + if isinstance(c.get("max_query_bytes"), int) and isinstance( + c.get("max_performance_suite_bytes"), int + ): + if c["max_query_bytes"] > c["max_performance_suite_bytes"]: + failures.append(f"{env}: max_query_bytes exceeds suite ceiling") + +if failures: + print("PERFORMANCE/COST FAIL:") + for f in failures: + print(f" - {f}") + sys.exit(1) +print(f"performance/cost OK: {len(envs)} environments, ceilings coherent") +PY +} + +gate_governance() { + # Sprint 7 Phase 1/13: governance metadata is complete, uses one source of + # truth, and the generated catalog is not stale. Offline, no credentials. + PYTHONPATH="${ATLAS_ROOT}/src" python3 - <<'PY' +import sys + +from atlas.governance.registry import validate_governance +from atlas.governance.catalog import _committed_matches, build_catalog +from atlas.governance.retention import validate_retention_config + +errors = validate_governance() + validate_retention_config() +if errors: + print("GOVERNANCE INVALID:") + for e in errors: + print(f" - {e}") + sys.exit(1) +catalog = build_catalog() +if not _committed_matches(): + print("GOVERNANCE DRIFT: committed catalog stale; run " + "`python -m atlas.governance.catalog generate`") + sys.exit(1) +print(f"governance OK: {catalog['asset_count']} assets, one source of truth, catalog current") +PY +} + +gate_reference_handoff() { + # Sprint 8 Phase 18: the reference-architecture package and handoff artifacts + # are internally consistent and free of hidden-context dependencies. Offline. + PYTHONPATH="${ATLAS_ROOT}/src" python3 -m atlas.reference.validate || return 1 + python3 "${ATLAS_ROOT}/scripts/validate_public_extraction.py" >/dev/null || return 1 + ATLAS_ROOT="${ATLAS_ROOT}" python3 - <<'PY' +import os +import re +import subprocess +import sys +from pathlib import Path + +root = Path(os.environ["ATLAS_ROOT"]) +errors: list[str] = [] + +# 1. START_HERE is the canonical entry point. +if not (root / "START_HERE.md").exists(): + errors.append("START_HERE.md is missing") + +# 2. Required reference + handoff documents exist. +required = [ + "docs/reference-architecture/README.md", + "docs/reference-architecture/reference-manifest.yml", + "docs/reference-architecture/architecture-invariants.md", + "docs/reference-architecture/evidence-index.md", + "docs/reference-architecture/unresolved-risks.md", + "docs/reference-architecture/capability-evidence-map.md", + "docs/reference-architecture/public-extraction-review.md", + "docs/handoff/operator-onboarding.md", + "docs/handoff/agent-onboarding.md", + "docs/handoff/clean-clone-reproduction.md", + "docs/handoff/handoff-scorecard.md", + "docs/evidence-sprint8/clean-clone-results.md", + "docs/evidence-sprint8/independent-handoff-results.md", + "governance/unresolved_risks.yml", + "config/public_extraction_manifest.yml", +] +for rel in required: + if not (root / rel).exists(): + errors.append(f"required handoff artifact missing: {rel}") + +# Scope for current onboarding/reference docs (excludes historical preflight +# and context-pack which legitimately discuss patterns). +scope: list[Path] = [root / "START_HERE.md"] +for sub in ("docs/reference-architecture", "docs/handoff"): + scope.extend(sorted((root / sub).glob("*.md"))) + +# 3. No forbidden absolute local paths in current onboarding docs. +abs_re = re.compile(r"/(?:home|Users|workspace)/") +# 4. No dependency on prior conversation context. +convo_re = re.compile( + r"chatgpt|(?:see|refer to)\s+(?:the\s+)?(?:prior|previous)\s+" + r"(?:conversation|chat|session)|ask\s+russell", + re.IGNORECASE, +) +# 5. No unresolved placeholders in current docs. +placeholder_re = re.compile(r"\b(?:TODO|FIXME|XXX|TBD)\b") +for path in scope: + text = path.read_text(encoding="utf-8") + rel = path.relative_to(root) + if abs_re.search(text): + errors.append(f"{rel}: contains a forbidden absolute local path") + if convo_re.search(text): + errors.append(f"{rel}: depends on prior conversation context") + if placeholder_re.search(text): + errors.append(f"{rel}: contains an unresolved placeholder") + +# 6. Capability evidence contains limitations. +cap = (root / "docs/reference-architecture/capability-evidence-map.md").read_text(encoding="utf-8") +if "imitation" not in cap: + errors.append("capability-evidence-map.md must record limitations") + +# 7. Sprint 8 tag is not claimed before it exists. +readme = (root / "README.md").read_text(encoding="utf-8") +if "atlas-sprint-8-complete" in readme: + try: + tags = subprocess.run( + ["git", "tag", "-l", "atlas-sprint-8-complete"], + cwd=str(root), capture_output=True, text=True, check=False, + ).stdout.strip() + except OSError: + tags = "" + if not tags: + errors.append("README references atlas-sprint-8-complete before the tag exists") + +if errors: + print("reference-handoff gate FAILED:") + for e in errors: + print(f" - {e}") + sys.exit(1) +print("reference-handoff OK: package consistent, no hidden-context dependencies") +PY +} + +# --------------------------------------------------------------------------- +# Mode composition +# --------------------------------------------------------------------------- + +if [[ "$MODE" == "static" || "$MODE" == "all" ]]; then + if in_group security-shell; then + run_gate secret_scan gate_secret_scan + run_gate shell_syntax gate_shell_syntax + if command -v shellcheck >/dev/null 2>&1; then + run_gate shell_static gate_shellcheck + elif [[ "$GROUP" == "security-shell" ]]; then + run_gate shell_static false + else + skip_gate shell_static "shellcheck not installed" + fi + run_gate workflow_yaml gate_workflow_yaml + run_gate sql_migrations gate_sql_migrations + fi + if in_group python; then + run_gate python_format gate_python_format + run_gate python_lint gate_python_lint + run_gate python_types gate_python_types + run_gate python_tests gate_python_tests + run_gate config_validation gate_config_validation + run_gate observability_config gate_observability_config + run_gate failure_injection gate_failure_injection + run_gate governance gate_governance + run_gate schema_compatibility gate_schema_compatibility + run_gate lineage_impact gate_lineage_impact + run_gate security_policy gate_security_policy + run_gate performance_cost gate_performance_cost + run_gate reference_handoff gate_reference_handoff + fi + if in_group airflow; then + if airflow_installed; then + run_gate airflow_environment gate_airflow_environment + run_gate dag_import gate_dag_import + elif [[ "$GROUP" == "airflow" ]]; then + echo "apache-airflow is required for --group airflow" >&2 + run_gate airflow_environment false + run_gate dag_import false + else + skip_gate airflow_environment "apache-airflow not installed in this interpreter" + skip_gate dag_import "apache-airflow not installed in this interpreter" + fi + fi + if in_group dbt; then + if command -v dbt >/dev/null 2>&1; then + run_gate dbt_static gate_dbt_static + elif [[ "$GROUP" == "dbt" ]]; then + echo "dbt is required for --group dbt" >&2 + run_gate dbt_static false + else + skip_gate dbt_static "dbt not installed" + fi + fi +fi + +if [[ "$MODE" == "integration" || "$MODE" == "all" ]]; then + if [[ -x "${ATLAS_ROOT}/scripts/validate_gcp_integration.sh" ]]; then + run_gate gcp_integration bash "${ATLAS_ROOT}/scripts/validate_gcp_integration.sh" + else + skip_gate gcp_integration "scripts/validate_gcp_integration.sh not present yet" + fi +fi + +# --------------------------------------------------------------------------- +# Summary +# --------------------------------------------------------------------------- + +python3 - "$RESULTS_FILE" <<'PY' +import json +import sys +from pathlib import Path + +tmp = Path(sys.argv[1] + ".tmp") +gates = [json.loads(line) for line in tmp.read_text().splitlines() if line.strip()] +payload = { + "overall_status": "FAIL" if any(g["status"] == "FAIL" for g in gates) else "PASS", + "gates": gates, +} +Path(sys.argv[1]).write_text(json.dumps(payload, indent=2) + "\n") +tmp.unlink() +PY + +echo "" +echo "===================== GATE SUMMARY =====================" +for i in "${!GATE_NAMES[@]}"; do + printf ' %-24s %s\n' "${GATE_NAMES[$i]}" "${GATE_STATUSES[$i]}" +done +echo "=========================================================" +echo "Machine-readable results: ${RESULTS_FILE}" + +if [[ "$FAILED" -ne 0 ]]; then + echo "validate_ci: FAIL" + exit 1 +fi +echo "validate_ci: PASS" diff --git a/scripts/validate_clean_clone.sh b/scripts/validate_clean_clone.sh new file mode 100644 index 0000000..3881f23 --- /dev/null +++ b/scripts/validate_clean_clone.sh @@ -0,0 +1,109 @@ +#!/usr/bin/env bash +# Clean-clone reproduction test (Sprint 8, Phase 11). +# +# Proves a new engineer can reach a green local validation state from a fresh +# clone using only documented commands — no reuse of the current virtualenv, +# generated data, dbt target, cached credentials, or untracked files. +# +# Usage: +# bash scripts/validate_clean_clone.sh [--ref ] [--keep] +# +# Steps: clone -> checkout -> fresh venv -> documented install -> static CI -> +# generate synthetic data -> unit tests -> governance -> lineage -> reference +# validation. Records duration and per-step outcome; removes the temp dir unless +# --keep. Credentialless: never touches GCP. +set -uo pipefail + +REF="" +KEEP=0 +while [[ $# -gt 0 ]]; do + case "$1" in + --ref) REF="$2"; shift 2 ;; + --keep) KEEP=1; shift ;; + *) echo "unknown arg: $1" >&2; exit 2 ;; + esac +done + +# Repository root (parent of ). +SRC_ROOT="$(cd "$(dirname "$0")/../.." && pwd)" +[[ -z "$REF" ]] && REF="$(git -C "$SRC_ROOT" rev-parse HEAD)" + +TMP_DIR="$(mktemp -d -t atlas-clean-clone-XXXXXX)" +CLONE="$TMP_DIR/Atlas-GCP-Build" +START_TS=$(date +%s) +declare -a RESULTS=() + +record() { RESULTS+=("$1: $2"); echo "----- $1: $2 -----"; } + +cleanup() { + if [[ "$KEEP" -eq 1 ]]; then + echo "temp dir kept: $TMP_DIR" + else + rm -rf "$TMP_DIR" + fi +} +trap cleanup EXIT + +echo "== clean-clone from $SRC_ROOT @ $REF ==" +if ! git clone --quiet "$SRC_ROOT" "$CLONE"; then + echo "clone FAILED" >&2; exit 1 +fi +git -C "$CLONE" checkout --quiet "$REF" || { echo "checkout FAILED" >&2; exit 1; } + +ATLAS="$CLONE/project-atlas" +cd "$ATLAS" || { echo "no project-atlas dir" >&2; exit 1; } + +# Fresh, isolated virtualenv (no reuse of the caller's environment). +python3 -m venv "$TMP_DIR/venv" || { echo "venv FAILED" >&2; exit 1; } +# shellcheck disable=SC1091 +source "$TMP_DIR/venv/bin/activate" +export PATH="$TMP_DIR/venv/bin:$PATH" +unset ATLAS_ROOT PYTHONPATH 2>/dev/null || true + +echo "== documented install ==" +if python -m pip install --quiet --upgrade pip \ + && python -m pip install --quiet -r requirements.txt -r requirements-ci.txt; then + record install PASS +else + record install FAIL +fi + +echo "== static CI ==" +if bash scripts/validate_ci.sh --mode static; then record static_ci PASS; else record static_ci FAIL; fi + +echo "== generate synthetic data ==" +if python scripts/generate_events.py >/dev/null 2>&1; then record generate PASS; else record generate SKIP_OR_FAIL; fi + +echo "== unit tests ==" +if python -m pytest tests/ -q >/dev/null 2>&1; then record unit_tests PASS; else record unit_tests FAIL; fi + +# atlas.* modules live under src/ (no installed package); use the documented +# PYTHONPATH=src convention (same as pytest.ini and validate_ci.sh). +export PYTHONPATH="src" + +echo "== governance ==" +if python -m atlas.governance.catalog check; then record governance PASS; else record governance FAIL; fi + +echo "== lineage ==" +if python -m atlas.governance.lineage >/dev/null 2>&1; then record lineage PASS; else record lineage FAIL; fi + +echo "== reference validation ==" +if python -m atlas.reference.validate; then record reference PASS; else record reference FAIL; fi + +END_TS=$(date +%s) +DURATION=$((END_TS - START_TS)) + +echo "" +echo "===================== CLEAN-CLONE SUMMARY =====================" +printf ' %s\n' "${RESULTS[@]}" +echo " duration_seconds: $DURATION" +echo " ref: $REF" +echo "===============================================================" + +# Fail if any required step failed (generate may SKIP without GCP config). +for r in "${RESULTS[@]}"; do + case "$r" in + *": FAIL") echo "clean_clone: FAIL"; exit 1 ;; + esac +done +echo "clean_clone: PASS" diff --git a/scripts/validate_dbt_sprint2.sh b/scripts/validate_dbt_sprint2.sh new file mode 100755 index 0000000..d1689ff --- /dev/null +++ b/scripts/validate_dbt_sprint2.sh @@ -0,0 +1,169 @@ +#!/usr/bin/env bash +# Validate Atlas Sprint 2 dbt outputs and emit timestamped JSON evidence. +set -Eeuo pipefail +IFS=$'\n\t' +umask 077 + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +DBT_PROJECT_DIR="${ATLAS_ROOT}/dbt/atlas_dbt" +VENV_DIR="${ATLAS_ROOT}/.venv-dbt" +PROFILES_DIR="${DBT_PROFILES_DIR:-$HOME/.dbt}" +LOG_DIR="${ATLAS_ROOT}/logs" +TIMESTAMP="$(date -u +%Y%m%dT%H%M%SZ)" +REPORT_PATH="${LOG_DIR}/validation-sprint2-${TIMESTAMP}.json" +VALIDATED_RUN_ID="${ATLAS_VALIDATED_RUN_ID:-atlas-20260714T163527Z-19a0e4f6}" +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-${GCP_PROJECT_ID:-}}" +LOCATION="${DBT_LOCATION:-US}" + +fail() { + echo "error: $*" >&2 + exit 1 +} + +print_command() { + printf '+' + printf ' %q' "$@" + printf '\n' +} + +run() { + print_command "$@" + "$@" +} + +require_command() { + command -v "$1" >/dev/null 2>&1 || fail "$1 is required but was not found on PATH" +} + +require_command bq +require_command python3 +[[ -x "${VENV_DIR}/bin/dbt" ]] || fail "missing ${VENV_DIR}/bin/dbt; run scripts/setup_dbt.sh first" +[[ -n "$PROJECT_ID" ]] || fail "ATLAS_GCP_PROJECT_ID or GCP_PROJECT_ID must be set" + +run mkdir -p "$LOG_DIR" +DBT_BIN="${VENV_DIR}/bin/dbt" +DBT_FLAGS=(--project-dir "$DBT_PROJECT_DIR" --profiles-dir "$PROFILES_DIR" --target dev) + +run_dbt() { + run "$DBT_BIN" "$@" "${DBT_FLAGS[@]}" +} + +overall_status="PASS" + +record_gate() { + local name="$1" + local status="$2" + if [[ "$status" != "PASS" ]]; then + overall_status="FAIL" + fi + printf '[gate] %-40s %s\n' "$name" "$status" +} + +run_query_scalar() { + local sql="$1" + bq query --use_legacy_sql=false --format=csv --max_rows=1 --quiet "$sql" | tail -n 1 +} + +echo "Running Sprint 2 validation for run_id=${VALIDATED_RUN_ID}" + +if run_dbt test --select test_type:singular; then + record_gate "singular_tests" "PASS" +else + record_gate "singular_tests" "FAIL" +fi + +if run_dbt test --exclude test_type:singular; then + record_gate "generic_and_unit_tests" "PASS" +else + record_gate "generic_and_unit_tests" "FAIL" +fi + +raw_rows="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_raw.events\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +accepted_rows="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_intermediate.int_accepted_events\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +rejected_rows="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_quarantine.int_rejected_events\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +fact_rows="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_core.fct_events\`")" +mart_rows="$(run_query_scalar "SELECT COALESCE(SUM(event_count), 0) FROM \`${PROJECT_ID}.atlas_marts.mart_daily_event_metrics\`")" + +if [[ "$raw_rows" == "$((accepted_rows + rejected_rows))" ]]; then + record_gate "raw_accepted_rejected_reconciliation" "PASS" +else + record_gate "raw_accepted_rejected_reconciliation" "FAIL" +fi + +if [[ "$fact_rows" == "$mart_rows" ]]; then + record_gate "fact_mart_reconciliation" "PASS" +else + record_gate "fact_mart_reconciliation" "FAIL" +fi + +duplicate_extra="$(run_query_scalar "SELECT COUNTIF(is_duplicate_extra) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +null_users="$(run_query_scalar "SELECT COUNTIF(user_id IS NULL) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +invalid_countries="$(run_query_scalar "SELECT COUNTIF(NOT is_valid_country) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +future_dated="$(run_query_scalar "SELECT COUNTIF(is_future_dated) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +event_time_late="$(run_query_scalar "SELECT COUNTIF(is_event_time_late_arriving) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +backdated="$(run_query_scalar "SELECT COUNTIF(is_backdated_event_date) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" +mismatch="$(run_query_scalar "SELECT COUNTIF(has_event_date_timestamp_mismatch) FROM \`${PROJECT_ID}.atlas_intermediate.int_event_classification\` WHERE pipeline_run_id = '${VALIDATED_RUN_ID}'")" + +check_anomaly() { + local name="$1" + local expected="$2" + local actual="$3" + if [[ "$expected" == "$actual" ]]; then + record_gate "anomaly_${name}" "PASS" + else + record_gate "anomaly_${name}" "FAIL" + fi +} + +check_anomaly "duplicate_extra" 50 "$duplicate_extra" +check_anomaly "null_users" 500 "$null_users" +check_anomaly "invalid_countries" 200 "$invalid_countries" +check_anomaly "future_dated" 150 "$future_dated" +check_anomaly "event_time_late" 0 "$event_time_late" +check_anomaly "backdated_event_date" 300 "$backdated" +check_anomaly "date_timestamp_mismatch" 300 "$mismatch" + +export VALIDATED_RUN_ID PROJECT_ID LOCATION OVERALL_STATUS="$overall_status" REPORT_PATH +export RAW_ROWS="$raw_rows" ACCEPTED_ROWS="$accepted_rows" REJECTED_ROWS="$rejected_rows" +export FACT_ROWS="$fact_rows" MART_ROWS="$mart_rows" +export DUPLICATE_EXTRA="$duplicate_extra" NULL_USERS="$null_users" +export INVALID_COUNTRIES="$invalid_countries" FUTURE_DATED="$future_dated" +export EVENT_TIME_LATE="$event_time_late" BACKDATED="$backdated" MISMATCH="$mismatch" + +python3 - <<'PY' +import json +import os +from datetime import datetime, timezone + +report = { + "timestamp": datetime.now(timezone.utc).isoformat(), + "validated_run_id": os.environ["VALIDATED_RUN_ID"], + "project_id": os.environ["PROJECT_ID"], + "location": os.environ["LOCATION"], + "overall_status": os.environ["OVERALL_STATUS"], + "counts": { + "raw_rows": int(os.environ["RAW_ROWS"]), + "accepted_rows": int(os.environ["ACCEPTED_ROWS"]), + "rejected_rows": int(os.environ["REJECTED_ROWS"]), + "fact_rows": int(os.environ["FACT_ROWS"]), + "mart_event_total": int(os.environ["MART_ROWS"]), + }, + "anomalies": { + "duplicate_extra": int(os.environ["DUPLICATE_EXTRA"]), + "null_users": int(os.environ["NULL_USERS"]), + "invalid_countries": int(os.environ["INVALID_COUNTRIES"]), + "future_dated": int(os.environ["FUTURE_DATED"]), + "event_time_late_arriving": int(os.environ["EVENT_TIME_LATE"]), + "backdated_event_date": int(os.environ["BACKDATED"]), + "date_timestamp_mismatch": int(os.environ["MISMATCH"]), + }, +} +with open(os.environ["REPORT_PATH"], "w", encoding="utf-8") as handle: + json.dump(report, handle, indent=2) +print(f"Wrote validation report to {os.environ['REPORT_PATH']}") +PY + +echo "Overall validation status: ${overall_status}" +if [[ "$overall_status" != "PASS" ]]; then + exit 1 +fi diff --git a/scripts/validate_dbt_sprint2_incremental.sh b/scripts/validate_dbt_sprint2_incremental.sh new file mode 100755 index 0000000..92f15d6 --- /dev/null +++ b/scripts/validate_dbt_sprint2_incremental.sh @@ -0,0 +1,133 @@ +#!/usr/bin/env bash +# Confirm Sprint 2 incremental idempotency on an unchanged raw source. +set -Eeuo pipefail +IFS=$'\n\t' +umask 077 + +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +DBT_PROJECT_DIR="${ATLAS_ROOT}/dbt/atlas_dbt" +VENV_DIR="${ATLAS_ROOT}/.venv-dbt" +PROFILES_DIR="${DBT_PROFILES_DIR:-$HOME/.dbt}" +LOG_DIR="${ATLAS_ROOT}/logs" +TIMESTAMP="$(date -u +%Y%m%dT%H%M%SZ)" +REPORT_PATH="${LOG_DIR}/validation-sprint2-incremental-${TIMESTAMP}.json" +PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-${GCP_PROJECT_ID:-}}" + +fail() { + echo "error: $*" >&2 + exit 1 +} + +print_command() { + printf '+' + printf ' %q' "$@" + printf '\n' +} + +run() { + print_command "$@" + "$@" +} + +require_command() { + command -v "$1" >/dev/null 2>&1 || fail "$1 is required but was not found on PATH" +} + +require_command bq +require_command python3 +[[ -x "${VENV_DIR}/bin/dbt" ]] || fail "missing ${VENV_DIR}/bin/dbt; run scripts/setup_dbt.sh first" +[[ -n "$PROJECT_ID" ]] || fail "ATLAS_GCP_PROJECT_ID or GCP_PROJECT_ID must be set" + +run mkdir -p "$LOG_DIR" +DBT_BIN="${VENV_DIR}/bin/dbt" +DBT_FLAGS=(--project-dir "$DBT_PROJECT_DIR" --profiles-dir "$PROFILES_DIR" --target dev) + +run_dbt() { + run "$DBT_BIN" "$@" "${DBT_FLAGS[@]}" +} + +run_query_scalar() { + local sql="$1" + bq query --use_legacy_sql=false --format=csv --max_rows=1 --quiet "$sql" | tail -n 1 +} + +echo "Sprint 2 incremental idempotency check (unchanged raw source expected)" + +before_fct="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_core.fct_events\`")" +before_mart="$(run_query_scalar "SELECT COALESCE(SUM(event_count), 0) FROM \`${PROJECT_ID}.atlas_marts.mart_daily_event_metrics\`")" +before_rejected="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_quarantine.int_rejected_events\`")" + +echo "Before: fct_events=${before_fct} mart_total=${before_mart} rejected=${before_rejected}" + +run_dbt build + +after_fct="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_core.fct_events\`")" +after_mart="$(run_query_scalar "SELECT COALESCE(SUM(event_count), 0) FROM \`${PROJECT_ID}.atlas_marts.mart_daily_event_metrics\`")" +after_rejected="$(run_query_scalar "SELECT COUNT(*) FROM \`${PROJECT_ID}.atlas_quarantine.int_rejected_events\`")" + +echo "After: fct_events=${after_fct} mart_total=${after_mart} rejected=${after_rejected}" + +overall_status="PASS" +if [[ "$before_fct" != "$after_fct" ]]; then + echo "[gate] fct_events_row_count_unchanged FAIL (${before_fct} -> ${after_fct})" + overall_status="FAIL" +else + echo "[gate] fct_events_row_count_unchanged PASS" +fi + +if [[ "$before_mart" != "$after_mart" ]]; then + echo "[gate] mart_event_total_unchanged FAIL (${before_mart} -> ${after_mart})" + overall_status="FAIL" +else + echo "[gate] mart_event_total_unchanged PASS" +fi + +if [[ "$before_rejected" != "$after_rejected" ]]; then + echo "[gate] rejected_row_count_unchanged FAIL (${before_rejected} -> ${after_rejected})" + overall_status="FAIL" +else + echo "[gate] rejected_row_count_unchanged PASS" +fi + +if run_dbt test --select test_type:singular; then + echo "[gate] singular_tests PASS" +else + echo "[gate] singular_tests FAIL" + overall_status="FAIL" +fi + +export REPORT_PATH PROJECT_ID OVERALL_STATUS="$overall_status" +export BEFORE_FCT="$before_fct" AFTER_FCT="$after_fct" +export BEFORE_MART="$before_mart" AFTER_MART="$after_mart" +export BEFORE_REJECTED="$before_rejected" AFTER_REJECTED="$after_rejected" + +python3 - <<'PY' +import json +import os +from datetime import datetime, timezone + +report = { + "timestamp": datetime.now(timezone.utc).isoformat(), + "check": "incremental_idempotency", + "project_id": os.environ["PROJECT_ID"], + "overall_status": os.environ["OVERALL_STATUS"], + "counts_before": { + "fct_events": int(os.environ["BEFORE_FCT"]), + "mart_event_total": int(os.environ["BEFORE_MART"]), + "rejected_rows": int(os.environ["BEFORE_REJECTED"]), + }, + "counts_after": { + "fct_events": int(os.environ["AFTER_FCT"]), + "mart_event_total": int(os.environ["AFTER_MART"]), + "rejected_rows": int(os.environ["AFTER_REJECTED"]), + }, +} +with open(os.environ["REPORT_PATH"], "w", encoding="utf-8") as handle: + json.dump(report, handle, indent=2) +print(f"Wrote validation report to {os.environ['REPORT_PATH']}") +PY + +echo "Overall incremental validation status: ${overall_status}" +if [[ "$overall_status" != "PASS" ]]; then + exit 1 +fi diff --git a/scripts/validate_events.py b/scripts/validate_events.py new file mode 100755 index 0000000..6df392a --- /dev/null +++ b/scripts/validate_events.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +"""Validate a loaded Atlas pipeline run.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(PROJECT_ROOT / "src")) + +from atlas.config.settings import load_settings +from atlas.validation.checks import validate_anomaly_detection, validate_loaded_run + + +def main() -> int: + parser = argparse.ArgumentParser(description="Validate Atlas loaded events") + parser.add_argument("--run-id") + parser.add_argument("--event-date", required=True) + parser.add_argument("--batch-id") + parser.add_argument("--processing-date") + parser.add_argument( + "--mode", + choices=("sprint1", "airflow"), + default="sprint1", + help="airflow mode scopes by batch_id and uses structural PASS criteria", + ) + args = parser.parse_args() + + if not args.run_id and not args.batch_id: + parser.error("Provide --run-id or --batch-id") + + settings = load_settings() + report = validate_loaded_run( + settings, + args.run_id or args.batch_id or "", + args.event_date, + batch_id=args.batch_id, + processing_date=args.processing_date or args.event_date, + mode=args.mode, + ) + if args.mode == "sprint1": + report = validate_anomaly_detection(report, settings) + print(json.dumps(report.to_dict(), indent=2)) + return 0 if report.overall_status == "PASS" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/validate_gcp_integration.sh b/scripts/validate_gcp_integration.sh new file mode 100755 index 0000000..76a0252 --- /dev/null +++ b/scripts/validate_gcp_integration.sh @@ -0,0 +1,247 @@ +#!/usr/bin/env bash +# Isolated GCP integration test for Project Atlas (Sprint 4, Phase 7). +# +# Runs the full pipeline against ephemeral, run-scoped resources: +# raw dataset: atlas_ci__raw +# dbt datasets: atlas_ci__{staging,intermediate,core,marts,quarantine} +# GCS prefix: gs:///atlas-ci// +# +# Never touches canonical Atlas datasets or the canonical events bucket. +# Cleanup always runs (EXIT trap); failures are recorded and fail the script. +# Orphan recovery (TTL): the CI bucket auto-deletes objects after 7 days; +# stale datasets can be listed with +# bq ls --project_id | grep atlas_ci_ +# and removed with `bq rm -r -f -d :`. +set -uo pipefail + +cd "$(dirname "${BASH_SOURCE[0]}")/.." || exit 1 +ATLAS_DIR="$(pwd)" + +RUN_TOKEN="${GITHUB_RUN_ID:-local$(date -u +%s)}" +RUN_TOKEN="${RUN_TOKEN//[^a-zA-Z0-9]/}" +export ATLAS_GCP_PROJECT_ID="${ATLAS_GCP_PROJECT_ID:-example-gcp-project}" +export ATLAS_GCS_BUCKET="${ATLAS_CI_BUCKET:-atlas-ci-${ATLAS_GCP_PROJECT_ID}}" +export ATLAS_GCS_PREFIX="atlas-ci/${RUN_TOKEN}/raw" +export ATLAS_BQ_DATASET="atlas_ci_${RUN_TOKEN}_raw" +export ATLAS_DBT_DATASET="atlas_ci_${RUN_TOKEN}" +export DBT_LOCATION="${DBT_LOCATION:-US}" + +BATCH_ID="atlas-ci-${RUN_TOKEN}" +PIPELINE_RUN_ID="atlas-ci-run-${RUN_TOKEN}" +PROCESSING_DATE="${ATLAS_CI_PROCESSING_DATE:-$(date -u +%F)}" +EXPECTED_ROWS=50000 +PY="${ATLAS_PYTHON:-python3}" +export PYTHONPATH="${ATLAS_DIR}/src:${PYTHONPATH:-}" + +RESULTS_DIR="${ATLAS_DIR}/logs/ci" +RESULTS_FILE="${RESULTS_DIR}/integration-results.json" +mkdir -p "$RESULTS_DIR" +declare -a GATE_NAMES=() GATE_STATUSES=() +OVERALL=0 + +record() { # name status + GATE_NAMES+=("$1") + GATE_STATUSES+=("$2") + if [[ "$2" == "FAIL" ]]; then OVERALL=1; fi + printf '[%s] %s\n' "$2" "$1" +} + +run_gate() { # name command... + local name="$1" + shift + echo "" + echo "=== gate: ${name} ===" + if "$@"; then record "$name" "PASS"; else record "$name" "FAIL"; fi +} + +write_results() { + { + echo '{' + echo " \"run_token\": \"${RUN_TOKEN}\"," + echo " \"batch_id\": \"${BATCH_ID}\"," + echo " \"raw_dataset\": \"${ATLAS_BQ_DATASET}\"," + echo " \"dbt_dataset_prefix\": \"${ATLAS_DBT_DATASET}\"," + echo " \"gcs_prefix\": \"gs://${ATLAS_GCS_BUCKET}/atlas-ci/${RUN_TOKEN}/\"," + echo ' "gates": [' + local i + for i in "${!GATE_NAMES[@]}"; do + local sep=',' + [[ "$i" -eq $((${#GATE_NAMES[@]} - 1)) ]] && sep='' + echo " {\"name\": \"${GATE_NAMES[$i]}\", \"status\": \"${GATE_STATUSES[$i]}\"}${sep}" + done + echo ' ],' + if [[ "$OVERALL" -eq 0 ]]; then + echo ' "overall": "PASS"' + else + echo ' "overall": "FAIL"' + fi + echo '}' + } >"$RESULTS_FILE" + echo "" + echo "Results written to ${RESULTS_FILE}" +} + +# --- Cleanup (always runs) ----------------------------------------------------- +CLEANED=0 +cleanup() { + if [[ "$CLEANED" -eq 1 ]]; then return; fi + CLEANED=1 + echo "" + echo "=== cleanup: removing ephemeral resources for run ${RUN_TOKEN} ===" + local cleanup_failed=0 + local ds + for ds in \ + "${ATLAS_BQ_DATASET}" \ + "${ATLAS_DBT_DATASET}" \ + "${ATLAS_DBT_DATASET}_staging" \ + "${ATLAS_DBT_DATASET}_intermediate" \ + "${ATLAS_DBT_DATASET}_core" \ + "${ATLAS_DBT_DATASET}_marts" \ + "${ATLAS_DBT_DATASET}_quarantine"; do + if bq --project_id="$ATLAS_GCP_PROJECT_ID" show --dataset "$ds" >/dev/null 2>&1; then + if bq --project_id="$ATLAS_GCP_PROJECT_ID" rm -r -f -d "$ds" >/dev/null 2>&1; then + echo " deleted dataset ${ds}" + else + echo " FAILED to delete dataset ${ds}" + cleanup_failed=1 + fi + fi + done + if gcloud storage ls "gs://${ATLAS_GCS_BUCKET}/atlas-ci/${RUN_TOKEN}/" >/dev/null 2>&1; then + if gcloud storage rm -r "gs://${ATLAS_GCS_BUCKET}/atlas-ci/${RUN_TOKEN}/**" >/dev/null 2>&1; then + echo " deleted gs://${ATLAS_GCS_BUCKET}/atlas-ci/${RUN_TOKEN}/" + else + echo " FAILED to delete gs://${ATLAS_GCS_BUCKET}/atlas-ci/${RUN_TOKEN}/ (7-day TTL will reap it)" + cleanup_failed=1 + fi + fi + # Verify nothing scoped to this run remains. + local leftovers + leftovers="$(bq --project_id="$ATLAS_GCP_PROJECT_ID" ls --max_results=1000 2>/dev/null | grep -c "atlas_ci_${RUN_TOKEN}" || true)" + if [[ "$leftovers" != "0" ]]; then + echo " cleanup verification FAILED: ${leftovers} dataset(s) remain for run ${RUN_TOKEN}" + cleanup_failed=1 + else + echo " cleanup verified: no atlas_ci_${RUN_TOKEN}* datasets remain" + fi + if [[ "$cleanup_failed" -eq 1 ]]; then + record "cleanup" "FAIL" + else + record "cleanup" "PASS" + fi + write_results +} +trap cleanup EXIT + +echo "Isolated integration run" +echo " project: ${ATLAS_GCP_PROJECT_ID}" +echo " run token: ${RUN_TOKEN}" +echo " raw dataset: ${ATLAS_BQ_DATASET}" +echo " dbt datasets: ${ATLAS_DBT_DATASET}_{staging,intermediate,core,marts,quarantine}" +echo " gcs prefix: gs://${ATLAS_GCS_BUCKET}/${ATLAS_GCS_PREFIX}" +echo " batch id: ${BATCH_ID}" + +# --- 1. Authentication and API reachability ------------------------------------- +gate_auth() { + local identity + identity="$(gcloud auth list --filter=status:ACTIVE --format='value(account)' 2>/dev/null)" || return 1 + [[ -n "$identity" ]] || { echo "no active gcloud identity"; return 1; } + echo "active identity: ${identity}" + bq --project_id="$ATLAS_GCP_PROJECT_ID" query --use_legacy_sql=false --format=none 'SELECT 1' || return 1 + gcloud storage ls "gs://${ATLAS_GCS_BUCKET}/" >/dev/null || return 1 + echo "BigQuery and GCS reachable" +} +run_gate "auth_and_apis" gate_auth + +# --- 2. Migration validation (plan mode, no mutation) ---------------------------- +gate_migration_plan() { + bash "${ATLAS_DIR}/scripts/apply_atlas_migrations.sh" --mode plan +} +if [[ -f "${ATLAS_DIR}/scripts/apply_atlas_migrations.sh" ]]; then + run_gate "migration_plan" gate_migration_plan +else + record "migration_plan" "SKIP" +fi + +# --- 3. Deterministic generation -------------------------------------------------- +GEN_DIR="$(mktemp -d)" +gate_generation() { + local out1 out2 sum1 sum2 + out1="$("$PY" "${ATLAS_DIR}/scripts/generate_events.py" \ + --processing-date "$PROCESSING_DATE" --batch-id "$BATCH_ID" \ + --pipeline-run-id "$PIPELINE_RUN_ID" --seed 42 \ + --output-path "${GEN_DIR}/events-a.jsonl")" || return 1 + out2="$("$PY" "${ATLAS_DIR}/scripts/generate_events.py" \ + --processing-date "$PROCESSING_DATE" --batch-id "$BATCH_ID" \ + --pipeline-run-id "$PIPELINE_RUN_ID" --seed 42 \ + --output-path "${GEN_DIR}/events-b.jsonl")" || return 1 + sum1="$(echo "$out1" | "$PY" -c 'import json,sys; print(json.load(sys.stdin)["checksum_sha256"])')" + sum2="$(echo "$out2" | "$PY" -c 'import json,sys; print(json.load(sys.stdin)["checksum_sha256"])')" + echo "checksum A: ${sum1}" + echo "checksum B: ${sum2}" + [[ -n "$sum1" && "$sum1" == "$sum2" ]] || { echo "generation is not deterministic"; return 1; } +} +run_gate "deterministic_generation" gate_generation + +# --- 4. Upload to isolated GCS prefix --------------------------------------------- +gate_upload() { + "$PY" "${ATLAS_DIR}/scripts/upload_events.py" \ + --local-path "${GEN_DIR}/events-a.jsonl" \ + --event-date "$PROCESSING_DATE" \ + --run-id "$PIPELINE_RUN_ID" \ + --batch-id "$BATCH_ID" +} +run_gate "gcs_upload" gate_upload + +GCS_URI="gs://${ATLAS_GCS_BUCKET}/${ATLAS_GCS_PREFIX}/event_date=${PROCESSING_DATE}/batch_id=${BATCH_ID}/events.jsonl" + +# --- 5. Raw load into isolated dataset --------------------------------------------- +gate_raw_load() { + "$PY" "${ATLAS_DIR}/scripts/load_events.py" \ + --gcs-uri "$GCS_URI" --run-id "$PIPELINE_RUN_ID" \ + --batch-id "$BATCH_ID" --processing-date "$PROCESSING_DATE" \ + --expected-row-count "$EXPECTED_ROWS" +} +run_gate "raw_load" gate_raw_load + +# --- 6. Idempotent rerun: second load must skip, count must not change --------------- +gate_idempotent_rerun() { + local rerun already rows + rerun="$("$PY" "${ATLAS_DIR}/scripts/load_events.py" \ + --gcs-uri "$GCS_URI" --run-id "${PIPELINE_RUN_ID}-rerun" \ + --batch-id "$BATCH_ID" --processing-date "$PROCESSING_DATE" \ + --expected-row-count "$EXPECTED_ROWS")" || return 1 + already="$(echo "$rerun" | "$PY" -c 'import json,sys; print(json.load(sys.stdin)["already_loaded"])')" + [[ "$already" == "True" ]] || { echo "rerun did not skip (already_loaded=${already})"; return 1; } + rows="$(bq --project_id="$ATLAS_GCP_PROJECT_ID" query --use_legacy_sql=false --format=csv \ + "SELECT COUNT(1) FROM \`${ATLAS_GCP_PROJECT_ID}.${ATLAS_BQ_DATASET}.events\` WHERE batch_id = '${BATCH_ID}'" \ + | tail -1)" + echo "raw rows after rerun: ${rows}" + [[ "$rows" == "$EXPECTED_ROWS" ]] || { echo "duplicate rows detected"; return 1; } +} +run_gate "idempotent_rerun" gate_idempotent_rerun + +# --- 7. dbt build against isolated schemas ------------------------------------------- +DBT_DIR="${ATLAS_DIR}/dbt/atlas_dbt" +DBT_PROFILES_TMP="$(mktemp -d)" +cp "${DBT_DIR}/profiles.yml.example" "${DBT_PROFILES_TMP}/profiles.yml" +gate_dbt_build() { + (cd "$DBT_DIR" \ + && dbt deps --profiles-dir "$DBT_PROFILES_TMP" --quiet \ + && dbt build --profiles-dir "$DBT_PROFILES_TMP" \ + --vars "{\"validated_batch_id\": \"${BATCH_ID}\"}") +} +run_gate "dbt_build_isolated" gate_dbt_build + +# --- 8. Batch-scoped warehouse reconciliation ----------------------------------------- +gate_reconciliation() { + "$PY" "${ATLAS_DIR}/scripts/atlas_step_runner.py" validate_warehouse \ + "{\"batch_id\": \"${BATCH_ID}\", \"processing_date\": \"${PROCESSING_DATE}\"}" +} +run_gate "warehouse_reconciliation" gate_reconciliation + +echo "" +echo "=== integration gates complete (cleanup follows) ===" +cleanup +trap - EXIT +exit "$OVERALL" diff --git a/scripts/validate_public_extraction.py b/scripts/validate_public_extraction.py new file mode 100644 index 0000000..d3819ab --- /dev/null +++ b/scripts/validate_public_extraction.py @@ -0,0 +1,255 @@ +#!/usr/bin/env python3 +"""Public-repository extraction review validator (Sprint 8, Phase 14). + +Dry-run by default. Scans the in-scope repository for private/sensitive patterns +and verifies that every file containing a *forbidden-public* pattern is given an +explicit disposition in ``config/public_extraction_manifest.yml``. + +Guarantees: +- never prints a matched secret value (only file paths + pattern ids), +- never pushes a repository, changes visibility, publishes artifacts, or removes + private evidence from the primary repository, +- exits nonzero on any unresolved sensitive file or manifest error. + +Under ``ATLAS_APPROVE_PUBLIC_EXTRACTION=true`` with ``--emit-candidate `` it +may copy PUBLIC_READY files into a *local* candidate directory for inspection. + +Usage: + python scripts/validate_public_extraction.py # dry-run report + check + python scripts/validate_public_extraction.py --emit-candidate /tmp/atlas-public +""" + +from __future__ import annotations + +import argparse +import os +import re +import subprocess +import sys +from pathlib import Path + +import yaml + +ROOT = Path(__file__).resolve().parent.parent # project-atlas/ +MANIFEST_REL = "config/public_extraction_manifest.yml" + +VALID_DISPOSITIONS = { + "PUBLIC_READY", + "REDACT", + "REPLACE_WITH_SAMPLE", + "EXCLUDE", + "PRIVATE_ONLY", + "REVIEW_REQUIRED", +} + +# Out-of-scope trees (separate lifecycle) and non-source dirs. +EXCLUDED_DIR_PARTS = { + ".git", + "artifact-platform", + "node_modules", + "__pycache__", + "target", + ".venv", + "dist", + "data", + "logs", + ".gcp", +} + +# Patterns that must NEVER appear in a PUBLIC_READY file. Built so this file +# itself contains no literal secret (regex fragments only). +FORBIDDEN_PUBLIC = { + "personal_email": re.compile(r"[A-Za-z0-9._%+-]+@(?:gmail|yahoo|hotmail|outlook)\.com", re.IGNORECASE), + "private_key_header": re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----"), + "gcp_api_key": re.compile(r"AIza[0-9A-Za-z_-]{35}"), + "service_account_json": re.compile(r'"type"\s*:\s*"service_account"'), + "slack_webhook": re.compile(r"https://hooks\.slack\.com/services/\S+"), + "bearer_literal": re.compile(r"Authorization:\s*Bearer\s+[A-Za-z0-9._-]{12,}"), +} + +TEXT_SUFFIXES = { + ".md", + ".py", + ".sh", + ".yaml", + ".yml", + ".json", + ".sql", + ".txt", + ".cfg", + ".ini", + ".toml", +} + + +def _in_scope(path: Path) -> bool: + parts = set(path.relative_to(ROOT).parts) + return not (parts & EXCLUDED_DIR_PARTS) + + +def _iter_text_files() -> list[Path]: + """Enumerate git-tracked, in-scope text files (never gitignored venvs/data).""" + try: + out = subprocess.run( + ["git", "ls-files", "-z"], + cwd=str(ROOT), + capture_output=True, + check=True, + text=True, + ).stdout + except (OSError, subprocess.CalledProcessError) as exc: + raise SystemExit(f"cannot list tracked files: {exc}") from exc + + files: list[Path] = [] + for rel in out.split("\0"): + if not rel: + continue + path = ROOT / rel + if path.suffix.lower() not in TEXT_SUFFIXES: + continue + if not path.is_file(): + continue + if not _in_scope(path): + continue + files.append(path) + return files + + +def load_manifest() -> dict: + path = ROOT / MANIFEST_REL + if not path.exists(): + raise SystemExit(f"missing manifest: {MANIFEST_REL}") + data = yaml.safe_load(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise SystemExit("manifest is not a mapping") + return data + + +def manifest_dispositions(manifest: dict) -> dict[str, str]: + result: dict[str, str] = {} + for entry in manifest.get("files", []) or []: + if not isinstance(entry, dict): + continue + rel = str(entry.get("path", "")).strip() + disp = str(entry.get("disposition", "")).strip() + if rel: + result[rel] = disp + return result + + +def detector_allowlist(manifest: dict) -> set[str]: + """Files that define/test the scanners: their pattern strings are detector + definitions or fixtures, not real secrets (mirrors gate_secret_scan).""" + allow = {"scripts/validate_public_extraction.py", MANIFEST_REL} + for rel in manifest.get("detector_allowlist", []) or []: + allow.add(str(rel).strip()) + return allow + + +def scan(manifest: dict | None = None) -> dict[str, list[str]]: + """Return {relpath: [pattern_ids]} for files with forbidden-public matches.""" + allow = detector_allowlist(manifest or {}) + findings: dict[str, list[str]] = {} + for path in _iter_text_files(): + rel = str(path.relative_to(ROOT)) + if rel in allow: + continue + try: + text = path.read_text(encoding="utf-8", errors="ignore") + except OSError: + continue + hits = [pid for pid, rx in FORBIDDEN_PUBLIC.items() if rx.search(text)] + if hits: + findings[rel] = sorted(hits) + return findings + + +def validate(manifest: dict) -> tuple[list[str], dict[str, list[str]]]: + errors: list[str] = [] + + for entry in manifest.get("files", []) or []: + disp = str(entry.get("disposition", "")).strip() + rel = str(entry.get("path", "")).strip() + if disp not in VALID_DISPOSITIONS: + errors.append(f"{rel or ''}: invalid disposition '{disp}'") + for field in ("reason", "replacement_strategy"): + if disp != "PUBLIC_READY" and not str(entry.get(field, "")).strip(): + errors.append(f"{rel}: missing '{field}' for non-public disposition") + + dispositions = manifest_dispositions(manifest) + findings = scan(manifest) + + for rel, pattern_ids in findings.items(): + disp = dispositions.get(rel) + if disp is None: + errors.append( + f"{rel}: contains forbidden-public pattern " + f"({', '.join(pattern_ids)}) but has no manifest disposition" + ) + elif disp == "PUBLIC_READY": + errors.append( + f"{rel}: marked PUBLIC_READY but contains forbidden-public pattern ({', '.join(pattern_ids)})" + ) + + return errors, findings + + +def emit_candidate(manifest: dict, dest: Path) -> None: + import shutil + + dispositions = manifest_dispositions(manifest) + dest.mkdir(parents=True, exist_ok=True) + copied = 0 + for path in _iter_text_files(): + rel = str(path.relative_to(ROOT)) + disp = dispositions.get(rel, "PUBLIC_READY") + if disp in {"EXCLUDE", "PRIVATE_ONLY", "REVIEW_REQUIRED"}: + continue + target = dest / rel + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(path, target) + copied += 1 + print(f"candidate emitted: {copied} files -> {dest} (no publication performed)") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Atlas public-extraction review") + parser.add_argument( + "--emit-candidate", + metavar="DIR", + help="copy public-safe files into a local candidate dir " + "(requires ATLAS_APPROVE_PUBLIC_EXTRACTION=true)", + ) + args = parser.parse_args(argv) + + manifest = load_manifest() + errors, findings = validate(manifest) + + print("Public-extraction dry-run report:") + print(f" files with forbidden-public patterns: {len(findings)}") + for rel, pattern_ids in sorted(findings.items()): + disp = manifest_dispositions(manifest).get(rel, "") + print(f" - {rel}: {', '.join(pattern_ids)} -> disposition {disp}") + + if errors: + print("public-extraction validation FAILED:", file=sys.stderr) + for err in errors: + print(f" - {err}", file=sys.stderr) + return 1 + + print("public-extraction validation OK: all sensitive files have dispositions") + + if args.emit_candidate: + if os.environ.get("ATLAS_APPROVE_PUBLIC_EXTRACTION") != "true": + print( + "refusing to emit candidate: set ATLAS_APPROVE_PUBLIC_EXTRACTION=true", + file=sys.stderr, + ) + return 2 + emit_candidate(manifest, Path(args.emit_candidate)) + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/verify_mcp_access.sh b/scripts/verify_mcp_access.sh new file mode 100755 index 0000000..eb16eeb --- /dev/null +++ b/scripts/verify_mcp_access.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Verify BigQuery and dbt MCP prerequisites for desktop and cloud agents. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +ATLAS_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +cd "$ROOT" + +PROJECT_ID="${GCP_PROJECT_ID:-example-gcp-project}" +DBT_BIN="${DBT_PATH:-${ROOT}/.venv/bin/dbt}" + +if [[ -n "${ATLAS_GCP_SERVICE_ACCOUNT_KEY:-}" ]]; then + # Materialize credentials OUTSIDE the git worktree (default ~/.gcp), matching + # setup_cloud_agent.sh. umask 077 in a subshell closes the window where the + # file would otherwise be world-readable before chmod. + KEY_DIR="${ATLAS_CLOUD_KEY_DIR:-${HOME}/.gcp}" + mkdir -p "${KEY_DIR}" + KEY_PATH="${KEY_DIR}/atlas-service-account.json" + ( umask 077; echo "${ATLAS_GCP_SERVICE_ACCOUNT_KEY}" | base64 -d > "${KEY_PATH}" ) + chmod 600 "${KEY_PATH}" + export GOOGLE_APPLICATION_CREDENTIALS="${KEY_PATH}" + echo "==> Materialized cloud service account credentials at ${KEY_PATH}" +fi + +if [[ ! -x "${DBT_BIN}" ]]; then + echo "error: dbt executable not found at ${DBT_BIN}" + exit 1 +fi + +echo "==> Verifying dbt BigQuery profile" +GCP_PROJECT_ID="${PROJECT_ID}" DBT_TARGET=bigquery "${DBT_BIN}" debug \ + --project-dir "${ROOT}/transform/dbt" \ + --profiles-dir "${ROOT}/transform/dbt" + +PYTHON_BIN="${ROOT}/.venv/bin/python" +if [[ ! -x "${PYTHON_BIN}" ]]; then + PYTHON_BIN="$(command -v python3)" +fi + +echo "==> Verifying BigQuery query access" +"${PYTHON_BIN}" - <<'PY' +from google.cloud import bigquery +import os + +project_id = os.environ.get("GCP_PROJECT_ID", "example-gcp-project") +client = bigquery.Client(project=project_id) +rows = list(client.query("SELECT 1 AS ok").result()) +assert rows and rows[0]["ok"] == 1 +print(f"BigQuery access verified for project {project_id}") +PY + +echo "==> Verifying Atlas package imports" +( + cd "${ATLAS_ROOT}" + PYTHONPATH="${ATLAS_ROOT}/src" "${PYTHON_BIN}" - <<'PY' +from atlas.config.settings import load_settings +settings = load_settings() +print(f"Atlas settings loaded for project {settings.gcp.project_id}") +PY +) + +echo "MCP verification complete." diff --git a/sql/create_deployments_table.sql b/sql/create_deployments_table.sql new file mode 100644 index 0000000..b8151e1 --- /dev/null +++ b/sql/create_deployments_table.sql @@ -0,0 +1,27 @@ +-- Deployment audit: one row per deployment or rollback attempt (Sprint 4, Phase 10). +CREATE TABLE IF NOT EXISTS `{project_id}.atlas_ops.deployments` ( + deployment_id STRING NOT NULL, + git_sha STRING NOT NULL, + git_ref STRING, + release_tag STRING, + environment STRING NOT NULL, + workflow_run_id STRING, + actor STRING, + deployment_type STRING NOT NULL, + started_at TIMESTAMP NOT NULL, + completed_at TIMESTAMP, + status STRING NOT NULL, + artifact_uri STRING, + artifact_checksum STRING, + composer_environment STRING, + composer_region STRING, + smoke_pipeline_run_id STRING, + previous_git_sha STRING, + migration_count INT64, + failure_stage STRING, + error_type STRING, + error_summary STRING, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +) +CLUSTER BY status, environment, git_sha; diff --git a/sql/create_events_table.sql b/sql/create_events_table.sql new file mode 100644 index 0000000..0c1d37a --- /dev/null +++ b/sql/create_events_table.sql @@ -0,0 +1,21 @@ +-- Project Atlas raw events table DDL. +-- Partitioned by event_date and clustered by event_name, country_code. +-- Metadata columns support lineage and idempotent run tracking. + +CREATE TABLE IF NOT EXISTS `{project_id}.{dataset_id}.{table_id}` ( + event_id STRING NOT NULL, + user_id STRING, + event_name STRING NOT NULL, + event_timestamp TIMESTAMP NOT NULL, + event_date DATE NOT NULL, + country_code STRING, + platform STRING, + app_version STRING, + ingested_at TIMESTAMP NOT NULL, + source_file STRING NOT NULL, + pipeline_run_id STRING NOT NULL, + batch_id STRING, + processing_date DATE +) +PARTITION BY event_date +CLUSTER BY event_name, country_code; diff --git a/sql/create_ops_schema.sql b/sql/create_ops_schema.sql new file mode 100644 index 0000000..7c52675 --- /dev/null +++ b/sql/create_ops_schema.sql @@ -0,0 +1,6 @@ +-- Operational metadata dataset for Project Atlas orchestration. +CREATE SCHEMA IF NOT EXISTS `{project_id}.atlas_ops` +OPTIONS ( + location = '{location}', + description = 'Operational metadata and audit records for Project Atlas pipelines.' +); diff --git a/sql/create_pipeline_runs_table.sql b/sql/create_pipeline_runs_table.sql new file mode 100644 index 0000000..b312db6 --- /dev/null +++ b/sql/create_pipeline_runs_table.sql @@ -0,0 +1,26 @@ +-- One row per Airflow DAG execution, keyed by pipeline_run_id. +CREATE TABLE IF NOT EXISTS `{project_id}.atlas_ops.pipeline_runs` ( + pipeline_run_id STRING NOT NULL, + batch_id STRING NOT NULL, + airflow_run_id STRING NOT NULL, + dag_id STRING NOT NULL, + processing_date DATE NOT NULL, + started_at TIMESTAMP NOT NULL, + completed_at TIMESTAMP, + status STRING NOT NULL, + attempt_number INT64, + gcs_uri STRING, + rows_generated INT64, + rows_loaded INT64, + rows_accepted INT64, + rows_rejected INT64, + fact_rows INT64, + mart_event_count INT64, + failed_task_id STRING, + error_type STRING, + error_message STRING, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +) +PARTITION BY processing_date +CLUSTER BY status, batch_id, dag_id; diff --git a/sql/create_schema_migrations_table.sql b/sql/create_schema_migrations_table.sql new file mode 100644 index 0000000..10c7d51 --- /dev/null +++ b/sql/create_schema_migrations_table.sql @@ -0,0 +1,12 @@ +-- Migration ledger: one row per applied schema migration (Sprint 4, Phase 9). +CREATE TABLE IF NOT EXISTS `{project_id}.atlas_ops.schema_migrations` ( + migration_id STRING NOT NULL, + migration_checksum STRING NOT NULL, + git_sha STRING, + applied_at TIMESTAMP NOT NULL, + workflow_run_id STRING, + applied_by STRING, + status STRING NOT NULL, + error_summary STRING +) +CLUSTER BY migration_id; diff --git a/sql/migrate_sprint3.sql b/sql/migrate_sprint3.sql new file mode 100644 index 0000000..ab6df83 --- /dev/null +++ b/sql/migrate_sprint3.sql @@ -0,0 +1,18 @@ +-- Sprint 3 additive migration for stable batch identity on raw events. +-- Existing Sprint 1 rows remain batch_id = NULL. + +ALTER TABLE `{project_id}.{dataset_id}.events` +ADD COLUMN IF NOT EXISTS batch_id STRING +OPTIONS ( + description = 'Stable logical batch identifier used for idempotency, reruns, and backfills.' +); + +-- Logical batch processing date. Drives reproducible, ingestion-independent +-- temporal anomaly flags so historical backfills classify identically to the +-- original run (see ADR-003). Legacy Sprint 1 rows remain NULL and fall back to +-- DATE(ingested_at) in staging. +ALTER TABLE `{project_id}.{dataset_id}.events` +ADD COLUMN IF NOT EXISTS processing_date DATE +OPTIONS ( + description = 'Logical batch processing date for reproducible temporal semantics.' +); diff --git a/sql/migrations/004_create_task_events_table.sql b/sql/migrations/004_create_task_events_table.sql new file mode 100644 index 0000000..28d425a --- /dev/null +++ b/sql/migrations/004_create_task_events_table.sql @@ -0,0 +1,27 @@ +-- Migration 004 (Sprint 5, Phase 3): task-attempt audit table. +-- Grain: one row per (pipeline_run_id, task_id, attempt_number, event_type). +-- Additive only — written via idempotent MERGE from atlas.ops.task_events. +CREATE TABLE IF NOT EXISTS `atlas_ops.task_events` ( + pipeline_run_id STRING NOT NULL, + batch_id STRING, + airflow_run_id STRING, + dag_id STRING, + task_id STRING NOT NULL, + attempt_number INT64 NOT NULL, + event_type STRING NOT NULL, + status STRING, + started_at TIMESTAMP, + completed_at TIMESTAMP, + duration_ms INT64, + operator_type STRING, + environment STRING, + git_sha STRING, + rows_affected INT64, + error_type STRING, + error_message STRING, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +) +OPTIONS ( + description = 'Atlas task-attempt audit (Sprint 5). One row per task attempt event — MERGE-idempotent — errors sanitized. Controlled event types: STARTED, RETRY, SUCCESS, FAILED, SKIPPED, UPSTREAM_FAILED.' +); diff --git a/sql/migrations/005_create_quality_results_table.sql b/sql/migrations/005_create_quality_results_table.sql new file mode 100644 index 0000000..5d7872b --- /dev/null +++ b/sql/migrations/005_create_quality_results_table.sql @@ -0,0 +1,23 @@ +-- Migration 005 (Sprint 5, Phase 4): durable data-quality check results. +-- Grain: one row per data-quality check per pipeline run. +CREATE TABLE IF NOT EXISTS `atlas_ops.quality_results` ( + pipeline_run_id STRING NOT NULL, + batch_id STRING, + check_name STRING NOT NULL, + check_category STRING NOT NULL, + severity STRING NOT NULL, + status STRING NOT NULL, + observed_value FLOAT64, + expected_value FLOAT64, + lower_bound FLOAT64, + upper_bound FLOAT64, + evaluated_at TIMESTAMP NOT NULL, + model_name STRING, + details_json STRING, + git_sha STRING, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +) +OPTIONS ( + description = 'Atlas data-quality results (Sprint 5). One row per check per pipeline run — MERGE-idempotent on (pipeline_run_id, check_name). Categories: FRESHNESS, COMPLETENESS, UNIQUENESS, REFERENTIAL_INTEGRITY, SCHEMA, VOLUME, REJECTION_RATE, RECONCILIATION.' +); diff --git a/sql/migrations/006_create_monitor_evaluations_table.sql b/sql/migrations/006_create_monitor_evaluations_table.sql new file mode 100644 index 0000000..c750e5c --- /dev/null +++ b/sql/migrations/006_create_monitor_evaluations_table.sql @@ -0,0 +1,22 @@ +-- Migration 006 (Sprint 5, Phase 4): observability monitor evaluations. +-- Grain: one row per monitor check per evaluation window. +CREATE TABLE IF NOT EXISTS `atlas_ops.monitor_evaluations` ( + evaluation_id STRING NOT NULL, + check_name STRING NOT NULL, + environment STRING NOT NULL, + window_start TIMESTAMP, + window_end TIMESTAMP, + status STRING NOT NULL, + severity STRING, + observed_value FLOAT64, + threshold FLOAT64, + incident_key STRING, + source STRING, + evaluated_at TIMESTAMP NOT NULL, + details_json STRING, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +) +OPTIONS ( + description = 'Atlas observability monitor evaluations (Sprint 5). One row per monitor check per window — MERGE-idempotent on evaluation_id. Statuses: PASS, WARN, FAIL, NO_DATA, DISABLED.' +); diff --git a/sql/migrations/007_create_recovery_actions_table.sql b/sql/migrations/007_create_recovery_actions_table.sql new file mode 100644 index 0000000..a02ed35 --- /dev/null +++ b/sql/migrations/007_create_recovery_actions_table.sql @@ -0,0 +1,30 @@ +-- Migration 007 (Sprint 6, Phase 4): recovery-action audit table. +-- Grain: one row per recovery action attempt, keyed by recovery_id. +-- Additive only — written via idempotent MERGE from atlas.ops.recovery_actions. +-- Recovery actions are a separate grain from pipeline_runs and deployments — +-- they link to those records but never mutate them. +CREATE TABLE IF NOT EXISTS `atlas_ops.recovery_actions` ( + recovery_id STRING NOT NULL, + incident_id STRING, + scenario_id STRING, + pipeline_run_id STRING, + batch_id STRING, + deployment_id STRING, + action_type STRING NOT NULL, + operator STRING, + environment STRING, + started_at TIMESTAMP, + completed_at TIMESTAMP, + status STRING NOT NULL, + source_state STRING, + target_state STRING, + verification_status STRING, + error_type STRING, + error_summary STRING, + git_sha STRING, + created_at TIMESTAMP NOT NULL, + updated_at TIMESTAMP NOT NULL +) +OPTIONS ( + description = 'Atlas recovery-action audit (Sprint 6). One row per recovery attempt — MERGE-idempotent — errors sanitized. Controlled action types: RETRY_TASK, RERUN_BATCH, REPAIR_PARTIAL_LOAD, QUARANTINE_BATCH, BACKFILL, RESTORE_RELEASE, FORWARD_MIGRATION, RESTORE_IAM, REBUILD_PARTITION, PAUSE_SCHEDULE, RESUME_SCHEDULE, RECONSTRUCT_AUDIT, RESET_MONITOR, MANUAL_CONTAINMENT. Statuses: RUNNING, SUCCESS, FAILED, PARTIAL, ABORTED. SUCCESS requires verification_status=VERIFIED.' +); diff --git a/sql/migrations/008_add_task_event_timing_columns.sql b/sql/migrations/008_add_task_event_timing_columns.sql new file mode 100644 index 0000000..cc6d8e4 --- /dev/null +++ b/sql/migrations/008_add_task_event_timing_columns.sql @@ -0,0 +1,11 @@ +-- Migration 008 (Sprint 6, Phase 1): timing provenance for task events. +-- Sprint 5 limitation: FAILED rows written by the Airflow failure callback +-- carried NULL started_at/completed_at/duration_ms. Timing is now derived +-- from reliable evidence only (runner clock or Airflow task-instance +-- timestamps) and each row records where its timing came from and how +-- trustworthy it is. NULL timing stays NULL — timestamps are never invented. +ALTER TABLE `atlas_ops.task_events` + ADD COLUMN IF NOT EXISTS timing_source STRING + OPTIONS (description = 'Timing evidence origin: step_runner_clock, airflow_task_instance, or finalizer_reconciliation'), + ADD COLUMN IF NOT EXISTS timing_confidence STRING + OPTIONS (description = 'Timing trustworthiness: exact, partial (completion bounded by callback clock), or none (no reliable evidence — timing left NULL)'); diff --git a/sql/migrations/checksums.lock b/sql/migrations/checksums.lock new file mode 100644 index 0000000..b7d3451 --- /dev/null +++ b/sql/migrations/checksums.lock @@ -0,0 +1,23 @@ +{ + "breaking": { + "001_create_pipeline_runs_table": false, + "002_sprint3_raw_batch_columns": false, + "003_create_deployments_table": false, + "004_create_task_events_table": false, + "005_create_quality_results_table": false, + "006_create_monitor_evaluations_table": false, + "007_create_recovery_actions_table": false, + "008_add_task_event_timing_columns": false + }, + "checksums": { + "001_create_pipeline_runs_table": "5fb06a83e1b37918c638ec618ddd0be7e5bb57c939854d2701dd04fbeac93a1b", + "002_sprint3_raw_batch_columns": "db8b53e68ee6f5414f1b5980241e9b0c9622a1b93c6ad0f9b0fdc4453732442f", + "003_create_deployments_table": "d581c625ad1e74a54d283021c70fcf9cd4b95d13ea70958018145fb558590da5", + "004_create_task_events_table": "69cb66c50b6d2adea6aeb551eeebfa5aceec7be781f71f9b0390393560b295fe", + "005_create_quality_results_table": "09a3affd7ba91aa51ab5c2a06e97b1952a0a4ba25170584676135f4e0ad14769", + "006_create_monitor_evaluations_table": "7a964b19a87476b686c345bc577e5231330e566c3ee4ea343100488f568c2362", + "007_create_recovery_actions_table": "6cb878b1a436485a8121c7adc74b23707ee4c4b4342e3ae35123aa339606d703", + "008_add_task_event_timing_columns": "6b486af2c26e924711d6ac1f08f54e10b76ab147c5c1f7e10360cb1c1e066cec" + }, + "version": 1 +} diff --git a/sql/migrations/manifest.txt b/sql/migrations/manifest.txt new file mode 100644 index 0000000..62bf451 --- /dev/null +++ b/sql/migrations/manifest.txt @@ -0,0 +1,12 @@ +# Atlas schema migration manifest (Sprint 4, Phase 9). +# Format: | +# Applied strictly top-to-bottom. Append only — never reorder, rename, or edit +# a shipped migration; the ledger refuses to re-apply a changed file. +001_create_pipeline_runs_table|../create_pipeline_runs_table.sql +002_sprint3_raw_batch_columns|../migrate_sprint3.sql +003_create_deployments_table|../create_deployments_table.sql +004_create_task_events_table|004_create_task_events_table.sql +005_create_quality_results_table|005_create_quality_results_table.sql +006_create_monitor_evaluations_table|006_create_monitor_evaluations_table.sql +007_create_recovery_actions_table|007_create_recovery_actions_table.sql +008_add_task_event_timing_columns|008_add_task_event_timing_columns.sql diff --git a/src/atlas/__init__.py b/src/atlas/__init__.py new file mode 100644 index 0000000..3a1a359 --- /dev/null +++ b/src/atlas/__init__.py @@ -0,0 +1,3 @@ +"""Project Atlas batch ELT pipeline package.""" + +__version__ = "0.2.0" diff --git a/src/atlas/batch/__init__.py b/src/atlas/batch/__init__.py new file mode 100644 index 0000000..9195aab --- /dev/null +++ b/src/atlas/batch/__init__.py @@ -0,0 +1,19 @@ +"""Batch identity helpers for Project Atlas orchestration.""" + +from atlas.batch.context import ( + BatchContext, + build_batch_artifact_paths, + default_batch_id, + default_seed_for_date, + resolve_batch_context, + validate_batch_id, +) + +__all__ = [ + "BatchContext", + "build_batch_artifact_paths", + "default_batch_id", + "default_seed_for_date", + "resolve_batch_context", + "validate_batch_id", +] diff --git a/src/atlas/batch/context.py b/src/atlas/batch/context.py new file mode 100644 index 0000000..4474f29 --- /dev/null +++ b/src/atlas/batch/context.py @@ -0,0 +1,93 @@ +"""Stable batch identity and artifact path resolution.""" + +from __future__ import annotations + +import hashlib +import re +from dataclasses import dataclass +from datetime import date, datetime +from pathlib import Path + +from atlas.config.settings import atlas_root + +_BATCH_ID_PATTERN = re.compile(r"^[a-zA-Z0-9][a-zA-Z0-9._-]{0,127}$") + + +@dataclass(frozen=True) +class BatchContext: + """Resolved identifiers for one orchestrated batch.""" + + processing_date: str + batch_id: str + pipeline_run_id: str + seed: int + local_file_path: Path + manifest_path: Path + + +def validate_batch_id(batch_id: str) -> str: + """Validate a user-supplied batch identifier.""" + if not _BATCH_ID_PATTERN.fullmatch(batch_id): + raise ValueError( + f"batch_id must match ^[a-zA-Z0-9][a-zA-Z0-9._-]{{0,127}}$ but received {batch_id!r}" + ) + return batch_id + + +def default_batch_id(processing_date: str) -> str: + """Return the default scheduled batch identifier.""" + parsed = date.fromisoformat(processing_date) + return f"atlas-{parsed.strftime('%Y%m%d')}" + + +def default_seed_for_date(processing_date: str) -> int: + """Derive a deterministic seed from the processing date.""" + digest = hashlib.sha256(processing_date.encode("utf-8")).hexdigest() + return int(digest[:8], 16) + + +def build_batch_artifact_paths(batch_id: str) -> tuple[Path, Path]: + """Return local JSONL and manifest paths for a batch.""" + run_dir = atlas_root() / "data" / "runs" / batch_id + return run_dir / "events.jsonl", run_dir / "manifest.json" + + +def sanitize_airflow_run_id(airflow_run_id: str) -> str: + """Convert an Airflow run id into a path-safe suffix.""" + return re.sub(r"[^a-zA-Z0-9._-]+", "-", airflow_run_id).strip("-")[:120] + + +def build_pipeline_run_id(processing_date: str, airflow_run_id: str) -> str: + """Build a unique execution identity for one Airflow run.""" + suffix = sanitize_airflow_run_id(airflow_run_id) + parsed = date.fromisoformat(processing_date) + return f"atlas-airflow-{parsed.strftime('%Y%m%d')}-{suffix}" + + +def resolve_batch_context( + *, + processing_date: str | None = None, + batch_id: str | None = None, + pipeline_run_id: str | None = None, + seed: int | None = None, + airflow_run_id: str | None = None, +) -> BatchContext: + """Resolve batch context from explicit orchestration inputs.""" + if processing_date is None: + raise ValueError("processing_date is required") + datetime.strptime(processing_date, "%Y-%m-%d") + resolved_batch_id = validate_batch_id(batch_id or default_batch_id(processing_date)) + resolved_pipeline_run_id = pipeline_run_id + if resolved_pipeline_run_id is None: + if airflow_run_id is None: + raise ValueError("pipeline_run_id or airflow_run_id is required") + resolved_pipeline_run_id = build_pipeline_run_id(processing_date, airflow_run_id) + local_file_path, manifest_path = build_batch_artifact_paths(resolved_batch_id) + return BatchContext( + processing_date=processing_date, + batch_id=resolved_batch_id, + pipeline_run_id=resolved_pipeline_run_id, + seed=seed if seed is not None else default_seed_for_date(processing_date), + local_file_path=local_file_path, + manifest_path=manifest_path, + ) diff --git a/src/atlas/batch/manifest.py b/src/atlas/batch/manifest.py new file mode 100644 index 0000000..b96fd14 --- /dev/null +++ b/src/atlas/batch/manifest.py @@ -0,0 +1,58 @@ +"""Artifact manifest helpers for idempotent batch generation.""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import asdict, dataclass +from pathlib import Path + + +@dataclass(frozen=True) +class BatchManifest: + """Checksum metadata for one generated batch artifact.""" + + batch_id: str + processing_date: str + pipeline_run_id: str + seed: int + row_count: int + checksum_sha256: str + output_path: str + + def to_dict(self) -> dict[str, object]: + return asdict(self) + + +def compute_file_checksum(path: Path) -> str: + """Return the SHA-256 digest for a local file.""" + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def write_manifest(path: Path, manifest: BatchManifest) -> None: + """Write a batch manifest JSON file.""" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(manifest.to_dict(), indent=2), encoding="utf-8") + + +def read_manifest(path: Path) -> BatchManifest | None: + """Read a batch manifest when present.""" + if not path.exists(): + return None + payload = json.loads(path.read_text(encoding="utf-8")) + return BatchManifest(**payload) + + +def manifests_match(existing: BatchManifest, requested: BatchManifest) -> bool: + """Return True when an existing artifact matches the requested batch identity.""" + return ( + existing.batch_id == requested.batch_id + and existing.processing_date == requested.processing_date + and existing.seed == requested.seed + and existing.row_count == requested.row_count + and existing.checksum_sha256 == requested.checksum_sha256 + ) diff --git a/src/atlas/config/__init__.py b/src/atlas/config/__init__.py new file mode 100644 index 0000000..b2c80ae --- /dev/null +++ b/src/atlas/config/__init__.py @@ -0,0 +1,15 @@ +"""Configuration package for Project Atlas.""" + +from atlas.config.settings import ( + AtlasSettings, + load_settings, + staging_table_id, + table_fqn, +) + +__all__ = [ + "AtlasSettings", + "load_settings", + "staging_table_id", + "table_fqn", +] diff --git a/src/atlas/config/settings.py b/src/atlas/config/settings.py new file mode 100644 index 0000000..dcf322e --- /dev/null +++ b/src/atlas/config/settings.py @@ -0,0 +1,228 @@ +"""Configuration loading and validation for Project Atlas. + +Purpose: + Centralizes runtime settings so every pipeline step reads the same + project, bucket, dataset, and validation thresholds. + +Interactions: + Used by generator, ingestion, loader, validation, and CLI scripts. + Reads ``config/atlas.yaml`` and ``config/anomaly_profile.yaml``. + +Engineering principles: + - Configuration over hardcoding for reproducibility across Cloud Shell + and Cursor Cloud Agent environments. + - Environment variable overrides keep secrets out of source control. + +Common failure modes: + - Missing ``GCP_PROJECT_ID`` or ``ATLAS_GCP_PROJECT_ID`` in cloud runs. + - Bucket name collisions if the logical name is used without project suffix. + +Implementation choice: + YAML + dataclasses were chosen over environment-only config because Sprint 1 + needs documented defaults and anomaly profiles that acceptance tests can + assert against. Alternatives considered: pure env vars (harder to review) + and Pydantic Settings (heavier dependency for a focused pipeline). +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import yaml + +PROJECT_ROOT = Path(__file__).resolve().parents[3] +DEFAULT_CONFIG_PATH = PROJECT_ROOT / "config" / "atlas.yaml" +DEFAULT_ANOMALY_PATH = PROJECT_ROOT / "config" / "anomaly_profile.yaml" + + +def atlas_root() -> Path: + """Return the Atlas project root, honoring ATLAS_ROOT for Composer layouts.""" + override = _env("ATLAS_ROOT") + if override: + return Path(override).expanduser().resolve() + return PROJECT_ROOT + + +@dataclass(frozen=True) +class GcpConfig: + """GCP resource identifiers for Atlas Sprint 1.""" + + project_id: str + location: str + bucket_name: str + bucket_logical_name: str + dataset_id: str + table_id: str + + +@dataclass(frozen=True) +class GeneratorConfig: + """Synthetic event generator settings.""" + + event_count: int + random_seed: int + output_dir: Path + + +@dataclass(frozen=True) +class IngestionConfig: + """Cloud Storage ingestion settings.""" + + gcs_prefix: str + + +@dataclass(frozen=True) +class LoaderConfig: + """BigQuery loader settings.""" + + staging_table_suffix: str + + +@dataclass(frozen=True) +class ValidationConfig: + """Validation thresholds.""" + + expected_event_count: int + future_date_field: str + + +@dataclass(frozen=True) +class LoggingConfig: + """Structured logging settings.""" + + log_dir: Path + log_format: str + + +@dataclass(frozen=True) +class AnomalyProfile: + """Expected seeded anomaly counts for generator and acceptance tests.""" + + anomalies: dict[str, dict[str, Any]] + valid_country_codes: list[str] + event_names: list[str] + platforms: list[str] + app_versions: list[str] + + def expected_count(self, anomaly_type: str) -> int: + """Return configured anomaly count for a named anomaly type.""" + return int(self.anomalies[anomaly_type]["count"]) + + +@dataclass(frozen=True) +class AtlasSettings: + """Fully resolved Atlas runtime settings.""" + + gcp: GcpConfig + generator: GeneratorConfig + ingestion: IngestionConfig + loader: LoaderConfig + validation: ValidationConfig + logging: LoggingConfig + anomaly_profile: AnomalyProfile + config_path: Path + anomaly_path: Path + + +def _env(name: str, default: str | None = None) -> str | None: + """Read an environment variable with optional default.""" + return os.environ.get(name, default) + + +def _env_str(name: str, default: str) -> str: + """Read an environment variable with a required string default.""" + value = os.environ.get(name) + return value if value else default + + +def _require_env(*names: str) -> str: + """Return the first populated environment variable.""" + for name in names: + value = _env(name) + if value: + return value + joined = ", ".join(names) + raise ValueError(f"Required environment variable not set. Provide one of: {joined}") + + +def load_yaml(path: Path) -> dict[str, Any]: + """Load a YAML document from disk.""" + with path.open("r", encoding="utf-8") as handle: + return yaml.safe_load(handle) + + +def load_settings( + config_path: Path | None = None, + anomaly_path: Path | None = None, +) -> AtlasSettings: + """Load and validate Atlas settings from YAML with env overrides.""" + root = atlas_root() + config_path = config_path or root / "config" / "atlas.yaml" + anomaly_path = anomaly_path or root / "config" / "anomaly_profile.yaml" + + raw = load_yaml(config_path) + anomaly_raw = load_yaml(anomaly_path) + + project_id = _env_str("ATLAS_GCP_PROJECT_ID", _env_str("GCP_PROJECT_ID", raw["gcp"]["project_id"])) + bucket_name = _env_str("ATLAS_GCS_BUCKET", raw["gcp"]["bucket_name"]) + + gcp = GcpConfig( + project_id=project_id, + location=raw["gcp"]["location"], + bucket_name=bucket_name, + bucket_logical_name=raw["gcp"]["bucket_logical_name"], + # ATLAS_BQ_DATASET matches the override dbt sources already honor, + # and lets CI redirect raw loads into isolated atlas_ci_* datasets. + dataset_id=_env_str("ATLAS_BQ_DATASET", raw["gcp"]["dataset_id"]), + table_id=raw["gcp"]["table_id"], + ) + generator = GeneratorConfig( + event_count=int(_env_str("ATLAS_EVENT_COUNT", str(raw["generator"]["event_count"]))), + random_seed=int(_env_str("ATLAS_RANDOM_SEED", str(raw["generator"]["random_seed"]))), + output_dir=root / raw["generator"]["output_dir"], + ) + ingestion = IngestionConfig(gcs_prefix=_env_str("ATLAS_GCS_PREFIX", raw["ingestion"]["gcs_prefix"])) + loader = LoaderConfig(staging_table_suffix=raw["loader"]["staging_table_suffix"]) + validation = ValidationConfig( + expected_event_count=int( + _env_str("ATLAS_EXPECTED_EVENT_COUNT", str(raw["validation"]["expected_event_count"])) + ), + future_date_field=raw["validation"]["future_date_field"], + ) + logging_cfg = LoggingConfig( + log_dir=root / raw["logging"]["log_dir"], + log_format=raw["logging"]["log_format"], + ) + anomaly_profile = AnomalyProfile( + anomalies=anomaly_raw["anomalies"], + valid_country_codes=anomaly_raw["valid_country_codes"], + event_names=anomaly_raw["event_names"], + platforms=anomaly_raw["platforms"], + app_versions=anomaly_raw["app_versions"], + ) + + return AtlasSettings( + gcp=gcp, + generator=generator, + ingestion=ingestion, + loader=loader, + validation=validation, + logging=logging_cfg, + anomaly_profile=anomaly_profile, + config_path=config_path, + anomaly_path=anomaly_path, + ) + + +def table_fqn(settings: AtlasSettings) -> str: + """Return fully qualified BigQuery table name.""" + return f"{settings.gcp.project_id}.{settings.gcp.dataset_id}.{settings.gcp.table_id}" + + +def staging_table_id(settings: AtlasSettings, run_id: str) -> str: + """Return a run-scoped staging table id.""" + safe_run_id = run_id.replace("-", "_") + return f"{settings.gcp.table_id}{settings.loader.staging_table_suffix}_{safe_run_id}" diff --git a/src/atlas/failure_injection/__init__.py b/src/atlas/failure_injection/__init__.py new file mode 100644 index 0000000..52e9b26 --- /dev/null +++ b/src/atlas/failure_injection/__init__.py @@ -0,0 +1 @@ +"""Controlled fault-injection framework (Sprint 6, ADR-013).""" diff --git a/src/atlas/failure_injection/cli.py b/src/atlas/failure_injection/cli.py new file mode 100644 index 0000000..811eccf --- /dev/null +++ b/src/atlas/failure_injection/cli.py @@ -0,0 +1,162 @@ +"""Failure-scenario lifecycle CLI (Sprint 6, ADR-013). + +Invoked through ``scripts/run_failure_scenario.sh``. Commands: + +- ``plan`` — print the scenario spec, affected resources, and gates (safe) +- ``status`` — print gate/approval state without mutating anything (safe) +- ``run`` — authorize and start one scenario (gated) +- ``verify`` — print the scenario's verification queries to execute (safe) +- ``recover`` — print the controlled recovery action and audit template (safe) +- ``cleanup`` — print/emit the scenario cleanup contract (gated telemetry) + +``run`` performs the authorization chain and emits structured telemetry, then +prints the exact injection steps for the operator/agent to execute inside the +live game-day window. It never mutates cloud resources by itself: every +mutation is an explicit, logged operator command from the printed plan, which +keeps LOW/MEDIUM/HIGH scenarios reviewable and prevents this CLI from +becoming an unattended destruction engine. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from typing import Any + +from atlas.failure_injection.framework import ( + APPROVAL_VAR, + SCENARIO_VAR, + InjectionRefused, + authorize_injection, + is_injection_requested, +) +from atlas.failure_injection.registry import get_scenario, load_catalog, validate_catalog +from atlas.observability.logging import emit_event + +SAFE_COMMANDS = frozenset({"plan", "status", "verify", "recover"}) +GATED_COMMANDS = frozenset({"run", "cleanup"}) + + +def _print_spec(spec: dict[str, Any]) -> None: + print(json.dumps(spec, indent=2, default=str)) + + +def _plan(spec: dict[str, Any]) -> int: + print(f"== PLAN {spec['scenario_id']} ({spec['category']}, risk {spec['risk_level']}) ==") + _print_spec(spec) + print("\nAffected resources / target component:") + print(f" {spec['target_component']}") + print("Approvals required before `run`:") + for approval in spec["approval_required"]: + print(f" {approval}=true") + print(f"Maximum duration: {spec['maximum_duration_minutes']} minutes") + print(f"Maximum cost: ${spec['maximum_cost_usd']}") + return 0 + + +def _status(spec: dict[str, Any], environment: str) -> int: + state = { + "scenario_id": spec["scenario_id"], + "environment": environment, + "scenario_requested": is_injection_requested(), + "requested_scenario": os.environ.get(SCENARIO_VAR, ""), + "approvals": {a: os.environ.get(a, "unset") for a in spec["approval_required"]}, + "execution_mode": spec["execution_mode"], + } + print(json.dumps(state, indent=2)) + return 0 + + +def _run(spec: dict[str, Any], environment: str, batch_id: str | None) -> int: + try: + authorization = authorize_injection(spec["scenario_id"], environment=environment, batch_id=batch_id) + except InjectionRefused as exc: + print(f"REFUSED: {exc}", file=sys.stderr) + return 2 + print(f"== AUTHORIZED {spec['scenario_id']} until {authorization.deadline.isoformat()} ==") + print("Injection method (execute exactly, inside the game-day window):") + print(f" {spec['injection_method']}") + print("Expected detection:") + print(f" {spec['expected_detection']}") + print("Expected containment:") + print(f" {spec['expected_containment']}") + print("Allowed data impact (anything beyond this aborts the scenario):") + print(f" {spec['allowed_data_impact']}") + return 0 + + +def _verify(spec: dict[str, Any]) -> int: + print(f"== VERIFY {spec['scenario_id']} ==") + for query in spec["verification_queries"]: + print(f" - {query}") + return 0 + + +def _recover(spec: dict[str, Any]) -> int: + print(f"== RECOVER {spec['scenario_id']} ==") + print(f"Controlled recovery action(s): {spec['recovery_action']}") + print( + "Record the attempt in atlas_ops.recovery_actions via " + "atlas.ops.recovery_actions (SUCCESS requires verification_status=VERIFIED)." + ) + return 0 + + +def _cleanup(spec: dict[str, Any], environment: str) -> int: + emit_event( + "failure_injection_cleanup", + severity="INFO", + component="failure_injection", + check_name=spec["scenario_id"], + status="CLEANUP", + environment=environment, + ) + print(f"== CLEANUP {spec['scenario_id']} ==") + print(f" {spec['cleanup']}") + return 0 + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Atlas controlled failure-scenario lifecycle") + parser.add_argument("command", choices=sorted(SAFE_COMMANDS | GATED_COMMANDS | {"validate"})) + parser.add_argument("--scenario", help="explicit scenario id (required for all but validate)") + parser.add_argument("--environment", help="explicit target environment (required for run/cleanup)") + parser.add_argument("--batch-id", default=None, help="isolated batch id (atlas-s6- prefix)") + args = parser.parse_args(argv) + + if args.command == "validate": + errors = validate_catalog(load_catalog()) + for error in errors: + print(f"INVALID {error}", file=sys.stderr) + print(f"{'INVALID' if errors else 'VALID'}: failure-scenario catalog") + return 1 if errors else 0 + + if not args.scenario: + parser.error("--scenario is required (fault injection never runs implicitly)") + spec = get_scenario(args.scenario) + + if args.command in {"run", "cleanup", "status"} and not args.environment: + parser.error("--environment is required (no implicit environment)") + + if args.command == "plan": + return _plan(spec) + if args.command == "status": + return _status(spec, args.environment) + if args.command == "run": + if os.environ.get(APPROVAL_VAR, "").lower() != "true": + print(f"REFUSED: {APPROVAL_VAR}=true is required", file=sys.stderr) + return 2 + return _run(spec, args.environment, args.batch_id) + if args.command == "verify": + return _verify(spec) + if args.command == "recover": + return _recover(spec) + if args.command == "cleanup": + return _cleanup(spec, args.environment) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/failure_injection/framework.py b/src/atlas/failure_injection/framework.py new file mode 100644 index 0000000..bde301a --- /dev/null +++ b/src/atlas/failure_injection/framework.py @@ -0,0 +1,180 @@ +"""Fault-injection activation guards (Sprint 6, ADR-013). + +Safety contract, enforced here and regression-tested in CI: + +- **Disabled by default.** Injection activates only when an explicit scenario + id is supplied AND ``ATLAS_APPROVE_FAILURE_INJECTION=true`` AND every + scenario-specific approval variable is true. +- **Never scheduled.** Activation is refused inside scheduled Airflow runs; + only manually triggered runs may carry a drill. +- **Never canonical.** Batch ids must carry the isolated ``atlas-s6-`` prefix; + canonical and smoke batch identities are refused. +- **Never production-like.** Only the approved development environment is + injectable. +- **Never inherited.** Activation requires the explicit + ``ATLAS_INJECTION_SCENARIO`` parameter naming the scenario; a lingering + approval variable alone can never activate an injection. +- **Bounded.** Every scenario carries a maximum duration; ``deadline`` turns + it into an absolute timeout. + +There is no fallback path: a refused injection raises ``InjectionRefused`` +and the caller must stop. Injection helpers never degrade into normal +execution silently. +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from datetime import UTC, datetime, timedelta +from typing import Any + +from atlas.failure_injection.registry import get_scenario +from atlas.observability.logging import emit_event + +APPROVAL_VAR = "ATLAS_APPROVE_FAILURE_INJECTION" +SCENARIO_VAR = "ATLAS_INJECTION_SCENARIO" +ISOLATED_BATCH_PREFIX = "atlas-s6-" + +INJECTABLE_ENVIRONMENTS = frozenset({"atlas-dev"}) +_PRODUCTION_MARKERS = ("prod", "production") + + +class InjectionRefused(RuntimeError): + """A fault-injection request failed the safety gates.""" + + +@dataclass(frozen=True) +class InjectionAuthorization: + """Proof that one scenario passed every activation gate.""" + + scenario_id: str + environment: str + batch_id: str | None + deadline: datetime + spec: dict[str, Any] + + +def _is_true(value: str | None) -> bool: + return (value or "").strip().lower() == "true" + + +def is_injection_requested(env: dict[str, str] | None = None) -> bool: + """True only when an explicit scenario parameter is present.""" + env = env if env is not None else dict(os.environ) + return bool(env.get(SCENARIO_VAR, "").strip()) + + +def authorize_injection( + scenario_id: str, + *, + environment: str, + batch_id: str | None = None, + env: dict[str, str] | None = None, + catalog: dict[str, Any] | None = None, +) -> InjectionAuthorization: + """Validate every activation gate for one scenario or raise InjectionRefused.""" + env = env if env is not None else dict(os.environ) + spec = get_scenario(scenario_id, catalog) + + requested = env.get(SCENARIO_VAR, "").strip() + if requested != scenario_id: + raise InjectionRefused( + f"{SCENARIO_VAR} must explicitly name {scenario_id!r} (got {requested!r}); " + "fault injection never activates through environment inheritance" + ) + + for approval in spec["approval_required"]: + if not _is_true(env.get(approval)): + raise InjectionRefused(f"missing approval: {approval}=true is required for {scenario_id}") + + if environment not in INJECTABLE_ENVIRONMENTS or any( + m in environment.lower() for m in _PRODUCTION_MARKERS + ): + raise InjectionRefused( + f"environment {environment!r} is not injectable (allowed: {sorted(INJECTABLE_ENVIRONMENTS)})" + ) + + run_type = env.get("AIRFLOW_CTX_DAG_RUN_TYPE", "").lower() + if run_type == "scheduled": + raise InjectionRefused("fault injection refuses scheduled execution; trigger manually") + run_id = env.get("AIRFLOW_CTX_DAG_RUN_ID", "") + if run_id.startswith("scheduled__"): + raise InjectionRefused("fault injection refuses scheduled run ids") + + if batch_id is not None and not batch_id.startswith(ISOLATED_BATCH_PREFIX): + raise InjectionRefused( + f"batch_id {batch_id!r} is not isolated; injection requires the " + f"{ISOLATED_BATCH_PREFIX!r} prefix and refuses canonical batch ids" + ) + + deadline = datetime.now(tz=UTC) + timedelta(minutes=float(spec["maximum_duration_minutes"])) + authorization = InjectionAuthorization( + scenario_id=scenario_id, + environment=environment, + batch_id=batch_id, + deadline=deadline, + spec=spec, + ) + emit_event( + "failure_injection_authorized", + severity="WARNING", + component="failure_injection", + batch_id=batch_id, + check_name=scenario_id, + status="AUTHORIZED", + environment=environment, + ) + return authorization + + +def enforce_deadline(authorization: InjectionAuthorization) -> None: + """Raise when a scenario has exceeded its maximum duration.""" + if datetime.now(tz=UTC) > authorization.deadline: + emit_event( + "failure_injection_timeout", + severity="ERROR", + component="failure_injection", + check_name=authorization.scenario_id, + status="TIMEOUT", + ) + raise InjectionRefused( + f"{authorization.scenario_id} exceeded maximum_duration; abort and run cleanup" + ) + + +def injection_active_for( + scenario_id: str, + *, + batch_id: str | None = None, + env: dict[str, str] | None = None, +) -> bool: + """Cheap hook check used inside pipeline code paths. + + Returns True only when the full authorization chain passes. Any refusal + returns False — a hook can never break a normal (non-drill) run — but the + refusal is NOT silent when a scenario was explicitly requested: that + misconfiguration is logged before returning False. + """ + env = env if env is not None else dict(os.environ) + if env.get(SCENARIO_VAR, "").strip() != scenario_id: + return False + try: + authorize_injection( + scenario_id, + environment=env.get("ATLAS_ENVIRONMENT", "atlas-dev"), + batch_id=batch_id, + env=env, + ) + return True + except InjectionRefused as exc: + emit_event( + "failure_injection_refused", + severity="ERROR", + component="failure_injection", + check_name=scenario_id, + status="REFUSED", + error_type="InjectionRefused", + error_message=str(exc), + ) + return False diff --git a/src/atlas/failure_injection/registry.py b/src/atlas/failure_injection/registry.py new file mode 100644 index 0000000..af05ac0 --- /dev/null +++ b/src/atlas/failure_injection/registry.py @@ -0,0 +1,171 @@ +"""Failure-scenario catalog loading and schema validation (ADR-013). + +The catalog (``config/failure_scenarios.yaml``) is the single source of truth +for every controlled failure scenario. CI validates the full catalog schema on +every run so a malformed or under-specified scenario can never reach a game +day. +""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from typing import Any + +import yaml + +CATALOG_PATH = Path(__file__).resolve().parents[3] / "config" / "failure_scenarios.yaml" + +REQUIRED_FIELDS = ( + "category", + "description", + "risk_level", + "target_component", + "preconditions", + "injection_method", + "expected_detection", + "expected_alert", + "expected_containment", + "allowed_data_impact", + "recovery_action", + "verification_queries", + "cleanup", + "recurrence_prevention", + "execution_mode", +) + +ALLOWED_CATEGORIES = frozenset( + { + "INGESTION", + "ORCHESTRATION", + "WAREHOUSE", + "SCHEMA", + "IAM", + "DEPLOYMENT", + "ROLLBACK", + "OBSERVABILITY", + "COST", + } +) +ALLOWED_RISK_LEVELS = frozenset({"LOW", "MEDIUM", "HIGH"}) +ALLOWED_EXECUTION_MODES = frozenset({"unit", "live", "both"}) + +# Recovery actions must come from the controlled atlas_ops.recovery_actions +# vocabulary; compound values like "QUARANTINE_BATCH then RERUN_BATCH" are +# allowed as long as every referenced action is controlled. +_CONTROLLED_ACTIONS = ( + "RETRY_TASK", + "RERUN_BATCH", + "REPAIR_PARTIAL_LOAD", + "QUARANTINE_BATCH", + "BACKFILL", + "RESTORE_RELEASE", + "FORWARD_MIGRATION", + "RESTORE_IAM", + "REBUILD_PARTITION", + "PAUSE_SCHEDULE", + "RESUME_SCHEDULE", + "RECONSTRUCT_AUDIT", + "RESET_MONITOR", + "MANUAL_CONTAINMENT", +) + + +def load_catalog(path: Path | None = None) -> dict[str, Any]: + """Load and parse the failure-scenario catalog.""" + return yaml.safe_load((path or CATALOG_PATH).read_text(encoding="utf-8")) + + +def validate_catalog(catalog: dict[str, Any]) -> list[str]: + """Return every schema violation in the catalog (empty list == valid).""" + errors: list[str] = [] + scenarios = catalog.get("scenarios") + if not isinstance(scenarios, dict) or not scenarios: + return ["catalog has no scenarios mapping"] + defaults = catalog.get("defaults", {}) + + for scenario_id, spec in scenarios.items(): + prefix = f"{scenario_id}: " + if not scenario_id.startswith("S6-"): + errors.append(prefix + "scenario id must start with S6-") + if not isinstance(spec, dict): + errors.append(prefix + "scenario body must be a mapping") + continue + for field in REQUIRED_FIELDS: + if field not in spec: + errors.append(prefix + f"missing required field {field!r}") + category = spec.get("category") + if category not in ALLOWED_CATEGORIES: + errors.append(prefix + f"invalid category {category!r}") + elif category: + expected_prefixes = { + "INGESTION": "S6-ING-", + "ORCHESTRATION": "S6-AIR-", + "WAREHOUSE": "S6-DBT-", + "SCHEMA": "S6-SCH-", + "IAM": "S6-IAM-", + "DEPLOYMENT": "S6-DEP-", + "ROLLBACK": "S6-RBK-", + "OBSERVABILITY": "S6-OBS-", + "COST": "S6-COST-", + } + if not scenario_id.startswith(expected_prefixes[category]): + errors.append(prefix + f"id prefix does not match category {category}") + risk = spec.get("risk_level") + if risk not in ALLOWED_RISK_LEVELS: + errors.append(prefix + f"invalid risk_level {risk!r} (no CRITICAL scenarios exist)") + mode = spec.get("execution_mode") + if mode not in ALLOWED_EXECUTION_MODES: + errors.append(prefix + f"invalid execution_mode {mode!r}") + recovery = str(spec.get("recovery_action", "")) + if recovery and not any(action in recovery for action in _CONTROLLED_ACTIONS): + errors.append(prefix + f"recovery_action {recovery!r} references no controlled action type") + approvals = spec.get("approval_required", defaults.get("approval_required", [])) + if "ATLAS_APPROVE_FAILURE_INJECTION" not in approvals: + errors.append(prefix + "ATLAS_APPROVE_FAILURE_INJECTION must always be required") + max_cost = spec.get("maximum_cost_usd", defaults.get("maximum_cost_usd")) + if not isinstance(max_cost, (int, float)) or max_cost > 1.0: + errors.append(prefix + f"maximum_cost_usd {max_cost!r} missing or above the $1 scenario ceiling") + max_duration = spec.get("maximum_duration_minutes", defaults.get("maximum_duration_minutes")) + if not isinstance(max_duration, (int, float)) or max_duration > 120: + errors.append(prefix + f"maximum_duration_minutes {max_duration!r} missing or above 120") + return errors + + +def get_scenario(scenario_id: str, catalog: dict[str, Any] | None = None) -> dict[str, Any]: + """Return one scenario spec with catalog defaults merged in.""" + catalog = catalog or load_catalog() + scenarios = catalog.get("scenarios", {}) + if scenario_id not in scenarios: + raise KeyError(f"unknown failure scenario: {scenario_id}") + defaults = catalog.get("defaults", {}) + merged = {**defaults, **scenarios[scenario_id], "scenario_id": scenario_id} + merged.setdefault("approval_required", defaults.get("approval_required", [])) + return merged + + +def main() -> int: + parser = argparse.ArgumentParser(description="Atlas failure-scenario catalog tools") + parser.add_argument("--validate", action="store_true", help="validate the catalog schema") + parser.add_argument("--show", metavar="SCENARIO_ID", help="print one merged scenario spec") + args = parser.parse_args() + + catalog = load_catalog() + if args.validate: + errors = validate_catalog(catalog) + if errors: + for error in errors: + print(f"INVALID {error}") + return 1 + print(f"VALID {len(catalog['scenarios'])} scenarios pass schema validation") + return 0 + if args.show: + print(json.dumps(get_scenario(args.show, catalog), indent=2, default=str)) + return 0 + parser.print_help() + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/generator/__init__.py b/src/atlas/generator/__init__.py new file mode 100644 index 0000000..124dff8 --- /dev/null +++ b/src/atlas/generator/__init__.py @@ -0,0 +1,5 @@ +"""Event generator package.""" + +from atlas.generator.events import EventRecord, GenerationResult, generate_events, write_jsonl + +__all__ = ["EventRecord", "GenerationResult", "generate_events", "write_jsonl"] diff --git a/src/atlas/generator/events.py b/src/atlas/generator/events.py new file mode 100644 index 0000000..385cc04 --- /dev/null +++ b/src/atlas/generator/events.py @@ -0,0 +1,290 @@ +"""Synthetic event generator for Project Atlas Sprint 1. + +Purpose: + Produce 50,000 reproducible JSON events with seeded data-quality anomalies. + +Interactions: + Writes local JSONL consumed by ingestion upload and referenced by validation + acceptance tests via ``config/anomaly_profile.yaml``. + +Engineering principles: + - Deterministic random seed for reproducibility. + - Explicit anomaly injection for validation and failure simulation. + +Common failure modes: + - Output directory missing or not writable. + - Anomaly counts exceeding total event count. + +Implementation choice: + Pure Python generation keeps Cloud Shell execution simple and testable. + Alternatives considered: Faker (extra dependency) and SQL generation + (premature for raw JSONL ingestion). +""" + +from __future__ import annotations + +import json +import random +import uuid +from collections.abc import Iterator +from dataclasses import asdict, dataclass, replace +from datetime import UTC, date, datetime, timedelta +from pathlib import Path + +from atlas.batch.context import resolve_batch_context +from atlas.batch.manifest import ( + BatchManifest, + compute_file_checksum, + manifests_match, + read_manifest, + write_manifest, +) +from atlas.config.settings import AnomalyProfile, AtlasSettings + + +@dataclass(frozen=True) +class EventRecord: + """Canonical Atlas raw event schema.""" + + event_id: str + user_id: str | None + event_name: str + event_timestamp: str + event_date: str + country_code: str + platform: str + app_version: str + ingested_at: str | None = None + + def to_dict(self) -> dict[str, str | None]: + """Convert the event to JSONL fields (ingested_at is added at load time).""" + payload = asdict(self) + payload.pop("ingested_at", None) + return payload + + +@dataclass(frozen=True) +class GenerationResult: + """Summary of a generator run.""" + + output_path: Path + event_count: int + anomaly_counts: dict[str, int] + primary_event_date: str + batch_id: str | None = None + pipeline_run_id: str | None = None + seed: int | None = None + reused_existing: bool = False + checksum_sha256: str | None = None + + +def _random_timestamp(rng: random.Random, base_day: date) -> datetime: + """Create a timestamp within the base day.""" + hour = rng.randint(0, 23) + minute = rng.randint(0, 59) + second = rng.randint(0, 59) + return datetime(base_day.year, base_day.month, base_day.day, hour, minute, second, tzinfo=UTC) + + +def _base_event( + rng: random.Random, + profile: AnomalyProfile, + base_day: date, + event_id: str | None = None, +) -> EventRecord: + """Create a valid baseline event.""" + timestamp = _random_timestamp(rng, base_day) + # Derive the UUID from the seeded RNG (not uuid4/os.urandom) so the same + # batch identity regenerates byte-identical artifacts on any machine. + return EventRecord( + event_id=event_id or str(uuid.UUID(int=rng.getrandbits(128), version=4)), + user_id=str(rng.randint(1, 100000)), + event_name=rng.choice(profile.event_names), + event_timestamp=timestamp.isoformat(), + event_date=timestamp.date().isoformat(), + country_code=rng.choice(profile.valid_country_codes), + platform=rng.choice(profile.platforms), + app_version=rng.choice(profile.app_versions), + ) + + +def _generate_event_rows( + settings: AtlasSettings, + *, + seed: int, + processing_date: str, +) -> tuple[list[EventRecord], dict[str, int], str]: + profile = settings.anomaly_profile + rng = random.Random(seed) + base_day = date.fromisoformat(processing_date) + events: list[EventRecord] = [] + anomaly_counts = {name: 0 for name in profile.anomalies} + + total = settings.generator.event_count + for _ in range(total): + events.append(_base_event(rng, profile, base_day)) + + duplicate_count = profile.expected_count("duplicate_event_ids") + group_size = 2 + num_groups = duplicate_count // group_size + source_indices = rng.sample(range(total), num_groups) + reserved = set(source_indices) + target_candidates = [index for index in range(total) if index not in reserved] + target_indices = rng.sample(target_candidates, num_groups) + for source_index, target_index in zip(source_indices, target_indices, strict=True): + events[target_index] = replace(events[target_index], event_id=events[source_index].event_id) + anomaly_counts["duplicate_event_ids"] += 1 + + for index in rng.sample(range(total), profile.expected_count("null_user_ids")): + events[index] = replace(events[index], user_id=None) + anomaly_counts["null_user_ids"] += 1 + + invalid_countries = ["XX", "ZZ", "INVALID"] + for index in rng.sample(range(total), profile.expected_count("invalid_country_codes")): + events[index] = replace(events[index], country_code=rng.choice(invalid_countries)) + anomaly_counts["invalid_country_codes"] += 1 + + for index in rng.sample(range(total), profile.expected_count("future_timestamps")): + original = events[index] + future_day = base_day + timedelta(days=rng.randint(1, 7)) + future_ts = datetime( + future_day.year, + future_day.month, + future_day.day, + 12, + 0, + 0, + tzinfo=UTC, + ) + events[index] = replace( + original, + event_timestamp=future_ts.isoformat(), + event_date=future_ts.date().isoformat(), + ) + anomaly_counts["future_timestamps"] += 1 + + for index in rng.sample(range(total), profile.expected_count("late_arriving_events")): + original = events[index] + late_date = base_day - timedelta(days=rng.randint(1, 5)) + timestamp = datetime.fromisoformat(original.event_timestamp) + events[index] = replace( + original, + event_date=late_date.isoformat(), + event_timestamp=timestamp.isoformat(), + ) + anomaly_counts["late_arriving_events"] += 1 + + return events, anomaly_counts, base_day.isoformat() + + +def write_jsonl(path: Path, records: Iterator[dict[str, str | None]]) -> None: + """Write records to a JSONL file.""" + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as handle: + for record in records: + handle.write(json.dumps(record, default=str)) + handle.write("\n") + + +def generate_events(settings: AtlasSettings) -> GenerationResult: + """Generate seeded synthetic events and write JSONL output.""" + return generate_events_for_batch( + settings, + processing_date=date.today().isoformat(), + batch_id=None, + pipeline_run_id=None, + seed=settings.generator.random_seed, + output_path=settings.generator.output_dir / f"events_seed_{settings.generator.random_seed}.jsonl", + ) + + +def generate_events_for_batch( + settings: AtlasSettings, + *, + processing_date: str, + batch_id: str | None, + pipeline_run_id: str | None, + seed: int | None, + output_path: Path | None = None, +) -> GenerationResult: + """Generate or reuse a batch artifact for orchestrated runs.""" + resolved_batch_id: str | None + resolved_pipeline_run_id: str | None + if batch_id and pipeline_run_id: + context = resolve_batch_context( + processing_date=processing_date, + batch_id=batch_id, + pipeline_run_id=pipeline_run_id, + seed=seed, + ) + target_path = output_path or context.local_file_path + manifest_path = context.manifest_path + resolved_seed = context.seed + resolved_batch_id = context.batch_id + resolved_pipeline_run_id = context.pipeline_run_id + + if target_path.exists() and manifest_path.exists(): + existing_manifest = read_manifest(manifest_path) + if existing_manifest is not None: + requested_manifest = BatchManifest( + batch_id=resolved_batch_id, + processing_date=processing_date, + pipeline_run_id=resolved_pipeline_run_id, + seed=resolved_seed, + row_count=existing_manifest.row_count, + checksum_sha256=compute_file_checksum(target_path), + output_path=str(target_path), + ) + if manifests_match(existing_manifest, requested_manifest): + return GenerationResult( + output_path=target_path, + event_count=existing_manifest.row_count, + anomaly_counts={}, + primary_event_date=processing_date, + batch_id=existing_manifest.batch_id, + pipeline_run_id=existing_manifest.pipeline_run_id, + seed=existing_manifest.seed, + reused_existing=True, + checksum_sha256=existing_manifest.checksum_sha256, + ) + raise ValueError( + f"Existing batch artifact conflicts with requested batch identity for {target_path}" + ) + else: + target_path = output_path or ( + settings.generator.output_dir / f"events_seed_{settings.generator.random_seed}.jsonl" + ) + manifest_path = target_path.with_suffix(".manifest.json") + resolved_seed = seed if seed is not None else settings.generator.random_seed + resolved_batch_id = batch_id + resolved_pipeline_run_id = pipeline_run_id + + events, anomaly_counts, primary_event_date = _generate_event_rows( + settings, + seed=resolved_seed, + processing_date=processing_date, + ) + write_jsonl(target_path, (event.to_dict() for event in events)) + checksum = compute_file_checksum(target_path) + manifest = BatchManifest( + batch_id=resolved_batch_id or f"legacy-{resolved_seed}", + processing_date=processing_date, + pipeline_run_id=resolved_pipeline_run_id or f"legacy-{resolved_seed}", + seed=resolved_seed, + row_count=len(events), + checksum_sha256=checksum, + output_path=str(target_path), + ) + write_manifest(manifest_path, manifest) + + return GenerationResult( + output_path=target_path, + event_count=len(events), + anomaly_counts=anomaly_counts, + primary_event_date=primary_event_date, + batch_id=resolved_batch_id, + pipeline_run_id=resolved_pipeline_run_id, + seed=resolved_seed, + reused_existing=False, + checksum_sha256=checksum, + ) diff --git a/src/atlas/governance/__init__.py b/src/atlas/governance/__init__.py new file mode 100644 index 0000000..756b16f --- /dev/null +++ b/src/atlas/governance/__init__.py @@ -0,0 +1,28 @@ +"""Atlas governance package (Sprint 7). + +Governance source of truth (ADR-016): +- dbt models are governed by their dbt ``meta.governance`` blocks. +- Non-dbt assets are governed by ``governance/non_dbt_assets.yml``. +- A consolidated catalog is *generated* from both; it is never hand-edited. + +This package is import-safe with no cloud dependencies so it can run in +credentialless CI. +""" + +from atlas.governance.registry import ( + GovernanceError, + build_asset_index, + load_dbt_model_governance, + load_non_dbt_assets, + load_policy, + validate_governance, +) + +__all__ = [ + "GovernanceError", + "build_asset_index", + "load_dbt_model_governance", + "load_non_dbt_assets", + "load_policy", + "validate_governance", +] diff --git a/src/atlas/governance/catalog.py b/src/atlas/governance/catalog.py new file mode 100644 index 0000000..13bd6a7 --- /dev/null +++ b/src/atlas/governance/catalog.py @@ -0,0 +1,125 @@ +"""Generate the consolidated Atlas asset catalog (Sprint 7, ADR-016). + +The catalog is *derived* from the two authoritative sources (dbt meta + the +non-dbt registry). It is regenerated, never hand-edited. CI checks that the +committed catalog matches a fresh generation so the two never drift. + +Usage: + python -m atlas.governance.catalog generate + python -m atlas.governance.catalog check +""" + +from __future__ import annotations + +import argparse +import json +import sys +from typing import Any + +from atlas.governance.registry import ( + atlas_root, + build_asset_index, + load_policy, + validate_governance, +) + +GENERATED_JSON = "governance/generated/catalog.json" +GENERATED_MD = "governance/generated/catalog.md" + + +def build_catalog() -> dict[str, Any]: + policy = load_policy() + index = build_asset_index() + assets = [] + for asset_id in sorted(index): + record = {k: v for k, v in index[asset_id].items() if not k.startswith("_")} + record["asset_id"] = asset_id + record["origin"] = index[asset_id].get("_origin", "unknown") + assets.append(record) + by_type: dict[str, int] = {} + by_class: dict[str, int] = {} + for asset in assets: + by_type[asset.get("asset_type", "?")] = by_type.get(asset.get("asset_type", "?"), 0) + 1 + by_class[asset.get("classification", "?")] = by_class.get(asset.get("classification", "?"), 0) + 1 + return { + "generator": "atlas.governance.catalog", + "policy_version": policy.get("version"), + "asset_count": len(assets), + "counts_by_type": dict(sorted(by_type.items())), + "counts_by_classification": dict(sorted(by_class.items())), + "assets": assets, + } + + +def _render_markdown(catalog: dict[str, Any]) -> str: + lines = [ + "# Atlas Generated Asset Catalog", + "", + "> Generated by `python -m atlas.governance.catalog generate`. Do not edit by hand.", + "", + f"Total assets: **{catalog['asset_count']}**", + "", + "| asset_id | type | owner | classification | retention | lifecycle | contract | origin |", + "| --- | --- | --- | --- | --- | --- | --- | --- |", + ] + for a in catalog["assets"]: + lines.append( + f"| `{a['asset_id']}` | {a.get('asset_type', '')} | {a.get('technical_owner', '')} " + f"| {a.get('classification', '')} | {a.get('retention_class', '')} " + f"| {a.get('lifecycle_status', '')} | {a.get('contract_version', '')} " + f"| {a.get('origin', '')} |" + ) + lines.append("") + return "\n".join(lines) + + +def write_catalog() -> dict[str, Any]: + catalog = build_catalog() + root = atlas_root() + json_path = root / GENERATED_JSON + md_path = root / GENERATED_MD + json_path.parent.mkdir(parents=True, exist_ok=True) + json_path.write_text(json.dumps(catalog, indent=2, sort_keys=True) + "\n", encoding="utf-8") + md_path.write_text(_render_markdown(catalog), encoding="utf-8") + return catalog + + +def _committed_matches() -> bool: + root = atlas_root() + json_path = root / GENERATED_JSON + if not json_path.exists(): + return False + fresh = json.dumps(build_catalog(), indent=2, sort_keys=True) + "\n" + return json_path.read_text(encoding="utf-8") == fresh + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Atlas governance catalog") + parser.add_argument("command", choices=["generate", "check"]) + args = parser.parse_args(argv) + + errors = validate_governance() + if errors: + print("GOVERNANCE INVALID:", file=sys.stderr) + for e in errors: + print(f" - {e}", file=sys.stderr) + return 1 + + if args.command == "generate": + catalog = write_catalog() + print(f"catalog generated: {catalog['asset_count']} assets -> {GENERATED_JSON}") + return 0 + + # check + if not _committed_matches(): + print( + "GOVERNANCE DRIFT: committed catalog is stale; run `python -m atlas.governance.catalog generate`", + file=sys.stderr, + ) + return 1 + print("governance catalog matches sources") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/governance/impact.py b/src/atlas/governance/impact.py new file mode 100644 index 0000000..349540c --- /dev/null +++ b/src/atlas/governance/impact.py @@ -0,0 +1,150 @@ +"""Consumer-impact analysis for a proposed change (Sprint 7, Phase 5). + +Given a governed asset (and optionally a change record), reports the direct and +transitive downstream assets, affected tests, affected contracts, consumers and +owners, and migrations/runbooks involved. Uses only repository artifacts. + +Usage:: + + python -m atlas.governance.impact --asset fct_events \\ + --change governance/changes/CHG-....yml --output-dir /tmp/impact +""" + +from __future__ import annotations + +import argparse +import json +import re +from pathlib import Path +from typing import Any + +import yaml + +from atlas.config.settings import atlas_root +from atlas.governance.lineage import build_lineage +from atlas.governance.registry import build_asset_index, load_consumers + + +def _tests_dir() -> Path: + return atlas_root() / "dbt" / "atlas_dbt" / "tests" + + +def _models_dir() -> Path: + return atlas_root() / "dbt" / "atlas_dbt" / "models" + + +def _affected_tests(assets: set[str]) -> list[str]: + """dbt test/property files that reference any affected asset by name.""" + hits: set[str] = set() + names = {a.split(".")[-1] for a in assets} + search_roots = [_tests_dir(), _models_dir()] + for root in search_roots: + if not root.exists(): + continue + for path in list(root.glob("**/*.sql")) + list(root.glob("**/*.yml")): + text = path.read_text(encoding="utf-8") + for name in names: + if re.search(rf"\b{re.escape(name)}\b", text): + hits.add(str(path.relative_to(atlas_root()))) + break + return sorted(hits) + + +def analyze(asset_id: str, change_file: Path | None = None) -> dict[str, Any]: + graph = build_lineage() + index = build_asset_index() + consumers = load_consumers() + + key = asset_id if asset_id in graph.node_types else asset_id.split(".")[-1] + direct = sorted(graph.downstream.get(key, set())) + transitive = sorted(graph.transitive_downstream(key)) + upstream = sorted(graph.transitive_upstream(key)) + + affected_assets = set(transitive) | {key} + # Consumers among the downstream set + any consumer registry entry that reads + # the asset directly. + affected_consumers = {n for n in transitive if n in consumers} + for consumer, spec in consumers.items(): + reads = set(spec.get("reads", []) or []) + if asset_id in reads or key in {r.split(".")[-1] for r in reads}: + affected_consumers.add(consumer) + + owners = sorted( + { + index[a].get("technical_owner", "?") + for a in affected_assets + if a in index and index[a].get("technical_owner") + } + ) + contracts = { + a: index[a].get("contract_version") + for a in sorted(affected_assets) + if a in index and index[a].get("contract_version") + } + runbooks = sorted( + {index[a]["runbook"] for a in affected_assets if a in index and index[a].get("runbook")} + ) + + report: dict[str, Any] = { + "asset": asset_id, + "resolved_node": key, + "upstream": upstream, + "direct_downstream": direct, + "transitive_downstream": transitive, + "affected_tests": _affected_tests(affected_assets), + "affected_contracts": contracts, + "affected_consumers": sorted(affected_consumers), + "owners_to_notify": owners, + "runbooks": runbooks, + } + + if change_file and change_file.exists(): + change = yaml.safe_load(change_file.read_text(encoding="utf-8")) or {} + report["change"] = { + "change_id": change.get("change_id"), + "compatibility_class": change.get("compatibility_class"), + "new_contract_version": change.get("new_contract_version"), + "approval_reference": bool(str(change.get("approval_reference", "")).strip()), + } + + return report + + +def render_summary(report: dict[str, Any]) -> str: + lines = [ + f"# Impact report: {report['asset']}", + "", + f"- Resolved node: `{report['resolved_node']}`", + f"- Direct downstream ({len(report['direct_downstream'])}): " + + ", ".join(f"`{d}`" for d in report["direct_downstream"]) + or "- Direct downstream: none", + f"- Transitive downstream ({len(report['transitive_downstream'])}): " + + ", ".join(f"`{d}`" for d in report["transitive_downstream"]), + f"- Affected consumers: {', '.join(report['affected_consumers']) or 'none'}", + f"- Owners to notify: {', '.join(report['owners_to_notify']) or 'none'}", + f"- Affected contracts: {report['affected_contracts']}", + f"- Affected tests/props: {len(report['affected_tests'])} file(s)", + f"- Runbooks: {', '.join(report['runbooks']) or 'none'}", + ] + return "\n".join(lines) + "\n" + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Atlas consumer-impact analysis") + parser.add_argument("--asset", required=True) + parser.add_argument("--change", type=Path) + parser.add_argument("--output-dir", type=Path) + args = parser.parse_args(argv) + + report = analyze(args.asset, args.change) + summary = render_summary(report) + if args.output_dir: + args.output_dir.mkdir(parents=True, exist_ok=True) + (args.output_dir / "impact.json").write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + (args.output_dir / "impact.md").write_text(summary) + print(summary) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/governance/lineage.py b/src/atlas/governance/lineage.py new file mode 100644 index 0000000..7de7769 --- /dev/null +++ b/src/atlas/governance/lineage.py @@ -0,0 +1,145 @@ +"""Repository-artifact lineage for Atlas (Sprint 7, Phase 5). + +Builds a source-to-mart lineage graph WITHOUT a graph database or metadata +service — it parses the dbt model SQL for ``ref()``/``source()`` edges, the +source registry, and the governance consumer registry. Offline and +credentialless so it runs in CI. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from functools import lru_cache +from pathlib import Path + +import yaml + +from atlas.config.settings import atlas_root +from atlas.governance.registry import load_consumers + +_REF_RE = re.compile(r"\bref\(\s*['\"]([^'\"]+)['\"]\s*\)") +_SOURCE_RE = re.compile(r"\bsource\(\s*['\"]([^'\"]+)['\"]\s*,\s*['\"]([^'\"]+)['\"]\s*\)") + + +@dataclass +class LineageGraph: + upstream: dict[str, set[str]] = field(default_factory=dict) + downstream: dict[str, set[str]] = field(default_factory=dict) + node_types: dict[str, str] = field(default_factory=dict) + + def add_edge(self, src: str, dst: str) -> None: + self.upstream.setdefault(dst, set()).add(src) + self.downstream.setdefault(src, set()).add(dst) + self.upstream.setdefault(src, set()) + self.downstream.setdefault(dst, set()) + + def add_node(self, node: str, node_type: str) -> None: + self.node_types.setdefault(node, node_type) + self.upstream.setdefault(node, set()) + self.downstream.setdefault(node, set()) + + def transitive_downstream(self, node: str) -> set[str]: + seen: set[str] = set() + stack = list(self.downstream.get(node, set())) + while stack: + n = stack.pop() + if n in seen: + continue + seen.add(n) + stack.extend(self.downstream.get(n, set())) + return seen + + def transitive_upstream(self, node: str) -> set[str]: + seen: set[str] = set() + stack = list(self.upstream.get(node, set())) + while stack: + n = stack.pop() + if n in seen: + continue + seen.add(n) + stack.extend(self.upstream.get(n, set())) + return seen + + +def _models_dir() -> Path: + return atlas_root() / "dbt" / "atlas_dbt" / "models" + + +@lru_cache(maxsize=1) +def build_lineage() -> LineageGraph: + graph = LineageGraph() + + # dbt sources -> nodes (e.g. source('atlas_raw','events') == atlas_raw.events). + sources_yml = _models_dir() / "sources" / "sources.yml" + if sources_yml.exists(): + data = yaml.safe_load(sources_yml.read_text(encoding="utf-8")) or {} + for src in data.get("sources", []) or []: + for tbl in src.get("tables", []) or []: + node = f"{src['name']}.{tbl['name']}" + graph.add_node(node, "source") + + # dbt models: parse ref()/source() edges from the SQL. + for sql in sorted(_models_dir().glob("*/*.sql")): + model = sql.stem + layer = sql.parent.name + graph.add_node(model, f"{layer}_model") + text = sql.read_text(encoding="utf-8") + for upstream in _REF_RE.findall(text): + graph.add_node(upstream, "model_or_seed") + graph.add_edge(upstream, model) + for src_name, tbl in _SOURCE_RE.findall(text): + node = f"{src_name}.{tbl}" + graph.add_node(node, "source") + graph.add_edge(node, model) + + # Governance consumers -> downstream consumer nodes reading governed assets. + for consumer, spec in load_consumers().items(): + graph.add_node(consumer, f"consumer:{spec.get('type', 'unknown')}") + for asset in spec.get("reads", []) or []: + # Consumers may read by dbt name or fully-qualified id; normalize the + # trailing name so 'core.fct_events' links to model 'fct_events'. + candidates = {asset, asset.split(".")[-1]} + linked = candidates & set(graph.node_types) + targets = linked or {asset} + for target in targets: + graph.add_node(target, graph.node_types.get(target, "asset")) + graph.add_edge(target, consumer) + + return graph + + +def to_dict(graph: LineageGraph) -> dict[str, object]: + return { + "nodes": [ + { + "id": node, + "type": graph.node_types.get(node, "unknown"), + "upstream": sorted(graph.upstream.get(node, set())), + "downstream": sorted(graph.downstream.get(node, set())), + } + for node in sorted(graph.node_types) + ], + "edge_count": sum(len(v) for v in graph.downstream.values()), + } + + +def main(argv: list[str] | None = None) -> int: + import argparse + import json + + parser = argparse.ArgumentParser(description="Atlas repository lineage") + parser.add_argument("--output", type=Path, help="write machine-readable lineage JSON") + args = parser.parse_args(argv) + + graph = build_lineage() + payload = to_dict(graph) + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n") + print(json.dumps(payload, indent=2, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/governance/registry.py b/src/atlas/governance/registry.py new file mode 100644 index 0000000..211c699 --- /dev/null +++ b/src/atlas/governance/registry.py @@ -0,0 +1,320 @@ +"""Load and validate Atlas governance metadata (Sprint 7, ADR-016). + +Two authoritative sources are merged into one asset index: + +1. dbt models -> ``meta.governance`` blocks in ``dbt/atlas_dbt/models/**/*.yml`` +2. non-dbt assets -> ``governance/non_dbt_assets.yml`` + +``validate_governance`` returns a list of human-readable error strings; an empty +list means the governance metadata satisfies ``governance/policy.yml``. This is +pure/offline so the governance CI gate needs no credentials. +""" + +from __future__ import annotations + +import re +from pathlib import Path +from typing import Any + +import yaml + +from atlas.config.settings import atlas_root + + +class GovernanceError(ValueError): + """Raised when governance metadata cannot be loaded.""" + + +def governance_dir() -> Path: + return atlas_root() / "governance" + + +def _load_yaml(path: Path) -> dict[str, Any]: + if not path.exists(): + raise GovernanceError(f"missing governance file: {path}") + data = yaml.safe_load(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise GovernanceError(f"governance file is not a mapping: {path}") + return data + + +def load_policy() -> dict[str, Any]: + return _load_yaml(governance_dir() / "policy.yml") + + +def load_non_dbt_assets() -> list[dict[str, Any]]: + data = _load_yaml(governance_dir() / "non_dbt_assets.yml") + assets = data.get("assets", []) + if not isinstance(assets, list): + raise GovernanceError("non_dbt_assets.yml: 'assets' must be a list") + return assets + + +def load_classifications() -> dict[str, Any]: + return _load_yaml(governance_dir() / "classifications.yml") + + +def load_retention() -> dict[str, Any]: + return _load_yaml(governance_dir() / "retention.yml") + + +def load_consumers() -> dict[str, Any]: + data = _load_yaml(governance_dir() / "consumers.yml") + return data.get("consumers", {}) or {} + + +def _dbt_models_dir() -> Path: + return atlas_root() / "dbt" / "atlas_dbt" / "models" + + +def load_dbt_model_governance() -> list[dict[str, Any]]: + """Extract governance metadata from dbt model ``meta.governance`` blocks. + + Parses the model property YAML files directly (no dbt runtime needed), so + this works in credentialless CI. Each returned record is normalized into the + same shape as a non-dbt asset, with ``asset_id`` = the dbt model name and + ``asset_type`` inferred from the model's directory. + """ + dir_to_type = { + "staging": "staging_model", + "intermediate": "intermediate_model", + "core": "core_model", # refined below by name + "marts": "mart_model", + } + records: list[dict[str, Any]] = [] + for yml in sorted(_dbt_models_dir().glob("*/*.yml")): + layer = yml.parent.name + data = yaml.safe_load(yml.read_text(encoding="utf-8")) or {} + for model in data.get("models", []) or []: + name = model.get("name") + meta = (model.get("meta") or {}).get("governance") + if not name or meta is None: + # Models without governance meta are reported by validation. + records.append( + { + "asset_id": name or f"", + "asset_type": dir_to_type.get(layer, "staging_model"), + "repository_path": str(yml.relative_to(atlas_root())), + "_missing_meta": True, + "source": "dbt", + } + ) + continue + asset_type = dir_to_type.get(layer, "staging_model") + if layer == "core": + asset_type = "fact_model" if name.startswith("fct_") else "dimension_model" + record = dict(meta) + record["asset_id"] = name + record["asset_type"] = asset_type + # dbt `description` is the authoritative purpose text; do not + # duplicate it inside meta.governance. + description = (model.get("description") or "").strip() + if description and not record.get("purpose"): + record["purpose"] = description + record.setdefault("repository_path", str(yml.relative_to(atlas_root()))) + record.setdefault("source", str(yml.relative_to(atlas_root()))) + record["_origin"] = "dbt_meta" + records.append(record) + return records + + +def build_asset_index() -> dict[str, dict[str, Any]]: + """Merge dbt and non-dbt governance records keyed by asset_id.""" + index: dict[str, dict[str, Any]] = {} + for record in load_dbt_model_governance(): + index[record["asset_id"]] = record + for asset in load_non_dbt_assets(): + record = dict(asset) + record["_origin"] = "registry" + index[record["asset_id"]] = record + return index + + +def _is_permanent(retention: dict[str, Any], retention_class: str) -> bool: + cls = (retention.get("classes", {}) or {}).get(retention_class, {}) + return bool(cls.get("is_permanent_evidence")) + + +def _change_record_ids() -> set[str]: + """Change_ids declared under governance/changes/ (excluding the template).""" + ids: set[str] = set() + changes_dir = governance_dir() / "changes" + if not changes_dir.exists(): + return ids + for path in changes_dir.glob("*.yml"): + if path.name == "TEMPLATE.yml": + continue + data = yaml.safe_load(path.read_text(encoding="utf-8")) or {} + if data.get("change_id"): + ids.add(str(data["change_id"])) + return ids + + +def _days_between(start: str, end: str) -> int | None: + from datetime import date + + try: + s = date.fromisoformat(str(start)) + e = date.fromisoformat(str(end)) + except (ValueError, TypeError): + return None + return (e - s).days + + +def deprecation_errors( + asset_id: str, + record: dict[str, Any], + consumers: dict[str, Any], + change_ids: set[str], + policy: dict[str, Any], +) -> list[str]: + """Validate the deprecation lifecycle for a single asset (pure function).""" + status = record.get("lifecycle_status") + if status in (None, "ACTIVE"): + return [] + errors: list[str] = [] + dep_policy = policy.get("deprecation", {}) + min_window = int(dep_policy.get("minimum_window_days", 30)) + dep = record.get("deprecation") or {} + + if not dep: + return [f"{asset_id}: lifecycle '{status}' requires a 'deprecation' block"] + + # 1. Replacement required. + if dep_policy.get("require_replacement", True) and not str(dep.get("replacement", "")).strip(): + errors.append(f"{asset_id}: deprecated/removed asset must declare a 'replacement'") + + # 2. Lifecycle change must cite a change record that exists. + change_ref = str(dep.get("change_record", "")).strip() + if dep_policy.get("require_change_record", True): + if not change_ref: + errors.append(f"{asset_id}: lifecycle change requires a 'change_record' reference") + elif change_ref not in change_ids: + errors.append(f"{asset_id}: change_record '{change_ref}' not found under governance/changes/") + + # 3. Removal date must respect the minimum window. + start = dep.get("deprecation_start") + removal = dep.get("earliest_removal_date") + if start and removal: + gap = _days_between(start, removal) + if gap is None: + errors.append(f"{asset_id}: invalid deprecation dates") + elif gap < min_window: + errors.append( + f"{asset_id}: earliest_removal_date is {gap}d after start (minimum window is {min_window}d)" + ) + elif status in ("REMOVAL_SCHEDULED", "REMOVED"): + errors.append(f"{asset_id}: {status} requires deprecation_start and earliest_removal_date") + + # 4. Removal approval required for REMOVAL_SCHEDULED / REMOVED. + if status in ("REMOVAL_SCHEDULED", "REMOVED") and not str(dep.get("removal_approval", "")).strip(): + errors.append(f"{asset_id}: {status} requires a 'removal_approval'") + + # 5. A REMOVED / REMOVAL_SCHEDULED asset must have no active consumers. + if status in ("REMOVAL_SCHEDULED", "REMOVED"): + short = asset_id.split(".")[-1] + active_readers = [] + for consumer, spec in consumers.items(): + reads = set(spec.get("reads", []) or []) + if asset_id in reads or short in {r.split(".")[-1] for r in reads}: + active_readers.append(consumer) + if active_readers: + errors.append( + f"{asset_id}: {status} but still has active consumers " + f"{sorted(active_readers)} — migrate them first" + ) + + return errors + + +def validate_governance() -> list[str]: # noqa: C901 - explicit sequential checks + """Return a list of governance policy violations (empty == valid).""" + errors: list[str] = [] + try: + policy = load_policy() + classifications = load_classifications() + retention = load_retention() + consumers = load_consumers() + dbt_records = load_dbt_model_governance() + non_dbt = load_non_dbt_assets() + except GovernanceError as exc: + return [str(exc)] + + required = set(policy["required_fields"]) + valid_types = set(policy["asset_types"]) + valid_lifecycle = set(policy["lifecycle_statuses"]) + valid_class = set(policy["classifications"]) + valid_retention = set(policy["retention_classes"]) + owner_pattern = re.compile(policy["owner_rules"]["allowed_owner_pattern"]) + disallow_email = policy["owner_rules"]["disallow_email_addresses"] + known_consumers = set(consumers) + # dbt-core layer types map onto the policy's model-type vocabulary. + type_alias = { + "staging_model": "staging_model", + "intermediate_model": "intermediate_model", + "dimension_model": "dimension_model", + "fact_model": "fact_model", + "mart_model": "mart_model", + } + + # 1. Single source of truth: a dbt model id must not also be a registry id. + dbt_ids = {r["asset_id"] for r in dbt_records} + registry_ids = {a.get("asset_id") for a in non_dbt} + overlap = dbt_ids & registry_ids + for dup in sorted(overlap): + errors.append(f"duplicate source of truth: '{dup}' defined in both dbt meta and registry") + + # 2. Per-asset validation across the merged index. + index = build_asset_index() + for asset_id, record in sorted(index.items()): + if record.get("_missing_meta"): + errors.append(f"{asset_id}: dbt model missing meta.governance block") + continue + for field in required: + value = record.get(field) + if value is None or (isinstance(value, str) and not value.strip()): + errors.append(f"{asset_id}: missing required field '{field}'") + atype = str(record.get("asset_type", "")) + if atype not in valid_types and type_alias.get(atype) not in valid_types: + errors.append(f"{asset_id}: invalid asset_type '{atype}'") + if record.get("classification") not in valid_class: + errors.append(f"{asset_id}: invalid classification '{record.get('classification')}'") + if record.get("lifecycle_status") not in valid_lifecycle: + errors.append(f"{asset_id}: invalid lifecycle_status '{record.get('lifecycle_status')}'") + rclass = record.get("retention_class") + if rclass not in valid_retention: + errors.append(f"{asset_id}: invalid retention_class '{rclass}'") + owner = record.get("technical_owner") + if isinstance(owner, str) and owner: + if disallow_email and "@" in owner: + errors.append(f"{asset_id}: technical_owner must be a role id, not an email") + elif not owner_pattern.match(owner): + errors.append(f"{asset_id}: technical_owner '{owner}' violates owner pattern") + for consumer in record.get("consumers", []) or []: + # Consumers may reference other governed assets or the consumer + # registry; unknown free-text consumers are allowed only if they + # look like an asset id (contain a '.') — otherwise they must be + # registered. + if consumer in known_consumers or "." in consumer or consumer in index: + continue + errors.append(f"{asset_id}: unknown consumer '{consumer}' (not in consumers.yml)") + + # 2b. Deprecation lifecycle validation. + change_ids = _change_record_ids() + for asset_id, record in sorted(index.items()): + if record.get("_missing_meta"): + continue + errors.extend(deprecation_errors(asset_id, record, consumers, change_ids, policy)) + + # 3. Retention permanence invariant. + for cls_name, cls in (retention.get("classes", {}) or {}).items(): + if cls.get("is_permanent_evidence") and cls.get("expiration_days") is not None: + errors.append(f"retention class '{cls_name}': permanent evidence cannot have an expiration") + + # 4. Classification completeness: no RESTRICTED asset if policy asserts none. + if classifications.get("no_restricted_assets_present"): + for asset_id, record in index.items(): + if record.get("classification") == "RESTRICTED": + errors.append(f"{asset_id}: RESTRICTED but classifications.yml asserts none present") + + return errors diff --git a/src/atlas/governance/retention.py b/src/atlas/governance/retention.py new file mode 100644 index 0000000..6f5711b --- /dev/null +++ b/src/atlas/governance/retention.py @@ -0,0 +1,97 @@ +"""Classification & retention validation and disposal planning (Sprint 7, P9). + +Offline validation of ``governance/retention.yml`` plus a dry-run planner that +maps assets to their desired expiration. Live lifecycle/expiration changes +require ``ATLAS_APPROVE_RETENTION_MUTATION=true`` and are applied separately — +this module never mutates cloud resources. +""" + +from __future__ import annotations + +from typing import Any + +from atlas.governance.registry import ( + build_asset_index, + load_policy, + load_retention, +) + +# Non-permanent classes that legitimately retain data indefinitely because it is +# deterministically rebuildable (not disposable evidence). +_REBUILDABLE = {"canonical_warehouse", "raw_landing"} +# Transient classes that MUST declare a disposal (expiration). +_TRANSIENT = {"observability_logs", "temporary_integration", "test_fixture"} + + +def validate_retention_config() -> list[str]: + """Return retention-policy violations (empty == valid).""" + errors: list[str] = [] + retention = load_retention() + policy = load_policy() + classes = retention.get("classes", {}) or {} + + # 1. policy.retention_classes must match the retention.yml class keys exactly + # (single source of truth — no drift between the two files). + policy_classes = set(policy.get("retention_classes", []) or []) + yaml_classes = set(classes) + if policy_classes != yaml_classes: + missing = policy_classes - yaml_classes + extra = yaml_classes - policy_classes + if missing: + errors.append(f"retention classes in policy.yml but not retention.yml: {sorted(missing)}") + if extra: + errors.append(f"retention classes in retention.yml but not policy.yml: {sorted(extra)}") + + # 2. Per-class consistency. + for name, cls in classes.items(): + for req in ("description", "retention", "expiration_days", "is_permanent_evidence"): + if req not in cls: + errors.append(f"retention class '{name}': missing '{req}'") + permanent = bool(cls.get("is_permanent_evidence")) + expiration = cls.get("expiration_days") + # Conflict: permanent evidence cannot expire. + if permanent and expiration is not None: + errors.append(f"retention class '{name}': permanent evidence cannot have an expiration") + # Conflict: transient class must declare a disposal window. + if name in _TRANSIENT and (expiration is None): + errors.append(f"retention class '{name}': transient class must set expiration_days") + # Conflict: a non-permanent, non-rebuildable, non-transient class with no + # disposal is ambiguous. + if not permanent and name not in _REBUILDABLE and name not in _TRANSIENT and expiration is None: + errors.append(f"retention class '{name}': ambiguous — declare expiration or permanence") + + return errors + + +def desired_expiration_days(retention_class: str) -> int | None: + classes = load_retention().get("classes", {}) or {} + return classes.get(retention_class, {}).get("expiration_days") + + +def plan_expirations() -> list[dict[str, Any]]: + """Dry-run: map each governed asset to its desired expiration disposition. + + Returns records without touching any cloud resource. Permanent audit and + release evidence are explicitly flagged as ``keep_forever`` so a live applier + can assert it never expires them. + """ + retention = load_retention().get("classes", {}) or {} + plan: list[dict[str, Any]] = [] + for asset_id, record in sorted(build_asset_index().items()): + if record.get("_missing_meta"): + continue + rclass = record.get("retention_class") + cls = retention.get(rclass, {}) + exp = cls.get("expiration_days") + permanent = bool(cls.get("is_permanent_evidence")) + plan.append( + { + "asset_id": asset_id, + "asset_type": record.get("asset_type"), + "retention_class": rclass, + "expiration_days": exp, + "disposition": "keep_forever" if permanent or exp is None else f"expire_{exp}d", + "is_permanent_evidence": permanent, + } + ) + return plan diff --git a/src/atlas/governance/schema_check.py b/src/atlas/governance/schema_check.py new file mode 100644 index 0000000..5e887f3 --- /dev/null +++ b/src/atlas/governance/schema_check.py @@ -0,0 +1,348 @@ +"""Schema compatibility checker (Sprint 7, ADR-017). + +Compares two schema manifests and classifies every difference into one of four +compatibility classes: + + COMPATIBLE additive / widening / metadata-only + CONDITIONALLY_COMPATIBLE requires consumer migration or approved evidence + BREAKING removes/renames/tightens; changes grain/partition/id + PROHIBITED unversioned replacement, contract downgrade + +A manifest is a JSON document:: + + { + "version": 1, + "assets": { + "": { + "contract_version": "1.0", + "grain": "one row per event_id", + "partition_field": "event_date", # optional + "event_identity": ["event_id"], # optional + "fields": { + "": { + "type": "string", + "nullable": true, + "accepted_values": ["a", "b"] # optional + } + } + } + } + } + +Usage:: + + python -m atlas.governance.schema_check --baseline base.json \\ + --candidate cand.json --output report.json + python -m atlas.governance.schema_check --generate manifest.json +""" + +from __future__ import annotations + +import argparse +import json +import sys +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any + +import yaml + +from atlas.config.settings import atlas_root + +COMPATIBLE = "COMPATIBLE" +CONDITIONALLY_COMPATIBLE = "CONDITIONALLY_COMPATIBLE" +BREAKING = "BREAKING" +PROHIBITED = "PROHIBITED" + +# Severity ordering (higher == worse) used to pick the overall class. +_SEVERITY = { + COMPATIBLE: 0, + CONDITIONALLY_COMPATIBLE: 1, + BREAKING: 2, + PROHIBITED: 3, +} + + +@dataclass +class Change: + asset_id: str + change_type: str + detail: str + compatibility_class: str + + +@dataclass +class CompatibilityReport: + overall_class: str = COMPATIBLE + changes: list[Change] = field(default_factory=list) + + def add(self, change: Change) -> None: + self.changes.append(change) + if _SEVERITY[change.compatibility_class] > _SEVERITY[self.overall_class]: + self.overall_class = change.compatibility_class + + def to_dict(self) -> dict[str, Any]: + return { + "overall_class": self.overall_class, + "change_count": len(self.changes), + "changes": [asdict(c) for c in self.changes], + } + + +def _version_tuple(v: str) -> tuple[int, int]: + try: + major, minor = str(v).split(".")[:2] + return int(major), int(minor) + except (ValueError, AttributeError): + return (0, 0) + + +def _compare_asset( + asset_id: str, base: dict[str, Any], cand: dict[str, Any], report: CompatibilityReport +) -> None: + base_fields = base.get("fields", {}) or {} + cand_fields = cand.get("fields", {}) or {} + schema_changed = False + + # Grain / partition / identity — structural, breaking when changed. + for key, ctype in ( + ("grain", "grain_changed"), + ("partition_field", "partition_field_changed"), + ("event_identity", "event_identity_changed"), + ): + if base.get(key) is not None and base.get(key) != cand.get(key): + schema_changed = True + report.add( + Change( + asset_id, + ctype, + f"{key}: {base.get(key)!r} -> {cand.get(key)!r}", + BREAKING, + ) + ) + + # Removed fields -> BREAKING. + for name in base_fields: + if name not in cand_fields: + schema_changed = True + report.add(Change(asset_id, "field_removed", f"field '{name}' removed", BREAKING)) + + # Added fields -> COMPATIBLE if nullable else CONDITIONALLY_COMPATIBLE. + for name, spec in cand_fields.items(): + if name not in base_fields: + schema_changed = True + nullable = spec.get("nullable", True) + cls = COMPATIBLE if nullable else CONDITIONALLY_COMPATIBLE + report.add( + Change( + asset_id, + "field_added", + f"field '{name}' added (nullable={nullable})", + cls, + ) + ) + + # Changed fields. + for name in base_fields.keys() & cand_fields.keys(): + b = base_fields[name] + c = cand_fields[name] + if b.get("type") and c.get("type") and b["type"] != c["type"]: + schema_changed = True + report.add( + Change( + asset_id, + "type_changed", + f"field '{name}' type {b['type']} -> {c['type']}", + BREAKING, + ) + ) + b_nullable = b.get("nullable", True) + c_nullable = c.get("nullable", True) + if b_nullable and not c_nullable: + schema_changed = True + report.add( + Change( + asset_id, + "nullability_tightened", + f"field '{name}' nullable -> required", + BREAKING, + ) + ) + elif not b_nullable and c_nullable: + schema_changed = True + report.add( + Change( + asset_id, + "nullability_loosened", + f"field '{name}' required -> nullable", + COMPATIBLE, + ) + ) + b_vals = b.get("accepted_values") + c_vals = c.get("accepted_values") + if b_vals is not None and c_vals is not None and set(b_vals) != set(c_vals): + schema_changed = True + removed = set(b_vals) - set(c_vals) + if removed: + report.add( + Change( + asset_id, + "accepted_values_narrowed", + f"field '{name}' drops values {sorted(removed)}", + BREAKING, + ) + ) + else: + report.add( + Change( + asset_id, + "accepted_values_widened", + f"field '{name}' widens values {sorted(set(c_vals) - set(b_vals))}", + COMPATIBLE, + ) + ) + + # Contract versioning: any schema change with an unchanged or decreased + # contract version is a PROHIBITED unversioned replacement. + b_ver = base.get("contract_version", "0.0") + c_ver = cand.get("contract_version", "0.0") + if schema_changed: + if _version_tuple(c_ver) < _version_tuple(b_ver): + report.add( + Change( + asset_id, + "contract_version_downgraded", + f"contract_version {b_ver} -> {c_ver}", + PROHIBITED, + ) + ) + elif _version_tuple(c_ver) == _version_tuple(b_ver): + report.add( + Change( + asset_id, + "unversioned_change", + f"schema changed but contract_version stayed {b_ver}", + PROHIBITED, + ) + ) + + +def compare_manifests(baseline: dict[str, Any], candidate: dict[str, Any]) -> CompatibilityReport: + report = CompatibilityReport() + base_assets = baseline.get("assets", {}) or {} + cand_assets = candidate.get("assets", {}) or {} + for asset_id in base_assets: + if asset_id not in cand_assets: + report.add(Change(asset_id, "asset_removed", "asset removed from manifest", BREAKING)) + continue + _compare_asset(asset_id, base_assets[asset_id], cand_assets[asset_id], report) + # Newly added assets are always compatible. + for asset_id in cand_assets: + if asset_id not in base_assets: + report.add(Change(asset_id, "asset_added", "new asset added", COMPATIBLE)) + return report + + +# --------------------------------------------------------------------------- +# Manifest generation from repository sources (dbt YAML + governance meta). +# --------------------------------------------------------------------------- + +# Structural properties the dbt SQL config expresses that are not in the YAML +# column list; kept as a small explicit map so the manifest captures grain- +# critical attributes without parsing Jinja. +_STRUCTURAL: dict[str, dict[str, Any]] = { + "fct_events": {"partition_field": "event_date", "event_identity": ["event_id"]}, +} + + +def _dbt_models_dir() -> Path: + return atlas_root() / "dbt" / "atlas_dbt" / "models" + + +def generate_manifest() -> dict[str, Any]: + assets: dict[str, Any] = {} + for yml in sorted(_dbt_models_dir().glob("*/*.yml")): + data = yaml.safe_load(yml.read_text(encoding="utf-8")) or {} + for model in data.get("models", []) or []: + name = model.get("name") + if not name: + continue + gov = (model.get("meta") or {}).get("governance", {}) or {} + fields: dict[str, Any] = {} + for col in model.get("columns", []) or []: + col_name = col.get("name") + if not col_name: + continue + tests = col.get("tests", []) or [] + nullable = not any(_is_not_null(t) for t in tests) + accepted = _extract_accepted_values(tests) + spec: dict[str, Any] = { + "type": col.get("data_type", "unknown"), + "nullable": nullable, + } + if accepted is not None: + spec["accepted_values"] = accepted + fields[col_name] = spec + entry: dict[str, Any] = { + "contract_version": gov.get("contract_version", "0.0"), + "grain": gov.get("grain", ""), + "fields": fields, + } + entry.update(_STRUCTURAL.get(name, {})) + assets[name] = entry + return {"version": 1, "assets": assets} + + +def _is_not_null(test: Any) -> bool: + return test == "not_null" + + +def _extract_accepted_values(tests: list[Any]) -> list[Any] | None: + for t in tests: + if isinstance(t, dict) and "accepted_values" in t: + args = t["accepted_values"].get("arguments", t["accepted_values"]) + return args.get("values") if isinstance(args, dict) else None + return None + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Atlas schema compatibility checker") + parser.add_argument("--baseline", type=Path) + parser.add_argument("--candidate", type=Path) + parser.add_argument("--output", type=Path) + parser.add_argument("--generate", type=Path, help="write a manifest from repo sources") + parser.add_argument( + "--fail-on", + default="BREAKING", + choices=[COMPATIBLE, CONDITIONALLY_COMPATIBLE, BREAKING, PROHIBITED], + help="exit non-zero when overall class is at/above this severity", + ) + args = parser.parse_args(argv) + + if args.generate: + manifest = generate_manifest() + args.generate.parent.mkdir(parents=True, exist_ok=True) + args.generate.write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n") + print(f"manifest written: {len(manifest['assets'])} assets -> {args.generate}") + return 0 + + if not args.baseline or not args.candidate: + parser.error("--baseline and --candidate are required unless --generate is used") + + baseline = json.loads(args.baseline.read_text()) + candidate = json.loads(args.candidate.read_text()) + report = compare_manifests(baseline, candidate) + payload = report.to_dict() + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(payload, indent=2) + "\n") + print(json.dumps(payload, indent=2)) + + if _SEVERITY[report.overall_class] >= _SEVERITY[args.fail_on]: + print(f"schema check: {report.overall_class} (>= {args.fail_on})", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/governance/security_policy.py b/src/atlas/governance/security_policy.py new file mode 100644 index 0000000..4f39287 --- /dev/null +++ b/src/atlas/governance/security_policy.py @@ -0,0 +1,115 @@ +"""Security-policy scanners for governed Atlas artifacts (Sprint 7, Phase 8). + +Two offline scanners used by ``gate_security_policy``: + +1. ``scan_managed_iam`` — managed IAM/bootstrap scripts must never grant + prohibited roles to Atlas principals or create service-account keys. +2. ``scan_data_exposure`` — governed Atlas artifacts (config, governance, + observability, sprint docs) must not commit literal secrets, recipient + addresses, webhook URLs, or verification codes. Variable references + (``$TOKEN``, ``${NOTIFICATION_CHANNEL}``) are allowed. + +Both return a list of ``(path, lineno, reason)`` findings; empty == clean. +Out-of-scope trees (artifact-platform, examples, infra/artifact-platform) are +excluded — they are reviewed at extraction time (Sprint 8). +""" + +from __future__ import annotations + +import re +from pathlib import Path + +from atlas.config.settings import atlas_root + +Finding = tuple[str, int, str] + +_PROHIBITED_ROLES = ( + "roles/owner", + "roles/editor", + "roles/resourcemanager.projectIamAdmin", +) + +# Literal-secret patterns. Deliberately do NOT match shell variable references. +_PRIVATE_KEY = re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----") +_GCP_API_KEY = re.compile(r"AIza[0-9A-Za-z_\-]{35}") +_SA_JSON = re.compile(r'"type"\s*:\s*"service_account"') +_SLACK_WEBHOOK = re.compile(r"https://hooks\.slack\.com/services/[A-Za-z0-9/_-]+") +_SLACK_TOKEN = re.compile(r"xox[baprs]-[A-Za-z0-9-]{10,}") +_LITERAL_BEARER = re.compile(r"Bearer\s+[A-Za-z0-9]{20,}") +_EMAIL = re.compile(r"[a-zA-Z0-9._%+-]+@(?:gmail|yahoo|hotmail|outlook)\.com") + + +def _iter_files(root: Path, patterns: list[str]) -> list[Path]: + files: list[Path] = [] + for pat in patterns: + files.extend(root.glob(pat)) + excluded = ("artifact-platform", "examples/artifact-dashboard", "infra/artifact-platform") + return sorted(f for f in files if f.is_file() and not any(x in str(f) for x in excluded)) + + +def scan_managed_iam() -> list[Finding]: + root = atlas_root() + findings: list[Finding] = [] + for path in _iter_files(root, ["scripts/*.sh", "infra/**/*.tf", "infra/**/*.sh"]): + for lineno, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): + low = line.strip() + if low.startswith("#"): + continue + for role in _PROHIBITED_ROLES: + if role in line and ("add-iam-policy-binding" in line or "role" in line.lower()): + findings.append((str(path.relative_to(root)), lineno, f"grants prohibited {role}")) + if "iam service-accounts keys create" in line or "--key-file" in line: + findings.append((str(path.relative_to(root)), lineno, "creates/uses a service-account key")) + return findings + + +def scan_data_exposure() -> list[Finding]: + root = atlas_root() + findings: list[Finding] = [] + targets = _iter_files( + root, + [ + "config/*.yaml", + "governance/**/*.yml", + "governance/**/*.json", + "observability/**/*.json", + "observability/**/*.txt", + "docs/evidence-sprint7/**/*", + ], + ) + checks = [ + (_PRIVATE_KEY, "private key material"), + (_GCP_API_KEY, "GCP API key"), + (_SA_JSON, "service-account JSON"), + (_SLACK_WEBHOOK, "Slack webhook URL"), + (_SLACK_TOKEN, "Slack token"), + (_LITERAL_BEARER, "literal bearer token"), + (_EMAIL, "personal email address"), + ] + for path in targets: + for lineno, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): + for pattern, reason in checks: + if pattern.search(line): + findings.append((str(path.relative_to(root)), lineno, reason)) + return findings + + +def scan_text(text: str) -> list[str]: + """Scan an arbitrary string (for regression tests). Never echoes the value.""" + reasons: list[str] = [] + for pattern, reason in [ + (_PRIVATE_KEY, "private key material"), + (_GCP_API_KEY, "GCP API key"), + (_SA_JSON, "service-account JSON"), + (_SLACK_WEBHOOK, "Slack webhook URL"), + (_SLACK_TOKEN, "Slack token"), + (_LITERAL_BEARER, "literal bearer token"), + (_EMAIL, "personal email address"), + ]: + if pattern.search(text): + reasons.append(reason) + return reasons + + +def all_findings() -> list[Finding]: + return scan_managed_iam() + scan_data_exposure() diff --git a/src/atlas/ingestion/__init__.py b/src/atlas/ingestion/__init__.py new file mode 100644 index 0000000..c5720f0 --- /dev/null +++ b/src/atlas/ingestion/__init__.py @@ -0,0 +1,10 @@ +"""Ingestion package for Project Atlas.""" + +from atlas.ingestion.upload import UploadResult, build_gcs_uri, build_object_name, upload_events_file + +__all__ = [ + "UploadResult", + "build_gcs_uri", + "build_object_name", + "upload_events_file", +] diff --git a/src/atlas/ingestion/upload.py b/src/atlas/ingestion/upload.py new file mode 100644 index 0000000..beb40e8 --- /dev/null +++ b/src/atlas/ingestion/upload.py @@ -0,0 +1,114 @@ +"""Cloud Storage ingestion for Project Atlas. + +Purpose: + Upload generated JSONL files to immutable, run-scoped GCS object paths. + +Interactions: + Reads local JSONL from the generator and writes objects consumed by the + BigQuery loader. Uses ``google.cloud.storage`` when credentials exist. + +Engineering principles: + - History is never overwritten: each run writes a unique object key. + - Idempotent upload checks for an existing object before writing. + +Common failure modes: + - Bucket does not exist or caller lacks ``storage.objects.create``. + - Attempting to overwrite an existing run object. + +Implementation choice: + Run-scoped keys ``raw/event_date=YYYY-MM-DD/batch_id=/events.jsonl`` extend + the Sprint 1 folder format without sacrificing immutability. Alternatives + considered: date-only keys (overwrite risk) and version IDs (harder to audit). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path + +import google.cloud.storage as storage + +from atlas.config.settings import AtlasSettings + + +@dataclass(frozen=True) +class UploadResult: + """Summary of a GCS upload.""" + + gcs_uri: str + object_name: str + bytes_uploaded: int + already_exists: bool + checksum_sha256: str | None = None + + +def build_object_name( + settings: AtlasSettings, + event_date: str, + identifier: str, + *, + use_batch_id: bool = False, +) -> str: + """Build the immutable GCS object key for a pipeline run or batch.""" + key_name = "batch_id" if use_batch_id else "run_id" + return f"{settings.ingestion.gcs_prefix}/event_date={event_date}/{key_name}={identifier}/events.jsonl" + + +def build_gcs_uri(settings: AtlasSettings, object_name: str) -> str: + """Build a gs:// URI for an object key.""" + return f"gs://{settings.gcp.bucket_name}/{object_name}" + + +def upload_events_file( + settings: AtlasSettings, + local_path: Path, + event_date: str, + run_id: str, + *, + client: storage.Client | None = None, + batch_id: str | None = None, + expected_checksum: str | None = None, + fail_once: bool = False, +) -> UploadResult: + """Upload a local JSONL file to Cloud Storage without overwriting history.""" + if fail_once: + raise RuntimeError("Injected transient upload failure for retry testing") + + use_batch = batch_id is not None + identifier = batch_id if batch_id is not None else run_id + object_name = build_object_name(settings, event_date, identifier, use_batch_id=use_batch) + gcs_uri = build_gcs_uri(settings, object_name) + storage_client = client or storage.Client(project=settings.gcp.project_id) + bucket = storage_client.bucket(settings.gcp.bucket_name) + blob = bucket.blob(object_name) + + if blob.exists(): + metadata = blob.metadata or {} + existing_checksum = metadata.get("checksum_sha256") + if expected_checksum and existing_checksum and existing_checksum != expected_checksum: + raise ValueError( + f"Existing GCS object checksum mismatch for {gcs_uri}: " + f"{existing_checksum} != {expected_checksum}" + ) + return UploadResult( + gcs_uri=gcs_uri, + object_name=object_name, + bytes_uploaded=blob.size or 0, + already_exists=True, + checksum_sha256=existing_checksum, + ) + + blob.metadata = {} + if expected_checksum: + blob.metadata["checksum_sha256"] = expected_checksum + blob.metadata["pipeline_run_id"] = run_id + if batch_id: + blob.metadata["batch_id"] = batch_id + blob.upload_from_filename(local_path, content_type="application/jsonl") + return UploadResult( + gcs_uri=gcs_uri, + object_name=object_name, + bytes_uploaded=local_path.stat().st_size, + already_exists=False, + checksum_sha256=expected_checksum, + ) diff --git a/src/atlas/loader/__init__.py b/src/atlas/loader/__init__.py new file mode 100644 index 0000000..ebef930 --- /dev/null +++ b/src/atlas/loader/__init__.py @@ -0,0 +1,15 @@ +"""BigQuery loader package.""" + +from atlas.loader.bigquery import ( + LoadResult, + ensure_events_table, + load_events_from_gcs, + render_create_table_sql, +) + +__all__ = [ + "LoadResult", + "ensure_events_table", + "load_events_from_gcs", + "render_create_table_sql", +] diff --git a/src/atlas/loader/bigquery.py b/src/atlas/loader/bigquery.py new file mode 100644 index 0000000..5457ad3 --- /dev/null +++ b/src/atlas/loader/bigquery.py @@ -0,0 +1,290 @@ +"""BigQuery loader for Project Atlas Sprint 1. + +Purpose: + Create dataset/table if missing, load JSONL from GCS, append metadata, and + preserve partition and cluster definitions. + +Interactions: + Reads GCS objects uploaded by ingestion and writes to ``atlas_raw.events``. + Uses run-scoped staging tables to make replays idempotent. + +Engineering principles: + - Append-only raw layer compatible with future dbt staging models. + - Explicit metadata columns for lineage and recovery. + +Common failure modes: + - Missing dataset or load job permissions. + - Duplicate batch attempted twice (guarded by batch check). + - Schema mismatch between JSONL and table definition. + +Implementation choice: + Load JSON to a run-scoped staging table, then INSERT into the partitioned + target table. Alternatives considered: direct append load (weaker replay + control) and external tables (less explicit metadata enrichment). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path + +from google.api_core.exceptions import NotFound +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, staging_table_id, table_fqn +from atlas.observability.cost import labeled_bigquery_client + +RAW_SCHEMA = [ + bigquery.SchemaField("event_id", "STRING", mode="REQUIRED"), + bigquery.SchemaField("user_id", "STRING", mode="NULLABLE"), + bigquery.SchemaField("event_name", "STRING", mode="REQUIRED"), + bigquery.SchemaField("event_timestamp", "TIMESTAMP", mode="REQUIRED"), + bigquery.SchemaField("event_date", "DATE", mode="REQUIRED"), + bigquery.SchemaField("country_code", "STRING", mode="NULLABLE"), + bigquery.SchemaField("platform", "STRING", mode="NULLABLE"), + bigquery.SchemaField("app_version", "STRING", mode="NULLABLE"), +] + +TARGET_SCHEMA = RAW_SCHEMA + [ + bigquery.SchemaField("ingested_at", "TIMESTAMP", mode="REQUIRED"), + bigquery.SchemaField("source_file", "STRING", mode="REQUIRED"), + bigquery.SchemaField("pipeline_run_id", "STRING", mode="REQUIRED"), + bigquery.SchemaField("batch_id", "STRING", mode="NULLABLE"), + bigquery.SchemaField("processing_date", "DATE", mode="NULLABLE"), +] + + +@dataclass(frozen=True) +class BatchLoadState: + """Existing raw rows for one stable batch identifier.""" + + row_count: int + ingestion_run_count: int + + +@dataclass(frozen=True) +class LoadResult: + """Summary of a BigQuery load operation.""" + + rows_loaded: int + target_table: str + staging_table: str + already_loaded: bool + source_file: str + batch_id: str | None = None + + +def render_create_table_sql(settings: AtlasSettings) -> str: + """Render the create-table SQL template.""" + template_path = Path(__file__).resolve().parents[3] / "sql" / "create_events_table.sql" + template = template_path.read_text(encoding="utf-8") + return template.format( + project_id=settings.gcp.project_id, + dataset_id=settings.gcp.dataset_id, + table_id=settings.gcp.table_id, + ) + + +def ensure_dataset(client: bigquery.Client, settings: AtlasSettings) -> None: + """Create the raw dataset if it does not exist.""" + dataset_ref = bigquery.Dataset(f"{settings.gcp.project_id}.{settings.gcp.dataset_id}") + dataset_ref.location = settings.gcp.location + try: + client.get_dataset(dataset_ref.dataset_id) + except NotFound: + client.create_dataset(dataset_ref, exists_ok=True) + + +def ensure_events_table(client: bigquery.Client, settings: AtlasSettings) -> None: + """Create the partitioned events table if it does not exist.""" + ensure_dataset(client, settings) + table_id = table_fqn(settings) + try: + client.get_table(table_id) + except NotFound: + table = bigquery.Table(table_id, schema=TARGET_SCHEMA) + table.time_partitioning = bigquery.TimePartitioning(field="event_date") + table.clustering_fields = ["event_name", "country_code"] + client.create_table(table) + + +def apply_sprint3_migration(client: bigquery.Client, settings: AtlasSettings) -> None: + """Apply additive batch_id migration when needed.""" + migration_path = Path(__file__).resolve().parents[3] / "sql" / "migrate_sprint3.sql" + rendered = migration_path.read_text(encoding="utf-8").format( + project_id=settings.gcp.project_id, + dataset_id=settings.gcp.dataset_id, + ) + client.query(rendered).result() + + +def batch_load_state( + client: bigquery.Client, + settings: AtlasSettings, + batch_id: str, +) -> BatchLoadState: + """Return existing raw row counts for one batch identifier.""" + query = f""" + SELECT + COUNT(1) AS row_count, + COUNT(DISTINCT pipeline_run_id) AS ingestion_run_count + FROM `{table_fqn(settings)}` + WHERE batch_id = @batch_id + """ + rows = list( + client.query( + query, + job_config=bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter("batch_id", "STRING", batch_id)] + ), + ).result() + ) + if not rows: + return BatchLoadState(row_count=0, ingestion_run_count=0) + return BatchLoadState( + row_count=int(rows[0]["row_count"] or 0), + ingestion_run_count=int(rows[0]["ingestion_run_count"] or 0), + ) + + +def run_already_loaded(client: bigquery.Client, settings: AtlasSettings, run_id: str) -> bool: + """Return True when the target table already contains rows for a run.""" + query = f""" + SELECT COUNT(1) AS row_count + FROM `{table_fqn(settings)}` + WHERE pipeline_run_id = @run_id + """ + job_config = bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter("run_id", "STRING", run_id)] + ) + rows = list(client.query(query, job_config=job_config).result()) + return bool(rows and rows[0]["row_count"] > 0) + + +def evaluate_batch_load( + state: BatchLoadState, + expected_row_count: int, +) -> str: + """Return load action: load, skip, or fail.""" + if state.row_count == 0: + return "load" + if state.row_count == expected_row_count: + return "skip" + if 0 < state.row_count < expected_row_count: + raise ValueError(f"Partial batch detected: expected {expected_row_count}, found {state.row_count}") + raise ValueError(f"Conflicting batch detected: expected {expected_row_count}, found {state.row_count}") + + +def load_events_from_gcs( + settings: AtlasSettings, + gcs_uri: str, + source_file: str, + run_id: str, + *, + client: bigquery.Client | None = None, + batch_id: str | None = None, + processing_date: str | None = None, + expected_row_count: int | None = None, +) -> LoadResult: + """Load a GCS JSONL file into the partitioned events table.""" + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "ingestion") + ensure_events_table(bq_client, settings) + apply_sprint3_migration(bq_client, settings) + + expected_rows = expected_row_count or settings.validation.expected_event_count + + if batch_id is not None: + state = batch_load_state(bq_client, settings, batch_id) + action = evaluate_batch_load(state, expected_rows) + if action == "skip": + return LoadResult( + rows_loaded=0, + target_table=table_fqn(settings), + staging_table="", + already_loaded=True, + source_file=source_file, + batch_id=batch_id, + ) + elif run_already_loaded(bq_client, settings, run_id): + return LoadResult( + rows_loaded=0, + target_table=table_fqn(settings), + staging_table="", + already_loaded=True, + source_file=source_file, + batch_id=batch_id, + ) + + staging_id = staging_table_id(settings, run_id) + staging_fqn = f"{settings.gcp.project_id}.{settings.gcp.dataset_id}.{staging_id}" + staging_table = bigquery.Table(staging_fqn, schema=RAW_SCHEMA) + bq_client.delete_table(staging_table, not_found_ok=True) + bq_client.create_table(staging_table) + + load_job_config = bigquery.LoadJobConfig( + source_format=bigquery.SourceFormat.NEWLINE_DELIMITED_JSON, + schema=RAW_SCHEMA, + write_disposition=bigquery.WriteDisposition.WRITE_TRUNCATE, + ignore_unknown_values=True, + ) + load_job = bq_client.load_table_from_uri(gcs_uri, staging_fqn, job_config=load_job_config) + load_job.result() + + ingested_at = datetime.now(tz=UTC).isoformat() + insert_sql = f""" + INSERT INTO `{table_fqn(settings)}` ( + event_id, + user_id, + event_name, + event_timestamp, + event_date, + country_code, + platform, + app_version, + ingested_at, + source_file, + pipeline_run_id, + batch_id, + processing_date + ) + SELECT + event_id, + user_id, + event_name, + event_timestamp, + event_date, + country_code, + platform, + app_version, + TIMESTAMP(@ingested_at) AS ingested_at, + @source_file AS source_file, + @run_id AS pipeline_run_id, + @batch_id AS batch_id, + DATE(@processing_date) AS processing_date + FROM `{staging_fqn}` + """ + query_job = bq_client.query( + insert_sql, + job_config=bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter("ingested_at", "STRING", ingested_at), + bigquery.ScalarQueryParameter("source_file", "STRING", source_file), + bigquery.ScalarQueryParameter("run_id", "STRING", run_id), + bigquery.ScalarQueryParameter("batch_id", "STRING", batch_id), + bigquery.ScalarQueryParameter("processing_date", "STRING", processing_date), + ] + ), + ) + query_job.result() + inserted_rows = query_job.num_dml_affected_rows or 0 + bq_client.delete_table(staging_table, not_found_ok=True) + + return LoadResult( + rows_loaded=inserted_rows, + target_table=table_fqn(settings), + staging_table=staging_fqn, + already_loaded=False, + source_file=source_file, + batch_id=batch_id, + ) diff --git a/src/atlas/logging/__init__.py b/src/atlas/logging/__init__.py new file mode 100644 index 0000000..01a828e --- /dev/null +++ b/src/atlas/logging/__init__.py @@ -0,0 +1,5 @@ +"""Logging package for Project Atlas.""" + +from atlas.logging.structured import StepLogger, configure_logging, new_pipeline_run_id + +__all__ = ["StepLogger", "configure_logging", "new_pipeline_run_id"] diff --git a/src/atlas/logging/structured.py b/src/atlas/logging/structured.py new file mode 100644 index 0000000..7d8668b --- /dev/null +++ b/src/atlas/logging/structured.py @@ -0,0 +1,128 @@ +"""Structured pipeline logging for Project Atlas. + +Purpose: + Emit consistent, machine-readable logs for every pipeline step. + +Interactions: + Called by generator, upload, loader, validation, and orchestrator. + Writes JSON lines to ``logs/.jsonl``. + +Engineering principles: + - Observability without external monitoring in Sprint 1. + - Every log record includes run identity and source file for recovery. + +Common failure modes: + - Missing log directory permissions in Cloud Shell. + - Duplicate handlers if ``configure_logging`` is called repeatedly. + +Implementation choice: + Standard library logging with a JSON formatter keeps dependencies minimal. + Alternatives considered: structlog (extra dependency) and print-based logs + (insufficient for automated acceptance tests). +""" + +from __future__ import annotations + +import json +import logging +import sys +import time +import uuid +from dataclasses import dataclass, field +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Literal + + +class JsonLogFormatter(logging.Formatter): + """Format log records as single-line JSON objects.""" + + def format(self, record: logging.LogRecord) -> str: + payload = { + "timestamp": datetime.fromtimestamp(record.created, tz=UTC).isoformat(), + "level": record.levelname, + "logger": record.name, + "message": record.getMessage(), + } + for key in ( + "pipeline_run_id", + "step", + "status", + "duration_ms", + "rows_processed", + "source_file", + "details", + ): + if hasattr(record, key): + payload[key] = getattr(record, key) + return json.dumps(payload, default=str) + + +def configure_logging(log_dir: Path, pipeline_run_id: str) -> logging.Logger: + """Configure root Atlas logger with console and file handlers.""" + log_dir.mkdir(parents=True, exist_ok=True) + logger = logging.getLogger("atlas") + logger.setLevel(logging.INFO) + logger.handlers.clear() + logger.propagate = False + + formatter = JsonLogFormatter() + stream_handler = logging.StreamHandler(sys.stdout) + stream_handler.setFormatter(formatter) + logger.addHandler(stream_handler) + + file_handler = logging.FileHandler(log_dir / f"{pipeline_run_id}.jsonl") + file_handler.setFormatter(formatter) + logger.addHandler(file_handler) + return logger + + +def new_pipeline_run_id(prefix: str = "atlas") -> str: + """Create a unique pipeline run identifier.""" + timestamp = datetime.now(tz=UTC).strftime("%Y%m%dT%H%M%SZ") + return f"{prefix}-{timestamp}-{uuid.uuid4().hex[:8]}" + + +@dataclass +class StepLogger: + """Context manager that logs step start, success, and failure.""" + + logger: logging.Logger + pipeline_run_id: str + step: str + source_file: str | None = None + rows_processed: int | None = None + details: dict[str, Any] = field(default_factory=dict) + _started_at: float = field(default=0.0, init=False) + + def __enter__(self) -> StepLogger: + self._started_at = time.perf_counter() + self._log("STARTED", rows_processed=self.rows_processed) + return self + + def __exit__(self, exc_type, exc, exc_tb) -> Literal[False]: + duration_ms = int((time.perf_counter() - self._started_at) * 1000) + if exc_type is None: + self._log("SUCCEEDED", duration_ms=duration_ms, rows_processed=self.rows_processed) + return False + self.details["error"] = str(exc) + self._log("FAILED", duration_ms=duration_ms, rows_processed=self.rows_processed) + return False + + def _log( + self, + status: str, + duration_ms: int | None = None, + rows_processed: int | None = None, + ) -> None: + extra = { + "pipeline_run_id": self.pipeline_run_id, + "step": self.step, + "status": status, + "source_file": self.source_file, + "rows_processed": rows_processed, + "details": self.details, + } + if duration_ms is not None: + extra["duration_ms"] = duration_ms + self.logger.info(f"{self.step} {status.lower()}", extra=extra) diff --git a/src/atlas/observability/__init__.py b/src/atlas/observability/__init__.py new file mode 100644 index 0000000..664c4ee --- /dev/null +++ b/src/atlas/observability/__init__.py @@ -0,0 +1 @@ +"""Atlas observability plane: structured logging contract, metrics, checks.""" diff --git a/src/atlas/observability/checks.py b/src/atlas/observability/checks.py new file mode 100644 index 0000000..d62d536 --- /dev/null +++ b/src/atlas/observability/checks.py @@ -0,0 +1,106 @@ +"""Bridge warehouse validation results into durable quality records (Phase 4). + +The Sprint 4 warehouse validator (atlas.validation.warehouse) computes ten +batch-scoped reconciliation checks and returns a WarehouseReport. Sprint 5 +persists each check into ``atlas_ops.quality_results`` so correctness +evidence survives the task log. dbt test evidence is summarized here, not +re-implemented: the dbt_build task already fails on test failures, and the +reconciliation checks verify the resulting tables directly. +""" + +from __future__ import annotations + +import numbers +from datetime import UTC, datetime +from typing import TYPE_CHECKING, Any + +from atlas.config.settings import AtlasSettings +from atlas.observability.logging import emit_event +from atlas.ops.quality_results import ( + QualityResultRecord, + details_to_json, + upsert_quality_result, +) + +if TYPE_CHECKING: + from google.cloud import bigquery + + from atlas.validation.warehouse import WarehouseReport + +# Category mapping for the warehouse reconciliation checks (ADR-011 Plane 1). +CHECK_CATEGORIES: dict[str, str] = { + "batch_nonempty": "COMPLETENESS", + "raw_equals_classification": "RECONCILIATION", + "accepted_plus_rejected_equals_raw": "RECONCILIATION", + "accepted_equals_fact": "RECONCILIATION", + "fact_event_ids_unique": "UNIQUENESS", + "fact_user_fk_resolves": "REFERENTIAL_INTEGRITY", + "fact_country_fk_resolves": "REFERENTIAL_INTEGRITY", + "mart_totals_reconcile": "RECONCILIATION", + "processing_date_semantics": "COMPLETENESS", + "batch_lineage_semantics": "COMPLETENESS", +} +_DEFAULT_CATEGORY = "RECONCILIATION" + + +def _as_float(value: Any) -> float | None: + if isinstance(value, bool) or not isinstance(value, numbers.Real): + return None + return float(value) + + +def persist_warehouse_report( + report: WarehouseReport, + pipeline_run_id: str, + *, + git_sha: str | None = None, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> int: + """Write one quality_results row per warehouse check; returns rows written. + + Persistence is telemetry: failures emit a structured error and are + reported, but never mask the validation outcome itself. + """ + written = 0 + evaluated_at = datetime.now(tz=UTC).isoformat() + for check in report.checks: + record = QualityResultRecord( + pipeline_run_id=pipeline_run_id, + batch_id=report.batch_id, + check_name=check.name, + check_category=CHECK_CATEGORIES.get(check.name, _DEFAULT_CATEGORY), + severity="CRITICAL" if check.status == "FAIL" else "INFO", + status=check.status, + evaluated_at=evaluated_at, + observed_value=_as_float(check.actual), + expected_value=_as_float(check.expected), + details_json=details_to_json( + {"message": check.message, "expected": check.expected, "actual": check.actual} + ), + git_sha=git_sha, + ) + try: + upsert_quality_result(record, settings, client=client) + written += 1 + except Exception as exc: # noqa: BLE001 - telemetry must not mask validation + emit_event( + "quality_result_write_failed", + severity="ERROR", + component="quality_results", + pipeline_run_id=pipeline_run_id, + batch_id=report.batch_id, + check_name=check.name, + error_type=type(exc).__name__, + error_message=str(exc), + ) + emit_event( + "warehouse_quality_persisted", + severity="INFO" if written == len(report.checks) else "WARNING", + component="quality_results", + pipeline_run_id=pipeline_run_id, + batch_id=report.batch_id, + status=report.overall_status, + details={"checks": len(report.checks), "persisted": written}, + ) + return written diff --git a/src/atlas/observability/cost.py b/src/atlas/observability/cost.py new file mode 100644 index 0000000..49b95db --- /dev/null +++ b/src/atlas/observability/cost.py @@ -0,0 +1,51 @@ +"""BigQuery cost attribution for Atlas (Sprint 5, Phase 7 / ADR-012). + +Attribution strategy, in evidence order: +1. Job labels (this module for Python jobs; dbt query-comment job-label for dbt). +2. Runtime identity (atlas-composer-runtime / atlas-github-* service accounts). +3. Referenced/destination Atlas datasets. + +``labeled_bigquery_client`` returns a client whose default query job config +carries the bounded attribution labels, so every Atlas Python query is +attributable in region-qualified ``INFORMATION_SCHEMA.JOBS`` without touching +individual call sites' query logic. Run/batch identifiers are deliberately +excluded from job labels (unnecessary cardinality; correlation lives in +Planes 1-2). +""" + +from __future__ import annotations + +from google.cloud import bigquery + +# Bounded label vocabulary (BigQuery label charset: lowercase, digits, _ , -). +ALLOWED_COMPONENTS = frozenset( + { + "pipeline", + "monitor", + "deployment", + "validation", + "audit", + "migration", + "ingestion", + "adhoc", + } +) + + +def attribution_labels(component: str, environment: str = "atlas-dev") -> dict[str, str]: + """Return the standard Atlas job labels for one bounded component.""" + if component not in ALLOWED_COMPONENTS: + raise ValueError(f"component {component!r} not in bounded set {sorted(ALLOWED_COMPONENTS)}") + return {"application": "atlas", "component": component, "environment": environment} + + +def labeled_bigquery_client( + project_id: str, + component: str, + environment: str = "atlas-dev", +) -> bigquery.Client: + """BigQuery client whose queries default to Atlas attribution labels.""" + return bigquery.Client( + project=project_id, + default_query_job_config=bigquery.QueryJobConfig(labels=attribution_labels(component, environment)), + ) diff --git a/src/atlas/observability/cost_guard.py b/src/atlas/observability/cost_guard.py new file mode 100644 index 0000000..37f2984 --- /dev/null +++ b/src/atlas/observability/cost_guard.py @@ -0,0 +1,167 @@ +"""Cost-guard CLI + control loading (Sprint 7, ADR-020). + +Extends the Sprint 6 guards (`atlas.observability.cost_guards`) with a +config-driven estimator and static checks. + +Usage:: + + python -m atlas.observability.cost_guard estimate \\ + --sql-file q.sql --project

--location US [--environment atlas-dev] + python -m atlas.observability.cost_guard check-partition-filter \\ + --sql-file q.sql [--asset atlas_raw.events] + +`estimate` always dry-runs first (bills $0), reports estimated bytes, compares +with the environment threshold, and refuses over-limit execution unless an +explicit approved override is provided. It never executes on estimation failure. +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import sys +from pathlib import Path +from typing import Any + +import yaml + +from atlas.config.settings import atlas_root +from atlas.observability.cost_guards import CostGuardViolation + +CONTROLS_PATH = "config/cost_controls.yaml" +OVERRIDE_VAR = "ATLAS_APPROVE_COST_OVERRIDE" +SUITE_BYTES_ENV = "ATLAS_MAX_PERFORMANCE_TEST_BYTES" + + +def load_cost_controls() -> dict[str, Any]: + path = atlas_root() / CONTROLS_PATH + return yaml.safe_load(path.read_text(encoding="utf-8")) + + +def environment_controls(environment: str = "atlas-dev") -> dict[str, Any]: + controls = load_cost_controls() + envs = controls.get("environments", {}) + if environment not in envs: + raise CostGuardViolation(f"unknown cost-control environment '{environment}'") + return envs[environment] + + +def max_query_bytes(environment: str = "atlas-dev") -> int: + return int(environment_controls(environment)["max_query_bytes"]) + + +def max_performance_suite_bytes(environment: str = "atlas-dev") -> int: + override = os.environ.get(SUITE_BYTES_ENV) + if override and override.strip().isdigit(): + return int(override) + return int(environment_controls(environment)["max_performance_suite_bytes"]) + + +# A "partition filter" is a WHERE/AND predicate on a partition column. Kept +# deliberately simple and offline for the static check. +_PARTITION_COLS = ("event_date", "_partitiondate", "_partitiontime", "processing_date") + + +def has_partition_filter(sql: str) -> bool: + lowered = sql.lower() + if "where" not in lowered: + return False + return any(re.search(rf"\b{col}\b", lowered) for col in _PARTITION_COLS) + + +def check_partition_filter(sql: str, asset: str | None, environment: str = "atlas-dev") -> None: + controls = environment_controls(environment) + required = set(controls.get("require_partition_filter_assets", []) or []) + asset_requires = ( + asset in required + if asset + else bool( + re.search(r"\b(atlas_raw\.events|atlas_core\.fct_events|fct_events|`?events`?)\b", sql.lower()) + ) + ) + if asset_requires and not has_partition_filter(sql): + raise CostGuardViolation( + f"query over {asset or 'a partitioned asset'} is missing a required " + "partition filter (event_date/processing_date)" + ) + + +def estimate( + sql: str, + project: str, + location: str, + environment: str = "atlas-dev", + allow_override: bool = False, +) -> dict[str, Any]: + """Dry-run estimate + threshold enforcement. Returns structured evidence.""" + from google.cloud import bigquery # imported lazily so static tests need no cloud + + client = bigquery.Client(project=project, location=location) + job = client.query(sql, job_config=bigquery.QueryJobConfig(dry_run=True, use_query_cache=False)) + estimated = int(job.total_bytes_processed or 0) + ceiling = max_query_bytes(environment) + override = allow_override or os.environ.get(OVERRIDE_VAR, "").lower() == "true" + evidence = { + "estimated_bytes": estimated, + "ceiling_bytes": ceiling, + "environment": environment, + "within_ceiling": estimated <= ceiling, + "override_applied": override and estimated > ceiling, + "location": location, + "project": project, + } + if estimated > ceiling and not override: + evidence["decision"] = "BLOCKED" + print(json.dumps(evidence, indent=2)) + raise CostGuardViolation( + f"estimate {estimated} bytes exceeds ceiling {ceiling} bytes for '{environment}'; " + f"set {OVERRIDE_VAR}=true only after a documented cost review" + ) + evidence["decision"] = "ALLOW" + return evidence + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Atlas cost guard") + sub = parser.add_subparsers(dest="command", required=True) + + est = sub.add_parser("estimate") + est.add_argument("--sql-file", type=Path, required=True) + est.add_argument("--project", required=True) + est.add_argument("--location", default="US") + est.add_argument("--environment", default="atlas-dev") + est.add_argument("--allow-override", action="store_true") + + pf = sub.add_parser("check-partition-filter") + pf.add_argument("--sql-file", type=Path, required=True) + pf.add_argument("--asset") + pf.add_argument("--environment", default="atlas-dev") + + args = parser.parse_args(argv) + sql = args.sql_file.read_text(encoding="utf-8") + + try: + if args.command == "estimate": + evidence = estimate( + sql, + args.project, + args.location, + args.environment, + args.allow_override, + ) + print(json.dumps(evidence, indent=2)) + return 0 + if args.command == "check-partition-filter": + check_partition_filter(sql, args.asset, args.environment) + print("partition filter present or not required") + return 0 + except CostGuardViolation as exc: + print(f"COST GUARD: {exc}", file=sys.stderr) + return 2 + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/observability/cost_guards.py b/src/atlas/observability/cost_guards.py new file mode 100644 index 0000000..9b5222a --- /dev/null +++ b/src/atlas/observability/cost_guards.py @@ -0,0 +1,140 @@ +"""BigQuery cost guardrails (Sprint 6, Phase 12). + +Guards fail *before* material spend: + +- ``estimate_query_bytes`` — dry-run estimate (bills nothing) +- ``enforce_dry_run_ceiling`` — refuse queries whose estimate exceeds the ceiling +- ``guarded_query_config`` — hard ``maximum_bytes_billed`` enforcement +- ``validate_backfill_window`` — bounded backfill windows, explicit override only +- ``require_full_refresh_approval`` — full refresh is an approved exception, + never a default + +Every rejection raises ``CostGuardViolation`` with the evidence (estimated +bytes, requested window) so the responsible component is identifiable without +running the expensive work. +""" + +from __future__ import annotations + +import os +from datetime import date, timedelta +from typing import Any + +from google.cloud import bigquery + +from atlas.observability.logging import emit_event + +# Initial operational ceilings for the synthetic Atlas workload (not SLOs). +# A full scan of every Atlas dataset today is < 100 MB; 1 GiB catches an +# unpartitioned-scan mistake with an order-of-magnitude margin. +DEFAULT_MAX_ESTIMATED_BYTES = 1 * 1024**3 +DEFAULT_MAX_BACKFILL_DAYS = 7 +FULL_REFRESH_APPROVAL_VAR = "ATLAS_APPROVE_FULL_REFRESH" +BACKFILL_OVERRIDE_VAR = "ATLAS_APPROVE_UNBOUNDED_BACKFILL" + + +class CostGuardViolation(RuntimeError): + """A guarded operation would exceed its cost boundary.""" + + +def estimate_query_bytes(client: bigquery.Client, sql: str) -> int: + """Dry-run a query and return the estimated bytes processed (bills $0).""" + job = client.query(sql, job_config=bigquery.QueryJobConfig(dry_run=True, use_query_cache=False)) + return int(job.total_bytes_processed or 0) + + +def enforce_dry_run_ceiling( + client: bigquery.Client, + sql: str, + *, + max_estimated_bytes: int = DEFAULT_MAX_ESTIMATED_BYTES, + component: str = "adhoc", +) -> int: + """Refuse execution when the dry-run estimate exceeds the ceiling. + + Returns the estimate so callers can record it as evidence. + """ + estimated = estimate_query_bytes(client, sql) + if estimated > max_estimated_bytes: + emit_event( + "cost_guard_blocked", + severity="ERROR", + component=component, + check_name="dry_run_ceiling", + observed_value=estimated, + threshold=max_estimated_bytes, + status="BLOCKED", + ) + raise CostGuardViolation( + f"query estimate {estimated} bytes exceeds ceiling {max_estimated_bytes} bytes; " + "add a partition filter or raise the ceiling with documented approval" + ) + return estimated + + +def guarded_query_config( + *, + maximum_bytes_billed: int = DEFAULT_MAX_ESTIMATED_BYTES, + labels: dict[str, str] | None = None, +) -> bigquery.QueryJobConfig: + """Job config that hard-fails the query at the BigQuery layer before spend.""" + config = bigquery.QueryJobConfig(maximum_bytes_billed=maximum_bytes_billed) + if labels: + config.labels = labels + return config + + +def validate_backfill_window( + start_date: date, + end_date: date, + *, + max_days: int = DEFAULT_MAX_BACKFILL_DAYS, + env: dict[str, str] | None = None, +) -> int: + """Reject backfill windows beyond policy unless explicitly overridden. + + Returns the window size in days. The override variable must be exactly + 'true'; an unbounded backfill can never happen by accident. + """ + env = env if env is not None else dict(os.environ) + if end_date < start_date: + raise CostGuardViolation(f"backfill window end {end_date} precedes start {start_date}") + days = (end_date - start_date).days + 1 + if days > max_days and env.get(BACKFILL_OVERRIDE_VAR, "").lower() != "true": + emit_event( + "cost_guard_blocked", + severity="ERROR", + component="pipeline", + check_name="backfill_window", + observed_value=days, + threshold=max_days, + status="BLOCKED", + ) + raise CostGuardViolation( + f"backfill window of {days} days exceeds the {max_days}-day policy; " + f"set {BACKFILL_OVERRIDE_VAR}=true only after a documented cost review" + ) + return days + + +def require_full_refresh_approval(env: dict[str, str] | None = None) -> None: + """Block dbt full refresh unless the approval variable is explicitly true.""" + env = env if env is not None else dict(os.environ) + if env.get(FULL_REFRESH_APPROVAL_VAR, "").lower() != "true": + emit_event( + "cost_guard_blocked", + severity="ERROR", + component="pipeline", + check_name="full_refresh_approval", + status="BLOCKED", + ) + raise CostGuardViolation( + f"full refresh requires {FULL_REFRESH_APPROVAL_VAR}=true; " + "incremental processing is the default and targeted repair is the " + "first response to corruption (ADR-014)" + ) + + +def backfill_dates(start_date: date, days: int) -> list[Any]: + """Enumerate the dates of a validated backfill window.""" + return [start_date + timedelta(days=offset) for offset in range(days)] diff --git a/src/atlas/observability/logging.py b/src/atlas/observability/logging.py new file mode 100644 index 0000000..117d050 --- /dev/null +++ b/src/atlas/observability/logging.py @@ -0,0 +1,263 @@ +"""Structured logging contract for Project Atlas (Sprint 5, ADR-011). + +One JSON-per-line contract for every Atlas structured event, across the step +runner, Airflow callbacks, deployment scripts, and the observability monitor. +Events go to stdout so Composer/Airflow routes them into task logs and, via +the Atlas log sink, into the dedicated log bucket. + +Contract guarantees (tested in tests/unit/test_observability_logging.py): +- deterministic field names drawn from a fixed allowlist, +- controlled severity and event vocabulary, +- UTC ISO-8601 timestamps, +- centralized error sanitization and truncation (reuses the audited + sanitizer from atlas.ops.audit), +- non-serializable values degrade to strings instead of raising, +- emission failures never propagate into the caller's data path. +""" + +from __future__ import annotations + +import json +import os +import sys +import uuid +from datetime import UTC, datetime +from typing import Any, TextIO + +from atlas.ops.audit import sanitize_error_message + +# Stable envelope marker so log filters can select Atlas contract events +# without matching unrelated JSON output. +EVENT_MARKER = "atlas_event" + +ALLOWED_SEVERITIES = frozenset({"DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"}) + +# Correlation hierarchy (ADR-011): +# deployment_id -> airflow_run_id -> pipeline_run_id -> batch_id -> task_id -> attempt_number +CORRELATION_FIELDS = ( + "deployment_id", + "airflow_run_id", + "pipeline_run_id", + "batch_id", + "task_id", + "attempt_number", +) + +# Full field allowlist. Anything not listed here is rejected in strict mode +# and dropped (with a contract_violation note) otherwise. +ALLOWED_FIELDS = frozenset( + { + "timestamp", + "severity", + "event_type", + "component", + "environment", + "git_sha", + "dag_id", + "processing_date", + "status", + "duration_ms", + "rows_generated", + "rows_loaded", + "rows_accepted", + "rows_rejected", + "fact_rows", + "mart_event_count", + "check_name", + "check_category", + "observed_value", + "threshold", + "expected_value", + "error_type", + "error_message", + "correlation_id", + "message", + "details", + *CORRELATION_FIELDS, + } +) + +_INT_FIELDS = frozenset( + { + "attempt_number", + "duration_ms", + "rows_generated", + "rows_loaded", + "rows_accepted", + "rows_rejected", + "fact_rows", + "mart_event_count", + } +) + +_MAX_ERROR_LENGTH = 2000 +_MAX_DETAILS_LENGTH = 4000 + +# Direct Cloud Logging emission (Sprint 5 live-acceptance mitigation). +# Composer 3 build.13 was observed not exporting any Airflow component logs +# to the customer project (documented in the Sprint 5 incident/validation +# reports), which silently strands stdout-only telemetry. When this env var +# is "true", contract events are ALSO written straight to the Cloud Logging +# API under logName atlas-events, where the Atlas sink filter +# (jsonPayload.atlas_event=true) routes them to the atlas-observability +# bucket. Off by default: local runs, unit tests, and CI stay offline. +CLOUD_EMIT_ENV_VAR = "ATLAS_LOG_TO_CLOUD_LOGGING" +CLOUD_LOG_NAME = "atlas-events" + +_SEVERITY_RANK = {"DEBUG": 100, "INFO": 200, "WARNING": 400, "ERROR": 500, "CRITICAL": 600} + +_cloud_logger: Any = None +_cloud_logger_failed = False + + +def _get_cloud_logger() -> Any: + """Lazily build (and cache) a Cloud Logging logger; never raises.""" + global _cloud_logger, _cloud_logger_failed + if _cloud_logger is not None or _cloud_logger_failed: + return _cloud_logger + try: + import google.cloud.logging as gcloud_logging + + _cloud_logger = gcloud_logging.Client().logger(CLOUD_LOG_NAME) + except Exception: # noqa: BLE001 - degraded telemetry must not break callers + _cloud_logger_failed = True + return _cloud_logger + + +def _emit_to_cloud(event: dict[str, Any]) -> None: + """Best-effort direct write of one contract event to Cloud Logging.""" + logger = _get_cloud_logger() + if logger is None: + return + try: + logger.log_struct(event, severity=event.get("severity", "INFO")) + except Exception: # noqa: BLE001, S110 - fallback path; stdout copy already exists + pass + + +class ContractViolation(ValueError): + """Raised in strict mode when an event violates the logging contract.""" + + +def _coerce(value: Any) -> Any: + """Return a JSON-serializable representation without raising.""" + if value is None or isinstance(value, (str, int, float, bool)): + return value + if isinstance(value, datetime): + return value.astimezone(UTC).isoformat() + if isinstance(value, dict): + return {str(k): _coerce(v) for k, v in value.items()} + if isinstance(value, (list, tuple)): + return [_coerce(v) for v in value] + try: + json.dumps(value) + return value + except (TypeError, ValueError): + return repr(value)[:500] + + +def build_event( + event_type: str, + *, + severity: str = "INFO", + strict: bool = False, + **fields: Any, +) -> dict[str, Any]: + """Build a contract-conformant event dict. + + In strict mode unknown fields or invalid severities raise + ContractViolation; otherwise they are dropped/normalized and noted under + ``contract_violations`` so telemetry bugs stay visible without breaking + the caller. + """ + if not event_type or not isinstance(event_type, str): + raise ContractViolation("event_type is required") + severity = severity.upper() + violations: list[str] = [] + if severity not in ALLOWED_SEVERITIES: + if strict: + raise ContractViolation(f"invalid severity: {severity}") + violations.append(f"severity:{severity}") + severity = "INFO" + + event: dict[str, Any] = { + EVENT_MARKER: True, + "timestamp": datetime.now(tz=UTC).isoformat(), + "severity": severity, + "event_type": event_type, + } + + for key, value in fields.items(): + if value is None: + continue + if key not in ALLOWED_FIELDS: + if strict: + raise ContractViolation(f"field not in contract: {key}") + violations.append(f"field:{key}") + continue + if key in _INT_FIELDS: + try: + value = int(value) + except (TypeError, ValueError): + violations.append(f"type:{key}") + continue + if key == "error_message": + value = sanitize_error_message(str(value), max_length=_MAX_ERROR_LENGTH) + if key == "details": + value = _coerce(value) + rendered = json.dumps(value, default=str) + if len(rendered) > _MAX_DETAILS_LENGTH: + value = {"truncated": True, "preview": rendered[:_MAX_DETAILS_LENGTH]} + event[key] = _coerce(value) + + if violations: + event["contract_violations"] = violations + return event + + +def emit_event( + event_type: str, + *, + severity: str = "INFO", + stream: TextIO | None = None, + strict: bool = False, + **fields: Any, +) -> dict[str, Any] | None: + """Build and print one structured event line; never raises in non-strict mode. + + Returns the event dict (useful for tests) or None when emission failed + and a fallback error line was printed instead. + """ + out = stream if stream is not None else sys.stdout + try: + event = build_event(event_type, severity=severity, strict=strict, **fields) + print(json.dumps(event, default=str), file=out, flush=True) + if os.environ.get(CLOUD_EMIT_ENV_VAR, "").lower() == "true": + _emit_to_cloud(event) + return event + except ContractViolation: + raise + except Exception as exc: # noqa: BLE001 - telemetry must not break the data path + fallback = { + EVENT_MARKER: True, + "timestamp": datetime.now(tz=UTC).isoformat(), + "severity": "ERROR", + "event_type": "telemetry_emit_failed", + "error_type": type(exc).__name__, + "error_message": sanitize_error_message(str(exc), max_length=_MAX_ERROR_LENGTH), + } + try: + print(json.dumps(fallback, default=str), file=out, flush=True) + except Exception: # noqa: BLE001, S110 - last resort: stay silent, never raise + pass + return None + + +def new_correlation_id() -> str: + """Random identifier linking events emitted by one logical operation.""" + return uuid.uuid4().hex + + +def correlation_fields_from_context(context: dict[str, Any]) -> dict[str, Any]: + """Extract the standard correlation identifiers from a run-context dict.""" + return {key: context.get(key) for key in CORRELATION_FIELDS if context.get(key) is not None} diff --git a/src/atlas/observability/metrics.py b/src/atlas/observability/metrics.py new file mode 100644 index 0000000..26c0111 --- /dev/null +++ b/src/atlas/observability/metrics.py @@ -0,0 +1,195 @@ +"""Cloud Monitoring metric publication for Atlas (Sprint 5, Phase 6). + +Descriptors are declared once in ``observability/metrics/metric-descriptors.json`` +(the versioned catalog and cardinality budget) and created idempotently by +``ensure_descriptors``. Publishing validates every point against the catalog: +unknown metric types or labels outside the bounded sets are rejected before +they can create unbudgeted time series. + +Telemetry-safety: ``publish_gauge_safely`` never raises; a Monitoring outage +degrades to a structured ``metric_publish_failed`` event (ADR-011). + +The google-cloud-monitoring dependency is imported lazily so DAG parsing and +credentialless static validation never require it. +""" + +from __future__ import annotations + +import argparse +import json +import time +from pathlib import Path +from typing import Any + +from atlas.observability.logging import emit_event + +_CATALOG_PATH = Path(__file__).resolve().parents[3] / "observability" / "metrics" / "metric-descriptors.json" + +MONITOR_STATUS_VALUES = {"PASS": 0, "WARN": 1, "FAIL": 2, "NO_DATA": -1, "DISABLED": -2} + +_ALLOWED_LABEL_VALUES: dict[str, Any] = { + "environment": {"atlas-dev"}, + "dag_id": {"atlas_batch_pipeline", "atlas_observability_monitor"}, + "component": {"pipeline", "monitor", "deployment", "cost"}, + "severity": {"INFO", "WARNING", "CRITICAL"}, + "mode": {"normal", "drill"}, + "status": {"SUCCESS", "FAILED", "ROLLED_BACK", "ROLLBACK_FAILED"}, + # check_name is bounded by config/observability.yaml; validated for shape only. + "check_name": None, +} +_MAX_CHECK_NAME_LENGTH = 64 + + +class MetricContractError(ValueError): + """Raised when a publish request violates the metric catalog.""" + + +def load_catalog(path: Path | None = None) -> dict[str, dict[str, Any]]: + """Return {metric_type: descriptor} from the versioned catalog.""" + raw = json.loads((path or _CATALOG_PATH).read_text(encoding="utf-8")) + return {d["type"]: d for d in raw["descriptors"]} + + +def validate_point( + metric_type: str, + labels: dict[str, str], + catalog: dict[str, dict[str, Any]] | None = None, +) -> dict[str, Any]: + """Validate one metric point against the catalog; returns the descriptor.""" + catalog = catalog or load_catalog() + descriptor = catalog.get(metric_type) + if descriptor is None: + raise MetricContractError(f"metric not in catalog: {metric_type}") + allowed_labels = set(descriptor["labels"]) + for key, value in labels.items(): + if key not in allowed_labels: + raise MetricContractError(f"label {key!r} not allowed on {metric_type}") + bounded = _ALLOWED_LABEL_VALUES.get(key) + if bounded is not None and value not in bounded: + raise MetricContractError(f"label value {key}={value!r} outside bounded set") + if key == "check_name" and (not value or len(value) > _MAX_CHECK_NAME_LENGTH): + raise MetricContractError("check_name label must be short and non-empty") + missing = allowed_labels - set(labels) + if missing: + raise MetricContractError(f"missing required labels for {metric_type}: {sorted(missing)}") + return descriptor + + +def ensure_descriptors(project_id: str, *, catalog_path: Path | None = None) -> list[str]: + """Idempotently create catalog descriptors; returns the created types.""" + import google.cloud.monitoring_v3 as monitoring_v3 + from google.api import label_pb2, metric_pb2 + + client = monitoring_v3.MetricServiceClient() + project_name = f"projects/{project_id}" + existing = { + d.type + for d in client.list_metric_descriptors( + request={ + "name": project_name, + "filter": 'metric.type = starts_with("custom.googleapis.com/atlas/")', + } + ) + } + created: list[str] = [] + kind_map = { + "GAUGE": metric_pb2.MetricDescriptor.MetricKind.GAUGE, + "CUMULATIVE": metric_pb2.MetricDescriptor.MetricKind.CUMULATIVE, + } + value_map = { + "DOUBLE": metric_pb2.MetricDescriptor.ValueType.DOUBLE, + "INT64": metric_pb2.MetricDescriptor.ValueType.INT64, + } + for metric_type, spec in load_catalog(catalog_path).items(): + if metric_type in existing: + continue + descriptor = metric_pb2.MetricDescriptor( + type=metric_type, + metric_kind=kind_map[spec["metric_kind"]], + value_type=value_map[spec["value_type"]], + unit=spec.get("unit", "1"), + description=spec["description"], + labels=[ + label_pb2.LabelDescriptor(key=key, value_type=label_pb2.LabelDescriptor.ValueType.STRING) + for key in spec["labels"] + ], + ) + client.create_metric_descriptor(name=project_name, metric_descriptor=descriptor) + created.append(metric_type) + return created + + +def publish_gauge( + project_id: str, + metric_type: str, + value: float | int, + labels: dict[str, str], + *, + catalog: dict[str, dict[str, Any]] | None = None, +) -> None: + """Write one gauge point after catalog validation.""" + descriptor = validate_point(metric_type, labels, catalog) + + import google.cloud.monitoring_v3 as monitoring_v3 + + client = monitoring_v3.MetricServiceClient() + series = monitoring_v3.TimeSeries() + series.metric.type = metric_type + for key, val in labels.items(): + series.metric.labels[key] = val + series.resource.type = "global" + series.resource.labels["project_id"] = project_id + + now = time.time() + interval = monitoring_v3.TimeInterval({"end_time": {"seconds": int(now), "nanos": int((now % 1) * 1e9)}}) + point = monitoring_v3.Point({"interval": interval}) + if descriptor["value_type"] == "INT64": + point.value.int64_value = int(value) + else: + point.value.double_value = float(value) + series.points = [point] + client.create_time_series(name=f"projects/{project_id}", time_series=[series]) + + +def publish_gauge_safely( + project_id: str, + metric_type: str, + value: float | int, + labels: dict[str, str], + *, + catalog: dict[str, dict[str, Any]] | None = None, +) -> bool: + """Publish one point without ever propagating telemetry failure.""" + try: + publish_gauge(project_id, metric_type, value, labels, catalog=catalog) + return True + except Exception as exc: # noqa: BLE001 - telemetry must not break the caller + emit_event( + "metric_publish_failed", + severity="ERROR", + component="metrics", + check_name=labels.get("check_name"), + error_type=type(exc).__name__, + error_message=f"{metric_type}: {exc}", + ) + return False + + +def main() -> int: + parser = argparse.ArgumentParser(description="Atlas metric descriptor management") + parser.add_argument("--ensure-descriptors", action="store_true") + parser.add_argument("--project-id", default=None) + args = parser.parse_args() + if args.ensure_descriptors: + from atlas.config.settings import load_settings + + project_id = args.project_id or load_settings().gcp.project_id + created = ensure_descriptors(project_id) + print(json.dumps({"created": created, "catalog_size": len(load_catalog())})) + return 0 + parser.print_help() + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/observability/monitor.py b/src/atlas/observability/monitor.py new file mode 100644 index 0000000..2a94894 --- /dev/null +++ b/src/atlas/observability/monitor.py @@ -0,0 +1,595 @@ +"""Atlas observability monitor engine (Sprint 5, Phase 8). + +Evaluates system health independently of the business pipeline. Read-only +against all canonical data; its only writes are ``atlas_ops.monitor_evaluations`` +rows and Cloud Monitoring metric points. Every check: + +- uses bounded time windows from ``config/observability.yaml``, +- handles NO_DATA explicitly (and DISABLED when monitoring_enabled=false, + e.g. before intentional Composer teardown), +- emits one structured log event and one ``atlas/monitor/check_status`` + metric point (0=PASS 1=WARN 2=FAIL -1=NO_DATA -2=DISABLED), +- persists one durable evaluation row. + +Composer platform health is deliberately NOT re-implemented here: native +``composer.googleapis.com/environment/healthy`` metrics feed that alert +policy directly (ADR-011: never re-create native platform metrics). +""" + +from __future__ import annotations + +import json +import os +import uuid +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +import yaml + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.logging import emit_event +from atlas.observability.metrics import MONITOR_STATUS_VALUES, publish_gauge_safely +from atlas.ops.quality_results import MonitorEvaluationRecord, upsert_monitor_evaluation + +_CONFIG_PATH = Path(__file__).resolve().parents[3] / "config" / "observability.yaml" + +CHECK_NAMES = ( + "latest_run_state", + "freshness", + "missing_scheduled_run", + "telemetry_completeness", + "reconciliation", + "volume_deviation", + "rejection_rate", + "schema_drift", + "deployment_failure", + "rollback_failure", + "cost_anomaly", +) + + +def load_config(path: Path | None = None) -> dict[str, Any]: + """Load observability.yaml and apply drill overrides (file + env).""" + config = yaml.safe_load((path or _CONFIG_PATH).read_text(encoding="utf-8")) + overrides = dict(config.get("drill_overrides") or {}) + env_overrides = os.environ.get("ATLAS_OBSERVABILITY_OVERRIDES_JSON") + if env_overrides: + overrides.update(json.loads(env_overrides)) + for section, values in overrides.items(): + if isinstance(values, dict) and isinstance(config.get(section), dict): + config[section].update(values) + else: + config[section] = values + return config + + +@dataclass +class CheckResult: + """Outcome of one monitor check before persistence.""" + + check_name: str + status: str # PASS | WARN | FAIL | NO_DATA | DISABLED + severity: str = "INFO" + observed_value: float | None = None + threshold: float | None = None + details: dict[str, Any] | None = None + extra_metrics: list[tuple[str, float, dict[str, str]]] | None = None + + +def _rows(client: Any, sql: str) -> list[dict[str, Any]]: + return [dict(row) for row in client.query(sql).result()] + + +def check_latest_run_state(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + window = int(config.get("telemetry", {}).get("window_hours", 48)) + rows = _rows( + client, + f""" + SELECT status, pipeline_run_id, started_at, + TIMESTAMP_DIFF(COALESCE(completed_at, CURRENT_TIMESTAMP()), started_at, SECOND) AS duration_s + FROM `{project_id}.atlas_ops.pipeline_runs` + WHERE started_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window} HOUR) + ORDER BY started_at DESC LIMIT 1 + """, + ) + if not rows: + return CheckResult("latest_run_state", "NO_DATA", details={"window_hours": window}) + row = rows[0] + failed = row["status"] == "FAILED" + return CheckResult( + "latest_run_state", + "FAIL" if failed else ("WARN" if row["status"] == "RUNNING" else "PASS"), + severity="CRITICAL" if failed else "INFO", + observed_value=float(row["duration_s"] or 0), + details={"pipeline_run_id": row["pipeline_run_id"], "status": row["status"]}, + extra_metrics=[ + ( + "custom.googleapis.com/atlas/pipeline/run_duration_seconds", + float(row["duration_s"] or 0), + {"dag_id": "atlas_batch_pipeline", "status": "FAILED" if failed else "SUCCESS"}, + ) + ], + ) + + +def check_freshness(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + warn = float(config["freshness"]["warn_seconds"]) + fail = float(config["freshness"]["fail_seconds"]) + rows = _rows( + client, + f""" + SELECT TIMESTAMP_DIFF(CURRENT_TIMESTAMP(), MAX(completed_at), SECOND) AS age_s + FROM `{project_id}.atlas_ops.pipeline_runs` + WHERE status = 'SUCCESS' + AND completed_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 30 DAY) + """, + ) + age = rows[0]["age_s"] if rows and rows[0]["age_s"] is not None else None + if age is None: + return CheckResult("freshness", "NO_DATA", details={"reason": "no successful run in 30d"}) + status = "FAIL" if age >= fail else ("WARN" if age >= warn else "PASS") + return CheckResult( + "freshness", + status, + severity="CRITICAL" if status == "FAIL" else ("WARNING" if status == "WARN" else "INFO"), + observed_value=float(age), + threshold=fail if status == "FAIL" else warn, + extra_metrics=[ + ( + "custom.googleapis.com/atlas/pipeline/last_success_age_seconds", + float(age), + {"dag_id": "atlas_batch_pipeline"}, + ) + ], + ) + + +def check_missing_scheduled_run(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + grace = int(config["expected_schedule"]["grace_seconds"]) + expected_interval = 86400 + grace # daily schedule + grace + rows = _rows( + client, + f""" + SELECT TIMESTAMP_DIFF(CURRENT_TIMESTAMP(), MAX(started_at), SECOND) AS since_any_s + FROM `{project_id}.atlas_ops.pipeline_runs` + WHERE started_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 30 DAY) + """, + ) + since = rows[0]["since_any_s"] if rows and rows[0]["since_any_s"] is not None else None + if since is None: + # No runs at all in 30d: the DAG is deliberately paused between + # acceptance windows (default state), which is disabled runtime, not + # staleness. monitoring_enabled=false turns the whole monitor off. + return CheckResult( + "missing_scheduled_run", "NO_DATA", details={"reason": "no runs in 30d (DAG paused)"} + ) + status = "FAIL" if since > expected_interval else "PASS" + return CheckResult( + "missing_scheduled_run", + status, + severity="WARNING" if status == "FAIL" else "INFO", + observed_value=float(since), + threshold=float(expected_interval), + ) + + +def check_telemetry_completeness(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + window = int(config.get("telemetry", {}).get("window_hours", 48)) + rows = _rows( + client, + f""" + WITH latest AS ( + -- Only terminal runs: an in-progress run legitimately has missing + -- terminal task events, so evaluating it produces false positives + -- (defect found live during Sprint 5 acceptance). + SELECT pipeline_run_id FROM `{project_id}.atlas_ops.pipeline_runs` + WHERE started_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window} HOUR) + AND status IN ('SUCCESS', 'FAILED') + ORDER BY started_at DESC LIMIT 1 + ) + SELECT l.pipeline_run_id, + (SELECT COUNT(DISTINCT task_id) FROM `{project_id}.atlas_ops.task_events` te + WHERE te.pipeline_run_id = l.pipeline_run_id + AND te.event_type IN ('SUCCESS','FAILED','SKIPPED','UPSTREAM_FAILED')) AS terminal_tasks + FROM latest l + """, + ) + if not rows: + return CheckResult("telemetry_completeness", "NO_DATA", details={"window_hours": window}) + from atlas.ops.task_events import EXPECTED_TERMINAL_TASKS + + expected = len(EXPECTED_TERMINAL_TASKS) + missing = max(0, expected - int(rows[0]["terminal_tasks"])) + return CheckResult( + "telemetry_completeness", + "FAIL" if missing else "PASS", + severity="WARNING" if missing else "INFO", + observed_value=float(missing), + threshold=0.0, + details={"pipeline_run_id": rows[0]["pipeline_run_id"], "expected_tasks": expected}, + extra_metrics=[ + ( + "custom.googleapis.com/atlas/pipeline/telemetry_incomplete_count", + float(missing), + {"dag_id": "atlas_batch_pipeline"}, + ) + ], + ) + + +def check_reconciliation(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + window = int(config.get("reconciliation", {}).get("window_hours", 48)) + rows = _rows( + client, + f""" + WITH latest AS ( + SELECT pipeline_run_id FROM `{project_id}.atlas_ops.quality_results` + WHERE evaluated_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window} HOUR) + ORDER BY evaluated_at DESC LIMIT 1 + ) + SELECT l.pipeline_run_id, + (SELECT COUNTIF(status = 'FAIL') FROM `{project_id}.atlas_ops.quality_results` qr + WHERE qr.pipeline_run_id = l.pipeline_run_id) AS failed_checks + FROM latest l + """, + ) + if not rows: + return CheckResult("reconciliation", "NO_DATA", details={"window_hours": window}) + failed = int(rows[0]["failed_checks"]) + return CheckResult( + "reconciliation", + "FAIL" if failed else "PASS", + severity="CRITICAL" if failed else "INFO", + observed_value=float(failed), + threshold=0.0, + details={"pipeline_run_id": rows[0]["pipeline_run_id"]}, + extra_metrics=[("custom.googleapis.com/atlas/data/reconciliation_failure_count", float(failed), {})], + ) + + +def check_volume_deviation(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + cfg = config["volume"] + n = int(cfg["baseline_window_runs"]) + rows = _rows( + client, + f""" + WITH recent AS ( + SELECT rows_loaded, started_at + FROM `{project_id}.atlas_ops.pipeline_runs` + WHERE status = 'SUCCESS' AND rows_loaded IS NOT NULL + AND started_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 30 DAY) + ORDER BY started_at DESC LIMIT {n + 1} + ) + SELECT + (SELECT rows_loaded FROM recent ORDER BY started_at DESC LIMIT 1) AS latest_rows, + (SELECT APPROX_QUANTILES(rows_loaded, 2)[OFFSET(1)] + FROM (SELECT rows_loaded FROM recent ORDER BY started_at DESC LIMIT {n} OFFSET 1) + ) AS baseline_rows + """, + ) + latest = rows[0]["latest_rows"] if rows else None + baseline = rows[0]["baseline_rows"] if rows else None + if latest is None: + return CheckResult("volume_deviation", "NO_DATA", details={"reason": "no successful runs"}) + if baseline is None or baseline < int(cfg["min_baseline_rows"]): + return CheckResult( + "volume_deviation", + "NO_DATA", + observed_value=float(latest), + details={"reason": "insufficient baseline", "baseline": baseline}, + extra_metrics=[("custom.googleapis.com/atlas/data/raw_row_count", float(latest), {})], + ) + ratio = float(latest) / float(baseline) + deviation = abs(1.0 - ratio) + warn, fail = float(cfg["warn_deviation"]), float(cfg["fail_deviation"]) + status = "FAIL" if deviation >= fail else ("WARN" if deviation >= warn else "PASS") + return CheckResult( + "volume_deviation", + status, + severity="CRITICAL" if status == "FAIL" else ("WARNING" if status == "WARN" else "INFO"), + observed_value=ratio, + threshold=fail if status == "FAIL" else warn, + details={"latest_rows": latest, "baseline_rows": baseline}, + extra_metrics=[ + ("custom.googleapis.com/atlas/data/raw_row_count", float(latest), {}), + ("custom.googleapis.com/atlas/data/volume_deviation_ratio", ratio, {}), + ], + ) + + +def check_rejection_rate(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + cfg = config["rejection_rate"] + rows = _rows( + client, + f""" + SELECT rows_loaded, rows_accepted, rows_rejected + FROM `{project_id}.atlas_ops.pipeline_runs` + WHERE status = 'SUCCESS' AND rows_loaded IS NOT NULL AND rows_rejected IS NOT NULL + AND started_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL 30 DAY) + ORDER BY started_at DESC LIMIT 1 + """, + ) + if not rows or not rows[0]["rows_loaded"]: + return CheckResult("rejection_rate", "NO_DATA", details={"reason": "no volume data"}) + row = rows[0] + rate = float(row["rows_rejected"]) / float(row["rows_loaded"]) + warn, fail = float(cfg["warn"]), float(cfg["fail"]) + status = "FAIL" if rate >= fail else ("WARN" if rate >= warn else "PASS") + return CheckResult( + "rejection_rate", + status, + severity="CRITICAL" if status == "FAIL" else ("WARNING" if status == "WARN" else "INFO"), + observed_value=rate, + threshold=fail if status == "FAIL" else warn, + extra_metrics=[ + ("custom.googleapis.com/atlas/data/rejection_rate", rate, {}), + ("custom.googleapis.com/atlas/data/accepted_row_count", float(row["rows_accepted"] or 0), {}), + ("custom.googleapis.com/atlas/data/rejected_row_count", float(row["rows_rejected"]), {}), + ], + ) + + +def check_schema_drift(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + from atlas.observability.schema_drift import detect_drift, summarize + + findings = detect_drift( + client, + project_id, + allowed_new_fields=config.get("schema", {}).get("allowed_new_fields") or [], + ) + counts = summarize(findings) + if counts["BREAKING"]: + status, severity = "FAIL", "CRITICAL" + elif counts["WARNING"]: + status, severity = "WARN", "WARNING" + else: + status, severity = "PASS", "INFO" + sample = [f.__dict__ for f in findings if f.classification != "ALLOWED"][:10] + return CheckResult( + "schema_drift", + status, + severity=severity, + observed_value=float(counts["BREAKING"] + counts["WARNING"]), + threshold=0.0, + details={"counts": counts, "sample": sample}, + extra_metrics=[ + ( + "custom.googleapis.com/atlas/data/schema_drift_count", + float(count), + {"severity": label}, + ) + for label, count in ( + ("CRITICAL", counts["BREAKING"]), + ("WARNING", counts["WARNING"]), + ("INFO", counts["ALLOWED"]), + ) + ], + ) + + +def _latest_deployment( + client: Any, project_id: str, window_hours: int, deployment_type: str | None +) -> dict[str, Any] | None: + type_clause = f"AND deployment_type = '{deployment_type}'" if deployment_type else "" + rows = _rows( + client, + f""" + SELECT deployment_id, status, deployment_type, + TIMESTAMP_DIFF(COALESCE(completed_at, CURRENT_TIMESTAMP()), started_at, SECOND) AS duration_s + FROM `{project_id}.atlas_ops.deployments` + WHERE started_at >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window_hours} HOUR) + {type_clause} + ORDER BY started_at DESC LIMIT 1 + """, + ) + return rows[0] if rows else None + + +def check_deployment_failure(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + window = int(config.get("deployment", {}).get("window_hours", 168)) + row = _latest_deployment(client, project_id, window, None) + if row is None: + return CheckResult("deployment_failure", "NO_DATA", details={"window_hours": window}) + failed = row["status"] in {"FAILED", "ROLLBACK_FAILED"} + status_label = "FAILED" if failed else ("ROLLED_BACK" if row["status"] == "ROLLED_BACK" else "SUCCESS") + return CheckResult( + "deployment_failure", + "FAIL" if failed else "PASS", + severity="CRITICAL" if failed else "INFO", + observed_value=1.0 if failed else 0.0, + threshold=0.0, + details={"deployment_id": row["deployment_id"], "status": row["status"]}, + extra_metrics=[ + ("custom.googleapis.com/atlas/deployment/latest_failed", 1.0 if failed else 0.0, {}), + ( + "custom.googleapis.com/atlas/deployment/duration_seconds", + float(row["duration_s"] or 0), + {"status": status_label}, + ), + ], + ) + + +def check_rollback_failure(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + window = int(config.get("deployment", {}).get("window_hours", 168)) + row = _latest_deployment(client, project_id, window, "rollback") + if row is None: + return CheckResult("rollback_failure", "NO_DATA", details={"window_hours": window}) + failed = row["status"] == "ROLLBACK_FAILED" + return CheckResult( + "rollback_failure", + "FAIL" if failed else "PASS", + severity="CRITICAL" if failed else "INFO", + observed_value=1.0 if failed else 0.0, + threshold=0.0, + details={"deployment_id": row["deployment_id"], "status": row["status"]}, + ) + + +def check_cost_anomaly(client: Any, config: dict[str, Any], project_id: str) -> CheckResult: + cfg = config["cost"] + window_h = int(cfg["window_hours"]) + baseline_d = int(cfg["baseline_window_days"]) + rows = _rows( + client, + f""" + SELECT + SUM(IF(creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window_h} HOUR), + total_bytes_billed, 0)) AS window_bytes, + SUM(IF(creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window_h} HOUR), + 1, 0)) AS window_jobs, + SAFE_DIVIDE( + SUM(IF(creation_time < TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {window_h} HOUR), + total_bytes_billed, 0)), + {baseline_d}) AS baseline_daily_bytes + FROM `{project_id}.region-us.INFORMATION_SCHEMA.JOBS` + WHERE creation_time >= TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {baseline_d + 1} DAY) + AND ('application', 'atlas') IN (SELECT (key, value) FROM UNNEST(labels)) + AND statement_type != 'SCRIPT' + """, + ) + if not rows: + return CheckResult("cost_anomaly", "NO_DATA") + window_bytes = float(rows[0]["window_bytes"] or 0) + window_jobs = float(rows[0]["window_jobs"] or 0) + baseline = float(rows[0]["baseline_daily_bytes"] or 0) + metrics: list[tuple[str, float, dict[str, str]]] = [ + ("custom.googleapis.com/atlas/cost/bigquery_bytes_billed", window_bytes, {}), + ("custom.googleapis.com/atlas/cost/bigquery_job_count", window_jobs, {}), + ] + if window_bytes < float(cfg["min_bytes_billed"]): + return CheckResult( + "cost_anomaly", + "PASS", + observed_value=window_bytes, + details={"reason": "below absolute floor", "baseline_daily_bytes": baseline}, + extra_metrics=metrics, + ) + if baseline <= 0: + return CheckResult( + "cost_anomaly", + "NO_DATA", + observed_value=window_bytes, + details={"reason": "no baseline"}, + extra_metrics=metrics, + ) + ratio = window_bytes / baseline + warn, fail = float(cfg["warn_ratio"]), float(cfg["fail_ratio"]) + status = "FAIL" if ratio >= fail else ("WARN" if ratio >= warn else "PASS") + return CheckResult( + "cost_anomaly", + status, + severity="CRITICAL" if status == "FAIL" else ("WARNING" if status == "WARN" else "INFO"), + observed_value=ratio, + threshold=fail if status == "FAIL" else warn, + details={"window_bytes": window_bytes, "baseline_daily_bytes": baseline}, + extra_metrics=metrics, + ) + + +CHECKS = { + "latest_run_state": check_latest_run_state, + "freshness": check_freshness, + "missing_scheduled_run": check_missing_scheduled_run, + "telemetry_completeness": check_telemetry_completeness, + "reconciliation": check_reconciliation, + "volume_deviation": check_volume_deviation, + "rejection_rate": check_rejection_rate, + "schema_drift": check_schema_drift, + "deployment_failure": check_deployment_failure, + "rollback_failure": check_rollback_failure, + "cost_anomaly": check_cost_anomaly, +} + + +def run_monitor( + *, + settings: AtlasSettings | None = None, + client: Any | None = None, + config: dict[str, Any] | None = None, + window_start: str | None = None, + window_end: str | None = None, + persist: bool = True, + publish: bool = True, +) -> list[CheckResult]: + """Run every monitor check; persist evaluations and publish metrics.""" + settings = settings or load_settings() + config = config or load_config() + environment = config.get("environment", "atlas-dev") + mode = config.get("runtime_mode", "normal") + enabled = bool(config.get("monitoring_enabled", True)) + now = datetime.now(tz=UTC).isoformat() + window_end = window_end or now + + if client is None: + from atlas.observability.cost import labeled_bigquery_client + + client = labeled_bigquery_client(settings.gcp.project_id, "monitor") + + results: list[CheckResult] = [] + for name, func in CHECKS.items(): + if not enabled: + result = CheckResult(name, "DISABLED", details={"monitoring_enabled": False}) + else: + try: + result = func(client, config, settings.gcp.project_id) + except Exception as exc: # noqa: BLE001 - one broken check must not hide the rest + result = CheckResult( + name, + "NO_DATA", + severity="WARNING", + details={"error_type": type(exc).__name__, "error": str(exc)[:300]}, + ) + results.append(result) + + emit_event( + "monitor_evaluation", + severity="ERROR" if result.status == "FAIL" else "INFO", + component="monitor", + environment=environment, + check_name=result.check_name, + status=result.status, + observed_value=result.observed_value, + threshold=result.threshold, + details=result.details, + ) + if persist: + evaluation = MonitorEvaluationRecord( + evaluation_id=f"{result.check_name}-{uuid.uuid4().hex[:12]}", + check_name=result.check_name, + environment=environment, + status=result.status, + severity=result.severity, + observed_value=result.observed_value, + threshold=result.threshold, + incident_key=f"atlas-{result.check_name}", + source="atlas_observability_monitor", + evaluated_at=now, + window_start=window_start, + window_end=window_end, + details_json=json.dumps(result.details, default=repr) if result.details else None, + ) + try: + upsert_monitor_evaluation(evaluation, settings, client=client) + except Exception as exc: # noqa: BLE001 - visible degradation, no crash + emit_event( + "monitor_evaluation_write_failed", + severity="ERROR", + component="monitor", + check_name=result.check_name, + error_type=type(exc).__name__, + error_message=str(exc), + ) + if publish: + base_labels = {"environment": environment, "mode": mode} + publish_gauge_safely( + settings.gcp.project_id, + "custom.googleapis.com/atlas/monitor/check_status", + MONITOR_STATUS_VALUES[result.status], + {**base_labels, "check_name": result.check_name}, + ) + for metric_type, value, extra in result.extra_metrics or []: + publish_gauge_safely(settings.gcp.project_id, metric_type, value, {**base_labels, **extra}) + return results diff --git a/src/atlas/observability/schema_drift.py b/src/atlas/observability/schema_drift.py new file mode 100644 index 0000000..3768969 --- /dev/null +++ b/src/atlas/observability/schema_drift.py @@ -0,0 +1,240 @@ +"""Schema-drift detection for governed Atlas tables (Sprint 5, Phase 9). + +An expected-schema manifest (``observability/schema/expected-schemas.json``, +generated from the live governed tables and reviewed into Git) is compared +against ``INFORMATION_SCHEMA.COLUMNS``. Findings are classified: + +- ALLOWED — configured new nullable field (``allowed_new_fields``) or + metadata-only difference that does not affect consumers +- WARNING — unapproved new nullable field, partition/clustering metadata + drift, missing description-level metadata +- BREAKING — removed field, incompatible type change, required field made + nullable/unavailable, partition-field change, missing table + +Detection is read-only. Drills use fixture datasets or expected-manifest +overrides; canonical tables are never mutated to prove this monitor. +""" + +from __future__ import annotations + +import argparse +import json +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from google.cloud import bigquery + +_MANIFEST_PATH = Path(__file__).resolve().parents[3] / "observability" / "schema" / "expected-schemas.json" + +# Governed tables monitored for drift (dataset.table). +DEFAULT_MONITORED_TABLES = ( + "atlas_raw.events", + "atlas_core.fct_events", + "atlas_core.dim_users", + "atlas_core.dim_countries", + "atlas_marts.mart_daily_event_metrics", + "atlas_ops.pipeline_runs", + "atlas_ops.deployments", + "atlas_ops.schema_migrations", + "atlas_ops.task_events", + "atlas_ops.quality_results", + "atlas_ops.monitor_evaluations", + "atlas_ops.recovery_actions", +) + + +@dataclass(frozen=True) +class DriftFinding: + """One classified schema difference.""" + + table: str + column: str | None + classification: str # ALLOWED | WARNING | BREAKING + kind: str + detail: str + + +def load_manifest(path: Path | None = None) -> dict[str, Any]: + return json.loads((path or _MANIFEST_PATH).read_text(encoding="utf-8")) + + +def fetch_live_schema( + client: bigquery.Client, + project_id: str, + dataset_id: str, +) -> dict[str, dict[str, dict[str, str]]]: + """Return {table: {column: {data_type, is_nullable}}} for one dataset.""" + sql = f""" + SELECT table_name, column_name, data_type, is_nullable, is_partitioning_column + FROM `{project_id}.{dataset_id}.INFORMATION_SCHEMA.COLUMNS` + ORDER BY table_name, ordinal_position + """ + tables: dict[str, dict[str, dict[str, str]]] = {} + for row in client.query(sql).result(): + tables.setdefault(row["table_name"], {})[row["column_name"]] = { + "data_type": row["data_type"], + "is_nullable": row["is_nullable"], + "is_partitioning_column": row["is_partitioning_column"], + } + return tables + + +def compare_table( + table: str, + expected: dict[str, Any], + live_columns: dict[str, dict[str, str]] | None, + allowed_new_fields: list[str] | None = None, +) -> list[DriftFinding]: + """Classify differences between one expected table schema and live columns.""" + findings: list[DriftFinding] = [] + allowed_new = set(allowed_new_fields or []) + + if live_columns is None: + return [DriftFinding(table, None, "BREAKING", "missing_table", "table not found in live dataset")] + + expected_columns: dict[str, Any] = expected["columns"] + for name, spec in expected_columns.items(): + live = live_columns.get(name) + if live is None: + findings.append(DriftFinding(table, name, "BREAKING", "removed_field", "expected column missing")) + continue + if live["data_type"] != spec["data_type"]: + findings.append( + DriftFinding( + table, + name, + "BREAKING", + "type_change", + f"expected {spec['data_type']}, live {live['data_type']}", + ) + ) + if spec["is_nullable"] == "NO" and live["is_nullable"] == "YES": + findings.append( + DriftFinding( + table, name, "BREAKING", "required_made_nullable", "REQUIRED column now NULLABLE" + ) + ) + expected_partition = expected.get("partition_column") + if expected_partition == name and live.get("is_partitioning_column") != "YES": + findings.append( + DriftFinding(table, name, "BREAKING", "partition_change", "expected partition column lost") + ) + + for name, live in live_columns.items(): + if name in expected_columns: + continue + if f"{table}.{name}" in allowed_new or name in allowed_new: + findings.append( + DriftFinding(table, name, "ALLOWED", "approved_new_field", "configured additive field") + ) + elif live["is_nullable"] == "YES": + findings.append( + DriftFinding( + table, name, "WARNING", "unapproved_new_field", "new nullable column not in manifest" + ) + ) + else: + findings.append( + DriftFinding( + table, name, "BREAKING", "unapproved_required_field", "new REQUIRED column breaks writers" + ) + ) + return findings + + +def detect_drift( + client: bigquery.Client, + project_id: str, + *, + manifest: dict[str, Any] | None = None, + allowed_new_fields: list[str] | None = None, +) -> list[DriftFinding]: + """Compare every manifest table against live INFORMATION_SCHEMA.""" + manifest = manifest or load_manifest() + findings: list[DriftFinding] = [] + live_cache: dict[str, dict[str, dict[str, dict[str, str]]]] = {} + for table, expected in manifest["tables"].items(): + dataset_id, table_name = table.split(".", 1) + if dataset_id not in live_cache: + live_cache[dataset_id] = fetch_live_schema(client, project_id, dataset_id) + findings.extend( + compare_table( + table, + expected, + live_cache[dataset_id].get(table_name), + allowed_new_fields, + ) + ) + return findings + + +def summarize(findings: list[DriftFinding]) -> dict[str, int]: + counts = {"ALLOWED": 0, "WARNING": 0, "BREAKING": 0} + for finding in findings: + counts[finding.classification] += 1 + return counts + + +def generate_manifest( + client: bigquery.Client, + project_id: str, + tables: tuple[str, ...] = DEFAULT_MONITORED_TABLES, +) -> dict[str, Any]: + """Snapshot live governed schemas into manifest form (review before commit).""" + manifest: dict[str, Any] = {"generated_from": project_id, "tables": {}} + live_cache: dict[str, dict[str, dict[str, dict[str, str]]]] = {} + for table in tables: + dataset_id, table_name = table.split(".", 1) + if dataset_id not in live_cache: + live_cache[dataset_id] = fetch_live_schema(client, project_id, dataset_id) + columns = live_cache[dataset_id].get(table_name) + if columns is None: + raise RuntimeError(f"table {table} not found while generating manifest") + partition_column = next( + (c for c, spec in columns.items() if spec.get("is_partitioning_column") == "YES"), + None, + ) + manifest["tables"][table] = { + "partition_column": partition_column, + "columns": { + name: {"data_type": spec["data_type"], "is_nullable": spec["is_nullable"]} + for name, spec in columns.items() + }, + } + return manifest + + +def main() -> int: + parser = argparse.ArgumentParser(description="Atlas schema-drift tooling") + parser.add_argument("--generate", action="store_true", help="snapshot live schemas to stdout") + parser.add_argument("--check", action="store_true", help="compare manifest against live schemas") + parser.add_argument("--project-id", default=None) + args = parser.parse_args() + + from atlas.config.settings import load_settings + from atlas.observability.cost import labeled_bigquery_client + + project_id = args.project_id or load_settings().gcp.project_id + client = labeled_bigquery_client(project_id, "monitor") + if args.generate: + print(json.dumps(generate_manifest(client, project_id), indent=2, sort_keys=True)) + return 0 + if args.check: + findings = detect_drift(client, project_id) + print( + json.dumps( + { + "summary": summarize(findings), + "findings": [finding.__dict__ for finding in findings], + }, + indent=2, + ) + ) + return 1 if any(f.classification == "BREAKING" for f in findings) else 0 + parser.print_help() + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/ops/__init__.py b/src/atlas/ops/__init__.py new file mode 100644 index 0000000..306aac8 --- /dev/null +++ b/src/atlas/ops/__init__.py @@ -0,0 +1,21 @@ +"""Operational audit and resource helpers for Project Atlas.""" + +from atlas.ops.audit import ( + PipelineRunRecord, + finalize_pipeline_run, + query_pipeline_run, + sanitize_error_message, + start_pipeline_run, + upsert_pipeline_run, +) +from atlas.ops.resources import ensure_audit_resources + +__all__ = [ + "PipelineRunRecord", + "ensure_audit_resources", + "finalize_pipeline_run", + "query_pipeline_run", + "sanitize_error_message", + "start_pipeline_run", + "upsert_pipeline_run", +] diff --git a/src/atlas/ops/audit.py b/src/atlas/ops/audit.py new file mode 100644 index 0000000..5b6ed6e --- /dev/null +++ b/src/atlas/ops/audit.py @@ -0,0 +1,336 @@ +"""BigQuery operational audit records for orchestrated pipeline runs.""" + +from __future__ import annotations + +import json +import os +import re +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings, table_fqn +from atlas.observability.cost import labeled_bigquery_client + +ALLOWED_STATUSES = frozenset({"RUNNING", "SUCCESS", "FAILED", "PARTIAL"}) +_SECRET_PATTERNS = ( + re.compile(r"BEGIN PRIVATE KEY"), + # Redact the value that follows a private_key/client_email field, not just + # the field name, so quoted JSON payloads cannot leak the secret itself. + re.compile(r"private_key(_id)?\"?\s*[:=]\s*\"?[^\",}\s]*", re.IGNORECASE), + re.compile(r"private_key(_id)?", re.IGNORECASE), + re.compile(r"client_email\"?\s*[:=]\s*\"?[^\",}\s]*", re.IGNORECASE), + re.compile(r"client_email", re.IGNORECASE), + re.compile(r"AIza[0-9A-Za-z\-_]{35}"), + re.compile(r"Bearer\s+[A-Za-z0-9\-._~+/]+=*", re.IGNORECASE), +) + + +@dataclass(frozen=True) +class PipelineRunRecord: + """One durable audit row for an Airflow execution.""" + + pipeline_run_id: str + batch_id: str + airflow_run_id: str + dag_id: str + processing_date: str + started_at: str + completed_at: str | None + status: str + attempt_number: int + gcs_uri: str | None = None + rows_generated: int | None = None + rows_loaded: int | None = None + rows_accepted: int | None = None + rows_rejected: int | None = None + fact_rows: int | None = None + mart_event_count: int | None = None + failed_task_id: str | None = None + error_type: str | None = None + error_message: str | None = None + + +def sanitize_error_message(message: str | None, *, max_length: int = 2000) -> str | None: + """Remove sensitive values and truncate operational error text.""" + if not message: + return None + sanitized = message + for pattern in _SECRET_PATTERNS: + sanitized = pattern.sub("[REDACTED]", sanitized) + sanitized = sanitized.strip() + if len(sanitized) > max_length: + return sanitized[: max_length - 3] + "..." + return sanitized or None + + +def _table_fqn(settings: AtlasSettings) -> str: + return f"{settings.gcp.project_id}.atlas_ops.pipeline_runs" + + +def upsert_pipeline_run( + record: PipelineRunRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Merge one pipeline run audit row keyed by pipeline_run_id.""" + if record.status not in ALLOWED_STATUSES: + raise ValueError(f"Unsupported audit status: {record.status}") + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + query = f""" + MERGE `{_table_fqn(settings)}` AS target + USING ( + SELECT + @pipeline_run_id AS pipeline_run_id, + @batch_id AS batch_id, + @airflow_run_id AS airflow_run_id, + @dag_id AS dag_id, + DATE(@processing_date) AS processing_date, + TIMESTAMP(@started_at) AS started_at, + TIMESTAMP(@completed_at) AS completed_at, + @status AS status, + @attempt_number AS attempt_number, + @gcs_uri AS gcs_uri, + @rows_generated AS rows_generated, + @rows_loaded AS rows_loaded, + @rows_accepted AS rows_accepted, + @rows_rejected AS rows_rejected, + @fact_rows AS fact_rows, + @mart_event_count AS mart_event_count, + @failed_task_id AS failed_task_id, + @error_type AS error_type, + @error_message AS error_message + ) AS source + ON target.pipeline_run_id = source.pipeline_run_id + WHEN MATCHED THEN UPDATE SET + batch_id = source.batch_id, + airflow_run_id = source.airflow_run_id, + dag_id = source.dag_id, + processing_date = source.processing_date, + completed_at = source.completed_at, + status = source.status, + attempt_number = source.attempt_number, + gcs_uri = source.gcs_uri, + rows_generated = source.rows_generated, + rows_loaded = source.rows_loaded, + rows_accepted = source.rows_accepted, + rows_rejected = source.rows_rejected, + fact_rows = source.fact_rows, + mart_event_count = source.mart_event_count, + failed_task_id = source.failed_task_id, + error_type = source.error_type, + error_message = source.error_message, + updated_at = CURRENT_TIMESTAMP() + WHEN NOT MATCHED THEN INSERT ( + pipeline_run_id, + batch_id, + airflow_run_id, + dag_id, + processing_date, + started_at, + completed_at, + status, + attempt_number, + gcs_uri, + rows_generated, + rows_loaded, + rows_accepted, + rows_rejected, + fact_rows, + mart_event_count, + failed_task_id, + error_type, + error_message, + created_at, + updated_at + ) VALUES ( + source.pipeline_run_id, + source.batch_id, + source.airflow_run_id, + source.dag_id, + source.processing_date, + source.started_at, + source.completed_at, + source.status, + source.attempt_number, + source.gcs_uri, + source.rows_generated, + source.rows_loaded, + source.rows_accepted, + source.rows_rejected, + source.fact_rows, + source.mart_event_count, + source.failed_task_id, + source.error_type, + source.error_message, + CURRENT_TIMESTAMP(), + CURRENT_TIMESTAMP() + ) + """ + params = asdict(record) + params["error_message"] = sanitize_error_message(record.error_message) + job_config = bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter(name, _param_type(name, value), value) + for name, value in params.items() + ] + ) + bq_client.query(query, job_config=job_config).result() + + +# Integer-typed audit columns: a NULL value must still carry the INT64 type so the +# MERGE source column matches the target schema (a STRING NULL cannot be assigned +# to an INT64 column in BigQuery). +_INT64_FIELDS = frozenset( + { + "attempt_number", + "rows_generated", + "rows_loaded", + "rows_accepted", + "rows_rejected", + "fact_rows", + "mart_event_count", + } +) + + +def _param_type(name: str, value: Any) -> str: + if isinstance(value, bool): + return "BOOL" + if isinstance(value, int): + return "INT64" + if name in _INT64_FIELDS: + return "INT64" + return "STRING" + + +def start_pipeline_run( + *, + pipeline_run_id: str, + batch_id: str, + airflow_run_id: str, + dag_id: str, + processing_date: str, + attempt_number: int = 1, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> PipelineRunRecord: + """Create or refresh a RUNNING audit row.""" + started_at = datetime.now(tz=UTC).isoformat() + record = PipelineRunRecord( + pipeline_run_id=pipeline_run_id, + batch_id=batch_id, + airflow_run_id=airflow_run_id, + dag_id=dag_id, + processing_date=processing_date, + started_at=started_at, + completed_at=None, + status="RUNNING", + attempt_number=attempt_number, + ) + upsert_pipeline_run(record, settings=settings, client=client) + return record + + +def finalize_pipeline_run( + record: PipelineRunRecord, + *, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> PipelineRunRecord: + """Upsert the terminal audit state for one execution.""" + completed = record.completed_at or datetime.now(tz=UTC).isoformat() + final_record = PipelineRunRecord(**{**asdict(record), "completed_at": completed}) + upsert_pipeline_run(final_record, settings=settings, client=client) + return final_record + + +def query_pipeline_run( + pipeline_run_id: str, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> dict[str, Any] | None: + """Fetch one audit row as a dictionary.""" + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + query = f""" + SELECT * + FROM `{_table_fqn(settings)}` + WHERE pipeline_run_id = @pipeline_run_id + LIMIT 1 + """ + rows = list( + bq_client.query( + query, + job_config=bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter("pipeline_run_id", "STRING", pipeline_run_id)] + ), + ).result() + ) + if not rows: + return None + row = dict(rows[0].items()) + for key, value in row.items(): + if hasattr(value, "isoformat"): + row[key] = value.isoformat() + return row + + +def collect_batch_metrics( + batch_id: str, + *, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, + dbt_dataset: str | None = None, +) -> dict[str, int]: + """Return best-effort batch-scoped row counts for the audit record. + + Each COUNT is guarded independently: a missing relation or query error + yields an absent metric rather than raising, so populating audit volumes can + never fail the finalizer. int_rejected_events lives in the quarantine schema; + the other relations follow the {dbt_dataset}_{folder} layout. + """ + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + dataset = dbt_dataset or os.environ.get("ATLAS_DBT_DATASET", "atlas") + project = settings.gcp.project_id + relations = { + "rows_loaded": table_fqn(settings), + "rows_accepted": f"{project}.{dataset}_intermediate.int_accepted_events", + "rows_rejected": f"{project}.{dataset}_quarantine.int_rejected_events", + "fact_rows": f"{project}.{dataset}_core.fct_events", + } + metrics: dict[str, int] = {} + for metric, relation in relations.items(): + try: + rows = list( + bq_client.query( + f"SELECT COUNT(1) AS n FROM `{relation}` WHERE batch_id = @batch_id", + job_config=bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter("batch_id", "STRING", batch_id)] + ), + ).result() + ) + metrics[metric] = int(rows[0]["n"]) if rows else 0 + except Exception: # noqa: BLE001 - metrics are best-effort observability + continue + return metrics + + +def write_local_run_summary( + pipeline_run_id: str, + payload: dict[str, Any], + settings: AtlasSettings | None = None, +) -> str: + """Persist detailed task-level evidence beside Airflow logs.""" + settings = settings or load_settings() + summary_dir = settings.logging.log_dir / "airflow" / pipeline_run_id + summary_dir.mkdir(parents=True, exist_ok=True) + summary_path = summary_dir / "run-summary.json" + summary_path.write_text(json.dumps(payload, indent=2), encoding="utf-8") + return str(summary_path) diff --git a/src/atlas/ops/deployments.py b/src/atlas/ops/deployments.py new file mode 100644 index 0000000..3b52466 --- /dev/null +++ b/src/atlas/ops/deployments.py @@ -0,0 +1,300 @@ +"""Durable deployment and rollback audit records (Sprint 4, Phase 10). + +Grain: one row per deployment or rollback attempt in +``atlas_ops.deployments``, keyed by ``deployment_id`` and written with +idempotent parameterized MERGE statements. Deployments are a separate grain +from ``atlas_ops.pipeline_runs``: a deployment may reference the smoke +pipeline run it triggered, but never duplicates its row. +""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.cost import labeled_bigquery_client +from atlas.ops.audit import sanitize_error_message + +ALLOWED_DEPLOYMENT_STATUSES = frozenset( + {"RUNNING", "SUCCESS", "FAILED", "ROLLING_BACK", "ROLLED_BACK", "ROLLBACK_FAILED"} +) +ALLOWED_DEPLOYMENT_TYPES = frozenset({"deploy", "rollback"}) + + +@dataclass(frozen=True) +class DeploymentRecord: + """One durable audit row for a deployment or rollback attempt.""" + + deployment_id: str + git_sha: str + environment: str + deployment_type: str + started_at: str + status: str + git_ref: str | None = None + release_tag: str | None = None + workflow_run_id: str | None = None + actor: str | None = None + completed_at: str | None = None + artifact_uri: str | None = None + artifact_checksum: str | None = None + composer_environment: str | None = None + composer_region: str | None = None + smoke_pipeline_run_id: str | None = None + previous_git_sha: str | None = None + migration_count: int | None = None + failure_stage: str | None = None + error_type: str | None = None + error_summary: str | None = None + + +def _table_fqn(settings: AtlasSettings) -> str: + return f"{settings.gcp.project_id}.atlas_ops.deployments" + + +_INT64_FIELDS = frozenset({"migration_count"}) + + +def _param_type(name: str, value: Any) -> str: + if isinstance(value, bool): + return "BOOL" + if isinstance(value, int) or name in _INT64_FIELDS: + return "INT64" + return "STRING" + + +def upsert_deployment( + record: DeploymentRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Merge one deployment audit row keyed by deployment_id.""" + if record.status not in ALLOWED_DEPLOYMENT_STATUSES: + raise ValueError(f"Unsupported deployment status: {record.status}") + if record.deployment_type not in ALLOWED_DEPLOYMENT_TYPES: + raise ValueError(f"Unsupported deployment type: {record.deployment_type}") + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "deployment") + query = f""" + MERGE `{_table_fqn(settings)}` AS target + USING ( + SELECT + @deployment_id AS deployment_id, + @git_sha AS git_sha, + @git_ref AS git_ref, + @release_tag AS release_tag, + @environment AS environment, + @workflow_run_id AS workflow_run_id, + @actor AS actor, + @deployment_type AS deployment_type, + TIMESTAMP(@started_at) AS started_at, + TIMESTAMP(@completed_at) AS completed_at, + @status AS status, + @artifact_uri AS artifact_uri, + @artifact_checksum AS artifact_checksum, + @composer_environment AS composer_environment, + @composer_region AS composer_region, + @smoke_pipeline_run_id AS smoke_pipeline_run_id, + @previous_git_sha AS previous_git_sha, + @migration_count AS migration_count, + @failure_stage AS failure_stage, + @error_type AS error_type, + @error_summary AS error_summary + ) AS source + ON target.deployment_id = source.deployment_id + WHEN MATCHED THEN UPDATE SET + git_sha = source.git_sha, + git_ref = source.git_ref, + release_tag = source.release_tag, + environment = source.environment, + workflow_run_id = source.workflow_run_id, + actor = source.actor, + deployment_type = source.deployment_type, + completed_at = source.completed_at, + status = source.status, + artifact_uri = source.artifact_uri, + artifact_checksum = source.artifact_checksum, + composer_environment = source.composer_environment, + composer_region = source.composer_region, + smoke_pipeline_run_id = source.smoke_pipeline_run_id, + previous_git_sha = source.previous_git_sha, + migration_count = source.migration_count, + failure_stage = source.failure_stage, + error_type = source.error_type, + error_summary = source.error_summary, + updated_at = CURRENT_TIMESTAMP() + WHEN NOT MATCHED THEN INSERT ( + deployment_id, git_sha, git_ref, release_tag, environment, + workflow_run_id, actor, deployment_type, started_at, completed_at, + status, artifact_uri, artifact_checksum, composer_environment, + composer_region, smoke_pipeline_run_id, previous_git_sha, + migration_count, failure_stage, error_type, error_summary, + created_at, updated_at + ) VALUES ( + source.deployment_id, source.git_sha, source.git_ref, source.release_tag, + source.environment, source.workflow_run_id, source.actor, + source.deployment_type, source.started_at, source.completed_at, + source.status, source.artifact_uri, source.artifact_checksum, + source.composer_environment, source.composer_region, + source.smoke_pipeline_run_id, source.previous_git_sha, + source.migration_count, source.failure_stage, source.error_type, + source.error_summary, CURRENT_TIMESTAMP(), CURRENT_TIMESTAMP() + ) + """ + params = asdict(record) + params["error_summary"] = sanitize_error_message(record.error_summary) + job_config = bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter(name, _param_type(name, value), value) + for name, value in params.items() + ] + ) + bq_client.query(query, job_config=job_config).result() + + +def start_deployment( + *, + deployment_id: str, + git_sha: str, + environment: str, + deployment_type: str = "deploy", + git_ref: str | None = None, + release_tag: str | None = None, + workflow_run_id: str | None = None, + actor: str | None = None, + artifact_uri: str | None = None, + artifact_checksum: str | None = None, + previous_git_sha: str | None = None, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> DeploymentRecord: + """Create or refresh a RUNNING (or ROLLING_BACK) deployment row.""" + status = "ROLLING_BACK" if deployment_type == "rollback" else "RUNNING" + record = DeploymentRecord( + deployment_id=deployment_id, + git_sha=git_sha, + environment=environment, + deployment_type=deployment_type, + started_at=datetime.now(tz=UTC).isoformat(), + status=status, + git_ref=git_ref, + release_tag=release_tag, + workflow_run_id=workflow_run_id, + actor=actor, + artifact_uri=artifact_uri, + artifact_checksum=artifact_checksum, + previous_git_sha=previous_git_sha, + ) + upsert_deployment(record, settings=settings, client=client) + return record + + +def finalize_deployment( + record: DeploymentRecord, + *, + status: str, + failure_stage: str | None = None, + error_type: str | None = None, + error_summary: str | None = None, + smoke_pipeline_run_id: str | None = None, + composer_environment: str | None = None, + composer_region: str | None = None, + migration_count: int | None = None, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> DeploymentRecord: + """Upsert the terminal state for one deployment attempt. Idempotent.""" + final = DeploymentRecord( + **{ + **asdict(record), + "completed_at": record.completed_at or datetime.now(tz=UTC).isoformat(), + "status": status, + "failure_stage": failure_stage, + "error_type": error_type, + "error_summary": error_summary, + "smoke_pipeline_run_id": smoke_pipeline_run_id or record.smoke_pipeline_run_id, + "composer_environment": composer_environment or record.composer_environment, + "composer_region": composer_region or record.composer_region, + "migration_count": migration_count if migration_count is not None else record.migration_count, + } + ) + upsert_deployment(final, settings=settings, client=client) + return final + + +def query_deployment( + deployment_id: str, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> dict[str, Any] | None: + """Fetch one deployment audit row as a dictionary.""" + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "deployment") + query = f""" + SELECT * FROM `{_table_fqn(settings)}` + WHERE deployment_id = @deployment_id + LIMIT 1 + """ + rows = list( + bq_client.query( + query, + job_config=bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter("deployment_id", "STRING", deployment_id)] + ), + ).result() + ) + if not rows: + return None + row = dict(rows[0].items()) + for key, value in row.items(): + if hasattr(value, "isoformat"): + row[key] = value.isoformat() + return row + + +def latest_successful_deployment( + environment: str, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, + exclude_git_sha: str | None = None, +) -> dict[str, Any] | None: + """Return the most recent SUCCESS (or ROLLED_BACK target) deployment row. + + Used by rollback to select the restore target: the newest deployment whose + artifacts were fully validated, optionally excluding the currently broken SHA. + """ + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "deployment") + query = f""" + SELECT * FROM `{_table_fqn(settings)}` + WHERE environment = @environment + AND status = 'SUCCESS' + AND (@exclude_git_sha IS NULL OR git_sha != @exclude_git_sha) + ORDER BY completed_at DESC + LIMIT 1 + """ + rows = list( + bq_client.query( + query, + job_config=bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter("environment", "STRING", environment), + bigquery.ScalarQueryParameter("exclude_git_sha", "STRING", exclude_git_sha), + ] + ), + ).result() + ) + if not rows: + return None + row = dict(rows[0].items()) + for key, value in row.items(): + if hasattr(value, "isoformat"): + row[key] = value.isoformat() + return row diff --git a/src/atlas/ops/finalizer.py b/src/atlas/ops/finalizer.py new file mode 100644 index 0000000..6a33569 --- /dev/null +++ b/src/atlas/ops/finalizer.py @@ -0,0 +1,30 @@ +"""Run summary reconciliation for orchestrated pipeline finalization.""" + +from __future__ import annotations + +from typing import Any + + +def reconcile_run_summary( + local_summary: dict[str, Any], + audit_row: dict[str, Any] | None, +) -> dict[str, Any]: + """Compare local JSON summary with the BigQuery audit row.""" + mismatches: list[str] = [] + if audit_row is None: + mismatches.append("missing BigQuery audit row") + else: + for key in ("pipeline_run_id", "batch_id", "status"): + if local_summary.get(key) != audit_row.get(key): + mismatches.append(f"{key} mismatch") + return { + "reconciled": not mismatches, + "mismatches": mismatches, + "local_status": local_summary.get("status"), + "audit_status": None if audit_row is None else audit_row.get("status"), + } + + +def finalizer_should_fail(summary: dict[str, Any]) -> bool: + """Return True when the all-done finalizer must raise to fail the DAG.""" + return summary.get("status") in {"FAILED", "PARTIAL"} diff --git a/src/atlas/ops/migrations.py b/src/atlas/ops/migrations.py new file mode 100644 index 0000000..ecac391 --- /dev/null +++ b/src/atlas/ops/migrations.py @@ -0,0 +1,288 @@ +"""Ledger-driven additive schema migrations for Project Atlas. + +Migrations are declared in ``sql/migrations/manifest.txt`` and applied in +manifest order. Every applied migration is recorded in +``atlas_ops.schema_migrations`` with the SHA-256 checksum of its source file: + +- re-applying a recorded, unchanged migration is an idempotent no-op; +- a recorded migration whose file content changed fails hard; +- a failed migration is recorded as FAILED and blocks promotion. + +Destructive changes are never reversed automatically (ADR-010). +""" + +from __future__ import annotations + +import hashlib +import os +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.cost import labeled_bigquery_client +from atlas.ops.audit import sanitize_error_message + +SQL_ROOT = Path(__file__).resolve().parents[3] / "sql" +MANIFEST_PATH = SQL_ROOT / "migrations" / "manifest.txt" + +_LEDGER_DDL = """ +CREATE SCHEMA IF NOT EXISTS `{project_id}.atlas_ops` OPTIONS (location = '{location}'); +CREATE TABLE IF NOT EXISTS `{project_id}.atlas_ops.schema_migrations` ( + migration_id STRING NOT NULL, + migration_checksum STRING NOT NULL, + git_sha STRING, + applied_at TIMESTAMP NOT NULL, + workflow_run_id STRING, + applied_by STRING, + status STRING NOT NULL, + error_summary STRING +) +CLUSTER BY migration_id; +""" + + +@dataclass(frozen=True) +class Migration: + """One manifest entry.""" + + migration_id: str + sql_path: Path + checksum: str + # Sprint 6 (ADR-015): a breaking migration makes releases built before it + # ineligible as rollback targets — old runtimes cannot run against the + # post-migration schema and a forward fix is required instead. + breaking: bool = False + + +@dataclass(frozen=True) +class MigrationPlanEntry: + """Planned action for one migration.""" + + migration_id: str + checksum: str + state: str # PENDING | APPLIED | CHECKSUM_MISMATCH | FAILED_PREVIOUSLY + + +def load_manifest(manifest_path: Path | None = None) -> list[Migration]: + """Parse the migration manifest into ordered migrations.""" + path = manifest_path or MANIFEST_PATH + migrations: list[Migration] = [] + seen: set[str] = set() + for line in path.read_text(encoding="utf-8").splitlines(): + line = line.strip() + if not line or line.startswith("#"): + continue + parts = [part.strip() for part in line.split("|")] + if len(parts) < 2 or not parts[0] or not parts[1]: + raise ValueError(f"Malformed manifest line: {line!r}") + migration_id, rel_path = parts[0], parts[1] + flags = set(parts[2:]) + if flags - {"breaking"}: + raise ValueError(f"Unknown manifest flags {sorted(flags - {'breaking'})} on line: {line!r}") + if migration_id in seen: + raise ValueError(f"Duplicate migration id: {migration_id}") + seen.add(migration_id) + sql_path = (path.parent / rel_path).resolve() + if not sql_path.is_file(): + raise FileNotFoundError(f"Migration {migration_id} references missing file {sql_path}") + checksum = hashlib.sha256(sql_path.read_bytes()).hexdigest() + migrations.append( + Migration( + migration_id=migration_id, + sql_path=sql_path, + checksum=checksum, + breaking="breaking" in flags, + ) + ) + return migrations + + +def render_migration_sql(migration: Migration, settings: AtlasSettings) -> str: + """Render placeholder fields against canonical Atlas identifiers.""" + return migration.sql_path.read_text(encoding="utf-8").format( + project_id=settings.gcp.project_id, + dataset_id=settings.gcp.dataset_id, + location=settings.gcp.location, + ) + + +def _ledger_fqn(settings: AtlasSettings) -> str: + return f"{settings.gcp.project_id}.atlas_ops.schema_migrations" + + +def ensure_ledger(client: bigquery.Client, settings: AtlasSettings) -> None: + """Create the atlas_ops schema and migration ledger when missing.""" + ddl = _LEDGER_DDL.format(project_id=settings.gcp.project_id, location=settings.gcp.location) + for statement in ddl.split(";"): + if statement.strip(): + client.query(statement).result() + + +def recorded_migrations(client: bigquery.Client, settings: AtlasSettings) -> dict[str, dict[str, Any]]: + """Return the latest ledger row per migration_id, or {} when no ledger exists.""" + query = f""" + SELECT migration_id, migration_checksum, status + FROM `{_ledger_fqn(settings)}` + QUALIFY ROW_NUMBER() OVER (PARTITION BY migration_id ORDER BY applied_at DESC) = 1 + """ + try: + rows = list(client.query(query).result()) + except Exception: # noqa: BLE001 - ledger absent on first run + return {} + return {row["migration_id"]: dict(row.items()) for row in rows} + + +def plan_migrations( + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, + manifest_path: Path | None = None, +) -> list[MigrationPlanEntry]: + """Compute the action for every manifest migration without mutating anything.""" + settings = settings or load_settings() + bq = client or labeled_bigquery_client(settings.gcp.project_id, "migration") + recorded = recorded_migrations(bq, settings) + plan: list[MigrationPlanEntry] = [] + for migration in load_manifest(manifest_path): + row = recorded.get(migration.migration_id) + if row is None: + state = "PENDING" + elif row["status"] != "APPLIED": + # A migration that never succeeded is retryable, including with + # corrected file content: immutability protects applied schema + # changes, not broken attempts (Sprint 5 fix, regression-tested). + state = "FAILED_PREVIOUSLY" + elif row["migration_checksum"] != migration.checksum: + state = "CHECKSUM_MISMATCH" + else: + state = "APPLIED" + plan.append( + MigrationPlanEntry(migration_id=migration.migration_id, checksum=migration.checksum, state=state) + ) + return plan + + +def _record_ledger_row( + client: bigquery.Client, + settings: AtlasSettings, + migration: Migration, + status: str, + error_summary: str | None, +) -> None: + query = f""" + MERGE `{_ledger_fqn(settings)}` AS target + USING ( + SELECT + @migration_id AS migration_id, + @migration_checksum AS migration_checksum, + @git_sha AS git_sha, + CURRENT_TIMESTAMP() AS applied_at, + @workflow_run_id AS workflow_run_id, + @applied_by AS applied_by, + @status AS status, + @error_summary AS error_summary + ) AS source + ON target.migration_id = source.migration_id + WHEN MATCHED THEN UPDATE SET + migration_checksum = source.migration_checksum, + git_sha = source.git_sha, + applied_at = source.applied_at, + workflow_run_id = source.workflow_run_id, + applied_by = source.applied_by, + status = source.status, + error_summary = source.error_summary + WHEN NOT MATCHED THEN INSERT ( + migration_id, migration_checksum, git_sha, applied_at, + workflow_run_id, applied_by, status, error_summary + ) VALUES ( + source.migration_id, source.migration_checksum, source.git_sha, source.applied_at, + source.workflow_run_id, source.applied_by, source.status, source.error_summary + ) + """ + params = { + "migration_id": migration.migration_id, + "migration_checksum": migration.checksum, + "git_sha": os.environ.get("ATLAS_DEPLOYED_GIT_SHA") or os.environ.get("GITHUB_SHA"), + "workflow_run_id": os.environ.get("GITHUB_RUN_ID"), + "applied_by": os.environ.get("GITHUB_ACTOR") or os.environ.get("USER"), + "status": status, + "error_summary": sanitize_error_message(error_summary, max_length=500), + } + job_config = bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter(name, "STRING", value) for name, value in params.items() + ] + ) + client.query(query, job_config=job_config).result() + + +def apply_migrations( + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, + manifest_path: Path | None = None, +) -> list[MigrationPlanEntry]: + """Apply pending migrations in manifest order, recording each in the ledger. + + Raises RuntimeError on the first checksum mismatch or execution failure so a + failed migration blocks runtime promotion. + """ + settings = settings or load_settings() + bq = client or labeled_bigquery_client(settings.gcp.project_id, "migration") + ensure_ledger(bq, settings) + results: list[MigrationPlanEntry] = [] + recorded = recorded_migrations(bq, settings) + for migration in load_manifest(manifest_path): + row = recorded.get(migration.migration_id) + if row is not None and row["status"] == "APPLIED" and row["migration_checksum"] != migration.checksum: + # Only successfully applied migrations are immutable; a FAILED + # attempt may be retried with corrected content (Sprint 5 fix). + raise RuntimeError( + f"Migration {migration.migration_id} content changed after being recorded " + f"(ledger {row['migration_checksum'][:12]}…, file {migration.checksum[:12]}…). " + "Shipped migrations are immutable; add a new migration instead." + ) + if row is not None and row["status"] == "APPLIED": + results.append(MigrationPlanEntry(migration.migration_id, migration.checksum, "APPLIED")) + continue + rendered = render_migration_sql(migration, settings) + try: + for statement in rendered.split(";"): + if statement.strip(): + bq.query(statement).result() + except Exception as exc: + _record_ledger_row(bq, settings, migration, "FAILED", str(exc)) + raise RuntimeError( + f"Migration {migration.migration_id} failed and was recorded as FAILED: " + f"{sanitize_error_message(str(exc), max_length=200)}" + ) from exc + _record_ledger_row(bq, settings, migration, "APPLIED", None) + results.append(MigrationPlanEntry(migration.migration_id, migration.checksum, "APPLIED_NOW")) + return results + + +def migration_status( + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> list[dict[str, Any]]: + """Return all ledger rows ordered by applied_at.""" + settings = settings or load_settings() + bq = client or labeled_bigquery_client(settings.gcp.project_id, "migration") + query = f"SELECT * FROM `{_ledger_fqn(settings)}` ORDER BY applied_at" + try: + rows = list(bq.query(query).result()) + except Exception: # noqa: BLE001 - ledger absent + return [] + out = [] + for row in rows: + item = dict(row.items()) + for key, value in item.items(): + if isinstance(value, datetime): + item[key] = value.astimezone(UTC).isoformat() + out.append(item) + return out diff --git a/src/atlas/ops/preflight.py b/src/atlas/ops/preflight.py new file mode 100644 index 0000000..37ebfc0 --- /dev/null +++ b/src/atlas/ops/preflight.py @@ -0,0 +1,85 @@ +"""Environment preflight checks for orchestrated Atlas runs.""" + +from __future__ import annotations + +import shutil +from dataclasses import dataclass +from pathlib import Path + +from atlas.config.settings import AtlasSettings, atlas_root, load_settings +from atlas.loader.bigquery import ensure_events_table +from atlas.observability.cost import labeled_bigquery_client +from atlas.ops.resources import ensure_audit_resources + + +@dataclass(frozen=True) +class PreflightResult: + """Summary of environment validation.""" + + status: str + checks: list[str] + atlas_root: str + dbt_project_dir: str + + +def _check_path_exists(path: Path, label: str, checks: list[str]) -> None: + if path.exists(): + checks.append(f"PASS {label}: {path}") + else: + checks.append(f"FAIL {label}: missing {path}") + + +def preflight_environment( + settings: AtlasSettings | None = None, + *, + dbt_project_dir: Path | None = None, + skip_gcp: bool = False, +) -> PreflightResult: + """Validate Atlas runtime paths, scripts, and optional GCP resources.""" + settings = settings or load_settings() + root = atlas_root() + dbt_dir = dbt_project_dir or (root / "dbt" / "atlas_dbt") + checks: list[str] = [] + + _check_path_exists(root / "config" / "atlas.yaml", "atlas config", checks) + _check_path_exists(root / "scripts" / "generate_events.py", "generate script", checks) + _check_path_exists(root / "scripts" / "upload_events.py", "upload script", checks) + _check_path_exists(root / "scripts" / "load_events.py", "load script", checks) + _check_path_exists(root / "scripts" / "validate_events.py", "validate script", checks) + _check_path_exists(root / "scripts" / "run_atlas_step.sh", "step dispatcher", checks) + _check_path_exists(dbt_dir / "dbt_project.yml", "dbt project", checks) + + if shutil.which("dbt") is None: + checks.append("WARN dbt CLI not on PATH (expected in Cloud Shell / venv)") + else: + checks.append("PASS dbt CLI available") + + if settings.gcp.project_id: + checks.append(f"PASS GCP project configured: {settings.gcp.project_id}") + else: + checks.append("FAIL GCP project not configured") + + if settings.gcp.bucket_name: + checks.append(f"PASS GCS bucket configured: {settings.gcp.bucket_name}") + else: + checks.append("FAIL GCS bucket not configured") + + if not skip_gcp: + try: + ensure_audit_resources(settings=settings) + checks.append("PASS atlas_ops audit resources verified") + ensure_events_table( + labeled_bigquery_client(settings.gcp.project_id, "pipeline"), + settings, + ) + checks.append(f"PASS raw table verified: {settings.gcp.dataset_id}.{settings.gcp.table_id}") + except Exception as exc: # noqa: BLE001 - preflight captures operational failures + checks.append(f"FAIL GCP preflight: {exc}") + + status = "PASS" if all(item.startswith("PASS") or item.startswith("WARN") for item in checks) else "FAIL" + return PreflightResult( + status=status, + checks=checks, + atlas_root=str(root), + dbt_project_dir=str(dbt_dir), + ) diff --git a/src/atlas/ops/quality_results.py b/src/atlas/ops/quality_results.py new file mode 100644 index 0000000..bc7c51b --- /dev/null +++ b/src/atlas/ops/quality_results.py @@ -0,0 +1,187 @@ +"""Durable data-quality and monitor-evaluation records (Sprint 5, Phase 4). + +Two grains, two tables: + +- ``atlas_ops.quality_results`` — one row per data-quality check per pipeline + run (MERGE key: pipeline_run_id + check_name). Populated by the warehouse + validator and any future check producers; dbt evidence is summarized and + linked, not duplicated. +- ``atlas_ops.monitor_evaluations`` — one row per monitor check per + evaluation window (MERGE key: evaluation_id). Populated by the + atlas_observability_monitor DAG. +""" + +from __future__ import annotations + +import json +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.cost import labeled_bigquery_client + +ALLOWED_CHECK_CATEGORIES = frozenset( + { + "FRESHNESS", + "COMPLETENESS", + "UNIQUENESS", + "REFERENTIAL_INTEGRITY", + "SCHEMA", + "VOLUME", + "REJECTION_RATE", + "RECONCILIATION", + } +) +ALLOWED_QUALITY_STATUSES = frozenset({"PASS", "WARN", "FAIL", "NO_DATA"}) +ALLOWED_EVALUATION_STATUSES = frozenset({"PASS", "WARN", "FAIL", "NO_DATA", "DISABLED"}) +ALLOWED_SEVERITIES = frozenset({"INFO", "WARNING", "CRITICAL"}) + +_MAX_DETAILS_LENGTH = 4000 + + +@dataclass(frozen=True) +class QualityResultRecord: + """One durable data-quality check result.""" + + pipeline_run_id: str + check_name: str + check_category: str + severity: str + status: str + evaluated_at: str + batch_id: str | None = None + observed_value: float | None = None + expected_value: float | None = None + lower_bound: float | None = None + upper_bound: float | None = None + model_name: str | None = None + details_json: str | None = None + git_sha: str | None = None + + +@dataclass(frozen=True) +class MonitorEvaluationRecord: + """One durable monitor evaluation for a bounded window.""" + + evaluation_id: str + check_name: str + environment: str + status: str + evaluated_at: str + window_start: str | None = None + window_end: str | None = None + severity: str | None = None + observed_value: float | None = None + threshold: float | None = None + incident_key: str | None = None + source: str | None = None + details_json: str | None = None + + +def _truncate_details(details_json: str | None) -> str | None: + if details_json and len(details_json) > _MAX_DETAILS_LENGTH: + return details_json[: _MAX_DETAILS_LENGTH - 3] + "..." + return details_json + + +def details_to_json(details: dict[str, Any] | None) -> str | None: + """Serialize a details dict defensively (non-serializable -> repr).""" + if not details: + return None + return _truncate_details(json.dumps(details, default=repr, sort_keys=True)) + + +_QUALITY_FLOAT_FIELDS = frozenset({"observed_value", "expected_value", "lower_bound", "upper_bound"}) +_EVAL_FLOAT_FIELDS = frozenset({"observed_value", "threshold"}) + + +def _merge( + table: str, + payload: dict[str, Any], + key_fields: tuple[str, ...], + float_fields: frozenset[str], + timestamp_fields: frozenset[str], + client: bigquery.Client, +) -> None: + now = datetime.now(tz=UTC).isoformat() + params: list[bigquery.ScalarQueryParameter] = [] + for key, value in payload.items(): + if key in float_fields: + params.append(bigquery.ScalarQueryParameter(key, "FLOAT64", value)) + elif key in timestamp_fields: + params.append(bigquery.ScalarQueryParameter(key, "TIMESTAMP", value)) + else: + params.append(bigquery.ScalarQueryParameter(key, "STRING", value)) + params.append(bigquery.ScalarQueryParameter("now", "TIMESTAMP", now)) + + on_clause = " AND ".join(f"target.{k} = @{k}" for k in key_fields) + update_cols = [k for k in payload if k not in key_fields] + set_clause = ", ".join(f"{col} = @{col}" for col in update_cols) + insert_cols = ", ".join([*payload.keys(), "created_at", "updated_at"]) + insert_vals = ", ".join([f"@{col}" for col in payload] + ["@now", "@now"]) + sql = f""" + MERGE `{table}` AS target + USING (SELECT @{key_fields[0]} AS join_key) AS source + ON {on_clause} + WHEN MATCHED THEN + UPDATE SET {set_clause}, updated_at = @now + WHEN NOT MATCHED THEN + INSERT ({insert_cols}) + VALUES ({insert_vals}) + """ + client.query(sql, job_config=bigquery.QueryJobConfig(query_parameters=params)).result() + + +def upsert_quality_result( + record: QualityResultRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Merge one quality result keyed by (pipeline_run_id, check_name).""" + if record.check_category not in ALLOWED_CHECK_CATEGORIES: + raise ValueError(f"Unsupported check category: {record.check_category}") + if record.status not in ALLOWED_QUALITY_STATUSES: + raise ValueError(f"Unsupported quality status: {record.status}") + if record.severity not in ALLOWED_SEVERITIES: + raise ValueError(f"Unsupported severity: {record.severity}") + settings = settings or load_settings() + client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + payload = asdict(record) + payload["details_json"] = _truncate_details(payload.get("details_json")) + _merge( + f"{settings.gcp.project_id}.atlas_ops.quality_results", + payload, + ("pipeline_run_id", "check_name"), + _QUALITY_FLOAT_FIELDS, + frozenset({"evaluated_at"}), + client, + ) + + +def upsert_monitor_evaluation( + record: MonitorEvaluationRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Merge one monitor evaluation keyed by evaluation_id.""" + if record.status not in ALLOWED_EVALUATION_STATUSES: + raise ValueError(f"Unsupported evaluation status: {record.status}") + if record.severity is not None and record.severity not in ALLOWED_SEVERITIES: + raise ValueError(f"Unsupported severity: {record.severity}") + settings = settings or load_settings() + client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + payload = asdict(record) + payload["details_json"] = _truncate_details(payload.get("details_json")) + _merge( + f"{settings.gcp.project_id}.atlas_ops.monitor_evaluations", + payload, + ("evaluation_id",), + _EVAL_FLOAT_FIELDS, + frozenset({"evaluated_at", "window_start", "window_end"}), + client, + ) diff --git a/src/atlas/ops/recovery_actions.py b/src/atlas/ops/recovery_actions.py new file mode 100644 index 0000000..c697d53 --- /dev/null +++ b/src/atlas/ops/recovery_actions.py @@ -0,0 +1,256 @@ +"""Recovery-action audit records (Sprint 6, Phase 4 / ADR-014). + +Grain: one row per recovery action attempt in ``atlas_ops.recovery_actions``, +keyed by ``recovery_id`` and written with idempotent parameterized MERGE. +Recovery actions are a separate grain from pipeline runs and deployments: +they *link* to incidents, pipeline runs, batches, and deployments but never +mutate those records. + +Verification contract: a recovery attempt may only be finalized as SUCCESS +when its verification passed (``verification_status="VERIFIED"``). Finalizing +SUCCESS without verification raises — an unverified "recovery" is not a +recovery (ADR-014). +""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, replace +from datetime import UTC, datetime +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.cost import labeled_bigquery_client +from atlas.observability.logging import emit_event +from atlas.ops.audit import sanitize_error_message + +ALLOWED_ACTION_TYPES = frozenset( + { + "RETRY_TASK", + "RERUN_BATCH", + "REPAIR_PARTIAL_LOAD", + "QUARANTINE_BATCH", + "BACKFILL", + "RESTORE_RELEASE", + "FORWARD_MIGRATION", + "RESTORE_IAM", + "REBUILD_PARTITION", + "PAUSE_SCHEDULE", + "RESUME_SCHEDULE", + "RECONSTRUCT_AUDIT", + "RESET_MONITOR", + "MANUAL_CONTAINMENT", + } +) + +ALLOWED_STATUSES = frozenset({"RUNNING", "SUCCESS", "FAILED", "PARTIAL", "ABORTED"}) +ALLOWED_VERIFICATION_STATUSES = frozenset({"PENDING", "VERIFIED", "FAILED", "SKIPPED"}) + + +@dataclass(frozen=True) +class RecoveryActionRecord: + """One durable recovery-action audit row.""" + + recovery_id: str + action_type: str + status: str + incident_id: str | None = None + scenario_id: str | None = None + pipeline_run_id: str | None = None + batch_id: str | None = None + deployment_id: str | None = None + operator: str | None = None + environment: str | None = None + started_at: str | None = None + completed_at: str | None = None + source_state: str | None = None + target_state: str | None = None + verification_status: str | None = None + error_type: str | None = None + error_summary: str | None = None + git_sha: str | None = None + + +def _table_fqn(settings: AtlasSettings) -> str: + return f"{settings.gcp.project_id}.atlas_ops.recovery_actions" + + +_TIMESTAMP_FIELDS = frozenset({"started_at", "completed_at"}) +_KEY_FIELDS = ("recovery_id",) + + +def _validate(record: RecoveryActionRecord) -> None: + if record.action_type not in ALLOWED_ACTION_TYPES: + raise ValueError(f"Unsupported recovery action type: {record.action_type}") + if record.status not in ALLOWED_STATUSES: + raise ValueError(f"Unsupported recovery status: {record.status}") + if ( + record.verification_status is not None + and record.verification_status not in ALLOWED_VERIFICATION_STATUSES + ): + raise ValueError(f"Unsupported verification status: {record.verification_status}") + if record.status == "SUCCESS" and record.verification_status != "VERIFIED": + raise ValueError( + "recovery status SUCCESS requires verification_status=VERIFIED (ADR-014: " + "no recovery is successful before verification passes)" + ) + + +def upsert_recovery_action( + record: RecoveryActionRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Merge one recovery-action row keyed by recovery_id.""" + _validate(record) + settings = settings or load_settings() + client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + + payload = asdict(record) + payload["error_summary"] = sanitize_error_message(payload.get("error_summary")) + now = datetime.now(tz=UTC).isoformat() + + params: list[bigquery.ScalarQueryParameter] = [] + for key, value in payload.items(): + kind = "TIMESTAMP" if key in _TIMESTAMP_FIELDS else "STRING" + params.append(bigquery.ScalarQueryParameter(key, kind, value)) + params.append(bigquery.ScalarQueryParameter("now", "TIMESTAMP", now)) + + update_cols = [k for k in payload if k not in _KEY_FIELDS] + set_clause = ", ".join(f"{col} = @{col}" for col in update_cols) + insert_cols = ", ".join([*payload.keys(), "created_at", "updated_at"]) + insert_vals = ", ".join([f"@{col}" for col in payload] + ["@now", "@now"]) + + sql = f""" + MERGE `{_table_fqn(settings)}` AS target + USING (SELECT @recovery_id AS recovery_id) AS source + ON target.recovery_id = @recovery_id + WHEN MATCHED THEN + UPDATE SET {set_clause}, updated_at = @now + WHEN NOT MATCHED THEN + INSERT ({insert_cols}) + VALUES ({insert_vals}) + """ + client.query(sql, job_config=bigquery.QueryJobConfig(query_parameters=params)).result() + + +def start_recovery_action( + *, + recovery_id: str, + action_type: str, + incident_id: str | None = None, + scenario_id: str | None = None, + pipeline_run_id: str | None = None, + batch_id: str | None = None, + deployment_id: str | None = None, + operator: str | None = None, + environment: str | None = None, + source_state: str | None = None, + target_state: str | None = None, + git_sha: str | None = None, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> RecoveryActionRecord: + """Create or refresh a RUNNING recovery-action row.""" + record = RecoveryActionRecord( + recovery_id=recovery_id, + action_type=action_type, + status="RUNNING", + incident_id=incident_id, + scenario_id=scenario_id, + pipeline_run_id=pipeline_run_id, + batch_id=batch_id, + deployment_id=deployment_id, + operator=operator, + environment=environment, + started_at=datetime.now(tz=UTC).isoformat(), + source_state=source_state, + target_state=target_state, + verification_status="PENDING", + git_sha=git_sha, + ) + upsert_recovery_action(record, settings, client=client) + emit_event( + "recovery_action_started", + severity="INFO", + component="recovery", + pipeline_run_id=pipeline_run_id, + batch_id=batch_id, + deployment_id=deployment_id, + status="RUNNING", + check_name=action_type, + correlation_id=recovery_id, + ) + return record + + +def finalize_recovery_action( + record: RecoveryActionRecord, + *, + status: str, + verification_status: str, + error_type: str | None = None, + error_summary: str | None = None, + target_state: str | None = None, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> RecoveryActionRecord: + """Upsert the terminal state for one recovery attempt. Idempotent. + + SUCCESS is refused unless verification_status is VERIFIED — verification + is not optional decoration on a recovery, it *is* the recovery evidence. + """ + final = replace( + record, + status=status, + verification_status=verification_status, + completed_at=record.completed_at or datetime.now(tz=UTC).isoformat(), + error_type=error_type, + error_summary=error_summary, + target_state=target_state or record.target_state, + ) + upsert_recovery_action(final, settings, client=client) + emit_event( + "recovery_action_finalized", + severity="INFO" if status == "SUCCESS" else "ERROR", + component="recovery", + pipeline_run_id=final.pipeline_run_id, + batch_id=final.batch_id, + deployment_id=final.deployment_id, + status=status, + check_name=final.action_type, + correlation_id=final.recovery_id, + error_type=error_type, + error_message=error_summary, + ) + return final + + +def query_recovery_actions( + *, + scenario_id: str | None = None, + incident_id: str | None = None, + batch_id: str | None = None, + settings: AtlasSettings | None = None, + client: bigquery.Client | None = None, +) -> list[dict[str, Any]]: + """Return recovery actions filtered by scenario, incident, or batch.""" + settings = settings or load_settings() + client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + sql = f""" + SELECT * FROM `{_table_fqn(settings)}` + WHERE (@scenario_id IS NULL OR scenario_id = @scenario_id) + AND (@incident_id IS NULL OR incident_id = @incident_id) + AND (@batch_id IS NULL OR batch_id = @batch_id) + ORDER BY started_at + """ + job_config = bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter("scenario_id", "STRING", scenario_id), + bigquery.ScalarQueryParameter("incident_id", "STRING", incident_id), + bigquery.ScalarQueryParameter("batch_id", "STRING", batch_id), + ] + ) + return [dict(row) for row in client.query(sql, job_config=job_config).result()] diff --git a/src/atlas/ops/resources.py b/src/atlas/ops/resources.py new file mode 100644 index 0000000..ff84d77 --- /dev/null +++ b/src/atlas/ops/resources.py @@ -0,0 +1,36 @@ +"""Idempotent creation of operational BigQuery resources.""" + +from __future__ import annotations + +from pathlib import Path + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.cost import labeled_bigquery_client + + +def _sql_path(name: str) -> Path: + return Path(__file__).resolve().parents[3] / "sql" / name + + +def ensure_audit_resources( + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Create atlas_ops dataset and pipeline_runs table when missing.""" + settings = settings or load_settings() + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + schema_sql = (_sql_path("create_ops_schema.sql")).read_text(encoding="utf-8") + table_sql = (_sql_path("create_pipeline_runs_table.sql")).read_text(encoding="utf-8") + for template in (schema_sql, table_sql): + rendered = template.format( + project_id=settings.gcp.project_id, + location=settings.gcp.location, + ) + bq_client.query(rendered).result() + + # Verify the table exists after DDL. + table_id = f"{settings.gcp.project_id}.atlas_ops.pipeline_runs" + bq_client.get_table(table_id) diff --git a/src/atlas/ops/rollback_compatibility.py b/src/atlas/ops/rollback_compatibility.py new file mode 100644 index 0000000..bbba2d0 --- /dev/null +++ b/src/atlas/ops/rollback_compatibility.py @@ -0,0 +1,74 @@ +"""Rollback schema-compatibility decisions (Sprint 6, ADR-015 / ADR-010). + +Rule: runtime rollback to a prior release is allowed only when that release is +compatible with the currently applied schema. With additive-only migrations +that is normally true; a migration flagged ``breaking`` in +``sql/migrations/manifest.txt`` marks the boundary after which releases built +before it can no longer run. Rolling back across a breaking migration is +blocked — the operator gets forward-recovery guidance instead, and nothing is +ever reversed automatically. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +from atlas.ops.migrations import Migration + + +@dataclass(frozen=True) +class RollbackDecision: + """Outcome of one rollback eligibility evaluation.""" + + eligible: bool + reason: str + blocking_migrations: tuple[str, ...] = () + + +def evaluate_rollback_compatibility( + *, + applied_migration_ids: list[str], + target_release_migration_ids: list[str], + manifest: list[Migration], +) -> RollbackDecision: + """Decide whether a prior release may be restored against the live schema. + + - Migrations pending for the target release block promotion (unchanged + Sprint 4 rule; handled by the caller as PENDING_MIGRATIONS). + - Applied migrations the target release does not know about are tolerated + when additive, and block the rollback when flagged breaking. + """ + known_to_target = set(target_release_migration_ids) + breaking_by_id = {m.migration_id: m.breaking for m in manifest} + + newer_applied = [m for m in applied_migration_ids if m not in known_to_target] + blocking = tuple(m for m in newer_applied if breaking_by_id.get(m, False)) + if blocking: + return RollbackDecision( + eligible=False, + reason=( + "applied schema contains breaking migration(s) the target release " + f"predates: {list(blocking)}. Runtime rollback is blocked; recover " + "forward (fix on a new release) instead. Breaking BigQuery " + "migrations are never reversed automatically (ADR-010/ADR-015)." + ), + blocking_migrations=blocking, + ) + unknown = [m for m in newer_applied if m not in breaking_by_id] + if unknown: + return RollbackDecision( + eligible=False, + reason=( + f"applied migration(s) {unknown} are not in the current repository " + "manifest, so their compatibility cannot be classified. Refusing " + "rollback rather than guessing." + ), + blocking_migrations=tuple(unknown), + ) + return RollbackDecision( + eligible=True, + reason=( + "target release is schema-compatible: every newer applied migration " + f"({newer_applied or 'none'}) is additive" + ), + ) diff --git a/src/atlas/ops/task_events.py b/src/atlas/ops/task_events.py new file mode 100644 index 0000000..838c3b5 --- /dev/null +++ b/src/atlas/ops/task_events.py @@ -0,0 +1,212 @@ +"""Task-attempt audit records (Sprint 5, Phase 3). + +Grain: one row per (pipeline_run_id, task_id, attempt_number, event_type) in +``atlas_ops.task_events``, written with idempotent parameterized MERGE. +Repeated callbacks update the existing row instead of duplicating it, so a +failed attempt 1 and a successful attempt 2 remain distinguishable rows. + +Telemetry-safety contract (ADR-011): ``record_task_event_safely`` never +raises — a telemetry failure emits a structured fallback event and returns +False, leaving the caller's data path untouched. The finalizer separately +verifies telemetry completeness so degradation stays visible. +""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.observability.cost import labeled_bigquery_client +from atlas.observability.logging import emit_event +from atlas.ops.audit import sanitize_error_message + +ALLOWED_EVENT_TYPES = frozenset({"STARTED", "RETRY", "SUCCESS", "FAILED", "SKIPPED", "UPSTREAM_FAILED"}) + +# Task ids the finalizer expects telemetry for on every non-skipped run. +EXPECTED_TERMINAL_TASKS = ( + "resolve_run_context", + "ensure_audit_resources", + "start_run_audit", + "preflight_environment", + "generate_events", + "upload_events", + "load_bigquery_raw", + "validate_raw_load", + "dbt_seed", + "dbt_source_freshness", + "dbt_build", + "validate_warehouse", + "publish_success_marker", +) + + +# Where a row's timing came from (Sprint 6, Phase 1). Ordered by preference. +TIMING_SOURCES = frozenset({"step_runner_clock", "airflow_task_instance", "finalizer_reconciliation"}) +TIMING_CONFIDENCES = frozenset({"exact", "partial", "none"}) + + +@dataclass(frozen=True) +class TaskEventRecord: + """One durable task-attempt audit row.""" + + pipeline_run_id: str + task_id: str + attempt_number: int + event_type: str + batch_id: str | None = None + airflow_run_id: str | None = None + dag_id: str | None = None + status: str | None = None + started_at: str | None = None + completed_at: str | None = None + duration_ms: int | None = None + operator_type: str | None = None + environment: str | None = None + git_sha: str | None = None + rows_affected: int | None = None + error_type: str | None = None + error_message: str | None = None + timing_source: str | None = None + timing_confidence: str | None = None + + +def _table_fqn(settings: AtlasSettings) -> str: + return f"{settings.gcp.project_id}.atlas_ops.task_events" + + +_INT64_FIELDS = frozenset({"attempt_number", "duration_ms", "rows_affected"}) +_TIMESTAMP_FIELDS = frozenset({"started_at", "completed_at"}) +_KEY_FIELDS = ("pipeline_run_id", "task_id", "attempt_number", "event_type") + + +def upsert_task_event( + record: TaskEventRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> None: + """Merge one task event row keyed by (run, task, attempt, event_type).""" + if record.event_type not in ALLOWED_EVENT_TYPES: + raise ValueError(f"Unsupported task event type: {record.event_type}") + if record.attempt_number < 1: + raise ValueError("attempt_number must be >= 1") + if record.timing_source is not None and record.timing_source not in TIMING_SOURCES: + raise ValueError(f"Unsupported timing_source: {record.timing_source}") + if record.timing_confidence is not None and record.timing_confidence not in TIMING_CONFIDENCES: + raise ValueError(f"Unsupported timing_confidence: {record.timing_confidence}") + + settings = settings or load_settings() + client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + + payload = asdict(record) + payload["error_message"] = sanitize_error_message(payload.get("error_message")) + now = datetime.now(tz=UTC).isoformat() + + params: list[bigquery.ScalarQueryParameter] = [] + for key, value in payload.items(): + if key in _INT64_FIELDS: + params.append(bigquery.ScalarQueryParameter(key, "INT64", value)) + elif key in _TIMESTAMP_FIELDS: + params.append(bigquery.ScalarQueryParameter(key, "TIMESTAMP", value)) + else: + params.append(bigquery.ScalarQueryParameter(key, "STRING", value)) + params.append(bigquery.ScalarQueryParameter("now", "TIMESTAMP", now)) + + update_cols = [k for k in payload if k not in _KEY_FIELDS] + set_clause = ", ".join(f"{col} = @{col}" for col in update_cols) + insert_cols = ", ".join([*payload.keys(), "created_at", "updated_at"]) + insert_vals = ", ".join([f"@{col}" for col in payload] + ["@now", "@now"]) + + sql = f""" + MERGE `{_table_fqn(settings)}` AS target + USING (SELECT @pipeline_run_id AS pipeline_run_id) AS source + ON target.pipeline_run_id = @pipeline_run_id + AND target.task_id = @task_id + AND target.attempt_number = @attempt_number + AND target.event_type = @event_type + WHEN MATCHED THEN + UPDATE SET {set_clause}, updated_at = @now + WHEN NOT MATCHED THEN + INSERT ({insert_cols}) + VALUES ({insert_vals}) + """ + client.query(sql, job_config=bigquery.QueryJobConfig(query_parameters=params)).result() + + +def record_task_event_safely( + record: TaskEventRecord, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> bool: + """Write a task event without ever propagating telemetry failure. + + Returns True when the row was written. On failure, emits a structured + ``task_telemetry_write_failed`` event and returns False. + """ + try: + upsert_task_event(record, settings, client=client) + return True + except Exception as exc: # noqa: BLE001 - telemetry must not break the data path + emit_event( + "task_telemetry_write_failed", + severity="ERROR", + component="task_events", + pipeline_run_id=record.pipeline_run_id, + task_id=record.task_id, + attempt_number=record.attempt_number, + error_type=type(exc).__name__, + error_message=str(exc), + ) + return False + + +def query_task_events( + pipeline_run_id: str, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> list[dict[str, Any]]: + """Return all task events for one pipeline run, ordered for diagnosis.""" + settings = settings or load_settings() + client = client or labeled_bigquery_client(settings.gcp.project_id, "audit") + sql = f""" + SELECT * FROM `{_table_fqn(settings)}` + WHERE pipeline_run_id = @pipeline_run_id + ORDER BY task_id, attempt_number, event_type + """ + job_config = bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter("pipeline_run_id", "STRING", pipeline_run_id)] + ) + return [dict(row) for row in client.query(sql, job_config=job_config).result()] + + +def telemetry_completeness( + pipeline_run_id: str, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, + expected_tasks: tuple[str, ...] = EXPECTED_TERMINAL_TASKS, +) -> dict[str, Any]: + """Report which expected tasks are missing a terminal event for a run. + + A task is "complete" when it has at least one terminal event + (SUCCESS/FAILED/SKIPPED/UPSTREAM_FAILED). STARTED-only rows indicate the + telemetry stream was cut mid-task. + """ + events = query_task_events(pipeline_run_id, settings, client=client) + terminal = {"SUCCESS", "FAILED", "SKIPPED", "UPSTREAM_FAILED"} + seen_terminal = {e["task_id"] for e in events if e["event_type"] in terminal} + started_only = {e["task_id"] for e in events if e["event_type"] == "STARTED"} - seen_terminal + missing = [t for t in expected_tasks if t not in seen_terminal] + return { + "pipeline_run_id": pipeline_run_id, + "complete": not missing, + "missing_terminal": missing, + "started_without_terminal": sorted(started_only), + "event_count": len(events), + } diff --git a/src/atlas/pipeline/__init__.py b/src/atlas/pipeline/__init__.py new file mode 100644 index 0000000..ba3bb46 --- /dev/null +++ b/src/atlas/pipeline/__init__.py @@ -0,0 +1,5 @@ +"""Pipeline orchestration package.""" + +from atlas.pipeline.orchestrator import PipelineResult, run_pipeline, summarize_result + +__all__ = ["PipelineResult", "run_pipeline", "summarize_result"] diff --git a/src/atlas/pipeline/orchestrator.py b/src/atlas/pipeline/orchestrator.py new file mode 100644 index 0000000..edacc09 --- /dev/null +++ b/src/atlas/pipeline/orchestrator.py @@ -0,0 +1,167 @@ +"""Pipeline orchestration for Project Atlas Sprint 1. + +Purpose: + Coordinate generate → upload → load → validate as an end-to-end batch run. + +Interactions: + Invoked by ``scripts/run_pipeline.py`` and acceptance tests. Each step uses + shared settings, logging, and run identifiers. + +Engineering principles: + - Independent scripts remain runnable on their own. + - Orchestrator adds sequencing, logging, and failure propagation only. + +Common failure modes: + - Partial success leaves GCS object without BigQuery rows. + - Validation FAIL is expected for seeded anomalies; callers must inspect + acceptance checks separately. + +Implementation choice: + A thin Python orchestrator was chosen over Airflow for Sprint 1 because + orchestration belongs to v0.4. Alternatives considered: Makefile-only flow + (weaker error propagation) and Cloud Functions (out of scope). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import google.cloud.storage as storage +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings +from atlas.generator.events import GenerationResult, generate_events +from atlas.ingestion.upload import UploadResult, upload_events_file +from atlas.loader.bigquery import LoadResult, load_events_from_gcs +from atlas.logging.structured import StepLogger, configure_logging, new_pipeline_run_id +from atlas.validation.checks import ValidationReport, validate_anomaly_detection, validate_loaded_run + + +@dataclass(frozen=True) +class PipelineResult: + """Aggregate result of a full pipeline run.""" + + pipeline_run_id: str + generation: GenerationResult + upload: UploadResult | None + load: LoadResult | None + validation: ValidationReport | None + log_file: Path + + +def run_pipeline( + settings: AtlasSettings | None = None, + *, + pipeline_run_id: str | None = None, + skip_upload: bool = False, + skip_load: bool = False, + skip_validation: bool = False, + storage_client: storage.Client | None = None, + bigquery_client: bigquery.Client | None = None, +) -> PipelineResult: + """Execute the Sprint 1 Atlas pipeline.""" + settings = settings or load_settings() + pipeline_run_id = pipeline_run_id or new_pipeline_run_id() + logger = configure_logging(settings.logging.log_dir, pipeline_run_id) + + with StepLogger(logger, pipeline_run_id, "generate") as step: + generation = generate_events(settings) + step.rows_processed = generation.event_count + step.details = {"output_path": str(generation.output_path)} + + upload_result: UploadResult | None = None + load_result: LoadResult | None = None + validation_report: ValidationReport | None = None + + if not skip_upload: + with StepLogger( + logger, + pipeline_run_id, + "upload", + source_file=str(generation.output_path), + ) as step: + upload_result = upload_events_file( + settings, + generation.output_path, + generation.primary_event_date, + pipeline_run_id, + client=storage_client, + ) + step.rows_processed = generation.event_count + step.details = { + "gcs_uri": upload_result.gcs_uri, + "already_exists": upload_result.already_exists, + } + + if not skip_load and upload_result is not None: + with StepLogger( + logger, + pipeline_run_id, + "load", + source_file=upload_result.gcs_uri, + ) as step: + load_result = load_events_from_gcs( + settings, + upload_result.gcs_uri, + upload_result.gcs_uri, + pipeline_run_id, + client=bigquery_client, + ) + step.rows_processed = load_result.rows_loaded + step.details = { + "target_table": load_result.target_table, + "already_loaded": load_result.already_loaded, + } + + if not skip_validation and load_result is not None: + with StepLogger( + logger, + pipeline_run_id, + "validate", + source_file=load_result.source_file, + ) as step: + validation_report = validate_loaded_run( + settings, + pipeline_run_id, + generation.primary_event_date, + client=bigquery_client, + ) + validation_report = validate_anomaly_detection(validation_report, settings) + step.details = validation_report.to_dict() + step.rows_processed = generation.event_count + + log_file = settings.logging.log_dir / f"{pipeline_run_id}.jsonl" + return PipelineResult( + pipeline_run_id=pipeline_run_id, + generation=generation, + upload=upload_result, + load=load_result, + validation=validation_report, + log_file=log_file, + ) + + +def summarize_result(result: PipelineResult) -> dict[str, Any]: + """Return a concise pipeline summary for CLI output.""" + return { + "pipeline_run_id": result.pipeline_run_id, + "generated_events": result.generation.event_count, + "upload_uri": result.upload.gcs_uri if result.upload else None, + "loaded_rows": result.load.rows_loaded if result.load else None, + "validation_status": result.validation.overall_status if result.validation else None, + "acceptance_status": ( + "PASS" + if result.validation + and all( + check.status == "PASS" + for check in result.validation.checks + if check.name.startswith("acceptance_") + ) + else "FAIL" + if result.validation + else None + ), + "log_file": str(result.log_file), + } diff --git a/src/atlas/reference/__init__.py b/src/atlas/reference/__init__.py new file mode 100644 index 0000000..2fbfc55 --- /dev/null +++ b/src/atlas/reference/__init__.py @@ -0,0 +1,9 @@ +"""Atlas reference-architecture validation (Sprint 8). + +Validates the curated reference package and the machine-readable evidence index +against repository truth: referenced files exist, IDs are unique, live claims are +backed by live evidence, and blocked work is never presented as complete. + +Import the API from :mod:`atlas.reference.validate` (kept out of package import +to avoid a runpy double-import warning under ``python -m atlas.reference.validate``). +""" diff --git a/src/atlas/reference/validate.py b/src/atlas/reference/validate.py new file mode 100644 index 0000000..8717428 --- /dev/null +++ b/src/atlas/reference/validate.py @@ -0,0 +1,286 @@ +"""Reference-architecture and evidence-index validator (Sprint 8, Phase 8/18). + +``python -m atlas.reference.validate`` fails (exit 1) when: + +- a referenced file/ADR/evidence path is missing, +- duplicate document ids or claim ids exist, +- required fields are missing, +- a document status or claim status is outside the controlled vocabulary, +- a LIVE claim is backed only by documentation (not real live evidence), +- a blocked claim is presented as complete, +- a verification/last-verified commit is absent, +- a current document references a command whose script does not exist. + +Pure/offline: no credentials, no network. Repository files are the only input. +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from pathlib import Path +from typing import Any + +import yaml + +from atlas.config.settings import atlas_root + +MANIFEST_REL = "docs/reference-architecture/reference-manifest.yml" +EVIDENCE_REL = "governance/generated/evidence-index.json" + +DOC_STATUSES = {"CURRENT", "HISTORICAL", "SUPERSEDED", "PLANNED", "BLOCKED"} +CLAIM_STATUSES = { + "PROVEN_LIVE", + "PROVEN_STATIC", + "PROVEN_TEST", + "PLANNED", + "BLOCKED", + "NOT_APPLICABLE", +} +EVIDENCE_TYPES = { + "TEST", + "CI_RUN", + "LIVE_DEPLOYMENT", + "LIVE_QUERY", + "DRY_RUN", + "INCIDENT", + "RECOVERY", + "DOCUMENTED_DECISION", + "CONFIGURATION", + "CODE_INSPECTION", +} +# Evidence types that count as "live" proof. +LIVE_EVIDENCE_TYPES = {"LIVE_DEPLOYMENT", "LIVE_QUERY", "INCIDENT", "RECOVERY"} +# Evidence types that are only documentation (never sufficient for a LIVE claim). +DOC_ONLY_EVIDENCE_TYPES = {"DOCUMENTED_DECISION"} + +MANIFEST_REQUIRED_FIELDS = ( + "document_id", + "title", + "purpose", + "audience", + "status", + "source_of_truth", + "last_verified_commit", + "owner", +) +MANIFEST_PATH_FIELDS = ( + "source_of_truth", + "related_adrs", + "related_runbooks", + "related_tests", + "related_evidence", +) +CLAIM_REQUIRED_FIELDS = ( + "claim_id", + "claim", + "scope", + "evidence_type", + "evidence_path", + "live_or_static", + "verification_commit", +) + +# A crude command reference matcher: `bash scripts/foo.sh` / `python -m atlas.x`. +_SCRIPT_RE = re.compile(r"\b(?:bash|sh)\s+(scripts/[A-Za-z0-9_./-]+\.(?:sh|py))") + + +class ReferenceError(ValueError): + """Raised when the reference package cannot be loaded.""" + + +def _root() -> Path: + return atlas_root() + + +def _resolve(rel: str) -> Path: + return _root() / rel + + +def load_manifest() -> dict[str, Any]: + path = _resolve(MANIFEST_REL) + if not path.exists(): + raise ReferenceError(f"missing reference manifest: {MANIFEST_REL}") + data = yaml.safe_load(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise ReferenceError("reference manifest is not a mapping") + return data + + +def load_evidence_index() -> dict[str, Any]: + path = _resolve(EVIDENCE_REL) + if not path.exists(): + raise ReferenceError(f"missing evidence index: {EVIDENCE_REL}") + data = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(data, dict): + raise ReferenceError("evidence index is not a mapping") + return data + + +def _as_path_list(value: Any) -> list[str]: + if value is None: + return [] + if isinstance(value, str): + return [value] + if isinstance(value, list): + return [str(v) for v in value] + return [] + + +def validate_manifest(manifest: dict[str, Any] | None = None) -> list[str]: + """Return human-readable errors for the reference manifest (empty = OK).""" + if manifest is None: + manifest = load_manifest() + errors: list[str] = [] + documents = manifest.get("documents") + if not isinstance(documents, list) or not documents: + return ["reference manifest has no 'documents' list"] + + seen_ids: set[str] = set() + for entry in documents: + if not isinstance(entry, dict): + errors.append("manifest document entry is not a mapping") + continue + doc_id = str(entry.get("document_id", "")).strip() + label = doc_id or "" + + for field in MANIFEST_REQUIRED_FIELDS: + value = entry.get(field) + if value is None or (isinstance(value, str) and not value.strip()): + errors.append(f"{label}: missing required field '{field}'") + + if doc_id: + if doc_id in seen_ids: + errors.append(f"duplicate document_id '{doc_id}'") + seen_ids.add(doc_id) + + status = str(entry.get("status", "")).strip() + if status and status not in DOC_STATUSES: + errors.append(f"{label}: invalid status '{status}'") + + # Referenced files must exist. + for field in MANIFEST_PATH_FIELDS: + for rel in _as_path_list(entry.get(field)): + if not _resolve(rel).exists(): + errors.append(f"{label}: {field} path not found: {rel}") + + # CURRENT documents must not reference nonexistent scripts. + if status == "CURRENT": + src = str(entry.get("source_of_truth", "")).strip() + if src.endswith(".md") and _resolve(src).exists(): + text = _resolve(src).read_text(encoding="utf-8") + for match in _SCRIPT_RE.finditer(text): + rel = match.group(1) + if not _resolve(rel).exists(): + errors.append(f"{label}: references missing script '{rel}'") + return errors + + +def _evidence_is_doc_only(claim: dict[str, Any]) -> bool: + etype = str(claim.get("evidence_type", "")).strip() + return etype in DOC_ONLY_EVIDENCE_TYPES + + +def validate_evidence_index(index: dict[str, Any] | None = None) -> list[str]: + """Return human-readable errors for the evidence index (empty = OK).""" + if index is None: + index = load_evidence_index() + errors: list[str] = [] + claims = index.get("claims") + if not isinstance(claims, list) or not claims: + return ["evidence index has no 'claims' list"] + + seen: set[str] = set() + for claim in claims: + if not isinstance(claim, dict): + errors.append("claim entry is not a mapping") + continue + claim_id = str(claim.get("claim_id", "")).strip() + label = claim_id or "" + + for field in CLAIM_REQUIRED_FIELDS: + value = claim.get(field) + if value is None or (isinstance(value, str) and not value.strip()): + errors.append(f"{label}: missing required field '{field}'") + + if claim_id: + if claim_id in seen: + errors.append(f"duplicate claim_id '{claim_id}'") + seen.add(claim_id) + + etype = str(claim.get("evidence_type", "")).strip() + if etype and etype not in EVIDENCE_TYPES: + errors.append(f"{label}: invalid evidence_type '{etype}'") + + live_or_static = str(claim.get("live_or_static", "")).strip().upper() + + # Evidence path must resolve (skip external run ids like CI runs). + for rel in _as_path_list(claim.get("evidence_path")): + if not _resolve(rel).exists(): + errors.append(f"{label}: evidence_path not found: {rel}") + + # A LIVE claim cannot be backed only by documentation. + if live_or_static == "LIVE": + if _evidence_is_doc_only(claim): + errors.append(f"{label}: LIVE claim has documentation-only evidence") + if etype in {"CONFIGURATION", "CODE_INSPECTION"}: + errors.append(f"{label}: LIVE claim backed only by static {etype}") + + # Blocked work must not be presented as complete. + status = str(claim.get("status", "")).strip().upper() + if status == "BLOCKED": + if live_or_static == "LIVE" and etype in LIVE_EVIDENCE_TYPES: + errors.append(f"{label}: BLOCKED claim presented with live evidence") + completed_flag = claim.get("completed") + if completed_flag is True: + errors.append(f"{label}: BLOCKED claim marked completed=true") + if status and status not in CLAIM_STATUSES: + errors.append(f"{label}: invalid status '{status}'") + + return errors + + +def validate_all() -> list[str]: + """Validate both the manifest and the evidence index.""" + errors: list[str] = [] + try: + errors.extend(validate_manifest()) + except ReferenceError as exc: + errors.append(str(exc)) + try: + errors.extend(validate_evidence_index()) + except ReferenceError as exc: + errors.append(str(exc)) + return errors + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate the Atlas reference package") + parser.add_argument( + "--only", + choices=["manifest", "evidence", "all"], + default="all", + help="which artifact to validate (default: all)", + ) + args = parser.parse_args(argv) + + if args.only == "manifest": + errors = validate_manifest() + elif args.only == "evidence": + errors = validate_evidence_index() + else: + errors = validate_all() + + if errors: + print("reference validation FAILED:", file=sys.stderr) + for err in errors: + print(f" - {err}", file=sys.stderr) + return 1 + print("reference validation OK: manifest + evidence index consistent") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/atlas/validation/__init__.py b/src/atlas/validation/__init__.py new file mode 100644 index 0000000..406978e --- /dev/null +++ b/src/atlas/validation/__init__.py @@ -0,0 +1,15 @@ +"""Validation package for Project Atlas.""" + +from atlas.validation.checks import ( + ValidationCheck, + ValidationReport, + validate_anomaly_detection, + validate_loaded_run, +) + +__all__ = [ + "ValidationCheck", + "ValidationReport", + "validate_anomaly_detection", + "validate_loaded_run", +] diff --git a/src/atlas/validation/checks.py b/src/atlas/validation/checks.py new file mode 100644 index 0000000..ced33f9 --- /dev/null +++ b/src/atlas/validation/checks.py @@ -0,0 +1,504 @@ +"""Validation engine for Project Atlas Sprint 1. + +Purpose: + Evaluate loaded raw events and emit PASS/FAIL with detailed check results. + +Interactions: + Queries ``atlas_raw.events`` after load and compares findings against the + seeded anomaly profile for acceptance testing. + +Engineering principles: + - Fail loudly on unexpected data quality issues. + - Treat expected seeded anomalies as validation failures overall, while + acceptance tests verify each expected defect was detected. + +Common failure modes: + - Row count mismatch versus generated file. + - Missing partition rows for the primary event_date. + - Duplicate event_id count lower than seeded profile. + +Implementation choice: + SQL-first validation keeps checks close to the warehouse and prepares for + dbt tests in v0.3. Alternatives considered: pandas validation (extra runtime + dependency in Cloud Shell) and Great Expectations (future phase). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, table_fqn +from atlas.observability.cost import labeled_bigquery_client + + +@dataclass(frozen=True) +class ValidationCheck: + """Result of an individual validation check.""" + + name: str + status: str + expected: Any + actual: Any + message: str + + +@dataclass(frozen=True) +class ValidationReport: + """Aggregate validation report for a pipeline run.""" + + pipeline_run_id: str + overall_status: str + checks: list[ValidationCheck] + + def to_dict(self) -> dict[str, Any]: + """Convert the report to a JSON-serializable dictionary.""" + return { + "pipeline_run_id": self.pipeline_run_id, + "overall_status": self.overall_status, + "checks": [ + { + "name": check.name, + "status": check.status, + "expected": check.expected, + "actual": check.actual, + "message": check.message, + } + for check in self.checks + ], + } + + +def _query_scalar(client: bigquery.Client, sql: str, params: dict[str, Any] | None = None) -> Any: + """Execute a scalar query and return the first column of the first row.""" + query_params: list[bigquery.ScalarQueryParameter | bigquery.ArrayQueryParameter] = [] + for name, value in (params or {}).items(): + if isinstance(value, bool): + param_type = "BOOL" + elif isinstance(value, int): + param_type = "INT64" + elif isinstance(value, list): + param_type = "STRING" + value = value + else: + param_type = "STRING" + if isinstance(value, list): + query_params.append(bigquery.ArrayQueryParameter(name, "STRING", value)) + else: + query_params.append(bigquery.ScalarQueryParameter(name, param_type, value)) + rows = list( + client.query( + sql, + job_config=bigquery.QueryJobConfig(query_parameters=query_params), + ).result() + ) + if not rows: + return None + return next(iter(rows[0].values())) + + +def _exact_match(actual: Any, expected: int) -> bool: + """Return True when an observed anomaly count matches the seeded profile.""" + return actual == expected + + +def validate_loaded_run( + settings: AtlasSettings, + pipeline_run_id: str, + primary_event_date: str, + *, + client: bigquery.Client | None = None, + batch_id: str | None = None, + processing_date: str | None = None, + mode: str = "sprint1", +) -> ValidationReport: + """Validate a loaded pipeline run and return PASS/FAIL results. + + When ``batch_id`` is provided (Airflow mode), scope checks to the stable batch + identifier and compare future-dated rows against ``processing_date``. + """ + bq_client = client or labeled_bigquery_client(settings.gcp.project_id, "validation") + target = table_fqn(settings) + profile = settings.anomaly_profile + checks: list[ValidationCheck] = [] + + if batch_id is not None: + scope_filter = "batch_id = @batch_id" + scope_params: dict[str, Any] = {"batch_id": batch_id} + report_id = batch_id + else: + scope_filter = "pipeline_run_id = @run_id" + scope_params = {"run_id": pipeline_run_id} + report_id = pipeline_run_id + + row_count = _query_scalar( + bq_client, + f"SELECT COUNT(1) AS value FROM `{target}` WHERE {scope_filter}", + scope_params, + ) + checks.append( + ValidationCheck( + name="row_count", + status="PASS" if row_count == settings.validation.expected_event_count else "FAIL", + expected=settings.validation.expected_event_count, + actual=row_count, + message="Loaded row count matches generated event count.", + ) + ) + + partition_count = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND event_date = DATE(@event_date) + """, + {**scope_params, "event_date": primary_event_date}, + ) + checks.append( + ValidationCheck( + name="partition_presence", + status="PASS" if partition_count and partition_count > 0 else "FAIL", + expected=f">0 rows for {primary_event_date}", + actual=partition_count, + message="Primary partition contains rows for the run.", + ) + ) + + schema_ok = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) = 0 AS value + FROM `{target}` + WHERE {scope_filter} + AND ( + event_id IS NULL + OR event_name IS NULL + OR event_timestamp IS NULL + OR event_date IS NULL + OR ingested_at IS NULL + OR source_file IS NULL + OR pipeline_run_id IS NULL + ) + """, + scope_params, + ) + checks.append( + ValidationCheck( + name="schema_required_fields", + status="PASS" if schema_ok else "FAIL", + expected=True, + actual=bool(schema_ok), + message="Required non-null columns are populated.", + ) + ) + + if batch_id is not None: + batch_id_present = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND batch_id IS NULL + """, + scope_params, + ) + checks.append( + ValidationCheck( + name="batch_id_populated", + status="PASS" if batch_id_present == 0 else "FAIL", + expected=0, + actual=batch_id_present, + message="All batch-scoped rows carry batch_id.", + ) + ) + + distinct_event_ids = _query_scalar( + bq_client, + f""" + SELECT COUNT(DISTINCT event_id) AS value + FROM `{target}` + WHERE {scope_filter} + """, + scope_params, + ) + duplicate_rows = (row_count or 0) - (distinct_event_ids or 0) + expected_duplicate_groups = profile.expected_count("duplicate_event_ids") // 2 + checks.append( + ValidationCheck( + name="distinct_event_ids", + status="PASS" + if distinct_event_ids == settings.validation.expected_event_count - expected_duplicate_groups + else "FAIL", + expected=settings.validation.expected_event_count - expected_duplicate_groups, + actual=distinct_event_ids, + message="Distinct event_id count reconciles with duplicate rows.", + ) + ) + checks.append( + ValidationCheck( + name="duplicate_rows", + status="PASS" if duplicate_rows == expected_duplicate_groups else "FAIL", + expected=expected_duplicate_groups, + actual=duplicate_rows, + message="Duplicate physical rows reconcile with seeded duplicate event_ids.", + ) + ) + + duplicate_count = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM ( + SELECT event_id + FROM `{target}` + WHERE {scope_filter} + GROUP BY event_id + HAVING COUNT(1) > 1 + ) + """, + scope_params, + ) + expected_duplicates = profile.expected_count("duplicate_event_ids") + checks.append( + ValidationCheck( + name="duplicates", + status="FAIL" if duplicate_count else "PASS", + expected=f"detect {expected_duplicates // 2} duplicate id groups", + actual=duplicate_count, + message="Duplicate event_id values detected.", + ) + ) + + null_user_ids = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND user_id IS NULL + """, + scope_params, + ) + expected_null_users = profile.expected_count("null_user_ids") + checks.append( + ValidationCheck( + name="null_user_ids", + status="FAIL" if null_user_ids else "PASS", + expected=f"detect {expected_null_users} null user_id rows", + actual=null_user_ids, + message="Null user_id values detected.", + ) + ) + + invalid_countries = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND country_code NOT IN UNNEST(@valid_countries) + """, + { + **scope_params, + "valid_countries": profile.valid_country_codes, + }, + ) + expected_invalid_countries = profile.expected_count("invalid_country_codes") + checks.append( + ValidationCheck( + name="invalid_country_codes", + status="FAIL" if invalid_countries else "PASS", + expected=f"detect {expected_invalid_countries} invalid countries", + actual=invalid_countries, + message="Invalid country_code values detected.", + ) + ) + + if batch_id is not None and processing_date is not None: + future_timestamps = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND event_date > DATE(@processing_date) + """, + {**scope_params, "processing_date": processing_date}, + ) + else: + future_timestamps = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND event_date > CURRENT_DATE() + """, + scope_params, + ) + expected_future = profile.expected_count("future_timestamps") + checks.append( + ValidationCheck( + name="future_timestamps", + status="FAIL" if future_timestamps else "PASS", + expected=f"detect {expected_future} future-dated rows", + actual=future_timestamps, + message="Future calendar-date event_date values detected.", + ) + ) + + other_partition_rows = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND event_date != DATE(@event_date) + """, + {**scope_params, "event_date": primary_event_date}, + ) + checks.append( + ValidationCheck( + name="partition_reconciliation", + status="PASS" + if (partition_count or 0) + (other_partition_rows or 0) == (row_count or 0) + else "FAIL", + expected={ + "primary_event_date_rows": partition_count, + "other_partition_rows": other_partition_rows, + "total_rows": row_count, + }, + actual={ + "primary_event_date_rows": partition_count, + "other_partition_rows": other_partition_rows, + "total_rows": row_count, + }, + message=("Primary-partition rows plus other-partition rows reconcile to total row count."), + ) + ) + + late_arrivals = _query_scalar( + bq_client, + f""" + SELECT COUNT(1) AS value + FROM `{target}` + WHERE {scope_filter} + AND event_date < DATE(event_timestamp) + """, + scope_params, + ) + expected_late = profile.expected_count("late_arriving_events") + checks.append( + ValidationCheck( + name="late_arriving_events", + status="FAIL" if late_arrivals else "PASS", + expected=f"detect {expected_late} late-arriving rows", + actual=late_arrivals, + message="Late-arriving event_date values detected.", + ) + ) + + if mode == "airflow" and batch_id is not None: + overall_status = ( + "PASS" + if all( + check.status == "PASS" + for check in checks + if check.name + not in { + "duplicates", + "null_user_ids", + "invalid_country_codes", + "future_timestamps", + "late_arriving_events", + } + ) + else "FAIL" + ) + else: + overall_status = "PASS" if all(check.status == "PASS" for check in checks) else "FAIL" + + return ValidationReport( + pipeline_run_id=report_id, + overall_status=overall_status, + checks=checks, + ) + + +def validate_anomaly_detection(report: ValidationReport, settings: AtlasSettings) -> ValidationReport: + """Build acceptance-oriented checks proving expected anomalies were detected.""" + profile = settings.anomaly_profile + by_name = {check.name: check for check in report.checks} + acceptance_checks = [] + + duplicate_actual = by_name["duplicates"].actual or 0 + expected_duplicate_groups = profile.expected_count("duplicate_event_ids") // 2 + acceptance_checks.append( + ValidationCheck( + name="acceptance_duplicate_detection", + status="PASS" if _exact_match(duplicate_actual, expected_duplicate_groups) else "FAIL", + expected=expected_duplicate_groups, + actual=duplicate_actual, + message="Expected duplicate event_id groups were detected exactly.", + ) + ) + + null_actual = by_name["null_user_ids"].actual or 0 + expected_null_users = profile.expected_count("null_user_ids") + acceptance_checks.append( + ValidationCheck( + name="acceptance_null_user_detection", + status="PASS" if _exact_match(null_actual, expected_null_users) else "FAIL", + expected=expected_null_users, + actual=null_actual, + message="Expected null user_id rows were detected exactly.", + ) + ) + + invalid_country_actual = by_name["invalid_country_codes"].actual or 0 + expected_invalid_countries = profile.expected_count("invalid_country_codes") + acceptance_checks.append( + ValidationCheck( + name="acceptance_invalid_country_detection", + status="PASS" if _exact_match(invalid_country_actual, expected_invalid_countries) else "FAIL", + expected=expected_invalid_countries, + actual=invalid_country_actual, + message="Expected invalid country_code rows were detected exactly.", + ) + ) + + future_actual = by_name["future_timestamps"].actual or 0 + expected_future = profile.expected_count("future_timestamps") + acceptance_checks.append( + ValidationCheck( + name="acceptance_future_timestamp_detection", + status="PASS" if _exact_match(future_actual, expected_future) else "FAIL", + expected=expected_future, + actual=future_actual, + message="Expected future-dated rows were detected exactly.", + ) + ) + + late_actual = by_name["late_arriving_events"].actual or 0 + expected_late = profile.expected_count("late_arriving_events") + acceptance_checks.append( + ValidationCheck( + name="acceptance_late_arrival_detection", + status="PASS" if _exact_match(late_actual, expected_late) else "FAIL", + expected=expected_late, + actual=late_actual, + message="Expected late-arriving rows were detected exactly.", + ) + ) + + merged_checks = report.checks + acceptance_checks + return ValidationReport( + pipeline_run_id=report.pipeline_run_id, + overall_status=report.overall_status, + checks=merged_checks, + ) diff --git a/src/atlas/validation/schema_versions.py b/src/atlas/validation/schema_versions.py new file mode 100644 index 0000000..1d589d8 --- /dev/null +++ b/src/atlas/validation/schema_versions.py @@ -0,0 +1,75 @@ +"""Event schema-version discrimination and normalization (Sprint 6, ADR-015). + +Atlas raw events carry an optional ``schema_version`` discriminator (absent +means version 1, the Sprint 1 contract). Normalization maps every supported +version onto the current logical schema explicitly: + +- unknown versions are rejected, never guessed; +- unknown fields are rejected, never silently dropped (no silent coercion); +- fields added by a newer version are backfilled as None for older inputs so + consumers see one stable shape with explicit nullability. +""" + +from __future__ import annotations + +from typing import Any + +# Version 1: the original Sprint 1 event contract. +_V1_FIELDS = frozenset( + { + "event_id", + "event_type", + "event_timestamp", + "user_id", + "country_code", + "device_type", + "session_id", + "payload_size_bytes", + "batch_id", + "processing_date", + } +) + +# Version 2: version 1 plus one approved additive nullable field (S6-SCH-001 +# flow). The discriminator itself is part of the v2 contract. +_V2_ONLY_FIELDS = frozenset({"schema_version", "client_app_version"}) + +SUPPORTED_SCHEMA_VERSIONS: dict[int, frozenset[str]] = { + 1: _V1_FIELDS, + 2: _V1_FIELDS | _V2_ONLY_FIELDS, +} +CURRENT_SCHEMA_VERSION = 2 + + +class SchemaVersionError(ValueError): + """An event failed schema-version discrimination or normalization.""" + + +def detect_schema_version(event: dict[str, Any]) -> int: + """Return the event's declared version (absent discriminator == 1).""" + raw = event.get("schema_version", 1) + try: + version = int(raw) + except (TypeError, ValueError) as exc: + raise SchemaVersionError(f"schema_version {raw!r} is not an integer") from exc + if version not in SUPPORTED_SCHEMA_VERSIONS: + raise SchemaVersionError( + f"unsupported schema_version {version}; supported: {sorted(SUPPORTED_SCHEMA_VERSIONS)}" + ) + return version + + +def normalize_event(event: dict[str, Any]) -> dict[str, Any]: + """Map one event of any supported version onto the current logical schema.""" + version = detect_schema_version(event) + allowed = SUPPORTED_SCHEMA_VERSIONS[version] + unknown = set(event) - allowed + if unknown: + raise SchemaVersionError( + f"fields {sorted(unknown)} are not part of schema version {version}; " + "unknown fields are rejected, not silently coerced" + ) + current_fields = SUPPORTED_SCHEMA_VERSIONS[CURRENT_SCHEMA_VERSION] + normalized = {field: event.get(field) for field in sorted(current_fields)} + normalized["schema_version"] = version + return normalized diff --git a/src/atlas/validation/warehouse.py b/src/atlas/validation/warehouse.py new file mode 100644 index 0000000..0f8c0be --- /dev/null +++ b/src/atlas/validation/warehouse.py @@ -0,0 +1,274 @@ +"""Batch-scoped warehouse validation for the Airflow validate_warehouse step. + +Purpose: + Replace the Sprint 3 hardcoded PASS with real reconciliation between the + raw layer, the dbt classification layer, the fact table, and the marts. + +Interactions: + Called by ``scripts/atlas_step_runner.py`` (step ``validate_warehouse``) + after ``dbt_build`` succeeds. Queries BigQuery directly; it never mutates + warehouse state. + +Engineering principles: + - Every check is batch-scoped where the model carries batch identity, so + historical backfills validate identically to same-day runs. + - The step fails loudly: any FAILed check makes the Airflow task fail. +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from typing import Any + +from google.cloud import bigquery + +from atlas.config.settings import AtlasSettings, load_settings, table_fqn +from atlas.observability.cost import labeled_bigquery_client + + +@dataclass(frozen=True) +class WarehouseCheck: + """Result of one warehouse reconciliation check.""" + + name: str + status: str + expected: Any + actual: Any + message: str + + +@dataclass(frozen=True) +class WarehouseReport: + """Aggregate warehouse validation result for one batch.""" + + batch_id: str + overall_status: str + checks: list[WarehouseCheck] + + def to_dict(self) -> dict[str, Any]: + return { + "batch_id": self.batch_id, + "overall_status": self.overall_status, + "checks": [ + { + "name": c.name, + "status": c.status, + "expected": c.expected, + "actual": c.actual, + "message": c.message, + } + for c in self.checks + ], + } + + +def _dbt_dataset_prefix() -> str: + """Return the dbt target dataset prefix (default ``atlas``).""" + return os.environ.get("ATLAS_DBT_DATASET", "atlas") + + +def warehouse_table(project_id: str, layer: str, table: str) -> str: + """Return the fully qualified name of one dbt-managed warehouse table.""" + return f"{project_id}.{_dbt_dataset_prefix()}_{layer}.{table}" + + +def _scalar(client: bigquery.Client, sql: str, params: dict[str, str]) -> Any: + job_config = bigquery.QueryJobConfig( + query_parameters=[ + bigquery.ScalarQueryParameter(name, "STRING", value) for name, value in params.items() + ] + ) + rows = list(client.query(sql, job_config=job_config).result()) + if not rows: + return None + return next(iter(rows[0].values())) + + +def validate_warehouse( + batch_id: str, + processing_date: str, + settings: AtlasSettings | None = None, + *, + client: bigquery.Client | None = None, +) -> WarehouseReport: + """Run batch-scoped reconciliation across raw, classification, fact, and marts.""" + settings = settings or load_settings() + bq = client or labeled_bigquery_client(settings.gcp.project_id, "validation") + project = settings.gcp.project_id + raw = table_fqn(settings) + classification = warehouse_table(project, "intermediate", "int_event_classification") + accepted = warehouse_table(project, "intermediate", "int_accepted_events") + rejected = warehouse_table(project, "quarantine", "int_rejected_events") + fact = warehouse_table(project, "core", "fct_events") + dim_users = warehouse_table(project, "core", "dim_users") + dim_countries = warehouse_table(project, "core", "dim_countries") + mart = warehouse_table(project, "marts", "mart_daily_event_metrics") + scope = {"batch_id": batch_id} + checks: list[WarehouseCheck] = [] + + def add(name: str, passed: bool, expected: Any, actual: Any, message: str) -> None: + checks.append( + WarehouseCheck( + name=name, + status="PASS" if passed else "FAIL", + expected=expected, + actual=actual, + message=message, + ) + ) + + raw_count = _scalar(bq, f"SELECT COUNT(1) FROM `{raw}` WHERE batch_id = @batch_id", scope) + add( + "batch_nonempty", + bool(raw_count), + "> 0 raw rows", + raw_count, + "The validated batch exists in the raw layer (guards against trivially passing on a missing batch).", + ) + + classified_count = _scalar( + bq, f"SELECT COUNT(1) FROM `{classification}` WHERE batch_id = @batch_id", scope + ) + add( + "raw_equals_classification", + raw_count == classified_count, + raw_count, + classified_count, + "Every batch-scoped raw row is classified exactly once.", + ) + + accepted_count = _scalar(bq, f"SELECT COUNT(1) FROM `{accepted}` WHERE batch_id = @batch_id", scope) + rejected_count = _scalar(bq, f"SELECT COUNT(1) FROM `{rejected}` WHERE batch_id = @batch_id", scope) + add( + "accepted_plus_rejected_equals_raw", + (accepted_count or 0) + (rejected_count or 0) == (raw_count or 0), + raw_count, + {"accepted": accepted_count, "rejected": rejected_count}, + "Accepted plus rejected rows reconcile to the raw batch.", + ) + + fact_count = _scalar( + bq, + f""" + SELECT COUNT(1) + FROM `{fact}` f + INNER JOIN `{accepted}` a USING (event_id) + WHERE a.batch_id = @batch_id + """, + scope, + ) + add( + "accepted_equals_fact", + fact_count == accepted_count, + accepted_count, + fact_count, + "Batch-scoped fact rows reconcile to accepted events.", + ) + + duplicate_fact_ids = _scalar( + bq, + f""" + SELECT COUNT(1) + FROM ( + SELECT event_id + FROM `{fact}` + GROUP BY event_id + HAVING COUNT(1) > 1 + ) + """, + {}, + ) + add( + "fact_event_ids_unique", + duplicate_fact_ids == 0, + 0, + duplicate_fact_ids, + "fct_events.event_id is globally unique.", + ) + + orphan_users = _scalar( + bq, + f""" + SELECT COUNT(1) + FROM `{fact}` f + LEFT JOIN `{dim_users}` u USING (user_id) + WHERE f.user_id IS NOT NULL + AND u.user_id IS NULL + """, + {}, + ) + add( + "fact_user_fk_resolves", + orphan_users == 0, + 0, + orphan_users, + "Every non-null fct_events.user_id resolves in dim_users.", + ) + + orphan_countries = _scalar( + bq, + f""" + SELECT COUNT(1) + FROM `{fact}` f + LEFT JOIN `{dim_countries}` c USING (country_code) + WHERE c.country_code IS NULL + """, + {}, + ) + add( + "fact_country_fk_resolves", + orphan_countries == 0, + 0, + orphan_countries, + "Every fct_events.country_code resolves in dim_countries.", + ) + + mart_total = _scalar(bq, f"SELECT COALESCE(SUM(event_count), 0) FROM `{mart}`", {}) + fact_total = _scalar(bq, f"SELECT COUNT(1) FROM `{fact}`", {}) + add( + "mart_totals_reconcile", + mart_total == fact_total, + fact_total, + mart_total, + "mart_daily_event_metrics total event_count equals fct_events row count.", + ) + + bad_processing_dates = _scalar( + bq, + f""" + SELECT COUNT(1) + FROM `{raw}` + WHERE batch_id = @batch_id + AND (processing_date IS NULL OR processing_date != DATE(@processing_date)) + """, + {**scope, "processing_date": processing_date}, + ) + add( + "processing_date_semantics", + bad_processing_dates == 0, + 0, + bad_processing_dates, + "Every batch-scoped raw row carries the batch's logical processing_date.", + ) + + null_batch_ids = _scalar( + bq, + f""" + SELECT COUNT(1) + FROM `{classification}` + WHERE batch_id = @batch_id + AND (event_id IS NULL OR pipeline_run_id IS NULL) + """, + scope, + ) + add( + "batch_lineage_semantics", + null_batch_ids == 0, + 0, + null_batch_ids, + "Batch-scoped classification rows carry event and pipeline lineage.", + ) + + overall = "PASS" if all(c.status == "PASS" for c in checks) else "FAIL" + return WarehouseReport(batch_id=batch_id, overall_status=overall, checks=checks) diff --git a/tests/acceptance/test_sprint1_acceptance.py b/tests/acceptance/test_sprint1_acceptance.py new file mode 100644 index 0000000..0852ea2 --- /dev/null +++ b/tests/acceptance/test_sprint1_acceptance.py @@ -0,0 +1,50 @@ +"""Acceptance tests for Sprint 1 definition of done (local portions).""" + +from __future__ import annotations + +import json +from pathlib import Path + +from atlas.config.settings import load_settings +from atlas.generator.events import generate_events +from atlas.pipeline.orchestrator import run_pipeline + + +def test_sprint1_local_artifacts_exist() -> None: + settings = load_settings() + result = run_pipeline(settings, skip_upload=True, skip_load=True, skip_validation=True) + + assert result.generation.output_path.exists() + lines = result.generation.output_path.read_text(encoding="utf-8").strip().splitlines() + assert len(lines) == 50000 + + first = json.loads(lines[0]) + assert "ingested_at" not in first + assert { + "event_id", + "user_id", + "event_name", + "event_timestamp", + "event_date", + "country_code", + "platform", + "app_version", + }.issubset(first.keys()) + + assert result.log_file.exists() + log_lines = result.log_file.read_text(encoding="utf-8").strip().splitlines() + assert any('"step": "generate"' in line for line in log_lines) + + +def test_generator_anomaly_profile_matches_config() -> None: + settings = load_settings() + result = generate_events(settings) + profile = settings.anomaly_profile + assert result.anomaly_counts["null_user_ids"] == profile.expected_count("null_user_ids") + assert result.anomaly_counts["invalid_country_codes"] == profile.expected_count("invalid_country_codes") + + +def test_repository_layout() -> None: + root = Path(__file__).resolve().parents[2] + for relative in ["src", "data", "docs", "sql", "tests", "requirements.txt", "README.md", ".gitignore"]: + assert (root / relative).exists() diff --git a/tests/acceptance/test_sprint2_dbt_environment.py b/tests/acceptance/test_sprint2_dbt_environment.py new file mode 100644 index 0000000..ccf735c --- /dev/null +++ b/tests/acceptance/test_sprint2_dbt_environment.py @@ -0,0 +1,155 @@ +"""Static acceptance checks for the Sprint 2 Atlas dbt environment.""" + +from __future__ import annotations + +import json +import subprocess +import unittest +from pathlib import Path + +REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +ATLAS_ROOT = REPOSITORY_ROOT +DBT_ROOT = ATLAS_ROOT / "dbt" +DBT_PROJECT_ROOT = DBT_ROOT / "atlas_dbt" + + +def _ignore_rules(path: Path) -> set[str]: + return { + line.strip() + for line in path.read_text(encoding="utf-8").splitlines() + if line.strip() and not line.lstrip().startswith("#") + } + + +class Sprint2DbtEnvironmentAcceptanceTest(unittest.TestCase): + def test_environment_artifacts_exist(self) -> None: + required_files = [ + DBT_ROOT / "requirements-dbt.txt", + DBT_PROJECT_ROOT / "dbt_project.yml", + DBT_PROJECT_ROOT / "packages.yml", + DBT_PROJECT_ROOT / "package-lock.yml", + DBT_PROJECT_ROOT / "profiles.yml.example", + ATLAS_ROOT / "scripts" / "setup_dbt.sh", + ] + + missing = [str(path.relative_to(REPOSITORY_ROOT)) for path in required_files if not path.is_file()] + self.assertEqual([], missing, f"missing Sprint 2 dbt environment files: {missing}") + + def test_root_and_atlas_ignore_rules_protect_dbt_secrets_and_artifacts(self) -> None: + shared_security_rules = { + ".env.*", + "credentials/", + "secrets/", + "service-account*.json", + "*-key.json", + } + root_rules = _ignore_rules(REPOSITORY_ROOT / ".gitignore") + atlas_rules = _ignore_rules(ATLAS_ROOT / ".gitignore") + + self.assertTrue(shared_security_rules <= root_rules) + self.assertTrue(shared_security_rules <= atlas_rules) + self.assertTrue( + { + ".venv-dbt/", + "dbt/atlas_dbt/target/", + "dbt/atlas_dbt/logs/", + "dbt/atlas_dbt/dbt_packages/", + "dbt/atlas_dbt/profiles.yml", + } + <= root_rules + ) + self.assertTrue( + { + ".venv-dbt/", + "dbt/atlas_dbt/target/", + "dbt/atlas_dbt/logs/", + "dbt/atlas_dbt/dbt_packages/", + "dbt/atlas_dbt/profiles.yml", + } + <= atlas_rules + ) + + def test_safe_examples_and_arbitrary_json_remain_trackable(self) -> None: + safe_paths = [ + ".env.example", + "dbt/atlas_dbt/profiles.yml.example", + "config/events.json", + ] + for relative_path in safe_paths: + result = subprocess.run( + ["git", "check-ignore", "--no-index", "--quiet", relative_path], + cwd=REPOSITORY_ROOT, + check=False, + ) + self.assertEqual(1, result.returncode, f"{relative_path} must remain trackable") + + def test_dbt_versions_package_and_project_defaults_are_pinned(self) -> None: + requirements = (DBT_ROOT / "requirements-dbt.txt").read_text(encoding="utf-8").splitlines() + self.assertEqual(["dbt-core==1.11.12", "dbt-bigquery==1.11.3"], requirements) + + packages = (DBT_PROJECT_ROOT / "packages.yml").read_text(encoding="utf-8") + package_lock = (DBT_PROJECT_ROOT / "package-lock.yml").read_text(encoding="utf-8") + for content in (packages, package_lock): + self.assertIn("dbt-labs/dbt_utils", content) + self.assertIn("version: 1.4.1", content) + + project = (DBT_PROJECT_ROOT / "dbt_project.yml").read_text(encoding="utf-8") + self.assertIn("name: atlas_dbt", project) + self.assertIn("profile: atlas_dbt", project) + self.assertIn("lookback_days: 3", project) + self.assertIn('validated_run_id: "atlas-20260714T163527Z-19a0e4f6"', project) + self.assertNotIn("snapshot-paths:", project) + self.assertFalse((DBT_PROJECT_ROOT / "snapshots").exists()) + + def test_profile_example_is_oauth_only_and_uses_environment_variables(self) -> None: + profile = (DBT_PROJECT_ROOT / "profiles.yml.example").read_text(encoding="utf-8") + self.assertIn("method: oauth", profile) + self.assertIn("env_var('ATLAS_GCP_PROJECT_ID')", profile) + self.assertIn("env_var('ATLAS_DBT_DATASET', 'atlas')", profile) + self.assertIn("env_var('DBT_LOCATION')", profile) + self.assertNotIn("service-account", profile) + self.assertNotIn("keyfile:", profile) + + def test_setup_script_and_atlas_mcp_use_the_isolated_environment(self) -> None: + setup_script_path = ATLAS_ROOT / "scripts" / "setup_dbt.sh" + setup_script = setup_script_path.read_text(encoding="utf-8") + self.assertTrue(setup_script_path.stat().st_mode & 0o100) + for required_text in [ + "set -Eeuo pipefail", + "require_command gcloud", + "require_command bq", + "require_command git", + "python3", + "atlas_raw", + "DBT_LOCATION", + ".venv-dbt", + "import ensurepip", + "-m pip --version", + "requirements-dbt.txt", + 'dbt" deps', + "${DBT_PROFILES_DIR:-$HOME/.dbt}", + "GOOGLE_APPLICATION_CREDENTIALS", + "service-account", + "chmod 600", + 'dbt" debug', + ]: + self.assertIn(required_text, setup_script) + self.assertNotIn("set -x", setup_script) + + mcp_config = json.loads((REPOSITORY_ROOT / ".cursor" / "mcp.json").read_text(encoding="utf-8")) + servers = mcp_config["mcpServers"] + self.assertTrue({"bigquery", "dbt-atlas"} <= servers.keys()) + atlas_env = servers["dbt-atlas"]["env"] + self.assertEqual( + "${workspaceFolder}/dbt/atlas_dbt", + atlas_env["DBT_PROJECT_DIR"], + ) + self.assertEqual( + "${workspaceFolder}/.venv-dbt/bin/dbt", + atlas_env["DBT_PATH"], + ) + self.assertNotIn("DBT_PROFILES_DIR", atlas_env) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/acceptance/test_sprint2_dbt_warehouse.py b/tests/acceptance/test_sprint2_dbt_warehouse.py new file mode 100644 index 0000000..f6eaf93 --- /dev/null +++ b/tests/acceptance/test_sprint2_dbt_warehouse.py @@ -0,0 +1,84 @@ +"""Static acceptance checks for the Sprint 2 Atlas dbt warehouse artifacts.""" + +from __future__ import annotations + +import unittest +from pathlib import Path + +REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +ATLAS_ROOT = REPOSITORY_ROOT +DBT_PROJECT_ROOT = ATLAS_ROOT / "dbt" / "atlas_dbt" + + +class Sprint2DbtWarehouseAcceptanceTest(unittest.TestCase): + def test_required_models_seeds_and_tests_exist(self) -> None: + required_paths = [ + DBT_PROJECT_ROOT / "models/sources/sources.yml", + DBT_PROJECT_ROOT / "models/staging/stg_events.sql", + DBT_PROJECT_ROOT / "models/staging/staging.yml", + DBT_PROJECT_ROOT / "models/intermediate/int_event_classification.sql", + DBT_PROJECT_ROOT / "models/intermediate/int_accepted_events.sql", + DBT_PROJECT_ROOT / "models/intermediate/int_rejected_events.sql", + DBT_PROJECT_ROOT / "models/intermediate/intermediate.yml", + DBT_PROJECT_ROOT / "models/core/dim_users.sql", + DBT_PROJECT_ROOT / "models/core/dim_countries.sql", + DBT_PROJECT_ROOT / "models/core/fct_events.sql", + DBT_PROJECT_ROOT / "models/core/core.yml", + DBT_PROJECT_ROOT / "models/marts/mart_daily_event_metrics.sql", + DBT_PROJECT_ROOT / "models/marts/marts.yml", + DBT_PROJECT_ROOT / "seeds/valid_country_codes.csv", + DBT_PROJECT_ROOT / "seeds/seeds.yml", + DBT_PROJECT_ROOT / "tests/assert_source_anomaly_profile.sql", + DBT_PROJECT_ROOT / "tests/assert_raw_classification_reconciliation.sql", + DBT_PROJECT_ROOT / "tests/assert_fact_rejected_reconciliation.sql", + DBT_PROJECT_ROOT / "tests/assert_mart_fact_reconciliation.sql", + ATLAS_ROOT / "scripts/run_dbt_sprint2.sh", + ATLAS_ROOT / "scripts/validate_dbt_sprint2.sh", + ATLAS_ROOT / "docs/architecture-sprint2.md", + ATLAS_ROOT / "docs/model-catalog-sprint2.md", + ATLAS_ROOT / "docs/runbook-sprint2.md", + ATLAS_ROOT / "docs/validation-report-sprint2.md", + ATLAS_ROOT / "docs/adr/ADR-002-isolated-atlas-dbt-project.md", + ATLAS_ROOT / "docs/adr/ADR-003-corrected-temporal-semantics.md", + ATLAS_ROOT / "docs/adr/ADR-004-no-snapshots-sprint2.md", + ] + missing = [str(path.relative_to(REPOSITORY_ROOT)) for path in required_paths if not path.is_file()] + self.assertEqual([], missing, f"missing Sprint 2 warehouse files: {missing}") + + def test_dbt_project_declares_layer_schemas(self) -> None: + project = (DBT_PROJECT_ROOT / "dbt_project.yml").read_text(encoding="utf-8") + for marker in [ + "+schema: staging", + "+schema: intermediate", + "+schema: core", + "+schema: marts", + "validated_run_id:", + ]: + self.assertIn(marker, project) + + def test_classification_and_scripts_are_executable(self) -> None: + for script_name in ("run_dbt_sprint2.sh", "validate_dbt_sprint2.sh"): + script_path = ATLAS_ROOT / "scripts" / script_name + self.assertTrue(script_path.stat().st_mode & 0o100, f"{script_name} must be executable") + + classification = (DBT_PROJECT_ROOT / "models/intermediate/int_event_classification.sql").read_text( + encoding="utf-8" + ) + staging = (DBT_PROJECT_ROOT / "models/staging/stg_events.sql").read_text(encoding="utf-8") + for snippet in [ + "missing_user_id", + "invalid_country_code", + "future_dated", + "duplicate_extra", + ]: + self.assertIn(snippet, classification) + for snippet in [ + "is_backdated_event_date", + "has_event_date_timestamp_mismatch", + "is_event_time_late_arriving", + ]: + self.assertIn(snippet, staging) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/airflow/test_callback_timing.py b/tests/airflow/test_callback_timing.py new file mode 100644 index 0000000..8e670e5 --- /dev/null +++ b/tests/airflow/test_callback_timing.py @@ -0,0 +1,93 @@ +"""Failed-task timing derivation tests (Sprint 6, Phase 1). + +Sprint 5 limitation: FAILED rows written by the failure callback carried NULL +started_at/completed_at/duration_ms. Timing must now come from reliable +evidence (Airflow task-instance timestamps) with recorded provenance — and +must stay NULL when no reliable evidence exists. +""" + +from __future__ import annotations + +from datetime import UTC, datetime, timedelta + +import pytest +from atlas_orchestration.callbacks import build_callback_context, derive_callback_timing + + +class _TaskInstance: + def __init__(self, start_date=None, end_date=None) -> None: + self.task_id = "dbt_build" + self.try_number = 1 + self.state = "failed" + self.start_date = start_date + self.end_date = end_date + + +def _meta(start_date=None, end_date=None) -> dict: + return build_callback_context({"task_instance": _TaskInstance(start_date, end_date), "dag_run": None}) + + +def test_exact_timing_from_task_instance_dates() -> None: + start = datetime(2026, 7, 19, 6, 33, 54, tzinfo=UTC) + end = start + timedelta(seconds=90) + timing = derive_callback_timing(_meta(start, end)) + assert timing["timing_source"] == "airflow_task_instance" + assert timing["timing_confidence"] == "exact" + assert timing["started_at"] == start.isoformat() + assert timing["completed_at"] == end.isoformat() + assert timing["duration_ms"] == 90_000 + + +def test_partial_timing_bounds_completion_with_callback_clock() -> None: + start = datetime.now(tz=UTC) - timedelta(seconds=30) + timing = derive_callback_timing(_meta(start, None)) + assert timing["timing_confidence"] == "partial" + assert timing["started_at"] == start.isoformat() + assert timing["completed_at"] is not None + assert timing["duration_ms"] >= 29_000 + + +def test_no_evidence_preserves_null_and_records_none_confidence() -> None: + """Timestamps are never invented: no start date means NULL timing.""" + timing = derive_callback_timing(_meta(None, None)) + assert timing["started_at"] is None + assert timing["completed_at"] is None + assert timing["duration_ms"] is None + assert timing["timing_source"] == "airflow_task_instance" + assert timing["timing_confidence"] == "none" + + +def test_negative_clock_skew_clamped_to_zero() -> None: + start = datetime.now(tz=UTC) + timing = derive_callback_timing(_meta(start, start - timedelta(seconds=5))) + assert timing["duration_ms"] == 0 + + +def test_task_event_record_rejects_unknown_timing_vocabulary() -> None: + from atlas.config.settings import load_settings + from atlas.ops.task_events import TaskEventRecord, upsert_task_event + + class _Client: + def query(self, sql, job_config=None): # pragma: no cover - must not be reached + raise AssertionError("validation must fail before any query") + + record = TaskEventRecord( + pipeline_run_id="pr-1", + task_id="dbt_build", + attempt_number=1, + event_type="FAILED", + timing_source="vibes", + ) + with pytest.raises(ValueError, match="Unsupported timing_source"): + upsert_task_event(record, load_settings(), client=_Client()) + + record2 = TaskEventRecord( + pipeline_run_id="pr-1", + task_id="dbt_build", + attempt_number=1, + event_type="FAILED", + timing_source="airflow_task_instance", + timing_confidence="pretty_sure", + ) + with pytest.raises(ValueError, match="Unsupported timing_confidence"): + upsert_task_event(record2, load_settings(), client=_Client()) diff --git a/tests/airflow/test_commands.py b/tests/airflow/test_commands.py new file mode 100644 index 0000000..505c3f2 --- /dev/null +++ b/tests/airflow/test_commands.py @@ -0,0 +1,16 @@ +"""Command builder tests.""" + +from __future__ import annotations + +from atlas_orchestration.commands import atlas_root, run_atlas_step_command + + +def test_run_atlas_step_command_includes_step_and_context() -> None: + cmd = run_atlas_step_command("generate_events", {"batch_id": "atlas-20260715"}) + assert "run_atlas_step.sh" in cmd + assert "generate_events" in cmd + + +def test_atlas_root_honors_env(monkeypatch) -> None: + monkeypatch.setenv("ATLAS_ROOT", "/composer/data/project-atlas") + assert str(atlas_root()) == "/composer/data/project-atlas" diff --git a/tests/airflow/test_composer_path_configuration.py b/tests/airflow/test_composer_path_configuration.py new file mode 100644 index 0000000..60fdf4c --- /dev/null +++ b/tests/airflow/test_composer_path_configuration.py @@ -0,0 +1,27 @@ +"""Composer path configuration tests.""" + +from __future__ import annotations + +from pathlib import Path + + +def test_no_hardcoded_de_project_path_in_dags() -> None: + dags_dir = Path(__file__).resolve().parents[2] / "dags" + for path in dags_dir.rglob("*.py"): + content = path.read_text(encoding="utf-8") + assert "Atlas-GCP-Build" not in content + assert "~/project-atlas" not in content + + +def test_atlas_root_override_in_settings(tmp_path, monkeypatch) -> None: + monkeypatch.setenv("ATLAS_ROOT", str(tmp_path)) + from atlas.config.settings import atlas_root + + assert atlas_root() == tmp_path.resolve() + + +def test_command_paths_use_atlas_root_env(monkeypatch) -> None: + monkeypatch.setenv("ATLAS_ROOT", "/home/airflow/gcs/data/project-atlas") + from atlas_orchestration.commands import scripts_dir + + assert str(scripts_dir()).startswith("/home/airflow/gcs/data/project-atlas") diff --git a/tests/airflow/test_dag_import.py b/tests/airflow/test_dag_import.py new file mode 100644 index 0000000..f9247aa --- /dev/null +++ b/tests/airflow/test_dag_import.py @@ -0,0 +1,22 @@ +"""Airflow DAG import safety tests.""" + +from __future__ import annotations + +from pathlib import Path + + +def test_dag_file_exists() -> None: + dag_path = Path(__file__).resolve().parents[2] / "dags" / "atlas_batch_pipeline.py" + assert dag_path.exists() + + +def test_orchestration_helpers_import_without_airflow_when_mocked(monkeypatch) -> None: + monkeypatch.setenv("ATLAS_ROOT", str(Path(__file__).resolve().parents[2])) + from atlas_orchestration.context import resolve_run_context_dict + + ctx = resolve_run_context_dict( + airflow_run_id="manual__2026-07-15", + dag_id="atlas_batch_pipeline", + conf={"processing_date": "2026-07-15", "batch_id": "atlas-20260715"}, + ) + assert ctx["batch_id"] == "atlas-20260715" diff --git a/tests/airflow/test_dag_structure.py b/tests/airflow/test_dag_structure.py new file mode 100644 index 0000000..5d87ca2 --- /dev/null +++ b/tests/airflow/test_dag_structure.py @@ -0,0 +1,37 @@ +"""DAG structure assertions.""" + +from __future__ import annotations + + +def test_expected_task_chain_order() -> None: + expected = [ + "resolve_run_context", + "ensure_audit_resources", + "start_run_audit", + "preflight_environment", + "generate_events", + "upload_events", + "load_bigquery_raw", + "validate_raw_load", + "dbt_seed", + "dbt_source_freshness", + "dbt_build", + "validate_warehouse", + "publish_success_marker", + "write_run_summary", + ] + # Static contract documented for parse-safe environments without Airflow runtime. + assert len(expected) == 14 + assert expected[0] == "resolve_run_context" + assert expected[-1] == "write_run_summary" + + +def test_schedule_and_start_date_constants() -> None: + from pathlib import Path + + dag_file = Path(__file__).resolve().parents[2] / "dags" / "atlas_batch_pipeline.py" + source = dag_file.read_text(encoding="utf-8") + assert 'schedule="0 6 * * *"' in source + assert "catchup=False" in source + assert "max_active_runs=1" in source + assert "START_DATE = datetime(2026, 7, 1, tzinfo=UTC)" in source diff --git a/tests/airflow/test_finalizer.py b/tests/airflow/test_finalizer.py new file mode 100644 index 0000000..fcf6246 --- /dev/null +++ b/tests/airflow/test_finalizer.py @@ -0,0 +1,17 @@ +"""Finalizer reconciliation tests.""" + +from __future__ import annotations + +from atlas.ops.finalizer import finalizer_should_fail, reconcile_run_summary + + +def test_reconcile_run_summary_detects_mismatch() -> None: + local = {"pipeline_run_id": "a", "batch_id": "b", "status": "SUCCESS"} + audit = {"pipeline_run_id": "a", "batch_id": "b", "status": "FAILED"} + result = reconcile_run_summary(local, audit) + assert result["reconciled"] is False + + +def test_finalizer_should_fail_on_failed_status() -> None: + assert finalizer_should_fail({"status": "FAILED"}) is True + assert finalizer_should_fail({"status": "SUCCESS"}) is False diff --git a/tests/airflow/test_parse_safety.py b/tests/airflow/test_parse_safety.py new file mode 100644 index 0000000..2ace5c4 --- /dev/null +++ b/tests/airflow/test_parse_safety.py @@ -0,0 +1,13 @@ +"""Parse-time side-effect guards.""" + +from __future__ import annotations + +from pathlib import Path + + +def test_context_module_has_no_subprocess_or_network_imports() -> None: + path = Path(__file__).resolve().parents[2] / "dags" / "atlas_orchestration" / "context.py" + source = path.read_text(encoding="utf-8") + assert "subprocess" not in source + assert "google.cloud" not in source + assert "requests" not in source diff --git a/tests/airflow/test_run_atlas_step_ctx.py b/tests/airflow/test_run_atlas_step_ctx.py new file mode 100644 index 0000000..d944f96 --- /dev/null +++ b/tests/airflow/test_run_atlas_step_ctx.py @@ -0,0 +1,44 @@ +"""Regression tests for run_atlas_step.sh JSON run-context passing. + +Guards against the ``${2:-{}}`` bash default-value bug: bash parsed the default +as ``{`` plus a literal trailing ``}``, appending a stray ``}`` to a JSON object +argument and breaking ``json.loads`` in every orchestrated task ("Extra data"). + +The dispatcher parses the context before dispatching on the step name, so an +unrecognised step with a valid JSON context reaches the "Unknown step" branch +without importing GCP libraries or touching the network. +""" + +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +ATLAS_ROOT = Path(__file__).resolve().parents[2] +STEP_SCRIPT = ATLAS_ROOT / "scripts" / "run_atlas_step.sh" +_NOOP_STEP = "__regression_noop__" + + +def _run(*args: str) -> str: + result = subprocess.run( + ["bash", str(STEP_SCRIPT), _NOOP_STEP, *args], + capture_output=True, + text=True, + ) + return result.stdout + result.stderr + + +def test_json_context_is_passed_through_intact() -> None: + ctx = json.dumps({"batch_id": "atlas-20260718", "seed": 1, "upload_once": True}) + combined = _run(ctx) + # Reaching the "Unknown step" branch proves json.loads succeeded. + assert f"Unknown step: {_NOOP_STEP}" in combined, combined + assert "invalid JSON context" not in combined + assert "Extra data" not in combined + + +def test_missing_context_defaults_to_valid_json() -> None: + combined = _run() + assert f"Unknown step: {_NOOP_STEP}" in combined, combined + assert "invalid JSON context" not in combined diff --git a/tests/airflow/test_run_context.py b/tests/airflow/test_run_context.py new file mode 100644 index 0000000..b27fbab --- /dev/null +++ b/tests/airflow/test_run_context.py @@ -0,0 +1,52 @@ +"""Run context resolution tests.""" + +from __future__ import annotations + +from datetime import UTC, datetime, timedelta + +import pytest +from atlas_orchestration.context import resolve_processing_date, resolve_run_context_dict + + +def test_resolve_processing_date_from_manual_override() -> None: + assert resolve_processing_date(None, manual_processing_date="2026-07-10") == "2026-07-10" + + +def test_backfill_mode_when_manual_date_differs(monkeypatch) -> None: + monkeypatch.setenv("ATLAS_ROOT", "/tmp/atlas") + recent = (datetime.now(tz=UTC).date() - timedelta(days=2)).isoformat() + ctx = resolve_run_context_dict( + airflow_run_id="manual__1", + dag_id="atlas_batch_pipeline", + conf={"processing_date": recent, "batch_id": f"atlas-{recent.replace('-', '')}"}, + ) + assert ctx["backfill_mode"] is True + + +def test_backfill_beyond_policy_window_is_blocked(monkeypatch) -> None: + """S6-COST-002: an oversized backfill window is rejected without override.""" + from atlas.observability.cost_guards import BACKFILL_OVERRIDE_VAR, CostGuardViolation + + monkeypatch.setenv("ATLAS_ROOT", "/tmp/atlas") + monkeypatch.delenv(BACKFILL_OVERRIDE_VAR, raising=False) + old = (datetime.now(tz=UTC).date() - timedelta(days=30)).isoformat() + with pytest.raises(CostGuardViolation, match="exceeds the .*-day policy"): + resolve_run_context_dict( + airflow_run_id="manual__2", + dag_id="atlas_batch_pipeline", + conf={"processing_date": old, "batch_id": f"atlas-{old.replace('-', '')}"}, + ) + + +def test_backfill_beyond_policy_window_allowed_with_override(monkeypatch) -> None: + from atlas.observability.cost_guards import BACKFILL_OVERRIDE_VAR + + monkeypatch.setenv("ATLAS_ROOT", "/tmp/atlas") + monkeypatch.setenv(BACKFILL_OVERRIDE_VAR, "true") + old = (datetime.now(tz=UTC).date() - timedelta(days=30)).isoformat() + ctx = resolve_run_context_dict( + airflow_run_id="manual__3", + dag_id="atlas_batch_pipeline", + conf={"processing_date": old, "batch_id": f"atlas-{old.replace('-', '')}"}, + ) + assert ctx["backfill_mode"] is True diff --git a/tests/airflow/test_sprint4_hygiene.py b/tests/airflow/test_sprint4_hygiene.py new file mode 100644 index 0000000..5809852 --- /dev/null +++ b/tests/airflow/test_sprint4_hygiene.py @@ -0,0 +1,68 @@ +"""Regression tests for Sprint 4 Phase 1 hygiene fixes. + +Guards against reintroducing: +- `|| true` suppression on required Airflow gates in test_airflow_sprint3.sh, +- trigger-without-poll behavior in run_airflow_sprint3.sh, +- the hardcoded validate_warehouse PASS in atlas_step_runner.py, +- obsolete feature-branch checkouts in the README quick starts. +""" + +from __future__ import annotations + +import re +from pathlib import Path + +ATLAS_ROOT = Path(__file__).resolve().parents[2] + + +def _read(relative: str) -> str: + return (ATLAS_ROOT / relative).read_text(encoding="utf-8") + + +def test_airflow_gate_does_not_suppress_import_errors() -> None: + content = _read("scripts/test_airflow_sprint3.sh") + for line in content.splitlines(): + if "list-import-errors" in line and "import_errors=" not in line: + assert "|| true" not in line, f"import-error gate is suppressed: {line.strip()}" + assert "exit 1" in content, "gate must be able to fail" + assert "atlas_batch_pipeline" in content + + +def test_airflow_gate_fails_when_dag_missing() -> None: + content = _read("scripts/test_airflow_sprint3.sh") + # The registration check must be a hard gate, not a soft grep. + assert re.search(r"if\s+!\s+airflow dags list.*grep.*atlas_batch_pipeline", content, re.S) + + +def test_run_airflow_polls_to_terminal_state() -> None: + content = _read("scripts/run_airflow_sprint3.sh") + assert "dags state" in content, "must poll the triggered run" + assert "RUN_TIMEOUT_SECONDS" in content, "must bound the poll" + assert "dag_run_id" in content, "must capture the triggered run id" + assert "states-for-dag-run" in content, "must report failed tasks" + assert "query_pipeline_run" in content, "must report the audit row" + assert "run-summary.json" in content, "must report the local summary path" + # Success path exits 0, failure and timeout paths exit 1. + assert re.search(r"success\)\s*\n.*", content) + assert content.count("exit 1") >= 3 + + +def test_step_runner_validate_warehouse_is_real() -> None: + content = _read("scripts/atlas_step_runner.py") + assert "dbt build tests cover warehouse validation" not in content, ( + "validate_warehouse must not return a hardcoded PASS" + ) + assert "from atlas.validation.warehouse import validate_warehouse" in content + assert 'report.overall_status == "PASS"' in content + + +def test_readme_quick_starts_use_main() -> None: + content = _read("README.md") + assert "git checkout cursor/atlas-sprint-2-dbt-warehouse-3660" not in content + assert "git checkout cursor/atlas-sprint-3-airflow-orchestration-3660" not in content + assert "git checkout main" in content + + +def test_readme_release_table_includes_sprint3() -> None: + content = _read("README.md") + assert "atlas-sprint-3-complete" in content diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..72cd607 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,13 @@ +"""Pytest configuration for Project Atlas.""" + +from __future__ import annotations + +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +SRC = ROOT / "src" +DAGS = ROOT / "dags" +for path in (SRC, DAGS): + if str(path) not in sys.path: + sys.path.insert(0, str(path)) diff --git a/tests/integration/test_pipeline_local.py b/tests/integration/test_pipeline_local.py new file mode 100644 index 0000000..4328152 --- /dev/null +++ b/tests/integration/test_pipeline_local.py @@ -0,0 +1,21 @@ +"""Integration tests for local pipeline execution.""" + +from __future__ import annotations + +from atlas.config.settings import load_settings +from atlas.pipeline.orchestrator import run_pipeline + + +def test_local_generate_only_pipeline() -> None: + settings = load_settings() + result = run_pipeline( + settings, + skip_upload=True, + skip_load=True, + skip_validation=True, + ) + assert result.generation.event_count == 50000 + assert result.upload is None + assert result.load is None + assert result.validation is None + assert result.log_file.exists() diff --git a/tests/unit/test_audit.py b/tests/unit/test_audit.py new file mode 100644 index 0000000..a5d4e23 --- /dev/null +++ b/tests/unit/test_audit.py @@ -0,0 +1,31 @@ +"""Unit tests for operational audit helpers.""" + +from __future__ import annotations + +from unittest.mock import MagicMock + +from atlas.ops.audit import PipelineRunRecord, sanitize_error_message, upsert_pipeline_run + + +def test_sanitize_error_message_redacts_secrets() -> None: + msg = "Bearer abc.def.ghi and BEGIN PRIVATE KEY" + sanitized = sanitize_error_message(msg) + assert "Bearer" not in sanitized + assert "BEGIN PRIVATE KEY" not in sanitized + + +def test_upsert_pipeline_run_executes_merge() -> None: + client = MagicMock() + record = PipelineRunRecord( + pipeline_run_id="atlas-airflow-20260715-test", + batch_id="atlas-20260715", + airflow_run_id="manual__1", + dag_id="atlas_batch_pipeline", + processing_date="2026-07-15", + started_at="2026-07-15T06:00:00+00:00", + completed_at=None, + status="RUNNING", + attempt_number=1, + ) + upsert_pipeline_run(record, client=client) + client.query.assert_called_once() diff --git a/tests/unit/test_batch_context.py b/tests/unit/test_batch_context.py new file mode 100644 index 0000000..863d9d1 --- /dev/null +++ b/tests/unit/test_batch_context.py @@ -0,0 +1,43 @@ +"""Unit tests for batch context resolution.""" + +from __future__ import annotations + +import pytest + +from atlas.batch.context import ( + build_pipeline_run_id, + default_batch_id, + default_seed_for_date, + resolve_batch_context, + validate_batch_id, +) + + +def test_default_batch_id_from_processing_date() -> None: + assert default_batch_id("2026-07-15") == "atlas-20260715" + + +def test_default_seed_is_deterministic() -> None: + assert default_seed_for_date("2026-07-15") == default_seed_for_date("2026-07-15") + + +def test_validate_batch_id_rejects_invalid() -> None: + with pytest.raises(ValueError): + validate_batch_id("bad batch") + + +def test_resolve_batch_context_builds_paths(tmp_path, monkeypatch) -> None: + monkeypatch.setenv("ATLAS_ROOT", str(tmp_path)) + ctx = resolve_batch_context( + processing_date="2026-07-15", + batch_id="atlas-20260715", + airflow_run_id="manual__2026-07-15T06:00:00+00:00", + ) + assert ctx.batch_id == "atlas-20260715" + assert ctx.pipeline_run_id.startswith("atlas-airflow-20260715-") + assert ctx.local_file_path == tmp_path / "data/runs/atlas-20260715/events.jsonl" + + +def test_build_pipeline_run_id_sanitizes_airflow_run_id() -> None: + run_id = build_pipeline_run_id("2026-07-15", "manual__2026-07-15T06:00:00+00:00") + assert "manual__2026-07-15T06" in run_id diff --git a/tests/unit/test_batch_manifest.py b/tests/unit/test_batch_manifest.py new file mode 100644 index 0000000..9848296 --- /dev/null +++ b/tests/unit/test_batch_manifest.py @@ -0,0 +1,33 @@ +"""Unit tests for batch manifest checksum helpers.""" + +from __future__ import annotations + +from pathlib import Path + +from atlas.batch.manifest import ( + BatchManifest, + compute_file_checksum, + manifests_match, + read_manifest, + write_manifest, +) + + +def test_manifest_round_trip(tmp_path: Path) -> None: + file_path = tmp_path / "events.jsonl" + file_path.write_text('{"event_id":"1"}\n', encoding="utf-8") + checksum = compute_file_checksum(file_path) + manifest = BatchManifest( + batch_id="atlas-20260715", + processing_date="2026-07-15", + pipeline_run_id="run-1", + seed=42, + row_count=1, + checksum_sha256=checksum, + output_path=str(file_path), + ) + manifest_path = tmp_path / "manifest.json" + write_manifest(manifest_path, manifest) + loaded = read_manifest(manifest_path) + assert loaded is not None + assert manifests_match(loaded, manifest) diff --git a/tests/unit/test_cost_guard.py b/tests/unit/test_cost_guard.py new file mode 100644 index 0000000..b3967f3 --- /dev/null +++ b/tests/unit/test_cost_guard.py @@ -0,0 +1,67 @@ +"""Sprint 7 Phase 12: cost-control loading and static checks.""" + +from __future__ import annotations + +import pytest + +from atlas.observability import cost_guard +from atlas.observability.cost_guards import CostGuardViolation + + +def test_cost_controls_load_and_have_required_fields() -> None: + controls = cost_guard.load_cost_controls() + for env in ("atlas-dev", "atlas-ci"): + c = controls["environments"][env] + for field in ( + "max_query_bytes", + "max_performance_suite_bytes", + "max_backfill_days", + "require_partition_filter_assets", + "temporary_dataset_ttl_hours", + "log_retention_days", + ): + assert field in c, f"{env} missing {field}" + + +def test_max_query_bytes_positive() -> None: + assert cost_guard.max_query_bytes("atlas-dev") == 1073741824 + + +def test_unknown_environment_raises() -> None: + with pytest.raises(CostGuardViolation): + cost_guard.environment_controls("does-not-exist") + + +def test_performance_suite_env_override(monkeypatch) -> None: + monkeypatch.setenv(cost_guard.SUITE_BYTES_ENV, "123456") + assert cost_guard.max_performance_suite_bytes("atlas-dev") == 123456 + + +def test_partition_filter_detected() -> None: + assert cost_guard.has_partition_filter("select * from t where event_date = '2026-07-19'") + assert cost_guard.has_partition_filter( + "select * from t where user_id = 'u' and processing_date >= '2026-07-01'" + ) + + +def test_partition_filter_absent() -> None: + assert not cost_guard.has_partition_filter("select count(*) from t") + assert not cost_guard.has_partition_filter("select * from t where user_id = 'u'") + + +def test_required_partition_filter_missing_raises() -> None: + with pytest.raises(CostGuardViolation): + cost_guard.check_partition_filter("select count(*) from `p.atlas_raw.events`", "atlas_raw.events") + + +def test_required_partition_filter_present_ok() -> None: + cost_guard.check_partition_filter( + "select count(*) from `p.atlas_raw.events` where event_date = '2026-07-19'", + "atlas_raw.events", + ) + + +def test_non_required_asset_without_filter_ok() -> None: + cost_guard.check_partition_filter( + "select count(*) from `p.atlas_ops.pipeline_runs`", "atlas_ops.pipeline_runs" + ) diff --git a/tests/unit/test_cost_guards.py b/tests/unit/test_cost_guards.py new file mode 100644 index 0000000..97a8477 --- /dev/null +++ b/tests/unit/test_cost_guards.py @@ -0,0 +1,102 @@ +"""BigQuery cost-guard tests (Sprint 6, Phase 12).""" + +from __future__ import annotations + +from datetime import date +from typing import Any + +import pytest + +from atlas.observability.cost_guards import ( + BACKFILL_OVERRIDE_VAR, + FULL_REFRESH_APPROVAL_VAR, + CostGuardViolation, + backfill_dates, + enforce_dry_run_ceiling, + estimate_query_bytes, + guarded_query_config, + require_full_refresh_approval, + validate_backfill_window, +) + + +class FakeDryRunJob: + def __init__(self, total_bytes: int) -> None: + self.total_bytes_processed = total_bytes + + +class FakeClient: + def __init__(self, estimate: int) -> None: + self.estimate = estimate + self.configs: list[Any] = [] + + def query(self, sql: str, job_config: Any = None) -> FakeDryRunJob: + self.configs.append(job_config) + return FakeDryRunJob(self.estimate) + + +def test_dry_run_estimate_uses_dry_run_config() -> None: + client = FakeClient(estimate=1234) + assert estimate_query_bytes(client, "SELECT 1") == 1234 + assert client.configs[0].dry_run is True + + +def test_ceiling_blocks_unpartitioned_scan(capsys: pytest.CaptureFixture[str]) -> None: + """S6-COST-001: an over-ceiling estimate is refused before execution.""" + client = FakeClient(estimate=5 * 1024**3) + with pytest.raises(CostGuardViolation, match="exceeds ceiling"): + enforce_dry_run_ceiling(client, "SELECT * FROM atlas_raw.events", max_estimated_bytes=1024**3) + assert "cost_guard_blocked" in capsys.readouterr().out + + +def test_ceiling_allows_bounded_query_and_returns_estimate() -> None: + client = FakeClient(estimate=10_000) + assert enforce_dry_run_ceiling(client, "SELECT 1", max_estimated_bytes=1024**3) == 10_000 + + +def test_guarded_query_config_sets_maximum_bytes_billed() -> None: + config = guarded_query_config(maximum_bytes_billed=42, labels={"application": "atlas"}) + assert config.maximum_bytes_billed == 42 + assert config.labels == {"application": "atlas"} + + +def test_backfill_window_within_policy_allowed() -> None: + days = validate_backfill_window(date(2026, 7, 10), date(2026, 7, 14), max_days=7, env={}) + assert days == 5 + assert len(backfill_dates(date(2026, 7, 10), days)) == 5 + + +def test_backfill_window_beyond_policy_blocked_without_override() -> None: + """S6-COST-002: unbounded backfill requires an explicit override.""" + with pytest.raises(CostGuardViolation, match="exceeds the 7-day policy"): + validate_backfill_window(date(2026, 6, 1), date(2026, 7, 19), max_days=7, env={}) + + +def test_backfill_window_beyond_policy_allowed_with_explicit_override() -> None: + days = validate_backfill_window( + date(2026, 6, 1), date(2026, 7, 19), max_days=7, env={BACKFILL_OVERRIDE_VAR: "true"} + ) + assert days == 49 + + +def test_backfill_override_must_be_exactly_true() -> None: + for sloppy in ("1", "yes", "TRUEISH"): + with pytest.raises(CostGuardViolation): + validate_backfill_window( + date(2026, 6, 1), date(2026, 7, 19), max_days=7, env={BACKFILL_OVERRIDE_VAR: sloppy} + ) + + +def test_inverted_backfill_window_rejected() -> None: + with pytest.raises(CostGuardViolation, match="precedes start"): + validate_backfill_window(date(2026, 7, 19), date(2026, 7, 1), env={}) + + +def test_full_refresh_blocked_without_approval() -> None: + """S6-COST-003: full refresh is an approved exception, never a default.""" + with pytest.raises(CostGuardViolation, match=FULL_REFRESH_APPROVAL_VAR): + require_full_refresh_approval(env={}) + + +def test_full_refresh_allowed_with_approval() -> None: + require_full_refresh_approval(env={FULL_REFRESH_APPROVAL_VAR: "true"}) diff --git a/tests/unit/test_deployments_audit.py b/tests/unit/test_deployments_audit.py new file mode 100644 index 0000000..e1034ba --- /dev/null +++ b/tests/unit/test_deployments_audit.py @@ -0,0 +1,154 @@ +"""Unit tests for atlas_ops.deployments audit records (Sprint 4 Phase 10).""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from atlas.config.settings import load_settings +from atlas.ops.deployments import ( + DeploymentRecord, + finalize_deployment, + start_deployment, + upsert_deployment, +) + + +class FakeJob: + def result(self) -> list[Any]: + return [] + + +class FakeClient: + def __init__(self) -> None: + self.queries: list[str] = [] + self.params: list[dict[str, Any]] = [] + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + self.queries.append(sql) + if job_config is not None: + self.params.append({p.name: p.value for p in job_config.query_parameters}) + return FakeJob() + + +SETTINGS = load_settings() + + +def _record(**overrides: Any) -> DeploymentRecord: + base: dict[str, Any] = { + "deployment_id": "atlas-dev-20260718-abc123", + "git_sha": "a" * 40, + "environment": "atlas-dev", + "deployment_type": "deploy", + "started_at": "2026-07-18T21:00:00+00:00", + "status": "RUNNING", + } + base.update(overrides) + return DeploymentRecord(**base) + + +def test_upsert_uses_merge_keyed_by_deployment_id() -> None: + client = FakeClient() + upsert_deployment(_record(), SETTINGS, client=client) + assert len(client.queries) == 1 + sql = client.queries[0] + assert "MERGE" in sql + assert "atlas_ops.deployments" in sql + assert "target.deployment_id = source.deployment_id" in sql + + +def test_upsert_rejects_unknown_status_and_type() -> None: + client = FakeClient() + with pytest.raises(ValueError, match="Unsupported deployment status"): + upsert_deployment(_record(status="EXPLODED"), SETTINGS, client=client) + with pytest.raises(ValueError, match="Unsupported deployment type"): + upsert_deployment(_record(deployment_type="yolo"), SETTINGS, client=client) + assert client.queries == [] + + +def test_start_deployment_writes_running_row() -> None: + client = FakeClient() + record = start_deployment( + deployment_id="d-1", + git_sha="b" * 40, + environment="atlas-dev", + workflow_run_id="12345", + settings=SETTINGS, + client=client, + ) + assert record.status == "RUNNING" + assert client.params[0]["status"] == "RUNNING" + assert client.params[0]["workflow_run_id"] == "12345" + + +def test_start_rollback_writes_rolling_back_row() -> None: + client = FakeClient() + record = start_deployment( + deployment_id="rb-1", + git_sha="c" * 40, + environment="atlas-dev", + deployment_type="rollback", + previous_git_sha="d" * 40, + settings=SETTINGS, + client=client, + ) + assert record.status == "ROLLING_BACK" + assert client.params[0]["previous_git_sha"] == "d" * 40 + + +def test_finalize_is_idempotent_per_deployment_id() -> None: + client = FakeClient() + record = _record() + first = finalize_deployment(record, status="SUCCESS", settings=SETTINGS, client=client) + second = finalize_deployment(first, status="SUCCESS", settings=SETTINGS, client=client) + # Same deployment_id and completed_at on repeat finalization: the MERGE + # updates the same row rather than inserting another attempt. + assert first.deployment_id == second.deployment_id + assert first.completed_at == second.completed_at + assert all("WHEN MATCHED THEN UPDATE" in sql for sql in client.queries) + + +def test_finalize_failure_records_stage_and_sanitized_error() -> None: + client = FakeClient() + record = _record() + finalize_deployment( + record, + status="FAILED", + failure_stage="smoke_validation", + error_type="SmokeFailure", + error_summary='dbt exploded with keyfile {"private_key": "SECRET"} attached', + settings=SETTINGS, + client=client, + ) + params = client.params[0] + assert params["status"] == "FAILED" + assert params["failure_stage"] == "smoke_validation" + assert "SECRET" not in (params["error_summary"] or "") + assert "[REDACTED]" in params["error_summary"] + + +def test_rollback_links_previous_sha() -> None: + client = FakeClient() + record = start_deployment( + deployment_id="rb-2", + git_sha="0" * 40, + environment="atlas-dev", + deployment_type="rollback", + previous_git_sha="f" * 40, + settings=SETTINGS, + client=client, + ) + final = finalize_deployment(record, status="ROLLED_BACK", settings=SETTINGS, client=client) + assert final.previous_git_sha == "f" * 40 + assert client.params[-1]["status"] == "ROLLED_BACK" + + +def test_deployments_and_pipeline_runs_are_separate_grains() -> None: + client = FakeClient() + upsert_deployment(_record(smoke_pipeline_run_id="atlas-smoke-abc-1"), SETTINGS, client=client) + sql = client.queries[0] + # Deployments reference the smoke run by id but never write pipeline_runs. + assert "atlas_ops.deployments" in sql + assert "pipeline_runs" not in sql + assert client.params[0]["smoke_pipeline_run_id"] == "atlas-smoke-abc-1" diff --git a/tests/unit/test_deprecation.py b/tests/unit/test_deprecation.py new file mode 100644 index 0000000..1d43f9b --- /dev/null +++ b/tests/unit/test_deprecation.py @@ -0,0 +1,87 @@ +"""Sprint 7 Phase 6: deprecation lifecycle enforcement tests (fixtures only).""" + +from __future__ import annotations + +from atlas.governance.registry import deprecation_errors + +POLICY = { + "deprecation": {"minimum_window_days": 30, "require_replacement": True, "require_change_record": True} +} +CHANGES = {"CHG-20260719-deprecate-legacy"} +CONSUMERS = { + "reader_a": {"type": "dag", "reads": ["legacy_model"]}, +} + + +def _base_dep() -> dict: + return { + "replacement": "new_model", + "owner": "atlas-data-eng", + "deprecation_start": "2026-07-19", + "earliest_removal_date": "2026-09-01", + "removal_approval": "ATLAS_APPROVE_SCHEMA_MUTATION", + "change_record": "CHG-20260719-deprecate-legacy", + } + + +def test_active_asset_has_no_deprecation_errors() -> None: + record = {"lifecycle_status": "ACTIVE"} + assert deprecation_errors("m", record, {}, CHANGES, POLICY) == [] + + +def test_deprecated_without_block_fails() -> None: + record = {"lifecycle_status": "DEPRECATED"} + errors = deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) + assert any("requires a 'deprecation' block" in e for e in errors) + + +def test_deprecated_without_replacement_fails() -> None: + dep = _base_dep() + dep["replacement"] = "" + record = {"lifecycle_status": "DEPRECATED", "deprecation": dep} + errors = deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) + assert any("must declare a 'replacement'" in e for e in errors) + + +def test_removal_before_minimum_window_fails() -> None: + dep = _base_dep() + dep["earliest_removal_date"] = "2026-07-25" # 6 days after start + record = {"lifecycle_status": "DEPRECATED", "deprecation": dep} + errors = deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) + assert any("minimum window" in e for e in errors) + + +def test_lifecycle_change_without_change_record_fails() -> None: + dep = _base_dep() + dep["change_record"] = "" + record = {"lifecycle_status": "DEPRECATED", "deprecation": dep} + errors = deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) + assert any("requires a 'change_record'" in e for e in errors) + + +def test_change_record_not_found_fails() -> None: + dep = _base_dep() + dep["change_record"] = "CHG-does-not-exist" + record = {"lifecycle_status": "DEPRECATED", "deprecation": dep} + errors = deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) + assert any("not found under governance/changes" in e for e in errors) + + +def test_removed_asset_with_active_consumer_fails() -> None: + record = {"lifecycle_status": "REMOVED", "deprecation": _base_dep()} + errors = deprecation_errors("legacy_model", record, CONSUMERS, CHANGES, POLICY) + assert any("still has active consumers" in e for e in errors) + + +def test_valid_deprecation_passes() -> None: + record = {"lifecycle_status": "DEPRECATED", "deprecation": _base_dep()} + # No active consumers passed -> DEPRECATED (not removed) is allowed. + assert deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) == [] + + +def test_removal_scheduled_requires_approval() -> None: + dep = _base_dep() + dep["removal_approval"] = "" + record = {"lifecycle_status": "REMOVAL_SCHEDULED", "deprecation": dep} + errors = deprecation_errors("legacy_model", record, {}, CHANGES, POLICY) + assert any("removal_approval" in e for e in errors) diff --git a/tests/unit/test_failure_injection.py b/tests/unit/test_failure_injection.py new file mode 100644 index 0000000..8525156 --- /dev/null +++ b/tests/unit/test_failure_injection.py @@ -0,0 +1,169 @@ +"""Fault-injection framework safety tests (Sprint 6, Phases 3/16).""" + +from __future__ import annotations + +from datetime import UTC, datetime, timedelta + +import pytest + +from atlas.failure_injection.framework import ( + APPROVAL_VAR, + SCENARIO_VAR, + InjectionRefused, + authorize_injection, + enforce_deadline, + injection_active_for, + is_injection_requested, +) +from atlas.failure_injection.registry import ( + ALLOWED_RISK_LEVELS, + get_scenario, + load_catalog, + validate_catalog, +) + +CATALOG = load_catalog() + + +def _env(**overrides: str) -> dict[str, str]: + base = { + SCENARIO_VAR: "S6-ING-001", + APPROVAL_VAR: "true", + } + base.update(overrides) + return base + + +# ---------------------------------------------------------------- catalog + + +def test_catalog_schema_is_valid() -> None: + assert validate_catalog(CATALOG) == [] + + +def test_no_critical_risk_scenarios_exist() -> None: + assert "CRITICAL" not in ALLOWED_RISK_LEVELS + for spec in CATALOG["scenarios"].values(): + assert spec["risk_level"] in ALLOWED_RISK_LEVELS + + +def test_every_scenario_requires_injection_approval() -> None: + defaults = CATALOG["defaults"]["approval_required"] + for scenario_id, spec in CATALOG["scenarios"].items(): + approvals = spec.get("approval_required", defaults) + assert APPROVAL_VAR in approvals, scenario_id + + +def test_destructive_scenarios_require_destructive_fixture_approval() -> None: + for scenario_id in ("S6-ING-004", "S6-DBT-006", "S6-DBT-007"): + spec = get_scenario(scenario_id, CATALOG) + assert "ATLAS_APPROVE_DESTRUCTIVE_FIXTURE" in spec["approval_required"] + + +def test_iam_scenarios_require_iam_approval() -> None: + for scenario_id in ("S6-IAM-001", "S6-IAM-002", "S6-IAM-003", "S6-IAM-005"): + spec = get_scenario(scenario_id, CATALOG) + assert "ATLAS_APPROVE_IAM" in spec["approval_required"] + + +def test_unknown_scenario_rejected() -> None: + with pytest.raises(KeyError, match="unknown failure scenario"): + get_scenario("S6-NOPE-999", CATALOG) + + +# ---------------------------------------------------------------- activation + + +def test_disabled_by_default_empty_environment() -> None: + assert is_injection_requested(env={}) is False + assert injection_active_for("S6-ING-001", env={}) is False + + +def test_approval_alone_never_activates_injection() -> None: + """A lingering approval variable without the explicit scenario is inert.""" + env = {APPROVAL_VAR: "true"} + assert is_injection_requested(env=env) is False + assert injection_active_for("S6-ING-001", env=env) is False + + +def test_scenario_without_approval_is_refused() -> None: + with pytest.raises(InjectionRefused, match="missing approval"): + authorize_injection("S6-ING-001", environment="atlas-dev", env=_env(**{APPROVAL_VAR: ""})) + + +def test_scenario_env_var_must_match_requested_scenario() -> None: + with pytest.raises(InjectionRefused, match="must explicitly name"): + authorize_injection( + "S6-ING-002", + environment="atlas-dev", + env=_env(), # env names S6-ING-001 + ) + + +def test_production_style_environment_refused() -> None: + for environment in ("atlas-prod", "production", "prod-us"): + with pytest.raises(InjectionRefused, match="not injectable"): + authorize_injection("S6-ING-001", environment=environment, env=_env()) + + +def test_scheduled_execution_refused() -> None: + with pytest.raises(InjectionRefused, match="refuses scheduled execution"): + authorize_injection( + "S6-ING-001", + environment="atlas-dev", + env=_env(AIRFLOW_CTX_DAG_RUN_TYPE="scheduled"), + ) + with pytest.raises(InjectionRefused, match="refuses scheduled run ids"): + authorize_injection( + "S6-ING-001", + environment="atlas-dev", + env=_env(AIRFLOW_CTX_DAG_RUN_ID="scheduled__2026-07-19T00:00:00+00:00"), + ) + + +def test_canonical_batch_ids_refused() -> None: + for batch_id in ("atlas-20260719", "atlas-smoke-abc123-run", "atlas-drillb-20260719"): + with pytest.raises(InjectionRefused, match="not isolated"): + authorize_injection("S6-ING-001", environment="atlas-dev", batch_id=batch_id, env=_env()) + + +def test_isolated_batch_id_accepted_with_full_approvals() -> None: + authorization = authorize_injection( + "S6-ING-001", environment="atlas-dev", batch_id="atlas-s6-ing001-20260719", env=_env() + ) + assert authorization.scenario_id == "S6-ING-001" + assert authorization.deadline > datetime.now(tz=UTC) + + +def test_scenario_specific_approvals_enforced() -> None: + env = _env(**{SCENARIO_VAR: "S6-ING-004"}) + with pytest.raises(InjectionRefused, match="ATLAS_APPROVE_DESTRUCTIVE_FIXTURE"): + authorize_injection("S6-ING-004", environment="atlas-dev", env=env) + env["ATLAS_APPROVE_DESTRUCTIVE_FIXTURE"] = "true" + authorization = authorize_injection( + "S6-ING-004", environment="atlas-dev", batch_id="atlas-s6-ing004-x", env=env + ) + assert authorization.spec["risk_level"] == "HIGH" + + +def test_timeout_enforcement() -> None: + authorization = authorize_injection( + "S6-ING-001", environment="atlas-dev", batch_id="atlas-s6-a", env=_env() + ) + enforce_deadline(authorization) # within budget: no raise + expired = type(authorization)( + scenario_id=authorization.scenario_id, + environment=authorization.environment, + batch_id=authorization.batch_id, + deadline=datetime.now(tz=UTC) - timedelta(seconds=1), + spec=authorization.spec, + ) + with pytest.raises(InjectionRefused, match="exceeded maximum_duration"): + enforce_deadline(expired) + + +def test_refused_hook_logs_and_returns_false(capsys: pytest.CaptureFixture[str]) -> None: + """A requested-but-unapproved scenario is refused loudly, not silently.""" + env = {SCENARIO_VAR: "S6-ING-001"} # approval missing + assert injection_active_for("S6-ING-001", env=env) is False + assert "failure_injection_refused" in capsys.readouterr().out diff --git a/tests/unit/test_generator.py b/tests/unit/test_generator.py new file mode 100644 index 0000000..55a9479 --- /dev/null +++ b/tests/unit/test_generator.py @@ -0,0 +1,94 @@ +"""Unit tests for the synthetic event generator.""" + +from __future__ import annotations + +from dataclasses import replace + +from atlas.config.settings import load_settings +from atlas.generator.events import EventRecord, generate_events, generate_events_for_batch + + +def test_event_record_excludes_ingested_at_from_jsonl() -> None: + record = EventRecord( + event_id="1", + user_id="u1", + event_name="app_open", + event_timestamp="2026-07-14T12:00:00+00:00", + event_date="2026-07-14", + country_code="US", + platform="web", + app_version="1.0.0", + ) + assert "ingested_at" not in record.to_dict() + + +def test_generate_events_count_and_seed(tmp_path) -> None: + settings = load_settings() + settings = replace( + settings, + generator=replace(settings.generator, output_dir=tmp_path), + ) + first = generate_events(settings) + second = generate_events(settings) + assert first.event_count == 50000 + assert first.output_path.exists() + assert first.output_path == second.output_path + assert first.anomaly_counts["null_user_ids"] == 500 + assert first.anomaly_counts["duplicate_event_ids"] == 50 + + first_line = first.output_path.read_text(encoding="utf-8").splitlines()[0] + assert "ingested_at" not in first_line + + +def test_generate_events_for_batch_is_deterministic(tmp_path, monkeypatch) -> None: + settings = load_settings() + output = tmp_path / "events.jsonl" + first = generate_events_for_batch( + settings, + processing_date="2026-07-15", + batch_id="atlas-20260715", + pipeline_run_id="atlas-airflow-20260715-test", + seed=12345, + output_path=output, + ) + second = generate_events_for_batch( + settings, + processing_date="2026-07-15", + batch_id="atlas-20260715", + pipeline_run_id="atlas-airflow-20260715-test2", + seed=12345, + output_path=output, + ) + assert first.reused_existing is False + assert second.reused_existing is True + assert first.checksum_sha256 == second.checksum_sha256 + + +def test_regeneration_to_fresh_path_is_byte_identical(tmp_path) -> None: + """Same batch identity must regenerate identical bytes without artifact reuse. + + Guards against nondeterministic sources (uuid4/os.urandom) sneaking into + generation: cross-machine reproducibility is what makes deployment smoke + batches and integration tests comparable. + """ + settings = load_settings() + first = generate_events_for_batch( + settings, + processing_date="2026-07-15", + batch_id="atlas-20260715", + pipeline_run_id="atlas-airflow-20260715-a", + seed=12345, + output_path=tmp_path / "a.jsonl", + ) + second = generate_events_for_batch( + settings, + processing_date="2026-07-15", + batch_id="atlas-20260715", + pipeline_run_id="atlas-airflow-20260715-b", + seed=12345, + output_path=tmp_path / "b.jsonl", + ) + assert first.reused_existing is False + assert second.reused_existing is False + assert first.checksum_sha256 == second.checksum_sha256 + assert (tmp_path / "a.jsonl").read_bytes() == (tmp_path / "b.jsonl").read_bytes() diff --git a/tests/unit/test_governance_demos.py b/tests/unit/test_governance_demos.py new file mode 100644 index 0000000..aefaab3 --- /dev/null +++ b/tests/unit/test_governance_demos.py @@ -0,0 +1,106 @@ +"""Sprint 7 Phase 14: controlled governance-failure demonstrations (fixtures). + +These prove CI *rejects* unsafe governance changes without ever committing the +defect to main. They inject crafted records via loader monkeypatching so the +real committed governance stays valid. +""" + +from __future__ import annotations + +import json + +import pytest + +from atlas.governance import registry + + +def _good_asset(**overrides): + asset = { + "asset_id": "demo_asset", + "asset_type": "operational_table", + "purpose": "demo", + "technical_owner": "atlas-data-eng", + "business_owner_or_role": "atlas-platform", + "grain": "one row per demo", + "source": "demo", + "consumers": [], + "classification": "INTERNAL", + "retention_class": "operational_audit", + "freshness_expectation": "per run", + "contract_version": "1.0", + "lifecycle_status": "ACTIVE", + "repository_path": "demo", + "runbook": "docs/demo.md", + "last_reviewed": "2026-07-19", + } + asset.update(overrides) + return asset + + +@pytest.fixture +def patch_registry(monkeypatch): + def _apply(assets, dbt_records=None): + monkeypatch.setattr(registry, "load_non_dbt_assets", lambda: assets) + monkeypatch.setattr(registry, "load_dbt_model_governance", lambda: dbt_records or []) + + return _apply + + +def test_valid_asset_passes(patch_registry) -> None: + patch_registry([_good_asset()]) + assert registry.validate_governance() == [] + + +def test_missing_owner_fails(patch_registry) -> None: + patch_registry([_good_asset(technical_owner="")]) + errors = registry.validate_governance() + assert any("missing required field 'technical_owner'" in e for e in errors) + + +def test_missing_grain_fails(patch_registry) -> None: + patch_registry([_good_asset(grain="")]) + errors = registry.validate_governance() + assert any("missing required field 'grain'" in e for e in errors) + + +def test_invalid_classification_fails(patch_registry) -> None: + patch_registry([_good_asset(classification="TOP_SECRET")]) + errors = registry.validate_governance() + assert any("invalid classification" in e for e in errors) + + +def test_invalid_retention_class_fails(patch_registry) -> None: + patch_registry([_good_asset(retention_class="forever_and_ever")]) + errors = registry.validate_governance() + assert any("invalid retention_class" in e for e in errors) + + +def test_email_owner_fails(patch_registry) -> None: + patch_registry([_good_asset(technical_owner="someone@example.com")]) + errors = registry.validate_governance() + assert any("must be a role id" in e for e in errors) + + +def test_duplicate_source_of_truth_fails(patch_registry) -> None: + dbt_record = _good_asset(asset_id="fct_events", asset_type="fact_model") + patch_registry([_good_asset(asset_id="fct_events")], dbt_records=[dbt_record]) + errors = registry.validate_governance() + assert any("duplicate source of truth" in e for e in errors) + + +def test_migration_checksum_tamper_is_detected() -> None: + # Demonstrate: modifying an applied migration changes its checksum and would + # diverge from the committed lock (the check gate_schema_compatibility runs). + from atlas.config.settings import atlas_root + from atlas.ops.migrations import load_manifest + + lock = json.loads((atlas_root() / "sql/migrations/checksums.lock").read_text()) + locked = lock["checksums"] + tampered = dict(locked) + # Simulate a tampered file checksum for an applied migration. + first = next(iter(tampered)) + tampered[first] = "0" * 64 + # The live files still match the lock ... + assert all(m.checksum == locked[m.migration_id] for m in load_manifest()) + # ... but a tampered checksum would not, which is exactly what the gate flags. + assert tampered[first] != locked[first] diff --git a/tests/unit/test_lineage_impact.py b/tests/unit/test_lineage_impact.py new file mode 100644 index 0000000..d99e3ee --- /dev/null +++ b/tests/unit/test_lineage_impact.py @@ -0,0 +1,45 @@ +"""Sprint 7 Phase 5: lineage + consumer-impact tests.""" + +from __future__ import annotations + +import json + +from atlas.config.settings import atlas_root +from atlas.governance import impact, lineage + + +def test_source_reaches_mart() -> None: + graph = lineage.build_lineage() + downstream = graph.transitive_downstream("atlas_raw.events") + assert "stg_events" in downstream + assert "fct_events" in downstream + assert "mart_daily_event_metrics" in downstream + + +def test_fct_events_upstream_includes_source() -> None: + graph = lineage.build_lineage() + upstream = graph.transitive_upstream("fct_events") + assert "stg_events" in upstream + assert "atlas_raw.events" in upstream + + +def test_impact_identifies_downstream_models() -> None: + report = impact.analyze("fct_events") + assert "mart_daily_event_metrics" in report["transitive_downstream"] + assert "analytics_mart_readers" in report["affected_consumers"] + assert "atlas-analytics" in report["owners_to_notify"] + assert report["affected_tests"], "expected at least one affected test/property file" + + +def test_impact_of_staging_change_propagates() -> None: + report = impact.analyze("stg_events") + # A change to staging must surface the whole downstream chain. + for expected in ("int_event_classification", "fct_events", "mart_daily_event_metrics"): + assert expected in report["transitive_downstream"], expected + + +def test_committed_lineage_matches_fresh_generation() -> None: + committed = json.loads((atlas_root() / "governance/generated/lineage.json").read_text()) + lineage.build_lineage.cache_clear() + fresh = lineage.to_dict(lineage.build_lineage()) + assert committed == fresh, "committed lineage.json is stale; regenerate it" diff --git a/tests/unit/test_loader_batch.py b/tests/unit/test_loader_batch.py new file mode 100644 index 0000000..c2f12f2 --- /dev/null +++ b/tests/unit/test_loader_batch.py @@ -0,0 +1,22 @@ +"""Unit tests for batch-scoped BigQuery load evaluation.""" + +from __future__ import annotations + +import pytest + +from atlas.loader.bigquery import BatchLoadState, evaluate_batch_load + + +def test_evaluate_batch_load_actions() -> None: + assert evaluate_batch_load(BatchLoadState(0, 0), 50000) == "load" + assert evaluate_batch_load(BatchLoadState(50000, 1), 50000) == "skip" + + +def test_evaluate_batch_load_partial_fails() -> None: + with pytest.raises(ValueError, match="Partial batch"): + evaluate_batch_load(BatchLoadState(100, 1), 50000) + + +def test_evaluate_batch_load_excess_fails() -> None: + with pytest.raises(ValueError, match="Conflicting batch"): + evaluate_batch_load(BatchLoadState(60000, 1), 50000) diff --git a/tests/unit/test_migrations.py b/tests/unit/test_migrations.py new file mode 100644 index 0000000..d3fad31 --- /dev/null +++ b/tests/unit/test_migrations.py @@ -0,0 +1,205 @@ +"""Unit tests for the ledger-driven migration system (Sprint 4 Phase 9).""" + +from __future__ import annotations + +import hashlib +from pathlib import Path +from typing import Any + +import pytest + +from atlas.config.settings import load_settings +from atlas.ops.migrations import ( + Migration, + apply_migrations, + load_manifest, + plan_migrations, + render_migration_sql, +) + + +class FakeJob: + def __init__(self, rows: list[dict[str, Any]] | Exception) -> None: + self._rows = rows + + def result(self) -> list[Any]: + if isinstance(self._rows, Exception): + raise self._rows + + class Row(dict): + def items(self): # noqa: ANN202 + return dict.items(self) + + def __getitem__(self, key): # noqa: ANN001, ANN204 + return dict.__getitem__(self, key) + + return [Row(r) for r in self._rows] + + +class FakeClient: + """Scripted BigQuery client: ledger reads return preset rows, DDL succeeds.""" + + def __init__( + self, + ledger_rows: list[dict[str, Any]] | None = None, + fail_sql_containing: str | None = None, + ) -> None: + self.ledger_rows = ledger_rows or [] + self.fail_sql_containing = fail_sql_containing + self.executed: list[str] = [] + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + self.executed.append(sql) + if self.fail_sql_containing and self.fail_sql_containing in sql: + return FakeJob(RuntimeError("synthetic failure")) + if "FROM" in sql and "schema_migrations" in sql and "MERGE" not in sql: + return FakeJob(self.ledger_rows) + return FakeJob([]) + + +def _write_manifest(tmp_path: Path, entries: list[tuple[str, str, str]]) -> Path: + """entries: (migration_id, filename, sql content).""" + manifest_dir = tmp_path / "migrations" + manifest_dir.mkdir() + lines = [] + for migration_id, filename, content in entries: + (tmp_path / filename).write_text(content, encoding="utf-8") + lines.append(f"{migration_id}|../{filename}") + manifest = manifest_dir / "manifest.txt" + manifest.write_text("\n".join(lines) + "\n", encoding="utf-8") + return manifest + + +def test_load_manifest_orders_and_checksums(tmp_path: Path) -> None: + manifest = _write_manifest( + tmp_path, + [("001_a", "a.sql", "SELECT 1"), ("002_b", "b.sql", "SELECT 2")], + ) + migrations = load_manifest(manifest) + assert [m.migration_id for m in migrations] == ["001_a", "002_b"] + assert migrations[0].checksum == hashlib.sha256(b"SELECT 1").hexdigest() + + +def test_load_manifest_rejects_duplicates(tmp_path: Path) -> None: + manifest = _write_manifest( + tmp_path, + [("001_a", "a.sql", "SELECT 1"), ("001_a", "b.sql", "SELECT 2")], + ) + with pytest.raises(ValueError, match="Duplicate migration id"): + load_manifest(manifest) + + +def test_load_manifest_rejects_missing_file(tmp_path: Path) -> None: + manifest_dir = tmp_path / "migrations" + manifest_dir.mkdir() + manifest = manifest_dir / "manifest.txt" + manifest.write_text("001_a|../missing.sql\n", encoding="utf-8") + with pytest.raises(FileNotFoundError): + load_manifest(manifest) + + +def test_load_manifest_rejects_malformed_line(tmp_path: Path) -> None: + manifest_dir = tmp_path / "migrations" + manifest_dir.mkdir() + manifest = manifest_dir / "manifest.txt" + manifest.write_text("001_a_no_pipe\n", encoding="utf-8") + with pytest.raises(ValueError, match="Malformed manifest line"): + load_manifest(manifest) + + +def test_shipped_manifest_parses_and_is_additive() -> None: + migrations = load_manifest() + assert [m.migration_id for m in migrations][:2] == [ + "001_create_pipeline_runs_table", + "002_sprint3_raw_batch_columns", + ] + settings = load_settings() + for migration in migrations: + sql = render_migration_sql(migration, settings).upper() + assert "DROP TABLE" not in sql + assert "DELETE FROM" not in sql + assert "TRUNCATE" not in sql + + +def test_plan_reports_pending_applied_and_mismatch(tmp_path: Path) -> None: + manifest = _write_manifest( + tmp_path, + [ + ("001_a", "a.sql", "SELECT 1"), + ("002_b", "b.sql", "SELECT 2"), + ("003_c", "c.sql", "SELECT 3"), + ], + ) + checksum_a = hashlib.sha256(b"SELECT 1").hexdigest() + client = FakeClient( + ledger_rows=[ + {"migration_id": "001_a", "migration_checksum": checksum_a, "status": "APPLIED"}, + {"migration_id": "002_b", "migration_checksum": "tampered", "status": "APPLIED"}, + ] + ) + plan = plan_migrations(load_settings(), client=client, manifest_path=manifest) + states = {e.migration_id: e.state for e in plan} + assert states == {"001_a": "APPLIED", "002_b": "CHECKSUM_MISMATCH", "003_c": "PENDING"} + + +def test_apply_refuses_changed_recorded_migration(tmp_path: Path) -> None: + manifest = _write_manifest(tmp_path, [("001_a", "a.sql", "SELECT 1 -- edited")]) + client = FakeClient( + ledger_rows=[{"migration_id": "001_a", "migration_checksum": "original", "status": "APPLIED"}] + ) + with pytest.raises(RuntimeError, match="content changed"): + apply_migrations(load_settings(), client=client, manifest_path=manifest) + + +def test_failed_migration_with_corrected_content_is_retryable(tmp_path: Path) -> None: + # Regression (Sprint 5): a FAILED attempt is not immutable — the corrected + # file must plan as FAILED_PREVIOUSLY and re-apply, not CHECKSUM_MISMATCH. + manifest = _write_manifest(tmp_path, [("001_a", "a.sql", "SELECT 1 -- fixed")]) + client = FakeClient( + ledger_rows=[{"migration_id": "001_a", "migration_checksum": "broken-original", "status": "FAILED"}] + ) + plan = plan_migrations(load_settings(), client=client, manifest_path=manifest) + assert plan[0].state == "FAILED_PREVIOUSLY" + results = apply_migrations(load_settings(), client=client, manifest_path=manifest) + assert [r.state for r in results] == ["APPLIED_NOW"] + + +def test_apply_skips_recorded_migrations_idempotently(tmp_path: Path) -> None: + manifest = _write_manifest(tmp_path, [("001_a", "a.sql", "SELECT 1")]) + checksum = hashlib.sha256(b"SELECT 1").hexdigest() + client = FakeClient( + ledger_rows=[{"migration_id": "001_a", "migration_checksum": checksum, "status": "APPLIED"}] + ) + results = apply_migrations(load_settings(), client=client, manifest_path=manifest) + assert [r.state for r in results] == ["APPLIED"] + assert not any("SELECT 1" in sql for sql in client.executed if "MERGE" not in sql and "CREATE" not in sql) + + +def test_apply_records_failure_and_blocks(tmp_path: Path) -> None: + manifest = _write_manifest(tmp_path, [("001_bad", "bad.sql", "SELECT boom_marker")]) + client = FakeClient(fail_sql_containing="boom_marker") + with pytest.raises(RuntimeError, match="001_bad failed"): + apply_migrations(load_settings(), client=client, manifest_path=manifest) + merges = [sql for sql in client.executed if "MERGE" in sql] + assert merges, "failed migration must still be recorded in the ledger" + + +def test_apply_retry_after_failure_reruns_migration(tmp_path: Path) -> None: + manifest = _write_manifest(tmp_path, [("001_a", "a.sql", "SELECT 1")]) + checksum = hashlib.sha256(b"SELECT 1").hexdigest() + client = FakeClient( + ledger_rows=[{"migration_id": "001_a", "migration_checksum": checksum, "status": "FAILED"}] + ) + results = apply_migrations(load_settings(), client=client, manifest_path=manifest) + assert [r.state for r in results] == ["APPLIED_NOW"] + + +def test_render_migration_sql_parameterizes_dataset(tmp_path: Path) -> None: + sql_file = tmp_path / "m.sql" + sql_file.write_text("ALTER TABLE `{project_id}.{dataset_id}.events` ADD COLUMN x STRING", "utf-8") + migration = Migration("001_x", sql_file, "abc") + settings = load_settings() + rendered = render_migration_sql(migration, settings) + assert settings.gcp.project_id in rendered + assert settings.gcp.dataset_id in rendered + assert "{" not in rendered diff --git a/tests/unit/test_observability_logging.py b/tests/unit/test_observability_logging.py new file mode 100644 index 0000000..23843f9 --- /dev/null +++ b/tests/unit/test_observability_logging.py @@ -0,0 +1,221 @@ +"""Contract tests for the Sprint 5 structured logging module.""" + +from __future__ import annotations + +import io +import json +from datetime import datetime + +import pytest + +from atlas.observability.logging import ( + ALLOWED_FIELDS, + CORRELATION_FIELDS, + EVENT_MARKER, + ContractViolation, + build_event, + correlation_fields_from_context, + emit_event, + new_correlation_id, +) + + +def test_minimal_event_schema() -> None: + event = build_event("task_started", component="step_runner") + assert event[EVENT_MARKER] is True + assert event["event_type"] == "task_started" + assert event["severity"] == "INFO" + assert event["component"] == "step_runner" + # UTC ISO-8601 timestamp + parsed = datetime.fromisoformat(event["timestamp"]) + assert parsed.tzinfo is not None and parsed.utcoffset().total_seconds() == 0 + + +def test_correlation_hierarchy_fields_accepted() -> None: + event = build_event( + "task_finished", + deployment_id="dep-1", + airflow_run_id="run-1", + pipeline_run_id="pr-1", + batch_id="b-1", + task_id="load", + attempt_number=2, + ) + for field in CORRELATION_FIELDS: + assert field in event + assert event["attempt_number"] == 2 + + +def test_unknown_field_dropped_and_noted() -> None: + event = build_event("task_finished", bogus_field="x") + assert "bogus_field" not in event + assert "field:bogus_field" in event["contract_violations"] + + +def test_unknown_field_raises_in_strict_mode() -> None: + with pytest.raises(ContractViolation): + build_event("task_finished", strict=True, bogus_field="x") + + +def test_invalid_severity_normalized_or_strict() -> None: + event = build_event("x", severity="LOUD") + assert event["severity"] == "INFO" + assert "severity:LOUD" in event["contract_violations"] + with pytest.raises(ContractViolation): + build_event("x", severity="LOUD", strict=True) + + +def test_error_message_is_sanitized_and_truncated() -> None: + # Concatenated so the repo secret scanner never sees a contiguous PEM header. + pem_header = "-----BEGIN " + "PRIVATE KEY-----" + secret = '{"private_key": "' + pem_header + 'abc"}' + "x" * 5000 + event = build_event("task_failed", error_message=secret, error_type="RuntimeError") + assert "BEGIN PRIVATE KEY" not in event["error_message"] + assert "[REDACTED]" in event["error_message"] + assert len(event["error_message"]) <= 2000 + + +def test_no_secret_tokens_survive_common_fields() -> None: + event = build_event( + "deploy_failed", + error_message="Authorization: Bearer abc123token failed", + ) + assert "abc123token" not in json.dumps(event) + + +def test_non_serializable_values_degrade_to_strings() -> None: + class Weird: + def __repr__(self) -> str: + return "" + + event = build_event("x", details={"obj": Weird()}) + assert event["details"]["obj"] == "" + json.dumps(event) # must be serializable end to end + + +def test_details_truncation() -> None: + event = build_event("x", details={"blob": "y" * 10000}) + assert event["details"]["truncated"] is True + + +def test_int_fields_coerced() -> None: + event = build_event("x", rows_loaded="50000", duration_ms=12.7) + assert event["rows_loaded"] == 50000 + assert event["duration_ms"] == 12 + + +def test_emit_writes_one_json_line() -> None: + stream = io.StringIO() + emit_event("task_started", stream=stream, pipeline_run_id="pr-1") + lines = stream.getvalue().strip().splitlines() + assert len(lines) == 1 + parsed = json.loads(lines[0]) + assert parsed["event_type"] == "task_started" + assert parsed["pipeline_run_id"] == "pr-1" + + +def test_emit_never_raises_and_writes_fallback(monkeypatch) -> None: + stream = io.StringIO() + + def boom(*args, **kwargs): # noqa: ANN002, ANN003 + raise RuntimeError("emitter broke") + + monkeypatch.setattr("atlas.observability.logging.build_event", boom) + result = emit_event("task_started", stream=stream) + assert result is None + parsed = json.loads(stream.getvalue().strip()) + assert parsed["event_type"] == "telemetry_emit_failed" + assert parsed["severity"] == "ERROR" + + +def test_field_names_are_deterministic() -> None: + # The allowlist is the contract; renaming a field is a breaking change + # that must be made consciously here and in downstream log filters. + expected_core = { + "timestamp", + "severity", + "event_type", + "component", + "environment", + "git_sha", + "deployment_id", + "dag_id", + "task_id", + "airflow_run_id", + "pipeline_run_id", + "batch_id", + "processing_date", + "attempt_number", + "status", + "duration_ms", + "rows_generated", + "rows_loaded", + "rows_accepted", + "rows_rejected", + "fact_rows", + "mart_event_count", + "check_name", + "observed_value", + "threshold", + "error_type", + "error_message", + "correlation_id", + } + assert expected_core <= set(ALLOWED_FIELDS) + + +def test_correlation_extraction_from_context() -> None: + ctx = { + "pipeline_run_id": "pr-1", + "batch_id": "b-1", + "airflow_run_id": "ar-1", + "processing_date": "2026-07-19", + "unrelated": "x", + } + fields = correlation_fields_from_context(ctx) + assert fields == {"pipeline_run_id": "pr-1", "batch_id": "b-1", "airflow_run_id": "ar-1"} + + +def test_new_correlation_id_unique() -> None: + assert new_correlation_id() != new_correlation_id() + + +def test_cloud_emit_disabled_by_default(monkeypatch: pytest.MonkeyPatch) -> None: + """Without the opt-in env var, no Cloud Logging client is ever touched.""" + import atlas.observability.logging as obs_logging + + monkeypatch.delenv(obs_logging.CLOUD_EMIT_ENV_VAR, raising=False) + calls: list[dict] = [] + monkeypatch.setattr(obs_logging, "_emit_to_cloud", lambda e: calls.append(e)) + out = io.StringIO() + assert emit_event("task_started", stream=out) is not None + assert calls == [] + + +def test_cloud_emit_enabled_forwards_event(monkeypatch: pytest.MonkeyPatch) -> None: + import atlas.observability.logging as obs_logging + + monkeypatch.setenv(obs_logging.CLOUD_EMIT_ENV_VAR, "true") + calls: list[dict] = [] + monkeypatch.setattr(obs_logging, "_emit_to_cloud", lambda e: calls.append(e)) + out = io.StringIO() + emit_event("task_started", stream=out, pipeline_run_id="pr-1") + assert len(calls) == 1 + assert calls[0]["pipeline_run_id"] == "pr-1" + + +def test_cloud_emit_failure_does_not_break_stdout(monkeypatch: pytest.MonkeyPatch) -> None: + import atlas.observability.logging as obs_logging + + monkeypatch.setenv(obs_logging.CLOUD_EMIT_ENV_VAR, "true") + + def _boom(event: dict) -> None: + raise RuntimeError("cloud logging down") + + monkeypatch.setattr(obs_logging, "_emit_to_cloud", _boom) + out = io.StringIO() + # emit_event must not raise; the original contract line is printed before + # the cloud fan-out, so it is always present in the stream. + emit_event("task_started", stream=out) + first_line = out.getvalue().splitlines()[0] + assert json.loads(first_line)["event_type"] == "task_started" diff --git a/tests/unit/test_observability_metrics.py b/tests/unit/test_observability_metrics.py new file mode 100644 index 0000000..6bf50af --- /dev/null +++ b/tests/unit/test_observability_metrics.py @@ -0,0 +1,129 @@ +"""Cardinality-budget and catalog tests for atlas.observability.metrics.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from atlas.observability.metrics import ( + MONITOR_STATUS_VALUES, + MetricContractError, + load_catalog, + publish_gauge_safely, + validate_point, +) + +CATALOG = load_catalog() +CATALOG_PATH = Path(__file__).resolve().parents[2] / "observability" / "metrics" / "metric-descriptors.json" + +FORBIDDEN_LABELS = {"pipeline_run_id", "batch_id", "deployment_id", "error_message", "airflow_run_id"} + + +def test_catalog_loads_and_is_nonempty() -> None: + assert len(CATALOG) >= 10 + assert all(t.startswith("custom.googleapis.com/atlas/") for t in CATALOG) + + +def test_no_high_cardinality_labels_in_catalog() -> None: + for metric_type, spec in CATALOG.items(): + overlap = set(spec["labels"]) & FORBIDDEN_LABELS + assert not overlap, f"{metric_type} declares forbidden labels {overlap}" + + +def test_catalog_declares_kind_unit_and_value_type() -> None: + for spec in CATALOG.values(): + assert spec["metric_kind"] in {"GAUGE", "CUMULATIVE"} + assert spec["value_type"] in {"INT64", "DOUBLE"} + assert spec.get("unit") + assert spec.get("description") + + +def test_projected_cardinality_within_budget() -> None: + budget = json.loads(CATALOG_PATH.read_text(encoding="utf-8"))["cardinality_budget"] + max_series = budget["max_projected_time_series"] + sizes = { + "environment": 1, + "dag_id": 2, + "component": 4, + "severity": 3, + "mode": 2, + "status": 4, + "check_name": 15, + } + projected = 0 + for spec in CATALOG.values(): + series = 1 + for label in spec["labels"]: + series *= sizes[label] + projected += series + assert projected <= max_series, f"projected {projected} series exceeds budget {max_series}" + + +def test_validate_point_accepts_valid_labels() -> None: + descriptor = validate_point( + "custom.googleapis.com/atlas/monitor/check_status", + {"environment": "atlas-dev", "check_name": "freshness", "mode": "normal"}, + CATALOG, + ) + assert descriptor["value_type"] == "INT64" + + +def test_validate_point_rejects_unknown_metric() -> None: + with pytest.raises(MetricContractError, match="not in catalog"): + validate_point("custom.googleapis.com/atlas/bogus/metric", {}, CATALOG) + + +def test_validate_point_rejects_forbidden_label() -> None: + with pytest.raises(MetricContractError, match="not allowed"): + validate_point( + "custom.googleapis.com/atlas/data/rejection_rate", + {"environment": "atlas-dev", "mode": "normal", "pipeline_run_id": "pr-1"}, + CATALOG, + ) + + +def test_validate_point_rejects_unbounded_label_value() -> None: + with pytest.raises(MetricContractError, match="outside bounded set"): + validate_point( + "custom.googleapis.com/atlas/data/rejection_rate", + {"environment": "prod-42", "mode": "normal"}, + CATALOG, + ) + + +def test_validate_point_requires_all_labels() -> None: + with pytest.raises(MetricContractError, match="missing required labels"): + validate_point( + "custom.googleapis.com/atlas/data/rejection_rate", + {"environment": "atlas-dev"}, + CATALOG, + ) + + +def test_drill_mode_is_a_separate_series_not_a_pollutant() -> None: + # Drill points carry mode=drill so they never mix with normal series. + for labels_mode in ("normal", "drill"): + validate_point( + "custom.googleapis.com/atlas/pipeline/last_success_age_seconds", + {"environment": "atlas-dev", "dag_id": "atlas_batch_pipeline", "mode": labels_mode}, + CATALOG, + ) + + +def test_monitor_status_value_mapping_is_stable() -> None: + assert MONITOR_STATUS_VALUES == {"PASS": 0, "WARN": 1, "FAIL": 2, "NO_DATA": -1, "DISABLED": -2} + + +def test_publish_gauge_safely_never_raises(capsys) -> None: + # Contract violation inside safely-wrapper degrades to a structured event. + ok = publish_gauge_safely( + "example-gcp-project", + "custom.googleapis.com/atlas/bogus/metric", + 1, + {}, + catalog=CATALOG, + ) + assert ok is False + assert "metric_publish_failed" in capsys.readouterr().out diff --git a/tests/unit/test_observability_monitor.py b/tests/unit/test_observability_monitor.py new file mode 100644 index 0000000..bb0de3a --- /dev/null +++ b/tests/unit/test_observability_monitor.py @@ -0,0 +1,146 @@ +"""Unit tests for the observability monitor engine (Sprint 5 P8).""" + +from __future__ import annotations + +import json +from typing import Any + +from atlas.config.settings import load_settings +from atlas.observability.monitor import ( + CHECK_NAMES, + CHECKS, + check_cost_anomaly, + check_freshness, + check_latest_run_state, + check_rejection_rate, + check_volume_deviation, + load_config, + run_monitor, +) + +SETTINGS = load_settings() +CONFIG = load_config() + + +class FakeJob: + def __init__(self, rows: list[dict[str, Any]]) -> None: + self._rows = rows + + def result(self) -> list[dict[str, Any]]: + return self._rows + + +class FakeClient: + """Returns queued row sets per query, records all SQL.""" + + def __init__(self, row_sets: list[list[dict[str, Any]]] | None = None) -> None: + self._row_sets = list(row_sets or []) + self.queries: list[str] = [] + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + self.queries.append(sql) + if sql.strip().upper().startswith("MERGE"): + return FakeJob([]) + return FakeJob(self._row_sets.pop(0) if self._row_sets else []) + + +def test_check_names_and_registry_agree() -> None: + assert set(CHECK_NAMES) == set(CHECKS) + + +def test_latest_run_state_fail_on_failed_run() -> None: + client = FakeClient( + [[{"status": "FAILED", "pipeline_run_id": "pr-1", "started_at": None, "duration_s": 100}]] + ) + result = check_latest_run_state(client, CONFIG, "p") + assert result.status == "FAIL" and result.severity == "CRITICAL" + + +def test_latest_run_state_no_data() -> None: + assert check_latest_run_state(FakeClient([[]]), CONFIG, "p").status == "NO_DATA" + + +def test_freshness_thresholds() -> None: + warn = CONFIG["freshness"]["warn_seconds"] + fail = CONFIG["freshness"]["fail_seconds"] + assert check_freshness(FakeClient([[{"age_s": warn - 1}]]), CONFIG, "p").status == "PASS" + assert check_freshness(FakeClient([[{"age_s": warn + 1}]]), CONFIG, "p").status == "WARN" + assert check_freshness(FakeClient([[{"age_s": fail + 1}]]), CONFIG, "p").status == "FAIL" + assert check_freshness(FakeClient([[{"age_s": None}]]), CONFIG, "p").status == "NO_DATA" + + +def test_volume_deviation_classification() -> None: + def result_for(latest: int, baseline: int | None): + return check_volume_deviation( + FakeClient([[{"latest_rows": latest, "baseline_rows": baseline}]]), CONFIG, "p" + ) + + assert result_for(50000, 50000).status == "PASS" + assert result_for(20000, 50000).status == "WARN" # 60 % deviation + assert result_for(5000, 50000).status == "FAIL" # 90 % deviation + assert result_for(50000, None).status == "NO_DATA" + assert result_for(50000, 10).status == "NO_DATA" # baseline below floor + + +def test_rejection_rate_classification() -> None: + def result_for(rejected: int): + rows = [{"rows_loaded": 50000, "rows_accepted": 50000 - rejected, "rows_rejected": rejected}] + return check_rejection_rate(FakeClient([rows]), CONFIG, "p") + + assert result_for(5000).status == "PASS" # 10 % + assert result_for(12500).status == "WARN" # 25 % + assert result_for(20000).status == "FAIL" # 40 % + + +def test_cost_anomaly_below_floor_passes() -> None: + rows = [{"window_bytes": 10_000, "window_jobs": 5, "baseline_daily_bytes": 100}] + result = check_cost_anomaly(FakeClient([rows]), CONFIG, "p") + assert result.status == "PASS" + assert result.details["reason"] == "below absolute floor" + + +def test_cost_anomaly_ratio_fail() -> None: + floor = CONFIG["cost"]["min_bytes_billed"] + rows = [{"window_bytes": floor * 20, "window_jobs": 5, "baseline_daily_bytes": floor}] + assert check_cost_anomaly(FakeClient([rows]), CONFIG, "p").status == "FAIL" + + +def test_run_monitor_disabled_emits_disabled_everywhere(capsys) -> None: + config = json.loads(json.dumps(CONFIG)) + config["monitoring_enabled"] = False + results = run_monitor(settings=SETTINGS, client=FakeClient(), config=config, persist=False, publish=False) + assert {r.status for r in results} == {"DISABLED"} + assert len(results) == len(CHECK_NAMES) + + +def test_run_monitor_one_broken_check_does_not_hide_others() -> None: + class ExplodingClient(FakeClient): + def query(self, sql: str, job_config: Any = None) -> FakeJob: + raise RuntimeError("backend down") + + results = run_monitor( + settings=SETTINGS, client=ExplodingClient(), config=CONFIG, persist=False, publish=False + ) + assert len(results) == len(CHECK_NAMES) + assert all(r.status in {"NO_DATA"} for r in results) + + +def test_drill_override_via_env(monkeypatch) -> None: + monkeypatch.setenv( + "ATLAS_OBSERVABILITY_OVERRIDES_JSON", + json.dumps({"freshness": {"fail_seconds": 60}, "runtime_mode": "drill"}), + ) + config = load_config() + assert config["freshness"]["fail_seconds"] == 60 + assert config["runtime_mode"] == "drill" + # untouched sections survive + assert config["volume"]["baseline_window_runs"] == CONFIG["volume"]["baseline_window_runs"] + + +def test_config_defaults_are_sane() -> None: + assert CONFIG["monitoring_enabled"] is True + assert CONFIG["runtime_mode"] == "normal" + assert CONFIG["freshness"]["warn_seconds"] < CONFIG["freshness"]["fail_seconds"] + assert CONFIG["volume"]["warn_deviation"] < CONFIG["volume"]["fail_deviation"] + assert CONFIG["rejection_rate"]["warn"] < CONFIG["rejection_rate"]["fail"] + assert CONFIG["cost"]["warn_ratio"] < CONFIG["cost"]["fail_ratio"] diff --git a/tests/unit/test_quality_results.py b/tests/unit/test_quality_results.py new file mode 100644 index 0000000..5797278 --- /dev/null +++ b/tests/unit/test_quality_results.py @@ -0,0 +1,177 @@ +"""Unit tests for quality_results / monitor_evaluations and the warehouse bridge.""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from atlas.config.settings import load_settings +from atlas.observability.checks import CHECK_CATEGORIES, persist_warehouse_report +from atlas.ops.quality_results import ( + MonitorEvaluationRecord, + QualityResultRecord, + details_to_json, + upsert_monitor_evaluation, + upsert_quality_result, +) +from atlas.validation.warehouse import WarehouseCheck, WarehouseReport + + +class FakeJob: + def result(self) -> list[Any]: + return [] + + +class FakeClient: + def __init__(self, fail: bool = False) -> None: + self.queries: list[str] = [] + self.params: list[dict[str, Any]] = [] + self._fail = fail + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + if self._fail: + raise RuntimeError("BigQuery unavailable") + self.queries.append(sql) + if job_config is not None: + self.params.append({p.name: p.value for p in job_config.query_parameters}) + return FakeJob() + + +SETTINGS = load_settings() + + +def _quality(**overrides: Any) -> QualityResultRecord: + base: dict[str, Any] = { + "pipeline_run_id": "pr-1", + "check_name": "raw_equals_classification", + "check_category": "RECONCILIATION", + "severity": "INFO", + "status": "PASS", + "evaluated_at": "2026-07-19T03:00:00+00:00", + "batch_id": "b-1", + "observed_value": 50000.0, + "expected_value": 50000.0, + } + base.update(overrides) + return QualityResultRecord(**base) + + +def test_quality_merge_keyed_by_run_and_check() -> None: + client = FakeClient() + upsert_quality_result(_quality(), SETTINGS, client=client) + sql = client.queries[0] + assert "MERGE" in sql and "atlas_ops.quality_results" in sql + assert "target.pipeline_run_id = @pipeline_run_id" in sql + assert "target.check_name = @check_name" in sql + + +def test_quality_rejects_invalid_vocabulary() -> None: + client = FakeClient() + with pytest.raises(ValueError, match="check category"): + upsert_quality_result(_quality(check_category="VIBES"), SETTINGS, client=client) + with pytest.raises(ValueError, match="quality status"): + upsert_quality_result(_quality(status="MEH"), SETTINGS, client=client) + with pytest.raises(ValueError, match="severity"): + upsert_quality_result(_quality(severity="LOUD"), SETTINGS, client=client) + assert client.queries == [] + + +def test_quality_details_truncated() -> None: + client = FakeClient() + upsert_quality_result(_quality(details_json="x" * 10000), SETTINGS, client=client) + assert len(client.params[0]["details_json"]) <= 4000 + + +def test_monitor_evaluation_merge_keyed_by_evaluation_id() -> None: + client = FakeClient() + upsert_monitor_evaluation( + MonitorEvaluationRecord( + evaluation_id="eval-1", + check_name="freshness", + environment="atlas-dev", + status="PASS", + evaluated_at="2026-07-19T03:00:00+00:00", + observed_value=120.0, + threshold=93600.0, + ), + SETTINGS, + client=client, + ) + sql = client.queries[0] + assert "atlas_ops.monitor_evaluations" in sql + assert "target.evaluation_id = @evaluation_id" in sql + + +def test_monitor_evaluation_allows_no_data_and_disabled() -> None: + client = FakeClient() + for status in ("NO_DATA", "DISABLED"): + upsert_monitor_evaluation( + MonitorEvaluationRecord( + evaluation_id=f"eval-{status}", + check_name="freshness", + environment="atlas-dev", + status=status, + evaluated_at="2026-07-19T03:00:00+00:00", + ), + SETTINGS, + client=client, + ) + assert len(client.queries) == 2 + + +def test_monitor_evaluation_rejects_invalid_status() -> None: + client = FakeClient() + with pytest.raises(ValueError, match="evaluation status"): + upsert_monitor_evaluation( + MonitorEvaluationRecord( + evaluation_id="eval-x", + check_name="freshness", + environment="atlas-dev", + status="ON_FIRE", + evaluated_at="2026-07-19T03:00:00+00:00", + ), + SETTINGS, + client=client, + ) + + +def test_details_to_json_handles_non_serializable() -> None: + class Weird: + def __repr__(self) -> str: + return "" + + rendered = details_to_json({"obj": Weird()}) + assert rendered is not None and "" in rendered + assert details_to_json(None) is None + + +def _report(status: str = "PASS") -> WarehouseReport: + checks = [ + WarehouseCheck(name=name, status=status, expected=1, actual=1, message="m") + for name in CHECK_CATEGORIES + ] + return WarehouseReport(batch_id="b-1", overall_status=status, checks=checks) + + +def test_persist_warehouse_report_writes_one_row_per_check() -> None: + client = FakeClient() + written = persist_warehouse_report(_report(), "pr-1", settings=SETTINGS, client=client) + assert written == len(CHECK_CATEGORIES) + names = {p["check_name"] for p in client.params} + assert names == set(CHECK_CATEGORIES) + categories = {p["check_name"]: p["check_category"] for p in client.params} + assert categories == CHECK_CATEGORIES + + +def test_persist_warehouse_report_failed_checks_are_critical() -> None: + client = FakeClient() + persist_warehouse_report(_report(status="FAIL"), "pr-1", settings=SETTINGS, client=client) + assert all(p["severity"] == "CRITICAL" and p["status"] == "FAIL" for p in client.params) + + +def test_persist_warehouse_report_survives_backend_failure(capsys) -> None: + written = persist_warehouse_report(_report(), "pr-1", settings=SETTINGS, client=FakeClient(fail=True)) + assert written == 0 + out = capsys.readouterr().out + assert "quality_result_write_failed" in out diff --git a/tests/unit/test_recovery_actions.py b/tests/unit/test_recovery_actions.py new file mode 100644 index 0000000..4ae12ca --- /dev/null +++ b/tests/unit/test_recovery_actions.py @@ -0,0 +1,144 @@ +"""Recovery-action audit tests (Sprint 6, Phase 4 / ADR-014).""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from atlas.config.settings import load_settings +from atlas.ops.recovery_actions import ( + ALLOWED_ACTION_TYPES, + RecoveryActionRecord, + finalize_recovery_action, + start_recovery_action, + upsert_recovery_action, +) + + +class FakeJob: + def result(self) -> list: + return [] + + +class FakeClient: + def __init__(self) -> None: + self.queries: list[str] = [] + self.params: list[dict[str, Any]] = [] + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + self.queries.append(sql) + if job_config is not None: + self.params.append({p.name: p.value for p in job_config.query_parameters}) + return FakeJob() + + +SETTINGS = load_settings() + + +def _record(**overrides: Any) -> RecoveryActionRecord: + base: dict[str, Any] = { + "recovery_id": "rec-1", + "action_type": "RERUN_BATCH", + "status": "RUNNING", + "scenario_id": "S6-ING-001", + "batch_id": "atlas-s6-a", + "verification_status": "PENDING", + } + base.update(overrides) + return RecoveryActionRecord(**base) + + +def test_upsert_uses_idempotent_merge_on_recovery_id() -> None: + client = FakeClient() + upsert_recovery_action(_record(), SETTINGS, client=client) + sql = client.queries[0] + assert "MERGE" in sql and "atlas_ops.recovery_actions" in sql + assert "ON target.recovery_id = @recovery_id" in sql + + +def test_repeated_finalization_is_merge_not_duplicate_insert() -> None: + client = FakeClient() + record = _record() + for _ in range(3): + finalize_recovery_action( + record, status="SUCCESS", verification_status="VERIFIED", client=client, settings=SETTINGS + ) + assert all("WHEN MATCHED THEN" in sql for sql in client.queries) + assert {p["recovery_id"] for p in client.params} == {"rec-1"} + + +def test_success_requires_verified_status() -> None: + client = FakeClient() + with pytest.raises(ValueError, match="requires verification_status=VERIFIED"): + upsert_recovery_action( + _record(status="SUCCESS", verification_status="PENDING"), SETTINGS, client=client + ) + with pytest.raises(ValueError, match="requires verification_status=VERIFIED"): + finalize_recovery_action( + _record(), status="SUCCESS", verification_status="FAILED", client=client, settings=SETTINGS + ) + assert client.queries == [] + + +def test_partial_and_failed_states_allowed_without_verification() -> None: + client = FakeClient() + finalize_recovery_action( + _record(), status="PARTIAL", verification_status="FAILED", client=client, settings=SETTINGS + ) + finalize_recovery_action( + _record(recovery_id="rec-2"), + status="FAILED", + verification_status="SKIPPED", + error_type="RuntimeError", + error_summary="repair query failed", + client=client, + settings=SETTINGS, + ) + assert len(client.queries) == 2 + + +def test_invalid_action_type_and_status_rejected() -> None: + client = FakeClient() + with pytest.raises(ValueError, match="Unsupported recovery action type"): + upsert_recovery_action(_record(action_type="WISH_HARDER"), SETTINGS, client=client) + with pytest.raises(ValueError, match="Unsupported recovery status"): + upsert_recovery_action(_record(status="MAYBE"), SETTINGS, client=client) + assert client.queries == [] + + +def test_all_controlled_action_types_accepted() -> None: + client = FakeClient() + for i, action in enumerate(sorted(ALLOWED_ACTION_TYPES)): + upsert_recovery_action(_record(recovery_id=f"rec-{i}", action_type=action), SETTINGS, client=client) + assert len(client.queries) == len(ALLOWED_ACTION_TYPES) + + +def test_error_summary_is_sanitized() -> None: + client = FakeClient() + secret = "-----BEGIN " + "PRIVATE KEY-----abc" + upsert_recovery_action(_record(status="FAILED", error_summary=f"boom {secret}"), SETTINGS, client=client) + assert "BEGIN PRIVATE KEY" not in client.params[0]["error_summary"] + assert "[REDACTED]" in client.params[0]["error_summary"] + + +def test_start_links_incident_scenario_and_pipeline_grains(capsys: pytest.CaptureFixture[str]) -> None: + client = FakeClient() + record = start_recovery_action( + recovery_id="rec-9", + action_type="RECONSTRUCT_AUDIT", + incident_id="0.abc123", + scenario_id="S6-AIR-003", + pipeline_run_id="atlas-s6-air003-run", + batch_id="atlas-s6-air003", + deployment_id="atlas-dev-20260720T000000Z-deadbeef", + settings=SETTINGS, + client=client, + ) + assert record.status == "RUNNING" + assert record.verification_status == "PENDING" + params = client.params[0] + assert params["incident_id"] == "0.abc123" + assert params["scenario_id"] == "S6-AIR-003" + assert params["deployment_id"] == "atlas-dev-20260720T000000Z-deadbeef" + assert "recovery_action_started" in capsys.readouterr().out diff --git a/tests/unit/test_retention.py b/tests/unit/test_retention.py new file mode 100644 index 0000000..f0ecfdb --- /dev/null +++ b/tests/unit/test_retention.py @@ -0,0 +1,89 @@ +"""Sprint 7 Phase 9: classification & retention validation tests.""" + +from __future__ import annotations + +from atlas.governance import retention + + +def test_live_retention_config_is_valid() -> None: + assert retention.validate_retention_config() == [] + + +def test_permanent_audit_has_no_expiration() -> None: + assert retention.desired_expiration_days("operational_audit") is None + assert retention.desired_expiration_days("release_evidence") is None + + +def test_transient_classes_have_expiration() -> None: + assert retention.desired_expiration_days("temporary_integration") is not None + assert retention.desired_expiration_days("observability_logs") == 30 + + +def test_plan_marks_permanent_evidence_keep_forever() -> None: + plan = {p["asset_id"]: p for p in retention.plan_expirations()} + # Operational audit tables must be keep_forever (never expired). + audit = plan["atlas_ops.pipeline_runs"] + assert audit["disposition"] == "keep_forever" + assert audit["is_permanent_evidence"] is True + + +def test_plan_release_bundles_retained() -> None: + plan = {p["asset_id"]: p for p in retention.plan_expirations()} + bundles = plan["gcs://atlas-deployments-example-gcp-project"] + assert bundles["disposition"] == "keep_forever" + + +def test_conflicting_permanent_expiration_fails(monkeypatch) -> None: + bad = { + "classes": { + "operational_audit": { + "description": "x", + "retention": "indefinite", + "expiration_days": 7, # conflict: permanent + expiration + "is_permanent_evidence": True, + } + } + } + monkeypatch.setattr(retention, "load_retention", lambda: bad) + monkeypatch.setattr(retention, "load_policy", lambda: {"retention_classes": ["operational_audit"]}) + errors = retention.validate_retention_config() + assert any("permanent evidence cannot have an expiration" in e for e in errors) + + +def test_transient_without_expiration_fails(monkeypatch) -> None: + bad = { + "classes": { + "temporary_integration": { + "description": "x", + "retention": "short", + "expiration_days": None, # conflict: transient must expire + "is_permanent_evidence": False, + } + } + } + monkeypatch.setattr(retention, "load_retention", lambda: bad) + monkeypatch.setattr(retention, "load_policy", lambda: {"retention_classes": ["temporary_integration"]}) + errors = retention.validate_retention_config() + assert any("transient class must set expiration_days" in e for e in errors) + + +def test_policy_retention_class_drift_fails(monkeypatch) -> None: + monkeypatch.setattr( + retention, + "load_retention", + lambda: { + "classes": { + "canonical_warehouse": { + "description": "x", + "retention": "y", + "expiration_days": None, + "is_permanent_evidence": False, + } + } + }, + ) + monkeypatch.setattr( + retention, "load_policy", lambda: {"retention_classes": ["canonical_warehouse", "raw_landing"]} + ) + errors = retention.validate_retention_config() + assert any("but not retention.yml" in e for e in errors) diff --git a/tests/unit/test_rollback_compatibility.py b/tests/unit/test_rollback_compatibility.py new file mode 100644 index 0000000..4539615 --- /dev/null +++ b/tests/unit/test_rollback_compatibility.py @@ -0,0 +1,88 @@ +"""Rollback schema-compatibility decision tests (Sprint 6, ADR-015).""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from atlas.ops.migrations import Migration, load_manifest +from atlas.ops.rollback_compatibility import evaluate_rollback_compatibility + + +def _migration(migration_id: str, breaking: bool = False) -> Migration: + return Migration( + migration_id=migration_id, + sql_path=Path("/dev/null"), + checksum="0" * 64, + breaking=breaking, + ) + + +MANIFEST = [ + _migration("001_base"), + _migration("002_additive"), + _migration("003_breaking_type_change", breaking=True), +] + + +def test_rollback_allowed_when_newer_migrations_are_additive() -> None: + decision = evaluate_rollback_compatibility( + applied_migration_ids=["001_base", "002_additive"], + target_release_migration_ids=["001_base"], + manifest=MANIFEST, + ) + assert decision.eligible is True + assert "additive" in decision.reason + + +def test_rollback_blocked_across_breaking_migration() -> None: + """S6-RBK-003: irreversible migration blocks runtime rollback.""" + decision = evaluate_rollback_compatibility( + applied_migration_ids=["001_base", "002_additive", "003_breaking_type_change"], + target_release_migration_ids=["001_base", "002_additive"], + manifest=MANIFEST, + ) + assert decision.eligible is False + assert decision.blocking_migrations == ("003_breaking_type_change",) + assert "recover forward" in decision.reason + assert "never reversed automatically" in decision.reason + + +def test_target_knowing_breaking_migration_is_eligible() -> None: + """A release built after the breaking migration may still be a target.""" + decision = evaluate_rollback_compatibility( + applied_migration_ids=["001_base", "002_additive", "003_breaking_type_change"], + target_release_migration_ids=["001_base", "002_additive", "003_breaking_type_change"], + manifest=MANIFEST, + ) + assert decision.eligible is True + + +def test_unclassifiable_applied_migration_refuses_rollback() -> None: + """Unknown applied migrations are refused rather than guessed compatible.""" + decision = evaluate_rollback_compatibility( + applied_migration_ids=["001_base", "999_mystery"], + target_release_migration_ids=["001_base"], + manifest=MANIFEST, + ) + assert decision.eligible is False + assert "cannot be classified" in decision.reason + + +def test_manifest_parses_breaking_flag(tmp_path: Path) -> None: + sql = tmp_path / "001.sql" + sql.write_text("SELECT 1") + manifest = tmp_path / "manifest.txt" + manifest.write_text("001_base|001.sql\n002_breaking|001.sql|breaking\n") + migrations = load_manifest(manifest) + assert [m.breaking for m in migrations] == [False, True] + + +def test_manifest_rejects_unknown_flags(tmp_path: Path) -> None: + sql = tmp_path / "001.sql" + sql.write_text("SELECT 1") + manifest = tmp_path / "manifest.txt" + manifest.write_text("001_base|001.sql|yolo\n") + with pytest.raises(ValueError, match="Unknown manifest flags"): + load_manifest(manifest) diff --git a/tests/unit/test_schema_check.py b/tests/unit/test_schema_check.py new file mode 100644 index 0000000..f87c61c --- /dev/null +++ b/tests/unit/test_schema_check.py @@ -0,0 +1,145 @@ +"""Sprint 7 Phase 4: schema compatibility checker classification tests.""" + +from __future__ import annotations + +import copy + +from atlas.governance import schema_check as sc + + +def _baseline() -> dict: + return { + "version": 1, + "assets": { + "fct_events": { + "contract_version": "1.0", + "grain": "one row per event_id", + "partition_field": "event_date", + "event_identity": ["event_id"], + "fields": { + "event_id": {"type": "string", "nullable": False}, + "user_id": {"type": "string", "nullable": False}, + "platform": { + "type": "string", + "nullable": False, + "accepted_values": ["ios", "android", "web"], + }, + }, + } + }, + } + + +def test_identical_manifests_are_compatible() -> None: + report = sc.compare_manifests(_baseline(), _baseline()) + assert report.overall_class == sc.COMPATIBLE + assert report.changes == [] + + +def test_added_nullable_field_is_compatible() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["app_version"] = {"type": "string", "nullable": True} + cand["assets"]["fct_events"]["contract_version"] = "1.1" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.COMPATIBLE + assert any(c.change_type == "field_added" for c in report.changes) + + +def test_added_required_field_is_conditionally_compatible() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["tenant_id"] = {"type": "string", "nullable": False} + cand["assets"]["fct_events"]["contract_version"] = "1.1" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.CONDITIONALLY_COMPATIBLE + + +def test_removed_field_is_breaking() -> None: + cand = _baseline() + del cand["assets"]["fct_events"]["fields"]["user_id"] + cand["assets"]["fct_events"]["contract_version"] = "2.0" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.BREAKING + assert any(c.change_type == "field_removed" for c in report.changes) + + +def test_type_change_is_breaking() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["user_id"]["type"] = "int64" + cand["assets"]["fct_events"]["contract_version"] = "2.0" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.BREAKING + assert any(c.change_type == "type_changed" for c in report.changes) + + +def test_grain_change_is_breaking() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["grain"] = "one row per (event_id, event_date)" + cand["assets"]["fct_events"]["contract_version"] = "2.0" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.BREAKING + assert any(c.change_type == "grain_changed" for c in report.changes) + + +def test_partition_change_is_breaking() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["partition_field"] = "ingested_date" + cand["assets"]["fct_events"]["contract_version"] = "2.0" + report = sc.compare_manifests(_baseline(), cand) + assert any(c.change_type == "partition_field_changed" for c in report.changes) + + +def test_narrowed_enum_is_breaking() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["platform"]["accepted_values"] = ["ios", "android"] + cand["assets"]["fct_events"]["contract_version"] = "2.0" + report = sc.compare_manifests(_baseline(), cand) + assert any(c.change_type == "accepted_values_narrowed" for c in report.changes) + + +def test_widened_enum_is_compatible() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["platform"]["accepted_values"] = [ + "ios", + "android", + "web", + "desktop", + ] + cand["assets"]["fct_events"]["contract_version"] = "1.1" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.COMPATIBLE + assert any(c.change_type == "accepted_values_widened" for c in report.changes) + + +def test_unversioned_change_is_prohibited() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["app_version"] = {"type": "string", "nullable": True} + # contract_version left at 1.0 despite the schema change. + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.PROHIBITED + assert any(c.change_type == "unversioned_change" for c in report.changes) + + +def test_contract_downgrade_is_prohibited() -> None: + cand = _baseline() + cand["assets"]["fct_events"]["fields"]["app_version"] = {"type": "string", "nullable": True} + cand["assets"]["fct_events"]["contract_version"] = "0.9" + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.PROHIBITED + + +def test_generated_manifest_matches_committed_baseline() -> None: + # The committed baseline must equal a fresh generation (drift guard). + import json + + from atlas.config.settings import atlas_root + + committed = json.loads((atlas_root() / "governance/schemas/manifests/baseline.json").read_text()) + fresh = sc.generate_manifest() + assert committed == fresh, "baseline manifest is stale; regenerate with --generate" + + +def test_new_asset_is_compatible() -> None: + cand = copy.deepcopy(_baseline()) + cand["assets"]["new_model"] = {"contract_version": "1.0", "grain": "x", "fields": {}} + report = sc.compare_manifests(_baseline(), cand) + assert report.overall_class == sc.COMPATIBLE diff --git a/tests/unit/test_schema_drift.py b/tests/unit/test_schema_drift.py new file mode 100644 index 0000000..35c3f1b --- /dev/null +++ b/tests/unit/test_schema_drift.py @@ -0,0 +1,103 @@ +"""Classification tests for atlas.observability.schema_drift (Sprint 5 P9).""" + +from __future__ import annotations + +from atlas.observability.schema_drift import compare_table, load_manifest, summarize + +EXPECTED = { + "partition_column": "event_date", + "columns": { + "event_id": {"data_type": "STRING", "is_nullable": "NO"}, + "event_date": {"data_type": "DATE", "is_nullable": "NO"}, + "amount": {"data_type": "FLOAT64", "is_nullable": "YES"}, + }, +} + + +def _live(**overrides): + base = { + "event_id": {"data_type": "STRING", "is_nullable": "NO", "is_partitioning_column": "NO"}, + "event_date": {"data_type": "DATE", "is_nullable": "NO", "is_partitioning_column": "YES"}, + "amount": {"data_type": "FLOAT64", "is_nullable": "YES", "is_partitioning_column": "NO"}, + } + base.update(overrides) + return base + + +def test_identical_schema_has_no_findings() -> None: + assert compare_table("ds.t", EXPECTED, _live()) == [] + + +def test_missing_table_is_breaking() -> None: + findings = compare_table("ds.t", EXPECTED, None) + assert [f.classification for f in findings] == ["BREAKING"] + assert findings[0].kind == "missing_table" + + +def test_removed_field_is_breaking() -> None: + live = _live() + del live["amount"] + findings = compare_table("ds.t", EXPECTED, live) + assert any(f.kind == "removed_field" and f.classification == "BREAKING" for f in findings) + + +def test_type_change_is_breaking() -> None: + live = _live(amount={"data_type": "STRING", "is_nullable": "YES", "is_partitioning_column": "NO"}) + findings = compare_table("ds.t", EXPECTED, live) + assert any(f.kind == "type_change" and f.classification == "BREAKING" for f in findings) + + +def test_required_made_nullable_is_breaking() -> None: + live = _live(event_id={"data_type": "STRING", "is_nullable": "YES", "is_partitioning_column": "NO"}) + findings = compare_table("ds.t", EXPECTED, live) + assert any(f.kind == "required_made_nullable" and f.classification == "BREAKING" for f in findings) + + +def test_partition_change_is_breaking() -> None: + live = _live(event_date={"data_type": "DATE", "is_nullable": "NO", "is_partitioning_column": "NO"}) + findings = compare_table("ds.t", EXPECTED, live) + assert any(f.kind == "partition_change" and f.classification == "BREAKING" for f in findings) + + +def test_unapproved_nullable_field_is_warning() -> None: + live = _live(new_col={"data_type": "STRING", "is_nullable": "YES", "is_partitioning_column": "NO"}) + findings = compare_table("ds.t", EXPECTED, live) + assert [f.classification for f in findings] == ["WARNING"] + assert findings[0].kind == "unapproved_new_field" + + +def test_approved_new_field_is_allowed() -> None: + live = _live(new_col={"data_type": "STRING", "is_nullable": "YES", "is_partitioning_column": "NO"}) + findings = compare_table("ds.t", EXPECTED, live, allowed_new_fields=["ds.t.new_col"]) + assert [f.classification for f in findings] == ["ALLOWED"] + + +def test_new_required_field_is_breaking() -> None: + live = _live(new_req={"data_type": "STRING", "is_nullable": "NO", "is_partitioning_column": "NO"}) + findings = compare_table("ds.t", EXPECTED, live) + assert [f.classification for f in findings] == ["BREAKING"] + assert findings[0].kind == "unapproved_required_field" + + +def test_summarize_counts_by_classification() -> None: + live = _live( + new_col={"data_type": "STRING", "is_nullable": "YES", "is_partitioning_column": "NO"}, + event_id={"data_type": "INT64", "is_nullable": "NO", "is_partitioning_column": "NO"}, + ) + counts = summarize(compare_table("ds.t", EXPECTED, live)) + assert counts == {"ALLOWED": 0, "WARNING": 1, "BREAKING": 1} + + +def test_shipped_manifest_covers_governed_tables() -> None: + manifest = load_manifest() + tables = set(manifest["tables"]) + required = { + "atlas_raw.events", + "atlas_core.fct_events", + "atlas_ops.pipeline_runs", + "atlas_ops.deployments", + "atlas_ops.task_events", + "atlas_ops.quality_results", + "atlas_ops.monitor_evaluations", + } + assert required <= tables diff --git a/tests/unit/test_schema_versions.py b/tests/unit/test_schema_versions.py new file mode 100644 index 0000000..0bd1186 --- /dev/null +++ b/tests/unit/test_schema_versions.py @@ -0,0 +1,79 @@ +"""Multi-version event schema tests (Sprint 6, S6-SCH-008).""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from atlas.validation.schema_versions import ( + CURRENT_SCHEMA_VERSION, + SchemaVersionError, + detect_schema_version, + normalize_event, +) + + +def _v1_event(**overrides: Any) -> dict[str, Any]: + base: dict[str, Any] = { + "event_id": "e-1", + "event_type": "page_view", + "event_timestamp": "2026-07-19T00:00:00+00:00", + "user_id": "u-1", + "country_code": "US", + "device_type": "mobile", + "session_id": "s-1", + "payload_size_bytes": 512, + "batch_id": "atlas-s6-sch008", + "processing_date": "2026-07-19", + } + base.update(overrides) + return base + + +def test_missing_discriminator_means_version_1() -> None: + assert detect_schema_version(_v1_event()) == 1 + + +def test_v2_discriminator_detected() -> None: + assert detect_schema_version(_v1_event(schema_version=2, client_app_version="3.1.0")) == 2 + + +def test_unknown_version_rejected_not_guessed() -> None: + with pytest.raises(SchemaVersionError, match="unsupported schema_version 99"): + detect_schema_version(_v1_event(schema_version=99)) + + +def test_non_integer_version_rejected() -> None: + with pytest.raises(SchemaVersionError, match="not an integer"): + detect_schema_version(_v1_event(schema_version="latest")) + + +def test_v1_normalizes_with_explicit_nulls_for_newer_fields() -> None: + normalized = normalize_event(_v1_event()) + assert normalized["schema_version"] == 1 + assert normalized["client_app_version"] is None + assert normalized["event_id"] == "e-1" + + +def test_v2_normalizes_to_current_shape() -> None: + normalized = normalize_event(_v1_event(schema_version=2, client_app_version="3.1.0")) + assert normalized["schema_version"] == CURRENT_SCHEMA_VERSION + assert normalized["client_app_version"] == "3.1.0" + + +def test_normalized_shape_is_identical_across_versions() -> None: + v1_keys = set(normalize_event(_v1_event())) + v2_keys = set(normalize_event(_v1_event(schema_version=2, client_app_version=None))) + assert v1_keys == v2_keys + + +def test_unknown_field_rejected_no_silent_coercion() -> None: + with pytest.raises(SchemaVersionError, match="not part of schema version 1"): + normalize_event(_v1_event(surprise_field="boo")) + + +def test_v2_only_field_rejected_on_v1_event() -> None: + """A v1 event smuggling a v2 field is rejected — versions are explicit.""" + with pytest.raises(SchemaVersionError, match="not part of schema version 1"): + normalize_event(_v1_event(client_app_version="3.1.0")) diff --git a/tests/unit/test_security_policy.py b/tests/unit/test_security_policy.py new file mode 100644 index 0000000..59d07ad --- /dev/null +++ b/tests/unit/test_security_policy.py @@ -0,0 +1,47 @@ +"""Sprint 7 Phase 8: security-policy scanner regression tests.""" + +from __future__ import annotations + +from atlas.governance import security_policy as sp + + +def test_live_managed_iam_is_clean() -> None: + assert sp.scan_managed_iam() == [] + + +def test_live_data_exposure_is_clean() -> None: + # Governed Atlas artifacts must never commit secret-like values. + assert sp.scan_data_exposure() == [] + + +def test_scan_text_flags_private_key() -> None: + reasons = sp.scan_text("-----BEGIN RSA PRIVATE KEY-----\nabc\n-----END-----") + assert "private key material" in reasons + + +def test_scan_text_flags_gcp_api_key() -> None: + reasons = sp.scan_text("key=AIza" + "A" * 35) + assert "GCP API key" in reasons + + +def test_scan_text_flags_service_account_json() -> None: + assert "service-account JSON" in sp.scan_text('{"type": "service_account"}') + + +def test_scan_text_flags_slack_webhook() -> None: + reasons = sp.scan_text("url: https://hooks.slack.com/services/T000/B000/xxxxxxxx") + assert "Slack webhook URL" in reasons + + +def test_scan_text_flags_personal_email() -> None: + assert "personal email address" in sp.scan_text("recipient: someone@gmail.com") + + +def test_scan_text_allows_variable_references() -> None: + # Variable references are not secrets and must not be flagged. + assert sp.scan_text("Authorization: Bearer $token") == [] + assert sp.scan_text("channel: ${NOTIFICATION_CHANNEL}") == [] + + +def test_scan_text_clean_string() -> None: + assert sp.scan_text("just a normal config line with no secrets") == [] diff --git a/tests/unit/test_settings.py b/tests/unit/test_settings.py new file mode 100644 index 0000000..09ac5eb --- /dev/null +++ b/tests/unit/test_settings.py @@ -0,0 +1,18 @@ +"""Unit tests for Atlas settings.""" + +from __future__ import annotations + +from atlas.config.settings import load_settings, staging_table_id, table_fqn + + +def test_load_settings_defaults() -> None: + settings = load_settings() + assert settings.gcp.project_id == "example-gcp-project" + assert settings.generator.event_count == 50000 + assert settings.anomaly_profile.expected_count("null_user_ids") == 500 + + +def test_table_fqn() -> None: + settings = load_settings() + assert table_fqn(settings) == "example-gcp-project.atlas_raw.events" + assert staging_table_id(settings, "atlas-run-1").endswith("_staging_atlas_run_1") diff --git a/tests/unit/test_task_events.py b/tests/unit/test_task_events.py new file mode 100644 index 0000000..0af7cbd --- /dev/null +++ b/tests/unit/test_task_events.py @@ -0,0 +1,169 @@ +"""Unit tests for atlas_ops.task_events (Sprint 5 Phase 3).""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from atlas.config.settings import load_settings +from atlas.ops.task_events import ( + EXPECTED_TERMINAL_TASKS, + TaskEventRecord, + record_task_event_safely, + telemetry_completeness, + upsert_task_event, +) + + +class FakeJob: + def __init__(self, rows: list[dict[str, Any]] | None = None) -> None: + self._rows = rows or [] + + def result(self) -> list[dict[str, Any]]: + return self._rows + + +class FakeClient: + def __init__(self, select_rows: list[dict[str, Any]] | None = None) -> None: + self.queries: list[str] = [] + self.params: list[dict[str, Any]] = [] + self._select_rows = select_rows or [] + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + self.queries.append(sql) + if job_config is not None: + self.params.append({p.name: p.value for p in job_config.query_parameters}) + if sql.strip().upper().startswith("SELECT"): + return FakeJob(self._select_rows) + return FakeJob() + + +class BrokenClient: + def query(self, sql: str, job_config: Any = None) -> FakeJob: + raise RuntimeError("BigQuery unavailable") + + +SETTINGS = load_settings() + + +def _record(**overrides: Any) -> TaskEventRecord: + base: dict[str, Any] = { + "pipeline_run_id": "pr-1", + "task_id": "load_bigquery_raw", + "attempt_number": 1, + "event_type": "SUCCESS", + "batch_id": "b-1", + "status": "SUCCESS", + } + base.update(overrides) + return TaskEventRecord(**base) + + +def test_upsert_uses_merge_on_full_attempt_key() -> None: + client = FakeClient() + upsert_task_event(_record(), SETTINGS, client=client) + sql = client.queries[0] + assert "MERGE" in sql and "atlas_ops.task_events" in sql + for key in ("task_id = @task_id", "attempt_number = @attempt_number", "event_type = @event_type"): + assert key in sql + + +def test_first_attempt_success_row() -> None: + client = FakeClient() + upsert_task_event(_record(), SETTINGS, client=client) + assert client.params[0]["attempt_number"] == 1 + assert client.params[0]["event_type"] == "SUCCESS" + + +def test_retry_then_success_are_distinct_rows() -> None: + client = FakeClient() + upsert_task_event( + _record(attempt_number=1, event_type="FAILED", status="FAILED"), SETTINGS, client=client + ) + upsert_task_event(_record(attempt_number=2, event_type="SUCCESS"), SETTINGS, client=client) + # Different attempt numbers hit different MERGE keys: two writes, two keys. + assert (client.params[0]["attempt_number"], client.params[0]["event_type"]) == (1, "FAILED") + assert (client.params[1]["attempt_number"], client.params[1]["event_type"]) == (2, "SUCCESS") + + +def test_repeated_callback_is_idempotent_merge_not_insert() -> None: + client = FakeClient() + for _ in range(3): + upsert_task_event(_record(event_type="FAILED", status="FAILED"), SETTINGS, client=client) + assert all("WHEN MATCHED THEN" in sql for sql in client.queries) + keys = {(p["pipeline_run_id"], p["task_id"], p["attempt_number"], p["event_type"]) for p in client.params} + assert len(keys) == 1 + + +def test_invalid_event_type_and_attempt_rejected() -> None: + client = FakeClient() + with pytest.raises(ValueError, match="Unsupported task event type"): + upsert_task_event(_record(event_type="EXPLODED"), SETTINGS, client=client) + with pytest.raises(ValueError, match="attempt_number"): + upsert_task_event(_record(attempt_number=0), SETTINGS, client=client) + assert client.queries == [] + + +def test_error_message_is_sanitized() -> None: + client = FakeClient() + upsert_task_event( + _record( + event_type="FAILED", + status="FAILED", + # Concatenated so the repo secret scanner never sees a contiguous PEM header. + error_message='failed: {"private_key": "' + "-----BEGIN " + 'PRIVATE KEY-----xyz"}', + ), + SETTINGS, + client=client, + ) + assert "BEGIN PRIVATE KEY" not in client.params[0]["error_message"] + assert "[REDACTED]" in client.params[0]["error_message"] + + +def test_missing_audit_backend_returns_false_and_emits_fallback(capsys) -> None: + ok = record_task_event_safely(_record(), SETTINGS, client=BrokenClient()) + assert ok is False + out = capsys.readouterr().out + assert "task_telemetry_write_failed" in out + assert "BigQuery unavailable" in out + + +def test_skipped_and_upstream_failed_event_types_allowed() -> None: + client = FakeClient() + upsert_task_event(_record(event_type="SKIPPED", status="SKIPPED"), SETTINGS, client=client) + upsert_task_event( + _record(event_type="UPSTREAM_FAILED", status="UPSTREAM_FAILED"), SETTINGS, client=client + ) + assert len(client.queries) == 2 + + +def _events(*rows: tuple[str, str]) -> list[dict[str, Any]]: + return [{"task_id": t, "event_type": e} for t, e in rows] + + +def test_telemetry_completeness_all_terminal() -> None: + rows = _events(*[(t, "SUCCESS") for t in EXPECTED_TERMINAL_TASKS]) + client = FakeClient(select_rows=rows) + report = telemetry_completeness("pr-1", SETTINGS, client=client) + assert report["complete"] is True + assert report["missing_terminal"] == [] + + +def test_telemetry_completeness_detects_missing_and_started_only() -> None: + rows = _events( + ("resolve_run_context", "SUCCESS"), + ("generate_events", "STARTED"), + ) + client = FakeClient(select_rows=rows) + report = telemetry_completeness("pr-1", SETTINGS, client=client) + assert report["complete"] is False + assert "generate_events" in report["missing_terminal"] + assert report["started_without_terminal"] == ["generate_events"] + + +def test_telemetry_completeness_counts_failed_as_terminal() -> None: + rows = _events(*[(t, "FAILED" if t == "dbt_build" else "SUCCESS") for t in EXPECTED_TERMINAL_TASKS]) + client = FakeClient(select_rows=rows) + report = telemetry_completeness("pr-1", SETTINGS, client=client) + assert report["complete"] is True diff --git a/tests/unit/test_upload.py b/tests/unit/test_upload.py new file mode 100644 index 0000000..2e5824a --- /dev/null +++ b/tests/unit/test_upload.py @@ -0,0 +1,41 @@ +"""Unit tests for GCS object naming and upload idempotency.""" + +from __future__ import annotations + +from pathlib import Path +from unittest.mock import MagicMock + +from atlas.config.settings import load_settings +from atlas.ingestion.upload import build_object_name, upload_events_file + + +def test_build_object_name_supports_run_id_and_batch_id() -> None: + settings = load_settings() + run_path = build_object_name(settings, "2026-07-14", "atlas-run-123") + batch_path = build_object_name(settings, "2026-07-14", "atlas-20260714", use_batch_id=True) + assert run_path == "raw/event_date=2026-07-14/run_id=atlas-run-123/events.jsonl" + assert batch_path == "raw/event_date=2026-07-14/batch_id=atlas-20260714/events.jsonl" + + +def test_upload_skips_existing_object(tmp_path: Path) -> None: + settings = load_settings() + local_path = tmp_path / "events.jsonl" + local_path.write_text('{"event_id":"1"}\n', encoding="utf-8") + + blob = MagicMock() + blob.exists.return_value = True + blob.size = 42 + bucket = MagicMock() + bucket.blob.return_value = blob + client = MagicMock() + client.bucket.return_value = bucket + + result = upload_events_file( + settings, + local_path, + "2026-07-14", + "atlas-run-123", + client=client, + ) + assert result.already_exists is True + blob.upload_from_filename.assert_not_called() diff --git a/tests/unit/test_validation.py b/tests/unit/test_validation.py new file mode 100644 index 0000000..abfde20 --- /dev/null +++ b/tests/unit/test_validation.py @@ -0,0 +1,41 @@ +"""Unit tests for validation reporting.""" + +from __future__ import annotations + +from atlas.config.settings import load_settings +from atlas.validation.checks import ValidationCheck, ValidationReport, validate_anomaly_detection + + +def test_validate_anomaly_detection_requires_exact_seeded_counts() -> None: + settings = load_settings() + base_checks = [ + ValidationCheck("duplicates", "FAIL", 50, 50, ""), + ValidationCheck("null_user_ids", "FAIL", 500, 500, ""), + ValidationCheck("invalid_country_codes", "FAIL", 200, 200, ""), + ValidationCheck("future_timestamps", "FAIL", 150, 150, ""), + ValidationCheck("late_arriving_events", "FAIL", 300, 300, ""), + ] + report = ValidationReport("run-1", "FAIL", base_checks) + enriched = validate_anomaly_detection(report, settings) + acceptance = [check for check in enriched.checks if check.name.startswith("acceptance_")] + assert len(acceptance) == 5 + assert all(check.status == "PASS" for check in acceptance) + + +def test_validate_anomaly_detection_rejects_inflated_future_timestamp_count() -> None: + settings = load_settings() + base_checks = [ + ValidationCheck("duplicates", "FAIL", 50, 50, ""), + ValidationCheck("null_user_ids", "FAIL", 500, 500, ""), + ValidationCheck("invalid_country_codes", "FAIL", 200, 200, ""), + ValidationCheck("future_timestamps", "FAIL", 150, 15028, ""), + ValidationCheck("late_arriving_events", "FAIL", 300, 300, ""), + ] + report = ValidationReport("run-1", "FAIL", base_checks) + enriched = validate_anomaly_detection(report, settings) + future_acceptance = next( + check for check in enriched.checks if check.name == "acceptance_future_timestamp_detection" + ) + assert future_acceptance.status == "FAIL" + assert future_acceptance.expected == 150 + assert future_acceptance.actual == 15028 diff --git a/tests/unit/test_warehouse_validation.py b/tests/unit/test_warehouse_validation.py new file mode 100644 index 0000000..f145c02 --- /dev/null +++ b/tests/unit/test_warehouse_validation.py @@ -0,0 +1,131 @@ +"""Unit tests for batch-scoped warehouse validation (Sprint 4 Phase 1).""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from atlas.config.settings import load_settings +from atlas.validation.warehouse import validate_warehouse, warehouse_table + + +class FakeRow: + def __init__(self, value: Any) -> None: + self._value = value + + def values(self) -> list[Any]: + return [self._value] + + +class FakeJob: + def __init__(self, value: Any) -> None: + self._value = value + + def result(self) -> list[FakeRow]: + return [FakeRow(self._value)] + + +class FakeClient: + """Returns scripted scalar answers keyed by an ordered list.""" + + def __init__(self, answers: list[Any]) -> None: + self._answers = list(answers) + self.queries: list[str] = [] + + def query(self, sql: str, job_config: Any = None) -> FakeJob: + self.queries.append(sql) + return FakeJob(self._answers.pop(0)) + + +RAW = 50_000 +ACCEPTED = 48_800 +REJECTED = 1_200 + + +def _happy_answers() -> list[Any]: + # Order matches the check sequence in validate_warehouse. + return [ + RAW, # raw count (batch_nonempty + raw_equals_classification) + RAW, # classification count + ACCEPTED, # accepted count + REJECTED, # rejected count + ACCEPTED, # fact join count + 0, # duplicate fact ids + 0, # orphan users + 0, # orphan countries + 123_456, # mart total + 123_456, # fact total + 0, # bad processing dates + 0, # null lineage + ] + + +def test_validate_warehouse_passes_when_all_reconcile() -> None: + client = FakeClient(_happy_answers()) + report = validate_warehouse("atlas-20260718", "2026-07-18", load_settings(), client=client) + assert report.overall_status == "PASS" + assert {c.name for c in report.checks} == { + "batch_nonempty", + "raw_equals_classification", + "accepted_plus_rejected_equals_raw", + "accepted_equals_fact", + "fact_event_ids_unique", + "fact_user_fk_resolves", + "fact_country_fk_resolves", + "mart_totals_reconcile", + "processing_date_semantics", + "batch_lineage_semantics", + } + + +def test_validate_warehouse_fails_on_missing_batch() -> None: + answers = _happy_answers() + answers[0] = 0 + answers[1] = 0 + client = FakeClient(answers) + report = validate_warehouse("atlas-19990101", "1999-01-01", load_settings(), client=client) + assert report.overall_status == "FAIL" + failed = {c.name for c in report.checks if c.status == "FAIL"} + assert "batch_nonempty" in failed + + +@pytest.mark.parametrize( + ("index", "bad_value", "expected_failed_check"), + [ + (1, RAW - 10, "raw_equals_classification"), + (3, REJECTED + 1, "accepted_plus_rejected_equals_raw"), + (4, ACCEPTED - 5, "accepted_equals_fact"), + (5, 3, "fact_event_ids_unique"), + (6, 2, "fact_user_fk_resolves"), + (7, 1, "fact_country_fk_resolves"), + (9, 999, "mart_totals_reconcile"), + (10, 42, "processing_date_semantics"), + (11, 7, "batch_lineage_semantics"), + ], +) +def test_validate_warehouse_fails_each_reconciliation( + index: int, bad_value: Any, expected_failed_check: str +) -> None: + answers = _happy_answers() + answers[index] = bad_value + client = FakeClient(answers) + report = validate_warehouse("atlas-20260718", "2026-07-18", load_settings(), client=client) + assert report.overall_status == "FAIL" + failed = {c.name for c in report.checks if c.status == "FAIL"} + assert expected_failed_check in failed + + +def test_validate_warehouse_never_trivially_passes_regression() -> None: + """Regression: the Sprint 3 step printed a hardcoded PASS for any input.""" + answers = [0] * 8 + [0, 0, 0, 0] + client = FakeClient(answers) + report = validate_warehouse("atlas-empty", "2026-01-01", load_settings(), client=client) + assert report.overall_status == "FAIL" + + +def test_warehouse_table_uses_dbt_dataset_prefix(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("ATLAS_DBT_DATASET", "atlas_ci_123") + assert warehouse_table("proj", "core", "fct_events") == "proj.atlas_ci_123_core.fct_events" + monkeypatch.delenv("ATLAS_DBT_DATASET") + assert warehouse_table("proj", "marts", "m") == "proj.atlas_marts.m"