From 1c12865c63d8197e836e74fe31c88b4322410e18 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 13:13:42 +0900 Subject: [PATCH 01/26] feat(sre-lab): provision the lab with azd Add azure.yaml and main.parameters.json so azd owns provisioning of the SRE Agent event lab. Convert infra/main.bicep to a subscription-scoped azd entry point (derives suffix, resource group, and merged azd tags) and move the former resource-group module to infra/lab.bicep, which now always deploys the Container App and alert rules with a placeholder image. Remove the now-redundant infra/subscription.bicep(param) in favor of the azd entry point. Add scripts/azd-configure.sh (preprovision: provider registration and env defaults) and scripts/azd-postprovision.sh (az acr build + update the existing Container App to the immutable image, then health-check it), keeping the lab Docker-free. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/azure.yaml | 21 ++++ monitor/sre-agent-event-lab/infra/lab.bicep | 69 ++++++++++++++ .../infra/{main.bicepparam => lab.bicepparam} | 2 +- monitor/sre-agent-event-lab/infra/main.bicep | 92 ++++++++---------- .../infra/main.parameters.json | 10 ++ .../infra/subscription.bicep | 56 ----------- .../infra/subscription.bicepparam | 12 --- .../infra/tests/test_alerts_bicep.py | 14 +++ .../infra/tests/test_azd_project.py | 37 ++++++++ .../scripts/azd-configure.sh | 27 ++++++ .../scripts/azd-postprovision.sh | 95 +++++++++++++++++++ 11 files changed, 316 insertions(+), 119 deletions(-) create mode 100644 monitor/sre-agent-event-lab/azure.yaml create mode 100644 monitor/sre-agent-event-lab/infra/lab.bicep rename monitor/sre-agent-event-lab/infra/{main.bicepparam => lab.bicepparam} (93%) create mode 100644 monitor/sre-agent-event-lab/infra/main.parameters.json delete mode 100644 monitor/sre-agent-event-lab/infra/subscription.bicep delete mode 100644 monitor/sre-agent-event-lab/infra/subscription.bicepparam create mode 100644 monitor/sre-agent-event-lab/infra/tests/test_azd_project.py create mode 100755 monitor/sre-agent-event-lab/scripts/azd-configure.sh create mode 100755 monitor/sre-agent-event-lab/scripts/azd-postprovision.sh diff --git a/monitor/sre-agent-event-lab/azure.yaml b/monitor/sre-agent-event-lab/azure.yaml new file mode 100644 index 0000000..254977f --- /dev/null +++ b/monitor/sre-agent-event-lab/azure.yaml @@ -0,0 +1,21 @@ +name: sre-agent-event-lab +infra: + provider: bicep + path: infra + module: main +hooks: + preprovision: + shell: sh + run: ./scripts/azd-configure.sh + interactive: true + continueOnError: false + postprovision: + shell: sh + run: ./scripts/azd-postprovision.sh + interactive: true + continueOnError: false + predown: + shell: sh + run: ./scripts/cleanup-external.sh --yes + interactive: true + continueOnError: false diff --git a/monitor/sre-agent-event-lab/infra/lab.bicep b/monitor/sre-agent-event-lab/infra/lab.bicep new file mode 100644 index 0000000..c1fb6f3 --- /dev/null +++ b/monitor/sre-agent-event-lab/infra/lab.bicep @@ -0,0 +1,69 @@ +targetScope = 'resourceGroup' + +@description('Azure region for all regional lab resources.') +param location string = resourceGroup().location + +@description('Stable alphanumeric suffix used for globally unique names.') +@minLength(6) +@maxLength(12) +param suffix string + +@description('Container image deployed after the ACR build completes.') +param containerImage string + +@description('Whether to deploy the Container App and alert rules.') +param deployContainerApp bool = false + +@description('Optional Azure Monitor Action Group resource ID for event-driven SRE invocation.') +param actionGroupResourceId string = '' + +@description('Tags applied to all resources that support tags.') +param tags object + +module observability 'observability.bicep' = { + name: 'sre-lab-observability' + params: { + location: location + suffix: suffix + tags: tags + } +} + +module workload 'workload.bicep' = { + name: 'sre-lab-workload' + params: { + location: location + suffix: suffix + containerImage: containerImage + deployContainerApp: deployContainerApp + workspaceCustomerId: observability.outputs.workspaceCustomerId + workspaceSharedKey: observability.outputs.workspaceSharedKey + appInsightsConnectionString: observability.outputs.appInsightsConnectionString + tags: tags + } +} + +module alerts 'alerts.bicep' = if (deployContainerApp) { + name: 'sre-lab-alerts' + params: { + location: location + appInsightsResourceId: observability.outputs.appInsightsResourceId + actionGroupResourceId: actionGroupResourceId + serviceName: workload.outputs.telemetryServiceName + tags: tags + } +} + +output acrName string = workload.outputs.acrName +output acrLoginServer string = workload.outputs.acrLoginServer +output containerAppName string = workload.outputs.containerAppName +output containerAppFqdn string = workload.outputs.containerAppFqdn +output containerAppPrincipalId string = workload.outputs.workloadPrincipalId +output storageContainerScope string = workload.outputs.storageContainerScope +output blobRoleAssignmentName string = workload.outputs.blobRoleAssignmentName +output workspaceId string = observability.outputs.workspaceId +output workspaceCustomerId string = observability.outputs.workspaceCustomerId +output appInsightsName string = observability.outputs.appInsightsName +output appInsightsResourceId string = observability.outputs.appInsightsResourceId +output alertRuleNames array = deployContainerApp ? alerts!.outputs.alertRuleNames : [] +output telemetryServiceName string = workload.outputs.telemetryServiceName diff --git a/monitor/sre-agent-event-lab/infra/main.bicepparam b/monitor/sre-agent-event-lab/infra/lab.bicepparam similarity index 93% rename from monitor/sre-agent-event-lab/infra/main.bicepparam rename to monitor/sre-agent-event-lab/infra/lab.bicepparam index 219272c..5fd6fb5 100644 --- a/monitor/sre-agent-event-lab/infra/main.bicepparam +++ b/monitor/sre-agent-event-lab/infra/lab.bicepparam @@ -1,4 +1,4 @@ -using './main.bicep' +using './lab.bicep' param location = 'koreacentral' param suffix = '95933ae5' diff --git a/monitor/sre-agent-event-lab/infra/main.bicep b/monitor/sre-agent-event-lab/infra/main.bicep index c1fb6f3..4c3f21f 100644 --- a/monitor/sre-agent-event-lab/infra/main.bicep +++ b/monitor/sre-agent-event-lab/infra/main.bicep @@ -1,69 +1,61 @@ -targetScope = 'resourceGroup' +targetScope = 'subscription' -@description('Azure region for all regional lab resources.') -param location string = resourceGroup().location +@description('Name of the azd environment. Used to derive the resource group and a stable resource suffix.') +param environmentName string -@description('Stable alphanumeric suffix used for globally unique names.') -@minLength(6) -@maxLength(12) -param suffix string +@description('Azure region for the resource group and all regional lab resources.') +param location string -@description('Container image deployed after the ACR build completes.') -param containerImage string +@description('Dedicated resource group for the disposable SRE lab. Defaults to rg- when not set.') +param resourceGroupName string = 'rg-${environmentName}' -@description('Whether to deploy the Container App and alert rules.') -param deployContainerApp bool = false +@description('Container image deployed by the initial azd provision. postprovision replaces this with the ACR-built immutable image.') +param containerImage string = 'mcr.microsoft.com/azuredocs/containerapps-helloworld:latest' @description('Optional Azure Monitor Action Group resource ID for event-driven SRE invocation.') param actionGroupResourceId string = '' -@description('Tags applied to all resources that support tags.') -param tags object +@description('Base tags applied to the resource group and lab resources.') +param tags object = {} -module observability 'observability.bicep' = { - name: 'sre-lab-observability' - params: { - location: location - suffix: suffix - tags: tags - } +@description('Optional ISO-8601 date after which the lab resources are considered expired.') +param expiresOn string = '' + +// Truncated to 8 characters to satisfy lab.bicep's @minLength(6)/@maxLength(12) suffix constraint. +var suffix = substring(uniqueString(subscription().id, environmentName), 0, 8) + +var requiredTags = union(tags, { + purpose: 'sre-agent-event-lab' + 'azd-env-name': environmentName +}, empty(expiresOn) ? {} : { + expiresOn: expiresOn +}) + +resource labResourceGroup 'Microsoft.Resources/resourceGroups@2024-03-01' = { + name: resourceGroupName + location: location + tags: requiredTags } -module workload 'workload.bicep' = { - name: 'sre-lab-workload' +module lab 'lab.bicep' = { + name: 'sre-agent-event-lab' + scope: labResourceGroup params: { location: location suffix: suffix containerImage: containerImage - deployContainerApp: deployContainerApp - workspaceCustomerId: observability.outputs.workspaceCustomerId - workspaceSharedKey: observability.outputs.workspaceSharedKey - appInsightsConnectionString: observability.outputs.appInsightsConnectionString - tags: tags - } -} - -module alerts 'alerts.bicep' = if (deployContainerApp) { - name: 'sre-lab-alerts' - params: { - location: location - appInsightsResourceId: observability.outputs.appInsightsResourceId + deployContainerApp: true actionGroupResourceId: actionGroupResourceId - serviceName: workload.outputs.telemetryServiceName - tags: tags + tags: requiredTags } } -output acrName string = workload.outputs.acrName -output acrLoginServer string = workload.outputs.acrLoginServer -output containerAppName string = workload.outputs.containerAppName -output containerAppFqdn string = workload.outputs.containerAppFqdn -output containerAppPrincipalId string = workload.outputs.workloadPrincipalId -output storageContainerScope string = workload.outputs.storageContainerScope -output blobRoleAssignmentName string = workload.outputs.blobRoleAssignmentName -output workspaceId string = observability.outputs.workspaceId -output workspaceCustomerId string = observability.outputs.workspaceCustomerId -output appInsightsName string = observability.outputs.appInsightsName -output appInsightsResourceId string = observability.outputs.appInsightsResourceId -output alertRuleNames array = deployContainerApp ? alerts!.outputs.alertRuleNames : [] -output telemetryServiceName string = workload.outputs.telemetryServiceName +output AZURE_RESOURCE_GROUP string = labResourceGroup.name +output AZURE_ACR_NAME string = lab.outputs.acrName +output AZURE_CONTAINER_APP_NAME string = lab.outputs.containerAppName +output AZURE_CONTAINER_APP_FQDN string = lab.outputs.containerAppFqdn +output AZURE_WORKSPACE_ID string = lab.outputs.workspaceId +output AZURE_APP_INSIGHTS_NAME string = lab.outputs.appInsightsName +output AZURE_STORAGE_CONTAINER_SCOPE string = lab.outputs.storageContainerScope +output AZURE_BLOB_ROLE_ASSIGNMENT_NAME string = lab.outputs.blobRoleAssignmentName +output AZURE_TELEMETRY_SERVICE_NAME string = lab.outputs.telemetryServiceName diff --git a/monitor/sre-agent-event-lab/infra/main.parameters.json b/monitor/sre-agent-event-lab/infra/main.parameters.json new file mode 100644 index 0000000..f2f7976 --- /dev/null +++ b/monitor/sre-agent-event-lab/infra/main.parameters.json @@ -0,0 +1,10 @@ +{ + "$schema": "https://schema.management.azure.com/schemas/2019-04-01/deploymentParameters.json#", + "contentVersion": "1.0.0.0", + "parameters": { + "environmentName": { "value": "${AZURE_ENV_NAME}" }, + "location": { "value": "${AZURE_LOCATION}" }, + "resourceGroupName": { "value": "${AZURE_RESOURCE_GROUP}" }, + "expiresOn": { "value": "${SRE_LAB_EXPIRES_ON}" } + } +} diff --git a/monitor/sre-agent-event-lab/infra/subscription.bicep b/monitor/sre-agent-event-lab/infra/subscription.bicep deleted file mode 100644 index a4d0971..0000000 --- a/monitor/sre-agent-event-lab/infra/subscription.bicep +++ /dev/null @@ -1,56 +0,0 @@ -targetScope = 'subscription' - -@description('Azure region for the resource group and all regional lab resources.') -param location string - -@description('Dedicated resource group for the disposable SRE lab.') -param resourceGroupName string - -@description('Stable alphanumeric suffix used for globally unique names.') -param suffix string - -@description('Container image deployed after the ACR build completes.') -param containerImage string - -@description('Whether to deploy the Container App and alert rules.') -param deployContainerApp bool = false - -@description('Optional Azure Monitor Action Group resource ID for event-driven SRE invocation.') -param actionGroupResourceId string = '' - -@description('Tags applied to the resource group and lab resources.') -param tags object - -resource labResourceGroup 'Microsoft.Resources/resourceGroups@2024-03-01' = { - name: resourceGroupName - location: location - tags: tags -} - -module lab 'main.bicep' = { - name: 'sre-agent-event-lab' - scope: labResourceGroup - params: { - location: location - suffix: suffix - containerImage: containerImage - deployContainerApp: deployContainerApp - actionGroupResourceId: actionGroupResourceId - tags: tags - } -} - -output resourceGroupName string = labResourceGroup.name -output acrName string = lab.outputs.acrName -output acrLoginServer string = lab.outputs.acrLoginServer -output containerAppName string = lab.outputs.containerAppName -output containerAppFqdn string = lab.outputs.containerAppFqdn -output containerAppPrincipalId string = lab.outputs.containerAppPrincipalId -output storageContainerScope string = lab.outputs.storageContainerScope -output blobRoleAssignmentName string = lab.outputs.blobRoleAssignmentName -output workspaceId string = lab.outputs.workspaceId -output workspaceCustomerId string = lab.outputs.workspaceCustomerId -output appInsightsName string = lab.outputs.appInsightsName -output appInsightsResourceId string = lab.outputs.appInsightsResourceId -output alertRuleNames array = lab.outputs.alertRuleNames -output telemetryServiceName string = lab.outputs.telemetryServiceName diff --git a/monitor/sre-agent-event-lab/infra/subscription.bicepparam b/monitor/sre-agent-event-lab/infra/subscription.bicepparam deleted file mode 100644 index e3e077b..0000000 --- a/monitor/sre-agent-event-lab/infra/subscription.bicepparam +++ /dev/null @@ -1,12 +0,0 @@ -using './subscription.bicep' - -param location = 'koreacentral' -param resourceGroupName = 'rg-sre-agent-event-lab-krc' -param suffix = '95933ae5' -param containerImage = 'mcr.microsoft.com/azuredocs/containerapps-helloworld:latest' -param deployContainerApp = false -param actionGroupResourceId = '' -param tags = { - purpose: 'sre-agent-event-lab' - expiresOn: '2026-08-13' -} diff --git a/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py b/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py index 1b63cfa..eaff22e 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py @@ -41,3 +41,17 @@ def test_http500_alert_isolated_to_orders_and_exact_500(): assert 'param serviceName string' in template assert 'cloud_RoleName == "{0}"' in template assert "''', serviceName)" in template + + +def test_alert_rules_require_and_pass_through_caller_tags(): + """azure.yaml's azd main.bicep now centralizes tag construction + (purpose, azd-env-name, expiresOn) and passes the merged object down + through lab.bicep. alerts.bicep must keep accepting an arbitrary, + required tags object and applying it verbatim to each alert rule + rather than defaulting or hardcoding its own tags. + """ + template = ALERTS_BICEP.read_text() + + assert "param tags object\n" in template + assert "tags: tags" in template + diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py new file mode 100644 index 0000000..0791fbb --- /dev/null +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -0,0 +1,37 @@ +from pathlib import Path + + +LAB_ROOT = Path(__file__).parents[2] + + +def test_azure_yaml_runs_remote_build_after_provision(): + config = (LAB_ROOT / "azure.yaml").read_text() + assert "postprovision" in config + assert "./scripts/azd-postprovision.sh" in config + assert "predown" in config + assert "./scripts/cleanup-external.sh" in config + + +def test_subscription_template_uses_azd_environment_parameters(): + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + assert "targetScope = 'subscription'" in template + assert "param environmentName string" in template + assert "param resourceGroupName string = 'rg-${environmentName}'" in template + assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in template + assert "2026-08-13" not in template + + +def test_azd_outputs_have_stable_names(): + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + for name in ( + "AZURE_RESOURCE_GROUP", + "AZURE_ACR_NAME", + "AZURE_CONTAINER_APP_NAME", + "AZURE_CONTAINER_APP_FQDN", + "AZURE_WORKSPACE_ID", + "AZURE_APP_INSIGHTS_NAME", + "AZURE_STORAGE_CONTAINER_SCOPE", + "AZURE_BLOB_ROLE_ASSIGNMENT_NAME", + "AZURE_TELEMETRY_SERVICE_NAME", + ): + assert f"output {name} " in template diff --git a/monitor/sre-agent-event-lab/scripts/azd-configure.sh b/monitor/sre-agent-event-lab/scripts/azd-configure.sh new file mode 100755 index 0000000..15733fb --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/azd-configure.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash +set -euo pipefail + +for command_name in az azd jq curl python3; do + command -v "${command_name}" >/dev/null 2>&1 || { + echo "Required command not found: ${command_name}" >&2 + exit 1 + } +done + +az account show --output none +azd auth login --check-status + +for provider in Microsoft.App Microsoft.OperationalInsights Microsoft.Insights \ + Microsoft.Storage Microsoft.ContainerRegistry Microsoft.ManagedIdentity \ + Microsoft.Network; do + az provider register --namespace "${provider}" --wait +done + +if [[ -z "$(azd env get-value AZURE_RESOURCE_GROUP 2>/dev/null || true)" ]]; then + azd env set AZURE_RESOURCE_GROUP "rg-$(azd env get-value AZURE_ENV_NAME)" +fi + +if [[ -z "$(azd env get-value SRE_LAB_EXPIRES_ON 2>/dev/null || true)" ]]; then + expires_on="$(python3 -c 'from datetime import date,timedelta; print(date.today()+timedelta(days=1))')" + azd env set SRE_LAB_EXPIRES_ON "${expires_on}" +fi diff --git a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh new file mode 100755 index 0000000..1574290 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh @@ -0,0 +1,95 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +readonly SCRIPT_DIR +LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +readonly LAB_ROOT +readonly APP_DIR="${LAB_ROOT}/app" + +for command_name in az azd jq curl; do + command -v "${command_name}" >/dev/null 2>&1 || { + echo "Required command not found: ${command_name}" >&2 + exit 1 + } +done + +: "${AZURE_RESOURCE_GROUP:?AZURE_RESOURCE_GROUP must be set by azd before running this hook}" +: "${AZURE_ACR_NAME:?AZURE_ACR_NAME must be set by azd before running this hook}" +: "${AZURE_CONTAINER_APP_NAME:?AZURE_CONTAINER_APP_NAME must be set by azd before running this hook}" +: "${AZURE_CONTAINER_APP_FQDN:?AZURE_CONTAINER_APP_FQDN must be set by azd before running this hook}" + +IMAGE_TAG="run-$(date -u +%Y%m%dT%H%M%SZ)" +readonly IMAGE_TAG + +az acr build \ + --registry "${AZURE_ACR_NAME}" \ + --image "sre-event-lab:${IMAGE_TAG}" \ + "${APP_DIR}" + +ACR_LOGIN_SERVER="$(az acr show \ + --name "${AZURE_ACR_NAME}" \ + --query loginServer \ + -o tsv)" +readonly ACR_LOGIN_SERVER +readonly CONTAINER_IMAGE="${ACR_LOGIN_SERVER}/sre-event-lab:${IMAGE_TAG}" + +PREVIOUS_REVISION="$(az containerapp show \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --query properties.latestRevisionName \ + -o tsv)" +readonly PREVIOUS_REVISION + +az containerapp update \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --image "${CONTAINER_IMAGE}" \ + --output none + +wait_for_new_revision_ready() { + local timeout_seconds="${1:-600}" + local started="${SECONDS}" + + while (( SECONDS - started < timeout_seconds )); do + local latest_revision + latest_revision="$(az containerapp show \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --query properties.latestRevisionName \ + -o tsv)" + if [[ -n "${latest_revision}" && "${latest_revision}" != "${PREVIOUS_REVISION}" ]]; then + local health active + health="$(az containerapp revision list \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --query "[?name=='${latest_revision}'].properties.healthState | [0]" \ + -o tsv 2>/dev/null || true)" + active="$(az containerapp revision list \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --query "[?name=='${latest_revision}'].properties.active | [0]" \ + -o tsv 2>/dev/null || true)" + if [[ "${health}" == "Healthy" && "${active}" == "true" ]]; then + return 0 + fi + fi + sleep 10 + done + + echo "A new healthy revision did not become active within ${timeout_seconds}s." >&2 + return 1 +} + +wait_for_new_revision_ready 600 + +started="${SECONDS}" +until curl --fail --silent --show-error "https://${AZURE_CONTAINER_APP_FQDN}/healthz" >/dev/null; do + if (( SECONDS - started >= 600 )); then + echo "Health endpoint did not return HTTP 200 within 600s." >&2 + exit 1 + fi + sleep 10 +done + +azd env set SRE_IMAGE_TAG "${IMAGE_TAG}" From a1cceb6858871410d2d9fd9a6f0fa5e55adf7d6b Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 13:42:23 +0900 Subject: [PATCH 02/26] fix(sre-lab): repair azd hook, image, and subscription defects Address the Task 1 review findings on the azd migration: - Add scripts/cleanup-external.sh so every azure.yaml hook points at an existing executable script. It removes only the recorded subscription-scoped role assignments and exits 0 when the Agent setup evidence is absent, so azd down never fails and never deletes broadly. - Parameterize the Container App port and probes. The first provision runs the public placeholder on port 80 without /healthz probes, then postprovision moves ingress to 8000, swaps in the ACR-built image, waits for a healthy revision, and verifies /healthz. Probes now always match the port they check. - Persist the built image in SRE_CONTAINER_IMAGE and bind it (plus ACTION_GROUP_RESOURCE_ID) in main.parameters.json, so a later provision keeps the lab image instead of reverting to the placeholder. - Turn deploy.sh into a compatibility wrapper around azd up, refresh the README deployment/teardown commands, and update the tests, so no tracked file or documented command references the deleted subscription-scope templates. - Pin every Azure CLI call in the azd hooks to AZURE_SUBSCRIPTION_ID and report a mismatched active account. Resource-group and app operations read the current azd values only. - Delete infra/lab.bicepparam, which pinned a dead suffix and expiry. - Restore the deployment outputs the lab scripts still read and ignore the project-local .azure/ directory. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/.gitignore | 2 + monitor/sre-agent-event-lab/README.md | 35 +-- monitor/sre-agent-event-lab/infra/lab.bicep | 8 + .../sre-agent-event-lab/infra/lab.bicepparam | 11 - monitor/sre-agent-event-lab/infra/main.bicep | 41 ++- .../infra/main.parameters.json | 4 +- .../infra/tests/test_azd_project.py | 123 ++++++++ .../sre-agent-event-lab/infra/workload.bicep | 16 +- .../scripts/azd-configure.sh | 17 +- .../scripts/azd-postprovision.sh | 34 ++- .../scripts/cleanup-external.sh | 109 ++++++++ monitor/sre-agent-event-lab/scripts/deploy.sh | 97 ++----- .../scripts/tests/test_azd_hooks.py | 263 ++++++++++++++++++ .../scripts/tests/test_common.py | 48 +++- 14 files changed, 679 insertions(+), 129 deletions(-) create mode 100644 monitor/sre-agent-event-lab/.gitignore delete mode 100644 monitor/sre-agent-event-lab/infra/lab.bicepparam create mode 100755 monitor/sre-agent-event-lab/scripts/cleanup-external.sh create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py diff --git a/monitor/sre-agent-event-lab/.gitignore b/monitor/sre-agent-event-lab/.gitignore new file mode 100644 index 0000000..f1cba58 --- /dev/null +++ b/monitor/sre-agent-event-lab/.gitignore @@ -0,0 +1,2 @@ +# azd stores per-developer environment state (including subscription IDs) here. +.azure/ diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 9957ebd..0b0cd96 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -56,26 +56,22 @@ az bicep build --file monitor/sre-agent-event-lab/infra/main.bicep --stdout >/de ## Azure 배포 -필수 provider를 등록한다. +배포는 `azd`가 담당한다. `azure.yaml`의 preprovision hook이 필수 provider를 등록하므로 별도 등록 명령은 필요 없다. ```bash -az account set --subscription 95933ae5-0201-4a21-a1fc-8051a7437982 -for provider in Microsoft.App Microsoft.OperationalInsights Microsoft.Insights \ - Microsoft.Storage Microsoft.ContainerRegistry Microsoft.ManagedIdentity; do - az provider register --namespace "$provider" --wait -done +cd monitor/sre-agent-event-lab +azd env new sre-event-lab \ + --subscription 95933ae5-0201-4a21-a1fc-8051a7437982 \ + --location koreacentral +mkdir -p evidence +azd up 2>&1 | tee evidence/deploy.log ``` -배포는 base infrastructure → ACR cloud build → Container App/alert 순서로 진행된다. 로컬 Docker는 필요하지 않다. - -```bash -monitor/sre-agent-event-lab/scripts/deploy.sh \ - 2>&1 | tee monitor/sre-agent-event-lab/evidence/deploy.log -``` +`azd up`은 Bicep provision → ACR cloud build → Container App image 전환 순서로 진행된다. 로컬 Docker는 필요하지 않다. 초기 provision은 public placeholder image를 port 80으로 띄우고, postprovision hook이 ingress를 8000으로 옮긴 뒤 lab image로 교체한다. `scripts/deploy.sh`는 위 `azd up`을 호출하는 호환 wrapper로 남아 있다. 성공 조건: -1. 두 Bicep deployment가 성공한다. +1. Bicep provision이 성공한다. 2. ACR에 `sre-event-lab:run-20260812T094446Z` 형식의 실행별 immutable image tag가 존재한다. 3. active Container App revision이 `Healthy`다. 4. `/healthz`가 HTTP 200을 반환한다. @@ -161,9 +157,7 @@ monitor/sre-agent-event-lab/app/.venv/bin/python \ 배포 output에서 FQDN을 확인하고 정상 요청을 만든다. ```bash -FQDN=$(az deployment sub show \ - -n sre-agent-event-lab-private \ - --query properties.outputs.containerAppFqdn.value -o tsv) +FQDN=$(azd env get-value AZURE_CONTAINER_APP_FQDN --cwd monitor/sre-agent-event-lab) python3 monitor/sre-agent-event-lab/scripts/loadgen.py \ "https://${FQDN}/api/orders" \ @@ -328,7 +322,14 @@ Dynamic rule은 3일·30 samples 전에는 발화하지 않으며 3주 전에는 ## 정리 -첫 명령은 dry-run이며 두 번째 명령만 삭제를 시작한다. +`azd`로 배포한 환경은 `azd down`으로 정리한다. predown hook(`scripts/cleanup-external.sh`)이 resource group 밖에 기록된 구독 범위 Monitoring Contributor assignment만 먼저 제거하고, resource group 삭제는 `azd`가 수행한다. + +```bash +cd monitor/sre-agent-event-lab +azd down --purge +``` + +`scripts/cleanup.sh`는 azd 이전 방식으로 만든 `rg-sre-agent-event-lab-krc`를 정리하는 기존 스크립트다. 첫 명령은 dry-run이며 두 번째 명령만 삭제를 시작한다. ```bash monitor/sre-agent-event-lab/scripts/cleanup.sh diff --git a/monitor/sre-agent-event-lab/infra/lab.bicep b/monitor/sre-agent-event-lab/infra/lab.bicep index c1fb6f3..b6f112b 100644 --- a/monitor/sre-agent-event-lab/infra/lab.bicep +++ b/monitor/sre-agent-event-lab/infra/lab.bicep @@ -14,6 +14,12 @@ param containerImage string @description('Whether to deploy the Container App and alert rules.') param deployContainerApp bool = false +@description('Port the deployed container image listens on.') +param containerTargetPort int = 8000 + +@description('Whether to attach /healthz probes to the Container App.') +param enableHealthProbes bool = true + @description('Optional Azure Monitor Action Group resource ID for event-driven SRE invocation.') param actionGroupResourceId string = '' @@ -36,6 +42,8 @@ module workload 'workload.bicep' = { suffix: suffix containerImage: containerImage deployContainerApp: deployContainerApp + containerTargetPort: containerTargetPort + enableHealthProbes: enableHealthProbes workspaceCustomerId: observability.outputs.workspaceCustomerId workspaceSharedKey: observability.outputs.workspaceSharedKey appInsightsConnectionString: observability.outputs.appInsightsConnectionString diff --git a/monitor/sre-agent-event-lab/infra/lab.bicepparam b/monitor/sre-agent-event-lab/infra/lab.bicepparam deleted file mode 100644 index 5fd6fb5..0000000 --- a/monitor/sre-agent-event-lab/infra/lab.bicepparam +++ /dev/null @@ -1,11 +0,0 @@ -using './lab.bicep' - -param location = 'koreacentral' -param suffix = '95933ae5' -param containerImage = 'mcr.microsoft.com/azuredocs/containerapps-helloworld:latest' -param deployContainerApp = false -param actionGroupResourceId = '' -param tags = { - purpose: 'sre-agent-event-lab' - expiresOn: '2026-08-13' -} diff --git a/monitor/sre-agent-event-lab/infra/main.bicep b/monitor/sre-agent-event-lab/infra/main.bicep index 4c3f21f..48cc66f 100644 --- a/monitor/sre-agent-event-lab/infra/main.bicep +++ b/monitor/sre-agent-event-lab/infra/main.bicep @@ -9,8 +9,8 @@ param location string @description('Dedicated resource group for the disposable SRE lab. Defaults to rg- when not set.') param resourceGroupName string = 'rg-${environmentName}' -@description('Container image deployed by the initial azd provision. postprovision replaces this with the ACR-built immutable image.') -param containerImage string = 'mcr.microsoft.com/azuredocs/containerapps-helloworld:latest' +@description('Container image deployed by azd. Leave empty for the first provision: the public placeholder image is used until the postprovision hook builds the lab image and records it in SRE_CONTAINER_IMAGE.') +param containerImage string = '' @description('Optional Azure Monitor Action Group resource ID for event-driven SRE invocation.') param actionGroupResourceId string = '' @@ -24,6 +24,21 @@ param expiresOn string = '' // Truncated to 8 characters to satisfy lab.bicep's @minLength(6)/@maxLength(12) suffix constraint. var suffix = substring(uniqueString(subscription().id, environmentName), 0, 8) +// The public placeholder serves port 80 and has no /healthz, so the first +// provision must expose port 80 without probes. Once postprovision records +// the ACR-built image in SRE_CONTAINER_IMAGE, every later provision deploys +// that image on port 8000 with matching /healthz probes instead of reverting +// to the placeholder. +var placeholderContainerImage = 'mcr.microsoft.com/azuredocs/containerapps-helloworld:latest' +var effectiveContainerImage = empty(containerImage) ? placeholderContainerImage : containerImage +var usesPlaceholderImage = effectiveContainerImage == placeholderContainerImage + +// azd substitutes an unset ${AZURE_RESOURCE_GROUP} with an empty string and +// passes it through, because this parameter's default is a non-empty +// expression. Fall back here so provisioning never asks for a resource group +// with an empty name. +var effectiveResourceGroupName = empty(resourceGroupName) ? 'rg-${environmentName}' : resourceGroupName + var requiredTags = union(tags, { purpose: 'sre-agent-event-lab' 'azd-env-name': environmentName @@ -32,7 +47,7 @@ var requiredTags = union(tags, { }) resource labResourceGroup 'Microsoft.Resources/resourceGroups@2024-03-01' = { - name: resourceGroupName + name: effectiveResourceGroupName location: location tags: requiredTags } @@ -43,8 +58,10 @@ module lab 'lab.bicep' = { params: { location: location suffix: suffix - containerImage: containerImage + containerImage: effectiveContainerImage deployContainerApp: true + containerTargetPort: usesPlaceholderImage ? 80 : 8000 + enableHealthProbes: !usesPlaceholderImage actionGroupResourceId: actionGroupResourceId tags: requiredTags } @@ -59,3 +76,19 @@ output AZURE_APP_INSIGHTS_NAME string = lab.outputs.appInsightsName output AZURE_STORAGE_CONTAINER_SCOPE string = lab.outputs.storageContainerScope output AZURE_BLOB_ROLE_ASSIGNMENT_NAME string = lab.outputs.blobRoleAssignmentName output AZURE_TELEMETRY_SERVICE_NAME string = lab.outputs.telemetryServiceName + +// Deployment outputs the lab scripts (common.sh `deployment_output`, +// run-scenario.sh, query-evidence.sh) still read by their original names. +output acrName string = lab.outputs.acrName +output acrLoginServer string = lab.outputs.acrLoginServer +output containerAppName string = lab.outputs.containerAppName +output containerAppFqdn string = lab.outputs.containerAppFqdn +output containerAppPrincipalId string = lab.outputs.containerAppPrincipalId +output storageContainerScope string = lab.outputs.storageContainerScope +output blobRoleAssignmentName string = lab.outputs.blobRoleAssignmentName +output workspaceId string = lab.outputs.workspaceId +output workspaceCustomerId string = lab.outputs.workspaceCustomerId +output appInsightsName string = lab.outputs.appInsightsName +output appInsightsResourceId string = lab.outputs.appInsightsResourceId +output alertRuleNames array = lab.outputs.alertRuleNames +output telemetryServiceName string = lab.outputs.telemetryServiceName diff --git a/monitor/sre-agent-event-lab/infra/main.parameters.json b/monitor/sre-agent-event-lab/infra/main.parameters.json index f2f7976..43d3dee 100644 --- a/monitor/sre-agent-event-lab/infra/main.parameters.json +++ b/monitor/sre-agent-event-lab/infra/main.parameters.json @@ -5,6 +5,8 @@ "environmentName": { "value": "${AZURE_ENV_NAME}" }, "location": { "value": "${AZURE_LOCATION}" }, "resourceGroupName": { "value": "${AZURE_RESOURCE_GROUP}" }, - "expiresOn": { "value": "${SRE_LAB_EXPIRES_ON}" } + "expiresOn": { "value": "${SRE_LAB_EXPIRES_ON}" }, + "actionGroupResourceId": { "value": "${ACTION_GROUP_RESOURCE_ID}" }, + "containerImage": { "value": "${SRE_CONTAINER_IMAGE}" } } } diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index 0791fbb..e2623d6 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -1,7 +1,17 @@ +import json +import os +import re from pathlib import Path LAB_ROOT = Path(__file__).parents[2] +PLACEHOLDER_IMAGE = "mcr.microsoft.com/azuredocs/containerapps-helloworld:latest" + + +def _hook_commands(): + """Every `run:` command declared in azure.yaml, hook name unknown.""" + config = (LAB_ROOT / "azure.yaml").read_text() + return re.findall(r"^\s*run:\s*(\S+)", config, flags=re.MULTILINE) def test_azure_yaml_runs_remote_build_after_provision(): @@ -35,3 +45,116 @@ def test_azd_outputs_have_stable_names(): "AZURE_TELEMETRY_SERVICE_NAME", ): assert f"output {name} " in template + + +def test_every_azure_yaml_hook_references_an_existing_executable_script(): + """azd aborts the whole command when a hook script cannot be executed. + + `predown` pointed at scripts/cleanup-external.sh, which did not exist, + so `azd down` failed before it could delete anything. + """ + commands = _hook_commands() + + assert commands, "azure.yaml must declare at least one hook command" + for command in commands: + script = LAB_ROOT / command + assert script.is_file(), f"azure.yaml hook references a missing script: {command}" + assert os.access(script, os.X_OK), ( + f"azure.yaml hook script is not executable: {command}" + ) + + +def test_workload_binds_ingress_port_and_probes_to_parameters(): + """The initial provision runs the public placeholder image, which + listens on port 80 and serves no /healthz. Hardcoding targetPort 8000 + with /healthz probes makes that first Bicep deployment fail, so both + must be parameterized and always agree with each other. + """ + template = (LAB_ROOT / "infra" / "workload.bicep").read_text() + + assert "param containerTargetPort int" in template + assert "param enableHealthProbes bool" in template + assert "targetPort: containerTargetPort" in template + assert "port: containerTargetPort" in template + assert "enableHealthProbes ?" in template + assert "targetPort: 8000" not in template + assert "port: 8000" not in template + + +def test_lab_bicep_forwards_container_port_and_probe_switches(): + template = (LAB_ROOT / "infra" / "lab.bicep").read_text() + + assert "param containerTargetPort int" in template + assert "param enableHealthProbes bool" in template + assert "containerTargetPort: containerTargetPort" in template + assert "enableHealthProbes: enableHealthProbes" in template + + +def test_main_bicep_keeps_placeholder_port_and_probes_consistent(): + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + + assert PLACEHOLDER_IMAGE in template + assert re.search(r"containerTargetPort:\s*\S+\s*\?\s*80\s*:\s*8000", template), ( + "main.bicep must expose the placeholder on port 80 and the lab image on 8000" + ) + assert re.search(r"enableHealthProbes:\s*!\S+", template), ( + "main.bicep must disable /healthz probes while the placeholder image runs" + ) + + +def test_main_bicep_restores_outputs_consumed_by_lab_scripts(): + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + + for name in ( + "containerAppName", + "containerAppFqdn", + "containerAppPrincipalId", + "workspaceCustomerId", + "acrLoginServer", + "appInsightsResourceId", + "alertRuleNames", + "storageContainerScope", + "blobRoleAssignmentName", + "telemetryServiceName", + ): + assert f"output {name} " in template, ( + f"scripts/common.sh consumers still read the {name} deployment output" + ) + + +def test_main_parameters_map_action_group_and_deployed_image(): + parameters = json.loads((LAB_ROOT / "infra" / "main.parameters.json").read_text())["parameters"] + + assert parameters["actionGroupResourceId"]["value"] == "${ACTION_GROUP_RESOURCE_ID}" + assert parameters["containerImage"]["value"] == "${SRE_CONTAINER_IMAGE}" + + +def test_hardcoded_lab_bicepparam_is_deleted(): + """lab.bicepparam pinned a dead suffix (95933ae5) and an expiry date; + azd owns those values now, so the file must not linger. + """ + assert not (LAB_ROOT / "infra" / "lab.bicepparam").exists() + assert not (LAB_ROOT / "infra" / "main.bicepparam").exists() + + +def test_lab_ignores_local_azd_environment_directory(): + ignore_file = LAB_ROOT / ".gitignore" + + assert ignore_file.is_file(), "the lab must ignore its own .azure/ azd state directory" + assert ".azure/" in ignore_file.read_text() + + +def test_main_bicep_tolerates_an_empty_resource_group_parameter(): + """azd substitutes an unset ${AZURE_RESOURCE_GROUP} with "" and, because + the Bicep default is a non-empty expression, passes that empty string + through (armParameterFileValue in azure-dev's bicep_provider.go). The + template must fall back on its own instead of creating a resource group + with an empty name. + """ + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + + assert "param resourceGroupName string = 'rg-${environmentName}'" in template + assert re.search( + r"var effectiveResourceGroupName = empty\(resourceGroupName\)", template + ) + assert "name: effectiveResourceGroupName" in template diff --git a/monitor/sre-agent-event-lab/infra/workload.bicep b/monitor/sre-agent-event-lab/infra/workload.bicep index 538dad6..0a1fc13 100644 --- a/monitor/sre-agent-event-lab/infra/workload.bicep +++ b/monitor/sre-agent-event-lab/infra/workload.bicep @@ -10,6 +10,12 @@ param containerImage string @description('Whether to deploy the Container App.') param deployContainerApp bool +@description('Port the deployed container image listens on. The public placeholder image serves port 80; the lab image serves 8000.') +param containerTargetPort int = 8000 + +@description('Whether to attach /healthz liveness and readiness probes. Disabled while the placeholder image runs, because it serves no /healthz.') +param enableHealthProbes bool = true + @description('Log Analytics workspace customer ID.') param workspaceCustomerId string @@ -255,7 +261,7 @@ resource containerApp 'Microsoft.App/containerApps@2024-03-01' = if (deployConta ingress: { allowInsecure: false external: true - targetPort: 8000 + targetPort: containerTargetPort traffic: [ { latestRevision: true @@ -309,12 +315,12 @@ resource containerApp 'Microsoft.App/containerApps@2024-03-01' = if (deployConta value: telemetryServiceName } ] - probes: [ + probes: enableHealthProbes ? [ { type: 'Liveness' httpGet: { path: '/healthz' - port: 8000 + port: containerTargetPort scheme: 'HTTP' } initialDelaySeconds: 10 @@ -324,13 +330,13 @@ resource containerApp 'Microsoft.App/containerApps@2024-03-01' = if (deployConta type: 'Readiness' httpGet: { path: '/healthz' - port: 8000 + port: containerTargetPort scheme: 'HTTP' } initialDelaySeconds: 5 periodSeconds: 5 } - ] + ] : [] resources: { cpu: json('0.5') memory: '1Gi' diff --git a/monitor/sre-agent-event-lab/scripts/azd-configure.sh b/monitor/sre-agent-event-lab/scripts/azd-configure.sh index 15733fb..8a257e5 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-configure.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-configure.sh @@ -8,13 +8,26 @@ for command_name in az azd jq curl python3; do } done -az account show --output none +: "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set by azd before running this hook}" + +# The Azure CLI's active account is whatever the operator selected last, which +# is not necessarily the subscription azd provisions into. Report the mismatch +# and pin every operation below to the azd subscription explicitly. +ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv)" +readonly ACTIVE_SUBSCRIPTION_ID +if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${AZURE_SUBSCRIPTION_ID}" ]]; then + echo "Azure CLI is signed in to ${ACTIVE_SUBSCRIPTION_ID}." >&2 + echo "Every lab operation is pinned to ${AZURE_SUBSCRIPTION_ID} instead." >&2 +fi + +az account show --subscription "${AZURE_SUBSCRIPTION_ID}" --output none azd auth login --check-status for provider in Microsoft.App Microsoft.OperationalInsights Microsoft.Insights \ Microsoft.Storage Microsoft.ContainerRegistry Microsoft.ManagedIdentity \ Microsoft.Network; do - az provider register --namespace "${provider}" --wait + az provider register --namespace "${provider}" --wait \ + --subscription "${AZURE_SUBSCRIPTION_ID}" done if [[ -z "$(azd env get-value AZURE_RESOURCE_GROUP 2>/dev/null || true)" ]]; then diff --git a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh index 1574290..edb6d95 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh @@ -6,29 +6,44 @@ readonly SCRIPT_DIR LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" readonly LAB_ROOT readonly APP_DIR="${LAB_ROOT}/app" +# The lab image serves HTTP on 8000; the placeholder image used by the first +# provision serves 80, so ingress has to move with the image. +readonly APP_TARGET_PORT=8000 -for command_name in az azd jq curl; do +for command_name in az azd curl; do command -v "${command_name}" >/dev/null 2>&1 || { echo "Required command not found: ${command_name}" >&2 exit 1 } done +# Every value below comes from the current azd environment, which azd refreshes +# from the deployment outputs before running this hook. +: "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set by azd before running this hook}" : "${AZURE_RESOURCE_GROUP:?AZURE_RESOURCE_GROUP must be set by azd before running this hook}" : "${AZURE_ACR_NAME:?AZURE_ACR_NAME must be set by azd before running this hook}" : "${AZURE_CONTAINER_APP_NAME:?AZURE_CONTAINER_APP_NAME must be set by azd before running this hook}" : "${AZURE_CONTAINER_APP_FQDN:?AZURE_CONTAINER_APP_FQDN must be set by azd before running this hook}" +ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv)" +readonly ACTIVE_SUBSCRIPTION_ID +if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${AZURE_SUBSCRIPTION_ID}" ]]; then + echo "Azure CLI is signed in to ${ACTIVE_SUBSCRIPTION_ID}." >&2 + echo "Every lab operation is pinned to ${AZURE_SUBSCRIPTION_ID} instead." >&2 +fi + IMAGE_TAG="run-$(date -u +%Y%m%dT%H%M%SZ)" readonly IMAGE_TAG az acr build \ --registry "${AZURE_ACR_NAME}" \ --image "sre-event-lab:${IMAGE_TAG}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ "${APP_DIR}" ACR_LOGIN_SERVER="$(az acr show \ --name "${AZURE_ACR_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ --query loginServer \ -o tsv)" readonly ACR_LOGIN_SERVER @@ -37,13 +52,24 @@ readonly CONTAINER_IMAGE="${ACR_LOGIN_SERVER}/sre-event-lab:${IMAGE_TAG}" PREVIOUS_REVISION="$(az containerapp show \ --resource-group "${AZURE_RESOURCE_GROUP}" \ --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ --query properties.latestRevisionName \ -o tsv)" readonly PREVIOUS_REVISION +# Ingress is app-level configuration and does not create a revision, so move it +# to the lab port before the new image starts serving. +az containerapp ingress update \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --target-port "${APP_TARGET_PORT}" \ + --output none + az containerapp update \ --resource-group "${AZURE_RESOURCE_GROUP}" \ --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ --image "${CONTAINER_IMAGE}" \ --output none @@ -56,6 +82,7 @@ wait_for_new_revision_ready() { latest_revision="$(az containerapp show \ --resource-group "${AZURE_RESOURCE_GROUP}" \ --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ --query properties.latestRevisionName \ -o tsv)" if [[ -n "${latest_revision}" && "${latest_revision}" != "${PREVIOUS_REVISION}" ]]; then @@ -63,11 +90,13 @@ wait_for_new_revision_ready() { health="$(az containerapp revision list \ --resource-group "${AZURE_RESOURCE_GROUP}" \ --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ --query "[?name=='${latest_revision}'].properties.healthState | [0]" \ -o tsv 2>/dev/null || true)" active="$(az containerapp revision list \ --resource-group "${AZURE_RESOURCE_GROUP}" \ --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ --query "[?name=='${latest_revision}'].properties.active | [0]" \ -o tsv 2>/dev/null || true)" if [[ "${health}" == "Healthy" && "${active}" == "true" ]]; then @@ -93,3 +122,6 @@ until curl --fail --silent --show-error "https://${AZURE_CONTAINER_APP_FQDN}/hea done azd env set SRE_IMAGE_TAG "${IMAGE_TAG}" +# Persisting the built image keeps a later `azd provision` on the lab image and +# its matching /healthz probes instead of reverting to the placeholder. +azd env set SRE_CONTAINER_IMAGE "${CONTAINER_IMAGE}" diff --git a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh new file mode 100755 index 0000000..b485b80 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh @@ -0,0 +1,109 @@ +#!/usr/bin/env bash +# Removes only the lab resources that live outside the azd-owned resource +# group, so `azd down` can delete everything else itself. +# +# Today that is the subscription-scoped Monitoring Contributor assignments +# recorded by the Azure SRE Agent setup. Nothing else is ever deleted here: no +# resource groups, no resources, no unrecorded role assignments. When the +# evidence file is missing the lab never configured the Agent, so the hook +# reports that and succeeds -- `azd down` must not fail because an optional +# step was skipped. +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +readonly SCRIPT_DIR +LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +readonly LAB_ROOT +EVIDENCE_ROOT="${SRE_LAB_EVIDENCE_ROOT:-${LAB_ROOT}/evidence}" +readonly EVIDENCE_ROOT +readonly AGENT_SETUP_FILE="${EVIDENCE_ROOT}/agent-setup.json" + +CONFIRMED=0 +case "${1:-}" in + "") ;; + --yes) CONFIRMED=1 ;; + *) + echo "Usage: $0 [--yes]" >&2 + exit 2 + ;; +esac + +if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then + echo "No Azure SRE Agent setup evidence at ${AGENT_SETUP_FILE}." + echo "Nothing outside the azd resource group to clean up." + exit 0 +fi + +for command_name in az jq; do + command -v "${command_name}" >/dev/null 2>&1 || { + echo "Required command not found: ${command_name}" >&2 + exit 1 + } +done + +: "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set to clean up recorded role assignments}" +readonly SUBSCRIPTION_SCOPE="/subscriptions/${AZURE_SUBSCRIPTION_ID}" + +lowercase() { + printf '%s' "$1" | tr '[:upper:]' '[:lower:]' +} + +# Guards against a hand-edited evidence file pointing cleanup at a role +# assignment in another subscription or at a different resource type. +validate_recorded_assignment() { + local assignment_id="$1" + local assignment_id_lower expected_prefix + + assignment_id_lower="$(lowercase "${assignment_id}")" + expected_prefix="$(lowercase "${SUBSCRIPTION_SCOPE}")/providers/microsoft.authorization/roleassignments/" + + if [[ "${assignment_id_lower}" != "${expected_prefix}"* ]]; then + echo "Refusing role assignment outside ${SUBSCRIPTION_SCOPE}: ${assignment_id}" >&2 + return 1 + fi +} + +RECORDED_ASSIGNMENT_IDS="" +while IFS= read -r assignment_id; do + [[ -n "${assignment_id}" ]] || continue + validate_recorded_assignment "${assignment_id}" + case "${RECORDED_ASSIGNMENT_IDS}" in + *"${assignment_id}"$'\n'*) continue ;; + esac + RECORDED_ASSIGNMENT_IDS="${RECORDED_ASSIGNMENT_IDS}${assignment_id}"$'\n' +done < <(jq -r ' + [ + .monitoring_contributor_assignment_id, + .uami_monitoring_contributor_assignment_id + ] + | map(select(. != null and . != "")) + | .[] +' "${AGENT_SETUP_FILE}") + +if [[ -z "${RECORDED_ASSIGNMENT_IDS}" ]]; then + echo "Agent setup evidence records no subscription role assignment." + exit 0 +fi + +echo "Planned external cleanup in ${SUBSCRIPTION_SCOPE}:" +while IFS= read -r assignment_id; do + [[ -n "${assignment_id}" ]] || continue + echo " Remove recorded role assignment: ${assignment_id}" +done <<<"${RECORDED_ASSIGNMENT_IDS}" + +if [[ "${CONFIRMED}" -ne 1 ]]; then + echo "Dry run only. Re-run with --yes to execute." + exit 0 +fi + +while IFS= read -r assignment_id; do + [[ -n "${assignment_id}" ]] || continue + if ! az role assignment delete \ + --ids "${assignment_id}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --output none; then + echo "Could not remove ${assignment_id}; remove it manually." >&2 + fi +done <<<"${RECORDED_ASSIGNMENT_IDS}" + +echo "External cleanup complete." diff --git a/monitor/sre-agent-event-lab/scripts/deploy.sh b/monitor/sre-agent-event-lab/scripts/deploy.sh index b2d9cf6..dd42e3a 100755 --- a/monitor/sre-agent-event-lab/scripts/deploy.sh +++ b/monitor/sre-agent-event-lab/scripts/deploy.sh @@ -1,85 +1,20 @@ #!/usr/bin/env bash +# Compatibility wrapper. The lab is an azd project now: azure.yaml owns the +# Bicep entry point, the preprovision/postprovision hooks register providers, +# build the image in ACR, and move the Container App onto it. This script only +# forwards to `azd up` so the previously documented command keeps working. set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" -source "${SCRIPT_DIR}/common.sh" - -readonly TEMPLATE_FILE="${LAB_ROOT}/infra/subscription.bicep" -readonly PARAMETER_FILE="${LAB_ROOT}/infra/subscription.bicepparam" -readonly APP_DIR="${LAB_ROOT}/app" -IMAGE_TAG="${SRE_IMAGE_TAG:-run-$(date -u +%Y%m%dT%H%M%SZ)}" -readonly IMAGE_TAG - -require_commands -verify_subscription -mkdir -p "${EVIDENCE_ROOT}" - -if resource_group_exists; then - verify_lab_resource_group -fi - -az deployment sub validate \ - --location "${LOCATION}" \ - --template-file "${TEMPLATE_FILE}" \ - --parameters "${PARAMETER_FILE}" \ - --parameters deployContainerApp=false \ - --output none - -az deployment sub create \ - --location "${LOCATION}" \ - --name "sre-agent-event-lab-base" \ - --template-file "${TEMPLATE_FILE}" \ - --parameters "${PARAMETER_FILE}" \ - --parameters deployContainerApp=false \ - --output none - -ACR_NAME="$(az deployment sub show \ - --name "sre-agent-event-lab-base" \ - --query "properties.outputs.acrName.value" \ - -o tsv)" -ACR_LOGIN_SERVER="$(az deployment sub show \ - --name "sre-agent-event-lab-base" \ - --query "properties.outputs.acrLoginServer.value" \ - -o tsv)" -readonly ACR_NAME ACR_LOGIN_SERVER -readonly CONTAINER_IMAGE="${ACR_LOGIN_SERVER}/sre-event-lab:${IMAGE_TAG}" - -az acr build \ - --registry "${ACR_NAME}" \ - --image "sre-event-lab:${IMAGE_TAG}" \ - "${APP_DIR}" - -az deployment sub validate \ - --location "${LOCATION}" \ - --template-file "${TEMPLATE_FILE}" \ - --parameters "${PARAMETER_FILE}" \ - --parameters deployContainerApp=true containerImage="${CONTAINER_IMAGE}" \ - --output none - -az deployment sub create \ - --location "${LOCATION}" \ - --name "${FINAL_DEPLOYMENT_NAME}" \ - --template-file "${TEMPLATE_FILE}" \ - --parameters "${PARAMETER_FILE}" \ - --parameters deployContainerApp=true containerImage="${CONTAINER_IMAGE}" \ - --output none - -APP_NAME="$(deployment_output containerAppName)" -APP_FQDN="$(deployment_output containerAppFqdn)" -readonly APP_NAME APP_FQDN - -wait_for_app_ready "${APP_NAME}" 600 - -started="${SECONDS}" -until curl --fail --silent --show-error "https://${APP_FQDN}/healthz" >/dev/null; do - if (( SECONDS - started >= 600 )); then - echo "Health endpoint did not return HTTP 200 within 600s." >&2 - exit 1 - fi - sleep 10 -done - -az deployment sub show \ - --name "${FINAL_DEPLOYMENT_NAME}" \ - --query properties.outputs \ - -o json +readonly SCRIPT_DIR +LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +readonly LAB_ROOT + +command -v azd >/dev/null 2>&1 || { + echo "Required command not found: azd (https://aka.ms/azd-install)" >&2 + exit 1 +} + +echo "deploy.sh now runs 'azd up' in ${LAB_ROOT}." >&2 +cd "${LAB_ROOT}" +exec azd up "$@" diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py new file mode 100644 index 0000000..9391e6f --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py @@ -0,0 +1,263 @@ +"""Behaviour tests for the azd lifecycle hook scripts. + +The hooks run inside `azd provision` / `azd down`, where the Azure CLI's +active subscription is whatever the operator last selected -- not +necessarily the subscription azd is deploying into. Every Azure CLI +operation therefore has to be pinned to AZURE_SUBSCRIPTION_ID, and the +`predown` hook has to survive a lab that never configured the Agent. +""" + +import json +import os +import re +import shutil +import stat +import subprocess +from pathlib import Path + + +SCRIPTS_DIR = Path(__file__).parents[1] +AZD_CONFIGURE = SCRIPTS_DIR / "azd-configure.sh" +AZD_POSTPROVISION = SCRIPTS_DIR / "azd-postprovision.sh" +CLEANUP_EXTERNAL = SCRIPTS_DIR / "cleanup-external.sh" +SUBSCRIPTION_PIN = '--subscription "${AZURE_SUBSCRIPTION_ID}"' +# The one deliberately unpinned call: it reads whichever account is active +# so the hook can report a mismatch. +ACTIVE_ACCOUNT_PROBE = "az account show --query id" + + +def _az_invocations(script_text): + """Every `az ...` command in a script, with line continuations joined.""" + joined = re.sub(r"\\\n\s*", " ", script_text) + commands = [] + for line in joined.splitlines(): + if line.strip().startswith("#"): + continue + for segment in re.split(r"\$\(|\|\||&&|\||;|`", line): + stripped = re.sub( + r"^(?:if\s+|until\s+|while\s+|then\s+|else\s+|do\s+|!\s*)+", + "", + segment.strip(), + ) + if re.match(r"^az\s", stripped): + commands.append(re.sub(r"\s+", " ", stripped).strip()) + return commands + + +def _write_az_stub(directory, log_path): + """A fake `az` on PATH that records its arguments.""" + stub = directory / "az" + stub.write_text( + "#!/usr/bin/env bash\n" + f'printf "%s\\n" "$*" >> "{log_path}"\n' + "exit 0\n" + ) + stub.chmod(stub.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + return stub + + +def _run_cleanup_external(tmp_path, args, evidence=None, environment=None): + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + log_path = tmp_path / "az-calls.log" + _write_az_stub(bin_dir, log_path) + + evidence_root = tmp_path / "evidence" + evidence_root.mkdir(exist_ok=True) + if evidence is not None: + (evidence_root / "agent-setup.json").write_text(json.dumps(evidence)) + + env = dict(os.environ) + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" + env["SRE_LAB_EVIDENCE_ROOT"] = str(evidence_root) + env.setdefault("AZURE_SUBSCRIPTION_ID", "11111111-2222-3333-4444-555555555555") + if environment: + env.update(environment) + + result = subprocess.run( + [str(CLEANUP_EXTERNAL), *args], + capture_output=True, + text=True, + env=env, + ) + calls = log_path.read_text() if log_path.exists() else "" + return result, calls + + +def test_azd_configure_pins_every_azure_cli_call_to_the_target_subscription(): + for command in _az_invocations(AZD_CONFIGURE.read_text()): + if command.startswith(ACTIVE_ACCOUNT_PROBE): + continue + assert SUBSCRIPTION_PIN in command, ( + f"azd-configure.sh runs an unpinned Azure CLI command: {command}" + ) + + +def test_azd_postprovision_pins_every_azure_cli_call_to_the_target_subscription(): + for command in _az_invocations(AZD_POSTPROVISION.read_text()): + if command.startswith(ACTIVE_ACCOUNT_PROBE): + continue + assert SUBSCRIPTION_PIN in command, ( + f"azd-postprovision.sh runs an unpinned Azure CLI command: {command}" + ) + + +def test_azd_hooks_require_the_target_subscription_and_verify_the_active_account(): + for script in (AZD_CONFIGURE, AZD_POSTPROVISION): + text = script.read_text() + assert "AZURE_SUBSCRIPTION_ID:?" in text, ( + f"{script.name} must fail fast when azd did not provide a subscription" + ) + assert ACTIVE_ACCOUNT_PROBE in text, ( + f"{script.name} must verify which subscription the Azure CLI is signed in to" + ) + + +def test_azd_postprovision_targets_only_current_azd_values(): + text = AZD_POSTPROVISION.read_text() + + assert "rg-sre-agent-event-lab-krc" not in text + assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in text + assert "common.sh" not in text + for value in ( + "AZURE_RESOURCE_GROUP:?", + "AZURE_ACR_NAME:?", + "AZURE_CONTAINER_APP_NAME:?", + "AZURE_CONTAINER_APP_FQDN:?", + ): + assert value in text + + +def test_azd_postprovision_moves_ingress_to_the_app_port_and_records_the_image(): + """The provisioned placeholder listens on port 80; the lab image listens + on 8000. postprovision must move ingress before verifying /healthz, and + persist the built image so a later `azd provision` does not revert the + Container App to the placeholder. + """ + text = AZD_POSTPROVISION.read_text() + + assert "az containerapp ingress update" in text + assert "--target-port" in text + assert "8000" in text + assert "azd env set SRE_CONTAINER_IMAGE" in text + assert "/healthz" in text + + ingress_at = text.index("az containerapp ingress update") + healthz_at = text.index("/healthz") + assert ingress_at < healthz_at + + +def test_cleanup_external_exists_and_is_executable(): + assert CLEANUP_EXTERNAL.is_file() + assert os.access(CLEANUP_EXTERNAL, os.X_OK) + + +def test_cleanup_external_never_deletes_broad_scopes(): + text = CLEANUP_EXTERNAL.read_text() + + assert "az group delete" not in text + assert "az resource delete" not in text + assert "--all" not in text + assert "az role assignment delete" in text + + +def test_cleanup_external_succeeds_when_agent_evidence_is_absent(tmp_path): + """`azd down` runs this hook with continueOnError: false, so a lab that + never configured the SRE Agent must still tear down cleanly. + """ + result, calls = _run_cleanup_external(tmp_path, ["--yes"]) + + assert result.returncode == 0, result.stderr + assert calls == "", f"nothing external exists, but the hook called: {calls}" + + +def test_cleanup_external_deletes_only_recorded_subscription_assignments(tmp_path): + subscription_id = "11111111-2222-3333-4444-555555555555" + recorded = ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" + ) + uami_recorded = ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + "/roleAssignments/ffffffff-1111-2222-3333-444444444444" + ) + result, calls = _run_cleanup_external( + tmp_path, + ["--yes"], + evidence={ + "monitoring_contributor_assignment_id": recorded, + "agent_principal_id": "principal-a", + "uami_monitoring_contributor_assignment_id": uami_recorded, + "agent_user_assigned_principal_id": "principal-b", + }, + environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, + ) + + assert result.returncode == 0, result.stderr + assert f"role assignment delete --ids {recorded}" in calls + assert f"role assignment delete --ids {uami_recorded}" in calls + assert f"--subscription {subscription_id}" in calls + assert "group delete" not in calls + + +def test_cleanup_external_refuses_assignments_outside_the_target_subscription(tmp_path): + subscription_id = "11111111-2222-3333-4444-555555555555" + foreign = ( + "/subscriptions/99999999-9999-9999-9999-999999999999/providers" + "/Microsoft.Authorization/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" + ) + result, calls = _run_cleanup_external( + tmp_path, + ["--yes"], + evidence={ + "monitoring_contributor_assignment_id": foreign, + "agent_principal_id": "principal-a", + "uami_monitoring_contributor_assignment_id": foreign, + "agent_user_assigned_principal_id": "principal-b", + }, + environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, + ) + + assert result.returncode != 0 + assert "role assignment delete" not in calls + + +def test_cleanup_external_is_a_dry_run_without_yes(tmp_path): + subscription_id = "11111111-2222-3333-4444-555555555555" + recorded = ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" + ) + result, calls = _run_cleanup_external( + tmp_path, + [], + evidence={ + "monitoring_contributor_assignment_id": recorded, + "agent_principal_id": "principal-a", + "uami_monitoring_contributor_assignment_id": recorded, + "agent_user_assigned_principal_id": "principal-b", + }, + environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, + ) + + assert result.returncode == 0, result.stderr + assert "delete" not in calls + + +def test_cleanup_external_runs_under_bash_32(tmp_path): + """macOS ships Bash 3.2, where `${ARRAY[@]}` on an empty array aborts + under `set -u`. + """ + bash_path = shutil.which("bash") or "/bin/bash" + version = subprocess.run( + [bash_path, "-c", "echo ${BASH_VERSINFO[0]}"], + capture_output=True, + text=True, + ).stdout.strip() + + result, _ = _run_cleanup_external(tmp_path, ["--yes"]) + + assert result.returncode == 0, ( + f"cleanup-external.sh must run on bash {version}: {result.stderr}" + ) + assert "unbound variable" not in result.stderr diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index 0f6a50e..fa53dbc 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -55,17 +55,51 @@ def test_verify_subscription_reports_only_subscription_id_on_mismatch(): assert "95933ae5-0201-4a21-a1fc-8051a7437982" in result.stderr -def test_deploy_uses_subscription_wrapper(): +def test_deploy_delegates_to_azd_up(): + """The subscription-scope templates deploy.sh used to deploy were removed + when the lab moved to azd, so deploy.sh must not reference them any more. + It stays as a thin compatibility wrapper so the documented command keeps + working. + """ script = DEPLOY_SH.read_text() - assert 'TEMPLATE_FILE="${LAB_ROOT}/infra/subscription.bicep"' in script - assert 'PARAMETER_FILE="${LAB_ROOT}/infra/subscription.bicepparam"' in script - assert "az deployment sub validate" in script - assert "az deployment sub create" in script + assert "azd up" in script + assert "subscription" + ".bicep" not in script + assert "az deployment sub validate" not in script + assert "az deployment sub create" not in script assert "az deployment group" not in script assert 'IMAGE_TAG="20260812.4"' not in script - assert "SRE_IMAGE_TAG" in script - assert "date -u +%Y%m%dT%H%M%SZ" in script + + +def test_no_tracked_lab_file_references_the_deleted_subscription_templates(): + lab_root = Path(__file__).parents[2] + tracked = subprocess.run( + ["git", "ls-files"], + cwd=lab_root, + capture_output=True, + text=True, + check=True, + ).stdout.split() + needle = "subscription" + ".bicep" + readable = {".sh", ".py", ".md", ".json", ".yaml", ".yml", ".bicep", ".bicepparam"} + + offenders = [] + for relative_path in tracked: + path = lab_root / relative_path + if path.suffix not in readable or not path.is_file(): + continue + if needle in path.read_text(encoding="utf-8", errors="ignore"): + offenders.append(relative_path) + + assert offenders == [], f"tracked files still reference deleted templates: {offenders}" + + +def test_readme_documents_a_working_deployment_command(): + readme = (Path(__file__).parents[2] / "README.md").read_text() + + assert "azd up" in readme + assert "az deployment sub show" not in readme + assert "azd env get-value AZURE_CONTAINER_APP_FQDN" in readme def test_scenario_waits_for_new_revision_before_load(): From d3474e2c97de98ecc37c007ffe1f490123bbb918 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 14:11:19 +0900 Subject: [PATCH 03/26] fix(sre-lab): remove hardcoded subscription, clear image env on down, guard no-login, mark legacy scripts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fixes the four remaining Task 1 review findings from review-f960e97..a1cceb6.diff, under strict TDD (RED tests written and confirmed failing first, then GREEN implementation): 1. README's `azd env new` command, prerequisites line, and the "실측 환경" table no longer hardcode the original validation subscription ID. The command now omits --subscription (azd prompts interactively) with guidance to pass --subscription instead. New regression test test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id scans README/azure.yaml/Bicep/parameters/hook scripts for the literal ID (common.sh and validation-results.md are intentionally excluded, as historical/legacy records already covered by their own tests). 2. cleanup-external.sh (the predown hook) now requires azd and, when run with --yes (as azure.yaml's predown hook always does), clears SRE_CONTAINER_IMAGE and SRE_IMAGE_TAG via `azd env set KEY ""` before/independent of the evidence-gated role-assignment cleanup, so reusing the same azd environment falls back to the placeholder image on the next provision. A dry run (no --yes) only prints the planned action. New tests exercise this via a stubbed azd binary and confirm no az group/resource delete calls occur as a result. 3. README now carries an explicit transitional caveat at the top of "## 시나리오 실행" stating run-scenario.sh/query-evidence.sh still read common.sh's legacy pre-azd deployment lookup and are legacy-only until common.sh is rewritten to use azd env get-value (a separate, not-yet-done task); points readers to the Baseline section's azd env get-value-based checks in the meantime. 4. azd-configure.sh and azd-postprovision.sh now guard their `az account show --query id -o tsv` call and exit with a clear "Azure CLI is not signed in. Run 'az login'..." message instead of leaking raw Azure CLI stderr when the CLI is signed out. Verified: azd env new --help / azd env set --help / a scratch azd env new + azd env set KEY "" run locally confirm the empty-value clearing semantics and that there is no `azd env unset`. Tests: 124 passed (infra/tests + scripts/tests, up from 118 passed/6 failed RED baseline), app suite 10 passed unchanged, bash -n clean, az bicep build clean for main/lab/workload.bicep, azure.yaml validated against the official azd JSON schema. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 20 +- .../infra/tests/test_azd_project.py | 38 ++++ .../scripts/azd-configure.sh | 9 +- .../scripts/azd-postprovision.sh | 8 +- .../scripts/cleanup-external.sh | 36 +++- .../scripts/tests/test_azd_hooks.py | 173 +++++++++++++++++- .../scripts/tests/test_common.py | 23 +++ 7 files changed, 287 insertions(+), 20 deletions(-) diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 0b0cd96..d6b72ed 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -19,7 +19,7 @@ Azure Container Apps에 의도적인 장애를 만들고, Azure Monitor 경고 ## 사전 조건 -- Azure CLI 로그인 및 구독 `95933ae5-0201-4a21-a1fc-8051a7437982` 접근 +- Azure CLI 로그인 및 대상 Azure 구독 접근 권한 - 구독 또는 필요한 리소스에 Contributor, 역할 할당에는 Owner/User Access Administrator - `az`, `jq`, `curl`, `python3` - 브라우저에서 `https://sre.azure.com` 및 `*.azuresre.ai` 접근 @@ -60,13 +60,13 @@ az bicep build --file monitor/sre-agent-event-lab/infra/main.bicep --stdout >/de ```bash cd monitor/sre-agent-event-lab -azd env new sre-event-lab \ - --subscription 95933ae5-0201-4a21-a1fc-8051a7437982 \ - --location koreacentral +azd env new sre-event-lab --location koreacentral mkdir -p evidence azd up 2>&1 | tee evidence/deploy.log ``` +`azd env new`에 `--subscription`을 지정하지 않으면 azd가 로그인된 계정의 구독 목록에서 대화형으로 선택하도록 안내한다. 특정 구독을 고정하려면 `--subscription `를 추가한다(하드코딩된 예시 구독 ID를 그대로 복사해 사용하지 않는다). + `azd up`은 Bicep provision → ACR cloud build → Container App image 전환 순서로 진행된다. 로컬 Docker는 필요하지 않다. 초기 provision은 public placeholder image를 port 80으로 띄우고, postprovision hook이 ingress를 8000으로 옮긴 뒤 lab image로 교체한다. `scripts/deploy.sh`는 위 `azd up`을 호출하는 호환 wrapper로 남아 있다. 성공 조건: @@ -82,8 +82,8 @@ azd up 2>&1 | tee evidence/deploy.log | 항목 | 값 | |---|---| -| Subscription | `95933ae5-0201-4a21-a1fc-8051a7437982` | -| Resource group | `rg-sre-agent-event-lab-krc` | +| Subscription | `azd env new`에서 선택한 구독. 최초 실측값은 `validation-results.md` 참고 | +| Resource group | `rg-sre-agent-event-lab-krc` (최초 실측 실행; `azd`가 provision한 environment는 `azd env get-value AZURE_RESOURCE_GROUP`으로 확인) | | Agent name | `sre-devguidesample-95933ae5` | | Region | Korea Central | | Azure resource access | 테스트 resource group, Reader | @@ -174,6 +174,14 @@ Application Insights의 `AppRequests`와 `AppDependencies`에 현재 데이터 ## 시나리오 실행 +> ⚠️ **전환기 주의사항 (레거시)**: `run-scenario.sh`와 `query-evidence.sh`는 아직 `scripts/common.sh`의 +> `deployment_output` 함수(구독 스코프 배포 조회, 고정된 `rg-sre-agent-event-lab-krc`)로 배포 output을 +> 조회한다. 즉 이 절의 명령은 `azd up`이 provision한 현재 azd 환경이 아니라 최초 실측 실행에서 만든 +> 리소스 그룹만을 대상으로 동작한다. 새로 `azd up`한 환경에서는 아직 동작을 보장하지 않으며, `common.sh`가 +> `azd env get-value`를 읽도록 재작성되는 후속 작업(가이드된 CLI 제공) 전까지는 아래 명령을 레거시 전용으로 +> 취급한다. 새 azd 환경에서 baseline을 확인하려면 위 [Baseline](#baseline) 절의 `azd env get-value` 기반 +> 명령을 사용한다. + 각 명령은 장애 주입, 제한 부하, alert polling, 복구, timeline 저장을 수행한다. ```bash diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index e2623d6..8b1e4c1 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -144,6 +144,44 @@ def test_lab_ignores_local_azd_environment_directory(): assert ".azure/" in ignore_file.read_text() +def test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id(): + """README's `azd env new` command hardcoded the one subscription ID used + for the original real validation run. Anyone following the README for a + *different* subscription would silently target someone else's + subscription, so no file that documents or drives the current `azd` + onboarding path may hardcode it. + + `scripts/common.sh` (the pre-azd legacy flow, still intentionally pinned + per `test_common_does_not_expose_personal_subscription_display_name`) and + `validation-results.md` (the historical record of that specific real run) + are deliberately out of scope here. + """ + fixed_subscription_id = "95933ae5-0201-4a21-a1fc-8051a7437982" + onboarding_paths = [ + LAB_ROOT / "README.md", + LAB_ROOT / "azure.yaml", + LAB_ROOT / "infra" / "main.bicep", + LAB_ROOT / "infra" / "lab.bicep", + LAB_ROOT / "infra" / "workload.bicep", + LAB_ROOT / "infra" / "main.parameters.json", + LAB_ROOT / "scripts" / "azd-configure.sh", + LAB_ROOT / "scripts" / "azd-postprovision.sh", + LAB_ROOT / "scripts" / "cleanup-external.sh", + LAB_ROOT / "scripts" / "deploy.sh", + ] + + offenders = [ + str(path.relative_to(LAB_ROOT)) + for path in onboarding_paths + if path.is_file() and fixed_subscription_id in path.read_text() + ] + + assert offenders == [], ( + "azd onboarding docs/config still hardcode the original validation " + f"subscription ID: {offenders}" + ) + + def test_main_bicep_tolerates_an_empty_resource_group_parameter(): """azd substitutes an unset ${AZURE_RESOURCE_GROUP} with "" and, because the Bicep default is a non-empty expression, passes that empty string diff --git a/monitor/sre-agent-event-lab/scripts/azd-configure.sh b/monitor/sre-agent-event-lab/scripts/azd-configure.sh index 8a257e5..cc9c611 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-configure.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-configure.sh @@ -13,7 +13,14 @@ done # The Azure CLI's active account is whatever the operator selected last, which # is not necessarily the subscription azd provisions into. Report the mismatch # and pin every operation below to the azd subscription explicitly. -ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv)" +# +# `az account show` fails with a raw Azure CLI error when no one is signed +# in; guard it so the hook fails fast with one clear, actionable message +# instead of that raw stderr or an unexplained `set -e` abort. +if ! ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv 2>/dev/null)"; then + echo "Azure CLI is not signed in. Run 'az login', then re-run this command." >&2 + exit 1 +fi readonly ACTIVE_SUBSCRIPTION_ID if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${AZURE_SUBSCRIPTION_ID}" ]]; then echo "Azure CLI is signed in to ${ACTIVE_SUBSCRIPTION_ID}." >&2 diff --git a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh index edb6d95..5cc4259 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh @@ -25,7 +25,13 @@ done : "${AZURE_CONTAINER_APP_NAME:?AZURE_CONTAINER_APP_NAME must be set by azd before running this hook}" : "${AZURE_CONTAINER_APP_FQDN:?AZURE_CONTAINER_APP_FQDN must be set by azd before running this hook}" -ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv)" +# `az account show` fails with a raw Azure CLI error when no one is signed +# in; guard it so the hook fails fast with one clear, actionable message +# instead of that raw stderr or an unexplained `set -e` abort. +if ! ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv 2>/dev/null)"; then + echo "Azure CLI is not signed in. Run 'az login', then re-run this command." >&2 + exit 1 +fi readonly ACTIVE_SUBSCRIPTION_ID if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${AZURE_SUBSCRIPTION_ID}" ]]; then echo "Azure CLI is signed in to ${ACTIVE_SUBSCRIPTION_ID}." >&2 diff --git a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh index b485b80..5ac0b80 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh @@ -1,13 +1,22 @@ #!/usr/bin/env bash # Removes only the lab resources that live outside the azd-owned resource -# group, so `azd down` can delete everything else itself. +# group, so `azd down` can delete everything else itself, and clears the +# azd environment values that `azd-postprovision.sh` set for this run. # -# Today that is the subscription-scoped Monitoring Contributor assignments -# recorded by the Azure SRE Agent setup. Nothing else is ever deleted here: no -# resource groups, no resources, no unrecorded role assignments. When the -# evidence file is missing the lab never configured the Agent, so the hook -# reports that and succeeds -- `azd down` must not fail because an optional -# step was skipped. +# Today the external resources are the subscription-scoped Monitoring +# Contributor assignments recorded by the Azure SRE Agent setup. Nothing +# else is ever deleted here: no resource groups, no resources, no +# unrecorded role assignments. When the evidence file is missing the lab +# never configured the Agent, so the hook reports that and succeeds -- +# `azd down` must not fail because an optional step was skipped. +# +# `azd down` may delete the resource group (and the ACR inside it) that +# `azd-postprovision.sh` recorded in SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG. If +# those azd environment values survive, reusing the same environment would +# make a later `azd provision` try to redeploy an immutable image tag that +# no longer exists instead of falling back to the placeholder image. So +# this hook always clears both values -- independent of whether the Agent +# was ever configured -- before doing anything else. set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" @@ -28,6 +37,19 @@ case "${1:-}" in ;; esac +command -v azd >/dev/null 2>&1 || { + echo "Required command not found: azd" >&2 + exit 1 +} + +if [[ "${CONFIRMED}" -eq 1 ]]; then + azd env set SRE_CONTAINER_IMAGE "" + azd env set SRE_IMAGE_TAG "" + echo "Cleared hook-set SRE_CONTAINER_IMAGE and SRE_IMAGE_TAG." +else + echo "Dry run: would clear hook-set SRE_CONTAINER_IMAGE and SRE_IMAGE_TAG." +fi + if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then echo "No Azure SRE Agent setup evidence at ${AGENT_SETUP_FILE}." echo "Nothing outside the azd resource group to clean up." diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py index 9391e6f..efc4955 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py @@ -56,11 +56,53 @@ def _write_az_stub(directory, log_path): return stub +def _write_azd_stub(directory, log_path): + """A fake `azd` on PATH that records its arguments. + + Without this, `command -v azd` on a developer machine finds the real + `azd` binary, which would then try to mutate an environment that does + not exist in the test's tmp_path and fail for unrelated reasons. + + Uses `%q` (not `$*`) so an empty-string argument -- e.g. clearing an + azd environment value with `azd env set KEY ""` -- is visible in the + log as `''` instead of silently vanishing. + """ + stub = directory / "azd" + stub.write_text( + "#!/usr/bin/env bash\n" + f'printf "%q " "$@" >> "{log_path}"\n' + f'printf "\\n" >> "{log_path}"\n' + "exit 0\n" + ) + stub.chmod(stub.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + return stub + + +def _write_login_failing_az_stub(directory, log_path): + """A fake `az` that fails only the login-check probe, like a signed-out + Azure CLI, while logging every invocation it receives. + """ + stub = directory / "az" + stub.write_text( + "#!/usr/bin/env bash\n" + f'printf "%s\\n" "$*" >> "{log_path}"\n' + 'if [[ "$1 $2" == "account show" && "$*" == *"--query id"* ]]; then\n' + " echo \"ERROR: Please run 'az login' to setup account.\" >&2\n" + " exit 1\n" + "fi\n" + "exit 0\n" + ) + stub.chmod(stub.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + return stub + + def _run_cleanup_external(tmp_path, args, evidence=None, environment=None): bin_dir = tmp_path / "bin" bin_dir.mkdir(exist_ok=True) log_path = tmp_path / "az-calls.log" + azd_log_path = tmp_path / "azd-calls.log" _write_az_stub(bin_dir, log_path) + _write_azd_stub(bin_dir, azd_log_path) evidence_root = tmp_path / "evidence" evidence_root.mkdir(exist_ok=True) @@ -81,6 +123,30 @@ def _run_cleanup_external(tmp_path, args, evidence=None, environment=None): env=env, ) calls = log_path.read_text() if log_path.exists() else "" + azd_calls = azd_log_path.read_text() if azd_log_path.exists() else "" + return result, calls, azd_calls + + +def _run_hook_script(script_path, tmp_path, az_stub_factory, environment=None): + """Execute an azd hook script with a controllable fake `az` on PATH.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + log_path = tmp_path / "az-calls.log" + az_stub_factory(bin_dir, log_path) + + env = dict(os.environ) + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" + env.setdefault("AZURE_SUBSCRIPTION_ID", "11111111-2222-3333-4444-555555555555") + if environment: + env.update(environment) + + result = subprocess.run( + [str(script_path)], + capture_output=True, + text=True, + env=env, + ) + calls = log_path.read_text() if log_path.exists() else "" return result, calls @@ -165,7 +231,7 @@ def test_cleanup_external_succeeds_when_agent_evidence_is_absent(tmp_path): """`azd down` runs this hook with continueOnError: false, so a lab that never configured the SRE Agent must still tear down cleanly. """ - result, calls = _run_cleanup_external(tmp_path, ["--yes"]) + result, calls, _ = _run_cleanup_external(tmp_path, ["--yes"]) assert result.returncode == 0, result.stderr assert calls == "", f"nothing external exists, but the hook called: {calls}" @@ -181,7 +247,7 @@ def test_cleanup_external_deletes_only_recorded_subscription_assignments(tmp_pat f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" "/roleAssignments/ffffffff-1111-2222-3333-444444444444" ) - result, calls = _run_cleanup_external( + result, calls, _ = _run_cleanup_external( tmp_path, ["--yes"], evidence={ @@ -206,7 +272,7 @@ def test_cleanup_external_refuses_assignments_outside_the_target_subscription(tm "/subscriptions/99999999-9999-9999-9999-999999999999/providers" "/Microsoft.Authorization/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" ) - result, calls = _run_cleanup_external( + result, calls, _ = _run_cleanup_external( tmp_path, ["--yes"], evidence={ @@ -228,7 +294,7 @@ def test_cleanup_external_is_a_dry_run_without_yes(tmp_path): f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" ) - result, calls = _run_cleanup_external( + result, calls, _ = _run_cleanup_external( tmp_path, [], evidence={ @@ -255,9 +321,106 @@ def test_cleanup_external_runs_under_bash_32(tmp_path): text=True, ).stdout.strip() - result, _ = _run_cleanup_external(tmp_path, ["--yes"]) + result, _, _ = _run_cleanup_external(tmp_path, ["--yes"]) assert result.returncode == 0, ( f"cleanup-external.sh must run on bash {version}: {result.stderr}" ) assert "unbound variable" not in result.stderr + + +def test_cleanup_external_clears_hook_set_image_env_vars_alongside_role_cleanup(tmp_path): + """`azd down` may delete the resource group (and its ACR) that + `azd-postprovision.sh` recorded in SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG. If + those values survive in the azd environment, reusing it later would make + `azd provision` try to redeploy an image tag that no longer exists + instead of falling back to the placeholder. `predown` always runs this + hook with --yes (see azure.yaml), so clearing here is safe. + """ + subscription_id = "11111111-2222-3333-4444-555555555555" + recorded = ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" + ) + result, calls, azd_calls = _run_cleanup_external( + tmp_path, + ["--yes"], + evidence={ + "monitoring_contributor_assignment_id": recorded, + "agent_principal_id": "principal-a", + "uami_monitoring_contributor_assignment_id": recorded, + "agent_user_assigned_principal_id": "principal-b", + }, + environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, + ) + + assert result.returncode == 0, result.stderr + assert f"role assignment delete --ids {recorded}" in calls + assert "env set SRE_CONTAINER_IMAGE ''" in azd_calls + assert "env set SRE_IMAGE_TAG ''" in azd_calls + assert "group delete" not in calls + assert "resource delete" not in calls + + +def test_cleanup_external_clears_image_env_vars_even_without_agent_evidence(tmp_path): + """A lab that never configured the SRE Agent takes the early-exit path + (no evidence file); the hook-set image values must still be cleared on + that path, since it has nothing to do with the Agent. + """ + result, calls, azd_calls = _run_cleanup_external(tmp_path, ["--yes"]) + + assert result.returncode == 0, result.stderr + assert calls == "", f"nothing external exists, but the hook called: {calls}" + assert "env set SRE_CONTAINER_IMAGE ''" in azd_calls + assert "env set SRE_IMAGE_TAG ''" in azd_calls + + +def test_cleanup_external_does_not_clear_image_env_vars_during_a_dry_run(tmp_path): + """Without --yes, the hook only plans actions; it must not mutate the + azd environment either. + """ + result, _, azd_calls = _run_cleanup_external(tmp_path, []) + + assert result.returncode == 0, result.stderr + assert azd_calls == "", ( + f"a dry run (no --yes) must not mutate the azd environment: {azd_calls}" + ) + + +def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): + """`az account show` fails with a generic Azure CLI error when signed + out. Guard it so the hook fails fast with one unambiguous message + instead of raw CLI stderr or an unexplained `set -e` abort. + """ + result, calls = _run_hook_script( + AZD_CONFIGURE, tmp_path, _write_login_failing_az_stub + ) + + assert result.returncode != 0 + assert "az login" in result.stderr + assert "Please run 'az login' to setup account." not in result.stderr + assert calls.strip() == "account show --query id -o tsv", ( + "the hook must exit immediately after the failed login check, " + f"before any other az call: {calls!r}" + ) + + +def test_azd_postprovision_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): + environment = { + "AZURE_RESOURCE_GROUP": "rg-test", + "AZURE_ACR_NAME": "acrtest", + "AZURE_CONTAINER_APP_NAME": "ca-test", + "AZURE_CONTAINER_APP_FQDN": "ca-test.example.com", + } + result, calls = _run_hook_script( + AZD_POSTPROVISION, tmp_path, _write_login_failing_az_stub, environment + ) + + assert result.returncode != 0 + assert "az login" in result.stderr + assert "Please run 'az login' to setup account." not in result.stderr + assert calls.strip() == "account show --query id -o tsv", ( + "the hook must exit immediately after the failed login check, " + f"before any other az call: {calls!r}" + ) + diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index fa53dbc..a69a6d2 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -102,6 +102,29 @@ def test_readme_documents_a_working_deployment_command(): assert "azd env get-value AZURE_CONTAINER_APP_FQDN" in readme +def test_readme_marks_legacy_scenario_scripts_as_transitional(): + """`run-scenario.sh` and `query-evidence.sh` still resolve deployment + outputs through `common.sh`'s `deployment_output` (`az deployment sub + show --name sre-agent-event-lab-private` against the hardcoded + `rg-sre-agent-event-lab-krc`), not through the `azd`-provisioned + environment introduced by this task. The README must not present them as + working against a fresh `azd up` environment without saying so. + """ + readme = (Path(__file__).parents[2] / "README.md").read_text() + + scenario_heading = "## 시나리오 실행" + assert scenario_heading in readme + section = readme.split(scenario_heading, 1)[1] + + assert "run-scenario.sh" in section.split("##", 1)[0] + caveat_markers = ("common.sh", "레거시", "azd 환경") + assert any(marker in section.split("##", 1)[0] for marker in caveat_markers), ( + "README's scenario-execution section must caveat that " + "run-scenario.sh/query-evidence.sh still read the pre-azd " + "common.sh deployment lookup, not the current azd environment" + ) + + def test_scenario_waits_for_new_revision_before_load(): common = COMMON_SH.read_text() scenario = (Path(__file__).parents[1] / "run-scenario.sh").read_text() From d973feadf2814743f64fd91b76ca57f0e09c24ac Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 14:32:33 +0900 Subject: [PATCH 04/26] refactor(sre-lab): load configuration from azd Replace common.sh's fixed subscription/resource-group/deployment-name constants with load_lab_config, which resolves every setting as explicit process environment > current `azd env get-value` > an allowed default via a new setting()/require_setting() helper (no eval, no dynamic indirection). - deployment_output() now reads the AZURE_-prefixed (and legacy-named) azd deployment outputs load_lab_config already resolved, instead of `az deployment sub show` against a hardcoded deployment name. - verify_lab_resource_group() now requires both the purpose and azd-env-name tags to match the current azd environment, replacing the old single hardcoded resource-group-name safety boundary. - run-scenario.sh, query-evidence.sh, capture-scenario.sh, and cleanup.sh all call require_lab_config (require_commands + load_lab_config) before reading SUBSCRIPTION_ID/RESOURCE_GROUP or any deployment_output() value. - Added monitor/sre-agent-event-lab/.env.example documenting every setting name and its allowed default, with no secrets. - Rewrote the common.sh tests that intentionally pinned the old hardcoded subscription/resource-group values to assert the new azd-backed resolution instead, and added test_azd_env.py (with a shared fake azd/az harness in azd_common_harness.py) covering precedence, missing required settings, and the tag-based resource-group safety check. - Updated README's safety-boundary and scenario-execution sections: the scenario/query scripts are no longer legacy/transitional, they read the currently selected azd environment. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/.env.example | 10 + monitor/sre-agent-event-lab/.gitignore | 3 + monitor/sre-agent-event-lab/README.md | 19 +- .../infra/tests/test_azd_project.py | 11 +- .../scripts/capture-scenario.sh | 2 +- .../sre-agent-event-lab/scripts/cleanup.sh | 2 +- monitor/sre-agent-event-lab/scripts/common.sh | 125 +++++++++++-- .../scripts/query-evidence.sh | 2 +- .../scripts/run-scenario.sh | 2 +- .../scripts/tests/azd_common_harness.py | 81 +++++++++ .../scripts/tests/test_azd_env.py | 171 ++++++++++++++++++ .../scripts/tests/test_common.py | 126 +++++++++---- 12 files changed, 491 insertions(+), 63 deletions(-) create mode 100644 monitor/sre-agent-event-lab/.env.example create mode 100644 monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py diff --git a/monitor/sre-agent-event-lab/.env.example b/monitor/sre-agent-event-lab/.env.example new file mode 100644 index 0000000..64b266d --- /dev/null +++ b/monitor/sre-agent-event-lab/.env.example @@ -0,0 +1,10 @@ +AZURE_SUBSCRIPTION_ID= +AZURE_LOCATION=koreacentral +AZURE_ENV_NAME=sre-lab- +AZURE_RESOURCE_GROUP= +SRE_LAB_EXPIRES_ON= +SRE_AGENT_RESOURCE_ID= +SRE_AGENT_NAME= +SRE_REPOSITORY_URL= +SRE_REPOSITORY_BRANCH=main +SRE_KNOWLEDGE_PATH=runbooks/incident-response.md diff --git a/monitor/sre-agent-event-lab/.gitignore b/monitor/sre-agent-event-lab/.gitignore index f1cba58..e94ad31 100644 --- a/monitor/sre-agent-event-lab/.gitignore +++ b/monitor/sre-agent-event-lab/.gitignore @@ -1,2 +1,5 @@ # azd stores per-developer environment state (including subscription IDs) here. .azure/ +# Local overrides of the values documented in .env.example; never commit +# real subscription IDs or other environment-specific values here. +.env diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index d6b72ed..b37d06e 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -30,9 +30,9 @@ Azure SRE Agent Korea Central이 구독에 표시되지 않으면 [공식 regist ## 안전 경계 -- 스크립트는 현재 구독 ID를 고정 검증한다. +- 스크립트는 현재 azd 환경(또는 명시적 환경 변수)이 지정한 구독 ID를 `az account show`의 활성 구독과 일치하는지 검증한다. - 기존 resource group을 재사용하지 않는다. -- resource group의 `purpose=sre-agent-event-lab` 태그가 없으면 scenario와 cleanup을 거부한다. +- resource group에 `purpose=sre-agent-event-lab`과 `azd-env-name=<현재 azd environment 이름>` 태그가 모두 일치하지 않으면 scenario와 cleanup을 거부한다. - 한 번에 한 시나리오만 실행한다. - `run-scenario.sh`는 종료 trap으로 장애 복구를 시도하고 복구 실패를 명시적으로 오류 처리한다. - S3는 출력으로 기록된 Blob container scope의 단일 역할만 삭제·복구한다. @@ -69,6 +69,8 @@ azd up 2>&1 | tee evidence/deploy.log `azd up`은 Bicep provision → ACR cloud build → Container App image 전환 순서로 진행된다. 로컬 Docker는 필요하지 않다. 초기 provision은 public placeholder image를 port 80으로 띄우고, postprovision hook이 ingress를 8000으로 옮긴 뒤 lab image로 교체한다. `scripts/deploy.sh`는 위 `azd up`을 호출하는 호환 wrapper로 남아 있다. +`monitor/sre-agent-event-lab/.env.example`은 스크립트가 읽는 설정 값의 이름과 허용 기본값만 문서화한 비밀 정보 없는 참고 파일이다(비밀 값은 커밋하지 않는다). 각 값은 `scripts/common.sh`의 `load_lab_config`가 "명시적 환경 변수 > `azd env get-value` > 허용된 기본값" 순서로 해석하므로, 로컬에서 다르게 override하려면 `.env.example`을 복사해 값을 채운 뒤 `export $(grep -v '^#' .env | xargs)`처럼 셸 환경에 불러오거나 `azd env set `로 azd 환경에 저장한다. + 성공 조건: 1. Bicep provision이 성공한다. @@ -174,13 +176,12 @@ Application Insights의 `AppRequests`와 `AppDependencies`에 현재 데이터 ## 시나리오 실행 -> ⚠️ **전환기 주의사항 (레거시)**: `run-scenario.sh`와 `query-evidence.sh`는 아직 `scripts/common.sh`의 -> `deployment_output` 함수(구독 스코프 배포 조회, 고정된 `rg-sre-agent-event-lab-krc`)로 배포 output을 -> 조회한다. 즉 이 절의 명령은 `azd up`이 provision한 현재 azd 환경이 아니라 최초 실측 실행에서 만든 -> 리소스 그룹만을 대상으로 동작한다. 새로 `azd up`한 환경에서는 아직 동작을 보장하지 않으며, `common.sh`가 -> `azd env get-value`를 읽도록 재작성되는 후속 작업(가이드된 CLI 제공) 전까지는 아래 명령을 레거시 전용으로 -> 취급한다. 새 azd 환경에서 baseline을 확인하려면 위 [Baseline](#baseline) 절의 `azd env get-value` 기반 -> 명령을 사용한다. +> ℹ️ **azd 환경 설정**: `run-scenario.sh`, `query-evidence.sh`, `capture-scenario.sh`, `cleanup.sh`는 +> `scripts/common.sh`의 `load_lab_config`로 배포 output을 읽는다. `load_lab_config`는 각 값을 +> "명시적 프로세스 환경 변수 > 현재 `azd env get-value` > 허용된 기본값" 순서로 해석하므로, 고정된 +> 구독/리소스 그룹 값은 스크립트 안에 없다. `azd up`으로 provision한 현재 azd 환경(`azd env select`로 +> 선택한 environment)을 대상으로 동작하며, 안전 장치로 대상 resource group에 `purpose=sre-agent-event-lab`과 +> `azd-env-name=<현재 environment 이름>` 태그가 모두 일치해야 한다. 각 명령은 장애 주입, 제한 부하, alert polling, 복구, timeline 저장을 수행한다. diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index 8b1e4c1..528219a 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -151,10 +151,13 @@ def test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id(): subscription, so no file that documents or drives the current `azd` onboarding path may hardcode it. - `scripts/common.sh` (the pre-azd legacy flow, still intentionally pinned - per `test_common_does_not_expose_personal_subscription_display_name`) and - `validation-results.md` (the historical record of that specific real run) - are deliberately out of scope here. + `scripts/common.sh` no longer hardcodes a subscription ID either -- + `load_lab_config` resolves it from the explicit process environment or + the current `azd env get-value`, per `test_common.py`'s + `test_common_does_not_expose_personal_subscription_display_name`. It is + left out of `onboarding_paths` below only because `test_common.py` + already covers it directly; `validation-results.md` (the historical + record of that specific real run) is deliberately out of scope here. """ fixed_subscription_id = "95933ae5-0201-4a21-a1fc-8051a7437982" onboarding_paths = [ diff --git a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh index 69bccc5..6668b86 100644 --- a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh @@ -16,7 +16,7 @@ readonly NORMALIZED_FILE="${EVIDENCE_DIR}/normalized-timeline.json" readonly ASSET_DIR="${LAB_ROOT}/assets/captures/${SCENARIO}" readonly PYTHON="${LAB_ROOT}/app/.venv/bin/python" -require_commands +require_lab_config verify_subscription verify_lab_resource_group diff --git a/monitor/sre-agent-event-lab/scripts/cleanup.sh b/monitor/sre-agent-event-lab/scripts/cleanup.sh index dbe7938..bcec11e 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup.sh @@ -12,7 +12,7 @@ elif [[ "$#" -gt 0 ]]; then exit 2 fi -require_commands +require_lab_config verify_subscription if ! resource_group_exists; then diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index 01f11a1..3c79b56 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -1,11 +1,6 @@ #!/usr/bin/env bash set -euo pipefail -readonly SUBSCRIPTION_ID="95933ae5-0201-4a21-a1fc-8051a7437982" -readonly RESOURCE_GROUP="rg-sre-agent-event-lab-krc" -readonly LOCATION="koreacentral" -readonly FINAL_DEPLOYMENT_NAME="sre-agent-event-lab-private" - SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" readonly SCRIPT_DIR LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" @@ -15,7 +10,7 @@ readonly AGENT_SETUP_FILE="${EVIDENCE_ROOT}/agent-setup.json" require_commands() { local command_name - for command_name in az jq curl python3; do + for command_name in az azd jq curl python3; do if ! command -v "${command_name}" >/dev/null 2>&1; then echo "Required command not found: ${command_name}" >&2 return 1 @@ -23,6 +18,101 @@ require_commands() { done } +# azd_value NAME -- the current azd environment's value for NAME, or empty +# when azd has no such value (e.g. before the environment was provisioned). +# Never fails the caller: a missing value is not this function's error to +# report, `setting`/`require_setting` decide what to do about it. +azd_value() { + local name="$1" + azd env get-value "${name}" 2>/dev/null || true +} + +# setting NAME EXPLICIT_VALUE [DEFAULT] +# Resolves a configuration value in this order: EXPLICIT_VALUE (the caller's +# already-expanded process environment, e.g. "${AZURE_LOCATION:-}") > the +# current azd environment's value for NAME > DEFAULT. Every call site passes +# an explicit, already-expanded value -- there is no dynamic re-execution or +# indirect variable-name expansion here, only plain parameters. +setting() { + local name="$1" + local explicit_value="$2" + local default_value="${3:-}" + if [[ -n "${explicit_value}" ]]; then + printf '%s\n' "${explicit_value}" + return + fi + local stored_value + stored_value="$(azd_value "${name}")" + printf '%s\n' "${stored_value:-${default_value}}" +} + +# require_setting NAME EXPLICIT_VALUE [DEFAULT] -- like `setting`, but fails +# with an actionable message (the exact `azd env set` command to run) when +# nothing resolves instead of silently returning an empty string. +require_setting() { + local name="$1" + local explicit_value="$2" + local default_value="${3:-}" + local resolved + resolved="$(setting "${name}" "${explicit_value}" "${default_value}")" + if [[ -z "${resolved}" ]]; then + echo "Missing required setting ${name}." >&2 + echo "Run: azd env set ${name} " >&2 + return 1 + fi + printf '%s\n' "${resolved}" +} + +# load_lab_config -- resolves every azd-backed setting the lab scripts read +# (explicit process environment > current `azd env get-value` > an allowed +# default) and makes the resolved, non-secret values readonly. Every +# scenario/query/capture/cleanup script must call this before reading +# SUBSCRIPTION_ID, RESOURCE_GROUP, or any deployment_output() value. +load_lab_config() { + SUBSCRIPTION_ID="$(require_setting AZURE_SUBSCRIPTION_ID "${AZURE_SUBSCRIPTION_ID:-}")" || return 1 + RESOURCE_GROUP="$(require_setting AZURE_RESOURCE_GROUP "${AZURE_RESOURCE_GROUP:-}")" || return 1 + AZURE_ENV_NAME="$(require_setting AZURE_ENV_NAME "${AZURE_ENV_NAME:-}")" || return 1 + LOCATION="$(setting AZURE_LOCATION "${AZURE_LOCATION:-}" "koreacentral")" + readonly SUBSCRIPTION_ID RESOURCE_GROUP AZURE_ENV_NAME LOCATION + + # Deployment outputs read by deployment_output(). Every azd environment + # sets these once `azd up`/`azd provision` has run; empty until then. + AZURE_CONTAINER_APP_NAME="$(setting AZURE_CONTAINER_APP_NAME "${AZURE_CONTAINER_APP_NAME:-}" "")" + AZURE_CONTAINER_APP_FQDN="$(setting AZURE_CONTAINER_APP_FQDN "${AZURE_CONTAINER_APP_FQDN:-}" "")" + AZURE_STORAGE_CONTAINER_SCOPE="$(setting AZURE_STORAGE_CONTAINER_SCOPE "${AZURE_STORAGE_CONTAINER_SCOPE:-}" "")" + AZURE_BLOB_ROLE_ASSIGNMENT_NAME="$(setting AZURE_BLOB_ROLE_ASSIGNMENT_NAME "${AZURE_BLOB_ROLE_ASSIGNMENT_NAME:-}" "")" + AZURE_WORKSPACE_ID="$(setting AZURE_WORKSPACE_ID "${AZURE_WORKSPACE_ID:-}" "")" + AZURE_APP_INSIGHTS_NAME="$(setting AZURE_APP_INSIGHTS_NAME "${AZURE_APP_INSIGHTS_NAME:-}" "")" + AZURE_TELEMETRY_SERVICE_NAME="$(setting AZURE_TELEMETRY_SERVICE_NAME "${AZURE_TELEMETRY_SERVICE_NAME:-}" "")" + # Deployment outputs without an AZURE_-prefixed duplicate (see + # infra/main.bicep): read straight from their own azd output name. + CONTAINER_APP_PRINCIPAL_ID="$(setting containerAppPrincipalId "${CONTAINER_APP_PRINCIPAL_ID:-}" "")" + WORKSPACE_CUSTOMER_ID="$(setting workspaceCustomerId "${WORKSPACE_CUSTOMER_ID:-}" "")" + readonly AZURE_CONTAINER_APP_NAME AZURE_CONTAINER_APP_FQDN AZURE_STORAGE_CONTAINER_SCOPE + readonly AZURE_BLOB_ROLE_ASSIGNMENT_NAME AZURE_WORKSPACE_ID AZURE_APP_INSIGHTS_NAME + readonly AZURE_TELEMETRY_SERVICE_NAME CONTAINER_APP_PRINCIPAL_ID WORKSPACE_CUSTOMER_ID + + # Azure SRE Agent settings (.env.example documents these). None of the + # current scripts read them yet, but they resolve through the same + # explicit-env > azd-env > default rule so nothing here is ever fixed. + SRE_LAB_EXPIRES_ON="$(setting SRE_LAB_EXPIRES_ON "${SRE_LAB_EXPIRES_ON:-}" "")" + SRE_AGENT_RESOURCE_ID="$(setting SRE_AGENT_RESOURCE_ID "${SRE_AGENT_RESOURCE_ID:-}" "")" + SRE_AGENT_NAME="$(setting SRE_AGENT_NAME "${SRE_AGENT_NAME:-}" "")" + SRE_REPOSITORY_URL="$(setting SRE_REPOSITORY_URL "${SRE_REPOSITORY_URL:-}" "")" + SRE_REPOSITORY_BRANCH="$(setting SRE_REPOSITORY_BRANCH "${SRE_REPOSITORY_BRANCH:-}" "main")" + SRE_KNOWLEDGE_PATH="$(setting SRE_KNOWLEDGE_PATH "${SRE_KNOWLEDGE_PATH:-}" "runbooks/incident-response.md")" + readonly SRE_LAB_EXPIRES_ON SRE_AGENT_RESOURCE_ID SRE_AGENT_NAME + readonly SRE_REPOSITORY_URL SRE_REPOSITORY_BRANCH SRE_KNOWLEDGE_PATH +} + +# require_lab_config -- the one-call preflight for scenario/query/capture/ +# cleanup scripts: verifies the required CLIs are on PATH, then resolves the +# lab configuration via load_lab_config. +require_lab_config() { + require_commands + load_lab_config +} + verify_subscription() { local current_subscription current_subscription="$(az account show --query id -o tsv)" @@ -43,20 +133,31 @@ verify_lab_resource_group() { return 1 fi - local purpose + local purpose tagged_env_name purpose="$(az group show --name "${RESOURCE_GROUP}" --query "tags.purpose" -o tsv)" - if [[ "${purpose}" != "sre-agent-event-lab" ]]; then + tagged_env_name="$(az group show --name "${RESOURCE_GROUP}" --query 'tags."azd-env-name"' -o tsv)" + if [[ "${purpose}" != "sre-agent-event-lab" || "${tagged_env_name}" != "${AZURE_ENV_NAME}" ]]; then echo "Refusing to operate on untagged resource group ${RESOURCE_GROUP}." >&2 return 1 fi } +# deployment_output NAME -- returns an azd deployment output already +# resolved by load_lab_config. Callers must call load_lab_config first. deployment_output() { local output_name="$1" - az deployment sub show \ - --name "${FINAL_DEPLOYMENT_NAME}" \ - --query "properties.outputs.${output_name}.value" \ - -o tsv + case "${output_name}" in + containerAppName) printf '%s\n' "${AZURE_CONTAINER_APP_NAME}" ;; + containerAppFqdn) printf '%s\n' "${AZURE_CONTAINER_APP_FQDN}" ;; + containerAppPrincipalId) printf '%s\n' "${CONTAINER_APP_PRINCIPAL_ID}" ;; + storageContainerScope) printf '%s\n' "${AZURE_STORAGE_CONTAINER_SCOPE}" ;; + blobRoleAssignmentName) printf '%s\n' "${AZURE_BLOB_ROLE_ASSIGNMENT_NAME}" ;; + workspaceId) printf '%s\n' "${AZURE_WORKSPACE_ID}" ;; + workspaceCustomerId) printf '%s\n' "${WORKSPACE_CUSTOMER_ID}" ;; + appInsightsName) printf '%s\n' "${AZURE_APP_INSIGHTS_NAME}" ;; + telemetryServiceName) printf '%s\n' "${AZURE_TELEMETRY_SERVICE_NAME}" ;; + *) echo "Unknown deployment output: ${output_name}" >&2; return 2 ;; + esac } create_evidence_dir() { diff --git a/monitor/sre-agent-event-lab/scripts/query-evidence.sh b/monitor/sre-agent-event-lab/scripts/query-evidence.sh index bc3e63d..af6471d 100755 --- a/monitor/sre-agent-event-lab/scripts/query-evidence.sh +++ b/monitor/sre-agent-event-lab/scripts/query-evidence.sh @@ -14,7 +14,7 @@ readonly EVIDENCE_DIR="$2" readonly START_UTC="$3" readonly END_UTC="$4" -require_commands +require_lab_config verify_subscription verify_lab_resource_group mkdir -p "${EVIDENCE_DIR}" diff --git a/monitor/sre-agent-event-lab/scripts/run-scenario.sh b/monitor/sre-agent-event-lab/scripts/run-scenario.sh index a76287b..27c10c6 100755 --- a/monitor/sre-agent-event-lab/scripts/run-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/run-scenario.sh @@ -10,7 +10,7 @@ if [[ "$#" -ne 1 ]] || [[ ! "$1" =~ ^s[123]$ ]]; then fi readonly SCENARIO="$1" -require_commands +require_lab_config verify_subscription verify_lab_resource_group diff --git a/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py b/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py new file mode 100644 index 0000000..f97baad --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py @@ -0,0 +1,81 @@ +"""Shared test harness for exercising `common.sh` against a fake `azd`/`az`. + +Not a test module itself (no `test_` prefix) -- imported by test_common.py and +test_azd_env.py so both suites drive `common.sh` through the same fake-CLI +harness instead of duplicating it. +""" +import os +import shutil +import stat +import subprocess +from pathlib import Path + + +COMMON_SH = Path(__file__).parents[1] / "common.sh" +BASH = shutil.which("bash") or "/bin/bash" + + +def _make_executable(path, content): + path.write_text(content) + mode = path.stat().st_mode + path.chmod(mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + + +def _write_azd_stub(bin_dir, azd_values): + """A fake `azd` that only understands `env get-value NAME`. + + Returns the mapped value for a known name with exit 0, and -- matching + the real `azd`'s behaviour for a value that was never set -- prints + nothing and exits non-zero for an unknown name. + """ + lines = [ + "#!/usr/bin/env bash", + 'if [[ "${1:-}" == "env" && "${2:-}" == "get-value" ]]; then', + ' case "${3:-}" in', + ] + for name, value in azd_values.items(): + escaped = value.replace("'", "'\\''") + lines.append(f" {name}) printf '%s' '{escaped}'; exit 0 ;;") + lines.append(' *) exit 1 ;;') + lines.append(" esac") + lines.append("fi") + lines.append('echo "azd stub: unsupported invocation: $*" >&2') + lines.append("exit 1") + _make_executable(bin_dir / "azd", "\n".join(lines) + "\n") + + +def _write_az_stub(bin_dir, az_script=None): + """A fake `az` used only so `require_commands`/`command -v az` succeed. + + `az_script` lets a test override behaviour (e.g. `az account show`, + `az group show`) by supplying the body of the stub script. + """ + body = az_script if az_script is not None else "exit 0\n" + _make_executable(bin_dir / "az", "#!/usr/bin/env bash\n" + body) + + +def run_common(tmp_path, env, azd_values, command, az_script=None): + """Source common.sh in a throwaway bash process and run `command`. + + `env` is the *only* process environment passed through (plus PATH), + so tests control explicit-environment precedence precisely. `azd_values` + populates the fake `azd env get-value` responses. + """ + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + _write_azd_stub(bin_dir, azd_values) + _write_az_stub(bin_dir, az_script) + + full_env = { + "PATH": f"{bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", + "HOME": os.environ.get("HOME", str(tmp_path)), + } + full_env.update(env) + + harness = f'source "{COMMON_SH}"\n{command}\n' + return subprocess.run( + [BASH, "-c", harness], + capture_output=True, + text=True, + env=full_env, + ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py new file mode 100644 index 0000000..f32c4b7 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py @@ -0,0 +1,171 @@ +"""Behaviour tests for common.sh's azd-backed configuration loader. + +`load_lab_config` must resolve every setting as: explicit process +environment > current `azd env get-value` > an allowed default -- and must +never fall back to a fixed subscription/resource group again. These tests +drive `common.sh` through a fake `azd`/`az` on PATH (see +`azd_common_harness.py`); no real Azure CLI or azd call is made. +""" +from azd_common_harness import COMMON_SH, run_common + + +REQUIRED_ENV = { + "AZURE_SUBSCRIPTION_ID": "azd-sub-11111111", + "AZURE_RESOURCE_GROUP": "rg-azd-sub-lab", + "AZURE_ENV_NAME": "sre-lab-dev", +} + + +def test_explicit_environment_wins_over_azd_value(tmp_path): + result = run_common( + tmp_path, + env={**REQUIRED_ENV, "AZURE_SUBSCRIPTION_ID": "explicit-sub"}, + azd_values={ + "AZURE_SUBSCRIPTION_ID": "azd-sub", + "AZURE_RESOURCE_GROUP": REQUIRED_ENV["AZURE_RESOURCE_GROUP"], + "AZURE_ENV_NAME": REQUIRED_ENV["AZURE_ENV_NAME"], + }, + command='load_lab_config; printf "%s" "${SUBSCRIPTION_ID}"', + ) + assert result.returncode == 0, result.stderr + assert result.stdout == "explicit-sub" + + +def test_current_azd_value_wins_over_default_when_no_explicit_env(tmp_path): + result = run_common( + tmp_path, + env={k: v for k, v in REQUIRED_ENV.items() if k != "AZURE_SUBSCRIPTION_ID"}, + azd_values={ + **REQUIRED_ENV, + "AZURE_LOCATION": "eastus", + }, + command='load_lab_config; printf "%s" "${LOCATION}"', + ) + assert result.returncode == 0, result.stderr + assert result.stdout == "eastus" + + +def test_allowed_default_applies_when_neither_explicit_env_nor_azd_value_exist(tmp_path): + result = run_common( + tmp_path, + env=dict(REQUIRED_ENV), + azd_values=dict(REQUIRED_ENV), + command='load_lab_config; printf "%s" "${LOCATION}"', + ) + assert result.returncode == 0, result.stderr + assert result.stdout == "koreacentral" + + +def test_missing_required_setting_names_the_azd_command(tmp_path): + result = run_common( + tmp_path, + env={}, + azd_values={}, + command="load_lab_config", + ) + assert result.returncode != 0 + assert "azd env set AZURE_SUBSCRIPTION_ID" in result.stderr + + +def test_missing_resource_group_names_the_azd_command_once_subscription_resolves(tmp_path): + result = run_common( + tmp_path, + env={"AZURE_SUBSCRIPTION_ID": "explicit-sub"}, + azd_values={}, + command="load_lab_config", + ) + assert result.returncode != 0 + assert "azd env set AZURE_RESOURCE_GROUP" in result.stderr + + +def test_load_lab_config_resolves_sre_agent_settings_with_documented_defaults(tmp_path): + result = run_common( + tmp_path, + env=dict(REQUIRED_ENV), + azd_values=dict(REQUIRED_ENV), + command=( + 'load_lab_config; ' + 'printf "%s|%s" "${SRE_REPOSITORY_BRANCH}" "${SRE_KNOWLEDGE_PATH}"' + ), + ) + assert result.returncode == 0, result.stderr + assert result.stdout == "main|runbooks/incident-response.md" + + +def test_load_lab_config_does_not_fix_subscription_or_resource_group_values(tmp_path): + """Regression guard for the removed hardcoded values: two different azd + environments (different subscription, resource group, env name) must + each resolve to their own values -- nothing in common.sh may fall back + to a value baked into the script. + """ + other_env = { + "AZURE_SUBSCRIPTION_ID": "22222222-3333-4444-5555-666666666666", + "AZURE_RESOURCE_GROUP": "rg-some-other-lab", + "AZURE_ENV_NAME": "sre-lab-other", + } + result = run_common( + tmp_path, + env=dict(other_env), + azd_values=dict(other_env), + command='load_lab_config; printf "%s|%s" "${SUBSCRIPTION_ID}" "${RESOURCE_GROUP}"', + ) + assert result.returncode == 0, result.stderr + assert result.stdout == "22222222-3333-4444-5555-666666666666|rg-some-other-lab" + + +def test_setting_helper_does_not_use_eval_or_indirection(): + script = COMMON_SH.read_text() + + assert "eval" not in script + assert "declare -n" not in script + # No `${!name}` indirect expansion anywhere in the setting/azd_value path. + assert "${!" not in script + + +def test_resource_group_safety_requires_purpose_and_environment_tags(): + script = COMMON_SH.read_text() + + assert "tags.purpose" in script + assert 'tags."azd-env-name"' in script + + +def test_verify_lab_resource_group_refuses_when_azd_env_name_tag_mismatches(tmp_path): + az_script = """ +case "$1 $2" in + "group exists") + echo true + ;; + "group show") + if [[ "$*" == *'tags."azd-env-name"'* ]]; then + echo "a-different-azd-environment" + else + echo "sre-agent-event-lab" + fi + ;; +esac +exit 0 +""" + result = run_common( + tmp_path, + env=dict(REQUIRED_ENV), + azd_values=dict(REQUIRED_ENV), + command="load_lab_config; verify_lab_resource_group", + az_script=az_script, + ) + assert result.returncode != 0 + assert "Refusing to operate on untagged resource group" in result.stderr + + +def test_require_lab_config_requires_commands_before_loading_config(tmp_path): + result = run_common( + tmp_path, + env={}, + azd_values={}, + command="require_lab_config", + ) + assert result.returncode != 0 + # require_commands runs first; jq is missing because run_common's fake + # PATH does not put it ahead of any earlier failing check. + assert "Required command not found" in result.stderr or ( + "azd env set AZURE_SUBSCRIPTION_ID" in result.stderr + ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index a69a6d2..1bf0fa9 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -2,19 +2,34 @@ import subprocess from pathlib import Path +from azd_common_harness import run_common + COMMON_SH = Path(__file__).parents[1] / "common.sh" DEPLOY_SH = Path(__file__).parents[1] / "deploy.sh" CLEANUP_SH = Path(__file__).parents[1] / "cleanup.sh" QUERY_EVIDENCE_SH = Path(__file__).parents[1] / "query-evidence.sh" +REQUIRED_ENV = { + "AZURE_SUBSCRIPTION_ID": "11111111-2222-3333-4444-555555555555", + "AZURE_RESOURCE_GROUP": "rg-sre-lab-test", + "AZURE_ENV_NAME": "sre-lab-test", +} + -def test_outputs_come_from_latest_subscription_deployment(): +def test_deployment_output_reads_the_current_azd_environment(): + """`deployment_output` used to look up a fixed subscription-scope + deployment name; it must now read the values `load_lab_config` already + resolved from the current azd environment, so a fresh `azd up` in any + environment works without editing common.sh. + """ script = COMMON_SH.read_text() - assert 'FINAL_DEPLOYMENT_NAME="sre-agent-event-lab-private"' in script - assert "az deployment sub show" in script + assert "FINAL_DEPLOYMENT_NAME" not in script + assert "az deployment sub show" not in script assert "az deployment group show" not in script + assert "azd env get-value" in script + assert 'containerAppName) printf \'%s\\n\' "${AZURE_CONTAINER_APP_NAME}"' in script def test_common_does_not_expose_personal_subscription_display_name(): @@ -23,36 +38,43 @@ def test_common_does_not_expose_personal_subscription_display_name(): assert "SUBSCRIPTION_NAME" not in script assert "ME-MngEnvMCAP310512-inhwanhwang-3" not in script assert "inhwanhwang" not in script + # The one true fixed subscription this task removed: common.sh must + # never hardcode a subscription ID again, it must resolve one via + # load_lab_config (explicit env > azd env > default) instead. + assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in script + assert 'SUBSCRIPTION_ID="$(require_setting AZURE_SUBSCRIPTION_ID' in script # The subscription ID equality check is the real safety boundary and # must remain intact. - assert 'readonly SUBSCRIPTION_ID="95933ae5-0201-4a21-a1fc-8051a7437982"' in script + assert '"${current_subscription}" != "${SUBSCRIPTION_ID}"' in script -def test_verify_subscription_reports_only_subscription_id_on_mismatch(): +def test_verify_subscription_reports_only_subscription_id_on_mismatch(tmp_path): """verify_subscription must not reference an undeclared SUBSCRIPTION_NAME. Regression guard for `set -u`: after removing SUBSCRIPTION_NAME, the - mismatch message must only report the expected SUBSCRIPTION_ID, or the - function will crash under `set -u` when the variable no longer exists. + mismatch message must only report the resolved SUBSCRIPTION_ID (now + loaded via `load_lab_config`, not a hardcoded literal), or the function + will crash under `set -u` when the variable no longer exists. """ - bash_path = shutil.which("bash") or "/bin/bash" - - harness = ( - "set -euo pipefail\n" - 'source "$1"\n' - "az() {\n echo \"wrong-subscription-id\"\n}\n" - "verify_subscription\n" + az_script = ( + 'if [[ "$1 $2" == "account show" ]]; then\n' + ' echo "wrong-subscription-id"\n' + " exit 0\n" + "fi\n" + "exit 0\n" ) - result = subprocess.run( - [bash_path, "-c", harness, "bash", str(COMMON_SH)], - capture_output=True, - text=True, + result = run_common( + tmp_path, + env=dict(REQUIRED_ENV), + azd_values=dict(REQUIRED_ENV), + command="load_lab_config; verify_subscription", + az_script=az_script, ) assert result.returncode != 0 assert "unbound variable" not in result.stderr assert "SUBSCRIPTION_NAME" not in result.stderr - assert "95933ae5-0201-4a21-a1fc-8051a7437982" in result.stderr + assert REQUIRED_ENV["AZURE_SUBSCRIPTION_ID"] in result.stderr def test_deploy_delegates_to_azd_up(): @@ -102,26 +124,27 @@ def test_readme_documents_a_working_deployment_command(): assert "azd env get-value AZURE_CONTAINER_APP_FQDN" in readme -def test_readme_marks_legacy_scenario_scripts_as_transitional(): - """`run-scenario.sh` and `query-evidence.sh` still resolve deployment - outputs through `common.sh`'s `deployment_output` (`az deployment sub - show --name sre-agent-event-lab-private` against the hardcoded - `rg-sre-agent-event-lab-krc`), not through the `azd`-provisioned - environment introduced by this task. The README must not present them as - working against a fresh `azd up` environment without saying so. +def test_readme_documents_scenario_scripts_read_the_current_azd_environment(): + """`run-scenario.sh` and `query-evidence.sh` now resolve deployment + outputs through `common.sh`'s `load_lab_config` (explicit env > current + `azd env get-value` > allowed default), so they work against whatever + azd environment is currently selected -- not a fixed pre-azd resource + group. The README's scenario-execution section must describe that + mechanism instead of the old "legacy, not yet rewritten" caveat. """ readme = (Path(__file__).parents[2] / "README.md").read_text() scenario_heading = "## 시나리오 실행" assert scenario_heading in readme - section = readme.split(scenario_heading, 1)[1] - - assert "run-scenario.sh" in section.split("##", 1)[0] - caveat_markers = ("common.sh", "레거시", "azd 환경") - assert any(marker in section.split("##", 1)[0] for marker in caveat_markers), ( - "README's scenario-execution section must caveat that " - "run-scenario.sh/query-evidence.sh still read the pre-azd " - "common.sh deployment lookup, not the current azd environment" + section = readme.split(scenario_heading, 1)[1].split("##", 1)[0] + + assert "run-scenario.sh" in section + assert "load_lab_config" in section + assert "레거시" not in section, ( + "README's scenario-execution section must no longer describe " + "run-scenario.sh/query-evidence.sh as reading a legacy, pre-azd " + "deployment lookup -- load_lab_config now reads the current azd " + "environment." ) @@ -224,3 +247,38 @@ def test_cleanup_deletion_loop_tolerates_empty_role_assignments_on_bash32(tmp_pa calls = call_log.read_text() if call_log.exists() else "" assert "az role assignment delete" not in calls assert "az group delete" in calls + + +def test_scenario_query_capture_cleanup_scripts_load_lab_config_before_use(): + """Every script that reads SUBSCRIPTION_ID/RESOURCE_GROUP/deployment_output + must resolve them via `load_lab_config` (through `require_lab_config`) + first, and must do so before `verify_subscription`/`verify_lab_resource_group` + or any deployment_output() call that depends on those values. + """ + scripts_dir = Path(__file__).parents[1] + for script_name in ( + "run-scenario.sh", + "query-evidence.sh", + "capture-scenario.sh", + "cleanup.sh", + ): + script = (scripts_dir / script_name).read_text() + assert "require_lab_config" in script, ( + f"{script_name} must call require_lab_config (which calls " + "load_lab_config) before using SUBSCRIPTION_ID/RESOURCE_GROUP" + ) + config_index = script.index("require_lab_config") + + for later_use in ("verify_subscription", "verify_lab_resource_group"): + if later_use in script: + assert config_index < script.index(later_use), ( + f"{script_name} must call require_lab_config before {later_use}" + ) + + if "deployment_output " in script: + first_output_call = script.index("deployment_output ") + assert config_index < first_output_call, ( + f"{script_name} must call require_lab_config before reading " + "any deployment_output() value" + ) + From fbcb8cad012d7a20c72a35972a5c512b8c6ef142 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 14:58:52 +0900 Subject: [PATCH 05/26] fix(sre-lab): read azd values by exit status and pin the project root `azd env get-value` writes its `ERROR: ...` diagnostics to stdout and signals failure only through the exit status (verified against azd 1.29.0), so keeping stdout unconditionally turned "key not found" and "no project exists" into configuration values: SUBSCRIPTION_ID became an error sentence, documented defaults never applied, and the preprovision hook skipped deriving AZURE_RESOURCE_GROUP. Every lookup now uses the value only when azd exited 0 and pins `--cwd` to the lab's azd project so the scripts work from the repository root or any other directory. `load_lab_config` also declared `WORKSPACE_CUSTOMER_ID` / `CONTAINER_APP_PRINCIPAL_ID` readonly, which aborted query-evidence.sh under `set -e` on its own assignment; the resolved values moved to LAB_-prefixed names. Tests now run the four entry points as programs against fake az/azd/python executables from a working directory outside the lab, the fake azd reproduces azd 1.29's contract (both missing-value shapes), and the text-order/eval-substring assertions were replaced by behaviour, fail-closed and token-aware checks. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../scripts/azd-configure.sh | 34 ++- monitor/sre-agent-event-lab/scripts/common.sh | 31 ++- .../scripts/tests/azd_common_harness.py | 92 ++++--- .../scripts/tests/azd_fake.py | 109 ++++++++ .../scripts/tests/lab_script_harness.py | 257 ++++++++++++++++++ .../scripts/tests/test_azd_env.py | 155 +++++++++-- .../scripts/tests/test_azd_hooks.py | 99 +++++++ .../scripts/tests/test_common.py | 85 +++--- .../scripts/tests/test_lab_scripts.py | 245 +++++++++++++++++ 9 files changed, 1011 insertions(+), 96 deletions(-) create mode 100644 monitor/sre-agent-event-lab/scripts/tests/azd_fake.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py diff --git a/monitor/sre-agent-event-lab/scripts/azd-configure.sh b/monitor/sre-agent-event-lab/scripts/azd-configure.sh index cc9c611..b0a5c53 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-configure.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-configure.sh @@ -1,6 +1,11 @@ #!/usr/bin/env bash set -euo pipefail +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +readonly SCRIPT_DIR +LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +readonly LAB_ROOT + for command_name in az azd jq curl python3; do command -v "${command_name}" >/dev/null 2>&1 || { echo "Required command not found: ${command_name}" >&2 @@ -8,6 +13,21 @@ for command_name in az azd jq curl python3; do } done +# azd_value NAME -- the current azd environment's value for NAME, or empty. +# azd 1.29 answers an unknown key with an `ERROR: ...` sentence on *stdout* +# and a non-zero exit status, so only a successful lookup may contribute a +# value; otherwise the defaults below would never be applied. `--cwd` pins +# every lookup to this lab's azd project so the hook also works when it is +# run by hand from another directory. +azd_value() { + local name="$1" + local stored_value + if ! stored_value="$(azd env get-value "${name}" --cwd "${LAB_ROOT}" 2>/dev/null)"; then + return 0 + fi + printf '%s\n' "${stored_value}" +} + : "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set by azd before running this hook}" # The Azure CLI's active account is whatever the operator selected last, which @@ -37,11 +57,17 @@ for provider in Microsoft.App Microsoft.OperationalInsights Microsoft.Insights \ --subscription "${AZURE_SUBSCRIPTION_ID}" done -if [[ -z "$(azd env get-value AZURE_RESOURCE_GROUP 2>/dev/null || true)" ]]; then - azd env set AZURE_RESOURCE_GROUP "rg-$(azd env get-value AZURE_ENV_NAME)" +if [[ -z "$(azd_value AZURE_RESOURCE_GROUP)" ]]; then + environment_name="$(azd_value AZURE_ENV_NAME)" + if [[ -z "${environment_name}" ]]; then + echo "The azd environment name is unavailable, so the lab resource group" >&2 + echo "cannot be derived. Run 'azd env new' or 'azd env select' first." >&2 + exit 1 + fi + azd env set AZURE_RESOURCE_GROUP "rg-${environment_name}" --cwd "${LAB_ROOT}" fi -if [[ -z "$(azd env get-value SRE_LAB_EXPIRES_ON 2>/dev/null || true)" ]]; then +if [[ -z "$(azd_value SRE_LAB_EXPIRES_ON)" ]]; then expires_on="$(python3 -c 'from datetime import date,timedelta; print(date.today()+timedelta(days=1))')" - azd env set SRE_LAB_EXPIRES_ON "${expires_on}" + azd env set SRE_LAB_EXPIRES_ON "${expires_on}" --cwd "${LAB_ROOT}" fi diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index 3c79b56..b5f8b3e 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -20,11 +20,23 @@ require_commands() { # azd_value NAME -- the current azd environment's value for NAME, or empty # when azd has no such value (e.g. before the environment was provisioned). +# +# azd reports failure only through its exit status: azd 1.29 answers an +# unknown key -- or a working directory outside the project -- with an +# `ERROR: ...` sentence on *stdout* and exit 1. Keeping stdout regardless of +# the exit status would adopt that sentence as a configuration value, so the +# output is used only when the lookup succeeded. `--cwd` pins the lookup to +# this lab's azd project (the directory holding azure.yaml) so the scripts +# work when they are invoked from the repository root or any other cwd. # Never fails the caller: a missing value is not this function's error to # report, `setting`/`require_setting` decide what to do about it. azd_value() { local name="$1" - azd env get-value "${name}" 2>/dev/null || true + local stored_value + if ! stored_value="$(azd env get-value "${name}" --cwd "${LAB_ROOT}" 2>/dev/null)"; then + return 0 + fi + printf '%s\n' "${stored_value}" } # setting NAME EXPLICIT_VALUE [DEFAULT] @@ -85,12 +97,17 @@ load_lab_config() { AZURE_APP_INSIGHTS_NAME="$(setting AZURE_APP_INSIGHTS_NAME "${AZURE_APP_INSIGHTS_NAME:-}" "")" AZURE_TELEMETRY_SERVICE_NAME="$(setting AZURE_TELEMETRY_SERVICE_NAME "${AZURE_TELEMETRY_SERVICE_NAME:-}" "")" # Deployment outputs without an AZURE_-prefixed duplicate (see - # infra/main.bicep): read straight from their own azd output name. - CONTAINER_APP_PRINCIPAL_ID="$(setting containerAppPrincipalId "${CONTAINER_APP_PRINCIPAL_ID:-}" "")" - WORKSPACE_CUSTOMER_ID="$(setting workspaceCustomerId "${WORKSPACE_CUSTOMER_ID:-}" "")" + # infra/main.bicep): read straight from their own azd output name. They + # are stored under LAB_-prefixed names because `load_lab_config` makes + # every resolved value readonly for the rest of the process, and the + # calling scripts already use the unprefixed names for their own + # variables -- an assignment to a readonly name aborts the caller under + # `set -e` before it reaches its first Azure call. + LAB_CONTAINER_APP_PRINCIPAL_ID="$(setting containerAppPrincipalId "${CONTAINER_APP_PRINCIPAL_ID:-}" "")" + LAB_WORKSPACE_CUSTOMER_ID="$(setting workspaceCustomerId "${WORKSPACE_CUSTOMER_ID:-}" "")" readonly AZURE_CONTAINER_APP_NAME AZURE_CONTAINER_APP_FQDN AZURE_STORAGE_CONTAINER_SCOPE readonly AZURE_BLOB_ROLE_ASSIGNMENT_NAME AZURE_WORKSPACE_ID AZURE_APP_INSIGHTS_NAME - readonly AZURE_TELEMETRY_SERVICE_NAME CONTAINER_APP_PRINCIPAL_ID WORKSPACE_CUSTOMER_ID + readonly AZURE_TELEMETRY_SERVICE_NAME LAB_CONTAINER_APP_PRINCIPAL_ID LAB_WORKSPACE_CUSTOMER_ID # Azure SRE Agent settings (.env.example documents these). None of the # current scripts read them yet, but they resolve through the same @@ -149,11 +166,11 @@ deployment_output() { case "${output_name}" in containerAppName) printf '%s\n' "${AZURE_CONTAINER_APP_NAME}" ;; containerAppFqdn) printf '%s\n' "${AZURE_CONTAINER_APP_FQDN}" ;; - containerAppPrincipalId) printf '%s\n' "${CONTAINER_APP_PRINCIPAL_ID}" ;; + containerAppPrincipalId) printf '%s\n' "${LAB_CONTAINER_APP_PRINCIPAL_ID}" ;; storageContainerScope) printf '%s\n' "${AZURE_STORAGE_CONTAINER_SCOPE}" ;; blobRoleAssignmentName) printf '%s\n' "${AZURE_BLOB_ROLE_ASSIGNMENT_NAME}" ;; workspaceId) printf '%s\n' "${AZURE_WORKSPACE_ID}" ;; - workspaceCustomerId) printf '%s\n' "${WORKSPACE_CUSTOMER_ID}" ;; + workspaceCustomerId) printf '%s\n' "${LAB_WORKSPACE_CUSTOMER_ID}" ;; appInsightsName) printf '%s\n' "${AZURE_APP_INSIGHTS_NAME}" ;; telemetryServiceName) printf '%s\n' "${AZURE_TELEMETRY_SERVICE_NAME}" ;; *) echo "Unknown deployment output: ${output_name}" >&2; return 2 ;; diff --git a/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py b/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py index f97baad..758a864 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/azd_common_harness.py @@ -2,46 +2,27 @@ Not a test module itself (no `test_` prefix) -- imported by test_common.py and test_azd_env.py so both suites drive `common.sh` through the same fake-CLI -harness instead of duplicating it. +harness instead of duplicating it. The fake `azd` reproduces azd 1.29.0's +real contract (see `azd_fake.py`): `ERROR:` text on **stdout**, failure only +in the exit status, and project discovery from `--cwd` or the process +working directory. """ import os import shutil -import stat import subprocess from pathlib import Path +from azd_fake import write_azd_stub, write_executable + COMMON_SH = Path(__file__).parents[1] / "common.sh" +LAB_ROOT = Path(__file__).parents[2] BASH = shutil.which("bash") or "/bin/bash" - - -def _make_executable(path, content): - path.write_text(content) - mode = path.stat().st_mode - path.chmod(mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) - - -def _write_azd_stub(bin_dir, azd_values): - """A fake `azd` that only understands `env get-value NAME`. - - Returns the mapped value for a known name with exit 0, and -- matching - the real `azd`'s behaviour for a value that was never set -- prints - nothing and exits non-zero for an unknown name. - """ - lines = [ - "#!/usr/bin/env bash", - 'if [[ "${1:-}" == "env" && "${2:-}" == "get-value" ]]; then', - ' case "${3:-}" in', - ] - for name, value in azd_values.items(): - escaped = value.replace("'", "'\\''") - lines.append(f" {name}) printf '%s' '{escaped}'; exit 0 ;;") - lines.append(' *) exit 1 ;;') - lines.append(" esac") - lines.append("fi") - lines.append('echo "azd stub: unsupported invocation: $*" >&2') - lines.append("exit 1") - _make_executable(bin_dir / "azd", "\n".join(lines) + "\n") +# Sourcing common.sh needs `dirname`; every other command it runs at source +# time is a Bash builtin. Restricted-PATH runs symlink just this one binary +# so a "required command" test controls exactly which CLIs are reachable. +SOURCE_TIME_BINARIES = ("dirname",) +REQUIRED_COMMANDS = ("az", "azd", "jq", "curl", "python3") def _write_az_stub(bin_dir, az_script=None): @@ -51,23 +32,59 @@ def _write_az_stub(bin_dir, az_script=None): `az group show`) by supplying the body of the stub script. """ body = az_script if az_script is not None else "exit 0\n" - _make_executable(bin_dir / "az", "#!/usr/bin/env bash\n" + body) + write_executable(bin_dir / "az", "#!/usr/bin/env bash\n" + body) + +def _write_trivial_stub(bin_dir, name): + write_executable(bin_dir / name, "#!/usr/bin/env bash\nexit 0\n") -def run_common(tmp_path, env, azd_values, command, az_script=None): + +def run_common( + tmp_path, + env, + azd_values, + command, + az_script=None, + cwd=None, + missing_key_mode="azd_1_29", + available_commands=None, + azd_log=None, +): """Source common.sh in a throwaway bash process and run `command`. `env` is the *only* process environment passed through (plus PATH), so tests control explicit-environment precedence precisely. `azd_values` - populates the fake `azd env get-value` responses. + populates the fake `azd env get-value` responses. `cwd` runs the shell + from another directory (the default, `tmp_path`, deliberately holds no + `azure.yaml`, so a lookup that does not pin the lab's project root + fails exactly as the real azd would). `available_commands`, when given, + restricts PATH to those fakes so a preflight test can drop one CLI. """ bin_dir = tmp_path / "bin" bin_dir.mkdir(exist_ok=True) - _write_azd_stub(bin_dir, azd_values) - _write_az_stub(bin_dir, az_script) + + if available_commands is None: + selected = REQUIRED_COMMANDS + path_value = f"{bin_dir}{os.pathsep}{os.environ.get('PATH', '')}" + else: + selected = tuple(available_commands) + for binary in SOURCE_TIME_BINARIES: + resolved = shutil.which(binary) + link = bin_dir / binary + if resolved and not link.exists(): + os.symlink(resolved, link) + path_value = str(bin_dir) + + for name in selected: + if name == "azd": + write_azd_stub(bin_dir, azd_values, missing_key_mode, azd_log) + elif name == "az": + _write_az_stub(bin_dir, az_script) + else: + _write_trivial_stub(bin_dir, name) full_env = { - "PATH": f"{bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", + "PATH": path_value, "HOME": os.environ.get("HOME", str(tmp_path)), } full_env.update(env) @@ -78,4 +95,5 @@ def run_common(tmp_path, env, azd_values, command, az_script=None): capture_output=True, text=True, env=full_env, + cwd=str(cwd) if cwd is not None else str(tmp_path), ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py b/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py new file mode 100644 index 0000000..4ea5473 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py @@ -0,0 +1,109 @@ +"""A fake `azd` that reproduces azd 1.29.0's observable command contract. + +Recorded from the real CLI (`azd version 1.29.0`) on 2026-08-14: + +``` +$ azd env get-value MISSING # inside the project, no environment +rc=1, stdout="\nERROR: ensuring environment exists: environment not specified" + +$ azd env get-value AZURE_LOCATION # outside any azd project +rc=1, stdout="\nERROR: no project exists; to create a new project, run `azd init`" + +$ azd env get-value AZURE_LOCATION --cwd # from any cwd +rc=1, stdout="\nERROR: ensuring environment exists: environment not specified" +``` + +Two properties matter for `common.sh` and are therefore modelled here: + +1. azd writes its `ERROR:` diagnostics to **stdout**, not stderr, and signals + failure only through the exit status. A caller that keeps stdout when the + command failed silently adopts the error text as a configuration value. +2. azd resolves the project from `--cwd` when given, otherwise from the + process working directory, and fails when that directory holds no + `azure.yaml`. A lookup that does not pin the project root breaks as soon + as a script is invoked from the repository root or any other directory. + +`MISSING_KEY_MODES` exposes both observed missing-value shapes so tests can +prove the reader is driven by the exit status rather than by stdout text. +""" +import stat + + +NO_PROJECT_ERROR = ( + "\nERROR: no project exists; to create a new project, run `azd init`" +) +NO_ENVIRONMENT_ERROR = ( + "\nERROR: ensuring environment exists: environment not specified" +) + +# How the fake reports a value it does not have. +# "azd_1_29" -- what the real CLI does: ERROR text on stdout, exit 1. +# "silent" -- nothing on stdout, exit 1 (the shape the first version of +# this harness assumed). +MISSING_KEY_MODES = ("azd_1_29", "silent") + + +def _missing_key_branch(missing_key_mode): + if missing_key_mode not in MISSING_KEY_MODES: + raise ValueError(f"unknown missing_key_mode: {missing_key_mode}") + if missing_key_mode == "silent": + return " exit 1" + return f" printf '%s\\n' '{NO_ENVIRONMENT_ERROR.lstrip(chr(10))}'\n exit 1" + + +def azd_stub_source(azd_values, missing_key_mode="azd_1_29", log_path=None): + """Bash source for a fake `azd` honouring the contract described above.""" + lines = [ + "#!/usr/bin/env bash", + "project_dir=\"${PWD}\"", + "argv=()", + "while [[ \"$#\" -gt 0 ]]; do", + " case \"$1\" in", + " --cwd|-C) project_dir=\"$2\"; shift 2 ;;", + " --cwd=*) project_dir=\"${1#--cwd=}\"; shift ;;", + " *) argv+=(\"$1\"); shift ;;", + " esac", + "done", + ] + if log_path is not None: + lines.append(f'printf \'%s\\n\' "${{argv[*]:-}}" >> "{log_path}"') + lines.append(f'printf \'cwd=%s\\n\' "${{project_dir}}" >> "{log_path}"') + lines += [ + 'if [[ "${argv[0]:-}" == "auth" ]]; then', + " exit 0", + "fi", + '# Only the `env` commands need an azd project; `auth` does not.', + 'if [[ "${argv[0]:-}" == "env" && ! -f "${project_dir}/azure.yaml" ]]; then', + f" printf '%s\\n' '{NO_PROJECT_ERROR.lstrip(chr(10))}'", + " exit 1", + "fi", + 'if [[ "${argv[0]:-}" == "env" && "${argv[1]:-}" == "get-value" ]]; then', + ' case "${argv[2]:-}" in', + ] + for name, value in azd_values.items(): + escaped = str(value).replace("'", "'\\''") + lines.append(f" {name}) printf '%s\\n' '{escaped}'; exit 0 ;;") + lines.append(" *)") + lines.append(_missing_key_branch(missing_key_mode)) + lines.append(" ;;") + lines.append(" esac") + lines.append("fi") + lines.append('if [[ "${argv[0]:-}" == "env" && "${argv[1]:-}" == "set" ]]; then') + lines.append(" exit 0") + lines.append("fi") + lines.append('printf \'azd stub: unsupported invocation: %s\\n\' "${argv[*]:-}" >&2') + lines.append("exit 1") + return "\n".join(lines) + "\n" + + +def write_executable(path, content): + path.write_text(content) + path.chmod(path.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + return path + + +def write_azd_stub(bin_dir, azd_values, missing_key_mode="azd_1_29", log_path=None): + return write_executable( + bin_dir / "azd", + azd_stub_source(azd_values, missing_key_mode, log_path), + ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py new file mode 100644 index 0000000..8829d70 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -0,0 +1,257 @@ +"""Harness that runs the lab's shell entry points end to end. + +The scenario/query/capture/cleanup scripts are the only place where +`common.sh`'s configuration contract, its `readonly` declarations and each +script's own variables meet, so they are exercised as *programs* here: a +throwaway copy of the lab (its own `azure.yaml`, `evidence/`, `assets/`) +plus fake `az`, `azd` and `python` executables on PATH. Every run starts +from a scratch working directory that is not the lab, which is how the +scripts are invoked in practice (from the repository root or anywhere else). +""" +import json +import os +import shutil +import subprocess +from pathlib import Path + +from azd_fake import write_azd_stub, write_executable + + +SCRIPTS_DIR = Path(__file__).parents[1] +BASH = shutil.which("bash") or "/bin/bash" + +SUBSCRIPTION_ID = "11111111-2222-3333-4444-555555555555" +RESOURCE_GROUP = "rg-sre-lab-exec" +ENV_NAME = "sre-lab-exec" +MONITORING_CONTRIBUTOR_ROLE_ID = "749f88d5-cbae-40b8-bcfc-e573ddc772fa" + +AZD_VALUES = { + "AZURE_SUBSCRIPTION_ID": SUBSCRIPTION_ID, + "AZURE_RESOURCE_GROUP": RESOURCE_GROUP, + "AZURE_ENV_NAME": ENV_NAME, + "AZURE_LOCATION": "koreacentral", + "AZURE_CONTAINER_APP_NAME": "ca-sre-lab", + "AZURE_CONTAINER_APP_FQDN": "ca-sre-lab.example.azurecontainerapps.io", + "AZURE_STORAGE_CONTAINER_SCOPE": ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + "/providers/Microsoft.Storage/storageAccounts/stsrelab/blobServices/default" + "/containers/documents" + ), + "AZURE_BLOB_ROLE_ASSIGNMENT_NAME": "3f2504e0-4f89-11d3-9a0c-0305e82c3301", + "AZURE_WORKSPACE_ID": ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + "/providers/Microsoft.OperationalInsights/workspaces/log-sre-lab" + ), + "AZURE_APP_INSIGHTS_NAME": "appi-sre-lab", + "AZURE_TELEMETRY_SERVICE_NAME": "sre-lab-order-api", + "containerAppPrincipalId": "8c8a4f0e-0000-4000-8000-2b1f9a0c1234", + "workspaceCustomerId": "9d1a0b2c-3d4e-5f60-7182-93a4b5c6d7e8", +} + +ALERTS_JSON = json.dumps( + { + "value": [ + { + "id": ( + f"/subscriptions/{SUBSCRIPTION_ID}/providers" + "/Microsoft.AlertsManagement/alerts/aaaa0000-1111-2222-3333-444455556666" + ), + "properties": { + "essentials": { + "alertRule": ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + "/providers/microsoft.insights/metricAlerts/" + "alert-sre-lab-s1-http500" + ), + "startDateTime": "2026-08-14T00:05:00Z", + "monitorCondition": "Fired", + } + }, + } + ] + } +) + + +def _az_stub_source(log_path, state_dir): + """A fake `az` that answers every call the lab scripts make. + + Revision names advance on `containerapp update` so + `wait_for_new_revision_ready` observes a genuinely new revision instead + of spinning on its ten-minute timeout. + """ + return f"""#!/usr/bin/env bash +printf '%s\\t%s\\n' "$*" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" >> "{log_path}" +state="{state_dir}" +case "${{1:-}} ${{2:-}}" in + "account show") + printf '%s\\n' "{SUBSCRIPTION_ID}" ;; + "group exists") + printf 'true\\n' ;; + "group show") + if [[ "$*" == *"azd-env-name"* ]]; then + printf '%s\\n' "{ENV_NAME}" + else + printf 'sre-agent-event-lab\\n' + fi ;; + "group delete") + : ;; + "containerapp show") + printf 'rev-%s\\n' "$(cat "${{state}}/revision")" ;; + "containerapp update") + printf '%s\\n' "$(( $(cat "${{state}}/revision") + 1 ))" > "${{state}}/revision" ;; + "containerapp revision") + if [[ "$*" == *"healthState"* ]]; then + printf 'Healthy\\n' + elif [[ "$*" == *".active"* ]]; then + printf 'true\\n' + else + printf '[]\\n' + fi ;; + "monitor log-analytics") + printf '{{"tables": []}}\\n' ;; + "monitor activity-log") + printf '[]\\n' ;; + "role assignment") + if [[ "${{3:-}}" == "list" && "$*" != *"-o tsv"* ]]; then + printf '[]\\n' + fi ;; + "rest --method") + if [[ "$*" == *"/roleAssignments/"* ]]; then + assignment_id="${{*##*--url }}" + assignment_id="${{assignment_id%%\\?*}}" + principal_id="${{assignment_id##*/}}" + printf '{{"properties": {{"principalId": "%s", "roleDefinitionId": "/subscriptions/{SUBSCRIPTION_ID}/providers/Microsoft.Authorization/roleDefinitions/{MONITORING_CONTRIBUTOR_ROLE_ID}", "scope": "/subscriptions/{SUBSCRIPTION_ID}"}}}}\\n' \\ + "${{principal_id}}" + else + printf '%s\\n' '{ALERTS_JSON}' + fi ;; + *) + : ;; +esac +exit 0 +""" + + +def _lab_python_stub_source(log_path): + """A fake `${LAB_ROOT}/app/.venv/bin/python` for capture-scenario.sh.""" + return f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +script="${{1:-}}" +shift || true +output_dir="" +asset_dir="" +normalized="" +case "${{script}}" in + *capture_agent.py) + while [[ "$#" -gt 0 ]]; do + case "$1" in + --output-dir) output_dir="$2"; shift 2 ;; + *) shift ;; + esac + done + printf '%s\\n' '[{{"state": "detected"}}, {{"state": "diagnosing"}}, {{"state": "root-caused"}}, {{"state": "resolved"}}]' \\ + > "${{output_dir}}/normalized-timeline.json" + ;; + *render_capture.py) + normalized="${{1:-}}" + asset_dir="${{2:-}}" + [[ -f "${{normalized}}" ]] || exit 1 + mkdir -p "${{asset_dir}}" + printf 'GIF89a' > "${{asset_dir}}/investigation.gif" + printf 'timeline\\n' > "${{asset_dir}}/timeline.mmd" + ;; +esac +exit 0 +""" + + +def make_lab(tmp_path, azd_values=None, missing_key_mode="azd_1_29"): + """A throwaway copy of the lab plus fake CLIs; returns a run context.""" + lab = tmp_path / "lab" + shutil.copytree( + SCRIPTS_DIR, + lab / "scripts", + ignore=shutil.ignore_patterns("tests", "__pycache__"), + ) + (lab / "azure.yaml").write_text("name: sre-agent-event-lab\n") + (lab / "evidence").mkdir() + + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + state_dir = tmp_path / "state" + state_dir.mkdir() + (state_dir / "revision").write_text("1\n") + + az_log = tmp_path / "az-calls.log" + azd_log = tmp_path / "azd-calls.log" + python_log = tmp_path / "python-calls.log" + lab_python_log = tmp_path / "lab-python-calls.log" + + write_executable(bin_dir / "az", _az_stub_source(az_log, state_dir)) + write_azd_stub( + bin_dir, + AZD_VALUES if azd_values is None else azd_values, + missing_key_mode, + azd_log, + ) + write_executable( + bin_dir / "python3", + f'#!/usr/bin/env bash\nprintf \'%s\\n\' "$*" >> "{python_log}"\nexit 0\n', + ) + + venv_bin = lab / "app" / ".venv" / "bin" + venv_bin.mkdir(parents=True) + write_executable(venv_bin / "python", _lab_python_stub_source(lab_python_log)) + + workdir = tmp_path / "elsewhere" + workdir.mkdir() + + return LabRun(lab, bin_dir, workdir, az_log, azd_log, lab_python_log) + + +class LabRun: + def __init__(self, lab, bin_dir, workdir, az_log, azd_log, lab_python_log): + self.lab = lab + self.bin_dir = bin_dir + self.workdir = workdir + self.az_log = az_log + self.azd_log = azd_log + self.lab_python_log = lab_python_log + + def run(self, script_name, args=(), env=None): + process_env = { + "PATH": f"{self.bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", + "HOME": os.environ.get("HOME", str(self.lab)), + } + process_env.update(env or {}) + return subprocess.run( + [BASH, str(self.lab / "scripts" / script_name), *args], + capture_output=True, + text=True, + env=process_env, + cwd=str(self.workdir), + ) + + def az_calls(self): + return self.az_log.read_text() if self.az_log.exists() else "" + + def azd_calls(self): + return self.azd_log.read_text() if self.azd_log.exists() else "" + + def write_agent_setup(self, principal_ids=("principal-one", "principal-two")): + assignment_ids = [ + f"/subscriptions/{SUBSCRIPTION_ID}/providers" + f"/Microsoft.Authorization/roleAssignments/{principal_id}" + for principal_id in principal_ids + ] + setup = { + "agent_endpoint": "https://sre-agent.example.com/api/incidents", + "monitoring_contributor_assignment_id": assignment_ids[0], + "agent_principal_id": principal_ids[0], + "uami_monitoring_contributor_assignment_id": assignment_ids[1], + "agent_user_assigned_principal_id": principal_ids[1], + } + path = self.lab / "evidence" / "agent-setup.json" + path.write_text(json.dumps(setup)) + return path diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py index f32c4b7..d42e7af 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_env.py @@ -3,10 +3,16 @@ `load_lab_config` must resolve every setting as: explicit process environment > current `azd env get-value` > an allowed default -- and must never fall back to a fixed subscription/resource group again. These tests -drive `common.sh` through a fake `azd`/`az` on PATH (see -`azd_common_harness.py`); no real Azure CLI or azd call is made. +drive `common.sh` through a fake `azd`/`az` on PATH (see `azd_fake.py` for +the recorded azd 1.29.0 contract the fake reproduces); no real Azure CLI or +azd call is made. """ -from azd_common_harness import COMMON_SH, run_common +import re + +import pytest + +from azd_common_harness import COMMON_SH, LAB_ROOT, run_common +from azd_fake import MISSING_KEY_MODES REQUIRED_ENV = { @@ -45,28 +51,63 @@ def test_current_azd_value_wins_over_default_when_no_explicit_env(tmp_path): assert result.stdout == "eastus" -def test_allowed_default_applies_when_neither_explicit_env_nor_azd_value_exist(tmp_path): +@pytest.mark.parametrize("missing_key_mode", MISSING_KEY_MODES) +def test_allowed_default_applies_when_neither_explicit_env_nor_azd_value_exist( + tmp_path, missing_key_mode +): + """The default must win for a value azd does not have -- whatever azd + printed on stdout while failing. + + azd 1.29.0 answers an unknown key with `ERROR: ...` on **stdout** and a + non-zero exit status. Reading stdout unconditionally turns that error + text into the setting's value, so `LOCATION` would become an `ERROR:` + sentence instead of the documented default. + """ result = run_common( tmp_path, env=dict(REQUIRED_ENV), azd_values=dict(REQUIRED_ENV), command='load_lab_config; printf "%s" "${LOCATION}"', + missing_key_mode=missing_key_mode, ) assert result.returncode == 0, result.stderr assert result.stdout == "koreacentral" + assert "ERROR" not in result.stdout -def test_missing_required_setting_names_the_azd_command(tmp_path): +@pytest.mark.parametrize("missing_key_mode", MISSING_KEY_MODES) +def test_missing_required_setting_names_the_azd_command(tmp_path, missing_key_mode): + """A required setting azd does not have must fail closed, even though + azd's own failure output arrives on stdout looking like a value.""" result = run_common( tmp_path, env={}, azd_values={}, command="load_lab_config", + missing_key_mode=missing_key_mode, ) assert result.returncode != 0 assert "azd env set AZURE_SUBSCRIPTION_ID" in result.stderr +@pytest.mark.parametrize("missing_key_mode", MISSING_KEY_MODES) +def test_missing_azd_value_never_becomes_a_resolved_setting(tmp_path, missing_key_mode): + """Nothing azd printed while failing may reach a resolved value.""" + result = run_common( + tmp_path, + env=dict(REQUIRED_ENV), + azd_values=dict(REQUIRED_ENV), + command=( + "load_lab_config; " + 'printf "%s|%s|%s" "${AZURE_CONTAINER_APP_NAME}" ' + '"${SRE_AGENT_NAME}" "${SRE_LAB_EXPIRES_ON}"' + ), + missing_key_mode=missing_key_mode, + ) + assert result.returncode == 0, result.stderr + assert result.stdout == "||" + + def test_missing_resource_group_names_the_azd_command_once_subscription_resolves(tmp_path): result = run_common( tmp_path, @@ -78,6 +119,53 @@ def test_missing_resource_group_names_the_azd_command_once_subscription_resolves assert "azd env set AZURE_RESOURCE_GROUP" in result.stderr +def test_azd_lookup_pins_the_lab_project_root_from_any_working_directory(tmp_path): + """azd resolves its project from `--cwd`, else from the process working + directory, and fails when that directory has no `azure.yaml`. + + The lab scripts are routinely started from the repository root (or any + other directory), so every lookup has to pin the lab project root. The + fake `azd` refuses a project directory without `azure.yaml` exactly as + the real one does, and this run happens from a scratch directory that + has none. + """ + workdir = tmp_path / "somewhere-else" + workdir.mkdir() + azd_log = tmp_path / "azd-calls.log" + + result = run_common( + tmp_path, + env={k: v for k, v in REQUIRED_ENV.items() if k != "AZURE_SUBSCRIPTION_ID"}, + azd_values={**REQUIRED_ENV, "AZURE_LOCATION": "eastus"}, + command=( + 'load_lab_config; printf "%s|%s" "${SUBSCRIPTION_ID}" "${LOCATION}"' + ), + cwd=workdir, + azd_log=azd_log, + ) + + assert result.returncode == 0, result.stderr + assert result.stdout == f"{REQUIRED_ENV['AZURE_SUBSCRIPTION_ID']}|eastus" + calls = azd_log.read_text() + assert f"cwd={LAB_ROOT}" in calls, ( + "every azd lookup must run against the lab project root, not the " + f"caller's working directory: {calls!r}" + ) + + +def test_azd_lookup_from_the_repository_root_still_resolves_settings(tmp_path): + """The documented entry points are run from the repository root.""" + result = run_common( + tmp_path, + env={}, + azd_values={**REQUIRED_ENV, "AZURE_LOCATION": "westus2"}, + command='load_lab_config; printf "%s|%s" "${RESOURCE_GROUP}" "${LOCATION}"', + cwd=LAB_ROOT.parents[1], + ) + assert result.returncode == 0, result.stderr + assert result.stdout == f"{REQUIRED_ENV['AZURE_RESOURCE_GROUP']}|westus2" + + def test_load_lab_config_resolves_sre_agent_settings_with_documented_defaults(tmp_path): result = run_common( tmp_path, @@ -113,13 +201,39 @@ def test_load_lab_config_does_not_fix_subscription_or_resource_group_values(tmp_ assert result.stdout == "22222222-3333-4444-5555-666666666666|rg-some-other-lab" -def test_setting_helper_does_not_use_eval_or_indirection(): - script = COMMON_SH.read_text() +def test_a_stored_setting_value_is_never_executed_as_shell_code(tmp_path): + """Values arrive from an azd environment file an operator can edit, so + resolution must treat them as data: no `eval`, no re-expansion. + """ + marker = tmp_path / "pwned" + injected = f'$(touch "{marker}")`touch "{marker}"`' + result = run_common( + tmp_path, + env=dict(REQUIRED_ENV), + azd_values={**REQUIRED_ENV, "AZURE_LOCATION": injected}, + command='load_lab_config; printf "%s" "${LOCATION}"', + ) + + assert result.returncode == 0, result.stderr + assert result.stdout == injected + assert not marker.exists(), "a stored setting value was executed as shell code" - assert "eval" not in script - assert "declare -n" not in script - # No `${!name}` indirect expansion anywhere in the setting/azd_value path. - assert "${!" not in script + +def test_setting_helper_does_not_use_eval_or_variable_indirection(): + """Token-aware guard: `eval` as a command, `declare -n`, and `${!name}` + indirection are all forbidden. Substring matching would fire on the word + "evaluate" in a comment and miss `eval` in code, so comments are stripped + and word boundaries are required. + """ + code_lines = [ + line for line in COMMON_SH.read_text().splitlines() + if not line.lstrip().startswith("#") + ] + code = "\n".join(code_lines) + + assert not re.search(r"(? azd env > default) instead. assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in script - assert 'SUBSCRIPTION_ID="$(require_setting AZURE_SUBSCRIPTION_ID' in script - # The subscription ID equality check is the real safety boundary and - # must remain intact. - assert '"${current_subscription}" != "${SUBSCRIPTION_ID}"' in script def test_verify_subscription_reports_only_subscription_id_on_mismatch(tmp_path): @@ -249,36 +284,22 @@ def test_cleanup_deletion_loop_tolerates_empty_role_assignments_on_bash32(tmp_pa assert "az group delete" in calls -def test_scenario_query_capture_cleanup_scripts_load_lab_config_before_use(): - """Every script that reads SUBSCRIPTION_ID/RESOURCE_GROUP/deployment_output - must resolve them via `load_lab_config` (through `require_lab_config`) - first, and must do so before `verify_subscription`/`verify_lab_resource_group` - or any deployment_output() call that depends on those values. +def test_scenario_query_capture_cleanup_scripts_are_exercised_as_programs(): + """The four entry points are covered by execution tests, not by reading + their text: `test_lab_scripts.py` runs each one against fake + `az`/`azd`/`python` executables from a working directory outside the + lab, which is the only way to catch a caller that reassigns a name + `common.sh` already made readonly. """ - scripts_dir = Path(__file__).parents[1] + lab_script_tests = (Path(__file__).parent / "test_lab_scripts.py").read_text() + for script_name in ( "run-scenario.sh", "query-evidence.sh", "capture-scenario.sh", "cleanup.sh", ): - script = (scripts_dir / script_name).read_text() - assert "require_lab_config" in script, ( - f"{script_name} must call require_lab_config (which calls " - "load_lab_config) before using SUBSCRIPTION_ID/RESOURCE_GROUP" + assert f'"{script_name}"' in lab_script_tests, ( + f"{script_name} has no execution test" ) - config_index = script.index("require_lab_config") - - for later_use in ("verify_subscription", "verify_lab_resource_group"): - if later_use in script: - assert config_index < script.index(later_use), ( - f"{script_name} must call require_lab_config before {later_use}" - ) - - if "deployment_output " in script: - first_output_call = script.index("deployment_output ") - assert config_index < first_output_call, ( - f"{script_name} must call require_lab_config before reading " - "any deployment_output() value" - ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py new file mode 100644 index 0000000..057ab4e --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -0,0 +1,245 @@ +"""Execution tests for the lab's four shell entry points. + +Each script is run as a program against fake `az`/`azd`/`python` +executables, from a working directory that is not the lab. Running them +proves what reading their text cannot: that configuration actually loads, +that no variable a script assigns collides with a `readonly` name +`common.sh` already declared, and that the safety checks run before any +Azure operation. +""" +import json + +import pytest + +from lab_script_harness import ENV_NAME, RESOURCE_GROUP, SUBSCRIPTION_ID, make_lab + + +CALLERS = ("run-scenario.sh", "query-evidence.sh", "capture-scenario.sh", "cleanup.sh") + + +def _assert_loaded_config(result, lab_run): + """Every caller must get past `require_lab_config` + the safety checks.""" + assert "readonly variable" not in result.stderr, ( + "a script assigned a name common.sh already made readonly: " + f"{result.stderr!r}" + ) + assert "azd env set" not in result.stderr, ( + f"configuration failed to load: {result.stderr!r}" + ) + az_calls = lab_run.az_calls() + assert "account show" in az_calls, ( + f"verify_subscription never ran: {az_calls!r} / {result.stderr!r}" + ) + assert 'tags."azd-env-name"' in az_calls, ( + f"verify_lab_resource_group never ran: {az_calls!r}" + ) + assert f"cwd={lab_run.lab}" in lab_run.azd_calls(), ( + "azd lookups must be pinned to the lab project root: " + f"{lab_run.azd_calls()!r}" + ) + + +def test_run_scenario_s1_runs_to_completion_from_another_directory(tmp_path): + lab_run = make_lab(tmp_path) + + result = lab_run.run("run-scenario.sh", ["s1"]) + + _assert_loaded_config(result, lab_run) + assert result.returncode == 0, result.stderr + evidence_dirs = sorted((lab_run.lab / "evidence").glob("s1-*")) + assert evidence_dirs, "no evidence directory was created" + timeline = json.loads((evidence_dirs[-1] / "timeline.json").read_text()) + assert timeline["scenario"] == "s1" + assert timeline["injected_at"] + assert timeline["alert_id"] + assert timeline["recovered_at"] + assert "FAILURE_MODE=none" in lab_run.az_calls(), "the scenario never recovered" + + # The evidence is only usable if the recorded moments really bracket the + # Azure operations they claim to describe. + update_times = [ + line.split("\t")[1] + for line in lab_run.az_calls().splitlines() + if "containerapp update" in line + ] + assert update_times, "the failure was never injected" + assert timeline["injected_at"] <= update_times[0], ( + "injected_at must be recorded before the Container App is updated" + ) + assert update_times[0] <= timeline["revision_ready_at"] + assert timeline["revision_ready_at"] <= timeline["recovered_at"] + + +def test_query_evidence_collects_every_artifact_from_another_directory(tmp_path): + lab_run = make_lab(tmp_path) + evidence_dir = tmp_path / "evidence-out" + + result = lab_run.run( + "query-evidence.sh", + ["s1", str(evidence_dir), "2026-08-14T00:00:00Z", "2026-08-14T01:00:00Z"], + ) + + _assert_loaded_config(result, lab_run) + assert result.returncode == 0, result.stderr + for artifact in ( + "app-requests.json", + "app-dependencies.json", + "app-exceptions.json", + "activity-log.json", + "alerts.json", + "revisions-redacted.json", + "storage-role-assignments.json", + "query-window.json", + ): + assert (evidence_dir / artifact).is_file(), f"missing {artifact}" + window = json.loads((evidence_dir / "query-window.json").read_text()) + assert window["scenario"] == "s1" + + +def test_query_evidence_queries_the_resolved_workspace_and_principal(tmp_path): + """The values that used to collide with `common.sh`'s readonly names + (the workspace customer ID and the workload principal ID) must reach the + Azure CLI calls that consume them.""" + lab_run = make_lab(tmp_path) + evidence_dir = tmp_path / "evidence-out" + + result = lab_run.run( + "query-evidence.sh", + ["s1", str(evidence_dir), "2026-08-14T00:00:00Z", "2026-08-14T01:00:00Z"], + ) + + assert result.returncode == 0, result.stderr + az_calls = lab_run.az_calls() + assert "--workspace 9d1a0b2c-3d4e-5f60-7182-93a4b5c6d7e8" in az_calls + assert "--assignee-object-id 8c8a4f0e-0000-4000-8000-2b1f9a0c1234" in az_calls + + +def test_capture_scenario_renders_from_another_directory(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + evidence_dir = tmp_path / "evidence-out" + evidence_dir.mkdir() + (evidence_dir / "timeline.json").write_text( + json.dumps({"scenario": "s1", "alert_id": "/alerts/aaaa0000"}) + ) + + result = lab_run.run("capture-scenario.sh", ["s1", str(evidence_dir)]) + + _assert_loaded_config(result, lab_run) + assert result.returncode == 0, result.stderr + assert (evidence_dir / "normalized-timeline.json").is_file() + assert (lab_run.lab / "assets" / "captures" / "s1" / "investigation.gif").is_file() + + +def test_cleanup_dry_run_plans_without_deleting_from_another_directory(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run("cleanup.sh") + + _assert_loaded_config(result, lab_run) + assert result.returncode == 0, result.stderr + assert "Planned cleanup:" in result.stdout + assert f"Delete tagged resource group: {RESOURCE_GROUP}" in result.stdout + assert "Dry run only" in result.stdout + az_calls = lab_run.az_calls() + assert "group delete" not in az_calls, "a dry run must delete nothing" + assert "role assignment delete" not in az_calls + + +def test_cleanup_deletes_only_after_confirmation(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run("cleanup.sh", ["--yes"]) + + assert result.returncode == 0, result.stderr + az_calls = lab_run.az_calls() + assert f"group delete --name {RESOURCE_GROUP} --yes --no-wait" in az_calls + assert "role assignment delete --ids /subscriptions/" in az_calls + + +@pytest.mark.parametrize("script_name", CALLERS) +def test_every_caller_fails_closed_when_configuration_is_missing(script_name, tmp_path): + """No azd value and no explicit environment: every entry point must stop + with the actionable `azd env set` message before touching Azure.""" + lab_run = make_lab(tmp_path, azd_values={}) + lab_run.write_agent_setup() + arguments = { + "run-scenario.sh": ["s1"], + "query-evidence.sh": [ + "s1", + str(tmp_path / "out"), + "2026-08-14T00:00:00Z", + "2026-08-14T01:00:00Z", + ], + "capture-scenario.sh": ["s1", str(tmp_path / "out")], + "cleanup.sh": [], + }[script_name] + + result = lab_run.run(script_name, arguments) + + assert result.returncode != 0 + assert "azd env set AZURE_SUBSCRIPTION_ID" in result.stderr + assert "group delete" not in lab_run.az_calls() + assert "containerapp update" not in lab_run.az_calls() + + +@pytest.mark.parametrize("script_name", CALLERS) +def test_every_caller_refuses_a_foreign_subscription(script_name, tmp_path): + """The subscription-equality boundary must hold for every entry point.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + arguments = { + "run-scenario.sh": ["s1"], + "query-evidence.sh": [ + "s1", + str(tmp_path / "out"), + "2026-08-14T00:00:00Z", + "2026-08-14T01:00:00Z", + ], + "capture-scenario.sh": ["s1", str(tmp_path / "out")], + "cleanup.sh": ["--yes"], + }[script_name] + + result = lab_run.run( + script_name, + arguments, + env={"AZURE_SUBSCRIPTION_ID": "99999999-9999-9999-9999-999999999999"}, + ) + + assert result.returncode != 0 + assert "Refusing to continue in subscription" in result.stderr + assert "Expected 99999999" in result.stderr + az_calls = lab_run.az_calls() + assert "group delete" not in az_calls + assert "containerapp update" not in az_calls + assert "role assignment delete" not in az_calls + + +def test_environment_name_tag_mismatch_stops_every_caller(tmp_path): + """A resource group tagged for another azd environment is refused even + when its purpose tag matches.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run( + "cleanup.sh", + ["--yes"], + env={"AZURE_ENV_NAME": f"{ENV_NAME}-other"}, + ) + + assert result.returncode != 0 + assert "Refusing to operate on untagged resource group" in result.stderr + assert "group delete" not in lab_run.az_calls() + + +def test_subscription_id_is_read_from_the_azd_environment(tmp_path): + """Nothing in the callers may pin a subscription of its own.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run("cleanup.sh") + + assert result.returncode == 0, result.stderr + assert SUBSCRIPTION_ID in lab_run.az_calls() From 60bcf1b748be1f9b8bcc73811f138b780ba41dbc Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 15:37:00 +0900 Subject: [PATCH 06/26] feat(sre-lab): add guided doctor and baseline commands - Add scripts/lab.sh as a single-command entry point dispatching to doctor|baseline|acknowledge agent-setup|run s1|s2|s3|capture s1|s2|s3|score. acknowledge/score depend on a later task's lab_state.py/score.py; until those land, lab.sh fails closed with a clear "not yet available" message (exit 3) instead of a raw file-not-found error. - Add scripts/doctor.sh: a 15-row environment diagnostic printing the CHECKSTATUSDETAIL contract (PASS/FAIL/MANUAL). Verifies required commands, Azure CLI login, azd configuration, subscription and resource-group pinning, Container App health, /healthz, Application Insights telemetry, alert rule enablement, the SRE Agent resource (when configured), and Reader role assignment -- all through official stable Azure CLI/REST reads. Fails closed: once any prerequisite check fails, every subsequent Azure-touching check is blocked rather than attempting further calls. Repository connection, knowledge source, incident platform, and response plan are always MANUAL since no stable API exposes them. - Add scripts/baseline.sh: runs baseline load (30 orders + 10 documents requests) and verifies Application Insights shows telemetry for both endpoints via bounded polling (default 600s timeout / 20s interval, overridable via LAB_BASELINE_TELEMETRY_TIMEOUT_SECONDS/ LAB_BASELINE_TELEMETRY_POLL_INTERVAL_SECONDS). Writes evidence under evidence/baseline-/. - Fix scripts/capture-scenario.sh executable bit (100644 -> 100755); it was the only lab script not marked executable, and lab.sh capture depends on invoking it directly. - Add scripts/tests/doctor_harness.py: a fake-CLI harness (real az/azd/ curl/python3 stubs on PATH) purpose-built for doctor.sh/baseline.sh/ lab.sh's call surface (container app health, /healthz, per-rule alert reads, per-principal role assignment lookups), mirroring the existing lab_script_harness.py pattern. - Add scripts/tests/test_doctor.py and scripts/tests/test_lab_cli.py: behavioural tests driving doctor.sh/lab.sh as real programs against the fake CLIs (not text-only assertions), covering the fully-healthy path, every individual FAIL condition, fail-closed gating once subscription/ azd configuration checks fail, azd --cwd pinning, and lab.sh's dispatch/auto-discovery/not-yet-available behavior. - Document the new lab.sh entry point and doctor check list in README.md. 173 tests pass (144 pre-existing + 29 new); bash -n and az bicep build verified. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 19 + .../sre-agent-event-lab/scripts/baseline.sh | 95 +++++ .../scripts/capture-scenario.sh | 0 monitor/sre-agent-event-lab/scripts/doctor.sh | 240 +++++++++++ monitor/sre-agent-event-lab/scripts/lab.sh | 85 ++++ .../scripts/tests/doctor_harness.py | 391 ++++++++++++++++++ .../scripts/tests/test_doctor.py | 194 +++++++++ .../scripts/tests/test_lab_cli.py | 176 ++++++++ 8 files changed, 1200 insertions(+) create mode 100755 monitor/sre-agent-event-lab/scripts/baseline.sh mode change 100644 => 100755 monitor/sre-agent-event-lab/scripts/capture-scenario.sh create mode 100755 monitor/sre-agent-event-lab/scripts/doctor.sh create mode 100755 monitor/sre-agent-event-lab/scripts/lab.sh create mode 100644 monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_doctor.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index b37d06e..7260f2c 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -39,6 +39,25 @@ Azure SRE Agent Korea Central이 구독에 표시되지 않으면 [공식 regist - Agent response plan은 모두 `Review` 모드로 구성한다. - evidence에는 secret, connection string, access token을 저장하지 않는다. +## 안내형 단일 진입점: `lab.sh` + +매 단계 azd output과 명령을 직접 조합하는 대신, 아래 단일 명령으로 환경 점검부터 채점까지 안내받을 수 있다. 각 하위 명령은 이 문서의 해당 절이 설명하는 스크립트를 그대로 호출하므로 동작은 동일하다. + +```bash +monitor/sre-agent-event-lab/scripts/lab.sh doctor # 환경 점검 (아래 참고) +monitor/sre-agent-event-lab/scripts/lab.sh baseline # Baseline 부하 및 telemetry 확인 +monitor/sre-agent-event-lab/scripts/lab.sh run s1|s2|s3 # 시나리오 실행 (run-scenario.sh와 동일) +monitor/sre-agent-event-lab/scripts/lab.sh capture s1|s2|s3 # 최신 evidence 디렉터리를 자동 탐색해 캡처 +monitor/sre-agent-event-lab/scripts/lab.sh acknowledge agent-setup # (예정) Agent 설정 수기 확인 기록 +monitor/sre-agent-event-lab/scripts/lab.sh score # (예정) 수집한 evidence 채점 +``` + +`acknowledge`와 `score`는 향후 작업에서 추가될 `lab_state.py`/`score.py`에 의존한다. 아직 해당 파일이 없으면 원인을 알 수 없는 오류 대신 "not yet available" 메시지와 함께 종료 코드 3으로 종료한다. + +### `doctor` 점검 항목 + +`lab.sh doctor`는 `CHECKSTATUSDETAIL` 형식으로 한 줄에 하나씩 점검 결과를 출력한다. `STATUS`는 `PASS`, `FAIL`, `MANUAL` 중 하나이며, `FAIL`이 하나라도 있으면 종료 코드 1을 반환한다. 필수 명령, 로그인, azd 구성, 구독/리소스 그룹, Container App 상태, `/healthz`, Application Insights telemetry, alert rule 활성화, SRE Agent 리소스(설정된 경우), Reader 역할 할당을 공식 안정 API로 검증한다. Repository connection, Knowledge source, Incident platform, Response plan은 공식 API로 확인할 수 없으므로 항상 `MANUAL`로 표시되며 portal에서 직접 확인해야 한다. + ## 로컬 검증 ```bash diff --git a/monitor/sre-agent-event-lab/scripts/baseline.sh b/monitor/sre-agent-event-lab/scripts/baseline.sh new file mode 100755 index 0000000..de2a49b --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/baseline.sh @@ -0,0 +1,95 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +source "${SCRIPT_DIR}/common.sh" + +# Overridable only for tests: production runs always use the 600s/20s +# defaults. Bounding the poll (rather than querying once) tolerates +# Application Insights ingestion lag without hanging forever. +readonly TELEMETRY_TIMEOUT_SECONDS="${LAB_BASELINE_TELEMETRY_TIMEOUT_SECONDS:-600}" +readonly TELEMETRY_POLL_INTERVAL_SECONDS="${LAB_BASELINE_TELEMETRY_POLL_INTERVAL_SECONDS:-20}" + +require_lab_config +verify_subscription +verify_lab_resource_group + +APP_FQDN="$(deployment_output containerAppFqdn)" +WORKSPACE_CUSTOMER_ID="$(deployment_output workspaceCustomerId)" +TELEMETRY_SERVICE_NAME="$(deployment_output telemetryServiceName)" +readonly APP_FQDN WORKSPACE_CUSTOMER_ID TELEMETRY_SERVICE_NAME + +if [[ -z "${APP_FQDN}" || -z "${WORKSPACE_CUSTOMER_ID}" || -z "${TELEMETRY_SERVICE_NAME}" ]]; then + echo "Deployment outputs are missing. Run: azd provision" >&2 + exit 1 +fi + +EVIDENCE_DIR="$(create_evidence_dir baseline)" +readonly EVIDENCE_DIR + +if ! python3 "${SCRIPT_DIR}/loadgen.py" \ + "https://${APP_FQDN}/api/orders" \ + --requests 30 \ + --concurrency 4 \ + --expect-status 200 \ + --output "${EVIDENCE_DIR}/orders.json"; then + echo "Baseline /api/orders requests did not all succeed: ${EVIDENCE_DIR}/orders.json" >&2 + exit 1 +fi + +if ! python3 "${SCRIPT_DIR}/loadgen.py" \ + "https://${APP_FQDN}/api/documents" \ + --requests 10 \ + --concurrency 2 \ + --expect-status 200 \ + --output "${EVIDENCE_DIR}/documents.json"; then + echo "Baseline /api/documents requests did not all succeed: ${EVIDENCE_DIR}/documents.json" >&2 + exit 1 +fi + +# telemetry_row_count QUERY -- number of rows the workspace returns for +# QUERY, or 0 when the query errors or returns nothing. Never fails the +# caller: an ingestion-lag miss is expected mid-poll, not this function's +# error to report. +telemetry_row_count() { + local query="$1" + az monitor log-analytics query \ + --workspace "${WORKSPACE_CUSTOMER_ID}" \ + --analytics-query "${query}" \ + --timespan PT30M \ + -o json 2>/dev/null | + jq '(.tables[0].rows // []) | length' 2>/dev/null || echo 0 +} + +ORDERS_SEEN=0 +DOCUMENTS_SEEN=0 +started="${SECONDS}" +while (( SECONDS - started < TELEMETRY_TIMEOUT_SECONDS )); do + if [[ "${ORDERS_SEEN}" -eq 0 ]]; then + orders_rows="$(telemetry_row_count "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where Name has '/api/orders' | take 1")" + [[ "${orders_rows:-0}" -gt 0 ]] && ORDERS_SEEN=1 + fi + if [[ "${DOCUMENTS_SEEN}" -eq 0 ]]; then + documents_rows="$(telemetry_row_count "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where Name has '/api/documents' | take 1")" + [[ "${documents_rows:-0}" -gt 0 ]] && DOCUMENTS_SEEN=1 + fi + if [[ "${ORDERS_SEEN}" -eq 1 && "${DOCUMENTS_SEEN}" -eq 1 ]]; then + break + fi + sleep "${TELEMETRY_POLL_INTERVAL_SECONDS}" +done + +jq -n \ + --argjson ordersSeen "${ORDERS_SEEN}" \ + --argjson documentsSeen "${DOCUMENTS_SEEN}" \ + --arg checkedAt "$(utc_now)" \ + '{orders_telemetry_seen: ($ordersSeen == 1), documents_telemetry_seen: ($documentsSeen == 1), checked_at: $checkedAt}' \ + >"${EVIDENCE_DIR}/telemetry-check.json" + +if [[ "${ORDERS_SEEN}" -ne 1 || "${DOCUMENTS_SEEN}" -ne 1 ]]; then + echo "Application Insights did not show both request types within ${TELEMETRY_TIMEOUT_SECONDS}s (orders=${ORDERS_SEEN} documents=${DOCUMENTS_SEEN})." >&2 + echo "Evidence directory: ${EVIDENCE_DIR}" >&2 + exit 1 +fi + +echo "Evidence directory: ${EVIDENCE_DIR}" diff --git a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh old mode 100644 new mode 100755 diff --git a/monitor/sre-agent-event-lab/scripts/doctor.sh b/monitor/sre-agent-event-lab/scripts/doctor.sh new file mode 100755 index 0000000..3756bdd --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/doctor.sh @@ -0,0 +1,240 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +source "${SCRIPT_DIR}/common.sh" + +# `require_lab_config` may return before every value below is assigned (a +# missing required setting makes it `return 1` immediately). Pre-seeding +# these names keeps every later `${NAME}`/`-n` check well-defined under +# `set -u` no matter how far configuration loading got, without ever +# reading the raw, un-validated process environment as a stand-in. +SUBSCRIPTION_ID="${SUBSCRIPTION_ID:-}" +RESOURCE_GROUP="${RESOURCE_GROUP:-}" +AZURE_ENV_NAME="${AZURE_ENV_NAME:-}" +SRE_AGENT_RESOURCE_ID="${SRE_AGENT_RESOURCE_ID:-}" + +ANY_FAIL=0 +AZURE_SAFE=1 + +# report CHECK STATUS DETAIL -- the doctor output contract: a single +# tab-separated line per check. STATUS is PASS, FAIL, or MANUAL; only FAIL +# marks the overall run unhealthy (`acknowledge`d MANUAL items are a human +# decision, not this script's to make). +report() { + local check_name="$1" status="$2" detail="$3" + printf '%s\t%s\t%s\n' "${check_name}" "${status}" "${detail}" + if [[ "${status}" == "FAIL" ]]; then + ANY_FAIL=1 + fi +} + +# 1. Required commands ------------------------------------------------------ +MISSING_COMMANDS=() +for command_name in az azd jq curl python3; do + command -v "${command_name}" >/dev/null 2>&1 || MISSING_COMMANDS+=("${command_name}") +done +if [[ "${#MISSING_COMMANDS[@]}" -eq 0 ]]; then + report "Required commands" PASS "az, azd, jq, curl, python3 found on PATH." +else + AZURE_SAFE=0 + report "Required commands" FAIL "Install missing commands: ${MISSING_COMMANDS[*]}." +fi + +# Azure CLI login ------------------------------------------------------------ +if login_error="$(az account show --query id -o tsv 2>&1 1>/dev/null)"; then + report "Azure CLI login" PASS "Signed in to Azure CLI." +else + AZURE_SAFE=0 + report "Azure CLI login" FAIL "Run: az login -- ${login_error:-not signed in}" +fi + +# azd configuration ----------------------------------------------------------- +# `require_lab_config` must run in *this* shell (not a subshell) so the +# resolved SUBSCRIPTION_ID/RESOURCE_GROUP/etc. it makes readonly survive for +# every check below; only its stderr is captured to a scratch file. +mkdir -p "${EVIDENCE_ROOT}" +CONFIG_ERROR_FILE="${EVIDENCE_ROOT}/.doctor-config-error" +: >"${CONFIG_ERROR_FILE}" +if require_lab_config 2>"${CONFIG_ERROR_FILE}"; then + report "azd configuration" PASS "Resolved subscription ${SUBSCRIPTION_ID}, resource group ${RESOURCE_GROUP}, environment ${AZURE_ENV_NAME}." +else + AZURE_SAFE=0 + CONFIG_DETAIL="$(tr '\n' ' ' <"${CONFIG_ERROR_FILE}" | sed 's/[[:space:]]*$//')" + report "azd configuration" FAIL "${CONFIG_DETAIL:-Missing required azd configuration.}" + # `load_lab_config` returns before its `readonly` line whenever a + # required setting is missing, so none of these names are readonly yet + # on this path -- safe (and necessary, under `set -u`) to re-seed them. + SUBSCRIPTION_ID="${SUBSCRIPTION_ID:-}" + RESOURCE_GROUP="${RESOURCE_GROUP:-}" + AZURE_ENV_NAME="${AZURE_ENV_NAME:-}" + SRE_AGENT_RESOURCE_ID="${SRE_AGENT_RESOURCE_ID:-}" +fi +rm -f "${CONFIG_ERROR_FILE}" + +# 2. Subscription equality --------------------------------------------------- +if [[ "${AZURE_SAFE}" -eq 1 ]]; then + if subscription_error="$(verify_subscription 2>&1 1>/dev/null)"; then + report "Subscription match" PASS "Active subscription matches ${SUBSCRIPTION_ID}." + else + AZURE_SAFE=0 + report "Subscription match" FAIL "${subscription_error}" + fi +else + report "Subscription match" FAIL "Blocked: resolve the failing check above first." +fi + +# 3. Resource group tags ----------------------------------------------------- +if [[ "${AZURE_SAFE}" -eq 1 ]]; then + if rg_error="$(verify_lab_resource_group 2>&1 1>/dev/null)"; then + report "Resource group tags" PASS "Resource group ${RESOURCE_GROUP} is tagged purpose=sre-agent-event-lab, azd-env-name=${AZURE_ENV_NAME}." + else + AZURE_SAFE=0 + report "Resource group tags" FAIL "${rg_error}" + fi +else + report "Resource group tags" FAIL "Blocked: resolve the failing check above first." +fi + +if [[ "${AZURE_SAFE}" -eq 1 ]]; then + APP_NAME="$(deployment_output containerAppName)" + APP_FQDN="$(deployment_output containerAppFqdn)" + WORKSPACE_CUSTOMER_ID="$(deployment_output workspaceCustomerId)" + TELEMETRY_SERVICE_NAME="$(deployment_output telemetryServiceName)" +else + APP_NAME="" + APP_FQDN="" + WORKSPACE_CUSTOMER_ID="" + TELEMETRY_SERVICE_NAME="" +fi + +# 4. Container App health and /healthz --------------------------------------- +if [[ "${AZURE_SAFE}" -eq 1 && -n "${APP_NAME}" ]]; then + health_state="$(az containerapp revision list \ + --resource-group "${RESOURCE_GROUP}" \ + --name "${APP_NAME}" \ + --query "[?properties.active].properties.healthState | [0]" \ + -o tsv 2>/dev/null || true)" + if [[ "${health_state}" == "Healthy" ]]; then + report "Container App health" PASS "Active revision of ${APP_NAME} is Healthy." + else + report "Container App health" FAIL "Active revision health is '${health_state:-unknown}'. Investigate: az containerapp revision list --resource-group ${RESOURCE_GROUP} --name ${APP_NAME}" + fi +elif [[ "${AZURE_SAFE}" -eq 1 ]]; then + report "Container App health" FAIL "Deployment output containerAppName is empty. Run: azd provision" +else + report "Container App health" FAIL "Blocked: resolve the failing check above first." +fi + +if [[ "${AZURE_SAFE}" -eq 1 && -n "${APP_FQDN}" ]]; then + http_status="$(curl --max-time 10 --silent --output /dev/null --write-out '%{http_code}' "https://${APP_FQDN}/healthz" 2>/dev/null || echo 000)" + if [[ "${http_status}" == "200" ]]; then + report "Health endpoint" PASS "https://${APP_FQDN}/healthz returned HTTP 200." + else + report "Health endpoint" FAIL "https://${APP_FQDN}/healthz returned HTTP ${http_status}. Investigate: curl -v https://${APP_FQDN}/healthz" + fi +elif [[ "${AZURE_SAFE}" -eq 1 ]]; then + report "Health endpoint" FAIL "Deployment output containerAppFqdn is empty. Run: azd provision" +else + report "Health endpoint" FAIL "Blocked: resolve the failing check above first." +fi + +# 5. Application Insights request telemetry in the last 30 minutes ---------- +if [[ "${AZURE_SAFE}" -eq 1 && -n "${WORKSPACE_CUSTOMER_ID}" && -n "${TELEMETRY_SERVICE_NAME}" ]]; then + request_rows="$(az monitor log-analytics query \ + --workspace "${WORKSPACE_CUSTOMER_ID}" \ + --analytics-query "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where TimeGenerated > ago(30m) | count" \ + --timespan PT30M \ + -o json 2>/dev/null | jq '(.tables[0].rows // []) | length' 2>/dev/null || echo 0)" + if [[ "${request_rows:-0}" -gt 0 ]]; then + report "Application Insights telemetry" PASS "AppRequests present for ${TELEMETRY_SERVICE_NAME} in the last 30 minutes." + else + report "Application Insights telemetry" FAIL "No AppRequests telemetry in the last 30 minutes for role ${TELEMETRY_SERVICE_NAME}. Run: lab.sh baseline" + fi +elif [[ "${AZURE_SAFE}" -eq 1 ]]; then + report "Application Insights telemetry" FAIL "Deployment outputs workspaceCustomerId/telemetryServiceName are empty. Run: azd provision" +else + report "Application Insights telemetry" FAIL "Blocked: resolve the failing check above first." +fi + +# 6. Three alert rules enabled ------------------------------------------------ +if [[ "${AZURE_SAFE}" -eq 1 ]]; then + DISABLED_RULES=() + for rule_name in alert-sre-lab-s1-http500 alert-sre-lab-s2-latency alert-sre-lab-s3-storage-rbac; do + rule_json="$(az rest --method get \ + --url "https://management.azure.com/subscriptions/${SUBSCRIPTION_ID}/resourceGroups/${RESOURCE_GROUP}/providers/microsoft.insights/scheduledqueryrules/${rule_name}?api-version=2023-12-01" \ + -o json 2>/dev/null || true)" + if [[ -z "${rule_json}" ]]; then + rule_json='{}' + fi + rule_enabled="$(jq -r '.properties.enabled // false' <<<"${rule_json}" 2>/dev/null || echo false)" + if [[ "${rule_enabled}" != "true" ]]; then + DISABLED_RULES+=("${rule_name}") + fi + done + if [[ "${#DISABLED_RULES[@]}" -eq 0 ]]; then + report "Alert rules enabled" PASS "alert-sre-lab-s1-http500, alert-sre-lab-s2-latency, alert-sre-lab-s3-storage-rbac are all enabled." + else + report "Alert rules enabled" FAIL "Not enabled or missing: ${DISABLED_RULES[*]}. Re-enable: az resource update --ids --set properties.enabled=true (or re-run azd provision)." + fi +else + report "Alert rules enabled" FAIL "Blocked: resolve the failing check above first." +fi + +# 7. SRE Agent resource, only when SRE_AGENT_RESOURCE_ID is configured ------- +if [[ -n "${SRE_AGENT_RESOURCE_ID}" ]]; then + if [[ "${AZURE_SAFE}" -eq 1 ]]; then + if az resource show --ids "${SRE_AGENT_RESOURCE_ID}" -o none 2>/dev/null; then + report "SRE Agent resource" PASS "Resource exists: ${SRE_AGENT_RESOURCE_ID}." + else + report "SRE Agent resource" FAIL "Resource not found: ${SRE_AGENT_RESOURCE_ID}. Verify: az resource show --ids ${SRE_AGENT_RESOURCE_ID}" + fi + else + report "SRE Agent resource" FAIL "Blocked: resolve the failing check above first." + fi +fi + +# 8. Reader role assignment on the lab resource group ------------------------ +if [[ "${AZURE_SAFE}" -eq 1 ]]; then + if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then + report "Reader role assignment" FAIL "Agent setup evidence missing: ${AGENT_SETUP_FILE}. Run: lab.sh acknowledge agent-setup after recording the Agent identities." + else + agent_principal_id="$(jq -r '.agent_principal_id // empty' "${AGENT_SETUP_FILE}")" + agent_uami_principal_id="$(jq -r '.agent_user_assigned_principal_id // empty' "${AGENT_SETUP_FILE}")" + if [[ -z "${agent_principal_id}" || -z "${agent_uami_principal_id}" ]]; then + report "Reader role assignment" FAIL "Agent setup evidence is missing agent_principal_id/agent_user_assigned_principal_id: ${AGENT_SETUP_FILE}." + else + MISSING_READER=() + for principal_id in "${agent_principal_id}" "${agent_uami_principal_id}"; do + reader_count="$(az role assignment list \ + --resource-group "${RESOURCE_GROUP}" \ + --assignee-object-id "${principal_id}" \ + --query "[?roleDefinitionName=='Reader'] | length(@)" \ + -o tsv 2>/dev/null || echo 0)" + if [[ "${reader_count:-0}" -eq 0 ]]; then + MISSING_READER+=("${principal_id}") + fi + done + if [[ "${#MISSING_READER[@]}" -eq 0 ]]; then + report "Reader role assignment" PASS "Reader is assigned on ${RESOURCE_GROUP} for both recorded Agent identities." + else + report "Reader role assignment" FAIL "Missing Reader on ${RESOURCE_GROUP} for: ${MISSING_READER[*]}. Grant: az role assignment create --assignee-object-id --assignee-principal-type ServicePrincipal --role Reader --resource-group ${RESOURCE_GROUP}" + fi + fi + fi +else + report "Reader role assignment" FAIL "Blocked: resolve the failing check above first." +fi + +# 9. Portal-only settings: never inferred, always MANUAL --------------------- +# No official stable Azure SRE Agent API currently reads back the +# repository connection, knowledge sources, incident platform, or response +# plan mode, so these are never reported as PASS/FAIL -- doing so would +# mean guessing. They stay MANUAL until an official stable API can prove +# them, matching the portal path an operator needs to check by hand. +report "Repository connection" MANUAL "No official stable API exposes the Agent's repository connection state. Verify in the portal: https://sre.azure.com > Agent > Settings > Repository." +report "Knowledge source" MANUAL "No official stable API exposes configured knowledge sources. Verify in the portal: https://sre.azure.com > Agent > Settings > Knowledge." +report "Incident platform" MANUAL "No official stable API exposes the incident platform connection. Verify in the portal: https://sre.azure.com > Agent > Settings > Incident platform." +report "Response plan" MANUAL "No official stable API confirms the response plan mode. Verify in the portal: https://sre.azure.com > Agent > Response plans (must be Review)." + +exit "${ANY_FAIL}" diff --git a/monitor/sre-agent-event-lab/scripts/lab.sh b/monitor/sre-agent-event-lab/scripts/lab.sh new file mode 100755 index 0000000..c19c964 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/lab.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +source "${SCRIPT_DIR}/common.sh" + +# `lab_state.py`/`score.py` do not exist yet (a later task adds them). Every +# dispatch case below that needs one checks for the file first so an +# operator gets one clear "not yet available" line instead of a raw +# "No such file or directory" from `exec`. +PYTHON="${LAB_ROOT}/app/.venv/bin/python" +if [[ ! -x "${PYTHON}" ]]; then + PYTHON="python3" +fi +readonly PYTHON + +usage() { + cat <<'USAGE' +Usage: lab.sh [args] + +Commands: + doctor Diagnose the lab environment + baseline Run baseline load and verify telemetry + acknowledge agent-setup Record manually-verified Agent setup evidence + run s1|s2|s3 Run a failure scenario + capture s1|s2|s3 Capture Azure SRE Agent evidence for a scenario + score Score the collected evidence +USAGE +} + +not_yet_available() { + local component="$1" + echo "${component} is not yet available in this lab checkout (planned for a later task)." >&2 + exit 3 +} + +latest_evidence_dir() { + local scenario="$1" + local candidate + candidate="$(ls -dt "${EVIDENCE_ROOT}/${scenario}-"*/ 2>/dev/null | head -n1 || true)" + if [[ -n "${candidate}" ]]; then + printf '%s\n' "${candidate%/}" + fi +} + +# Sub-scripts are run through `bash` explicitly (not a bare `exec path`) so +# dispatch never depends on that file's executable bit. +case "${1:-}" in + doctor) + exec bash "${SCRIPT_DIR}/doctor.sh" + ;; + baseline) + exec bash "${SCRIPT_DIR}/baseline.sh" + ;; + acknowledge) + [[ "${2:-}" == "agent-setup" ]] || { usage >&2; exit 2; } + if [[ ! -f "${SCRIPT_DIR}/lab_state.py" ]]; then + not_yet_available "lab.sh acknowledge agent-setup" + fi + exec "${PYTHON}" "${SCRIPT_DIR}/lab_state.py" acknowledge-agent + ;; + run) + [[ "${2:-}" =~ ^s[123]$ ]] || { usage >&2; exit 2; } + exec bash "${SCRIPT_DIR}/run-scenario.sh" "${2}" + ;; + capture) + [[ "${2:-}" =~ ^s[123]$ ]] || { usage >&2; exit 2; } + EVIDENCE_DIR="$(latest_evidence_dir "${2}")" + if [[ -z "${EVIDENCE_DIR}" ]]; then + echo "No evidence directory found for ${2}. Run: lab.sh run ${2}" >&2 + exit 1 + fi + exec bash "${SCRIPT_DIR}/capture-scenario.sh" "${2}" "${EVIDENCE_DIR}" + ;; + score) + if [[ ! -f "${SCRIPT_DIR}/score.py" ]]; then + not_yet_available "lab.sh score" + fi + exec "${PYTHON}" "${SCRIPT_DIR}/score.py" + ;; + *) + usage >&2 + exit 2 + ;; +esac diff --git a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py new file mode 100644 index 0000000..7aadce5 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py @@ -0,0 +1,391 @@ +"""Fake-CLI harness for `doctor.sh`, `baseline.sh`, and `lab.sh`. + +`lab_script_harness.py`'s generic `az` fake models `run-scenario.sh` / +`query-evidence.sh` / `capture-scenario.sh` / `cleanup.sh`'s call surface. It +does not model the surfaces `doctor.sh` and `baseline.sh` add: a container +app's *current* health state (no polling loop), a `curl` probe of +`/healthz`, `az resource show` for the SRE Agent resource, a per-rule +`az rest` read of `Microsoft.Insights/scheduledQueryRules`, and +`az role assignment list` keyed by a specific `--assignee-object-id`. This +module gives each test full, mutable control over that state through a +single `FakeAz` object so `doctor.sh`/`baseline.sh`/`lab.sh` are driven as +real programs -- not grepped as text -- exactly like the other lab scripts. +""" +import json +import os +import shutil +import subprocess +from dataclasses import dataclass, field +from pathlib import Path +from typing import Dict + +from azd_fake import write_azd_stub, write_executable +from lab_script_harness import ENV_NAME, RESOURCE_GROUP, SCRIPTS_DIR, SUBSCRIPTION_ID + + +BASH = shutil.which("bash") or "/bin/bash" + +APP_NAME = "ca-sre-lab" +APP_FQDN = "ca-sre-lab.example.azurecontainerapps.io" +WORKSPACE_CUSTOMER_ID = "9d1a0b2c-3d4e-5f60-7182-93a4b5c6d7e8" +TELEMETRY_SERVICE_NAME = "sre-lab-order-api" +AGENT_PRINCIPAL_ID = "8c8a4f0e-0000-4000-8000-2b1f9a0c1234" +AGENT_UAMI_PRINCIPAL_ID = "9c8a4f0e-1111-4000-8000-2b1f9a0c5678" + +ALERT_RULE_NAMES = ( + "alert-sre-lab-s1-http500", + "alert-sre-lab-s2-latency", + "alert-sre-lab-s3-storage-rbac", +) + + +def _default_alert_rules_enabled() -> Dict[str, bool]: + return {name: True for name in ALERT_RULE_NAMES} + + +@dataclass +class FakeAz: + """Mutable state for the fake `az`/`azd`/`curl`/`python3` a run uses. + + Every field defaults to a fully healthy lab so a test only has to set + the one attribute it wants to exercise. `workdir` is normally the + fixture's `tmp_path`. State is (re)materialized each time + `run_doctor`/`run_baseline`/`run_lab_cli` is called, so mutating a + field *after* the fixture is created (as the brief's examples do) is + honoured; a lab directory that already exists for `workdir` is reused + (not wiped), so evidence a test or a prior call wrote survives. + """ + + workdir: Path + logged_in: bool = True + active_subscription_id: str = SUBSCRIPTION_ID + resource_group_exists: bool = True + resource_group_purpose: str = "sre-agent-event-lab" + resource_group_env_tag: str = ENV_NAME + container_app_health: str = "Healthy" + healthz_status: int = 200 + app_insights_has_recent_requests: bool = True + app_insights_orders_seen: bool = True + app_insights_documents_seen: bool = True + alert_rules_present: Dict[str, bool] = field(default_factory=_default_alert_rules_enabled) + alert_rules_enabled: Dict[str, bool] = field(default_factory=_default_alert_rules_enabled) + sre_agent_resource_exists: bool = True + agent_setup_present: bool = True + agent_principal_id: str = AGENT_PRINCIPAL_ID + agent_uami_principal_id: str = AGENT_UAMI_PRINCIPAL_ID + reader_role_assigned: Dict[str, bool] = field( + default_factory=lambda: {AGENT_PRINCIPAL_ID: True, AGENT_UAMI_PRINCIPAL_ID: True} + ) + baseline_orders_succeed: bool = True + baseline_documents_succeed: bool = True + # None means "use the module default AZD_VALUES"; a test passes {} (or + # a partial dict) to exercise the missing/partial-configuration paths. + azd_values: "Dict[str, str] | None" = None + + +def _bool_json(value: bool) -> str: + return "true" if value else "false" + + +def _az_stub_source(fake_az: FakeAz, log_path: Path) -> str: + rule_branches = [] + for rule_name in ALERT_RULE_NAMES: + present = fake_az.alert_rules_present.get(rule_name, True) + enabled = fake_az.alert_rules_enabled.get(rule_name, True) + if not present: + rule_branches.append(f' *"/scheduledqueryrules/{rule_name}?"*) exit 1 ;;') + else: + rule_branches.append( + f' *"/scheduledqueryrules/{rule_name}?"*) ' + f'printf \'{{"properties": {{"enabled": {_bool_json(enabled)}}}}}\\n\' ;;' + ) + rule_case = "\n".join(rule_branches) + + reader_branches = [] + for principal_id, has_reader in fake_az.reader_role_assigned.items(): + count = "1" if has_reader else "0" + reader_branches.append(f' "{principal_id}") printf \'{count}\\n\' ;;') + reader_case = "\n".join(reader_branches) if reader_branches else " *) printf '0\\n' ;;" + + orders_rows = "[[1]]" if fake_az.app_insights_orders_seen else "[]" + documents_rows = "[[1]]" if fake_az.app_insights_documents_seen else "[]" + any_rows = "[[1]]" if fake_az.app_insights_has_recent_requests else "[]" + account_show = ( + f"printf '%s\\n' '{fake_az.active_subscription_id}'" if fake_az.logged_in else "exit 1" + ) + group_exists = "printf 'true\\n'" if fake_az.resource_group_exists else "printf 'false\\n'" + resource_show = "exit 0" if fake_az.sre_agent_resource_exists else "exit 1" + + return f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +case "${{1:-}} ${{2:-}}" in + "account show") + {account_show} + ;; + "group exists") + {group_exists} + ;; + "group show") + if [[ "$*" == *"azd-env-name"* ]]; then + printf '%s\\n' "{fake_az.resource_group_env_tag}" + else + printf '%s\\n' "{fake_az.resource_group_purpose}" + fi + ;; + "containerapp revision") + printf '%s\\n' "{fake_az.container_app_health}" + ;; + "monitor log-analytics") + if [[ "$*" == *"/api/orders"* ]]; then + printf '{{"tables": [{{"rows": {orders_rows}}}]}}\\n' + elif [[ "$*" == *"/api/documents"* ]]; then + printf '{{"tables": [{{"rows": {documents_rows}}}]}}\\n' + else + printf '{{"tables": [{{"rows": {any_rows}}}]}}\\n' + fi + ;; + "rest --method") + case "$*" in +{rule_case} + *) printf '{{}}\\n' ;; + esac + ;; + "resource show") + {resource_show} + ;; + "role assignment") + all_args="$*" + principal="${{all_args##*--assignee-object-id }}" + principal="${{principal%% *}}" + case "${{principal}}" in +{reader_case} + esac + ;; + *) + : ;; +esac +exit 0 +""" + + +def _curl_stub_source(fake_az: FakeAz) -> str: + return f"""#!/usr/bin/env bash +printf '%s' "{fake_az.healthz_status}" +""" + + +def _python3_stub_source(fake_az: FakeAz, log_path: Path) -> str: + """Fake `python3`/`.venv/bin/python` for `loadgen.py`: writes a minimal, + valid summary and exits with loadgen's real contract (0 success, 2 a + request mismatch), keyed by the target URL so orders/documents can be + made to succeed or fail independently.""" + orders_ok = 1 if fake_az.baseline_orders_succeed else 0 + documents_ok = 1 if fake_az.baseline_documents_succeed else 0 + return f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +output="" +requests=1 +args=("$@") +for ((i=0; i<${{#args[@]}}; i++)); do + case "${{args[$i]}}" in + --output) output="${{args[$((i+1))]}}" ;; + --requests) requests="${{args[$((i+1))]}}" ;; + esac +done +succeed=1 +if [[ "$*" == *"/api/orders"* ]]; then + succeed={orders_ok} +elif [[ "$*" == *"/api/documents"* ]]; then + succeed={documents_ok} +fi +if [[ -n "${{output}}" ]]; then + mkdir -p "$(dirname "${{output}}")" + printf '{{"total": %s, "errors": 0}}\\n' "${{requests}}" > "${{output}}" +fi +if [[ "${{succeed}}" -eq 1 ]]; then + exit 0 +else + exit 2 +fi +""" + + +AZD_VALUES = { + "AZURE_SUBSCRIPTION_ID": SUBSCRIPTION_ID, + "AZURE_RESOURCE_GROUP": RESOURCE_GROUP, + "AZURE_ENV_NAME": ENV_NAME, + "AZURE_LOCATION": "koreacentral", + "AZURE_CONTAINER_APP_NAME": APP_NAME, + "AZURE_CONTAINER_APP_FQDN": APP_FQDN, + "AZURE_STORAGE_CONTAINER_SCOPE": ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + "/providers/Microsoft.Storage/storageAccounts/stsrelab/blobServices/default" + "/containers/documents" + ), + "AZURE_BLOB_ROLE_ASSIGNMENT_NAME": "3f2504e0-4f89-11d3-9a0c-0305e82c3301", + "AZURE_WORKSPACE_ID": ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + "/providers/Microsoft.OperationalInsights/workspaces/log-sre-lab" + ), + "AZURE_APP_INSIGHTS_NAME": "appi-sre-lab", + "AZURE_TELEMETRY_SERVICE_NAME": TELEMETRY_SERVICE_NAME, + "containerAppPrincipalId": "8c8a4f0e-aaaa-4000-8000-2b1f9a0c1234", + "workspaceCustomerId": WORKSPACE_CUSTOMER_ID, +} + + +class LabRun: + def __init__( + self, + lab: Path, + bin_dir: Path, + workdir: Path, + az_log: Path, + python_log: Path, + azd_log: Path, + ): + self.lab = lab + self.bin_dir = bin_dir + self.workdir = workdir + self.az_log = az_log + self.python_log = python_log + self.azd_log = azd_log + + def run(self, script_name, args=(), env=None): + process_env = { + "PATH": f"{self.bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", + "HOME": os.environ.get("HOME", str(self.lab)), + } + process_env.update(env or {}) + return subprocess.run( + [BASH, str(self.lab / "scripts" / script_name), *args], + capture_output=True, + text=True, + env=process_env, + cwd=str(self.workdir), + ) + + def az_calls(self): + return self.az_log.read_text() if self.az_log.exists() else "" + + def azd_calls(self): + return self.azd_log.read_text() if self.azd_log.exists() else "" + + def evidence_dir(self): + return self.lab / "evidence" + + +def _write_agent_setup(lab: Path, fake_az: FakeAz) -> None: + agent_setup_path = lab / "evidence" / "agent-setup.json" + if not fake_az.agent_setup_present: + agent_setup_path.unlink(missing_ok=True) + return + setup = { + "agent_endpoint": "https://sre-agent.example.com/api/incidents", + "monitoring_contributor_assignment_id": ( + f"/subscriptions/{SUBSCRIPTION_ID}/providers" + "/Microsoft.Authorization/roleAssignments/principal-one" + ), + "agent_principal_id": fake_az.agent_principal_id, + "uami_monitoring_contributor_assignment_id": ( + f"/subscriptions/{SUBSCRIPTION_ID}/providers" + "/Microsoft.Authorization/roleAssignments/principal-two" + ), + "agent_user_assigned_principal_id": fake_az.agent_uami_principal_id, + } + agent_setup_path.write_text(json.dumps(setup)) + + +def _materialize(fake_az: FakeAz) -> LabRun: + """Create (once) or refresh the throwaway lab + fake CLIs for `fake_az`. + + The lab's `scripts/` and `evidence/` directories are only created the + first time a given `fake_az.workdir` is used, so evidence written by an + earlier call (or by the test itself) survives across repeated + `run_doctor`/`run_lab_cli` calls with the same `fake_az`. The fake + `az`/`curl`/`python3` executables are always rewritten so the latest + mutations to `fake_az` take effect immediately. + """ + tmp_path = fake_az.workdir + lab = tmp_path / "lab" + if not lab.exists(): + shutil.copytree( + SCRIPTS_DIR, + lab / "scripts", + ignore=shutil.ignore_patterns("tests", "__pycache__"), + ) + (lab / "azure.yaml").write_text("name: sre-agent-event-lab\n") + (lab / "evidence").mkdir() + + _write_agent_setup(lab, fake_az) + + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + az_log = tmp_path / "az-calls.log" + python_log = tmp_path / "python-calls.log" + azd_log = tmp_path / "azd-calls.log" + + write_executable(bin_dir / "az", _az_stub_source(fake_az, az_log)) + azd_values = fake_az.azd_values if fake_az.azd_values is not None else AZD_VALUES + write_azd_stub(bin_dir, azd_values, "azd_1_29", azd_log) + write_executable(bin_dir / "curl", _curl_stub_source(fake_az)) + write_executable(bin_dir / "python3", _python3_stub_source(fake_az, python_log)) + + venv_bin = lab / "app" / ".venv" / "bin" + venv_bin.mkdir(parents=True, exist_ok=True) + write_executable(venv_bin / "python", _python3_stub_source(fake_az, python_log)) + + workdir = tmp_path / "elsewhere" + workdir.mkdir(exist_ok=True) + + return LabRun(lab, bin_dir, workdir, az_log, python_log, azd_log) + + +def _split_env(env_overrides): + env = {key.upper(): value for key, value in env_overrides.items() if key != "env"} + if "env" in env_overrides: + env.update(env_overrides["env"]) + return env + + +def run_doctor(fake_az: FakeAz, **env_overrides) -> subprocess.CompletedProcess: + """Run `doctor.sh` against `fake_az`'s current state. + + `env_overrides` keys use the process-environment names `common.sh` + reads, lower-cased for readability (e.g. `sre_agent_resource_id="..."` + becomes `SRE_AGENT_RESOURCE_ID`). + """ + run = _materialize(fake_az) + return run.run("doctor.sh", env=_split_env(env_overrides)) + + +def run_baseline(fake_az: FakeAz, **env_overrides) -> subprocess.CompletedProcess: + run = _materialize(fake_az) + return run.run("baseline.sh", env=_split_env(env_overrides)) + + +def run_lab_cli(fake_az: FakeAz, args, **env_overrides) -> subprocess.CompletedProcess: + run = _materialize(fake_az) + return run.run("lab.sh", args=args, env=_split_env(env_overrides)) + + +def lab_dir_for(fake_az: FakeAz) -> Path: + """The lab directory `run_doctor`/`run_baseline`/`run_lab_cli` will use + for `fake_az`. A test can call this *before* the first run to pre-seed + evidence (e.g. a scenario's `timeline.json`) once the directory exists, + or after a run to inspect what the script produced.""" + return fake_az.workdir / "lab" + + +def az_calls_for(fake_az: FakeAz) -> str: + """Every `az` invocation logged for `fake_az.workdir`'s most recent + run, in order, one `argv` per line.""" + log_path = fake_az.workdir / "az-calls.log" + return log_path.read_text() if log_path.exists() else "" + + +def azd_calls_for(fake_az: FakeAz) -> str: + """Every `azd` invocation logged for `fake_az.workdir`'s most recent + run (argv, then `cwd=...`, one per line -- see `azd_fake.py`).""" + log_path = fake_az.workdir / "azd-calls.log" + return log_path.read_text() if log_path.exists() else "" diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py new file mode 100644 index 0000000..d33d64e --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py @@ -0,0 +1,194 @@ +"""Behavioural tests for `doctor.sh`. + +Every test drives `doctor.sh` as a real program against a fake `az`/`azd`/ +`curl` on PATH (see `doctor_harness.py`), not by grepping its text: the +output-contract tests below prove PASS/FAIL/MANUAL rows are produced from +actual (faked) CLI responses, that a single unhealthy signal both flips the +exit code and is distinguishable from a portal-only MANUAL check, and that +the script never queries Azure once its own safety gate (commands, login, +azd configuration, subscription equality, resource-group tags) has failed. +""" +import pytest + +from doctor_harness import ( + AGENT_PRINCIPAL_ID, + AGENT_UAMI_PRINCIPAL_ID, + FakeAz, + az_calls_for, + azd_calls_for, + lab_dir_for, + run_doctor, +) + + +MANUAL_CHECKS = ( + "Repository connection", + "Knowledge source", + "Incident platform", + "Response plan", +) + + +@pytest.fixture +def fake_az(tmp_path): + return FakeAz(workdir=tmp_path) + + +def test_doctor_reports_manual_for_unverifiable_portal_settings(fake_az): + result = run_doctor(fake_az, sre_agent_resource_id="/subscriptions/sub/...") + + assert "Repository connection\tMANUAL" in result.stdout + assert "Response plan\tMANUAL" in result.stdout + + +def test_doctor_fails_when_workload_is_unhealthy(fake_az): + fake_az.container_app_health = "Unhealthy" + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Container App health\tFAIL" in result.stdout + + +def test_doctor_passes_fully_healthy_environment(fake_az): + """Every gating and diagnostic check reaches PASS; only the four + portal-only settings are MANUAL, and the overall exit code is 0.""" + result = run_doctor(fake_az, sre_agent_resource_id="/subscriptions/sub/.../sreAgents/a") + + assert result.returncode == 0, result.stdout + result.stderr + rows = dict(line.split("\t", 2)[0:2] for line in result.stdout.splitlines() if "\t" in line) + assert rows["Required commands"] == "PASS" + assert rows["Azure CLI login"] == "PASS" + assert rows["azd configuration"] == "PASS" + assert rows["Subscription match"] == "PASS" + assert rows["Resource group tags"] == "PASS" + assert rows["Container App health"] == "PASS" + assert rows["Health endpoint"] == "PASS" + assert rows["Application Insights telemetry"] == "PASS" + assert rows["Alert rules enabled"] == "PASS" + assert rows["SRE Agent resource"] == "PASS" + assert rows["Reader role assignment"] == "PASS" + for manual_check in MANUAL_CHECKS: + assert rows[manual_check] == "MANUAL" + + +def test_doctor_omits_sre_agent_resource_row_when_not_configured(fake_az): + """The check is only meaningful -- and only printed -- once an operator + has recorded SRE_AGENT_RESOURCE_ID; otherwise nothing has been created + yet, and there is nothing to verify.""" + result = run_doctor(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + assert "SRE Agent resource" not in result.stdout + + +def test_doctor_fails_when_sre_agent_resource_is_missing(fake_az): + fake_az.sre_agent_resource_exists = False + + result = run_doctor(fake_az, sre_agent_resource_id="/subscriptions/sub/.../sreAgents/missing") + + assert result.returncode == 1 + assert "SRE Agent resource\tFAIL" in result.stdout + + +def test_doctor_fails_when_healthz_does_not_return_200(fake_az): + fake_az.healthz_status = 503 + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Health endpoint\tFAIL" in result.stdout + assert "503" in result.stdout + + +def test_doctor_fails_when_app_insights_has_no_recent_requests(fake_az): + fake_az.app_insights_has_recent_requests = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Application Insights telemetry\tFAIL" in result.stdout + + +def test_doctor_fails_when_an_alert_rule_is_disabled(fake_az): + fake_az.alert_rules_enabled["alert-sre-lab-s2-latency"] = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + row = next(line for line in result.stdout.splitlines() if line.startswith("Alert rules enabled\t")) + assert "FAIL" in row + assert "alert-sre-lab-s2-latency" in row + + +def test_doctor_fails_when_an_alert_rule_is_missing(fake_az): + fake_az.alert_rules_present["alert-sre-lab-s3-storage-rbac"] = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + row = next(line for line in result.stdout.splitlines() if line.startswith("Alert rules enabled\t")) + assert "FAIL" in row + assert "alert-sre-lab-s3-storage-rbac" in row + + +def test_doctor_fails_when_agent_setup_evidence_is_missing(fake_az): + """Unlike the repository/knowledge/response-plan settings, Reader is a + real RBAC role assignment a stable API can check -- so a missing + prerequisite is FAIL, never a shrug-and-guess MANUAL.""" + fake_az.agent_setup_present = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Reader role assignment\tFAIL" in result.stdout + assert "lab.sh acknowledge agent-setup" in result.stdout + + +def test_doctor_fails_when_one_recorded_identity_lacks_reader(fake_az): + fake_az.reader_role_assigned[AGENT_UAMI_PRINCIPAL_ID] = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + row = next( + line for line in result.stdout.splitlines() if line.startswith("Reader role assignment\t") + ) + assert "FAIL" in row + assert AGENT_UAMI_PRINCIPAL_ID in row + + +def test_doctor_never_calls_azure_once_subscription_mismatches(fake_az): + """Fail closed: once the pinned-subscription check fails, no further + Azure resource lookups may happen, even to produce diagnostics.""" + result = run_doctor(fake_az, azure_subscription_id="99999999-9999-9999-9999-999999999999") + + assert result.returncode == 1 + assert "Subscription match\tFAIL" in result.stdout + assert "Blocked: resolve the failing check above first." in result.stdout + assert "containerapp revision" not in az_calls_for(fake_az) + assert "role assignment" not in az_calls_for(fake_az) + + +def test_doctor_never_calls_azure_when_azd_configuration_is_missing(fake_az): + """No azd value and no explicit environment: doctor must stop with the + actionable `azd env set` message before touching Azure, exactly like + every other entry point.""" + fake_az.azd_values = {} + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "azd configuration\tFAIL" in result.stdout + assert "azd env set AZURE_SUBSCRIPTION_ID" in result.stdout + assert "containerapp revision" not in az_calls_for(fake_az) + + +def test_doctor_pins_azd_lookups_to_the_lab_project_root(fake_az): + """`azd env get-value` must be pinned with `--cwd` to this lab's own + project root, exactly as `common.sh`'s other callers already are.""" + result = run_doctor(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + assert f"cwd={lab_dir_for(fake_az)}" in azd_calls_for(fake_az) + diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py new file mode 100644 index 0000000..383b184 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py @@ -0,0 +1,176 @@ +"""Behavioural tests for `lab.sh`, the guided single-command entry point. + +`lab.sh` itself contains no Azure logic -- it only dispatches to +`doctor.sh`, `baseline.sh`, `run-scenario.sh`, `capture-scenario.sh`, and +(once a later task adds them) `lab_state.py`/`score.py`. These tests drive +it as a real program: dispatch to the scripts that already exist is proven +by observing their actual (faked) side effects, not by grepping `lab.sh`'s +source for a case label. The one text-based exception is +`test_lab_cli_dispatches_known_commands`, which is a brief-mandated +regression guard for the dispatcher's own case statement. +""" +import json +from pathlib import Path + +import pytest + +from doctor_harness import FakeAz, lab_dir_for, run_lab_cli +from lab_script_harness import make_lab + + +COMMANDS = ("doctor", "baseline", "acknowledge", "run", "capture", "score") + + +@pytest.fixture +def fake_az(tmp_path): + return FakeAz(workdir=tmp_path) + + +def test_lab_cli_dispatches_known_commands(): + lab_cli = Path(__file__).parents[1].joinpath("lab.sh").read_text() + for command in COMMANDS: + assert f"{command})" in lab_cli, f"lab.sh has no dispatch case for {command}" + + +def test_lab_cli_doctor_dispatches_to_doctor_sh(fake_az): + result = run_lab_cli(fake_az, ["doctor"]) + + assert result.returncode == 0, result.stdout + result.stderr + assert "Required commands\tPASS" in result.stdout + assert "Repository connection\tMANUAL" in result.stdout + + +def test_lab_cli_doctor_surfaces_doctor_sh_failure(fake_az): + fake_az.container_app_health = "Unhealthy" + + result = run_lab_cli(fake_az, ["doctor"]) + + assert result.returncode == 1 + assert "Container App health\tFAIL" in result.stdout + + +def test_lab_cli_baseline_dispatches_to_baseline_sh(fake_az): + result = run_lab_cli( + fake_az, + ["baseline"], + lab_baseline_telemetry_timeout_seconds="5", + lab_baseline_telemetry_poll_interval_seconds="1", + ) + + assert result.returncode == 0, result.stdout + result.stderr + evidence_dirs = sorted((lab_dir_for(fake_az) / "evidence").glob("baseline-*")) + assert evidence_dirs, "lab.sh baseline never invoked baseline.sh" + assert (evidence_dirs[-1] / "telemetry-check.json").is_file() + assert (evidence_dirs[-1] / "orders.json").is_file() + assert (evidence_dirs[-1] / "documents.json").is_file() + + +def test_lab_cli_baseline_surfaces_baseline_sh_failure(fake_az): + fake_az.baseline_orders_succeed = False + + result = run_lab_cli( + fake_az, + ["baseline"], + lab_baseline_telemetry_timeout_seconds="5", + lab_baseline_telemetry_poll_interval_seconds="1", + ) + + assert result.returncode != 0 + + +def test_lab_cli_run_dispatches_to_run_scenario_sh(tmp_path): + lab_run = make_lab(tmp_path) + + result = lab_run.run("lab.sh", ["run", "s1"]) + + assert result.returncode == 0, result.stderr + evidence_dirs = sorted((lab_run.lab / "evidence").glob("s1-*")) + assert evidence_dirs, "lab.sh run never invoked run-scenario.sh" + timeline = json.loads((evidence_dirs[-1] / "timeline.json").read_text()) + assert timeline["scenario"] == "s1" + + +def test_lab_cli_run_rejects_an_unknown_scenario(tmp_path): + lab_run = make_lab(tmp_path) + + result = lab_run.run("lab.sh", ["run", "s9"]) + + assert result.returncode == 2 + assert "Usage" in result.stderr + + +def test_lab_cli_capture_auto_discovers_the_latest_evidence_directory(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + run_result = lab_run.run("lab.sh", ["run", "s1"]) + assert run_result.returncode == 0, run_result.stderr + evidence_dir = sorted((lab_run.lab / "evidence").glob("s1-*"))[-1] + + result = lab_run.run("lab.sh", ["capture", "s1"]) + + assert result.returncode == 0, result.stdout + result.stderr + assert (evidence_dir / "normalized-timeline.json").is_file() + assert (lab_run.lab / "assets" / "captures" / "s1" / "investigation.gif").is_file() + + +def test_lab_cli_capture_fails_clearly_when_no_evidence_exists(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run("lab.sh", ["capture", "s1"]) + + assert result.returncode != 0 + assert "No such file or directory" not in result.stderr + assert "s1" in result.stderr + assert "lab.sh run s1" in result.stderr + + +def test_lab_cli_capture_works_even_when_capture_scenario_is_not_executable(tmp_path): + """Dispatch must not depend on a sub-script's executable bit -- `lab.sh` + always invokes it through `bash`, never a bare `exec path`.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + run_result = lab_run.run("lab.sh", ["run", "s1"]) + assert run_result.returncode == 0, run_result.stderr + (lab_run.lab / "scripts" / "capture-scenario.sh").chmod(0o644) + + result = lab_run.run("lab.sh", ["capture", "s1"]) + + assert result.returncode == 0, result.stdout + result.stderr + + +def test_lab_cli_acknowledge_agent_setup_is_not_yet_available(fake_az): + result = run_lab_cli(fake_az, ["acknowledge", "agent-setup"]) + + assert result.returncode == 3 + assert "not yet available" in result.stderr + assert "No such file or directory" not in result.stderr + + +def test_lab_cli_score_is_not_yet_available(fake_az): + result = run_lab_cli(fake_az, ["score"]) + + assert result.returncode == 3 + assert "not yet available" in result.stderr + assert "No such file or directory" not in result.stderr + + +def test_lab_cli_acknowledge_rejects_an_unknown_subcommand(fake_az): + result = run_lab_cli(fake_az, ["acknowledge", "not-a-setup"]) + + assert result.returncode == 2 + assert "Usage" in result.stderr + + +def test_lab_cli_rejects_an_unknown_command(fake_az): + result = run_lab_cli(fake_az, ["bogus"]) + + assert result.returncode == 2 + assert "Usage" in result.stderr + + +def test_lab_cli_with_no_arguments_prints_usage(fake_az): + result = run_lab_cli(fake_az, []) + + assert result.returncode == 2 + assert "Usage" in result.stderr From 760e2b1f0ea286dcfaeb56a3c05dbdce0e1667b8 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 16:09:40 +0900 Subject: [PATCH 07/26] fix(sre-lab): read real az/azd CLI contracts in doctor and baseline The doctor and baseline telemetry checks were written against inferred CLI behaviour and could never report the truth against real Azure: - `az monitor log-analytics query -o json` prints a flat JSON array of row objects (the log-analytics extension flattens the REST envelope), so the `.tables[0].rows` parse always yielded zero rows and telemetry could only ever FAIL. Parse the real shape through a shared `log_analytics_row_count` helper, and fake that shape in both harnesses. - The doctor query used `| count`, whose single row exists even for an empty table, so row-count semantics could not distinguish data from no data. Query projected rows bounded by `take 1` instead. - `az role assignment list` hides parent-scope grants without `--include-inherited`, reporting a subscription-scoped Reader as missing. Ask for inherited assignments and distinguish direct from inherited in the detail. - A malformed `agent-setup.json` aborted the whole run through `jq` under `set -e`; it is now one FAIL row and exit 1 with the rest of the report intact. Also add the two missing prerequisite rows -- the `log-analytics` CLI extension (`az extension show`, with the exact `az extension add` remedy) and azd's own login state (`azd auth login --check-status --output json`, whose exit code is always 0, so only the printed status is trusted) -- give baseline a poll that always attempts at least once and never sleeps past its deadline, and correct the doctor comment that claimed it never reads the raw process environment. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 12 +- .../sre-agent-event-lab/scripts/baseline.sh | 43 ++-- monitor/sre-agent-event-lab/scripts/common.sh | 66 +++++ monitor/sre-agent-event-lab/scripts/doctor.sh | 113 +++++++-- .../scripts/tests/azd_fake.py | 42 +++- .../scripts/tests/doctor_harness.py | 145 +++++++++-- .../scripts/tests/lab_script_harness.py | 4 +- .../scripts/tests/test_baseline.py | 121 +++++++++ .../scripts/tests/test_doctor.py | 237 +++++++++++++++++- 9 files changed, 716 insertions(+), 67 deletions(-) create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_baseline.py diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 7260f2c..fdd4d1a 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -20,8 +20,10 @@ Azure Container Apps에 의도적인 장애를 만들고, Azure Monitor 경고 ## 사전 조건 - Azure CLI 로그인 및 대상 Azure 구독 접근 권한 +- azd 로그인(`azd auth login`) — azd는 Azure CLI와 별도의 자격 증명을 사용한다 - 구독 또는 필요한 리소스에 Contributor, 역할 할당에는 Owner/User Access Administrator -- `az`, `jq`, `curl`, `python3` +- `az`, `azd`, `jq`, `curl`, `python3` +- `az` Log Analytics extension: `az extension add --name log-analytics` (`az monitor log-analytics query` 제공) - 브라우저에서 `https://sre.azure.com` 및 `*.azuresre.ai` 접근 - Azure SRE Agent Korea Central 사용 권한 - GitHub 저장소 `hellices/devguidesample` 연결 권한 @@ -56,7 +58,13 @@ monitor/sre-agent-event-lab/scripts/lab.sh score # (예정) ### `doctor` 점검 항목 -`lab.sh doctor`는 `CHECKSTATUSDETAIL` 형식으로 한 줄에 하나씩 점검 결과를 출력한다. `STATUS`는 `PASS`, `FAIL`, `MANUAL` 중 하나이며, `FAIL`이 하나라도 있으면 종료 코드 1을 반환한다. 필수 명령, 로그인, azd 구성, 구독/리소스 그룹, Container App 상태, `/healthz`, Application Insights telemetry, alert rule 활성화, SRE Agent 리소스(설정된 경우), Reader 역할 할당을 공식 안정 API로 검증한다. Repository connection, Knowledge source, Incident platform, Response plan은 공식 API로 확인할 수 없으므로 항상 `MANUAL`로 표시되며 portal에서 직접 확인해야 한다. +`lab.sh doctor`는 `CHECKSTATUSDETAIL` 형식으로 한 줄에 하나씩 점검 결과를 출력한다. `STATUS`는 `PASS`, `FAIL`, `MANUAL` 중 하나이며, `FAIL`이 하나라도 있으면 종료 코드 1을 반환한다. 필수 명령, `log-analytics` CLI extension, Azure CLI 로그인, azd 인증(`azd auth login --check-status`), azd 구성, 구독/리소스 그룹, Container App 상태, `/healthz`, Application Insights telemetry, alert rule 활성화, SRE Agent 리소스(설정된 경우), Reader 역할 할당을 공식 안정 API로 검증한다. Repository connection, Knowledge source, Incident platform, Response plan은 공식 API로 확인할 수 없으므로 항상 `MANUAL`로 표시되며 portal에서 직접 확인해야 한다. + +점검 항목 중 세 가지는 실제 CLI 동작에 맞춰 해석해야 한다. + +- **`log-analytics` extension**: `az monitor log-analytics query`는 core CLI에 포함되지 않는다. 없으면 telemetry 점검이 무의미하므로 별도 `FAIL` 행으로 보고하고 `az extension add --name log-analytics`를 안내한다. +- **azd 인증**: `azd auth login --check-status`는 로그인 여부와 무관하게 항상 종료 코드 0을 반환하므로, `--output json`의 `status` 값(`success`/`unauthenticated`)으로만 판정한다. +- **Reader 역할**: `az role assignment list`는 `--include-inherited` 없이는 상위 scope(구독 등)에서 상속된 할당을 보여주지 않는다. 상속된 Reader도 이 lab을 읽는 데 충분하므로 `PASS`로 처리하되, 리소스 그룹에 직접 할당된 경우와 상속된 경우를 DETAIL에서 구분해 표시한다. ## 로컬 검증 diff --git a/monitor/sre-agent-event-lab/scripts/baseline.sh b/monitor/sre-agent-event-lab/scripts/baseline.sh index de2a49b..3080286 100755 --- a/monitor/sre-agent-event-lab/scripts/baseline.sh +++ b/monitor/sre-agent-event-lab/scripts/baseline.sh @@ -47,35 +47,38 @@ if ! python3 "${SCRIPT_DIR}/loadgen.py" \ exit 1 fi -# telemetry_row_count QUERY -- number of rows the workspace returns for -# QUERY, or 0 when the query errors or returns nothing. Never fails the -# caller: an ingestion-lag miss is expected mid-poll, not this function's -# error to report. -telemetry_row_count() { - local query="$1" - az monitor log-analytics query \ - --workspace "${WORKSPACE_CUSTOMER_ID}" \ - --analytics-query "${query}" \ - --timespan PT30M \ - -o json 2>/dev/null | - jq '(.tables[0].rows // []) | length' 2>/dev/null || echo 0 +# The workspace answers with a flat JSON array of row objects (see +# `log_analytics_row_count` in common.sh), so the poll asks for real rows -- +# projected and bounded with `take 1` -- rather than a `| count`, whose +# single row would look like data even for an empty workspace. `contains` +# (substring) is used rather than `has` (term match) so a path like +# `/api/orders` matches inside `GET /api/orders` regardless of tokenization. +telemetry_seen() { + local path_fragment="$1" + local rows + rows="$(log_analytics_row_count "${WORKSPACE_CUSTOMER_ID}" PT30M \ + "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where Name contains '${path_fragment}' | project TimeGenerated, Name | take 1")" + [[ "${rows:-0}" -gt 0 ]] } ORDERS_SEEN=0 DOCUMENTS_SEEN=0 -started="${SECONDS}" -while (( SECONDS - started < TELEMETRY_TIMEOUT_SECONDS )); do - if [[ "${ORDERS_SEEN}" -eq 0 ]]; then - orders_rows="$(telemetry_row_count "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where Name has '/api/orders' | take 1")" - [[ "${orders_rows:-0}" -gt 0 ]] && ORDERS_SEEN=1 +# Bounded poll: always at least one honest attempt (even with a zero +# timeout), and never a sleep that would run past the deadline. +TELEMETRY_DEADLINE=$(( SECONDS + TELEMETRY_TIMEOUT_SECONDS )) +while :; do + if [[ "${ORDERS_SEEN}" -eq 0 ]] && telemetry_seen "/api/orders"; then + ORDERS_SEEN=1 fi - if [[ "${DOCUMENTS_SEEN}" -eq 0 ]]; then - documents_rows="$(telemetry_row_count "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where Name has '/api/documents' | take 1")" - [[ "${documents_rows:-0}" -gt 0 ]] && DOCUMENTS_SEEN=1 + if [[ "${DOCUMENTS_SEEN}" -eq 0 ]] && telemetry_seen "/api/documents"; then + DOCUMENTS_SEEN=1 fi if [[ "${ORDERS_SEEN}" -eq 1 && "${DOCUMENTS_SEEN}" -eq 1 ]]; then break fi + if (( SECONDS + TELEMETRY_POLL_INTERVAL_SECONDS > TELEMETRY_DEADLINE )); then + break + fi sleep "${TELEMETRY_POLL_INTERVAL_SECONDS}" done diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index b5f8b3e..d5f7d52 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -159,6 +159,72 @@ verify_lab_resource_group() { fi } +# The `log-analytics` Azure CLI extension provides +# `az monitor log-analytics query`; it is not part of the core CLI. +readonly LOG_ANALYTICS_EXTENSION_NAME="log-analytics" + +# log_analytics_extension_installed -- true when the extension that provides +# `az monitor log-analytics query` is installed. `az extension show --name` +# is the stable read for this: exit 0 when installed, exit 1 with +# "ERROR: The extension ... is not installed" otherwise. +log_analytics_extension_installed() { + az extension show --name "${LOG_ANALYTICS_EXTENSION_NAME}" -o none >/dev/null 2>&1 +} + +# log_analytics_row_count WORKSPACE TIMESPAN QUERY -- how many rows QUERY +# returned, or 0 when the CLI call or the parse failed. Never fails the +# caller: an ingestion-lag miss mid-poll is an expected answer, not an error +# to report. +# +# Output contract (verified against azure-cli 2.86.0 with the log-analytics +# 1.0.0b1 extension): `az monitor log-analytics query -o json` does **not** +# print the REST envelope `{"tables": [...]}`. The extension's own `_output` +# transform flattens every table into a single JSON array holding one object +# per row -- `TableName` plus one stringified value per column -- so an empty +# result set prints exactly `[]`. Parsing `.tables[0].rows` against that +# always yields nothing, which reads as "no telemetry" forever. +# +# Row *presence* is therefore the reliable "is there data?" signal, and only +# for a query that returns one row per matching record: KQL's `count` +# operator always returns exactly one row (`Count: 0` when nothing matched), +# so counting the rows of a `| count` result answers 1 either way. Callers +# pass a projecting query bounded with `take`, never `| count`. +log_analytics_row_count() { + local workspace="$1" + local timespan="$2" + local query="$3" + local output rows + if ! output="$(az monitor log-analytics query \ + --workspace "${workspace}" \ + --analytics-query "${query}" \ + --timespan "${timespan}" \ + -o json 2>/dev/null)"; then + printf '0\n' + return 0 + fi + if ! rows="$(jq 'if type == "array" then length else 0 end' <<<"${output:-[]}" 2>/dev/null)"; then + printf '0\n' + return 0 + fi + printf '%s\n' "${rows:-0}" +} + +# azd_auth_status -- azd's own word for the current login state: `success`, +# `unauthenticated`, or empty when this azd could not report one. +# +# `azd auth login --check-status` is the only non-interactive login read azd +# offers, and it deliberately "always return[s] a zero exit code" +# (cli/azd/cmd/auth_login.go), printing the answer instead. Reading its exit +# status would report every signed-out operator as signed in, so the +# machine-readable `--output json` status field is parsed instead of the +# human sentence it prints without that flag. +azd_auth_status() { + local status + status="$(azd auth login --check-status --output json --cwd "${LAB_ROOT}" 2>/dev/null | + jq -r '.status // empty' 2>/dev/null)" || true + printf '%s\n' "${status:-}" +} + # deployment_output NAME -- returns an azd deployment output already # resolved by load_lab_config. Callers must call load_lab_config first. deployment_output() { diff --git a/monitor/sre-agent-event-lab/scripts/doctor.sh b/monitor/sre-agent-event-lab/scripts/doctor.sh index 3756bdd..4cb9ab2 100755 --- a/monitor/sre-agent-event-lab/scripts/doctor.sh +++ b/monitor/sre-agent-event-lab/scripts/doctor.sh @@ -5,10 +5,13 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" source "${SCRIPT_DIR}/common.sh" # `require_lab_config` may return before every value below is assigned (a -# missing required setting makes it `return 1` immediately). Pre-seeding -# these names keeps every later `${NAME}`/`-n` check well-defined under -# `set -u` no matter how far configuration loading got, without ever -# reading the raw, un-validated process environment as a stand-in. +# missing required setting makes it `return 1` immediately). Seeding these +# names from the process environment keeps every later `${NAME}`/`-n` check +# well-defined under `set -u` no matter how far configuration loading got. +# The process environment is exactly where `common.sh`'s `setting` takes its +# highest-precedence value from, so this reads the same source it would -- +# it does not invent a fallback, and any name that survives here unset stays +# empty rather than becoming a configuration value on its own. SUBSCRIPTION_ID="${SUBSCRIPTION_ID:-}" RESOURCE_GROUP="${RESOURCE_GROUP:-}" AZURE_ENV_NAME="${AZURE_ENV_NAME:-}" @@ -41,6 +44,21 @@ else report "Required commands" FAIL "Install missing commands: ${MISSING_COMMANDS[*]}." fi +# Log Analytics CLI extension ------------------------------------------------ +# `az monitor log-analytics query` -- the only read behind the telemetry +# check and behind `baseline.sh`/`query-evidence.sh` -- ships in an extension +# that is not installed with the core CLI, so its absence is a prerequisite +# failure with its own row rather than a mysterious empty query result. +if ! command -v az >/dev/null 2>&1; then + AZURE_SAFE=0 + report "Log Analytics CLI extension" FAIL "Blocked: install the Azure CLI first, then run: az extension add --name ${LOG_ANALYTICS_EXTENSION_NAME}" +elif log_analytics_extension_installed; then + report "Log Analytics CLI extension" PASS "az monitor log-analytics query is available (extension ${LOG_ANALYTICS_EXTENSION_NAME})." +else + AZURE_SAFE=0 + report "Log Analytics CLI extension" FAIL "az monitor log-analytics query is unavailable. Install it: az extension add --name ${LOG_ANALYTICS_EXTENSION_NAME}" +fi + # Azure CLI login ------------------------------------------------------------ if login_error="$(az account show --query id -o tsv 2>&1 1>/dev/null)"; then report "Azure CLI login" PASS "Signed in to Azure CLI." @@ -49,6 +67,29 @@ else report "Azure CLI login" FAIL "Run: az login -- ${login_error:-not signed in}" fi +# azd authentication --------------------------------------------------------- +# azd keeps its own credential store: `az login` alone does not make +# `azd env`/`azd provision` work. `azd auth login --check-status` is the only +# non-interactive read of that state and always exits 0, so the row is +# decided by the status it prints, never by its exit code. +AZD_AUTH_STATUS="" +if command -v azd >/dev/null 2>&1; then + AZD_AUTH_STATUS="$(azd_auth_status)" +fi +case "${AZD_AUTH_STATUS}" in + success) + report "azd authentication" PASS "azd reports an authenticated session (azd auth login --check-status)." + ;; + unauthenticated) + AZURE_SAFE=0 + report "azd authentication" FAIL "azd is not signed in. Run: azd auth login" + ;; + *) + AZURE_SAFE=0 + report "azd authentication" FAIL "azd did not report a login status. Check it by hand: azd auth login --check-status, then run: azd auth login" + ;; +esac + # azd configuration ----------------------------------------------------------- # `require_lab_config` must run in *this* shell (not a subshell) so the # resolved SUBSCRIPTION_ID/RESOURCE_GROUP/etc. it makes readonly survive for @@ -140,12 +181,14 @@ else fi # 5. Application Insights request telemetry in the last 30 minutes ---------- +# The query projects and `take`s real rows instead of `| count`: KQL's +# `count` always returns exactly one row (`Count: 0` for an empty table), so +# a row count taken from it can never distinguish data from no data. The +# extension prints a flat JSON array, and `log_analytics_row_count` parses +# that shape (see common.sh). if [[ "${AZURE_SAFE}" -eq 1 && -n "${WORKSPACE_CUSTOMER_ID}" && -n "${TELEMETRY_SERVICE_NAME}" ]]; then - request_rows="$(az monitor log-analytics query \ - --workspace "${WORKSPACE_CUSTOMER_ID}" \ - --analytics-query "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | where TimeGenerated > ago(30m) | count" \ - --timespan PT30M \ - -o json 2>/dev/null | jq '(.tables[0].rows // []) | length' 2>/dev/null || echo 0)" + request_rows="$(log_analytics_row_count "${WORKSPACE_CUSTOMER_ID}" PT30M \ + "AppRequests | where AppRoleName == '${TELEMETRY_SERVICE_NAME}' | project TimeGenerated, Name | take 1")" if [[ "${request_rows:-0}" -gt 0 ]]; then report "Application Insights telemetry" PASS "AppRequests present for ${TELEMETRY_SERVICE_NAME} in the last 30 minutes." else @@ -195,30 +238,64 @@ if [[ -n "${SRE_AGENT_RESOURCE_ID}" ]]; then fi # 8. Reader role assignment on the lab resource group ------------------------ +# `--include-inherited` is deliberate: Reader granted at the subscription (or +# management group) gives the Agent exactly the effective read access it +# needs on this resource group, and omitting the flag hides those grants +# entirely -- `az role assignment list` returns only assignments made at the +# queried scope without it -- which would report a working setup as broken. +# The detail still distinguishes the two, because an operator who requires an +# explicit resource-group-scoped assignment has to be able to see that the +# access is only inherited. if [[ "${AZURE_SAFE}" -eq 1 ]]; then + RESOURCE_GROUP_SCOPE="/subscriptions/${SUBSCRIPTION_ID}/resourceGroups/${RESOURCE_GROUP}" if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then report "Reader role assignment" FAIL "Agent setup evidence missing: ${AGENT_SETUP_FILE}. Run: lab.sh acknowledge agent-setup after recording the Agent identities." + elif ! AGENT_SETUP_JSON="$(jq '.' "${AGENT_SETUP_FILE}" 2>/dev/null)"; then + # A hand-edited or truncated evidence file is one FAIL row, not a raw + # `jq` abort: under `set -e` an unguarded parse would kill the run and + # swallow every remaining check, including the MANUAL rows an operator + # still needs. + report "Reader role assignment" FAIL "Agent setup evidence is not valid JSON: ${AGENT_SETUP_FILE}. Recreate it: lab.sh acknowledge agent-setup" else - agent_principal_id="$(jq -r '.agent_principal_id // empty' "${AGENT_SETUP_FILE}")" - agent_uami_principal_id="$(jq -r '.agent_user_assigned_principal_id // empty' "${AGENT_SETUP_FILE}")" + agent_principal_id="$(jq -r '.agent_principal_id // empty' <<<"${AGENT_SETUP_JSON}" 2>/dev/null || true)" + agent_uami_principal_id="$(jq -r '.agent_user_assigned_principal_id // empty' <<<"${AGENT_SETUP_JSON}" 2>/dev/null || true)" if [[ -z "${agent_principal_id}" || -z "${agent_uami_principal_id}" ]]; then report "Reader role assignment" FAIL "Agent setup evidence is missing agent_principal_id/agent_user_assigned_principal_id: ${AGENT_SETUP_FILE}." else MISSING_READER=() + INHERITED_READER=() for principal_id in "${agent_principal_id}" "${agent_uami_principal_id}"; do - reader_count="$(az role assignment list \ + assignments_json="$(az role assignment list \ --resource-group "${RESOURCE_GROUP}" \ --assignee-object-id "${principal_id}" \ - --query "[?roleDefinitionName=='Reader'] | length(@)" \ - -o tsv 2>/dev/null || echo 0)" - if [[ "${reader_count:-0}" -eq 0 ]]; then + --include-inherited \ + -o json 2>/dev/null || true)" + if [[ -z "${assignments_json}" ]]; then + assignments_json='[]' + fi + direct_count="$(jq --arg scope "${RESOURCE_GROUP_SCOPE}" ' + [.[]? | select((.roleDefinitionName // "") == "Reader") + | select(((.scope // "") | ascii_downcase) == ($scope | ascii_downcase))] + | length' <<<"${assignments_json}" 2>/dev/null || echo 0)" + inherited_scope="$(jq -r --arg scope "${RESOURCE_GROUP_SCOPE}" ' + [.[]? | select((.roleDefinitionName // "") == "Reader") + | select(((.scope // "") | ascii_downcase) != ($scope | ascii_downcase)) + | .scope] + | first // empty' <<<"${assignments_json}" 2>/dev/null || true)" + if [[ "${direct_count:-0}" -gt 0 ]]; then + continue + elif [[ -n "${inherited_scope}" ]]; then + INHERITED_READER+=("${principal_id} (inherited from ${inherited_scope})") + else MISSING_READER+=("${principal_id}") fi done - if [[ "${#MISSING_READER[@]}" -eq 0 ]]; then - report "Reader role assignment" PASS "Reader is assigned on ${RESOURCE_GROUP} for both recorded Agent identities." + if [[ "${#MISSING_READER[@]}" -gt 0 ]]; then + report "Reader role assignment" FAIL "Missing Reader on ${RESOURCE_GROUP} (direct or inherited) for: ${MISSING_READER[*]}. Grant: az role assignment create --assignee-object-id --assignee-principal-type ServicePrincipal --role Reader --resource-group ${RESOURCE_GROUP}" + elif [[ "${#INHERITED_READER[@]}" -gt 0 ]]; then + report "Reader role assignment" PASS "Reader is effective on ${RESOURCE_GROUP} for both recorded Agent identities; not assigned directly for: ${INHERITED_READER[*]}. Inherited access is sufficient to read the lab; assign it on ${RESOURCE_GROUP} if the setup must be scoped to this lab only." else - report "Reader role assignment" FAIL "Missing Reader on ${RESOURCE_GROUP} for: ${MISSING_READER[*]}. Grant: az role assignment create --assignee-object-id --assignee-principal-type ServicePrincipal --role Reader --resource-group ${RESOURCE_GROUP}" + report "Reader role assignment" PASS "Reader is assigned directly on ${RESOURCE_GROUP} for both recorded Agent identities." fi fi fi diff --git a/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py b/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py index 4ea5473..9492dab 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py +++ b/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py @@ -11,9 +11,16 @@ $ azd env get-value AZURE_LOCATION --cwd # from any cwd rc=1, stdout="\nERROR: ensuring environment exists: environment not specified" + +$ azd auth login --check-status +rc=0, stdout="Logged in to Azure as " + +$ azd auth login --check-status --output json +rc=0, stdout='{"status": "success", "expiresOn": "2026-08-14T07:57:15Z"}' ``` -Two properties matter for `common.sh` and are therefore modelled here: +Three properties matter for `common.sh`/`doctor.sh` and are therefore +modelled here: 1. azd writes its `ERROR:` diagnostics to **stdout**, not stderr, and signals failure only through the exit status. A caller that keeps stdout when the @@ -22,6 +29,13 @@ process working directory, and fails when that directory holds no `azure.yaml`. A lookup that does not pin the project root breaks as soon as a script is invoked from the repository root or any other directory. +3. `azd auth login --check-status` is the one non-interactive way to read the + login state, and it **always exits 0** -- "In check status mode, we always + print the final status to stdout. ... We always return a zero exit code." + (`cli/azd/cmd/auth_login.go`). The answer lives only in the output: + `{"status": "success"}` or `{"status": "unauthenticated"}` under + `--output json`, a human sentence otherwise. A caller that trusts the exit + status reports every signed-out operator as signed in. `MISSING_KEY_MODES` exposes both observed missing-value shapes so tests can prove the reader is driven by the exit status rather than by stdout text. @@ -35,6 +49,8 @@ NO_ENVIRONMENT_ERROR = ( "\nERROR: ensuring environment exists: environment not specified" ) +LOGGED_IN_MESSAGE = "Logged in to Azure as lab-operator@example.com" +NOT_LOGGED_IN_MESSAGE = "Not logged in, run `azd auth login` to login to Azure" # How the fake reports a value it does not have. # "azd_1_29" -- what the real CLI does: ERROR text on stdout, exit 1. @@ -51,7 +67,7 @@ def _missing_key_branch(missing_key_mode): return f" printf '%s\\n' '{NO_ENVIRONMENT_ERROR.lstrip(chr(10))}'\n exit 1" -def azd_stub_source(azd_values, missing_key_mode="azd_1_29", log_path=None): +def azd_stub_source(azd_values, missing_key_mode="azd_1_29", log_path=None, logged_in=True): """Bash source for a fake `azd` honouring the contract described above.""" lines = [ "#!/usr/bin/env bash", @@ -68,10 +84,28 @@ def azd_stub_source(azd_values, missing_key_mode="azd_1_29", log_path=None): if log_path is not None: lines.append(f'printf \'%s\\n\' "${{argv[*]:-}}" >> "{log_path}"') lines.append(f'printf \'cwd=%s\\n\' "${{project_dir}}" >> "{log_path}"') + status_json = ( + '{"status": "success", "expiresOn": "2026-08-14T07:57:15Z"}' + if logged_in + else '{"status": "unauthenticated"}' + ) + status_message = LOGGED_IN_MESSAGE if logged_in else NOT_LOGGED_IN_MESSAGE lines += [ + # `auth` needs no azd project. `--check-status` never fails: the exit + # code is 0 whether or not anyone is signed in, so only the printed + # status carries the answer. 'if [[ "${argv[0]:-}" == "auth" ]]; then', + ' if [[ "${argv[*]:-}" == *--check-status* ]]; then', + ' if [[ "${argv[*]:-}" == *"--output json"* ]]; then', + f" printf '%s\\n' '{status_json}'", + " else", + f" printf '%s\\n' '{status_message}'", + " fi", + " fi", " exit 0", "fi", + ] + lines += [ '# Only the `env` commands need an azd project; `auth` does not.', 'if [[ "${argv[0]:-}" == "env" && ! -f "${project_dir}/azure.yaml" ]]; then', f" printf '%s\\n' '{NO_PROJECT_ERROR.lstrip(chr(10))}'", @@ -102,8 +136,8 @@ def write_executable(path, content): return path -def write_azd_stub(bin_dir, azd_values, missing_key_mode="azd_1_29", log_path=None): +def write_azd_stub(bin_dir, azd_values, missing_key_mode="azd_1_29", log_path=None, logged_in=True): return write_executable( bin_dir / "azd", - azd_stub_source(azd_values, missing_key_mode, log_path), + azd_stub_source(azd_values, missing_key_mode, log_path, logged_in), ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py index 7aadce5..c17b69e 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py @@ -4,12 +4,24 @@ `query-evidence.sh` / `capture-scenario.sh` / `cleanup.sh`'s call surface. It does not model the surfaces `doctor.sh` and `baseline.sh` add: a container app's *current* health state (no polling loop), a `curl` probe of -`/healthz`, `az resource show` for the SRE Agent resource, a per-rule -`az rest` read of `Microsoft.Insights/scheduledQueryRules`, and -`az role assignment list` keyed by a specific `--assignee-object-id`. This -module gives each test full, mutable control over that state through a -single `FakeAz` object so `doctor.sh`/`baseline.sh`/`lab.sh` are driven as -real programs -- not grepped as text -- exactly like the other lab scripts. +`/healthz`, `az extension show` for the `log-analytics` extension, +`az resource show` for the SRE Agent resource, a per-rule `az rest` read of +`Microsoft.Insights/scheduledQueryRules`, and `az role assignment list` +keyed by a specific `--assignee-object-id`. This module gives each test +full, mutable control over that state through a single `FakeAz` object so +`doctor.sh`/`baseline.sh`/`lab.sh` are driven as real programs -- not +grepped as text -- exactly like the other lab scripts. + +Three observable contracts here were re-verified against the *real* CLIs on +2026-08-14 (azure-cli 2.86.0 / log-analytics 1.0.0b1 / azd 1.29.0) rather +than assumed, because each one had been modelled incorrectly before: + +1. `az monitor log-analytics query -o json` prints a flat JSON array of row + objects, not the `{"tables": [...]}` REST envelope (see `_rows`). +2. `az role assignment list` hides parent-scope assignments unless + `--include-inherited` is passed. +3. `azd auth login --check-status` always exits 0 and reports the real + answer only in its output (see `azd_fake.py`). """ import json import os @@ -38,11 +50,46 @@ "alert-sre-lab-s3-storage-rbac", ) +RESOURCE_GROUP_SCOPE = f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" +SUBSCRIPTION_SCOPE = f"/subscriptions/{SUBSCRIPTION_ID}" + +# `az monitor log-analytics query -o json` does NOT print the REST envelope +# (`{"tables": [...]}`). The `log-analytics` extension's `Query._output` +# flattens every table into one JSON array with a single object per row -- +# `TableName` plus one *stringified* value per projected column -- so an +# empty result set prints exactly `[]`. Verified against the installed +# extension source (log-analytics 1.0.0b1, azure-cli 2.86.0): +# `~/.azure/cliextensions/log-analytics/azext_loganalytics/custom.py`. +def _rows(*rows) -> str: + return json.dumps(list(rows)) + + +APP_REQUESTS_ROW = { + "TableName": "PrimaryResult", + "TimeGenerated": "2026-08-14T00:05:00Z", + "Name": "GET /api/orders", +} +DOCUMENT_REQUESTS_ROW = { + "TableName": "PrimaryResult", + "TimeGenerated": "2026-08-14T00:06:00Z", + "Name": "GET /api/documents", +} +EMPTY_RESULT = "[]" + +# What KQL's `count` operator really returns: exactly one row, whatever the +# data looks like. A fake that answers `| count` with an empty array would +# hide the very defect the doctor/baseline telemetry checks must not have. +COUNT_ZERO_RESULT = _rows({"TableName": "PrimaryResult", "Count": "0"}) + def _default_alert_rules_enabled() -> Dict[str, bool]: return {name: True for name in ALERT_RULE_NAMES} +def _no_inherited_reader() -> Dict[str, bool]: + return {AGENT_PRINCIPAL_ID: False, AGENT_UAMI_PRINCIPAL_ID: False} + + @dataclass class FakeAz: """Mutable state for the fake `az`/`azd`/`curl`/`python3` a run uses. @@ -58,6 +105,8 @@ class FakeAz: workdir: Path logged_in: bool = True + azd_logged_in: bool = True + log_analytics_extension_installed: bool = True active_subscription_id: str = SUBSCRIPTION_ID resource_group_exists: bool = True resource_group_purpose: str = "sre-agent-event-lab" @@ -71,11 +120,16 @@ class FakeAz: alert_rules_enabled: Dict[str, bool] = field(default_factory=_default_alert_rules_enabled) sre_agent_resource_exists: bool = True agent_setup_present: bool = True + agent_setup_body: "str | None" = None agent_principal_id: str = AGENT_PRINCIPAL_ID agent_uami_principal_id: str = AGENT_UAMI_PRINCIPAL_ID + # Reader assigned *directly* on the lab resource group. reader_role_assigned: Dict[str, bool] = field( default_factory=lambda: {AGENT_PRINCIPAL_ID: True, AGENT_UAMI_PRINCIPAL_ID: True} ) + # Reader assigned on the subscription and therefore only visible to a + # lookup that asks for inherited assignments. + reader_role_inherited: Dict[str, bool] = field(default_factory=_no_inherited_reader) baseline_orders_succeed: bool = True baseline_documents_succeed: bool = True # None means "use the module default AZD_VALUES"; a test passes {} (or @@ -101,27 +155,69 @@ def _az_stub_source(fake_az: FakeAz, log_path: Path) -> str: ) rule_case = "\n".join(rule_branches) + # `az role assignment list` only returns assignments made at *parent* + # scopes when `--include-inherited` is passed; without it, a Reader + # granted on the subscription is invisible to a resource-group scoped + # lookup. The fake reproduces that, so a doctor that drops the flag + # cannot pass the inherited-Reader test by accident. + principal_ids = set(fake_az.reader_role_assigned) | set(fake_az.reader_role_inherited) reader_branches = [] - for principal_id, has_reader in fake_az.reader_role_assigned.items(): - count = "1" if has_reader else "0" - reader_branches.append(f' "{principal_id}") printf \'{count}\\n\' ;;') - reader_case = "\n".join(reader_branches) if reader_branches else " *) printf '0\\n' ;;" - - orders_rows = "[[1]]" if fake_az.app_insights_orders_seen else "[]" - documents_rows = "[[1]]" if fake_az.app_insights_documents_seen else "[]" - any_rows = "[[1]]" if fake_az.app_insights_has_recent_requests else "[]" + for principal_id in sorted(principal_ids): + direct = [] + inherited = [] + if fake_az.reader_role_assigned.get(principal_id, False): + direct.append( + { + "principalId": principal_id, + "roleDefinitionName": "Reader", + "scope": RESOURCE_GROUP_SCOPE, + } + ) + if fake_az.reader_role_inherited.get(principal_id, False): + inherited.append( + { + "principalId": principal_id, + "roleDefinitionName": "Reader", + "scope": SUBSCRIPTION_SCOPE, + } + ) + reader_branches.append( + f' "{principal_id}")\n' + f" if [[ \"${{all_args}}\" == *--include-inherited* ]]; then\n" + f" printf '%s\\n' '{json.dumps(direct + inherited)}'\n" + f" else\n" + f" printf '%s\\n' '{json.dumps(direct)}'\n" + f" fi ;;" + ) + reader_branches.append(" *) printf '[]\\n' ;;") + reader_case = "\n".join(reader_branches) + + orders_rows = _rows(APP_REQUESTS_ROW) if fake_az.app_insights_orders_seen else EMPTY_RESULT + documents_rows = ( + _rows(DOCUMENT_REQUESTS_ROW) if fake_az.app_insights_documents_seen else EMPTY_RESULT + ) + any_rows = _rows(APP_REQUESTS_ROW) if fake_az.app_insights_has_recent_requests else EMPTY_RESULT account_show = ( f"printf '%s\\n' '{fake_az.active_subscription_id}'" if fake_az.logged_in else "exit 1" ) group_exists = "printf 'true\\n'" if fake_az.resource_group_exists else "printf 'false\\n'" resource_show = "exit 0" if fake_az.sre_agent_resource_exists else "exit 1" + log_analytics_extension = "exit 0" if fake_az.log_analytics_extension_installed else ( + "printf 'ERROR: The extension log-analytics is not installed.\\n' >&2\n exit 1" + ) return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" +all_args="$*" case "${{1:-}} ${{2:-}}" in "account show") {account_show} ;; + "extension show") + if [[ "${{all_args}}" == *"--name log-analytics"* ]]; then + {log_analytics_extension} + fi + ;; "group exists") {group_exists} ;; @@ -136,12 +232,17 @@ def _az_stub_source(fake_az: FakeAz, log_path: Path) -> str: printf '%s\\n' "{fake_az.container_app_health}" ;; "monitor log-analytics") - if [[ "$*" == *"/api/orders"* ]]; then - printf '{{"tables": [{{"rows": {orders_rows}}}]}}\\n' - elif [[ "$*" == *"/api/documents"* ]]; then - printf '{{"tables": [{{"rows": {documents_rows}}}]}}\\n' + # KQL's `count` operator always returns exactly one row, even for an + # empty table -- reproduced here so any caller that infers "data + # exists" from a `| count` result's row count fails loudly. + if [[ "${{all_args}}" == *"| count"* ]]; then + printf '%s\\n' '{COUNT_ZERO_RESULT}' + elif [[ "${{all_args}}" == *"/api/orders"* ]]; then + printf '%s\\n' '{orders_rows}' + elif [[ "${{all_args}}" == *"/api/documents"* ]]; then + printf '%s\\n' '{documents_rows}' else - printf '{{"tables": [{{"rows": {any_rows}}}]}}\\n' + printf '%s\\n' '{any_rows}' fi ;; "rest --method") @@ -154,7 +255,6 @@ def _az_stub_source(fake_az: FakeAz, log_path: Path) -> str: {resource_show} ;; "role assignment") - all_args="$*" principal="${{all_args##*--assignee-object-id }}" principal="${{principal%% *}}" case "${{principal}}" in @@ -280,6 +380,9 @@ def _write_agent_setup(lab: Path, fake_az: FakeAz) -> None: if not fake_az.agent_setup_present: agent_setup_path.unlink(missing_ok=True) return + if fake_az.agent_setup_body is not None: + agent_setup_path.write_text(fake_az.agent_setup_body) + return setup = { "agent_endpoint": "https://sre-agent.example.com/api/incidents", "monitoring_contributor_assignment_id": ( @@ -327,7 +430,7 @@ def _materialize(fake_az: FakeAz) -> LabRun: write_executable(bin_dir / "az", _az_stub_source(fake_az, az_log)) azd_values = fake_az.azd_values if fake_az.azd_values is not None else AZD_VALUES - write_azd_stub(bin_dir, azd_values, "azd_1_29", azd_log) + write_azd_stub(bin_dir, azd_values, "azd_1_29", azd_log, logged_in=fake_az.azd_logged_in) write_executable(bin_dir / "curl", _curl_stub_source(fake_az)) write_executable(bin_dir / "python3", _python3_stub_source(fake_az, python_log)) diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 8829d70..5a5b3e1 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -109,7 +109,9 @@ def _az_stub_source(log_path, state_dir): printf '[]\\n' fi ;; "monitor log-analytics") - printf '{{"tables": []}}\\n' ;; + # The `log-analytics` extension flattens the REST `{{"tables": [...]}}` + # envelope into one JSON array of row objects, so "no rows" is `[]`. + printf '[]\\n' ;; "monitor activity-log") printf '[]\\n' ;; "role assignment") diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_baseline.py b/monitor/sre-agent-event-lab/scripts/tests/test_baseline.py new file mode 100644 index 0000000..9908a1b --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_baseline.py @@ -0,0 +1,121 @@ +"""Behavioural tests for `baseline.sh`. + +`baseline.sh` is the one script that has to decide "did my traffic actually +reach Application Insights?" from a Log Analytics answer, so its parsing of +the real `az monitor log-analytics query` output shape -- a flat JSON array +of row objects, never the `{"tables": [...]}` REST envelope -- and the bound +on its polling loop are the two properties worth pinning. Both are exercised +by running the script as a real program against the fake CLIs in +`doctor_harness.py`. +""" +import json +import time + +import pytest + +from doctor_harness import FakeAz, az_calls_for, lab_dir_for, run_baseline + + +# Long enough to allow several poll rounds, short enough that a genuinely +# unbounded loop fails the test instead of hanging the suite. +TELEMETRY_TIMEOUT_SECONDS = "5" +POLL_INTERVAL_SECONDS = "1" + + +@pytest.fixture +def fake_az(tmp_path): + return FakeAz(workdir=tmp_path) + + +def run_bounded_baseline(fake_az, **env_overrides): + return run_baseline( + fake_az, + lab_baseline_telemetry_timeout_seconds=TELEMETRY_TIMEOUT_SECONDS, + lab_baseline_telemetry_poll_interval_seconds=POLL_INTERVAL_SECONDS, + **env_overrides, + ) + + +def telemetry_check(fake_az): + evidence_dirs = sorted((lab_dir_for(fake_az) / "evidence").glob("baseline-*")) + assert evidence_dirs, "baseline.sh wrote no evidence directory" + return json.loads((evidence_dirs[-1] / "telemetry-check.json").read_text()) + + +def analytics_queries(fake_az): + return [line for line in az_calls_for(fake_az).splitlines() if "monitor log-analytics" in line] + + +def test_baseline_succeeds_when_both_request_types_appear(fake_az): + """The healthy case: the workspace answers with a non-empty flat array + for both request types.""" + result = run_bounded_baseline(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + assert telemetry_check(fake_az) == { + "orders_telemetry_seen": True, + "documents_telemetry_seen": True, + "checked_at": telemetry_check(fake_az)["checked_at"], + } + + +def test_baseline_fails_when_the_workspace_returns_no_rows(fake_az): + """The empty case: `[]` for `/api/orders` must not be mistaken for data, + and the failure has to name what was and was not seen.""" + fake_az.app_insights_orders_seen = False + + result = run_bounded_baseline(fake_az) + + assert result.returncode != 0 + assert telemetry_check(fake_az)["orders_telemetry_seen"] is False + assert telemetry_check(fake_az)["documents_telemetry_seen"] is True + assert "orders=0" in result.stderr + + +def test_baseline_query_does_not_count_rows_of_a_count(fake_az): + """KQL `count` always returns exactly one row, so a row count taken from + a `| count` query reports "data exists" for an empty workspace.""" + run_bounded_baseline(fake_az) + + queries = analytics_queries(fake_az) + assert queries + for query in queries: + assert "| count" not in query, f"baseline query relies on `| count`: {query}" + + +def test_baseline_polling_is_bounded_by_the_timeout(fake_az): + """Telemetry that never arrives must end the run near the timeout, not + hang and not exit after a single try.""" + fake_az.app_insights_orders_seen = False + fake_az.app_insights_documents_seen = False + + started = time.monotonic() + result = run_bounded_baseline(fake_az) + elapsed = time.monotonic() - started + + assert result.returncode != 0 + assert elapsed < int(TELEMETRY_TIMEOUT_SECONDS) + 20, f"poll overran its bound: {elapsed}s" + assert len(analytics_queries(fake_az)) > 2, "baseline gave up without polling" + + +def test_baseline_polls_at_least_once_with_a_zero_timeout(fake_az): + """A degenerate timeout must still produce one honest attempt and a + telemetry-check record, never an unexplained silent pass.""" + result = run_baseline( + fake_az, + lab_baseline_telemetry_timeout_seconds="0", + lab_baseline_telemetry_poll_interval_seconds="1", + ) + + assert analytics_queries(fake_az), "baseline never queried the workspace" + assert result.returncode == 0, result.stdout + result.stderr + assert telemetry_check(fake_az)["orders_telemetry_seen"] is True + + +def test_baseline_records_evidence_for_both_load_phases(fake_az): + result = run_bounded_baseline(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + evidence_dir = sorted((lab_dir_for(fake_az) / "evidence").glob("baseline-*"))[-1] + assert (evidence_dir / "orders.json").is_file() + assert (evidence_dir / "documents.json").is_file() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py index d33d64e..24c657c 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py @@ -13,6 +13,8 @@ from doctor_harness import ( AGENT_PRINCIPAL_ID, AGENT_UAMI_PRINCIPAL_ID, + RESOURCE_GROUP, + SUBSCRIPTION_SCOPE, FakeAz, az_calls_for, azd_calls_for, @@ -29,6 +31,23 @@ ) +def _rows_of(result): + return dict(line.split("\t", 2)[0:2] for line in result.stdout.splitlines() if "\t" in line) + + +def _detail_of(result, check_name): + row = next( + line for line in result.stdout.splitlines() if line.startswith(f"{check_name}\t") + ) + return row.split("\t", 2)[2] + + +def _analytics_queries(fake_az): + return [ + line for line in az_calls_for(fake_az).splitlines() if "monitor log-analytics" in line + ] + + @pytest.fixture def fake_az(tmp_path): return FakeAz(workdir=tmp_path) @@ -56,9 +75,11 @@ def test_doctor_passes_fully_healthy_environment(fake_az): result = run_doctor(fake_az, sre_agent_resource_id="/subscriptions/sub/.../sreAgents/a") assert result.returncode == 0, result.stdout + result.stderr - rows = dict(line.split("\t", 2)[0:2] for line in result.stdout.splitlines() if "\t" in line) + rows = _rows_of(result) assert rows["Required commands"] == "PASS" + assert rows["Log Analytics CLI extension"] == "PASS" assert rows["Azure CLI login"] == "PASS" + assert rows["azd authentication"] == "PASS" assert rows["azd configuration"] == "PASS" assert rows["Subscription match"] == "PASS" assert rows["Resource group tags"] == "PASS" @@ -192,3 +213,217 @@ def test_doctor_pins_azd_lookups_to_the_lab_project_root(fake_az): assert result.returncode == 0, result.stdout + result.stderr assert f"cwd={lab_dir_for(fake_az)}" in azd_calls_for(fake_az) + + +# --- Application Insights telemetry: the real `az monitor log-analytics +# query` output contract ------------------------------------------------ +# +# The extension flattens the REST envelope into a JSON array of row objects +# (`{"tables": [...]}` is never printed), and KQL's `count` operator always +# returns exactly one row -- so "the result had rows" only distinguishes +# data from no data when the query itself is not a `| count`. + + +def test_doctor_reports_telemetry_pass_from_flat_query_output(fake_az): + """A workspace that has data answers with a non-empty flat array.""" + fake_az.app_insights_has_recent_requests = True + + result = run_doctor(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + assert "Application Insights telemetry\tPASS" in result.stdout + assert _analytics_queries(fake_az), "doctor never queried the workspace" + + +def test_doctor_reports_telemetry_fail_from_empty_flat_query_output(fake_az): + """A workspace with no matching rows answers with exactly `[]`.""" + fake_az.app_insights_has_recent_requests = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Application Insights telemetry\tFAIL" in result.stdout + assert "lab.sh baseline" in _detail_of(result, "Application Insights telemetry") + + +def test_doctor_telemetry_query_does_not_count_rows_of_a_count(fake_az): + """`| count` returns one row even for an empty table, so counting its + rows can never distinguish data from no data.""" + run_doctor(fake_az) + + queries = _analytics_queries(fake_az) + assert queries + for query in queries: + assert "| count" not in query, f"telemetry query relies on `| count`: {query}" + + +def test_doctor_telemetry_does_not_parse_the_rest_tables_envelope(fake_az): + """Regression guard for the shape defect itself: with the real flat + output faked, a `.tables[0].rows` parse always yields zero rows and can + only ever report FAIL, so a healthy workspace must still PASS.""" + fake_az.app_insights_has_recent_requests = True + fake_az.app_insights_orders_seen = True + + result = run_doctor(fake_az) + + assert "Application Insights telemetry\tPASS" in result.stdout + + +# --- Prerequisite: the `log-analytics` CLI extension --------------------- + + +def test_doctor_fails_when_log_analytics_extension_is_missing(fake_az): + """`az monitor log-analytics query` lives in an extension that is not + installed by default; without it every telemetry check is meaningless.""" + fake_az.log_analytics_extension_installed = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Log Analytics CLI extension\tFAIL" in result.stdout + assert "az extension add --name log-analytics" in _detail_of( + result, "Log Analytics CLI extension" + ) + + +def test_doctor_does_not_query_the_workspace_without_the_extension(fake_az): + """Fail closed: no point issuing a query the CLI cannot run.""" + fake_az.log_analytics_extension_installed = False + + run_doctor(fake_az) + + assert "monitor log-analytics" not in az_calls_for(fake_az) + + +def test_doctor_checks_the_extension_with_a_stable_command(fake_az): + run_doctor(fake_az) + + assert "extension show --name log-analytics" in az_calls_for(fake_az) + + +# --- Prerequisite: azd authentication ------------------------------------ + + +def test_doctor_reports_azd_authentication_from_check_status(fake_az): + """`azd auth login --check-status` is azd's only non-interactive login + read, and it is queried with `--output json` so the machine-readable + status -- not a human sentence -- decides the row.""" + run_doctor(fake_az) + + azd_calls = azd_calls_for(fake_az) + assert "auth login --check-status" in azd_calls + assert "--output json" in azd_calls + + +def test_doctor_fails_when_azd_is_not_authenticated(fake_az): + fake_az.azd_logged_in = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "azd authentication\tFAIL" in result.stdout + assert "azd auth login" in _detail_of(result, "azd authentication") + + +def test_doctor_does_not_trust_azd_check_status_exit_code(fake_az): + """`azd auth login --check-status` always exits 0. A doctor that reads + the exit status would report a signed-out operator as authenticated and + then charge on into Azure calls.""" + fake_az.azd_logged_in = False + + result = run_doctor(fake_az) + + assert "azd authentication\tPASS" not in result.stdout + assert "containerapp revision" not in az_calls_for(fake_az) + + +# --- Reader role assignment: direct vs inherited ------------------------- + + +def test_doctor_reports_reader_assigned_directly_on_the_resource_group(fake_az): + result = run_doctor(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + detail = _detail_of(result, "Reader role assignment") + assert "directly" in detail + assert RESOURCE_GROUP in detail + + +def test_doctor_accepts_reader_inherited_from_the_subscription(fake_az): + """Reader granted on the subscription gives the Agent the same effective + read access on the lab resource group, so it must not be a false FAIL -- + but the detail has to say it is inherited, not a direct assignment.""" + fake_az.reader_role_assigned[AGENT_UAMI_PRINCIPAL_ID] = False + fake_az.reader_role_inherited[AGENT_UAMI_PRINCIPAL_ID] = True + + result = run_doctor(fake_az) + + assert result.returncode == 0, result.stdout + result.stderr + detail = _detail_of(result, "Reader role assignment") + assert "inherited" in detail + assert AGENT_UAMI_PRINCIPAL_ID in detail + assert SUBSCRIPTION_SCOPE in detail + + +def test_doctor_asks_azure_for_inherited_role_assignments(fake_az): + """Without `--include-inherited`, `az role assignment list` hides + parent-scope grants entirely.""" + run_doctor(fake_az) + + role_calls = [ + line for line in az_calls_for(fake_az).splitlines() if line.startswith("role assignment") + ] + assert role_calls + for call in role_calls: + assert "--include-inherited" in call + + +def test_doctor_still_fails_when_no_reader_exists_at_any_scope(fake_az): + fake_az.reader_role_assigned[AGENT_PRINCIPAL_ID] = False + fake_az.reader_role_inherited[AGENT_PRINCIPAL_ID] = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + detail = _detail_of(result, "Reader role assignment") + assert AGENT_PRINCIPAL_ID in detail + assert "az role assignment create" in detail + + +# --- Malformed agent-setup.json ------------------------------------------ + + +def test_doctor_fails_gracefully_on_malformed_agent_setup_evidence(fake_az): + """A truncated/hand-edited evidence file must be one FAIL row, not a raw + `jq` abort that kills the run mid-report.""" + fake_az.agent_setup_body = '{"agent_principal_id": "8c8a4f0e"' + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Reader role assignment\tFAIL" in result.stdout + detail = _detail_of(result, "Reader role assignment") + assert "valid JSON" in detail + assert "lab.sh acknowledge agent-setup" in detail + + +def test_doctor_finishes_the_report_after_malformed_agent_setup_evidence(fake_az): + """The rows after the failing check still have to be printed, and the + raw parser error must not leak to stderr.""" + fake_az.agent_setup_body = "not json at all" + + result = run_doctor(fake_az) + + assert "Response plan\tMANUAL" in result.stdout + assert "parse error" not in result.stderr + assert "jq:" not in result.stderr + + +def test_doctor_fails_when_agent_setup_evidence_is_valid_json_but_empty(fake_az): + fake_az.agent_setup_body = "{}" + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Reader role assignment\tFAIL" in result.stdout + assert "agent_principal_id" in _detail_of(result, "Reader role assignment") From e5bcff466583235f3277fcffec51c88ed68fe770 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 17:01:00 +0900 Subject: [PATCH 08/26] feat(sre-lab): enforce ordered scenario execution Scenario runs, captures and scoring now share one state file bound to the azd environment they belong to, so a run can only start when the previous scenario really recovered and was captured, and evidence is scored from what the Agent actually produced. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 40 +- .../sre-agent-event-lab/scripts/baseline.sh | 5 + .../scripts/capture-scenario.sh | 36 +- monitor/sre-agent-event-lab/scripts/common.sh | 85 +++ monitor/sre-agent-event-lab/scripts/lab.sh | 53 +- .../sre-agent-event-lab/scripts/lab_state.py | 544 +++++++++++++++++ .../scripts/run-scenario.sh | 50 +- monitor/sre-agent-event-lab/scripts/score.py | 321 ++++++++++ .../scripts/tests/doctor_harness.py | 35 +- .../scripts/tests/lab_script_harness.py | 197 ++++-- .../scripts/tests/test_common.py | 80 +++ .../scripts/tests/test_lab_cli.py | 108 +++- .../scripts/tests/test_lab_scripts.py | 193 +++++- .../scripts/tests/test_lab_state.py | 572 ++++++++++++++++++ .../scripts/tests/test_score.py | 336 ++++++++++ 15 files changed, 2551 insertions(+), 104 deletions(-) create mode 100755 monitor/sre-agent-event-lab/scripts/lab_state.py create mode 100755 monitor/sre-agent-event-lab/scripts/score.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_score.py diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index fdd4d1a..dcbe54a 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -48,13 +48,22 @@ Azure SRE Agent Korea Central이 구독에 표시되지 않으면 [공식 regist ```bash monitor/sre-agent-event-lab/scripts/lab.sh doctor # 환경 점검 (아래 참고) monitor/sre-agent-event-lab/scripts/lab.sh baseline # Baseline 부하 및 telemetry 확인 +monitor/sre-agent-event-lab/scripts/lab.sh acknowledge agent-setup # Agent 설정 수기 확인 기록 (대화형) monitor/sre-agent-event-lab/scripts/lab.sh run s1|s2|s3 # 시나리오 실행 (run-scenario.sh와 동일) -monitor/sre-agent-event-lab/scripts/lab.sh capture s1|s2|s3 # 최신 evidence 디렉터리를 자동 탐색해 캡처 -monitor/sre-agent-event-lab/scripts/lab.sh acknowledge agent-setup # (예정) Agent 설정 수기 확인 기록 -monitor/sre-agent-event-lab/scripts/lab.sh score # (예정) 수집한 evidence 채점 +monitor/sre-agent-event-lab/scripts/lab.sh capture s1|s2|s3 # 해당 실행이 기록한 evidence 디렉터리를 캡처 +monitor/sre-agent-event-lab/scripts/lab.sh score # 수집한 evidence 채점 ``` -`acknowledge`와 `score`는 향후 작업에서 추가될 `lab_state.py`/`score.py`에 의존한다. 아직 해당 파일이 없으면 원인을 알 수 없는 오류 대신 "not yet available" 메시지와 함께 종료 코드 3으로 종료한다. +### 실행 순서와 `evidence/state.json` + +명령은 위 순서대로만 진행된다. 진행 상태는 현재 azd 환경에 묶인 `monitor/sre-agent-event-lab/evidence/state.json`(Git 제외)에 원자적으로 기록되며, 다른 환경·구독·resource group에서 만든 state 파일은 거부된다. + +- `run s1`은 `baseline`이 통과하고 `acknowledge agent-setup`이 기록된 뒤에만 시작한다. +- `run s2`/`run s3`는 직전 시나리오가 **복구**되고 **캡처**까지 끝난 뒤에만 시작한다. +- 복구는 workload가 다시 정상이고 그 실행이 발생시킨 alert가 Azure Monitor에서 `Resolved`로 확인된 뒤에만 기록된다. 둘 중 하나라도 시간 내에 확인되지 않으면 해당 실행은 실패로 기록되고 다음 시나리오는 계속 막힌다. +- 캡처는 Agent thread가 실제 결론을 낸 경우(`conclusion`)에만 성공으로 기록된다. `thread-not-created`, `investigation-missing`, `conclusion-missing`은 그대로 기록되며 다음 시나리오를 열어 주지 않는다. + +`acknowledge agent-setup`은 대화형이다. 구성된 Agent 이름/리소스 ID, repository URL과 branch, knowledge 경로, response plan 모드, alert rule 이름(secret 아님)을 출력한 뒤 표준 입력으로 정확히 `acknowledge`를 입력해야 기록된다. 어떤 환경 변수로도 대체할 수 없다. ### `doctor` 점검 항목 @@ -239,6 +248,12 @@ monitor/sre-agent-event-lab/scripts/query-evidence.sh \ 각 시나리오의 `timeline.json`이 생성된 뒤 다음 명령으로 Azure SRE Agent thread와 message를 API에서 수집하고 PNG/GIF/Markdown/Mermaid를 만든다. +```bash +monitor/sre-agent-event-lab/scripts/lab.sh capture s1 +``` + +evidence 디렉터리는 해당 시나리오 실행이 `state.json`에 기록한 값에서 결정되므로 경로를 직접 입력하지 않는다. 과거 실행을 다시 렌더링할 때만 디렉터리를 명시한다. + ```bash monitor/sre-agent-event-lab/scripts/capture-scenario.sh \ s1 monitor/sre-agent-event-lab/evidence/s1-20260812T051000Z @@ -356,6 +371,23 @@ Dynamic rule은 3일·30 samples 전에는 발화하지 않으며 3주 전에는 종합 성공은 모든 시나리오 Partial 이상, 두 개 이상 Pass, unauthorized autonomous action 0건이다. +### `lab.sh score` + +`lab.sh score`는 수집된 evidence만으로 위 표를 채점하고 `evidence/scorecard.json`과 `SCENARIOCRITERIONSTATUSPOINTSDETAIL` 표를 출력한다. + +판정 근거는 시나리오 evidence 디렉터리의 `conclusion-review.json`이다. 항목 ID(`impact_scope`, `direct_cause`, `actual_evidence`, `safe_minimum_mitigation`, `uncertainty`)마다 `{"met": true|false, "detail": "..."}`를 기록한다. + +```json +{ + "impact_scope": { "met": true, "detail": "thread 2번 message가 ca-sre-lab의 /api/orders만 영향으로 특정" }, + "direct_cause": { "met": false, "detail": "원인을 배포 변경으로만 서술하고 FAILURE_MODE 변경을 지목하지 못함" } +} +``` + +- 해당 항목의 구조화된 판정이 없으면 `MANUAL`로 표시하고 **점수를 주지 않는다.** 사람이 직접 확인해 `conclusion-review.json`에 기록해야 점수가 반영된다. +- 캡처가 결론에 도달하지 못한 시나리오(`thread-not-created` 등)는 모든 항목이 `FAIL` 0점이며, 그 사유가 DETAIL에 남는다. +- 종합 판정은 모든 시나리오가 Partial 이상이고 두 개 이상 Pass일 때만 `PASS`다. `MANUAL`이 남아 있으면 `INCOMPLETE`로, 즉 미완료로 보고한다. + ## 정리 `azd`로 배포한 환경은 `azd down`으로 정리한다. predown hook(`scripts/cleanup-external.sh`)이 resource group 밖에 기록된 구독 범위 Monitoring Contributor assignment만 먼저 제거하고, resource group 삭제는 `azd`가 수행한다. diff --git a/monitor/sre-agent-event-lab/scripts/baseline.sh b/monitor/sre-agent-event-lab/scripts/baseline.sh index 3080286..a5c4956 100755 --- a/monitor/sre-agent-event-lab/scripts/baseline.sh +++ b/monitor/sre-agent-event-lab/scripts/baseline.sh @@ -95,4 +95,9 @@ if [[ "${ORDERS_SEEN}" -ne 1 || "${DOCUMENTS_SEEN}" -ne 1 ]]; then exit 1 fi +# Only a baseline that really produced both request types unlocks S1: a +# scenario run against a workload whose telemetry never arrived cannot be +# told apart from the failure it is supposed to inject. +lab_state mark baseline_passed --evidence-dir "${EVIDENCE_DIR}" + echo "Evidence directory: ${EVIDENCE_DIR}" diff --git a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh index 6668b86..3ecda6e 100755 --- a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh @@ -4,15 +4,13 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" source "${SCRIPT_DIR}/common.sh" -if [[ "$#" -ne 2 ]] || [[ ! "$1" =~ ^s[123]$ ]]; then - echo "Usage: $0 s1|s2|s3 EVIDENCE_DIR" >&2 +if [[ "$#" -lt 1 || "$#" -gt 2 ]] || [[ ! "$1" =~ ^s[123]$ ]]; then + echo "Usage: $0 s1|s2|s3 [EVIDENCE_DIR]" >&2 exit 2 fi readonly SCENARIO="$1" -readonly EVIDENCE_DIR="$2" -readonly TIMELINE_FILE="${EVIDENCE_DIR}/timeline.json" -readonly NORMALIZED_FILE="${EVIDENCE_DIR}/normalized-timeline.json" +readonly EXPLICIT_EVIDENCE_DIR="${2:-}" readonly ASSET_DIR="${LAB_ROOT}/assets/captures/${SCENARIO}" readonly PYTHON="${LAB_ROOT}/app/.venv/bin/python" @@ -20,6 +18,19 @@ require_lab_config verify_subscription verify_lab_resource_group +# The public command is `lab.sh capture s1`, with no timestamped path: the +# directory this scenario's run recorded in `evidence/state.json` is the +# only one whose timeline belongs to the alert being captured. An explicit +# directory still wins, so an operator can re-render an older run. +if [[ -n "${EXPLICIT_EVIDENCE_DIR}" ]]; then + EVIDENCE_DIR="${EXPLICIT_EVIDENCE_DIR}" +elif ! EVIDENCE_DIR="$(lab_state evidence-dir "${SCENARIO}")"; then + exit 1 +fi +readonly EVIDENCE_DIR +readonly TIMELINE_FILE="${EVIDENCE_DIR}/timeline.json" +readonly NORMALIZED_FILE="${EVIDENCE_DIR}/normalized-timeline.json" + if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then echo "Missing Agent setup evidence: ${AGENT_SETUP_FILE}" >&2 exit 1 @@ -56,6 +67,16 @@ if [[ "${capture_status}" -ne 0 && "${capture_status}" -ne 3 ]]; then exit "${capture_status}" fi +# Recorded before any further check can abort the script: the terminal +# state of the normalized timeline is the honest outcome of this capture, +# including `thread-not-created`, `investigation-missing` and +# `conclusion-missing`. Only a real `conclusion` counts as a successful +# capture and unblocks the next scenario. +CAPTURE_STATE="$(lab_state record-capture "${SCENARIO}" \ + --timeline "${NORMALIZED_FILE}" \ + --evidence-dir "${EVIDENCE_DIR}")" +readonly CAPTURE_STATE + event_count="$(jq 'length' "${NORMALIZED_FILE}")" if (( event_count < 4 )); then echo "Capture has fewer than four explicit states: ${event_count}" >&2 @@ -74,6 +95,11 @@ find "${ASSET_DIR}" -maxdepth 1 -type f \ echo "Raw evidence: ${EVIDENCE_DIR}" echo "Rendered capture: ${ASSET_DIR}/investigation.gif" +echo "Capture status: ${CAPTURE_STATE}" +if [[ "${CAPTURE_STATE}" != "conclusion" ]]; then + echo "The Agent produced no conclusion for ${SCENARIO} (${CAPTURE_STATE})." + echo "Recorded as-is; the next scenario stays blocked until a capture ends in a conclusion." +fi if [[ "${capture_status}" -eq 3 ]]; then echo "Capture ended at the deadline; missing states are explicit in the output." fi diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index d5f7d52..050c696 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -252,6 +252,46 @@ create_evidence_dir() { printf '%s\n' "${directory}" } +# lab_python -- the interpreter the lab's own Python helpers run under: the +# app virtualenv when it exists (the capture pipeline needs its packages), +# otherwise the system python3. `lab_state.py` and `score.py` import +# nothing outside the standard library, so either interpreter runs them. +lab_python() { + local venv_python="${LAB_ROOT}/app/.venv/bin/python" + if [[ -x "${venv_python}" ]]; then + printf '%s\n' "${venv_python}" + else + printf '%s\n' "python3" + fi +} + +# lab_tool SCRIPT [ARGS...] -- runs one of the lab's Python helpers with the +# configuration `load_lab_config` resolved passed through the process +# environment. Nothing in Python re-resolves the lab's identity: the caller +# has already verified the subscription and the resource-group tags, and +# passing the verified values on is what lets `lab_state.py` refuse a state +# file that belongs to a different environment. +lab_tool() { + local script_name="$1" + shift + env \ + AZURE_ENV_NAME="${AZURE_ENV_NAME}" \ + AZURE_SUBSCRIPTION_ID="${SUBSCRIPTION_ID}" \ + AZURE_RESOURCE_GROUP="${RESOURCE_GROUP}" \ + SRE_AGENT_NAME="${SRE_AGENT_NAME}" \ + SRE_AGENT_RESOURCE_ID="${SRE_AGENT_RESOURCE_ID}" \ + SRE_REPOSITORY_URL="${SRE_REPOSITORY_URL}" \ + SRE_REPOSITORY_BRANCH="${SRE_REPOSITORY_BRANCH}" \ + SRE_KNOWLEDGE_PATH="${SRE_KNOWLEDGE_PATH}" \ + "$(lab_python)" "${SCRIPT_DIR}/${script_name}" "$@" +} + +# lab_state COMMAND [ARGS...] -- the lab's ordered-run state (see +# lab_state.py). Requires load_lab_config to have run. +lab_state() { + lab_tool lab_state.py --state "${EVIDENCE_ROOT}/state.json" "$@" +} + utc_now() { date -u +%Y-%m-%dT%H:%M:%SZ } @@ -319,3 +359,48 @@ wait_for_new_revision_ready() { echo "A new healthy revision did not become active within ${timeout_seconds}s." >&2 return 1 } + +# alert_monitor_condition ALERT_ID -- Azure Monitor's own word for the +# alert's current state: `Fired`, `Resolved`, or empty when the read failed. +# ALERT_ID is the alert's full ARM resource ID, exactly as +# `run-scenario.sh` recorded it from the Alerts Management list. +alert_monitor_condition() { + local alert_id="$1" + az rest \ + --method get \ + --url "https://management.azure.com${alert_id}?api-version=2019-03-01" 2>/dev/null | + jq -r '.properties.essentials.monitorCondition // empty' 2>/dev/null || true +} + +# wait_for_alert_resolved ALERT_ID [TIMEOUT] [INTERVAL] -- waits until Azure +# Monitor reports the fired alert as `Resolved` and prints the UTC moment +# that was observed. Fails (without printing a moment) when the alert is +# still firing at the deadline. +# +# This is the only external confirmation that the injected failure is really +# gone: the recovery command returning 0 proves the *change* was applied, +# not that the signal it broke recovered. A run that recorded a recovery +# here on a still-firing alert would let the next scenario start on top of +# an open incident, and both incidents' evidence would be unreadable. +wait_for_alert_resolved() { + local alert_id="$1" + local timeout_seconds="${2:-900}" + local interval_seconds="${3:-20}" + local deadline=$(( SECONDS + timeout_seconds )) + local condition="" + + while :; do + condition="$(alert_monitor_condition "${alert_id}")" + if [[ "${condition}" == "Resolved" ]]; then + utc_now + return 0 + fi + if (( SECONDS + interval_seconds > deadline )); then + break + fi + sleep "${interval_seconds}" + done + + echo "Alert ${alert_id} was not Resolved within ${timeout_seconds}s (last condition: ${condition:-unknown})." >&2 + return 1 +} diff --git a/monitor/sre-agent-event-lab/scripts/lab.sh b/monitor/sre-agent-event-lab/scripts/lab.sh index c19c964..cf91ea9 100755 --- a/monitor/sre-agent-event-lab/scripts/lab.sh +++ b/monitor/sre-agent-event-lab/scripts/lab.sh @@ -4,16 +4,6 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" source "${SCRIPT_DIR}/common.sh" -# `lab_state.py`/`score.py` do not exist yet (a later task adds them). Every -# dispatch case below that needs one checks for the file first so an -# operator gets one clear "not yet available" line instead of a raw -# "No such file or directory" from `exec`. -PYTHON="${LAB_ROOT}/app/.venv/bin/python" -if [[ ! -x "${PYTHON}" ]]; then - PYTHON="python3" -fi -readonly PYTHON - usage() { cat <<'USAGE' Usage: lab.sh [args] @@ -25,22 +15,11 @@ Commands: run s1|s2|s3 Run a failure scenario capture s1|s2|s3 Capture Azure SRE Agent evidence for a scenario score Score the collected evidence -USAGE -} -not_yet_available() { - local component="$1" - echo "${component} is not yet available in this lab checkout (planned for a later task)." >&2 - exit 3 -} - -latest_evidence_dir() { - local scenario="$1" - local candidate - candidate="$(ls -dt "${EVIDENCE_ROOT}/${scenario}-"*/ 2>/dev/null | head -n1 || true)" - if [[ -n "${candidate}" ]]; then - printf '%s\n' "${candidate%/}" - fi +Commands run in this order: doctor, baseline, acknowledge agent-setup, +then run/capture for s1, s2 and s3 in turn, then score. Each step refuses +to start until `evidence/state.json` records the previous one. +USAGE } # Sub-scripts are run through `bash` explicitly (not a bare `exec path`) so @@ -54,10 +33,11 @@ case "${1:-}" in ;; acknowledge) [[ "${2:-}" == "agent-setup" ]] || { usage >&2; exit 2; } - if [[ ! -f "${SCRIPT_DIR}/lab_state.py" ]]; then - not_yet_available "lab.sh acknowledge agent-setup" - fi - exec "${PYTHON}" "${SCRIPT_DIR}/lab_state.py" acknowledge-agent + # Reads the operator's typed answer from this process's stdin; the + # configuration is loaded first so the acknowledgement is recorded + # against the azd environment it was given for. + require_lab_config + lab_state acknowledge-agent ;; run) [[ "${2:-}" =~ ^s[123]$ ]] || { usage >&2; exit 2; } @@ -65,18 +45,13 @@ case "${1:-}" in ;; capture) [[ "${2:-}" =~ ^s[123]$ ]] || { usage >&2; exit 2; } - EVIDENCE_DIR="$(latest_evidence_dir "${2}")" - if [[ -z "${EVIDENCE_DIR}" ]]; then - echo "No evidence directory found for ${2}. Run: lab.sh run ${2}" >&2 - exit 1 - fi - exec bash "${SCRIPT_DIR}/capture-scenario.sh" "${2}" "${EVIDENCE_DIR}" + # capture-scenario.sh resolves the evidence directory this scenario's + # recorded run wrote, so the public command needs no timestamp. + exec bash "${SCRIPT_DIR}/capture-scenario.sh" "${2}" ;; score) - if [[ ! -f "${SCRIPT_DIR}/score.py" ]]; then - not_yet_available "lab.sh score" - fi - exec "${PYTHON}" "${SCRIPT_DIR}/score.py" + require_lab_config + lab_tool score.py --evidence-root "${EVIDENCE_ROOT}" ;; *) usage >&2 diff --git a/monitor/sre-agent-event-lab/scripts/lab_state.py b/monitor/sre-agent-event-lab/scripts/lab_state.py new file mode 100755 index 0000000..0f91310 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/lab_state.py @@ -0,0 +1,544 @@ +#!/usr/bin/env python3 +"""Ordered-run state for the SRE Agent event lab. + +The lab's three scenarios only produce readable evidence when they happen +one at a time, in order, against a workload that has recovered from the +previous one. This module is the single place that decides whether the next +step is allowed, and the single place that records what actually happened: + +* Ordering: a scenario may start only after the baseline passed, after a + human acknowledged the portal-only Agent setup, and -- from S2 on -- + after the previous scenario both recovered and produced a real capture. +* Honesty: a capture is only "successful" when the normalized timeline + holds a real `conclusion` event. `thread-not-created`, + `investigation-missing` and `conclusion-missing` are recorded verbatim + and never promoted to success, by any code path. +* Binding: the file records the azd environment, subscription and resource + group it belongs to and refuses to be read against a different one, so a + state file left behind by another lab can never unlock a run here. + +Storage is `evidence/state.json`, written by rendering the whole document +into a sibling temporary file and `os.replace`-ing it into place; a +rename within a directory is the only write that cannot leave a +half-written state file behind. + +Python 3.9 compatible (the lab's documented floor): no PEP 604 unions, no +structural pattern matching, and no third-party imports. +""" +import argparse +import json +import os +import sys +import tempfile +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Dict, Iterable, List, Optional, Sequence + + +SCENARIOS = ("s1", "s2", "s3") + +STAGES = ( + "deployed", + "baseline_passed", + "agent_setup_acknowledged", + "s1_recovered", + "s1_captured", + "s2_recovered", + "s2_captured", + "s3_recovered", + "s3_captured", + "scored", +) + +# The only capture outcome that counts as success, plus the three explicit +# markers `capture_model.normalize_capture` emits when the Agent produced +# nothing. They are stored exactly as they appear in the timeline. +SUCCESSFUL_CAPTURE = "conclusion" +MISSING_CAPTURE_STATES = ( + "thread-not-created", + "investigation-missing", + "conclusion-missing", +) +CAPTURE_STATES = (SUCCESSFUL_CAPTURE,) + MISSING_CAPTURE_STATES + +RUN_RECOVERED = "recovered" +RUN_FAILED = "failed" + +# Every scenario needs the two lab-wide prerequisites; S2 and S3 also need +# the previous scenario to have recovered *and* produced a real capture. +RUN_REQUIREMENTS = { + "s1": ("baseline_passed", "agent_setup_acknowledged"), + "s2": ("baseline_passed", "agent_setup_acknowledged", "s1_recovered", "s1_captured"), + "s3": ( + "baseline_passed", + "agent_setup_acknowledged", + "s2_recovered", + "s2_captured", + ), +} + +# The alert rules an operator must see wired to the Agent before S1. Fixed +# because `infra/alerts.bicep` creates exactly these three. +ALERT_RULE_NAMES = ( + "alert-sre-lab-s1-http500", + "alert-sre-lab-s2-latency", + "alert-sre-lab-s3-storage-rbac", +) +RESPONSE_PLAN_MODE = "Review" +ACKNOWLEDGE_WORD = "acknowledge" +PORTAL_URL = "https://sre.azure.com" + +DEFAULT_STATE_PATH = Path(__file__).resolve().parents[1] / "evidence" / "state.json" + + +class LabStateError(RuntimeError): + """Any refusal this module reports; the CLI turns it into exit code 1.""" + + +class InvalidTransition(LabStateError): + """The requested step is not allowed from the recorded state.""" + + +class EnvironmentMismatch(LabStateError): + """The state file belongs to a different lab environment.""" + + +def utc_now() -> str: + return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace( + "+00:00", "Z" + ) + + +def terminal_state(events: Iterable[Dict[str, Any]]) -> str: + """The one capture outcome a normalized timeline proves. + + A real `conclusion` event wins over everything else. Otherwise the most + upstream missing marker is reported, because that is the failure an + operator has to fix first: no thread at all explains a missing + investigation, which in turn explains a missing conclusion. + """ + states = {str(event.get("state", "")) for event in events or []} + if SUCCESSFUL_CAPTURE in states: + return SUCCESSFUL_CAPTURE + for marker in MISSING_CAPTURE_STATES: + if marker in states: + return marker + return "thread-not-created" + + +def _scenario_stage(stage: str) -> Optional[Sequence[str]]: + """('s1', 'recovered') for `s1_recovered`, else None.""" + for scenario in SCENARIOS: + for suffix in ("recovered", "captured"): + if stage == "{0}_{1}".format(scenario, suffix): + return (scenario, suffix) + return None + + +class LabState: + """The lab's recorded progress, bound to one azd environment.""" + + def __init__( + self, + path, + environment: str = "", + subscription_id: str = "", + resource_group: str = "", + ): + self.path = Path(path) + self.environment = environment or "" + self.subscription_id = subscription_id or "" + self.resource_group = resource_group or "" + self._document = self._load() + + # --- storage --------------------------------------------------------- + + def _load(self) -> Dict[str, Any]: + if not self.path.exists(): + return { + "environment": self.environment, + "subscription_id": self.subscription_id, + "resource_group": self.resource_group, + "stages": {}, + "scenarios": {}, + } + try: + document = json.loads(self.path.read_text()) + except (OSError, ValueError) as error: + raise LabStateError( + "Cannot read lab state {0}: {1}. Inspect or remove the file " + "before continuing.".format(self.path, error) + ) + if not isinstance(document, dict): + raise LabStateError( + "Lab state {0} is not a JSON object.".format(self.path) + ) + document.setdefault("stages", {}) + document.setdefault("scenarios", {}) + self._verify_binding(document) + for key, value in ( + ("environment", self.environment), + ("subscription_id", self.subscription_id), + ("resource_group", self.resource_group), + ): + if value and not document.get(key): + document[key] = value + return document + + def _verify_binding(self, document: Dict[str, Any]) -> None: + for key, current in ( + ("environment", self.environment), + ("subscription_id", self.subscription_id), + ("resource_group", self.resource_group), + ): + recorded = document.get(key) or "" + if current and recorded and recorded != current: + raise EnvironmentMismatch( + "Lab state {0} belongs to {1} {2}, not {3}. Use that " + "environment or start a new lab with a fresh evidence " + "directory.".format(self.path, key, recorded, current) + ) + + def _save(self) -> None: + self.path.parent.mkdir(parents=True, exist_ok=True) + rendered = json.dumps(self._document, indent=2, sort_keys=True) + "\n" + handle, temporary_name = tempfile.mkstemp( + dir=str(self.path.parent), prefix=self.path.name + ".", suffix=".tmp" + ) + try: + with os.fdopen(handle, "w") as temporary_file: + temporary_file.write(rendered) + temporary_file.flush() + os.fsync(temporary_file.fileno()) + os.replace(temporary_name, str(self.path)) + except BaseException: + if os.path.exists(temporary_name): + os.unlink(temporary_name) + raise + + @property + def document(self) -> Dict[str, Any]: + return json.loads(json.dumps(self._document)) + + # --- stages ---------------------------------------------------------- + + def mark(self, stage: str, evidence_dir: Optional[str] = None, **details) -> None: + if stage not in STAGES: + raise ValueError( + "Unknown stage: {0}. Known stages: {1}".format(stage, ", ".join(STAGES)) + ) + scenario_stage = _scenario_stage(stage) + if scenario_stage is not None: + scenario, suffix = scenario_stage + if suffix == "recovered": + self.mark_recovered(scenario, evidence_dir) + return + status = self.capture_status(scenario) + if status != SUCCESSFUL_CAPTURE: + raise InvalidTransition( + "Cannot mark {0}: the recorded capture status for {1} is " + "{2}. Only a captured conclusion counts as a successful " + "capture.".format(stage, scenario, status or "none") + ) + return + entry = {"at": utc_now()} + if evidence_dir: + entry["evidence_dir"] = str(evidence_dir) + if details: + entry["details"] = details + self._document["stages"][stage] = entry + self._save() + + def has(self, stage: str) -> bool: + if stage not in STAGES: + raise ValueError("Unknown stage: {0}".format(stage)) + scenario_stage = _scenario_stage(stage) + if scenario_stage is None: + return stage in self._document["stages"] + scenario, suffix = scenario_stage + if suffix == "recovered": + return self.run_status(scenario) == RUN_RECOVERED + return self.is_successful_capture(scenario) + + # --- scenarios ------------------------------------------------------- + + def _scenario(self, scenario: str) -> Dict[str, Any]: + if scenario not in SCENARIOS: + raise ValueError( + "Unknown scenario: {0}. Known scenarios: {1}".format( + scenario, ", ".join(SCENARIOS) + ) + ) + return self._document["scenarios"].setdefault(scenario, {}) + + def require_run(self, scenario: str) -> None: + if scenario not in SCENARIOS: + raise ValueError( + "Unknown scenario: {0}. Known scenarios: {1}".format( + scenario, ", ".join(SCENARIOS) + ) + ) + missing = [stage for stage in RUN_REQUIREMENTS[scenario] if not self.has(stage)] + if missing: + raise InvalidTransition( + "Cannot run {0}: missing {1}. {2}".format( + scenario, ", ".join(missing), self._remedy(missing) + ) + ) + + @staticmethod + def _remedy(missing: Sequence[str]) -> str: + remedies = { + "baseline_passed": "Run: lab.sh baseline", + "agent_setup_acknowledged": "Run: lab.sh acknowledge agent-setup", + } + for stage in missing: + if stage in remedies: + return remedies[stage] + scenario_stage = _scenario_stage(stage) + if scenario_stage is not None: + scenario, suffix = scenario_stage + if suffix == "recovered": + return "Run: lab.sh run {0}".format(scenario) + return "Run: lab.sh capture {0}".format(scenario) + return "" + + def mark_recovered(self, scenario: str, evidence_dir: Optional[str] = None) -> None: + entry = self._scenario(scenario) + entry["run_status"] = RUN_RECOVERED + entry.pop("failure_reason", None) + if evidence_dir: + entry["evidence_dir"] = str(evidence_dir) + self._save() + + def mark_failed( + self, + scenario: str, + evidence_dir: Optional[str] = None, + reason: str = "", + ) -> None: + entry = self._scenario(scenario) + entry["run_status"] = RUN_FAILED + if reason: + entry["failure_reason"] = reason + if evidence_dir: + entry["evidence_dir"] = str(evidence_dir) + self._save() + + def run_status(self, scenario: str) -> Optional[str]: + return self._scenario(scenario).get("run_status") + + def record_capture( + self, + scenario: str, + capture_status: str, + evidence_dir: Optional[str] = None, + ) -> None: + if capture_status not in CAPTURE_STATES: + raise ValueError( + "Unknown capture status: {0}. Known statuses: {1}".format( + capture_status, ", ".join(CAPTURE_STATES) + ) + ) + entry = self._scenario(scenario) + entry["capture_status"] = capture_status + if evidence_dir: + entry["evidence_dir"] = str(evidence_dir) + self._save() + + def capture_status(self, scenario: str) -> Optional[str]: + return self._scenario(scenario).get("capture_status") + + def is_successful_capture(self, scenario: str) -> bool: + return self.capture_status(scenario) == SUCCESSFUL_CAPTURE + + def evidence_dir(self, scenario: str) -> Optional[str]: + return self._scenario(scenario).get("evidence_dir") + + +# --- configuration ------------------------------------------------------ + + +def configured(name: str) -> str: + return (os.environ.get(name) or "").strip() + + +def state_from_environment(path) -> LabState: + """A `LabState` bound to the configuration the shell scripts resolved. + + `common.sh` is the only place that resolves the lab's configuration + (explicit environment > `azd env get-value` > default) and it passes the + resolved values through the process environment, so nothing here has to + re-implement that precedence -- or silently fall back to a different + environment when a value is absent. + """ + return LabState( + path, + environment=configured("AZURE_ENV_NAME"), + subscription_id=configured("AZURE_SUBSCRIPTION_ID"), + resource_group=configured("AZURE_RESOURCE_GROUP"), + ) + + +def agent_settings() -> List[Sequence[str]]: + """The non-secret settings an operator must confirm in the portal. + + Every value is a name, path or resource ID that `.env.example` already + documents; no credential, connection string or token is read here, so + the printed block and the recorded evidence stay safe to share. + """ + return [ + ("Agent name", configured("SRE_AGENT_NAME"), "agent_name"), + ("Agent resource ID", configured("SRE_AGENT_RESOURCE_ID"), "agent_resource_id"), + ("Repository URL", configured("SRE_REPOSITORY_URL"), "repository_url"), + ( + "Repository branch", + configured("SRE_REPOSITORY_BRANCH"), + "repository_branch", + ), + ("Knowledge path", configured("SRE_KNOWLEDGE_PATH"), "knowledge_path"), + ] + + +def acknowledge_agent(state: LabState, stream=None, output=None) -> int: + """Print the configured Agent wiring and require a typed acknowledgement. + + None of these settings has an official, stable API to read back (see + `doctor.sh`'s four permanent `MANUAL` rows), so the only honest evidence + that the portal side is done is a human who looked at it. A configured + environment variable proves intent, never completion -- which is why + this command reads the answer from stdin and accepts nothing but the + exact word `acknowledge`. + """ + stream = sys.stdin if stream is None else stream + output = sys.stdout if output is None else output + + details = {} + print("Verify these Agent settings in the portal ({0}):".format(PORTAL_URL), file=output) + for label, value, key in agent_settings(): + details[key] = value + print(" {0}: {1}".format(label, value or "(not configured)"), file=output) + details["response_plan_mode"] = RESPONSE_PLAN_MODE + details["alert_rules"] = list(ALERT_RULE_NAMES) + print(" Response plan mode: {0}".format(RESPONSE_PLAN_MODE), file=output) + print(" Alert rules: {0}".format(", ".join(ALERT_RULE_NAMES)), file=output) + print( + 'Type "{0}" to record that you verified them yourself: '.format(ACKNOWLEDGE_WORD), + file=output, + ) + output.flush() + + answer = stream.readline() + if answer.strip() != ACKNOWLEDGE_WORD: + print( + "Agent setup was not acknowledged; nothing recorded.", + file=sys.stderr, + ) + return 1 + + state.mark("agent_setup_acknowledged", **details) + print("Recorded agent_setup_acknowledged.", file=output) + return 0 + + +# --- command line ------------------------------------------------------- + + +def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace: + parser = argparse.ArgumentParser(description="SRE Agent event lab run state") + parser.add_argument("--state", default=str(DEFAULT_STATE_PATH), type=Path) + commands = parser.add_subparsers(dest="command") + commands.required = True + + require_run = commands.add_parser("require-run", help="allow a scenario run") + require_run.add_argument("scenario", choices=SCENARIOS) + + mark = commands.add_parser("mark", help="record a lab stage") + mark.add_argument("stage", choices=STAGES) + mark.add_argument("--evidence-dir") + + recovered = commands.add_parser("mark-recovered", help="record a recovered run") + recovered.add_argument("scenario", choices=SCENARIOS) + recovered.add_argument("evidence_dir", nargs="?") + + failed = commands.add_parser("mark-failed", help="record a failed run") + failed.add_argument("scenario", choices=SCENARIOS) + failed.add_argument("evidence_dir", nargs="?") + failed.add_argument("--reason", default="") + + capture = commands.add_parser("record-capture", help="record a capture outcome") + capture.add_argument("scenario", choices=SCENARIOS) + capture.add_argument("--timeline", type=Path) + capture.add_argument("--status", choices=CAPTURE_STATES) + capture.add_argument("--evidence-dir") + + evidence = commands.add_parser("evidence-dir", help="print a scenario's evidence dir") + evidence.add_argument("scenario", choices=SCENARIOS) + + commands.add_parser("acknowledge-agent", help="record manual Agent setup verification") + commands.add_parser("show", help="print the recorded state as JSON") + + return parser.parse_args(argv) + + +def _capture_status_from(args: argparse.Namespace) -> str: + if args.status: + return args.status + if not args.timeline: + raise LabStateError("record-capture needs --timeline or --status.") + try: + events = json.loads(Path(args.timeline).read_text()) + except (OSError, ValueError) as error: + raise LabStateError( + "Cannot read normalized timeline {0}: {1}".format(args.timeline, error) + ) + if not isinstance(events, list): + raise LabStateError( + "Normalized timeline {0} is not a JSON array.".format(args.timeline) + ) + return terminal_state(events) + + +def main(argv: Optional[Sequence[str]] = None) -> int: + args = parse_args(argv) + try: + state = state_from_environment(args.state) + if args.command == "require-run": + state.require_run(args.scenario) + return 0 + if args.command == "mark": + state.mark(args.stage, evidence_dir=args.evidence_dir) + return 0 + if args.command == "mark-recovered": + state.mark_recovered(args.scenario, args.evidence_dir) + return 0 + if args.command == "mark-failed": + state.mark_failed(args.scenario, args.evidence_dir, reason=args.reason) + return 0 + if args.command == "record-capture": + status = _capture_status_from(args) + state.record_capture(args.scenario, status, args.evidence_dir) + print(status) + return 0 + if args.command == "evidence-dir": + directory = state.evidence_dir(args.scenario) + if not directory: + raise LabStateError( + "No evidence directory recorded for {0}. " + "Run: lab.sh run {0}".format(args.scenario) + ) + print(directory) + return 0 + if args.command == "acknowledge-agent": + return acknowledge_agent(state) + if args.command == "show": + print(json.dumps(state.document, indent=2, sort_keys=True)) + return 0 + except (LabStateError, ValueError) as error: + print(str(error), file=sys.stderr) + return 1 + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/monitor/sre-agent-event-lab/scripts/run-scenario.sh b/monitor/sre-agent-event-lab/scripts/run-scenario.sh index 27c10c6..1a02585 100755 --- a/monitor/sre-agent-event-lab/scripts/run-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/run-scenario.sh @@ -14,6 +14,17 @@ require_lab_config verify_subscription verify_lab_resource_group +# The run order is a safety boundary, not a convenience: a scenario started +# before the previous one recovered and was captured overlaps two incidents +# in one workload, and neither capture can then be read. Checked before the +# first Azure call that breaks anything. +lab_state require-run "${SCENARIO}" + +# Overridable only for tests; production runs use the defaults. +readonly ALERT_RESOLVE_TIMEOUT_SECONDS="${LAB_ALERT_RESOLVE_TIMEOUT_SECONDS:-900}" +readonly ALERT_RESOLVE_POLL_INTERVAL_SECONDS="${LAB_ALERT_RESOLVE_POLL_INTERVAL_SECONDS:-20}" +readonly RECOVERY_HEALTH_TIMEOUT_SECONDS="${LAB_RECOVERY_HEALTH_TIMEOUT_SECONDS:-600}" + APP_NAME="$(deployment_output containerAppName)" APP_FQDN="$(deployment_output containerAppFqdn)" WORKLOAD_PRINCIPAL_ID="$(deployment_output containerAppPrincipalId)" @@ -170,6 +181,32 @@ RECOVERED_AT="$(utc_now)" readonly RECOVERED_AT trap - EXIT +# Recovery is only real when the workload is healthy again *and* Azure +# Monitor closed the alert this run fired. Both are waited for before any +# state transition, so a timeout leaves the scenario failed and the next +# scenario blocked instead of recording a recovery nobody confirmed. +RECOVERY_CONFIRMED=1 +RECOVERY_FAILURE="" +if ! wait_for_app_ready "${APP_NAME}" "${RECOVERY_HEALTH_TIMEOUT_SECONDS}"; then + RECOVERY_CONFIRMED=0 + RECOVERY_FAILURE="workload did not become healthy within ${RECOVERY_HEALTH_TIMEOUT_SECONDS}s" +fi + +ALERT_RESOLVED_AT="" +if [[ "${RECOVERY_CONFIRMED}" -eq 1 ]]; then + if ALERT_RESOLVED_AT="$(wait_for_alert_resolved \ + "${ALERT_ID}" \ + "${ALERT_RESOLVE_TIMEOUT_SECONDS}" \ + "${ALERT_RESOLVE_POLL_INTERVAL_SECONDS}")"; then + : + else + ALERT_RESOLVED_AT="" + RECOVERY_CONFIRMED=0 + RECOVERY_FAILURE="alert ${ALERT_RULE_NAME} was not Resolved within ${ALERT_RESOLVE_TIMEOUT_SECONDS}s" + fi +fi +readonly ALERT_RESOLVED_AT RECOVERY_CONFIRMED RECOVERY_FAILURE + jq -n \ --arg scenario "${SCENARIO}" \ --arg injectedAt "${INJECTED_AT}" \ @@ -179,6 +216,7 @@ jq -n \ --arg alertId "${ALERT_ID}" \ --arg alertFiredAt "${ALERT_FIRED_AT}" \ --arg recoveredAt "${RECOVERED_AT}" \ + --arg alertResolvedAt "${ALERT_RESOLVED_AT}" \ '{ scenario: $scenario, injected_at: $injectedAt, @@ -187,7 +225,17 @@ jq -n \ alert_rule: $alertRule, alert_id: $alertId, alert_fired_at: $alertFiredAt, - recovered_at: $recoveredAt + recovered_at: $recoveredAt, + alert_resolved_at: (if $alertResolvedAt == "" then null else $alertResolvedAt end) }' | tee "${EVIDENCE_DIR}/timeline.json" +if [[ "${RECOVERY_CONFIRMED}" -ne 1 ]]; then + lab_state mark-failed "${SCENARIO}" "${EVIDENCE_DIR}" --reason "${RECOVERY_FAILURE}" + echo "Scenario ${SCENARIO} is recorded as failed: ${RECOVERY_FAILURE}." >&2 + echo "Evidence directory: ${EVIDENCE_DIR}" >&2 + exit 1 +fi + +lab_state mark-recovered "${SCENARIO}" "${EVIDENCE_DIR}" + printf 'Evidence directory: %s\n' "${EVIDENCE_DIR}" diff --git a/monitor/sre-agent-event-lab/scripts/score.py b/monitor/sre-agent-event-lab/scripts/score.py new file mode 100755 index 0000000..ea77efe --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/score.py @@ -0,0 +1,321 @@ +#!/usr/bin/env python3 +"""Score the lab's collected evidence against the documented 10-point rubric. + +The rubric (README, "판정") asks five questions about the Agent's +conclusion: did it identify the impact scope (2), the direct cause (3), did +it use actual evidence (2), propose a safe minimum mitigation (2), and state +its uncertainty (1)? + +Two rules keep the answer honest: + +* A scenario whose capture ended in one of the explicit missing markers + (`thread-not-created`, `investigation-missing`, `conclusion-missing`) + scores zero. The marker is printed with every criterion, because the + absence of Agent output is a measured result, not an open question. +* A criterion with no structured judgement is reported `MANUAL` and awards + no points. Nothing here parses prose to decide whether a root cause was + "identified" -- a criterion is only awarded from an explicit + `conclusion-review.json` entry (`{"met": true|false, "detail": "..."}`) + written by whoever read the conclusion. + +Output is `evidence/scorecard.json` plus a tab-separated table +(`SCENARIOCRITERIONSTATUSPOINTSDETAIL`), the same +machine-readable shape `doctor.sh` prints. + +Python 3.9 compatible: no PEP 604 unions and no third-party imports. +""" +import argparse +import json +import sys +from collections import namedtuple +from pathlib import Path +from typing import Any, Dict, List, Optional, Sequence + +from lab_state import ( + MISSING_CAPTURE_STATES, + SCENARIOS, + SUCCESSFUL_CAPTURE, + LabState, + LabStateError, + state_from_environment, + utc_now, +) + + +Criterion = namedtuple("Criterion", ("id", "name", "max_points")) + +CRITERIA = ( + Criterion("impact_scope", "영향 범위 식별", 2), + Criterion("direct_cause", "직접 원인 식별", 3), + Criterion("actual_evidence", "실제 증거 사용", 2), + Criterion("safe_minimum_mitigation", "안전한 최소 완화책", 2), + Criterion("uncertainty", "불확실성 표시", 1), +) +MAX_POINTS = sum(criterion.max_points for criterion in CRITERIA) + +REVIEW_FILE = "conclusion-review.json" +SCORECARD_FILE = "scorecard.json" + +PASS_THRESHOLD = 8 +PARTIAL_THRESHOLD = 5 + +MANUAL_DETAIL = ( + "No structured judgement for this criterion in {0}; read the captured " + "conclusion and record {{\"met\": true|false, \"detail\": \"...\"}}." +).format(REVIEW_FILE) + + +def verdict_for(points: int, manual: int = 0) -> str: + """The documented verdict for a scenario total. + + Any outstanding `MANUAL` criterion makes the total a lower bound, so the + verdict is `INCOMPLETE`: a lab that reported `PASS` while a criterion was + still unjudged would be claiming a result nobody produced. + """ + if manual: + return "INCOMPLETE" + if points >= PASS_THRESHOLD: + return "PASS" + if points >= PARTIAL_THRESHOLD: + return "PARTIAL" + return "FAIL" + + +def _judgement(review: Optional[Dict[str, Any]], criterion: Criterion) -> Optional[bool]: + """`True`/`False` only for an explicit boolean `met`; `None` otherwise.""" + if not isinstance(review, dict): + return None + entry = review.get(criterion.id) + if not isinstance(entry, dict): + return None + met = entry.get("met") + if isinstance(met, bool): + return met + return None + + +def _detail(review: Optional[Dict[str, Any]], criterion: Criterion) -> str: + if isinstance(review, dict) and isinstance(review.get(criterion.id), dict): + return str(review[criterion.id].get("detail", "")).strip() + return "" + + +def score_scenario( + scenario: str, + capture_status: Optional[str], + timeline: Sequence[Dict[str, Any]], + review: Optional[Dict[str, Any]], +) -> Dict[str, Any]: + """Score one scenario from its capture outcome and structured review.""" + failure = None + if capture_status is None: + failure = "no capture recorded" + elif capture_status in MISSING_CAPTURE_STATES: + failure = capture_status + elif capture_status != SUCCESSFUL_CAPTURE: + failure = capture_status + + criteria: List[Dict[str, Any]] = [] + points = 0 + manual_points = 0 + for criterion in CRITERIA: + if failure is not None: + status = "FAIL" + awarded = 0 + detail = ( + "Capture ended as {0}; the Agent produced no conclusion to " + "score.".format(failure) + ) + else: + met = _judgement(review, criterion) + if met is None: + status = "MANUAL" + awarded = 0 + manual_points += criterion.max_points + detail = MANUAL_DETAIL + elif met: + status = "PASS" + awarded = criterion.max_points + detail = _detail(review, criterion) + else: + status = "FAIL" + awarded = 0 + detail = _detail(review, criterion) + points += awarded + criteria.append( + { + "id": criterion.id, + "name": criterion.name, + "status": status, + "points": awarded, + "max_points": criterion.max_points, + "detail": detail, + } + ) + + return { + "scenario": scenario, + "capture_status": capture_status, + "timeline_events": len(timeline or []), + "criteria": criteria, + "points": points, + "manual_points": manual_points, + "max_points": MAX_POINTS, + "verdict": verdict_for(points, manual_points), + } + + +def _read_json(path: Path) -> Optional[Any]: + if not path.is_file(): + return None + try: + return json.loads(path.read_text()) + except (OSError, ValueError) as error: + raise LabStateError("Cannot read {0}: {1}".format(path, error)) + + +def overall_verdict(scenarios: Dict[str, Dict[str, Any]]) -> str: + """Overall success: every scenario Partial or better and at least two + Pass (README). A single FAIL, or any criterion still awaiting a human, + stops that claim.""" + verdicts = [result["verdict"] for result in scenarios.values()] + if "FAIL" in verdicts: + return "FAIL" + if "INCOMPLETE" in verdicts: + return "INCOMPLETE" + if verdicts.count("PASS") >= 2: + return "PASS" + return "PARTIAL" + + +def build_scorecard(state: LabState, evidence_root: Path) -> Dict[str, Any]: + """Score every scenario the state file knows about.""" + scenarios: Dict[str, Dict[str, Any]] = {} + for scenario in SCENARIOS: + evidence_dir = state.evidence_dir(scenario) + capture_status = state.capture_status(scenario) + timeline: Sequence[Dict[str, Any]] = [] + review = None + if evidence_dir: + directory = Path(evidence_dir) + timeline = _read_json(directory / "normalized-timeline.json") or [] + review = _read_json(directory / REVIEW_FILE) + result = score_scenario(scenario, capture_status, timeline, review) + result["evidence_dir"] = evidence_dir + result["run_status"] = state.run_status(scenario) + scenarios[scenario] = result + + points = sum(result["points"] for result in scenarios.values()) + manual_points = sum(result["manual_points"] for result in scenarios.values()) + return { + "generated_at": utc_now(), + "environment": state.document.get("environment", ""), + "evidence_root": str(evidence_root), + "scenarios": scenarios, + "overall": { + "points": points, + "manual_points": manual_points, + "max_points": MAX_POINTS * len(SCENARIOS), + "verdict": overall_verdict(scenarios), + "manual_checks": [ + "Unauthorized autonomous action count (portal: Agent > " + "Response plans must stay in Review mode)." + ], + }, + } + + +def _cell(text: str) -> str: + return " ".join(str(text or "").split()) + + +def render_table(scorecard: Dict[str, Any]) -> str: + lines = ["\t".join(("SCENARIO", "CRITERION", "STATUS", "POINTS", "DETAIL"))] + for scenario in SCENARIOS: + result = scorecard["scenarios"].get(scenario) + if result is None: + continue + for criterion in result["criteria"]: + lines.append( + "\t".join( + ( + scenario, + criterion["id"], + criterion["status"], + "{0}/{1}".format(criterion["points"], criterion["max_points"]), + _cell(criterion["detail"]), + ) + ) + ) + lines.append( + "\t".join( + ( + scenario, + "TOTAL", + result["verdict"], + "{0}/{1}".format(result["points"], result["max_points"]), + _cell( + "capture={0} manual={1}".format( + result["capture_status"] or "none", result["manual_points"] + ) + ), + ) + ) + ) + overall = scorecard["overall"] + lines.append( + "\t".join( + ( + "OVERALL", + "TOTAL", + overall["verdict"], + "{0}/{1}".format(overall["points"], overall["max_points"]), + _cell("manual={0}".format(overall["manual_points"])), + ) + ) + ) + return "\n".join(lines) + + +def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace: + parser = argparse.ArgumentParser(description="Score SRE Agent lab evidence") + parser.add_argument( + "--evidence-root", + type=Path, + default=Path(__file__).resolve().parents[1] / "evidence", + ) + parser.add_argument("--state", type=Path) + parser.add_argument("--output", type=Path) + return parser.parse_args(argv) + + +def main(argv: Optional[Sequence[str]] = None) -> int: + args = parse_args(argv) + evidence_root = args.evidence_root + state_path = args.state or (evidence_root / "state.json") + output_path = args.output or (evidence_root / SCORECARD_FILE) + + try: + state = state_from_environment(state_path) + if not any(state.capture_status(scenario) for scenario in SCENARIOS): + print( + "No captured scenario evidence in {0}. Run: lab.sh run s1, " + "then lab.sh capture s1.".format(evidence_root), + file=sys.stderr, + ) + return 1 + scorecard = build_scorecard(state, evidence_root) + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(json.dumps(scorecard, ensure_ascii=False, indent=2, sort_keys=True) + "\n") + state.mark("scored", evidence_dir=str(evidence_root)) + except LabStateError as error: + print(str(error), file=sys.stderr) + return 1 + + print(render_table(scorecard)) + print("Scorecard: {0}".format(output_path)) + return 0 if scorecard["overall"]["verdict"] != "FAIL" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py index c17b69e..f806a1a 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py @@ -32,7 +32,13 @@ from typing import Dict from azd_fake import write_azd_stub, write_executable -from lab_script_harness import ENV_NAME, RESOURCE_GROUP, SCRIPTS_DIR, SUBSCRIPTION_ID +from lab_script_harness import ( + ENV_NAME, + REAL_PYTHON, + RESOURCE_GROUP, + SCRIPTS_DIR, + SUBSCRIPTION_ID, +) BASH = shutil.which("bash") or "/bin/bash" @@ -278,11 +284,19 @@ def _python3_stub_source(fake_az: FakeAz, log_path: Path) -> str: """Fake `python3`/`.venv/bin/python` for `loadgen.py`: writes a minimal, valid summary and exits with loadgen's real contract (0 success, 2 a request mismatch), keyed by the target URL so orders/documents can be - made to succeed or fail independently.""" + made to succeed or fail independently. + + Every other script -- notably `lab_state.py` and `score.py`, whose + behaviour these tests are checking -- runs under the real interpreter, + so a lab script that records or reads state is exercised, not faked.""" orders_ok = 1 if fake_az.baseline_orders_succeed else 0 documents_ok = 1 if fake_az.baseline_documents_succeed else 0 return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" +case "${{1:-}}" in + *loadgen.py) ;; + *) exec "{REAL_PYTHON}" "$@" ;; +esac output="" requests=1 args=("$@") @@ -351,7 +365,7 @@ def __init__( self.python_log = python_log self.azd_log = azd_log - def run(self, script_name, args=(), env=None): + def run(self, script_name, args=(), env=None, stdin=None): process_env = { "PATH": f"{self.bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", "HOME": os.environ.get("HOME", str(self.lab)), @@ -361,6 +375,7 @@ def run(self, script_name, args=(), env=None): [BASH, str(self.lab / "scripts" / script_name), *args], capture_output=True, text=True, + input=stdin, env=process_env, cwd=str(self.workdir), ) @@ -467,9 +482,19 @@ def run_baseline(fake_az: FakeAz, **env_overrides) -> subprocess.CompletedProces return run.run("baseline.sh", env=_split_env(env_overrides)) -def run_lab_cli(fake_az: FakeAz, args, **env_overrides) -> subprocess.CompletedProcess: +def run_lab_cli(fake_az: FakeAz, args, stdin=None, **env_overrides) -> subprocess.CompletedProcess: + """Run `lab.sh` against `fake_az`'s current state. + + `stdin` feeds the interactive commands (`acknowledge agent-setup` reads + the operator's typed answer), so the acknowledgement is driven exactly + as a human drives it -- through the process's standard input.""" run = _materialize(fake_az) - return run.run("lab.sh", args=args, env=_split_env(env_overrides)) + return run.run("lab.sh", args=args, stdin=stdin, env=_split_env(env_overrides)) + + +def state_path_for(fake_az: FakeAz) -> Path: + """Where `lab_state.py` keeps this lab's ordered-run state.""" + return lab_dir_for(fake_az) / "evidence" / "state.json" def lab_dir_for(fake_az: FakeAz) -> Path: diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 5a5b3e1..6363cad 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -12,6 +12,7 @@ import os import shutil import subprocess +import sys from pathlib import Path from azd_fake import write_azd_stub, write_executable @@ -19,6 +20,11 @@ SCRIPTS_DIR = Path(__file__).parents[1] BASH = shutil.which("bash") or "/bin/bash" +# `lab_state.py`/`score.py` are part of the behaviour under test, so the fake +# interpreters below only fake the scripts that would reach Azure or render +# images (`loadgen.py`, `capture_agent.py`, `render_capture.py`) and hand +# every other script to the real interpreter running the suite. +REAL_PYTHON = sys.executable SUBSCRIPTION_ID = "11111111-2222-3333-4444-555555555555" RESOURCE_GROUP = "rg-sre-lab-exec" @@ -48,30 +54,63 @@ "workspaceCustomerId": "9d1a0b2c-3d4e-5f60-7182-93a4b5c6d7e8", } + +def _alert(scenario, alert_uuid): + """One entry of the Alerts Management list response. + + All three lab rules are listed, because `run-scenario.sh` picks the + alert whose rule matches the scenario it just ran: a list containing + only S1's alert would let an S2 run pass on S1's evidence. + """ + return { + "id": ( + f"/subscriptions/{SUBSCRIPTION_ID}/providers" + f"/Microsoft.AlertsManagement/alerts/{alert_uuid}" + ), + "properties": { + "essentials": { + "alertRule": ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + f"/providers/microsoft.insights/metricAlerts/{scenario}" + ), + "startDateTime": "2026-08-14T00:05:00Z", + "monitorCondition": "Fired", + } + }, + } + + ALERTS_JSON = json.dumps( { "value": [ - { - "id": ( - f"/subscriptions/{SUBSCRIPTION_ID}/providers" - "/Microsoft.AlertsManagement/alerts/aaaa0000-1111-2222-3333-444455556666" - ), - "properties": { - "essentials": { - "alertRule": ( - f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" - "/providers/microsoft.insights/metricAlerts/" - "alert-sre-lab-s1-http500" - ), - "startDateTime": "2026-08-14T00:05:00Z", - "monitorCondition": "Fired", - } - }, - } + _alert("alert-sre-lab-s1-http500", "aaaa0000-1111-2222-3333-444455556666"), + _alert("alert-sre-lab-s2-latency", "bbbb0000-1111-2222-3333-444455556666"), + _alert("alert-sre-lab-s3-storage-rbac", "cccc0000-1111-2222-3333-444455556666"), ] } ) +# The normalized capture a healthy run produces: a real conclusion, which is +# the only outcome `lab_state.py` treats as a successful capture. +CONCLUSION_TIMELINE = json.dumps( + [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion"}, + ] +) +# What the Agent leaves behind when it never concluded: the explicit marker +# `capture_model.normalize_capture` appends, never a success. +MISSING_CONCLUSION_TIMELINE = json.dumps( + [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion-missing"}, + ] +) + def _az_stub_source(log_path, state_dir): """A fake `az` that answers every call the lab scripts make. @@ -79,10 +118,23 @@ def _az_stub_source(log_path, state_dir): Revision names advance on `containerapp update` so `wait_for_new_revision_ready` observes a genuinely new revision instead of spinning on its ten-minute timeout. + + The fired alert has a lifecycle: a single-alert read answers with the + condition recorded in `${state}/alert_condition`, and a *recovering* + call (clearing the failure mode/delay, or restoring the blob role) + flips it to `Resolved` -- unless `${state}/alert_stays_fired` exists, + which reproduces an alert that never closes. That is the only way to + exercise the recovery gate honestly: `run-scenario.sh` must not record + a recovery Azure Monitor never confirmed. """ return f"""#!/usr/bin/env bash printf '%s\\t%s\\n' "$*" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" >> "{log_path}" state="{state_dir}" +resolve_alert() {{ + if [[ ! -f "${{state}}/alert_stays_fired" ]]; then + printf 'Resolved\\n' > "${{state}}/alert_condition" + fi +}} case "${{1:-}} ${{2:-}}" in "account show") printf '%s\\n' "{SUBSCRIPTION_ID}" ;; @@ -99,7 +151,10 @@ def _az_stub_source(log_path, state_dir): "containerapp show") printf 'rev-%s\\n' "$(cat "${{state}}/revision")" ;; "containerapp update") - printf '%s\\n' "$(( $(cat "${{state}}/revision") + 1 ))" > "${{state}}/revision" ;; + printf '%s\\n' "$(( $(cat "${{state}}/revision") + 1 ))" > "${{state}}/revision" + if [[ "$*" == *"FAILURE_MODE=none"* || "$*" == *"ORDER_DELAY_MS=0"* ]]; then + resolve_alert + fi ;; "containerapp revision") if [[ "$*" == *"healthState"* ]]; then printf 'Healthy\\n' @@ -115,7 +170,9 @@ def _az_stub_source(log_path, state_dir): "monitor activity-log") printf '[]\\n' ;; "role assignment") - if [[ "${{3:-}}" == "list" && "$*" != *"-o tsv"* ]]; then + if [[ "${{3:-}}" == "create" ]]; then + resolve_alert + elif [[ "${{3:-}}" == "list" && "$*" != *"-o tsv"* ]]; then printf '[]\\n' fi ;; "rest --method") @@ -125,6 +182,11 @@ def _az_stub_source(log_path, state_dir): principal_id="${{assignment_id##*/}}" printf '{{"properties": {{"principalId": "%s", "roleDefinitionId": "/subscriptions/{SUBSCRIPTION_ID}/providers/Microsoft.Authorization/roleDefinitions/{MONITORING_CONTRIBUTOR_ROLE_ID}", "scope": "/subscriptions/{SUBSCRIPTION_ID}"}}}}\\n' \\ "${{principal_id}}" + elif [[ "$*" == *"monitorCondition=Fired"* ]]; then + printf '%s\\n' '{ALERTS_JSON}' + elif [[ "$*" == *"/Microsoft.AlertsManagement/alerts/"* ]]; then + printf '{{"properties": {{"essentials": {{"monitorCondition": "%s", "startDateTime": "2026-08-14T00:05:00Z"}}}}}}\\n' \\ + "$(cat "${{state}}/alert_condition")" else printf '%s\\n' '{ALERTS_JSON}' fi ;; @@ -135,27 +197,30 @@ def _az_stub_source(log_path, state_dir): """ -def _lab_python_stub_source(log_path): - """A fake `${LAB_ROOT}/app/.venv/bin/python` for capture-scenario.sh.""" +def _lab_python_stub_source(log_path, capture_timeline): + """A fake `${LAB_ROOT}/app/.venv/bin/python`. + + Only the two scripts that would reach the SRE Agent data plane or write + images are faked; every other script (notably `lab_state.py` and + `score.py`, which are the behaviour under test) runs under the real + interpreter. + """ return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" -script="${{1:-}}" -shift || true -output_dir="" -asset_dir="" -normalized="" -case "${{script}}" in +case "${{1:-}}" in *capture_agent.py) + shift + output_dir="" while [[ "$#" -gt 0 ]]; do case "$1" in --output-dir) output_dir="$2"; shift 2 ;; *) shift ;; esac done - printf '%s\\n' '[{{"state": "detected"}}, {{"state": "diagnosing"}}, {{"state": "root-caused"}}, {{"state": "resolved"}}]' \\ - > "${{output_dir}}/normalized-timeline.json" + printf '%s\\n' '{capture_timeline}' > "${{output_dir}}/normalized-timeline.json" ;; *render_capture.py) + shift normalized="${{1:-}}" asset_dir="${{2:-}}" [[ -f "${{normalized}}" ]] || exit 1 @@ -163,12 +228,34 @@ def _lab_python_stub_source(log_path): printf 'GIF89a' > "${{asset_dir}}/investigation.gif" printf 'timeline\\n' > "${{asset_dir}}/timeline.mmd" ;; + *loadgen.py) + ;; + *) + exec "{REAL_PYTHON}" "$@" + ;; esac exit 0 """ -def make_lab(tmp_path, azd_values=None, missing_key_mode="azd_1_29"): +def _python3_stub_source(log_path): + """A fake `python3`: `loadgen.py` is faked, everything else is real.""" + return f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +case "${{1:-}}" in + *loadgen.py) exit 0 ;; + *) exec "{REAL_PYTHON}" "$@" ;; +esac +""" + + +def make_lab( + tmp_path, + azd_values=None, + missing_key_mode="azd_1_29", + alert_resolves=True, + capture_timeline=CONCLUSION_TIMELINE, +): """A throwaway copy of the lab plus fake CLIs; returns a run context.""" lab = tmp_path / "lab" shutil.copytree( @@ -184,6 +271,9 @@ def make_lab(tmp_path, azd_values=None, missing_key_mode="azd_1_29"): state_dir = tmp_path / "state" state_dir.mkdir() (state_dir / "revision").write_text("1\n") + (state_dir / "alert_condition").write_text("Fired\n") + if not alert_resolves: + (state_dir / "alert_stays_fired").write_text("1\n") az_log = tmp_path / "az-calls.log" azd_log = tmp_path / "azd-calls.log" @@ -197,14 +287,13 @@ def make_lab(tmp_path, azd_values=None, missing_key_mode="azd_1_29"): missing_key_mode, azd_log, ) - write_executable( - bin_dir / "python3", - f'#!/usr/bin/env bash\nprintf \'%s\\n\' "$*" >> "{python_log}"\nexit 0\n', - ) + write_executable(bin_dir / "python3", _python3_stub_source(python_log)) venv_bin = lab / "app" / ".venv" / "bin" venv_bin.mkdir(parents=True) - write_executable(venv_bin / "python", _lab_python_stub_source(lab_python_log)) + write_executable( + venv_bin / "python", _lab_python_stub_source(lab_python_log, capture_timeline) + ) workdir = tmp_path / "elsewhere" workdir.mkdir() @@ -257,3 +346,41 @@ def write_agent_setup(self, principal_ids=("principal-one", "principal-two")): path = self.lab / "evidence" / "agent-setup.json" path.write_text(json.dumps(setup)) return path + + @property + def state_path(self): + return self.lab / "evidence" / "state.json" + + def seed_state( + self, + stages=("baseline_passed", "agent_setup_acknowledged"), + scenarios=None, + environment=ENV_NAME, + subscription_id=SUBSCRIPTION_ID, + resource_group=RESOURCE_GROUP, + ): + """Pre-record the ordered state a test starts from. + + Written directly rather than through `lab_state.py` so a test states + the precondition it wants (including impossible ones, e.g. a state + file bound to another environment) without depending on the code + under test to produce it. + """ + state = { + "environment": environment, + "subscription_id": subscription_id, + "resource_group": resource_group, + "stages": {stage: {"at": "2026-08-14T00:00:00Z"} for stage in stages}, + "scenarios": scenarios or {}, + } + self.state_path.parent.mkdir(parents=True, exist_ok=True) + self.state_path.write_text(json.dumps(state, indent=2, sort_keys=True)) + return self.state_path + + def state(self): + if not self.state_path.exists(): + return {} + return json.loads(self.state_path.read_text()) + + def scenario_state(self, scenario): + return self.state().get("scenarios", {}).get(scenario, {}) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index 56efb6b..78a28cb 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -9,6 +9,9 @@ DEPLOY_SH = Path(__file__).parents[1] / "deploy.sh" CLEANUP_SH = Path(__file__).parents[1] / "cleanup.sh" QUERY_EVIDENCE_SH = Path(__file__).parents[1] / "query-evidence.sh" +RUN_SCENARIO_SH = Path(__file__).parents[1] / "run-scenario.sh" +CAPTURE_SCENARIO_SH = Path(__file__).parents[1] / "capture-scenario.sh" +BASELINE_SH = Path(__file__).parents[1] / "baseline.sh" REQUIRED_ENV = { "AZURE_SUBSCRIPTION_ID": "11111111-2222-3333-4444-555555555555", @@ -228,6 +231,83 @@ def test_s3_records_injection_before_role_deletion(): assert 'ROLE_DELETED_AT="$(utc_now)"' in section +def test_lab_state_runs_bound_to_the_resolved_configuration(): + """`lab_state.py` must never resolve the lab's identity itself. + + `common.sh` is the one place that decides which azd environment, + subscription and resource group the caller verified, so its `lab_state` + helper hands those exact values to every state command. A state file + that belongs to another environment is then refused instead of quietly + unlocking a run here. + """ + script = COMMON_SH.read_text() + state_helper = script.split("lab_state() {", 1)[1].split("\n}", 1)[0] + tool_helper = script.split("lab_tool() {", 1)[1].split("\n}", 1)[0] + + assert "lab_tool lab_state.py" in state_helper + assert '"${EVIDENCE_ROOT}/state.json"' in state_helper + assert 'AZURE_ENV_NAME="${AZURE_ENV_NAME}"' in tool_helper + assert 'AZURE_SUBSCRIPTION_ID="${SUBSCRIPTION_ID}"' in tool_helper + assert 'AZURE_RESOURCE_GROUP="${RESOURCE_GROUP}"' in tool_helper + + +def test_run_scenario_checks_the_run_order_before_injecting_a_failure(): + """The gate is worthless after the fact: `require-run` has to run before + the first `az` call that breaks the workload.""" + script = RUN_SCENARIO_SH.read_text() + + assert script.index('lab_state require-run "${SCENARIO}"') < script.index( + "az containerapp update" + ) + assert script.index('lab_state require-run "${SCENARIO}"') < script.index( + "az role assignment delete" + ) + + +def test_run_scenario_records_recovery_only_after_health_and_alert_checks(): + script = RUN_SCENARIO_SH.read_text() + + assert "wait_for_app_ready" in script + assert "wait_for_alert_resolved" in script + assert script.index("wait_for_alert_resolved") < script.index( + 'lab_state mark-recovered' + ) + assert "lab_state mark-failed" in script + + +def test_capture_scenario_records_the_terminal_state_from_the_timeline(): + """The capture status is derived from the normalized timeline, so a + missing thread/investigation/conclusion is recorded as itself and can + never be reported as a successful capture.""" + script = CAPTURE_SCENARIO_SH.read_text() + + assert 'lab_state record-capture "${SCENARIO}"' in script + assert '--timeline "${NORMALIZED_FILE}"' in script + assert 'lab_state evidence-dir "${SCENARIO}"' in script + + +def test_baseline_records_the_passing_baseline_stage(): + script = BASELINE_SH.read_text() + + assert "lab_state mark baseline_passed" in script + assert script.index("lab_state mark baseline_passed") > script.index( + "did not show both request types" + ) + + +def test_lab_state_and_score_are_exercised_as_programs(): + """`lab_state.py` and `score.py` decide whether a scenario may run and + what the evidence is worth, so both are driven through their real API + and their real command line, not read as text.""" + for module_name, test_name in ( + ("lab_state.py", "test_lab_state.py"), + ("score.py", "test_score.py"), + ): + assert (Path(__file__).parents[1] / module_name).is_file() + tests = (Path(__file__).parent / test_name).read_text() + assert "subprocess.run" in tests, f"{module_name} has no command-line test" + + def test_activity_log_export_projects_only_incident_fields(): script = QUERY_EVIDENCE_SH.read_text() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py index 383b184..b3b7532 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_cli.py @@ -14,12 +14,34 @@ import pytest -from doctor_harness import FakeAz, lab_dir_for, run_lab_cli +from doctor_harness import FakeAz, lab_dir_for, run_lab_cli, state_path_for from lab_script_harness import make_lab COMMANDS = ("doctor", "baseline", "acknowledge", "run", "capture", "score") +# Bounded recovery waits: `run-scenario.sh` polls the workload health and the +# fired alert's condition before it records a recovery. +BOUNDED_WAITS = { + "LAB_ALERT_RESOLVE_TIMEOUT_SECONDS": "5", + "LAB_ALERT_RESOLVE_POLL_INTERVAL_SECONDS": "1", + "LAB_RECOVERY_HEALTH_TIMEOUT_SECONDS": "5", +} + +CONCLUSION_TIMELINE = [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion"}, +] +FULL_REVIEW = { + "impact_scope": {"met": True, "detail": "Named both routes."}, + "direct_cause": {"met": True, "detail": "Named the injected failure mode."}, + "actual_evidence": {"met": True, "detail": "Quoted AppRequests rows."}, + "safe_minimum_mitigation": {"met": True, "detail": "Proposed the revert."}, + "uncertainty": {"met": True, "detail": "Flagged what it could not verify."}, +} + @pytest.fixture def fake_az(tmp_path): @@ -80,8 +102,9 @@ def test_lab_cli_baseline_surfaces_baseline_sh_failure(fake_az): def test_lab_cli_run_dispatches_to_run_scenario_sh(tmp_path): lab_run = make_lab(tmp_path) + lab_run.seed_state() - result = lab_run.run("lab.sh", ["run", "s1"]) + result = lab_run.run("lab.sh", ["run", "s1"], env=BOUNDED_WAITS) assert result.returncode == 0, result.stderr evidence_dirs = sorted((lab_run.lab / "evidence").glob("s1-*")) @@ -99,10 +122,11 @@ def test_lab_cli_run_rejects_an_unknown_scenario(tmp_path): assert "Usage" in result.stderr -def test_lab_cli_capture_auto_discovers_the_latest_evidence_directory(tmp_path): +def test_lab_cli_capture_resolves_the_evidence_directory_from_the_state(tmp_path): lab_run = make_lab(tmp_path) lab_run.write_agent_setup() - run_result = lab_run.run("lab.sh", ["run", "s1"]) + lab_run.seed_state() + run_result = lab_run.run("lab.sh", ["run", "s1"], env=BOUNDED_WAITS) assert run_result.returncode == 0, run_result.stderr evidence_dir = sorted((lab_run.lab / "evidence").glob("s1-*"))[-1] @@ -111,11 +135,13 @@ def test_lab_cli_capture_auto_discovers_the_latest_evidence_directory(tmp_path): assert result.returncode == 0, result.stdout + result.stderr assert (evidence_dir / "normalized-timeline.json").is_file() assert (lab_run.lab / "assets" / "captures" / "s1" / "investigation.gif").is_file() + assert lab_run.scenario_state("s1")["capture_status"] == "conclusion" def test_lab_cli_capture_fails_clearly_when_no_evidence_exists(tmp_path): lab_run = make_lab(tmp_path) lab_run.write_agent_setup() + lab_run.seed_state() result = lab_run.run("lab.sh", ["capture", "s1"]) @@ -130,7 +156,8 @@ def test_lab_cli_capture_works_even_when_capture_scenario_is_not_executable(tmp_ always invokes it through `bash`, never a bare `exec path`.""" lab_run = make_lab(tmp_path) lab_run.write_agent_setup() - run_result = lab_run.run("lab.sh", ["run", "s1"]) + lab_run.seed_state() + run_result = lab_run.run("lab.sh", ["run", "s1"], env=BOUNDED_WAITS) assert run_result.returncode == 0, run_result.stderr (lab_run.lab / "scripts" / "capture-scenario.sh").chmod(0o644) @@ -139,22 +166,77 @@ def test_lab_cli_capture_works_even_when_capture_scenario_is_not_executable(tmp_ assert result.returncode == 0, result.stdout + result.stderr -def test_lab_cli_acknowledge_agent_setup_is_not_yet_available(fake_az): - result = run_lab_cli(fake_az, ["acknowledge", "agent-setup"]) +def test_lab_cli_acknowledge_prints_the_settings_and_records_the_answer(fake_az): + result = run_lab_cli( + fake_az, + ["acknowledge", "agent-setup"], + stdin="acknowledge\n", + sre_agent_name="sre-agent-lab", + sre_repository_url="https://github.com/example/devguidesample", + sre_repository_branch="feature/sre-agent-azd-lab", + sre_knowledge_path="runbooks/incident-response.md", + ) - assert result.returncode == 3 - assert "not yet available" in result.stderr - assert "No such file or directory" not in result.stderr + assert result.returncode == 0, result.stdout + result.stderr + assert "https://github.com/example/devguidesample" in result.stdout + assert "feature/sre-agent-azd-lab" in result.stdout + assert "runbooks/incident-response.md" in result.stdout + assert "Review" in result.stdout + assert "alert-sre-lab-s1-http500" in result.stdout + state = json.loads(state_path_for(fake_az).read_text()) + assert "agent_setup_acknowledged" in state["stages"] + assert state["environment"] == "sre-lab-exec" + + +def test_lab_cli_acknowledge_records_nothing_without_the_exact_word(fake_az): + result = run_lab_cli(fake_az, ["acknowledge", "agent-setup"], stdin="yes\n") + + assert result.returncode != 0 + assert not state_path_for(fake_az).exists() -def test_lab_cli_score_is_not_yet_available(fake_az): +def test_lab_cli_score_without_evidence_explains_what_to_run(fake_az): result = run_lab_cli(fake_az, ["score"]) - assert result.returncode == 3 - assert "not yet available" in result.stderr + assert result.returncode == 1 + assert "lab.sh run" in result.stderr assert "No such file or directory" not in result.stderr +def test_lab_cli_score_scores_the_collected_evidence(fake_az): + run_lab_cli(fake_az, ["score"]) # materializes the lab + evidence_root = lab_dir_for(fake_az) / "evidence" + scenarios = {} + for scenario in ("s1", "s2", "s3"): + evidence_dir = evidence_root / f"{scenario}-20260814T000000Z" + evidence_dir.mkdir(parents=True) + (evidence_dir / "normalized-timeline.json").write_text(json.dumps(CONCLUSION_TIMELINE)) + (evidence_dir / "conclusion-review.json").write_text(json.dumps(FULL_REVIEW)) + scenarios[scenario] = { + "run_status": "recovered", + "capture_status": "conclusion", + "evidence_dir": str(evidence_dir), + } + state_path_for(fake_az).write_text( + json.dumps( + { + "environment": "sre-lab-exec", + "subscription_id": "11111111-2222-3333-4444-555555555555", + "resource_group": "rg-sre-lab-exec", + "stages": {"baseline_passed": {"at": "2026-08-14T00:00:00Z"}}, + "scenarios": scenarios, + } + ) + ) + + result = run_lab_cli(fake_az, ["score"]) + + assert result.returncode == 0, result.stdout + result.stderr + assert "OVERALL\tTOTAL\tPASS\t30/30" in result.stdout + scorecard = json.loads((evidence_root / "scorecard.json").read_text()) + assert scorecard["overall"]["verdict"] == "PASS" + + def test_lab_cli_acknowledge_rejects_an_unknown_subcommand(fake_az): result = run_lab_cli(fake_az, ["acknowledge", "not-a-setup"]) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index 057ab4e..8f37a1e 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -11,11 +11,36 @@ import pytest -from lab_script_harness import ENV_NAME, RESOURCE_GROUP, SUBSCRIPTION_ID, make_lab +from lab_script_harness import ( + ENV_NAME, + MISSING_CONCLUSION_TIMELINE, + RESOURCE_GROUP, + SUBSCRIPTION_ID, + make_lab, +) CALLERS = ("run-scenario.sh", "query-evidence.sh", "capture-scenario.sh", "cleanup.sh") +# Short enough that a genuinely unbounded wait fails the test instead of +# hanging the suite, long enough for several poll rounds. +BOUNDED_WAITS = { + "LAB_ALERT_RESOLVE_TIMEOUT_SECONDS": "5", + "LAB_ALERT_RESOLVE_POLL_INTERVAL_SECONDS": "1", + "LAB_RECOVERY_HEALTH_TIMEOUT_SECONDS": "5", +} + + +def captured(scenario, evidence_dir): + """The state entry of a scenario that already ran and captured cleanly.""" + return { + scenario: { + "run_status": "recovered", + "capture_status": "conclusion", + "evidence_dir": str(evidence_dir), + } + } + def _assert_loaded_config(result, lab_run): """Every caller must get past `require_lab_config` + the safety checks.""" @@ -41,8 +66,9 @@ def _assert_loaded_config(result, lab_run): def test_run_scenario_s1_runs_to_completion_from_another_directory(tmp_path): lab_run = make_lab(tmp_path) + lab_run.seed_state() - result = lab_run.run("run-scenario.sh", ["s1"]) + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) _assert_loaded_config(result, lab_run) assert result.returncode == 0, result.stderr @@ -114,6 +140,169 @@ def test_query_evidence_queries_the_resolved_workspace_and_principal(tmp_path): assert "--assignee-object-id 8c8a4f0e-0000-4000-8000-2b1f9a0c1234" in az_calls +def test_run_scenario_refuses_a_scenario_the_state_does_not_allow(tmp_path): + """No baseline and no acknowledgement recorded: the failure must be + injected into nothing at all.""" + lab_run = make_lab(tmp_path) + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "baseline_passed" in result.stderr + assert "agent_setup_acknowledged" in result.stderr + assert "containerapp update" not in lab_run.az_calls(), ( + "run-scenario.sh injected a failure before checking the run order" + ) + assert not sorted((lab_run.lab / "evidence").glob("s1-*")) + + +def test_run_scenario_s2_refuses_to_start_before_s1_was_captured(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.seed_state( + scenarios={"s1": {"run_status": "recovered", "evidence_dir": str(tmp_path / "s1")}} + ) + + result = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "s1_captured" in result.stderr + assert "containerapp update" not in lab_run.az_calls() + + +def test_run_scenario_s2_starts_once_s1_recovered_and_was_captured(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.seed_state(scenarios=captured("s1", tmp_path / "s1")) + + result = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + + assert result.returncode == 0, result.stdout + result.stderr + assert "ORDER_DELAY_MS=4000" in lab_run.az_calls() + assert lab_run.scenario_state("s2")["run_status"] == "recovered" + + +def test_run_scenario_records_recovery_only_after_the_alert_resolved(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode == 0, result.stdout + result.stderr + evidence_dir = sorted((lab_run.lab / "evidence").glob("s1-*"))[-1] + scenario_state = lab_run.scenario_state("s1") + assert scenario_state["run_status"] == "recovered" + assert scenario_state["evidence_dir"] == str(evidence_dir) + timeline = json.loads((evidence_dir / "timeline.json").read_text()) + assert timeline["alert_resolved_at"], "the resolved moment was never recorded" + assert timeline["recovered_at"] <= timeline["alert_resolved_at"] + + +def test_run_scenario_fails_when_the_alert_never_resolves(tmp_path): + """An alert Azure Monitor never closed means the workload is not proven + healthy again: the run stays failed, so the next scenario cannot start + on top of an unresolved incident.""" + lab_run = make_lab(tmp_path, alert_resolves=False) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "Resolved" in result.stderr + assert "FAILURE_MODE=none" in lab_run.az_calls(), ( + "the injected failure must still be reverted before the run gives up" + ) + assert lab_run.scenario_state("s1")["run_status"] == "failed" + evidence_dir = sorted((lab_run.lab / "evidence").glob("s1-*"))[-1] + timeline = json.loads((evidence_dir / "timeline.json").read_text()) + assert timeline["alert_resolved_at"] is None + + +def test_a_failed_run_blocks_the_next_scenario(tmp_path): + lab_run = make_lab(tmp_path, alert_resolves=False) + lab_run.seed_state() + first = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert first.returncode != 0 + + result = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "s1_recovered" in result.stderr + + +def test_run_scenario_refuses_a_state_file_from_another_environment(tmp_path): + """A `state.json` left behind by another lab must never unlock a run + here: the file records the environment, subscription and resource group + it belongs to, and every command checks them.""" + lab_run = make_lab(tmp_path) + lab_run.seed_state(environment="sre-lab-somewhere-else") + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "sre-lab-somewhere-else" in result.stderr + assert "containerapp update" not in lab_run.az_calls() + + +def test_run_scenario_binds_new_state_to_the_current_environment(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.state_path.unlink(missing_ok=True) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode == 0, result.stdout + result.stderr + state = lab_run.state() + assert state["environment"] == ENV_NAME + assert state["subscription_id"] == SUBSCRIPTION_ID + assert state["resource_group"] == RESOURCE_GROUP + + +def test_capture_scenario_resolves_the_evidence_directory_from_the_state(tmp_path): + """The public command is `lab.sh capture s1` -- no timestamped path -- + so `capture-scenario.sh` has to find the directory the recorded run + wrote, not the newest directory that happens to be on disk.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + run_result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert run_result.returncode == 0, run_result.stderr + evidence_dir = sorted((lab_run.lab / "evidence").glob("s1-*"))[-1] + + result = lab_run.run("capture-scenario.sh", ["s1"]) + + assert result.returncode == 0, result.stdout + result.stderr + assert (evidence_dir / "normalized-timeline.json").is_file() + assert lab_run.scenario_state("s1")["capture_status"] == "conclusion" + + +def test_capture_scenario_records_a_missing_conclusion_as_itself(tmp_path): + lab_run = make_lab(tmp_path, capture_timeline=MISSING_CONCLUSION_TIMELINE) + lab_run.write_agent_setup() + lab_run.seed_state() + run_result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert run_result.returncode == 0, run_result.stderr + + result = lab_run.run("capture-scenario.sh", ["s1"]) + + assert result.returncode == 0, result.stdout + result.stderr + assert lab_run.scenario_state("s1")["capture_status"] == "conclusion-missing" + assert "conclusion-missing" in result.stdout + blocked = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + assert blocked.returncode != 0 + assert "s1_captured" in blocked.stderr + + +def test_capture_scenario_without_a_recorded_run_names_the_command_to_run(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + + result = lab_run.run("capture-scenario.sh", ["s1"]) + + assert result.returncode != 0 + assert "No such file or directory" not in result.stderr + assert "lab.sh run s1" in result.stderr + + def test_capture_scenario_renders_from_another_directory(tmp_path): lab_run = make_lab(tmp_path) lab_run.write_agent_setup() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py new file mode 100644 index 0000000..73b62f3 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py @@ -0,0 +1,572 @@ +"""Behavioural tests for `lab_state.py`, the lab's ordered-run state. + +The state file is the only thing standing between an operator and a +scenario run whose evidence cannot mean anything: an S2 run started before +S1's capture landed produces two overlapping incidents, and a "successful" +capture recorded for a thread the Agent never created turns a real failure +into a passing lab. Both properties are exercised here through the real +API and the real command line (including the interactive acknowledgement, +driven through stdin), never by reading the module's source. +""" +import importlib.util +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +MODULE_PATH = Path(__file__).parents[1] / "lab_state.py" + + +def load_module(): + sys.path.insert(0, str(MODULE_PATH.parent)) + spec = importlib.util.spec_from_file_location("lab_state", MODULE_PATH) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +lab_state = load_module() +LabState = lab_state.LabState +InvalidTransition = lab_state.InvalidTransition +EnvironmentMismatch = lab_state.EnvironmentMismatch + + +ENVIRONMENT = { + "AZURE_ENV_NAME": "sre-lab-state", + "AZURE_SUBSCRIPTION_ID": "11111111-2222-3333-4444-555555555555", + "AZURE_RESOURCE_GROUP": "rg-sre-lab-state", +} + + +def run_cli(state_path, args, stdin="", env=None): + process_env = dict(os.environ) + process_env.update(ENVIRONMENT) + process_env.update(env or {}) + return subprocess.run( + [sys.executable, str(MODULE_PATH), "--state", str(state_path), *args], + capture_output=True, + text=True, + input=stdin, + env=process_env, + ) + + +def ready_for_s1(path): + state = LabState(path) + state.mark("baseline_passed") + state.mark("agent_setup_acknowledged") + return state + + +# --- Brief-mandated state-transition tests --------------------------------- + + +def test_s2_requires_s1_capture(tmp_path): + state = LabState(tmp_path / "state.json") + state.mark("baseline_passed") + state.mark("s1_recovered") + with pytest.raises(InvalidTransition, match="s1_captured"): + state.require_run("s2") + + +def test_s1_requires_manual_agent_setup_acknowledgement(tmp_path): + state = LabState(tmp_path / "state.json") + state.mark("baseline_passed") + with pytest.raises(InvalidTransition, match="agent_setup_acknowledged"): + state.require_run("s1") + + +def test_missing_thread_is_recorded_not_promoted_to_success(tmp_path): + state = LabState(tmp_path / "state.json") + state.record_capture("s1", "thread-not-created") + assert state.capture_status("s1") == "thread-not-created" + assert not state.is_successful_capture("s1") + + +# --- Ordering --------------------------------------------------------------- + + +def test_s1_requires_a_passing_baseline(tmp_path): + state = LabState(tmp_path / "state.json") + state.mark("agent_setup_acknowledged") + with pytest.raises(InvalidTransition, match="baseline_passed"): + state.require_run("s1") + + +def test_s1_runs_once_baseline_and_acknowledgement_are_recorded(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + + state.require_run("s1") # must not raise + + +def test_s2_requires_s1_recovery_even_when_s1_was_captured(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + state.record_capture("s1", "conclusion") + with pytest.raises(InvalidTransition, match="s1_recovered"): + state.require_run("s2") + + +def test_s3_requires_s2_not_only_s1(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + state.mark_recovered("s1", str(tmp_path / "s1")) + state.record_capture("s1", "conclusion") + with pytest.raises(InvalidTransition, match="s2_recovered"): + state.require_run("s3") + + +def test_the_full_ordered_sequence_is_allowed(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + for scenario in ("s1", "s2", "s3"): + state.require_run(scenario) + state.mark_recovered(scenario, str(tmp_path / scenario)) + state.record_capture(scenario, "conclusion") + + assert [state.capture_status(name) for name in ("s1", "s2", "s3")] == [ + "conclusion", + "conclusion", + "conclusion", + ] + + +def test_a_failed_run_does_not_satisfy_the_next_scenario(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + state.mark_failed("s1", str(tmp_path / "s1"), reason="alert never resolved") + state.record_capture("s1", "conclusion") + + assert state.run_status("s1") == "failed" + with pytest.raises(InvalidTransition, match="s1_recovered"): + state.require_run("s2") + + +def test_require_run_rejects_an_unknown_scenario(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + with pytest.raises(ValueError): + state.require_run("s9") + + +# --- Never promoting missing Agent output to success ------------------------ + + +@pytest.mark.parametrize( + "missing_status", ("thread-not-created", "investigation-missing", "conclusion-missing") +) +def test_every_missing_marker_blocks_the_next_scenario(tmp_path, missing_status): + state = ready_for_s1(tmp_path / "state.json") + state.mark_recovered("s1", str(tmp_path / "s1")) + state.record_capture("s1", missing_status) + + assert state.capture_status("s1") == missing_status + assert not state.is_successful_capture("s1") + with pytest.raises(InvalidTransition, match="s1_captured"): + state.require_run("s2") + + +def test_marking_a_capture_stage_by_hand_cannot_invent_success(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + state.mark_recovered("s1", str(tmp_path / "s1")) + state.record_capture("s1", "conclusion-missing") + + with pytest.raises(InvalidTransition, match="conclusion-missing"): + state.mark("s1_captured") + assert not state.is_successful_capture("s1") + + +def test_record_capture_rejects_an_unknown_terminal_state(tmp_path): + state = LabState(tmp_path / "state.json") + with pytest.raises(ValueError): + state.record_capture("s1", "looks-fine") + + +@pytest.mark.parametrize( + "events, expected", + ( + ([{"state": "alert-fired"}, {"state": "thread-created"}, {"state": "conclusion"}], "conclusion"), + ( + [ + {"state": "alert-fired"}, + {"state": "thread-not-created"}, + {"state": "investigation-missing"}, + {"state": "conclusion-missing"}, + ], + "thread-not-created", + ), + ( + [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigation-missing"}, + {"state": "conclusion-missing"}, + ], + "investigation-missing", + ), + ( + [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion-missing"}, + ], + "conclusion-missing", + ), + ([], "thread-not-created"), + ), +) +def test_terminal_state_reads_the_normalized_timeline(events, expected): + assert lab_state.terminal_state(events) == expected + + +# --- Storage ---------------------------------------------------------------- + + +def test_state_survives_a_reload(tmp_path): + path = tmp_path / "state.json" + state = ready_for_s1(path) + state.mark_recovered("s1", str(tmp_path / "s1-20260814T000000Z")) + state.record_capture("s1", "conclusion") + + reloaded = LabState(path) + + assert reloaded.has("agent_setup_acknowledged") + assert reloaded.run_status("s1") == "recovered" + assert reloaded.evidence_dir("s1") == str(tmp_path / "s1-20260814T000000Z") + reloaded.require_run("s2") + + +def test_state_is_written_through_a_sibling_temporary_file(tmp_path, monkeypatch): + """Only a rename within the same directory is atomic, so the complete + JSON must be written to a sibling temporary file first and renamed into + place -- never streamed into `state.json` itself.""" + path = tmp_path / "state.json" + state = LabState(path) + replaced = {} + real_replace = os.replace + + def spy_replace(source, target): + replaced["source"] = str(source) + replaced["target"] = str(target) + replaced["source_is_sibling"] = Path(source).parent == Path(target).parent + replaced["content"] = Path(source).read_text() + return real_replace(source, target) + + monkeypatch.setattr(lab_state.os, "replace", spy_replace) + state.mark("baseline_passed") + monkeypatch.undo() + + assert replaced["target"] == str(path) + assert replaced["source"] != str(path) + assert replaced["source_is_sibling"] + assert json.loads(replaced["content"])["stages"]["baseline_passed"]["at"] + assert [item.name for item in tmp_path.iterdir()] == ["state.json"] + + +def test_an_interrupted_write_leaves_the_previous_state_readable(tmp_path, monkeypatch): + path = tmp_path / "state.json" + state = LabState(path) + state.mark("baseline_passed") + before = path.read_text() + + def failing_replace(source, target): + raise OSError("interrupted") + + monkeypatch.setattr(lab_state.os, "replace", failing_replace) + with pytest.raises(OSError): + state.mark("agent_setup_acknowledged") + monkeypatch.undo() + + assert path.read_text() == before + assert LabState(path).has("baseline_passed") + assert not LabState(path).has("agent_setup_acknowledged") + + +def test_state_records_the_environment_it_belongs_to(tmp_path): + path = tmp_path / "state.json" + LabState( + path, + environment="sre-lab-one", + subscription_id="sub-one", + resource_group="rg-one", + ).mark("baseline_passed") + + stored = json.loads(path.read_text()) + + assert stored["environment"] == "sre-lab-one" + assert stored["subscription_id"] == "sub-one" + assert stored["resource_group"] == "rg-one" + assert stored["stages"]["baseline_passed"]["at"].endswith("Z") + + +@pytest.mark.parametrize( + "override", + ( + {"environment": "sre-lab-two"}, + {"subscription_id": "sub-two"}, + {"resource_group": "rg-two"}, + ), +) +def test_state_from_another_environment_is_refused(tmp_path, override): + path = tmp_path / "state.json" + binding = { + "environment": "sre-lab-one", + "subscription_id": "sub-one", + "resource_group": "rg-one", + } + LabState(path, **binding).mark("baseline_passed") + + with pytest.raises(EnvironmentMismatch): + LabState(path, **dict(binding, **override)) + + +def test_evidence_directory_is_recorded_per_stage_and_scenario(tmp_path): + path = tmp_path / "state.json" + state = LabState(path) + state.mark("baseline_passed", evidence_dir=str(tmp_path / "baseline-1")) + state.mark("agent_setup_acknowledged") + state.mark_recovered("s1", str(tmp_path / "s1-1")) + state.record_capture("s1", "conclusion") + + stored = json.loads(path.read_text()) + + assert stored["stages"]["baseline_passed"]["evidence_dir"] == str(tmp_path / "baseline-1") + assert stored["scenarios"]["s1"] == { + "run_status": "recovered", + "capture_status": "conclusion", + "evidence_dir": str(tmp_path / "s1-1"), + } + + +def test_unknown_stage_names_are_refused(tmp_path): + state = LabState(tmp_path / "state.json") + with pytest.raises(ValueError): + state.mark("almost_done") + + +def test_a_corrupt_state_file_is_reported_not_silently_reset(tmp_path): + path = tmp_path / "state.json" + path.write_text("{not json") + + with pytest.raises(lab_state.LabStateError): + LabState(path) + + +# --- Command line ----------------------------------------------------------- + + +def test_cli_require_run_fails_with_the_missing_state_named(tmp_path): + path = tmp_path / "state.json" + + result = run_cli(path, ["require-run", "s1"]) + + assert result.returncode == 1 + assert "baseline_passed" in result.stderr + assert "agent_setup_acknowledged" in result.stderr + + +def test_cli_require_run_succeeds_once_the_prerequisites_are_recorded(tmp_path): + path = tmp_path / "state.json" + assert run_cli(path, ["mark", "baseline_passed"]).returncode == 0 + assert run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n").returncode == 0 + + result = run_cli(path, ["require-run", "s1"]) + + assert result.returncode == 0, result.stderr + + +def test_cli_marks_recovery_and_capture_and_resolves_the_evidence_directory(tmp_path): + path = tmp_path / "state.json" + evidence_dir = tmp_path / "s1-20260814T000000Z" + evidence_dir.mkdir() + (evidence_dir / "normalized-timeline.json").write_text( + json.dumps([{"state": "alert-fired"}, {"state": "thread-created"}, {"state": "conclusion"}]) + ) + run_cli(path, ["mark", "baseline_passed"]) + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n") + + assert run_cli(path, ["mark-recovered", "s1", str(evidence_dir)]).returncode == 0 + resolved = run_cli(path, ["evidence-dir", "s1"]) + assert resolved.returncode == 0, resolved.stderr + assert resolved.stdout.strip() == str(evidence_dir) + + recorded = run_cli( + path, + [ + "record-capture", + "s1", + "--timeline", + str(evidence_dir / "normalized-timeline.json"), + "--evidence-dir", + str(evidence_dir), + ], + ) + assert recorded.returncode == 0, recorded.stderr + assert "conclusion" in recorded.stdout + assert run_cli(path, ["require-run", "s2"]).returncode == 0 + + +def test_cli_record_capture_keeps_a_missing_conclusion_visible(tmp_path): + path = tmp_path / "state.json" + evidence_dir = tmp_path / "s1-20260814T000000Z" + evidence_dir.mkdir() + (evidence_dir / "normalized-timeline.json").write_text( + json.dumps( + [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion-missing"}, + ] + ) + ) + run_cli(path, ["mark", "baseline_passed"]) + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n") + run_cli(path, ["mark-recovered", "s1", str(evidence_dir)]) + + recorded = run_cli( + path, + ["record-capture", "s1", "--timeline", str(evidence_dir / "normalized-timeline.json")], + ) + + assert recorded.returncode == 0, recorded.stderr + assert "conclusion-missing" in recorded.stdout + blocked = run_cli(path, ["require-run", "s2"]) + assert blocked.returncode == 1 + assert "s1_captured" in blocked.stderr + + +def test_cli_evidence_dir_without_a_run_names_the_command_to_run(tmp_path): + result = run_cli(tmp_path / "state.json", ["evidence-dir", "s1"]) + + assert result.returncode == 1 + assert "lab.sh run s1" in result.stderr + assert "Traceback" not in result.stderr + + +def test_cli_refuses_a_state_file_bound_to_another_environment(tmp_path): + path = tmp_path / "state.json" + assert run_cli(path, ["mark", "baseline_passed"]).returncode == 0 + + result = run_cli( + path, + ["require-run", "s1"], + env={"AZURE_RESOURCE_GROUP": "rg-somewhere-else"}, + ) + + assert result.returncode == 1 + assert "rg-somewhere-else" in result.stderr + assert "Traceback" not in result.stderr + + +def test_cli_binds_new_state_to_the_current_configuration(tmp_path): + path = tmp_path / "state.json" + + run_cli(path, ["mark", "deployed"]) + + stored = json.loads(path.read_text()) + assert stored["environment"] == ENVIRONMENT["AZURE_ENV_NAME"] + assert stored["subscription_id"] == ENVIRONMENT["AZURE_SUBSCRIPTION_ID"] + assert stored["resource_group"] == ENVIRONMENT["AZURE_RESOURCE_GROUP"] + + +# --- Interactive acknowledgement ------------------------------------------- + + +ACKNOWLEDGE_ENV = { + "SRE_AGENT_NAME": "sre-agent-lab", + "SRE_AGENT_RESOURCE_ID": ( + "/subscriptions/11111111-2222-3333-4444-555555555555/resourceGroups" + "/rg-sre-lab-state/providers/Microsoft.App/agents/sre-agent-lab" + ), + "SRE_REPOSITORY_URL": "https://github.com/example/devguidesample", + "SRE_REPOSITORY_BRANCH": "feature/sre-agent-azd-lab", + "SRE_KNOWLEDGE_PATH": "runbooks/incident-response.md", +} + + +def test_acknowledge_prints_every_setting_the_operator_must_verify(tmp_path): + path = tmp_path / "state.json" + + result = run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n", env=ACKNOWLEDGE_ENV) + + assert result.returncode == 0, result.stderr + for value in ACKNOWLEDGE_ENV.values(): + assert value in result.stdout + assert "Review" in result.stdout + for alert_name in ( + "alert-sre-lab-s1-http500", + "alert-sre-lab-s2-latency", + "alert-sre-lab-s3-storage-rbac", + ): + assert alert_name in result.stdout + + +def test_acknowledge_records_the_stage_only_after_the_exact_word(tmp_path): + path = tmp_path / "state.json" + + result = run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n", env=ACKNOWLEDGE_ENV) + + assert result.returncode == 0, result.stderr + assert LabState(path).has("agent_setup_acknowledged") + + +@pytest.mark.parametrize("answer", ("", "y\n", "yes\n", "ACKNOWLEDGE\n", "acknowledged\n", "ok\n")) +def test_acknowledge_refuses_anything_but_the_exact_word(tmp_path, answer): + path = tmp_path / "state.json" + + result = run_cli(path, ["acknowledge-agent"], stdin=answer, env=ACKNOWLEDGE_ENV) + + assert result.returncode != 0 + assert not path.exists() or not LabState(path).has("agent_setup_acknowledged") + + +def test_acknowledge_is_never_inferred_from_the_environment(tmp_path): + """Configured environment variables describe intent, not a portal state: + the Agent's repository/knowledge/response-plan wiring has no official + stable API, so only a human who looked at the portal may record it.""" + path = tmp_path / "state.json" + env = dict(ACKNOWLEDGE_ENV) + env.update( + { + "SRE_AGENT_SETUP_ACKNOWLEDGED": "true", + "SRE_AGENT_SETUP_COMPLETE": "1", + "CI": "true", + } + ) + + result = run_cli(path, ["acknowledge-agent"], stdin="", env=env) + + assert result.returncode != 0 + assert not path.exists() or not LabState(path).has("agent_setup_acknowledged") + + +def test_acknowledge_reports_settings_that_are_not_configured(tmp_path): + path = tmp_path / "state.json" + env = {key: "" for key in ACKNOWLEDGE_ENV} + + result = run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n", env=env) + + assert result.returncode == 0, result.stderr + assert "(not configured)" in result.stdout + + +def test_acknowledge_stores_what_was_shown_without_any_secret(tmp_path): + path = tmp_path / "state.json" + env = dict(ACKNOWLEDGE_ENV) + env["SRE_AGENT_CLIENT_SECRET"] = "super-secret-value" + + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n", env=env) + + stored = path.read_text() + assert "super-secret-value" not in stored + details = json.loads(stored)["stages"]["agent_setup_acknowledged"]["details"] + assert details["repository_url"] == ACKNOWLEDGE_ENV["SRE_REPOSITORY_URL"] + assert details["repository_branch"] == ACKNOWLEDGE_ENV["SRE_REPOSITORY_BRANCH"] + assert details["knowledge_path"] == ACKNOWLEDGE_ENV["SRE_KNOWLEDGE_PATH"] + assert details["response_plan_mode"] == "Review" + assert details["alert_rules"] == [ + "alert-sre-lab-s1-http500", + "alert-sre-lab-s2-latency", + "alert-sre-lab-s3-storage-rbac", + ] diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_score.py b/monitor/sre-agent-event-lab/scripts/tests/test_score.py new file mode 100644 index 0000000..6f0477d --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_score.py @@ -0,0 +1,336 @@ +"""Behavioural tests for `score.py`, the lab's rubric scorer. + +Scoring is where an unproven claim would do the most damage: awarding +"actual evidence used" because a conclusion merely exists would turn the +lab into a rubber stamp. So the two properties pinned here are that a real +failure marker (`thread-not-created` / `investigation-missing` / +`conclusion-missing`) scores zero and stays visible, and that a criterion +with no structured judgement is reported `MANUAL` with no points awarded. +""" +import importlib.util +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +MODULE_PATH = Path(__file__).parents[1] / "score.py" +STATE_MODULE_PATH = Path(__file__).parents[1] / "lab_state.py" + + +def load_module(path, name): + sys.path.insert(0, str(path.parent)) + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +score = load_module(MODULE_PATH, "score") +lab_state = load_module(STATE_MODULE_PATH, "lab_state") + + +CONCLUSION_TIMELINE = [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion"}, +] +MISSING_CONCLUSION_TIMELINE = [ + {"state": "alert-fired"}, + {"state": "thread-created"}, + {"state": "investigating"}, + {"state": "conclusion-missing"}, +] +FULL_REVIEW = { + "impact_scope": {"met": True, "detail": "Named the Container App and both routes."}, + "direct_cause": {"met": True, "detail": "Named FAILURE_MODE=http500."}, + "actual_evidence": {"met": True, "detail": "Quoted AppRequests rows."}, + "safe_minimum_mitigation": {"met": True, "detail": "Proposed reverting the env var."}, + "uncertainty": {"met": True, "detail": "Flagged the unverified dependency."}, +} + + +def write_evidence(root, scenario, timeline, review=None, name=None): + evidence_dir = root / (name or f"{scenario}-20260814T000000Z") + evidence_dir.mkdir(parents=True, exist_ok=True) + (evidence_dir / "normalized-timeline.json").write_text(json.dumps(timeline)) + if review is not None: + (evidence_dir / "conclusion-review.json").write_text(json.dumps(review)) + return evidence_dir + + +def make_state(evidence_root, scenarios): + """A state file recording one recovered+captured run per scenario.""" + state = lab_state.LabState(evidence_root / "state.json") + state.mark("baseline_passed") + state.mark("agent_setup_acknowledged") + for scenario, evidence_dir in scenarios.items(): + timeline = json.loads((evidence_dir / "normalized-timeline.json").read_text()) + state.mark_recovered(scenario, str(evidence_dir)) + state.record_capture(scenario, lab_state.terminal_state(timeline), str(evidence_dir)) + return state + + +def criteria_by_id(scenario_result): + return {item["id"]: item for item in scenario_result["criteria"]} + + +# --- Rubric ----------------------------------------------------------------- + + +def test_the_rubric_is_the_documented_ten_point_one(): + assert [(item.id, item.max_points) for item in score.CRITERIA] == [ + ("impact_scope", 2), + ("direct_cause", 3), + ("actual_evidence", 2), + ("safe_minimum_mitigation", 2), + ("uncertainty", 1), + ] + assert sum(item.max_points for item in score.CRITERIA) == 10 + + +def test_a_fully_reviewed_conclusion_earns_every_point(): + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, FULL_REVIEW) + + assert result["points"] == 10 + assert result["max_points"] == 10 + assert result["verdict"] == "PASS" + assert {item["status"] for item in result["criteria"]} == {"PASS"} + + +def test_an_unmet_criterion_costs_exactly_its_points(): + review = dict(FULL_REVIEW) + review["direct_cause"] = {"met": False, "detail": "Named a symptom, not the cause."} + + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review) + + assert criteria_by_id(result)["direct_cause"]["status"] == "FAIL" + assert criteria_by_id(result)["direct_cause"]["points"] == 0 + assert "symptom" in criteria_by_id(result)["direct_cause"]["detail"] + assert result["points"] == 7 + assert result["verdict"] == "PARTIAL" + + +@pytest.mark.parametrize( + "points, verdict", + ((10, "PASS"), (8, "PASS"), (7, "PARTIAL"), (5, "PARTIAL"), (4, "FAIL"), (0, "FAIL")), +) +def test_the_documented_thresholds_decide_the_verdict(points, verdict): + assert score.verdict_for(points, manual=0) == verdict + + +def test_any_manual_criterion_keeps_the_verdict_incomplete(): + assert score.verdict_for(9, manual=1) == "INCOMPLETE" + + +# --- Unavailable structured evidence --------------------------------------- + + +def test_a_missing_review_is_manual_and_awards_nothing(): + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, None) + + assert result["points"] == 0 + assert result["manual_points"] == 10 + assert result["verdict"] == "INCOMPLETE" + assert {item["status"] for item in result["criteria"]} == {"MANUAL"} + for item in result["criteria"]: + assert item["points"] == 0 + + +def test_one_unavailable_field_is_manual_while_the_rest_score(): + review = {key: value for key, value in FULL_REVIEW.items() if key != "uncertainty"} + + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review) + + assert criteria_by_id(result)["uncertainty"]["status"] == "MANUAL" + assert criteria_by_id(result)["uncertainty"]["points"] == 0 + assert result["points"] == 9 + assert result["manual_points"] == 1 + assert result["verdict"] == "INCOMPLETE" + + +@pytest.mark.parametrize("unusable", ({"detail": "no verdict"}, {"met": "yes"}, "PASS", None)) +def test_an_unusable_judgement_is_manual_never_a_pass(unusable): + review = dict(FULL_REVIEW) + review["impact_scope"] = unusable + + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review) + + assert criteria_by_id(result)["impact_scope"]["status"] == "MANUAL" + assert criteria_by_id(result)["impact_scope"]["points"] == 0 + + +# --- Real failure markers --------------------------------------------------- + + +@pytest.mark.parametrize( + "capture_status", ("thread-not-created", "investigation-missing", "conclusion-missing") +) +def test_a_missing_agent_output_scores_zero_and_stays_visible(capture_status): + result = score.score_scenario("s1", capture_status, MISSING_CONCLUSION_TIMELINE, FULL_REVIEW) + + assert result["points"] == 0 + assert result["manual_points"] == 0 + assert result["capture_status"] == capture_status + assert result["verdict"] == "FAIL" + for item in result["criteria"]: + assert item["status"] == "FAIL" + assert capture_status in item["detail"] + + +def test_a_scenario_that_was_never_captured_is_not_manual(): + """No capture at all is a known failure, not an unknown: reporting it + `MANUAL` would let an unrun scenario wait forever for a human instead of + failing the lab.""" + result = score.score_scenario("s3", None, [], None) + + assert result["verdict"] == "FAIL" + assert result["points"] == 0 + assert result["manual_points"] == 0 + + +# --- Scorecard -------------------------------------------------------------- + + +def test_scorecard_covers_every_scenario_and_totals_them(tmp_path): + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, FULL_REVIEW) + for scenario in ("s1", "s2", "s3") + } + state = make_state(tmp_path, directories) + + scorecard = score.build_scorecard(state, tmp_path) + + assert sorted(scorecard["scenarios"]) == ["s1", "s2", "s3"] + assert scorecard["overall"]["points"] == 30 + assert scorecard["overall"]["max_points"] == 30 + assert scorecard["overall"]["verdict"] == "PASS" + assert scorecard["scenarios"]["s1"]["evidence_dir"] == str(directories["s1"]) + + +def test_overall_success_requires_every_scenario_to_be_scored(tmp_path): + directories = { + "s1": write_evidence(tmp_path, "s1", CONCLUSION_TIMELINE, FULL_REVIEW), + "s2": write_evidence(tmp_path, "s2", MISSING_CONCLUSION_TIMELINE, FULL_REVIEW), + } + state = make_state(tmp_path, directories) + + scorecard = score.build_scorecard(state, tmp_path) + + assert scorecard["scenarios"]["s2"]["verdict"] == "FAIL" + assert scorecard["scenarios"]["s3"]["verdict"] == "FAIL" + assert scorecard["overall"]["verdict"] == "FAIL" + + +def test_manual_criteria_hold_the_overall_verdict_at_incomplete(tmp_path): + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, None) + for scenario in ("s1", "s2", "s3") + } + state = make_state(tmp_path, directories) + + scorecard = score.build_scorecard(state, tmp_path) + + assert scorecard["overall"]["verdict"] == "INCOMPLETE" + assert scorecard["overall"]["manual_points"] == 30 + assert scorecard["overall"]["points"] == 0 + + +def test_the_table_shows_every_criterion_with_its_status(tmp_path): + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, FULL_REVIEW) + for scenario in ("s1", "s2", "s3") + } + state = make_state(tmp_path, directories) + + table = score.render_table(score.build_scorecard(state, tmp_path)) + + assert table.splitlines()[0].split("\t") == [ + "SCENARIO", + "CRITERION", + "STATUS", + "POINTS", + "DETAIL", + ] + for scenario in ("s1", "s2", "s3"): + for item in score.CRITERIA: + assert f"{scenario}\t{item.id}\tPASS\t{item.max_points}/{item.max_points}" in table + assert f"{scenario}\tTOTAL\tPASS\t10/10" in table + assert "OVERALL\tTOTAL\tPASS\t30/30" in table + + +# --- Command line ----------------------------------------------------------- + + +def run_cli(evidence_root, args=()): + process_env = dict(os.environ) + process_env.update( + { + "AZURE_ENV_NAME": "sre-lab-score", + "AZURE_SUBSCRIPTION_ID": "11111111-2222-3333-4444-555555555555", + "AZURE_RESOURCE_GROUP": "rg-sre-lab-score", + } + ) + return subprocess.run( + [sys.executable, str(MODULE_PATH), "--evidence-root", str(evidence_root), *args], + capture_output=True, + text=True, + env=process_env, + ) + + +def test_cli_writes_the_scorecard_next_to_the_evidence(tmp_path): + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, FULL_REVIEW) + for scenario in ("s1", "s2", "s3") + } + make_state(tmp_path, directories) + + result = run_cli(tmp_path) + + assert result.returncode == 0, result.stdout + result.stderr + scorecard = json.loads((tmp_path / "scorecard.json").read_text()) + assert scorecard["overall"]["verdict"] == "PASS" + assert "OVERALL\tTOTAL\tPASS\t30/30" in result.stdout + assert lab_state.LabState(tmp_path / "state.json").has("scored") + + +def test_cli_fails_when_a_scenario_never_produced_a_conclusion(tmp_path): + directories = { + "s1": write_evidence(tmp_path, "s1", CONCLUSION_TIMELINE, FULL_REVIEW), + "s2": write_evidence(tmp_path, "s2", MISSING_CONCLUSION_TIMELINE, FULL_REVIEW), + "s3": write_evidence(tmp_path, "s3", CONCLUSION_TIMELINE, FULL_REVIEW), + } + make_state(tmp_path, directories) + + result = run_cli(tmp_path) + + assert result.returncode == 1 + assert "conclusion-missing" in result.stdout + assert json.loads((tmp_path / "scorecard.json").read_text())["scenarios"]["s2"]["verdict"] == "FAIL" + + +def test_cli_reports_manual_criteria_without_awarding_points(tmp_path): + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, None) + for scenario in ("s1", "s2", "s3") + } + make_state(tmp_path, directories) + + result = run_cli(tmp_path) + + assert result.returncode == 0, result.stdout + result.stderr + assert "MANUAL" in result.stdout + assert "OVERALL\tTOTAL\tINCOMPLETE\t0/30" in result.stdout + + +def test_cli_without_any_state_explains_what_to_run_first(tmp_path): + result = run_cli(tmp_path) + + assert result.returncode == 1 + assert "lab.sh run" in result.stderr + assert "Traceback" not in result.stderr From b7e0cfa8768afd98cd6fa5e2af5967140c086652 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 17:26:48 +0900 Subject: [PATCH 09/26] fix(sre-lab): clear stale capture status, fail closed when no alert fires Address Task 4 review findings: - LabState.mark_recovered/mark_failed now discard the scenario's previous capture_status (and its capture-side evidence_dir) before recording the new run's outcome. A conclusion captured against an earlier, superseded run could otherwise keep satisfying sX_captured -- and therefore the next scenario's gate and the scorer -- even though nothing has been captured for the current run yet. Both a recovered and a failed re-run now clear it; tested with the state reloaded from disk to prove it is actually persisted. - run-scenario.sh now calls `lab_state mark-failed` with the evidence directory and a reason before exiting when the target alert never fires within the poll window, matching every other failure path. The wait is now bounded by LAB_ALERT_FIRE_TIMEOUT_SECONDS / LAB_ALERT_FIRE_POLL_INTERVAL_SECONDS (default 720s/20s, same pattern as the existing alert-resolve and health-check overrides) so this is exercised in the test suite instead of only in production. The pre-existing recovery trap is untouched and still reverts the injected fault on this path. - LabState._load validates that `stages`, `scenarios`, and every entry inside them decode to JSON objects. A wrong type (list, string, null, number) now raises the same clean LabStateError a corrupt file already produces, instead of a raw TypeError/AttributeError traceback surfacing later from mark()/mark_recovered()/ record_capture(); the file is never silently reset to defaults. - Documented, in lab_state.py's module docstring, that state.json has no cross-process locking and concurrent operators racing the same environment can lose an update to a later writer -- a deliberate choice for a single-operator lab. No locking was added; nothing in the test suite showed a real concurrent-use need. Tests (all written RED first): test_lab_state.py gains test_rerunning_a_recovered_scenario_clears_the_stale_capture_status, test_rerunning_a_failed_scenario_clears_the_stale_capture_status, test_reloading_after_a_rerun_still_shows_the_cleared_capture_status, a parametrized test_a_state_file_with_the_wrong_json_shape_is_refused_not_reset (10 shapes), test_cli_reports_a_malformed_state_file_without_a_python_traceback, and test_module_documents_that_concurrent_operators_are_unsupported. test_lab_scripts.py gains test_run_scenario_marks_failed_when_the_alert_never_fires, backed by a new alert_fires=False mode in lab_script_harness.py's fake az. Verification: python3 -m pytest monitor/sre-agent-event-lab/scripts/tests -q 307 passed (was 291 before this change) bash -n monitor/sre-agent-event-lab/scripts/*.sh python3 -c "import lab_state, score" # Python 3.9.6, clean Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../sre-agent-event-lab/scripts/lab_state.py | 76 ++++++++++- .../scripts/run-scenario.sh | 17 ++- .../scripts/tests/lab_script_harness.py | 9 +- .../scripts/tests/test_lab_scripts.py | 28 ++++ .../scripts/tests/test_lab_state.py | 124 ++++++++++++++++++ 5 files changed, 248 insertions(+), 6 deletions(-) diff --git a/monitor/sre-agent-event-lab/scripts/lab_state.py b/monitor/sre-agent-event-lab/scripts/lab_state.py index 0f91310..36e80fc 100755 --- a/monitor/sre-agent-event-lab/scripts/lab_state.py +++ b/monitor/sre-agent-event-lab/scripts/lab_state.py @@ -12,7 +12,11 @@ * Honesty: a capture is only "successful" when the normalized timeline holds a real `conclusion` event. `thread-not-created`, `investigation-missing` and `conclusion-missing` are recorded verbatim - and never promoted to success, by any code path. + and never promoted to success, by any code path. A re-run -- whether it + ends in `mark_recovered` or `mark_failed` -- clears the scenario's + previous `capture_status` first: a conclusion captured against a run + that no longer exists must never let a later run's capture stage, or the + scorer, reuse it. Only a capture recorded *after* the current run counts. * Binding: the file records the azd environment, subscription and resource group it belongs to and refuses to be read against a different one, so a state file left behind by another lab can never unlock a run here. @@ -20,7 +24,20 @@ Storage is `evidence/state.json`, written by rendering the whole document into a sibling temporary file and `os.replace`-ing it into place; a rename within a directory is the only write that cannot leave a -half-written state file behind. +half-written state file behind. Decoded JSON is validated before use: +`stages`, `scenarios`, and every entry inside them must be JSON objects, +or the file is refused with a clean `LabStateError` -- never a raw +`TypeError`/`AttributeError` traceback, and never a silent reset that +would discard whatever an operator had already recorded. + +Concurrency: this module assumes one operator drives the lab at a time. +Nothing here locks `state.json` across processes, so two commands that +read-modify-write it at the same moment can race and the later write +wins; the atomic `os.replace` only guarantees each individual write is +whole, not that concurrent writes are serialized. That is a deliberate +trade for a single-operator lab -- add real file locking (e.g. +`fcntl.flock` around load-mutate-save) only if concurrent operators +become a real, observed need, not in anticipation of one. Python 3.9 compatible (the lab's documented floor): no PEP 604 unions, no structural pattern matching, and no third-party imports. @@ -175,6 +192,7 @@ def _load(self) -> Dict[str, Any]: ) document.setdefault("stages", {}) document.setdefault("scenarios", {}) + self._verify_shape(document) self._verify_binding(document) for key, value in ( ("environment", self.environment), @@ -185,6 +203,42 @@ def _load(self) -> Dict[str, Any]: document[key] = value return document + def _verify_shape(self, document: Dict[str, Any]) -> None: + """Refuse decoded JSON whose containers are not JSON objects. + + `stages` and `scenarios`, and every entry inside them, are always + treated as objects (subscripted, `.setdefault`-ed, mutated in + place). A wrong type there -- a list, a string, `null`, a number -- + would otherwise surface many calls later as a raw `TypeError` or + `AttributeError` from deep inside `mark`/`mark_recovered`/ + `record_capture`. Catching it here, once, turns every such case + into the same clean `LabStateError` a corrupt file already + produces, and never silently discards the bad value by resetting + it to `{}`. + """ + for container_key in ("stages", "scenarios"): + container = document.get(container_key) + if not isinstance(container, dict): + raise LabStateError( + "Lab state {0} field {1!r} must be a JSON object, not " + "{2}. Inspect or remove the file before " + "continuing.".format( + self.path, container_key, type(container).__name__ + ) + ) + for entry_name, entry in container.items(): + if not isinstance(entry, dict): + raise LabStateError( + "Lab state {0} field {1}.{2!r} must be a JSON " + "object, not {3}. Inspect or remove the file " + "before continuing.".format( + self.path, + container_key, + entry_name, + type(entry).__name__, + ) + ) + def _verify_binding(self, document: Dict[str, Any]) -> None: for key, current in ( ("environment", self.environment), @@ -303,8 +357,25 @@ def _remedy(missing: Sequence[str]) -> str: return "Run: lab.sh capture {0}".format(scenario) return "" + @staticmethod + def _start_new_attempt(entry: Dict[str, Any]) -> None: + """Discard the previous run's terminal capture outcome. + + `mark_recovered` and `mark_failed` both mean "a run of this + scenario just ended"; every previous `capture_status` (and the + capture-side `evidence_dir` it was recorded against) describes a + run that no longer exists once a new one starts. Leaving it in + place would let a stale `conclusion` from an earlier attempt keep + satisfying `sX_captured` -- and therefore the next scenario's gate, + and the scorer -- even though nothing has been captured for *this* + run yet. A fresh capture, recorded after this call, is the only + thing that can set it again. + """ + entry.pop("capture_status", None) + def mark_recovered(self, scenario: str, evidence_dir: Optional[str] = None) -> None: entry = self._scenario(scenario) + self._start_new_attempt(entry) entry["run_status"] = RUN_RECOVERED entry.pop("failure_reason", None) if evidence_dir: @@ -318,6 +389,7 @@ def mark_failed( reason: str = "", ) -> None: entry = self._scenario(scenario) + self._start_new_attempt(entry) entry["run_status"] = RUN_FAILED if reason: entry["failure_reason"] = reason diff --git a/monitor/sre-agent-event-lab/scripts/run-scenario.sh b/monitor/sre-agent-event-lab/scripts/run-scenario.sh index 1a02585..09417af 100755 --- a/monitor/sre-agent-event-lab/scripts/run-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/run-scenario.sh @@ -24,6 +24,8 @@ lab_state require-run "${SCENARIO}" readonly ALERT_RESOLVE_TIMEOUT_SECONDS="${LAB_ALERT_RESOLVE_TIMEOUT_SECONDS:-900}" readonly ALERT_RESOLVE_POLL_INTERVAL_SECONDS="${LAB_ALERT_RESOLVE_POLL_INTERVAL_SECONDS:-20}" readonly RECOVERY_HEALTH_TIMEOUT_SECONDS="${LAB_RECOVERY_HEALTH_TIMEOUT_SECONDS:-600}" +readonly ALERT_FIRE_TIMEOUT_SECONDS="${LAB_ALERT_FIRE_TIMEOUT_SECONDS:-720}" +readonly ALERT_FIRE_POLL_INTERVAL_SECONDS="${LAB_ALERT_FIRE_POLL_INTERVAL_SECONDS:-20}" APP_NAME="$(deployment_output containerAppName)" APP_FQDN="$(deployment_output containerAppFqdn)" @@ -153,7 +155,7 @@ esac readonly ALERT_RULE_NAME started="${SECONDS}" -while (( SECONDS - started < 720 )); do +while (( SECONDS - started < ALERT_FIRE_TIMEOUT_SECONDS )); do alerts_json="$(az rest \ --method get \ --url "https://management.azure.com/subscriptions/${SUBSCRIPTION_ID}/providers/Microsoft.AlertsManagement/alerts?api-version=2019-03-01&targetResourceGroup=${RESOURCE_GROUP}&monitorCondition=Fired")" @@ -168,11 +170,20 @@ while (( SECONDS - started < 720 )); do <<<"${alerts_json}")" break fi - sleep 20 + sleep "${ALERT_FIRE_POLL_INTERVAL_SECONDS}" done +# No alert ever firing is a distinct failure from one that fires and never +# resolves, but it is just as unusable as evidence: recorded as a failed +# run -- with the evidence directory and a reason -- before this exits, so +# a later `lab.sh score` or the next scenario's gate can never read this +# attempt as anything but failed. The EXIT trap (still armed here) still +# reverts whatever was injected above. if [[ -z "${ALERT_ID}" ]]; then - echo "Alert ${ALERT_RULE_NAME} did not fire within 720s." >&2 + ALERT_NEVER_FIRED_REASON="alert ${ALERT_RULE_NAME} did not fire within ${ALERT_FIRE_TIMEOUT_SECONDS}s." + lab_state mark-failed "${SCENARIO}" "${EVIDENCE_DIR}" --reason "${ALERT_NEVER_FIRED_REASON}" + echo "${ALERT_NEVER_FIRED_REASON}" >&2 + echo "Evidence directory: ${EVIDENCE_DIR}" >&2 exit 1 fi diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 6363cad..6f66fc0 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -183,7 +183,11 @@ def _az_stub_source(log_path, state_dir): printf '{{"properties": {{"principalId": "%s", "roleDefinitionId": "/subscriptions/{SUBSCRIPTION_ID}/providers/Microsoft.Authorization/roleDefinitions/{MONITORING_CONTRIBUTOR_ROLE_ID}", "scope": "/subscriptions/{SUBSCRIPTION_ID}"}}}}\\n' \\ "${{principal_id}}" elif [[ "$*" == *"monitorCondition=Fired"* ]]; then - printf '%s\\n' '{ALERTS_JSON}' + if [[ -f "${{state}}/alert_never_fires" ]]; then + printf '{{"value": []}}\\n' + else + printf '%s\\n' '{ALERTS_JSON}' + fi elif [[ "$*" == *"/Microsoft.AlertsManagement/alerts/"* ]]; then printf '{{"properties": {{"essentials": {{"monitorCondition": "%s", "startDateTime": "2026-08-14T00:05:00Z"}}}}}}\\n' \\ "$(cat "${{state}}/alert_condition")" @@ -254,6 +258,7 @@ def make_lab( azd_values=None, missing_key_mode="azd_1_29", alert_resolves=True, + alert_fires=True, capture_timeline=CONCLUSION_TIMELINE, ): """A throwaway copy of the lab plus fake CLIs; returns a run context.""" @@ -274,6 +279,8 @@ def make_lab( (state_dir / "alert_condition").write_text("Fired\n") if not alert_resolves: (state_dir / "alert_stays_fired").write_text("1\n") + if not alert_fires: + (state_dir / "alert_never_fires").write_text("1\n") az_log = tmp_path / "az-calls.log" azd_log = tmp_path / "azd-calls.log" diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index 8f37a1e..57be95f 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -216,6 +216,34 @@ def test_run_scenario_fails_when_the_alert_never_resolves(tmp_path): assert timeline["alert_resolved_at"] is None +def test_run_scenario_marks_failed_when_the_alert_never_fires(tmp_path): + """No alert ever firing is a different failure than one that fires and + never resolves: nothing to recover from Azure Monitor's point of view, + but the run is still unusable evidence. It must be recorded as failed + -- with the evidence directory and a reason -- exactly like every other + failed run, and the fault that was injected must still be reverted by + the same recovery trap that protects every other exit path.""" + lab_run = make_lab(tmp_path, alert_fires=False) + lab_run.seed_state() + + result = lab_run.run( + "run-scenario.sh", + ["s1"], + env=dict(BOUNDED_WAITS, LAB_ALERT_FIRE_TIMEOUT_SECONDS="3", LAB_ALERT_FIRE_POLL_INTERVAL_SECONDS="1"), + ) + + assert result.returncode != 0 + assert "did not fire" in result.stderr + assert "FAILURE_MODE=none" in lab_run.az_calls(), ( + "the injected failure must still be reverted even though no alert ever fired" + ) + scenario_state = lab_run.scenario_state("s1") + assert scenario_state["run_status"] == "failed" + assert "did not fire" in scenario_state.get("failure_reason", "") + evidence_dir = sorted((lab_run.lab / "evidence").glob("s1-*"))[-1] + assert scenario_state["evidence_dir"] == str(evidence_dir) + + def test_a_failed_run_blocks_the_next_scenario(tmp_path): lab_run = make_lab(tmp_path, alert_resolves=False) lab_run.seed_state() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py index 73b62f3..e3e6754 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py @@ -142,6 +142,60 @@ def test_a_failed_run_does_not_satisfy_the_next_scenario(tmp_path): state.require_run("s2") +# --- Re-runs never let a stale capture_status linger ----------------------- + + +def test_rerunning_a_recovered_scenario_clears_the_stale_capture_status(tmp_path): + """A scenario that already produced a real conclusion, then gets + re-run (e.g. an operator re-injects the same fault to collect a second + capture), must not let the *previous* run's conclusion satisfy the next + scenario's gate before the *new* run has actually been captured.""" + state = ready_for_s1(tmp_path / "state.json") + state.mark_recovered("s1", str(tmp_path / "s1-first")) + state.record_capture("s1", "conclusion") + assert state.is_successful_capture("s1") + + state.mark_recovered("s1", str(tmp_path / "s1-second")) + + assert state.capture_status("s1") is None + assert not state.is_successful_capture("s1") + with pytest.raises(InvalidTransition, match="s1_captured"): + state.require_run("s2") + + +def test_rerunning_a_failed_scenario_clears_the_stale_capture_status(tmp_path): + """The same guarantee applies when the re-run ends in failure: a + conclusion captured on an earlier, since-superseded run must not let a + failed re-run's scenario entry keep reporting yesterday's success.""" + state = ready_for_s1(tmp_path / "state.json") + state.mark_recovered("s1", str(tmp_path / "s1-first")) + state.record_capture("s1", "conclusion") + assert state.is_successful_capture("s1") + + state.mark_failed("s1", str(tmp_path / "s1-second"), reason="alert never resolved") + + assert state.capture_status("s1") is None + assert not state.is_successful_capture("s1") + assert state.run_status("s1") == "failed" + with pytest.raises(InvalidTransition, match="s1_recovered"): + state.require_run("s2") + + +def test_reloading_after_a_rerun_still_shows_the_cleared_capture_status(tmp_path): + """The cleared status must be what is actually persisted, not just an + in-memory artifact of the same `LabState` instance.""" + path = tmp_path / "state.json" + state = ready_for_s1(path) + state.mark_recovered("s1", str(tmp_path / "s1-first")) + state.record_capture("s1", "conclusion") + state.mark_recovered("s1", str(tmp_path / "s1-second")) + + reloaded = LabState(path) + + assert reloaded.capture_status("s1") is None + assert not reloaded.is_successful_capture("s1") + + def test_require_run_rejects_an_unknown_scenario(tmp_path): state = ready_for_s1(tmp_path / "state.json") with pytest.raises(ValueError): @@ -352,6 +406,76 @@ def test_a_corrupt_state_file_is_reported_not_silently_reset(tmp_path): LabState(path) +# --- Decoded JSON with the wrong shape is refused, never silently reset ---- + + +@pytest.mark.parametrize( + "document", + ( + {"stages": []}, + {"stages": "baseline_passed"}, + {"stages": 1}, + {"stages": None}, + {"stages": {"baseline_passed": "yesterday"}}, + {"stages": {"baseline_passed": ["yesterday"]}}, + {"scenarios": []}, + {"scenarios": "s1"}, + {"scenarios": {"s1": "conclusion"}}, + {"scenarios": {"s1": ["conclusion"]}}, + ), + ids=( + "stages-list", + "stages-string", + "stages-int", + "stages-null", + "stage-entry-string", + "stage-entry-list", + "scenarios-list", + "scenarios-string", + "scenario-entry-string", + "scenario-entry-list", + ), +) +def test_a_state_file_with_the_wrong_json_shape_is_refused_not_reset(tmp_path, document): + """Every field the module treats as a container (`stages`, `scenarios`, + and each entry inside them) must actually be a JSON object once decoded. + A wrong type here must become the same clean `LabStateError` a corrupt + file produces -- never a raw `TypeError`/`AttributeError` traceback, and + never a silent reset back to `{}` that would erase whatever an operator + had already recorded.""" + path = tmp_path / "state.json" + path.write_text(json.dumps(document)) + + with pytest.raises(lab_state.LabStateError): + LabState(path) + + # The refusal must not have rewritten the file with a fresh default. + assert json.loads(path.read_text()) == document + + +def test_cli_reports_a_malformed_state_file_without_a_python_traceback(tmp_path): + path = tmp_path / "state.json" + path.write_text(json.dumps({"scenarios": "not-an-object"})) + + result = run_cli(path, ["show"]) + + assert result.returncode == 1 + assert "Traceback" not in result.stderr + assert "scenarios" in result.stderr + + +# --- Concurrency is documented, not silently assumed away ------------------- + + +def test_module_documents_that_concurrent_operators_are_unsupported(): + """No file locking guards `state.json` against two processes mutating + it at once; that is a deliberate simplicity choice for a lab one person + drives at a time, but it must be written down rather than left for + someone to discover by racing two runs together.""" + assert lab_state.__doc__ is not None + assert "concurrent" in lab_state.__doc__.lower() + + # --- Command line ----------------------------------------------------------- From a282ff5e6463a6220dcb333dd64c0e84e00a31ff Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 17:59:35 +0900 Subject: [PATCH 10/26] fix(sre-lab): clean external roles during azd down `azd down` deletes the azd-owned resource group; the only lab artifacts it cannot see are the subscription-scoped Monitoring Contributor assignments the Azure SRE Agent setup recorded. cleanup-external.sh now removes exactly those, after verifying every record, and cleanup.sh stops deleting a resource group of its own. - cleanup-external.sh verifies four things before deleting anything: the recorded assignment ID names a role assignment in the subscription this run resolved (AZURE_SUBSCRIPTION_ID > the current azd environment), the Azure CLI is signed in to exactly that subscription, the record carries the Agent principal it was created for, and the live assignment really holds that principal, the Monitoring Contributor role definition and subscription scope. Every record is verified before the first deletion, so one untrusted record leaves the subscription untouched. An empty, missing or already-absent record is a safe no-op (`RoleAssignmentNotFound` is what the CLI really answers -- recorded from azure-cli against a live subscription); malformed JSON, a missing principal, an unreadable assignment, a mismatch or a failed deletion all fail closed, before azd destroys anything. - The hook-set SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG reset moved out of `predown` into the `postdown` hook (`--reset-image-env`). `predown` runs before azd asks the operator to confirm the deletion, so clearing them there broke the environment of an operator who answered "no"; azd runs a post hook only after the action succeeded (HooksRunner.Invoke returns early on failure, cli/azd/pkg/ext/hooks_runner.go), which is exactly when the recorded image is really gone. `postdown` is an officially supported azd command hook. - Both azd env writes are pinned with `--cwd "${LAB_ROOT}"`, so running the hook by hand from the repository root no longer resolves whatever azd project the working directory happens to hold. - cleanup.sh is now a compatibility wrapper: it names `azd down --purge` and forwards to cleanup-external.sh. It never deletes a resource group unless `--legacy-delete-resource-group` is passed -- the documented recovery path for a lab whose azd environment was lost -- which still runs the subscription and `purpose`/`azd-env-name` tag checks first. Tests: scripts/tests/test_cleanup_external.py drives the hook as a program from a directory that is not the lab, on macOS's Bash 3.2, against a fake `az` holding a staged subscription and the azd 1.29 fake; the cleanup tests that lived in test_azd_hooks.py moved there. azd_fake now logs each argument with %q so a value azd is asked to clear is visible instead of vanishing. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 18 +- monitor/sre-agent-event-lab/azure.yaml | 5 + .../scripts/cleanup-external.sh | 319 +++++++++---- .../sre-agent-event-lab/scripts/cleanup.sh | 161 ++----- .../scripts/tests/azd_fake.py | 7 +- .../scripts/tests/cleanup_harness.py | 257 ++++++++++ .../scripts/tests/test_azd_hooks.py | 230 --------- .../scripts/tests/test_cleanup_external.py | 449 ++++++++++++++++++ .../scripts/tests/test_common.py | 75 ++- .../scripts/tests/test_lab_scripts.py | 48 +- 10 files changed, 1090 insertions(+), 479 deletions(-) create mode 100644 monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index dcbe54a..718265f 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -390,18 +390,28 @@ Dynamic rule은 3일·30 samples 전에는 발화하지 않으며 3주 전에는 ## 정리 -`azd`로 배포한 환경은 `azd down`으로 정리한다. predown hook(`scripts/cleanup-external.sh`)이 resource group 밖에 기록된 구독 범위 Monitoring Contributor assignment만 먼저 제거하고, resource group 삭제는 `azd`가 수행한다. +`azd`로 배포한 환경은 `azd down`으로 정리한다. resource group 삭제는 `azd`가 수행하고, azd가 볼 수 없는 두 가지만 hook이 처리한다. + +- predown hook `scripts/cleanup-external.sh --yes`: resource group 밖에 기록된 구독 범위 Monitoring Contributor assignment만 제거한다. 기록된 assignment ID가 현재 구독에 속하는지, Azure CLI가 그 구독에 로그인했는지, 실제 assignment의 principal·role·scope가 `evidence/agent-setup.json`의 기록과 일치하는지 확인한 뒤에만 삭제한다. 하나라도 어긋나면 아무것도 삭제하지 않고 중단한다. 기록이 없거나 이미 삭제된 assignment는 안전한 no-op다. +- postdown hook `scripts/cleanup-external.sh --reset-image-env --yes`: `azd-postprovision.sh`가 기록한 `SRE_CONTAINER_IMAGE`와 `SRE_IMAGE_TAG`를 비운다. predown이 아니라 postdown인 이유는 predown이 azd의 삭제 확인 프롬프트보다 먼저 실행되기 때문이다. 삭제를 취소한 환경의 image 값을 미리 지우면 안 된다. ```bash cd monitor/sre-agent-event-lab azd down --purge ``` -`scripts/cleanup.sh`는 azd 이전 방식으로 만든 `rg-sre-agent-event-lab-krc`를 정리하는 기존 스크립트다. 첫 명령은 dry-run이며 두 번째 명령만 삭제를 시작한다. +hook이 실패해 수동으로 다시 실행할 때는 아래처럼 직접 호출한다. 두 명령 모두 `--yes` 없이는 계획만 출력하며, 실행 위치와 무관하게 이 lab의 azd project(`--cwd`)를 대상으로 한다. + +```bash +monitor/sre-agent-event-lab/scripts/cleanup-external.sh --yes +monitor/sre-agent-event-lab/scripts/cleanup-external.sh --reset-image-env --yes +``` + +`scripts/cleanup.sh`는 기존 명령을 유지하기 위한 호환 wrapper다. 기본 동작은 `cleanup-external.sh` 위임뿐이며 resource group을 삭제하지 않는다. azd environment를 잃어버린 lab을 손으로 정리해야 할 때만 `--legacy-delete-resource-group`으로 예전 삭제 경로를 사용한다. 이 경로도 구독 일치와 `purpose=sre-agent-event-lab`/`azd-env-name` 태그 확인을 거치며, 첫 명령은 dry-run이고 두 번째 명령만 삭제를 시작한다. ```bash -monitor/sre-agent-event-lab/scripts/cleanup.sh -monitor/sre-agent-event-lab/scripts/cleanup.sh --yes +monitor/sre-agent-event-lab/scripts/cleanup.sh --legacy-delete-resource-group +monitor/sre-agent-event-lab/scripts/cleanup.sh --legacy-delete-resource-group --yes ``` 정리 후 resource group 부재와 기록된 Monitoring Contributor assignment 제거를 별도로 확인한다. diff --git a/monitor/sre-agent-event-lab/azure.yaml b/monitor/sre-agent-event-lab/azure.yaml index 254977f..1442191 100644 --- a/monitor/sre-agent-event-lab/azure.yaml +++ b/monitor/sre-agent-event-lab/azure.yaml @@ -19,3 +19,8 @@ hooks: run: ./scripts/cleanup-external.sh --yes interactive: true continueOnError: false + postdown: + shell: sh + run: ./scripts/cleanup-external.sh --reset-image-env --yes + interactive: true + continueOnError: false diff --git a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh index 5ac0b80..9155db7 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh @@ -1,109 +1,253 @@ #!/usr/bin/env bash -# Removes only the lab resources that live outside the azd-owned resource -# group, so `azd down` can delete everything else itself, and clears the -# azd environment values that `azd-postprovision.sh` set for this run. +# Teardown hook for `azd down`. Two modes, one per hook: # -# Today the external resources are the subscription-scoped Monitoring -# Contributor assignments recorded by the Azure SRE Agent setup. Nothing -# else is ever deleted here: no resource groups, no resources, no -# unrecorded role assignments. When the evidence file is missing the lab -# never configured the Agent, so the hook reports that and succeeds -- -# `azd down` must not fail because an optional step was skipped. +# predown (default) Remove the lab resources that live *outside* +# the azd-owned resource group, so azd can +# delete everything else itself. +# postdown --reset-image-env Clear the azd environment values +# `azd-postprovision.sh` recorded, once the +# resources they point at are really gone. # -# `azd down` may delete the resource group (and the ACR inside it) that -# `azd-postprovision.sh` recorded in SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG. If -# those azd environment values survive, reusing the same environment would -# make a later `azd provision` try to redeploy an immutable image tag that -# no longer exists instead of falling back to the placeholder image. So -# this hook always clears both values -- independent of whether the Agent -# was ever configured -- before doing anything else. +# The only external resources are the subscription-scoped Monitoring +# Contributor assignments the Azure SRE Agent setup recorded in +# `evidence/agent-setup.json`. Nothing else is ever deleted here: no +# resource groups, no resources, no unrecorded role assignment. When the +# evidence file is missing the lab never configured the Agent, so the hook +# reports that and succeeds -- `azd down` must not fail because an optional +# step was skipped. +# +# Before deleting anything, every recorded record has to survive four +# checks, because the evidence file is a plain JSON file an operator can +# edit: +# +# 1. the assignment ID names a role assignment in the subscription this +# run resolved (`AZURE_SUBSCRIPTION_ID` > the current azd environment); +# 2. the Azure CLI is signed in to exactly that subscription; +# 3. the record carries the Agent principal the assignment was created +# for; +# 4. the live assignment really holds that principal, the Monitoring +# Contributor role definition, and subscription scope. +# +# A record that fails any of them stops the hook before *any* deletion. A +# recorded assignment that is empty or already gone is a safe no-op, so +# re-running `azd down` works. +# +# Why the image values are cleared in `postdown` and not here: `predown` +# runs before azd asks the operator to confirm the deletion. An operator who +# answers "no" keeps every resource, so clearing SRE_CONTAINER_IMAGE / +# SRE_IMAGE_TAG at that point would break an environment nothing happened +# to. `postdown` is the documented counterpart hook (azd command hooks: +# pre/post for restore, provision, package, deploy, publish, up and down), +# and azd runs a post hook only after the action itself succeeded -- +# `HooksRunner.Invoke` returns early when the action fails +# (cli/azd/pkg/ext/hooks_runner.go), so a cancelled or failed `azd down` +# leaves the recorded image values alone. set -euo pipefail -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" -readonly SCRIPT_DIR -LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" -readonly LAB_ROOT -EVIDENCE_ROOT="${SRE_LAB_EVIDENCE_ROOT:-${LAB_ROOT}/evidence}" -readonly EVIDENCE_ROOT -readonly AGENT_SETUP_FILE="${EVIDENCE_ROOT}/agent-setup.json" +CLEANUP_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +# shellcheck source=./common.sh +source "${CLEANUP_SCRIPT_DIR}/common.sh" + +# `SRE_LAB_EVIDENCE_ROOT` lets a test point the hook at a scratch evidence +# directory; every real run reads the lab's own `evidence/`. +readonly CLEANUP_EVIDENCE_ROOT="${SRE_LAB_EVIDENCE_ROOT:-${EVIDENCE_ROOT}}" +readonly CLEANUP_SETUP_FILE="${CLEANUP_EVIDENCE_ROOT}/agent-setup.json" +readonly MONITORING_CONTRIBUTOR_ROLE_ID="749f88d5-cbae-40b8-bcfc-e573ddc772fa" + +usage() { + cat <<'USAGE' +Usage: cleanup-external.sh [--reset-image-env] [--yes] + + (default) Remove the recorded subscription-scoped Monitoring + Contributor assignments that live outside the azd + resource group. Run by `azd down` as its predown hook. + --reset-image-env Clear the azd environment values azd-postprovision.sh + recorded (SRE_CONTAINER_IMAGE, SRE_IMAGE_TAG) instead. + Run by `azd down` as its postdown hook. + --yes Execute. Without it, both modes only print their plan. +USAGE +} +MODE="roles" CONFIRMED=0 -case "${1:-}" in - "") ;; - --yes) CONFIRMED=1 ;; - *) - echo "Usage: $0 [--yes]" >&2 - exit 2 - ;; -esac - -command -v azd >/dev/null 2>&1 || { - echo "Required command not found: azd" >&2 - exit 1 +while [[ "$#" -gt 0 ]]; do + case "$1" in + --yes) CONFIRMED=1 ;; + --reset-image-env) MODE="image-env" ;; + -h | --help) + usage + exit 0 + ;; + *) + usage >&2 + exit 2 + ;; + esac + shift +done +readonly MODE CONFIRMED + +require_command() { + command -v "$1" >/dev/null 2>&1 || { + echo "Required command not found: $1" >&2 + exit 1 + } +} + +lowercase() { + printf '%s' "$1" | tr '[:upper:]' '[:lower:]' } -if [[ "${CONFIRMED}" -eq 1 ]]; then - azd env set SRE_CONTAINER_IMAGE "" - azd env set SRE_IMAGE_TAG "" +if [[ "${MODE}" == "image-env" ]]; then + require_command azd + if [[ "${CONFIRMED}" -ne 1 ]]; then + echo "Planned azd environment reset:" + echo " Clear hook-set SRE_CONTAINER_IMAGE and SRE_IMAGE_TAG." + echo "Dry run only. Re-run with --yes to execute." + exit 0 + fi + # `--cwd` pins the write to this lab's azd project, so running the hook + # by hand from the repository root does not resolve another project. + azd env set SRE_CONTAINER_IMAGE "" --cwd "${LAB_ROOT}" + azd env set SRE_IMAGE_TAG "" --cwd "${LAB_ROOT}" echo "Cleared hook-set SRE_CONTAINER_IMAGE and SRE_IMAGE_TAG." -else - echo "Dry run: would clear hook-set SRE_CONTAINER_IMAGE and SRE_IMAGE_TAG." + exit 0 fi -if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then - echo "No Azure SRE Agent setup evidence at ${AGENT_SETUP_FILE}." +if [[ ! -f "${CLEANUP_SETUP_FILE}" ]]; then + echo "No Azure SRE Agent setup evidence at ${CLEANUP_SETUP_FILE}." echo "Nothing outside the azd resource group to clean up." exit 0 fi -for command_name in az jq; do - command -v "${command_name}" >/dev/null 2>&1 || { - echo "Required command not found: ${command_name}" >&2 - exit 1 - } -done +require_command az +require_command jq -: "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set to clean up recorded role assignments}" -readonly SUBSCRIPTION_SCOPE="/subscriptions/${AZURE_SUBSCRIPTION_ID}" +SUBSCRIPTION_ID="$(require_setting AZURE_SUBSCRIPTION_ID "${AZURE_SUBSCRIPTION_ID:-}")" +readonly SUBSCRIPTION_ID +readonly SUBSCRIPTION_SCOPE="/subscriptions/${SUBSCRIPTION_ID}" -lowercase() { - printf '%s' "$1" | tr '[:upper:]' '[:lower:]' -} +# The Azure CLI's active subscription is whatever the operator last +# selected, so it is read (the one deliberately unpinned call) and compared +# before anything is verified or deleted. A signed-out CLI is reported as +# itself instead of as a raw CLI error. +if ! ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv 2>/dev/null)"; then + echo "Azure CLI is not signed in, so recorded role assignments cannot be removed." >&2 + echo "Run: az login" >&2 + exit 1 +fi +readonly ACTIVE_SUBSCRIPTION_ID +if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${SUBSCRIPTION_ID}" ]]; then + echo "Refusing to continue in subscription ${ACTIVE_SUBSCRIPTION_ID}." >&2 + echo "Expected ${SUBSCRIPTION_ID}." >&2 + echo "Run: az account set --subscription ${SUBSCRIPTION_ID}" >&2 + exit 1 +fi -# Guards against a hand-edited evidence file pointing cleanup at a role -# assignment in another subscription or at a different resource type. -validate_recorded_assignment() { - local assignment_id="$1" - local assignment_id_lower expected_prefix +# Only these two keys are ever read, each with the Agent principal it was +# created for. Any other content of the evidence file is ignored. +if ! RECORDED_ASSIGNMENTS="$(jq -r ' + [ + { + assignment: (.monitoring_contributor_assignment_id // ""), + principal: (.agent_principal_id // "") + }, + { + assignment: (.uami_monitoring_contributor_assignment_id // ""), + principal: (.agent_user_assigned_principal_id // "") + } + ] + | .[] + | [.assignment, .principal] + | @tsv +' "${CLEANUP_SETUP_FILE}" 2>/dev/null)"; then + echo "Agent setup evidence is not valid JSON: ${CLEANUP_SETUP_FILE}" >&2 + echo "Recreate it: lab.sh acknowledge agent-setup" >&2 + exit 1 +fi +readonly RECORDED_ASSIGNMENTS - assignment_id_lower="$(lowercase "${assignment_id}")" +# verify_recorded_assignment ID PRINCIPAL -- 0 when the live assignment is +# the recorded Agent one, 2 when it is already gone, 1 when the record +# cannot be trusted. +verify_recorded_assignment() { + local assignment_id="$1" + local expected_principal_id="$2" + local expected_prefix expected_prefix="$(lowercase "${SUBSCRIPTION_SCOPE}")/providers/microsoft.authorization/roleassignments/" + if [[ "$(lowercase "${assignment_id}")" != "${expected_prefix}"* ]]; then + echo "Recorded role assignment does not belong to current subscription ${SUBSCRIPTION_ID}: ${assignment_id}" >&2 + return 1 + fi + + local assignment_json + # A role assignment that no longer exists answers, verbatim (recorded + # from azure-cli against a live subscription on 2026-08-14): + # ERROR: Not Found({"error":{"code":"RoleAssignmentNotFound", ...}}) + # which is an expected state during teardown, not a failure. + if ! assignment_json="$(az rest --method get --url "https://management.azure.com${assignment_id}?api-version=2022-04-01" --subscription "${SUBSCRIPTION_ID}" 2>&1)"; then + case "${assignment_json}" in + *RoleAssignmentNotFound* | *RoleAssignmentDoesNotExist* | *ResourceNotFound*) + echo "Recorded role assignment is already absent: ${assignment_id}" + return 2 + ;; + esac + echo "Unable to verify recorded role assignment: ${assignment_id}" >&2 + echo "${assignment_json}" >&2 + return 1 + fi - if [[ "${assignment_id_lower}" != "${expected_prefix}"* ]]; then - echo "Refusing role assignment outside ${SUBSCRIPTION_SCOPE}: ${assignment_id}" >&2 + local actual_principal_id actual_role_id actual_scope + actual_principal_id="$(jq -r '.properties.principalId // empty' <<<"${assignment_json}" 2>/dev/null || true)" + actual_role_id="$(jq -r '.properties.roleDefinitionId // empty' <<<"${assignment_json}" 2>/dev/null || true)" + actual_scope="$(jq -r '.properties.scope // empty' <<<"${assignment_json}" 2>/dev/null || true)" + local expected_role_id="${SUBSCRIPTION_SCOPE}/providers/Microsoft.Authorization/roleDefinitions/${MONITORING_CONTRIBUTOR_ROLE_ID}" + + if [[ "$(lowercase "${actual_principal_id}")" != "$(lowercase "${expected_principal_id}")" ]]; then + echo "Refusing role assignment held by another principal: ${assignment_id}" >&2 + echo "Recorded principal ${expected_principal_id}, assigned principal ${actual_principal_id:-unknown}." >&2 + return 1 + fi + if [[ "$(lowercase "${actual_role_id}")" != "$(lowercase "${expected_role_id}")" ]]; then + echo "Refusing role assignment of another role definition: ${assignment_id}" >&2 + return 1 + fi + if [[ "$(lowercase "${actual_scope}")" != "$(lowercase "${SUBSCRIPTION_SCOPE}")" ]]; then + echo "Refusing role assignment scoped to ${actual_scope:-unknown}, not ${SUBSCRIPTION_SCOPE}: ${assignment_id}" >&2 return 1 fi } -RECORDED_ASSIGNMENT_IDS="" -while IFS= read -r assignment_id; do - [[ -n "${assignment_id}" ]] || continue - validate_recorded_assignment "${assignment_id}" - case "${RECORDED_ASSIGNMENT_IDS}" in +# Every record is verified before the first deletion, so a single untrusted +# record leaves the whole subscription untouched. Held as a newline-joined +# string: Bash 3.2 (macOS) aborts under `set -u` when an empty array is +# expanded. +VERIFIED_ASSIGNMENT_IDS="" +while IFS=$'\t' read -r assignment_id expected_principal_id; do + if [[ -z "${assignment_id}" ]]; then + continue + fi + if [[ -z "${expected_principal_id}" ]]; then + echo "Incomplete Agent setup evidence: ${assignment_id} was recorded without its Agent principal ID." >&2 + echo "Recreate it: lab.sh acknowledge agent-setup" >&2 + exit 1 + fi + case "${VERIFIED_ASSIGNMENT_IDS}" in *"${assignment_id}"$'\n'*) continue ;; esac - RECORDED_ASSIGNMENT_IDS="${RECORDED_ASSIGNMENT_IDS}${assignment_id}"$'\n' -done < <(jq -r ' - [ - .monitoring_contributor_assignment_id, - .uami_monitoring_contributor_assignment_id - ] - | map(select(. != null and . != "")) - | .[] -' "${AGENT_SETUP_FILE}") -if [[ -z "${RECORDED_ASSIGNMENT_IDS}" ]]; then - echo "Agent setup evidence records no subscription role assignment." + verification_status=0 + verify_recorded_assignment "${assignment_id}" "${expected_principal_id}" || verification_status="$?" + case "${verification_status}" in + 0) VERIFIED_ASSIGNMENT_IDS="${VERIFIED_ASSIGNMENT_IDS}${assignment_id}"$'\n' ;; + 2) ;; + *) exit 1 ;; + esac +done <<<"${RECORDED_ASSIGNMENTS}" +readonly VERIFIED_ASSIGNMENT_IDS + +if [[ -z "${VERIFIED_ASSIGNMENT_IDS}" ]]; then + echo "Agent setup evidence records no subscription role assignment to remove." exit 0 fi @@ -111,21 +255,28 @@ echo "Planned external cleanup in ${SUBSCRIPTION_SCOPE}:" while IFS= read -r assignment_id; do [[ -n "${assignment_id}" ]] || continue echo " Remove recorded role assignment: ${assignment_id}" -done <<<"${RECORDED_ASSIGNMENT_IDS}" +done <<<"${VERIFIED_ASSIGNMENT_IDS}" if [[ "${CONFIRMED}" -ne 1 ]]; then echo "Dry run only. Re-run with --yes to execute." exit 0 fi +# A role assignment left behind is the one outcome this hook exists to +# prevent, so a failed deletion stops `azd down` before it destroys the +# resource group -- nothing is lost, and the run can be repeated. +DELETION_FAILED=0 while IFS= read -r assignment_id; do [[ -n "${assignment_id}" ]] || continue - if ! az role assignment delete \ - --ids "${assignment_id}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --output none; then - echo "Could not remove ${assignment_id}; remove it manually." >&2 + if ! az role assignment delete --ids "${assignment_id}" --subscription "${SUBSCRIPTION_ID}" --output none; then + echo "Could not remove recorded role assignment: ${assignment_id}" >&2 + DELETION_FAILED=1 fi -done <<<"${RECORDED_ASSIGNMENT_IDS}" +done <<<"${VERIFIED_ASSIGNMENT_IDS}" + +if [[ "${DELETION_FAILED}" -ne 0 ]]; then + echo "External cleanup incomplete; remove the assignment above and re-run." >&2 + exit 1 +fi echo "External cleanup complete." diff --git a/monitor/sre-agent-event-lab/scripts/cleanup.sh b/monitor/sre-agent-event-lab/scripts/cleanup.sh index bcec11e..187e612 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup.sh @@ -1,15 +1,58 @@ #!/usr/bin/env bash +# Compatibility wrapper for the documented cleanup command. +# +# The lab is provisioned with azd, so `azd down --purge` is what tears it +# down: azd deletes the resource group it created, and its predown/postdown +# hooks run `cleanup-external.sh` for the two things azd cannot see (the +# recorded subscription-scoped role assignments, and the image values the +# postprovision hook stored). This script therefore only forwards to +# `cleanup-external.sh` -- it never deletes a resource group of its own, +# because a broad deletion here would also take resources azd did not +# create and knows nothing about. +# +# `--legacy-delete-resource-group` keeps the pre-azd recovery path +# available: a lab whose azd environment was lost still has to be +# deletable by hand. It runs the same subscription and tag checks the old +# script ran, so it can only ever delete a resource group tagged +# `purpose=sre-agent-event-lab` for the current azd environment. set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" source "${SCRIPT_DIR}/common.sh" CONFIRMED=0 -if [[ "${1:-}" == "--yes" ]]; then - CONFIRMED=1 -elif [[ "$#" -gt 0 ]]; then - echo "Usage: $0 [--yes]" >&2 - exit 2 +LEGACY_RESOURCE_GROUP=0 +EXTERNAL_ARGS=() +while [[ "$#" -gt 0 ]]; do + case "$1" in + --yes) + CONFIRMED=1 + EXTERNAL_ARGS+=("--yes") + ;; + --legacy-delete-resource-group) LEGACY_RESOURCE_GROUP=1 ;; + *) + echo "Usage: $0 [--yes] [--legacy-delete-resource-group]" >&2 + exit 2 + ;; + esac + shift +done +readonly CONFIRMED LEGACY_RESOURCE_GROUP + +echo "Use 'azd down --purge' for complete cleanup." + +external_cleanup() { + # Bash 3.2 aborts on "${ARRAY[@]}" for an empty array under `set -u`. + if (( ${#EXTERNAL_ARGS[@]} > 0 )); then + "${SCRIPT_DIR}/cleanup-external.sh" "${EXTERNAL_ARGS[@]}" + else + "${SCRIPT_DIR}/cleanup-external.sh" + fi +} + +if [[ "${LEGACY_RESOURCE_GROUP}" -ne 1 ]]; then + external_cleanup + exit 0 fi require_lab_config @@ -17,110 +60,14 @@ verify_subscription if ! resource_group_exists; then echo "Resource group ${RESOURCE_GROUP} is already absent." + external_cleanup exit 0 fi verify_lab_resource_group -readonly MONITORING_CONTRIBUTOR_ROLE_ID="749f88d5-cbae-40b8-bcfc-e573ddc772fa" -readonly SUBSCRIPTION_SCOPE="/subscriptions/${SUBSCRIPTION_ID}" -ROLE_ASSIGNMENT_IDS=() - -validate_recorded_assignment() { - local assignment_id="$1" - local expected_principal_id="$2" - local assignment_json - local assignment_id_lower subscription_scope_lower - assignment_id_lower="$(printf '%s' "${assignment_id}" | tr '[:upper:]' '[:lower:]')" - subscription_scope_lower="$(printf '%s' "${SUBSCRIPTION_SCOPE}" | tr '[:upper:]' '[:lower:]')" - - if [[ "${assignment_id_lower}" != "${subscription_scope_lower}/providers/microsoft.authorization/roleassignments/"* ]]; then - echo "Refusing non-subscription role assignment: ${assignment_id}" >&2 - return 1 - fi - - if ! assignment_json="$(az rest --method get \ - --url "https://management.azure.com${assignment_id}?api-version=2022-04-01" \ - 2>&1)"; then - if [[ "${assignment_json}" == *"RoleAssignmentDoesNotExist"* \ - || "${assignment_json}" == *"ResourceNotFound"* ]]; then - echo "Recorded role assignment is already absent: ${assignment_id}" - return 2 - fi - echo "Unable to verify recorded role assignment: ${assignment_id}" >&2 - echo "${assignment_json}" >&2 - return 1 - fi - - local actual_principal_id actual_role_id actual_scope - actual_principal_id="$(jq -r '.properties.principalId // empty' <<<"${assignment_json}")" - actual_role_id="$(jq -r '.properties.roleDefinitionId // empty' <<<"${assignment_json}")" - actual_scope="$(jq -r '.properties.scope // empty' <<<"${assignment_json}")" - local expected_role_id="${SUBSCRIPTION_SCOPE}/providers/Microsoft.Authorization/roleDefinitions/${MONITORING_CONTRIBUTOR_ROLE_ID}" - local actual_principal_lower expected_principal_lower actual_role_lower - local expected_role_lower actual_scope_lower - actual_principal_lower="$(printf '%s' "${actual_principal_id}" | tr '[:upper:]' '[:lower:]')" - expected_principal_lower="$(printf '%s' "${expected_principal_id}" | tr '[:upper:]' '[:lower:]')" - actual_role_lower="$(printf '%s' "${actual_role_id}" | tr '[:upper:]' '[:lower:]')" - expected_role_lower="$(printf '%s' "${expected_role_id}" | tr '[:upper:]' '[:lower:]')" - actual_scope_lower="$(printf '%s' "${actual_scope}" | tr '[:upper:]' '[:lower:]')" +external_cleanup - if [[ "${actual_principal_lower}" != "${expected_principal_lower}" \ - || "${actual_role_lower}" != "${expected_role_lower}" \ - || "${actual_scope_lower}" != "${subscription_scope_lower}" ]]; then - echo "Refusing mismatched role assignment: ${assignment_id}" >&2 - return 1 - fi -} - -if [[ ! -f "${AGENT_SETUP_FILE}" ]]; then - echo "Agent setup evidence is required for cleanup: ${AGENT_SETUP_FILE}" >&2 - exit 1 -fi - -required_setup_values=( - "$(jq -r '.monitoring_contributor_assignment_id // empty' "${AGENT_SETUP_FILE}")" - "$(jq -r '.agent_principal_id // empty' "${AGENT_SETUP_FILE}")" - "$(jq -r '.uami_monitoring_contributor_assignment_id // empty' "${AGENT_SETUP_FILE}")" - "$(jq -r '.agent_user_assigned_principal_id // empty' "${AGENT_SETUP_FILE}")" -) -for required_value in "${required_setup_values[@]}"; do - if [[ -z "${required_value}" ]]; then - echo "Incomplete Agent setup evidence; refusing cleanup." >&2 - exit 1 - fi -done - -while IFS=$'\t' read -r assignment_id expected_principal_id; do - if validate_recorded_assignment "${assignment_id}" "${expected_principal_id}"; then - ROLE_ASSIGNMENT_IDS+=("${assignment_id}") - else - validation_status="$?" - [[ "${validation_status}" -eq 2 ]] || exit "${validation_status}" - fi -done < <(jq -r ' - [ - { - assignment: .monitoring_contributor_assignment_id, - principal: .agent_principal_id - }, - { - assignment: .uami_monitoring_contributor_assignment_id, - principal: .agent_user_assigned_principal_id - } - ] - | .[] - | [.assignment, .principal] - | @tsv -' "${AGENT_SETUP_FILE}") - -echo "Planned cleanup:" -if (( ${#ROLE_ASSIGNMENT_IDS[@]} > 0 )); then - for assignment_id in "${ROLE_ASSIGNMENT_IDS[@]}"; do - echo " Remove recorded role assignment: ${assignment_id}" - done -else - echo " No recorded subscription role assignment found." -fi +echo "Planned legacy cleanup:" echo " Delete tagged resource group: ${RESOURCE_GROUP}" if [[ "${CONFIRMED}" -ne 1 ]]; then @@ -128,12 +75,6 @@ if [[ "${CONFIRMED}" -ne 1 ]]; then exit 0 fi -if (( ${#ROLE_ASSIGNMENT_IDS[@]} > 0 )); then - for assignment_id in "${ROLE_ASSIGNMENT_IDS[@]}"; do - az role assignment delete --ids "${assignment_id}" - done -fi - az group delete \ --name "${RESOURCE_GROUP}" \ --yes \ diff --git a/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py b/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py index 9492dab..72d1283 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py +++ b/monitor/sre-agent-event-lab/scripts/tests/azd_fake.py @@ -82,7 +82,12 @@ def azd_stub_source(azd_values, missing_key_mode="azd_1_29", log_path=None, logg "done", ] if log_path is not None: - lines.append(f'printf \'%s\\n\' "${{argv[*]:-}}" >> "{log_path}"') + # `%q` per argument, not `"${argv[*]}"`: a value azd is asked to + # clear (`azd env set KEY ""`) is an empty argument, which would + # otherwise vanish from the log and read exactly like a call that + # never happened. + lines.append(f'printf \'%q \' "${{argv[@]}}" >> "{log_path}"') + lines.append(f'printf \'\\n\' >> "{log_path}"') lines.append(f'printf \'cwd=%s\\n\' "${{project_dir}}" >> "{log_path}"') status_json = ( '{"status": "success", "expiresOn": "2026-08-14T07:57:15Z"}' diff --git a/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py b/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py new file mode 100644 index 0000000..49a878e --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py @@ -0,0 +1,257 @@ +"""Harness that runs `cleanup-external.sh` as a program. + +The script is the `predown`/`postdown` hook of `azd down`, so it is +exercised the way azd runs it: as an executable, from a working directory +that is not the lab, with fake `az`/`azd` binaries on PATH and nothing but +PATH/HOME inherited from the developer's shell. That is the only way to +prove what reading the text cannot -- that it deletes exactly the recorded +role assignments it verified, and nothing else. + +The fake `az` answers the three calls the script makes: + +* `az account show --query id -o tsv` -- the active subscription, or the + signed-out failure the real CLI produces. +* `az rest --method get --url .../roleAssignments/?...` -- the stored + assignment document, or, when no such assignment was staged, the failure + the real CLI produces (recorded from azure-cli against a live + subscription on 2026-08-14): + + ``` + rc=1, stderr='ERROR: Not Found({"error":{"code":"RoleAssignmentNotFound", + "message":"The role assignment \'...\' is not found."}})' + ``` + + A document stored as `@error:` fails with that text instead, which + is how an unreadable assignment (no permission, throttling) is staged. +* `az role assignment delete ...` -- success, or a staged failure. +""" +import json +import os +import shutil +import subprocess +from pathlib import Path + +from azd_fake import write_azd_stub, write_executable + + +SCRIPTS_DIR = Path(__file__).parents[1] +LAB_ROOT = Path(__file__).parents[2] +CLEANUP_EXTERNAL = SCRIPTS_DIR / "cleanup-external.sh" +BASH = shutil.which("bash") or "/bin/bash" + +SUBSCRIPTION_ID = "11111111-2222-3333-4444-555555555555" +OTHER_SUBSCRIPTION_ID = "99999999-9999-9999-9999-999999999999" +MONITORING_CONTRIBUTOR_ROLE_ID = "749f88d5-cbae-40b8-bcfc-e573ddc772fa" +READER_ROLE_ID = "acdd72a7-3385-48ef-bd42-f606fba81ae7" + +AGENT_PRINCIPAL_ID = "aaaaaaaa-0000-4000-8000-aaaaaaaaaaaa" +AGENT_UAMI_PRINCIPAL_ID = "bbbbbbbb-0000-4000-8000-bbbbbbbbbbbb" +AGENT_ASSIGNMENT_NAME = "cccccccc-1111-2222-3333-cccccccccccc" +UAMI_ASSIGNMENT_NAME = "dddddddd-1111-2222-3333-dddddddddddd" + +# Not staged in the fake tenant: reads back as an assignment that no longer +# exists, which is the "someone already deleted it" case. +ABSENT_ASSIGNMENT_NAME = "eeeeeeee-1111-2222-3333-eeeeeeeeeeee" + + +def assignment_id(name, subscription_id=SUBSCRIPTION_ID): + return ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + f"/roleAssignments/{name}" + ) + + +def assignment_document( + principal_id, + subscription_id=SUBSCRIPTION_ID, + role_definition_id=None, + scope=None, +): + """The ARM document `az rest` returns for one role assignment.""" + if role_definition_id is None: + role_definition_id = ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + f"/roleDefinitions/{MONITORING_CONTRIBUTOR_ROLE_ID}" + ) + return { + "properties": { + "principalId": principal_id, + "roleDefinitionId": role_definition_id, + "scope": scope if scope is not None else f"/subscriptions/{subscription_id}", + } + } + + +def agent_setup( + monitoring_assignment_id=None, + agent_principal_id=AGENT_PRINCIPAL_ID, + uami_assignment_id=None, + uami_principal_id=AGENT_UAMI_PRINCIPAL_ID, +): + """The evidence file `lab.sh acknowledge agent-setup` leaves behind.""" + if monitoring_assignment_id is None: + monitoring_assignment_id = assignment_id(AGENT_ASSIGNMENT_NAME) + if uami_assignment_id is None: + uami_assignment_id = assignment_id(UAMI_ASSIGNMENT_NAME) + return { + "agent_endpoint": "https://sre-agent.example.com/api/incidents", + "monitoring_contributor_assignment_id": monitoring_assignment_id, + "agent_principal_id": agent_principal_id, + "uami_monitoring_contributor_assignment_id": uami_assignment_id, + "agent_user_assigned_principal_id": uami_principal_id, + } + + +def staged_assignments(): + """The two assignments a healthy lab recorded, as the tenant holds them.""" + return { + AGENT_ASSIGNMENT_NAME: assignment_document(AGENT_PRINCIPAL_ID), + UAMI_ASSIGNMENT_NAME: assignment_document(AGENT_UAMI_PRINCIPAL_ID), + } + + +def _az_stub_source(log_path, state_dir): + return f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +state="{state_dir}" +case "${{1:-}} ${{2:-}}" in + "account show") + if [[ -f "${{state}}/signed_out" ]]; then + echo "ERROR: Please run 'az login' to setup account." >&2 + exit 1 + fi + cat "${{state}}/active_subscription" + ;; + "rest --method") + url="" + while [[ "$#" -gt 0 ]]; do + if [[ "$1" == "--url" ]]; then + url="$2" + shift 2 + continue + fi + shift + done + name="${{url%%\\?*}}" + name="${{name##*/}}" + document="${{state}}/assignments/${{name}}.json" + if [[ ! -f "${{document}}" ]]; then + printf 'ERROR: Not Found({{"error":{{"code":"RoleAssignmentNotFound","message":"The role assignment %s is not found."}}}})\\n' "${{name}}" >&2 + exit 1 + fi + if [[ "$(head -c 7 "${{document}}")" == "@error:" ]]; then + tail -c +8 "${{document}}" >&2 + exit 1 + fi + cat "${{document}}" + ;; + "role assignment") + if [[ -f "${{state}}/delete_fails" ]]; then + echo "ERROR: AuthorizationFailed" >&2 + exit 1 + fi + ;; +esac +exit 0 +""" + + +class CleanupRun: + def __init__(self, result, az_log, azd_log): + self.result = result + self._az_log = az_log + self._azd_log = azd_log + + @property + def returncode(self): + return self.result.returncode + + @property + def stdout(self): + return self.result.stdout + + @property + def stderr(self): + return self.result.stderr + + @property + def az_calls(self): + return self._az_log.read_text() if self._az_log.exists() else "" + + @property + def azd_calls(self): + return self._azd_log.read_text() if self._azd_log.exists() else "" + + +def run_cleanup( + tmp_path, + args=(), + evidence=None, + raw_evidence=None, + assignments=None, + subscription_id=SUBSCRIPTION_ID, + active_subscription_id=None, + azd_values=None, + signed_out=False, + delete_fails=False, + env=None, + script=CLEANUP_EXTERNAL, +): + """Run `cleanup-external.sh` against a staged fake subscription.""" + tmp_path.mkdir(parents=True, exist_ok=True) + bin_dir = tmp_path / "bin" + bin_dir.mkdir(parents=True, exist_ok=True) + state_dir = tmp_path / "state" + (state_dir / "assignments").mkdir(parents=True, exist_ok=True) + (state_dir / "active_subscription").write_text( + f"{active_subscription_id or subscription_id}\n" + ) + if signed_out: + (state_dir / "signed_out").write_text("1\n") + if delete_fails: + (state_dir / "delete_fails").write_text("1\n") + for name, document in ( + staged_assignments() if assignments is None else assignments + ).items(): + path = state_dir / "assignments" / f"{name}.json" + path.write_text( + document if isinstance(document, str) else json.dumps(document) + ) + + az_log = tmp_path / "az-calls.log" + azd_log = tmp_path / "azd-calls.log" + write_executable(bin_dir / "az", _az_stub_source(az_log, state_dir)) + write_azd_stub( + bin_dir, + {"AZURE_SUBSCRIPTION_ID": subscription_id} + if azd_values is None + else azd_values, + "azd_1_29", + azd_log, + ) + + evidence_root = tmp_path / "evidence" + evidence_root.mkdir(parents=True, exist_ok=True) + if raw_evidence is not None: + (evidence_root / "agent-setup.json").write_text(raw_evidence) + elif evidence is not None: + (evidence_root / "agent-setup.json").write_text(json.dumps(evidence)) + + workdir = tmp_path / "elsewhere" + workdir.mkdir(parents=True, exist_ok=True) + + process_env = { + "PATH": f"{bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", + "HOME": os.environ.get("HOME", str(tmp_path)), + "SRE_LAB_EVIDENCE_ROOT": str(evidence_root), + } + process_env.update(env or {}) + + result = subprocess.run( + [BASH, str(script), *args], + capture_output=True, + text=True, + env=process_env, + cwd=str(workdir), + ) + return CleanupRun(result, az_log, azd_log) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py index e0b50c4..0b2017a 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py @@ -7,10 +7,8 @@ `predown` hook has to survive a lab that never configured the Agent. """ -import json import os import re -import shutil import stat import subprocess from pathlib import Path @@ -22,7 +20,6 @@ LAB_ROOT = Path(__file__).parents[2] AZD_CONFIGURE = SCRIPTS_DIR / "azd-configure.sh" AZD_POSTPROVISION = SCRIPTS_DIR / "azd-postprovision.sh" -CLEANUP_EXTERNAL = SCRIPTS_DIR / "cleanup-external.sh" SUBSCRIPTION_PIN = '--subscription "${AZURE_SUBSCRIPTION_ID}"' # The one deliberately unpinned call: it reads whichever account is active # so the hook can report a mismatch. @@ -59,28 +56,6 @@ def _write_az_stub(directory, log_path): return stub -def _write_azd_stub(directory, log_path): - """A fake `azd` on PATH that records its arguments. - - Without this, `command -v azd` on a developer machine finds the real - `azd` binary, which would then try to mutate an environment that does - not exist in the test's tmp_path and fail for unrelated reasons. - - Uses `%q` (not `$*`) so an empty-string argument -- e.g. clearing an - azd environment value with `azd env set KEY ""` -- is visible in the - log as `''` instead of silently vanishing. - """ - stub = directory / "azd" - stub.write_text( - "#!/usr/bin/env bash\n" - f'printf "%q " "$@" >> "{log_path}"\n' - f'printf "\\n" >> "{log_path}"\n' - "exit 0\n" - ) - stub.chmod(stub.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) - return stub - - def _write_login_failing_az_stub(directory, log_path): """A fake `az` that fails only the login-check probe, like a signed-out Azure CLI, while logging every invocation it receives. @@ -99,37 +74,6 @@ def _write_login_failing_az_stub(directory, log_path): return stub -def _run_cleanup_external(tmp_path, args, evidence=None, environment=None): - bin_dir = tmp_path / "bin" - bin_dir.mkdir(exist_ok=True) - log_path = tmp_path / "az-calls.log" - azd_log_path = tmp_path / "azd-calls.log" - _write_az_stub(bin_dir, log_path) - _write_azd_stub(bin_dir, azd_log_path) - - evidence_root = tmp_path / "evidence" - evidence_root.mkdir(exist_ok=True) - if evidence is not None: - (evidence_root / "agent-setup.json").write_text(json.dumps(evidence)) - - env = dict(os.environ) - env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" - env["SRE_LAB_EVIDENCE_ROOT"] = str(evidence_root) - env.setdefault("AZURE_SUBSCRIPTION_ID", "11111111-2222-3333-4444-555555555555") - if environment: - env.update(environment) - - result = subprocess.run( - [str(CLEANUP_EXTERNAL), *args], - capture_output=True, - text=True, - env=env, - ) - calls = log_path.read_text() if log_path.exists() else "" - azd_calls = azd_log_path.read_text() if azd_log_path.exists() else "" - return result, calls, azd_calls - - def _run_hook_script(script_path, tmp_path, az_stub_factory, environment=None): """Execute an azd hook script with a controllable fake `az` on PATH.""" bin_dir = tmp_path / "bin" @@ -216,180 +160,6 @@ def test_azd_postprovision_moves_ingress_to_the_app_port_and_records_the_image() assert ingress_at < healthz_at -def test_cleanup_external_exists_and_is_executable(): - assert CLEANUP_EXTERNAL.is_file() - assert os.access(CLEANUP_EXTERNAL, os.X_OK) - - -def test_cleanup_external_never_deletes_broad_scopes(): - text = CLEANUP_EXTERNAL.read_text() - - assert "az group delete" not in text - assert "az resource delete" not in text - assert "--all" not in text - assert "az role assignment delete" in text - - -def test_cleanup_external_succeeds_when_agent_evidence_is_absent(tmp_path): - """`azd down` runs this hook with continueOnError: false, so a lab that - never configured the SRE Agent must still tear down cleanly. - """ - result, calls, _ = _run_cleanup_external(tmp_path, ["--yes"]) - - assert result.returncode == 0, result.stderr - assert calls == "", f"nothing external exists, but the hook called: {calls}" - - -def test_cleanup_external_deletes_only_recorded_subscription_assignments(tmp_path): - subscription_id = "11111111-2222-3333-4444-555555555555" - recorded = ( - f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" - "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" - ) - uami_recorded = ( - f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" - "/roleAssignments/ffffffff-1111-2222-3333-444444444444" - ) - result, calls, _ = _run_cleanup_external( - tmp_path, - ["--yes"], - evidence={ - "monitoring_contributor_assignment_id": recorded, - "agent_principal_id": "principal-a", - "uami_monitoring_contributor_assignment_id": uami_recorded, - "agent_user_assigned_principal_id": "principal-b", - }, - environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, - ) - - assert result.returncode == 0, result.stderr - assert f"role assignment delete --ids {recorded}" in calls - assert f"role assignment delete --ids {uami_recorded}" in calls - assert f"--subscription {subscription_id}" in calls - assert "group delete" not in calls - - -def test_cleanup_external_refuses_assignments_outside_the_target_subscription(tmp_path): - subscription_id = "11111111-2222-3333-4444-555555555555" - foreign = ( - "/subscriptions/99999999-9999-9999-9999-999999999999/providers" - "/Microsoft.Authorization/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" - ) - result, calls, _ = _run_cleanup_external( - tmp_path, - ["--yes"], - evidence={ - "monitoring_contributor_assignment_id": foreign, - "agent_principal_id": "principal-a", - "uami_monitoring_contributor_assignment_id": foreign, - "agent_user_assigned_principal_id": "principal-b", - }, - environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, - ) - - assert result.returncode != 0 - assert "role assignment delete" not in calls - - -def test_cleanup_external_is_a_dry_run_without_yes(tmp_path): - subscription_id = "11111111-2222-3333-4444-555555555555" - recorded = ( - f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" - "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" - ) - result, calls, _ = _run_cleanup_external( - tmp_path, - [], - evidence={ - "monitoring_contributor_assignment_id": recorded, - "agent_principal_id": "principal-a", - "uami_monitoring_contributor_assignment_id": recorded, - "agent_user_assigned_principal_id": "principal-b", - }, - environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, - ) - - assert result.returncode == 0, result.stderr - assert "delete" not in calls - - -def test_cleanup_external_runs_under_bash_32(tmp_path): - """macOS ships Bash 3.2, where `${ARRAY[@]}` on an empty array aborts - under `set -u`. - """ - bash_path = shutil.which("bash") or "/bin/bash" - version = subprocess.run( - [bash_path, "-c", "echo ${BASH_VERSINFO[0]}"], - capture_output=True, - text=True, - ).stdout.strip() - - result, _, _ = _run_cleanup_external(tmp_path, ["--yes"]) - - assert result.returncode == 0, ( - f"cleanup-external.sh must run on bash {version}: {result.stderr}" - ) - assert "unbound variable" not in result.stderr - - -def test_cleanup_external_clears_hook_set_image_env_vars_alongside_role_cleanup(tmp_path): - """`azd down` may delete the resource group (and its ACR) that - `azd-postprovision.sh` recorded in SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG. If - those values survive in the azd environment, reusing it later would make - `azd provision` try to redeploy an image tag that no longer exists - instead of falling back to the placeholder. `predown` always runs this - hook with --yes (see azure.yaml), so clearing here is safe. - """ - subscription_id = "11111111-2222-3333-4444-555555555555" - recorded = ( - f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" - "/roleAssignments/aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" - ) - result, calls, azd_calls = _run_cleanup_external( - tmp_path, - ["--yes"], - evidence={ - "monitoring_contributor_assignment_id": recorded, - "agent_principal_id": "principal-a", - "uami_monitoring_contributor_assignment_id": recorded, - "agent_user_assigned_principal_id": "principal-b", - }, - environment={"AZURE_SUBSCRIPTION_ID": subscription_id}, - ) - - assert result.returncode == 0, result.stderr - assert f"role assignment delete --ids {recorded}" in calls - assert "env set SRE_CONTAINER_IMAGE ''" in azd_calls - assert "env set SRE_IMAGE_TAG ''" in azd_calls - assert "group delete" not in calls - assert "resource delete" not in calls - - -def test_cleanup_external_clears_image_env_vars_even_without_agent_evidence(tmp_path): - """A lab that never configured the SRE Agent takes the early-exit path - (no evidence file); the hook-set image values must still be cleared on - that path, since it has nothing to do with the Agent. - """ - result, calls, azd_calls = _run_cleanup_external(tmp_path, ["--yes"]) - - assert result.returncode == 0, result.stderr - assert calls == "", f"nothing external exists, but the hook called: {calls}" - assert "env set SRE_CONTAINER_IMAGE ''" in azd_calls - assert "env set SRE_IMAGE_TAG ''" in azd_calls - - -def test_cleanup_external_does_not_clear_image_env_vars_during_a_dry_run(tmp_path): - """Without --yes, the hook only plans actions; it must not mutate the - azd environment either. - """ - result, _, azd_calls = _run_cleanup_external(tmp_path, []) - - assert result.returncode == 0, result.stderr - assert azd_calls == "", ( - f"a dry run (no --yes) must not mutate the azd environment: {azd_calls}" - ) - - def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): """`az account show` fails with a generic Azure CLI error when signed out. Guard it so the hook fails fast with one unambiguous message diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py b/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py new file mode 100644 index 0000000..f282a7f --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py @@ -0,0 +1,449 @@ +"""Behaviour tests for `cleanup-external.sh`, the azd teardown hook. + +`azd down` deletes the azd-owned resource group itself. The only lab +artifacts it cannot see are the subscription-scoped Monitoring Contributor +assignments the Azure SRE Agent setup recorded, so this hook removes those +-- and only those. Everything here is a behaviour test: the script runs as a +program, from a directory that is not the lab, against a fake `az` holding a +staged subscription, exactly as azd runs it. + +Two lifecycle rules are covered as well: + +* `predown` deletes the recorded external role assignments *before* azd + destroys anything, and must never touch the azd environment values -- an + operator who answers "no" at azd's delete prompt keeps a working + environment. +* `postdown` clears the image values `azd-postprovision.sh` recorded, once + the resources they point at are really gone. +""" +import os +import re +import subprocess + +from cleanup_harness import ( + ABSENT_ASSIGNMENT_NAME, + AGENT_ASSIGNMENT_NAME, + AGENT_PRINCIPAL_ID, + CLEANUP_EXTERNAL, + LAB_ROOT, + OTHER_SUBSCRIPTION_ID, + READER_ROLE_ID, + SUBSCRIPTION_ID, + UAMI_ASSIGNMENT_NAME, + agent_setup, + assignment_document, + assignment_id, + run_cleanup, + staged_assignments, +) + + +RECORDED_ASSIGNMENT_ID = assignment_id(AGENT_ASSIGNMENT_NAME) +RECORDED_UAMI_ASSIGNMENT_ID = assignment_id(UAMI_ASSIGNMENT_NAME) +# The one deliberately unpinned call: it reads whichever account is active +# so the hook can refuse to act on the wrong one. +ACTIVE_ACCOUNT_PROBE = "az account show --query id" +SUBSCRIPTION_PIN = '--subscription "${SUBSCRIPTION_ID}"' + + +def _az_invocations(script_text): + """Every `az ...` command in a script, with line continuations joined.""" + joined = re.sub(r"\\\n\s*", " ", script_text) + commands = [] + for line in joined.splitlines(): + if line.strip().startswith("#"): + continue + for segment in re.split(r"\$\(|\|\||&&|\||;|`", line): + stripped = re.sub( + r"^(?:if\s+|until\s+|while\s+|then\s+|else\s+|do\s+|!\s*)+", + "", + segment.strip(), + ) + if re.match(r"^az\s", stripped): + commands.append(re.sub(r"\s+", " ", stripped).strip()) + return commands + + +def test_cleanup_external_exists_and_is_executable(): + assert CLEANUP_EXTERNAL.is_file() + assert os.access(CLEANUP_EXTERNAL, os.X_OK) + + +def test_cleanup_never_deletes_broad_scopes(): + text = CLEANUP_EXTERNAL.read_text() + + assert "az group delete" not in text + assert "az resource delete" not in text + assert "--all" not in text + assert "az role assignment delete" in text + + +def test_cleanup_pins_every_azure_call_except_the_active_account_probe(): + """The Azure CLI's active subscription is whatever the operator last + selected; every call that reads or deletes must name the target one.""" + for command in _az_invocations(CLEANUP_EXTERNAL.read_text()): + if command.startswith(ACTIVE_ACCOUNT_PROBE): + continue + assert SUBSCRIPTION_PIN in command, ( + f"cleanup-external.sh runs an unpinned Azure CLI command: {command}" + ) + + +def test_cleanup_refuses_assignment_from_other_subscription(tmp_path): + """A hand-edited evidence file must never point cleanup at a role + assignment that lives in somebody else's subscription.""" + run = run_cleanup( + tmp_path, + ["--yes"], + evidence=agent_setup( + monitoring_assignment_id=assignment_id( + AGENT_ASSIGNMENT_NAME, OTHER_SUBSCRIPTION_ID + ) + ), + ) + + assert run.returncode != 0 + assert "does not belong to current subscription" in run.stderr + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_dry_run_never_deletes(tmp_path): + run = run_cleanup(tmp_path, [], evidence=agent_setup()) + + assert run.returncode == 0, run.stderr + assert "az role assignment delete" not in run.az_calls + assert "role assignment delete" not in run.az_calls + assert RECORDED_ASSIGNMENT_ID in run.stdout + assert "Dry run only" in run.stdout + + +def test_cleanup_deletes_both_recorded_assignments_after_verifying_them(tmp_path): + run = run_cleanup(tmp_path, ["--yes"], evidence=agent_setup()) + + assert run.returncode == 0, run.stderr + assert f"role assignment delete --ids {RECORDED_ASSIGNMENT_ID}" in run.az_calls + assert f"role assignment delete --ids {RECORDED_UAMI_ASSIGNMENT_ID}" in run.az_calls + assert f"--subscription {SUBSCRIPTION_ID}" in run.az_calls + assert "group delete" not in run.az_calls + # Each assignment is read back before it is deleted. + assert run.az_calls.index("rest --method get") < run.az_calls.index( + "role assignment delete" + ) + + +def test_cleanup_refuses_an_assignment_held_by_another_principal(tmp_path): + """The recorded ID names an assignment; only the principal proves it is + the Agent's. A recycled ID belonging to anything else is left alone.""" + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = assignment_document( + "ffffffff-0000-4000-8000-ffffffffffff" + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode != 0 + assert AGENT_ASSIGNMENT_NAME in run.stderr + assert "role assignment delete" not in run.az_calls, ( + "no assignment may be deleted once any recorded record failed validation" + ) + + +def test_cleanup_refuses_an_assignment_scoped_below_the_subscription(tmp_path): + """Only the subscription-scoped assignment lives outside the azd resource + group; anything narrower is azd's to delete, not this hook's.""" + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = assignment_document( + AGENT_PRINCIPAL_ID, + scope=f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/rg-sre-lab", + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode != 0 + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_refuses_an_assignment_of_another_role(tmp_path): + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = assignment_document( + AGENT_PRINCIPAL_ID, + role_definition_id=( + f"/subscriptions/{SUBSCRIPTION_ID}/providers/Microsoft.Authorization" + f"/roleDefinitions/{READER_ROLE_ID}" + ), + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode != 0 + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_accepts_an_assignment_that_is_already_absent(tmp_path): + """Re-running `azd down`, or an operator who removed the assignment by + hand, must not fail the teardown.""" + run = run_cleanup( + tmp_path, + ["--yes"], + evidence=agent_setup( + monitoring_assignment_id=assignment_id(ABSENT_ASSIGNMENT_NAME) + ), + ) + + assert run.returncode == 0, run.stderr + assert "already absent" in run.stdout + assert ABSENT_ASSIGNMENT_NAME not in run.az_calls.split("role assignment delete")[-1] + assert f"role assignment delete --ids {RECORDED_UAMI_ASSIGNMENT_ID}" in run.az_calls + + +def test_cleanup_deletes_a_record_repeated_under_both_keys_once(tmp_path): + """An Agent configured with a single identity records the same + assignment under both keys; it must be removed once, not twice.""" + run = run_cleanup( + tmp_path, + ["--yes"], + evidence=agent_setup( + uami_assignment_id=RECORDED_ASSIGNMENT_ID, + uami_principal_id=AGENT_PRINCIPAL_ID, + ), + ) + + assert run.returncode == 0, run.stderr + deletions = [ + line + for line in run.az_calls.splitlines() + if line.startswith("role assignment delete") + ] + assert deletions == [ + f"role assignment delete --ids {RECORDED_ASSIGNMENT_ID} " + f"--subscription {SUBSCRIPTION_ID} --output none" + ] + + +def test_cleanup_refuses_an_assignment_it_cannot_read(tmp_path): + """A read that fails for any other reason (no permission, throttling) + leaves the record unverified, and an unverified deletion is exactly + what this hook must never do.""" + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = ( + "@error:ERROR: Forbidden({\"error\":{\"code\":\"AuthorizationFailed\"}})" + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode != 0 + assert "Unable to verify recorded role assignment" in run.stderr + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_succeeds_without_agent_evidence(tmp_path): + """`azd down` runs this hook with continueOnError: false, so a lab that + never configured the Agent must still tear down cleanly -- without + needing an Azure CLI session at all.""" + run = run_cleanup(tmp_path, ["--yes"]) + + assert run.returncode == 0, run.stderr + assert run.az_calls == "", f"nothing external exists, but the hook called: {run.az_calls}" + + +def test_cleanup_succeeds_when_the_evidence_records_no_assignment(tmp_path): + run = run_cleanup( + tmp_path, + ["--yes"], + evidence=agent_setup( + monitoring_assignment_id="", + agent_principal_id="", + uami_assignment_id="", + uami_principal_id="", + ), + ) + + assert run.returncode == 0, run.stderr + assert "role assignment delete" not in run.az_calls + assert "no subscription role assignment" in run.stdout + + +def test_cleanup_refuses_an_assignment_recorded_without_its_principal(tmp_path): + """Without the principal the record cannot be verified, and an + unverifiable deletion is exactly what this hook must never do.""" + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(agent_principal_id="") + ) + + assert run.returncode != 0 + assert "Incomplete Agent setup evidence" in run.stderr + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_refuses_malformed_evidence(tmp_path): + run = run_cleanup(tmp_path, ["--yes"], raw_evidence="{not json at all") + + assert run.returncode != 0 + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_refuses_to_act_from_another_active_subscription(tmp_path): + run = run_cleanup( + tmp_path, + ["--yes"], + evidence=agent_setup(), + active_subscription_id=OTHER_SUBSCRIPTION_ID, + ) + + assert run.returncode != 0 + assert "Refusing to continue in subscription" in run.stderr + assert f"Expected {SUBSCRIPTION_ID}" in run.stderr + assert "role assignment delete" not in run.az_calls + assert "rest --method" not in run.az_calls + + +def test_cleanup_reports_a_signed_out_azure_cli(tmp_path): + run = run_cleanup(tmp_path, ["--yes"], evidence=agent_setup(), signed_out=True) + + assert run.returncode != 0 + assert "az login" in run.stderr + assert "Please run 'az login' to setup account." not in run.stderr + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_fails_closed_without_a_configured_subscription(tmp_path): + run = run_cleanup(tmp_path, ["--yes"], evidence=agent_setup(), azd_values={}) + + assert run.returncode != 0 + assert "azd env set AZURE_SUBSCRIPTION_ID" in run.stderr + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_fails_when_a_recorded_assignment_cannot_be_deleted(tmp_path): + """Leaving a subscription-scoped assignment behind is the one outcome + this hook exists to prevent, so a failed deletion stops `azd down` + before it destroys the resource group.""" + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), delete_fails=True + ) + + assert run.returncode != 0 + assert RECORDED_ASSIGNMENT_ID in run.stderr + + +def test_role_cleanup_leaves_the_azd_environment_untouched(tmp_path): + """`predown` runs before azd asks the operator to confirm the deletion. + Clearing the recorded image values there would break an environment + whose owner answered "no", so the role-cleanup mode never writes to the + azd environment.""" + run = run_cleanup(tmp_path, ["--yes"], evidence=agent_setup()) + + assert run.returncode == 0, run.stderr + assert "env set" not in run.azd_calls, ( + f"predown must not mutate the azd environment: {run.azd_calls!r}" + ) + + +def test_reset_image_env_clears_only_the_hook_set_image_values(tmp_path): + """`azd down` deletes the ACR that `azd-postprovision.sh` recorded in + SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG. Reusing the environment afterwards + would make `azd provision` redeploy an image tag that no longer exists, + so `postdown` clears both once the resources are really gone.""" + run = run_cleanup(tmp_path, ["--reset-image-env", "--yes"], evidence=agent_setup()) + + assert run.returncode == 0, run.stderr + assert "env set SRE_CONTAINER_IMAGE ''" in run.azd_calls + assert "env set SRE_IMAGE_TAG ''" in run.azd_calls + assert run.az_calls == "", ( + f"the environment reset must not touch Azure: {run.az_calls!r}" + ) + assert "role assignment delete" not in run.az_calls + + +def test_reset_image_env_writes_to_the_lab_project_from_any_directory(tmp_path): + """Run by hand from the repository root, `azd env set` would otherwise + resolve whatever project the working directory happens to hold.""" + run = run_cleanup(tmp_path, ["--reset-image-env", "--yes"]) + + assert run.returncode == 0, run.stderr + assert f"cwd={LAB_ROOT}" in run.azd_calls, ( + f"azd env writes were not pinned to the lab project: {run.azd_calls!r}" + ) + assert "no project exists" not in run.azd_calls + + +def test_reset_image_env_dry_run_does_not_mutate_the_azd_environment(tmp_path): + run = run_cleanup(tmp_path, ["--reset-image-env"]) + + assert run.returncode == 0, run.stderr + assert "env set" not in run.azd_calls, ( + f"a dry run (no --yes) must not mutate the azd environment: {run.azd_calls!r}" + ) + + +def test_cleanup_rejects_an_unknown_option(tmp_path): + run = run_cleanup(tmp_path, ["--purge-everything"], evidence=agent_setup()) + + assert run.returncode == 2 + assert "Usage" in run.stderr + assert run.az_calls == "" + + +def test_cleanup_runs_under_bash_32(tmp_path): + """macOS ships Bash 3.2, where `"${ARRAY[@]}"` on an empty array aborts + under `set -u` and none of Bash 4's builtins exist.""" + system_bash = "/bin/bash" + version = subprocess.run( + [system_bash, "-c", "echo ${BASH_VERSINFO[0]}"], + capture_output=True, + text=True, + ).stdout.strip() + + for arguments, evidence in ( + (["--yes"], None), + (["--yes"], agent_setup()), + (["--reset-image-env", "--yes"], None), + ): + run = run_cleanup( + tmp_path / f"run-{len(arguments)}-{evidence is None}", + arguments, + evidence=evidence, + ) + assert run.returncode == 0, ( + f"cleanup-external.sh must run on bash {version}: {run.stderr}" + ) + assert "unbound variable" not in run.stderr + assert "syntax error" not in run.stderr + + text = CLEANUP_EXTERNAL.read_text() + for bash_4_only in ("mapfile", "readarray", "declare -A", "${_,,}"): + assert bash_4_only not in text, ( + f"{bash_4_only} does not exist in the Bash 3.2 macOS ships" + ) + + +def test_azure_yaml_removes_roles_predown_and_resets_image_values_postdown(): + config = (LAB_ROOT / "azure.yaml").read_text() + + predown = config.split("predown:", 1)[1].split("postdown:", 1)[0] + postdown = config.split("postdown:", 1)[1] + + assert "./scripts/cleanup-external.sh --yes" in predown + assert "--reset-image-env" not in predown, ( + "predown runs before azd's delete confirmation: an operator who " + "cancels must keep the recorded image values" + ) + assert "./scripts/cleanup-external.sh --reset-image-env --yes" in postdown + + +def test_readme_documents_the_teardown_hooks_and_manual_recovery(): + readme = (LAB_ROOT / "README.md").read_text() + section = readme.split("## 정리", 1)[1].split("\n## ", 1)[0] + + assert "predown" in section + assert "postdown" in section + assert "cleanup-external.sh" in section + assert "--reset-image-env" in section diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index 78a28cb..9fc3d78 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -1,4 +1,3 @@ -import shutil import subprocess from pathlib import Path @@ -8,11 +7,14 @@ COMMON_SH = Path(__file__).parents[1] / "common.sh" DEPLOY_SH = Path(__file__).parents[1] / "deploy.sh" CLEANUP_SH = Path(__file__).parents[1] / "cleanup.sh" +CLEANUP_EXTERNAL_SH = Path(__file__).parents[1] / "cleanup-external.sh" QUERY_EVIDENCE_SH = Path(__file__).parents[1] / "query-evidence.sh" RUN_SCENARIO_SH = Path(__file__).parents[1] / "run-scenario.sh" CAPTURE_SCENARIO_SH = Path(__file__).parents[1] / "capture-scenario.sh" BASELINE_SH = Path(__file__).parents[1] / "baseline.sh" +LEGACY_RESOURCE_GROUP_FLAG = "--legacy-delete-resource-group" + REQUIRED_ENV = { "AZURE_SUBSCRIPTION_ID": "11111111-2222-3333-4444-555555555555", "AZURE_RESOURCE_GROUP": "rg-sre-lab-test", @@ -196,7 +198,14 @@ def test_scenario_waits_for_new_revision_before_load(): def test_cleanup_removes_both_subscription_monitoring_assignments(): - script = CLEANUP_SH.read_text() + """The verified deletion of the two subscription-scoped assignments now + lives in `cleanup-external.sh` -- the script `azd down`'s `predown` hook + runs and the one `cleanup.sh` forwards to -- so both recorded records, + their principals, the Monitoring Contributor role definition and the + read-back that verifies them must be there. `test_cleanup_external.py` + exercises the resulting behaviour against a staged subscription. + """ + script = CLEANUP_EXTERNAL_SH.read_text() assert "monitoring_contributor_assignment_id" in script assert "uami_monitoring_contributor_assignment_id" in script @@ -204,8 +213,13 @@ def test_cleanup_removes_both_subscription_monitoring_assignments(): assert "agent_user_assigned_principal_id" in script assert "749f88d5-cbae-40b8-bcfc-e573ddc772fa" in script assert "az rest --method get" in script - assert "Agent setup evidence is required for cleanup" in script assert "Incomplete Agent setup evidence" in script + # A lab that never configured the Agent has no evidence file at all, + # and `azd down` runs this hook with continueOnError: false -- so a + # missing file is a no-op here, not the refusal the standalone script + # used to report. + assert "Agent setup evidence is required for cleanup" not in script + assert "Nothing outside the azd resource group to clean up." in script def test_s1_and_s2_record_injection_before_container_app_update(): @@ -317,51 +331,24 @@ def test_activity_log_export_projects_only_incident_fields(): assert "claims:" not in script -def test_cleanup_deletion_loop_tolerates_empty_role_assignments_on_bash32(tmp_path): - """Regression test for the macOS Bash 3.2 empty-array bug. - - Bash 3.2 (macOS's default /bin/bash) raises "unbound variable" when - expanding "${ARRAY[@]}" for an empty array under `set -u`, even though - Bash 4+ treats it as an empty expansion. cleanup.sh must guard the - ROLE_ASSIGNMENT_IDS deletion loop so a lab run with zero recorded role - assignments still proceeds to delete the resource group instead of - crashing. +def test_cleanup_delegates_to_external_cleanup_and_keeps_recovery_deletion_behind_a_flag(): + """`cleanup.sh` used to delete the whole resource group itself, which is + now `azd down`'s job. It stays as a compatibility wrapper: it names the + supported command, forwards to `cleanup-external.sh`, and only deletes a + resource group when an operator explicitly asks for the documented + recovery path. `test_lab_scripts.py` runs both paths as programs. """ - bash_path = shutil.which("bash") or "/bin/bash" - script = CLEANUP_SH.read_text() - dry_run_marker = ( - 'if [[ "${CONFIRMED}" -ne 1 ]]; then\n' - ' echo "Dry run only. Re-run with --yes to execute."\n' - " exit 0\n" - "fi\n" - ) - assert dry_run_marker in script - deletion_tail = script.split(dry_run_marker, 1)[1] - - call_log = tmp_path / "az-calls.log" - harness = f""" -set -euo pipefail -RESOURCE_GROUP="rg-test-empty-assignments" -ROLE_ASSIGNMENT_IDS=() -az() {{ - echo "az $*" >> "{call_log}" -}} -{deletion_tail} -""" - result = subprocess.run( - [bash_path, "-c", harness], - capture_output=True, - text=True, - ) + readme = (Path(__file__).parents[2] / "README.md").read_text() - assert result.returncode == 0, ( - "cleanup.sh's deletion loop must not crash on Bash 3.2 when " - f"ROLE_ASSIGNMENT_IDS is empty. stderr:\n{result.stderr}" + assert "azd down --purge" in script + assert "cleanup-external.sh" in script + assert LEGACY_RESOURCE_GROUP_FLAG in script + assert LEGACY_RESOURCE_GROUP_FLAG in readme, ( + "the legacy resource-group deletion must be documented for recovery" ) - calls = call_log.read_text() if call_log.exists() else "" - assert "az role assignment delete" not in calls - assert "az group delete" in calls + # Nothing may delete a resource group before the legacy flag is parsed. + assert script.index(LEGACY_RESOURCE_GROUP_FLAG) < script.index("az group delete") def test_scenario_query_capture_cleanup_scripts_are_exercised_as_programs(): diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index 57be95f..a00c323 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -349,33 +349,68 @@ def test_capture_scenario_renders_from_another_directory(tmp_path): def test_cleanup_dry_run_plans_without_deleting_from_another_directory(tmp_path): + """`cleanup.sh` is a compatibility wrapper around the external cleanup + `azd down` runs: it plans the recorded role-assignment removal and never + proposes a resource-group deletion of its own.""" lab_run = make_lab(tmp_path) lab_run.write_agent_setup() result = lab_run.run("cleanup.sh") - _assert_loaded_config(result, lab_run) assert result.returncode == 0, result.stderr - assert "Planned cleanup:" in result.stdout - assert f"Delete tagged resource group: {RESOURCE_GROUP}" in result.stdout + assert "azd down --purge" in result.stdout + assert "Planned external cleanup" in result.stdout + assert f"Delete tagged resource group: {RESOURCE_GROUP}" not in result.stdout assert "Dry run only" in result.stdout az_calls = lab_run.az_calls() assert "group delete" not in az_calls, "a dry run must delete nothing" assert "role assignment delete" not in az_calls -def test_cleanup_deletes_only_after_confirmation(tmp_path): +def test_cleanup_deletes_only_the_recorded_external_assignments(tmp_path): + """Even with --yes, the wrapper must not delete a resource group: that + is `azd down`'s job, and a broad deletion here would take resources azd + never created with it.""" lab_run = make_lab(tmp_path) lab_run.write_agent_setup() result = lab_run.run("cleanup.sh", ["--yes"]) + assert result.returncode == 0, result.stderr + az_calls = lab_run.az_calls() + assert "role assignment delete --ids /subscriptions/" in az_calls + assert "group delete" not in az_calls + + +def test_cleanup_legacy_flag_deletes_the_tagged_resource_group(tmp_path): + """The pre-azd resource groups still have to be recoverable by hand, so + the broad deletion stays available behind an explicit flag -- after the + same tag and subscription checks it always ran.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run( + "cleanup.sh", ["--legacy-delete-resource-group", "--yes"] + ) + + _assert_loaded_config(result, lab_run) assert result.returncode == 0, result.stderr az_calls = lab_run.az_calls() assert f"group delete --name {RESOURCE_GROUP} --yes --no-wait" in az_calls assert "role assignment delete --ids /subscriptions/" in az_calls +def test_cleanup_legacy_dry_run_plans_the_resource_group_deletion(tmp_path): + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + + result = lab_run.run("cleanup.sh", ["--legacy-delete-resource-group"]) + + assert result.returncode == 0, result.stderr + assert f"Delete tagged resource group: {RESOURCE_GROUP}" in result.stdout + assert "group delete" not in lab_run.az_calls() + + @pytest.mark.parametrize("script_name", CALLERS) def test_every_caller_fails_closed_when_configuration_is_missing(script_name, tmp_path): """No azd value and no explicit environment: every entry point must stop @@ -436,13 +471,14 @@ def test_every_caller_refuses_a_foreign_subscription(script_name, tmp_path): def test_environment_name_tag_mismatch_stops_every_caller(tmp_path): """A resource group tagged for another azd environment is refused even - when its purpose tag matches.""" + when its purpose tag matches -- including on the legacy recovery path, + the only one that still deletes a resource group.""" lab_run = make_lab(tmp_path) lab_run.write_agent_setup() result = lab_run.run( "cleanup.sh", - ["--yes"], + ["--legacy-delete-resource-group", "--yes"], env={"AZURE_ENV_NAME": f"{ENV_NAME}-other"}, ) From f80be214377eba9f5a780e60b2780158e2d8c6cd Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 18:37:44 +0900 Subject: [PATCH 11/26] fix(sre-lab): split az rest streams, exact dedupe, honest predown docs - cleanup-external.sh: capture az rest's stdout and stderr into separate files instead of merging them with 2>&1. A successful read that also prints an azure-cli warning to stderr no longer corrupts the ARM JSON, and a failing read's RoleAssignmentNotFound check now inspects stderr alone instead of a stream that could contain stray stdout content. Added --only-show-errors to ask azure-cli itself to drop most warnings first. - Fixed the 'already verified' dedupe check: it matched any recorded ID that was merely a suffix of an already-verified one (case pattern *"${id}"$'\n'* has no left boundary), silently skipping validation of a second, distinct record. Now bounded by a leading newline as well, requiring a full \n\n match, still a plain string (Bash 3.2 safe). - README: documented that predown runs before azd's own delete confirmation prompt, so canceling that prompt does not restore already -removed role assignments; an operator must re-create the assignment and re-run 'lab.sh acknowledge agent-setup'. No longer implies canceling keeps a fully working environment. - test_cleanup_runs_under_bash_32 now pins /bin/bash explicitly via run_cleanup(..., bash=system_bash) instead of relying on whatever 'bash' resolves to on PATH. - Added behaviour tests for all of the above plus fake az/az-fake harness support for staged stderr warnings and stdout noise (cleanup_harness.py Staged, lab_script_harness.py dispatch fix for --only-show-errors). Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 4 +- .../scripts/cleanup-external.sh | 57 +++++-- .../scripts/tests/cleanup_harness.py | 60 +++++++- .../scripts/tests/lab_script_harness.py | 11 +- .../scripts/tests/test_cleanup_external.py | 141 +++++++++++++++++- .../scripts/tests/test_common.py | 2 +- 6 files changed, 252 insertions(+), 23 deletions(-) diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 718265f..7c7a46e 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -393,7 +393,9 @@ Dynamic rule은 3일·30 samples 전에는 발화하지 않으며 3주 전에는 `azd`로 배포한 환경은 `azd down`으로 정리한다. resource group 삭제는 `azd`가 수행하고, azd가 볼 수 없는 두 가지만 hook이 처리한다. - predown hook `scripts/cleanup-external.sh --yes`: resource group 밖에 기록된 구독 범위 Monitoring Contributor assignment만 제거한다. 기록된 assignment ID가 현재 구독에 속하는지, Azure CLI가 그 구독에 로그인했는지, 실제 assignment의 principal·role·scope가 `evidence/agent-setup.json`의 기록과 일치하는지 확인한 뒤에만 삭제한다. 하나라도 어긋나면 아무것도 삭제하지 않고 중단한다. 기록이 없거나 이미 삭제된 assignment는 안전한 no-op다. -- postdown hook `scripts/cleanup-external.sh --reset-image-env --yes`: `azd-postprovision.sh`가 기록한 `SRE_CONTAINER_IMAGE`와 `SRE_IMAGE_TAG`를 비운다. predown이 아니라 postdown인 이유는 predown이 azd의 삭제 확인 프롬프트보다 먼저 실행되기 때문이다. 삭제를 취소한 환경의 image 값을 미리 지우면 안 된다. +- postdown hook `scripts/cleanup-external.sh --reset-image-env --yes`: `azd-postprovision.sh`가 기록한 `SRE_CONTAINER_IMAGE`와 `SRE_IMAGE_TAG`를 비운다. predown이 아니라 postdown인 이유는 postdown이 `azd down`의 리소스 삭제가 실제로 성공한 뒤에만 실행되기 때문이다 -- 삭제 자체가 취소되거나 실패한 환경의 image 값을 미리 지우면 안 된다. + +중요: predown hook은 `azd down`이 **삭제 확인 프롬프트를 띄우기 전에** 실행된다(azd의 command hook은 명령 전체를 감싸고, 그 확인 프롬프트는 명령 자체의 일부다). 즉 `azd down`을 실행하는 순간 기록된 Monitoring Contributor assignment는 이미 제거되며, 뒤이어 나오는 확인 프롬프트에서 **취소해도 이미 제거된 assignment는 되돌아오지 않는다**. resource group과 그 안의 리소스는 그대로 남지만, Agent의 구독 범위 role assignment는 사라진 상태가 된다 -- `azd down` 취소가 lab을 완전히 예전 상태로 되돌린다고 가정하면 안 된다. 취소한 뒤 lab을 계속 쓰려면 role assignment를 다시 만들고 `lab.sh acknowledge agent-setup`을 다시 실행해서 `evidence/agent-setup.json`을 새 assignment ID로 갱신해야 한다. ```bash cd monitor/sre-agent-event-lab diff --git a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh index 9155db7..19f7445 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh @@ -180,20 +180,46 @@ verify_recorded_assignment() { return 1 fi - local assignment_json - # A role assignment that no longer exists answers, verbatim (recorded - # from azure-cli against a live subscription on 2026-08-14): - # ERROR: Not Found({"error":{"code":"RoleAssignmentNotFound", ...}}) - # which is an expected state during teardown, not a failure. - if ! assignment_json="$(az rest --method get --url "https://management.azure.com${assignment_id}?api-version=2022-04-01" --subscription "${SUBSCRIPTION_ID}" 2>&1)"; then - case "${assignment_json}" in + # `az rest`'s stdout is the ARM document this function parses with jq; + # its stderr is diagnostics only -- an azure-cli warning (a preview + # notice, an extension-update nag) on a *successful* call, or the real + # error body on a failed one. A plain `2>&1` would merge the two: a + # warning on an otherwise-healthy read corrupts the JSON and makes a live + # assignment look unreadable, and a failure's real error can end up + # interleaved with unrelated output. The two streams are therefore kept + # apart with separate capture files (no array, no process substitution, + # portable to Bash 3.2) and read back into their own variables. + # `--only-show-errors` additionally asks azure-cli itself to drop most + # warnings before they are ever written. + local rest_stdout_file rest_stderr_file rest_status=0 + rest_stdout_file="$(mktemp)" + rest_stderr_file="$(mktemp)" + if ! az rest --only-show-errors --method get \ + --url "https://management.azure.com${assignment_id}?api-version=2022-04-01" \ + --subscription "${SUBSCRIPTION_ID}" \ + >"${rest_stdout_file}" 2>"${rest_stderr_file}"; then + rest_status=1 + fi + local assignment_json assignment_stderr + assignment_json="$(cat "${rest_stdout_file}")" + assignment_stderr="$(cat "${rest_stderr_file}")" + rm -f "${rest_stdout_file}" "${rest_stderr_file}" + + if [[ "${rest_status}" -ne 0 ]]; then + # A role assignment that no longer exists answers, verbatim (recorded + # from azure-cli against a live subscription on 2026-08-14): + # ERROR: Not Found({"error":{"code":"RoleAssignmentNotFound", ...}}) + # which is an expected state during teardown, not a failure. Only + # stderr is inspected for it -- never stdout, which a failing call must + # not be trusted to have left empty. + case "${assignment_stderr}" in *RoleAssignmentNotFound* | *RoleAssignmentDoesNotExist* | *ResourceNotFound*) echo "Recorded role assignment is already absent: ${assignment_id}" return 2 ;; esac echo "Unable to verify recorded role assignment: ${assignment_id}" >&2 - echo "${assignment_json}" >&2 + echo "${assignment_stderr}" >&2 return 1 fi @@ -220,9 +246,14 @@ verify_recorded_assignment() { # Every record is verified before the first deletion, so a single untrusted # record leaves the whole subscription untouched. Held as a newline-joined -# string: Bash 3.2 (macOS) aborts under `set -u` when an empty array is -# expanded. -VERIFIED_ASSIGNMENT_IDS="" +# string bounded by a leading newline as well as a trailing one, so the +# "already verified" check below can require a full `\n\n` match -- +# matching bare `\n` (no leading boundary) would also accept any +# recorded ID that merely *ends with* the same characters as an +# already-verified one, silently treating a distinct, unverified record as +# a duplicate. Bash 3.2 (macOS) aborts under `set -u` when an empty array +# is expanded, which is why this stays a string instead of an array. +VERIFIED_ASSIGNMENT_IDS=$'\n' while IFS=$'\t' read -r assignment_id expected_principal_id; do if [[ -z "${assignment_id}" ]]; then continue @@ -233,7 +264,7 @@ while IFS=$'\t' read -r assignment_id expected_principal_id; do exit 1 fi case "${VERIFIED_ASSIGNMENT_IDS}" in - *"${assignment_id}"$'\n'*) continue ;; + *$'\n'"${assignment_id}"$'\n'*) continue ;; esac verification_status=0 @@ -246,7 +277,7 @@ while IFS=$'\t' read -r assignment_id expected_principal_id; do done <<<"${RECORDED_ASSIGNMENTS}" readonly VERIFIED_ASSIGNMENT_IDS -if [[ -z "${VERIFIED_ASSIGNMENT_IDS}" ]]; then +if [[ "${VERIFIED_ASSIGNMENT_IDS}" == $'\n' ]]; then echo "Agent setup evidence records no subscription role assignment to remove." exit 0 fi diff --git a/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py b/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py index 49a878e..99f54de 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/cleanup_harness.py @@ -29,10 +29,21 @@ import os import shutil import subprocess +from collections import namedtuple from pathlib import Path from azd_fake import write_azd_stub, write_executable +# A staged `az rest` reply that also exercises the stderr/stdout split: the +# fake `az` writes `stderr_extra` to its own stderr (simulating an +# azure-cli warning -- a preview notice, an extension update nag -- that a +# real, successful call can still print) and/or `stdout_noise` to its own +# stdout (simulating stray content on a failing call, which a real `az +# rest` never produces but which a robust reader must still not be misled +# by) around the normal `document` reply. +Staged = namedtuple("Staged", ["document", "stderr_extra", "stdout_noise"]) +Staged.__new__.__defaults__ = (None, None) + SCRIPTS_DIR = Path(__file__).parents[1] LAB_ROOT = Path(__file__).parents[2] @@ -114,15 +125,28 @@ def _az_stub_source(log_path, state_dir): return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" state="{state_dir}" -case "${{1:-}} ${{2:-}}" in - "account show") +# Real azure-cli global flags (`--only-show-errors`, etc.) can appear +# anywhere before or between an azd/az command's own arguments, so the +# fake dispatches on which recognisable sub-command is present anywhere in +# "$@", not on the literal position of $1/$2. +subcommand="other" +case " $* " in + *" --method "*) subcommand="rest" ;; + *) case "${{1:-}} ${{2:-}}" in + "account show") subcommand="account-show" ;; + "role assignment") subcommand="role-assignment" ;; + esac ;; +esac + +case "${{subcommand}}" in + "account-show") if [[ -f "${{state}}/signed_out" ]]; then echo "ERROR: Please run 'az login' to setup account." >&2 exit 1 fi cat "${{state}}/active_subscription" ;; - "rest --method") + "rest") url="" while [[ "$#" -gt 0 ]]; do if [[ "$1" == "--url" ]]; then @@ -135,17 +159,28 @@ def _az_stub_source(log_path, state_dir): name="${{url%%\\?*}}" name="${{name##*/}}" document="${{state}}/assignments/${{name}}.json" + stderr_extra="${{state}}/assignments/${{name}}.stderr_extra" + stdout_noise="${{state}}/assignments/${{name}}.stdout_noise" if [[ ! -f "${{document}}" ]]; then printf 'ERROR: Not Found({{"error":{{"code":"RoleAssignmentNotFound","message":"The role assignment %s is not found."}}}})\\n' "${{name}}" >&2 exit 1 fi if [[ "$(head -c 7 "${{document}}")" == "@error:" ]]; then + if [[ -f "${{stdout_noise}}" ]]; then + cat "${{stdout_noise}}" + fi + if [[ -f "${{stderr_extra}}" ]]; then + cat "${{stderr_extra}}" >&2 + fi tail -c +8 "${{document}}" >&2 exit 1 fi + if [[ -f "${{stderr_extra}}" ]]; then + cat "${{stderr_extra}}" >&2 + fi cat "${{document}}" ;; - "role assignment") + "role-assignment") if [[ -f "${{state}}/delete_fails" ]]; then echo "ERROR: AuthorizationFailed" >&2 exit 1 @@ -196,6 +231,7 @@ def run_cleanup( delete_fails=False, env=None, script=CLEANUP_EXTERNAL, + bash=BASH, ): """Run `cleanup-external.sh` against a staged fake subscription.""" tmp_path.mkdir(parents=True, exist_ok=True) @@ -213,10 +249,24 @@ def run_cleanup( for name, document in ( staged_assignments() if assignments is None else assignments ).items(): + stderr_extra = None + stdout_noise = None + if isinstance(document, Staged): + stderr_extra = document.stderr_extra + stdout_noise = document.stdout_noise + document = document.document path = state_dir / "assignments" / f"{name}.json" path.write_text( document if isinstance(document, str) else json.dumps(document) ) + if stderr_extra is not None: + (state_dir / "assignments" / f"{name}.stderr_extra").write_text( + stderr_extra + ) + if stdout_noise is not None: + (state_dir / "assignments" / f"{name}.stdout_noise").write_text( + stdout_noise + ) az_log = tmp_path / "az-calls.log" azd_log = tmp_path / "azd-calls.log" @@ -248,7 +298,7 @@ def run_cleanup( process_env.update(env or {}) result = subprocess.run( - [BASH, str(script), *args], + [bash, str(script), *args], capture_output=True, text=True, env=process_env, diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 6f66fc0..8aba66e 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -135,7 +135,16 @@ def _az_stub_source(log_path, state_dir): printf 'Resolved\\n' > "${{state}}/alert_condition" fi }} -case "${{1:-}} ${{2:-}}" in +# Real azure-cli global flags (`--only-show-errors`, etc.) can land before +# or between an az command's own arguments, shifting whatever the +# positional `$1 $2` dispatch below expects. `az rest` is the one call +# `cleanup-external.sh` adds such a flag to, so its dispatch key is derived +# from the presence of `--method` in "$@" rather than its position. +dispatch_key="${{1:-}} ${{2:-}}" +case " $* " in + *" --method "*) dispatch_key="rest --method" ;; +esac +case "${{dispatch_key}}" in "account show") printf '%s\\n' "{SUBSCRIPTION_ID}" ;; "group exists") diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py b/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py index f282a7f..e4e3373 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py @@ -24,10 +24,12 @@ ABSENT_ASSIGNMENT_NAME, AGENT_ASSIGNMENT_NAME, AGENT_PRINCIPAL_ID, + AGENT_UAMI_PRINCIPAL_ID, CLEANUP_EXTERNAL, LAB_ROOT, OTHER_SUBSCRIPTION_ID, READER_ROLE_ID, + Staged, SUBSCRIPTION_ID, UAMI_ASSIGNMENT_NAME, agent_setup, @@ -126,7 +128,7 @@ def test_cleanup_deletes_both_recorded_assignments_after_verifying_them(tmp_path assert f"--subscription {SUBSCRIPTION_ID}" in run.az_calls assert "group delete" not in run.az_calls # Each assignment is read back before it is deleted. - assert run.az_calls.index("rest --method get") < run.az_calls.index( + assert run.az_calls.index("rest --only-show-errors --method get") < run.az_calls.index( "role assignment delete" ) @@ -226,6 +228,120 @@ def test_cleanup_deletes_a_record_repeated_under_both_keys_once(tmp_path): ] +def test_cleanup_does_not_dedupe_a_second_id_that_is_only_a_suffix_of_the_first( + tmp_path, +): + """The "already verified" check must match a whole recorded ID, not any + suffix of the newline-joined string it is held in. A UAMI-key record + that merely *ends with* the same characters as the already-verified + monitoring-key record is a different value and must still be validated + on its own -- a suffix-only "match" would silently skip validating (and + so deleting) whatever it actually names, exactly like the untrusted + records this hook exists to stop.""" + monitoring_id = RECORDED_ASSIGNMENT_ID + suffix_only_id = monitoring_id[-40:] + assert not suffix_only_id.startswith("/subscriptions/"), ( + "the crafted ID must not independently pass the subscription-prefix " + "check -- it must only ever be seen through the dedupe path" + ) + + run = run_cleanup( + tmp_path, + ["--yes"], + evidence=agent_setup( + uami_assignment_id=suffix_only_id, + uami_principal_id=AGENT_UAMI_PRINCIPAL_ID, + ), + ) + + assert run.returncode != 0, ( + "a suffix-only match let a second, distinct recorded ID go " + f"unvalidated: stdout={run.stdout!r} stderr={run.stderr!r}" + ) + assert "does not belong to current subscription" in run.stderr + assert "role assignment delete" not in run.az_calls, ( + "a single untrusted record must leave the whole subscription untouched" + ) + + +def test_cleanup_tolerates_a_cli_warning_on_a_successful_verification_read(tmp_path): + """A verified read that also emits an azure-cli warning on stderr (a + preview notice, an extension-update nag) must still parse the ARM + document from stdout and proceed -- merging the two streams (as a bare + `2>&1` would) corrupts the JSON with the warning text and makes a + legitimate, live assignment look unreadable.""" + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = Staged( + assignment_document(AGENT_PRINCIPAL_ID), + stderr_extra="WARNING: This command is in preview and under development.\n", + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode == 0, run.stderr + assert f"role assignment delete --ids {RECORDED_ASSIGNMENT_ID}" in run.az_calls + assert f"role assignment delete --ids {RECORDED_UAMI_ASSIGNMENT_ID}" in run.az_calls + + +def test_cleanup_absent_detection_reads_stderr_alone_ignoring_stray_stdout(tmp_path): + """The already-absent check inspects stderr alone. Unrelated content on + stdout during a failing call (a real `az rest` failure never produces + any, but a robust reader must not depend on that) must not stop the + 'already absent' no-op the RoleAssignmentNotFound text on stderr asks + for.""" + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = Staged( + '@error:ERROR: Not Found({"error":{"code":"RoleAssignmentNotFound",' + '"message":"The role assignment is not found."}})', + stdout_noise='{"unexpected": "stdout noise"}\n', + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode == 0, run.stderr + assert "already absent" in run.stdout + assert ( + f"role assignment delete --ids {RECORDED_UAMI_ASSIGNMENT_ID}" in run.az_calls + ) + assert AGENT_ASSIGNMENT_NAME not in run.az_calls.split("role assignment delete")[-1] + + +def test_cleanup_reports_the_real_error_unmixed_with_warning_noise(tmp_path): + """A verification read that fails for a real reason (not + RoleAssignmentNotFound) must surface that real error text -- unmixed + with any warning also present on stderr -- and must never delete.""" + assignments = staged_assignments() + assignments[AGENT_ASSIGNMENT_NAME] = Staged( + '@error:ERROR: Forbidden({"error":{"code":"AuthorizationFailed"}})', + stderr_extra="WARNING: This command is in preview and under development.\n", + ) + + run = run_cleanup( + tmp_path, ["--yes"], evidence=agent_setup(), assignments=assignments + ) + + assert run.returncode != 0 + assert "Unable to verify recorded role assignment" in run.stderr + assert "AuthorizationFailed" in run.stderr + assert "already absent" not in run.stdout + assert "role assignment delete" not in run.az_calls + + +def test_cleanup_verification_read_requests_only_show_errors(): + """`--only-show-errors` asks azure-cli itself to drop most warnings + before they are ever written, on top of the script's own stream split.""" + text = CLEANUP_EXTERNAL.read_text() + rest_calls = [ + command for command in _az_invocations(text) if command.startswith("az rest") + ] + assert rest_calls, "expected an `az rest` verification call" + assert all("--only-show-errors" in command for command in rest_calls) + + def test_cleanup_refuses_an_assignment_it_cannot_read(tmp_path): """A read that fails for any other reason (no permission, throttling) leaves the record unverified, and an unverified deletion is exactly @@ -394,7 +510,11 @@ def test_cleanup_rejects_an_unknown_option(tmp_path): def test_cleanup_runs_under_bash_32(tmp_path): """macOS ships Bash 3.2, where `"${ARRAY[@]}"` on an empty array aborts - under `set -u` and none of Bash 4's builtins exist.""" + under `set -u` and none of Bash 4's builtins exist. The script must run + under the *system* `/bin/bash` specifically -- not whatever `bash` + happens to resolve first on PATH, which on a developer machine with a + newer Bash installed (Homebrew's, for example) would silently stop + exercising Bash 3.2 at all despite this test's name and purpose.""" system_bash = "/bin/bash" version = subprocess.run( [system_bash, "-c", "echo ${BASH_VERSINFO[0]}"], @@ -411,6 +531,7 @@ def test_cleanup_runs_under_bash_32(tmp_path): tmp_path / f"run-{len(arguments)}-{evidence is None}", arguments, evidence=evidence, + bash=system_bash, ) assert run.returncode == 0, ( f"cleanup-external.sh must run on bash {version}: {run.stderr}" @@ -447,3 +568,19 @@ def test_readme_documents_the_teardown_hooks_and_manual_recovery(): assert "postdown" in section assert "cleanup-external.sh" in section assert "--reset-image-env" in section + # azd runs `predown` before its own delete confirmation prompt (the + # prompt lives inside the same action the hook wraps), so canceling + # that prompt does not undo a predown hook that already ran. The + # recorded external role assignments are gone either way; only the + # README's claim about what "cancel" leaves behind must not overstate + # it as a fully working environment. + assert "취소" in section, "the README must describe what canceling azd down leaves behind" + assert "확인" in section, "the README must name azd's delete confirmation prompt" + assert not re.search(r"완전히\s*(작동|동작)", section), ( + "the README must not claim canceling leaves a fully working environment " + "-- predown already removed the recorded roles before the prompt" + ) + assert "acknowledge agent-setup" in section, ( + "the README must name the recovery step (re-running Agent setup/role " + "assignment) an operator needs after canceling" + ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index 9fc3d78..7ab9f91 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -212,7 +212,7 @@ def test_cleanup_removes_both_subscription_monitoring_assignments(): assert "agent_principal_id" in script assert "agent_user_assigned_principal_id" in script assert "749f88d5-cbae-40b8-bcfc-e573ddc772fa" in script - assert "az rest --method get" in script + assert "az rest --only-show-errors --method get" in script assert "Incomplete Agent setup evidence" in script # A lab that never configured the Agent has no evidence file at all, # and `azd down` runs this hook with continueOnError: false -- so a From f14edd7dda732184a7c166580872c961868e3573 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 19:17:27 +0900 Subject: [PATCH 12/26] docs(sre-lab): add an azd-first guided walkthrough Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 445 ++++------------ .../official/portal-complete-setup-page.png | Bin 0 -> 92722 bytes .../portal-incident-response-plans-list.png | Bin 0 -> 81682 bytes .../portal-response-plan-autonomy-step.png | Bin 0 -> 67961 bytes .../official/portal-setup-status-bar.png | Bin 0 -> 72653 bytes .../guides/01-agent-setup.md | 133 +++++ .../guides/02-scenario-s1.md | 90 ++++ .../guides/03-scenario-s2.md | 56 ++ .../guides/04-scenario-s3.md | 60 +++ .../sre-agent-event-lab/guides/05-results.md | 84 +++ .../runbooks/incident-response.md | 7 +- .../scripts/tests/test_briefing_docs.py | 24 +- .../scripts/tests/test_lab_guides.py | 498 ++++++++++++++++++ 13 files changed, 1038 insertions(+), 359 deletions(-) create mode 100644 monitor/sre-agent-event-lab/assets/official/portal-complete-setup-page.png create mode 100644 monitor/sre-agent-event-lab/assets/official/portal-incident-response-plans-list.png create mode 100644 monitor/sre-agent-event-lab/assets/official/portal-response-plan-autonomy-step.png create mode 100644 monitor/sre-agent-event-lab/assets/official/portal-setup-status-bar.png create mode 100644 monitor/sre-agent-event-lab/guides/01-agent-setup.md create mode 100644 monitor/sre-agent-event-lab/guides/02-scenario-s1.md create mode 100644 monitor/sre-agent-event-lab/guides/03-scenario-s2.md create mode 100644 monitor/sre-agent-event-lab/guides/04-scenario-s3.md create mode 100644 monitor/sre-agent-event-lab/guides/05-results.md create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 7c7a46e..a2b03b5 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -1,81 +1,48 @@ -# Azure SRE Agent 이벤트 기반 장애 분석 실험실 +# Azure SRE Agent 이벤트 기반 장애 분석 실습 -제품 개요와 활용 방법은 [Azure SRE Agent 소개 자료](../azure-sre-agent.md)를 참고하세요. 이 문서에서는 실험 환경을 배포하고 장애를 재현하며 조사 근거를 수집하는 방법을 설명합니다. +Azure Container Apps에 장애를 세 번 주입하고, Azure Monitor 경고를 받은 Azure SRE Agent가 실제로 조사·결론까지 도달하는지 확인합니다. 제품 개요는 [Azure SRE Agent 소개](../azure-sre-agent.md)를 먼저 읽어 주세요. -Azure Container Apps에 의도적인 장애를 만들고, Azure Monitor 경고를 받은 Azure SRE Agent가 자동으로 조사하는지 실제 Azure 리소스에서 확인합니다. +> ⚠️ 이 실습은 실제 Azure 리소스를 만들고 **과금**합니다. 끝나면 반드시 [정리](#정리) 절차로 지우세요. -## 구성 +## 결과물 -| 구성 요소 | 역할 | +| 산출물 | 위치 | |---|---| -| Container App | HTTP 500, latency, Blob dependency 장애 재현 | -| Application Insights | request, exception, dependency telemetry | -| Log Analytics | workspace 기반 Application Insights 및 Container Apps 로그 | -| Azure Monitor | Sev2 scheduled-query alert 3개 | -| Azure SRE Agent | 1분 scanner, incident 조사, Review-mode 완화 제안 | -| ACR / Storage | 이미지 저장 및 실제 managed identity dependency | +| 시나리오별 Agent 조사 타임라인(PNG/GIF/Markdown) | `assets/captures/s1`, `s2`, `s3` | +| 원본 API 근거와 실행 상태 | `evidence/`(Git 제외) | +| 10점 만점 채점 결과 | `evidence/scorecard.json` | -모든 workload 자산은 `rg-sre-agent-event-lab-krc`에 생성된다. SRE Agent의 Azure Monitor scanner에 필요한 구독 범위 `Monitoring Contributor`만 예외이며, 설정 시 assignment ID를 기록하고 정리 시 제거한다. +## 비용과 안전 경계 -## 사전 조건 - -- Azure CLI 로그인 및 대상 Azure 구독 접근 권한 -- azd 로그인(`azd auth login`) — azd는 Azure CLI와 별도의 자격 증명을 사용한다 -- 구독 또는 필요한 리소스에 Contributor, 역할 할당에는 Owner/User Access Administrator -- `az`, `azd`, `jq`, `curl`, `python3` -- `az` Log Analytics extension: `az extension add --name log-analytics` (`az monitor log-analytics query` 제공) -- 브라우저에서 `https://sre.azure.com` 및 `*.azuresre.ai` 접근 -- Azure SRE Agent Korea Central 사용 권한 -- GitHub 저장소 `hellices/devguidesample` 연결 권한 - -Azure SRE Agent Korea Central이 구독에 표시되지 않으면 [공식 registration request](https://github.com/microsoft/sre-agent/issues/new?labels=registration&title=Subscription+registration+request)를 제출해야 한다. - -## 안전 경계 - -- 스크립트는 현재 azd 환경(또는 명시적 환경 변수)이 지정한 구독 ID를 `az account show`의 활성 구독과 일치하는지 검증한다. -- 기존 resource group을 재사용하지 않는다. -- resource group에 `purpose=sre-agent-event-lab`과 `azd-env-name=<현재 azd environment 이름>` 태그가 모두 일치하지 않으면 scenario와 cleanup을 거부한다. -- 한 번에 한 시나리오만 실행한다. -- `run-scenario.sh`는 종료 trap으로 장애 복구를 시도하고 복구 실패를 명시적으로 오류 처리한다. -- S3는 출력으로 기록된 Blob container scope의 단일 역할만 삭제·복구한다. -- Agent response plan은 모두 `Review` 모드로 구성한다. -- evidence에는 secret, connection string, access token을 저장하지 않는다. - -## 안내형 단일 진입점: `lab.sh` - -매 단계 azd output과 명령을 직접 조합하는 대신, 아래 단일 명령으로 환경 점검부터 채점까지 안내받을 수 있다. 각 하위 명령은 이 문서의 해당 절이 설명하는 스크립트를 그대로 호출하므로 동작은 동일하다. - -```bash -monitor/sre-agent-event-lab/scripts/lab.sh doctor # 환경 점검 (아래 참고) -monitor/sre-agent-event-lab/scripts/lab.sh baseline # Baseline 부하 및 telemetry 확인 -monitor/sre-agent-event-lab/scripts/lab.sh acknowledge agent-setup # Agent 설정 수기 확인 기록 (대화형) -monitor/sre-agent-event-lab/scripts/lab.sh run s1|s2|s3 # 시나리오 실행 (run-scenario.sh와 동일) -monitor/sre-agent-event-lab/scripts/lab.sh capture s1|s2|s3 # 해당 실행이 기록한 evidence 디렉터리를 캡처 -monitor/sre-agent-event-lab/scripts/lab.sh score # 수집한 evidence 채점 -``` - -### 실행 순서와 `evidence/state.json` +배포가 만드는 과금 대상은 다음과 같습니다. 금액은 구독·지역·사용량에 따라 달라지므로 [Azure 가격 계산기](https://azure.microsoft.com/pricing/calculator/)로 확인하세요. -명령은 위 순서대로만 진행된다. 진행 상태는 현재 azd 환경에 묶인 `monitor/sre-agent-event-lab/evidence/state.json`(Git 제외)에 원자적으로 기록되며, 다른 환경·구독·resource group에서 만든 state 파일은 거부된다. +- Container Apps 환경과 앱 1개(0.5 vCPU / 1Gi, 최소 replica 1이라 유휴 상태에도 과금됩니다) +- Container Registry Basic +- Log Analytics 작업 영역(PerGB2018, 30일 보존)과 Application Insights 수집량 +- Storage 계정(Standard_LRS) +- 1분 주기 로그 검색 경고 규칙 3개(평가 주기가 짧을수록 규칙당 단가가 올라갑니다) -- `run s1`은 `baseline`이 통과하고 `acknowledge agent-setup`이 기록된 뒤에만 시작한다. -- `run s2`/`run s3`는 직전 시나리오가 **복구**되고 **캡처**까지 끝난 뒤에만 시작한다. -- 복구는 workload가 다시 정상이고 그 실행이 발생시킨 alert가 Azure Monitor에서 `Resolved`로 확인된 뒤에만 기록된다. 둘 중 하나라도 시간 내에 확인되지 않으면 해당 실행은 실패로 기록되고 다음 시나리오는 계속 막힌다. -- 캡처는 Agent thread가 실제 결론을 낸 경우(`conclusion`)에만 성공으로 기록된다. `thread-not-created`, `investigation-missing`, `conclusion-missing`은 그대로 기록되며 다음 시나리오를 열어 주지 않는다. +Azure SRE Agent는 이 실습이 만들지 않습니다. 미리 만들어 둔 Agent를 사용하며 [별도로 과금](https://azure.microsoft.com/pricing/details/sre-agent/)됩니다. -`acknowledge agent-setup`은 대화형이다. 구성된 Agent 이름/리소스 ID, repository URL과 branch, knowledge 경로, response plan 모드, alert rule 이름(secret 아님)을 출력한 뒤 표준 입력으로 정확히 `acknowledge`를 입력해야 기록된다. 어떤 환경 변수로도 대체할 수 없다. +안전 경계는 스크립트가 강제합니다. -### `doctor` 점검 항목 +- 현재 azd 환경이 가리키는 구독과 `az account show`의 활성 구독이 다르면 실행을 거부합니다. +- 리소스 그룹에 `purpose=sre-agent-event-lab`과 `azd-env-name=<현재 environment>` 태그가 모두 없으면 거부합니다. +- 한 번에 한 시나리오만 실행하며, 이전 시나리오가 복구·캡처될 때까지 다음 시나리오를 막습니다. +- 응답 계획은 모두 `Review` 모드로 두어 Agent가 승인 없이 변경하지 못하게 합니다. +- `evidence/`에는 비밀 값, 연결 문자열, 액세스 토큰을 저장하지 않습니다. -`lab.sh doctor`는 `CHECKSTATUSDETAIL` 형식으로 한 줄에 하나씩 점검 결과를 출력한다. `STATUS`는 `PASS`, `FAIL`, `MANUAL` 중 하나이며, `FAIL`이 하나라도 있으면 종료 코드 1을 반환한다. 필수 명령, `log-analytics` CLI extension, Azure CLI 로그인, azd 인증(`azd auth login --check-status`), azd 구성, 구독/리소스 그룹, Container App 상태, `/healthz`, Application Insights telemetry, alert rule 활성화, SRE Agent 리소스(설정된 경우), Reader 역할 할당을 공식 안정 API로 검증한다. Repository connection, Knowledge source, Incident platform, Response plan은 공식 API로 확인할 수 없으므로 항상 `MANUAL`로 표시되며 portal에서 직접 확인해야 한다. - -점검 항목 중 세 가지는 실제 CLI 동작에 맞춰 해석해야 한다. +## 사전 조건 -- **`log-analytics` extension**: `az monitor log-analytics query`는 core CLI에 포함되지 않는다. 없으면 telemetry 점검이 무의미하므로 별도 `FAIL` 행으로 보고하고 `az extension add --name log-analytics`를 안내한다. -- **azd 인증**: `azd auth login --check-status`는 로그인 여부와 무관하게 항상 종료 코드 0을 반환하므로, `--output json`의 `status` 값(`success`/`unauthenticated`)으로만 판정한다. -- **Reader 역할**: `az role assignment list`는 `--include-inherited` 없이는 상위 scope(구독 등)에서 상속된 할당을 보여주지 않는다. 상속된 Reader도 이 lab을 읽는 데 충분하므로 `PASS`로 처리하되, 리소스 그룹에 직접 할당된 경우와 상속된 경우를 DETAIL에서 구분해 표시한다. +- `az`, `azd`, `jq`, `curl`, `python3` +- `az extension add --name log-analytics` (`az monitor log-analytics query` 제공) +- `az login`과 `azd auth login` — 두 CLI는 자격 증명을 따로 관리합니다. +- 구독 Contributor, 역할 할당을 위한 Owner 또는 User Access Administrator +- Azure SRE Agent를 만들 수 있는 [지원 지역](https://learn.microsoft.com/azure/sre-agent/supported-regions) 접근 권한 +- 브라우저에서 `https://sre.azure.com` 및 `*.azuresre.ai` 접근 +- Agent에 연결할 GitHub 저장소 권한 -## 로컬 검증 +로컬 검증만 먼저 해 보려면 다음을 실행합니다. ```bash cd monitor/sre-agent-event-lab/app @@ -83,349 +50,117 @@ python3 -m venv .venv .venv/bin/pip install -r requirements-dev.txt .venv/bin/python -m pytest -q -cd ../../.. -monitor/sre-agent-event-lab/app/.venv/bin/python -m pytest \ - monitor/sre-agent-event-lab/scripts/tests/test_loadgen.py -q -bash -n monitor/sre-agent-event-lab/scripts/*.sh -az bicep build --file monitor/sre-agent-event-lab/infra/main.bicep --stdout >/dev/null +cd .. +bash -n scripts/*.sh +az bicep build --file infra/main.bicep --stdout >/dev/null ``` -## Azure 배포 - -배포는 `azd`가 담당한다. `azure.yaml`의 preprovision hook이 필수 provider를 등록하므로 별도 등록 명령은 필요 없다. +## azd 환경 만들기 ```bash cd monitor/sre-agent-event-lab azd env new sre-event-lab --location koreacentral -mkdir -p evidence -azd up 2>&1 | tee evidence/deploy.log ``` -`azd env new`에 `--subscription`을 지정하지 않으면 azd가 로그인된 계정의 구독 목록에서 대화형으로 선택하도록 안내한다. 특정 구독을 고정하려면 `--subscription `를 추가한다(하드코딩된 예시 구독 ID를 그대로 복사해 사용하지 않는다). +`--subscription`을 생략하면 azd가 로그인된 계정의 구독 목록에서 고르게 합니다. 특정 구독을 고정하려면 `--subscription `를 붙이세요. 리소스 그룹 이름은 지정하지 않으면 `rg-`이 됩니다. -`azd up`은 Bicep provision → ACR cloud build → Container App image 전환 순서로 진행된다. 로컬 Docker는 필요하지 않다. 초기 provision은 public placeholder image를 port 80으로 띄우고, postprovision hook이 ingress를 8000으로 옮긴 뒤 lab image로 교체한다. `scripts/deploy.sh`는 위 `azd up`을 호출하는 호환 wrapper로 남아 있다. +`.env.example`은 스크립트가 읽는 설정 이름과 허용 기본값만 적어 둔 참고 파일입니다. 값을 바꾸려면 `azd env set `로 현재 azd 환경에 저장합니다. -`monitor/sre-agent-event-lab/.env.example`은 스크립트가 읽는 설정 값의 이름과 허용 기본값만 문서화한 비밀 정보 없는 참고 파일이다(비밀 값은 커밋하지 않는다). 각 값은 `scripts/common.sh`의 `load_lab_config`가 "명시적 환경 변수 > `azd env get-value` > 허용된 기본값" 순서로 해석하므로, 로컬에서 다르게 override하려면 `.env.example`을 복사해 값을 채운 뒤 `export $(grep -v '^#' .env | xargs)`처럼 셸 환경에 불러오거나 `azd env set `로 azd 환경에 저장한다. - -성공 조건: - -1. Bicep provision이 성공한다. -2. ACR에 `sre-event-lab:run-20260812T094446Z` 형식의 실행별 immutable image tag가 존재한다. -3. active Container App revision이 `Healthy`다. -4. `/healthz`가 HTTP 200을 반환한다. - -## Azure SRE Agent 설정 - -실측 환경은 공식 ARM/data-plane API로 생성했다. - -| 항목 | 값 | -|---|---| -| Subscription | `azd env new`에서 선택한 구독. 최초 실측값은 `validation-results.md` 참고 | -| Resource group | `rg-sre-agent-event-lab-krc` (최초 실측 실행; `azd`가 provision한 environment는 `azd env get-value AZURE_RESOURCE_GROUP`으로 확인) | -| Agent name | `sre-devguidesample-95933ae5` | -| Region | Korea Central | -| Azure resource access | 테스트 resource group, Reader | -| Repository | `hellices/devguidesample` | -| Knowledge | `runbooks/incident-response.md` | -| Model | Microsoft Foundry / Automatic | -| Action mode | Review / Low | - -### 실제 event bridge - -제품의 표준 Azure Monitor 연계는 Azure Monitor incident platform과 response plan을 통해 Agent로 직접 전달되며 Logic App bridge가 필요하지 않다. - -이 실험은 response plan 공개 API 자동 구성 제약 때문에 Azure SRE Agent의 HTTP Trigger 기능 앞에 Action Group + Logic App 인증 bridge를 둔 lab-specific 경로를 구성했다. 아래 bridge는 표준 도입의 필수 구성 요소가 아니다. - -```text -Azure Monitor scheduled-query alert - → Action Group (common alert schema) - → Logic App request trigger - → Logic App managed identity token - → Azure SRE Agent HTTP Trigger (Review) - → Agent thread / investigation -``` - -| 구성 | 값 | -|---|---| -| HTTP Trigger | `sre-lab-alerts`, Review | -| Logic App | `logic-sre-agent-alert-bridge` | -| Logic App role | SRE Agent Standard User, Agent scope | -| Action Group | `ag-sre-agent-event-lab` | -| Alert action | S1/S2/S3 모두 동일 Action Group | -| Common schema | Enabled | - -중요: 2026-08-12 실측에서 HTTP Trigger endpoint는 `https://management.azure.com/` audience token을 HTTP 401로 거부하고 `https://azuresre.dev` audience token을 수락했다. Logic App HTTP action의 managed identity audience도 `https://azuresre.dev`로 설정해야 한다. - -Agent principal, endpoint, subscription-scope assignment ID는 Git에서 제외되는 `monitor/sre-agent-event-lab/evidence/agent-setup.json`에 기록한다. - -### Azure MCP - -VS Code에서 `ms-azuretools.vscode-azure-mcp-server`와 `ms-azuretools.vscode-azure-github-copilot`을 설치하면 Resource Graph, Monitor, Policy/RBAC 등 구조화된 Azure 도구를 사용할 수 있다. extension 설치 후 VS Code window를 reload하고 Agent Mode의 tools 목록에서 Azure MCP Server를 확인한다. - -## 티켓과 이메일 운영 output - -S1 Agent conclusion에서 실제 GitHub Issue와 Outlook-compatible email draft를 생성했다. - -- 실제 ticket: [GitHub Issue #43](https://github.com/hellices/devguidesample/issues/43) -- Issue body: `assets/notifications/s1-github-issue.md` -- HTML draft: `assets/notifications/s1-incident-summary.html` -- RFC 5322 email: `assets/notifications/s1-incident-summary.eml` -- Email preview: `assets/notifications/s1-email-preview.png` - -artifact 재생성: +## 배포 ```bash -monitor/sre-agent-event-lab/app/.venv/bin/python \ - monitor/sre-agent-event-lab/scripts/generate_notifications.py \ - --timeline monitor/sre-agent-event-lab/evidence/s1-20260812T080606Z/normalized-timeline.json \ - --output-dir monitor/sre-agent-event-lab/assets/notifications \ - --report-url "monitor/sre-agent-event-lab/validation-results.md" \ - --issue-url https://github.com/hellices/devguidesample/issues/43 -``` - -이번 lab은 외부 수신자와 Outlook OAuth consent가 없으므로 email을 보내지 않고 `DRAFT`로 보존한다. Production에서는 Outlook managed connector의 Send email operation을 사용한다. - -- `To`: User-defined parameter로 on-call distribution list에 고정 -- `Subject`, `Body`: Agent-defined -- Review workflow: write operation을 `Ask` -- Autonomous workflow: `Ask`가 bypass될 수 있으므로 별도 최소권한 connector 사용 - -## Baseline - -배포 output에서 FQDN을 확인하고 정상 요청을 만든다. - -```bash -FQDN=$(azd env get-value AZURE_CONTAINER_APP_FQDN --cwd monitor/sre-agent-event-lab) - -python3 monitor/sre-agent-event-lab/scripts/loadgen.py \ - "https://${FQDN}/api/orders" \ - --requests 30 --concurrency 4 --expect-status 200 \ - --output monitor/sre-agent-event-lab/evidence/baseline-orders.json - -python3 monitor/sre-agent-event-lab/scripts/loadgen.py \ - "https://${FQDN}/api/documents" \ - --requests 10 --concurrency 2 --expect-status 200 \ - --output monitor/sre-agent-event-lab/evidence/baseline-documents.json +mkdir -p evidence +azd up 2>&1 | tee evidence/deploy.log ``` -Application Insights의 `AppRequests`와 `AppDependencies`에 현재 데이터가 들어오고 세 alert가 Resolved인지 확인한다. - -## 시나리오 실행 - -> ℹ️ **azd 환경 설정**: `run-scenario.sh`, `query-evidence.sh`, `capture-scenario.sh`, `cleanup.sh`는 -> `scripts/common.sh`의 `load_lab_config`로 배포 output을 읽는다. `load_lab_config`는 각 값을 -> "명시적 프로세스 환경 변수 > 현재 `azd env get-value` > 허용된 기본값" 순서로 해석하므로, 고정된 -> 구독/리소스 그룹 값은 스크립트 안에 없다. `azd up`으로 provision한 현재 azd 환경(`azd env select`로 -> 선택한 environment)을 대상으로 동작하며, 안전 장치로 대상 resource group에 `purpose=sre-agent-event-lab`과 -> `azd-env-name=<현재 environment 이름>` 태그가 모두 일치해야 한다. +Bicep provision → ACR 클라우드 빌드 → Container App 이미지 교체 순서로 진행되며 로컬 Docker는 필요 없습니다. 처음에는 공개 placeholder 이미지가 80 포트로 뜨고, postprovision hook이 ingress를 8000으로 옮긴 뒤 실습 이미지로 교체합니다. -각 명령은 장애 주입, 제한 부하, alert polling, 복구, timeline 저장을 수행한다. +성공 조건은 provision 성공, 활성 revision `Healthy`, `/healthz` HTTP 200 세 가지입니다. -```bash -monitor/sre-agent-event-lab/scripts/run-scenario.sh s1 -monitor/sre-agent-event-lab/scripts/run-scenario.sh s2 -monitor/sre-agent-event-lab/scripts/run-scenario.sh s3 -``` +## Azure SRE Agent 설정 -각 실행 후 다음 조건을 확인하기 전 다음 시나리오를 시작하지 않는다. +포털에서만 할 수 있는 설정이 남아 있습니다. 저장소 연결, 지식 문서 업로드, **Azure Monitor incident platform** 연결, `Review` 모드 응답 계획, 역할 할당을 [guides/01-agent-setup.md](guides/01-agent-setup.md)가 순서대로 안내합니다. -- 원래 endpoint가 정상 상태다. -- 해당 Azure Monitor alert가 Resolved다. -- SRE Agent incident thread의 첫 구조화 결론을 기록했다. -- `query-evidence.sh`로 동일 UTC 구간의 증거를 내보냈다. +기본 실습에는 Logic App bridge를 배포하지 않습니다. 제품 표준 경로는 Azure Monitor를 incident platform으로 연결하는 것이고, 예전 실측에서 쓰던 Action Group + Logic App 인증 경로는 레거시 기록으로만 남아 있습니다([validation-results.md](validation-results.md)). -증거 export: +## 점검과 승인 ```bash -monitor/sre-agent-event-lab/scripts/query-evidence.sh \ - s1 monitor/sre-agent-event-lab/evidence/s1-YYYYMMDDTHHMMSSZ \ - 2026-08-12T00:00:00Z 2026-08-12T00:30:00Z +./scripts/lab.sh doctor +./scripts/lab.sh baseline +./scripts/lab.sh acknowledge agent-setup ``` -실제 timeline의 UTC 값을 사용해야 한다. +`doctor`는 `CHECKSTATUSDETAIL` 한 줄씩 출력하고 `FAIL`이 하나라도 있으면 종료 코드 1을 반환합니다. 저장소 연결, 지식 원본, incident platform, 응답 계획은 공식 안정 API로 읽을 수 없어 항상 `MANUAL`입니다. -## SRE Agent 실제 동작 캡처 +`baseline`은 정상 부하를 넣고 Application Insights에 두 요청 종류가 모두 보일 때까지 최대 10분 기다립니다. `acknowledge agent-setup`은 대화형이며, 설정 값을 출력한 뒤 표준 입력으로 정확히 `acknowledge`를 입력해야 기록됩니다. -각 시나리오의 `timeline.json`이 생성된 뒤 다음 명령으로 Azure SRE Agent thread와 message를 API에서 수집하고 PNG/GIF/Markdown/Mermaid를 만든다. - -```bash -monitor/sre-agent-event-lab/scripts/lab.sh capture s1 -``` - -evidence 디렉터리는 해당 시나리오 실행이 `state.json`에 기록한 값에서 결정되므로 경로를 직접 입력하지 않는다. 과거 실행을 다시 렌더링할 때만 디렉터리를 명시한다. +## 시나리오 실행 ```bash -monitor/sre-agent-event-lab/scripts/capture-scenario.sh \ - s1 monitor/sre-agent-event-lab/evidence/s1-20260812T051000Z +./scripts/lab.sh run s1 +./scripts/lab.sh capture s1 +./scripts/lab.sh run s2 +./scripts/lab.sh capture s2 +./scripts/lab.sh run s3 +./scripts/lab.sh capture s3 ``` -원본 API snapshot과 normalized timeline은 Git에서 제외되는 evidence 폴더에 남는다. - -```text -monitor/sre-agent-event-lab/evidence/s1-20260812T051000Z/ - alert.json - normalized-timeline.json - thread-snapshots/ -``` +`run-scenario.sh`와 `capture-scenario.sh`는 `scripts/common.sh`의 `load_lab_config`로 "명시적 환경 변수 > 현재 `azd env get-value` > 허용된 기본값" 순서로 설정을 읽으므로, 고정된 구독이나 리소스 그룹이 스크립트 안에 없습니다. 진행 상태는 현재 azd 환경에 묶인 `evidence/state.json`에 기록되며 순서를 어기면 실행이 거부됩니다. -redaction을 통과한 시각 자료만 commit 대상이다. - -```text -monitor/sre-agent-event-lab/assets/captures/s1/ - 01-alert-fired.png - 02-thread-created.png - 03-investigating.png - 04-investigating.png - 05-investigating.png - 06-investigating.png - 07-conclusion.png - investigation.gif - timeline.mmd - timeline.md -``` +| 시나리오 | 주입하는 장애 | 안내 문서 | +|---|---|---| +| S1 | HTTP 500 응답 | [guides/02-scenario-s1.md](guides/02-scenario-s1.md) | +| S2 | 주문 API 지연 | [guides/03-scenario-s2.md](guides/03-scenario-s2.md) | +| S3 | Blob 읽기 권한 제거 | [guides/04-scenario-s3.md](guides/04-scenario-s3.md) | -GIF frame 수와 크기를 확인한다. +## 결과 확인 ```bash -monitor/sre-agent-event-lab/app/.venv/bin/python - <<'PY' -from PIL import Image - -path = "monitor/sre-agent-event-lab/assets/captures/s1/investigation.gif" -with Image.open(path) as image: - print({"frames": image.n_frames, "size": image.size}) - assert image.n_frames >= 4 - assert image.size == (1280, 720) -PY +./scripts/lab.sh score ``` -Agent가 thread를 만들지 않았거나 조사 결론을 내리지 못한 경우에도 GIF는 빈 성공 화면을 만들지 않는다. `thread-not-created`, `investigation-missing`, `conclusion-missing` frame으로 누락 상태와 마지막 polling 시각을 표시한다. +채점 기준, 사람이 채워야 하는 판정, 종합 판정 해석은 [guides/05-results.md](guides/05-results.md)에 있습니다. -### 선택: Portal UI 수동 녹화 - -UI 모양 자체를 보존해야 할 때만 API evidence와 별도로 녹화한다. - -1. `https://sre.azure.com`에서 해당 incident thread를 연다. -2. macOS에서 `Shift+Command+5` → 선택한 부분 기록을 사용한다. -3. alert card → investigation plan → evidence → conclusion 순서로 30~60초 녹화한다. -4. 계정 메뉴, token, unrelated resource는 화면에 포함하지 않는다. -5. MP4를 GIF로 변환한다. +## 정리 ```bash -ffmpeg -i sre-agent-s1.mp4 \ - -vf "fps=8,scale=1280:-1:flags=lanczos,split[s0][s1];[s0]palettegen[p];[s1][p]paletteuse" \ - monitor/sre-agent-event-lab/assets/captures/s1/portal-investigation.gif -``` - -Portal 녹화는 API evidence를 대체하지 않는다. 사실 판정은 `alert.json`, thread snapshots, KQL 결과를 기준으로 한다. - -## Static Threshold에서 Dynamic Threshold로 - -이번 실험은 같은 날 세 장애를 결정론적으로 발생시키기 위해 1분 scheduled-query와 static threshold를 사용했다. 모든 S1/S2/S3 점수와 GIF는 static rule의 실측 결과다. Azure Monitor Dynamic Threshold는 장기 운영에서 정상 패턴을 학습해 anomaly를 찾는 다음 단계이며 이번 세션에서는 **미실증**이다. - -| 기준 | Static | Dynamic | -|---|---|---| -| 목적 | known failure의 빠른 재현·hard limit 보호 | 시간대·일간·주간 baseline을 벗어난 anomaly | -| 준비 시간 | 즉시 | 최소 3일·30 samples | -| 학습 | 수동 threshold | 최근 10일 data, 3주 후 weekly seasonality | -| Log Search frequency | 1분 가능 | 1분 미지원, 5분 이상 | -| 운영 | deterministic safety rule | shadow 검증 후 adaptive alert | - -### 후보 numeric signal - -| Scenario | KQL 결과 | Dynamic 조건 | -|---|---|---| -| S1 | 5분당 5xx count 또는 error rate | upper bound 초과 | -| S2 | `/api/orders` p95 duration(ms) | upper bound 초과 | -| S3 | Blob 403 count 또는 failure rate | upper bound 초과 | - -Boolean 식인 `count() > 10`이 아니라 `summarize ErrorCount=count()`처럼 numeric series를 반환해야 한다. - -### 권장 시작값 - -- Frequency 5분, lookback 15~20분 -- Sensitivity Medium, noise가 크면 Low -- 4회 평가 중 2회 위반 -- 정상 telemetry 시작 UTC를 learning start로 지정 -- action을 연결하지 않은 shadow mode로 시작 -- 학습 gate 통과 후 기존 `ag-sre-agent-event-lab`을 연결해 같은 Logic App → Review-mode SRE Agent 경로 재사용 - -Dynamic rule은 3일·30 samples 전에는 발화하지 않으며 3주 전에는 weekly seasonality가 충분하지 않다. 최근 behavior change는 10일 baseline에 즉시 반영되지 않고 slowly evolving issue를 놓칠 수 있으므로, cold start·hard limit·보안 경계용 static rule을 함께 유지한다. - -공식 자료: [Azure Monitor alerts with dynamic thresholds](https://learn.microsoft.com/azure/azure-monitor/alerts/alerts-dynamic-thresholds) - -## 판정 - -각 시나리오는 10점 만점이다. - -| 항목 | 점수 | -|---|---:| -| 영향 범위 식별 | 2 | -| 직접 원인 식별 | 3 | -| 실제 증거 사용 | 2 | -| 안전한 최소 완화책 | 2 | -| 불확실성 표시 | 1 | - -- Pass: 8-10 -- Partial: 5-7 -- Fail: 0-4 - -종합 성공은 모든 시나리오 Partial 이상, 두 개 이상 Pass, unauthorized autonomous action 0건이다. - -### `lab.sh score` - -`lab.sh score`는 수집된 evidence만으로 위 표를 채점하고 `evidence/scorecard.json`과 `SCENARIOCRITERIONSTATUSPOINTSDETAIL` 표를 출력한다. - -판정 근거는 시나리오 evidence 디렉터리의 `conclusion-review.json`이다. 항목 ID(`impact_scope`, `direct_cause`, `actual_evidence`, `safe_minimum_mitigation`, `uncertainty`)마다 `{"met": true|false, "detail": "..."}`를 기록한다. - -```json -{ - "impact_scope": { "met": true, "detail": "thread 2번 message가 ca-sre-lab의 /api/orders만 영향으로 특정" }, - "direct_cause": { "met": false, "detail": "원인을 배포 변경으로만 서술하고 FAILURE_MODE 변경을 지목하지 못함" } -} +azd down --purge ``` -- 해당 항목의 구조화된 판정이 없으면 `MANUAL`로 표시하고 **점수를 주지 않는다.** 사람이 직접 확인해 `conclusion-review.json`에 기록해야 점수가 반영된다. -- 캡처가 결론에 도달하지 못한 시나리오(`thread-not-created` 등)는 모든 항목이 `FAIL` 0점이며, 그 사유가 DETAIL에 남는다. -- 종합 판정은 모든 시나리오가 Partial 이상이고 두 개 이상 Pass일 때만 `PASS`다. `MANUAL`이 남아 있으면 `INCOMPLETE`로, 즉 미완료로 보고한다. - -## 정리 +리소스 그룹 삭제는 azd가 하고, azd가 볼 수 없는 두 가지만 hook이 처리합니다. -`azd`로 배포한 환경은 `azd down`으로 정리한다. resource group 삭제는 `azd`가 수행하고, azd가 볼 수 없는 두 가지만 hook이 처리한다. +- predown hook `scripts/cleanup-external.sh --yes`: `evidence/agent-setup.json`에 기록된 구독 범위 Monitoring Contributor 할당만 제거합니다. 기록된 principal·역할·범위가 실제 할당과 모두 일치할 때만 삭제하고, 하나라도 어긋나면 아무것도 지우지 않습니다. +- postdown hook `scripts/cleanup-external.sh --reset-image-env --yes`: 기록된 `SRE_CONTAINER_IMAGE`와 `SRE_IMAGE_TAG`를 비웁니다. 삭제가 실제로 성공한 뒤에만 실행되어야 하므로 predown이 아니라 postdown입니다. -- predown hook `scripts/cleanup-external.sh --yes`: resource group 밖에 기록된 구독 범위 Monitoring Contributor assignment만 제거한다. 기록된 assignment ID가 현재 구독에 속하는지, Azure CLI가 그 구독에 로그인했는지, 실제 assignment의 principal·role·scope가 `evidence/agent-setup.json`의 기록과 일치하는지 확인한 뒤에만 삭제한다. 하나라도 어긋나면 아무것도 삭제하지 않고 중단한다. 기록이 없거나 이미 삭제된 assignment는 안전한 no-op다. -- postdown hook `scripts/cleanup-external.sh --reset-image-env --yes`: `azd-postprovision.sh`가 기록한 `SRE_CONTAINER_IMAGE`와 `SRE_IMAGE_TAG`를 비운다. predown이 아니라 postdown인 이유는 postdown이 `azd down`의 리소스 삭제가 실제로 성공한 뒤에만 실행되기 때문이다 -- 삭제 자체가 취소되거나 실패한 환경의 image 값을 미리 지우면 안 된다. +중요: predown hook은 `azd down`이 삭제 **확인** 프롬프트를 띄우기 **전에** 실행됩니다. 그 프롬프트에서 **취소**해도 이미 제거된 Monitoring Contributor 할당은 돌아오지 않습니다. 리소스 그룹은 남지만 Agent의 구독 범위 권한은 사라진 상태이므로, 계속 쓰려면 역할 할당을 다시 만들고 `./scripts/lab.sh acknowledge agent-setup`으로 `evidence/agent-setup.json`을 새 할당 ID로 갱신해야 합니다. -중요: predown hook은 `azd down`이 **삭제 확인 프롬프트를 띄우기 전에** 실행된다(azd의 command hook은 명령 전체를 감싸고, 그 확인 프롬프트는 명령 자체의 일부다). 즉 `azd down`을 실행하는 순간 기록된 Monitoring Contributor assignment는 이미 제거되며, 뒤이어 나오는 확인 프롬프트에서 **취소해도 이미 제거된 assignment는 되돌아오지 않는다**. resource group과 그 안의 리소스는 그대로 남지만, Agent의 구독 범위 role assignment는 사라진 상태가 된다 -- `azd down` 취소가 lab을 완전히 예전 상태로 되돌린다고 가정하면 안 된다. 취소한 뒤 lab을 계속 쓰려면 role assignment를 다시 만들고 `lab.sh acknowledge agent-setup`을 다시 실행해서 `evidence/agent-setup.json`을 새 assignment ID로 갱신해야 한다. +hook이 실패해 손으로 다시 실행할 때는 아래를 직접 호출합니다. `--yes` 없이는 계획만 출력합니다. ```bash -cd monitor/sre-agent-event-lab -azd down --purge +./scripts/cleanup-external.sh --yes +./scripts/cleanup-external.sh --reset-image-env --yes ``` -hook이 실패해 수동으로 다시 실행할 때는 아래처럼 직접 호출한다. 두 명령 모두 `--yes` 없이는 계획만 출력하며, 실행 위치와 무관하게 이 lab의 azd project(`--cwd`)를 대상으로 한다. +azd 환경을 잃어버린 실습을 정리할 때만 `./scripts/cleanup.sh --legacy-delete-resource-group`으로 예전 삭제 경로를 씁니다. 이 경로도 구독 일치와 태그 확인을 거치며, 첫 명령은 dry-run입니다. -```bash -monitor/sre-agent-event-lab/scripts/cleanup-external.sh --yes -monitor/sre-agent-event-lab/scripts/cleanup-external.sh --reset-image-env --yes -``` +## 문제 해결 -`scripts/cleanup.sh`는 기존 명령을 유지하기 위한 호환 wrapper다. 기본 동작은 `cleanup-external.sh` 위임뿐이며 resource group을 삭제하지 않는다. azd environment를 잃어버린 lab을 손으로 정리해야 할 때만 `--legacy-delete-resource-group`으로 예전 삭제 경로를 사용한다. 이 경로도 구독 일치와 `purpose=sre-agent-event-lab`/`azd-env-name` 태그 확인을 거치며, 첫 명령은 dry-run이고 두 번째 명령만 삭제를 시작한다. +먼저 `./scripts/lab.sh doctor`를 실행해 어떤 검사가 `FAIL`인지 확인하세요. 각 명령의 실패 처리와 복구 절차는 해당 단계 문서에 있습니다. -```bash -monitor/sre-agent-event-lab/scripts/cleanup.sh --legacy-delete-resource-group -monitor/sre-agent-event-lab/scripts/cleanup.sh --legacy-delete-resource-group --yes -``` +| 증상 | 확인할 곳 | +|---|---| +| 배포 후 앱이 응답하지 않음 | `azd env get-value AZURE_CONTAINER_APP_FQDN`으로 FQDN을 확인한 뒤 `/healthz` 호출 | +| Agent가 스레드를 만들지 않음 | [guides/01-agent-setup.md](guides/01-agent-setup.md)의 incident platform·응답 계획 확인 | +| 경고가 발생하지 않음 | 시나리오 문서의 "복구 확인" 절 | +| 채점이 `INCOMPLETE` | [guides/05-results.md](guides/05-results.md)의 수동 판정 절 | -정리 후 resource group 부재와 기록된 Monitoring Contributor assignment 제거를 별도로 확인한다. +정적 임계값 대신 Dynamic Threshold로 확장하는 설계는 [dynamic-thresholds.md](dynamic-thresholds.md)에 정리해 두었습니다. ## 공식 자료 -- [Azure SRE Agent 제품 소개](../azure-sre-agent.md) -- [Incident response](https://learn.microsoft.com/azure/sre-agent/incident-response) -- [Root cause analysis](https://learn.microsoft.com/azure/sre-agent/root-cause-analysis) -- [Agent reasoning](https://learn.microsoft.com/azure/sre-agent/agent-reasoning) -- [Memory and knowledge](https://learn.microsoft.com/azure/sre-agent/memory) -- [Azure Monitor alerts in Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/azure-monitor-alerts) +- [Azure SRE Agent 개요](https://learn.microsoft.com/azure/sre-agent/overview) +- [Complete setup](https://learn.microsoft.com/azure/sre-agent/complete-setup) - [Automate incident response](https://learn.microsoft.com/azure/sre-agent/automate-incidents) -- [Log Analytics and Application Insights connectors](https://learn.microsoft.com/azure/sre-agent/log-analytics-app-insights) -- [Supported regions](https://learn.microsoft.com/azure/sre-agent/supported-regions) +- [Azure Monitor alerts in Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/azure-monitor-alerts) +- [Incident response plans](https://learn.microsoft.com/azure/sre-agent/incident-response-plans) diff --git a/monitor/sre-agent-event-lab/assets/official/portal-complete-setup-page.png b/monitor/sre-agent-event-lab/assets/official/portal-complete-setup-page.png new file mode 100644 index 0000000000000000000000000000000000000000..75ceb9afcec6727af25529083888aa37c0650083 GIT binary patch literal 92722 zcmeEuWmr^g7cL?q-Q8W%osyD64I* z=l?m^_iwJv>^;w3&suk^wKoxJDst#3Bq(rjaOev1G8%AjPkz9`!M{ap^Y!i?}Rp3Enc9RXrmQ` zzm6sQ&}+TVuX2?&{jMS-Wqx*cwsHC(T5t@gqvKUrH@&}G+N7dWG@d|-EH43%6Z;l{ zj_T3B9|I?_zke;U)=&Ijj{p8j0k&GUGx`7VJOW&`Z0kdWsfPdgdcV(uorV72vmOHd z-)Q`2+Wj{g|Bc3fapQLm1pfbvqTf%?&#P-t`8ocBZ2o>+=YJ$-wSIOZ^5$PA`uj)l zqeo9ttvQiBB>xX|RjVu!XyGu}oSXE&6P5`HmRfcSV!R3A1M>a{+a=E;;QjX}|6dC1 zcjchxQ&0XTNZB^h=h4k!k|OP*Tyk#XzqtWmhNTO0r|sGUtO97^)l;nj2-cxDv1*VM z*-;CfKl$Jq^cKEyh6jm@#4<>gPnu21tK=fbaBsk`=| zPR^Jl=zXykQZxrE`BV?fh;2#0kv~5@RknVt&HlG&^rW&4O1?BX_nzL*c(uPEy%snw zYP+{xtP1E^bLkhi5Y$n`{LBJkd85pbq~~ukY3ga{KV|4)T)wz$$FKR!xhtPux=j6H zo$DB|nAY-4z6bseGp7y`RpFO9kK-4XrFxz#EeD+M03qbc1v~|upsjEzO>}N_ah5fl z6s)A^*N&U6#tU|zy?Ub8V+73`D3V=&q8{7A&X_Gx+52%e`$0QuDwu4C zwkr8oZPh99Pr~;n*p_g$+J?rc(9)p2L4DUJ@2^||Nzhm`D@orheUeSgo~THFV}0=% zl{}@ZarGyzs!+0+>P==3yMFbz6|M%sprYm1g8 zTeH9(@W>_r99Q^SGer-W;>HIWuh83GJKZ}p4Ky0`CvqsBIx|Ys*rh2h#9^r#kYsXpf~dTi`$Bmh zCO+C8bPApwJDN}@X1{NY>@(XfozKo*y2**$>Hk{=$d9T8m65fdf&Sp4Nd*#J9`VSM zWr3*76}>W9itT;F9XP;_-mvo}1Bhi=EWVItbWCw#p%;j2*hZ0`0~F!|vL&)euZH-9GLW3^Y$)d{j!ehaN~C;3Xf_3W*`FyR$QN# zzj8k$5ks0p9`|?gaYYE!l3D+|g-FFT|7zS1$)GT_EZafDfZ672?rCg%dd=ff>MsCx zPX^ULbw3W{vex2^X7~aA8iUU>iFD_8RuO0pPaZuRWx~dbK| z)ax^^emx=<(iR7WilLBSZ=m*$Y>KMCF1dD9&S3s(?i#|o|4sIzg1LsgQ1FND)giZ7 z+lw*0)Dcb&`$7-G*{Pqw!tH}OrU_)ZY6R z-GRYwbTSD#_x~A4%RLw^YrFA)Imt~hBm?iyP4WB=Zwncn8Vjc5lt(w%K46!!;R|&Y27${? zL7_HJ)wiPYw9@GY_g<4hcu%4JR{G7@KTE9-hh-rKY4@x}X22 zr^R*sGeZ#Tp^ab;L5xL*U#NhSjw_$Lg?`f%lyB6i@AUNG4y~0;(wt)$ru+W6qnLlr0?Ns#1X-pq*D6x2>Iqu5`B4!=fUYzpTUQ~6U6fSrrnNv&~ zH3=vY)BHI=AnYnQ{B=bj=y|-f9R$J?j9k(<4mr!*O_dnc1nA5ZZ35_d(mAwP8VV!0 zQ|zh?3RhiITyoaU;|H~ldMh)k@xAcRIURipwZopZSI?K7w$LUi&sY`9|HYac4j4P! zuPt#r)S3r=U@8Ik@=BY4;`hR0!qOcv$K&-qUwunHF(8DU36!|bGaFjuRrLh|ep)e& zzP7dAUsSJ?;o1EDPP+ch%#4U@$eCT!V`~QepEPa&hP2+Xu@vkWL~9F~9LWNRclQtp zyD*Z#ngNKrN?qFvTYyHHD=tOfkM-G#_e*- zDkAk!mrS_Fb2Q%c*0u5#@c4y6J=bUJU4C-!ShKj99~;WVQ~w0)W~EBRlOtnBOSJRn zr!w~M!Yha9I85F_=5i{>_$8gc=#>afCh%~|zXA>Q-X4NzN!ra|f>x5m%$)}-EfW92V;>Yeo zVS5A1aEf++`le|bttd843YKzd$U4zzqO{9WK#GlhJ8l=txFT3jDm8~RWycBB(q3!h zekb7&eycog2LaNLaFQrP=Dyjcv#i+#I0}U;+VF~qU1~oNt^_VjJt=1=(?2A30uIs^ zn&g|Xu&`}o38#Gy=TF2$_X5TkJUSqN%`f4vIuzd!lND0)m06KU8y@JuWurArR`@DRyW*yIsRp0#nZ6$&8R{iTxKrZz{$c!+j^r~ogp&d6NvF@2;IYhK2vqvL8+rUNzPDc-n?bOK4-_Zhj*Vb+bq6(5%nf=(m}tOsq(69 zjN)SM;A+>dN{@;JW65d+axN)!*gJIg5mViPuy)ZTH-GR-j`e*4&e0fq$(2a>N@Iy3U<3-3QgE zYWDPFzOTP#+bxFhgq#`Z`}7BeB9JJv8>1pb?*l-F%J>WI356D;)vtSck3nKd2};v0 z#^vAvI~H?9#gq}XazS00LF$S9wim}Pvxtr*4)eAq`l>l;0`f~i@yaU=(qm=`Ic|00 zd>z}lu@*L^0foEnX0t#aegEl56%t@7q(AX**BJa3Myz-eRf-Rw50#@r;;khsRVnnV z6R!I(cl%uD%fOjq zU#L1$K~NdqAFJ{_^1#7=k4FdDh-HN|iU+R9%;?oi{8)6^++!H15G?i*uHj;Kwt< zvqx1!JeR)u{hkbqncY?9Hjbu>TA@hW`m8!(JrMi6rB)5+QoE<>)<=1Fis0(FS&eR* z(bb-m%(kJ#IVWR2Er4aHrt~w7hG9s3>6G>kvz89#q`~VcRS;yoLa^tc&Hih$OS!In z{sfCi8cUv8VXj!UowsT0UKJ6q|AUi^ZNY9YqeCj`yzC(_p{Z8984|St#F{h>MU1|B^_&v?c^>GyAb8CGZHQd`haPn(e6! z9X3Lg9)q?ffF-4IY>C_{8vLqX7J)W&p-y}{t{)VNmGB8aNBZnc^9fGD^tP9|W*xA& z(bT6bj+li|31Uz=~hGL7Z{WtcsU~n!g=%T;!MpR1w-n`8Qs<6p9WW_C&pwuNXCfGP;+au*MXJN z`0sk%P7)EJXZY188@MZ4HN7~jPuITi$InvVj2zftQ-~!p3e8a zPz^xd(5pzk=}=jSD^*5seyDA0Xar;Wiv8jm4Ttwu&)Io?($1px>@M#S^2o@p_)XGZ z@0d&nTXQ_Zh|R+WOgY1clcxiHiukb2G&ax9-Y@5+J-z3Nb+jc#Mob@E;Tu)QwW`>$ zlgeJ|RHEg^X!Y~%*RZJpl9RAtq9tBItIpAvxhu^D7t{JNUeEBOn#?_6MsJqUzH$k9=-8m-}c`5@H-A+!>l3uTiL^Rn$aXP)P78#71KQqNU? zrkwc_ET)uIGN+OB=*-&g+tC=HHSvS+u+0=ZZJY)9M0IOTUwLQEM*u1rlIHb#N+<>h z3ODh0L3aQTBbFqUjh=80kPWrg?s~=?M6oLoXg+K5(bI!4O03h(p(a%Uzn2*!kA>E8 zN(G-gByc1Sz!w0pgckF^d6Ejze>}nZM-;q;kFou$cM`E{Ze!#<1@k{oqI-ILB}}yB zKmELcX~A=80FEubjUU7zO4bo^PqbYW>;STulbMZxT$Hn!F)dj-1A85q(`)4Pq3O@G z4x+lXM>2LHnP+bL`YiGNtA(2=?G#U#P#KgDU4CA{pY`A|tJ-*U==(y%Uk>fAHHv1cn{c&)eTW;CBpiwkK8fkYtlm zo`z5PNbMI=zdTf~*G3E(`1AO`&f8n*B;_sJJtZ)HB=3PXn2~}h-Gl#Rp(8h;c-kSvmK}AD+*_+ zRMWzb>vfmAM-tl6PQ1o;3nxfdAs+{&e(=@wS>vayBWcR#0@gD;D6VKan2Mpz?^k)a z1>8q;ma+qY_l1 zpXOCQnX~7Qqhrm|Qs+$G2%>ZMXXFLafmDty9S;{KhJx6*n|h~SSnaif7P zHp8O5>Cf13oi1<|pLn5g%xHIuK@s4!gyYXw_1d#muYYMOP7x&co+|S*OA2-?gIJsd zcuAK}OA(v4Z!9)R4hH*I@ zp1=~pP;#380UE3n;l_h*vi*L18p+}q$kLI=+_1sm4t{HTj7yqYLHsqz4H+4H_Rjt} zXr?IHYg=FjT|L80%VQmHp~a_p1iiQ7Hq$Kf(n=f3%oovCIC#SS87tt}(I{xTWF zn?`Ih%CYetUcs3UuHh}hH%VAege;bo<{^v6u#n>9fRGr<1YX4|kbB)^Q zIN=*|z8+116ML59_V+6HHPnn+sCUTCjTAKITwtj{Q}eBjlo~_jZtN33CGTF(-5@)e zbDJiS{~%%@4vdHu>n8rF?AVrM1gR6^RYJGenWD;G!9HM^>v+?i)OlU29?&G+!e-y4 z@V5S}zc(Y&+~}R*akgYzH?VTrMmgKMcQtg#B(w+NG1x@e200a_d5TgCpksSKq^NIM z^2$ixN7bk6Q%PIfd%LIm6YRPDV|3Xr^dJR$F*no23XjhzQaRv(gC(_#Xd%4+u#Fk8 zLaNlf5PkVz8Ugl-xc(K91964w5-#K3s$PH4$<_D)9s3zPj?*v^6jX zU4bNE=b*S|lLq1@-<@8{*=cCXSxUu{jwZn)S-9$0XS%Ud+)Ncvq1X8i_52DW6nD0t z4#qR1&>E513{`x=@1)sI^-JU`5c{WWd&UbNyZXWFBJlzzeZh>AxDU;wsb7LXM6vJa zK@b2)9yM4<8P*hJ51*W$ml{|5J6XkwDW>tXZ%Wymqi~tqG;#77leiI|q}lPs2**PX zlwZRxp0UpaEFuru9Y32@vkVOljT*PP^l%3gw1C@Q+J_i0W_f2AqWaVk3=VSa^mrWd z@3|E7BapzXR7Gd*ISbq7vi+<)kDVP2Cx+pOr^N`vY~({p6UD=LG}%j(>E9%juwIg$ zm&{%A83ilf`rURQ3@lSavJS;}N zfhlI742_0rWP7VDt#R`d?|IKPIz7xfK(V^`L(t?+l797$=#%Yb9(MYk{_R;HQ+qy; zU69D-5~F8_ zw+wwO8w=g15k==MfX(>0-WGB0@m?0V@oYtqD> zUYZzcTWa6)xia{Acjvo1lu&rKBI6Q}6W5Im0!y)yJ|B1U-}aw{ver5i4W{*|bq%@K z-elN5z4`l0N}!<<3}+^fpnMOwE{jCfls)`|R?{_$m|`n)QY_%kcf^V}C@4t3+Tle_ zH+X6ecx^{};z(Wjc_hXvNLI5uzpxf+lapYUgyT5J(8BSYZ`>tQz9q#aFSa;wM#CI_ znraAGd1(MGW}8jyhnkPnYB0!^rBS0yewumh(s*q@SK}pJ?nQc~5>l*wt5fAp9uX8% zN%x?Q5IndZ-QP`hw%DsYZ15~HmU-wT-QiNPf)ALpQVLjyy<-mSb$-b#Ks`8zwt%%2E^XBLhmx7TmBM(q)=# zZg(EH_nf)LCTmyZW{(p~#&msu88kc0$K@#v+X{A|CJ9E`NC&5;%JLj>=QZ`1u*}wT)>DBL za0NrZH?8fd$YgFo1KlEjwjIMHZVYPuhU3-%nqzmu^}&}`{ZK#E6aaP%Lk1~p)mo^q zt5_px;pS4zHT%m-t8770&LiPHUNcpQXh2Q)Q&r=VIwrT$iT`VrV8c_N0D=QCG&Emp zKzyI8CjW4^on)W27f82a!s1O`STH&hrY=#7yJ>O>;LFg9)emmboSo%8dc){z%5urS zFl^XgeU64{fwBgCBUv?)Sc0}h8n1|{b9@Q3;LKpgSM;-Q)DJfZ5_X-9D-P;JhoJQI zo79fP_n=v7*p>BY^#oNl>Uhe_mlypVHU1ju!B|D3;olZi05X;D!KQKTI=J#)bPP5s zncr);>BzNt&Wpsf=|&AAp&-$#x9A@+@4b`DgW*R|hgUlTskpVS_WLgInhe-^_Y934SruO*E-VZdsN`}LrH?PpQm z2B?8e{)mmaHS}yGACRh%a+#;`Gy^ST;=_7#yrtqpDl#*`^jhcJVxxxx*zmN=0C$t| z)NS$m-57xq)_fk)k{al%r5o?C-I!}MIomr<2=E{mGHLb3XL>uK=l#yhp4W(b2f3!# z8dVYwp-F;=@ADO@lSoPu>*{(kjw^2crnl5;hf>%vfunggHbzyzAGfO8XCyV|&fc>v z6Lz8JPJ|j&)$;wlV3&^U)3H^iauJ0w>ULey_ z)|{?uEnaCm*cwx3Z9M~3Y`WYl7u8`H1(LiYRi@V@R(~ z!?EzQ>)J@0TV9btH|S}7VLPgbuumoN!cilq2+dHza9?fb*4sbaSPYBm=oWFA-yc=X zO8@p-DvTeXk`He}Q&pAHonPIX`+9ZE~f@!muVuxS11-2E^ppv%-P2Q-z zLQH7iM=##ZVtv~~D=gO`=czs3ib>=c3-w)k)v8!72&<8x%oFy?EM`qM?VhTG=G}4A zbQ{HSyEnU^&;{IzNy+r2{n@1>Skb|X8F{Ulc#c08panwvO%c`13l zqo#iZC72u*U~2(-k3Y7P6BQ}jAy_p}gIayYt5=z@LCCWjTB5A9j=3MdX%J_VWxTmb zY%m8ZK5E*TNlQCOTH=YWSK=GjwpwXa*`6pitaFr)>60Oaq0uJ+DUB-} zur==R$&9sd;SG@wlFOap|li|Z21ZPX%FW*tQ-n;NeE?Xc5Qn^xre;l`YE*6V# zi+zcE1`Iuax!73O>`G;vIyqf!=bytyIYxj0ONs+mPOMm-YNGhXQy>7IN2mB64UiY+io~EVQ*bJyy&!Shm zv%b{T1aj!13<06SCTk0GM%$L zLBG2@ok`!IvUMgNlVAbfuBiAU5&~cTqTmUY;>Yp7$Ddyl;R;c&(^$fJeHFuu?n?LZ z6RV5m9Yxl|Z+((Vzk7PnbWoKJb+>|SOYUwk1?Fp5<|n*|a4BlK9>0fWPP(3^!p!XU zLSu|<`%AJl83PN%tEWtJ;tlqOHsha4=$YSxxN9bWF5?>DxvT}>dHcy?3X;@nw`zmB zXBp(|r~8Qmu3i-8BRg*uL1AGiA@5ty&NB0Q34`smeX9vFivPH1I?kuRhWxDxp5K3O zNT8u4oOj+?_vh)VughiEqHYJf)rPm{B9!juJx1pa`#X_aeM3E3)&7S*?bTltK)sz! zBk0Y7p3}%-mUXr>{okjnc4bXVM=P+!Q$H$>?zl+7nH}RPD(mZ$D&s}97gf)t4py!4 zc%Sj%_ex9K=__iTzkH+St#t|o>3DDuv#ic4HTsxu*grdElS=Pi53{0;@4@HJc&5$D zhKDj1B#fK-!$-%Y0{ubg&~Zvp%}-7;qWpWH7UA7B8aK*}qO(`BAaBj^4~?K|)HjN9 za*W7$u{rvVM#tlhAc2#lO7cc&5U|OwHL_7#bR%avb4DACeNQTr>;lnUZd7U5vF;wp zLt>Il&H3fObHG5G94m+OF=FeGm8v9R^~WKQ*G^UkGijyzwQHf7^BO6EYalZnIn()} zmI@d4j6p&|JmK^CSNq<_b(=(u%eZ{0-R5TY4|y+df(HL)DEd*C^urNAIvftNqMTE3 zD^H$=xu%9P7i5!)RwOM{d!c4eDQm#~*}$x8){TSD?r;^==y?2$y)z_s0IboP60$BQ z^+;z{BP3^OsmgYO)esu3_Gk1tfMo$oy5qmIZhlaQ}` zEU`aD&3@A+;1rxz{WZyD&3?W%bLmCB&iSmZv!|zkp;RhzoSNvz%KucvvaoVY%7Ie< z;hs5~wv89g^G4eygHiiWS_5w$*v4OK3m{O({t}-+wMtu zPam~T?sc%c1;Dcs3{l;wME*B&y+}+$JKSUZe-v?YE<&v6yB}==StD#4r~aq4m-cxj z_)nqaWr89chHME@sOzWfFAXHpYgfiqq>CmOPoiAIpy;g4+rvt#q7}uB#Rd?2?Y6-u z(;*0304%El+=^~#^;~LfQ7Z>I<6?@kMI)c$DRlJw#*AvO$#(-|5S}p}xs7hqhr%Dy zfZ;A%r{l_>x)+jw>wb(*?@Oq2>)Pv?mAR`w+Sv8(?C~rQY7Yduw3K<7Mp)Y72_Dk+ z@y0FeIVh|W@Ljv!E~zJ0usEgHp_n|~x%8YVEmlL7#$6-A!2{16AX6#CBP}X?m8_u7 z64U<4oq&@6R5iUUishy1k-_Lg+RK{|{xVljV(q1Vt1nJDWQ3Q&jvkEE0SVf-u4_Vr zedJKDVX&f{s=mkFI?#V~sG)t5O&Gx%afA5gC6l&6;}(>sDP^VxW>ro=HaQ{H2JY<* zv}cm*IdDuo#Ggjy?&27F;&GcunMd8Q+@y#{ss#CHeX`YJkVJ5Oodm^vNo+a)9k1_@ zz#>y0T3P(iykiQ4H4;aw^_cCOe%x;F*Be`&A1#fwW(X|CJ_Chr>_R=3#40b}XAiU# z)U1Q~#P94p^zYl|wNG|xb?9e@Cn`U$4BZwyLZ`2*LzoUpFE3vxsh>OK7>E@4EFYFN zhV}k~AxQ3_pYXq_=j3gKgN#;db1SO?12R3Kvh}*tUBe)qLC1ZgIir5?UWLn@=IFG? zD)HC2RD{Y)Xqx{uc+3>y?Xz2(UEv6}wPq17t+VT%$PJU1X(rMXJM~>`ZS~Z)B}Cb=;c?F zj6%9ORqxwWN4m$lpdERUi}Ot(go;_l|p8 z6sD2ud7t=knB8-8r)gV_T5oc4@{3M2IKs4IeILio#?$3A*i&(5h2xFeKB(2eL>RZ1 zg)wn2y-;dJW%$mE+tM2`7JCDzDMbFd$NcZF+Wms${;u4IBY;@ABe@}~A(`;b-m_18 zoxLjFclUW3<97V8yYx=eP&0BWXHL2>&Px!wifr}|XblX>vwwyu{xc#iiItw_KprCH zYdVVHvts>lMR}7qzed0d$FKJPD0rR}2`4>jc8(*I6$yx`fTo&#MXjB3M)yWV3hNab z{YCBS!L{h8d$Gy6la2PXS|#^szh|4N!tKmUhDzxwA-)#eIM@F^Ihl(4>uE|pakG-Y zbGw+?u1FXOtH$PZxF_Jg<9nnkweoc*fb-{b1s=9hGHHRyj(~L@gfA?M>3@<-8RI-AJxgC^GhRq}cMz6j?0@N@ zq+%9G>ez5NyLfk2T)iM1;T~ETYgf{EHuE9iW(CvxyEJ?4YFx&bfLW-0Gw~&F@zdPf z!GODCRsW-Suv|W`>!RN|`e~_d`KRJyd&Sw*x}Ku(EiE0x(>0C&ZDdz+IJKTm8jH`U zA6F~=$m8&1#(CW7q#?K*P7Brd*FceBgow;3Z1E^!f0(vJr`jz?&(s`@Ozj79r`mN6kLHn=DfEszMoKWsA%7B}pp&^-> zgY=hja9nFI2j6IuO?WK>y4wDu6KOyN%Q1sw^R8dR4-;U>hTn-dU1iR6&85~scqMuYQQh6!lWi2c z7r%_WegEh=ECQI9f~N>s-2P2JUvRWhBHT6K-|kN%zu?l|#LYZFFmRMIwGiETGHYKw zPf~4F&e;U}b6A|}n~`|kRchrQMicm`aq6!26;}JMS=VJITLf$&)9f95QsO!#1`6FT zq6nyKm$y4g=)Ee+-y=QGQE6&EjB#{Aj-}#17?z_{epe&!me!%bnBjKN`ZPYA(kAyf zI*Lg4Yp*)I)XHnmZ^V`-XJ;DGMYP|3B3l}|4Zk=^sp<)fLfRKKC@Bz{sEs%LMDf3_ixuxh)-ri{gNt5ZN9e%!We|OHP+~ifj75MyUqPppnBj8eLB?pGA z)#^V@AAi?hI7) z2^vi>{q$wzqflHf&n4X8x39H`VuAQ}CS~FqLW7sEdr^J_Fn7x;_%MVL{74>y`UTRW z&lV(~VLM_oUv!O`7cbyF5BB1+jOFr_P!e&#Wc|VmcrH}D?o0Z3{G$j}-%k#nI2BKh zqj0cOEhxBxq=hg0ZF90)&BAAtkC--H2gN+^ZZ7G*HLF1DSGQ7;#F5MU;HmoS3z`o? zID!Eu9bMvghtx6fM@LD`WbPa{Xp3EBu4wA+So;^=>nWB^E9CoBfuBzD#c!D-J#$t) zo39Pog+~z1FQ*6n3HZIssrR%>-{9B1X2Na86xrb{G^}{7&}X_^-ZpGyrXoK?qeCz< zcnvGUI_aeiTjT2)*o8Jf4>Px+LYX7J;G)uMHZ9|q7LXE}(d$s!J`c{YcvKnP&iq~a zdwpyKUk8E)?K5iD4;h;{?`S=VY>-Ho&2H(kd&0N)b31byH^lZOgRfQ@#ZwwmsG59# zN>fS&$DC&4<0wN8xne<7QKF}PGgP_1*$`K^ z{1GJZltEj~lEirl_^y~@j5u8tUyf-hFXiFBM|k0?9m}V^;v<2XMEEd97+0%*CmduEwi99}Bk<{1WtCVp0*?0Biz($tozHZy5y3cdxL)-V0}B!nS&Z1>xS`&mq}*Oi+vLvpu{QyJlmu;{!K*}} zKTf3_x$MbXba;9EhT=_dnuDT)6f!@i1-|U06{yTUFn@H3Np-9|;ay(j$y?-LW4N%X ztB%m{iv}_6&QK&)z^4={z=3b(#u=H^DUt@ZYV9u-*)>VA^!%Wc!yo$@_LE@7%knrd z@@n)!IwdZ~$0yiyL?tIE&eSLVF6%~d$Ec0q^2ND6`1zf@KpoIk)f;2?s_tpZuJirZ9xlw$FAvO9XkZEYFyXuR;oNip; zH|N^rt32eoD_Hqvbp+Q3KhyNWBx(Q5;*l7_&ZS$@nr)dw6srGnKD$Pu=BF2-E|%p= zq*?VC<^4sMQ#US624qzR96cURX;6UhgmHCpeDhz{xjA>9_EOZB{oQbbdxW-0^~l88 zH0=cgQEYO;&1$#n!De_^9CGw-nI~LPsKB+1I6Q?6?L@Xna2t_zccm1GJEAM9(Bt~| z3{oA-9DbWhOH663pM<3cP>nCzvW-plj^G-~p~=z>5mfrIK?)H^4)#^?>A};uqz1ZlqDg5yMJK(8Ej05f z=R}}F+e?uX^7dt{3q-~Yst~ouZ9{2`H|t=nCEhS!?MZe*y&9UD_zv^Ylaq3umll%4 zxkM}!A{S;!M65Pclc-?_>WT|ZW*JsVbK#%1$s%pWvXK>VBVQyN$SmQYUh*aww3n;= zX>9}^!rG<8nDo8>xF`v+N6{N5ic6)~0%X|FX+L*4Wwm-ls*}cIO3{~l?8x@z9N-SV z`|M7_%@sB^gxwq8lGsikh5%OQ^^<-i6N}|YRLSqvMW+e?x7+4_Z#APcvprJdCKr&% zK`+ms_73NruujNx0zyQ328sCi$;@5>k;K|nDRu-B9s6489~H&TO7!;Tf&Z!K=(mAOUJd#IA@Z%RvqZsXxGfX&M}ZS((RF zcHnl8RUDoKWY8P4Y(6nSm1CCeK~@vYFqEn`d}7}rYJTX7{?U|sF`o{Q8iS|uXa zn*r*ZvSsFv6uCvxgz%$knNdjU(Fj*mr9F`(C_PkM#s~BfqzOzsfvOEmx2RwOb+ z0N>G059Ymiz27hqunF_al}%D*>L(!$+iY>K3|hr?W?&zxabxRK+RI6gVT0u#MUYcP zBYoJ8_frD5EBTsXh@>s^^Kv7bh(f~&RMSroIAS8*m)GBB;>*cAzvFA2i!nE&sT1u; zCs*zWg@vgD*Yj&<2J@h{NCMtjZutnTX@g^=KyW6o6pV8&pdPS8x<~5H(Vz;Y5ZF;; zqqYvujFQ7&{DOjqVw4>XHFd7Jk$A_oJ>h%N(QQs@ya%@t z`#Lo9K;DRr_l(%((H=rwH{V41ytXjSu`-QrmI*6qaYCYo)+IA7a4Bh*BW(+{Vwzq4 zNand;kC+Q=%&Y%u2PY%I>zW0ltQS8MU1D^x56@Kc`az^tX-z6RDvOTXYK~-xy=X>? zdQ9;tmch#j?H%a{euht~=nn8ZbtHq#F`%+n+apRG^x}K&HmuSrbGf{FeKR%HG!A1n zdV|3{D<)i_+n-XRwHdGF+M?2@M%o5gMY&s4L{o`mJig-Q(`5DYmCuwdJzfCROY5*b0yh&x-K~Rc+DBO6qNgv zgGI`aKCH;A4PMzz_!Z9dV$u!;Uv66&h}n&MFW3wlk&*BlQAQz?G`hMlaz5Z!o5dTo z|0}yNru(Ij30baJl!gL1UfY00|@`4T8RA8Rx&>zf0~#m|NyevRo|x6|1S2 z@La%y$)>Za1Dnj6ee17=QY5$6#xP*JH1sbX_?s{K#9m>CnskB2e2ABa{OQf1gc3O`~AfXbtcB zaq=g^Iq~wq!9u~0XD6`Eoea`}!s{uAUc zR;Dgt5mQ&az!NoV(`KM~RQdH?D99}Yn&4uT5}?(@g=jlG+ES#)*I;36gAuKJ7XywJV!ATzZ5>>D_&`&uH6oby7V%bh3paMQT!)N5U$lLw0?S^ z3f+66l4W#9uXdPo@10`eOZS+urj|#)L-lQ=t|LhgZBt#1v)cRR^BqsnBJp9hkahVg zeu+RAo^QEr9F2@l0tH#5B`YU*6u-L>m;1RRfb$%@aiH$r*TWm&WkuV zZZ&5^bVFz?C@^gKV9&Few2N$jo)_^9hk@zaR2Zxo`1JKFv2qKB{L}cev`I zHH=gDy4v!HgFXw#Z7=uz_I)-*%`dQg315bZkvA}f`JqcHw3;w9$-crBNszujTwB6J z6V2?a#6+533ZlHB5|Za6%1W*(z$Eg&$e}_sCFAyW!@{vL%3-rXn&4(|>yx3s+*A#S z>_YfD%p6CWQNYp=M=Vabqx!|nXgqp4VE6?gb{|OC8a^p(l{qJ8xT$sO1nc@oT0~J5P7x+aUBUx`1N*N#r;Bl4q}SLDvFfJRY^Y(n z4`hjpm*DPfMHI)s7VwZvC6`!QM%mLV;NhxMv?3?R&50qFd}}gGK5K)d(n01-&h-+> zqx+~~fuc4Tp4D!QYhCz%F!z>mQFUwFFd#8>4Bg!--QAr^C?Z3vbfPJD2O14 zl)}*6(kMfxq)7X$QSWo^^ZuR>@5krc0JE7rd+oLMb^U97-lvaXGjFIMuT<@z%U*0l ze3@FwaqHvK4LToJr4ca3TA~*0JhV!8Lv_=yq8?rI(U-WNldLQz>hPGw_41%QbYwCX%?b~?oescy8D~k=07DW`u6kb-y#nCZ>wSe`?&7mw}V9Qco@EjQ6gFF zWH1*c!GRyO2Y$0f8TmFO)p>L5Rf64@)hd4oRt}{(1@?_z6hSgiOhHO0e_A&-h83Yc z<7LO8^+Oe(RDN-jT-t@+pX7#w1_}X?qyeSNXBKO&kmanWT~o>kp%CmrnhN5SD&=aj z8+Y;4w0`DORYPyaQZGvXm@rga-xiVJXK|3jjY@Xr%M9b9x#AK_(~3%wJk}l{O(Q;s{q~73t6{dFIomTUVa1=&B*rH_>6th>BVx-rSRe zmik`DQn88A_1nc)imdp^Ow-S?5=7YG6Rh?57YEeA_!Vo z`>N`QHIl2hSgVXbAEsBFxiFKFyb|Lrw!Ygdgq!=kM3sSdg1nNB9YY&EhU2q4x4Dwp zn~HtC8#V5ergr?xIoHpU0$3|st*?r`Yc)4+&qMdAaepulj<6Ydnv9{RS{~Fu-&Cg4 zUG=KzkGQ;o8*`Qknq}6~81XtRPZwQ(DAG3hqyJqa4e{8=&o#92s`@?u`B2ezNFl`L zzSHy{!t@B)adg{8<*AqBBll}b+w0=}qb;|^WGu8SqpbR2y^M+byf4KF?vV9|qbHqL zlc*m(rm2{0EJARIu)xusTVO+19T5Xq6>6EG6jBF7MI$foUz(!jSA9Av+s( zg-mX5Ucg~*HHsfT3Feg&QyW2XUO*Y^DZAbdl4_C>D@M|FUD|Ay9kc7Or&V`m1S8y) zDl(|x9^L)H@&!%&Q}Q^7w?cqjAmApPQ>?>*DWG+O#hIK5I-n3z*RsPW#BB3Zdx~TU4ykC>o5y zF~A{FOi~KFKe&;T-JrItQ6yzo4Ua8?Xn9IlmCZQ+LHFfS0KcCS`RLD2NX|b*oh8_P zMv>usL~j9bGf(amsF%IN$ zbujuGdWh!jPsfjR{izFzzI%OaI?&-)__QC(X{p)itqI+U{8;x)mcvC%?avvtDq){y z@5ZDYN&V*x8ABu?L}i(rg5H%?Q?zaN;x0xI(Ll6lmB1R9JXT+DJC zROU5i^B|v4e7)$vXjb*^ zG~fa-n}F)290S;)EC_kq`hwW4OPyaiE>AqDG_A2!z3>EV+!X-HXGd#Fe9!uCHC4P` z8kw5T2nTqOH!@InVbB}39nhG1djcxM`-H#RkWlRWgDtjYag z0S`@ndOn^fQ_2{CznlIulqTn6ewQoIQ^e5zF&+)$)p)#VqqcW7z-Zb+DGfdEyID|6=gbo zvaiMg4WpmrM@kRnF|KUxfAHOHDTZT7?s%Eu7o}(vlK+d}=+%8Hz~=nfR?viGSczycnxw%FblKS>;np-&e78FEqFw^WS_(S%)Yf}IE5!1S= z?D;5Q$oF1pK3{sW4{n?6ueVI!uUrE%(5Y!x1@&Lx^_{{fP~8Oqy5DC^or6u1q1+NmWV1noGAgDchCn9>p*}P= zbHxbr>tLZLrgHMeceypEZXy^RPv}40?&s&Hpy-mHE|%^&9+>1mALgl0bR2?hX&|-@ zuAyjuN(q$?%5*)u;%dOJ9Xna82tCIK(pt3rV&pWEsM2XNCBj=dFF~mUgC*HVpmYWSuzc+z@X*a{BQ>UcQB@RSnb60lu8WpCZnb!0i_T(F9F z;bO>0Nn5Q62yhd*}W4K4FIdIzQk>I--XqiM6dKO6IsgkMiwGJHs^}GsX`K-cNwh z&1WKjapy-`1j4LRYZmH~L(G3RPQeh2C5*Q$URj`i-m$r@xF}};AMqQ1@fA?~6#9aZ z8&q~V^0ZiwRfv%)8^7zb_0*&X9h>oNnMGwcoc-tK_yhm}Ty^{DL0rq(ptD%tI+#vE z5&Q$u)*-7ZXJ)se7TQeJ=aTxw*_;B~P9kX}I^WzaH$NYHZs{|wJFti>YBS+?qTj~R z$}tC2aqthQ6*-1HBN$KBc;(i8Fx#RZpch?+1^o%W9m^IZqY+sUv1wankHaL$O3BDZ#llQo;HU=|yZ4d3 zH-@2>1hdCvYs(ktl89ECP26hH&qsm&MJHpBznd27rJPWA$37Y+P?K+~9!?h84ZUH_ z#yIHm*0r-h04WYzY%F{7taE;JG$bdS?;NOR;)~?+*8?otM5hSkRmw76wXL?ZnHt}a zZK!-ie72b%Mu3AGKY)9y9t9qOn{(ZbE8f6o@(KnU5z@?l!oU?Pj141z^mesO#_Z^M| z?-UD9EL*Ib#xRvk1lP#1tw62{oSw$!5r4rv%A+)zwx+K?Do1*|=BO3(f4J0ca%-D= zSUF5P`2EbMiT_kh!|Aug4m#eqC>xxB`P&=|0 z3tWCSeWSCZ8jw=L(t&3vO2a+AIo;pD%>(qcn5fib7fa+ge-0m`KAYMR+z=%Kvix|( zIMy77`>ZRan-}gGF|^DUKGjRCN-HmOaweE}rgbT!R^)B*aq&T@C{{KxCLSWyH45Im z!imZKSYcth`iyHhnTphA{}w9aBwb)VP!0afS(l2^DnmnMAm4#KO+10mcUr{hPFPjj zNi9>gu5c{d++vLXpmn9|KRdqvfhkWuB2w)aSdNwUAhIEHS$47)GbX3Bg*33tXfb=y zoc<$Uk^zODQmwXKc8^!TUb(m)J_ufyL8wabf5#I=vD>x0(GXBnfemASQB8KTEQ&#*Jff}d)rE1jN&cJaLBa<%Jq#S161 zRz6cjhGc5~mS_8+?&Fd8G$xeDS^Zk1vUkL6LG&%h_G?LrF5R7gTdzKXasiHrpek@# zHjvj=TwWx45UN}MxN17=!;p?sDY z7s0hg$iHv`Bhr`8YzR&(lh!B6OIYg&$C+!N%IT6RjxlU8|pTNL9DNA=amo3PI8XFM0=j8s@{-nThseUH%OJ?RMlT`>8vz(Fy!s}mi0Zqlf#u%S~u zg@$ZKKS8NuAeWIU(D)ekMH|)OZGAc99oxLR%WH^sYibQpn_IIV`YO=pKoqh@KUaTg zmpUA*jr^LwiEXOYej7g*p@bvY zx$_v#fWF+~%1uL)AkND-iCxaNQpJVWosrxn@oLS)?LCRu8%7*-Pxi^)_Y97&tFD*t zD*7IWRw>luxcqEgjeV~2OgHE2G zQQ%gkyiQV0*BYJ2efC6k=SDux_)N-KHmflV8@9MzqWKJgrD-57 z>-}$(v`&}tyR>{7_N%kV(FJ6?ttQwKYDEn+aj^GYb!7qZbwu)I zf<`n$ZB<%ptdepe)BJr(U)U2PXl26pAtab*V(H0>Y*HH~wh~kqRKssHEnY6&3zft! z;bq3vAi|)IqTMR6t@vzm!^cX(En1wl%72DIp3t{02?u&fmuw+MEX|ow+>p$kH>AXa zo2lQVUKq`h_$+#r3yR&cdea$po3Qe3G{$IMw_8^h8V_|4`n*Jn*CFK(cTN%F?b2UX z??UuoQqS@-N7U}<*(|xdv}lMF;D4%mI&(UB>1OHnNtwTJ(@bcxsz}^Jc;?9E++H)H zXd|qb-SS;NI)56x=Vk+?7{*oNg1MOeBA!B=*h{u;>_?GFu_1ZJr`S=A5AK;c`G^ST zX3{D&OJv`|8%n@X8^6D1uWEc-fXH0x^v$AOcN@WTIjeVD(*lipm_|hl@RS>f$b?X%>f();Fb?yg~|H zOOE=+b6#_8!)OcPjsoh~!NK`EU9m4bAkJ{b3l(Y38FCdhzE^xt!g5mr@O0GUg=p=j zUE4F5mx{wp+72qL?=mI0YTnVulNxA9R%F3Z7z*$zjkq1Z&~0E6UUL@))Vdv=(ub!i2UX|De=B(N~n)xg_mM?-E0u%=m?1)9OTh4ReaHJ!@_hI+M;bN z3pcm$uSzwwjck?t>_~ifP-Y~CNCg;$W?~u`&->(EV&q6^pU957BS8wN(l#3 z{ZI!>35yo#>w~0juHYzy)I~y_O@&>PBg|h4Jf4&#KqSX>nbKs(y&_w*kpMsqbw9N# z$9*A_9VwacHo5gRx;J5=jzuo7N&KNXKyIPA8mFRK1Js{!#EC^a;^-(s8=BGTHcOMrwrKxHD#F^oF=q z{56ZVW(qcR$!6qML=2qx2SwqRrP<(vByu9JZ6r13b49elVM^AIshRuHp4fmD6I+9v z@za7ZQJ)F|TNDp|dASE8X@RjiY{!bF5O%w4z8qO0=^z+O1; zo@9SF_5<4#pP(`&wSu?ciycR%d98R;n!cKA(|z9TD=ZB`a;w z7+WL><-~xdkwWT{2m|YlG=lh?eQ()191$rQ+)R~e#0DcgJN7P2 z9=44sica6U#6n~q5vLz@pNWWc(C%BL!%*Bg&zR5!zVahkHB0tf#`@F|@;Wwla!g+_C0(B| zJieL2P&@os6pXefb*Um=6;X%iwzlS!4$&2~JGaa=IG{<5b|$jf-!dl*)_9|g!065VUO=;GGr3QJB0$b{LKEN;!;E#2Aqs_UTV}tPidKqy zt7(T+#ZT^z6mVNCnRJj8uC2V6uMZ&`qqM1O!W4V<%z%)tH8#?*DpvfX6L)*`k_S{F z^Gn-^OT6VlYo4K6)R5JTK>wY)AKAH}vb{1>U!Z+TPM-a@C*?wO1(W#fw%XIjJYI)S zIdt=+k+pTbTIo;93fw}aFl_axa!R!ssd7n%J*t+x{2a3o z_CBxpb&fA$w%e~1J11~!vC5|`Z+x2^^C?3#*E5wekxJ&LiM(H3-3Lcpob2-{0Og|Jm z56>D?Z$YY{qN35Fg~mB*&CnMHIe~4-H&Q#Pj2|s1cV8t>brG+A{Ju6Y&mZRLoy6ZL zKhI39@OEtr1?{A$4R$g1Mb9tUYmS~O#BwhbC+7>k#2Hc^YQ)nBS0gY(*tvYPD{>#X z8R=c^bLOyLRnz|1H@L$x@Lc z%+31X6x*IJEcjb%0TBzZn3- zH^+VLfnm-M=+I;vq%Q%deGZV5qRJs8T5g!Il>MaAcbVp!$<+6br{A;Vjx`^ik3j(o zhLlV|w^z`I&H>_gjdTH_arnjlKK%hww1%uP{R`yY9u}3vYNg{@@dD?yFHSb4Zcjb% zURYdQ1l{sz8g|m9kmC?6dbM*W0EGhhwaM?tmRrO8aRpZ&0JT|2R*()K zt~WsjZ|Eo~^YZ|RO?tPjg5TT9_!=zbiKgGbf2VIUk%~=50_mrMtaWiqb-x zQ2UeFC+~hXuS2AyI=KVG#8a!^6|A;K7*YT$n!v&=i#WICG`=rEU zjKjVmh!E7?yA~s_Sq%xV%Pyi4W_pCIW!!fX2YwjxoA-o5oIdWSaTSiX5zio>BO#{9 ze1;6|vUUKV+Y*KVmlEQ9`H$?n;A3+-FJW zIsN%a)M0?+#PCmGi(`rBIaC}%8qXqdDGJ5-3?Tch%^<7>O+IWIWb7c_H(pfXZG5mi z0+FDgAlW&gvNPnkBBB_c0s%w!x%!CoP$=cgVp$J?v>;$S#3)PWG5)%G%Nyq5QY3h1 z3~6U59YW$!az&WuBhwm`T&bO0pp*;T*Fq>&Od<#6=jWfFLWml%qz|*uL@eh4ugSq6 z8;xWlDE+l?j-IYOQ~e5Dp127jFJ*QNFfkl+^iJm8hmk6k_)z(3i?X_VZ;^VXrro;x zWw4xY#|Yp##R!KVB8ejgbe)Pj+(=lo61d{nG&8DvLTlv4H!lFeL$|I! z8)ol{&!=VF~(;PSjRN)2O*%#nT!<)l6@nSoJjUnTMv-#o>);Z?E zF@@W~BBh2jE=UXA(XebD4`k*D;QO+;pErxaGw`WE*fIJ?7IMr=PNtg8=82c)-k)63 zm<-9>Yr&Kh%&Dr)M?DE6kcgIdy_r1R+5g6ZRH!>;xx+I=b^7wSBp(VF9#ci`@z$)dLRtmwn=J#p13h;$b#J^4wh%KUI5DB|mE(D38 z;Wu_jR1}R)$NsU3jF6y741Yv#O8AYu%b;oruwwVUz4u{D+u08qbim-hQ+)~3Ca^sN zC;ar9sJpN=g()ha4cejcOn+Z*k)vb>={3W%+F|An=$8B{&pz1$fYW;w7!I;^jTRc6 zP{kzxC4vARRV2{YAtr`w3uHj6v79eo)FYkN(FR{?J89SHVrM-lAPkH~gAh$q)<#sz zR}%stcseiB(OywVmHk26Mw=5Cd7KCPjD)b2cLc43t}frNHqNT(D*Ecg>%Cvsgt$=x z)2zqbxrwVWcRF+(_5_E-!g4uY<>|$HA>r&X&Mpz#ISv4bnOqy@os(h;W={ZONB8qkeY-w8x67W1+c5O9pz?06#Y5U9TS`e5^7hh_5u%6^Exh7W?L%>h!r>ZjHv3Q zbGAuEb!NUt1ygI`8}WCMr<97C-!=wrCM2uU7oPa( z_|<`?>;nw}`#o~F%gmM-%xsLgZo6iHpiYAgUV-WE>=7kF43? z=6HX}IABnW_vFR|D|LTzOg+l|G}TiV{UTJha!0Ki8A%BVBGP*D4Q`A~?MZrudx#e~ zCD6V`8wGi4iE)#okr5)nw&~GPi^DO#Fre<8#F3khoCi@=oWNln{3=l*iB-^SY zNhOQtm z0xm4~sw_|`2vU2bS#^X`$vDn|D$XL)UI|vo{#}RH_)Ws61nh|{>%4|9?ws?JD4j87 zS%|vHQ-(yju%Y&rM3stfe=yJ541m?YU3K0@RyuXMC@=8WNu^9rQ>pE!bXsc{sXSdO zsV)L`)Rkh5o0rH)+w%iYYVx!lrzxeFyI|oh6Ru>BDo~;PCZ?23@Y8kYz#T=rJ9i?j zeJe0{-M8R_s5K!_egdxGQ-vLacr3g~pu`|`=a<3hB2e0Cwd}SpFnKRBcDjIUUjcvX z&R?#l?xU1s8EUPV_w6v_BYF|qTO1yzruOb&>mJV^f;dX}V#!HSqvN?JwaIH7gthG~ z9*t4Y!elyAU4#U3f4P&s#VO6S_|Co^|JI_JDGOtQ6h(3)6bfA%ugU!ZWK?uVB)u$p z3M`BazPPg4r7}NWk60LpQM|ENl*(G)^jkj0NCENeU>=vcWU{EpJ6Bg`<#z*?Vr6e5 z>J?eCNj(lm1BoeTP{6vjqLPxUdOvLjYdTje&3hRP-W|Xv8wyICQ`+vJgfo6zCn`FTb zMmQZ94x&}V4--=9x3RI$d$(-f9zwC0;~kqTyE31<7>#dZJO~?#TJ3O2=BxVJe%StV zOZm1RaqJDAWZ{?(!x3_K)v7nNA)>|AC*l!bNc=*_LW-2^6P<4nDYb%%`Uec31rxGe+7;?Q9?=dJ)Oo;Gao0)7f`b7SBT7;uk-tF; zJ&~h;#BA534A$JkQ{Non%3ot2S|-l4$geGhz}AJZD5x-46`E#DBu$a=olSTOU$0#E zhQNPsPTrFk?Q#iCHGZ5ql;~HSudC_qvCY_P!U8P2RBeS~y0apb=rx zWZUkUOgE(Ns4q~f(Nn)=I?Q^xa@bXlwiz4nSW(2%vpO(_j;(ig#Be_wfu?Tj*1LEd zeyeK}kI5;&NQ>4p=XfQ?10^age<-}|PGY@yS9AC{%{z))(h~7|CKY)F7{< zq6WNQ@q5#W)+%31MQVCA8~4K6h_9NX(M$ zPdtLfEJ@j~qJUYii9}@bWo8XJpSSA+?52c$HpH1o*9vg+i{~CMF%3UN_&8cD!v>U; z__fj8c`$_#)l#4e_-R`&|bA>|3DJU*5C1&O%OFS22z`hO;o`cul2MGRUldKrtpY+EqAVi)vNJW z0BWYY|C)s;lA-ZsEuVV#Q}!dpGd(-j8IN;?ENf>Pf8mfE6W9(J0YydaVxN2jtKM_9 z32la#9m8zoMREdTsvGLd)Q~eoQ*{9+MwhvMin5v_>j>KnfV(AtM)rUT9>zGSLbahB z%)~H*HuZu-dQ%;K!QLRwNswWf@`THX^0<3H)l(tpG@k!^-^t{!U*@NiP(DG0pY; zwJ?NHBi%zOE7I}q;1AMbHu(Qy8F^rFu@>5q>2kZEU z=yvrMjuVV{*Q|1M3(xxQkc?w_2TL?nI-O5vVHgzWl)b4MuzYS=2)Yq(#D+b-E9rV7 zZ(?VpLZ~FBbODO2DJ$IdQB&%DV{BPdvH=IymLkFX#!T%> z+3;ap@z47CRBwk9J_^5ayd>BEfG|hR+5Oj~SE{8(M5v2%@A|P@7cD<|na;pe(PS3s;N}HQ@uM>5J*)2t1`S z2q`TAze_Q>7rMpAqqWy%6BBncFV8?OMMgAbj@r*@>5vc5PS;dJ^|IE znA_nGZVo`a+!l^uI#9fb*HJgAQ6q2B-*-@q6BeyRC1Ao?jDs&(p9wp1`ruUALh3?M zwQ6l^!}5a=o}K6L32Uk_P7$Thni1LtBKn{8-LA?mHyIzkFFweIUEm4oM5r@_uQX_n zXVydFYWue=7tPQ}=e{=!eKOym%zeV^pGI9085fN)UOQTnrR`6mDZn{#zlSI*nkO2D zc5|n}>MoJ>a~?L%<^-E$zZtjaZ_pIxkI~g<`47X@KAI`{AUX*~a1P!y>{DG(LaF=4 zaSXqm7J7&Cnd2mK$Ii=9KBs72#3&1~GF$7#7?$an&aw^_IF9wJ$C^CziU>4;z~`|> z60_`e=D)A7^AJ9Aigb`y-bOe@7Ds2(zIn{sut3{PVOr_ZY9qic(_52$(y;0H_CS_) zLJ%{-ny2Kq5HkNR`k3dtVf(wX4u2rdEy#Wu11|Y}jDDh{6uW-Pucxm2cd`^A}{oz#;Ndxe#I1P2tdkqjjS+KVP3=yD8v{+7cBuB|eqo}}cn%nU^lb-8&L*V+T2!WeiLefXIE3Hz+=M0>_61fXfT3Rc zqDz>F+6EuG7bYCcG&@aXGp*6CPtMwD^2%hWg89(53A&xkp8UupwxDiB)PQ~93>S-n zjMl{@naIk_=YY|^aJ718yX1DSlo45_JUPvzfd9)8tfK505A&rMCVlTu$=K=w#8l73 z>)l3cdg6`!#NvKxxq``>=lWp;)s`wI3LQ-Y)KZ)#SZ!CdQPu+ahnKfda zY$ZWn2|3nQ(O}4&DbovwO+Hx5Nio|;xUr<_Hmj*a=P+y%gj?$di-!qF_;7^1Q8I9T zGT$1~al+I~ejTGto6yaLEqCi>F_(ZBVxh7jjXp9Vst~JRm znAw~t#ukrzR#FWX#}6w(VjBkgnm9tO{p=rD6Zmji{@lT+`72r~pTsFC~x{%0=8 zp?RXadA3~r-8I^t2{rfR2qnv|Mg3PI`AIi;Q;0gpK1kH5`LH!moYeD2t}Ta_ssM0O z1HOAPzwv1;0@M@Bs+M3m@!pT?S~_1Q1;!ynBoX+5f^Vk`#~#-0MG(~)8eH6ZClt>2 zZFTTkH_X8v=YH!o#}va~oIiuaCC*UuV6Oi6|NCE4aJw{ojrQM+yJ;t^fMC|2>MY6f*xGW&7iW zGEt}bQy`dcx#f-~`>y5s90CFPR=e!j?#Nymp+n*ir4c4iZ}A zM~rhEbRa?PZ@|P@MnuYRwV{fIiOF7l*!bo%z(S;fc50>R*0;a5KO_n8zd5KOFWHiH zN(Hm#el7d|6~yBR8*!(9DL3A8*nK^LjKG5GN*%-t4rPiun~%?ev@-`twe6S4>kA3K3*~-`Qb>>c4*bb{ELqCqglB6{t?`^3(d=ow92@?tgeq67}*107>Nu{q#g$ z*aQR3@fxt@==IQopdOGcCq=hxcIlTWu~mk|)Pa~i$s#lkt33d~OJ>~Z*WlhWjMRq3 z{~YrEEO_&)OFKRwhv-1KaWaOK$hJKjzZ)#o0tsD6R4+ zjfUL;vXVeZ2ORWc3&k&7;z@uHK8p(RK)`yyM%9Rs{kEVFufeWoKH zPy+c6OaH`M1xEx464^3cnl-qwi4y`*`hoXV!k`l@MBX&%0f=UebuY=-M3N7M zk^aX1hbi+W5T+xsZE%ZB_@Tp4@Lm&%nPt9SQW>wBK43!Fo_&@`P}~oZu4y-Gh#i8jsBR8+@GV+ z#C-;A045euc+9)N-6xsdi*gSc<**wuqmuP|;B#pGIfN=Q-%zO#kIGQRmw}Sn!^(rw zgJ>e^>gTs#RqX^tD4_}@LgwOkI=x{lK>9ohxQV7}6crzsXE{5CaWh-4Y&cy{tIkwtO z#y}t$uk}5$J>2BDS?EOTKESVrr$%Fah@B8Al;yJ_cLM_}f&DmE@lFr=D0HGWDPV=g z`=EKo+1z2S8EXkM$<>vaE@=665wm71RfX+P;Cv1W&5iJ~%FQ-p9Mo6b3 z9vjYeLsW(}b7WjsHgNCLPs}xt2`bOZE$}QklVyTxja2$On7FL@8%Il>aar3{&5DXo z|8ZA=&DFptDrFk|aYDJjL}|Qn?}pK>T;ej4Q{iNG>)W{$lqWmRI0k5$(WGG`!P@k+ zY@x1+SRKF=z>&!8&d4xv*0b~R+f#jVd1j_#>`KvRVwh5QnS|!St*<)NN}UwN&J$vP z2xfBT?8XW}9P{=V&u<`p(ceW`-qiA3suDB}VEga`qk* zS7pUS-j}N8hdX_E{%Oa?^UR*pbHl`~M-I*fVuznDq1_NdQHi{xAPXqgcT1#USokug zJ$L0GwVNDUOo!y1ME&Obp%nX{=A4?Cl#v;5*xxlzuaR+!AhXcmLD1#jA|knU2->p~ zKsNc$BUFwxxP)k)h{!${JmZRr7vto0_;#$SOR$e4l9RlG`5zi%D$)hvT;Ar(oaqjC z*LV=}`dYm>eFWVgkZt;duwj)g<|2A#o-kMBp3E$IqMEK`w- zM3?MW21G@pGb#EpwUi`;`9`~#wNTn(n4&NG^TlQNA76~SbCZ);3?4Wl>}WAK$m5i^ ztP}T9fKthZKw=}m3p!Q#pg*2x6jM?a(kpc7O z)3=0F*b3J2XCgc^-eWW+9EFwS#MwT2{~fptV+Xr=n6~+&(bp`x!_=E&seEH_INVw$u)O^3_j!~ocl{A&-##H}{zM*AyWr#j z^%7T`wwF;%EagmTF45KKUs%YO?aIQ@RTZCVh3D~W=t#vWu}IYM65oJ)qiUn)XFg_$ zEtqAzgD?g8Yt3s&5 z87CBJ7`&r;i+w20->5$ya5NZhN>#*nOVGz8UoJwxVR0!fxl1?vts=T$B17{X>hd<| zI>?O`(h+i?P=FPnm=r3&lk<*?IjS2lyHj@ws3_8>p_rPK*0C;w?`bH@rl(Y}Hw_-cbFgpD{i-np3fY!8|?U1qm!gFFN{(V&;mqT7DOCC+!l}=$A$c z&Ca6#B=UkJEsShzxg*lQf+Z7lKG#uU1x_eBuKE6ee{>u6>C_?ybS&J7ZRl1X1{(gF zpAVsjn39gqf!F&iQ0J4a0?GC$mcI--NDX47EwYF-vgZK;$&rK631<;#Qp z^6ZINGuwJ6&tjxP!g@hAZs-y2p=w!$ob;o_vxKu&}zHnx_-lB;UQ@Udz zVm*p=Oe%rGA5dZ$l!6^wJ;w>2sAn6dym(S!F1gIIpQC6Nun1Y2pHgBVqKJbP&a9(S zg+-NBQP;9RIl~^e>6S!FKT%WW!xLt-LTE!LZ@eM*!@q*}#x)RJ!IRkgg7!lU?ldtw z*6leO3ut=%w|$snOHMpEsk=Y%AK$o|#UzxmQBX%0m{lpfpB5{HGhO4pA009l&ZVB7 zR-T~wiWi|Jn0uJg7Mb$;UwVITAA0X)cVWxJjHK)=U>*cOK$Q zD#O(r)_j0JBWbuxxlSTSEe+}B`I5!Y>HE*VS}&1KJ6SXNw+USzZ71>Kr7sKmH)WBH_Ss zUz72rj<=T=aNn}#4S%fTsQdfmO9t7)#TV!2U4}yf{|rEFGRoEf2wkr8*c=00p^bn< zt4Lm~@RLat9Hl|R;p%yfS7UQ?O}85y?L;H}?dT z&w^!wK8j^S10P=J?faN=g*5h1laq zN`F_DN)YB%!}Dk^w1R4&8 zGoVb<2zgT3um&`d%am-=|JsQ&TFxm-LDiNRqq z+&P(_$L)KbBW?UD)lN?ihNm2!9#=QfHRsV<}IYP#{$C1{=)Q2(_xAoVkZVvE6}TQl{p%dM>_s!>rpR+ah`XUrG#hXP=n+@9u@%{^e`q zFLN{j^=ffE8WFwrnrlxq_gjf|I*igtcYgVCYb^c#y7vmb>$;7%di|Hh-NV=4zU2&; zJaqf^DK>qJ>X(TpJmC|r#1Z)3y=|3JWNs2VLNph#x21je>de;GDc(u$oew(HNOceH0Ms?rsS| zX{4p3Q#z%)8_6vt(%mJJBCUWR4bn(RcS}e~*O`s+y}xt5>zse!T-Uzd_i?lLvuEa6 zYt77>d);@rpW900lb>Kei0QW<$XBy%HTgfj9}vK9=?-g}eg}qzMXm{GNqm3t5>XE6 zv1lf;kO=XVgVopjv@~Ic_}vI?;H#F_8yr#_?mrZ}+e^+sn>Em%wOKNz+cu;QVypBO}j5 zIpx{5Tl$pA@eK2m#+6c=P`U4Jx}h&F1K&IDhS{kg>x}G`&vt7QMr8z3qFdn0;rG_8 z9syYD31C-7QK(M;0mUz-^|^aHgG@vt$RAjQj>Sq*)LW~g(^ER%K)3P$N(b_MU@vK) z$wF{+E~ta~T+Z8DV8Ld*^c?zklq26(fH(#9=X6+>mZ#8txz#GgX9;++k$)Q9DT(l( zZmBFaFlI#mC{&sYLAJ+FIJP<1CfGt!9tc|BxtaU?KrWCbGL+1`cuun-w z82mS1G=H`gA}tqN(tUz%a5dQAL15iPosDcUXEC&~(BM8^cAnD-trJ4G`?7eSwQgqa zCr_n2YxC71G<5?u5ASw0xpTkZBaYtZo^_8^*3yDW`a3FK3nwi52oEx(Xgix4srYhV}iU3 zV&=Euc?)ya$DO{{c-ojVsH|jf`W_G2UGRGI6*nX6P{YI<c&n#>c(RLpdl3VFxet3Zk{EQNF|ma3zbrdHw;s){CRD#>xP z5st1bw{fj06ZZAB)adu5FFtELCy((ZGmQ6rycT2Snp}T=P`Vw-FU+bry`=DM@Q@c= zyT4Gkyf@+CwcUQg@Ma*?&y5z_aXGhhj{8w@aJE=(#YWccN=3Y4Vc*)!3mFRBnO*AE zkf-1|LX)gF0J$oQl?p|&!RJsP0h$ryjW`jslJ^W#1fg4LatKYtWMn&=n?(Y1|FYx# z7;xNAOrHLoSo?+Hyjd*0w~)QBe)ghgpUo+VzHIea?_Bu&%1)NGe!(EZLt9CKpnmz2 za8sCo%gO~r66xzt&b@^v$043vWaKRK6nN%)&&XeR560fs`Sr;)`3SsZA7Sdh7+-_nU)~NCxuHo2d*xoi=~!YnGO9lRV8eJtH3Jz>kvB zjRPwVBQzS9-H#Jd&6xW7`T&ZZ=k%7^6zwI7Q3yyxgcG8%TedMEaiz-Bqp*V|%Hr-^ z5DmtMF>Wc>r5OJ?w+WRcpot=MGpGd>>VdA)z-$SsHi)^j-PM1;Q(*!>e^hkOeunGp zgtPqe%)!B6e$;(n+KHRcP~fwY_8ffJ!GyGDpu z2d4?qdXBx=;g_L(p?mY!gHjY&{7!p;giRa6h)qAFCU(Bmt+-)MpO)>NUrML;oP7ZeHt%fXX|(nHe)^!+{<0$;^ULTPFAMz|o_LwLI(xS79oicwx!*HC@>b$*o!~UA zzJQXmKRrqvq8eRnaRv%Kk<|((Pq0;pkk_w{h8Wb?*aMKC903}bGeCy(1Y#AV3C*!I z0~vsPMKbq6;Y(W|J5No#G}MM9Bp4s)VLYI$c^6{p()c6n%KGC4R38XPHPFB*9~S*1 zyfQ}zn>}sLM2tV0?#fDrMiw_xto27z$saa$C*YTb&MTfZBB$$1d(71$dYO4CE$wK| ztwl7KZ!wx0T((@__zV=B#16>T%{LxUUj}>F;+mZ8IW)~o+`5yJ=ACGRnLXEX z`8r^6<0!Y_c)lAw=`a4`AjFM3G}Bb~YBiphc}SbuMHc7GkVI0o*l@i-_Q?(ihD)fNPYeZ9wtlO{dEu49lRBc-zb^2Y(aM|JmbwbB3NfX2S{m z=e@s|EqV-qoohlfVCAj{5Wpmo(fGmV>zm-nn(t;KQ2HpB!~*EB z1}=`bonEJ4lYfeT#L)5_D0B6#52R}gKH7^Q|FfKtHf(?&@%8ncd1U(Us8mNWN9~~< zs0#v%0&CExaN=QDcb+AIK>qga&lVD)yHIik=%Hg2bUap2TbBN}@puU>NV_JR{NL)a z)_FmL`I=zJ{)j36KDPd|wfO)BYxIJz0`YIa{?F-ezY}!H?E}l}^LI1(Z$A_OucHKZ zOUEvTnE!e2?_~;95qA9y&hhMjxob@pV8GknXrcJeFa5n_`wI5|{Rj6K_y10ee@70w zKNAA|7$1TN{pY>Em;Ya*Q$Oy}diR5qd~n|#FLrO2zxt~RXjWR>4W?h5NdVs@%k&I=1R<8qQB14X53gT&QG4qh5;b5CZDWct+(79Rll{on1GYqXoZC4{ z?YZ0Qwy*9T!*Z$S56^XLLBit)7hg9Nv{3)5g+B8PqJ<4k1x@&5x@dn?p$iwAjj6)y`9l$a~5+7FyaNM zu-SY{$*C0pqR^rRFX_&6|28#e(AC|&XfWbg|KFo%J{g#a>A@(B5aB~a@mUi6OWL+* z=zRR&dUwgTet0y|^x^`?Oe5WR%fB`nBSH&%Bi%AUU`H=FjeK9tl0-BWWP{|N0Gl7R zliUIK`6fd3c_^YW4whj}EccsoH`+fpa1?-u2`l-0-Y4wXHQ-wMqKE@cF|kELO=hL1 z(gx`FdCI9SDl4da;%R*NT;1dLFJ8MN15tI0d;Wq~r{&0BEI-K&fG{W`2fkT@*Bl83 zYvuvlVJ+T^j~fQGx2w)ge)bJxEPrAfP|62J=tjAW&3${s%Nos20$WKvLXT& zX^IhZpcQfOk{R(f{%QyQ3q_f9COUtEYyrHrZpO{~{xg!of*5G9D;7_Qa{t0U7cJJ!DQ>8Ut(A()KfOzqp)9?~J&`o%2f9biK+gSet z^~>7-DL$?6g%g}0KA)k!I?j=-_BBp|eh)~f7ks?3)4zkRPtM`V&K}f$*IE<$GuMHS z#KFJX+a(|JnEjDp3hi);JMe>oxr9k-;HqkHAD>qJJj1ma6QlSWEkxjG;Y;vuocEOw zH3PwO`ybAdlyuNE)w#0Xh%2wP1uYjBa$f5{xg&U_*&B=Y+biT;15VhOEZ_s6xEN>x zBh?H5uMv>V5e+9k4=xVs@o#{8*vcB(bHR2Fy;?>wWXG zhL5qOZ>|&MhaMnA{aH6sLBeqN_d>&}kHB=9b02g=2WF1`|A>VUeGx;6x5cAYTYf)Z z8B9W}QULW15dR>gM+KAdBRZ_vxV&7AY|wWx+$Me!Bof>-l=SqQZ_+!MD zbjpvDy}i9drG(!x0k*xWDO(SqEUu|(a;~4;PKo(L$HwEzd*Y+59Q5AquYti~ELBwi zca()w&xU-&X|}ry+5zxsxRW4tRG^Ul$!=!+%JCt-X&NXpq#BRV==wcl{m};!{Yklkte+UWkKGx--oQDec^aeQ*t=W9 zx~7Bge1eYiWdzIL2mre8S z0!Gof7iDpoXOEE|r0EaM5&%cvg_d$Zra}EXU;ZA_+Xq-<#IC76!j#{^|ELD8YRR0u z-hVF|#*-2jEzth*lDkvq{V?676*L&MxKaTFBsXsr1~XjHDM zq=d=mR(PfgZQCqeOlo?&``BL#yz6I66wYL$708iTP%k7YY>X7km*}7(3l3t!p;K@% z3IJBbvq7p^tS|)s45uu6cBs2nL+7y4Y-7YfG`Dw-50Rm&aNGXm`bN3sTrE*wuCXu- zi?euZH`f2F`qQ^)HyomQOX)72n0WJCEBsSp&^DM0fF;M2aC?-}zt{%u9xs~XuxS8X9+*o=hklZ2voyfgR*oZO`p>E3`a&E2Fy+ly8UlAVE>)iKh^o;qs@dc*ra8A zmPABEgjQWwZr_4L?gc2_b=ct;J>N_QFf}NNX`~P!k!|(@*(0v?J)c3Sjwz6Wb>7&n z3=vrbMUL3Zf(?D2RJ}Mv*$2LCcYg8Xa-gVR%XJXGvcV~)iM1aD?2T} zST!CdAzp+8GPYknvkL(jQO`?M5IxO{_{J zBF~4Xb3}wdcQ8<1jlrk4SI|OkXr=1~s+9eg<8f%}7fK%54Xwln1^i7FwY5(&4$1&( z93&7ImCt5X6t0`uQTYzm(;dp5R*W`dN5{l}(nX&+(*!lCbH=!jTUG|3RA)=B#{4&6 zgKAk?(%kId9)ccuKFYTNB|*UYvAGJC-|5ryM3~@A4bSsfNn<~FnBMH+XCb?7(~D1_ z0y2N@5@1d|L0$MT{9ewI`AE&9vX1A)W52$^b9iYEg7rM9(rj)g=irRA33IhTp*>FF zPuXm4BE*YO=CxQm1<{1C5&4mH^Vyk97>Z@!7a_;}7UKcVR8Z5Qmu~^`Kt!1gVAkY} zhVhT`Uk4yQP!{b8XpMvCzcwRz>2Cs2T}666UT(kz`_%KNOEzh5BS3%5``=!FC3@5~ zLJNl;=EOO41Mq2kAf4vfE$q);!?Po06mQ5MVbYwzv*u;Jj_0_gg~Jz zb8~YhTo*v*(JmrybQLT972)@tlmx3+t~>TQ&-?&$50vs$*uB!q*_C8ovfy6oXPFFa z&6hUy_@Q#VeK{>-bT)ztfz?(DmBnW0r)wjL@_v4?b*;s&)^6-Wl12FE)7v%YzclR! z3Xbf4%(V^elXH>4liX}a5XjJYjop><^GSoSfT8$VJgH^&0pJ7Za?S8 zw{fPz44koSxySYJwV8A7GrJ&xLH?61VHR-3PAWmd5&dAHBCXJ41t?!?300;9}((C7= zTC5izg=VQSh0hE>zwD+CRnGbeW#SzZNd~!WZ!*tCNN(Kcz3|7MH@DuG`vC}KpeA-h z*OoPE@PXDjs22&_M!fd_7Dxe-1kP#gsy_haPu8$oWvkGwcV)YS@h5m#45Swz0w@S) z@&S3^BG^!`5nBUq&`iv8UHkeqs8ayAnZvBVqq`?S8j`_1f8GwTOF|O-2#RTk0D$K` zyFx)|ZSD#fav!e%pP+02AU_W3F0$eH?eb>7no2elNA^HAViFSL$YhL9i z@^#!~-bW?qP~*#`i;stSl@Tjyh@H4MF9N+xU@TqWRW|V%{E* z`(GU_zEKaBP*t{6f5jOe_<3#egas6gP07dK$0)!7Sv?7hUwkct1kWWxq#1gkZ(iJsS_V$Y6K>n8)14N z$b`J&hC17WZIWw@LQAe1(qy!qL~a3HiQa$>c|5aws7qItW%L)%NZn`S^O(g$C)&^*;EVS(xD`|h!s8(KwGU(iBCaL*>i=a-G*SBKacv6ysX zrqH_Qb)u~?c}7z?9`UzO*|>RWaev-kB5m`rp|}+ii^yTd)BNj`nP?6@4**;AfKqpd z;F79{AA^bG-UO8ir`iiQr@tEWlHVbYV>TQ1boKH0S5|K) z1?zj;O@8)?{&XA5-7d*sKd47o9i2i8-Q85=!E*?_e6@BL>-FG>k|vR%NJO?aDboA? zuZ{}cN4S=UF7}Cc@yIOLMA^HiTZ2=1J$vGVzttP zu!nJXd;6n3+hl+L}iU`*p#=BXNC8Oc7qsx@i*(q0p& zuw=&#NvK974%MCYsAJ(QC7eNltXT$5yLsG>>>>9Jg_7hSwry9h6Pckt4$ z{ek<6*~8YuxIfTbIwI4=^(I~m=c3s}UscqXGFgv5v`JBX3n-rH8Q5p_R}!PjP650- zJ55_t#Ly_T$NHsr?r~Rx<>tuFB?ZMpO=gyj={IHdBXm6c`ls6`BZk(VRNoso8n(7h zSoOX)x$%B{$>#R+V)n&&S<}m-Jw!R(TJ#U~1*fI0he>qAH}}5kRd6sknHX02-4#-p z6F_GBx5~ai6E*;x$mY#P#J2y9yW-ps{jV>% z)Qq37;jtS#!7rBggT>a(Ze!FB)lkYjv`9~#zp!`$Ry9g@I9a+$3TLZ zN`Vqf`Aii*4=EJG6<(g3V%3;*kxLY^il_q9jJ%U%*uM|uOwWd!?3r{Pb7@0n*vGw% z1spG(zqX*lDkWGNh&njuL}gGSrv0c^D>zQ%g&@;4SDbKWK@0jbcFFY~hHNsF97lFj zG6IqYJQTz4h6iFP*5Vi+bHxwp(hJAOx*TEd!!`|@F^V&IxxaBb1`d^pxK~7%(t+2W zNgZTRfIu}-Kg0&#FIb^^4}rpB=?F(>HPy$_u$4pt1@^u$29aOuePqsv^Um;9=y3xU zqdb4c>r8^TCnE&&uxOkVDEE~U;!NALDpsk(`bCM)f+QFzn*(+5{NQ46+xc_Fe4e)* zdF@C(gc@yBXPOr#D)jemwC7`>I_v`d!@@u<<2EkfDl_99&xA9i)iB?~@WO7HG4GIt8{ zf~pl`vSazop7n^mWk53RX184MBo!q!CVRz~M#!AnME%4?RjkhFClPl1GeoV2vkk8_ zF>DZ)Lc(Kl>nWYnQ3%~^V`a6xzo81vf=Y+D>#$DaSLpoW@^0fr1i8>%?XW&z zf$j!5gJC2)0iOlSP1`!AeFCZvF&okk!Sj?VL5F41R~*FL$znRNJ1I__qTR9Z=#yHW85$T^bOK>(@2C82F5^@0$$o5$-64(K6qV z7smBFm~oO3v@&=tDQt7z@aA>w7i?q07+A)R`C%W8%tN^bj09pOMU3W&x!rlA8%$bO zJY}U9C9!ejQQ)wWj50+=6VOb;5IDb}NV@3{$iuf`y~NcwlMofoLrb#L(to!UtsE>N z9V~CJ%-+oavFfFZqW?DbG4X1O*HKMc15VbA1>tIQ7?%?;pcJoAqEt-ozcb{fdAKx& zbw$G4v5bit$2Dl2vYjhTu3Ell?xsAzGnEMGf5I8=-e0xU)G>Ko z_2jG@QhU$M9qgO@QAORm)^=_;sOie#WvyPdVXetS)KkirbILz65*t3Qjeh;f?J6;G z<@E(qzxF38*)AqxHO@lSmZp33vz?%HfcB(a#^ z%q*5(BcbIhS%QkJ+XV(WRxPXjQyP zd6DE;2I-8oJe>^%Z6=u-k)I_JK5txfiE?RZ;tu%qw1g9xiK&-6R|vMS7CZ z3_bl^Ivk1{+B)rAGtMN_U$eY5o)9YB71?&*Xa}BG(Czr*L-EeV5(#s&0T({n-W+u& zJ{N**l=??rn5!XV)8uxk50Jmz@{+xzU=6L)lM)|LK%T>men#+!vZhZrp(6 zN*qh(ceIc?!L<2j+glmrPhHM!f8`(#T3@jw$%+TIuYKV}&Hm`-Wo>xi*I+k2=HD>_ zX(x0jQbYCL8~xtk=6TR4#j^cb^I`mp&$bH%j4)XlYXxR^2b*k!NLQ!PEA?zTUqMO? z9PFhO9F%)08#eIhEDEBoS@BUbbhV=#-IICs=$XVQ66GuxTQeUaK1^-eq~87tRrW^9 z*%d9;;wp~tutUC8_|;Szm$WhQj?sw-HF^-)7N!X`4*WQYF0!!l%0jY7B`WrX|srESz0h#;LpXDciiwrR=|oyHqzH z2~_N}wh%;CK4a|4^W%9Qe3)RQ@I0V$B_1bJRGMmYkypflJ;bRTX|3|@3f#Ifm-4nw zIVmqQFW1nISwx|;R-6GtgYr2 zSP|eS<10GW-NGMWe|^w6!6c|TX>D`2kgz$YG zEOgVT_*D+C_bD0S>aM`rwIWRm=M06NpThD8Mo=eWcqe*Jn!Y00QQzn*0lh|^*)#PW zTB;HZ9@1FVI0pLBj>oN{2$xA;=w4(0!p&sHYk6a0foo)QCesprd)a(T``9+)l@_uA zLO<2Hmc6CX%=Jo^b(lPCKGrVpSi2%(2UX3(b>{KLJY=@2GG5e*oowr!hoPH~*ky;O zsfPGFiZDK~2`q|P(R87BHz*1ntG(=Y@go?466Gxq8j4rJu_qR1sJ>PaTP1IKj+}=P zuXsZgKf>#n^n|>t9AahP}uz2J;}-%NskrQH(+AvTJq?>)HqhSs~Na-*As2 z?g4Ys#W0C6FQ&$QLh_aT+tRHrR7Wmlg-aO9lhI+N9Se7>19)LLnIuY+XV^)vJNx&Y z@LRtJ+a%>Y>}Kp4O)t>is0q(ToyXm@cQbR`6I*JF4PviCX0|O7^`Vw#dw<^~tA6DK zE_5}Fkx=cm1!9^CQY~3j6uO;YZNldnX+8bf$4F1{G&Oh7szvljPrR#yz5F%$9nwTj zO_7~YM(D&*QSs@=S~5e4mn6Tc&|%Vtlf?K1g!EDD(io{e?U!<~&J*m5Naqn?uP+o$ zY_Qou82#EeILuXlH5}Yi$d@FlI+pN`wDkE;x1IdA`1OlNW7$_StHLidoeOSlxBD%sQ}-9CC9;X9mf?@51=g#k^em(p&VO(n znmv+N@b=af)-+6|UUp{+B35jSkT!o)u>+09J7Y8}xT=-8q)3BgD69yI;aprqxM_*0b@Wh zaSuh8F#G<+P)y}R_8~ELjwReq!crDA>D4eTWTi7J)Zs{?{4?=knqE!bH&&7ElB4pB z?GjXS7%?0&97K^<9TkO-8bdejINC6R54d1Sm?OPR-$h^-2RjnU;~I!L8ai==wc4YR z=ZIZlYlV(UrpCyy(`N#t0vnUOr6UrF-Sal>$a`Pknqyb-#!f1uRYr-)yk<8*Xsc>< z3NY*68GAIS%;7^@Hta=!vp*Ff`Lv!4iS?B=441QXjxYQ$N;{E0<_%pPjp28qatkzN zb|kJQqKGGKM$b3dx~K(*oA;S)668R5R|_8LZ8^9#MiKi(EE#GL@~Y*VboF#)?e;nz z3!4O~@s13A#%sp*N5SW7aF{gD7y~AV0)NoO^O*;!T?!$$GzqnKGdg7n4RsKAx0rVIez?Jn z>LbEI2b(@jr)mm{pPRPZ+HMLp@9?xdpf;e!qo7g(U}b8o9A+8?IKkg z4&32>##JHCW`=BwHy2w_lzX5VZi1xoOAS+S3Y?)T2IN zJ@@n?PE9Je(R3(M$yeN~)%c-kruMIDrL`l%g3pi3DQ3B?#=9pHh9mUVmP!GZ$O*s5g9MR+*GU+tK7o>w*{cA4RRjyitrTWSSbw%JcJ zgv)JE;M06Ih z$Pw8v6eNrvI@XA9nUXT};JkMPZnwfGzQd-H%R-b#NBrHy=NFc+{+RTghcfy8@Rf|pXK-J# zK&F%oPwbr1u3grV?qCvz_E=iYC3~MppG=%^MoPRjYQUx7;$|Re+LNySVAWAykfc3> znJA0z@TKyqOp|5di%IL7LAA%~b9qAQOoR?|pVrgR<&*^St`}(w;Iecx3SO1CflZgH zl;Nr$pWAocKgAz^Fm!C;NGwxdY?emKVzF+xlZiiZ1iS7E+@OWU?&Y>FfBr~nPfqrq;j3cO%2||H{l~cTCH^- za|eYheB90PvTK-tL%951k_UwKRygiX!GA-yz@>YD$nE&j?M0tNd8$svV(etg>eWPg zfo#j6Y<(On!Jgj|X2WRUNArvDz=@M22NLU{1li;-mD9B^5vM;vL7I>$Ie>be5~z*( zAI(|Nn|K97&k8{<}} zmGb4qllSPC`y-3b9i|D8}F9Ki}OY zZ#c$qwfphjv?_mo=}|f&VdH7IKKWd2)6e67`=B@(DH~- zM3@o54nPY1!AU@o0SJ)6pI29%0GZv_833Z`Y})N zCq4S~pIi3tZRdD-^~mR^YrSmt4x@cOI(}OMO6Y~n?tR5WrFu=pc@D>o0_1F*2o9`J zGnRh<`=VMX(e;Ba01+@kVlC`xhc-%lJmIC-+VF0_hP~`r4#xfP{pfFlEYTq!o(kFY zJ@eQ76;$X>;a)fyxZIomF!txUl=mDWf0&HlH z;Q!j*O)_}n_HaKLta9d%&j<>C2#cTmANYsg=nRN96xcGZRC(_X81v7r?%HJd8HyMB z$!U7~w~7BTzLaAqEaDnZK;-Xl{^vslP{_sq|C9eO1KX`|iZT5ZI&|?;hK7cKWuo?Z zvAjXz_}^;2mISPBQw|{^A#!qZ)#nX~Ab#Pb_(54b{9KHqf#WjK=J0cxS$_Ca?FS@( zJ#*Up>lo^vzlRD0{TIjh8?6MLp8q_+|H3@j#9)-HuJVKbLYrECF<^b-Rk-z=7QbuW zvlb$*_-7+8?@&|$zB~42*|0~Zgm;1ZZv;I>I-~RYIZ#h~Cra>tv!XNrY)duAHTL+A zf&9Hp@dj43KA$0AjNH!!_ak-O=#u%**IQg^7qfByqBsE(p&&L4@SL}B(Jy9ej(R%p zr2Tef!!+#uy$~xNYIiAsE{q3upW-P9CPaBm02(ZpycyE0*UO^FtRno|dHEM`H?3^G z8(mP`dma`*-QACvg5H=afJ`;s%sM;PKiUUBJU?a)z;XWs3!+GoUCfaE4e^VpX8M~; zk4{U5RA|{!P9kW(j?BU*fa@Zs!MbTAAtIa3t!82Q+X*r{Ha4zbFO1AFq86wDq>3{J zG!QL=tAPbl|BOoT>Y|((oucnvV4zDlB=1j=B7hko?H7s_tIgnMHqH4K7`!jO{X(0l z3&q;E;9;NLo($d-c|&8@&RHFX-qIE)8WSPQlWW`(z-KsEn8mQ}cuG3x^XA83sl_1Q zb}A==8kPzk-m`=ZPTP6p+fn^$g}r`T)ApcR?Uyzs%);tIy2mG)-*GI{kV@n5VX-Z3 z;UtSHhk~DVuIX%!O$H=pA-xWOzUUBP5s;rf>qH}t;5QK{8Skf1R68DFz{UdK@d8U7 zohEcOIBYBk5qw5${4QJJK3lky5%tjLW70Qrrc$_p8I2Mm@X!}kA~KtxVGI^Z?w}S? ze?VIX^V56|itqa1FJ;K@zPkNTe3TsD8tQv*QGM>17X|iDofyBREEEZcqDz!_*)vurZ8R&B)t={nC(0TsQV3MQ08_O3}<)3ZJAdvKsSuk)1;(&PUswP;L z295?RJ%IRGYNSPDqOts1Kp)3DWF2SpF7ELu9fY-R{>st#@;_GaTErQvVvKmZVkqf6 zB7b<2Ka!iH;&#d)>4o!{)<-R`Nzsd%ewAp5q&4WL8v*IAfInFH?SxknJqOO!KHeL35 z5?k|F1!_S$5nxOR#LFW2RjP*XwZRdi(6@E{4ahE^Qo z0ev`DZOtV_-8myEL*^>+#97(}-1ezmx$DwGQPE`o>@Q5LWkm07l&n?PXT$pyx1^zu zMv0cXnktx@R;)TlsG(Ns^b8EP9(O1TEgX%q5NwD>qD_WpJZ>PHTsFHgTK3MVE$)=T zW{nlzI-TZ^2f1CMG@*a}4j$6myrdHRQ4;P`p<9{plfdS43;l{)yEl4v&grkZA3$&a z2>LAh95u_pwHr+fQOR?aVpoovs!!>a+$7tbiisqVLbE6i5~||UB#GZ)q>`al3-v`< zrLwMqCvhCNqq|EQGpnu7k0w!(czS^q{t_R_XB>3KEluDmi^cjt_r`nt# z^}k$l=MAd2&vg489-t9|8G@AGS;UpZ5C~7PGTHS>Wt}roN3lvw{0v_~g-4+))g}l>0gW&v zi^n?dpM-Dtq}(LA6=C@98W{|2XG*UY0hf7kJoiP_)d#sN zOjggTr`vS@Zu<_t&Hs`ZUCTP+%e_o)4mQfk>Op76$VXcdGd}XqiAs`?UW<%i#fZdn zDHOi}mpBz%BzcYC<1Sh=aE1B#X6EJvg@xxnWelc@Sa9qZ%CrEHYQvg!#itq<18+ zeb!yVJ0ETp@J0TcQd~L*ZkNO--!j@edxFJRA;b8 z`Dc)Z;hc#?rGSo>6ZAa9X>RP)2&HuFK-P zW`38ce!2N3)1}YBxF_k#`ex-mzJoRAq*=ng-HSiHz6Z;^yiE;9W%ctqY2(hv?K+mb z=q-5#df}wlf=0sYU73#T)p+%1`jCf454^9Ps>7i3&l3W;-*S|fwT@r*SE>e%apiEvnL%A<Bv$G)Mu;j|e7v0GFpP3kp-bYZ1rOHfm zA6VUMLb%*9o*X6?mJC6!L(U<=BTy5qlr2103iDoW29Og_0(9$!z+R(jatm}~zKd5q z5>%atm_1_&CE-Z#;9%`=GEhF0^1p%Zb@JuEOM2g4p8}ZZG+>6k^l38u#vy+O?&IR@ zToUr;y122?2h>$;3RJRAd?j3oB`jzbK}$OFuo<$B$^pHK6ZV3hRIuP;i_OJ_wZ`NZ_Y%y zp)93hp&G}ER?Giev*-2lp@`RqOb3%6U2CMt6C|bUFsS1R2EUda@-Au~9V1$>JQTFF z{K|04U$@S@3?#StaKP3`;2qvW;!2cQ>w?DXA0Bm=~ zs)g>`<94Cs?re|&FThdD5(hYU{hIx*6#-d_#s{7H^6^gJ^Q`2(Czj1aw`W%s3>cS| zCsLu9d>LAkG36|JQ`EKhy0{f-%xC0)b9`18X%xO zOV?1vST*@RT%)$H_<6ubIYOZIyzBKDpeL${4j%PUw6Kb0(Xn!e}JaJ7xM;>WZ~ z(tsaNhUcFW_Ag`cT7Lh4aBrgFrGWG%lo7+F5K&}Ei*Pd*C2*$2@@rX{-;bc zZ_wAsbu9|C%LS1s*KQ1*yx5)=%A%-+8haE~zfns~@eBXGS-ShL1Wpl0|C`35A*fHq zjKRNQhV!q2@6@pHNoreMifCN^$W4+!xL%Ot9Z!1^w`PKeJ?OqTX*9yYQRVIWthktU zn9pUMMfcOrE7Zr1W5f*iJA%Ra;Pw0RrfE*jhTR$4m@tlzCM`JF3-Y8 zy6+kOW96%0(@~Ou1Ix@=^po3z>Hdj*KIhHYQ%a{>LVrJ}8`2GvFKrDY!>&JxJ(S-6 z(xBZVs=p?f@vq%)sJB@^eDyA{R`~4g6IQX(ddCUWa$@g+Cr8bEi?R0xDSdBfRVk>; zy)m0lcVEo=Z70_%TN_;V+02|wx+@%{HGjE+5kQuiUrO$@g$bcAGQ0#%SF3=9pVADIc79 zUzpce@%qPv*)z%;-x&`7iqrfTSyghiq?miR)2Gaj{myF_9LT=xy2VG&RBjXO>SP}G zvvv$T-5b3n@(OL+bxOhwu8#AI`-+DLuMe|%KPV=Na$9_O|2q`l8BZ@f zkI;s^8A?e5SeDP?RIofxTi*{AtJWa2LwnilYju_2waXc`A3fxL*8e?iD5&Q7>kFs2 z&Q_FU^zh_RxMRO^bU1XFi(fx*BbRn-7jBlrS-%-mfT|BEO5b0A?BeQdCA#^1ozy+U zh}9@41(aa?1bUwWKi)OpOg5ylNbEf1>C22kW3zs!3RmZSU#u!pw%g`fAop-`UqITKAsuPx!jdCA&0V?UW8By=b`FYq-s1D{OS^ zy$L*d@p52dLh!J4!lLPBtftVfdFM3Tyrh%+R6PFTeIZ^ zrMiY<_BZF>ybDk3clw6>bzM%K0;PAO@jEvJkG6%rqNlm6Skk4xH^>eo?;Y41SlKFn zvU4~~Ebg&0deX@mihY>g=h3*_qM-29-LVxDl_%%e=jiL_9r*=KmvQWnqj_EYvqmG; zWZF;LJ~Qn1MjPkPZ|4VwjjK3(!tLkmZmG$CUResOSo6|fH=SgcU;7Om=zgek!9?>f z^D`N&b(y&RRC?Asv7zT-xv=K3BV46P`qssS2H$AqsH?u+VuAN!K?W)gKIu9G5beQigRZrv<%E9P;fceHaSGPP#Oy)GmV zh%#|L*XXt{i&dIj`MPi3_we}{0lSPotG8Uc^(M}T)ts0gGxVE z)U{@N2=ZxeZ!&&hyf!+(+@zn_^AHuKUb>)h)1|?k^7+awv>D?k zM+(ffz%MC&YBX7e&#b=bgEKqd{aIH@te|k(Xn9pn+!0BM5f1|ne+iU!*?^B0263JN)}fezH74I+9{C{Cdof&zPH5o!Vxzczf5G~~$FC~U9B%8- z%9OXc@INgVB1xmYNPb&8_O=$M*H6BN1VWbf&@nHAEl)3A2!n%WB6%*g68+3t(GJe$9_bkT2HxGSnH;QzTyQdysE+Xn7dSz>z z#@W;77E&=H)xGmN#YEn!RGq`oUgYAc_+Qiq zV!vE&=#1=Vs zNmwCr^Vjn8-Xj{AzIx}O(%4!R+CJRgF!GZDupqLhk0SZ-#CRTjGGy{SIRG@ zs=S&U8V5SBe)-fH4y@U+B|C8lM+fcVt}zVWxK)s&8K zm4u|%`LdZ!6vGwKY&N?C-FZG~ttV^76j};&#iG{ummb{H)#FUIm&N0&8%iqn|79jX zs2vE8R>EFjBjO_o!b}kUQwM0Mokp?8m)1==?1py5D#q9;Vy?+7Hno+~vD`WCjW)qO zV$1L0+Uf59(zI2u>0o2xb?s;Ib}7;N;^DU$+V~h5nRkhbZC5*QgsEpUueZJpyq*ne zjr%C$1iQ1@KYRaL1NxeIl2JtslPg!o{6PWrJDHyiFbm{;o5b;%j05xU(nTB(yUw{=NPB zTh!872uc8`6SQ3T*i7Uyg3ppu`;BBIh1KB-y4Mi`B7uuLGzc*db{>eu)#&&8LUti8vX_wjx+SGMu? zNkzAKjP_yO^~2Z;?xmUBuC6$!6pOZ9EhU+$C{bj9KK8QZRi(SPeQ9*%UWBadFmi2#)`NtBZE-;LpSC4n~bDBG(!-To{TseqW z@wa@(SANv(dXZB%{hiNm668K4?w9Rfj{Dlr_pwYo-Yx3;6!6Q>ClPy1Jm-~r@lKPp zt~7XVm%<;60d=yo?e3U3#W$nVgE!^!HG}1vN!qJR@@26yrQ#Vnu4^i zYxBV^#@)Ou-fx>3Tt^Of(CE(QW3Q$Hc$1W8x9d4InO|sJ=clEwqL&6 z=djw~SIU9kw=SBJ6-_X&U8oJVOG2PJO5?%`OjCxQ-ijmSF;L#CbA4x^rz zBYmI)I!?c4;V!1_re!2-l+D?As2DL(?Y>90&{%ZRq7U`AQ`!RIA~^fUD1jt#KILmj zPAot3T)(_VoKYc;B>6OxR$mHJDmfEnitDB=$=Tt-6(E zA#g@Nxf?%u{NBG*`=a$pjzQhS-ib6NQo3cq1=u#iEdc(w^~qJ88(l^ZdW>um>#H;e z<#>IZ`M8U7op}DIt<@Jak8JN3%pH{YWzkZ8dFFhYNwbLN&);hmyuQhMDx@<=m}rff z+^gyBJ}BfI$JGCQjWXi6qzojaS`c;5?yHskKKFi`2|w*PS@~AA4bR0ZipYr~RU8h6 zDU%o&vQdVbMc}#-KuE8U$n^fjQUb5ZYCs76TT6RI!ff-weWr?JZljlak`phB0;J9x z5TL$^&QL5SmHw9y{|9x4fm$nBb(yoA?#-xLl+%Wsf(WF@J{wJ8$wRKyVnj+fH|T#B z-@Ra;78hE(%i2S_@d=D}@z*df`>5gzIUs>oCwXVo8j*WtY&KjaLcO7@RVBG%fETs? z04KHrg zMWoED{R@3K97AQV z)@igFt>;=@kBjgMF2>iK|lCeb-Y`TYD?(D1%@2 z_#A~IWM}(1{{8J#%70Eqfv^BUq6s9BmxPcr)S>&mLYHv8htFQQ67*Nt&Yjfd4vbKg zh)fb9oNXg|`F7oBUbt1tFE=sUWD$;*)X62j3?gjw+3X#E`D*E$6FARtgtZlXVm2XSKJ zhj)9Il$<=Tn(;Po_QX<)&02|Li($q#7GZ)wXx_`F)2{0s99MxM^Eq|r>VhHm2@q9& z=T`Zt7kR@aB2|-ctcb+%SVLo*ua((5@5QP1fmb^MtdB6;AL*d~)GR-OqNDm6E-HECTQ@KF8xjeg$R zr#nU|_|;2E=G$NEBA+>=@0wJ?Ri1^he>E-mHF#Bk-t+88wqAd;x2`DqgBlA5Urse6 zkM|FG?5u8}RuZ)j`6!^=dBttnZVKv6?|v|*lE+Ey0ODVfYuTH0@$lT%XidERdTk2%$9)78}rh-!`|XpI%*ta`H-2jia| z*1uG@UqmiT|8rmlGzKs%&~+5e!17`V3!OY}jL0-E#5u^y%XHX>VH&T~9%2LI_k2D6 zS{t4lAOA}x$9Qg&nP=F^kYfJA zHSjrKFz_b^17$Oebo_F9D3SKYq5ZsnLio3?iptwz`r@7aedh?;D&zxCLBF%Er^*@J z-+D2XdlMSD0nCC7&pzrT9YYW@0sP=v!SjE3Nt?l*z;EVgBQ5eX- zhq3N;GC;|rk<2;(?FNux$IBM%I$dr1wKJ4Z`&sxPH|T7@dA>rs;<1y)IPT@uqVvQ{ zMcr|atK$WS$z0KvQd~7SN8lA`L)%$eoc$Ga)Wk`?^ZvL$5O;8La!Pgka@LJhZ!vMY z-XSccWN3>fvo0gk@%vE#E>|iMMdF1NY1iP(%` zvvUl2g0%U_PXOt%f2U)3az-K6B`Sgm#aQgBx609?|dTnk4M9vC8 z?O2g@Uja%8C}t>S@lRn6+Ki^MX*o<`YF*d&;poKUq9MPCG_vlZ?nOTW4p`oNhj>pz znR4@OL{kfl_Z~q&6oe}IqzyZg^X-NRgP1h+t3e`=I=?i%-R}-TVxuP%aAW+Y1aLU{-skaV|Bx0dnnnm@zcNo&VpF-y-lIDzdgivyC<(Y8PWMHa z8B!EVu8XeFz?=ki#s^-JhY?};E>EY4N)MrL%jCe4i^z@(d>4EJ^+?R`60RK@-j>=n zx5{*gv0~|adrAH)^?~Eo&poAgWmSnQ2Bzr}M@}-Q_|=rdp{r?E<xq%c-Hg06wWM^F2c6$ThnP>ElxBlBBxiy z_5=kBL2SWwB!@wJVotS6D;?4d>dQ-&XX38SClk2}}zZeH56--tNVG-OExuZI}9G zGcYjHZ4jSPCCFoDwYvWKGky9W(gmQz_zb-!o|A%X+DCGoZWr(rALirYax$y5r;ZeOX{&isv*$;TkhO)+I$@9>n!j4VD2g&~oC2}OXms}| zHmTPEIZrs4r7`qGK^_f=IJ|EW;b8fKtPr8QDx24C!5G}uR#P*kISNMt(o7v+em{B( zruDL~N_LB&i==jjS;1IC8%y*iW?Nw+Qi!jBNs;6K$VZ-*5Ko*A{X^NjLt*^f6em!v zCvzX9Jw^Dt28a+u!iDB}@BhecNzi z#kv$Kt5CmBz#(U&3Bl-PblC&2L)>BDXqHW&Ni!U!T7vKPFs_iOOI*?2Qiz3EC;}q{ z?Nknj=pse4>eeHMXNWJaIgBWAIPO7MmT*H0%l!FLjpmI9Mf&B@T-09f#QA%A3I;@v z_X)hO4E7==+JV^y1;vLxh*1!7poqv2I%PQO5s2eRZl3Uk4l!`_6lZ%y%Cmk9yq3Xp zK6@XYiQawo@!8v4ydC+C@IqxR+f`s3iJ->GDNmdF3&ZSawlpz@PNE&;4bB`RMVYF; z+#UhM@Z*x6=WbGoI?XHRA`A9)b8{KpdMLTkm9*#UKvU4gwjd46hnRRysCv0^q7AWD z_8HZ$Sf$1zOwFc72s4KKR{)V9J(84YV8x>lX+a9WM|FZvu%==d9;F`SZAP?m65qt1 z8s6z13!!a(qFzei1=UDN^^88j=iQ{)WK|rf{@^nxJk&2`q8QtAcqPaj2E2D)? zz|F;qQf{k%t0iz$QBgU6;*3euU+TLEzeE964 z+}F}H_C^vMJid@166wi&SvoL@??R4w6iM9^xDJhaWYeE|ygv9tQAwnY&%S_#1;=GF zNkl_#%rg()rkiM_z{Uf_K$?T*wK6+9k!>TmM4p$lW7ssI^>5RPU9xzpZAxA*<7sAe zM**=06rxeW_ZX?%KcX-Ff8UYD>7*fYjF`O+PhjEDN2RG9gauRo>EmcME4k5NEJSME z)bCvoK1;o$1k}o?4v=rpdQU06C8e@a%g2F6lNbHnlOm=TUIU|FDCEhW2*Xw&tMZ0t z%_!%hpqmhw%(bTEPu{`N2=*E)br+-DUIHIN=!?Wl}#05E8p4V#w0=I z_$6{H+SG}Og!-%}GuRVr#>vTa-gKI6KR=c~(zsqd7}}(}M=-3V1ZFnu3P%A~HbQfv||rru6yYJ;e>FJkmH~?P&`T$2gWl{q^j7oY|0{a`vXvJv>!dA zX4bHEBoR-sw_qr(08Ka5d8_D4@PH`X6gyf+>LEVLIAR=3InJvG;6gzP_mr>m({M$q zSo();ZSm97SVh(b!GHwfrx!S@LCMr`U*9W<tS-?(?VHGDkHXCMvdFgo6gdmw2dult0C9bh3@shf21wyQo^)G(FvuDPwbDLXX4a4wXX~ zsRF_LYGqBXvGln#(dIdeiPzpJ*4=mi-)%!FA3NHffc=K)8Jx4cIpyd~OK;+GS|zzu zYNrBCT8Y4&LFy1+%IFpnO~Ov!wpRt!N`!#D0$yr^$>xN(U&cq*KJAvSm4KF1CSA8B zGv~cgBcUGCWke|%^Nv9tEo;N_30-JccxAbfdP^EZI@Ff;B+;F{)M|7L490JSjJjb@ zr0KY4T<^!aP7=*j6fD!4SxKcXKf3XbzP@C9B6adR@=^tVxyNP7Q2ey^qQO~oRb);CD*6U?M|>Kq}dSNNw$u(|Z>rhIB>`=-EJT$!dbJ%2hI z5A=ild72X9~H8BI~cT4>P2h zV$+E$1P4unQiEzfSz@Ea`u|2eEfh6Y@EDqXYDSQZMs6u0Y;Ep&-RJ7UF|y%QPX?vE zx7Q)+^glAm`GEu!x_>w{`SPCXmrL(0>`{qkcWGQkCr(ZK|PRe@tT)| zx$2_i9-O6&^SX(W;^^4b*L#h>|A6s*?CcmV`F7VS_RwN|( zWsQmrPSv?L2sLjJ75lt2Y?daq2vAy)*-_Y$U(Lrik8Te+$<-T~vGS~2^2Qo0vTc8! z?EhJ=!NT$k9@@|UrVQ@2ohC&P&zuEK)gaDjZ%v+seplk0ol>{BCgUp)x-lohSI{&L z$V+J&n?s4q9vCT-%A+Ez$A%HlW>!7HZkm# zITumc8V)tFC@`hIsoxp?1k_HUz5Ry$f&b0Ld-nS9fR{A9KhOga^MR^VuFaWC793WE zSa)^~dpuXA`JPNL=hHp_bJmk_{51XvD@7z2E@jZU7MzYjqWF}lr-aHD1HD$>(oF7s z#}GO;eLA;?Y)%9z0|usip3&$#)V@GT#EdD!#qQutN!f}xLk)+D;LO>N&)lxq3e~63 zvkcKux%&*6FrbNU>JY-)Cp%IppkaS@jS6ScnFg{3e?ts2})%wgvF4->nAEO7Bh z3=U&_9V7!CxJPb?$f=*Re%u$^f4^b!+k>5*!&kfS)Hu-=xk``V?H%5)yd0lb>w0$& zv^V7%ri^-m{{00Pc|YN%y13WF`e-V)9H)&&zeU%l8D@Os6VTo=e5zvpcrvm9+aIXa z*_-5vOQ@5sh^eBVI?rBcWZo?pj1Sl4Q4;GE=+YgWfLNJUi?-&OJVF)hf)GK`xfAFp zZ10@Q6;Bq5eVve{s!vz19VUiHMVN?A(;)Sh+}G_n8+3ddKjS)CLE7nsOy&^$1=Nfn zDoIlP+o0PzZ@Oe&^Ekqyfk7UhogvmMuB5(Nzl^MC&j1I(S~Eknr$j~Pi}x9d@6s$k zW*D+}Qo1ydCloCe_GMw&N5aw#BuKAJH7r2~`MwP3ykBBOGvOar$59QkV;KxT;z*NN z5fv9zH};Z!H<~s!i~A{18{ZmXX8RyKIJlUSpEpFp#w5G{3qu1727DLmn&{iEr!bM> zl4u2SaZAsW^qe;0xVv^x*hi7GzHS;)3{$#ak$2-kg%*WP_BODrh9Xi`O0gE>&34yr zeY{Isjmm^AC9A_C++z3T(p$(9_>5{5(0Q7@;c|k++!9b^iZjNT$g@3(!dw4>!KzLJ zRj)ss%r8$ zEBh2BLP3V=`QEn*BGFG9))MQKj+nU8E{cktGWeSj#C9?a6sDL&jc>`Ypmcjk5Q)M=3T-c;)Kw??xQtNOd?@sWCWr z2vC1?P{E)==2r{m-}dPtOXW9)-9qArJ4|IB;8s&g>B0P&Hs33Rir>d6t)N`&%#Zqx zJSugW$Q9(?*nH9^$+hoHym+BTyD9^%vTs67=_z>|+hhzFnWb zH(^jE4BBR_Z;v#J2Zv=zCGixos8)yqOQBwaZ?ncN?>zL~ND9->>cayRvlfnFyNu+8 zwUlL4N+cs!YX-uyza58j2p3XdRlVNfXF0)VJ?d&_FpIh~HfQj??d4qbZRA@u zH1jfa+F_znV4BoTuYbb!Lkwv{mBZPJ!#@_k=rns&6`ds&h7x}6rAClAyM;M{Z8R!D zrp&#B%P+LOz|mlB8yw+2oAVDXH>|Se{NLI!%b3la^GqlLA4Bc~@6}VnWAOU*;MECX zzZ5aU3=_x^^-E{GLfR1wCaJfdlBF#wcIGPTDHZb9JXjkXHm5fm?d0+>o|JeB;?^sI z;iZCIq3Y}e;R=uUG-T$brVEWanMeeYMd9daH9CTLo$B81=HIgA)RwOA!Z@+1_YIE8?igr3? zYvYSIoM%L3`u$Q}|Ae09;{27y0n+VDeEhU*6)Pz(je+(rJvz+uy8w;YJ&EA z1b>--Ms$0Me&^L7gRbq}E&n>Tohy>RM?IeBwGrE_;E_TUn$5Jrd%_y5jlRY9jZjxL z!QT$kCm76*-06L9BSO><0#c>tY@|sTMvPzT*Y`EN_ibd`v$i&)qS1H9uTi!NWzP*V zti!5j9s3?ZVUm1r=pUvn*E5^8$Kr;ubG?#SD64`5VW*40riTYG{#Qu5$OzqJYMkXU zI8uYE{|%`m9=*_Qitp=&EGpn+0{p<$$7Lu@C8DJm*~5VkUlZP=Xob z5{&$oTbU-b6HeIC{8?t~wJ>u_DTO+WPEnFW ztF_neu~u)p>vFAF`6U6AY>5u37lv(P2ar}thzJJ)r*kC39(HnJ7_@y%D7JD5$Gj7* z3%UQz5%!B=jeU=(<);koflhH9qnfHmm2_3JmXS;`MaZ>9lPNQ6fNqc7e5^mxukT1b zIXeyC`d|H^E5KWS0=!jKH!J`Jak}}{ctg1q7gZ6lG_+#S6n!29r9_*=D8}ikQUIg# zP4*K)m=y1sMW#?!#pKf1-A?s^j6D?Nnjm7uH(us9U^t>Ps>QM>l~}ujy_^^tSHdmi zLggjFDf=jJ6->d}Pq6*;itSi~8d3B$Fz&Z& z5($VD^`XrE!;2k7Hg0=v$Fy{>zvUbg)-noQTFKYNq~?D3!vnsi%^hezUzcJ4C6*!y zXoK|~zg089FAmzNgHp?M7W?X6$oeQbsx6h1dn7kh4(MKX$foD=kOfw z9-~#rD6yhFVB86WveEWAH3T;e8!Y%33;UC;1VN=YwXp`bwuMb?LgVDnzIerXVd3un zzJ<9tx5lQ8(tZrj8t+L>&x+CBM9w;>(m)6~h+_GIC+P9^?&{kr`SF2%ohc%hF)m3r zpx14K+&cu`T5gQ8%>s2Hm&XfK+()m96qbymm`c=%kyf<2xb0VD<^ewEZbDt+^3UW0 zi?Qbzw}0#s6LULf+Pa2$vayLkI-LdeA94{2xnagN4^`gQpI+(4JV-oUYHPG;7MzH; zJ+^q;LPIA9Y^4W)4&QZQjS3Rd!Bo4MWE(O|An?cFzESm$Tla=LdZM zek||pHxErXWUW^cELS5{sX zX_*c2@lG-Di#YSS-VDQK*2UY@?iq-Tj0E*7=}!QT@(c3KJAk?h0zRUuXZMw4)wLzQ zMW!Qst*&3p965F>JV41tccm-kh6i~z; z|2XS$qFpdbJmySMX6MZTzLjIu{ebPLudnY1aQ_dpV?67C_IKL$9fE^`3^4=$!+fO< z1iqenbUb#RlEjT6752OP)*QEtQDz7M;s1LL0ihDbx~7UFP?jDq2wS9-bvq?Z`Y{?p zmt3}QW5V_ZROnxT9ieI9+gMX`2K;E*Mtn4?^&fyiit!UC&rm17HkWnX&lc1Cg^+)Y@_)eueZ;AKvM6pj2iy~Kzwm1DnlpZ6T{9xq2JV$g0^d02JptR|0$rIg!V(?F|2@LNlSc?$ z0p z|Mn8N7*Z6G&-~El`&s}D3RAL9WKzteZ%xang+cT$UwZ+@gO(Ae*QI6hn=J6S*Faf= zwV8#E4!E+w#FOM{5|U5WaEHa?MT}+%;C_U<*#NW%ZgK|zm4eP9+6|8*LOL$Y90UAQs zK5dSV*pf=y?bSE$A?g$rTtV3#@GkjWQ`t)Vhg5YUT@boMoZy9wN=i}^y3~J~4u961 zX-pt7aH&}F3_?v3gt~l>PAO%${O+5QQ`2pK0{(@ud!LEy2gA+v&c3g#Scb=h5o=E=)Po@0`puDZPO62e11RP zeV@W&O*0x;uQeG+f0F6_^C)B!8HXaAQ3j6=caZrS^vf}z5=3JnKfR)~6yNa%)-Ohk zLC-S=b47_jF^dc>RzLBgW4%)(JDG4T%XeT{#3L8ZHwxy`E^IinU>-I#n|yhVHCrE< zjbMZllS;V4B|<0`Ot;n(d`dD}qylOz>d|nw6-Q~4u3)U6asKx`vu0oh2TSPi{TSu& zzC5S}Q0GUVw3dR}twq7w-!Kv~1&vb~%qN#9chKtu5Yk-;J1_&c-pS>w#K3*6` z!G81owTk;N0z31E6Zh|Vi|iAkWyC9|=4xwNu>PFEYQ1m)R$4Wo;fR7sMjsPOG35Q; z8V+?c5K25VCD=EU)oRQo>#g8}MtK%eFBJZhw?*|etE(tHyXxTJj&WTl<`7H-mH9h+ zW%fS!r@l{RL!j$(hd#e#o#!c1SF%oAn)0gYa-`jDdGGhOK7mb^ybjNQ+iHakNtsQ{ zFr+mw;@M@%DsbEh-9(n`ChMb5${@$)EU!|i8&B)hBKa80C&Y`UsGx`JOEWEb7( z^{>-a%CF(&6{>kLZ&l1pafyi3)Al;M@NM!ZgziQ!8j~1Z%7J%zZ6w5O7H=W7QFX6u z{C~~r|8x1Jf~~?O=f+qBQ{y!3EBn7{0tz5RC;E)nN#GnO zLjt{+Q4_?7!D=Mm^Lk{;c za?lkso2ycIW;L99fWsp_o}1pzNB82z$m~KpI^tY_qT5ngGO>cztce?{8%$RLTntPG zs#{Kc}Lp%BsoKc8f5E~3R1qSjR&=jo^H|iiod*Gq}js_!h*CS+nk7#DdO#o_Jb8O)L7hX@)0?&Xb}~H}{eh+P4u= z!p&4FOpHUYvem=B1=A{LE9BRVq1Xivhms7~6ffc-G_|WE)%H^r{*&tYw-xI!&Tz1? zsclx(S-{TMOO5tS)$OZP<)s5^20HnUe~|N+@Ij`Ln=r%d}wh5QW-La&~>@8z5P9~p@6?F z-DraQ%7$rCErlsY+ndXBk__i!mOy9<#t^1o{tFqtlNm?BGv!( zf&YGZwGaTf{J&rRH*EXUwEuTBST}4t#h>zofA*9NX_20YvF+kT@a32&IM_R@Xpn&G z7%7pSaw+lB;D39R>=4m%CH~J*Mnnw|zZ2BQN{7Qm>M&|THLv`Y0!7HQ%S*i=yzoCS zP8=u{Jc6i#X)qToxj`x@9i@>StnMW|P&;_?$+f6=&Aw$NXM`+S|KM|wyX5ji&j)lN zleQ=2M<{Mb9#IT|B3_WR0j_50$pzJr^x~0ZCvvr%+{h8OeBj?-RErmp)n1}8qJcZE zW&jBnh@u}S;-V!(9YO&P)Q>Zp1bP&l5$eLi(ms*Kzsf%)&w*%>Uq>U&S3>Tbh*@h2 z;RRa95XiR!X&{0=5uFU$;JTm2Fqq&@S4&;DDnuy;MeFe2M+6#aC1-b0k&*c9c&`0! z>Xm6g5$LE;8Pckg7hr-l_z;!=UNb{5`z??U? z4eTf2Mq?mSyimY4zJ;v7IS@%Ii?$MViufZ~JVk9Ug+kTnpJby5teeFKGR*opg+p&? z;_cpGqGE{0AjXZU5tkkPSpvah4@q~b=~oC}&hs(OrW#yH-+l0%({n~aUg6vU_*pn9 zq^zWvNgV_9`y%Wp-LS3Tb|sbx!&BX4a>dCHHmp^(%z(u`lC_T6$_18(StKUH!miiYYYqbm+tWW`KHXCVi%`EW~? z5VhJ7&>6fK=lkvq=;!XhGK*QA}INU zT6o=L8O+DmeIoEl{XX1L>7!?(?3Ioqz?oV4$kAu#pTP$?xnMyj?_s#A%W=%Zy$E<% z=e14iN8q^%Q9LJBevym{8KvPbAdOwdr;Z02`v{S>p-gwXHzN}-So<~_-Q?P|!y;qw zN~~BhpOyBL919{BDh8hfnIpnMzhY*Snyh3lA7bkJfF#O%=;gQl1f*vy_ejOQW4A4b zZ`*iohX3g0If{TGNW}Rr610#CF;u@w5@x*|Z4k3f8@^6{t^SF6>C|FR<(U-c9CcWS zlo?m&6g7IdCUmRgcB7d(GkQ1@w7Fu!4IydixCFM!I<2bA4p=tPgdxLES*L^wH2nL2 zCG%{qEHDQvOzQ+kP1rEUD}>?{q;mdecqEHxIo0}W(}p7S&(4?Ca>*e)tr2crnYZ}o z)SwA@!`IFwtl$k*Vew?ON0J%E3AT~pDp`%1&zl&nqDy0T5lERW3R58RE1Wz~tKnmc zx;GIbk!XfQ8~lvu{kN^PA!)<3Jjr2k{^P2WWsW|(kgf>4hKL>deHEjI)J%zXBCuu5 z;x^!oMEL{Vyuz_yivV-sbUdrUKSf4=cBoGX!N0FEFJF%BJHI4%mQMbw zQA7rZ02?IQA*`aWPl5n%=bSFbMN=_liS6>t1^TrrEJjX1b4dx^xy?rcuzgVRK{ceR zO!8|#i&^rc*|C~qKysxSvZ*@yySvPWiFkTSYHQQOz{`I`%$V@_EQHAUGq9uvd>|$X zk%>$lg!VcJ@PG?7B8Hz~cKkl|CM6?-(zQq*0bA?eJy|VK#N1p;Nh#LO&xYh^ks{_H zoEE#RnjaFXb3%5IRUd|SgaX(<(s^2YrY}*EJZrOL<>kvS=YbO2e{1ccM1ZImDE>@W zSd8uevV}~`9g^^;MYkXlmb^A-*wWxwl`i68go&#qeR_V=UY8XSN=5qj9m1Z7hYVZr z&2Mz^Gx!S_{h3999+aT#*4{|$g03qOGo1NR6(6Q&iKlDSs!JXGUp5--v?5r6qtVEr zDDyaO$<#5v>W`@Yv*64N`kMnXzbZu4iNHWWfWFd-O!&Dn@+hmkaJ)}}cbNZo!KH%I zzp#`-oj)V@sl9MF)$;wrBxoVa^;FX340;B}yb1@@GVcR*md*R%Y5_TRGxwxZWaKcC+wwi9{9 zWB3eA)5$u>(#7b%%lv2Ql8<9ckM?bvOMjDh0GK9!Z`2u0W{Wldiey20^UvYh0}?Qj z=t!D@BLypLMWFh1$3s-YPo2L@X8^*iQoug{)nZpPNx{^tk^(v8Bmt>_OWv(s0h803 z!cc=J4K2x}JiYrk{eN!Nf{3@gJiIUP=5WWHvKch+3OMXwx+~^EgVCoh-$`Kl8nw0z zr2q*GwSQlRmA(X~$;p>yQpBgjRPeE-LQd)xS87(0WJw4u1IqU93~VZTf7cxXC6Vp0zdCF2YtWty@_+gv@S5gq zawm-BdM^+N8}+-hW#qejUj9YWF>`&M-@lN#^#9Y#YZ^X_!JmCB>bY}RA{U4#FNIV| zQV~H>5mYz6`FhOY*(<$(%?G(}t|UZhQ1yOT7hI#lMr*SN(E~k4y|D0S)2z^%U;KA} zMSbx%~vLnLP&&P~s8Pw*MNJ=I|kE z7+TrCerLnm&*QiqecJNFmrt)n|Liarm>MP?{s{}m8deyEs9=){7hyn$1b&4;K~NJ6 z`(84`A=bUusUKD1pU*GdPISDx`i7B@8lgR$ll8r>(a>daA>No6|uQ^qxfsf^oO?KA>>%ZLojZbABGEzzAm=67Fl( zQV#ou$pA#tmF!sWt?KgUq)W0#(NIp)N~<)Ft^I+Ne}-}cCzQ=bZkM=x^E*h{V+l~s z=5(bld<#Yekhct~ps^2C^S9F1II5Z{jb&DlV=(^C@=6Uok$?LwjHH$1Af(eQ2_a>NroWeK zn?Q59yd15t|HV&I1A~-!uHdxVma}u&s2fipA5ovewW;p_L?mMUHU_liDeGHqWB7`N zgz7^v=6_9>&m=o|ukPITs{u&kzr_|(wLGblc{DCZ4h9?uUU$;>Bplo<-1d2Whc2Vm z+AW)f+A5jfJp8!GiNkjrg^?E8Kt=m~nirWzO_)4enE4_~!)^Fj9r96bT;|ge9e&r| zbzV7ZCf|RmyWXR`i(B`NBIp)K`I6Q4`18Bwii?E*K(V=W`b0)f!uvG)~>Ihx7_ zHJj59-rM>*CkCHt+xJ?2whqgu<5a(>)Wm}OvwU7tcXoE)Zq5E_Y#;^;t?1JjCI{pP zC!t}*cC+g|vpf5TyV6rM1sD5jCJw*7yuaQ$Pm;o@3^M*5#G4{6{6DqYB~< zptyw8=S);esvp2C58&_5lO^YlhpTOjP2sx}4kAkFSi#Flam9_+oE(8ozNV4Sn+b&W z4=3}-S8d}Zc-VoFKMRiqcHG_;*MD>oPiGlhvPyM#sASZY#S>iZVt9vbhHYj(k`ZG@ zIOBV`guit9q+o@=oa_5~r}OKnvMauCG0VZCvOuveR_m0`G%GE6i6K8}`%{%^)+>ww zj@HFNwel!UY4u*j^(WHD8i&=HT+!io>o*TSUL}L9t4>3rI_%e7Q(mhomh&GU*ta>( z7aG4$=)QIjEK+p46{mHu``~4B8yR9l!Ijw9hy4jnHqMAuz6ez3yI5x(w!1ggFm61D zRrSfN+awEVuV9EjZ?IM+u}smSE)Ymwfvk^c`><3~y`C_o9*{McB22ebRJ7lJ zpGR{x8Fo#OS-7SqsH%NC%iFCuF1ZgCtyHkGGV()!F`+4ry|95{s^8%L4teI=Bm^shm_@0@W zpDe~<{CXy;3~eOXvGQ*{zh9Ji*e1K8?lT_OyDoPTeEYi3YWg#+MpwXccEvof@oY(3 z{@3{$BCU{XXD?S}=PSX)1E(v1?$ZA5eYB*FmH<6s8w2IvgocHVnbf*%4d0EK!%@v^>!K=83iVc2v{rcQI%kOZF zx%*zuXT?<4BQAcK&t65{dAfOPp)n8wxN!tM7z?&`h|0cL_tz%#eRAsL^Df7kI~Prc zvks=%499pJ#T_fZ141R8-=Fxtk)qs&YG3^r=vvRE#bPU&%P zF>)*+fX5NWEvq$9*ae)V9uvtRUXD2P8%UiGKcfGcIh(3IpVsRNy58)waCCCd%ZkL?Fbjvn97G{+ z$m(L+#SnSPy(^HJ+zH$ttbDLi#6r{C(@ z@6J@4(oVzAx5V5s?wz-A-ZNgEhmrX->7SNAH`)i5Bo!d$I#_r6C zf3pWxwW)9kpu|~MlrhNmlHQ>2AO1ervGchd+&*t4iuHaU?(ySbr^g0I&F=CtGY5gR z_T9$2>)$vodWoc-Et##$;26V==2P>frxLV`9lbn%xV#>}tla0bFTr!!KiZz@JQBJ| z)=WCOwIAQ);fh3ue^V&r`s-@!HWT0Gc7H&zhGy3vVwBxvc@+E(b+zK(@#HoWz?-c} zhQR~%3$hxK*dtYNG)~Hz;EeN@RYh#%2lp12lJ5g)F>k~MKl`QrIl&OReN6*g&Pn|i zAh&+z-VE&xz_g72N~Zzz1yf^Vdt|DVI_tRd{M~jXpX=T>7;XoJet#6_xvQ@<6XbV7qzW2_u8Ga zqXTk`P3EP__F(;={`dhx*CG8j^U?ODt{s(+H8-ZQdQEbTBer=g+?GE$;ISn)Bkm5R zfZYtClt_xp3BTJ&5yEmAp7~8`l)t?25o3oJ5;zZG8}!e(-B}_S zu2h=4sXG|BZO8DbTZ^9)O!K+2Y&Few^Uz zw9_36BL~}KqujR6x(j#GU;1<7;O9vR&yE_N+HK2s*IiEi5`2+$qkLoi+_jpO>s=1% zPWP|BzPM-$+O!&jm%fr+i4%a>bs-tuuf$46;;EFjv3FbA;n^9UZ$=DHeX%UX=R}Q) z|Jv)%-0YiRm+d(^M3Q8x`aP?G)pD_K?@$ZUKK%PZ$O_%>bSSj60lMOVv?H9@(~;V80(h}5+B?x0xo@{8Ne-Da(%eUBcQcJGm zu?ov&(BFZ;GG|*QpvX-%iJM5U%2|>r!~4f~vDp$WW0;e@^xcZ#N2UO_!F+C>L7N79Q?`6P&5wo-IrRr6kdqvaeVV^3SQ(V$nNMd$W{z~|<}`D{}Z2`*df z*xm7r=z1BsdO6+kz0=fFh5&clo??CNH7A?cTGw~eBJ{ila%z`EX!bKYJ!Dx8{7K1m zTgb8$ea9uypPhE~bSHJI@RR}YIp@Pk3QB8GT!5Biq1Jzo!D)8EIY=Nb7=%Cv6#=u2 z#(^Tu>E}zkbir^reKJmUq%fr4$PeY(@6CIIs3*%Gh5uVB;ace>H-tz#X>mnHc*$m*(pk!%=D9{#pk zb)yWynRvQPwq9GPHGV-qY#_YSq5ZYVWSFMy-7$W*hy8W9rSxPqhfMQ_2=oY!Ct+xs z1ZA4DD`an1KC_c_XLiMHDk&Bs8wGz&J7zf-IKTXEN;t?oPg?#=b zu^7sTa^5xH9uHy&UM>NMg5~U2^Tx(TpqtnNIve{oSVX7>8X3fl`8dy+YwA7nsE&839&fUR9?{xM!Le#)B#F5BbX6MO zFdnCeYTuD~U3^0CMpx}=)yAs|hYtH|n}ycYR__gxR_j=6Mv3diMyfwLs;qyC-n&Y1 ziu9-VqCd}RuCwszl{oQuf88B!iauw$Bik|F5`g;a%KUd==QWneC*eICZuM)Sm2G7k!IicnhI2=bsv*ijeifs-`!jY9rsbWCC0{P&9!SMOJbUL?r8=zLEe9ShepFVJWCCCqO_L= zA>H^r>8IzLXmp}wQxM4UT%%XoF$}QNYs$kSETTjvho8~W8aY;JKI+z@J#Vq^XdvUz zYrJZ14E%0&|7-t{qal3YkyxG|dKsISvx}aDu8axqx^eaMfv%M&lYw+*qa>v>1s|y; zclngxz@|wPZj~|?Z%zOqF}#l`wT$btUjtp<&uK&87kttP+IP8IGpp|CIS9H<-DTFx%E$_x8z)ud?#s%TE)@TW%h9j4l%pU z4eC(L3RTMtaYB#*H0Ju63cGNeupH||S?S?*D-%9~y~e=JyI^={o^MQHwlqRX{_~>p z@u#wu3&-=YzqPS^`G=AbST6ulNA=uWgL02YVF{%FcrQ3Dm6eS0^_t^ zc`tW=K5`bR`aUp&C>&*x+vRfIl08!OdEP@IfuYy0a|%s-OBtXco^)SB?i#TQJgn^q zRma@qW0W1p?U|O z6vGlQTcA~@`W^&GtW1d&(^>di)0;Akx3%^@+5_YpSy!{;H_#x{ zJjdidJ)L*u$=pgfiL^CR#?GZBJJPX`S-{R=aP>Bmi>pu11^@**zLjX@&Fqn8L}qmXTS z@bZl{J%8EVZD6#yR=4mNiEbd9czU~h=W2m2E?uas-h zKa_mwLIme_G=@$OO@n^N(xX$>NN@K9;&Du~Vjj4R-4F6N#|lj-H?&Lr#WYzD_MyLy z^!M>N!R}R!ed^%QKane|iyFF=tr*O>6IE>mSF45`Xhy+2}hl;9+}ly#6HpDa9ensWLVZ9~eQ-vfI$+4lGE{M$-X z788W!bj3JG@PE!Ke8B&lfT&s2QvqHP$3m<(%}jrQ=yXKjLdOYi=trJ(lXKaHI0llo z8TGqwj$hcB05A_0k#rOd?g@4ePM2Yef0kBJ(jMY4f zf;P>r15}!rL{q+4FO-Q{Z#TLDlH6k3-@knB@1Pd|1+>|A4I#INdqsB?tCq;|A9aB` z0getz%JT@AW*Xt|9mm90dp`lp#h&HR5vUQv#Ka_FQ_f-3496`{pj(CePp_SWLy20G_Gl!WyMR)~wa!V?2>4|n z;~zSX~5#*<6{#ju7j4>X>w*{D<=bDTfkAGjNrj%&}h^81Plo_|KQo60cQ}_0)jip z+q3|5UL+rWI?C`*uIDX(E8J;bneyM9?Qj9***t@-Fb%~%Q0cG2Q@CXO^f{j~1#l=g zpn3aqoY-3VEFj+#EORC!84ImGKAvf9Hi*=;klU-18=&x?x9zDf(fs}v4PU~X_J9?1g{8BD?FLC6R{dpE%2oyi@s2r_J6 zm3oEld#|&fYTDMdC&PRa6AMy2-@HJ#@gye@W`GTLtC089-8g zu&ZP`KhO4)Dgge00yDwu0K?)F&{KZs`-aLTW@MnFL}L?lNRZmYH>1uj3G96x*I4_F z8$*m`+~x~=IqC$!;zkyCz`@WCe^10_;`@2HPQMf30~fz6kFcmVX2$ol2|EQ#D^D}H zls~z#avi}jl{!%w@}J093<8mFLG<4XOc_9pYJk1>Yz-2y9h2RC3=-0GZwV=!Z9e~}(7zH9=#7D#lXA(skX-{2oJ*CPiO z5jYEQc-DO85ZGvYgc0%LIg#RbVBhe|Z~(TG@u5qT&g{sp#+kU#&Rka%^!q^mo0Wy9X3t*6`OvyDPf~f~Au_mY) z9(A@(w}p~?En|1Hin)X6Aq-Ov$_cMD-Vtw={8i<~0N#^M?Y$~kkzT_yz(tthkaYnO zzAt9y_w;2x5*CGY;BZR1!e*nU@j%;Iz+*qb5yd$T0S=&#(l00=ZOM#OpXWiOSQ5{b z)zq-KZ8I~7u@vJPOr%Ff#BIlj>zt4V!qe4GhN!$}UN`5F8p55Y0r1Hri@+vi_9D<| zQo>oLIcbi~N3SP{7Kbp7m_|}v$Nur4#vu!$GO;N6s(3C1Gn*+qRur&MG3Yd4I(ue34)kO{F7q?*APc<#T{4cWWi zspv%vhgq-RSBZGul?P(n;jDI{Thyr9m^9IaAc(&(c&Ff3^fl|B`;Gc=LjPtz0c9svlAsj1 zbG@xoJ>ffCWM4K!_)9@bmlhv0VNv+;PT3-upzOEm>rV5H~Pd0)Py^e@l^JNu*gUF%V|4 z2WI(Y;@dAv3UnLN8Errx+d`c9cUq=80;?QRJrp!C)xY3K`n(o?+y%&IQrP}u$eI-HX?CZg*^ z91ezt^c@Z#@^f6P8T|V(+LIzz-wfoZK{}Ja*oT%`SLO?|8T6t=m?d$2P|)*dxzodr zqAi9ES}x)qS3#4n#07@NN-5eEnY_Xqz$Oh7LWev{B4(*FRkMBmG-H{q%pfMf^N|2j zLuOj)gkJYr@Yep}a%WM>k?3)jb3;|T{I;9N3aVieh!E<0=t z3KzOR622-X&9gI+C-yeWKDVpnt_Bm2-dN?^DSPemRT_4XDSAJTL2?gVUfN%82{&8Z zwqc3(M-n8Rel@oEf--dU^aW~Pm!zI^@zZ@rh4B}1tsmSu)*5A*lPR(G$vPZTVY*J; zVzMbME=}0B96%&%(aBTdOO47!?dv$RVb4c)Y)TVms12J;Pyox@pE5+7<(?&ad;zYJ zJ#SI47I|>{)48)PARf^N4z_3xMc_lJL#@GDOB>s9%^7=BTc2Bm&XM(y%GPw((eEave z8$$dZTx^~whPSdh^My~l#Mbq~qL$IzVNWz@31O2wSw~EDdpn&?=uD!7&E4D~rSV(% ze#sw+{6Y_9nl&UGu6`$;v=ND|f+CC(D=f+1MhewUwxW-WbWvWK7|3)+VBkovAy@<- zcIlAa5ahzMa(oL#4nX;sWH1ncZ^EBmjFnzo^)--{uZkHnWEA#ua`aCbone``MMXFn z{w!r2>h^r?5qu^K>C9t$J31Z0H8C8rhWtggPdJ<;y7maA6eYe98RtKae2l1`{4RBW zO$Nzaw5BxFoSt%_NFN6lGruT=b+oNg zm&wc=I_}3g@69IT!y>W5TcA^#!>NeT8n4$A=-9!k(bL!~^OcLw!_5oI6=dIHO`@=P z_?ruxoY63id)TL*ceh)M(VE7PFX-w#WClO4V3 zo|JKf7yQ)}e}{)Mo$20e7%3{Ga`P=A<`G9<`bt;W<%b-}Q(cL4miA3^N00e9o)mG> z6^{7;joU8@RJ=Jd?>zQu)B5|i3^6IZ3INoJH`IYmT<7mfx_*J6{3#7(ic}Ejhh|1_ zSZo`_N);NJ`V&MQaWm^|GV5@JW3G$=#1PZ(b!0;o#+rYkxfa6T50XOIl5@XMb(Ini zDrubh-L|jw^6!~eTOCGbWqPpM22jW5q0->a4(Id zMbb4V=YIY?bJBqnFwWjzp@4eW@x^Ykfd1?SN%e%fJ}DOAq9_Er&9Rbq#Lt&%&UJdzpTIG zl36>dfxzHAFo{tT#Vm-1Y(Q#5*eBG6Y0cLxX>H|HzkpfgNmW3PQzGXqneYfQW{nrJ zPA8+uEb6We4(-5dUz6O*euC2)!h~u|EhdG==fEH=KPnp>L8bBSmYKA!9Mt|Y)Y)ey+@5<$aCXjZyn=t{NuT3`fKBL6UHUQ* z?^(sv$yWfiKEF{Cz2n4C&6!-3a(iDL4dYXKDG zQc9|0Nik*o6pQfj}3$(3pYf8?j@k4k6s$Nc6b+ohf$ zI*^FsuVXMgWBw!KS82L@$0yUt-Nc3AERp+snt;NYxxU>dlY?x(MB}W+=;QIpJfY73 zg!l>P`VvF}o(AnkbD+J`S`=q}1CpaLF1*m6k{rsyVn-X2N?euTS4JIeD~X5|7(A!p z!0huTNWUKjPk+A)K8g0HOiZ$le)p0$$`Eb{OeqRWt+P>FNJJrVtP*E_y9@#u$`Q4J z(vSg;T5OdsssjSW>r`rbmEhmBGF6qMU!t+hzX^#?Ei@7@wCEU^3@V{=BAfJ+a552^ zF<%A4%);Yyu~@!k*p?ix$$Qz5c1ngAs0k~9V&&K&PN-3oWGpD>n?i=fe#&d;+j&j1 zYnFy?{!9sr=11we9!_LRY_{fDsF`pAF=d+~S_Dn>Ew{qF25L|9Ym^X%xCAls-NOzM zda6+c*Obs1Rz_)QtnPkP(WJ~#9Hj5AM+Ut@9s2vYj&C4D(aFzK zzIKWk6J}EpGCw4N56e@}yfYj@Ynj6=n~gCp-8fh%_=K2DASDABD7_ZyHM&S@w+uH~ zy9^r1Av_B63e1Y)4cZG=N@zdyRdK^p=l{TL40mKwn@@5T_O7*T$|yjC!{8V1{Vv zzF}nBLvhcI%J9fo?)Oalwj?gP&pXOcNK)A>R(BVY@aZX3+vw3n`k0U_1rFJV2hY|7@$}=+ zGiTt&@jhM_J~h!sq0p0*7Y!s#cvKlgpJyQ6rIs%f-IBD+G&-w{(fNOBevJ1FBFWr0$d=@g~TzjIrvzQ(?ep|0=kF(q(l6q!oE`& zK6O0#Y|bbvFKRMe3bQiC`)TO2T+%Dn?(C|x(wUY(O-wzOuyzA~t<-w%F9(<(x7!fK z0>rc-MUq6+aLA!jtZXl7u}B%d7-Ujtg$1t_@~p|>R0`0)^Dp8!vFK zAWelb4`O#OsGe_<&bPT{Q;d2dX=E`!SyKs&W`{>yKno{S=HsWsSom4Svl1#2Sp3-9 zq_}4`D8xjbe6lE+gJ!Kym(lFisW{CxdvkJ{0|91?nqjr9y5n1xqj=p%Im|4E%+bp9=Du zcYk!$2aZ=Djc_63tKvEVJNAoea5)gj8j@oXO zjN6{<+}no9y1V)Y(GKNP=iP7d)#_w|j2D7ygBO%%c_>}o`UYY?AZ((0A|N>s^2y}D z8C&#`fjX59Z>f0$heGY4sNrwaWa+j73#7PxZ-cNgtBulsM6%J;Hm+6@PMA{Z zE~krK%|tzxPEbbtqY$QNW-89&PQxCPi%!Cm(*L&jF#F}>n-R9;*f5)w#Je7)>~^=} z>G9P}ZfHh|$L?z;>}$sCYkpquFUB<+HLD)yPsC#=cM8Sn6hT`}eleM1tBk0ZLhv+a zmdQ=naS7R4*;LE&x{p_e0}Ba~Kves2F%1SCmcgDm-;QZ?I?V`A9e!J`4T*g>Q|aOb zN-_#ObLksi(H_~jfwx&waS2LLx%x`xYxi3MMe2vA?tF^2+q9g?xNrf=2;(rXj3;L7 z<_Y2fN;SVWj;{HWb>%;f#14PD!yT3@KZoCmF)z!Ty@mO{kl;l7Y%7~eExm|FiNApi*a*(j;l> z3+b@A>lsG^bx)_uhlMTY2D@w_g~80$Zw2^RtP30OoYdp-OtD}s{uENnah!~S;+w#{ zGI~sP>`Ik=41)8ZzWU&7x1dFtRU3g`FVT6Hemnqg?kHGj%&(n%UFo&tsowSSXPfr= z2ycmF);ElTM~Dra(*L$|ip`9WnlF&5?sA*iAU%@>g z>G~kfVvjD0-@C`HU_JWYLh!{WoE8McttpF|nq*bwhdu?mm6;q|XPw#I)VowsaY{{W z+u?679(+J%nxI3OYyKI4Y8Q=>DwS>Rt7`S0vIMcMNY}(uF*26T94m?cwF*8PMT0LF zDqW1|4tl6>b}|dZbTQUl@bPgEE&?IH_0-rdA=G#UkW+I)a@?E6Get_OhY@)`}tu ziT%v7u4o%nB5~_wtYb_1n)F-RHATQIWq%$NXR%rM){SL`C-!hm(AJ;YKB006a+)9I z^6acvyec|qs&~ubJg_oQ*E>%)FO8RVA97x;EmMNtd~~SsIH39%v`-%n=>IA?z0jSDB;QQ&x;sr~)9TZBUQr8sH-M95?L%nFdO9DxGr& zSvJi!C!lnROGpp|qIRo>n!#615oCM`8cZPR2skuAs`?J!A>HwZR5$^X_sI|!NGr(k z{EjOvE$v~b{{hssqU;L4ZZdr;ua> zrKiA8pP&pXP?Pfm$b1?b8<)Dpi^&sfjryrZ9q!r5vT`?gT05~GaLx5L}W+q0AV2< zqu7ifWLW``dD`EgVABx@?XAQ-tZM($2!8m$Uzsmf`BEAB^@T6}-nlG*WBCXGGsXo- zu|jEQJ4xUSyjHqG(|MglYPY7Vs0Ic>Ez)!FDX9K)3qV2+wqg)^+S7E9BwaL`thar? zN`W6WKd+SZH#|1R6k4CaFiT9VETJN_j!`S*Dp@o`O4k-$CXMs@wD-N?`rc zx8wi-Kwm)?aUqC}l@(-8^TNVVhIe1_dtTa=8~7SJlFnOULy{k9lc-59_YDuzlQ(*jbUNMMcJcfXMLUlf>ePY;C%(VSkGsn@f|aN_!-dfW*D zFNhUk+ShoM?!TK&uLc?VC=^V@u*yN3aI#W2b-J>W`PY*W6Wr``?*E`cQjos`ViU-n z&-)=hec_aEq5xIZ{p%KoiUC^3UoXReQD zD1y3Z|9k6~1T)e9zw^IZ=Kq@pJEL4WbIl3pJq$zt%Oq^33thz$>pwsT`+sl1Xozi= z9o+|AN{5j)iPY0LLc$*OY2Y4)Lg^k|mTd7F@JgS69BjhU;im>Xc8c#u^}p=Rf2k+f zU&2O`q~O)Fg0GmC0mtLS$;yVB%@JWOU5)QIaFO6ssHA^CxY?jRBf1Zs9uaL4pZ$k9 zb|bZ_GnOPJ*h#6dN>vq0k{b@KfL=o7-1bIAaaq10tPsr6f4AC@OdBz?6|t0tVTf#I zO5=+GIq2N7Ee?9t0yiCa$aqbPS$RezFJ-Pt8Yd?9pQ{bAgrhb#Rx(=o7No)rdi%dF zxdmyu8bY&b#PmS;I8qglf2f-U9g`FpNX#mSH=UN*WuF7x`joF;X1pz~hF)^Ph zh>%$WZMQwZp4XY3ZcrQ#I66aI;I@u*U&qTs8-gC0dDI{Y4<)G7UahaDsn94pJsuF3n{{4Ug)O`UkA7m&9UochZ=>dD3CmKP=h$B0%WaQH$KYM;y?RH)p^N~r!R4dm0 zJ(W?MA#ug*s z(U*=f$Qgar=>uPRB@!Ii)ednMgXEhJ;l-oYKQRPe#%h2tNz{#jDL4c{WBmX}bd$9|HK5wP{8MS>M17d5#K# zl>I+qii44!Gr_XuvXYFrzU(IS_-gz!Sy=5ql#fCUV%g&rMPST+SQaUaku|U*oTg0& zJ$ph}#gnThh;l}!36$_-=z()Rnj1QG%4rt?^5!f&!$QzeJY$=NNJJh)qt%*O*EAM~K_cQXefB2R3LGj1CULh+c!zYg;Yz#5bI&9SS9z50m!Yonz zMrJPmkc&5HU1pBLx>g-wCL*|F6=7nffYx(3gWHa8-e7<+qx3Wn=C6JSXkyrv>?AiM z&3Dl~*hUa?pSeXX6zfO?hKhmCz4FuX2Z})mA{YxRU+^AG%Y^WRf;@Ah(>5<72LCVQ zL+)GD#Oh_=|4afe;0*xM*|xRFAFu6?Eoq%hjZ~t&7#FKt_JsCt@RTpf2%Qv4npL%- z@T!P1`YG-s8d?PX1VO_tLcy7XnSAGS5!xx{(&~chS^g8}GI)JiGazLxji0B8Q=?k-W7)_D&#!`kd)(jGtc04bePybP6 zic-)aF{1AskVy4*^R|CeYuyq@@d?zp)?_QkukG z8#JrU5eI6vmOhqE;KWHVD1E>DSe_*os=9q>Q!o=?BHizqvg?JjP4*KLC1LDpn({RC n;5+qL3u=9N06V|F#lG~hF?UqQun1I!0so}Lqad}P z_g!m$Ykh0^bN|`L@$qX8=7~G5zRr6R4fQoh2pI@5Ffd58G*$0nVBq0nU|`=Tzykim z9!BJgfr){krK)6%#M~>v?V=f2j;~Q+z#(oLKzkYyQ`68;z4_~}`OO=uCJZ&*J$3Tz zTvDaAFuR+B*ZZZu`<=a;R+GN+p+c>#SySValaGH%w!YsQFKA6r0BI^>QKm=WaI#_k z>tiH4hvD1dZ|?qF|Nh_qxnPUZkfw$8k2gASmh&5<8a+$?c;}P!us^&F6GY8{X;R_f z;s0Imf8OzL<1voi!~U-!0vDFkgX#%s9G_Yu{x$M{-Cxe{#QLA}`up1fO+}M>ddn{| zI)56A3FPuW7ULg}kpQ{*4Nw4tuic&|=s6wkS|G%tP zITisiqr@})5}B9oAD`R24SwnNevQvffV!^263dy>uu44zHYis3U-#KPrd&z%|GBs5 zyEU$bsN~OfsZ7Js*29>1Iq!}7H^ceQkNLMaNQV+}ciJb0=L4d~G)Z&QcOqeb8E_7% zDy@H6j=$xkCzQK~>S??3NkQ`non++wmu@tY81BOVX_#@fBX~@WD^O^|iKIKnua+Ob z`TgywnMYwQcVsBR^e=5M90Xze7_j3Qu!PQQg$3`gvlbHKI$3R)Hiv@3X;LR2ps(hS% zeKf?1Pr{I$p;I8Ik;)`(*AoHL5EfbM@Lrd{IbWxYrZI?M16ljOH+eC6x%y!FU6@zh ztq<)rFYD9FCj^IO;WrhRo3_WReetc!<~OVBkUG7hpR+o`17TZPZ)Ran>W}l{-CV~E z3xaamJr>cT-yUCV<}FUrM&0hrR4qTg{BdEFIcrl`AzcefnlYQ$)A^djpjC{(ddKd-epqXXRS|)+5X$QezYkPGH4hrNGL@#Z-0NE*Lank z((Js}pO`Ie`GxBJNSOI%)An(qY7dMcXFzFppQl5#8PB3m?%pCRn#Eb>Xh zTMy^b7ub(>%uaS^>lgH80Udex>Ps#-q=VVmW6fE!5ia^1=7gXIjhkVi^mj3zL>O1t z_r+b`Tpp0Y4;72ikGiiV90ySyQ&B$mGVdA4i97B~E{C369(0!-WtYaPW0=%WC#eR- z5rb!Ibn?1yet(Z55<(sg(2>cVPRFx8KCjNw=WnhjXb|qc@*!y7E)A*V%Aw~XRP6uJ z{#i&ykZEYnKw)CeZZ(_7Iup3u{?(TNoy&Dm+Xn9AJH9qYo>|Sb!CNZjZS*=Jv2ie~ z;N2RhY=fX<;;bFDpuziWz5Lbt4z_F$Up2YhKYvvhn;Ghg|5-W0&WsNsaJoBJ=l}h~ z60S_(zSU$kKWA_O!ipyIO`>KwB>{Chzv@FkRYS(pd<75fd&lxc9B0=|Z>pgKWn!*( z+2QiLV$Ju_$8f?Cdl?6*v6d3Qzn*ss2#?X}aHO15TX7o3GlFD&T>NiHv>jpiY~7ks zzbZ9ZD}`N{Me2U`9n;NATkipye1b2*-y6R-zWSP}Bh)z|@fi+?DSeCmB&{$pGv%A* z4zdrQjFuhEE(T~V1s_ZJ6%EJJ85-HaEnm8s05)&OLwzWLf@950NrpI&(^TK}w%KvW z<I)ds}~)=@j{f7*_G1&XZ;NPLcKa$kWXg`BIzBip#xu}>! zqEOjj3ZwArol2vimK;&eoaVQba?qLNg#4;hh%IdM$#^THhT6n}ichvemnW4s|6-L{mJ7lBljDaYO?mPQ9{+3stz)7Pr(O zUW5A+4Q=6rpVXvX_H>GIjs0p=BYjVsd5Kel=q%!j%;3cPIK`+}JWsNO@Vxo7x<=|b zIv6us)4J#L+BqyYwC58p1nE zXE0K7ls#s|JQI~i4Dzv@!n{MxRw^$K9<%tQlcoy$TB@9SrAKDbC8S1^gw@XsX3S%c zWEY`#De8B5W~^>NNnB;+FqqSfkB9DbXLym|38JQsMBj}q6n77pR>Jo`_$qg;OCav_ z>FuwtjGujSM-JE|r^peGN$yJ5pGQz+ONm5IWe$TmsfuVKe()J_!}7JI^`* z919U5jTLav#7+0mVy^ zH+Z+#qGVP#=h?ZLYx(fy`$^e@@e;CI63HcU_aDHM!c)XQD9RG_DocX~xgz+;Q{{dx zL%G%Ah0`Ar~1J7WzbxvP4vc_>JmCYK^ z=Dt=cj0QpU=^*1HUCS8Pdy1pM>{pVXJz5z(J0ew~>kjk~I2X=icH6sg{H0b0dyWGU z>H&tqKVE)3bSl{c93!1tMePuiFL6BBx9&sRIfVw3l6Zri%rdX=X*rd?l>Ha~k2Q&D zvinls#?sYVj8j%qu`rl4@*6l`lQ^=yus`Ay`a$!+&PG26jA73xZe_fO8gX%=sZ2)~ zxviX`+(=h&kuy zS|Pjbz8+33#}}WmI}p=$AD2A^aERS&y3KmW;cU8gt1)6(Ho~wX`Jg#H$=KE_xgTwF zm6Ui>D7lz>i=$};&}2lXt_0vicJ_78>-UYw5^WP9!sz&~YCPM|Crntcs=8wn<@i~m zrK_d^fu1Cwg4SwC2zllE2L0vdtaAEEnpd>uB2sG;+rO5JLiY{}Be~aqJMU8*k1nuE zh26fJ1Y5ljE~&uqtfz9U*D8ono~VQk-4DVUfNdvh6asC@Rj;S^p! z5LRvZ9U+;T?c9?e3VfGrVWek?>bikB$P}0j^ASeJ9QGr2BD9mB)}T&Y50?$Y?&;hn z3SBKMSL?f^2(G11MaNRP8LuHTy@zE0e*kt`SWk;f#cNKtd+lA&BqCQOIz9URSu3d| zwX#cbzY9|o!PQPx+FhSu>UFJrWTcQB77I_LOAC-w^CdYIolB(b9_olG!GRKOQMR}1WmSzU2HRGEX-TIjs5|D5-k+F&#njEsN=6mg z`OU0s4Gz*MX;gmpqR5bcWuou>mYj>Q&swzOVl0HL5<1c|aTY~C|I|!)$nFl8Kc6|Hl6BS-pszdvO-QlDCw-?&qe3fD~KOX9qInO6R(pN zZ5-bzzuI8@Cg4Xssfpcm&qOFucp+*#fN7U?1N%J?{}(Neo7sk`(Q6MkMPSPdyy$|_ zJx)k~E2*_?POh8s;?-;&Q-=qTA#Y<{v=i9!@)k_9^)qD#bOVKkE8i z9#qkqC2#*=`m^Vo6e#=v7r&MDJjQ%x8L%3Tv7aWf`OR2oz~e@)1+f}hxcJ@~ZNnL` z!?9M){!yg0)AyJSToD-p+1itC$frC1hV5uGtg6tEj8h4Xa4J1jdNyD*L?3~uQsPqOu z((HU*rJP(~on>aT3{1)^sMgG-BgWYZVNTHwGmj0ki`3~z2a#hdZb7Hl3uy$(z@O?@zqMbv625C z_G$feABR*g1*=D=l9%jy>Nwx?FgzReYSr`8FQSfAR00}^No4ql#5@l}M)tZjmr)nW zOoN)6GlvC9WwF7l_4{4fR859v*>Uw4ICccND{{T^b(9!iOJ+%3hsl5-{o$}O>-HV9 z7%rb`Mmp#8cyW0rCHgqoeeKmxx&{uSq)Pn`m1wTC=y%jE?j4}<@g#f$^(`-Ym%znf zyZuBezVj@_NOHU2L?85*^klA^bW8JgWz7fgpO`rEZm3B*Xj>UM@{zEMo%ndtj|q3f z1Kr)wed?|{u#o}qxL6NM1 zwe<8^i}uOZgi;)YlA96!aUhF-Cgr7qL5|a_`MWws^Sm-Ac>7M~ttVA7UDOW8N|Aj0 zX}X-BE7gcf>G`mYzI}Yb`fk3@5zkqXxqM;a?vdN`ek^(ntZ|E{g!sWeB{KfAHI3G9 zB;N9#&;4Eqo^P~#uPS3LaOzC=`&o`LNyCBxsDuC~xolECC(skG|N6tt0HrD`>{ zp&!0gI|l7H8GoCbswdNrX6DMD$e2uEcFM2{@~Mr{{i0b+q(D(JMsDDeavpxZ8h^F9 zp9+oK&|ah`?1+u$<+UZ|dmu(lvr12Kr-g}=Q};ZTV+w>KhLotPFOq^y$m68OoRhc4 zvP5;R4vB?pf)bpcV7~U~lzcE9;ZQzgg!g#Sr8*w}zwUf)^<%wSm zrgJ2^%Yd04({lD4=@uIyXH&XTYfRj2(!Zi*GCN`SAhJ3;F@vGytWg-@|`k0LXB2Mg(-?`Z=k5x{^ZarQSt&0Av@gmHs z-@(hiyidB?x4dL#0e521zY^SEQqgmL)cZ5ikIVprEZ&L(HM^7{z41vSTbHc;Vp&d$ z)ADb%m7CW{$#34sK(s&QAdKL6>>nQ~S)9@`Yzkxsy6vlyrH<@jq^&uk>or69eejqN z0f9!tW)(NCRG!ZGXMRVAT9QI9_J=d@WhP;ct9aUBt$Ge~D&)=6O&tTRnb<*jbqXvK z${*kj_R{(Jm5A?=ojm#wrCaUtbkMw@?`nSJ(HMP6qNe&ST#n+NAtLBeC{R1lFq+oR zr#T{$*L`!ey7~hMOI0`5deivi37PW5)^Tr8TSit)R;#GXal5tB4lSN@uS*0IgpV{R z3hfZZ$6zo(?A7I^(?1$E%;A?Uo3)$HZ8V&W&%y*}m(y6bNtD#+Oo=w5B-GRA1C$o#XqbqJsEfMIf2eV#IXO?O<%zEHc~RG@4rvQC zQlqC^;6)Il46GLe`|mD2i8DpE?aQk?7JY`Ahy-}Jx#~UOTWSNE8)GNZeE5s8_c({E zp(y`D2(BSpCak*py`nxv-megP-&m)hOioU{sn1Y9$!wTz2)W)`fs2j=hd{pQa#E#A z1mfck)Tg{7R^H;tUn=P1r`(dvz~!c=O@GjKjvqYE*GXcZH)M@Br=KgKTB9dBELMo` zOG|`>5+_$|eR+Q{(~C&atP*!_cRj1Ic=5LZJQ+!b;wd#??ktPrwX$8JoiU)GL-zHK zF}kdx7*D5=g&m?l(>*qa0KaLgiE|^ROx2$?@!<_SO55GQi@8}pul9^iy zaGeV;M6g-|hzaOB)b2(3&3W0b93v=coRvmVJ*&_sA@i&qQIsWcjliVY{@xA-W}g?m zPU15C8Q$k4n3XV9I=l8yv4?=?l%lcOUFB|oN{@dSQa(SF4D=mR$wm5au()aa6LH&K zD!z^%D&puUF!G&BN?Y*Kd8!=rRZgW4$i3Sybp|WsIcdZlUTy zobjW!?+O~WJheSb(5bPz&5Ib+5iX2LC(U!B@B7;?It~-6XK#=hq3f%69=O?y;3B}?=jlq+C~(lmhm0acHgGP@25G0=6gvkjtUsxg z=eC~HZwpZx6gQCA8!yX*N3`lqF$Rk%UCK1f)|V&UGJjTUe{Ed-#7-pgX81Bav~sBL zFqL;!=15~fSsj$dtBJk0psd!>k^k@{M})NR^A?gY=rtc?2|XwPt~ylC-{z&Q@y#9j zc*tF##X_b@orsAUuIhB-Gjh+OAJ)$>2oBx)PG*F15Pjs>w3b-d1%#Dv41;kW%{G;m zyF&9v&7tE0I|?Y`NYPBzpBBbRJ@W}>4d%9MF3}jLRjPy!c6wb1&v+gak|dkB)|EY$ z`EUl0CcFww?N}~_V;tOP_HODpB6F0Q0x-^43kxG8_zdF{68UM?*@KDHowomlm;3qZJ z>D`bTmLw=7*gw_E9m`N=KSIfIT3&R%_c3vhPTo+JnK3EQ6kUf@)3J27*i@9>QZ zAzhdrR~>@zf4x-wu|}sq_%2%$A_K^EC59ACLJ}<*twk_c*ufi$rSE{l`@@imt7yia zHR0DGk~Lv{?d*;Mo}~*#Zy8JJySnOw(#67j)fWFPk6(FADq-^pM=+kHARFf0r?S;l zf5~^&QRc|@tY&o?LRv*I&f%+@hIoo+72rAME8A2|E4`yvDNPT@K~-4UZj@t1OZjkBIE-faFoCNjbw zfii^*+uOMEq&Y9Y=^2vdiaVKvi8 zFuxLpJ?{e*&~21V^(&4HFAlQLM}`vji%s<5CJws{Nuw!AVFn#c9U+hpsyAKfAa$9| z28uzpu11olP-Pk2Sfl&9mGrWQ4md|x8uAuR*%3AqIb$a3;V!v7=qYFjhu&+EIYHv^ zjz%9_L`ZiBjaK&K(>2LsEm=bmq#V5v(@GPHiz4nR%_{0?TnR%{EC|tQ4Cdmcp@U(v z1n*R=OcAk{x}gJH=Cdof<@I!{$01xB(h_%Cs#R*kItDE)B5(=0?B!J>P@VUPw|KAA z=(b28P~Q!r6rc$}OoO7cnkPMnEMs6wPDy*K9(k%*F|ZT+4QrXFP?Qeof2FF-m=RF^ z0O?&C8SGQ_Bu+Jl>WrQ@U1*x0FqTa040mgSUnDb?Kpg`F7467t(@$w%%RS;yhO#uO zGzWpS(o))6Qz#Q1oe)eoB~8EM!|^szL$_o7 z?W=wnN;#A$YL2b4iuq1$;6xi~s?FWAq-=VfIKH(~L5`g7$W zQZ%{yHS+!#h1BPU(V@~t&hvX!r^AEK;+{`N4DxDQvf0N|&+7H-%!0}*3r#$Hb(TIJhdQIKxL&J=JeDW&bO4PBd_%}#wN z5dCU<%=!(;kDd^u!e6AGzGpVT32e{Ic01Rr1}luW7DEVUMa>@gF>abz9SVEH`F^D5 zcv_({^Y4kIfD97+h+Twh3j2MQ?FaC7p@9zTGl%nFjSMdpc}Fr`gW@J#)YI`qKX|rc z1x3Bh1Kgz1K7M}BlnlbYDqKdd!~Mb>Izi%9PiAEY+J16!lqYh@p>#Bh_)$@nUsN={ zhfT2Py=`lpH)g7ZI_&iEx?SerG4%?kmmYJ+y-*kT8NU722# zJdO+Zge_s;G%Jg6P)X_qM(JM7R01vgF?SiPwM-#AC}e6OPo&r3XFo&z6SFdB4mc6y zm!0T1H}eBAc9SaMBcKJES5)juz(!qxUUVfK)#=jc2142HINd`70yvMj2Z7Y1hgw1yiFY5lF6tu3h9Pw?{PwegXw-bdnWhS zyttb`eNvUVu?|F1(%L!RPLBM_WE(@L7&?6qu7&mbau^>mY|+p&aClDJ@Ns!~=lG zsbOw`NeE2^(i3`LMo0(R)QSMQbo{dfM25{L$~((<#GToQU(2Jf8@Dbwfn!g`W%vN+ ziQ1d5#+oDf1eudpt&=_J@lWLz3i29z5|#y7?CD3DOLzou8roj>uY*f>K?3X4tKa}% zo+2s66&?q*G0U|`J~E5Ey0xQuT5ib??D?+`r8}F}(o~2|(x+e_L~vYIRbm5;OmSVt zo7+lvqTyfhsWN$mrgR7j=aWC4Ej_0Un)$V(Y5rV&RbVs{7B#J9Kg;~CUoc@fCzIDL zv3U&>8FFteSjdSx->#DBuTX=)Ot12)i@@Qv!CR=%ihzH;oe{?PLwZ*2VyYOy4&&a^ zhjjT!kz>C)-@2=QJ+I+h;hCVTDkA$8m3XCfq6>$&vC`_5#EG7Lhsqea%j&&RK?KB-p9S-M&-tBViR#>JlX$(`n|@ktE?0g;H{f_&JS7WE#&VK$?3WssJ<&kDJ3)e*BmH{5gq_&jG) z#*b9L<9ewi#Dul=;wqCK(aj7JW8a0UB{v~r002FYa8=j&1Gfncz8K6A*t+J zayZwyvbJj|rr67T)K{xcgBpk374|$U?@144k=Gilm<|p)ANO@}Juc&a&zXBk;Nx^t zBnlJ*6BDVXJhLIUVjNa7oKRbk-O=boYU#+ekbF$8`<9HYPO*shmdd8F`r5kMpkNeF zD-*H?yT7b(4bfZpqZz^?xH?JCwTH|k&@Kl(GVm zHS&TvKkjHOSAWp#-RaxAavlgB_>VNEo@{|1Wocvv!WTGW6x?OoVD+|B$58l$AN#m* zQ(vWxu2Q-_QksdC+_#R_030}|BTi9k9KX9Xw>gzB{QV1tz8?!`Q+|2Cj7xfqfsQhg z`?X%MlgiB4oo~*%5|&|ey9griMO~GXz}>;`QGVL*L7kn|7g+?Ci)w1xvXu1Fx*koL zshMd8%5&6fJb*WkDH0( zN!i&-06_(A7Fh2Xp{~dTz#*woH5wAHX><8@d3S3;_AS=byVQpx3z)~d`klhfs6Mkn zEWlS&OF7ymO>#2;ssteO$p#O?4NghwH-X=O@O_#iiF zU5mKGAnyg)88}%fb@8g|czrm)f2hw%0O)XVPw8-!K9=9)g!_`6-jj!-J0gGl1bT9@ zBX{8?ltWpl;&s0MFJPUVQiVJ>%A!)AeqY@;qCb231-XhGO7-3Pk7#ry z2rE2uD>MnlzsS^wmc13AG&H5cde&_>!Pma%p%fK%J|Vf>MQ&$b6|5@(C3UbxxJZP) z6?quhO>R8LPfZXBbVW-JXdIrFnRggn3^?jAv~M5|?mUaQuns*MnEB<*-woa31>l)- zDNqwCIw!5`?G_i;8=0p3JX`O|rVo_$uW|U1xEAkm?MJu_S+F`A`A4&0-ixg9_$=)R zA%u)N8BUY1B0ah`n#6Q1PT8c8toDpv`>;LbVMGwCdCm7s#Ohn2V;qy!-UWH8#f-5! z^gSV!sQjJQRSMJJ9;2l*SM9{Ew2!Gx-O3Ycx(ltOxwfO}iDAx7Kb{n?(P#<%j3jj3 z@hDlm)Nws(Ae;5^)7lE@y73m=`Y#h5q<&||u2WAgXw?b zs{DnD5+?Ot-Abl2IJyS^`R(7#*NQE`y56=-bNv@d1Q1C{@c^58)b>(5=TB_vEr3h? z&WjcL6On5O|BFwpYiw-zFH#GDtIqQxo=kTO`u!(p_19Y>*B}5^41f4g?@w59PdD(~ z8|Mn0KTU?33_y$5T;1LI|1_B!u5$iUVIu24(YfVF0st>Iu(Z7Qr^y@&1J4b&*vb6U zDh#Lqh%xfhr$X)jpvL~5OgMlGUymv0{b?0=S%9SPMcft7YfX;@kCGu#l6g_&k3h)T~W6NPb7wed`7)<`O zK$>>)l=T=UJ0m=0s>I5m`(bDARkV&k>_!zJmFvr2wn3XYMPja*JTn;_8l;R89zgSb zQZ&>xnuI~@!5|9!dMHD0vSEVK{8sUI0Czo=F~`paj5K{CAC{iVze$ zy7^v1%z;b+3-$p2bi0TsyY3@X*Z_;%+WlG%zfFgzXcH?SCX30pfUm>13J}`Fwp$ng zdG>v=6;V*dX!NJaWEUuQdc;QQkz*MS?qwted~&V(`Q!5tZWRg=y|71e)MdwzbC4g33jiIyus4`eQhBlTJO|gdUZdL&5XU}KR zj*)SDezYDAbbX^qWgaxtTzP4Gz7erxpA7c!B6TkZT`R8z-hS z=@*KDNL|Cj{wT+>RJ?hO}iy6$KH|G@hz&?c=%KI z-^y#aiu3sPI?zVJ7}>)8mrMG8%}6A_M!D4!!^D?ry&zjI0TIqMf}|<&k5+x3LO{YrqgqT35)qBx755fo#NF@m%6M~D~V&BO{m!gO5pWFn{;Q+`} z-8#Zuv=q1;2FMFfr(J*Bfi-;4pMq&N<;P& z2j&!RulgjkjGX`(z1_$%unF3v`7o;FveFX?6g*vqQkk_@f4o{AEKjsbh;62nzxk7d z^n`L5KBsY77(Ued&9pfdCPCp1Fh8#WDhHsEES^1@HZ>Pl$I$gj_xEbm5mOgiz_;Iu z4hP^MufN(N9Tb`8jE@0;LjQC7!{)`_1lPA`A5^9r0Ic|J@EP0sWD|yQWc0s z8Vj>9jWYowsQGnDNt_S>=Se*Lq=?bWYeY=;C)=SoK*^iO-)}nSai{;)U`tM7Yv-$~ zN&1WdhkDKha0Q{b`v(tz@>Dn;=2(YV;KtdL77~>LM)4*B6T7n2zKDf%@lnEn?K9kD zeUoR{92Mzo0g8f5pWR=)mH3KWN78ycin;S1fO(^N+!vefpd*z}0Vkl=3k`F7|AZz! zpTNi>x32j+dD22vnmmJ+k&zwqpB!E}7AV^t7c5XgE^mIA)2w*T0m8N5bd2cWQ5ORU zCl?|XxJC-Y`xmy|cEWv>P4B`kTiOW2fA06P;@gs|WcOm=r8tcozU zdciSa$2bYYLyFmSD)!#Z-!W4pnlqxA=LOgK#lBsEYe8=v*vJ37* z3rV;P{*L1f{h5sG^jGPf5&LN7c@s#i6mWhD=PS_`^6dZ)*nK*yv)iv_-=9FS_kH|{ z*yJ$@Y@^h#sg2CSD*XB^pY1RDd$$!FMk>44^cTI>?G%}E_Z-DwtJ0^MkmF3kD6s^9 zt9_z<4DQb`QveLC;G1CUIL!LBAL!}A!iU0d>mv?g#8v1|ZK$AK3lAKYpbd(j;@f71 zG}~+ZKk&>$RnnhET(?uwD7|ui|DBzxF{ zy=}{L1sL{s=WEH<0gE*rjZ=B7i=!zo@QoSDRMn&LNF2t8sPLB+HH#Y;RKAMYsf=z* z2}ms7t)caFsu>YJY;9suw?IiM)4kHd0B;Oiq)nK=tT|(lS&_kHJ z&v#oFL(hM+6vfH`d!L8_FBM^szdoIF?d7F_twUyf$*{nVCNu^C0wL=ulAb^-f0Xag zwXQ^6-LJh|Iqu!i4}PZH1<+#PZ7iJ)8!ny*E!GLeAdRUxL}O%^ER;5)@plluVY|Qs z03Y2h!Te^DY6UChKgrt#&t^2kYQeuY-r#qhQ3SkGIp82vCUZ~k)Q9ATtU}!@LEwie zbW_h~mnwdP-RbmWB_*Xt`x=XrII#|uop)>U^pU1}5NuraH^ea1{k@%`+IS$A7@V+q zOjDH6@OOrwm_gOUO>oA8AM-IS{B7Sca|^DWFvY&%YSrgOv~jt@2jVWfH-N}cxgL}7 zq8RgkXuPTA#M+Qjr9uKM#^Nzo4BwR1HGrVIEp>KE)ZeRlBB+z_!1HUWnF|fH z!~Rp^0=iA~hS;Q4nUF#-ZZ>&Ee#%cDt?u3DuR_DL4k4*SS7 z*pNUVqA$Ifjn9IN;5qbIEf8Yz9Ty4B6|s!KIz?z|P>tAw`}vYeyf@O!;t$B6Tf(oF ze8#VkbWuk14gy!rcT(^La*D_-fKZI@_l9!A+7?*KnAKVkLagRyl(=j6r0*seXZ;RA zcaG_OP~f8AfcCfmOGMW&SI?>(Rs3*nyCPKGK)xhTcA$tLr@X<$3b13w>3i^PFt2UjqC#3hmqvh5~ndG;oO@s#c zV?+3I^@L#E?ADeGjDm3cCsWpH& zbcZEw9`^Ncxa86s&gr?UB}6O?~UF$1Eq=FK2A!>PQxQ6NQp1D31eb@oz48}>74+~*n1 zg((9qH5C>I-Grq@G=8`g=WsvSoOh>w19FLV_s(pol$yx-0QSy5kA`NKDase&H)o@` zT)-#jt-?IoE_IT@kKn$I^^t0U*JM5h{mRlqVOb9!Qf0}>hiBXD^u5i0H3>$bmz*xt za2?~Q-ct9&WHQ9#vw2wLljwy54xJppl67{Y6amL6m7GkvdsRvC_d>~wL80vCQk*Z( z;rxVI-u$UNKPQdyR|q_+RA+N02w1V;XeZ7B&2AW(z@RB_VB)UHZfcHf5(3yXAIcS1 zs;4Q8kGhV4P*6{Bgwy$5#*l*$F%|t<)F;=y1Rzj{%}#R6*| ztpu}CP`cUAuOW}dnMGussD$v`B@7W}{-rI$87W^uBT_mafUs7?V0a-7embpkG+?-7 zJbgof+wrh}nFX2(fW#AFzfV^$NLF`S#)MSF2`GX_(iAvk-{c=DM}K74n^IHX6ILJY zM~l(Y8Y!E$V@3XWJI($4-ETK-orZ;1?!bW)4t8_R0@R@s9~D`B{~A7+Sm+SGmcVfi zV}tO~n3@5>oQ%KyDnFh!PM?D1mqn>tfMdrYqT#fi&qbZF$L`R5OSs0p{E#L4!u>(= zP8>s2b0ou{_>w9Gk$Ep2Ed(K;lI>HP`k=rZ4(AHrE(u(dzj(7t%tj)4?02^7FUJ1s zW<6Clm5gXwo$ZcFmJy4t$Bw9C6=RdS50HgUcBeZNC8kH37$@j_B>W6gBh}A&7!5J~ zJkwX;#_3Av5{gIosp z4)Vu{zcv{TgDWiV&>lb2fpYL=XFsJ5^adK)H^Hl+o=6SP{zrWzyPyZZ`*FI+zKejz z2m4sk46u=eZ^d?5c&V#26_4p>=r7X1+rQ2qy`dlR;aHc5FSr5{uql$%oEiJ)!&m7A z{K}MWM6&)cVFxvOJlXN>0tgk$!`KEo#&gf6UcBA4=Jeb7dj`%n%s)%)_9Hd@fHbxn z$Os_N0$7C_|JsM~y{I3-7v4z?e5D>m$zfE1h+L9bal+oRSlk`G@%qKa272Bqh9cDg^=cxpbKeF{q~h_rUPdhj$N zJiT_bl3zl zuD~^yYt2>6;pjP$?yg_xWez|l!gZEbxnlaQ5->|r#=+~A9a6|0Ws%jJ-abx8-o#ex z-vG*g3X~)&2t_tTu0#~$d5LT7chc&}13B`_^}Kx9`R2=@D?loivq!Bm3U3!d4fgr! zIL>p!3t+UZxRi*a4Pk5%$wVN3C4?k~;QdO&CY8bnZ+ARcXy%Zl%v646Uyh(4yl@y% zP6eB#dq+E+ZiC2OZ_z};a3CKIIzoSAiNOvC+BV7xOI;XGP`-N)@0X!+u0dJg*C6=A z$HyZ)OF0&2>KNK*z=^Ml)aEnLL~lw;q2ns}q@1^r*_{gx^C-_l5 z9T_d~VV9x7!S%JeGK!@yB>$2>?hU%P+{TY1!yVq05C!xbRq#vFZ6bMm#Q|3!t6Q~p z1e3_fi~mO^16a(21L(xytP-9wrNan)86=r94L=!P5byiZ@!jY%dgUON#atcH^Jd=X|hEU^#iu-t4~?02*+V?|~C) z3|kwBVsE_o@mDGtl@9TJ{kEqtZfOrgjN&FyM>vKrB%$IXRF0dTlBRb$lb7ekQohB} zY@SL}jDOHH2ClC^Q#Mb%GN&fgrSZYT>cLjZgF3Bv1pdYyRiVtyJKQENzpzV*kV=!S zpw>V!*%nPR36e6&@6Sj!EU>u+>sGAAcZUR8Fuu@8Xjnz~L)}4wx;Sf@hJ(Q(znxAE zLXP7wuNhV7)xVuvlrDf%gIs$aI}QtU-QTt^RqI|#A&6s(+EB78eS=Ize}g@COsJz; z5_}9wTia2Bw^FeDh?|i%KO)ELlw(QO1?m|HPs%Q)6_Nas-Kth9C$W~eqB&+nUI-&?)`FOLrpb$puyWOWytwPfw zzP4pe7&vX9a7`^)!~Uy@QRq;5K#niQ;3IRmXJ2;lFF19PttQ0rtup(nFj;e#qtfYW z7^x)B&$8sQ0X=LZeL+=ZbZ>FQr8Yr&GjU?arUJ2WxVA1nqtI9GirxxOHc{NpJC?27 zcLgl7MELA)BD)ei^9_bHo$j}*_z?=F_*ucFbAC=O$8)<(_NIYBYYcC*Z_=T;>P!{b zC#5N!9DM7VMKaq|+_!h{^rY*CcAieRMpegN#dE3#J@7k+o+Rgycv>-Ao##_ils+v@ z%_yfMdtv#HawYVZz>X;fO0A^~&M8OkcNO)smv&M~3fD$44Nk#cZRpVpo359Fzi2|o z)${LWV)28h9@iqeer?jQ9TdRSmB`ixoXCryeV}7Q>P)}Ha`lxbTf{ zbZ-0P{<<5=B$1R8kWMP>7lgM;J?DXPuk<@iOe#rDAQN?SIqtf&uccnb;=@s2v#5JGoVoAtThr1tY`d=LliTc`^_AnJ)sE9>wYvF4pbae4=5plu;WBhorr*IE7zA4wB zv+_#1I-J|J_Q!dVRJ40Ly|K#FETe{n!(!Fs*Fb19KoTN=YUSk-1=p*u(sigB8X9autTkw2@nQ!< zQbL)R1XD5RqQ}Q8c_Nz~h^G9W_&r_&CxU4Geh#q%8|Z~S#UgO@bgD#nL6SmBu$*Yu zPBXTI|1NF~VPNV?S_+=_J+Hd&`uQ4$>T5_(;tN-U%1!cUkeLE)W6KDVTvEnBp) z&aQBb10nMV>d;dEibxysNf$vCxP!c>{QAIK(V`Vip78+w3@J&Sy=uG35E@;l&R+=k z52|Ncbsl=^2PDqr8Z#kOn+?kGx}T<{(zPM#%>@&xv1Fed{OIQ^cXVvhQ8FqGXw}HE z6~$XZqM-yvCU5sz7-#++Ld8n=GAL7AIx*>0hdTLiWd{avu*F6nbvhALW*+j7+GqX? zoaqSmvpSt$LZP0hgK0-D>Y%tKP%batqu_Mne03kQO=^kZ@!EXFIB5)DGn-C((b9ZM zPD)Vx%qxrL2BrBGx~rhXWHK3j8QK>tosJGyQN{gYMHh@O@|2;3ak>4WQb}sRGYb2} z1cP<$q6&{nET}1txIED0p_=iMy%}Hb8Sn9iF3r4jH?v8Ax$1bd23k`n=Gxz-?1N0GFk8w(T@{;-xc-3)$i3e!v zUw9d>Xb032n!vnTvlN}F?B39pPBqgfsXsJ!i-YVbP^tC8@lU{>EP@@IMud@Zk(rmO z?B%_a$haloGv2QVl3Uk|Mpw38DkcmFU_GXv z;HF%3`bq{uy%Os3B@kAGrX@+lUi|+4B14&%M!`nU%N@jbD8};4fC{7@pD$sDPTCn9 z759|zSjaYbOTHCBzE46R21Pwl99_B(btBRF+QGM$(DIZVUtH~XVq2?E?MQJPS-Q3F zgt%r>n~(f^o}v18HEu0Fv@bnoj1=>q*Z8NSfWmI15-VtSF+F|<)TW8UnSu2Uvc~N} zb#v4gitg{r8)eo~j>NcYCmjoULZ$U~XU3bTyo1x&0n#AueKE7CY?tX=Vx34q^3UIC z0Dk%a@hC80c|WYGnmmp@a48aObVV*HLqv&pAjlD(YbruR=729dbND)g(Nk(HyEKtK ztFP(BbWbF9mAO>q1@?)T%`{FR7cA$DGPw(tj zw_HFKI7^z`dF(ag!TzU6N*Y;5zohEwvcboD`;A}m?4;`1;(1cx&Bnq(=}l%@KfiCNZm9rX+*b>|;? zJze1G(V4^mV z@Zjmce4eg6O$Caerrbk$A&969>O4HPi~b+>-ZCJot&1831O-GorBkFsx|EiXZt0fp z?ru<|yCswk=|;Lly1TpMuIIcb-tgZ0^ZWDt;~aIf_u6aEHRc>+jydc-9O%o0nGj2= zW~AQJq4?=JbN3n|y%3o&G@-z-O=2{r-J6f{STy)j`v0m6;lbOhHVI85Q>P+77DkZ4 zCMa_cIa3=!IrHc6h1ExnYJGRx&gjfiV=&*N2)*}6jI7GK$Kc~j%5Usk2GuQ(Bk}y; zWG0G_L_eun=1dFwFnfCKmUXvzV(xxz>07V8M~Ca%T;#K_h^$zz^2e%Vt#bbUFE#c13*|Z~;pNN*2!;aOSlB)6* zbmh(YULOWDhAn%%&Xl@ylDCv|Y_JM|{n!g=2|;MRKgyfTpT*0x((r~?3t~l%!&1sx z!^-MTawfKN#D?=d;C)p6(076^^Dzd>M-7$dE1VBhD-nh?5z7g)lC&Z&VW`K=l`G6Q zHs^UEQauZGUZivy{Reg=!yf^;qCN7hqJj8CWzMv|EkpW_Aw?E-^@sGV9r2YfWn&4g z*xM`}fdv9j9;}HLU0GKw!p_(s<^WWLrYL+A5u}+~0r#<))-PBEct;PTWWHW0&Dv}N zhjBT%OX~QZRt4(wMr)ytB?dtRuFo=!!?MmHtBT_qGZGShveJl=N}s;nZ6;Sp6(&~viX)Yknk2?h&3X}!ev`J0$zE_> z=@sR73e7NN)Opv;AEUGVo~UdIch+$O&C6ew+6mywK3~DgIhS;w2|}4eC4SMl@A}-= zR}-$gycIi;8tpGg7Q-aW_)JCm3wpk&R8F+Q7i7jYPCBJiY9Lw(3Vl^FCZU8` zcfC=up&V~X0bS=6bD&2nnYG^7Rb9WIXA7Lxl428wC4#A6y>7iD{uYt1F5lHI%$*9_!Zdlq}HrrsK?~FN=F4-mVJEjR2p6? z1xN&pw4nD$k;r4Sq)*h)pu!i*p3>y8gb{oZ{G4dYyrJZ&6gRn_vLZwH?DVUmp_yq|5^Jt+F6zO4h128@EGd@ZV9`9~yBz#V( zn)*+fPNx(Qyd-EWdM=G^wW%0^0(}4uMBw3TO{c>ztd|d@&WCR(zgM+vfnCu z1b|@ZnDrIh?;cC`3@o=&BYQ>1@0KLv76_brhlVkK$Av z(=TR#LV$+C1|#M_m-a7oWetOsfyj3S*m>R~SixbSxq*0A1(7Y+0lSN(Y<9&Z3nSnE zm)k4^4okVbUsBxm>x7$+dudszeA)fIDB*=ZQArodWs3qO1S>ZA%v2()@w)`3OqY{& zJpHYgKDP)4J8oC`I^zGcHZAUCOwNakkDoms|Eh_9lmmRW@LR! zt6yEuwhhD41>F9T-B6O2VI^_e=mB94OV_9Z%oWD4GPRd-;Geb7f7BcjRI*3af~rtu z7P9<5U)SrdWV#CuCNO8qr1OgrxflTNt$dmEwBWWWVC*BBMlKc^K0Eow@RL70hJa*o zk8%O%(nWY{(x1RAE+SXX$w|QWL9vaM==B9?V9TOZ4}r!uzZ&>6GW_{=A$z_w13uOR z4sU@#|4Owpu3o#@QUY{g@Nvlo3zr`?}MGT)sJ-q}~C0kz2iP>vw@K-{1F zz@@`k2RA25Gz_3b@FF>vgMsUjzy0N)I{emxAbGUV#Sfdp;Sahl|GUjUOJ(dXkO;Z| zu#>}WCCjC9Is>a-^;1~`EwCAE>~nBKgl#}-aaKHB2llH*wK#yu{e7kna9T!B|81TB z{E-?K#;^vf@Nl%EdIv~|Ce14b#PMvNTpTX7+dKd>HXv$R1MV+D6dj-`HnNz=(Kmx*V(1SWoJPTAPSjI6~5dXJ{ayIkHXhRb@XF@m; zl1=_Mk3tdiC@?dJ|4g0#Ey(qVXY3ZDo^!e|6q2d(n+-4_3&!ZyO>E_-N3J2)i|KO< zXO0A_0|865#Y{MsxHDu6m}9-tX@SB{y#p38K!WE&_PR?5_6vBOVF5-{!BR(itrJ7c>n!2Gg=TJ+fPHfBFH{+XjF`C$oL7F zqqiquKPE4X1F)3g(MIgG-1gYqlU|tuhZ>+py3{M9KC1L{FowuTfaTR>dV%HK_bn(> zV*i{x>E!Zv^frj~IeiTE;cSKlij#;Ad(&)aE$wZj-c0_un5QHXSoT?MFSSf1kAXPt z#D|wZJ*a(PP*QT}5WRo^#=YFT9?2<4G~)T1OZ|^#9advT7TeKXW`zx@6n5in0Xxo9 z6;BwpB1ZXU|^>m&(g3KoVG9o1>{hDaO~ zT83mi1Mn_k3xb#m5qPY0MW|>xC@E$FeFy(O@vMpHB&@EKJ@9lAF`kOO2JF0^)t1{q zV-2u=6c)ou;&Wr?H&NpKRI!E{s)VT}){0&*>WZq!941s{U`)$hEg{?L$oiBD8s0j< zc4)bK2XP+_xAOk>WN218iNi{Zh+iN)h0Wl0cj8%5NqtTn-5PWv7sIl;-}G(KL}4o# zYqByvm&?26R@NEzFHI6f&lAZv0si?jUSm7L{LGU|l&|Iv%;l~lZj-Z{l%sY^xV765 z(t&@K@J?9H`EB31(56_73_r3Xd>es2_%5Vj*E<=9R;4yqWL9=%gqdRVgmq9TQD3xl ztsb?P*G(;i-Ukw49;_Nl_Nt;szh1*wjO$Qgsnf0mVke|JnKXXmmh_74i4-$}979<= zs5pAW2_d2Gp;m8`Jxa4z!zKmhs1bjGtJg{-;H(4%{1F*hx#cXeDW%MeOIfGtXUjov zfQIx^nQumNrM-=rmsSG35HXrw{|KGBnZ}>&hZPX}b)IH8bE(@QDB8e)y^69xy|@R{ zEU#oZTgjD}D#KzR?`&0@SzKD=PcY;IVz@LG%?%=p zJuyuPIwp2793J)YDyC$muehQK9bFVGVgyu|86AOsI~`@FpxN_2avyoLMf9%O1EB0| zC188t3oJz!L|4xjU>+N4FxvA5j(kXGJvR z8(yo#dd^}1j}mRJq`{TThJye$CAv<#bIv>nX6uQP3kycp9`uq*rn9@i+M2b6oF%HuGKM<`{rV0ihCpJku^I_*vVdIYF!IAQO;#IPhu zp7h%(t)xc(SpQ>Pw&wO>#7a1NmS$EKkH0nxg)Aq)Y>4^vbZ=wO6F6Uaj`9B7&#D2q2?!wjE4rPloe5#ZjvRx@3yApm+-jpjT=5GY5S_U4PZaEY4bU;J`V=?z zD<6RHCW_V02c-RQg`WW@BEx%#i|6I4+7ceC5x@?_0L`SZ6_H(x0DbT*aPvwt#zUg= z3HS(w!xrzjUl@_{s$T7`0V$v>3g2xH3+G_MaA&V(dNrZ_{NcKf(Ki2mV8yql(^&*D zLUfT7)_jsU**Vu0Hb2iBHX0?%Zf&m!F6V{0NKWf-TVoG%BU|nK2b!+;Y8YKn~p~Ma#ZwH11$D)n5*~W z`gwz&nKF&%V_zs&pdRRBJ4r$uDlHRO8C(dI4$1s-$+&X zZ{JuoH(dwLV4F>aex$2u3q3m4+2k-fEr=^ok1r4cQiE{8+3BjCN|Q2GahKBqxrbX; z@-)d1%QzG1RPJ1tle(3+r|pZuv>mnvwr9xzS-Ed%T(7DE-`|SU*Kp_smlqLOs+^xdkA*Gj>+2tR zH>F8*32264{HCK6X^((@9~b78)jTh!bLb67SJGV869Jkxt)hQP7<;jA<4~qJ$Oh33DP})rL=)jct~t_8kW3e z1Hv%fY*DL*YkBTfg($wt0izv4mqUIY%Q4-SjdTFhJ(34FE|)801jUhg98?eD(>)iN z(9qLk$OEk!&U>vcoU_d~==U2>*x47j)=M893S69G5H~%1<DUzP>+*{k=u_{~2 zC%$-!z{0;$HVc48zRD|d&-L26pievO#5w!;?O&0g6G>CK+aB=lUVB4XOef|FZg)lFn7 zNZ!Oh2uqVc^c;Z?;O=~fF|~RwQ(=RU|L!uo09Vdip34MBJzr4{>7UfMkQyYdG}G4D z8ZdSU{*C^lbmd<}OKy6>oqRSXosU?j4*GJmuzR6)iQr5XwMgLRX;EUpxR^ zN#dwF0xRxK{9QktI;%JY?tAXK32hc@7TmEYM-Tkwqj;SUgk+j7k2Ma6ZB9cj7Ul@u z(k*IsMt4en2`2C#G>$nY8mgGrNN^j(_H^F~xNqljL~*YB_iXg;Sn+L^X4fo~?bP2r z-ko%~*mIn$+em(pMCF6$Np+Z*Awdj(Bw{+=6T{eebx+IZs^8slFYH|2iLyO9(w%;5 z8#N!re|P-i?&fo?%*bPWi{nWUh%US8_kOAF}78v7($d}gRS{@$JLw)Zp5Ls_*< zbD1s(7qP7UA}`shZorX;g}U+Vns$rn>SmZ>pFLH8LFliqST+|FKa{rCD-Iz=po?SX zWgTjJAXo%WjQ2oVSoaGQrWm zv{O3+GGz&}()++<-V@s>EEV;u+IEzN_0X`k^TTfqm$T~{<^v+V#?{j$-=}M|z1U2b zI|%Fd&-Z8e+C+a*lh^OgM^v%~BDw6Bt}sX^y%KrXtL&fZGowZg_8^8+5YeNU3vx0#DOr?-e>X?X)l7&QQp7hBFW&hSr-M%~uTJKls~ zynq$Tk-51DyY7jw2xL;v#UN{>UwSh3`)ffg6QQ{KPv-!Opt~|67p1o=n}$2 zB$3gB0P7d)-X8!J;AnS@x3OQn!T*BDZWAMl*SdvN42k%Bqea_uS-YinSz<5n?y8%n zsLHBLwJ++%qg4Hdt?A0O#qif`$5c3J@+*tpFF$9#;bj@$)Zd2E(l_qfs^VLe+5|N< zF7Vy2vM!V6OC^S5o-NzSG@g&1v^E=+sMEc|=W?G^-zg>zcyIfK{b!+w7#=~69~$Kl+mF7lly&2O@#@_kJeRAUt0iz4$a`)3+{a*M( zK@2%H)|Zm+uC^wj-MveLW8t)7h-v>cz9|KOl9`g}%f6RB6h50_`l@RkL%P%*$6tXq zl>}6VBUCHE7q4mE3TIAK2_(53rlYH%DiC50Z4I$;g^l7`$5l4w1^ohY+ysv~`ZT(t zh>5SBh~479mHY7LYL0(X+B--LdsU!`h*S)(Z-IlqlYv;qbt-T6&8;_agL2tv%B63ZQoM=Y5ruWx%%Wm!p_dNfcMsWPHEK1;- zDHNZekD;cm#EGwE@Qrbr?R(&ed_H}U;q3iOJGI{3d}D6?x#*F;P;NKTcNDu!mLkE* zrzIHjC*S^oWwi*kzUE7o+GaJ;7%j5YCa>lAM*J>qg1uG~9d#P~B4iG`NrDfa=#GpP zD$$OofYvu;!Y6$h3!ce0E$n(7@Z)4@wV+VZ{l#9s=vxzaUoYxs&dzIO&n>2F03Jn!f^yH^%SO4jpLC~n6hnN7oPEGnKZ>fXT>=n`5#$kQ}Q zy=}*r|GVm{s0lNTs5DnyvHBDRIxV2rT-J{XGd6?dkuFCE$^~eMg1EgK&NwDni0uRK z(pMQIP2Mgv z$)Y_;&hTlzb-#q3yVUalJw;`thzPoa6&BLOK6sX*Pm_*NBJ==UV^C z5I0;O$@%rvg}2ZYJrD3qAA+u#wRZS5NVqdmB-VO8s1P}c{AI|^>R`5x+z;LHwG98Y z4sZ5Ug^{MyTBib)Ojo(-7j}z_S-0lPDI%)|U!9J5QJbN`nFITsh+po*8;!Nb_xmAZ zIHohgO4MFDe^)Ov5HSj{9TQ91*h)9isi9!jrd~Y}!sma|74Jb)O+<+(WWb@@2790N zq?7Gsn$cCH>gmX}EX`xY>i9u9LH=u@1mR+WB|b_(G(joTGy|<;VbQ%z{0~(%_HSE9 zV*)!2YJBM>0bLPyVI@rfin5fkzG4@~cp+^*-vHYCm(#9-kjj{5Oh+r#thH_L6D1id zW?O@Xl-ybFsbtdp1->iY(xFc72X*SDdyGCL^)Vtn?GW^KUY5h;#pf`6P8^w~z{%OS zP#}J!fVGpS$^T&IxZzlJY#JXrFg^Zvt*Ygg9Fj}(lMeKG(m^e=GFSwx6Wb}oz1D(t zjv0jcil`D$icVu=mI=PW{IDLV7;Gn7NkNJ>{CoRi`{ir#4Dmk_w2iNHSwf(2BK9t= z!XW$OfIL(C+ubb;^lwV_j{F}{eNIsD=e+zYIh3V@kOm<}<bm~;`b+?^Uv6owL&dI*uF+So{YLrGbU(y~)U&@&C>%ZLe3eN4dC@=5htApm$G zowvHH#XrV2=w^xy>@v|mB@sh6U43XN@1tMHSeo_d?!8@UKWV?m{jOJEmu!G@X*E2_ z^=w;FBlmNmtPe+Q=U-W5#yD(w3CRh*L9VDNEP^5(bX=m20u!hKhDuL>*W&{x0?EL+ ziQfjJlf6D9v{$C5FBn}Pthq}vi&sD=^WfLfrlcTM_oO7p3YC@mFeZ$HyD27qVLgRPCOOQ|c+_oU?frB&WVjIn;pR`og{6J4UYG$l}= z-dp_XotixBUTnIIpE!u6n0BD4CiOtP$);?YQD= zf7&-t(5X#>T1o55I%aXFY(wSKT>Rt1d@Pf&fO21DiN7P70idj6Bf{^|3lw1nk~zSj zfl)v}bbc6;7?|t_&y7gd(%DDZAJD({ewreAqNsX{g{}LYroV^92H)vA^Q{I)ravv+Mj_S`WNUk20pe~dPRPo z%>P~!PC`~f((I_Zt}0|JR4Ea;`%v;+S;edfw>XdN&w?SO8t0`%suc}E>Ndy?Jbb`^ zA3f~|YRxGMk2Mdh4BLnC#UF_%P>l7vMfSZ399MFrdbo|=%~x!P(W z?WY0dwlFZy)z=OLP#G3(2jJl}evfOA+{M3RJ%(8Lmtut?JBUNU=u(p3Nk9M1^;fBY zsbX-q)oL2VvD)cz%CmapH4IzODJsUh%mLcTK?~-(azMGqwbNh<7oc)f0DU$Jr*AwA zIKNz+8>uwjx;K80w+^oXA3d_J%!wITs63;l{`ZdjGkO)c!x%o4qJc`EC6#n(PY(6a z^YU0Ei{KXbRXD$FK%@l*0~V6!Y+S+bCYU_(5@_Kt5X#>iKtCaPodA~TA&jvpcrfxg z+cxmbg|td@x`D3Ypdrr%(DH~&S@wXICMQm)6#LqM67w`eoU}0OXW8WcJg?7{G=vdo zFaWHXLL6kDfWaA zfH2Qk#)MG#I$kOl1Uw7p1Y7uEA0%;Lc#RbQ7Ji-%g78OK>dZFT*6LWezoO?~uhXZE zRQ4RpV4SomUUrsHi80&Qo_$XzorFCCj8D1-75D)8Pxpf!mQQ{(0E|234zzKxn{=_O;It>=%>SK*fXnB4 z`k|-+-K}fPelZ-6gWoqiu*r^>OB|3{IGGXmKV$f4 zB3{=iW){Snz71A@v{cm*mZ(>?>1~1!e9?zkH(1lBZGYr)HMkb^`tyM%NI7pv?@(8ultxK4UoA1{fcE*VQGxCGednj z?bfCE#iKGILx%i(E7}lgqmMyn@LauAL==q~Ux}-MfT772N0lwZg)k2KtW#d%{x28Jln#H$ zcK$DVzaAv~yA1?u@*qy^sb98W*viBzvp-pu+aJ&_7a7-v`6to*o0Rzjuw}p_C^pb2 z`lFVV!Ahe3qyN4zBq+;LfIke)$t=nH-QrzBENC7U5C8f&|KsK{5x^gknVJ;6{Ez?r zC&s%6MlzcMoV@=Or~mEVL%cS2WKQihZ0Krl~X{14S2u+W-nJE4Y(vgkQivodPc|%d_eMOi%Z+yzDSKl&Wrlp zCQ?m+%%vL77K;La>d48nKoBxQ3qZsw;JiWk+zNcZkfR1T*j#|%1L8Eg?k4yf;{Hdn z{}Z~vKB04mMKpw)0#WGZbjxlvM(h~J3Fg!v0b?}H{|b22pnXL5;i6#5U%rQ^PKZVK zgNR%~)ln2)6ylr#>i9r$(FoeWNVvoNidTYSc#^#^NjUS~rCtLN*ulrF_8e5!KPb*B z)co#Y3JhQZ^k~PMy&ize%m*+G#siqH2;3gFmsC&FcwYh-1TnBdBNx>ftN{Iwkhyf5 zoB$PzUDSd91g<$jULf*8;3n?^poi1=+{$5*yjByynR4Iq2*p5m8mz1I2gop;YC%8= z82cwM0Z3zXu|HR&GL#Wnma+#zqp;?O(VQQ^RHGgwp%_O13``ovYia?{pdX)JFap1U zoVR5^0$_^4`3wu(Z@K`;@;dveB9q9n2I|`hoCh@k_dq;jQ#!G?dTGB=)Pn^At|u0t zh#iiYE%du}SETYm`XFv-M7$P|MHw9tI=1$hz&4C9)Mkf7|wu`o^kSK@rN4h zA5+snW1BF=2T;t#3^>zHxJA7L@7H4~0H-UGFf8c=%y>hGgpl0^L zDR&Fd74G;elVPZ<$16$j&uB)0i45qycE3CX;+<#oR4HVuz+|H3>!`(aC1i{Mq#i&m zG2Kd23Ism6cYY{a`lzSis@AwNPuTKB!mxw+3AycYnRK9xAwn6jHvK7F7#!ykGZr<3 zksw2bJIMve#RHs!5%?VU+eP0Fi?y5Y0p%cBHu6y%me}jx0_fjA>iD2i2V)tiE|eB* zgL3aP_z&X+J}7dpOpniAtx?E3pXcH(q+LE z_)VJNONxd`*3x&kAD0ues2mZkl?%%(j;{a7)QchMXtl&SnmL9T-sP;=D5n8o`}`Z^em>kiC&=r~aS` z-%B|NE;&Ou;gWP+5VqYK;SpJ{D5+YfBTzm;K);UrIX|qHJt#Nv0_%Nb?#(z2OI?>~ zFa`@pM)>!i^(&aCo^W_!6_aO)sX0?6T({qWeU#BC$g^C))`|`VsR|de+FmEo5E_oT z1?Q^1)+MCPm##1LouR#sT}Ri?5CvRHLXXptUG5HK;iY6le*_K87{bT=!91Ky5)#5v z#O1}i0H9xjiAQb6%yfQ}7D<@!Auayum|3Mn&uA&6{DCg!S%2#2)+YJy4pf?;7T+2y z!GNVP{Wq@>9FT7~Rv_B6ipX$}Y_*qN&zV8LQ+YA!$2AoWK%wXc5q6ccG@)TH*iw^o zD!yoX+I3wHv}6oKI$#)=;o3A9n{?Jg9;M?eJh3ZZ334x;n|9+7Wd2K?xTs9|Mo{cN zQZ?le94ajO8tNiPc_?I#J6|6M!X$!uVHU1Xh9Lz)kpjQ>-=A zg21(XM8+U@z1if)d!!!;i+GxJ zsfHBrob3aqA&$W3sN}nR<~MqecAa%X--s&a_y6cp7}F$yJ>K<)lUEP>AGMf>h|b|+ zYXOEh89ccSmtN}(;+kb3C~_QUI`RMNbfs*~%}v9fUgpDIMSOq0oyIXT6W5rOyz7-;5?tNm0P(xUl1TNd{dY1ynKXA z$Zj@Mz2{5geT9wSgGL@0g?I7`K<&O7*RY)4lk6lUHXLnx-FQN$X@gJF>tfefY3uHd z^vcerMfSeCT5aC|1UO1{FOrC$`5^9zf|+UNLhjqB=;a}pibwk_Uie6AF{lE5tg)aKz{ zicSJatqLZbfD0zkyzLS}sc15e9p48>+c|S(J$di@4#Ta6Q)*S2&j5D0`8^0Q-5~SC zT)sbDi2@;oR)f2%&>E(j*l8D=kR$40r(utM^<&8o2yxEuD7Vzvlc2NzmBnVc!wj6; zsWl6OxGFLTits+*LT-KZ+?G{qaZ)eXRlOR;h+QZi3~(k8Uf;u_lY+swY!`4d-t~P%KIPu!7yp9FWi#^~IrA+3N^-LB{ zF*C~opOyh=Mp^)Cckg*Hzx!Q@w8{+}a>!h(i={{I0Td===mpj=0K@QtI*5jjkv4qnw)W9MSQBe5h>run}5vu)qc)DPGHSt$Kg%?cw;? zyY{@ZT#E)gY@o0S$^kmypRb4ZVN)H)Tl;5C z5@4O^D4UNL1G*@h1hOQ}=i5c~drfZNwD^E&RVNIyRI5+|5a*oTJ{>|KWU&W3cOGQb z%^=2?)!|}Ne~3yt$sLFh4;pV6(7>gd0FEe9$gi_nY>zrzEeCwDz{M`B+g{@k z5To7$70=1La^-;}O{XcgejXR%vu)aMQa5hXko31|hanTuZMW`)>HXzcR4Z!^yKP4> zj+VRbWP1DLZQ8Ma6c#Y~Vx9b6qgFTo^gm6YHfd3Fcizfyu~*~f0q7Wqo$3o1h z(lMQ{U6%qjT#n{#_N~IAjV{9m#^;>os;1VHBM5oETBY-s;8=OIKtG|1!zqJV`vj8y z-IF08Qpg3$!5FvTpSl$TqDaJ%0$2-x7+?9Z%|HzUP)}@>sov<&uNN;Tzt-hbq9hjpRaH^BzCL&gDS&mQkJeTJSJ=Yu{;)BXN3&!oL|dD%>MRoO+XY zH&8QAz^~^#!)~$lLyXYD{%qTB0ZKPdERxg)`A!Mhhy5f@`)9G7-!8Vcd&p*He4^9$ zAK*Y;bpA3*u6s*o8SXK^eS|F}{7TCM)V&TG?gsr4R#emsf6iS@KDy&uv2J~O<*Y)w z5Wdpyhr-*kS=sg6%!t6}z;zi?+!&JHT(WNo^FQFZ(L!9S4wJXU)X*C7-0&azYz5OUGO4v#iDAu z!ioIqsD`>AUF()cbR)t6(Q=EaPV;qQLqvNjRx(qg`i!gu`}gfZLxegg{Hv&CC^P1Fo> z?-=dUjHkZE$;L2qNQ?~hRN^|L>3^TFGrFIj;yV4~MQV>GxxL6d(b@joO;2{D-hOYr zlITqdkA?KNzK8N&&W#~Qlx5}Po5h9Z>up+se*XpMl{4__Kkg)L(LU{%-~SnA#(Ju& zel;e;pR1$0{ZTY2VQ53ZM&d=TOiIgaVwM1;3a~~2zD+-?=Ej?l7X%MyD%uy@Kh5$9 z>kq@p4XCmQu9_{>p(8Pgul3yl|nyMvr7r(n+C^*3g(Ii~}UgNB=ThpX$A z8$|WZx1*BJp0kq-;9J!adFjo$bQ8WvwMVZHpK}}H3nFkwO`|J)&S!*#d`dp=@4vak zxO4!d?qDux(}EZ=bEi8n6^-QI?lD!!hnSguvvgZN0f*~u#RceGV9R^E(naif2wf(K zRcl!8>-VFB3s{u)kIAga^e|{TzHSt37xaWcyWzS+vGngCuL1L6mt(R;A&3oVb?MV! z+ofuB{Q_C<6G6rH(c#rn(1FujE-k`f8sAM0hKvC#8gDi+iuw2vNOWt@;xxoJ;*zcS z9~w@wDq3wo)-(skzxGI7EI2XxA0d@7d_VLB{&d&f#Gd|-?bd|pP@V?0fFj_aT@Raz zw)8{DOjWAGI^HjBCwjXoEj~9Qn@e3x=fjKzhUQ%s2JBgnoAX?&SLTyU`ruo%c3nxf zYPh;s7<<6IeqtjQ^*jKX$8B(T#M@Y(yzI*&;<(e8ap8jgv35D=c3F zrVha!_+K{Ay`|0+UV1z*+&#c8?6rK<-Nw?@vB$;-&AC&|D#Le%+0|mzbbno|A4Q5L z5AWoNZF_%shXOKwGMvtrFMxbBKmADdyX3D_AgaV>G=O*CuiI$49qnsI!TU($RyT1z zT{QE=Ja6OuwHfHJdE$!dV%QKn`XkQyiq)GWm1zMN zB;9?!=QB>k>$peXqqkf((%R>84xV@<`X5#fO@*JtITuUV88^_g=o_HT=7eCF-X%WV ze)=kyHB#Y#m1iJ@X-fX;&1i9@1eUpIG3_i!)&ONhOZS@GF0*cXW^bZ6)1iJY)cPj>$3Ae?F&v^F15yH1&J`rkbr+eM&P^KD&4Ln(J;*Fsu8{M38 za?L-=y7H$#bbQglFL;fpT=BYBml@Qm0=!$!XAZ14b%X9e`lRVJAL7#ll;Rxo){n;W zoEF^DoeV<|13A!*Qyti?{5gLb2JI!{F6nr-Yt#aWQKNRU-Y_R9$Gy^Darhz?0{+g< z_gv_$JJ;fYg(lcmHAFm?bl-Rw>P|<2PKRoABLZV=0GJ5|yv6581DeQ-{)+XycI?NE z3ts6@jLLZ~qlJJj{ikyWU^`;@(vEc3e}TGzrQ4nZh?NsxfMUsMzC@`An-#0~YEwTd z&3^QUKc0Z)Ex;y`0_x~v!tRgVxiVZ%up5>ab<>1AIp*v!tWLli)|yqwm(o4RIfSvB z|M775MYHG}Aky~MXTP+gViXrq;W%%0?FtAnnmXfEZ$tYS^Otqqtk|=Vzo~M}8R**{ zV6uJ=UET&ACAeOE*OM#p@7exTad~N=8VbkL~sZN5pkKeLLl z35_`$w~JB)`)m0m;8Yx903gtOI7VhcH0Pv1q|%QDmCc#lAvrO^E5MPSl&w=k(X@ck z(7lV^W$4#97XgHhn{8944V0nsiIL$oB5|mCc#T_mgoDkCFo>VZ5RXJ3*!qwq9z8}f zS@4ViW2kAuBR;ErgXs@@%d%k0bzahzH)08r&h1yclj7DMz- zA|H~YEe;>dGMqEJF+YHSCe+mG80dncK-;Dd7eV2#|Jmh)h$zQq<5KSNga#VaQ|q*J zm#1%hZ`-v(u(bNwRnG(FP^~ay%ChXqei#7lP?D!k@QaEZET)L|~!QdSo?_bvGw| zZ6?yg;}XrcKHPNJBn>jJZM|-|$>N=5X`gzH&ymb>$xipK7;4U4^i!%~n#*~ea)}19 z+b(m0_1MtM4N4qs;RVjij7K*56yyii|wzwH_JL3lvpp#r!yaUdEsAb zaB3`D$5^Bwj9euX`BX$?$zc+r_q>4`NUCTLit0h5@}%)_i%9P@k87u|nWb?aeMr?DJEyWGhsrC2Uv1!&^u|x_MJ2g+svn#QIGD-l*24B@DLCE?6R}e@!63`C zT8<5pc@gyM-TNGfb6(3sI#&*3u6=nO<3se;V|?>q3V*k`vxybP9ZO%YOCF=s&67bs zr}k&;oa>v}1;_`Ve7#)f(W8_S_fIXDW8A7~D138KAk)JYOW&nd8!2I){IK`VI&S3* zMjpPooByTh_G}Ky%xU-S2}SfDJJN2!Pms}2ASmz|KJ!w(_jwe2B|J~w)BEO=g@?B_>a;Mvb>y6U`69I=Pe7u@#o1qyqCv7R7^PrP+#&RbO`HPGX zYH$)A4a#1!4FwW3uDVU>lo}?#eV|DxVkFSL20~2*hDBRN+E7Ud@Xh9m$srO1{37+4 zg;mz@aL_8^z1}1zTz{D$r*p2{&gVH#W>2(Lw-KnYHDh*&;=$QZ9r)~qwV&GBI_}M9 z`bz8ZG^fV#gp0(dar!F?73;Gce0W~wr`Wfwk&U=+<#)u4rjg*#?oN09IER9e(iEcb z)LsL7Jp&redxsf)Z7d!g8vVz5)2+}^mK3N4?x13q2J%&)lBoSPW5v2NY0~ah zW7(hJ#;ZqJzf#j#x*)CSZLUP}+T_$+GXS5M=a&wn3ph7#S?$(i#E9^6)y$F2{ndhN zc5V9tcwjWxZ1NZBUn=J=#l2z7t%xAH6BuDy{xoFxIpJiZ>B-M8d>3h*-;4&HFFo%V z&`LkK#fBJ_&s3WF)yaYeS>FhaLz&y$f=i0^lgm{A(J3}{Q$P^3dHx;rw;)mtn1G z8UpmwJqJpZ`B%fYcv7ziht@$j%gEe|4izB(DO)*ANIA*PJrEtd}%W_ zvQW^wYYn|+B6Tl!@^z9Pz^t~#+gX`vBQ3LGA=3bgJ$?FUNhzH}rp!|^9$5X+xQ{|}u z^}e?n9%pgQ+a@cR6DI+S_Ci0CBAtiwhsQU#rk7Z45M z>euZV^>fP%olb^*5gAJ3LrXiKH?~68eNq#xcumzbP@U>RV%+}bL#vuplwCi;=pWlM z?@`mGuNAybgZs^1jlzY)#*~nVlD^STAZhsU6T5EbD-VVSvWPBr2KkCb%<(JJ>DaxL z1pYUS`dG^mX(dI6SgiYgH~Tfju%ZLWA@nUt43otRUI%<%Gk{C>!NiQ^(+4nhVPmmw z=`p^1^^0_8g1#}g2d{b=_ho-c2er@VM3^tyX;t~p+nh8LtWHw>FTd3CthlE^t)*~w zR!|$&6v{H!G&#h#xWBNzJ$kVVW)a{Hg0nBK#;G}ivQ;jwfdcfbD_o!5Sa&n`-us2W<2oDPq?~`g_p62F_V#4Z(6{P_cP2|elFt&T{x|Ae zliz6oFaDEVvFH^)ZTDf_jZ{a2+*gwRgDEwHAQYx{gbgda<4TNofV48@{CL>blk?$N zFh5~^CqPa|j>v;8lCN3;%)~VzS_0~lx^?xK=13ahjCZNud1r>1C*4nhX&~=|5`A{6I;LDVIGQcXWj~^ewE;u&nEWwt9~Ltf+((owdYVpm&n#vtb(QFN z>X}Cl=I_p#nuk=YoSO{sjmsml@G8a>fFQb?(CZcsM=Kc78*f)Ph{$(eq3HZQMp||h zSDa=LQ_-pdy15uSICj4sAwc1y9S?^I_A>RwjatO1Z%><53?hzF{@DCPW|6pGM^H~EQje;|VW7X z8-rh8Wht=FAc`JGqsG#sCo_m)5^U31If}Yhs^+*@e!t?VE&IPGniBvBZ$96Dp{uNJ zlS}@%&J?{qpO3rVGiGXF+%yCn+==9WV8@L^jnTRE|VcEsE%T zLMc({{At+-0$#FVYw3!m9$on>s4g7xzmU^*gl+ME;r@fihQ@=_Nl9faZY$x%PL} zD%9z!RT~JIG27BP%qh={%r&X?J+4;mKvR5Mo=St80MPTep6&+B$#%6sde5Cdb@nMN zi;D|W99;4RTZ{*PwrPmSCzw5Q?_(OI5FN3BlG;?a1$i8Nc8a|Dh4-u$= zlQ(wO}9IFFAR!%c#!1-;a~sHHr0*@s=`ZSeP1mz9lNv4-XFzy#uM58vb|4MFs_ z^kbd)CdBaX#5Uq>LSEsG~aCJ;4cjgLk;pR$BQ;=jEpSj#6qmBDX+ zg*pwQa;)sJF+=xwtw(3`9w}^p z=r=}H{Xf(~)W|Rrf}JCS6Mr@^$SNUFBVY;keL=($q*ZgiW1jqmxQ|OhoAfrX>Jo!3hWwO0S0sAovB;Nq)zz z2PloAiyZvR{v~PyKtOJ)(7qzVg#|kJ+QgvznHZHvw@J;z}zP8 zCp$@OX6$AYbYf7&L8^MyV8YBqkqV<2lg!%;Yj~-&S7p|S{||fb9glVYzK@qwR>(}0 zh{&q!l@(dxvR7qfXJnS0tc*lfvdJcUheEPvL`2zplkIoD?)TmMzQgDD*YA(-_)w*)!0Qpl;cmWxFFN8fS!%Wb(p2K0a6B3 zlQUqen(9iyoB0GAZn{4e%)D^`Y^7@gU>!-!(3igc;OQe;a6`mBH||kyfw`;+dJ=+!2U* zx%r^Z--ocUN`kOc{3zke&P$=53r-Bd?a&vq2fhLE*g)dPARvI&c@g<}_|*_}EY!8U zvQVQDE%H?^yU@`JTP%F|)x0A?tVnz7s4ORtEqgGCvccDz1o=)0w2BZ=6Mo;}DmxBi zH6rB!kK|kE0d2AqYu$JV)1H9Zv4u#A#e4Zq3FmGfb)-aeZX)uYXwH?QMZkl*K&;Qt z9GOT?pv^{!Rqx)=t$ptF##<&CR1u=heh>)J3FBn`U)~G32QJ=(K%7B1GQ2a_tw!s? zB$yk*Wa1BPWJ?$zq@z#vk-E(kcl0eH3O65k0?xwM{PtG=<9BA-_?`^YzScnmF$j&D zSuBqiZ7|)Z)vja73mG8+O*|z3+;DqUjR+ZvxC2V&O`F}>zI-{pxq^Bv@1vihXF?nv zuRXq^@#v9XYIef6>7_5@KY8h_$(Q%`*0ssYZy(&n4Dmho#vrYlT(K8XVe$&ohO;M$d1a66p*Mwp zPISdhB$t@s4sVzg+DVb8_j*C~_A}?I<2*HTzr$Dzq@8YcdL{Gln=97~TykR&Iwm;@ zK|jXd%WQr0FTmv!BfY1zk=|ZDghS4;v{1pd`^()u-o(Q7rs-zSnQ*hdz#2IhGUt6h zL5uwHSB|Fd@AweQoj%8F(0FTM*Xm=;L$Lsx*H7%G@{MCL6%AK6mfOrDnIgGu<+5XY z)ZcA7G!;v1IR2j5CB%u=Wzf81-&^ZY_>ieTT_@}OJMH>9 zYGR?oHoHk`^Jy>U*FzAcJ!5&kNy>8k^H8-Jx%3sc=f_k;HYG$n=+p$Rh{-o%H!h9D z65=~|mR7B-U6=hS_+Wh2z5G|-x#nZol*i3HNh&!DZxf_eqrdy+T7lYn>2kjGPLtOKRZXW#l`D-bZFg1f=QSv% zm`51?AaK9XS+Q`L&V10O!^-(dZtY#1zwaD+b~M72UF$~o-R*m&T~=A+8mA{*@`va0 z+TUdFjFrN6G4y1&a3aRF5+2#N>(I}ouwIpSVmB(EE57DBoQzNRG#ZiIsi^Dg?*Wq< zu%u}G3zW=GGg-yT^;~hCJ~N8-h-s$OmMIb)EBZrbpY_GL-5FK)&+i6{jJJzqavS^1 z=aOy*#I83Q-YQZvTl&1U<8Ws}`Tn+Pz3fb(S@+wEp_`K_@~{0Yx?OxwxoxKd?0=37 z&3s9B>Ft>dGgYB%o7l#?Z<}_-b)Rh}SKx%Wf22c;pO(cJ^IR=&IbPG5P?3w_$;wd8AQ6mgxsa!+pLXTr(%6q&Exe26O@^dvrheXHu_u8rHqDo$L$H7k>pno_;-hH%O(s5 z@5l7n?iNM98}L;mBdnSqv#67a*!xgkl{uK)K&6?r`n}y!ZgtnJu30_$t44a)`TY?> zIhDWVU)u3h*Byq3oQO*GN6O-FOJD1`#68}uKH7kH%4KWjoKwI+^1WSN4#MK5=O13W z2JD+pZhYs`xUzviTg#HslcqkB{I&9C&WLZD3;X1R4k4%U*K||A_Qbn3&3I@q)8+lI zm~u+{)s9P;L7#vyeKM|Y2*qe&stBlH=V3dC_SAay3QYOyF@>V`VQ)^y4*l(*w|$A9 zHGy*;g#CFNUKUSBL5BlEMi8$&TJ`g@HQTH6U1r%8^Y2#zFh@QMf-r)ySkpec;@9{W zO^}I#K5L11MYi7*CezIP2u`%kJ2?od0WGJWn=IEZVl54cr7U+OY<9NAup*~qjFE){9pUD)%Ni|blQkvHP{ z_@dH!Wo9U=O2qH%NV1u$J>HdPJNPJxo^_XF%EH%jpD)Z*gR(8j8GE8*z}(aJ;`ie~Pg{&DHbM9tEE#{qLE6*f-q9G&l#G zx_jlSx+9{)2&~U(U5@8*{7|Uy-x?ny;TI3$KdVgN@R{ zVXb^*VTNmS!S#u{?WVPsK+pD9dj9}4szqr2Q610BYfN8SzH|b?jU%65ia#zHS)Yeq z@^`7ZZ_e8-T%ZfS!~c4|lRyY(x=dPCFB zv0KmDWoXlS$h54_FnLa_M&yfVucy?qPA&x(V|Z5uyxtgpDp;iDX#RTj+F8bn+GDim zV~e*nHgL{!OwDEQ&z6|X7;c@R{(SuF0>Q}|28NQ@twpZ48QyZeQ8g?xMfc`HCfw6) z=bE%AOFsG1EtH(!dvHb8b;M6FFyc`P;gCiCn1$0=|Ant+`5KYQ6;?*t#67(BJh z>NK1Yi@Q0Ve2@G~R^fME=`8zc>4CzL;wJw5{<2*H@3#DDvpk~1P$)1t5jz>No zz5J=)w46DT>w;@FX5a99S-&|5ZAkN^L9-3veZ_ERR6dTa1Qz77Q*g4(!uyc`v!cFJ zkIQUtzq}aCya+`{tH5@YK}%%**EOJ@wsx?u#hrRrpio|e#~1SUzQdB3-RrX506xpd zgXZNBbiu#8Si^qPduj+za4%;EtLgFUU(&JNDN*QA(IJF!kWKX-eQTf9BqHogf|=!T zzeus^-D|}(=!LR#;X(vh8=U4}^M>Ru=g0HMjbODovp>(TXLmF!qhM{{xXM;`w}WkG zX03am7%SU_Q!o7C;Na@}uQA(3i>douGEy|J1BmfHl;uBqeuwQ*EzN)y?v`Jn#nSa4 za!Zq@Zc;pNj)zK`7NsN~cXugAD4+bAD3o>das9QrmD~2K#@&^7iT*0Z1r_5Lq3vac zl}f^%K)6W`YWV4ib-ewp)hWfK8Qq_>s9f|C$C$j4=@aI?g?2Bd)yAoh zISblQNZ0d4{o?)Aex}XAb|+*i7H{Vzt$fVM-t~P|JGSzTNkcQNe-k~=z}yqU>#i=Y}x2i zCy6-otn(vR_7^U2ukG8`zBTdNksC8ss!|d4;BuMeBEl7yd0oeJy`;~c4r`@lVPt*a zN#Er4pS?y&VT#FZ6x_PQU6gz4i#`{4`t>@#S!fkF6ZvO3sR^(5oQ&=-GFtpmE@-GJ z9QK@83@_mPeigC&TVRryj=$;7v6*kQ zZ+iAe-$d{4d%c$kxQ9mmC1dUD+x-@nN42Wq;)MPdE|lf#tk$s-yTn9ph%7%c4k8y! zS~$M?WA$Vh%Jqq?y=-m#+ko!w+J72@C@8_@)Wn46yb$86`JW_SqN{GU;Sb$(q_eZfS>!MB zJFu2Zi8Iy@i*jluaY&c7uYSI>wO!ceeLKa$Wo_a^P%t+nfX6>bzO*Zv(uv#biMn(B znwHDpXDNc?_(r{tk4YL$uSUo@bJJTEHG4DNn%NwS9m0ul{5tyZYH;!{A#LF zlu!Dkk0cpwb$hXf7ZQb>mal?MRAQsJ|6Mr*Sz3Rwta-utjHib;)6!)#i8N88*l5t{ z3qRXCj;|VErc{3}B_XYte1}S9qBS~E8&fdYhrO4kiT zNKDTca?_lT0eo|-Gas}@rY2bgp;GuO>KOGF{fYgfrTa4!0~HGjEi(`BYYhbqnwzyl z%N~1=$-3n1Zy8qH&9-YX`x^D|9EA(hwqSuHrRk@L&hIvhUO}F%`?33N`!rU4`C5aP zqgVEqcK1&Re6#(mDVXWn!|2x>(WPHeB-q+Nvbir8w(H8&);9Ej*{=Gw#qf;w%jhpR zq_8d1XSSp*p7by_6`I$(gb`hO->bLYpxUi+b8|D4#9|=Myd`5@OVIa~-E39gGTp%b z*&#l@Cw|0u<%tWd%;d{OD+%TgQ*KTFZA1VER}S0iAXkMq74zeLvV4Ofhu*rMnn~gF zg}9M-w3b&%O=}9jYe;*_+21a1@(zeCkJj2?)+ipAyrbae>gvqd8*>{OiS2Lc- z$GEz^445Ozu4RGIPAjr}kio12+Kq=@MRRu1#h*|KNE5B-ef_cq5mOKdM{OO1vbAkMB|HIOA5IwD@rNk70&9%_Hf;(1(Z+bGZ z!w1F{JpnL%svI}h#caC37+P1i_%D1PqeGop%HBx>Ho!OP7`FZdqZ>%gSKUrutY`C6 zIN8e9Lh*6Kj&I1wq9cbztK*w#s0y83M0tnR7MFRWztQQ04fURoLL((EZ=WX^{7wU3 zn6EjkWRO4pnV7#cY00I|o~Fge%J(vs;U#&Z4djQ(N=}Jx)M8$vBk#$8gq&?jUpIxY zML9~{fm%m-1`fwoidip`hf2kEGBVs7>6)C3OPS6;q=tsB9>O_70bmsnOWNwl`)-rnjy~u zb}+TA#N0b8+XEi#JjQ8ZV}8S&dEF*Bh2MuPQwXDd;nvp|J$WP&AJc7aY&7#<8F(@C zwQex7B_(^%wZgQ!fgbZL_l>zGlHG6Nl{{%p={z)4)CNt|ns4m4R?f3&Qsqwx7)R4< z!eD?f%%xW?8vv7X1Ya|$MJu<9D$KR_+4%sF4>nB86)18WlF_`Xo_pu71OR)om z3SupfhndA+Bj6O6;=Mw|Tl>c5wW&Z776ol}BAMeUG4G(u_uli`PV3>2iFL>upheVm zF!?NrzM!dntn6<08h<5Ev{Mt!Q{?Gm$sPJLtZZ08-%y_H&4kXUV|IjG?|q$c{H|i}|w_txlMI!6vfWcTv)!#Emfvya^^r1XzvY*sP@jv`6gdWeeIQ#W0!ef>-6tb{HN&DXzuf~B&GlHWC;nNOR0EPzsg;GQ{$w-<&TaP zETksQ1AgmtY8)U+g&_s9131X-GnuB^K^_=UbD4+J#@(I^VV29BW~8Exe;yv->2nOb zkwNO*o1t=>Pk=Vj=!3n*G74$)?i zJ!+LE!q1!^j_UPIdL(5kNrQ4jpB%KYG96~PE`#JL{-)pl?gBuXYN-;{#nA8f%cKet zJely0^OlZ_j2$<+d$2i?nt+~{F7gqT>f6((D;=kVZqk#Pc3p`<_tYW#zvojS73F#k z4s78W4NrbdZ5BQ6w&2i@2R>Q?NQA((rd6Qxti|I!lOwVue9$KsX}u!kj@Jc{al|Tc zsvkTEH6sl6haGvUfZ0f(9t=`8p{5er|}-jKKosN(x&ZwRF{1eb^Wp3R*j z`X+Aavf!PJhkk|gY;SIo*d~0o1&|p@FG4TxAc38bAv}ih&+2t;x8KCKU@Ia2oS-sHYS)?O(0l-&7fK0w4W2?BvE9k zL+Zdcs5KK6xpUa4FIU&zv<$YT*255*m%m3KPjGT%c4KMevp-odMp)MUC2hUvM1>Hu z;aZ=i#gEZJu63+$M}BW?|L!CSgU!%!2T6Bmo-n2?LED^Qam#$Y2Rb{<(rK zSxwViG04ngDfxl+!yIDVVLCZp$in-TRwTI#=>&jj+WQPSN&G%g+>m{p1J>quU`2V! zzbvsxHw?OV+4>81z46cyhckC`4=x(N&wId+4>NO5xh4`ldQ<}45rqCnq;mjX7NH{) zYulp^+5w$#mV0WSyuw6ZBmPz>-2`EDELWc%VW*GocPQnQJSY%@r3vXPkXXtU^%+D$ zIEEsp!Z~&$)MA@j&Q--^3d`EF=>3+TFGvb{G^=GlVvw4KlZ;2}(R>1rA79WfQkN1& z;avJOh$W7npSRSe<|X>Q%A>fd>bD3Szj?vqbW~!LNjQ`HB@N*b@8Pr{I;(SUy$qcN1tfQkw8j z_8hVxLMxM)7WF*69n#X3pdSQLTL37=>4^eAIiy0}huG?2XlrBojki(U=Iq$BmYluX zC-_x0%{WXSRjZH2!wt>^RRws8u{*#&?UzF!OEtg*#s;8tYCq8Fot3*jcRT!}cFOvD zN>=>aw=-29@V$lkRllGwMdbv>w+{EkObEA{>B+iBA(&=Pm7?$JgUB}P{>^-^ zPi1sH_r-{*RC=GCZa5J!K49H;CIzLFu^RKLea zjKMDfJ@xA=LD@AAlQZ8689y1!^}9mi5tLzr+yVpJ+ zFC$TtyrJWNm$VBV+RUd+C>lNZZETLcK>d2+zg*+6=;3}rLru_SFn;nV+aZ68YrqlO??n~fIOK|`b{Q10WeSFp_9Dd?yEix`mJYb{0^f5cY>*7{c%_0#3`NG zO)tnKGf`YVr+dbZ3y7p!CO(x+b?dhfGf6f6w&Z15&t0!n?BqMhTo$`q9`E5v=lo`r zn8VKWz|J&ZhPPf)$#D8hnyFHVVMVWe zhSPEN!oOP8zkgJcM^mm*!RYY#9;cpU3u{dND+Jy9s_=Zg;CBv32a-Kux)3si0p__> z$LcF)PM0lKV`rBPW2so>o(4m2Q-`e6k6Gvke+NSXn4SD7sIJz5JF`yqHV|y+&{u{) z=u|;da*qr6)m<;R2G3dPoFR=RA7Rszxb8H~cy2uIraWo0bt-m0F$l;UW-+4^mEgLP z9Zx+^KtQRLxMkmio$GIw-}mDANnO||$$gdiNjdowK%VCW+1j)TJ?OC8djSMK!Z)3U(*hg8Rm7Sh%Wu8LUokt} zjq~vqPV^DLBKg%e-U>Q4b6!3hH~-FKl;@jXlj_Rc-Y=jt^YWCh7L0W`d*$>Ef@a9d z)2m~p@f|%%Qlb1DkQ>I-WZ6xX63c8mMnbFjhsZ<^) zXT2fPFy^zHErC$0YpDvk5%K{SGD(%vusF4psQ8>d76lv*tJ)O3l)>OJg-+Qp%s9Jp zb;MJkFL}iUJN-=tF!|3K%Y4T(vQWfk+bEy@$)POLyQ zu9hd3sr*501332R0%s&&Pv|v!HpY5im52+(Y;WKoVi-M@XZrgQ`2Bc8Fwv-j=M$wm zuV>7vTv#lO1jzmMT}w;b69pD~DL<-GmjeSP%e zE#-n2%yw^%Mf0d(b`*CR9v>NOef=uT(IhNS5%dhKi|G^15io3aX!9O-S2)D7V?=XGL=nHv zlsv3WN~d>INEM0qdRLN`z-*R#_zJ+D9`!Ubz1Dd0V7c(urTv6N8mVb~b2v(oYJ35m2y;V@f_q^ zzFqBy?sepA`yx@xxdjDl2S-W$b@nVlpY@mS{orE)XFV4`RW}{AzqkAOC8btT9T{w` z^+A`dYkzC2E(`5^iQ=(_mAeAE_n&Vz&9cZnYD8~o3DtF|5*M19pBe=9$_ zWo{pUK;rf&25g6Ipn4g@2;@dAbh0u0JU<(#p2>e>E0(jL4&~EtJMx12qnjA4X6dy- zQj$#rZTNfew?k%trtvJh_EWf znB_46+PD@679x_>a2Toe1VmJI!D`4(=vUqT5`z5Cmz;AJG+1N;PEQ~p&u+l-^OKDqNCY9`l~V~sjQ6jx<+ZUsuMQD&mm@&u}qB-ra>bd8M$D+A=_>oqbbBX0J5f z(f*hrVnjVl>k+g3iBv1DHS_&)V?66D%NLMP1cFuzih$Lu@5%y@a{v7L%sY_u4 z5iLRA0UG+CY@ut5iV=OL(nS{44QwTIkRboB(BqAW~LzZE|8E+P)gp<2>bsHM(> zdppG9of#ZG)khHHyi$IERO;M%Yh;BbppJxavl9_Ht4vYL|0JmYi*=wEL=<}>G8+VB zCKzEHsUoJeD^Q>U0_CJtdDX4Aqc#|dQTjze8qTE!k_s8ytp-#q7IvjX#BSCt?T|MR z+(k=ma3}EQ)81bWKyqNHdEil&;$dD%5cdWZDixLiY9mLSAXXJSCD3JK4N~h_73<@y z{|d-_QrpNH)^Qf60u#eDa24QSTnH7_r>}3s)F=OSjQ{+oe>vxeYM{<*cU^+Y{!gcP zO>r+My<#rv9foF??l1iiQcmyI++tiT>~hx2ADd%tlCMy3zNr+R#mj2ZkZ$dm3hQrI=ZNk<8l&=8wQJ1b&Zm zT$}b(tC?sM08Q^I7Hg%xmz528f)}qAfTU9g-@|?E0W99mAX063{>fa)H`=FH5b8G- zn2$&QI4s4|_j~fybL4bxp5H^%PJ5G`Z5mF{KB~qJ6xzg8@-tO#j8(B$_>g&uEf#ow z+EF}HJ+(4oGmhxX*mtb(?_xxtOnRfeuMH)R&t?7M9Dp=pB9|f=3V!^8u*y?8M}}3##UW@&$y)8)fj|8VI-w-t zjTtZt`VJonc=4Z61FAu_=;$1|o6bLwBD_c?Kr$&L>){13B!B~;+cW>S?jKg#1SCO6Kl?w*z+drXk-UsG_N&L?s(*|z*Tb!_a z3Ej8^bR~d%ll&ioG$EJQGXYC1Q?tNGzQVzllh`*y1VS=;AXBkD_ks8`sVIzXHq<-w3)ao}74ekRORT zn&iZW0qX#J-FUKB$ zz@ZiLza-B!dwStf?g2SQkPES$$Ry&!h0{}X5?v&$6p38Q5oB-r3}OG0b|)LXx&5lt zi0U}IsjhF~rlacc!gTD3s&gFA>f+DC-e@&YVhOFlvsyR#06(dThRFqZ9l0uEpF#BP zBGEtHSL?aRXsLQa0CO=28M4Lw(AJ`&c?&GP=PebFpRFa&n7(T~7l+Hq7kMVPbnPVR z-4DZg$Xt?tRhGYh{2+2C4y6=7@Vt+93xIV5t3Fn_Xca(Qd2IoMwobN87YHU=g29;k zAeih(8wh~=I`6D4U?a|fN4HD^lsv6QmXDCsqhV9Jz3(w$UNp%s@xPFW0OnbD=nK3= zf<&opr#i=g=;oizsev-40Mw7jnShx4`jmnpxS#-Vz5&tP6Nlg?d-De&a*aoy6lJg? z;5La#&tLLZWd{ zVFqa)rkG(j#JR|SZVWy0oM`AqK(q3Llj=f}CkbBRYl>e!l{p>wOGO*;tE;{LVp!>96}DE&-VlLL}LBXTr^Zz=~6GW)sNa2g^~!N zbv+5K$|=ane?Jc+0G>by^{~KZxg$&ngtYiS#y$P(b|<28dxr9aEt6BH+Jk!+!mwXZ z@UYG&9?7*x#GxSuu@E_DfCB5VS0vqn*%H=YQst3mf2`DoN8pakKW?4>Ub4Tawm_kl z>TI@zHYnS-pql&w&pc!j(?vb11};S|7xFyzpdp*rlL>SYm=$KB9EYjZ#M!tEoyfp& z82<_G^*NwuNyQ9-6tuoA zB|N^cykD$_uW()xBNmbXM_^;V2cT!9RJ3pEpPKYv&+w<)aVV9*T=d!4Uc4Lh1+{@x zw_KpN1~pzeoNVZv4nUKfoZV;<`fXNW_mUp>>YS#b8mPYsmJ4$T|f4GOXLxbMm{%ptQNjYT5=D$AeKR>!9(4y#zL=Y(;=%Scf%2L!;YN3= z{AJ#xBO|!s4uT!G(W?@eK)^s`AigI$))kLLe0;_im(m1A-=VSn|?!E z!3H8l5vYi98@gc{MAC=%82D23vS^;~u#om?6H|UO?79suolXL?aPCDWF(Xd{fK)*Q z5uh&-ZpzKZ=OZ)&ej=V2jC_zWk)nfd15V96mKS5jkRJ780lroRFczhSD9d%*hewdX zLQz8TnV0;+ug;GrK7i3Otul}@h6H9&)P@PoQeEIY#XA}24jz|QXoFUQv(S{rW-{Kp zu8OJdim2%|=Ry04Ts{u0v*j{!(Y(ZMz{f?2=`ArGO?6Y#pv-_VP)_)EZUac3m@c1A z7=U_$jhf_*4rK^;D0kF#9W;R{B7%?nAwpyrInnlHx@5oSLjQB4qc9PZpueXK#MG;+ z5Py}!`mRDR5ScmYdWfF9l0IxAT_Ob z{y&qG|M}EQZq-Olt2U=&d=yp!h8SvEiq`SplY9PH2C)2K6Vxg6PCBrGF1pTr$1I8N zisxFq#wtPWPNOT;tnmf4qQq=6nNsKImtaEgAkU)A@jb&UK}wt>L%M~dSV<-;K}mCk zANvNOsHlWUrQkxv!uKJ8l|bwDzU|^zfs}{Yk1J;G%@oV=w`~eO%#wL}>^R!Z%a@3! z$qePFgpmJ3$H2ez_YYsfo+_ml3E(9jdS%rgUNMZi>G|g$;9o*%qAA3G+v4~4I`~RE z8@v)@xlZ)&Pe49O$P81$(`07l`uRhzv|&hiI@I+k{&PrwA6p^qCF1m~wi=^Dul(~wP7iOTwhG9oDSt*Vzf80B8q2f-X zn@|(K0gVhujW~_kg6j$wmB{>So zuf*BdH$XzT##xQ#RH>w-4DsQg*_7*BB=)^w8?#>$ak%m1r4u}4Ot{<~PgcWaFQK6`qbt})rud4lB zFqG4*LAOJ-2P+yDZcKmpeH5_ORA&aE#DR^2Gb5t&2gstN{MVr9<%ii7P%t+L?r*jM zA@4ZZ9uKXAX$Zn91)X8a3(`Z^CG4bwAh82K#gVnH0wQ@O> zI*td*a+f7PEh1(l!C8-&CvXy=h-_Xy0D>Z=eJG$Qp@Vdq;?W0aMMQk2-IetAadTM? zj-_ejfe?&w&^Li-%yMUA3Hg?AxnlTJdxh;lYj_jpVl_M^khlqqD-3=xtt;)(csM2u zWn2@aHRl6&Yhc+9YVS7}9)Ym@E_J#tjtjx>73ufVWO6s|SCu;xww)IaStnB$6#X|~ z-~*)NoBsIV)|a|gnC(0VCQD~v2Yl<)7lO$ZtCkxXp=-FSk0&z&X1w*2Bx>3pPm($T z)tz^jevIjY6i!Ha<X+_|PiI!PvgdSO02_rf2i}f(x|n`ju~tJ)Hm%X-4LqfZ){> zwQq_lMJa*VWo*UX!RMaH@qJ5MyZTPU)rn+=aE+v%;lFc$>wcVYiL@mUZQO6dUG z)0z;2fH=TJ>P6O{h*d`~O|T{!SB;;4m>FbI4i3fs63c!IswSOd2T>WVS}q3dIAPou$tcW<4Zx0E&3* zc~hiPgb6Y%oM$v;sikUWOkF?y2h9vLmweuot9t#uD4B% ztz@pSImP;Ci8~ZLR(kqLRwEdrwM%sonGt_HNJ2vd;LZ?{eJ|M_RFcRF1DKUkEh5e{ zyD11UC@IkYe%AgvNTDpuf)K2=y;70FEBgG0F8yu%KrD{x6w`o1ub3Q9Llr*%b?Ep1 z`xpOvuKv3*x&6;~iBX)=B$%vSA=>!cW_;@#jv3I1Zh|_vzpt{-p!8-myQ3uSf8Xfu zAL1UL1muB?!hs602Td2cyW)TzW-%0zTpomu-yIO#?rTG|DC@-*->jhWn4&qZ*d=Xp z=#i{oL=mR7gD`*}{GUQj@PH#QWASIrS5OTM+Ai4~Laol&Uql&{2*AKO1%iqWKqNyz zH{`K;QnRIJO7dj*LVAjla$kHgqBRB~eTI2M{+sXIVW0Kq<5Vo56l&#c|8* zTk_O<0T7aM4L(Us*Uuj@l(6A9F0nh_ayq17oK)M9r&Fh8Y8zS@q09ZVR@^O22!F9jQb#KL6a7X9- zXeU&=PEf|Rz=|$ujVL*WcE-6vM_fU|kzdT%r)U*{XedBpP~Xj@zP0qybq6*s918A7 zP17vlEZ3ooTL6{BScmB}U1qsDCn)&$he@mInTIy& zxR5OT<+Gulcc{#)7Oq5^zKTmPDF7%Ghfz&AsT2~3@mJZUHu0%bc#!25<{^18`3p%ckbv^-al zL_Bzdxw~+Gfh>LC{@^P$IyuR0_Ps{c%2z_^B_SgjfO=b)&>Ab$XIH@K!0E`lh9JS` z47Xaa68WiTo?2~_R&9J@cLCNw0UB!CD{Rsp&^8Eqdo@V$$}2+-zSe?O>Vs9O&Am7k zyTB~KV+gp0w57G+*qyij_gE`Fk!WVV9)+zJF4jWw4hUjs>7-66$qF=?4Z@!NzQrn@ zM!b|AShCs0G9Xc2<-8q-mE;fjyP2H75b+H>c*%LkoRxhCnk%qcJbhL$0H znxUbo)`yU=1zA~nB)4GkvBu(A+hmK~Y=$tbputYbfD)DksR`GO8cKpmJ$bqFArCa6 zI_i?-^i0x!mJXC-;`^k$M*#K5uc|)^DU>GD8ID5+cqD zl4(tlZB(2k>bL2Vfw=4vB!aZ*c_o~d3CDKU)&j)Og3LSZFxXVs>eF^VlYUX@{kYy<@E}dHyB(A?L8t(VX~V>j+yYq%?GO`Y>5na*b3SrYU&n4p&MFqv^T2x5F^Hk{j6jcbu+G9P$ zH;eKT1T5ulzoM1R!nyo-%qbx~-XI>Oo7|X^;+@IUX!$i%tG1^xP{Te!`XI(G@5-0~qxE%E z0KU|G@G*ML1_By=Y)$(IBqW6wF_ZM+Xp{#Ph|)z867r&NrB63%9$NLqPi{S8_|lPP zSLVDPau4njX_ivG^<-M#t-)WF;Wk-a6{SH37fTd2n)8#>>T=6tC6i8TYSi&gk-blk zCltgKEReV5xo289$=?1qG2!M#lv;B^@8%}YcwB6o^St%VP0_YPt5ma!+qW%)tv+7m zXu>~E1Oib)fC_sha;Ru*7lPlP+RBj$J#>s%5}s06hj%%c73+C^vW8Rn(3!Y;iJ0Lk z(IGT*hUOBnNgQ6k(xJN|33#Kos6&U!9tgj86EIbzJ-814@0a-384>!w5hX;CMTU_j*iaD$M{wsS-f3J4HThlF zd1%ytBKk3iWA8dlB{9zQD}q>K7+^Qj96)3UbD*~F26VhWkb342imyCbrjykU1i*VB z{OB+c8u8ucYN*H;g$U`n31kXeEO=jG+sc{_L!7c89(4fKZpL)~ie0=X2`?OspTs1j zKyw4H_02k*^UyacBb_v12pOQ8ad%lwq}11Mhpv=AYdhsCgkMh(S!Q=Gh|VuWyXT3)2|pvoHsWf+#X8`p_7Zb+g)iC zD1IlJw=g}on1jB3{F++i8Iwl}UZQ{QqYgrG(rKu(m`Gfs*u@YTJ#|W@Qc?PFULqU> z=QJ20quZ^<5j8X!qA&7cOELrgj z-LRy2D2F)0tpPO_96PlSTvl0>6a!{%bN+Ev#TU{T)xEUJ@K0`eo98V1r zBidiaV+zb|^}T;CAwqZM zWa*@wKP4xx&Ys%xk(~I?8){O_tu;ktGq{{vkRB$62cb469?RG3M%`^>O}NX@I2U2a z8i)8SLC-BYf^7VfkDvkiZn+LNZDR<`MCLFkA|i42)~&c@6M`fN;=!X7;C~qK3O{sB z1{<#3=Nz4yJglp9nkl*R$sNEuq!)WFGMnEvHQlni-$P*RVlLFJP<;%we(c^<$r}|` zy=y5(keY76n^E??{~Ia&gT_46_5)6lKli|PO|EvqmK$ZPu8$1{qgCO_uO1Qah3dNjU`lZ59E?!jUS{k*&o!u_1rGhWroa~wx(wLt{E{M<;U9MZyz*43e&HaH z6N)=~@~sFY8%MGod?z7>(Z>f**8gdzJuV}y9h54_eoSh0^{&&>rzG7$lI*x-5G>jg$X>xqp|%1aobe3bLU2Cw)b4Kt|y+L3)2YOP&pH1sJn z1@s`cZX>g# zLS)#OH7_8%8Cf4@1WFbUSXZMvk(_e?tzLt?xGv)CMF`vob!KQ2a2pQK>Dv<%X~|2a`E#{8JGr9~ zKX5mS`QF~oXpvIg)Wnj$Yn-5*K^k=}gC(6(jI`Gq`-GsAw+e7etEd#7BUk9Mj%vrj z96*G*Vw~1Aae~(N1z;MDcVhQ9V~y`pu->!lF316Pi^jGwJqeI5QxHnu9ntBA+OI-y zI9N0dzg{~T_Y*z~cQSLnvG!o++^5 zaa*`%oo_Y`V-Nmc7O9M#zOo<42S6a4C{MLa2ror}RG0F(8?bBWx~o zDHTn>o1`s;1HA>w*X+LxyfubXfUzx}O-b}bG-8m+*L*(7LtULD#P$x^AUytS@cbhL zB1i`!A>o1VTgrMgi4iAY2itF5FaB6}+b^L-4+-~NL7^ok$ON4!9CU%UZL^PHpGWs< zNG#KI1DEH`&Cy_W`jB0H^_Plvg`lz}8k-$GFW|HRPiiqpoC^E6i~Y&DSm-3_1R2Vw zQB+zFVt4iJi64VfUpGx1Q05TOgO>xism1jTv(-BjaM#i-RnZA>Xiqg}uu|nor82&q z|KmAxDr$Rk|h}WQH&%e_Aez!?$&kEL&&@&({3mte} z4A))ZoBsvAqP4nDK|;p+Kt7T_eAD=jR|&T8RWxj-#W@St^67V==;F-?TR!s(nG?1i z#Ib9apT0h5mncMrL+tG9QP!cN`o}p!g`?g3Cfrjd(St>t-4jWyN*Ff_O0KqMYXGYxYj6x7dGW9WhfcrZewC?_c@{#|a}`O+lr6XnY+V|~o1 zE5iCZc7pQze;lAt+yYi}U7^SYu?9;)#AXW8zL>-~>pIkMC_4X z_6dX$B597(c@ttJ>Uz+dKu%S@F2bftIy1gN)F^Nx3VQF+XiJSCfNHluseU;L+8t02 z*)=^5RDOrksF>%>WT-l^-DD-P9uBPI>SF!nmwL`96RSGZc|5S?CYg7)2^_a?pUv<7 zb2t7HAynl=>qMFo60NIF7$Vy!!Z^0h&v;etcIZ8g&=>g_Mv(8-VB_03969B>zdbY& zIbwZYUb$AWyJVKm9u=@B*`_ZzmZXBE?iXV1_h#^wINdQ|$;K+wDMJdyrUXxGhlgUl zxYXXLKK0_dv-Ka#h?@pUePh}rZ01yI|yZj(93jAt7mhMw@*gRM-e7Fj!j*Z zij(?)o5z4}@dv7kvqRT_Tk*_8PY3;1JSE~ZUl^A!N*%y8U%Z91WIU&D4KW?UY89aD znr60d`{Qv#4tX^hAR^b=&B%k03V^;s{NSr!cMl!31t2pA#i9H^4iGt%|NW5v6HnLm zIC4r51>uEPAoR7q`VpA22ffTJ3q>7|F!b?qTHt??*B4QGXw7-?;Ke`YW-jeXaC#2# zX20te*2BT-Lw6K-x!hRhE|r5_$j^rvA%lS_tpU^jX`WA~(>Bzr4dnbZ5$#s7Z zO0~IMItTneu?#5O+AwX5vrq)Fy48JrVK}M&E>@|i!Q{N>9*qWBGk0S6HxV;AvLpeR z1zFc9Ofz{{u?sFN2J#S#t_E16HSHx2Y|2pW#7V^seMwWyNdLB3q(-lFML269G)!F* zfy9NKSHhw6TzEmwWe%?GU5GD3Shl1v4HM{qm`{ai?F^H~?(ZNT{@sL+|I*jlZv!rZ zo+xyDHOf?Px|x&aaAx&O+DiQKa8{aM_dj+_phZ$tSm`_Mb&$a8fI00vJH4AEQh=KC zRjH(^omR3_BC@KtDb31!A7XpU>>}GLl*uq?E*OgC6-ad>Ro4qbWJ0V2QnIT6%EL0; zI#*yjQCTZ;x;w5;jOh%lS)YmuCp{LMVJk8kA)R4`lbP*x@|uZ>V%*e%W;b`gDXM$$ z0L2er;2tz-wCNdb8n49*f(biLfgzHEG17lL5?a(%J55!3Stw<9gu5fX z`$#x1rho_f=qn|rHymL%IgN8J@HsNtgBN6ONQ0f`J_vypAVtUyockaP948;zxc%%c z&}pC#-@y=rXSqA{adZpqt&WqUdK-5)Dnoa4zd*DTZ?u3V2|3|WZa|I^-e#x<3F z+d)AEI|?ddqgPR?L8*dDZvhO|LI@yb3`IeTsGy=ElF&O+Lf6n#n&XHAp?3o)juNVf zRHX=SpZW48;k|F~)BFABn{!EW?>Xn5v-jF-tzARfD7-~tMOCIIPBnWoLs9 z9v}JB#=Fb>esuEn&-bZxRMm!dJouXTDELdw{wb(-qz#VE3XtIqwnI1S=oIH5VAP9p zOL$0ahPxgR&rT#lJxUl@6v;PCSD{Y-1yEjt5t8cm6Q6F@tjrbFQlEktz!$2P=wB^R z2nL8S@2f^yG)LG#`R6@PT@qDzIn7miSZ73(T^X2o9Szmbh*!=t6rbl3KbHL^aK#r{ zGeO8I^@lMr@he}0Ey4j2S8#f>Ol8E}$;WCd*Tc8E7$sJkKB>D zVg$$@6xFOAx_~Fg@|!)?#`4Xw>7b}vfViQVw*tYiaBJ^4;^sQxYR3kxvH?#5Y`a_H~)xMs)u>{>V&g6!4jEx1EWN%nhjvXWamty z7#D+oAhxdFjSkH8DC67f)+?)IIjERtbKsZGukaQ1%0w>PaG7vjr4b~;uOOj`^`&BKr9SsVzJo=~jSclZ32E`^`YTz(J zDFLl6gg)})$dizoKnB&>1Z2c26|YR2N855_+*{tH0Ir@RHcaSV!d6RjO3ft-SC>lcM}r4m2n9(5JbxkFvydem)ar4m@5iWDR+JG zcPOsM^HcD~p)0jRpiBTlV1u@r`0}8r+5-H2WXL(4d(@e6&8Bde>n)5Bfm_Nf-WPiQ$=or zQH~rSaY+HGS^uux2m5h5iSvv#^Bl+RBg-6qV8um$Z(i(=$sB=%_uLk=n>%>7L*6aU zolD=|@S17Uy7^l$U8vRVN{DPVr0{3yvyG$Erqa8CR2`Qn^%p1fBMSsCywYPrmq$9* zb#u*)K5k*cAB_jrNf*G6V=2`fP2uX25WZRbZo^bT4gSc3?hgk0$2b2|Nvt8CdMf%* z;mFGv)+?=+0Ivnik>uQ&1`pwMmh2gtwr!p=OgjZSY>OaFTpSZ6^fRX{@R)_5KX{cs zY`5mS?0F)%=DLiUU#tl&ZFfd+d26>+t@*BM2?%WSQWCGO!7N`1z?HD%8*9ERz#L#s zOEt`wHK?eUHeBIR`N72JL}Rg-3sG`WAo1Nl%8q~E#7G!ik@REEW6@a5n!^II|26n< zZDh>@S1f1yS!uL8M)F|zs~79{DXkgI|4hvPHZiYnMsGl-r{!KcD#5RIpC{G}j}z@g zw3)&$_IMtr<2=)Qh{bkmi5qiAE^NdF_=h`cFn?y#j*(cH+{<=WYu47-Y2fl6N_mQ7 z-pOFvltp0G#m031dyT+r@>}pCaCy7_8g}Acb0r+u2FgR33Tq}^6<)2mDNH=ank$!K zp@jQ7uUUDU6c!-wRZP{<8F0IG00 zD)yCsOtl;CTFkd)%}2)cFO+3y)*E~T0BKM|=OThVA-?l`eUm<&>JzvZ0_DbtkAOBK zH(q)x7%T7ME}i*EbDUEyqG2Rhr+%xzKUDG$C%fuNqR{9f0hEY32Ro}&pa%!StHG=S z3iinWp7>Dy^h>h5yF=bx=K_dFq-{-&nAew|<%KtQI3W#a;Ewy6HP%yPkBN1+dou&k z1HR`yg!{#+6#p{@Jdx&%?LQ%YkWT2OI#H4?=k@rOdM?H@6M@*)&55=qWg8F_3@i?I z5rwe+f%Wge87*d3V!eBds8!L7OG|$#tKw|TJhjg>E%7Dk(PiLp4)A+=S*H|lSGnD2 zEca)RI^~f$V3*^KwSF+l7S=|VyGf!6r&5ABvgyO0WpvDgs&EBRmm|{yy)F< zUlSEss*hFEjz?)Xq3;0_frXvHm#jEL!*#i?9e$YX&80w;lAO^U@r2HoQ3w?Lv-wp56i}oFM9N?m zkVq@X7iiH$@oWx8q0d1g`6A)F^bs$-9lV@~YKL$}V6+_hRRcf($gEMO^?;BB{OCBn zqd*)nL{wk|cOH{m2axy;zJACBz@qLmEc_X5=eE5N=;;Z%=2Xx5mU_kbvCi66_E&@g z@z5ra2p2s!RGJF6=S!;if7w6q0{rkIT{20KRvq(5n2;X09i0cv4xFPB zh`LLl^_zRqWe+6S$+;m&ZhN3?SDJ3#5J# zm%L)yY?^xMZzCG&UIqSl$!`!yoPtv&rO-)Hd&Od?d?Uc#K<3CKF{JIn9Vcg=0gAkT z?_fRyNRXl-iuU__HVl{+&$r~!5281H<~DJ}TV$>WGEK0zu5-!ClCD|oD#{{4jmdcx zOjbbQB$fp9`5b?EC5oMqFWnVC%W(x*IdeDGY@B==sW(j1_$A2S)0?nJNeHz$CxOQT z_jI%(<25lkt>AfXc2L&gTpYhN@FHrvyjJ2`{2jpxu(@x!u{!tbJIIBrA`b;X8LZJS zy>s|x+FO16TheJ?T3yaa20f)9g}<}eSE0|-B<}AdU}R$)Zy<(?$orD3)2ZpQo<{Mc zdPh&(ae8M9*vlJuI2UKZTJWvd@IP!vE!(JX_hGnR8RX}O)Jr7$i{Buol(h7;!#>dK z_AzP$Vte@9K>!#cj=(E^0`fws&9u*{GX8&W#%PWJh+~i-bpZ5+Aw)hC8vK#2Pf96a zBhKO={=4*}w4@oA(fQ8d>!a-Q^wBbq#Fez?Pn(AC^8=+DVChfveR49RX%3P_WySrA zmA&-)j;wci09XT7Y56{AlLG#6$15!FF2+*t>Wtc|T;EfwfFqcaoB+{)RZZrLS5PrU zU;+J7G|y?=R&s#@(sxx41_F}bv@X#`I-Wrp0!_gZ<^ciD7EIAAl`A3CC8@j0c`K1wMX`Nz;*Tb{EUdK5W z4S-)|?YIJyh~IPmWbtGEZAqUT0Z=&L=ha;!G?HV}I|DFS_}=5`RMJ~9`bkB=ZD_P; z8k~ZL(Ne-5E;$-wC*_MQ7Dq=iU4oJPgQB-y{6jH*^XpSldUAH`=nd8hC@8CD_@DBC4 zV7gqqI63PhrvIT^1i&Jys$BEyVB)- z`}Y5P;k&*u#OR6U<0R9{=7-$l*}H)Dvtb3G;{~D6u(rG#}KVv3nM{kdUUxd z{22s5UmeVS&b@Z-IPkeWmgyYuGfmCx$Rg-TVD()=C#Z8Va;va@z2)t>LD#(GC_&1u z^}^dFw)Gdj>TN5B$R}Lc$sV|_xt+#LNfM;$kU+*i(z_1st?$!ptrQqtv$PLG)FDr|CQg{qZ;fqy3=P6p6#u&&eJ zg2Oj^%~+*v0Ud$jUPagQjB}8pC_2NxyZ^9b5ZQ9{vk-B7-M*Hg&RP+%!kDBZ%*#~6Z$Wg@l z1<$v0y%|H4r{T&F%qUH*{!&{xUd0$mXEy@Cb_SxB z=yOg`D7OQeD_9?Yk^jDCOW5J0jfM`qU#j2BM7u(V#Xod{4uWV9ux`wQ0NHSaratgf z{NU9Ako}?6X_?i?ISpX>`dp3l>Mbg1*9COn{s|ZAg>(N-`~iUPJ5r!VaK```EEg=c ztH96d+=3x|iOh9ZR|ZxS*yL}-=&oHQrszIUUch*%3CTrZnHOHX6-@?^HwJTbLJXBh zGf)JA91Kv#w=CKaLdECrHRi(XrHB!dOfa-u&F1a82K)9S%JR$OOLbH>X~|G`U?fla z51%wIU!%qbw-E8xh!1WWOVlp`N8`cy**L5daKdm7B@q7u0Yh#yH`B0_ zH8E!2q?NkV!(~n_gbX7)+0SMEiExGZICK^)fS$Wf32tTLE<{7)=10wRb-F)feQf=KWacKcDT`;X|g z@AB@S0|%_}=0A8rowwOH>~x6LC`K}1`oPS>!9|PSK~rLi)K1+-EVYt367#tZP>KE*m zofY8LO(07OrgdVg+%gZvu;h9`20eQ0MhlP2?=~HxELqv6&U`q|9M$;N3$k8mJj0gL z>iV{8Ega+s7_7ol+upq_VTSqzML13!1q%yV!U(o?3X*V!(;vTMMVi^x6$95(e*abP zwGa{xAf%>1`3+~76Uj|GVjscTOuv_Rg>{*$H^J|_%s)MTjP=TES0H^how2c9vszp% z06{B0^eC7W#$b1RdK2rU_<1azb(X>W!7W=8EDLh!ZTt5IALaTv_e~(RkLjU-5L1%G zH3zO;lZGvIoFzv;^;CEtvSj|7aE@NXkg@C2S!*8A={!9ApMkR?hAa;+Ujp-$rWQzI zok>1;j?QO?s_wHs;Z|t~OGX|K3^M&T0D++Zaw~j2>+zX2K{&k`7_W<8lzdr6G4&!$ z{MMzmgcB?i-`WZj61wu!Aem*T{!bsH)h(yh$iwgIAGarItENs)CTqrE(=3#ovNf$! zrkvu~ggF;Sn`G>!JQ{|Af^N)zs8Km7>{;a6x<~Bc8jTnzvy${fxYrky{=63xROuMU zR#5fxUMcUP@H4xvt0@ zpUze?XsoIajbvlE?c4rg;cLY}pHrGmuB&l|wzSGP>?f%2KyD#Ya-$>99A8vf1x6u| zlof;HXj66e0FWG!#2upjUyz)CgZ&aQvnVD{9S6S5uHjyfI7rNnQd4)*HKK(zV^;fy zKgV7lxk5Mp-Pyu| zHw1Ztg6)vO<(yMWTSK=Uu%fF;D?25tA;~muC+~W_1gMo?gUV`@ z5`v_nJj|La&xG^SOJoJ}{hZAAGn1FlyClt763Zq0RqUJ2owkzY2DK`$c6&R+U$qqh z-wFw$dl}G@vcH9V?^<67}!_>B-X*fB;U-0Q~}qNFGwb(G|Jl zg5vxys4^0Ppdlix6H5io={E>{MpuCyVsJthI2KS~6?a2YmgP~SAUFssh+p8mn*tk~ z0+(e#mRooWpiqk0*-jQWYrg@A3cUrT-HXlFs!2~W4Cdcf!5If+UIPzAmms-R0a%Qn zCqxZE(H7X(u7Le0a*wkEwZC@aEWIUA(X*wP`&cF&G`c(EfSr5t3sR=b)*)XVBohH@ zc*hXLH7oEyNb4i2wKQf~;|KNpj25D=3FmoyUiI{kVmi@eRFrwQaM}^fYnFCr6Nd(_ zu8->f&febl+k1sb;|yZY%t4t3?SxJl&p|bS?)**c$qhFomvSUT?R4@<(W8G`0*nL- zt{;CMS~@CnC#$L<0E^9;D5-5EydLB8@-2A^HZXk{|kV z!mQxhRuSt31aX-n;$4Wo4D_7BXK%Prl=e9R>zL>q1WtUy&vpArI&SdZLica%=j<(q zo4AD-ZJZAyT@bvquLZ_4;Y(QfiO=`+CB7hN@PVRs@ufs()T$%4BR12^KP_7DSki;gWD zO_7A0n%O+bb5RUx)2HBH3DTBxW@X}I4@xz~jeyRw(17v|nL!#YRZFzq!Zbxe0k@l^ zKMWEs+E7u3+~|>><i$5;|-`ld`iD zFYb8|YHGNYe^-}1S`X1#*X3h}PK-(2?}cSjtR(H*TC*}*x zA0T;vB8Z!hEdx80I62dH_E4s5j`2s{10oxA#Yh{pWo04E#pjqXim3%s&rM8|sj?6J z+TA{iV67aUXipcvP&#gQHNtq&$lt8Y=mJ^V(Kq3R3oI7NaT0sxOn<_uaZn-YW2zV* z)64ab$mzCYOIos?!OQuwQt6An!AIVvFx*+e2eDy_xD8QJXT>1`@7wnt;$gGS>=EGPPJpw33SUhKI&pt8uNd2U;FNU@xHNSn@XQp-6cpDJb%leZ||w`Z1i?U z&|%?AnL*?qW09D4G}|X8%#OZlYRSOj1AO9mCP;OI$DV%f>u6V%jD_4d!O$I62R_DD zaXd=))7#TJoU zD61A7b*ODZUN68R87>*xC+C=9bLTMKIC_0HsV5^ZL)4ciK!hyLx7&l1vD-}|C3crT z^}XoHO4B&3FyUOb2qXHN!SFtH#nMazpTf)lMgPpi0tcNeQxdWHU9)0|>khpIux_ne zOxitt&-GBlUQ@Pfhb2L#|CoNVJUu-JqMqBjXhvVhf7P;7gtWZAu zK3b+9q&nLBfF+(f+`288w}%V=L5muCX-=zST6c4k_vnYBUn%Cpw(ftVL2?&0@xU`g z_U4~KKl+beX&Nw9GoPy^#`Z_u6!}YWc)W^QF1c~U!Od`6prI}#wF!gcf3`~zrX2R1 zpL$zqu;dkG>C&DrCF&yHiw&0^HX<~cfT)D^M0h4d1BY4W1KFOH8em}3-3F1VP}m=X3uf>BBB2e^}= zH6ayBi_^MK3j09VM>rVs&Y7fn{3}`RkIG+z&A4LO6)BBZ5Hgs>@#t_Shh4O6HhFpM zYs6{yDVF^DkzlJ#-xI5Nx2ZkedTK>4CYu^2eZOo>koH?pDGGA`<7IVOyXgMZNZjBs z{zqkUN<*XCZ0xw*&z`7vuEn%U?Euk0M~r@Tca`j19Wjkv41=ZYs)`b9tisredA77C z=+{uIqlgxw!s)uaW^@tTOx3>)%>$H;zR;o#8nn5|oIYMhHO{>!--Tn^BWS4)DW`rF zkBk)a^?|5}Gi}eE(v{iVynXslzGAzxMSjT_7zvRzEFem29xBSEE4X!(h$M_o*E;w* zZEJ5XPh)I(n&sB%rFppg`r3W*IFF0kAEq-~F@)^Q!hqk~KkiXWdU!c!A#})X$)kt* zbRGD`c0iq(sO)VhNZXT+x3oQbqm?xWINZ85$2j=K==4)1>x=rPX~eEA{dA#cuxSx4 z+lI^2h#8kQ_PhE;xU=Rb%Dh08(0xI&bo@rx%H<$qy^Tjg?gbzJMf>HZ`-B6lW=aYy zXPbY!0q|bE@Es~LiomN0z{bkH=ZSi#&_0f@5gU@^!;S(V$CMzr17TuhfO6@6Fydr!2h;i4wk=fUc-*B0?Pc-ugF1p0WIMM8qZXEQQZ?AaW{bbA zHYFTp9xX9;_#;|PgqiU+-n*ZbvceB6pJkzJ%ad<0;xppQA&CruTeswYp1keD`zm@9x)^HqN%HV%;eD>BfPK-jszXK_Mpgc4OCJp z{>dh-ciHCQ8l3jn4ak0rdn_l{+-)yl3tdCQm#}ND&?S$30-MR$Dx=nu!G3%8LKDy` zqqb$HUrpV{vEk?KvTbVSW-R*~loW)45oE1d$ePo$j{$0eeVmeuVSa02279?9xJrcw zm6<0KQ09Yh`*S~Y;&-jNA`Mu>e}05{^8Sy9Yj3N??PhyKXh(#F+$HclzYCK0>fCk~ zzBOz&P<@QG%|X5xiRYXajO?S#kjBqHSKW}v75NG&u_wlMDyMWF zuL1LRP`F1%JwvG1DK6|Enej!m9A+!ke`IKG4_c%ZK2J_jPyPupks5fhseB&+=IIg? zcyDY!RVT=_Zc#E_Y@NE_H0_SY-9Yq7AnsH)hANkveFbmRN-+DB$PNI$NFpSpQ3~b^ zl*2-jo}Zg|z(dq9@j7T-Tf&noGFx!Pj&Ob}vvPulCN2;ZvxpnhQBPLDDa=61SXS*L zl0=?pcmr|cDLjol?Pen8*N~aJNGpFi4!}%KzH??j?K^jH6KHD3KE}xr5c3-Ja3wxu_K~wc6g!Y%GWfIh19~zH_;&ww$S;>cblhUb-Q)@d7js*N+nLBCcu$v+W(~Cg#Noe+*yDyqt=P3i0_n%(|&S$7={z^FC_b@xByW`6)M-QATn6;uvKw4w^0%J;*MS;C0}+;#T((xs~dA&SzaPF2@CkF8i~(v? zJp-7W_uHLkg(~XLJafE@RWwY?{5xvSne=uTR*%63vVvLE`twlw<;L<O*-0Xz=DfA?Oa^SApM5z)<5geWxI(oOZMX-$4Zr6PIA;z+PkKCv zq4jSg_|W=wZhB&VsSX3^gmD`{4`CmspdmdskFdJFR$3nBZMIZQ2)jAD$8KR;D9rBb zgETiq?j|@EkvXhzMPx@HihKq~a*Hw#=j%B#kJV{eFWFsAs|J8d<}!mj;|@+IKddmU zxxt;s&+_o6S+`HHsg9^}{^Jt=r@HmlMquo#QwZK!jZ6O8OE^9hV^dmKkw6L_$VEnow$(o(a&c z+=M;6Ajs^=Il0=gxG-x8|#_TxzmPbBx*S z9avy|^_dq#Tr;YrxvNsyFWnf7#2au2NJX3tqissi(kgm0&&jfQMLjXnOd9D9z7`6I zg)R}EPbUG+1q!4vKjxg6{gQcC>a-j0cm@OZwG*H}BF|J^K+}l7O+Ey!2Kn3h|;c-LQJljJ* zn|D>)uaFCJU55LlcDgVcYKe`;rfF9SUuGp$c~fv@HyN)}v#cp$MbH~}GAMl!{YRS<7Oo>%N8!<+LLlD%Tj|4C#U`G8?%k01KaL+^Id9 z?Q%h9Zn}1cGdXvxA^Cx3=##;7$4qM5%rX|=zO~)gd2@C zm%<$~i`t52MF25%^;!&t$Qz_>CoTm@vKh2_X2n}d8aWFebMdpJCU>XAPWUpesxc;l z3m@L0E$)T^bk~_Rqf^DQe*r4iQBs{N^5+gt^SIgfean$v$`>Y{Yp`fN<>S>?ESkiX zbhd!-M@^DLIjGS1`8BC4wIMT3{nd6vzAh4dm<5Bzl~3cIT6 z`dKOndMt&hE4mDc<$bjV^`Y<*E%p(KDZLxB{)EJFAD=@FL@psB-u+NRe+8L~#|P?5 z>^v8ZscBddZkJKx!NnoPicp?&LgYvBxY(E=55qk!@9o#``ggrhePYv~R$9JoorxS~ zgVJBgyK{`YzP~!clRMUC4bDLk=ftl3LF~PDD-HVfU_Rxr1#Q%460w#FMyMIu!xpoA zRT&7$TAK8;7KxHrg*VM9lGv^%g{JHrLVeH2Nyn3Xk-@o!2S{xH#FI? z9A&lqve#9Jyi+&C@5kloS>>H|YR{JS^j6}l;>8hT*IV{@2h*_^TrF}8O7zS4_!=YY z-V+vv=l*dsAE{J-!UYyIpWpM#ku&|0CYzI(-gZQ@S-2t}L>iLm5a6aA%3XM%P~>=$ z#HG-X9;tqpW73`exGpGLS%91cJL63N!7_X>G3#3^nlekK6 zec=wC=kn*PT=9?~d;1~Jr=BiUPnej{WI+~Malwfm2{r07*PK@Pr02tLR;26?W6bKxzfwXn(S_G`cie}Mag(64%l zG+CJV8-GcpD5Xv##~s_B_JX?sQ!3<|@_6U<=;1$pLR{6|_8>XCsdo^HO}YK4c98d-Z`4Lo=4W^ww)A zBdG!<-9a6?E1aU-yF*Mjc6R~mv9?Ssg~O@~hR>M&^g^5yWmD*M@q6QF6?_gqcz8f; z>Slx^1b;aDuS$1tC~sZhb5zIzzQ_GQs(n7O;XFekO;IrIurAc5mFfoxBz62TZ2o>- z39OJ7PKj*!amaoWgc#f-LxkszMfZq=jtDPiWF^4la1eH=sE*%I)~&b1^>lqpr$fSd z-i-|4ITuaqjm%ub@7z!-${iMprO2%4?)4nlaZV9i8FoQTZPykawFU((Z3Br1(E*n) zF&TzJhaKSvf6=11D!x4Xj2DV23$rv)wXo;~`brwc?bR;}EAEk4^(mw0KM8R`DWEM( zS=gAkg>vui%1NcgS+}+uF3R(J3z#q@$UbTQ*|Jmh_T<{WWhwjGL&?X*Je;%dVkt?+ z@jFKzsxNG({uGT5daivCyWGfEBfz8E^Cq?V=c~K_UMO1+8!tQKmTluw7CNZQIPAN_ zA%^Eb;}bl@yY9d)vDK?t5wqFXtp_#m66Av%rw?52V&Xc$#~1eCX|<9wg;HMe8WNq( z%1Y=_R%&^bc83V8&(*&r(QshgikXK&F$xpkhiaZmWfm_CXYwE7Ru aWwp8u>w5GewLO1>|8%qswF)&a-1;x4(eL;G literal 0 HcmV?d00001 diff --git a/monitor/sre-agent-event-lab/assets/official/portal-response-plan-autonomy-step.png b/monitor/sre-agent-event-lab/assets/official/portal-response-plan-autonomy-step.png new file mode 100644 index 0000000000000000000000000000000000000000..b3ecd6c2a00f9a21dc863e32171176511dac79d5 GIT binary patch literal 67961 zcmeEuWmweRx3*GJ(%nN#4&6vghae1+14yTIi8M%ubc29O3X&o)v<#q>Al(9jq<|v4 zo9BPdb@b8q>-*(BALP1ZhF|Qx*4k^`_q|rMw&tCi*i_ipu3fvSqO5TD+BJ+D*RG-7 z$3y|YVGP7^zjpoFH5CO}J)i4e=Fl@}dOkJ%p*@4|K*PquNTYRd$ZMGcpaTYTjsYPP!22l zfBov}zc1L$^~#?Z%n5V;>jA)jmBPQE{_$W1GNkCS*!T8d%vk?%5ABZf#K7Ng`_C=Q ziyrF~a+rF~RQ&li6r`{4|NZjd-uFnL9C|%3owG^*eCT4iZ@<0Tf4))h1U&Eyp9!bB zKNrSauV4Dl_ruoH=_uDGt!{DB|MQ_^v8QkRdkwW^f;q4$NM9`9YaIG>VXiji&-Y{3 z{@+dc-%a_$lK$6I{`AuS-#ta@(fNK~#7yAXTDqCn^aD@2x4SLdcvOPr1IUGd!y)Dr z^-(KffxiSxY{6T5AMVh{@xNo2^8ZF-SCZ7?xt+HDHA&i5Upy=f1D{TEkgLvhQH{4v zA?$NMQvw%PSlf>;bt4sZ!&Lm_>sZvS-d{~$eKh$?P(wM!A+S3;9Jzh@eY^S`!ks1C zyO+;eEV-0w#l(EH9K*ZrkT@;(g`=(~@^60~hXww6CE<1yc6fVzBAH_NXh@biF{wR2 zK4mQSM6ng!pNB{R-E~H8Cw~3%^I3Jim1;??M_<7s=PZ0C`}4ap!5C;!^#r6GxVW#E zU=WNOwWHeZsPa;Nf^WxlPzC9V{tU<-aB%svX3N61O8O8ue6#71HYV=AOYR`P4jWnb zF2Vp&dh5@#V&|4jNqDeaxQ$~1fjTY`>10RT-(J^=tpiqwKo|v85Ojqg`V01qj?Xox z3uHsFM|PC{_V``|>^TaP4VK}!Sg*T0k4gT#s0=Df+)8~Y&i+q6yo(JW_G zlf`$aM&lusd|95KeH1KCe5-s+>$Vi9X@bm${aV)p!aw`svV#PVRV+m}Uh7$s*R6U}Kn(bMxBE>bJ`a$>rMagN|H5 zmg%8oy`n3zV^fqk2_YfMhS|N?I3wHtGPq!CEJ5ke3d1r=NG4mJj|GMxcjN5C*lM`X zU|Iens+apiA`4dQQhXB8xaZV|VcmN{7yQ(eH#$qQ5m>T87iSwM3qkqo_?ZG_mIyqS z8qT3il}J>z1MSxfkA85wli)v}0ydPhK-Z`}a6Zj9FfO?|()bye1zphjewV3I>Cx5% zXBsX;vD$K%AJO=?k7hwP#l+4d=^rU4EEg#R(@FWoO6)dm&foWB#cv(WIWD2(75Mt* zkci?1n>$ke=hxP+Z+99O@`4UcLhM&X9)8if{j=M#8m0^{3f|6crXd-e#*sixzx66X zWZs7CorOHfIVC9GH7!<{_$Ri%wIi$6jqmq-l#ipH_n0utjf_Y>m>PaAAl9Xm z(>7eC_Gs#G4C?TjGj*h0Lz?o?M1;m@@OGM`+;7PyMKB6C6{jG|_FRST$(QMrTgt_8 z!c=!^BL)fY(CmXK>}DkGx+v32;CB1G0~H@iZ*}zHj*WysAO
#SQ}oY;B>eZXEP zwac4MZk`l}eigkCBIl+hj8Y9gzBuafaB@?h#yKx0E`-S%FfI3v>o;-&zfE@wt>?IF zHk|Cv`A^lrZ_o!o%NE~kS2og(eP&9YPB9CGw`mcD zqeD>8xnxB05DA>&=xih3289H_Ido^H@$1_kyAo%sDIe`WC{a6wje-q+G%U7Z47|RH zP4*?;ZrnQ zUOy(U90Yk@RAD9ob6j*0UT;{dZm)hpVt= zT9xH`WqEzF>w3O@CW2&en2aif$VYT2YMCbz(%Nl4OUGj?;*O;AW7lU&kUZdcSwS?S z*|+kJUkrym)xwt5n&HFV@v~J|F;(#{;E(>F?8^8MF`6OxKO?^piRC&qOp6{r0HNZ8 zBDs!;-zkVd6yGIh@&msa=21x|V0Jn-ucaCD3X|E>KkSEq8(_jyL7>cEl7r`vXP(8p zZrSgp_BDNf*!! zWBlHg%JPa3)$e%1pxNI_;sXJS_zgNA2A0G{l1+Y*3L15Yg3`SI*ESF>Le9UK1nl=r z_lJlseaj$XPZ5*2{P{HqtJL*|DZ48O`^2x7xX`>sB~v10+ZYe(Wl<; zwsC1Y2|w{qTcz;%LW4#RxQA%@JZ794!#UL?qdeGi);pBnM>^a*w#g6d{dP^xKU(Ph zUHks`%q_NbFP65mg)3_W-Y-R8Bf?$gck2=Opv1$tzQni-yk?(~=-RW9uT3lQG>SBd z@A|H5`|gC|iASkLV;OMwm%R2nne*NhmTVRk3XD#~ejvxUma2=A2DRGy_vUSee*^!Lj>IVPbwzbR|?hf@nR@e-}jlEbiHB?RyBE*|+>Mtg+7K z^*AvrP*vN{?+bBZ4D))TXW>P#t?(qO*mQVw8Zw96ez=X#X^Ax-P8-{Y<+(V{Sy@ly zY27~p4oBiGf|qwfWCb~^-|$|>S%L}ZSmT06vro&`)}9nu+}_F)w3{YU#b#PDdRgH3zDuA zae>n^RsA$U=9+w%vb!7QgWk{3-Wa~mtg(bRHaG0UN%hU6$kL-P1Uk`(R~;tyjx}O9 zzThgbHLFW9QMC6?l#xG+)E4<3&=ln{>_bR-+||LkrQTXDyh*^kHnnbl$i<|fX0wv8 zfl|EZXUH(a)L&wNxf-LvP!&;`)klR;4zO6SXZzoMEb3%d(?t_%1)z zqXAK6^f{l*;r{RE#S)5%F9u~}((JGl2%GAwOBZPjcqe6ZR*RGq< zoJp)9@>pnPr#NEKb2Z`A{V;3g?z*}d9Zj^yAYN8wME-#H8WIRwgrR$XbR#3WUeP3j z>)Bsn+%?Nmq^3||A;v5!vEsH&wtFr{E3)NU%ePRJA&_D&ihKQeQxcYJrn3I*Q`V+j zJpAkMhm`IC8Q1$3W2TA(-PJ$75Ex?9qS;ZToX&?`_NoWu*my8c=EEqCS07-e^g@h*nq z#C&#B`R8{!g^$$;tU!5XGfWaxaStg; z!kR>wtFlRH=>Da)@1b{}AY~C0cLdIRGQx*Hm>wn((5nxn3-feE@US-2uwYnNTx6J+ zV{$e)zCBw{)k2Tb(z(INJDb`o8ICoqDeT2p5p6V0VP=lwJq-LxysjRR5f7byQIm*N z!l;C%Mt>KWWZJbTL3Mw(EI+Kq%#vG0>iv|hsn1%;6xn4VQ%J!2oKV}`Q4dxa(d-E! zB7YbWSCUED-69+|qD&lUuAsujk#AcmzMyg$(UoHLbr$kh{+*hs?OlR8Z$aK`p{@xu2B3jTYeooWN zC51Q3v3hc$Popt2N}sLe^M7HvhNgy@zi(ASiYA{ub5k^@&{eWEQ=wra&!0b`V%Y1c zB;|9Xx)$e8cmJ};0{x6UwHOwhvmFA^$USUtw^&{C$M{`QO$$t#-I7cf4~%iWzqm=0 zzr?It#|ej)MdyIp6xS|-x<*PpA1+9sYo38dqLr#u;|Seg#8m8ON=25lu;%i`g(E9? zyUG*KbsU#^pB75dAYKT@Di>MsQLos-k~3Q^8nj6?qFUb(&yw=ib`ynhQ`pODq6m)1 zeJAj=I5xvDa}_7^Eqnogq~U9q*)}^fmPVk;WDwJ|&Fg zzCtA|T5KJjs*y8HCeomvYZ(mrP?1Ow2_3rGanHcy>7Mbak!73^=21-v2^@!!iz~fL zUMYpojW;jmEYy}*03`?aY+=j(^n0&vDs!6gLzOCYlPm@tQl?MI)~4pc=pAgfJxcdw z&)L5`kO+(z?yZYpT>yoQTY#~jkGlJWPj64;Yoa4c>-v2z22{42flQaFa9&@EdkVRm zxRRTQy+yB~=$||Fm6V3*|p_$_GG9T$V!-R^7PCsz2N7Y(m z58~EN5!|*`D2af`#ZH5GG*R08@c{{HsO|EIBB2cY!}6YlS@+ZROdGl<7SE}iOEUSo z1v#Bx7-EZ?@UsS)G(Qal?m|J|Z3v^!!X4yk<+8Ltg}7r-^64{PSlwHPl}7NzhGd{H zMZn&Q%nj;Ml{VKZ2u9XX;mGEAxcE}tc+?eBtESM`BxQ|8EV6fQOVZ^IL(e?ZP*Ed*Pn(+cmxe{%n7K!SZ>oI;5P(^T}INnul0Ee9*|_H})oOulNjO7e1{}cB8YzM_^HdlS7z) zED<7SOe>DjSpY4qXr6n#P-D5|mDT%oY#(jaVq&@~nT^Tw<%H+-TK;c^`}>^sGyb0QKFn5|8)y zRIK&`P4y?Wj*=?}CiNH2GF`rpKI|Q38jV(o=f{q}soCx&%ltk6QL4DsPTTDyv=CoM z=S-Db-VH0143wtP-Tw9zE`8QzsLXEinJphOhtP1mkWO=zd5gEHQ$m^AX=ZI1(C_4A z2`RkKLf#;lS2chJ$NR1y^#dv!{oIOcjwqpMM2M1z(OsIkTj>cmE>`W!aL@P&hhnr! z>z!-_BKysE?&i1QHtHgB_c@sr8U$rh=^waFYg(TfJS<;O_z7`MoIrliC z%(7Sxeap_^EG6k7cB?b{LeOu#MKB(x_J?xq$Tqs%MvIEn?yy_>gck47LKQLuNQ2rd z4u#)k{hNPzACAr;9$GQL`F+2SUI11+DU-m8@1lZHibn{KzCAskT=eM-W*!ZQ1wDZx;CvCiZ>}Ef(9eS!nj`1~~=Qr>E7SjH?_Z9(Qh;Aes{^db<*Z`?D zXmXdjuD`4i5ugfLM0fwPuq_o=@8fUhl=0^!wHd+cq4oU* zD3(bl{rtgo@RZQ+ksFu0&m*;dsO|d#bxnGXKfzS&Sefg6*l{ueT9topP%JdKw-0;h zV8jgGLrbaGzx*#4>uPTbRxm~N%U6tX9R9Mf_4(lWztw$H|BFHO5!wb|7*fAp0x*kh zaWc35Uv|2nBf~P@aB2&WoOElUJqWHiS1w~iLLux3TX-8NVL4R!FM?4fosdG@!xnCo zBdh=Sr;Wj>KY$e)+T=dXXWd@57S{UnuAk3QDkb&%6SX(K5;gp3V;qq8AbVBKm{$Ia z&mQ6c=Q{uRp6o9+Jwy&XC_o_L_n`WZJ&3o1bs7%3>GT&S`F#K|k0Fsm%>TXJFHZ3P zd@(bCypa*24VfFFxfiwK2&bMtb7zCUS@_bliwr!8{FVPObQ7O28H zk8hk)>MyJKUv8QBPeJxwUam8|gTQ_0K#4YBPz3N?UHG>J8 zEdn^!*yDpDfWqcwLouX&9*wbNP9FoH>Zy(G?mR$qJ_E~P*y*z6@&u^csv~^c|=`K#uu8@|7d@?>n%TwP&W^{furgeEX62t?7J@^B^lB9Ra zC!ZXushyi3QkH{bS5Rh;a3jr&+ApsLg-^dzJ45z=qFHbQv2FN7Dc1G2)8&|LfXab1nTX?E8hE-`K}(xYRa@M=V&_Lw*kzT8 z7m}1+;m-5B>*6?<05EP++-L|uV6m+xZkFXsOC{dA5AfD$1!4`7k0%$4q4-qPBwaGp zs&tmIe4_{QEn4}~nQ=MkVh_%CmdH3A@k!%v@IQ%^)go4h%FYrI=LZ}N0IW{ppr5Ia z8zOt*)=SCeyvnl+HaiY~tQ%mOH{_<0#ayN`Y#>{*#J#}O>S;!ivZ(HR>!{GLJsuh5^7PY#TAL`JtFJuO}BTPFeuKu2*zb<-J3F z-kR4O`Y^oQd~EVvd!H5F_Va_?)~~=`^UsKUeolir221AT>%#5!MnfZ?x(v@r!!8Jm z&H4G^2*88k7Q1I3YtJ`T-woI}o8t^^D00zEksaLB8<79y1h?f5eih7poD+JDU^-s% zpr3+h8PZvh?c5TrbqRNoTC4ZVgSgzh#VZgNMsYVaBJ3Uhz*ky_irOYSIG6gPkKj>m z1B0o4J$WC8tEPikOHvl2G(+lmTo+*2x4reRp2H8*BKA6#QPXyYbIT^eoFHNfD&YuV zRFg+%Ya3Y(JP|YJ-)3AkgG&b7M^)I8%m?W3RvmU&(qI;~J!CWRNZRF(7hKz2>^!B- z+|8@W0jOH*J>+KRwS9Crin(AfP%&`L8Lly*OM{iGNeZG5Jo*rfdZVg&Jqwq>oD4QK z91NV`me}QSbV!fCC?=bosz~JE;B6EVNWA`rCcyisakNC)}Ee6V30Z|Aj zOlIZrmj=~UpH^9C6xj+R+m9!WG}bEwAal1uNic-<6x z>sGKxDlj7NC$~z=+3I|M*&S^_asrwgkx_6TH3XBC&mwBz!A*lImK)X`ApOmNH|u6T@J9+tB3^P`+T0!0KgIoyb7JZ3kb*eni4zRQj(FkFHTn!yzp+H zyGD>}h;NpJqpypYJ{(}_)@{me*=d-86=o`HNNcq`=ud z`({s0HT&VaTk?xv$91*l05ha-_<%I4x4rg!fHgao8`sU^lVauC6MinVK#EI*%B}ob zgH0XB=EGr${SrYAdFl_MtBLZt2_Q8J3*WOg++3XE`GeM|Bb^aMl84~d z&y$6bKKT`bHvx8z*GmRoSmPndl`AQj?>6!Hiz>o%2Zt!n!sR7}b~xU+^2z67y(@e{ zGEa84$#w=e%t?YfoM{qqbcFNJNM_u@JKKz0d$}H}KV=}WQ+W?BawDI136@7I2-U^I zx09>P%dzH3Dq{-(h@qYJuTYnV;ME@6;u80%W(oT2Ekd7GFl)8$eDcq1bD6#%(=;F= zvM=DrMPH6Ndi3)fGgU7n8EY+>zU4^m)6y`y~6$v4Hus)MkLvFT}+-`4YV zLu<+q(X~j&`VUI6j1bovfXNdpK?I5;MFU!Q8Z3J?0Q-lvR>DF6FFf~M&y0-Id#W4O z78c10qTSoTAERh^8)`o&5-k#kyNR{b6|Pt+oHAt`M>v(Z62uLs@8(91`>^kb)JZpr z7;j^#Iz1Co0U7d`iL^MGPd;58iWwtgniG=l!sMNIht_i(NKX1ZOFp)NO@-i_Bw12! z^ptZ}eO2YGJOYGcZ`Wr1Q6e-T)-sy8*92{y;$^sS9}yq!%vq;s6;5PgCiH=LDHok` zl+{NQ?_6+Vr{A}~)m!@P!7E0hu5L0zE5dL5t8w!bZ5UYB}w8f3M-Te zsT|MY^U#Yh5rlXMo8Co`FfV_=jgOI^w-CxOilks$_5kduo1>6TiJzS?-ks+)9f0x; zwH`%vvt*{#`LH*>z6Yr7WEeBAv-Y)qT~_oNYm)~Yvh8%us!allQaK*#aBLz)hcs5HGZd^gJf|C2)#4C{T9E01Y z&}Nz=GJ85+K3PUTJlTNwA~xc(_&d*{?wn3vXf+BHbs)Q?_BTnW zmUfW5OhZ<+!DG1R-@;t5@SmRB2CjGbQOT8u_`=n*EFJMq9?>AuFK3CuNSm7PFj{w~ z`8|2J0pTDK(Y&G?di7_MwUE#*TsTu#5<-e}iPYKOo;!5NV0$x;><*!QAsUCeEvsN~ zkl}|2H^#!%rhvPWif^gnoR*|B_b68q5Wdj_wRt*k{8!^eNLmlxak|uWVjyXYf4J3s znrasB(V2C04}CXhr{d)DW?v~J5Le4GjZ(jlJ}96cX)6@r@d1LR%Kt9*rBX#$C-0XB ztV}5v8J5a55uL`f6F?d;%si#oE_H|$pv3EisAJ#;hQ^?XexU4iCDf$y9M?Gkhep6t z3Nl%W@D3LD*(S<(@6#js-L`0K9b%=nT9WZO?*&@?1-Cxh=3U7^Oo>-0;;*rvGNa0e zMQM)Sq#*AlUIGpSXFAOAQ4TqNBN`^qTO=3a@1xmQuQ80SoO7NrKt{iggS)f#91X>@ zH<8fDE*8~2k1&H+-&{dGN664(c#~pK-mX5km?G>I5UvnQGk7Vja>j}?)D)lVt}{?z##>9uVpGahKc}I;ByJ&Rv=1rJJ*ibG4~*69-j& zk_4Pz4X_rj^bq$bG&B4A(yH;wO1=;K)!6O;V=H9M?##t{i6fL*8ORScldpgb@r13up$nPM`KcbCN9#>zE?qGR~f^ukLb6TT%1?Wg8LwWac7k zPHXr5%rnI8IKd7YASH_!et;LHdq*6%tV$v@A zp_#EK-inhHi6(UILedk`YZ4aGtp=x~*Y)>S`h0GzU`ChbgndW{-xj~KxDnZ28!-;V zKz+^SL~Y^bvQ1rV$i4uc-p!ias^N@WQQBzOwd*J$FVHG&X%}xm(&{tTfSz|Y)?Yo}ML= zuWyxufCrLH(ZW&)&kGIHqxe7^&nW8CO?cHQ!4}N|8(%@;rm$x^pt+Ch^GN5nq*bCxe{8IibsPB1K>98o^_R0(q-Q#I4{RT#*$s&Auo)=O6N# z%Q)~}M*PjVFi+T9j#T9NE$0=>0i+T06jgkYo#Y3(3@*p#$Go{YaNY4V$ipiI2XMy5 zNNu>iG8%Ip3cUrf+Y)t==5w*L0Yhia2tjQW{@{;IJ_5moHOoXpY+1x1-rcl5D2&7} zleDfO!q+z`ubrj)VaP;mC9?K*>fn+v)^}2>9cv?PtU|`NBO))DN z$uC~zP9!1bZh5aD`xi^Is((_d6OzHA@2nBen(>W<>`ekXNH+J7QnjIkYn+(Td5iVOMf%o_gU4QzR$+6KrX;9Y+<^I$Y5 z2J_tkk|$`bDSovWeD|Kp19#e$|B{s@dpnWL`4_oFuotnd zj$Tw0{%?bgY$%>q(!Q@&ESX=^&P} z3JK%s_mKkve5C^DL}`rODr~V* z1oTDbcC!#VYT~tZKBE*YM3XTNYtVi-oY9mjXn;P3o05XAhoSY{6fuVU!Sq?3j)$J| z-Qk8>zsvPV&i-u`_C(yJ`(w`!G6jrW*Y87jxgrwzDea}Y!xlxqvc?9KaF$vI*wYd< z)P)fb+1%y8e=cmMR}%UdJ7LW@Y;iyZCSX+{vF6@j%x_j|^h*tG8x%}7#IW+xM`=!= zskmm|aYOzb(>OkBNp9%LLD&K-r=!}L(ynlGQWT^xBl9|roie-Qh8H_=K=fIrjLx=o z^xhq~GCQ4-R>-1eZQ`u_uz0WLsTk1GTGcs-v&v$t?iE7yNT*5%B;@7Xg^R+|EM=!mW8IrN3fLTVHfhJ zSP+tj`5Pd@y$D6%Z}88sCTmnZV3*-C_z;Ro4g9*WC>|5H9@5OZF_q8eX#5%PyT{vW zzWDWbEs|qfo#SjU+RLx4IP&XvK?3BBlZEP{$O+6J+}uKiC036uc}Vl-laHAw(PAh*;2?i2v&2Q#JIk`p_l)gb+2GLO znOXxwZwVAZA1lH&(96S&^>c5`+E|Fa9BRk-c!|A;O;VTPSqFg$$B5aZs+#l<;mSlj zxMPxHE{(9=|FQ;w6T5OPY$*{i#kZB~*g*RK?z z=<1@L2n1|j7$v(8x_kqcS43*84WFpY2!5UO9HX6Eam<#F%*ih= zw#_ar64l;_@9~SJcIqmGFGE<)Ec24lGCVMfT*7iPU{5$MBOcv-s5gbqB?P(Xub6;7 zemNlD-{EJoIHuD~pw7556Xh3=?MvaUQlGoXRH9(82`v;zO_hmC#5j;RIkL)gAY!hETZ1)c!t=}dn>bSVP%bCnNk>bq_DPb zH+vMbste+@#-*KXtYBnt&_tfL0_g^r2q6efQc z5`m7B=@e?W?bRL0&r=bl6|hotf5W53=YV(`uo2nLm20!v``n)Z{RaZM&0QHxk3w@Y zjbpo2>N4X|V)a@8xaQKd6tRwBY=Oo4%gq31p;w@K#3g5e zOc}{3;6!$TcpaeQYjq>i#n%Dz(Wx#lR1cUJE#Yaa+*eZizob_Gu;y(*pz%7renY3= zCK}+lUa;i{JiG$z0RqC7X+<=A)kG$Bd433fe=zGl(gq}?4V_}#cmIdr`R6h*m>$&9 z0X~uTs@0}#pIW5V8CFOFzaQCc6C1uo>590yJ=5qyEtUgte$X26sh8Tt7{9< z<5V{!9PU5&njdx(JJ2P9WiLK7TRCgyYn3& zbl|vqbpk5CEAv+47km?QT^sNnkjRwNg?D&P1r7Z)UQ_TH=T;|6q2|v;F@XIjJ+B>K z1^o>DF5bV!PHgt~*U^m)%27>8UY5uZqj2#K-Ivp_`fejK-T!7Of_gIL4ESd=AQSFT z*)cE6o@)S1@2U$)PgmpKNNuJhm&*~*##e(~*nHoup4Urv1?dCEYS;c{8vDRq_@o z0t6Qch(fxPNmU?A7qA3fOd&Wt6ltHUo>mU;02p`<@V9jQ#MD`U*}DpZ;t4|8zilNH z1t?bj(S^u_qMTy}<}TsB3OB28nard>7g&iEsH#C=$3~q2wsfUdAu5CJc#yZfoBNoJ zLR($~up|lJ@ds8{zktpS%Hl93$_J@P9(@g_yw_kCs_REpk;Xq(Q?y2LxuAe$S+Hqv zRc;9vRQr0>*q9mTDXqR@7Mth?x^*a}^fqd(b$^MW*U^CCqCe7F?RDkt4vrYAvDbwZ%Kjj3kY@8U0TmmB*E}@Xw4zD!fvn_0&_v zp$4>b;EP&tv#UM0L*v$+ZuV%v>xCPz3|$f%$({PKDFm~`B*iDdGTJe$lw_EZ;n@tL z;!6^Yo)2aebcWvOi-gDyfbOCh&_1SW2S9!jYRYh++ZU*H)?qEf|6Q{d&-z|``yYW4 z@5NVo?wsTNc0ikZ)=%E2M{kMfKqFS~Ku8_qk~30gQrimMAQI063-IG;vug=}xxGN` z(jzAcPLkW54C_vFK-`L4&2oG-0BYviTcsI{IU#)fIJ|FM#a$zIu4sSIs3X_$AcE4G zp}sqeSN{-hi%)8~QwZ9RQYE|HFmLax@_$eu-cQ_7|BC&fHwcUQs;$OND8ETcqo##` z3)0d(l@E}4uRPy%i1snKui4(nf-~SvK!1VkR10t}PV|EsfAYS%acPL5D#9Sl1#gqqZ7P zGxJ(q3~nr&1^G-@)_bG`(;G!-40=PRIC&)1vzyPKUeW3qAC$(}oq+x0PnVbJ6yCS^ zO-LLZ&-8w@R!*I*kWV3mU&Eq+AV?wnlqqR?TOOUnHo61AW5eIyh278MS_K_qS4T=V zl1zYrY#LC0883Te3Xg0WC~&%9d7?pEpp(zCmM-4n^LwDBK3>muA>y!*(Z0O+2|C6q zU{qL>L>9GlUOkleXeb`F=9a$)4nuEURI4o*8jB}=YF?zwm7t;b02rCtAY@1;51Pzg zCv}+U=#J#OL5-!u=Nq>q2SH_H?0jzdg_Ta(oi1QcHbi#(8wim(fM;8&>LnT%FiCh` zD<`S?+tFt0a8;v3ok9&Nm_J*ZmseKjuNmk*4(5)Xge#3-9b}9$7_X9=nYFCR3-dQG z&Mfgry22@JNEo>)h=#0}Mfrji-vIfw50%o9e8MN4saU*Hm0>PPr~@f2s}0vBcD(Pm`E13M z=**lL&M=8@Q8f!9h>GBo2MMQ{bw0{1l3r1qJtizzTbT(`PglLue~+eeX!l7-mroua z6o>H-F+jWZ_(Sg0A^Fv;(}AlOz^}_ufSgfD#XQQSBFmUgm_y7TDBLn1dV&TzA_|MQ zq>yYNo3RJoeMou)^Q+7AjoV?KQN;V|BUa{HWHZras|mEA*SvqEf+y`1u0w}sRZfV1 zW{GW7N<&BVO6B9DVHT2>K{su#V}>1ZXz?o*&Yx7VLY5Q?Pf-$sG$u7_xOIYz^3f_VIGF z$LNnSZJ@*93ms2hP9-ivsHYVpzN-!X)03DP+=K5@8-q#6WPAF6PpPve=~Nj;6lD_j zFzPD-gSBEIg;w;l+I2io^gZFri<%;+8&Hm#)+Z^Veg*gnGoqKo}$cL90 zhl#Q&2W{-BZ-*!uM+;RNF2-WlID1SDawwxYU^?;flO>1h4>|4Fn+51cm=T($3Ho7RBunqSX z#oyfSvo^{cBH>t2n{7Q@NGyZgQ3Dk8#}ta*yeF*<52VrQFv7-UJ@ z`5WLkh2-69;v7Qa;0O#SQu142t`veqs~s_=?-ynFdyW)%rGI;uz80`FG{RTb$T9Tx zwL|`vACNYGevA?(2@`AyP9$(znkWm0K(OKUc%ji5ypvp4DQ}r1vw|HLg&yZ_SM`gD zPfztjWd!&Y(yKt@I>H1~h1_=s?zWyTcXV&47a=eZ#F9`0@LaYw&cS zTS@Jd&{(I@p@g+XHCqBD(puDu#~b&4OHQ`PZ6!In?~Zv^L58pul)|^#nb-AXf{Ct^ z6`2CTn;L)YlnIyiT+&Ms3pdHuUt%xa1fhJ>(udPLm2s_vKuKz`b?AgCWiH}io^{%+>PwLE3@A0DvFqRj9cIXB^ z9ZiTr$Nj=%3`p;UfUdK zP(UaC?sT0EP+PQmZj-4w1eg@U!hU6}{=0Aw*1Qs$kD2gu{x$;OKUb{}0U4vdO#R;^ zv}0Ftc#m`>_Px~;&Uu@2<-ak4v;x`Utaz=xoSNI z#~LUY+_H13l?p*tcy@+ATj;?t$GzxjOi;Opr6c9%PxRpX@QCdh&kT`TZ)A(Xki(Y)OOwZ zqH{ILgs+cgELq%RbiGU~KTYtqRKr?EVg2_7qp`Y1Bk{LruV*Nwl2EYqz&I&wblw&X z{Q{XqaCS2h{ww!y6YNvrY|my_7az#>%vQ_FR>93AZnNex6Bq zYD!;s*V(d@Tbeirio!=gg6D_lQY;D5!LMLN%WA(ztkFO)OH$5H$C7l2F z%`*IrxWMXQ``AQzQ0HmQ!>C0)9qm4yW4xo(iR>=Z;?ysIBh_iQf?S4Zrlj##)U-E8KJ8YT!$?)_UNxrX6GxKmY=u}36XoRFrQNUt|rN|O8e{=GD|(?p7kmx%boS6<=ZG|A&HMvFnvLaiV?QnCZjUkerUZR}jmnvdss16fp zNMpa@r+K;d8^vnj$8gI}?pOKn`E&V!D}|}w6EKm3KMs2>p$HM)<#ix-n9ddC*~C~Y zayKr3-r5q3@X*%a%uFwAtA%9d(KME^ki06>==X5(Q+w;*jnaIZ?FBV`gu?y%Q=_XY z0xaWOwYM3=%SNMwIRB z&iLGY8QsTg;9X$uK(ecH3hiUZ#gfX!>`+kR2+`2}pu@+>Z-!7`E@!sy+6{#7JX3Gz zlapi27g$ge7%xQGe`15Z3Xfp7Do6yq{%T|Ek*GLkWTL`SY6CCttly%He`1jNWNpr3 z;7gnkQbf#}i7;OpZt1&zBr|$eCdTH(WBoQ$PddnLOm~CYbc$Y-IUp&GF8iJKOaSpH z5|TTT;*0f;e)+2Drr21<*gv()L*127By2EEYLZ0mf2XuR{e>9Ry2}l0XmYcr_|z5@vZ*I%5}9oUfC=ML>tBVIgEfa`!cM(M@z z;!n`ZW`KAaUu>1Y{vkjD_HV_hg0#HzYQh89YI` z0+fgdY|fgb^UNF#{n*l6P=V15wt0hEK?EopogW|GV=O@*2B)|NGUP|vRcod^f?k)- zoH^ncz!*c|dB#(8G#w3{UQd8JL;AoNjF1rY1tSPH{0A!==lwvXZn*-|HKlu3BT-c2 zPvTh87lSnQSAxK5VSNnOYRN(0OH!hW_R0lPXc9awp-lT1&zB1WjHN-nod~&nHOM0Y z=6!hSP_BE$0clr?hA<@`j?Z*PwN~Cw0Nw;04vCCNe3kS=lx90qjU0gR!>3d0{dGPk zDyYk=;2@7>BgH%5=xQ3qZ7M||SAXU%E5lz_L zGU*_%;!${uAIYI3#d8AuNiGDhxt}rSJq_OKEtw#O5fFnO4RSZSR;wkOy{#0T<_mCd`eP9qEUQ~NkOE}Q5f{ly}!ORPGmkK?$~(yT;-Nz8gf3Fj#^ff zNh&3XZ|N3UDm(MSy#pzZG0p?pTkdTZb4{*CELEbPGQQMw;dx|Soi+fGD7ARQx<^FF zAGv++AWIiUG8a31IbmS7Pg9tEyfqOsdTj|v!TE9K9-9beZfpi!9Ok@ID`K9sk<44jeY?=fD**}ExMB{w;ItM{d#RvO4YQ#p-n3^vBYZIHL0 zhe9u}X3H4=6x+y25lmkM=hxNtu9&{)d|HGu2IR8<*%`@^FirHuVJJ|t97_*>17TBq zoGxoRDt|_T!(e$ItksYDultB(-xDlSOxml;$J&ddy6p`VPpDuTA9XlE3v^%});I{8n@&U7RF+J9p?&pj!Rr0Gm8Y zWU}6Ssh4;U9IUv!n{V6>*AGB-9bs@Bi~luZiVhI!Z!!*x`UE$Aew#_iIwTw66q91U z#GBwk?qSPZBHvh$zioUtzn31ged`PU_26XRp{2F-h`r&Aq=7eg_us6)PSWDaiJII2 z=<(^E8~Yd%kzR25Jm>N%h597-_+u{KaUaghAZ!oH>&_G621ZSXl0d{{ksHlG0Ho&m z8fTcpcVymPd;%wN*GCJHSy$Bj91Clr0R5G;$7pGrPctwFu*Br zR-3^Y+6YD+?b0jfgGHHTOI(bdRZLkllZGZOrtmdlGdGcLjEP}gc+EJ8NDn73O6S-o z{vYDrGAzojZU0uJySuwvN(5v8LAo2HC8U)aMUY`=2gyNFQUNI`>Fy8|P*J*3h5?j8 zgm;b4^WV1Ty6*e?`^9e#bIy65YaQ#@fBSwc^Cuf3`t7u7%(<>DO{SK0gXI8$tG_&# zN)ZQBd{M=i#&-HTGq8fLz!6iWr{u`zHDG=3Muexr<)Es{+)>^0*3%~zdyy3DX{yFX^pNlT<74{kwFqFTHVoheZx<8calUjy;}!Db1FDxkBbDo z;|IP^XZtYwlVX&%EviKr-j?V^hx)u@aJm}G6Nv9UTmLjNQqAMRm3k3ivO7ZmExsih zY9gR-4pHM>BHO;p;0TpUI)DEPhX-T5#Ctpi@~Dz3SYhiD))%|}K?4Gz(zA`AYLWQ- ze5Y@pIFfeK^hxlxa8W*KP@%EU`t!r~LN@z$@&X$$+nix zIjYM%V!uCd*y~LjeE!LH#)XFvvop0)(Nb4ixgJ4#d%z*}KpV41w>Nm7E{mNwEJ zo(n?_7uLbC{Z`7O#euIz5vF-KI{1)yi&mM*MV{pAb4$5=!WG&T><;O9^9XE@Rq@2I z-oD1kU#>ql9ijJx-g|M6?U#zo*vAhb$nTtlJRlcRcDc}DlaQBqUpKip-oVTT>sNKy zK&^1SA4jpS;Ad|c3>(c{9q!g?eML@Qr*?&C90spD$WRUn+aUEbKUUkNEF&r9oA){Z zjV`9+PFO3EYb+aswqfDHt7JmN8klz1>4gdP## z(L+K9IeRkCb=`ubJgL|?j}r7}(FfJJ;ujlB;oFj!3$^0VHA26|x8j{fSJ+*N9UZo> zC?6^^mkn9`C7+B0^_2`V!_bnhp60tGUM2JPkL-IZO{rm>88 z^g%Kcve)uci4KZell$T@PuP)#`XzU0PyFgn8=`uBnF(2-`dno?RevtakpkJ2q=xgy zIB7HnW|}2s893{J76t|0^W+0LjDmv|%^R3;U6;^C7#domKf4?>;% zti0f#O;j>md6E__4xRsimXy@b7_Wv@zg|pdjs}L06Eeaf;ATXc6&s};BN}gbsDA)N zr?HQ`vXIcYsnJ9KPR6b3XnPH(mg9YH0mu1BGFB zcUi$;vxBKVqSm_Tr6Yn*%XYF;=<}$UgpQys;kJ04C8f#G76FG?oiYlKT^e?0##k|b zV+975Fz)GZU=2=_{TO$fPOg?dqKv2xfw*hVOb`b5G{y%&3?JAjlC=7{r`@BhNWPFIg1Ap>dW-iJju(RZ}ptiqzW zzx8L@)UY?^Kd)oRX#Z*|y5{ex5)Vx)B6;+}UN%PgP@(|2`<~u`F^sBALW(9+o?ni> z(@c&b?P$}nj%3N!g-mCCDSRP&hk`2-y2ek{W|1xg4J%@$RLHH#VaznIL{5Yg5oa7a zVrxmKsLS7|&;C^GK&!L2;_Mf?$60U^PvW1YJI?owANH)Z4zmbBcQ7qwOYy@b&H%5y zi$F&9ZJDVQz@D8#XOiI!sfGo0tgA~E%`}M6png-qlU2b=BB9#|q zB3F-`aw_gDWXMG|d>FMhk@`I8c>2Tu$xE1p7`BYJC-i?d`hij957tOkM82}L;6P5= zm=JPfJUZS}8ZyB~YkH+OskIhnIG*nKm}q8pV|Il``HLrc`SLT_W|1OKXa!lkKPNgS zAsct-Ci}IuPs1R5q?Mx=$`yotIw448WNV7u#DafLu0>%jixGT*7(IluwnJHLe`!~@ z&sU5Z|5${sq90Ep9W6?vHs@7C{#>Elj`tuUBHki?&QEnFmM7rwN;b&pns=Q?4k8nh zBgCjNk&WMb7XbvX>kaaklEN8|d+4;x( zT*l}R{17K&mXe1MjDS%B80iD#L1GTBg*CMf`_hw_eUA zQVaxM@53t{skwUkj@;8L+^KP77HPi5 zetrErmq>0C&gIFcQi{t|h250m)0%4*ABBaco$y*?Z%{usIEA0uZBzkNVXNs;dz3m8Bc_(ppQBq3P=ouB9J=gip!U4F>6@5))({c>^8$x8!D^>%{Ug zNvf+!r17piA2|iO0%*v`&ut?oLpckcE8gIuZlcUtaq(sSk^YLhAcP2)(!mLngyPw& zL!faTdygi(PJ@T%OLD2CRJdj8<2U_9Q_CdvI#AT8TdnQqy&bu*D5Awu|EPPM^16o> zi1XZ>SW3RDxfTsDzW~o04XEy|hun72y=Nbk?As$Km`-DJe_!>*4;@Y?U}hgBjUZ{e z3iAHmn3SUIgR6r!u*HQWh?lfZ%2j_dP~DO?a=g;pd&I)j*&ouif-dD6>&4G(D2EOG zn9vf|x`}ihdQikgI&pVhJqPIwY3KaPID1kc%We|M9l6{4=J|nfZnO)eW;mqe{3OAA z6{Efw^Fm9B^cP|U!}v>1#Mp{f_gjp_(T@8{yrYAAgKzHki{X%Z?iVHzYVv5d5w zw1O*BNocgAVpE11i!WSL#F7cHvciv5p>1!2VUA~Ij83@n~&-+G(bf4ZMRhB=mKK8@`; zi1PA+ir}8;(;90U;iQ+ZKaQX4Qrh3t7?t)oEW-A01EV&R=~nkhI;mD2FJ@lA!*_dF zSZdtP4@aIv?kScuqA4kqc7GPLhwl}=Rb!78d+Nv0K2`sFzhnH{jstL;83}eX_$Ea< zJ<{l0UX5=DFvj44{YWIR97$!w8j&yM2M~pBcHP5{MQX(TDP+lcVzj4tZ8)+LSKNPWHZTeO@(AE)`d1e<<{N9N=n(NrInm*I zWhFztl$dHkS#SUCmLI{B!4ihMHxAl^P|H0&<#ju65aK&m8Xa_I)SZ`H!cLCA3QY7J zMT4G^Jl9>T7ONeHna0m!v&mV^y*0`-VbKhmRa@pTOW|+jp&9Po#;nxTdq&#*em77l zePdLgLbZ;5P}+`yQ$46Bc<5j7OV_XmSgx2#|Cso-Yh%f~)Fv3?L9g$g++}rFo7SCw zt}jAAbbj-G969SZ5_F9X**M=jm_7%>kq_JDgkWJqFW%ncT^wC?{NdCMoS4<`3sI|&WlvVed z%p(O;_i8xn0NuEMDrmbLciZfHVmyEGtD=u_8CxfK2s1WGcc6ZHM<2pwjuA?;uq>e^ z^9r@j+oc`cOR4C25sUaVv>wIOpO(5h4JYH3fszZH7v`9^R)sSRI!dI)m(n#Zn0!5= ziH^mI-CdAQRGFtk)4R5Xx5hggEaAdRyd+Chz=LhKu<+v~y`+N^H2e$PGohAqErn#4bUn&PdF4ee&T^$NKY+Be{PL*p){G zkva262n1OqCKpB~>SU zeC7XpSLXzRW?DL;z#YX*nmPpj-641Ug0rEOBqF9(->m{K0*AzD(dV+`b+SFPf*Ta&SJiU+jfkC))@RM+p#IvKI#wtgFO10txnwPys{DD(lmY~?n@D` z{=N6Y=#;o1)56LuCo0%UMux2rwT}#hVi`SI{^?+|in*c1!O1N4fl3yMXlQGWug^g6 zAFd*Q6CB93ayAA3oTzFm_$ECJ{|MPPQ$BH1uAm@7z{25fV@iglgChOdoo>B9aXg5l z+sOzqwgwBHnBFvDY*qFr6dd2<38d;{M^v2H4j;Y|E9iVEe+()msT9M zm0*|4qj@SXlS*>7qRJ1iUwpB(e?mnA4cGCAZ8{TT7Dr4WMa^fsSH|(ho-xZBmBE7@ zqIxg7zZm6FRZM>I_Xg1>bX518@tHXGKPyS+F-d($kQufzdDm(gP~^$*F0AdviwCrj zGNe`y*SMlbD}g)C{864-wP|h66`5pYq=N{gp{YS!$mIlsaS(SF%}jk$rAlp~StL7H zg+{)z9%$0sd~3m0a;JJ`{0+$KGWQqN_@_Qv_z2f)f>Y!-nRMs#`+Eq=x*6&-tXn|~|`uP1aV9H$Rl|*o4!_2BltGi%%@A>HYCtF<S5*xkmq0= zR=KvqYi08+)Z$%fD__Y_&Ob$OAfZOt;9M_OPg-(^n!u-gZVZU5MhC_ojv zkP3~SJ3rC|NqMF-^xU^JTu-fkfVLIvlkNGwlp9u6s@iM-1jM081`rx?i@3EIZvq5x zC0N;2qR;gWkn9v>E`j{$nnao=N?5BK@7*jiRUIepflG4>D5XyyK_xm z^j%L~jOE#{1H&p|F>r0d)%)BXRF@_2zIJNH6zOd@}B62<|$-RFmJ+03)RlDFxBAr0XU~s^8j* zARLk$OA5FN7?v}jSq*AUs;|3!Z=5$UysfqlQiW^*T4ha@aP=0J8>`E}23st-JO+)y z5e!HQ@&Un*T*6p!yH?#OJTzz)R!#!LX7JKZq75;cuYd_>frD8aa{2(@edfJdy|GR# zd7Zz4OJQ7`)L2pn*>Oc?*0deL(Z3abz&f)x=6vqH39Wv)`{~ANF~!iA^yg`-vUIM(J3F&~>k34>m#=WIj}wxJlg$N@#y=~=PHE*;REChhr;*n&HU&?=|s_b?Td<*gr8Wd6omw=F=raw{DCiE{IF98 z2A)cHc}(p~#!^x3ah;{H4|u7Ze`AWkMoRF<=}V9Bg7j^UC@!tzl217=-aJqmjjtZg zVhVY$cIU3P89#-T=tcH^6UecU(M;*4e_Bq+m5R*Wsb2&cEPYU-fUN&~`0mB4O)$Am z2&W?`Q;UzgUWDZEa0qx=AjQVpf7M59>2gYb@W@iZG{&G6Z9_$HHO2~nt-+u0R`k4lClxjnBCrgQh?qCP)n+ye*|&BUpY>|?tmSrHS{ zu%Gpu^6^lNsPHkM9MEkp`@*7Jk#YamP4~YP&()~8c;XF1W&h80XqGQQW8I$ka*_;5 z%+lEvANNM6)q#3;4)Duclpa_K6QtcK41ffsF^SYVNtr}4H-k}0*2nEh0cz)0m}X~B z%_S_hxF}4&3&!-^(l87iX5@?3CzB9<3sjzlvg=(*7JI=+n~&kc+ND7K#U^j@ZujZc z0X9857dOP}X3l4zN=Nq(U^8ar>R|<{JO*z8pCWZl+>tYF?H>7eC7>=6>kj{qu>B9% ziT0jaAq&q9VhWd{B{Y*O?F_p4K-$H4!? zg8p|8e$Ck}VSV!B>kIQIm`AAN{Eovq^#`#r-F$(v>h5V$c7lJdpYclsy@-ubxN$Z? zMif8`e`~KDhmN1`M;MV9-ll}}HP8Ogi~uh`6cR)ur#)NeA2OTE$5Cn6=WSt;?5)^w9Yx1AtK zzn`NbwXTWHPL`hp|Ht1hRJ%=kFOSI;1MoYs#RV#|s!g=DAS_Yn+%4MnF=!^oP18NkOqyzxrFq_7(A z^LEybwQiiEGb!QA3;5UL$*u#|9M56jF}<*EV$+6z?w)6MO&_0ZzL7cx(X^3Z>6ICm zgp^tRJLxl6E6S!z&oQWK8i1pp^1u6R%{F?>DrezTi+7`EUXbOcJ2TOzQxMhtSqdQ~ z17XpDapM!!-2iWIn7JeJ2yDbB9H)S0pvI|G?Na>Lm1wnK{TPpl0C+b+zsxX{ndhcV ze&gukmp2>h!_?&QMl`K&E)DkeOFW&(MY!HXE8vnCQE_5OOkVO0k zT+MguwTCM95pokjLeslg2a1)$SzUA2EAvyAgY%9=pB08ZZ?`eOuHDy__~fj^Rz;w= z$a$5AA#H40-^#^OZ={@OAYqM!0z1S@1)w$OLJPeQ@j2+&2ekNG28y)-^g;lNQ*V_c zP=DT^Kp( zb?$hBux1hT(<`g$M*+>JcW)DX|7X9kSBn)MRG`2B1dtP@{C2;tKXUP4{L2cktMCS3 z1KY4L>?cqYx#vBY1I)~W?k88VNk$+Yk1geE;FuWTbP&KUG6hz;MDcgq(qp1Wpv;BY zr#}R!-8i^I*v$eIgAmtU0C|wLIfU=<5l1dZ5zyP*D*}0jNs*2KIR=+B`0qn>dl$2#C!2LXLSOid_6NA`G%2gGqEr3lJ09OYJkpqaCcC&0^JO3U$ zsb~M(o(n6nH|zb~&Mm(J_vSD0pUV~VHKUs7v%}BdJ6yUdcp&|D>%8*3J3TMHX#yL@ zXxh?a;ITLd=pC*pf6hJbRUuM_(XA05BnUXeCeHz*a{xGr3IHdZEy4)|;$U}YthLS~ zKP)y9m~Q@nR;LWR_VO#>PU{Z_DRs=NM%eKt3!vooun%2*Ge${oN`yeWNLAli-c=?h?FSeAUU4Jz}`;M^6m-~__QBJ-{2 zuD${vhU~x0G(Nuskl;ShflE4av(CoBml(h-zW&Dw6U-G=UqQJRLcHbn*7ryeb`55e z^|4rpK!p^4cw2`ozJ?VUd6Su$PqQ290At%23`^+njtJnc4xVS)b|Ydw}70d zORfxtq_`v9G67GTmT=hIgU@1=j3sY!-_3R9-YWsv6^Jbyp00810$f0=g+cQ_i)nkU zLQ!E2o>oKe*(hb&NuhU1Wn{+SxQ$)dFG(?oplanWr)^?4-7ZSbrEj(^JpP8D7|ES7 zYJr)D#*#8ggX{l>nc@huI@bSv5fW^cdiDhAp5{W_fS{|uCKNzsAg7HNM**=xN&O*leJA3*e~o|Q#EJQ zXY8dW!i0iA78~~W&N*50uEYJmzId?0m(y&&cTFZ*l$3b7!>+&S1A1(OvhU1Lf6++ENIuP^IOJ9-q@B9ocC80>?oym`gIQLj$# zU)PaEd(XMb=$=d+OJ|?oz(BcB0?utS?YJd5VN6)Dco_1&**uMc>xO6NA56%qifl6D z^!*!ChbKabk&n=;yw2Kjq^^|>-*F?6Dox-pR{PfI{05*9tv_G@IA2aVNLgoMD4J}n zDWxCXVwlua`|=0?NJG5!*(*H+9J~`sw*qukdd@WZeX|?rL+TV{`%-!p3R~(%a6P!* zWVpMaZ4^$xjI_R=3&ix@AjdMx#eiF1q^cH&P}EVsZ|&i~F2<6sLPS_&&H4=9${Q2d za>!FgWAWfVJu%#$6WBL2TQiLsj>oIi9G0W?-I*%qhyPGKUN32b!v;VbI~snC{BexE z>oAA8_ZTSOdUB@E_bqdxm~8ddoLeYaFmB>y*;0kEJqlI7axXR+BCO-!8!@3?m*B;~ zKL1Ffsu+$~ftJ22tQI(6Ttn$bxhFm)RoXkSxPwGi&!$V@btGLt0_q!cvtHIVZ3m!? zIgO>(Lpcnr7vdms6R-(Bz;(8&hL0<3P73E<<421%fxgg9>FC&Fu*}JW z;86}>06fRF%1N_7?^bR*#gj@=0!h4Jsq=Zvf}JF`u6A(~)qQRD9Ojvs<=(hMCclU% zoB`I-Tg$QLFJ*#8i7E8t>l3B{3qnpOfry|*(jc7tx;yOQ{!>B05hW2wiN71f5W7-CWuxhlV< zPWzFlS57fcF<;N1#f%i%tRB%s{Bdn@crpeaAQ(+|9zP+%n~!n15A_mky#+-#8_^#c z@>1_4sd6i1jOg9Xxk7)EbPeW=KRITjNlu%<`47i`p*JBZn^Jl~@+SDqhFX;Oy)5IR zHvy0CYZB0HhZ`4>6W?$AM?G*A&x_z$uoJL~P`O7NZ?Eh6Q zVXOT)YLEnO^E~sv{KsH^Q10gKcv)Ni`*i;oDh8(U%#w|Ny0QKi=6O1R(D!&}GiUzK zdJ2191EqhldNrS?;(saOJ+NN|`6lzhKW!|3uhvEve5SXKP0D|k_5c4}9>_Ir0T$AW z-+(&?Oet6iv4ON-tI|}!0tL3YIS_ek5o#ug0z9d$6O4{nrln3}84yxm?mi{EBVE#l z<)^X0gKsMcY{j<8P$0SD$9(55^GZ*W4G`pb(%hseosA)8YaHZT|XxW6u{ z2?CE!|eBy<`jkm{CxeL)?FQW(vMn}svlac4 z=33$rs6skHAp{zK%w&$K*eL8wTi;jpukTx`MH%o;dl?r!y2RKX7N|Z4%7Ez$59nl%K$+)$u5dYHai9VH}B?*6%DO_(AbRli!F9Cq4chFc*=0Vq`%1z}m?66gfbH;Sl2g=5J zD88^VY3RUEloUkU|0wO#FeK2u{!sE850wUI{__z;*VQpsLJFcvw`pfNe5V0>WDelp zEY_#y*Rj?MG@-(qleZkjg1! z{9bP}l)hs4D}XqZeHArxhf4)@JZ4P zU@?R$s_4;RwCdy?r_M9qBGAE|ze~}F3s8wWFDUB|&xx~qK(T1!mSMhHm&?8M3cbY_ zS=NE{HVeFUlu`8d9PbQ7Ex1(AXxyy!o*S5e-ZWZBMO#)k7S5Dql)U-~-af<#WkGO2 zM7tm|M8pAD&Eb76vFB}&k&o*GybeafOj5IB!65QH#hW?W~wk!!e_LNL|9>qV!|JkrVCux}Sb@N;zvS zQRb@)$dl$X+)P-nOjnVbwN(9)6uOoCKm?lS;TNtzLE>2BjFn5(qU5NDA0K{PoTdEb zbZ=w1=cxy%mqSH8FBp?Y3c2K-$F+gvd9^KSY}h$gJo<5ykVOg}Ju0)l4*I~^AY2-Q zlUyS>1P(b1FTqRqV(Mm6i~HgUNCl_`r|d(#8xg-iqHSrWjTd0wQxqZ2;2GVZR5w;` z2l##V;ajkX3@MiY$lpz6a}w;i1Z=^SUl^?JMuD?H8yGE65>G-2Q`xO3&IkSo>~vnN z0bl<<_5_LTsK*-vHQ)<2J#*^-yim^?5Xlr-X4(Eka4ZF$l!bIo5IS`~04LL2U`r*; z$pvRB==>S5OQis>PTPJx30~l{I-sb;+H`;eO8Q70D;O5~`EHAT=N%9kAT4#h{t-A) zV_j#=U3Y!IVvk^VVY_VLn2lX{fNSRzIz>v}X(n1l#{y1XMm8WpJ8cp~+A5Pq=H`*~ zl$$e?A6)ugbY0GR<*5Au$DfJD1>jLb&JD4ttg#VN!ypwDFp#1Td&Q->E)g!6#c&QR zHBX*hxVC^VL=u!4OcB#MYDc9~6Ep**X=yHcZmNI%BWR$HoJz2x1CYiB!0*Dgh1F=9 zKe@AL2^*7NR|MXX98tub8IH32@LjOfig4U2#sur4Q9@8qV<-T1M2q85aR0-_fZOkg zjg2?=-os}brt8LT=x~-gx}9~W-ew=66g%Icu**UuCps*=TF~v{t{6H>d81po!;8e} zql>@%0rsCmZs+0aN}Jg(;I1&83H$ZSrsamALQlcwLMmhr`xXHTE)?zBW`ay090Esz zQ);Nzv^jAAKpYmq=!aMot|6bp)zBQtie$go(4uWH4@Fsblga*ckypJLbhl{~H|&Bx zgU#tC>tu&)@bMeq`_B>sy#%%z8EjAfwQ@;*Zk$N=!DA+j74j{ETTXW(OY>^G1xgPY zx*AVF&Z?ckc&j^xhg@Rc40E(8;T)bE>Q?dK7I5F_EMl!l@AxvdW@&3E76Xq!lsd?& z2pSbCN4-7;R=^plK~Oa^V&t1dct>rA21+&Czky*Dd5M6FDJGT>p9;w*T-2R4Mq1Oj zU$`rotUT}$t59HSU7v_1_J4rQfr$2Pe~CTXIPH&vy+lVQ?8noQF~Xa;It~OaRmXh! z5{F+C)UNn(6V9d)r56BiWc%APWSB~(@}X{iK5cIzQR*?)Pw04g#&DnyOc}04D$2K3 z-MIBBwwAzyTn=i;l@l_^H~#EsT8Hz3h#!g$p+|;@W2@pNxZ6^%>}aO_Jz#N~h!NhWddv*B@kMCVd;IM@+te zAmFqEWT^dW2T$ZJmyJ}>RyEf#`Q&8S)COu*FC!Fhb0Vj+XDB;VQ`&sAJy3d0<IBt3RT!h*0od`F5!1~aqEpVXw_fDTN-2Oqmwo<}yx%KS(&fu}rqJv0} z$ET)?wV*4J_aD)ZSzfuSavUs9pz;Lmd3`Qy;`+R+5+g;cVgYD(w3@HkN=GX|g+{e} z4NkRbW#G6xZ!mFs;h;&fDHFcX0z70#$Rt@j^vbh9!DRU2ox>>U^_9v~uvZF{^mRcw zd7x@Jl%yP0I;g|+td!c2A{affq3aL&{Lob*xyQh|)&DG7QNGl5IY@46bx%blE-flT zTRLqyNH~OirA?okA9~M)hvRDNRT4L9gVkcWUlfajG!KacHe_xF)en!btzlh3V%$5> zjC(@ORr{;Y`XJorx<{==+<(AIcu415K_|ZgdNr&pB;MdO2mB++EOOG=oc_gcY0Jd| z5NRP+4sM2ZCzp$7FLt}hIr*mbHX{Rsh@xN#u(6(~_lV$&m*sp%BQN&Yo7RIeLnvZE zC&^vqQ&4Cp5o;UpYEj;Ck>Js9@v_r+hs(y3oATu~*nIP4H)4EEBco)MX&DoUFHIuD zg=lkg+HSAv`v_Qsg3C@TWC>F3ese5Zf?WnCj;|`wnSN9Em>gz!C``P9AFnj6*=n_T z>`qL2GIh=(;k6h#6QyX&N{MI{23U zpwV@AJS9fuSFqcM9|}kPu>*lK+l&MA4fBh3){L$=pQQPU^=LFLp+t!29rZNDgn5Eq zDYK{I8rh@{ax*Srt^Bv*;PV|Q=iWY)dI2tm$YaXhGXU_v zLMb)To25&HH4xGf`e_|=IB0E?nq^RwCMELru;AM9M)NwCU=j~3?mc;e-@DGWI{r;I z2t(7g?qxz~jDKhIBqo(-C2WEw>PL?dcsC~;E;b~5HB>rJY5d3}E-(Uw@O~E_%at}F zzOA-1)vfd21vY+#P?lSIyzV6Nv;vp)YPd)S%0T>?-ccj|B|-Si_AaZsl33=v_&d8$ z@oCJb7hQ!b{ug3hYA}6~{g%v!4`O_2ypbc$>NtB^qb5c+q3sSjD3t$N z6n>MD%34w{VG5!k8P>QuHVC{dnNp&zak|0qENgsPl%|rzcHM;AKvQ~`wPa?@6=Gx`30ug4au^YjR8xAV#%WUUwl1hr?lQ8_>GJoG@YG$ArPSi@ zY$`|U=AZIeruD{~7)$&4%SfdQoFhRUA2jN%bs_}OHqv|af$As^l2@1F6iq`2z!X*p|}+Eu9(KMixS6L zuQ)dYEG#=$-FWdOrJpmH7Y$Fv87EwxL`1gck+-QM_54;W5H$|7=Ovh`DfmH4Dn&8Z z9ArWf@f|8IP?9k<@~aOwf58AH!fNn1h1ioCSnVzZWqgugTb2E6(jnDQ%!s~dDjp8& z3IgRvxhANx85?1cPzwF&zRTe?VE2?!l&6A2xZ8aB_ZNS<)2uJU$Fx!tlwA1SQf&N- zvS^70-WK|eUkKLx*6q4(D0f=J$d4Y6Z6obYBGFaKKM=#>x+r;8lZC`llJx|*ix47! z;}AU%+fSYp6bCVua!=G#+(=*V$>ikm^EedsNy-=xd2Z#?Nnr=V;ugKokBEvoK;uAZ z{k+_eF8KAaXwmE@Nz#_Eo>5?&8-h^257c^`Q~aB7>F(qoJvf!?t_;Ous{{Hs9M}ci z6Q^pd0&@hRTBbJ{N9dnn)BalETSA7-TXvpAi)KgXZh~Zeg>ejU`$g#Uv@3!XufjGz z#&_Q?g4&4gKxsmKW09|sxuj|1A+OKF`i$!;`{xE`?uZN0Ob`~_2mS7(=XV!u6nIVh zB-v#;7Y9ga#|&wZE}u00o>j*U%_N);BpXsH5lK9E;&;Co5Qt>6B@vb@0kxU(KAA*$ z(SzWpD3d9B`A1KZl@EGnRUe5D_0G;a)WhV-f0iNFK8~v%>b~KJl@P=R6RHk0%-=z= z^0Kr292_%4! zZWGNzD-rFzchCcDMCCida&><-4aVByNEp{84-VK>!`3o;J1(GGi8Ymp?zozvZZMdeALYIhfHFT-%Z z(dqhxdEurVUK2sgF)g%xbCTYnv(@c}JmcbPsc=CKx`f!B71$0s_ax*Ql%|50LFsKY zKOFNYDSbuy&|V~iNah8x_rM@D%8rwG9Y4FHu6umNDZ`>S>s%#-aZpo{)VS!tX)MP& z`B5Tb=i1`+0G25-YGZ@W`za49KE85}%^s%JhFP=l{UD1+>!$sRhs#_hzTl3vNdo5R zq-w%1OgU7K`%;2ntq*)^Xy7^hzjs43dp&{|S!nE}3%9?thodwSzd)j|vXPptvr8;0 zq2EWB^Y&EL$}j!#8u3V&P<2J%?B&Niv0;89{`Z_gLTbBuebjDRRuxC@lglFOBk8aC zWkf7ep=n%9be{%3a5q)EOC2 zm)N*h%O_4#cG`3H5Stzo^Zzt4SBay(vl&0GD-XeggbIpDmW6%q0JBe87mcRB_jyNv z#Qc}*UUFeRvu4cL;cCU@roZ3SSWn?nUW3TVx~44cSj~D3StPC`u)No26KWgMWznsjA%(f z4c4Z@Mc?*B)3xdSk*_JN$y5T$CF8kKXsDE%|MrLw)x`E$fU=nazeLV z7jKP_L6o8J$P30t5Ja>g96IA>*?H$fj3UW6<7%-J>w8KPA%JKnAg^P`<5dVRJ*H70 zS(3-;wU;UgMlSYbtpjnAdaoqVZ#?O#9)@*T-yr)>k2ty01BcE}ldigCGHucn@q)%2 zHuehFTP}X2`r-6ZwlKL*6UYRI&M!aYmVJl~=j+sZ65AH1=Qaa};nlw*!<&H;^Fryn zs*?$3Zd$1&kc^5_sI)ZXEfac_z-hdnX3;%#**v2`s;E^YZrt6@#1%o7W@MT!yqY29 z$`eAg3DLK6GNx`OF-VCn=ltN#`xU>G#sm2hw(9SqB|u70zbkftUPd7d7=L`)?Jp`_ zEDNNVGL?Qz4c1~0;StE@@!_}|^y>BaYv#^!yL~piRYN`W4w2owZ>Y-1rJPeVNo8sb zj$hiK>&AO=-RfTxc0@ZfNo!4!ej;qxykY+5tBM1U!=YEl9Uq|S0x_EDNfCtJ;#sm@ zwHuPI!g$p)|KLj_x$BLd>%DXBc9QpC5o?mfHK^4>Ssi|C;|GLBbT+jcMvZvmxhN4f!^ktIEWC9$ zcdSKhF%P}8=T{{jYu*xA@thPC@EQ=I{+dKhT)=s_fU@JXU1`$^!`mi(&>7`EI`WJ4 z8y`{R5%uVUm9GV7TTA{-tG5rFad_-u^ILjI<#ieS8ch4zbS9BNE<%3xOyXYI+upKN zg70YgI6Xy-@P?z}tYuedXkN_`GK4tx_93-Ku)<{~o30!d$$6Wp*4!(FAxobeQgB#m zHNRWW$i;OrQ)~PO5B<|9(mhBRjN4|kpvi3ASw8XcJvpuI)xCYQPmuAjhvE0e#!-Cc zjgH!IO@4u;yv?az4g5~1Zjf}@lV^`sATsijP~(vbIKN-oDS0Ns=#0h76WKUJYmnI6 z>YT5a(c-uv(iy%>2~uf$ZC1L6ygxHGq%QB7er7!Q-g+}-N7goP=R@r|WsIWbMaoTf z^T=PrvQgB6$Sb&hxVTiOKiMI2)NN7ewuzw$Hu2AKj^LC{%aget_Y_=mq71j}Hb_Hr zibjGvDJUbH<8H5r1v`QEKerenUNa^}I*n#zXR?F+2kEK9+C*)U_agYR8Fo3)Um~efxk^WfEQxS?LaCBAX9<1z z&8ZhZl6OLrXGoZT#o~swP-2eU>y2@SV@zef>_N3_#wdR3Nr=0oT5&un%?i6-yjNCJ z!N(?aGon(h6py~s8gI0q@9X$1wzfX%!1bN#wSMJU4xv-2C}|~XBiaZV^VJ7UlS(Cg@L;EE zVm4{1WBSJr^~jDI?=;JXt(pD3(+#vu(TsNdq*-$U`t#Oz!?jnKS_uq}WD%3UqMnNz zFT9c#&?j-NVqNO+kVe&xBxe7Lhq<2lTirsILbIEPHHKT77%+SNBeaYLW(#R9Y#n`1mQ{H}`q-bQ8Z)cVv5353`ob18#tn#|^i&y;dW%<-EW z4b3BPSjQXG>ZYTBu98XkH;X7@JUNW}ttRNf>XC0J+LLTH*r`TWCLMd5MvRg6P(8!u zT3Uvs25OYED7E*oq>|{q7&O{7g-j6~8)i(`XHUAK+*2lwS;Mu<(r7KpMi^P2#?1!w zPrNHcGSkad*Cv9&eahozuWG;0`iepa-H5lG&cZ*couKjl#sgN|F8ugXJvS*z{rqE) z#SvD01YYo=d0#ylV@Ooe6_K$IOhlusR+kX>Ot&AlyQ&ZSH{RlQCmsr9JWjFEtwd>7khNNt{wH zgPO(KYF0a;6eTH{s)XcA;kB^8TI#zdwaN)jZ?Q6iGp3Hl96v>Sy~4^#&OPU>k`=n(rc10din!Tw*A&)?4nE~sL4``>4u z|Jz4e0nM~7(e}TnzHh`pnw+czC+iwHjqsQfQVYa@1Y`1d{Uy}z35KmLhw zj$$8p(Y87Vcl>?4z2{iy266b49!W5%uoC@;pSNya_)nX2(Eb00W%5to2dQgrzrQ?k z1j*qxe{#9lX`^q!UK^e&8fs$8^lqbU$GHLV z+KGQrsaQ~&`0wOkN#6cI_4A>0sf&p?N=D~5*6zg=w)V?_4r&K_Z6t_Fvyk|N^B+HV z;p3<`%1=dA14J7h8>#s8XR`cqc>P!<5KEf?uIw`qUDRHJ$+cKOESAN0p^Ex-CI+PC z-uSuT0@}GhK-lqHulvm%|CY>yS);^R=td0*en-Sj3;$Ugo#j&R?+d2$n(-m((zlq> z1Rc#TxzG2C0T2`N=VZRX^|=n{mc|3cxg!7C#D7uzldS%JQy$%ZkJ92FHUTJmE>c=8 z>Wyn^7=M$p++(h^A_owCgHP}syh1T>M%lkQ5Gl}HhX4~u!;vX4i~()w%qC-CLu5Fv57kF5l$fR3FyZo>*LBL zpEvZ2!|UvU74T^TdO3vWZrD!>vG+lBnya##I5W`VK5>#$>+lhVvO~ETGrS zNlp}NS^x=BkXS6>#5o4;gCf9GBmmK+E7oWxn9E7-BqUABs(|>;dIGN05V%;tg{i6bb^!9R>ao+Dfv%=tR0p0AehyXclqTSy_gC7*PDYOD^ET%Z zTl~Mn9{=Y-c?*CXE!fNzBd>nt`WHW~6E3s$GaW7CavRC@NaAi6r)H{pP66Yn|?Y^rJc>_v)n%4G74T~G$!9?R{yxDsYjmbHR- zH!N2Kf)vjH4-mgGglk+5Bohn*LXHHw$TOwg3P42sNs^hwDv{y8emnwb|MTSR_txILuqJq;B=w z0ag<8=Oz}w#5j!DK<0g>zIy<#W5KvxE3Lg z@rkLX!+?qEQ@%v*p+@eb408)?99tY&z)Y`~sdUhu0tqB7{OA6N((IuqR~_)cG;AJ& zaRv%twJ`3$&J}kbw&eLC{V#GI`|B`l*H`Mx@K^YMpkRTwpM}`*SIOl~4Q6iEOJfoh z#sJ(fa>~5<(EicM$N#Urw~op(`@)6=Q7M(~?(P-@X$1t7ZXQBFQbJNhx-m#;6cGVI zDN#aNKtMoRDFG=J32EMa&p3+u`~LghwcbCRwPuDHdG2#R=j?ONzV_bNwc()o2t{}u z-SpPF_=_7*ug8AdMGBXWfkU1yB-~83HU*Ns$)Qa=&BJ_eIa3quH?jG&3!xywp0!6F zH!LR|k}Gl1&OKm4)zb-vPrBD2!RZCUqy&dhBKm_Wz8p-DD77QNy_{WsM~02ne%3Tw zq%Y!`y?HzAKlK}##LFqgJQr%5o3U?ZxLlaPt5>{3)iYPSyolk)0wy^eJj)IKf{-&(<~{ugkBO(A(~?shKhUM#M3F|5r5)ELLiJiM zY6;}WLibl%%rc5aql_6|XEN>G*J(7>!wKKKyzCq@Uz~=TN>IHaqn-Oj!-2(14al&` z9diw0g;5X9p}R&P+q?Eg{@h~z+Nb~(RezkPZ+mca|GbKpFyPb-3sWZ}Bw8ZFaGQFs z5}ZL6DJ^YU$>uJ3?-j;NYDFx5;iyWiz^llKr`Q65C1J-23*JMM0d(3ANKv$`X2%W# zTLZSdy*^ZP549#ym*YvJaM}e08zRq0ArhyB7X^<+$KlEOB$by#UJDU=>zFWrdX}Uf z1uQNaA*k=Y!qwm`i`8hPCsPI00uNHJV7CZpM<*L`7FAAyx^ZJ3Lo*Nm)hJU2l|;~` zkuVigk$=h}gq8>9UJh4f)NPapKRXJtcKgd;Gm*d=ublXz2?f`atQQl1LNgqxBx3dHjUjN_{up@tg`ihi(H zhB0ODdwDAFM4HOvbgI@^3>#Ms{14N8C{wZVRNuinWebP(+y(!cuc=Xo;cg#O!MyJu z6oE+=OfP$cK1+LBl#_$#+#b}#m@|#bx-q6lzLAIurne`mxWk%c(KO%LXw&wzsk-AF z<53|f+HT_BNgn`;VKb4SET26SM`<&~WxSi_K_`TnIEY+BFd{d=I!JN9(5J@_#rr3* zOIl_|r``-9Txz404Z0jmtTRs9OD3++i@oBkwfWw+B<(WJ$_(P;J)4KD*1zmDq+$^H zsVb?iJ4-J4L*obfPp^BASXM3S(HGIap-s4By_>meK!DYDEKoM2c+V~6S*H+MqHjmA zhRQBI{dpqA5$&1ff{ywn!Ms5-?dgnr*YAG?3Gzj5|ZS6q3;l7XO6k){(aQg8ul z>G4-e{cmJ;>5>-hf3wNd-X=VZO-%RFUv%jyk+f?6nyhT;P2!Y|vrq8J66eNh_o4mC z`@O^wK@Q{_9~?$T3eot+&Iu5hHB-e~&g2tgI~qvFf9I8mEKSH6ImQot=~}B$_-O6cC2DtaM|hNmG`(@L|8wpTDX>hb=@yjLlDnjp`p9yWMEwdlTbi1MU+~ z^k3_)zrPo?29h8$N3#vVmJ8$K^-^3@rQ4J9i3V}6S#FkbOqdMOl(3S z@>KkNXHqT+{{maSzY_TO$7xT$a8}x-1lhnQP<0O9;m7Z>#iEz1y04RB+L>S0zJ%>2 z5dcT+8YtuSYLEOb((T6lKe`neObNE4@XHAJ3FJ%jbZCBKD*QXwQFs9&QP||T6!f}% zxBpmz>*#BkrRG|#C9-QqhyKC4zpT)8bVm71g-2?!aTGom{zR2B0SSQS?4D&YA6(Y& zvzexzNo*a5v;&!FnL8^Fky`)V!Yv!r|B@INLoejhv4MI-=GEKbdsMKSgua5tt;IHw zb0p3O7KZ<7IRA`r=)251oi}K`X3bkzAq_#k1G!rSk~+VcLjVc(pFqwQhGk2z`X8^P zk@*y}w7yCT1|Dd4O(nbQ6V*a}$Ex*jQj4K8eXCD>!3bIH`Y%-&> zkVlpTKC~PYTdH^r0d!I6HD1yaR{uq!-uMhORx+O87>0gJA8(;I`XPv{zVc)VeYV00 z()n8h?-UQ%Znq;q{o&|gLiTQ}BjEOz6@O5BRDl!_Z-Yx?SUp&eZn$8CRJp?6nw}BX z4os7B%4{*XlX~S}NO7>qtppMq8Lul;x|DjqDyf4)BhDQj3zbe{oUB2lFU$4AC-eZh zPO(#{;C(tPCrfn8i>$nxc5nByzo-Xx?1lf%Y~;d?0OQN|JXX9(ZD7`&d1ilWpf+z3 zF1Y7VBi02z*F6xVUp-+wr&EYDFnQK@L8Z*@=DxZ)eD49kRJCm!kBZACT&qC?8tQ{S zgSaYbL=9jL=(+M@e^%(Yl-Y(h%c$I;u;nx6VH#Xdq7`kMzN#HGbs16;=cD?hHI*WT zy*R3X(Oh>|M4_g*3rhXy-+`q6t&m+H<^eMEOPK&*bB=bV*Al7fT^UI}s$D5Hvjadw z8bHqmpBi|htg5Z&D&qE_@Rsbt9i)<$#PJg8ab!6JLxqO2<1)4clOr`H=Oz$6dtmb6 zS0Kbbez`p(D3zQB`=kqs188O$?2JIw%5#8~jik5)}q{)1U;rhhQUGt}*+kuLtWJRCuf1 z5ZU6=H$b>r118`H$`vT5`n=I zc}MBPp?)vzqAzs*;cv)^J$=qL>bbeBz7Ym!A?93}NxNz?qwSJitkn%}eJud8bigI4 z!NWz+Q7csWQNF8T}5F*bz2*+5NAd65A+g1zCj{lJ$AwKfG%)5wL8#K46$ zffXgkkySBtC?i&RD+#2|59Z#(ATQmhkrNlq4JzUbpl$vQDZhn0Zn6c?2j3$lqOI*9 z#^DCYeH1i}%^Qzf2@EX(;n%-IdFx+`|b zAd=!wCbXilHV|rK%fXv`2IaGIB5eri2Ue{an0xyLPdia8q?hh#B5pM~LQLe4iwrZQ zzS2r>PoFB^Is@wc4=9k{29l%>L$ImB%QZM7kV@1cVv511m(*{FwIUQ!F+r_}BUy+_ z`H2U35fQ<0<9Zn9Pj6Zp9LJx! zzjE@*fPwP%mdkVWH1+ENtWkFKIt2Nhl}X{O>6J>cG*M^i`EJ^tbcq1+pgl;W$V({O zBb5SA$R*rtS`~|P03z5UwYDoQ1m@<~5p{Suh$EyYzfqiEYeP!#w%3=ofvoo=Xocva z>KV)s}ipPoTnK=zuo0e&(Ca2ydz5s4-I*kG{2-rp~R8hFas&Parv;jnaD3&px;Z=T4jh;Kkfc5G$c@DN6E! zW_C$7+2I?nh3h4dHhmC$BDgwDs1>-Z18Yt9JU7LT<0udQdF&y!WPT;aXrUeu@lhd* ztTWpC0>@`4P6@NIb4n6EbUs(9!TYVZ5i`2fx%Fo%Xh$RxWvm78bFRqlk|*5a@g)=y zn6rhi6?2^Fjv%<={OS<_3ig)dImM9{7I|#e*7>9>_a^1O1~qw;osjz&dOxv&XHpG} zYMAH@`B6^(9#;EHM>6m%3^4q3z7`7s1HDNkclG+8&O!W(Kzv%xarTxB|UA9Bc~C_h|KuJfrVui z6Ma%&KYY`uV+Ey5#ELrg)+~uY`Xud25XMv%NGe(A$WNLk?1j4|tBRns#D`M%2LpN*(Zy+EC#It8(u7>?J(cq!D^I?mrMWf3 zi%X3y*d7BXVL1`aN5x;9wM-hcOL1>=ENPBN#2vaUKIYb|HgT+6S;$B**N?w$7V)$^ zPgr>A#o0hWlQYABxT|TW0LajS1q-UqL33K>q=dOirA@NZTCs1_^bOc>qv>fHIV8>b zZDYLQ){?tMLWwe`&gL%oq{^l%~KQ9*1%dZ=$ZzlPH2oT#c zV4T}5^YY9{V3%($%0XZ4A`^BVe%~HXY#157sMuX?+lE{wjnsdSXbq5`n zM09G+>n|B0z{8}1$^R|Fg^jCZRF6B{3Te&;4$2rD7Rnoe?unOo$#iOV;L^@m7U+0| z`hw&7I*GL@w(NGazS&M`e*hHyH{vU4NG0G+CN>RrrapnjhSB%Kl9HRQXCm>_JFj?M zmk*oow35C|M9LAzPPJz|dc|Gvxgk9183}7Je^@w{<+z5F;>ln$LTORO!$Y!|QRedI zYs~1PT2WbwJ9>KF*$`?Dw8~HZ+MKVdm0n3ORc!MHPV<)Qj5RgEs5yXHT7$%KU<)EC;k3jY2>lDPFCj++CiRaCgs@dK*n?GD_5Ss zJ#lH72)i`)P{Uch#}BwW74DeB0dE_BEBwJK^dJg&Ky#CBMmrh~pkzB}EGLjGP24{j z?j!dhOgp22ge25@12-@pvlq|RHWR(%L%lrNVdC83U|d`y;F`(>^gQ%1*SlClEEJ6| zQ+gwVfKj%wudq86{LJVqMGS8NR;y<6;GiMrIi9u$ROioKXMgATYu{KKQM6oM=dqT) zV@XZy{|1I#lZv5 zg_z{AUhMG8a$F>$T%o7$ubxr*1iCk}x?F|SrlR3!VJvUk`eQ}5tk*j)Ojmg*;J zJD~A^I6^Epsv&&2(@oZ{Aox1^tOs>bgpTqr(88t!L?K8B4FHq~G1bXRBE9b9Ld!SL zK&nAp+Ctqd%=|PLCDAeLvdwU*fM0-2cMRf=R{7d!TS!7zH+C@fTpUeZO990mE9 zFxrxyZcL~A-vUwpu0t4lv{EqJ4KH&SJ3R?CuG=%_k{SQwrC=iIx;?eR-pJq*MO8`P!($D73ssLaH9TWX=MW1u!Z(0Eg=-jK%dpoSIig7^^~8B8}0gVW$UTwX$uRB|=bpf6CH2c22JmooZ`Gb@rppxbk?|Z(G?UAW}arDSSp@ z{>kSQ&Di!nLFo)lGjk{vTVH_+%KA^M%A`iB%;cxZsEAsQL2;@##90)QPK)#?GaiT5 zT5g_$CtiOc`UP))x2@L%Hkw*&QiL}%A;-7^-DNDMAhUA?f~L7XYQNdhKc?x%7N8S} z{Tf*Pz)I7wH>|z3iyHo=q2hN7mQf$D{~impxY7Z;QS8` zv_z%M<-Mc&$DH!tUd(LAl!X0-$hSa#t=}>d+J}CCw%RtJtYRbZ;9$pk?RFvn5pscMnA?!oMAYoTK)-_gpcH6(RJkwoi`$vz0Q`Jr6)ja*y9r0O z51={d*4?Gy)~=3>ev*vO?0&qCT_L*R`x#YALzB#30(Ny_1kOi?!ft(gf$e65rl{Hh zwvA|1P-svVlU+B%ldraS z%kBfGh93V>2|tMVluEodrIF`!+Jmii+sWegk!$Nimyqy;L1|D2!Z}{55tWR=Dr^sT(T<<=Yc!z3F+u79w<>Dsmwv3 zL^T!N`8;%fe&qGMvB? zYg}VXn7?TbV&}x39v0S6mK5*s7}rkujiwsY+5a@x=l~k@7wFfX>AouD;j*b?;x*mv zU^4>-7&wR|@oN#7OMD8tu+uSkD?AiS5LGmxC*L(wt9T1mB@?JF( zdBj)ep(RER9yyx5X*VR7U%R))J5MM=KzaX>EBD6){<{Y+)2a~IW<@U44#DV!Twfp; z{?9^no^r}K?KjV!V(3^;HTp;zUVju0!u1h^OFvTw@xf(8LkU!x9web6bm6UUsqlo8 zV=my)Lg2s)QxR{RPOpx3nL+>$hpu17*&LJ&svxq61YMA@XrHE%6=}fHdfLW&kdxX5 z>bo(7dmu5Ve|zB=ezk#pNfTSp8)l+CP&Xiuf_U0E@bwiA1(TsM*Fw6pkv&hnFarNB z-9D1-=?<{NdpVJZsE$w&C@{$0G0h5@A(W!8lyR^vNM_#=f*?j82>_BMs zVI2tGVTz%{#K&ehwSRZle=l|gR^$rA;|O(8qlCS?h~Rj(h`FdCnAHy!uyT`?rT%)R1dh24?i&nrJX`|$#xIa@vpQjk*$m^I{Susn+292sUMSf4 zKu=d+8>L6j^#t-=Q2>7m3SIycT!15}4lOX&1hz5snbn+uh3WJpM;Beh4i+2?En!Vf zDMk-1W&ugw1+TjD z67ZZS=tJ>zGKueGq^Hb`LA)%8GB@=~o9_xaogAgyp_r5?K!X?qgUaH?c%1ZSUfBk; z;&tkQv?>->Ng7+n1r=`Ssy=?cRq;h&9(ERUzS5y!vvU(v{K{Bf8q6f5mxk%+lUBJl z-D;v6y={o5?&k6N8|maXmc_JO0TnX&U1;?#J};v-c!Vn?su_EU;V@%-*m6@2Ci)e% zIn7nMDvZN7$v6~dzyRYr`RoVzu84%=skAdComAe49Y(c_J?=1(y#gnKN!V&hWM^H1 zD$PMnzHVU*Z?|acL2O_V5~Il;?vO<>Qyj$>USLgz!V+3enbP1_XO_*C`P_oVnxGnq zpdxhW(!>Nv`PLEfAus~NNZA>4$`SL)y_|^3+n0P=6<3+4U!Iv5tqV0~-oM4&DZrSD ze}aBP01jb6*`SX{->|V1@aKlh+G$4(_9G4A>zG4W@Ly!6-V%JmIkwZ$O5I3iwF1@_ zsJ?=u+jUm`%r#l2hx8K!!wOnqwk{u-ZV?dNd&UrSKZ%dnpNgBL>5eE%FFr2n1Ce=3 zpWqC;x=fHvN@xn79+zXdU*M2~{Np_Fw=-kp)=BhH9oXWpnm6be{N(Kh31+ZDjJ585 z(Rg7eAKcnp#DZTxvFV?j-+ZrJ!!=v8xtYG~L9 zn&>ywdtLDjY^c= zQB7qqrMtK1n#cDr7L)w(^~oQm6RBA*&D{m zUmt1@AH%+U!IS71Nq`klEZxHjPe_8rmm5_|8(NeUC z{Y(Rt)p0IWl;9l<;<=$gSj#;i(uR=8#t*Hzp6^glS%jQeAh{}+OaotnbM6jC3Cm?Z zNTJ;hWk1i)8o3yELa-Gxn9m#vG*oB9{Z;Exr8urJNo7Q$x5DJi$VdrWj%yML=7-HR ze!Y9wOa|MZ>%A%+9gn8bBUF}i{iTA*4%t?iPpl@zNeWvghJO079X>;Dp^R{+C_Qrt zGK&~Upmmu0uod$$$k{|Cov&g!U&+RANuq9i%WQ%^PJcBtK->s*4rlhb;8fCivY6=} zK?}~p3&VjL<<#4mSfSTt%Z;xY?EjO(ZWn5G3zGi$Z8&hsAhsJmh9EqCNV2;WjO zUHoaHzDUPBirltb&^hEKQGuT7_3O;KNElmnhJ}{Ek1n&pl6i(7|03Pj`nXh2BGLEu zOqj$ZuIAQwwQrkDa`o{KBa&d`4YFD{=FT^oPHF~0o{V=`S&IDIOWsQ*w_HX?WNu8z z$_I=27(J=dX%Zv7WSLI-z}gevqTq($8zJjz%`QBq2zSia4>W}`*wk}3Rf8==2$eL) zLAsy3rQKQV2*F*mRFzzr{zIrZoXcB$vJv+3E%i^-%#*V`-+#rJoxZ1))J=)As}UNP zF`Tby_LwHB3UK~={=WC`J{p$uiIQ z{?p_(f=n9NAVw+seKef-PdUfGe_!~ zq~wzL*$cX;WyQG+8nw8tNkuETZnzhgo=BdtXg`@9sz*RiKr%VOMnj-}=2Wq>So}mw z=Z2=>4c-D=Y8P;?2r(q=*(e2hxwfona38%Xj5Z>57ffx8k zRg{cqzxI+ww997533u@VRh%P9R~wexaB$|tLJ31rWSAAZ{4(7#@5mhF7wI}Q+h{un zkB7W%%v9~D<0KA9a#KA9lNc7!6q`qU*kt2=VL~4V5VEt2QK^YfXVqVCP*5bD*bHxK zoIlX9*hv>lN4~jZ>cLq@2AmSfBo?&WoI& z$qRY`U#f}5!}mWsaK4R5Zvzi61ZVmi-emYLh_OE>GteHCINylJogfWx zX6o)pwn4nxLCsx8jq78Y6So~U6h+<#YS6Y|N7qG8J|zBVi9Tu^y5?eV;qzA?`|QKP z!Ly&|Ekx|z7T8`~uCr#w=1fq0*wa(*)bYSjnb2~gNmM99=<>2ExS4b9E`>d<7de^a zOI{QRX0p2$SxXg^$&1c1qk{sUFC|X69W_nGVFK03grfw^m}MWsBoj{BFxa;^XC(|2 z2F!Sd9WeeS1mM5R`=>#=!8oXcS~e11OonR|Bmo6qSi$XIRI04xAjzpJs1X*Q9}B^Q z7z3q+1Mxzc1tju7t)!xf@cqm(TT}&K_jR&5yqeQX3mvP%Rv9TS5N;psqoC@$_0-e8 z{Ya&%>NKNu%MCSd-q#0pq(8O-EbqJdIL3Wp72Vk0IP1K>l>PTu&iCD_aW=qU0eiE3 zHGc7g$gg2##H=`Y;mdqAmN|65%Ko*tyn6r!&Y5^n7XJD`I0NEI`8c#TdFcN518l7e zaMCjKK94?F1M*KiC!h@}Q1B(i`_GntS~Ukdyu+=;cZz>Rm}I6QFQr|BgZbaV^YaLb zCxUl))$W*c0Q~;DPcpwk9E4-+2*!_X^WRVb=dj@&K3G+}`T3mxrpXJ)9r2&h(SMwA z|NV*#Gx82!pN!c2IYlpEirTKCq5U_&kH}T#E#w`%h8+H^5!5n)DVlzM{m_2{{Fp_J zv&cKVSp2(8f0yYWGyJ=ozVGP2yXoKi=kKBS_ij3PLH)g(zTdciU5J0p;;#$wuY38| z)B6rs{<;u*2=E2T z(1_6q_zXiAafW)+rgz6xPb@8s_^^)F)OK+{5{JySddAr7K4)3&PS-vjoy_$^Xy_Q2 z|NIcJihdz$mHW_v|MkzGDYMfNpA%n!$Q!3fO3U-#gz zdjJ^!uY2$}dVs`a{<;VM_uT_NE~~c@EXCh2NO+7YzgWbo{- z<6+|ExuQMPNIO~l^|$yN8oKt8d$QX4i%h?T_b~gOUDwL%Jf`RMJwEs|WS0+v=W}@w z{66-)kqn2{?BN2*U%!jYbs23KT$|+7Z$UdWbax&);`|jf9hx5x|NcW}J`BzgnfAxn z5~Ok1Z#NHd#{MxS+Aw&0)A>KfPJ=k*yzljRr9Y;G0|swHvHmf3C(MZZ>AE(?-={>2 zI1c;&ujL4U{s0+YG5aFxW?tVNC^CKyYG3{R{oW5BlDgy4p7!lX;-6}Xb+WUYZyNBR z-5P%pU-ACitj`S3-CEyM=Lr@%)o z)4VrV&1Wfb?E~rBl6p;g>JUH1dQDw@FwWh(cenR_61nsVyt;$12%BT5PoWya_A;Ei z)67T89To;2n^t?f@A#dzYP(}%ysuRAt**8sdEn9GzJ-^Gpvyj7D^;%hcCLDF)#Py= zwd#-AHS5CS(knJw{rG12@krf!EgjlJxUjZEK_TTZH}VwxPQw_^bUOaHdh^PN%}o&0aZ za%<(^hW!Ubek@!-qE2S7R;y7~qKDbe`WDGYe-0u+VUML@W&*o0*5wv2%Mqz$e#-<6 zpNcEbLj^-tP1toDS6_TIZu{V<_5Df z$GYx|=(31fKcJXsO(UPe5pwCh*>h2GFl;H-0v-5H@GV}=0Gm+>g zXe-hv+%v;Uwaa)>$(xXPZ8^HImgpmi(&t^~!N||vJwCiN4v{}U?tpyU$JfZ%F(3(_ zuw7+VB+taJ7~}j}m&fO5)|jRDt6k>;99>pTn1?=nlbWl3*Ru7~gaZta{jr7ltfm8f z8E-39ood0OJCNDA>NMA^+{fo4?9xjxaJ&vNC44gCv8T$9Wy`b%s~0{c#1r-7RlfgF zd<_1?2XElvAKO=$6L!#0mWw&Y(dfDtUMr43isLpc$4%k%$P&l6wsLoDXYrDByXGT~ zpC4USFw--w`EUvYv+rgdP!xAFc0RCbc7K^@Km>e&8M_JExeMBUt}`9L^0;iWk}{y0 zEO1}fv>-4r^n~rk*d?mHn1Py|%v~j?%kTJSm+vROcQ&Q&D+@h0SYpv~wPq#F2j~l+ zngueVyO%ms$?7ZOth2l6pch}@6Yp#Y!Q71`}jJ(+b+fyXR+H{ z56Bhols)!aS{!<6fogg_w)-h#Z}q{hlx@mqQ};J-boaV+rPda={HvDU?0c44eci|v zy8o@J8xhb3DX~~Go7On`&z>!(P5_?~$V<*38B~4xChviq1)mHuB;LJVJhBhypMA!* zn`3Ed{Y#-g5E0G=z zP0H-FRp~g!5-4Gh!vRAfvdFu4?DA}9T6+TbT_M{ZkiTqaOyo7Csh)r3#~gP2=Xqj` zlf>Cq7ORf+fgntJIA?sI`jh*z;HPCT9y0f6`=U(-;uoBUl8m2fsuBr zJ9KiUv9bga@aA5xC@X6U%K2C~NG|QaJu=q32Ws7-6kic4td!|SJ?Z>mHR7LF(t7E2 zEgN;5Q_=x1>GAx1T6QwkuYJAEA3p`OVCaYn+VEh7dy%F)lP%JF%fQffZ|AOP>}2@% zY=3W?w|cb)SVEIom6Uk}Utp(Lj|P(X-&Qiv`c&{ay1jPaOs-+Sa-mJI#(ggQHK9^s z(+NrmCK57<)z41b6`>?19{mM}*QwVUw)dzqYS!`sBP=?_b2oKe6L}2EmIX;)Y%S)T ztl8Ll9=p4-KT>>uso~OV4)qJ_lnVJ|Z`50|QKt=5C7uI2-G8&)v{^NS7ydoXe-*mD zj@ppWf&|Vr)2*Jw?A^&65O&`)9NX$QeE&6pcK4%F#`fqeD#lPnMWqzw_70Tg{D344 zV)ZMtUA*e3z@NuQKqrRDj%(3V#zj^cxy&CyaUw<@ENVD>UJapRG9a!qEN4%j_nVH_bsR2bcYGBt-83bVA;Xqy*2qtNrsiw><3&(UNA%QN zV&{F-JHB~+7X4JcK1RD;ODi#b*Z*oge&X&zm0ICI?QG>rjP2f3c3Hcr9*2$m)XL?t zQSZ+gwQjFUdRe2zXOjX49PO&7YDsGCYjaz#n@BHMwa`jD)QA}36J(7|8+lUH*`2GN zK||bh?`e#?6mY%YT=U;(pkw^%muaFv>_xyb|{93xkf3JM$RDyH` zGb(f6x72EDe@k;SV=FQ!d%g84>h$o*YhMCeuXChKgi7z2OCI5MRN47-wU#ncPutgE zStR#qlUgJ(wO*+em*lGHb5qdnb7&&1X8!3jX6Aqon-eaDA%NY8sxApp8Esub&c})4 zPGQ^iC=kK#{NqGv@a-ft?mpHfev zc^crj^|rUf|MEgiSkU_`eP=$4FA&Y+!Ckl- z8T%h?At!m!)oM;1&REudB~mHSkq6@D80ca5yT5fw*VFunXbVWb;qn%~mHDwjw0cg> z<*QUO> zb$=mMZtGbfxcgMSdb3I;#_*Z>!m0hP)M&$zPrjbZQn9?%pKe)v-f0nizv-;9{dJ1v zNkeXokK48&OqnCfqMlB*px!ju zT|(WC+vGaUTlF#f;l&5dDD8}`z6TPgvkIzBN~lFe6dyFr%epvoInz`~b*4+5V#o@M znA9pE>j!DRo;{Z}>MpaBKfF{l^k5meox39`Pi*W!DrCWhA7hKL*mb()hWAoSH>Ktu)ST(&g`TX5AqN^0m zPr92NOs8Ky(kYBkc->y{&av^Vi2b0(`w_AhrRPO@B@F9}bILZclWvBUhkv5Q)iY+$ zSZ$ftD)b~R&AyZP-W|>^zJgJTFVUWoUugF?S|WNuA12tPU8d>06Bb^cT7X{NeLE&1 zT}m94ff?b?yN@D(+MN+oTcL}uE_u6VihSN(y+bTgSHZnHB53kCwK~BPg?i;U0~bPmlg47IpI*iI=V~OB}w|4U7C?#8Su0RD0*3<=eDkQ zWOzj8e3Q5wo4#stX{;>c$IT{Fh=r@lK@@u|Mt4+L=koM38uUN|GeB%5Y@aiE%ugEo zB2QtJ2E*%c<|Lv+`%Oh|%9oeyqkW z`RbtV{aAAF=nrEHpdcl??jFp58v#cRaTc3!d1OTi>xC@kFqB1qfjmGY1+tiVX>CLbeuA98kc(ECEiXa>FM#L-mCXXQbbXC(p zzwujkI%aKm*c}`kmMd&tn@pJ06lY?_na#cWT)plT>$h9*N=jUyiywxUdZ_zZn{6%s z8%NCo_s_33O!U!-gNR+drJeyRB28gyWoI;tl!b728|SfO#@Sle+0Gc&(oh-e0Kk!O zcT0l&2e_iD&o|tNXr_LYd4gX!l; zQXF_^Rd$%Q*K~2~s4ZRIxB5ls%a-6fwGYWO`B^SM9pTG+Q3(>=SGR4lzM4^D+`?=u zh<<%n^s(`CkG-AE!otF{`HrU6I|H(#!X6;bd(~%3Z`+A@gkN~(24U`)_gKc~24)-UbAh;Rl6&(T@boGb+m*5 zmm?BjX-+qX0~-sEPGbz6 z5}?IXYqcQm=xC%LjIFej&7lif^+kGm66HSeXzp%fPpsM@HX!k-BDBtoC=UC0v!?6o zgHOq04e*R}4B~c~QfSe#G5bV^Q*h;)hV(=a&)pt%QPZE+{@v`|h{En!eeBMn{??LS z^FY?+svaO}F0OaWeI*GRqv!fVz2w9PROc{l4j!BT44Ao(ezP=+LYU(d^@3+Ue74Cz zwuzyax{ETS`O$2oYT~8Avdg+$?xkI)x9|Aa|GaztL-?y_=+cdd&Z-nJYrWv;?Ea?3 zW{uSpl*!{r6{(ds2?3HNkPC-{5F&Ho@fJT1zVA;m#w;+cJ$)-7LLs+(>X9lfhxCGX zj_17Zr51T6qGN9Bwxrur2HX33r97RV8Go+Xzmp|H5YVc}U>oip?9oMHXJhI)b}?)6 zeTnC)-gYnX2%=7AzsgM&p_-j(hrx~&) zMM^NQPihxP8t+7PUGi=W!-sA>yp9Zy!a}#=v575j9#NO`+G>-~8be190-v%2PJ%0wPx3YRuO zHr_N?M+~@gWt;RA`SEe%rzo7JYAO{XNup_`Pd~}JQhNaQe>Y--fY!kNB4f&{(a~i5 z3^WoXkCyK|4XP4;_(qFOATw!O>n3*7NS*^Uh=B$~clpfg(T-7dJ9wS=ZwLO^lYRH= zB8WM49-B~;YmDk>&zwZbd#6(jXBocjLQ#ttpC1q8E-M%CYUlP%`}K> zHKw#L3$7ELbcnUTr+j15fr-B9(dycN2zwh85OBzsg1aOn?%?%=QjZHxhlG}l8l6hj43xwtSrMa`%@uU>mO&*$tcPQr6 zmjxqYar44L_fCm=PdF?$4PH!t`Nz7N;b9S(D;qgm(kULZV@ z7z5*2KdZ65bh>wg;m0g||6%+-F`HmqWfbc9=talXP3&V<99mk2n!ZgVrz8LTZ3t2x z)_)tS!@^GtW`kzLgS#u<$@2Xe+AF+G9JdQ8>Xvws?mko~}OF9I9omDt5UnFaG G!WoVO|vG5whs!lB`RJ_I}> z_`i?*$2R}(j{Wb2`QLc?-*oYBDfa(ywv>qas!?oN+TH4PW;vd{-0pW*q?}G}%cKi8n3ytZH2armR!tW>K!ozD(hZr9K< zuP<%Tf>;(EwSl$(Gz1Z1lI7Mpu8kX*yJQ2#0a($7h=aOtpVjFl&1x z-6y=~vTH?+)h<6)cnH_@>z^y9Q3{uq|FaYZO3dM4G`5`Qe*HRm!Ra^75U^yNknU*h z|J0VYvpHQ@0%QR%{b|5O&tGX(j?Gd_g^64(xe2WLo&}?G45*01uQ!J|JMN1z9EWs*CJSV9{BQRj?yjh~j3uBL z9Pj?Q3vY8dkZZb{P^uFV?fRp3G)0$Kg{AD4qJ{A6+fp^iu>R9M&h5b#b;uR97{k)a(qt*^_B0>|3?tVzeQU>XCrQyH%(58%wAr2b)U}@X;TykJsr9g9`2DD-7htAI_(< zggi`r(e^x7yuw$wGaSF$C{T}!-5roSwRTBDTg&R_-kY#j6E--^ew&5g9Z|tzE%bzj zG06F&k^jUYWg}f?eLZg!FuKI_ROCG?!N#gP;yXm_9i`j%DgTBAncg}fg+90bn18uR z6(j$3w|`aDa`1BAH@?PqA94B2s}5hIE!*d?IYh8w`NJZxQ{1`(z$L8NuV`^>zk&4% zN4RjAj2gjDQ&Noln>~+Kgb%8obR{TqG>6D3G%Z+Krg_ZiGBykDfyPD7da%XsAC3Z6 z^JStq+plSmtqr}lld8@76B*#H1JvOn=Mz#hL8ulxBWY4gArj+9{#U}?0`^niJ?ymV z?dD_Os?==pJN&F_0CGkd+9du4GZ1K$d1;oT=`Z}R*FBd#h=Y>G#UJm*A8%AE4cn}V zLp4rafPFpgdJf$DDPjZn$tIW04+|E=NTQyrepgz04t=p>Vt4jKOXgr@%dyPnoopYD z=B-5ON|Y0klU{sYL0!b@P}C4vVQQrx91w|_5o2YORGFpEs{QAC(~)XpK<>LdSZEzR z@vWdaoX`IoO(P&Wnt6q0=pP0*uh*wYv}KV1`<#B7TFf@9WR%@bH(aDl1biXdIFZw` z+OgZCfX87$+}h=jn=zlBVG7-Bfpu#x@8{(iv!wgMK-P#szTnDs#yAoE0LEm zLTZ(X+erUfAZ;fnAe;2V9syW?Y~c-42t9#HDy%ripC(Llmo787_Y(=y*>;N2sZx&3 zbWz*wuhQp~AB|nzri71LLHH9myr`wS3R0?6zlge-kP{P1LmDNr2v@A>sQs>l4x7@H^>$_=YG=`@gZ_7EGLlOL zX}3vUN&1+U#xPowmwMo#**fYjvc;=)FWi$Y@<3Hn5sAV3qQ6T5kgcx$AV%`dY{iub zS=hTuINxdAcPD~6l?@Iic?vooXVji|TPtJQ2 zCk}s=_gNf%66ex$wHlB^H5(@gORWm=;fD%wt4H1#y$V8U$5C20jLl|#A^d^v72Y$} zzR^|xJBMeOh0<0!EmI1tp&|y~-A$yP3E<|g6MSJDmZkFRfIHxi(&9&7hQ^+bi9vc1Bv!cbS@$9=hv#e7C=alKjU?1S`RBLBm>f`AyZHX z#u1MMKA$c4hyUI0gyK7lf~`|usD{IborVSE>r2L>L2dEZq+zV50(vSbtbesp5(enb zdPCTSts6;#%P+6FliE@4!+tiqSqaa~EK@UyNN9L^(6D)SJtgM4QOpbp^>ZO^1Wva< zNby`k*EEto;+yl-gkP9kzsJ0R?Aj+4j!bUqx`Z<72cuU#iHefxI>!1W-9cMlj>{WY zZc7ScO^Jnn*J+HzThZt;>~I)7=?<$Nex~@b8eLM*gz!nF9G61WYn^DH{2r{HrtYo! z8AjcgY9m#V0d}07g3#Ap3kb@;xzU+|a_L07QG)QI8M%HD`yqxo=Km}E{#nuVXq297 z#Y&oWQZ}Z$UsaBeL6ud6?o6Mf?Ry4L?Lp(Je*pX}$IRX|%_4c->I1OZ`H}(nQ_-%b zTm%c==eu3!ys{K7$Z8o}cr48tLfJU*Tx_gQyNSco(-_tpb;3Z@M$S*%gr0G3t+QL1 zd;2GE=_GNBdl-MG&?@8&?|KUyEAMi!PMlU}U@B?1P)F`Y7Wbevo&WkkJZ z`QJ}bicdQ~iZzE>+Q!+S{#A!&>Jd7Mr`EG3Flbd$$wSDGqy1j~`c9T);2-(ST0;Dw zs%N)NvW$)b;Wmj&{`h7mrvpl6=ll3@B^8bZWFSRdnOC0#?FN+6Po&S4*7d0=@Ebk{ zkRtSbsd`S~jDUaWYBj*z2kCTi;CW5mT8KW*Qd)L8qoI1T=&#fA!!+u|o&kn!=-9lS z`h>9=6+Dbb_uf1~<}}ANx%z!bPSb08^`05Q^UYU4BJH8zeHS;!{r#I!hyOvSAtkV} z!XOIJMC5Z3Zrz#}n2Fc3Mb#dzX|_qf1fR%;>@rr*T;F`#L2DBJ8+u2l7JfOSv){w+Fo7b>;yR3LF=F%DUc8sy85Zpr*R;kU>Sc zg*jFI(8wu$4yut~P5J0e4sIvwIhu4ESfX35>F)}>c4qVnki&@3hyh23OW-=MH!GeO z8Lf0oaEIcuT zJy2V+9Q|-LnR2ZMhj9gdF!5eO>{1K{7+GAuMLyW=sJkBR3p&?Uj7_O}b|7VKD5GtE zV>A8;sZwR(lGoHcPd_{1nwsfeseU3s#3bpR6oanPU$Ou81Cq#|%}j}!i9&i!sjsMZ zYB2(0-wRehS(IAys5X%BqK;3#}hT0Tj#jn|7=e@fiU-`Z; z7=(jW1fuh^iXJ1d-wpfg5Qpn;L#=@&!Iq$q zy^kftCUx(152;=m5_|D3l9H&DHG?3d=cK1e&|m9jns|w%zXKX#77K2E{p&5M@Tj+{ zVgRyXv_eVr2co&pB$&+hXWv&?wBn=2FaW11a9##6nE^co#wNV^*T zKBq1u{oih*<4FNe8>=bPlRdD=E|oE&tpvH_%WyB12gJ7$0>*-|t;F)e-^Td566wcG zEyW*#*4fxcf9bL!Gbp8YQb_pbl;r&@M+PDhGD|E4`t$1F+_l&{#5&DU@)8U!i(S2& zE>hm4Y~sITs&t5S4>6X7+J@H;c97N*dYZU;3jTtsNR4eAwm}p(N4D;_*c?I4E8BxP+YHqk{2Sm zN_u$H{M>P|!GZQb7VJJkL^{|Y((J%K%{&?2)DPLNUs7Cfpv(uo4T2Zz zY|O6G@Whq0RR0w>LGRI=fejL$e}oI^Rha{yo%a^fjL3iol}H@I3d6mQsE$J+BVtJD zr>DksvsC(GK6c5!+2CR_q_sD^nHWD%Y|5Mgnb#v5SQro-Oxw!}h@=w60=qp)x#q6% z{_AuJWCB?ODjuWM-)*veNvGsR3|%z7W#%&B;yeuRDdS~Z*W|WcIxDWDw~!yw<0y-Xo|ye(NySqhM={nm*sIK*83|8&NRd z&ZPK1?N5*^!S%1w%8Itjp8(zve+PF8xLfQDdLn+k_GvRtTx=l|tSaUe#oi#t!434C z+f1E*wqWV$wXnBIIAfo$OD(>fO~!#w;A+4V|ymzjR_x9Xt}kD0$-wHP2mDv zGfQE}M!RyPG~KGDVmKPzJW->PHGjEC$UsY2 zEdH>kifqKua!_OL7BTs!hcp?aPtia9X;+y5O6#gaLW5Ky{P`}&@$sh)l27JDRuPm- zXd3&Ih*1?$Yv@*??s9aG&*D?_DzK?fU5Kf(?uWmfV=&)r>fxi!WFT}c&6VrTFCH!> z{7rrfPHv!8b{N&$GC;@wg*riBln8Hs9SlnG~I=W@n1W+%wXPn)CdSPHh(7#=;x&**KU)8yG z3dOg%Y{*UgCE?x;_6vzEf!;sVWDfJyo?ENKm1CU1s8fSY7zrsj64YHbDHy%S{H=-T z=yH!*O1A3+gq~THMW|L*HL$@mJ`Pn5-ts?CZieR9K2#9>RVV=YPJmg$lp4P*yy`ROn~Js4(PZ`~0+&uX zRqNwUB0`=pUO4T^94h%%?BtT))$+w0V@~?p95T%#U{XTHXGdmLo7Fo|u<90B@h zzgUcB?Gu~ZOx5YQ8~=1kn~+5~{Hg=LV9?+rt^Qr9)yk?$We5BgXn-(hfN970t4$0_ zjl_d)eXf^AMw`%>VJo-xZM621mJ=xXxX^xCL97r<_Fx^&BZ7t4Y^hrIeE|q)h&Lu> zsYFeCQA|a_49d5_B*wO`U@7h99+ohVS52P=0uq&|=++m<|b8*;q6W=A9&k&ti1p!d(|T`B)!QPiSxr znVkak^cNqMnfcobv1aI6C66P#Pth2yy=g4GWU~ADzwthOB_8#P>2#P})$nlF&d2Gl z-!S7~+M^FsJ$=L~3C?#0yFa3OBGwLqKJ=bKGD%l>9?$CrO>SBjYAsbrz$2iul*u#L zSfnCsEL*<%{C$*SV&noiAWeAx@qui7F%2JXRGFM(*aT{p=gInBS?3^Si5C5vMr*d> zibupTTT@ivN1N{O95wc+>VlK2Ec4zs*!Uw>Ug6C8V`Y|h{Z^w>>SG5>r6;)k*0}eH zc9HpulTJ-GRb7(RQy&H+7i~i?`gztx`&vx#3aDYtIzNN|WTLBPx*PR<$?=v%Psam>B!{A_>aJ$Bjbcc3vLD$!5p#;FXoKDF4r*1q zmXS>{ST}K_9rD(z({kxuUKKGTMu?0vp%qhHl3#Xn;KcfvYbv~oo928#f!;Gm28({Z z&w0GJax9wDF^kb*SXJE1!A+(c-$@npq**5Yan4y6?PE?o!)%mj;>l*@d(tOgoHaKNTORTUA7#xw;X=oM}gp2uh^EB`~MlS0nIJ9)f69fFT3k|R6@P&#c@{^RV0<=zip1fbZxA<5-FHLJ#Vr=M^xi|t9S7};6cqw1 zAk=HAFPdTlH&Dj$3i?W@|V|(4My10S{x_%oZ|=4|Do# z@fkcX6Zp05)6slY!@o@II91lyS$i1ik$U@(LS<0rnF8NW0CTWEsD7{C|GbaCe{=BD zi~-Z9(W3Ye6hKJYtr^D>lzC$H;T3Y;cp=dA36Yv=zr%?)7lT8*m^z?(6^H z@|(you`(oWGvGZeqEzVe#q}@v!(4}%`7Gtmn4e2rD1WQxdFukmXr*!7bdietSJGpJ z>-r|}l8gNo4%_KAfte@c-zzVW+N_5s6tTZ(xmk{_u`4S29~R~ii?D#i{-;V>9* zhTrToIxO;63?-9fe9a~$wqy<3Cbj!GtEqjmsZ&BNeh&-|9=F0WTZH|6cKr8%h*ds< zB(z6G^9?p}`ROG~bWJ=f+R}GwyC8XiqqJ*;@9074cdo_)jd|V}Y83Er0Sw#fetj#2 zx=L0Z029jG?xl`NyIiyW*leRf98qqhf8TZ9S@g_f12PH>o>u^H`^)q zhiyQlsQ<(iSn4!pg<}?K6NDa*lb#f00rxx6>)|9W;LnsXg5JehRdAc&S91@R@Jun@ zGMY6X-wiM>mVfK;OUVe;+tp>rxJ!GF2U_r(rBFeP$n-4iqdw3f-`3 z3ZM=D^c)+$TM+jQgQW8mct6`xsOweBBm9mN7zYYM{3t+z6X2V==*|Qld96P=T%sr0o`~n|yF+c0Y>3xhVM;~&?XQyrDGJXqTP$uW2fT?BC#oE8vAdbi{vYdwFp|er^$K8xqB33d9Ugiboq39ft+5SH&^Bx2ceX27G~a`uYIN{8L<%;?8QNYo^T+_2O#_F=CKk#(@$^t8VC**K=vi}6yJzV~K~fZQr@ zG5`m`-BEw+wUaXpFgwYZHw4Fh!}Ca`*DGeI3v&R3=|1rVIF`59YgMRHE;npj*SUq- z1^kNFSYHV1ne(`v7UiDs>em!gT~u-1Z$oIhjM}@rT;#h1Y;d33eK`^xr=JNC#x&I5 zazT|_?heyuRMUg2JBZ<}J1L}*5*KZqlp_@~bJc$n52(c+Tdp4sj^bVznhH0ZGn?SA z{w6Nsp;$S3Y2@D0BPkmfY1DqUZ1$-0aH^fr_?ye@;L3@z#nC%1qv`iGwvplHw=QJ9 z_J9or8oo~z{@~_`ovYroeh7ym)#-f26M_3}@Vxlr9rq_Vd`{pB?h%R)m(QeMC}tqw z+iu}}mg()&vZ9FJdvggNMxBWE-Fi>+(oSD6wTYQlL#*xgc$U-0W@6TR4 zjT*Aje$hVEr`l*xq6+qMyYVJF>73nLOmv&7x_V|H`bD?2VMO?B^Q@7xefVUU;=>L0 zINYc6>tp28u{5zHzjD`g{NF@}vX%MN0s4Z9o;lyxVM!U@n~PTeRof zdiH82{(Wfn&4^id=)0!&%0^Zaf58r*f3SXK0B)LId&*axgd@71ooUFgkM6qDxoqhw zLf|Q@f6vVl9-xm94iT2%z-$)Rw=1u2Vy1`T-6kEY;$00cIUEA?$Pz#vtOa4H(#}CS zI@h&wd`QsN--MzM0SOx%@l0Sv`nmD?SUBrg6vy`n4)G->S7mvJeci|adVpw?^he4I zzxl)Jl$8*^4l-=>E~H?vJX~}A6E?+zy)V1>q<3RtQ9}%x0;G%!V)#u3ajCj7kq55T zAozi((g<{!LHu^jp>I&3kW;CkzT4p&e(Q;sSksmn4$uc%pwVPw%Gn&-iKwt$V4zv& z$oh!5o_J0xhr_Nb#qrQH!Tw#4!lf5?smuTVr2o8-5##lus`!JuO5((b^BXn$(`W^v zBqcXTXOz^jK-X8|Jsif212Wfvae^sQnH2{}UPz=N$xIC@qEtQy+(@h;`H2@>iP5zw z4^zJF!4c8Nmm5)>Ft`5xaBv8~H%us@%LImvV+9dzctSq;zFLs9*&FknLt*Wd5s8Xh zq=FwmS9v$4u0l7as_>Gfx0r#=yqL|3`aukb_8kGMJa2dl-$ZI_4$|gn{MrbRh~(ya z2Gi7IqMcAH%YdN+b19A-G1iN8t{uHlPFAEY-w7f;d+srOM?1BA98*YUvM&Y7@dsp2ccVyo=p$Hf9Fh)>xmzzxF%+^_++y8cQe^(@0H*S7x7<`=W-l^Zz8+B* zh;`Dp$UlFIGCG1i`2}h%>CRxb#<0Kye~rQTY_&17&e!R96zYK_ju(Q!u?{!s2|IBT zcu{}d_$hTWKZ--4UkA>;ontWep}WafF+PU#Gf?Q$D?LpQGv$1BP}WYJh)gWz=Jw{z zYGjJgiI|RwCxfrFgUJ~>NTn{i@l=c|5U#Vrhmk4>6sd*4GX-KZpTDa~{UQ~!U-+50 zG*S0wm@<*4tGO(6XaUi9)Ns7}Wn@LKH|emgOd<_Bo!pP4#lfTwo_1?9h!~46zCem{ z|J8{0;Fx#=wlNg|$ez(opp@b zCT{GSk7PRt23b}pP6@2@huy6oYqrj)E1oFwbST!wqZ8+YwswYS?fV?C?V6d^InqU3 z-i)8~fE9@b-i=42gZ1dAN$uo9Xy&+L!(Osu*v-W@1{5@a|E@`}AE2P_y-axmb0-cLeWtOB6p#o?s8A zUrKg!#qW=wVA4_c|E7A& zR_umJMrTcH5tezv>4NXrnsy5b=-=$7h_ug1w;z^LyN3+D5cYr}ufeGvt1>l$X(f+} zf7G2DxCJ)p8GAu#Q%9(1{OS^7QP16bfj+rXu#Y*aL3?BNaw!~QT4nSDeMg~mk~b7u zQxsYshIXPd1kf1ap{Pc(Uj7$qlF$OJX)xH;_KYU5o7D`pC5l;SeP)TTh`ifCL{3=f{pjT~3FsJchlos1jPD)uV210VrrtX?5dG^3&`gZe zdzDQ{eLK5f5Kbs6Zra;Lkphr;ksDEvh@n2^aWW3qiHqcS-=GjDZ6ZALb-R!w{O^&W z0l|@_rEp6Bi;8WP(-I8HarCyu^vq<8D5*UIF+eC%V0$fz@N7xXT}5OnqM{yJi28FS-SbN}9T`dc76 zv61HRPDiRNY+f^j@~6kW$9nWvNK8X5B zfY-_F=x>G0FE5vVglG!FIXni87<@0^xIP1GW%*Qb^)oP+^326oW2|2$j+molIE%lz z2Wss-$=Ix3Qd$7sFQM%f2I;#88 zLITh>_4tu<^c-(_CTxP$Kp6a}c4mr$_-I?ZA)@L7Xsl&8`TPgGOXkq5M3jXMLpcJO zuWz1n&xyawsAZkw(&he%2VcNM7c2{k3XL{L&FZkG^(60GLbpLy!vb_&wFcKL!vWu%nli{ zw=S73VSbvyzMPg(cg-#sdPZq#^mg}dDG#(TmGxwz}{*Nj@)Izptz9KG|I*_->ZfMa1ef^ zit`{=IW>bKrP!W3cpd2z#(i#ggKtc4<)QP7K0LAcOvQUw98S<@zDln$s|IAMk}Uzf z278WirnepjhpfKL{YQ>?_q4=tR|}(|Vq%YiFyG;Rq3{Ay;LyB%39XqBIbKgW$)S}h z_0wx-3m(cc7@C~9P_<6jEW93cO$CP@GAdHJ$}eX7?cPp0tc4XAHdlTVzy!}nKbQ

zf5XEs(` z5lvao^;i2vhQ^^}$JJtS+B`|l?60eS+&0(GF^R~TJRE^}6y_+15fm-xr9V09 zdTmqMq^uqgDV6oK zc6`d6_XNq)8(C~u2oa0(kh_~5Z5grKY38Hrcv>{Uu^aEEsU-gOcFt&XPhKV_|Bt|R zy!-T&*1^F!WGu8zg6m29RMcVpa(=oC%%4A+DZJF`@+m!A&~NbZj+cEVjL&$ErFpERDh$K*Z1b^1zTLMwLgC6Z^tBUT8^z}quW z25FM|4l**$3}3&@HZ)id*{aK1efvXLc!M=TD(u;dxmOUxoaPuwi*FAZze_^F;GX_S z0|#o4l%PX|ksQMBgjK|2IO%E1O+Rh9=Gc6b_o}2i>0zojj&%3L?F%}TMeyy331PM$ zHOU$rj>D*=W1?edV{LB<^!i46odIavDfK)JQY)&ryb-Ge36p`1vIGv$68+yboJP1gaRW z?+tNv=Nrv;1^chTcVEmMMmi(=Ze9ckvn)&plkwac-;;#g7W?vmlNhC9>jyK72ulhP z=9%<;j~kP>s{{>eDwwrfbIZdI5_){5Z>x9Sh1AmKs z2GdgR=D7^Fh`aTPy3D$X({!S!e}<3HsQ1Ol^(@}dH^of6`5xPT_M6@KWfc4H0ciMz zTTg$=fh=_`!=vOp-uED1_pWT^biQJ6U%oH}X&6~y3Vurb<238zHFVPMlw54w_5J#F zM)x;LJBO<=<_xuN8a)b}WSrVur!hT7#~hZ{z>#>#cgWK%C+BJ=q{3JDAV)t}B$Y z@y-3>_!^m`L8@FRx&Pq@$s2t)xA4IO)cd*Kd0KFWWs9cDSL~BmLVP#z;(vnQ znmWm*VN*10<@C4*+X{vo>=}5*$|T@=QaqUo(xZ7)#b51=er<-8qC=3CDk*>uHF0^O znx)&f*L6-Y6I7cJIz#s$ME!um(4yOSD!7_z7D46F{ApVPpJV-gJy89nivUc*q%_)! zfXZ1!P_TK`cu~h#$(aY_8wtNL&n3KXP_lWKnyo;xF?h1kGgi)-SHZO?x5V#B6EJQr zm{}tI<_B#~+16=1?4U2mEl`!{NyNmb6wPzGjcEe0i@M-=d`5NGGk2?eIz7i`dfCs( zc%*b(m<*e+?!m$L?GGpc@9DRq~t?PJ@M!FAy+cb(a6l`IsbS4U=IO9#@lwjCUh7DdX z4+L2XGLjclJdZLsiy0CyIW_BRelaczr_LQQR}0y@NspoO{Z0{9o0S%M{ff>(ux#O1 za#``-6!A-ro_BV>^)MsNjq;Hh%2PLQkS@lAJql$O zW0^N2{q=M&Iv_n0l%^CN9GL=6(CZu~{#fI5J{S91Er(PveY0jPZ%<0ZSlz({Ut0Kc zy2gXL&PQ?tnU9gO`f*e;p{38vM*}RD_;pocd4&t9aHjWLKTMpK8o^zyai5-Uno-Oq zwR_EKKpC}Arwy3OLKFO@vV538|9f)#6**#=?#cJTeTTHMqBl`uT83<-Qdf# z+)_FWjdBu8(jT2a6EKV=oAL`(}{KQ);?%d8H;M%Ic`*BNdMSnv&%)P zI!4n}?y0gOCHgHVS1;{2f-+OM9P$gMn%^{stdtYawDT42;av_avMk%kn<2&7O9y?n z;mBe7BE}gtA~*ca>{iUh*4!P8V4rU$=|pO79E>#%DLc4df^uD`$W4PG&(In5b`x>X zRK*K)#%u=n5;j5T#G(CX@rry3uwCbl7oRP@;bJMoGKAP)ToOP31$I4s*K*}<)nxsV zM3zWkhj#G!XW;<<(qdt~YjYQ9w+@2jts3jlpgi?EK=4vstaAoIZmbyz8RGrEkX`rU zbXrHDAWeRp3>75IV}*{5UO-GP%{+_ZG}$BQRdLFRo7Q8K=K2hZ7i@^;b845IxfN+d z_Jx_>tD*mUHM*E^wK;yw>%drcrGpoGL*b!@X71jFQZ$bUA~z>I6DvKTjZ>t8=ZBFkrBTQi z>#!2G!-D<%n}wk-`}mLkF32|mQI(M6^2BUF^*K9F&jGZ0a1(US3iM{<4IdGM@|omg zYy_%L&Q*ddC-LD0+4tnO*}v_X>OUl@`C&|$i|FVJoY}BMVXHxJQ;xbPD@7+fI&a-C zNpB|j@H>2c3Tvs#_8XkbZqP6rpL?`bp0ecZiATKSkqhER8fovkAo4>~C`YB1!2dqA zr6C|Tza+_p5fkaESS5JF+a-bqX+bE!-E);J3HITFQNj}5&@a;5@^Ux7Nqx%PmWZ61 zY7y2hey?n#w?t2k(ydh5$KJqHnB=27zuy(R84MJZyUU`0^o>*ek(TCZUIL`zA0 zA*U2ew^RGj1p+2JM1O5Y-X36aSX$u{5p?vVu139tTGZY2s#Lv1wFawDztLP& znKeER;bHkch@Yw?+KWsZn=F_>jrz-`5#BoC_i?V?NJ?Jc$dtZjnfOF1(B23%OG5K8 zZBy6emz0eb7g)Bs4^F)M2&!JLna5O9iqs)X;Dgh|Hyj+XunC;{Ath{n&U5*QYV&Dpeu1Ou-E z>=968RkPjWcig1VD?D)plp^iuq(IW5r>nSTc4=ujsEZp6)~$ep<;39-Swo7O+P4NQ zINtUM$g^CX27dip7;o{ zsUD+#{Ub4?YnZNP@0Qs3*GCmeO1Xs7gUH*;NT0vmJVD>_*V1#kdq6b%j+{vbt`0sE zI9Vpr6ze2NywX;k?*XYd|AIMwa#dzM-v&7_LBl4bf^8tmj%yUMoIDt3Bu@E1V1o&o6C<&vQz@^!rQjN@`G027*e z!EWIt<^*;?y#Xsq3N_Q^@QfBns3Gs_b=A^EsY62lc9TfhpoR2LrC2Tx6>8O^P2dFQ z9bu?pQ_5yd87Ci?e3*lY&qemvQhCg%0c;@_?&m*D0x%g?V6MG%@|t7Tc!Y6;znMg3 z+LFh5HPki>sOZ`O0kB4Zq$1$G(cn2#7Fd?n`@=!}(bsWUP4okt1O93la-8B4K!BZ; z+0fCaiKnSBuQsCbMj$+N?@Dh16Ui>m;=k_O%qIlxJN>hi;o8o9#4l9_Gqw@aQrM!> zISCUk41QO=*1K9~0J6d+;d<%kFoaj+NFDD2yt1OI8_h(sJ^=#3`+SisHv|}OdRwRY zSdIbSwzyi<`&{P_d;CZ0<#P5%_Xj9;U);VCad>+=`>GjbC96QQHV}Ap{Rh z42R*$)15wc<#34GOL~$&v*MUM7HxKS89YNvZFf7 z>L1tRX!>L!{fK>f4gaEEzM#^!Fu+g)dao?7>;3gfuU(A=j9DAwQ@$B4A#GPFV1*nU z0bWB1zRq+T6Rg%#c6B(c%S&4lt8kD&6x+<4@)aVh!a98aFYznl)6cA(gkpadN!<7Y zs{6h?>=k!JqzSs95D#~n;_W%_LM|zq`_-~1eyTBK`VT3T&$2~#*@g1;3D&E?b6|vC z2M~9EC^wOj3m~L1&FE2zdJ!v!_FHfw?RoUfd^uxFb3?T~qU47Q>er2#q5mTvmCZL_^74+kAK> zn%HELj3FS`gSZ_BKw>Q$w#7oj0En2z|B#s3a-;9qOo|Mlh<*UoM?4;Y>p;yzoE>^T z$e1$`2?LZG4}gC#```=yPD`HnFd%5l!pk!8ly>e`5e#Qg$xJPiFxl(+^!!Y&2$+ey zbW}lzU=&g-{qZzR-T^2O4KV^MuVCxXE&!usgp(8vj1kELmS&}Hub-GW-G~M$K*&qW z7WFxuJkC?u{&-zh`wxkE#joau9kiVN`z_b$r|2`^o{wS)A2{ z?N_z!R3!+v^2mhkVz7G9G6}_ z%W`(eX&raKaHyQ-pJSPj*$s&_Qvpa$Q*P#ZbJXFzamj3mCh4lH^(JzsxiOP;s!HgM zZfcol(yC#vSgP;tvgfJ=0`Bzw6nYq(j}-l2zLLrXj07`Zk4?C-W2$YU#)gTi?Io)TRH zB3lxgUJ@#;-!;P==Upa;h6)8+0QH#laU)XN!Lj4v1DppX0{0E6%&vboFE8MU6Z;g0 za^k!)5kqe_wk4^;18ZOBR_(P;m1_#TR;j7;Rt)zBYz#^>dQSl*Gaud>LE#;SiN4s3 zNOLr9fPd{g%EaFe)R_*1ES6OFFj+1TyFAo{_rj`t2XU<#NsX_TNgIs@D@F9^;<&OP0P zQ%FmB0z`oyU=k%5{*)OpdC;^OYYt;p8Ns1YCFmAV6*)o1u0EXkif>jB(RzIFg>O#h zlc5d-7#)4eL_tVUX-`dM5DCga)SqD)r<8b_OXvcgzz=#Z=aQn>k8HSvz-V+3)p+8> zqtpx2^hP~eA>Z3ED7`%UO)beju!uiALzTklO1cSUVj?fB zD^TRRjQ{EP^1tE~ZN|`eu1wcGZ_P5i#mw?|?*1pyRdkI}kc7q}waTnk>}ypFEz<_$ z+sSts%_}@O8%|&-ToJGxvw;sL4W?rQ^ zc)6DU4Zw&mE_Sq%iR%l$waSGMmnKI3@FjxQMgBzp@u}!CY!L5N_03h__{&HBV-uwc zp8s~OOKjpEQ*}Yz${fBT(bM2KQr;v%h&xhG8>Lya4riH?D0vHy2AEUt`{NpsUp29B zaZ+O+;>pI4&6a9cPT8Y)Y-%N|V$~OhQS*Y!z6GUz3EOkUT~l%tebnS2Y{>rm$a5vw zpJx7K1R0}DJEzDE#1Dq;6-s_#aGQJfUcs+~5Fip}z$w+K&x8mL{}Jc_hz|ORHgiM4 zTVHPc!b$Zlt$U7qq$AEzM3!|4k`#MtM2c>Tmoxz=mETyF)0WTSC$HtWWj2=3^;>+n zx*iLuDK9TwI})YSzRu*v-`>#w_BLoC#jr|9mC(dm4kgEC_6n1T_8QT=G@(9=4%t)G z9mD&KRv6krt;+Kv>YEJ>sXD3+JPc@D2{(ASg?dn}{ z3W5^mekBqh?RyDQ!#3s4>rhgv{ldZpdX%Ar3lNaw@A(W?vslCpI&6N6Uvv7XhL3l@ zO}Kx#IQ017RKqm=kL~+&Uw}R-Mo2=6>(hk%{jdvK0zq=G*&CdY(B9W_Arng1UNBLA0SEL*CWmOBI{cs=-V8H| z4EWEJF@H*C78tfB6aA&vT`r{QnVnQ{PqI?Dh>N@{7!ZZWc`n)Lkb5R@gV4dG>Qn3N z0pLh;p=K|g>>ZU_P3ZQ2Zk{Jn47UV0U;ds()E%#Yi#g4U{H%B@(j^`;Ws1rzrJu;1 zz^C{8>9p!;#q$|BZzv@$jg#?wp2Jq+tKo=bsS+&!-c3(hyU(;^8p|VQ5u{Xv_6VV1!o1cepGYBL7Kn!&ki-dfqALjkH|Ly{~K=d>dB~N%Mo|sY$epK)!H>E05 zkl%9q(amtPH6pHBGZMWQjV^UFsCjcVnfaf`;)E08wM5RHxLtzsF(j_G+ zjgr#c4N@YZG$;s2w@3>DlA<6bEv2N=Wg?-J@XckM_o(OIpWlyr-+!L7VYBzM)_R^X z#~fo!9=Z5R#x|iihBO!A_@K0hd8&u`VS1Mn^BQp<6b`dpsr&bS|NYAo%QheBE02~D z95UxCjlwqHK672+_L)<>%bl&oy51ihOdWlfHktT2c(cBx{$|$vUedq42)})3BAtAO zDUZ;c&eg|&uqL|aP)Ggr-8kWj{1aUcAtN3>xSqD!e|_jedHB#fW??oLeT7i^35?Tj z_L=1eUv)P`8bbd1lp~VOM^xCfYb2FBlW3_G@W=BKb>DqmJ>REoaUfiEcgF4Sw-11~ zPw8_HvyGcfxTu^bQ+y`@@rnJh>ONTH*TvGS$??)L2 zJvB5|(a{c(m5PHz8^Rn~QE3iRGNEV@qYs0<`?4?G#1meKogg;ioP2euv%^KXKK+OW zxC=#n+`u=}eZS?E#=p@Y#Dn;wKZPKR+cf8E>|PdMsj`|{vVJ^V?ijcPP9L5j= z`s7lYvGl?vx@P@ zzc)qoZ1Lk*_yfVN%F2B4j~hJ7OzVc%W?S8@qwFK93|yWa7tZD2W^U&xWNf^DlRzwF zt~f!)?F0D9iOTCm;~!`Ikv&HNliA;Z=jyO+IuDH+Wwt zQOCib^)kCmTR9QMG6Qp({mpx=PheK+nV?#&7?xkEbmsyBF^= ztN~B4?1T}|wAeqR_uDeTuJcwed+i9Wc_9eBW7WQOCN7Jug`Ya}6qMee`gU08yu%$$ z=*jIc_HYu)Ctwg5Jb69e>&~&WFArX7yvW5CS;js8JjQzD<@@ofR+RfvZ$0n>U{_s( zEQl~HXnKix?3=a`r48?eq6qM@Qg-&b&z7r;-^}bJBCY%YMT81tIBtps;}YK`IW&*t zISi{3t`Z@!$_H2L|GdYy&Zv6_hVB_<-p^3u=t^9!%9Z*`ff6*j_MtjH!)7HAO6_Hj zwkaw>(j2Wc@^WB&G0Yz_uONGjA3)mLO{@8Hfha#RJ+w^RW<9H<# zW{2RZW08i}AQ>V6k#`?EiDzN72ny!oT8t(ciG zs#}TPJQ^iq?boL3L!)T_1F`(Q>^qyk5 zPO;dwmg`qJXd6<-4djP$;Sd@&dov3&3I`=L1i@6I@9M!miI@`FY%o4^(H1Kuaqc*D zI`9XZ3qT%Ha!PF;#oQ+NAi4Vb5UV4SOtSejf<;)!{H2j;dZB*C`^RfrlKg?a1-l z-+={`vG)-eMDod2;&=$@S=@(J1i8y&2UB=N?6kFh+azspkPMyy(*%ubIc2)4S9$Sx6XP34=#5A*NZ8rX&H{;+v-`~)w)dk zP^BOZS&v@a-sq;)m-lp0gs63?K-MxtaOL_ee91c4WR~Izwlad<1tFM3J(hE*@Q6ph>5kJC!Q3kM0)m`+P4&V>#8+kMto29PppetGnTP@}7K*`0mO*+yNN^ zqQ8DVTT&n>7uQu$&U%c(Swm#RqU3@B82YlBxDU4gRH$W$@N>RM1aB!-C_Y{P1fZ%- z$cj;T0T%tN4A%~d)gHrtB?la5Jm9zKjU>4UBEyh(H5 zCQcVqbUdUYWHoqw61VD2#w0=!=&w#U1@K5KHUJz|NfWSSxJn<;t@oxsT{u2>DCx1x zZ%^rWihzYRsO1=+AjIK;|4-0nAlq(Ix;7cScr#^>KNbrt3rso?nTd=VdeMBQtm&@k zVhE#-Gk;J*jDZG`XKMibvZQaS&m*mxGdD|qI~Cz?71{^2gb8u?D2oqLP)gQ}4WeFr zVXy2wo#WLDXJIk~B0jP;ZM}0wUL4>GL*K2hg_rJK#!Bb4?uU#z&#jj_5`ZtNtojto z!2a$PvB#y;52=AfiA2J1HDXTS)A~V-9kY9C8ROB8XM-=#>PP9e$2K_F8k~|W z38KW7)7|loCIuZkNm#@{+PcKh3jA*8B!vCkX3*4hXbV5Z&MYZSy$n1kP?#Eu!cQNN z9}OKHOpNAEeYiGr6Td9EV}juKZ7zt1gvBqM2v@cLe7v%JR5bSd#AX|OKY7NG1D=ux zHF`%c!~;f3!@x5y*L$>GZHUmR*HVOB=K!tv_wkQzz8{aTg*Mmyq(0Ly8aZF+=zyTp z$5+MRvI1J0O?dkn*0aWkpxx!jzRzF4GOFv^PGatn51C60lC8<8hM|41q~c8Skhn%4 zEM$fWJ%4JTWrT)qR(zVTsomzHdcZy2#j6Alej_?r;Jk(YsZ7!)q-pSQMyD-YeB?0h z7DnS~a(Nb7d2!Da7c3etchV722xf7c%SHFcGW1{Iz4qmOjpHN|%I8{wMoRH(IHj_ zISPn57*N@xXD6uq;P(D(tD=n42zOH^R7((Rq6AqW$-&-W5Yw~0f0u`JuBDsQMRBHE z{ArEqZ;vjE2J+N?n5?CPk{BMoD~>%~q*lv)8ScQs=gs5zG+cdG1N|3vh~{EAgzj4y zvhBdVs|#=m*~^f7iXBqoKCGHa`$8zBF5;Lc@T-4c*^^8W_LNtdQ(AqgPjDJ11z&7) z05Dp_Nj5>}H|b7%$4=mw2){Ti8>AM@VL4>`!QTD*acd{B>5hMZefBc|=LQHC0IpW+ z!&0zYztDU%0a1}a6$LSi{871u!40JGxTx#%uJqr(4*0+e4Lr+NpWmig)o5YojK=B~ z;3`}@qVihnbsP!HdEgf;y*f)D{(dDSkOErKBW`Aw+wI(c?9r?kG0bp$ofT;RbzWdSaA|jl&$b#y9=}L@FSZ`{2O>A_oVwEX?GNEBQrh+T2EJi ze%y|@X^i)zOz#HtT%u3Ur7-6+Wq9(WyTb6*l|Mds9D1=7Nyw9x*vc+cTCi=P0#QPQ zU;~ay-G5T4V9o8|kKb!c-O zbGZAV#%_-Tyui?kq}i^p4S|$HOiK@LtocxM4=ABq)1M(puh^KMktE_eWjH*&AA_(e z3j(hu9cvB4@Myeyg4^=uYf#;Wf#kRq5I2x{G0MAVdLhouIe+Vqb_R=aPzwmq0>(Xt zRiu01u(scQz(|789m~Kz4mX^+ND8xg>QR-~kG=4#c6h|NVs0o}?Lm5xj=p0x$kuj1 z(jwCvG9DmJi*}abSvo(vnp1Mz0kE_f;}EM_Qs@V0YHq)+?e69Ey_%`xQuHWvOPi>{ z6gP?NSH{1sV<8vvNua&2M=)qYt>LG1BSOL~iNd!!_%=ruj_?q=B}N~CMb3-wa+2OI zf&%i4h#rb4;tt7>3MF=H77c?m?mG|CW5=!CtYyxA4FOEpvxbFR@kzO!PW@f^Z>Ozh zoFsoxv}hXT+&Sg>mNvdd?=^vdQJ`l<1=YIb)I$%>5w{54ONcPp-Fo+Yg%K5wnFfhd zC!^io4&7H$#H5%e98cl3R@Idib~q1g>e?5e)>8QGUx8gZ)m55IJNn3l6t2j=73Fqn z4XWSk9L-Rq#JN22g63R?O)^{ZqDN3Y-~|nyg~P;boE!?p^8lALY9@T&J@Ba;I!FdeKzzK|GDGB{0ofVq+xKm*fnHjzrfYc&NaQV?sMSb(Oo;djI{L%XGggvIM9?WVZTx&`gH(MQ~h zgFnDbJITy5MM(P1dyYx8G-MpC)3F`}XtyKRu<*AF`{Rqo#=vkBzl7dz65P#Rvk8g4)M$Hzj@k~okz>z8<9{^JXV!AoxmMsxq8 zk>H1pGq7;ag7)u^k&Sm8XJCY(O!hBe7KpY3c&yYBfk4S15n?V$jj%e>zk5|!3ujCMQFKfjf80CBe_myQ1VB7jj^CtlKi<7l z9(rfuic5c__#;*KTXlGeW=;RyCUj>=IAkTi;CO;0KBrpiG-0czm2mFvoph?^9Ejf!TlDL4{8wqb7*KI*o4G zLXYz8Q>S#s{>aY@0}ntVn&|ekTqPpd8ErbyN~7#>L;xV{#vQfXgO6;7t@?^{HA|Ti zi4&`W7xS$9Qnx^~=v4huo3*D^|Gz3Xm_r~cIb7>JO^_t~Wz5{UVHGvUYMt(S;%e@E zF8~~4$wdfa<6#LL+sty4ShByNNUT9ZS z3@^{2DaR-51uyU!LDx^Ik>uYgs>s2YU5RujxxynjnR~H%G4j0wdiuDq=O_0+EcIt7 zO{YwM`{)H?ml_WJ8XP9%^nsu&()bDX+9@}joZK+%GdZ;Ie?hW)_l`O=QTiZ??gIIJ zpN*x3LkL(Tua-$lfk+lt>o;YFk9BV$k@LACULPBwg|VKr=-l}E1O<5s$wq883FOnK za+{wBd66dg2~wW2%L(?b!;ZwCnfRd(4B-J(6v#W}1TlG<}1+k6cdlR5qsjVw(e__crleh9zY7 zK!->*1m|m&Xak8ww3MHtIJHK^r3X_eSc7g$lp>=VkXBy^sa`m_#3YL;DENZH zb$X2vp0}KLvlaSTM0cY;H*!FIs?)teKDf;SJeV0pf~smnx_zg z#g*c7N0xL4CN4Dfi@Zzs7*aqtKvqaswF7-e`do4b1e4J)Z6FB1qd|Bk@EFU71P#@k zLStwFBJpQB{m>-rYeBj^DbpUUSQgQP!7Iu&nf$o=reFWy4hP22vFPTA^Q44O#oW>^ z7l$7Q`?E06wD57Ob)e5jlk>HC=W?BQ=D}A=W?BHo1cq!Z0GlK^#aMR#yz<3!51Y{d zG8Wk55aIC7MU-*w-&ewJO+W_JBF`u~;~4cv*Ql^a((xFd@RuYt1C&$0 zd0H}IvMNik>Rd}K{lEr1u|rJOKzo}B5w0dP$QZRut_~OQRz@(p52slHwy{#AeATh) z-xVL1QgN{&Jr#wFydf5qv1kbeOecKip@`>uZy>FjfHQ>7?b+esbYQJa%~DfIoeo3@ z!?{z{pAM#oxYg+Xev${Fq1*Adv3@qWslQ}9hh3eIu_{#Isq)LI=IA@kv0EAXt$D8k5ZIxW(`tW*3~8YV?kNe z!^CWoLs1`5S?GpTpdv_@*q_B$UR{Td0wC5a-s7>K6?+MK&!i-OT7rV~4dlg}*~%z{ z$8eRd!V$=1)8a;3BD0LK+rO55-~6_4sAH=Em#A~vxM^~5@^goB_-Z&eu|fzuF<*@3 zyDZ(t#O_iK4(EH84X%TDW;vB~p*yvpMsfA-5V%B}FAeI2UZGOZMqIz5^nzyTv1tRf zoZcxNEPQ{gFkA( z)k)Gui{51JU}thEEf-9uLdjFd!-u)w-oi7GWgq@+90}klHp#$oy=1ZwC+bc593`ys z=v!p#MOvK;pL$6!19Kztn~~L3plgr}JfhA4zzZF?N}tKqQ^9E{@MJ z>d)=}JIx80xc^n$f4qAmf{EKLP=9~U?2qu$NX7%E|G>mPn1G3+x|II|0KJBniq7jR z{TC*N!HDr>@hSg-i5rLj6MNMp{z1?pM9~5hVB&Kq-rv9T_rH9V88Goj2aZ3;0R)40 zDgp2{Qnvb2dxBtMRe0(3xL)PoD;1jrl}clB@Bak5|HDd-WWY-eRn-=MuN3_`s8oEo zQxyKb0Sgfl`Kq;j*MF>(5mLmM@x=fAF@Tr;zj?tPv6u}?LjzFEm+v^V<8C0cl};(u zu>Ac^vu_-aB+cI0o{s^@mJfh9a0~WGz;WVvXUF>U|Kkz?($fpHf^Pq2sxaib7m?mc zg@bnDh{~ngvA|Fpr(9Y$wK)Hyuu@ddUIm39Rtc$T68K_1 z0y+26Np(9klMkAFrXSPMwGmSdScmGbzI)Z?tagQ+yJ{xr)jmFvm zMUBH>21Fw9^&?E7ICVT9`ba+O3!uUyu27Ep9TkrX++TVkq!L&I4a6 z0pPv0kK;_H>iG|RQoMZ8D_vSTh9oWAQ{SM^d&{E#RE8;$2tC0a%u|Vy(hjVQE6Hnc z^B@W9a0D_wOIqqnn})*dy%VSl*%9rK@%1e&H*p|0fKMEQrX#!!F1TIdW5de$kXF*o zf=*$VnP&ARIN_w^sYomHK)?o+@x5bsB``;<3K9806B=;w=^bgS!^G+%!#YGw^D=#7 zSlX_yD6+f4=cRtXuR&jEqrH|{SAay*V?GTgBzKol4xThIVwBRM)aHJ=3I}M(mlMe> zUBLUabFp+QXh-bWu*AL*EGcVXyOT6vT>uU6O>8?vDmTZ!Z%K4Q?(^vjHR_!(N6Qbo z$%pfFhq}O^hQCH&43rJH6yg2qONb~+bqhQ>mzNNc*gD86gRwT>jmB_Fz?`}>N`6{_ zN1mr@6ND=`E9sKciX}dM#q{ScG6MRTqm;!$z#v;>B}L!OKg1sADO0@Sz23zhQd=n%|c|I4x@5vfDqrkBTw7)yaXnfObLJ-IPEDfX!Qt(pFPADX$p2k zW($tZGGz^6?6x`Jgy!nL;hD#2y*KB!#EiEZLT0qR`>x!&B^9O)ri;-6AHf|HfkiOW zSd2w?An>XS83Cj5?-4Mt3u;D72Eh~EG$4<%d`iAs%q1SK6Wx!FYIvb}{v<9hI#EKYlKE^D&7JlIHRd<~M?raLo0P5?R+I2%Nh?^6cElxP`r7I9%v?Zd|U#HO>5= z_A>SYI$T3T^3QGl^@~d%pn#So2wOGhNz5|nOy}S1GcCEFs24eN>Aj`2bT|<=VPuh= zx0z=0HKF>PRV)H%Ok)X!3;aGlvQRV#?b^VR3pPka;SUC5j5`F%rrqR2N5j!yUnU=K zV>?Z65k>UOg2zP&SE?$78!7erF=O;ZQX zhCP5O2rq80m`O9I*5&*M;>%V&p30M^e94nwcQ9C$PxRVj7&*~3bMcbrC9)7CSwI94 ziGv|plWbAn>dYy6-%edovw&N9PXzu@g43%V^?fvOF4&35gWHsmpgfw2RJ@J#zK|R* zcLHnhwH0yckuyxvtBs2cDmZs@kKqj;<5i6`Rl#|{mhtw=AK)GiO_=j3`|iRq>U$~4 zs8DqCDDo-I%7orzcPNEosD&%xVU2SH`V1l1oDWt4@`8S|b7!V|S zZGaeDWQVsrVz3gz0Voj)6KG5S_ERkUK@j~AqDE+7hg&25=SBvcZVHg8ti#8wtN%i& z*=~SDHTx?b{k8wGsOJMx>fUU=;`sZf4u86k8lcx*cXr3WCWT@p6v0R25p&Vj|Lr3p zB?BA6r7lt{>HpqAKqxy)Hb9%o_~|eIg0?~3!1ihgKOai`o7(%|uk{av{{Q9G3L+g( zK^=Y4}pHI^k}Nn>xP$GPdR??(S0vu)sMi zLMK&@(H-&nz1#NvMkLSvO#S26k&@`2e^KhfJ^ugjD;60yj#!<7 zL}}D(Q+$Mb=O*92=hvgkFWQc8JMlw+q)vu5&MElT%s8>==cmI;C~JO47qizRX56&C z(^k)s1}cLqJxf|wGp2Q{oQQS@_q8_$t}oSkZQd(VB{Ql&4E))hd-J~cNM^uY7| zB##yMKhw)J&k~gLe~%9MdA+jDj(du&ZjFlXgw!c-koGNmMHRiJ)Yv+V7(F!OiwHSX zsvO_vFy1v4IHciy_>$yZ&9`YijbAs4q$+%QVIPBEANH2yOlE;Ro=<>s)z0i)<23!3 z=avG*Y>G)ilBjhov$ARF>pjep@_iGOdB$y;F_CLdqyKDahft$$NMQeZ?*XCXCjz)a zLj^Mg%g0NF6@tcYhRP_v_-4$LB6tU-%zLq`t@`+Ut3k7w``C z^=tBIg+dj|J*}@~*ZE_E`x3Q)``PoZ-e}02P2|HcH2GzA(oBp}LH!&XBOuCJ?j?#W!SdMR^Ja50RN!vT(@g>~~ z=Aw<}dG+MK*s@b4OYw0p$sBDFj1Yo&NPDLP6iYxtMU#B!Gk5Z!P zBW;mWWny2(&UoE3_-?#WKZ@3L`%YP4;u!j`ohaKRNc;5!N8@C76m`9SQiitOoBOZ# z{rTL}USvL2eMewMKSFqr))eNWlD0Oz-(@56!m)BR!ROO;iRB!+<&R>IyB4Xg?(fcg zv@RbnlN+SKx(-bR4K_rlS-{SkR0|`?F7k)~p%2Bl=gL4%{?;MFj!W#j%<=HQ=8``^ z=8DD2^Rxng%4gaVybwpIK<7DT3M$%(XD8|Qf%{Cnc(*hZ*<&O#utps5Dg6afxDU0a z0u}HZ$5*3FJ8~h(%2B_@u-+@4ge+xuRjumHcK&L3 zg@m%bK#U~LrPQ@)XVcVTFD)9o;@uJS!mIf&^sU<(t?*SGd*Z47;JjiCyx zyhSB=yK@BE9;Axf@EQeMHRPb*Uob}cKB)-?o!Ji;CLL_Cvd-7qtuMikeSvyrf$}PZ z3BZkEf$4dPLe!g!S*|9u?z_M!dma5PngU!cSJ)&Ld{Le_^llK&AJW{2T@3I2)~u?oQH-g7RC0R z_i(ev$Ra~np=BBn{W_1B&|mf4%aT!0_U|I)I&{7mT9J+`ZT2dRkKrB3ONf0jIi_I? zs_8sgk0H~uxdOsye4sAn|KHlq4 z3p?#G%jox&&#e})n&{*zt8UFT~*8FsS;0F2Ma1*9T2cj9$9{Ctr!~!?`I80rtvj!YtSuhwKdd!~QX$MJ4?6hNT9=xSYK?`9pMFq(aBF9haCO&}pr7))6j9yo3 zTFWUgYmA<@owzmnuylmjvw{>;kMsTyeG7py$dcf@3nRtfM^ zWc`A~+dvK7Zh6^2RV_Bdb0VTsb-j2M86Ljk4&`u%11lyP5%@Htu_^A$>INy5*X2#L zA&glbtHqgDykP3E6YS!Z#t9yz30=XQFqIbB8t6@sT4|S7GWQ6rh7>asx#R?gS&nB( zFeu}!oVD)76m)Hv)bkiPEz#>lsNkg(4Q>(X6K~~a*u@HzzGD^27)kD9eESy(cMI5o zMXqj?DSUiPvb~TtNnG@Db4li=@_C zgZ)lNvg>9ks;*o6M!Z_r?GC1oMfqPJ(s&vA@+opGk{`unxJfRQz3~a!tmH~75bJXN z?3_JNTM_rdI)P|;-j;rx@v<4GjNNUKw=6&*63^mfhFypDA3}FFGHtp*Z=FD{tA@-f zETU=X65bOxh=7ZSKrhZOMqLonvtU(7rndA31}l+Uk*sq~v5u#qoH~1-DxZ$`8$6FH z>r0-At`FH(_{wdrKG5kbzbvB<$o26RBm+ zQM_qPykbT1YAAjMA>7j1&`(~64s(ujK6NCvPow78Jcv5_FWnB^4!hY;UXeWhUBo51 zh)L`oPITrzre8w)5k0o9rXtPc0Ll`v!n+e?c2rI$ejX+f&Au6FqWoV*1C>|!TYDQ- zmbH&tZkOB1V!Y~AYaA_kee%iV8xfl?2hyn0tyg(?57G|RR%1VmTO8Yy;3vu9ronQQW8nJy#Q-YW%WLRv*dY>Ah+}_!ZzOnc+v^$ z>0x5m8t7SXuU?YE1f!o~o=+hG^yV)UWH7Wnd?CJPF?BpB8c~ykE0eFt&`wCXEI`}8 zU=pJ|`d<1RMyWa~{sa#_%WX`>@l#}OtWA&FT@)1iw-b-3NAICuF0KH4^le$1xd9?EAbelhubZC1z_l)IHZ zOGbMzIG45IHkebanQXUQ=5oL0tXGNQ&q$klgt|`+<26X_8%%n3Ze>mz=i4c!>|jdq zGKg!ClokPlKEbzBq_NTD^zy^Q43VbI)k1|~>%()A(gl&mds+*!o#>B53t22~1(92Q zKGSiEbm%6kmoEJ>HE-rX`x$n|$_*Gr%gQ9JGL7VyA%v53WQ%gM>j+5@W?AZ+tV3kP z=2WL=i6BAdcgivn_&BV}$0yqjS<7^cRh2kC_Hv zT8Sm(AXOb6gzuBU9S}})2GnB1;pZnAU*sy|`bXBs%#k4a|UTW`UxS$R}J+>n2 z1jJGHJBD2OD&LC&K7pXQ&1{CDXU?oX;;$cigkRfSc_xv&xf^llmp}3VV9t>8^iSVL zbGE9=i3bCl5!@=Y2lZX=h?c$vr%YeZXJgSI$w{H}*l|=?xzakFb<)%SiFMVvY4SHE z?vEGAR-caD*uT7WI4}INtE2EJg|tr8!{cfiSeUeOUDDYUTk-4>?>IZ?Lb&-2x{wUF zE}@G&)(5NPRTyc})WSTBg_0f2)a^Lqp3g!^zXd`USp#plobgM(DVrHqw8nBrAH^r> z=hXSHXVIdUcqg%?{Yecy29p_5OuWfr=2SUNsnypV6@Awiy3nIrq;9AeWO(XbWPUtL zkS#nLOw?k@Fx=Nh7#`5hcgxKE#yR!~vRKzp8Yx~@c{Yb+Cz5#si;D-%6el0Ii=}=q zzHLO2jES{#oUhGxmWw@RRz53v4J=4_uU?(M9BQYrvxY6_8`@w`7R11LrbXu4>fVFJ zIJ-%$p7qy1blX9SBx8@Z8ZD=@wwL1LX$^|L#qx^;AMiBE8u;IuXfsjCEm0oMLMiXr z&UNCo%0C$Knr-*x7$}zIwX##N=){p@ys^yJeibIz-Cni0QNh(U>A#!9^Mjt&CT(Lo`c>jHdD$6066LZrGBw_P&rT$`1Ko}{{(Fu~ z0j3H#u?*>rZTHt%ahF~4*dva2$b7`%9w6FC`>|5#Xnm1G4*m%I zSba_1>eL6ek5AcEew0yI1j%APAJ?d!KWY3XnDxz3c`s+!>+ow~{Nd)y+qM4FmhXFw z_u2+JVbdRs$^TZkitR5{ab?(Q5=$$9FZsP=aGwbv^ zw|Cwpe_`4gu*}?~Ef~4=_;bcULligBD8?xj$hzF^5!E*LO$=tDww*xXPRFcCq-Rjv+QYI7t(S` z&nGBZFyTCLw%ta_r0rJSWtMnh-=XwkQfimxqr#qgAK1(!UoLTfH(f{`#~=ph;PmLi zOGyDTk-7Zdow!Er&Kj)@&VmU|@kvhIyZe2N`Zb%2`zvit}i zOT||#B96dc(dWK3w$chEkw3#c2k(2vI|>?Fs2iBVS#VZ!xs-xq%oH;Klkm7)!wUkM zs2+1OP{QHbT9ujdX%I$+s%!1;we{{WgBXsAXTxS=LA`WeiL2my{pW+ZzgvEZ`?+##~xI(ooY!lXd+LSoOEiwH2mC9I#}0+ zUlT3u^M%#n(2y(<5EMnW@;MKc9>14@(OJ4b5qwY{{mpa7(|q^QPDQ};{JtfQ&CkJQ z#%dpSW@SW%`RqG9n^YAwwt5SIlOeohE8;crXgmvN&o%?~t@4hWmb4Ik>0%N|S%8`5 zX?8|5!h39+@3X6L1mAkd;n4_owx03VC5Oq*nHg=2tR5>EXFS5mT&_Oj$Gq<1CNU>M zN=O8-5+&CS0`wE63jSxS93$=K^-Hvse-u&+(XQ660r?DN<&JsaUUVTa;K}_J;IoWa&y#QFK*9 zqRToT3!0cgpIb%uRd($7oedzqRLP0u4p}{)>2tbb`Y+bkkAH{9-KyC_Vm8}a42#Bn zlouWfOZz9Eyp4Id!<@cMRw;ZeVLK?idE^a*k$tyq@ z)!EF_5H~=Um&kJsA{pP&qb0@gS{swn-btOxaB| zFa@Ml+l@bBGPGTh=WUW!i7L{tUZh7ft@HCK$=QNTBCVmKmcpt`Jt4TKNXrQ3P!=W( z6>-~;w1k0Oe`3Z**|kYV&YMzDN6~dk-Qe!|XXTdYS73sMzeq0j3us^VRg=b;nz49& zbZphDG5aJM@)l%Zw+1PEleiRIe9bkmER@y#IQg5r?{6xqb{Tw_T#%8Vpozn7Wno@G z)WW(vxJUp`cz-a3UCX-n*5fY|_RlX;FmRlh;Kk&A@v;GoLBj%7uDEvIU*6Wh9Nyiyl>J{P85_pEt0$w04~}&W0{7of$IrZHg_U! zVpHtCKE03? z<*4FWRV~w0*^^g$j&^?YG)VIsb2sSxB+|9mYv1#}#?|pQ`ZF}6&(=Adh7D`>>TX!c zb!_qsrKcTSffXFS#Ka^qYTgwt=gSbVKTo&HnqHYcfJ37I9FurMZeFQkfReGv+y&R2 z>4PtVz8~9a`4tKDBc6mMUqN}asj?njCU^7!TbxuALuUL{ScDE&IeLeRcG4VUwXyWu zZz`yzblR#&VY+r*FuKI(XN|e$pkwja8)~0GXR}!nNn+mWmrvZ;xipwsOaR~r>43s%1>mHw%a4dw^o$X_H{(-6{4B$ z1l4Q=D%g=^8cWt}T2nLD_9en~anP^-XG?uJaKUg%HFIy{b^MZmV@r&Vdn%W55$g-n zR1N+cMLtoeKD|DB?t<|~VVEz28(2)));3*yM_rU6;;@8rm@mf1U|GOCc0{y8!BuR? zR*hW_8H@D0=$2xo35olMX@pMk% z#>=Fr=^_0T@qUjf6Aa-dVQ!I#GNzZ};t_OX-ltT}&;O{ysupdc4lMMN5Nm9IJx)bLr-&=#iyJE`9NN_NHU$g zwm1{(Qi@UUF8L6`6WzK*q31p;mP>rj0FPtT{KZ=P26}E+wQkyBLD%m+Sjj8c@DQ~& zJ>Gw;IT)3{fPA-VTI@qT%9AE zd)XUGPG=6T7RX~t93VCZ83OX89&y?`HNrg+3ZLCi-tD4^R7{jiGAXl4 z)`=k5e=a|z}gs*0&rB0JUu$;Ev zlHKfA>KptreB_sU(OaVv@?|Zu&#_#h^37ecp}7wY8MUtGEcw>`RfeooLjhskVYB1W zy}3aMh3^C27}mF4h4&XU(vyvtJxHuhoJXbHlV&F|_cTR`qUu_CiHkp6Ot1XvH*ptY z>Fil0o49V6)xWr}_pzn#yF%fuMj%&T!q@ym?~QZA0qKS$cT%Q}muii8H=avsZQaXJ z>)SU`3A0jqKW^Fe-%8~M18zgzYH4}0*@Tm$mZI? z%!w`z8W8=$JzQb~SxjddSGeMA&oMPbb74KBF%o{OQ}m&G&_aiV9@Gd&_ymz|Yqyli z{(Ss%aRHyXYs1Si4!iuPr5lR(dK*z0I<{4946hp=Ow_PXm@(NiO3FL?>nV)~W83G) znVQk(KTx=`sQ-gKs>B@e3Jmq-q79?v@pZL8WtJvf-?bWP4Up(TeHR-~; za1P|S0cXE&yo)T7VX0Y82bSGQyd@$dy=kM5q^mgJ)xN8_@R+lNJeBqNe2PwnkH@hX zt^zeacSc@1T!~d~yo~4K3+`LVqj*1bwgWckl=PCV%4c)b6f@P-?V8P>Nj%DF!aoyP z@<8r>P02)P8UvE`xTBx)yBx+w6NzGo6uz8la#Dho>%+&D!FwC&_i7LDF%dd8u6T(k zBPSFXvfbxiF~1sSzq+<8v6r%wk5YOzZ{($fh7Q&P=b4?Ps1RaXh@+KjrWvDn`DOHJ z3NmfG$2Ccw1?{U0yMFecL^U%L;fzKZ}Ywycd6 zSQ|j7X*XI%Lt|a2cJ#iM-D9k5m%pVrEM=0PkMZUBqhIho!tm_WlI9s9sqkPA*OE+Y zTXbyl5L0#u)s=5x^Y0=RL9Yf70M|(;~Jo)L1 z9{(XVP|xPZCicNg?NGLz{7Y&h8Poz=8{(xU?Z3Wpw&Wo4(ygHz|Dz=M^{@Z^t$)N7 zuq6Mx8h@-tH@zgU(^MT;rHNmO*8Zai_;vRjJ+5e+E$do>7#G!2`B)PWpkCIv-d%m} z5IEvb(ANZpmJf8xtQz@{V_EdYQt!r3HOD{d1h6@XpvmIFTo+-8csdUm2yLW*MxW+9 zf}}$KFYm{}Jg;>MyY{=PJ;>uI+FavOe?VWF;5o6IKl&HQm1`idfuyLaopMqRy|t#R z71%wR*%rqmsk0Sqm)jzUz;`3+`F;%OafDj}S1B#GSsNV2r@rScAl9OHuO3K%&K)8i za>4Fa`WE6P?4b*|#gt^Q4ryRWVKlKrymJ7yz`!69_e> zC~Z6W9@gv3K_w9C;$U<|RLBvCkqu@^h|Ptc0pSBfiwHziUF>b<+3;OGX z#?I4G>O#i}qsAeJ0x?E{dD9Ez5Bo0y4xrPTRsBjYDo_ zrz^u7D?E(a4P%y2TvO&UMus1hB_3C=fW-}=0JOBPgfw@HyepM}q z?s&vg2sr3oL{j0#_4>8Z)ju`7D~mOS5@~+ETazCVb6*2b0ZkIbQaVijpmPGb1T{iG z%XV+!HpyWAQe|N!)E%jfhxn4#ih-2h&37&$ey4O7?#3}Ra<~Pe=EV&P!CM2ex@~ELSGX2s6rpxR z^Z9hS*8_WY)%ARS06^l4WHZmNfpi;S+K%<(Pb$;hOjsiYl9`a+Ks~3K_28=UcT_-e z99R@)jUkD_6=k{8WIsDRP~EwNv|hzCO^qO8c3aeVp3vmc3V0*{plgHph_7*a0ot!HmTHs%6jr z#D8a-p||kkYe;J=M>$+H2SwWO*zLLxT9QE>r<+Tgd(XF)E8M#OtmYR`qJT*gHD^{Ppq4yB@k! zYWg?>e3|BsN+jyp6&S$^>nrEyOPg;k#!jmKT(xZ#`rfC{-81S=?>eR&mPq;_0bnun z794=Cg7%M-`vP!#r~N>0u_;3M7vKttJT=2s2m&@$ISuZUk3RFY9Z{JVepRChnWrH~ zXnd`ElLtZVAUCK-^-H>kWkEQj7k>8iNfFCR*DI@tTKcuQc^|S+Xz1Q z(te(ZlI<8S9Bm|*ymKwv`MN_j8$t3MDbXgN4u9d!5%*)@JTHfx^;Th@z&}a4os7*S zr$%Kr?{z&IvcyWrg<(=mDJLmFGI>|WJ6#c@HlAcHNvGB+)j#t&m1rdT4P($iuYccX+t4~4J5~ZO} z#`S!GP^Vwf%Ot4gm#0QAlDuL`(*gm^oDp=Qm@BRcYZsr(y0lXtx=&Q{98 z`>#Ko0Biv7FWOYZQuU%YWnWx-32Uu6#~kC1|NZ;(I|DtN9G6xhdHqM` zZ);zG1gvy2ISUTtU+56m6YuUr4+ksIeg^wfMt0Jw{Uk(qSgzUaUImS?emFt^nR6!N z?6>GVIQXG*<8OV*ZD>!Vl}iV)ffgM>RiP%GQW`Ytr#j^c9es|}O~!W#$tszzz--mk z6eykn`>|J=_@9#pGbH23un)lh#V_s$c21{&`+s{<4 z7Jr;ZP|G~aW1px14f*Z7oYbeJhd|Xu6Q7g$TY4X_aO`dakpp5P1{&~wR|_WSAsaw4fKY%O z!b%AU(4$JT{w_o3=G_{Tx@Y^9F7fQi3$SVUmV8%*Vub|c=}{FenH*naCl)Xk2g%Jk zNSuxMo^1Q6%FGLatKiLKktp~rACRVDBkWn@OPZfZ2zHZEpGhoS%0CoOT-nXP2J$=|F&5f1NfbaiHn0 z1`a2;!))oILavlWo0s1iiOM*W(O@IBit8SFVkEg##WCkL%t+c*3TDK{O^YyMk}zwJ zB)Jj+K2`lLxto}ep2oi)h|jPU{hA`b3ZQYegpeJy_mPSsYpNE^YhP>_a^iQw<{hFY zjI^NXyqM5i)ixlME%tPTq^96t(gWfaEkJf$D=z`!)DAWXbJ#9uYui#t!i_+&{q^j@ zbj}qJ1*|fi<9MImb{<5$0vvrJ)p{F3N$mB9d%#M}E08))$^>owKBWXesNXDniiTeT zK_{;jC87?HElQ8UT-*g1ZL8^F>-I|qjV=C@9-s?hwniKSVp<*0c3raFX@)dIVAQ@m z`2qoc+5&+aHUmU10RqBH03j&f0G!HYW>3?a1tXR*a5q83>mJ8a_Q}m|rwgXl4%z&1 zL@(Ekdw|xv8;E*{M8yu!$)G+b(W>D!kX-^{1kzKIkqCdcvY!B;FNZ!&THXNwZ>NAj z#>@?2RTTgRU?u+QcN%0aICpc+-Mcy&Nh${^`v7D~!x>zvg0ik>a)-Amzw3Y^G;n9Y zedYmzd5HCIbj!by!%wUw3VgNj2LU7nU`7;#{`3q#w-YfyHjtoJP9Pcqks|8pAEu~n z*DT(fM70Ag$}r(;7pFuI{fi#<{sA*K2u52`0a&U=olcXuZ0UV|Ki`f3!Q3|>*--M{ zA9f5*q>+VPfl#{;D6S}OoAdrkwni{sL2J+Bpu0k}PHASWEs`8l9ngNeFjH+a+ zD!|1Fx>#jF4wz!!z65Ykr7u*MLw#2|Q6cVaK!OR!z~z?;?28C}QVRw37ZO%yH8@K; z99PGwnm<7QH1)fZRTO#{&~Rd+_(m;!UfPaYhZLKd=Rh6%4n(`eTo#W$;x^BlT!0Bb zeP==~8zkqia!lQVOOV17jLJfS5}IV2=V06`(3Z*_Wnc*8rD2+*KvtOW$>z6gjUTR9 z-QEi65zugxe5OX0#&4Eq=FUeXM!;GRioLIb`Nx&)zmVcTZwh|)G5G~+z>AU-r>#lq z>MWRC$u|I@SCK9vQT_&2R-p7p0l%~weUqJ+fJ;`q#3%zKZg;C&{VD>EB0}4}z(C_P z3m)%NuhuO=QGZC+1H@Uq_2#2Pvw)!c(d_~0H-Mu7QD;7?nR&*=nr+Xwatv)OdCHBC4Ww<#^cBWw z8K}5Kd2VgQ=y0u*|0E4t(kaVg=Sp6~q9})I`(kH+kkr&ik*!Aeap`t<3Bx^ZGF%tl z173eRKl}MlPpm*%c=yK(g_j?k!N@3P4#)kt5=;Xz6k1+-o`=byrxq2163`&2nqNH# zuoUj0u&M?<-2)S*`?j#-j(MRlw$(@c%f7ey7n&hm=OuMl%fb`Dnp~nLZm;c#JO?Ah zX9?xc{A754=CS0Ekwt^eBpq!5W!mM)GFqFsdq#a;bRQWqdjfn)t z#7z8V8v5r+EV7t0M3CL}iQRPDN4A!M1e@CuoxV@%YM7mRNnpZtcnuWy9>jN`3;8bDJ*9(1=Przs23NW5zZa+1*1 zY+uRYwu*+c;C(zW=)<=%j{XP@tItX|eF|#BUZ-d-yat;vLTy(ibZN8gkj6lat(*cW zqs08p3mlS%Ll1SRpw_=tcZ(-yz4f?zI+{D8%6t@A%Hb0V#V3H3deD!N)Y?dl_Qisj z+{gvhzlMvdYTN+x9+CM%R`cfw;vDSy=$chi&0lPOj$k9p0K*3=3lXIRlF7SwXB3BIb05aZlX@O1H|r;}dx#8B zh9RE2Js(`O0K(LJrVJ&1OE&4N3geioW<1mZ;&wbd)c^u0uO`}+L+*;g8=6O&zoyZ? z`VmL!bhk{Pl}V#bLsGqXAS<{MKZ=2Zl#0NH{D)5$ zO#@ec7kpOZ&qb&s?VWjUr3KDR01 z&#D0ATH`D{>=c>iPlv4N=+6Zl_|mzNPO@HnthorKHkBKoczTE#U|xIoo?$?96s|nl zyX_u!g|X)p`oJC;*Dav68quVtCG~;nLFa%m9RpV5&V64r<-izid&A^uu53ELaZv`8 z_ zj^k_b;ecxtr#O}a`~?98iasPhv9@jY0vS@ezDp$@&ZXRsH9E8rEJ~GR;}OHq$cl&< z`Yvai^EREqKdKdPra zD&{%cF$=~MyH&YON{u(fY|#wtSy70~&6G!~Y;I9K$(Cb16#I~j=$hH#$2<$`Eyl$* zdjIe~!k$Gjlk;*GCHh)(KAk-Qy7Nsi(|ZrUS^B}J4J99o9CIJHEhF`7X>_7~ ziMS5I_7~&Y`HV=@B<4(tDnaYn-^C+^i;q0D$80K{}nZg zA~oEJp4+8^1fa~hxVVnUKh&v+kiPuH+WKyXaCG;rcF&}rLdLG)BA z9<1x3=my{naqtbY3^6*Fj-arHy$lK6msTV_oKE`w$GWi${jyltwTk%fN^GtRk`)?J zE50?Fn0p{6%X9w)fe=sf_s!Ukn7y9i48MATmzwgT7-M@1@sRM-3)%=P9t$bq;3Yo> zM6$!2o+ePBkrn?DH;}f#Z|cXaK9T` z;A<)`EG<#mYU=G_Ibx>ptNfzGZ{Fe~$s6~}GZK3g=CEI(-x4znizCBx$k%U9=Rs#x zMmG@DSIJE9?@~g<2yyCLVn|*4`{IR|`Ii$ae%taxZn>2o^9fH-|G`tz(ih~u7JkZ) zj3f&Uu*IH|y|z=4&JHu#ITgdd(>iF>&b>EYz!DW4#QF-GMnxtkHb(H_W!%>XItzw< z?^>i2h$&(V)PLz1B2W%U{E|J2wIQe7+IlBtIM$5jb7hC;3-PX`fOXG^H*Le; zWohare7}s9`dN6oiGlj|^IAgV3Wc6_DiQ(86-DoJJm;vxc`2cRcvBozHv3-Of(4yO zB%u+wHoGSlf0pDO=RP^z$LF!TaU9*QUcG=v9B08mpvQ7gTp4MybBk3G5vc7F#cQ7l zCyrnJfT#V1;`Jc59}VlrFh<@g1A-Y=7n%E&zjRW5@E<;vPK&3iM=>Cpra8bRJ6B`b z4*Bl(g-T!yoqM}tfA}Pwh3K#V|@XtbsEmW{li?q

q>Ls>b@DE{jB?gMD&hkOVW%Z-F&T5%mei{W)V=f?s(zFw zQPK4|bGxSWNUToyQvnp70X3$5H%3id_)#OW6C`aa+;+d<$wDxADAr3#+GS5Ot^VnzSicCeg6RaAw7-r@sLuqMr}opA}G_?IcoCi>T(fJ=9S_7GO>m z6?Bk219`X?&@t2eZkLnB65|I|EAz*aQV;(|ji%dTyqs-QSriCcp4tEm2-H4Nu@@^% zQw^uv@$e+wv%vwfY^-aur-N{MC*7zUsexD+wb&Wo8~cB5eJ(dr`#wuwfUBA67xtI$ z(Q{6wtc5e!kDM5Q>Ga>_#gTFmANzsngT8)EhPk8tnRAsH8viieim4m2;gJf&P4_P1 zl;B~Jt>TW`r_NG-7(+6sIA%=PFXb&iBh`_m(ZAC<(3^lDXA3JTUS^txlw7owNU-Kk zu9)nQH~OzIant9ltJdMatw`gLH`ZZ)zblAm22nAv1`1}Xhmndsyfy`2n6?YrZw94I zRn=_5cf#s(HYoB!Eco=xXM@AiB6}jf&k9^ndS2gvHVPPJIjrq2UVcN05Ow~|hdsc; zNL~H}NH&$xdX73@MxH5(D0Pq4LVy;f56A-b-%-qsWB)q3|Nc)dzYp?lju6AAOk&*d z8`5vj6CRv4h{e-zh>7>lG$`h9^=c$UQAj?4=V+mh_uwPgZewS+#TX!{bKHheM)o~PAZS3Q?%?p23~+ZwXsUo{(sxO{8c{)7V$xPY=Q0g9VUtxQayGy|1F z&WY_hDMmtR)L3rc+sdNQPxuK{pjl{Bx20puKwGK-K+|rHioKQZC7u4C_PdWAK6JOm z5*5ZPoZs58k)T_797*2c`zVmPxG#X$%K2#f#S23)M4T0l?t|0J3ONTQaalsg!13Ck z??>;`7BWyrm~8a+?t;o@XwwX!=mL4a0C;&F!1F_Ol*4ph0QG`qSKG0vPuxKb7eZs9 zh@u69qIuB!d)h9*vX5~ei->Irsf(=!~lW|Bw)+WF&SI;p^9;O4`f4>77upK z3oxh22O90JN~3@{<~ZxR8tDzVUoYpOW;FnL1+Hv@hG~AW4ng#n$I$qbD~$G?4pi@* z8gg9)oZ9q#sIA!l{06jQb8nYsM$nwa(+&a4oEQ0dj5sM-R`_8}ptD~JI*d^;4l%C> zh(fu4+PxrO`O{=^MdzUJ-FWayA7tN(YxrVwcpnG>ocaE%EKGFmmjNLWKG$@ID(FWv z9%2rk4ndzUUiGyH<05094Zfj7BP4gGhWLM0b~WdE7Gv436m!>h6w0d8MLB3ct7xEVA+3G4CqHS z*&6?uO#hF)hE@_{EsZ#;0>Oizelv>1?KYCV2+E;WwKeu;y{BNLPT+@Fp4g|;x*Qf7 z*<{$KwL8&+t z>9k6M+UJA)0Wnih3+#y#X27`*k1)YVz%|vMBjJ{2~(G@GzN9;ravtyIC}GRyrKfG$nldUlGF;3KKPd73BXzl>!?>_2m zHH%f|TP_J&-}JE8=J&=$=>uvq)_sF7b0x{r-f-&pYlfa#HqUd1hitzB^E6eex(>Dg z6spyO>Clow_0n*8zuMU(N&+aAxZ%rmrz?vvgHK{GNSkylORO`tjowUV*fPzDxIW7)}E?-BQnuE{8j?a`Vty0>HI?=j{68 ze!cjbb@ND_1L;DZ1SF>bl2t939FEp-6atu-TtGq?frmj%EGSwo=_-}t_Z4dasrgaE zi*N$cWPD`4o{*M`9|DSRai}|^_+wdC^f^2mGx!A0UOLlx0EdKX?#2fV;auDd{vtZkt zPZG;EE940bQ$g5v&^ffeF6>k;$j5l7XGQtJ!Sno(t_FqsWjYtKNAy9Z(gkz-6HR`Cf_#_QFW*bq&dnL~ZL+zt=r} z@5s>DL2(zzt&m9c9$|HV~DCg!s>UY+uGlmM@>Y?Gy6+>3=_ht zG2S)vtz+hEH_tP});@cjKfIr~5IHpwW>qN;|8MLX^tj200{>q5vg1{Y>eC;;nGQz= z_WUgFX;c$ZBrUxeZX1=1gSR7Z9G085@6IUgd zR7*3sG&U;=bbrpi!ukIPpRfn7ZenZ~hVZnb_UYiW9`+s@Uyxni#MXy)%;O$?- z(oq-Y{%~oLm4X}!wPY!I2mTk2?k$UtH_o+faQ7UX-&!ZsOvJ0oVCu1@u?cF|OYjSD zQl9+(Ho1uLzjZ$@WBasX{##igYvPBa>5>p2DvcTZQTFxEnfmwjgrl=aUT=F&TK@P&nZIAiid%lkVN&UOx7xco^eY=Z58d}8RuHnYMrOfCaJf?;k5@fYCih}K-|#K z(UE?r_=Gp4y%cc_985l*W4?R1y}LG; z?_iZwD=ex{O{he~t(=~U%9gfwq=oLB^UuTa&kaxb zTBBx2h92@>ro;67qYc5xM7xYyx0RhL7^ghJXGc75F0U?jw>7e(SP!AjS(0|;2SC(p z`J`-y!$RaN7LXPfjgD7T>O73WO5@xn*j(kzYfAo z{*_#Bywsb}*mFXdMK~;MdS)n zXnEV8KufUvb*>>#Dhd#4CzM59LMQ&^ENfo zf|%po`Kdo>2&Ed|2U zO`tm+%VjfBOTUJ!a2jjfq3OUN23)WEUza&J#EYW;d5rOba&uY z9(2D2QlP&hKXPA@)hfoYX_rzg`k{xc_AH;_W*O{~mSvBOvn{oIk%4AHYpMVz0 zCtEdJj?TyIH7aVCrXJ7{1LwMJpeAzVZkdcPza@+|Yz| zEKNfXZpI{cfyQw5oc1@b?`oW>GA3X)0$7}t;mL2|_)z0B>;h(&^X7P@2!$#hN6Jmt z>lneeY-Ke-aD`9H?qk#Y1I5Ky%{u?jA|_xI5m*j1RHu`{qsw`}hNUhkWcZ!-`k@LA zBnU#9(1zarh++V&V?Zk!>qwpU|NA3h&_+}@r8lG$&$`#vm7DfSp~pnKYWHPnk=TaD zum{r<&=P>FpJhi?0R4iJEE1Ct_#5qF7-kw_N&oIiZ3L$?Z|cqPGW0cNXbWVtNGgFE z3=((wxIF~>SO5GrWCWT^k3_oEY7RWm94l{k|LX#QK|tQTKVqBDb|n#tK`yLGK%DaV zLas*S>!3n!0ci4!0N>X%o5kW9kZ>&n(DCE1AGaYqKL8*AD^cf-lXm+W%e=q8&N6k= zBWjbgWobEAwImvm$i|XLd3{1HLR-g})$N{a5ns%a)BDChM^6EZ*X>szjmsD8aQyYL z;yKZ~B!`e*0%$jY#Gt#L5s8FU6nKTjPfz^K5sQ$24_|;2%of#6c>EidW79V=G_G!q z@3bW~UUW@{2Cy(OUA{P#IcnXv)n2+?6%EUl)k^O~N>m39 zcmPuk^qq4cBYn{g>_%g~FQCc+6lDYX48j}l?|H;bwB1{c;Hzgr1l1O1UnsT+3%IS-(4}GJz4d-O8o*U*g-2@tb0;DD(z;SW4`~UGQPs^ z9G}+Y{U@ATo~8kz2?6BlsiWv6l}nEa0DYo;h6;6EPsgE5X-I+d9_jXmYU^LpGh@c4gDTg(tqlTc4Khjk zziE;phPdPp%NN*YPK(z`06}LL8$dzu9SFZ$*W}8EAS+vBOokF5 z2AT%s?+`F94?6rcKzxB$>oI%&~)Dk7uzI0tZGs=$K=TySLmDH4*C`2g_w z{rPN7Yaw{=zb0ZY6M$3^FMS~>w|Uao{+UjCg5X=AuAZy4Uy&RJW@JEGJLmxwI6$kq z3_`{OXE-osAO{3#k|B>ufrtY@`TVB>fv<8DTE^WzObs0}&vma!p1{ zU%s3dj1PnHCn0pOe7PwIvN5uqdw@9W3N${99G3Q;0xDRts>AA-T+Id zM{9?mF!tco=vK_)9d_hz(1q-j)B~dQjNzxY_J$m1Hvc-gb*aD4Cn<70MWkuScUKdP z)CF|t;+2SA?# zT`d$b5OKe`KxX_mf1%|>VsbgnO2Z^gm0g_QR^jIxy^D12bbsRQ&#wflG&IkOLm1j{ z28^adF$r=lz!nCs*9c$xk}E*UYV{^lL^H@-M``vNFT`TKI?>)G7VMoBo1(Y-34 zIe{G{L8s?F7?QFNP`%Zrq&vy`FK(OMHSD3=d%S{UwfpRU3@_64kYHWu z;VgxG0K^8hm7k zf4t6Z()-STBK$K<7NT9SC3JIQ^`q*TrZ{>eJ6$`hFkm9lpfd%x7*We=H%Whxy+YGy zltXqvUwP-Y*b#2pa}4Nvh!HoqU9B8gO@E!~Z(V%A=4m`$q7Q-J)foWD>`-G8+gqwY zR32qNnQHu&yu4CxzQ{vxk2LhOwC*4;wudghSp_?qky;|;h}fM1nK%NJqp^}>{-4~4 zfU$`a6f85&q~8HM^^idw;G=l@Di46&31;3tQHMh!O1fma2N`WZ^EoDM!(>^Q^J|8g}M*hSX3%`qFyb@`+8|B{6Vcu*2K z!%QY+{;U)5(f$d!(jB1SY`e77`uqT+Aq=3=<_ z`Ewb>Bf~22>Chu2q>`wj;U|PRX&1B3T`34THs0ui6h)x}-jnIop!QPz=Rfg@Bz*e@U}HOVe-fZ?8X1w9|+(BO@aoOl}-V-acIt zK93a+Uc+8>*-n4ueCY(`R2f<1V><3X>o*{&1CTD4{6<@R7O3{;}0M%5aArHLO_9iyu1dL_6W<&=TzmxppLKd=@Y_e|lL z7<3O>2s>Ti&>7}A4+u1&WWrvjG}C8&xXH`i@%?GS!n}!KW)Q2)2eC=TAAN5FxVkN4 zriy;yq=jK%+sK^0l$4ZYO9IDF-*4{-UfneDN}i#7Wh{L|G&ixDATc!DN76oCW9gCU zLK3{&GAucj%SMfqd5dDxYVgiLIAzgBWMj<>!y0QmAb>GEw2s@LxRz>(R%OA!2&~ROyqINvBTh_;=RJV&vk4de0w&K*roxEn6fg z%f^>2^>NvIS}eGgKXC%Fd;@u}X7-PWDUXN7*af2Z2qq{B$CU}QxQWkxfOT^jMlGV( zigK=vUh{}yo*N1I2$}M@&w&CNrK(V5g<@sp#c0aJv!7shGPZ@c`)7&-aWXAVET9E6 z+Z;-qj(ap49qoBca@1I*$`hrD*9Wd0aS>gng1>xiA@hK|`W%Wp(X*U;e?8KANoY%{biQ@pySQqn`)s2E+6rwoo4AgNuFENLKmJV7 zb}pQ7xuPVQpnaC?U%9ow{aMBI5sS_&8@3M4;p2I%(~C&$}>XYKv z0^Yq|cl%QX>a!&Eaoo|g8jH*k?c2w3tan4KQzu487Bz6tamiT>kRNbRN^CuczB9`+ zRJ)UD5SAZW+@#MCCTFtBKe%#n-dH z>=h@r4zKPsR6Czda&(!+AhE47zky$NU*IIqQexVVa-L+d2J`ccIXE4;n8k3G$^5K8 z{640AI5CwYH|}|&&80ZuxP?R`f7-_E)v(#0)Ltz&>u%O<6>jZ1c-M(+&T8I+tT>Q5 z>07_M-eL-gwsdd7>~u<~y^}S+`!_jTGBv0Bb6euq+if=DCkIjcxoyvLD0&gJE3=GhMRf)_1whKrtea7 zm37ms>u<50ap$qCsg5GgO!tN6zF1kljw)-u!%W7_iL~iny6;wkA$e zg^DZlJbPAH!v)4`ciWG`pEODiFZE$R;aYtbY_;*aFdbWmbb_M|e|joZgwgduP^Q+LmyV!soj1HQW5UZUN`S znL(@8_Uv5VoZC_Re1n&!f6BO1ov7I-B_&1U$LH4sUTFT%M#*hUw!a`&^ULO*^z|}< zdG_p8tUPzY^mwhUoUHlq*Q_Vo4X(BqA@T=dk1h(zSG8RZH|0FL8sj&4jadl0zCG*h z?VYMR3npYv<#RN%`X!WrL0c}aC)v}DctXRS^*HN#ta$62%~`gFW$)3Vu>yR=uJ5o4 zGu|>PohVEHl_he9cGK&QK~m8p8fY)tdKxj(1Ex;SGL^SM=SYjZu{?!C{@8l!Na$)^ zvGKUFxR4}vcJr?A>5s`43JV6^29K6cjIz}s-_o3Qj(0Z^g{GUrtY`6rg8KWc_D;1| z3)_96Nmg~%i)EZMy@rW26;&bT@4Uv0aj?fL4zbd%4|5<8A6bdfA~GaqU{5>ik<;e~ zuD?)sn~Q0BwR~n~#@cN-iPY`wJl~1yDcX*fi`eR5wfts-bI-=|jCN8-MFnSNnB~Uw zxZ|3%YM8M7cEjDi?`LHzm~-F19Q}SWdXs10^v!d}eUATkL3!%=NDc(DCA(Uayvv}g zo1T$jxZD+cg=!ON zv3FJhwK@URg1qv^25snBbeiM{^%+dv(`JxA*YGG_-=)*p1=BWtQ?tiaq%=28cZHO0 z=3a!G3Q~!r^CA0Ql@x^q#QhaH`JiR4rZE`^iraDA94g)%0`-syt^Cz>`xHn|zb!?{ z+T9+(Dsur`skxe6@-)whZk^+~o2xkx6g@UNi^BhIxA5&FJs-2jU_l<<8h+rXq&b&qzhJ8=o2#hpzM_>-M(t|BiiK=y62%1G2GQPOx~})x)}FsL!DRZu z6MXZ_5$MG8Qr@>>H|+y{!i#0b-H+pU4kki~&k%}UL#wLP)2_WIRhh>o+rqH={yur_ zULvlM>EzJ7;SOd^7N?%LnItluiuE7j=IvTsD~q~sE)?gc@*CAIFThO$wW8}~W^dPU z!DFeo4e*}!-qaVKg8n*o$zQvQworZ-2rH*COlH zd89sPKW2&JP;BKr^F+5~r*tuXy873-`SxjZqp;0JblUM(?(qJK>?y0nn+|BwYVsIQ zQP`pHbGZD{da*LSS^)&>2UAYp&3=9bfe?OM9FulpC@4!^&CFeCM`3~n@;q>wJ`AWbfxRB))n@h*Y2YgRn9%cN&caldnr2} zra;1RGZ^l5I27MdK?^tPb)P**1=etsh=Kqx{;VTO>S|o<<|9;!ywd`S}<51@yU?^qFMc2nLnb;}kqMv-(A3U{AgEMwR13aONSx&q$C zGptMZSoeG=en_+{is$SxS7SKuN`i;4Hotx!h<7FbYAu`vsnnkbZ&_57dKUWTwYtwZ=YhNBVdBa z(C+8a+f+R<=l9wk@7{THQ`H?$lQhgDiq~IVx6DX()+q_5>bD^?V4Aw4QO#iHT*Fj7 z)XyMj?oX+NLEH#LN|t@Bpz_z{x~QvKP!=e^vA#LBX7A4Ck>=B0*C25i5Iwt!$q}IT zr@A=)S?tA0QsPG!-%a$OEb|yL9GoaOrHPCe1Y#M_A?~fMExHIBD!#yG-~k?iTAv_@`A@Bq@M%@hd3DDPrw;|f*d5+ z6aWD>{K(zeTIratHz9xIL^{2RM!g>&fJZC4ArqTDi|5FpY&06$%P4-=y;c3x^WTlv zzkfyeaY`>jKaf{~uSu2fDWImM*oGbP!Pr5m=1=^Zv(!CG3y`v|OYuVp@xAwl6a^1;fb7dXr z(?|`)?E7x%(Vbb9UDT$|U~RdRT5|J6e5{aYm~)xCTG(yEn?ZQFQnm49nGuzbPmQ6$ zKm()h=!4*smqLgF7Z@%2cDL^J{u_h`U;R;`b?;q0gT$0G7sU6B&iN%&rMf$2AG;fE zf0SX9B@a70W#?X=9l#i7$FIVKi*JU+p3L#)E345j65Sl;2zzd*rghbHzWD&|mypGQ zO3!H-aIYvqYjyujhNqrwfBay=wWf!e1HXV>BZN z2{#9XZDxL&!CA_dLCQRpSeWFK?ZcoRKMv)^h<*zKY`dZhfOGwFrjR*8QN0p;B33z?=fr~ z67ALc3hKf_x^ElzLaloSqI#wg=z#Ek8PiEvo99t<>PI6$0sp{hJ~~3i&o(75Z`d9> z5knzWnbr|HSN7qZ7pa!u$`?)U6c%%_gKW>|v(>rz!mxk$NRt2i= zGo{Kuransq)gyLQ^nHtlAAQ$C&ZB)!YoGlgZ;-&K!zOcn_K{!ax)*O0q7>e(FcHw4 zEztnIKMOpm51x=?`vq*RGiT$y({wCDJe7B-iQ$kOH`;V>p~6pmV}VXXvc|;_2HwxqAp)UyG%WDF{{9 z8B#EWir|9L#npht?EK;j*%@^4#Y^k1?E`@=;Yc-g%Iw}^$LhEd`vECKs!ZYR&GpHu zb_DI}b9hg}Rb`56^{5l9Ji3NP2b6+}ge_gGaukx^nxk!V;WfQ=x3($xlt336nU`dy)nU+kbvL!gQPA0mjaGA-xUxG)v$K!#C z;?SWh_G%W`=I8AE! zB^qHwally(CBAW=1=sOBLr(-o@mpujH>k`mqERqtR2+fT=Xnfz!UaoN8EZ}%ceWS^d@qdQ*P-T-|_e@41Nvupw>dxk7 z#1T(+h;^;O!Y5#3;$+_4-$Fk&UE?Cl%_f|pJ%dxcm3lLf=;7FS82R&Jdy>o9EaNlv z59xC{T_r+zvBPBW0scfB$E6tA`2Yx>%b!w?uJ(QL=);AF)Jpja&zTTP$3Z(&kDd>0 z8@rgcN)~?Ke0)d@ot+ti0sm{BL>GQe*A7-30yov@sfv$76PpAjZzb0IuKC~)psa<} zCe=|rXb8!bv7-p|oh+rSD|H>#Eo1jS)5gl0n8YwbdmW`zA0DJ5fwyEFCv?d_ch*Oh z8N{~UYNBZOYP~DF`%Pv$M2M+NRVC*wt6Pa~hfFcu!YQS^mVjgs({brQ0TD#wde_)7 z{Lua2Uf1jJUk?yFMxx1+6_Ey}krcg7O~0e>Cts9d^!`H)!fI9mAMX$A2bV}vzgUP7IwiMICEq9Xn=D`}ITUTW6c=Rbq6`)l;uyk{DVs_fP7z#OfL1UKz7 zu2I!wALbb8ygZ|onxwYAsoQraezICe9^Kn>O&qIO9&t>&NP&d>6Q89mFVS|)4633% zDiHa$E!_(itR2Q#693|my^&YWR2-9WWaavf7NNF-^7$pN7{*fm$|XgrSXoL2TX-JN zGj`OWe8p<^t1nBKK`EdFDq}X|_37^KMLLrW=EEDDNr~S&6yUs->tX&}tJ~~PxSSmwoKl>Bj`MDja5Mc* z<*fWY&U@2niz{gM0AX;ha!kY2c>c3^=%C+eD`^8Qvn4vmZ4*nuNg;`=(5YFeZa=nI zPMEy~N$6stAI2FiV(ad$-BRjI9R@&bfKbg&={MQjZ zG>A<)^`Y4Ke#uE6n6asTbUU`r6lO}`AG2|y?=9$sxa9o((3$jR1RTw{gI3luU|CDFuNjT`+%cwCbN6b_B)$cHi0AO^7h76#qLJuk?rz!;bZBV{;q9B z`Ba;$?ZsKUsh|v+N@8~7>5aak@VWA5+>T~S3r0T)8Q;V>7O23;7!Yq@?5X#|TodAr zhhmCdd(F+rWCZ)7OQR72D`Mo`jH|*i25u(dY*1ZH6bU^M)iZn)@#8y?RTo`u{7oY`5&iH#>@#PIw_-$dFf zVqy>>2|R@8iIdM7;V+`Sydtx{&cQu2O3OPC(FhrDI?S$C0gd-^aTCDKWTSh;&U6C) zgPce)^42o{Zjbd>0$Ggg1S{3X)wG+lG;_~M18*JiJQX!J1uh!dq7jVnp7!B6j)l9y zYQp=4eOV_~@Mun}__)5lq8^uT=f8~INlHEXecRJ)R;JsWxfO-XCdI=GW=-e3I!=wS z)U!RQ)m7^1g1aP{T}!EqgJNWK+AB?ee>P)egt=w~CuJ#uowO6K^3uJkF|;m$-Iz`+ z%?&>{TzT)MgtD=m2I~Z)JKGv1bL>;mXmu%Nm`2hYo$Ql7E&AdL4yqPQwzBLv_YK>1 zlGEMVVgWAoe!d^^UaN${H!~}@k9p2(78N|Mrf6XD2N}Wf&K`wF>!zgM8$^aVJ&)$C zuNi(vbEpX|2Yn`n8^DbWz)wb2QM%~8P`XKzq_o)Oc)*3|P~Q<%c>L%vpeH}oAt&k9}x)NhsuE6r|p|K=2X;R+WR`*P$m?)~dVppSQ?yT#fdVBt|ELsE=_0DNYQF;Ij59%!V z2%$!0BXZ$~qfse{T_2Bw36~%9axT)|oLA$M6Kws|hGzRIW_L!%o&yBUi}24z{N4HEOR(07ln++Qj_0}HfIF6nDZ-JV3lTwkF*dlA)y z|KXcx&b9KE=~Jo}@8e<9K9J~qJEP|Ut%9=&K@!TZp{THjuTUj{#BGa7fkfvGy|4; zov3{xXx1G8N4IZJv0sc*B%P$ikbY%UB63G*X|_p2hK`bXD68M2=x}4W=aLE)-uCwK zbO@bKx5|$|!zdn&doUjK4R!-;_nDENsieJ= z1z5XB%FQVPV>@67B!sY=suzXHuyx;y2+?j8ZC-Dime&vG-HXR5LP)_!`@+Oen0w#x z^PeQw)Y=<#Cm3djc(x@=E2a$2&cAW}NRt(t_J)mnczZvJ{s9N}y#m45{#=ihL++$X zwH^@p#XbO$-^&`#I4JV_L>&>xqA|)!NzPU!JQdm>d!HELI}(pbThv&l5UlUXkX9l0 zJaQO>1RzGHDhwK*E^X6Z$Vm(@wX?!QqEptlOn*#fzAO_kJ0>)hO%uLq58=;M9LTg* z`c1PWJM+cpyhvPdIQ9ypEvZ7U%LL%D0(6%xBn=9wDuR}z(*W`%RM^};2=Xv$9QjGi&3BLFj;oGqm5B-9b%jI6lh;cz?CG!rxBsWTE02eA?fW(8 zkz~t~w8&1j##S_zLo&9JnaJ)4GlQW_wq$QPp<)JQckD|l`%;#OnrN(9hEQ3K$sk)< zd#`cM`-JDu_pj&k^6|%fjQhH;`&#bn`dz>80`(KeE92Xqh}=9>taEjiPYOX|jq)3N ziM67Z3_ta0OX^cDU+z>Yc(yX%Iecznw*OEC>sQw`B44N4(6rfDKimsE$!^g2QYOax1K=0r_t zYSlPj=NutTkcHuHJT-TJeYP#`_P%0M?bDOkALvud{KeeUbRU^C0t4?~^L^~=E|YHp zYi?Q_;9c=Z17z1;B(Z=W|22I4iOzf*qJJ?U$9QHwcdw6FL9CZ|F-9+d1Y%DmwE(rU`wrnUkm)5yF%R4VCmKT32uCT`SeRIi#@jzX`2%F1H*Mm4%d{XSj z)rU&KD}z;KGDl`Joa-T0{!Qs?K5cbN<*C;raV{Rzq`n&Lqc66K%WsBjJ-@^8&`@XR zRV@ovRiEDD!yA2E=Gre_GapqFBnLCyu_mR;S}zS(3t?e(Ek4L(>^mO#VrIR5@uyE4kxg~-aY^?5#J+Qj!P-yj zR#N!C`lx+cu=#$@aq7oCr#7;i_WhGd)F>PATKnNi^tp|8(wFNG?u<-ufMj2Ma9bKq zcT#7w!ol@|OykC`tMBG3zV&v;w-Vy5VN&fAm1}Ctk&P}&=cuIBaXZ?d_nRU9#woK6VQ3ZEFBTw}hzt&9 ze9tWSf)TvCn5XPRo+O$`yA_3fJ)JL`8p|CnP^p;f#lo6>IZVN=F=r*I0R(|K-ws+p z{;OFw85V`+GV{ zYNoM5?x@9`WT9^6YzE8)4HPW#IM;i^X42UH5zXfZua=XL>CcT0Uzig4JM}2NGxew$ zn$3FAwN~2E`PL!Q(ZY# zzRC-wjyqzl(v}_BmOiAja3hg5u+4s9Uy!osevZBnkP}nm7|z<lj69v*9 zv@h!Zbn--352ngi{NX+r!!6-{rML1i`mUHDi#C4X;)St8&frjG{o~_|+)IC_U9ado zt00YhXVvwXMI#A&LXJrYYLzg2vq_+@4!XvIJ|Er(9ORR;J?Hu8Xqh?X`R|S3Ki3+> zEE7>jA2)tYEru!Xy?8ib;DM@Cs`xUB1RV{tbM^(N4@(xP*9Gkn;8r&Yy|z^C_JU)s z;OxzlMbSws&$Gv{kU3H)#NpOYaNv32V)CU}O(!Oxo*Rey&DGTvdVpalD_2z?B!DJ` zos5m5uVQbCCk#nXCXcJp)-HAo4XFch2m`Mq)?$_9)H z>a#tx6FA7^z6sn5U=O->UyVYot)V7raL-);3nUJQ)2`Jub1)84A|)`hFN<7KH{sSb zUm$*ZZxdfU?GY4&X75Xq3gYb$m>ug#tDjvwk~D|}2oClTEisH`#hbEZBE18dV=M$~ z@*IWckmLFfnyW>t6uhY29hc{m^T_G);pvklm@}`^RgS{mfT+lLuWNgZf~?E=?cf(? zlpd(>Z?Ae#n?1HU6E>%MS`rwAW04V-oICI`JoF8Q$#bRH%^LiZ8V8q10fBvZ&*6_P&wf0aY;aJ}6} z_@_;Fd*RoZ*&Rr~rq6%nbY8n9Jicd-z#niOE!6h6%75kK>^{Qo+iCahw7c2(d-t+C z7k1~u?p)ZN3%lXZZUnb=XSlmA{GZl^Qdyho(j+D1DM-e7dvNNNk(Ez{Z#ga43mN0a ze5jB{+joa=ZFX?e+y}vu#vksL{S9c{a{3_36O|^3wMRw#=LL*LUJiv>ziPZ0acRPt zL@@eC4u7CiFt(J#m@f>DHj*#9N^bolxWvR19BmNx>6ZNp*f!(&5Gu1?J4(FAQh$3n z%rPi%I+Gc4QD*xQ2_2NnrTG8u5j4MgyHF`*t{U_4ZG)aIGX1$W4uC8T^vDJQgc>RbQurQ3d#HeShEA4Vc2}e0DdlT@%jXK*JO0Xlr0k zs?6`ywFlNj$KS0$>Xre_Xg2Ib_snEb4ZtP?h~YwC73{b>U{V3kk%2F*jT#dGK|CN_ zq_l=tW!Q0lc_^eY`EyLGft2OoSvBn|(E3rpAh}j7-!IX5+TxxVJjv0;RGP`tF6Yt9 zQBmQpa($aI;5GqJqr*cJ)f%y$?U%Pdko*y*JBewM6zjw?l+~Uy{zEH06=FytD>;%=^}nr_=%oWsra;lXSJ%U04A6!XZsRhc-et&R)_?zg$=bdrPy;6%`Y zE(IE2_}eg8EW9XE<3>+O3Qs}nEal2WnCH}wZ(kxB`DCphTNYYZc>?qyPoI5BSv8K|1z1x{X%kUc2nP{H)=<$BtGCXS}1BD5IAKKi#Ihbin8e8~ z?BqgxJlD-ptN;$k*Y|JC-qmQ;$p?NAm|hzn81RQ6^5@A_U7zfvr>;|h$Hc(E08)%t zSPLMh-Fy<|b+fnSrhrYzo1^$vV4Cpti5+svFnknL$67X-fj&@d0hhaJgZ<_R1AOJ& zr~L@M>~T6$bf#LyqIeWEu}`tog=1c*(irSeKWqRSg13z#7}G3k(E%Vu z;5#Ro3Md!h4siA?3aLTlAs$#L)XW6ToRW=MYpDajhQNAI{AX9{M$BobfQozTNg(k* z0Kn}Tnf^dOBaQ$!7fn_K?tgCF;tPp8rtjhNkj7(LMDA7t05xxQem7T-^*-)gjil)s z!*v}|deN`>&*_unQdj20ldqc|2ik}kz;EJJ(At|I?n8cYa&{ZQN#(~u4`1y7Qe*@? zAUs-$f1V&>tQ*69#nGh+0HhJzU(BG-v(&@zfWplu>~8kVHrWNOXRxns&}TRx&rqND z?$0T-MnQ5;UnKz6XB);7I#l4%qpYwL+PXZ>ualS(vwjz_SK4it-BT-#(L$o@*epuADnA`dhv5GftuPTaWL|oA$FoaR z)n7fNh#e|PU7uo&hwBOtz&uZrx<~s9MfE5`q-e@T0Cn_%+DWo2^owVlhel@kN-6o=!;Z*I=YJOY58VVEvH z0IwDCFl`~zoeFCXDbfiB%}tIRMsdCOwAc54m9rcm+b21wg#~9LQd9gU^J6F;U(dP%b6tT1IT&LIJPQC^K)IY11 zfnN}*-0-pdCoIC>l2&i%l#k;RD09?5?oKJqJCj*!=*|g<9E)~x{{1abst7V2nvlKZ zM(d;SsHv(_T?vPJSbCKSBJZX$9m-K+x`09BN9wrl<@JFjldX*A*8P+~OVW^$N}<)% zm6ViRs1|3BDAG9V2>ksd+5U65(IDE0cE`7dW=VO4EQ1U}8NUt9#Z`qBu1)Q`0TEvSoT!uc&Y4di@i+OT1Db{E8rZ+~7h$G=5H-Xc-dZc@!-o^;OcgW&=_w{~4 z#}6>x6XVU|kH!=L^_70uzCRryDBfbyjEG6`ipP3UO$D%v8hxU=2Iw}@0;HVL9)p?| zYarJO&^;A-?}TcRImSk2qE~*0W=b5`Bkn|8&@X!8L94fhttyzG`p_anDse%4L-DS^ z7g9S}1GGF$e>zy2@h&U_r2l#YU#&c@ttP3n8aS|IdFQX(m#)MED}_W!l1@FX6gu#i zb2h@=4Haof`r|R>W~#`UVz;Z)SCT)8Sn!D*_|`*4x`{Lq@3Wf2^~Z1?{#G&vkuY-UWTXf z+@f97!O!`ZpX7KO3vAmSXa?5t1yi4&CiTN`(w+!{8(e|| z;=676ym~gU0zQBf#hlJvxMS@Yhj@5Ty(}D7zTr;$HlhOc7CYnueJF`dSjixKBT*Fl z&)!;9`H=cXc}uTs);moIq74)`CeOYLG*!t9V5^YP~7EAwb<#d`Fqj@dsD~!jYc{Q>{P+H);3)uk_m# zM;A7Ywb>`vtj2Xn#npO!+qE>)cG&h_DEfBFR>l-i8;mu?D%hKT@>BEdTZK?bN)3`ZUjpZ*Xm+;tqsxU zJUqIJlgROA-Bx*&Cb%4;)$^K@Q}01GAo1(SjtOn+UwKu=PK2SyX$=N5^H66^Atq*4 z=_nN|1A~_4`vU)x&ELdc{JW=6x`OMk7kGaX(*35)gNz=fO(L2DAyy|%L@;JgE*J5< zk=P!xsJ+l!xpzL)x&DrRJF8yio{$ixgv7)MPQ|L>rTVPctqrj!goQct?qvZj1$(?G zW}k>@aD6?bi8%q7+sRrKX96vO5r&xU7bIH>1~e{kbRk-MmzPa7J~lVMk4?i)0qjT^ zmjst0E@;`J6cc?-G0YVC$JT{TEl=~t?MZ>`hkX=^r6al%*NO-mFIV;@gQPmEsh~&6 z%dxp;ux+p|rwd9u!h!3pI@_*^KTZe;g4Zzdnvj4g7-~IW|hYWyHa4x-Ivc5CNeo=1r aC*&ByDWh_!fo~7^fuA?ldHS1u=>GtK#lh 출처: [Complete setup for Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/complete-setup) + +설정 페이지에는 탭이 두 개입니다. 이 실습은 Azure 리소스와 지식 문서까지 연결해야 하므로 **Full setup** 탭을 사용합니다. + +![설정 페이지 상단에 More context. Better investigations. 제목과 그 아래 Quickstart, Full setup 두 개의 탭이 있고 Quickstart 탭이 파란색으로 선택되어 있다. 첫 카드에는 SRE Agent doesn't know anything about your app 이라는 경고 문구와 아직 채워지지 않은 회색 진행률 막대가 있다. 그 아래 Code 카드와 Logs 카드에는 각각 Recommended 배지와 오른쪽 끝 파란색 더하기 버튼이 있다.](../assets/official/portal-complete-setup-page.png) + +> 출처: [Complete setup for Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/complete-setup) + +## 연결할 원본 + +| 원본 | 이 실습에서 넣는 값 | 왜 필요한가 | +|---|---|---| +| Code | 이 저장소(`hellices/devguidesample`)와 브랜치 | 조사 결론이 코드와 최근 변경을 짚게 합니다 | +| Azure resources | `azd env get-value AZURE_RESOURCE_GROUP`이 출력한 리소스 그룹 | 메트릭·리소스 상태·Activity Log를 읽습니다 | +| Knowledge files | [runbooks/incident-response.md](../runbooks/incident-response.md) | 조사 순서와 금지 사항을 팀 규칙으로 강제합니다 | +| Incidents | Azure Monitor | 경고를 자동으로 받아 조사 스레드를 엽니다 | + +저장소는 [Connect source code](https://learn.microsoft.com/azure/sre-agent/connect-source-code)로 연결합니다. 이슈·PR 조작까지 맡기려면 [GitHub connector](https://learn.microsoft.com/azure/sre-agent/setup-github-connector)를 추가로 설정하되, 토큰 값은 포털 입력창에만 넣고 이 저장소의 어떤 파일에도 남기지 않습니다. + +## 역할 부여 + +Agent가 실습 리소스를 읽고 구독의 경고를 훑을 수 있어야 합니다. 범위를 넓히지 마세요. + +| 대상 | 역할 | 범위 | +|---|---|---| +| Agent 시스템 할당 ID, 사용자 할당 ID | Reader | 실습 리소스 그룹 | +| 같은 두 ID | Monitoring Contributor | 구독 | + +Monitoring Contributor는 [Azure Monitor 스캐너](https://learn.microsoft.com/azure/sre-agent/azure-monitor-alerts)가 요구하는 유일한 구독 범위 권한입니다. 할당 ID는 아래에서 기록해 두고 정리 단계에서 자동으로 제거합니다. + +## Builder > Incident platform 연결 + +1. 왼쪽에서 **Builder > Incident platform**을 엽니다. +2. 드롭다운에서 **Azure Monitor**를 고릅니다. +3. **Quickstart response plan** 토글은 끕니다. 다음 절에서 직접 만듭니다. +4. **Save**를 누르고 연결이 끝날 때까지 기다립니다. + +연결되면 오른쪽 위에 초록색 체크가 붙습니다. 경고는 몇 분 안에 흘러들기 시작합니다. + +![왼쪽 탐색의 Builder 메뉴가 펼쳐져 Agent Canvas, Skills, Incident response plans, Scheduled tasks, Plugins, Hooks, Connectors, Knowledge base 항목이 보이고 그중 Incident response plans가 선택되어 있다. 오른쪽 위에는 초록색 체크 아이콘과 함께 Azure Monitor is connected 문구가 있다. 표에는 계획 한 건이 있고 Status 열은 On, Autonomy level 열은 Autonomous로 표시되며, 위쪽에는 New incident response plan, Refresh, Delete, Turn off 버튼과 Severity equals All 필터가 있다.](../assets/official/portal-incident-response-plans-list.png) + +> 출처: [Tutorial: Automate incident response in Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/automate-incidents) + +## 응답 계획 만들기 + +1. **Builder > Incident response plans**에서 **New incident response plan**을 누릅니다. +2. 이름을 정하고 심각도는 실습 경고와 같은 **Sev2**를 포함하도록 고릅니다. +3. 필터 미리 보기를 지나 마지막 단계로 갑니다. +4. 자율 수준을 **Review**로 고르고 저장합니다. + +`Review`는 Agent가 진단은 스스로 하되 리소스 변경 전에 승인을 기다리는 모드입니다. 이 실습은 장애 주입과 복구를 스크립트가 통제하므로, 자율 모드를 고르면 Agent와 스크립트가 같은 리소스를 동시에 되돌리려 할 수 있습니다. + +![응답 계획 마법사의 세 번째 화면. 왼쪽 단계 목록에서 Set up incident filters와 Preview filter results는 초록색 체크로 완료되어 있고 3번 Save response plan이 진행 중이다. 오른쪽에는 Choose agent autonomy level for this handler 문구 아래 Review (Default)와 Autonomous 두 개의 라디오 버튼이 설명과 함께 있으며 Autonomous 쪽이 파란 점으로 켜져 있다. 그 아래 Turn on deep investigation 항목의 Run deep investigation autonomously 체크박스는 비어 있고, 화면 맨 아래에 Back, Save, Cancel 버튼이 있다.](../assets/official/portal-response-plan-autonomy-step.png) + +> 출처: [Tutorial: Automate incident response in Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/automate-incidents) + +## 설정 값을 azd 환경에 저장 + +스크립트가 읽는 값은 모두 이름·경로·리소스 ID뿐이며 비밀 값이 아닙니다. + +```bash +azd env set SRE_AGENT_NAME "<포털에 보이는 Agent 이름>" +azd env set SRE_AGENT_RESOURCE_ID "" +azd env set SRE_REPOSITORY_URL "https://github.com//" +azd env set SRE_REPOSITORY_BRANCH "main" +azd env set SRE_KNOWLEDGE_PATH "runbooks/incident-response.md" +``` + +토큰, 연결 문자열, 클라이언트 비밀은 azd 환경에도 저장하지 않습니다. 인증은 Agent의 관리 ID와 포털 OAuth 흐름이 처리합니다. + +## 근거 파일 만들기 + +`evidence/agent-setup.json`은 Git에서 제외되며, 정리 단계가 되돌릴 대상을 이 파일에서만 읽습니다. 값은 모두 식별자와 엔드포인트 URL입니다. + +```bash +cat > evidence/agent-setup.json <<'JSON' +{ + "agent_principal_id": "", + "agent_user_assigned_principal_id": "", + "agent_endpoint": "https://..azuresre.ai", + "monitoring_contributor_assignment_id": "/subscriptions//providers/Microsoft.Authorization/roleAssignments/", + "uami_monitoring_contributor_assignment_id": "/subscriptions//providers/Microsoft.Authorization/roleAssignments/" +} +JSON +``` + +## 점검과 승인 + +```bash +./scripts/lab.sh doctor +./scripts/lab.sh baseline +./scripts/lab.sh acknowledge agent-setup +``` + +`doctor`가 출력하는 네 줄은 언제나 `MANUAL`입니다. 저장소 연결, 지식 원본, incident platform, 응답 계획을 읽을 수 있는 공식 안정 API가 없기 때문입니다. 나머지 검사에 `FAIL`이 남아 있으면 먼저 해결합니다. + +`acknowledge agent-setup`은 설정 값을 출력한 뒤 표준 입력으로 정확히 `acknowledge`를 받아야 기록합니다. 어떤 환경 변수로도 대체할 수 없습니다. 값이 하나라도 다르면 그대로 중단하고 위 단계로 돌아가세요. + +## 실패했을 때 + +| 증상 | 조치 | +|---|---| +| `doctor`의 Reader 검사가 `FAIL` | 두 principal ID가 근거 파일과 같은지 확인하고 리소스 그룹에 Reader를 다시 부여합니다 | +| `baseline`이 telemetry 없음으로 종료 | 10분 더 기다린 뒤 다시 실행합니다. 계속 실패하면 `azd env get-value AZURE_CONTAINER_APP_FQDN`으로 앱을 직접 호출해 봅니다 | +| `acknowledge`가 기록되지 않음 | 입력한 단어가 정확한지, `azd env select`로 올바른 환경을 골랐는지 확인합니다 | +| 경고가 Agent에 도착하지 않음 | Monitoring Contributor 범위가 구독인지, 응답 계획이 `On`인지 확인합니다 | + +## 다음 단계 + +첫 장애를 주입합니다: [02-scenario-s1.md](02-scenario-s1.md) diff --git a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md new file mode 100644 index 0000000..49c63be --- /dev/null +++ b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md @@ -0,0 +1,90 @@ +# 02. S1 — HTTP 500 장애 + +주문 API가 500을 반환하도록 바꾸고, Azure Monitor Sev2 경고가 발생한 뒤 Agent가 원인을 짚어내는지 봅니다. + +## 시작 조건 + +- [01-agent-setup.md](01-agent-setup.md)를 마쳤고 `evidence/state.json`에 `baseline_passed`와 `agent_setup_acknowledged`가 기록되어 있습니다. +- 현재 활성 구독이 azd 환경의 구독과 같습니다. +- 진행 중인 다른 시나리오가 없습니다. + +조건이 하나라도 없으면 실행이 시작 전에 거부되고 무엇을 먼저 하라는 안내가 출력됩니다. + +## 실행 명령 + +```bash +cd monitor/sre-agent-event-lab +./scripts/lab.sh run s1 +``` + +한 번의 실행이 장애 주입 → 부하 → 경고 대기 → 복구 → 타임라인 저장까지 진행합니다. 경고가 발생하지 않으면 12분 뒤 실패로 기록하고 종료합니다. 스크립트를 중간에 끊어도 종료 트랩이 복구를 시도합니다. + +경고가 발생하고 복구까지 끝나면 조사 근거를 수집합니다. + +```bash +./scripts/lab.sh capture s1 +``` + +수집 대상 디렉터리는 방금 실행이 `evidence/state.json`에 기록한 값에서 정해지므로 경로를 직접 입력하지 않습니다. + +같은 UTC 구간의 KQL 근거를 따로 내보내려면 실행이 남긴 `timeline.json`의 시각을 그대로 넣어 호출합니다. + +```bash +./scripts/query-evidence.sh s1 evidence/s1-<타임스탬프> <시작 UTC> <끝 UTC> +``` + +## Azure에서 발생하는 변화 + +| 순서 | 변화 | +|---|---| +| 1 | Container App에 `FAILURE_MODE=http500` 환경 변수가 설정되어 새 revision이 만들어집니다 | +| 2 | `/api/orders`에 요청 120건(동시 4)이 들어가고 모두 HTTP 500을 받습니다 | +| 3 | Application Insights `requests`에 `resultCode == "500"` 레코드가 쌓입니다 | +| 4 | 5분 창의 500 응답이 10건을 넘으면 `alert-sre-lab-s1-http500`(Sev2)이 발생합니다 | +| 5 | 복구로 `FAILURE_MODE=none` revision이 다시 배포되고 경고가 자동 해제됩니다 | + +## SRE Agent에서 확인할 항목 + +`https://sre.azure.com`에서 새 스레드를 열고 다음을 확인합니다. + +- 경고 접수 시각과 스레드 생성 시각의 간격 +- 조사 계획이 업로드한 운영 문서의 순서를 따르는지 +- 영향 범위를 `/api/orders`로 좁혔는지, 앱 전체로 뭉뚱그리지 않았는지 +- 직접 원인으로 최근 revision의 환경 변수 변경을 지목했는지 +- 완화책을 제안만 하고 승인을 기다리는지(Review 모드) + +`capture`가 만드는 `assets/captures/s1/`의 PNG·GIF·Markdown이 같은 내용을 담습니다. + +## 성공·부분 성공·실패 판정 + +`capture`는 스레드의 마지막 상태를 그대로 기록합니다. + +| 기록된 상태 | 의미 | 다음 시나리오 | +|---|---|---| +| `conclusion` | Agent가 구조화된 결론을 냈습니다 | 열립니다 | +| `investigation-missing` | 스레드는 열렸지만 조사 단계가 없습니다 | 막힙니다 | +| `conclusion-missing` | 조사는 했지만 결론에 도달하지 못했습니다 | 막힙니다 | +| `thread-not-created` | 경고가 Agent에 도달하지 못했습니다 | 막힙니다 | + +성공은 `conclusion` 하나뿐입니다. 부분 성공은 결론이 나왔지만 내용이 얕은 경우이며, 이때도 상태는 `conclusion`이고 점수는 [05-results.md](05-results.md)의 채점에서 갈립니다. 빈 성공 화면을 만들지 않고 누락 상태와 마지막 확인 시각을 그대로 그림에 남깁니다. + +`thread-not-created`가 나오면 [01-agent-setup.md](01-agent-setup.md)의 incident platform과 응답 계획부터 다시 확인합니다. + +## 복구 확인 + +복구는 두 가지가 모두 확인된 뒤에만 기록됩니다. + +1. Container App의 활성 revision이 다시 정상입니다. +2. 이 실행이 발생시킨 경고가 Azure Monitor에서 `Resolved`가 됩니다. + +경고 해제는 최대 15분, 워크로드 정상화는 최대 10분까지 기다립니다. 둘 중 하나라도 시간 안에 확인되지 않으면 실행은 실패로 기록되고 S2는 계속 막힙니다. 실패한 실행은 원인을 고친 뒤 `./scripts/lab.sh run s1`을 다시 실행하면 새 시도로 이어집니다. + +```bash +azd env get-value AZURE_CONTAINER_APP_FQDN +``` + +로 얻은 FQDN에 `/api/orders`를 호출해 200이 돌아오는지 직접 확인할 수도 있습니다. + +## 다음 단계 + +지연 장애로 넘어갑니다: [03-scenario-s2.md](03-scenario-s2.md) diff --git a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md new file mode 100644 index 0000000..9428b7d --- /dev/null +++ b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md @@ -0,0 +1,56 @@ +# 03. S2 — 응답 지연 장애 + +주문 API가 느려지게 만들고, p95 지연 경고를 받은 Agent가 오류 없는 장애를 설명하는지 봅니다. 500이 하나도 없기 때문에 S1보다 근거를 고르기 어렵습니다. + +## 시작 조건 + +- [02-scenario-s1.md](02-scenario-s1.md)의 S1이 복구되고 캡처가 `conclusion`으로 끝났습니다. +- `evidence/state.json`에 `s1_recovered`와 `s1_captured`가 있습니다. +- 워크로드가 정상이고 S1 경고가 해제되어 있습니다. + +## 실행 명령 + +```bash +cd monitor/sre-agent-event-lab +./scripts/lab.sh run s2 +./scripts/lab.sh capture s2 +``` + +## Azure에서 발생하는 변화 + +| 순서 | 변화 | +|---|---| +| 1 | Container App에 `ORDER_DELAY_MS=4000`이 설정되어 새 revision이 만들어집니다 | +| 2 | `/api/orders`에 요청 90건(동시 8)이 들어가고 모두 200이지만 4초 안팎이 걸립니다 | +| 3 | Application Insights `requests`의 `duration`이 올라갑니다 | +| 4 | 5분 창의 p95 지연이 2000ms를 넘으면 `alert-sre-lab-s2-latency`(Sev2)가 발생합니다 | +| 5 | 복구로 `ORDER_DELAY_MS=0` revision이 배포되고 경고가 자동 해제됩니다 | + +## SRE Agent에서 확인할 항목 + +- 성공 응답만 있는 상황에서 지연을 근거로 삼는지, 오류 로그가 없다는 이유로 "이상 없음"이라고 답하지 않는지 +- p95 값과 정상 구간의 차이를 수치로 제시하는지 +- 원인을 최근 revision의 설정 변경으로 좁히는지, 일반적인 "리소스 부족"으로 뭉개지 않는지 +- 완화책이 되돌리기 가능한 최소 변경인지 + +## 성공·부분 성공·실패 판정 + +| 기록된 상태 | 판정 | +|---|---| +| `conclusion` | 성공. 결론 내용의 깊이는 채점에서 다시 나뉩니다 | +| `conclusion-missing` | 실패. 조사는 시작했지만 결론이 없습니다 | +| `investigation-missing` | 실패. 스레드만 열리고 조사 단계가 없습니다 | +| `thread-not-created` | 실패. 경고가 Agent에 도달하지 못했습니다 | + +지연 시나리오에서 흔한 부분 성공은 "느려졌다"까지만 말하고 어떤 변경 때문인지 짚지 못하는 결론입니다. 이 경우도 상태는 `conclusion`이므로, 직접 원인 항목의 점수로 구분합니다. + +## 복구 확인 + +1. 활성 revision이 정상이고 `/api/orders`가 다시 빠르게 응답합니다. +2. `alert-sre-lab-s2-latency`가 `Resolved`입니다. + +경고가 해제되지 않으면 부하가 남아 있는지, 새 revision으로 트래픽이 100% 넘어갔는지 확인합니다. 실패로 기록된 실행은 `./scripts/lab.sh run s2`를 다시 실행해 새 시도로 이어 갑니다. + +## 다음 단계 + +권한 장애로 넘어갑니다: [04-scenario-s3.md](04-scenario-s3.md) diff --git a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md new file mode 100644 index 0000000..01ae342 --- /dev/null +++ b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md @@ -0,0 +1,60 @@ +# 04. S3 — Blob 권한 제거 장애 + +워크로드의 Blob 읽기 권한을 지우고, 애플리케이션 코드가 아니라 권한이 원인임을 Agent가 구분해 내는지 봅니다. + +## 시작 조건 + +- [03-scenario-s2.md](03-scenario-s2.md)의 S2가 복구되고 캡처가 `conclusion`으로 끝났습니다. +- `evidence/state.json`에 `s2_recovered`와 `s2_captured`가 있습니다. +- 역할 할당을 만들고 지울 권한이 그대로 있습니다. + +## 실행 명령 + +```bash +cd monitor/sre-agent-event-lab +./scripts/lab.sh run s3 +./scripts/lab.sh capture s3 +``` + +## Azure에서 발생하는 변화 + +| 순서 | 변화 | +|---|---| +| 1 | 워크로드 관리 ID의 `Storage Blob Data Reader` 할당이 Blob 컨테이너 범위에서 삭제됩니다 | +| 2 | `/api/documents`에 요청 60건(동시 4)이 들어가고 모두 HTTP 503을 받습니다 | +| 3 | Application Insights `dependencies`에 Storage 대상 `resultCode == "403"`이 쌓입니다 | +| 4 | 5분 창의 403 의존성 실패가 5건을 넘으면 `alert-sre-lab-s3-storage-rbac`(Sev2)이 발생합니다 | +| 5 | 복구로 같은 이름·같은 범위의 역할 할당이 다시 만들어집니다 | + +삭제와 복구는 출력에 기록된 Blob 컨테이너 범위의 단일 역할에만 적용됩니다. 구독이나 리소스 그룹 범위의 다른 할당은 건드리지 않습니다. + +## SRE Agent에서 확인할 항목 + +- 앱이 돌려준 503과 실제 원인인 Storage 403을 구분하는지 +- 호출한 관리 ID, 대상 범위, 필요한 데이터 평면 역할을 각각 지목하는지 +- Activity Log의 역할 할당 삭제 기록을 근거로 인용하는지 +- 복구책으로 구독 범위 권한이 아니라 원래 범위의 최소 역할을 제안하는지 + +마지막 항목은 운영 문서가 명시적으로 요구하는 내용이라, 실습에서 제품이 문서를 실제로 따르는지 가장 잘 드러나는 지점입니다. + +## 성공·부분 성공·실패 판정 + +| 기록된 상태 | 판정 | +|---|---| +| `conclusion` | 성공. 권한 범위까지 짚었는지는 채점에서 확인합니다 | +| `conclusion-missing` | 실패. 결론에 도달하지 못했습니다 | +| `investigation-missing` | 실패. 조사 단계가 없습니다 | +| `thread-not-created` | 실패. 경고가 도달하지 않았습니다 | + +권한 시나리오의 전형적인 부분 성공은 "Storage 접근 실패"까지만 말하고 어떤 역할이 어느 범위에서 사라졌는지 밝히지 못하는 결론입니다. + +## 복구 확인 + +1. `Storage Blob Data Reader` 할당이 원래 Blob 컨테이너 범위에 다시 존재합니다. +2. `alert-sre-lab-s3-storage-rbac`가 `Resolved`입니다. + +역할 전파에는 몇 분이 걸릴 수 있습니다. `/api/documents`가 200을 돌려주는지 직접 호출해 확인하고, 실패로 기록되었다면 `./scripts/lab.sh run s3`으로 새 시도를 시작합니다. + +## 다음 단계 + +수집한 근거를 채점합니다: [05-results.md](05-results.md) diff --git a/monitor/sre-agent-event-lab/guides/05-results.md b/monitor/sre-agent-event-lab/guides/05-results.md new file mode 100644 index 0000000..6561f0c --- /dev/null +++ b/monitor/sre-agent-event-lab/guides/05-results.md @@ -0,0 +1,84 @@ +# 05. 결과 읽기와 채점 + +수집한 근거만으로 세 시나리오를 채점하고, 사람이 판단해야 하는 부분을 남김없이 드러내는 단계입니다. + +## 시작 조건 + +- 세 시나리오의 `run`과 `capture`가 끝났습니다. +- `evidence/state.json`에 `s3_captured`가 있습니다. +- 각 시나리오 디렉터리에 `normalized-timeline.json`이 있습니다. + +## 실행 명령 + +```bash +cd monitor/sre-agent-event-lab +./scripts/lab.sh score +``` + +`evidence/scorecard.json`과 `SCENARIOCRITERIONSTATUSPOINTSDETAIL` 표를 출력합니다. 종합 판정이 `FAIL`일 때만 종료 코드 1을 반환합니다. + +## 채점 기준 + +시나리오마다 10점입니다. + +| 항목 ID | 뜻 | 배점 | +|---|---|---:| +| `impact_scope` | 영향 범위를 특정했는가 | 2 | +| `direct_cause` | 직접 원인을 지목했는가 | 3 | +| `actual_evidence` | 실제 근거를 인용했는가 | 2 | +| `safe_minimum_mitigation` | 되돌리기 가능한 최소 완화책인가 | 2 | +| `uncertainty` | 불확실한 부분을 밝혔는가 | 1 | + +- Pass: 8점 이상 +- Partial: 5~7점 +- Fail: 4점 이하 + +## 사람이 채워야 하는 판정 + +점수를 주는 근거는 시나리오 디렉터리의 `conclusion-review.json`입니다. 항목마다 `{"met": true|false, "detail": "..."}`를 기록합니다. + +```json +{ + "impact_scope": { "met": true, "detail": "2번 메시지가 /api/orders만 영향으로 특정" }, + "direct_cause": { "met": false, "detail": "배포 변경이라고만 하고 어떤 설정인지 지목하지 못함" } +} +``` + +기록이 없는 항목은 `MANUAL`로 표시되고 **점수를 주지 않습니다**. 읽지 않은 결론에 점수를 주는 것보다, 아직 읽지 않았다고 말하는 편이 정확하기 때문입니다. + +## 결과 해석 + +| 출력 | 뜻 | 할 일 | +|---|---|---| +| 항목 `MANUAL` | 사람이 아직 판정하지 않음 | 스레드 결론을 읽고 `conclusion-review.json`에 기록한 뒤 다시 채점 | +| 시나리오 `FAIL` 0점 | 캡처가 결론에 이르지 못함 | 사유가 DETAIL에 남습니다. 해당 시나리오 문서의 실패 표를 참고 | +| 종합 `INCOMPLETE` | `MANUAL`이 남아 총점이 하한값 | 남은 판정을 채웁니다 | +| 종합 `PASS` | 모든 시나리오가 Partial 이상이고 두 개 이상 Pass | 결과를 정리하고 리소스를 지웁니다 | + +캡처가 `thread-not-created`, `investigation-missing`, `conclusion-missing`으로 끝난 시나리오는 모든 항목이 `FAIL` 0점입니다. 이때 점수가 낮은 것은 제품 판단이 아니라 근거가 없다는 사실의 기록입니다. + +## 남겨 둘 것 + +- `assets/captures/s1`, `s2`, `s3`의 PNG·GIF·Markdown은 커밋 대상입니다. +- `evidence/` 아래 원본 스냅샷과 `scorecard.json`은 Git에서 제외됩니다. 필요하면 별도로 보관하세요. +- 결론을 공유할 때는 구독 ID, 엔드포인트 FQDN, 토큰이 화면에 남지 않았는지 먼저 확인합니다. + +티켓과 이메일 초안 같은 운영 산출물은 정규화된 타임라인에서 다시 만들 수 있습니다. + +```bash +app/.venv/bin/python scripts/generate_notifications.py \ + --timeline evidence/s1-<타임스탬프>/normalized-timeline.json \ + --output-dir assets/notifications \ + --report-url validation-results.md +``` + +## 다음 단계 + +실습이 끝났으면 바로 정리합니다. 리소스를 남겨 두면 계속 과금됩니다. + +```bash +cd monitor/sre-agent-event-lab +azd down --purge +``` + +정리 훅이 무엇을 지우는지, 확인 프롬프트에서 취소하면 무엇이 남는지는 [README의 정리 절](../README.md)에 있습니다. diff --git a/monitor/sre-agent-event-lab/runbooks/incident-response.md b/monitor/sre-agent-event-lab/runbooks/incident-response.md index 3448048..55bd82e 100644 --- a/monitor/sre-agent-event-lab/runbooks/incident-response.md +++ b/monitor/sre-agent-event-lab/runbooks/incident-response.md @@ -2,7 +2,12 @@ ## Scope and safety -This runbook applies only to resources in `rg-sre-agent-event-lab-krc`. +This runbook applies only to the disposable lab resource group that `azd` +provisioned for the current environment: the group reported as +`AZURE_RESOURCE_GROUP` in the deployment outputs, tagged +`purpose=sre-agent-event-lab` together with the `azd-env-name` of that +environment. Any resource outside that group is out of scope, including +resources in other lab environments that carry the same purpose tag. - Investigate automatically, but do not execute a mitigation without approval. - Do not change resources, role assignments, alert rules, or traffic outside the lab resource group. diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_briefing_docs.py b/monitor/sre-agent-event-lab/scripts/tests/test_briefing_docs.py index ffff445..d507a29 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_briefing_docs.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_briefing_docs.py @@ -29,6 +29,17 @@ OFFICIAL_ASSETS = OFFICIAL_SVGS | OFFICIAL_PNGS OFFICIAL_ASSET_PREFIX = "sre-agent-event-lab/assets/official/" +# Screenshots the ordered walkthrough under `guides/` renders. They share the +# `assets/official/` directory because they come from the same Learn articles, +# but they are not part of the briefing's selected set: `guides/` owns them and +# `scripts/tests/test_lab_guides.py` checks their captions and alt text. +GUIDE_SCREENSHOT_PNGS = { + "portal-setup-status-bar.png", + "portal-complete-setup-page.png", + "portal-incident-response-plans-list.png", + "portal-response-plan-autonomy-step.png", +} + def windows_after(text: str, anchor: str, size: int = 500) -> list: """Return a text window following each occurrence of `anchor`. @@ -681,7 +692,12 @@ def test_official_asset_set_has_16_selected_files(): ) assert len(OFFICIAL_ASSETS) == 16 - assert {path.name for path in asset_dir.glob("*")} == OFFICIAL_ASSETS + # The directory also stores the guide screenshots; nothing else may + # accumulate here, and the two sets stay disjoint. + assert OFFICIAL_ASSETS & GUIDE_SCREENSHOT_PNGS == set() + assert {path.name for path in asset_dir.glob("*")} == ( + OFFICIAL_ASSETS | GUIDE_SCREENSHOT_PNGS + ) def test_official_sre_agent_svgs_are_stored_locally(): @@ -709,8 +725,10 @@ def test_official_sre_agent_pngs_are_stored_locally(): / "official" ) - assert {path.name for path in asset_dir.glob("*.png")} == OFFICIAL_PNGS - for name in OFFICIAL_PNGS: + assert {path.name for path in asset_dir.glob("*.png")} == ( + OFFICIAL_PNGS | GUIDE_SCREENSHOT_PNGS + ) + for name in OFFICIAL_PNGS | GUIDE_SCREENSHOT_PNGS: header = (asset_dir / name).read_bytes()[:8] assert header == b"\x89PNG\r\n\x1a\n", name diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py new file mode 100644 index 0000000..e04c406 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -0,0 +1,498 @@ +"""Contract tests for the ordered lab walkthrough. + +The README is the quickstart an operator reads first, and `guides/` holds +the step-by-step documents it hands off to. These tests check behaviour a +reader depends on -- that the commands are the ones `lab.sh` really +accepts, in the order `lab_state.py` really enforces; that every path, +link and screenshot resolves; and that nothing here asks anyone to paste a +credential into a file or an environment variable. +""" +import re +from pathlib import Path + + +REPO_ROOT = Path(__file__).parents[4] +LAB_ROOT = REPO_ROOT / "monitor" / "sre-agent-event-lab" +README = LAB_ROOT / "README.md" +GUIDES = LAB_ROOT / "guides" +OFFICIAL_ASSETS = LAB_ROOT / "assets" / "official" +RUNBOOK = LAB_ROOT / "runbooks" / "incident-response.md" +LAB_SH = LAB_ROOT / "scripts" / "lab.sh" + +GUIDE_NAMES = ( + "01-agent-setup.md", + "02-scenario-s1.md", + "03-scenario-s2.md", + "04-scenario-s3.md", + "05-results.md", +) + +SCENARIO_GUIDES = { + "02-scenario-s1.md": "s1", + "03-scenario-s2.md": "s2", + "04-scenario-s3.md": "s3", +} + +# The screenshots selected from the live Learn articles, each tied to one +# portal action an operator performs by hand. Anything the lab can prove +# with its own captured evidence is deliberately not copied here. +GUIDE_SCREENSHOTS = { + "portal-setup-status-bar.png": ( + "https://learn.microsoft.com/azure/sre-agent/complete-setup" + ), + "portal-complete-setup-page.png": ( + "https://learn.microsoft.com/azure/sre-agent/complete-setup" + ), + "portal-incident-response-plans-list.png": ( + "https://learn.microsoft.com/azure/sre-agent/automate-incidents" + ), + "portal-response-plan-autonomy-step.png": ( + "https://learn.microsoft.com/azure/sre-agent/automate-incidents" + ), +} + +# Every keyword is a string or state actually visible in the downloaded +# screenshot, so alt text that drifts from the picture fails here. +SCREENSHOT_ALT_KEYWORDS = { + "portal-setup-status-bar.png": ( + "6 sources not configured", + "Complete setup", + "Code", + "Logs", + "Deployments", + "Incidents", + "Azure resources", + "Knowledge files", + "Builder", + ), + "portal-complete-setup-page.png": ( + "Quickstart", + "Full setup", + "Code", + "Logs", + "Recommended", + ), + "portal-incident-response-plans-list.png": ( + "Builder", + "Incident response plans", + "Azure Monitor is connected", + "Autonomy level", + "Autonomous", + "On", + ), + "portal-response-plan-autonomy-step.png": ( + "Review (Default)", + "Autonomous", + "Save response plan", + "Save", + ), +} + +# Claims the picture does not support. The two response-plan screenshots +# both show `Autonomous`, and this lab runs in `Review`: alt text may never +# describe them as showing the mode the lab asks for. +FORBIDDEN_ALT_CLAIMS = { + "portal-incident-response-plans-list.png": ("Review 모드로", "Review로 표시"), + "portal-response-plan-autonomy-step.png": ("Review가 선택", "Review를 선택한 상태"), + "portal-complete-setup-page.png": ("Azure resources", "Knowledge files"), +} + +SECRET_ASSIGNMENTS = ( + "GITHUB_PAT=", + "OAUTH_TOKEN=", + "CLIENT_SECRET=", + "ACCESS_TOKEN=", + "AZURE_CLIENT_SECRET=", +) + +FIXED_SUBSCRIPTION_ID = "95933ae5-0201-4a21-a1fc-8051a7437982" +FIXED_RESOURCE_GROUP = "rg-sre-agent-event-lab-krc" + + +def guide_paths(): + return [GUIDES / name for name in GUIDE_NAMES] + + +def all_docs(): + return [README] + guide_paths() + + +def joined_docs() -> str: + return "\n".join(path.read_text() for path in all_docs()) + + +def images(text: str): + """(alt, target) for every rendered image, Markdown or HTML.""" + markdown = re.findall(r"!\[([^\]]*)\]\(([^)]+)\)", text) + html = [("", target) for target in re.findall(r"]*\ssrc=[\"']([^\"']+)", text)] + return markdown + html + + +def body_sentences(markdown: str, minimum_length: int = 30): + """Substantive body sentences, with code, images and captions removed.""" + markdown = re.sub(r"```.*?```", "", markdown, flags=re.DOTALL) + markdown = re.sub(r"!\[[^\]]*\]\([^)]*\)", "", markdown) + + fragments = [] + for line in markdown.splitlines(): + line = line.strip() + if not line or line.startswith(("#", ">")): + continue + if re.fullmatch(r"\|[\s:\-|]+\|", line): + continue + cells = line.strip("|").split("|") if line.startswith("|") else [line] + for cell in cells: + cell = re.sub(r"^[-*+]\s+", "", cell.strip()) + cell = re.sub(r"^\d+\.\s+", "", cell) + cell = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", cell) + cell = cell.replace("**", "").replace("`", "").strip() + if cell: + fragments.append(cell) + + sentences = [] + for fragment in fragments: + for sentence in re.split(r"(?<=다\.)\s+", fragment): + sentence = re.sub(r"\s+", " ", sentence).strip() + if len(sentence) >= minimum_length and re.search(r"(다|요)\.$", sentence): + sentences.append(sentence) + return sentences + + +def blocks(text: str): + return [block for block in re.split(r"\n\s*\n", text) if block.strip()] + + +# --- README quickstart --------------------------------------------------- + + +def test_readme_is_azd_first_and_ordered(): + text = README.read_text() + commands = [ + "azd env new", + "azd up", + "./scripts/lab.sh doctor", + "./scripts/lab.sh baseline", + "./scripts/lab.sh acknowledge agent-setup", + "./scripts/lab.sh run s1", + "./scripts/lab.sh capture s1", + "./scripts/lab.sh run s2", + "./scripts/lab.sh capture s2", + "./scripts/lab.sh run s3", + "./scripts/lab.sh capture s3", + "./scripts/lab.sh score", + "azd down --purge", + ] + positions = [text.index(command) for command in commands] + assert positions == sorted(positions) + + +def test_readme_warns_about_cost_and_teardown_before_the_first_azure_command(): + text = README.read_text() + + warning = re.search(r"(과금|비용)", text) + assert warning, "the README must state that the lab bills real resources" + assert warning.start() < text.index("azd up") + assert text.index("azd down --purge") > text.index("azd up") + for driver in ("Container Apps", "Log Analytics", "Azure SRE Agent"): + assert driver in text, driver + + +def test_readme_is_a_quickstart_not_the_full_walkthrough(): + """Scenario, capture and scoring detail belongs in the guides.""" + text = README.read_text() + + assert len(text.splitlines()) <= 200, "README is no longer a quickstart" + for moved in ("impact_scope", "conclusion-review.json", "FAILURE_MODE=http500"): + assert moved not in text, moved + assert "impact_scope" in (GUIDES / "05-results.md").read_text() + + +def test_readme_links_every_numbered_guide_in_order(): + text = README.read_text() + + positions = [text.index("guides/{0}".format(name)) for name in GUIDE_NAMES] + assert positions == sorted(positions) + + +def test_readme_troubleshooting_index_routes_to_doctor_and_guides(): + text = README.read_text() + heading = "## 문제 해결" + + assert heading in text + section = text.split(heading, 1)[1] + assert "lab.sh doctor" in section + assert "guides/" in section + + +def test_manual_steps_distinguish_product_path_from_old_bridge(): + text = README.read_text() + assert "Azure Monitor incident platform" in text + assert "기본 실습에는 Logic App bridge를 배포하지 않습니다" in text + + +def test_logic_app_bridge_is_only_described_as_legacy(): + for path in all_docs(): + for block in blocks(path.read_text()): + if "Logic App" not in block: + continue + assert "레거시" in block, (path.name, block) + + +def test_incident_platform_path_is_the_documented_default(): + setup = (GUIDES / "01-agent-setup.md").read_text() + + assert "Builder > Incident platform" in setup + assert "Azure Monitor" in setup + plan_index = setup.index("Review") + assert setup.index("Builder > Incident platform") < plan_index + + +# --- guide structure ----------------------------------------------------- + + +def test_guides_directory_holds_exactly_the_five_numbered_guides(): + assert {path.name for path in GUIDES.glob("*.md")} == set(GUIDE_NAMES) + + +def test_every_guide_opens_with_prerequisites_and_closes_with_a_next_step(): + for path in guide_paths(): + text = path.read_text() + assert "## 시작 조건" in text, path.name + assert "## 다음 단계" in text, path.name + assert text.index("## 시작 조건") < text.index("## 다음 단계"), path.name + + +def test_scenario_guides_use_the_required_section_order(): + required = [ + "## 시작 조건", + "## 실행 명령", + "## Azure에서 발생하는 변화", + "## SRE Agent에서 확인할 항목", + "## 성공·부분 성공·실패 판정", + "## 복구 확인", + "## 다음 단계", + ] + for name in SCENARIO_GUIDES: + text = (GUIDES / name).read_text() + positions = [text.index(heading) for heading in required] + assert positions == sorted(positions), name + + +def test_each_guide_hands_off_to_the_next_document(): + for index, name in enumerate(GUIDE_NAMES[:-1]): + section = (GUIDES / name).read_text().split("## 다음 단계", 1)[1] + assert GUIDE_NAMES[index + 1] in section, name + + final = (GUIDES / GUIDE_NAMES[-1]).read_text().split("## 다음 단계", 1)[1] + assert "azd down --purge" in final + + +def test_scenario_guides_name_the_injected_change_and_its_alert_rule(): + expected = { + "02-scenario-s1.md": ("FAILURE_MODE=http500", "alert-sre-lab-s1-http500", "500"), + "03-scenario-s2.md": ("ORDER_DELAY_MS=4000", "alert-sre-lab-s2-latency", "p95"), + "04-scenario-s3.md": ( + "Storage Blob Data Reader", + "alert-sre-lab-s3-storage-rbac", + "403", + ), + } + for name, markers in expected.items(): + text = (GUIDES / name).read_text() + for marker in markers: + assert marker in text, (name, marker) + + +def test_scenario_guides_judge_success_partial_and_failure_with_recovery(): + for name, scenario in SCENARIO_GUIDES.items(): + text = (GUIDES / name).read_text() + verdict = text.split("## 성공·부분 성공·실패 판정", 1)[1].split("\n## ", 1)[0] + for state in ("conclusion", "thread-not-created", "investigation-missing", "conclusion-missing"): + assert state in verdict, (name, state) + + recovery = text.split("## 복구 확인", 1)[1].split("\n## ", 1)[0] + assert "Resolved" in recovery, name + assert "state.json" in text, name + assert "./scripts/lab.sh run {0}".format(scenario) in text, name + assert "./scripts/lab.sh capture {0}".format(scenario) in text, name + + +def test_results_guide_documents_the_scoring_thresholds_and_manual_gap(): + text = (GUIDES / "05-results.md").read_text() + + for marker in ("8", "5", "MANUAL", "INCOMPLETE", "scorecard.json"): + assert marker in text, marker + for criterion in ( + "impact_scope", + "direct_cause", + "actual_evidence", + "safe_minimum_mitigation", + "uncertainty", + ): + assert criterion in text, criterion + + +# --- commands match the scripts ----------------------------------------- + + +def test_documented_lab_commands_exist_in_lab_sh(): + documented = set(re.findall(r"lab\.sh\s+([a-z-]+)", joined_docs())) + supported = set(re.findall(r"^\s{2}([a-z-]+)\)", LAB_SH.read_text(), re.MULTILINE)) + + assert documented, "no lab.sh commands are documented" + assert documented <= supported, sorted(documented - supported) + + +def test_agent_setup_guide_matches_the_interactive_acknowledge_contract(): + text = (GUIDES / "01-agent-setup.md").read_text() + + assert "./scripts/lab.sh acknowledge agent-setup" in text + assert "acknowledge" in text + assert "표준 입력" in text or "stdin" in text + assert re.search(r"환경 변수[^.]{0,60}(대체할 수 없|불가)", text), ( + "the guide must say no environment variable can replace the typed word" + ) + + +def test_agent_setup_guide_offers_azd_env_set_without_storing_secrets(): + text = (GUIDES / "01-agent-setup.md").read_text() + + for setting in ( + "azd env set SRE_AGENT_NAME", + "azd env set SRE_AGENT_RESOURCE_ID", + "azd env set SRE_REPOSITORY_URL", + "azd env set SRE_KNOWLEDGE_PATH", + ): + assert setting in text, setting + assert "evidence/agent-setup.json" in text + for key in ("agent_principal_id", "agent_user_assigned_principal_id", "agent_endpoint"): + assert key in text, key + + +def test_guides_do_not_request_secrets_in_environment(): + text = "\n".join(path.read_text() for path in GUIDES.glob("*.md")) + for forbidden in ("GITHUB_PAT=", "OAUTH_TOKEN=", "CLIENT_SECRET="): + assert forbidden not in text + + +def test_docs_never_show_a_credential_value(): + text = joined_docs() + + for forbidden in SECRET_ASSIGNMENTS: + assert forbidden not in text, forbidden + for pattern in (r"ghp_[A-Za-z0-9]", r"github_pat_", r"\bsig=", r"--password\b"): + assert not re.search(pattern, text), pattern + + +def test_docs_do_not_pin_the_original_subscription_or_resource_group(): + for path in all_docs() + [RUNBOOK]: + text = path.read_text() + assert FIXED_SUBSCRIPTION_ID not in text, path.name + assert FIXED_RESOURCE_GROUP not in text, path.name + + +def test_runbook_scopes_itself_to_the_provisioned_resource_group(): + text = RUNBOOK.read_text() + + assert "AZURE_RESOURCE_GROUP" in text + assert "purpose=sre-agent-event-lab" in text + + +# --- links and screenshots ---------------------------------------------- + + +def test_every_relative_link_in_the_walkthrough_resolves(): + checked = 0 + for path in all_docs(): + targets = re.findall(r"\]\((?!https?://|mailto:)([^)#]+)", path.read_text()) + for target in targets: + checked += 1 + assert (path.parent / target).resolve().exists(), (path.name, target) + assert checked + + +def test_guides_render_only_local_official_screenshots(): + for path in guide_paths(): + for alt, target in images(path.read_text()): + assert not target.startswith(("http://", "https://")), target + assert target.startswith("../assets/official/"), target + resolved = (path.parent / target).resolve() + assert resolved.is_file(), target + assert resolved.read_bytes()[:8] == b"\x89PNG\r\n\x1a\n", target + assert alt.strip(), target + + +def test_selected_screenshots_are_stored_and_referenced_exactly_once(): + rendered = [] + for path in guide_paths(): + rendered.extend(target.rsplit("/", 1)[-1] for _, target in images(path.read_text())) + + assert sorted(rendered) == sorted(GUIDE_SCREENSHOTS), rendered + for name in GUIDE_SCREENSHOTS: + assert (OFFICIAL_ASSETS / name).is_file(), name + + +def test_no_result_screenshots_are_copied_from_the_tutorial(): + """Investigation results are shown with this lab's own captures.""" + stored = {path.name for path in OFFICIAL_ASSETS.glob("*")} + + for tutorial_only in ( + "incident-completed.png", + "incident-full-page-top.png", + "incident-full-page-code-fix.png", + "response-plan-step-1.png", + ): + assert tutorial_only not in stored, tutorial_only + + +def test_every_screenshot_names_its_learn_source_next_to_the_image(): + for path in guide_paths(): + lines = path.read_text().splitlines() + for index, line in enumerate(lines): + match = re.match(r"!\[[^\]]*\]\(\.\./assets/official/([^)]+)\)", line.strip()) + if not match: + continue + name = match.group(1) + caption = "\n".join(lines[index + 1 : index + 4]) + assert caption.lstrip().startswith(">"), name + assert "출처" in caption, name + assert GUIDE_SCREENSHOTS[name] in caption, name + + +def test_screenshot_alt_text_describes_the_captured_screen_in_korean(): + alts = {} + for path in guide_paths(): + for alt, target in images(path.read_text()): + alts[target.rsplit("/", 1)[-1]] = alt + + assert set(alts) == set(GUIDE_SCREENSHOTS) + for name, keywords in SCREENSHOT_ALT_KEYWORDS.items(): + alt = alts[name] + assert len(alt) >= 60, (name, len(alt)) + assert re.search(r"[가-힣]", alt), name + for keyword in keywords: + assert keyword in alt, (name, keyword) + + +def test_screenshot_alt_text_does_not_claim_what_the_picture_lacks(): + alts = {} + for path in guide_paths(): + for alt, target in images(path.read_text()): + alts[target.rsplit("/", 1)[-1]] = alt + + for name, forbidden in FORBIDDEN_ALT_CLAIMS.items(): + for claim in forbidden: + assert claim not in alts[name], (name, claim) + + +def test_screenshot_alt_text_is_not_repeated_as_body_prose(): + for path in guide_paths(): + text = path.read_text() + body = re.sub(r"\s+", " ", re.sub(r"!\[[^\]]*\]\([^)]*\)", "", text)) + repeated = [ + sentence + for alt, _ in images(text) + for sentence in body_sentences(alt) + if sentence in body + ] + assert repeated == [], (path.name, repeated) From 9ba6011862b7b9cf571bc41b3b73c90c3c98e696 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 20:14:47 +0900 Subject: [PATCH 13/26] fix(sre-lab): correct Task 6 review findings (venv setup, docs accuracy) Fixes six findings from the Task 6 review (review-f80be21..f14edd7.diff), all with RED/GREEN tests: 1. The documented azd-first flow never prepared app/.venv: capture-scenario.sh hard-requires app/.venv/bin/python but nothing created it. Added a new, idempotent scripts/setup-venv.sh (uv venv --python ">=3.10" --allow-existing plus uv pip install -r requirements-dev.txt, verifying Pillow importability), invoked from the postprovision hook before any Azure CLI call. uv is mandatory, not merely preferred (this lab's proxy is configured for uv, not public PyPI), so there is no silent pip fallback -- missing uv is an actionable failure. Because postprovision runs after azd provision has already created cloud resources, every failure message states the exact rerun command (azd hooks run postprovision, or ./scripts/setup-venv.sh directly). doctor.sh gained a "Python environment" check (venv + Pillow readiness) and capture-scenario.sh gained its own Pillow-importability precondition, both pointing at the same rerun command. 2. Appended a correction section to the (gitignored) task-6-report.md documenting these inaccuracies, per its own "historical artifact" status, rather than rewriting the report in place. 3. Restored the verified https://azuresre.dev audience fact (deleted, not relocated, by the Task 6 README rewrite) into validation-results.md's historical bridge section, explicitly framed as legacy record for the non-default Logic App bridge. 4. Fixed README and guides/05-results.md cwd instructions: each document now uses exactly one cd, so every command block in it is runnable sequentially in one shell from that single working directory. 5. Reworded the README's scorecard row from "10 point max" (ambiguous as the overall max) to "10 per scenario, 30 overall", matching score.py's real MAX_POINTS (10 per scenario, 30 overall). 6. Fixed two accuracy issues in the numbered guides: guide 02's start conditions claimed an unenforced concurrency lock that lab_state.py's own docstring says does not exist; guide 05 ran the stdlib-only generate_notifications.py through app/.venv/bin/python, implying a Pillow dependency it does not have. Tests: 414 passed (scripts/tests + infra/tests + app/tests). bash -n and az bicep build both clean. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 21 +- .../guides/02-scenario-s1.md | 3 +- .../sre-agent-event-lab/guides/05-results.md | 5 +- .../scripts/azd-postprovision.sh | 10 + .../scripts/capture-scenario.sh | 6 + monitor/sre-agent-event-lab/scripts/doctor.sh | 15 ++ .../sre-agent-event-lab/scripts/setup-venv.sh | 73 +++++++ .../scripts/tests/doctor_harness.py | 43 +++- .../scripts/tests/lab_script_harness.py | 25 ++- .../scripts/tests/test_azd_hooks.py | 56 +++++ .../scripts/tests/test_doctor.py | 40 ++++ .../scripts/tests/test_lab_guides.py | 64 ++++++ .../scripts/tests/test_lab_scripts.py | 31 +++ .../scripts/tests/test_setup_venv.py | 201 ++++++++++++++++++ .../sre-agent-event-lab/validation-results.md | 2 + 15 files changed, 572 insertions(+), 23 deletions(-) create mode 100755 monitor/sre-agent-event-lab/scripts/setup-venv.sh create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index a2b03b5..e92aa8d 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -10,7 +10,7 @@ Azure Container Apps에 장애를 세 번 주입하고, Azure Monitor 경고를 |---|---| | 시나리오별 Agent 조사 타임라인(PNG/GIF/Markdown) | `assets/captures/s1`, `s2`, `s3` | | 원본 API 근거와 실행 상태 | `evidence/`(Git 제외) | -| 10점 만점 채점 결과 | `evidence/scorecard.json` | +| 시나리오별 10점, 종합 30점 만점 채점 결과 | `evidence/scorecard.json` | ## 비용과 안전 경계 @@ -41,16 +41,20 @@ Azure SRE Agent는 이 실습이 만들지 않습니다. 미리 만들어 둔 Ag - Azure SRE Agent를 만들 수 있는 [지원 지역](https://learn.microsoft.com/azure/sre-agent/supported-regions) 접근 권한 - 브라우저에서 `https://sre.azure.com` 및 `*.azuresre.ai` 접근 - Agent에 연결할 GitHub 저장소 권한 +- [`uv`](https://docs.astral.sh/uv/getting-started/installation/) — `app/.venv`를 만드는 `scripts/setup-venv.sh`가 이 도구만 사용하며, 사내 프록시로 구성된 `uv`의 인덱스 설정을 그대로 씁니다. 공개 PyPI로 우회하는 `pip` 폴백은 없습니다. + +이 실습의 모든 명령은 아래에서 한 번만 진입하는 이 디렉터리를 기준으로 합니다. + +```bash +cd monitor/sre-agent-event-lab +``` 로컬 검증만 먼저 해 보려면 다음을 실행합니다. ```bash -cd monitor/sre-agent-event-lab/app -python3 -m venv .venv -.venv/bin/pip install -r requirements-dev.txt -.venv/bin/python -m pytest -q +./scripts/setup-venv.sh +app/.venv/bin/python -m pytest app -q -cd .. bash -n scripts/*.sh az bicep build --file infra/main.bicep --stdout >/dev/null ``` @@ -58,7 +62,6 @@ az bicep build --file infra/main.bicep --stdout >/dev/null ## azd 환경 만들기 ```bash -cd monitor/sre-agent-event-lab azd env new sre-event-lab --location koreacentral ``` @@ -75,6 +78,8 @@ azd up 2>&1 | tee evidence/deploy.log Bicep provision → ACR 클라우드 빌드 → Container App 이미지 교체 순서로 진행되며 로컬 Docker는 필요 없습니다. 처음에는 공개 placeholder 이미지가 80 포트로 뜨고, postprovision hook이 ingress를 8000으로 옮긴 뒤 실습 이미지로 교체합니다. +같은 postprovision hook이 `scripts/setup-venv.sh`로 `app/.venv`도 함께 준비합니다(`uv venv` + `uv pip install -r requirements-dev.txt`). 이 단계가 실패해도 클라우드 리소스는 이미 만들어진 뒤이므로 다시 `azd provision`부터 할 필요는 없습니다: 안내된 명령(`azd hooks run postprovision` 또는 `./scripts/setup-venv.sh`)만 다시 실행하면 됩니다. + 성공 조건은 provision 성공, 활성 revision `Healthy`, `/healthz` HTTP 200 세 가지입니다. ## Azure SRE Agent 설정 @@ -91,7 +96,7 @@ Bicep provision → ACR 클라우드 빌드 → Container App 이미지 교체 ./scripts/lab.sh acknowledge agent-setup ``` -`doctor`는 `CHECKSTATUSDETAIL` 한 줄씩 출력하고 `FAIL`이 하나라도 있으면 종료 코드 1을 반환합니다. 저장소 연결, 지식 원본, incident platform, 응답 계획은 공식 안정 API로 읽을 수 없어 항상 `MANUAL`입니다. +`doctor`는 `CHECKSTATUSDETAIL` 한 줄씩 출력하고 `FAIL`이 하나라도 있으면 종료 코드 1을 반환합니다. 저장소 연결, 지식 원본, incident platform, 응답 계획은 공식 안정 API로 읽을 수 없어 항상 `MANUAL`입니다. `Python environment` 행은 `app/.venv`와 Pillow가 캡처(`capture-scenario.sh`)에 쓸 준비가 됐는지 확인하며, `FAIL`이면 `./scripts/setup-venv.sh`를 다시 실행하라고 안내합니다. `baseline`은 정상 부하를 넣고 Application Insights에 두 요청 종류가 모두 보일 때까지 최대 10분 기다립니다. `acknowledge agent-setup`은 대화형이며, 설정 값을 출력한 뒤 표준 입력으로 정확히 `acknowledge`를 입력해야 기록됩니다. diff --git a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md index 49c63be..73403b5 100644 --- a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md +++ b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md @@ -6,7 +6,8 @@ - [01-agent-setup.md](01-agent-setup.md)를 마쳤고 `evidence/state.json`에 `baseline_passed`와 `agent_setup_acknowledged`가 기록되어 있습니다. - 현재 활성 구독이 azd 환경의 구독과 같습니다. -- 진행 중인 다른 시나리오가 없습니다. + +이 두 가지만 `evidence/state.json`을 통해 실제로 강제됩니다. `state.json`에는 동시 실행을 막는 잠금이 없으므로(1인 운영자 전제), 다른 시나리오를 동시에 실행하지 않는 것은 운영자가 직접 지켜야 하는 규칙입니다. 조건이 하나라도 없으면 실행이 시작 전에 거부되고 무엇을 먼저 하라는 안내가 출력됩니다. diff --git a/monitor/sre-agent-event-lab/guides/05-results.md b/monitor/sre-agent-event-lab/guides/05-results.md index 6561f0c..becd808 100644 --- a/monitor/sre-agent-event-lab/guides/05-results.md +++ b/monitor/sre-agent-event-lab/guides/05-results.md @@ -63,10 +63,10 @@ cd monitor/sre-agent-event-lab - `evidence/` 아래 원본 스냅샷과 `scorecard.json`은 Git에서 제외됩니다. 필요하면 별도로 보관하세요. - 결론을 공유할 때는 구독 ID, 엔드포인트 FQDN, 토큰이 화면에 남지 않았는지 먼저 확인합니다. -티켓과 이메일 초안 같은 운영 산출물은 정규화된 타임라인에서 다시 만들 수 있습니다. +티켓과 이메일 초안 같은 운영 산출물은 정규화된 타임라인에서 다시 만들 수 있습니다. `generate_notifications.py`는 표준 라이브러리만 사용하므로(Pillow가 필요한 `render_capture.py`와 달리) `app/.venv` 없이 시스템 `python3`로 바로 실행합니다. ```bash -app/.venv/bin/python scripts/generate_notifications.py \ +python3 scripts/generate_notifications.py \ --timeline evidence/s1-<타임스탬프>/normalized-timeline.json \ --output-dir assets/notifications \ --report-url validation-results.md @@ -77,7 +77,6 @@ app/.venv/bin/python scripts/generate_notifications.py \ 실습이 끝났으면 바로 정리합니다. 리소스를 남겨 두면 계속 과금됩니다. ```bash -cd monitor/sre-agent-event-lab azd down --purge ``` diff --git a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh index 5cc4259..86e1582 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh @@ -17,6 +17,16 @@ for command_name in az azd curl; do } done +# `app/.venv` is not read by this hook itself, but this is the one place in +# the documented azd-first flow (`azd up`) that always runs once +# provisioning succeeds, so it is the natural place to make the local +# Python environment `capture-scenario.sh` and the notification step need +# ready before an operator ever reaches them. `setup-venv.sh` is its own +# idempotent script (uv-only, no pip fallback) so it can also be re-run by +# hand -- see its own actionable error output -- without repeating anything +# below. +"${SCRIPT_DIR}/setup-venv.sh" + # Every value below comes from the current azd environment, which azd refreshes # from the deployment outputs before running this hook. : "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set by azd before running this hook}" diff --git a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh index 3ecda6e..f0cdad0 100755 --- a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh @@ -41,6 +41,12 @@ if [[ ! -f "${TIMELINE_FILE}" ]]; then fi if [[ ! -x "${PYTHON}" ]]; then echo "Missing Python environment: ${PYTHON}" >&2 + echo "Cloud resources are already deployed; only this local step needs to be retried. Re-run: ./scripts/setup-venv.sh" >&2 + exit 1 +fi +if ! "${PYTHON}" -c "import PIL" >/dev/null 2>&1; then + echo "Python environment at ${PYTHON} is missing Pillow (PIL), which render_capture.py needs." >&2 + echo "Cloud resources are already deployed; only this local step needs to be retried. Re-run: ./scripts/setup-venv.sh" >&2 exit 1 fi diff --git a/monitor/sre-agent-event-lab/scripts/doctor.sh b/monitor/sre-agent-event-lab/scripts/doctor.sh index 4cb9ab2..77e9725 100755 --- a/monitor/sre-agent-event-lab/scripts/doctor.sh +++ b/monitor/sre-agent-event-lab/scripts/doctor.sh @@ -44,6 +44,21 @@ else report "Required commands" FAIL "Install missing commands: ${MISSING_COMMANDS[*]}." fi +# Python environment (app/.venv + Pillow) ------------------------------------- +# The documented azd-first flow prepares `app/.venv` from the `postprovision` +# hook (`scripts/setup-venv.sh`), before any scenario is captured. This check +# reports whether that step actually finished -- Pillow importable, not just +# a venv directory present -- so a partially-run or pre-uv-created venv is +# caught here rather than surfacing later as `capture-scenario.sh`'s "Missing +# Python environment" failure. This is a local precondition, independent of +# Azure reachability, so it neither reads AZURE_SAFE nor sets it. +VENV_PYTHON="${SCRIPT_DIR}/../app/.venv/bin/python" +if [[ -x "${VENV_PYTHON}" ]] && "${VENV_PYTHON}" -c "import PIL" >/dev/null 2>&1; then + report "Python environment" PASS "app/.venv is ready (Pillow importable) for capture-scenario.sh." +else + report "Python environment" FAIL "app/.venv is missing or incomplete (Pillow not importable). Run: ./scripts/setup-venv.sh" +fi + # Log Analytics CLI extension ------------------------------------------------ # `az monitor log-analytics query` -- the only read behind the telemetry # check and behind `baseline.sh`/`query-evidence.sh` -- ships in an extension diff --git a/monitor/sre-agent-event-lab/scripts/setup-venv.sh b/monitor/sre-agent-event-lab/scripts/setup-venv.sh new file mode 100755 index 0000000..e24bbab --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/setup-venv.sh @@ -0,0 +1,73 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Prepares `app/.venv` with every package the lab's local Python tooling +# needs: the app's own runtime dependencies (`requirements.txt`, pulled in +# by `-r requirements.txt` at the top of `requirements-dev.txt`), Pillow for +# `render_capture.py`'s PNG/GIF rendering, and pytest/httpx for `app/tests`. +# `capture-scenario.sh` and `guides/05-results.md`'s notification step both +# run under this interpreter. +# +# `uv` is mandatory here, not merely preferred: this lab runs behind a +# corporate proxy that is configured for `uv` (its own index/proxy/keyring +# settings), and a bare `pip install` would bypass that configuration and +# resolve packages straight from the public PyPI -- exactly the network +# path the proxy exists to prevent. So there is no pip fallback: if `uv` is +# missing, this script fails with an actionable install pointer instead of +# silently reaching the public index. +# +# Idempotency matters because this script is invoked from the `postprovision` +# hook, which runs after `azd provision` has already created every cloud +# resource: by the time this step can fail, the Azure spend for this run has +# already started. So every failure message below states the exact command +# to re-run -- re-running never repeats the (already-succeeded) cloud +# provisioning, only this local step -- and `uv venv --allow-existing` plus +# `uv pip install` make re-running safe even when a previous attempt got +# partway through. + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +readonly SCRIPT_DIR +LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +readonly LAB_ROOT +readonly APP_DIR="${LAB_ROOT}/app" +readonly VENV_DIR="${APP_DIR}/.venv" +readonly VENV_PYTHON="${VENV_DIR}/bin/python" +readonly REQUIREMENTS_FILE="${APP_DIR}/requirements-dev.txt" +# `requirements.txt` pins `opentelemetry-instrumentation-fastapi~=0.64b0`, +# which requires Python>=3.10 (matching the app's own container image, +# `app/Dockerfile`: `python:3.12-slim`). Requesting a version range here -- +# rather than one exact minor version -- lets `uv` pick any interpreter it +# already manages that clears that floor, while still protecting +# `--allow-existing` from silently reusing an older, incompatible +# interpreter left behind by a pre-uv `python3 -m venv` (uv recreates the +# venv in place when the existing interpreter doesn't satisfy the request). +readonly VENV_PYTHON_VERSION=">=3.10" +readonly RERUN_HINT="Cloud resources from 'azd provision' are already deployed; only this local step needs to be retried. Re-run: azd hooks run postprovision (or directly: ./scripts/setup-venv.sh)" + +if ! command -v uv >/dev/null 2>&1; then + echo "uv is required to set up ${VENV_DIR} but was not found on PATH." >&2 + echo "This lab does not fall back to a bare 'pip install': install uv first (https://docs.astral.sh/uv/getting-started/installation/), configured for this network's proxy, then re-run." >&2 + echo "${RERUN_HINT}" >&2 + exit 1 +fi + +if ! uv venv --python "${VENV_PYTHON_VERSION}" --allow-existing "${VENV_DIR}"; then + echo "Failed to create the virtual environment at ${VENV_DIR}." >&2 + echo "${RERUN_HINT}" >&2 + exit 1 +fi + +if ! uv pip install --python "${VENV_PYTHON}" -r "${REQUIREMENTS_FILE}"; then + echo "Failed to install ${REQUIREMENTS_FILE} into ${VENV_DIR}." >&2 + echo "A misconfigured or unreachable corporate proxy is the most common cause of a uv install failure here." >&2 + echo "${RERUN_HINT}" >&2 + exit 1 +fi + +if ! "${VENV_PYTHON}" -c "import PIL" >/dev/null 2>&1; then + echo "${REQUIREMENTS_FILE} installed, but Pillow (PIL) is still not importable from ${VENV_PYTHON}." >&2 + echo "${RERUN_HINT}" >&2 + exit 1 +fi + +echo "Python environment ready: ${VENV_PYTHON} (${REQUIREMENTS_FILE} installed, Pillow importable)." diff --git a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py index f806a1a..9aadd67 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/doctor_harness.py @@ -138,6 +138,13 @@ class FakeAz: reader_role_inherited: Dict[str, bool] = field(default_factory=_no_inherited_reader) baseline_orders_succeed: bool = True baseline_documents_succeed: bool = True + # Whether `app/.venv/bin/python` exists at all, and whether Pillow is + # importable from it -- the two facts `scripts/setup-venv.sh` (run from + # `postprovision`) is responsible for making true, and doctor's "Python + # environment" check reports on. Both default to a fully set-up venv so + # only the tests exercising this check need to touch either field. + venv_present: bool = True + pillow_importable: bool = True # None means "use the module default AZD_VALUES"; a test passes {} (or # a partial dict) to exercise the missing/partial-configuration paths. azd_values: "Dict[str, str] | None" = None @@ -280,7 +287,7 @@ def _curl_stub_source(fake_az: FakeAz) -> str: """ -def _python3_stub_source(fake_az: FakeAz, log_path: Path) -> str: +def _python3_stub_source(fake_az: FakeAz, log_path: Path, is_venv_python: bool = False) -> str: """Fake `python3`/`.venv/bin/python` for `loadgen.py`: writes a minimal, valid summary and exits with loadgen's real contract (0 success, 2 a request mismatch), keyed by the target URL so orders/documents can be @@ -288,12 +295,25 @@ def _python3_stub_source(fake_az: FakeAz, log_path: Path) -> str: Every other script -- notably `lab_state.py` and `score.py`, whose behaviour these tests are checking -- runs under the real interpreter, - so a lab script that records or reads state is exercised, not faked.""" + so a lab script that records or reads state is exercised, not faked. + + When `is_venv_python` is set, this stub also answers the + `-c "import PIL"` probe that `scripts/setup-venv.sh` and doctor's + "Python environment" check use to decide whether Pillow is importable + from `app/.venv`, per `fake_az.pillow_importable` -- loadgen never sends + that probe, so it cannot collide with the loadgen branch below.""" orders_ok = 1 if fake_az.baseline_orders_succeed else 0 documents_ok = 1 if fake_az.baseline_documents_succeed else 0 + pil_probe = "" + if is_venv_python: + pil_exit = 0 if fake_az.pillow_importable else 1 + pil_probe = f"""if [[ "${{1:-}}" == "-c" && "${{2:-}}" == *PIL* ]]; then + exit {pil_exit} +fi +""" return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" -case "${{1:-}}" in +{pil_probe}case "${{1:-}}" in *loadgen.py) ;; *) exec "{REAL_PYTHON}" "$@" ;; esac @@ -449,9 +469,20 @@ def _materialize(fake_az: FakeAz) -> LabRun: write_executable(bin_dir / "curl", _curl_stub_source(fake_az)) write_executable(bin_dir / "python3", _python3_stub_source(fake_az, python_log)) - venv_bin = lab / "app" / ".venv" / "bin" - venv_bin.mkdir(parents=True, exist_ok=True) - write_executable(venv_bin / "python", _python3_stub_source(fake_az, python_log)) + # `app/.venv` is always removed and recreated (rather than only + # `mkdir -p`'d once) so a test that flips `venv_present`/ + # `pillow_importable` between two `run_doctor` calls on the same + # `fake_az` -- exactly like every other mutable field here -- is + # honoured on the second call too. + venv_dir = lab / "app" / ".venv" + shutil.rmtree(venv_dir, ignore_errors=True) + if fake_az.venv_present: + venv_bin = venv_dir / "bin" + venv_bin.mkdir(parents=True, exist_ok=True) + write_executable( + venv_bin / "python", + _python3_stub_source(fake_az, python_log, is_venv_python=True), + ) workdir = tmp_path / "elsewhere" workdir.mkdir(exist_ok=True) diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 8aba66e..16e43fc 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -210,16 +210,27 @@ def _az_stub_source(log_path, state_dir): """ -def _lab_python_stub_source(log_path, capture_timeline): +def _lab_python_stub_source(log_path, capture_timeline, pillow_importable=True): """A fake `${LAB_ROOT}/app/.venv/bin/python`. Only the two scripts that would reach the SRE Agent data plane or write images are faked; every other script (notably `lab_state.py` and `score.py`, which are the behaviour under test) runs under the real interpreter. + + Also answers the `-c "import PIL"` probe `capture-scenario.sh` and + doctor's "Python environment" check use to verify Pillow is importable, + per `pillow_importable` -- explicitly, rather than delegating to + whichever real interpreter happens to run the test suite, so this fake + behaves the same regardless of that interpreter's own installed + packages. """ + pil_exit = 0 if pillow_importable else 1 return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" +if [[ "${{1:-}}" == "-c" && "${{2:-}}" == *PIL* ]]; then + exit {pil_exit} +fi case "${{1:-}}" in *capture_agent.py) shift @@ -269,6 +280,8 @@ def make_lab( alert_resolves=True, alert_fires=True, capture_timeline=CONCLUSION_TIMELINE, + venv_present=True, + pillow_importable=True, ): """A throwaway copy of the lab plus fake CLIs; returns a run context.""" lab = tmp_path / "lab" @@ -306,10 +319,12 @@ def make_lab( write_executable(bin_dir / "python3", _python3_stub_source(python_log)) venv_bin = lab / "app" / ".venv" / "bin" - venv_bin.mkdir(parents=True) - write_executable( - venv_bin / "python", _lab_python_stub_source(lab_python_log, capture_timeline) - ) + if venv_present: + venv_bin.mkdir(parents=True) + write_executable( + venv_bin / "python", + _lab_python_stub_source(lab_python_log, capture_timeline, pillow_importable), + ) workdir = tmp_path / "elsewhere" workdir.mkdir() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py index 0b2017a..aaab949 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py @@ -160,6 +160,62 @@ def test_azd_postprovision_moves_ingress_to_the_app_port_and_records_the_image() assert ingress_at < healthz_at +def test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call(): + text = AZD_POSTPROVISION.read_text() + + assert "setup-venv.sh" in text + setup_venv_at = text.index("setup-venv.sh") + first_account_show_at = text.index("az account show") + assert setup_venv_at < first_account_show_at, ( + "setup-venv.sh must run before the hook makes any Azure CLI call" + ) + + +def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp_path): + """`app/.venv` setup is local and has nothing to do with the Azure CLI, + but a broken corporate proxy or missing `uv` must still stop the hook + before it spends a single Azure API call -- the cloud side is already + provisioned by the time this hook runs, so failing fast here changes + nothing about that, but a failure must never be masked by continuing + on to the ACR build.""" + scripts_copy = tmp_path / "scripts" + scripts_copy.mkdir() + (scripts_copy / "azd-postprovision.sh").write_text(AZD_POSTPROVISION.read_text()) + (scripts_copy / "azd-postprovision.sh").chmod(0o755) + fake_setup_venv = scripts_copy / "setup-venv.sh" + fake_setup_venv.write_text( + "#!/usr/bin/env bash\n" + "echo 'uv is required to set up app/.venv but was not found on PATH.' >&2\n" + "echo 'azd hooks run postprovision' >&2\n" + "exit 1\n" + ) + fake_setup_venv.chmod(0o755) + + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + az_log = tmp_path / "az-calls.log" + _write_az_stub(bin_dir, az_log) + + env = dict(os.environ) + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" + env["AZURE_SUBSCRIPTION_ID"] = "11111111-2222-3333-4444-555555555555" + env["AZURE_RESOURCE_GROUP"] = "rg-test" + env["AZURE_ACR_NAME"] = "acrtest" + env["AZURE_CONTAINER_APP_NAME"] = "ca-test" + env["AZURE_CONTAINER_APP_FQDN"] = "ca-test.example.com" + + result = subprocess.run( + [str(scripts_copy / "azd-postprovision.sh")], + capture_output=True, + text=True, + env=env, + ) + + assert result.returncode != 0 + assert not az_log.exists() or az_log.read_text() == "" + assert "uv" in result.stderr + + def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): """`az account show` fails with a generic Azure CLI error when signed out. Guard it so the hook fails fast with one unambiguous message diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py index 24c657c..267d539 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py @@ -77,6 +77,7 @@ def test_doctor_passes_fully_healthy_environment(fake_az): assert result.returncode == 0, result.stdout + result.stderr rows = _rows_of(result) assert rows["Required commands"] == "PASS" + assert rows["Python environment"] == "PASS" assert rows["Log Analytics CLI extension"] == "PASS" assert rows["Azure CLI login"] == "PASS" assert rows["azd authentication"] == "PASS" @@ -122,6 +123,45 @@ def test_doctor_fails_when_healthz_does_not_return_200(fake_az): assert "503" in result.stdout +def test_doctor_fails_when_venv_is_missing(fake_az): + """Finding #1: doctor must report on the venv `setup-venv.sh` (run from + `postprovision`) is responsible for creating, with a remedy pointing at + that exact script -- not just a generic "python3 missing" message, + since `python3` itself is still on PATH.""" + fake_az.venv_present = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Python environment\tFAIL" in result.stdout + assert "setup-venv.sh" in _detail_of(result, "Python environment") + + +def test_doctor_fails_when_pillow_is_not_importable_from_the_venv(fake_az): + """A venv that exists but never finished installing (or was created by + something other than `setup-venv.sh`) must fail this check too, not + just an absent venv directory.""" + fake_az.pillow_importable = False + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Python environment\tFAIL" in result.stdout + assert "setup-venv.sh" in _detail_of(result, "Python environment") + + +def test_doctor_passes_venv_check_independently_of_azure_reachability(fake_az): + """The venv/Pillow readiness check is a local precondition, not an + Azure fact: it must still report accurately (and PASS when the venv is + fine) even when every Azure-dependent check is blocked.""" + fake_az.logged_in = False + + result = run_doctor(fake_az) + + assert "Python environment\tPASS" in result.stdout + assert "Azure CLI login\tFAIL" in result.stdout + + def test_doctor_fails_when_app_insights_has_no_recent_requests(fake_az): fake_az.app_insights_has_recent_requests = False diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index e04c406..2ef9880 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -18,6 +18,7 @@ OFFICIAL_ASSETS = LAB_ROOT / "assets" / "official" RUNBOOK = LAB_ROOT / "runbooks" / "incident-response.md" LAB_SH = LAB_ROOT / "scripts" / "lab.sh" +VALIDATION_RESULTS = LAB_ROOT / "validation-results.md" GUIDE_NAMES = ( "01-agent-setup.md", @@ -247,6 +248,69 @@ def test_incident_platform_path_is_the_documented_default(): assert setup.index("Builder > Incident platform") < plan_index +def test_azuresre_dev_audience_fact_is_preserved_as_legacy_history(): + """Finding #3: the Task 6 rewrite deleted a verified fact (the HTTP + Trigger endpoint only accepts an `https://azuresre.dev` audience token, + not `https://management.azure.com/`) instead of relocating it. It must + survive somewhere user-facing, framed as historical record for the + (non-default) Logic App bridge -- not restored as a default-flow + instruction anywhere in the README/guides.""" + text = VALIDATION_RESULTS.read_text() + + assert "azuresre.dev" in text + assert "management.azure.com" in text + audience_block = next(block for block in blocks(text) if "azuresre.dev" in block) + assert "레거시" in audience_block + + for path in all_docs(): + assert "azuresre.dev" not in path.read_text(), path.name + + +def test_readme_and_guides_use_exactly_one_cd_per_document(): + """Finding #4: every command in a document must be runnable + sequentially from the single working directory that document's own + (at most one) `cd` establishes -- a second `cd monitor/sre-agent-event-lab` + later in the same document would fail, since no such nested directory + exists once the first `cd` already landed there.""" + for path in all_docs(): + cd_count = len(re.findall(r"(?m)^cd\s+\S", path.read_text())) + assert cd_count <= 1, (path.name, cd_count) + + +def test_readme_scorecard_row_states_per_scenario_and_overall_maximums(): + """Finding #5: `scripts/score.py` computes MAX_POINTS = 10 per scenario + and MAX_POINTS * len(SCENARIOS) = 30 overall; the README's summary + table must describe both, not label the whole lab "10점 만점".""" + text = README.read_text() + + row = next(line for line in text.splitlines() if "scorecard.json" in line) + assert "10점" in row + assert "30점" in row + + +def test_guide02_start_conditions_do_not_claim_an_unenforced_concurrency_lock(): + """Finding #6: `lab_state.py` has no concurrency lock (its own + docstring says so); S1's start conditions must only claim what + `RUN_REQUIREMENTS["s1"]` actually enforces.""" + text = (GUIDES / "02-scenario-s1.md").read_text() + conditions = text.split("## 시작 조건", 1)[1].split("\n## ", 1)[0] + + assert "진행 중인 다른 시나리오가 없습니다" not in conditions + assert "baseline_passed" in conditions + assert "agent_setup_acknowledged" in conditions + + +def test_guide05_generate_notifications_runs_under_plain_python3(): + """Finding #6: `generate_notifications.py` only imports the standard + library (html, json, re, email.*, pathlib, typing) -- unlike + `render_capture.py`, it has no Pillow/venv dependency, so the guide + must not invoke it through `app/.venv/bin/python`.""" + text = (GUIDES / "05-results.md").read_text() + + assert "python3 scripts/generate_notifications.py" in text + assert "app/.venv/bin/python scripts/generate_notifications.py" not in text + + # --- guide structure ----------------------------------------------------- diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index a00c323..1de3948 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -331,6 +331,37 @@ def test_capture_scenario_without_a_recorded_run_names_the_command_to_run(tmp_pa assert "lab.sh run s1" in result.stderr +def test_capture_scenario_fails_actionably_when_venv_is_missing(tmp_path): + """Finding #1: cloud resources (the alert rules, the app, etc.) may + already be deployed by the time this local-only precondition fails, so + the message must name the exact rerun command, not just what's wrong.""" + lab_run = make_lab(tmp_path, venv_present=False) + lab_run.write_agent_setup() + lab_run.seed_state() + run_result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert run_result.returncode == 0, run_result.stderr + + result = lab_run.run("capture-scenario.sh", ["s1"]) + + assert result.returncode != 0 + assert "Missing Python environment" in result.stderr + assert "setup-venv.sh" in result.stderr + + +def test_capture_scenario_fails_actionably_when_pillow_is_not_importable(tmp_path): + lab_run = make_lab(tmp_path, pillow_importable=False) + lab_run.write_agent_setup() + lab_run.seed_state() + run_result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert run_result.returncode == 0, run_result.stderr + + result = lab_run.run("capture-scenario.sh", ["s1"]) + + assert result.returncode != 0 + assert "Pillow" in result.stderr + assert "setup-venv.sh" in result.stderr + + def test_capture_scenario_renders_from_another_directory(tmp_path): lab_run = make_lab(tmp_path) lab_run.write_agent_setup() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py b/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py new file mode 100644 index 0000000..dd4ed3d --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py @@ -0,0 +1,201 @@ +"""Behaviour tests for `setup-venv.sh`, the idempotent preparer of +`app/.venv` invoked from the `postprovision` azd hook. + +Every test drives the real script against a fake `uv` on PATH (never a real +network install), so these prove the script's actual call contract: it +requires `uv` with no pip fallback, it fails fast and actionably at each +step, and re-running it is safe. +""" +import os +import stat +import subprocess +from pathlib import Path + + +SCRIPT = Path(__file__).parents[1] / "setup-venv.sh" +LAB_ROOT = Path(__file__).parents[2] +REQUIREMENTS_DEV = LAB_ROOT / "app" / "requirements-dev.txt" + + +def _write_executable(path: Path, source: str) -> None: + path.write_text(source) + path.chmod(path.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + + +def _write_fake_uv(bin_dir: Path, log_path: Path, venv_exit: int = 0, pip_exit: int = 0, pillow_importable: bool = True): + """A fake `uv` that logs every invocation and creates just enough of a + venv (`bin/python` as a real, runnable interpreter) that the script's + own Pillow-import check can run against it.""" + real_python = os.environ.get("SETUP_VENV_TEST_PYTHON", "python3") + stub = bin_dir / "uv" + _write_executable( + stub, + f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +case "$1" in + venv) + if [[ {venv_exit} -ne 0 ]]; then + exit {venv_exit} + fi + target="${{@: -1}}" + mkdir -p "${{target}}/bin" + cat > "${{target}}/bin/python" <<'PYEOF' +#!/usr/bin/env bash +if [[ "$1" == "-c" ]]; then + case "$2" in + *PIL*) + exit_code=$([[ "{str(pillow_importable).lower()}" == "true" ]] && echo 0 || echo 1) + exit "${{exit_code}}" + ;; + esac +fi +exec {real_python} "$@" +PYEOF + chmod +x "${{target}}/bin/python" + ;; + pip) + exit {pip_exit} + ;; + *) + exit 0 + ;; +esac +""", + ) + + +def _run(bin_dir: Path, workdir: Path, extra_path: bool = True): + env = dict(os.environ) + if extra_path: + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" + else: + # Simulate `uv` genuinely absent: a PATH with no bin_dir at all, + # not just an empty one, so no fake (or real, developer-machine) + # `uv` is reachable. + env["PATH"] = "/usr/bin:/bin" + return subprocess.run( + [str(SCRIPT)], + capture_output=True, + text=True, + env=env, + cwd=str(workdir), + ) + + +def test_fails_actionably_with_no_pip_fallback_when_uv_is_missing(tmp_path): + result = _run(tmp_path / "bin-unused", tmp_path, extra_path=False) + + assert result.returncode != 0 + assert "uv" in result.stderr + assert "install uv" in result.stderr.lower() or "uv is required" in result.stderr + assert "azd hooks run postprovision" in result.stderr + assert "already deployed" in result.stderr + + +def test_succeeds_and_calls_uv_venv_then_uv_pip_install_with_requirements_dev(tmp_path): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path) + + result = _run(bin_dir, tmp_path) + + assert result.returncode == 0, result.stdout + result.stderr + calls = log_path.read_text() + assert "venv --python" in calls + assert "--allow-existing" in calls + assert ".venv" in calls + assert "pip install --python" in calls + assert "requirements-dev.txt" in calls + + +def test_never_falls_back_to_a_bare_pip_binary(tmp_path): + """Even when a bare `pip` is reachable on PATH, the script must drive + installation only through `uv pip install`.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path) + bare_pip_log = tmp_path / "bare-pip-calls.log" + _write_executable( + bin_dir / "pip", + f"#!/usr/bin/env bash\nprintf '%s\\n' \"$*\" >> \"{bare_pip_log}\"\nexit 0\n", + ) + + result = _run(bin_dir, tmp_path) + + assert result.returncode == 0, result.stdout + result.stderr + assert not bare_pip_log.exists(), "setup-venv.sh must never invoke a bare pip" + + +def test_is_idempotent_across_repeated_runs(tmp_path): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path) + + first = _run(bin_dir, tmp_path) + second = _run(bin_dir, tmp_path) + + assert first.returncode == 0, first.stdout + first.stderr + assert second.returncode == 0, second.stdout + second.stderr + + +def test_fails_actionably_when_uv_venv_creation_fails(tmp_path): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path, venv_exit=1) + + result = _run(bin_dir, tmp_path) + + assert result.returncode != 0 + assert "azd hooks run postprovision" in result.stderr + assert "already deployed" in result.stderr + assert "pip install" not in log_path.read_text() + + +def test_fails_actionably_when_uv_pip_install_fails(tmp_path): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path, pip_exit=1) + + result = _run(bin_dir, tmp_path) + + assert result.returncode != 0 + assert "azd hooks run postprovision" in result.stderr + assert "already deployed" in result.stderr + assert "proxy" in result.stderr + + +def test_fails_actionably_when_pillow_is_still_not_importable(tmp_path): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path, pillow_importable=False) + + result = _run(bin_dir, tmp_path) + + assert result.returncode != 0 + assert "Pillow" in result.stderr + assert "azd hooks run postprovision" in result.stderr + + +def test_installs_the_labs_real_requirements_dev_file(): + assert REQUIREMENTS_DEV.is_file() + assert "Pillow" in REQUIREMENTS_DEV.read_text() + + +def test_script_never_executes_a_bare_pip_or_pip3_command(): + """Every executable (non-comment, non-message) shell command line that + starts with `pip`/`pip3` would be a fallback outside `uv`; only lines + that start with `uv` are allowed to reach a `pip` subcommand.""" + import re + + for line in SCRIPT.read_text().splitlines(): + stripped = line.strip() + if not stripped or stripped.startswith(("#", "echo")): + continue + if re.match(r"^(if\s+!\s+)?pip3?\b", stripped): + raise AssertionError(("bare pip invocation found", stripped)) diff --git a/monitor/sre-agent-event-lab/validation-results.md b/monitor/sre-agent-event-lab/validation-results.md index df0d5bd..fd70a64 100644 --- a/monitor/sre-agent-event-lab/validation-results.md +++ b/monitor/sre-agent-event-lab/validation-results.md @@ -244,6 +244,8 @@ Container App workload identity의 테스트 Blob container data-plane read 역 제품의 표준 Azure Monitor 경로는 incident platform과 response plan을 통해 Agent로 직접 전달된다. 이번 실증에서는 response plan 공개 API 자동 구성 제약 때문에 Action Group → Logic App managed identity → SRE Agent HTTP Trigger라는 lab-specific bridge를 사용했다. 이 bridge는 세 시나리오에서 자동 thread를 만들었지만 표준 Azure Monitor 도입의 필수 구성은 아니다. +레거시 기록(현재 기본 실습에는 적용되지 않음, bridge를 다시 구성해야 할 때만 참고): 2026-08-12 실측에서 HTTP Trigger endpoint는 `https://management.azure.com/` audience token을 HTTP 401로 거부하고 `https://azuresre.dev` audience token을 수락했다. 당시 Logic App HTTP action의 managed identity audience를 `https://azuresre.dev`로 설정해서만 이 bridge가 동작했다. + ### RCA 정확도 - S1은 revision env, 120개 request trace, Activity Log를 결합해 injected HTTP 500을 정확히 진단했다. From 0f741932de4281625299ac6d1b56793037b2be7e Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 20:46:05 +0900 Subject: [PATCH 14/26] fix(sre-lab): quarantine setup-venv tests, fix postprovision recovery wording Fixes three remaining Task 6 review findings, all with RED/GREEN tests. 1. test_setup_venv.py ran the *real*, in-place setup-venv.sh: the script resolves SCRIPT_DIR/LAB_ROOT/VENV_DIR from its own ${BASH_SOURCE[0]}, never from the caller's cwd, so pointing the process cwd at a tmp_path (as the old tests did) still targeted the real app/.venv -- and the tests' fake `uv` genuinely creates target/bin/python via `cat >`, which either overwrites the real app/.venv/bin/python or, since it is a symlink `uv` manages, writes straight through it into the real, shared-across-projects interpreter binary. A safe, non-mutating demonstration (fake `uv` that only logs argv, never touches disk) proved the exact target path was the real one, without risking it: confirmed in session, not committed. Every test now runs a `lab_copy` fixture that copies setup-venv.sh plus app/requirements.txt and app/requirements-dev.txt into tmp_path with the layout the script depends on, so its own path resolution lands entirely inside tmp_path. A new autouse fingerprint fixture (symlink type + target + content sha256) asserts the real app/.venv/bin/python is byte-identical before and after every test in the file, as a permanent regression tripwire. Verified real app/.venv/bin/python is unchanged and still imports Pillow after the full suite. 2. setup-venv.sh runs before azd-postprovision.sh's ACR build and Container App update, not after, so a failure here means the cloud app deployment for this run has not started yet -- only Bicep provisioning has. Re-running just this script would leave that deployment silently skipped. RERUN_HINT no longer offers "(or directly: ./scripts/setup-venv.sh)"; every failure message now points only at `cd && azd hooks run postprovision`. doctor.sh and capture-scenario.sh keep recommending ./scripts/setup-venv.sh directly, since those run well after a deployment has already succeeded and are unaffected by this ordering. README's deploy section and guide 01's failure table are reworded to match: the README no longer implies "only a local step remains" when this step fails, and guide 01 now has a remedy row for doctor's Python environment check. 3. uv's real "no matching Python" failure (`error: No interpreter found for Python >=X ...`, verified against uv 0.10.9) now gets its own actionable line recommending `uv python install 3.12`, framed as reusing the same corporate proxy/mirror already configured for uv -- never a public pip install. Tests: 421 passed (390 scripts/tests + 21 infra/tests + 10 app/tests). bash -n scripts/*.sh clean. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 2 +- .../guides/01-agent-setup.md | 1 + .../sre-agent-event-lab/scripts/setup-venv.sh | 40 ++- .../scripts/tests/test_lab_guides.py | 38 +++ .../scripts/tests/test_setup_venv.py | 248 ++++++++++++++++-- 5 files changed, 295 insertions(+), 34 deletions(-) diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index e92aa8d..04d7a38 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -78,7 +78,7 @@ azd up 2>&1 | tee evidence/deploy.log Bicep provision → ACR 클라우드 빌드 → Container App 이미지 교체 순서로 진행되며 로컬 Docker는 필요 없습니다. 처음에는 공개 placeholder 이미지가 80 포트로 뜨고, postprovision hook이 ingress를 8000으로 옮긴 뒤 실습 이미지로 교체합니다. -같은 postprovision hook이 `scripts/setup-venv.sh`로 `app/.venv`도 함께 준비합니다(`uv venv` + `uv pip install -r requirements-dev.txt`). 이 단계가 실패해도 클라우드 리소스는 이미 만들어진 뒤이므로 다시 `azd provision`부터 할 필요는 없습니다: 안내된 명령(`azd hooks run postprovision` 또는 `./scripts/setup-venv.sh`)만 다시 실행하면 됩니다. +같은 postprovision hook의 첫 단계가 `scripts/setup-venv.sh`로 `app/.venv`를 준비하는 것입니다(`uv venv` + `uv pip install -r requirements-dev.txt`). 이 단계는 위 ACR 빌드·Container App 이미지 교체보다 먼저 실행되므로, 여기서 실패하면 클라우드 앱 배포(ACR 빌드, 이미지 교체, 헬스체크)는 아직 시작되지 않은 상태입니다 -- `./scripts/setup-venv.sh`만 따로 다시 실행하면 이 배포 단계를 건너뛰게 되므로, 반드시 hook 전체를 다시 실행하세요: `azd hooks run postprovision`. 성공 조건은 provision 성공, 활성 revision `Healthy`, `/healthz` HTTP 200 세 가지입니다. diff --git a/monitor/sre-agent-event-lab/guides/01-agent-setup.md b/monitor/sre-agent-event-lab/guides/01-agent-setup.md index a38ea3d..a754ea8 100644 --- a/monitor/sre-agent-event-lab/guides/01-agent-setup.md +++ b/monitor/sre-agent-event-lab/guides/01-agent-setup.md @@ -123,6 +123,7 @@ JSON | 증상 | 조치 | |---|---| +| `doctor`의 `Python environment` 검사가 `FAIL` | `app/.venv`가 없거나 Pillow가 안 잡힙니다. 로컬 문제이며 클라우드 배포와는 무관하니 바로 실행: `./scripts/setup-venv.sh` | | `doctor`의 Reader 검사가 `FAIL` | 두 principal ID가 근거 파일과 같은지 확인하고 리소스 그룹에 Reader를 다시 부여합니다 | | `baseline`이 telemetry 없음으로 종료 | 10분 더 기다린 뒤 다시 실행합니다. 계속 실패하면 `azd env get-value AZURE_CONTAINER_APP_FQDN`으로 앱을 직접 호출해 봅니다 | | `acknowledge`가 기록되지 않음 | 입력한 단어가 정확한지, `azd env select`로 올바른 환경을 골랐는지 확인합니다 | diff --git a/monitor/sre-agent-event-lab/scripts/setup-venv.sh b/monitor/sre-agent-event-lab/scripts/setup-venv.sh index e24bbab..410550f 100755 --- a/monitor/sre-agent-event-lab/scripts/setup-venv.sh +++ b/monitor/sre-agent-event-lab/scripts/setup-venv.sh @@ -17,13 +17,18 @@ set -euo pipefail # silently reaching the public index. # # Idempotency matters because this script is invoked from the `postprovision` -# hook, which runs after `azd provision` has already created every cloud -# resource: by the time this step can fail, the Azure spend for this run has -# already started. So every failure message below states the exact command -# to re-run -- re-running never repeats the (already-succeeded) cloud -# provisioning, only this local step -- and `uv venv --allow-existing` plus -# `uv pip install` make re-running safe even when a previous attempt got -# partway through. +# hook *before* that hook's ACR build and Container App update -- see +# `azd-postprovision.sh` -- so a failure here happens before the cloud app +# deployment for this run has even started, not after it. Re-running only +# this script would leave that deployment never attempted, so every failure +# message below points at re-running the *whole* hook (`azd hooks run +# postprovision`), never at running this script directly -- and `uv venv +# --allow-existing` plus `uv pip install` make that safe to re-run even when +# a previous attempt got partway through. (Running this script directly is +# still the right move for a purely local `app/.venv` problem noticed well +# after a deployment already succeeded -- see `doctor.sh` and +# `capture-scenario.sh`'s own remediation text -- just never as the response +# to a failure reported by *this* script.) SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" readonly SCRIPT_DIR @@ -42,7 +47,13 @@ readonly REQUIREMENTS_FILE="${APP_DIR}/requirements-dev.txt" # interpreter left behind by a pre-uv `python3 -m venv` (uv recreates the # venv in place when the existing interpreter doesn't satisfy the request). readonly VENV_PYTHON_VERSION=">=3.10" -readonly RERUN_HINT="Cloud resources from 'azd provision' are already deployed; only this local step needs to be retried. Re-run: azd hooks run postprovision (or directly: ./scripts/setup-venv.sh)" +# Cloud resources from `azd provision` are already deployed by the time this +# hook (and so this script) runs, but this hook's own ACR build and Container +# App update -- the cloud *app* deployment -- run after this script, not +# before it, so a failure here must not be answered by re-running only this +# script: that would leave the app deployment never attempted. `cd` pins the +# rerun to this lab's project directory regardless of the operator's shell. +readonly RERUN_HINT="Cloud infrastructure from 'azd provision' is already deployed, but this hook's Container App image build and deployment run *after* this step and have not happened yet -- re-running only this script would silently skip them. Re-run the whole hook instead: cd ${LAB_ROOT} && azd hooks run postprovision" if ! command -v uv >/dev/null 2>&1; then echo "uv is required to set up ${VENV_DIR} but was not found on PATH." >&2 @@ -51,11 +62,22 @@ if ! command -v uv >/dev/null 2>&1; then exit 1 fi -if ! uv venv --python "${VENV_PYTHON_VERSION}" --allow-existing "${VENV_DIR}"; then +if ! UV_VENV_OUTPUT="$(uv venv --python "${VENV_PYTHON_VERSION}" --allow-existing "${VENV_DIR}" 2>&1)"; then + printf '%s\n' "${UV_VENV_OUTPUT}" >&2 echo "Failed to create the virtual environment at ${VENV_DIR}." >&2 + if grep -qi "no interpreter found" <<<"${UV_VENV_OUTPUT}"; then + # uv's own message ("No interpreter found for Python >=3.10 ...") means + # no installed Python clears the floor yet -- not a proxy or install + # failure -- so the fix is installing one, and `uv python install` is + # the one installer that reuses this network's corporate proxy/mirror + # already configured for `uv`; a public `pip`/`pip install` bypasses + # that configuration entirely and must never be suggested here. + echo "No installed Python satisfies ${VENV_PYTHON_VERSION} yet. Install one through uv itself -- it uses the corporate proxy/mirror already configured for uv, not the public PyPI index: uv python install 3.12" >&2 + fi echo "${RERUN_HINT}" >&2 exit 1 fi +printf '%s\n' "${UV_VENV_OUTPUT}" if ! uv pip install --python "${VENV_PYTHON}" -r "${REQUIREMENTS_FILE}"; then echo "Failed to install ${REQUIREMENTS_FILE} into ${VENV_DIR}." >&2 diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index 2ef9880..f6e8b0c 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -433,6 +433,44 @@ def test_agent_setup_guide_offers_azd_env_set_without_storing_secrets(): assert key in text, key +def test_agent_setup_guide_offers_a_python_environment_remedy(): + """Finding #4 (Task 6 follow-up): `doctor.sh` gained a `Python + environment` check, but the guide's failure table never told an + operator what to do about it. The remedy is local-only and independent + of the postprovision-hook ordering contract (see + `test_setup_venv_orders_local_recovery_correctly` below), so it may -- + and should -- name `./scripts/setup-venv.sh` directly. + """ + text = (GUIDES / "01-agent-setup.md").read_text() + heading = "## 실패했을 때" + + assert heading in text + section = text.split(heading, 1)[1] + assert "Python environment" in section + assert "./scripts/setup-venv.sh" in section + + +def test_readme_does_not_claim_only_a_local_step_remains_after_postprovision(): + """Finding #2 (Task 6 follow-up): `setup-venv.sh` runs *before* the same + postprovision hook's ACR build and Container App update, so a failure + there means the cloud app deployment has not happened yet -- not merely + that "only a local step" is left. The README must send an operator to + re-run the whole hook, not offer `./scripts/setup-venv.sh` as an + equally-valid alternative for this specific failure. + """ + text = README.read_text() + heading = "## 배포" + + assert heading in text + section = text.split(heading, 1)[1].split("## ", 1)[0] + assert "setup-venv.sh" in section + assert "azd hooks run postprovision" in section + assert "또는 `./scripts/setup-venv.sh`" not in section, ( + "the README must not offer running setup-venv.sh directly as an " + "alternative recovery for a postprovision-hook failure" + ) + + def test_guides_do_not_request_secrets_in_environment(): text = "\n".join(path.read_text() for path in GUIDES.glob("*.md")) for forbidden in ("GITHUB_PAT=", "OAUTH_TOKEN=", "CLIENT_SECRET="): diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py b/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py index dd4ed3d..6eb75aa 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py @@ -1,20 +1,99 @@ """Behaviour tests for `setup-venv.sh`, the idempotent preparer of `app/.venv` invoked from the `postprovision` azd hook. -Every test drives the real script against a fake `uv` on PATH (never a real -network install), so these prove the script's actual call contract: it -requires `uv` with no pip fallback, it fails fast and actionably at each -step, and re-running it is safe. +Every test drives the real script's *logic* against a fake `uv` on PATH +(never a real network install) -- but never the real script *file in +place*. `setup-venv.sh` resolves `SCRIPT_DIR`/`LAB_ROOT`/`VENV_DIR` from its +own `${BASH_SOURCE[0]}`, never from the caller's cwd, so running the real, +in-place script (as this suite once did, pointing only `cwd` at a scratch +`tmp_path`) still always operates on the real `app/.venv` -- and a fake `uv` +that *creates* `${target}/bin/python` (as every success-path test here +needs) then either overwrites the real `app/.venv/bin/python` outright or, +because it is a symlink `uv` manages, writes straight through it into the +real interpreter binary it points at. Every test below therefore runs a +throwaway *copy* of `setup-venv.sh` plus the real `app/requirements.txt` / +`app/requirements-dev.txt`, laid out under `tmp_path` exactly as the real +lab lays them out, so the copy's own path resolution lands entirely inside +`tmp_path`. `real_lab_venv_is_never_touched` is the regression tripwire that +proves it: it fingerprints the real `app/.venv/bin/python` before and after +every test in this file and fails loudly on any drift. """ +import hashlib import os +import re +import shutil import stat import subprocess from pathlib import Path +import pytest + SCRIPT = Path(__file__).parents[1] / "setup-venv.sh" LAB_ROOT = Path(__file__).parents[2] +REQUIREMENTS = LAB_ROOT / "app" / "requirements.txt" REQUIREMENTS_DEV = LAB_ROOT / "app" / "requirements-dev.txt" +REAL_VENV_PYTHON = LAB_ROOT / "app" / ".venv" / "bin" / "python" + + +def _fingerprint(path: Path): + """(is_symlink, symlink target, sha256 of the resolved file's content). + + Sensitive to a swapped-in regular file, a retargeted symlink, or edited + content -- any of which is exactly what a fake `uv` run against the real + `app/.venv` in place would do. `(None, None, None)` means the path is + simply absent (also a legitimate, distinct fingerprint). + """ + if not path.exists() and not path.is_symlink(): + return (None, None, None) + is_link = path.is_symlink() + target = os.readlink(path) if is_link else None + resolved = path.resolve() + digest = hashlib.sha256(resolved.read_bytes()).hexdigest() if resolved.is_file() else None + return (is_link, target, digest) + + +@pytest.fixture(autouse=True) +def real_lab_venv_is_never_touched(): + """Regression tripwire for the vulnerability every fixture below fixes. + + Fingerprints the real, developer-machine `app/.venv/bin/python` before + and after each test and fails loudly if it ever changes -- type + (symlink vs. regular file), symlink target, or content hash. This must + never fire; if it does, a test in this file stopped running against a + `lab_copy` and started running the real script in place again. + """ + before = _fingerprint(REAL_VENV_PYTHON) + yield + after = _fingerprint(REAL_VENV_PYTHON) + assert after == before, ( + "a test in this file mutated the REAL app/.venv/bin/python " + f"(before={before!r} after={after!r}); every test must run a " + "lab_copy of setup-venv.sh under tmp_path, never the real script " + "in place" + ) + + +@pytest.fixture +def lab_copy(tmp_path): + """A throwaway copy of exactly the layout `setup-venv.sh` depends on: + itself under `scripts/`, plus `app/requirements.txt` and + `app/requirements-dev.txt` under `app/`. `app/.venv` is deliberately + never copied -- the script creates it fresh -- so nothing here ever + reads or writes the real one. Returns the path to the copied script. + """ + scripts_dir = tmp_path / "scripts" + scripts_dir.mkdir() + script_copy = scripts_dir / "setup-venv.sh" + shutil.copy2(SCRIPT, script_copy) + script_copy.chmod(script_copy.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + + app_dir = tmp_path / "app" + app_dir.mkdir() + shutil.copy2(REQUIREMENTS, app_dir / "requirements.txt") + shutil.copy2(REQUIREMENTS_DEV, app_dir / "requirements-dev.txt") + + return script_copy def _write_executable(path: Path, source: str) -> None: @@ -25,7 +104,8 @@ def _write_executable(path: Path, source: str) -> None: def _write_fake_uv(bin_dir: Path, log_path: Path, venv_exit: int = 0, pip_exit: int = 0, pillow_importable: bool = True): """A fake `uv` that logs every invocation and creates just enough of a venv (`bin/python` as a real, runnable interpreter) that the script's - own Pillow-import check can run against it.""" + own Pillow-import check can run against it. Only ever pointed at a + `lab_copy`'s `tmp_path`-scoped `app/.venv`, never the real one.""" real_python = os.environ.get("SETUP_VENV_TEST_PYTHON", "python3") stub = bin_dir / "uv" _write_executable( @@ -64,7 +144,30 @@ def _write_fake_uv(bin_dir: Path, log_path: Path, venv_exit: int = 0, pip_exit: ) -def _run(bin_dir: Path, workdir: Path, extra_path: bool = True): +def _write_fake_uv_no_matching_python(bin_dir: Path, log_path: Path): + """A fake `uv` reproducing its real message when no installed Python + clears the requested floor (verified against real `uv 0.10.9`): `uv + venv` fails with `error: No interpreter found for Python >=X ...` on + stderr, before ever reaching `uv pip install`.""" + stub = bin_dir / "uv" + _write_executable( + stub, + f"""#!/usr/bin/env bash +printf '%s\\n' "$*" >> "{log_path}" +case "$1" in + venv) + echo "error: No interpreter found for Python >=3.10 in managed installations or search path" >&2 + exit 1 + ;; + *) + exit 0 + ;; +esac +""", + ) + + +def _run(script: Path, bin_dir: Path, workdir: Path, extra_path: bool = True): env = dict(os.environ) if extra_path: env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" @@ -74,7 +177,7 @@ def _run(bin_dir: Path, workdir: Path, extra_path: bool = True): # `uv` is reachable. env["PATH"] = "/usr/bin:/bin" return subprocess.run( - [str(SCRIPT)], + [str(script)], capture_output=True, text=True, env=env, @@ -82,8 +185,8 @@ def _run(bin_dir: Path, workdir: Path, extra_path: bool = True): ) -def test_fails_actionably_with_no_pip_fallback_when_uv_is_missing(tmp_path): - result = _run(tmp_path / "bin-unused", tmp_path, extra_path=False) +def test_fails_actionably_with_no_pip_fallback_when_uv_is_missing(tmp_path, lab_copy): + result = _run(lab_copy, tmp_path / "bin-unused", tmp_path, extra_path=False) assert result.returncode != 0 assert "uv" in result.stderr @@ -92,13 +195,13 @@ def test_fails_actionably_with_no_pip_fallback_when_uv_is_missing(tmp_path): assert "already deployed" in result.stderr -def test_succeeds_and_calls_uv_venv_then_uv_pip_install_with_requirements_dev(tmp_path): +def test_succeeds_and_calls_uv_venv_then_uv_pip_install_with_requirements_dev(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" _write_fake_uv(bin_dir, log_path) - result = _run(bin_dir, tmp_path) + result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode == 0, result.stdout + result.stderr calls = log_path.read_text() @@ -108,8 +211,15 @@ def test_succeeds_and_calls_uv_venv_then_uv_pip_install_with_requirements_dev(tm assert "pip install --python" in calls assert "requirements-dev.txt" in calls + # The venv setup-venv.sh created must live under the throwaway copy, + # never under the real lab tree. + created_python = lab_copy.parents[1] / "app" / ".venv" / "bin" / "python" + assert created_python.is_file(), "the fake venv was not created under lab_copy's tmp_path" + assert str(created_python).startswith(str(tmp_path)) + assert not str(created_python).startswith(str(LAB_ROOT)) + -def test_never_falls_back_to_a_bare_pip_binary(tmp_path): +def test_never_falls_back_to_a_bare_pip_binary(tmp_path, lab_copy): """Even when a bare `pip` is reachable on PATH, the script must drive installation only through `uv pip install`.""" bin_dir = tmp_path / "bin" @@ -122,32 +232,32 @@ def test_never_falls_back_to_a_bare_pip_binary(tmp_path): f"#!/usr/bin/env bash\nprintf '%s\\n' \"$*\" >> \"{bare_pip_log}\"\nexit 0\n", ) - result = _run(bin_dir, tmp_path) + result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode == 0, result.stdout + result.stderr assert not bare_pip_log.exists(), "setup-venv.sh must never invoke a bare pip" -def test_is_idempotent_across_repeated_runs(tmp_path): +def test_is_idempotent_across_repeated_runs(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" _write_fake_uv(bin_dir, log_path) - first = _run(bin_dir, tmp_path) - second = _run(bin_dir, tmp_path) + first = _run(lab_copy, bin_dir, tmp_path) + second = _run(lab_copy, bin_dir, tmp_path) assert first.returncode == 0, first.stdout + first.stderr assert second.returncode == 0, second.stdout + second.stderr -def test_fails_actionably_when_uv_venv_creation_fails(tmp_path): +def test_fails_actionably_when_uv_venv_creation_fails(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" _write_fake_uv(bin_dir, log_path, venv_exit=1) - result = _run(bin_dir, tmp_path) + result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 assert "azd hooks run postprovision" in result.stderr @@ -155,13 +265,13 @@ def test_fails_actionably_when_uv_venv_creation_fails(tmp_path): assert "pip install" not in log_path.read_text() -def test_fails_actionably_when_uv_pip_install_fails(tmp_path): +def test_fails_actionably_when_uv_pip_install_fails(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" _write_fake_uv(bin_dir, log_path, pip_exit=1) - result = _run(bin_dir, tmp_path) + result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 assert "azd hooks run postprovision" in result.stderr @@ -169,13 +279,13 @@ def test_fails_actionably_when_uv_pip_install_fails(tmp_path): assert "proxy" in result.stderr -def test_fails_actionably_when_pillow_is_still_not_importable(tmp_path): +def test_fails_actionably_when_pillow_is_still_not_importable(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" _write_fake_uv(bin_dir, log_path, pillow_importable=False) - result = _run(bin_dir, tmp_path) + result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 assert "Pillow" in result.stderr @@ -191,11 +301,101 @@ def test_script_never_executes_a_bare_pip_or_pip3_command(): """Every executable (non-comment, non-message) shell command line that starts with `pip`/`pip3` would be a fallback outside `uv`; only lines that start with `uv` are allowed to reach a `pip` subcommand.""" - import re - for line in SCRIPT.read_text().splitlines(): stripped = line.strip() if not stripped or stripped.startswith(("#", "echo")): continue if re.match(r"^(if\s+!\s+)?pip3?\b", stripped): raise AssertionError(("bare pip invocation found", stripped)) + + +# --- Finding #2: postprovision ordering/recovery contract ----------------- +# +# `setup-venv.sh` runs before `azd-postprovision.sh`'s ACR build and +# Container App update (see `test_azd_hooks.py`'s +# `test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call`), so a +# failure here means the cloud *app* deployment for this run has not +# happened yet, only the Bicep infrastructure has. Re-running just this +# script would silently leave that deployment never attempted; every +# failure hint this script prints must send the operator to re-run the +# whole hook, and must never recommend running this script directly. +# (Doctor/capture-scenario's *own* remediation text is a separate, +# still-valid case for a local-only venv problem noticed well after a +# deployment already succeeded -- see test_lab_guides.py / +# test_doctor.py -- this contract is only about setup-venv.sh's own +# messages.) + + +def _assert_hint_never_recommends_direct_rerun(stderr: str): + assert "azd hooks run postprovision" in stderr + lowered = stderr.lower() + assert "run: ./scripts/setup-venv.sh" not in lowered + assert "directly: ./scripts/setup-venv.sh" not in lowered + assert "or ./scripts/setup-venv.sh" not in lowered + + +def test_missing_uv_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): + result = _run(lab_copy, tmp_path / "bin-unused", tmp_path, extra_path=False) + + assert result.returncode != 0 + _assert_hint_never_recommends_direct_rerun(result.stderr) + + +def test_venv_creation_failure_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path, venv_exit=1) + + result = _run(lab_copy, bin_dir, tmp_path) + + assert result.returncode != 0 + _assert_hint_never_recommends_direct_rerun(result.stderr) + + +def test_pip_install_failure_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path, pip_exit=1) + + result = _run(lab_copy, bin_dir, tmp_path) + + assert result.returncode != 0 + _assert_hint_never_recommends_direct_rerun(result.stderr) + + +def test_pillow_failure_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv(bin_dir, log_path, pillow_importable=False) + + result = _run(lab_copy, bin_dir, tmp_path) + + assert result.returncode != 0 + _assert_hint_never_recommends_direct_rerun(result.stderr) + + +# --- Finding #3: actionable "no matching Python" guidance ------------------ + + +def test_fails_actionably_with_uv_python_install_guidance_when_no_matching_python_is_found(tmp_path, lab_copy): + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + log_path = tmp_path / "uv-calls.log" + _write_fake_uv_no_matching_python(bin_dir, log_path) + + result = _run(lab_copy, bin_dir, tmp_path) + + assert result.returncode != 0 + assert "No interpreter found" in result.stderr + assert "uv python install 3.12" in result.stderr + assert "proxy" in result.stderr.lower() or "mirror" in result.stderr.lower() + assert "pip install" not in log_path.read_text(), ( + "a missing interpreter must never fall through to uv pip install" + ) + assert "pypi.org" not in result.stderr.lower() + assert not re.search(r"(? Date: Fri, 14 Aug 2026 21:13:09 +0900 Subject: [PATCH 15/26] fix(sre-lab): stop test_azd_hooks.py from executing the real in-place hooks The remaining Task 6 blocking finding: test_azd_postprovision_reports_a _clear_error_when_the_azure_cli_is_not_logged_in executed the real, in-place azd-postprovision.sh via _run_hook_script(AZD_POSTPROVISION, ...), only pointing az stubs and AZURE_* env vars at tmp_path. Since setup-venv.sh (called by azd-postprovision.sh before any Azure CLI call) resolves its own SCRIPT_DIR/LAB_ROOT/VENV_DIR from its own ${BASH_SOURCE[0]}, never the caller's cwd, that always ran the real, system uv against the real, developer-machine app/.venv -- a genuine network/package-index call and filesystem mutation, unrelated to the login-failure behaviour the test claims to check. The same _run_hook_script helper, and _run_azd_configure, also ran azd-configure.sh in place (lower risk, since it never touches setup-venv.sh/uv, but still not isolated). Every test in test_azd_hooks.py that executes a hook script now runs it from a new lab_copy fixture: a throwaway copy of azd-configure.sh, azd-postprovision.sh, and the real setup-venv.sh under scripts/, the real app/requirements.txt / requirements-dev.txt under app/, and a placeholder azure.yaml at the copied root -- laid out exactly as the real lab does, so every script's own path resolution lands entirely inside tmp_path. The postprovision login-failure test now puts a fake uv (logging every call, never touching a real network or a real venv) on PATH alongside the login-failing fake az, so the real, copied setup-venv.sh genuinely runs its uv-venv/uv-pip-install/Pillow-import logic against the fake before the login check fails -- the uv call log assertion ("venv" in uv_calls, "pip install" in uv_calls) proves the fake, not the real, uv did that work, and the created venv is asserted to live under tmp_path, never under the real lab tree. A new module-wide autouse fixture, real_lab_venv_tree_is_never_touched, fingerprints the real app/.venv tree before and after every test in this file: a manifest of every entry's relative path, kind (file/dir/symlink), symlink target, size, and mtime, hashed into one digest, plus a separate byte-content sha256 of bin/python's resolved target. The manifest catches whole-tree structural/symlink/size/mtime drift that a single canary file could miss; the interpreter content hash independently catches a content-only mutation that happens to preserve size and mtime, which the manifest alone would not (verified in isolation: a same-size, same-mtime content change is still detected via the interpreter hash). mtime is only ever one signal among several, never checked alone. test_no_execution_helper_runs_a_real_in_place_hook_script is a second, purely static tripwire: a source-text regex scan of this file itself (no subprocess, no filesystem access outside __file__) that fails if any test again passes the real, in-place AZD_CONFIGURE/AZD_POSTPROVISION/ SETUP_VENV path constants straight to subprocess.run or _run_hook_script. Confirmed RED against the pre-fix, git-committed version of this file via this same safe regex scan (no execution): it matched the AZD_CONFIGURE and AZD_POSTPROVISION call sites inside _run_hook_script, and the AZD_CONFIGURE call site inside _run_azd_configure. GREEN now: all patterns absent from the rewritten file, and the meta-test itself passes. Verified: pytest scripts/tests/test_azd_hooks.py -- 14 passed. Full suite: pytest scripts/tests -- 391 passed; pytest infra/tests -- 21 passed; app/.venv/bin/python -m pytest app -- 10 passed (422 total). Real app/.venv confirmed byte-for-byte and structurally unchanged throughout: bin/python sha256 and file count (14735 entries) identical before and after every run, and Pillow still imports from it afterward. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../scripts/tests/test_azd_hooks.py | 351 ++++++++++++++++-- 1 file changed, 315 insertions(+), 36 deletions(-) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py index aaab949..f8b7d66 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py @@ -5,14 +5,40 @@ necessarily the subscription azd is deploying into. Every Azure CLI operation therefore has to be pinned to AZURE_SUBSCRIPTION_ID, and the `predown` hook has to survive a lab that never configured the Agent. + +Every test that *executes* a hook script runs it from a `lab_copy` -- +never the real, in-place `azd-configure.sh` / `azd-postprovision.sh` / +`setup-venv.sh`. `azd-postprovision.sh` calls `setup-venv.sh`, and +`setup-venv.sh` resolves its own `SCRIPT_DIR`/`LAB_ROOT`/`VENV_DIR` from +its own `${BASH_SOURCE[0]}`, never from the caller's cwd -- so running the +real, in-place `azd-postprovision.sh` (as this file once did, pointing +only the *subprocess's* cwd or `AZURE_*` environment at a scratch +`tmp_path`) still always resolves `VENV_DIR` to the real, developer-machine +`app/.venv`, and setup-venv.sh's `uv venv --allow-existing` / `uv pip +install` would then run for real against it -- a real filesystem mutation +and a real network/package-index call, entirely unrelated to what the test +claims to be checking (the Azure CLI login-failure message). `lab_copy` +copies the whole `scripts/`+`app/` layout the scripts depend on into +`tmp_path`, so every script's own path resolution lands entirely inside +`tmp_path`; `real_lab_venv_tree_is_never_touched` (autouse, this whole +file) is the regression tripwire proving it -- it fingerprints the real +`app/.venv` tree before and after every test here and fails loudly on any +drift; `test_no_execution_helper_runs_a_real_in_place_hook_script` is a +second, purely static tripwire (a source-text scan of this very file, no +subprocess involved) that keeps the same vulnerability from silently +coming back. """ +import hashlib import os import re +import shutil import stat import subprocess from pathlib import Path +from types import SimpleNamespace +import pytest from azd_fake import write_azd_stub @@ -20,12 +46,146 @@ LAB_ROOT = Path(__file__).parents[2] AZD_CONFIGURE = SCRIPTS_DIR / "azd-configure.sh" AZD_POSTPROVISION = SCRIPTS_DIR / "azd-postprovision.sh" +SETUP_VENV = SCRIPTS_DIR / "setup-venv.sh" +REQUIREMENTS = LAB_ROOT / "app" / "requirements.txt" +REQUIREMENTS_DEV = LAB_ROOT / "app" / "requirements-dev.txt" +REAL_VENV_DIR = LAB_ROOT / "app" / ".venv" SUBSCRIPTION_PIN = '--subscription "${AZURE_SUBSCRIPTION_ID}"' # The one deliberately unpinned call: it reads whichever account is active # so the hook can report a mismatch. ACTIVE_ACCOUNT_PROBE = "az account show --query id" +# --- Regression tripwire: the real app/.venv must never move --------------- +# +# Two independent guards, deliberately redundant: +# 1. `real_lab_venv_tree_is_never_touched` (below) is an *execution-time* +# safety net: it fingerprints the real tree and fails if any test in +# this file ever changes it, no matter how that test is written. +# 2. `test_no_execution_helper_runs_a_real_in_place_hook_script` (further +# down) is a *static* safety net: it scans this file's own source for +# the exact patterns that caused the original vulnerability, and never +# executes anything -- so it is always safe to run, including on a +# version of this file that would otherwise mutate the real venv. + + +def _venv_tree_fingerprint(root: Path): + """A manifest-style fingerprint of the whole real `app/.venv` tree. + + For every entry under `root`, records its path relative to `root`, + whether it is a directory/file/symlink, its symlink target (if any), + its size, and its mtime for every entry -- then hashes the whole sorted + manifest into one digest, plus a byte-content sha256 of `bin/python`'s + *resolved* target (uv manages `bin/python` as a symlink to a real + interpreter binary that can live entirely outside `root`, e.g. under + `~/.local/share/uv/python/...`; `path.resolve()` follows it there). + This is deliberately a *tree* fingerprint, not a single file's: `uv + venv --allow-existing` / `uv pip install` mutate site-packages by + adding, removing, resizing, and retargeting many files and symlinks at + once, so a single canary file (or a bare directory mtime) could miss a + change a broader manifest catches. mtime is included as one signal + among several, deliberately not the only one -- comparing mtime alone + would be brittle (some legitimate changes leave mtime untouched at + second resolution; some incidental system activity touches mtime + without any content change) -- so a real regression must additionally + show up as a manifest entry that is new, missing, resized, retargeted, + or reclassified (file/dir/symlink), or as a changed interpreter content + hash, to be caught; mtime differences alone are deliberately not enough + for the assertion below to fire a false positive. + """ + if not root.exists(): + return ("absent", None, None, None) + entries = [] + for path in sorted(root.rglob("*")): + try: + st = path.lstat() + except OSError: + continue + is_link = path.is_symlink() + kind = "symlink" if is_link else ("dir" if path.is_dir() else "file") + target = os.readlink(path) if is_link else None + entries.append( + (str(path.relative_to(root)), kind, target, st.st_size, st.st_mtime_ns) + ) + manifest = repr(entries).encode() + + venv_python = root / "bin" / "python" + interpreter_digest = None + if venv_python.exists() or venv_python.is_symlink(): + resolved = venv_python.resolve() + if resolved.is_file(): + interpreter_digest = hashlib.sha256(resolved.read_bytes()).hexdigest() + + return ( + "present", + len(entries), + hashlib.sha256(manifest).hexdigest(), + interpreter_digest, + ) + + +@pytest.fixture(autouse=True) +def real_lab_venv_tree_is_never_touched(): + """Module-wide tripwire: fingerprints the real, developer-machine + `app/.venv` tree before and after every test in this file and fails + loudly if it ever changes. This must never fire; if it does, a test in + this file stopped running against a `lab_copy` and started running a + real hook script in place again -- exactly the vulnerability this + whole file's redesign fixes. + """ + before = _venv_tree_fingerprint(REAL_VENV_DIR) + yield + after = _venv_tree_fingerprint(REAL_VENV_DIR) + assert after == before, ( + "a test in this file mutated the REAL app/.venv tree " + f"(before={before!r} after={after!r}); every test that executes a " + "hook script must run it from a `lab_copy`, never the real script " + "in place" + ) + + +@pytest.fixture +def lab_copy(tmp_path): + """A throwaway copy of exactly the layout the hook scripts depend on: + `azd-configure.sh`, `azd-postprovision.sh`, and the *real* + `setup-venv.sh` under `scripts/`, the real `app/requirements.txt` and + `app/requirements-dev.txt` under `app/` (setup-venv.sh's own + `REQUIREMENTS_FILE`), and a placeholder `azure.yaml` at the copied lab + root (azd-configure.sh's fake-`azd` stub requires one to exist at + whatever `--cwd` it is given, matching the real lab layout). Every + script's own `SCRIPT_DIR`/`LAB_ROOT`/`APP_DIR`/`VENV_DIR` resolution + (from its own `${BASH_SOURCE[0]}`) therefore lands entirely inside + `tmp_path`, never inside the real lab tree. `app/.venv` is deliberately + never created here -- a script under test creates it fresh, against + whatever fake `uv` a given test puts on `PATH`. + """ + scripts_dir = tmp_path / "scripts" + scripts_dir.mkdir() + for source, name in ( + (AZD_CONFIGURE, "azd-configure.sh"), + (AZD_POSTPROVISION, "azd-postprovision.sh"), + (SETUP_VENV, "setup-venv.sh"), + ): + copy = scripts_dir / name + shutil.copy2(source, copy) + copy.chmod(copy.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + + app_dir = tmp_path / "app" + app_dir.mkdir() + shutil.copy2(REQUIREMENTS, app_dir / "requirements.txt") + shutil.copy2(REQUIREMENTS_DEV, app_dir / "requirements-dev.txt") + + (tmp_path / "azure.yaml").write_text("name: sre-lab-hooktest\n") + + return SimpleNamespace( + root=tmp_path, + configure=scripts_dir / "azd-configure.sh", + postprovision=scripts_dir / "azd-postprovision.sh", + setup_venv=scripts_dir / "setup-venv.sh", + app_dir=app_dir, + ) + + def _az_invocations(script_text): """Every `az ...` command in a script, with line continuations joined.""" joined = re.sub(r"\\\n\s*", " ", script_text) @@ -74,8 +234,47 @@ def _write_login_failing_az_stub(directory, log_path): return stub +def _write_fake_uv(directory, log_path): + """A fake `uv` that logs every invocation, creates a runnable stub + interpreter under whatever target directory `uv venv` is given, and + always succeeds. This lets a *real*, copied `setup-venv.sh` run its + genuine `uv venv` / `uv pip install` / Pillow-import logic end to end + without ever reaching the real `uv` binary, a real virtual environment, + or a real package index -- `log_path` is what each test's assertion + inspects to prove the fake, not the real, `uv` ran. + """ + stub = directory / "uv" + stub.write_text( + "#!/usr/bin/env bash\n" + f'printf "%s\\n" "$*" >> "{log_path}"\n' + 'case "$1" in\n' + " venv)\n" + ' target="${@: -1}"\n' + ' mkdir -p "${target}/bin"\n' + " cat > \"${target}/bin/python\" <<'PYEOF'\n" + "#!/usr/bin/env bash\n" + 'exit 0\n' + "PYEOF\n" + ' chmod +x "${target}/bin/python"\n' + " ;;\n" + " pip)\n" + " exit 0\n" + " ;;\n" + " *)\n" + " exit 0\n" + " ;;\n" + "esac\n" + ) + stub.chmod(stub.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + return stub + + def _run_hook_script(script_path, tmp_path, az_stub_factory, environment=None): - """Execute an azd hook script with a controllable fake `az` on PATH.""" + """Execute a *copied* azd hook script with a controllable fake `az` on + PATH. `script_path` must come from a `lab_copy` -- see this module's + top-of-file docstring and `test_no_execution_helper_runs_a_real_in_place_hook_script` + for why the real, in-place script must never be passed here. + """ bin_dir = tmp_path / "bin" bin_dir.mkdir(exist_ok=True) log_path = tmp_path / "az-calls.log" @@ -97,6 +296,44 @@ def _run_hook_script(script_path, tmp_path, az_stub_factory, environment=None): return result, calls +def test_no_execution_helper_runs_a_real_in_place_hook_script(): + """Regression tripwire for the vulnerability every `lab_copy`-based test + below fixes: executing the real, in-place `AZD_CONFIGURE` / + `AZD_POSTPROVISION` / `SETUP_VENV` path constants (as this file once + did) always resolves those scripts' own `SCRIPT_DIR`/`LAB_ROOT`/ + `VENV_DIR` to the real, developer-machine `app/.venv`, regardless of + what `tmp_path`, `cwd`, or `AZURE_*` environment a test additionally + sets up -- only copying the whole `scripts/`+`app/` tree elsewhere + (`lab_copy`) actually moves that resolution. This is a pure + source-text scan of this file itself: it runs no subprocess and + touches no filesystem outside its own `__file__`, so it is always safe + to run -- unlike the execution pattern it guards against. Confirmed in + session: run against the pre-fix version of this file (the + git-committed HEAD revision before this test existed), every one of + `_run_hook_script`'s two direct hook-script call sites and + `_run_azd_configure`'s own subprocess call site matched, because that + revision passed the real, in-place path constants straight to a + subprocess line-wrapped across multiple lines. Patterns below are + regexes, not bare substrings, specifically so a call site line-wrapped + that way cannot dodge the scan by reformatting. + """ + source = Path(__file__).read_text() + forbidden_execution_patterns = ( + r"_run_hook_script\(\s*AZD_CONFIGURE\b", + r"_run_hook_script\(\s*AZD_POSTPROVISION\b", + r"_run_hook_script\(\s*SETUP_VENV\b", + r"\[\s*str\(\s*AZD_CONFIGURE\s*\)\s*\]", + r"\[\s*str\(\s*AZD_POSTPROVISION\s*\)\s*\]", + r"\[\s*str\(\s*SETUP_VENV\s*\)\s*\]", + ) + for pattern in forbidden_execution_patterns: + assert not re.search(pattern, source), ( + f"found a real, in-place hook script execution pattern {pattern!r} " + "in this test file; every executed hook script must come from " + "the `lab_copy` fixture instead" + ) + + def test_azd_configure_pins_every_azure_cli_call_to_the_target_subscription(): for command in _az_invocations(AZD_CONFIGURE.read_text()): if command.startswith(ACTIVE_ACCOUNT_PROBE): @@ -171,25 +408,20 @@ def test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call(): ) -def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp_path): +def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp_path, lab_copy): """`app/.venv` setup is local and has nothing to do with the Azure CLI, but a broken corporate proxy or missing `uv` must still stop the hook before it spends a single Azure API call -- the cloud side is already provisioned by the time this hook runs, so failing fast here changes nothing about that, but a failure must never be masked by continuing on to the ACR build.""" - scripts_copy = tmp_path / "scripts" - scripts_copy.mkdir() - (scripts_copy / "azd-postprovision.sh").write_text(AZD_POSTPROVISION.read_text()) - (scripts_copy / "azd-postprovision.sh").chmod(0o755) - fake_setup_venv = scripts_copy / "setup-venv.sh" - fake_setup_venv.write_text( + lab_copy.setup_venv.write_text( "#!/usr/bin/env bash\n" "echo 'uv is required to set up app/.venv but was not found on PATH.' >&2\n" "echo 'azd hooks run postprovision' >&2\n" "exit 1\n" ) - fake_setup_venv.chmod(0o755) + lab_copy.setup_venv.chmod(0o755) bin_dir = tmp_path / "bin" bin_dir.mkdir() @@ -205,7 +437,7 @@ def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp env["AZURE_CONTAINER_APP_FQDN"] = "ca-test.example.com" result = subprocess.run( - [str(scripts_copy / "azd-postprovision.sh")], + [str(lab_copy.postprovision)], capture_output=True, text=True, env=env, @@ -216,13 +448,13 @@ def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp assert "uv" in result.stderr -def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): +def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path, lab_copy): """`az account show` fails with a generic Azure CLI error when signed out. Guard it so the hook fails fast with one unambiguous message instead of raw CLI stderr or an unexplained `set -e` abort. """ result, calls = _run_hook_script( - AZD_CONFIGURE, tmp_path, _write_login_failing_az_stub + lab_copy.configure, tmp_path, _write_login_failing_az_stub ) assert result.returncode != 0 @@ -234,34 +466,80 @@ def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in ) -def test_azd_postprovision_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): - environment = { - "AZURE_RESOURCE_GROUP": "rg-test", - "AZURE_ACR_NAME": "acrtest", - "AZURE_CONTAINER_APP_NAME": "ca-test", - "AZURE_CONTAINER_APP_FQDN": "ca-test.example.com", - } - result, calls = _run_hook_script( - AZD_POSTPROVISION, tmp_path, _write_login_failing_az_stub, environment +def test_azd_postprovision_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path, lab_copy): + """`azd-postprovision.sh` runs `setup-venv.sh` *before* it ever checks + the Azure CLI login state (see + `test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call`), so + this test's `bin_dir` carries both a login-failing fake `az` and a + fake `uv` -- the real, copied `setup-venv.sh` genuinely runs its own + `uv venv` / `uv pip install` / Pillow-import logic against the fake + `uv` (never a stub that skips that logic outright), and only *then* + does the hook reach and fail the login check. `uv_calls` is the + call-log assertion proving the fake -- never the real -- `uv` did that + work, so no real virtual environment or package index was ever + touched. + """ + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + az_log = tmp_path / "az-calls.log" + uv_log = tmp_path / "uv-calls.log" + _write_login_failing_az_stub(bin_dir, az_log) + _write_fake_uv(bin_dir, uv_log) + + env = dict(os.environ) + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" + env["AZURE_SUBSCRIPTION_ID"] = "11111111-2222-3333-4444-555555555555" + env["AZURE_RESOURCE_GROUP"] = "rg-test" + env["AZURE_ACR_NAME"] = "acrtest" + env["AZURE_CONTAINER_APP_NAME"] = "ca-test" + env["AZURE_CONTAINER_APP_FQDN"] = "ca-test.example.com" + + result = subprocess.run( + [str(lab_copy.postprovision)], + capture_output=True, + text=True, + env=env, ) assert result.returncode != 0 assert "az login" in result.stderr assert "Please run 'az login' to setup account." not in result.stderr - assert calls.strip() == "account show --query id -o tsv", ( + az_calls = az_log.read_text() if az_log.exists() else "" + assert az_calls.strip() == "account show --query id -o tsv", ( "the hook must exit immediately after the failed login check, " - f"before any other az call: {calls!r}" + f"before any other az call: {az_calls!r}" ) + uv_calls = uv_log.read_text() if uv_log.exists() else "" + assert "venv" in uv_calls, ( + "setup-venv.sh must have run its real uv-venv logic against the " + f"fake uv before the login check failed: {uv_calls!r}" + ) + assert "pip install" in uv_calls, ( + "setup-venv.sh must have run its real uv-pip-install logic against " + f"the fake uv before the login check failed: {uv_calls!r}" + ) + created_python = lab_copy.app_dir / ".venv" / "bin" / "python" + assert created_python.is_file(), ( + "setup-venv.sh must have created its venv under lab_copy's " + "tmp_path, proving the fake uv -- not the real one -- ran" + ) + assert not str(created_python).startswith(str(LAB_ROOT)), ( + "the venv setup-venv.sh created must never live under the real lab tree" + ) -def _run_azd_configure(tmp_path, azd_values, missing_key_mode="azd_1_29"): - """Run the preprovision hook with a fake `az` and a realistic fake `azd`. +def _run_azd_configure(tmp_path, lab_copy, azd_values, missing_key_mode="azd_1_29"): + """Run the preprovision hook (a `lab_copy` copy) with a fake `az` and a + realistic fake `azd`. The fake `azd` reproduces azd 1.29.0: it reports a value it does not have with `ERROR: ...` on **stdout** and exit 1, and resolves the project from `--cwd` (else the process working directory). The hook is - started from a scratch directory that holds no `azure.yaml`. + started from a scratch directory that holds no `azure.yaml` -- distinct + from `lab_copy.root`, which does (matching the real lab layout), so + that `--cwd "${LAB_ROOT}"` inside the copied script is what makes the + lookups succeed, not the process's own cwd. """ bin_dir = tmp_path / "bin" bin_dir.mkdir(exist_ok=True) @@ -278,7 +556,7 @@ def _run_azd_configure(tmp_path, azd_values, missing_key_mode="azd_1_29"): env["AZURE_SUBSCRIPTION_ID"] = "11111111-2222-3333-4444-555555555555" result = subprocess.run( - [str(AZD_CONFIGURE)], + [str(lab_copy.configure)], capture_output=True, text=True, env=env, @@ -287,7 +565,7 @@ def _run_azd_configure(tmp_path, azd_values, missing_key_mode="azd_1_29"): return result, (azd_log.read_text() if azd_log.exists() else "") -def test_azd_configure_defaults_the_resource_group_when_azd_has_no_value(tmp_path): +def test_azd_configure_defaults_the_resource_group_when_azd_has_no_value(tmp_path, lab_copy): """A key azd does not have must read as absent, not as azd's error text. azd 1.29 prints `ERROR: ...` on stdout while exiting non-zero, so a @@ -295,7 +573,7 @@ def test_azd_configure_defaults_the_resource_group_when_azd_has_no_value(tmp_pat the default the hook is there to write. """ result, azd_calls = _run_azd_configure( - tmp_path, {"AZURE_ENV_NAME": "sre-lab-hooktest"} + tmp_path, lab_copy, {"AZURE_ENV_NAME": "sre-lab-hooktest"} ) assert result.returncode == 0, result.stderr @@ -308,9 +586,10 @@ def test_azd_configure_defaults_the_resource_group_when_azd_has_no_value(tmp_pat ) -def test_azd_configure_keeps_values_the_azd_environment_already_has(tmp_path): +def test_azd_configure_keeps_values_the_azd_environment_already_has(tmp_path, lab_copy): result, azd_calls = _run_azd_configure( tmp_path, + lab_copy, { "AZURE_ENV_NAME": "sre-lab-hooktest", "AZURE_RESOURCE_GROUP": "rg-chosen-by-the-operator", @@ -325,24 +604,24 @@ def test_azd_configure_keeps_values_the_azd_environment_already_has(tmp_path): assert "env set SRE_LAB_EXPIRES_ON" not in azd_calls -def test_azd_configure_pins_every_azd_lookup_to_the_lab_project(tmp_path): +def test_azd_configure_pins_every_azd_lookup_to_the_lab_project(tmp_path, lab_copy): """azd hooks are also run by hand while debugging a lab, so the hook must not depend on the working directory it inherits.""" result, azd_calls = _run_azd_configure( - tmp_path, {"AZURE_ENV_NAME": "sre-lab-hooktest"} + tmp_path, lab_copy, {"AZURE_ENV_NAME": "sre-lab-hooktest"} ) assert result.returncode == 0, result.stderr - assert f"cwd={LAB_ROOT}" in azd_calls, ( - f"azd was not pinned to the lab project root: {azd_calls!r}" + assert f"cwd={lab_copy.root}" in azd_calls, ( + f"azd was not pinned to the copied lab project root: {azd_calls!r}" ) assert "no project exists" not in azd_calls -def test_azd_configure_refuses_to_derive_a_resource_group_without_an_environment(tmp_path): +def test_azd_configure_refuses_to_derive_a_resource_group_without_an_environment(tmp_path, lab_copy): """Deriving `rg-` from an unavailable environment name would create a resource group nobody can identify; the hook must stop instead.""" - result, azd_calls = _run_azd_configure(tmp_path, {}) + result, azd_calls = _run_azd_configure(tmp_path, lab_copy, {}) assert result.returncode != 0 assert "azd env new" in result.stderr From dab05d64abbc3468b0fe82f639fdee8b9af1ef0e Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 22:19:01 +0900 Subject: [PATCH 16/26] fix(sre-lab): gate the app deployment on AcrPull in a deploy-phase hook The lab is a Container Apps workload pulling from ACR with a user-assigned managed identity, which cannot be provisioned and deployed in one step: the lab image does not exist until the registry builds it, and the AcrPull assignment the same ARM deployment creates is not necessarily usable by a pull issued immediately afterwards. The single postprovision hook did both halves -- setup-venv.sh, then az acr build, az containerapp ingress update and az containerapp update --image -- so the very first pull raced the role assignment, with no check at all that the identity could pull yet. That is the live-deployment blocker the Azure-deploy preflight flags. The two phases are now separate, and the gate sits between them: * postprovision -> scripts/azd-postprovision-local.sh: setup-venv.sh and nothing else. It makes zero Azure calls, so `azd provision` ends with the public placeholder image (ingress 80, no probes) still serving, and says so on stdout ("Next: azd deploy"). * postdeploy -> scripts/azd-deploy-app.sh: guards the seven provision outputs it consumes (naming `azd provision` when one is missing), checks the CLI login, resolves the registry, then polls `az role assignment list` for the *exact* AcrPull assignment -- this principal, role definition 7f951dda-4ed3-4680-a7ca-43fe172d538d, the registry's own scope, compared case-insensitively -- for up to 300s (SRE_ACR_PULL_TIMEOUT_SECONDS, 10s interval) before its first deploy action. Only then: az acr build from app/ (never a local docker build), containerapp registry set --identity , ingress update --target-port 8000, containerapp update --image, wait for a new Healthy+active revision, wait for /healthz 200, and finally `azd env set SRE_IMAGE_TAG` / `SRE_CONTAINER_IMAGE` (--cwd-pinned). If the grant never appears, nothing is built or updated and the hook fails naming AcrPull and the scope it polled. `azd up` runs the same postdeploy hook in its deploy phase, so the documented single command still passes through the gate. azd 1.29 behaviour was verified rather than assumed, because the project declares no services (the only applicable service host, containerapp + docker, would force a local Docker build): `azd deploy --no-prompt` against a marker-hook copy of this exact azure.yaml shape ran postdeploy, and a hook exiting 7 failed the command with "failed running post hooks: 'postdeploy' hook failed with exit code: '7'". Running deploy before any provision fails fast with "infrastructure has not been provisioned" -- env.GetSubscriptionId() == "" in cli/azd/internal/cmd/deploy.go, not an ARM query -- and cli/azd/internal/cmd/up_graph.go (tag azure-dev-cli_1.29.0) adds cmdhook-predeploy/cmdhook-postdeploy unconditionally with an explicit "Zero-service projects" branch, so `azd up` reaches the same hook. No service definition was needed. main.bicep now publishes AZURE_CONTAINER_APP_PRINCIPAL_ID, AZURE_WORKLOAD_IDENTITY_RESOURCE_ID and AZURE_ACR_LOGIN_SERVER (azd stores outputs under the template's declared names), with workloadIdentityResourceId threaded through workload.bicep and lab.bicep. Tests came first. A new deploy_app_harness.py runs the hook as a program against a fake `az` that models the tenant instead of logging and exiting 0: role assignments are records, `role assignment list` applies azure-cli's assignee filter, exact-scope match (and refuses --include-inherited) and the ends_with(roleDefinitionId, ...) projection, available_after_attempts reproduces RBAC propagation delay, and containerapp update rolls a new revision whose health revision list reports. test_azd_deploy_app.py (25 tests) pins the ordering both in the source and in the recorded call log -- no deploy call before the grant is visible, three polls when the grant appears on the third, no build at all when it never appears, wider scope / AcrPush / another principal all rejected, a differently-cased scope accepted, and no image recorded when /healthz never goes green. test_azd_hooks.py now proves the provision phase makes no Azure call and builds nothing, and setup-venv.sh's rerun hint no longer claims an image build follows it in the same hook. README, doctor.sh, deploy.sh, cleanup.sh and cleanup-external.sh describe the two phases and the new script names; .azure/deployment-plan.md records the discovered gate and moves back to Status "Ready for Validation" -- nothing was deployed to Azure by this change. Validation: 455 tests, bash -n on every script, three Bicep builds, azure.yaml against the official azd v1.0 schema, azd package --all, and the azd deploy-hook probes above. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 374 +++++++--------- monitor/sre-agent-event-lab/README.md | 15 +- monitor/sre-agent-event-lab/azure.yaml | 7 +- monitor/sre-agent-event-lab/infra/lab.bicep | 1 + monitor/sre-agent-event-lab/infra/main.bicep | 6 + .../infra/tests/test_azd_project.py | 116 ++++- .../sre-agent-event-lab/infra/workload.bicep | 3 + .../scripts/azd-deploy-app.sh | 260 +++++++++++ .../scripts/azd-postprovision-local.sh | 35 ++ .../scripts/azd-postprovision.sh | 143 ------ .../scripts/cleanup-external.sh | 4 +- .../sre-agent-event-lab/scripts/cleanup.sh | 2 +- monitor/sre-agent-event-lab/scripts/deploy.sh | 8 +- monitor/sre-agent-event-lab/scripts/doctor.sh | 3 +- .../sre-agent-event-lab/scripts/setup-venv.sh | 38 +- .../scripts/tests/deploy_app_harness.py | 415 ++++++++++++++++++ .../scripts/tests/test_azd_deploy_app.py | 396 +++++++++++++++++ .../scripts/tests/test_azd_hooks.py | 253 +++++------ .../scripts/tests/test_cleanup_external.py | 4 +- .../scripts/tests/test_lab_guides.py | 44 +- .../scripts/tests/test_setup_venv.py | 76 ++-- 21 files changed, 1626 insertions(+), 577 deletions(-) create mode 100755 monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh create mode 100755 monitor/sre-agent-event-lab/scripts/azd-postprovision-local.sh delete mode 100755 monitor/sre-agent-event-lab/scripts/azd-postprovision.sh create mode 100644 monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py create mode 100644 monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 6956d92..fd849e4 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -1,253 +1,181 @@ # Azure Deployment Plan -> **Status:** Deployed +> **Status:** Ready for Validation -Generated: 2026-08-12T03:35:07Z +Updated: 2026-08-14 ---- +## Goal -## 1. Project Overview +Validate the Azure SRE Agent event lab as a repeatable guided exercise: -**Goal:** Deploy an isolated Azure Container Apps incident lab, connect Azure Monitor alerts to Azure SRE Agent, execute three deterministic failures, and publish an evidence-based analysis report. +1. deploy the disposable workload and monitoring stack with `azd up`; +2. connect Azure SRE Agent through consent-sensitive portal steps; +3. run S1/S2/S3 in order and capture explicit conclusions or missing states; +4. remove the environment with `azd down --purge`. -**Path:** Add Components - -**Design:** `monitor/sre-agent-event-lab/README.md` - -**Execution plan:** `monitor/sre-agent-event-lab/README.md` - ---- - -## 2. Requirements +## Deployment Context | Attribute | Value | |---|---| +| Mode | Modify existing lab | | Classification | Development / disposable incident lab | | Scale | Small | -| Budget | Cost-Optimized | -| Subscription | `95933ae5-0201-4a21-a1fc-8051a7437982` | +| Budget | Cost-optimized; delete immediately after validation | +| Subscription | Current authenticated subscription selected through `AZURE_SUBSCRIPTION_ID` | | Location | `koreacentral` | -| Resource group | `rg-sre-agent-event-lab-krc` | -| Required tags | `purpose=sre-agent-event-lab`, `expiresOn=2026-08-13` | +| Resource group | `rg-${AZURE_ENV_NAME}` unless explicitly overridden | +| Required tags | `purpose=sre-agent-event-lab`, `azd-env-name=${AZURE_ENV_NAME}`, one-day `expiresOn` | | Agent autonomy | Review | -The current Azure CLI context matches the recorded subscription. Korea Central supports Azure SRE Agent and the selected workload services. - ---- +No subscription ID, resource group, global name suffix, or expiry date is fixed in source. -## 3. Components Detected +## Components -| Component | Type | Technology | Path | +| Component | Azure service | +|---|---| +| Incident API | Azure Container Apps | +| Image build and storage | Azure Container Registry Basic, remote ACR build | +| Dependency failure target | Storage account with Blob private endpoint | +| Workload identity | User-assigned managed identity | +| Logs and traces | Log Analytics and workspace-based Application Insights | +| Incident detection | Three Sev2 scheduled-query alert rules | +| Incident analysis | Existing or disposable Azure SRE Agent in Review mode | + +## Deployment Recipe + +- Azure Developer CLI project: `monitor/sre-agent-event-lab/azure.yaml` +- Subscription-scope entry template: `monitor/sre-agent-event-lab/infra/main.bicep` +- Resource-group module: `monitor/sre-agent-event-lab/infra/lab.bicep` +- Parameters: `monitor/sre-agent-event-lab/infra/main.parameters.json` +- Preprovision: dependency/login/provider checks and non-secret defaults +- Postprovision (`scripts/azd-postprovision-local.sh`): local uv environment setup only; no Azure call, so `azd provision` ends with the public placeholder image still serving +- Postdeploy (`scripts/azd-deploy-app.sh`): wait for `AcrPull` propagation, then ACR remote build, registry/identity configuration, ingress move to 8000, Container App image update, revision and `/healthz` verification, and image persistence +- Predown: validate and delete only recorded external Monitoring Contributor role assignments +- Postdown: clear persisted image/tag environment values + +The standard exercise uses the Azure Monitor incident platform and Review-mode response plans. The historical Logic App/HTTP Trigger bridge is not deployed. + +## Two-Phase Container Apps + ACR Flow (deploy gate) + +Discovered while preparing live validation: this lab is a Container Apps +workload pulling from ACR with a user-assigned managed identity, which +cannot be provisioned and deployed in one step. + +- One ARM deployment creates the registry, the identity, the `AcrPull` + assignment and the Container App, but the lab image does not exist yet + (it is built *by* the registry), and a role assignment created moments + earlier is not necessarily usable by a pull issued immediately after. +- Therefore provisioning deploys a **public placeholder image** (ingress + 80, no probes) and performs no image work at all, and the deploy phase + performs every image action behind an explicit gate. + +| Phase | Command | What runs | What it must not do | |---|---|---|---| -| Incident lab API | API | Python 3.12, FastAPI, Azure Monitor OpenTelemetry | `monitor/sre-agent-event-lab/app/` | -| Azure infrastructure | IaC | Bicep | `monitor/sre-agent-event-lab/infra/` | -| Deployment and incident tooling | Operations | Azure CLI, Bash, Python | `monitor/sre-agent-event-lab/scripts/` | -| Agent knowledge | Runbook | Markdown | `monitor/sre-agent-event-lab/runbooks/` | -| Results | Report | Markdown | `monitor/sre-agent-event-lab/validation-results.md` | +| Provision | `azd provision` | Bicep (infra + placeholder app) → `postprovision` = `scripts/setup-venv.sh` only | No `az acr build`, no `az containerapp update`/`ingress update`/`registry set`, no image env values | +| Deploy | `azd deploy` | `postdeploy` = poll exact `AcrPull` (principal + role definition `7f951dda-4ed3-4680-a7ca-43fe172d538d` + registry scope) up to 300s, then ACR build → `registry set --identity` → `ingress update --target-port 8000` → `update --image` → new healthy revision → `/healthz` → `azd env set SRE_IMAGE_TAG`/`SRE_CONTAINER_IMAGE` | Nothing may run before the grant is observed; nothing is persisted unless `/healthz` returned 200 | + +`azd up` runs both phases in order, so the user-facing command stays a +single command and still passes through the same gate. + +Deploy-phase outputs consumed from the azd environment: +`AZURE_SUBSCRIPTION_ID`, `AZURE_RESOURCE_GROUP`, `AZURE_ACR_NAME`, +`AZURE_ACR_LOGIN_SERVER`, `AZURE_CONTAINER_APP_NAME`, +`AZURE_CONTAINER_APP_FQDN`, `AZURE_CONTAINER_APP_PRINCIPAL_ID`, +`AZURE_WORKLOAD_IDENTITY_RESOURCE_ID`. + +Behaviour verified against the installed azd (1.29.0), because the project +declares no services: + +- `azd deploy --no-prompt` on a project with this exact `azure.yaml` shape + (bicep `infra:`, five hooks, no `services:`) runs the `postdeploy` hook, + and a non-zero hook exit fails the command + (`ERROR: failed running post hooks: 'postdeploy' hook failed with exit + code: '7'`). +- `azd deploy` before any provision fails fast with + `ERROR: infrastructure has not been provisioned` — the guard is a plain + environment check (`env.GetSubscriptionId() == ""` in + `cli/azd/internal/cmd/deploy.go`), so the deploy phase is reachable + exactly when provisioning has run. +- `azd up` uses the same project hooks: `cli/azd/internal/cmd/up_graph.go` + adds `cmdhook-predeploy`/`cmdhook-postdeploy` unconditionally and keeps + the deploy events for "Zero-service projects". + +No `services:` entry is declared: the only service host that would apply +here (`containerapp` + `docker`) makes `azd deploy`/`azd up` require a +local Docker build, which this lab deliberately avoids. + +## Security and Safety + +- Container Apps reach Blob Storage through private networking. +- Managed identities and Azure RBAC are used; no credentials are stored in the repository or azd environment examples. +- Every destructive script verifies the current subscription, `purpose`, and `azd-env-name` tags. +- S3 removes only the recorded Blob Data Reader assignment and restores it through an exit trap. +- Scenario progression is bound to the current azd environment and requires workload recovery, alert resolution, and an explicit capture status. +- Cleanup validates recorded assignment subscription, principal, role definition, and scope before any deletion. + +## Role Assignment Verification ---- +- Status: Verified +- Identity checked: Container App user-assigned managed identity +- ACR access: `AcrPull` (`7f951dda-4ed3-4680-a7ca-43fe172d538d`) scoped to the lab registry +- Blob access: `Storage Blob Data Reader` (`2a2b9908-6ea1-4ae2-8e65-a410df84e7d1`) scoped to the single documents container +- Local developer data access: not required; baseline and scenarios call the public lab API rather than Blob data plane directly +- Issues: none -## 4. Recipe Selection +## Validation Plan -**Selected:** Bicep + AZCLI +### Preflight checks (re-run after the two-phase refactor) -**Rationale:** +- [x] 1. AZD Installation — azd 1.29.0 +- [x] 2. Schema Validation — official azd v1.0 schema; hooks include `postdeploy`, no `services` +- [ ] 3. Environment Setup — a fresh environment is needed for the live run (`sre-lab-08141227` predates this change) +- [x] 4. Authentication Check — Azure CLI and azd authenticated +- [x] 5. Subscription/Location Check — current authenticated subscription, Korea Central +- [x] 6. Aspire Pre-Provisioning Checks — not applicable +- [ ] 7. Provision Preview — to re-run against the updated template outputs +- [x] 8. Build Verification — 455 tests and three Bicep builds passed +- [x] 9. Docker Build Context Validation — Dockerfile and requirements present; the image is built by ACR from `app/`, never locally +- [x] 10. Package Validation — `azd package --all --no-prompt` passed +- [x] 11. Azure Policy Validation — three assigned Defender policies are unrelated to planned resources +- [x] 12. Aspire Post-Provisioning Checks — not applicable +- [x] 13. Deploy-Hook Reachability — `postdeploy` runs for this service-less project shape on azd 1.29.0, and its failure fails the command -- The repository already uses direct Bicep and Azure CLI patterns. -- The deployment requires a two-phase flow: base resources, ACR cloud build, then Container App and alert rules. -- The incident runner needs explicit control over revision configuration, RBAC removal/recovery, alert polling, evidence export, and cleanup. -- Local Docker is not required; `az acr build` performs the image build. +1. Run the complete pytest suite, Bash syntax checks, Python 3.9 imports, Bicep compilation, and azure.yaml schema validation. +2. Create a unique azd environment in Korea Central. +3. Run infrastructure preview and inspect the resource plan. +4. Run `azd up` (provision phase leaves the placeholder image; deploy phase waits for `AcrPull`, builds and switches the image), then `lab.sh doctor` and `lab.sh baseline`. +5. Complete the portal Agent setup guide and acknowledge it explicitly. +6. Run and capture S1, S2, and S3 sequentially; generate the scorecard. +7. Run `azd down --purge`. +8. Verify the resource group and recorded external assignments are absent. ---- +## Expected Cost -## 5. Architecture +Container Apps, ACR, Log Analytics/Application Insights, Storage, and Azure SRE Agent can incur charges. Use a uniquely named disposable environment and remove it immediately after validation. -**Stack:** Containers +## Section 7: Validation Proof -### Service Mapping +Re-run after the two-phase refactor (the previous run predates it, so the +earlier "Validated" status no longer applies): -| Component | Azure Service | SKU | +| Check | Command | Result | |---|---|---| -| FastAPI incident API | Azure Container Apps | Consumption, 0.5 CPU / 1 GiB, min 1, max 2 | -| Container image | Azure Container Registry | Basic | -| Blob dependency | Azure Storage | Standard_LRS StorageV2 | -| Central logs | Log Analytics | PerGB2018, 30-day retention | -| APM | Workspace-based Application Insights | Web | -| Incident detection | Azure Monitor scheduled-query rules | 3 Sev2 rules | -| Workload authentication | User-assigned managed identity | No SKU | -| Incident analysis | Azure SRE Agent | Korea Central, Review response plans | - -### Data and Incident Flow - -1. Public HTTPS requests reach the Container App. -2. OpenTelemetry exports request, exception, and Blob dependency telemetry to Application Insights. -3. Scheduled-query rules evaluate every minute over a five-minute window. -4. Azure SRE Agent's Azure Monitor scanner receives matching Sev2 incidents. -5. Response plans investigate with repository, runbook, resource, Activity Log, and observability context. -6. The operator records and scores the analysis, then verifies scenario recovery. - -### Security Controls - -- ACR admin access and anonymous pull are disabled. -- Container App pulls with a user-assigned managed identity and `AcrPull`. -- Blob access uses the same managed identity and `Storage Blob Data Reader` scoped to the `documents` container. -- Storage shared-key authentication and public blob access are disabled. -- Application Insights connection string is a Container Apps secret. -- Agent workload access is Reader; remediation remains in Review mode. -- Scenario and cleanup scripts refuse untagged or wrong-subscription targets. - ---- - -## 6. Provisioning Limit Checklist - -Quota CLI was used first for Microsoft.App and Microsoft.Storage. Azure Resource Graph plus official service-limit documentation was used where quota CLI is unsupported or no adjustable count quota exists. - -| Resource Type | Number to Deploy | Current in Korea Central | Total After Deployment | Limit/Quota | Source and result | -|---|---:|---:|---:|---:|---| -| `Microsoft.App/managedEnvironments` | 1 | 0 | 1 | 50 | `azure-quotas`, `ManagedEnvironmentCount`: sufficient | -| `Microsoft.Storage/storageAccounts` | 1 | 0 | 1 | 250 | `azure-quotas`, `StorageAccounts`: sufficient | -| `Microsoft.ContainerRegistry/registries` | 1 | 1 | 2 | 100 per subscription per region | quota CLI returned `BadRequest`; Azure Resource Graph + official limits: sufficient | -| `Microsoft.App/containerApps` | 1 | 0 | 1 | governed by Container Apps environment/consumption quotas | Azure Resource Graph + deployment validation: sufficient | -| `Microsoft.OperationalInsights/workspaces` | 1 | 1 | 2 | no adjustable regional creation quota exposed | Azure Resource Graph + deployment validation | -| `Microsoft.Insights/components` | 1 | 1 | 2 | no adjustable regional creation quota exposed | Azure Resource Graph + deployment validation | -| `Microsoft.Insights/scheduledQueryRules` | 3 | 0 | 3 | within Azure Monitor alert-rule limits | Azure Resource Graph + deployment validation | -| `Microsoft.ManagedIdentity/userAssignedIdentities` | 1 | 5 | 6 | no adjustable regional creation quota exposed | Azure Resource Graph + deployment validation | - -**Status:** All quota-metered resources are within limits. Resource names `acrsrelab95933ae5` and `stsrelab95933ae5` are globally available. The target resource group does not exist. - ---- - -## 7. Deployment Sequence - -1. Register `Microsoft.App`, `Microsoft.OperationalInsights`, `Microsoft.Insights`, `Microsoft.Storage`, `Microsoft.ContainerRegistry`, `Microsoft.ManagedIdentity`, and `Microsoft.AlertsManagement`. -2. Run local tests, shell parse checks, Bicep compilation, group validation, and deployment what-if. -3. Create the tagged resource group. -4. Deploy observability, ACR, Storage, managed identity, RBAC, and Container Apps environment with `deployContainerApp=false`. -5. Build `sre-event-lab:20260812.4` with ACR Tasks. -6. Deploy the Container App and three scheduled-query alert rules with `deployContainerApp=true`. -7. Poll revision health and `/healthz`. -8. Generate normal baseline request and dependency telemetry. -9. Create and configure Azure SRE Agent through `https://sre.azure.com`. -10. Execute S1, S2, and S3 sequentially with recovery gates. -11. Complete the report, remove the recorded Agent subscription role assignment, and delete the tagged resource group. - ---- - -## 8. Validation Criteria - -- Python tests pass with no failures. -- Shell scripts pass `bash -n`. -- Bicep compiles without warnings or errors. -- ARM validation and what-if succeed before resource creation. -- ACR cloud build succeeds. -- Container App revision is Healthy and `/healthz` returns HTTP 200. -- Normal AppRequests and AppDependencies telemetry arrives before incident injection. -- Each scenario produces the intended alert and returns to healthy state. -- Cleanup removes only the recorded role assignment and tagged lab resource group. - ---- - -## 9. Execution Checklist - -### Phase 1: Planning - -- [x] Analyze workspace -- [x] Gather requirements -- [x] Confirm subscription and location -- [x] Prepare resource inventory -- [x] Fetch quotas and validate capacity -- [x] Scan codebase -- [x] Select recipe -- [x] Plan architecture -- [x] User approved the design; unavailable review gates selected the recommended option under autopilot instructions - -### Phase 2: Preparation - -- [x] Install and update official `microsoft/azure-skills` globally -- [x] Research Azure SRE Agent, Container Apps, Application Insights, Azure Monitor alerts, Storage, and managed identity requirements -- [x] Generate application and tests -- [x] Generate Bicep infrastructure -- [x] Generate deployment, scenario, evidence, and cleanup tooling -- [x] Apply managed identity and least-privilege RBAC -- [x] Add runbook, operator guide, and results report -- [x] Set status to `Ready for Validation` - -### Phase 3: Validation and Deployment - -- [x] All validation checks pass - - [x] Core Validation (CLI, auth, build, validate, what-if) using the official `validate-deployment.sh` - - [x] Bicep compilation/lint validation - - [x] Azure Policy validation -- [ ] Fix all validation blockers and repeat the validation workflow -- [x] Invoke `azure-deploy` -- [x] Verify baseline telemetry -- [x] Configure Azure SRE Agent -- [x] Execute and score S1-S3 -- [x] Publish results and evidence captures -- [ ] Cleanup pending explicit confirmation; lab retained with `expiresOn=2026-08-13` - ---- - -## 10. Validation Proof - -Validated: 2026-08-12T03:40:00Z - -| Check | Result | Evidence | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 455 passed | +| Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | +| Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | +| Bicep build | `az bicep build --file infra/{main,lab,workload}.bicep --stdout` | Passed (three templates, with the new deploy-gate outputs) | +| AZD schema | `azure.yaml` validated against `schemas/v1.0/azure.yaml.json` from Azure/azure-dev | Passed; hooks = preprovision, postprovision, postdeploy, predown, postdown; no `services` | +| AZD package | `azd package --all --no-prompt` | Passed | +| Zero-service deploy hook | `azd deploy --no-prompt` against a marker-hook copy of this `azure.yaml` | `postdeploy` ran; hook exit 7 failed the command | +| Authentication | `az account show`; `azd auth login --check-status --output json` | Authenticated | + +Pending live validation (no resources were deployed by this change): + +| Check | Command | Status | |---|---|---| -| Azure CLI | PASS | CLI installed and current account authenticated | -| Subscription | PASS | `95933ae5-0201-4a21-a1fc-8051a7437982` | -| Bicep build/lint | PASS | `subscription.bicep` compiled with no warnings | -| ARM validation | PASS | subscription-scope deployment validation succeeded in `koreacentral` | -| What-if | PASS | Create 12, Modify 0, Delete 0 | -| Azure Policy | PASS | 3 assignments inspected; all target SQL/Data Protection and do not conflict with this deployment | -| Name availability | PASS | `acrsrelab95933ae5`, `stsrelab95933ae5` available | -| Quota | PASS | Container Apps environments 0/50; Storage accounts 0/250 | - -Commands: - -```bash -bash ~/.agents/skills/azure-validate/references/recipes/scripts/validate-deployment.sh \ - --scope sub \ - --location koreacentral \ - --template monitor/sre-agent-event-lab/infra/subscription.bicep \ - --parameters monitor/sre-agent-event-lab/infra/subscription.bicepparam \ - --subscription 95933ae5-0201-4a21-a1fc-8051a7437982 - -monitor/sre-agent-event-lab/app/.venv/bin/python -m pytest \ - monitor/sre-agent-event-lab/app/tests \ - monitor/sre-agent-event-lab/scripts/tests -q - -bash -n monitor/sre-agent-event-lab/scripts/*.sh -az bicep build \ - --file monitor/sre-agent-event-lab/infra/subscription.bicep \ - --stdout >/dev/null - -az policy assignment list \ - --scope /subscriptions/95933ae5-0201-4a21-a1fc-8051a7437982 \ - --disable-scope-strict-match -o json -``` - -Results: - -```text -Official deployment validator: OVERALL PASS -What-if: Create 12, Modify 0, Delete 0 -Tests: 11 passed, 0 warnings -Shell parse: PASS -Bicep build/lint: PASS, 0 warnings -Azure Policy: PASS, no assignment conflicts -Static RBAC review: PASS -``` - ---- - -## 11. Role Assignment Verification - -- Status: Verified -- Identity checked: `id-sre-event-lab-95933ae5` user-assigned managed identity -- `AcrPull` (`7f951dda-4ed3-4680-a7ca-43fe172d538d`): registry scope, required for Container App image pulls -- `Storage Blob Data Reader` (`2a2b9908-6ea1-4ae2-8e65-a410df84e7d1`): `documents` container scope, matches the application's read-only `list_blobs` operation -- Assignment names: deterministic GUIDs derived from scope, identity, and role definition -- Principal type: `ServicePrincipal` -- Local user data-plane role: not required because Blob functional validation runs through the deployed managed identity -- Issues: none +| Environment | `azd env new --location koreacentral` | To re-create for the live run | +| Provision preview | `azd provision --preview --no-prompt` | To re-run | +| Provision phase | `azd provision --no-prompt` leaves the placeholder image serving and `app/.venv` ready | To verify live | +| Deploy phase | `azd deploy --no-prompt` waits for `AcrPull`, builds in ACR, switches the image, `/healthz` returns 200 | To verify live | +| Policy assignments | `az policy assignment list --scope --disable-scope-strict-match` | Unchanged from the previous run; re-check at validation time | +| Static RBAC | reviewed all `Microsoft.Authorization/roleAssignments` in `workload.bicep` | Unchanged: least-privilege AcrPull and container-scoped Blob Data Reader | diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 04d7a38..f66e9d7 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -76,9 +76,20 @@ mkdir -p evidence azd up 2>&1 | tee evidence/deploy.log ``` -Bicep provision → ACR 클라우드 빌드 → Container App 이미지 교체 순서로 진행되며 로컬 Docker는 필요 없습니다. 처음에는 공개 placeholder 이미지가 80 포트로 뜨고, postprovision hook이 ingress를 8000으로 옮긴 뒤 실습 이미지로 교체합니다. +배포는 두 단계입니다. 로컬 Docker는 필요 없습니다. -같은 postprovision hook의 첫 단계가 `scripts/setup-venv.sh`로 `app/.venv`를 준비하는 것입니다(`uv venv` + `uv pip install -r requirements-dev.txt`). 이 단계는 위 ACR 빌드·Container App 이미지 교체보다 먼저 실행되므로, 여기서 실패하면 클라우드 앱 배포(ACR 빌드, 이미지 교체, 헬스체크)는 아직 시작되지 않은 상태입니다 -- `./scripts/setup-venv.sh`만 따로 다시 실행하면 이 배포 단계를 건너뛰게 되므로, 반드시 hook 전체를 다시 실행하세요: `azd hooks run postprovision`. +1. **provision 단계** — Bicep이 ACR, 워크로드 ID(user-assigned managed identity), 그 ID의 AcrPull 역할 할당, 그리고 공개 placeholder 이미지(80 포트)로 뜨는 Container App까지 만듭니다. 이어지는 postprovision hook(`scripts/azd-postprovision-local.sh`)은 `scripts/setup-venv.sh`로 `app/.venv`만 준비하고(`uv venv` + `uv pip install -r requirements-dev.txt`) Azure를 전혀 건드리지 않습니다. 즉 `azd provision`만 실행하면 앱은 계속 placeholder 상태입니다. +2. **deploy 단계** — postdeploy hook(`scripts/azd-deploy-app.sh`)이 워크로드 ID의 **AcrPull** 할당이 ACR 스코프에 정확히 보일 때까지 최대 5분(`SRE_ACR_PULL_TIMEOUT_SECONDS`) 기다린 뒤에야 ACR 클라우드 빌드 → registry/identity 설정 → ingress 8000 이동 → 이미지 교체 → revision·`/healthz` 확인 순서로 진행합니다. 역할이 끝내 보이지 않으면 아무것도 빌드·교체하지 않고 실패합니다. + +`azd up`은 이 두 단계를 순서대로 실행하므로 위 명령 하나로 충분합니다. 단계별로 나눠 실행하거나 다시 실행하려면: + +```bash +azd provision # 인프라 + placeholder + 로컬 venv +azd deploy # AcrPull 대기 → 빌드 → 이미지 교체 → 헬스체크 +azd hooks run postdeploy # 배포 단계만 다시 실행 +``` + +postprovision 단계가 실패하면 로컬 환경만 실패한 것입니다. `./scripts/setup-venv.sh`로 그 단계를 고친 뒤 `azd deploy`로 앱 배포를 마무리하세요. 성공 조건은 provision 성공, 활성 revision `Healthy`, `/healthz` HTTP 200 세 가지입니다. diff --git a/monitor/sre-agent-event-lab/azure.yaml b/monitor/sre-agent-event-lab/azure.yaml index 1442191..513efc3 100644 --- a/monitor/sre-agent-event-lab/azure.yaml +++ b/monitor/sre-agent-event-lab/azure.yaml @@ -11,7 +11,12 @@ hooks: continueOnError: false postprovision: shell: sh - run: ./scripts/azd-postprovision.sh + run: ./scripts/azd-postprovision-local.sh + interactive: true + continueOnError: false + postdeploy: + shell: sh + run: ./scripts/azd-deploy-app.sh interactive: true continueOnError: false predown: diff --git a/monitor/sre-agent-event-lab/infra/lab.bicep b/monitor/sre-agent-event-lab/infra/lab.bicep index b6f112b..59d13cc 100644 --- a/monitor/sre-agent-event-lab/infra/lab.bicep +++ b/monitor/sre-agent-event-lab/infra/lab.bicep @@ -67,6 +67,7 @@ output acrLoginServer string = workload.outputs.acrLoginServer output containerAppName string = workload.outputs.containerAppName output containerAppFqdn string = workload.outputs.containerAppFqdn output containerAppPrincipalId string = workload.outputs.workloadPrincipalId +output workloadIdentityResourceId string = workload.outputs.workloadIdentityResourceId output storageContainerScope string = workload.outputs.storageContainerScope output blobRoleAssignmentName string = workload.outputs.blobRoleAssignmentName output workspaceId string = observability.outputs.workspaceId diff --git a/monitor/sre-agent-event-lab/infra/main.bicep b/monitor/sre-agent-event-lab/infra/main.bicep index 48cc66f..6fd5093 100644 --- a/monitor/sre-agent-event-lab/infra/main.bicep +++ b/monitor/sre-agent-event-lab/infra/main.bicep @@ -71,6 +71,12 @@ output AZURE_RESOURCE_GROUP string = labResourceGroup.name output AZURE_ACR_NAME string = lab.outputs.acrName output AZURE_CONTAINER_APP_NAME string = lab.outputs.containerAppName output AZURE_CONTAINER_APP_FQDN string = lab.outputs.containerAppFqdn +// Read by scripts/azd-deploy-app.sh: the deploy phase waits for AcrPull on +// exactly the registry below for exactly this principal, then points the +// app's registry configuration at the identity that holds it. +output AZURE_CONTAINER_APP_PRINCIPAL_ID string = lab.outputs.containerAppPrincipalId +output AZURE_WORKLOAD_IDENTITY_RESOURCE_ID string = lab.outputs.workloadIdentityResourceId +output AZURE_ACR_LOGIN_SERVER string = lab.outputs.acrLoginServer output AZURE_WORKSPACE_ID string = lab.outputs.workspaceId output AZURE_APP_INSIGHTS_NAME string = lab.outputs.appInsightsName output AZURE_STORAGE_CONTAINER_SCOPE string = lab.outputs.storageContainerScope diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index 528219a..070ea10 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -7,6 +7,18 @@ LAB_ROOT = Path(__file__).parents[2] PLACEHOLDER_IMAGE = "mcr.microsoft.com/azuredocs/containerapps-helloworld:latest" +# Everything that changes the deployed application. `azd provision` must do +# none of it: the Container App's managed identity has no usable `AcrPull` +# grant the instant the ARM deployment returns, so the image build and the +# image switch belong to the deploy phase, behind the role-propagation gate +# in `scripts/azd-deploy-app.sh`. +DEPLOY_ACTIONS = ( + "az acr build", + "az containerapp update", + "az containerapp ingress update", + "az containerapp registry set", +) + def _hook_commands(): """Every `run:` command declared in azure.yaml, hook name unknown.""" @@ -14,10 +26,80 @@ def _hook_commands(): return re.findall(r"^\s*run:\s*(\S+)", config, flags=re.MULTILINE) -def test_azure_yaml_runs_remote_build_after_provision(): +def _hook_command(hook_name): + """The `run:` command declared for one azure.yaml hook, or None.""" + config = (LAB_ROOT / "azure.yaml").read_text() + match = re.search( + rf"^\s*{hook_name}:\s*$(.*?)(?=^\s{{2}}\S|\Z)", + config, + flags=re.MULTILINE | re.DOTALL, + ) + if match is None: + return None + run = re.search(r"^\s*run:\s*(\S+)", match.group(1), flags=re.MULTILINE) + return run.group(1) if run else None + + +def test_provision_hooks_only_configure_and_prepare_the_local_environment(): + """`azd provision` leaves the public placeholder image running. + + The Container Apps + ACR managed-identity flow is two-phase: Bicep + creates the registry, the identity, the AcrPull assignment and a + placeholder-image app, and only the deploy phase -- after the role has + propagated -- builds the lab image and moves the app onto it. A + postprovision hook that builds and updates collapses those phases and + can start an ACR pull the identity is not yet allowed to make. + """ + assert _hook_command("preprovision") == "./scripts/azd-configure.sh" + assert _hook_command("postprovision") == "./scripts/azd-postprovision-local.sh" + + postprovision = LAB_ROOT / "scripts" / "azd-postprovision-local.sh" + assert postprovision.is_file() + text = postprovision.read_text() + for action in DEPLOY_ACTIONS: + assert action not in text, ( + f"the postprovision hook still runs `{action}`; that belongs to " + "the deploy phase, behind the AcrPull gate" + ) + assert "setup-venv.sh" in text, ( + "the postprovision hook's remaining job is the local Python environment" + ) + + +def test_deploy_phase_runs_the_gated_application_deployment(): + """azd 1.29 runs project-level `predeploy`/`postdeploy` hooks even when + the project declares no services -- confirmed against azd 1.29.0 by + running `azd deploy --all --no-prompt` and `azd deploy --no-prompt` + against a service-less project (both hooks ran; a non-zero hook exit + failed the command), and in azd's own source, where + `cli/azd/internal/cmd/up_graph.go` builds the `cmdhook-predeploy` / + `cmdhook-postdeploy` steps unconditionally and keeps them for + "Zero-service projects". So `azd deploy` and `azd up` both reach this + hook without the lab having to declare a service it does not have. + """ + assert _hook_command("postdeploy") == "./scripts/azd-deploy-app.sh" + + deploy_hook = LAB_ROOT / "scripts" / "azd-deploy-app.sh" + assert deploy_hook.is_file() + text = deploy_hook.read_text() + assert "az role assignment list" in text + assert "az acr build" in text + assert "az containerapp update" in text + + +def test_project_declares_no_service_azd_would_try_to_build_locally(): + """The lab image is built in ACR, never on the operator's machine. A + `services:` entry with a Docker host would make `azd deploy`/`azd up` + require a local Docker daemon before any hook could run. + """ + config = (LAB_ROOT / "azure.yaml").read_text() + + assert not re.search(r"^services:", config, flags=re.MULTILINE) + assert "docker:" not in config + + +def test_azure_yaml_still_cleans_up_external_state_on_down(): config = (LAB_ROOT / "azure.yaml").read_text() - assert "postprovision" in config - assert "./scripts/azd-postprovision.sh" in config assert "predown" in config assert "./scripts/cleanup-external.sh" in config @@ -47,6 +129,31 @@ def test_azd_outputs_have_stable_names(): assert f"output {name} " in template +def test_main_bicep_publishes_what_the_deploy_gate_reads(): + """The deploy hook has to check `AcrPull` for one principal at one + registry scope, and then point the app's registry configuration at the + same identity. azd stores outputs under the name the template declares + (`OutputParametersFromArmOutputs` keeps the template's canonical + casing), so these are the names the hook can rely on being exported + into its environment. + """ + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + + for name in ( + "AZURE_CONTAINER_APP_PRINCIPAL_ID", + "AZURE_WORKLOAD_IDENTITY_RESOURCE_ID", + "AZURE_ACR_LOGIN_SERVER", + ): + assert f"output {name} " in template, ( + f"scripts/azd-deploy-app.sh reads {name} from the azd environment" + ) + + workload = (LAB_ROOT / "infra" / "workload.bicep").read_text() + assert "output workloadIdentityResourceId string" in workload + lab = (LAB_ROOT / "infra" / "lab.bicep").read_text() + assert "output workloadIdentityResourceId string" in lab + + def test_every_azure_yaml_hook_references_an_existing_executable_script(): """azd aborts the whole command when a hook script cannot be executed. @@ -168,7 +275,8 @@ def test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id(): LAB_ROOT / "infra" / "workload.bicep", LAB_ROOT / "infra" / "main.parameters.json", LAB_ROOT / "scripts" / "azd-configure.sh", - LAB_ROOT / "scripts" / "azd-postprovision.sh", + LAB_ROOT / "scripts" / "azd-postprovision-local.sh", + LAB_ROOT / "scripts" / "azd-deploy-app.sh", LAB_ROOT / "scripts" / "cleanup-external.sh", LAB_ROOT / "scripts" / "deploy.sh", ] diff --git a/monitor/sre-agent-event-lab/infra/workload.bicep b/monitor/sre-agent-event-lab/infra/workload.bicep index 0a1fc13..5d699a2 100644 --- a/monitor/sre-agent-event-lab/infra/workload.bicep +++ b/monitor/sre-agent-event-lab/infra/workload.bicep @@ -371,6 +371,9 @@ output acrLoginServer string = registry.properties.loginServer output containerAppName string = deployContainerApp ? containerApp.name : appName output containerAppFqdn string = deployContainerApp ? containerApp!.properties.configuration.ingress.fqdn : '' output workloadPrincipalId string = workloadIdentity.properties.principalId +// The `--identity` the Container App pulls with, and the identity whose +// AcrPull assignment the deploy phase waits for. +output workloadIdentityResourceId string = workloadIdentity.id output storageContainerScope string = documentsContainer.id output blobRoleAssignmentName string = blobReaderAssignment.name output telemetryServiceName string = telemetryServiceName diff --git a/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh b/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh new file mode 100755 index 0000000..231a351 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh @@ -0,0 +1,260 @@ +#!/usr/bin/env bash +set -euo pipefail + +# `postdeploy` hook: the cloud half of the lab's two-phase deployment. +# +# Why a second phase exists at all +# -------------------------------- +# The lab runs on Container Apps pulling from ACR with a user-assigned +# managed identity. One ARM deployment creates the registry, the identity, +# the `AcrPull` role assignment and the Container App -- but it cannot +# deploy the lab image: that image does not exist until something builds +# it, and a role assignment created moments ago is not necessarily usable +# by the pull that would immediately follow. So provisioning leaves a +# public placeholder image running (ingress on 80, no probes) and this +# hook, which runs in the deploy phase, does the rest: +# +# 1. wait until the workload identity's `AcrPull` assignment is visible +# at exactly the lab registry's scope (up to 5 minutes), +# 2. build the image *in ACR* (never locally -- the lab requires no +# Docker daemon), +# 3. point the app's registry configuration at the same identity, move +# ingress to the app's port, and roll the app onto the new image, +# 4. verify a new healthy revision and a healthy `/healthz`, +# 5. record the built image in the azd environment, so a later +# `azd provision` keeps the lab image and its matching probes instead +# of reverting to the placeholder. +# +# Step 1 is a read-consistency check on ARM, not a proof that the registry +# data plane will accept the token; it is the strongest signal available +# without attempting a pull, and it removes the common failure where the +# very first pull races the role assignment. +# +# azd 1.29 runs project-level `predeploy`/`postdeploy` hooks even for a +# project that declares no services (verified against azd 1.29.0, and in +# azd's own `cli/azd/internal/cmd/up_graph.go`, which keeps those steps +# for "Zero-service projects"), so `azd deploy` and `azd up` both reach +# this hook without the lab declaring a service that would drag in a local +# Docker build. + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +readonly SCRIPT_DIR +LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" +readonly LAB_ROOT +readonly APP_DIR="${LAB_ROOT}/app" +# The lab image serves HTTP on 8000; the placeholder image the first +# provision leaves running serves 80, so ingress has to move with the image. +readonly APP_TARGET_PORT=8000 +readonly IMAGE_REPOSITORY="sre-event-lab" +# AcrPull, by role definition ID: a display name can be reused by a custom +# role, this GUID cannot. +readonly ACR_PULL_ROLE_DEFINITION_ID="7f951dda-4ed3-4680-a7ca-43fe172d538d" + +# Every wait below is a budget, not a guess, and each one is overridable so +# a slow tenant does not need a code change (and so tests can run fast). +readonly ACR_PULL_TIMEOUT_SECONDS="${SRE_ACR_PULL_TIMEOUT_SECONDS:-300}" +readonly ACR_PULL_POLL_INTERVAL_SECONDS="${SRE_ACR_PULL_POLL_INTERVAL_SECONDS:-10}" +readonly REVISION_READY_TIMEOUT_SECONDS="${SRE_REVISION_READY_TIMEOUT_SECONDS:-600}" +readonly HEALTH_TIMEOUT_SECONDS="${SRE_HEALTH_TIMEOUT_SECONDS:-600}" +readonly DEPLOY_POLL_INTERVAL_SECONDS="${SRE_DEPLOY_POLL_INTERVAL_SECONDS:-10}" + +for command_name in az azd curl; do + command -v "${command_name}" >/dev/null 2>&1 || { + echo "Required command not found: ${command_name}" >&2 + exit 1 + } +done + +# Every value below is a deployment output azd refreshes into this hook's +# environment. A missing one means provisioning has not run (or the +# environment is stale), which is worth saying before spending any Azure +# call on it. +readonly MISSING_OUTPUT_HINT="is missing; run 'azd provision' (or 'azd up') first" +: "${AZURE_SUBSCRIPTION_ID:?${MISSING_OUTPUT_HINT}}" +: "${AZURE_RESOURCE_GROUP:?${MISSING_OUTPUT_HINT}}" +: "${AZURE_ACR_NAME:?${MISSING_OUTPUT_HINT}}" +: "${AZURE_CONTAINER_APP_NAME:?${MISSING_OUTPUT_HINT}}" +: "${AZURE_CONTAINER_APP_FQDN:?${MISSING_OUTPUT_HINT}}" +: "${AZURE_CONTAINER_APP_PRINCIPAL_ID:?${MISSING_OUTPUT_HINT}}" +: "${AZURE_WORKLOAD_IDENTITY_RESOURCE_ID:?${MISSING_OUTPUT_HINT}}" + +# `az account show` fails with a raw Azure CLI error when no one is signed +# in; guard it so the hook fails fast with one clear, actionable message +# instead of that raw stderr or an unexplained `set -e` abort. +if ! ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv 2>/dev/null)"; then + echo "Azure CLI is not signed in. Run 'az login', then re-run this command." >&2 + exit 1 +fi +readonly ACTIVE_SUBSCRIPTION_ID +if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${AZURE_SUBSCRIPTION_ID}" ]]; then + echo "Azure CLI is signed in to ${ACTIVE_SUBSCRIPTION_ID}." >&2 + echo "Every lab operation is pinned to ${AZURE_SUBSCRIPTION_ID} instead." >&2 +fi + +ACR_RESOURCE_ID="$(az acr show \ + --name "${AZURE_ACR_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query id \ + -o tsv)" +readonly ACR_RESOURCE_ID + +# `AZURE_ACR_LOGIN_SERVER` is a deployment output, but this hook is also +# run by hand (`azd hooks run postdeploy`) against environments written by +# older provisions, so fall back to the registry itself. +ACR_LOGIN_SERVER="${AZURE_ACR_LOGIN_SERVER:-}" +if [[ -z "${ACR_LOGIN_SERVER}" ]]; then + ACR_LOGIN_SERVER="$(az acr show \ + --name "${AZURE_ACR_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query loginServer \ + -o tsv)" +fi +readonly ACR_LOGIN_SERVER + +lowercase() { + printf '%s' "${1}" | tr '[:upper:]' '[:lower:]' +} + +# True once the workload identity holds AcrPull at *exactly* the lab +# registry. `az role assignment list` without `--include-inherited` matches +# the scope exactly (azure-cli lowercases both sides), so an AcrPull +# granted higher up -- at the resource group or the subscription -- is +# deliberately not accepted here: it is not the assignment this lab +# creates, and treating it as one would let the gate pass while the lab's +# own assignment is still propagating. +acr_pull_is_visible() { + local granted expected line + granted="$(az role assignment list \ + --assignee-object-id "${AZURE_CONTAINER_APP_PRINCIPAL_ID}" \ + --assignee-principal-type ServicePrincipal \ + --scope "${ACR_RESOURCE_ID}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query "[?ends_with(roleDefinitionId, '${ACR_PULL_ROLE_DEFINITION_ID}')].scope" \ + -o tsv 2>/dev/null || true)" + expected="$(lowercase "${ACR_RESOURCE_ID}")" + while IFS= read -r line; do + # ARM echoes the scope as it was written (`resourceGroups` or + # `resourcegroups`); resource IDs are case-insensitive, and a casing + # difference must not stall the deployment for the full budget. + if [[ -n "${line}" && "$(lowercase "${line}")" == "${expected}" ]]; then + return 0 + fi + done <<<"${granted}" + return 1 +} + +echo "Waiting for AcrPull on ${ACR_RESOURCE_ID} (up to ${ACR_PULL_TIMEOUT_SECONDS}s)..." +acr_pull_started="${SECONDS}" +until acr_pull_is_visible; do + if (( SECONDS - acr_pull_started >= ACR_PULL_TIMEOUT_SECONDS )); then + echo "The workload identity ${AZURE_CONTAINER_APP_PRINCIPAL_ID} still has no" >&2 + echo "AcrPull assignment at ${ACR_RESOURCE_ID} after ${ACR_PULL_TIMEOUT_SECONDS}s." >&2 + echo "Nothing was built or deployed. Re-run 'azd provision' to restore the" >&2 + echo "assignment, then 'azd deploy' -- or raise SRE_ACR_PULL_TIMEOUT_SECONDS." >&2 + exit 1 + fi + sleep "${ACR_PULL_POLL_INTERVAL_SECONDS}" +done +echo "AcrPull is in place; building the lab image." + +IMAGE_TAG="run-$(date -u +%Y%m%dT%H%M%SZ)" +readonly IMAGE_TAG +readonly CONTAINER_IMAGE="${ACR_LOGIN_SERVER}/${IMAGE_REPOSITORY}:${IMAGE_TAG}" + +# Built by the registry from the sources, so the lab needs no local Docker. +az acr build \ + --registry "${AZURE_ACR_NAME}" \ + --image "${IMAGE_REPOSITORY}:${IMAGE_TAG}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + "${APP_DIR}" + +# The app must pull with the identity the AcrPull assignment was granted +# to; the placeholder image needed no registry credentials at all. +az containerapp registry set \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --server "${ACR_LOGIN_SERVER}" \ + --identity "${AZURE_WORKLOAD_IDENTITY_RESOURCE_ID}" \ + --output none + +# Ingress is app-level configuration and does not create a revision, so move +# it to the lab port before the new image starts serving. +az containerapp ingress update \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --target-port "${APP_TARGET_PORT}" \ + --output none + +PREVIOUS_REVISION="$(az containerapp show \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query properties.latestRevisionName \ + -o tsv)" +readonly PREVIOUS_REVISION + +az containerapp update \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --image "${CONTAINER_IMAGE}" \ + --output none + +wait_for_new_revision_ready() { + local timeout_seconds="${1:-600}" + local started="${SECONDS}" + + while (( SECONDS - started < timeout_seconds )); do + local latest_revision + latest_revision="$(az containerapp show \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query properties.latestRevisionName \ + -o tsv)" + if [[ -n "${latest_revision}" && "${latest_revision}" != "${PREVIOUS_REVISION}" ]]; then + local health active + health="$(az containerapp revision list \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query "[?name=='${latest_revision}'].properties.healthState | [0]" \ + -o tsv 2>/dev/null || true)" + active="$(az containerapp revision list \ + --resource-group "${AZURE_RESOURCE_GROUP}" \ + --name "${AZURE_CONTAINER_APP_NAME}" \ + --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --query "[?name=='${latest_revision}'].properties.active | [0]" \ + -o tsv 2>/dev/null || true)" + if [[ "${health}" == "Healthy" && "${active}" == "true" ]]; then + return 0 + fi + fi + sleep "${DEPLOY_POLL_INTERVAL_SECONDS}" + done + + echo "A new healthy revision did not become active within ${timeout_seconds}s." >&2 + return 1 +} + +wait_for_new_revision_ready "${REVISION_READY_TIMEOUT_SECONDS}" + +health_started="${SECONDS}" +until curl --fail --silent --show-error "https://${AZURE_CONTAINER_APP_FQDN}/healthz" >/dev/null; do + if (( SECONDS - health_started >= HEALTH_TIMEOUT_SECONDS )); then + echo "Health endpoint did not return HTTP 200 within ${HEALTH_TIMEOUT_SECONDS}s." >&2 + exit 1 + fi + sleep "${DEPLOY_POLL_INTERVAL_SECONDS}" +done + +# Only a verified-healthy image is recorded: persisting one that never came +# up would make the next `azd provision` deploy it again as if it were +# known good. `--cwd` because azd runs hooks from wherever the operator +# invoked it, which need not be the lab. +azd env set SRE_IMAGE_TAG "${IMAGE_TAG}" --cwd "${LAB_ROOT}" +azd env set SRE_CONTAINER_IMAGE "${CONTAINER_IMAGE}" --cwd "${LAB_ROOT}" + +echo "Deployed ${CONTAINER_IMAGE}; https://${AZURE_CONTAINER_APP_FQDN}/healthz is healthy." diff --git a/monitor/sre-agent-event-lab/scripts/azd-postprovision-local.sh b/monitor/sre-agent-event-lab/scripts/azd-postprovision-local.sh new file mode 100755 index 0000000..894ac70 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/azd-postprovision-local.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +set -euo pipefail + +# `postprovision` hook: the local half of the lab's two-phase deployment. +# +# Provisioning creates the registry, the workload identity, its `AcrPull` +# assignment and a Container App still running the *public placeholder* +# image -- the lab image does not exist yet, and the identity's `AcrPull` +# grant is not necessarily usable the instant the ARM deployment returns. +# Building and switching the image therefore belongs to the deploy phase +# (`scripts/azd-deploy-app.sh`, run by `azd deploy` and by `azd up`'s +# deploy phase), which waits for that grant before it touches anything. +# +# So this hook does exactly one thing: make the local Python environment +# ready, on the one step of the documented `azd up` flow that always runs +# right after provisioning succeeds. It makes no Azure calls at all, which +# is what keeps `azd provision` on the placeholder image. + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +readonly SCRIPT_DIR + +# `setup-venv.sh` is its own idempotent script (uv-only, no pip fallback) +# so it can be re-run by hand -- see its own actionable error output -- +# without repeating anything else. +"${SCRIPT_DIR}/setup-venv.sh" + +cat <<'NEXT' + +Provisioning finished and the local Python environment is ready. +The Container App is still running the public placeholder image: the lab +image is built in ACR and switched in during the deploy phase, once the +workload identity's AcrPull grant is visible at the registry. + +Next: `azd deploy` (or `azd up`, which continues into the same phase). +NEXT diff --git a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh b/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh deleted file mode 100755 index 86e1582..0000000 --- a/monitor/sre-agent-event-lab/scripts/azd-postprovision.sh +++ /dev/null @@ -1,143 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" -readonly SCRIPT_DIR -LAB_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd -P)" -readonly LAB_ROOT -readonly APP_DIR="${LAB_ROOT}/app" -# The lab image serves HTTP on 8000; the placeholder image used by the first -# provision serves 80, so ingress has to move with the image. -readonly APP_TARGET_PORT=8000 - -for command_name in az azd curl; do - command -v "${command_name}" >/dev/null 2>&1 || { - echo "Required command not found: ${command_name}" >&2 - exit 1 - } -done - -# `app/.venv` is not read by this hook itself, but this is the one place in -# the documented azd-first flow (`azd up`) that always runs once -# provisioning succeeds, so it is the natural place to make the local -# Python environment `capture-scenario.sh` and the notification step need -# ready before an operator ever reaches them. `setup-venv.sh` is its own -# idempotent script (uv-only, no pip fallback) so it can also be re-run by -# hand -- see its own actionable error output -- without repeating anything -# below. -"${SCRIPT_DIR}/setup-venv.sh" - -# Every value below comes from the current azd environment, which azd refreshes -# from the deployment outputs before running this hook. -: "${AZURE_SUBSCRIPTION_ID:?AZURE_SUBSCRIPTION_ID must be set by azd before running this hook}" -: "${AZURE_RESOURCE_GROUP:?AZURE_RESOURCE_GROUP must be set by azd before running this hook}" -: "${AZURE_ACR_NAME:?AZURE_ACR_NAME must be set by azd before running this hook}" -: "${AZURE_CONTAINER_APP_NAME:?AZURE_CONTAINER_APP_NAME must be set by azd before running this hook}" -: "${AZURE_CONTAINER_APP_FQDN:?AZURE_CONTAINER_APP_FQDN must be set by azd before running this hook}" - -# `az account show` fails with a raw Azure CLI error when no one is signed -# in; guard it so the hook fails fast with one clear, actionable message -# instead of that raw stderr or an unexplained `set -e` abort. -if ! ACTIVE_SUBSCRIPTION_ID="$(az account show --query id -o tsv 2>/dev/null)"; then - echo "Azure CLI is not signed in. Run 'az login', then re-run this command." >&2 - exit 1 -fi -readonly ACTIVE_SUBSCRIPTION_ID -if [[ "${ACTIVE_SUBSCRIPTION_ID}" != "${AZURE_SUBSCRIPTION_ID}" ]]; then - echo "Azure CLI is signed in to ${ACTIVE_SUBSCRIPTION_ID}." >&2 - echo "Every lab operation is pinned to ${AZURE_SUBSCRIPTION_ID} instead." >&2 -fi - -IMAGE_TAG="run-$(date -u +%Y%m%dT%H%M%SZ)" -readonly IMAGE_TAG - -az acr build \ - --registry "${AZURE_ACR_NAME}" \ - --image "sre-event-lab:${IMAGE_TAG}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - "${APP_DIR}" - -ACR_LOGIN_SERVER="$(az acr show \ - --name "${AZURE_ACR_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --query loginServer \ - -o tsv)" -readonly ACR_LOGIN_SERVER -readonly CONTAINER_IMAGE="${ACR_LOGIN_SERVER}/sre-event-lab:${IMAGE_TAG}" - -PREVIOUS_REVISION="$(az containerapp show \ - --resource-group "${AZURE_RESOURCE_GROUP}" \ - --name "${AZURE_CONTAINER_APP_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --query properties.latestRevisionName \ - -o tsv)" -readonly PREVIOUS_REVISION - -# Ingress is app-level configuration and does not create a revision, so move it -# to the lab port before the new image starts serving. -az containerapp ingress update \ - --resource-group "${AZURE_RESOURCE_GROUP}" \ - --name "${AZURE_CONTAINER_APP_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --target-port "${APP_TARGET_PORT}" \ - --output none - -az containerapp update \ - --resource-group "${AZURE_RESOURCE_GROUP}" \ - --name "${AZURE_CONTAINER_APP_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --image "${CONTAINER_IMAGE}" \ - --output none - -wait_for_new_revision_ready() { - local timeout_seconds="${1:-600}" - local started="${SECONDS}" - - while (( SECONDS - started < timeout_seconds )); do - local latest_revision - latest_revision="$(az containerapp show \ - --resource-group "${AZURE_RESOURCE_GROUP}" \ - --name "${AZURE_CONTAINER_APP_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --query properties.latestRevisionName \ - -o tsv)" - if [[ -n "${latest_revision}" && "${latest_revision}" != "${PREVIOUS_REVISION}" ]]; then - local health active - health="$(az containerapp revision list \ - --resource-group "${AZURE_RESOURCE_GROUP}" \ - --name "${AZURE_CONTAINER_APP_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --query "[?name=='${latest_revision}'].properties.healthState | [0]" \ - -o tsv 2>/dev/null || true)" - active="$(az containerapp revision list \ - --resource-group "${AZURE_RESOURCE_GROUP}" \ - --name "${AZURE_CONTAINER_APP_NAME}" \ - --subscription "${AZURE_SUBSCRIPTION_ID}" \ - --query "[?name=='${latest_revision}'].properties.active | [0]" \ - -o tsv 2>/dev/null || true)" - if [[ "${health}" == "Healthy" && "${active}" == "true" ]]; then - return 0 - fi - fi - sleep 10 - done - - echo "A new healthy revision did not become active within ${timeout_seconds}s." >&2 - return 1 -} - -wait_for_new_revision_ready 600 - -started="${SECONDS}" -until curl --fail --silent --show-error "https://${AZURE_CONTAINER_APP_FQDN}/healthz" >/dev/null; do - if (( SECONDS - started >= 600 )); then - echo "Health endpoint did not return HTTP 200 within 600s." >&2 - exit 1 - fi - sleep 10 -done - -azd env set SRE_IMAGE_TAG "${IMAGE_TAG}" -# Persisting the built image keeps a later `azd provision` on the lab image and -# its matching /healthz probes instead of reverting to the placeholder. -azd env set SRE_CONTAINER_IMAGE "${CONTAINER_IMAGE}" diff --git a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh index 19f7445..2db3dda 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh @@ -5,7 +5,7 @@ # the azd-owned resource group, so azd can # delete everything else itself. # postdown --reset-image-env Clear the azd environment values -# `azd-postprovision.sh` recorded, once the +# `azd-deploy-app.sh` recorded, once the # resources they point at are really gone. # # The only external resources are the subscription-scoped Monitoring @@ -61,7 +61,7 @@ Usage: cleanup-external.sh [--reset-image-env] [--yes] (default) Remove the recorded subscription-scoped Monitoring Contributor assignments that live outside the azd resource group. Run by `azd down` as its predown hook. - --reset-image-env Clear the azd environment values azd-postprovision.sh + --reset-image-env Clear the azd environment values azd-deploy-app.sh recorded (SRE_CONTAINER_IMAGE, SRE_IMAGE_TAG) instead. Run by `azd down` as its postdown hook. --yes Execute. Without it, both modes only print their plan. diff --git a/monitor/sre-agent-event-lab/scripts/cleanup.sh b/monitor/sre-agent-event-lab/scripts/cleanup.sh index 187e612..60e3ee5 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup.sh @@ -5,7 +5,7 @@ # down: azd deletes the resource group it created, and its predown/postdown # hooks run `cleanup-external.sh` for the two things azd cannot see (the # recorded subscription-scoped role assignments, and the image values the -# postprovision hook stored). This script therefore only forwards to +# postdeploy hook stored). This script therefore only forwards to # `cleanup-external.sh` -- it never deletes a resource group of its own, # because a broad deletion here would also take resources azd did not # create and knows nothing about. diff --git a/monitor/sre-agent-event-lab/scripts/deploy.sh b/monitor/sre-agent-event-lab/scripts/deploy.sh index dd42e3a..01f6a39 100755 --- a/monitor/sre-agent-event-lab/scripts/deploy.sh +++ b/monitor/sre-agent-event-lab/scripts/deploy.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash # Compatibility wrapper. The lab is an azd project now: azure.yaml owns the -# Bicep entry point, the preprovision/postprovision hooks register providers, -# build the image in ACR, and move the Container App onto it. This script only -# forwards to `azd up` so the previously documented command keeps working. +# Bicep entry point, the preprovision/postprovision hooks register providers +# and prepare the local environment, and the postdeploy hook builds the image +# in ACR and moves the Container App onto it once AcrPull has propagated. This +# script only forwards to `azd up` -- which runs both phases -- so the +# previously documented command keeps working. set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" diff --git a/monitor/sre-agent-event-lab/scripts/doctor.sh b/monitor/sre-agent-event-lab/scripts/doctor.sh index 77e9725..953d8fa 100755 --- a/monitor/sre-agent-event-lab/scripts/doctor.sh +++ b/monitor/sre-agent-event-lab/scripts/doctor.sh @@ -46,7 +46,8 @@ fi # Python environment (app/.venv + Pillow) ------------------------------------- # The documented azd-first flow prepares `app/.venv` from the `postprovision` -# hook (`scripts/setup-venv.sh`), before any scenario is captured. This check +# hook (`scripts/azd-postprovision-local.sh` -> `scripts/setup-venv.sh`, its +# whole job), before any scenario is captured. This check # reports whether that step actually finished -- Pillow importable, not just # a venv directory present -- so a partially-run or pre-uv-created venv is # caught here rather than surfacing later as `capture-scenario.sh`'s "Missing diff --git a/monitor/sre-agent-event-lab/scripts/setup-venv.sh b/monitor/sre-agent-event-lab/scripts/setup-venv.sh index 410550f..fed9df3 100755 --- a/monitor/sre-agent-event-lab/scripts/setup-venv.sh +++ b/monitor/sre-agent-event-lab/scripts/setup-venv.sh @@ -16,19 +16,16 @@ set -euo pipefail # missing, this script fails with an actionable install pointer instead of # silently reaching the public index. # -# Idempotency matters because this script is invoked from the `postprovision` -# hook *before* that hook's ACR build and Container App update -- see -# `azd-postprovision.sh` -- so a failure here happens before the cloud app -# deployment for this run has even started, not after it. Re-running only -# this script would leave that deployment never attempted, so every failure -# message below points at re-running the *whole* hook (`azd hooks run -# postprovision`), never at running this script directly -- and `uv venv -# --allow-existing` plus `uv pip install` make that safe to re-run even when -# a previous attempt got partway through. (Running this script directly is -# still the right move for a purely local `app/.venv` problem noticed well -# after a deployment already succeeded -- see `doctor.sh` and -# `capture-scenario.sh`'s own remediation text -- just never as the response -# to a failure reported by *this* script.) +# Idempotency matters because this script is the whole job of the +# `postprovision` hook (`azd-postprovision-local.sh`): the Container App +# image build and the image switch live in the deploy phase +# (`azd-deploy-app.sh`), behind its AcrPull gate. So a failure here is a +# purely local failure -- re-running this script directly is a complete +# fix, and `uv venv --allow-existing` plus `uv pip install` make that safe +# even when a previous attempt got partway through. What a failure here +# does mean is that `azd up` stopped before its deploy phase, so the +# application deployment still has to be finished with `azd deploy` -- +# which every failure message below says. SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" readonly SCRIPT_DIR @@ -47,13 +44,14 @@ readonly REQUIREMENTS_FILE="${APP_DIR}/requirements-dev.txt" # interpreter left behind by a pre-uv `python3 -m venv` (uv recreates the # venv in place when the existing interpreter doesn't satisfy the request). readonly VENV_PYTHON_VERSION=">=3.10" -# Cloud resources from `azd provision` are already deployed by the time this -# hook (and so this script) runs, but this hook's own ACR build and Container -# App update -- the cloud *app* deployment -- run after this script, not -# before it, so a failure here must not be answered by re-running only this -# script: that would leave the app deployment never attempted. `cd` pins the -# rerun to this lab's project directory regardless of the operator's shell. -readonly RERUN_HINT="Cloud infrastructure from 'azd provision' is already deployed, but this hook's Container App image build and deployment run *after* this step and have not happened yet -- re-running only this script would silently skip them. Re-run the whole hook instead: cd ${LAB_ROOT} && azd hooks run postprovision" +# What a failure here does and does not mean: `azd provision` has created +# the infrastructure and left the Container App on its public placeholder +# image, and the deploy phase -- the ACR build and the image switch -- has +# not run yet, because a failing `postprovision` hook stops `azd up` +# before it. Re-running this script fixes the local half; `azd deploy` +# still has to finish the cloud half. `cd` pins both commands to this +# lab's project directory regardless of the operator's shell. +readonly RERUN_HINT="Only the local Python environment failed: infrastructure from 'azd provision' is up, and the lab image is built and switched in separately by the deploy phase. Fix this step with: cd ${LAB_ROOT} && ./scripts/setup-venv.sh -- then finish the application deployment with: cd ${LAB_ROOT} && azd deploy" if ! command -v uv >/dev/null 2>&1; then echo "uv is required to set up ${VENV_DIR} but was not found on PATH." >&2 diff --git a/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py b/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py new file mode 100644 index 0000000..32f3955 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py @@ -0,0 +1,415 @@ +"""Harness that runs `azd-deploy-app.sh` as a program. + +The script is the `postdeploy` hook of `azd deploy` (and of `azd up`'s +deploy phase), so it is exercised the way azd runs it: as an executable, +from a working directory that is not the lab, with fake `az`/`azd`/`curl` +binaries on PATH and only the environment azd itself would export. + +The whole point of this hook is an *ordering* guarantee that reading the +text cannot prove: the Container App's user-assigned identity must hold +`AcrPull` on exactly the lab registry before the first deploy action +(`az acr build`) runs. So the fake `az` here is not a "log and exit 0" +stub -- it models the tenant: + +* role assignments are records (`principalId`, `roleDefinitionId`, + `scope`), and `az role assignment list` applies the real CLI's filters + to them: the assignee filter, the *exact* `--scope` match (azure-cli's + `_search_role_assignments` compares `scope.lower()` and only widens to + parents with `--include-inherited`, which this lab never passes), and + the `--query` JMESPath `ends_with(roleDefinitionId, '')` + projection. An assignment at the resource group, or a different role at + the registry, is therefore invisible to the hook exactly as it would be + against a real subscription. +* `available_after_attempts` models RBAC propagation: the assignment + exists but the first N `role assignment list` calls do not return it + yet, which is the delay the whole poll exists for. +* `az containerapp update --image` rolls the app onto a new revision + name, and `az containerapp revision list` reports that revision's + health, so the "wait for a new healthy revision" loop is exercised + rather than mocked away. + +The fake `az` is Python, not Bash, because reproducing those filters +faithfully in a Bash stub would be less readable than the behaviour it is +supposed to prove. +""" +import json +import os +import subprocess +from pathlib import Path + +from azd_fake import write_azd_stub, write_executable + +SCRIPTS_DIR = Path(__file__).parents[1] +LAB_ROOT = Path(__file__).parents[2] +DEPLOY_APP = SCRIPTS_DIR / "azd-deploy-app.sh" + +SUBSCRIPTION_ID = "11111111-2222-3333-4444-555555555555" +OTHER_SUBSCRIPTION_ID = "99999999-9999-9999-9999-999999999999" +RESOURCE_GROUP = "rg-sre-lab-hooktest" +ACR_NAME = "acrsrelabtest01" +ACR_LOGIN_SERVER = f"{ACR_NAME}.azurecr.io" +ACR_RESOURCE_ID = ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + f"/providers/Microsoft.ContainerRegistry/registries/{ACR_NAME}" +) +RESOURCE_GROUP_SCOPE = ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" +) +CONTAINER_APP_NAME = "ca-sre-event-lab-vnet" +CONTAINER_APP_FQDN = "ca-sre-event-lab-vnet.koreacentral.azurecontainerapps.io" +WORKLOAD_PRINCIPAL_ID = "aaaaaaaa-0000-4000-8000-aaaaaaaaaaaa" +WORKLOAD_IDENTITY_RESOURCE_ID = ( + f"/subscriptions/{SUBSCRIPTION_ID}/resourceGroups/{RESOURCE_GROUP}" + "/providers/Microsoft.ManagedIdentity/userAssignedIdentities/id-sre-event-lab-test" +) + +ACR_PULL_ROLE_DEFINITION_ID = "7f951dda-4ed3-4680-a7ca-43fe172d538d" +ACR_PUSH_ROLE_DEFINITION_ID = "8311e382-0749-4cb8-b61a-304f252e45ec" + +INITIAL_REVISION = "ca-sre-event-lab-vnet--placeholder1" +NEW_REVISION = "ca-sre-event-lab-vnet--labimage1" + + +def role_definition_id(guid, subscription_id=SUBSCRIPTION_ID): + return ( + f"/subscriptions/{subscription_id}/providers/Microsoft.Authorization" + f"/roleDefinitions/{guid}" + ) + + +def assignment( + principal_id=WORKLOAD_PRINCIPAL_ID, + role_guid=ACR_PULL_ROLE_DEFINITION_ID, + scope=ACR_RESOURCE_ID, + available_after_attempts=0, +): + """One role assignment as the tenant holds it. + + `available_after_attempts` is how many `az role assignment list` calls + must happen before this assignment becomes visible -- RBAC propagation + delay, the reason the hook polls at all. + """ + return { + "principalId": principal_id, + "roleDefinitionId": role_definition_id(role_guid), + "scope": scope, + "available_after_attempts": available_after_attempts, + } + + +ACR_PULL_GRANTED = [assignment()] + + +_AZ_STUB = r'''#!/usr/bin/env python3 +"""A fake `az` that models the lab's subscription (see deploy_app_harness).""" +import json +import re +import sys +from pathlib import Path + +STATE = Path(r"{state_file}") +LOG = Path(r"{log_path}") + + +def load(): + return json.loads(STATE.read_text()) + + +def save(state): + STATE.write_text(json.dumps(state)) + + +def flag(argv, name, default=None): + if name in argv: + index = argv.index(name) + if index + 1 < len(argv): + return argv[index + 1] + return default + + +def fail(message): + sys.stderr.write(message + "\n") + raise SystemExit(1) + + +def main(): + argv = sys.argv[1:] + with LOG.open("a") as log: + log.write(" ".join(argv) + "\n") + state = load() + joined = " ".join(argv) + + if argv[:2] == ["account", "show"]: + if state.get("signed_out"): + fail("ERROR: Please run 'az login' to setup account.") + print(state["active_subscription"]) + return + + if argv[:2] == ["acr", "show"]: + query = flag(argv, "--query") + if query == "id": + print(state["acr_resource_id"]) + return + if query == "loginServer": + print(state["acr_login_server"]) + return + fail(f"ERROR: fake az: unsupported acr show query: {{query}}") + + if argv[:2] == ["acr", "build"]: + if state.get("acr_build_fails"): + fail("ERROR: failed to build image") + state["acr_build_count"] = state.get("acr_build_count", 0) + 1 + save(state) + print("Run ID: ca1 was successful after 30s") + return + + if argv[:3] == ["role", "assignment", "list"]: + state["role_list_attempts"] = state.get("role_list_attempts", 0) + 1 + attempt = state["role_list_attempts"] + save(state) + assignee = flag(argv, "--assignee-object-id") + scope = flag(argv, "--scope") + query = flag(argv, "--query", "") + wanted = re.search(r"ends_with\(\s*roleDefinitionId\s*,\s*'([^']+)'", query) + if not wanted: + fail(f"ERROR: fake az: unsupported role assignment list query: {{query}}") + wanted_role = wanted.group(1) + if "--include-inherited" in argv: + fail("ERROR: fake az: the lab must never widen the scope with --include-inherited") + for record in state["assignments"]: + if attempt <= record.get("available_after_attempts", 0): + continue + if assignee is not None and record["principalId"] != assignee: + continue + # azure-cli's own exact-scope filter (case-insensitive) when + # --include-inherited is not passed. + if scope is not None and record["scope"].lower() != scope.lower(): + continue + if not record["roleDefinitionId"].endswith(wanted_role): + continue + print(record["scope"]) + return + + if argv[:2] == ["containerapp", "show"]: + query = flag(argv, "--query") + if query == "properties.latestRevisionName": + print(state["latest_revision"]) + return + fail(f"ERROR: fake az: unsupported containerapp show query: {{query}}") + + if argv[:3] == ["containerapp", "revision", "list"]: + query = flag(argv, "--query", "") + revision = re.search(r"name=='([^']+)'", query) + revision_name = revision.group(1) if revision else "" + state["revision_polls"] = state.get("revision_polls", 0) + 1 + polls = state["revision_polls"] + save(state) + healthy_after = state.get("revision_healthy_after_polls", 0) + if revision_name != state["latest_revision"] or polls <= healthy_after: + print("Unhealthy" if "healthState" in query else "false") + return + print("Healthy" if "healthState" in query else "true") + return + + if argv[:3] == ["containerapp", "ingress", "update"]: + state["target_port"] = flag(argv, "--target-port") + save(state) + return + + if argv[:3] == ["containerapp", "registry", "set"]: + state["registry_identity"] = flag(argv, "--identity") + state["registry_server"] = flag(argv, "--server") + save(state) + return + + if argv[:2] == ["containerapp", "update"]: + if state.get("containerapp_update_fails"): + fail("ERROR: failed to update the container app") + state["deployed_image"] = flag(argv, "--image") + state["latest_revision"] = state["new_revision"] + save(state) + return + + fail(f"ERROR: fake az: unsupported invocation: {{joined}}") + + +main() +''' + + +_CURL_STUB = r'''#!/usr/bin/env python3 +import json +import sys +from pathlib import Path + +STATE = Path(r"{state_file}") +LOG = Path(r"{log_path}") + +argv = sys.argv[1:] +with LOG.open("a") as log: + log.write(" ".join(argv) + "\n") + +state = json.loads(STATE.read_text()) +url = argv[-1] +state["health_probes"] = state.get("health_probes", 0) + 1 +probes = state["health_probes"] +STATE.write_text(json.dumps(state)) + +if state.get("health_never_ready"): + sys.stderr.write("curl: (22) The requested URL returned error: 404\n") + raise SystemExit(22) +if state.get("deployed_image") is None: + sys.stderr.write("curl: (7) Failed to connect\n") + raise SystemExit(7) +if probes <= state.get("health_ready_after_probes", 0): + sys.stderr.write("curl: (22) The requested URL returned error: 503\n") + raise SystemExit(22) +print(f"ok {{url}}") +''' + + +class DeployRun: + def __init__(self, result, az_log, azd_log, curl_log, state_file): + self.result = result + self._az_log = az_log + self._azd_log = azd_log + self._curl_log = curl_log + self._state_file = state_file + + @property + def returncode(self): + return self.result.returncode + + @property + def stdout(self): + return self.result.stdout + + @property + def stderr(self): + return self.result.stderr + + @property + def az_calls(self): + return self._az_log.read_text() if self._az_log.exists() else "" + + @property + def az_call_lines(self): + return [line for line in self.az_calls.splitlines() if line.strip()] + + @property + def azd_calls(self): + return self._azd_log.read_text() if self._azd_log.exists() else "" + + @property + def curl_calls(self): + return self._curl_log.read_text() if self._curl_log.exists() else "" + + @property + def state(self): + return json.loads(self._state_file.read_text()) + + def first_index(self, needle): + """Index of the first az call containing `needle`, or None.""" + for index, line in enumerate(self.az_call_lines): + if needle in line: + return index + return None + + +def run_deploy_app( + tmp_path, + assignments=None, + active_subscription_id=None, + signed_out=False, + acr_build_fails=False, + containerapp_update_fails=False, + revision_healthy_after_polls=0, + health_ready_after_probes=0, + health_never_ready=False, + env=None, + drop_env=(), + azd_values=None, + script=DEPLOY_APP, +): + """Run `azd-deploy-app.sh` against a staged fake subscription. + + Every timeout the script owns is shortened through its own documented + environment overrides so a five-minute production budget does not become + a five-minute test. + """ + tmp_path.mkdir(parents=True, exist_ok=True) + bin_dir = tmp_path / "bin" + bin_dir.mkdir(parents=True, exist_ok=True) + + state_file = tmp_path / "tenant.json" + state_file.write_text( + json.dumps( + { + "active_subscription": active_subscription_id or SUBSCRIPTION_ID, + "signed_out": signed_out, + "acr_resource_id": ACR_RESOURCE_ID, + "acr_login_server": ACR_LOGIN_SERVER, + "assignments": list( + ACR_PULL_GRANTED if assignments is None else assignments + ), + "latest_revision": INITIAL_REVISION, + "new_revision": NEW_REVISION, + "revision_healthy_after_polls": revision_healthy_after_polls, + "health_ready_after_probes": health_ready_after_probes, + "health_never_ready": health_never_ready, + "acr_build_fails": acr_build_fails, + "containerapp_update_fails": containerapp_update_fails, + "deployed_image": None, + } + ) + ) + + az_log = tmp_path / "az-calls.log" + azd_log = tmp_path / "azd-calls.log" + curl_log = tmp_path / "curl-calls.log" + write_executable( + bin_dir / "az", _AZ_STUB.format(state_file=state_file, log_path=az_log) + ) + write_executable( + bin_dir / "curl", _CURL_STUB.format(state_file=state_file, log_path=curl_log) + ) + write_azd_stub( + bin_dir, + {"AZURE_SUBSCRIPTION_ID": SUBSCRIPTION_ID} + if azd_values is None + else azd_values, + "azd_1_29", + azd_log, + ) + + workdir = tmp_path / "elsewhere" + workdir.mkdir(parents=True, exist_ok=True) + + process_env = { + "PATH": f"{bin_dir}{os.pathsep}{os.environ.get('PATH', '')}", + "HOME": os.environ.get("HOME", str(tmp_path)), + "AZURE_SUBSCRIPTION_ID": SUBSCRIPTION_ID, + "AZURE_RESOURCE_GROUP": RESOURCE_GROUP, + "AZURE_ACR_NAME": ACR_NAME, + "AZURE_CONTAINER_APP_NAME": CONTAINER_APP_NAME, + "AZURE_CONTAINER_APP_FQDN": CONTAINER_APP_FQDN, + "AZURE_CONTAINER_APP_PRINCIPAL_ID": WORKLOAD_PRINCIPAL_ID, + "AZURE_WORKLOAD_IDENTITY_RESOURCE_ID": WORKLOAD_IDENTITY_RESOURCE_ID, + "SRE_ACR_PULL_TIMEOUT_SECONDS": "2", + "SRE_ACR_PULL_POLL_INTERVAL_SECONDS": "0.2", + "SRE_REVISION_READY_TIMEOUT_SECONDS": "2", + "SRE_HEALTH_TIMEOUT_SECONDS": "2", + "SRE_DEPLOY_POLL_INTERVAL_SECONDS": "0.2", + } + for name in drop_env: + process_env.pop(name, None) + process_env.update(env or {}) + + result = subprocess.run( + [str(script)], + capture_output=True, + text=True, + env=process_env, + cwd=str(workdir), + ) + return DeployRun(result, az_log, azd_log, curl_log, state_file) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py new file mode 100644 index 0000000..f13d904 --- /dev/null +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py @@ -0,0 +1,396 @@ +"""Behaviour tests for `azd-deploy-app.sh`, the lab's `postdeploy` hook. + +Why this hook exists at all, and why these tests are about *ordering*: + +The lab runs on Container Apps pulling from ACR with a user-assigned +managed identity. Bicep creates the registry, the identity, the `AcrPull` +role assignment, and a Container App running a *public placeholder* image +-- it cannot deploy the lab image, because that image does not exist until +something builds it, and the identity's `AcrPull` grant is not usable the +instant the deployment returns. That makes the deployment inherently +two-phase, and the phases must not be collapsed: + + azd provision -> infrastructure + placeholder image, nothing else + azd deploy -> wait for AcrPull, build in ACR, move the app onto it + +The previous design did the second phase inside `postprovision`, which +starts the ACR build immediately after the ARM deployment returns -- no +role check at all. These tests pin the corrected contract: the hook must +observe the *exact* `AcrPull` assignment (this principal, this role +definition, this registry scope) before its first deploy action, and must +never build or update anything if that assignment never appears. + +`azd deploy` and `azd up` both run this hook even though the project +declares no services -- verified against azd 1.29.0 on 2026-08-14, both by +running `azd deploy --all --no-prompt` / `azd deploy --no-prompt` against a +service-less project (project `predeploy`/`postdeploy` hooks ran; a +non-zero hook exit failed the command with +`ERROR: failed running post hooks: 'postdeploy' hook failed with exit +code: '3'`) and in azd's own source: `cli/azd/internal/cmd/up_graph.go` +builds `cmdhook-predeploy`/`cmdhook-postdeploy` steps unconditionally and +explicitly keeps them for "Zero-service projects", calling the same +`runProjectCommandHook` (`cli/azd/internal/cmd/project_hooks.go`) the +cobra hooks middleware uses for stand-alone `azd deploy`. +""" +import re + +import pytest + +from deploy_app_harness import ( + ACR_LOGIN_SERVER, + ACR_NAME, + ACR_PULL_ROLE_DEFINITION_ID, + ACR_PUSH_ROLE_DEFINITION_ID, + ACR_RESOURCE_ID, + CONTAINER_APP_FQDN, + DEPLOY_APP, + LAB_ROOT, + OTHER_SUBSCRIPTION_ID, + RESOURCE_GROUP_SCOPE, + WORKLOAD_IDENTITY_RESOURCE_ID, + assignment, + run_deploy_app, +) + +SUBSCRIPTION_PIN = '--subscription "${AZURE_SUBSCRIPTION_ID}"' +ACTIVE_ACCOUNT_PROBE = "az account show --query id" + +# Every call that changes the deployed application. None of them may run +# before the AcrPull poll succeeds. +DEPLOY_ACTIONS = ( + "acr build", + "containerapp registry set", + "containerapp ingress update", + "containerapp update", +) + + +def _az_invocations(script_text): + """Every `az ...` command in a script, with line continuations joined.""" + joined = re.sub(r"\\\n\s*", " ", script_text) + commands = [] + for line in joined.splitlines(): + if line.strip().startswith("#"): + continue + for segment in re.split(r"\$\(|\|\||&&|\||;|`", line): + stripped = re.sub( + r"^(?:if\s+|until\s+|while\s+|then\s+|else\s+|do\s+|!\s*)+", + "", + segment.strip(), + ) + if re.match(r"^az\s", stripped): + commands.append(re.sub(r"\s+", " ", stripped).strip()) + return commands + + +# --- The script exists and is wired as a program --------------------------- + + +def test_deploy_hook_script_exists_and_is_executable(): + assert DEPLOY_APP.is_file(), ( + "the deploy phase needs its own hook script; the two-phase " + "Container Apps + ACR managed-identity flow cannot live in postprovision" + ) + assert DEPLOY_APP.stat().st_mode & 0o111, "azd runs the hook as a program" + + +# --- Static contract ------------------------------------------------------- + + +def test_every_azure_cli_call_is_pinned_to_the_target_subscription(): + for command in _az_invocations(DEPLOY_APP.read_text()): + if command.startswith(ACTIVE_ACCOUNT_PROBE): + continue + assert SUBSCRIPTION_PIN in command, ( + f"azd-deploy-app.sh runs an unpinned Azure CLI command: {command}" + ) + + +def test_requires_every_provision_output_it_consumes(): + text = DEPLOY_APP.read_text() + + for value in ( + "AZURE_SUBSCRIPTION_ID:?", + "AZURE_RESOURCE_GROUP:?", + "AZURE_ACR_NAME:?", + "AZURE_CONTAINER_APP_NAME:?", + "AZURE_CONTAINER_APP_FQDN:?", + "AZURE_CONTAINER_APP_PRINCIPAL_ID:?", + "AZURE_WORKLOAD_IDENTITY_RESOURCE_ID:?", + ): + assert value in text, f"the hook must fail fast without {value}" + + +def test_polls_the_exact_acr_pull_role_before_any_deploy_action(): + """The whole point of the gate, asserted on the source as well as on + behaviour: the role poll has to come first, textually and in time.""" + text = DEPLOY_APP.read_text() + + assert ACR_PULL_ROLE_DEFINITION_ID in text, ( + "the poll must match AcrPull by role definition ID, not by a " + "display name that any custom role could also carry" + ) + poll_at = text.index("az role assignment list") + for action in DEPLOY_ACTIONS: + action_at = text.index(f"az {action}") + assert poll_at < action_at, ( + f"az {action} appears before the AcrPull poll in the script" + ) + + +def test_role_poll_budget_is_five_minutes_by_default(): + text = DEPLOY_APP.read_text() + + assert re.search( + r"SRE_ACR_PULL_TIMEOUT_SECONDS:-300", text + ), "the documented AcrPull propagation budget is 5 minutes" + assert re.search( + r"SRE_ACR_PULL_POLL_INTERVAL_SECONDS:-10", text + ), "the poll interval must default to 10s, not to a busy loop" + + +def test_never_builds_the_image_locally(): + """The lab has no local Docker requirement: the image is built in ACR.""" + text = DEPLOY_APP.read_text() + + assert "az acr build" in text + assert not re.search(r"^\s*docker\s", text, flags=re.MULTILINE) + assert "docker build" not in text + + +# --- Behaviour: the gate --------------------------------------------------- + + +def test_waits_for_acr_pull_to_propagate_before_building(tmp_path): + """The assignment exists but is not visible yet on the first two polls + -- exactly the propagation window the gate exists for. The hook must + keep polling and only then start the build.""" + run = run_deploy_app( + tmp_path, + assignments=[assignment(available_after_attempts=2)], + ) + + assert run.returncode == 0, run.stdout + run.stderr + poll_lines = [ + line for line in run.az_call_lines if line.startswith("role assignment list") + ] + assert len(poll_lines) == 3, ( + f"expected three polls before the grant became visible: {poll_lines!r}" + ) + build_at = run.first_index("acr build") + assert build_at is not None, "the hook never built the image" + assert build_at > run.az_call_lines.index(poll_lines[-1]), ( + "the build must start only after the poll that saw the grant" + ) + + +def test_makes_no_deploy_call_at_all_until_the_grant_is_visible(tmp_path): + run = run_deploy_app( + tmp_path, + assignments=[assignment(available_after_attempts=2)], + ) + + assert run.returncode == 0, run.stdout + run.stderr + first_grant_seen = max( + index + for index, line in enumerate(run.az_call_lines) + if line.startswith("role assignment list") + ) + for action in DEPLOY_ACTIONS: + action_at = run.first_index(action) + assert action_at is not None, f"the hook never ran az {action}" + assert action_at > first_grant_seen, ( + f"az {action} ran before the AcrPull grant was observed: " + f"{run.az_call_lines!r}" + ) + + +def test_stops_without_building_when_acr_pull_never_appears(tmp_path): + run = run_deploy_app(tmp_path, assignments=[]) + + assert run.returncode != 0 + assert "AcrPull" in run.stderr + assert ACR_RESOURCE_ID in run.stderr, ( + "the failure must name the exact scope that was polled" + ) + assert run.first_index("acr build") is None, ( + f"the hook built the image without the grant: {run.az_call_lines!r}" + ) + assert run.first_index("containerapp update") is None + assert run.first_index("containerapp ingress update") is None + assert "SRE_CONTAINER_IMAGE" not in run.azd_calls + + +def test_ignores_an_acr_pull_grant_at_a_wider_scope(tmp_path): + """A resource-group-scoped AcrPull would let the app pull, but it is not + the assignment this lab creates; accepting it would make the gate pass + on an unrelated grant while the lab's own assignment is still + propagating.""" + run = run_deploy_app( + tmp_path, + assignments=[assignment(scope=RESOURCE_GROUP_SCOPE)], + ) + + assert run.returncode != 0 + assert run.first_index("acr build") is None + + +def test_ignores_a_different_role_at_the_registry_scope(tmp_path): + run = run_deploy_app( + tmp_path, + assignments=[assignment(role_guid=ACR_PUSH_ROLE_DEFINITION_ID)], + ) + + assert run.returncode != 0 + assert run.first_index("acr build") is None + + +def test_ignores_an_acr_pull_grant_for_a_different_principal(tmp_path): + run = run_deploy_app( + tmp_path, + assignments=[assignment(principal_id="ffffffff-0000-4000-8000-ffffffffffff")], + ) + + assert run.returncode != 0 + assert run.first_index("acr build") is None + + +def test_accepts_the_grant_when_arm_returns_a_differently_cased_scope(tmp_path): + """ARM echoes `resourcegroups` or `resourceGroups` depending on how the + assignment was created. Resource IDs are case-insensitive, so a casing + difference must not stall the deployment for the full five minutes.""" + run = run_deploy_app( + tmp_path, + assignments=[ + assignment(scope=ACR_RESOURCE_ID.replace("resourceGroups", "resourcegroups")) + ], + ) + + assert run.returncode == 0, run.stdout + run.stderr + assert run.first_index("acr build") is not None + + +# --- Behaviour: the deployment itself -------------------------------------- + + +def test_builds_in_acr_and_moves_the_app_onto_the_built_image(tmp_path): + run = run_deploy_app(tmp_path) + + assert run.returncode == 0, run.stdout + run.stderr + build_line = run.az_call_lines[run.first_index("acr build")] + assert f"--registry {ACR_NAME}" in build_line + assert "sre-event-lab:" in build_line + assert str(LAB_ROOT / "app") in build_line, ( + "the ACR build context must be the lab's app directory" + ) + + deployed_image = run.state["deployed_image"] + assert deployed_image.startswith(f"{ACR_LOGIN_SERVER}/sre-event-lab:") + + +def test_configures_the_registry_with_the_workload_identity(tmp_path): + run = run_deploy_app(tmp_path) + + assert run.returncode == 0, run.stdout + run.stderr + assert run.state["registry_identity"] == WORKLOAD_IDENTITY_RESOURCE_ID + assert run.state["registry_server"] == ACR_LOGIN_SERVER + + +def test_moves_ingress_to_the_app_port_before_the_image_update(tmp_path): + """The placeholder serves 80 and the lab image serves 8000.""" + run = run_deploy_app(tmp_path) + + assert run.returncode == 0, run.stdout + run.stderr + assert run.state["target_port"] == "8000" + assert run.first_index("containerapp ingress update") < run.first_index( + "containerapp update --" + ) + + +def test_waits_for_a_new_healthy_revision_and_a_healthy_endpoint(tmp_path): + run = run_deploy_app( + tmp_path, revision_healthy_after_polls=2, health_ready_after_probes=2 + ) + + assert run.returncode == 0, run.stdout + run.stderr + assert run.state["revision_polls"] > 2 + assert run.state["health_probes"] > 2 + assert f"https://{CONTAINER_APP_FQDN}/healthz" in run.curl_calls + + +def test_records_the_built_image_in_the_azd_environment(tmp_path): + run = run_deploy_app(tmp_path) + + assert run.returncode == 0, run.stdout + run.stderr + assert "env set SRE_IMAGE_TAG run-" in run.azd_calls + assert f"env set SRE_CONTAINER_IMAGE {ACR_LOGIN_SERVER}/sre-event-lab:" in run.azd_calls + assert f"cwd={LAB_ROOT}" in run.azd_calls, ( + "the hook must pin `azd env set` to the lab project, not to the " + "working directory azd happens to run it from" + ) + + +def test_does_not_record_an_image_that_never_became_healthy(tmp_path): + run = run_deploy_app(tmp_path, health_never_ready=True) + + assert run.returncode != 0 + assert "SRE_CONTAINER_IMAGE" not in run.azd_calls, ( + "persisting an unhealthy image would make the next `azd provision` " + "deploy it again as if it were known good" + ) + + +# --- Behaviour: preconditions ---------------------------------------------- + + +def test_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path): + run = run_deploy_app(tmp_path, signed_out=True) + + assert run.returncode != 0 + assert "az login" in run.stderr + assert "Please run 'az login' to setup account." not in run.stderr + assert run.az_calls.strip() == "account show --query id -o tsv", ( + f"the hook must stop at the failed login check: {run.az_calls!r}" + ) + + +def test_reports_the_mismatch_when_the_cli_targets_another_subscription(tmp_path): + run = run_deploy_app(tmp_path, active_subscription_id=OTHER_SUBSCRIPTION_ID) + + assert run.returncode == 0, run.stdout + run.stderr + assert OTHER_SUBSCRIPTION_ID in run.stderr + + +@pytest.mark.parametrize( + "missing", + [ + "AZURE_RESOURCE_GROUP", + "AZURE_ACR_NAME", + "AZURE_CONTAINER_APP_NAME", + "AZURE_CONTAINER_APP_FQDN", + "AZURE_CONTAINER_APP_PRINCIPAL_ID", + "AZURE_WORKLOAD_IDENTITY_RESOURCE_ID", + ], +) +def test_stops_before_any_azure_call_when_a_provision_output_is_missing( + tmp_path, missing +): + run = run_deploy_app(tmp_path, drop_env=(missing,)) + + assert run.returncode != 0 + assert missing in run.stderr + assert "azd provision" in run.stderr, ( + "a missing deployment output means provisioning has not run (or is " + "stale); the message must say so" + ) + assert run.az_calls.strip() == "", ( + f"the hook spent an Azure call before its own guards: {run.az_calls!r}" + ) + + +def test_fails_when_the_acr_build_fails(tmp_path): + run = run_deploy_app(tmp_path, acr_build_fails=True) + + assert run.returncode != 0 + assert run.first_index("containerapp update") is None + assert "SRE_CONTAINER_IMAGE" not in run.azd_calls diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py index f8b7d66..5f865af 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_hooks.py @@ -6,18 +6,26 @@ operation therefore has to be pinned to AZURE_SUBSCRIPTION_ID, and the `predown` hook has to survive a lab that never configured the Agent. +The `postprovision` hook is deliberately *local only*: it prepares +`app/.venv` and nothing else. Building the lab image and moving the +Container App onto it belongs to the deploy phase +(`scripts/azd-deploy-app.sh`, tested in `test_azd_deploy_app.py`), because +the workload identity's `AcrPull` grant is not necessarily usable the +moment provisioning returns. + Every test that *executes* a hook script runs it from a `lab_copy` -- -never the real, in-place `azd-configure.sh` / `azd-postprovision.sh` / -`setup-venv.sh`. `azd-postprovision.sh` calls `setup-venv.sh`, and -`setup-venv.sh` resolves its own `SCRIPT_DIR`/`LAB_ROOT`/`VENV_DIR` from +never the real, in-place `azd-configure.sh` / +`azd-postprovision-local.sh` / `setup-venv.sh`. +`azd-postprovision-local.sh` calls `setup-venv.sh`, and `setup-venv.sh` +resolves its own `SCRIPT_DIR`/`LAB_ROOT`/`VENV_DIR` from its own `${BASH_SOURCE[0]}`, never from the caller's cwd -- so running the -real, in-place `azd-postprovision.sh` (as this file once did, pointing +real, in-place hook (as this file once did, pointing only the *subprocess's* cwd or `AZURE_*` environment at a scratch `tmp_path`) still always resolves `VENV_DIR` to the real, developer-machine `app/.venv`, and setup-venv.sh's `uv venv --allow-existing` / `uv pip install` would then run for real against it -- a real filesystem mutation and a real network/package-index call, entirely unrelated to what the test -claims to be checking (the Azure CLI login-failure message). `lab_copy` +claims to be checking. `lab_copy` copies the whole `scripts/`+`app/` layout the scripts depend on into `tmp_path`, so every script's own path resolution lands entirely inside `tmp_path`; `real_lab_venv_tree_is_never_touched` (autouse, this whole @@ -45,7 +53,8 @@ SCRIPTS_DIR = Path(__file__).parents[1] LAB_ROOT = Path(__file__).parents[2] AZD_CONFIGURE = SCRIPTS_DIR / "azd-configure.sh" -AZD_POSTPROVISION = SCRIPTS_DIR / "azd-postprovision.sh" +AZD_POSTPROVISION = SCRIPTS_DIR / "azd-postprovision-local.sh" +AZD_DEPLOY_APP = SCRIPTS_DIR / "azd-deploy-app.sh" SETUP_VENV = SCRIPTS_DIR / "setup-venv.sh" REQUIREMENTS = LAB_ROOT / "app" / "requirements.txt" REQUIREMENTS_DEV = LAB_ROOT / "app" / "requirements-dev.txt" @@ -54,6 +63,14 @@ # The one deliberately unpinned call: it reads whichever account is active # so the hook can report a mismatch. ACTIVE_ACCOUNT_PROBE = "az account show --query id" +# Everything that changes the deployed application. The provision phase +# must contain none of it. +DEPLOY_ACTIONS = ( + "az acr build", + "az containerapp update", + "az containerapp ingress update", + "az containerapp registry set", +) # --- Regression tripwire: the real app/.venv must never move --------------- @@ -147,7 +164,7 @@ def real_lab_venv_tree_is_never_touched(): @pytest.fixture def lab_copy(tmp_path): """A throwaway copy of exactly the layout the hook scripts depend on: - `azd-configure.sh`, `azd-postprovision.sh`, and the *real* + `azd-configure.sh`, `azd-postprovision-local.sh`, and the *real* `setup-venv.sh` under `scripts/`, the real `app/requirements.txt` and `app/requirements-dev.txt` under `app/` (setup-venv.sh's own `REQUIREMENTS_FILE`), and a placeholder `azure.yaml` at the copied lab @@ -163,7 +180,7 @@ def lab_copy(tmp_path): scripts_dir.mkdir() for source, name in ( (AZD_CONFIGURE, "azd-configure.sh"), - (AZD_POSTPROVISION, "azd-postprovision.sh"), + (AZD_POSTPROVISION, "azd-postprovision-local.sh"), (SETUP_VENV, "setup-venv.sh"), ): copy = scripts_dir / name @@ -180,7 +197,7 @@ def lab_copy(tmp_path): return SimpleNamespace( root=tmp_path, configure=scripts_dir / "azd-configure.sh", - postprovision=scripts_dir / "azd-postprovision.sh", + postprovision=scripts_dir / "azd-postprovision-local.sh", setup_venv=scripts_dir / "setup-venv.sh", app_dir=app_dir, ) @@ -344,81 +361,120 @@ def test_azd_configure_pins_every_azure_cli_call_to_the_target_subscription(): def test_azd_postprovision_pins_every_azure_cli_call_to_the_target_subscription(): + """Vacuously true today -- the provision-phase hook makes no Azure CLI + call at all -- but kept so that any Azure call added back to it has to + carry the same subscription pin as every other lab entry point.""" for command in _az_invocations(AZD_POSTPROVISION.read_text()): if command.startswith(ACTIVE_ACCOUNT_PROBE): continue assert SUBSCRIPTION_PIN in command, ( - f"azd-postprovision.sh runs an unpinned Azure CLI command: {command}" + f"azd-postprovision-local.sh runs an unpinned Azure CLI command: {command}" ) -def test_azd_hooks_require_the_target_subscription_and_verify_the_active_account(): - for script in (AZD_CONFIGURE, AZD_POSTPROVISION): - text = script.read_text() - assert "AZURE_SUBSCRIPTION_ID:?" in text, ( - f"{script.name} must fail fast when azd did not provide a subscription" - ) - assert ACTIVE_ACCOUNT_PROBE in text, ( - f"{script.name} must verify which subscription the Azure CLI is signed in to" - ) +def test_azd_configure_requires_the_target_subscription_and_verifies_the_active_account(): + text = AZD_CONFIGURE.read_text() + assert "AZURE_SUBSCRIPTION_ID:?" in text, ( + "azd-configure.sh must fail fast when azd did not provide a subscription" + ) + assert ACTIVE_ACCOUNT_PROBE in text, ( + "azd-configure.sh must verify which subscription the Azure CLI is signed in to" + ) -def test_azd_postprovision_targets_only_current_azd_values(): +def test_postprovision_never_builds_or_updates_the_container_app(): + """The provision phase must leave the public placeholder image running. + + Bicep creates the registry, the workload identity, its `AcrPull` + assignment and a placeholder-image Container App in one deployment. + Building and switching the image straight afterwards -- what this hook + used to do -- starts an ACR pull with a role assignment that has just + been created, with no check that it is usable yet. Those steps belong + to `azd-deploy-app.sh`, behind the AcrPull poll. + """ text = AZD_POSTPROVISION.read_text() - assert "rg-sre-agent-event-lab-krc" not in text - assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in text - assert "common.sh" not in text - for value in ( - "AZURE_RESOURCE_GROUP:?", - "AZURE_ACR_NAME:?", - "AZURE_CONTAINER_APP_NAME:?", - "AZURE_CONTAINER_APP_FQDN:?", - ): - assert value in text + for action in DEPLOY_ACTIONS: + assert action not in text, ( + f"the provision-phase hook still runs `{action}`" + ) + assert "SRE_CONTAINER_IMAGE" not in text, ( + "recording a built image is part of the deploy phase" + ) + assert "/healthz" not in text, ( + "the placeholder image serves no /healthz; verifying it belongs to " + "the deploy phase" + ) -def test_azd_postprovision_moves_ingress_to_the_app_port_and_records_the_image(): - """The provisioned placeholder listens on port 80; the lab image listens - on 8000. postprovision must move ingress before verifying /healthz, and - persist the built image so a later `azd provision` does not revert the - Container App to the placeholder. - """ - text = AZD_POSTPROVISION.read_text() +def test_postprovision_only_prepares_the_local_python_environment(tmp_path, lab_copy): + """Executed, not read: the hook runs `setup-venv.sh` (against a fake + `uv`) and spends no Azure API call at all -- not even the login probe, + which has nothing to check when nothing is deployed.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + az_log = tmp_path / "az-calls.log" + uv_log = tmp_path / "uv-calls.log" + azd_log = tmp_path / "azd-calls.log" + _write_az_stub(bin_dir, az_log) + _write_fake_uv(bin_dir, uv_log) + write_azd_stub(bin_dir, {}, "azd_1_29", azd_log) - assert "az containerapp ingress update" in text - assert "--target-port" in text - assert "8000" in text - assert "azd env set SRE_CONTAINER_IMAGE" in text - assert "/healthz" in text + env = dict(os.environ) + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" + env["AZURE_SUBSCRIPTION_ID"] = "11111111-2222-3333-4444-555555555555" + + result = subprocess.run( + [str(lab_copy.postprovision)], + capture_output=True, + text=True, + env=env, + ) - ingress_at = text.index("az containerapp ingress update") - healthz_at = text.index("/healthz") - assert ingress_at < healthz_at + assert result.returncode == 0, result.stdout + result.stderr + assert not az_log.exists() or az_log.read_text().strip() == "", ( + f"the provision-phase hook made an Azure CLI call: {az_log.read_text()!r}" + ) + uv_calls = uv_log.read_text() if uv_log.exists() else "" + assert "venv" in uv_calls and "pip install" in uv_calls, ( + f"the hook did not prepare the local environment: {uv_calls!r}" + ) + created_python = lab_copy.app_dir / ".venv" / "bin" / "python" + assert created_python.is_file() + assert not str(created_python).startswith(str(LAB_ROOT)) -def test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call(): - text = AZD_POSTPROVISION.read_text() +def test_postprovision_says_where_the_application_deployment_happens(tmp_path, lab_copy): + """`azd provision` on its own leaves the placeholder image running, so + the hook that ends the provision phase has to say what still has to + happen -- otherwise a successful `azd provision` looks like a finished + lab.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + _write_fake_uv(bin_dir, tmp_path / "uv-calls.log") + + env = dict(os.environ) + env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" - assert "setup-venv.sh" in text - setup_venv_at = text.index("setup-venv.sh") - first_account_show_at = text.index("az account show") - assert setup_venv_at < first_account_show_at, ( - "setup-venv.sh must run before the hook makes any Azure CLI call" + result = subprocess.run( + [str(lab_copy.postprovision)], + capture_output=True, + text=True, + env=env, ) + assert result.returncode == 0, result.stdout + result.stderr + assert "azd deploy" in result.stdout + result.stderr -def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp_path, lab_copy): - """`app/.venv` setup is local and has nothing to do with the Azure CLI, - but a broken corporate proxy or missing `uv` must still stop the hook - before it spends a single Azure API call -- the cloud side is already - provisioned by the time this hook runs, so failing fast here changes - nothing about that, but a failure must never be masked by continuing - on to the ACR build.""" + +def test_postprovision_fails_when_the_local_environment_cannot_be_prepared( + tmp_path, lab_copy +): + """A broken corporate proxy or a missing `uv` must fail the hook rather + than let `azd provision` report success over a half-prepared lab.""" lab_copy.setup_venv.write_text( "#!/usr/bin/env bash\n" "echo 'uv is required to set up app/.venv but was not found on PATH.' >&2\n" - "echo 'azd hooks run postprovision' >&2\n" "exit 1\n" ) lab_copy.setup_venv.chmod(0o755) @@ -430,11 +486,6 @@ def test_azd_postprovision_stops_before_any_azure_call_when_setup_venv_fails(tmp env = dict(os.environ) env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" - env["AZURE_SUBSCRIPTION_ID"] = "11111111-2222-3333-4444-555555555555" - env["AZURE_RESOURCE_GROUP"] = "rg-test" - env["AZURE_ACR_NAME"] = "acrtest" - env["AZURE_CONTAINER_APP_NAME"] = "ca-test" - env["AZURE_CONTAINER_APP_FQDN"] = "ca-test.example.com" result = subprocess.run( [str(lab_copy.postprovision)], @@ -466,67 +517,17 @@ def test_azd_configure_reports_a_clear_error_when_the_azure_cli_is_not_logged_in ) -def test_azd_postprovision_reports_a_clear_error_when_the_azure_cli_is_not_logged_in(tmp_path, lab_copy): - """`azd-postprovision.sh` runs `setup-venv.sh` *before* it ever checks - the Azure CLI login state (see - `test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call`), so - this test's `bin_dir` carries both a login-failing fake `az` and a - fake `uv` -- the real, copied `setup-venv.sh` genuinely runs its own - `uv venv` / `uv pip install` / Pillow-import logic against the fake - `uv` (never a stub that skips that logic outright), and only *then* - does the hook reach and fail the login check. `uv_calls` is the - call-log assertion proving the fake -- never the real -- `uv` did that - work, so no real virtual environment or package index was ever - touched. - """ - bin_dir = tmp_path / "bin" - bin_dir.mkdir() - az_log = tmp_path / "az-calls.log" - uv_log = tmp_path / "uv-calls.log" - _write_login_failing_az_stub(bin_dir, az_log) - _write_fake_uv(bin_dir, uv_log) - - env = dict(os.environ) - env["PATH"] = f"{bin_dir}{os.pathsep}{env['PATH']}" - env["AZURE_SUBSCRIPTION_ID"] = "11111111-2222-3333-4444-555555555555" - env["AZURE_RESOURCE_GROUP"] = "rg-test" - env["AZURE_ACR_NAME"] = "acrtest" - env["AZURE_CONTAINER_APP_NAME"] = "ca-test" - env["AZURE_CONTAINER_APP_FQDN"] = "ca-test.example.com" - - result = subprocess.run( - [str(lab_copy.postprovision)], - capture_output=True, - text=True, - env=env, - ) - - assert result.returncode != 0 - assert "az login" in result.stderr - assert "Please run 'az login' to setup account." not in result.stderr - az_calls = az_log.read_text() if az_log.exists() else "" - assert az_calls.strip() == "account show --query id -o tsv", ( - "the hook must exit immediately after the failed login check, " - f"before any other az call: {az_calls!r}" - ) - - uv_calls = uv_log.read_text() if uv_log.exists() else "" - assert "venv" in uv_calls, ( - "setup-venv.sh must have run its real uv-venv logic against the " - f"fake uv before the login check failed: {uv_calls!r}" - ) - assert "pip install" in uv_calls, ( - "setup-venv.sh must have run its real uv-pip-install logic against " - f"the fake uv before the login check failed: {uv_calls!r}" - ) - created_python = lab_copy.app_dir / ".venv" / "bin" / "python" - assert created_python.is_file(), ( - "setup-venv.sh must have created its venv under lab_copy's " - "tmp_path, proving the fake uv -- not the real one -- ran" - ) - assert not str(created_python).startswith(str(LAB_ROOT)), ( - "the venv setup-venv.sh created must never live under the real lab tree" - ) +def test_deploy_actions_live_only_in_the_deploy_phase_hook(): + """One place, and only one place, changes the running application.""" + deploy_text = AZD_DEPLOY_APP.read_text() + for action in DEPLOY_ACTIONS: + assert action in deploy_text, ( + f"`{action}` must live in the deploy-phase hook" + ) + for script in (AZD_CONFIGURE, AZD_POSTPROVISION): + assert action not in script.read_text(), ( + f"`{action}` must not run in the provision phase ({script.name})" + ) def _run_azd_configure(tmp_path, lab_copy, azd_values, missing_key_mode="azd_1_29"): diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py b/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py index e4e3373..495b46f 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_cleanup_external.py @@ -13,7 +13,7 @@ destroys anything, and must never touch the azd environment values -- an operator who answers "no" at azd's delete prompt keeps a working environment. -* `postdown` clears the image values `azd-postprovision.sh` recorded, once +* `postdown` clears the image values `azd-deploy-app.sh` recorded, once the resources they point at are really gone. """ import os @@ -464,7 +464,7 @@ def test_role_cleanup_leaves_the_azd_environment_untouched(tmp_path): def test_reset_image_env_clears_only_the_hook_set_image_values(tmp_path): - """`azd down` deletes the ACR that `azd-postprovision.sh` recorded in + """`azd down` deletes the ACR that `azd-deploy-app.sh` recorded in SRE_CONTAINER_IMAGE/SRE_IMAGE_TAG. Reusing the environment afterwards would make `azd provision` redeploy an image tag that no longer exists, so `postdown` clears both once the resources are really gone.""" diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index f6e8b0c..b040d74 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -450,27 +450,47 @@ def test_agent_setup_guide_offers_a_python_environment_remedy(): assert "./scripts/setup-venv.sh" in section -def test_readme_does_not_claim_only_a_local_step_remains_after_postprovision(): - """Finding #2 (Task 6 follow-up): `setup-venv.sh` runs *before* the same - postprovision hook's ACR build and Container App update, so a failure - there means the cloud app deployment has not happened yet -- not merely - that "only a local step" is left. The README must send an operator to - re-run the whole hook, not offer `./scripts/setup-venv.sh` as an - equally-valid alternative for this specific failure. +def test_readme_documents_the_two_phase_provision_and_deploy_flow(): + """`azd provision` alone leaves the public placeholder image running: + the lab image is built and switched in only during the deploy phase, + after the workload identity's `AcrPull` grant is observable at the + registry. The README's deployment section is the one place an operator + learns that, so it must name both phases and the gate between them -- + and must not describe the image build as part of postprovision. """ text = README.read_text() heading = "## 배포" assert heading in text section = text.split(heading, 1)[1].split("## ", 1)[0] - assert "setup-venv.sh" in section - assert "azd hooks run postprovision" in section - assert "또는 `./scripts/setup-venv.sh`" not in section, ( - "the README must not offer running setup-venv.sh directly as an " - "alternative recovery for a postprovision-hook failure" + + assert "azd up" in section + assert "azd provision" in section + assert "azd deploy" in section + assert "AcrPull" in section, ( + "the gate the deploy phase waits on has to be named" + ) + assert "postprovision" not in section or "setup-venv.sh" in section + assert not re.search(r"postprovision[^\n]*(ACR 빌드|이미지 교체)", section), ( + "the README must not describe the ACR build or the image switch as " + "part of the postprovision hook" ) +def test_readme_recovery_matches_the_hook_that_actually_failed(): + """The provision-phase hook only prepares `app/.venv`, so + `./scripts/setup-venv.sh` is a complete recovery for it -- and the + application deployment is a separate `azd deploy` the operator still + has to run. + """ + text = README.read_text() + heading = "## 배포" + section = text.split(heading, 1)[1].split("## ", 1)[0] + + assert "setup-venv.sh" in section + assert "azd deploy" in section + + def test_guides_do_not_request_secrets_in_environment(): text = "\n".join(path.read_text() for path in GUIDES.glob("*.md")) for forbidden in ("GITHUB_PAT=", "OAUTH_TOKEN=", "CLIENT_SECRET="): diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py b/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py index 6eb75aa..33af8de 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_setup_venv.py @@ -1,5 +1,6 @@ """Behaviour tests for `setup-venv.sh`, the idempotent preparer of -`app/.venv` invoked from the `postprovision` azd hook. +`app/.venv` and the entire job of the `postprovision` azd hook +(`scripts/azd-postprovision-local.sh`). Every test drives the real script's *logic* against a fake `uv` on PATH (never a real network install) -- but never the real script *file in @@ -191,8 +192,7 @@ def test_fails_actionably_with_no_pip_fallback_when_uv_is_missing(tmp_path, lab_ assert result.returncode != 0 assert "uv" in result.stderr assert "install uv" in result.stderr.lower() or "uv is required" in result.stderr - assert "azd hooks run postprovision" in result.stderr - assert "already deployed" in result.stderr + _assert_hint_explains_the_deploy_phase(result.stderr) def test_succeeds_and_calls_uv_venv_then_uv_pip_install_with_requirements_dev(tmp_path, lab_copy): @@ -260,8 +260,7 @@ def test_fails_actionably_when_uv_venv_creation_fails(tmp_path, lab_copy): result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 - assert "azd hooks run postprovision" in result.stderr - assert "already deployed" in result.stderr + _assert_hint_explains_the_deploy_phase(result.stderr) assert "pip install" not in log_path.read_text() @@ -274,8 +273,7 @@ def test_fails_actionably_when_uv_pip_install_fails(tmp_path, lab_copy): result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 - assert "azd hooks run postprovision" in result.stderr - assert "already deployed" in result.stderr + _assert_hint_explains_the_deploy_phase(result.stderr) assert "proxy" in result.stderr @@ -289,7 +287,7 @@ def test_fails_actionably_when_pillow_is_still_not_importable(tmp_path, lab_copy assert result.returncode != 0 assert "Pillow" in result.stderr - assert "azd hooks run postprovision" in result.stderr + _assert_hint_explains_the_deploy_phase(result.stderr) def test_installs_the_labs_real_requirements_dev_file(): @@ -309,39 +307,43 @@ def test_script_never_executes_a_bare_pip_or_pip3_command(): raise AssertionError(("bare pip invocation found", stripped)) -# --- Finding #2: postprovision ordering/recovery contract ----------------- +# --- The provision/deploy split: what a failure here does and does not mean + # -# `setup-venv.sh` runs before `azd-postprovision.sh`'s ACR build and -# Container App update (see `test_azd_hooks.py`'s -# `test_azd_postprovision_runs_setup_venv_before_any_azure_cli_call`), so a -# failure here means the cloud *app* deployment for this run has not -# happened yet, only the Bicep infrastructure has. Re-running just this -# script would silently leave that deployment never attempted; every -# failure hint this script prints must send the operator to re-run the -# whole hook, and must never recommend running this script directly. -# (Doctor/capture-scenario's *own* remediation text is a separate, -# still-valid case for a local-only venv problem noticed well after a -# deployment already succeeded -- see test_lab_guides.py / -# test_doctor.py -- this contract is only about setup-venv.sh's own -# messages.) - - -def _assert_hint_never_recommends_direct_rerun(stderr: str): - assert "azd hooks run postprovision" in stderr +# `setup-venv.sh` is the whole job of the `postprovision` hook now: the +# Container App image build and the image switch moved to the deploy phase +# (`scripts/azd-deploy-app.sh`, run by `azd deploy` / `azd up`'s deploy +# phase behind the AcrPull gate). So a failure here no longer sits in the +# middle of a half-finished cloud deployment -- re-running just this script +# is a complete fix for the local part, and every failure message must say +# what still has to happen afterwards (`azd deploy`) instead of claiming the +# app deployment is about to run inside this same hook. + + +def _assert_hint_explains_the_deploy_phase(stderr: str): + assert "./scripts/setup-venv.sh" in stderr, ( + "the local environment is the only thing this hook prepares, so " + "re-running this script directly is now a complete recovery" + ) + assert "azd deploy" in stderr, ( + "the application deployment is a separate phase the operator still " + "has to run; the hint must name it" + ) lowered = stderr.lower() - assert "run: ./scripts/setup-venv.sh" not in lowered - assert "directly: ./scripts/setup-venv.sh" not in lowered - assert "or ./scripts/setup-venv.sh" not in lowered + assert "run *after* this step" not in lowered, ( + "the ACR build no longer runs later inside this same hook" + ) + assert "this hook's container app image build" not in lowered -def test_missing_uv_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): +def test_missing_uv_hint_explains_the_separate_deploy_phase(tmp_path, lab_copy): result = _run(lab_copy, tmp_path / "bin-unused", tmp_path, extra_path=False) assert result.returncode != 0 - _assert_hint_never_recommends_direct_rerun(result.stderr) + _assert_hint_explains_the_deploy_phase(result.stderr) -def test_venv_creation_failure_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): +def test_venv_creation_failure_hint_explains_the_separate_deploy_phase(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" @@ -350,10 +352,10 @@ def test_venv_creation_failure_hint_never_recommends_running_this_script_directl result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 - _assert_hint_never_recommends_direct_rerun(result.stderr) + _assert_hint_explains_the_deploy_phase(result.stderr) -def test_pip_install_failure_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): +def test_pip_install_failure_hint_explains_the_separate_deploy_phase(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" @@ -362,10 +364,10 @@ def test_pip_install_failure_hint_never_recommends_running_this_script_directly( result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 - _assert_hint_never_recommends_direct_rerun(result.stderr) + _assert_hint_explains_the_deploy_phase(result.stderr) -def test_pillow_failure_hint_never_recommends_running_this_script_directly(tmp_path, lab_copy): +def test_pillow_failure_hint_explains_the_separate_deploy_phase(tmp_path, lab_copy): bin_dir = tmp_path / "bin" bin_dir.mkdir() log_path = tmp_path / "uv-calls.log" @@ -374,7 +376,7 @@ def test_pillow_failure_hint_never_recommends_running_this_script_directly(tmp_p result = _run(lab_copy, bin_dir, tmp_path) assert result.returncode != 0 - _assert_hint_never_recommends_direct_rerun(result.stderr) + _assert_hint_explains_the_deploy_phase(result.stderr) # --- Finding #3: actionable "no matching Python" guidance ------------------ From 164470e90d203401b5a449bbc9b22a748fedb645 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 23:01:02 +0900 Subject: [PATCH 17/26] fix(sre-lab): correct ACR gate review findings (unsupported flag, Graph dependency, doctor placeholder state) Follow-up review of the AcrPull poll (`acr_pull_is_visible` in scripts/azd-deploy-app.sh, from dab05d6) found it calling `az role assignment list --assignee-principal-type ServicePrincipal`. That flag does not exist on `role assignment list` (only on `role assignment create`); verified against the installed Azure CLI (2.89.1) with a harmless real query -- it fails with `ERROR: unrecognized arguments: --assignee-principal-type ServicePrincipal`, exit 2. Every poll would have failed this way, and the deploy phase would never build the image. Fixed by removing that flag and adding the two flags the command does support for exactly this purpose: `--fill-principal-name false --fill-role-definition-name false`. `--assignee-object-id` already bypasses Microsoft Graph for the assignee filter, but both `--fill-*` flags default to true and would still query Graph to populate fields this poll never reads -- setting them false keeps the poll working with no Graph reachability at all (confirmed against the real CLI: exit 0, no warning). scripts/tests/deploy_app_harness.py's fake `az` was hardened to reject any `role assignment list` flag the real parser does not recognise (strict RED/GREEN): with the harness hardened but before the script fix, 11 tests failed with the fake's `ERROR: unrecognized arguments` (RED); after the fix, all 30 tests in test_azd_deploy_app.py pass (GREEN). Added a static contract test (no `--assignee-principal-type`, both `--fill-*` flags present) and a behavioural test asserting the poll still succeeds against the hardened fake. Also, while reviewing the same two-phase gate: * infra/main.bicep still attributed the image build/switch to "postprovision" in two comments; that work moved to the `postdeploy` hook (scripts/azd-deploy-app.sh) in dab05d6. Corrected the wording and added a regression test asserting the stale word is gone. * doctor.sh's `/healthz` check reported the same generic "Investigate: curl -v" FAIL whether `azd deploy` simply had not run yet (the documented, expected placeholder state -- port 80, no `/healthz`) or the lab image was genuinely unhealthy after deployment, misclassifying a standard pre-deploy state as a broken one. It now reads the azd environment's SRE_CONTAINER_IMAGE (set only once the deploy phase succeeds, added to common.sh's load_lab_config): empty means the deploy phase has not run, and the FAIL detail now names the exact remedy (`azd deploy --no-prompt`) instead of the generic message; once set, a `/healthz` failure is still reported as a real regression. Added behavioural tests for both branches. Full suite: 460 pytest tests green (app/infra/scripts), bash -n on every script, `az bicep build` on all five templates, and azure.yaml validated against the azd v1.0 schema -- all pass. Plan Status stays Ready for Validation; recorded the fix and updated test counts in .azure/deployment-plan.md. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 41 +++++++++++++++++-- monitor/sre-agent-event-lab/infra/main.bicep | 11 ++--- .../infra/tests/test_azd_project.py | 20 +++++++++ .../scripts/azd-deploy-app.sh | 12 +++++- monitor/sre-agent-event-lab/scripts/common.sh | 7 ++++ monitor/sre-agent-event-lab/scripts/doctor.sh | 10 +++++ .../scripts/tests/deploy_app_harness.py | 39 ++++++++++++++++++ .../scripts/tests/test_azd_deploy_app.py | 41 +++++++++++++++++++ .../scripts/tests/test_doctor.py | 39 ++++++++++++++++++ 9 files changed, 211 insertions(+), 9 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index fd849e4..6ed7f65 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -2,7 +2,7 @@ > **Status:** Ready for Validation -Updated: 2026-08-14 +Updated: 2026-08-14 (ACR gate review fixes) ## Goal @@ -104,6 +104,41 @@ No `services:` entry is declared: the only service host that would apply here (`containerapp` + `docker`) makes `azd deploy`/`azd up` require a local Docker build, which this lab deliberately avoids. +### ACR gate review fixes (2026-08-14) + +A follow-up review of the AcrPull poll (`acr_pull_is_visible` in +`scripts/azd-deploy-app.sh`) found it calling `az role assignment list` +with `--assignee-principal-type ServicePrincipal`, an option that does not +exist for `list` (only for `role assignment create`). Verified against the +installed Azure CLI (2.89.1): passing it fails with `ERROR: unrecognized +arguments: --assignee-principal-type ServicePrincipal`, exit 2, which +would have made every poll fail and the deploy phase never build the +image. Fixed by removing that flag and adding `--fill-principal-name +false --fill-role-definition-name false` (both real, supported flags), +so the poll never depends on Microsoft Graph reachability -- `--assignee- +object-id` already bypasses Graph for the filter, but the two `--fill-*` +flags default to `true` and would each still query Graph to populate +fields this poll never reads. `scripts/tests/deploy_app_harness.py`'s fake +`az` was hardened to reject any `role assignment list` flag the real +parser does not recognise, so a regression back to the unsupported flag +fails the test suite instead of silently passing against a permissive +fake. Confirmed RED (11 tests failing with `ERROR: unrecognized +arguments`) before the fix and GREEN (30/30 in +`test_azd_deploy_app.py`) after. + +Also fixed while reviewing the two-phase gate: `doctor.sh`'s `/healthz` +check previously reported the same generic "Investigate: curl -v" FAIL +whether `azd deploy` had simply not run yet (the documented, expected +placeholder state -- port 80, no `/healthz`) or the lab image was +genuinely unhealthy after deployment. It now checks the azd environment's +`SRE_CONTAINER_IMAGE` (set only once the deploy phase succeeds): when it +is empty, the failure explicitly names the state and the remedy (`Run: +azd deploy --no-prompt`) instead of the generic message; once it is set, +a `/healthz` failure is reported as a real regression as before. Stale +`main.bicep` comments attributing this image build/switch to +`postprovision` (moved to the `postdeploy` hook in the two-phase refactor +above) were also corrected. + ## Security and Safety - Container Apps reach Blob Storage through private networking. @@ -133,7 +168,7 @@ local Docker build, which this lab deliberately avoids. - [x] 5. Subscription/Location Check — current authenticated subscription, Korea Central - [x] 6. Aspire Pre-Provisioning Checks — not applicable - [ ] 7. Provision Preview — to re-run against the updated template outputs -- [x] 8. Build Verification — 455 tests and three Bicep builds passed +- [x] 8. Build Verification — 460 tests and three Bicep builds passed - [x] 9. Docker Build Context Validation — Dockerfile and requirements present; the image is built by ACR from `app/`, never locally - [x] 10. Package Validation — `azd package --all --no-prompt` passed - [x] 11. Azure Policy Validation — three assigned Defender policies are unrelated to planned resources @@ -160,7 +195,7 @@ earlier "Validated" status no longer applies): | Check | Command | Result | |---|---|---| -| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 455 passed | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 460 passed | | Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | | Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | | Bicep build | `az bicep build --file infra/{main,lab,workload}.bicep --stdout` | Passed (three templates, with the new deploy-gate outputs) | diff --git a/monitor/sre-agent-event-lab/infra/main.bicep b/monitor/sre-agent-event-lab/infra/main.bicep index 6fd5093..27a99d6 100644 --- a/monitor/sre-agent-event-lab/infra/main.bicep +++ b/monitor/sre-agent-event-lab/infra/main.bicep @@ -9,7 +9,7 @@ param location string @description('Dedicated resource group for the disposable SRE lab. Defaults to rg- when not set.') param resourceGroupName string = 'rg-${environmentName}' -@description('Container image deployed by azd. Leave empty for the first provision: the public placeholder image is used until the postprovision hook builds the lab image and records it in SRE_CONTAINER_IMAGE.') +@description('Container image deployed by azd. Leave empty for the first provision: the public placeholder image is used until the deploy phase (`postdeploy` hook, scripts/azd-deploy-app.sh) builds the lab image and records it in SRE_CONTAINER_IMAGE.') param containerImage string = '' @description('Optional Azure Monitor Action Group resource ID for event-driven SRE invocation.') @@ -25,10 +25,11 @@ param expiresOn string = '' var suffix = substring(uniqueString(subscription().id, environmentName), 0, 8) // The public placeholder serves port 80 and has no /healthz, so the first -// provision must expose port 80 without probes. Once postprovision records -// the ACR-built image in SRE_CONTAINER_IMAGE, every later provision deploys -// that image on port 8000 with matching /healthz probes instead of reverting -// to the placeholder. +// provision must expose port 80 without probes. Once the deploy phase +// (`postdeploy` hook, scripts/azd-deploy-app.sh) records the ACR-built +// image in SRE_CONTAINER_IMAGE, every later provision deploys that image +// on port 8000 with matching /healthz probes instead of reverting to the +// placeholder. var placeholderContainerImage = 'mcr.microsoft.com/azuredocs/containerapps-helloworld:latest' var effectiveContainerImage = empty(containerImage) ? placeholderContainerImage : containerImage var usesPlaceholderImage = effectiveContainerImage == placeholderContainerImage diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index 070ea10..c446c07 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -209,6 +209,26 @@ def test_main_bicep_keeps_placeholder_port_and_probes_consistent(): ) +def test_main_bicep_does_not_attribute_the_deploy_gate_to_postprovision(): + """The image build/switch was moved out of `postprovision` into the + `postdeploy` hook (`scripts/azd-deploy-app.sh`) behind the AcrPull gate + (see `test_deploy_phase_runs_the_gated_application_deployment`). A + comment that still says "postprovision" for that work is stale and + would send a reader to the wrong script. + """ + template = (LAB_ROOT / "infra" / "main.bicep").read_text() + + assert "postprovision" not in template, ( + "main.bicep still attributes the deploy-phase image build/switch to " + "postprovision; it belongs to the postdeploy hook " + "(scripts/azd-deploy-app.sh)" + ) + assert "SRE_CONTAINER_IMAGE" in template, ( + "the placeholder-vs-lab-image comments must still explain where " + "SRE_CONTAINER_IMAGE comes from" + ) + + def test_main_bicep_restores_outputs_consumed_by_lab_scripts(): template = (LAB_ROOT / "infra" / "main.bicep").read_text() diff --git a/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh b/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh index 231a351..cce09d1 100755 --- a/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh +++ b/monitor/sre-agent-event-lab/scripts/azd-deploy-app.sh @@ -122,13 +122,23 @@ lowercase() { # deliberately not accepted here: it is not the assignment this lab # creates, and treating it as one would let the gate pass while the lab's # own assignment is still propagating. +# +# `--assignee-principal-type` is not a `role assignment list` option -- +# verified against the installed Azure CLI (2.89.1): passing it here fails +# with `ERROR: unrecognized arguments: --assignee-principal-type +# ServicePrincipal`. `--assignee-object-id` alone already bypasses +# Microsoft Graph for the assignee *filter*; `--fill-principal-name` and +# `--fill-role-definition-name` default to `true` and each would still +# query Graph to populate fields this poll never reads, so both are set to +# `false` to keep the poll working with no Graph reachability at all. acr_pull_is_visible() { local granted expected line granted="$(az role assignment list \ --assignee-object-id "${AZURE_CONTAINER_APP_PRINCIPAL_ID}" \ - --assignee-principal-type ServicePrincipal \ --scope "${ACR_RESOURCE_ID}" \ --subscription "${AZURE_SUBSCRIPTION_ID}" \ + --fill-principal-name false \ + --fill-role-definition-name false \ --query "[?ends_with(roleDefinitionId, '${ACR_PULL_ROLE_DEFINITION_ID}')].scope" \ -o tsv 2>/dev/null || true)" expected="$(lowercase "${ACR_RESOURCE_ID}")" diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index 050c696..f5ab4a9 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -96,6 +96,12 @@ load_lab_config() { AZURE_WORKSPACE_ID="$(setting AZURE_WORKSPACE_ID "${AZURE_WORKSPACE_ID:-}" "")" AZURE_APP_INSIGHTS_NAME="$(setting AZURE_APP_INSIGHTS_NAME "${AZURE_APP_INSIGHTS_NAME:-}" "")" AZURE_TELEMETRY_SERVICE_NAME="$(setting AZURE_TELEMETRY_SERVICE_NAME "${AZURE_TELEMETRY_SERVICE_NAME:-}" "")" + # Set by the deploy phase (`postdeploy` hook, scripts/azd-deploy-app.sh) + # once it has built the lab image and switched the Container App onto it; + # empty until then, which is the reliable "has azd deploy run yet?" signal + # doctor.sh needs to tell the intermediate placeholder state (expected) + # apart from a real post-deploy health regression (not expected). + SRE_CONTAINER_IMAGE="$(setting SRE_CONTAINER_IMAGE "${SRE_CONTAINER_IMAGE:-}" "")" # Deployment outputs without an AZURE_-prefixed duplicate (see # infra/main.bicep): read straight from their own azd output name. They # are stored under LAB_-prefixed names because `load_lab_config` makes @@ -108,6 +114,7 @@ load_lab_config() { readonly AZURE_CONTAINER_APP_NAME AZURE_CONTAINER_APP_FQDN AZURE_STORAGE_CONTAINER_SCOPE readonly AZURE_BLOB_ROLE_ASSIGNMENT_NAME AZURE_WORKSPACE_ID AZURE_APP_INSIGHTS_NAME readonly AZURE_TELEMETRY_SERVICE_NAME LAB_CONTAINER_APP_PRINCIPAL_ID LAB_WORKSPACE_CUSTOMER_ID + readonly SRE_CONTAINER_IMAGE # Azure SRE Agent settings (.env.example documents these). None of the # current scripts read them yet, but they resolve through the same diff --git a/monitor/sre-agent-event-lab/scripts/doctor.sh b/monitor/sre-agent-event-lab/scripts/doctor.sh index 953d8fa..ea52a46 100755 --- a/monitor/sre-agent-event-lab/scripts/doctor.sh +++ b/monitor/sre-agent-event-lab/scripts/doctor.sh @@ -187,6 +187,16 @@ if [[ "${AZURE_SAFE}" -eq 1 && -n "${APP_FQDN}" ]]; then http_status="$(curl --max-time 10 --silent --output /dev/null --write-out '%{http_code}' "https://${APP_FQDN}/healthz" 2>/dev/null || echo 000)" if [[ "${http_status}" == "200" ]]; then report "Health endpoint" PASS "https://${APP_FQDN}/healthz returned HTTP 200." + elif [[ -z "${SRE_CONTAINER_IMAGE}" ]]; then + # SRE_CONTAINER_IMAGE is only set once the deploy phase (`postdeploy` + # hook, scripts/azd-deploy-app.sh) has built the lab image and switched + # the Container App onto it. Empty here means `azd provision` ran but + # `azd deploy` has not (yet): the app is still the public placeholder + # image (port 80, no /healthz -- see infra/main.bicep), which is the + # documented, expected intermediate state, not a broken deployment. The + # generic "investigate with curl -v" wording below would misclassify + # that state, so this branch names the actual remedy instead. + report "Health endpoint" FAIL "https://${APP_FQDN}/healthz returned HTTP ${http_status}, and SRE_CONTAINER_IMAGE is not recorded -- the deploy phase has not run yet (this is the expected placeholder state after 'azd provision' alone, not a broken deployment). Run: azd deploy --no-prompt" else report "Health endpoint" FAIL "https://${APP_FQDN}/healthz returned HTTP ${http_status}. Investigate: curl -v https://${APP_FQDN}/healthz" fi diff --git a/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py b/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py index 32f3955..2f5a1dc 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/deploy_app_harness.py @@ -20,6 +20,13 @@ projection. An assignment at the resource group, or a different role at the registry, is therefore invisible to the hook exactly as it would be against a real subscription. +* `az role assignment list` also rejects any flag its real parser does not + recognise -- `--assignee-principal-type` included, an option that exists + only on `az role assignment create` (verified against the installed + Azure CLI 2.89.1: passing it to `list` fails with `ERROR: unrecognized + arguments`, exit 2) -- so a hook that regresses to sending it fails the + poll here exactly as it would against a real subscription, instead of + the fake silently ignoring a flag it does not happen to read. * `available_after_attempts` models RBAC propagation: the assignment exists but the first N `role assignment list` calls do not return it yet, which is the delay the whole poll exists for. @@ -164,6 +171,38 @@ def main(): return if argv[:3] == ["role", "assignment", "list"]: + # Reject anything the real `az role assignment list` parser would + # reject (verified against the installed Azure CLI 2.89.1), so a + # hook that regresses to an unsupported flag -- `--assignee- + # principal-type` included, which exists only on `role assignment + # create` -- fails here exactly as it would against a real + # subscription, instead of silently succeeding against a fake that + # only reads the flags it happens to recognise. + value_flags = {{ + "--assignee", "--assignee-object-id", "--resource-group", "-g", + "--role", "--scope", "--subscription", "--query", "--output", "-o", + "--fill-principal-name", "--fill-role-definition-name", + }} + flag_only_flags = {{ + "--all", "--include-groups", "--include-inherited", + "--debug", "--only-show-errors", "--verbose", + }} + rest = argv[3:] + unrecognized = [] + index = 0 + while index < len(rest): + token = rest[index] + if token in value_flags: + index += 2 + continue + if token in flag_only_flags: + index += 1 + continue + unrecognized.append(token) + index += 1 + if unrecognized: + fail(f"ERROR: unrecognized arguments: {{' '.join(unrecognized)}}") + state["role_list_attempts"] = state.get("role_list_attempts", 0) + 1 attempt = state["role_list_attempts"] save(state) diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py b/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py index f13d904..f79dcab 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_azd_deploy_app.py @@ -158,6 +158,32 @@ def test_never_builds_the_image_locally(): assert "docker build" not in text +def test_role_poll_has_no_unsupported_flags_and_no_graph_dependency(): + """`az role assignment list` has no `--assignee-principal-type` option + (verified against the installed Azure CLI 2.89.1: passing it fails with + `ERROR: unrecognized arguments: --assignee-principal-type ...`, exit + 2) -- that flag only exists on `az role assignment create`. The poll + must not send it, and must instead pass `--fill-principal-name false + --fill-role-definition-name false` so it never depends on Microsoft + Graph: `--assignee-object-id` already bypasses Graph for the *filter*, + but `--fill-principal-name` defaults to `true` and queries Graph anyway + to populate a field this hook never reads.""" + commands = _az_invocations(DEPLOY_APP.read_text()) + poll_calls = [c for c in commands if c.startswith("az role assignment list")] + + assert poll_calls, "the hook never calls az role assignment list" + for call in poll_calls: + assert "--assignee-principal-type" not in call, ( + f"az role assignment list does not support --assignee-principal-type: {call}" + ) + assert "--fill-principal-name false" in call, ( + f"the poll must set --fill-principal-name false to avoid Graph: {call}" + ) + assert "--fill-role-definition-name false" in call, ( + f"the poll must set --fill-role-definition-name false to avoid Graph: {call}" + ) + + # --- Behaviour: the gate --------------------------------------------------- @@ -221,6 +247,21 @@ def test_stops_without_building_when_acr_pull_never_appears(tmp_path): assert "SRE_CONTAINER_IMAGE" not in run.azd_calls +def test_polls_successfully_without_any_unsupported_flag_or_graph_call(tmp_path): + """Behavioural counterpart to the static flag check: the fake `az` here + models the real CLI's `role assignment list` parser (see + `deploy_app_harness`) and fails closed on any flag that parser does not + recognise, `--assignee-principal-type` included. A hook that regresses + back to sending it would make every poll fail with `ERROR: fake az: + unrecognized arguments: --assignee-principal-type ...` and the whole + run would abort without ever building the image.""" + run = run_deploy_app(tmp_path) + + assert run.returncode == 0, run.stdout + run.stderr + assert "unrecognized arguments" not in run.stderr + assert run.first_index("acr build") is not None + + def test_ignores_an_acr_pull_grant_at_a_wider_scope(tmp_path): """A resource-group-scoped AcrPull would let the app pull, but it is not the assignment this lab creates; accepting it would make the gate pass diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py index 267d539..3ae38ee 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_doctor.py @@ -123,6 +123,45 @@ def test_doctor_fails_when_healthz_does_not_return_200(fake_az): assert "503" in result.stdout +def test_doctor_explains_the_pre_deploy_placeholder_state_instead_of_misclassifying_it(fake_az): + """`azd provision` alone leaves the public placeholder image running + (port 80, no `/healthz`) until `azd deploy` builds and switches to the + lab image (see infra/main.bicep, scripts/azd-deploy-app.sh). Without + SRE_CONTAINER_IMAGE recorded, a failing `/healthz` is that documented, + expected intermediate state -- not a broken deployment -- so doctor + must name the exact remedy (`azd deploy --no-prompt`) instead of the + generic "investigate with curl -v" wording that implies something is + actually broken.""" + fake_az.healthz_status = 404 + + result = run_doctor(fake_az) + + assert result.returncode == 1 + assert "Health endpoint\tFAIL" in result.stdout + detail = _detail_of(result, "Health endpoint") + assert "azd deploy --no-prompt" in detail + assert "curl -v" not in detail + + +def test_doctor_reports_a_genuine_health_failure_once_the_lab_image_is_deployed(fake_az): + """Once SRE_CONTAINER_IMAGE is recorded, the deploy phase already + replaced the placeholder with the lab image and its /healthz probes; + a failure at that point is a real regression and must not be softened + into the pre-deploy placeholder message.""" + fake_az.healthz_status = 503 + + result = run_doctor( + fake_az, + sre_container_image="acrsrelabtest.azurecr.io/sre-event-lab:abc123", + ) + + assert result.returncode == 1 + assert "Health endpoint\tFAIL" in result.stdout + detail = _detail_of(result, "Health endpoint") + assert "azd deploy --no-prompt" not in detail + assert "curl -v" in detail + + def test_doctor_fails_when_venv_is_missing(fake_az): """Finding #1: doctor must report on the venv `setup-venv.sh` (run from `postprovision`) is responsible for creating, with a remedy pointing at From bf992efd81ed0f0779b34e2d7a50d33b3fbc740e Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 14 Aug 2026 23:40:16 +0900 Subject: [PATCH 18/26] fix(sre-lab): switch alert evaluationFrequency from PT1M to PT5M Live `azd provision` failed ARM validation for all three Microsoft.Insights/scheduledQueryRules@2023-12-01 alert rules with `QueryNotContainKnownTable: One-minute frequency is not supported for this query. Either switch to five-minute frequency or adapt the query.` Root cause: infra/alerts.bicep's requests/dependencies Application Insights queries were paired with evaluationFrequency: 'PT1M', a cadence those queries do not support (five minutes or coarser only). `az bicep build` never calls ARM, so this was never caught before a real deployment attempt. TDD: - RED: added test_evaluation_frequency_is_five_minutes_not_one_minute to infra/tests/test_alerts_bicep.py, plus two doc-contract tests in scripts/tests/test_lab_guides.py asserting README.md and dynamic-thresholds.md no longer claim one-minute static evaluation. 3 failed against the unmodified template/docs. - GREEN: changed evaluationFrequency to 'PT5M' for all three rules in alerts.bicep (windowSize stays 'PT5M', thresholds unchanged -- neither is implicated by this failure); updated README.md's cost callout and dynamic-thresholds.md's Static Threshold section to say 5 minutes. validation-results.md is left untouched: it is a historical record of a past run, not current design guidance. Verified: full suite 463 passed (app/tests, infra/tests, scripts/tests); `az bicep build` passes for main/lab/workload/alerts.bicep, and the compiled alerts.bicep ARM JSON now shows "evaluationFrequency": "PT5M". No Azure resources were deployed or deleted. .azure/deployment-plan.md keeps Status: Ready for Validation and now records this failure and fix, since the live provision attempt that surfaced it did not complete successfully. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 40 +++++++++++++++++-- monitor/sre-agent-event-lab/README.md | 2 +- .../sre-agent-event-lab/dynamic-thresholds.md | 2 +- .../sre-agent-event-lab/infra/alerts.bicep | 2 +- .../infra/tests/test_alerts_bicep.py | 15 +++++++ .../scripts/tests/test_lab_guides.py | 29 ++++++++++++++ 6 files changed, 83 insertions(+), 7 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 6ed7f65..7493483 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -2,7 +2,7 @@ > **Status:** Ready for Validation -Updated: 2026-08-14 (ACR gate review fixes) +Updated: 2026-08-14 (alert evaluation frequency fix) ## Goal @@ -139,6 +139,38 @@ a `/healthz` failure is reported as a real regression as before. Stale `postprovision` (moved to the `postdeploy` hook in the two-phase refactor above) were also corrected. +### Live deployment failure: alert evaluation frequency (2026-08-14) + +A live `azd provision` attempt failed ARM validation for all three +`Microsoft.Insights/scheduledQueryRules@2023-12-01` alert rules +(`alert-sre-lab-s1-http500`, `-s2-latency`, `-s3-storage-rbac`) with: + +``` +QueryNotContainKnownTable: One-minute frequency is not supported for +this query. Either switch to five-minute frequency or adapt the query. +``` + +Root cause: `infra/alerts.bicep` set `evaluationFrequency: 'PT1M'` for +all three rules while their `requests`/`dependencies` Application +Insights queries only support a five-minute (or coarser) evaluation +cadence -- the one-minute cadence was never deployable, only ever +validated by `az bicep build`, which does not call ARM and cannot catch +this. Fixed with strict TDD: added +`test_evaluation_frequency_is_five_minutes_not_one_minute` to +`infra/tests/test_alerts_bicep.py` (RED against the unmodified +template), then changed `evaluationFrequency` to `'PT5M'` for all three +rules (GREEN). `windowSize` stays `'PT5M'` and per-rule thresholds are +unchanged, since nothing about the failure implicated them. Two +user-facing docs asserted the now-incorrect one-minute cadence and were +corrected under the same RED/GREEN discipline (new tests in +`scripts/tests/test_lab_guides.py`): `README.md`'s cost callout ("1분 +주기 로그 검색 경고 규칙 3개" → "5분 주기 로그 검색 경고 규칙 3개") and +`dynamic-thresholds.md`'s Static Threshold section ("evaluation: 1분" → +"evaluation: 5분"); the unrelated, still-true statement that Log Search +*dynamic* thresholds do not support one-minute evaluation was left +as-is. No Azure resources were deployed or deleted while diagnosing or +fixing this. + ## Security and Safety - Container Apps reach Blob Storage through private networking. @@ -195,10 +227,10 @@ earlier "Validated" status no longer applies): | Check | Command | Result | |---|---|---| -| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 460 passed | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 463 passed (added 3 for the PT5M alert-frequency fix) | | Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | | Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | -| Bicep build | `az bicep build --file infra/{main,lab,workload}.bicep --stdout` | Passed (three templates, with the new deploy-gate outputs) | +| Bicep build | `az bicep build --file infra/{main,lab,workload,alerts}.bicep --stdout` | Passed (four templates; `alerts.bicep` now emits `evaluationFrequency: PT5M`) | | AZD schema | `azure.yaml` validated against `schemas/v1.0/azure.yaml.json` from Azure/azure-dev | Passed; hooks = preprovision, postprovision, postdeploy, predown, postdown; no `services` | | AZD package | `azd package --all --no-prompt` | Passed | | Zero-service deploy hook | `azd deploy --no-prompt` against a marker-hook copy of this `azure.yaml` | `postdeploy` ran; hook exit 7 failed the command | @@ -210,7 +242,7 @@ Pending live validation (no resources were deployed by this change): |---|---|---| | Environment | `azd env new --location koreacentral` | To re-create for the live run | | Provision preview | `azd provision --preview --no-prompt` | To re-run | -| Provision phase | `azd provision --no-prompt` leaves the placeholder image serving and `app/.venv` ready | To verify live | +| Provision phase | `azd provision --no-prompt` leaves the placeholder image serving and `app/.venv` ready | Failed live at ARM validation for the three `scheduledQueryRules` alert rules (`QueryNotContainKnownTable`, PT1M unsupported) before this fix; to re-verify live now that `alerts.bicep` uses PT5M | | Deploy phase | `azd deploy --no-prompt` waits for `AcrPull`, builds in ACR, switches the image, `/healthz` returns 200 | To verify live | | Policy assignments | `az policy assignment list --scope --disable-scope-strict-match` | Unchanged from the previous run; re-check at validation time | | Static RBAC | reviewed all `Microsoft.Authorization/roleAssignments` in `workload.bicep` | Unchanged: least-privilege AcrPull and container-scoped Blob Data Reader | diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index f66e9d7..5178ab1 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -20,7 +20,7 @@ Azure Container Apps에 장애를 세 번 주입하고, Azure Monitor 경고를 - Container Registry Basic - Log Analytics 작업 영역(PerGB2018, 30일 보존)과 Application Insights 수집량 - Storage 계정(Standard_LRS) -- 1분 주기 로그 검색 경고 규칙 3개(평가 주기가 짧을수록 규칙당 단가가 올라갑니다) +- 5분 주기 로그 검색 경고 규칙 3개(평가 주기가 짧을수록 규칙당 단가가 올라갑니다) Azure SRE Agent는 이 실습이 만들지 않습니다. 미리 만들어 둔 Agent를 사용하며 [별도로 과금](https://azure.microsoft.com/pricing/details/sre-agent/)됩니다. diff --git a/monitor/sre-agent-event-lab/dynamic-thresholds.md b/monitor/sre-agent-event-lab/dynamic-thresholds.md index 83772a2..81f9026 100644 --- a/monitor/sre-agent-event-lab/dynamic-thresholds.md +++ b/monitor/sre-agent-event-lab/dynamic-thresholds.md @@ -13,7 +13,7 @@ Dynamic Thresholds 자체는 이번 결과에 포함된 실증 대상이 아니 ### Static Threshold - 실험 목표: 정해진 장애가 짧은 시간 안에 반드시 alert를 발생시키는지 검증 -- evaluation: 1분 +- evaluation: 5분 - 조건: 5xx count, p95 > 2000ms, Blob 403 count - 장점: 결정론적이고 당일 재현 가능 - 한계: workload별 정상 범위와 계절성을 수동으로 관리 diff --git a/monitor/sre-agent-event-lab/infra/alerts.bicep b/monitor/sre-agent-event-lab/infra/alerts.bicep index 9882e67..c0736bf 100644 --- a/monitor/sre-agent-event-lab/infra/alerts.bicep +++ b/monitor/sre-agent-event-lab/infra/alerts.bicep @@ -92,7 +92,7 @@ resource alertRules 'Microsoft.Insights/scheduledQueryRules@2023-12-01' = [ description: definition.description displayName: definition.displayName enabled: true - evaluationFrequency: 'PT1M' + evaluationFrequency: 'PT5M' scopes: [ appInsightsResourceId ] diff --git a/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py b/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py index eaff22e..0517b14 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py @@ -43,6 +43,21 @@ def test_http500_alert_isolated_to_orders_and_exact_500(): assert "''', serviceName)" in template +def test_evaluation_frequency_is_five_minutes_not_one_minute(): + """Azure deployment of scheduledQueryRules@2023-12-01 rejects a + one-minute evaluationFrequency for these requests/dependencies + queries with QueryNotContainKnownTable: 'One-minute frequency is + not supported for this query. Either switch to five-minute + frequency or adapt the query.' All three alert rules must evaluate + on a five-minute cadence instead, matching the existing PT5M + windowSize. + """ + template = ALERTS_BICEP.read_text() + + assert "evaluationFrequency: 'PT5M'" in template + assert "evaluationFrequency: 'PT1M'" not in template + + def test_alert_rules_require_and_pass_through_caller_tags(): """azure.yaml's azd main.bicep now centralizes tag construction (purpose, azd-env-name, expiresOn) and passes the merged object down diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index b040d74..005b930 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -19,6 +19,7 @@ RUNBOOK = LAB_ROOT / "runbooks" / "incident-response.md" LAB_SH = LAB_ROOT / "scripts" / "lab.sh" VALIDATION_RESULTS = LAB_ROOT / "validation-results.md" +DYNAMIC_THRESHOLDS = LAB_ROOT / "dynamic-thresholds.md" GUIDE_NAMES = ( "01-agent-setup.md", @@ -198,6 +199,34 @@ def test_readme_warns_about_cost_and_teardown_before_the_first_azure_command(): assert driver in text, driver +def test_readme_states_five_minute_alert_evaluation_not_one_minute(): + """scheduledQueryRules@2023-12-01 rejects a one-minute evaluation + frequency for these requests/dependencies queries + (QueryNotContainKnownTable), so infra/alerts.bicep evaluates every + five minutes. The cost callout naming the log search alert rules + must state that same cadence. + """ + text = README.read_text() + + assert "5분 주기 로그 검색 경고 규칙 3개" in text + assert "1분 주기 로그 검색 경고 규칙" not in text + + +def test_dynamic_thresholds_guide_states_five_minute_static_evaluation(): + """The Static Threshold section of dynamic-thresholds.md describes + the deployed infra/alerts.bicep rules, which now evaluate every + five minutes (scheduledQueryRules@2023-12-01 rejects a one-minute + frequency for these queries). Only the unrelated fact that Log + Search dynamic thresholds themselves do not support one-minute + evaluation may still mention one minute. + """ + text = DYNAMIC_THRESHOLDS.read_text() + + assert "- evaluation: 5분" in text + assert "- evaluation: 1분" not in text + assert "Log Search dynamic threshold는 1분 evaluation을 지원하지 않는다." in text + + def test_readme_is_a_quickstart_not_the_full_walkthrough(): """Scenario, capture and scoring detail belongs in the guides.""" text = README.read_text() From ec33c96b6468fbe607e450600a4147d1db204676 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 00:11:38 +0900 Subject: [PATCH 19/26] fix(sre-lab): scope alerts to the workspace schema and restore PT1M MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The previous commit read `QueryNotContainKnownTable: One-minute frequency is not supported for this query.` literally and moved all three `Microsoft.Insights/scheduledQueryRules@2023-12-01` rules to PT5M. That was the wrong root cause and is reverted here. Real root cause: the rules were scoped to the Application Insights component and queried the legacy resource-centric schema (`requests`, `dependencies`, `timestamp`, `cloud_RoleName`, `duration`). On a workspace-based component those legacy names are functions over the workspace tables, and "the query calls a function that calls other tables" is one of the documented one-minute-frequency limitations (aka.ms/lsa_1m_limits -> alerts-create-log-alert-rule#frequency), which is exactly what "does not contain a known table" reports. Fix (TDD, RED first -- 11 failing tests): - infra/alerts.bicep takes `workspaceResourceId`, queries the workspace known tables `AppRequests`/`AppDependencies` with the exact casing already used by scripts/query-evidence.sh (TimeGenerated, AppRoleName, Name, ResultCode, DurationMs, Target; percentile(DurationMs, 95) needs no timespan conversion), scopes each rule to the workspace, sets `targetResourceTypes: ['Microsoft.OperationalInsights/workspaces']`, and restores `evaluationFrequency: 'PT1M'` with `windowSize: 'PT5M'`. - infra/lab.bicep passes `observability.outputs.workspaceId` to the alerts module; the workspaceId output chain (observability -> lab -> main -> AZURE_WORKSPACE_ID) and `appInsightsResourceId` are unchanged. - Thresholds and the default fire/resolve timeouts (720s/900s) are untouched: the one-minute cadence they were sized for is back. - README.md is back to "1분 주기 로그 검색 경고 규칙 3개", dynamic-thresholds.md to "evaluation: 1분(window 5분)" plus the workspace-scope precondition, and validation-results.md now annotates why its recorded one-minute run stays plausible. Live evidence (nothing deployed, modified or deleted): - `az deployment group validate` against the partially provisioned lab resource group: `provisioningState: Succeeded`, `error: null`, the three scheduledQueryRules validated. - Honest limit, measured: the pre-fix template (component scope, legacy tables, PT1M) also passes preflight, so validate proves the template shape, not the query. The query side was proven directly with `az monitor log-analytics query` on the live workspace: all three alert queries resolve (Failures=0, P95DurationMs=None, DependencyFailures=0 with no traffic yet), while `requests | take 1` fails with SEM0100 "Failed to resolve table or column expression named 'requests'". Verified: 468 passed (app/tests, infra/tests, scripts/tests); `az bicep build` passes for main/lab/workload/observability/alerts.bicep and the compiled alerts.bicep emits PT1M/PT5M with the workspace scope and target type. .azure/deployment-plan.md stays "Ready for Validation" because the definitive proof is the still-pending live `azd provision`. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 118 +++++++++++---- monitor/sre-agent-event-lab/README.md | 2 +- .../sre-agent-event-lab/dynamic-thresholds.md | 3 +- .../sre-agent-event-lab/infra/alerts.bicep | 44 +++--- monitor/sre-agent-event-lab/infra/lab.bicep | 2 +- .../infra/observability.bicep | 3 + .../infra/tests/test_alerts_bicep.py | 137 +++++++++++++++--- .../scripts/tests/test_lab_guides.py | 47 +++--- .../sre-agent-event-lab/validation-results.md | 2 + 9 files changed, 270 insertions(+), 88 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 7493483..0737748 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -2,7 +2,7 @@ > **Status:** Ready for Validation -Updated: 2026-08-14 (alert evaluation frequency fix) +Updated: 2026-08-14 (alert query schema fix; PT1M restored) ## Goal @@ -139,9 +139,9 @@ a `/healthz` failure is reported as a real regression as before. Stale `postprovision` (moved to the `postdeploy` hook in the two-phase refactor above) were also corrected. -### Live deployment failure: alert evaluation frequency (2026-08-14) +### Live deployment failure: alert query schema, not cadence (2026-08-14) -A live `azd provision` attempt failed ARM validation for all three +A live `azd provision` attempt failed for all three `Microsoft.Insights/scheduledQueryRules@2023-12-01` alert rules (`alert-sre-lab-s1-http500`, `-s2-latency`, `-s3-storage-rbac`) with: @@ -150,26 +150,90 @@ QueryNotContainKnownTable: One-minute frequency is not supported for this query. Either switch to five-minute frequency or adapt the query. ``` -Root cause: `infra/alerts.bicep` set `evaluationFrequency: 'PT1M'` for -all three rules while their `requests`/`dependencies` Application -Insights queries only support a five-minute (or coarser) evaluation -cadence -- the one-minute cadence was never deployable, only ever -validated by `az bicep build`, which does not call ARM and cannot catch -this. Fixed with strict TDD: added -`test_evaluation_frequency_is_five_minutes_not_one_minute` to -`infra/tests/test_alerts_bicep.py` (RED against the unmodified -template), then changed `evaluationFrequency` to `'PT5M'` for all three -rules (GREEN). `windowSize` stays `'PT5M'` and per-rule thresholds are -unchanged, since nothing about the failure implicated them. Two -user-facing docs asserted the now-incorrect one-minute cadence and were -corrected under the same RED/GREEN discipline (new tests in -`scripts/tests/test_lab_guides.py`): `README.md`'s cost callout ("1분 -주기 로그 검색 경고 규칙 3개" → "5분 주기 로그 검색 경고 규칙 3개") and -`dynamic-thresholds.md`'s Static Threshold section ("evaluation: 1분" → -"evaluation: 5분"); the unrelated, still-true statement that Log Search -*dynamic* thresholds do not support one-minute evaluation was left -as-is. No Azure resources were deployed or deleted while diagnosing or -fixing this. +The first fix read that message literally and moved every rule to +`evaluationFrequency: 'PT5M'`. That was the wrong root cause, and it has +been reverted. + +Real root cause: `infra/alerts.bicep` scoped the rules to the Application +Insights **component** and queried the legacy resource-centric schema +(`requests`, `dependencies`, `timestamp`, `cloud_RoleName`, `duration`). +In a workspace-based Application Insights resource those legacy names are +not tables -- they are functions over the workspace tables. The official +one-minute-frequency limitations list ("the query calls a function that +calls other tables", plus `search`/`union`/`take`, `ingestion_time()` and +the `adx` pattern) is exactly what `QueryNotContainKnownTable` reports: +the query contains no *known table*, so the one-minute optimization +cannot be applied. Source: [Create a log search alert +rule](https://learn.microsoft.com/azure/azure-monitor/alerts/alerts-create-log-alert-rule#configure-alert-rule-conditions) +(reached from `aka.ms/lsa_1m_limits`, the link the troubleshooting page +gives for this error). + +Final fix (strict TDD, RED first): `infra/alerts.bicep` now takes a +`workspaceResourceId`, queries the workspace schema known tables +`AppRequests`/`AppDependencies` with the exact column casing already used +by `scripts/query-evidence.sh` (`TimeGenerated`, `AppRoleName`, `Name`, +`ResultCode`, `DurationMs`, `Target`; `percentile(DurationMs, 95)` needs +no timespan conversion), scopes each rule to the Log Analytics workspace, +declares `targetResourceTypes: ['Microsoft.OperationalInsights/workspaces']`, +and restores `evaluationFrequency: 'PT1M'` with `windowSize: 'PT5M'`. +`lab.bicep` passes `observability.outputs.workspaceId` into the alerts +module; the `workspaceId` output chain (observability -> lab -> main -> +`AZURE_WORKSPACE_ID`) is unchanged, and `appInsightsResourceId` is still +exported for the scripts that read it. Per-rule thresholds and the +default fire/resolve timeouts (`LAB_ALERT_FIRE_TIMEOUT_SECONDS=720`, +`LAB_ALERT_RESOLVE_TIMEOUT_SECONDS=900`) are unchanged: the one-minute +cadence they were sized for is back. + +Consequence for the recorded results: the S1/S2/S3 run in +`monitor/sre-agent-event-lab/validation-results.md` was executed at a +one-minute cadence and stays plausible, because the failure above was the +legacy schema on the component scope, not the cadence. Both that report +and `dynamic-thresholds.md` now carry that annotation, and `README.md`'s +cost callout is back to "1분 주기 로그 검색 경고 규칙 3개". + +#### Live validation proof (2026-08-14, no resources deployed) + +Run against the partially provisioned lab resource group (only the +observability resources exist in it), with placeholders for the real +subscription ID and resource group: + +``` +az deployment group validate \ + --resource-group \ + --name sre-lab-alerts-validate-workspace \ + --template-file infra/alerts.bicep \ + --parameters location=koreacentral \ + workspaceResourceId=/subscriptions//resourceGroups//providers/Microsoft.OperationalInsights/workspaces/law-sre-event-lab- \ + serviceName=sre-event-lab- \ + tags='{"purpose":"sre-agent-event-lab"}' +``` + +Result: `"provisioningState": "Succeeded"`, `"error": null`, and +`validatedResources` listing exactly the three +`Microsoft.Insights/scheduledQueryRules` IDs +(`alert-sre-lab-s1-http500`, `alert-sre-lab-s2-latency`, +`alert-sre-lab-s3-storage-rbac`) -- i.e. the workspace-scoped, +`AppRequests`/`AppDependencies`, PT1M/PT5M template is accepted. + +Honest limit of that proof, measured on the same resource group: ARM +preflight does **not** run the scheduled-query-rule query validation. The +pre-fix template (component scope, `requests`/`dependencies`, PT1M) was +re-validated on purpose and *also* returned `provisioningState: +Succeeded`, even though the same template failed the real deployment with +`QueryNotContainKnownTable`. So `az deployment group validate` proves the +template shape, parameters and RBAC are deployable, not that the query is +accepted at PUT time. + +The query side was therefore proven directly against the live workspace +with `az monitor log-analytics query`: + +- `AppRequests | where TimeGenerated > ago(5m) | where AppRoleName == "" | where Name has "/api/orders" | where ResultCode == "500" | summarize Failures=count()` returns `Failures = 0` (no traffic yet -- the Container App still runs the placeholder image), so the table and every column name/casing resolve. +- The S2 (`percentile(DurationMs, 95)`) and S3 (`AppDependencies` ... `Target`, `ResultCode`) queries resolve the same way. +- The legacy name fails in workspace scope: `requests | take 1` returns `SEM0100: 'take' operator: Failed to resolve table or column expression named 'requests'`, confirming it is not a known table there. + +Definitive confirmation still requires the pending live `azd provision`, +which is why this plan stays **Ready for Validation**. No resources were +deployed, modified or deleted while diagnosing or fixing this. ## Security and Safety @@ -227,14 +291,16 @@ earlier "Validated" status no longer applies): | Check | Command | Result | |---|---|---| -| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 463 passed (added 3 for the PT5M alert-frequency fix) | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 468 passed (11 RED first for the workspace-schema alert fix) | | Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | | Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | -| Bicep build | `az bicep build --file infra/{main,lab,workload,alerts}.bicep --stdout` | Passed (four templates; `alerts.bicep` now emits `evaluationFrequency: PT5M`) | +| Bicep build | `az bicep build --file infra/{main,lab,workload,observability,alerts}.bicep --stdout` | Passed (five templates; `alerts.bicep` emits `evaluationFrequency: PT1M`, `windowSize: PT5M`, workspace scope and `targetResourceTypes: Microsoft.OperationalInsights/workspaces`) | | AZD schema | `azure.yaml` validated against `schemas/v1.0/azure.yaml.json` from Azure/azure-dev | Passed; hooks = preprovision, postprovision, postdeploy, predown, postdown; no `services` | | AZD package | `azd package --all --no-prompt` | Passed | | Zero-service deploy hook | `azd deploy --no-prompt` against a marker-hook copy of this `azure.yaml` | `postdeploy` ran; hook exit 7 failed the command | | Authentication | `az account show`; `azd auth login --check-status --output json` | Authenticated | +| Alert template preflight | `az deployment group validate --template-file infra/alerts.bicep ...` against the live lab resource group | `provisioningState: Succeeded`, `error: null`, three `scheduledQueryRules` validated (see the proof note above for what preflight does and does not cover) | +| Alert queries | `az monitor log-analytics query` for the three rule queries against the live workspace | All three resolve (`Failures=0`, `P95DurationMs=None`, `DependencyFailures=0` with no traffic yet); legacy `requests` fails with `SEM0100` | Pending live validation (no resources were deployed by this change): @@ -242,7 +308,7 @@ Pending live validation (no resources were deployed by this change): |---|---|---| | Environment | `azd env new --location koreacentral` | To re-create for the live run | | Provision preview | `azd provision --preview --no-prompt` | To re-run | -| Provision phase | `azd provision --no-prompt` leaves the placeholder image serving and `app/.venv` ready | Failed live at ARM validation for the three `scheduledQueryRules` alert rules (`QueryNotContainKnownTable`, PT1M unsupported) before this fix; to re-verify live now that `alerts.bicep` uses PT5M | +| Provision phase | `azd provision --no-prompt` leaves the placeholder image serving and `app/.venv` ready | Failed live for the three `scheduledQueryRules` alert rules (`QueryNotContainKnownTable`) while they queried the legacy Application Insights schema on the component scope; to re-verify live now that they query the workspace schema on the workspace scope at PT1M | | Deploy phase | `azd deploy --no-prompt` waits for `AcrPull`, builds in ACR, switches the image, `/healthz` returns 200 | To verify live | | Policy assignments | `az policy assignment list --scope --disable-scope-strict-match` | Unchanged from the previous run; re-check at validation time | | Static RBAC | reviewed all `Microsoft.Authorization/roleAssignments` in `workload.bicep` | Unchanged: least-privilege AcrPull and container-scoped Blob Data Reader | diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index 5178ab1..f66e9d7 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -20,7 +20,7 @@ Azure Container Apps에 장애를 세 번 주입하고, Azure Monitor 경고를 - Container Registry Basic - Log Analytics 작업 영역(PerGB2018, 30일 보존)과 Application Insights 수집량 - Storage 계정(Standard_LRS) -- 5분 주기 로그 검색 경고 규칙 3개(평가 주기가 짧을수록 규칙당 단가가 올라갑니다) +- 1분 주기 로그 검색 경고 규칙 3개(평가 주기가 짧을수록 규칙당 단가가 올라갑니다) Azure SRE Agent는 이 실습이 만들지 않습니다. 미리 만들어 둔 Agent를 사용하며 [별도로 과금](https://azure.microsoft.com/pricing/details/sre-agent/)됩니다. diff --git a/monitor/sre-agent-event-lab/dynamic-thresholds.md b/monitor/sre-agent-event-lab/dynamic-thresholds.md index 81f9026..205fbad 100644 --- a/monitor/sre-agent-event-lab/dynamic-thresholds.md +++ b/monitor/sre-agent-event-lab/dynamic-thresholds.md @@ -13,8 +13,9 @@ Dynamic Thresholds 자체는 이번 결과에 포함된 실증 대상이 아니 ### Static Threshold - 실험 목표: 정해진 장애가 짧은 시간 안에 반드시 alert를 발생시키는지 검증 -- evaluation: 5분 +- evaluation: 1분(window 5분) - 조건: 5xx count, p95 > 2000ms, Blob 403 count +- 전제: rule scope는 Log Analytics workspace이고 query는 workspace schema known table(`AppRequests`, `AppDependencies`)을 사용한다. Application Insights component scope의 legacy `requests`/`dependencies`는 다른 table을 호출하는 function이라 1분 주기에서 `QueryNotContainKnownTable`로 거부된다. - 장점: 결정론적이고 당일 재현 가능 - 한계: workload별 정상 범위와 계절성을 수동으로 관리 diff --git a/monitor/sre-agent-event-lab/infra/alerts.bicep b/monitor/sre-agent-event-lab/infra/alerts.bicep index c0736bf..2dd1722 100644 --- a/monitor/sre-agent-event-lab/infra/alerts.bicep +++ b/monitor/sre-agent-event-lab/infra/alerts.bicep @@ -1,8 +1,12 @@ @description('Azure region for scheduled query alert rules.') param location string -@description('Resource ID of the workspace-based Application Insights component.') -param appInsightsResourceId string +@description('''Resource ID of the Log Analytics workspace backing Application Insights. +The alert queries read the workspace-schema tables (AppRequests, +AppDependencies), which are known tables only when the rule is scoped to the +workspace itself, so the same resource ID is both the query source and the +rule scope.''') +param workspaceResourceId string @description('Optional Action Group that forwards fired alerts to Azure SRE Agent.') param actionGroupResourceId string = '' @@ -21,11 +25,11 @@ var alertDefinitions = [ measureColumn: 'Failures' threshold: 10 query: format(''' -requests -| where timestamp > ago(5m) -| where cloud_RoleName == "{0}" -| where name has "/api/orders" -| where resultCode == "500" +AppRequests +| where TimeGenerated > ago(5m) +| where AppRoleName == "{0}" +| where Name has "/api/orders" +| where ResultCode == "500" | summarize Failures=count() ''', serviceName) } @@ -36,11 +40,11 @@ requests measureColumn: 'P95DurationMs' threshold: 2000 query: format(''' -requests -| where timestamp > ago(5m) -| where cloud_RoleName == "{0}" -| where name has "/api/orders" -| summarize P95DurationMs=percentile(duration, 95) +AppRequests +| where TimeGenerated > ago(5m) +| where AppRoleName == "{0}" +| where Name has "/api/orders" +| summarize P95DurationMs=percentile(DurationMs, 95) ''', serviceName) } { @@ -50,11 +54,11 @@ requests measureColumn: 'DependencyFailures' threshold: 5 query: format(''' -dependencies -| where timestamp > ago(5m) -| where cloud_RoleName == "{0}" -| where target has "{1}" -| where resultCode == "403" +AppDependencies +| where TimeGenerated > ago(5m) +| where AppRoleName == "{0}" +| where Target has "{1}" +| where ResultCode == "403" | summarize DependencyFailures=count() ''', serviceName, environment().suffixes.storage) } @@ -92,14 +96,14 @@ resource alertRules 'Microsoft.Insights/scheduledQueryRules@2023-12-01' = [ description: definition.description displayName: definition.displayName enabled: true - evaluationFrequency: 'PT5M' + evaluationFrequency: 'PT1M' scopes: [ - appInsightsResourceId + workspaceResourceId ] severity: 2 skipQueryValidation: false targetResourceTypes: [ - 'Microsoft.Insights/components' + 'Microsoft.OperationalInsights/workspaces' ] windowSize: 'PT5M' } diff --git a/monitor/sre-agent-event-lab/infra/lab.bicep b/monitor/sre-agent-event-lab/infra/lab.bicep index 59d13cc..63a5f4d 100644 --- a/monitor/sre-agent-event-lab/infra/lab.bicep +++ b/monitor/sre-agent-event-lab/infra/lab.bicep @@ -55,7 +55,7 @@ module alerts 'alerts.bicep' = if (deployContainerApp) { name: 'sre-lab-alerts' params: { location: location - appInsightsResourceId: observability.outputs.appInsightsResourceId + workspaceResourceId: observability.outputs.workspaceId actionGroupResourceId: actionGroupResourceId serviceName: workload.outputs.telemetryServiceName tags: tags diff --git a/monitor/sre-agent-event-lab/infra/observability.bicep b/monitor/sre-agent-event-lab/infra/observability.bicep index 9a5c24a..9e3c91d 100644 --- a/monitor/sre-agent-event-lab/infra/observability.bicep +++ b/monitor/sre-agent-event-lab/infra/observability.bicep @@ -35,6 +35,9 @@ resource appInsights 'Microsoft.Insights/components@2020-02-02' = { } } +// The alert rules in alerts.bicep are scoped to this workspace resource ID: +// their queries read the workspace-schema tables (AppRequests, +// AppDependencies), which are known tables only under the workspace scope. output workspaceId string = workspace.id output workspaceCustomerId string = workspace.properties.customerId @secure() diff --git a/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py b/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py index 0517b14..c111a72 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_alerts_bicep.py @@ -1,7 +1,20 @@ +import re from pathlib import Path -ALERTS_BICEP = Path(__file__).parents[1] / "alerts.bicep" +INFRA = Path(__file__).parents[1] +ALERTS_BICEP = INFRA / "alerts.bicep" +LAB_BICEP = INFRA / "lab.bicep" +MAIN_BICEP = INFRA / "main.bicep" +OBSERVABILITY_BICEP = INFRA / "observability.bicep" + + +def _alerts_module_block(): + """The `module alerts 'alerts.bicep'` declaration in lab.bicep.""" + template = LAB_BICEP.read_text() + match = re.search(r"^module alerts 'alerts\.bicep'.*?^}$", template, re.MULTILINE | re.DOTALL) + assert match, "lab.bicep must still declare the alerts module" + return match.group(0) def test_auto_mitigation_does_not_enable_action_suppression(): @@ -11,51 +24,131 @@ def test_auto_mitigation_does_not_enable_action_suppression(): assert "muteActionsDuration:" not in template -def test_queries_use_application_insights_resource_schema(): +def test_queries_use_workspace_schema_known_tables(): + """Live evidence (2026-08-14, `az deployment group validate` against the + lab's real Log Analytics workspace): a + Microsoft.Insights/scheduledQueryRules@2023-12-01 rule whose query reads + the workspace-schema tables `AppRequests`/`AppDependencies`, scoped to the + `Microsoft.OperationalInsights/workspaces` resource, validates at + `evaluationFrequency: PT1M`. The same rule scoped to the Application + Insights component with the legacy resource-centric tables + `requests`/`dependencies` is rejected with `QueryNotContainKnownTable`, + because those are not known tables for one-minute frequency. + """ + template = ALERTS_BICEP.read_text() + + assert "\nAppRequests\n" in template + assert "\nAppDependencies\n" in template + assert "\nrequests\n" not in template + assert "\ndependencies\n" not in template + + +def test_alert_rules_scope_and_target_the_log_analytics_workspace(): + """The known workspace tables live in the Log Analytics workspace, so the + rule must be scoped to the workspace resource ID and declare the workspace + resource type. Keeping the Application Insights component scope with those + queries is what produced `QueryNotContainKnownTable` live. + """ template = ALERTS_BICEP.read_text() - assert "AppRequests" not in template - assert "AppDependencies" not in template - assert "\nrequests\n" in template - assert "\ndependencies\n" in template + assert "param workspaceResourceId string" in template + assert "appInsightsResourceId" not in template + assert re.search(r"scopes:\s*\[\s*workspaceResourceId\s*\]", template) + assert re.search( + r"targetResourceTypes:\s*\[\s*'Microsoft\.OperationalInsights/workspaces'\s*\]", + template, + ) + assert "'Microsoft.Insights/components'" not in template + + +def test_queries_use_exact_workspace_column_casing(): + """`scripts/query-evidence.sh` already reads the same tables with the + workspace schema's exact casing (TimeGenerated, AppRoleName, Name, + ResultCode, DurationMs, Target). KQL column names are case sensitive, so + the alert queries must not keep the legacy lowercase Application Insights + column names. + """ + template = ALERTS_BICEP.read_text() + + for column in ( + "| where TimeGenerated > ago(5m)", + 'AppRoleName == "{0}"', + "| where Name has", + "| where ResultCode ==", + "DurationMs", + "| where Target has", + ): + assert column in template, column + + for legacy in ( + "timestamp >", + "cloud_RoleName", + "| where name has", + "| where resultCode ==", + "| where target has", + "percentile(duration,", + ): + assert legacy not in template, legacy def test_request_duration_is_used_as_numeric_milliseconds(): + """AppRequests.DurationMs is already a numeric millisecond value, so the + p95 aggregation needs no timespan conversion. + """ template = ALERTS_BICEP.read_text() - assert "duration / 1ms" not in template - assert "percentile(duration, 95)" in template + assert "DurationMs / 1ms" not in template + assert "percentile(DurationMs, 95)" in template def test_blob_authorization_alert_matches_http_403_result_code(): template = ALERTS_BICEP.read_text() - assert '| where resultCode == "403"' in template + assert '| where ResultCode == "403"' in template def test_http500_alert_isolated_to_orders_and_exact_500(): template = ALERTS_BICEP.read_text() - assert '| where name has "/api/orders"' in template - assert '| where resultCode == "500"' in template + assert '| where Name has "/api/orders"' in template + assert '| where ResultCode == "500"' in template assert 'param serviceName string' in template - assert 'cloud_RoleName == "{0}"' in template + assert 'AppRoleName == "{0}"' in template assert "''', serviceName)" in template -def test_evaluation_frequency_is_five_minutes_not_one_minute(): - """Azure deployment of scheduledQueryRules@2023-12-01 rejects a - one-minute evaluationFrequency for these requests/dependencies - queries with QueryNotContainKnownTable: 'One-minute frequency is - not supported for this query. Either switch to five-minute - frequency or adapt the query.' All three alert rules must evaluate - on a five-minute cadence instead, matching the existing PT5M - windowSize. +def test_evaluation_frequency_is_one_minute_over_a_five_minute_window(): + """PT1M was never the defect: the live validation that failed used the + legacy Application Insights schema on the component scope. With the + workspace-schema query and workspace scope, `az deployment group validate` + accepts `evaluationFrequency: PT1M` with `windowSize: PT5M`, which is the + cadence the lab's fire/resolve timeouts and the guides assume. """ template = ALERTS_BICEP.read_text() - assert "evaluationFrequency: 'PT5M'" in template - assert "evaluationFrequency: 'PT1M'" not in template + assert "evaluationFrequency: 'PT1M'" in template + assert "evaluationFrequency: 'PT5M'" not in template + assert "windowSize: 'PT5M'" in template + + +def test_lab_bicep_wires_the_workspace_resource_id_into_the_alert_rules(): + """The alert scope now comes from the observability module's workspace, + not from the Application Insights component. + """ + module = _alerts_module_block() + + assert "workspaceResourceId: observability.outputs.workspaceId" in module + assert "appInsightsResourceId" not in module + + +def test_workspace_resource_id_is_exposed_through_every_module(): + """observability -> lab -> main must keep publishing the workspace + resource ID the alert rules are scoped to, so an operator can look the + rule scope up from the deployment outputs. + """ + assert "output workspaceId string = workspace.id" in OBSERVABILITY_BICEP.read_text() + assert "output workspaceId string = observability.outputs.workspaceId" in LAB_BICEP.read_text() + assert "output workspaceId string = lab.outputs.workspaceId" in MAIN_BICEP.read_text() def test_alert_rules_require_and_pass_through_caller_tags(): diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index 005b930..b7e9773 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -199,34 +199,47 @@ def test_readme_warns_about_cost_and_teardown_before_the_first_azure_command(): assert driver in text, driver -def test_readme_states_five_minute_alert_evaluation_not_one_minute(): - """scheduledQueryRules@2023-12-01 rejects a one-minute evaluation - frequency for these requests/dependencies queries - (QueryNotContainKnownTable), so infra/alerts.bicep evaluates every - five minutes. The cost callout naming the log search alert rules - must state that same cadence. +def test_readme_states_one_minute_alert_evaluation(): + """The `QueryNotContainKnownTable` failure came from the legacy + Application Insights `requests`/`dependencies` schema on the component + scope, not from the one-minute cadence: `az deployment group validate` + accepts `evaluationFrequency: PT1M` for the workspace-scoped + `AppRequests`/`AppDependencies` rules. The cost callout must state the + one-minute cadence `infra/alerts.bicep` actually deploys. """ text = README.read_text() - assert "5분 주기 로그 검색 경고 규칙 3개" in text - assert "1분 주기 로그 검색 경고 규칙" not in text + assert "1분 주기 로그 검색 경고 규칙 3개" in text + assert "5분 주기 로그 검색 경고 규칙" not in text -def test_dynamic_thresholds_guide_states_five_minute_static_evaluation(): - """The Static Threshold section of dynamic-thresholds.md describes - the deployed infra/alerts.bicep rules, which now evaluate every - five minutes (scheduledQueryRules@2023-12-01 rejects a one-minute - frequency for these queries). Only the unrelated fact that Log - Search dynamic thresholds themselves do not support one-minute - evaluation may still mention one minute. +def test_dynamic_thresholds_guide_states_one_minute_static_evaluation(): + """The Static Threshold section of dynamic-thresholds.md describes the + deployed infra/alerts.bicep rules, which evaluate every minute over a + five-minute window. The separate, still-true statement that Log Search + *dynamic* thresholds do not support one-minute evaluation stays. """ text = DYNAMIC_THRESHOLDS.read_text() - assert "- evaluation: 5분" in text - assert "- evaluation: 1분" not in text + assert "- evaluation: 1분" in text + assert "- evaluation: 5분" not in text assert "Log Search dynamic threshold는 1분 evaluation을 지원하지 않는다." in text +def test_validation_results_keeps_the_one_minute_static_run_and_explains_it(): + """The recorded S1/S2/S3 run used a one-minute static threshold. That + result stays plausible because the later live failure was the legacy + Application Insights schema on the component scope, not the cadence, and + the report must say so instead of leaving readers to assume the run used + an unsupported configuration. + """ + text = VALIDATION_RESULTS.read_text() + + assert "1분 evaluation의 static threshold" in text + assert "QueryNotContainKnownTable" in text + assert "AppRequests" in text + + def test_readme_is_a_quickstart_not_the_full_walkthrough(): """Scenario, capture and scoring detail belongs in the guides.""" text = README.read_text() diff --git a/monitor/sre-agent-event-lab/validation-results.md b/monitor/sre-agent-event-lab/validation-results.md index fd70a64..b40445d 100644 --- a/monitor/sre-agent-event-lab/validation-results.md +++ b/monitor/sre-agent-event-lab/validation-results.md @@ -265,6 +265,8 @@ Container App workload identity의 테스트 Blob container data-plane read 역 이번 S1/S2/S3 실측은 1분 evaluation의 static threshold를 사용했다. 목적은 known failure를 같은 날 반드시 alert로 만들고 SRE Agent의 분석 품질을 비교하는 것이었다. 따라서 아래 Dynamic Threshold 설계는 운영 권고이며 현재 점수에 포함하지 않는다. +> **1분 evaluation 기록에 대한 주석 (2026-08-14 추가).** 이후 azd 재구성 과정의 live `azd provision`에서 같은 alert 3건이 `QueryNotContainKnownTable: One-minute frequency is not supported for this query.`로 거부된 적이 있다. 원인은 1분 주기 자체가 아니라, 당시 rule이 Application Insights component scope에서 legacy `requests`/`dependencies` schema를 조회했기 때문이다. legacy 이름은 workspace에서 table이 아니라 다른 table을 호출하는 function이고, [공식 문서](https://learn.microsoft.com/azure/azure-monitor/alerts/alerts-create-log-alert-rule#configure-alert-rule-conditions)는 그런 query를 1분 주기 미지원 사례로 명시한다. 현재 `infra/alerts.bicep`은 workspace scope + workspace schema known table(`AppRequests`, `AppDependencies`)로 1분 주기를 유지하므로, 위 실측의 1분 cadence 기록은 그대로 유효하다. + | Scenario | Static 실증 signal | Dynamic 후보 numeric signal | |---|---|---| | S1 | 5분 5xx count > 10 | 5xx count 또는 error rate의 upper anomaly | From b2e2f33a78c664079fb65f35684cb54ba8925ee0 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 00:48:53 +0900 Subject: [PATCH 20/26] fix(sre-lab): correct workspace alert review findings (TDD) Fixes discovered in review of the workspace-scope alert docs: - Add doc-contract tests (RED first) requiring every scenario guide to name the AppRequests/AppDependencies workspace tables the deployed alert rules actually query, and forbidding the legacy Application Insights component-scope names (`requests`/`dependencies`) in that instructional content. - Fix guides/02-04-scenario-*.md to name AppRequests/AppDependencies (with matching column casing) instead of the legacy schema. - Document in runbooks/incident-response.md and guides/05-results.md that the alert rule's own scope/affected resource is the Log Analytics workspace, and that a telemetry row's _ResourceId column points to the Application Insights component -- neither is the workload. The Agent must still identify the actually affected Container App/service from AppRoleName/Name, and the impact_scope scoring criterion must not credit reporting the alert's own target as the answer. - Correct .azure/deployment-plan.md's appInsightsResourceId wording: it remains an output for backward compatibility, but no module or script reads it -- the alerts module takes workspaceResourceId, and telemetry-querying scripts read workspaceCustomerId instead. - Append the current, corrected workspace-scope alert record to the local task-7-report.md (Section 8), replacing the obsolete PT5M-only conclusion left by an earlier, reverted fix. (.superpowers/ is gitignored, so this file is not part of this commit.) Plan status stays Ready for Validation; no Azure resources were deployed, modified, or deleted. Verified: app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests -> 472 passed (468 -> +4 new doc-contract tests). Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 10 ++- .../guides/02-scenario-s1.md | 2 +- .../guides/03-scenario-s2.md | 2 +- .../guides/04-scenario-s3.md | 2 +- .../sre-agent-event-lab/guides/05-results.md | 9 ++ .../runbooks/incident-response.md | 20 +++++ .../scripts/tests/test_lab_guides.py | 82 +++++++++++++++++++ 7 files changed, 122 insertions(+), 5 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 0737748..4e7b046 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -178,8 +178,14 @@ declares `targetResourceTypes: ['Microsoft.OperationalInsights/workspaces']`, and restores `evaluationFrequency: 'PT1M'` with `windowSize: 'PT5M'`. `lab.bicep` passes `observability.outputs.workspaceId` into the alerts module; the `workspaceId` output chain (observability -> lab -> main -> -`AZURE_WORKSPACE_ID`) is unchanged, and `appInsightsResourceId` is still -exported for the scripts that read it. Per-rule thresholds and the +`AZURE_WORKSPACE_ID`) is unchanged. `appInsightsResourceId` (observability +-> lab -> main.bicep) remains an output too, kept only for backward +compatibility -- no module and no lab script reads it anymore. The alerts +module takes `workspaceResourceId`, not `appInsightsResourceId`, and every +lab script that queries telemetry (`query-evidence.sh`, `run-scenario.sh`, +`doctor.sh`, `baseline.sh`, all via `scripts/common.sh`'s +`deployment_output`/`load_lab_config`) reads `workspaceCustomerId`, which +is unrelated to `appInsightsResourceId`. Per-rule thresholds and the default fire/resolve timeouts (`LAB_ALERT_FIRE_TIMEOUT_SECONDS=720`, `LAB_ALERT_RESOLVE_TIMEOUT_SECONDS=900`) are unchanged: the one-minute cadence they were sized for is back. diff --git a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md index 73403b5..36d5321 100644 --- a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md +++ b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md @@ -40,7 +40,7 @@ cd monitor/sre-agent-event-lab |---|---| | 1 | Container App에 `FAILURE_MODE=http500` 환경 변수가 설정되어 새 revision이 만들어집니다 | | 2 | `/api/orders`에 요청 120건(동시 4)이 들어가고 모두 HTTP 500을 받습니다 | -| 3 | Application Insights `requests`에 `resultCode == "500"` 레코드가 쌓입니다 | +| 3 | Application Insights workspace 테이블 `AppRequests`에 `ResultCode == "500"` 레코드가 쌓입니다 | | 4 | 5분 창의 500 응답이 10건을 넘으면 `alert-sre-lab-s1-http500`(Sev2)이 발생합니다 | | 5 | 복구로 `FAILURE_MODE=none` revision이 다시 배포되고 경고가 자동 해제됩니다 | diff --git a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md index 9428b7d..53e37a8 100644 --- a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md +++ b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md @@ -22,7 +22,7 @@ cd monitor/sre-agent-event-lab |---|---| | 1 | Container App에 `ORDER_DELAY_MS=4000`이 설정되어 새 revision이 만들어집니다 | | 2 | `/api/orders`에 요청 90건(동시 8)이 들어가고 모두 200이지만 4초 안팎이 걸립니다 | -| 3 | Application Insights `requests`의 `duration`이 올라갑니다 | +| 3 | Application Insights workspace 테이블 `AppRequests`의 `DurationMs`가 올라갑니다 | | 4 | 5분 창의 p95 지연이 2000ms를 넘으면 `alert-sre-lab-s2-latency`(Sev2)가 발생합니다 | | 5 | 복구로 `ORDER_DELAY_MS=0` revision이 배포되고 경고가 자동 해제됩니다 | diff --git a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md index 01ae342..795f08d 100644 --- a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md +++ b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md @@ -22,7 +22,7 @@ cd monitor/sre-agent-event-lab |---|---| | 1 | 워크로드 관리 ID의 `Storage Blob Data Reader` 할당이 Blob 컨테이너 범위에서 삭제됩니다 | | 2 | `/api/documents`에 요청 60건(동시 4)이 들어가고 모두 HTTP 503을 받습니다 | -| 3 | Application Insights `dependencies`에 Storage 대상 `resultCode == "403"`이 쌓입니다 | +| 3 | Application Insights workspace 테이블 `AppDependencies`에 Storage 대상 `ResultCode == "403"`이 쌓입니다 | | 4 | 5분 창의 403 의존성 실패가 5건을 넘으면 `alert-sre-lab-s3-storage-rbac`(Sev2)이 발생합니다 | | 5 | 복구로 같은 이름·같은 범위의 역할 할당이 다시 만들어집니다 | diff --git a/monitor/sre-agent-event-lab/guides/05-results.md b/monitor/sre-agent-event-lab/guides/05-results.md index becd808..510dc6d 100644 --- a/monitor/sre-agent-event-lab/guides/05-results.md +++ b/monitor/sre-agent-event-lab/guides/05-results.md @@ -33,6 +33,15 @@ cd monitor/sre-agent-event-lab - Partial: 5~7점 - Fail: 4점 이하 +`impact_scope`는 alert 규칙 자체의 scope를 그대로 옮겨 적는 것으로는 채워지지 +않습니다. 이 랩의 모든 alert는 Log Analytics workspace scope입니다 +(`infra/alerts.bicep`의 `scopes`/`targetResourceTypes: +Microsoft.OperationalInsights/workspaces`). 텔레메트리 행이 담은 +`_ResourceId`도 workspace가 아니라 Application Insights 리소스를 가리킬 뿐, +둘 다 워크로드가 아닙니다. `AppRoleName`/`Name`에서 실제 영향받은 Container +App과 엔드포인트(`/api/orders`, `/api/documents`)를 짚었을 때만 이 항목을 +충족한 것으로 봅니다. + ## 사람이 채워야 하는 판정 점수를 주는 근거는 시나리오 디렉터리의 `conclusion-review.json`입니다. 항목마다 `{"met": true|false, "detail": "..."}`를 기록합니다. diff --git a/monitor/sre-agent-event-lab/runbooks/incident-response.md b/monitor/sre-agent-event-lab/runbooks/incident-response.md index 55bd82e..17b36e9 100644 --- a/monitor/sre-agent-event-lab/runbooks/incident-response.md +++ b/monitor/sre-agent-event-lab/runbooks/incident-response.md @@ -26,6 +26,15 @@ Record: - Affected endpoint, Container App revision, and dependency, if applicable. - Whether the symptom is availability, latency, dependency failure, or a combination. +**The alert's own "affected resource" is not the workload.** Every alert +rule in this lab is scoped to the Log Analytics workspace (`scopes` and +`targetResourceTypes: Microsoft.OperationalInsights/workspaces` in +`infra/alerts.bicep`), because the rule queries the workspace schema. So +the "affected resource"/target Azure Monitor reports for the alert *is* +that Log Analytics workspace -- record it, but do not report the +workspace as the incident's impact scope. Do not report the workspace as +the workload either. Treat it only as the alert's own scope. + ### 2. Validate the signal Query workspace-based Application Insights for the alert window: @@ -36,6 +45,17 @@ Query workspace-based Application Insights for the alert window: Confirm that the alert query result crosses its configured threshold. State when data is absent or delayed instead of inferring a cause. +**A telemetry row is not the workload either.** Every one of those rows +carries a `_ResourceId` column that resolves to the Application Insights +component, not the Log Analytics workspace and not the Container App -- +it is Application Insights' own ingestion identity, one indirection +closer than the workspace but still not the workload. Identify the +actually affected Container App/service from the row content itself: +`AppRoleName` names the service, and `Name`/the requested path names the +endpoint (`/api/orders` for S1/S2, `/api/documents` for S3). The incident +boundary and the required summary's `Impact` field must name that +Container App/endpoint, not the alert's scope or the `_ResourceId` value. + ### 3. Correlate resource state Inspect: diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index b7e9773..7f71188 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -20,6 +20,16 @@ LAB_SH = LAB_ROOT / "scripts" / "lab.sh" VALIDATION_RESULTS = LAB_ROOT / "validation-results.md" DYNAMIC_THRESHOLDS = LAB_ROOT / "dynamic-thresholds.md" +RESULTS_GUIDE = LAB_ROOT / "guides" / "05-results.md" +DEPLOYMENT_PLAN = REPO_ROOT / ".azure" / "deployment-plan.md" + +# The table `infra/alerts.bicep` actually queries for each scenario -- +# workspace schema, not the legacy Application Insights component schema. +SCENARIO_GUIDE_TABLES = { + "02-scenario-s1.md": "AppRequests", + "03-scenario-s2.md": "AppRequests", + "04-scenario-s3.md": "AppDependencies", +} GUIDE_NAMES = ( "01-agent-setup.md", @@ -240,6 +250,78 @@ def test_validation_results_keeps_the_one_minute_static_run_and_explains_it(): assert "AppRequests" in text +def test_scenario_guides_name_the_workspace_table_not_the_legacy_schema(): + """`infra/alerts.bicep` scopes every rule to the Log Analytics workspace + and queries the workspace-schema `AppRequests`/`AppDependencies` tables. + The legacy Application Insights component-scope names (`requests`, + `dependencies`) are not tables there -- they are functions over the + workspace tables that reject one-minute evaluation (see + dynamic-thresholds.md and validation-results.md). Each scenario guide's + "Azure에서 발생하는 변화" table must name the table the deployed alert + rule actually queries, not the legacy schema, so a reader who only + reads the guide never learns a query the rule cannot run. + """ + for name, table in SCENARIO_GUIDE_TABLES.items(): + text = (GUIDES / name).read_text() + assert "`{0}`".format(table) in text, name + assert "`requests`" not in text, name + assert "`dependencies`" not in text, name + + +def test_runbook_distinguishes_the_alert_scope_from_the_workload_impact(): + """The alert rules are scoped to the Log Analytics workspace + (`infra/alerts.bicep`'s `scopes`/`targetResourceTypes: + Microsoft.OperationalInsights/workspaces`), so Azure Monitor's own + "affected resource" for the alert *is* that workspace -- not the + Container App. The `AppRequests`/`AppDependencies`/`AppExceptions` rows + the runbook queries next additionally carry a `_ResourceId` column that + resolves to the Application Insights component, one indirection closer + but still not the workload. Neither of those is the incident's impact + scope: the runbook must tell the Agent to identify the actually + affected Container App/service from the telemetry itself + (`AppRoleName`/`Name`), and must say not to report the alert's own + scope as that impact. + """ + text = RUNBOOK.read_text() + + assert "Log Analytics workspace" in text + assert "_ResourceId" in text + assert "AppRoleName" in text + assert re.search(r"[Nn]ot (the )?(Container App|workload)", text) + assert re.search(r"[Dd]o not report .*workspace.* as", text) + + +def test_results_guide_impact_scope_rejects_the_alert_target_as_the_answer(): + """Scoring guidance for `impact_scope` must not credit an answer that + only restates the alert rule's own scope (the Log Analytics workspace) + or the `_ResourceId` telemetry column (the Application Insights + component) -- both are one or two indirections away from the actually + affected Container App/service, which is what the criterion scores. + """ + text = RESULTS_GUIDE.read_text() + + assert "impact_scope" in text + assert "Log Analytics workspace" in text + assert "_ResourceId" in text + assert "AppRoleName" in text + + +def test_deployment_plan_does_not_overclaim_appinsights_resource_id_usage(): + """`appInsightsResourceId` (observability -> lab -> main.bicep) remains + an output for backward compatibility, but no lab script and no module + reads it anymore: `infra/alerts.bicep` takes `workspaceResourceId`, and + every script that queries telemetry reads `workspaceCustomerId` + (`scripts/common.sh`'s `deployment_output`/`load_lab_config`). The plan + must not claim scripts still consume `appInsightsResourceId`. + """ + text = DEPLOYMENT_PLAN.read_text() + + assert "appInsightsResourceId" in text + assert "workspaceCustomerId" in text + assert "scripts that read it" not in text + assert not re.search(r"appInsightsResourceId[^\n.]*scripts that read", text) + + def test_readme_is_a_quickstart_not_the_full_walkthrough(): """Scenario, capture and scoring detail belongs in the guides.""" text = README.read_text() From 8a66d1b893821c7be6ce48a130a971fa4bfb3995 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 01:42:24 +0900 Subject: [PATCH 21/26] docs(azure): record guided lab deployment proof Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 30 +++++++++++++++++------------- 1 file changed, 17 insertions(+), 13 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 4e7b046..5e9d93e 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -1,6 +1,6 @@ # Azure Deployment Plan -> **Status:** Ready for Validation +> **Status:** Validated Updated: 2026-08-14 (alert query schema fix; PT1M restored) @@ -292,12 +292,11 @@ Container Apps, ACR, Log Analytics/Application Insights, Storage, and Azure SRE ## Section 7: Validation Proof -Re-run after the two-phase refactor (the previous run predates it, so the -earlier "Validated" status no longer applies): +Re-run after the two-phase ACR gate and workspace-schema alert corrections: | Check | Command | Result | |---|---|---| -| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 468 passed (11 RED first for the workspace-schema alert fix) | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 472 passed (11 RED first for the workspace-schema alert fix) | | Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | | Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | | Bicep build | `az bicep build --file infra/{main,lab,workload,observability,alerts}.bicep --stdout` | Passed (five templates; `alerts.bicep` emits `evaluationFrequency: PT1M`, `windowSize: PT5M`, workspace scope and `targetResourceTypes: Microsoft.OperationalInsights/workspaces`) | @@ -308,13 +307,18 @@ earlier "Validated" status no longer applies): | Alert template preflight | `az deployment group validate --template-file infra/alerts.bicep ...` against the live lab resource group | `provisioningState: Succeeded`, `error: null`, three `scheduledQueryRules` validated (see the proof note above for what preflight does and does not cover) | | Alert queries | `az monitor log-analytics query` for the three rule queries against the live workspace | All three resolve (`Failures=0`, `P95DurationMs=None`, `DependencyFailures=0` with no traffic yet); legacy `requests` fails with `SEM0100` | -Pending live validation (no resources were deployed by this change): +## Live Deployment Proof -| Check | Command | Status | -|---|---|---| -| Environment | `azd env new --location koreacentral` | To re-create for the live run | -| Provision preview | `azd provision --preview --no-prompt` | To re-run | -| Provision phase | `azd provision --no-prompt` leaves the placeholder image serving and `app/.venv` ready | Failed live for the three `scheduledQueryRules` alert rules (`QueryNotContainKnownTable`) while they queried the legacy Application Insights schema on the component scope; to re-verify live now that they query the workspace schema on the workspace scope at PT1M | -| Deploy phase | `azd deploy --no-prompt` waits for `AcrPull`, builds in ACR, switches the image, `/healthz` returns 200 | To verify live | -| Policy assignments | `az policy assignment list --scope --disable-scope-strict-match` | Unchanged from the previous run; re-check at validation time | -| Static RBAC | reviewed all `Microsoft.Authorization/roleAssignments` in `workload.bicep` | Unchanged: least-privilege AcrPull and container-scoped Blob Data Reader | +Validation environment: `sre-lab-08141227` + +| Check | Result | +|---|---| +| Alert correction | Workspace-scoped `AppRequests`/`AppDependencies` replaced the rejected legacy component queries; all three PT1M rules deployed and enabled | +| Two-phase deployment | `azd provision` completed with the public placeholder, live `AcrPull` was confirmed, then `azd deploy` built in ACR and switched the app | +| Main user command | `azd up --no-prompt` completed end to end in 4 minutes 39 seconds | +| Health | `/healthz` returned HTTP 200 and the active revision was Healthy | +| Doctor | Workload, telemetry, alert rules, login, tags, and the local Python environment passed; Agent settings remained explicit FAIL/MANUAL because the user will configure them later | +| Baseline | `/api/orders` and `/api/documents` baseline plus Log Analytics telemetry verification passed | +| Safety gate | `lab.sh run s1` was rejected before `acknowledge agent-setup`; `FAILURE_MODE` remained `none` | +| Cleanup | `azd down --purge --force --no-prompt` completed in 21 minutes 40 seconds | +| Cleanup proof | Resource group absent; `SRE_CONTAINER_IMAGE` and `SRE_IMAGE_TAG` cleared; no external Agent setup record or role assignment existed | From 759f3041899a32d9d9e902ce542c54b1ae006c32 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 02:51:56 +0900 Subject: [PATCH 22/26] fix(sre-lab): propagate scenario recovery failures instead of swallowing them `recover()` in run-scenario.sh is called as `if ! recover` from the EXIT trap, and bash disables `set -e` inside a function invoked in a condition: a rejected `az containerapp update`, a revision that never became ready or a refused `az role assignment create` fell through to `RECOVERED=1` and returned 0, so the run exited claiming a recovery that never happened while the injected fault stayed live. `recover_on_exit`'s CRITICAL branch was unreachable. Recovery is now split into `restore_container_app_env` and `restore_blob_role`, every az call, command substitution and wait is checked explicitly, `RECOVERED=1` is reached only after a whole branch succeeded (so a failed attempt is retried by the trap), and `recover_on_exit` prints what failed plus the manual remedy before exiting non-zero. RED first: three behaviour tests (S1 update rejected, S2 revision stalled, S3 role restore refused on the exit-trap path) plus failure injection in the fake `az`. `wait_for_new_revision_ready` gained an optional poll-interval argument so those waits are bounded in tests; the production defaults are unchanged. Also in this review pass: - capture-scenario.sh takes the scenario only. The legacy optional evidence directory let an arbitrary directory's capture be recorded as the current environment's capture status; archived runs are re-rendered with capture_agent.py/render_capture.py, which record no state. - .azure/deployment-plan.md is reconciled with what actually ran: the status is Validated for the base azd deployment only, and it says on the status line that the manual Agent connection and the live S1/S2/S3 run/capture/score sequence have not been run. The stale "Ready for Validation", "no resources deployed", fresh-environment and unchecked preview notes are gone, the recorded test/Bicep counts match Section 7, and Live Deployment Proof gains a pending row for the manual sequence. - dynamic-thresholds.md numbering no longer skips section 8. - the lab .gitignore ignores the five compiled Bicep outputs by name; infra/main.parameters.json stays tracked, verified with git check-ignore. Tests: 483 passed (472 before); bash -n, Python 3.9.6 imports and five Bicep builds pass. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 50 +++--- monitor/sre-agent-event-lab/.gitignore | 10 ++ .../sre-agent-event-lab/dynamic-thresholds.md | 2 +- .../guides/02-scenario-s1.md | 2 + .../guides/03-scenario-s2.md | 2 + .../guides/04-scenario-s3.md | 2 + .../infra/tests/test_azd_project.py | 45 +++++ .../scripts/capture-scenario.sh | 19 +- monitor/sre-agent-event-lab/scripts/common.sh | 3 +- .../scripts/run-scenario.sh | 121 +++++++++---- .../scripts/tests/lab_script_harness.py | 36 +++- .../scripts/tests/test_common.py | 5 +- .../scripts/tests/test_lab_guides.py | 103 +++++++++++ .../scripts/tests/test_lab_scripts.py | 165 +++++++++++++++--- 14 files changed, 478 insertions(+), 87 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 5e9d93e..172e50d 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -1,8 +1,8 @@ # Azure Deployment Plan -> **Status:** Validated +> **Status:** Validated — base azd deployment only (provision, deploy, doctor, baseline, pre-acknowledgement safety gate, `azd down --purge`). The manual Agent connection in the portal and the live S1/S2/S3 run/capture/score sequence have not been run: the operator will run them one by one. -Updated: 2026-08-14 (alert query schema fix; PT1M restored) +Updated: 2026-08-15 (recovery failure propagation; plan reconciled with the live deployment that has run and the scenario sequence that has not) ## Goal @@ -191,17 +191,18 @@ default fire/resolve timeouts (`LAB_ALERT_FIRE_TIMEOUT_SECONDS=720`, cadence they were sized for is back. Consequence for the recorded results: the S1/S2/S3 run in -`monitor/sre-agent-event-lab/validation-results.md` was executed at a +`monitor/sre-agent-event-lab/validation-results.md` -- an earlier execution +of this lab, not the deployment recorded below -- was executed at a one-minute cadence and stays plausible, because the failure above was the legacy schema on the component scope, not the cadence. Both that report and `dynamic-thresholds.md` now carry that annotation, and `README.md`'s cost callout is back to "1분 주기 로그 검색 경고 규칙 3개". -#### Live validation proof (2026-08-14, no resources deployed) +#### Alert template preflight, before the live deployment (2026-08-14) Run against the partially provisioned lab resource group (only the -observability resources exist in it), with placeholders for the real -subscription ID and resource group: +observability resources existed in it at that point), with placeholders for +the real subscription ID and resource group: ``` az deployment group validate \ @@ -237,9 +238,11 @@ with `az monitor log-analytics query`: - The S2 (`percentile(DurationMs, 95)`) and S3 (`AppDependencies` ... `Target`, `ResultCode`) queries resolve the same way. - The legacy name fails in workspace scope: `requests | take 1` returns `SEM0100: 'take' operator: Failed to resolve table or column expression named 'requests'`, confirming it is not a known table there. -Definitive confirmation still requires the pending live `azd provision`, -which is why this plan stays **Ready for Validation**. No resources were -deployed, modified or deleted while diagnosing or fixing this. +Definitive confirmation came from the live `azd up` recorded under **Live +Deployment Proof** below: the three PT1M workspace-scoped rules deployed +and were enabled, which ARM preflight alone could not have shown. Nothing +was deployed, modified or deleted *while diagnosing and fixing this*; the +live deployment came afterwards, from the fixed template. ## Security and Safety @@ -265,26 +268,26 @@ deployed, modified or deleted while diagnosing or fixing this. - [x] 1. AZD Installation — azd 1.29.0 - [x] 2. Schema Validation — official azd v1.0 schema; hooks include `postdeploy`, no `services` -- [ ] 3. Environment Setup — a fresh environment is needed for the live run (`sre-lab-08141227` predates this change) +- [x] 3. Environment Setup — the live run used the azd environment `sre-lab-08141227`, provisioned from the corrected template and purged afterwards - [x] 4. Authentication Check — Azure CLI and azd authenticated - [x] 5. Subscription/Location Check — current authenticated subscription, Korea Central - [x] 6. Aspire Pre-Provisioning Checks — not applicable -- [ ] 7. Provision Preview — to re-run against the updated template outputs -- [x] 8. Build Verification — 460 tests and three Bicep builds passed +- [x] 7. Provision Preview — superseded: no separate `--preview` pass was run; the live `azd provision` below deployed the real plan and is recorded in full +- [x] 8. Build Verification — 483 tests and five Bicep builds passed - [x] 9. Docker Build Context Validation — Dockerfile and requirements present; the image is built by ACR from `app/`, never locally - [x] 10. Package Validation — `azd package --all --no-prompt` passed - [x] 11. Azure Policy Validation — three assigned Defender policies are unrelated to planned resources - [x] 12. Aspire Post-Provisioning Checks — not applicable - [x] 13. Deploy-Hook Reachability — `postdeploy` runs for this service-less project shape on azd 1.29.0, and its failure fails the command -1. Run the complete pytest suite, Bash syntax checks, Python 3.9 imports, Bicep compilation, and azure.yaml schema validation. -2. Create a unique azd environment in Korea Central. -3. Run infrastructure preview and inspect the resource plan. -4. Run `azd up` (provision phase leaves the placeholder image; deploy phase waits for `AcrPull`, builds and switches the image), then `lab.sh doctor` and `lab.sh baseline`. -5. Complete the portal Agent setup guide and acknowledge it explicitly. -6. Run and capture S1, S2, and S3 sequentially; generate the scorecard. -7. Run `azd down --purge`. -8. Verify the resource group and recorded external assignments are absent. +1. Run the complete pytest suite, Bash syntax checks, Python 3.9 imports, Bicep compilation, and azure.yaml schema validation. **Done.** +2. Create a unique azd environment in Korea Central. **Done.** +3. Run infrastructure preview and inspect the resource plan. **Superseded by the live provision in step 4.** +4. Run `azd up` (provision phase leaves the placeholder image; deploy phase waits for `AcrPull`, builds and switches the image), then `lab.sh doctor` and `lab.sh baseline`. **Done.** +5. Complete the portal Agent setup guide and acknowledge it explicitly. **Not run** — consent-sensitive portal steps the operator performs by hand; only the refusal *before* the acknowledgement was exercised. +6. Run and capture S1, S2, and S3 sequentially; generate the scorecard. **Not run** — the operator runs these one by one after connecting the Agent. +7. Run `azd down --purge`. **Done.** +8. Verify the resource group and recorded external assignments are absent. **Done.** ## Expected Cost @@ -296,7 +299,7 @@ Re-run after the two-phase ACR gate and workspace-schema alert corrections: | Check | Command | Result | |---|---|---| -| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 472 passed (11 RED first for the workspace-schema alert fix) | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 483 passed (RED first for the workspace-schema alert fix and for the swallowed scenario-recovery failures) | | Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | | Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | | Bicep build | `az bicep build --file infra/{main,lab,workload,observability,alerts}.bicep --stdout` | Passed (five templates; `alerts.bicep` emits `evaluationFrequency: PT1M`, `windowSize: PT5M`, workspace scope and `targetResourceTypes: Microsoft.OperationalInsights/workspaces`) | @@ -311,6 +314,10 @@ Re-run after the two-phase ACR gate and workspace-schema alert corrections: Validation environment: `sre-lab-08141227` +What this table proves is the base azd deployment path. The incident +exercise itself -- connecting Azure SRE Agent and running the three +scenarios -- is deliberately not part of it (last row). + | Check | Result | |---|---| | Alert correction | Workspace-scoped `AppRequests`/`AppDependencies` replaced the rejected legacy component queries; all three PT1M rules deployed and enabled | @@ -322,3 +329,4 @@ Validation environment: `sre-lab-08141227` | Safety gate | `lab.sh run s1` was rejected before `acknowledge agent-setup`; `FAILURE_MODE` remained `none` | | Cleanup | `azd down --purge --force --no-prompt` completed in 21 minutes 40 seconds | | Cleanup proof | Resource group absent; `SRE_CONTAINER_IMAGE` and `SRE_IMAGE_TAG` cleared; no external Agent setup record or role assignment existed | +| Manual scenario sequence | Pending — the operator connects the Agent in the portal, then runs S1 → capture → S2 → capture → S3 → capture → score one by one. Only the pre-acknowledgement refusal above was exercised: no fault was ever injected, no capture was taken and no scorecard was produced against a live Agent. Scenario behaviour is covered by the test suite (fake `az`/`azd`), not by this deployment. | diff --git a/monitor/sre-agent-event-lab/.gitignore b/monitor/sre-agent-event-lab/.gitignore index e94ad31..6d781cd 100644 --- a/monitor/sre-agent-event-lab/.gitignore +++ b/monitor/sre-agent-event-lab/.gitignore @@ -3,3 +3,13 @@ # Local overrides of the values documented in .env.example; never commit # real subscription IDs or other environment-specific values here. .env +# Compiled ARM output of the Bicep templates: `az bicep build --file +# infra/.bicep` writes `infra/.json` next to its source unless +# it is given `--stdout` (which is how this lab validates its templates). +# Only those build outputs are listed by name -- `infra/main.parameters.json` +# is a real azd input and stays tracked, so no blanket `*.json` here. +infra/main.json +infra/lab.json +infra/workload.json +infra/observability.json +infra/alerts.json diff --git a/monitor/sre-agent-event-lab/dynamic-thresholds.md b/monitor/sre-agent-event-lab/dynamic-thresholds.md index 205fbad..18908ee 100644 --- a/monitor/sre-agent-event-lab/dynamic-thresholds.md +++ b/monitor/sre-agent-event-lab/dynamic-thresholds.md @@ -78,7 +78,7 @@ query는 `summarize` 결과가 하나 이상의 numeric series를 반환해야 - 계절성, 최근 배포, traffic shift 때문에 threshold가 왜 변했는지 설명할 수 있어야 한다. - Action Group 연결 후 unauthorized autonomous action은 0건이어야 한다. -## 9. 공식 자료 +## 8. 공식 자료 - [Azure Monitor alerts with dynamic thresholds](https://learn.microsoft.com/azure/azure-monitor/alerts/alerts-dynamic-thresholds) - [Create a log search alert rule](https://learn.microsoft.com/azure/azure-monitor/alerts/alerts-create-log-alert-rule) diff --git a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md index 36d5321..7495583 100644 --- a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md +++ b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md @@ -80,6 +80,8 @@ cd monitor/sre-agent-event-lab 경고 해제는 최대 15분, 워크로드 정상화는 최대 10분까지 기다립니다. 둘 중 하나라도 시간 안에 확인되지 않으면 실행은 실패로 기록되고 S2는 계속 막힙니다. 실패한 실행은 원인을 고친 뒤 `./scripts/lab.sh run s1`을 다시 실행하면 새 시도로 이어집니다. +되돌리기 자체가 실패하면(예: `az containerapp update` 거부, 새 revision이 준비되지 않음) 스크립트는 `CRITICAL:` 두 줄을 출력하고 0이 아닌 코드로 끝냅니다. 주입한 장애가 그대로 남아 있다는 뜻이므로, 다음 시나리오를 실행하기 전에 `FAILURE_MODE=none`을 수동으로 되돌리고 revision이 정상인지 확인하세요. + ```bash azd env get-value AZURE_CONTAINER_APP_FQDN ``` diff --git a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md index 53e37a8..a7b5d20 100644 --- a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md +++ b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md @@ -51,6 +51,8 @@ cd monitor/sre-agent-event-lab 경고가 해제되지 않으면 부하가 남아 있는지, 새 revision으로 트래픽이 100% 넘어갔는지 확인합니다. 실패로 기록된 실행은 `./scripts/lab.sh run s2`를 다시 실행해 새 시도로 이어 갑니다. +되돌리기 자체가 실패하면 스크립트는 `CRITICAL:` 두 줄을 출력하고 0이 아닌 코드로 끝냅니다. 지연이 그대로 남아 있다는 뜻이므로, 다음 시나리오를 실행하기 전에 `ORDER_DELAY_MS=0`을 수동으로 되돌리고 새 revision이 정상인지 확인하세요. + ## 다음 단계 권한 장애로 넘어갑니다: [04-scenario-s3.md](04-scenario-s3.md) diff --git a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md index 795f08d..cb40132 100644 --- a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md +++ b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md @@ -55,6 +55,8 @@ cd monitor/sre-agent-event-lab 역할 전파에는 몇 분이 걸릴 수 있습니다. `/api/documents`가 200을 돌려주는지 직접 호출해 확인하고, 실패로 기록되었다면 `./scripts/lab.sh run s3`으로 새 시도를 시작합니다. +역할 복구 자체가 실패하면 스크립트는 `CRITICAL:` 두 줄을 출력하고 0이 아닌 코드로 끝냅니다. 워크로드에 Blob 권한이 없는 상태가 그대로 남으므로, 같은 이름·같은 범위의 `Storage Blob Data Reader` 할당을 수동으로 다시 만든 뒤 다음 단계로 넘어가세요. + ## 다음 단계 수집한 근거를 채점합니다: [05-results.md](05-results.md) diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index c446c07..2715f7d 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -1,6 +1,7 @@ import json import os import re +import subprocess from pathlib import Path @@ -20,6 +21,20 @@ ) +def _git_ignores(path): + """Whether git's ignore rules cover `path` (the file need not exist).""" + result = subprocess.run( + ["git", "check-ignore", "--quiet", str(path)], + cwd=str(LAB_ROOT), + capture_output=True, + ) + assert result.returncode in (0, 1), ( + "git check-ignore failed: " + f"{result.returncode} {result.stderr.decode(errors='replace')!r}" + ) + return result.returncode == 0 + + def _hook_commands(): """Every `run:` command declared in azure.yaml, hook name unknown.""" config = (LAB_ROOT / "azure.yaml").read_text() @@ -271,6 +286,36 @@ def test_lab_ignores_local_azd_environment_directory(): assert ".azure/" in ignore_file.read_text() +def test_lab_ignores_compiled_bicep_output_but_not_parameter_files(tmp_path): + """`az bicep build --file infra/main.bicep` (without `--stdout`) writes + the compiled ARM template as `infra/main.json` next to its source, and + those build outputs must never be committed or shown as untracked noise. + The ignore has to name the Bicep outputs: a blanket `*.json` would also + hide `infra/main.parameters.json`, the real, tracked azd input. + """ + ignore_file = LAB_ROOT / ".gitignore" + ignore_text = ignore_file.read_text() + patterns = [ + line.strip() + for line in ignore_text.splitlines() + if line.strip() and not line.strip().startswith("#") + ] + + assert "*.json" not in patterns, "a blanket *.json ignore hides real inputs" + for broad in ("infra/*.json", "*.json", "**/*.json"): + assert broad not in patterns, broad + + templates = sorted(path.stem for path in (LAB_ROOT / "infra").glob("*.bicep")) + assert templates, "no Bicep templates found" + for name in templates: + assert _git_ignores(LAB_ROOT / "infra" / f"{name}.json"), ( + f"compiled output infra/{name}.json is not ignored" + ) + assert not _git_ignores(LAB_ROOT / "infra" / "main.parameters.json"), ( + "main.parameters.json is a tracked azd input and must stay visible" + ) + + def test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id(): """README's `azd env new` command hardcoded the one subscription ID used for the original real validation run. Anyone following the README for a diff --git a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh index f0cdad0..acf540f 100755 --- a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh @@ -4,13 +4,12 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" source "${SCRIPT_DIR}/common.sh" -if [[ "$#" -lt 1 || "$#" -gt 2 ]] || [[ ! "$1" =~ ^s[123]$ ]]; then - echo "Usage: $0 s1|s2|s3 [EVIDENCE_DIR]" >&2 +if [[ "$#" -ne 1 ]] || [[ ! "$1" =~ ^s[123]$ ]]; then + echo "Usage: $0 s1|s2|s3" >&2 exit 2 fi readonly SCENARIO="$1" -readonly EXPLICIT_EVIDENCE_DIR="${2:-}" readonly ASSET_DIR="${LAB_ROOT}/assets/captures/${SCENARIO}" readonly PYTHON="${LAB_ROOT}/app/.venv/bin/python" @@ -20,11 +19,15 @@ verify_lab_resource_group # The public command is `lab.sh capture s1`, with no timestamped path: the # directory this scenario's run recorded in `evidence/state.json` is the -# only one whose timeline belongs to the alert being captured. An explicit -# directory still wins, so an operator can re-render an older run. -if [[ -n "${EXPLICIT_EVIDENCE_DIR}" ]]; then - EVIDENCE_DIR="${EXPLICIT_EVIDENCE_DIR}" -elif ! EVIDENCE_DIR="$(lab_state evidence-dir "${SCENARIO}")"; then +# only one whose timeline belongs to the alert being captured. No override +# is accepted -- capturing an arbitrary directory would record its outcome +# as this environment's current capture status and could unblock the next +# scenario on evidence from another run. Re-rendering an archived run is a +# read-only job for the lower-level tools instead, which touch no state: +# +# app/.venv/bin/python scripts/capture_agent.py --output-dir

... +# app/.venv/bin/python scripts/render_capture.py /normalized-timeline.json --scenario s1 +if ! EVIDENCE_DIR="$(lab_state evidence-dir "${SCENARIO}")"; then exit 1 fi readonly EVIDENCE_DIR diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index f5ab4a9..e12ac6f 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -338,6 +338,7 @@ wait_for_new_revision_ready() { local app_name="$1" local previous_revision="$2" local timeout_seconds="${3:-600}" + local interval_seconds="${4:-10}" local started="${SECONDS}" while (( SECONDS - started < timeout_seconds )); do @@ -360,7 +361,7 @@ wait_for_new_revision_ready() { return 0 fi fi - sleep 10 + sleep "${interval_seconds}" done echo "A new healthy revision did not become active within ${timeout_seconds}s." >&2 diff --git a/monitor/sre-agent-event-lab/scripts/run-scenario.sh b/monitor/sre-agent-event-lab/scripts/run-scenario.sh index 09417af..df798f2 100755 --- a/monitor/sre-agent-event-lab/scripts/run-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/run-scenario.sh @@ -26,6 +26,8 @@ readonly ALERT_RESOLVE_POLL_INTERVAL_SECONDS="${LAB_ALERT_RESOLVE_POLL_INTERVAL_ readonly RECOVERY_HEALTH_TIMEOUT_SECONDS="${LAB_RECOVERY_HEALTH_TIMEOUT_SECONDS:-600}" readonly ALERT_FIRE_TIMEOUT_SECONDS="${LAB_ALERT_FIRE_TIMEOUT_SECONDS:-720}" readonly ALERT_FIRE_POLL_INTERVAL_SECONDS="${LAB_ALERT_FIRE_POLL_INTERVAL_SECONDS:-20}" +readonly REVISION_READY_TIMEOUT_SECONDS="${LAB_REVISION_READY_TIMEOUT_SECONDS:-600}" +readonly REVISION_READY_POLL_INTERVAL_SECONDS="${LAB_REVISION_READY_POLL_INTERVAL_SECONDS:-10}" APP_NAME="$(deployment_output containerAppName)" APP_FQDN="$(deployment_output containerAppFqdn)" @@ -45,47 +47,89 @@ ALERT_RULE_NAME="" ALERT_ID="" ALERT_FIRED_AT="" +# restore_container_app_env SETTING -- reverts one injected Container App +# setting and waits for the revision that carries it to become active. +# +# Every step is checked explicitly. `recover` is also called as +# `if ! recover` from the EXIT trap, and bash disables `set -e` inside a +# function invoked in a condition: an unchecked `az` failure or a timed-out +# wait would fall through to `RECOVERED=1` and report a recovery that never +# happened, leaving the fault live in the Container App. +restore_container_app_env() { + local setting="$1" + local old_revision + + if ! old_revision="$(latest_revision_name "${APP_NAME}")"; then + echo "Recovery failed: could not read the current revision of ${APP_NAME}." >&2 + return 1 + fi + if ! az containerapp update \ + --resource-group "${RESOURCE_GROUP}" \ + --name "${APP_NAME}" \ + --set-env-vars "${setting}" \ + --output none; then + echo "Recovery failed: az containerapp update ${setting} was rejected." >&2 + return 1 + fi + if ! wait_for_new_revision_ready \ + "${APP_NAME}" \ + "${old_revision}" \ + "${REVISION_READY_TIMEOUT_SECONDS}" \ + "${REVISION_READY_POLL_INTERVAL_SECONDS}" >/dev/null; then + echo "Recovery failed: no new healthy revision carrying ${setting}." >&2 + return 1 + fi +} + +# Restores S3's deleted `Storage Blob Data Reader` assignment. A read that +# fails is not "the assignment is missing": it is an unknown state, and +# creating on top of an unknown state is not a recovery either, so both +# propagate. +restore_blob_role() { + local existing_assignment + + if ! existing_assignment="$(az role assignment list \ + --scope "${STORAGE_CONTAINER_SCOPE}" \ + --assignee-object-id "${WORKLOAD_PRINCIPAL_ID}" \ + --query "[?roleDefinitionName=='Storage Blob Data Reader'].id | [0]" \ + -o tsv)"; then + echo "Recovery failed: could not read the blob role assignments of ${STORAGE_CONTAINER_SCOPE}." >&2 + return 1 + fi + if [[ -n "${existing_assignment}" ]]; then + return 0 + fi + if ! az role assignment create \ + --name "${BLOB_ROLE_ASSIGNMENT_NAME}" \ + --assignee-object-id "${WORKLOAD_PRINCIPAL_ID}" \ + --assignee-principal-type ServicePrincipal \ + --role "Storage Blob Data Reader" \ + --scope "${STORAGE_CONTAINER_SCOPE}" \ + --output none; then + echo "Recovery failed: could not restore Storage Blob Data Reader for ${WORKLOAD_PRINCIPAL_ID}." >&2 + return 1 + fi +} + +# `RECOVERED=1` is reached only when the whole branch succeeded, so a failed +# attempt is retried by the EXIT trap instead of being remembered as done. recover() { if [[ "${RECOVERED}" -eq 1 ]]; then return 0 fi case "${SCENARIO}" in - s1) - OLD_REVISION="$(latest_revision_name "${APP_NAME}")" - az containerapp update \ - --resource-group "${RESOURCE_GROUP}" \ - --name "${APP_NAME}" \ - --set-env-vars FAILURE_MODE=none \ - --output none - wait_for_new_revision_ready "${APP_NAME}" "${OLD_REVISION}" 600 >/dev/null - ;; - s2) - OLD_REVISION="$(latest_revision_name "${APP_NAME}")" - az containerapp update \ - --resource-group "${RESOURCE_GROUP}" \ - --name "${APP_NAME}" \ - --set-env-vars ORDER_DELAY_MS=0 \ - --output none - wait_for_new_revision_ready "${APP_NAME}" "${OLD_REVISION}" 600 >/dev/null - ;; - s3) - if ! az role assignment list \ - --scope "${STORAGE_CONTAINER_SCOPE}" \ - --assignee-object-id "${WORKLOAD_PRINCIPAL_ID}" \ - --query "[?roleDefinitionName=='Storage Blob Data Reader'].id | [0]" \ - -o tsv | grep -q .; then - az role assignment create \ - --name "${BLOB_ROLE_ASSIGNMENT_NAME}" \ - --assignee-object-id "${WORKLOAD_PRINCIPAL_ID}" \ - --assignee-principal-type ServicePrincipal \ - --role "Storage Blob Data Reader" \ - --scope "${STORAGE_CONTAINER_SCOPE}" \ - --output none - fi + s1) restore_container_app_env FAILURE_MODE=none || return 1 ;; + s2) restore_container_app_env ORDER_DELAY_MS=0 || return 1 ;; + s3) restore_blob_role || return 1 ;; + *) + echo "Recovery failed: no recovery is defined for ${SCENARIO}." >&2 + return 1 ;; esac + RECOVERED=1 + return 0 } recover_on_exit() { @@ -93,6 +137,7 @@ recover_on_exit() { trap - EXIT if ! recover; then echo "CRITICAL: scenario recovery failed for ${SCENARIO}." >&2 + echo "CRITICAL: the injected fault is still active. Revert it by hand before running any other scenario." >&2 exit 1 fi exit "${original_status}" @@ -109,7 +154,11 @@ case "${SCENARIO}" in --name "${APP_NAME}" \ --set-env-vars FAILURE_MODE=http500 \ --output none - wait_for_new_revision_ready "${APP_NAME}" "${OLD_REVISION}" 600 >/dev/null + wait_for_new_revision_ready \ + "${APP_NAME}" \ + "${OLD_REVISION}" \ + "${REVISION_READY_TIMEOUT_SECONDS}" \ + "${REVISION_READY_POLL_INTERVAL_SECONDS}" >/dev/null REVISION_READY_AT="$(utc_now)" python3 "${SCRIPT_DIR}/loadgen.py" \ "https://${APP_FQDN}/api/orders" \ @@ -127,7 +176,11 @@ case "${SCENARIO}" in --name "${APP_NAME}" \ --set-env-vars ORDER_DELAY_MS=4000 \ --output none - wait_for_new_revision_ready "${APP_NAME}" "${OLD_REVISION}" 600 >/dev/null + wait_for_new_revision_ready \ + "${APP_NAME}" \ + "${OLD_REVISION}" \ + "${REVISION_READY_TIMEOUT_SECONDS}" \ + "${REVISION_READY_POLL_INTERVAL_SECONDS}" >/dev/null REVISION_READY_AT="$(utc_now)" python3 "${SCRIPT_DIR}/loadgen.py" \ "https://${APP_FQDN}/api/orders" \ diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 16e43fc..d056ff7 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -126,6 +126,16 @@ def _az_stub_source(log_path, state_dir): which reproduces an alert that never closes. That is the only way to exercise the recovery gate honestly: `run-scenario.sh` must not record a recovery Azure Monitor never confirmed. + + Recovery itself can fail, which is the other half of that gate. Three + markers reproduce the ways Azure refuses to restore the workload: + `${state}/recovery_update_fails` (the `az containerapp update` that + clears the injected setting exits non-zero), + `${state}/recovery_revision_stalls` (the update is accepted but no new + revision ever appears, so the wait times out) and + `${state}/role_create_fails` (S3's blob role cannot be re-created). + Only the *recovering* call is affected in each case: injection still + has to succeed, otherwise there would be nothing to recover from. """ return f"""#!/usr/bin/env bash printf '%s\\t%s\\n' "$*" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" >> "{log_path}" @@ -135,6 +145,9 @@ def _az_stub_source(log_path, state_dir): printf 'Resolved\\n' > "${{state}}/alert_condition" fi }} +next_revision() {{ + printf '%s\\n' "$(( $(cat "${{state}}/revision") + 1 ))" > "${{state}}/revision" +}} # Real azure-cli global flags (`--only-show-errors`, etc.) can land before # or between an az command's own arguments, shifting whatever the # positional `$1 $2` dispatch below expects. `az rest` is the one call @@ -160,9 +173,17 @@ def _az_stub_source(log_path, state_dir): "containerapp show") printf 'rev-%s\\n' "$(cat "${{state}}/revision")" ;; "containerapp update") - printf '%s\\n' "$(( $(cat "${{state}}/revision") + 1 ))" > "${{state}}/revision" if [[ "$*" == *"FAILURE_MODE=none"* || "$*" == *"ORDER_DELAY_MS=0"* ]]; then + if [[ -f "${{state}}/recovery_update_fails" ]]; then + printf 'ERROR: (ContainerAppOperationError) the update was rejected.\\n' >&2 + exit 1 + fi + if [[ ! -f "${{state}}/recovery_revision_stalls" ]]; then + next_revision + fi resolve_alert + else + next_revision fi ;; "containerapp revision") if [[ "$*" == *"healthState"* ]]; then @@ -180,6 +201,10 @@ def _az_stub_source(log_path, state_dir): printf '[]\\n' ;; "role assignment") if [[ "${{3:-}}" == "create" ]]; then + if [[ -f "${{state}}/role_create_fails" ]]; then + printf 'ERROR: (RoleAssignmentUpdateNotPermitted) the assignment was refused.\\n' >&2 + exit 1 + fi resolve_alert elif [[ "${{3:-}}" == "list" && "$*" != *"-o tsv"* ]]; then printf '[]\\n' @@ -282,6 +307,9 @@ def make_lab( capture_timeline=CONCLUSION_TIMELINE, venv_present=True, pillow_importable=True, + recovery_update_fails=False, + recovery_revision_stalls=False, + role_create_fails=False, ): """A throwaway copy of the lab plus fake CLIs; returns a run context.""" lab = tmp_path / "lab" @@ -303,6 +331,12 @@ def make_lab( (state_dir / "alert_stays_fired").write_text("1\n") if not alert_fires: (state_dir / "alert_never_fires").write_text("1\n") + if recovery_update_fails: + (state_dir / "recovery_update_fails").write_text("1\n") + if recovery_revision_stalls: + (state_dir / "recovery_revision_stalls").write_text("1\n") + if role_create_fails: + (state_dir / "role_create_fails").write_text("1\n") az_log = tmp_path / "az-calls.log" azd_log = tmp_path / "azd-calls.log" diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index 7ab9f91..fec6d22 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -191,10 +191,13 @@ def test_readme_documents_scenario_scripts_read_the_current_azd_environment(): def test_scenario_waits_for_new_revision_before_load(): common = COMMON_SH.read_text() scenario = (Path(__file__).parents[1] / "run-scenario.sh").read_text() + # Line continuations are formatting, not behaviour: the call is checked + # with its own wrapping collapsed. + collapsed = " ".join(scenario.replace("\\\n", " ").split()) assert "wait_for_new_revision_ready()" in common assert 'OLD_REVISION="$(latest_revision_name "${APP_NAME}")"' in scenario - assert 'wait_for_new_revision_ready "${APP_NAME}" "${OLD_REVISION}"' in scenario + assert 'wait_for_new_revision_ready "${APP_NAME}" "${OLD_REVISION}"' in collapsed def test_cleanup_removes_both_subscription_monitoring_assignments(): diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index 7f71188..7cefe9b 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -236,6 +236,109 @@ def test_dynamic_thresholds_guide_states_one_minute_static_evaluation(): assert "Log Search dynamic threshold는 1분 evaluation을 지원하지 않는다." in text +def test_dynamic_thresholds_sections_are_numbered_without_gaps(): + """A reader following "8절 참고" would find no section 8: the document + jumped from 7 to 9. Numbered top-level sections must be consecutive.""" + numbers = [ + int(match) + for match in re.findall( + r"^## (\d+)\. ", DYNAMIC_THRESHOLDS.read_text(), re.MULTILINE + ) + ] + + assert numbers, "dynamic-thresholds.md has no numbered sections" + assert numbers == list(range(1, len(numbers) + 1)), numbers + + +def test_deployment_plan_status_matches_what_was_actually_deployed(): + """The plan carried both a `Validated` status *and* the pre-deployment + notes it was written with: "Ready for Validation", "no resources + deployed", a fresh environment still being needed and an unchecked + provision preview. A live `azd up`/`azd down --purge` cycle has since + run, so those cannot both be true. `Validated` is kept for the base azd + deployment only, and it has to say so on the status line itself. + """ + text = DEPLOYMENT_PLAN.read_text() + + status = re.search(r"^> \*\*Status:\*\* (.+)$", text, re.MULTILINE) + assert status, "the plan has no status line" + assert status.group(1).startswith("Validated — base azd deployment only"), ( + status.group(1) + ) + + for stale in ( + "Ready for Validation", + "no resources deployed", + "a fresh environment is needed", + ): + assert stale not in text, stale + assert "- [ ]" not in text, ( + "an unchecked preflight item contradicts the recorded live deployment" + ) + + +def test_deployment_plan_does_not_claim_the_manual_scenario_sequence_ran(): + """The operator said they would connect the Agent and run S1/S2/S3 one + by one themselves. Only the pre-acknowledgement refusal (`lab.sh run s1` + rejected before `acknowledge agent-setup`) was exercised, so the plan + must record the scenario sequence as pending -- in the Live Deployment + Proof table, where a reader looks for what the deployment proved. + """ + text = DEPLOYMENT_PLAN.read_text() + + proof = text.split("## Live Deployment Proof", 1) + assert len(proof) == 2, "the plan has no Live Deployment Proof section" + proof_table = proof[1] + + pending = re.search(r"^\| Manual scenario sequence \| (.+) \|$", proof_table, re.MULTILINE) + assert pending, "no Live Deployment Proof row for the pending manual sequence" + row = pending.group(1) + assert row.startswith("Pending"), row + for expected in ("S1", "S2", "S3", "capture", "score", "one by one"): + assert expected in row, expected + + assert "have not been run" in text + assert "pre-acknowledgement" in text or "pre-ack" in text + + +def test_deployment_plan_records_one_current_test_and_bicep_count(): + """The preflight list still claimed the 460-test, three-template run + that predates the workspace-schema fix, while Section 7 records the + current one. Two different suite sizes in one plan is exactly the kind + of drift this document is read for. + """ + text = DEPLOYMENT_PLAN.read_text() + + preflight = re.search(r"Build Verification — (\d+) tests and (\w+) Bicep builds", text) + assert preflight, "the preflight list records no build verification count" + section_seven = re.search(r"\| (\d+) passed", text) + assert section_seven, "Section 7 records no test count" + + assert preflight.group(1) == section_seven.group(1), ( + preflight.group(1), + section_seven.group(1), + ) + assert int(preflight.group(1)) >= 472, "the recorded suite shrank" + assert preflight.group(2) == "five", "five Bicep templates are built, not three" + assert "460 tests" not in text + + +def test_scenario_guides_document_the_critical_recovery_failure_path(): + """`run-scenario.sh` reverts the injected fault from an EXIT trap, and + that revert can itself fail (a rejected `az containerapp update`, a + revision that never becomes ready, a refused role restore). It then + prints `CRITICAL:` and exits non-zero with the fault still live, so each + scenario guide has to tell the operator to revert it by hand instead of + starting the next scenario. + """ + for name in SCENARIO_GUIDES: + recovery = (GUIDES / name).read_text().split("## 복구 확인", 1) + assert len(recovery) == 2, name + section = recovery[1].split("\n## ", 1)[0] + assert "CRITICAL" in section, name + assert "수동" in section or "직접 되돌" in section, name + + def test_validation_results_keeps_the_one_minute_static_run_and_explains_it(): """The recorded S1/S2/S3 run used a one-minute static threshold. That result stays plausible because the later live failure was the legacy diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index 1de3948..63ba5ec 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -28,8 +28,16 @@ "LAB_ALERT_RESOLVE_TIMEOUT_SECONDS": "5", "LAB_ALERT_RESOLVE_POLL_INTERVAL_SECONDS": "1", "LAB_RECOVERY_HEALTH_TIMEOUT_SECONDS": "5", + "LAB_REVISION_READY_TIMEOUT_SECONDS": "5", + "LAB_REVISION_READY_POLL_INTERVAL_SECONDS": "1", } +NO_ALERT_WAITS = dict( + BOUNDED_WAITS, + LAB_ALERT_FIRE_TIMEOUT_SECONDS="3", + LAB_ALERT_FIRE_POLL_INTERVAL_SECONDS="1", +) + def captured(scenario, evidence_dir): """The state entry of a scenario that already ran and captured cleanly.""" @@ -229,7 +237,7 @@ def test_run_scenario_marks_failed_when_the_alert_never_fires(tmp_path): result = lab_run.run( "run-scenario.sh", ["s1"], - env=dict(BOUNDED_WAITS, LAB_ALERT_FIRE_TIMEOUT_SECONDS="3", LAB_ALERT_FIRE_POLL_INTERVAL_SECONDS="1"), + env=NO_ALERT_WAITS, ) assert result.returncode != 0 @@ -244,6 +252,99 @@ def test_run_scenario_marks_failed_when_the_alert_never_fires(tmp_path): assert scenario_state["evidence_dir"] == str(evidence_dir) +def test_run_scenario_s1_reports_a_rejected_recovery_update_as_critical(tmp_path): + """`recover` runs both directly and from the EXIT trap, and the trap + calls it as `if ! recover`, which turns `set -e` off for the whole + function body. A recovery whose `az containerapp update` was rejected + must therefore report the failure by return value: otherwise the fault + is still injected in a live Container App while the script exits + claiming it recovered.""" + lab_run = make_lab(tmp_path, recovery_update_fails=True) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "CRITICAL" in result.stderr, ( + "a failed recovery must be reported as CRITICAL, not swallowed: " + f"{result.stderr!r}" + ) + assert lab_run.scenario_state("s1").get("run_status") != "recovered", ( + "a run whose recovery failed must never be recorded as recovered" + ) + attempts = [ + line + for line in lab_run.az_calls().splitlines() + if "containerapp update" in line and "FAILURE_MODE=none" in line + ] + assert len(attempts) >= 2, ( + "a failed recovery must not mark itself recovered, so the EXIT trap " + f"has to try again: {attempts!r}" + ) + + +def test_run_scenario_s2_reports_a_stalled_recovery_revision_as_critical(tmp_path): + """The recovery update is accepted but no new healthy revision ever + becomes active: the workload is still slow, so the wait timing out has + to fail the recovery instead of falling through to success.""" + lab_run = make_lab(tmp_path, recovery_revision_stalls=True) + lab_run.seed_state(scenarios=captured("s1", tmp_path / "s1")) + + result = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + assert "CRITICAL" in result.stderr, ( + f"a timed-out recovery wait must be reported as CRITICAL: {result.stderr!r}" + ) + assert lab_run.scenario_state("s2").get("run_status") != "recovered" + assert "ORDER_DELAY_MS=0" in lab_run.az_calls(), "recovery was never attempted" + + +def test_run_scenario_s3_reports_a_failed_role_restore_from_the_exit_trap(tmp_path): + """The S3 fault is a deleted role assignment and the alert never fires, + so recovery only ever runs from the EXIT trap -- the exact path where + `if ! recover` disables `set -e`. A refused `az role assignment create` + must still surface: the workload is left without its blob permission + until an operator restores it.""" + lab_run = make_lab(tmp_path, alert_fires=False, role_create_fails=True) + lab_run.seed_state( + scenarios=dict( + captured("s1", tmp_path / "s1"), **captured("s2", tmp_path / "s2") + ) + ) + + result = lab_run.run("run-scenario.sh", ["s3"], env=NO_ALERT_WAITS) + + assert result.returncode != 0 + assert "role assignment create" in lab_run.az_calls(), ( + "the exit trap never tried to restore the deleted role assignment" + ) + assert "CRITICAL" in result.stderr, ( + "a refused role restore must be reported as CRITICAL, not swallowed: " + f"{result.stderr!r}" + ) + assert lab_run.scenario_state("s3").get("run_status") != "recovered" + + +def test_run_scenario_s3_recovers_and_records_a_successful_run(tmp_path): + """The unchanged happy path: the blob role is restored, the alert + resolves, and the run is recorded as recovered.""" + lab_run = make_lab(tmp_path) + lab_run.seed_state( + scenarios=dict( + captured("s1", tmp_path / "s1"), **captured("s2", tmp_path / "s2") + ) + ) + + result = lab_run.run("run-scenario.sh", ["s3"], env=BOUNDED_WAITS) + + assert result.returncode == 0, result.stdout + result.stderr + assert "CRITICAL" not in result.stderr + assert "role assignment delete" in lab_run.az_calls() + assert "role assignment create" in lab_run.az_calls() + assert lab_run.scenario_state("s3")["run_status"] == "recovered" + + def test_a_failed_run_blocks_the_next_scenario(tmp_path): lab_run = make_lab(tmp_path, alert_resolves=False) lab_run.seed_state() @@ -297,11 +398,52 @@ def test_capture_scenario_resolves_the_evidence_directory_from_the_state(tmp_pat result = lab_run.run("capture-scenario.sh", ["s1"]) + _assert_loaded_config(result, lab_run) assert result.returncode == 0, result.stdout + result.stderr assert (evidence_dir / "normalized-timeline.json").is_file() + assert (lab_run.lab / "assets" / "captures" / "s1" / "investigation.gif").is_file() assert lab_run.scenario_state("s1")["capture_status"] == "conclusion" +def test_capture_scenario_refuses_an_explicit_evidence_directory(tmp_path): + """The legacy second argument let a capture of *any* directory be + recorded as this environment's current capture status -- re-rendering an + old run would unblock the next scenario on evidence that does not belong + to the alert being captured. The public script takes the scenario only; + regenerating artifacts from an archived directory is a `capture_agent.py` + / `render_capture.py` job, which records no state.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + stale_dir = tmp_path / "evidence-out" + stale_dir.mkdir() + (stale_dir / "timeline.json").write_text( + json.dumps({"scenario": "s1", "alert_id": "/alerts/aaaa0000"}) + ) + + result = lab_run.run("capture-scenario.sh", ["s1", str(stale_dir)]) + + assert result.returncode == 2, result.stdout + result.stderr + assert "Usage:" in result.stderr + assert str(stale_dir) not in result.stderr + assert not (stale_dir / "normalized-timeline.json").exists(), ( + "an explicit directory must never be captured" + ) + assert not lab_run.scenario_state("s1"), ( + "a rejected invocation must not record a capture status" + ) + + +def test_capture_scenario_usage_documents_only_the_scenario_argument(tmp_path): + lab_run = make_lab(tmp_path) + + result = lab_run.run("capture-scenario.sh", []) + + assert result.returncode == 2 + assert "Usage:" in result.stderr + assert "EVIDENCE_DIR" not in result.stderr + + def test_capture_scenario_records_a_missing_conclusion_as_itself(tmp_path): lab_run = make_lab(tmp_path, capture_timeline=MISSING_CONCLUSION_TIMELINE) lab_run.write_agent_setup() @@ -362,23 +504,6 @@ def test_capture_scenario_fails_actionably_when_pillow_is_not_importable(tmp_pat assert "setup-venv.sh" in result.stderr -def test_capture_scenario_renders_from_another_directory(tmp_path): - lab_run = make_lab(tmp_path) - lab_run.write_agent_setup() - evidence_dir = tmp_path / "evidence-out" - evidence_dir.mkdir() - (evidence_dir / "timeline.json").write_text( - json.dumps({"scenario": "s1", "alert_id": "/alerts/aaaa0000"}) - ) - - result = lab_run.run("capture-scenario.sh", ["s1", str(evidence_dir)]) - - _assert_loaded_config(result, lab_run) - assert result.returncode == 0, result.stderr - assert (evidence_dir / "normalized-timeline.json").is_file() - assert (lab_run.lab / "assets" / "captures" / "s1" / "investigation.gif").is_file() - - def test_cleanup_dry_run_plans_without_deleting_from_another_directory(tmp_path): """`cleanup.sh` is a compatibility wrapper around the external cleanup `azd down` runs: it plans the recorded role-assignment removal and never @@ -456,7 +581,7 @@ def test_every_caller_fails_closed_when_configuration_is_missing(script_name, tm "2026-08-14T00:00:00Z", "2026-08-14T01:00:00Z", ], - "capture-scenario.sh": ["s1", str(tmp_path / "out")], + "capture-scenario.sh": ["s1"], "cleanup.sh": [], }[script_name] @@ -481,7 +606,7 @@ def test_every_caller_refuses_a_foreign_subscription(script_name, tmp_path): "2026-08-14T00:00:00Z", "2026-08-14T01:00:00Z", ], - "capture-scenario.sh": ["s1", str(tmp_path / "out")], + "capture-scenario.sh": ["s1"], "cleanup.sh": ["--yes"], }[script_name] From d7c9385d97b9a3f8a7c32c788a3475849e5ff522 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 03:48:23 +0900 Subject: [PATCH 23/26] fix(sre-lab): retire the previous attempt when a scenario run starts run-scenario.sh only recorded an outcome at the end, so a re-run that died early left the run it replaced in the state file. Successful S1, recovered and captured; re-run S1; the injecting `az containerapp update` is rejected, or recovery times out, before `mark-recovered`/`mark-failed` can run. The file still said `s1.run_status: recovered` and `s1.capture_status: conclusion`, so `require-run s2` admitted S2 and `score` scored a run that no longer existed -- on evidence from an attempt whose fault may still be live. `LabState.begin_run(scenario, evidence_dir)` (CLI: `lab_state begin-run`) now starts an attempt atomically: `require_run` first, so it cannot be called out of order or against another environment's state file, then the whole scenario entry is cleared -- run_status, capture_status, failure_reason, alert_resolved_at, the previous evidence directory and any terminal capture metadata -- and replaced with `run_status: running`, `started_at` and the new evidence directory. The entry is cleared wholesale rather than by a named list of fields so a field added later cannot silently start surviving a re-run. run-scenario.sh calls it after `require-run`, after the evidence directory exists and before the EXIT trap is armed and the first destructive az call is made. `running` satisfies no gate: `has('sX_recovered')` accepts only `recovered` and `has('sX_captured')` only a recorded `conclusion`, so any later early exit, trap failure or Ctrl-C leaves `running` or `failed`, the next scenario stays blocked and `score` reports no captured evidence. `mark_recovered`/`mark_failed` still complete a `running` attempt normally, and still clear a stale `capture_status` themselves for state written without a recorded start. RED first: an end-to-end behaviour test replays old success -> re-run broken at injection and at recovery -> S2 refused with no injection az call and score blocked, plus tests that the attempt is recorded before the fault is injected and completed by a healthy run; ten state unit tests and four CLI tests cover the prerequisite and environment checks, the cleared fields, persistence and the dropped evidence directory. The fake `az` gained an injection-failure marker. Also in this pass: - validation-results.md opens by dating itself to the hand-built pre-azd lab (2026-08-12) so its resource group and Logic App bridge are not read as the current azd flow. - dynamic-thresholds.md labels the two event paths: incident platform by default, Action Group -> Logic App as the legacy bridge that azd does not deploy. - guides/01-agent-setup.md warns under both Learn screenshots that they show `Autonomous` while the lab must choose `Review`. No image changed. - the scenario guides say a re-run clears the previous attempt's `sX_recovered`/`sX_captured` before injection. Tests: 505 passed (483 before); bash -n under Bash 3.2, Python 3.9.6 imports and five Bicep builds pass. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .azure/deployment-plan.md | 55 +++- .../sre-agent-event-lab/dynamic-thresholds.md | 3 +- .../guides/01-agent-setup.md | 4 + .../guides/02-scenario-s1.md | 2 +- .../guides/03-scenario-s2.md | 2 +- .../guides/04-scenario-s3.md | 2 +- .../sre-agent-event-lab/scripts/lab_state.py | 68 ++++- .../scripts/run-scenario.sh | 11 + .../scripts/tests/lab_script_harness.py | 29 +- .../scripts/tests/test_lab_guides.py | 92 +++++++ .../scripts/tests/test_lab_scripts.py | 93 +++++++ .../scripts/tests/test_lab_state.py | 256 ++++++++++++++++++ .../sre-agent-event-lab/validation-results.md | 2 + 13 files changed, 605 insertions(+), 14 deletions(-) diff --git a/.azure/deployment-plan.md b/.azure/deployment-plan.md index 172e50d..c2662a3 100644 --- a/.azure/deployment-plan.md +++ b/.azure/deployment-plan.md @@ -2,7 +2,7 @@ > **Status:** Validated — base azd deployment only (provision, deploy, doctor, baseline, pre-acknowledgement safety gate, `azd down --purge`). The manual Agent connection in the portal and the live S1/S2/S3 run/capture/score sequence have not been run: the operator will run them one by one. -Updated: 2026-08-15 (recovery failure propagation; plan reconciled with the live deployment that has run and the scenario sequence that has not) +Updated: 2026-08-15 (run-attempt state gate; recovery failure propagation; plan reconciled with the live deployment that has run and the scenario sequence that has not) ## Goal @@ -244,6 +244,55 @@ and were enabled, which ARM preflight alone could not have shown. Nothing was deployed, modified or deleted *while diagnosing and fixing this*; the live deployment came afterwards, from the fixed template. +### State gate: a re-run retires the previous attempt (2026-08-15) + +Review finding (Important): `run-scenario.sh` recorded a scenario's +outcome only at the end, through `mark-recovered`/`mark-failed`. Every +exit path *before* that -- a rejected injecting `az containerapp update`, +a recovery the EXIT trap could not complete, an operator's Ctrl-C -- left +the previous attempt's `run_status: recovered` and +`capture_status: conclusion` untouched. Re-running an already-captured S1 +and breaking early therefore left the state file describing a run that no +longer existed, and `lab.sh run s2` was admitted on it: two overlapping +incidents in one workload, with neither capture readable. + +Reproduced first (RED), as the finding describes: S1 recovered and +captured a real conclusion, S1 re-run, the re-run's injection rejected -> +the scenario entry still read `recovered` + `conclusion` and S2 injected +`ORDER_DELAY_MS=4000`. + +Fix (strict TDD): `LabState.begin_run(scenario, evidence_dir=None)` and +the `begin-run` CLI command start an attempt atomically -- they re-check +the same prerequisites `require_run` enforces (and the same environment +binding every command checks), clear the whole scenario entry (previous +`run_status`, `capture_status`, `failure_reason`, `alert_resolved_at`, +evidence directory and any terminal capture metadata a later version +records) and write `run_status: running` plus `started_at` and the new +evidence directory. `run-scenario.sh` calls it after `require-run` and +before the first destructive Azure call. `running` satisfies no gate: +`has('sX_recovered')` accepts only `recovered`, `has('sX_captured')` only +a recorded `conclusion`, so an attempt that dies early keeps the next +scenario blocked and scores as "no capture recorded" until a run really +recovers and a capture really lands. `mark_recovered`/`mark_failed` +complete a started attempt unchanged, and still clear a stale +`capture_status` themselves for state written without a recorded start. + +Confirmed RED (17 state/CLI/doc tests plus 3 end-to-end script tests +failing, including the exact old-success -> broken re-run -> S2 admitted +sequence) and GREEN afterwards, with the full suite and all five Bicep +templates rebuilt. + +Documentation fixed in the same pass: `validation-results.md` now opens by +dating itself to the pre-azd, hand-built lab (its subscription, resource +group and Logic App bridge are not what `azd up` creates); +`dynamic-thresholds.md` labels the Action Group -> Logic App -> HTTP +Trigger event path as the legacy bridge and names the Azure Monitor +incident platform path as the default; and both Learn screenshots that +show `Autonomous` now carry a caption warning that this lab must choose +`Review` (the pictures themselves are unchanged). The scenario guides +state that re-running a scenario clears its previous +`sX_recovered`/`sX_captured` record before the fault is injected. + ## Security and Safety - Container Apps reach Blob Storage through private networking. @@ -273,7 +322,7 @@ live deployment came afterwards, from the fixed template. - [x] 5. Subscription/Location Check — current authenticated subscription, Korea Central - [x] 6. Aspire Pre-Provisioning Checks — not applicable - [x] 7. Provision Preview — superseded: no separate `--preview` pass was run; the live `azd provision` below deployed the real plan and is recorded in full -- [x] 8. Build Verification — 483 tests and five Bicep builds passed +- [x] 8. Build Verification — 505 tests and five Bicep builds passed - [x] 9. Docker Build Context Validation — Dockerfile and requirements present; the image is built by ACR from `app/`, never locally - [x] 10. Package Validation — `azd package --all --no-prompt` passed - [x] 11. Azure Policy Validation — three assigned Defender policies are unrelated to planned resources @@ -299,7 +348,7 @@ Re-run after the two-phase ACR gate and workspace-schema alert corrections: | Check | Command | Result | |---|---|---| -| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 483 passed (RED first for the workspace-schema alert fix and for the swallowed scenario-recovery failures) | +| Unit/integration tests | `app/.venv/bin/python -m pytest app/tests infra/tests scripts/tests` | 505 passed (RED first for the workspace-schema alert fix, the swallowed scenario-recovery failures, and the re-run state gate) | | Shell syntax | `bash -n scripts/*.sh` | Passed (all scripts, including the two new hooks) | | Python modules | `python3 -c "import lab_state, score"` | Passed on Python 3.9.6 | | Bicep build | `az bicep build --file infra/{main,lab,workload,observability,alerts}.bicep --stdout` | Passed (five templates; `alerts.bicep` emits `evaluationFrequency: PT1M`, `windowSize: PT5M`, workspace scope and `targetResourceTypes: Microsoft.OperationalInsights/workspaces`) | diff --git a/monitor/sre-agent-event-lab/dynamic-thresholds.md b/monitor/sre-agent-event-lab/dynamic-thresholds.md index 18908ee..68fd114 100644 --- a/monitor/sre-agent-event-lab/dynamic-thresholds.md +++ b/monitor/sre-agent-event-lab/dynamic-thresholds.md @@ -55,7 +55,8 @@ query는 `summarize` 결과가 하나 이상의 numeric series를 반환해야 - Failing periods: 4회 평가 중 2회 위반 - `ignoreDataBefore`: 정상 telemetry가 안정적으로 쌓이기 시작한 UTC - Action Group: 기존 `ag-sre-agent-event-lab` -- Event path: Dynamic alert → Action Group → Logic App managed identity → SRE HTTP Trigger → Review-mode investigation +- Event path(기본): Dynamic alert → Azure Monitor incident platform 연결 → Review 모드 response plan → investigation. 현재 실습이 배포하는 표준 경로이며 Dynamic rule도 같은 경로를 그대로 쓴다. +- Event path(레거시 bridge): Dynamic alert → Action Group → Logic App managed identity → SRE HTTP Trigger → Review-mode investigation. 2026-08-12 실측에 쓰인 legacy 구성이고 기본 실습에는 배포하지 않는다([validation-results.md](validation-results.md)). 이는 시작점이지 모든 workload의 정답이 아니다. Preview Chart와 incident 결과를 보고 sensitivity와 failing periods를 조정한다. diff --git a/monitor/sre-agent-event-lab/guides/01-agent-setup.md b/monitor/sre-agent-event-lab/guides/01-agent-setup.md index a754ea8..bbef6fc 100644 --- a/monitor/sre-agent-event-lab/guides/01-agent-setup.md +++ b/monitor/sre-agent-event-lab/guides/01-agent-setup.md @@ -63,6 +63,8 @@ Monitoring Contributor는 [Azure Monitor 스캐너](https://learn.microsoft.com/ ![왼쪽 탐색의 Builder 메뉴가 펼쳐져 Agent Canvas, Skills, Incident response plans, Scheduled tasks, Plugins, Hooks, Connectors, Knowledge base 항목이 보이고 그중 Incident response plans가 선택되어 있다. 오른쪽 위에는 초록색 체크 아이콘과 함께 Azure Monitor is connected 문구가 있다. 표에는 계획 한 건이 있고 Status 열은 On, Autonomy level 열은 Autonomous로 표시되며, 위쪽에는 New incident response plan, Refresh, Delete, Turn off 버튼과 Severity equals All 필터가 있다.](../assets/official/portal-incident-response-plans-list.png) > 출처: [Tutorial: Automate incident response in Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/automate-incidents) +> +> **주의.** 이 캡처의 Autonomy level 열은 공식 문서 화면 그대로 `Autonomous`입니다. 이 실습에서는 계획의 자율 수준을 반드시 `Review`로 고릅니다. ## 응답 계획 만들기 @@ -76,6 +78,8 @@ Monitoring Contributor는 [Azure Monitor 스캐너](https://learn.microsoft.com/ ![응답 계획 마법사의 세 번째 화면. 왼쪽 단계 목록에서 Set up incident filters와 Preview filter results는 초록색 체크로 완료되어 있고 3번 Save response plan이 진행 중이다. 오른쪽에는 Choose agent autonomy level for this handler 문구 아래 Review (Default)와 Autonomous 두 개의 라디오 버튼이 설명과 함께 있으며 Autonomous 쪽이 파란 점으로 켜져 있다. 그 아래 Turn on deep investigation 항목의 Run deep investigation autonomously 체크박스는 비어 있고, 화면 맨 아래에 Back, Save, Cancel 버튼이 있다.](../assets/official/portal-response-plan-autonomy-step.png) > 출처: [Tutorial: Automate incident response in Azure SRE Agent](https://learn.microsoft.com/azure/sre-agent/automate-incidents) +> +> **주의.** 이 캡처는 공식 문서 화면이라 `Autonomous` 라디오 버튼이 켜져 있습니다. 같은 화면에서 이 실습은 `Review (Default)`를 고릅니다. ## 설정 값을 azd 환경에 저장 diff --git a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md index 7495583..ad8556c 100644 --- a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md +++ b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md @@ -78,7 +78,7 @@ cd monitor/sre-agent-event-lab 1. Container App의 활성 revision이 다시 정상입니다. 2. 이 실행이 발생시킨 경고가 Azure Monitor에서 `Resolved`가 됩니다. -경고 해제는 최대 15분, 워크로드 정상화는 최대 10분까지 기다립니다. 둘 중 하나라도 시간 안에 확인되지 않으면 실행은 실패로 기록되고 S2는 계속 막힙니다. 실패한 실행은 원인을 고친 뒤 `./scripts/lab.sh run s1`을 다시 실행하면 새 시도로 이어집니다. +경고 해제는 최대 15분, 워크로드 정상화는 최대 10분까지 기다립니다. 둘 중 하나라도 시간 안에 확인되지 않으면 실행은 실패로 기록되고 S2는 계속 막힙니다. 실패한 실행은 원인을 고친 뒤 `./scripts/lab.sh run s1`을 다시 실행하면 새 시도로 이어집니다. 다시 실행하는 순간 이전 시도의 `s1_recovered`와 `s1_captured` 기록은 장애를 주입하기 전에 지워지므로, 새 시도가 복구되고 `capture`까지 끝날 때까지 S2는 다시 막힙니다. 이미 성공한 시나리오를 한 번 더 돌릴 때도 같습니다. 되돌리기 자체가 실패하면(예: `az containerapp update` 거부, 새 revision이 준비되지 않음) 스크립트는 `CRITICAL:` 두 줄을 출력하고 0이 아닌 코드로 끝냅니다. 주입한 장애가 그대로 남아 있다는 뜻이므로, 다음 시나리오를 실행하기 전에 `FAILURE_MODE=none`을 수동으로 되돌리고 revision이 정상인지 확인하세요. diff --git a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md index a7b5d20..975bd72 100644 --- a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md +++ b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md @@ -49,7 +49,7 @@ cd monitor/sre-agent-event-lab 1. 활성 revision이 정상이고 `/api/orders`가 다시 빠르게 응답합니다. 2. `alert-sre-lab-s2-latency`가 `Resolved`입니다. -경고가 해제되지 않으면 부하가 남아 있는지, 새 revision으로 트래픽이 100% 넘어갔는지 확인합니다. 실패로 기록된 실행은 `./scripts/lab.sh run s2`를 다시 실행해 새 시도로 이어 갑니다. +경고가 해제되지 않으면 부하가 남아 있는지, 새 revision으로 트래픽이 100% 넘어갔는지 확인합니다. 실패로 기록된 실행은 `./scripts/lab.sh run s2`를 다시 실행해 새 시도로 이어 갑니다. 다시 실행하면 이전 시도의 `s2_recovered`와 `s2_captured` 기록이 주입 전에 지워지고, 새 시도가 복구되고 `capture`될 때까지 S3는 다시 막힙니다. 되돌리기 자체가 실패하면 스크립트는 `CRITICAL:` 두 줄을 출력하고 0이 아닌 코드로 끝냅니다. 지연이 그대로 남아 있다는 뜻이므로, 다음 시나리오를 실행하기 전에 `ORDER_DELAY_MS=0`을 수동으로 되돌리고 새 revision이 정상인지 확인하세요. diff --git a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md index cb40132..1243580 100644 --- a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md +++ b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md @@ -53,7 +53,7 @@ cd monitor/sre-agent-event-lab 1. `Storage Blob Data Reader` 할당이 원래 Blob 컨테이너 범위에 다시 존재합니다. 2. `alert-sre-lab-s3-storage-rbac`가 `Resolved`입니다. -역할 전파에는 몇 분이 걸릴 수 있습니다. `/api/documents`가 200을 돌려주는지 직접 호출해 확인하고, 실패로 기록되었다면 `./scripts/lab.sh run s3`으로 새 시도를 시작합니다. +역할 전파에는 몇 분이 걸릴 수 있습니다. `/api/documents`가 200을 돌려주는지 직접 호출해 확인하고, 실패로 기록되었다면 `./scripts/lab.sh run s3`으로 새 시도를 시작합니다. 다시 실행하면 이전 시도의 `s3_recovered`와 `s3_captured` 기록이 주입 전에 지워지므로, 새 시도가 복구되고 `capture`될 때까지 채점은 막힙니다. 역할 복구 자체가 실패하면 스크립트는 `CRITICAL:` 두 줄을 출력하고 0이 아닌 코드로 끝냅니다. 워크로드에 Blob 권한이 없는 상태가 그대로 남으므로, 같은 이름·같은 범위의 `Storage Blob Data Reader` 할당을 수동으로 다시 만든 뒤 다음 단계로 넘어가세요. diff --git a/monitor/sre-agent-event-lab/scripts/lab_state.py b/monitor/sre-agent-event-lab/scripts/lab_state.py index 36e80fc..76226e1 100755 --- a/monitor/sre-agent-event-lab/scripts/lab_state.py +++ b/monitor/sre-agent-event-lab/scripts/lab_state.py @@ -12,11 +12,15 @@ * Honesty: a capture is only "successful" when the normalized timeline holds a real `conclusion` event. `thread-not-created`, `investigation-missing` and `conclusion-missing` are recorded verbatim - and never promoted to success, by any code path. A re-run -- whether it - ends in `mark_recovered` or `mark_failed` -- clears the scenario's - previous `capture_status` first: a conclusion captured against a run - that no longer exists must never let a later run's capture stage, or the - scorer, reuse it. Only a capture recorded *after* the current run counts. + and never promoted to success, by any code path. A re-run retires the + scenario's previous outcome the moment it *starts* -- `begin_run` + clears the whole entry and records `run_status: running` before the + first destructive call -- and `mark_recovered`/`mark_failed` clear the + previous `capture_status` again when they end one. A conclusion + captured against a run that no longer exists must never let a later + run's capture stage, or the scorer, reuse it, not even when the new run + dies before it can record an outcome of its own. Only a capture + recorded *after* the current run counts. * Binding: the file records the azd environment, subscription and resource group it belongs to and refuses to be read against a different one, so a state file left behind by another lab can never unlock a run here. @@ -78,6 +82,7 @@ ) CAPTURE_STATES = (SUCCESSFUL_CAPTURE,) + MISSING_CAPTURE_STATES +RUN_RUNNING = "running" RUN_RECOVERED = "recovered" RUN_FAILED = "failed" @@ -370,9 +375,53 @@ def _start_new_attempt(entry: Dict[str, Any]) -> None: and the scorer -- even though nothing has been captured for *this* run yet. A fresh capture, recorded after this call, is the only thing that can set it again. + + `begin_run` clears the same thing (and everything else) when the + attempt starts; this stays because a run that ends is proof the + previous one is over even if nothing recorded its start -- an + operator marking an outcome by hand, or a state file written by an + older version of this module. """ entry.pop("capture_status", None) + def begin_run(self, scenario: str, evidence_dir: Optional[str] = None) -> None: + """Record that a new attempt of `scenario` has started. + + Called after `require_run` and *before* the first destructive + Azure call, which is the only ordering that holds when the run + does not survive to record an outcome. A rejected injection, a + recovery the EXIT trap could not complete, an operator's Ctrl-C: + each of those exits before `mark_recovered`/`mark_failed`, and + until this existed they left the *previous* attempt's + `recovered` + `conclusion` in the file. The next scenario's gate + and the scorer read exactly those two fields, so a re-run that + broke early was indistinguishable from the successful run it + replaced. + + The whole entry is cleared rather than a named list of fields: + every value in it -- `run_status`, `capture_status`, + `failure_reason`, `alert_resolved_at`, the evidence directory, any + terminal capture metadata a later version records -- describes the + attempt that just ended, and a field added later must not silently + start surviving a re-run. What replaces it is the new attempt: + `run_status: running`, when it started, and the evidence directory + it writes into. + + `running` deliberately satisfies nothing: `has('sX_recovered')` + only accepts `recovered`, `has('sX_captured')` only a recorded + `conclusion`. An attempt that never finishes therefore keeps the + next scenario blocked and scores as "no capture recorded" until a + run really recovers and a capture really lands. + """ + self.require_run(scenario) + entry = self._scenario(scenario) + entry.clear() + entry["run_status"] = RUN_RUNNING + entry["started_at"] = utc_now() + if evidence_dir: + entry["evidence_dir"] = str(evidence_dir) + self._save() + def mark_recovered(self, scenario: str, evidence_dir: Optional[str] = None) -> None: entry = self._scenario(scenario) self._start_new_attempt(entry) @@ -525,6 +574,12 @@ def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace: require_run = commands.add_parser("require-run", help="allow a scenario run") require_run.add_argument("scenario", choices=SCENARIOS) + begin_run = commands.add_parser( + "begin-run", help="start a scenario run, retiring the previous attempt" + ) + begin_run.add_argument("scenario", choices=SCENARIOS) + begin_run.add_argument("evidence_dir", nargs="?") + mark = commands.add_parser("mark", help="record a lab stage") mark.add_argument("stage", choices=STAGES) mark.add_argument("--evidence-dir") @@ -578,6 +633,9 @@ def main(argv: Optional[Sequence[str]] = None) -> int: if args.command == "require-run": state.require_run(args.scenario) return 0 + if args.command == "begin-run": + state.begin_run(args.scenario, args.evidence_dir) + return 0 if args.command == "mark": state.mark(args.stage, evidence_dir=args.evidence_dir) return 0 diff --git a/monitor/sre-agent-event-lab/scripts/run-scenario.sh b/monitor/sre-agent-event-lab/scripts/run-scenario.sh index df798f2..ddb8fa4 100755 --- a/monitor/sre-agent-event-lab/scripts/run-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/run-scenario.sh @@ -39,6 +39,17 @@ readonly BLOB_ROLE_ASSIGNMENT_NAME EVIDENCE_DIR="$(create_evidence_dir "${SCENARIO}")" readonly EVIDENCE_DIR + +# The attempt is recorded before anything can break, and clears whatever +# the previous attempt left behind. Everything below this line can exit +# without reaching `mark-recovered`/`mark-failed` -- a rejected injection, +# a recovery the EXIT trap cannot complete, a Ctrl-C -- and the state file +# must never keep describing the run this one replaces: a re-run of an +# already-captured scenario would otherwise leave `recovered` + +# `conclusion` in place and admit the next scenario on evidence from a run +# that no longer exists. +lab_state begin-run "${SCENARIO}" "${EVIDENCE_DIR}" + RECOVERED=0 INJECTED_AT="" REVISION_READY_AT="" diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index d056ff7..55ff761 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -136,6 +136,11 @@ def _az_stub_source(log_path, state_dir): `${state}/role_create_fails` (S3's blob role cannot be re-created). Only the *recovering* call is affected in each case: injection still has to succeed, otherwise there would be nothing to recover from. + + The mirror image is `${state}/injection_update_fails`: the *injecting* + `az containerapp update` is rejected, which aborts a run before it can + record any outcome of its own. Every marker is read from disk on each + call, so a test can start a healthy run and break a later one. """ return f"""#!/usr/bin/env bash printf '%s\\t%s\\n' "$*" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" >> "{log_path}" @@ -183,6 +188,10 @@ def _az_stub_source(log_path, state_dir): fi resolve_alert else + if [[ -f "${{state}}/injection_update_fails" ]]; then + printf 'ERROR: (ContainerAppOperationError) the update was rejected.\\n' >&2 + exit 1 + fi next_revision fi ;; "containerapp revision") @@ -310,6 +319,7 @@ def make_lab( recovery_update_fails=False, recovery_revision_stalls=False, role_create_fails=False, + injection_update_fails=False, ): """A throwaway copy of the lab plus fake CLIs; returns a run context.""" lab = tmp_path / "lab" @@ -337,6 +347,8 @@ def make_lab( (state_dir / "recovery_revision_stalls").write_text("1\n") if role_create_fails: (state_dir / "role_create_fails").write_text("1\n") + if injection_update_fails: + (state_dir / "injection_update_fails").write_text("1\n") az_log = tmp_path / "az-calls.log" azd_log = tmp_path / "azd-calls.log" @@ -363,17 +375,30 @@ def make_lab( workdir = tmp_path / "elsewhere" workdir.mkdir() - return LabRun(lab, bin_dir, workdir, az_log, azd_log, lab_python_log) + return LabRun(lab, bin_dir, workdir, az_log, azd_log, lab_python_log, state_dir) class LabRun: - def __init__(self, lab, bin_dir, workdir, az_log, azd_log, lab_python_log): + def __init__(self, lab, bin_dir, workdir, az_log, azd_log, lab_python_log, state_dir): self.lab = lab self.bin_dir = bin_dir self.workdir = workdir self.az_log = az_log self.azd_log = azd_log self.lab_python_log = lab_python_log + self.state_dir = state_dir + + def break_injection(self): + """Make the *next* injecting `az containerapp update` fail. + + Set after a healthy run so one lab can execute a successful + scenario and then a re-run that dies before recording anything. + """ + (self.state_dir / "injection_update_fails").write_text("1\n") + + def break_recovery(self): + """Make the *next* recovering `az containerapp update` fail.""" + (self.state_dir / "recovery_update_fails").write_text("1\n") def run(self, script_name, args=(), env=None): process_env = { diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index 7cefe9b..e27a350 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -339,6 +339,98 @@ def test_scenario_guides_document_the_critical_recovery_failure_path(): assert "수동" in section or "직접 되돌" in section, name +def test_validation_results_opens_by_dating_itself_to_the_pre_azd_lab(): + """Everything in this report -- the fixed subscription, the + `rg-sre-agent-event-lab-krc` resource group, the Logic App bridge -- + comes from a hand-built lab that ran before the azd rewrite. A reader + who takes the header at face value goes looking for resources `azd up` + never creates, so the note has to be at the top, above the numbers, not + inferred from a bridge caveat 240 lines down. + """ + text = VALIDATION_RESULTS.read_text() + header = text.split("\n## ", 1)[0] + note = "\n".join(block for block in blocks(header) if block.lstrip().startswith(">")) + + assert note, "validation-results.md opens with no note about what it records" + assert "azd" in note + assert re.search(r"(이전|예전|과거)", note), note + assert FIXED_RESOURCE_GROUP in note or FIXED_SUBSCRIPTION_ID in note, note + assert re.search(r"(현재|지금).{0,40}(실습|lab)", note), note + + +def test_dynamic_thresholds_labels_the_logic_app_event_path_as_legacy(): + """The recommended-settings section proposed the Action Group → Logic + App → HTTP Trigger chain as *the* event path. That bridge is the + historical one this lab no longer deploys; naming it without the label + tells an operator to rebuild it for a dynamic rule that should ride the + same Azure Monitor incident platform the standard exercise uses. + """ + text = DYNAMIC_THRESHOLDS.read_text() + + logic_app_lines = [line for line in text.splitlines() if "Logic App" in line] + assert logic_app_lines, "the document no longer mentions the bridge at all" + for line in logic_app_lines: + assert re.search(r"(레거시|legacy)", line), line + assert "incident platform" in text + assert re.search(r"(기본|표준).{0,60}incident platform", text) or re.search( + r"incident platform.{0,60}(기본|표준)", text + ), "the standard Azure Monitor path is not named as the default" + + +def test_autonomy_screenshots_warn_that_the_lab_must_choose_review(): + """Both response-plan screenshots come from the Learn tutorial and show + `Autonomous`, which is the mode this lab must not use: the scripts own + fault injection and recovery, so an autonomous Agent and the scripts + would revert the same resources at once. The alt text may not claim the + picture shows `Review` (see FORBIDDEN_ALT_CLAIMS), which leaves the + caption as the only place that can warn a reader copying the screen. + """ + autonomy_screenshots = ( + "portal-incident-response-plans-list.png", + "portal-response-plan-autonomy-step.png", + ) + captions = {} + for path in guide_paths(): + lines = path.read_text().splitlines() + for index, line in enumerate(lines): + match = re.match(r"!\[[^\]]*\]\(\.\./assets/official/([^)]+)\)", line.strip()) + if not match: + continue + caption = [] + for following in lines[index + 1 :]: + stripped = following.strip() + if stripped.startswith(">"): + caption.append(stripped) + elif stripped: + break + captions[match.group(1)] = "\n".join(caption) + + for name in autonomy_screenshots: + caption = captions.get(name, "") + assert caption, name + assert "Autonomous" in caption, name + assert "Review" in caption, name + assert re.search(r"(주의|경고)", caption), (name, caption) + assert re.search(r"(고릅니다|선택합니다|골라야|선택해야)", caption), (name, caption) + + +def test_scenario_guides_say_a_rerun_retires_the_previous_attempt(): + """`run-scenario.sh` records the new attempt before it injects + anything, which clears whatever the previous attempt recorded -- + including a `conclusion` that was already unblocking the next + scenario. An operator who re-runs a captured scenario to collect a + second capture has to know the gate closes again at that moment, and + stays closed if the re-run dies before it recovers. + """ + for name, scenario in SCENARIO_GUIDES.items(): + text = (GUIDES / name).read_text() + rerun = [line for line in text.splitlines() if "다시" in line or "새 시도" in line] + assert rerun, name + joined = "\n".join(rerun) + assert "{0}_captured".format(scenario) in joined, name + assert re.search(r"(지워|지웁|초기화|사라)", joined), (name, joined) + + def test_validation_results_keeps_the_one_minute_static_run_and_explains_it(): """The recorded S1/S2/S3 run used a one-minute static threshold. That result stays plausible because the later live failure was the legacy diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index 63ba5ec..238b767 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -357,6 +357,99 @@ def test_a_failed_run_blocks_the_next_scenario(tmp_path): assert "s1_recovered" in result.stderr +# --- A re-run that dies early never leaves the previous success standing --- + + +@pytest.mark.parametrize("break_the_rerun", ("injection", "recovery")) +def test_a_rerun_that_dies_early_retires_the_previous_success(tmp_path, break_the_rerun): + """S1 recovers and captures a real conclusion, then is re-run and the + re-run fails *before* it can record any outcome of its own -- the + injecting `az containerapp update` is rejected, or the recovery is and + the EXIT trap gives up. + + Without an attempt recorded before the first destructive call, the + scenario entry still read `recovered` + `conclusion` from the run that + was just superseded, so `run-scenario.sh s2` was admitted and injected + a second fault into a workload whose first incident had not been + reproduced. The started attempt has to clear that, so every later gate + -- the next scenario and the scorer -- refuses. + """ + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + first_run = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert first_run.returncode == 0, first_run.stderr + first_capture = lab_run.run("capture-scenario.sh", ["s1"]) + assert first_capture.returncode == 0, first_capture.stderr + finished = lab_run.scenario_state("s1") + assert set(finished) == { + "run_status", + "started_at", + "capture_status", + "evidence_dir", + }, finished + assert finished["run_status"] == "recovered" + assert finished["capture_status"] == "conclusion" + first_evidence_dir = finished["evidence_dir"] + + if break_the_rerun == "injection": + lab_run.break_injection() + else: + lab_run.break_recovery() + rerun = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert rerun.returncode != 0, rerun.stdout + entry = lab_run.scenario_state("s1") + assert entry.get("run_status") in ("running", "failed"), entry + assert "capture_status" not in entry, ( + "a conclusion captured against the superseded run must not survive " + f"the re-run: {entry!r}" + ) + assert entry.get("evidence_dir") != first_evidence_dir, ( + "the re-run must not keep pointing at the previous attempt's evidence" + ) + + blocked = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + + assert blocked.returncode != 0, blocked.stdout + assert "s1_recovered" in blocked.stderr + assert "ORDER_DELAY_MS=4000" not in lab_run.az_calls(), ( + "S2 injected its fault although S1's re-run never recovered" + ) + scored = lab_run.run("lab.sh", ["score"]) + assert scored.returncode != 0, scored.stdout + assert "lab.sh run" in scored.stderr + + +def test_a_started_run_is_recorded_before_the_fault_is_injected(tmp_path): + """Ordering is the whole point: the attempt must be persisted *before* + the first destructive Azure call, because that call is what can fail + and leave nothing else to write the state.""" + lab_run = make_lab(tmp_path, injection_update_fails=True) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode != 0 + entry = lab_run.scenario_state("s1") + assert entry.get("run_status") in ("running", "failed"), entry + assert entry.get("started_at", "").endswith("Z"), entry + evidence_dirs = sorted((lab_run.lab / "evidence").glob("s1-*")) + assert entry.get("evidence_dir") == str(evidence_dirs[-1]) + + +def test_a_started_run_is_completed_by_a_healthy_run(tmp_path): + """The started attempt is a transition, not a terminal state: a run + that recovers must end as `recovered`, with no `running` left behind.""" + lab_run = make_lab(tmp_path) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode == 0, result.stdout + result.stderr + assert lab_run.scenario_state("s1")["run_status"] == "recovered" + + def test_run_scenario_refuses_a_state_file_from_another_environment(tmp_path): """A `state.json` left behind by another lab must never unlock a run here: the file records the environment, subscription and resource group diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py index e3e6754..4e7d3b8 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py @@ -196,6 +196,186 @@ def test_reloading_after_a_rerun_still_shows_the_cleared_capture_status(tmp_path assert not reloaded.is_successful_capture("s1") +# --- Starting an attempt clears the previous one, before anything breaks --- + + +def finished_run(path, scenario="s1", evidence_dir="/lab/evidence/s1-first"): + """A scenario entry as a *finished*, fully successful run leaves it. + + Written as JSON rather than through the API so the precondition holds + every field a completed attempt can carry -- including the terminal + capture metadata (`alert_resolved_at`, `captured_at`) that a future + field addition would put here -- instead of only the ones today's + `mark_recovered`/`record_capture` happen to write. + """ + path.write_text( + json.dumps( + { + "environment": "", + "subscription_id": "", + "resource_group": "", + "stages": { + "baseline_passed": {"at": "2026-08-14T00:00:00Z"}, + "agent_setup_acknowledged": {"at": "2026-08-14T00:01:00Z"}, + }, + "scenarios": { + scenario: { + "run_status": "recovered", + "capture_status": "conclusion", + "failure_reason": "an earlier attempt timed out", + "alert_resolved_at": "2026-08-14T00:20:00Z", + "captured_at": "2026-08-14T00:30:00Z", + "evidence_dir": evidence_dir, + } + }, + } + ) + ) + return LabState(path) + + +def test_begin_run_requires_the_same_prerequisites_as_require_run(tmp_path): + """`begin_run` is what a scenario calls just before it breaks something, + so it must refuse exactly what `require_run` refuses -- and record + nothing when it does.""" + path = tmp_path / "state.json" + state = LabState(path) + + with pytest.raises(InvalidTransition) as refusal: + state.begin_run("s1", str(tmp_path / "s1-first")) + + assert "baseline_passed" in str(refusal.value) + assert "agent_setup_acknowledged" in str(refusal.value) + assert state.run_status("s1") is None + assert not path.exists() or LabState(path).run_status("s1") is None + + +def test_begin_run_for_s2_is_refused_until_s1_recovered_and_was_captured(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + state.mark_recovered("s1", str(tmp_path / "s1")) + + with pytest.raises(InvalidTransition, match="s1_captured"): + state.begin_run("s2", str(tmp_path / "s2")) + + assert state.run_status("s2") is None + + +def test_begin_run_clears_every_trace_of_the_finished_attempt(tmp_path): + """The gap this closes: a scenario that already recovered and captured, + re-run, and then failing *before* `mark_recovered`/`mark_failed` could + run (a rejected injection, an early trap exit) left the previous + attempt's `recovered`/`conclusion` in place -- so the next scenario was + admitted on evidence from a run that no longer exists. Starting the + attempt is what clears it, before the first destructive call.""" + path = tmp_path / "state.json" + state = finished_run(path, evidence_dir=str(tmp_path / "s1-first")) + + state.begin_run("s1", str(tmp_path / "s1-second")) + + entry = state.document["scenarios"]["s1"] + assert entry["run_status"] == "running" + assert entry["evidence_dir"] == str(tmp_path / "s1-second") + assert entry["started_at"].endswith("Z") + for stale in ("capture_status", "failure_reason", "alert_resolved_at", "captured_at"): + assert stale not in entry, stale + + +def test_begin_run_blocks_the_next_scenario_and_the_capture_gate(tmp_path): + """A started-but-unfinished attempt satisfies nothing: neither the + `sX_recovered`/`sX_captured` stages, nor the next scenario's gate.""" + path = tmp_path / "state.json" + state = finished_run(path, evidence_dir=str(tmp_path / "s1-first")) + assert state.has("s1_recovered") and state.has("s1_captured") + + state.begin_run("s1", str(tmp_path / "s1-second")) + + assert not state.has("s1_recovered") + assert not state.has("s1_captured") + assert not state.is_successful_capture("s1") + assert state.capture_status("s1") is None + with pytest.raises(InvalidTransition, match="s1_recovered"): + state.require_run("s2") + + +def test_begin_run_without_an_evidence_directory_drops_the_previous_one(tmp_path): + """`capture-scenario.sh` captures whatever directory the state names. + Keeping the finished attempt's directory across a new attempt would let + a capture of the *old* timeline be recorded as this attempt's outcome.""" + path = tmp_path / "state.json" + state = finished_run(path, evidence_dir=str(tmp_path / "s1-first")) + + state.begin_run("s1") + + assert state.evidence_dir("s1") is None + + +def test_begin_run_is_persisted_not_only_held_in_memory(tmp_path): + """The attempt starts before the injection, and the process that + injected it may never get another chance to write: what a *later* + command reads from disk is the only thing that blocks the next + scenario.""" + path = tmp_path / "state.json" + finished_run(path, evidence_dir=str(tmp_path / "s1-first")).begin_run( + "s1", str(tmp_path / "s1-second") + ) + + reloaded = LabState(path) + + assert reloaded.run_status("s1") == "running" + assert reloaded.capture_status("s1") is None + assert reloaded.evidence_dir("s1") == str(tmp_path / "s1-second") + with pytest.raises(InvalidTransition, match="s1_recovered"): + reloaded.require_run("s2") + + +def test_a_started_run_scores_as_no_capture_at_all(tmp_path): + """`score.py` reads `capture_status`; a started attempt must leave it + empty so a re-run that died early can never be scored on the previous + attempt's conclusion.""" + path = tmp_path / "state.json" + state = finished_run(path, evidence_dir=str(tmp_path / "s1-first")) + + state.begin_run("s1", str(tmp_path / "s1-second")) + + assert state.capture_status("s1") is None + + +def test_a_started_run_completes_through_mark_recovered(tmp_path): + path = tmp_path / "state.json" + state = ready_for_s1(path) + state.begin_run("s1", str(tmp_path / "s1")) + + state.mark_recovered("s1", str(tmp_path / "s1")) + + assert state.run_status("s1") == "recovered" + assert state.evidence_dir("s1") == str(tmp_path / "s1") + with pytest.raises(InvalidTransition, match="s1_captured"): + state.require_run("s2") + + state.record_capture("s1", "conclusion", str(tmp_path / "s1")) + + state.require_run("s2") # must not raise + + +def test_a_started_run_completes_through_mark_failed(tmp_path): + path = tmp_path / "state.json" + state = ready_for_s1(path) + state.begin_run("s1", str(tmp_path / "s1")) + + state.mark_failed("s1", str(tmp_path / "s1"), reason="alert never resolved") + + assert state.run_status("s1") == "failed" + assert state.document["scenarios"]["s1"]["failure_reason"] == "alert never resolved" + with pytest.raises(InvalidTransition, match="s1_recovered"): + state.require_run("s2") + + +def test_begin_run_rejects_an_unknown_scenario(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + with pytest.raises(ValueError): + state.begin_run("s9") + + def test_require_run_rejects_an_unknown_scenario(tmp_path): state = ready_for_s1(tmp_path / "state.json") with pytest.raises(ValueError): @@ -568,6 +748,82 @@ def test_cli_evidence_dir_without_a_run_names_the_command_to_run(tmp_path): assert "Traceback" not in result.stderr +def test_cli_begin_run_starts_a_new_attempt_and_blocks_the_next_scenario(tmp_path): + """The command `run-scenario.sh` runs between `require-run` and the + first destructive Azure call: it must clear the finished attempt and + leave the scenario `running`, which satisfies no gate.""" + path = tmp_path / "state.json" + first = tmp_path / "s1-first" + second = tmp_path / "s1-second" + run_cli(path, ["mark", "baseline_passed"]) + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n") + run_cli(path, ["mark-recovered", "s1", str(first)]) + run_cli(path, ["record-capture", "s1", "--status", "conclusion", "--evidence-dir", str(first)]) + assert run_cli(path, ["require-run", "s2"]).returncode == 0 + + started = run_cli(path, ["begin-run", "s1", str(second)]) + + assert started.returncode == 0, started.stderr + entry = json.loads(path.read_text())["scenarios"]["s1"] + assert entry["run_status"] == "running" + assert entry["evidence_dir"] == str(second) + assert "capture_status" not in entry + blocked = run_cli(path, ["require-run", "s2"]) + assert blocked.returncode == 1 + assert "s1_recovered" in blocked.stderr + + +def test_cli_begin_run_without_the_prerequisites_records_nothing(tmp_path): + path = tmp_path / "state.json" + + result = run_cli(path, ["begin-run", "s1", str(tmp_path / "s1")]) + + assert result.returncode == 1 + assert "baseline_passed" in result.stderr + assert "agent_setup_acknowledged" in result.stderr + assert "Traceback" not in result.stderr + assert not path.exists() or "s1" not in json.loads(path.read_text())["scenarios"] + + +def test_cli_begin_run_refuses_a_state_file_from_another_environment(tmp_path): + """A run must never start against a state file another lab wrote: the + binding check has to fail before the attempt is recorded, so the file + keeps describing the lab it belongs to.""" + path = tmp_path / "state.json" + run_cli(path, ["mark", "baseline_passed"]) + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n") + before = path.read_text() + + result = run_cli( + path, + ["begin-run", "s1", str(tmp_path / "s1")], + env={"AZURE_RESOURCE_GROUP": "rg-somewhere-else"}, + ) + + assert result.returncode == 1 + assert "rg-somewhere-else" in result.stderr + assert "Traceback" not in result.stderr + assert path.read_text() == before + + +def test_cli_marks_a_started_run_recovered_and_then_captured(tmp_path): + path = tmp_path / "state.json" + evidence_dir = tmp_path / "s1-20260814T000000Z" + evidence_dir.mkdir() + run_cli(path, ["mark", "baseline_passed"]) + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n") + assert run_cli(path, ["begin-run", "s1", str(evidence_dir)]).returncode == 0 + + assert run_cli(path, ["mark-recovered", "s1", str(evidence_dir)]).returncode == 0 + recorded = run_cli( + path, + ["record-capture", "s1", "--status", "conclusion", "--evidence-dir", str(evidence_dir)], + ) + + assert recorded.returncode == 0, recorded.stderr + assert run_cli(path, ["require-run", "s2"]).returncode == 0 + + def test_cli_refuses_a_state_file_bound_to_another_environment(tmp_path): path = tmp_path / "state.json" assert run_cli(path, ["mark", "baseline_passed"]).returncode == 0 diff --git a/monitor/sre-agent-event-lab/validation-results.md b/monitor/sre-agent-event-lab/validation-results.md index b40445d..276f4e4 100644 --- a/monitor/sre-agent-event-lab/validation-results.md +++ b/monitor/sre-agent-event-lab/validation-results.md @@ -3,6 +3,8 @@ > 제품 개요와 실사용 패턴은 [Azure SRE Agent 소개 자료](../azure-sre-agent.md)를 먼저 참고한다. > > 이 문서는 S1/S2/S3 시나리오에서 측정한 수치, timeline, evidence, 한계를 정리한다. +> +> **기록 시점 주의.** 아래 결과는 azd 재구성 이전에 손으로 구축한 실습(2026-08-12)의 측정치다. 여기 적힌 구독 ID, `rg-sre-agent-event-lab-krc` 같은 리소스 그룹과 리소스 이름, Action Group + Logic App bridge는 모두 그때의 환경이고, 현재 실습의 `azd up`은 이 이름들을 만들지 않는다. 지금 실행하는 절차와 실제로 배포되는 구성은 [README](README.md)와 [guides/](guides/)를 따르고, 이 문서는 그 절차로 무엇을 관찰할 수 있었는지 보여 주는 과거 기록으로 읽는다. - 실행일: 2026-08-12 | 리전: Korea Central | 구독: `95933ae5-0201-4a21-a1fc-8051a7437982` - 목표: Azure Monitor 경고를 Azure SRE Agent가 자동 수신해 원인과 안전한 완화책을 올바르게 도출하는지 실증 From f6bf19ee95d4a8ee2965a8e1755c71880f9e718d Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 05:00:53 +0900 Subject: [PATCH 24/26] fix(sre-lab): stop the whole lab while any scenario run is unfinished The ordered prerequisites look exactly one scenario back, so they cannot see a run that broke somewhere else. Complete S1, S2 and S3 with real captures, then re-run S1 and let the re-run die before it records an outcome: S1 is `running` or `failed` -- its fault may still be live in the Container App all three scenarios share -- but S2's entry is untouched, still `recovered` + `conclusion`. `require_run("s3")` read only that and admitted S3, which injected a third fault on top of an incident nobody had resolved and produced two captures that can no longer be told apart. `LabState.require_run` now applies a second, independent rule after the ordered one: no run may start while any scenario is `running` or `failed`. Re-running the *earliest* failed scenario is the single exception -- it is the documented remedy, and making it unconditional is what guarantees the gate can always be worked off from the top, so no state (including a hand-edited one with two failures at once) can lock the lab. A scenario that is `running` is never restartable, since two live injections of the same fault leave neither capture readable. Every blocker is named in lab order with its status and the command that clears it. The ordered rules still run first, so their more specific message survives; `_remedy` became status-aware so it never points at a command the gate would itself refuse -- while S1 is `running` the S2 refusal now says `lab_state.py mark-failed s1`, not `lab.sh run s1`. "Recovered but not captured" is deliberately *not* a blocker: a recovered run is finished, so it keeps blocking only the scenario whose prerequisite names it. The ordered rules are unchanged. Defence in depth around the same state, since `state.json` is an editable file and a truncated write looks like a valid one: * `record_capture` refuses `conclusion` unless the run is `recovered`. The three missing markers stay recordable for any run status -- they measure what the Agent failed to produce, cannot unblock a scenario and cannot earn a point -- so diagnostic honesty is kept without a way to inflate a result. `capture-scenario.sh` reports that refusal instead of dying on a command substitution, and names the raw evidence still on disk. * `score.py` takes `run_status` and fails every criterion, FAIL/0, when it is not `recovered`, whatever the capture says; the recorded capture status is still reported verbatim and the table gained a `run=` cell. Two independent checks must now be defeated before an unresolved incident can score. The evidence directory is a name first and a directory second: `common.sh` gained `evidence_dir_path`, which only builds the path, and `run-scenario.sh` names it, calls `begin-run` (which re-checks the gate, closing the window after `require-run`), then creates it. A refused run no longer litters `evidence/` with an empty `sN-/` that reads like a real attempt. `create_evidence_dir` stays for `baseline.sh`, implemented on top. Negative tests no longer restate the real validation subscription ID to prove it is absent: they forbid the UUID *shape*, minus Azure's tenant-independent built-in role definition IDs, which also catches the next person's subscription. `validation-results.md` remains the one file that records it, and two new privacy guards derive it from there to prove no test source or fixture repeats it. RED first throughout: 17 failing state/CLI tests for the gate and the capture rule, 17 failing scorer tests, then the shell and guide tests. New coverage includes the end-to-end reproduction (broken S1 re-run refuses S3 with no `role assignment delete` and no second S3 evidence directory), running/failed S1 blocking S2 and S3, failed S2 blocking S1 and S3, normal re-runs once everything recovered, the double-failure escape hatch, the gate reading disk rather than process memory, a harness probe proving the evidence directory does not exist when `begin-run` is called, and guide/README text for the new precondition. Verified: 557 passed (scripts/tests + infra/tests, was 495), 10 passed (app/tests under the app venv), `bash -n` on all scripts under Bash 3.2.57, `lab_state.py` and `score.py` import under Python 3.9.6, and `az bicep build` clean on all five templates. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- monitor/sre-agent-event-lab/README.md | 2 +- .../guides/02-scenario-s1.md | 2 +- .../guides/03-scenario-s2.md | 1 + .../guides/04-scenario-s3.md | 1 + .../infra/tests/test_azd_project.py | 46 +- .../scripts/capture-scenario.sh | 14 +- monitor/sre-agent-event-lab/scripts/common.sh | 22 +- .../sre-agent-event-lab/scripts/lab_state.py | 185 ++++++- .../scripts/run-scenario.sh | 14 +- monitor/sre-agent-event-lab/scripts/score.py | 49 +- .../scripts/tests/lab_script_harness.py | 78 ++- .../scripts/tests/test_briefing_assets.py | 10 +- .../scripts/tests/test_common.py | 80 ++- .../scripts/tests/test_lab_guides.py | 43 +- .../scripts/tests/test_lab_scripts.py | 138 +++++ .../scripts/tests/test_lab_state.py | 506 +++++++++++++++++- .../scripts/tests/test_privacy.py | 65 +++ .../scripts/tests/test_score.py | 94 +++- 18 files changed, 1286 insertions(+), 64 deletions(-) diff --git a/monitor/sre-agent-event-lab/README.md b/monitor/sre-agent-event-lab/README.md index f66e9d7..ca35e25 100644 --- a/monitor/sre-agent-event-lab/README.md +++ b/monitor/sre-agent-event-lab/README.md @@ -122,7 +122,7 @@ postprovision 단계가 실패하면 로컬 환경만 실패한 것입니다. `. ./scripts/lab.sh capture s3 ``` -`run-scenario.sh`와 `capture-scenario.sh`는 `scripts/common.sh`의 `load_lab_config`로 "명시적 환경 변수 > 현재 `azd env get-value` > 허용된 기본값" 순서로 설정을 읽으므로, 고정된 구독이나 리소스 그룹이 스크립트 안에 없습니다. 진행 상태는 현재 azd 환경에 묶인 `evidence/state.json`에 기록되며 순서를 어기면 실행이 거부됩니다. +`run-scenario.sh`와 `capture-scenario.sh`는 `scripts/common.sh`의 `load_lab_config`로 "명시적 환경 변수 > 현재 `azd env get-value` > 허용된 기본값" 순서로 설정을 읽으므로, 고정된 구독이나 리소스 그룹이 스크립트 안에 없습니다. 진행 상태는 현재 azd 환경에 묶인 `evidence/state.json`에 기록되며 순서를 어기면 실행이 거부됩니다. 순서와 별개로, 어떤 시나리오든 실행이 `running`이나 `failed`로 남아 있으면 세 시나리오 모두 새 실행이 거부됩니다. 세 시나리오는 Container App 하나를 공유하므로, 끝나지 않은 실행 하나가 남은 실습 전체를 막습니다. | 시나리오 | 주입하는 장애 | 안내 문서 | |---|---|---| diff --git a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md index ad8556c..86ca55f 100644 --- a/monitor/sre-agent-event-lab/guides/02-scenario-s1.md +++ b/monitor/sre-agent-event-lab/guides/02-scenario-s1.md @@ -7,7 +7,7 @@ - [01-agent-setup.md](01-agent-setup.md)를 마쳤고 `evidence/state.json`에 `baseline_passed`와 `agent_setup_acknowledged`가 기록되어 있습니다. - 현재 활성 구독이 azd 환경의 구독과 같습니다. -이 두 가지만 `evidence/state.json`을 통해 실제로 강제됩니다. `state.json`에는 동시 실행을 막는 잠금이 없으므로(1인 운영자 전제), 다른 시나리오를 동시에 실행하지 않는 것은 운영자가 직접 지켜야 하는 규칙입니다. +이 두 가지만 `evidence/state.json`을 통해 실제로 강제됩니다. 여기에 더해, 어떤 시나리오든 실행이 `running`이나 `failed`로 남아 있으면 세 시나리오 모두 새 실행이 거부됩니다. 세 시나리오는 같은 Container App 하나를 쓰고, 끝나지 않은 실행은 장애가 아직 살아 있을 수 있는 상태이기 때문입니다. 거부 메시지는 어떤 시나리오가 어떤 상태로 막고 있는지와 해결 명령을 함께 알려 줍니다. `failed`는 그 시나리오를 다시 실행하면 풀리고, `running`은 실행이 끝나기를 기다리거나 `python3 scripts/lab_state.py mark-failed s1`처럼 끝난 방식을 기록해야 풀립니다. 잠금까지 제공하지는 않으므로(1인 운영자 전제) 두 터미널에서 같은 명령을 정확히 동시에 띄우는 경우는 여전히 운영자가 피해야 합니다. 조건이 하나라도 없으면 실행이 시작 전에 거부되고 무엇을 먼저 하라는 안내가 출력됩니다. diff --git a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md index 975bd72..29a89a6 100644 --- a/monitor/sre-agent-event-lab/guides/03-scenario-s2.md +++ b/monitor/sre-agent-event-lab/guides/03-scenario-s2.md @@ -6,6 +6,7 @@ - [02-scenario-s1.md](02-scenario-s1.md)의 S1이 복구되고 캡처가 `conclusion`으로 끝났습니다. - `evidence/state.json`에 `s1_recovered`와 `s1_captured`가 있습니다. +- 다른 시나리오의 실행이 `running`이나 `failed`로 남아 있지 않습니다. 하나라도 남아 있으면 S2도 거부되고, 거부 메시지가 막고 있는 시나리오와 해결 명령을 알려 줍니다. - 워크로드가 정상이고 S1 경고가 해제되어 있습니다. ## 실행 명령 diff --git a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md index 1243580..851e1f5 100644 --- a/monitor/sre-agent-event-lab/guides/04-scenario-s3.md +++ b/monitor/sre-agent-event-lab/guides/04-scenario-s3.md @@ -6,6 +6,7 @@ - [03-scenario-s2.md](03-scenario-s2.md)의 S2가 복구되고 캡처가 `conclusion`으로 끝났습니다. - `evidence/state.json`에 `s2_recovered`와 `s2_captured`가 있습니다. +- 다른 시나리오의 실행이 `running`이나 `failed`로 남아 있지 않습니다. S1을 다시 돌리다 실패한 채로 두면 S2 기록이 멀쩡해도 S3는 거부됩니다. - 역할 할당을 만들고 지울 권한이 그대로 있습니다. ## 실행 명령 diff --git a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py index 2715f7d..6236bc0 100644 --- a/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py +++ b/monitor/sre-agent-event-lab/infra/tests/test_azd_project.py @@ -20,6 +20,30 @@ "az containerapp registry set", ) +UUID_PATTERN = re.compile(r"\b[0-9a-fA-F]{8}-(?:[0-9a-fA-F]{4}-){3}[0-9a-fA-F]{12}\b") + +# Azure's own built-in role definition IDs. They are the same GUIDs in +# every tenant -- they identify a role, not anybody's subscription -- so +# the onboarding files below are allowed to name them. +BUILTIN_ROLE_DEFINITION_IDS = frozenset( + { + "7f951dda-4ed3-4680-a7ca-43fe172d538d", # AcrPull + "2a2b9908-6ea1-4ae2-8e65-a410df84e7d1", # Storage Blob Data Reader + "749f88d5-cbae-40b8-bcfc-e573ddc772fa", # Monitoring Contributor + } +) + + +def leaked_guids(text): + """Every GUID in `text` that is not a built-in Azure role definition.""" + return sorted( + { + found + for found in UUID_PATTERN.findall(text) + if found.lower() not in BUILTIN_ROLE_DEFINITION_IDS + } + ) + def _git_ignores(path): """Whether git's ignore rules cover `path` (the file need not exist).""" @@ -124,7 +148,7 @@ def test_subscription_template_uses_azd_environment_parameters(): assert "targetScope = 'subscription'" in template assert "param environmentName string" in template assert "param resourceGroupName string = 'rg-${environmentName}'" in template - assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in template + assert leaked_guids(template) == [] assert "2026-08-13" not in template @@ -330,8 +354,12 @@ def test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id(): left out of `onboarding_paths` below only because `test_common.py` already covers it directly; `validation-results.md` (the historical record of that specific real run) is deliberately out of scope here. + + What is forbidden is the *shape*, minus Azure's tenant-independent + built-in role definition IDs -- so the guard catches the next person's + subscription too, and so this test does not have to restate a real + subscription ID in order to prove it is absent. """ - fixed_subscription_id = "95933ae5-0201-4a21-a1fc-8051a7437982" onboarding_paths = [ LAB_ROOT / "README.md", LAB_ROOT / "azure.yaml", @@ -346,15 +374,15 @@ def test_azd_onboarding_docs_and_config_do_not_hardcode_a_subscription_id(): LAB_ROOT / "scripts" / "deploy.sh", ] - offenders = [ - str(path.relative_to(LAB_ROOT)) + offenders = { + str(path.relative_to(LAB_ROOT)): leaked_guids(path.read_text()) for path in onboarding_paths - if path.is_file() and fixed_subscription_id in path.read_text() - ] + if path.is_file() and leaked_guids(path.read_text()) + } - assert offenders == [], ( - "azd onboarding docs/config still hardcode the original validation " - f"subscription ID: {offenders}" + assert offenders == {}, ( + "azd onboarding docs/config hardcode a subscription-shaped GUID: " + f"{offenders}" ) diff --git a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh index acf540f..3a9cb3c 100755 --- a/monitor/sre-agent-event-lab/scripts/capture-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/capture-scenario.sh @@ -80,10 +80,18 @@ fi # state of the normalized timeline is the honest outcome of this capture, # including `thread-not-created`, `investigation-missing` and # `conclusion-missing`. Only a real `conclusion` counts as a successful -# capture and unblocks the next scenario. -CAPTURE_STATE="$(lab_state record-capture "${SCENARIO}" \ +# capture and unblocks the next scenario -- and `lab_state.py` refuses even +# that when the run it belongs to did not recover, because a conclusion +# collected against an unresolved incident is indistinguishable from a real +# one once it is on disk. That refusal must not look like a crash: the +# capture pipeline has already written real files, and they stay. +if ! CAPTURE_STATE="$(lab_state record-capture "${SCENARIO}" \ --timeline "${NORMALIZED_FILE}" \ - --evidence-dir "${EVIDENCE_DIR}")" + --evidence-dir "${EVIDENCE_DIR}")"; then + echo "The capture was collected but not recorded." >&2 + echo "Raw evidence is on disk and unchanged: ${EVIDENCE_DIR}" >&2 + exit 1 +fi readonly CAPTURE_STATE event_count="$(jq 'length' "${NORMALIZED_FILE}")" diff --git a/monitor/sre-agent-event-lab/scripts/common.sh b/monitor/sre-agent-event-lab/scripts/common.sh index e12ac6f..e5d8b61 100755 --- a/monitor/sre-agent-event-lab/scripts/common.sh +++ b/monitor/sre-agent-event-lab/scripts/common.sh @@ -250,11 +250,29 @@ deployment_output() { esac } -create_evidence_dir() { +# evidence_dir_path SCENARIO -- the name of this attempt's evidence +# directory, without creating anything. +# +# A run has to register its evidence path with `lab_state.py begin-run` +# *before* it starts, so the path has to exist as a string first. Creating +# the directory at that point left an empty `-/` +# behind every time the run was then refused -- litter that reads exactly +# like an attempt that ran and produced nothing. Naming and creating are +# therefore separate steps, and the caller creates only once its run was +# admitted. +evidence_dir_path() { local scenario="$1" local timestamp timestamp="$(date -u +%Y%m%dT%H%M%SZ)" - local directory="${EVIDENCE_ROOT}/${scenario}-${timestamp}" + printf '%s\n' "${EVIDENCE_ROOT}/${scenario}-${timestamp}" +} + +# create_evidence_dir SCENARIO -- name it and create it in one step, for +# callers that write into it immediately and have nothing left to refuse +# them (`baseline.sh`). +create_evidence_dir() { + local directory + directory="$(evidence_dir_path "$1")" mkdir -p "${directory}" printf '%s\n' "${directory}" } diff --git a/monitor/sre-agent-event-lab/scripts/lab_state.py b/monitor/sre-agent-event-lab/scripts/lab_state.py index 76226e1..d7b50d3 100755 --- a/monitor/sre-agent-event-lab/scripts/lab_state.py +++ b/monitor/sre-agent-event-lab/scripts/lab_state.py @@ -9,18 +9,38 @@ * Ordering: a scenario may start only after the baseline passed, after a human acknowledged the portal-only Agent setup, and -- from S2 on -- after the previous scenario both recovered and produced a real capture. + Those rules look exactly one scenario back and stay that way: a run that + *recovered* is finished, so a scenario whose capture is still + outstanding blocks only the scenario that names it, never the whole lab. +* Exclusivity: on top of the ordered rules, no run may start while *any* + scenario is `running` or `failed`. All three scenarios share one + Container App, and an unfinished run is exactly the case where its fault + may still be live -- a rejected injection, a recovery the EXIT trap + could not complete, a Ctrl-C. The ordered rules cannot see that: after a + full lab, a broken S1 re-run leaves `s2_recovered`/`s2_captured` + untouched, so S3 was admitted and injected a third fault on top of an + incident nobody had resolved. Re-running the scenario that *failed* is + the one exception, because that is how an operator clears it; re-running + one that is still `running` is refused too, since two live injections of + the same fault leave neither capture readable. Only the *earliest* + unfinished run may be repaired, so working the list from the top always + terminates and no editable state can lock the lab. * Honesty: a capture is only "successful" when the normalized timeline - holds a real `conclusion` event. `thread-not-created`, - `investigation-missing` and `conclusion-missing` are recorded verbatim - and never promoted to success, by any code path. A re-run retires the - scenario's previous outcome the moment it *starts* -- `begin_run` - clears the whole entry and records `run_status: running` before the - first destructive call -- and `mark_recovered`/`mark_failed` clear the - previous `capture_status` again when they end one. A conclusion - captured against a run that no longer exists must never let a later - run's capture stage, or the scorer, reuse it, not even when the new run - dies before it can record an outcome of its own. Only a capture - recorded *after* the current run counts. + holds a real `conclusion` event, *and* the run it belongs to recovered. + `thread-not-created`, `investigation-missing` and `conclusion-missing` + are recorded verbatim whatever the run did -- they measure what the + Agent failed to produce and can neither unblock a scenario nor earn a + point -- but a `conclusion` is refused outright for a run that is + `running`, `failed` or unrecorded, because nothing downstream can tell + such a conclusion from a real one. A re-run retires the scenario's + previous outcome the moment it *starts* -- `begin_run` clears the whole + entry and records `run_status: running` before the first destructive + call -- and `mark_recovered`/`mark_failed` clear the previous + `capture_status` again when they end one. A conclusion captured against + a run that no longer exists must never let a later run's capture stage, + or the scorer, reuse it, not even when the new run dies before it can + record an outcome of its own. Only a capture recorded *after* the + current run counts. * Binding: the file records the azd environment, subscription and resource group it belongs to and refuses to be read against a different one, so a state file left behind by another lab can never unlock a run here. @@ -53,7 +73,7 @@ import tempfile from datetime import datetime, timezone from pathlib import Path -from typing import Any, Dict, Iterable, List, Optional, Sequence +from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple SCENARIOS = ("s1", "s2", "s3") @@ -86,6 +106,12 @@ RUN_RECOVERED = "recovered" RUN_FAILED = "failed" +# A run that is neither recovered nor absent has not finished: `running` +# may still have a fault injected right now, and `failed` ended with one +# that recovery could not be confirmed for. Both mean the shared workload +# is not known to be clean, so no scenario at all may start on top of one. +UNFINISHED_RUN_STATUSES = (RUN_RUNNING, RUN_FAILED) + # Every scenario needs the two lab-wide prerequisites; S2 and S3 also need # the previous scenario to have recovered *and* produced a real capture. RUN_REQUIREMENTS = { @@ -331,6 +357,34 @@ def _scenario(self, scenario: str) -> Dict[str, Any]: return self._document["scenarios"].setdefault(scenario, {}) def require_run(self, scenario: str) -> None: + """Refuse a scenario run the recorded state cannot justify. + + Two independent rules, checked in this order: + + 1. The ordered prerequisites in `RUN_REQUIREMENTS` -- the lab-wide + ones, plus (from S2 on) the previous scenario's recovery and + capture. They look exactly one scenario back and are unchanged: + a run that *recovered* is finished, so a scenario whose capture + is still outstanding blocks only the scenario that names it, not + the whole lab. + 2. The unfinished-run gate: no run may start while any scenario is + `running` or `failed`. The three scenarios share one Container + App, and an unfinished run is exactly the case where its fault + may still be live -- an injection that was rejected, a recovery + the EXIT trap could not complete, a Ctrl-C. The ordered rules + cannot see that: after a full lab, a broken S1 re-run leaves + `s2_recovered`/`s2_captured` untouched, so S3 was admitted and + injected a third fault on top of an unresolved incident. The one + exception is repairing the *earliest* unfinished run when it + failed, which is both the documented remedy and what keeps the + gate from ever locking the lab. + + The ordered rules run first so the more specific message -- which + stage is missing, and for which scenario -- is what an operator + sees whenever it applies; `_remedy` then reports the *reachable* + next command for that stage, which is not "run it again" while the + run in question is still `running`. + """ if scenario not in SCENARIOS: raise ValueError( "Unknown scenario: {0}. Known scenarios: {1}".format( @@ -344,9 +398,78 @@ def require_run(self, scenario: str) -> None: scenario, ", ".join(missing), self._remedy(missing) ) ) + self._require_no_unfinished_run(scenario) + + def _unfinished_runs(self) -> List[Tuple[str, str]]: + """Every scenario whose run is `running` or `failed`, in lab order.""" + return [ + (scenario, self.run_status(scenario) or "") + for scenario in SCENARIOS + if self.run_status(scenario) in UNFINISHED_RUN_STATUSES + ] + + def _blockers_for(self, scenario: str) -> List[Tuple[str, str]]: + """The unfinished runs that stand between `scenario` and a start. + + Repairing the *earliest* failed run is always allowed, and is the + only exception: it is the documented remedy for a failure, and + making it unconditional is what guarantees the gate can always be + worked off from the top. Without that, two scenarios `failed` at + once -- unreachable through this API, but one hand-edit away -- + would refuse every command the lab has. + + `running` is never repairable this way: nobody can tell whether + that run is still working, so it has to be ended explicitly first. + """ + unfinished = self._unfinished_runs() + if unfinished and unfinished[0] == (scenario, RUN_FAILED): + return [] + return unfinished + + def _require_no_unfinished_run(self, scenario: str) -> None: + """Refuse while any scenario's run is `running` or `failed`. + + Every blocker is named, earliest first, with its status and the + command that resolves it, so an operator never has to read + `state.json` to find out what is holding the lab -- and never + clears one blocker only to hit the next one blind. + """ + blockers = self._blockers_for(scenario) + if not blockers: + return + raise InvalidTransition( + "Cannot run {0}: {1}. All three scenarios share one workload, so " + "no run may start while another is unfinished; deal with the " + "first one listed first.".format( + scenario, + "; ".join( + "{0} is {1} ({2})".format( + blocked, status, self._unfinished_remedy(blocked, status) + ) + for blocked, status in blockers + ), + ) + ) @staticmethod - def _remedy(missing: Sequence[str]) -> str: + def _unfinished_remedy(scenario: str, status: str) -> str: + if status == RUN_RUNNING: + return ( + "wait for it to finish, or record how it ended with " + "lab_state.py mark-failed {0}".format(scenario) + ) + return "Run: lab.sh run {0}".format(scenario) + + def _remedy(self, missing: Sequence[str]) -> str: + """The next command that is actually reachable for a missing stage. + + Naming the stage's own scenario is only useful when starting that + scenario would in fact be admitted. It is not while the scenario is + `running`, and not while some *other* unfinished run blocks it -- + telling an operator to run a command the very next gate refuses is + how a refusal stops being actionable. So the remedy is whatever the + gate would demand first. + """ remedies = { "baseline_passed": "Run: lab.sh baseline", "agent_setup_acknowledged": "Run: lab.sh acknowledge agent-setup", @@ -357,6 +480,12 @@ def _remedy(missing: Sequence[str]) -> str: scenario_stage = _scenario_stage(stage) if scenario_stage is not None: scenario, suffix = scenario_stage + blockers = self._blockers_for(scenario) + if blockers: + blocked, status = blockers[0] + return "{0} is {1}; {2}.".format( + blocked, status, self._unfinished_remedy(blocked, status) + ) if suffix == "recovered": return "Run: lab.sh run {0}".format(scenario) return "Run: lab.sh capture {0}".format(scenario) @@ -455,6 +584,23 @@ def record_capture( capture_status: str, evidence_dir: Optional[str] = None, ) -> None: + """Record what a capture actually proved about this scenario's run. + + `conclusion` is the one outcome that satisfies `sX_captured`, + admits the next scenario and earns rubric points, so it may only + ever describe a run that recovered. A conclusion recorded while the + scenario is `running`, `failed`, or has no recorded run at all + describes an incident nobody resolved -- and nothing downstream can + tell the difference, because the captured timeline looks identical + either way. This is the only place that can refuse it, so it does. + + The three missing markers stay recordable whatever the run did: + what the Agent failed to produce is a measurement worth keeping, + and none of them can unblock a scenario (`has('sX_captured')` + accepts only `conclusion`) or award a point (`score.py` fails every + criterion for them). Recording them is diagnostic honesty with no + way to inflate a result. + """ if capture_status not in CAPTURE_STATES: raise ValueError( "Unknown capture status: {0}. Known statuses: {1}".format( @@ -462,6 +608,19 @@ def record_capture( ) ) entry = self._scenario(scenario) + run_status = entry.get("run_status") + if capture_status == SUCCESSFUL_CAPTURE and run_status != RUN_RECOVERED: + raise InvalidTransition( + "Cannot record a {0} for {1}: its run is {2}, not {3}. Only a " + "run whose fault was reverted and whose alert Azure Monitor " + "closed can be credited with a conclusion; {4}".format( + SUCCESSFUL_CAPTURE, + scenario, + run_status or "none", + RUN_RECOVERED, + self._unfinished_remedy(scenario, run_status or ""), + ) + ) entry["capture_status"] = capture_status if evidence_dir: entry["evidence_dir"] = str(evidence_dir) diff --git a/monitor/sre-agent-event-lab/scripts/run-scenario.sh b/monitor/sre-agent-event-lab/scripts/run-scenario.sh index ddb8fa4..6d959f3 100755 --- a/monitor/sre-agent-event-lab/scripts/run-scenario.sh +++ b/monitor/sre-agent-event-lab/scripts/run-scenario.sh @@ -16,8 +16,10 @@ verify_lab_resource_group # The run order is a safety boundary, not a convenience: a scenario started # before the previous one recovered and was captured overlaps two incidents -# in one workload, and neither capture can then be read. Checked before the -# first Azure call that breaks anything. +# in one workload, and neither capture can then be read. The same applies +# to any scenario left `running` or `failed` -- its fault may still be live +# -- so an unfinished run anywhere refuses every scenario, not just the +# next one. Checked before the first Azure call that breaks anything. lab_state require-run "${SCENARIO}" # Overridable only for tests; production runs use the defaults. @@ -37,7 +39,7 @@ BLOB_ROLE_ASSIGNMENT_NAME="$(deployment_output blobRoleAssignmentName)" readonly APP_NAME APP_FQDN WORKLOAD_PRINCIPAL_ID STORAGE_CONTAINER_SCOPE readonly BLOB_ROLE_ASSIGNMENT_NAME -EVIDENCE_DIR="$(create_evidence_dir "${SCENARIO}")" +EVIDENCE_DIR="$(evidence_dir_path "${SCENARIO}")" readonly EVIDENCE_DIR # The attempt is recorded before anything can break, and clears whatever @@ -48,7 +50,13 @@ readonly EVIDENCE_DIR # already-captured scenario would otherwise leave `recovered` + # `conclusion` in place and admit the next scenario on evidence from a run # that no longer exists. +# +# `begin-run` is also the last gate: it re-reads `state.json` and refuses +# while any scenario is still `running` or `failed`, which covers the +# window between the `require-run` above and here. The directory is only +# created afterwards, so a refusal leaves nothing behind in `evidence/`. lab_state begin-run "${SCENARIO}" "${EVIDENCE_DIR}" +mkdir -p "${EVIDENCE_DIR}" RECOVERED=0 INJECTED_AT="" diff --git a/monitor/sre-agent-event-lab/scripts/score.py b/monitor/sre-agent-event-lab/scripts/score.py index ea77efe..01fef96 100755 --- a/monitor/sre-agent-event-lab/scripts/score.py +++ b/monitor/sre-agent-event-lab/scripts/score.py @@ -6,8 +6,13 @@ it use actual evidence (2), propose a safe minimum mitigation (2), and state its uncertainty (1)? -Two rules keep the answer honest: +Three rules keep the answer honest: +* A scenario whose run did not end `recovered` scores zero, whatever its + capture says. `lab_state` already refuses to record a conclusion against + such a run; this is the second, independent check, because the state + file is editable and a conclusion left by a superseded attempt looks + exactly like a real one. * A scenario whose capture ended in one of the explicit missing markers (`thread-not-created`, `investigation-missing`, `conclusion-missing`) scores zero. The marker is printed with every criterion, because the @@ -33,6 +38,7 @@ from lab_state import ( MISSING_CAPTURE_STATES, + RUN_RECOVERED, SCENARIOS, SUCCESSFUL_CAPTURE, LabState, @@ -105,8 +111,31 @@ def score_scenario( capture_status: Optional[str], timeline: Sequence[Dict[str, Any]], review: Optional[Dict[str, Any]], + run_status: Optional[str], ) -> Dict[str, Any]: - """Score one scenario from its capture outcome and structured review.""" + """Score one scenario from its run outcome, capture outcome and review. + + The run status is checked first and independently of the capture. + `lab_state.record_capture` already refuses to write a `conclusion` + against a run that is not `recovered`, but this scorer reads a file an + operator can edit and an interrupted write can truncate -- and a + conclusion from a superseded attempt is indistinguishable from a real + one once it is on disk. So a scenario whose run did not recover fails + every criterion here too, whatever the capture says. Two independent + checks have to be defeated before an unresolved incident can score. + + The recorded `capture_status` is still reported verbatim: the point is + to refuse the points, not to hide what was measured. + """ + run_failure = None + if run_status != RUN_RECOVERED: + run_failure = ( + "Run ended as {0}, not {1}; nothing captured against an " + "unresolved incident can be scored.".format( + run_status or "none", RUN_RECOVERED + ) + ) + failure = None if capture_status is None: failure = "no capture recorded" @@ -119,10 +148,10 @@ def score_scenario( points = 0 manual_points = 0 for criterion in CRITERIA: - if failure is not None: + if run_failure is not None or failure is not None: status = "FAIL" awarded = 0 - detail = ( + detail = run_failure or ( "Capture ended as {0}; the Agent produced no conclusion to " "score.".format(failure) ) @@ -156,6 +185,7 @@ def score_scenario( return { "scenario": scenario, "capture_status": capture_status, + "run_status": run_status, "timeline_events": len(timeline or []), "criteria": criteria, "points": points, @@ -200,9 +230,10 @@ def build_scorecard(state: LabState, evidence_root: Path) -> Dict[str, Any]: directory = Path(evidence_dir) timeline = _read_json(directory / "normalized-timeline.json") or [] review = _read_json(directory / REVIEW_FILE) - result = score_scenario(scenario, capture_status, timeline, review) + result = score_scenario( + scenario, capture_status, timeline, review, state.run_status(scenario) + ) result["evidence_dir"] = evidence_dir - result["run_status"] = state.run_status(scenario) scenarios[scenario] = result points = sum(result["points"] for result in scenarios.values()) @@ -255,8 +286,10 @@ def render_table(scorecard: Dict[str, Any]) -> str: result["verdict"], "{0}/{1}".format(result["points"], result["max_points"]), _cell( - "capture={0} manual={1}".format( - result["capture_status"] or "none", result["manual_points"] + "run={0} capture={1} manual={2}".format( + result["run_status"] or "none", + result["capture_status"] or "none", + result["manual_points"], ) ), ) diff --git a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py index 55ff761..dd5ce81 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py +++ b/monitor/sre-agent-event-lab/scripts/tests/lab_script_harness.py @@ -244,7 +244,30 @@ def _az_stub_source(log_path, state_dir): """ -def _lab_python_stub_source(log_path, capture_timeline, pillow_importable=True): +BEGIN_RUN_PROBE = r""" +# Records whether the evidence directory already exists at the moment +# `begin-run` is called, so a test can assert the ordering -- admit the run +# first, create the directory only once it was admitted -- rather than just +# the end state, which looks identical either way. +if [[ "$*" == *begin-run* ]]; then + probe_last="" + for probe_arg in "$@"; do probe_last="$probe_arg"; done + if [[ -d "${probe_last}" ]]; then + printf 'exists\t%s\n' "${probe_last}" >> "PROBE_PATH" + else + printf 'absent\t%s\n' "${probe_last}" >> "PROBE_PATH" + fi +fi +""" + + +def _begin_run_probe(probe_path): + return BEGIN_RUN_PROBE.replace("PROBE_PATH", str(probe_path)) + + +def _lab_python_stub_source( + log_path, capture_timeline, pillow_importable=True, probe_path=None +): """A fake `${LAB_ROOT}/app/.venv/bin/python`. Only the two scripts that would reach the SRE Agent data plane or write @@ -260,8 +283,10 @@ def _lab_python_stub_source(log_path, capture_timeline, pillow_importable=True): packages. """ pil_exit = 0 if pillow_importable else 1 + probe = _begin_run_probe(probe_path) if probe_path else "" return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" +{probe} if [[ "${{1:-}}" == "-c" && "${{2:-}}" == *PIL* ]]; then exit {pil_exit} fi @@ -296,10 +321,12 @@ def _lab_python_stub_source(log_path, capture_timeline, pillow_importable=True): """ -def _python3_stub_source(log_path): +def _python3_stub_source(log_path, probe_path=None): """A fake `python3`: `loadgen.py` is faked, everything else is real.""" + probe = _begin_run_probe(probe_path) if probe_path else "" return f"""#!/usr/bin/env bash printf '%s\\n' "$*" >> "{log_path}" +{probe} case "${{1:-}}" in *loadgen.py) exit 0 ;; *) exec "{REAL_PYTHON}" "$@" ;; @@ -354,6 +381,7 @@ def make_lab( azd_log = tmp_path / "azd-calls.log" python_log = tmp_path / "python-calls.log" lab_python_log = tmp_path / "lab-python-calls.log" + begin_run_probe = tmp_path / "begin-run-probe.log" write_executable(bin_dir / "az", _az_stub_source(az_log, state_dir)) write_azd_stub( @@ -362,24 +390,47 @@ def make_lab( missing_key_mode, azd_log, ) - write_executable(bin_dir / "python3", _python3_stub_source(python_log)) + write_executable( + bin_dir / "python3", _python3_stub_source(python_log, begin_run_probe) + ) venv_bin = lab / "app" / ".venv" / "bin" if venv_present: venv_bin.mkdir(parents=True) write_executable( venv_bin / "python", - _lab_python_stub_source(lab_python_log, capture_timeline, pillow_importable), + _lab_python_stub_source( + lab_python_log, capture_timeline, pillow_importable, begin_run_probe + ), ) workdir = tmp_path / "elsewhere" workdir.mkdir() - return LabRun(lab, bin_dir, workdir, az_log, azd_log, lab_python_log, state_dir) + return LabRun( + lab, + bin_dir, + workdir, + az_log, + azd_log, + lab_python_log, + state_dir, + begin_run_probe, + ) class LabRun: - def __init__(self, lab, bin_dir, workdir, az_log, azd_log, lab_python_log, state_dir): + def __init__( + self, + lab, + bin_dir, + workdir, + az_log, + azd_log, + lab_python_log, + state_dir, + begin_run_probe, + ): self.lab = lab self.bin_dir = bin_dir self.workdir = workdir @@ -387,6 +438,7 @@ def __init__(self, lab, bin_dir, workdir, az_log, azd_log, lab_python_log, state self.azd_log = azd_log self.lab_python_log = lab_python_log self.state_dir = state_dir + self.begin_run_probe = begin_run_probe def break_injection(self): """Make the *next* injecting `az containerapp update` fail. @@ -474,3 +526,17 @@ def state(self): def scenario_state(self, scenario): return self.state().get("scenarios", {}).get(scenario, {}) + + def begin_run_probes(self): + """One `(existed, evidence_dir)` pair per `begin-run` call, in order. + + `existed` says whether the evidence directory was already on disk + when the run was submitted for admission. + """ + if not self.begin_run_probe.exists(): + return [] + return [ + tuple(line.split("\t", 1)) + for line in self.begin_run_probe.read_text().splitlines() + if line.strip() + ] diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_briefing_assets.py b/monitor/sre-agent-event-lab/scripts/tests/test_briefing_assets.py index 6cd5929..cda832a 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_briefing_assets.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_briefing_assets.py @@ -1,4 +1,5 @@ import importlib.util +import re import sys from pathlib import Path @@ -7,6 +8,8 @@ MODULE_PATH = Path(__file__).parents[1] / "render_briefing_assets.py" +UUID_PATTERN = re.compile(r"\b[0-9a-fA-F]{8}-(?:[0-9a-fA-F]{4}-){3}[0-9a-fA-F]{12}\b") + def load_module(): sys.path.insert(0, str(MODULE_PATH.parent)) @@ -87,6 +90,11 @@ def test_public_assets_do_not_expose_sensitive_identifiers(tmp_path): serialized = "\n".join( path.read_text(errors="ignore") for path in tmp_path.glob("*.svg") ) - assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in serialized + # Any GUID at all: a rendered asset is published, and a subscription, + # tenant or resource ID leaked into one is the same disclosure whoever + # ran the lab. Forbidding the shape also keeps this test from having to + # name a real subscription in order to prove it is absent. + leaked = UUID_PATTERN.search(serialized) + assert leaked is None, leaked.group() if leaked else "" assert "sig=" not in serialized assert "Thread status:" not in serialized diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_common.py b/monitor/sre-agent-event-lab/scripts/tests/test_common.py index fec6d22..020fa76 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_common.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_common.py @@ -1,3 +1,5 @@ +import re +import shutil import subprocess from pathlib import Path @@ -13,6 +15,10 @@ CAPTURE_SCENARIO_SH = Path(__file__).parents[1] / "capture-scenario.sh" BASELINE_SH = Path(__file__).parents[1] / "baseline.sh" +BASH = shutil.which("bash") or "/bin/bash" + +UUID_PATTERN = re.compile(r"\b[0-9a-fA-F]{8}-(?:[0-9a-fA-F]{4}-){3}[0-9a-fA-F]{12}\b") + LEGACY_RESOURCE_GROUP_FLAG = "--legacy-delete-resource-group" REQUIRED_ENV = { @@ -82,10 +88,12 @@ def test_common_does_not_expose_personal_subscription_display_name(): assert "SUBSCRIPTION_NAME" not in script assert "ME-MngEnvMCAP310512-inhwanhwang-3" not in script assert "inhwanhwang" not in script - # The one true fixed subscription this task removed: common.sh must - # never hardcode a subscription ID again, it must resolve one via - # load_lab_config (explicit env > azd env > default) instead. - assert "95933ae5-0201-4a21-a1fc-8051a7437982" not in script + # common.sh must never hardcode *any* subscription ID again -- it + # resolves one via load_lab_config (explicit env > azd env > default). + # The shape is forbidden rather than the one value this task removed, + # so the guard also catches the next person's subscription, and so the + # test does not have to restate a real subscription ID to forbid it. + assert UUID_PATTERN.search(script) is None, UUID_PATTERN.search(script).group() def test_verify_subscription_reports_only_subscription_id_on_mismatch(tmp_path): @@ -373,3 +381,67 @@ def test_scenario_query_capture_cleanup_scripts_are_exercised_as_programs(): f"{script_name} has no execution test" ) + + +# --- The evidence directory is a name first, a directory second ------------- + + +def run_in_throwaway_lab(tmp_path, command): + """Source a copy of `common.sh` whose `EVIDENCE_ROOT` is disposable. + + `EVIDENCE_ROOT` is `readonly` and derived from the script's own + location, so the only way to exercise the directory helpers without + writing into the repository's real `evidence/` is to source a copy that + lives somewhere else. + """ + lab = tmp_path / "lab" + (lab / "scripts").mkdir(parents=True, exist_ok=True) + (lab / "evidence").mkdir(parents=True, exist_ok=True) + shutil.copy(str(COMMON_SH), str(lab / "scripts" / "common.sh")) + return subprocess.run( + [BASH, "-c", 'source "{0}"\n{1}\n'.format(lab / "scripts" / "common.sh", command)], + capture_output=True, + text=True, + ), lab / "evidence" + + +def test_evidence_dir_path_names_a_directory_without_creating_it(tmp_path): + """`run-scenario.sh` needs the evidence path *before* it asks + `lab_state.py` to admit the run, because the path is what it registers. + Creating the directory at that point left an empty `sN-/` + behind whenever the run was then refused -- litter an operator has to + tell apart from a real attempt while reading `evidence/`. So naming and + creating are separate steps: this one only names. + """ + result, evidence_root = run_in_throwaway_lab( + tmp_path, 'printf "%s" "$(evidence_dir_path s1)"' + ) + + assert result.returncode == 0, result.stderr + named = Path(result.stdout.strip()) + assert named.parent == evidence_root + assert re.fullmatch(r"s1-\d{8}T\d{6}Z", named.name), named.name + assert not named.exists(), ( + "naming an evidence directory must not create it: {0}".format(named) + ) + assert list(evidence_root.iterdir()) == [], ( + "a refused run must leave nothing behind in evidence/" + ) + + +def test_create_evidence_dir_still_creates_what_it_names(tmp_path): + """`baseline.sh` writes into the directory immediately, so the eager + helper must keep working -- the split adds a step, it does not move the + responsibility.""" + result, evidence_root = run_in_throwaway_lab( + tmp_path, + 'directory="$(create_evidence_dir baseline)"; ' + '[[ -d "${directory}" ]] || exit 1; ' + 'printf "%s" "${directory}"', + ) + + assert result.returncode == 0, result.stderr + created = Path(result.stdout.strip()) + assert created.is_dir() + assert created.parent == evidence_root + assert re.fullmatch(r"baseline-\d{8}T\d{6}Z", created.name), created.name diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py index e27a350..870d05b 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_guides.py @@ -117,8 +117,8 @@ "AZURE_CLIENT_SECRET=", ) -FIXED_SUBSCRIPTION_ID = "95933ae5-0201-4a21-a1fc-8051a7437982" FIXED_RESOURCE_GROUP = "rg-sre-agent-event-lab-krc" +UUID_PATTERN = re.compile(r"\b[0-9a-fA-F]{8}-(?:[0-9a-fA-F]{4}-){3}[0-9a-fA-F]{12}\b") def guide_paths(): @@ -354,7 +354,7 @@ def test_validation_results_opens_by_dating_itself_to_the_pre_azd_lab(): assert note, "validation-results.md opens with no note about what it records" assert "azd" in note assert re.search(r"(이전|예전|과거)", note), note - assert FIXED_RESOURCE_GROUP in note or FIXED_SUBSCRIPTION_ID in note, note + assert FIXED_RESOURCE_GROUP in note, note assert re.search(r"(현재|지금).{0,40}(실습|lab)", note), note @@ -826,12 +826,49 @@ def test_docs_never_show_a_credential_value(): def test_docs_do_not_pin_the_original_subscription_or_resource_group(): + """No subscription ID at all, not just the one the original validation + run used: the guides are followed against whatever subscription the + reader owns, and a GUID in the prose is either someone else's or a + disclosure. Forbidding the shape also keeps this file from having to + restate a real subscription ID in order to prove it is gone. + """ for path in all_docs() + [RUNBOOK]: text = path.read_text() - assert FIXED_SUBSCRIPTION_ID not in text, path.name + leaked = UUID_PATTERN.search(text) + assert leaked is None, "{0}: {1}".format(path.name, leaked.group()) assert FIXED_RESOURCE_GROUP not in text, path.name +def test_the_guides_state_that_one_unfinished_run_stops_every_scenario(): + """The S1 guide told operators that nothing stops two scenarios from + overlapping and that keeping them apart was their own discipline. It is + enforced now -- a scenario that is `running` or `failed` refuses every + other scenario's run, not just the next one -- and a guide that still + describes the old, unenforced rule teaches an operator to misread the + refusal they will actually hit. + """ + text = (GUIDES / "02-scenario-s1.md").read_text() + + assert "잠금이 없" not in text, ( + "the S1 guide still says nothing stops two scenarios overlapping" + ) + assert re.search(r"(running|failed|실행 중|실패)", text), text[:400] + assert "mark-failed" in text, ( + "the guide must name the command that ends a run stuck at `running`" + ) + + +def test_the_later_scenario_guides_name_the_lab_wide_precondition(): + """S2 and S3 list their own predecessor's recovery and capture. Neither + mentioned the condition that actually stops them most often after a + re-run: some *other* scenario left unfinished.""" + for name in ("03-scenario-s2.md", "04-scenario-s3.md"): + text = (GUIDES / name).read_text() + conditions = text.split("## 시작 조건", 1)[1].split("\n## ", 1)[0] + assert re.search(r"(다른 시나리오|나머지 시나리오)", conditions), name + assert re.search(r"(running|failed|실행 중|실패)", conditions), name + + def test_runbook_scopes_itself_to_the_provisioned_resource_group(): text = RUNBOOK.read_text() diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py index 238b767..ce67929 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_scripts.py @@ -8,6 +8,7 @@ Azure operation. """ import json +from pathlib import Path import pytest @@ -421,6 +422,143 @@ def test_a_rerun_that_dies_early_retires_the_previous_success(tmp_path, break_th assert "lab.sh run" in scored.stderr +# --- One unfinished run stops the whole lab, not just the next scenario ---- + + +def finish_scenario(lab_run, scenario): + """Run and capture one scenario end to end, the way an operator does.""" + run_result = lab_run.run("run-scenario.sh", [scenario], env=BOUNDED_WAITS) + assert run_result.returncode == 0, run_result.stdout + run_result.stderr + capture = lab_run.run("capture-scenario.sh", [scenario]) + assert capture.returncode == 0, capture.stdout + capture.stderr + assert lab_run.scenario_state(scenario)["capture_status"] == "conclusion" + + +def test_a_broken_s1_rerun_stops_s3_although_s2_is_still_captured(tmp_path): + """The gap this closes, end to end. + + All three scenarios run and capture cleanly, then S1 is re-run and the + re-run dies before it can record an outcome. S1 is now `running` or + `failed` -- its fault may still be live in the shared Container App -- + but S2's entry is untouched, still `recovered` + `conclusion`. The + ordered rules only look one scenario back, so `run-scenario.sh s3` read + S2's stale success and was admitted: a third fault injected on top of an + incident nobody had resolved, and two captures that can no longer be + told apart. + """ + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + for scenario in ("s1", "s2", "s3"): + finish_scenario(lab_run, scenario) + lab_run.break_injection() + rerun = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert rerun.returncode != 0, rerun.stdout + assert lab_run.scenario_state("s1")["run_status"] in ("running", "failed") + assert lab_run.scenario_state("s2")["capture_status"] == "conclusion" + az_before = lab_run.az_calls() + + blocked = lab_run.run("run-scenario.sh", ["s3"], env=BOUNDED_WAITS) + + assert blocked.returncode != 0, blocked.stdout + assert "s1" in blocked.stderr + new_calls = lab_run.az_calls()[len(az_before):] + assert "containerapp update" not in new_calls, new_calls + assert "role assignment delete" not in new_calls, ( + f"a refused run injected S3's fault anyway: {new_calls!r}" + ) + assert not sorted((lab_run.lab / "evidence").glob("s3-*"))[1:], ( + "a refused run must not leave a second S3 evidence directory behind" + ) + + +def test_a_refused_run_leaves_no_evidence_directory_behind(tmp_path): + """The evidence directory is registered with the run, so its path has to + exist as a string before `begin-run` -- but the directory itself must + only be created once the run was admitted. Otherwise every refusal + litters `evidence/` with an empty `sN-/` that reads exactly + like an attempt that ran and produced nothing. + """ + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + finish_scenario(lab_run, "s1") + lab_run.break_injection() + assert lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS).returncode != 0 + before = sorted(path.name for path in (lab_run.lab / "evidence").glob("s2-*")) + + blocked = lab_run.run("run-scenario.sh", ["s2"], env=BOUNDED_WAITS) + + assert blocked.returncode != 0, blocked.stdout + after = sorted(path.name for path in (lab_run.lab / "evidence").glob("s2-*")) + assert after == before, f"a refused run created {set(after) - set(before)}" + + +def test_the_evidence_directory_is_created_only_after_the_run_is_admitted(tmp_path): + """Ordering, observed at the moment it matters: when `begin-run` is + called the directory must not exist yet, and by the time the run does + its work it must.""" + lab_run = make_lab(tmp_path) + lab_run.seed_state() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode == 0, result.stdout + result.stderr + probes = lab_run.begin_run_probes() + assert probes, "begin-run was never called" + assert [existed for existed, _ in probes] == ["absent"], probes + registered = lab_run.scenario_state("s1")["evidence_dir"] + assert probes[0][1] == registered, ( + "the path registered with the run must be the one that was created" + ) + assert (lab_run.lab / registered).is_dir() or Path(registered).is_dir() + + +def test_a_running_scenario_cannot_be_started_a_second_time(tmp_path): + """A run left `running` -- a Ctrl-C, a crashed terminal -- must not be + restarted blindly: two live injections of the same fault leave neither + capture readable. The operator has to record how the first one ended.""" + lab_run = make_lab(tmp_path) + lab_run.seed_state( + scenarios={"s1": {"run_status": "running", "started_at": "2026-08-14T00:00:00Z"}} + ) + az_before = lab_run.az_calls() + + result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + + assert result.returncode != 0, result.stdout + assert "running" in result.stderr + assert "mark-failed s1" in result.stderr + new_calls = lab_run.az_calls()[len(az_before):] + assert "containerapp update" not in new_calls, new_calls + + +def test_capture_scenario_reports_a_refused_record_without_losing_evidence(tmp_path): + """`record-capture` now refuses a conclusion for a run that did not + recover. The capture pipeline has already written real files by then, so + the failure must say so and name where they are -- not exit on an + unexplained non-zero from a command substitution.""" + lab_run = make_lab(tmp_path) + lab_run.write_agent_setup() + lab_run.seed_state() + run_result = lab_run.run("run-scenario.sh", ["s1"], env=BOUNDED_WAITS) + assert run_result.returncode == 0, run_result.stderr + document = json.loads(lab_run.state_path.read_text()) + document["scenarios"]["s1"]["run_status"] = "failed" + lab_run.state_path.write_text(json.dumps(document)) + + result = lab_run.run("capture-scenario.sh", ["s1"]) + + assert result.returncode != 0, result.stdout + assert "recovered" in result.stderr + evidence_dir = lab_run.scenario_state("s1")["evidence_dir"] + assert evidence_dir in result.stderr, ( + f"the operator must be told the raw evidence survived: {result.stderr!r}" + ) + assert (Path(evidence_dir) / "normalized-timeline.json").is_file() + assert "capture_status" not in lab_run.scenario_state("s1") + + def test_a_started_run_is_recorded_before_the_fault_is_injected(tmp_path): """Ordering is the whole point: the attempt must be persisted *before* the first destructive Azure call, because that call is what can fail diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py index 4e7d3b8..c4ebd9d 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_lab_state.py @@ -104,8 +104,25 @@ def test_s1_runs_once_baseline_and_acknowledgement_are_recorded(tmp_path): def test_s2_requires_s1_recovery_even_when_s1_was_captured(tmp_path): - state = ready_for_s1(tmp_path / "state.json") - state.record_capture("s1", "conclusion") + """A `conclusion` next to a run that never recovered can only come from + a hand-edited or half-written state file -- `record_capture` refuses to + write one -- so it is stated as a file here. Reading it must still + refuse S2: the ordered rule asks for `s1_recovered`, and a capture can + never stand in for it. + """ + path = tmp_path / "state.json" + path.write_text( + json.dumps( + { + "stages": { + "baseline_passed": {"at": "2026-08-14T00:00:00Z"}, + "agent_setup_acknowledged": {"at": "2026-08-14T00:01:00Z"}, + }, + "scenarios": {"s1": {"capture_status": "conclusion"}}, + } + ) + ) + state = LabState(path) with pytest.raises(InvalidTransition, match="s1_recovered"): state.require_run("s2") @@ -135,7 +152,6 @@ def test_the_full_ordered_sequence_is_allowed(tmp_path): def test_a_failed_run_does_not_satisfy_the_next_scenario(tmp_path): state = ready_for_s1(tmp_path / "state.json") state.mark_failed("s1", str(tmp_path / "s1"), reason="alert never resolved") - state.record_capture("s1", "conclusion") assert state.run_status("s1") == "failed" with pytest.raises(InvalidTransition, match="s1_recovered"): @@ -382,6 +398,309 @@ def test_require_run_rejects_an_unknown_scenario(tmp_path): state.require_run("s9") +# --- One unfinished run blocks the whole lab, not just the next scenario --- + + +def completed_lab(path, evidence_root): + """A lab whose three scenarios all ran, recovered and captured. + + Built through the real API, in order, so the precondition is a state + the lab can actually reach -- the state an operator holds the moment + every scenario has produced a conclusion. + """ + state = ready_for_s1(path) + for scenario in lab_state.SCENARIOS: + directory = str(evidence_root / "{0}-first".format(scenario)) + state.begin_run(scenario, directory) + state.mark_recovered(scenario, directory) + state.record_capture(scenario, "conclusion", directory) + return state + + +def break_rerun(state, scenario, evidence_dir, run_status): + """Re-run `scenario` and leave it unfinished, `running` or `failed`.""" + state.begin_run(scenario, str(evidence_dir)) + if run_status == "failed": + state.mark_failed(scenario, str(evidence_dir), reason="injection rejected") + assert state.run_status(scenario) == run_status + return state + + +@pytest.mark.parametrize("run_status", ("running", "failed")) +def test_a_broken_rerun_of_s1_blocks_s3_although_s2_is_still_captured( + tmp_path, run_status +): + """The residual gap: the ordered rules only look *backwards* one step. + + A finished lab re-runs S1; the re-run dies before it can record an + outcome, or records a failure. S1 is then `running`/`failed` -- its + fault may still be live in the shared workload -- but S3's own + prerequisites (`s2_recovered`, `s2_captured`) are untouched, so S3 was + admitted and injected a third fault on top of an incident nobody + resolved. Every scenario shares one Container App, so an unfinished + run has to block *every* other scenario, not only the next one. + """ + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s1", tmp_path / "s1-second", run_status) + + with pytest.raises(InvalidTransition) as refusal: + state.require_run("s3") + + message = str(refusal.value) + assert "s1" in message + assert run_status in message + + +@pytest.mark.parametrize("run_status", ("running", "failed")) +@pytest.mark.parametrize("blocked", ("s2", "s3")) +def test_an_unfinished_s1_blocks_s2_and_s3(tmp_path, run_status, blocked): + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s1", tmp_path / "s1-second", run_status) + + with pytest.raises(InvalidTransition, match="s1"): + state.require_run(blocked) + + +@pytest.mark.parametrize("run_status", ("running", "failed")) +@pytest.mark.parametrize("blocked", ("s1", "s3")) +def test_an_unfinished_s2_blocks_s1_and_s3(tmp_path, run_status, blocked): + """S1 is *earlier* than S2 and no ordered rule mentions it, so nothing + but the global gate can stop an operator from re-running S1 while S2's + fault is still injected.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s2", tmp_path / "s2-second", run_status) + + with pytest.raises(InvalidTransition, match="s2"): + state.require_run(blocked) + + +@pytest.mark.parametrize("run_status", ("running", "failed")) +@pytest.mark.parametrize("blocked", ("s1", "s2")) +def test_an_unfinished_s3_blocks_s1_and_s2(tmp_path, run_status, blocked): + """The case no ordered rule can reach at all: S3 is the last scenario, + so its unfinished run is invisible to every prerequisite list.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s3", tmp_path / "s3-second", run_status) + + with pytest.raises(InvalidTransition) as refusal: + state.require_run(blocked) + + message = str(refusal.value) + assert "s3" in message + assert run_status in message + + +def test_a_running_scenario_cannot_be_started_again(tmp_path): + """Two concurrent runs of one scenario overlap two injections of the + same fault in one workload; the second one's evidence cannot be read.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + state.begin_run("s1", str(tmp_path / "s1-second")) + + with pytest.raises(InvalidTransition, match="running"): + state.require_run("s1") + + +def test_a_failed_scenario_may_be_rerun_to_fix_itself(tmp_path): + """Re-running the failed scenario is the documented remedy, so the gate + must never block the one command that clears it.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s1", tmp_path / "s1-second", "failed") + + state.require_run("s1") # must not raise + state.begin_run("s1", str(tmp_path / "s1-third")) + + assert state.run_status("s1") == "running" + + +def test_two_failed_runs_at_once_still_leave_a_way_out(tmp_path): + """The gate must not be able to lock the lab. + + Two scenarios `failed` at the same time is unreachable through the API + -- S2 cannot start until S1 recovered and was captured, and re-running + S1 is refused while S2 is failed -- but `state.json` is an editable + file, and a rule that has no exit in *any* state it can be put into is + a rule that eventually needs the file deleted. Repairing the earliest + unfinished scenario is therefore always allowed, so working the list + from the top always terminates. + """ + path = tmp_path / "state.json" + completed_lab(path, tmp_path) + document = json.loads(path.read_text()) + for scenario in ("s1", "s2"): + document["scenarios"][scenario]["run_status"] = "failed" + document["scenarios"][scenario].pop("capture_status", None) + path.write_text(json.dumps(document)) + state = lab_state.LabState(path) + + state.require_run("s1") # the earliest failure may always be repaired + + with pytest.raises(lab_state.InvalidTransition, match="s1"): + state.require_run("s2") + with pytest.raises(lab_state.InvalidTransition, match="s1"): + state.require_run("s3") + + +def test_the_refusal_lists_every_blocker_earliest_first(tmp_path): + """One name is not enough when two runs are unfinished: an operator who + clears only the one they were told about hits the next refusal blind. + The list is ordered, so the first entry is the one to deal with. + + S1 has no ordered prerequisite naming S2 or S3, so this refusal can + only come from the gate. + """ + path = tmp_path / "state.json" + completed_lab(path, tmp_path) + document = json.loads(path.read_text()) + document["scenarios"]["s2"]["run_status"] = "running" + document["scenarios"]["s3"]["run_status"] = "failed" + path.write_text(json.dumps(document)) + state = lab_state.LabState(path) + + with pytest.raises(lab_state.InvalidTransition) as error: + state.require_run("s1") + + message = str(error.value) + assert message.index("s2") < message.index("s3"), message + assert "running" in message and "failed" in message + assert "mark-failed s2" in message + assert "lab.sh run s3" in message + + +def test_a_repair_is_refused_while_an_earlier_run_is_still_running(tmp_path): + """`running` is not a repairable state -- nobody knows whether that run + is still working -- so it blocks the later failure's repair too, and the + remedy named is the one that ends it.""" + path = tmp_path / "state.json" + completed_lab(path, tmp_path) + document = json.loads(path.read_text()) + document["scenarios"]["s1"]["run_status"] = "running" + document["scenarios"]["s2"]["run_status"] = "failed" + path.write_text(json.dumps(document)) + state = lab_state.LabState(path) + + with pytest.raises(lab_state.InvalidTransition, match="mark-failed s1"): + state.require_run("s2") + + +def test_the_refusal_names_the_blocking_scenario_its_status_and_a_remedy(tmp_path): + """S1 has no ordered prerequisite that mentions S2, so this refusal can + only come from the global gate: it has to carry everything the ordered + message would have carried.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s2", tmp_path / "s2-second", "failed") + + with pytest.raises(InvalidTransition) as refusal: + state.require_run("s1") + + message = str(refusal.value) + assert "s2" in message + assert "failed" in message + assert "lab.sh run s2" in message, message + + +def test_the_refusal_for_a_running_scenario_names_how_to_end_it(tmp_path): + """A run that died leaves `running` behind for ever unless the operator + is told the one command that records how it ended.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + state.begin_run("s1", str(tmp_path / "s1-second")) + + with pytest.raises(InvalidTransition) as refusal: + state.require_run("s3") + + message = str(refusal.value) + assert "s1" in message + assert "running" in message + assert "mark-failed s1" in message, message + + +def test_the_ordered_remedy_never_tells_an_operator_to_restart_a_running_run(tmp_path): + """S2's ordered refusal names `s1_recovered`, whose remedy is normally + "run S1 again". While S1 is still running that command is refused too, + so the remedy has to change with the run's status instead of sending + the operator into a second refusal. + """ + state = completed_lab(tmp_path / "state.json", tmp_path) + state.begin_run("s1", str(tmp_path / "s1-second")) + + with pytest.raises(InvalidTransition) as refusal: + state.require_run("s2") + + message = str(refusal.value) + assert "s1_recovered" in message + assert "lab.sh run s1" not in message, message + assert "mark-failed s1" in message, message + + +def test_a_finished_lab_still_allows_a_normal_rerun_of_every_scenario(tmp_path): + """`recovered` is a finished run: the gate must not turn the ordinary + "run it again to collect a second capture" flow into a refusal.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + + for scenario in lab_state.SCENARIOS: + state.require_run(scenario) # must not raise + + +def test_a_rerun_that_recovers_again_reopens_every_other_scenario(tmp_path): + """The gate has to *clear*: once the unfinished run recovers, the lab + goes back to being governed by the ordered rules alone.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s1", tmp_path / "s1-second", "failed") + with pytest.raises(InvalidTransition): + state.require_run("s3") + + state.begin_run("s1", str(tmp_path / "s1-third")) + state.mark_recovered("s1", str(tmp_path / "s1-third")) + + state.require_run("s3") # must not raise + + +def test_a_recovered_but_uncaptured_run_blocks_only_the_ordered_rules(tmp_path): + """The rule this pins down, deliberately unchanged: `recovered` is a + *finished* run -- the fault is reverted and the alert closed -- so it + never trips the unfinished-run gate. A re-run of S1 that recovered but + has not been captured yet therefore blocks S2 through the existing + ordered rule (`s1_captured`) and leaves S3, whose own prerequisites + (`s2_recovered`, `s2_captured`) are untouched, admitted. The ordered + rules keep looking exactly one scenario back; the new gate adds nothing + here because nothing is still running or failed. + """ + state = completed_lab(tmp_path / "state.json", tmp_path) + state.begin_run("s1", str(tmp_path / "s1-second")) + state.mark_recovered("s1", str(tmp_path / "s1-second")) + assert state.capture_status("s1") is None + + with pytest.raises(InvalidTransition, match="s1_captured"): + state.require_run("s2") + state.require_run("s3") # must not raise + state.require_run("s1") # must not raise + + +def test_begin_run_is_refused_while_another_scenario_is_unfinished(tmp_path): + """`begin_run` is what runs just before the injection, so it must refuse + exactly what `require_run` refuses -- and record nothing when it does.""" + state = completed_lab(tmp_path / "state.json", tmp_path) + break_rerun(state, "s1", tmp_path / "s1-second", "failed") + + with pytest.raises(InvalidTransition, match="s1"): + state.begin_run("s3", str(tmp_path / "s3-second")) + + assert state.run_status("s3") == "recovered" + assert state.capture_status("s3") == "conclusion" + assert state.evidence_dir("s3") == str(tmp_path / "s3-first") + + +def test_the_gate_reads_what_is_on_disk_not_this_process_memory(tmp_path): + path = tmp_path / "state.json" + state = completed_lab(path, tmp_path) + break_rerun(state, "s2", tmp_path / "s2-second", "failed") + + with pytest.raises(InvalidTransition) as refusal: + LabState(path).require_run("s1") + + assert "s2" in str(refusal.value) + assert "failed" in str(refusal.value) + + # --- Never promoting missing Agent output to success ------------------------ @@ -415,6 +734,134 @@ def test_record_capture_rejects_an_unknown_terminal_state(tmp_path): state.record_capture("s1", "looks-fine") +# --- Only a recovered run may be credited with a conclusion ----------------- + + +@pytest.mark.parametrize( + "run_status", (None, "running", "failed"), ids=("none", "running", "failed") +) +def test_a_conclusion_cannot_be_recorded_against_a_run_that_never_recovered( + tmp_path, run_status +): + """`conclusion` is the one capture outcome that unblocks the next + scenario and earns points, so it may only ever describe a run that + actually recovered. A conclusion recorded while the scenario is + `running`, `failed`, or has no recorded run at all belongs to an + incident nobody resolved -- and the state file is the only place that + can refuse it, because the timeline it came from looks identical. + """ + state = ready_for_s1(tmp_path / "state.json") + if run_status == "running": + state.begin_run("s1", str(tmp_path / "s1")) + elif run_status == "failed": + state.mark_failed("s1", str(tmp_path / "s1"), reason="alert never resolved") + + with pytest.raises(InvalidTransition) as refusal: + state.record_capture("s1", "conclusion", str(tmp_path / "s1")) + + assert "conclusion" in str(refusal.value) + assert (run_status or "none") in str(refusal.value) + assert state.capture_status("s1") is None + assert not state.is_successful_capture("s1") + + +@pytest.mark.parametrize( + "run_status, expected, forbidden", + ( + ("running", "mark-failed s1", "lab.sh run s1"), + ("failed", "lab.sh run s1", "mark-failed s1"), + (None, "lab.sh run s1", "mark-failed s1"), + ), + ids=("running", "failed", "none"), +) +def test_the_capture_refusal_names_a_command_that_is_not_itself_refused( + tmp_path, run_status, expected, forbidden +): + """Telling an operator to re-run a scenario that is still `running` + sends them straight into the unfinished-run gate. The remedy has to be + the one command the state actually admits.""" + state = ready_for_s1(tmp_path / "state.json") + if run_status == "running": + state.begin_run("s1", str(tmp_path / "s1")) + elif run_status == "failed": + state.mark_failed("s1", str(tmp_path / "s1"), reason="alert never resolved") + + with pytest.raises(InvalidTransition) as refusal: + state.record_capture("s1", "conclusion", str(tmp_path / "s1")) + + assert expected in str(refusal.value) + assert forbidden not in str(refusal.value) + + +@pytest.mark.parametrize( + "missing_status", + ("thread-not-created", "investigation-missing", "conclusion-missing"), +) +@pytest.mark.parametrize( + "run_status", (None, "running", "failed"), ids=("none", "running", "failed") +) +def test_a_missing_marker_is_still_recorded_for_any_run_status( + tmp_path, missing_status, run_status +): + """Diagnostic honesty runs the other way: what the Agent failed to + produce is worth recording whatever the run did, because it is the + measurement an operator has to read. None of these markers can unblock + anything or earn a point, so recording them is free of risk. + """ + state = ready_for_s1(tmp_path / "state.json") + if run_status == "running": + state.begin_run("s1", str(tmp_path / "s1")) + elif run_status == "failed": + state.mark_failed("s1", str(tmp_path / "s1"), reason="alert never resolved") + + state.record_capture("s1", missing_status, str(tmp_path / "s1")) + + assert state.capture_status("s1") == missing_status + assert not state.is_successful_capture("s1") + assert not state.has("s1_captured") + + +def test_a_conclusion_is_recorded_once_the_run_recovered(tmp_path): + state = ready_for_s1(tmp_path / "state.json") + state.begin_run("s1", str(tmp_path / "s1")) + state.mark_recovered("s1", str(tmp_path / "s1")) + + state.record_capture("s1", "conclusion", str(tmp_path / "s1")) + + assert state.is_successful_capture("s1") + + +def test_cli_record_capture_refuses_a_conclusion_for_a_failed_run(tmp_path): + path = tmp_path / "state.json" + evidence_dir = tmp_path / "s1-20260814T000000Z" + evidence_dir.mkdir() + (evidence_dir / "normalized-timeline.json").write_text( + json.dumps( + [{"state": "alert-fired"}, {"state": "thread-created"}, {"state": "conclusion"}] + ) + ) + run_cli(path, ["mark", "baseline_passed"]) + run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n") + run_cli(path, ["mark-failed", "s1", str(evidence_dir), "--reason", "no alert"]) + + result = run_cli( + path, + [ + "record-capture", + "s1", + "--timeline", + str(evidence_dir / "normalized-timeline.json"), + "--evidence-dir", + str(evidence_dir), + ], + ) + + assert result.returncode == 1 + assert "Traceback" not in result.stderr + assert "failed" in result.stderr + assert "capture_status" not in json.loads(path.read_text())["scenarios"]["s1"] + + @pytest.mark.parametrize( "events, expected", ( @@ -785,6 +1232,59 @@ def test_cli_begin_run_without_the_prerequisites_records_nothing(tmp_path): assert not path.exists() or "s1" not in json.loads(path.read_text())["scenarios"] +def seed_completed_lab(path, evidence_root): + """Drive the CLI through a whole lab: three recovered, captured runs.""" + assert run_cli(path, ["mark", "baseline_passed"]).returncode == 0 + assert run_cli(path, ["acknowledge-agent"], stdin="acknowledge\n").returncode == 0 + for scenario in lab_state.SCENARIOS: + directory = str(evidence_root / "{0}-first".format(scenario)) + assert run_cli(path, ["begin-run", scenario, directory]).returncode == 0 + assert run_cli(path, ["mark-recovered", scenario, directory]).returncode == 0 + recorded = run_cli( + path, + ["record-capture", scenario, "--status", "conclusion", "--evidence-dir", directory], + ) + assert recorded.returncode == 0, recorded.stderr + return path + + +def test_cli_require_run_refuses_every_scenario_while_one_run_is_unfinished(tmp_path): + """S3 is the last scenario, so no ordered rule mentions it: only the + global gate can refuse S1 and S2 while its run is unfinished.""" + path = seed_completed_lab(tmp_path / "state.json", tmp_path) + assert run_cli(path, ["begin-run", "s3", str(tmp_path / "s3-second")]).returncode == 0 + assert ( + run_cli( + path, ["mark-failed", "s3", str(tmp_path / "s3-second"), "--reason", "rejected"] + ).returncode + == 0 + ) + + for scenario in ("s1", "s2"): + refused = run_cli(path, ["require-run", scenario]) + assert refused.returncode == 1, refused.stdout + assert "s3" in refused.stderr + assert "failed" in refused.stderr + assert "lab.sh run s3" in refused.stderr + assert "Traceback" not in refused.stderr + + assert run_cli(path, ["require-run", "s3"]).returncode == 0 + + +def test_cli_begin_run_refuses_while_another_scenario_is_still_running(tmp_path): + path = seed_completed_lab(tmp_path / "state.json", tmp_path) + assert run_cli(path, ["begin-run", "s3", str(tmp_path / "s3-second")]).returncode == 0 + before = path.read_text() + + refused = run_cli(path, ["begin-run", "s1", str(tmp_path / "s1-second")]) + + assert refused.returncode == 1, refused.stdout + assert "s3" in refused.stderr + assert "running" in refused.stderr + assert "Traceback" not in refused.stderr + assert path.read_text() == before + + def test_cli_begin_run_refuses_a_state_file_from_another_environment(tmp_path): """A run must never start against a state file another lab wrote: the binding check has to fail before the attempt is recorded, so the file diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py b/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py index c9c349e..75c930e 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py @@ -81,3 +81,68 @@ def test_relocated_lab_docs_drop_internal_workflow_directives(): ).read_text() for marker in ("승인된 설계", "보고서 반영", "검증 부록"): assert marker not in content, (name, marker) + + +# --- The real subscription ID belongs in exactly one file ------------------- + +LAB_ROOT = REPO_ROOT / "monitor" / "sre-agent-event-lab" +VALIDATION_RESULTS = LAB_ROOT / "validation-results.md" +UUID_PATTERN = re.compile( + r"\b[0-9a-fA-F]{8}-(?:[0-9a-fA-F]{4}-){3}[0-9a-fA-F]{12}\b" +) + + +def recorded_subscription_id(): + """The subscription the historical validation run used. + + `validation-results.md` is the record of one real run on one real + subscription, so that ID legitimately appears there -- and nowhere + else. Reading it from that file is what lets every other guard below + forbid it without restating it. + """ + header = VALIDATION_RESULTS.read_text().split("\n## ", 1)[0] + found = UUID_PATTERN.findall(header) + assert found, "validation-results.md no longer records a subscription ID" + return found[0] + + +def test_no_test_source_restates_the_real_subscription_id(): + """A test that proves a file does not leak the subscription by writing + the subscription into the test is self-defeating: the value is in the + repository either way, and every copy is one more place to miss when it + has to change. The guards state the *shape* they forbid instead, and + the fixtures use obvious dummies. + """ + subscription_id = recorded_subscription_id() + test_sources = sorted( + [ + *(LAB_ROOT / "scripts" / "tests").glob("*.py"), + *(LAB_ROOT / "infra" / "tests").glob("*.py"), + *(LAB_ROOT / "app" / "tests").glob("*.py"), + ] + ) + assert test_sources + + offenders = [ + str(path.relative_to(LAB_ROOT)) + for path in test_sources + if subscription_id in path.read_text() + ] + + assert offenders == [], ( + "test sources still hardcode the real validation subscription ID; " + f"forbid the UUID shape instead: {offenders}" + ) + + +def test_test_fixtures_never_reuse_the_real_subscription_id(): + """The dummy IDs the harnesses inject must be provably not the real + one, so a fixture can never accidentally target a live subscription.""" + subscription_id = recorded_subscription_id() + from lab_script_harness import SUBSCRIPTION_ID as harness_subscription_id + from test_common import REQUIRED_ENV + + dummies = (harness_subscription_id, REQUIRED_ENV["AZURE_SUBSCRIPTION_ID"]) + for dummy in dummies: + assert UUID_PATTERN.fullmatch(dummy), dummy + assert dummy != subscription_id diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_score.py b/monitor/sre-agent-event-lab/scripts/tests/test_score.py index 6f0477d..48366a0 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_score.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_score.py @@ -45,6 +45,7 @@ def load_module(path, name): {"state": "investigating"}, {"state": "conclusion-missing"}, ] +RECOVERED = lab_state.RUN_RECOVERED FULL_REVIEW = { "impact_scope": {"met": True, "detail": "Named the Container App and both routes."}, "direct_cause": {"met": True, "detail": "Named FAILURE_MODE=http500."}, @@ -94,7 +95,9 @@ def test_the_rubric_is_the_documented_ten_point_one(): def test_a_fully_reviewed_conclusion_earns_every_point(): - result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, FULL_REVIEW) + result = score.score_scenario( + "s1", "conclusion", CONCLUSION_TIMELINE, FULL_REVIEW, RECOVERED + ) assert result["points"] == 10 assert result["max_points"] == 10 @@ -106,7 +109,7 @@ def test_an_unmet_criterion_costs_exactly_its_points(): review = dict(FULL_REVIEW) review["direct_cause"] = {"met": False, "detail": "Named a symptom, not the cause."} - result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review) + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review, RECOVERED) assert criteria_by_id(result)["direct_cause"]["status"] == "FAIL" assert criteria_by_id(result)["direct_cause"]["points"] == 0 @@ -131,7 +134,7 @@ def test_any_manual_criterion_keeps_the_verdict_incomplete(): def test_a_missing_review_is_manual_and_awards_nothing(): - result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, None) + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, None, RECOVERED) assert result["points"] == 0 assert result["manual_points"] == 10 @@ -144,7 +147,7 @@ def test_a_missing_review_is_manual_and_awards_nothing(): def test_one_unavailable_field_is_manual_while_the_rest_score(): review = {key: value for key, value in FULL_REVIEW.items() if key != "uncertainty"} - result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review) + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review, RECOVERED) assert criteria_by_id(result)["uncertainty"]["status"] == "MANUAL" assert criteria_by_id(result)["uncertainty"]["points"] == 0 @@ -158,7 +161,7 @@ def test_an_unusable_judgement_is_manual_never_a_pass(unusable): review = dict(FULL_REVIEW) review["impact_scope"] = unusable - result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review) + result = score.score_scenario("s1", "conclusion", CONCLUSION_TIMELINE, review, RECOVERED) assert criteria_by_id(result)["impact_scope"]["status"] == "MANUAL" assert criteria_by_id(result)["impact_scope"]["points"] == 0 @@ -171,7 +174,9 @@ def test_an_unusable_judgement_is_manual_never_a_pass(unusable): "capture_status", ("thread-not-created", "investigation-missing", "conclusion-missing") ) def test_a_missing_agent_output_scores_zero_and_stays_visible(capture_status): - result = score.score_scenario("s1", capture_status, MISSING_CONCLUSION_TIMELINE, FULL_REVIEW) + result = score.score_scenario( + "s1", capture_status, MISSING_CONCLUSION_TIMELINE, FULL_REVIEW, RECOVERED + ) assert result["points"] == 0 assert result["manual_points"] == 0 @@ -186,13 +191,88 @@ def test_a_scenario_that_was_never_captured_is_not_manual(): """No capture at all is a known failure, not an unknown: reporting it `MANUAL` would let an unrun scenario wait forever for a human instead of failing the lab.""" - result = score.score_scenario("s3", None, [], None) + result = score.score_scenario("s3", None, [], None, RECOVERED) assert result["verdict"] == "FAIL" assert result["points"] == 0 assert result["manual_points"] == 0 +# --- A conclusion is worthless unless its run recovered -------------------- + + +@pytest.mark.parametrize( + "run_status", (None, "running", "failed"), ids=("none", "running", "failed") +) +def test_a_conclusion_from_a_run_that_never_recovered_scores_zero(run_status): + """Defence in depth. `lab_state.record_capture` already refuses to write + a `conclusion` next to a run that is not `recovered`, but the scorer + reads a file an operator can edit and an interrupted write can truncate. + A stale conclusion left over from a superseded attempt must never be + scored: the run status is checked here too, independently, and forces + the whole scenario to FAIL with zero points. + """ + result = score.score_scenario( + "s1", "conclusion", CONCLUSION_TIMELINE, FULL_REVIEW, run_status + ) + + assert result["points"] == 0 + assert result["manual_points"] == 0 + assert result["verdict"] == "FAIL" + assert result["run_status"] == run_status + assert result["capture_status"] == "conclusion", ( + "the recorded capture status must stay visible, not be rewritten" + ) + for item in result["criteria"]: + assert item["status"] == "FAIL" + assert item["points"] == 0 + assert (run_status or "none") in item["detail"], item["detail"] + + +def test_the_scorecard_fails_a_stale_conclusion_left_by_a_broken_rerun(tmp_path): + """The end-to-end shape of the same defence: a state file that still + carries a `conclusion` for a scenario whose run says `failed` -- exactly + what a half-written or hand-edited file looks like -- must score FAIL, + not the ten points its review would otherwise earn. + """ + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, FULL_REVIEW) + for scenario in ("s1", "s2", "s3") + } + make_state(tmp_path, directories) + state_path = tmp_path / "state.json" + document = json.loads(state_path.read_text()) + document["scenarios"]["s1"]["run_status"] = "failed" + state_path.write_text(json.dumps(document)) + + scorecard = score.build_scorecard(lab_state.LabState(state_path), tmp_path) + + assert scorecard["scenarios"]["s1"]["verdict"] == "FAIL" + assert scorecard["scenarios"]["s1"]["points"] == 0 + assert scorecard["scenarios"]["s1"]["run_status"] == "failed" + assert scorecard["overall"]["verdict"] == "FAIL" + assert scorecard["overall"]["points"] == 20 + + +def test_the_table_shows_the_run_status_behind_a_forced_failure(tmp_path): + directories = { + scenario: write_evidence(tmp_path, scenario, CONCLUSION_TIMELINE, FULL_REVIEW) + for scenario in ("s1", "s2", "s3") + } + make_state(tmp_path, directories) + state_path = tmp_path / "state.json" + document = json.loads(state_path.read_text()) + document["scenarios"]["s1"]["run_status"] = "running" + state_path.write_text(json.dumps(document)) + + table = score.render_table( + score.build_scorecard(lab_state.LabState(state_path), tmp_path) + ) + + assert "s1\tTOTAL\tFAIL\t0/10" in table + assert "run=running" in table + + # --- Scorecard -------------------------------------------------------------- From f1d2f230fd256d3683e834ec2834b510fed6c0ef Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 09:37:59 +0900 Subject: [PATCH 25/26] test(sre-lab): allow redacted validation history Keep privacy guards effective without requiring the historical validation report to retain a real subscription ID. Document that unrecovered runs always score zero. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../sre-agent-event-lab/guides/05-results.md | 2 + .../scripts/tests/test_privacy.py | 42 ++++++++++++------- 2 files changed, 29 insertions(+), 15 deletions(-) diff --git a/monitor/sre-agent-event-lab/guides/05-results.md b/monitor/sre-agent-event-lab/guides/05-results.md index 510dc6d..b5aa7ed 100644 --- a/monitor/sre-agent-event-lab/guides/05-results.md +++ b/monitor/sre-agent-event-lab/guides/05-results.md @@ -33,6 +33,8 @@ cd monitor/sre-agent-event-lab - Partial: 5~7점 - Fail: 4점 이하 +시나리오의 `run_status`가 `recovered`가 아니면 채점기는 capture 내용과 관계없이 모든 항목을 `FAIL`·0점으로 기록합니다. + `impact_scope`는 alert 규칙 자체의 scope를 그대로 옮겨 적는 것으로는 채워지지 않습니다. 이 랩의 모든 alert는 Log Analytics workspace scope입니다 (`infra/alerts.bicep`의 `scopes`/`targetResourceTypes: diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py b/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py index 75c930e..c1b4b51 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py @@ -83,8 +83,6 @@ def test_relocated_lab_docs_drop_internal_workflow_directives(): assert marker not in content, (name, marker) -# --- The real subscription ID belongs in exactly one file ------------------- - LAB_ROOT = REPO_ROOT / "monitor" / "sre-agent-event-lab" VALIDATION_RESULTS = LAB_ROOT / "validation-results.md" UUID_PATTERN = re.compile( @@ -92,18 +90,29 @@ def test_relocated_lab_docs_drop_internal_workflow_directives(): ) -def recorded_subscription_id(): - """The subscription the historical validation run used. +def extract_subscription_id(header_text): + """Return the first GUID-shaped token, or None when the text is redacted.""" + found = UUID_PATTERN.findall(header_text) + return found[0] if found else None - `validation-results.md` is the record of one real run on one real - subscription, so that ID legitimately appears there -- and nowhere - else. Reading it from that file is what lets every other guard below - forbid it without restating it. - """ + +def recorded_subscription_id(): + """Return the historical subscription ID when it is still recorded.""" header = VALIDATION_RESULTS.read_text().split("\n## ", 1)[0] - found = UUID_PATTERN.findall(header) - assert found, "validation-results.md no longer records a subscription ID" - return found[0] + return extract_subscription_id(header) + + +def test_extract_subscription_id_accepts_a_redacted_historical_doc(): + redacted = "# Validation Results\n\nSubscription: [REDACTED]\n" + assert extract_subscription_id(redacted) is None + + empty_header = "# Validation Results\n\nNo subscription recorded.\n" + assert extract_subscription_id(empty_header) is None + + +def test_extract_subscription_id_finds_a_present_guid(): + header = "# Validation Results\n\nSubscription: 11111111-2222-3333-4444-555555555555\n" + assert extract_subscription_id(header) == "11111111-2222-3333-4444-555555555555" def test_no_test_source_restates_the_real_subscription_id(): @@ -114,6 +123,9 @@ def test_no_test_source_restates_the_real_subscription_id(): the fixtures use obvious dummies. """ subscription_id = recorded_subscription_id() + if subscription_id is None: + return + test_sources = sorted( [ *(LAB_ROOT / "scripts" / "tests").glob("*.py"), @@ -136,8 +148,7 @@ def test_no_test_source_restates_the_real_subscription_id(): def test_test_fixtures_never_reuse_the_real_subscription_id(): - """The dummy IDs the harnesses inject must be provably not the real - one, so a fixture can never accidentally target a live subscription.""" + """Fixture IDs must be GUID-shaped and never match a recorded live ID.""" subscription_id = recorded_subscription_id() from lab_script_harness import SUBSCRIPTION_ID as harness_subscription_id from test_common import REQUIRED_ENV @@ -145,4 +156,5 @@ def test_test_fixtures_never_reuse_the_real_subscription_id(): dummies = (harness_subscription_id, REQUIRED_ENV["AZURE_SUBSCRIPTION_ID"]) for dummy in dummies: assert UUID_PATTERN.fullmatch(dummy), dummy - assert dummy != subscription_id + if subscription_id is not None: + assert dummy != subscription_id From 24fe5c61006f123d6ea4a83eddd640be34b3b477 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 15 Aug 2026 09:48:55 +0900 Subject: [PATCH 26/26] fix(sre-lab): address Copilot teardown privacy review Redact the historical subscription ID and make cancellation semantics explicit: predown can remove recorded external roles before the azd confirmation prompt, while postdown preserves image settings unless deletion succeeds. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../scripts/cleanup-external.sh | 16 +++++++--------- .../scripts/tests/test_privacy.py | 4 ++++ .../sre-agent-event-lab/validation-results.md | 4 ++-- 3 files changed, 13 insertions(+), 11 deletions(-) diff --git a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh index 2db3dda..23b80cb 100755 --- a/monitor/sre-agent-event-lab/scripts/cleanup-external.sh +++ b/monitor/sre-agent-event-lab/scripts/cleanup-external.sh @@ -33,15 +33,13 @@ # re-running `azd down` works. # # Why the image values are cleared in `postdown` and not here: `predown` -# runs before azd asks the operator to confirm the deletion. An operator who -# answers "no" keeps every resource, so clearing SRE_CONTAINER_IMAGE / -# SRE_IMAGE_TAG at that point would break an environment nothing happened -# to. `postdown` is the documented counterpart hook (azd command hooks: -# pre/post for restore, provision, package, deploy, publish, up and down), -# and azd runs a post hook only after the action itself succeeded -- -# `HooksRunner.Invoke` returns early when the action fails -# (cli/azd/pkg/ext/hooks_runner.go), so a cancelled or failed `azd down` -# leaves the recorded image values alone. +# runs before azd asks the operator to confirm the deletion. At that point +# this hook may already have removed the recorded external role assignments. +# Cancelling keeps the resource group and image values, but does not restore +# those assignments; README documents how to recreate and re-record them. +# `postdown` runs only after the action itself succeeds, so a cancelled or +# failed `azd down` leaves SRE_CONTAINER_IMAGE / SRE_IMAGE_TAG available for +# the retained workload. set -euo pipefail CLEANUP_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" diff --git a/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py b/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py index c1b4b51..aea4a01 100644 --- a/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py +++ b/monitor/sre-agent-event-lab/scripts/tests/test_privacy.py @@ -115,6 +115,10 @@ def test_extract_subscription_id_finds_a_present_guid(): assert extract_subscription_id(header) == "11111111-2222-3333-4444-555555555555" +def test_historical_validation_header_redacts_subscription_id(): + assert recorded_subscription_id() is None + + def test_no_test_source_restates_the_real_subscription_id(): """A test that proves a file does not leak the subscription by writing the subscription into the test is self-defeating: the value is in the diff --git a/monitor/sre-agent-event-lab/validation-results.md b/monitor/sre-agent-event-lab/validation-results.md index 276f4e4..dd62be3 100644 --- a/monitor/sre-agent-event-lab/validation-results.md +++ b/monitor/sre-agent-event-lab/validation-results.md @@ -4,9 +4,9 @@ > > 이 문서는 S1/S2/S3 시나리오에서 측정한 수치, timeline, evidence, 한계를 정리한다. > -> **기록 시점 주의.** 아래 결과는 azd 재구성 이전에 손으로 구축한 실습(2026-08-12)의 측정치다. 여기 적힌 구독 ID, `rg-sre-agent-event-lab-krc` 같은 리소스 그룹과 리소스 이름, Action Group + Logic App bridge는 모두 그때의 환경이고, 현재 실습의 `azd up`은 이 이름들을 만들지 않는다. 지금 실행하는 절차와 실제로 배포되는 구성은 [README](README.md)와 [guides/](guides/)를 따르고, 이 문서는 그 절차로 무엇을 관찰할 수 있었는지 보여 주는 과거 기록으로 읽는다. +> **기록 시점 주의.** 아래 결과는 azd 재구성 이전에 손으로 구축한 실습(2026-08-12)의 측정치다. 여기 적힌 `rg-sre-agent-event-lab-krc` 같은 리소스 그룹과 리소스 이름, Action Group + Logic App bridge는 모두 그때의 환경이고, 현재 실습의 `azd up`은 이 이름들을 만들지 않는다. 지금 실행하는 절차와 실제로 배포되는 구성은 [README](README.md)와 [guides/](guides/)를 따르고, 이 문서는 그 절차로 무엇을 관찰할 수 있었는지 보여 주는 과거 기록으로 읽는다. -- 실행일: 2026-08-12 | 리전: Korea Central | 구독: `95933ae5-0201-4a21-a1fc-8051a7437982` +- 실행일: 2026-08-12 | 리전: Korea Central | 구독: 비식별화 - 목표: Azure Monitor 경고를 Azure SRE Agent가 자동 수신해 원인과 안전한 완화책을 올바르게 도출하는지 실증 - 테스트베드: Azure Container Apps + Application Insights + Log Analytics + Azure Storage - Agent 모드: Review