From 9f2b50a9da64d7b1b050603cbb4663b761406723 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 16:39:34 +0900 Subject: [PATCH 01/14] =?UTF-8?q?docs(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20=EC=8B=A4=EC=8A=B5=EA=B3=BC=20Airflow=20Parquet=20=EC=A0=81?= =?UTF-8?q?=EC=9E=AC=20=EC=B8=A1=EC=A0=95=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 프라이빗 엔드포인트 뒤 Azure DocumentDB(M30)에서 PyMongo change stream의 지원 범위, 재개, 중복과 지연을 AKS consumer로 측정한 lab 추가 - AKS 위 Airflow DAG가 5분마다 change stream을 읽어 ADLS Gen2에 Parquet으로 쓰는 구성의 지연, 재시도 중복, 400 MB 활성 change log를 넘긴 재개와 파일 크기를 측정한 research 문서와 구성 다이어그램 추가 - maxAwaitTimeMS가 getMore 실행 제한으로 적용되어 큰 백로그 재개가 code 50으로 반복 실패하는 동작을 기록하고 내보내기 코드의 기본값에서 제외 - runnable sample(python-aks): Bicep, consumer·생성기·검증기, 내보내기와 DAG - taxonomy에 azure-documentdb 서비스와 airflow, mongodb 기술 추가 Co-Authored-By: Claude Opus 5.5 --- docs-taxonomy.yml | 3 + .../airflow-parquet/images/architecture.svg | 61 ++++ .../change-streams/airflow-parquet/index.md | 231 +++++++++++++ .../azure-documentdb/change-streams/index.md | 239 +++++++++++++ .../samples/python-aks/README.md | 144 ++++++++ .../samples/python-aks/airflow/Dockerfile | 2 + .../airflow/dags/change_stream_to_parquet.py | 54 +++ .../samples/python-aks/airflow/values.yaml | 32 ++ .../samples/python-aks/app/Dockerfile | 8 + .../samples/python-aks/app/common.py | 59 ++++ .../samples/python-aks/app/consumer.py | 163 +++++++++ .../samples/python-aks/app/generator.py | 109 ++++++ .../samples/python-aks/app/lake_export.py | 179 ++++++++++ .../samples/python-aks/app/probe.py | 322 ++++++++++++++++++ .../samples/python-aks/app/requirements.txt | 5 + .../samples/python-aks/app/verify.py | 103 ++++++ .../samples/python-aks/app/verify_lake.py | 107 ++++++ .../samples/python-aks/infra/lake.bicep | 158 +++++++++ .../samples/python-aks/infra/main.bicep | 213 ++++++++++++ .../samples/python-aks/k8s/airflow-rbac.yaml | 34 ++ .../samples/python-aks/k8s/consumer.yaml | 49 +++ .../samples/python-aks/k8s/job.yaml | 44 +++ .../samples/python-aks/k8s/lake.yaml | 55 +++ .../samples/python-aks/k8s/toolbox.yaml | 22 ++ .../samples/python-aks/sample.yml | 6 + tests/docs/test_faceted_discovery.py | 10 +- tests/docs/test_topics.py | 4 +- 27 files changed, 2409 insertions(+), 7 deletions(-) create mode 100644 docs/services/azure-documentdb/change-streams/airflow-parquet/images/architecture.svg create mode 100644 docs/services/azure-documentdb/change-streams/airflow-parquet/index.md create mode 100644 docs/services/azure-documentdb/change-streams/index.md create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/README.md create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/Dockerfile create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/Dockerfile create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/common.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/job.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/toolbox.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml diff --git a/docs-taxonomy.yml b/docs-taxonomy.yml index 0f3ce8e7..9e20a8b3 100644 --- a/docs-taxonomy.yml +++ b/docs-taxonomy.yml @@ -35,6 +35,7 @@ services: azure-container-apps: Azure Container Apps azure-cosmos-db: Azure Cosmos DB azure-database-for-mysql: Azure Database for MySQL + azure-documentdb: Azure DocumentDB azure-functions: Azure Functions azure-hdinsight: Azure HDInsight azure-kubernetes-service: Azure Kubernetes Service @@ -47,6 +48,7 @@ services: microsoft-foundry: Microsoft Foundry technologies: + airflow: Apache Airflow azure-cli: Azure CLI bicep: Bicep cosmos-db: Azure Cosmos DB @@ -59,6 +61,7 @@ technologies: kafka: Apache Kafka kubernetes: Kubernetes mcp: Model Context Protocol + mongodb: MongoDB mysql: MySQL nodejs: Node.js open-telemetry: OpenTelemetry diff --git a/docs/services/azure-documentdb/change-streams/airflow-parquet/images/architecture.svg b/docs/services/azure-documentdb/change-streams/airflow-parquet/images/architecture.svg new file mode 100644 index 00000000..12711e3e --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/airflow-parquet/images/architecture.svg @@ -0,0 +1,61 @@ + + AKS 위 Airflow DAG가 Azure DocumentDB change stream을 ADLS Gen2 Parquet으로 적재하는 구성 + AKS의 airflow 네임스페이스에서 Airflow scheduler가 5분마다 cslab 네임스페이스에 내보내기 파드를 만든다. 내보내기 파드는 프라이빗 엔드포인트를 거쳐 Azure DocumentDB의 change stream을 읽고, blob과 dfs 프라이빗 엔드포인트를 거쳐 ADLS Gen2에 Parquet 청크와 checkpoint를 쓴다. 스토리지 인증은 Microsoft Entra ID의 관리 ID와 워크로드 ID로 한다. + + + + Airflow DAG로 change stream을 Parquet에 적재하는 구성 + + + VNet + + AKS · Standard_D4s_v6 4대 · 워크로드 ID + + + namespace airflow + + Airflow scheduler + LocalExecutor · 5분 주기 + + PostgreSQL + 메타데이터 + + + namespace cslab + + 내보내기 파드 + lake_export.py · 실행마다 생성 + 서비스 계정 cs-lake + + + KubernetesPodOperator + + + PE · MongoCluster + + PE · blob, dfs + + + 읽기 + + + 쓰기 + + + + Azure DocumentDB + M30 · shard 1개 · 서버 7.0 + 공용 액세스 끔 · change stream + + + ADLS Gen2 · 파일 시스템 cdc + orders/dt=…/hour=…/part-*.parquet + _checkpoints/orders.json + 공용 액세스·공유 키 끔 + + + Microsoft Entra ID + 관리 ID · federated credential + + 워크로드 ID 토큰 + diff --git a/docs/services/azure-documentdb/change-streams/airflow-parquet/index.md b/docs/services/azure-documentdb/change-streams/airflow-parquet/index.md new file mode 100644 index 00000000..238ca6a8 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/airflow-parquet/index.md @@ -0,0 +1,231 @@ +--- +title: Azure DocumentDB change stream을 Airflow DAG로 ADLS Gen2 Parquet에 적재하기 +description: AKS 위 Airflow DAG가 5분마다 PyMongo로 change stream을 읽어 프라이빗 엔드포인트 뒤 ADLS Gen2에 Parquet으로 쓸 때의 지연, 재시도 중복, 400 MB 활성 change log를 넘긴 재개와 파일 크기를 측정했습니다. +document_type: research +services: [azure-documentdb, azure-storage, azure-kubernetes-service] +technologies: [python, mongodb, airflow, kubernetes] +tags: [evaluate, build, storage] +status: current +verification_status: verified +sources_checked_at: 2026-10-02 +published_at: 2026-10-02 +topic_order: 1 +official_sources: + - title: Change streams in Azure DocumentDB + url: https://learn.microsoft.com/azure/documentdb/change-streams + - title: Use private endpoints for Azure Storage + url: https://learn.microsoft.com/azure/storage/common/storage-private-endpoints + - title: Use Microsoft Entra Workload ID with Azure Kubernetes Service (AKS) + url: https://learn.microsoft.com/azure/aks/workload-identity-overview +--- + +# Azure DocumentDB change stream을 Airflow DAG로 ADLS Gen2 Parquet에 적재하기 + +[상위 실습](../index.md)은 change stream을 상시 consumer로 읽었습니다. 이 문서는 +AKS에 설치한 Airflow가 5분마다 파드를 띄워 그동안 쌓인 이벤트를 ADLS Gen2에 +Parquet 파일로 쓰는 구성을 실제 구독에서 측정한 결과입니다. 확인한 질문은 다음과 +같습니다. + +- 이벤트가 파일로 저장되기까지 얼마나 걸리는가 +- 업로드와 checkpoint 사이에서 파드가 죽으면 중복 행이 생기는가 +- DAG를 멈춘 사이 400 MB 활성 change log보다 많이 쌓여도 이어 읽을 수 있는가 +- Parquet 파일은 얼마나 커지는가 + +배포와 실행 방법은 sample의 +[README](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md)와 +코드에 있습니다. + +## 구성 + +![AKS의 Airflow scheduler가 5분마다 내보내기 파드를 만들고, 파드가 프라이빗 엔드포인트를 거쳐 DocumentDB change stream을 읽어 ADLS Gen2에 Parquet 청크와 checkpoint를 쓰며 Entra ID 워크로드 ID로 인증하는 구성](images/architecture.svg) + +| 구성 요소 | 내용 | +| --- | --- | +| DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 서버 7.0.0 | +| AKS | Kubernetes 1.35, Standard_D4s_v6 노드 4개 | +| Airflow | Helm chart 1.22.0, Airflow 3.2.2, `LocalExecutor`, 차트에 포함된 PostgreSQL | +| DAG | `KubernetesPodOperator` 태스크 하나. 5분 주기, `max_active_runs=1`, 재시도 3회 | +| 내보내기 파드 | Python 3.12, PyMongo 4.18.2, pyarrow 25.0.1. 청크당 100,000건, snappy 압축 | +| ADLS Gen2 | Standard_LRS, 계층 구조 네임스페이스, 공용 액세스와 공유 키 끔 | + +스토리지에는 blob과 dfs 프라이빗 엔드포인트를 둘 다 만들었습니다. Learn은 Data +Lake Storage에서 dfs 엔드포인트만 만들면 blob 엔드포인트를 쓰는 작업이 실패할 수 +있다고 설명합니다. 파드는 `azure.workload.identity/use: "true"` 레이블이 있어야 +워크로드 ID 토큰을 받습니다. + +실행 한 번은 checkpoint의 resume token에서 재개해 실행을 시작한 시각까지의 이벤트를 +읽습니다. 100,000건마다 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. 청크 +파일 이름은 그 청크를 시작한 resume token의 해시라서 재시도는 같은 파일을 덮어씁니다. + +## 결과 요약 + +| 질문 | 결과 | +| --- | --- | +| 지연 | 초당 약 1,000건 부하에서 이벤트 기록부터 파일 저장까지 p50 156초, p99 302초, 최대 311초 | +| 재시도 중복 | 첫 청크 업로드 직후 프로세스를 죽인 뒤 재시도한 결과 230,000건 중 중복 0, 누락 0 | +| 활성 change log 초과 | 문서 본문 2.6 GB가 넘는 백로그를 누락·중복 없이 이어 읽음. 단 `maxAwaitTimeMS=1000`에서는 같은 위치에서 code 50으로 반복 실패 | +| 처리 속도 | 1 KB 문서는 초당 약 13,000–18,000건, 4 KB 문서 백로그는 초당 약 3,600건. 실행마다 파드 기동과 정리에 약 7초 | +| 파일 크기 | 이벤트에 실린 문서 크기와 거의 같음. 시험 데이터의 랜덤 채움 문자열은 snappy로 줄지 않았고 zstd로 다시 쓰면 71% 줄어듦 | + +모든 시나리오에서 이벤트가 빠짐없이 한 번씩 파일에 들어갔습니다. 검증은 +`verify_lake.py`가 Parquet 파일 전체를 생성기가 만든 이벤트와 대조했습니다. 중복은 +같은 resume token이 두 행 이상인 경우입니다. 저장 지연은 파일 last-modified(초 +단위)에서 이벤트 `wallTime`을 뺀 값입니다. + +## 5분 주기 지연 + +부하는 문서 780,000개, 1 KB 채움, 초당 1,000건 제한으로 31분 동안 실행했습니다. DAG는 +5분 주기로 계속 돌았습니다. 생성기가 실제로 낸 속도는 초당 958건이었습니다. 이벤트 +1,794,000건이 모두 파일에 들어갔고 중복, 누락, 문서별 순서 역전은 0건이었습니다. + +| 지표 | p50 | p95 | p99 | 최대 | +| --- | --- | --- | --- | --- | +| 이벤트 기록부터 파일 저장까지 | 156초 | 290초 | 302초 | 311초 | +| 이벤트 기록부터 파드가 읽기까지 | 154초 | 284초 | 296초 | 301초 | + +- 지연 분포는 주기 간격과 거의 같습니다. 직전 실행 직후에 기록된 이벤트는 약 5분을 + 기다리고 실행 직전에 기록된 이벤트는 몇 초 만에 저장됩니다. +- 실행 한 번은 이벤트 약 300,000건을 16–24초에 썼습니다. Airflow가 실행을 시작해 + 파드를 정리하기까지는 21–30초였습니다. 차이인 약 7초가 파드 예약, 기동, 연결과 + 정리에 쓰였습니다. +- 파일은 21개였고 대부분 100,000행, 약 69 MB였습니다. 실행 끝의 나머지 청크는 더 + 작았습니다. +- 생성기만 돌 때 클러스터 CPU는 1분 평균 약 23–25%였습니다. 내보내기가 도는 분에도 + 같은 범위여서 1분 단위에서는 읽기 부하가 드러나지 않았습니다. + +## 업로드 직후 장애와 재시도 + +DAG를 멈춘 상태에서 문서 100,000개(이벤트 230,000건)를 쓰고 첫 시도만 첫 청크 업로드 +직후 종료하도록 실행했습니다. 첫 시도는 checkpoint를 쓰기 전에 종료 코드 137로 +끝났습니다. Airflow가 30초 뒤 재시도했고 두 번째 시도는 같은 checkpoint에서 시작해 +첫 청크를 같은 경로에 덮어썼습니다. + +| 기대 이벤트 | 저장된 이벤트 | 중복 | 누락 | 파일 | +| --- | --- | --- | --- | --- | +| 230,000 | 230,000 | 0 | 0 | 3 | + +## 활성 change log를 넘긴 재개 + +DAG를 멈춘 상태에서 생성기가 31분 동안 문서 600,000개를 4 KB씩 채워 이벤트 +1,380,000건을 썼습니다. 문서 본문만 2.6 GB가 넘어 Learn이 설명하는 활성 change log +400 MB의 6배 이상입니다. 그동안 클러스터 스토리지 사용률은 약 30%에서 44%로 +올랐습니다. + +### maxAwaitTimeMS가 재개를 막음 + +DAG를 다시 켜자 첫 시도와 재시도 모두 약 13초 만에 실패했습니다. 청크는 하나도 +쓰지 못했고 checkpoint도 그대로였습니다. + +```text +pymongo.errors.ExecutionTimeout: Query exceeded command timeout of 1000ms +full error: {'ok': 0.0, 'code': 50, 'codeName': 'ExceededTimeLimit', ...} +``` + +당시 코드는 `max_await_time_ms=1000`으로 stream을 열었습니다. 같은 checkpoint에서 +읽기만 하는 프로브로 비교했습니다. + +| `maxAwaitTimeMS` | 결과 | +| --- | --- | +| 1000 | 58,000번째 이벤트 뒤 `getMore`에서 code 50으로 실패 | +| 지정 안 함 | 1,380,000건을 311초에 모두 읽음(초당 4,436건). 0.5초 넘는 `getMore` 65회, 최대 7.4초 | + +- 이 클러스터는 `maxAwaitTimeMS`를 새 이벤트를 기다리는 시간이 아니라 `getMore` + 전체의 실행 제한으로 적용했습니다. 그 시간을 넘긴 `getMore`는 빈 배치 대신 오류를 + 돌려주었습니다. +- 지정하지 않은 실행에서 처음 느려진 `getMore`도 58,000번째 이벤트 뒤였습니다(1.4초). + 느린 `getMore`는 그 뒤로도 불규칙하게 나타났습니다. +- 같은 위치에서 매번 실패하므로 Airflow 재시도로는 넘어가지 못합니다. 고치지 않으면 + 이후 주기 실행도 모두 같은 지점에서 실패합니다. +- 새 이벤트가 없을 때 `try_next()`는 값을 지정했을 때와 지정하지 않았을 때 모두 + 1.0초 뒤 `None`을 돌려주었습니다. 따라서 값을 빼도 실행 종료 조건은 그대로 + 동작합니다. + +### 수정 후 결과 + +`lake_export.py`가 기본적으로 `max_await_time_ms`를 지정하지 않도록 바꾼 뒤 Airflow의 +세 번째 시도가 같은 checkpoint에서 시작했습니다. + +| 지표 | 값 | +| --- | --- | +| 저장된 이벤트 | 1,380,000건(기대값과 같음), 중복 0, 누락 0, 문서별 순서 역전 0 | +| 내보내기 시간 | 380.5초(초당 약 3,600건). Airflow 태스크 전체 398초 | +| 파일 | 14개, 대부분 100,000행·약 374 MB, 합계 5.13 GB | +| 내보내기 파드 메모리 | 20초 간격 측정에서 최대 약 1.9 GiB | +| 클러스터 CPU(1분 평균) | 따라잡는 동안 15–32%, 직전 약 5% | + +바로 다음 주기 실행은 새 이벤트 0건으로 1.2초 만에 끝났습니다. + +## 파일 크기 + +백로그 재개 시험의 Parquet 합계 5.13 GB는 생성기가 쓴 문서 본문 2.6 GB의 약 두 배입니다. 가장 큰 +파일(100,000행, 374 MB)을 열어 원인을 확인했습니다. + +| 열 | 압축 전 | snappy 압축 후 | 비율 | +| --- | --- | --- | --- | +| `full_document` | 382.3 MB | 371.1 MB | 0.97 | +| 나머지 6개 열 합계 | 7.2 MB | 2.7 MB | 0.38 | + +- 파일 크기의 99%가 문서 본문입니다. 이 파일의 행은 insert 43,567건, update 47,720건, + delete 8,713건이었고 delete를 뺀 91,287행에 문서 전체가 들어 있었습니다. 문서는 + BSON 기준 평균 4,129바이트였습니다. +- change stream은 변경마다 문서 전체를 보냅니다. 이 클러스터는 update에도 + `fullDocument`를 넣습니다. 그래서 문서 하나가 insert와 update에 한 번씩 저장됩니다. + 문서가 실린 이벤트 약 1,260,000건 × 약 4.1 KB가 약 5.2 GB이고 Parquet 합계와 거의 + 같습니다. +- 크기 대부분은 생성기가 채운 4,000자 랜덤 16진 문자열입니다. 반복이 없어 snappy는 + 거의 줄이지 못했습니다. + +같은 파일을 다른 방식으로 다시 써 비교했습니다. + +| 방식 | 크기 | +| --- | --- | +| snappy(현재) | 374 MB | +| gzip | 210 MB | +| zstd level 3 | 107 MB | +| zstd level 9 | 111 MB | +| snappy, 채움 문자열 제거 | 4.8 MB | + +zstd는 insert와 update에 반복된 같은 채움 문자열까지 찾아 줄였습니다. 채움 문자열을 +빼면 100,000행이 4.8 MB여서 Parquet의 열 구조가 더하는 크기는 작습니다. 실제 문서는 +랜덤 문자열보다 잘 압축되므로 이번 숫자는 압축 측면의 최악에 가깝습니다. + +## 판단 + +- 몇 분 지연을 받아들일 수 있고 결과가 Parquet 파일이어야 하면 이 방식이 단순합니다. + 상시 파드가 없고 중간 메시지 계층도 없습니다. +- 정확히 한 번의 결과는 파일 이름과 checkpoint 순서로 얻습니다. checkpoint를 먼저 쓰고 + 파일을 나중에 쓰면 그 사이 장애로 이벤트를 잃습니다. +- `max_active_runs=1`과 ETag 조건은 둘 다 필요합니다. 첫째는 Airflow 안에서 실행이 + 겹치지 않게 하고 둘째는 수동 Job 같은 외부 실행이 checkpoint를 덮어쓰지 못하게 합니다. +- change stream을 여는 코드에 짧은 `maxAwaitTimeMS`를 주지 않습니다. 평소에는 문제가 + 없다가 백로그가 커진 뒤에야 code 50으로 드러나고 재시도로도 풀리지 않습니다. Learn의 + C# 예제도 `MaxAwaitTime`을 1초로 둡니다. 값을 꼭 줘야 하면 같은 클러스터에서 백로그 + 재개를 시험해 정합니다. +- DAG를 오래 멈추면 Learn이 말하는 재개 범위(최대 35일 또는 클러스터 초기화 시점 중 + 이른 쪽)를 넘을 수 있습니다. 멈춘 기간을 모니터링합니다. Learn은 보관된 로그를 + 처리하는 작업을 트래픽이 적은 시간에 하라고 권장합니다. 백로그 재개 시험에서 따라잡는 동안 + 클러스터 CPU가 15–32%로 올랐습니다. +- 파일은 이벤트 이력이므로 문서 하나가 여러 번 저장됩니다. 문서별 최신 상태만 필요하면 + 뒤 단계에서 `doc_id` 기준으로 합칩니다. update를 바뀐 필드만으로 줄이는 방법은 이 + 클러스터가 `updateDescription`을 돌려주지 않아 쓸 수 없습니다. +- 압축은 zstd를 먼저 검토합니다. 문서가 크면 청크 크기(`CS_CHUNK_EVENTS`)를 줄여 + 파일 크기와 파드 메모리를 맞춥니다. + +## 한계 + +- 단일 shard M30 클러스터, 컬렉션 하나, 측정 하루의 결과입니다. +- change log의 실제 크기는 조회할 수 없어 문서 본문 크기로 추정했습니다. +- `maxAwaitTimeMS`가 `getMore` 실행 제한으로 적용되는 동작은 Learn에 설명이 없습니다. + 이 클러스터에서 관측한 결과입니다. +- 압축 비교는 백로그 재개 시험의 파일 하나를 다시 써서 얻었습니다. zstd로 쓸 때의 시간과 CPU는 + 측정하지 않았습니다. +- 파일 저장 지연은 초 단위 last-modified로 계산했습니다. +- Parquet 스키마는 원본 문서를 JSON 문자열 열 하나로 둡니다. 분석 쿼리에서 열로 + 펼치는 작업은 다루지 않았습니다. +- 차트에 포함된 PostgreSQL은 시험용입니다. 운영에서는 외부 데이터베이스를 씁니다. + +## 공식 참고 자료 + +- [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) +- [Use private endpoints for Azure Storage](https://learn.microsoft.com/azure/storage/common/storage-private-endpoints) +- [Use Microsoft Entra Workload ID with AKS](https://learn.microsoft.com/azure/aks/workload-identity-overview) diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md new file mode 100644 index 00000000..295858e8 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -0,0 +1,239 @@ +--- +title: Azure DocumentDB change stream을 Python으로 AKS에서 검증하기 +description: 프라이빗 엔드포인트 뒤의 Azure DocumentDB(vCore) 클러스터에서 PyMongo change stream의 지원 범위, 재개, 중복 처리와 지연을 AKS consumer로 측정합니다. +document_type: lab +services: [azure-documentdb, azure-kubernetes-service] +technologies: [python, mongodb, kubernetes, bicep] +tags: [build, evaluate] +status: verified +verification_status: verified +sources_checked_at: 2026-10-02 +official_sources: + - title: Change streams in Azure DocumentDB + url: https://learn.microsoft.com/azure/documentdb/change-streams + - title: $changeStream (Azure DocumentDB aggregation operator) + url: https://learn.microsoft.com/documentdb/query/operators/aggregation/$changestream + - title: Use Azure Private Link in Azure DocumentDB + url: https://learn.microsoft.com/azure/documentdb/how-to-private-link + - title: Compute and storage configurations for Azure DocumentDB + url: https://learn.microsoft.com/azure/documentdb/compute-storage + - title: Release notes for Azure DocumentDB + url: https://learn.microsoft.com/azure/documentdb/release-notes + - title: Microsoft.DocumentDB mongoClusters (Bicep reference) + url: https://learn.microsoft.com/azure/templates/microsoft.documentdb/2026-06-01/mongoclusters + - title: AzureCosmosDB/changestream-driver-compatibility + url: https://github.com/AzureCosmosDB/changestream-driver-compatibility +last_verified: 2026-10-02 +review_cycle_days: 90 +estimated_time: 90m +cost: paid +cleanup_required: true +--- + +# Azure DocumentDB change stream을 Python으로 AKS에서 검증하기 + +Azure DocumentDB(이전 이름 Azure Cosmos DB for MongoDB vCore)는 MongoDB +change stream을 제공합니다. 다만 지원 범위가 MongoDB 서버와 다르므로 코드를 +옮기기 전에 실제 클러스터에서 확인해야 합니다. 이 실습은 PyMongo consumer를 AKS에 +띄우고 프라이빗 엔드포인트로 클러스터에 연결해 다음 네 가지를 측정합니다. + +- 어떤 옵션과 이벤트 필드가 실제로 동작하는가 +- 파드를 죽였다 살려도 이벤트가 빠지지 않는가 +- 중복이 생기는 지점은 어디이고 어떻게 흡수하는가 +- 초당 1,000건 쓰기에서 지연은 얼마인가 + +상시 consumer 대신 Airflow DAG가 주기마다 change stream을 읽어 ADLS Gen2에 +Parquet으로 쓰는 방식은 [Airflow DAG로 Parquet 적재](airflow-parquet/index.md)에서 +측정했습니다. + +## 목표 + +- 프라이빗 엔드포인트만 열린 DocumentDB 클러스터와 AKS를 Bicep으로 배포합니다. +- `probe.py`로 change stream 옵션별 동작을 확인합니다. +- 생성기가 만든 이벤트와 sink에 기록된 이벤트를 `verify.py`로 대조해 누락, + 중복, 순서 역전과 지연을 숫자로 남깁니다. + +## 사전 조건 + +- Azure 구독에서 리소스 그룹을 만들 수 있는 권한 +- Azure CLI, `kubectl`, `envsubst` +- 이 저장소의 [python-aks sample](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md) + +이 문서의 결과는 2026-10-02에 아래 환경에서 측정했습니다. + +| 항목 | 값 | +| --- | --- | +| 리전 | East US 2 | +| DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 고가용성 끔, 서버 버전 7.0.0 | +| 네트워크 | 공용 액세스 끔, 프라이빗 엔드포인트(`privatelink.mongocluster.cosmos.azure.com`, 포트 10260) | +| AKS | Kubernetes 1.35, Standard_D4s_v6 노드, Azure CNI overlay | +| 클라이언트 | Python 3.12, PyMongo 4.18.2 | + +## 비용과 안전 경계 + +- 클러스터, AKS 노드, ACR과 프라이빗 엔드포인트는 실행하는 동안 과금됩니다. + 측정이 끝나면 바로 리소스 그룹을 삭제합니다. +- 관리자 비밀번호와 연결 문자열은 Kubernetes Secret에만 넣습니다. sample의 + `.env`는 `.gitignore` 대상이며 커밋하지 않습니다. +- 생성기는 지정한 데이터베이스(`cslab`)의 `orders` 컬렉션에 직접 씁니다. + 운영 클러스터를 대상으로 실행하지 않습니다. +- 이 문서에 나오는 리소스 이름과 레지스트리 주소는 가상 값입니다. + +## 배포 + +sample 디렉터리에서 실행합니다. + +```bash +RG=rg-docdb-changestream +az group create -n $RG -l eastus2 +az deployment group create -g $RG -f infra/main.bicep \ + -p adminPassword="$ADMIN_PASSWORD" + +ACR=docdbcsacrexample +az acr build -r $ACR -t cslab:v1 app/ +az aks get-credentials -g $RG -n aks-docdbcs + +kubectl create namespace cslab +kubectl -n cslab create secret generic docdb --from-literal=uri="$MONGO_URI" +``` + +`main.bicep`은 VNet, `Microsoft.DocumentDB/mongoClusters` 클러스터(M30, shard 1개), +프라이빗 엔드포인트와 사설 DNS 영역, ACR, AKS를 만듭니다. 프라이빗 엔드포인트의 +group ID는 `MongoCluster`입니다. Learn은 프라이빗 엔드포인트로 연결할 때 +`mongodb+srv` 형식의 연결 문자열을 쓰라고 안내합니다. AKS 파드에서는 클러스터 +호스트 이름이 사설 DNS 영역을 거쳐 프라이빗 IP로 확인됩니다. + +감시할 컬렉션은 consumer보다 먼저 만듭니다. 컬렉션이 없으면 `watch()`가 +`NamespaceNotFound`(code 26)로 실패합니다. + +```bash +export IMAGE=docdbcsacrexample.azurecr.io/cslab:v1 +envsubst < k8s/toolbox.yaml | kubectl apply -f - +kubectl -n cslab exec cs-toolbox -- python -c \ + "from common import *; get_database(get_client('setup')).create_collection('orders')" +envsubst < k8s/consumer.yaml | kubectl apply -f - +``` + +## 시나리오 실행 + +모든 시나리오는 `k8s/job.yaml`에 `SCRIPT`와 매개변수를 넣어 Job으로 실행합니다. + +| 실행 | 스크립트와 매개변수 | 목적 | +| --- | --- | --- | +| probe | `SCRIPT=probe.py` | 옵션, 이벤트 필드, 재개 방식과 오류 코드 확인 | +| r1 | `generator.py`, `DOCS=10000` | 기본 정합성과 지연 | +| r2 | `generator.py`, `DOCS=100000`, 실행 중 consumer 파드 반복 삭제 | 재시작 후 누락 여부 | +| r3 | `generator.py`, `DOCS=100000`, consumer에 `FAULT_EXIT_AFTER_WRITE=200` 설정 | sink 쓰기와 checkpoint 사이에서 프로세스가 죽을 때의 중복 | +| r4 | `generator.py`, `RATE=1000`, 5분 | 초당 1,000건에서의 지연 | + +생성기는 문서마다 insert와 update를 한 번씩 실행합니다. 10번째 문서마다 +replace, 5번째 문서마다 delete를 더합니다. 따라서 문서 10,000개는 이벤트 +23,000건이 됩니다. 쓰기는 모두 `w=majority`입니다. + +```bash +JOB_NAME=gen-r4 SCRIPT=generator.py RUN_ID=r4 DOCS=300000 WORKERS=16 RATE=1000 \ + envsubst < k8s/job.yaml | kubectl apply -f - +# 생성기가 끝나고 consumer가 따라잡은 뒤 +JOB_NAME=verify-r4 SCRIPT=verify.py RUN_ID=r4 DOCS=0 WORKERS=0 RATE=0 \ + envsubst < k8s/job.yaml | kubectl apply -f - +kubectl -n cslab logs job/verify-r4 +``` + +consumer는 다음 방식으로 동작합니다. + +- `collection.watch(full_document="updateLookup", max_await_time_ms=1000)`로 열고 + `try_next()`로 읽습니다. +- 이벤트를 최대 200건씩 sink 컬렉션에 `bulk_write`한 뒤 resume token을 + `_cs_checkpoints`에 저장합니다. 대기 중에는 post-batch resume token을 저장합니다. +- sink는 resume token의 `_data`를 키로 upsert합니다. 같은 이벤트가 다시 오면 + 행을 추가하지 않고 `deliveries`를 1 올립니다. +- 지연은 consumer가 이벤트를 받은 시각에서 생성기가 문서에 기록한 + `updated_at` 또는 `created_at`을 뺀 값입니다. + +## 예상 결과 + +### 옵션과 이벤트 필드 + +Learn 문서에 나온 동작과 이번 클러스터에서 관찰한 동작을 구분했습니다. +관찰 결과는 M30, shard 1개, 서버 7.0.0 클러스터에서 2026-10-02에 확인한 값입니다. + +| 항목 | Learn 문서 | 관찰 결과 | +| --- | --- | --- | +| 이벤트 필드 | insert, update, delete 예시에 `_id`, `operationType`, `fullDocument`, `ns`, `documentKey` | `_id`, `operationType`, `fullDocument`, `ns`, `documentKey`, `wallTime`. `clusterTime`은 없음 | +| replace | 예시 없음 | `operationType: update`로 오고 `fullDocument`에 교체 후 문서 전체가 있음 | +| update의 `fullDocument` | 변경 후 문서 전체를 보여 주는 예시 | 옵션 없이도 포함됨. `updateLookup`, `whenAvailable`, `required` 모두 오류 없이 열림 | +| `updateDescription` | 별도 옵션 예시로 제시. 파이프라인 안의 update에서는 지원하지 않음 | 파이프라인이 있든 없든 반환되지 않음 | +| pre-image | 미리 보기. 지원 요청으로 클러스터에서 켜야 함 | `collMod`는 성공. `whenAvailable`은 `null`, `required`는 code 10065 오류. 지원 요청은 하지 않음 | +| 파이프라인 단계 | `$addFields`, `$match`, `$project`, `$set`, `$unset` | 다섯 개 모두 동작. 목록에 없는 `$replaceRoot`, `$redact`도 오류 없이 동작 | +| 감시 범위 | 컬렉션 예시 | `db.watch()`는 code 26. `client.watch()`에 `ns.db` 조건을 건 `$match`는 동작 | +| 재개 | `resumeAfter`, `startAt`, `startAtOperationTime` 지원 | `resume_after`, `start_after` 동작. 세션의 `operationTime`이 비어 있어 이를 쓴 `start_at_operation_time`은 실패. 현재 시각에서 10분 뺀 `Timestamp`는 동작 | +| 잘못된 resume token | 언급 없음 | code 2(BadValue) | +| `showExpandedEvents` | 지원하지 않음 | code 115(CommandNotSupported) | +| 감시 중인 컬렉션 drop, rename | 언급 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | +| 트랜잭션 | 언급 없음 | 이벤트는 오지만 `txnNumber`, `lsid`는 없음 | +| 큰 문서 | 언급 없음 | 14 MiB 문서의 insert와 update 이벤트 모두 전달됨 | +| 대기 중 resume token | 언급 없음 | 이벤트가 없어도 post-batch resume token이 전진함 | + +Learn은 이력 재개에 대해 다음을 설명합니다. 기본 change stream은 400 MB 크기의 +활성 change log 안의 이벤트만 읽습니다. PITR 로그와 통합되면 최대 35일 또는 클러스터 +초기화 시점 중 이른 쪽까지 재개 범위가 늘어납니다. 이번 실습은 이 범위를 시험하지 +않았습니다. + +### 정합성과 지연 + +모든 실행에서 `verify.py`가 보고한 누락, 예상 밖 이벤트, 문서별 순서 역전은 +0건이었습니다. + +| 실행 | 부하 | 이벤트 | 누락 | 재전달 | 지연 p50 / p99 | +| --- | --- | --- | --- | --- | --- | +| r1 | 문서 10,000개, 속도 제한 없음 | 23,000 | 0 | 0 | 57 / 177 ms | +| r2 | 문서 100,000개, consumer 파드 반복 삭제 | 230,000 | 0 | 0 | 측정 대상 아님 | +| r3 | 문서 100,000개, 200건 쓰기 직후 프로세스 종료를 5회 주입 | 230,000 | 0 | 1,000 | 측정 대상 아님 | +| r4 | 초당 1,000건, 5분 | 299,000 | 0 | 0 | 55 / 123 ms | + +- r2에서 파드를 삭제하면 이전 파드가 종료되는 동안 ReplicaSet이 새 파드를 + 띄웠습니다. `Recreate` 전략은 롤아웃에만 적용되므로 잠깐 두 consumer가 함께 + 읽었지만 upsert 덕분에 sink에 중복 행은 생기지 않았습니다. +- r3의 재전달 1,000건은 5회 × 200건입니다. checkpoint보다 sink 쓰기가 먼저 + 끝난 배치를 재시작 후 다시 받은 것입니다. change stream 소비는 + at-least-once이므로 sink가 멱등이어야 합니다. +- r3에서 밀린 이벤트를 따라잡는 속도는 초당 약 6,800건이었습니다. +- 클러스터 CPU는 초당 약 1,000건에서 약 30%, 속도 제한 없이 초당 약 2,900건을 + 쓸 때 약 60%였습니다. + +## 검증 + +- `verify.py` 출력의 `missing`, `unexpected`, `per_doc_out_of_order`가 모두 0인지 확인합니다. +- `redelivered_events`는 장애를 주입하지 않은 실행에서 0, r3에서는 주입 횟수 × 배치 + 크기와 같아야 합니다. +- `kubectl -n cslab logs deploy/cs-consumer`에서 재시작 직후 `start` 로그의 + `resume` 값이 `true`인지 확인합니다. 저장된 token으로 이어 읽었다는 뜻입니다. + +## 정리 + +```bash +az group delete -n $RG +``` + +삭제 전에 `kubectl config delete-context`로 로컬 kubeconfig의 AKS 항목도 지웁니다. + +## 문제 해결 + +| 증상 | 원인과 조치 | +| --- | --- | +| `watch()`가 code 26으로 실패 | 컬렉션이 없거나 `db.watch()`를 호출했습니다. 컬렉션을 먼저 만들거나 `client.watch()`와 `$match`를 씁니다 | +| 컬렉션을 지운 뒤 consumer가 반복 실패 | drop과 rename은 `invalidate` 없이 code 26을 반환합니다. 컬렉션을 다시 만들고 새 stream을 엽니다 | +| `start_at_operation_time`에 넘길 값이 없음 | 세션의 `operationTime`이 비어 있습니다. 시각 기반 `Timestamp`를 만들거나 resume token을 저장합니다 | +| `full_document_before_change="required"`가 code 10065로 실패 | pre-image는 미리 보기이며 지원 요청으로 켜야 합니다 | +| 큰 백로그를 재개할 때 code 50 `ExceededTimeLimit`로 반복 실패 | 이 클러스터는 `maxAwaitTimeMS`를 `getMore` 실행 제한으로 적용합니다. 오래된 change log를 읽는 `getMore`는 몇 초 걸릴 수 있습니다. consumer는 `MAX_AWAIT_MS`를 늘리고 직접 작성한 코드는 `max_await_time_ms`를 지정하지 않습니다. [Airflow DAG로 Parquet 적재](airflow-parquet/index.md)에서 측정했습니다 | +| 노드 크기 오류로 AKS 배포 실패 | 구독에서 허용되지 않는 VM 크기입니다. `nodeVmSize` 매개변수를 바꿉니다 | + +## 공식 참고 자료 + +- [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) +- [$changeStream](https://learn.microsoft.com/documentdb/query/operators/aggregation/$changestream) +- [Use Azure Private Link in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/how-to-private-link) +- [Compute and storage configurations](https://learn.microsoft.com/azure/documentdb/compute-storage) +- [Release notes for Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/release-notes) +- [Microsoft.DocumentDB mongoClusters Bicep reference](https://learn.microsoft.com/azure/templates/microsoft.documentdb/2026-06-01/mongoclusters) +- [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md new file mode 100644 index 00000000..f6cb5e26 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -0,0 +1,144 @@ +# Python change stream consumer on AKS + +This runnable sample tests MongoDB change streams on Azure DocumentDB with +PyMongo. The cluster has public network access disabled. Pods on AKS reach it +through a private endpoint. + +| Path | Purpose | +| --- | --- | +| `infra/main.bicep` | VNet, DocumentDB cluster (M30, one shard), private endpoint and DNS zone, ACR, AKS | +| `app/probe.py` | Checks which change stream options and event fields the cluster returns | +| `app/consumer.py` | Long-running consumer. Writes events to a sink collection and keeps the resume token in a checkpoint collection | +| `app/generator.py` | Deterministic insert, update, replace and delete workload | +| `app/verify.py` | Compares the generator's expected events with a sink collection | +| `infra/lake.bicep` | ADLS Gen2 account with public access and shared keys disabled, blob and dfs private endpoints, a managed identity federated to the `cs-lake` service account | +| `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes Parquet chunks and exits | +| `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | +| `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | +| `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job and service account, RBAC for the Airflow scheduler | + +## Run + +The commands use fictitious names. Replace them with your own values. + +```bash +RG=rg-docdb-changestream +az group create -n $RG -l eastus2 +az deployment group create -g $RG -f infra/main.bicep \ + -p adminPassword="$ADMIN_PASSWORD" + +ACR=docdbcsacrexample +az acr build -r $ACR -t cslab:v1 app/ +az aks get-credentials -g $RG -n aks-docdbcs + +kubectl create namespace cslab +kubectl -n cslab create secret generic docdb --from-literal=uri="$MONGO_URI" +``` + +`MONGO_URI` uses the cluster's `mongodb+srv://` connection string. The cluster +host name resolves to the private endpoint address inside the VNet. + +The watched collection must exist before the consumer starts. The cluster +under test returned `NamespaceNotFound` (code 26) for a missing collection. + +```bash +export IMAGE=docdbcsacrexample.azurecr.io/cslab:v1 +envsubst < k8s/toolbox.yaml | kubectl apply -f - +kubectl -n cslab exec cs-toolbox -- python -c \ + "from common import *; get_database(get_client('setup')).create_collection('orders')" + +envsubst < k8s/consumer.yaml | kubectl apply -f - + +JOB_NAME=gen-r1 SCRIPT=generator.py RUN_ID=r1 DOCS=100000 WORKERS=16 RATE=0 \ + envsubst < k8s/job.yaml | kubectl apply -f - +# After the generator finishes and the consumer catches up: +JOB_NAME=verify-r1 SCRIPT=verify.py RUN_ID=r1 DOCS=0 WORKERS=0 RATE=0 \ + envsubst < k8s/job.yaml | kubectl apply -f - +kubectl -n cslab logs job/verify-r1 +``` + +Run `probe.py` the same way with `SCRIPT=probe.py`. + +## Consumer behavior + +- The sink write is an upsert keyed by the resume token `_data`. A replayed + event increments `deliveries` on the existing row instead of adding a row. +- The checkpoint is saved after each sink batch. When the stream is idle, the + consumer saves the post-batch resume token instead. +- Delivery is at-least-once. `FAULT_EXIT_AFTER_WRITE=` makes the + process exit between the sink write and the checkpoint so you can see the + replay. Use it only in tests. +- On SIGTERM the consumer writes the pending batch and its checkpoint before + exiting. +- Deleting a pod makes the ReplicaSet start a replacement while the old pod is + still terminating. The `Recreate` strategy covers rollouts only, so for a + short time two consumers can run. The upsert keeps the sink correct. + +## Airflow to Parquet + +AKS needs the OIDC issuer and workload identity. `infra/main.bicep` enables +both. Deploy the lake next to the cluster. + +```bash +ISSUER=$(az aks show -g $RG -n aks-docdbcs --query oidcIssuerProfile.issuerUrl -o tsv) +az deployment group create -g $RG -f infra/lake.bicep -p oidcIssuerUrl="$ISSUER" +export LAKE_CLIENT_ID= +export LAKE_URL=https://docdbcslakeexample.dfs.core.windows.net/ +``` + +Install Airflow with the DAG baked into the image. The values use +`LocalExecutor`, so tasks run in the scheduler pod and its service account +launches the export pods in `cslab`. + +```bash +az acr build -r $ACR -t cslab-airflow:v1 airflow/ +kubectl apply -f k8s/airflow-rbac.yaml +ACR=docdbcsacrexample.azurecr.io AIRFLOW_IMAGE_TAG=v1 envsubst < airflow/values.yaml > values.rendered.yaml +helm repo add apache-airflow https://airflow.apache.org +helm install airflow apache-airflow/airflow --version 1.22.0 -n airflow --create-namespace \ + -f values.rendered.yaml --wait --timeout 10m +``` + +With chart defaults the database migration job is a post-install hook, so +`--wait` never sees it run and the Airflow pods wait for the migration forever. +`values.yaml` sets `useHelmHooks` and `applyCustomEnv` to `false` for +`migrateDatabaseJob` and `createUserJob`, as the +[chart documentation](https://airflow.apache.org/docs/helm-chart/stable/index.html) +advises for `--wait`. + +`lake.yaml` creates the `cs-lake` service account. Apply it once before the +first DAG run, for example with the verifier job. + +```bash +JOB_NAME=verify-lake-r1 SCRIPT=verify_lake.py LAKE_PREFIX=orders RUN_ID=r1 MAX_EVENTS=0 \ + FAULT_EXIT_AFTER_UPLOAD=0 envsubst < k8s/lake.yaml | kubectl apply -f - +``` + +Export behavior: + +- A run stops at the first event written after the run started, or when the + stream has nothing to return. Under steady writes `try_next()` rarely + returns `None`, so the time boundary is what ends the run. +- A run with no checkpoint starts at the current position of the stream. +- Each chunk is named after the resume token before its first event. A retry + reads the same events from the same checkpoint and overwrites the same file. +- The checkpoint is `_checkpoints/.json` in the same file system. It is + written with an ETag condition after each chunk upload. +- Trigger the DAG with `{"fault_after_chunks": 1}` to make the first try exit + after one upload and before the checkpoint. The retry finishes the run. +- `max_await_time_ms` is not set unless `MAX_AWAIT_MS` is non-zero. The test + cluster applied it as a limit on the whole `getMore`. With 1000 ms, resuming + a backlog larger than the active change log failed with code 50 + (`ExceededTimeLimit`) at the same event on every retry. An idle `getMore` + without it still returns after about one second. +- With 4 KB documents a 100,000-event chunk is about 374 MB and the export pod + used up to about 1.9 GiB. For large documents lower `CS_CHUNK_EVENTS` in the + Airflow scheduler environment. The DAG passes it to the pod as `CHUNK_EVENTS`. +- `consumer.py` always passes `MAX_AWAIT_MS`, 1000 by default. Raise it before + the consumer resumes a large backlog. + +## Clean up + +```bash +az group delete -n $RG +``` diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/Dockerfile b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/Dockerfile new file mode 100644 index 00000000..923325f6 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/Dockerfile @@ -0,0 +1,2 @@ +FROM apache/airflow:3.2.2 +COPY dags/ /opt/airflow/dags/ diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py new file mode 100644 index 00000000..0b885903 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py @@ -0,0 +1,54 @@ +"""Export Azure DocumentDB change stream events to Parquet in ADLS Gen2. + +Each run starts one pod in the ``cslab`` namespace. The pod resumes from the +checkpoint in the lake, writes the events recorded before the run started and +exits. ``max_active_runs=1`` keeps a single reader per stream. + +Trigger with ``{"fault_after_chunks": N}`` to make the first try exit after the +Nth upload and before the checkpoint, then let the retry finish the run. +""" + +import os +from datetime import datetime, timedelta + +from airflow.providers.cncf.kubernetes.operators.pod import KubernetesPodOperator +from airflow.providers.cncf.kubernetes.secret import Secret +from airflow.sdk import DAG +from kubernetes.client import models as k8s + +INTERVAL = timedelta(minutes=int(os.getenv("CS_EXPORT_INTERVAL_MIN", "5"))) + +with DAG( + dag_id="change_stream_to_parquet", + schedule=INTERVAL, + start_date=datetime(2026, 1, 1), + catchup=False, + max_active_runs=1, + default_args={"retries": 3, "retry_delay": timedelta(seconds=30)}, + tags=["change-stream"], +): + KubernetesPodOperator( + task_id="export", + name="cs-lake-export", + namespace="cslab", + image=os.getenv("CS_EXPORT_IMAGE", "cslab:latest"), + cmds=["python", "lake_export.py"], + env_vars={ + "LAKE_URL": os.getenv("CS_LAKE_URL", ""), + "LAKE_FILESYSTEM": "cdc", + "LAKE_PREFIX": "orders", + "STREAM_ID": "orders", + "CHUNK_EVENTS": os.getenv("CS_CHUNK_EVENTS", "100000"), + "MAX_EVENTS": "{{ dag_run.conf.get('max_events', 0) }}", + "FAULT_EXIT_AFTER_UPLOAD": + "{{ dag_run.conf.get('fault_after_chunks', 0) if ti.try_number == 1 else 0 }}", + }, + secrets=[Secret("env", "MONGO_URI", "docdb", "uri")], + service_account_name="cs-lake", + labels={"azure.workload.identity/use": "true"}, + container_resources=k8s.V1ResourceRequirements( + requests={"cpu": "1", "memory": "1Gi"}, limits={"memory": "4Gi"}), + get_logs=True, + startup_timeout_seconds=300, + on_finish_action="delete_succeeded_pod", + ) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml new file mode 100644 index 00000000..96451a03 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml @@ -0,0 +1,32 @@ +# Values for apache-airflow/airflow 1.22.0. Rendered with envsubst. +executor: LocalExecutor +images: + airflow: + repository: ${ACR}/cslab-airflow + tag: "${AIRFLOW_IMAGE_TAG}" +env: + - name: CS_EXPORT_IMAGE + value: "${IMAGE}" + - name: CS_LAKE_URL + value: "${LAKE_URL}" + - name: CS_EXPORT_INTERVAL_MIN + value: "5" +config: + core: + load_examples: "False" + scheduler: + dag_dir_list_interval: 30 +triggerer: + enabled: false +redis: + enabled: false +statsd: + enabled: false +# Run the migration and user jobs as normal resources so `helm --wait` does +# not wait on pods that are blocked behind a post-install hook. +migrateDatabaseJob: + useHelmHooks: false + applyCustomEnv: false +createUserJob: + useHelmHooks: false + applyCustomEnv: false diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/Dockerfile b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/Dockerfile new file mode 100644 index 00000000..e300e8de --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/Dockerfile @@ -0,0 +1,8 @@ +FROM python:3.12-slim +ENV PYTHONUNBUFFERED=1 PIP_NO_CACHE_DIR=1 +WORKDIR /app +COPY requirements.txt . +RUN pip install -r requirements.txt +COPY *.py ./ +USER 65532:65532 +CMD ["python", "consumer.py"] diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/common.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/common.py new file mode 100644 index 00000000..5595198d --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/common.py @@ -0,0 +1,59 @@ +"""Shared helpers for the Azure DocumentDB change stream lab.""" + +import json +import logging +import os +import sys +from datetime import datetime, timezone +from typing import Any, Optional + +import certifi +from pymongo import MongoClient +from pymongo.collection import Collection +from pymongo.database import Database + +CHECKPOINT_COLLECTION = "_cs_checkpoints" +SINK_COLLECTION = "_cs_sink" +RUN_COLLECTION = "_cs_runs" + + +def env(name: str, default: Optional[str] = None) -> str: + value = os.getenv(name, default) + if value is None or value == "": + raise SystemExit(f"Missing required environment variable: {name}") + return value + + +def utcnow() -> datetime: + return datetime.now(timezone.utc) + + +def setup_logging() -> logging.Logger: + logging.basicConfig( + stream=sys.stdout, + level=os.getenv("LOG_LEVEL", "INFO"), + format="%(asctime)s %(levelname)s %(message)s", + ) + return logging.getLogger("changestream") + + +def log_json(logger: logging.Logger, event: str, **fields: Any) -> None: + logger.info(json.dumps({"event": event, **fields}, default=str)) + + +def get_client(app_name: str) -> MongoClient: + return MongoClient( + env("MONGO_URI"), + appname=app_name, + serverSelectionTimeoutMS=15000, + tlsCAFile=certifi.where(), + tz_aware=True, + ) + + +def get_database(client: MongoClient) -> Database: + return client[env("MONGO_DB", "cslab")] + + +def get_source(db: Database) -> Collection: + return db[env("MONGO_COLLECTION", "orders")] diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py new file mode 100644 index 00000000..522a3929 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py @@ -0,0 +1,163 @@ +"""Change stream consumer for AKS. + +The consumer stores processed events in a sink collection and saves the +resume token in a checkpoint collection, so a replacement pod continues from +the last committed event instead of a local file. +""" + +import json +import os +import signal +import socket +import time +from typing import Any, Optional + +from pymongo import UpdateOne +from pymongo.errors import OperationFailure, PyMongoError + +from common import ( + CHECKPOINT_COLLECTION, + SINK_COLLECTION, + env, + get_client, + get_database, + get_source, + log_json, + setup_logging, + utcnow, +) + +logger = setup_logging() +stopping = False + + +def handle_sigterm(signum: int, _frame: Any) -> None: + global stopping + stopping = True + log_json(logger, "signal", signum=signum) + + +def load_checkpoint(db, consumer_id: str) -> Optional[dict]: + doc = db[CHECKPOINT_COLLECTION].find_one({"_id": consumer_id}) + return doc["token"] if doc else None + + +def save_checkpoint(db, consumer_id: str, token: dict, events: int) -> None: + db[CHECKPOINT_COLLECTION].update_one( + {"_id": consumer_id}, + { + "$set": {"token": token, "updated_at": utcnow(), "pod": socket.gethostname()}, + "$inc": {"events": events}, + }, + upsert=True, + ) + + +def to_sink(change: dict, pod: str) -> UpdateOne: + received = utcnow() + full = change.get("fullDocument") or {} + key = change.get("documentKey", {}).get("_id") + written = full.get("updated_at") or full.get("created_at") + lag_ms = (received - written).total_seconds() * 1000 if written else None + doc = { + "op": change["operationType"], + "doc_id": key, + "run_id": full.get("run_id") or (key.split(":")[0] if isinstance(key, str) else None), + "seq": full.get("seq"), + "version": full.get("version"), + "has_full_document": "fullDocument" in change and change["fullDocument"] is not None, + "has_update_description": "updateDescription" in change, + "received_at": received, + "recv_ns": time.time_ns(), + "lag_ms": lag_ms, + "pod": pod, + } + # Keyed by resume token: a replayed event updates the same row and bumps deliveries. + return UpdateOne({"_id": change["_id"]["_data"]}, + {"$setOnInsert": doc, "$inc": {"deliveries": 1}, "$set": {"last_pod": pod}}, + upsert=True) + + +def build_watch_kwargs(token: Optional[dict]) -> dict: + kwargs: dict = {"max_await_time_ms": int(os.getenv("MAX_AWAIT_MS", "1000"))} + full_document = os.getenv("FULL_DOCUMENT", "updateLookup") + if full_document != "default": + kwargs["full_document"] = full_document + batch_size = os.getenv("CS_BATCH_SIZE") + if batch_size: + kwargs["batch_size"] = int(batch_size) + if token: + kwargs["resume_after"] = token + return kwargs + + +def main() -> None: + signal.signal(signal.SIGTERM, handle_sigterm) + consumer_id = env("CONSUMER_ID", "orders-consumer") + batch_limit = int(os.getenv("SINK_BATCH", "200")) + pipeline = json.loads(os.getenv("PIPELINE_JSON", "[]")) + pod = socket.gethostname() + + client = get_client(f"cs-consumer-{pod}") + db = get_database(client) + source = get_source(db) + sink = db[SINK_COLLECTION] + + token = load_checkpoint(db, consumer_id) + log_json(logger, "start", consumer_id=consumer_id, pod=pod, resume=bool(token), + namespace=source.full_name, pipeline=pipeline) + + # Test-only fault injection: exit hard after a sink write but before the + # checkpoint, so the next pod replays events that are already in the sink. + fault_every = int(os.getenv("FAULT_EXIT_AFTER_WRITE", "0")) + + backoff = 1.0 + total = 0 + while not stopping: + try: + with source.watch(pipeline, **build_watch_kwargs(token)) as stream: + log_json(logger, "stream_open", resume=bool(token)) + backoff = 1.0 + pending: list = [] + last_token = None + while True: + change = None if stopping else stream.try_next() + if change is not None: + pending.append(to_sink(change, pod)) + last_token = change["_id"] + if len(pending) < batch_limit: + continue + if pending: + sink.bulk_write(pending, ordered=False) + if fault_every and total + len(pending) >= fault_every: + log_json(logger, "fault_exit", total=total + len(pending)) + os._exit(137) + save_checkpoint(db, consumer_id, last_token, len(pending)) + token = last_token + total += len(pending) + log_json(logger, "commit", batch=len(pending), total=total) + pending = [] + elif stopping: + break + elif stream.resume_token and stream.resume_token != token: + # Idle: advance the checkpoint to the post-batch resume token. + token = stream.resume_token + save_checkpoint(db, consumer_id, token, 0) + except OperationFailure as error: + log_json(logger, "operation_failure", code=error.code, error=str(error)[:500]) + if error.code in (280, 286): # ChangeStreamFatalError, ChangeStreamHistoryLost + raise + # 26 NamespaceNotFound: the watched collection is missing (not created + # yet, dropped or renamed). Observed instead of drop/invalidate events. + except PyMongoError as error: + log_json(logger, "stream_error", type=type(error).__name__, error=str(error)[:500]) + if not stopping: + time.sleep(backoff) + backoff = min(backoff * 2, 30) + + log_json(logger, "stopped", total=total) + client.close() + + +if __name__ == "__main__": + main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py new file mode 100644 index 00000000..b1fc938b --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py @@ -0,0 +1,109 @@ +"""Write a deterministic insert/update/replace/delete workload. + +Each document id is ``:`` so the verifier can match every +change event, including deletes that carry only ``documentKey``. +""" + +import os +import secrets +import threading +import time + +from pymongo import WriteConcern + +from common import RUN_COLLECTION, env, get_client, get_database, get_source, log_json, setup_logging, utcnow + +logger = setup_logging() + + +def pad(size: int) -> str: + """Random hex so storage compression does not shrink the filler.""" + return secrets.token_hex(size // 2) if size else "" + + +def worker(source, run_id: str, indexes: range, rate: float, pad_bytes: int, counts: dict, + lock: threading.Lock) -> None: + interval = 1.0 / rate if rate > 0 else 0.0 + next_at = time.monotonic() + local = {"insert": 0, "update": 0, "replace": 0, "delete": 0} + + def pace() -> None: + nonlocal next_at + if interval: + next_at += interval + delay = next_at - time.monotonic() + if delay > 0: + time.sleep(delay) + + for i in indexes: + doc_id = f"{run_id}:{i:07d}" + source.insert_one({ + "_id": doc_id, "run_id": run_id, "seq": i, "version": 1, + "status": "new", "qty": i % 100, "created_at": utcnow(), "pad": pad(pad_bytes), + }) + local["insert"] += 1 + pace() + source.update_one({"_id": doc_id}, {"$set": {"status": "paid", "updated_at": utcnow()}, + "$inc": {"version": 1}}) + local["update"] += 1 + pace() + if i % 10 == 0: + source.replace_one({"_id": doc_id}, { + "run_id": run_id, "seq": i, "version": 3, "status": "replaced", "updated_at": utcnow(), + "pad": pad(pad_bytes), + }) + local["replace"] += 1 + pace() + if i % 5 == 0: + source.delete_one({"_id": doc_id}) + local["delete"] += 1 + pace() + with lock: + for key, value in local.items(): + counts[key] += value + + +def main() -> None: + run_id = env("RUN_ID") + docs = int(os.getenv("DOCS", "1000")) + workers = int(os.getenv("WORKERS", "4")) + rate = float(os.getenv("RATE", "0")) # total ops/s, 0 = unthrottled + # Filler on inserts and replaces so the change log grows like a real payload. + pad_bytes = int(os.getenv("PAD_BYTES", "0")) + + client = get_client(f"cs-generator-{run_id}") + db = get_database(client) + source = get_source(db).with_options(write_concern=WriteConcern(w="majority")) + runs = db[RUN_COLLECTION] + + counts = {"insert": 0, "update": 0, "replace": 0, "delete": 0} + lock = threading.Lock() + started = utcnow() + runs.replace_one({"_id": run_id}, {"_id": run_id, "docs": docs, "workers": workers, "rate": rate, + "pad_bytes": pad_bytes, "started_at": started, "state": "running"}, upsert=True) + log_json(logger, "generator_start", run_id=run_id, docs=docs, workers=workers, rate=rate, + pad_bytes=pad_bytes) + + threads = [] + per_worker_rate = rate / workers if rate else 0.0 + for w in range(workers): + t = threading.Thread(target=worker, args=(source, run_id, range(w, docs, workers), + per_worker_rate, pad_bytes, counts, lock)) + t.start() + threads.append(t) + for t in threads: + t.join() + + finished = utcnow() + elapsed = (finished - started).total_seconds() + total_ops = sum(counts.values()) + runs.update_one({"_id": run_id}, {"$set": {"expected": counts, "finished_at": finished, + "elapsed_s": elapsed, "ops_per_s": total_ops / elapsed, + "state": "done"}}) + log_json(logger, "generator_done", run_id=run_id, expected=counts, elapsed_s=round(elapsed, 2), + ops_per_s=round(total_ops / elapsed, 1)) + client.close() + + +if __name__ == "__main__": + main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py new file mode 100644 index 00000000..545f479a --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -0,0 +1,179 @@ +"""Export change stream events to Parquet files in ADLS Gen2. + +One run is one Airflow task: resume from the checkpoint, read the events +written before the run started (or up to a limit), write Parquet chunks of a +fixed size and move the checkpoint after every uploaded chunk. + +A chunk is named after the resume token that precedes its first event. A retry +starts from the same checkpoint, reads the same events in the same order and +overwrites the same file with the same or a longer chunk, so a crash between +the upload and the checkpoint does not leave duplicate rows. +""" + +import hashlib +import io +import json +import logging +import os +import socket +import time +from datetime import datetime, timezone +from typing import Optional + +import pyarrow as pa +import pyarrow.parquet as pq +from azure.core import MatchConditions +from azure.core.exceptions import ResourceNotFoundError +from azure.identity import DefaultAzureCredential +from azure.storage.filedatalake import DataLakeServiceClient +from bson import json_util + +from common import env, get_client, get_database, get_source, log_json, setup_logging, utcnow + +logger = setup_logging() +# The storage SDK logs every HTTP request at INFO. +logging.getLogger("azure").setLevel(logging.WARNING) + +SCHEMA = pa.schema([ + ("resume_token", pa.string()), + ("op", pa.string()), + ("doc_id", pa.string()), + ("ns", pa.string()), + ("wall_time", pa.timestamp("ms", tz="UTC")), + ("read_at", pa.timestamp("us", tz="UTC")), + ("full_document", pa.string()), +]) + + +class Checkpoint: + """Resume token stored next to the data, updated with an ETag condition.""" + + def __init__(self, fs, stream_id: str): + self.file = fs.get_file_client(f"_checkpoints/{stream_id}.json") + self.etag: Optional[str] = None + + def load(self) -> Optional[dict]: + try: + download = self.file.download_file() + except ResourceNotFoundError: + return None + self.etag = download.properties.etag + return json_util.loads(download.readall())["token"] + + def save(self, token: dict, events: int) -> None: + body = json_util.dumps({"token": token, "updated_at": utcnow(), "events": events, + "pod": socket.gethostname()}) + # IfNotModified fails if another run moved the checkpoint since we read it. + condition = (dict(etag=self.etag, match_condition=MatchConditions.IfNotModified) + if self.etag else dict(match_condition=MatchConditions.IfMissing)) + result = self.file.upload_data(body, overwrite=True, **condition) + self.etag = result["etag"] + + +def to_row(change: dict) -> dict: + full = change.get("fullDocument") + return { + "resume_token": change["_id"]["_data"], + "op": change["operationType"], + "doc_id": str(change.get("documentKey", {}).get("_id")), + "ns": f"{change['ns']['db']}.{change['ns']['coll']}", + "wall_time": change.get("wallTime"), + "read_at": utcnow(), + "full_document": json_util.dumps(full, json_options=json_util.RELAXED_JSON_OPTIONS) + if full is not None else None, + } + + +def chunk_path(prefix: str, start_token: Optional[dict], rows: list) -> str: + first = rows[0]["wall_time"] or datetime.now(timezone.utc) + key = start_token["_data"] if start_token else "origin" + digest = hashlib.sha256(key.encode()).hexdigest()[:20] + return f"{prefix}/dt={first:%Y-%m-%d}/hour={first:%H}/part-{digest}.parquet" + + +def write_chunk(fs, path: str, rows: list) -> int: + table = pa.Table.from_pylist(rows, schema=SCHEMA) + buffer = io.BytesIO() + pq.write_table(table, buffer, compression="snappy") + data = buffer.getvalue() + fs.get_file_client(path).upload_data(data, overwrite=True) + return len(data) + + +def main() -> None: + stream_id = env("STREAM_ID", "orders") + prefix = env("LAKE_PREFIX", "orders") + chunk_events = int(os.getenv("CHUNK_EVENTS", "100000")) + max_events = int(os.getenv("MAX_EVENTS", "0")) # 0 = until caught up + max_seconds = float(os.getenv("MAX_SECONDS", "0")) + # Test-only: exit after uploading this many chunks, before the checkpoint. + fault_after_chunks = int(os.getenv("FAULT_EXIT_AFTER_UPLOAD", "0")) + + service = DataLakeServiceClient(env("LAKE_URL"), credential=DefaultAzureCredential()) + fs = service.get_file_system_client(env("LAKE_FILESYSTEM", "cdc")) + checkpoint = Checkpoint(fs, stream_id) + token = checkpoint.load() + + client = get_client(f"cs-lake-export-{socket.gethostname()}") + source = get_source(get_database(client)) + kwargs: dict = {"batch_size": int(os.getenv("CS_BATCH_SIZE", "1000"))} + # The cluster applies maxAwaitTimeMS as a hard limit on getMore. Reading a + # backlog older than the active change log can take several seconds per + # getMore and then fails with ExceededTimeLimit, so leave it unset by + # default. An idle getMore still returns after about one second. + max_await_ms = int(os.getenv("MAX_AWAIT_MS", "0")) + if max_await_ms: + kwargs["max_await_time_ms"] = max_await_ms + if token: + kwargs["resume_after"] = token + + started = time.monotonic() + run_started_at = utcnow() + log_json(logger, "export_start", stream_id=stream_id, resume=bool(token), chunk_events=chunk_events) + total = chunks = written_bytes = 0 + caught_up = False + with source.watch(**kwargs) as stream: + rows: list = [] + chunk_start = token + last_token = token + while True: + change = stream.try_next() + if change is not None: + rows.append(to_row(change)) + last_token = change["_id"] + # Under steady load try_next rarely returns None, so stop once + # the stream reaches events written after this run started. + wall_time = change.get("wallTime") + caught_up = wall_time is not None and wall_time >= run_started_at + else: + caught_up = True + full = len(rows) >= chunk_events + limit = (max_events and total + len(rows) >= max_events) or \ + (max_seconds and time.monotonic() - started >= max_seconds) + if rows and (full or caught_up or limit): + path = chunk_path(prefix, chunk_start, rows) + size = write_chunk(fs, path, rows) + chunks += 1 + if fault_after_chunks and chunks >= fault_after_chunks: + log_json(logger, "fault_exit", chunks=chunks, path=path) + os._exit(137) + checkpoint.save(last_token, len(rows)) + total += len(rows) + written_bytes += size + log_json(logger, "chunk", path=path, rows=len(rows), bytes=size, total=total) + rows = [] + chunk_start = last_token + if caught_up or limit: + break + # Nothing new: move the checkpoint to the post-batch resume token so + # the next run does not scan the same idle range again. + if total == 0 and stream.resume_token and stream.resume_token != token: + checkpoint.save(stream.resume_token, 0) + + log_json(logger, "export_done", events=total, chunks=chunks, bytes=written_bytes, + caught_up=caught_up, elapsed_s=round(time.monotonic() - started, 2)) + client.close() + + +if __name__ == "__main__": + main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py new file mode 100644 index 00000000..f42b68ca --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py @@ -0,0 +1,322 @@ +"""Probe which change stream options Azure DocumentDB accepts through PyMongo. + +Each check runs against its own collection and reports PASS, FAIL or an +observed value. Nothing here asserts documented behavior; it records what the +cluster under test actually returned. +""" + +import json +import time +from datetime import timedelta +from typing import Any, Callable, Optional + +import pymongo +from bson import Timestamp +from pymongo.errors import OperationFailure, PyMongoError + +from common import get_client, get_database, utcnow + +client = get_client("cs-probe") +db = get_database(client) +results: list = [] +WAIT_S = 15 + + +def record(name: str, status: str, **detail: Any) -> None: + item = {"check": name, "status": status, **detail} + results.append(item) + print(json.dumps(item, default=str), flush=True) + + +def fresh(name: str): + coll = db[f"probe_{name}"] + coll.drop() + db.create_collection(coll.name) + return coll + + +def next_event(stream, timeout: float = WAIT_S) -> Optional[dict]: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + change = stream.try_next() + if change is not None: + return change + return None + + +def check(name: str) -> Callable: + def wrap(fn: Callable) -> Callable: + try: + fn() + except OperationFailure as error: + record(name, "FAIL", code=error.code, codeName=error.details.get("codeName") if error.details else None, + error=str(error)[:300]) + except PyMongoError as error: + record(name, "FAIL", type=type(error).__name__, error=str(error)[:300]) + except Exception as error: # noqa: BLE001 - probe must keep going + record(name, "ERROR", type=type(error).__name__, error=str(error)[:300]) + return fn + return wrap + + +@check("server_info") +def _server_info(): + info = client.admin.command("buildInfo") + hello = client.admin.command("hello") + record("server_info", "INFO", version=info.get("version"), pymongo=pymongo.version, + maxWireVersion=hello.get("maxWireVersion"), setName=hello.get("setName"), + msg=hello.get("msg")) + + +@check("event_shapes") +def _event_shapes(): + coll = fresh("shapes") + with coll.watch(max_await_time_ms=500) as s: + coll.insert_one({"_id": 1, "a": 1, "arr": [1, 2, 3]}) + coll.update_one({"_id": 1}, {"$set": {"a": 2}, "$unset": {"arr": ""}}) + coll.replace_one({"_id": 1}, {"a": 3}) + coll.delete_one({"_id": 1}) + shapes = {} + for _ in range(4): + e = next_event(s) + if e is None: + break + shapes[e["operationType"]] = { + "keys": sorted(e.keys()), + "updateDescription": e.get("updateDescription"), + "clusterTime_type": type(e.get("clusterTime")).__name__, + "wallTime": e.get("wallTime"), + } + status = "PASS" if set(shapes) == {"insert", "update", "replace", "delete"} else "FAIL" + record("event_shapes", status, shapes=shapes) + + +@check("update_full_document_default") +def _update_default(): + coll = fresh("upd_default") + coll.insert_one({"_id": 1, "a": 1}) + with coll.watch(max_await_time_ms=500) as s: + coll.update_one({"_id": 1}, {"$set": {"a": 2}}) + e = next_event(s) + record("update_full_document_default", "INFO", has_fullDocument="fullDocument" in e, + fullDocument=e.get("fullDocument"), updateDescription=e.get("updateDescription")) + + +for mode in ("updateLookup", "whenAvailable", "required"): + @check(f"full_document={mode}") + def _full_doc(mode=mode): + coll = fresh(f"fd_{mode}") + coll.insert_one({"_id": 1, "a": 1}) + with coll.watch(full_document=mode, max_await_time_ms=500) as s: + coll.update_one({"_id": 1}, {"$set": {"a": 2}}) + e = next_event(s) + record(f"full_document={mode}", "PASS" if e and e.get("fullDocument") else "FAIL", + fullDocument=e.get("fullDocument") if e else None) + + +@check("pre_image_collmod") +def _pre_image_collmod(): + coll = fresh("preimage") + db.command({"collMod": coll.name, "changeStreamPreAndPostImages": {"enabled": True}}) + record("pre_image_collmod", "PASS") + + +for mode in ("whenAvailable", "required"): + @check(f"full_document_before_change={mode}") + def _pre_image(mode=mode): + coll = db["probe_preimage"] + coll.delete_many({}) + coll.insert_one({"_id": mode, "a": 1}) + with coll.watch(full_document_before_change=mode, max_await_time_ms=500) as s: + coll.update_one({"_id": mode}, {"$set": {"a": 2}}) + e = next_event(s) + record(f"full_document_before_change={mode}", + "PASS" if e and e.get("fullDocumentBeforeChange") else "FAIL", + fullDocumentBeforeChange=e.get("fullDocumentBeforeChange") if e else None) + + +PIPELINES = { + "$match": [{"$match": {"operationType": "insert", "fullDocument.dept": "IT"}}], + "$project": [{"$project": {"operationType": 1, "fullDocument.name": 1, "ns": 1, "documentKey": 1}}], + "$addFields": [{"$addFields": {"probe": "added"}}], + "$set": [{"$set": {"probe": "set"}}], + "$unset": [{"$unset": "fullDocument.dept"}], + "$replaceRoot": [{"$replaceRoot": {"newRoot": {"_id": "$_id", "op": "$operationType"}}}], + "$redact": [{"$redact": "$$KEEP"}], +} +for stage, pipeline in PIPELINES.items(): + @check(f"pipeline {stage}") + def _pipeline(stage=stage, pipeline=pipeline): + coll = fresh(f"pl_{stage[1:]}") + with coll.watch(pipeline, max_await_time_ms=500) as s: + coll.insert_one({"name": "x", "dept": "HR"}) + coll.insert_one({"name": "y", "dept": "IT"}) + events = [e for e in (next_event(s, 8), next_event(s, 3)) if e] + record(f"pipeline {stage}", "PASS" if events else "FAIL", events=len(events), + sample={k: v for k, v in events[0].items() if k != "_id"} if events else None) + + +@check("pipeline_update_updateDescription") +def _pipeline_update(): + coll = fresh("pipeline_update") + coll.insert_one({"_id": 1, "a": 1}) + with coll.watch(max_await_time_ms=500) as s: + coll.update_one({"_id": 1}, [{"$set": {"a": {"$add": ["$a", 1]}}}]) + e = next_event(s) + record("pipeline_update_updateDescription", "INFO", operationType=e.get("operationType") if e else None, + has_updateDescription=bool(e and "updateDescription" in e), + updateDescription=e.get("updateDescription") if e else None) + + +@check("database_watch") +def _db_watch(): + coll = fresh("dbwatch") + with db.watch(max_await_time_ms=500) as s: + coll.insert_one({"a": 1}) + e = next_event(s) + record("database_watch", "PASS" if e else "FAIL", ns=e.get("ns") if e else None) + + +@check("cluster_watch") +def _cluster_watch(): + coll = fresh("clusterwatch") + with client.watch(max_await_time_ms=500) as s: + coll.insert_one({"a": 1}) + e = next_event(s) + record("cluster_watch", "PASS" if e else "FAIL", ns=e.get("ns") if e else None) + + +@check("resume_after") +def _resume_after(): + coll = fresh("resume") + with coll.watch(max_await_time_ms=500) as s: + coll.insert_one({"_id": 1}) + first = next_event(s) + coll.insert_one({"_id": 2}) + coll.insert_one({"_id": 3}) + with coll.watch(resume_after=first["_id"], max_await_time_ms=500) as s: + got = [next_event(s, 8), next_event(s, 8)] + ids = [e["documentKey"]["_id"] for e in got if e] + record("resume_after", "PASS" if ids == [2, 3] else "FAIL", resumed_ids=ids, + token_sample=first["_id"]) + + +@check("start_after") +def _start_after(): + coll = fresh("startafter") + with coll.watch(max_await_time_ms=500) as s: + coll.insert_one({"_id": 1}) + first = next_event(s) + coll.insert_one({"_id": 2}) + with coll.watch(start_after=first["_id"], max_await_time_ms=500) as s: + e = next_event(s, 8) + record("start_after", "PASS" if e and e["documentKey"]["_id"] == 2 else "FAIL", + got=e["documentKey"] if e else None) + + +@check("start_at_operation_time") +def _start_at_op_time(): + coll = fresh("optime") + with client.start_session() as session: + coll.insert_one({"_id": 1}, session=session) + op_time = session.operation_time + coll.insert_one({"_id": 2}) + with coll.watch(start_at_operation_time=op_time, max_await_time_ms=500) as s: + got = [next_event(s, 8), next_event(s, 8)] + ids = [e["documentKey"]["_id"] for e in got if e] + record("start_at_operation_time", "PASS" if 2 in ids else "FAIL", operation_time=op_time, ids=ids) + + +@check("start_at_operation_time_minus_10m") +def _historical(): + coll = db["probe_optime"] + past = Timestamp(int((utcnow() - timedelta(minutes=10)).timestamp()), 1) + with coll.watch(start_at_operation_time=past, max_await_time_ms=500) as s: + e = next_event(s, 8) + record("start_at_operation_time_minus_10m", "PASS" if e else "INFO", first=e.get("documentKey") if e else None) + + +@check("invalid_resume_token") +def _invalid_token(): + coll = fresh("badtoken") + with coll.watch(resume_after={"_data": "AAAAAAAAAAAA"}, max_await_time_ms=500) as s: + coll.insert_one({"a": 1}) + e = next_event(s, 5) + record("invalid_resume_token", "INFO", note="stream opened with bogus token", got=bool(e)) + + +@check("show_expanded_events") +def _expanded(): + coll = fresh("expanded") + with coll.watch(show_expanded_events=True, max_await_time_ms=500) as s: + coll.insert_one({"a": 1}) + e = next_event(s, 5) + record("show_expanded_events", "INFO", note="accepted", got=bool(e)) + + +@check("post_batch_resume_token_idle") +def _pbrt(): + coll = fresh("pbrt") + with coll.watch(max_await_time_ms=500) as s: + before = s.resume_token + s.try_next() + time.sleep(1) + s.try_next() + after = s.resume_token + record("post_batch_resume_token_idle", "PASS" if after else "FAIL", before=before, after=after, + advanced=before != after) + + +@check("transaction_events") +def _txn(): + coll = fresh("txn") + with coll.watch(max_await_time_ms=500) as s: + with client.start_session() as session: + with session.start_transaction(): + coll.insert_one({"_id": 1}, session=session) + coll.insert_one({"_id": 2}, session=session) + got = [next_event(s, 8), next_event(s, 8)] + got = [e for e in got if e] + record("transaction_events", "PASS" if len(got) == 2 else "FAIL", events=len(got), + txn_fields={k: str(got[0].get(k)) for k in ("lsid", "txnNumber") if got and k in got[0]}) + + +@check("drop_and_invalidate") +def _drop(): + coll = fresh("drop") + with coll.watch(max_await_time_ms=500) as s: + coll.insert_one({"a": 1}) + coll.drop() + ops = [] + for _ in range(3): + e = next_event(s, 6) + if e is None: + break + ops.append(e["operationType"]) + alive = s.alive + record("drop_and_invalidate", "INFO", ops=ops, stream_alive_after=alive) + + +@check("rename") +def _rename(): + coll = fresh("rename") + db["probe_renamed"].drop() + with coll.watch(max_await_time_ms=500) as s: + coll.rename("probe_renamed") + e = next_event(s, 6) + record("rename", "INFO", op=e.get("operationType") if e else None, to=e.get("to") if e else None) + + +@check("large_document_event") +def _large(): + coll = fresh("large") + payload = "x" * (7 * 1024 * 1024) # doc 14 MiB; event with updateDescription exceeds 16 MiB + with coll.watch(full_document="updateLookup", max_await_time_ms=500) as s: + coll.insert_one({"_id": 1, "blob": payload}) + coll.update_one({"_id": 1}, {"$set": {"blob2": payload}}) + got = [next_event(s, 15), next_event(s, 15)] + record("large_document_event", "INFO", ops=[e["operationType"] if e else None for e in got]) + + +print(json.dumps({"summary": {r["check"]: r["status"] for r in results}}, indent=2)) +client.close() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt new file mode 100644 index 00000000..4934b775 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt @@ -0,0 +1,5 @@ +pymongo[srv]==4.18.2 +certifi==2026.7.22 +pyarrow==25.0.1 +azure-storage-file-datalake==12.26.0 +azure-identity==1.26.0 diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py new file mode 100644 index 00000000..a8c16d55 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py @@ -0,0 +1,103 @@ +"""Compare the generator's expected events with what the consumer committed.""" + +import json +import os +import sys +from collections import Counter, defaultdict + +from common import RUN_COLLECTION, SINK_COLLECTION, env, get_client, get_database + +def percentile(values: list, pct: float): + if not values: + return None + values = sorted(values) + k = max(0, min(len(values) - 1, round(pct / 100 * (len(values) - 1)))) + return round(values[k], 1) + + +def summarize(values: list) -> dict: + return {"p50": percentile(values, 50), "p95": percentile(values, 95), "p99": percentile(values, 99), + "max": percentile(values, 100), "n": len(values)} + + +def expected_events(run_id: str, docs: int) -> Counter: + """Expected (doc_id, op) counts. + + ``replace_one`` is counted under ``update`` because the cluster under test + emits replacements as ``operationType: update``. + """ + expected = Counter() + for i in range(docs): + doc_id = f"{run_id}:{i:07d}" + expected[(doc_id, "insert")] += 1 + expected[(doc_id, "update")] += 2 if i % 10 == 0 else 1 + if i % 5 == 0: + expected[(doc_id, "delete")] += 1 + return expected + + +def main() -> None: + run_id = env("RUN_ID") + client = get_client(f"cs-verify-{run_id}") + db = get_database(client) + run = db[RUN_COLLECTION].find_one({"_id": run_id}) + if not run or run.get("state") != "done": + print(json.dumps({"run_id": run_id, "error": "generator run not finished", "run": run}, default=str)) + sys.exit(2) + + sink = os.getenv("SINK") or SINK_COLLECTION + events = list(db[sink].find( + {"doc_id": {"$regex": f"^{run_id}:"}}, + {"doc_id": 1, "op": 1, "recv_ns": 1, "pod": 1, "has_full_document": 1, + "has_update_description": 1, "deliveries": 1, "lag_ms": 1}, + )) + # The sink is keyed by resume token, so each row is one distinct event and + # replays show up as deliveries > 1 instead of extra rows. + got = Counter((e["doc_id"], e["op"]) for e in events) + expected = expected_events(run_id, run["docs"]) + missing = sorted((expected - got).elements()) + unexpected = sorted((got - expected).elements()) + + # Per-document order: insert -> update(s) -> delete. + order = {"insert": 0, "update": 1, "replace": 1, "delete": 2} + per_doc = defaultdict(list) + for e in events: + per_doc[e["doc_id"]].append((e["recv_ns"], order[e["op"]])) + out_of_order = [d for d, seq in per_doc.items() + if [o for _, o in sorted(seq)] != sorted(o for _, o in seq)] + + lag_by_op = defaultdict(list) + for e in events: + if e.get("lag_ms") is not None: + lag_by_op[e["op"]].append(e["lag_ms"]) + + result = { + "run_id": run_id, + "sink": sink, + "docs": run["docs"], + "workers": run["workers"], + "generator_ops_per_s": round(run["ops_per_s"], 1), + "generator_elapsed_s": round(run["elapsed_s"], 1), + "generator_ops": run["expected"], + "expected_events": sum(expected.values()), + "received_events": len(events), + "received_by_op": dict(Counter(e["op"] for e in events)), + "pods": dict(Counter(e["pod"] for e in events)), + "missing": len(missing), + "missing_sample": missing[:10], + "unexpected": len(unexpected), + "unexpected_sample": unexpected[:10], + "redelivered_events": sum(1 for e in events if e.get("deliveries", 1) > 1), + "redeliveries_total": sum(e.get("deliveries", 1) - 1 for e in events), + "per_doc_out_of_order": len(out_of_order), + "update_without_update_description": sum(1 for e in events if e["op"] == "update" + and not e["has_update_description"]), + "lag_ms": {op: summarize(v) for op, v in lag_by_op.items()}, + } + print(json.dumps(result, default=str, indent=None if os.getenv("COMPACT") else 2)) + client.close() + sys.exit(0 if not missing and not unexpected else 1) + + +if __name__ == "__main__": + main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py new file mode 100644 index 00000000..df6d3301 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py @@ -0,0 +1,107 @@ +"""Compare the generator's expected events with the Parquet files in ADLS Gen2.""" + +import io +import json +import logging +import os +import sys +from collections import Counter, defaultdict +from datetime import timezone + +import pyarrow.parquet as pq +from azure.identity import DefaultAzureCredential +from azure.storage.filedatalake import DataLakeServiceClient + +from common import RUN_COLLECTION, env, get_client, get_database +from verify import expected_events, percentile, summarize + +logging.getLogger("azure").setLevel(logging.WARNING) + + +def main() -> None: + run_id = env("RUN_ID") + client = get_client(f"cs-verify-lake-{run_id}") + run = get_database(client)[RUN_COLLECTION].find_one({"_id": run_id}) + if not run or run.get("state") != "done": + print(json.dumps({"run_id": run_id, "error": "generator run not finished", "run": run}, default=str)) + sys.exit(2) + + service = DataLakeServiceClient(env("LAKE_URL"), credential=DefaultAzureCredential()) + fs = service.get_file_system_client(env("LAKE_FILESYSTEM", "cdc")) + prefix = env("LAKE_PREFIX", "orders") + columns = ["resume_token", "op", "doc_id", "wall_time", "read_at"] + + rows = [] + files = [] + for path in fs.get_paths(path=prefix, recursive=True): + if path.is_directory or not path.name.endswith(".parquet"): + continue + data = fs.get_file_client(path.name).download_file().readall() + table = pq.read_table(io.BytesIO(data), columns=columns) + mine = [r for r in table.to_pylist() if r["doc_id"].startswith(f"{run_id}:")] + if not mine: + continue + files.append({"rows": len(mine), "file_rows": table.num_rows, "bytes": path.content_length}) + # The listing returns last-modified as a naive UTC datetime. + committed_at = path.last_modified.replace(tzinfo=timezone.utc) + for i, r in enumerate(mine): + r["committed_at"] = committed_at + r["idx"] = i + rows.append(r) + + # Every event has a unique resume token, so extra rows with the same token + # are duplicates written by a retried or overlapping export. + by_token = Counter(r["resume_token"] for r in rows) + duplicates = sum(c - 1 for c in by_token.values() if c > 1) + seen = set() + events = [] + for r in sorted(rows, key=lambda r: (r["read_at"], r["committed_at"], r["idx"])): + if r["resume_token"] not in seen: + seen.add(r["resume_token"]) + events.append(r) + + got = Counter((e["doc_id"], e["op"]) for e in events) + expected = expected_events(run_id, run["docs"]) + missing = sorted((expected - got).elements()) + unexpected = sorted((got - expected).elements()) + + order = {"insert": 0, "update": 1, "replace": 1, "delete": 2} + per_doc = defaultdict(list) + for e in events: + per_doc[e["doc_id"]].append(order[e["op"]]) + out_of_order = [d for d, seq in per_doc.items() if seq != sorted(seq)] + + read_lag_ms = [(e["read_at"] - e["wall_time"]).total_seconds() * 1000 for e in events if e["wall_time"]] + commit_lag_s = [(e["committed_at"] - e["wall_time"]).total_seconds() for e in events if e["wall_time"]] + file_rows = [f["file_rows"] for f in files] + file_bytes = [f["bytes"] for f in files] + + result = { + "run_id": run_id, + "docs": run["docs"], + "generator_ops_per_s": round(run["ops_per_s"], 1), + "generator_elapsed_s": round(run["elapsed_s"], 1), + "expected_events": sum(expected.values()), + "received_events": len(events), + "duplicate_rows": duplicates, + "missing": len(missing), + "missing_sample": missing[:10], + "unexpected": len(unexpected), + "unexpected_sample": unexpected[:10], + "per_doc_out_of_order": len(out_of_order), + "read_lag_ms": summarize(read_lag_ms), + # Storage last-modified has one-second resolution. + "commit_lag_s": summarize(commit_lag_s), + "files": len(files), + "file_rows": {"min": min(file_rows, default=None), "p50": percentile(file_rows, 50), + "max": max(file_rows, default=None)}, + "file_bytes": {"min": min(file_bytes, default=None), "p50": percentile(file_bytes, 50), + "max": max(file_bytes, default=None), "total": sum(file_bytes)}, + } + print(json.dumps(result, default=str, indent=None if os.getenv("COMPACT") else 2)) + client.close() + sys.exit(0 if not missing and not unexpected and not duplicates else 1) + + +if __name__ == "__main__": + main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep new file mode 100644 index 00000000..c2b54db2 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep @@ -0,0 +1,158 @@ +// ADLS Gen2 landing zone for the Airflow export path. +// Public network access and shared key access are disabled. AKS pods reach the +// account through blob and dfs private endpoints and authenticate with a +// user-assigned managed identity federated to a Kubernetes service account. + +targetScope = 'resourceGroup' + +@description('Region for every resource.') +param location string = resourceGroup().location + +@description('Short prefix used in resource names. Must match main.bicep.') +param prefix string = 'docdbcs' + +@description('OIDC issuer URL of the AKS cluster.') +param oidcIssuerUrl string + +@description('Kubernetes namespace and service account that run the export task.') +param serviceAccountNamespace string = 'cslab' +param serviceAccountName string = 'cs-lake' + +@description('File system (container) for exported change events.') +param fileSystemName string = 'cdc' + +var suffix = uniqueString(resourceGroup().id) +var storageName = '${prefix}lake${suffix}' + +resource vnet 'Microsoft.Network/virtualNetworks@2024-05-01' existing = { + name: 'vnet-${prefix}' +} + +resource storage 'Microsoft.Storage/storageAccounts@2024-01-01' = { + name: storageName + location: location + kind: 'StorageV2' + sku: { + name: 'Standard_LRS' + } + properties: { + isHnsEnabled: true + publicNetworkAccess: 'Disabled' + allowBlobPublicAccess: false + allowSharedKeyAccess: false + minimumTlsVersion: 'TLS1_2' + supportsHttpsTrafficOnly: true + networkAcls: { + defaultAction: 'Deny' + bypass: 'None' + } + } +} + +resource blobService 'Microsoft.Storage/storageAccounts/blobServices@2024-01-01' = { + parent: storage + name: 'default' +} + +resource fileSystem 'Microsoft.Storage/storageAccounts/blobServices/containers@2024-01-01' = { + parent: blobService + name: fileSystemName +} + +// Data Lake Storage needs both blob and dfs private endpoints. +var endpoints = [ + { + groupId: 'blob' + zone: 'privatelink.blob.${environment().suffixes.storage}' + } + { + groupId: 'dfs' + zone: 'privatelink.dfs.${environment().suffixes.storage}' + } +] + +resource zones 'Microsoft.Network/privateDnsZones@2024-06-01' = [for ep in endpoints: { + name: ep.zone + location: 'global' +}] + +resource zoneLinks 'Microsoft.Network/privateDnsZones/virtualNetworkLinks@2024-06-01' = [for (ep, i) in endpoints: { + parent: zones[i] + name: 'link-${prefix}' + location: 'global' + properties: { + registrationEnabled: false + virtualNetwork: { + id: vnet.id + } + } +}] + +resource privateEndpoints 'Microsoft.Network/privateEndpoints@2024-05-01' = [for ep in endpoints: { + name: 'pe-${storageName}-${ep.groupId}' + location: location + properties: { + subnet: { + id: '${vnet.id}/subnets/snet-pe' + } + privateLinkServiceConnections: [ + { + name: 'plsc-${storageName}-${ep.groupId}' + properties: { + privateLinkServiceId: storage.id + groupIds: [ + ep.groupId + ] + } + } + ] + } +}] + +resource zoneGroups 'Microsoft.Network/privateEndpoints/privateDnsZoneGroups@2024-05-01' = [for (ep, i) in endpoints: { + parent: privateEndpoints[i] + name: 'default' + properties: { + privateDnsZoneConfigs: [ + { + name: ep.groupId + properties: { + privateDnsZoneId: zones[i].id + } + } + ] + } +}] + +resource identity 'Microsoft.ManagedIdentity/userAssignedIdentities@2023-01-31' = { + name: 'id-${prefix}-lake' + location: location +} + +resource federation 'Microsoft.ManagedIdentity/userAssignedIdentities/federatedIdentityCredentials@2023-01-31' = { + parent: identity + name: 'aks-${serviceAccountNamespace}-${serviceAccountName}' + properties: { + issuer: oidcIssuerUrl + subject: 'system:serviceaccount:${serviceAccountNamespace}:${serviceAccountName}' + audiences: [ + 'api://AzureADTokenExchange' + ] + } +} + +// Storage Blob Data Contributor on the account. +resource lakeWriter 'Microsoft.Authorization/roleAssignments@2022-04-01' = { + name: guid(storage.id, identity.id, 'blobcontrib') + scope: storage + properties: { + principalId: identity.properties.principalId + principalType: 'ServicePrincipal' + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', 'ba92f5b4-2d11-453d-a403-e96b0029c9fe') + } +} + +output storageAccountName string = storage.name +output dfsEndpoint string = storage.properties.primaryEndpoints.dfs +output fileSystemName string = fileSystem.name +output identityClientId string = identity.properties.clientId diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep new file mode 100644 index 00000000..dacc9fb0 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep @@ -0,0 +1,213 @@ +// Azure DocumentDB (vCore) + AKS + ACR for the Python change stream lab. +// The cluster has public network access disabled and is reached from AKS +// through a private endpoint and the privatelink.mongocluster.cosmos.azure.com zone. + +targetScope = 'resourceGroup' + +@description('Region for every resource.') +param location string = resourceGroup().location + +@description('Short prefix used in resource names.') +param prefix string = 'docdbcs' + +@description('DocumentDB administrator user name.') +param adminUserName string = 'csadmin' + +@secure() +@description('DocumentDB administrator password.') +param adminPassword string + +@description('DocumentDB compute tier, for example M30.') +param clusterTier string = 'M30' + +@description('Storage size per shard in GiB.') +param storageSizeGb int = 32 + +@description('AKS node VM size.') +param nodeVmSize string = 'Standard_D4s_v6' + +@description('AKS node count.') +param nodeCount int = 2 + +var suffix = uniqueString(resourceGroup().id) +var clusterName = '${prefix}-${suffix}' +var privateDnsZoneName = 'privatelink.mongocluster.cosmos.azure.com' + +resource vnet 'Microsoft.Network/virtualNetworks@2024-05-01' = { + name: 'vnet-${prefix}' + location: location + properties: { + addressSpace: { + addressPrefixes: [ + '10.40.0.0/16' + ] + } + subnets: [ + { + name: 'snet-aks' + properties: { + addressPrefix: '10.40.0.0/22' + } + } + { + name: 'snet-pe' + properties: { + addressPrefix: '10.40.8.0/24' + privateEndpointNetworkPolicies: 'Disabled' + } + } + ] + } +} + +resource cluster 'Microsoft.DocumentDB/mongoClusters@2025-09-01' = { + name: clusterName + location: location + properties: { + administrator: { + userName: adminUserName + password: adminPassword + } + compute: { + tier: clusterTier + } + storage: { + sizeGb: storageSizeGb + } + sharding: { + shardCount: 1 + } + highAvailability: { + targetMode: 'Disabled' + } + publicNetworkAccess: 'Disabled' + } +} + +resource privateDnsZone 'Microsoft.Network/privateDnsZones@2024-06-01' = { + name: privateDnsZoneName + location: 'global' +} + +resource privateDnsLink 'Microsoft.Network/privateDnsZones/virtualNetworkLinks@2024-06-01' = { + parent: privateDnsZone + name: 'link-${prefix}' + location: 'global' + properties: { + registrationEnabled: false + virtualNetwork: { + id: vnet.id + } + } +} + +resource privateEndpoint 'Microsoft.Network/privateEndpoints@2024-05-01' = { + name: 'pe-${clusterName}' + location: location + properties: { + subnet: { + id: '${vnet.id}/subnets/snet-pe' + } + privateLinkServiceConnections: [ + { + name: 'plsc-${clusterName}' + properties: { + privateLinkServiceId: cluster.id + groupIds: [ + 'MongoCluster' + ] + } + } + ] + } +} + +resource privateDnsZoneGroup 'Microsoft.Network/privateEndpoints/privateDnsZoneGroups@2024-05-01' = { + parent: privateEndpoint + name: 'default' + properties: { + privateDnsZoneConfigs: [ + { + name: 'mongocluster' + properties: { + privateDnsZoneId: privateDnsZone.id + } + } + ] + } +} + +resource acr 'Microsoft.ContainerRegistry/registries@2023-07-01' = { + name: '${prefix}acr${suffix}' + location: location + sku: { + name: 'Basic' + } + properties: { + adminUserEnabled: false + } +} + +resource aks 'Microsoft.ContainerService/managedClusters@2024-09-01' = { + name: 'aks-${prefix}' + location: location + identity: { + type: 'SystemAssigned' + } + properties: { + dnsPrefix: 'aks-${prefix}-${suffix}' + // Workload identity lets the Airflow export pods use a managed identity (lake.bicep). + oidcIssuerProfile: { + enabled: true + } + securityProfile: { + workloadIdentity: { + enabled: true + } + } + agentPoolProfiles: [ + { + name: 'system' + mode: 'System' + count: nodeCount + vmSize: nodeVmSize + osType: 'Linux' + vnetSubnetID: '${vnet.id}/subnets/snet-aks' + } + ] + networkProfile: { + networkPlugin: 'azure' + networkPluginMode: 'overlay' + podCidr: '192.168.0.0/16' + serviceCidr: '172.16.0.0/16' + dnsServiceIP: '172.16.0.10' + } + } +} + +// AcrPull for the kubelet identity so pods can pull the lab image. +resource acrPull 'Microsoft.Authorization/roleAssignments@2022-04-01' = { + name: guid(acr.id, aks.id, 'acrpull') + scope: acr + properties: { + principalId: aks.properties.identityProfile.kubeletidentity.objectId + principalType: 'ServicePrincipal' + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '7f951dda-4ed3-4680-a7ca-43fe172d538d') + } +} + +// AKS needs Network Contributor on its subnet when using a custom VNet. +resource aksSubnetRole 'Microsoft.Authorization/roleAssignments@2022-04-01' = { + name: guid(vnet.id, aks.id, 'netcontrib') + scope: vnet + properties: { + principalId: aks.identity.principalId + principalType: 'ServicePrincipal' + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '4d97b98b-1d4f-4787-a291-c67834d212e7') + } +} + +output clusterName string = cluster.name +output acrName string = acr.name +output acrLoginServer string = acr.properties.loginServer +output aksName string = aks.name diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml new file mode 100644 index 00000000..83c70204 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml @@ -0,0 +1,34 @@ +# Lets the Airflow scheduler (LocalExecutor runs tasks in it) manage export +# pods in the cslab namespace. +apiVersion: rbac.authorization.k8s.io/v1 +kind: Role +metadata: + name: airflow-pod-launcher + namespace: cslab +rules: + - apiGroups: [""] + resources: ["pods"] + verbs: ["create", "get", "list", "watch", "delete", "patch"] + - apiGroups: [""] + resources: ["pods/log"] + verbs: ["get", "list"] + - apiGroups: [""] + resources: ["pods/exec"] + verbs: ["create", "get"] + - apiGroups: [""] + resources: ["events"] + verbs: ["list", "watch"] +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: RoleBinding +metadata: + name: airflow-pod-launcher + namespace: cslab +subjects: + - kind: ServiceAccount + name: airflow-scheduler + namespace: airflow +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: Role + name: airflow-pod-launcher diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml new file mode 100644 index 00000000..d369114a --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml @@ -0,0 +1,49 @@ +apiVersion: v1 +kind: Namespace +metadata: + name: cslab +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: cs-consumer + namespace: cslab +spec: + replicas: 1 # one stream owner per CONSUMER_ID; checkpoint is shared + strategy: + type: Recreate # avoid two pods committing the same checkpoint during rollout + selector: + matchLabels: + app: cs-consumer + template: + metadata: + labels: + app: cs-consumer + spec: + terminationGracePeriodSeconds: 30 + containers: + - name: consumer + image: ${IMAGE} + command: ["python", "consumer.py"] + env: + - name: MONGO_URI + valueFrom: + secretKeyRef: + name: docdb + key: uri + - name: MONGO_DB + value: cslab + - name: MONGO_COLLECTION + value: orders + - name: CONSUMER_ID + value: orders-consumer + - name: FULL_DOCUMENT + value: updateLookup + - name: SINK_BATCH + value: "200" + resources: + requests: + cpu: 250m + memory: 256Mi + limits: + memory: 512Mi diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/job.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/job.yaml new file mode 100644 index 00000000..4c2b6789 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/job.yaml @@ -0,0 +1,44 @@ +# Generic one-shot job: SCRIPT is generator.py, verify.py or probe.py. +apiVersion: batch/v1 +kind: Job +metadata: + name: ${JOB_NAME} + namespace: cslab +spec: + backoffLimit: 0 + ttlSecondsAfterFinished: 3600 + template: + spec: + restartPolicy: Never + containers: + - name: job + image: ${IMAGE} + command: ["python", "${SCRIPT}"] + env: + - name: MONGO_URI + valueFrom: + secretKeyRef: + name: docdb + key: uri + - name: MONGO_DB + value: cslab + - name: MONGO_COLLECTION + value: orders + - name: RUN_ID + value: "${RUN_ID}" + - name: DOCS + value: "${DOCS}" + - name: WORKERS + value: "${WORKERS}" + - name: RATE + value: "${RATE}" + - name: SINK + value: "${SINK}" + - name: PAD_BYTES + value: "${PAD_BYTES}" + resources: + requests: + cpu: "1" + memory: 256Mi + limits: + memory: 1Gi diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml new file mode 100644 index 00000000..14cb646a --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml @@ -0,0 +1,55 @@ +# Service account federated to the lake identity (infra/lake.bicep) and a +# one-shot job that runs lake_export.py or verify_lake.py with it. +apiVersion: v1 +kind: ServiceAccount +metadata: + name: cs-lake + namespace: cslab + annotations: + azure.workload.identity/client-id: "${LAKE_CLIENT_ID}" +--- +apiVersion: batch/v1 +kind: Job +metadata: + name: ${JOB_NAME} + namespace: cslab +spec: + backoffLimit: 0 + ttlSecondsAfterFinished: 3600 + template: + metadata: + labels: + azure.workload.identity/use: "true" + spec: + serviceAccountName: cs-lake + restartPolicy: Never + containers: + - name: job + image: ${IMAGE} + command: ["python", "${SCRIPT}"] + env: + - name: MONGO_URI + valueFrom: + secretKeyRef: + name: docdb + key: uri + - name: LAKE_URL + value: "${LAKE_URL}" + - name: LAKE_FILESYSTEM + value: cdc + - name: LAKE_PREFIX + value: "${LAKE_PREFIX}" + - name: STREAM_ID + value: "${LAKE_PREFIX}" + - name: RUN_ID + value: "${RUN_ID}" + - name: MAX_EVENTS + value: "${MAX_EVENTS}" + - name: FAULT_EXIT_AFTER_UPLOAD + value: "${FAULT_EXIT_AFTER_UPLOAD}" + resources: + requests: + cpu: "1" + memory: 512Mi + limits: + memory: 4Gi diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/toolbox.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/toolbox.yaml new file mode 100644 index 00000000..740f3c4a --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/toolbox.yaml @@ -0,0 +1,22 @@ +# Interactive pod for ad-hoc probes: kubectl -n cslab exec -i cs-toolbox -- python - < script.py +apiVersion: v1 +kind: Pod +metadata: + name: cs-toolbox + namespace: cslab +spec: + containers: + - name: toolbox + image: ${IMAGE} + command: ["sleep", "infinity"] + workingDir: /app + env: + - name: MONGO_URI + valueFrom: + secretKeyRef: + name: docdb + key: uri + - name: MONGO_DB + value: cslab + - name: PYTHONWARNINGS + value: ignore diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml b/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml new file mode 100644 index 00000000..6454e587 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml @@ -0,0 +1,6 @@ +title: Python change stream consumer on AKS +description: Runnable PyMongo change stream probe, consumer, Airflow Parquet export, load generator and verifiers for Azure DocumentDB behind a private endpoint. +kind: runnable +used_by: +- index +- airflow-parquet diff --git a/tests/docs/test_faceted_discovery.py b/tests/docs/test_faceted_discovery.py index df1565e1..996e1bf6 100644 --- a/tests/docs/test_faceted_discovery.py +++ b/tests/docs/test_faceted_discovery.py @@ -160,11 +160,11 @@ def test_explore_has_exact_member_data_and_one_card_per_topic(taxonomy, catalog) assert PurePosixPath("explore/index.md") in pages page = pages[PurePosixPath("explore/index.md")] assert yaml.safe_load(page.split("---", 2)[1])["title"] == "글 찾기" - assert "41개 주제 · 66개 문서" in page + assert "42개 주제 · 68개 문서" in page parser = Elements(page) - assert len(parser.select("data-explore-topic")) == len(catalog.topics) == 41 + assert len(parser.select("data-explore-topic")) == len(catalog.topics) == 42 members = parser.select("data-explore-member") - assert len(members) == len(catalog.documents) == 66 + assert len(members) == len(catalog.documents) == 68 by_path = {row["data-explore-member"]: row for row in members} for doc in catalog.documents: row = by_path[doc.relative_path.as_posix()] @@ -230,7 +230,7 @@ def test_home_navigation_and_article_chips_use_explore(taxonomy, catalog): nav = yaml.safe_load((ROOT / "docs/.nav.yml").read_text())["nav"] assert [next(iter(item)) for item in nav if "glob" not in item] == ["홈", "서비스별 보기", "글 찾기", "기여하기"] home = build_home_page((ROOT / "docs/index.md").read_text(), catalog.documents, taxonomy, catalog=catalog) - assert "41개 주제 · 66개 문서" in home + assert "42개 주제 · 68개 문서" in home assert "전체 글" not in home and "태그별 보기" not in home assert "(explore/index.md)" in home doc = catalog.documents[0] @@ -253,7 +253,7 @@ def test_explore_assets_are_configured_and_accessible(): def test_service_discovery_reports_topics_and_documents(taxonomy, catalog): pages = build_index_pages(catalog.documents, taxonomy, catalog=catalog) - assert "41개 주제 · 66개 문서" in pages[PurePosixPath("services/index.md")] + assert "42개 주제 · 68개 문서" in pages[PurePosixPath("services/index.md")] documents = [doc for doc in catalog.documents if "azure-monitor" in doc.metadata["services"]] topics = {catalog.by_document[doc.relative_path].entry.relative_path for doc in documents} assert f"{len(topics)}개 주제 · {len(documents)}개 문서" in pages[PurePosixPath("services/azure-monitor/index.md")] diff --git a/tests/docs/test_topics.py b/tests/docs/test_topics.py index 45e71e1f..df00b85f 100644 --- a/tests/docs/test_topics.py +++ b/tests/docs/test_topics.py @@ -1065,8 +1065,8 @@ def test_repository_uses_only_canonical_topic_packages() -> None: ROOT / "docs", load_taxonomy(ROOT / "docs-taxonomy.yml") ) - assert len(catalog.documents) == 66 - assert len(catalog.topics) == 41 + assert len(catalog.documents) == 68 + assert len(catalog.topics) == 42 assert not any( (ROOT / "docs" / name).exists() for name in ("cases", "guides", "labs", "research") From 61415c9d2a953b81ec6a18e27881ef8c8a66273e Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 17:49:31 +0900 Subject: [PATCH 02/14] =?UTF-8?q?docs(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20sample=20=EC=9D=B8=ED=94=84=EB=9D=BC=EB=A5=BC=20azd=EB=A1=9C?= =?UTF-8?q?=20=EB=B0=B0=ED=8F=AC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - azure.yaml과 구독 범위 main.bicep을 추가하고 기존 템플릿은 cluster.bicep으로 옮김 - 관리자 비밀번호는 azd .env에 넣지 않고 셸 환경 변수로만 전달 - 이미지, 파드와 Airflow는 README의 az acr build, kubectl, helm 절차로 배포 - README 절차대로 새 환경에서 r1과 af1을 다시 실행해 누락·중복 0건 확인 - 수동으로 만드는 namespace와 서비스 계정을 매니페스트에서 제거 Co-Authored-By: Claude Opus 5.5 --- .../azure-documentdb/change-streams/index.md | 56 +++-- .../samples/python-aks/README.md | 140 ++++++++--- .../samples/python-aks/app/generator.py | 2 +- .../samples/python-aks/azure.yaml | 9 + .../samples/python-aks/infra/cluster.bicep | 217 +++++++++++++++++ .../samples/python-aks/infra/lake.bicep | 2 +- .../samples/python-aks/infra/main.bicep | 228 +++--------------- .../python-aks/infra/main.parameters.json | 9 + .../samples/python-aks/k8s/consumer.yaml | 5 - .../samples/python-aks/k8s/lake.yaml | 12 +- 10 files changed, 422 insertions(+), 258 deletions(-) create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/azure.yaml create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.parameters.json diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index 295858e8..44a567b3 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -21,6 +21,8 @@ official_sources: url: https://learn.microsoft.com/azure/documentdb/release-notes - title: Microsoft.DocumentDB mongoClusters (Bicep reference) url: https://learn.microsoft.com/azure/templates/microsoft.documentdb/2026-06-01/mongoclusters + - title: Work with Azure Developer CLI environment variables + url: https://learn.microsoft.com/azure/developer/azure-developer-cli/manage-environment-variables - title: AzureCosmosDB/changestream-driver-compatibility url: https://github.com/AzureCosmosDB/changestream-driver-compatibility last_verified: 2026-10-02 @@ -48,7 +50,7 @@ Parquet으로 쓰는 방식은 [Airflow DAG로 Parquet 적재](airflow-parquet/i ## 목표 -- 프라이빗 엔드포인트만 열린 DocumentDB 클러스터와 AKS를 Bicep으로 배포합니다. +- 프라이빗 엔드포인트만 열린 DocumentDB 클러스터와 AKS를 azd로 배포합니다. - `probe.py`로 change stream 옵션별 동작을 확인합니다. - 생성기가 만든 이벤트와 sink에 기록된 이벤트를 `verify.py`로 대조해 누락, 중복, 순서 역전과 지연을 숫자로 남깁니다. @@ -56,7 +58,7 @@ Parquet으로 쓰는 방식은 [Airflow DAG로 Parquet 적재](airflow-parquet/i ## 사전 조건 - Azure 구독에서 리소스 그룹을 만들 수 있는 권한 -- Azure CLI, `kubectl`, `envsubst` +- Azure Developer CLI(azd) 1.29 이상, Azure CLI, `kubectl`, `envsubst` - 이 저장소의 [python-aks sample](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md) 이 문서의 결과는 2026-10-02에 아래 환경에서 측정했습니다. @@ -73,41 +75,53 @@ Parquet으로 쓰는 방식은 [Airflow DAG로 Parquet 적재](airflow-parquet/i - 클러스터, AKS 노드, ACR과 프라이빗 엔드포인트는 실행하는 동안 과금됩니다. 측정이 끝나면 바로 리소스 그룹을 삭제합니다. -- 관리자 비밀번호와 연결 문자열은 Kubernetes Secret에만 넣습니다. sample의 - `.env`는 `.gitignore` 대상이며 커밋하지 않습니다. +- 관리자 비밀번호는 셸 환경 변수로만 azd에 넘기고 연결 문자열은 Kubernetes + Secret에만 넣습니다. Learn은 azd 환경의 `.env` 파일에 비밀을 넣지 말라고 + 경고합니다. 그 파일이 있는 `.azure/`는 `.gitignore` 대상이며 커밋하지 않습니다. - 생성기는 지정한 데이터베이스(`cslab`)의 `orders` 컬렉션에 직접 씁니다. 운영 클러스터를 대상으로 실행하지 않습니다. - 이 문서에 나오는 리소스 이름과 레지스트리 주소는 가상 값입니다. ## 배포 -sample 디렉터리에서 실행합니다. +Azure 리소스는 azd로 배포하고 이미지와 파드는 `az acr build`와 `kubectl`로 직접 +배포합니다. 명령은 sample 디렉터리에서 실행하며 전체 절차는 sample README에 +있습니다. ```bash -RG=rg-docdb-changestream -az group create -n $RG -l eastus2 -az deployment group create -g $RG -f infra/main.bicep \ - -p adminPassword="$ADMIN_PASSWORD" +azd env new docdb-changestream +azd env set AZURE_LOCATION eastus2 +export DOCDB_ADMIN_PASSWORD="Cs$(openssl rand -hex 12)Aa9" +azd provision -ACR=docdbcsacrexample -az acr build -r $ACR -t cslab:v1 app/ -az aks get-credentials -g $RG -n aks-docdbcs +set -a; eval "$(azd env get-values)"; set +a +export IMAGE=$AZURE_CONTAINER_REGISTRY_ENDPOINT/cslab:v1 +az acr build -r $AZURE_CONTAINER_REGISTRY_NAME -t cslab:v1 app/ +az aks get-credentials -g $AZURE_RESOURCE_GROUP -n $AZURE_AKS_CLUSTER_NAME kubectl create namespace cslab +MONGO_URI="${DOCDB_CONNECTION_STRING/:/$DOCDB_ADMIN_USER:$DOCDB_ADMIN_PASSWORD}" kubectl -n cslab create secret generic docdb --from-literal=uri="$MONGO_URI" ``` -`main.bicep`은 VNet, `Microsoft.DocumentDB/mongoClusters` 클러스터(M30, shard 1개), -프라이빗 엔드포인트와 사설 DNS 영역, ACR, AKS를 만듭니다. 프라이빗 엔드포인트의 -group ID는 `MongoCluster`입니다. Learn은 프라이빗 엔드포인트로 연결할 때 -`mongodb+srv` 형식의 연결 문자열을 쓰라고 안내합니다. AKS 파드에서는 클러스터 -호스트 이름이 사설 DNS 영역을 거쳐 프라이빗 IP로 확인됩니다. +`azd provision`은 `rg-<환경 이름>` 리소스 그룹을 만들고 `infra/cluster.bicep`과 +`infra/lake.bicep`을 배포합니다. `cluster.bicep`은 VNet, +`Microsoft.DocumentDB/mongoClusters` 클러스터(M30, shard 1개), 프라이빗 +엔드포인트와 사설 DNS 영역, ACR, AKS를 만듭니다. `lake.bicep`은 +[Airflow DAG로 Parquet 적재](airflow-parquet/index.md)에서 쓰는 ADLS Gen2와 워크로드 +ID를 만듭니다. 배포 출력은 azd 환경 값으로 저장되고 위의 `azd env get-values`로 +셸 변수가 됩니다. + +프라이빗 엔드포인트의 group ID는 `MongoCluster`입니다. Learn은 프라이빗 +엔드포인트로 연결할 때 `mongodb+srv` 형식의 연결 문자열을 쓰라고 안내합니다. 배포 +출력의 연결 문자열은 ``와 `` 자리 표시자를 담고 있어 위 명령이 +관리자 계정으로 바꿉니다. AKS 파드에서는 클러스터 호스트 이름이 사설 DNS 영역을 +거쳐 프라이빗 IP로 확인됩니다. 감시할 컬렉션은 consumer보다 먼저 만듭니다. 컬렉션이 없으면 `watch()`가 `NamespaceNotFound`(code 26)로 실패합니다. ```bash -export IMAGE=docdbcsacrexample.azurecr.io/cslab:v1 envsubst < k8s/toolbox.yaml | kubectl apply -f - kubectl -n cslab exec cs-toolbox -- python -c \ "from common import *; get_database(get_client('setup')).create_collection('orders')" @@ -212,10 +226,11 @@ Learn은 이력 재개에 대해 다음을 설명합니다. 기본 change stream ## 정리 ```bash -az group delete -n $RG +azd down +kubectl config delete-context $AZURE_AKS_CLUSTER_NAME ``` -삭제 전에 `kubectl config delete-context`로 로컬 kubeconfig의 AKS 항목도 지웁니다. +`azd down`은 리소스 그룹과 그 안의 리소스를 모두 삭제합니다. ## 문제 해결 @@ -236,4 +251,5 @@ az group delete -n $RG - [Compute and storage configurations](https://learn.microsoft.com/azure/documentdb/compute-storage) - [Release notes for Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/release-notes) - [Microsoft.DocumentDB mongoClusters Bicep reference](https://learn.microsoft.com/azure/templates/microsoft.documentdb/2026-06-01/mongoclusters) +- [Work with Azure Developer CLI environment variables](https://learn.microsoft.com/azure/developer/azure-developer-cli/manage-environment-variables) - [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index f6cb5e26..fb03221f 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -6,48 +6,109 @@ through a private endpoint. | Path | Purpose | | --- | --- | -| `infra/main.bicep` | VNet, DocumentDB cluster (M30, one shard), private endpoint and DNS zone, ACR, AKS | +| `azure.yaml` | azd project. azd provisions the Azure resources only | +| `infra/main.bicep` | azd entry point. Creates the resource group, deploys the two modules below and outputs the values the Kubernetes steps need | +| `infra/cluster.bicep` | VNet, DocumentDB cluster (M30, one shard), private endpoint and DNS zone, ACR, AKS with OIDC issuer and workload identity | +| `infra/lake.bicep` | ADLS Gen2 account with public access and shared keys disabled, blob and dfs private endpoints, a managed identity federated to the `cs-lake` service account | | `app/probe.py` | Checks which change stream options and event fields the cluster returns | | `app/consumer.py` | Long-running consumer. Writes events to a sink collection and keeps the resume token in a checkpoint collection | | `app/generator.py` | Deterministic insert, update, replace and delete workload | | `app/verify.py` | Compares the generator's expected events with a sink collection | -| `infra/lake.bicep` | ADLS Gen2 account with public access and shared keys disabled, blob and dfs private endpoints, a managed identity federated to the `cs-lake` service account | | `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes Parquet chunks and exits | | `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | | `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | | `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job and service account, RBAC for the Airflow scheduler | -## Run +Azure resources are provisioned with azd. Images, pods and Airflow are +deployed by hand with `az acr build`, `kubectl` and `helm` so each step can be +inspected. Run every command from this directory. + +## Prerequisites + +- Azure Developer CLI 1.29 or later and Azure CLI +- `kubectl`, `helm` and `envsubst` (gettext) +- Permission to create a resource group and role assignments in the + subscription -The commands use fictitious names. Replace them with your own values. +## 1. Provision Azure resources with azd ```bash -RG=rg-docdb-changestream -az group create -n $RG -l eastus2 -az deployment group create -g $RG -f infra/main.bicep \ - -p adminPassword="$ADMIN_PASSWORD" +azd auth login +azd env new docdb-changestream +azd env set AZURE_LOCATION eastus2 +export DOCDB_ADMIN_PASSWORD="Cs$(openssl rand -hex 12)Aa9" +azd provision --preview +azd provision +``` + +`infra/main.parameters.json` maps `adminPassword` to `DOCDB_ADMIN_PASSWORD`. +Export it in the shell instead of storing it with `azd env set`, because azd +environment values are kept in a plain-text `.env` file. azd stops with +`missing required inputs` when the variable is not set. Keep the value +somewhere safe: every later `azd provision` needs the same password, and a +different one changes the administrator password. + +`azd provision` creates `rg-` and deploys `cluster.bicep` +and `lake.bicep` into it. The lake module receives the AKS OIDC issuer from +the cluster module, so one run sets up the workload identity federation too. + +The deployment outputs are saved as azd environment values: + +| Value | Used for | +| --- | --- | +| `AZURE_RESOURCE_GROUP`, `AZURE_AKS_CLUSTER_NAME` | `az aks get-credentials` | +| `AZURE_CONTAINER_REGISTRY_NAME`, `AZURE_CONTAINER_REGISTRY_ENDPOINT` | Image builds and image names | +| `DOCDB_CONNECTION_STRING`, `DOCDB_ADMIN_USER` | `MONGO_URI` with the exported password. The connection string keeps the `` and `` placeholders | +| `LAKE_URL` | `k8s/lake.yaml` and `airflow/values.yaml` | +| `LAKE_CLIENT_ID` | Annotation on the `cs-lake` service account | + +`.azure/` holds the environment values and is ignored by Git. Do not commit +it. + +Load the values into the shell. Later steps read them as variables. + +```bash +set -a; eval "$(azd env get-values)"; set +a +``` -ACR=docdbcsacrexample -az acr build -r $ACR -t cslab:v1 app/ -az aks get-credentials -g $RG -n aks-docdbcs +## 2. Build the images +```bash +export ACR=$AZURE_CONTAINER_REGISTRY_ENDPOINT +export IMAGE=$ACR/cslab:v1 +az acr build -r $AZURE_CONTAINER_REGISTRY_NAME -t cslab:v1 app/ +az acr build -r $AZURE_CONTAINER_REGISTRY_NAME -t cslab-airflow:v1 airflow/ +``` + +The AKS kubelet identity has `AcrPull` on the registry, so the pods need no +pull secret. + +## 3. Connect to AKS and store the connection string + +```bash +az aks get-credentials -g $AZURE_RESOURCE_GROUP -n $AZURE_AKS_CLUSTER_NAME kubectl create namespace cslab + +MONGO_URI="${DOCDB_CONNECTION_STRING/:/$DOCDB_ADMIN_USER:$DOCDB_ADMIN_PASSWORD}" kubectl -n cslab create secret generic docdb --from-literal=uri="$MONGO_URI" ``` -`MONGO_URI` uses the cluster's `mongodb+srv://` connection string. The cluster -host name resolves to the private endpoint address inside the VNet. +The connection string uses `mongodb+srv://`. The cluster host name resolves to +the private endpoint address inside the VNet, so it works only from AKS. + +## 4. Run the consumer The watched collection must exist before the consumer starts. The cluster under test returned `NamespaceNotFound` (code 26) for a missing collection. ```bash -export IMAGE=docdbcsacrexample.azurecr.io/cslab:v1 envsubst < k8s/toolbox.yaml | kubectl apply -f - +kubectl -n cslab wait --for=condition=Ready pod/cs-toolbox --timeout=5m kubectl -n cslab exec cs-toolbox -- python -c \ "from common import *; get_database(get_client('setup')).create_collection('orders')" envsubst < k8s/consumer.yaml | kubectl apply -f - +kubectl -n cslab rollout status deployment/cs-consumer JOB_NAME=gen-r1 SCRIPT=generator.py RUN_ID=r1 DOCS=100000 WORKERS=16 RATE=0 \ envsubst < k8s/job.yaml | kubectl apply -f - @@ -74,29 +135,28 @@ Run `probe.py` the same way with `SCRIPT=probe.py`. still terminating. The `Recreate` strategy covers rollouts only, so for a short time two consumers can run. The upsert keeps the sink correct. -## Airflow to Parquet +## 5. Export to Parquet with Airflow -AKS needs the OIDC issuer and workload identity. `infra/main.bicep` enables -both. Deploy the lake next to the cluster. +The storage account, its private endpoints and the federated identity already +exist from step 1. Create the `cs-lake` service account that the identity is +federated to. The export pods use it to get a workload identity token. ```bash -ISSUER=$(az aks show -g $RG -n aks-docdbcs --query oidcIssuerProfile.issuerUrl -o tsv) -az deployment group create -g $RG -f infra/lake.bicep -p oidcIssuerUrl="$ISSUER" -export LAKE_CLIENT_ID= -export LAKE_URL=https://docdbcslakeexample.dfs.core.windows.net/ +kubectl -n cslab create serviceaccount cs-lake +kubectl -n cslab annotate serviceaccount cs-lake \ + azure.workload.identity/client-id="$LAKE_CLIENT_ID" ``` Install Airflow with the DAG baked into the image. The values use -`LocalExecutor`, so tasks run in the scheduler pod and its service account -launches the export pods in `cslab`. +`LocalExecutor`, so tasks run in the scheduler pod. `k8s/airflow-rbac.yaml` +lets the scheduler's service account create the export pods in `cslab`. ```bash -az acr build -r $ACR -t cslab-airflow:v1 airflow/ kubectl apply -f k8s/airflow-rbac.yaml -ACR=docdbcsacrexample.azurecr.io AIRFLOW_IMAGE_TAG=v1 envsubst < airflow/values.yaml > values.rendered.yaml helm repo add apache-airflow https://airflow.apache.org -helm install airflow apache-airflow/airflow --version 1.22.0 -n airflow --create-namespace \ - -f values.rendered.yaml --wait --timeout 10m +AIRFLOW_IMAGE_TAG=v1 envsubst < airflow/values.yaml | \ + helm install airflow apache-airflow/airflow --version 1.22.0 \ + -n airflow --create-namespace -f - --wait --timeout 10m ``` With chart defaults the database migration job is a post-install hook, so @@ -106,12 +166,27 @@ With chart defaults the database migration job is a post-install hook, so [chart documentation](https://airflow.apache.org/docs/helm-chart/stable/index.html) advises for `--wait`. -`lake.yaml` creates the `cs-lake` service account. Apply it once before the -first DAG run, for example with the verifier job. +The DAG is paused when Airflow first loads it. Unpause it to start the +5-minute schedule. ```bash -JOB_NAME=verify-lake-r1 SCRIPT=verify_lake.py LAKE_PREFIX=orders RUN_ID=r1 MAX_EVENTS=0 \ +kubectl -n airflow exec airflow-scheduler-0 -c scheduler -- \ + airflow dags unpause change_stream_to_parquet +kubectl -n airflow exec airflow-scheduler-0 -c scheduler -- \ + airflow dags list-runs change_stream_to_parquet +``` + +The first run has no checkpoint, so it starts at the current position of the +stream and saves it. Write events after that run and compare them with the +Parquet files once a later run has exported them. + +```bash +JOB_NAME=gen-af1 SCRIPT=generator.py RUN_ID=af1 DOCS=100000 WORKERS=16 RATE=0 \ + envsubst < k8s/job.yaml | kubectl apply -f - +# After the next DAG run finishes: +JOB_NAME=verify-lake-af1 SCRIPT=verify_lake.py LAKE_PREFIX=orders RUN_ID=af1 MAX_EVENTS=0 \ FAULT_EXIT_AFTER_UPLOAD=0 envsubst < k8s/lake.yaml | kubectl apply -f - +kubectl -n cslab logs job/verify-lake-af1 ``` Export behavior: @@ -140,5 +215,8 @@ Export behavior: ## Clean up ```bash -az group delete -n $RG +azd down +kubectl config delete-context $AZURE_AKS_CLUSTER_NAME ``` + +`azd down` deletes the resource group and everything in it. diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py index b1fc938b..b570b48c 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py @@ -69,7 +69,7 @@ def main() -> None: workers = int(os.getenv("WORKERS", "4")) rate = float(os.getenv("RATE", "0")) # total ops/s, 0 = unthrottled # Filler on inserts and replaces so the change log grows like a real payload. - pad_bytes = int(os.getenv("PAD_BYTES", "0")) + pad_bytes = int(os.getenv("PAD_BYTES") or "0") # envsubst leaves "" when unset client = get_client(f"cs-generator-{run_id}") db = get_database(client) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/azure.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/azure.yaml new file mode 100644 index 00000000..ac509444 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/azure.yaml @@ -0,0 +1,9 @@ +# azd provisions the Azure resources only. Images and Kubernetes resources are +# deployed by hand with az acr build, kubectl and helm (see README.md). +name: docdb-changestream-lab +requiredVersions: + azd: ">=1.29.0" +infra: + provider: bicep + path: infra + module: main diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep new file mode 100644 index 00000000..507c14b3 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep @@ -0,0 +1,217 @@ +// Azure DocumentDB (vCore) + AKS + ACR for the Python change stream lab. +// Deployed by main.bicep into the azd resource group. +// The cluster has public network access disabled and is reached from AKS +// through a private endpoint and the privatelink.mongocluster.cosmos.azure.com zone. + +targetScope = 'resourceGroup' + +@description('Region for every resource.') +param location string = resourceGroup().location + +@description('Short prefix used in resource names.') +param prefix string = 'docdbcs' + +@description('DocumentDB administrator user name.') +param adminUserName string = 'csadmin' + +@secure() +@description('DocumentDB administrator password.') +param adminPassword string + +@description('DocumentDB compute tier, for example M30.') +param clusterTier string = 'M30' + +@description('Storage size per shard in GiB.') +param storageSizeGb int = 32 + +@description('AKS node VM size.') +param nodeVmSize string = 'Standard_D4s_v6' + +@description('AKS node count.') +param nodeCount int = 2 + +var suffix = uniqueString(resourceGroup().id) +var clusterName = '${prefix}-${suffix}' +var privateDnsZoneName = 'privatelink.mongocluster.cosmos.azure.com' + +resource vnet 'Microsoft.Network/virtualNetworks@2024-05-01' = { + name: 'vnet-${prefix}' + location: location + properties: { + addressSpace: { + addressPrefixes: [ + '10.40.0.0/16' + ] + } + subnets: [ + { + name: 'snet-aks' + properties: { + addressPrefix: '10.40.0.0/22' + } + } + { + name: 'snet-pe' + properties: { + addressPrefix: '10.40.8.0/24' + privateEndpointNetworkPolicies: 'Disabled' + } + } + ] + } +} + +resource cluster 'Microsoft.DocumentDB/mongoClusters@2025-09-01' = { + name: clusterName + location: location + properties: { + administrator: { + userName: adminUserName + password: adminPassword + } + compute: { + tier: clusterTier + } + storage: { + sizeGb: storageSizeGb + } + sharding: { + shardCount: 1 + } + highAvailability: { + targetMode: 'Disabled' + } + publicNetworkAccess: 'Disabled' + } +} + +resource privateDnsZone 'Microsoft.Network/privateDnsZones@2024-06-01' = { + name: privateDnsZoneName + location: 'global' +} + +resource privateDnsLink 'Microsoft.Network/privateDnsZones/virtualNetworkLinks@2024-06-01' = { + parent: privateDnsZone + name: 'link-${prefix}' + location: 'global' + properties: { + registrationEnabled: false + virtualNetwork: { + id: vnet.id + } + } +} + +resource privateEndpoint 'Microsoft.Network/privateEndpoints@2024-05-01' = { + name: 'pe-${clusterName}' + location: location + properties: { + subnet: { + id: '${vnet.id}/subnets/snet-pe' + } + privateLinkServiceConnections: [ + { + name: 'plsc-${clusterName}' + properties: { + privateLinkServiceId: cluster.id + groupIds: [ + 'MongoCluster' + ] + } + } + ] + } +} + +resource privateDnsZoneGroup 'Microsoft.Network/privateEndpoints/privateDnsZoneGroups@2024-05-01' = { + parent: privateEndpoint + name: 'default' + properties: { + privateDnsZoneConfigs: [ + { + name: 'mongocluster' + properties: { + privateDnsZoneId: privateDnsZone.id + } + } + ] + } +} + +resource acr 'Microsoft.ContainerRegistry/registries@2023-07-01' = { + name: '${prefix}acr${suffix}' + location: location + sku: { + name: 'Basic' + } + properties: { + adminUserEnabled: false + } +} + +resource aks 'Microsoft.ContainerService/managedClusters@2024-09-01' = { + name: 'aks-${prefix}' + location: location + identity: { + type: 'SystemAssigned' + } + properties: { + dnsPrefix: 'aks-${prefix}-${suffix}' + // Workload identity lets the Airflow export pods use a managed identity (lake.bicep). + oidcIssuerProfile: { + enabled: true + } + securityProfile: { + workloadIdentity: { + enabled: true + } + } + agentPoolProfiles: [ + { + name: 'system' + mode: 'System' + count: nodeCount + vmSize: nodeVmSize + osType: 'Linux' + vnetSubnetID: '${vnet.id}/subnets/snet-aks' + } + ] + networkProfile: { + networkPlugin: 'azure' + networkPluginMode: 'overlay' + podCidr: '192.168.0.0/16' + serviceCidr: '172.16.0.0/16' + dnsServiceIP: '172.16.0.10' + } + } +} + +// AcrPull for the kubelet identity so pods can pull the lab image. +resource acrPull 'Microsoft.Authorization/roleAssignments@2022-04-01' = { + name: guid(acr.id, aks.id, 'acrpull') + scope: acr + properties: { + principalId: aks.properties.identityProfile.kubeletidentity.objectId + principalType: 'ServicePrincipal' + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '7f951dda-4ed3-4680-a7ca-43fe172d538d') + } +} + +// AKS needs Network Contributor on its subnet when using a custom VNet. +resource aksSubnetRole 'Microsoft.Authorization/roleAssignments@2022-04-01' = { + name: guid(vnet.id, aks.id, 'netcontrib') + scope: vnet + properties: { + principalId: aks.identity.principalId + principalType: 'ServicePrincipal' + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '4d97b98b-1d4f-4787-a291-c67834d212e7') + } +} + +output clusterName string = cluster.name +output acrName string = acr.name +output acrLoginServer string = acr.properties.loginServer +output aksName string = aks.name +output oidcIssuerUrl string = aks.properties.oidcIssuerProfile.issuerURL +// Template with and placeholders. It holds no secret. +output connectionString string = cluster.properties.connectionString diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep index c2b54db2..1d3dc210 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/lake.bicep @@ -8,7 +8,7 @@ targetScope = 'resourceGroup' @description('Region for every resource.') param location string = resourceGroup().location -@description('Short prefix used in resource names. Must match main.bicep.') +@description('Short prefix used in resource names. Must match cluster.bicep.') param prefix string = 'docdbcs' @description('OIDC issuer URL of the AKS cluster.') diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep index dacc9fb0..2b40bd4a 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep @@ -1,213 +1,61 @@ -// Azure DocumentDB (vCore) + AKS + ACR for the Python change stream lab. -// The cluster has public network access disabled and is reached from AKS -// through a private endpoint and the privatelink.mongocluster.cosmos.azure.com zone. +// azd entry point. Creates the resource group, then deploys the cluster, AKS +// and ACR (cluster.bicep) and the ADLS Gen2 landing zone (lake.bicep) into it. +// The outputs become azd environment values that the README's kubectl steps use. -targetScope = 'resourceGroup' +targetScope = 'subscription' -@description('Region for every resource.') -param location string = resourceGroup().location +@minLength(1) +@maxLength(60) +@description('azd environment name. The resource group is rg-.') +param environmentName string -@description('Short prefix used in resource names.') -param prefix string = 'docdbcs' +@description('Region for every resource.') +param location string @description('DocumentDB administrator user name.') param adminUserName string = 'csadmin' @secure() -@description('DocumentDB administrator password.') +@description('DocumentDB administrator password, from the DOCDB_ADMIN_PASSWORD azd environment value.') param adminPassword string -@description('DocumentDB compute tier, for example M30.') -param clusterTier string = 'M30' - -@description('Storage size per shard in GiB.') -param storageSizeGb int = 32 - -@description('AKS node VM size.') -param nodeVmSize string = 'Standard_D4s_v6' - @description('AKS node count.') param nodeCount int = 2 -var suffix = uniqueString(resourceGroup().id) -var clusterName = '${prefix}-${suffix}' -var privateDnsZoneName = 'privatelink.mongocluster.cosmos.azure.com' - -resource vnet 'Microsoft.Network/virtualNetworks@2024-05-01' = { - name: 'vnet-${prefix}' - location: location - properties: { - addressSpace: { - addressPrefixes: [ - '10.40.0.0/16' - ] - } - subnets: [ - { - name: 'snet-aks' - properties: { - addressPrefix: '10.40.0.0/22' - } - } - { - name: 'snet-pe' - properties: { - addressPrefix: '10.40.8.0/24' - privateEndpointNetworkPolicies: 'Disabled' - } - } - ] - } -} - -resource cluster 'Microsoft.DocumentDB/mongoClusters@2025-09-01' = { - name: clusterName - location: location - properties: { - administrator: { - userName: adminUserName - password: adminPassword - } - compute: { - tier: clusterTier - } - storage: { - sizeGb: storageSizeGb - } - sharding: { - shardCount: 1 - } - highAvailability: { - targetMode: 'Disabled' - } - publicNetworkAccess: 'Disabled' - } -} - -resource privateDnsZone 'Microsoft.Network/privateDnsZones@2024-06-01' = { - name: privateDnsZoneName - location: 'global' -} - -resource privateDnsLink 'Microsoft.Network/privateDnsZones/virtualNetworkLinks@2024-06-01' = { - parent: privateDnsZone - name: 'link-${prefix}' - location: 'global' - properties: { - registrationEnabled: false - virtualNetwork: { - id: vnet.id - } - } -} - -resource privateEndpoint 'Microsoft.Network/privateEndpoints@2024-05-01' = { - name: 'pe-${clusterName}' - location: location - properties: { - subnet: { - id: '${vnet.id}/subnets/snet-pe' - } - privateLinkServiceConnections: [ - { - name: 'plsc-${clusterName}' - properties: { - privateLinkServiceId: cluster.id - groupIds: [ - 'MongoCluster' - ] - } - } - ] - } -} - -resource privateDnsZoneGroup 'Microsoft.Network/privateEndpoints/privateDnsZoneGroups@2024-05-01' = { - parent: privateEndpoint - name: 'default' - properties: { - privateDnsZoneConfigs: [ - { - name: 'mongocluster' - properties: { - privateDnsZoneId: privateDnsZone.id - } - } - ] - } -} - -resource acr 'Microsoft.ContainerRegistry/registries@2023-07-01' = { - name: '${prefix}acr${suffix}' - location: location - sku: { - name: 'Basic' - } - properties: { - adminUserEnabled: false - } -} - -resource aks 'Microsoft.ContainerService/managedClusters@2024-09-01' = { - name: 'aks-${prefix}' +resource group 'Microsoft.Resources/resourceGroups@2024-03-01' = { + name: 'rg-${environmentName}' location: location - identity: { - type: 'SystemAssigned' - } - properties: { - dnsPrefix: 'aks-${prefix}-${suffix}' - // Workload identity lets the Airflow export pods use a managed identity (lake.bicep). - oidcIssuerProfile: { - enabled: true - } - securityProfile: { - workloadIdentity: { - enabled: true - } - } - agentPoolProfiles: [ - { - name: 'system' - mode: 'System' - count: nodeCount - vmSize: nodeVmSize - osType: 'Linux' - vnetSubnetID: '${vnet.id}/subnets/snet-aks' - } - ] - networkProfile: { - networkPlugin: 'azure' - networkPluginMode: 'overlay' - podCidr: '192.168.0.0/16' - serviceCidr: '172.16.0.0/16' - dnsServiceIP: '172.16.0.10' - } + tags: { + 'azd-env-name': environmentName } } -// AcrPull for the kubelet identity so pods can pull the lab image. -resource acrPull 'Microsoft.Authorization/roleAssignments@2022-04-01' = { - name: guid(acr.id, aks.id, 'acrpull') - scope: acr - properties: { - principalId: aks.properties.identityProfile.kubeletidentity.objectId - principalType: 'ServicePrincipal' - roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '7f951dda-4ed3-4680-a7ca-43fe172d538d') +module cluster 'cluster.bicep' = { + name: 'cluster' + scope: group + params: { + location: location + adminUserName: adminUserName + adminPassword: adminPassword + nodeCount: nodeCount } } -// AKS needs Network Contributor on its subnet when using a custom VNet. -resource aksSubnetRole 'Microsoft.Authorization/roleAssignments@2022-04-01' = { - name: guid(vnet.id, aks.id, 'netcontrib') - scope: vnet - properties: { - principalId: aks.identity.principalId - principalType: 'ServicePrincipal' - roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '4d97b98b-1d4f-4787-a291-c67834d212e7') +module lake 'lake.bicep' = { + name: 'lake' + scope: group + params: { + location: location + oidcIssuerUrl: cluster.outputs.oidcIssuerUrl } } -output clusterName string = cluster.name -output acrName string = acr.name -output acrLoginServer string = acr.properties.loginServer -output aksName string = aks.name +output AZURE_RESOURCE_GROUP string = group.name +output AZURE_AKS_CLUSTER_NAME string = cluster.outputs.aksName +output AZURE_CONTAINER_REGISTRY_NAME string = cluster.outputs.acrName +output AZURE_CONTAINER_REGISTRY_ENDPOINT string = cluster.outputs.acrLoginServer +output DOCDB_CLUSTER_NAME string = cluster.outputs.clusterName +output DOCDB_ADMIN_USER string = adminUserName +output DOCDB_CONNECTION_STRING string = cluster.outputs.connectionString +output LAKE_URL string = lake.outputs.dfsEndpoint +output LAKE_CLIENT_ID string = lake.outputs.identityClientId diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.parameters.json b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.parameters.json new file mode 100644 index 00000000..3e0a59af --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.parameters.json @@ -0,0 +1,9 @@ +{ + "$schema": "https://schema.management.azure.com/schemas/2019-04-01/deploymentParameters.json#", + "contentVersion": "1.0.0.0", + "parameters": { + "environmentName": { "value": "${AZURE_ENV_NAME}" }, + "location": { "value": "${AZURE_LOCATION}" }, + "adminPassword": { "value": "${DOCDB_ADMIN_PASSWORD}" } + } +} diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml index d369114a..d2900b47 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/consumer.yaml @@ -1,8 +1,3 @@ -apiVersion: v1 -kind: Namespace -metadata: - name: cslab ---- apiVersion: apps/v1 kind: Deployment metadata: diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml index 14cb646a..66472543 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/lake.yaml @@ -1,13 +1,5 @@ -# Service account federated to the lake identity (infra/lake.bicep) and a -# one-shot job that runs lake_export.py or verify_lake.py with it. -apiVersion: v1 -kind: ServiceAccount -metadata: - name: cs-lake - namespace: cslab - annotations: - azure.workload.identity/client-id: "${LAKE_CLIENT_ID}" ---- +# One-shot job that runs lake_export.py or verify_lake.py as the cs-lake service +# account, which the README creates and federates to the lake identity. apiVersion: batch/v1 kind: Job metadata: From 1e74c7876d72d003d61833eec096c17f3d38ac88 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 18:15:52 +0900 Subject: [PATCH 03/14] =?UTF-8?q?docs(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20=EB=AC=B8=EC=84=9C=EB=A5=BC=20Airflow=20Parquet=20=EC=A0=81?= =?UTF-8?q?=EC=9E=AC=20=EA=B2=80=EC=A6=9D=20=EC=A4=91=EC=8B=AC=EC=9C=BC?= =?UTF-8?q?=EB=A1=9C=20=EC=9E=AC=EA=B5=AC=EC=84=B1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 주제 첫 문서를 research로 바꾸고 기능 소개, 동작 여부, 공식 예제가 놓친 부분, 운영 전 확인 항목 위주로 정리 - 측정 수치와 상시 consumer 기준선, 옵션별 관찰은 measurements 하위 문서로 이동 - 배포와 실행 절차는 sample README로 옮기고 측정 시나리오 매개변수를 추가 Co-Authored-By: Claude Opus 5.5 --- .../images/architecture.svg | 0 .../azure-documentdb/change-streams/index.md | 360 +++++++----------- .../index.md | 160 ++++---- .../samples/python-aks/README.md | 29 +- .../samples/python-aks/sample.yml | 6 +- 5 files changed, 230 insertions(+), 325 deletions(-) rename docs/services/azure-documentdb/change-streams/{airflow-parquet => }/images/architecture.svg (100%) rename docs/services/azure-documentdb/change-streams/{airflow-parquet => measurements}/index.md (51%) diff --git a/docs/services/azure-documentdb/change-streams/airflow-parquet/images/architecture.svg b/docs/services/azure-documentdb/change-streams/images/architecture.svg similarity index 100% rename from docs/services/azure-documentdb/change-streams/airflow-parquet/images/architecture.svg rename to docs/services/azure-documentdb/change-streams/images/architecture.svg diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index 44a567b3..a85c6571 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -1,255 +1,163 @@ --- -title: Azure DocumentDB change stream을 Python으로 AKS에서 검증하기 -description: 프라이빗 엔드포인트 뒤의 Azure DocumentDB(vCore) 클러스터에서 PyMongo change stream의 지원 범위, 재개, 중복 처리와 지연을 AKS consumer로 측정합니다. -document_type: lab -services: [azure-documentdb, azure-kubernetes-service] -technologies: [python, mongodb, kubernetes, bicep] -tags: [build, evaluate] -status: verified +title: Azure DocumentDB change stream을 Airflow DAG로 Parquet에 내리기 +description: AKS 위 Airflow DAG가 Azure DocumentDB(vCore) change stream을 주기적으로 읽어 ADLS Gen2에 Parquet으로 쓸 때 이벤트 누락과 중복이 없는지 확인했습니다. 공식 예제가 다루지 않는 부분과 운영 전에 더 확인할 항목도 정리했습니다. +document_type: research +services: [azure-documentdb, azure-storage, azure-kubernetes-service] +technologies: [python, mongodb, airflow, kubernetes] +tags: [evaluate, build, storage] +status: current verification_status: verified sources_checked_at: 2026-10-02 +published_at: 2026-10-02 official_sources: - title: Change streams in Azure DocumentDB url: https://learn.microsoft.com/azure/documentdb/change-streams - - title: $changeStream (Azure DocumentDB aggregation operator) - url: https://learn.microsoft.com/documentdb/query/operators/aggregation/$changestream - - title: Use Azure Private Link in Azure DocumentDB - url: https://learn.microsoft.com/azure/documentdb/how-to-private-link - - title: Compute and storage configurations for Azure DocumentDB - url: https://learn.microsoft.com/azure/documentdb/compute-storage - - title: Release notes for Azure DocumentDB - url: https://learn.microsoft.com/azure/documentdb/release-notes - - title: Microsoft.DocumentDB mongoClusters (Bicep reference) - url: https://learn.microsoft.com/azure/templates/microsoft.documentdb/2026-06-01/mongoclusters - - title: Work with Azure Developer CLI environment variables - url: https://learn.microsoft.com/azure/developer/azure-developer-cli/manage-environment-variables - title: AzureCosmosDB/changestream-driver-compatibility url: https://github.com/AzureCosmosDB/changestream-driver-compatibility -last_verified: 2026-10-02 -review_cycle_days: 90 -estimated_time: 90m -cost: paid -cleanup_required: true --- -# Azure DocumentDB change stream을 Python으로 AKS에서 검증하기 +# Azure DocumentDB change stream을 Airflow DAG로 Parquet에 내리기 -Azure DocumentDB(이전 이름 Azure Cosmos DB for MongoDB vCore)는 MongoDB -change stream을 제공합니다. 다만 지원 범위가 MongoDB 서버와 다르므로 코드를 -옮기기 전에 실제 클러스터에서 확인해야 합니다. 이 실습은 PyMongo consumer를 AKS에 -띄우고 프라이빗 엔드포인트로 클러스터에 연결해 다음 네 가지를 측정합니다. +Azure DocumentDB(이전 이름 Azure Cosmos DB for MongoDB vCore)의 change stream을 +Airflow DAG가 주기적으로 읽어 Parquet 파일로 내리는 방식이 실제로 쓸 만한지 +확인했습니다. 질문은 세 가지입니다. -- 어떤 옵션과 이벤트 필드가 실제로 동작하는가 -- 파드를 죽였다 살려도 이벤트가 빠지지 않는가 -- 중복이 생기는 지점은 어디이고 어떻게 흡수하는가 -- 초당 1,000건 쓰기에서 지연은 얼마인가 +- 이벤트를 빠뜨리거나 두 번 쓰지 않고 파일로 내릴 수 있는가 +- 공식 예제를 그대로 옮기면 어디서 문제가 생기는가 +- 운영에 올리기 전에 무엇을 더 확인해야 하는가 -상시 consumer 대신 Airflow DAG가 주기마다 change stream을 읽어 ADLS Gen2에 -Parquet으로 쓰는 방식은 [Airflow DAG로 Parquet 적재](airflow-parquet/index.md)에서 -측정했습니다. +시험 환경은 이 질문에 답하려고 만든 것입니다. 배포와 실행 절차는 sample +[README](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md)에, +시나리오별 수치는 [측정 상세](measurements/index.md)에 있습니다. -## 목표 +## change stream 소개 -- 프라이빗 엔드포인트만 열린 DocumentDB 클러스터와 AKS를 azd로 배포합니다. -- `probe.py`로 change stream 옵션별 동작을 확인합니다. -- 생성기가 만든 이벤트와 sink에 기록된 이벤트를 `verify.py`로 대조해 누락, - 중복, 순서 역전과 지연을 숫자로 남깁니다. +change stream은 컬렉션의 변경을 이벤트로 받아 보는 MongoDB 기능입니다. 변경을 +찾으려고 컬렉션을 반복해서 조회할 필요가 없습니다. Learn 문서에 나온 DocumentDB의 +동작은 다음과 같습니다. -## 사전 조건 +- 이벤트의 `_id`가 resume token입니다. 이 값을 저장했다가 `resumeAfter`로 넘기면 + 그 다음 이벤트부터 이어 읽습니다. +- 기본 change stream은 400 MB 활성 change log 안의 이벤트만 읽습니다. PITR 로그와 + 통합되면 최대 35일 또는 클러스터 초기화 시점 중 이른 쪽까지 재개할 수 있습니다. +- 파이프라인에는 `$addFields`, `$match`, `$project`, `$set`, `$unset`을 쓸 수 + 있습니다. `showExpandedEvents`는 지원하지 않습니다. +- 변경 전 문서(pre-image)와 다중 shard 클러스터 지원은 미리 보기이며 지원 + 요청으로 켭니다. -- Azure 구독에서 리소스 그룹을 만들 수 있는 권한 -- Azure Developer CLI(azd) 1.29 이상, Azure CLI, `kubectl`, `envsubst` -- 이 저장소의 [python-aks sample](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md) +## 확인한 구성 -이 문서의 결과는 2026-10-02에 아래 환경에서 측정했습니다. +![AKS의 Airflow scheduler가 5분마다 내보내기 파드를 만들면 그 파드가 프라이빗 엔드포인트를 거쳐 DocumentDB change stream을 읽어 ADLS Gen2에 Parquet 청크와 checkpoint를 쓰며 Entra ID 워크로드 ID로 인증하는 구성](images/architecture.svg) -| 항목 | 값 | +Airflow가 5분마다 `KubernetesPodOperator`로 내보내기 파드를 하나 띄웁니다. 파드는 +다음 순서로 동작하고 끝나면 사라집니다. + +1. ADLS Gen2의 checkpoint 파일에서 resume token을 읽습니다. 없으면 stream의 현재 + 위치에서 시작합니다. +2. 실행을 시작한 시각까지 기록된 이벤트를 읽어 100,000건씩 Parquet 청크로 + 올립니다. 청크 파일 이름은 청크 첫 이벤트 바로 앞의 resume token으로 만듭니다. +3. 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. + +DAG는 `max_active_runs=1`이고 실패하면 세 번까지 재시도합니다. 재시도는 같은 +checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁니다. + +## 결과: 동작하는가 + +동작합니다. 모든 시나리오에서 생성기가 만든 이벤트가 Parquet 파일에 한 번씩 +들어갔고 누락, 중복, 문서별 순서 역전은 0건이었습니다. 단 아래 표의 조건을 +지켰을 때입니다. + +| 확인 항목 | 결과 | | --- | --- | -| 리전 | East US 2 | -| DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 고가용성 끔, 서버 버전 7.0.0 | -| 네트워크 | 공용 액세스 끔, 프라이빗 엔드포인트(`privatelink.mongocluster.cosmos.azure.com`, 포트 10260) | -| AKS | Kubernetes 1.35, Standard_D4s_v6 노드, Azure CNI overlay | -| 클라이언트 | Python 3.12, PyMongo 4.18.2 | - -## 비용과 안전 경계 - -- 클러스터, AKS 노드, ACR과 프라이빗 엔드포인트는 실행하는 동안 과금됩니다. - 측정이 끝나면 바로 리소스 그룹을 삭제합니다. -- 관리자 비밀번호는 셸 환경 변수로만 azd에 넘기고 연결 문자열은 Kubernetes - Secret에만 넣습니다. Learn은 azd 환경의 `.env` 파일에 비밀을 넣지 말라고 - 경고합니다. 그 파일이 있는 `.azure/`는 `.gitignore` 대상이며 커밋하지 않습니다. -- 생성기는 지정한 데이터베이스(`cslab`)의 `orders` 컬렉션에 직접 씁니다. - 운영 클러스터를 대상으로 실행하지 않습니다. -- 이 문서에 나오는 리소스 이름과 레지스트리 주소는 가상 값입니다. - -## 배포 - -Azure 리소스는 azd로 배포하고 이미지와 파드는 `az acr build`와 `kubectl`로 직접 -배포합니다. 명령은 sample 디렉터리에서 실행하며 전체 절차는 sample README에 -있습니다. - -```bash -azd env new docdb-changestream -azd env set AZURE_LOCATION eastus2 -export DOCDB_ADMIN_PASSWORD="Cs$(openssl rand -hex 12)Aa9" -azd provision - -set -a; eval "$(azd env get-values)"; set +a -export IMAGE=$AZURE_CONTAINER_REGISTRY_ENDPOINT/cslab:v1 -az acr build -r $AZURE_CONTAINER_REGISTRY_NAME -t cslab:v1 app/ -az aks get-credentials -g $AZURE_RESOURCE_GROUP -n $AZURE_AKS_CLUSTER_NAME - -kubectl create namespace cslab -MONGO_URI="${DOCDB_CONNECTION_STRING/:/$DOCDB_ADMIN_USER:$DOCDB_ADMIN_PASSWORD}" -kubectl -n cslab create secret generic docdb --from-literal=uri="$MONGO_URI" -``` - -`azd provision`은 `rg-<환경 이름>` 리소스 그룹을 만들고 `infra/cluster.bicep`과 -`infra/lake.bicep`을 배포합니다. `cluster.bicep`은 VNet, -`Microsoft.DocumentDB/mongoClusters` 클러스터(M30, shard 1개), 프라이빗 -엔드포인트와 사설 DNS 영역, ACR, AKS를 만듭니다. `lake.bicep`은 -[Airflow DAG로 Parquet 적재](airflow-parquet/index.md)에서 쓰는 ADLS Gen2와 워크로드 -ID를 만듭니다. 배포 출력은 azd 환경 값으로 저장되고 위의 `azd env get-values`로 -셸 변수가 됩니다. - -프라이빗 엔드포인트의 group ID는 `MongoCluster`입니다. Learn은 프라이빗 -엔드포인트로 연결할 때 `mongodb+srv` 형식의 연결 문자열을 쓰라고 안내합니다. 배포 -출력의 연결 문자열은 ``와 `` 자리 표시자를 담고 있어 위 명령이 -관리자 계정으로 바꿉니다. AKS 파드에서는 클러스터 호스트 이름이 사설 DNS 영역을 -거쳐 프라이빗 IP로 확인됩니다. - -감시할 컬렉션은 consumer보다 먼저 만듭니다. 컬렉션이 없으면 `watch()`가 -`NamespaceNotFound`(code 26)로 실패합니다. - -```bash -envsubst < k8s/toolbox.yaml | kubectl apply -f - -kubectl -n cslab exec cs-toolbox -- python -c \ - "from common import *; get_database(get_client('setup')).create_collection('orders')" -envsubst < k8s/consumer.yaml | kubectl apply -f - -``` - -## 시나리오 실행 - -모든 시나리오는 `k8s/job.yaml`에 `SCRIPT`와 매개변수를 넣어 Job으로 실행합니다. - -| 실행 | 스크립트와 매개변수 | 목적 | +| 5분 주기 지연 | 초당 약 1,000건 쓰기에서 이벤트 기록부터 파일 저장까지 p50 156초, p99 302초 | +| 업로드 직후 장애와 재시도 | 첫 청크를 올리고 checkpoint를 쓰기 전에 파드를 죽여도 재시도 후 중복 0, 누락 0 | +| 400 MB 활성 change log를 넘긴 재개 | DAG를 31분 멈춘 사이 쌓인 문서 본문 2.6 GB 이상의 백로그를 누락 없이 따라잡음. `maxAwaitTimeMS`를 지정하지 않았을 때만 성공 | +| 처리 속도 | 1 KB 문서는 초당 약 13,000–18,000건, 4 KB 문서 백로그는 초당 약 3,600건 | +| 파일 크기 | 이벤트에 실린 문서 크기와 비슷함. update 이벤트에도 문서 전체가 실려 문서 하나가 여러 번 저장됨 | + +지켜야 할 조건은 다음과 같습니다. + +- change stream을 열 때 짧은 `maxAwaitTimeMS`를 주지 않습니다. 1초로 두면 큰 + 백로그를 재개할 때 code 50으로 실패하고 재시도로도 넘어가지 못했습니다. +- 파일을 먼저 쓰고 checkpoint를 나중에 씁니다. 순서가 반대면 그 사이 장애로 + 이벤트를 잃습니다. +- 파일 이름을 resume token으로 정해 재시도가 같은 파일을 덮어쓰게 합니다. +- 실행이 겹치지 않게 `max_active_runs=1`과 checkpoint ETag 조건을 함께 씁니다. +- 감시할 컬렉션을 먼저 만듭니다. 없으면 `watch()`가 code 26으로 실패합니다. + +## 공식 예제가 놓친 부분 + +Learn 문서의 시작 예제(Python, Java, C#, Ruby, Node.js)와 +[changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) +저장소의 sample은 계속 떠 있는 consumer가 이벤트를 출력하는 예제입니다. 기능 +확인에는 충분하지만 주기 실행하는 DAG로 옮기면 다음 부분이 문제가 됩니다. + +| 공식 예제 | 옮겼을 때의 문제 | 이번 구현 | | --- | --- | --- | -| probe | `SCRIPT=probe.py` | 옵션, 이벤트 필드, 재개 방식과 오류 코드 확인 | -| r1 | `generator.py`, `DOCS=10000` | 기본 정합성과 지연 | -| r2 | `generator.py`, `DOCS=100000`, 실행 중 consumer 파드 반복 삭제 | 재시작 후 누락 여부 | -| r3 | `generator.py`, `DOCS=100000`, consumer에 `FAULT_EXIT_AFTER_WRITE=200` 설정 | sink 쓰기와 checkpoint 사이에서 프로세스가 죽을 때의 중복 | -| r4 | `generator.py`, `RATE=1000`, 5분 | 초당 1,000건에서의 지연 | - -생성기는 문서마다 insert와 update를 한 번씩 실행합니다. 10번째 문서마다 -replace, 5번째 문서마다 delete를 더합니다. 따라서 문서 10,000개는 이벤트 -23,000건이 됩니다. 쓰기는 모두 `w=majority`입니다. - -```bash -JOB_NAME=gen-r4 SCRIPT=generator.py RUN_ID=r4 DOCS=300000 WORKERS=16 RATE=1000 \ - envsubst < k8s/job.yaml | kubectl apply -f - -# 생성기가 끝나고 consumer가 따라잡은 뒤 -JOB_NAME=verify-r4 SCRIPT=verify.py RUN_ID=r4 DOCS=0 WORKERS=0 RATE=0 \ - envsubst < k8s/job.yaml | kubectl apply -f - -kubectl -n cslab logs job/verify-r4 -``` - -consumer는 다음 방식으로 동작합니다. - -- `collection.watch(full_document="updateLookup", max_await_time_ms=1000)`로 열고 - `try_next()`로 읽습니다. -- 이벤트를 최대 200건씩 sink 컬렉션에 `bulk_write`한 뒤 resume token을 - `_cs_checkpoints`에 저장합니다. 대기 중에는 post-batch resume token을 저장합니다. -- sink는 resume token의 `_data`를 키로 upsert합니다. 같은 이벤트가 다시 오면 - 행을 추가하지 않고 `deliveries`를 1 올립니다. -- 지연은 consumer가 이벤트를 받은 시각에서 생성기가 문서에 기록한 - `updated_at` 또는 `created_at`을 뺀 값입니다. - -## 예상 결과 - -### 옵션과 이벤트 필드 - -Learn 문서에 나온 동작과 이번 클러스터에서 관찰한 동작을 구분했습니다. -관찰 결과는 M30, shard 1개, 서버 7.0.0 클러스터에서 2026-10-02에 확인한 값입니다. +| 저장소의 `mongo_utils.py`가 resume token을 이벤트마다 로컬 파일 `.resume_token.json`에 씀 | 실행마다 새 파드가 뜨면 파일이 없어 현재 위치부터 읽음. 그 사이 이벤트를 잃음 | checkpoint를 ADLS Gen2 파일로 두고 청크마다 ETag 조건으로 갱신 | +| Python 예제가 대상 컬렉션에 `insert_one`을 한 뒤 token을 저장 | 두 동작 사이에서 프로세스가 죽으면 재시작 후 같은 이벤트가 한 번 더 들어감 | resume token으로 파일 이름을 정해 덮어씀. 상시 consumer는 token을 키로 upsert | +| `for change in stream`처럼 끝없이 읽음 | 배치 실행이 끝나지 않음 | `try_next()`로 읽고 실행 시작 시각 이후 이벤트를 만나면 종료 | +| C# 예제와 저장소의 지원 확인 스크립트가 대기 시간을 1초로 지정 | 큰 백로그 재개에서 code 50(`ExceededTimeLimit`)이 같은 위치에서 반복 | `max_await_time_ms`를 지정하지 않음 | +| 오류가 나면 메시지를 출력하고 끝남 | Learn 제한 사항은 장애 조치 뒤 커서를 다시 열어야 한다고 설명함 | Airflow 재시도가 새 파드에서 checkpoint로 stream을 다시 엶 | +| 감시할 컬렉션이 있다고 가정 | 컬렉션이 없으면 `watch()`가 code 26 | 컬렉션을 먼저 만듦 | + +Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 있었습니다. 2026-10-02 +기준 M30, shard 1개, 서버 7.0.0 클러스터에서 관찰한 결과입니다. | 항목 | Learn 문서 | 관찰 결과 | | --- | --- | --- | -| 이벤트 필드 | insert, update, delete 예시에 `_id`, `operationType`, `fullDocument`, `ns`, `documentKey` | `_id`, `operationType`, `fullDocument`, `ns`, `documentKey`, `wallTime`. `clusterTime`은 없음 | -| replace | 예시 없음 | `operationType: update`로 오고 `fullDocument`에 교체 후 문서 전체가 있음 | -| update의 `fullDocument` | 변경 후 문서 전체를 보여 주는 예시 | 옵션 없이도 포함됨. `updateLookup`, `whenAvailable`, `required` 모두 오류 없이 열림 | -| `updateDescription` | 별도 옵션 예시로 제시. 파이프라인 안의 update에서는 지원하지 않음 | 파이프라인이 있든 없든 반환되지 않음 | -| pre-image | 미리 보기. 지원 요청으로 클러스터에서 켜야 함 | `collMod`는 성공. `whenAvailable`은 `null`, `required`는 code 10065 오류. 지원 요청은 하지 않음 | -| 파이프라인 단계 | `$addFields`, `$match`, `$project`, `$set`, `$unset` | 다섯 개 모두 동작. 목록에 없는 `$replaceRoot`, `$redact`도 오류 없이 동작 | -| 감시 범위 | 컬렉션 예시 | `db.watch()`는 code 26. `client.watch()`에 `ns.db` 조건을 건 `$match`는 동작 | -| 재개 | `resumeAfter`, `startAt`, `startAtOperationTime` 지원 | `resume_after`, `start_after` 동작. 세션의 `operationTime`이 비어 있어 이를 쓴 `start_at_operation_time`은 실패. 현재 시각에서 10분 뺀 `Timestamp`는 동작 | -| 잘못된 resume token | 언급 없음 | code 2(BadValue) | -| `showExpandedEvents` | 지원하지 않음 | code 115(CommandNotSupported) | -| 감시 중인 컬렉션 drop, rename | 언급 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | -| 트랜잭션 | 언급 없음 | 이벤트는 오지만 `txnNumber`, `lsid`는 없음 | -| 큰 문서 | 언급 없음 | 14 MiB 문서의 insert와 update 이벤트 모두 전달됨 | -| 대기 중 resume token | 언급 없음 | 이벤트가 없어도 post-batch resume token이 전진함 | - -Learn은 이력 재개에 대해 다음을 설명합니다. 기본 change stream은 400 MB 크기의 -활성 change log 안의 이벤트만 읽습니다. PITR 로그와 통합되면 최대 35일 또는 클러스터 -초기화 시점 중 이른 쪽까지 재개 범위가 늘어납니다. 이번 실습은 이 범위를 시험하지 -않았습니다. - -### 정합성과 지연 - -모든 실행에서 `verify.py`가 보고한 누락, 예상 밖 이벤트, 문서별 순서 역전은 -0건이었습니다. - -| 실행 | 부하 | 이벤트 | 누락 | 재전달 | 지연 p50 / p99 | -| --- | --- | --- | --- | --- | --- | -| r1 | 문서 10,000개, 속도 제한 없음 | 23,000 | 0 | 0 | 57 / 177 ms | -| r2 | 문서 100,000개, consumer 파드 반복 삭제 | 230,000 | 0 | 0 | 측정 대상 아님 | -| r3 | 문서 100,000개, 200건 쓰기 직후 프로세스 종료를 5회 주입 | 230,000 | 0 | 1,000 | 측정 대상 아님 | -| r4 | 초당 1,000건, 5분 | 299,000 | 0 | 0 | 55 / 123 ms | - -- r2에서 파드를 삭제하면 이전 파드가 종료되는 동안 ReplicaSet이 새 파드를 - 띄웠습니다. `Recreate` 전략은 롤아웃에만 적용되므로 잠깐 두 consumer가 함께 - 읽었지만 upsert 덕분에 sink에 중복 행은 생기지 않았습니다. -- r3의 재전달 1,000건은 5회 × 200건입니다. checkpoint보다 sink 쓰기가 먼저 - 끝난 배치를 재시작 후 다시 받은 것입니다. change stream 소비는 - at-least-once이므로 sink가 멱등이어야 합니다. -- r3에서 밀린 이벤트를 따라잡는 속도는 초당 약 6,800건이었습니다. -- 클러스터 CPU는 초당 약 1,000건에서 약 30%, 속도 제한 없이 초당 약 2,900건을 - 쓸 때 약 60%였습니다. - -## 검증 - -- `verify.py` 출력의 `missing`, `unexpected`, `per_doc_out_of_order`가 모두 0인지 확인합니다. -- `redelivered_events`는 장애를 주입하지 않은 실행에서 0, r3에서는 주입 횟수 × 배치 - 크기와 같아야 합니다. -- `kubectl -n cslab logs deploy/cs-consumer`에서 재시작 직후 `start` 로그의 - `resume` 값이 `true`인지 확인합니다. 저장된 token으로 이어 읽었다는 뜻입니다. - -## 정리 - -```bash -azd down -kubectl config delete-context $AZURE_AKS_CLUSTER_NAME -``` - -`azd down`은 리소스 그룹과 그 안의 리소스를 모두 삭제합니다. - -## 문제 해결 - -| 증상 | 원인과 조치 | +| `maxAwaitTimeMS` | 설명 없음. C# 예제는 1초 | 새 이벤트 대기 시간이 아니라 `getMore` 전체의 실행 제한으로 적용됨 | +| 이벤트 필드 | `_id`, `operationType`, `fullDocument`, `ns`, `documentKey` | 같은 필드에 `wallTime`이 더 있고 `clusterTime`은 없음 | +| replace | 예시 없음 | `operationType: update`로 오고 교체 후 문서 전체가 실림 | +| update의 `fullDocument` | 변경 후 문서 전체를 보여 주는 예시 | `updateLookup` 없이도 포함됨 | +| `updateDescription` | 이벤트 예시가 있음 | 파이프라인이 있든 없든 반환되지 않음 | +| 감시 범위 | 컬렉션 예시만 있음 | `db.watch()`는 code 26. `client.watch()`에 `$match`로 `ns.db`를 거르면 동작 | +| 컬렉션 drop, rename | 설명 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | +| `startAtOperationTime` | 날짜로 `Timestamp`를 만드는 예시 | 날짜로 만든 값은 동작함. 세션의 `operationTime`은 비어 있어 쓸 수 없음 | +| pre-image | 미리 보기. 지원 요청으로 켬 | 지원 요청 없이 `required`로 열면 code 10065 | + +옵션별 관찰 전체는 [측정 상세](measurements/index.md)에 있습니다. + +## 운영 전에 더 확인할 것 + +이번 시험에서 다루지 않았거나 숫자로 확인하지 못한 항목입니다. + +- **장애 조치:** 시험 클러스터는 고가용성을 껐습니다. 고가용성을 켠 클러스터에서 + 실행 중 장애 조치가 나면 태스크가 실패하고 재시도가 이어 읽는지 확인합니다. +- **다중 shard:** 미리 보기 기능입니다. Learn은 전역 순서를 보장하지 않고 단일 + shard에서 다중 shard로 바꾸면 재개할 수 없다고 설명합니다. 파일 안 이벤트 순서에 + 기대는 다운스트림 처리를 다시 봐야 합니다. +- **재개 범위를 넘긴 정지:** DAG가 35일 또는 클러스터 초기화 시점보다 오래 멈추면 + 저장한 token으로 재개할 수 없습니다. 그때 나는 오류와 전체 재적재 절차를 + 정합니다. 마지막 checkpoint 갱신 시각을 모니터링합니다. +- **400 MB 경계:** change log 크기는 조회할 수 없어 문서 본문 크기로 추정했습니다. + 백로그 재개가 PITR 로그를 거쳤는지는 구분하지 못했습니다. +- **서버 업데이트:** `maxAwaitTimeMS`, `updateDescription`처럼 문서와 다르게 동작한 + 항목은 서버 버전이 바뀌면 다시 확인합니다. +- **부하와 주기:** 실행 한 번이 주기(5분) 안에 끝나야 지연이 쌓이지 않습니다. 더 + 높은 쓰기 속도와 다른 클러스터 tier에서 실행 시간을 다시 잽니다. +- **파일 크기와 압축:** 시험 데이터는 압축되지 않는 랜덤 문자열이었습니다. 실제 + 문서로 snappy와 zstd의 크기, 시간, CPU를 비교합니다. +- **다운스트림 처리:** 파일은 이벤트 이력입니다. 문서별 최신 상태 병합, 문서가 + 없는 delete 이벤트 처리, JSON 문자열 열 펼치기, 작은 파일 정리를 설계합니다. +- **운영 Airflow:** 차트에 포함된 PostgreSQL 대신 외부 데이터베이스를 쓰고 실패한 + 실행에 알림을 겁니다. +- **변경 전 문서가 필요할 때:** pre-image는 지원 요청으로 켠 뒤 저장 공간과 지연 + 영향을 측정합니다. + +## 시험 범위와 한계 + +| 항목 | 값 | | --- | --- | -| `watch()`가 code 26으로 실패 | 컬렉션이 없거나 `db.watch()`를 호출했습니다. 컬렉션을 먼저 만들거나 `client.watch()`와 `$match`를 씁니다 | -| 컬렉션을 지운 뒤 consumer가 반복 실패 | drop과 rename은 `invalidate` 없이 code 26을 반환합니다. 컬렉션을 다시 만들고 새 stream을 엽니다 | -| `start_at_operation_time`에 넘길 값이 없음 | 세션의 `operationTime`이 비어 있습니다. 시각 기반 `Timestamp`를 만들거나 resume token을 저장합니다 | -| `full_document_before_change="required"`가 code 10065로 실패 | pre-image는 미리 보기이며 지원 요청으로 켜야 합니다 | -| 큰 백로그를 재개할 때 code 50 `ExceededTimeLimit`로 반복 실패 | 이 클러스터는 `maxAwaitTimeMS`를 `getMore` 실행 제한으로 적용합니다. 오래된 change log를 읽는 `getMore`는 몇 초 걸릴 수 있습니다. consumer는 `MAX_AWAIT_MS`를 늘리고 직접 작성한 코드는 `max_await_time_ms`를 지정하지 않습니다. [Airflow DAG로 Parquet 적재](airflow-parquet/index.md)에서 측정했습니다 | -| 노드 크기 오류로 AKS 배포 실패 | 구독에서 허용되지 않는 VM 크기입니다. `nodeVmSize` 매개변수를 바꿉니다 | +| 측정일 | 2026-10-02, East US 2 | +| DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 고가용성 끔, 서버 7.0.0, 공용 액세스 끔 | +| AKS | Kubernetes 1.35, Standard_D4s_v6 노드 4개 | +| Airflow | Helm chart 1.22.0, Airflow 3.2.2, `LocalExecutor` | +| 내보내기 파드 | Python 3.12, PyMongo 4.18.2, pyarrow 25.0.1, snappy 압축 | + +컬렉션 하나를 하루 동안 측정한 결과입니다. -## 공식 참고 자료 +## 공식 출처 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) -- [$changeStream](https://learn.microsoft.com/documentdb/query/operators/aggregation/$changestream) -- [Use Azure Private Link in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/how-to-private-link) -- [Compute and storage configurations](https://learn.microsoft.com/azure/documentdb/compute-storage) -- [Release notes for Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/release-notes) -- [Microsoft.DocumentDB mongoClusters Bicep reference](https://learn.microsoft.com/azure/templates/microsoft.documentdb/2026-06-01/mongoclusters) -- [Work with Azure Developer CLI environment variables](https://learn.microsoft.com/azure/developer/azure-developer-cli/manage-environment-variables) - [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) diff --git a/docs/services/azure-documentdb/change-streams/airflow-parquet/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md similarity index 51% rename from docs/services/azure-documentdb/change-streams/airflow-parquet/index.md rename to docs/services/azure-documentdb/change-streams/measurements/index.md index 238ca6a8..326a7a8f 100644 --- a/docs/services/azure-documentdb/change-streams/airflow-parquet/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -1,10 +1,10 @@ --- -title: Azure DocumentDB change stream을 Airflow DAG로 ADLS Gen2 Parquet에 적재하기 -description: AKS 위 Airflow DAG가 5분마다 PyMongo로 change stream을 읽어 프라이빗 엔드포인트 뒤 ADLS Gen2에 Parquet으로 쓸 때의 지연, 재시도 중복, 400 MB 활성 change log를 넘긴 재개와 파일 크기를 측정했습니다. +title: Azure DocumentDB change stream Parquet 적재 측정 상세 +description: Airflow DAG의 5분 주기 지연, 업로드 직후 장애 재시도, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. document_type: research services: [azure-documentdb, azure-storage, azure-kubernetes-service] technologies: [python, mongodb, airflow, kubernetes] -tags: [evaluate, build, storage] +tags: [evaluate, storage] status: current verification_status: verified sources_checked_at: 2026-10-02 @@ -13,64 +13,18 @@ topic_order: 1 official_sources: - title: Change streams in Azure DocumentDB url: https://learn.microsoft.com/azure/documentdb/change-streams - - title: Use private endpoints for Azure Storage - url: https://learn.microsoft.com/azure/storage/common/storage-private-endpoints - - title: Use Microsoft Entra Workload ID with Azure Kubernetes Service (AKS) - url: https://learn.microsoft.com/azure/aks/workload-identity-overview --- -# Azure DocumentDB change stream을 Airflow DAG로 ADLS Gen2 Parquet에 적재하기 +# Azure DocumentDB change stream Parquet 적재 측정 상세 -[상위 실습](../index.md)은 change stream을 상시 consumer로 읽었습니다. 이 문서는 -AKS에 설치한 Airflow가 5분마다 파드를 띄워 그동안 쌓인 이벤트를 ADLS Gen2에 -Parquet 파일로 쓰는 구성을 실제 구독에서 측정한 결과입니다. 확인한 질문은 다음과 -같습니다. +[상위 문서](../index.md)의 결론을 뒷받침하는 측정값입니다. 시험 환경도 상위 문서에 +있습니다. 각 시나리오의 생성기 매개변수와 실행 방법은 sample +[README](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md)의 +Measured scenarios에 있습니다. -- 이벤트가 파일로 저장되기까지 얼마나 걸리는가 -- 업로드와 checkpoint 사이에서 파드가 죽으면 중복 행이 생기는가 -- DAG를 멈춘 사이 400 MB 활성 change log보다 많이 쌓여도 이어 읽을 수 있는가 -- Parquet 파일은 얼마나 커지는가 - -배포와 실행 방법은 sample의 -[README](https://github.com/hellices/devguidesample/blob/main/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md)와 -코드에 있습니다. - -## 구성 - -![AKS의 Airflow scheduler가 5분마다 내보내기 파드를 만들고, 파드가 프라이빗 엔드포인트를 거쳐 DocumentDB change stream을 읽어 ADLS Gen2에 Parquet 청크와 checkpoint를 쓰며 Entra ID 워크로드 ID로 인증하는 구성](images/architecture.svg) - -| 구성 요소 | 내용 | -| --- | --- | -| DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 서버 7.0.0 | -| AKS | Kubernetes 1.35, Standard_D4s_v6 노드 4개 | -| Airflow | Helm chart 1.22.0, Airflow 3.2.2, `LocalExecutor`, 차트에 포함된 PostgreSQL | -| DAG | `KubernetesPodOperator` 태스크 하나. 5분 주기, `max_active_runs=1`, 재시도 3회 | -| 내보내기 파드 | Python 3.12, PyMongo 4.18.2, pyarrow 25.0.1. 청크당 100,000건, snappy 압축 | -| ADLS Gen2 | Standard_LRS, 계층 구조 네임스페이스, 공용 액세스와 공유 키 끔 | - -스토리지에는 blob과 dfs 프라이빗 엔드포인트를 둘 다 만들었습니다. Learn은 Data -Lake Storage에서 dfs 엔드포인트만 만들면 blob 엔드포인트를 쓰는 작업이 실패할 수 -있다고 설명합니다. 파드는 `azure.workload.identity/use: "true"` 레이블이 있어야 -워크로드 ID 토큰을 받습니다. - -실행 한 번은 checkpoint의 resume token에서 재개해 실행을 시작한 시각까지의 이벤트를 -읽습니다. 100,000건마다 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. 청크 -파일 이름은 그 청크를 시작한 resume token의 해시라서 재시도는 같은 파일을 덮어씁니다. - -## 결과 요약 - -| 질문 | 결과 | -| --- | --- | -| 지연 | 초당 약 1,000건 부하에서 이벤트 기록부터 파일 저장까지 p50 156초, p99 302초, 최대 311초 | -| 재시도 중복 | 첫 청크 업로드 직후 프로세스를 죽인 뒤 재시도한 결과 230,000건 중 중복 0, 누락 0 | -| 활성 change log 초과 | 문서 본문 2.6 GB가 넘는 백로그를 누락·중복 없이 이어 읽음. 단 `maxAwaitTimeMS=1000`에서는 같은 위치에서 code 50으로 반복 실패 | -| 처리 속도 | 1 KB 문서는 초당 약 13,000–18,000건, 4 KB 문서 백로그는 초당 약 3,600건. 실행마다 파드 기동과 정리에 약 7초 | -| 파일 크기 | 이벤트에 실린 문서 크기와 거의 같음. 시험 데이터의 랜덤 채움 문자열은 snappy로 줄지 않았고 zstd로 다시 쓰면 71% 줄어듦 | - -모든 시나리오에서 이벤트가 빠짐없이 한 번씩 파일에 들어갔습니다. 검증은 -`verify_lake.py`가 Parquet 파일 전체를 생성기가 만든 이벤트와 대조했습니다. 중복은 -같은 resume token이 두 행 이상인 경우입니다. 저장 지연은 파일 last-modified(초 -단위)에서 이벤트 `wallTime`을 뺀 값입니다. +검증은 `verify_lake.py`가 Parquet 파일 전체를 생성기가 만든 이벤트와 대조했습니다. +중복은 같은 resume token이 두 행 이상인 경우입니다. 저장 지연은 파일 +last-modified(초 단위)에서 이벤트 `wallTime`을 뺀 값입니다. ## 5분 주기 지연 @@ -189,43 +143,65 @@ zstd는 insert와 update에 반복된 같은 채움 문자열까지 찾아 줄 빼면 100,000행이 4.8 MB여서 Parquet의 열 구조가 더하는 크기는 작습니다. 실제 문서는 랜덤 문자열보다 잘 압축되므로 이번 숫자는 압축 측면의 최악에 가깝습니다. -## 판단 - -- 몇 분 지연을 받아들일 수 있고 결과가 Parquet 파일이어야 하면 이 방식이 단순합니다. - 상시 파드가 없고 중간 메시지 계층도 없습니다. -- 정확히 한 번의 결과는 파일 이름과 checkpoint 순서로 얻습니다. checkpoint를 먼저 쓰고 - 파일을 나중에 쓰면 그 사이 장애로 이벤트를 잃습니다. -- `max_active_runs=1`과 ETag 조건은 둘 다 필요합니다. 첫째는 Airflow 안에서 실행이 - 겹치지 않게 하고 둘째는 수동 Job 같은 외부 실행이 checkpoint를 덮어쓰지 못하게 합니다. -- change stream을 여는 코드에 짧은 `maxAwaitTimeMS`를 주지 않습니다. 평소에는 문제가 - 없다가 백로그가 커진 뒤에야 code 50으로 드러나고 재시도로도 풀리지 않습니다. Learn의 - C# 예제도 `MaxAwaitTime`을 1초로 둡니다. 값을 꼭 줘야 하면 같은 클러스터에서 백로그 - 재개를 시험해 정합니다. -- DAG를 오래 멈추면 Learn이 말하는 재개 범위(최대 35일 또는 클러스터 초기화 시점 중 - 이른 쪽)를 넘을 수 있습니다. 멈춘 기간을 모니터링합니다. Learn은 보관된 로그를 - 처리하는 작업을 트래픽이 적은 시간에 하라고 권장합니다. 백로그 재개 시험에서 따라잡는 동안 - 클러스터 CPU가 15–32%로 올랐습니다. -- 파일은 이벤트 이력이므로 문서 하나가 여러 번 저장됩니다. 문서별 최신 상태만 필요하면 - 뒤 단계에서 `doc_id` 기준으로 합칩니다. update를 바뀐 필드만으로 줄이는 방법은 이 - 클러스터가 `updateDescription`을 돌려주지 않아 쓸 수 없습니다. -- 압축은 zstd를 먼저 검토합니다. 문서가 크면 청크 크기(`CS_CHUNK_EVENTS`)를 줄여 - 파일 크기와 파드 메모리를 맞춥니다. - -## 한계 - -- 단일 shard M30 클러스터, 컬렉션 하나, 측정 하루의 결과입니다. +## 상시 consumer 기준선 + +DAG와 비교하려고 계속 떠 있는 PyMongo consumer로도 같은 클러스터를 읽었습니다. +consumer는 이벤트를 최대 200건씩 MongoDB sink 컬렉션에 resume token을 키로 +upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 update를 한 번씩 +실행하고 10번째 문서마다 replace, 5번째 문서마다 delete를 더합니다. 문서 +10,000개가 이벤트 23,000건이 됩니다. + +모든 실행에서 `verify.py`가 보고한 누락, 예상 밖 이벤트, 문서별 순서 역전은 +0건이었습니다. + +| 실행 | 부하 | 이벤트 | 누락 | 재전달 | 지연 p50 / p99 | +| --- | --- | --- | --- | --- | --- | +| r1 | 문서 10,000개, 속도 제한 없음 | 23,000 | 0 | 0 | 57 / 177 ms | +| r2 | 문서 100,000개, consumer 파드 반복 삭제 | 230,000 | 0 | 0 | 측정 대상 아님 | +| r3 | 문서 100,000개, 200건 쓰기 직후 프로세스 종료를 5회 주입 | 230,000 | 0 | 1,000 | 측정 대상 아님 | +| r4 | 초당 1,000건, 5분 | 299,000 | 0 | 0 | 55 / 123 ms | + +- r2에서 파드를 삭제하면 이전 파드가 종료되는 동안 ReplicaSet이 새 파드를 + 띄웠습니다. `Recreate` 전략은 롤아웃에만 적용되므로 잠깐 두 consumer가 함께 + 읽었지만 upsert 덕분에 sink에 중복 행은 생기지 않았습니다. +- r3의 재전달 1,000건은 5회 × 200건입니다. checkpoint보다 sink 쓰기가 먼저 + 끝난 배치를 재시작 후 다시 받은 것입니다. change stream 소비는 + at-least-once이므로 sink가 멱등이어야 합니다. +- r3에서 밀린 이벤트를 따라잡는 속도는 초당 약 6,800건이었습니다. +- 클러스터 CPU는 초당 약 1,000건에서 약 30%, 속도 제한 없이 초당 약 2,900건을 + 쓸 때 약 60%였습니다. + +## change stream 동작 확인 + +`probe.py`로 옵션과 이벤트 필드를 확인했습니다. Learn 문서에 나온 동작과 이번 +클러스터에서 관찰한 동작을 구분했습니다. + +| 항목 | Learn 문서 | 관찰 결과 | +| --- | --- | --- | +| 이벤트 필드 | insert, update, delete 예시에 `_id`, `operationType`, `fullDocument`, `ns`, `documentKey` | `_id`, `operationType`, `fullDocument`, `ns`, `documentKey`, `wallTime`. `clusterTime`은 없음 | +| replace | 예시 없음 | `operationType: update`로 오고 `fullDocument`에 교체 후 문서 전체가 있음 | +| update의 `fullDocument` | 변경 후 문서 전체를 보여 주는 예시 | 옵션 없이도 포함됨. `updateLookup`, `whenAvailable`, `required` 모두 오류 없이 열림 | +| `updateDescription` | 별도 옵션 예시로 제시. 파이프라인 안의 update에서는 지원하지 않음 | 파이프라인이 있든 없든 반환되지 않음 | +| pre-image | 미리 보기. 지원 요청으로 클러스터에서 켜야 함 | `collMod`는 성공. `whenAvailable`은 `null`, `required`는 code 10065 오류. 지원 요청은 하지 않음 | +| 파이프라인 단계 | `$addFields`, `$match`, `$project`, `$set`, `$unset` | 다섯 개 모두 동작. 목록에 없는 `$replaceRoot`, `$redact`도 오류 없이 동작 | +| 감시 범위 | 컬렉션 예시 | `db.watch()`는 code 26. `client.watch()`에 `ns.db` 조건을 건 `$match`는 동작 | +| 재개 | `resumeAfter`, `startAt`, `startAtOperationTime` 지원 | `resume_after`, `start_after` 동작. 세션의 `operationTime`이 비어 있어 이를 쓴 `start_at_operation_time`은 실패. 현재 시각에서 10분 뺀 `Timestamp`는 동작 | +| 잘못된 resume token | 언급 없음 | code 2(BadValue) | +| `showExpandedEvents` | 지원하지 않음 | code 115(CommandNotSupported) | +| 감시 중인 컬렉션 drop, rename | 언급 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | +| 트랜잭션 | 언급 없음 | 이벤트는 오지만 `txnNumber`, `lsid`는 없음 | +| 큰 문서 | 언급 없음 | 14 MiB 문서의 insert와 update 이벤트 모두 전달됨 | +| 대기 중 resume token | 언급 없음 | 이벤트가 없어도 post-batch resume token이 전진함 | + +## 측정의 한계 + - change log의 실제 크기는 조회할 수 없어 문서 본문 크기로 추정했습니다. -- `maxAwaitTimeMS`가 `getMore` 실행 제한으로 적용되는 동작은 Learn에 설명이 없습니다. - 이 클러스터에서 관측한 결과입니다. -- 압축 비교는 백로그 재개 시험의 파일 하나를 다시 써서 얻었습니다. zstd로 쓸 때의 시간과 CPU는 - 측정하지 않았습니다. +- `maxAwaitTimeMS`가 `getMore` 실행 제한으로 적용되는 동작은 Learn에 설명이 + 없습니다. 이 클러스터에서 관측한 결과입니다. +- 압축 비교는 백로그 재개 시험의 파일 하나를 다시 써서 얻었습니다. zstd로 쓸 때의 + 시간과 CPU는 측정하지 않았습니다. - 파일 저장 지연은 초 단위 last-modified로 계산했습니다. -- Parquet 스키마는 원본 문서를 JSON 문자열 열 하나로 둡니다. 분석 쿼리에서 열로 - 펼치는 작업은 다루지 않았습니다. -- 차트에 포함된 PostgreSQL은 시험용입니다. 운영에서는 외부 데이터베이스를 씁니다. -## 공식 참고 자료 +## 공식 출처 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) -- [Use private endpoints for Azure Storage](https://learn.microsoft.com/azure/storage/common/storage-private-endpoints) -- [Use Microsoft Entra Workload ID with AKS](https://learn.microsoft.com/azure/aks/workload-identity-overview) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index fb03221f..7b261f9c 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -1,7 +1,9 @@ -# Python change stream consumer on AKS +# Change stream to Parquet with Airflow on AKS -This runnable sample tests MongoDB change streams on Azure DocumentDB with -PyMongo. The cluster has public network access disabled. Pods on AKS reach it +This runnable sample checks whether an Airflow DAG on AKS can export Azure +DocumentDB change stream events to Parquet files in ADLS Gen2 without losing +or duplicating events. A long-running PyMongo consumer is included as a +baseline. The cluster has public network access disabled. Pods on AKS reach it through a private endpoint. | Path | Purpose | @@ -17,7 +19,7 @@ through a private endpoint. | `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes Parquet chunks and exits | | `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | | `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | -| `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job and service account, RBAC for the Airflow scheduler | +| `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job, RBAC for the Airflow scheduler | Azure resources are provisioned with azd. Images, pods and Airflow are deployed by hand with `az acr build`, `kubectl` and `helm` so each step can be @@ -212,6 +214,25 @@ Export behavior: - `consumer.py` always passes `MAX_AWAIT_MS`, 1000 by default. Raise it before the consumer resumes a large backlog. +## 6. Measured scenarios + +The published results come from these runs. Each one uses the commands from +steps 4 and 5 with the parameters below. Verify every run with `verify.py` +(consumer) or `verify_lake.py` (Parquet) and the same `RUN_ID`. + +| Run | Generator parameters | Extra steps | +| --- | --- | --- | +| r1 | `DOCS=100000 RATE=0` | None | +| r2 | `DOCS=100000 RATE=0` | While the generator runs, delete the consumer pod repeatedly with `kubectl -n cslab delete pod -l app=cs-consumer --wait=false` | +| r3 | `DOCS=100000 RATE=0` | Before the generator, run `kubectl -n cslab set env deployment/cs-consumer FAULT_EXIT_AFTER_WRITE=200`. The container exits after each 200-event write and restarts. Remove it with `FAULT_EXIT_AFTER_WRITE-` after the number of faults you want | +| r4 | `DOCS=300000 RATE=1000` | None | +| Airflow latency | `DOCS=780000 RATE=1000 PAD_BYTES=1000` | DAG unpaused for the whole run | +| Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG, run the generator, then `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container | +| Airflow backlog | `DOCS=600000 RATE=0 PAD_BYTES=4000` | Pause the DAG, run the generator, then unpause it | + +All runs use `WORKERS=16`. Pause the DAG with `airflow dags pause +change_stream_to_parquet` in the scheduler container. + ## Clean up ```bash diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml b/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml index 6454e587..cf231826 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/sample.yml @@ -1,6 +1,6 @@ -title: Python change stream consumer on AKS -description: Runnable PyMongo change stream probe, consumer, Airflow Parquet export, load generator and verifiers for Azure DocumentDB behind a private endpoint. +title: Change stream to Parquet with Airflow on AKS +description: Runnable Airflow DAG that exports Azure DocumentDB change stream events to Parquet in ADLS Gen2, with a PyMongo baseline consumer, probe, load generator and verifiers behind private endpoints. kind: runnable used_by: - index -- airflow-parquet +- measurements From 77241aefe73d82a91a91f2f9cda1f5a30f88f3c0 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 20:42:27 +0900 Subject: [PATCH 04/14] =?UTF-8?q?fix(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20sample=20=EB=A6=AC=EB=B7=B0=EC=9D=98=20=EC=8B=A4=ED=8C=A8=20?= =?UTF-8?q?=EA=B2=BD=EB=A1=9C=20=EB=B3=B4=EC=99=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 첫 실행이 읽기 전에 시작 시각을 저장해 재시도가 같은 위치에서 읽도록 함 - 내보내기 실행이 lock 파일 lease를 잡아 겹친 실행의 파일 덮어쓰기를 막음 - wallTime이 없는 이벤트는 dt=unknown 파티션에 써서 재시도 경로를 고정 - generator가 worker 실패를 failed로 기록하고 exit 1로 종료 - verify.py, verify_lake.py가 문서별 순서 역전을 실패로 처리 - README의 r1, r4 생성기 매개변수를 측정값에 맞춤 Co-Authored-By: Claude Opus 5.5 --- .../azure-documentdb/change-streams/index.md | 21 +++- .../change-streams/measurements/index.md | 2 +- .../samples/python-aks/README.md | 34 ++++-- .../samples/python-aks/app/consumer.py | 24 ++-- .../samples/python-aks/app/generator.py | 76 ++++++------ .../samples/python-aks/app/lake_export.py | 110 +++++++++++++++--- .../samples/python-aks/app/verify.py | 2 +- .../samples/python-aks/app/verify_lake.py | 2 +- 8 files changed, 199 insertions(+), 72 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index a85c6571..cb4a475c 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -14,6 +14,8 @@ official_sources: url: https://learn.microsoft.com/azure/documentdb/change-streams - title: AzureCosmosDB/changestream-driver-compatibility url: https://github.com/AzureCosmosDB/changestream-driver-compatibility + - title: Lease Blob + url: https://learn.microsoft.com/rest/api/storageservices/lease-blob --- # Azure DocumentDB change stream을 Airflow DAG로 Parquet에 내리기 @@ -52,14 +54,15 @@ change stream은 컬렉션의 변경을 이벤트로 받아 보는 MongoDB 기 Airflow가 5분마다 `KubernetesPodOperator`로 내보내기 파드를 하나 띄웁니다. 파드는 다음 순서로 동작하고 끝나면 사라집니다. -1. ADLS Gen2의 checkpoint 파일에서 resume token을 읽습니다. 없으면 stream의 현재 - 위치에서 시작합니다. +1. ADLS Gen2의 lock 파일에 lease를 잡고 checkpoint 파일에서 resume token을 + 읽습니다. checkpoint가 없으면 현재 시각을 시작 위치로 먼저 저장합니다. 2. 실행을 시작한 시각까지 기록된 이벤트를 읽어 100,000건씩 Parquet 청크로 올립니다. 청크 파일 이름은 청크 첫 이벤트 바로 앞의 resume token으로 만듭니다. 3. 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. DAG는 `max_active_runs=1`이고 실패하면 세 번까지 재시도합니다. 재시도는 같은 -checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁니다. +checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁니다. 첫 실행의 +재시도도 저장해 둔 시작 시각부터 다시 읽습니다. ## 결과: 동작하는가 @@ -82,7 +85,11 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 - 파일을 먼저 쓰고 checkpoint를 나중에 씁니다. 순서가 반대면 그 사이 장애로 이벤트를 잃습니다. - 파일 이름을 resume token으로 정해 재시도가 같은 파일을 덮어쓰게 합니다. -- 실행이 겹치지 않게 `max_active_runs=1`과 checkpoint ETag 조건을 함께 씁니다. +- checkpoint가 없는 첫 실행은 시작 위치를 먼저 저장합니다. 저장하지 않으면 첫 + 청크를 쓰다 실패했을 때 재시도가 더 뒤에서 시작해 그 사이 이벤트를 잃습니다. +- 실행이 겹치지 않게 `max_active_runs=1`을 두고 실행 동안 lock 파일 lease를 + 잡습니다. checkpoint ETag 조건은 checkpoint만 보호합니다. 이미 올린 파일을 다른 + 실행이 짧은 청크로 덮어쓰는 것은 막지 못합니다. - 감시할 컬렉션을 먼저 만듭니다. 없으면 `watch()`가 code 26으로 실패합니다. ## 공식 예제가 놓친 부분 @@ -155,9 +162,13 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 | Airflow | Helm chart 1.22.0, Airflow 3.2.2, `LocalExecutor` | | 내보내기 파드 | Python 3.12, PyMongo 4.18.2, pyarrow 25.0.1, snappy 압축 | -컬렉션 하나를 하루 동안 측정한 결과입니다. +컬렉션 하나를 하루 동안 측정한 결과입니다. 측정 뒤 리뷰에서 나온 실패 경로를 +sample에 반영했습니다. 첫 실행 시작 위치 저장, lock 파일 lease, `wallTime`이 없는 +이벤트의 파티션 고정, 검증 스크립트의 순서 역전 실패 처리입니다. 측정에서 거치지 +않은 경로이며 로컬 fake로만 확인했고 Azure에서 다시 실행하지 않았습니다. ## 공식 출처 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) - [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) +- [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 326a7a8f..01b98390 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -159,7 +159,7 @@ upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 | r1 | 문서 10,000개, 속도 제한 없음 | 23,000 | 0 | 0 | 57 / 177 ms | | r2 | 문서 100,000개, consumer 파드 반복 삭제 | 230,000 | 0 | 0 | 측정 대상 아님 | | r3 | 문서 100,000개, 200건 쓰기 직후 프로세스 종료를 5회 주입 | 230,000 | 0 | 1,000 | 측정 대상 아님 | -| r4 | 초당 1,000건, 5분 | 299,000 | 0 | 0 | 55 / 123 ms | +| r4 | 문서 130,000개, 초당 1,000건 제한(약 5분) | 299,000 | 0 | 0 | 55 / 123 ms | - r2에서 파드를 삭제하면 이전 파드가 종료되는 동안 ReplicaSet이 새 파드를 띄웠습니다. `Recreate` 전략은 롤아웃에만 적용되므로 잠깐 두 consumer가 함께 diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index 7b261f9c..ed5ae8b2 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -112,7 +112,7 @@ kubectl -n cslab exec cs-toolbox -- python -c \ envsubst < k8s/consumer.yaml | kubectl apply -f - kubectl -n cslab rollout status deployment/cs-consumer -JOB_NAME=gen-r1 SCRIPT=generator.py RUN_ID=r1 DOCS=100000 WORKERS=16 RATE=0 \ +JOB_NAME=gen-r1 SCRIPT=generator.py RUN_ID=r1 DOCS=10000 WORKERS=16 RATE=0 \ envsubst < k8s/job.yaml | kubectl apply -f - # After the generator finishes and the consumer catches up: JOB_NAME=verify-r1 SCRIPT=verify.py RUN_ID=r1 DOCS=0 WORKERS=0 RATE=0 \ @@ -126,6 +126,9 @@ Run `probe.py` the same way with `SCRIPT=probe.py`. - The sink write is an upsert keyed by the resume token `_data`. A replayed event increments `deliveries` on the existing row instead of adding a row. +- On its first start the consumer saves the current time as its start position + before it reads. A retry before the first checkpoint opens the stream at that + time with `startAtOperationTime` and reads the same events again. - The checkpoint is saved after each sink batch. When the stream is idle, the consumer saves the post-batch resume token instead. - Delivery is at-least-once. `FAULT_EXIT_AFTER_WRITE=` makes the @@ -178,8 +181,8 @@ kubectl -n airflow exec airflow-scheduler-0 -c scheduler -- \ airflow dags list-runs change_stream_to_parquet ``` -The first run has no checkpoint, so it starts at the current position of the -stream and saves it. Write events after that run and compare them with the +The first run has no checkpoint, so it saves the current time as its start +position and reads from there. Write events after that run and compare them with the Parquet files once a later run has exported them. ```bash @@ -196,11 +199,20 @@ Export behavior: - A run stops at the first event written after the run started, or when the stream has nothing to return. Under steady writes `try_next()` rarely returns `None`, so the time boundary is what ends the run. -- A run with no checkpoint starts at the current position of the stream. +- A run with no checkpoint saves the current time as the start position before + it reads. Its retry starts at the same time. - Each chunk is named after the resume token before its first event. A retry reads the same events from the same checkpoint and overwrites the same file. + The partition comes from the first event's `wallTime`, or `dt=unknown` when + the event has none, so a retry in a later hour writes the same path. - The checkpoint is `_checkpoints/.json` in the same file system. It is written with an ETag condition after each chunk upload. +- A run holds a 20-second lease on `_checkpoints/.lock` and renews it + in the background. A second run waits up to 90 seconds for the lease and + then fails, so two runs never overwrite each other's chunks. The ETag + condition alone only protects the checkpoint, not files already uploaded. + A lease that expires instead of being released can take up to a minute to + become available again. - Trigger the DAG with `{"fault_after_chunks": 1}` to make the first try exit after one upload and before the checkpoint. The retry finishes the run. - `max_await_time_ms` is not set unless `MAX_AWAIT_MS` is non-zero. The test @@ -218,14 +230,22 @@ Export behavior: The published results come from these runs. Each one uses the commands from steps 4 and 5 with the parameters below. Verify every run with `verify.py` -(consumer) or `verify_lake.py` (Parquet) and the same `RUN_ID`. +(consumer) or `verify_lake.py` (Parquet) and the same `RUN_ID`. Both exit with +code 1 on a missing, unexpected or out-of-order event. The generator records a +run as `failed` and exits with code 1 if any write fails, and the verifiers +refuse such a run. + +The runs used the sample before these changes: the first-run start position, +the stream lease, the `dt=unknown` partition and the stricter exit codes. +They change only failure paths that the runs did not hit, and were checked +with local fakes, not on Azure. | Run | Generator parameters | Extra steps | | --- | --- | --- | -| r1 | `DOCS=100000 RATE=0` | None | +| r1 | `DOCS=10000 RATE=0` | None | | r2 | `DOCS=100000 RATE=0` | While the generator runs, delete the consumer pod repeatedly with `kubectl -n cslab delete pod -l app=cs-consumer --wait=false` | | r3 | `DOCS=100000 RATE=0` | Before the generator, run `kubectl -n cslab set env deployment/cs-consumer FAULT_EXIT_AFTER_WRITE=200`. The container exits after each 200-event write and restarts. Remove it with `FAULT_EXIT_AFTER_WRITE-` after the number of faults you want | -| r4 | `DOCS=300000 RATE=1000` | None | +| r4 | `DOCS=130000 RATE=1000` | None | | Airflow latency | `DOCS=780000 RATE=1000 PAD_BYTES=1000` | DAG unpaused for the whole run | | Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG, run the generator, then `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container | | Airflow backlog | `DOCS=600000 RATE=0 PAD_BYTES=4000` | Pause the DAG, run the generator, then unpause it | diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py index 522a3929..32ef1dd4 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py @@ -2,7 +2,8 @@ The consumer stores processed events in a sink collection and saves the resume token in a checkpoint collection, so a replacement pod continues from -the last committed event instead of a local file. +the last committed event instead of a local file. The first start saves its +start time there before reading, so a failed first batch is read again. """ import json @@ -12,6 +13,7 @@ import time from typing import Any, Optional +from bson.timestamp import Timestamp from pymongo import UpdateOne from pymongo.errors import OperationFailure, PyMongoError @@ -37,9 +39,14 @@ def handle_sigterm(signum: int, _frame: Any) -> None: log_json(logger, "signal", signum=signum) -def load_checkpoint(db, consumer_id: str) -> Optional[dict]: - doc = db[CHECKPOINT_COLLECTION].find_one({"_id": consumer_id}) - return doc["token"] if doc else None +def load_checkpoint(db, consumer_id: str) -> dict: + """Return the saved position, creating it with the current time on first start.""" + db[CHECKPOINT_COLLECTION].update_one( + {"_id": consumer_id}, + {"$setOnInsert": {"token": None, "start_at": Timestamp(int(time.time()), 0), "events": 0}}, + upsert=True, + ) + return db[CHECKPOINT_COLLECTION].find_one({"_id": consumer_id}) def save_checkpoint(db, consumer_id: str, token: dict, events: int) -> None: @@ -78,7 +85,7 @@ def to_sink(change: dict, pod: str) -> UpdateOne: upsert=True) -def build_watch_kwargs(token: Optional[dict]) -> dict: +def build_watch_kwargs(token: Optional[dict], start_at: Optional[Timestamp]) -> dict: kwargs: dict = {"max_await_time_ms": int(os.getenv("MAX_AWAIT_MS", "1000"))} full_document = os.getenv("FULL_DOCUMENT", "updateLookup") if full_document != "default": @@ -88,6 +95,8 @@ def build_watch_kwargs(token: Optional[dict]) -> dict: kwargs["batch_size"] = int(batch_size) if token: kwargs["resume_after"] = token + elif start_at: + kwargs["start_at_operation_time"] = start_at return kwargs @@ -103,7 +112,8 @@ def main() -> None: source = get_source(db) sink = db[SINK_COLLECTION] - token = load_checkpoint(db, consumer_id) + position = load_checkpoint(db, consumer_id) + token, start_at = position.get("token"), position.get("start_at") log_json(logger, "start", consumer_id=consumer_id, pod=pod, resume=bool(token), namespace=source.full_name, pipeline=pipeline) @@ -115,7 +125,7 @@ def main() -> None: total = 0 while not stopping: try: - with source.watch(pipeline, **build_watch_kwargs(token)) as stream: + with source.watch(pipeline, **build_watch_kwargs(token, start_at)) as stream: log_json(logger, "stream_open", resume=bool(token)) backoff = 1.0 pending: list = [] diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py index b570b48c..117dcb09 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/generator.py @@ -6,8 +6,10 @@ import os import secrets +import sys import threading import time +from concurrent.futures import ThreadPoolExecutor from pymongo import WriteConcern @@ -35,32 +37,35 @@ def pace() -> None: if delay > 0: time.sleep(delay) - for i in indexes: - doc_id = f"{run_id}:{i:07d}" - source.insert_one({ - "_id": doc_id, "run_id": run_id, "seq": i, "version": 1, - "status": "new", "qty": i % 100, "created_at": utcnow(), "pad": pad(pad_bytes), - }) - local["insert"] += 1 - pace() - source.update_one({"_id": doc_id}, {"$set": {"status": "paid", "updated_at": utcnow()}, - "$inc": {"version": 1}}) - local["update"] += 1 - pace() - if i % 10 == 0: - source.replace_one({"_id": doc_id}, { - "run_id": run_id, "seq": i, "version": 3, "status": "replaced", "updated_at": utcnow(), - "pad": pad(pad_bytes), + try: + for i in indexes: + doc_id = f"{run_id}:{i:07d}" + source.insert_one({ + "_id": doc_id, "run_id": run_id, "seq": i, "version": 1, + "status": "new", "qty": i % 100, "created_at": utcnow(), "pad": pad(pad_bytes), }) - local["replace"] += 1 + local["insert"] += 1 pace() - if i % 5 == 0: - source.delete_one({"_id": doc_id}) - local["delete"] += 1 + source.update_one({"_id": doc_id}, {"$set": {"status": "paid", "updated_at": utcnow()}, + "$inc": {"version": 1}}) + local["update"] += 1 pace() - with lock: - for key, value in local.items(): - counts[key] += value + if i % 10 == 0: + source.replace_one({"_id": doc_id}, { + "run_id": run_id, "seq": i, "version": 3, "status": "replaced", "updated_at": utcnow(), + "pad": pad(pad_bytes), + }) + local["replace"] += 1 + pace() + if i % 5 == 0: + source.delete_one({"_id": doc_id}) + local["delete"] += 1 + pace() + finally: + # Keep the writes that succeeded before a failure in the run record. + with lock: + for key, value in local.items(): + counts[key] += value def main() -> None: @@ -84,25 +89,28 @@ def main() -> None: log_json(logger, "generator_start", run_id=run_id, docs=docs, workers=workers, rate=rate, pad_bytes=pad_bytes) - threads = [] per_worker_rate = rate / workers if rate else 0.0 - for w in range(workers): - t = threading.Thread(target=worker, args=(source, run_id, range(w, docs, workers), - per_worker_rate, pad_bytes, counts, lock)) - t.start() - threads.append(t) - for t in threads: - t.join() + errors = [] + with ThreadPoolExecutor(max_workers=workers) as pool: + futures = [pool.submit(worker, source, run_id, range(w, docs, workers), per_worker_rate, + pad_bytes, counts, lock) for w in range(workers)] + for future in futures: + if future.exception(): + errors.append(repr(future.exception())[:500]) finished = utcnow() elapsed = (finished - started).total_seconds() total_ops = sum(counts.values()) + # A failed worker leaves the workload incomplete, so the verifiers refuse it. + state = "failed" if errors else "done" runs.update_one({"_id": run_id}, {"$set": {"expected": counts, "finished_at": finished, "elapsed_s": elapsed, "ops_per_s": total_ops / elapsed, - "state": "done"}}) - log_json(logger, "generator_done", run_id=run_id, expected=counts, elapsed_s=round(elapsed, 2), - ops_per_s=round(total_ops / elapsed, 1)) + "state": state, "errors": errors}}) + log_json(logger, f"generator_{state}", run_id=run_id, expected=counts, elapsed_s=round(elapsed, 2), + ops_per_s=round(total_ops / elapsed, 1), errors=errors) client.close() + if errors: + sys.exit(1) if __name__ == "__main__": diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py index 545f479a..88a0026f 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -7,26 +7,30 @@ A chunk is named after the resume token that precedes its first event. A retry starts from the same checkpoint, reads the same events in the same order and overwrites the same file with the same or a longer chunk, so a crash between -the upload and the checkpoint does not leave duplicate rows. +the upload and the checkpoint does not leave duplicate rows. The first run +saves its start time before reading, so its retry starts at the same place. + +A run holds a lease on a lock file next to the checkpoint, so two runs never +write chunks for the same stream at the same time. """ import hashlib import io -import json import logging import os import socket +import threading import time -from datetime import datetime, timezone from typing import Optional import pyarrow as pa import pyarrow.parquet as pq from azure.core import MatchConditions -from azure.core.exceptions import ResourceNotFoundError +from azure.core.exceptions import HttpResponseError, ResourceNotFoundError from azure.identity import DefaultAzureCredential from azure.storage.filedatalake import DataLakeServiceClient from bson import json_util +from bson.timestamp import Timestamp from common import env, get_client, get_database, get_source, log_json, setup_logging, utcnow @@ -46,7 +50,10 @@ class Checkpoint: - """Resume token stored next to the data, updated with an ETag condition.""" + """Stream position stored next to the data, updated with an ETag condition. + + The position is a resume token, or the start time saved by the first run. + """ def __init__(self, fs, stream_id: str): self.file = fs.get_file_client(f"_checkpoints/{stream_id}.json") @@ -58,11 +65,11 @@ def load(self) -> Optional[dict]: except ResourceNotFoundError: return None self.etag = download.properties.etag - return json_util.loads(download.readall())["token"] + return json_util.loads(download.readall()) - def save(self, token: dict, events: int) -> None: - body = json_util.dumps({"token": token, "updated_at": utcnow(), "events": events, - "pod": socket.gethostname()}) + def save(self, token: Optional[dict], events: int, start_at: Optional[Timestamp] = None) -> None: + body = json_util.dumps({"token": token, "start_at": start_at, "updated_at": utcnow(), + "events": events, "pod": socket.gethostname()}) # IfNotModified fails if another run moved the checkpoint since we read it. condition = (dict(etag=self.etag, match_condition=MatchConditions.IfNotModified) if self.etag else dict(match_condition=MatchConditions.IfMissing)) @@ -70,6 +77,60 @@ def save(self, token: dict, events: int) -> None: self.etag = result["etag"] +class StreamLock: + """Lease on ``_checkpoints/.lock`` held for the whole run. + + The lease is short and renewed in the background. A lease left by a killed + pod expires, but Blob Storage can take up to a minute to grant a new one, + so acquiring waits that long before the task fails. + """ + + DURATION_S = 20 + ACQUIRE_WAIT_S = 90 + + def __init__(self, fs, stream_id: str): + self.file = fs.get_file_client(f"_checkpoints/{stream_id}.lock") + self.lease = None + self.lost: Optional[Exception] = None + self.stopped = threading.Event() + + def acquire(self) -> None: + try: + self.file.create_file(match_condition=MatchConditions.IfMissing) + except HttpResponseError as error: + if error.status_code != 409: + raise + deadline = time.monotonic() + self.ACQUIRE_WAIT_S + while True: + try: + self.lease = self.file.acquire_lease(lease_duration=self.DURATION_S) + break + except HttpResponseError as error: + # 409 while another run holds the lease. + if error.status_code != 409 or time.monotonic() >= deadline: + raise + time.sleep(5) + threading.Thread(target=self._renew, daemon=True).start() + + def _renew(self) -> None: + while not self.stopped.wait(self.DURATION_S / 4): + try: + self.lease.renew() + except Exception as error: # noqa: BLE001 - any failure means the lease may be gone + self.lost = error + return + + def check(self) -> None: + """Call before every write; stop instead of writing without the lease.""" + if self.lost: + raise RuntimeError(f"stream lease lost: {self.lost}") + + def release(self) -> None: + self.stopped.set() + if self.lease and not self.lost: + self.lease.release() + + def to_row(change: dict) -> dict: full = change.get("fullDocument") return { @@ -84,11 +145,14 @@ def to_row(change: dict) -> dict: } -def chunk_path(prefix: str, start_token: Optional[dict], rows: list) -> str: - first = rows[0]["wall_time"] or datetime.now(timezone.utc) - key = start_token["_data"] if start_token else "origin" +def chunk_path(prefix: str, start: dict, rows: list) -> str: + """Path from the chunk start position only, so a retry rewrites the same file.""" + key = start["token"]["_data"] if start.get("token") else f"start-{start['start_at']}" digest = hashlib.sha256(key.encode()).hexdigest()[:20] - return f"{prefix}/dt={first:%Y-%m-%d}/hour={first:%H}/part-{digest}.parquet" + first = rows[0]["wall_time"] + # Without wallTime the read time would move the file on a later retry. + partition = f"dt={first:%Y-%m-%d}/hour={first:%H}" if first else "dt=unknown" + return f"{prefix}/{partition}/part-{digest}.parquet" def write_chunk(fs, path: str, rows: list) -> int: @@ -111,8 +175,16 @@ def main() -> None: service = DataLakeServiceClient(env("LAKE_URL"), credential=DefaultAzureCredential()) fs = service.get_file_system_client(env("LAKE_FILESYSTEM", "cdc")) + lock = StreamLock(fs, stream_id) + lock.acquire() checkpoint = Checkpoint(fs, stream_id) - token = checkpoint.load() + position = checkpoint.load() + if position is None: + # First run: save the start time before reading, so a retry after a + # failed first chunk reads the same events instead of starting later. + position = {"token": None, "start_at": Timestamp(int(time.time()), 0)} + checkpoint.save(None, 0, start_at=position["start_at"]) + token = position.get("token") client = get_client(f"cs-lake-export-{socket.gethostname()}") source = get_source(get_database(client)) @@ -126,6 +198,8 @@ def main() -> None: kwargs["max_await_time_ms"] = max_await_ms if token: kwargs["resume_after"] = token + else: + kwargs["start_at_operation_time"] = position["start_at"] started = time.monotonic() run_started_at = utcnow() @@ -134,7 +208,7 @@ def main() -> None: caught_up = False with source.watch(**kwargs) as stream: rows: list = [] - chunk_start = token + chunk_start = position last_token = token while True: change = stream.try_next() @@ -152,24 +226,28 @@ def main() -> None: (max_seconds and time.monotonic() - started >= max_seconds) if rows and (full or caught_up or limit): path = chunk_path(prefix, chunk_start, rows) + lock.check() size = write_chunk(fs, path, rows) chunks += 1 if fault_after_chunks and chunks >= fault_after_chunks: log_json(logger, "fault_exit", chunks=chunks, path=path) os._exit(137) + lock.check() checkpoint.save(last_token, len(rows)) total += len(rows) written_bytes += size log_json(logger, "chunk", path=path, rows=len(rows), bytes=size, total=total) rows = [] - chunk_start = last_token + chunk_start = {"token": last_token} if caught_up or limit: break # Nothing new: move the checkpoint to the post-batch resume token so # the next run does not scan the same idle range again. if total == 0 and stream.resume_token and stream.resume_token != token: + lock.check() checkpoint.save(stream.resume_token, 0) + lock.release() log_json(logger, "export_done", events=total, chunks=chunks, bytes=written_bytes, caught_up=caught_up, elapsed_s=round(time.monotonic() - started, 2)) client.close() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py index a8c16d55..f677cc9a 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py @@ -96,7 +96,7 @@ def main() -> None: } print(json.dumps(result, default=str, indent=None if os.getenv("COMPACT") else 2)) client.close() - sys.exit(0 if not missing and not unexpected else 1) + sys.exit(0 if not missing and not unexpected and not out_of_order else 1) if __name__ == "__main__": diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py index df6d3301..efe4131d 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py @@ -100,7 +100,7 @@ def main() -> None: } print(json.dumps(result, default=str, indent=None if os.getenv("COMPACT") else 2)) client.close() - sys.exit(0 if not missing and not unexpected and not duplicates else 1) + sys.exit(0 if not missing and not unexpected and not duplicates and not out_of_order else 1) if __name__ == "__main__": From 3db1e11028e5e748b0dc0cc6439a869d5f5d832d Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 22:09:28 +0900 Subject: [PATCH 05/14] =?UTF-8?q?fix(azure-documentdb):=20AKS=20=EC=84=9C?= =?UTF-8?q?=EB=B8=8C=EB=84=B7=20=EA=B6=8C=ED=95=9C=EC=9D=84=20=ED=81=B4?= =?UTF-8?q?=EB=9F=AC=EC=8A=A4=ED=84=B0=20=EC=83=9D=EC=84=B1=20=EC=A0=84?= =?UTF-8?q?=EC=97=90=20=EB=B6=80=EC=97=AC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 시스템 할당 ID는 클러스터 생성 후에야 생겨 Network Contributor 역할이 클러스터보다 늦게 붙었다. 사용자 할당 ID를 먼저 만들고 snet-aks 범위로 역할을 부여한 뒤 AKS가 그 역할 부여에 의존하도록 바꾼다. Co-Authored-By: Claude Opus 5.5 --- .../samples/python-aks/infra/cluster.bicep | 44 +++++++++++++------ 1 file changed, 31 insertions(+), 13 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep index 507c14b3..b333e598 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/cluster.bicep @@ -149,12 +149,41 @@ resource acr 'Microsoft.ContainerRegistry/registries@2023-07-01' = { } } +resource aksSubnet 'Microsoft.Network/virtualNetworks/subnets@2024-05-01' existing = { + parent: vnet + name: 'snet-aks' +} + +// AKS needs Network Contributor on its custom subnet before the cluster is created. +// A system-assigned identity exists only after creation, so the control plane uses +// a user-assigned identity that is granted the role first. +resource aksIdentity 'Microsoft.ManagedIdentity/userAssignedIdentities@2023-01-31' = { + name: 'id-aks-${prefix}' + location: location +} + +resource aksSubnetRole 'Microsoft.Authorization/roleAssignments@2022-04-01' = { + name: guid(aksSubnet.id, aksIdentity.id, 'netcontrib') + scope: aksSubnet + properties: { + principalId: aksIdentity.properties.principalId + principalType: 'ServicePrincipal' + roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '4d97b98b-1d4f-4787-a291-c67834d212e7') + } +} + resource aks 'Microsoft.ContainerService/managedClusters@2024-09-01' = { name: 'aks-${prefix}' location: location identity: { - type: 'SystemAssigned' + type: 'UserAssigned' + userAssignedIdentities: { + '${aksIdentity.id}': {} + } } + dependsOn: [ + aksSubnetRole + ] properties: { dnsPrefix: 'aks-${prefix}-${suffix}' // Workload identity lets the Airflow export pods use a managed identity (lake.bicep). @@ -173,7 +202,7 @@ resource aks 'Microsoft.ContainerService/managedClusters@2024-09-01' = { count: nodeCount vmSize: nodeVmSize osType: 'Linux' - vnetSubnetID: '${vnet.id}/subnets/snet-aks' + vnetSubnetID: aksSubnet.id } ] networkProfile: { @@ -197,17 +226,6 @@ resource acrPull 'Microsoft.Authorization/roleAssignments@2022-04-01' = { } } -// AKS needs Network Contributor on its subnet when using a custom VNet. -resource aksSubnetRole 'Microsoft.Authorization/roleAssignments@2022-04-01' = { - name: guid(vnet.id, aks.id, 'netcontrib') - scope: vnet - properties: { - principalId: aks.identity.principalId - principalType: 'ServicePrincipal' - roleDefinitionId: subscriptionResourceId('Microsoft.Authorization/roleDefinitions', '4d97b98b-1d4f-4787-a291-c67834d212e7') - } -} - output clusterName string = cluster.name output acrName string = acr.name output acrLoginServer string = acr.properties.loginServer From 22270d361cd06d0dc1dbc679ebc6d5639fe6c4b0 Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 22:31:32 +0900 Subject: [PATCH 06/14] =?UTF-8?q?fix(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20sample=20=EC=9E=AC=EB=A6=AC=EB=B7=B0=20=EC=A7=80=EC=A0=81=20?= =?UTF-8?q?=EB=B0=98=EC=98=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Airflow scheduler Role에서 쓰지 않는 pods/exec 권한 제거 - probe의 이벤트 형태 확인이 같은 operationType 이벤트를 덮어쓰지 않도록 목록으로 기록 - probe의 기본 update 확인이 이벤트를 받지 못하면 FAIL로 기록 - verify.py가 RUN_ID를 정규식에 넣기 전에 escape - AKS 노드 수 기본값을 측정 환경과 같은 4개로 변경 Co-Authored-By: Claude Opus 5.5 --- .../samples/python-aks/app/probe.py | 15 ++++++++++----- .../samples/python-aks/app/verify.py | 3 ++- .../samples/python-aks/infra/main.bicep | 4 ++-- .../samples/python-aks/k8s/airflow-rbac.yaml | 3 --- 4 files changed, 14 insertions(+), 11 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py index f42b68ca..af48bfc9 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/probe.py @@ -76,19 +76,21 @@ def _event_shapes(): coll.update_one({"_id": 1}, {"$set": {"a": 2}, "$unset": {"arr": ""}}) coll.replace_one({"_id": 1}, {"a": 3}) coll.delete_one({"_id": 1}) - shapes = {} + # A list, not a dict keyed by operationType: the replace may arrive as a + # second update and must not overwrite the first one. + events = [] for _ in range(4): e = next_event(s) if e is None: break - shapes[e["operationType"]] = { + events.append({ + "operationType": e["operationType"], "keys": sorted(e.keys()), "updateDescription": e.get("updateDescription"), "clusterTime_type": type(e.get("clusterTime")).__name__, "wallTime": e.get("wallTime"), - } - status = "PASS" if set(shapes) == {"insert", "update", "replace", "delete"} else "FAIL" - record("event_shapes", status, shapes=shapes) + }) + record("event_shapes", "PASS" if len(events) == 4 else "FAIL", events=events) @check("update_full_document_default") @@ -98,6 +100,9 @@ def _update_default(): with coll.watch(max_await_time_ms=500) as s: coll.update_one({"_id": 1}, {"$set": {"a": 2}}) e = next_event(s) + if e is None: + record("update_full_document_default", "FAIL", error=f"no update event within {WAIT_S}s") + return record("update_full_document_default", "INFO", has_fullDocument="fullDocument" in e, fullDocument=e.get("fullDocument"), updateDescription=e.get("updateDescription")) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py index f677cc9a..1d68547a 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py @@ -2,6 +2,7 @@ import json import os +import re import sys from collections import Counter, defaultdict @@ -47,7 +48,7 @@ def main() -> None: sink = os.getenv("SINK") or SINK_COLLECTION events = list(db[sink].find( - {"doc_id": {"$regex": f"^{run_id}:"}}, + {"doc_id": {"$regex": f"^{re.escape(run_id)}:"}}, {"doc_id": 1, "op": 1, "recv_ns": 1, "pod": 1, "has_full_document": 1, "has_update_description": 1, "deliveries": 1, "lag_ms": 1}, )) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep index 2b40bd4a..4acf0311 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/infra/main.bicep @@ -19,8 +19,8 @@ param adminUserName string = 'csadmin' @description('DocumentDB administrator password, from the DOCDB_ADMIN_PASSWORD azd environment value.') param adminPassword string -@description('AKS node count.') -param nodeCount int = 2 +@description('AKS node count. The measurements in the topic used four nodes.') +param nodeCount int = 4 resource group 'Microsoft.Resources/resourceGroups@2024-03-01' = { name: 'rg-${environmentName}' diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml index 83c70204..1efacd0b 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/k8s/airflow-rbac.yaml @@ -12,9 +12,6 @@ rules: - apiGroups: [""] resources: ["pods/log"] verbs: ["get", "list"] - - apiGroups: [""] - resources: ["pods/exec"] - verbs: ["create", "get"] - apiGroups: [""] resources: ["events"] verbs: ["list", "watch"] From 6b0ab49d968c5df8108a29433cfe67caa886f12d Mon Sep 17 00:00:00 2001 From: hellices Date: Fri, 2 Oct 2026 23:38:06 +0900 Subject: [PATCH 07/14] =?UTF-8?q?docs(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20sample=20=EC=9E=AC=EB=B0=B0=ED=8F=AC=20=ED=99=95=EC=9D=B8=20?= =?UTF-8?q?=EA=B2=B0=EA=B3=BC=20=EB=B0=98=EC=98=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 리뷰 반영 후 sample을 Azure에 다시 배포해 첫 실행 시작 위치, stream lease 경쟁, 100,000개 적재와 업로드 직후 장애 재시도를 실행했다. 로컬 fake로만 확인했다는 서술을 실측 결과로 바꾸고 wallTime 없는 이벤트와 순서 역전 경로만 로컬 확인으로 남긴다. pause 상태에서 trigger한 run은 unpause해야 실행되므로 retry 절차를 고친다. Co-Authored-By: Claude Opus 5.5 --- .../azure-documentdb/change-streams/index.md | 15 ++++++++++++--- .../change-streams/samples/python-aks/README.md | 11 ++++++++--- 2 files changed, 20 insertions(+), 6 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index cb4a475c..e88cd12f 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -163,9 +163,18 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 | 내보내기 파드 | Python 3.12, PyMongo 4.18.2, pyarrow 25.0.1, snappy 압축 | 컬렉션 하나를 하루 동안 측정한 결과입니다. 측정 뒤 리뷰에서 나온 실패 경로를 -sample에 반영했습니다. 첫 실행 시작 위치 저장, lock 파일 lease, `wallTime`이 없는 -이벤트의 파티션 고정, 검증 스크립트의 순서 역전 실패 처리입니다. 측정에서 거치지 -않은 경로이며 로컬 fake로만 확인했고 Azure에서 다시 실행하지 않았습니다. +sample에 반영했습니다. 같은 날 반영한 sample을 새 환경에 다시 배포해 아래 경로를 +확인했습니다. + +- **첫 실행 시작 위치 저장:** checkpoint 없이 시작한 첫 실행이 현재 시각을 저장하고 + 0건으로 끝났습니다. 다음 실행은 그 위치에서 이어 읽었습니다. +- **lock 파일 lease:** 다른 파드가 lease를 잡고 있는 동안 띄운 내보내기는 약 90초 + 기다린 뒤 change stream을 열지 않고 실패했습니다. +- **다시 실행한 적재:** 문서 100,000개 적재와 업로드 직후 장애 재시도를 다시 + 실행했습니다. 두 경우 모두 이벤트 230,000건이 중복과 누락 없이 들어갔습니다. +- **로컬 fake로만 확인한 경로:** `wallTime`이 없는 이벤트의 파티션 고정과 검증 + 스크립트의 순서 역전 실패 처리입니다. 이 클러스터의 이벤트에는 항상 `wallTime`이 + 있었고 순서 역전도 생기지 않았습니다. ## 공식 출처 diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index ed5ae8b2..485e7c7d 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -237,8 +237,13 @@ refuse such a run. The runs used the sample before these changes: the first-run start position, the stream lease, the `dt=unknown` partition and the stricter exit codes. -They change only failure paths that the runs did not hit, and were checked -with local fakes, not on Azure. +They change only failure paths that the runs did not hit. The changed sample +was then deployed again on Azure with the defaults in this README. r1, the +first DAG run without a checkpoint, a 100,000-document Airflow run and the +Airflow retry passed again. A second export started while another pod held the +stream lease waited about 90 seconds and failed with `LeaseAlreadyPresent` +before it opened the change stream. The `dt=unknown` partition and the +out-of-order failure were not hit there and were checked with local fakes only. | Run | Generator parameters | Extra steps | | --- | --- | --- | @@ -247,7 +252,7 @@ with local fakes, not on Azure. | r3 | `DOCS=100000 RATE=0` | Before the generator, run `kubectl -n cslab set env deployment/cs-consumer FAULT_EXIT_AFTER_WRITE=200`. The container exits after each 200-event write and restarts. Remove it with `FAULT_EXIT_AFTER_WRITE-` after the number of faults you want | | r4 | `DOCS=130000 RATE=1000` | None | | Airflow latency | `DOCS=780000 RATE=1000 PAD_BYTES=1000` | DAG unpaused for the whole run | -| Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG, run the generator, then `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container | +| Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG, run the generator, then run `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container and unpause the DAG. The triggered run stays queued while the DAG is paused and runs before the next scheduled run | | Airflow backlog | `DOCS=600000 RATE=0 PAD_BYTES=4000` | Pause the DAG, run the generator, then unpause it | All runs use `WORKERS=16`. Pause the DAG with `airflow dags pause From 69966e1ce61feb96110ac7a4fe55221580872a68 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 3 Oct 2026 12:15:39 +0900 Subject: [PATCH 08/14] =?UTF-8?q?feat(azure-documentdb):=20Parquet=20?= =?UTF-8?q?=EC=A0=81=EC=9E=AC=EB=A5=BC=2010=EB=B6=84=20=EA=B5=AC=EA=B0=84?= =?UTF-8?q?=C2=B750=20MB=20=EA=B7=9C=EC=B9=99=EC=9C=BC=EB=A1=9C=20?= =?UTF-8?q?=EB=B3=80=EA=B2=BD?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 이벤트를 wallTime 기준 10분 구간(UTC)으로 묶고 압축 전 50 MB 전에 파일을 나눈다. 열린 구간은 쓰지 않고 다음 실행으로 넘긴다. - DAG를 구간 종료 2분 뒤 CronTriggerTimetable로 실행한다. - 열린 구간 이벤트를 읽은 실행은 idle checkpoint를 저장하지 않는다. - verify_lake.py가 구간이 섞인 파일과 한도를 넘은 파일을 실패로 본다. - 내보내기 파드 메모리를 요청 512Mi, 제한 1Gi로 낮춘다. - 새 배포에서 측정한 지연, 재시도, 열린 구간 보류, 파일 크기와 메모리를 문서에 반영하고 ADLS 모범 사례를 출처에 추가한다. Co-Authored-By: Claude Opus 5.5 --- .../change-streams/images/architecture.svg | 4 +- .../azure-documentdb/change-streams/index.md | 75 +++++++-- .../change-streams/measurements/index.md | 75 ++++++++- .../samples/python-aks/README.md | 50 ++++-- .../airflow/dags/change_stream_to_parquet.py | 19 ++- .../samples/python-aks/airflow/values.yaml | 8 +- .../samples/python-aks/app/lake_export.py | 159 +++++++++++++----- .../samples/python-aks/app/verify_lake.py | 16 +- 8 files changed, 310 insertions(+), 96 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/images/architecture.svg b/docs/services/azure-documentdb/change-streams/images/architecture.svg index 12711e3e..a23b0ce1 100644 --- a/docs/services/azure-documentdb/change-streams/images/architecture.svg +++ b/docs/services/azure-documentdb/change-streams/images/architecture.svg @@ -1,6 +1,6 @@ AKS 위 Airflow DAG가 Azure DocumentDB change stream을 ADLS Gen2 Parquet으로 적재하는 구성 - AKS의 airflow 네임스페이스에서 Airflow scheduler가 5분마다 cslab 네임스페이스에 내보내기 파드를 만든다. 내보내기 파드는 프라이빗 엔드포인트를 거쳐 Azure DocumentDB의 change stream을 읽고, blob과 dfs 프라이빗 엔드포인트를 거쳐 ADLS Gen2에 Parquet 청크와 checkpoint를 쓴다. 스토리지 인증은 Microsoft Entra ID의 관리 ID와 워크로드 ID로 한다. + AKS의 airflow 네임스페이스에서 Airflow scheduler가 10분 구간이 닫힐 때마다 cslab 네임스페이스에 내보내기 파드를 만든다. 내보내기 파드는 프라이빗 엔드포인트를 거쳐 Azure DocumentDB의 change stream을 읽고, blob과 dfs 프라이빗 엔드포인트를 거쳐 ADLS Gen2에 Parquet 청크와 checkpoint를 쓴다. 스토리지 인증은 Microsoft Entra ID의 관리 ID와 워크로드 ID로 한다. @@ -15,7 +15,7 @@ namespace airflow Airflow scheduler - LocalExecutor · 5분 주기 + LocalExecutor · 10분 구간 PostgreSQL 메타데이터 diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index e88cd12f..9b62e30f 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -7,7 +7,7 @@ technologies: [python, mongodb, airflow, kubernetes] tags: [evaluate, build, storage] status: current verification_status: verified -sources_checked_at: 2026-10-02 +sources_checked_at: 2026-10-03 published_at: 2026-10-02 official_sources: - title: Change streams in Azure DocumentDB @@ -16,6 +16,8 @@ official_sources: url: https://github.com/AzureCosmosDB/changestream-driver-compatibility - title: Lease Blob url: https://learn.microsoft.com/rest/api/storageservices/lease-blob + - title: Best practices for using Azure Data Lake Storage + url: https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices --- # Azure DocumentDB change stream을 Airflow DAG로 Parquet에 내리기 @@ -49,16 +51,20 @@ change stream은 컬렉션의 변경을 이벤트로 받아 보는 MongoDB 기 ## 확인한 구성 -![AKS의 Airflow scheduler가 5분마다 내보내기 파드를 만들면 그 파드가 프라이빗 엔드포인트를 거쳐 DocumentDB change stream을 읽어 ADLS Gen2에 Parquet 청크와 checkpoint를 쓰며 Entra ID 워크로드 ID로 인증하는 구성](images/architecture.svg) +![AKS의 Airflow scheduler가 10분 구간이 닫힐 때마다 내보내기 파드를 만들면 그 파드가 프라이빗 엔드포인트를 거쳐 DocumentDB change stream을 읽어 ADLS Gen2에 Parquet 청크와 checkpoint를 쓰며 Entra ID 워크로드 ID로 인증하는 구성](images/architecture.svg) -Airflow가 5분마다 `KubernetesPodOperator`로 내보내기 파드를 하나 띄웁니다. 파드는 -다음 순서로 동작하고 끝나면 사라집니다. +Airflow가 매시 2분, 12분, 22분처럼 10분 구간이 닫히고 2분 뒤에 +`KubernetesPodOperator`로 내보내기 파드를 하나 띄웁니다. 파드는 다음 순서로 +동작하고 끝나면 사라집니다. 1. ADLS Gen2의 lock 파일에 lease를 잡고 checkpoint 파일에서 resume token을 읽습니다. checkpoint가 없으면 현재 시각을 시작 위치로 먼저 저장합니다. -2. 실행을 시작한 시각까지 기록된 이벤트를 읽어 100,000건씩 Parquet 청크로 - 올립니다. 청크 파일 이름은 청크 첫 이벤트 바로 앞의 resume token으로 만듭니다. -3. 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. +2. 이벤트를 `wallTime` 기준 10분 구간(UTC)으로 나눠 실행 전에 닫힌 구간만 + 읽습니다. 아직 열린 구간의 첫 이벤트를 만나면 쓰지 않고 멈춥니다. +3. 구간마다 Parquet 청크로 올립니다. 한 구간이 50 MB(50,000,000바이트)를 넘으면 + 그 앞에서 잘라 다음 파일로 넘깁니다. 청크 파일 이름은 구간 시작 시각과 청크 첫 + 이벤트 바로 앞의 resume token으로 만듭니다. +4. 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. DAG는 `max_active_runs=1`이고 실패하면 세 번까지 재시도합니다. 재시도는 같은 checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁니다. 첫 실행의 @@ -72,11 +78,13 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 | 확인 항목 | 결과 | | --- | --- | -| 5분 주기 지연 | 초당 약 1,000건 쓰기에서 이벤트 기록부터 파일 저장까지 p50 156초, p99 302초 | +| 10분 구간 지연 | 초당 약 960건 쓰기에서 이벤트 기록부터 파일 저장까지 p50 454초, p99 722초, 최대 731초 | | 업로드 직후 장애와 재시도 | 첫 청크를 올리고 checkpoint를 쓰기 전에 파드를 죽여도 재시도 후 중복 0, 누락 0 | -| 400 MB 활성 change log를 넘긴 재개 | DAG를 31분 멈춘 사이 쌓인 문서 본문 2.6 GB 이상의 백로그를 누락 없이 따라잡음. `maxAwaitTimeMS`를 지정하지 않았을 때만 성공 | -| 처리 속도 | 1 KB 문서는 초당 약 13,000–18,000건, 4 KB 문서 백로그는 초당 약 3,600건 | -| 파일 크기 | 이벤트에 실린 문서 크기와 비슷함. update 이벤트에도 문서 전체가 실려 문서 하나가 여러 번 저장됨 | +| 구간과 크기로 자르기 | 검증한 파일 58개가 모두 구간 하나의 이벤트만 담았고 50 MB를 넘은 파일은 없음 | +| 400 MB 활성 change log를 넘긴 재개 | DAG를 31분 멈춘 사이 쌓인 문서 본문 2.6 GB 이상의 백로그를 누락 없이 따라잡음. `maxAwaitTimeMS`를 지정하지 않았을 때만 성공. 이전 구성에서 측정 | +| 처리 속도 | 1 KB 문서 10분 구간 하나(약 600,000건)를 32–38초에 씀 | +| 파일 크기 | 압축 전 50 MB에서 자른 파일이 1 KB 랜덤 채움 문서는 약 28 MB, 채움 없는 작은 문서는 8.2 MB | +| 내보내기 파드 메모리 | 최대 RSS 358–386 MiB. 메모리 요청 512Mi, 제한 1Gi로 실행 | 지켜야 할 조건은 다음과 같습니다. @@ -85,6 +93,9 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 - 파일을 먼저 쓰고 checkpoint를 나중에 씁니다. 순서가 반대면 그 사이 장애로 이벤트를 잃습니다. - 파일 이름을 resume token으로 정해 재시도가 같은 파일을 덮어쓰게 합니다. + 자르는 위치도 이벤트의 `wallTime`과 압축 전 크기로만 정해야 재시도가 같은 + 위치에서 자릅니다. 실행 시각이나 압축 후 크기로 자르면 재시도가 다른 파일을 + 만들어 중복이 생깁니다. - checkpoint가 없는 첫 실행은 시작 위치를 먼저 저장합니다. 저장하지 않으면 첫 청크를 쓰다 실패했을 때 재시도가 더 뒤에서 시작해 그 사이 이벤트를 잃습니다. - 실행이 겹치지 않게 `max_active_runs=1`을 두고 실행 동안 lock 파일 lease를 @@ -103,7 +114,7 @@ Learn 문서의 시작 예제(Python, Java, C#, Ruby, Node.js)와 | --- | --- | --- | | 저장소의 `mongo_utils.py`가 resume token을 이벤트마다 로컬 파일 `.resume_token.json`에 씀 | 실행마다 새 파드가 뜨면 파일이 없어 현재 위치부터 읽음. 그 사이 이벤트를 잃음 | checkpoint를 ADLS Gen2 파일로 두고 청크마다 ETag 조건으로 갱신 | | Python 예제가 대상 컬렉션에 `insert_one`을 한 뒤 token을 저장 | 두 동작 사이에서 프로세스가 죽으면 재시작 후 같은 이벤트가 한 번 더 들어감 | resume token으로 파일 이름을 정해 덮어씀. 상시 consumer는 token을 키로 upsert | -| `for change in stream`처럼 끝없이 읽음 | 배치 실행이 끝나지 않음 | `try_next()`로 읽고 실행 시작 시각 이후 이벤트를 만나면 종료 | +| `for change in stream`처럼 끝없이 읽음 | 배치 실행이 끝나지 않음 | `try_next()`로 읽고 아직 열린 10분 구간의 이벤트를 만나면 쓰지 않고 종료 | | C# 예제와 저장소의 지원 확인 스크립트가 대기 시간을 1초로 지정 | 큰 백로그 재개에서 code 50(`ExceededTimeLimit`)이 같은 위치에서 반복 | `max_await_time_ms`를 지정하지 않음 | | 오류가 나면 메시지를 출력하고 끝남 | Learn 제한 사항은 장애 조치 뒤 커서를 다시 열어야 한다고 설명함 | Airflow 재시도가 새 파드에서 checkpoint로 stream을 다시 엶 | | 감시할 컬렉션이 있다고 가정 | 컬렉션이 없으면 `watch()`가 code 26 | 컬렉션을 먼저 만듦 | @@ -141,12 +152,19 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 백로그 재개가 PITR 로그를 거쳤는지는 구분하지 못했습니다. - **서버 업데이트:** `maxAwaitTimeMS`, `updateDescription`처럼 문서와 다르게 동작한 항목은 서버 버전이 바뀌면 다시 확인합니다. -- **부하와 주기:** 실행 한 번이 주기(5분) 안에 끝나야 지연이 쌓이지 않습니다. 더 - 높은 쓰기 속도와 다른 클러스터 tier에서 실행 시간을 다시 잽니다. +- **부하와 주기:** 실행 한 번이 구간 길이(10분) 안에 끝나야 지연이 쌓이지 + 않습니다. 초당 약 960건에서는 40초 안팎이었습니다. 더 높은 쓰기 속도와 다른 + 클러스터 tier에서 실행 시간을 다시 잽니다. 구간이 닫힌 뒤에야 쓰므로 지연은 최대 + 구간 길이에 대기 시간(2분)과 실행 시간을 더한 값입니다. +- **파일 크기 규칙:** ADLS 모범 사례는 분석용 파일 크기로 256 MB–100 GB를 권하고 + 작은 파일이 많으면 읽기 성능과 트랜잭션 비용이 나빠진다고 설명합니다. 50 MB는 + 지연을 줄이려고 그보다 작게 고른 값입니다. 이 한도는 압축 전 크기라서 실제 파일은 + 압축률만큼 더 작습니다. 시험에서는 28 MB와 8.2 MB였습니다. 쓰기가 적은 구간도 + 파일이 작아지므로 다운스트림에서 큰 파일로 합치는 작업을 둡니다. - **파일 크기와 압축:** 시험 데이터는 압축되지 않는 랜덤 문자열이었습니다. 실제 문서로 snappy와 zstd의 크기, 시간, CPU를 비교합니다. - **다운스트림 처리:** 파일은 이벤트 이력입니다. 문서별 최신 상태 병합, 문서가 - 없는 delete 이벤트 처리, JSON 문자열 열 펼치기, 작은 파일 정리를 설계합니다. + 없는 delete 이벤트 처리, JSON 문자열 열 펼치기를 설계합니다. - **운영 Airflow:** 차트에 포함된 PostgreSQL 대신 외부 데이터베이스를 쓰고 실패한 실행에 알림을 겁니다. - **변경 전 문서가 필요할 때:** pre-image는 지원 요청으로 켠 뒤 저장 공간과 지연 @@ -156,14 +174,15 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 | 항목 | 값 | | --- | --- | -| 측정일 | 2026-10-02, East US 2 | +| 측정일 | 2026-10-02(UTC), East US 2 | | DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 고가용성 끔, 서버 7.0.0, 공용 액세스 끔 | | AKS | Kubernetes 1.35, Standard_D4s_v6 노드 4개 | | Airflow | Helm chart 1.22.0, Airflow 3.2.2, `LocalExecutor` | | 내보내기 파드 | Python 3.12, PyMongo 4.18.2, pyarrow 25.0.1, snappy 압축 | -컬렉션 하나를 하루 동안 측정한 결과입니다. 측정 뒤 리뷰에서 나온 실패 경로를 -sample에 반영했습니다. 같은 날 반영한 sample을 새 환경에 다시 배포해 아래 경로를 +컬렉션 하나를 하루 동안 측정한 결과입니다. 처음에는 5분 주기와 100,000건 청크로 +측정했습니다. 백로그 재개와 consumer 기준선은 그 구성의 결과입니다. 이후 리뷰에서 +나온 실패 경로를 sample에 반영했고 반영한 sample을 새 환경에 다시 배포해 아래 경로를 확인했습니다. - **첫 실행 시작 위치 저장:** checkpoint 없이 시작한 첫 실행이 현재 시각을 저장하고 @@ -172,6 +191,9 @@ sample에 반영했습니다. 같은 날 반영한 sample을 새 환경에 다 기다린 뒤 change stream을 열지 않고 실패했습니다. - **다시 실행한 적재:** 문서 100,000개 적재와 업로드 직후 장애 재시도를 다시 실행했습니다. 두 경우 모두 이벤트 230,000건이 중복과 누락 없이 들어갔습니다. +- **10분 구간과 50 MB 규칙:** 같은 날 규칙을 바꾼 sample을 다시 새 환경에 배포해 + 지연, 재시도, 열린 구간 보류, 메모리를 측정했습니다. 표의 지연, 자르기, 처리 속도, + 파일 크기, 메모리 값이 이 측정입니다. - **로컬 fake로만 확인한 경로:** `wallTime`이 없는 이벤트의 파티션 고정과 검증 스크립트의 순서 역전 실패 처리입니다. 이 클러스터의 이벤트에는 항상 `wallTime`이 있었고 순서 역전도 생기지 않았습니다. @@ -181,3 +203,20 @@ sample에 반영했습니다. 같은 날 반영한 sample을 새 환경에 다 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) - [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) +- [Best practices for using Azure Data Lake Storage](https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices) + +## 더 읽을 문서 + +이번 sample에 그대로 채택하지 않았지만 구성을 바꿀 때 참고할 문서입니다. + +- AKS에 Airflow 배포: [개요](https://learn.microsoft.com/azure/aks/airflow-overview), + [인프라 만들기](https://learn.microsoft.com/azure/aks/airflow-create-infrastructure), + [배포](https://learn.microsoft.com/azure/aks/airflow-deploy). + Key Vault 비밀 연동과 운영 체크리스트가 있습니다. +- [Fabric open mirroring best practices](https://learn.microsoft.com/fabric/mirroring/open-mirroring-best-practices): + 임시 이름으로 올린 뒤 이름을 바꾸는 방식으로 파일을 원자적으로 게시합니다. +- pandas `to_parquet`로 ADLS에 쓰기: + [Synapse](https://learn.microsoft.com/azure/synapse-analytics/spark/tutorial-use-pandas-spark-pool), + [Fabric](https://learn.microsoft.com/fabric/data-science/read-write-pandas) +- [`pyarrow.fs.AzureFileSystem`](https://arrow.apache.org/docs/python/generated/pyarrow.fs.AzureFileSystem.html) +- [PyMongo `ChangeStream.try_next`](https://pymongo.readthedocs.io/en/stable/api/pymongo/change_stream.html#pymongo.change_stream.ChangeStream.try_next) diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 01b98390..bb600d00 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -1,6 +1,6 @@ --- title: Azure DocumentDB change stream Parquet 적재 측정 상세 -description: Airflow DAG의 5분 주기 지연, 업로드 직후 장애 재시도, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. +description: Airflow DAG의 10분 구간·50 MB 규칙 지연과 파일 크기, 이전 5분 주기 구성의 지연, 업로드 직후 장애 재시도, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. document_type: research services: [azure-documentdb, azure-storage, azure-kubernetes-service] technologies: [python, mongodb, airflow, kubernetes] @@ -26,7 +26,72 @@ Measured scenarios에 있습니다. 중복은 같은 resume token이 두 행 이상인 경우입니다. 저장 지연은 파일 last-modified(초 단위)에서 이벤트 `wallTime`을 뺀 값입니다. -## 5분 주기 지연 +## 10분 구간과 50 MB 규칙 + +현재 sample의 규칙입니다. 이벤트를 `wallTime` 기준 10분 구간(UTC)으로 나누고 한 +구간이 압축 전 50,000,000바이트를 넘으면 그 앞에서 다음 파일로 넘깁니다. DAG는 구간이 +닫히고 2분 뒤에 실행해 닫힌 구간만 씁니다. 2026-10-02(UTC)에 새로 배포한 환경에서 +측정했습니다. + +### 지연 + +부하는 문서 780,000개, 1 KB 채움, 초당 1,000건 제한으로 31분 동안 실행했습니다. DAG는 +계속 돌았고 생성기가 실제로 낸 속도는 초당 958건이었습니다. 이벤트 1,794,000건이 모두 +파일에 들어갔고 중복, 누락, 문서별 순서 역전은 0건이었습니다. + +| 지표 | p50 | p95 | p99 | 최대 | +| --- | --- | --- | --- | --- | +| 이벤트 기록부터 파일 저장까지 | 454초 | 701초 | 722초 | 731초 | +| 이벤트 기록부터 파드가 읽기까지 | 453초 | 700초 | 720초 | 729초 | + +- 이벤트는 자기 구간이 닫히고 2분 뒤 실행에서 저장됩니다. 구간 끝에 기록된 + 이벤트는 약 2분 반, 구간 시작에 기록된 이벤트는 약 12분을 기다립니다. p50은 구간 + 길이의 절반에 2분과 실행 시간을 더한 값과 맞습니다. +- 꽉 찬 구간 하나는 이벤트 549,336–600,001건이었고 내보내기 파드가 32–38초에 + 썼습니다. Airflow가 실행을 시작해 끝내기까지는 39–47초였습니다. +- 파일은 46개였습니다. 꽉 찬 구간은 14–15개로 나뉘었고 한도에서 자른 파일은 + 41,236–41,386행, 27.2–29.4 MB였습니다. 구간 끝의 나머지 파일과 부하가 시작하고 + 끝난 구간의 파일은 더 작았고 가장 작은 파일이 7.9 MB였습니다. 구간 둘에 걸친 파일과 50 MB를 + 넘은 파일은 없었습니다. + +### 업로드 직후 장애와 재시도 + +DAG를 멈춘 상태에서 문서 100,000개(이벤트 230,000건)를 썼습니다. 구간이 닫힌 뒤 첫 +시도만 첫 청크 업로드 직후 종료하도록 실행했습니다. 첫 시도는 checkpoint를 쓰기 전에 +종료 코드 137로 끝났습니다. 재시도는 같은 checkpoint에서 시작해 첫 청크를 같은 경로에 +덮어쓰고 나머지 5개를 이어 썼습니다. + +| 기대 이벤트 | 저장된 이벤트 | 중복 | 누락 | 파일 | +| --- | --- | --- | --- | --- | +| 230,000 | 230,000 | 0 | 0 | 6 | + +### 열린 구간 보류 + +19:22–19:23(UTC)에 이벤트 230,000건을 쓰고 19:23에 내보내기를 실행했습니다. 모든 +이벤트가 아직 열린 19:20 구간에 있어 0건을 쓰고 끝났습니다. 19:30 뒤 실행이 같은 +checkpoint에서 230,000건을 모두 읽어 6개 파일로 썼습니다. + +### 파일 크기와 메모리 + +`ru_maxrss`로 내보내기 프로세스의 최대 메모리를 쟀습니다. 각 실행은 이벤트 +230,000건입니다. + +| 문서 | 한도에서 자른 파일의 행 수 | 파일 크기 | 최대 RSS | +| --- | --- | --- | --- | +| 1 KB 랜덤 16진 채움 | 약 41,370 | 약 29 MB | 358 MiB | +| 채움 없음 | 169,109 | 8.2 MB | 386 MiB | + +- 한도는 압축 전 행 크기라서 파일은 압축률만큼 더 작습니다. 채움 없는 문서는 반복되는 + JSON 키가 잘 압축돼 50 MB 한도의 약 6분의 1이 됐습니다. +- 메모리는 청크 하나를 Python 객체와 Arrow 테이블로 함께 들고 있을 때 가장 큽니다. + 행이 작을수록 행 수가 늘어 조금 더 썼습니다. 내보내기 파드를 메모리 요청 512Mi, + 제한 1Gi로 바꾼 뒤에도 DAG가 1 KB 문서 이벤트 230,000건을 중복과 누락 없이 + 썼습니다. + +## 5분 주기 지연(이전 구성) + +이 절부터 파일 크기 절까지는 이전 구성의 측정입니다. 이전 구성은 5분마다 실행 시작 +시각까지의 이벤트를 100,000건씩 잘라 썼습니다. 부하는 문서 780,000개, 1 KB 채움, 초당 1,000건 제한으로 31분 동안 실행했습니다. DAG는 5분 주기로 계속 돌았습니다. 생성기가 실제로 낸 속도는 초당 958건이었습니다. 이벤트 @@ -47,7 +112,7 @@ last-modified(초 단위)에서 이벤트 `wallTime`을 뺀 값입니다. - 생성기만 돌 때 클러스터 CPU는 1분 평균 약 23–25%였습니다. 내보내기가 도는 분에도 같은 범위여서 1분 단위에서는 읽기 부하가 드러나지 않았습니다. -## 업로드 직후 장애와 재시도 +## 업로드 직후 장애와 재시도(이전 구성) DAG를 멈춘 상태에서 문서 100,000개(이벤트 230,000건)를 쓰고 첫 시도만 첫 청크 업로드 직후 종료하도록 실행했습니다. 첫 시도는 checkpoint를 쓰기 전에 종료 코드 137로 @@ -58,7 +123,7 @@ DAG를 멈춘 상태에서 문서 100,000개(이벤트 230,000건)를 쓰고 첫 | --- | --- | --- | --- | --- | | 230,000 | 230,000 | 0 | 0 | 3 | -## 활성 change log를 넘긴 재개 +## 활성 change log를 넘긴 재개(이전 구성) DAG를 멈춘 상태에서 생성기가 31분 동안 문서 600,000개를 4 KB씩 채워 이벤트 1,380,000건을 썼습니다. 문서 본문만 2.6 GB가 넘어 Learn이 설명하는 활성 change log @@ -109,7 +174,7 @@ full error: {'ok': 0.0, 'code': 50, 'codeName': 'ExceededTimeLimit', ...} 바로 다음 주기 실행은 새 이벤트 0건으로 1.2초 만에 끝났습니다. -## 파일 크기 +## 파일 크기(이전 구성) 백로그 재개 시험의 Parquet 합계 5.13 GB는 생성기가 쓴 문서 본문 2.6 GB의 약 두 배입니다. 가장 큰 파일(100,000행, 374 MB)을 열어 원인을 확인했습니다. diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index 485e7c7d..56bc4689 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -16,7 +16,7 @@ through a private endpoint. | `app/consumer.py` | Long-running consumer. Writes events to a sink collection and keeps the resume token in a checkpoint collection | | `app/generator.py` | Deterministic insert, update, replace and delete workload | | `app/verify.py` | Compares the generator's expected events with a sink collection | -| `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes Parquet chunks and exits | +| `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes the closed 10-minute windows as Parquet chunks of at most 50 MB of row data and exits | | `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | | `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | | `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job, RBAC for the Airflow scheduler | @@ -172,7 +172,8 @@ With chart defaults the database migration job is a post-install hook, so advises for `--wait`. The DAG is paused when Airflow first loads it. Unpause it to start the -5-minute schedule. +schedule. Runs start at minute 2, 12, 22 and so on (UTC), two minutes after +each 10-minute window closes. ```bash kubectl -n airflow exec airflow-scheduler-0 -c scheduler -- \ @@ -183,7 +184,8 @@ kubectl -n airflow exec airflow-scheduler-0 -c scheduler -- \ The first run has no checkpoint, so it saves the current time as its start position and reads from there. Write events after that run and compare them with the -Parquet files once a later run has exported them. +Parquet files once a later run has exported them. An event is exported by the +first run after its window closes. ```bash JOB_NAME=gen-af1 SCRIPT=generator.py RUN_ID=af1 DOCS=100000 WORKERS=16 RATE=0 \ @@ -196,15 +198,30 @@ kubectl -n cslab logs job/verify-lake-af1 Export behavior: -- A run stops at the first event written after the run started, or when the +- Events are grouped by `wallTime` into UTC windows of `WINDOW_MINUTES` + (10). A run reads only the windows that closed before it started. It stops + at the first event of the open window without writing it, or when the stream has nothing to return. Under steady writes `try_next()` rarely - returns `None`, so the time boundary is what ends the run. + returns `None`, so the window boundary is what ends the run. +- A chunk holds events of one window. It ends at the window boundary or + before its rows pass `CHUNK_BYTES` (50,000,000). The size is the + plain-encoded row size before compression, so a retry cuts at the same + events and the Parquet file is smaller than the limit. A window that gets + more data than the limit is split into several files. In the test, files + cut at the limit were about 29 MB with 1 KB random padding and 8.2 MB with + unpadded documents. +- An event is written about 2 to 12 minutes after it happens, plus the run + time: it waits for its window to close and for the offset. The test measured + p50 454 seconds and a maximum of 731 seconds. +- The export pod requests 512Mi of memory and is limited to 1Gi. With 50 MB + chunks its peak RSS was 358–386 MiB. - A run with no checkpoint saves the current time as the start position before it reads. Its retry starts at the same time. -- Each chunk is named after the resume token before its first event. A retry - reads the same events from the same checkpoint and overwrites the same file. - The partition comes from the first event's `wallTime`, or `dt=unknown` when - the event has none, so a retry in a later hour writes the same path. +- Each chunk is written to + `dt=/hour=/part--.parquet`. The hash comes + from the resume token before its first event. A retry reads the same events + from the same checkpoint and overwrites the same file. A chunk whose events + have no `wallTime` goes to `dt=unknown`. - The checkpoint is `_checkpoints/.json` in the same file system. It is written with an ETag condition after each chunk upload. - A run holds a 20-second lease on `_checkpoints/.lock` and renews it @@ -220,9 +237,10 @@ Export behavior: a backlog larger than the active change log failed with code 50 (`ExceededTimeLimit`) at the same event on every retry. An idle `getMore` without it still returns after about one second. -- With 4 KB documents a 100,000-event chunk is about 374 MB and the export pod - used up to about 1.9 GiB. For large documents lower `CS_CHUNK_EVENTS` in the - Airflow scheduler environment. The DAG passes it to the pod as `CHUNK_EVENTS`. +- To change the window, set `CS_WINDOW_MINUTES` in `airflow/values.yaml`. The + DAG passes it to the pod and builds its schedule from it, so the two stay + equal. `CS_EXPORT_OFFSET_MIN` (2) delays the run after the window closes. + `CS_CHUNK_BYTES` sets the file limit. - `consumer.py` always passes `MAX_AWAIT_MS`, 1000 by default. Raise it before the consumer resumes a large backlog. @@ -245,6 +263,11 @@ stream lease waited about 90 seconds and failed with `LeaseAlreadyPresent` before it opened the change stream. The `dt=unknown` partition and the out-of-order failure were not hit there and were checked with local fakes only. +The Airflow rows were run once more on a new deployment after the export +switched to 10-minute windows and 50 MB chunks. The published latency, retry, +file size and memory results for that rule come from those runs. The backlog +row was not repeated. + | Run | Generator parameters | Extra steps | | --- | --- | --- | | r1 | `DOCS=10000 RATE=0` | None | @@ -252,7 +275,8 @@ out-of-order failure were not hit there and were checked with local fakes only. | r3 | `DOCS=100000 RATE=0` | Before the generator, run `kubectl -n cslab set env deployment/cs-consumer FAULT_EXIT_AFTER_WRITE=200`. The container exits after each 200-event write and restarts. Remove it with `FAULT_EXIT_AFTER_WRITE-` after the number of faults you want | | r4 | `DOCS=130000 RATE=1000` | None | | Airflow latency | `DOCS=780000 RATE=1000 PAD_BYTES=1000` | DAG unpaused for the whole run | -| Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG, run the generator, then run `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container and unpause the DAG. The triggered run stays queued while the DAG is paused and runs before the next scheduled run | +| Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG and run the generator. After the window of its events closes, run `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container and unpause the DAG. The triggered run stays queued while the DAG is paused and runs before the next scheduled run | +| Export memory | `DOCS=100000 RATE=0`, once with `PAD_BYTES=1000` and once with `PAD_BYTES=0` | Pause the DAG. After the window closes, run `lake_export.py` with `k8s/lake.yaml` and wrap it to print `resource.getrusage(resource.RUSAGE_SELF).ru_maxrss` at exit | | Airflow backlog | `DOCS=600000 RATE=0 PAD_BYTES=4000` | Pause the DAG, run the generator, then unpause it | All runs use `WORKERS=16`. Pause the DAG with `airflow dags pause diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py index 0b885903..18e42690 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py @@ -1,8 +1,12 @@ """Export Azure DocumentDB change stream events to Parquet in ADLS Gen2. Each run starts one pod in the ``cslab`` namespace. The pod resumes from the -checkpoint in the lake, writes the events recorded before the run started and -exits. ``max_active_runs=1`` keeps a single reader per stream. +checkpoint in the lake, writes the events of the windows that closed before +the run started and exits. ``max_active_runs=1`` keeps a single reader per +stream. + +The DAG runs a few minutes after each window closes, so a run writes the +window that just closed. The window length must divide 60. Trigger with ``{"fault_after_chunks": N}`` to make the first try exit after the Nth upload and before the checkpoint, then let the retry finish the run. @@ -14,13 +18,15 @@ from airflow.providers.cncf.kubernetes.operators.pod import KubernetesPodOperator from airflow.providers.cncf.kubernetes.secret import Secret from airflow.sdk import DAG +from airflow.timetables.trigger import CronTriggerTimetable from kubernetes.client import models as k8s -INTERVAL = timedelta(minutes=int(os.getenv("CS_EXPORT_INTERVAL_MIN", "5"))) +WINDOW_MINUTES = int(os.getenv("CS_WINDOW_MINUTES", "10")) +OFFSET_MINUTES = int(os.getenv("CS_EXPORT_OFFSET_MIN", "2")) with DAG( dag_id="change_stream_to_parquet", - schedule=INTERVAL, + schedule=CronTriggerTimetable(f"{OFFSET_MINUTES}-59/{WINDOW_MINUTES} * * * *", timezone="UTC"), start_date=datetime(2026, 1, 1), catchup=False, max_active_runs=1, @@ -38,7 +44,8 @@ "LAKE_FILESYSTEM": "cdc", "LAKE_PREFIX": "orders", "STREAM_ID": "orders", - "CHUNK_EVENTS": os.getenv("CS_CHUNK_EVENTS", "100000"), + "WINDOW_MINUTES": str(WINDOW_MINUTES), + "CHUNK_BYTES": os.getenv("CS_CHUNK_BYTES", str(50_000_000)), "MAX_EVENTS": "{{ dag_run.conf.get('max_events', 0) }}", "FAULT_EXIT_AFTER_UPLOAD": "{{ dag_run.conf.get('fault_after_chunks', 0) if ti.try_number == 1 else 0 }}", @@ -47,7 +54,7 @@ service_account_name="cs-lake", labels={"azure.workload.identity/use": "true"}, container_resources=k8s.V1ResourceRequirements( - requests={"cpu": "1", "memory": "1Gi"}, limits={"memory": "4Gi"}), + requests={"cpu": "1", "memory": "512Mi"}, limits={"memory": "1Gi"}), get_logs=True, startup_timeout_seconds=300, on_finish_action="delete_succeeded_pod", diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml index 96451a03..cc75b9e7 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/values.yaml @@ -9,13 +9,13 @@ env: value: "${IMAGE}" - name: CS_LAKE_URL value: "${LAKE_URL}" - - name: CS_EXPORT_INTERVAL_MIN - value: "5" + - name: CS_WINDOW_MINUTES + value: "10" config: core: load_examples: "False" - scheduler: - dag_dir_list_interval: 30 + dag_processor: + refresh_interval: 30 triggerer: enabled: false redis: diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py index 88a0026f..3dd7e710 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -1,14 +1,20 @@ """Export change stream events to Parquet files in ADLS Gen2. -One run is one Airflow task: resume from the checkpoint, read the events -written before the run started (or up to a limit), write Parquet chunks of a -fixed size and move the checkpoint after every uploaded chunk. +One run is one Airflow task: resume from the checkpoint, read the events of +the time windows that closed before the run started (or up to a limit), write +Parquet chunks and move the checkpoint after every uploaded chunk. + +Events are grouped by ``wallTime`` into windows of ``WINDOW_MINUTES``. A chunk +holds events of one window and ends at the window boundary or before its rows +pass ``CHUNK_BYTES``. Both cut points depend only on the events, so a retry +cuts the same chunks. The run never writes the window that is still open, so +one window is not spread over several runs. A chunk is named after the resume token that precedes its first event. A retry starts from the same checkpoint, reads the same events in the same order and -overwrites the same file with the same or a longer chunk, so a crash between -the upload and the checkpoint does not leave duplicate rows. The first run -saves its start time before reading, so its retry starts at the same place. +overwrites the same file with the same chunk, so a crash between the upload +and the checkpoint does not leave duplicate rows. The first run saves its +start time before reading, so its retry starts at the same place. A run holds a lease on a lock file next to the checkpoint, so two runs never write chunks for the same stream at the same time. @@ -145,14 +151,57 @@ def to_row(change: dict) -> dict: } -def chunk_path(prefix: str, start: dict, rows: list) -> str: - """Path from the chunk start position only, so a retry rewrites the same file.""" +def row_bytes(row: dict) -> int: + """Plain-encoded size of a row: string bytes plus a 4-byte length each, + and 8 bytes per timestamp. ``read_at`` changes on a retry but its size + does not, so the same events always give the same size.""" + size = 16 + for column in ("resume_token", "op", "doc_id", "ns", "full_document"): + if row[column] is not None: + size += len(row[column].encode()) + 4 + return size + + +def window_start(wall_time, minutes: int): + if wall_time is None: + return None + return wall_time.replace(minute=wall_time.minute - wall_time.minute % minutes, + second=0, microsecond=0) + + +def chunk_path(prefix: str, start: dict, window) -> str: + """Path from the chunk start position and window only, so a retry + rewrites the same file.""" key = start["token"]["_data"] if start.get("token") else f"start-{start['start_at']}" digest = hashlib.sha256(key.encode()).hexdigest()[:20] - first = rows[0]["wall_time"] # Without wallTime the read time would move the file on a later retry. - partition = f"dt={first:%Y-%m-%d}/hour={first:%H}" if first else "dt=unknown" - return f"{prefix}/{partition}/part-{digest}.parquet" + if window is None: + return f"{prefix}/dt=unknown/part-{digest}.parquet" + return f"{prefix}/dt={window:%Y-%m-%d}/hour={window:%H}/part-{window:%Y%m%dT%H%M}-{digest}.parquet" + + +class Chunk: + """Rows of one window, started after the position in ``start``.""" + + def __init__(self, start: dict): + self.start = start + self.rows: list = [] + self.size = 0 + self.window = None + self.end_token: Optional[dict] = None + + def fits(self, size: int, window, max_bytes: int) -> bool: + if not self.rows: + return True + if window is not None and self.window is not None and window != self.window: + return False + return self.size + size <= max_bytes + + def add(self, row: dict, size: int, window, token: dict) -> None: + self.rows.append(row) + self.size += size + self.window = self.window or window + self.end_token = token def write_chunk(fs, path: str, rows: list) -> int: @@ -164,10 +213,27 @@ def write_chunk(fs, path: str, rows: list) -> int: return len(data) +def publish(fs, lock: StreamLock, checkpoint: Checkpoint, prefix: str, chunk: Chunk, fault: bool) -> int: + """Upload the chunk, then move the checkpoint past its last event.""" + path = chunk_path(prefix, chunk.start, chunk.window) + lock.check() + size = write_chunk(fs, path, chunk.rows) + if fault: + log_json(logger, "fault_exit", path=path) + os._exit(137) + lock.check() + checkpoint.save(chunk.end_token, len(chunk.rows)) + log_json(logger, "chunk", path=path, rows=len(chunk.rows), row_bytes=chunk.size, bytes=size) + return size + + def main() -> None: stream_id = env("STREAM_ID", "orders") prefix = env("LAKE_PREFIX", "orders") - chunk_events = int(os.getenv("CHUNK_EVENTS", "100000")) + window_minutes = int(os.getenv("WINDOW_MINUTES", "10")) + if window_minutes <= 0 or 60 % window_minutes: + raise SystemExit("WINDOW_MINUTES must divide 60") + chunk_bytes = int(os.getenv("CHUNK_BYTES", str(50_000_000))) max_events = int(os.getenv("MAX_EVENTS", "0")) # 0 = until caught up max_seconds = float(os.getenv("MAX_SECONDS", "0")) # Test-only: exit after uploading this many chunks, before the checkpoint. @@ -202,48 +268,49 @@ def main() -> None: kwargs["start_at_operation_time"] = position["start_at"] started = time.monotonic() - run_started_at = utcnow() - log_json(logger, "export_start", stream_id=stream_id, resume=bool(token), chunk_events=chunk_events) + # Events from the start of the open window on belong to a later run. + cutoff = window_start(utcnow(), window_minutes) + log_json(logger, "export_start", stream_id=stream_id, resume=bool(token), cutoff=cutoff, + window_minutes=window_minutes, chunk_bytes=chunk_bytes) total = chunks = written_bytes = 0 - caught_up = False + caught_up = read_any = False with source.watch(**kwargs) as stream: - rows: list = [] - chunk_start = position - last_token = token + chunk = Chunk(position) while True: change = stream.try_next() - if change is not None: - rows.append(to_row(change)) - last_token = change["_id"] - # Under steady load try_next rarely returns None, so stop once - # the stream reaches events written after this run started. - wall_time = change.get("wallTime") - caught_up = wall_time is not None and wall_time >= run_started_at - else: + if change is None: + # Every buffered row is older than the open window. caught_up = True - full = len(rows) >= chunk_events - limit = (max_events and total + len(rows) >= max_events) or \ + else: + read_any = True + wall_time = change.get("wallTime") + # Under steady load try_next rarely returns None, so stop at + # the first event of the open window. The next run reads it. + caught_up = wall_time is not None and wall_time >= cutoff + if change is not None and not caught_up: + row = to_row(change) + size = row_bytes(row) + window = window_start(wall_time, window_minutes) + if not chunk.fits(size, window, chunk_bytes): + chunks += 1 + written_bytes += publish(fs, lock, checkpoint, prefix, chunk, + fault_after_chunks and chunks >= fault_after_chunks) + total += len(chunk.rows) + chunk = Chunk({"token": chunk.end_token}) + chunk.add(row, size, window, change["_id"]) + limit = (max_events and total + len(chunk.rows) >= max_events) or \ (max_seconds and time.monotonic() - started >= max_seconds) - if rows and (full or caught_up or limit): - path = chunk_path(prefix, chunk_start, rows) - lock.check() - size = write_chunk(fs, path, rows) - chunks += 1 - if fault_after_chunks and chunks >= fault_after_chunks: - log_json(logger, "fault_exit", chunks=chunks, path=path) - os._exit(137) - lock.check() - checkpoint.save(last_token, len(rows)) - total += len(rows) - written_bytes += size - log_json(logger, "chunk", path=path, rows=len(rows), bytes=size, total=total) - rows = [] - chunk_start = {"token": last_token} if caught_up or limit: + if chunk.rows: + chunks += 1 + written_bytes += publish(fs, lock, checkpoint, prefix, chunk, + fault_after_chunks and chunks >= fault_after_chunks) + total += len(chunk.rows) break - # Nothing new: move the checkpoint to the post-batch resume token so - # the next run does not scan the same idle range again. - if total == 0 and stream.resume_token and stream.resume_token != token: + # Nothing to read: move the checkpoint to the post-batch resume token + # so the next run does not scan the same idle range again. Skip it + # when an event of the open window was read, or that event is lost. + if not read_any and stream.resume_token and stream.resume_token != token: lock.check() checkpoint.save(stream.resume_token, 0) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py index efe4131d..f3491120 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py @@ -13,6 +13,7 @@ from azure.storage.filedatalake import DataLakeServiceClient from common import RUN_COLLECTION, env, get_client, get_database +from lake_export import window_start from verify import expected_events, percentile, summarize logging.getLogger("azure").setLevel(logging.WARNING) @@ -29,6 +30,8 @@ def main() -> None: service = DataLakeServiceClient(env("LAKE_URL"), credential=DefaultAzureCredential()) fs = service.get_file_system_client(env("LAKE_FILESYSTEM", "cdc")) prefix = env("LAKE_PREFIX", "orders") + window_minutes = int(os.getenv("WINDOW_MINUTES", "10")) + chunk_bytes = int(os.getenv("CHUNK_BYTES", str(50_000_000))) columns = ["resume_token", "op", "doc_id", "wall_time", "read_at"] rows = [] @@ -41,7 +44,11 @@ def main() -> None: mine = [r for r in table.to_pylist() if r["doc_id"].startswith(f"{run_id}:")] if not mine: continue - files.append({"rows": len(mine), "file_rows": table.num_rows, "bytes": path.content_length}) + # A file holds one window, so every row with wallTime maps to it. + windows = {window_start(t, window_minutes) for t in table.column("wall_time").to_pylist() + if t is not None} + files.append({"rows": len(mine), "file_rows": table.num_rows, "bytes": path.content_length, + "windows": len(windows)}) # The listing returns last-modified as a naive UTC datetime. committed_at = path.last_modified.replace(tzinfo=timezone.utc) for i, r in enumerate(mine): @@ -75,6 +82,8 @@ def main() -> None: commit_lag_s = [(e["committed_at"] - e["wall_time"]).total_seconds() for e in events if e["wall_time"]] file_rows = [f["file_rows"] for f in files] file_bytes = [f["bytes"] for f in files] + mixed_windows = sum(1 for f in files if f["windows"] > 1) + oversize = sum(1 for b in file_bytes if b > chunk_bytes) result = { "run_id": run_id, @@ -97,10 +106,13 @@ def main() -> None: "max": max(file_rows, default=None)}, "file_bytes": {"min": min(file_bytes, default=None), "p50": percentile(file_bytes, 50), "max": max(file_bytes, default=None), "total": sum(file_bytes)}, + "files_with_mixed_windows": mixed_windows, + "files_over_chunk_bytes": oversize, } print(json.dumps(result, default=str, indent=None if os.getenv("COMPACT") else 2)) client.close() - sys.exit(0 if not missing and not unexpected and not duplicates and not out_of_order else 1) + ok = not (missing or unexpected or duplicates or out_of_order or mixed_windows or oversize) + sys.exit(0 if ok else 1) if __name__ == "__main__": From 7c77ef694d954a6a4e68f085d534e54633ee49dd Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 3 Oct 2026 12:35:20 +0900 Subject: [PATCH 09/14] =?UTF-8?q?fix(azure-documentdb):=20wallTime=20?= =?UTF-8?q?=EC=97=86=EB=8A=94=20=EC=9D=B4=EB=B2=A4=ED=8A=B8=EB=A5=BC=20?= =?UTF-8?q?=EB=B3=84=EB=8F=84=20=EA=B5=AC=EA=B0=84=EC=9C=BC=EB=A1=9C=20?= =?UTF-8?q?=EB=B6=84=EB=A6=AC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Chunk가 wallTime 없는 이벤트와 날짜가 있는 구간을 한 파일에 섞지 않도록 구간 값을 그대로 비교 - verify_lake.py가 wallTime 없는 행도 하나의 구간으로 세어 섞인 파일을 실패로 판정 - 측정 문서의 fullDocument 설명을 insert·update로 한정하고 큰 문서 크기를 7 MiB insert, 약 14 MiB update로 정정 Co-Authored-By: Claude Opus 5.5 --- .../azure-documentdb/change-streams/measurements/index.md | 7 ++++--- .../change-streams/samples/python-aks/app/lake_export.py | 6 ++++-- .../change-streams/samples/python-aks/app/verify_lake.py | 6 +++--- 3 files changed, 11 insertions(+), 8 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index bb600d00..636a6958 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -187,8 +187,9 @@ full error: {'ok': 0.0, 'code': 50, 'codeName': 'ExceededTimeLimit', ...} - 파일 크기의 99%가 문서 본문입니다. 이 파일의 행은 insert 43,567건, update 47,720건, delete 8,713건이었고 delete를 뺀 91,287행에 문서 전체가 들어 있었습니다. 문서는 BSON 기준 평균 4,129바이트였습니다. -- change stream은 변경마다 문서 전체를 보냅니다. 이 클러스터는 update에도 - `fullDocument`를 넣습니다. 그래서 문서 하나가 insert와 update에 한 번씩 저장됩니다. +- change stream은 insert와 update 이벤트에 문서 전체를 싣습니다. 이 클러스터는 + 옵션 없이도 update에 `fullDocument`를 넣고 delete에는 본문이 없습니다. 그래서 문서 + 하나가 insert와 update에 한 번씩 저장됩니다. 문서가 실린 이벤트 약 1,260,000건 × 약 4.1 KB가 약 5.2 GB이고 Parquet 합계와 거의 같습니다. - 크기 대부분은 생성기가 채운 4,000자 랜덤 16진 문자열입니다. 반복이 없어 snappy는 @@ -255,7 +256,7 @@ upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 | `showExpandedEvents` | 지원하지 않음 | code 115(CommandNotSupported) | | 감시 중인 컬렉션 drop, rename | 언급 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | | 트랜잭션 | 언급 없음 | 이벤트는 오지만 `txnNumber`, `lsid`는 없음 | -| 큰 문서 | 언급 없음 | 14 MiB 문서의 insert와 update 이벤트 모두 전달됨 | +| 큰 문서 | 언급 없음 | 7 MiB 문서의 insert와 약 14 MiB로 커진 update 이벤트 모두 전달됨 | | 대기 중 resume token | 언급 없음 | 이벤트가 없어도 post-batch resume token이 전진함 | ## 측정의 한계 diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py index 3dd7e710..67886860 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -193,14 +193,16 @@ def __init__(self, start: dict): def fits(self, size: int, window, max_bytes: int) -> bool: if not self.rows: return True - if window is not None and self.window is not None and window != self.window: + # None (no wallTime) is its own window, kept apart from dated ones. + if window != self.window: return False return self.size + size <= max_bytes def add(self, row: dict, size: int, window, token: dict) -> None: + if not self.rows: + self.window = window self.rows.append(row) self.size += size - self.window = self.window or window self.end_token = token diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py index f3491120..ad4fb9e6 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py @@ -44,9 +44,9 @@ def main() -> None: mine = [r for r in table.to_pylist() if r["doc_id"].startswith(f"{run_id}:")] if not mine: continue - # A file holds one window, so every row with wallTime maps to it. - windows = {window_start(t, window_minutes) for t in table.column("wall_time").to_pylist() - if t is not None} + # A file holds one window. Rows without wallTime count as their own + # window, so a file that mixes them with dated rows fails too. + windows = {window_start(t, window_minutes) for t in table.column("wall_time").to_pylist()} files.append({"rows": len(mine), "file_rows": table.num_rows, "bytes": path.content_length, "windows": len(windows)}) # The listing returns last-modified as a naive UTC datetime. From 92f3796d694269cb10630fe01a79d7e053f57027 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 3 Oct 2026 14:26:53 +0900 Subject: [PATCH 10/14] =?UTF-8?q?feat(azure-documentdb):=20Parquet=20?= =?UTF-8?q?=EC=B2=AD=ED=81=AC=20=ED=95=9C=EB=8F=84=EB=A5=BC=20=EC=95=95?= =?UTF-8?q?=EC=B6=95=20=EC=A0=84=2080=20MB=EB=A1=9C=20=EC=83=81=ED=96=A5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 행을 10,000건마다 Arrow RecordBatch로 옮기고 업로드할 때 버퍼 사본을 만들지 않도록 바꿔 50 MB 한도의 최대 RSS를 358/386 MiB에서 305/253 MiB로 줄임 - CHUNK_BYTES 기본값을 80,000,000으로 올림. 1 KB 채움 문서 파일은 43.7-46.7 MB, 채움 없는 문서는 13.3-13.4 MB - 50/80/100/200 MB 한도 비교, 80 MB Airflow 백로그 실행과 업로드 직후 장애 재시도 결과를 측정 상세에 추가 - 재시도가 다른 위치에서 잘라도 중복이 생기지 않는다는 점을 반영해 파일 이름 조건 설명을 고침 - Airflow 재시도 시험 절차를 DAG를 켠 채 수동 실행하도록 수정 Co-Authored-By: Claude Opus 5.5 --- .../azure-documentdb/change-streams/index.md | 49 ++++++----- .../change-streams/measurements/index.md | 81 ++++++++++++++----- .../samples/python-aks/README.md | 35 +++++--- .../airflow/dags/change_stream_to_parquet.py | 2 +- .../samples/python-aks/app/lake_export.py | 50 ++++++++---- .../samples/python-aks/app/verify_lake.py | 2 +- 6 files changed, 147 insertions(+), 72 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index 9b62e30f..2c5f014b 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -61,8 +61,8 @@ Airflow가 매시 2분, 12분, 22분처럼 10분 구간이 닫히고 2분 뒤에 읽습니다. checkpoint가 없으면 현재 시각을 시작 위치로 먼저 저장합니다. 2. 이벤트를 `wallTime` 기준 10분 구간(UTC)으로 나눠 실행 전에 닫힌 구간만 읽습니다. 아직 열린 구간의 첫 이벤트를 만나면 쓰지 않고 멈춥니다. -3. 구간마다 Parquet 청크로 올립니다. 한 구간이 50 MB(50,000,000바이트)를 넘으면 - 그 앞에서 잘라 다음 파일로 넘깁니다. 청크 파일 이름은 구간 시작 시각과 청크 첫 +3. 구간마다 Parquet 청크로 올립니다. 한 구간이 압축 전 80 MB(80,000,000바이트)를 + 넘으면 그 앞에서 잘라 다음 파일로 넘깁니다. 청크 파일 이름은 구간 시작 시각과 청크 첫 이벤트 바로 앞의 resume token으로 만듭니다. 4. 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. @@ -79,12 +79,12 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 | 확인 항목 | 결과 | | --- | --- | | 10분 구간 지연 | 초당 약 960건 쓰기에서 이벤트 기록부터 파일 저장까지 p50 454초, p99 722초, 최대 731초 | -| 업로드 직후 장애와 재시도 | 첫 청크를 올리고 checkpoint를 쓰기 전에 파드를 죽여도 재시도 후 중복 0, 누락 0 | -| 구간과 크기로 자르기 | 검증한 파일 58개가 모두 구간 하나의 이벤트만 담았고 50 MB를 넘은 파일은 없음 | +| 업로드 직후 장애와 재시도 | 첫 청크를 올리고 checkpoint를 쓰기 전에 파드를 죽여도 재시도 후 중복 0, 누락 0. 50 MB와 80 MB 한도에서 각각 확인 | +| 구간과 크기로 자르기 | 50, 80, 100, 200 MB 한도로 쓴 파일이 모두 구간 하나의 이벤트만 담았고 한도를 넘은 파일은 없음 | | 400 MB 활성 change log를 넘긴 재개 | DAG를 31분 멈춘 사이 쌓인 문서 본문 2.6 GB 이상의 백로그를 누락 없이 따라잡음. `maxAwaitTimeMS`를 지정하지 않았을 때만 성공. 이전 구성에서 측정 | -| 처리 속도 | 1 KB 문서 10분 구간 하나(약 600,000건)를 32–38초에 씀 | -| 파일 크기 | 압축 전 50 MB에서 자른 파일이 1 KB 랜덤 채움 문서는 약 28 MB, 채움 없는 작은 문서는 8.2 MB | -| 내보내기 파드 메모리 | 최대 RSS 358–386 MiB. 메모리 요청 512Mi, 제한 1Gi로 실행 | +| 처리 속도 | 1 KB 문서 10분 구간 하나(약 600,000건)를 32–38초에 씀. 한도를 50–200 MB로 바꿔도 시간은 비슷함 | +| 파일 크기 | 압축 전 80 MB에서 자른 파일이 1 KB 랜덤 채움 문서는 43.7–46.7 MB, 채움 없는 작은 문서는 13.3–13.4 MB. 파일 크기는 한도에 비례함 | +| 내보내기 파드 메모리 | 80 MB 한도에서 최대 RSS 336 MiB. 200 MB 한도는 620 MiB. 메모리 요청 512Mi, 제한 1Gi로 실행 | 지켜야 할 조건은 다음과 같습니다. @@ -92,10 +92,10 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 백로그를 재개할 때 code 50으로 실패하고 재시도로도 넘어가지 못했습니다. - 파일을 먼저 쓰고 checkpoint를 나중에 씁니다. 순서가 반대면 그 사이 장애로 이벤트를 잃습니다. -- 파일 이름을 resume token으로 정해 재시도가 같은 파일을 덮어쓰게 합니다. - 자르는 위치도 이벤트의 `wallTime`과 압축 전 크기로만 정해야 재시도가 같은 - 위치에서 자릅니다. 실행 시각이나 압축 후 크기로 자르면 재시도가 다른 파일을 - 만들어 중복이 생깁니다. +- 파일 이름을 청크 첫 이벤트 바로 앞의 resume token으로 정해 재시도가 같은 파일을 + 덮어쓰게 합니다. checkpoint보다 앞서 올라간 파일은 마지막 청크 하나뿐이라 재시도가 + 다른 위치에서 잘라도 그 파일을 덮어씁니다. 실행 시각처럼 재시도마다 바뀌는 값을 + 파일 이름에 넣으면 다른 파일이 생겨 중복이 됩니다. - checkpoint가 없는 첫 실행은 시작 위치를 먼저 저장합니다. 저장하지 않으면 첫 청크를 쓰다 실패했을 때 재시도가 더 뒤에서 시작해 그 사이 이벤트를 잃습니다. - 실행이 겹치지 않게 `max_active_runs=1`을 두고 실행 동안 lock 파일 lease를 @@ -157,10 +157,12 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 클러스터 tier에서 실행 시간을 다시 잽니다. 구간이 닫힌 뒤에야 쓰므로 지연은 최대 구간 길이에 대기 시간(2분)과 실행 시간을 더한 값입니다. - **파일 크기 규칙:** ADLS 모범 사례는 분석용 파일 크기로 256 MB–100 GB를 권하고 - 작은 파일이 많으면 읽기 성능과 트랜잭션 비용이 나빠진다고 설명합니다. 50 MB는 - 지연을 줄이려고 그보다 작게 고른 값입니다. 이 한도는 압축 전 크기라서 실제 파일은 - 압축률만큼 더 작습니다. 시험에서는 28 MB와 8.2 MB였습니다. 쓰기가 적은 구간도 - 파일이 작아지므로 다운스트림에서 큰 파일로 합치는 작업을 둡니다. + 작은 파일이 많으면 읽기 성능과 트랜잭션 비용이 나빠진다고 설명합니다. sample의 + 한도는 압축 전 80 MB이고 시험에서 파일은 47 MB 이하였습니다. 압축이 거의 되지 않는 + 문서라면 파일이 80 MB에 가까워집니다. 파일을 50 MB 아래로 지켜야 하면 한도를 + 50,000,000으로 둡니다. 한도를 올리면 메모리도 늘어 200 MB에서 620 MiB였습니다. + 쓰기가 적은 구간도 파일이 작아지므로 다운스트림에서 큰 파일로 합치는 작업을 + 둡니다. - **파일 크기와 압축:** 시험 데이터는 압축되지 않는 랜덤 문자열이었습니다. 실제 문서로 snappy와 zstd의 크기, 시간, CPU를 비교합니다. - **다운스트림 처리:** 파일은 이벤트 이력입니다. 문서별 최신 상태 병합, 문서가 @@ -174,7 +176,7 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 | 항목 | 값 | | --- | --- | -| 측정일 | 2026-10-02(UTC), East US 2 | +| 측정일 | 2026-10-02–03(UTC), East US 2 | | DocumentDB | M30(2 vCore, 8 GiB), shard 1개, 스토리지 32 GiB, 고가용성 끔, 서버 7.0.0, 공용 액세스 끔 | | AKS | Kubernetes 1.35, Standard_D4s_v6 노드 4개 | | Airflow | Helm chart 1.22.0, Airflow 3.2.2, `LocalExecutor` | @@ -192,11 +194,16 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 - **다시 실행한 적재:** 문서 100,000개 적재와 업로드 직후 장애 재시도를 다시 실행했습니다. 두 경우 모두 이벤트 230,000건이 중복과 누락 없이 들어갔습니다. - **10분 구간과 50 MB 규칙:** 같은 날 규칙을 바꾼 sample을 다시 새 환경에 배포해 - 지연, 재시도, 열린 구간 보류, 메모리를 측정했습니다. 표의 지연, 자르기, 처리 속도, - 파일 크기, 메모리 값이 이 측정입니다. -- **로컬 fake로만 확인한 경로:** `wallTime`이 없는 이벤트의 파티션 고정과 검증 - 스크립트의 순서 역전 실패 처리입니다. 이 클러스터의 이벤트에는 항상 `wallTime`이 - 있었고 순서 역전도 생기지 않았습니다. + 지연, 재시도, 열린 구간 보류를 측정했습니다. 표의 지연과 처리 속도 값이 이 + 측정입니다. +- **한도 비교와 80 MB:** 다음 날 같은 환경에서 내보내기가 행을 Arrow 배치로 옮기게 + 바꾸고 한도를 50, 80, 100, 200 MB로 바꿔 파일 크기와 메모리를 쟀습니다. 기본값을 + 80 MB로 올린 뒤 Airflow 실행과 업로드 직후 장애 재시도를 다시 확인했습니다. 표의 + 자르기, 파일 크기, 메모리, 재시도 값이 이 측정입니다. +- **로컬 fake로만 확인한 경로:** `wallTime`이 없는 이벤트의 파티션 고정, 검증 + 스크립트의 순서 역전 실패 처리, 재시도가 첫 시도와 다른 위치에서 자르는 경우입니다. + 이 클러스터의 이벤트에는 항상 `wallTime`이 있었고 순서 역전도 생기지 않았습니다. + Azure에서 재시도는 첫 시도와 같은 한도로 실행해 같은 위치에서 잘랐습니다. ## 공식 출처 diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 636a6958..4308f30f 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -1,6 +1,6 @@ --- title: Azure DocumentDB change stream Parquet 적재 측정 상세 -description: Airflow DAG의 10분 구간·50 MB 규칙 지연과 파일 크기, 이전 5분 주기 구성의 지연, 업로드 직후 장애 재시도, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. +description: Airflow DAG의 10분 구간 지연, 크기 한도별 파일 크기와 메모리, 이전 5분 주기 구성의 지연, 업로드 직후 장애 재시도, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. document_type: research services: [azure-documentdb, azure-storage, azure-kubernetes-service] technologies: [python, mongodb, airflow, kubernetes] @@ -26,12 +26,17 @@ Measured scenarios에 있습니다. 중복은 같은 resume token이 두 행 이상인 경우입니다. 저장 지연은 파일 last-modified(초 단위)에서 이벤트 `wallTime`을 뺀 값입니다. -## 10분 구간과 50 MB 규칙 +## 10분 구간과 크기 한도 현재 sample의 규칙입니다. 이벤트를 `wallTime` 기준 10분 구간(UTC)으로 나누고 한 -구간이 압축 전 50,000,000바이트를 넘으면 그 앞에서 다음 파일로 넘깁니다. DAG는 구간이 -닫히고 2분 뒤에 실행해 닫힌 구간만 씁니다. 2026-10-02(UTC)에 새로 배포한 환경에서 -측정했습니다. +구간이 압축 전 80,000,000바이트를 넘으면 그 앞에서 다음 파일로 넘깁니다. DAG는 구간이 +닫히고 2분 뒤에 실행해 닫힌 구간만 씁니다. + +처음에는 한도를 압축 전 50 MB로 두고 2026-10-02(UTC)에 새로 배포한 환경에서 +측정했습니다. 그 한도에서 자른 파일이 29 MB 안팎에 그쳐 2026-10-03에 한도별 파일 +크기와 메모리를 비교했고 기본값을 80 MB로 올렸습니다. 지연, 재시도, 열린 구간 보류 +절의 수치는 50 MB 구성에서 쟀습니다. 80 MB 구성의 Airflow 실행과 재시도는 이 절 끝의 +"80 MB 한도로 다시 실행"에 있습니다. ### 지연 @@ -71,22 +76,60 @@ DAG를 멈춘 상태에서 문서 100,000개(이벤트 230,000건)를 썼습니 이벤트가 아직 열린 19:20 구간에 있어 0건을 쓰고 끝났습니다. 19:30 뒤 실행이 같은 checkpoint에서 230,000건을 모두 읽어 6개 파일로 썼습니다. -### 파일 크기와 메모리 +### 한도별 파일 크기와 메모리 -`ru_maxrss`로 내보내기 프로세스의 최대 메모리를 쟀습니다. 각 실행은 이벤트 -230,000건입니다. +2026-10-03에 같은 이벤트를 한도만 바꿔 내보냈습니다. 문서 200,000개를 1 KB 랜덤 16진 +문자열로 채운 이벤트 460,000건과 채움 없는 문서 450,000개의 이벤트 1,035,000건입니다. +데이터마다 구간 하나에 쓰고 한도별 내보내기를 따로 실행해 `ru_maxrss`로 최대 메모리를 +쟀습니다. 80 MB 행의 파일 크기는 아래 Airflow 실행에서 나왔습니다. 그 메모리는 1 KB 채움 +문서 300,000개의 이벤트 690,000건을 같은 한도로 따로 내보내 쟀고 채움 없는 문서는 재지 +않았습니다. -| 문서 | 한도에서 자른 파일의 행 수 | 파일 크기 | 최대 RSS | -| --- | --- | --- | --- | -| 1 KB 랜덤 16진 채움 | 약 41,370 | 약 29 MB | 358 MiB | -| 채움 없음 | 169,109 | 8.2 MB | 386 MiB | - -- 한도는 압축 전 행 크기라서 파일은 압축률만큼 더 작습니다. 채움 없는 문서는 반복되는 - JSON 키가 잘 압축돼 50 MB 한도의 약 6분의 1이 됐습니다. -- 메모리는 청크 하나를 Python 객체와 Arrow 테이블로 함께 들고 있을 때 가장 큽니다. - 행이 작을수록 행 수가 늘어 조금 더 썼습니다. 내보내기 파드를 메모리 요청 512Mi, - 제한 1Gi로 바꾼 뒤에도 DAG가 1 KB 문서 이벤트 230,000건을 중복과 누락 없이 - 썼습니다. +| 압축 전 한도 | 1 KB 채움: 한도에서 자른 파일 | 1 KB 채움: 최대 RSS | 채움 없음: 한도에서 자른 파일 | 채움 없음: 최대 RSS | +| --- | --- | --- | --- | --- | +| 50 MB | 41,211–41,288행, 28.1–29.3 MB | 305 MiB | 168,510–169,106행, 8.3–8.5 MB | 253 MiB | +| 80 MB | 65,917–66,204행, 43.7–46.7 MB | 336 MiB | 269,618–270,400행, 13.3–13.4 MB | 재지 않음 | +| 100 MB | 82,475–82,560행, 57.6–58.4 MB | 390 MiB | 337,023–337,805행, 16.6–16.8 MB | 335 MiB | +| 200 MB | 165,006–165,099행, 115.4–116.7 MB | 620 MiB | 674,828행, 32.8 MB | 519 MiB | + +- 파일 크기는 한도에 비례했습니다. 1 KB 채움 문서는 한도의 약 0.55–0.58배, 채움 없는 + 문서는 약 0.17배였습니다. 채움 없는 문서는 반복되는 JSON 키가 잘 압축됩니다. +- 80 MB는 압축이 잘 안 되는 1 KB 채움 문서에서도 파일이 50 MB를 넘지 않는 값으로 + 골랐습니다. 압축이 거의 되지 않는 문서라면 파일이 80 MB에 가까워질 수 있습니다. +- 구간 하나를 쓰는 시간은 한도와 관계없이 1 KB 채움 문서 38–41초, 채움 없는 문서 + 38–40초였습니다. 세 한도를 동시에 실행한 값입니다. +- 메모리는 청크 하나를 Arrow 테이블로 만들어 Parquet으로 쓸 때 가장 큽니다. 이 + 측정부터 내보내기는 행을 10,000건마다 Arrow 배치로 옮겨 청크 전체를 Python 객체와 + Arrow 사본으로 함께 들고 있지 않습니다. 같은 50 MB 한도에서 이전 sample의 최대 RSS는 1 KB 채움 + 문서가 358 MiB, 채움 없는 문서가 386 MiB였습니다. +- 내보내기 파드는 메모리 요청 512Mi, 제한 1Gi로 실행합니다. 200 MB 한도의 620 MiB는 + 요청을 넘습니다. 한도를 올리면 파드 메모리도 함께 늘립니다. + +### 80 MB 한도로 다시 실행 + +기본값을 80 MB로 바꾼 이미지를 Airflow에 배포했습니다. DAG를 멈춘 사이 구간 세 개에 +이벤트 2,185,000건이 쌓였습니다. 위 비교에 쓴 두 데이터와 1 KB 채움 문서 300,000개의 +이벤트 690,000건입니다. DAG를 다시 켜자 정기 실행 한 번이 세 구간을 모두 읽어 +175.8초에 파일 22개를 썼습니다. + +| 데이터 | 이벤트 | 파일 | 중복 | 누락 | 순서 역전 | +| --- | --- | --- | --- | --- | --- | +| 1 KB 채움, 문서 200,000개 | 460,000 | 7 | 0 | 0 | 0 | +| 채움 없음, 문서 450,000개 | 1,035,000 | 4 | 0 | 0 | 0 | +| 1 KB 채움, 문서 300,000개 | 690,000 | 11 | 0 | 0 | 0 | + +구간 둘에 걸친 파일과 한도를 넘은 파일은 없었습니다. 한도에서 자른 파일의 크기는 위 +표의 80 MB 행과 같습니다. + +업로드 직후 장애와 재시도도 80 MB 한도로 다시 확인했습니다. DAG를 켠 채 1 KB 채움 문서 +300,000개(이벤트 690,000건)를 04:50 구간에 썼습니다. 구간이 닫힌 05:00에 첫 시도만 첫 +청크 업로드 직후 종료하도록 실행했습니다. 첫 시도는 checkpoint를 쓰기 전에 끝났고 +재시도가 같은 경로에 첫 청크를 덮어쓴 뒤 나머지 10개를 이어 썼습니다. 재시도는 +58.4초 걸렸고 다음 정기 실행은 재시도가 끝난 뒤 시작했습니다. + +| 기대 이벤트 | 저장된 이벤트 | 중복 | 누락 | 파일 | +| --- | --- | --- | --- | --- | +| 690,000 | 690,000 | 0 | 0 | 11 | ## 5분 주기 지연(이전 구성) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index 56bc4689..c99f25aa 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -16,7 +16,7 @@ through a private endpoint. | `app/consumer.py` | Long-running consumer. Writes events to a sink collection and keeps the resume token in a checkpoint collection | | `app/generator.py` | Deterministic insert, update, replace and delete workload | | `app/verify.py` | Compares the generator's expected events with a sink collection | -| `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes the closed 10-minute windows as Parquet chunks of at most 50 MB of row data and exits | +| `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes the closed 10-minute windows as Parquet chunks of at most 80 MB of row data and exits | | `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | | `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | | `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job, RBAC for the Airflow scheduler | @@ -204,17 +204,21 @@ Export behavior: stream has nothing to return. Under steady writes `try_next()` rarely returns `None`, so the window boundary is what ends the run. - A chunk holds events of one window. It ends at the window boundary or - before its rows pass `CHUNK_BYTES` (50,000,000). The size is the - plain-encoded row size before compression, so a retry cuts at the same - events and the Parquet file is smaller than the limit. A window that gets - more data than the limit is split into several files. In the test, files - cut at the limit were about 29 MB with 1 KB random padding and 8.2 MB with - unpadded documents. + before its rows pass `CHUNK_BYTES` (80,000,000). The size is the + plain-encoded row size before compression, so the Parquet file is smaller + than the limit. A window that gets more data than the limit is split into + several files. In the test, files cut at the limit were 43.7–46.7 MB with + 1 KB random padding and 13.3–13.4 MB with unpadded documents. The file is a + fixed share of the limit that depends on how well the data compresses, so + data that hardly compresses gets files close to 80 MB. Set `CS_CHUNK_BYTES` to 50,000,000 if files must stay under 50 MB. - An event is written about 2 to 12 minutes after it happens, plus the run time: it waits for its window to close and for the offset. The test measured p50 454 seconds and a maximum of 731 seconds. -- The export pod requests 512Mi of memory and is limited to 1Gi. With 50 MB - chunks its peak RSS was 358–386 MiB. +- Rows are moved into Arrow record batches every 10,000 rows, so a chunk is + not held as Python objects and an Arrow copy at the same time. The export + pod requests 512Mi of memory and is limited to 1Gi. Peak RSS was 336 MiB + with 80 MB chunks of 1 KB padded documents, 390 MiB at 100 MB and 620 MiB + at 200 MB. Raise the pod memory with the limit. - A run with no checkpoint saves the current time as the start position before it reads. Its retry starts at the same time. - Each chunk is written to @@ -264,9 +268,12 @@ before it opened the change stream. The `dt=unknown` partition and the out-of-order failure were not hit there and were checked with local fakes only. The Airflow rows were run once more on a new deployment after the export -switched to 10-minute windows and 50 MB chunks. The published latency, retry, -file size and memory results for that rule come from those runs. The backlog -row was not repeated. +switched to 10-minute windows and 50 MB chunks. The published latency results +come from those runs. The next day the export started to move rows into Arrow +batches and the default limit became 80 MB. The chunk limit comparison, the +Airflow 80 MB backlog and the Airflow retry were run on the same deployment +with that sample. The published file size, memory and 80 MB retry results come +from them. The backlog row was not repeated. | Run | Generator parameters | Extra steps | | --- | --- | --- | @@ -275,8 +282,10 @@ row was not repeated. | r3 | `DOCS=100000 RATE=0` | Before the generator, run `kubectl -n cslab set env deployment/cs-consumer FAULT_EXIT_AFTER_WRITE=200`. The container exits after each 200-event write and restarts. Remove it with `FAULT_EXIT_AFTER_WRITE-` after the number of faults you want | | r4 | `DOCS=130000 RATE=1000` | None | | Airflow latency | `DOCS=780000 RATE=1000 PAD_BYTES=1000` | DAG unpaused for the whole run | -| Airflow retry | `DOCS=100000 RATE=0 PAD_BYTES=1000` | Pause the DAG and run the generator. After the window of its events closes, run `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container and unpause the DAG. The triggered run stays queued while the DAG is paused and runs before the next scheduled run | +| Airflow retry | `DOCS=300000 RATE=0 PAD_BYTES=1000` (`DOCS=100000` with 50 MB chunks) | Keep the DAG unpaused. Start the generator right after a window opens so its events fall in that window. After the window closes and before the next scheduled run, in the first two minutes of the next window, run `airflow dags trigger change_stream_to_parquet -c '{"fault_after_chunks": 1}'` in the scheduler container. Do not pause the DAG for this run. Unpausing creates the missed scheduled run at once, and that run can read the events before the triggered run | | Export memory | `DOCS=100000 RATE=0`, once with `PAD_BYTES=1000` and once with `PAD_BYTES=0` | Pause the DAG. After the window closes, run `lake_export.py` with `k8s/lake.yaml` and wrap it to print `resource.getrusage(resource.RUSAGE_SELF).ru_maxrss` at exit | +| Chunk limit comparison | `DOCS=200000 RATE=0 PAD_BYTES=1000`, then `DOCS=450000 RATE=0 PAD_BYTES=0` in the next window | Pause the DAG. Before the generator, run `lake_export.py` once per limit with its own `LAKE_PREFIX`, which is also the stream ID, so each saves its own start position. After each window closes, run the exports again with `CHUNK_BYTES` added to the Job env (50000000, 100000000 or 200000000) and wrapped as in Export memory. Verify each prefix with `verify_lake.py`. The 80 MB memory was measured the same way on the third run of the next row | +| Airflow 80 MB backlog | The two runs above, then `DOCS=300000 RATE=0 PAD_BYTES=1000` in another window | Keep the DAG paused while the three runs write, then unpause it. One scheduled run reads the three windows | | Airflow backlog | `DOCS=600000 RATE=0 PAD_BYTES=4000` | Pause the DAG, run the generator, then unpause it | All runs use `WORKERS=16`. Pause the DAG with `airflow dags pause diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py index 18e42690..c48afc60 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/airflow/dags/change_stream_to_parquet.py @@ -45,7 +45,7 @@ "LAKE_PREFIX": "orders", "STREAM_ID": "orders", "WINDOW_MINUTES": str(WINDOW_MINUTES), - "CHUNK_BYTES": os.getenv("CS_CHUNK_BYTES", str(50_000_000)), + "CHUNK_BYTES": os.getenv("CS_CHUNK_BYTES", str(80_000_000)), "MAX_EVENTS": "{{ dag_run.conf.get('max_events', 0) }}", "FAULT_EXIT_AFTER_UPLOAD": "{{ dag_run.conf.get('fault_after_chunks', 0) if ti.try_number == 1 else 0 }}", diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py index 67886860..df7d0e7c 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -53,6 +53,9 @@ ("read_at", pa.timestamp("us", tz="UTC")), ("full_document", pa.string()), ]) +# Rows are moved into Arrow every BATCH_ROWS rows. A large chunk then holds +# the column data once instead of as Python objects plus an Arrow copy. +BATCH_ROWS = 10_000 class Checkpoint: @@ -185,13 +188,15 @@ class Chunk: def __init__(self, start: dict): self.start = start - self.rows: list = [] + self.batches: list = [] + self.pending: list = [] + self.count = 0 self.size = 0 self.window = None self.end_token: Optional[dict] = None def fits(self, size: int, window, max_bytes: int) -> bool: - if not self.rows: + if not self.count: return True # None (no wallTime) is its own window, kept apart from dated ones. if window != self.window: @@ -199,33 +204,44 @@ def fits(self, size: int, window, max_bytes: int) -> bool: return self.size + size <= max_bytes def add(self, row: dict, size: int, window, token: dict) -> None: - if not self.rows: + if not self.count: self.window = window - self.rows.append(row) + self.pending.append(row) + self.count += 1 self.size += size self.end_token = token + if len(self.pending) >= BATCH_ROWS: + self.batches.append(pa.RecordBatch.from_pylist(self.pending, schema=SCHEMA)) + self.pending = [] + def table(self) -> pa.Table: + if self.pending: + self.batches.append(pa.RecordBatch.from_pylist(self.pending, schema=SCHEMA)) + self.pending = [] + return pa.Table.from_batches(self.batches, schema=SCHEMA) -def write_chunk(fs, path: str, rows: list) -> int: - table = pa.Table.from_pylist(rows, schema=SCHEMA) + +def write_chunk(fs, path: str, table: pa.Table) -> int: buffer = io.BytesIO() pq.write_table(table, buffer, compression="snappy") - data = buffer.getvalue() - fs.get_file_client(path).upload_data(data, overwrite=True) - return len(data) + size = buffer.tell() + # Upload from the buffer itself instead of a bytes copy of the file. + buffer.seek(0) + fs.get_file_client(path).upload_data(buffer, length=size, overwrite=True) + return size def publish(fs, lock: StreamLock, checkpoint: Checkpoint, prefix: str, chunk: Chunk, fault: bool) -> int: """Upload the chunk, then move the checkpoint past its last event.""" path = chunk_path(prefix, chunk.start, chunk.window) lock.check() - size = write_chunk(fs, path, chunk.rows) + size = write_chunk(fs, path, chunk.table()) if fault: log_json(logger, "fault_exit", path=path) os._exit(137) lock.check() - checkpoint.save(chunk.end_token, len(chunk.rows)) - log_json(logger, "chunk", path=path, rows=len(chunk.rows), row_bytes=chunk.size, bytes=size) + checkpoint.save(chunk.end_token, chunk.count) + log_json(logger, "chunk", path=path, rows=chunk.count, row_bytes=chunk.size, bytes=size) return size @@ -235,7 +251,7 @@ def main() -> None: window_minutes = int(os.getenv("WINDOW_MINUTES", "10")) if window_minutes <= 0 or 60 % window_minutes: raise SystemExit("WINDOW_MINUTES must divide 60") - chunk_bytes = int(os.getenv("CHUNK_BYTES", str(50_000_000))) + chunk_bytes = int(os.getenv("CHUNK_BYTES", str(80_000_000))) max_events = int(os.getenv("MAX_EVENTS", "0")) # 0 = until caught up max_seconds = float(os.getenv("MAX_SECONDS", "0")) # Test-only: exit after uploading this many chunks, before the checkpoint. @@ -297,17 +313,17 @@ def main() -> None: chunks += 1 written_bytes += publish(fs, lock, checkpoint, prefix, chunk, fault_after_chunks and chunks >= fault_after_chunks) - total += len(chunk.rows) + total += chunk.count chunk = Chunk({"token": chunk.end_token}) chunk.add(row, size, window, change["_id"]) - limit = (max_events and total + len(chunk.rows) >= max_events) or \ + limit = (max_events and total + chunk.count >= max_events) or \ (max_seconds and time.monotonic() - started >= max_seconds) if caught_up or limit: - if chunk.rows: + if chunk.count: chunks += 1 written_bytes += publish(fs, lock, checkpoint, prefix, chunk, fault_after_chunks and chunks >= fault_after_chunks) - total += len(chunk.rows) + total += chunk.count break # Nothing to read: move the checkpoint to the post-batch resume token # so the next run does not scan the same idle range again. Skip it diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py index ad4fb9e6..8eb46517 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py @@ -31,7 +31,7 @@ def main() -> None: fs = service.get_file_system_client(env("LAKE_FILESYSTEM", "cdc")) prefix = env("LAKE_PREFIX", "orders") window_minutes = int(os.getenv("WINDOW_MINUTES", "10")) - chunk_bytes = int(os.getenv("CHUNK_BYTES", str(50_000_000))) + chunk_bytes = int(os.getenv("CHUNK_BYTES", str(80_000_000))) columns = ["resume_token", "op", "doc_id", "wall_time", "read_at"] rows = [] From 9f19bddf16b0f8a508e01f84ae00df15f7caa2d2 Mon Sep 17 00:00:00 2001 From: hellices Date: Sat, 3 Oct 2026 23:54:50 +0900 Subject: [PATCH 11/14] =?UTF-8?q?fix(azure-documentdb):=20lease=EB=A5=BC?= =?UTF-8?q?=20=EC=9E=83=EC=9D=80=20=EC=8B=A4=ED=96=89=EC=9D=98=20=EB=8A=A6?= =?UTF-8?q?=EC=9D=80=20=EC=B2=AD=ED=81=AC=20=EC=97=85=EB=A1=9C=EB=93=9C=20?= =?UTF-8?q?=EC=B0=A8=EB=8B=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 청크 경로를 checkpoint에 먼저 기록하고 청크 파일 lease를 잡은 뒤 lease ID로 업로드 - stale_writer_test.py 추가, Azure에서 이전/현재 sample 결과를 문서에 반영 - 메모리 측정 파드와 Airflow DAG 파드의 리소스 조건을 구분 - README 사전 요구 사항에 Helm 3.19.0 이상 명시 Co-Authored-By: Claude Opus 5.5 --- .../azure-documentdb/change-streams/index.md | 25 ++++- .../change-streams/measurements/index.md | 50 +++++++++- .../samples/python-aks/README.md | 44 +++++++-- .../samples/python-aks/app/lake_export.py | 75 ++++++++++++--- .../python-aks/app/stale_writer_test.py | 94 +++++++++++++++++++ 5 files changed, 256 insertions(+), 32 deletions(-) create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index 2c5f014b..c2ef5141 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -16,6 +16,8 @@ official_sources: url: https://github.com/AzureCosmosDB/changestream-driver-compatibility - title: Lease Blob url: https://learn.microsoft.com/rest/api/storageservices/lease-blob + - title: Path - Update + url: https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update - title: Best practices for using Azure Data Lake Storage url: https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices --- @@ -64,7 +66,9 @@ Airflow가 매시 2분, 12분, 22분처럼 10분 구간이 닫히고 2분 뒤에 3. 구간마다 Parquet 청크로 올립니다. 한 구간이 압축 전 80 MB(80,000,000바이트)를 넘으면 그 앞에서 잘라 다음 파일로 넘깁니다. 청크 파일 이름은 구간 시작 시각과 청크 첫 이벤트 바로 앞의 resume token으로 만듭니다. -4. 청크를 올린 뒤 checkpoint를 ETag 조건으로 갱신합니다. +4. 올릴 청크 파일 경로를 checkpoint에 먼저 기록합니다. 그 다음 청크 파일에 + lease를 잡고 checkpoint가 자기 기록 그대로인지 확인한 뒤 그 lease ID로 + 올립니다. 마지막으로 checkpoint를 ETag 조건으로 갱신합니다. DAG는 `max_active_runs=1`이고 실패하면 세 번까지 재시도합니다. 재시도는 같은 checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁니다. 첫 실행의 @@ -79,12 +83,13 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 | 확인 항목 | 결과 | | --- | --- | | 10분 구간 지연 | 초당 약 960건 쓰기에서 이벤트 기록부터 파일 저장까지 p50 454초, p99 722초, 최대 731초 | +| lock lease를 잃은 실행 | 멈춰 있던 실행이 다음 실행 뒤에 같은 파일을 쓰려 하면 업로드가 412로 거부되거나 checkpoint 확인에서 멈춤. 중복 0, 누락 0. 이전 sample은 45,000행 중복 | | 업로드 직후 장애와 재시도 | 첫 청크를 올리고 checkpoint를 쓰기 전에 파드를 죽여도 재시도 후 중복 0, 누락 0. 50 MB와 80 MB 한도에서 각각 확인 | | 구간과 크기로 자르기 | 50, 80, 100, 200 MB 한도로 쓴 파일이 모두 구간 하나의 이벤트만 담았고 한도를 넘은 파일은 없음 | | 400 MB 활성 change log를 넘긴 재개 | DAG를 31분 멈춘 사이 쌓인 문서 본문 2.6 GB 이상의 백로그를 누락 없이 따라잡음. `maxAwaitTimeMS`를 지정하지 않았을 때만 성공. 이전 구성에서 측정 | | 처리 속도 | 1 KB 문서 10분 구간 하나(약 600,000건)를 32–38초에 씀. 한도를 50–200 MB로 바꿔도 시간은 비슷함 | | 파일 크기 | 압축 전 80 MB에서 자른 파일이 1 KB 랜덤 채움 문서는 43.7–46.7 MB, 채움 없는 작은 문서는 13.3–13.4 MB. 파일 크기는 한도에 비례함 | -| 내보내기 파드 메모리 | 80 MB 한도에서 최대 RSS 336 MiB. 200 MB 한도는 620 MiB. 메모리 요청 512Mi, 제한 1Gi로 실행 | +| 내보내기 파드 메모리 | 80 MB 한도에서 최대 RSS 336 MiB. 200 MB 한도는 620 MiB. 측정 파드는 요청 512Mi, 제한 4Gi. Airflow DAG의 파드는 요청 512Mi, 제한 1Gi | 지켜야 할 조건은 다음과 같습니다. @@ -99,8 +104,11 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 - checkpoint가 없는 첫 실행은 시작 위치를 먼저 저장합니다. 저장하지 않으면 첫 청크를 쓰다 실패했을 때 재시도가 더 뒤에서 시작해 그 사이 이벤트를 잃습니다. - 실행이 겹치지 않게 `max_active_runs=1`을 두고 실행 동안 lock 파일 lease를 - 잡습니다. checkpoint ETag 조건은 checkpoint만 보호합니다. 이미 올린 파일을 다른 - 실행이 짧은 청크로 덮어쓰는 것은 막지 못합니다. + 잡습니다. checkpoint ETag 조건은 checkpoint만 보호합니다. lock lease를 잃은 실행이 + 늦게 업로드하면 다음 실행이 쓴 더 짧은 청크를 덮어씁니다. 그래서 올릴 파일 + 경로를 checkpoint에 먼저 기록하고 청크 파일에도 lease를 잡아 그 lease ID로 + 올립니다. 같은 파일을 쓰는 다음 실행은 자기 기록을 남긴 뒤 lease를 끊습니다. + 이전 실행은 기록 확인에서 멈추거나 업로드가 412로 거부됩니다. - 감시할 컬렉션을 먼저 만듭니다. 없으면 `watch()`가 code 26으로 실패합니다. ## 공식 예제가 놓친 부분 @@ -200,16 +208,23 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 바꾸고 한도를 50, 80, 100, 200 MB로 바꿔 파일 크기와 메모리를 쟀습니다. 기본값을 80 MB로 올린 뒤 Airflow 실행과 업로드 직후 장애 재시도를 다시 확인했습니다. 표의 자르기, 파일 크기, 메모리, 재시도 값이 이 측정입니다. +- **lock lease를 잃은 실행:** 실행 하나를 첫 업로드 전에 멈추고 lock lease를 잃게 한 + 뒤 다음 실행이 같은 파일을 쓰게 했습니다. 이전 sample은 늦은 업로드가 다음 실행의 + 청크를 덮어써 후속 실행 뒤 45,000행이 중복됐습니다. 현재 sample은 업로드 직전과 lease + 직전 두 위치에서 모두 늦은 실행이 실패했고 중복과 누락은 0이었습니다. - **로컬 fake로만 확인한 경로:** `wallTime`이 없는 이벤트의 파티션 고정, 검증 스크립트의 순서 역전 실패 처리, 재시도가 첫 시도와 다른 위치에서 자르는 경우입니다. 이 클러스터의 이벤트에는 항상 `wallTime`이 있었고 순서 역전도 생기지 않았습니다. - Azure에서 재시도는 첫 시도와 같은 한도로 실행해 같은 위치에서 잘랐습니다. + Azure에서 재시도는 첫 시도와 같은 한도로 실행해 같은 위치에서 잘랐습니다. 다음 + 실행이 업로드를 마치고 checkpoint를 갱신하기 전에 이전 실행이 파일 lease를 끊는 + 순서도 fake로만 재현했습니다. ## 공식 출처 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) - [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) +- [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) - [Best practices for using Azure Data Lake Storage](https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices) ## 더 읽을 문서 diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 4308f30f..85b2b4c3 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -1,18 +1,22 @@ --- title: Azure DocumentDB change stream Parquet 적재 측정 상세 -description: Airflow DAG의 10분 구간 지연, 크기 한도별 파일 크기와 메모리, 이전 5분 주기 구성의 지연, 업로드 직후 장애 재시도, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. +description: Airflow DAG의 10분 구간 지연, 크기 한도별 파일 크기와 메모리, 이전 5분 주기 구성의 지연, 업로드 직후 장애 재시도, lock lease를 잃은 실행의 늦은 업로드, 400 MB 활성 change log를 넘긴 재개, Parquet 파일 크기와 상시 PyMongo consumer 기준선, change stream 옵션별 동작을 측정한 수치입니다. document_type: research services: [azure-documentdb, azure-storage, azure-kubernetes-service] technologies: [python, mongodb, airflow, kubernetes] tags: [evaluate, storage] status: current verification_status: verified -sources_checked_at: 2026-10-02 +sources_checked_at: 2026-10-03 published_at: 2026-10-02 topic_order: 1 official_sources: - title: Change streams in Azure DocumentDB url: https://learn.microsoft.com/azure/documentdb/change-streams + - title: Lease Blob + url: https://learn.microsoft.com/rest/api/storageservices/lease-blob + - title: Path - Update + url: https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update --- # Azure DocumentDB change stream Parquet 적재 측정 상세 @@ -102,8 +106,10 @@ checkpoint에서 230,000건을 모두 읽어 6개 파일로 썼습니다. 측정부터 내보내기는 행을 10,000건마다 Arrow 배치로 옮겨 청크 전체를 Python 객체와 Arrow 사본으로 함께 들고 있지 않습니다. 같은 50 MB 한도에서 이전 sample의 최대 RSS는 1 KB 채움 문서가 358 MiB, 채움 없는 문서가 386 MiB였습니다. -- 내보내기 파드는 메모리 요청 512Mi, 제한 1Gi로 실행합니다. 200 MB 한도의 620 MiB는 - 요청을 넘습니다. 한도를 올리면 파드 메모리도 함께 늘립니다. +- 이 표의 메모리는 `k8s/lake.yaml`로 띄운 파드에서 쟀습니다. 이 파드는 메모리 요청 + 512Mi, 제한 4Gi입니다. Airflow DAG가 띄우는 내보내기 파드는 요청 512Mi, 제한 + 1Gi입니다. 200 MB 한도의 620 MiB는 그 요청을 넘으니 한도를 올리면 DAG의 파드 + 메모리도 함께 늘립니다. ### 80 MB 한도로 다시 실행 @@ -131,6 +137,39 @@ checkpoint에서 230,000건을 모두 읽어 6개 파일로 썼습니다. | --- | --- | --- | --- | --- | | 690,000 | 690,000 | 0 | 0 | 11 | +## lock lease를 잃은 실행의 늦은 업로드 + +2026-10-03에 `stale_writer_test.py`로 확인했습니다. 실행 A는 첫 청크를 올리기 전에 +멈추고 stream lock lease 갱신을 끊습니다. lease가 만료되면 실행 B가 같은 checkpoint에서 +시작해 1,000건짜리 첫 청크를 같은 파일에 쓰고 checkpoint를 옮깁니다. 그 뒤 A가 이어서 +진행합니다. 파일이 B의 청크로 남아 마지막 resume token이 checkpoint와 같으면 +통과입니다. + +데이터는 1 KB 채움 문서 20,000개의 이벤트 46,000건이고 10분 구간 하나에 썼습니다. +경우마다 prefix와 stream ID를 따로 두었습니다. 시험 뒤 같은 prefix로 내보내기를 한 번 더 +실행해 나머지 이벤트를 쓰고 `verify_lake.py`로 대조했습니다. + +| sample | A가 멈춘 위치 | A의 결과 | 파일 행 수 | 후속 실행 뒤 중복 | 누락 | +| --- | --- | --- | --- | --- | --- | +| 이전(청크 파일 lease 없음) | 업로드 직전 | B의 청크를 46,000행으로 덮어쓰고 checkpoint 저장에서 `ConditionNotMet` | 46,000 | 45,000 | 0 | +| 현재 | 파일 lease를 잡은 뒤 업로드 직전 | B가 lease를 끊어 업로드가 412 `LeaseNotPresent`로 거부됨 | 1,000 | 0 | 0 | +| 현재 | checkpoint에 파일을 기록한 뒤 lease 전 | checkpoint 확인에서 멈춤 | 1,000 | 0 | 0 | + +- 이전 sample에서는 A의 업로드가 성공했습니다. checkpoint는 B의 1,000건 뒤를 가리키는데 + 파일에는 46,000건이 있어 후속 실행이 쓴 45,000건이 그대로 중복됐습니다. +- 같은 데이터로 업로드 직후 장애도 다시 실행했습니다. 첫 시도는 종료 코드 137로 끝났고 + 후속 실행이 같은 파일을 덮어써 46,000건이 중복과 누락 없이 들어갔습니다. 첫 시도의 + 파일 lease는 후속 실행 전에 만료됐습니다. +- 별도 시험 작업에서 lease가 걸린 파일에 lease ID 없이 올리면 `LeaseIdMissing`, 끊긴 + lease ID로 올리면 `LeaseIdMismatch`가 나왔고 모두 412였습니다. +- `create_file`에 lease ID를 함께 넘겨 만든 파일은 그 lease를 Blob 엔드포인트로 해제할 + 때 `LeaseIdMismatchWithLeaseOperation`이 났습니다. 만든 직후에도 같았습니다. 그래서 + 빈 파일을 먼저 만들고 lease를 따로 잡습니다. +- 중간 구현은 파일 lease만 잡고 checkpoint가 그대로인지 확인했습니다. 이 방식은 B가 + 업로드를 마치고 checkpoint를 갱신하기 전에 A가 lease를 끊으면 확인을 통과합니다. 파일 + 경로를 checkpoint에 먼저 기록하는 단계를 더해 막았고 이 순서는 로컬 fake로만 + 재현했습니다. + ## 5분 주기 지연(이전 구성) 이 절부터 파일 크기 절까지는 이전 구성의 측정입니다. 이전 구성은 5분마다 실행 시작 @@ -310,7 +349,10 @@ upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 - 압축 비교는 백로그 재개 시험의 파일 하나를 다시 써서 얻었습니다. zstd로 쓸 때의 시간과 CPU는 측정하지 않았습니다. - 파일 저장 지연은 초 단위 last-modified로 계산했습니다. +- lock lease를 잃은 실행의 시험은 경우마다 한 번씩 실행했습니다. ## 공식 출처 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) +- [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) +- [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index c99f25aa..330ab49e 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -18,6 +18,7 @@ through a private endpoint. | `app/verify.py` | Compares the generator's expected events with a sink collection | | `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes the closed 10-minute windows as Parquet chunks of at most 80 MB of row data and exits | | `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | +| `app/stale_writer_test.py` | Test only. Stops one export run before its first upload, lets a second run write the same file and checks that the first run cannot overwrite it | | `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | | `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job, RBAC for the Airflow scheduler | @@ -28,7 +29,8 @@ inspected. Run every command from this directory. ## Prerequisites - Azure Developer CLI 1.29 or later and Azure CLI -- `kubectl`, `helm` and `envsubst` (gettext) +- `kubectl`, Helm 3.19.0 or later (Airflow chart 1.22.0 requires it) and + `envsubst` (gettext) - Permission to create a resource group and role assignments in the subscription @@ -215,10 +217,11 @@ Export behavior: time: it waits for its window to close and for the offset. The test measured p50 454 seconds and a maximum of 731 seconds. - Rows are moved into Arrow record batches every 10,000 rows, so a chunk is - not held as Python objects and an Arrow copy at the same time. The export - pod requests 512Mi of memory and is limited to 1Gi. Peak RSS was 336 MiB - with 80 MB chunks of 1 KB padded documents, 390 MiB at 100 MB and 620 MiB - at 200 MB. Raise the pod memory with the limit. + not held as Python objects and an Arrow copy at the same time. Peak RSS was + 336 MiB with 80 MB chunks of 1 KB padded documents, 390 MiB at 100 MB and + 620 MiB at 200 MB. These runs used `k8s/lake.yaml` pods, which request + 512Mi and are limited to 4Gi. The pod the DAG starts requests 512Mi and is + limited to 1Gi. Raise its memory in the DAG when you raise the limit. - A run with no checkpoint saves the current time as the start position before it reads. Its retry starts at the same time. - Each chunk is written to @@ -230,10 +233,28 @@ Export behavior: written with an ETag condition after each chunk upload. - A run holds a 20-second lease on `_checkpoints/.lock` and renews it in the background. A second run waits up to 90 seconds for the lease and - then fails, so two runs never overwrite each other's chunks. The ETag - condition alone only protects the checkpoint, not files already uploaded. - A lease that expires instead of being released can take up to a minute to - become available again. + then fails, so two runs do not write chunks at the same time. A lease that + expires instead of being released can take up to a minute to become + available again. +- A run can lose the stream lease while it serializes or uploads a chunk. The + next run then starts from the same checkpoint and may write a shorter chunk + to the same file. The ETag condition only protects the checkpoint, so a + chunk is written in three steps: + 1. The run saves the checkpoint again at the same position with the chunk + path in `writing`, under the ETag condition. This fails if another run + moved the checkpoint or recorded its own chunk. + 2. It creates the chunk file empty if it is missing and acquires a + 60-second lease on it. A lease left by an older run is broken. It then + checks that the checkpoint ETag is still the one it saved. + 3. It uploads with the lease ID and saves the checkpoint. + + A later run that writes the same file saves its own record before it breaks + the lease. So an older run either fails the check in step 2 or gets 412 from + storage on the upload instead of overwriting the file. One chunk upload has + to finish within the 60 seconds. A lease proposed when the file is created + could not be released through the Blob endpoint in the test + (`LeaseIdMismatchWithLeaseOperation`), so the run creates the file first + and leases it after. - Trigger the DAG with `{"fault_after_chunks": 1}` to make the first try exit after one upload and before the checkpoint. The retry finishes the run. - `max_await_time_ms` is not set unless `MAX_AWAIT_MS` is non-zero. The test @@ -273,7 +294,9 @@ come from those runs. The next day the export started to move rows into Arrow batches and the default limit became 80 MB. The chunk limit comparison, the Airflow 80 MB backlog and the Airflow retry were run on the same deployment with that sample. The published file size, memory and 80 MB retry results come -from them. The backlog row was not repeated. +from them. The backlog row was not repeated. The stale writer row was run +with the current sample and, for comparison, with the sample before the chunk +file lease. | Run | Generator parameters | Extra steps | | --- | --- | --- | @@ -287,6 +310,7 @@ from them. The backlog row was not repeated. | Chunk limit comparison | `DOCS=200000 RATE=0 PAD_BYTES=1000`, then `DOCS=450000 RATE=0 PAD_BYTES=0` in the next window | Pause the DAG. Before the generator, run `lake_export.py` once per limit with its own `LAKE_PREFIX`, which is also the stream ID, so each saves its own start position. After each window closes, run the exports again with `CHUNK_BYTES` added to the Job env (50000000, 100000000 or 200000000) and wrapped as in Export memory. Verify each prefix with `verify_lake.py`. The 80 MB memory was measured the same way on the third run of the next row | | Airflow 80 MB backlog | The two runs above, then `DOCS=300000 RATE=0 PAD_BYTES=1000` in another window | Keep the DAG paused while the three runs write, then unpause it. One scheduled run reads the three windows | | Airflow backlog | `DOCS=600000 RATE=0 PAD_BYTES=4000` | Pause the DAG, run the generator, then unpause it | +| Stale writer | `DOCS=20000 RATE=0 PAD_BYTES=1000` | Pause the DAG. Before the generator, run `lake_export.py` once per case with its own `LAKE_PREFIX`. After the window closes, run `stale_writer_test.py` with `k8s/lake.yaml` (`SCRIPT=stale_writer_test.py`), once as is and once with `STALL_AT=lease` added to the Job env. It exits with code 1 when the older run overwrote the newer chunk. Then run `lake_export.py` again on each prefix and verify it with `verify_lake.py` | All runs use `WORKERS=16`. Pause the DAG with `airflow dags pause change_stream_to_parquet` in the scheduler container. diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py index df7d0e7c..85a00fd4 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -17,7 +17,13 @@ start time before reading, so its retry starts at the same place. A run holds a lease on a lock file next to the checkpoint, so two runs never -write chunks for the same stream at the same time. +write chunks for the same stream at the same time. A run can still lose that +lease while it writes, so a chunk is written in three steps. The run first +records the chunk file in the checkpoint with the ETag condition, then leases +the file and checks that the checkpoint still holds its own record, and then +uploads with the lease ID. A later run that writes the same file records it +and breaks the lease, so the older run fails the check, or Blob Storage +rejects its upload. """ import hashlib @@ -34,7 +40,7 @@ from azure.core import MatchConditions from azure.core.exceptions import HttpResponseError, ResourceNotFoundError from azure.identity import DefaultAzureCredential -from azure.storage.filedatalake import DataLakeServiceClient +from azure.storage.filedatalake import DataLakeLeaseClient, DataLakeServiceClient from bson import json_util from bson.timestamp import Timestamp @@ -76,15 +82,23 @@ def load(self) -> Optional[dict]: self.etag = download.properties.etag return json_util.loads(download.readall()) - def save(self, token: Optional[dict], events: int, start_at: Optional[Timestamp] = None) -> None: + def save(self, token: Optional[dict], events: int, start_at: Optional[Timestamp] = None, + writing: Optional[str] = None) -> None: body = json_util.dumps({"token": token, "start_at": start_at, "updated_at": utcnow(), - "events": events, "pod": socket.gethostname()}) + "events": events, "writing": writing, "pod": socket.gethostname()}) # IfNotModified fails if another run moved the checkpoint since we read it. condition = (dict(etag=self.etag, match_condition=MatchConditions.IfNotModified) if self.etag else dict(match_condition=MatchConditions.IfMissing)) result = self.file.upload_data(body, overwrite=True, **condition) self.etag = result["etag"] + def claim(self, start: dict, path: str) -> None: + """Record the chunk file about to be written, at the same position.""" + self.save(start.get("token"), 0, start_at=start.get("start_at"), writing=path) + + def unchanged(self) -> bool: + return self.file.get_file_properties().etag == self.etag + class StreamLock: """Lease on ``_checkpoints/.lock`` held for the whole run. @@ -221,26 +235,61 @@ def table(self) -> pa.Table: return pa.Table.from_batches(self.batches, schema=SCHEMA) -def write_chunk(fs, path: str, table: pa.Table) -> int: - buffer = io.BytesIO() - pq.write_table(table, buffer, compression="snappy") - size = buffer.tell() - # Upload from the buffer itself instead of a bytes copy of the file. - buffer.seek(0) - fs.get_file_client(path).upload_data(buffer, length=size, overwrite=True) - return size +def lease_file(file, duration: int = 60) -> DataLakeLeaseClient: + """Lease the chunk file, creating it empty if it does not exist. + + A lease already on the file belongs to a run that crashed or lost the + stream lock. Breaking it makes that run's upload fail. + """ + # A lease proposed on create could not be released through the Blob + # endpoint in the test, so create the file first and lease it after. + try: + file.create_file(match_condition=MatchConditions.IfMissing) + except HttpResponseError as error: + if error.status_code != 409: + raise + lease = DataLakeLeaseClient(file) + try: + lease.acquire(lease_duration=duration) + except HttpResponseError as error: + if error.status_code != 409: + raise + lease.break_lease(lease_break_period=0) + log_json(logger, "file_lease_broken", path=file.path_name) + lease.acquire(lease_duration=duration) + return lease def publish(fs, lock: StreamLock, checkpoint: Checkpoint, prefix: str, chunk: Chunk, fault: bool) -> int: """Upload the chunk, then move the checkpoint past its last event.""" path = chunk_path(prefix, chunk.start, chunk.window) + buffer = io.BytesIO() + pq.write_table(chunk.table(), buffer, compression="snappy") + size = buffer.tell() lock.check() - size = write_chunk(fs, path, chunk.table()) + # Fails if another run moved the checkpoint or recorded its own chunk. + checkpoint.claim(chunk.start, path) + file = fs.get_file_client(path) + lease = lease_file(file) + # Another run that writes this file records it before it breaks the lease. + # If our record is still there, any later writer has to break this lease + # first, and the upload below then fails. + if not checkpoint.unchanged(): + raise RuntimeError("another run recorded a chunk after this run") + # Upload from the buffer itself instead of a bytes copy of the file. + buffer.seek(0) + file.upload_data(buffer, length=size, overwrite=True, lease=lease) if fault: log_json(logger, "fault_exit", path=path) os._exit(137) lock.check() checkpoint.save(chunk.end_token, chunk.count) + try: + lease.release() + except HttpResponseError: + # A run that lost the stream lock broke the lease after the upload. + # The checkpoint is saved, so that run fails its check. + log_json(logger, "file_lease_release_failed", path=path) log_json(logger, "chunk", path=path, rows=chunk.count, row_bytes=chunk.size, bytes=size) return size diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py new file mode 100644 index 00000000..569a3844 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py @@ -0,0 +1,94 @@ +"""Check that a run that lost the stream lease cannot overwrite a newer chunk. + +Run A (this process) stops before its first Parquet upload, stops renewing +the stream lease and starts run B as a subprocess. B waits for the lease, +writes a shorter first chunk (``B_MAX_EVENTS``) to the same file and moves the +checkpoint. Then A goes on. The check passes when the file still holds B's +chunk, so its last resume token is the one in the checkpoint. + +``STALL_AT=upload`` (default) stops A after it leased the file, right before +the upload. ``STALL_AT=lease`` stops A before it leases the file. + +Needs unread events in a closed window, more than ``B_MAX_EVENTS``. Runs with +the same environment as ``lake_export.py``. +""" + +import io +import json +import os +import subprocess +import sys + +import pyarrow.parquet as pq +from azure.storage.filedatalake import DataLakeFileClient + +import lake_export + +locks = [] +held: dict = {} +acquire = lake_export.StreamLock.acquire +upload_data = DataLakeFileClient.upload_data + + +def acquire_and_keep(self): + acquire(self) + locks.append(self) + + +lease_file = getattr(lake_export, "lease_file", None) + + +def run_b(path: str) -> None: + held["path"] = path + # The stream lease expires about 20 seconds after the last renewal. + locks[0].stopped.set() + env = dict(os.environ, MAX_EVENTS=os.getenv("B_MAX_EVENTS", "1000")) + b = subprocess.run([sys.executable, "lake_export.py"], env=env, capture_output=True, text=True) + held["b_exit"] = b.returncode + held["b_log"] = [line for line in b.stdout.splitlines() + b.stderr.splitlines() + if '"event"' in line or "Error" in line][-8:] + + +def upload_after_b(self, data, *args, **kwargs): + if self.path_name.endswith(".parquet") and not held: + run_b(self.path_name) + return upload_data(self, data, *args, **kwargs) + + +def lease_after_b(file, *args, **kwargs): + if not held: + run_b(file.path_name) + return lease_file(file, *args, **kwargs) + + +def main() -> None: + lake_export.StreamLock.acquire = acquire_and_keep + if os.getenv("STALL_AT", "upload") == "lease": + lake_export.lease_file = lease_after_b + else: + DataLakeFileClient.upload_data = upload_after_b + try: + lake_export.main() + a_error = None + except Exception as error: # noqa: BLE001 - A is expected to fail + a_error = f"{type(error).__name__}: {str(error).splitlines()[0]}" + DataLakeFileClient.upload_data = upload_data + if "path" not in held: + raise SystemExit("run A wrote no chunk; write more events in a closed window first") + + service = lake_export.DataLakeServiceClient(lake_export.env("LAKE_URL"), + credential=lake_export.DefaultAzureCredential()) + fs = service.get_file_system_client(lake_export.env("LAKE_FILESYSTEM", "cdc")) + data = fs.get_file_client(held["path"]).download_file().readall() + tokens = pq.read_table(io.BytesIO(data), columns=["resume_token"]).column("resume_token").to_pylist() + checkpoint = lake_export.Checkpoint(fs, lake_export.env("STREAM_ID", "orders")).load() + ok = held["b_exit"] == 0 and tokens[-1] == checkpoint["token"]["_data"] + print(json.dumps({"result": "pass" if ok else "fail", "path": held["path"], "file_rows": len(tokens), + "file_ends_at_checkpoint": tokens[-1] == checkpoint["token"]["_data"], + "checkpoint_events": checkpoint["events"], "a_error": a_error, + "b_exit": held["b_exit"], "b_log": held["b_log"]}, default=str)) + sys.exit(0 if ok else 1) + + +if __name__ == "__main__": + main() From 811cb80f741e46198843a6105bbb97a34d5e6ba2 Mon Sep 17 00:00:00 2001 From: hellices Date: Thu, 8 Oct 2026 19:37:46 +0900 Subject: [PATCH 12/14] =?UTF-8?q?docs(azure-documentdb):=20change=20stream?= =?UTF-8?q?=20=EC=97=AD=ED=95=A0=C2=B7=EA=B3=BC=EA=B1=B0=20=EC=9E=AC?= =?UTF-8?q?=EA=B0=9C=20=EC=8B=A4=EC=B8=A1=20=EB=B3=B4=EA=B0=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit secondary user 역할별 startAtOperationTime 차이와 grant/revoke 통제 실험을 기록하고, 큰 백로그의 maxAwaitTimeMS timeout과 분리해 운영 절차를 정리한다. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../azure-documentdb/change-streams/index.md | 46 ++++++++++++- .../change-streams/measurements/index.md | 59 ++++++++++++++++- .../samples/python-aks/README.md | 66 +++++++++++++++++++ 3 files changed, 169 insertions(+), 2 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index c2ef5141..4f7c6f3d 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -7,11 +7,13 @@ technologies: [python, mongodb, airflow, kubernetes] tags: [evaluate, build, storage] status: current verification_status: verified -sources_checked_at: 2026-10-03 +sources_checked_at: 2026-10-08 published_at: 2026-10-02 official_sources: - title: Change streams in Azure DocumentDB url: https://learn.microsoft.com/azure/documentdb/change-streams + - title: Read and read/write privileges with secondary native users + url: https://learn.microsoft.com/azure/documentdb/secondary-users - title: AzureCosmosDB/changestream-driver-compatibility url: https://github.com/AzureCosmosDB/changestream-driver-compatibility - title: Lease Blob @@ -87,6 +89,7 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 | 업로드 직후 장애와 재시도 | 첫 청크를 올리고 checkpoint를 쓰기 전에 파드를 죽여도 재시도 후 중복 0, 누락 0. 50 MB와 80 MB 한도에서 각각 확인 | | 구간과 크기로 자르기 | 50, 80, 100, 200 MB 한도로 쓴 파일이 모두 구간 하나의 이벤트만 담았고 한도를 넘은 파일은 없음 | | 400 MB 활성 change log를 넘긴 재개 | DAG를 31분 멈춘 사이 쌓인 문서 본문 2.6 GB 이상의 백로그를 누락 없이 따라잡음. `maxAwaitTimeMS`를 지정하지 않았을 때만 성공. 이전 구성에서 측정 | +| secondary user의 과거 시점 시작 | `readWriteAnyDatabase + clusterAdmin`만 가진 사용자는 `startAtOperationTime`이 code 13. `readAnyDatabase`를 grant하면 같은 사용자가 성공했고 revoke하면 다시 실패 | | 처리 속도 | 1 KB 문서 10분 구간 하나(약 600,000건)를 32–38초에 씀. 한도를 50–200 MB로 바꿔도 시간은 비슷함 | | 파일 크기 | 압축 전 80 MB에서 자른 파일이 1 KB 랜덤 채움 문서는 43.7–46.7 MB, 채움 없는 작은 문서는 13.3–13.4 MB. 파일 크기는 한도에 비례함 | | 내보내기 파드 메모리 | 80 MB 한도에서 최대 RSS 336 MiB. 200 MB 한도는 620 MiB. 측정 파드는 요청 512Mi, 제한 4Gi. Airflow DAG의 파드는 요청 512Mi, 제한 1Gi | @@ -95,6 +98,9 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 - change stream을 열 때 짧은 `maxAwaitTimeMS`를 주지 않습니다. 1초로 두면 큰 백로그를 재개할 때 code 50으로 실패하고 재시도로도 넘어가지 못했습니다. +- secondary user로 `startAtOperationTime`을 쓸 때는 `readAnyDatabase` 역할을 + 확인합니다. `readWriteAnyDatabase + clusterAdmin`만 있으면 일반 `watch()`와 + `resumeAfter`는 동작해도 과거 시점 시작은 code 13으로 실패했습니다. - 파일을 먼저 쓰고 checkpoint를 나중에 씁니다. 순서가 반대면 그 사이 장애로 이벤트를 잃습니다. - 파일 이름을 청크 첫 이벤트 바로 앞의 resume token으로 정해 재시도가 같은 파일을 @@ -111,6 +117,43 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 이전 실행은 기록 확인에서 멈추거나 업로드가 412로 거부됩니다. - 감시할 컬렉션을 먼저 만듭니다. 없으면 `watch()`가 code 26으로 실패합니다. +## secondary user로 과거 시점부터 읽기 + +Learn의 secondary native user 문서는 읽기 전용 사용자에 `readAnyDatabase`, +읽기·쓰기 사용자에 `readWriteAnyDatabase + clusterAdmin`을 지정합니다. 2026-10-08 +M25, shard 1개, 서버 8.0 클러스터에서는 후자의 사용자가 일반 `watch()`와 +`resumeAfter`는 열었지만 `startAtOperationTime`만 code 13(`Unauthorized`)으로 +실패했습니다. 같은 timestamp와 이벤트를 사용한 raw aggregate와 PyMongo 시험에서 +built-in 관리자와 `readAnyDatabase` 사용자는 성공했습니다. + +기존 읽기·쓰기 사용자에는 built-in 관리자로 역할을 grant할 수 있었습니다. + +```javascript +use admin + +db.runCommand({ + grantRolesToUser: "cdc_user", + roles: [ + { role: "readAnyDatabase", db: "admin" } + ] +}) +``` + +세 역할을 `createUser`에 한꺼번에 주면 code 31, `updateUser`로 역할을 바꾸면 code 2로 +실패했습니다. 반면 위 grant는 성공했고 `usersInfo`에 세 역할이 저장됐습니다. +`connectionStatus`에는 추가 역할이 보이지 않았으므로 적용 여부는 아래 명령으로 +확인합니다. 기존 client pool도 닫고 새 연결에서 다시 시험합니다. + +```javascript +db.runCommand({ usersInfo: "cdc_user" }) +``` + +DocumentDB에서 읽기만 하고 checkpoint와 결과를 외부 저장소에 쓰는 CDC라면 +`readAnyDatabase` 전용 사용자가 더 단순합니다. 같은 계정이 DocumentDB 안에 +checkpoint나 sink도 써야 하면 읽기·쓰기 역할을 유지하고 `readAnyDatabase`를 +grant합니다. rollback은 `revokeRolesFromUser`에 같은 역할을 넘깁니다. 이 역할별 +차이는 Learn에 설명되지 않은 관찰 결과이므로 서버 업데이트 뒤 다시 확인합니다. + ## 공식 예제가 놓친 부분 Learn 문서의 시작 예제(Python, Java, C#, Ruby, Node.js)와 @@ -140,6 +183,7 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 | 감시 범위 | 컬렉션 예시만 있음 | `db.watch()`는 code 26. `client.watch()`에 `$match`로 `ns.db`를 거르면 동작 | | 컬렉션 drop, rename | 설명 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | | `startAtOperationTime` | 날짜로 `Timestamp`를 만드는 예시 | 날짜로 만든 값은 동작함. 세션의 `operationTime`은 비어 있어 쓸 수 없음 | +| secondary user 역할 | 읽기 전용과 읽기·쓰기 역할을 설명하지만 change stream 옵션별 차이는 없음 | `readAnyDatabase`는 과거 시점 시작 성공. `readWriteAnyDatabase + clusterAdmin`은 code 13이고 `readAnyDatabase` grant 뒤 성공 | | pre-image | 미리 보기. 지원 요청으로 켬 | 지원 요청 없이 `required`로 열면 code 10065 | 옵션별 관찰 전체는 [측정 상세](measurements/index.md)에 있습니다. diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 85b2b4c3..32ebd718 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -7,12 +7,14 @@ technologies: [python, mongodb, airflow, kubernetes] tags: [evaluate, storage] status: current verification_status: verified -sources_checked_at: 2026-10-03 +sources_checked_at: 2026-10-08 published_at: 2026-10-02 topic_order: 1 official_sources: - title: Change streams in Azure DocumentDB url: https://learn.microsoft.com/azure/documentdb/change-streams + - title: Read and read/write privileges with secondary native users + url: https://learn.microsoft.com/azure/documentdb/secondary-users - title: Lease Blob url: https://learn.microsoft.com/rest/api/storageservices/lease-blob - title: Path - Update @@ -319,6 +321,59 @@ upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 - 클러스터 CPU는 초당 약 1,000건에서 약 30%, 속도 제한 없이 초당 약 2,900건을 쓸 때 약 60%였습니다. +## secondary user 역할과 `startAtOperationTime` + +2026-10-08 M25, shard 1개, 서버 8.0 클러스터에서 역할별 권한을 분리했습니다. +동일 컬렉션에 서버 `operationTime` 직후 marker 문서를 쓰고 같은 BSON +`Timestamp`를 세 principal이 읽었습니다. 매 라운드마다 raw `$changeStream`, +raw 명령에 `updateLookup`과 실제 CDC `$match`를 더한 경우, PyMongo `watch()`를 +각각 실행했고 이 과정을 세 번 반복했습니다. + +| principal | raw 최소 옵션 | raw CDC 옵션 | PyMongo CDC 옵션 | +| --- | ---: | ---: | ---: | +| built-in 관리자(`root`) | 3/3 marker 확인 | 3/3 marker 확인 | 3/3 marker 확인 | +| `readAnyDatabase` | 3/3 marker 확인 | 3/3 marker 확인 | 3/3 marker 확인 | +| `readWriteAnyDatabase + clusterAdmin` | 3/3 code 13 | 3/3 code 13 | 3/3 code 13 | + +마지막 principal도 옵션 없는 `watch()`와 event token을 쓴 `resumeAfter`는 +성공했습니다. 따라서 연결, 컬렉션 read, pipeline, resume token이 아니라 +`startAtOperationTime` 권한 경로에서만 차이가 났습니다. + +같은 읽기·쓰기 사용자의 역할을 바꾸며 인과관계도 확인했습니다. + +| 순서 | `usersInfo` 역할 | 같은 timestamp의 결과 | +| --- | --- | --- | +| 생성 직후 | `readWriteAnyDatabase`, `clusterAdmin` | code 13 | +| `readAnyDatabase` grant | 위 두 역할 + `readAnyDatabase` | marker 확인 | +| `readAnyDatabase` revoke | 다시 두 역할 | code 13 | +| `readAnyDatabase` re-grant | 다시 세 역할 | marker 확인 | + +grant와 revoke 뒤에는 매번 새 MongoClient를 만들었습니다. `usersInfo`에는 +`readAnyDatabase`가 추가·제거됐지만 `connectionStatus.authenticatedUserRoles`에는 +grant 뒤에도 기존 두 역할만 보였습니다. `connectionStatus(showPrivileges)`의 +action 목록에도 전후 모두 `changeStream`과 `find`가 있었으므로 역할 적용 여부와 이 +옵션의 동작은 `usersInfo`와 실제 요청으로 확인해야 합니다. + +역할을 만드는 명령 경로도 raw `mongosh`로 세 번 반복했습니다. + +| 명령 | 결과 | 직후 `usersInfo` | +| --- | --- | --- | +| 세 역할을 한 번에 `createUser` | code 31(`RoleNotFound`) | 사용자 없음 | +| 두 읽기·쓰기 역할로 `createUser` | 성공 | 두 역할 | +| 세 역할로 `updateUser` | code 2(`BadValue`), role update 미지원 | 두 역할 유지 | +| `grantRolesToUser(readAnyDatabase)` | 성공 | 세 역할 | + +이 결과는 secondary user 문서가 설명하는 생성 시 역할 조합과 grant 이후 실제로 +저장할 수 있는 조합이 다름을 보여 줍니다. 읽기 전용 CDC는 `readAnyDatabase`만 +쓰고, DocumentDB 안에도 데이터를 써야 하는 CDC는 두 읽기·쓰기 역할로 사용자를 만든 +뒤 `readAnyDatabase`를 grant하는 방식으로 확인했습니다. 관리형 서비스의 내부 권한 +검사 원인은 확인하지 않았습니다. + +이 시험의 1분·1시간·7일 전 timestamp는 모두 역할을 grant한 뒤 열렸습니다. 35일보다 +오래된 실제 archived event를 읽은 시험은 아니며, 역할 문제를 해결한 뒤 큰 백로그에서 +짧은 `maxAwaitTimeMS`를 주면 별도로 code 50이 날 수 있습니다. 이는 위 +"활성 change log를 넘긴 재개(이전 구성)" 절과 같은 문제입니다. + ## change stream 동작 확인 `probe.py`로 옵션과 이벤트 필드를 확인했습니다. Learn 문서에 나온 동작과 이번 @@ -334,6 +389,7 @@ upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 | 파이프라인 단계 | `$addFields`, `$match`, `$project`, `$set`, `$unset` | 다섯 개 모두 동작. 목록에 없는 `$replaceRoot`, `$redact`도 오류 없이 동작 | | 감시 범위 | 컬렉션 예시 | `db.watch()`는 code 26. `client.watch()`에 `ns.db` 조건을 건 `$match`는 동작 | | 재개 | `resumeAfter`, `startAt`, `startAtOperationTime` 지원 | `resume_after`, `start_after` 동작. 세션의 `operationTime`이 비어 있어 이를 쓴 `start_at_operation_time`은 실패. 현재 시각에서 10분 뺀 `Timestamp`는 동작 | +| secondary user | 읽기 전용과 읽기·쓰기 역할을 설명 | `readAnyDatabase`는 `startAtOperationTime` 성공. 읽기·쓰기 두 역할만 있으면 code 13이고 해당 역할을 grant하면 성공 | | 잘못된 resume token | 언급 없음 | code 2(BadValue) | | `showExpandedEvents` | 지원하지 않음 | code 115(CommandNotSupported) | | 감시 중인 컬렉션 drop, rename | 언급 없음 | `invalidate` 이벤트 없이 code 26으로 커서 종료 | @@ -354,5 +410,6 @@ upsert한 뒤 checkpoint를 저장합니다. 생성기는 문서마다 insert와 ## 공식 출처 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) +- [Read and read/write privileges with secondary native users](https://learn.microsoft.com/azure/documentdb/secondary-users) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) - [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index 330ab49e..ba2e2f5e 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -100,6 +100,72 @@ kubectl -n cslab create secret generic docdb --from-literal=uri="$MONGO_URI" The connection string uses `mongodb+srv://`. The cluster host name resolves to the private endpoint address inside the VNet, so it works only from AKS. +### Use a secondary user for CDC + +The commands above use the built-in administrator to keep the lab short. For +a long-running deployment, create a secondary user with the least privilege +needed by each workload. Azure DocumentDB documents two secondary-user role +sets: + +- A read-only CDC that stores checkpoints outside DocumentDB can use + `readAnyDatabase`. +- A consumer that writes its checkpoint or sink back to DocumentDB needs + `readWriteAnyDatabase` and `clusterAdmin`. + +In the service behavior measured on 2026-10-08, a user with only the two +read-write roles could open a plain stream and use `resumeAfter`, but +`startAtOperationTime` returned code 13. Granting `readAnyDatabase` after +creation enabled the same request. Creating the user with all three roles at +once returned code 31, and changing roles through `updateUser` returned code +2. Create the read-write user first, then grant the additional read role. + +```javascript +use admin + +db.runCommand({ + createUser: "cdc_writer", + pwd: "", + roles: [ + { role: "readWriteAnyDatabase", db: "admin" }, + { role: "clusterAdmin", db: "admin" } + ] +}) + +db.runCommand({ + grantRolesToUser: "cdc_writer", + roles: [ + { role: "readAnyDatabase", db: "admin" } + ] +}) +``` + +Verify the stored roles with `usersInfo`, not `connectionStatus`. +`connectionStatus.authenticatedUserRoles` did not display the granted +`readAnyDatabase` role in this test even though a new connection could use +`startAtOperationTime`. + +```javascript +db.runCommand({ usersInfo: "cdc_writer" }) +``` + +Close existing MongoClient pools after the grant and reconnect. To roll back +the added role: + +```javascript +db.runCommand({ + revokeRolesFromUser: "cdc_writer", + roles: [ + { role: "readAnyDatabase", db: "admin" } + ] +}) +``` + +This role-specific behavior is an observed Azure DocumentDB compatibility +detail, not a behavior described in the +[secondary-user documentation](https://learn.microsoft.com/azure/documentdb/secondary-users). +Recheck it after service upgrades. Do not put real credentials in this +repository or in `azd` environment files. + ## 4. Run the consumer The watched collection must exist before the consumer starts. The cluster From a752e1aeff7d4828212a264f9cbcf2d679237053 Mon Sep 17 00:00:00 2001 From: hellices Date: Thu, 8 Oct 2026 22:05:57 +0900 Subject: [PATCH 13/14] =?UTF-8?q?fix(azure-documentdb):=20historical=20get?= =?UTF-8?q?More=20=EB=8C=80=EA=B8=B0=20=EC=A0=9C=ED=95=9C=20=EC=A0=9C?= =?UTF-8?q?=EA=B1=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 공식 compatibility verifier의 1초 maxAwaitTimeMS가 지원 확인 범위임을 명확히 하고, historical backlog에서 getMore code 50을 유발할 수 있어 consumer 기본값에서는 생략한다. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../azure-documentdb/change-streams/index.md | 12 +++++--- .../change-streams/measurements/index.md | 28 +++++++++++++++++++ .../samples/python-aks/README.md | 5 ++++ .../samples/python-aks/app/consumer.py | 7 ++++- 4 files changed, 47 insertions(+), 5 deletions(-) diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index 4f7c6f3d..8cf94fdb 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -16,6 +16,8 @@ official_sources: url: https://learn.microsoft.com/azure/documentdb/secondary-users - title: AzureCosmosDB/changestream-driver-compatibility url: https://github.com/AzureCosmosDB/changestream-driver-compatibility + - title: MongoDB Change Streams Specification + url: https://github.com/mongodb/specifications/blob/master/source/change-streams/change-streams.md - title: Lease Blob url: https://learn.microsoft.com/rest/api/storageservices/lease-blob - title: Path - Update @@ -96,8 +98,9 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 지켜야 할 조건은 다음과 같습니다. -- change stream을 열 때 짧은 `maxAwaitTimeMS`를 주지 않습니다. 1초로 두면 큰 - 백로그를 재개할 때 code 50으로 실패하고 재시도로도 넘어가지 못했습니다. +- historical change stream을 열 때 짧은 `maxAwaitTimeMS`를 주지 않습니다. 이 + 값은 stream 전체 제한이 아니라 후속 `getMore`의 `maxTimeMS`가 됩니다. 1초로 + 두면 큰 백로그를 재개할 때 code 50으로 실패하고 재시도로도 넘어가지 못했습니다. - secondary user로 `startAtOperationTime`을 쓸 때는 `readAnyDatabase` 역할을 확인합니다. `readWriteAnyDatabase + clusterAdmin`만 있으면 일반 `watch()`와 `resumeAfter`는 동작해도 과거 시점 시작은 code 13으로 실패했습니다. @@ -166,7 +169,7 @@ Learn 문서의 시작 예제(Python, Java, C#, Ruby, Node.js)와 | 저장소의 `mongo_utils.py`가 resume token을 이벤트마다 로컬 파일 `.resume_token.json`에 씀 | 실행마다 새 파드가 뜨면 파일이 없어 현재 위치부터 읽음. 그 사이 이벤트를 잃음 | checkpoint를 ADLS Gen2 파일로 두고 청크마다 ETag 조건으로 갱신 | | Python 예제가 대상 컬렉션에 `insert_one`을 한 뒤 token을 저장 | 두 동작 사이에서 프로세스가 죽으면 재시작 후 같은 이벤트가 한 번 더 들어감 | resume token으로 파일 이름을 정해 덮어씀. 상시 consumer는 token을 키로 upsert | | `for change in stream`처럼 끝없이 읽음 | 배치 실행이 끝나지 않음 | `try_next()`로 읽고 아직 열린 10분 구간의 이벤트를 만나면 쓰지 않고 종료 | -| C# 예제와 저장소의 지원 확인 스크립트가 대기 시간을 1초로 지정 | 큰 백로그 재개에서 code 50(`ExceededTimeLimit`)이 같은 위치에서 반복 | `max_await_time_ms`를 지정하지 않음 | +| Microsoft의 driver compatibility verifier가 대기 시간을 1초로 지정 | verifier는 cursor가 열리는지만 확인하고 바로 닫음. 이 값을 historical consumer에 옮기면 오래 걸리는 `getMore`가 code 50(`ExceededTimeLimit`)으로 끝남 | historical read에는 `max_await_time_ms`를 지정하지 않음 | | 오류가 나면 메시지를 출력하고 끝남 | Learn 제한 사항은 장애 조치 뒤 커서를 다시 열어야 한다고 설명함 | Airflow 재시도가 새 파드에서 checkpoint로 stream을 다시 엶 | | 감시할 컬렉션이 있다고 가정 | 컬렉션이 없으면 `watch()`가 code 26 | 컬렉션을 먼저 만듦 | @@ -175,7 +178,7 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 | 항목 | Learn 문서 | 관찰 결과 | | --- | --- | --- | -| `maxAwaitTimeMS` | 설명 없음. C# 예제는 1초 | 새 이벤트 대기 시간이 아니라 `getMore` 전체의 실행 제한으로 적용됨 | +| `maxAwaitTimeMS` | Learn의 historical 예제에는 없음. 별도 compatibility verifier는 1초 | MongoDB spec대로 `getMore.maxTimeMS`가 됨. DocumentDB가 history를 스캔 중이면 빈 batch 대신 code 50을 반환할 수 있음 | | 이벤트 필드 | `_id`, `operationType`, `fullDocument`, `ns`, `documentKey` | 같은 필드에 `wallTime`이 더 있고 `clusterTime`은 없음 | | replace | 예시 없음 | `operationType: update`로 오고 교체 후 문서 전체가 실림 | | update의 `fullDocument` | 변경 후 문서 전체를 보여 주는 예시 | `updateLookup` 없이도 포함됨 | @@ -267,6 +270,7 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) - [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) +- [MongoDB Change Streams Specification](https://github.com/mongodb/specifications/blob/master/source/change-streams/change-streams.md) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) - [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) - [Best practices for using Azure Data Lake Storage](https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices) diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 32ebd718..8beefb62 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -15,6 +15,10 @@ official_sources: url: https://learn.microsoft.com/azure/documentdb/change-streams - title: Read and read/write privileges with secondary native users url: https://learn.microsoft.com/azure/documentdb/secondary-users + - title: AzureCosmosDB/changestream-driver-compatibility + url: https://github.com/AzureCosmosDB/changestream-driver-compatibility + - title: MongoDB Change Streams Specification + url: https://github.com/mongodb/specifications/blob/master/source/change-streams/change-streams.md - title: Lease Blob url: https://learn.microsoft.com/rest/api/storageservices/lease-blob - title: Path - Update @@ -258,6 +262,28 @@ full error: {'ok': 0.0, 'code': 50, 'codeName': 'ExceededTimeLimit', ...} 바로 다음 주기 실행은 새 이벤트 0건으로 1.2초 만에 끝났습니다. +### 1초 verifier 설정을 consumer에 옮기면 안 되는 이유 + +Microsoft의 `changestream-driver-compatibility` 저장소는 Python, C#, Java verifier에 +1초 `maxAwaitTimeMS`를 둡니다. 이 코드는 change stream cursor가 열리는지만 확인하고 +이벤트를 계속 읽지 않습니다. MongoDB Change Streams Specification은 이 옵션이 initial +`aggregate`나 `$changeStream` stage가 아니라 후속 `getMore`의 `maxTimeMS`라고 +명시합니다. 따라서 verifier의 1초를 historical consumer의 전체 실행 제한으로 쓰면 +적용 대상이 달라집니다. + +별도 M25, shard 1개, 서버 7.0 클러스터에서 timestamp와 marker 사이에 4 KiB update +event 본문을 약 6.14 GB 만들고 wire command를 기록했습니다. initial `aggregate`는 +약 52.5초에 성공했지만 다음 `getMore(maxTimeMS=1000)`가 1.0초에 code 50으로 +끝났습니다. 같은 timestamp에서 이 옵션을 생략하자 aggregate 뒤 네 번의 `getMore`가 +각각 88–114초 걸렸고 466초에 marker를 반환했습니다. `pymongo.timeout(30분)`은 +aggregate에 약 180만 ms의 `maxTimeMS`로 전달됐지만 별도 +`max_await_time_ms=1000`을 덮어쓰지 못했습니다. + +따라서 이 sample은 historical read에 `max_await_time_ms` 기본값을 두지 않습니다. +전체 실행 상한은 Airflow `execution_timeout`, PyMongo CSOT, 애플리케이션 deadline으로 +관리합니다. 실시간 tail의 polling 주기를 조정하려고 값을 추가할 때도 저장된 +checkpoint가 밀린 경우를 별도로 처리해야 합니다. + ## 파일 크기(이전 구성) 백로그 재개 시험의 Parquet 합계 5.13 GB는 생성기가 쓴 문서 본문 2.6 GB의 약 두 배입니다. 가장 큰 @@ -411,5 +437,7 @@ action 목록에도 전후 모두 `changeStream`과 `find`가 있었으므로 - [Change streams in Azure DocumentDB](https://learn.microsoft.com/azure/documentdb/change-streams) - [Read and read/write privileges with secondary native users](https://learn.microsoft.com/azure/documentdb/secondary-users) +- [AzureCosmosDB/changestream-driver-compatibility](https://github.com/AzureCosmosDB/changestream-driver-compatibility) +- [MongoDB Change Streams Specification](https://github.com/mongodb/specifications/blob/master/source/change-streams/change-streams.md) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) - [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index ba2e2f5e..151ba6e5 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -197,6 +197,11 @@ Run `probe.py` the same way with `SCRIPT=probe.py`. - On its first start the consumer saves the current time as its start position before it reads. A retry before the first checkpoint opens the stream at that time with `startAtOperationTime` and reads the same events again. +- The consumer does not set `maxAwaitTimeMS` by default. Microsoft's driver + compatibility verifier uses one second only to check that a cursor opens. + MongoDB drivers send this option as `getMore.maxTimeMS`; on DocumentDB a + historical scan that needs longer can return code 50 instead of an empty + batch. Set `MAX_AWAIT_MS` only after testing the largest expected backlog. - The checkpoint is saved after each sink batch. When the stream is idle, the consumer saves the post-batch resume token instead. - Delivery is at-least-once. `FAULT_EXIT_AFTER_WRITE=` makes the diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py index 32ef1dd4..17a20781 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/consumer.py @@ -86,7 +86,12 @@ def to_sink(change: dict, pod: str) -> UpdateOne: def build_watch_kwargs(token: Optional[dict], start_at: Optional[Timestamp]) -> dict: - kwargs: dict = {"max_await_time_ms": int(os.getenv("MAX_AWAIT_MS", "1000"))} + kwargs: dict = {} + max_await_time_ms = os.getenv("MAX_AWAIT_MS") + if max_await_time_ms: + # DocumentDB applies this as getMore.maxTimeMS. A short value can abort + # historical catch-up, so omit it unless the workload has been measured. + kwargs["max_await_time_ms"] = int(max_await_time_ms) full_document = os.getenv("FULL_DOCUMENT", "updateLookup") if full_document != "default": kwargs["full_document"] = full_document From 97faabdf207a24b39380c7c31cf51514561bb476 Mon Sep 17 00:00:00 2001 From: hellices Date: Thu, 8 Oct 2026 22:33:48 +0900 Subject: [PATCH 14/14] =?UTF-8?q?fix(azure-documentdb):=20sample=20?= =?UTF-8?q?=EA=B2=80=EC=A6=9D=C2=B7checkpoint=20=EC=95=88=EC=A0=84?= =?UTF-8?q?=EC=84=B1=20=EB=B3=B4=EC=99=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit update와 replace를 version으로 구분해 누락·순서를 검증하고, checkpoint를 조건부 단일 Put Blob으로 원자 교체한다. 빈 envsubst 값과 DAG window 대기 안내도 바로잡는다. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../azure-documentdb/change-streams/index.md | 9 +- .../change-streams/measurements/index.md | 8 ++ .../samples/python-aks/README.md | 12 +- .../samples/python-aks/app/lake_export.py | 46 +++++-- .../samples/python-aks/app/requirements.txt | 1 + .../python-aks/app/stale_writer_test.py | 10 +- .../python-aks/app/test_lake_export.py | 116 ++++++++++++++++++ .../python-aks/app/test_verification.py | 46 +++++++ .../samples/python-aks/app/verify.py | 45 +++++-- .../samples/python-aks/app/verify_lake.py | 16 ++- 10 files changed, 271 insertions(+), 38 deletions(-) create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_lake_export.py create mode 100644 docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_verification.py diff --git a/docs/services/azure-documentdb/change-streams/index.md b/docs/services/azure-documentdb/change-streams/index.md index 8cf94fdb..7142cc78 100644 --- a/docs/services/azure-documentdb/change-streams/index.md +++ b/docs/services/azure-documentdb/change-streams/index.md @@ -22,6 +22,8 @@ official_sources: url: https://learn.microsoft.com/rest/api/storageservices/lease-blob - title: Path - Update url: https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update + - title: Put Blob + url: https://learn.microsoft.com/rest/api/storageservices/put-blob - title: Best practices for using Azure Data Lake Storage url: https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices --- @@ -72,7 +74,8 @@ Airflow가 매시 2분, 12분, 22분처럼 10분 구간이 닫히고 2분 뒤에 이벤트 바로 앞의 resume token으로 만듭니다. 4. 올릴 청크 파일 경로를 checkpoint에 먼저 기록합니다. 그 다음 청크 파일에 lease를 잡고 checkpoint가 자기 기록 그대로인지 확인한 뒤 그 lease ID로 - 올립니다. 마지막으로 checkpoint를 ETag 조건으로 갱신합니다. + 올립니다. 마지막으로 checkpoint를 ETag 조건이 붙은 단일 `Put Blob`으로 + 교체해 갱신 중 장애가 나도 이전 위치를 보존합니다. DAG는 `max_active_runs=1`이고 실패하면 세 번까지 재시도합니다. 재시도는 같은 checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁니다. 첫 실행의 @@ -112,6 +115,9 @@ checkpoint에서 같은 이벤트를 읽어 같은 파일 이름으로 덮어씁 파일 이름에 넣으면 다른 파일이 생겨 중복이 됩니다. - checkpoint가 없는 첫 실행은 시작 위치를 먼저 저장합니다. 저장하지 않으면 첫 청크를 쓰다 실패했을 때 재시도가 더 뒤에서 시작해 그 사이 이벤트를 잃습니다. +- checkpoint JSON은 ETag 조건이 붙은 단일 `Put Blob`으로 교체합니다. DFS + `upload_data(overwrite=True)`는 기존 path를 자른 뒤 append/flush하는 중 장애가 + 나면 이전 checkpoint도 잃을 수 있습니다. - 실행이 겹치지 않게 `max_active_runs=1`을 두고 실행 동안 lock 파일 lease를 잡습니다. checkpoint ETag 조건은 checkpoint만 보호합니다. lock lease를 잃은 실행이 늦게 업로드하면 다음 실행이 쓴 더 짧은 청크를 덮어씁니다. 그래서 올릴 파일 @@ -273,6 +279,7 @@ Learn 문서와 다르게 동작했거나 문서에 설명이 없는 부분도 - [MongoDB Change Streams Specification](https://github.com/mongodb/specifications/blob/master/source/change-streams/change-streams.md) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) - [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) +- [Put Blob](https://learn.microsoft.com/rest/api/storageservices/put-blob) - [Best practices for using Azure Data Lake Storage](https://learn.microsoft.com/azure/storage/blobs/data-lake-storage-best-practices) ## 더 읽을 문서 diff --git a/docs/services/azure-documentdb/change-streams/measurements/index.md b/docs/services/azure-documentdb/change-streams/measurements/index.md index 8beefb62..2ff566c8 100644 --- a/docs/services/azure-documentdb/change-streams/measurements/index.md +++ b/docs/services/azure-documentdb/change-streams/measurements/index.md @@ -23,6 +23,8 @@ official_sources: url: https://learn.microsoft.com/rest/api/storageservices/lease-blob - title: Path - Update url: https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update + - title: Put Blob + url: https://learn.microsoft.com/rest/api/storageservices/put-blob --- # Azure DocumentDB change stream Parquet 적재 측정 상세 @@ -175,6 +177,11 @@ checkpoint에서 230,000건을 모두 읽어 6개 파일로 썼습니다. 업로드를 마치고 checkpoint를 갱신하기 전에 A가 lease를 끊으면 확인을 통과합니다. 파일 경로를 checkpoint에 먼저 기록하는 단계를 더해 막았고 이 순서는 로컬 fake로만 재현했습니다. +- 후속 리뷰에서 DFS `upload_data(overwrite=True)`가 checkpoint를 먼저 truncate한 뒤 + append/flush하는 다중 요청임을 확인했습니다. 작은 checkpoint JSON은 ETag 조건을 + 유지한 단일 `Put Blob`으로 교체했고, 업로드 실패 시 기존 body가 남는 회귀 테스트를 + 추가했습니다. 이 원자 교체 변경은 로컬 fake로 확인했으며 위 Azure 수치를 다시 + 측정한 것은 아닙니다. ## 5분 주기 지연(이전 구성) @@ -441,3 +448,4 @@ action 목록에도 전후 모두 `changeStream`과 `find`가 있었으므로 - [MongoDB Change Streams Specification](https://github.com/mongodb/specifications/blob/master/source/change-streams/change-streams.md) - [Lease Blob](https://learn.microsoft.com/rest/api/storageservices/lease-blob) - [Path - Update](https://learn.microsoft.com/rest/api/storageservices/datalakestoragegen2/path/update) +- [Put Blob](https://learn.microsoft.com/rest/api/storageservices/put-blob) diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md index 151ba6e5..2f0a4512 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/README.md @@ -18,6 +18,7 @@ through a private endpoint. | `app/verify.py` | Compares the generator's expected events with a sink collection | | `app/lake_export.py` | One export run: resumes from the checkpoint in the lake, writes the closed 10-minute windows as Parquet chunks of at most 80 MB of row data and exits | | `app/verify_lake.py` | Compares the generator's expected events with the Parquet files | +| `app/test_*.py` | Local regression tests for event identity/order and atomic checkpoint replacement | | `app/stale_writer_test.py` | Test only. Stops one export run before its first upload, lets a second run write the same file and checks that the first run cannot overwrite it | | `airflow/` | DAG that runs `lake_export.py` with `KubernetesPodOperator`, the Airflow image and chart values | | `k8s/` | Consumer Deployment, generic Job, toolbox pod, lake Job, RBAC for the Airflow scheduler | @@ -263,7 +264,8 @@ first run after its window closes. ```bash JOB_NAME=gen-af1 SCRIPT=generator.py RUN_ID=af1 DOCS=100000 WORKERS=16 RATE=0 \ envsubst < k8s/job.yaml | kubectl apply -f - -# After the next DAG run finishes: +# After the generator finishes, wait until its last event's window closes and +# the DAG run that starts after that boundary completes successfully: JOB_NAME=verify-lake-af1 SCRIPT=verify_lake.py LAKE_PREFIX=orders RUN_ID=af1 MAX_EVENTS=0 \ FAULT_EXIT_AFTER_UPLOAD=0 envsubst < k8s/lake.yaml | kubectl apply -f - kubectl -n cslab logs job/verify-lake-af1 @@ -301,7 +303,9 @@ Export behavior: from the same checkpoint and overwrites the same file. A chunk whose events have no `wallTime` goes to `dt=unknown`. - The checkpoint is `_checkpoints/.json` in the same file system. It is - written with an ETag condition after each chunk upload. + replaced with one conditional `Put Blob` after each chunk upload. The small + JSON write is atomic, so an interrupted replacement leaves the previous + checkpoint readable. - A run holds a 20-second lease on `_checkpoints/.lock` and renews it in the background. A second run waits up to 90 seconds for the lease and then fails, so two runs do not write chunks at the same time. A lease that @@ -337,8 +341,8 @@ Export behavior: DAG passes it to the pod and builds its schedule from it, so the two stay equal. `CS_EXPORT_OFFSET_MIN` (2) delays the run after the window closes. `CS_CHUNK_BYTES` sets the file limit. -- `consumer.py` always passes `MAX_AWAIT_MS`, 1000 by default. Raise it before - the consumer resumes a large backlog. +- `consumer.py` and `lake_export.py` omit `MAX_AWAIT_MS` by default. A short + value becomes `getMore.maxTimeMS` and can abort historical catch-up. ## 6. Measured scenarios diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py index 85a00fd4..fafc1926 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/lake_export.py @@ -40,6 +40,7 @@ from azure.core import MatchConditions from azure.core.exceptions import HttpResponseError, ResourceNotFoundError from azure.identity import DefaultAzureCredential +from azure.storage.blob import BlobServiceClient from azure.storage.filedatalake import DataLakeLeaseClient, DataLakeServiceClient from bson import json_util from bson.timestamp import Timestamp @@ -70,8 +71,9 @@ class Checkpoint: The position is a resume token, or the start time saved by the first run. """ - def __init__(self, fs, stream_id: str): + def __init__(self, fs, blobs, stream_id: str): self.file = fs.get_file_client(f"_checkpoints/{stream_id}.json") + self.blob = blobs.get_blob_client(f"_checkpoints/{stream_id}.json") self.etag: Optional[str] = None def load(self) -> Optional[dict]: @@ -86,10 +88,18 @@ def save(self, token: Optional[dict], events: int, start_at: Optional[Timestamp] writing: Optional[str] = None) -> None: body = json_util.dumps({"token": token, "start_at": start_at, "updated_at": utcnow(), "events": events, "writing": writing, "pod": socket.gethostname()}) - # IfNotModified fails if another run moved the checkpoint since we read it. - condition = (dict(etag=self.etag, match_condition=MatchConditions.IfNotModified) - if self.etag else dict(match_condition=MatchConditions.IfMissing)) - result = self.file.upload_data(body, overwrite=True, **condition) + # A small checkpoint is one conditional Put Blob. Unlike DFS + # upload_data(overwrite=True), it never truncates the old checkpoint + # before the replacement body is committed. + if self.etag: + result = self.blob.upload_blob( + body, + overwrite=True, + etag=self.etag, + match_condition=MatchConditions.IfNotModified, + ) + else: + result = self.blob.upload_blob(body, overwrite=False) self.etag = result["etag"] def claim(self, start: dict, path: str) -> None: @@ -294,6 +304,21 @@ def publish(fs, lock: StreamLock, checkpoint: Checkpoint, prefix: str, chunk: Ch return size +def get_storage_clients(): + credential = DefaultAzureCredential() + lake_url = env("LAKE_URL") + filesystem = env("LAKE_FILESYSTEM", "cdc") + fs = DataLakeServiceClient( + lake_url, + credential=credential, + ).get_file_system_client(filesystem) + blobs = BlobServiceClient( + lake_url.replace(".dfs.", ".blob."), + credential=credential, + ).get_container_client(filesystem) + return fs, blobs + + def main() -> None: stream_id = env("STREAM_ID", "orders") prefix = env("LAKE_PREFIX", "orders") @@ -301,16 +326,15 @@ def main() -> None: if window_minutes <= 0 or 60 % window_minutes: raise SystemExit("WINDOW_MINUTES must divide 60") chunk_bytes = int(os.getenv("CHUNK_BYTES", str(80_000_000))) - max_events = int(os.getenv("MAX_EVENTS", "0")) # 0 = until caught up - max_seconds = float(os.getenv("MAX_SECONDS", "0")) + max_events = int(os.getenv("MAX_EVENTS") or "0") # envsubst may leave "" + max_seconds = float(os.getenv("MAX_SECONDS") or "0") # Test-only: exit after uploading this many chunks, before the checkpoint. - fault_after_chunks = int(os.getenv("FAULT_EXIT_AFTER_UPLOAD", "0")) + fault_after_chunks = int(os.getenv("FAULT_EXIT_AFTER_UPLOAD") or "0") - service = DataLakeServiceClient(env("LAKE_URL"), credential=DefaultAzureCredential()) - fs = service.get_file_system_client(env("LAKE_FILESYSTEM", "cdc")) + fs, blobs = get_storage_clients() lock = StreamLock(fs, stream_id) lock.acquire() - checkpoint = Checkpoint(fs, stream_id) + checkpoint = Checkpoint(fs, blobs, stream_id) position = checkpoint.load() if position is None: # First run: save the start time before reading, so a retry after a diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt index 4934b775..f6a23862 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/requirements.txt @@ -2,4 +2,5 @@ pymongo[srv]==4.18.2 certifi==2026.7.22 pyarrow==25.0.1 azure-storage-file-datalake==12.26.0 +azure-storage-blob==12.31.0 azure-identity==1.26.0 diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py index 569a3844..216ca044 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/stale_writer_test.py @@ -76,12 +76,14 @@ def main() -> None: if "path" not in held: raise SystemExit("run A wrote no chunk; write more events in a closed window first") - service = lake_export.DataLakeServiceClient(lake_export.env("LAKE_URL"), - credential=lake_export.DefaultAzureCredential()) - fs = service.get_file_system_client(lake_export.env("LAKE_FILESYSTEM", "cdc")) + fs, blobs = lake_export.get_storage_clients() data = fs.get_file_client(held["path"]).download_file().readall() tokens = pq.read_table(io.BytesIO(data), columns=["resume_token"]).column("resume_token").to_pylist() - checkpoint = lake_export.Checkpoint(fs, lake_export.env("STREAM_ID", "orders")).load() + checkpoint = lake_export.Checkpoint( + fs, + blobs, + lake_export.env("STREAM_ID", "orders"), + ).load() ok = held["b_exit"] == 0 and tokens[-1] == checkpoint["token"]["_data"] print(json.dumps({"result": "pass" if ok else "fail", "path": held["path"], "file_rows": len(tokens), "file_ends_at_checkpoint": tokens[-1] == checkpoint["token"]["_data"], diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_lake_export.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_lake_export.py new file mode 100644 index 00000000..8fac312e --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_lake_export.py @@ -0,0 +1,116 @@ +import unittest +from types import SimpleNamespace + +from azure.core import MatchConditions +from bson import json_util + +from lake_export import Checkpoint + + +class State: + def __init__(self, body): + self.body = body + self.etag = '"etag-1"' + + +class Download: + def __init__(self, state): + self.state = state + self.properties = SimpleNamespace(etag=state.etag) + + def readall(self): + return self.state.body + + +class FileClient: + def __init__(self, state): + self.state = state + + def download_file(self): + return Download(self.state) + + def get_file_properties(self): + return SimpleNamespace(etag=self.state.etag) + + def upload_data(self, *_args, **_kwargs): + raise AssertionError("checkpoint save must not use DFS upload_data") + + +class FileSystem: + def __init__(self, state): + self.state = state + + def get_file_client(self, _path): + return FileClient(self.state) + + +class BlobClient: + def __init__(self, state): + self.state = state + self.fail = False + self.calls = [] + + def upload_blob(self, body, **kwargs): + self.calls.append(kwargs) + if self.fail: + raise RuntimeError("injected Put Blob failure") + if kwargs.get("etag") and kwargs["etag"] != self.state.etag: + raise AssertionError("wrong ETag") + self.state.body = body + self.state.etag = '"etag-2"' + return {"etag": self.state.etag} + + +class BlobContainer: + def __init__(self, blob): + self.blob = blob + + def get_blob_client(self, _path): + return self.blob + + +class CheckpointAtomicWriteTest(unittest.TestCase): + def setUp(self): + self.old = { + "token": {"_data": "old"}, + "start_at": None, + "events": 1, + "writing": None, + } + self.state = State(json_util.dumps(self.old)) + self.blob = BlobClient(self.state) + self.checkpoint = Checkpoint( + FileSystem(self.state), + BlobContainer(self.blob), + "orders", + ) + self.assertEqual(self.checkpoint.load()["token"], {"_data": "old"}) + + def test_existing_checkpoint_is_one_conditional_blob_upload(self): + self.checkpoint.save({"_data": "new"}, 2) + + self.assertEqual( + self.blob.calls, + [ + { + "overwrite": True, + "etag": '"etag-1"', + "match_condition": MatchConditions.IfNotModified, + } + ], + ) + self.assertEqual(self.checkpoint.load()["token"], {"_data": "new"}) + + def test_failed_upload_preserves_previous_checkpoint(self): + previous_body = self.state.body + self.blob.fail = True + + with self.assertRaisesRegex(RuntimeError, "injected Put Blob failure"): + self.checkpoint.save({"_data": "new"}, 2) + + self.assertEqual(self.state.body, previous_body) + self.assertEqual(self.checkpoint.load()["token"], {"_data": "old"}) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_verification.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_verification.py new file mode 100644 index 00000000..a73bfdb7 --- /dev/null +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/test_verification.py @@ -0,0 +1,46 @@ +import unittest +from collections import Counter + +from verify import event_identity, event_rank, expected_events + + +class EventIdentityTest(unittest.TestCase): + def test_replacement_is_distinct_from_ordinary_update(self): + expected = expected_events("run", 1) + + self.assertEqual( + expected, + Counter( + { + event_identity("run:0000000", "insert", 1): 1, + event_identity("run:0000000", "update", 2): 1, + event_identity("run:0000000", "update", 3): 1, + event_identity("run:0000000", "delete", None): 1, + } + ), + ) + + substituted = Counter( + { + event_identity("run:0000000", "insert", 1): 1, + event_identity("run:0000000", "update", 2): 2, + event_identity("run:0000000", "delete", None): 1, + } + ) + self.assertTrue(expected - substituted) + self.assertTrue(substituted - expected) + + def test_update_versions_have_distinct_order(self): + self.assertEqual( + [ + event_rank("insert", 1), + event_rank("update", 2), + event_rank("update", 3), + event_rank("delete", None), + ], + [0, 1, 2, 3], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py index 1d68547a..220c21a3 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify.py @@ -5,6 +5,7 @@ import re import sys from collections import Counter, defaultdict +from typing import Optional from common import RUN_COLLECTION, SINK_COLLECTION, env, get_client, get_database @@ -21,19 +22,36 @@ def summarize(values: list) -> dict: "max": percentile(values, 100), "n": len(values)} -def expected_events(run_id: str, docs: int) -> Counter: - """Expected (doc_id, op) counts. +def event_identity(doc_id: str, op: str, version: Optional[int]) -> tuple: + """Identity that distinguishes the ordinary update from a replacement.""" + return doc_id, op, version if op in ("insert", "update") else None + + +def event_rank(op: str, version: Optional[int]) -> int: + """Expected per-document order for the deterministic generator.""" + if op == "insert": + return 0 + if op == "update" and version == 2: + return 1 + if op == "update" and version == 3: + return 2 + if op == "delete": + return 3 + return 99 - ``replace_one`` is counted under ``update`` because the cluster under test - emits replacements as ``operationType: update``. - """ + +def expected_events(run_id: str, docs: int) -> Counter: + """Expected (doc_id, operationType, version) identities.""" expected = Counter() for i in range(docs): doc_id = f"{run_id}:{i:07d}" - expected[(doc_id, "insert")] += 1 - expected[(doc_id, "update")] += 2 if i % 10 == 0 else 1 + expected[event_identity(doc_id, "insert", 1)] += 1 + expected[event_identity(doc_id, "update", 2)] += 1 + if i % 10 == 0: + # DocumentDB reports replace_one as operationType: update. + expected[event_identity(doc_id, "update", 3)] += 1 if i % 5 == 0: - expected[(doc_id, "delete")] += 1 + expected[event_identity(doc_id, "delete", None)] += 1 return expected @@ -50,20 +68,21 @@ def main() -> None: events = list(db[sink].find( {"doc_id": {"$regex": f"^{re.escape(run_id)}:"}}, {"doc_id": 1, "op": 1, "recv_ns": 1, "pod": 1, "has_full_document": 1, - "has_update_description": 1, "deliveries": 1, "lag_ms": 1}, + "has_update_description": 1, "deliveries": 1, "lag_ms": 1, "version": 1}, )) # The sink is keyed by resume token, so each row is one distinct event and # replays show up as deliveries > 1 instead of extra rows. - got = Counter((e["doc_id"], e["op"]) for e in events) + got = Counter(event_identity(e["doc_id"], e["op"], e.get("version")) for e in events) expected = expected_events(run_id, run["docs"]) missing = sorted((expected - got).elements()) unexpected = sorted((got - expected).elements()) - # Per-document order: insert -> update(s) -> delete. - order = {"insert": 0, "update": 1, "replace": 1, "delete": 2} + # Per-document order: insert(v1) -> update(v2) -> replacement(v3) -> delete. per_doc = defaultdict(list) for e in events: - per_doc[e["doc_id"]].append((e["recv_ns"], order[e["op"]])) + per_doc[e["doc_id"]].append( + (e["recv_ns"], event_rank(e["op"], e.get("version"))) + ) out_of_order = [d for d, seq in per_doc.items() if [o for _, o in sorted(seq)] != sorted(o for _, o in seq)] diff --git a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py index 8eb46517..f6dd8010 100644 --- a/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py +++ b/docs/services/azure-documentdb/change-streams/samples/python-aks/app/verify_lake.py @@ -11,10 +11,11 @@ import pyarrow.parquet as pq from azure.identity import DefaultAzureCredential from azure.storage.filedatalake import DataLakeServiceClient +from bson import json_util from common import RUN_COLLECTION, env, get_client, get_database from lake_export import window_start -from verify import expected_events, percentile, summarize +from verify import event_identity, event_rank, expected_events, percentile, summarize logging.getLogger("azure").setLevel(logging.WARNING) @@ -32,7 +33,7 @@ def main() -> None: prefix = env("LAKE_PREFIX", "orders") window_minutes = int(os.getenv("WINDOW_MINUTES", "10")) chunk_bytes = int(os.getenv("CHUNK_BYTES", str(80_000_000))) - columns = ["resume_token", "op", "doc_id", "wall_time", "read_at"] + columns = ["resume_token", "op", "doc_id", "wall_time", "read_at", "full_document"] rows = [] files = [] @@ -52,6 +53,12 @@ def main() -> None: # The listing returns last-modified as a naive UTC datetime. committed_at = path.last_modified.replace(tzinfo=timezone.utc) for i, r in enumerate(mine): + full_document = ( + json_util.loads(r["full_document"]) + if r["full_document"] is not None + else {} + ) + r["version"] = full_document.get("version") r["committed_at"] = committed_at r["idx"] = i rows.append(r) @@ -67,15 +74,14 @@ def main() -> None: seen.add(r["resume_token"]) events.append(r) - got = Counter((e["doc_id"], e["op"]) for e in events) + got = Counter(event_identity(e["doc_id"], e["op"], e.get("version")) for e in events) expected = expected_events(run_id, run["docs"]) missing = sorted((expected - got).elements()) unexpected = sorted((got - expected).elements()) - order = {"insert": 0, "update": 1, "replace": 1, "delete": 2} per_doc = defaultdict(list) for e in events: - per_doc[e["doc_id"]].append(order[e["op"]]) + per_doc[e["doc_id"]].append(event_rank(e["op"], e.get("version"))) out_of_order = [d for d, seq in per_doc.items() if seq != sorted(seq)] read_lag_ms = [(e["read_at"] - e["wall_time"]).total_seconds() * 1000 for e in events if e["wall_time"]]