Repository navigation
Expand file tree
/
Copy pathconfig.example.env
More file actions
64 lines (55 loc) · 2.1 KB
/
Copy pathconfig.example.env
File metadata and controls
64 lines (55 loc) · 2.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
# Copy this file to config.env and edit the values for your Google Cloud project.
# Required Google Cloud settings.
GCP_PROJECT_ID="your-google-cloud-project-id"
GCP_REGION="us-central1"
# A short lowercase prefix used for Cloud Run service names and image names.
SERVICE_PREFIX="ollama-agent"
# Choose any Ollama model. Examples:
# qwen3-coder:30b
# gpt-oss:120b
# llama3.1:8b
# deepseek-coder:latest
# deepseek-r1:671b
OLLAMA_MODEL="qwen3-coder:30b"
# Public model id exposed by the proxy and Codex profile. This can be the same
# as OLLAMA_MODEL, but a colon-free alias avoids noisy telemetry warnings.
MODEL_ALIAS="qwen3-coder-30b"
# Artifact Registry repository for runtime and proxy images.
ARTIFACT_REPO="ollama-agent-images"
# Cloud Run runtime service sizing. Tune these for the model you choose.
# Notes for large-model deployments:
# - Cloud Run now supports nvidia-rtx-pro-6000 in supported regions such as
# europe-west4.
# - For nvidia-rtx-pro-6000, Cloud Run requires at least 20 CPU and 80Gi memory.
# - 30 CPU + 120Gi + 1x nvidia-rtx-pro-6000 was validated as an accepted Cloud
# Run deployment shape in europe-west4.
# - GPU VRAM and instance memory are separate. Do not assume 96Gi is the memory
# cap just because RTX PRO 6000 exposes 96 GB VRAM.
# - STARTUP_INITIAL_DELAY_SECONDS must be <= 240 or Cloud Run rejects deploy.
CLOUD_RUN_CPU="8"
CLOUD_RUN_MEMORY="32Gi"
GPU_TYPE="nvidia-l4"
MAX_INSTANCES="1"
CONCURRENCY="4"
STARTUP_INITIAL_DELAY_SECONDS="180"
# Example high-memory RTX PRO 6000 profile for very large Ollama models:
# GCP_REGION="europe-west4"
# SERVICE_PREFIX="deepseek-r1"
# OLLAMA_MODEL="deepseek-r1:671b"
# MODEL_ALIAS="deepseek-r1-671b-iq4xs"
# CLOUD_RUN_CPU="30"
# CLOUD_RUN_MEMORY="120Gi"
# GPU_TYPE="nvidia-rtx-pro-6000"
# MAX_INSTANCES="1"
# CONCURRENCY="1"
# STARTUP_INITIAL_DELAY_SECONDS="240"
# MODEL_CONTEXT_WINDOW="4096"
# REQUEST_TIMEOUT_SECONDS="7200"
# CPU-only proxy sizing.
PROXY_CPU="1"
PROXY_MEMORY="1Gi"
# Codex profile settings.
CODEX_PROFILE_NAME="ollama-cloud"
CODEX_PROVIDER_NAME="ollama_cloud_proxy"
MODEL_CONTEXT_WINDOW="32768"
REQUEST_TIMEOUT_SECONDS="3600"