-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathDockerfile
More file actions
110 lines (98 loc) · 5.13 KB
/
Copy pathDockerfile
File metadata and controls
110 lines (98 loc) · 5.13 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
# Single image for all of Aperture: builds the frontend (apps/web) and
# bundles it into the GPU backend (apps/api), which serves both the API
# and the built UI itself (see the WEB_DIST static mount at the end of
# apps/api/src/aperture_api/main.py) — one image, one process, one port.
# No nginx, no reverse proxy, no cross-origin requests to reason about.
#
# Needs an actual NVIDIA GPU at runtime:
# docker run --gpus all -p 8000:8000 -v $(pwd)/data:/app/data <image>
# (or docker-compose.yml, which wires up the GPU + volume for you). The
# host needs the NVIDIA Container Toolkit installed either way —
# https://github.com/NVIDIA/nvidia-container-toolkit.
# ARG declared before the first FROM is "global" in Docker — available to
# every FROM line in the file. Must live here, not inside a stage.
ARG CUDA_VERSION=12.8.1
# ---- frontend build stage ----
FROM node:22-alpine AS frontend
WORKDIR /app
# Copying the whole workspace before `npm ci` (rather than just the root
# package.json) is deliberate: this is an npm-workspaces monorepo, so npm
# needs every package's package.json present to link the workspace
# correctly. That trades away some Docker layer caching (a source-only
# change also invalidates the npm ci layer) for a Dockerfile that keeps
# working as packages are added or removed, which happens often here.
COPY . .
RUN npm ci
# Vite inlines VITE_* variables into the built bundle at build time, not
# runtime. Empty string is correct here (not just the default): the
# frontend and API are the same origin now (same process, same port), so
# every request is a plain relative /api/... or /data/... fetch — see
# setApiBase in apps/web/src/main.tsx.
ARG VITE_API_BASE_URL=""
ENV VITE_API_BASE_URL=$VITE_API_BASE_URL
RUN npm run build
# ---- backend runtime stage ----
# Full CUDA "devel" (not "runtime") on purpose — gpt-oss-style MXFP4
# checkpoints load through triton-compiled kernels (see the `kernels`
# dependency in apps/api/pyproject.toml), which can shell out to
# ptxas/nvcc from the CUDA toolkit at first use; the slimmer runtime image
# doesn't ship those.
#
# CUDA_VERSION and TORCH_CUDA_ARCH must match each other and the host
# driver — see the table below. Defaults target the latest tested pair;
# override via --build-arg or docker-compose's args / a .env file.
#
# Host CUDA driver ver | CUDA_VERSION | TORCH_CUDA_ARCH
# ----------------------|---------------|----------------
# 12.8.x (≥570 driver) | 12.8.1 | cu128 ← default
# 12.6.x (≥560 driver) | 12.6.3 | cu126
# 12.4.x (≥550 driver) | 12.4.1 | cu124
# 12.1.x (≥530 driver) | 12.1.0 | cu121
#
# Run `nvidia-smi` on the host; the "CUDA Version" field in the top-right
# corner tells you which row to pick.
FROM nvidia/cuda:${CUDA_VERSION}-devel-ubuntu22.04
# Re-declare after FROM so this stage can read it (build-arg scope resets
# at each FROM; the default must match CUDA_VERSION's default above).
ARG TORCH_CUDA_ARCH=cu128
# Ubuntu 22.04's python3 is 3.10; deadsnakes PPA gives us 3.12 to match
# apps/api/pyproject.toml's requires-python = ">=3.12".
# DEBIAN_FRONTEND=noninteractive + TZ short-circuit tzdata's debconf
# geographic-area prompt (pulled in transitively by
# software-properties-common), which otherwise hangs the build waiting on
# a TTY that a non-interactive `docker build` never provides.
ENV DEBIAN_FRONTEND=noninteractive TZ=Etc/UTC
RUN apt-get update && apt-get install -y --no-install-recommends software-properties-common && \
add-apt-repository ppa:deadsnakes/ppa && \
apt-get update && apt-get install -y --no-install-recommends \
python3.12 \
python3.12-dev \
python3.12-venv \
&& rm -rf /var/lib/apt/lists/*
# Preserve the monorepo's on-disk layout, not just apps/api in isolation:
# apps/api/src/aperture_api/paths.py derives the repo root (and so
# data/models/ and the frontend's dist/ below) from its own file location
# via `Path(__file__).resolve().parents[4]`, which only lands in the right
# place if apps/api keeps its real position four directories below the
# root.
WORKDIR /app
COPY . .
COPY --from=frontend /app/apps/web/dist ./apps/web/dist
# A venv here is mostly to keep apt's system Python's PEP 668
# "externally-managed-environment" guard out of the way, not for
# isolation (this container runs nothing else).
RUN python3.12 -m venv /opt/venv
ENV PATH="/opt/venv/bin:$PATH"
# --extra-index-url (rather than --index-url) is deliberate: it adds
# PyTorch's CUDA wheel index as a *second* source alongside regular PyPI,
# rather than replacing PyPI outright, so every other dependency below
# still resolves normally. Pick the index matching the CUDA base image
# above if you change it (see https://pytorch.org for the current list).
RUN pip install --no-cache-dir -e ./apps/api --extra-index-url https://download.pytorch.org/whl/${TORCH_CUDA_ARCH}
# data/models/ — downloaded checkpoints — is bind-mounted at runtime
# (docker-compose.yml) so it survives container recreation; this just
# makes sure the directory exists before the app's first write.
RUN mkdir -p /app/data/models
EXPOSE 8000
WORKDIR /app/apps/api
CMD ["uvicorn", "aperture_api.main:app", "--host", "0.0.0.0", "--port", "8000"]