From b3e2ed1d1e21b0e70e730d3e5b8bb07bfd6399ea Mon Sep 17 00:00:00 2001 From: Tyler <53561637+im-tyler@users.noreply.github.com> Date: Fri, 18 Sep 2026 04:00:43 -0700 Subject: [PATCH 1/4] =?UTF-8?q?feat(docker):=20RecreateSpec=20=E2=80=94=20?= =?UTF-8?q?full-fidelity=20container=20recreation=20(F20)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit docker run commands were rebuilt inline from a partial inspect struct, so recreation silently dropped everything the struct did not happen to list: log rotation (every recreated container lost max-size=10m), entrypoint overrides, stop timeout/signal, extra hosts, sysctls, tmpfs, capabilities, security opts, read-only/privileged flags, and secondary networks. InspectRecreate captures a complete RecreateSpec from one inspect; the pure RenderRecreateArgs renders it (values quoted exactly once at the shell boundary, TCL-11); Recreate applies the avoidPorts reallocation and force-removes + reruns. Restart is now InspectRecreate + Recreate with the same signature, so rollback/heal/displaced-workload restore all gain the full spec. A multi-element entrypoint that differs from the image's own fails closed (the docker CLI cannot represent argv there) instead of being joined into a wrong single string; one matching the image is dropped — the recreated container inherits it from the immutable image ID (F19 rule). HostPortFor resolves the host binding for a named CONTAINER port — the primary-port lookup fields[0] reads cannot do for multi-port containers (TCL-14); rollback consumes it via the F14 record. --- internal/docker/docker.go | 245 ------------- internal/docker/recreate.go | 590 +++++++++++++++++++++++++++++++ internal/docker/recreate_test.go | 325 +++++++++++++++++ 3 files changed, 915 insertions(+), 245 deletions(-) create mode 100644 internal/docker/recreate.go create mode 100644 internal/docker/recreate_test.go diff --git a/internal/docker/docker.go b/internal/docker/docker.go index 51ca145..5fd60b5 100644 --- a/internal/docker/docker.go +++ b/internal/docker/docker.go @@ -328,251 +328,6 @@ func (c *Client) Start(ctx context.Context, name string) error { return nil } -// containerInspect mirrors the subset of `docker inspect` JSON we use to -// reconstruct a container's `docker run` invocation. -type containerInspect struct { - // Image is the container's immutable, content-addressed image ID - // (sha256:…). This — not Config.Image, which is the original (possibly - // mutable tag) reference — is the identity a faithful recreation must - // preserve: after a newer image is pulled under the same tag, - // recreating from Config.Image silently runs the NEW bytes while - // reporting the old version (audit F19). - Image string - Config struct { - Image string - Env []string - Cmd []string - WorkingDir string - User string - Labels map[string]string - Healthcheck *struct { - Test []string - } - } - HostConfig struct { - NetworkMode string - PortBindings map[string][]struct { - HostIp string - HostPort string - } - Binds []string - Mounts []struct { - Type string // "bind" | "volume" | "tmpfs" - Source string - Target string - ReadOnly bool - } - RestartPolicy struct { - Name string - } - Memory int64 // bytes - NanoCpus int64 // nano-CPUs (1 CPU = 1e9) - } - NetworkSettings struct { - Networks map[string]struct { - Aliases []string - } - } -} - -// Restart fully recreates a container with its original configuration: -// inspect the existing container, force-remove it, then `docker run` with -// the same name + extracted args. -// -// Use this in rollback flows where plain Start() is insufficient. Docker -// 29 silently fails to re-publish HostConfig.PortBindings if another -// container has taken + released the host port since the original stop -// (the bind also detaches from custom networks). Recreating from scratch -// sidesteps that landmine. -// -// avoidPorts is the set of host ports currently held by containers this -// restart must not collide with — critically, the live container(s) being -// rolled back FROM, which are still running (and still holding their port) -// at the point Restart is called, since rollback starts the target before -// stopping the current version for zero-downtime. A single-hop rollback -// (to the immediately-previous version) can never collide: that version's -// port was freed by the deploy that superseded it and nothing has claimed -// it since. A --to rollback reaching back further can: Docker's ephemeral -// port allocator reuses freed ports, so an older version's original port -// may since have been handed to what is now the live container. Found via -// live testing (v1→49152, v1-tls→49153, v2 reused 49152 after v1-tls -// stopped; `rollback --to v1` then collided with the still-running v2). -// When a binding's original port is in avoidPorts, a fresh one is -// allocated instead — safe because HostPort() (not persisted state) is -// what the rest of the rollback reads back afterward. Pass nil to -// preserve every port binding exactly (no caller does today, but this -// keeps the zero-collision-checking path available for a non-rollback use -// of Restart in the future). -// -// Preserved across the recreate: image, network mode + aliases, port -// bindings (subject to the above), env, bind mounts, named-volume + tmpfs -// mounts, command, working dir, user, labels, restart policy, memory + CPU -// limits, and the --no-healthcheck NONE marker. -func (c *Client) Restart(ctx context.Context, name string, avoidPorts map[int]bool) error { - raw, err := c.exec.Run(ctx, "docker inspect "+ssh.ShellQuote(name)) - if err != nil { - return fmt.Errorf("inspecting %s: %w", name, err) - } - var arr []containerInspect - if err := json.Unmarshal([]byte(raw), &arr); err != nil { - return fmt.Errorf("parsing inspect of %s: %w", name, err) - } - if len(arr) == 0 { - return fmt.Errorf("container %s not found", name) - } - spec := arr[0] - - args := []string{"-d", "--name", ssh.ShellQuote(name)} - - // Network mode (e.g. "teploy"). Skip docker's "default" / "bridge". - // Inspect-derived values are DATA, not shell syntax — an image whose - // metadata carries spaces or metacharacters must never be interpreted - // by the host shell (TCL-11). Every value below is quoted exactly once - // at this shell boundary. - if spec.HostConfig.NetworkMode != "" && spec.HostConfig.NetworkMode != "default" && spec.HostConfig.NetworkMode != "bridge" { - args = append(args, "--network", ssh.ShellQuote(spec.HostConfig.NetworkMode)) - } - - // Network aliases on the primary network. Skip the container-name auto-alias - // and anything that's a prefix of the container ID (also auto-added). - primary := spec.HostConfig.NetworkMode - if netInfo, ok := spec.NetworkSettings.Networks[primary]; ok { - seenAlias := map[string]bool{name: true} - aliases := append([]string(nil), netInfo.Aliases...) - sort.Strings(aliases) - for _, alias := range aliases { - if alias == "" || seenAlias[alias] { - continue - } - seenAlias[alias] = true - args = append(args, "--network-alias", ssh.ShellQuote(alias)) - } - } - - // Port bindings: sorted for deterministic args ordering. - pbKeys := make([]string, 0, len(spec.HostConfig.PortBindings)) - for k := range spec.HostConfig.PortBindings { - pbKeys = append(pbKeys, k) - } - sort.Strings(pbKeys) - // Ports claimed so far this call, merged into avoidPorts as bindings are - // resolved — so two colliding bindings on the same container (unusual, - // but possible with multiple published ports) don't get handed the same - // replacement port. - claimed := make(map[int]bool, len(avoidPorts)) - for p := range avoidPorts { - claimed[p] = true - } - for _, containerPort := range pbKeys { - for _, b := range spec.HostConfig.PortBindings[containerPort] { - host := b.HostIp - if host == "" { - host = "0.0.0.0" - } - hostPort := b.HostPort - if origPort, err := strconv.Atoi(b.HostPort); err == nil && claimed[origPort] { - newPort, ferr := c.FindAvailablePortExcluding(ctx, claimed) - if ferr != nil { - return fmt.Errorf("port %s for %s is held by another running container and no replacement port is available: %w", b.HostPort, name, ferr) - } - hostPort = strconv.Itoa(newPort) - claimed[newPort] = true - } - args = append(args, "-p", ssh.ShellQuote(fmt.Sprintf("%s:%s:%s", host, hostPort, containerPort))) - } - } - - // Env vars. - for _, e := range spec.Config.Env { - args = append(args, "-e", ssh.ShellQuote(e)) - } - - // Bind mounts (already host:container[:mode] formatted by docker). - // Quoted: a host path with a space would otherwise word-split into a - // second -v fragment or bleed into the next argument (F17). - for _, b := range spec.HostConfig.Binds { - args = append(args, "-v", ssh.ShellQuote(b)) - } - - // Named volume + tmpfs + non-bind mounts (Binds covers bind mounts; this - // covers everything created via --mount). Use --mount syntax for fidelity. - for _, m := range spec.HostConfig.Mounts { - if m.Type == "" || m.Target == "" { - continue - } - parts := []string{"type=" + m.Type, "target=" + m.Target} - if m.Source != "" { - parts = append(parts, "source="+m.Source) - } - if m.ReadOnly { - parts = append(parts, "readonly") - } - args = append(args, "--mount", ssh.ShellQuote(strings.Join(parts, ","))) - } - - // Resource limits. - if spec.HostConfig.Memory > 0 { - args = append(args, "--memory", fmt.Sprintf("%db", spec.HostConfig.Memory)) - } - if spec.HostConfig.NanoCpus > 0 { - // NanoCpus → fractional --cpus (1e9 nano = 1.0 cpu). - cpus := float64(spec.HostConfig.NanoCpus) / 1e9 - args = append(args, "--cpus", strconv.FormatFloat(cpus, 'f', -1, 64)) - } - - // Labels (sorted for determinism). - labelKeys := make([]string, 0, len(spec.Config.Labels)) - for k := range spec.Config.Labels { - labelKeys = append(labelKeys, k) - } - sort.Strings(labelKeys) - for _, k := range labelKeys { - args = append(args, "--label", ssh.ShellQuote(k+"="+spec.Config.Labels[k])) - } - - // Preserve the explicit-NONE healthcheck (set by --no-healthcheck at run). - if spec.Config.Healthcheck != nil && len(spec.Config.Healthcheck.Test) == 1 && spec.Config.Healthcheck.Test[0] == "NONE" { - args = append(args, "--no-healthcheck") - } - - // Restart policy (skip docker default "no"). - if spec.HostConfig.RestartPolicy.Name != "" && spec.HostConfig.RestartPolicy.Name != "no" { - args = append(args, "--restart", ssh.ShellQuote(spec.HostConfig.RestartPolicy.Name)) - } - - if spec.Config.WorkingDir != "" { - args = append(args, "-w", ssh.ShellQuote(spec.Config.WorkingDir)) - } - if spec.Config.User != "" { - args = append(args, "-u", ssh.ShellQuote(spec.Config.User)) - } - - // Image (last positional before cmd). Prefer the container's immutable - // top-level image ID over the Config.Image tag reference so a tag that - // has moved to newer bytes cannot silently change what the recreated - // container runs (F19). Config.Image is kept only as a fallback for - // odd daemons that report no top-level ID. - imageRef := spec.Image - if imageRef == "" || !strings.HasPrefix(imageRef, "sha256:") { - imageRef = spec.Config.Image - } - args = append(args, ssh.ShellQuote(imageRef)) - - // Cmd. - for _, cmdPart := range spec.Config.Cmd { - args = append(args, ssh.ShellQuote(cmdPart)) - } - - // Force-remove old container, then run fresh. - if _, err := c.exec.Run(ctx, "docker rm -f "+ssh.ShellQuote(name)); err != nil { - return fmt.Errorf("removing old %s: %w", name, err) - } - if _, err := c.exec.Run(ctx, "docker run "+strings.Join(args, " ")); err != nil { - return fmt.Errorf("recreating %s: %w", name, err) - } - return nil -} - // Pull pulls a Docker image from a registry. // ScanImage runs a Trivy vulnerability scan against an image already // present on the server, streaming findings to out. Two passes because diff --git a/internal/docker/recreate.go b/internal/docker/recreate.go new file mode 100644 index 0000000..9050560 --- /dev/null +++ b/internal/docker/recreate.go @@ -0,0 +1,590 @@ +package docker + +import ( + "context" + "encoding/json" + "fmt" + "sort" + "strconv" + "strings" + + "github.com/useteploy/teploy/internal/ssh" +) + +// RecreateBinding is one host port binding of a container, normalized from +// docker's PortBindings map ("3000/tcp" keys) into comparable integers. +type RecreateBinding struct { + HostIP string `json:"host_ip,omitempty"` + HostPort int `json:"host_port,omitempty"` + ContainerPort int `json:"container_port"` + Proto string `json:"proto,omitempty"` // "" and "tcp" are the same +} + +// RecreateMount is one --mount-style mount (named volumes, tmpfs created via +// --mount, and anything else Binds does not cover). +type RecreateMount struct { + Type string `json:"type"` // "volume" | "bind" | "tmpfs" | "npipe" + Source string `json:"source,omitempty"` + Target string `json:"target"` + ReadOnly bool `json:"read_only,omitempty"` +} + +// RecreateSpec is a container's complete docker-run-representable +// configuration, captured from docker inspect — the "full RecreateSpec" of +// audit F20. Recreation renders a docker run from this spec alone, so every +// field captured here is a field that survives recreate; the test suite +// pins one rendered command per field. +// +// Known, deliberate limits (documented rather than silent): +// - Network aliases are captured for the primary network only; teploy +// containers are single-network. Additional networks are preserved as +// plain --network flags without their aliases. +// - A multi-element entrypoint that differs from the image's own cannot be +// represented through the docker CLI (which takes a single --entrypoint +// string); recreate fails closed instead of silently mangling it. +type RecreateSpec struct { + Name string `json:"name"` + ImageID string `json:"image_id,omitempty"` // immutable sha256:… (top-level .Image) + ImageRef string `json:"image_ref,omitempty"` // Config.Image original tag reference + Entrypoint []string `json:"entrypoint,omitempty"` + Cmd []string `json:"cmd,omitempty"` + Env []string `json:"env,omitempty"` + WorkingDir string `json:"working_dir,omitempty"` + User string `json:"user,omitempty"` + Labels map[string]string `json:"labels,omitempty"` + NoHealthcheck bool `json:"no_healthcheck,omitempty"` + Networks []string `json:"networks,omitempty"` // non-default networks, primary first + Aliases []string `json:"aliases,omitempty"` // primary network's aliases, sorted + PortBindings []RecreateBinding `json:"port_bindings,omitempty"` + Binds []string `json:"binds,omitempty"` + Mounts []RecreateMount `json:"mounts,omitempty"` + MemoryBytes int64 `json:"memory_bytes,omitempty"` + NanoCPUs int64 `json:"nano_cpus,omitempty"` + RestartPolicy string `json:"restart_policy,omitempty"` + StopTimeout int `json:"stop_timeout,omitempty"` + StopSignal string `json:"stop_signal,omitempty"` + LogDriver string `json:"log_driver,omitempty"` + LogOpts []string `json:"log_opts,omitempty"` // sorted "k=v" + ExtraHosts []string `json:"extra_hosts,omitempty"` + Sysctls map[string]string `json:"sysctls,omitempty"` + Tmpfs map[string]string `json:"tmpfs,omitempty"` + CapAdd []string `json:"cap_add,omitempty"` + CapDrop []string `json:"cap_drop,omitempty"` + SecurityOpt []string `json:"security_opt,omitempty"` + Privileged bool `json:"privileged,omitempty"` + ReadonlyRootfs bool `json:"readonly_rootfs,omitempty"` +} + +// containerInspect mirrors the subset of `docker inspect` JSON used to build +// a RecreateSpec. +type containerInspect struct { + // Image is the container's immutable, content-addressed image ID + // (sha256:…). This — not Config.Image, which is the original (possibly + // mutable tag) reference — is the identity a faithful recreation must + // preserve: after a newer image is pulled under the same tag, + // recreating from Config.Image silently runs the NEW bytes while + // reporting the old version (audit F19). + Image string + Config struct { + Image string + Env []string + Cmd []string + Entrypoint []string + WorkingDir string + User string + Labels map[string]string + StopSignal string + StopTimeout int + Healthcheck *struct { + Test []string + } + } + HostConfig struct { + NetworkMode string + PortBindings map[string][]struct { + HostIp string + HostPort string + } + Binds []string + Mounts []struct { + Type string // "bind" | "volume" | "tmpfs" + Source string + Target string + ReadOnly bool + } + RestartPolicy struct { + Name string + } + Memory int64 // bytes + NanoCpus int64 // nano-CPUs (1 CPU = 1e9) + LogConfig struct { + Type string + Config map[string]string + } + ExtraHosts []string + Sysctls map[string]string + Tmpfs map[string]string + CapAdd []string + CapDrop []string + SecurityOpt []string + Privileged bool + ReadonlyRootfs bool + } + NetworkSettings struct { + Networks map[string]struct { + Aliases []string + } + } +} + +// InspectRecreate reads a container's full inspect JSON and returns its +// RecreateSpec. One round trip; the same command Restart used to issue. +func (c *Client) InspectRecreate(ctx context.Context, name string) (*RecreateSpec, error) { + raw, err := c.exec.Run(ctx, "docker inspect "+ssh.ShellQuote(name)) + if err != nil { + return nil, fmt.Errorf("inspecting %s: %w", name, err) + } + var arr []containerInspect + if err := json.Unmarshal([]byte(raw), &arr); err != nil { + return nil, fmt.Errorf("parsing inspect of %s: %w", name, err) + } + if len(arr) == 0 { + return nil, fmt.Errorf("container %s not found", name) + } + return specFromInspect(name, arr[0]), nil +} + +// specFromInspect is the pure inspect-JSON -> RecreateSpec mapping, split out +// so tests can drive it without an executor. +func specFromInspect(name string, in containerInspect) *RecreateSpec { + spec := &RecreateSpec{ + Name: name, + ImageID: in.Image, + ImageRef: in.Config.Image, + Entrypoint: append([]string(nil), in.Config.Entrypoint...), + Cmd: append([]string(nil), in.Config.Cmd...), + Env: append([]string(nil), in.Config.Env...), + WorkingDir: in.Config.WorkingDir, + User: in.Config.User, + Labels: in.Config.Labels, + StopSignal: in.Config.StopSignal, + StopTimeout: in.Config.StopTimeout, + Binds: append([]string(nil), in.HostConfig.Binds...), + MemoryBytes: in.HostConfig.Memory, + NanoCPUs: in.HostConfig.NanoCpus, + RestartPolicy: in.HostConfig.RestartPolicy.Name, + LogDriver: in.HostConfig.LogConfig.Type, + ExtraHosts: append([]string(nil), in.HostConfig.ExtraHosts...), + Sysctls: in.HostConfig.Sysctls, + Tmpfs: in.HostConfig.Tmpfs, + CapAdd: append([]string(nil), in.HostConfig.CapAdd...), + CapDrop: append([]string(nil), in.HostConfig.CapDrop...), + SecurityOpt: append([]string(nil), in.HostConfig.SecurityOpt...), + Privileged: in.HostConfig.Privileged, + ReadonlyRootfs: in.HostConfig.ReadonlyRootfs, + } + if in.Config.Healthcheck != nil && len(in.Config.Healthcheck.Test) == 1 && in.Config.Healthcheck.Test[0] == "NONE" { + spec.NoHealthcheck = true + } + + // Networks: every non-default network, primary (HostConfig.NetworkMode) + // first, the rest sorted for determinism. + primary := in.HostConfig.NetworkMode + isDefault := func(n string) bool { return n == "" || n == "default" || n == "bridge" } + var others []string + for n := range in.NetworkSettings.Networks { + if !isDefault(n) && n != primary { + others = append(others, n) + } + } + sort.Strings(others) + if !isDefault(primary) { + spec.Networks = append(spec.Networks, primary) + } + spec.Networks = append(spec.Networks, others...) + + // Aliases on the primary network: skip the container-name auto-alias + // (docker adds it itself) and dedupe; sorted for determinism. + if netInfo, ok := in.NetworkSettings.Networks[primary]; ok { + seenAlias := map[string]bool{name: true} + aliases := append([]string(nil), netInfo.Aliases...) + sort.Strings(aliases) + for _, alias := range aliases { + if alias == "" || seenAlias[alias] { + continue + } + seenAlias[alias] = true + spec.Aliases = append(spec.Aliases, alias) + } + } + + // Port bindings, normalized; sorted by container port then host port for + // deterministic args ordering. + for containerPort, bindings := range in.HostConfig.PortBindings { + portStr, proto, _ := strings.Cut(containerPort, "/") + cp := atoiOrZero(portStr) + for _, b := range bindings { + hostIP := b.HostIp + if hostIP == "" { + hostIP = "0.0.0.0" + } + spec.PortBindings = append(spec.PortBindings, RecreateBinding{ + HostIP: hostIP, + HostPort: atoiOrZero(b.HostPort), + ContainerPort: cp, + Proto: proto, + }) + } + } + sort.Slice(spec.PortBindings, func(i, j int) bool { + if spec.PortBindings[i].ContainerPort != spec.PortBindings[j].ContainerPort { + return spec.PortBindings[i].ContainerPort < spec.PortBindings[j].ContainerPort + } + return spec.PortBindings[i].HostPort < spec.PortBindings[j].HostPort + }) + + for _, m := range in.HostConfig.Mounts { + if m.Type == "" || m.Target == "" { + continue + } + spec.Mounts = append(spec.Mounts, RecreateMount{Type: m.Type, Source: m.Source, Target: m.Target, ReadOnly: m.ReadOnly}) + } + + for k, v := range in.HostConfig.LogConfig.Config { + spec.LogOpts = append(spec.LogOpts, k+"="+v) + } + sort.Strings(spec.LogOpts) + + return spec +} + +func atoiOrZero(s string) int { + n, _ := strconv.Atoi(strings.TrimSpace(s)) + return n +} + +// RenderRecreateArgs renders the docker run arguments (excluding the leading +// "docker run") for the spec. Pure; every value is quoted exactly once at +// this shell boundary (inspect-derived values are DATA, not shell syntax — +// audit TCL-11). avoidPorts handling lives in Recreate, which adjusts the +// spec's host bindings before rendering. +func RenderRecreateArgs(spec *RecreateSpec) ([]string, error) { + q := ssh.ShellQuote + args := []string{"-d", "--name", q(spec.Name)} + + for _, n := range spec.Networks { + args = append(args, "--network", q(n)) + } + for _, alias := range spec.Aliases { + args = append(args, "--network-alias", q(alias)) + } + + for _, b := range spec.PortBindings { + containerPort := strconv.Itoa(b.ContainerPort) + if b.Proto != "" { + containerPort += "/" + b.Proto + } + hostPort := "" + if b.HostPort > 0 { + hostPort = strconv.Itoa(b.HostPort) + } + args = append(args, "-p", q(b.HostIP+":"+hostPort+":"+containerPort)) + } + + for _, e := range spec.Env { + args = append(args, "-e", q(e)) + } + + for _, b := range spec.Binds { + args = append(args, "-v", q(b)) + } + + for _, m := range spec.Mounts { + parts := []string{"type=" + m.Type, "target=" + m.Target} + if m.Source != "" { + parts = append(parts, "source="+m.Source) + } + if m.ReadOnly { + parts = append(parts, "readonly") + } + args = append(args, "--mount", q(strings.Join(parts, ","))) + } + + if spec.MemoryBytes > 0 { + args = append(args, "--memory", fmt.Sprintf("%db", spec.MemoryBytes)) + } + if spec.NanoCPUs > 0 { + // NanoCpus → fractional --cpus (1e9 nano = 1.0 cpu). + cpus := float64(spec.NanoCPUs) / 1e9 + args = append(args, "--cpus", strconv.FormatFloat(cpus, 'f', -1, 64)) + } + + if len(spec.Labels) > 0 { + labelKeys := make([]string, 0, len(spec.Labels)) + for k := range spec.Labels { + labelKeys = append(labelKeys, k) + } + sort.Strings(labelKeys) + for _, k := range labelKeys { + args = append(args, "--label", q(k+"="+spec.Labels[k])) + } + } + + if spec.NoHealthcheck { + args = append(args, "--no-healthcheck") + } + + if spec.RestartPolicy != "" && spec.RestartPolicy != "no" { + args = append(args, "--restart", q(spec.RestartPolicy)) + } + + if spec.WorkingDir != "" { + args = append(args, "-w", q(spec.WorkingDir)) + } + if spec.User != "" { + args = append(args, "-u", q(spec.User)) + } + + // Entry point: only a single element is representable through the docker + // CLI. The common cases are "no override" (empty) and a one-string + // override; a multi-element entrypoint is usually just the image's own, + // which the recreated container inherits from the (immutable) image + // anyway — Recreate verifies that before calling here. + switch { + case len(spec.Entrypoint) == 1: + args = append(args, "--entrypoint", q(spec.Entrypoint[0])) + case len(spec.Entrypoint) > 1: + return nil, fmt.Errorf("container %s has a multi-element entrypoint (%q) that the docker CLI cannot represent; verify it matches the image's own before recreating", spec.Name, spec.Entrypoint) + } + + if spec.StopTimeout > 0 { + args = append(args, "--stop-timeout", strconv.Itoa(spec.StopTimeout)) + } + if spec.StopSignal != "" { + args = append(args, "--stop-signal", q(spec.StopSignal)) + } + + if spec.LogDriver != "" && spec.LogDriver != "json-file" { + args = append(args, "--log-driver", q(spec.LogDriver)) + } + for _, opt := range spec.LogOpts { + args = append(args, "--log-opt", q(opt)) + } + + for _, h := range spec.ExtraHosts { + args = append(args, "--add-host", q(h)) + } + if len(spec.Sysctls) > 0 { + sysctlKeys := make([]string, 0, len(spec.Sysctls)) + for k := range spec.Sysctls { + sysctlKeys = append(sysctlKeys, k) + } + sort.Strings(sysctlKeys) + for _, k := range sysctlKeys { + args = append(args, "--sysctl", q(k+"="+spec.Sysctls[k])) + } + } + if len(spec.Tmpfs) > 0 { + tmpfsPaths := make([]string, 0, len(spec.Tmpfs)) + for k := range spec.Tmpfs { + tmpfsPaths = append(tmpfsPaths, k) + } + sort.Strings(tmpfsPaths) + for _, p := range tmpfsPaths { + mount := p + if opts := spec.Tmpfs[p]; opts != "" { + mount += ":" + opts + } + args = append(args, "--tmpfs", q(mount)) + } + } + for _, cap := range spec.CapAdd { + args = append(args, "--cap-add", q(cap)) + } + for _, cap := range spec.CapDrop { + args = append(args, "--cap-drop", q(cap)) + } + for _, opt := range spec.SecurityOpt { + args = append(args, "--security-opt", q(opt)) + } + if spec.Privileged { + args = append(args, "--privileged") + } + if spec.ReadonlyRootfs { + args = append(args, "--read-only") + } + + // Image (last positional before cmd). Prefer the container's immutable + // top-level image ID over the Config.Image tag reference so a tag that + // has moved to newer bytes cannot silently change what the recreated + // container runs (F19). Config.Image is kept only as a fallback for + // odd daemons that report no top-level ID. + imageRef := spec.ImageID + if imageRef == "" || !strings.HasPrefix(imageRef, "sha256:") { + imageRef = spec.ImageRef + } + args = append(args, q(imageRef)) + + for _, cmdPart := range spec.Cmd { + args = append(args, q(cmdPart)) + } + + return args, nil +} + +// Recreate force-removes the named container and runs a fresh one from the +// spec. avoidPorts is the set of host ports currently held by containers +// this recreation must not collide with; a binding whose original port is +// in the set gets a freshly allocated one (see Restart's doc comment for the +// rollback collision this exists for). +func (c *Client) Recreate(ctx context.Context, spec *RecreateSpec, avoidPorts map[int]bool) error { + if spec == nil || spec.Name == "" { + return fmt.Errorf("recreate requires a spec with a container name") + } + + claimed := make(map[int]bool, len(avoidPorts)) + for p := range avoidPorts { + claimed[p] = true + } + for i, b := range spec.PortBindings { + if b.HostPort > 0 && claimed[b.HostPort] { + newPort, err := c.FindAvailablePortExcluding(ctx, claimed) + if err != nil { + return fmt.Errorf("port %d for %s is held by another running container and no replacement port is available: %w", b.HostPort, spec.Name, err) + } + spec.PortBindings[i].HostPort = newPort + claimed[newPort] = true + } + } + + // A multi-element entrypoint cannot be rendered. Before failing, check + // whether it is simply the image's own (the overwhelmingly common case + // for containers created without --entrypoint): the recreated container + // inherits it from the immutable image ID, so nothing is lost by + // dropping it from the rendered command. + if len(spec.Entrypoint) > 1 { + same, err := c.entrypointMatchesImage(ctx, spec) + if err != nil { + return err + } + if same { + spec.Entrypoint = nil + } + } + + args, err := RenderRecreateArgs(spec) + if err != nil { + return err + } + + if _, err := c.exec.Run(ctx, "docker rm -f "+ssh.ShellQuote(spec.Name)); err != nil { + return fmt.Errorf("removing old %s: %w", spec.Name, err) + } + if _, err := c.exec.Run(ctx, "docker run "+strings.Join(args, " ")); err != nil { + return fmt.Errorf("recreating %s: %w", spec.Name, err) + } + return nil +} + +// entrypointMatchesImage reports whether the container's multi-element +// entrypoint equals its image's own entrypoint, comparing the JSON arrays. +func (c *Client) entrypointMatchesImage(ctx context.Context, spec *RecreateSpec) (bool, error) { + image := spec.ImageID + if image == "" || !strings.HasPrefix(image, "sha256:") { + image = spec.ImageRef + } + if image == "" { + return false, nil + } + out, err := c.exec.Run(ctx, "docker image inspect -f '{{json .Config.Entrypoint}}' "+ssh.ShellQuote(image)) + if err != nil { + // Unreadable image (pruned while the container lingered): cannot + // prove equality; fail closed. + return false, fmt.Errorf("comparing entrypoint against image %s: %w", image, err) + } + var imageEntry []string + if err := json.Unmarshal([]byte(strings.TrimSpace(out)), &imageEntry); err != nil { + return false, fmt.Errorf("parsing image entrypoint: %w", err) + } + if len(imageEntry) != len(spec.Entrypoint) { + return false, nil + } + for i := range imageEntry { + if imageEntry[i] != spec.Entrypoint[i] { + return false, nil + } + } + return true, nil +} + +// Restart fully recreates a container with its original configuration: +// inspect the existing container, force-remove it, then `docker run` with +// the same name + extracted args. +// +// Use this in rollback flows where plain Start() is insufficient. Docker +// 29 silently fails to re-publish HostConfig.PortBindings if another +// container has taken + released the host port since the original stop +// (the bind also detaches from custom networks). Recreating from scratch +// sidesteps that landmine. +// +// avoidPorts is the set of host ports currently held by containers this +// restart must not collide with — critically, the live container(s) being +// rolled back FROM, which are still running (and still holding their port) +// at the point Restart is called, since rollback starts the target before +// stopping the current version for zero-downtime. A single-hop rollback +// (to the immediately-previous version) can never collide: that version's +// port was freed by the deploy that superseded it and nothing has claimed +// it since. A --to rollback reaching back further can: Docker's ephemeral +// port allocator reuses freed ports, so an older version's original port +// may since have been handed to what is now the live container. Found via +// live testing (v1→49152, v1-tls→49153, v2 reused 49152 after v1-tls +// stopped; `rollback --to v1` then collided with the still-running v2). +// When a binding's original port is in avoidPorts, a fresh one is +// allocated instead — safe because HostPort() (not persisted state) is +// what the rest of the rollback reads back afterward. Pass nil to +// preserve every port binding exactly. +// +// Preserved across the recreate (audit F20): image (immutable ID), network +// mode + primary-network aliases + additional networks, port bindings +// (subject to the above), env, bind mounts, named-volume + tmpfs + --mount +// mounts, command, working dir, user, labels, restart policy, memory + CPU +// limits, the --no-healthcheck NONE marker, single-element entrypoint +// overrides, stop timeout + signal, log driver + options, extra hosts, +// sysctls, tmpfs mounts, capabilities, security options, privileged and +// read-only flags. +func (c *Client) Restart(ctx context.Context, name string, avoidPorts map[int]bool) error { + spec, err := c.InspectRecreate(ctx, name) + if err != nil { + return err + } + return c.Recreate(ctx, spec, avoidPorts) +} + +// HostPortFor returns the host port bound to the given CONTAINER port — the +// primary-port-aware form of HostPort (audit TCL-14). A multi-port +// container's "first" binding is map-order luck; a caller that knows the +// release's primary container port (from the F14 record) gets the port the +// health check should actually probe. +func (c *Client) HostPortFor(ctx context.Context, name string, containerPort int) (int, error) { + out, err := c.exec.Run(ctx, fmt.Sprintf( + "docker inspect -f '{{json .NetworkSettings.Ports}}' %s", ssh.ShellQuote(name))) + if err != nil { + return 0, fmt.Errorf("inspecting container %s ports: %w", name, err) + } + var ports map[string][]struct { + HostIp string `json:"HostIp"` + HostPort string `json:"HostPort"` + } + if err := json.Unmarshal([]byte(strings.TrimSpace(out)), &ports); err != nil { + return 0, fmt.Errorf("parsing ports of container %s: %w", name, err) + } + // docker keys the map "/"; try tcp first, then any proto. + for _, key := range []string{strconv.Itoa(containerPort) + "/tcp", strconv.Itoa(containerPort)} { + for _, b := range ports[key] { + if p, err := strconv.Atoi(strings.TrimSpace(b.HostPort)); err == nil && p > 0 { + return p, nil + } + } + } + return 0, fmt.Errorf("container %s has no host binding for container port %d", name, containerPort) +} diff --git a/internal/docker/recreate_test.go b/internal/docker/recreate_test.go new file mode 100644 index 0000000..250b11b --- /dev/null +++ b/internal/docker/recreate_test.go @@ -0,0 +1,325 @@ +package docker + +import ( + "context" + "fmt" + "strings" + "testing" + + "github.com/useteploy/teploy/internal/ssh" +) + +// richInspect is a container inspect fixture exercising every field the +// RecreateSpec must preserve (audit F20). Fields that older code dropped — +// entrypoint override, stop timeout/signal, log config, extra hosts, +// sysctls, tmpfs, capabilities, security opts, privileged/read-only, a +// second network — are all present. +func richInspect(name string) string { + return fmt.Sprintf(`[{ + "Image": "sha256:%s", + "Config": { + "Image": "myapp:v9", + "Env": ["PORT=3000", "TOKEN=sec;ret"], + "Cmd": ["npm", "start"], + "Entrypoint": ["./entrypoint.sh"], + "WorkingDir": "/app dir", + "User": "1000:1000", + "Labels": {"teploy.app": "myapp", "teploy.version": "v9"}, + "StopSignal": "SIGQUIT", + "StopTimeout": 25, + "Healthcheck": {"Test": ["NONE"]} + }, + "HostConfig": { + "NetworkMode": "teploy", + "PortBindings": { + "3000/tcp": [{"HostIp": "127.0.0.1", "HostPort": "49152"}], + "51820/udp": [{"HostIp": "0.0.0.0", "HostPort": "51820"}] + }, + "Binds": ["/deployments/myapp/volumes/data:/data:ro"], + "Mounts": [{"Type": "volume", "Source": "myapp-uploads", "Target": "/uploads"}], + "RestartPolicy": {"Name": "unless-stopped"}, + "Memory": 536870912, + "NanoCpus": 1500000000, + "LogConfig": {"Type": "json-file", "Config": {"max-size": "10m"}}, + "ExtraHosts": ["host.docker.internal:host-gateway"], + "Sysctls": {"net.core.somaxconn": "1024"}, + "Tmpfs": {"/scratch": "size=64m"}, + "CapAdd": ["NET_ADMIN"], + "CapDrop": ["CHOWN"], + "SecurityOpt": ["no-new-privileges"], + "Privileged": false, + "ReadonlyRootfs": true + }, + "NetworkSettings": { + "Networks": { + "teploy": {"Aliases": ["myapp", "extra-alias"]}, + "mesh": {"Aliases": ["myapp-mesh"]} + } + } +}]`, strings.Repeat("a", 64)) +} + +func TestInspectRecreate_CapturesEveryPreservedField(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect 'myapp-web-v9'", Output: richInspect("myapp-web-v9")}, + ) + spec, err := NewClient(mock).InspectRecreate(context.Background(), "myapp-web-v9") + if err != nil { + t.Fatalf("InspectRecreate: %v", err) + } + + if spec.ImageID != "sha256:"+strings.Repeat("a", 64) || spec.ImageRef != "myapp:v9" { + t.Errorf("image identity not captured: %+v", spec) + } + if len(spec.Entrypoint) != 1 || spec.Entrypoint[0] != "./entrypoint.sh" { + t.Errorf("entrypoint not captured: %v", spec.Entrypoint) + } + if spec.StopTimeout != 25 || spec.StopSignal != "SIGQUIT" { + t.Errorf("stop settings not captured: %d/%q", spec.StopTimeout, spec.StopSignal) + } + if spec.LogDriver != "json-file" || len(spec.LogOpts) != 1 || spec.LogOpts[0] != "max-size=10m" { + t.Errorf("log config not captured: %s %v", spec.LogDriver, spec.LogOpts) + } + if len(spec.ExtraHosts) != 1 || spec.ExtraHosts[0] != "host.docker.internal:host-gateway" { + t.Errorf("extra hosts not captured: %v", spec.ExtraHosts) + } + if spec.Sysctls["net.core.somaxconn"] != "1024" { + t.Errorf("sysctls not captured: %v", spec.Sysctls) + } + if spec.Tmpfs["/scratch"] != "size=64m" { + t.Errorf("tmpfs not captured: %v", spec.Tmpfs) + } + if len(spec.CapAdd) != 1 || spec.CapAdd[0] != "NET_ADMIN" || len(spec.CapDrop) != 1 || spec.CapDrop[0] != "CHOWN" { + t.Errorf("capabilities not captured: %v %v", spec.CapAdd, spec.CapDrop) + } + if len(spec.SecurityOpt) != 1 || spec.SecurityOpt[0] != "no-new-privileges" { + t.Errorf("security opts not captured: %v", spec.SecurityOpt) + } + if !spec.ReadonlyRootfs || spec.Privileged { + t.Errorf("privileged/read-only not captured: %v/%v", spec.Privileged, spec.ReadonlyRootfs) + } + if len(spec.Networks) != 2 || spec.Networks[0] != "teploy" || spec.Networks[1] != "mesh" { + t.Errorf("networks not captured (primary first): %v", spec.Networks) + } + // Aliases: primary network only, name auto-alias skipped, sorted. + if len(spec.Aliases) != 2 || spec.Aliases[0] != "extra-alias" || spec.Aliases[1] != "myapp" { + t.Errorf("aliases not captured as expected: %v", spec.Aliases) + } + if len(spec.PortBindings) != 2 { + t.Fatalf("bindings not captured: %+v", spec.PortBindings) + } + udp := spec.PortBindings[1] + if udp.ContainerPort != 51820 || udp.Proto != "udp" || udp.HostPort != 51820 { + t.Errorf("udp binding not normalized: %+v", udp) + } + tcp := spec.PortBindings[0] + if tcp.ContainerPort != 3000 || tcp.Proto != "tcp" || tcp.HostPort != 49152 || tcp.HostIP != "127.0.0.1" { + t.Errorf("tcp binding not normalized: %+v", tcp) + } + if !spec.NoHealthcheck { + t.Error("explicit-NONE healthcheck marker not captured") + } +} + +// TestRecreate_RendersEveryPreservedField is the F20 regression pin: the +// rendered docker run must carry every inspect field the spec captured. +// Each field that silently disappears from this command after a refactor +// is a field recreation stopped preserving. +func TestRecreate_RendersEveryPreservedField(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect 'myapp-web-v9'", Output: richInspect("myapp-web-v9")}, + ssh.MockCommand{Match: "docker rm -f", Output: ""}, + ssh.MockCommand{Match: "docker run", Output: ""}, + ) + if err := NewClient(mock).Restart(context.Background(), "myapp-web-v9", nil); err != nil { + t.Fatalf("Restart: %v", err) + } + + var run string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "docker run ") { + run = c + } + } + if run == "" { + t.Fatal("no docker run issued") + } + + for _, want := range []string{ + "--name 'myapp-web-v9'", + "--network 'teploy'", + "--network 'mesh'", + "--network-alias 'extra-alias'", + "--network-alias 'myapp'", + "-p '127.0.0.1:49152:3000/tcp'", + "-p '0.0.0.0:51820:51820/udp'", + "-e 'PORT=3000'", + "-e 'TOKEN=sec;ret'", + "-v '/deployments/myapp/volumes/data:/data:ro'", + "--mount 'type=volume,target=/uploads,source=myapp-uploads'", + "--memory 536870912b", + "--cpus 1.5", + "--label 'teploy.app=myapp'", + "--label 'teploy.version=v9'", + "--no-healthcheck", + "--restart 'unless-stopped'", + "-w '/app dir'", + "-u '1000:1000'", + "--entrypoint './entrypoint.sh'", + "--stop-timeout 25", + "--stop-signal 'SIGQUIT'", + "--log-opt 'max-size=10m'", + "--add-host 'host.docker.internal:host-gateway'", + "--sysctl 'net.core.somaxconn=1024'", + "--tmpfs '/scratch:size=64m'", + "--cap-add 'NET_ADMIN'", + "--cap-drop 'CHOWN'", + "--security-opt 'no-new-privileges'", + "--read-only", + "'sha256:" + strings.Repeat("a", 64) + "'", + "'npm' 'start'", + } { + if !strings.Contains(run, want) { + t.Errorf("rendered run is missing %s\n run: %s", want, run) + } + } + if strings.Contains(run, "--privileged") { + t.Errorf("privileged rendered for an unprivileged container: %s", run) + } +} + +// A multi-element entrypoint that MATCHES the image's own must be dropped +// from the command (the recreated container inherits it from the immutable +// image), while one that DIFFERS cannot be represented through the CLI and +// must fail closed instead of silently mangling it. +func TestRecreate_MultiElementEntrypoint(t *testing.T) { + inspect := fmt.Sprintf(`[{ + "Image": "sha256:%s", + "Config": {"Image": "myapp:v9", "Entrypoint": ["docker-entrypoint.sh", "serve"], "Labels": {}}, + "HostConfig": {"PortBindings": {}, "RestartPolicy": {}}, + "NetworkSettings": {"Networks": {"teploy": {"Aliases": ["myapp"]}}} +}]`, strings.Repeat("b", 64)) + + t.Run("matches image", func(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect 'c'", Output: inspect}, + ssh.MockCommand{Match: "docker image inspect", Output: `["docker-entrypoint.sh","serve"]`}, + ssh.MockCommand{Match: "docker rm -f", Output: ""}, + ssh.MockCommand{Match: "docker run", Output: ""}, + ) + if err := NewClient(mock).Restart(context.Background(), "c", nil); err != nil { + t.Fatalf("Restart with image-own entrypoint: %v", err) + } + var run string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "docker run ") { + run = c + } + } + if run == "" { + t.Fatal("no docker run issued") + } + if strings.Contains(run, "--entrypoint") { + t.Errorf("image-own entrypoint was rendered as an override: %s", run) + } + }) + + t.Run("differs from image", func(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect 'c'", Output: inspect}, + ssh.MockCommand{Match: "docker image inspect", Output: `["other-entrypoint"]`}, + ) + err := NewClient(mock).Restart(context.Background(), "c", nil) + if err == nil || !strings.Contains(err.Error(), "multi-element entrypoint") { + t.Fatalf("expected fail-closed multi-element entrypoint error, got %v", err) + } + for _, c := range mock.Calls { + if strings.HasPrefix(c, "docker rm -f") { + t.Fatalf("container was removed despite unrepresentable spec: %s", c) + } + } + }) +} + +// The recreated container keeps the immutable image ID even when the tag has +// moved (F19's rule, retained through the RecreateSpec refactor). +func TestRecreate_PrefersImmutableImageID(t *testing.T) { + inspect := fmt.Sprintf(`[{ + "Image": "sha256:%s", + "Config": {"Image": "myapp:latest", "Labels": {}}, + "HostConfig": {"PortBindings": {}, "RestartPolicy": {}}, + "NetworkSettings": {"Networks": {}} +}]`, strings.Repeat("c", 64)) + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect 'c'", Output: inspect}, + ssh.MockCommand{Match: "docker rm -f", Output: ""}, + ssh.MockCommand{Match: "docker run", Output: ""}, + ) + if err := NewClient(mock).Restart(context.Background(), "c", nil); err != nil { + t.Fatalf("Restart: %v", err) + } + var run string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "docker run ") { + run = c + } + } + if !strings.Contains(run, "'sha256:"+strings.Repeat("c", 64)+"'") { + t.Errorf("recreate did not pin the immutable image ID: %s", run) + } + if strings.Contains(run, "myapp:latest") { + t.Errorf("recreate used the mutable tag reference: %s", run) + } +} + +// HostPortFor resolves the host binding for a specific CONTAINER port — the +// primary-port lookup a fields[0] read cannot do for multi-port containers +// (TCL-14). +func TestHostPortFor_SelectsTheNamedContainerPort(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect -f '{{json .NetworkSettings.Ports}}'", + Output: `{"3000/tcp":[{"HostIp":"127.0.0.1","HostPort":"49152"}],"51820/udp":[{"HostIp":"0.0.0.0","HostPort":"51820"}]}`}, + ) + port, err := NewClient(mock).HostPortFor(context.Background(), "c", 3000) + if err != nil { + t.Fatalf("HostPortFor: %v", err) + } + if port != 49152 { + t.Errorf("HostPortFor(3000) = %d, want 49152", port) + } + if _, err := NewClient(mock).HostPortFor(context.Background(), "c", 9999); err == nil { + t.Error("expected an error for an unbound container port") + } +} + +// The avoidPorts collision reallocation (the --to rollback landmine) must +// survive the RecreateSpec refactor intact. +func TestRecreate_AvoidPortsReallocatesCollidingBinding(t *testing.T) { + inspect := fmt.Sprintf(`[{ + "Image": "sha256:%s", + "Config": {"Image": "myapp:latest", "Labels": {}}, + "HostConfig": {"PortBindings": {"3000/tcp":[{"HostIp":"127.0.0.1","HostPort":"49152"}]}, "RestartPolicy": {}}, + "NetworkSettings": {"Networks": {}} +}]`, strings.Repeat("d", 64)) + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker inspect 'c'", Output: inspect}, + ssh.MockCommand{Match: "ss -tln", Output: "LISTEN 0 128 0.0.0.0:49152 0.0.0.0:*"}, + ssh.MockCommand{Match: "docker rm -f", Output: ""}, + ssh.MockCommand{Match: "docker run", Output: ""}, + ) + if err := NewClient(mock).Restart(context.Background(), "c", map[int]bool{49152: true}); err != nil { + t.Fatalf("Restart: %v", err) + } + var run string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "docker run ") { + run = c + } + } + if strings.Contains(run, ":49152:") { + t.Errorf("colliding port was not reallocated: %s", run) + } + if !strings.Contains(run, "-p '127.0.0.1:49153:3000/tcp'") { + t.Errorf("expected reallocation to the next free port 49153: %s", run) + } +} From 97e6d03b19e8eacd98966374b1206f949710dcff Mon Sep 17 00:00:00 2001 From: Tyler <53561637+im-tyler@users.noreply.github.com> Date: Fri, 18 Sep 2026 04:01:49 -0700 Subject: [PATCH 2/4] feat(releasemeta): F14 per-release immutable metadata store keyed target+release MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One JSON record per (app, release id) at /deployments//meta/.json, written atomically at 0600 (backfills embed resolved env), following the file-per-target convention of state.json and pins: the server a command talks to IS the target, so the app directory is the target key. The record captures the full resolved spec a rollback or recreate needs and state.json never retained: image ref + digest, env references (server-side env-file paths; resolved values only for backfills), volumes, labels, ports with the TCL-14 primary designation and a fixed/ephemeral flag, replicas/processes/cmd, resource limits, stop timeout, bind address, the deploy health gate, the Caddy edge config (TLS/extra/cache/firewall/access), and for static apps the serving config (SPA/headers/cache). The primary web container's full RecreateSpec (F20) is embedded from docker's own view. Writers are the deploy paths only; readers never mutate. A same-version redeploy is the one allowed rewrite (the live containers just got that config — an unchanged record would lie). Rollback never writes for its target: a release does not change by being rolled back to. Migration: pre-F14 releases get a 'release-0' record backfilled from the live containers the first time a consumer needs one (flagged Backfilled) — existing installs converge without a redeploy. Backfill is honest about what inspect cannot recover (--env-file references, health gate, Caddy config) and leaves those empty for the legacy fallbacks. Read keeps the F15 discipline: confirmed-absent is (nil, nil); transport/parse/schema failures are errors. state.ReadRemoteFile exported for the same framed absence-vs-failure read. --- internal/releasemeta/releasemeta.go | 448 +++++++++++++++++++++++ internal/releasemeta/releasemeta_test.go | 328 +++++++++++++++++ internal/state/pins.go | 2 +- internal/state/state.go | 11 +- 4 files changed, 783 insertions(+), 6 deletions(-) create mode 100644 internal/releasemeta/releasemeta.go create mode 100644 internal/releasemeta/releasemeta_test.go diff --git a/internal/releasemeta/releasemeta.go b/internal/releasemeta/releasemeta.go new file mode 100644 index 0000000..378526f --- /dev/null +++ b/internal/releasemeta/releasemeta.go @@ -0,0 +1,448 @@ +// Package releasemeta stores the immutable per-release execution record +// (audit F14): the full resolved spec of every release this CLI deployed to +// a target, keyed by (app, release id). +// +// Design: +// +// - One JSON file per release at /deployments//meta/.json, +// written atomically (sibling temp + rename) with mode 0600. Backfilled +// records carry the container's RESOLVED environment, which docker +// inspect reveals on the same host anyway, but which has no business +// being world-readable on disk. +// - The store is per-target (per-server, per-app directory), exactly like +// state.json and pins: the server a command talks to IS the target; +// there is no client-side database to key otherwise. +// - Writers: successful deploys (container and static). Readers: rollback, +// state-only rollback, recreate. Rollback never writes a record for its +// target — the record describes the release, and a release does not +// change by being rolled back to. +// - Immutability: readers never mutate. The only writer allowed to change +// an existing record is a NEW deploy of the same release id (a +// same-version redeploy with changed config: the live containers were +// just recreated with that config, so the record follows or it lies). +// - Migration: releases deployed before this store existed have no record. +// The first consumer that needs one backfills it from the live +// containers (a "release-0" record, flagged Backfilled) so existing +// installs converge without a redeploy. Backfill is best-effort: fields +// docker cannot report (--env-file references, the teploy health/caddy +// config) stay empty and callers keep their legacy fallbacks for them. +package releasemeta + +import ( + "context" + "encoding/json" + "fmt" + "regexp" + "sort" + "strconv" + "strings" + "time" + + "github.com/useteploy/teploy/internal/caddy" + "github.com/useteploy/teploy/internal/docker" + "github.com/useteploy/teploy/internal/ssh" + "github.com/useteploy/teploy/internal/state" +) + +// SchemaVersion is the record schema this CLI writes and accepts. +const SchemaVersion = 1 + +// deploymentsDir mirrors state.DefaultStateDir without importing a cycle +// through the deploy package. +const deploymentsDir = "/deployments" + +// validHash matches release ids safe to use as a meta file-name segment: +// the same grammar the CLI already accepts for versions (imageTagPattern in +// internal/cli) — git short hashes, tags like v1.2.3, and image-derived +// sha256- labels all pass; path metacharacters do not. +var validHash = regexp.MustCompile(`^[A-Za-z0-9_][A-Za-z0-9._-]{0,127}$`) + +// Port is one published port of a release, with the primary designation that +// disambiguates multi-port containers (audit TCL-14): health checks probe +// the primary's host binding and Caddy dials the primary's container port. +type Port struct { + HostPort int `json:"host_port,omitempty"` + ContainerPort int `json:"container_port"` + Proto string `json:"proto,omitempty"` // "" and "tcp" are the same + Bind string `json:"bind,omitempty"` // host IP the port is published on + Primary bool `json:"primary,omitempty"` + // Fixed marks an operator-configured host port (publish entries, host + // ingress): a port two containers can never hold at once, which is what + // makes the deploy/rollback recreate strategy apply (audit F21). + Fixed bool `json:"fixed,omitempty"` +} + +// Health records the deploy-time health gate so a rollback probes the path +// the target release was actually deployed with, not whatever the current +// teploy.yml says. +type Health struct { + Path string `json:"path,omitempty"` + TimeoutSeconds int `json:"timeout_seconds,omitempty"` + IntervalSeconds int `json:"interval_seconds,omitempty"` +} + +// CaddyRoute records the Caddy-facing options a release was deployed with, +// so rollback restores the target release's edge config instead of the +// current config file's. +type CaddyRoute struct { + TLSCert string `json:"tls_cert,omitempty"` + TLSKey string `json:"tls_key,omitempty"` + TLSInternal bool `json:"tls_internal,omitempty"` + CaddyExtra string `json:"caddy_extra,omitempty"` + Cache map[string]string `json:"cache,omitempty"` + Firewall *caddy.Firewall `json:"firewall,omitempty"` + Access *caddy.Access `json:"access,omitempty"` +} + +// Static records a type:static release's serving configuration — the piece +// state.json never retained, which is what made state-only static rollback +// impossible before F13/F14. +type Static struct { + Domain string `json:"domain,omitempty"` + SPA bool `json:"spa,omitempty"` + SPAFallback string `json:"spa_fallback,omitempty"` + Cache map[string]string `json:"cache,omitempty"` + Headers map[string]string `json:"headers,omitempty"` + CaddyExtra string `json:"caddy_extra,omitempty"` + KeepReleases int `json:"keep_releases,omitempty"` +} + +// Record is the immutable per-release execution record. +type Record struct { + SchemaVersion int `json:"schema_version"` + App string `json:"app"` + Hash string `json:"hash"` + CreatedAt time.Time `json:"created_at"` + // Backfilled marks a "release-0" record synthesized from live containers + // rather than observed at deploy time (see the package doc). + Backfilled bool `json:"backfilled,omitempty"` + Generation uint64 `json:"generation,omitempty"` + + DeploymentType string `json:"deployment_type"` // container | static + IngressMode string `json:"ingress_mode"` + Domain string `json:"domain,omitempty"` + + ImageRef string `json:"image_ref,omitempty"` + ImageDigest string `json:"image_digest,omitempty"` + + Replicas int `json:"replicas,omitempty"` + Processes map[string]string `json:"processes,omitempty"` + Cmd string `json:"cmd,omitempty"` + Env map[string]string `json:"env,omitempty"` + EnvFiles []string `json:"env_files,omitempty"` + Volumes map[string]string `json:"volumes,omitempty"` + Publish []string `json:"publish,omitempty"` + Ports []Port `json:"ports,omitempty"` + Labels map[string]string `json:"labels,omitempty"` + Memory string `json:"memory,omitempty"` + CPU string `json:"cpu,omitempty"` + StopTimeout int `json:"stop_timeout,omitempty"` + Bind string `json:"bind,omitempty"` + + Health *Health `json:"health,omitempty"` + Caddy *CaddyRoute `json:"caddy,omitempty"` + Static *Static `json:"static,omitempty"` + + // Recreate is the primary web container's full inspect-derived recreation + // spec (audit F20), captured from docker's own view of the container the + // deploy started. + Recreate *docker.RecreateSpec `json:"recreate,omitempty"` +} + +// Path returns the record path for (app, hash). Both segments are grammar +// checked here so no caller can interpolate an unvalidated id into a remote +// path. +func Path(app, hash string) (string, error) { + if app == "" || !validHash.MatchString(hash) { + return "", fmt.Errorf("invalid release id %q for app %q", hash, app) + } + return fmt.Sprintf("%s/%s/meta/%s.json", deploymentsDir, app, hash), nil +} + +// Read loads the record for (app, hash). A confirmed-missing record returns +// (nil, nil); every other failure (transport, permission, malformed JSON, +// wrong schema) is an error — callers that can proceed without the record +// degrade explicitly with a warning, callers that cannot fail closed. +func Read(ctx context.Context, exec ssh.Executor, app, hash string) (*Record, error) { + path, err := Path(app, hash) + if err != nil { + return nil, err + } + data, present, err := state.ReadRemoteFile(ctx, exec, path) + if err != nil { + return nil, fmt.Errorf("reading release metadata for %s@%s: %w", app, hash, err) + } + if !present { + return nil, nil + } + var rec Record + if err := json.Unmarshal(data, &rec); err != nil { + return nil, fmt.Errorf("parsing release metadata for %s@%s: %w", app, hash, err) + } + if rec.SchemaVersion != SchemaVersion { + return nil, fmt.Errorf("unsupported release-metadata schema version %d for %s@%s", rec.SchemaVersion, app, hash) + } + return &rec, nil +} + +// Write persists the record atomically. It is the deploy paths' writer: an +// existing record for the same release id is replaced (same-version +// redeploy — see the package doc); readers never call this. +func Write(ctx context.Context, exec ssh.Executor, rec *Record) error { + if rec == nil { + return fmt.Errorf("record is required") + } + if rec.SchemaVersion == 0 { + rec.SchemaVersion = SchemaVersion + } + if rec.SchemaVersion != SchemaVersion { + return fmt.Errorf("cannot write release-metadata schema version %d", rec.SchemaVersion) + } + if rec.App == "" || rec.Hash == "" { + return fmt.Errorf("record requires app and hash") + } + if rec.CreatedAt.IsZero() { + rec.CreatedAt = time.Now().UTC() + } + path, err := Path(rec.App, rec.Hash) + if err != nil { + return err + } + if _, err := exec.Run(ctx, fmt.Sprintf("mkdir -p %s/%s/meta", deploymentsDir, rec.App)); err != nil { + return fmt.Errorf("creating metadata directory: %w", err) + } + data, err := json.Marshal(rec) + if err != nil { + return fmt.Errorf("marshaling release metadata: %w", err) + } + data = append(data, '\n') + // 0600: backfilled records embed resolved env; docker inspect shows the + // same values to someone already on the host, but the file must not + // make them readable to everyone with a shell account. + return ssh.UploadAtomic(ctx, exec, strings.NewReader(string(data)), path, "0600") +} + +// PrimaryContainerPort returns the release's designated primary container +// port (audit TCL-14) — the port health checks publish-probe and Caddy +// dials, which fields[0]-style port picks cannot identify once a container +// publishes more than one port. +func PrimaryContainerPort(rec *Record) (int, bool) { + if rec == nil { + return 0, false + } + for _, p := range rec.Ports { + if p.Primary && p.ContainerPort > 0 { + return p.ContainerPort, true + } + } + return 0, false +} + +// HasFixedHostPorts reports whether any recorded port is an +// operator-configured host port — the condition that makes two containers of +// the same app unable to run at once and therefore selects the recreate +// strategy on deploy and rollback (audit F21). +func HasFixedHostPorts(rec *Record) bool { + if rec == nil { + return false + } + if len(rec.Publish) > 0 { + return true + } + for _, p := range rec.Ports { + if p.Fixed { + return true + } + } + return false +} + +// PortFromPublishSpec parses a verbatim docker -p publish spec +// ("[ip:]host:container[/proto]", "[host:]container[/proto]") into a Port. +// IPv6 bind forms with more than three colon-separated segments are not +// parsed; the caller keeps the raw string in Publish, which is what recreate +// and displace decisions actually read. +func PortFromPublishSpec(spec string) (Port, bool) { + s := strings.TrimSpace(spec) + if s == "" { + return Port{}, false + } + parts := strings.Split(s, ":") + if len(parts) > 3 { + return Port{}, false + } + var bind, host, container string + switch len(parts) { + case 1: + container = parts[0] + case 2: + host, container = parts[0], parts[1] + case 3: + bind, host, container = parts[0], parts[1], parts[2] + } + proto := "" + if i := strings.Index(container, "/"); i >= 0 { + proto, container = container[i+1:], container[:i] + } + if container == "" { + return Port{}, false + } + p := Port{ContainerPort: atoiOrZero(container), Proto: proto, Bind: bind, Fixed: true} + p.HostPort = atoiOrZero(host) + return p, p.ContainerPort > 0 +} + +func atoiOrZero(s string) int { + n, _ := strconv.Atoi(s) + return n +} + +// Backfill synthesizes the "release-0" record for a pre-F14 release from the +// live containers (see the package doc). containers is the caller's +// inventory (docker.Client.ListContainers output) — rollback already holds +// it; re-listing would race the recreation it is about to perform. The +// returned record is also persisted; a persistence failure is returned as a +// warning error alongside the record so the caller can use it in-memory. +// +// What backfill cannot recover, by construction: --env-file references +// (inspect shows resolved values only), the teploy health gate, and the +// Caddy edge config. Those fields stay empty and readers keep their legacy +// fallbacks for them. +func Backfill(ctx context.Context, exec ssh.Executor, dk *docker.Client, containers []docker.Container, app, hash string, st *state.AppState) (*Record, error) { + if st == nil { + return nil, fmt.Errorf("backfilling %s@%s: no state to source ingress/domain from", app, hash) + } + var web []docker.Container + processes := map[string]string{} + for _, c := range containers { + if c.Labels["teploy.version"] != hash { + continue + } + proc := c.Labels["teploy.process"] + if proc != "" { + processes[proc] = "" + } + if proc == "web" { + web = append(web, c) + } + } + if len(web) == 0 { + return nil, fmt.Errorf("backfilling %s@%s: no web container to inspect", app, hash) + } + sort.Slice(web, func(i, j int) bool { return web[i].Name < web[j].Name }) + + spec, err := dk.InspectRecreate(ctx, web[0].Name) + if err != nil { + return nil, fmt.Errorf("backfilling %s@%s: %w", app, hash, err) + } + + rec := &Record{ + SchemaVersion: SchemaVersion, + App: app, + Hash: hash, + CreatedAt: time.Now().UTC(), + Backfilled: true, + + DeploymentType: st.DeploymentType, + IngressMode: st.IngressMode, + Domain: st.Domain, + Replicas: len(web), + Processes: processes, + Recreate: spec, + } + + if st.DeploymentType == "" { + rec.DeploymentType = "container" + } + if st.IngressMode == "" { + rec.IngressMode = "caddy" + } + + // Image identity: prefer the immutable image ID (F19's rule); keep the + // original tag reference alongside it. + rec.ImageDigest = spec.ImageID + rec.ImageRef = spec.ImageRef + + // Resolved env, straight from the container's own config. PORT is the + // primary-port signal teploy's docker run sets; it selects the primary + // binding below. + env := map[string]string{} + portEnv := 0 + for _, kv := range spec.Env { + k, v, ok := strings.Cut(kv, "=") + if !ok { + continue + } + env[k] = v + if k == "PORT" { + portEnv = atoiOrZero(v) + } + } + if len(env) > 0 { + rec.Env = env + } + + // Ports from the actual bindings. Fixed = outside the ephemeral range + // this CLI allocates from (49152-65535): an operator-chosen host port. + // A fixed port published inside the ephemeral range is misclassified as + // ephemeral, which degrades to today's clean docker-run error rather + // than an outage; fresh deploys record Publish verbatim and never rely + // on this heuristic. + primarySet := false + for _, b := range spec.PortBindings { + if b.ContainerPort <= 0 { + continue + } + p := Port{ + HostPort: b.HostPort, + ContainerPort: b.ContainerPort, + Proto: b.Proto, + Bind: b.HostIP, + } + p.Fixed = b.HostPort > 0 && (b.HostPort < 49152 || b.HostPort > 65535) + if !primarySet && b.Proto != "udp" && (b.ContainerPort == portEnv || portEnv == 0) { + p.Primary = true + primarySet = true + } + rec.Ports = append(rec.Ports, p) + } + + // Volumes: binds (host:container[:mode]) plus named-volume mounts. + volumes := map[string]string{} + for _, bind := range spec.Binds { + parts := strings.Split(bind, ":") + if len(parts) < 2 { + continue + } + volumes[parts[0]] = parts[1] + } + for _, m := range spec.Mounts { + if m.Type == "volume" && m.Source != "" && m.Target != "" { + volumes[m.Source] = m.Target + } + } + if len(volumes) > 0 { + rec.Volumes = volumes + } + + if len(spec.Labels) > 0 { + rec.Labels = spec.Labels + } + if spec.MemoryBytes > 0 { + rec.Memory = fmt.Sprintf("%db", spec.MemoryBytes) + } + if spec.NanoCPUs > 0 { + rec.CPU = strconv.FormatFloat(float64(spec.NanoCPUs)/1e9, 'f', -1, 64) + } + if len(spec.Cmd) > 0 { + rec.Cmd = strings.Join(spec.Cmd, " ") + } + if proc, ok := processes["web"]; ok && proc == "" { + processes["web"] = rec.Cmd + } + + if err := Write(ctx, exec, rec); err != nil { + return rec, fmt.Errorf("persisting the backfilled record: %w", err) + } + return rec, nil +} diff --git a/internal/releasemeta/releasemeta_test.go b/internal/releasemeta/releasemeta_test.go new file mode 100644 index 0000000..0062fe8 --- /dev/null +++ b/internal/releasemeta/releasemeta_test.go @@ -0,0 +1,328 @@ +package releasemeta + +import ( + "context" + "encoding/json" + "fmt" + "strings" + "testing" + + "github.com/useteploy/teploy/internal/docker" + "github.com/useteploy/teploy/internal/ssh" + "github.com/useteploy/teploy/internal/state" +) + +func recordJSON(t *testing.T, rec Record) string { + t.Helper() + rec.SchemaVersion = SchemaVersion + data, err := json.Marshal(&rec) + if err != nil { + t.Fatal(err) + } + return string(data) +} + +func TestRead_AbsentVsPresentVsUnreadable(t *testing.T) { + rec := recordJSON(t, Record{App: "myapp", Hash: "v1", DeploymentType: "container", IngressMode: "caddy"}) + + t.Run("absent", func(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", Output: "absent"}, + ) + got, err := Read(context.Background(), mock, "myapp", "v1") + if err != nil || got != nil { + t.Fatalf("expected (nil, nil) for a confirmed-missing record, got (%v, %v)", got, err) + } + }) + + t.Run("present", func(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", Output: "present\n" + rec}, + ) + got, err := Read(context.Background(), mock, "myapp", "v1") + if err != nil || got == nil { + t.Fatalf("expected the record, got (%v, %v)", got, err) + } + if got.App != "myapp" || got.Hash != "v1" || got.DeploymentType != "container" { + t.Errorf("record parsed wrong: %+v", got) + } + }) + + t.Run("malformed is an error, not absence", func(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", Output: "present\n{not json"}, + ) + got, err := Read(context.Background(), mock, "myapp", "v1") + if err == nil || got != nil { + t.Fatalf("malformed record must be an error (F15 discipline), got (%v, %v)", got, err) + } + }) + + t.Run("transport failure is an error, not absence", func(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4") // no fixture: unexpected command + if _, err := Read(context.Background(), mock, "myapp", "v1"); err == nil { + t.Fatal("transport failure must be an error, never silent absence") + } + }) + + t.Run("future schema rejected", func(t *testing.T) { + future := strings.Replace(rec, fmt.Sprintf(`"schema_version":%d`, SchemaVersion), `"schema_version":99`, 1) + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", Output: "present\n" + future}, + ) + if _, err := Read(context.Background(), mock, "myapp", "v1"); err == nil || !strings.Contains(err.Error(), "schema") { + t.Fatalf("expected schema rejection, got %v", err) + } + }) +} + +func TestWrite_RoundTripAndPathValidation(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "mkdir -p /deployments/myapp/meta", Output: ""}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", Output: "absent"}, + ssh.MockCommand{Match: "UPLOAD:", Output: ""}, + ssh.MockCommand{Match: "mv", Output: ""}, + ) + rec := &Record{ + App: "myapp", Hash: "v1", + DeploymentType: "container", IngressMode: "host", Domain: "myapp.io", + Ports: []Port{{HostPort: 7460, ContainerPort: 7460, Bind: "0.0.0.0", Primary: true, Fixed: true}}, + EnvFiles: []string{"/deployments/myapp/.env"}, + } + if err := Write(context.Background(), mock, rec); err != nil { + t.Fatalf("Write: %v", err) + } + + // 0600: records can carry resolved env from backfills. + var uploadCall string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "UPLOAD:/deployments/myapp/meta/v1.json.tmp-") { + uploadCall = c + } + } + if uploadCall == "" { + t.Fatal("record was not uploaded to its atomic temp path") + } + if !strings.Contains(uploadCall, "mode 0600") { + t.Errorf("record must be written 0600, got: %s", uploadCall) + } + written, ok := mock.Files["/deployments/myapp/meta/v1.json"] + if !ok { + t.Fatal("record not committed into place") + } + if !strings.Contains(string(written), `"hash":"v1"`) || !strings.Contains(string(written), `"fixed":true`) { + t.Errorf("unexpected record content: %s", written) + } + + // Invalid release ids never reach a remote path. + for _, bad := range []string{"", "../state", "a/b", ".hidden", strings.Repeat("x", 129)} { + if _, err := Path("myapp", bad); err == nil { + t.Errorf("Path accepted invalid hash %q", bad) + } + } + if _, err := Path("", "v1"); err == nil { + t.Error("Path accepted an empty app") + } +} + +func TestPrimaryPortSelection(t *testing.T) { + rec := &Record{Ports: []Port{ + {HostPort: 51820, ContainerPort: 51820, Proto: "udp", Fixed: true}, + {HostPort: 49152, ContainerPort: 3000, Primary: true}, + }} + if p, ok := PrimaryContainerPort(rec); !ok || p != 3000 { + t.Errorf("PrimaryContainerPort = (%d, %v), want (3000, true)", p, ok) + } + if p, ok := PrimaryContainerPort(&Record{}); ok || p != 0 { + t.Errorf("PrimaryContainerPort on a record without ports = (%d, %v)", p, ok) + } +} + +func TestHasFixedHostPorts(t *testing.T) { + if HasFixedHostPorts(nil) { + t.Error("nil record has no fixed ports") + } + if !HasFixedHostPorts(&Record{Publish: []string{"127.0.0.1:3001:3001"}}) { + t.Error("publish entries are fixed host ports") + } + if !HasFixedHostPorts(&Record{Ports: []Port{{HostPort: 7460, ContainerPort: 7460, Fixed: true}}}) { + t.Error("a Fixed-flagged port must count") + } + if HasFixedHostPorts(&Record{Ports: []Port{{HostPort: 49152, ContainerPort: 3000}}}) { + t.Error("ephemeral blue/green ports are not fixed") + } +} + +func TestPortFromPublishSpec(t *testing.T) { + cases := []struct { + in string + want Port + ok bool + }{ + {"0.0.0.0:3001:3001", Port{Bind: "0.0.0.0", HostPort: 3001, ContainerPort: 3001, Fixed: true}, true}, + {"127.0.0.1:9100:9000", Port{Bind: "127.0.0.1", HostPort: 9100, ContainerPort: 9000, Fixed: true}, true}, + {"3001:3001", Port{HostPort: 3001, ContainerPort: 3001, Fixed: true}, true}, + {"53:53/udp", Port{HostPort: 53, ContainerPort: 53, Proto: "udp", Fixed: true}, true}, + {"3000", Port{ContainerPort: 3000, Fixed: true}, true}, + {"[::]:80:80", Port{}, false}, // IPv6 multi-colon: raw-only, by design + {"", Port{}, false}, + } + for _, tc := range cases { + got, ok := PortFromPublishSpec(tc.in) + if ok != tc.ok || got != tc.want { + t.Errorf("PortFromPublishSpec(%q) = (%+v, %v), want (%+v, %v)", tc.in, got, ok, tc.want, tc.ok) + } + } +} + +// backfillInspect is a rich inspect for the target's web container: multiple +// bindings (primary 3000 via PORT env, plus a fixed 51820/udp publish), env, +// a named-volume mount, labels, resource limits, and a healthcheck override. +func backfillInspect() string { + return fmt.Sprintf(`[{ + "Image": "sha256:%s", + "Config": { + "Image": "myapp:v1", + "Env": ["PORT=3000", "API_KEY=resolved-secret"], + "Cmd": ["npm", "run", "start"], + "Labels": {"teploy.app": "myapp", "teploy.version": "v1", "teploy.process": "web"}, + "Healthcheck": {"Test": ["NONE"]} + }, + "HostConfig": { + "NetworkMode": "teploy", + "PortBindings": { + "3000/tcp": [{"HostIp": "127.0.0.1", "HostPort": "49152"}], + "9100/udp": [{"HostIp": "0.0.0.0", "HostPort": "9100"}] + }, + "Binds": [], + "Mounts": [{"Type": "volume", "Source": "myapp-uploads", "Target": "/uploads"}], + "RestartPolicy": {"Name": "unless-stopped"}, + "Memory": 536870912, + "NanoCpus": 0, + "LogConfig": {"Type": "json-file", "Config": {"max-size": "10m"}} + }, + "NetworkSettings": {"Networks": {"teploy": {"Aliases": ["myapp"]}}} +}]`, strings.Repeat("e", 64)) +} + +func backfillContainers() string { + return `{"ID":"aaa","Names":"myapp-web-v1","Image":"myapp:v1","State":"exited","Status":"Exited","Labels":"teploy.app=myapp,teploy.version=v1,teploy.process=web"}` + "\n" + + `{"ID":"bbb","Names":"myapp-worker-v1","Image":"myapp:v1","State":"exited","Status":"Exited","Labels":"teploy.app=myapp,teploy.version=v1,teploy.process=worker"}` +} + +// The convergence migration (F14's "release-0"): an install whose releases +// predate the store gets a record synthesized from the live containers the +// first time rollback needs one. +func TestBackfill_FromLiveContainers(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker ps --all --filter label=teploy.app='myapp'", Output: backfillContainers()}, + ssh.MockCommand{Match: "docker inspect 'myapp-web-v1'", Output: backfillInspect()}, + ssh.MockCommand{Match: "mkdir -p /deployments/myapp/meta", Output: ""}, + ssh.MockCommand{Match: "UPLOAD:", Output: ""}, + ssh.MockCommand{Match: "mv", Output: ""}, + ) + containers, err := docker.ParseContainers(backfillContainers()) + if err != nil { + t.Fatal(err) + } + st := &state.AppState{IngressMode: "caddy", Domain: "myapp.com", DeploymentType: "container"} + + rec, warn := Backfill(context.Background(), mock, docker.NewClient(mock), containers, "myapp", "v1", st) + if rec == nil { + t.Fatalf("Backfill returned no record (warn: %v)", warn) + } + if warn != nil { + t.Errorf("unexpected persistence warning: %v", warn) + } + + if !rec.Backfilled { + t.Error("backfilled record must be flagged") + } + if rec.ImageDigest != "sha256:"+strings.Repeat("e", 64) || rec.ImageRef != "myapp:v1" { + t.Errorf("image identity not captured: %s / %s", rec.ImageDigest, rec.ImageRef) + } + // Resolved env from inspect — the recoverable form of env references. + if rec.Env["API_KEY"] != "resolved-secret" || rec.Env["PORT"] != "3000" { + t.Errorf("resolved env not captured: %v", rec.Env) + } + if len(rec.EnvFiles) != 0 { + t.Errorf("env-file references are NOT recoverable; recording any would lie: %v", rec.EnvFiles) + } + // Ports: 3000 primary (via PORT env), 51820 fixed (outside ephemeral range). + var primary, fixed *Port + for i := range rec.Ports { + if rec.Ports[i].Primary { + primary = &rec.Ports[i] + } + if rec.Ports[i].ContainerPort == 9100 { + fixed = &rec.Ports[i] + } + } + if primary == nil || primary.ContainerPort != 3000 || primary.HostPort != 49152 { + t.Errorf("primary port not selected via PORT env: %+v", rec.Ports) + } + if fixed == nil || !fixed.Fixed { + t.Errorf("fixed publish port not detected: %+v", rec.Ports) + } + if rec.Volumes["myapp-uploads"] != "/uploads" { + t.Errorf("named volume not captured: %v", rec.Volumes) + } + if rec.Memory != "536870912b" { + t.Errorf("memory not captured: %q", rec.Memory) + } + if rec.Processes["worker"] != "" || rec.Processes["web"] == "" { + t.Errorf("process set not captured: %v", rec.Processes) + } + if rec.Cmd != "npm run start" { + t.Errorf("web cmd not captured: %q", rec.Cmd) + } + if rec.Recreate == nil || !rec.Recreate.NoHealthcheck { + t.Errorf("recreate spec not embedded: %+v", rec.Recreate) + } + if rec.Health != nil || rec.Caddy != nil { + t.Errorf("health/caddy config is not recoverable from containers; recording any would lie: %+v %+v", rec.Health, rec.Caddy) + } + + // The record was persisted: a second Read finds it without another + // backfill. + mock2 := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", + Output: "present\n" + string(mock.Files["/deployments/myapp/meta/v1.json"])}, + ) + reread, err := Read(context.Background(), mock2, "myapp", "v1") + if err != nil || reread == nil || !reread.Backfilled { + t.Fatalf("backfilled record did not round-trip: (%v, %v)", reread, err) + } +} + +func TestBackfill_NoWebContainerFails(t *testing.T) { + containers, err := docker.ParseContainers(`{"ID":"bbb","Names":"myapp-worker-v1","Image":"myapp:v1","State":"exited","Status":"Exited","Labels":"teploy.app=myapp,teploy.version=v1,teploy.process=worker"}`) + if err != nil { + t.Fatal(err) + } + mock := ssh.NewMockExecutor("1.2.3.4") + if _, err := Backfill(context.Background(), mock, docker.NewClient(mock), containers, "myapp", "v1", + &state.AppState{IngressMode: "caddy"}); err == nil || !strings.Contains(err.Error(), "no web container") { + t.Fatalf("expected no-web-container error, got %v", err) + } +} + +// A persistence failure still returns the synthesized record (usable +// in-memory) alongside a warning — the store must never be why a rollback +// that used to work stops working. +func TestBackfill_PersistFailureStillReturnsRecord(t *testing.T) { + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "docker ps --all", Output: backfillContainers()}, + ssh.MockCommand{Match: "docker inspect 'myapp-web-v1'", Output: backfillInspect()}, + ssh.MockCommand{Match: "mkdir -p /deployments/myapp/meta", Err: fmt.Errorf("disk full")}, + ) + containers, _ := docker.ParseContainers(backfillContainers()) + rec, warn := Backfill(context.Background(), mock, docker.NewClient(mock), containers, "myapp", "v1", + &state.AppState{IngressMode: "caddy"}) + if rec == nil { + t.Fatal("record must still be returned for in-memory use") + } + if warn == nil || !strings.Contains(warn.Error(), "persisting") { + t.Errorf("expected a persistence warning, got %v", warn) + } +} diff --git a/internal/state/pins.go b/internal/state/pins.go index b6afff4..24202bf 100644 --- a/internal/state/pins.go +++ b/internal/state/pins.go @@ -27,7 +27,7 @@ func pinsPath(app string) string { // retained (audit F78). Callers must fail CLOSED on error (skip pruning), // never prune as though no pins existed. func ReadPins(ctx context.Context, exec ssh.Executor, app string) ([]string, error) { - data, present, err := readRemoteFile(ctx, exec, pinsPath(app)) + data, present, err := ReadRemoteFile(ctx, exec, pinsPath(app)) if err != nil { return nil, fmt.Errorf("reading pins for %s: %w", app, err) } diff --git a/internal/state/state.go b/internal/state/state.go index bf098bf..83f9098 100644 --- a/internal/state/state.go +++ b/internal/state/state.go @@ -113,12 +113,13 @@ type LogEntry struct { Image string `json:"image,omitempty"` } -// readRemoteFile returns the file's contents and whether it exists. Absence +// ReadRemoteFile returns the file's contents and whether it exists. Absence // is CONFIRMED by the remote test — the old `cat path 2>/dev/null` shape // folded "missing", "unreadable", and "transport failed" into one silent // empty result, so a permission failure was indistinguishable from a first -// deploy (audit F15). -func readRemoteFile(ctx context.Context, exec ssh.Executor, path string) ([]byte, bool, error) { +// deploy (audit F15). Exported for releasemeta, which reads per-release +// records with the same absence/transport distinction. +func ReadRemoteFile(ctx context.Context, exec ssh.Executor, path string) ([]byte, bool, error) { cmd := fmt.Sprintf("if [ ! -e %s ]; then printf 'absent\\n'; else printf 'present\\n'; cat -- %s; fi", ssh.ShellQuote(path), ssh.ShellQuote(path)) out, err := exec.Run(ctx, cmd) @@ -146,7 +147,7 @@ func readRemoteFile(ctx context.Context, exec ssh.Executor, path string) ([]byte // (audit F15). func Read(ctx context.Context, exec ssh.Executor, app string) (*AppState, error) { v2Path := fmt.Sprintf("%s/%s/state.json", deploymentsDir, app) - data, present, err := readRemoteFile(ctx, exec, v2Path) + data, present, err := ReadRemoteFile(ctx, exec, v2Path) if err != nil { return nil, fmt.Errorf("reading canonical state for %s: %w", app, err) } @@ -173,7 +174,7 @@ func Read(ctx context.Context, exec ssh.Executor, app string) (*AppState, error) // Only a confirmed-missing canonical file reaches the legacy migration. path := fmt.Sprintf("%s/%s/state", deploymentsDir, app) - data, present, err = readRemoteFile(ctx, exec, path) + data, present, err = ReadRemoteFile(ctx, exec, path) if err != nil { return nil, fmt.Errorf("reading legacy state for %s: %w", app, err) } From 7314da71560ec5bc98657051a2b79b090e996282 Mon Sep 17 00:00:00 2001 From: Tyler <53561637+im-tyler@users.noreply.github.com> Date: Fri, 18 Sep 2026 04:11:47 -0700 Subject: [PATCH 3/4] =?UTF-8?q?feat(deploy):=20F13/F21/TCL-14=20=E2=80=94?= =?UTF-8?q?=20rollback=20and=20publish=20recreate=20restore=20from=20the?= =?UTF-8?q?=20recorded=20release?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Deploy records every release (F14 store) right after the state commit: ports with the primary designation and fixed/ephemeral flags, publish specs, env-file references, health gate, Caddy edge config, and the primary web container's RecreateSpec. Record failures warn — the containers are live and routed by then; the store must never turn a successful deploy into an outage. Rollback now restores FROM the record, not from the current teploy.yml or reverse-engineered container state: recorded health gate, domain, TLS / caddy_extra / cache / firewall / access, and the recorded ingress mode. A confirmed-missing record backfills from the live containers (release-0 convergence); an unreadable record degrades to the historical inspect-driven path with a warning — the store is an overlay, never a new way for rollback to refuse. TCL-14: the recorded primary container port decides which binding the health check probes (HostPortFor) and which port Caddy dials — a multi-port container's auxiliary listener can no longer steal either. Without a record, the old first-port behavior remains. F21: publish apps take the explicit recreate strategy on both paths. Deploy displaces the current web container before the replacement starts (the same restore-on-failure contract host ingress has) instead of dying mid-deploy on 'port is already allocated'; rollback likewise frees the fixed ports first and stops avoiding the live ports for fixed-port targets. F13: 'teploy rollback --app --host ...' is now complete — StaticDeployer.RollbackStateOnly flips the release symlink and rebuilds the Caddy block from the recorded serving config (SPA/headers/cache/ extra/domain). No record means an explicit redeploy-first request; a static release's serving config is unrecoverable from disk. The directory-based static rollback also prefers the record when one exists. --- internal/cli/rollback.go | 21 +- internal/deploy/deploy.go | 120 +++++- internal/deploy/f14_wiring_test.go | 656 +++++++++++++++++++++++++++++ internal/deploy/rollback.go | 142 ++++++- internal/deploy/rollback_test.go | 12 +- internal/deploy/static.go | 165 +++++++- 6 files changed, 1067 insertions(+), 49 deletions(-) create mode 100644 internal/deploy/f14_wiring_test.go diff --git a/internal/cli/rollback.go b/internal/cli/rollback.go index 994bbc3..d3d1150 100644 --- a/internal/cli/rollback.go +++ b/internal/cli/rollback.go @@ -131,12 +131,11 @@ func runRollback(flags *Flags, toHash string) error { // teploy.yml. Used by teploy-dash and for running rollback outside of an // app directory. // -// Known gap, not addressed here: this always calls the container -// deploy.Rollback — it never checks whether the app is actually -// type:static (that requires teploy.yml, which this path deliberately -// doesn't need) and branches to StaticDeployer.Rollback the way -// runRollback does. Calling `teploy rollback --app --host -// ...` will fail rather than correctly flip the release symlink. +// Both deploy types are complete on this path since F13/F14: container +// rollbacks restore from the recorded release spec (deploy.Rollback reads +// the F14 record, backfilling from live containers for pre-store releases), +// and static rollbacks read the recorded serving config +// (StaticDeployer.RollbackStateOnly). func runRollbackByApp(flags *Flags, appName, toHash string) error { if err := config.ValidateName(appName); err != nil { return err @@ -172,10 +171,12 @@ func runRollbackByApp(flags *Flags, appName, toHash string) error { return fmt.Errorf("no state found for app %q on %s — has it been deployed?", appName, host) } if appState.DeploymentType == "static" { - // Static rollback needs the release-serving config (SPA, headers, - // cache…) which state does not retain — deferred, see AUDIT_OPEN.md - // (F13): fail with direction instead of corrupting the route. - return fmt.Errorf("app %q is a static deploy; state-only static rollback is not supported yet — run `teploy rollback` from the app directory (teploy.yml provides the serving config)", appName) + // Static rollback reads the serving config (SPA, headers, cache…) + // from the recorded release metadata (F13/F14). Releases deployed + // before the store existed have no record — that is a redeploy-first + // request, not a guess at how the release was served. + d := deploy.NewStaticDeployer(executor, os.Stdout) + return d.RollbackStateOnly(ctx, appName, toHash) } ingress := appState.IngressMode if ingress == "" { diff --git a/internal/deploy/deploy.go b/internal/deploy/deploy.go index 3e525a2..063df5b 100644 --- a/internal/deploy/deploy.go +++ b/internal/deploy/deploy.go @@ -5,12 +5,14 @@ import ( "encoding/json" "fmt" "io" + "maps" "sort" "strings" "time" "github.com/useteploy/teploy/internal/caddy" "github.com/useteploy/teploy/internal/docker" + "github.com/useteploy/teploy/internal/releasemeta" "github.com/useteploy/teploy/internal/ssh" "github.com/useteploy/teploy/internal/state" ) @@ -194,6 +196,17 @@ func (d *Deployer) DeployLocked(ctx context.Context, cfg Config) error { webBindHost = "0.0.0.0" } + // recreateWeb selects the explicit recreate strategy (audit F21): stop + // the current web container(s) BEFORE starting the replacement. Host + // ingress needs it because the fixed published port cannot double-bind; + // `publish:` entries need it for the same reason — their host ports are + // operator-configured and identical across versions, so blue/green would + // just die on "port is already allocated" mid-deploy (previously a clean + // pre-mutation error; now a deliberate, brief-outage recreate with + // restore-on-failure, matching host ingress). Validation upstream keeps + // publish/host ingress single-replica. + recreateWeb := cfg.ingressHost() || len(cfg.Publish) > 0 + start := time.Now() if replicas > 1 { @@ -335,12 +348,13 @@ func (d *Deployer) DeployLocked(ctx context.Context, cfg Config) error { } } - // Host ingress recreates rather than blue/greens: the new container reuses - // the old one's fixed host port, so stop the running web container first to - // free it. Keep the stopped container until state commits so a failed commit - // can recreate it with its original configuration and fixed port. + // Host ingress and publish-apps recreate rather than blue/green: the new + // container reuses the old one's fixed host port(s), so stop the running + // web container first to free them. Keep the stopped container until + // state commits so a failed commit can recreate it with its original + // configuration and fixed port. var displacedHostWeb []string - if cfg.ingressHost() { + if recreateWeb { names, _ := d.exec.Run(ctx, fmt.Sprintf( "docker ps --filter label=teploy.app=%s --filter label=teploy.process=web --format '{{.Names}}'", cfg.App)) for _, name := range strings.Fields(names) { @@ -356,9 +370,9 @@ func (d *Deployer) DeployLocked(ctx context.Context, cfg Config) error { } cancel() if lastErr != nil { - return fmt.Errorf("stopping current host-ingress container %s: %w — restoring the displaced workload also failed: %v; %s needs manual attention", name, err, lastErr, cfg.App) + return fmt.Errorf("stopping current fixed-port container %s: %w — restoring the displaced workload also failed: %v; %s needs manual attention", name, err, lastErr, cfg.App) } - return fmt.Errorf("stopping current host-ingress container %s: %w — the displaced workload was restored", name, err) + return fmt.Errorf("stopping current fixed-port container %s: %w — the displaced workload was restored", name, err) } displacedHostWeb = append(displacedHostWeb, name) } @@ -572,6 +586,12 @@ func (d *Deployer) DeployLocked(ctx context.Context, cfg Config) error { return d.abortStateCommit(ctx, cfg, current, started, displacedHostWeb, start, err) } + // 13b. Record the release metadata (F14). The containers are live and + // the route/state are committed — a record failure is a degraded + // rollback window, not a failed deploy, and it converges on the next + // deploy or backfill. Never abort into abortStateCommit from here. + d.recordRelease(ctx, cfg, newState, ports, webBindHost, webContainerName) + // 14. Stop the predecessor workload snapshotted in step 6b (all // processes + all replicas). For same-version redeploys the old // containers were renamed to _replaced; remove them after stopping so @@ -692,7 +712,7 @@ func stopOldWorkloadsByName(ctx context.Context, dk *docker.Client, out io.Write func (d *Deployer) abortStateCommit(ctx context.Context, cfg Config, current *state.AppState, started, displacedHostWeb []string, start time.Time, commitErr error) error { d.logDeploy(ctx, cfg, false, start) - if cfg.ingressHost() { + if cfg.ingressHost() || len(cfg.Publish) > 0 { for _, name := range started { d.docker.Stop(ctx, name, 5) } @@ -782,6 +802,90 @@ func (d *Deployer) logDeploy(ctx context.Context, cfg Config, success bool, star }) } +// recordRelease persists the F14 record for the release this deploy just +// committed. Ports are the resolved live allocation (the primary's host +// binding plus every publish entry); env records the references (server-side +// env-file paths + the plaintext env map), never resolved secrets. The +// primary web container's full RecreateSpec is embedded from docker's own +// view of it. Every failure is a warning — see the call site. +func (d *Deployer) recordRelease(ctx context.Context, cfg Config, applied *state.AppState, ports []int, webBindHost, webContainerName string) { + containerPort := cfg.ContainerPort + if containerPort == 0 { + containerPort = 80 + } + healthCfg := cfg.Health.withDefaults() + + rec := &releasemeta.Record{ + App: cfg.App, + Hash: cfg.Version, + Generation: applied.Generation, + DeploymentType: "container", + IngressMode: cfg.Ingress, + Domain: cfg.Domain, + ImageRef: cfg.Image, + ImageDigest: applied.ImageDigest, + Replicas: len(ports), + Processes: maps.Clone(cfg.Processes), + Cmd: cfg.Cmd, + Env: maps.Clone(cfg.Env), + EnvFiles: append([]string(nil), cfg.EnvFiles...), + Volumes: maps.Clone(cfg.Volumes), + Publish: append([]string(nil), cfg.Publish...), + Memory: cfg.Memory, + CPU: cfg.CPU, + StopTimeout: cfg.StopTimeout, + Bind: cfg.Bind, + Health: &releasemeta.Health{ + Path: healthCfg.Path, + TimeoutSeconds: int(healthCfg.Timeout.Seconds()), + IntervalSeconds: int(healthCfg.Interval.Seconds()), + }, + } + if rec.IngressMode == "" { + rec.IngressMode = "caddy" + } + if len(ports) > 0 { + rec.Ports = append(rec.Ports, releasemeta.Port{ + HostPort: ports[0], + ContainerPort: containerPort, + Bind: webBindHost, + Primary: true, + Fixed: cfg.ingressHost(), + }) + } + for _, pub := range cfg.Publish { + if p, ok := releasemeta.PortFromPublishSpec(pub); ok { + rec.Ports = append(rec.Ports, p) + } + } + if cfg.usesCaddy() { + rec.Caddy = &releasemeta.CaddyRoute{ + TLSCert: cfg.TLSCert, + TLSKey: cfg.TLSKey, + TLSInternal: cfg.TLSInternal, + CaddyExtra: cfg.CaddyExtra, + Cache: maps.Clone(cfg.Cache), + } + if !cfg.Firewall.Empty() { + fw := cfg.Firewall + rec.Caddy.Firewall = &fw + } + if !cfg.Access.Empty() { + acc := cfg.Access + rec.Caddy.Access = &acc + } + } + + if spec, err := d.docker.InspectRecreate(ctx, webContainerName); err == nil { + rec.Recreate = spec + } else { + fmt.Fprintf(d.out, "Warning: could not capture the recreate spec for %s: %v (recreate falls back to live inspect)\n", webContainerName, err) + } + if err := releasemeta.Write(ctx, d.exec, rec); err != nil { + fmt.Fprintf(d.out, "Warning: could not record release metadata for %s@%s: %v (rollback for this release falls back to live inspection)\n", cfg.App, cfg.Version, err) + } +} + // sortedProcessNames returns process names with "web" first, then alphabetical. func sortedProcessNames(processes map[string]string) []string { var others []string diff --git a/internal/deploy/f14_wiring_test.go b/internal/deploy/f14_wiring_test.go new file mode 100644 index 0000000..2eaefce --- /dev/null +++ b/internal/deploy/f14_wiring_test.go @@ -0,0 +1,656 @@ +package deploy + +// Behavioral tests for the F14 wiring: deploy records the release spec, +// rollback restores from the record (backfilling pre-store releases), the +// recorded primary port drives health checks and Caddy upstreams (TCL-14), +// publish deploys/rollbacks take the explicit recreate strategy (F21), and +// state-only static rollback works from the record (F13). + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "strings" + "testing" + "time" + + "github.com/useteploy/teploy/internal/releasemeta" + "github.com/useteploy/teploy/internal/ssh" +) + +const f14Inspect = `[{ + "Image": "sha256:%s", + "Config": { + "Image": "myapp:v9", + "Env": ["PORT=3000", "API_KEY=x"], + "Cmd": ["npm", "start"], + "Labels": {"teploy.app": "myapp", "teploy.version": "v9", "teploy.process": "web"}, + "Healthcheck": {"Test": ["NONE"]} + }, + "HostConfig": { + "NetworkMode": "teploy", + "PortBindings": {"3000/tcp": [{"HostIp": "127.0.0.1", "HostPort": "49152"}]}, + "Binds": ["/deployments/myapp/volumes/data:/data"], + "RestartPolicy": {"Name": "unless-stopped"}, + "Memory": 268435456, + "LogConfig": {"Type": "json-file", "Config": {"max-size": "10m"}} + }, + "NetworkSettings": {"Networks": {"teploy": {"Aliases": ["myapp"]}}} +}]` + +func withDigest(s string) string { return fmt.Sprintf(s, strings.Repeat("9", 64)) } + +// deployRecordFixture is the standard successful-deploy mock. The web +// container's full inspect is registered BEFORE the generic "docker inspect" +// status fixture so InspectRecreate sees real JSON. Extra fixtures are +// registered FIRST (the mock matches in order), so per-test overrides of +// base fixtures win. +func deployRecordFixture(imageDigest string, extra ...ssh.MockCommand) *ssh.MockExecutor { + base := []ssh.MockCommand{ + {Match: "mkdir -p /deployments/myapp", Output: ""}, + {Match: "mkdir /deployments/myapp/.lock", Output: ""}, + {Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "absent"}, + {Match: "if [ ! -e '/deployments/myapp/state' ]", Output: "absent"}, + {Match: "ss -tln", Output: ssOutput}, + {Match: "docker ps --all --filter label=teploy.app='myapp'", Output: ""}, + {Match: "docker run", Output: "abc123def456"}, + {Match: "docker inspect 'myapp-web-v9'", Output: withDigest(f14Inspect)}, + {Match: "docker inspect -f '{{.Image}}'", Output: imageDigest}, + {Match: "docker inspect", Output: "running"}, + {Match: "curl -s -o /dev/null", Output: "200"}, + {Match: "curl -sf http://localhost:2019/config/apps/http/servers/srv0", Output: `{"listen":[":80",":443"]}`}, + {Match: "curl -sf -X PATCH", Err: fmt.Errorf("not found")}, + {Match: "curl -sf -X POST http://localhost:2019/config/apps/http/servers/srv0/routes", Output: ""}, + {Match: "rm -f /tmp/teploy_caddy", Output: ""}, + {Match: "cat /deployments/caddy/Caddyfile", Output: "{\n\tadmin 0.0.0.0:2019\n}\n"}, + {Match: "mv /tmp/teploy_caddyfile.tmp", Output: ""}, + {Match: "mkdir /deployments/caddy/.lock", Output: ""}, + {Match: "a=$(docker exec caddy md5sum", Output: "TEPLOY_CADDY_OK"}, + {Match: "docker exec caddy caddy reload", Output: ""}, + {Match: "rmdir /deployments/caddy/.lock", Output: ""}, + {Match: "printf %s", Output: ""}, + {Match: "rm -rf /deployments/myapp/.lock", Output: ""}, + {Match: "UPLOAD:", Output: ""}, + {Match: "mkdir -p", Output: ""}, + {Match: "mv", Output: ""}, + } + return ssh.NewMockExecutor("1.2.3.4", append(extra, base...)...) +} + +func TestDeploy_RecordsReleaseMetadata(t *testing.T) { + imageDigest := "sha256:" + strings.Repeat("a", 64) + mock := deployRecordFixture(imageDigest) + + var buf bytes.Buffer + err := NewDeployer(mock, &buf).Deploy(context.Background(), Config{ + App: "myapp", + Domain: "myapp.com", + Image: "myapp:v9", + Version: "v9", + ContainerPort: 3000, + Publish: []string{"0.0.0.0:3001:3001"}, + EnvFiles: []string{"/deployments/myapp/.env", "/deployments/myapp/.deploy-env"}, + Env: map[string]string{"PLAIN": "1"}, + Volumes: map[string]string{"/deployments/myapp/volumes/data": "/data"}, + Memory: "256m", + CPU: "1.5", + Health: HealthConfig{Path: "/ready", Timeout: 45 * time.Second, Interval: 2 * time.Second}, + }) + if err != nil { + t.Fatalf("Deploy: %v\n%s", err, buf.String()) + } + + raw, ok := mock.Files["/deployments/myapp/meta/v9.json"] + if !ok { + t.Fatalf("release record not written\n%s", buf.String()) + } + var rec releasemeta.Record + if err := json.Unmarshal(raw, &rec); err != nil { + t.Fatalf("invalid record: %v\n%s", err, raw) + } + + if rec.App != "myapp" || rec.Hash != "v9" || rec.DeploymentType != "container" || rec.IngressMode != "caddy" || rec.Domain != "myapp.com" { + t.Errorf("identity fields wrong: %+v", rec) + } + if rec.ImageRef != "myapp:v9" || rec.ImageDigest != imageDigest { + t.Errorf("image identity not recorded: %s / %s", rec.ImageRef, rec.ImageDigest) + } + if len(rec.EnvFiles) != 2 || rec.EnvFiles[0] != "/deployments/myapp/.env" { + t.Errorf("env-file references not recorded: %v", rec.EnvFiles) + } + if rec.Env["PLAIN"] != "1" || rec.Volumes["/deployments/myapp/volumes/data"] != "/data" { + t.Errorf("env/volumes not recorded: %v %v", rec.Env, rec.Volumes) + } + if rec.Memory != "256m" || rec.CPU != "1.5" { + t.Errorf("resource limits not recorded: %s %s", rec.Memory, rec.CPU) + } + if rec.Health == nil || rec.Health.Path != "/ready" || rec.Health.TimeoutSeconds != 45 { + t.Errorf("health gate not recorded: %+v", rec.Health) + } + if rec.Caddy == nil { + t.Fatalf("caddy route refs not recorded") + } + // Ports: primary (3000 on the allocated ephemeral host port) plus the + // parsed publish entry, flagged fixed (TCL-14/F21 contract). + var primary, publish *releasemeta.Port + for i := range rec.Ports { + switch { + case rec.Ports[i].Primary: + primary = &rec.Ports[i] + case rec.Ports[i].ContainerPort == 3001: + publish = &rec.Ports[i] + } + } + if primary == nil || primary.ContainerPort != 3000 || primary.HostPort != 49152 { + t.Errorf("primary port not recorded with its host binding: %+v", rec.Ports) + } + if primary.Fixed { + t.Errorf("a caddy-ingress primary is ephemeral blue/green, not fixed: %+v", *primary) + } + if publish == nil || !publish.Fixed || publish.HostPort != 3001 || publish.Bind != "0.0.0.0" { + t.Errorf("publish entry not recorded as a fixed port: %+v", rec.Ports) + } + if len(rec.Publish) != 1 || rec.Publish[0] != "0.0.0.0:3001:3001" { + t.Errorf("raw publish specs not recorded: %v", rec.Publish) + } + if rec.Recreate == nil || rec.Recreate.Name != "myapp-web-v9" || !rec.Recreate.NoHealthcheck { + t.Errorf("recreate spec not embedded from docker's view: %+v", rec.Recreate) + } +} + +// A record-write failure is degraded rollback metadata, never a failed +// deploy — the containers are live and routed by the time it happens. +func TestDeploy_RecordWriteFailureIsAWarning(t *testing.T) { + mock := deployRecordFixture("sha256:"+strings.Repeat("a", 64), + ssh.MockCommand{Match: "UPLOAD:/deployments/myapp/meta/", Err: fmt.Errorf("disk full")}, + ) + var buf bytes.Buffer + err := NewDeployer(mock, &buf).Deploy(context.Background(), Config{ + App: "myapp", Domain: "myapp.com", Image: "myapp:v9", Version: "v9", + }) + if err != nil { + t.Fatalf("deploy must not fail on a record-write failure: %v", err) + } + if !strings.Contains(buf.String(), "Warning: could not record release metadata") { + t.Errorf("expected a warning about the unwritten record:\n%s", buf.String()) + } +} + +func TestDeploy_PublishRecreatesByDisplacement(t *testing.T) { + currentState := `{"schema_version":2,"deployment_type":"container","ingress_mode":"caddy","domain":"myapp.com","updated_at":"2026-09-18T10:00:00Z","current_port":49152,"current_hash":"v8"}` + mock := deployRecordFixture("sha256:"+strings.Repeat("a", 64), + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "present\n" + currentState}, + // The displacement inventory: v8's web container is running and + // holds the fixed publish port. + ssh.MockCommand{Match: "docker ps --filter label=teploy.app=myapp --filter label=teploy.process=web", + Output: "myapp-web-v8\n"}, + ssh.MockCommand{Match: "docker stop", Output: ""}, + ) + + var buf bytes.Buffer + err := NewDeployer(mock, &buf).Deploy(context.Background(), Config{ + App: "myapp", + Domain: "myapp.com", + Image: "myapp:v9", + Version: "v9", + Publish: []string{"0.0.0.0:3001:3001"}, + }) + if err != nil { + t.Fatalf("Deploy: %v\n%s", err, buf.String()) + } + + // The publish port is verbatim on the new container's docker run. + var runCmd string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "docker run") && strings.Contains(c, "myapp-web-v9") { + runCmd = c + } + } + if runCmd == "" || !strings.Contains(runCmd, "-p '0.0.0.0:3001:3001'") { + t.Fatalf("publish spec missing from the new container's run:\n%s", runCmd) + } + + // F21's core: the current web container was stopped BEFORE the + // replacement started — blue/green would have died on the fixed port. + stopIdx, runIdx := -1, -1 + for i, c := range mock.Calls { + if stopIdx < 0 && strings.HasPrefix(c, "docker stop") && strings.Contains(c, "myapp-web-v8") { + stopIdx = i + } + if runIdx < 0 && strings.HasPrefix(c, "docker run") && strings.Contains(c, "myapp-web-v9") { + runIdx = i + } + } + if stopIdx < 0 { + t.Fatalf("the current web container was never displaced\ncalls: %v", mock.Calls) + } + if runIdx < 0 || stopIdx > runIdx { + t.Errorf("displacement (idx %d) must precede the replacement run (idx %d)", stopIdx, runIdx) + } +} + +func TestDeploy_PublishRecreateFailureRestoresDisplaced(t *testing.T) { + currentState := `{"schema_version":2,"deployment_type":"container","ingress_mode":"caddy","domain":"myapp.com","updated_at":"2026-09-18T10:00:00Z","current_port":49152,"current_hash":"v8"}` + mock := deployRecordFixture("sha256:"+strings.Repeat("a", 64), + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "present\n" + currentState}, + ssh.MockCommand{Match: "docker ps --filter label=teploy.app=myapp --filter label=teploy.process=web", + Output: "myapp-web-v8\n"}, + // The new container cannot start (its fixed publish port is still + // held); the once-flag lets the displaced workload's restore run + // succeed afterwards. + ssh.MockCommand{Match: "docker run", Err: fmt.Errorf("port already allocated"), Once: true}, + // The displaced workload's restore: full inspect (Restart), rm -f, + // and a docker run that succeeds again. + ssh.MockCommand{Match: "docker inspect 'myapp-web-v8'", Output: `[{"Config":{"Image":"myapp:v8","Labels":{"teploy.app":"myapp"}},"HostConfig":{"NetworkMode":"teploy","PortBindings":{"3000/tcp":[{"HostIp":"127.0.0.1","HostPort":"49152"}]},"RestartPolicy":{"Name":"no"}},"NetworkSettings":{"Networks":{"teploy":{"Aliases":["myapp"]}}}}]`}, + ssh.MockCommand{Match: "docker rm -f 'myapp-web-v8'", Output: ""}, + ssh.MockCommand{Match: "docker run", Err: nil}, + ssh.MockCommand{Match: "docker stop", Output: ""}, + ) + var buf bytes.Buffer + err := NewDeployer(mock, &buf).Deploy(context.Background(), Config{ + App: "myapp", + Domain: "myapp.com", + Image: "myapp:v9", + Version: "v9", + Publish: []string{"0.0.0.0:3001:3001"}, + }) + if err == nil { + t.Fatalf("expected the deploy to fail\n%s", buf.String()) + } + if !strings.Contains(buf.String(), "Restored myapp-web-v8") { + t.Errorf("the displaced fixed-port workload was not restored:\n%s", buf.String()) + } +} + +// rollbackFixtureCommands is the standard two-version rollback command +// table; record is the framed meta-read response ("absent", or +// "present\n"). Compose with extra fixtures prepended when a test +// needs additional inspect forms. +func rollbackFixtureCommands(record string) []ssh.MockCommand { + stateContent := `{"schema_version":2,"deployment_type":"container","ingress_mode":"caddy","domain":"myapp.com","updated_at":"2026-07-22T10:00:00Z","current_port":49153,"current_hash":"v2","previous_port":49152,"previous_hash":"v1"}` + return []ssh.MockCommand{ + {Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "present\n" + stateContent}, + {Match: "mkdir -p /deployments/myapp", Output: ""}, + {Match: "mkdir /deployments/myapp/.lock", Output: ""}, + {Match: "cat /deployments/myapp/.lock/info", Err: fmt.Errorf("none")}, + {Match: "if [ ! -e '/deployments/myapp/meta/v1.json' ]", Output: record}, + {Match: "docker ps --all --filter label=teploy.app='myapp'", + Output: `{"ID":"aaa","Names":"myapp-web-v1","Image":"myapp:latest","State":"exited","Status":"Exited","Labels":"teploy.app=myapp,teploy.version=v1,teploy.process=web"}` + "\n" + + `{"ID":"bbb","Names":"myapp-web-v2","Image":"myapp:latest","State":"running","Status":"Up 1h","Labels":"teploy.app=myapp,teploy.version=v2,teploy.process=web"}`, + }, + {Match: "docker inspect 'myapp-web-v1'", Output: `[{"Image":"sha256:` + strings.Repeat("1", 64) + `","Config":{"Image":"myapp:v1","Env":["PORT=3000"],"Labels":{"teploy.app":"myapp"}},"HostConfig":{"NetworkMode":"teploy","PortBindings":{"3000/tcp":[{"HostIp":"127.0.0.1","HostPort":"49152"}]},"RestartPolicy":{"Name":"no"}},"NetworkSettings":{"Networks":{"teploy":{"Aliases":["myapp"]}}}}]`}, + {Match: "docker rm -f 'myapp-web-v1'", Output: ""}, + {Match: "docker run", Output: ""}, + {Match: "curl -s -o /dev/null", Output: "200"}, + {Match: "docker inspect -f '{{range $p, $b := .NetworkSettings.Ports}}{{range $b}}{{.HostIp}}", Output: "127.0.0.1 "}, + {Match: "docker inspect -f '{{range $p, $b := .NetworkSettings.Ports}}", Output: "49152"}, + {Match: "docker inspect -f '{{range $p, $_ := .NetworkSettings.Ports}}", Output: "3000/tcp"}, + {Match: "caddy", Output: ""}, + {Match: "cat /deployments/caddy/Caddyfile", Output: "{\n\tadmin 0.0.0.0:2019\n}\n"}, + {Match: "mv /tmp/teploy_caddyfile.tmp", Output: ""}, + {Match: "mkdir /deployments/caddy/.lock", Output: ""}, + {Match: "a=$(docker exec caddy md5sum", Output: "TEPLOY_CADDY_OK"}, + {Match: "docker exec caddy caddy reload", Output: ""}, + {Match: "rmdir /deployments/caddy/.lock", Output: ""}, + {Match: "docker stop", Output: ""}, + {Match: "mkdir -p", Output: ""}, + {Match: "cat /tmp", Output: ""}, + {Match: "UPLOAD:", Output: ""}, + } +} + +func rollbackRecordFixture(record string) *ssh.MockExecutor { + return ssh.NewMockExecutor("1.2.3.4", rollbackFixtureCommands(record)...) +} + +func marshalRecord(t *testing.T, rec *releasemeta.Record) string { + t.Helper() + rec.SchemaVersion = releasemeta.SchemaVersion + if rec.App == "" { + rec.App = "myapp" + } + if rec.Hash == "" { + rec.Hash = "v1" + } + data, err := json.Marshal(rec) + if err != nil { + t.Fatal(err) + } + return "present\n" + string(data) +} + +func TestRollback_RestoresFromRecordedSpec(t *testing.T) { + rec := &releasemeta.Record{ + DeploymentType: "container", IngressMode: "caddy", Domain: "rec.example.com", + Health: &releasemeta.Health{Path: "/deep", TimeoutSeconds: 30, IntervalSeconds: 1}, + Caddy: &releasemeta.CaddyRoute{ + TLSInternal: true, + CaddyExtra: "header X-Recorded yes", + Cache: map[string]string{"assets/*": "max-age=60"}, + }, + } + mock := rollbackRecordFixture(marshalRecord(t, rec)) + + var buf bytes.Buffer + cfg := rollbackCfg() // passes /health, no TLS, domain myapp.com — all wrong for v1 + if err := Rollback(context.Background(), mock, &buf, cfg); err != nil { + t.Fatalf("Rollback: %v\n%s", err, buf.String()) + } + + // The health probe used the recorded path, not the CLI-passed /health. + var healthCurl string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "curl -s -o /dev/null") { + healthCurl = c + } + } + if !strings.Contains(healthCurl, "/deep") { + t.Errorf("health probe did not use the recorded path:\n%s", healthCurl) + } + + // The Caddy block carries the recorded edge config: the recorded domain, + // internal TLS, the recorded extra directive, and the cache rule. + caddyfile := string(mock.Files["/deployments/caddy/Caddyfile"]) + for _, want := range []string{"rec.example.com {", "tls internal", "header X-Recorded yes", "max-age=60"} { + if !strings.Contains(caddyfile, want) { + t.Errorf("restored route missing recorded %q:\n%s", want, caddyfile) + } + } + if strings.Contains(caddyfile, "myapp.com {") { + t.Errorf("rollback routed the CLI-passed domain instead of the recorded one:\n%s", caddyfile) + } +} + +func TestRollback_BackfillsRecordOnFirstUse(t *testing.T) { + // No record on disk: the read comes back absent, Backfill synthesizes + // one from the live v1 containers and persists it. + mock := rollbackRecordFixture("absent") + + var buf bytes.Buffer + if err := Rollback(context.Background(), mock, &buf, rollbackCfg()); err != nil { + t.Fatalf("Rollback: %v\n%s", err, buf.String()) + } + + raw, ok := mock.Files["/deployments/myapp/meta/v1.json"] + if !ok { + t.Fatalf("backfilled record not persisted\n%s", buf.String()) + } + var rec releasemeta.Record + if err := json.Unmarshal(raw, &rec); err != nil { + t.Fatalf("invalid backfilled record: %v", err) + } + if !rec.Backfilled { + t.Error("record must be flagged backfilled") + } + // Primary port derived from the container's PORT env (3000). + if p, ok := releasemeta.PrimaryContainerPort(&rec); !ok || p != 3000 { + t.Errorf("primary container port not recovered from PORT env: %+v", rec.Ports) + } + if rec.Env["PORT"] != "3000" { + t.Errorf("resolved env not captured: %v", rec.Env) + } + if rec.ImageDigest != "sha256:"+strings.Repeat("1", 64) { + t.Errorf("image digest not captured: %s", rec.ImageDigest) + } + if rec.Health != nil || rec.Caddy != nil { + t.Errorf("backfill must not invent health/caddy config: %+v %+v", rec.Health, rec.Caddy) + } +} + +// TCL-14: with a recorded primary container port, the health check probes +// that port's host binding and Caddy dials the primary container port — an +// auxiliary published listener can neither steal the probe nor the route. +func TestRollback_RecordedPrimaryPortDrivesHealthAndUpstream(t *testing.T) { + rec := &releasemeta.Record{ + DeploymentType: "container", IngressMode: "caddy", Domain: "myapp.com", + Health: &releasemeta.Health{Path: "/health", TimeoutSeconds: 30, IntervalSeconds: 1}, + Ports: []releasemeta.Port{ + {HostPort: 49152, ContainerPort: 8080, Primary: true}, + {HostPort: 9100, ContainerPort: 9000, Bind: "0.0.0.0", Fixed: true}, + }, + } + // HostPortFor reads the full ports map: 8080 is NOT the numerically + // first key (9000 is), which is exactly the fields[0] ambiguity being + // fixed. + mock := ssh.NewMockExecutor("1.2.3.4", + append([]ssh.MockCommand{ + {Match: "docker inspect -f '{{json .NetworkSettings.Ports}}'", + Output: `{"8080/tcp":[{"HostIp":"127.0.0.1","HostPort":"49160"}],"9000/tcp":[{"HostIp":"0.0.0.0","HostPort":"9100"}]}`}, + }, rollbackFixtureCommands(marshalRecord(t, rec))...)...) + + var buf bytes.Buffer + if err := Rollback(context.Background(), mock, &buf, rollbackCfg()); err != nil { + t.Fatalf("Rollback: %v\n%s", err, buf.String()) + } + + var healthCurl string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "curl -s -o /dev/null") { + healthCurl = c + } + } + if !strings.Contains(healthCurl, ":49160") { + t.Errorf("health probe did not target the primary port's host binding (49160):\n%s", healthCurl) + } + caddyfile := string(mock.Files["/deployments/caddy/Caddyfile"]) + if !strings.Contains(caddyfile, "myapp-web-v1:8080") { + t.Errorf("Caddy upstream is not the recorded primary container port (8080):\n%s", caddyfile) + } +} + +// F21 rollback: a target release with recorded publish entries takes the +// recreate order — the current web containers stop before the target's +// containers restart, because the fixed publish ports cannot double-bind. +func TestRollback_FixedPortTargetDisplacesCurrent(t *testing.T) { + rec := &releasemeta.Record{ + DeploymentType: "container", IngressMode: "caddy", Domain: "myapp.com", + Health: &releasemeta.Health{Path: "/health", TimeoutSeconds: 30, IntervalSeconds: 1}, + Publish: []string{"0.0.0.0:3001:3001"}, + Ports: []releasemeta.Port{ + {HostPort: 49152, ContainerPort: 3000, Primary: true}, + {HostPort: 3001, ContainerPort: 3001, Bind: "0.0.0.0", Fixed: true}, + }, + } + mock := ssh.NewMockExecutor("1.2.3.4", + append([]ssh.MockCommand{ + {Match: "docker inspect -f '{{json .NetworkSettings.Ports}}'", + Output: `{"3000/tcp":[{"HostIp":"127.0.0.1","HostPort":"49152"}]}`}, + }, rollbackFixtureCommands(marshalRecord(t, rec))...)...) + + var buf bytes.Buffer + if err := Rollback(context.Background(), mock, &buf, rollbackCfg()); err != nil { + t.Fatalf("Rollback: %v\n%s", err, buf.String()) + } + if !strings.Contains(buf.String(), "Freed the fixed port") { + t.Errorf("current web container was not displaced for the fixed-port target:\n%s", buf.String()) + } + stopIdx, runIdx := -1, -1 + for i, c := range mock.Calls { + if stopIdx < 0 && strings.HasPrefix(c, "docker stop") && strings.Contains(c, "myapp-web-v2") { + stopIdx = i + } + if runIdx < 0 && strings.HasPrefix(c, "docker run") && strings.Contains(c, "myapp-web-v1") { + runIdx = i + } + } + if stopIdx < 0 || runIdx < 0 || stopIdx > runIdx { + t.Errorf("displacement (idx %d) must precede the target recreation (idx %d)", stopIdx, runIdx) + } +} + +// --- static: F13 state-only rollback on top of the F14 record --- + +// staticMetaFixture returns the framed meta-read response for hash. +func staticMetaFixture(t *testing.T, rec *releasemeta.Record, hash string) string { + t.Helper() + rec.SchemaVersion = releasemeta.SchemaVersion + rec.App = "myapp" + rec.Hash = hash + data, err := json.Marshal(rec) + if err != nil { + t.Fatal(err) + } + return "present\n" + string(data) +} + +func TestStaticDeploy_RecordsReleaseMetadata(t *testing.T) { + src := staticTestSource(t) + hash, err := hashDir(src) + if err != nil { + t.Fatal(err) + } + shortHash := hash[:12] + + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "mkdir -p /deployments/myapp", Output: ""}, + ssh.MockCommand{Match: "mkdir /deployments/myapp/.lock", Output: ""}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "absent"}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/state' ]", Output: "present\n" + ""}, + ssh.MockCommand{Match: "mkdir -p /deployments/myapp/releases", Output: ""}, + ssh.MockCommand{Match: "test -d /deployments/myapp/releases/", Output: "yes"}, + ssh.MockCommand{Match: "ln -s -- releases/", Output: ""}, + ssh.MockCommand{Match: "curl -sf http://localhost:2019/config/apps/http/servers/srv0", Output: `{"listen":[":80",":443"]}`}, + ssh.MockCommand{Match: "curl -sf -X DELETE", Err: fmt.Errorf("not found")}, + ssh.MockCommand{Match: "cat /deployments/caddy/Caddyfile", Output: "{\n\tadmin 0.0.0.0:2019\n}\n"}, + ssh.MockCommand{Match: "mv /tmp/teploy_caddyfile.tmp", Output: ""}, + ssh.MockCommand{Match: "mkdir /deployments/caddy/.lock", Output: ""}, + ssh.MockCommand{Match: "a=$(docker exec caddy md5sum", Output: "TEPLOY_CADDY_OK"}, + ssh.MockCommand{Match: "rmdir /deployments/caddy/.lock", Output: ""}, + ssh.MockCommand{Match: "docker exec caddy caddy reload", Output: ""}, + ssh.MockCommand{Match: "ls -1t /deployments/myapp/releases", Output: ""}, + ssh.MockCommand{Match: "rm -rf /deployments/myapp/.lock", Output: ""}, + ssh.MockCommand{Match: "UPLOAD:", Output: ""}, + ssh.MockCommand{Match: "rm -f", Output: ""}, + ssh.MockCommand{Match: "mkdir -p", Output: ""}, + ssh.MockCommand{Match: "mv", Output: ""}, + ssh.MockCommand{Match: "cat", Output: ""}, + ) + + var buf bytes.Buffer + err = NewStaticDeployer(mock, &buf).Deploy(context.Background(), StaticConfig{ + App: "myapp", + Domain: "myapp.com", + Source: src, + SPA: true, + Headers: map[string]string{"X-Static": "1"}, + Cache: map[string]string{"assets/*": "max-age=3600"}, + }) + if err != nil { + t.Fatalf("Deploy: %v", err) + } + + raw, ok := mock.Files["/deployments/myapp/meta/"+shortHash+".json"] + if !ok { + t.Fatalf("static release record not written for %s", shortHash) + } + var rec releasemeta.Record + if err := json.Unmarshal(raw, &rec); err != nil { + t.Fatalf("invalid static record: %v", err) + } + if rec.DeploymentType != "static" || rec.IngressMode != "caddy" || rec.Hash != shortHash { + t.Errorf("identity fields wrong: %+v", rec) + } + if rec.Static == nil { + t.Fatalf("static serving config not recorded") + } + if !rec.Static.SPA || rec.Static.Domain != "myapp.com" || + rec.Static.Headers["X-Static"] != "1" || rec.Static.Cache["assets/*"] != "max-age=3600" { + t.Errorf("serving config not recorded faithfully: %+v", rec.Static) + } +} + +func TestRollbackStateOnly_RestoresRecordedServingConfig(t *testing.T) { + stateContent := `{"schema_version":2,"deployment_type":"static","ingress_mode":"caddy","domain":"myapp.com","updated_at":"2026-09-18T10:00:00Z","current_hash":"aaa111","previous_hash":"bbb222"}` + rec := &releasemeta.Record{ + DeploymentType: "static", IngressMode: "caddy", + Static: &releasemeta.Static{ + Domain: "rec.example.com", + SPA: true, + SPAFallback: "/index.html", + Headers: map[string]string{"X-Rec": "1"}, + Cache: map[string]string{"assets/*": "max-age=99"}, + CaddyExtra: "header X-Extra 2", + }, + } + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "mkdir /deployments/myapp/.lock", Output: ""}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "present\n" + stateContent}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/bbb222.json' ]", Output: staticMetaFixture(t, rec, "bbb222")}, + ssh.MockCommand{Match: "test -d /deployments/myapp/releases/bbb222", Output: "yes"}, + ssh.MockCommand{Match: "ln -s -- releases/", Output: ""}, + ssh.MockCommand{Match: "curl -sf http://localhost:2019/config/apps/http/servers/srv0", Output: `{"listen":[":80",":443"]}`}, + ssh.MockCommand{Match: "curl -sf -X DELETE", Err: fmt.Errorf("not found")}, + ssh.MockCommand{Match: "cat /deployments/caddy/Caddyfile", Output: "{\n\tadmin 0.0.0.0:2019\n}\n"}, + ssh.MockCommand{Match: "mv /tmp/teploy_caddyfile.tmp", Output: ""}, + ssh.MockCommand{Match: "mkdir /deployments/caddy/.lock", Output: ""}, + ssh.MockCommand{Match: "a=$(docker exec caddy md5sum", Output: "TEPLOY_CADDY_OK"}, + ssh.MockCommand{Match: "rmdir /deployments/caddy/.lock", Output: ""}, + ssh.MockCommand{Match: "docker exec caddy caddy reload", Output: ""}, + ssh.MockCommand{Match: "printf %s", Output: ""}, + ssh.MockCommand{Match: "rm -rf /deployments/myapp/.lock", Output: ""}, + ssh.MockCommand{Match: "UPLOAD:", Output: ""}, + ssh.MockCommand{Match: "mkdir -p", Output: ""}, + ssh.MockCommand{Match: "mv -Tf", Output: ""}, + ssh.MockCommand{Match: "mv", Output: ""}, + ssh.MockCommand{Match: "rm -f", Output: ""}, + ) + + var buf bytes.Buffer + if err := NewStaticDeployer(mock, &buf).RollbackStateOnly(context.Background(), "myapp", ""); err != nil { + t.Fatalf("RollbackStateOnly: %v\n%s", err, buf.String()) + } + + // The Caddyfile block was rebuilt from the RECORD: recorded domain, SPA + // fallback, headers, cache, and extra directive — none of which exist in + // state.json. + caddyfile := string(mock.Files["/deployments/caddy/Caddyfile"]) + for _, want := range []string{"rec.example.com {", "try_files {path} {path}/ {path}/index.html /index.html", "X-Rec", "max-age=99", "header X-Extra 2"} { + if !strings.Contains(caddyfile, want) { + t.Errorf("restored static block missing recorded %q:\n%s", want, caddyfile) + } + } + if strings.Contains(caddyfile, "myapp.com {") { + t.Errorf("state's domain was served instead of the recorded one:\n%s", caddyfile) + } + + // The symlink flipped to the target release and state swapped. + var swap string + for _, c := range mock.Calls { + if strings.HasPrefix(c, "ln -s -- releases/") { + swap = c + } + } + if !strings.Contains(swap, "releases/bbb222") { + t.Errorf("current symlink did not flip to bbb222: %s", swap) + } + stateData := writtenState(t, mock, "myapp") + if stateData.CurrentHash != "bbb222" || stateData.PreviousHash != "aaa111" { + t.Errorf("state not swapped: %+v", stateData) + } +} + +// Without a record there is nothing faithful to restore — static serving +// config is unrecoverable from disk (F48 territory) — so the request is a +// redeploy, never a guess. +func TestRollbackStateOnly_NoRecordFailsClosed(t *testing.T) { + stateContent := `{"schema_version":2,"deployment_type":"static","ingress_mode":"caddy","domain":"myapp.com","updated_at":"2026-09-18T10:00:00Z","current_hash":"aaa111","previous_hash":"bbb222"}` + mock := ssh.NewMockExecutor("1.2.3.4", + ssh.MockCommand{Match: "mkdir /deployments/myapp/.lock", Output: ""}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/state.json' ]", Output: "present\n" + stateContent}, + ssh.MockCommand{Match: "if [ ! -e '/deployments/myapp/meta/bbb222.json' ]", Output: "absent"}, + ) + + err := NewStaticDeployer(mock, &bytes.Buffer{}).RollbackStateOnly(context.Background(), "myapp", "") + if err == nil || !strings.Contains(err.Error(), "no recorded release metadata") { + t.Fatalf("expected fail-closed no-record error, got %v", err) + } + if !strings.Contains(err.Error(), "redeploy") { + t.Errorf("error must say what to do next: %v", err) + } + for _, c := range mock.Calls { + if strings.HasPrefix(c, "ln -s -- releases/") { + t.Errorf("the symlink was touched despite refusing: %s", c) + } + } +} diff --git a/internal/deploy/rollback.go b/internal/deploy/rollback.go index 30b6a9d..90d2146 100644 --- a/internal/deploy/rollback.go +++ b/internal/deploy/rollback.go @@ -10,6 +10,7 @@ import ( "github.com/useteploy/teploy/internal/caddy" "github.com/useteploy/teploy/internal/docker" + "github.com/useteploy/teploy/internal/releasemeta" "github.com/useteploy/teploy/internal/ssh" "github.com/useteploy/teploy/internal/state" ) @@ -147,6 +148,30 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac return fmt.Errorf("no web container found for version %s", target) } + // 3b. Load the target release's recorded spec (F14) — the rollback + // restores FROM THE RECORD, not from whatever the current teploy.yml + // happens to say. A confirmed-missing record is backfilled from the + // live containers (release-0 migration). The store must never be why a + // rollback that worked before stops working: an UNREADABLE record (or a + // failed backfill) degrades to the historical inspect-driven path with a + // warning, and only the genuinely recoverable overlays are skipped. + rec, recWarn := loadReleaseRecord(ctx, exec, dk, containers, cfg.App, target, current) + if recWarn != nil { + fmt.Fprintf(out, "Warning: %v — proceeding from live container inspection\n", recWarn) + } + if rec != nil { + applyRecordToRollback(&cfg, rec, &healthCfg) + } + + // The recorded primary container port (TCL-14): with more than one + // published port, a fields[0] pick cannot tell the HTTP surface from an + // auxiliary listener. Health checks probe the primary's host binding and + // Caddy dials the primary container port. + primaryContainerPort := 0 + if rec != nil { + primaryContainerPort, _ = releasemeta.PrimaryContainerPort(rec) + } + // Ports the current (about-to-be-stopped) version is live on right now — // must not be handed to the target's recreated containers. A single-hop // rollback could never collide (the immediately-previous version's port @@ -157,15 +182,16 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac // container. See docker.Client.Restart's doc comment for how this is // resolved (fresh port allocated instead of reusing a colliding one). // - // EXCEPT under host ingress, where this reasoning inverts. There the port is - // fixed by config, so every version shares it and current.CurrentPort ALWAYS - // equals the target's port — avoiding it would reallocate on every single - // rollback, silently republishing the app on a random ephemeral port. That is - // not a rare collision case, it is guaranteed, and it made `teploy rollback` - // unusable for any host-ingress app. The fixed port is freed instead, by + // EXCEPT under fixed host ports (host ingress, or an app with publish + // entries — F21), where this reasoning inverts. There the ports are + // fixed by config, so every version shares them and the current + // version's port ALWAYS equals the target's — avoiding it would + // reallocate on every single rollback, silently republishing the app on + // a random ephemeral port. The fixed ports are freed instead, by // stopping the current web containers before the target starts (below). + fixedPorts := cfg.ingressHost() || releasemeta.HasFixedHostPorts(rec) avoidPorts := make(map[int]bool, len(current.CurrentPorts)+1) - if !cfg.ingressHost() { + if !fixedPorts { for _, p := range current.CurrentPorts { avoidPorts[p] = true } @@ -179,10 +205,11 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac // not "-" — a suffix match silently skips every replica (leaving // them stopped on rollback / orphaned on the next deploy). Every teploy // container carries the version label. - // Host ingress recreates rather than blue/greens — the target reuses the same - // fixed host port, so the current web containers must stop before it starts. - // Deploy does exactly this (see deploy.go's displacedHostWeb); rollback did - // not, which is why it only ever "worked" by reallocating the port. + // Fixed host ports (host ingress or publish entries — F21) recreate + // rather than blue/green — the target reuses the same fixed port(s), so + // the current web containers must stop before it starts. Deploy does + // exactly this (see deploy.go's displacedHostWeb); rollback did not, + // which is why it only ever "worked" by reallocating the port. // // Recorded so a failed health check can bring them back: with a fixed port // there is no moment where both versions are live, so the window between @@ -203,7 +230,7 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac } } } - if cfg.ingressHost() { + if fixedPorts { for _, c := range containers { // Displace only the RUNNING web containers of the AUTHORITATIVE // current generation (TCL-07). The old filter (any non-target @@ -277,7 +304,19 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac // rollback of a host-ingress app with a bind address. healthBindHost := "" for _, c := range targetWeb { - p, err := dk.HostPort(ctx, c.Name) + var p int + // TCL-14: prefer the binding of the recorded PRIMARY container port + // over the first field docker's map happens to print — a multi-port + // container's auxiliary listeners must not steal the health probe. + if primaryContainerPort > 0 { + if hp, herr := dk.HostPortFor(ctx, c.Name, primaryContainerPort); herr == nil { + p = hp + } else { + p, err = dk.HostPort(ctx, c.Name) + } + } else { + p, err = dk.HostPort(ctx, c.Name) + } if err != nil { // A failed inspect must not strand the displaced fixed-port // workload (F12). @@ -323,10 +362,19 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac if cfg.usesCaddy() { fmt.Fprintln(out, "Updating routes...") tls := caddy.TLS{Cert: cfg.TLSCert, Key: cfg.TLSKey, Internal: cfg.TLSInternal} + // The Caddy upstream port is the recorded primary container port + // when there is one (TCL-14); without a record the first exposed + // port remains the historical best-effort pick. + upstreamPort := func(name string) (int, error) { + if primaryContainerPort > 0 { + return primaryContainerPort, nil + } + return dk.InternalPort(ctx, name) + } if len(targetWeb) > 1 { upstreams := make([]caddy.Upstream, 0, len(targetWeb)) for _, c := range targetWeb { - port, err := dk.InternalPort(ctx, c.Name) + port, err := upstreamPort(c.Name) if err != nil { return fmt.Errorf("inspecting target container port: %w", err) } @@ -337,7 +385,7 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac } fmt.Fprintf(out, " Traffic load-balanced across %d replicas\n", len(targetWeb)) } else { - port, err := dk.InternalPort(ctx, targetWeb[0].Name) + port, err := upstreamPort(targetWeb[0].Name) if err != nil { return fmt.Errorf("inspecting target container port: %w", err) } @@ -388,11 +436,11 @@ func Rollback(ctx context.Context, exec ssh.Executor, out io.Writer, cfg Rollbac newState.ImageDigest = digest } if err := state.Write(ctx, exec, cfg.App, newState); err != nil { - // Host ingress: the target holds the fixed port. Stop it, restore the + // Fixed host ports: the target holds them. Stop it, restore the // displaced workload, then remove the uncommitted target — previously // this branch skipped the restore and still claimed "the original // workload was left running" (audit F12). - if cfg.ingressHost() { + if fixedPorts { for _, name := range started { dk.Stop(ctx, name, 5) } @@ -472,3 +520,63 @@ func restoreRollbackRoute(ctx context.Context, cd *caddy.Client, dk *docker.Clie } return cd.SetLoadBalancerHealth(ctx, cfg.App, domain, upstreams, cfg.Health.withDefaults().Path, tls, cfg.CaddyExtra, cfg.Cache, cfg.Firewall, cfg.Access) } + +// loadReleaseRecord resolves the F14 record for the rollback target, +// backfilling a "release-0" record from the live containers when the release +// predates the store (the convergence migration). The returned warning is +// non-fatal by contract: rollback worked before the store existed and must +// keep working without it — an unreadable record or failed backfill means +// the overlays are skipped, never that the rollback refuses. +func loadReleaseRecord(ctx context.Context, exec ssh.Executor, dk *docker.Client, containers []docker.Container, app, hash string, current *state.AppState) (*releasemeta.Record, error) { + rec, err := releasemeta.Read(ctx, exec, app, hash) + if err != nil { + return nil, err + } + if rec == nil { + rec, err = releasemeta.Backfill(ctx, exec, dk, containers, app, hash, current) + if err != nil { + return nil, fmt.Errorf("backfilling release metadata for %s@%s: %w", app, hash, err) + } + } + return rec, nil +} + +// applyRecordToRollback overlays the recorded release spec onto the rollback +// config: the record is the target release's truth, the CLI-passed values +// (from the current teploy.yml) are the fallback. Wholesale replacement per +// field group — an empty recorded TLS block MEANS "this release had no +// custom TLS" and must not be silently upgraded to whatever the current +// config says. Backfilled records carry no health/caddy data (unrecoverable +// from containers), so those overlays simply don't fire for them. +func applyRecordToRollback(cfg *RollbackConfig, rec *releasemeta.Record, healthCfg *HealthConfig) { + if rec.IngressMode != "" { + cfg.Ingress = rec.IngressMode + } + if rec.Domain != "" { + cfg.Domain = rec.Domain + } + if rec.Health != nil { + if rec.Health.Path != "" { + healthCfg.Path = rec.Health.Path + } + if rec.Health.TimeoutSeconds > 0 { + healthCfg.Timeout = time.Duration(rec.Health.TimeoutSeconds) * time.Second + } + if rec.Health.IntervalSeconds > 0 { + healthCfg.Interval = time.Duration(rec.Health.IntervalSeconds) * time.Second + } + } + if rec.Caddy != nil { + cfg.TLSCert = rec.Caddy.TLSCert + cfg.TLSKey = rec.Caddy.TLSKey + cfg.TLSInternal = rec.Caddy.TLSInternal + cfg.CaddyExtra = rec.Caddy.CaddyExtra + cfg.Cache = rec.Caddy.Cache + if rec.Caddy.Firewall != nil { + cfg.Firewall = *rec.Caddy.Firewall + } + if rec.Caddy.Access != nil { + cfg.Access = *rec.Caddy.Access + } + } +} diff --git a/internal/deploy/rollback_test.go b/internal/deploy/rollback_test.go index 35d0649..d9eca01 100644 --- a/internal/deploy/rollback_test.go +++ b/internal/deploy/rollback_test.go @@ -632,15 +632,21 @@ func TestRollback_HostIngressKeepsTheFixedPort(t *testing.T) { } // Caddy ingress keeps blue/green: the current version's port must still be -// avoided there, since two versions genuinely do run at once. +// avoided there, since two versions genuinely do run at once. The guard now +// keys on fixedPorts (host ingress OR recorded fixed/publish ports — F21); +// a plain caddy app computes fixedPorts=false and keeps avoiding the live +// port. Pin both the derivation and the guard so neither regresses. func TestRollback_CaddyIngressStillAvoidsTheLivePort(t *testing.T) { body, err := osReadFile("rollback.go") if err != nil { t.Fatalf("read rollback.go: %v", err) } src := string(body) - if !strings.Contains(src, "if !cfg.ingressHost() {") { - t.Error("the avoidPorts guard is not conditioned on host ingress; caddy rollbacks need the live port avoided") + if !strings.Contains(src, "fixedPorts := cfg.ingressHost() || releasemeta.HasFixedHostPorts(rec)") { + t.Error("fixedPorts must derive from host ingress or recorded fixed/publish ports only — anything else would stop a plain caddy rollback avoiding the live port") + } + if !strings.Contains(src, "if !fixedPorts {") { + t.Error("the avoidPorts guard is not conditioned on fixedPorts; caddy rollbacks need the live port avoided") } } diff --git a/internal/deploy/static.go b/internal/deploy/static.go index 5c2c1dd..61d6099 100644 --- a/internal/deploy/static.go +++ b/internal/deploy/static.go @@ -33,6 +33,7 @@ import ( "time" "github.com/useteploy/teploy/internal/caddy" + "github.com/useteploy/teploy/internal/releasemeta" "github.com/useteploy/teploy/internal/ssh" "github.com/useteploy/teploy/internal/state" ) @@ -249,6 +250,31 @@ func (d *StaticDeployer) Deploy(ctx context.Context, cfg StaticConfig) error { return d.abortStaticStateCommit(ctx, cfg.App, cfg.StateDir, currentLink, prior, err) } + // 10b. Record the release metadata (F14): the serving configuration is + // exactly the piece state.json never retained, and it is what makes a + // state-only static rollback possible (F13). Warning-only — see the + // container path's recordRelease for the posture. + staticRec := &releasemeta.Record{ + App: cfg.App, + Hash: shortHash, + Generation: newState.Generation, + DeploymentType: "static", + IngressMode: "caddy", + Domain: cfg.Domain, + Static: &releasemeta.Static{ + Domain: cfg.Domain, + SPA: cfg.SPA, + SPAFallback: cfg.SPAFallback, + Cache: cfg.Cache, + Headers: cfg.Headers, + CaddyExtra: cfg.CaddyExtra, + KeepReleases: cfg.KeepReleases, + }, + } + if err := releasemeta.Write(ctx, d.exec, staticRec); err != nil { + fmt.Fprintf(d.out, " warning: could not record release metadata for %s@%s: %v (state-only rollback for this release will ask for a redeploy first)\n", cfg.App, shortHash, err) + } + // 11. Prune old releases (keep the most recent KeepReleases including // current). Best-effort; failure here doesn't fail the deploy. prev := priorHashOrEmpty(prior, shortHash) @@ -597,18 +623,27 @@ func (d *StaticDeployer) Rollback(ctx context.Context, cfg StaticRollbackConfig) return errors.New("no state found — has this app been deployed?") } - target := cfg.ToHash - if target == "" { - target = prior.PreviousHash - } - if target == "" { - return ErrNoPreviousDeploy - } - if target == prior.CurrentHash { - return fmt.Errorf("target hash %s is already current", target) + target, err := staticRollbackTarget(prior, cfg.ToHash) + if err != nil { + return err } - if !safeReleaseName(target) { - return fmt.Errorf("invalid target release %q — expected a release id from `teploy releases`", target) + + // The recorded serving config is the target release's truth (F13/F14); + // the teploy.yml-passed options are the fallback for releases that + // predate the store. Absent record → keep what the caller passed; + // unreadable record → warn and keep the caller's config (a static + // rollback can proceed with explicit config, unlike RollbackStateOnly). + if rec, rerr := releasemeta.Read(ctx, d.exec, cfg.App, target); rerr != nil { + fmt.Fprintf(d.out, "Warning: %v — using the serving config from the command line\n", rerr) + } else if rec != nil && rec.Static != nil { + cfg.SPA = rec.Static.SPA + cfg.SPAFallback = rec.Static.SPAFallback + cfg.Cache = rec.Static.Cache + cfg.Headers = rec.Static.Headers + cfg.CaddyExtra = rec.Static.CaddyExtra + if rec.Static.Domain != "" { + cfg.Domain = rec.Static.Domain + } } releasesDir := fmt.Sprintf("%s/%s/releases", cfg.StateDir, cfg.App) @@ -667,6 +702,114 @@ func (d *StaticDeployer) Rollback(ctx context.Context, cfg StaticRollbackConfig) return nil } +// RollbackStateOnly rolls a static app back with NO teploy.yml: server-side +// state plus the recorded release metadata supply everything the serving +// config used to (F13 on top of F14). This is the path `teploy rollback +// --app --host ...` takes — before the record existed it could only +// refuse, because SPA/headers/cache live nowhere in state. +// +// Fail-closed on the record: unlike a container rollback (which can rebuild +// from live inspection), a static release's serving config is unrecoverable +// from disk — the Caddyfile block is string fragments (F48) and the release +// tree says nothing about how it was served. No record means asking for one +// redeploy, not guessing. +func (d *StaticDeployer) RollbackStateOnly(ctx context.Context, app, toHash string) error { + if err := state.AcquireLock(ctx, d.exec, app); err != nil { + return fmt.Errorf("acquire lock: %w", err) + } + defer state.ReleaseLockDetached(d.exec, app) + + prior, err := state.Read(ctx, d.exec, app) + if err != nil { + return fmt.Errorf("reading state for %s: %w", app, err) + } + if prior == nil { + return errors.New("no state found — has this app been deployed?") + } + + target, err := staticRollbackTarget(prior, toHash) + if err != nil { + return err + } + + rec, err := releasemeta.Read(ctx, d.exec, app, target) + if err != nil { + return fmt.Errorf("reading release metadata for %s@%s: %w", app, target, err) + } + if rec == nil || rec.Static == nil { + return fmt.Errorf("no recorded release metadata for %s@%s — static rollback needs the recorded serving config; redeploy the release once with this CLI version to record it, then roll back", app, target) + } + st := rec.Static + + releasesDir := fmt.Sprintf("%s/%s/releases", DefaultStateDir, app) + if out, _ := d.exec.Run(ctx, fmt.Sprintf("test -d %s/%s && echo yes || true", releasesDir, target)); strings.TrimSpace(out) != "yes" { + return fmt.Errorf("release %s no longer on server (may have been pruned)", target) + } + + if err := d.swapCurrentLink(ctx, app, DefaultStateDir, target); err != nil { + return fmt.Errorf("symlink swap: %w", err) + } + + domain := st.Domain + if domain == "" { + domain = prior.Domain + } + if err := d.caddy.SetStaticRoute(ctx, app, domain, caddy.StaticBlockOpts{ + Root: fmt.Sprintf("%s/%s/current", DefaultStaticMount, app), + SPA: st.SPA, + SPAFallback: st.SPAFallback, + Cache: st.Cache, + Headers: st.Headers, + CaddyExtra: st.CaddyExtra, + }); err != nil { + if restoreErr := d.restoreStaticLink(ctx, app, DefaultStateDir, prior); restoreErr != nil { + return fmt.Errorf("caddy route: %w; restoring release %s failed: %v; the target release was left active", err, prior.CurrentHash, restoreErr) + } + return fmt.Errorf("caddy route: %w; release %s was restored", err, prior.CurrentHash) + } + + newState := state.NewAppliedState(prior, "static", "caddy", domain) + newState.CurrentHash = target + newState.PreviousHash = prior.CurrentHash + if prior.PreviousRelease != nil && prior.PreviousRelease.Hash == target { + newState.ApplyRelease(prior.PreviousRelease) + } + if err := state.Write(ctx, d.exec, app, newState); err != nil { + if restoreErr := d.restoreStaticLink(ctx, app, DefaultStateDir, prior); restoreErr != nil { + return fmt.Errorf("committing authoritative applied state after static rollback route switch: %w; restoring release %s failed: %v; the target release was left active", err, prior.CurrentHash, restoreErr) + } + return fmt.Errorf("committing authoritative applied state after static rollback route switch: %w; release %s was restored and state remains unchanged", err, prior.CurrentHash) + } + state.AppendLog(ctx, d.exec, state.LogEntry{ + Timestamp: time.Now().UTC(), + App: app, + Type: "rollback", + Hash: target, + Success: true, + }) + fmt.Fprintf(d.out, "Rolled back %s to %s (from recorded release metadata)\n", app, target) + return nil +} + +// staticRollbackTarget resolves the rollback target for a static app from +// state: an explicit --to hash, else the recorded previous release. +func staticRollbackTarget(prior *state.AppState, toHash string) (string, error) { + target := toHash + if target == "" { + target = prior.PreviousHash + } + if target == "" { + return "", ErrNoPreviousDeploy + } + if target == prior.CurrentHash { + return "", fmt.Errorf("target hash %s is already current", target) + } + if !safeReleaseName(target) { + return "", fmt.Errorf("invalid target release %q — expected a release id from `teploy releases`", target) + } + return target, nil +} + // ListReleases returns the retained releases on disk for an app, newest // first. Used by `teploy releases `. func (d *StaticDeployer) ListReleases(ctx context.Context, app, stateDir string) ([]string, error) { From 58732a7e8e103730b169d96f1186a90a1704f80f Mon Sep 17 00:00:00 2001 From: Tyler <53561637+im-tyler@users.noreply.github.com> Date: Fri, 18 Sep 2026 04:13:27 -0700 Subject: [PATCH 4/4] docs(audit): F14 + family resolved; dependents annotated F14 (store), F13 (state-only rollback incl. static serving), F20 (full RecreateSpec), F21 (publish recreate strategy), TCL-09/TCL-13/TCL-14 resolved with their commits. F04/TCL-10/TCL-15 annotated 'unblocked by F14, design remains'; F08 notes the record as its keying surface; F60 annotated largely-delivered with the access-contract residual; TCL-39's F13 half landed, F48 remains. F48/F49/F16/F57 untouched. --- AUDIT_OPEN.md | 132 ++++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 106 insertions(+), 26 deletions(-) diff --git a/AUDIT_OPEN.md b/AUDIT_OPEN.md index ea21200..8683f85 100644 --- a/AUDIT_OPEN.md +++ b/AUDIT_OPEN.md @@ -12,10 +12,13 @@ contained fixes (several narrowing pass-6 deferrals), the rest deferred — almost all of them the same architectural tail pass 6 already carries, now with the round-2 evidence folded in. -Open items: pass-6 deferred tail (29 sub-items across 24 findings), plus -the round-2 residual tail itemized in that section. The 2 upstream/owner -items are closed (below). The two upstream items received from -teploy-dash's 2026-09-17 pass are closed below. +Open items: the pass-6 deferred tail minus the F14 family (resolved +2026-09-18, bottom section: F13/F14/F20/F21 plus round-2's TCL-09/TCL-13/ +TCL-14 folded into them), the round-2 residual tail itemized in that +section, and the dependent designs F04/F48/F49/F57/F16 that stay deferred +with their annotations. The 2 upstream/owner items are closed (below). The +two upstream items received from teploy-dash's 2026-09-17 pass are closed +below. ## Resolved from this register @@ -112,25 +115,32 @@ defect could corrupt data today. - F04 — Generation-scoped container identities + a RouteSwitch handoff boundary (candidates unroutable before readiness, external-ingress policy). The contained F03 fix removed the deletion defect; the alias - handoff redesign spans deploy, rollback, and Caddy together. + handoff redesign spans deploy, rollback, and Caddy together. Unblocked + by F14 (the per-release record can now key generation-scoped identities + and attempt artifacts), design remains. - F08 — Attempt-scoped immutable artifacts (build dirs, env files, TLS) under one lease. The F07 lock split serializes container mutation; full - artifact generation needs F04's generation IDs. -- F13 — State-only rollback restoring a complete target-release spec - (incl. static serving config). Requires F14's per-release metadata. -- F14 — Per-release immutable metadata store keyed by target ID (beyond - the current one-level PreviousRelease). Schema + migration design. + artifact generation needs F04's generation IDs. F14's record exists as + the keying surface; the lease/generation design remains. +- F13 — RESOLVED 2026-09-18 with F14 (see the F14-family section at the + bottom): state-only rollback restores the complete target-release spec, + including static serving config. +- F14 — RESOLVED 2026-09-18 (see the F14-family section at the bottom): + per-release immutable metadata store keyed target+release at + /deployments//meta/.json, with live-container backfill so + existing installs converge. - F16 — Owner-token fenced locks with renewal (age-based breaking retained: TTLs are documented and manual locks never expire; a fencing redesign can strand apps mid-incident if it ships wrong). - F17 — Commands as argv arrays end-to-end (Cmd stays a deliberate operator-authored shell string; validated sources are quoted at the sinks). -- F20 — Full RecreateSpec/Engine-API recreation preserving every inspect - field (immutable image ID + quoted binds landed; full spec needs F14). -- F21 — Explicit recreate strategy for `publish` (product decision: the - current failure mode is a clean pre-mutation docker-run error, not an - outage). +- F20 — RESOLVED 2026-09-18 with F14 (see the F14-family section at the + bottom): full RecreateSpec recreation preserving every inspect field + the docker CLI can represent. +- F21 — RESOLVED 2026-09-18 with F14 (see the F14-family section at the + bottom): publish takes the explicit recreate strategy on deploy and + rollback. - F22 — Remaining argv exposure in `docker exec` channels (BAO_TOKEN, MYSQL_PWD, Restart -e): docker exec has no --env-file; needs container-side file plumbing shared with the engine images. @@ -164,7 +174,10 @@ defect could corrupt data today. - F57 — Presence-aware overlay semantics (explicit clearing of maps/lists, null handling) — schema design decision. - F60 — Protected full execution spec per release (public redacted - snapshot keeps its current role). + snapshot keeps its current role). Largely delivered 2026-09-18 by F14's + per-release record (full execution spec, 0600, server-side, embedded + RecreateSpec); if the intent included a reader-facing access contract + (CLI/dash surfacing of the record), that design remains. - F62 — Update selection policy (prereleases/downgrades) + extraction member bounds. - F63 — Supply-chain pinning. Workflow action pins landed 2026-09-17 @@ -283,26 +296,33 @@ into each rather than duplicated as new work items. - TCL-03 — F04 (generation-scoped identities / RouteSwitch handoff) + F21 (publish recreate strategy). Contained pieces already landed: candidate - name dedup (F03), publish+replicas rejected at validation. + name dedup (F03), publish+replicas rejected at validation. F21 RESOLVED + 2026-09-18 (F14-family section); the F04 remainder is unblocked by F14, + design remains. - TCL-04 — F08 (attempt-scoped immutable artifacts under one lease). - TCL-05 — F16 (fencing/renewal). Containment added this round: the caddy lock release is detached/bounded so cancellation cannot strand it. - TCL-08 — F05/F35 tail (durable journal). Contained pieces landed this round: cleanup failure reporting, caddy verify-failure compensation. -- TCL-09 — F14/F13 (immutable per-release execution record). +- TCL-09 — RESOLVED 2026-09-18: F14/F13 landed (immutable per-release + execution record; state-only rollback restores from it — F14-family + section below). - TCL-10 — ingress-transition transaction (caddy→host/external leaves the old route behind). Needs F04's generation handoff; a plain "reject ingress changes" would break the documented host-migration flow. Folded - into F04. + into F04. Unblocked by F14, design remains. - TCL-12 — F22, narrowed: docker-run channels COULD lift env to --env-file (Restart -e from inspect is now the registered contained follow-up); docker exec channels (BAO_TOKEN, MYSQL_PWD) remain blocked on container-side file plumbing shared with the engine images. -- TCL-13 — F20 (full RecreateSpec preservation; needs F14). -- TCL-14 — needs the recorded primary-port contract (F14 metadata); - multi-port ambiguity is real but not fixable contained without it. +- TCL-13 — RESOLVED 2026-09-18: F20 landed (full RecreateSpec preservation + — F14-family section below). +- TCL-14 — RESOLVED 2026-09-18: the recorded primary-port contract landed + (ports with primary designation in the F14 record; health checks and + Caddy targets resolve through it — F14-family section below). - TCL-15 — port allocation redesign (Docker-ephemeral publish + inspect). - With F14. + Unblocked by F14 (the record now carries the resolved port allocation + per release), design remains. - TCL-17 — F47 tail (explicit HTTP/TCP/auto probe modes; the 404/3xx TCP fallback is documented deliberate compat). - TCL-24 — F49 tail (foreign-block adoption by brace counting; parser/ @@ -321,9 +341,11 @@ into each rather than duplicated as new work items. drift detection). - TCL-37 — F05/F37 tail (readiness-gated accessory upgrade with verified recovery). -- TCL-39 — static route-policy restore needs F13/F48; renderer input - hardening (header-name grammar, fallback charset) registered as the - contained follow-up inside that item. +- TCL-39 — static route-policy restore needs F13/F48; F13 landed 2026-09-18 + (recorded serving config restored), F48 remains (structured route + representation for policy-layer preservation). Renderer input hardening + (header-name grammar, fallback charset) registered as the contained + follow-up inside that item. - TCL-40 — restore under the app lock + writer quiescence (F37-adjacent; the orchestrated quiesce/cutover boundary is new lifecycle surface). - TCL-41 — F37 (engine-specific consistency contract; verify-backup @@ -350,3 +372,61 @@ into each rather than duplicated as new work items. Gates at the closing commits: `go vet ./...` clean; `go test ./... -race` all packages ok. No push performed. + +## F14 family (2026-09-18) — resolved + +The per-release metadata architecture landed in three commits, closing +F14 and everything the register had blocked on it. + +- **F14** (`97e6d03`) — `internal/releasemeta`: one immutable JSON record + per (app, release id) at `/deployments//meta/.json`, atomic + write at 0600, keyed per-target like state.json/pins (the server a + command talks to IS the target). Records the full resolved spec: image + ref + digest, env-file references, volumes, labels, ports with the + TCL-14 primary designation and fixed/ephemeral flags, replicas/ + processes/cmd, resource limits, stop timeout, bind, health gate, Caddy + edge config (TLS/extra/cache/firewall/access), static serving config, + and the primary web container's full RecreateSpec. Writers: deploy paths + only (a same-version redeploy is the one allowed rewrite); readers never + mutate. Migration: a confirmed-missing record is backfilled from the + live containers on first use ("release-0", flagged Backfilled) — + existing installs converge without a redeploy; backfill records only + what inspect can prove and leaves env-file refs/health/caddy empty for + the legacy fallbacks. +- **F20** (`b3e2ed1`) — `internal/docker` RecreateSpec: InspectRecreate + captures the complete docker-run-representable config in one inspect; + RenderRecreateArgs is a pure renderer; Restart = Inspect + Recreate with + the same signature. Fields previously dropped on every recreate: log + rotation, entrypoint overrides, stop timeout/signal, extra hosts, + sysctls, tmpfs, capabilities, security opts, read-only/privileged, + secondary networks. Multi-element entrypoints that differ from the + image's own fail closed (the CLI cannot represent argv there) instead of + being silently joined. +- **F13 / F21 / TCL-14** (`7314da7`) — rollback and publish recreate + restore from the record, not from reverse-engineered current state: + recorded health gate, domain, TLS/caddy_extra/cache/firewall/access, + ingress mode; recorded primary container port drives health probes + (HostPortFor) and Caddy upstreams; publish apps take the explicit + recreate strategy on deploy (displace current web first, restore on + failure — the host-ingress contract) and rollback (free the fixed ports + first, stop avoiding them). `teploy rollback --app ` is + complete via StaticDeployer.RollbackStateOnly; no record means an + explicit redeploy-first request, never a guess. + +Degradation posture (deliberate): the store is an overlay, not a new +dependency — an unreadable record or failed backfill warns and falls back +to the historical inspect-driven path; only state-only static rollback +(which has nothing to fall back to) fails closed. Record-write failures +warn after the live commit rather than aborting a routed deploy. + +Still deferred, now annotated: F04/TCL-10/TCL-15 are unblocked by F14 +(designs remain — generation handoff, ingress-transition transaction, +Docker-ephemeral port allocation); F08 gains F14 as its keying surface; +F60's protected-spec-per-release is largely delivered by the record (a +reader-facing access contract, if wanted, is the remaining design); +TCL-39's F13 half landed, F48 remains. F48/F49 (Caddy adapt-API route +representation), F16 (lock fencing), F57 (overlay semantics) stay +deferred as before — F14 does not unblock them. + +Gates at the closing commits: `go vet ./...` clean; `go test ./... -race` +all packages ok. No push performed.