From 80db4de5082dc27c6546038396200bdd85274a52 Mon Sep 17 00:00:00 2001 From: Yuan Gao Date: Thu, 27 Aug 2026 15:52:14 -0700 Subject: [PATCH 1/4] hack: drop the IPv6 kubeconfig repoint Before, on an IPv6-only cluster the script rewrote kind's `https://[::1]:PORT` kubeconfig entry to `https://localhost:PORT` unconditionally. That breaks any host whose `/etc/hosts` leaves `localhost` off the `::1` line, including the Ubuntu cloud image Lima runs. Remove the repoint logic, but note that limactl re-forwards the published port to the host's v4 loopback only, so a macOS client cannot directly access kind clusters run inside a Lima VM via `[::1]`. Instead, operations should run inside the guest. Tested: manual tests creating IPv6-only kind clusters on a local env with IPv6 egress, namely macOS + Lima VM. --- hack/create-kind-cluster.sh | 15 --------------- 1 file changed, 15 deletions(-) diff --git a/hack/create-kind-cluster.sh b/hack/create-kind-cluster.sh index 280d9d135d..d6cc5aac4b 100755 --- a/hack/create-kind-cluster.sh +++ b/hack/create-kind-cluster.sh @@ -156,21 +156,6 @@ if [[ "${IP_FAMILY}" != "ipv4" && exit 1 fi -# For ipv6 kind writes a kubeconfig pointing at [::1], the address it published -# the apiserver on, which only works for a client on the Docker host itself: a -# VM-hosted daemon (Lima on macOS) forwards the port to the *v4* loopback, so -# every kubectl below fails at connect. localhost is a SAN on the apiserver -# cert and lets the client pick a family that works from either side. -if [[ "${IP_FAMILY}" == "ipv6" ]]; then - server="$(kubectl config view \ - -o jsonpath="{.clusters[?(@.name==\"${KUBECTL_CONTEXT}\")].cluster.server}")" - if [[ "${server}" == "https://[::1]:"* ]]; then - echo "Repointing the kubeconfig for '${KUBECTL_CONTEXT}' at localhost..." - kubectl config set-cluster "${KUBECTL_CONTEXT}" \ - --server="https://localhost:${server##*:}" >/dev/null - fi -fi - # 2.5 Enable Proxy ARP/NDP on kind nodes for gVisor loopback pod-to-pod networking echo "Enabling Proxy ARP/NDP on kind nodes..." for node in $("${ROOT}"/hack/kind.sh get nodes --name "${KIND_CLUSTER_NAME}"); do From 6c9d0ce6ba2f376864c82372fabba870be88b00b Mon Sep 17 00:00:00 2001 From: Yuan Gao Date: Thu, 27 Aug 2026 15:52:37 -0700 Subject: [PATCH 2/4] hack: fix CoreDNS on IPv6-only kind clusters MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CoreDNS inherits the node's IPv4 resolver, which a v6-only pod cannot reach. Change CoreDNS's Corefile to forward to an overridable IPv6 upstream. Also add a kind-registry:53 server block to CoreDNS. In addition, `kind create` returns before the apiserver answers, so add logic to wait for the control plane from inside the node first. Tested: manual tests of creation and installation on a local env with IPv6 egress, namely macOS + Lima VM. CI cannot verify this — the GitHub runner is IPv4-only, so an IPv6-only cluster there additionally needs DNS64 and NAT64. --- hack/create-kind-cluster.sh | 65 +++++++++++++++++++++++++++++++++++++ 1 file changed, 65 insertions(+) diff --git a/hack/create-kind-cluster.sh b/hack/create-kind-cluster.sh index d6cc5aac4b..d4cf95fded 100755 --- a/hack/create-kind-cluster.sh +++ b/hack/create-kind-cluster.sh @@ -21,6 +21,7 @@ KIND_CLUSTER_NAME="${KIND_CLUSTER_NAME:-kind}" KUBECTL_CONTEXT="kind-${KIND_CLUSTER_NAME}" reg_name="kind-registry" reg_port="${KIND_REGISTRY_PORT:-5001}" +IPV6_DNS_UPSTREAM="${IPV6_DNS_UPSTREAM:-2001:4860:4860::8888 2001:4860:4860::8844}" if [[ $# -gt 0 ]]; then case "$1" in @@ -31,6 +32,9 @@ if [[ $# -gt 0 ]]; then echo "Configured through the environment:" echo " KIND_CLUSTER_NAME Name of the cluster to create (default: kind)." echo " IP_FAMILY Address families for pods and Services: ipv4, ipv6 or dual (default: ipv4)." + echo " IPV6_DNS_UPSTREAM Space-separated IPv6 resolvers CoreDNS forwards to when IP_FAMILY=ipv6" + echo " (default: Google Public DNS). These replace the host's resolver, so any" + echo " split-horizon names it served stop resolving from pods." exit 0 ;; esac @@ -146,6 +150,23 @@ fi echo "Creating kind cluster '${KIND_CLUSTER_NAME}'..." "${ROOT}"/hack/kind.sh create cluster --name "${KIND_CLUSTER_NAME}" --config "${ROOT}/bin/kind-config.yaml" +# kind create returns before the apiserver answers, and every kubectl below races +# it. Poll from inside the node, where the answer does not depend on how the +# daemon published the port. +echo "Waiting for the control plane to answer..." +for attempt in $(seq 60); do + if docker exec "${KIND_CLUSTER_NAME}-control-plane" \ + kubectl --kubeconfig=/etc/kubernetes/admin.conf get --raw /healthz >/dev/null 2>&1; then + break + fi + if [[ "${attempt}" == 60 ]]; then + echo "error: the control plane did not answer /healthz within 2m of create:" >&2 + echo " docker logs ${KIND_CLUSTER_NAME}-control-plane" >&2 + exit 1 + fi + sleep 2 +done + # A daemon with IPv6 off hands kind a v4-only network whatever it asked for. if [[ "${IP_FAMILY}" != "ipv4" && "$(docker network inspect kind --format '{{.EnableIPv6}}')" != "true" ]]; then @@ -192,6 +213,50 @@ if [ "$(docker inspect -f='{{json .NetworkSettings.Networks.kind}}' "${reg_name} docker network connect "kind" "${reg_name}" fi +# 4.5. Point CoreDNS at an IPv6 resolver and teach it the registry's name +if [[ "${IP_FAMILY}" == "ipv6" ]]; then + echo "Repointing CoreDNS at an IPv6 resolver and teaching it '${reg_name}'..." + reg_v6="$(docker inspect "${reg_name}" \ + --format '{{.NetworkSettings.Networks.kind.GlobalIPv6Address}}' 2>/dev/null || true)" + if [[ -z "${reg_v6}" ]]; then + echo "error: '${reg_name}' has no IPv6 address on the 'kind' network" >&2 + exit 1 + fi + + # CoreDNS runs dnsPolicy: Default and inherits the node's IPv4 resolver, which + # no pod here can reach. + corefile="$(kubectl --context="${KUBECTL_CONTEXT}" -n kube-system get cm coredns \ + -o jsonpath='{.data.Corefile}')" + search="forward . /etc/resolv.conf" + # $search unquoted: bash 3.2 splices the quotes in literally. + patched="${corefile/$search/forward . ${IPV6_DNS_UPSTREAM}}" + if [[ "${patched}" == "${corefile}" ]]; then + echo "error: '${search}' not found in the CoreDNS Corefile" >&2 + echo " the Corefile layout changed upstream; update this block" >&2 + exit 1 + fi + + # Step 3's registry wiring is node-side, while atelet pulls from its own netns, + # where "kind-registry" does not resolve. Own zone, so no fallthrough is needed: + # only this name reaches the hosts stanza. + patched="${patched} +${reg_name}:53 { + errors + hosts { + ${reg_v6} ${reg_name} + } +}" + + # A YAML patch file avoids escaping the Corefile's newlines into JSON. + { printf 'data:\n Corefile: |\n'; printf '%s\n' "${patched}" | sed 's/^/ /'; } \ + > "${ROOT}/bin/coredns-patch.yaml" + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system patch cm coredns \ + --type=merge --patch-file "${ROOT}/bin/coredns-patch.yaml" + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system rollout restart deploy/coredns + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system rollout status deploy/coredns \ + --timeout=120s +fi + # 5. Document the local registry in kube-public ConfigMap echo "Documenting local registry in cluster..." cat < Date: Tue, 1 Sep 2026 14:54:01 -0700 Subject: [PATCH 3/4] docs: add a guide for IPv6-only kind clusters IP_FAMILY=ipv6 has no documentation, so the two things that reliably go wrong get rediscovered every time: a Docker daemon with IPv6 off, and a host that cannot route IPv6 at all. The second is usually measured with `curl -6 `, which cannot tell a resolver returning no AAAA apart from a host with no route, and reports neither. Also covers the kubeconfig rewrite kubectl needs when the cluster runs in a Lima VM on macOS, where kind publishes the apiserver on [::1] and Lima forwards that port to the host's 127.0.0.1. --- CONTRIBUTING.md | 3 +- docs/dev/ipv6-local.md | 80 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 82 insertions(+), 1 deletion(-) create mode 100644 docs/dev/ipv6-local.md diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 0b76edcff8..4d6ed03784 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -31,7 +31,8 @@ The [Quickstart (Development)](README.md#quickstart-development) in the README covers bringing up a local cluster with the default (gVisor) runtime. To run the microVM runtime locally — which needs `/dev/kvm`, or Lima nested virtualization on Apple Silicon — see -[docs/dev/microvm-local.md](docs/dev/microvm-local.md). +[docs/dev/microvm-local.md](docs/dev/microvm-local.md). To bring up an +IPv6-only cluster, see [docs/dev/ipv6-local.md](docs/dev/ipv6-local.md). ## Contribution process diff --git a/docs/dev/ipv6-local.md b/docs/dev/ipv6-local.md new file mode 100644 index 0000000000..48e0c54e57 --- /dev/null +++ b/docs/dev/ipv6-local.md @@ -0,0 +1,80 @@ +# Running an IPv6-only cluster locally + +## Overview + +```sh +IP_FAMILY=ipv6 ./hack/create-kind-cluster.sh +``` + +That is the whole setup on a host with IPv6 egress of its own, which includes a +Lima VM on macOS. + +[Measuring whether you have IPv6 egress](#measuring-whether-you-have-ipv6-egress) +tells you whether yours qualifies. On macOS the cluster runs inside the VM, so +`kubectl` from the host needs one extra step: +[Reaching the cluster from macOS](#reaching-the-cluster-from-macos). + +## Prerequisites + +**The Docker daemon needs IPv6.** The script checks this and prints the fix, so +you can let it fail rather than checking up front. On Linux, add to +`/etc/docker/daemon.json` and restart dockerd: + +```json +{"ipv6": true, "ip6tables": true} +``` + +## Measuring whether you have IPv6 egress + +Keep DNS out of the path. `curl -6 ` conflates the two: a resolver +that returns no AAAA fails identically to a host that cannot route, and the +error text (`Could not resolve host`) names neither. Ping and fetch a literal: + +```sh +ping6 -c2 2001:4860:4860::8888 +# Egress: 2 packets transmitted, 2 received, 0% packet loss +# No egress: ping6: connect: Network unreachable + +curl -6 -sS -m 5 -o /dev/null -w '%{http_code}\n' \ + 'http://[2607:f8b0:4004:c1b::5e]/generate_204' +# Egress: 204 +# No egress: curl: (7) Failed to connect ..., then 000 +``` + +Both succeed and you have IPv6 egress; either fails and you do not. The wording +of the failures varies by platform — what matters is that they come back +immediately rather than timing out, which is what a firewalled but routable path +looks like. + +Nor does the absence of a global address mean the absence of egress — a Lima +guest holds only a ULA on `lima0` and reaches the v6 internet through vzNAT. + +## Reaching the cluster from macOS + +Everything that needs Docker — `ko`, kind, the e2e harness — runs inside the +VM, so that is the primary path. For `kubectl` from macOS there is one wrinkle: +kind publishes the API server on the guest's `[::1]`, Lima forwards that to the +*host's* `127.0.0.1`, and the apiserver certificate has no `127.0.0.1` SAN. +Rewrite the address and verify against the `::1` SAN instead of turning +verification off: + +```sh +limactl shell docker-nested -- \ + bash -lc 'cd && ./hack/kind.sh get kubeconfig --name kind' \ + | sed 's|https://\[::1\]:|https://127.0.0.1:|' > /tmp/kind-ipv6.kubeconfig + +KUBECONFIG=~/.kube/config:/tmp/kind-ipv6.kubeconfig \ + kubectl config view --flatten > /tmp/merged && cp /tmp/merged ~/.kube/config +kubectl config set-cluster kind-kind --tls-server-name='::1' +``` + +The forwarded port changes every time the cluster is recreated, so this has to +be redone after each run. + +## Troubleshooting + +| Symptom | Cause | +|---|---| +| `the 'kind' Docker network has no IPv6` | Daemon IPv6 off; see [Prerequisites](#prerequisites). | +| `curl -6 ` fails but the cluster works | A resolver returning no AAAA, not an egress fault; see [Measuring whether you have IPv6 egress](#measuring-whether-you-have-ipv6-egress). | +| `certificate is valid for ... ::1, not 127.0.0.1` | kubectl on macOS against the Lima-forwarded port; see [Reaching the cluster from macOS](#reaching-the-cluster-from-macos). | From 872a8fb52ef27b54d472c12cd2f33fc3d52e976e Mon Sep 17 00:00:00 2001 From: Yuan Gao Date: Mon, 31 Aug 2026 15:13:46 -0700 Subject: [PATCH 4/4] hack: add opt-in NAT64 support to IPv6-only kind clusters An IPv6-only kind cluster on a host without IPv6 egress comes up but reaches nothing: external names resolve to unroutable addresses, and the tree had no way to set up the translation such a host needs. IPV6_DNS64_PREFIX now deploys the kubernetes-sigs/nat64 agent, points cluster DNS through it, and probes from a pod that both halves meet. Off by default, since translate_all routes every external name through a translator a host with working IPv6 egress does not need. The agent's IPv4 pool caps a node at a /120 Pod CIDR, so the prefix narrows podSubnet too. --- _LICENSES/third_party/nat64/LICENSE | 201 +++++++++++++++++++++++++++ docs/dev/ipv6-local.md | 103 ++++++++++++++ hack/create-kind-cluster.sh | 208 +++++++++++++++++++++++++++- hack/third_party/nat64/LICENSE | 201 +++++++++++++++++++++++++++ hack/third_party/nat64/README.md | 21 +++ hack/third_party/nat64/VERSION | 12 ++ hack/third_party/nat64/install.yaml | 99 +++++++++++++ hack/update/nat64.sh | 61 ++++++++ 8 files changed, 903 insertions(+), 3 deletions(-) create mode 100644 _LICENSES/third_party/nat64/LICENSE create mode 100644 hack/third_party/nat64/LICENSE create mode 100644 hack/third_party/nat64/README.md create mode 100644 hack/third_party/nat64/VERSION create mode 100644 hack/third_party/nat64/install.yaml create mode 100755 hack/update/nat64.sh diff --git a/_LICENSES/third_party/nat64/LICENSE b/_LICENSES/third_party/nat64/LICENSE new file mode 100644 index 0000000000..261eeb9e9f --- /dev/null +++ b/_LICENSES/third_party/nat64/LICENSE @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/docs/dev/ipv6-local.md b/docs/dev/ipv6-local.md index 48e0c54e57..7a2e5f7d73 100644 --- a/docs/dev/ipv6-local.md +++ b/docs/dev/ipv6-local.md @@ -9,6 +9,11 @@ IP_FAMILY=ipv6 ./hack/create-kind-cluster.sh That is the whole setup on a host with IPv6 egress of its own, which includes a Lima VM on macOS. +A host without it — a CI runner, most corporate networks — gets a cluster that +comes up clean and then reaches nothing; set `IPV6_DNS64_PREFIX` as well and the +script deploys a translator, see +[When you also need NAT64](#when-you-also-need-nat64). + [Measuring whether you have IPv6 egress](#measuring-whether-you-have-ipv6-egress) tells you whether yours qualifies. On macOS the cluster runs inside the VM, so `kubectl` from the host needs one extra step: @@ -49,6 +54,89 @@ looks like. Nor does the absence of a global address mean the absence of egress — a Lima guest holds only a ULA on `lima0` and reaches the v6 internet through vzNAT. +Without egress, [NAT64](#when-you-also-need-nat64) is the way out. + +## When you also need NAT64 + +`IPV6_DNS64_PREFIX` deploys the +[kubernetes-sigs/nat64](https://github.com/kubernetes-sigs/nat64) agent and +points cluster DNS at the prefix it translates. Two cases need it: + +- **A host with no IPv6 egress at all**, which is the CI runner case. Without + NAT64 the cluster comes up clean and then reaches nothing: every image pull + and outbound call from a pod times out. +- **Reaching IPv4-only destinations.** A plain IPv6-only cluster reaches + anything with a AAAA record, which today is most things — but not, for + example, `github.com`. + +It is off by default because it routes *every* external name through the +translator, including names whose AAAA records already work. + +### Forwarding must be on + +The script does *not* check this. Without it the agent deploys, the rollout +succeeds, the restart-count check passes, and the run fails at the connectivity +probe with nothing pointing at the cause: + +```sh +sudo sysctl -w net.ipv4.ip_forward=1 +sudo sysctl -w net.ipv6.conf.all.forwarding=1 +``` + +> [!WARNING] +> Forwarding also stops the kernel accepting router advertisements, which can +> drop the default route on a machine with real IPv6. Set it in the VM, not on +> your Mac. + +### In CI + +The IPv6-only e2e job configures nothing itself — it sets two variables and +lets this script build the cluster: + +```yaml +env: + IP_FAMILY: ipv6 + IPV6_DNS64_PREFIX: '64:ff9b::/96' +``` + +plus the two sysctls above, ordered *after* the dockerd restart that enables +IPv6, because that restart rebuilds the chains they affect. + +### On Linux + +```sh +sudo sysctl -w net.ipv4.ip_forward=1 +sudo sysctl -w net.ipv6.conf.all.forwarding=1 + +IP_FAMILY=ipv6 IPV6_DNS64_PREFIX='64:ff9b::/96' ./hack/create-kind-cluster.sh +``` + +### On Apple Silicon + +The published agent images do not run on arm64: every tag ships an x86-64 +binary in the arm64 slot of its image index +([upstream issue](https://github.com/kubernetes-sigs/nat64/issues/103)), so the +agent exits with `exec format error` and nothing is translated. Build one and +push it to the local registry: + +```sh +git clone https://github.com/kubernetes-sigs/nat64 /tmp/nat64 +docker build --build-arg GOARCH=arm64 -t localhost:5001/nat64:local /tmp/nat64 +docker push localhost:5001/nat64:local +``` + +Push to the registry rather than `kind load`: the script deletes and recreates +the cluster on every run, which drops a loaded image, while the registry +container outlives it. The registry is created by the cluster script, so on a +first run let it create the cluster once, then build, push, and re-run. + +```sh +IP_FAMILY=ipv6 \ +IPV6_DNS64_PREFIX='64:ff9b::/96' \ +NAT64_IMAGE=localhost:5001/nat64:local \ + ./hack/create-kind-cluster.sh +``` + ## Reaching the cluster from macOS Everything that needs Docker — `ko`, kind, the e2e harness — runs inside the @@ -71,10 +159,25 @@ kubectl config set-cluster kind-kind --tls-server-name='::1' The forwarded port changes every time the cluster is recreated, so this has to be redone after each run. +## Verifying + +With `IPV6_DNS64_PREFIX` set, the script proves both halves meet before it +exits — it fetches `NAT64_PROBE_URL` from a pod and prints `NAT64 is +translating.` To look at what it built: + +```sh +kubectl -n kube-system get pod -l app=nat64 +kubectl -n kube-system get cm coredns -o jsonpath='{.data.Corefile}' +``` + ## Troubleshooting | Symptom | Cause | |---|---| | `the 'kind' Docker network has no IPv6` | Daemon IPv6 off; see [Prerequisites](#prerequisites). | +| `a pod could not reach ... through 64:ff9b::/96` | Forwarding sysctls not set on the host. | +| `exec format error` in the agent log | arm64 node running a published image; set `NAT64_IMAGE`. | | `curl -6 ` fails but the cluster works | A resolver returning no AAAA, not an egress fault; see [Measuring whether you have IPv6 egress](#measuring-whether-you-have-ipv6-egress). | +| In-cluster Service names stop resolving | The Corefile's `dns64` block was merged into the cluster-zone block. It has to stay separate — `dns64` synthesizes AAAA from A, and an AAAA-only ClusterIP synthesizes to nothing. | +| Only names *without* AAAA records get translated | `translate_all` is not in effect. The prefix belongs inside the `dns64 { }` block; on the `dns64` line it parses and the block is silently dropped. | | `certificate is valid for ... ::1, not 127.0.0.1` | kubectl on macOS against the Lima-forwarded port; see [Reaching the cluster from macOS](#reaching-the-cluster-from-macos). | diff --git a/hack/create-kind-cluster.sh b/hack/create-kind-cluster.sh index d4cf95fded..b36de67ee0 100755 --- a/hack/create-kind-cluster.sh +++ b/hack/create-kind-cluster.sh @@ -21,7 +21,13 @@ KIND_CLUSTER_NAME="${KIND_CLUSTER_NAME:-kind}" KUBECTL_CONTEXT="kind-${KIND_CLUSTER_NAME}" reg_name="kind-registry" reg_port="${KIND_REGISTRY_PORT:-5001}" -IPV6_DNS_UPSTREAM="${IPV6_DNS_UPSTREAM:-2001:4860:4860::8888 2001:4860:4860::8844}" +IPV6_DNS_UPSTREAM="${IPV6_DNS_UPSTREAM:-}" +IPV6_DNS64_PREFIX="${IPV6_DNS64_PREFIX:-}" +NAT64_PROBE_URL="${NAT64_PROBE_URL:-http://connectivitycheck.gstatic.com/generate_204}" +NAT64_IMAGE="${NAT64_IMAGE:-}" +# curl, not the busybox in manifests/: busybox wget takes the first address the +# resolver hands back, which here can be one no pod can reach. +probe_image="curlimages/curl:8.11.1@sha256:c1fe1679c34d9784c1b0d1e5f62ac0a79fca01fb6377cdd33e90473c6f9f9a69" if [[ $# -gt 0 ]]; then case "$1" in @@ -33,8 +39,17 @@ if [[ $# -gt 0 ]]; then echo " KIND_CLUSTER_NAME Name of the cluster to create (default: kind)." echo " IP_FAMILY Address families for pods and Services: ipv4, ipv6 or dual (default: ipv4)." echo " IPV6_DNS_UPSTREAM Space-separated IPv6 resolvers CoreDNS forwards to when IP_FAMILY=ipv6" - echo " (default: Google Public DNS). These replace the host's resolver, so any" - echo " split-horizon names it served stop resolving from pods." + echo " (default: Google Public DNS, reached through IPV6_DNS64_PREFIX when" + echo " that is set). These replace the host's resolver, so any split-horizon" + echo " names it served stop resolving from pods." + echo " IPV6_DNS64_PREFIX NAT64 prefix to route external traffic through when IP_FAMILY=ipv6." + echo " Must be 64:ff9b::/96, the only prefix the vendored agent translates" + echo " (default: empty, no NAT64). Needed where the host has no IPv6 egress" + echo " of its own; leave unset where it does. Setup, prerequisites and" + echo " caveats: docs/dev/ipv6-local.md." + echo " NAT64_PROBE_URL URL a pod fetches to prove translation works (default:" + echo " ${NAT64_PROBE_URL})." + echo " NAT64_IMAGE Override the NAT64 agent image; required on arm64." exit 0 ;; esac @@ -52,6 +67,37 @@ case "${IP_FAMILY}" in ;; esac +if [[ -n "${IPV6_DNS64_PREFIX}" ]]; then + if [[ "${IP_FAMILY}" != "ipv6" ]]; then + echo "error: IPV6_DNS64_PREFIX only applies to IP_FAMILY=ipv6 (got '${IP_FAMILY}')" >&2 + exit 1 + fi + # The vendored manifest leaves the agent's --nat-v6-cidr at its 64:ff9b::/96 + # default, so any other prefix has DNS64 synthesizing addresses nothing + # translates. + if [[ "${IPV6_DNS64_PREFIX}" != "64:ff9b::/96" ]]; then + echo "error: IPV6_DNS64_PREFIX must be 64:ff9b::/96, the only prefix the deployed agent" >&2 + echo " translates (got '${IPV6_DNS64_PREFIX}')" >&2 + exit 1 + fi +fi + +# The resolver cluster DNS forwards to. With NAT64 it has to be an IPv4 resolver +# reached through the prefix, because a host that needs NAT64 has no reachable +# IPv6 resolver to send pods at. +derive_dns_upstream() { + if [[ -z "${IPV6_DNS64_PREFIX}" ]]; then + echo "2001:4860:4860::8888 2001:4860:4860::8844" + return + fi + local p="${IPV6_DNS64_PREFIX%/*}" + echo "${p}0808:0808 ${p}0808:0404" +} + +if [[ -z "${IPV6_DNS_UPSTREAM}" ]]; then + IPV6_DNS_UPSTREAM="$(derive_dns_upstream)" +fi + mkdir -p "${ROOT}/bin" # 1. Create registry container unless it already exists @@ -116,6 +162,17 @@ runtimeConfig: "certificates.k8s.io/v1beta1": "true" networking: ipFamily: ${IP_FAMILY} +EOF +if [[ -n "${IPV6_DNS64_PREFIX}" ]]; then + # The manifest leaves the agent's IPv4 pool (--nat-v4-cidr) at its /24 + # default, so a node's Pod CIDR cannot be wider than /120. This /112 is + # what makes kube-controller-manager hand out /120s; kind's IPv6 default + # /56 does not. + cat <> "${ROOT}/bin/kind-config.yaml" + podSubnet: "fd00:10:244::/112" +EOF +fi +cat <> "${ROOT}/bin/kind-config.yaml" # The install pulls ~570MB of third-party images (postgres, prometheus, the otel # collector, rustfs, envoy, jaeger) onto this one node. kubelet serializes image # pulls by default, so they queue behind one another and whichever workload draws @@ -128,6 +185,18 @@ kubeadmConfigPatches: serializeImagePulls: false maxParallelImagePulls: 4 EOF +if [[ -n "${IPV6_DNS64_PREFIX}" ]]; then + # Appends an item to the list above. A second kubeadmConfigPatches key makes + # the config unparseable and kind rejects it before creating anything. + cat <> "${ROOT}/bin/kind-config.yaml" +- | + kind: ClusterConfiguration + controllerManager: + extraArgs: + - name: node-cidr-mask-size-ipv6 + value: "120" +EOF +fi echo "Deleting existing kind cluster '${KIND_CLUSTER_NAME}' if it exists..." "${ROOT}"/hack/kind.sh delete cluster --name "${KIND_CLUSTER_NAME}" || true @@ -213,6 +282,61 @@ if [ "$(docker inspect -f='{{json .NetworkSettings.Networks.kind}}' "${reg_name} docker network connect "kind" "${reg_name}" fi +# 4.4. Deploy the NAT64 agent, before cluster DNS starts answering with the +# prefix it translates. +if [[ -n "${IPV6_DNS64_PREFIX}" ]]; then + echo "Deploying the NAT64 agent for ${IPV6_DNS64_PREFIX}..." + kubectl --context="${KUBECTL_CONTEXT}" apply -f "${ROOT}/hack/third_party/nat64/install.yaml" + if [[ -n "${NAT64_IMAGE}" ]]; then + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system set image ds/nat64 \ + "nat64=${NAT64_IMAGE}" + fi + + nat64_fail() { + echo "error: ${1}" >&2 + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system logs ds/nat64 --tail=50 >&2 || true + exit 1 + } + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system rollout status ds/nat64 \ + --timeout=180s || nat64_fail "the NAT64 agent did not roll out" + + # ds/nat64 has no readiness probe, so the rollout above returns as soon as the + # container is created. Poll instead. + echo "Waiting for the NAT64 agent to stay up..." + clean=0 + for attempt in $(seq 45); do + sleep 2 + total=0 + running=0 + while read -r count started; do + if [[ -z "${count}" ]]; then + continue + fi + total=$((total + 1)) + if [[ "${count}" != 0 ]]; then + nat64_fail "the NAT64 agent restarted ${count} time(s)" + fi + if [[ -n "${started}" ]]; then + running=$((running + 1)) + fi + done <<< "$(kubectl --context="${KUBECTL_CONTEXT}" -n kube-system get pod -l app=nat64 \ + -o jsonpath='{range .items[*].status.containerStatuses[*]}{.restartCount}{" "}{.state.running.startedAt}{"\n"}{end}')" + + if [[ "${total}" -gt 0 && "${running}" -eq "${total}" ]]; then + clean=$((clean + 1)) + if [[ "${clean}" -eq 5 ]]; then + break + fi + else + clean=0 + fi + + if [[ "${attempt}" -eq 45 ]]; then + nat64_fail "the NAT64 agent never ran for 10s without restarting" + fi + done +fi + # 4.5. Point CoreDNS at an IPv6 resolver and teach it the registry's name if [[ "${IP_FAMILY}" == "ipv6" ]]; then echo "Repointing CoreDNS at an IPv6 resolver and teaching it '${reg_name}'..." @@ -247,6 +371,49 @@ ${reg_name}:53 { } }" + if [[ -n "${IPV6_DNS64_PREFIX}" ]]; then + echo "Synthesizing external names into ${IPV6_DNS64_PREFIX}..." + # dns64 answers AAAA by synthesizing from A, so it cannot share a block with + # the cluster zones: an AAAA-only ClusterIP would synthesize to nothing. Narrow + # the block kind ships to those zones and lift its forwarder out. + rezoned="$(printf '%s\n' "${patched}" | awk ' + NR == 1 && /^\.:53[[:space:]]*\{/ { + print "cluster.local:53 in-addr.arpa:53 ip6.arpa:53 {"; first = 1; next + } + first && /^ forward([[:space:]].*)?\{$/ { skip = 1; next } + first && skip && /^ \}$/ { skip = 0; next } + first && skip { next } + first && /^\}$/ { first = 0 } + { print } + ')" + case "${rezoned}" in + "cluster.local:53"*) ;; + *) echo "error: the Corefile does not open with the '.:53' block kind ships" >&2 + echo " the Corefile layout changed upstream; update this block" >&2 + exit 1 ;; + esac + if printf '%s' "${rezoned}" | grep -q 'forward'; then + echo "error: a forward block survived the re-zone" >&2 + exit 1 + fi + # The prefix goes inside the block: `dns64 PREFIX { ... }` parses and then + # silently drops the block. + patched="${rezoned} +.:53 { + errors + dns64 { + prefix ${IPV6_DNS64_PREFIX} + translate_all + } + forward . ${IPV6_DNS_UPSTREAM} { + max_concurrent 1000 + } + cache 30 + loop + reload +}" + fi + # A YAML patch file avoids escaping the Corefile's newlines into JSON. { printf 'data:\n Corefile: |\n'; printf '%s\n' "${patched}" | sed 's/^/ /'; } \ > "${ROOT}/bin/coredns-patch.yaml" @@ -257,6 +424,41 @@ ${reg_name}:53 { --timeout=120s fi +# 4.6. Prove the two halves meet. +if [[ -n "${IPV6_DNS64_PREFIX}" ]]; then + echo "Probing ${NAT64_PROBE_URL} from a pod..." + for attempt in $(seq 30); do + kubectl --context="${KUBECTL_CONTEXT}" -n default get sa default >/dev/null 2>&1 && break + if [[ "${attempt}" == 30 ]]; then + echo "error: the 'default' ServiceAccount never appeared" >&2 + exit 1 + fi + sleep 2 + done + + kubectl --context="${KUBECTL_CONTEXT}" -n default delete pod nat64-probe \ + --ignore-not-found --wait >/dev/null + kubectl --context="${KUBECTL_CONTEXT}" -n default run nat64-probe \ + --image="${probe_image}" --restart=Never --command -- \ + sh -c "curl -6 -sS -m 25 -o /dev/null -w 'reached %{remote_ip}\n' '${NAT64_PROBE_URL}' \ + && echo NAT64-OK" >/dev/null + + # Read the log rather than attaching: attach races the container's exit. + kubectl --context="${KUBECTL_CONTEXT}" -n default wait --for=jsonpath='{.status.phase}'=Succeeded \ + pod/nat64-probe --timeout=120s >/dev/null || true + probe_out="$(kubectl --context="${KUBECTL_CONTEXT}" -n default logs nat64-probe 2>&1 || true)" + kubectl --context="${KUBECTL_CONTEXT}" -n default delete pod nat64-probe \ + --ignore-not-found --wait=false >/dev/null + + if [[ "${probe_out}" != *NAT64-OK* ]]; then + echo "error: a pod could not reach ${NAT64_PROBE_URL} through ${IPV6_DNS64_PREFIX}" >&2 + echo " probe output: ${probe_out:-(none)}" >&2 + kubectl --context="${KUBECTL_CONTEXT}" -n kube-system logs ds/nat64 --tail=50 >&2 || true + exit 1 + fi + echo "NAT64 is translating." +fi + # 5. Document the local registry in kube-public ConfigMap echo "Documenting local registry in cluster..." cat <&2 + exit 1 +fi + +{ + cat <"${DIR}/install.yaml" + +echo "Wrote ${DIR}/install.yaml from ${NAT64_COMMIT}"