diff --git a/README.md b/README.md index 9f37e9ecf..8593bb784 100644 --- a/README.md +++ b/README.md @@ -88,7 +88,8 @@ To quickly set up the complete environment: 2. Run the following steps: ```shell -# create cluster and local registry +# create cluster and local registry (IPv4; KIND_IP_FAMILY=dual|ipv6 overrides, +# and hack/verify-kind-networking.sh checks the result) hack/create-kind-cluster.sh # install ate, valkey, rustfs diff --git a/hack/create-kind-cluster.sh b/hack/create-kind-cluster.sh index 26548f2b4..64ac3db11 100755 --- a/hack/create-kind-cluster.sh +++ b/hack/create-kind-cluster.sh @@ -18,9 +18,36 @@ set -o errexit -o nounset -o pipefail ROOT="$(cd "$(dirname "$0")/.." && pwd)" KIND_CLUSTER_NAME="${KIND_CLUSTER_NAME:-kind}" +KUBECTL_CONTEXT="kind-${KIND_CLUSTER_NAME}" reg_name="kind-registry" reg_port="5001" +if [[ $# -gt 0 ]]; then + case "$1" in + -h|--help) + echo "Usage: $0" + echo "Creates the kind cluster '${KIND_CLUSTER_NAME}' and a local registry container on port ${reg_port}." + echo + echo "Configured through the environment:" + echo " KIND_CLUSTER_NAME Name of the cluster to create (default: kind)." + echo " KIND_IP_FAMILY Address families for pods and Services: ipv4, ipv6 or dual (default: ipv4)." + exit 0 + ;; + esac +fi + +# Only ipFamily is set; kind's per-family podSubnet/serviceSubnet defaults are +# already what we want. +KIND_IP_FAMILY="${KIND_IP_FAMILY:-ipv4}" +case "${KIND_IP_FAMILY}" in + ipv4|ipv6|dual) + ;; + *) + echo "error: KIND_IP_FAMILY must be one of ipv4, ipv6, dual (got '${KIND_IP_FAMILY}')" >&2 + exit 1 + ;; +esac + mkdir -p "${ROOT}/bin" # 1. Create registry container unless it already exists @@ -33,6 +60,9 @@ if [ "$(docker inspect -f '{{.State.Running}}' "${reg_name}" 2>/dev/null || true fi if [ "$(docker inspect -f '{{.State.Running}}' "${reg_name}" 2>/dev/null || true)" != "true" ]; then + # Published on both loopback families so `ko` reaches localhost:5001 whichever + # one its resolver picks. The node side is separate: it goes over the "kind" + # network in step 4. docker run \ -d --restart=always \ --label created-by=agent-substrate \ @@ -57,7 +87,7 @@ else echo "/dev/kvm not available: micro-VM support disabled (gVisor still works)." fi -echo "Creating kind configuration for cluster '${KIND_CLUSTER_NAME}'..." +echo "Creating kind configuration for cluster '${KIND_CLUSTER_NAME}' (ipFamily=${KIND_IP_FAMILY})..." cat < "${ROOT}/bin/kind-config.yaml" kind: Cluster apiVersion: kind.x-k8s.io/v1alpha4 @@ -83,18 +113,65 @@ featureGates: PodCertificateRequest: true runtimeConfig: "certificates.k8s.io/v1beta1": "true" +networking: + ipFamily: ${KIND_IP_FAMILY} EOF echo "Deleting existing kind cluster '${KIND_CLUSTER_NAME}' if it exists..." "${ROOT}"/hack/kind.sh delete cluster --name "${KIND_CLUSTER_NAME}" || true +# kind reuses an existing "kind" network as-is, so one created by an older kind +# or while the daemon had IPv6 off (kind falls back to a v4-only network rather +# than failing) leaves the nodes with no v6 address — seen much later as pods +# stuck at ContainerCreating. Deleting the cluster does not drop the network +# either: the registry is still attached to it. Step 4 reconnects the registry. +if [[ "${KIND_IP_FAMILY}" != "ipv4" && + "$(docker network inspect kind --format '{{.EnableIPv6}}' 2>/dev/null || echo absent)" == "false" ]]; then + echo "The 'kind' Docker network exists without IPv6; recreating it..." + docker network disconnect kind "${reg_name}" 2>/dev/null || true + if ! docker network rm kind >/dev/null; then + echo "error: could not remove the 'kind' Docker network. Something else is still" >&2 + echo " attached to it; disconnect it and re-run:" >&2 + echo " docker network inspect kind --format '{{json .Containers}}'" >&2 + exit 1 + fi +fi + echo "Creating kind cluster '${KIND_CLUSTER_NAME}'..." "${ROOT}"/hack/kind.sh create cluster --name "${KIND_CLUSTER_NAME}" --config "${ROOT}/bin/kind-config.yaml" -# 2.5 Enable Proxy ARP on kind nodes for gVisor loopback pod-to-pod networking -echo "Enabling Proxy ARP on kind nodes..." +# A daemon with IPv6 off hands kind a v4-only network whatever it asked for. +if [[ "${KIND_IP_FAMILY}" != "ipv4" && + "$(docker network inspect kind --format '{{.EnableIPv6}}')" != "true" ]]; then + echo "error: the 'kind' Docker network has no IPv6, so the nodes have no v6 address." >&2 + echo " Enable IPv6 in the Docker daemon and re-run. On Linux, add to" >&2 + echo " /etc/docker/daemon.json and restart dockerd:" >&2 + echo ' {"ipv6": true, "ip6tables": true}' >&2 + exit 1 +fi + +# For ipv6 kind writes a kubeconfig pointing at [::1], the address it published +# the apiserver on, which only works for a client on the Docker host itself: a +# VM-hosted daemon (Lima on macOS) forwards the port to the *v4* loopback, so +# every kubectl below fails at connect. localhost is a SAN on the apiserver +# cert and lets the client pick a family that works from either side. +if [[ "${KIND_IP_FAMILY}" == "ipv6" ]]; then + server="$(kubectl config view \ + -o jsonpath="{.clusters[?(@.name==\"${KUBECTL_CONTEXT}\")].cluster.server}")" + if [[ "${server}" == "https://[::1]:"* ]]; then + echo "Repointing the kubeconfig for '${KUBECTL_CONTEXT}' at localhost..." + kubectl config set-cluster "${KUBECTL_CONTEXT}" \ + --server="https://localhost:${server##*:}" >/dev/null + fi +fi + +# 2.5 Enable Proxy ARP/NDP on kind nodes for gVisor loopback pod-to-pod networking +echo "Enabling Proxy ARP/NDP on kind nodes..." for node in $("${ROOT}"/hack/kind.sh get nodes --name "${KIND_CLUSTER_NAME}"); do + # Unconditional: harmless on a v6-only cluster, where the nodes still carry + # IPv4 on the Docker bridge, and proxy_ndp just supports IPv6 if configured. docker exec "${node}" sysctl net.ipv4.conf.all.proxy_arp=1 + docker exec "${node}" sysctl net.ipv6.conf.all.proxy_ndp=1 done # 2.6 When KVM is available: make /dev/kvm usable inside the node and label @@ -103,7 +180,7 @@ if [ "${HAS_KVM}" = "1" ]; then echo "Preparing kind nodes for micro-VM (kata + cloud-hypervisor) runtime..." for node in $("${ROOT}"/hack/kind.sh get nodes --name "${KIND_CLUSTER_NAME}"); do docker exec "${node}" chmod 666 /dev/kvm - kubectl label node "${node}" ate.dev/sandboxClass=microvm --overwrite + kubectl --context="${KUBECTL_CONTEXT}" label node "${node}" ate.dev/sandboxClass=microvm --overwrite done fi @@ -125,7 +202,7 @@ fi # 5. Document the local registry in kube-public ConfigMap echo "Documenting local registry in cluster..." -cat < registry path. +REG_IMAGE="localhost:5001/kind-net-check/busybox:1" + +case "${KIND_IP_FAMILY}" in + ipv4) want_families=(ipv4); primary="ipv4" ;; + ipv6) want_families=(ipv6); primary="ipv6" ;; + dual) want_families=(ipv4 ipv6); primary="ipv4" ;; + *) + echo "error: KIND_IP_FAMILY must be one of ipv4, ipv6, dual (got '${KIND_IP_FAMILY}')" >&2 + exit 1 + ;; +esac + +run_kubectl() { kubectl --context="${KUBECTL_CONTEXT}" "$@"; } + +log_step() { echo; echo "[kind-net-check]: $*"; } +fail() { echo "FAIL: $*" >&2; exit 1; } + +cleanup() { + local code=$? + set +e + run_kubectl delete namespace "${NS}" --wait=false >/dev/null 2>&1 + if [[ "${code}" -eq 0 ]]; then + echo + echo "PASS: ${KIND_IP_FAMILY} cluster networking looks correct." + fi + exit "${code}" +} +trap cleanup EXIT + +# Kubernetes exposes no field for the family, so go by the colon. +family_of() { + case "$1" in + *:*) echo "ipv6" ;; + *) echo "ipv4" ;; + esac +} + +assert_family() { + local what="$1" addr="$2" want="$3" got + got="$(family_of "${addr}")" + [[ "${got}" == "${want}" ]] || fail "${what}: expected ${want}, got ${addr}" + echo " ok: ${what} = ${addr} (${got})" +} + +# ip_of_family WANT ADDR... — prints the first ADDR in family WANT, or fails. +ip_of_family() { + local want="$1" addr + shift + for addr in "$@"; do + if [[ "$(family_of "${addr}")" == "${want}" ]]; then + echo "${addr}" + return 0 + fi + done + return 1 +} + +# assert_covers_families WHAT ADDR... — every wanted family appears in ADDR... +assert_covers_families() { + local what="$1" want found + shift + [[ "$#" -gt 0 ]] || fail "${what}: no addresses at all" + for want in "${want_families[@]}"; do + found="$(ip_of_family "${want}" "$@")" || fail "${what}: no ${want} address (has: $*)" + echo " ok: ${what} ${want} = ${found}" + done +} + +log_step "0. cluster is reachable" +run_kubectl version -o yaml >/dev/null || fail "cannot reach context ${KUBECTL_CONTEXT}" + +log_step "1. node InternalIPs cover: ${want_families[*]}" +for node in $(run_kubectl get nodes -o name); do + # jsonpath joins multiple addresses with a space. + read -r -a addrs <<<"$(run_kubectl get "${node}" \ + -o jsonpath='{.status.addresses[?(@.type=="InternalIP")].address}')" + [[ "${#addrs[@]}" -gt 0 ]] || fail "${node} has no InternalIP" + assert_covers_families "${node} InternalIP" "${addrs[@]}" +done + +log_step "2. kubernetes Service ClusterIP is ${primary}" +assert_family "kubernetes ClusterIP" \ + "$(run_kubectl get svc kubernetes -n default -o jsonpath='{.spec.clusterIP}')" "${primary}" + +log_step "3. pods pull from the local registry and get podIPs covering: ${want_families[*]}" +run_kubectl create namespace "${NS}" >/dev/null +# Seed the registry so the pull below goes through it rather than a public +# mirror, covering the host -> registry direction on the way. +docker pull --quiet docker.io/library/busybox:1 >/dev/null +docker tag docker.io/library/busybox:1 "${REG_IMAGE}" +docker push --quiet "${REG_IMAGE}" >/dev/null +for name in server client; do + run_kubectl run "${name}" -n "${NS}" --image="${REG_IMAGE}" --restart=Never \ + --command -- sleep 3600 >/dev/null +done +run_kubectl wait --for=condition=Ready pod/server pod/client -n "${NS}" --timeout=180s >/dev/null +# Ready means containerd pulled the image, which is the step the "kind" network's +# address families actually gate: the node resolves kind-registry through +# Docker's embedded DNS there, not over the published loopback ports. +echo " ok: pods pulled ${REG_IMAGE} from the local registry" + +# podIP is podIPs[0] by definition, so it only ever shows the primary family; +# podIPs is where the second address of a dual-stack pod turns up. +assert_family "server status.podIP" \ + "$(run_kubectl get pod server -n "${NS}" -o jsonpath='{.status.podIP}')" "${primary}" +read -r -a server_ips <<<"$(run_kubectl get pod server -n "${NS}" \ + -o jsonpath='{.status.podIPs[*].ip}')" +[[ "${#server_ips[@]}" -gt 0 ]] || fail "server has no status.podIPs" +assert_family "server status.podIPs[0]" "${server_ips[0]}" "${primary}" +assert_covers_families "server status.podIPs" "${server_ips[@]}" + +log_step "4. pod -> pod over: ${want_families[*]}" +# One listener per family on its own port. A single [::] socket would serve IPv4 +# too through v4-mapped addresses, so it would pass without the v4 path ever +# being exercised — and it breaks outright wherever bindv6only is on. +# busybox httpd binds 0.0.0.0 given a bare port, which accepts nothing on a +# v6-only pod, so the v6 wildcard is named. URLs bracket a v6 literal, never a v4. +port=8080 +for want in "${want_families[@]}"; do + server_ip="$(ip_of_family "${want}" "${server_ips[@]}")" + if [[ "${want}" == "ipv6" ]]; then + listen_spec="[::]:${port}" + server_host="[${server_ip}]" + else + listen_spec="${port}" + server_host="${server_ip}" + fi + # httpd daemonizes without -f, so this exec returns once it is listening. + run_kubectl exec -n "${NS}" server -- \ + sh -c "mkdir -p /www && echo pong > /www/ping && httpd -p '${listen_spec}' -h /www" >/dev/null + got="$(run_kubectl exec -n "${NS}" client -- \ + wget -q -T 10 -O- "http://${server_host}:${port}/ping" 2>/dev/null || true)" + [[ "${got}" == "pong" ]] || + fail "pod -> pod to ${server_host}:${port} returned '${got}', want 'pong'" + echo " ok: client reached server at ${server_host}:${port}" + port=$((port + 1)) +done + +log_step "5. a PreferDualStack Service gets a ClusterIP in each of: ${want_families[*]}" +# Step 2's "kubernetes" Service is SingleStack, so it says nothing about whether +# kube-apiserver got a second Service CIDR. Every dual-stack Service in the tree +# depends on that allocation working. +run_kubectl apply -n "${NS}" -f - >/dev/null </dev/null +# Primary-only on purpose: kubernetes.default is SingleStack, so an AAAA query +# is a correct NODATA, and busybox nslookup cannot be relied on to ask for one +# type at a time. Per-family DNS belongs in the e2e suite, which has a real +# resolver and can tell NODATA from SERVFAIL. +# Skip to the answer section; everything before "Name:" describes the resolver. +# busybox has written both "Address:" and "Address 1:", so match loosely. +# The trailing "|| true" is load-bearing: errexit plus pipefail would otherwise +# abort the whole script on a failed lookup before reaching the fail() below, +# and a bare non-zero exit from a command substitution prints nothing at all. +resolved="$(run_kubectl exec -n "${NS}" client -- nslookup kubernetes.default.svc.cluster.local 2>/dev/null \ + | awk '/^Name:/{seen=1; next} seen && /^Address/{sub(/^Address[^:]*:[[:space:]]*/, ""); print $1; exit}' \ + || true)" +[[ -n "${resolved}" ]] || fail "kubernetes.default did not resolve" +assert_family "kubernetes.default resolves to" "${resolved}" "${primary}"