Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions deploy/aks/manifests/hosting-operator/operator-rbac.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -137,6 +137,17 @@ rules:
- apiGroups: [""]
resources: ["persistentvolumes"]
verbs: ["get", "list", "delete"]
# For hosting-tls, the step that gives a new instance its certificate. cert-manager's own
# controller does the issuing; the operator only declares the Certificate and reads it back to
# learn whether it went Ready. Measured 2026-09-15 on pearl, step 15/18 of its Provision:
# certificates.cert-manager.io "pearl-tls" is forbidden: User
# "system:serviceaccount:memex-ops:hosting-operator" cannot get resource "certificates"
# The run stopped there, so TLS and all three verification steps never ran and the host served
# another instance's certificate. No delete: a Certificate is owned by the release that renders
# the ingress, and removing one is not something a provisioning run may do.
- apiGroups: ["cert-manager.io"]
resources: ["certificates"]
verbs: ["get", "list", "watch", "create", "update", "patch"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
Expand Down
42 changes: 37 additions & 5 deletions deploy/aks/operator/bin/hosting-tls
Original file line number Diff line number Diff line change
Expand Up @@ -7,16 +7,27 @@
# the Secret minutes later, so a step that returns as soon as the object exists reports success over
# an ingress still serving another host's default certificate — which is what a browser shows the
# first visitor. This waits for the Secret to carry an actual key pair, and fails if it never does.
#
# 🚨 IT ALSO READS THE CERTIFICATE BACK. The Secret is the outcome; the Certificate's Ready
# condition is the REASON, and it is what separates "the challenge is still running" from "the
# challenge cannot succeed" (DNS not resolving to this ingress publicly — HTTP-01 validates over
# the internet, from Let's Encrypt's resolver, not from inside the cluster). The refusal names
# which of the two it was. The explicit `kubectl get certificates` below is also what makes the
# grant visible to deploy/aks/operator/test/check-rbac-coverage.sh: a Certificate created through
# `kubectl apply -f -` is invisible to that check, which is why this script reached main without
# its ClusterRole rule and failed pearl's Provision at step 15/18 on 2026-09-15 with
# `certificates.cert-manager.io "pearl-tls" is forbidden`.

source "$(dirname "$(readlink -f "${BASH_SOURCE[0]}")")/_common.sh"
# shellcheck disable=SC2034 # read by hosting::die in _common.sh, which shellcheck does not follow here
HOSTING_CMD="hosting-tls"

namespace="" host="" timeout="${HOSTING_TLS_TIMEOUT:-600}"
namespace="" host="" issuer="" timeout="${HOSTING_TLS_TIMEOUT:-600}"
while [ $# -gt 0 ]; do
case "$1" in
--namespace) namespace="${2:-}"; shift 2 ;;
--host) host="${2:-}"; shift 2 ;;
--issuer) issuer="${2:-}"; shift 2 ;;
--timeout) timeout="${2:-}"; shift 2 ;;
*) hosting::die "unknown argument '$1'" ;;
esac
Expand All @@ -28,14 +39,26 @@ hosting::safe_name namespace "$namespace"
hosting::safe_host host "$host"

secret="${HOSTING_TLS_SECRET:-${namespace}-tls}"
issuer="${HOSTING_TLS_ISSUER:-letsencrypt-prod}"
# --issuer is what the RECORD says (the plan passes it); the env var is the fleet fallback for a
# hand-run. `none` would mean "ask cert-manager for nothing", which is not this step's business:
# the plan does not compose a TLS step for such a record, so seeing it here is a contradiction.
[ -n "$issuer" ] || issuer="${HOSTING_TLS_ISSUER:-letsencrypt-prod}"
[ "$issuer" = "none" ] && hosting::die "the issuer is 'none' — that record asks cert-manager for nothing, so this step must not have been composed. Nothing was created."

ready() {
local tls_crt
tls_crt="$(kubectl -n "$namespace" get secret "$secret" -o jsonpath='{.data.tls\.crt}' 2>/dev/null || true)"
[ -n "$tls_crt" ]
}

# The Certificate's Ready condition — "True" once cert-manager has issued, plus its message while
# it has not. Read explicitly (see the header) so the refusal can say WHY, not just that nothing
# arrived.
cert_state() {
kubectl -n "$namespace" get certificates "$secret" \
-o jsonpath='{.status.conditions[?(@.type=="Ready")].status} {.status.conditions[?(@.type=="Ready")].message}' 2>/dev/null || true
}

if ready; then
hosting::log "TLS secret ${secret} already carries a certificate — keeping it"
hosting::say tls kept
Expand Down Expand Up @@ -66,10 +89,15 @@ spec:
kind: ClusterIssuer
YAML

# WAIT for the certificate to exist, not for the request to be accepted. Both halves must hold:
# the Certificate Ready, and the Secret carrying a key pair — the object can be Ready a moment
# before the Secret is readable here, and a Secret without `tls.crt` serves nothing.
deadline=$(( $(date +%s) + timeout ))
state=""
while [ "$(date +%s)" -lt "$deadline" ]; do
if ready; then
hosting::log "certificate issued into ${secret}"
state="$(cert_state)"
if ready && [ "${state%% *}" = "True" ]; then
hosting::log "certificate issued into ${secret} (Certificate ${secret} is Ready)"
hosting::say tls issued
exit 0
fi
Expand All @@ -79,5 +107,9 @@ done
# Name what to look at. A cert that never issues is almost always DNS not yet resolving to the
# ingress, and the Certificate's own events say so.
reason="$(kubectl -n "$namespace" describe certificate "$secret" 2>/dev/null | tail -20 || true)"
hosting::die "no certificate in ${secret} after ${timeout}s. Until one exists the ingress serves ANOTHER host's default certificate, so this is a failure rather than something to proceed past. Most often the A record has not propagated to the issuer's resolver yet. Certificate events:
if [ "${state%% *}" = "True" ]; then
hosting::die "Certificate ${secret} reports Ready but the Secret still carries no key pair after ${timeout}s. The ingress therefore still serves another host's certificate. Certificate events:
${reason}"
fi
hosting::die "no certificate in ${secret} after ${timeout}s (Certificate Ready: ${state:-no such object}). Until one exists the ingress serves ANOTHER host's default certificate, so this is a failure rather than something to proceed past. Most often the A record has not propagated to the issuer's resolver yet. Certificate events:
${reason}"
62 changes: 55 additions & 7 deletions deploy/aks/operator/bin/hosting-verify
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,15 @@
# DNS resolving to the ingress, the certificate matching the host, and the app returning 200. A
# rollout status proves none of them. This is the step that turns "helm says deployed" into "a
# browser can open it", so it deliberately does NOT pass --insecure.
#
# 🚨 AND IT NAMES WHICH ONE FAILED. Without a certificate of its own, an ingress serves the
# controller's fallback — another instance's certificate, valid but for the wrong host. curl then
# refuses the connection, so this step used to spend 30 attempts × 10 s and report "never
# answered", which reads like a dead app and sends the reader to the pods. It is not the app: it
# is the certificate. So on failure the served certificate is read and compared against the host,
# and the refusal says whether NO certificate was served, the WRONG one was, or TLS was fine and
# the application answered badly. pearl.meshweaver.cloud served CN=memex.meshweaver.cloud for nine
# hours on 2026-09-15 while its portal was healthy (Systemorph/Memex, the pearl Provision).

source "$(dirname "$(readlink -f "${BASH_SOURCE[0]}")")/_common.sh"
# shellcheck disable=SC2034 # read by hosting::die in _common.sh, which shellcheck does not follow here
Expand Down Expand Up @@ -34,22 +43,61 @@ fi
url="https://${host}${path}"
hosting::log "probing ${url}"

# The certificate the host actually serves, as "subject|SAN list" — empty when no handshake.
served_cert() {
local pem
pem="$(echo | openssl s_client -servername "$host" -connect "${host}:443" 2>/dev/null \
| sed -n '/-----BEGIN CERTIFICATE-----/,/-----END CERTIFICATE-----/p')"
[ -n "$pem" ] || return 1
local subject sans
subject="$(printf '%s\n' "$pem" | openssl x509 -noout -subject 2>/dev/null | sed 's/^subject= *//')"
sans="$(printf '%s\n' "$pem" | openssl x509 -noout -ext subjectAltName 2>/dev/null \
| tr ',' '\n' | sed -n 's/.*DNS://p' | tr -d ' ' | tr '\n' ' ')"
printf '%s|%s' "${subject:-unknown}" "${sans}"
}

# Does a name on the certificate cover this host? Exact match, or a wildcard covering ONE label.
covers_host() {
local names="$1" n
for n in $names; do
[ "$n" = "$host" ] && return 0
case "$n" in
\*.*) [ "${host#*.}" = "${n#\*.}" ] && [ "${host%%.*}" != "$host" ] && return 0 ;;
esac
done
return 1
}

last_status="" last_error=""
for i in $(seq 1 "$attempts"); do
# --fail-with-body is not universal on alpine curl; read the status explicitly instead.
last_status="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 20 "$url" 2>/tmp/hosting-verify.err)" || true
last_error="$(cat /tmp/hosting-verify.err 2>/dev/null || true)"
if [ "$last_status" = "200" ] || [ "$last_status" = "302" ] || [ "$last_status" = "401" ]; then
hosting::log "answered ${last_status} on attempt ${i}"
# curl verified the chain and the hostname to get here (no --insecure, deliberately). Read the
# certificate back anyway and assert it names THIS host: the assertion is what keeps a future
# --insecure, or a proxy that terminates elsewhere, from turning this gate into a 200-check.
cert="$(served_cert || true)"
names="${cert#*|}"
if [ -n "$cert" ] && ! covers_host "$names"; then
hosting::die "https://${host}${path} answered ${last_status}, but the certificate served is '${cert%%|*}' (names: ${names:-none}) — not this host's. An instance reachable only over the wrong certificate is not provisioned: every browser refuses it."
fi
hosting::log "answered ${last_status} on attempt ${i} over verified TLS, certificate ${cert%%|*}"
hosting::say verify "$last_status"
# Report the certificate the host actually served, so a wrong-cert ingress is visible in the log
# rather than only in a browser.
subject="$(echo | openssl s_client -servername "$host" -connect "${host}:443" 2>/dev/null \
| openssl x509 -noout -subject 2>/dev/null || true)"
[ -n "$subject" ] && hosting::log "served certificate ${subject}"
hosting::say tls_subject "${cert%%|*}"
exit 0
fi
[ "$i" -lt "$attempts" ] && sleep 10
done

hosting::die "https://${host}${path} never answered (last status '${last_status:-none}'${last_error:+, last transport error: ${last_error}}) after ${attempts} attempts. An instance that helm deployed but nobody can open is not provisioned. Check, in this order: the A record resolves to the ingress IP, the TLS secret matches this host, the portal pods are Ready."
# Which of the three failed? Read the certificate the host serves and say so plainly, instead of
# sending the reader to the pods for what is a certificate problem nine times out of ten.
cert="$(served_cert || true)"
if [ -z "$cert" ]; then
hosting::die "https://${host}${path} never answered and the host served NO certificate at all after ${attempts} attempts (last status '${last_status:-none}'${last_error:+, transport: ${last_error}}). Either the A record does not resolve to the ingress yet, or no TLS secret exists for this host — the TLS step is what creates it."
fi
names="${cert#*|}"
if ! covers_host "$names"; then
hosting::die "https://${host}${path} is served the WRONG certificate: '${cert%%|*}' (names: ${names:-none}), which does not cover ${host}. That is the ingress controller's fallback, served because this host has no certificate of its own — the portal may well be healthy behind it. Fix the certificate (the TLS step), not the pods."
fi
hosting::die "https://${host}${path} never answered (last status '${last_status:-none}'${last_error:+, last transport error: ${last_error}}) after ${attempts} attempts, although TLS is correct: the certificate '${cert%%|*}' covers this host. So DNS and the certificate are fine and the APPLICATION did not answer — check the portal pods and their logs."
1 change: 1 addition & 0 deletions deploy/aks/operator/test/check-rbac-coverage.sh
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,7 @@ resource_of() {
job|jobs) echo "batch jobs" ;;
cronjob|cronjobs|cj) echo "batch cronjobs" ;;
secretproviderclass|secretproviderclasses|spc) echo "secrets-store.csi.x-k8s.io secretproviderclasses" ;;
certificate|certificates|cert) echo "cert-manager.io certificates" ;;
*) echo "" ;;
esac
}
Expand Down
30 changes: 30 additions & 0 deletions deploy/aks/operator/test/run-tests.sh
Original file line number Diff line number Diff line change
Expand Up @@ -127,6 +127,36 @@ refuses "redirect needs a mode" "must be 'suspend' or 'restore'"
refuses "deploy needs --release" "missing required flag --release" hosting-deploy --namespace n --database d
refuses "verify needs --host" "missing required flag --host" hosting-verify
refuses "unknown flags are not ignored" "unknown argument" hosting-verify --host h --nope 1
# ---------------------------------------------------------------------------
# THE CERTIFICATE, end to end. A new instance is not provisioned until its own host answers over
# its OWN certificate: an ingress without one is served the controller's fallback — another
# instance's certificate — and pearl.meshweaver.cloud spent nine hours in exactly that state on
# 2026-09-15 while its portal was healthy. These assert the two halves that were missing: the TLS
# step refuses a record that asks cert-manager for nothing, and the verify step tells a wrong
# certificate apart from a dead application instead of blaming the pods for both.
# ---------------------------------------------------------------------------
refuses "tls refuses an issuer of 'none'" "asks cert-manager for nothing" \
hosting-tls --namespace n --host h.example.com --issuer none
refuses "tls needs --namespace" "missing required flag --namespace" hosting-tls --host h.example.com
refuses "tls rejects unknown flags" "unknown argument" hosting-tls --namespace n --host h.example.com --nope 1

VERIFY_STUBS="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/stubs/verify" && pwd)"
_verify() { env PATH="$VERIFY_STUBS:$PATH" HOSTING_VERIFY_ATTEMPTS=1 "$@" hosting-verify --host instance.example.com; }

# The fallback-certificate shape: the host answers nothing over TLS and serves a certificate for
# ANOTHER host. The refusal must name the certificate and send the reader to the certificate.
refuses "verify names a WRONG certificate rather than blaming the pods" "served the WRONG certificate" \
_verify HOSTING_VERIFY_STUB_CN=memex.meshweaver.cloud HOSTING_VERIFY_STUB_SANS=memex.meshweaver.cloud
# No certificate at all — DNS or the TLS step, not the application.
refuses "verify says when NO certificate is served" "served NO certificate at all" \
_verify HOSTING_VERIFY_STUB_NOCERT=1
# TLS is correct and the app is not: the one case where the pods ARE the answer.
refuses "verify blames the application only when TLS is correct" "the APPLICATION did not answer" \
_verify HOSTING_VERIFY_STUB_CN=instance.example.com HOSTING_VERIFY_STUB_SANS=instance.example.com
# A wildcard covers one label: *.example.com is this host's certificate, not a wrong one.
refuses "verify accepts a wildcard certificate as this host's" "the APPLICATION did not answer" \
_verify HOSTING_VERIFY_STUB_CN='*.example.com' HOSTING_VERIFY_STUB_SANS='*.example.com'

refuses "pull-secret needs --namespace" "missing required flag --namespace" hosting-pull-secret --registry r.example.test --vault V --secret S
refuses "pull-secret needs --registry" "missing required flag --registry" hosting-pull-secret --namespace n --vault V --secret S
refuses "pull-secret needs --vault" "missing required flag --vault" hosting-pull-secret --namespace n --registry r.example.test --secret S
Expand Down
8 changes: 8 additions & 0 deletions deploy/aks/operator/test/stubs/verify/curl
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
#!/usr/bin/env bash
# A stand-in `curl` for the hosting-verify certificate tests: answers with the status in
# $HOSTING_VERIFY_STUB_STATUS (default 000 = no answer, which is what a refused TLS handshake
# looks like to the real curl) and writes $HOSTING_VERIFY_STUB_ERR to the -o/2> error file.
set -u
printf '%s' "${HOSTING_VERIFY_STUB_STATUS:-000}"
[ -n "${HOSTING_VERIFY_STUB_ERR:-}" ] && printf '%s\n' "$HOSTING_VERIFY_STUB_ERR" >&2
exit 0
18 changes: 18 additions & 0 deletions deploy/aks/operator/test/stubs/verify/openssl
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
#!/usr/bin/env bash
# A stand-in `openssl` for the hosting-verify certificate tests. It serves ONE certificate, named
# by $HOSTING_VERIFY_STUB_CN and $HOSTING_VERIFY_STUB_SANS; with HOSTING_VERIFY_STUB_NOCERT set it
# serves none, which is the "no TLS at all" shape.
set -u
case "${1:-}" in
s_client)
[ -n "${HOSTING_VERIFY_STUB_NOCERT:-}" ] && exit 1
printf -- '-----BEGIN CERTIFICATE-----\nstub\n-----END CERTIFICATE-----\n' ;;
x509)
[ -n "${HOSTING_VERIFY_STUB_NOCERT:-}" ] && exit 1
cat >/dev/null
case "$*" in
*subjectAltName*) printf 'X509v3 Subject Alternative Name: \n DNS:%s\n' "${HOSTING_VERIFY_STUB_SANS:-${HOSTING_VERIFY_STUB_CN:-other.example.com}}" ;;
*subject*) printf 'subject=CN=%s\n' "${HOSTING_VERIFY_STUB_CN:-other.example.com}" ;;
esac ;;
*) exit 1 ;;
esac
Loading
Loading