From b13f3744a5ff5b4823f5e6aa129b3cd8de4e7dfb Mon Sep 17 00:00:00 2001 From: David Binney Date: Mon, 10 Aug 2026 22:52:37 +1000 Subject: [PATCH 1/4] fix: k8s probe defaults and drop imagePullPolicy Always Use versioned image so kube default pull is IfNotPresent. Align readiness with standard period/failureThreshold; keep short timeouts for delay demos. Startup still ~60s cold-start window. --- k8s-cluster-util-apis.yml | 28 +++++++++++++++++++--------- 1 file changed, 19 insertions(+), 9 deletions(-) diff --git a/k8s-cluster-util-apis.yml b/k8s-cluster-util-apis.yml index b1542de..67e8659 100644 --- a/k8s-cluster-util-apis.yml +++ b/k8s-cluster-util-apis.yml @@ -19,13 +19,18 @@ spec: spec: containers: - name: cluster-utils-api - image: donkeyx/cluster-utils-api:latest - imagePullPolicy: Always + # Prefer a version tag so kube default pull policy is IfNotPresent. + # :latest defaults to Always even if you omit imagePullPolicy. + image: donkeyx/cluster-utils-api:2.5.0 ports: - name: http containerPort: 8080 - # Short timeouts so LIVE/READY/STARTUP delaySeconds are easy to trip. - # startup runs until first success, then kube moves on to live+ready. + # Probes — close to kube defaults; short timeout so app delaySeconds demos work. + # Defaults if omitted: periodSeconds=10, timeoutSeconds=1, successThreshold=1, + # failureThreshold=3, initialDelaySeconds=0. + # + # startup: only until first success, then live+ready take over. + # period 2 × failureThreshold 30 ≈ 60s max cold-start window. startupProbe: httpGet: path: /startupz @@ -33,6 +38,8 @@ spec: periodSeconds: 2 timeoutSeconds: 1 failureThreshold: 30 + successThreshold: 1 + # liveness: restart if process is wedged (defaults are fine for this app) livenessProbe: httpGet: path: /livez @@ -40,13 +47,16 @@ spec: periodSeconds: 10 timeoutSeconds: 1 failureThreshold: 3 + successThreshold: 1 + # readiness: leave Service endpoints if not ready (~30s of fails with defaults) readinessProbe: httpGet: path: /readyz port: http - periodSeconds: 5 + periodSeconds: 10 timeoutSeconds: 1 - failureThreshold: 2 + failureThreshold: 3 + successThreshold: 1 env: - name: PORT value: "8080" @@ -91,11 +101,11 @@ spec: # value: "true" resources: requests: - cpu: "0.1" + cpu: "100m" memory: "50Mi" limits: - cpu: "0.5" - memory: 100Mi + cpu: "500m" + memory: "100Mi" --- apiVersion: v1 From defb005f769e8f49a27c2498140fb8ad23e81b73 Mon Sep 17 00:00:00 2001 From: David Binney Date: Mon, 10 Aug 2026 22:53:05 +1000 Subject: [PATCH 2/4] docs: align README with k8s probe and pull defaults --- README.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 99c9769..081dcb8 100644 --- a/README.md +++ b/README.md @@ -680,13 +680,13 @@ kubectl get pods,svc -n default # service: cluster-utils-api-svc:8080 ``` -Sample manifest probes: +Sample manifest probes (near kube defaults; timeout 1s so delay demos trip easily): -- **startupProbe** → `/startupz` (timeout 1s, period 2s) -- **livenessProbe** → `/livez` (timeout 1s) -- **readinessProbe** → `/readyz` (timeout 1s) +- **startupProbe** → `/startupz` (period 2s × failureThreshold 30 ≈ 60s cold start) +- **livenessProbe** → `/livez` (period 10s, failureThreshold 3) +- **readinessProbe** → `/readyz` (period 10s, failureThreshold 3) -Uncomment the env examples in the yaml to break things on purpose, or flip live via `/a/control/probes` after you grab the token from pod logs (or set `AUTH_TOKEN`). +Image is pinned to a version tag so default pull is **IfNotPresent** (bare `:latest` still forces Always in kube). Uncomment env examples to break probes, or flip via `/a/control/probes`. ```bash kubectl -n default port-forward svc/cluster-utils-api-svc 8080:8080 From 5a4a7580918f69f640ac1e6100dd8a5fbe2143ee Mon Sep 17 00:00:00 2001 From: David Binney Date: Mon, 10 Aug 2026 23:01:54 +1000 Subject: [PATCH 3/4] fix: GHCR image and resource requests from measured usage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prefer ghcr.io for the sample deploy. requests ~90% of observed process RSS (~40Mi → 36Mi) and ~10m CPU; limits with headroom for GC. --- k8s-cluster-util-apis.yml | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/k8s-cluster-util-apis.yml b/k8s-cluster-util-apis.yml index 67e8659..1558a6c 100644 --- a/k8s-cluster-util-apis.yml +++ b/k8s-cluster-util-apis.yml @@ -19,9 +19,10 @@ spec: spec: containers: - name: cluster-utils-api - # Prefer a version tag so kube default pull policy is IfNotPresent. - # :latest defaults to Always even if you omit imagePullPolicy. - image: donkeyx/cluster-utils-api:2.5.0 + # GHCR primary (Hub still works as mirror). Version tag → default pull IfNotPresent. + # :latest still forces Always in kube even without imagePullPolicy. + image: ghcr.io/donkeyx/cluster-utils-api:2.5.0 + # Alternative: docker.io/donkeyx/cluster-utils-api:2.5.0 ports: - name: http containerPort: 8080 @@ -99,13 +100,15 @@ spec: # include kube probe paths in traces (off by default — noisy) # - name: OTEL_TRACE_PROBES # value: "true" + # Sized from podman run of :2.5.0 (idle~14–15Mi cgroup, process_resident~40Mi, + # CPU ~0.3–1% under light load). requests ≈ 90% of observed process RSS; limits headroom. resources: requests: - cpu: "100m" - memory: "50Mi" + cpu: "10m" + memory: "36Mi" limits: - cpu: "500m" - memory: "100Mi" + cpu: "100m" + memory: "80Mi" --- apiVersion: v1 From 6f0b2be95cf0c02f18a878bf4dc76f3898129e74 Mon Sep 17 00:00:00 2001 From: David Binney Date: Mon, 10 Aug 2026 23:03:34 +1000 Subject: [PATCH 4/4] fix: bump memory request to 40Mi after recheck (~43Mi RSS) --- k8s-cluster-util-apis.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/k8s-cluster-util-apis.yml b/k8s-cluster-util-apis.yml index 1558a6c..0de1b4b 100644 --- a/k8s-cluster-util-apis.yml +++ b/k8s-cluster-util-apis.yml @@ -100,12 +100,12 @@ spec: # include kube probe paths in traces (off by default — noisy) # - name: OTEL_TRACE_PROBES # value: "true" - # Sized from podman run of :2.5.0 (idle~14–15Mi cgroup, process_resident~40Mi, - # CPU ~0.3–1% under light load). requests ≈ 90% of observed process RSS; limits headroom. + # Sized from podman run of :2.5.0 (cgroup ~15–16Mi, process_resident ~40–43Mi, + # CPU ~0.1–1% under light load). requests ≈ 90% of process RSS; limits headroom. resources: requests: cpu: "10m" - memory: "36Mi" + memory: "40Mi" limits: cpu: "100m" memory: "80Mi"