From 75a534e2000c6f0135eea4ae680265a09f3c2ca7 Mon Sep 17 00:00:00 2001 From: Andrei Dorin Oprea Date: Fri, 18 Sep 2026 10:51:17 +0300 Subject: [PATCH 1/2] feat: make the probe timings configurable The probes have been retuned in place three times, each time to suit one cluster: - #2254 added the liveness probe. - #2844 raised initialDelaySeconds because the operator "was running into timeouts on my arm64 mac machine ... so operator gets more time to load and apply CRDs". - #4791 found that delay now penalised fast startups - "ASO finish starting up after about 10s, then idle for an extra ~45s" - and replaced it with a startup probe at 10s x 12. That budget is now too tight at the other end of the range. The operator applies its CRDs and starts a controller per installed CRD before it serves /healthz, so on a cluster with a large number of CRDs the kubelet kills it mid-apply and it never finishes starting. Rather than move the constant a fourth time, this exposes the timings as Helm values. The defaults are what the chart applies today, so the only change to the rendered output is that the Kubernetes defaults for the liveness and readiness probes are now written out explicitly. The probe paths and ports stay fixed; they have never been the problem. --- .../guide/diagnosing-problems/_index.md | 25 +++++++++++++++++++ ...ureserviceoperator-controller-manager.yaml | 11 ++++++-- v2/charts/azure-service-operator/values.yaml | 21 ++++++++++++++++ v2/config/manager/manager.yaml | 7 ++++++ 4 files changed, 62 insertions(+), 2 deletions(-) diff --git a/docs/hugo/content/guide/diagnosing-problems/_index.md b/docs/hugo/content/guide/diagnosing-problems/_index.md index 1332e5588f3..9fc7655c3d9 100644 --- a/docs/hugo/content/guide/diagnosing-problems/_index.md +++ b/docs/hugo/content/guide/diagnosing-problems/_index.md @@ -26,6 +26,31 @@ Events: ... ``` +### Operator pod is restarted before it finishes applying CRDs + +On a cluster with a large number of CRDs, the operator can be killed by the kubelet while it is still starting. +The pod restarts repeatedly and never reports ready, and `kubectl describe pod` shows the startup probe failing: + +``` + Warning Unhealthy 2m (x12 over 4m) kubelet Startup probe failed: Get "http://10.244.0.9:8081/healthz": context deadline exceeded +``` + +The operator applies its CRDs and starts a controller for each installed CRD before it serves `/healthz`, so on +a large cluster that work can outlast the startup probe's budget of `periodSeconds` x `failureThreshold`. + +Give it more time by raising `failureThreshold`. In Helm: +```yaml +probes: + startup: + failureThreshold: 60 +``` + +That allows ten minutes rather than the default two. Raising it costs nothing when the operator starts quickly, +because the probe stops as soon as it first succeeds. + +Installing fewer CRDs also shortens startup, since the operator only applies and watches the CRDs it is asked +for. See [CRD management]( {{< relref "crd-management" >}} ). + ### Helm installation via Argo missing ClusterRole and other resources See reference issue [#4184](https://github.com/Azure/azure-service-operator/issues/4184). diff --git a/v2/charts/azure-service-operator/templates/apps_v1_deployment_azureserviceoperator-controller-manager.yaml b/v2/charts/azure-service-operator/templates/apps_v1_deployment_azureserviceoperator-controller-manager.yaml index 680e16c75c7..5cd23d41b0e 100644 --- a/v2/charts/azure-service-operator/templates/apps_v1_deployment_azureserviceoperator-controller-manager.yaml +++ b/v2/charts/azure-service-operator/templates/apps_v1_deployment_azureserviceoperator-controller-manager.yaml @@ -220,6 +220,9 @@ spec: httpGet: path: /healthz port: 8081 + periodSeconds: {{ .Values.probes.liveness.periodSeconds }} + failureThreshold: {{ .Values.probes.liveness.failureThreshold }} + timeoutSeconds: {{ .Values.probes.liveness.timeoutSeconds }} name: manager ports: - containerPort: {{ .Values.webhook.port }} @@ -237,6 +240,9 @@ spec: httpGet: path: /readyz port: 8081 + periodSeconds: {{ .Values.probes.readiness.periodSeconds }} + failureThreshold: {{ .Values.probes.readiness.failureThreshold }} + timeoutSeconds: {{ .Values.probes.readiness.timeoutSeconds }} {{- with .Values.resources }} resources: {{- toYaml . | nindent 10 }} @@ -249,8 +255,9 @@ spec: httpGet: path: /healthz port: 8081 - periodSeconds: 10 - failureThreshold: 12 + periodSeconds: {{ .Values.probes.startup.periodSeconds }} + failureThreshold: {{ .Values.probes.startup.failureThreshold }} + timeoutSeconds: {{ .Values.probes.startup.timeoutSeconds }} volumeMounts: - mountPath: /var/run/secrets/tokens name: azure-identity diff --git a/v2/charts/azure-service-operator/values.yaml b/v2/charts/azure-service-operator/values.yaml index 1789e87f0f6..a3712db2aa2 100644 --- a/v2/charts/azure-service-operator/values.yaml +++ b/v2/charts/azure-service-operator/values.yaml @@ -236,6 +236,27 @@ resources: go: memLimit: 400MiB # This should be set to ~80-90% of the hard memory limit set above in resources +# probes configures the timings of the operator's liveness, readiness and startup probes. The defaults are the +# values the chart has always applied, so leaving this alone changes nothing. +# The startup probe is the one most likely to need adjusting. The operator applies its CRDs and starts a +# controller per installed CRD before it serves /healthz, so on a cluster with a large number of CRDs the default +# budget of periodSeconds x failureThreshold can expire while it is still working, and the kubelet restarts it +# mid-apply. Raise failureThreshold in that case. +# The probe paths and ports are not configurable; only these timings are. +probes: + startup: + periodSeconds: 10 + failureThreshold: 12 + timeoutSeconds: 1 + liveness: + periodSeconds: 10 + failureThreshold: 3 + timeoutSeconds: 1 + readiness: + periodSeconds: 10 + failureThreshold: 3 + timeoutSeconds: 1 + # Number of old history to retain to allow rollback # Default Kubernetes value is set to 10 revisionHistoryLimit: 10 diff --git a/v2/config/manager/manager.yaml b/v2/config/manager/manager.yaml index 734438c90f3..388994a592b 100644 --- a/v2/config/manager/manager.yaml +++ b/v2/config/manager/manager.yaml @@ -61,14 +61,21 @@ spec: port: 8081 periodSeconds: 10 failureThreshold: 12 + timeoutSeconds: 1 livenessProbe: httpGet: path: /healthz port: 8081 + periodSeconds: 10 + failureThreshold: 3 + timeoutSeconds: 1 readinessProbe: httpGet: path: /readyz port: 8081 + periodSeconds: 10 + failureThreshold: 3 + timeoutSeconds: 1 image: controller:latest name: manager resources: From ea7e84024e4b7e96dc26263afb97dcec9bf8ed23 Mon Sep 17 00:00:00 2001 From: Andrei Dorin Oprea Date: Mon, 21 Sep 2026 11:37:40 +0300 Subject: [PATCH 2/2] docs: tighten the probes comment in values.yaml Applies the review suggestion on the probes block: drops the sentence about the defaults matching what the chart already applied, which belongs in the PR description rather than the values file. Co-Authored-By: Claude Opus 5 (1M context) --- v2/charts/azure-service-operator/values.yaml | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/v2/charts/azure-service-operator/values.yaml b/v2/charts/azure-service-operator/values.yaml index a3712db2aa2..6fd22eb95b3 100644 --- a/v2/charts/azure-service-operator/values.yaml +++ b/v2/charts/azure-service-operator/values.yaml @@ -236,13 +236,12 @@ resources: go: memLimit: 400MiB # This should be set to ~80-90% of the hard memory limit set above in resources -# probes configures the timings of the operator's liveness, readiness and startup probes. The defaults are the -# values the chart has always applied, so leaving this alone changes nothing. +# probes configures the timings of the operator's liveness, readiness and startup probes. # The startup probe is the one most likely to need adjusting. The operator applies its CRDs and starts a # controller per installed CRD before it serves /healthz, so on a cluster with a large number of CRDs the default # budget of periodSeconds x failureThreshold can expire while it is still working, and the kubelet restarts it # mid-apply. Raise failureThreshold in that case. -# The probe paths and ports are not configurable; only these timings are. +# The probe paths and ports are not configurable; only the timings are. probes: startup: periodSeconds: 10