diff --git a/README.md b/README.md index 3ab65085d..a85458285 100644 --- a/README.md +++ b/README.md @@ -202,6 +202,9 @@ kubectl -n kafka run kafka-consumer -it --image=adobe/kafka:2.13-3.9.1 --rm=true ## Documentation +For opt-in, broker-only KRaft metadata PVCs with an automatic, one-broker-at-a-time live migration, +see [dedicated broker metadata storage](docs/broker-metadata-storage.md). + For detailed documentation on the Koperator project, see the [Koperator documentation website](https://opensource.adobe.com/koperator/). ## Issues and contributions diff --git a/api/v1beta1/common_types.go b/api/v1beta1/common_types.go index 85e092e25..58d64dce7 100644 --- a/api/v1beta1/common_types.go +++ b/api/v1beta1/common_types.go @@ -50,6 +50,9 @@ type SSLClientAuthentication string // PerBrokerConfigurationState holds info about the per-broker configuration state type PerBrokerConfigurationState string +// MetadataStorageState holds info about the dedicated KRaft metadata storage of a broker +type MetadataStorageState string + // ExternalListenerConfigNames type describes a collection of external listener names type ExternalListenerConfigNames []string @@ -253,8 +256,17 @@ type BrokerState struct { Image string `json:"image,omitempty"` // Compressed data from broker configuration to restore broker pod in specific cases ConfigurationBackup string `json:"configurationBackup,omitempty"` + // MetadataStorageState is Ready once the broker runs from its dedicated metadata storage. + // Data disks of a broker with metadataStorage can be removed only after that. + MetadataStorageState MetadataStorageState `json:"metadataStorageState,omitempty"` } +const ( + // MetadataStorageReady states that a replacement broker pod became ready using its + // dedicated metadata storage, so the metadata no longer lives on any data disk + MetadataStorageReady MetadataStorageState = "Ready" +) + const ( // Configured states the broker is running Configured RackAwarenessState = "Configured" diff --git a/api/v1beta1/kafkacluster_types.go b/api/v1beta1/kafkacluster_types.go index 0e3642c4e..5001accb4 100644 --- a/api/v1beta1/kafkacluster_types.go +++ b/api/v1beta1/kafkacluster_types.go @@ -301,18 +301,23 @@ type BrokerConfig struct { // +kubebuilder:validation:Items:Type=string // +kubebuilder:validation:Items:Enum=controller;broker // +optional - Roles []string `json:"processRoles,omitempty"` - Image string `json:"image,omitempty"` - MetricsReporterImage string `json:"metricsReporterImage,omitempty"` - Config string `json:"config,omitempty"` - StorageConfigs []StorageConfig `json:"storageConfigs,omitempty"` - ServiceAccountName string `json:"serviceAccountName,omitempty"` - Resources *corev1.ResourceRequirements `json:"resourceRequirements,omitempty"` - ImagePullSecrets []corev1.LocalObjectReference `json:"imagePullSecrets,omitempty"` - NodeSelector map[string]string `json:"nodeSelector,omitempty"` - Tolerations []corev1.Toleration `json:"tolerations,omitempty"` - KafkaHeapOpts string `json:"kafkaHeapOpts,omitempty"` - KafkaJVMPerfOpts string `json:"kafkaJvmPerfOpts,omitempty"` + Roles []string `json:"processRoles,omitempty"` + Image string `json:"image,omitempty"` + MetricsReporterImage string `json:"metricsReporterImage,omitempty"` + Config string `json:"config,omitempty"` + StorageConfigs []StorageConfig `json:"storageConfigs,omitempty"` + // MetadataStorage optionally isolates KRaft metadata on a dedicated PVC. + // Only broker-only KRaft nodes may use it. It is not a data or Cruise Control disk. + // Once enabled it cannot be removed or relocated; reverse migration is not supported. + // +optional + MetadataStorage *StorageConfig `json:"metadataStorage,omitempty"` + ServiceAccountName string `json:"serviceAccountName,omitempty"` + Resources *corev1.ResourceRequirements `json:"resourceRequirements,omitempty"` + ImagePullSecrets []corev1.LocalObjectReference `json:"imagePullSecrets,omitempty"` + NodeSelector map[string]string `json:"nodeSelector,omitempty"` + Tolerations []corev1.Toleration `json:"tolerations,omitempty"` + KafkaHeapOpts string `json:"kafkaHeapOpts,omitempty"` + KafkaJVMPerfOpts string `json:"kafkaJvmPerfOpts,omitempty"` // Override for the default log4j configuration Log4jConfig string `json:"log4jConfig,omitempty"` // Custom annotations for the broker pods - e.g.: Prometheus scraping annotations: @@ -1300,12 +1305,24 @@ func (b *Broker) GetBrokerConfig(kafkaClusterSpec KafkaClusterSpec) (*BrokerConf } envs := mergeEnvs(kafkaClusterSpec, &groupConfig, bConfig) + // A broker override replaces this single storage object, rather than merging + // half of a group PVC spec into a broker-local mount. + metadataStorage := bConfig.MetadataStorage + if metadataStorage == nil { + metadataStorage = groupConfig.MetadataStorage + } + if metadataStorage != nil { + metadataStorage = metadataStorage.DeepCopy() + } err = mergo.Merge(bConfig, groupConfig, mergo.WithAppendSlice) if err != nil { return nil, errors.WrapIf(err, "could not merge brokerConfig with ConfigGroup") } bConfig.StorageConfigs = dedupStorageConfigs(bConfig.StorageConfigs) + if metadataStorage != nil { + bConfig.MetadataStorage = metadataStorage.DeepCopy() + } if groupConfig.Affinity != nil || bConfig.Affinity != nil { bConfig.Affinity = dstAffinity } diff --git a/api/v1beta1/metadata_storage.go b/api/v1beta1/metadata_storage.go new file mode 100644 index 000000000..64d9aad2d --- /dev/null +++ b/api/v1beta1/metadata_storage.go @@ -0,0 +1,92 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package v1beta1 + +import ( + "fmt" + "path" + "reflect" + "regexp" + "strings" +) + +// MetadataStorageVolumeName is the pod volume name reserved for the dedicated metadata PVC. +const MetadataStorageVolumeName = "kraft-metadata" + +var metadataMountPathPattern = regexp.MustCompile(`^/[a-zA-Z0-9_./-]+$`) + +// ValidateMetadataStorage validates the effective (group-merged) broker configuration. +func (b *BrokerConfig) ValidateMetadataStorage(kraft bool) error { + if b == nil || b.MetadataStorage == nil { + return nil + } + if !kraft || !b.IsBrokerOnlyNode() || len(b.Roles) != 1 { + return fmt.Errorf("metadataStorage requires a broker-only KRaft node") + } + m := b.MetadataStorage + if m.PvcSpec == nil || m.EmptyDir != nil { + return fmt.Errorf("metadataStorage requires pvcSpec and does not support emptyDir") + } + if m.PvcSpec.VolumeMode != nil && *m.PvcSpec.VolumeMode != "Filesystem" { + return fmt.Errorf("metadataStorage requires a filesystem PVC") + } + if !metadataMountPathPattern.MatchString(m.MountPath) || path.Clean(m.MountPath) != m.MountPath || m.MountPath == "/" { + return fmt.Errorf("metadataStorage mountPath must be a clean absolute path using letters, digits, '/', '.', '_' or '-'") + } + paths := []string{"/config", "/opt", "/etc", "/var/run", "/run", "/dev", "/proc", "/sys", "/usr", "/bin", "/sbin", "/lib", "/lib64", "/tmp"} + for _, s := range b.StorageConfigs { + // Ephemeral data cannot be a migration source, and durable metadata gains nothing + // beside ephemeral replicas. + if s.PvcSpec == nil || s.EmptyDir != nil { + return fmt.Errorf("metadataStorage requires all data storageConfigs to be PVC-backed; %q is not", s.MountPath) + } + paths = append(paths, s.MountPath) + } + for _, v := range b.VolumeMounts { + if v.Name == MetadataStorageVolumeName { + return fmt.Errorf("%s is a reserved volume name", MetadataStorageVolumeName) + } + paths = append(paths, v.MountPath) + } + for _, v := range b.Volumes { + if v.Name == MetadataStorageVolumeName { + return fmt.Errorf("%s is a reserved volume name", MetadataStorageVolumeName) + } + } + for _, p := range paths { + if StoragePathsOverlap(m.MountPath, p) { + return fmt.Errorf("metadataStorage mountPath overlaps %q", p) + } + } + return nil +} + +// StoragePathsOverlap includes nested mounts, which can hide an existing volume. +func StoragePathsOverlap(a, b string) bool { + a, b = path.Clean(a), path.Clean(b) + return a == b || strings.HasPrefix(a, b+"/") || strings.HasPrefix(b, a+"/") +} + +// MetadataStorageLocationEqual allows capacity changes, but no relocation or PVC replacement. +func MetadataStorageLocationEqual(a, b *StorageConfig) bool { + if a == nil || b == nil { + return a == b + } + a, b = a.DeepCopy(), b.DeepCopy() + if a.PvcSpec != nil && b.PvcSpec != nil { + a.PvcSpec.Resources = b.PvcSpec.Resources + } + return reflect.DeepEqual(a, b) +} diff --git a/api/v1beta1/metadata_storage_test.go b/api/v1beta1/metadata_storage_test.go new file mode 100644 index 000000000..892f11877 --- /dev/null +++ b/api/v1beta1/metadata_storage_test.go @@ -0,0 +1,98 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package v1beta1 + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" +) + +func TestMetadataStorageValidationAndMapping(t *testing.T) { + base := &BrokerConfig{ + Roles: []string{"broker"}, + StorageConfigs: []StorageConfig{{MountPath: "/csi-kafka-logs1", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}}, + MetadataStorage: &StorageConfig{MountPath: "/csi-kafka-metadata", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}, + } + for _, tc := range []struct { + name string + kraft bool + change func(*BrokerConfig) + valid bool + }{ + {"broker", true, func(*BrokerConfig) {}, true}, + {"default unchanged", false, func(b *BrokerConfig) { b.MetadataStorage = nil }, true}, + {"zk", false, func(*BrokerConfig) {}, false}, + {"controller", true, func(b *BrokerConfig) { b.Roles = []string{"controller"} }, false}, + {"combined", true, func(b *BrokerConfig) { b.Roles = []string{"broker", "controller"} }, false}, + {"emptydir", true, func(b *BrokerConfig) { b.MetadataStorage.EmptyDir = &corev1.EmptyDirVolumeSource{} }, false}, + {"missing pvc", true, func(b *BrokerConfig) { b.MetadataStorage.PvcSpec = nil }, false}, + {"emptydir data", true, func(b *BrokerConfig) { + b.StorageConfigs = append(b.StorageConfigs, StorageConfig{MountPath: "/kafka-logs", EmptyDir: &corev1.EmptyDirVolumeSource{}}) + }, false}, + {"emptydir data without metadata storage", true, func(b *BrokerConfig) { + b.MetadataStorage = nil + b.StorageConfigs = []StorageConfig{{MountPath: "/kafka-logs", EmptyDir: &corev1.EmptyDirVolumeSource{}}} + }, true}, + {"block pvc", true, func(b *BrokerConfig) { + mode := corev1.PersistentVolumeBlock + b.MetadataStorage.PvcSpec.VolumeMode = &mode + }, false}, + {"reserved volume", true, func(b *BrokerConfig) { b.Volumes = []corev1.Volume{{Name: "kraft-metadata"}} }, false}, + {"collision", true, func(b *BrokerConfig) { b.MetadataStorage.MountPath = "/csi-kafka-logs1" }, false}, + {"nested collision", true, func(b *BrokerConfig) { b.MetadataStorage.MountPath = "/csi-kafka-logs1/metadata" }, false}, + {"reserved mount", true, func(b *BrokerConfig) { b.MetadataStorage.MountPath = "/config/metadata" }, false}, + {"custom mount", true, func(b *BrokerConfig) { b.VolumeMounts = []corev1.VolumeMount{{MountPath: "/csi-kafka-metadata/kafka"}} }, false}, + {"relative path", true, func(b *BrokerConfig) { b.MetadataStorage.MountPath = "metadata" }, false}, + {"unclean path", true, func(b *BrokerConfig) { b.MetadataStorage.MountPath = "/metadata/../logs" }, false}, + } { + t.Run(tc.name, func(t *testing.T) { + b := base.DeepCopy() + tc.change(b) + if err := b.ValidateMetadataStorage(tc.kraft); (err == nil) != tc.valid { + t.Fatalf("valid=%v, got %v", tc.valid, err) + } + }) + } + group := base.DeepCopy() + group.MetadataStorage.PvcSpec.StorageClassName = ptrString("group-class") + broker := Broker{BrokerConfigGroup: "brokers"} + spec := KafkaClusterSpec{BrokerConfigGroups: map[string]BrokerConfig{"brokers": *group}} + inherited, err := broker.GetBrokerConfig(spec) + if err != nil || inherited.MetadataStorage.MountPath != base.MetadataStorage.MountPath || len(inherited.StorageConfigs) != 1 { + t.Fatalf("inheritance failed: %v, %v", inherited, err) + } + broker.BrokerConfig = &BrokerConfig{MetadataStorage: &StorageConfig{MountPath: "/local-metadata", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}} + local, err := broker.GetBrokerConfig(spec) + if err != nil || local.MetadataStorage.MountPath != "/local-metadata" || local.MetadataStorage.PvcSpec.StorageClassName != nil { + t.Fatalf("metadata override was merged rather than replaced: %v, %v", local, err) + } + local.MetadataStorage.MountPath = "/mutated" + if broker.BrokerConfig.MetadataStorage.MountPath != "/local-metadata" || group.MetadataStorage.MountPath != base.MetadataStorage.MountPath { + t.Fatal("merged config aliases its inputs") + } + a, b := base.MetadataStorage.DeepCopy(), base.MetadataStorage.DeepCopy() + b.PvcSpec.Resources.Requests = corev1.ResourceList{corev1.ResourceStorage: resource.MustParse("20Gi")} + if !MetadataStorageLocationEqual(a, b) { + t.Fatal("capacity change should not be a relocation") + } + b.MountPath = "/elsewhere" + if MetadataStorageLocationEqual(a, b) || MetadataStorageLocationEqual(a, nil) { + t.Fatal("removal/relocation allowed") + } +} + +func ptrString(value string) *string { return &value } diff --git a/api/v1beta1/zz_generated.deepcopy.go b/api/v1beta1/zz_generated.deepcopy.go index d5cfaf00d..b937991bb 100644 --- a/api/v1beta1/zz_generated.deepcopy.go +++ b/api/v1beta1/zz_generated.deepcopy.go @@ -78,6 +78,11 @@ func (in *BrokerConfig) DeepCopyInto(out *BrokerConfig) { (*in)[i].DeepCopyInto(&(*out)[i]) } } + if in.MetadataStorage != nil { + in, out := &in.MetadataStorage, &out.MetadataStorage + *out = new(StorageConfig) + (*in).DeepCopyInto(*out) + } if in.Resources != nil { in, out := &in.Resources, &out.Resources *out = new(v1.ResourceRequirements) diff --git a/charts/kafka-operator/crds/kafkaclusters.yaml b/charts/kafka-operator/crds/kafkaclusters.yaml index ab749490c..4bce156f7 100644 --- a/charts/kafka-operator/crds/kafkaclusters.yaml +++ b/charts/kafka-operator/crds/kafkaclusters.yaml @@ -4440,6 +4440,261 @@ spec: log4jConfig: description: Override for the default log4j configuration type: string + metadataStorage: + description: |- + MetadataStorage optionally isolates KRaft metadata on a dedicated PVC. + Only broker-only KRaft nodes may use it. It is not a data or Cruise Control disk. + Once enabled it cannot be removed or relocated; reverse migration is not supported. + properties: + emptyDir: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes#emptydir is used + as storage for Kafka broker log dirs. The use of empty dir as Kafka broker storage is useful in development + environments where data loss is not a concern as data stored on emptydir backed storage is lost at pod restarts. + Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + medium: + description: |- + medium represents what type of storage medium should back this directory. + The default is "" which means to use the node's default medium. + Must be an empty string (default) or Memory. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + type: string + mode: + description: |- + mode specifies the permission bits for the emptyDir directory, in numeric + notation (e.g., 0755, 01777). Must be a value between 0000 and 01777. + If not specified, defaults to 0777. + This might be in conflict with other options that affect the file + mode, like fsGroup. If fsGroup is specified, the fsGroup permissions + will override the mode specified here. + This field has no effect on Windows. + This field is alpha and requires EmptyDirVolumeMode featuregate to be enabled. + format: int32 + type: integer + sizeLimit: + anyOf: + - type: integer + - type: string + description: |- + sizeLimit is the total amount of local storage required for this EmptyDir volume. + The size limit is also applicable for memory medium. + The maximum usage on memory medium EmptyDir would be the minimum value between + the SizeLimit specified here and the sum of memory limits of all containers in a pod. + The default is nil which means that the limit is undefined. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + type: object + mountPath: + type: string + pvcSpec: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes/#persistentvolumeclaim is used + as storage for Kafka broker log dirs. Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + accessModes: + description: |- + accessModes contains the desired access modes the volume should have. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#access-modes-1 + items: + type: string + type: array + x-kubernetes-list-type: atomic + dataSource: + description: |- + dataSource field can be used to specify either: + * An existing VolumeSnapshot object (snapshot.storage.k8s.io/VolumeSnapshot) + * An existing PVC (PersistentVolumeClaim) + If the provisioner or an external controller can support the specified data source, + it will create a new volume based on the contents of the specified data source. + dataSource contents will be copied to dataSourceRef, and dataSourceRef contents will be + copied to dataSource when dataSourceRef.namespace is not specified. + If the namespace is specified, then dataSourceRef will not be copied to dataSource. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + required: + - kind + - name + type: object + x-kubernetes-map-type: atomic + dataSourceRef: + description: |- + dataSourceRef specifies the object from which to populate the volume with data, if a non-empty + volume is desired. This may be any object from a non-empty API group (non + core object) or a PersistentVolumeClaim object. + When this field is specified, volume binding will only succeed if the type of + the specified object matches some installed volume populator or dynamic + provisioner. + This field will replace the functionality of the dataSource field and as such + if both fields are non-empty, they must have the same value. For backwards + compatibility, when namespace isn't specified in dataSourceRef, + both fields (dataSource and dataSourceRef) will be set to the same + value automatically if one of them is empty and the other is non-empty. + When namespace is specified in dataSourceRef, + dataSource isn't set to the same value and must be empty. + There are three important differences between dataSource and dataSourceRef: + * While dataSource only allows two specific types of objects, dataSourceRef + allows any non-core object, as well as PersistentVolumeClaim objects. + * While dataSource ignores disallowed values (dropping them), dataSourceRef + preserves all values, and generates an error if a disallowed value is + specified. + * While dataSource only allows local objects, dataSourceRef allows objects + in any namespaces. + (Alpha) Using the namespace field of dataSourceRef requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + namespace: + description: |- + Namespace is the namespace of resource being referenced + Note that when a namespace is specified, a gateway.networking.k8s.io/ReferenceGrant object is required in the referent namespace to allow that namespace's owner to accept the reference. See the ReferenceGrant documentation for details. + (Alpha) This field requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + type: string + required: + - kind + - name + type: object + resources: + description: |- + resources represents the minimum resources the volume should have. + Users are allowed to specify resource requirements + that are lower than previous value but must still be higher than capacity recorded in the + status field of the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#resources + properties: + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + selector: + description: selector is a label query over volumes + to consider for binding. + properties: + matchExpressions: + description: matchExpressions is a list of label + selector requirements. The requirements are ANDed. + items: + description: |- + A label selector requirement is a selector that contains values, a key, and an operator that + relates the key and values. + properties: + key: + description: key is the label key that the + selector applies to. + type: string + operator: + description: |- + operator represents a key's relationship to a set of values. + Valid operators are In, NotIn, Exists and DoesNotExist. + type: string + values: + description: |- + values is an array of string values. If the operator is In or NotIn, + the values array must be non-empty. If the operator is Exists or DoesNotExist, + the values array must be empty. This array is replaced during a strategic + merge patch. + items: + type: string + type: array + x-kubernetes-list-type: atomic + required: + - key + - operator + type: object + type: array + x-kubernetes-list-type: atomic + matchLabels: + additionalProperties: + type: string + description: |- + matchLabels is a map of {key,value} pairs. A single {key,value} in the matchLabels + map is equivalent to an element of matchExpressions, whose key field is "key", the + operator is "In", and the values array contains only "value". The requirements are ANDed. + type: object + type: object + x-kubernetes-map-type: atomic + storageClassName: + description: |- + storageClassName is the name of the StorageClass required by the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#class-1 + type: string + volumeAttributesClassName: + description: |- + volumeAttributesClassName may be used to set the VolumeAttributesClass used by this claim. + If specified, the CSI driver will create or update the volume with the attributes defined + in the corresponding VolumeAttributesClass. This has a different purpose than storageClassName, + it can be changed after the claim is created. An empty string or nil value indicates that no + VolumeAttributesClass will be applied to the claim. If the claim enters an Infeasible error state, + this field can be reset to its previous value (including nil) to cancel the modification. + If the resource referred to by volumeAttributesClass does not exist, this PersistentVolumeClaim will be + set to a Pending state, as reflected by the modifyVolumeStatus field, until such as a resource + exists. + More info: https://kubernetes.io/docs/concepts/storage/volume-attributes-classes/ + type: string + volumeMode: + description: |- + volumeMode defines what type of volume is required by the claim. + Value of Filesystem is implied when not included in claim spec. + type: string + volumeName: + description: volumeName is the binding reference to + the PersistentVolume backing this claim. + type: string + type: object + required: + - mountPath + type: object metricsReporterImage: type: string networkConfig: @@ -11721,6 +11976,262 @@ spec: log4jConfig: description: Override for the default log4j configuration type: string + metadataStorage: + description: |- + MetadataStorage optionally isolates KRaft metadata on a dedicated PVC. + Only broker-only KRaft nodes may use it. It is not a data or Cruise Control disk. + Once enabled it cannot be removed or relocated; reverse migration is not supported. + properties: + emptyDir: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes#emptydir is used + as storage for Kafka broker log dirs. The use of empty dir as Kafka broker storage is useful in development + environments where data loss is not a concern as data stored on emptydir backed storage is lost at pod restarts. + Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + medium: + description: |- + medium represents what type of storage medium should back this directory. + The default is "" which means to use the node's default medium. + Must be an empty string (default) or Memory. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + type: string + mode: + description: |- + mode specifies the permission bits for the emptyDir directory, in numeric + notation (e.g., 0755, 01777). Must be a value between 0000 and 01777. + If not specified, defaults to 0777. + This might be in conflict with other options that affect the file + mode, like fsGroup. If fsGroup is specified, the fsGroup permissions + will override the mode specified here. + This field has no effect on Windows. + This field is alpha and requires EmptyDirVolumeMode featuregate to be enabled. + format: int32 + type: integer + sizeLimit: + anyOf: + - type: integer + - type: string + description: |- + sizeLimit is the total amount of local storage required for this EmptyDir volume. + The size limit is also applicable for memory medium. + The maximum usage on memory medium EmptyDir would be the minimum value between + the SizeLimit specified here and the sum of memory limits of all containers in a pod. + The default is nil which means that the limit is undefined. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + type: object + mountPath: + type: string + pvcSpec: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes/#persistentvolumeclaim is used + as storage for Kafka broker log dirs. Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + accessModes: + description: |- + accessModes contains the desired access modes the volume should have. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#access-modes-1 + items: + type: string + type: array + x-kubernetes-list-type: atomic + dataSource: + description: |- + dataSource field can be used to specify either: + * An existing VolumeSnapshot object (snapshot.storage.k8s.io/VolumeSnapshot) + * An existing PVC (PersistentVolumeClaim) + If the provisioner or an external controller can support the specified data source, + it will create a new volume based on the contents of the specified data source. + dataSource contents will be copied to dataSourceRef, and dataSourceRef contents will be + copied to dataSource when dataSourceRef.namespace is not specified. + If the namespace is specified, then dataSourceRef will not be copied to dataSource. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + required: + - kind + - name + type: object + x-kubernetes-map-type: atomic + dataSourceRef: + description: |- + dataSourceRef specifies the object from which to populate the volume with data, if a non-empty + volume is desired. This may be any object from a non-empty API group (non + core object) or a PersistentVolumeClaim object. + When this field is specified, volume binding will only succeed if the type of + the specified object matches some installed volume populator or dynamic + provisioner. + This field will replace the functionality of the dataSource field and as such + if both fields are non-empty, they must have the same value. For backwards + compatibility, when namespace isn't specified in dataSourceRef, + both fields (dataSource and dataSourceRef) will be set to the same + value automatically if one of them is empty and the other is non-empty. + When namespace is specified in dataSourceRef, + dataSource isn't set to the same value and must be empty. + There are three important differences between dataSource and dataSourceRef: + * While dataSource only allows two specific types of objects, dataSourceRef + allows any non-core object, as well as PersistentVolumeClaim objects. + * While dataSource ignores disallowed values (dropping them), dataSourceRef + preserves all values, and generates an error if a disallowed value is + specified. + * While dataSource only allows local objects, dataSourceRef allows objects + in any namespaces. + (Alpha) Using the namespace field of dataSourceRef requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + namespace: + description: |- + Namespace is the namespace of resource being referenced + Note that when a namespace is specified, a gateway.networking.k8s.io/ReferenceGrant object is required in the referent namespace to allow that namespace's owner to accept the reference. See the ReferenceGrant documentation for details. + (Alpha) This field requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + type: string + required: + - kind + - name + type: object + resources: + description: |- + resources represents the minimum resources the volume should have. + Users are allowed to specify resource requirements + that are lower than previous value but must still be higher than capacity recorded in the + status field of the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#resources + properties: + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + selector: + description: selector is a label query over volumes + to consider for binding. + properties: + matchExpressions: + description: matchExpressions is a list of label + selector requirements. The requirements are + ANDed. + items: + description: |- + A label selector requirement is a selector that contains values, a key, and an operator that + relates the key and values. + properties: + key: + description: key is the label key that + the selector applies to. + type: string + operator: + description: |- + operator represents a key's relationship to a set of values. + Valid operators are In, NotIn, Exists and DoesNotExist. + type: string + values: + description: |- + values is an array of string values. If the operator is In or NotIn, + the values array must be non-empty. If the operator is Exists or DoesNotExist, + the values array must be empty. This array is replaced during a strategic + merge patch. + items: + type: string + type: array + x-kubernetes-list-type: atomic + required: + - key + - operator + type: object + type: array + x-kubernetes-list-type: atomic + matchLabels: + additionalProperties: + type: string + description: |- + matchLabels is a map of {key,value} pairs. A single {key,value} in the matchLabels + map is equivalent to an element of matchExpressions, whose key field is "key", the + operator is "In", and the values array contains only "value". The requirements are ANDed. + type: object + type: object + x-kubernetes-map-type: atomic + storageClassName: + description: |- + storageClassName is the name of the StorageClass required by the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#class-1 + type: string + volumeAttributesClassName: + description: |- + volumeAttributesClassName may be used to set the VolumeAttributesClass used by this claim. + If specified, the CSI driver will create or update the volume with the attributes defined + in the corresponding VolumeAttributesClass. This has a different purpose than storageClassName, + it can be changed after the claim is created. An empty string or nil value indicates that no + VolumeAttributesClass will be applied to the claim. If the claim enters an Infeasible error state, + this field can be reset to its previous value (including nil) to cancel the modification. + If the resource referred to by volumeAttributesClass does not exist, this PersistentVolumeClaim will be + set to a Pending state, as reflected by the modifyVolumeStatus field, until such as a resource + exists. + More info: https://kubernetes.io/docs/concepts/storage/volume-attributes-classes/ + type: string + volumeMode: + description: |- + volumeMode defines what type of volume is required by the claim. + Value of Filesystem is implied when not included in claim spec. + type: string + volumeName: + description: volumeName is the binding reference + to the PersistentVolume backing this claim. + type: string + type: object + required: + - mountPath + type: object metricsReporterImage: type: string networkConfig: @@ -23818,6 +24329,11 @@ spec: description: Image specifies the current docker image of the broker type: string + metadataStorageState: + description: |- + MetadataStorageState is Ready once the broker runs from its dedicated metadata storage. + Data disks of a broker with metadataStorage can be removed only after that. + type: string perBrokerConfigurationState: description: PerBrokerConfigurationState holds info about the per-broker (dynamically updatable) config diff --git a/config/base/crds/kafka.banzaicloud.io_kafkaclusters.yaml b/config/base/crds/kafka.banzaicloud.io_kafkaclusters.yaml index ab749490c..4bce156f7 100644 --- a/config/base/crds/kafka.banzaicloud.io_kafkaclusters.yaml +++ b/config/base/crds/kafka.banzaicloud.io_kafkaclusters.yaml @@ -4440,6 +4440,261 @@ spec: log4jConfig: description: Override for the default log4j configuration type: string + metadataStorage: + description: |- + MetadataStorage optionally isolates KRaft metadata on a dedicated PVC. + Only broker-only KRaft nodes may use it. It is not a data or Cruise Control disk. + Once enabled it cannot be removed or relocated; reverse migration is not supported. + properties: + emptyDir: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes#emptydir is used + as storage for Kafka broker log dirs. The use of empty dir as Kafka broker storage is useful in development + environments where data loss is not a concern as data stored on emptydir backed storage is lost at pod restarts. + Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + medium: + description: |- + medium represents what type of storage medium should back this directory. + The default is "" which means to use the node's default medium. + Must be an empty string (default) or Memory. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + type: string + mode: + description: |- + mode specifies the permission bits for the emptyDir directory, in numeric + notation (e.g., 0755, 01777). Must be a value between 0000 and 01777. + If not specified, defaults to 0777. + This might be in conflict with other options that affect the file + mode, like fsGroup. If fsGroup is specified, the fsGroup permissions + will override the mode specified here. + This field has no effect on Windows. + This field is alpha and requires EmptyDirVolumeMode featuregate to be enabled. + format: int32 + type: integer + sizeLimit: + anyOf: + - type: integer + - type: string + description: |- + sizeLimit is the total amount of local storage required for this EmptyDir volume. + The size limit is also applicable for memory medium. + The maximum usage on memory medium EmptyDir would be the minimum value between + the SizeLimit specified here and the sum of memory limits of all containers in a pod. + The default is nil which means that the limit is undefined. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + type: object + mountPath: + type: string + pvcSpec: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes/#persistentvolumeclaim is used + as storage for Kafka broker log dirs. Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + accessModes: + description: |- + accessModes contains the desired access modes the volume should have. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#access-modes-1 + items: + type: string + type: array + x-kubernetes-list-type: atomic + dataSource: + description: |- + dataSource field can be used to specify either: + * An existing VolumeSnapshot object (snapshot.storage.k8s.io/VolumeSnapshot) + * An existing PVC (PersistentVolumeClaim) + If the provisioner or an external controller can support the specified data source, + it will create a new volume based on the contents of the specified data source. + dataSource contents will be copied to dataSourceRef, and dataSourceRef contents will be + copied to dataSource when dataSourceRef.namespace is not specified. + If the namespace is specified, then dataSourceRef will not be copied to dataSource. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + required: + - kind + - name + type: object + x-kubernetes-map-type: atomic + dataSourceRef: + description: |- + dataSourceRef specifies the object from which to populate the volume with data, if a non-empty + volume is desired. This may be any object from a non-empty API group (non + core object) or a PersistentVolumeClaim object. + When this field is specified, volume binding will only succeed if the type of + the specified object matches some installed volume populator or dynamic + provisioner. + This field will replace the functionality of the dataSource field and as such + if both fields are non-empty, they must have the same value. For backwards + compatibility, when namespace isn't specified in dataSourceRef, + both fields (dataSource and dataSourceRef) will be set to the same + value automatically if one of them is empty and the other is non-empty. + When namespace is specified in dataSourceRef, + dataSource isn't set to the same value and must be empty. + There are three important differences between dataSource and dataSourceRef: + * While dataSource only allows two specific types of objects, dataSourceRef + allows any non-core object, as well as PersistentVolumeClaim objects. + * While dataSource ignores disallowed values (dropping them), dataSourceRef + preserves all values, and generates an error if a disallowed value is + specified. + * While dataSource only allows local objects, dataSourceRef allows objects + in any namespaces. + (Alpha) Using the namespace field of dataSourceRef requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + namespace: + description: |- + Namespace is the namespace of resource being referenced + Note that when a namespace is specified, a gateway.networking.k8s.io/ReferenceGrant object is required in the referent namespace to allow that namespace's owner to accept the reference. See the ReferenceGrant documentation for details. + (Alpha) This field requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + type: string + required: + - kind + - name + type: object + resources: + description: |- + resources represents the minimum resources the volume should have. + Users are allowed to specify resource requirements + that are lower than previous value but must still be higher than capacity recorded in the + status field of the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#resources + properties: + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + selector: + description: selector is a label query over volumes + to consider for binding. + properties: + matchExpressions: + description: matchExpressions is a list of label + selector requirements. The requirements are ANDed. + items: + description: |- + A label selector requirement is a selector that contains values, a key, and an operator that + relates the key and values. + properties: + key: + description: key is the label key that the + selector applies to. + type: string + operator: + description: |- + operator represents a key's relationship to a set of values. + Valid operators are In, NotIn, Exists and DoesNotExist. + type: string + values: + description: |- + values is an array of string values. If the operator is In or NotIn, + the values array must be non-empty. If the operator is Exists or DoesNotExist, + the values array must be empty. This array is replaced during a strategic + merge patch. + items: + type: string + type: array + x-kubernetes-list-type: atomic + required: + - key + - operator + type: object + type: array + x-kubernetes-list-type: atomic + matchLabels: + additionalProperties: + type: string + description: |- + matchLabels is a map of {key,value} pairs. A single {key,value} in the matchLabels + map is equivalent to an element of matchExpressions, whose key field is "key", the + operator is "In", and the values array contains only "value". The requirements are ANDed. + type: object + type: object + x-kubernetes-map-type: atomic + storageClassName: + description: |- + storageClassName is the name of the StorageClass required by the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#class-1 + type: string + volumeAttributesClassName: + description: |- + volumeAttributesClassName may be used to set the VolumeAttributesClass used by this claim. + If specified, the CSI driver will create or update the volume with the attributes defined + in the corresponding VolumeAttributesClass. This has a different purpose than storageClassName, + it can be changed after the claim is created. An empty string or nil value indicates that no + VolumeAttributesClass will be applied to the claim. If the claim enters an Infeasible error state, + this field can be reset to its previous value (including nil) to cancel the modification. + If the resource referred to by volumeAttributesClass does not exist, this PersistentVolumeClaim will be + set to a Pending state, as reflected by the modifyVolumeStatus field, until such as a resource + exists. + More info: https://kubernetes.io/docs/concepts/storage/volume-attributes-classes/ + type: string + volumeMode: + description: |- + volumeMode defines what type of volume is required by the claim. + Value of Filesystem is implied when not included in claim spec. + type: string + volumeName: + description: volumeName is the binding reference to + the PersistentVolume backing this claim. + type: string + type: object + required: + - mountPath + type: object metricsReporterImage: type: string networkConfig: @@ -11721,6 +11976,262 @@ spec: log4jConfig: description: Override for the default log4j configuration type: string + metadataStorage: + description: |- + MetadataStorage optionally isolates KRaft metadata on a dedicated PVC. + Only broker-only KRaft nodes may use it. It is not a data or Cruise Control disk. + Once enabled it cannot be removed or relocated; reverse migration is not supported. + properties: + emptyDir: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes#emptydir is used + as storage for Kafka broker log dirs. The use of empty dir as Kafka broker storage is useful in development + environments where data loss is not a concern as data stored on emptydir backed storage is lost at pod restarts. + Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + medium: + description: |- + medium represents what type of storage medium should back this directory. + The default is "" which means to use the node's default medium. + Must be an empty string (default) or Memory. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + type: string + mode: + description: |- + mode specifies the permission bits for the emptyDir directory, in numeric + notation (e.g., 0755, 01777). Must be a value between 0000 and 01777. + If not specified, defaults to 0777. + This might be in conflict with other options that affect the file + mode, like fsGroup. If fsGroup is specified, the fsGroup permissions + will override the mode specified here. + This field has no effect on Windows. + This field is alpha and requires EmptyDirVolumeMode featuregate to be enabled. + format: int32 + type: integer + sizeLimit: + anyOf: + - type: integer + - type: string + description: |- + sizeLimit is the total amount of local storage required for this EmptyDir volume. + The size limit is also applicable for memory medium. + The maximum usage on memory medium EmptyDir would be the minimum value between + the SizeLimit specified here and the sum of memory limits of all containers in a pod. + The default is nil which means that the limit is undefined. + More info: https://kubernetes.io/docs/concepts/storage/volumes#emptydir + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + type: object + mountPath: + type: string + pvcSpec: + description: |- + If set https://kubernetes.io/docs/concepts/storage/volumes/#persistentvolumeclaim is used + as storage for Kafka broker log dirs. Either `pvcSpec` or `emptyDir` has to be set. + When both `pvcSpec` and `emptyDir` fields are set + the `pvcSpec` is used by default. + properties: + accessModes: + description: |- + accessModes contains the desired access modes the volume should have. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#access-modes-1 + items: + type: string + type: array + x-kubernetes-list-type: atomic + dataSource: + description: |- + dataSource field can be used to specify either: + * An existing VolumeSnapshot object (snapshot.storage.k8s.io/VolumeSnapshot) + * An existing PVC (PersistentVolumeClaim) + If the provisioner or an external controller can support the specified data source, + it will create a new volume based on the contents of the specified data source. + dataSource contents will be copied to dataSourceRef, and dataSourceRef contents will be + copied to dataSource when dataSourceRef.namespace is not specified. + If the namespace is specified, then dataSourceRef will not be copied to dataSource. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + required: + - kind + - name + type: object + x-kubernetes-map-type: atomic + dataSourceRef: + description: |- + dataSourceRef specifies the object from which to populate the volume with data, if a non-empty + volume is desired. This may be any object from a non-empty API group (non + core object) or a PersistentVolumeClaim object. + When this field is specified, volume binding will only succeed if the type of + the specified object matches some installed volume populator or dynamic + provisioner. + This field will replace the functionality of the dataSource field and as such + if both fields are non-empty, they must have the same value. For backwards + compatibility, when namespace isn't specified in dataSourceRef, + both fields (dataSource and dataSourceRef) will be set to the same + value automatically if one of them is empty and the other is non-empty. + When namespace is specified in dataSourceRef, + dataSource isn't set to the same value and must be empty. + There are three important differences between dataSource and dataSourceRef: + * While dataSource only allows two specific types of objects, dataSourceRef + allows any non-core object, as well as PersistentVolumeClaim objects. + * While dataSource ignores disallowed values (dropping them), dataSourceRef + preserves all values, and generates an error if a disallowed value is + specified. + * While dataSource only allows local objects, dataSourceRef allows objects + in any namespaces. + (Alpha) Using the namespace field of dataSourceRef requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being + referenced + type: string + name: + description: Name is the name of resource being + referenced + type: string + namespace: + description: |- + Namespace is the namespace of resource being referenced + Note that when a namespace is specified, a gateway.networking.k8s.io/ReferenceGrant object is required in the referent namespace to allow that namespace's owner to accept the reference. See the ReferenceGrant documentation for details. + (Alpha) This field requires the CrossNamespaceVolumeDataSource feature gate to be enabled. + type: string + required: + - kind + - name + type: object + resources: + description: |- + resources represents the minimum resources the volume should have. + Users are allowed to specify resource requirements + that are lower than previous value but must still be higher than capacity recorded in the + status field of the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#resources + properties: + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + selector: + description: selector is a label query over volumes + to consider for binding. + properties: + matchExpressions: + description: matchExpressions is a list of label + selector requirements. The requirements are + ANDed. + items: + description: |- + A label selector requirement is a selector that contains values, a key, and an operator that + relates the key and values. + properties: + key: + description: key is the label key that + the selector applies to. + type: string + operator: + description: |- + operator represents a key's relationship to a set of values. + Valid operators are In, NotIn, Exists and DoesNotExist. + type: string + values: + description: |- + values is an array of string values. If the operator is In or NotIn, + the values array must be non-empty. If the operator is Exists or DoesNotExist, + the values array must be empty. This array is replaced during a strategic + merge patch. + items: + type: string + type: array + x-kubernetes-list-type: atomic + required: + - key + - operator + type: object + type: array + x-kubernetes-list-type: atomic + matchLabels: + additionalProperties: + type: string + description: |- + matchLabels is a map of {key,value} pairs. A single {key,value} in the matchLabels + map is equivalent to an element of matchExpressions, whose key field is "key", the + operator is "In", and the values array contains only "value". The requirements are ANDed. + type: object + type: object + x-kubernetes-map-type: atomic + storageClassName: + description: |- + storageClassName is the name of the StorageClass required by the claim. + More info: https://kubernetes.io/docs/concepts/storage/persistent-volumes#class-1 + type: string + volumeAttributesClassName: + description: |- + volumeAttributesClassName may be used to set the VolumeAttributesClass used by this claim. + If specified, the CSI driver will create or update the volume with the attributes defined + in the corresponding VolumeAttributesClass. This has a different purpose than storageClassName, + it can be changed after the claim is created. An empty string or nil value indicates that no + VolumeAttributesClass will be applied to the claim. If the claim enters an Infeasible error state, + this field can be reset to its previous value (including nil) to cancel the modification. + If the resource referred to by volumeAttributesClass does not exist, this PersistentVolumeClaim will be + set to a Pending state, as reflected by the modifyVolumeStatus field, until such as a resource + exists. + More info: https://kubernetes.io/docs/concepts/storage/volume-attributes-classes/ + type: string + volumeMode: + description: |- + volumeMode defines what type of volume is required by the claim. + Value of Filesystem is implied when not included in claim spec. + type: string + volumeName: + description: volumeName is the binding reference + to the PersistentVolume backing this claim. + type: string + type: object + required: + - mountPath + type: object metricsReporterImage: type: string networkConfig: @@ -23818,6 +24329,11 @@ spec: description: Image specifies the current docker image of the broker type: string + metadataStorageState: + description: |- + MetadataStorageState is Ready once the broker runs from its dedicated metadata storage. + Data disks of a broker with metadataStorage can be removed only after that. + type: string perBrokerConfigurationState: description: PerBrokerConfigurationState holds info about the per-broker (dynamically updatable) config diff --git a/config/samples/kraft/simplekafkacluster_kraft.yaml b/config/samples/kraft/simplekafkacluster_kraft.yaml index 35818a4b3..e7543b3a0 100644 --- a/config/samples/kraft/simplekafkacluster_kraft.yaml +++ b/config/samples/kraft/simplekafkacluster_kraft.yaml @@ -33,6 +33,15 @@ spec: requests: storage: 10Gi broker: + # Optional for NEW clusters; for live migration enable on ONE broker at a time. + # See docs/broker-metadata-storage.md. Never add this to the controller group. + # metadataStorage: + # mountPath: /csi-kafka-metadata + # pvcSpec: + # accessModes: [ReadWriteOnce] + # resources: + # requests: + # storage: 20Gi kafkaJvmPerfOpts: >- -Dcom.sun.management.jmxremote.port=1090 -Dcom.sun.management.jmxremote.rmi.port=1090 -Dcom.sun.management.jmxremote.local.only=false -Djava.rmi.server.hostname=127.0.0.1 -Djute.maxbuffer=0x9fffff @@ -73,6 +82,7 @@ spec: brokerConfigGroup: "broker" rollingUpgradeConfig: failureThreshold: 1 + concurrentBrokerRestartCountPerRack: 1 listenersConfig: internalListeners: - type: "plaintext" diff --git a/docs/broker-metadata-storage.md b/docs/broker-metadata-storage.md new file mode 100644 index 000000000..6e896be07 --- /dev/null +++ b/docs/broker-metadata-storage.md @@ -0,0 +1,256 @@ +# Dedicated KRaft metadata storage for broker-only nodes + +`brokerConfig.metadataStorage` is **opt-in**. Omit it to preserve the existing +storage and startup behavior. It is supported only for KRaft nodes with exactly +`processRoles: [broker]`, not controllers, combined nodes, ZooKeeper nodes, or +brokers still using ZooKeeper during a ZK-to-KRaft migration. Brokers whose +ZK-to-KRaft migration is finalized are supported: their data directories keep +version 0 `meta.properties` (`broker.id`) and usually have no +`bootstrap.checkpoint`, which the migration accepts. + +The setting uses the existing `StorageConfig` shape, but requires `pvcSpec` and +disallows `emptyDir` and block-mode PVCs. Every data `storageConfigs` entry of +the broker must be PVC-backed as well: an `emptyDir` data directory is lost on +pod replacement, so it cannot be a migration source, and durable metadata gains +nothing beside ephemeral replicas. It creates a separate `--metadata` PVC, +mounts it at `mountPath`, and generates +`metadata.log.dir=/kafka`. Because the broker ConfigMap is live-mounted, +the operator adds `metadata.log.dir` only once no broker pod without the metadata +volume remains, and never creates a metadata-storage pod whose ConfigMap lacks it; +once published the setting is kept. It **does not** add that path to `log.dirs`, +Cruise Control capacity/add/remove/rebalance operations, or topic storage +assignments. `storageConfigs` remain topic-data disks. + +Group inheritance works like other broker configuration, with one distinction: +a broker-local `metadataStorage` replaces the entire group metadata storage +object; it does not merge partial PVC specs. Omitting a local value inherits the +group value. Do not put this setting in a group used by controllers. + +## Opt-in configuration + +Enable it once in the broker-only config group; keep broker IDs, roles and all +data storage unchanged. The operator migrates the brokers itself, one at a time +(see [Automatic rollout](#automatic-rollout)): + +```yaml +spec: + kRaft: true + brokerConfigGroups: + # Keep every existing field of the group used only by broker-only nodes. + broker: + processRoles: + - broker + metadataStorage: + mountPath: /csi-kafka-metadata + pvcSpec: + accessModes: + - ReadWriteOnce + storageClassName: YOUR_EXISTING_CSI_STORAGE_CLASS + resources: + requests: + storage: 20Gi +``` + +To pilot on a single broker first, put the same `metadataStorage` object in that +broker's `brokerConfig` instead, then move it to the group once satisfied. + +The storage class, size and availability-zone topology are examples, not sizing +recommendations. The metadata mount must not equal, contain, or be contained by +any data/custom/system mount. It must be a clean absolute path. For **new** +clusters, the same object can be placed directly in the broker-only config +group. See the optional example in +[`simplekafkacluster_kraft.yaml`](../config/samples/kraft/simplekafkacluster_kraft.yaml). +The operator generates this property's value, overriding a previous +`metadata.log.dir` pin when dedicated storage is enabled. KafkaCluster-producing +Helm charts must explicitly emit `metadataStorage`; adding it only to chart +values or to `additionalVolumes` does not enable this API feature. + +Once enabled, admission and reconciliation reject removal, mount relocation +and PVC-spec replacement. Storage requests may grow when the storage class +supports expansion, but may not shrink. Reverse migration is **not supported**. +Deleting a PVC, changing its purpose/identity annotations, downgrading the +operator, or bypassing admission is not a rollback procedure. + +## Preconditions for a live migration + +These are checked once, before enabling the flag; the operator automates the +per-broker steps. + +1. Install the updated operator, admission webhook and both the installed CRD + and chart CRD before opting in. Verify that admission is enabled and working. + Helm does not automatically upgrade existing CRDs. +2. Use the broker's existing Kafka **3.9 or newer** image with the static-quorum + formatter used by the existing operator. The image must provide Bash, Kafka's + `kafka-storage.sh`, `cp -a`, `mv`, `cmp`, and `sync`. Formatting must produce a + valid unique `directory.id`. Rehearse against that exact image and CSI driver + in a non-production cluster before a production rollout. Do not combine this + rollout with a Kafka/metadata-version upgrade. +3. Confirm a healthy controller quorum and healthy broker registration. This + feature does not migrate, wipe, format, or alter **any controller** metadata, + directory, PVC, or recovery mechanism. +4. For RF3 across three AZs, verify that **every affected partition**, including + internal topics, has three healthy in-sync replicas across the expected AZs. + Verify `min.insync.replicas` and producer acknowledgements; RF3 alone is not + enough. Two remaining replicas must have headroom for the unavailable + broker's leaders/traffic. A topic with `min.insync.replicas=3` cannot maintain + acknowledged writes while one RF3 replica is offline. +5. Pause unrelated upgrades, data-disk drains, reassignments, scaling and + disruptive node maintenance. Preserve the source data disk until migration, + restart and full ISR recovery are verified. Admission rejects changing data + `storageConfigs` in the same update that enables metadata storage. +6. Verify that all source data PVCs are mounted and readable, and that exactly + one contains `/kafka/__cluster_metadata-0`. The source may be + `/csi-kafka-logs1/kafka`, `/csi-kafka-logs2/kafka`, or another mounted data + directory. Validate its `meta.properties` cluster ID and node ID (`node.id` + for version 1, `broker.id` for version 0 or unversioned files). The source + `bootstrap.checkpoint` is copied when present; brokers migrated from + ZooKeeper normally have none, and the destination then has none either. Take CSI snapshots/backups using your existing + application-consistent procedure. Do not copy or move metadata while Kafka + is running. +7. Size the metadata PVC for the current complete metadata log, snapshots, + bootstrap checkpoint and future growth. Initial staging uses one payload + copy on that PVC; the source backup is a rename on the old data volume. + Ensure the new PVC can bind/schedule in the broker's AZ and is writable using + the broker's existing security context. Check the configured termination + grace period accommodates shutdown. + The migration init container uses the broker's resource requirements (not the + tiny reporter-copy init limit), with a 256 MiB maximum Java heap for formatting; + allow JVM/native overhead as well. + +## Automatic rollout + +Enabling `metadataStorage` is the only required change. The operator: + +1. Creates every opted-in broker's metadata PVC up front. +2. Replaces at most **one** broker pod per cluster for migration at a time, using + graceful deletion and the normal rolling-upgrade gates. Before each + migration restart it additionally requires that no other opted-in broker's + migration is in progress (old pod terminating/missing, or replacement not + yet `Ready`) and that no **other** broker has offline or out-of-sync replicas. + This applies regardless of `concurrentBrokerRestartCountPerRack`, and also to + crashed pods that would otherwise bypass the rolling-upgrade gates. +3. Runs the `migrate-broker-metadata` init container in the replacement pod, + which copies the stopped broker's metadata onto the new PVC before Kafka + starts. A failure keeps the pod in its init container and halts the rollout; + nothing else is migrated until it is resolved. +4. Records `status.brokersState[""].metadataStorageState: Ready` once the + replacement pod is Ready, then waits for in-sync replicas to recover before + migrating the next broker. + +Monitor progress: + +```sh +kubectl -n "$NAMESPACE" get kafkacluster "$CLUSTER" \ + -o jsonpath='{range .status.brokersState.*}{.metadataStorageState}{"\n"}{end}' +kubectl -n "$NAMESPACE" get pods,pvc -l "kafka_cr=$CLUSTER" +kubectl -n "$NAMESPACE" logs "$POD" -c migrate-broker-metadata +``` + +The init container logs each phase (validation, format, copy, publish, source +retirement, completion), so a slow copy is distinguishable from a hung one. +Success includes `KRaft metadata storage ready for broker `. A failure +blocks Kafka startup; diagnose it without deleting source metadata, +destination ownership markers or backups. + +Readiness is recorded durably as an annotation on the metadata PVC and projected +to the status. Admission rejects removing any data disk of a broker with +`metadataStorage` until that state is `Ready`, so a data-removal update cannot +race the migration. If admission was bypassed, reconciliation keeps every data +PVC and stops with an error naming the disk to restore. After every broker is +`Ready`, a later, separate change may remove old **data** disks through the +existing Cruise Control drain workflow. Back up retained metadata before +deleting a source PVC. + +The gates check broker registration and replica health, not client error rates +or capacity headroom; keep watching those during the rollout. + +### What the rollout gates do (and do not do) + +The normal rolling-upgrade path waits on missing/terminating/pending pods and +rejects further restarts when it observes offline/out-of-sync replicas beyond +`failureThreshold`. Migration restarts additionally pass the dedicated +one-per-cluster migration gate described above, including on the +terminated-container repair path. The unschedulable-pod repair path (a pod +pinned to a removed PVC) is not gated, because waiting cannot fix it. +Broker-only pods do not gain a new metadata-readiness probe in this change. + +No setting guarantees full ISR, spare capacity, client availability, or zero +transient client errors under all failure scenarios. Leader election/retries +can cause transient errors. Broker-only restarts do not use the controller +quorum-readiness gate, so check controller quorum health before enabling the flag. + +**Fencing limitation:** ordinary graceful Kubernetes pod deletion waits for +container shutdown, which is the offline-copy precondition. RWO is **not** +process fencing. Do not force-delete the old pod, reduce its grace period to +zero, manually create a replacement, or perform migration while a node is +partitioned/unreachable and its old Kafka process may still run. The operator +does not provide external STONITH/storage fencing. Resolve/fence that node and +prove the old process is stopped before continuing. + +## Durable phases and failure handling + +The init container validates broker-only role, generated config and all mounted +data-directory identities. A fresh broker is allowed only when the metadata PVC +was created without existing pods, data PVCs, or a known running broker version, and no +formatted data directory or source metadata is present. An existing broker with +missing or multiple sources fails closed rather than rebuilding an empty cache. + +On the metadata PVC: + +* `.koperator-metadata-migration` records immutable node ID, cluster ID and source + mount (or `fresh`) before mutation. +* `.koperator-metadata-stage` is formatted by Kafka with a **new** directory ID. + Only the complete stopped-source `__cluster_metadata-0` and, when present, + `bootstrap.checkpoint` are copied. Without a source bootstrap, the + formatter's bootstrap is removed so the destination mirrors the source. Source `meta.properties` and topic replicas + are never copied/reformatted. Partial copies can be retried from the intact + stopped source. +* The owned staging directory is synced and atomically renamed to `kafka`. + `.koperator-metadata-owner` identifies that publication. Unknown destination/ + staging content or mismatched identities causes an error. +* The source `__cluster_metadata-0` is atomically **moved** to + `/.koperator-metadata-backup-`, outside `log.dirs`. + Source topic directories, `meta.properties`, and bootstrap remain untouched. +* `.koperator-metadata-complete` is written after backup. An interruption after + publishing or backup resumes using the owned destination and retained backup. + A completed restart uses the destination and never resets its quorum cache. + Once complete, removing the old source disk is supported. + +Do not manually erase markers to bypass an error. Correct storage permissions, +identity/config conflicts, capacity or image/tool issues, then retry through +normal operator replacement. If source/destination evidence disagrees, stop the +rollout and use your incident/recovery process. Never restore the old active +metadata into a live data `kafka` root alongside the new destination. + +The original source backup is **not** a current replica after Kafka starts on +the destination. Keep it and the original data `meta.properties`/bootstrap for +your agreed recovery-retention window; preserve a CSI snapshot before the data +PVC is drained/deleted. Nothing automatically prunes backups. There is no safe +automatic reverse migration, operator downgrade, or failback to this stale +backup. Recovery after destination loss requires a separately reviewed, offline +broker recovery procedure. Controller disaster recovery remains unchanged. + +## Broker removal + +Removing a broker from `spec.brokers` uses the existing graceful downscale. When +the broker pod is deleted, the operator deletes its data PVCs **and** its +metadata PVC, so the broker ID can later be reused with or without metadata +storage. Back up the metadata PVC first if your recovery policy requires it. + +## Kafka 3.9.2 migration smoke test + +The shell-fixture unit tests exercise failure/retry phases. A separate smoke +test starts a real, isolated controller and broker, writes a topic record, +migrates from each of `logs1` and `logs2`, and verifies the record and metadata +quorum after two broker restarts: + +```sh +docker run --rm --network none --entrypoint /bin/bash \ + -v "$PWD/scripts/test-broker-metadata-migration.sh:/test/smoke.sh:ro" \ + -v "$PWD/pkg/resources/kafka/migrate-broker-metadata.sh:/test/migrate-broker-metadata.sh:ro" \ + apache/kafka:3.9.2 /test/smoke.sh +``` + +This checks real Kafka formatting and metadata recovery, not RF3/AZ disruption +gates, Kubernetes pod fencing, or production CSI behavior. Those still require +the staged deployment rehearsal and recovery checks described above. diff --git a/pkg/k8sutil/status.go b/pkg/k8sutil/status.go index de8b36194..710894440 100644 --- a/pkg/k8sutil/status.go +++ b/pkg/k8sutil/status.go @@ -156,6 +156,8 @@ func generateBrokerState(brokerIDs []string, cluster *banzaicloudv1beta1.KafkaCl brokerState.ConfigurationState = s case banzaicloudv1beta1.PerBrokerConfigurationState: brokerState.PerBrokerConfigurationState = s + case banzaicloudv1beta1.MetadataStorageState: + brokerState.MetadataStorageState = s case map[string]banzaicloudv1beta1.VolumeState: if brokerState.GracefulActionState.VolumeStates == nil { brokerState.GracefulActionState.VolumeStates = make(map[string]banzaicloudv1beta1.VolumeState) diff --git a/pkg/resources/cruisecontrol/metadata_storage_test.go b/pkg/resources/cruisecontrol/metadata_storage_test.go new file mode 100644 index 000000000..033dd90bf --- /dev/null +++ b/pkg/resources/cruisecontrol/metadata_storage_test.go @@ -0,0 +1,44 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package cruisecontrol + +import ( + "testing" + + "github.com/go-logr/logr" + "github.com/stretchr/testify/require" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + + "github.com/banzaicloud/koperator/api/v1beta1" +) + +func TestMetadataStorageExcludedFromCapacity(t *testing.T) { + pvc := &corev1.PersistentVolumeClaimSpec{Resources: corev1.VolumeResourceRequirements{ + Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse("100Gi")}, + }} + config := v1beta1.BrokerConfig{ + Roles: []string{"broker"}, + StorageConfigs: []v1beta1.StorageConfig{{MountPath: "/csi-kafka-logs1", PvcSpec: pvc}}, + MetadataStorage: &v1beta1.StorageConfig{MountPath: "/csi-kafka-metadata", PvcSpec: pvc.DeepCopy()}, + } + spec := v1beta1.KafkaClusterSpec{BrokerConfigGroups: map[string]v1beta1.BrokerConfig{"brokers": config}} + broker := v1beta1.Broker{Id: 1, BrokerConfigGroup: "brokers"} + disks, err := generateBrokerDisks(broker, spec, nil, nil, logr.Discard()) + require.NoError(t, err) + require.Len(t, disks, 1) + require.Contains(t, disks, "/csi-kafka-logs1/kafka") + require.NotContains(t, disks, "/csi-kafka-metadata/kafka") +} diff --git a/pkg/resources/kafka/configmap.go b/pkg/resources/kafka/configmap.go index 79cd62c6d..f266e802d 100644 --- a/pkg/resources/kafka/configmap.go +++ b/pkg/resources/kafka/configmap.go @@ -94,6 +94,11 @@ func (r *Reconciler) getConfigProperties(bConfig *v1beta1.BrokerConfig, broker v if err := config.Set(kafkautils.KafkaConfigBrokerLogDirectory, strings.Join(mountPathsMerged, ",")); err != nil { log.Error(err, fmt.Sprintf(kafkautils.BrokerConfigErrorMsgTemplate, kafkautils.KafkaConfigBrokerLogDirectory)) } + if bConfig.MetadataStorage != nil && r.metadataLogDirConfigurable(context.Background(), broker.Id, &brokerConfigMapOld, log) { + if err := config.Set(kafkautils.KafkaConfigMetadataLogDirectory, util.StorageConfigKafkaMountPath(bConfig.MetadataStorage.MountPath)); err != nil { + log.Error(err, fmt.Sprintf(kafkautils.BrokerConfigErrorMsgTemplate, kafkautils.KafkaConfigMetadataLogDirectory)) + } + } } // Add superuser configuration diff --git a/pkg/resources/kafka/kafka.go b/pkg/resources/kafka/kafka.go index 2b55bc38a..f467002a6 100644 --- a/pkg/resources/kafka/kafka.go +++ b/pkg/resources/kafka/kafka.go @@ -106,6 +106,9 @@ type Reconciler struct { resources.Reconciler kafkaClientProvider kafkaclient.Provider CruiseControlScalerFactory func(ctx context.Context, kafkaCluster *banzaiv1beta1.KafkaCluster) (scale.CruiseControlScaler, error) + // metadataMigrationStarted serializes metadata migrations within one reconcile pass, + // before the pod deletion is visible to listers. + metadataMigrationStarted bool } // New creates a new reconciler for Kafka @@ -160,6 +163,7 @@ func getCreatedPvcForBroker( return nil, err } + foundPvcList.Items = dataPVCs(foundPvcList.Items) var missing []string for i := range storageConfigs { if storageConfigs[i].PvcSpec == nil { @@ -331,6 +335,9 @@ func (r *Reconciler) Reconcile(log logr.Logger) error { } var brokerVolumes []*corev1.PersistentVolumeClaim + if err := r.reconcileMetadataStorage(ctx, broker, brokerConfig, log); err != nil { + return errors.WrapIfWithDetails(err, "failed to reconcile metadata storage", "brokerId", broker.Id) + } for index, storage := range brokerConfig.StorageConfigs { if storage.PvcSpec == nil && storage.EmptyDir == nil { return errors.WrapIfWithDetails(err, @@ -469,6 +476,16 @@ func (r *Reconciler) Reconcile(log logr.Logger) error { if err != nil { return errors.WrapIfWithDetails(err, "failed to list PVC's") } + if brokerConfig.MetadataStorage != nil { + metadataPvc, err := r.metadataPVC(ctx, broker.Id) + if err != nil { + return err + } + if metadataPvc == nil { + return errors.New("metadata PVC is not created") + } + pvcs = append(pvcs, *metadataPvc) + } if !r.KafkaCluster.Spec.HeadlessServiceEnabled { o := r.service(broker.Id, brokerConfig) @@ -695,21 +712,8 @@ func (r *Reconciler) reconcileKafkaPodDelete(ctx context.Context, log logr.Logge } log.V(1).Info("service for broker deleted", "service name", serviceName, banzaiv1beta1.BrokerIdLabelKey, broker.Labels[banzaiv1beta1.BrokerIdLabelKey]) } - for _, volume := range broker.Spec.Volumes { - if strings.HasPrefix(volume.Name, kafkaDataVolumeMount) { - err = r.Delete(context.TODO(), &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{ - Name: volume.PersistentVolumeClaim.ClaimName, - Namespace: r.KafkaCluster.Namespace, - }}) - if err != nil { - if apierrors.IsNotFound(err) { - // can happen when broker was not fully initialized and now is deleted - log.Info(fmt.Sprintf("PVC for Broker %s not found. Continue", broker.Labels[banzaiv1beta1.BrokerIdLabelKey])) - } - return errors.WrapIfWithDetails(err, "could not delete pvc for broker", "id", broker.Labels[banzaiv1beta1.BrokerIdLabelKey]) - } - log.V(1).Info("pvc for broker deleted", "pvc name", volume.PersistentVolumeClaim.ClaimName, banzaiv1beta1.BrokerIdLabelKey, broker.Labels[banzaiv1beta1.BrokerIdLabelKey]) - } + if err = r.deleteBrokerPVCs(ctx, broker, log); err != nil { + return err } err = k8sutil.DeleteBrokerStatus(r.Client, broker.Labels[banzaiv1beta1.BrokerIdLabelKey], r.KafkaCluster, log) if err != nil { @@ -720,6 +724,31 @@ func (r *Reconciler) reconcileKafkaPodDelete(ctx context.Context, log logr.Logge return nil } +// deleteBrokerPVCs deletes the data PVCs mounted by a removed broker pod and its metadata PVC. +func (r *Reconciler) deleteBrokerPVCs(ctx context.Context, broker corev1.Pod, log logr.Logger) error { + for _, volume := range broker.Spec.Volumes { + if strings.HasPrefix(volume.Name, kafkaDataVolumeMount) { + err := r.Delete(ctx, &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{ + Name: volume.PersistentVolumeClaim.ClaimName, + Namespace: r.KafkaCluster.Namespace, + }}) + if err != nil { + if apierrors.IsNotFound(err) { + // can happen when broker was not fully initialized and now is deleted + log.Info(fmt.Sprintf("PVC for Broker %s not found. Continue", broker.Labels[banzaiv1beta1.BrokerIdLabelKey])) + continue + } + return errors.WrapIfWithDetails(err, "could not delete pvc for broker", "id", broker.Labels[banzaiv1beta1.BrokerIdLabelKey]) + } + log.V(1).Info("pvc for broker deleted", "pvc name", volume.PersistentVolumeClaim.ClaimName, banzaiv1beta1.BrokerIdLabelKey, broker.Labels[banzaiv1beta1.BrokerIdLabelKey]) + } + } + if err := r.deleteMetadataPVC(ctx, broker.Labels[banzaiv1beta1.BrokerIdLabelKey], log); err != nil { + return errors.WrapIfWithDetails(err, "could not delete metadata pvc for broker", "id", broker.Labels[banzaiv1beta1.BrokerIdLabelKey]) + } + return nil +} + func arePodsAlreadyDeleted(pods []corev1.Pod, log logr.Logger) bool { for _, broker := range pods { if broker.DeletionTimestamp == nil { @@ -885,6 +914,9 @@ func (r *Reconciler) reconcileKafkaPod(log logr.Logger, desiredPod *corev1.Pod, } switch { case len(podList.Items) == 0: + if err := r.ensureMetadataLogDirConfigured(context.TODO(), desiredPod.Labels[banzaiv1beta1.BrokerIdLabelKey], bConfig); err != nil { + return err + } if err := patch.DefaultAnnotator.SetLastAppliedAnnotation(desiredPod); err != nil { return errors.WrapIf(err, "could not apply last state to annotation") } @@ -1059,7 +1091,15 @@ func (r *Reconciler) handleRollingUpgrade(log logr.Logger, desiredPod, currentPo return errors.WrapIf(err, "could not apply last state to annotation") } - if podUnschedulableReferencingRemovedPVC(currentPod, desiredPod) { + unschedulable := podUnschedulableReferencingRemovedPVC(currentPod, desiredPod) + migratesMetadata := isMetadataMigrationRestart(currentPod, desiredPod) + if migratesMetadata && !unschedulable { + if err := r.metadataMigrationGate(context.TODO(), currentPod); err != nil { + return err + } + } + + if unschedulable { // Self-heal: a broker pod stuck Pending because it references a PVC for a disk that was // removed while the pod was being (re)created can never schedule ("persistentvolumeclaim ... // not found"), and a never-started broker keeps the rolling-upgrade health check below @@ -1166,6 +1206,10 @@ func (r *Reconciler) handleRollingUpgrade(log logr.Logger, desiredPod, currentPo if err != nil { return errorfactory.New(errorfactory.APIFailure{}, err, "deleting resource failed", "kind", desiredType) } + if migratesMetadata { + r.metadataMigrationStarted = true + log.Info("broker pod deleted to migrate metadata to dedicated storage", banzaiv1beta1.BrokerIdLabelKey, currentPod.Labels[banzaiv1beta1.BrokerIdLabelKey]) + } // Print terminated container's statuses if k8sutil.IsPodContainsTerminatedContainer(currentPod) { @@ -1284,6 +1328,7 @@ func (r *Reconciler) reconcileKafkaPvc(ctx context.Context, log logr.Logger, bro if err != nil { return errorfactory.New(errorfactory.APIFailure{}, err, "getting resource failed", "kind", desiredType) } + pvcList.Items = dataPVCs(pvcList.Items) isController, err := r.isController(util.ConvertStringToInt32(brokerId)) if err != nil { @@ -1318,6 +1363,7 @@ func (r *Reconciler) reconcileKafkaPvc(ctx context.Context, log logr.Logger, bro if err != nil { return errorfactory.New(errorfactory.APIFailure{}, err, "getting resource failed", "kind", desiredType) } + pvcList.Items = dataPVCs(pvcList.Items) mountPath := currentPvc.Annotations[mountPathAnnotationKey] // Creating the first PersistentVolume For Pod diff --git a/pkg/resources/kafka/metadata_migration_test.go b/pkg/resources/kafka/metadata_migration_test.go new file mode 100644 index 000000000..4bf93d11b --- /dev/null +++ b/pkg/resources/kafka/metadata_migration_test.go @@ -0,0 +1,320 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package kafka + +import ( + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "github.com/stretchr/testify/require" +) + +type metadataFixture struct { + t *testing.T + root, metadata, config, bin string + data []string + meta []string + env []string +} + +func newMetadataFixture(t *testing.T, source int) *metadataFixture { + t.Helper() + root, err := os.MkdirTemp(".", ".metadata-fixture-") + require.NoError(t, err) + root, err = filepath.Abs(root) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, os.RemoveAll(root)) }) + f := &metadataFixture{t: t, root: root, metadata: filepath.Join(root, "metadata"), config: filepath.Join(root, "broker-config"), bin: filepath.Join(root, "bin")} + f.data = []string{filepath.Join(root, "csi-kafka-logs1"), filepath.Join(root, "csi-kafka-logs2")} + for _, path := range append(append([]string{}, f.data...), f.metadata, f.bin, filepath.Join(root, "wait")) { + require.NoError(t, os.MkdirAll(path, 0o755)) + } + f.write(f.config, fmt.Sprintf("process.roles=broker\nnode.id=1\nmetadata.log.dir=%s/kafka\nlog.dirs=%s/kafka,%s/kafka\n", f.metadata, f.data[0], f.data[1])) + f.write(filepath.Join(f.bin, "kafka-storage.sh"), `#!/bin/bash +set -eu +while [[ $# -gt 0 ]]; do + if [[ "$1" == -c ]]; then config=$2; shift; fi + shift +done +while IFS= read -r line; do + case "$line" in log.dirs=*) dir=${line#*=};; esac +done < "$config" +mkdir -p "$dir" +if [[ ! -f "$dir/meta.properties" ]]; then + printf 'version=1\ncluster.id=cluster-one\nnode.id=1\ndirectory.id=%s\n' "${FORMAT_ID:-AAAAAAAAAAAAAAAAAAAAAQ}" > "$dir/meta.properties" + if [[ "${FAULT_AT:-}" == partial-format ]]; then exit 42; fi + printf 'generated-bootstrap' > "$dir/bootstrap.checkpoint" +fi +if [[ "${FAULT_AT:-}" == format ]]; then exit 42; fi +`) + for _, name := range []string{"cp", "mv"} { + real, err := exec.LookPath(name) + require.NoError(t, err) + f.write(filepath.Join(f.bin, name), fmt.Sprintf(`#!/bin/bash +set -eu +if [[ "%s" == cp && "${FAULT_AT:-}" == copy && "$1" == -a && "$2" == */__cluster_metadata-0/. ]]; then + "%s" -a "$2/00000000000000000000.log" "$3" + exit 42 +fi +"%s" "$@" +if [[ "%s" == mv ]]; then + for arg in "$@"; do + if [[ "${FAULT_AT:-}" == manifest && "$arg" == "$METADATA_MOUNT/.koperator-metadata-migration" ]]; then exit 42; fi + if [[ "${FAULT_AT:-}" == stage-owner && "$arg" == */.koperator-metadata-stage-owner ]]; then exit 42; fi + if [[ "${FAULT_AT:-}" == publish && "$arg" == "$METADATA_MOUNT/kafka" ]]; then exit 42; fi + if [[ "${FAULT_AT:-}" == backup && "$arg" == */.koperator-metadata-backup-1 ]]; then exit 42; fi + if [[ "${FAULT_AT:-}" == complete && "$arg" == "$METADATA_MOUNT/.koperator-metadata-complete" ]]; then exit 42; fi + done +fi +`, name, real, real, name)) + } + f.env = append(os.Environ(), + "KAFKA_HOME="+root, "BROKER_CONFIG="+f.config, "WORK_DIR="+filepath.Join(root, "wait"), + "METADATA_MOUNT="+f.metadata, "NODE_ID=1", "CLUSTER_ID=cluster-one", "ALLOW_FRESH=true", + "PATH="+f.bin+":"+os.Getenv("PATH")) + if source >= 0 { + for i := range f.data { + f.setMeta(i, fmt.Sprintf("version=1\ncluster.id=cluster-one\nnode.id=1\ndirectory.id=BBBBBBBBBBBBBBBBBBBBB%d\n", i)) + f.write(filepath.Join(f.data[i], "kafka/orders-0/00000000000000000000.log"), "original-topic-replica") + } + f.addSource(source) + } + return f +} + +func (f *metadataFixture) write(path, content string) { + f.t.Helper() + require.NoError(f.t, os.MkdirAll(filepath.Dir(path), 0o755)) + require.NoError(f.t, os.WriteFile(path, []byte(content), 0o755)) +} + +func (f *metadataFixture) setMeta(i int, content string) { + f.t.Helper() + for len(f.meta) <= i { + f.meta = append(f.meta, "") + } + f.meta[i] = content + f.write(filepath.Join(f.data[i], "kafka/meta.properties"), content) +} + +// zkMigrated mimics a broker migrated from ZooKeeper: V0 data directories +// (broker.id, optionally without version) and no bootstrap.checkpoint. +func (f *metadataFixture) zkMigrated(source int, versionLine string) { + f.t.Helper() + for i := range f.data { + f.setMeta(i, versionLine+"cluster.id=cluster-one\nbroker.id=1\n") + } + require.NoError(f.t, os.Remove(filepath.Join(f.data[source], "kafka/bootstrap.checkpoint"))) +} + +func (f *metadataFixture) addSource(i int) { + f.write(filepath.Join(f.data[i], "kafka/bootstrap.checkpoint"), "original-bootstrap") + for _, name := range []string{"00000000000000000000.log", "00000000000000000000.index", "00000000000000000000.timeindex", "00000000000000000002-0000000001.checkpoint", "leader-epoch-checkpoint", "quorum-state"} { + f.write(filepath.Join(f.data[i], "kafka/__cluster_metadata-0", name), "original-"+name+"\x00\x01\x02") + } +} + +func (f *metadataFixture) run(fault string) ([]byte, error) { + cmd := exec.Command("bash", "-c", metadataMigrationScript) + cmd.Env = append(append([]string{}, f.env...), "FAULT_AT="+fault, "DATA_MOUNTS="+strings.Join(f.data, ",")) + return cmd.CombinedOutput() +} + +func (f *metadataFixture) assertRecovered(source int) { + f.t.Helper() + dest := filepath.Join(f.metadata, "kafka") + require.FileExists(f.t, filepath.Join(dest, "meta.properties")) + require.FileExists(f.t, filepath.Join(f.metadata, ".koperator-metadata-complete")) + for i := range f.data { + require.NoDirExists(f.t, filepath.Join(f.data[i], "kafka/__cluster_metadata-0")) + data, err := os.ReadFile(filepath.Join(f.data[i], "kafka/orders-0/00000000000000000000.log")) + require.NoError(f.t, err) + require.Equal(f.t, "original-topic-replica", string(data)) + meta, err := os.ReadFile(filepath.Join(f.data[i], "kafka/meta.properties")) + require.NoError(f.t, err) + require.Equal(f.t, f.meta[i], string(meta)) + } + backup := filepath.Join(f.data[source], ".koperator-metadata-backup-1") + entries, err := os.ReadDir(backup) + require.NoError(f.t, err) + for _, entry := range entries { + original, err := os.ReadFile(filepath.Join(backup, entry.Name())) + require.NoError(f.t, err) + copied, err := os.ReadFile(filepath.Join(dest, "__cluster_metadata-0", entry.Name())) + require.NoError(f.t, err) + require.Equal(f.t, original, copied) + } + sourceBootstrap, err := os.ReadFile(filepath.Join(f.data[source], "kafka/bootstrap.checkpoint")) + if os.IsNotExist(err) { + require.NoFileExists(f.t, filepath.Join(dest, "bootstrap.checkpoint")) + return + } + require.NoError(f.t, err) + bootstrap, err := os.ReadFile(filepath.Join(dest, "bootstrap.checkpoint")) + require.NoError(f.t, err) + require.Equal(f.t, sourceBootstrap, bootstrap) +} + +func TestMetadataMigrationAndRecovery(t *testing.T) { + for _, source := range []int{0, 1} { + t.Run(fmt.Sprintf("logs%d", source+1), func(t *testing.T) { + f := newMetadataFixture(t, source) + output, err := f.run("") + require.NoError(t, err, string(output)) + f.assertRecovered(source) + output, err = f.run("") + require.NoError(t, err, string(output)) + f.assertRecovered(source) + // The source data disk can subsequently be removed by the normal CC lifecycle. + f.data = []string{f.data[1-source]} + output, err = f.run("") + require.NoError(t, err, string(output)) + }) + } + for _, fault := range []string{"manifest", "stage-owner", "format", "partial-format", "copy", "publish", "backup", "complete"} { + t.Run("interrupted-"+fault, func(t *testing.T) { + f := newMetadataFixture(t, 1) + output, err := f.run(fault) + require.Error(t, err, string(output)) + output, err = f.run("") + require.NoError(t, err, string(output)) + f.assertRecovered(1) + }) + } + for _, tc := range []struct{ name, version string }{{"zk-migrated-v0", "version=0\n"}, {"zk-migrated-unversioned", ""}} { + t.Run(tc.name, func(t *testing.T) { + f := newMetadataFixture(t, 1) + f.zkMigrated(1, tc.version) + output, err := f.run("") + require.NoError(t, err, string(output)) + f.assertRecovered(1) + output, err = f.run("") + require.NoError(t, err, string(output)) + f.assertRecovered(1) + }) + } + for _, fault := range []string{"format", "copy", "publish", "backup"} { + t.Run("zk-migrated-interrupted-"+fault, func(t *testing.T) { + f := newMetadataFixture(t, 1) + f.zkMigrated(1, "version=0\n") + output, err := f.run(fault) + require.Error(t, err, string(output)) + output, err = f.run("") + require.NoError(t, err, string(output)) + f.assertRecovered(1) + }) + } + t.Run("fresh", func(t *testing.T) { + f := newMetadataFixture(t, -1) + output, err := f.run("") + require.NoError(t, err, string(output)) + output, err = f.run("") + require.NoError(t, err, string(output)) + }) +} + +func TestMetadataMigrationFailsClosed(t *testing.T) { + for _, tc := range []struct { + name string + change func(*metadataFixture) + }{ + {"multiple sources", func(f *metadataFixture) { f.addSource(1) }}, + {"unowned destination", func(f *metadataFixture) { f.write(filepath.Join(f.metadata, "kafka/meta.properties"), "version=1") }}, + {"unowned staging", func(f *metadataFixture) { + f.write(filepath.Join(f.metadata, ".koperator-metadata-stage/meta.properties"), "version=1") + }}, + {"wrong source node", func(f *metadataFixture) { + f.write(filepath.Join(f.data[0], "kafka/meta.properties"), "version=1\ncluster.id=cluster-one\nnode.id=2\n") + }}, + {"wrong source cluster", func(f *metadataFixture) { + f.write(filepath.Join(f.data[0], "kafka/meta.properties"), "version=1\ncluster.id=cluster-two\nnode.id=1\n") + }}, + {"wrong V0 source broker", func(f *metadataFixture) { + f.zkMigrated(0, "version=0\n") + f.setMeta(0, "version=0\ncluster.id=cluster-one\nbroker.id=2\n") + }}, + {"V0 source without cluster", func(f *metadataFixture) { + f.zkMigrated(0, "version=0\n") + f.setMeta(0, "version=0\nbroker.id=1\n") + }}, + {"unsupported source version", func(f *metadataFixture) { + f.setMeta(0, "version=2\ncluster.id=cluster-one\nnode.id=1\n") + }}, + {"duplicate source version", func(f *metadataFixture) { + f.setMeta(0, "version=1\nversion=1\ncluster.id=cluster-one\nnode.id=1\n") + }}, + {"symlinked source bootstrap", func(f *metadataFixture) { + bootstrap := filepath.Join(f.data[0], "kafka/bootstrap.checkpoint") + require.NoError(f.t, os.Remove(bootstrap)) + require.NoError(f.t, os.Symlink(filepath.Join(f.root, "broker-config"), bootstrap)) + }}, + {"combined", func(f *metadataFixture) { f.write(f.config, "process.roles=broker,controller\nnode.id=1\n") }}, + {"empty source", func(f *metadataFixture) { + require.NoError(f.t, os.RemoveAll(filepath.Join(f.data[0], "kafka/__cluster_metadata-0"))) + require.NoError(f.t, os.Mkdir(filepath.Join(f.data[0], "kafka/__cluster_metadata-0"), 0o755)) + }}, + {"lost source", func(f *metadataFixture) { + require.NoError(f.t, os.RemoveAll(filepath.Join(f.data[0], "kafka/__cluster_metadata-0"))) + }}, + {"cloned directory ID", func(f *metadataFixture) { f.env = append(f.env, "FORMAT_ID=BBBBBBBBBBBBBBBBBBBBB0") }}, + {"missing unique directory ID", func(f *metadataFixture) { f.env = append(f.env, "FORMAT_ID=invalid") }}, + } { + t.Run(tc.name, func(t *testing.T) { + f := newMetadataFixture(t, 0) + tc.change(f) + output, err := f.run("") + require.Error(t, err, string(output)) + require.NoFileExists(t, filepath.Join(f.metadata, ".koperator-metadata-complete")) + topic, err := os.ReadFile(filepath.Join(f.data[0], "kafka/orders-0/00000000000000000000.log")) + require.NoError(t, err) + require.Equal(t, "original-topic-replica", string(topic)) + }) + } + + t.Run("wrong published identity", func(t *testing.T) { + f := newMetadataFixture(t, 0) + output, err := f.run("") + require.NoError(t, err, string(output)) + f.write(filepath.Join(f.metadata, "kafka/meta.properties"), "version=1\ncluster.id=wrong\nnode.id=1\n") + output, err = f.run("") + require.Error(t, err, string(output)) + }) + t.Run("existing broker without source", func(t *testing.T) { + f := newMetadataFixture(t, -1) + f.env = append(f.env, "ALLOW_FRESH=false") + output, err := f.run("") + require.Error(t, err, string(output)) + }) +} + +func TestMetadataStartupFormatFailurePreventsKafka(t *testing.T) { + f := newMetadataFixture(t, -1) + f.write(filepath.Join(f.bin, "kafka-storage.sh"), "#!/bin/bash\nexit 42\n") + marker := filepath.Join(f.root, "kafka-started") + f.write(filepath.Join(f.bin, "kafka-server-start.sh"), "#!/bin/bash\ntouch '"+marker+"'\n") + cmd := exec.Command("bash", "-c", envoySidecarScript) + cmd.Env = append(append([]string{}, f.env...), "WAIT_DIR="+filepath.Join(f.root, "wait"), "METADATA_STORAGE_ENABLED=true") + output, err := cmd.CombinedOutput() + require.Error(t, err, string(output)) + var exitError *exec.ExitError + require.ErrorAs(t, err, &exitError) + require.Equal(t, 42, exitError.ExitCode()) + require.NoFileExists(t, marker) + require.NoFileExists(t, filepath.Join(f.root, "wait/do-not-exit-yet")) +} diff --git a/pkg/resources/kafka/metadata_storage.go b/pkg/resources/kafka/metadata_storage.go new file mode 100644 index 000000000..c7991e08f --- /dev/null +++ b/pkg/resources/kafka/metadata_storage.go @@ -0,0 +1,368 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package kafka + +import ( + "context" + "encoding/json" + "fmt" + + "emperror.dev/errors" + "github.com/go-logr/logr" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "sigs.k8s.io/controller-runtime/pkg/client" + + apiutil "github.com/banzaicloud/koperator/api/util" + "github.com/banzaicloud/koperator/api/v1beta1" + "github.com/banzaicloud/koperator/pkg/errorfactory" + "github.com/banzaicloud/koperator/pkg/k8sutil" + kafkautils "github.com/banzaicloud/koperator/pkg/util/kafka" + properties "github.com/banzaicloud/koperator/properties/pkg" +) + +const ( + storagePurposeLabel = "kafka.banzaicloud.io/storage-purpose" + metadataPurpose = "metadata" + metadataConfigAnnotation = "kafka.banzaicloud.io/metadata-storage-config" + metadataFreshAnnotation = "kafka.banzaicloud.io/metadata-fresh-storage" + metadataReadyAnnotation = "kafka.banzaicloud.io/metadata-storage-ready" + metadataVolumeName = v1beta1.MetadataStorageVolumeName +) + +func metadataMigrationRecovered(pod *corev1.Pod, pvcName string) bool { + if !pod.DeletionTimestamp.IsZero() || pod.Status.Phase != corev1.PodRunning || !isPodReady(pod) { + return false + } + mounted := false + for _, volume := range pod.Spec.Volumes { + if volume.Name == metadataVolumeName && volume.PersistentVolumeClaim != nil && + volume.PersistentVolumeClaim.ClaimName == pvcName { + mounted = true + break + } + } + if !mounted { + return false + } + for _, status := range pod.Status.InitContainerStatuses { + if status.Name == "migrate-broker-metadata" { + return status.State.Terminated != nil && status.State.Terminated.ExitCode == 0 + } + } + return false +} + +func isMetadataPVC(pvc corev1.PersistentVolumeClaim) bool { + return pvc.Labels[storagePurposeLabel] == metadataPurpose +} + +func dataPVCs(pvcs []corev1.PersistentVolumeClaim) []corev1.PersistentVolumeClaim { + result := make([]corev1.PersistentVolumeClaim, 0, len(pvcs)) + for _, pvc := range pvcs { + if !isMetadataPVC(pvc) { + result = append(result, pvc) + } + } + return result +} + +func (r *Reconciler) metadataPVC(ctx context.Context, brokerID int32) (*corev1.PersistentVolumeClaim, error) { + pvcs := &corev1.PersistentVolumeClaimList{} + if err := r.List(ctx, pvcs, client.InNamespace(r.KafkaCluster.Namespace), client.MatchingLabels(apiutil.MergeLabels( + apiutil.LabelsForKafka(r.KafkaCluster.Name), map[string]string{v1beta1.BrokerIdLabelKey: fmt.Sprint(brokerID)}, + ))); err != nil { + return nil, err + } + var result *corev1.PersistentVolumeClaim + for _, pvc := range pvcs.Items { + if !isMetadataPVC(pvc) { + continue + } + if result != nil || !pvc.DeletionTimestamp.IsZero() { + return nil, fmt.Errorf("broker %d metadata PVC is ambiguous or terminating", brokerID) + } + result = pvc.DeepCopy() + } + return result, nil +} + +// Metadata PVCs have their own lifecycle. They must never enter disk-draining +// reconciliation, capacity generation or Cruise Control volume states. +func (r *Reconciler) reconcileMetadataStorage(ctx context.Context, broker v1beta1.Broker, config *v1beta1.BrokerConfig, log logr.Logger) error { + if config == nil { + return fmt.Errorf("broker %d has no effective configuration", broker.Id) + } + if err := config.ValidateMetadataStorage(r.KafkaCluster.Spec.KRaftMode); err != nil { + return err + } + current, err := r.metadataPVC(ctx, broker.Id) + if err != nil { + return err + } + if config.MetadataStorage == nil { + if current != nil { + return fmt.Errorf("broker %d cannot remove enabled metadataStorage", broker.Id) + } + return nil + } + if !shouldUseKRaftModeForBroker(getBrokerReadOnlyConfig(broker, r.KafkaCluster, log)) { + return fmt.Errorf("metadataStorage cannot be used by a ZooKeeper migration broker") + } + pvcs := &corev1.PersistentVolumeClaimList{} + if err := r.List(ctx, pvcs, client.InNamespace(r.KafkaCluster.Namespace), client.MatchingLabels(apiutil.MergeLabels( + apiutil.LabelsForKafka(r.KafkaCluster.Name), map[string]string{v1beta1.BrokerIdLabelKey: fmt.Sprint(broker.Id)}, + ))); err != nil { + return err + } + pods := &corev1.PodList{} + if err := r.List(ctx, pods, client.InNamespace(r.KafkaCluster.Namespace), client.MatchingLabels(apiutil.MergeLabels( + apiutil.LabelsForKafka(r.KafkaCluster.Name), map[string]string{v1beta1.BrokerIdLabelKey: fmt.Sprint(broker.Id)}, + ))); err != nil { + return err + } + migrationReady := current != nil && current.Annotations[metadataReadyAnnotation] == configValueTrue + if current != nil && !migrationReady { + for i := range pods.Items { + if metadataMigrationRecovered(&pods.Items[i], current.Name) { + migrationReady = true + break + } + } + } + for _, pvc := range dataPVCs(pvcs.Items) { + if v1beta1.StoragePathsOverlap(config.MetadataStorage.MountPath, pvc.Annotations[mountPathAnnotationKey]) { + return fmt.Errorf("metadataStorage overlaps existing data PVC %s", pvc.Name) + } + if !migrationReady { + found := false + for _, storage := range config.StorageConfigs { + if storage.MountPath == pvc.Annotations[mountPathAnnotationKey] { + found = true + } + } + if !found { + // Admission rejects this using MetadataStorageState; this guards bypassed admission. + return fmt.Errorf("broker %d: keep all existing data disks until metadata migration has completed and the replacement pod is ready; "+ + "restore data storageConfig %s to resume reconciliation", broker.Id, pvc.Annotations[mountPathAnnotationKey]) + } + } + } + desired, err := r.pvc(broker.Id, 0, *config.MetadataStorage, config, true) + if err != nil { + return err + } + desired.Name = fmt.Sprintf("%s-%d-metadata", r.KafkaCluster.Name, broker.Id) + desired.GenerateName = "" + desired.Labels[storagePurposeLabel] = metadataPurpose + location := config.MetadataStorage.DeepCopy() + location.PvcSpec.Resources = corev1.VolumeResourceRequirements{} + locationJSON, err := json.Marshal(location) + if err != nil { + return err + } + desired.Annotations[metadataConfigAnnotation] = string(locationJSON) + if current == nil { + desired.Annotations[metadataFreshAnnotation] = fmt.Sprint(len(dataPVCs(pvcs.Items)) == 0 && len(pods.Items) == 0 && + r.KafkaCluster.Status.BrokersState[fmt.Sprint(broker.Id)].Version == "") + } + if current != nil { + if current.Annotations[metadataConfigAnnotation] != string(locationJSON) || + current.Annotations[mountPathAnnotationKey] != config.MetadataStorage.MountPath { + return fmt.Errorf("broker %d cannot relocate or replace enabled metadataStorage", broker.Id) + } + desired = current.DeepCopy() + desired.Spec.Resources = config.MetadataStorage.PvcSpec.Resources + if isDesiredStorageValueInvalid(desired, current) { + return fmt.Errorf("cannot reduce metadata PVC size") + } + if migrationReady { + desired.Annotations[metadataReadyAnnotation] = configValueTrue + } + } + if err := k8sutil.Reconcile(log, r.Client, desired, r.KafkaCluster); err != nil { + return err + } + // The PVC annotation is the durable record; the status projects it for admission. + if migrationReady && r.KafkaCluster.Status.BrokersState[fmt.Sprint(broker.Id)].MetadataStorageState != v1beta1.MetadataStorageReady { + return k8sutil.UpdateBrokerStatus(r.Client, []string{fmt.Sprint(broker.Id)}, r.KafkaCluster, v1beta1.MetadataStorageReady, log) + } + return nil +} + +// deleteMetadataPVC deletes the dedicated metadata PVC of a removed broker, so the +// broker ID can later be reused with or without metadata storage. +func (r *Reconciler) deleteMetadataPVC(ctx context.Context, brokerID string, log logr.Logger) error { + pvcs := &corev1.PersistentVolumeClaimList{} + if err := r.List(ctx, pvcs, client.InNamespace(r.KafkaCluster.Namespace), client.MatchingLabels(apiutil.MergeLabels( + apiutil.LabelsForKafka(r.KafkaCluster.Name), map[string]string{v1beta1.BrokerIdLabelKey: brokerID, storagePurposeLabel: metadataPurpose}, + ))); err != nil { + return err + } + for i := range pvcs.Items { + if !pvcs.Items[i].DeletionTimestamp.IsZero() { + continue + } + if err := client.IgnoreNotFound(r.Delete(ctx, &pvcs.Items[i])); err != nil { + return err + } + log.V(1).Info("metadata pvc for broker deleted", "pvc name", pvcs.Items[i].Name, v1beta1.BrokerIdLabelKey, brokerID) + } + return nil +} + +func podMountsMetadataVolume(pod *corev1.Pod) bool { + for _, volume := range pod.Spec.Volumes { + if volume.Name == metadataVolumeName && volume.PersistentVolumeClaim != nil { + return true + } + } + return false +} + +func configMapHasMetadataLogDir(configMap *corev1.ConfigMap) bool { + if configMap == nil { + return false + } + props, err := properties.NewFromString(configMap.Data[kafkautils.ConfigPropertyName]) + if err != nil { + return false + } + value, found := props.Get(kafkautils.KafkaConfigMetadataLogDirectory) + return found && value.Value() != "" +} + +// metadataLogDirConfigurable reports whether metadata.log.dir may be written into the +// broker ConfigMap. The ConfigMap is live-mounted, so a still-running pod without the +// metadata volume would otherwise format the unmounted directory on a container restart. +// Once published it is kept, because metadataStorage cannot be disabled. +func (r *Reconciler) metadataLogDirConfigurable(ctx context.Context, brokerID int32, current *corev1.ConfigMap, log logr.Logger) bool { + if configMapHasMetadataLogDir(current) { + return true + } + pods := &corev1.PodList{} + if err := r.List(ctx, pods, client.InNamespace(r.KafkaCluster.Namespace), client.MatchingLabels(apiutil.MergeLabels( + apiutil.LabelsForKafka(r.KafkaCluster.Name), map[string]string{v1beta1.BrokerIdLabelKey: fmt.Sprint(brokerID)}, + ))); err != nil { + log.Error(err, "deferring metadata.log.dir: could not list broker pods", v1beta1.BrokerIdLabelKey, brokerID) + return false + } + for i := range pods.Items { + if !podMountsMetadataVolume(&pods.Items[i]) { + log.Info("deferring metadata.log.dir until the broker pod without metadata storage is replaced", + v1beta1.BrokerIdLabelKey, brokerID, "pod", pods.Items[i].Name) + return false + } + } + return true +} + +// ensureMetadataLogDirConfigured fails closed before creating a metadata-storage pod whose +// ConfigMap does not yet point Kafka at the metadata volume. A missing ConfigMap (rack +// awareness creates it after the pod) is generated later with the setting. +func (r *Reconciler) ensureMetadataLogDirConfigured(ctx context.Context, brokerID string, bConfig *v1beta1.BrokerConfig) error { + if bConfig == nil || bConfig.MetadataStorage == nil { + return nil + } + configMap := &corev1.ConfigMap{} + err := r.Get(ctx, client.ObjectKey{Namespace: r.KafkaCluster.Namespace, Name: fmt.Sprintf(brokerConfigTemplate+"-%s", r.KafkaCluster.Name, brokerID)}, configMap) + if apierrors.IsNotFound(err) { + return nil + } + if err != nil { + return err + } + if !configMapHasMetadataLogDir(configMap) { + return errorfactory.New(errorfactory.ResourceNotReady{}, errors.New("metadata.log.dir not yet configured"), + "broker configmap does not set metadata.log.dir; deferring pod creation", v1beta1.BrokerIdLabelKey, brokerID) + } + return nil +} + +func isMetadataMigrationRestart(currentPod, desiredPod *corev1.Pod) bool { + return podMountsMetadataVolume(desiredPod) && !podMountsMetadataVolume(currentPod) +} + +// metadataMigrationGate lets metadataStorage be enabled for many brokers at once: +// it starts a migration only when no other broker's migration is in progress and +// no other broker has offline or out-of-sync replicas. +func (r *Reconciler) metadataMigrationGate(ctx context.Context, currentPod *corev1.Pod) error { + waiting := func(reason string, keysAndValues ...interface{}) error { + return errorfactory.New(errorfactory.ReconcileRollingUpgrade{}, errors.New(reason), + "waiting before migrating the next broker's metadata", keysAndValues...) + } + currentID := currentPod.Labels[v1beta1.BrokerIdLabelKey] + if r.metadataMigrationStarted { + return waiting("a metadata migration was started in this reconcile", v1beta1.BrokerIdLabelKey, currentID) + } + // Read through the API server: a just-deleted pod may not be in the cache yet. + var reader client.Reader = r.Client + if r.DirectClient != nil { + reader = r.DirectClient + } + pods := &corev1.PodList{} + if err := reader.List(ctx, pods, client.InNamespace(r.KafkaCluster.Namespace), + client.MatchingLabels(apiutil.LabelsForKafka(r.KafkaCluster.Name))); err != nil { + return errors.WrapIf(err, "failed to list broker pods for metadata migration gate") + } + podsByBroker := make(map[string][]corev1.Pod) + for _, pod := range pods.Items { + id := pod.Labels[v1beta1.BrokerIdLabelKey] + podsByBroker[id] = append(podsByBroker[id], pod) + } + for _, broker := range r.KafkaCluster.Spec.Brokers { + id := fmt.Sprint(broker.Id) + if id == currentID { + continue + } + config, err := broker.GetBrokerConfig(r.KafkaCluster.Spec) + if err != nil { + return err + } + if config.MetadataStorage == nil || r.KafkaCluster.Status.BrokersState[id].MetadataStorageState == v1beta1.MetadataStorageReady { + continue + } + // Not Ready and the old pod is gone, going, or already replaced: migration in progress. + inProgress := len(podsByBroker[id]) == 0 + for i := range podsByBroker[id] { + if podMountsMetadataVolume(&podsByBroker[id][i]) || !podsByBroker[id][i].DeletionTimestamp.IsZero() { + inProgress = true + } + } + if inProgress { + return waiting("metadata migration of another broker is not Ready yet", v1beta1.BrokerIdLabelKey, currentID, "migratingBrokerId", id) + } + } + kClient, closeClient, err := r.kafkaClientProvider.NewFromCluster(r.Client, r.KafkaCluster) + if err != nil { + return errorfactory.New(errorfactory.BrokersUnreachable{}, err, "could not connect to kafka brokers") + } + defer closeClient() + offline, err := kClient.AllOfflineReplicas() + if err != nil { + return errors.WrapIf(err, "metadata migration health check failed") + } + outOfSync, err := kClient.OutOfSyncReplicas() + if err != nil { + return errors.WrapIf(err, "metadata migration health check failed") + } + // The broker being migrated may itself be unhealthy (e.g. a crashed container). + for _, brokerID := range append(offline, outOfSync...) { + if fmt.Sprint(brokerID) != currentID { + return waiting("cluster has offline or out-of-sync replicas", v1beta1.BrokerIdLabelKey, currentID, "impactedBrokerId", brokerID) + } + } + return nil +} diff --git a/pkg/resources/kafka/metadata_storage_test.go b/pkg/resources/kafka/metadata_storage_test.go new file mode 100644 index 000000000..ab0248e93 --- /dev/null +++ b/pkg/resources/kafka/metadata_storage_test.go @@ -0,0 +1,532 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package kafka + +import ( + "context" + "reflect" + "testing" + + "emperror.dev/errors" + "github.com/go-logr/logr" + "github.com/stretchr/testify/require" + "go.uber.org/mock/gomock" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + apiutil "github.com/banzaicloud/koperator/api/util" + "github.com/banzaicloud/koperator/api/v1beta1" + "github.com/banzaicloud/koperator/pkg/errorfactory" + "github.com/banzaicloud/koperator/pkg/kafkaclient" + "github.com/banzaicloud/koperator/pkg/resources" + "github.com/banzaicloud/koperator/pkg/resources/kafka/mocks" + kafkautils "github.com/banzaicloud/koperator/pkg/util/kafka" + properties "github.com/banzaicloud/koperator/properties/pkg" +) + +func metadataTestReconciler(t *testing.T) (*Reconciler, v1beta1.Broker) { + t.Helper() + scheme := runtime.NewScheme() + require.NoError(t, corev1.AddToScheme(scheme)) + require.NoError(t, v1beta1.AddToScheme(scheme)) + config := &v1beta1.BrokerConfig{ + Roles: []string{"broker"}, + StorageConfigs: []v1beta1.StorageConfig{{MountPath: "/csi-kafka-logs1", PvcSpec: &corev1.PersistentVolumeClaimSpec{ + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + Resources: corev1.VolumeResourceRequirements{Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse("100Gi")}}, + }}}, + MetadataStorage: &v1beta1.StorageConfig{MountPath: "/csi-kafka-metadata", PvcSpec: &corev1.PersistentVolumeClaimSpec{ + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + Resources: corev1.VolumeResourceRequirements{Requests: corev1.ResourceList{corev1.ResourceStorage: resource.MustParse("10Gi")}}, + }}, + } + broker := v1beta1.Broker{Id: 1, BrokerConfig: config} + cluster := &v1beta1.KafkaCluster{ + ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "test", UID: "cluster-uid"}, + Spec: v1beta1.KafkaClusterSpec{KRaftMode: true, Brokers: []v1beta1.Broker{broker}}, + Status: v1beta1.KafkaClusterStatus{ClusterID: "cluster-one"}, + } + c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(cluster).WithStatusSubresource(cluster).Build() + return &Reconciler{Reconciler: resources.Reconciler{Client: c, KafkaCluster: cluster}}, broker +} + +func TestMetadataStorageIndependentLifecycle(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx, log := context.Background(), logr.Discard() + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + pvc, err := r.metadataPVC(ctx, broker.Id) + require.NoError(t, err) + require.NotNil(t, pvc) + require.Equal(t, "test-1-metadata", pvc.Name) + require.Equal(t, "true", pvc.Annotations[metadataFreshAnnotation]) + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + all := &corev1.PersistentVolumeClaimList{} + require.NoError(t, r.List(ctx, all)) + require.Len(t, all.Items, 1) + + data, err := r.pvc(1, 0, broker.BrokerConfig.StorageConfigs[0], broker.BrokerConfig, true) + require.NoError(t, err) + data.Name, data.GenerateName = "data", "" + data.Status.Phase = corev1.ClaimBound + require.NoError(t, r.Create(ctx, data)) + dataPvcs, err := getCreatedPvcForBroker(ctx, r.Client, nil, 1, broker.BrokerConfig.StorageConfigs, "test", "test") + require.NoError(t, err) + require.Len(t, dataPvcs, 1) + require.Equal(t, "data", dataPvcs[0].Name) + desired, err := r.pvc(1, 0, broker.BrokerConfig.StorageConfigs[0], broker.BrokerConfig, true) + require.NoError(t, err) + require.NoError(t, r.reconcileKafkaPvc(ctx, log, map[string][]*corev1.PersistentVolumeClaim{"1": {desired}}, map[string]struct{}{"1": {}})) + for _, state := range r.KafkaCluster.Status.BrokersState { + require.NotContains(t, state.GracefulActionState.VolumeStates, broker.BrokerConfig.MetadataStorage.MountPath) + } + require.NoError(t, r.Get(ctx, client.ObjectKeyFromObject(pvc), &corev1.PersistentVolumeClaim{})) + + pod := r.pod(1, broker.BrokerConfig, append(dataPvcs, *pvc), log).(*corev1.Pod) + require.Len(t, pod.Spec.InitContainers, 3) + init := pod.Spec.InitContainers[2] + require.Equal(t, "migrate-broker-metadata", init.Name) + require.Equal(t, pod.Spec.Containers[0].Image, init.Image) + require.Equal(t, pod.Spec.Containers[0].Resources, init.Resources) + for _, env := range init.Env { + if env.Name == "DATA_MOUNTS" { + require.Equal(t, "/csi-kafka-logs1", env.Value) + } + } + for _, env := range pod.Spec.Containers[0].Env { + if env.Name == "LOG_DIRS" { + require.Equal(t, "/csi-kafka-logs1", env.Value) + } + } + require.Contains(t, pod.Spec.Containers[0].VolumeMounts, corev1.VolumeMount{Name: metadataVolumeName, MountPath: "/csi-kafka-metadata"}) + config := r.generateBrokerConfig(broker, broker.BrokerConfig, nil, nil, nil, nil, nil, "", nil, log) + props, err := properties.NewFromString(config) + require.NoError(t, err) + logDirs, _ := props.Get("log.dirs") + metadataDir, _ := props.Get("metadata.log.dir") + require.Equal(t, "/csi-kafka-logs1/kafka", logDirs.Value()) + require.Equal(t, "/csi-kafka-metadata/kafka", metadataDir.Value()) + + for _, change := range []func(*v1beta1.BrokerConfig){ + func(b *v1beta1.BrokerConfig) { b.MetadataStorage = nil }, + func(b *v1beta1.BrokerConfig) { b.MetadataStorage.MountPath = "/relocated" }, + func(b *v1beta1.BrokerConfig) { + value := "different-class" + b.MetadataStorage.PvcSpec.StorageClassName = &value + }, + func(b *v1beta1.BrokerConfig) { + b.MetadataStorage.PvcSpec.Resources.Requests[corev1.ResourceStorage] = resource.MustParse("1Gi") + }, + } { + next := broker.BrokerConfig.DeepCopy() + change(next) + require.Error(t, r.reconcileMetadataStorage(ctx, broker, next, log)) + } + grown := broker.BrokerConfig.DeepCopy() + grown.MetadataStorage.PvcSpec.Resources.Requests[corev1.ResourceStorage] = resource.MustParse("20Gi") + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, grown, log)) +} + +func TestMetadataStorageExistingSourcesAndDefault(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx, log := context.Background(), logr.Discard() + existing := createPvc("old-data", "1", "/csi-kafka-logs1") + existing.Namespace = "test" + existing.Labels = apiutil.MergeLabels(existing.Labels, apiutil.LabelsForKafka("test")) + require.NoError(t, r.Create(ctx, existing)) + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + pvc, err := r.metadataPVC(ctx, 1) + require.NoError(t, err) + require.Equal(t, "false", pvc.Annotations[metadataFreshAnnotation]) + t.Run("existing pod without version or data PVC cannot be fresh", func(t *testing.T) { + r, broker := metadataTestReconciler(t) + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ + Name: "existing", Namespace: "test", + Labels: apiutil.MergeLabels(apiutil.LabelsForKafka("test"), map[string]string{v1beta1.BrokerIdLabelKey: "1"}), + }} + require.NoError(t, r.Create(ctx, pod)) + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + metadataPVC, err := r.metadataPVC(ctx, 1) + require.NoError(t, err) + require.Equal(t, "false", metadataPVC.Annotations[metadataFreshAnnotation]) + }) + + for _, roles := range [][]string{{"broker"}, {"controller"}, {"broker", "controller"}} { + b := broker.BrokerConfig.DeepCopy() + b.MetadataStorage = nil + b.Roles = roles + pod := r.pod(1, b, []corev1.PersistentVolumeClaim{*existing}, log).(*corev1.Pod) + require.Len(t, pod.Spec.InitContainers, 2) + require.Contains(t, pod.Spec.Containers[0].Command[2], "QUORUM_STATE_FILE=") + for _, volume := range pod.Spec.Volumes { + require.NotEqual(t, metadataVolumeName, volume.Name) + } + } +} + +func TestDataDiskRemovalRetainsMetadataPVC(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx, log := context.Background(), logr.Discard() + second := broker.BrokerConfig.StorageConfigs[0].DeepCopy() + second.MountPath = "/csi-kafka-logs2" + broker.BrokerConfig.StorageConfigs = append(broker.BrokerConfig.StorageConfigs, *second) + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + metadataPVC, err := r.metadataPVC(ctx, 1) + require.NoError(t, err) + for i, storage := range broker.BrokerConfig.StorageConfigs { + pvc, err := r.pvc(1, i, storage, broker.BrokerConfig, true) + require.NoError(t, err) + pvc.Name = "data-" + string(rune('1'+i)) + pvc.GenerateName = "" + pvc.Status.Phase = corev1.ClaimBound + require.NoError(t, r.Create(ctx, pvc)) + } + broker.BrokerConfig.StorageConfigs = []v1beta1.StorageConfig{*second} + require.ErrorContains(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log), "keep all existing data disks") + require.Empty(t, r.KafkaCluster.Status.BrokersState["1"].MetadataStorageState) + require.NoError(t, r.Get(ctx, client.ObjectKey{Namespace: "test", Name: "data-1"}, &corev1.PersistentVolumeClaim{})) + pod := &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{ + Name: "migrated", Namespace: "test", + Labels: apiutil.MergeLabels(apiutil.LabelsForKafka("test"), map[string]string{v1beta1.BrokerIdLabelKey: "1"}), + }, + Spec: corev1.PodSpec{Volumes: []corev1.Volume{{ + Name: metadataVolumeName, + VolumeSource: corev1.VolumeSource{PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ + ClaimName: metadataPVC.Name, + }}, + }}}, + Status: corev1.PodStatus{ + Phase: corev1.PodRunning, + Conditions: []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}}, + InitContainerStatuses: []corev1.ContainerStatus{{ + Name: "migrate-broker-metadata", + State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: 0}}, + }}, + }, + } + require.NoError(t, r.Create(ctx, pod)) + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + metadataPVC, err = r.metadataPVC(ctx, 1) + require.NoError(t, err) + require.Equal(t, "true", metadataPVC.Annotations[metadataReadyAnnotation]) + stored := &v1beta1.KafkaCluster{} + require.NoError(t, r.Get(ctx, client.ObjectKeyFromObject(r.KafkaCluster), stored)) + require.Equal(t, v1beta1.MetadataStorageReady, stored.Status.BrokersState["1"].MetadataStorageState) + + // The PVC annotation is durable: a lost status is projected again without the pod. + require.NoError(t, r.Delete(ctx, pod)) + r.KafkaCluster.Status.BrokersState = nil + require.NoError(t, r.Client.Status().Update(ctx, r.KafkaCluster)) + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + require.Equal(t, v1beta1.MetadataStorageReady, r.KafkaCluster.Status.BrokersState["1"].MetadataStorageState) + + r.KafkaCluster.Status.BrokersState = map[string]v1beta1.BrokerState{"1": { + GracefulActionState: v1beta1.GracefulActionState{VolumeStates: map[string]v1beta1.VolumeState{ + "/csi-kafka-logs1": {CruiseControlVolumeState: v1beta1.GracefulDiskRemovalSucceeded}, + "/csi-kafka-logs2": {CruiseControlVolumeState: v1beta1.GracefulDiskRebalanceSucceeded}, + }}, + }} + require.NoError(t, r.Client.Status().Update(ctx, r.KafkaCluster)) + desired, err := r.pvc(1, 1, *second, broker.BrokerConfig, true) + require.NoError(t, err) + require.NoError(t, r.reconcileKafkaPvc(ctx, log, map[string][]*corev1.PersistentVolumeClaim{"1": {desired}}, map[string]struct{}{"1": {}})) + require.Error(t, r.Get(ctx, client.ObjectKey{Namespace: "test", Name: "data-1"}, &corev1.PersistentVolumeClaim{})) + require.NoError(t, r.Get(ctx, client.ObjectKeyFromObject(metadataPVC), &corev1.PersistentVolumeClaim{})) + pvcs, err := getCreatedPvcForBroker(ctx, r.Client, nil, 1, broker.BrokerConfig.StorageConfigs, "test", "test") + require.NoError(t, err) + require.Len(t, pvcs, 1) + require.Equal(t, "data-2", pvcs[0].Name) + require.NotContains(t, r.KafkaCluster.Status.BrokersState["1"].GracefulActionState.VolumeStates, "/csi-kafka-metadata") +} + +func TestBrokerRemovalDeletesMetadataPVC(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx, log := context.Background(), logr.Discard() + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + other := broker.DeepCopy() + other.Id = 2 + require.NoError(t, r.reconcileMetadataStorage(ctx, *other, other.BrokerConfig, log)) + data, err := r.pvc(1, 0, broker.BrokerConfig.StorageConfigs[0], broker.BrokerConfig, true) + require.NoError(t, err) + data.Name, data.GenerateName = "data", "" + require.NoError(t, r.Create(ctx, data)) + + require.NoError(t, r.deleteMetadataPVC(ctx, "1", log)) + require.NoError(t, r.deleteMetadataPVC(ctx, "1", log)) + removed, err := r.metadataPVC(ctx, 1) + require.NoError(t, err) + require.Nil(t, removed) + kept, err := r.metadataPVC(ctx, 2) + require.NoError(t, err) + require.NotNil(t, kept) + require.NoError(t, r.Get(ctx, client.ObjectKey{Namespace: "test", Name: "data"}, &corev1.PersistentVolumeClaim{})) + + // A data PVC that is already gone must not prevent metadata PVC cleanup. + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + pod := corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Name: "removed", Namespace: "test", Labels: map[string]string{v1beta1.BrokerIdLabelKey: "1"}}, + Spec: corev1.PodSpec{Volumes: []corev1.Volume{ + {Name: kafkaDataVolumeMount + "-0", VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: "already-deleted"}}}, + {Name: kafkaDataVolumeMount + "-1", VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: "data"}}}, + }}, + } + require.NoError(t, r.deleteBrokerPVCs(ctx, pod, log)) + removed, err = r.metadataPVC(ctx, 1) + require.NoError(t, err) + require.Nil(t, removed) + require.Error(t, r.Get(ctx, client.ObjectKey{Namespace: "test", Name: "data"}, &corev1.PersistentVolumeClaim{})) + + // The broker ID can be reused without metadata storage. + withoutMetadata := broker.BrokerConfig.DeepCopy() + withoutMetadata.MetadataStorage = nil + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, withoutMetadata, log)) +} + +func TestMetadataMigrationRecoveryGate(t *testing.T) { + pod := &corev1.Pod{ + Spec: corev1.PodSpec{Volumes: []corev1.Volume{{ + Name: metadataVolumeName, + VolumeSource: corev1.VolumeSource{PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: "metadata"}}, + }}}, + Status: corev1.PodStatus{ + Phase: corev1.PodRunning, + Conditions: []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}}, + InitContainerStatuses: []corev1.ContainerStatus{{ + Name: "migrate-broker-metadata", + State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: 0}}, + }}, + }, + } + require.True(t, metadataMigrationRecovered(pod, "metadata")) + require.False(t, metadataMigrationRecovered(pod, "different-claim")) + for _, change := range []func(*corev1.Pod){ + func(p *corev1.Pod) { p.Status.Phase = corev1.PodPending }, + func(p *corev1.Pod) { p.Status.Conditions = nil }, + func(p *corev1.Pod) { p.Status.InitContainerStatuses = nil }, + func(p *corev1.Pod) { p.Status.InitContainerStatuses[0].State.Terminated.ExitCode = 1 }, + func(p *corev1.Pod) { now := metav1.Now(); p.DeletionTimestamp = &now }, + } { + next := pod.DeepCopy() + change(next) + require.False(t, metadataMigrationRecovered(next, "metadata")) + } +} + +func TestMetadataMigrationClusterIDEnvironment(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx, log := context.Background(), logr.Discard() + require.NoError(t, r.reconcileMetadataStorage(ctx, broker, broker.BrokerConfig, log)) + pvc, err := r.metadataPVC(ctx, broker.Id) + require.NoError(t, err) + for _, clusterID := range []corev1.EnvVar{ + {Name: clusterIDEnvVarName, Value: "explicit-cluster"}, + {Name: clusterIDEnvVarName, ValueFrom: &corev1.EnvVarSource{ + SecretKeyRef: &corev1.SecretKeySelector{ + LocalObjectReference: corev1.LocalObjectReference{Name: "cluster-identity"}, + Key: "id", + }, + }}, + } { + r.KafkaCluster.Spec.Envs = []corev1.EnvVar{clusterID} + pod := r.pod(broker.Id, broker.BrokerConfig, []corev1.PersistentVolumeClaim{*pvc}, log).(*corev1.Pod) + require.Contains(t, pod.Spec.InitContainers[len(pod.Spec.InitContainers)-1].Env, clusterID) + } +} + +func TestMetadataLogDirDeferredWhilePodWithoutMetadataVolumeExists(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx, log := context.Background(), logr.Discard() + metadataLogDir := func() (string, bool) { + props, err := properties.NewFromString(r.generateBrokerConfig(broker, broker.BrokerConfig, nil, nil, nil, nil, nil, "", nil, log)) + require.NoError(t, err) + value, found := props.Get(kafkautils.KafkaConfigMetadataLogDirectory) + if !found { + return "", false + } + return value.Value(), true + } + labels := apiutil.MergeLabels(apiutil.LabelsForKafka("test"), map[string]string{v1beta1.BrokerIdLabelKey: "1"}) + + // The running pre-migration pod live-mounts the ConfigMap but has no metadata volume. + oldPod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "test-1-old", Namespace: "test", Labels: labels}} + require.NoError(t, r.Create(ctx, oldPod)) + _, found := metadataLogDir() + require.False(t, found) + + // No ConfigMap yet (rack awareness creates it after the pod): creation is allowed. + require.NoError(t, r.ensureMetadataLogDirConfigured(ctx, "1", broker.BrokerConfig)) + configMap := &corev1.ConfigMap{ + ObjectMeta: metav1.ObjectMeta{Name: "test-config-1", Namespace: "test"}, + Data: map[string]string{kafkautils.ConfigPropertyName: "log.dirs=/csi-kafka-logs1/kafka\n"}, + } + require.NoError(t, r.Create(ctx, configMap)) + err := r.ensureMetadataLogDirConfigured(ctx, "1", broker.BrokerConfig) + require.Error(t, err) + require.Contains(t, err.Error(), "metadata.log.dir") + require.NoError(t, r.ensureMetadataLogDirConfigured(ctx, "1", &v1beta1.BrokerConfig{})) + + // Once the old pod is gone, the replacement is created with the setting. + require.NoError(t, r.Delete(ctx, oldPod)) + value, found := metadataLogDir() + require.True(t, found) + require.Equal(t, "/csi-kafka-metadata/kafka", value) + configMap.Data[kafkautils.ConfigPropertyName] = "log.dirs=/csi-kafka-logs1/kafka\nmetadata.log.dir=" + value + "\n" + require.NoError(t, r.Update(ctx, configMap)) + require.NoError(t, r.ensureMetadataLogDirConfigured(ctx, "1", broker.BrokerConfig)) + + // Once published it is never withdrawn, even if pod listing would defer it. + stray := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "test-1-stray", Namespace: "test", Labels: labels}} + require.NoError(t, r.Create(ctx, stray)) + _, found = metadataLogDir() + require.True(t, found) + + // A pod that mounts the metadata volume does not defer the setting. + require.NoError(t, r.Delete(ctx, stray)) + require.NoError(t, r.Delete(ctx, configMap)) + newPod := &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Name: "test-1-new", Namespace: "test", Labels: labels}, + Spec: corev1.PodSpec{Volumes: []corev1.Volume{{Name: metadataVolumeName, VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: "test-1-metadata"}, + }}}}, + } + require.NoError(t, r.Create(ctx, newPod)) + _, found = metadataLogDir() + require.True(t, found) +} + +func TestMetadataMigrationGate(t *testing.T) { + ctx := context.Background() + labelsFor := func(id string) map[string]string { + return apiutil.MergeLabels(apiutil.LabelsForKafka("test"), map[string]string{v1beta1.BrokerIdLabelKey: id}) + } + oldPod := func(id string) *corev1.Pod { + return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "test-" + id, Namespace: "test", Labels: labelsFor(id)}} + } + migratedPod := func(id string) *corev1.Pod { + pod := oldPod(id) + pod.Spec.Volumes = []corev1.Volume{{Name: metadataVolumeName, VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: "test-" + id + "-metadata"}, + }}} + return pod + } + for _, tc := range []struct { + name string + setup func(*Reconciler) + pods []*corev1.Pod + outOfSync []int32 + blocked bool + }{ + {name: "no migration in progress", pods: []*corev1.Pod{oldPod("2"), oldPod("3")}}, + {name: "migration already started in this pass", pods: []*corev1.Pod{oldPod("2"), oldPod("3")}, + setup: func(r *Reconciler) { r.metadataMigrationStarted = true }, blocked: true}, + {name: "replacement pod not Ready yet", pods: []*corev1.Pod{migratedPod("2"), oldPod("3")}, blocked: true}, + {name: "previous migration Ready", pods: []*corev1.Pod{migratedPod("2"), oldPod("3")}, + setup: func(r *Reconciler) { + r.KafkaCluster.Status.BrokersState = map[string]v1beta1.BrokerState{"2": {MetadataStorageState: v1beta1.MetadataStorageReady}} + }}, + {name: "old pod deleted, replacement missing", pods: []*corev1.Pod{oldPod("3")}, blocked: true}, + {name: "old pod terminating", pods: []*corev1.Pod{oldPod("2"), func() *corev1.Pod { + pod := oldPod("3") + pod.Finalizers = []string{"test"} + pod.DeletionTimestamp = &metav1.Time{} + return pod + }()}, blocked: true}, + {name: "broker without metadata storage is ignored", pods: []*corev1.Pod{oldPod("2")}, + setup: func(r *Reconciler) { + r.KafkaCluster.Spec.Brokers[2].BrokerConfig = &v1beta1.BrokerConfig{Roles: []string{"broker"}} + }}, + {name: "another broker out of sync", pods: []*corev1.Pod{oldPod("2"), oldPod("3")}, outOfSync: []int32{2}, blocked: true}, + {name: "only the migrating broker out of sync", pods: []*corev1.Pod{oldPod("2"), oldPod("3")}, outOfSync: []int32{1}}, + } { + t.Run(tc.name, func(t *testing.T) { + r, broker := metadataTestReconciler(t) + r.KafkaCluster.Spec.Brokers = []v1beta1.Broker{broker, {Id: 2, BrokerConfig: broker.BrokerConfig.DeepCopy()}, {Id: 3, BrokerConfig: broker.BrokerConfig.DeepCopy()}} + if tc.setup != nil { + tc.setup(r) + } + current := oldPod("1") + require.NoError(t, r.Create(ctx, current)) + for _, pod := range tc.pods { + deleting := pod.DeletionTimestamp + pod.DeletionTimestamp = nil + require.NoError(t, r.Create(ctx, pod)) + if deleting != nil { + require.NoError(t, r.Delete(ctx, pod)) + } + } + kafkaClient := mocks.NewMockKafkaClient(gomock.NewController(t)) + kafkaClient.EXPECT().AllOfflineReplicas().Return(nil, nil).AnyTimes() + kafkaClient.EXPECT().OutOfSyncReplicas().Return(tc.outOfSync, nil).AnyTimes() + provider := new(kafkaclient.MockedProvider) + provider.On("NewFromCluster", r.Client, r.KafkaCluster).Return(kafkaClient, func() {}, nil) + r.kafkaClientProvider = provider + + err := r.metadataMigrationGate(ctx, current) + if tc.blocked { + require.Error(t, err) + require.True(t, errors.As(err, &errorfactory.ReconcileRollingUpgrade{})) + } else { + require.NoError(t, err) + } + }) + } +} + +func TestMetadataMigrationRestartSerializedWithinPass(t *testing.T) { + r, broker := metadataTestReconciler(t) + ctx := context.Background() + r.KafkaCluster.Spec.Brokers = []v1beta1.Broker{broker, {Id: 2, BrokerConfig: broker.BrokerConfig.DeepCopy()}} + kafkaClient := mocks.NewMockKafkaClient(gomock.NewController(t)) + kafkaClient.EXPECT().AllOfflineReplicas().Return(nil, nil).AnyTimes() + kafkaClient.EXPECT().OutOfSyncReplicas().Return(nil, nil).AnyTimes() + provider := new(kafkaclient.MockedProvider) + provider.On("NewFromCluster", r.Client, r.KafkaCluster).Return(kafkaClient, func() {}, nil) + r.kafkaClientProvider = provider + + podFor := func(id string, metadata bool) *corev1.Pod { + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "test-" + id, Namespace: "test", + Labels: apiutil.MergeLabels(apiutil.LabelsForKafka("test"), map[string]string{v1beta1.BrokerIdLabelKey: id})}} + if metadata { + pod.Spec.Volumes = []corev1.Volume{{Name: metadataVolumeName, VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: "test-" + id + "-metadata"}, + }}} + } + return pod + } + require.True(t, isMetadataMigrationRestart(podFor("1", false), podFor("1", true))) + require.False(t, isMetadataMigrationRestart(podFor("1", true), podFor("1", true))) + require.False(t, isMetadataMigrationRestart(podFor("1", false), podFor("1", false))) + + // Crashed containers bypass the generic rolling-upgrade gates, not the migration gate. + crashed := func(pod *corev1.Pod) *corev1.Pod { + pod.Status.ContainerStatuses = []corev1.ContainerStatus{{Name: "kafka", State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: 1}}}} + return pod + } + first, second := crashed(podFor("1", false)), crashed(podFor("2", false)) + require.NoError(t, r.Create(ctx, first)) + require.NoError(t, r.Create(ctx, second)) + require.NoError(t, r.handleRollingUpgrade(logr.Discard(), podFor("1", true), first, reflect.TypeOf(first))) + require.True(t, r.metadataMigrationStarted) + err := r.handleRollingUpgrade(logr.Discard(), podFor("2", true), second, reflect.TypeOf(second)) + require.Error(t, err) + require.True(t, errors.As(err, &errorfactory.ReconcileRollingUpgrade{})) + require.NoError(t, r.Get(ctx, client.ObjectKeyFromObject(second), &corev1.Pod{})) +} diff --git a/pkg/resources/kafka/migrate-broker-metadata.sh b/pkg/resources/kafka/migrate-broker-metadata.sh new file mode 100644 index 000000000..5931476d6 --- /dev/null +++ b/pkg/resources/kafka/migrate-broker-metadata.sh @@ -0,0 +1,251 @@ +# Copyright 2026 Adobe. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Runs only in a broker-only init container, after the previous pod terminates. +# Source topic replicas and data-directory meta.properties are never changed. +set -euo pipefail +KAFKA_HOME=${KAFKA_HOME:-/opt/kafka} +BROKER_CONFIG=${BROKER_CONFIG:-/config/broker-config} +WORK_DIR=${WORK_DIR:-/var/run/wait} + +fail() { echo "KRaft metadata migration: $*" >&2; exit 1; } +# Phase progress on stdout so a slow copy is distinguishable from a hung one. +log() { echo "KRaft metadata migration [broker ${NODE_ID:-?}]: $*"; } +property() { + local key=$1 file=$2 line value="" count=0 + [[ -f "$file" && ! -L "$file" ]] || fail "missing or symlinked properties: $file" + while IFS= read -r line || [[ -n "$line" ]]; do + if [[ "$line" == "$key="* ]]; then + value=${line#*=} + count=$((count + 1)) + fi + done < "$file" + [[ $count == 1 && -n "$value" ]] || fail "missing/duplicate $key in $file" + printf '%s' "$value" +} +meta_version() { + local file=$1/meta.properties line value=0 count=0 + [[ -f "$file" && ! -L "$file" ]] || fail "missing or symlinked properties: $file" + while IFS= read -r line || [[ -n "$line" ]]; do + if [[ "$line" == version=* ]]; then + value=${line#*=} + count=$((count + 1)) + fi + done < "$file" + # Kafka treats a missing version as V0. + [[ $count -le 1 ]] || fail "duplicate version in $file" + printf '%s' "$value" +} +# Destination/staging directories are always formatted by Kafka as V1. +identity() { + [[ "$(property cluster.id "$1/meta.properties")" == "$CLUSTER_ID" ]] || fail "wrong cluster ID in $1" + [[ "$(property node.id "$1/meta.properties")" == "$NODE_ID" ]] || fail "wrong node ID in $1" + [[ "$(meta_version "$1")" == 1 ]] || fail "not KRaft storage: $1" +} +# Data directories of brokers migrated from ZooKeeper keep V0 (broker.id). +source_identity() { + [[ "$(property cluster.id "$1/meta.properties")" == "$CLUSTER_ID" ]] || fail "wrong cluster ID in $1" + case "$(meta_version "$1")" in + 0) [[ "$(property broker.id "$1/meta.properties")" == "$NODE_ID" ]] || fail "wrong broker ID in $1" ;; + 1) [[ "$(property node.id "$1/meta.properties")" == "$NODE_ID" ]] || fail "wrong node ID in $1" ;; + *) fail "unsupported meta.properties version in $1" ;; + esac +} +reject_links() { + local entry + for entry in "$1"/* "$1"/.[!.]* "$1"/..?*; do + [[ ! -L "$entry" ]] || fail "symlinked metadata payload: $entry" + if [[ -d "$entry" ]]; then reject_links "$entry"; fi + done +} +safe_path() { + [[ "$1" =~ ^/[a-zA-Z0-9_./-]+$ && "$1" != "/" && "$1" != */ && "$1" != *//* && "$1" != */../* && "$1" != */./* && "$1" != */.. && "$1" != */. ]] || fail "unsafe mount path: $1" +} +[[ -n "${CLUSTER_ID:-}" && "${NODE_ID:-}" =~ ^[0-9]+$ ]] || fail "missing cluster/node identity" +[[ "$(property process.roles "$BROKER_CONFIG")" == broker ]] || fail "not a broker-only configuration" +[[ "$(property node.id "$BROKER_CONFIG")" == "$NODE_ID" ]] || fail "node/config disagreement" +safe_path "$METADATA_MOUNT" +log "validating storage (metadata=$METADATA_MOUNT, data=${DATA_MOUNTS:-})" +dest="$METADATA_MOUNT/kafka" +stage="$METADATA_MOUNT/.koperator-metadata-stage" +manifest="$METADATA_MOUNT/.koperator-metadata-migration" +complete="$METADATA_MOUNT/.koperator-metadata-complete" +[[ "$(property metadata.log.dir "$BROKER_CONFIG")" == "$dest" ]] || fail "metadata/config disagreement" +[[ -d "$METADATA_MOUNT" && ! -L "$METADATA_MOUNT" ]] || fail "metadata volume is not mounted" +IFS=',' read -ra mounts <<< "$DATA_MOUNTS" +[[ ${#mounts[@]} -gt 0 ]] || fail "no mounted data directories" +sources=() +formatted=0 +directory_ids=("not-a-directory-id") +for mount in "${mounts[@]}"; do + safe_path "$mount" + [[ "$mount" != "$METADATA_MOUNT" && "$mount" != "$METADATA_MOUNT/"* && "$METADATA_MOUNT" != "$mount/"* ]] || fail "overlapping metadata/data mounts" + [[ -d "$mount" && ! -L "$mount" ]] || fail "data volume is not mounted: $mount" + root="$mount/kafka" + [[ ! -L "$root" && ! -L "$root/__cluster_metadata-0" ]] || fail "symlinked source" + if [[ -f "$root/meta.properties" ]]; then + source_identity "$root" + formatted=$((formatted + 1)) + # Older source directories may predate Kafka's directory.id field. + while IFS= read -r line; do + [[ "$line" != directory.id=* ]] || directory_ids+=("${line#*=}") + done < "$root/meta.properties" + fi + if [[ -e "$root/__cluster_metadata-0" ]]; then + [[ -d "$root/__cluster_metadata-0" ]] || fail "source is not a directory" + source_identity "$root" + sources+=("$mount") + fi +done +[[ ${#sources[@]} -le 1 ]] || fail "multiple active metadata sources" +log "found ${#sources[@]} active metadata source(s) across ${#mounts[@]} data dir(s), $formatted formatted" + +if [[ -f "$manifest" ]]; then + [[ ! -L "$manifest" ]] || fail "symlinked migration manifest" + [[ "$(property cluster.id "$manifest")" == "$CLUSTER_ID" && "$(property node.id "$manifest")" == "$NODE_ID" ]] || fail "migration identity conflict" + source=$(property source "$manifest") + log "resuming migration from source=$source" +else + [[ ! -e "$dest" && ! -e "$stage" && ! -e "$complete" ]] || fail "unowned destination/staging storage" + source=fresh + if [[ ${#sources[@]} == 1 ]]; then + source=${sources[0]} + elif [[ $formatted != 0 ]]; then + fail "formatted broker has no metadata source; refusing empty recovery" + elif [[ "${ALLOW_FRESH:-false}" != true ]]; then + fail "existing broker has no metadata source; refusing empty recovery" + fi + log "starting migration from source=$source" + printf 'cluster.id=%s\nnode.id=%s\nsource=%s\n' "$CLUSTER_ID" "$NODE_ID" "$source" > "$manifest.pending" + sync + mv "$manifest.pending" "$manifest" + sync +fi + +if [[ "$source" != fresh ]]; then + safe_path "$source" + backup="$source/.koperator-metadata-backup-$NODE_ID" +fi +if [[ ${#sources[@]} == 1 ]]; then + [[ "$source" == "${sources[0]}" && ! -e "$complete" ]] || fail "active source conflicts with migration state" + [[ ! -e "$backup" ]] || fail "both active metadata and backup exist" +fi + +if [[ ! -d "$dest" ]]; then + [[ ! -e "$dest" && ! -e "$complete" ]] || fail "destination missing or not a directory" + if [[ "$source" != fresh ]]; then + [[ ${#sources[@]} == 1 ]] || fail "source lost before publishing" + reject_links "$source/kafka/__cluster_metadata-0" + # A real cache has a segment or snapshot, not just an empty directory. + payload=false + for file in "$source/kafka/__cluster_metadata-0/"*.log "$source/kafka/__cluster_metadata-0/"*.checkpoint; do + [[ ! -f "$file" ]] || payload=true + done + [[ "$payload" == true ]] || fail "empty source metadata log" + # Brokers migrated from ZooKeeper have no bootstrap.checkpoint. + [[ ! -L "$source/kafka/bootstrap.checkpoint" ]] || fail "symlinked source bootstrap.checkpoint" + else + [[ ${#sources[@]} == 0 && $formatted == 0 ]] || fail "fresh migration conflicts with existing storage" + fi + mkdir -p "$stage" "$WORK_DIR" + [[ ! -L "$stage" ]] || fail "symlinked staging directory" + stage_owner="$stage/.koperator-metadata-stage-owner" + if [[ ! -e "$stage_owner" ]]; then + # An interruption before writing the first ownership marker can leave an + # empty directory or this single pending file, but cannot leave payload. + for entry in "$stage"/* "$stage"/.[!.]* "$stage"/..?*; do + if [[ -e "$entry" || -L "$entry" ]]; then + [[ "$entry" == "$stage_owner.pending" && ! -L "$entry" ]] || fail "unowned staging content" + fi + done + cp "$manifest" "$stage_owner.pending" + sync + mv "$stage_owner.pending" "$stage_owner" + sync + fi + [[ ! -L "$stage_owner" ]] || fail "symlinked staging ownership" + cmp "$manifest" "$stage_owner" || fail "staging belongs to another migration" + format_config="$WORK_DIR/metadata-format.config" + trap 'rm -f "$format_config"' EXIT + # Format only the staging directory. Kafka supplies a NEW directory.id. + # The old data directory's meta.properties must never be copied. + while IFS= read -r line || [[ -n "$line" ]]; do + case "$line" in log.dirs=*|log.dir=*|metadata.log.dir=*) ;; *) printf '%s\n' "$line";; esac + done < "$BROKER_CONFIG" > "$format_config" + printf 'log.dirs=%s\nmetadata.log.dir=%s\n' "$stage" "$stage" >> "$format_config" + if [[ -f "$stage/meta.properties" && ! -f "$stage/bootstrap.checkpoint" ]]; then + # An interrupted formatter has not yet published anything or touched the source. + rm "$stage/meta.properties" + fi + log "formatting staging directory $stage" + "$KAFKA_HOME/bin/kafka-storage.sh" format --cluster-id="$CLUSTER_ID" --ignore-formatted -c "$format_config" + identity "$stage" + [[ -f "$stage/bootstrap.checkpoint" ]] || fail "formatter did not write bootstrap.checkpoint" + new_id=$(property directory.id "$stage/meta.properties") + [[ "$new_id" =~ ^[a-zA-Z0-9_-]{22}$ ]] || fail "Kafka formatter lacks a valid directory.id (Kafka 3.7+ required)" + for old_id in "${directory_ids[@]}"; do + [[ "$new_id" != "$old_id" ]] || fail "destination reuses a data directory ID" + done + if [[ "$source" != fresh ]]; then + log "copying metadata log from $source/kafka/__cluster_metadata-0" + mkdir -p "$stage/__cluster_metadata-0" + cp -a "$source/kafka/__cluster_metadata-0/." "$stage/__cluster_metadata-0/" + # Relocate, do not transform: the destination mirrors the source bootstrap. + if [[ -f "$source/kafka/bootstrap.checkpoint" ]]; then + cp -a "$source/kafka/bootstrap.checkpoint" "$stage/bootstrap.checkpoint" + else + rm "$stage/bootstrap.checkpoint" + fi + fi + log "publishing $dest" + cp "$manifest" "$stage/.koperator-metadata-owner" + sync + mv "$stage" "$dest" + sync +else + log "destination $dest already published" +fi +log "verifying published destination" +[[ ! -L "$dest" ]] || fail "symlinked destination" +[[ ! -e "$stage" ]] || fail "both staging and published destination exist" +identity "$dest" +[[ -f "$dest/.koperator-metadata-owner" && ! -L "$dest/.koperator-metadata-owner" ]] || fail "destination ownership missing or symlinked" +cmp "$manifest" "$dest/.koperator-metadata-owner" || fail "destination belongs to another migration" +new_id=$(property directory.id "$dest/meta.properties") +[[ "$new_id" =~ ^[a-zA-Z0-9_-]{22}$ ]] || fail "invalid destination directory ID" +for old_id in "${directory_ids[@]}"; do + [[ "$new_id" != "$old_id" ]] || fail "destination/data directory ID conflict" +done +[[ ! -L "$dest/bootstrap.checkpoint" ]] || fail "symlinked destination bootstrap" +if [[ "$source" == fresh ]]; then + [[ -f "$dest/bootstrap.checkpoint" ]] || fail "destination bootstrap missing" +fi +if [[ "$source" != fresh ]]; then + [[ -d "$dest/__cluster_metadata-0" && ! -L "$dest/__cluster_metadata-0" ]] || fail "published metadata log missing" + reject_links "$dest/__cluster_metadata-0" +fi + +if [[ ! -e "$complete" ]]; then + if [[ "$source" != fresh ]]; then + if [[ ${#sources[@]} == 1 ]]; then + log "retiring source metadata to $backup" + # Same-volume rename takes the old active metadata OUTSIDE all log.dirs. + # No topic replicas or data-directory meta.properties are moved. + mv "$source/kafka/__cluster_metadata-0" "$backup" + sync + else + [[ -d "$backup" && ! -L "$backup" ]] || fail "source missing without a retained backup" + fi + fi + log "recording completion" + cp "$manifest" "$complete.pending" + sync + mv "$complete.pending" "$complete" + sync +fi +[[ -f "$complete" && ! -L "$complete" ]] || fail "completion marker missing or symlinked" +cmp "$manifest" "$complete" || fail "completion identity conflict" +for mount in "${mounts[@]}"; do + [[ ! -e "$mount/kafka/__cluster_metadata-0" ]] || fail "active metadata remains in log.dirs" +done +echo "KRaft metadata storage ready for broker $NODE_ID" diff --git a/pkg/resources/kafka/pod.go b/pkg/resources/kafka/pod.go index ee63004af..625b09b07 100644 --- a/pkg/resources/kafka/pod.go +++ b/pkg/resources/kafka/pod.go @@ -38,18 +38,42 @@ import ( pkicommon "github.com/banzaicloud/koperator/pkg/util/pki" ) +const ( + bashCommand = "bash" + exitFileVolumeName = "exitfile" +) + var ( //go:embed wait-for-envoy-sidecar.sh envoySidecarScript string + //go:embed migrate-broker-metadata.sh + metadataMigrationScript string ) func (r *Reconciler) pod(id int32, brokerConfig *v1beta1.BrokerConfig, pvcs []corev1.PersistentVolumeClaim, log logr.Logger) runtime.Object { const kafkaContainerName = "kafka" - dataVolume, dataVolumeMount := generateDataVolumeAndVolumeMount(pvcs, brokerConfig.StorageConfigs) + dataVolume, dataVolumeMount := generateDataVolumeAndVolumeMount(dataPVCs(pvcs), brokerConfig.StorageConfigs) + dataMountPaths := make([]string, 0, len(dataVolumeMount)) + allowFreshMetadata := "false" + for _, mount := range dataVolumeMount { + dataMountPaths = append(dataMountPaths, mount.MountPath) + } + if brokerConfig.MetadataStorage != nil { + for _, pvc := range pvcs { + if isMetadataPVC(pvc) { + allowFreshMetadata = pvc.Annotations[metadataFreshAnnotation] + dataVolume = append(dataVolume, corev1.Volume{ + Name: metadataVolumeName, + VolumeSource: corev1.VolumeSource{PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: pvc.Name}}, + }) + dataVolumeMount = append(dataVolumeMount, corev1.VolumeMount{Name: metadataVolumeName, MountPath: brokerConfig.MetadataStorage.MountPath}) + } + } + } // TODO remove this bash envoy sidecar checker script once sidecar precedence becomes available to Kubernetes(baluchicken) - command := []string{"bash", "-c", envoySidecarScript} + command := []string{bashCommand, "-c", envoySidecarScript} // Updating Controller pod names to say "controller" podname := fmt.Sprintf("%s-%d-", r.KafkaCluster.Name, id) @@ -63,7 +87,7 @@ func (r *Reconciler) pod(id int32, brokerConfig *v1beta1.BrokerConfig, pvcs []co Lifecycle: &corev1.Lifecycle{ PreStop: &corev1.LifecycleHandler{ Exec: &corev1.ExecAction{ - Command: []string{"bash", "-c", ` + Command: []string{bashCommand, "-c", ` if [[ -n "$ENVOY_SIDECAR_STATUS" ]]; then HEALTHYSTATUSCODE="200" SC=$(curl -s -o /dev/null -w "%{http_code}" http://localhost:15000/ready) @@ -155,6 +179,35 @@ fi`}, PriorityClassName: brokerConfig.GetPriorityClassName(), }, } + if brokerConfig.MetadataStorage != nil { + clusterID := corev1.EnvVar{Name: clusterIDEnvVarName, Value: r.KafkaCluster.Status.ClusterID} + for _, env := range r.KafkaCluster.Spec.Envs { + if env.Name == clusterIDEnvVarName { + clusterID = env + } + } + mounts := append([]corev1.VolumeMount{}, dataVolumeMount...) + mounts = append(mounts, + corev1.VolumeMount{Name: brokerConfigMapVolumeMount, MountPath: "/config", ReadOnly: true}, + corev1.VolumeMount{Name: exitFileVolumeName, MountPath: "/var/run/wait"}) + pod.Spec.InitContainers = append(pod.Spec.InitContainers, corev1.Container{ + Name: "migrate-broker-metadata", + Image: kafkaContainer.Image, + Command: []string{bashCommand, "-c", metadataMigrationScript}, + Env: []corev1.EnvVar{ + clusterID, + {Name: "NODE_ID", Value: strconv.Itoa(int(id))}, + {Name: "METADATA_MOUNT", Value: brokerConfig.MetadataStorage.MountPath}, + {Name: "DATA_MOUNTS", Value: strings.Join(dataMountPaths, ",")}, + {Name: "ALLOW_FRESH", Value: allowFreshMetadata}, + {Name: "KAFKA_HEAP_OPTS", Value: "-Xms64m -Xmx256m"}, + }, + VolumeMounts: mounts, + SecurityContext: brokerConfig.SecurityContext, + Resources: *brokerConfig.GetResources(), + }) + pod.Spec.Containers[0].Env = append(pod.Spec.Containers[0].Env, corev1.EnvVar{Name: "METADATA_STORAGE_ENABLED", Value: configValueTrue}) + } if r.KafkaCluster.Spec.HeadlessServiceEnabled { pod.Spec.Hostname = fmt.Sprintf("%s-%d", r.KafkaCluster.Name, id) @@ -321,7 +374,7 @@ func getVolumeMounts(brokerConfigVolumeMounts, dataVolumeMount []corev1.VolumeMo MountPath: "/etc/jmx-exporter/", }, { - Name: "exitfile", + Name: exitFileVolumeName, MountPath: "/var/run/wait", }, }...) @@ -346,7 +399,7 @@ func getVolumes(brokerConfigVolumes, dataVolume []corev1.Volume, kafkaClusterSpe volumes = append(volumes, generateVolumesForListenerCerts(kafkaClusterSpec.ListenersConfig, kafkaClusterName)...) volumes = append(volumes, []corev1.Volume{ { - Name: "exitfile", + Name: exitFileVolumeName, VolumeSource: corev1.VolumeSource{ EmptyDir: &corev1.EmptyDirVolumeSource{}, }, diff --git a/pkg/resources/kafka/wait-for-envoy-sidecar.sh b/pkg/resources/kafka/wait-for-envoy-sidecar.sh index 544fc2951..e4350c589 100644 --- a/pkg/resources/kafka/wait-for-envoy-sidecar.sh +++ b/pkg/resources/kafka/wait-for-envoy-sidecar.sh @@ -37,7 +37,19 @@ if [[ -n "${CLUSTER_ID}" ]]; then # If the storage is already formatted (e.g. broker restarts), the kafka-storage.sh will skip formatting for that storage # thus we can safely run the storage format command regardless if the storage has been formatted or not echo "Formatting KRaft storage with cluster ID ${CLUSTER_ID}" - ${KAFKA_HOME}/bin/kafka-storage.sh format --cluster-id="${CLUSTER_ID}" --ignore-formatted -c /config/broker-config + if [[ "${METADATA_STORAGE_ENABLED:-}" == true ]]; then + # The init container has recovered metadata. A format failure must not + # allow Kafka to start with partially initialized data storage. + if ${KAFKA_HOME}/bin/kafka-storage.sh format --cluster-id="${CLUSTER_ID}" --ignore-formatted -c /config/broker-config; then + : + else + FORMAT_EXIT=$? + rm -f "${WAIT_DIR}/do-not-exit-yet" + exit "$FORMAT_EXIT" + fi + else + ${KAFKA_HOME}/bin/kafka-storage.sh format --cluster-id="${CLUSTER_ID}" --ignore-formatted -c /config/broker-config + fi # Adding or removing controller nodes to the Kafka cluster would trigger cluster rolling upgrade so all the nodes in the cluster are aware of the newly added/removed controllers. # When this happens, Kafka's local quorum state file would be outdated since it is static and the Kafka server can't be started with conflicting controllers info (compared to info stored in ConfigMap), diff --git a/pkg/util/kafka/const.go b/pkg/util/kafka/const.go index 501f0ce7b..5b4e3ca51 100644 --- a/pkg/util/kafka/const.go +++ b/pkg/util/kafka/const.go @@ -25,8 +25,9 @@ const ( KafkaConfigBoostrapServers = "bootstrap.servers" KafkaConfigZooKeeperConnect = "zookeeper.connect" // KafkaConfigBrokerID is used in ZooKeeper mode - KafkaConfigBrokerID = "broker.id" - KafkaConfigBrokerLogDirectory = "log.dirs" + KafkaConfigBrokerID = "broker.id" + KafkaConfigBrokerLogDirectory = "log.dirs" + KafkaConfigMetadataLogDirectory = "metadata.log.dir" // Configuration keys for KRaft KafkaConfigNodeID = "node.id" diff --git a/pkg/webhooks/kafkacluster_validator.go b/pkg/webhooks/kafkacluster_validator.go index 486db357e..8325440f9 100644 --- a/pkg/webhooks/kafkacluster_validator.go +++ b/pkg/webhooks/kafkacluster_validator.go @@ -18,6 +18,8 @@ package webhooks import ( "context" "fmt" + "reflect" + "strconv" corev1 "k8s.io/api/core/v1" @@ -30,14 +32,17 @@ import ( banzaicloudv1beta1 "github.com/banzaicloud/koperator/api/v1beta1" "github.com/banzaicloud/koperator/pkg/util" + kafkautils "github.com/banzaicloud/koperator/pkg/util/kafka" + properties "github.com/banzaicloud/koperator/properties/pkg" ) type KafkaClusterValidator struct { Log logr.Logger } -func (s KafkaClusterValidator) ValidateUpdate(ctx context.Context, _, kafkaClusterNew *banzaicloudv1beta1.KafkaCluster) (warnings admission.Warnings, err error) { +func (s KafkaClusterValidator) ValidateUpdate(ctx context.Context, kafkaClusterOld, kafkaClusterNew *banzaicloudv1beta1.KafkaCluster) (warnings admission.Warnings, err error) { var allErrs field.ErrorList + allErrs = append(allErrs, checkMetadataStorage(kafkaClusterNew, kafkaClusterOld)...) log := s.Log.WithValues("name", kafkaClusterNew.GetName(), "namespace", kafkaClusterNew.GetNamespace()) listenerErrs := checkInternalAndExternalListeners(&kafkaClusterNew.Spec) @@ -57,6 +62,7 @@ func (s KafkaClusterValidator) ValidateUpdate(ctx context.Context, _, kafkaClust func (s KafkaClusterValidator) ValidateCreate(ctx context.Context, kafkaCluster *banzaicloudv1beta1.KafkaCluster) (warnings admission.Warnings, err error) { var allErrs field.ErrorList + allErrs = append(allErrs, checkMetadataStorage(kafkaCluster, nil)...) log := s.Log.WithValues("name", kafkaCluster.GetName(), "namespace", kafkaCluster.GetNamespace()) listenerErrs := checkInternalAndExternalListeners(&kafkaCluster.Spec) @@ -67,13 +73,95 @@ func (s KafkaClusterValidator) ValidateCreate(ctx context.Context, kafkaCluster if len(allErrs) == 0 { return nil, nil } - log.Info("rejected", "invalid field(s)", allErrs.ToAggregate().Error()) return nil, apierrors.NewInvalid( kafkaCluster.GroupVersionKind().GroupKind(), kafkaCluster.Name, allErrs) } +func checkMetadataStorage(cluster, old *banzaicloudv1beta1.KafkaCluster) field.ErrorList { + if !hasMetadataStorage(cluster) && (old == nil || !hasMetadataStorage(old)) { + return nil + } + var errs field.ErrorList + for i, broker := range cluster.Spec.Brokers { + p := field.NewPath("spec", "brokers").Index(i).Child("brokerConfig", "metadataStorage") + config, err := broker.GetBrokerConfig(cluster.Spec) + if err == nil { + err = config.ValidateMetadataStorage(cluster.Spec.KRaftMode) + } + if err != nil { + errs = append(errs, field.Invalid(p, nil, err.Error())) + continue + } + if config == nil { + config = &banzaicloudv1beta1.BrokerConfig{} + } + if config.MetadataStorage != nil { + props, parseErr := properties.NewFromString(cluster.Spec.ReadOnlyConfig + "\n" + broker.ReadOnlyConfig + "\n" + config.Config) + if parseErr != nil { + errs = append(errs, field.Invalid(p, nil, parseErr.Error())) + } else if mode, found := props.Get(kafkautils.MigrationBrokerKRaftMode); found && mode.Value() != "true" { + errs = append(errs, field.Invalid(p, nil, "metadataStorage cannot be used by a ZooKeeper migration broker")) + } + } + if old == nil { + continue + } + for _, previous := range old.Spec.Brokers { + if previous.Id != broker.Id { + continue + } + before, err := previous.GetBrokerConfig(old.Spec) + if err != nil { + errs = append(errs, field.Invalid(p, nil, err.Error())) + } else if before != nil && before.MetadataStorage != nil && + !banzaicloudv1beta1.MetadataStorageLocationEqual(before.MetadataStorage, config.MetadataStorage) { + errs = append(errs, field.Forbidden(p, "enabled metadataStorage cannot be removed or relocated; reverse migration is unsupported")) + } + if before != nil && before.MetadataStorage == nil && config.MetadataStorage != nil && + !reflect.DeepEqual(before.StorageConfigs, config.StorageConfigs) { + errs = append(errs, field.Forbidden(p, "enable metadataStorage without changing existing data storageConfigs")) + } + if before != nil && before.MetadataStorage != nil && config.MetadataStorage != nil && + removesDataDisk(before.StorageConfigs, config.StorageConfigs) && + old.Status.BrokersState[strconv.Itoa(int(broker.Id))].MetadataStorageState != banzaicloudv1beta1.MetadataStorageReady { + errs = append(errs, field.Forbidden(field.NewPath("spec", "brokers").Index(i).Child("brokerConfig", "storageConfigs"), + "data disks cannot be removed until status.brokersState metadataStorageState of this broker is Ready; "+ + "the disks may still hold the only copy of the broker metadata")) + } + } + } + return errs +} + +func removesDataDisk(before, after []banzaicloudv1beta1.StorageConfig) bool { + kept := make(map[string]struct{}, len(after)) + for _, storage := range after { + kept[storage.MountPath] = struct{}{} + } + for _, storage := range before { + if _, ok := kept[storage.MountPath]; !ok { + return true + } + } + return false +} + +func hasMetadataStorage(cluster *banzaicloudv1beta1.KafkaCluster) bool { + for _, group := range cluster.Spec.BrokerConfigGroups { + if group.MetadataStorage != nil { + return true + } + } + for _, broker := range cluster.Spec.Brokers { + if broker.BrokerConfig != nil && broker.BrokerConfig.MetadataStorage != nil { + return true + } + } + return false +} + func (s KafkaClusterValidator) ValidateDelete(_ context.Context, _ *banzaicloudv1beta1.KafkaCluster) (warnings admission.Warnings, err error) { return nil, nil } diff --git a/pkg/webhooks/metadata_storage_test.go b/pkg/webhooks/metadata_storage_test.go new file mode 100644 index 000000000..a27bdbd16 --- /dev/null +++ b/pkg/webhooks/metadata_storage_test.go @@ -0,0 +1,134 @@ +// Copyright 2026 Adobe. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package webhooks + +import ( + "context" + "testing" + + "github.com/go-logr/logr" + "github.com/stretchr/testify/require" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + + "github.com/banzaicloud/koperator/api/v1beta1" +) + +func TestMetadataStorageAdmission(t *testing.T) { + cluster := &v1beta1.KafkaCluster{Spec: v1beta1.KafkaClusterSpec{ + KRaftMode: true, + BrokerConfigGroups: map[string]v1beta1.BrokerConfig{"brokers": { + Roles: []string{"broker"}, + StorageConfigs: []v1beta1.StorageConfig{{MountPath: "/csi-kafka-logs1", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}}, + MetadataStorage: &v1beta1.StorageConfig{MountPath: "/csi-kafka-metadata", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}, + }}, + Brokers: []v1beta1.Broker{{Id: 1, BrokerConfigGroup: "brokers"}}, + }} + validator := KafkaClusterValidator{Log: logr.Discard()} + _, err := validator.ValidateCreate(context.Background(), cluster) + require.NoError(t, err) + for _, tc := range []struct { + name string + change func(*v1beta1.KafkaCluster) + }{ + {"removal", func(c *v1beta1.KafkaCluster) { + b := c.Spec.BrokerConfigGroups["brokers"] + b.MetadataStorage = nil + c.Spec.BrokerConfigGroups["brokers"] = b + }}, + {"relocation", func(c *v1beta1.KafkaCluster) { + b := c.Spec.BrokerConfigGroups["brokers"] + b.MetadataStorage.MountPath = "/new" + c.Spec.BrokerConfigGroups["brokers"] = b + }}, + {"role change", func(c *v1beta1.KafkaCluster) { + b := c.Spec.BrokerConfigGroups["brokers"] + b.Roles = []string{"controller"} + c.Spec.BrokerConfigGroups["brokers"] = b + }}, + {"zk mode", func(c *v1beta1.KafkaCluster) { c.Spec.KRaftMode = false }}, + {"zk migration", func(c *v1beta1.KafkaCluster) { c.Spec.Brokers[0].ReadOnlyConfig = "migration.broker.kRaftMode=false" }}, + } { + t.Run(tc.name, func(t *testing.T) { + next := cluster.DeepCopy() + tc.change(next) + _, err := validator.ValidateUpdate(context.Background(), cluster, next) + require.Error(t, err) + }) + } + grown := cluster.DeepCopy() + b := grown.Spec.BrokerConfigGroups["brokers"] + b.MetadataStorage.PvcSpec.Resources.Requests = corev1.ResourceList{corev1.ResourceStorage: resource.MustParse("20Gi")} + grown.Spec.BrokerConfigGroups["brokers"] = b + _, err = validator.ValidateUpdate(context.Background(), cluster, grown) + require.NoError(t, err) + old := cluster.DeepCopy() + b = old.Spec.BrokerConfigGroups["brokers"] + b.MetadataStorage = nil + old.Spec.BrokerConfigGroups["brokers"] = b + _, err = validator.ValidateUpdate(context.Background(), old, cluster) + require.NoError(t, err) + changedData := cluster.DeepCopy() + b = changedData.Spec.BrokerConfigGroups["brokers"] + b.StorageConfigs[0].MountPath = "/another-data-volume" + changedData.Spec.BrokerConfigGroups["brokers"] = b + _, err = validator.ValidateUpdate(context.Background(), old, changedData) + require.Error(t, err) +} + +func TestMetadataStorageDataDiskRemovalAdmission(t *testing.T) { + cluster := &v1beta1.KafkaCluster{Spec: v1beta1.KafkaClusterSpec{ + KRaftMode: true, + Brokers: []v1beta1.Broker{{Id: 1, BrokerConfig: &v1beta1.BrokerConfig{ + Roles: []string{"broker"}, + StorageConfigs: []v1beta1.StorageConfig{ + {MountPath: "/csi-kafka-logs1", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}, + {MountPath: "/csi-kafka-logs2", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}, + }, + MetadataStorage: &v1beta1.StorageConfig{MountPath: "/csi-kafka-metadata", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}, + }}}, + }} + validator := KafkaClusterValidator{Log: logr.Discard()} + removed := cluster.DeepCopy() + removed.Spec.Brokers[0].BrokerConfig.StorageConfigs = removed.Spec.Brokers[0].BrokerConfig.StorageConfigs[1:] + added := cluster.DeepCopy() + added.Spec.Brokers[0].BrokerConfig.StorageConfigs = append(added.Spec.Brokers[0].BrokerConfig.StorageConfigs, + v1beta1.StorageConfig{MountPath: "/csi-kafka-logs3", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}) + + _, err := validator.ValidateUpdate(context.Background(), cluster, removed) + require.ErrorContains(t, err, "metadataStorageState") + _, err = validator.ValidateUpdate(context.Background(), cluster, added) + require.NoError(t, err) + + ready := cluster.DeepCopy() + ready.Status.BrokersState = map[string]v1beta1.BrokerState{"1": {MetadataStorageState: v1beta1.MetadataStorageReady}} + _, err = validator.ValidateUpdate(context.Background(), ready, removed) + require.NoError(t, err) + + otherReady := cluster.DeepCopy() + otherReady.Status.BrokersState = map[string]v1beta1.BrokerState{"2": {MetadataStorageState: v1beta1.MetadataStorageReady}} + _, err = validator.ValidateUpdate(context.Background(), otherReady, removed) + require.Error(t, err) + + _, err = validator.ValidateCreate(context.Background(), &v1beta1.KafkaCluster{Spec: v1beta1.KafkaClusterSpec{ + KRaftMode: true, + Brokers: []v1beta1.Broker{{Id: 1, BrokerConfig: &v1beta1.BrokerConfig{ + Roles: []string{"broker"}, + StorageConfigs: []v1beta1.StorageConfig{{MountPath: "/kafka-logs", EmptyDir: &corev1.EmptyDirVolumeSource{}}}, + MetadataStorage: &v1beta1.StorageConfig{MountPath: "/csi-kafka-metadata", PvcSpec: &corev1.PersistentVolumeClaimSpec{}}, + }}}, + }}) + require.ErrorContains(t, err, "PVC-backed") +} diff --git a/scripts/test-broker-metadata-migration.sh b/scripts/test-broker-metadata-migration.sh new file mode 100644 index 000000000..0365e8404 --- /dev/null +++ b/scripts/test-broker-metadata-migration.sh @@ -0,0 +1,129 @@ +#!/usr/bin/env bash +# Copyright 2026 Adobe. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +set -euo pipefail + +KAFKA_HOME=${KAFKA_HOME:-/opt/kafka} +MIGRATOR=${MIGRATOR:-/test/migrate-broker-metadata.sh} +export KAFKA_HEAP_OPTS="-Xms128m -Xmx256m" +controller_pid="" +broker_pid="" + +cleanup() { + for pid in "$broker_pid" "$controller_pid"; do + if [[ -n "$pid" ]] && kill -0 "$pid" 2>/dev/null; then + kill "$pid" + wait "$pid" || echo "Kafka process $pid terminated during cleanup" >&2 + fi + done +} +trap cleanup EXIT + +stop_broker() { + kill "$broker_pid" + if wait "$broker_pid"; then + : + else + code=$? + [[ "$code" == 143 ]] || return "$code" + fi + broker_pid="" +} + +write_broker_config() { + cat > "$root/broker.properties" < "$root/topics" 2> "$root/client-error"; then + return + fi + if ! kill -0 "$broker_pid" 2>/dev/null; then + cat "$root/broker.log" >&2 + return 1 + fi + sleep 1 + done + cat "$root/broker.log" "$root/client-error" >&2 + return 1 +} + +for source in logs1 logs2; do + root=$(mktemp -d /tmp/koperator-metadata-smoke.XXXXXX) + mkdir -p "$root/logs1" "$root/logs2" "$root/metadata" "$root/wait" + cluster_id=$("$KAFKA_HOME/bin/kafka-storage.sh" random-uuid) + printf 'request.timeout.ms=3000\ndefault.api.timeout.ms=5000\n' > "$root/client.properties" + cat > "$root/controller.properties" < "$root/controller.log" 2>&1 & + controller_pid=$! + "$KAFKA_HOME/bin/kafka-server-start.sh" "$root/broker.properties" > "$root/broker.log" 2>&1 & + broker_pid=$! + wait_for_broker + "$KAFKA_HOME/bin/kafka-topics.sh" --bootstrap-server 127.0.0.1:19092 \ + --create --topic migration-smoke --partitions 1 --replication-factor 1 + printf 'before-migration\n' | "$KAFKA_HOME/bin/kafka-console-producer.sh" \ + --bootstrap-server 127.0.0.1:19092 --topic migration-smoke + stop_broker + + write_broker_config "$root/metadata/kafka" + CLUSTER_ID="$cluster_id" NODE_ID=1 METADATA_MOUNT="$root/metadata" \ + DATA_MOUNTS="$root/logs1,$root/logs2" BROKER_CONFIG="$root/broker.properties" \ + WORK_DIR="$root/wait" ALLOW_FRESH=false bash "$MIGRATOR" + [[ -d "$root/$source/.koperator-metadata-backup-1" ]] + [[ ! -e "$root/logs1/kafka/__cluster_metadata-0" && ! -e "$root/logs2/kafka/__cluster_metadata-0" ]] + + for restart in 1 2; do + CLUSTER_ID="$cluster_id" NODE_ID=1 METADATA_MOUNT="$root/metadata" \ + DATA_MOUNTS="$root/logs1,$root/logs2" BROKER_CONFIG="$root/broker.properties" \ + WORK_DIR="$root/wait" ALLOW_FRESH=false bash "$MIGRATOR" + "$KAFKA_HOME/bin/kafka-server-start.sh" "$root/broker.properties" > "$root/broker.log" 2>&1 & + broker_pid=$! + wait_for_broker + "$KAFKA_HOME/bin/kafka-console-consumer.sh" --bootstrap-server 127.0.0.1:19092 \ + --topic migration-smoke --from-beginning --max-messages 1 --timeout-ms 15000 \ + > "$root/consumed" + [[ "$(cat "$root/consumed")" == before-migration ]] + "$KAFKA_HOME/bin/kafka-metadata-quorum.sh" --bootstrap-server 127.0.0.1:19092 describe --status + stop_broker + done + kill "$controller_pid" + if wait "$controller_pid"; then + : + else + code=$? + [[ "$code" == 143 ]] || exit "$code" + fi + controller_pid="" + echo "Kafka metadata migration smoke test passed: $source" +done