diff --git a/CHANGELOG.md b/CHANGELOG.md index e89d30d4..2d9e38fa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,7 @@ All notable changes to this project will be documented in this file. - Support floating tags for product images via the new `spec.image.stackableVersionPolicy` field ([#831]). - Add `/ready` endpoint to the operator Deployment, which reports the CRD installation status ([#835]). +- Name nodes now have a default affinity to the OPA Pods when OPA authorization is configured ([#839]). ### Changed @@ -68,6 +69,7 @@ All notable changes to this project will be documented in this file. [#831]: https://github.com/stackabletech/hdfs-operator/pull/831 [#833]: https://github.com/stackabletech/hdfs-operator/pull/833 [#835]: https://github.com/stackabletech/hdfs-operator/pull/835 +[#839]: https://github.com/stackabletech/hdfs-operator/pull/839 ## [26.7.0] - 2026-07-21 diff --git a/docs/modules/hdfs/pages/usage-guide/operations/pod-placement.adoc b/docs/modules/hdfs/pages/usage-guide/operations/pod-placement.adoc index 9b9f52e5..32c1d718 100644 --- a/docs/modules/hdfs/pages/usage-guide/operations/pod-placement.adoc +++ b/docs/modules/hdfs/pages/usage-guide/operations/pod-placement.adoc @@ -28,6 +28,26 @@ affinity: weight: 70 ---- +If OPA authorization is configured, the name nodes additionally prefer to run on the same Kubernetes nodes as the OPA Pods (weight 50). +Only the name nodes are configured with the OPA authorizer, so data nodes and journal nodes do not get this affinity. + +[source,yaml] +---- +affinity: + podAffinity: + preferredDuringSchedulingIgnoredDuringExecution: + - podAffinityTerm: + labelSelector: + matchLabels: + app.kubernetes.io/component: server + app.kubernetes.io/instance: opa-cluster-name + app.kubernetes.io/name: opa + topologyKey: kubernetes.io/hostname + weight: 50 +---- + +`opa-cluster-name` is the `configMapName` from `spec.clusterConfig.authorization.opa`, which by convention is the name of the OpaCluster. + Default Pod placement constraints for data nodes: [source,yaml] diff --git a/rust/operator-binary/src/controller/validate.rs b/rust/operator-binary/src/controller/validate.rs index b61d90fd..00a2f863 100644 --- a/rust/operator-binary/src/controller/validate.rs +++ b/rust/operator-binary/src/controller/validate.rs @@ -97,7 +97,15 @@ pub fn validate_cluster( )?; let namenode_role_group_configs = validate_role_group_configs( hdfs.spec.name_nodes.as_ref(), - NameNodeConfigFragment::default_config(cluster_name.as_ref(), &HdfsNodeRole::Name), + NameNodeConfigFragment::default_config( + cluster_name.as_ref(), + &HdfsNodeRole::Name, + hdfs.spec + .cluster_config + .authorization + .as_ref() + .map(|authorization| &authorization.opa), + ), )?; let datanode_role_group_configs = validate_role_group_configs( hdfs.spec.data_nodes.as_ref(), diff --git a/rust/operator-binary/src/crd/affinity.rs b/rust/operator-binary/src/crd/affinity.rs index 19f7098b..49224446 100644 --- a/rust/operator-binary/src/crd/affinity.rs +++ b/rust/operator-binary/src/crd/affinity.rs @@ -1,18 +1,34 @@ use stackable_operator::{ - commons::affinity::{ - StackableAffinityFragment, affinity_between_cluster_pods, affinity_between_role_pods, + commons::{ + affinity::{ + StackableAffinityFragment, affinity_between_cluster_pods, affinity_between_role_pods, + }, + opa::OpaConfig, }, k8s_openapi::api::core::v1::{PodAffinity, PodAntiAffinity}, }; use crate::crd::{HdfsNodeRole, constants::APP_NAME}; -pub fn get_affinity(cluster_name: &str, role: &HdfsNodeRole) -> StackableAffinityFragment { +/// `opa_config` is only passed for roles that send authorization requests to OPA. +pub fn get_affinity( + cluster_name: &str, + role: &HdfsNodeRole, + opa_config: Option<&OpaConfig>, +) -> StackableAffinityFragment { + let mut pod_affinities = vec![affinity_between_cluster_pods(APP_NAME, cluster_name, 20)]; + if let Some(opa_config) = opa_config { + pod_affinities.push(affinity_between_role_pods( + "opa", + &opa_config.config_map_name, // The discovery cm has the same name as the OpaCluster itself + "server", + 50, + )); + } + StackableAffinityFragment { pod_affinity: Some(PodAffinity { - preferred_during_scheduling_ignored_during_execution: Some(vec![ - affinity_between_cluster_pods(APP_NAME, cluster_name, 20), - ]), + preferred_during_scheduling_ignored_during_execution: Some(pod_affinities), required_during_scheduling_ignored_during_execution: None, }), pod_anti_affinity: Some(PodAntiAffinity { @@ -63,6 +79,10 @@ spec: productVersion: 3.5.0 clusterConfig: zookeeperConfigMapName: hdfs-zk + authorization: + opa: + configMapName: simple-opa + package: hdfs journalNodes: roleGroups: default: @@ -80,31 +100,59 @@ spec: let validated_cluster = deserialize_and_validate_cluster(input); let merged_config = common_config(&validated_cluster, &role, &role_group_name("default")); + let mut expected_pod_affinities = vec![WeightedPodAffinityTerm { + pod_affinity_term: PodAffinityTerm { + label_selector: Some(LabelSelector { + match_expressions: None, + match_labels: Some(BTreeMap::from([ + ("app.kubernetes.io/name".to_string(), "hdfs".to_string()), + ( + "app.kubernetes.io/instance".to_string(), + "simple-hdfs".to_string(), + ), + ])), + }), + namespace_selector: None, + namespaces: None, + topology_key: "kubernetes.io/hostname".to_string(), + ..PodAffinityTerm::default() + }, + weight: 20, + }]; + // Only the NameNode is configured with the OPA authorizer. + if role == HdfsNodeRole::Name { + expected_pod_affinities.push(WeightedPodAffinityTerm { + pod_affinity_term: PodAffinityTerm { + label_selector: Some(LabelSelector { + match_expressions: None, + match_labels: Some(BTreeMap::from([ + ("app.kubernetes.io/name".to_string(), "opa".to_string()), + ( + "app.kubernetes.io/instance".to_string(), + "simple-opa".to_string(), + ), + ( + "app.kubernetes.io/component".to_string(), + "server".to_string(), + ), + ])), + }), + namespace_selector: None, + namespaces: None, + topology_key: "kubernetes.io/hostname".to_string(), + ..PodAffinityTerm::default() + }, + weight: 50, + }); + } + assert_eq!( merged_config.affinity, StackableAffinity { pod_affinity: Some(PodAffinity { - preferred_during_scheduling_ignored_during_execution: Some(vec![ - WeightedPodAffinityTerm { - pod_affinity_term: PodAffinityTerm { - label_selector: Some(LabelSelector { - match_expressions: None, - match_labels: Some(BTreeMap::from([ - ("app.kubernetes.io/name".to_string(), "hdfs".to_string(),), - ( - "app.kubernetes.io/instance".to_string(), - "simple-hdfs".to_string(), - ), - ])) - }), - namespace_selector: None, - namespaces: None, - topology_key: "kubernetes.io/hostname".to_string(), - ..PodAffinityTerm::default() - }, - weight: 20 - } - ]), + preferred_during_scheduling_ignored_during_execution: Some( + expected_pod_affinities + ), required_during_scheduling_ignored_during_execution: None, }), pod_anti_affinity: Some(PodAntiAffinity { diff --git a/rust/operator-binary/src/crd/mod.rs b/rust/operator-binary/src/crd/mod.rs index 8701670a..7a970501 100644 --- a/rust/operator-binary/src/crd/mod.rs +++ b/rust/operator-binary/src/crd/mod.rs @@ -13,6 +13,7 @@ use stackable_operator::{ commons::{ affinity::StackableAffinity, cluster_operation::ClusterOperation, + opa::OpaConfig, product_image_selection::ProductImage, resources::{ CpuLimitsFragment, MemoryLimitsFragment, NoRuntimeLimits, NoRuntimeLimitsFragment, @@ -640,7 +641,11 @@ pub struct NameNodeConfig { impl NameNodeConfigFragment { const DEFAULT_NAME_NODE_SECRET_LIFETIME: Duration = Duration::from_days_unchecked(1); - pub fn default_config(cluster_name: &str, role: &HdfsNodeRole) -> Self { + pub fn default_config( + cluster_name: &str, + role: &HdfsNodeRole, + opa_config: Option<&OpaConfig>, + ) -> Self { Self { resources: ResourcesFragment { cpu: CpuLimitsFragment { @@ -662,7 +667,7 @@ impl NameNodeConfigFragment { logging: product_logging::spec::default_logging(), listener_class: Some(DEFAULT_LISTENER_CLASS.clone()), common: CommonNodeConfigFragment { - affinity: get_affinity(cluster_name, role), + affinity: get_affinity(cluster_name, role, opa_config), graceful_shutdown_timeout: Some(DEFAULT_NAME_NODE_GRACEFUL_SHUTDOWN_TIMEOUT), requested_secret_lifetime: Some(Self::DEFAULT_NAME_NODE_SECRET_LIFETIME), }, @@ -750,7 +755,9 @@ impl DataNodeConfigFragment { logging: product_logging::spec::default_logging(), listener_class: Some(DEFAULT_LISTENER_CLASS.clone()), common: CommonNodeConfigFragment { - affinity: get_affinity(cluster_name, role), + // Only the NameNode is configured with the OPA authorizer, so this role gets no + // affinity to the OPA Pods. + affinity: get_affinity(cluster_name, role, None), graceful_shutdown_timeout: Some(DEFAULT_DATA_NODE_GRACEFUL_SHUTDOWN_TIMEOUT), requested_secret_lifetime: Some(Self::DEFAULT_DATA_NODE_SECRET_LIFETIME), }, @@ -826,7 +833,9 @@ impl JournalNodeConfigFragment { }, logging: product_logging::spec::default_logging(), common: CommonNodeConfigFragment { - affinity: get_affinity(cluster_name, role), + // Only the NameNode is configured with the OPA authorizer, so this role gets no + // affinity to the OPA Pods. + affinity: get_affinity(cluster_name, role, None), graceful_shutdown_timeout: Some(DEFAULT_JOURNAL_NODE_GRACEFUL_SHUTDOWN_TIMEOUT), requested_secret_lifetime: Some(Self::DEFAULT_JOURNAL_NODE_SECRET_LIFETIME), },