Skip to content

Commit 10bdf3e

Browse files
authored
feat: Add affinity to OPA Pods (#839)
* feat: Add affinity to OPA Pods * Updating Changelog
1 parent ea9aaee commit 10bdf3e

5 files changed

Lines changed: 119 additions & 32 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,7 @@ All notable changes to this project will be documented in this file.
99
- Support floating tags for product images via the new `spec.image.stackableVersionPolicy` field
1010
([#831]).
1111
- Add `/ready` endpoint to the operator Deployment, which reports the CRD installation status ([#835]).
12+
- Name nodes now have a default affinity to the OPA Pods when OPA authorization is configured ([#839]).
1213

1314
### Changed
1415

@@ -68,6 +69,7 @@ All notable changes to this project will be documented in this file.
6869
[#831]: https://github.com/stackabletech/hdfs-operator/pull/831
6970
[#833]: https://github.com/stackabletech/hdfs-operator/pull/833
7071
[#835]: https://github.com/stackabletech/hdfs-operator/pull/835
72+
[#839]: https://github.com/stackabletech/hdfs-operator/pull/839
7173

7274
## [26.7.0] - 2026-07-21
7375

‎docs/modules/hdfs/pages/usage-guide/operations/pod-placement.adoc‎

Lines changed: 20 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -28,6 +28,26 @@ affinity:
2828
weight: 70
2929
----
3030

31+
If OPA authorization is configured, the name nodes additionally prefer to run on the same Kubernetes nodes as the OPA Pods (weight 50).
32+
Only the name nodes are configured with the OPA authorizer, so data nodes and journal nodes do not get this affinity.
33+
34+
[source,yaml]
35+
----
36+
affinity:
37+
podAffinity:
38+
preferredDuringSchedulingIgnoredDuringExecution:
39+
- podAffinityTerm:
40+
labelSelector:
41+
matchLabels:
42+
app.kubernetes.io/component: server
43+
app.kubernetes.io/instance: opa-cluster-name
44+
app.kubernetes.io/name: opa
45+
topologyKey: kubernetes.io/hostname
46+
weight: 50
47+
----
48+
49+
`opa-cluster-name` is the `configMapName` from `spec.clusterConfig.authorization.opa`, which by convention is the name of the OpaCluster.
50+
3151
Default Pod placement constraints for data nodes:
3252

3353
[source,yaml]

‎rust/operator-binary/src/controller/validate.rs‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -97,7 +97,15 @@ pub fn validate_cluster(
9797
)?;
9898
let namenode_role_group_configs = validate_role_group_configs(
9999
hdfs.spec.name_nodes.as_ref(),
100-
NameNodeConfigFragment::default_config(cluster_name.as_ref(), &HdfsNodeRole::Name),
100+
NameNodeConfigFragment::default_config(
101+
cluster_name.as_ref(),
102+
&HdfsNodeRole::Name,
103+
hdfs.spec
104+
.cluster_config
105+
.authorization
106+
.as_ref()
107+
.map(|authorization| &authorization.opa),
108+
),
101109
)?;
102110
let datanode_role_group_configs = validate_role_group_configs(
103111
hdfs.spec.data_nodes.as_ref(),

‎rust/operator-binary/src/crd/affinity.rs‎

Lines changed: 75 additions & 27 deletions
Original file line numberDiff line numberDiff line change
@@ -1,18 +1,34 @@
11
use stackable_operator::{
2-
commons::affinity::{
3-
StackableAffinityFragment, affinity_between_cluster_pods, affinity_between_role_pods,
2+
commons::{
3+
affinity::{
4+
StackableAffinityFragment, affinity_between_cluster_pods, affinity_between_role_pods,
5+
},
6+
opa::OpaConfig,
47
},
58
k8s_openapi::api::core::v1::{PodAffinity, PodAntiAffinity},
69
};
710

811
use crate::crd::{HdfsNodeRole, constants::APP_NAME};
912

10-
pub fn get_affinity(cluster_name: &str, role: &HdfsNodeRole) -> StackableAffinityFragment {
13+
/// `opa_config` is only passed for roles that send authorization requests to OPA.
14+
pub fn get_affinity(
15+
cluster_name: &str,
16+
role: &HdfsNodeRole,
17+
opa_config: Option<&OpaConfig>,
18+
) -> StackableAffinityFragment {
19+
let mut pod_affinities = vec![affinity_between_cluster_pods(APP_NAME, cluster_name, 20)];
20+
if let Some(opa_config) = opa_config {
21+
pod_affinities.push(affinity_between_role_pods(
22+
"opa",
23+
&opa_config.config_map_name, // The discovery cm has the same name as the OpaCluster itself
24+
"server",
25+
50,
26+
));
27+
}
28+
1129
StackableAffinityFragment {
1230
pod_affinity: Some(PodAffinity {
13-
preferred_during_scheduling_ignored_during_execution: Some(vec![
14-
affinity_between_cluster_pods(APP_NAME, cluster_name, 20),
15-
]),
31+
preferred_during_scheduling_ignored_during_execution: Some(pod_affinities),
1632
required_during_scheduling_ignored_during_execution: None,
1733
}),
1834
pod_anti_affinity: Some(PodAntiAffinity {
@@ -63,6 +79,10 @@ spec:
6379
productVersion: 3.5.0
6480
clusterConfig:
6581
zookeeperConfigMapName: hdfs-zk
82+
authorization:
83+
opa:
84+
configMapName: simple-opa
85+
package: hdfs
6686
journalNodes:
6787
roleGroups:
6888
default:
@@ -80,31 +100,59 @@ spec:
80100
let validated_cluster = deserialize_and_validate_cluster(input);
81101
let merged_config = common_config(&validated_cluster, &role, &role_group_name("default"));
82102

103+
let mut expected_pod_affinities = vec![WeightedPodAffinityTerm {
104+
pod_affinity_term: PodAffinityTerm {
105+
label_selector: Some(LabelSelector {
106+
match_expressions: None,
107+
match_labels: Some(BTreeMap::from([
108+
("app.kubernetes.io/name".to_string(), "hdfs".to_string()),
109+
(
110+
"app.kubernetes.io/instance".to_string(),
111+
"simple-hdfs".to_string(),
112+
),
113+
])),
114+
}),
115+
namespace_selector: None,
116+
namespaces: None,
117+
topology_key: "kubernetes.io/hostname".to_string(),
118+
..PodAffinityTerm::default()
119+
},
120+
weight: 20,
121+
}];
122+
// Only the NameNode is configured with the OPA authorizer.
123+
if role == HdfsNodeRole::Name {
124+
expected_pod_affinities.push(WeightedPodAffinityTerm {
125+
pod_affinity_term: PodAffinityTerm {
126+
label_selector: Some(LabelSelector {
127+
match_expressions: None,
128+
match_labels: Some(BTreeMap::from([
129+
("app.kubernetes.io/name".to_string(), "opa".to_string()),
130+
(
131+
"app.kubernetes.io/instance".to_string(),
132+
"simple-opa".to_string(),
133+
),
134+
(
135+
"app.kubernetes.io/component".to_string(),
136+
"server".to_string(),
137+
),
138+
])),
139+
}),
140+
namespace_selector: None,
141+
namespaces: None,
142+
topology_key: "kubernetes.io/hostname".to_string(),
143+
..PodAffinityTerm::default()
144+
},
145+
weight: 50,
146+
});
147+
}
148+
83149
assert_eq!(
84150
merged_config.affinity,
85151
StackableAffinity {
86152
pod_affinity: Some(PodAffinity {
87-
preferred_during_scheduling_ignored_during_execution: Some(vec![
88-
WeightedPodAffinityTerm {
89-
pod_affinity_term: PodAffinityTerm {
90-
label_selector: Some(LabelSelector {
91-
match_expressions: None,
92-
match_labels: Some(BTreeMap::from([
93-
("app.kubernetes.io/name".to_string(), "hdfs".to_string(),),
94-
(
95-
"app.kubernetes.io/instance".to_string(),
96-
"simple-hdfs".to_string(),
97-
),
98-
]))
99-
}),
100-
namespace_selector: None,
101-
namespaces: None,
102-
topology_key: "kubernetes.io/hostname".to_string(),
103-
..PodAffinityTerm::default()
104-
},
105-
weight: 20
106-
}
107-
]),
153+
preferred_during_scheduling_ignored_during_execution: Some(
154+
expected_pod_affinities
155+
),
108156
required_during_scheduling_ignored_during_execution: None,
109157
}),
110158
pod_anti_affinity: Some(PodAntiAffinity {

‎rust/operator-binary/src/crd/mod.rs‎

Lines changed: 13 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -13,6 +13,7 @@ use stackable_operator::{
1313
commons::{
1414
affinity::StackableAffinity,
1515
cluster_operation::ClusterOperation,
16+
opa::OpaConfig,
1617
product_image_selection::ProductImage,
1718
resources::{
1819
CpuLimitsFragment, MemoryLimitsFragment, NoRuntimeLimits, NoRuntimeLimitsFragment,
@@ -640,7 +641,11 @@ pub struct NameNodeConfig {
640641
impl NameNodeConfigFragment {
641642
const DEFAULT_NAME_NODE_SECRET_LIFETIME: Duration = Duration::from_days_unchecked(1);
642643

643-
pub fn default_config(cluster_name: &str, role: &HdfsNodeRole) -> Self {
644+
pub fn default_config(
645+
cluster_name: &str,
646+
role: &HdfsNodeRole,
647+
opa_config: Option<&OpaConfig>,
648+
) -> Self {
644649
Self {
645650
resources: ResourcesFragment {
646651
cpu: CpuLimitsFragment {
@@ -662,7 +667,7 @@ impl NameNodeConfigFragment {
662667
logging: product_logging::spec::default_logging(),
663668
listener_class: Some(DEFAULT_LISTENER_CLASS.clone()),
664669
common: CommonNodeConfigFragment {
665-
affinity: get_affinity(cluster_name, role),
670+
affinity: get_affinity(cluster_name, role, opa_config),
666671
graceful_shutdown_timeout: Some(DEFAULT_NAME_NODE_GRACEFUL_SHUTDOWN_TIMEOUT),
667672
requested_secret_lifetime: Some(Self::DEFAULT_NAME_NODE_SECRET_LIFETIME),
668673
},
@@ -750,7 +755,9 @@ impl DataNodeConfigFragment {
750755
logging: product_logging::spec::default_logging(),
751756
listener_class: Some(DEFAULT_LISTENER_CLASS.clone()),
752757
common: CommonNodeConfigFragment {
753-
affinity: get_affinity(cluster_name, role),
758+
// Only the NameNode is configured with the OPA authorizer, so this role gets no
759+
// affinity to the OPA Pods.
760+
affinity: get_affinity(cluster_name, role, None),
754761
graceful_shutdown_timeout: Some(DEFAULT_DATA_NODE_GRACEFUL_SHUTDOWN_TIMEOUT),
755762
requested_secret_lifetime: Some(Self::DEFAULT_DATA_NODE_SECRET_LIFETIME),
756763
},
@@ -826,7 +833,9 @@ impl JournalNodeConfigFragment {
826833
},
827834
logging: product_logging::spec::default_logging(),
828835
common: CommonNodeConfigFragment {
829-
affinity: get_affinity(cluster_name, role),
836+
// Only the NameNode is configured with the OPA authorizer, so this role gets no
837+
// affinity to the OPA Pods.
838+
affinity: get_affinity(cluster_name, role, None),
830839
graceful_shutdown_timeout: Some(DEFAULT_JOURNAL_NODE_GRACEFUL_SHUTDOWN_TIMEOUT),
831840
requested_secret_lifetime: Some(Self::DEFAULT_JOURNAL_NODE_SECRET_LIFETIME),
832841
},

0 commit comments

Comments
 (0)