diff --git a/etcd/v1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml b/etcd/v1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml index dde9ce6e9f1..bcdd9e005d6 100644 --- a/etcd/v1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml +++ b/etcd/v1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml @@ -53,6 +53,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected nodes present" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -256,6 +261,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected nodes present" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -461,6 +471,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" - type: Extra status: "True" lastTransitionTime: "2024-01-01T00:00:00Z" @@ -498,6 +513,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -709,6 +729,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -919,6 +944,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1129,6 +1159,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1332,6 +1367,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1541,6 +1581,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1751,6 +1796,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1933,3 +1983,1123 @@ tests: reason: Schedulable message: "Schedulable" expectedStatusError: "must be a valid global unicast IPv4 or IPv6 address in canonical form" + - name: Should reject cluster status missing FencingEnabled condition + initial: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: Extra + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Extra + message: "Extra" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: [] + expectedStatusError: "conditions must contain a condition of type FencingEnabled" + - name: Should accept node with lastFenceEvent for a completed fence operation + initial: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "False" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Unclean + message: "Node was fenced" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: success + client: crmd + lastUpdated: "2024-01-01T00:00:00Z" + completedTime: "2024-01-01T00:00:05Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + expected: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "False" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Unclean + message: "Node was fenced" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: success + client: crmd + lastUpdated: "2024-01-01T00:00:00Z" + completedTime: "2024-01-01T00:00:05Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Should accept lastFenceEvent with pending status and no completedTime + initial: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Clean + message: "Clean" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: pending + lastUpdated: "2024-01-01T00:00:00Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + expected: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Clean + message: "Clean" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: pending + lastUpdated: "2024-01-01T00:00:00Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Should reject lastFenceEvent with invalid action + initial: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Clean + message: "Clean" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: poweroff + status: success + lastUpdated: "2024-01-01T00:00:00Z" + completedTime: "2024-01-01T00:00:05Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + expectedStatusError: "Unsupported value" diff --git a/etcd/v1/types_pacemakercluster.go b/etcd/v1/types_pacemakercluster.go index a481f5e1bd4..de038aa0d63 100644 --- a/etcd/v1/types_pacemakercluster.go +++ b/etcd/v1/types_pacemakercluster.go @@ -14,6 +14,7 @@ const ( // Specifically, it aggregates the following conditions: // - ClusterInServiceConditionType // - ClusterNodeCountAsExpectedConditionType + // - ClusterFencingEnabledConditionType // - NodeHealthyConditionType (for each node) // When True, the cluster is healthy with reason "ClusterHealthy". // When False, the cluster is unhealthy with reason "ClusterUnhealthy". @@ -30,6 +31,16 @@ const ( // When True, the expected number of nodes are present with reason "AsExpected". // When False, the node count is incorrect with reason "InsufficientNodes" or "ExcessiveNodes". ClusterNodeCountAsExpectedConditionType = "NodeCountAsExpected" + + // ClusterFencingEnabledConditionType tracks whether STONITH (fencing) is enabled in the cluster. + // Fencing is the mechanism that isolates failed nodes by powering them off via BMC. In Two Node + // OpenShift with Fencing, fencing is mandatory — without it, a network partition leaves both nodes + // running and data diverges. If someone runs `pcs property set stonith-enabled=false`, fencing is + // globally disabled and the cluster cannot recover from node failures. + // When True, fencing is enabled with reason "FencingEnabled". This is the normal operating state. + // When False, fencing is globally disabled with reason "FencingDisabled". This is a critical state. + // When Unknown, the fencing-enabled property has not yet been observed with reason "Pending". + ClusterFencingEnabledConditionType = "FencingEnabled" ) // ClusterHealthy condition reasons @@ -72,6 +83,23 @@ const ( ClusterNodeCountAsExpectedReasonExcessiveNodes = "ExcessiveNodes" ) +// ClusterFencingEnabled condition reasons +const ( + // ClusterFencingEnabledReasonEnabled means fencing (STONITH) is enabled in the cluster. + // This is the normal and expected operating state for Two Node OpenShift with Fencing. + ClusterFencingEnabledReasonEnabled = "FencingEnabled" + + // ClusterFencingEnabledReasonDisabled means fencing (STONITH) is globally disabled in the cluster. + // Without fencing, the cluster cannot isolate failed nodes and etcd quorum recovery cannot happen. + // This is a critical state that should be investigated immediately. + ClusterFencingEnabledReasonDisabled = "FencingDisabled" + + // ClusterFencingEnabledReasonPending means the fencing-enabled property has not yet been observed + // by the status collector. This is expected to be temporary, for example immediately after upgrade + // or before the first successful status collection. + ClusterFencingEnabledReasonPending = "Pending" +) + // Node-level condition types for PacemakerCluster.status.nodes[].conditions const ( // NodeHealthyConditionType tracks the overall health of a node in the pacemaker cluster. @@ -462,6 +490,84 @@ const ( FencingMethodIPMI FencingMethod = "IPMI" ) +// PacemakerFenceEventAction represents the type of fencing action performed. +// Values are lowercase per Pacemaker's fence-event XML schema (fence-event-2.15.rng). +// +kubebuilder:validation:Enum=reboot;power-off;power-on +// +enum +type PacemakerFenceEventAction string + +const ( + // PacemakerFenceEventActionReboot is a fence action that power-cycles the target node. + PacemakerFenceEventActionReboot PacemakerFenceEventAction = "reboot" + + // PacemakerFenceEventActionPowerOff is a fence action that powers off the target node. + PacemakerFenceEventActionPowerOff PacemakerFenceEventAction = "power-off" + + // PacemakerFenceEventActionPowerOn is a fence action that powers on the target node. + PacemakerFenceEventActionPowerOn PacemakerFenceEventAction = "power-on" +) + +// PacemakerFenceEventStatus represents the outcome of a fencing operation. +// Values are lowercase per Pacemaker's fence-event XML schema (fence-event-2.15.rng). +// +kubebuilder:validation:Enum=success;failed;pending +// +enum +type PacemakerFenceEventStatus string + +const ( + // PacemakerFenceEventStatusSuccess means the fencing operation completed successfully. + PacemakerFenceEventStatusSuccess PacemakerFenceEventStatus = "success" + + // PacemakerFenceEventStatusFailed means the fencing operation failed. + PacemakerFenceEventStatusFailed PacemakerFenceEventStatus = "failed" + + // PacemakerFenceEventStatusPending means the fencing operation is currently in progress. + PacemakerFenceEventStatusPending PacemakerFenceEventStatus = "pending" +) + +// PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events +// are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during +// split-brain recovery or manually via stonith_admin. This struct captures the last fence event +// targeting a given node, providing visibility into what happened, its outcome, and whether it was +// triggered automatically by the cluster or manually by an operator. +type PacemakerFenceEvent struct { + // action is the type of fencing action that was performed or is in progress. + // Valid values are "reboot" (power-cycle the node), "power-off" (power off the node), + // and "power-on" (power on the node). + // +required + Action PacemakerFenceEventAction `json:"action,omitempty"` + + // status is the outcome of the fencing operation. + // Valid values are "success" (the operation completed successfully), "failed" (the operation + // failed), and "pending" (the operation is currently in progress). + // +required + Status PacemakerFenceEventStatus `json:"status,omitempty"` + + // client identifies the daemon or tool that requested the fencing operation. Typical values + // are "crmd" for automatic recovery initiated by the cluster resource manager, or + // "stonith_admin" for manual fencing initiated by an operator. This is useful for + // distinguishing "the cluster fenced itself for cause" from "someone fenced a node manually" + // during incident review. This field is optional and is omitted when the client is not + // reported by Pacemaker. When provided, the value must be between 1 and 256 characters. + // +kubebuilder:validation:MinLength=1 + // +kubebuilder:validation:MaxLength=256 + // +optional + Client string `json:"client,omitempty"` + + // lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. + // This timestamp is always present in the CIB fence history, including for pending operations + // where completedTime is not yet set. It must be a valid timestamp in RFC3339 format. + // +kubebuilder:validation:Format=date-time + // +required + LastUpdated *metav1.Time `json:"lastUpdated,omitempty"` + + // completedTime is the timestamp when the fencing operation completed. This field is optional + // and is omitted when the fencing operation is still in progress (status "pending") or when + // Pacemaker has not recorded a completion time. + // +kubebuilder:validation:Format=date-time + // +optional + CompletedTime *metav1.Time `json:"completedTime,omitempty"` +} + // +genclient // +genclient:nonNamespaced // +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object @@ -502,18 +608,20 @@ type PacemakerCluster struct { // +kubebuilder:validation:XValidation:rule="!has(oldSelf.lastUpdated) || self.lastUpdated >= oldSelf.lastUpdated",message="lastUpdated may not be set to an earlier timestamp" type PacemakerClusterStatus struct { // conditions represent the observations of the pacemaker cluster's current state. - // Known condition types are: "Healthy", "InService", "NodeCountAsExpected". + // Known condition types are: "Healthy", "InService", "NodeCountAsExpected", "FencingEnabled". // The "Healthy" condition is an aggregate that tracks the overall health of the cluster. // The "InService" condition tracks whether the cluster is in service (not in maintenance mode). // The "NodeCountAsExpected" condition tracks whether the expected number of nodes are present. - // Each of these conditions is required, so the array must contain at least 3 items. + // The "FencingEnabled" condition tracks whether STONITH (fencing) is enabled in the cluster. + // Each of these conditions is required, so the array must contain at least 4 items. // +listType=map // +listMapKey=type - // +kubebuilder:validation:MinItems=3 + // +kubebuilder:validation:MinItems=4 // +kubebuilder:validation:MaxItems=8 // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'Healthy')",message="conditions must contain a condition of type Healthy" // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'InService')",message="conditions must contain a condition of type InService" // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'NodeCountAsExpected')",message="conditions must contain a condition of type NodeCountAsExpected" + // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'FencingEnabled')",message="conditions must contain a condition of type FencingEnabled" // +required Conditions []metav1.Condition `json:"conditions,omitempty"` @@ -619,6 +727,15 @@ type PacemakerClusterNodeStatus struct { // +kubebuilder:validation:XValidation:rule="self.all(x, self.exists_one(y, x.name == y.name))",message="fencing agent names must be unique" // +required FencingAgents []PacemakerClusterFencingAgentStatus `json:"fencingAgents,omitempty"` + + // lastFenceEvent is the most recent fencing event targeting this node, as recorded in + // Pacemaker's fence history in the CIB. This captures the last time this node was fenced + // (or a fence attempt was made), including the action taken and its outcome. When the + // Clean condition is False, this field provides the concrete fencing context behind the + // unclean state. This field is optional and is omitted when no fencing event has been + // observed for this node. + // +optional + LastFenceEvent PacemakerFenceEvent `json:"lastFenceEvent,omitempty,omitzero"` } // PacemakerClusterFencingAgentStatus represents the status of a fencing agent that can fence a node. @@ -716,6 +833,28 @@ type PacemakerClusterResourceStatus struct { // Fencing agents are tracked separately in the node's fencingAgents field. // +required Name PacemakerClusterResourceName `json:"name,omitempty"` + + // failCount is the current failure count Pacemaker records for this resource on + // this node, as reported by the CIB. Pacemaker increments this count each time an + // operation for this resource fails, and resets it to zero when a `pcs resource + // cleanup` is performed. The value must be zero or greater. This field is optional + // and is omitted when the status collector has not yet observed a fail count for + // this resource, for example on a freshly bootstrapped cluster or for a resource + // that has never failed. + // +kubebuilder:validation:Minimum=0 + // +optional + FailCount *int32 `json:"failCount,omitempty"` + + // migrationThreshold is the configured number of failures after which Pacemaker + // will no longer attempt to run this resource on this node, as reported by the + // CIB. Without this value, failCount alone is uninterpretable — whether + // failCount 3 is alarming depends on whether the threshold is 5 or 1000000 + // (Pacemaker's default INFINITY). The value must be zero or greater. This field + // is optional and is omitted when the status collector has not yet observed a + // migration threshold for this resource. + // +kubebuilder:validation:Minimum=0 + // +optional + MigrationThreshold *int32 `json:"migrationThreshold,omitempty"` } // +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object diff --git a/etcd/v1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml b/etcd/v1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml index 0cad7e278b2..ba042080da2 100644 --- a/etcd/v1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml +++ b/etcd/v1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml @@ -54,11 +54,12 @@ spec: conditions: description: |- conditions represent the observations of the pacemaker cluster's current state. - Known condition types are: "Healthy", "InService", "NodeCountAsExpected". + Known condition types are: "Healthy", "InService", "NodeCountAsExpected", "FencingEnabled". The "Healthy" condition is an aggregate that tracks the overall health of the cluster. The "InService" condition tracks whether the cluster is in service (not in maintenance mode). The "NodeCountAsExpected" condition tracks whether the expected number of nodes are present. - Each of these conditions is required, so the array must contain at least 3 items. + The "FencingEnabled" condition tracks whether STONITH (fencing) is enabled in the cluster. + Each of these conditions is required, so the array must contain at least 4 items. items: description: Condition contains details for one aspect of the current state of this API Resource. @@ -114,7 +115,7 @@ spec: - type type: object maxItems: 8 - minItems: 3 + minItems: 4 type: array x-kubernetes-list-map-keys: - type @@ -126,6 +127,8 @@ spec: rule: self.exists(c, c.type == 'InService') - message: conditions must contain a condition of type NodeCountAsExpected rule: self.exists(c, c.type == 'NodeCountAsExpected') + - message: conditions must contain a condition of type FencingEnabled + rule: self.exists(c, c.type == 'FencingEnabled') lastUpdated: description: |- lastUpdated is the timestamp when this status was last updated. This is useful for identifying @@ -437,6 +440,65 @@ spec: x-kubernetes-validations: - message: fencing agent names must be unique rule: self.all(x, self.exists_one(y, x.name == y.name)) + lastFenceEvent: + description: |- + lastFenceEvent is the most recent fencing event targeting this node, as recorded in + Pacemaker's fence history in the CIB. This captures the last time this node was fenced + (or a fence attempt was made), including the action taken and its outcome. When the + Clean condition is False, this field provides the concrete fencing context behind the + unclean state. This field is optional and is omitted when no fencing event has been + observed for this node. + properties: + action: + description: |- + action is the type of fencing action that was performed or is in progress. + Valid values are "reboot" (power-cycle the node), "power-off" (power off the node), + and "power-on" (power on the node). + enum: + - reboot + - power-off + - power-on + type: string + client: + description: |- + client identifies the daemon or tool that requested the fencing operation. Typical values + are "crmd" for automatic recovery initiated by the cluster resource manager, or + "stonith_admin" for manual fencing initiated by an operator. This is useful for + distinguishing "the cluster fenced itself for cause" from "someone fenced a node manually" + during incident review. This field is optional and is omitted when the client is not + reported by Pacemaker. When provided, the value must be between 1 and 256 characters. + maxLength: 256 + minLength: 1 + type: string + completedTime: + description: |- + completedTime is the timestamp when the fencing operation completed. This field is optional + and is omitted when the fencing operation is still in progress (status "pending") or when + Pacemaker has not recorded a completion time. + format: date-time + type: string + lastUpdated: + description: |- + lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. + This timestamp is always present in the CIB fence history, including for pending operations + where completedTime is not yet set. It must be a valid timestamp in RFC3339 format. + format: date-time + type: string + status: + description: |- + status is the outcome of the fencing operation. + Valid values are "success" (the operation completed successfully), "failed" (the operation + failed), and "pending" (the operation is currently in progress). + enum: + - success + - failed + - pending + type: string + required: + - action + - lastUpdated + - status + type: object nodeName: description: |- nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase @@ -571,6 +633,30 @@ spec: - message: conditions must contain a condition of type Schedulable rule: self.exists(c, c.type == 'Schedulable') + failCount: + description: |- + failCount is the current failure count Pacemaker records for this resource on + this node, as reported by the CIB. Pacemaker increments this count each time an + operation for this resource fails, and resets it to zero when a `pcs resource + cleanup` is performed. The value must be zero or greater. This field is optional + and is omitted when the status collector has not yet observed a fail count for + this resource, for example on a freshly bootstrapped cluster or for a resource + that has never failed. + format: int32 + minimum: 0 + type: integer + migrationThreshold: + description: |- + migrationThreshold is the configured number of failures after which Pacemaker + will no longer attempt to run this resource on this node, as reported by the + CIB. Without this value, failCount alone is uninterpretable — whether + failCount 3 is alarming depends on whether the threshold is 5 or 1000000 + (Pacemaker's default INFINITY). The value must be zero or greater. This field + is optional and is omitted when the status collector has not yet observed a + migration threshold for this resource. + format: int32 + minimum: 0 + type: integer name: description: |- name is the name of the pacemaker resource. diff --git a/etcd/v1/zz_generated.deepcopy.go b/etcd/v1/zz_generated.deepcopy.go index c529240e40c..2a7d364ee50 100644 --- a/etcd/v1/zz_generated.deepcopy.go +++ b/etcd/v1/zz_generated.deepcopy.go @@ -122,6 +122,7 @@ func (in *PacemakerClusterNodeStatus) DeepCopyInto(out *PacemakerClusterNodeStat (*in)[i].DeepCopyInto(&(*out)[i]) } } + in.LastFenceEvent.DeepCopyInto(&out.LastFenceEvent) return } @@ -145,6 +146,16 @@ func (in *PacemakerClusterResourceStatus) DeepCopyInto(out *PacemakerClusterReso (*in)[i].DeepCopyInto(&(*out)[i]) } } + if in.FailCount != nil { + in, out := &in.FailCount, &out.FailCount + *out = new(int32) + **out = **in + } + if in.MigrationThreshold != nil { + in, out := &in.MigrationThreshold, &out.MigrationThreshold + *out = new(int32) + **out = **in + } return } @@ -193,6 +204,30 @@ func (in *PacemakerClusterStatus) DeepCopy() *PacemakerClusterStatus { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *PacemakerFenceEvent) DeepCopyInto(out *PacemakerFenceEvent) { + *out = *in + if in.LastUpdated != nil { + in, out := &in.LastUpdated, &out.LastUpdated + *out = (*in).DeepCopy() + } + if in.CompletedTime != nil { + in, out := &in.CompletedTime, &out.CompletedTime + *out = (*in).DeepCopy() + } + return +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PacemakerFenceEvent. +func (in *PacemakerFenceEvent) DeepCopy() *PacemakerFenceEvent { + if in == nil { + return nil + } + out := new(PacemakerFenceEvent) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *PacemakerNodeAddress) DeepCopyInto(out *PacemakerNodeAddress) { *out = *in diff --git a/etcd/v1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml b/etcd/v1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml index c9ec52f3829..2f83b7739cb 100644 --- a/etcd/v1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml +++ b/etcd/v1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml @@ -55,11 +55,12 @@ spec: conditions: description: |- conditions represent the observations of the pacemaker cluster's current state. - Known condition types are: "Healthy", "InService", "NodeCountAsExpected". + Known condition types are: "Healthy", "InService", "NodeCountAsExpected", "FencingEnabled". The "Healthy" condition is an aggregate that tracks the overall health of the cluster. The "InService" condition tracks whether the cluster is in service (not in maintenance mode). The "NodeCountAsExpected" condition tracks whether the expected number of nodes are present. - Each of these conditions is required, so the array must contain at least 3 items. + The "FencingEnabled" condition tracks whether STONITH (fencing) is enabled in the cluster. + Each of these conditions is required, so the array must contain at least 4 items. items: description: Condition contains details for one aspect of the current state of this API Resource. @@ -115,7 +116,7 @@ spec: - type type: object maxItems: 8 - minItems: 3 + minItems: 4 type: array x-kubernetes-list-map-keys: - type @@ -127,6 +128,8 @@ spec: rule: self.exists(c, c.type == 'InService') - message: conditions must contain a condition of type NodeCountAsExpected rule: self.exists(c, c.type == 'NodeCountAsExpected') + - message: conditions must contain a condition of type FencingEnabled + rule: self.exists(c, c.type == 'FencingEnabled') lastUpdated: description: |- lastUpdated is the timestamp when this status was last updated. This is useful for identifying @@ -438,6 +441,65 @@ spec: x-kubernetes-validations: - message: fencing agent names must be unique rule: self.all(x, self.exists_one(y, x.name == y.name)) + lastFenceEvent: + description: |- + lastFenceEvent is the most recent fencing event targeting this node, as recorded in + Pacemaker's fence history in the CIB. This captures the last time this node was fenced + (or a fence attempt was made), including the action taken and its outcome. When the + Clean condition is False, this field provides the concrete fencing context behind the + unclean state. This field is optional and is omitted when no fencing event has been + observed for this node. + properties: + action: + description: |- + action is the type of fencing action that was performed or is in progress. + Valid values are "reboot" (power-cycle the node), "power-off" (power off the node), + and "power-on" (power on the node). + enum: + - reboot + - power-off + - power-on + type: string + client: + description: |- + client identifies the daemon or tool that requested the fencing operation. Typical values + are "crmd" for automatic recovery initiated by the cluster resource manager, or + "stonith_admin" for manual fencing initiated by an operator. This is useful for + distinguishing "the cluster fenced itself for cause" from "someone fenced a node manually" + during incident review. This field is optional and is omitted when the client is not + reported by Pacemaker. When provided, the value must be between 1 and 256 characters. + maxLength: 256 + minLength: 1 + type: string + completedTime: + description: |- + completedTime is the timestamp when the fencing operation completed. This field is optional + and is omitted when the fencing operation is still in progress (status "pending") or when + Pacemaker has not recorded a completion time. + format: date-time + type: string + lastUpdated: + description: |- + lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. + This timestamp is always present in the CIB fence history, including for pending operations + where completedTime is not yet set. It must be a valid timestamp in RFC3339 format. + format: date-time + type: string + status: + description: |- + status is the outcome of the fencing operation. + Valid values are "success" (the operation completed successfully), "failed" (the operation + failed), and "pending" (the operation is currently in progress). + enum: + - success + - failed + - pending + type: string + required: + - action + - lastUpdated + - status + type: object nodeName: description: |- nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase @@ -572,6 +634,30 @@ spec: - message: conditions must contain a condition of type Schedulable rule: self.exists(c, c.type == 'Schedulable') + failCount: + description: |- + failCount is the current failure count Pacemaker records for this resource on + this node, as reported by the CIB. Pacemaker increments this count each time an + operation for this resource fails, and resets it to zero when a `pcs resource + cleanup` is performed. The value must be zero or greater. This field is optional + and is omitted when the status collector has not yet observed a fail count for + this resource, for example on a freshly bootstrapped cluster or for a resource + that has never failed. + format: int32 + minimum: 0 + type: integer + migrationThreshold: + description: |- + migrationThreshold is the configured number of failures after which Pacemaker + will no longer attempt to run this resource on this node, as reported by the + CIB. Without this value, failCount alone is uninterpretable — whether + failCount 3 is alarming depends on whether the threshold is 5 or 1000000 + (Pacemaker's default INFINITY). The value must be zero or greater. This field + is optional and is omitted when the status collector has not yet observed a + migration threshold for this resource. + format: int32 + minimum: 0 + type: integer name: description: |- name is the name of the pacemaker resource. diff --git a/etcd/v1/zz_generated.model_name.go b/etcd/v1/zz_generated.model_name.go index 9e13a7167e4..0211fa49798 100644 --- a/etcd/v1/zz_generated.model_name.go +++ b/etcd/v1/zz_generated.model_name.go @@ -35,6 +35,11 @@ func (in PacemakerClusterStatus) OpenAPIModelName() string { return "com.github.openshift.api.etcd.v1.PacemakerClusterStatus" } +// OpenAPIModelName returns the OpenAPI model name for this type. +func (in PacemakerFenceEvent) OpenAPIModelName() string { + return "com.github.openshift.api.etcd.v1.PacemakerFenceEvent" +} + // OpenAPIModelName returns the OpenAPI model name for this type. func (in PacemakerNodeAddress) OpenAPIModelName() string { return "com.github.openshift.api.etcd.v1.PacemakerNodeAddress" diff --git a/etcd/v1/zz_generated.swagger_doc_generated.go b/etcd/v1/zz_generated.swagger_doc_generated.go index e9e47b47cfd..0b6dc668800 100644 --- a/etcd/v1/zz_generated.swagger_doc_generated.go +++ b/etcd/v1/zz_generated.swagger_doc_generated.go @@ -43,12 +43,13 @@ func (PacemakerClusterList) SwaggerDoc() map[string]string { } var map_PacemakerClusterNodeStatus = map[string]string{ - "": "PacemakerClusterNodeStatus represents the status of a single node in the pacemaker cluster including the node's conditions and the health of critical resources running on that node.", - "conditions": "conditions represent the observations of the node's current state. Known condition types are: \"Healthy\", \"Online\", \"InService\", \"Active\", \"Ready\", \"Clean\", \"Member\", \"FencingAvailable\", \"FencingHealthy\". The \"Healthy\" condition is an aggregate that tracks the overall health of the node. The \"Online\" condition tracks whether the node is online. The \"InService\" condition tracks whether the node is in service (not in maintenance mode). The \"Active\" condition tracks whether the node is active (not in standby mode). The \"Ready\" condition tracks whether the node is ready (not in a pending state). The \"Clean\" condition tracks whether the node is in a clean (status known) state. The \"Member\" condition tracks whether the node is a member of the cluster. The \"FencingAvailable\" condition tracks whether this node can be fenced by at least one healthy agent. The \"FencingHealthy\" condition tracks whether all fencing agents for this node are healthy. Each of these conditions is required, so the array must contain at least 9 items.", - "nodeName": "nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase RFC 1123 subdomain consisting of lowercase alphanumeric characters, '-' or '.', starting and ending with an alphanumeric character, and be at most 253 characters in length.", - "addresses": "addresses is a list of IP addresses for the node. Pacemaker allows multiple IP addresses for Corosync communication between nodes. The first address in this list is used for IP-based peer URLs for etcd membership. Each address must be a valid global unicast IPv4 or IPv6 address in canonical form (e.g., \"192.168.1.1\" not \"192.168.001.001\", or \"2001:db8::1\" not \"2001:0db8::1\"). This excludes loopback, link-local, and multicast addresses.", - "resources": "resources contains the status of pacemaker resources scheduled on this node. Each resource entry includes the resource name and its health conditions. For Two Node OpenShift with Fencing, we track Kubelet and Etcd resources per node. Both resources are required to be present, so the array must contain at least 2 items. Valid resource names are \"Kubelet\" and \"Etcd\". Fencing agents are tracked separately in the fencingAgents field.", - "fencingAgents": "fencingAgents contains the status of fencing agents that can fence this node. Unlike resources (which are scheduled to run on this node), fencing agents are mapped to the node they can fence (their target), not the node where monitoring operations run. Each fencing agent entry includes a unique name, fencing type, target node, and health conditions. A node is considered fence-capable if at least one fencing agent is healthy. A healthy node is expected to have at least 1 fencing agent, but the list may be empty when fencing agent discovery fails. Names must be unique within this array.", + "": "PacemakerClusterNodeStatus represents the status of a single node in the pacemaker cluster including the node's conditions and the health of critical resources running on that node.", + "conditions": "conditions represent the observations of the node's current state. Known condition types are: \"Healthy\", \"Online\", \"InService\", \"Active\", \"Ready\", \"Clean\", \"Member\", \"FencingAvailable\", \"FencingHealthy\". The \"Healthy\" condition is an aggregate that tracks the overall health of the node. The \"Online\" condition tracks whether the node is online. The \"InService\" condition tracks whether the node is in service (not in maintenance mode). The \"Active\" condition tracks whether the node is active (not in standby mode). The \"Ready\" condition tracks whether the node is ready (not in a pending state). The \"Clean\" condition tracks whether the node is in a clean (status known) state. The \"Member\" condition tracks whether the node is a member of the cluster. The \"FencingAvailable\" condition tracks whether this node can be fenced by at least one healthy agent. The \"FencingHealthy\" condition tracks whether all fencing agents for this node are healthy. Each of these conditions is required, so the array must contain at least 9 items.", + "nodeName": "nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase RFC 1123 subdomain consisting of lowercase alphanumeric characters, '-' or '.', starting and ending with an alphanumeric character, and be at most 253 characters in length.", + "addresses": "addresses is a list of IP addresses for the node. Pacemaker allows multiple IP addresses for Corosync communication between nodes. The first address in this list is used for IP-based peer URLs for etcd membership. Each address must be a valid global unicast IPv4 or IPv6 address in canonical form (e.g., \"192.168.1.1\" not \"192.168.001.001\", or \"2001:db8::1\" not \"2001:0db8::1\"). This excludes loopback, link-local, and multicast addresses.", + "resources": "resources contains the status of pacemaker resources scheduled on this node. Each resource entry includes the resource name and its health conditions. For Two Node OpenShift with Fencing, we track Kubelet and Etcd resources per node. Both resources are required to be present, so the array must contain at least 2 items. Valid resource names are \"Kubelet\" and \"Etcd\". Fencing agents are tracked separately in the fencingAgents field.", + "fencingAgents": "fencingAgents contains the status of fencing agents that can fence this node. Unlike resources (which are scheduled to run on this node), fencing agents are mapped to the node they can fence (their target), not the node where monitoring operations run. Each fencing agent entry includes a unique name, fencing type, target node, and health conditions. A node is considered fence-capable if at least one fencing agent is healthy. A healthy node is expected to have at least 1 fencing agent, but the list may be empty when fencing agent discovery fails. Names must be unique within this array.", + "lastFenceEvent": "lastFenceEvent is the most recent fencing event targeting this node, as recorded in Pacemaker's fence history in the CIB. This captures the last time this node was fenced (or a fence attempt was made), including the action taken and its outcome. When the Clean condition is False, this field provides the concrete fencing context behind the unclean state. This field is optional and is omitted when no fencing event has been observed for this node.", } func (PacemakerClusterNodeStatus) SwaggerDoc() map[string]string { @@ -56,9 +57,11 @@ func (PacemakerClusterNodeStatus) SwaggerDoc() map[string]string { } var map_PacemakerClusterResourceStatus = map[string]string{ - "": "PacemakerClusterResourceStatus represents the status of a pacemaker resource scheduled on a node. A pacemaker resource is a unit of work managed by pacemaker. In pacemaker terminology, resources are services or applications that pacemaker monitors, starts, stops, and moves between nodes to maintain high availability. For Two Node OpenShift with Fencing, we track two resources per node:\n - Kubelet (the Kubernetes node agent and a prerequisite for etcd)\n - Etcd (the distributed key-value store)\n\nFencing agents are tracked separately in the fencingAgents field because they are mapped to their target node (the node they can fence), not the node where monitoring operations are scheduled.", - "conditions": "conditions represent the observations of the resource's current state. Known condition types are: \"Healthy\", \"InService\", \"Managed\", \"Enabled\", \"Operational\", \"Active\", \"Started\", \"Schedulable\". The \"Healthy\" condition is an aggregate that tracks the overall health of the resource. The \"InService\" condition tracks whether the resource is in service (not in maintenance mode). The \"Managed\" condition tracks whether the resource is managed by pacemaker. The \"Enabled\" condition tracks whether the resource is enabled. The \"Operational\" condition tracks whether the resource is operational (not failed). The \"Active\" condition tracks whether the resource is active (available to be used). The \"Started\" condition tracks whether the resource is started. The \"Schedulable\" condition tracks whether the resource is schedulable (not blocked). Each of these conditions is required, so the array must contain at least 8 items.", - "name": "name is the name of the pacemaker resource. Valid values are \"Kubelet\" and \"Etcd\". The Kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments. The Etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations. Fencing agents are tracked separately in the node's fencingAgents field.", + "": "PacemakerClusterResourceStatus represents the status of a pacemaker resource scheduled on a node. A pacemaker resource is a unit of work managed by pacemaker. In pacemaker terminology, resources are services or applications that pacemaker monitors, starts, stops, and moves between nodes to maintain high availability. For Two Node OpenShift with Fencing, we track two resources per node:\n - Kubelet (the Kubernetes node agent and a prerequisite for etcd)\n - Etcd (the distributed key-value store)\n\nFencing agents are tracked separately in the fencingAgents field because they are mapped to their target node (the node they can fence), not the node where monitoring operations are scheduled.", + "conditions": "conditions represent the observations of the resource's current state. Known condition types are: \"Healthy\", \"InService\", \"Managed\", \"Enabled\", \"Operational\", \"Active\", \"Started\", \"Schedulable\". The \"Healthy\" condition is an aggregate that tracks the overall health of the resource. The \"InService\" condition tracks whether the resource is in service (not in maintenance mode). The \"Managed\" condition tracks whether the resource is managed by pacemaker. The \"Enabled\" condition tracks whether the resource is enabled. The \"Operational\" condition tracks whether the resource is operational (not failed). The \"Active\" condition tracks whether the resource is active (available to be used). The \"Started\" condition tracks whether the resource is started. The \"Schedulable\" condition tracks whether the resource is schedulable (not blocked). Each of these conditions is required, so the array must contain at least 8 items.", + "name": "name is the name of the pacemaker resource. Valid values are \"Kubelet\" and \"Etcd\". The Kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments. The Etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations. Fencing agents are tracked separately in the node's fencingAgents field.", + "failCount": "failCount is the current failure count Pacemaker records for this resource on this node, as reported by the CIB. Pacemaker increments this count each time an operation for this resource fails, and resets it to zero when a `pcs resource cleanup` is performed. The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a fail count for this resource, for example on a freshly bootstrapped cluster or for a resource that has never failed.", + "migrationThreshold": "migrationThreshold is the configured number of failures after which Pacemaker will no longer attempt to run this resource on this node, as reported by the CIB. Without this value, failCount alone is uninterpretable — whether failCount 3 is alarming depends on whether the threshold is 5 or 1000000 (Pacemaker's default INFINITY). The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a migration threshold for this resource.", } func (PacemakerClusterResourceStatus) SwaggerDoc() map[string]string { @@ -67,7 +70,7 @@ func (PacemakerClusterResourceStatus) SwaggerDoc() map[string]string { var map_PacemakerClusterStatus = map[string]string{ "": "PacemakerClusterStatus contains the actual pacemaker cluster status information. As part of validating the status object, we need to ensure that the lastUpdated timestamp may not be set to an earlier timestamp than the current value. The validation rule checks if oldSelf has lastUpdated before comparing, to handle the initial status creation case.", - "conditions": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. Each of these conditions is required, so the array must contain at least 3 items.", + "conditions": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\", \"FencingEnabled\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. The \"FencingEnabled\" condition tracks whether STONITH (fencing) is enabled in the cluster. Each of these conditions is required, so the array must contain at least 4 items.", "lastUpdated": "lastUpdated is the timestamp when this status was last updated. This is useful for identifying stale status reports. It must be a valid timestamp in RFC3339 format. Once set, this field cannot be removed and cannot be set to an earlier timestamp than the current value.", "nodes": "nodes provides detailed status for each control-plane node in the Pacemaker cluster. While Pacemaker supports up to 32 nodes, the limit is set to 5 (max OpenShift control-plane nodes). For Two Node OpenShift with Fencing, exactly 2 nodes are expected in a healthy cluster. An empty list indicates a catastrophic failure where Pacemaker reports no nodes.", } @@ -76,6 +79,19 @@ func (PacemakerClusterStatus) SwaggerDoc() map[string]string { return map_PacemakerClusterStatus } +var map_PacemakerFenceEvent = map[string]string{ + "": "PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during split-brain recovery or manually via stonith_admin. This struct captures the last fence event targeting a given node, providing visibility into what happened, its outcome, and whether it was triggered automatically by the cluster or manually by an operator.", + "action": "action is the type of fencing action that was performed or is in progress. Valid values are \"reboot\" (power-cycle the node), \"power-off\" (power off the node), and \"power-on\" (power on the node).", + "status": "status is the outcome of the fencing operation. Valid values are \"success\" (the operation completed successfully), \"failed\" (the operation failed), and \"pending\" (the operation is currently in progress).", + "client": "client identifies the daemon or tool that requested the fencing operation. Typical values are \"crmd\" for automatic recovery initiated by the cluster resource manager, or \"stonith_admin\" for manual fencing initiated by an operator. This is useful for distinguishing \"the cluster fenced itself for cause\" from \"someone fenced a node manually\" during incident review. This field is optional and is omitted when the client is not reported by Pacemaker. When provided, the value must be between 1 and 256 characters.", + "lastUpdated": "lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. This timestamp is always present in the CIB fence history, including for pending operations where completedTime is not yet set. It must be a valid timestamp in RFC3339 format.", + "completedTime": "completedTime is the timestamp when the fencing operation completed. This field is optional and is omitted when the fencing operation is still in progress (status \"pending\") or when Pacemaker has not recorded a completion time.", +} + +func (PacemakerFenceEvent) SwaggerDoc() map[string]string { + return map_PacemakerFenceEvent +} + var map_PacemakerNodeAddress = map[string]string{ "": "PacemakerNodeAddress contains information for a node's address. This is similar to corev1.NodeAddress but adds validation for IP addresses.", "type": "type is the type of node address. Currently only \"InternalIP\" is supported.", diff --git a/etcd/v1alpha1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml b/etcd/v1alpha1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml index 6c919c01946..0ec8ddbfccd 100644 --- a/etcd/v1alpha1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml +++ b/etcd/v1alpha1/tests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml @@ -53,6 +53,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected nodes present" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -256,6 +261,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected nodes present" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -461,6 +471,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" - type: Extra status: "True" lastTransitionTime: "2024-01-01T00:00:00Z" @@ -498,6 +513,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -709,6 +729,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -919,6 +944,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1129,6 +1159,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1332,6 +1367,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1541,6 +1581,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1751,6 +1796,11 @@ tests: lastTransitionTime: "2024-01-01T00:00:00Z" reason: AsExpected message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" lastUpdated: "2024-01-01T00:00:01Z" nodes: - nodeName: master-0.example.com @@ -1933,3 +1983,1123 @@ tests: reason: Schedulable message: "Schedulable" expectedStatusError: "must be a valid global unicast IPv4 or IPv6 address in canonical form" + - name: Should reject cluster status missing FencingEnabled condition + initial: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: Extra + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Extra + message: "Extra" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: [] + expectedStatusError: "conditions must contain a condition of type FencingEnabled" + - name: Should accept node with lastFenceEvent for a completed fence operation + initial: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "False" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Unclean + message: "Node was fenced" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: success + client: crmd + lastUpdated: "2024-01-01T00:00:00Z" + completedTime: "2024-01-01T00:00:05Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + expected: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "False" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Unclean + message: "Node was fenced" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: success + client: crmd + lastUpdated: "2024-01-01T00:00:00Z" + completedTime: "2024-01-01T00:00:05Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Should accept lastFenceEvent with pending status and no completedTime + initial: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Clean + message: "Clean" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: pending + lastUpdated: "2024-01-01T00:00:00Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + expected: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Clean + message: "Clean" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: reboot + status: pending + lastUpdated: "2024-01-01T00:00:00Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Should reject lastFenceEvent with invalid action + initial: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + updated: | + apiVersion: etcd.openshift.io/v1alpha1 + kind: PacemakerCluster + metadata: + name: cluster + status: + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ClusterHealthy + message: "Cluster is healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: NodeCountAsExpected + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: AsExpected + message: "Expected" + - type: FencingEnabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingEnabled + message: "Fencing enabled" + lastUpdated: "2024-01-01T00:00:01Z" + nodes: + - nodeName: master-0.example.com + addresses: + - type: InternalIP + address: "192.168.1.1" + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: NodeHealthy + message: "Node healthy" + - type: Online + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Online + message: "Online" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Ready + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Ready + message: "Ready" + - type: Clean + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Clean + message: "Clean" + - type: Member + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Member + message: "Member" + - type: FencingAvailable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingAvailable + message: "Fencing available" + - type: FencingHealthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: FencingHealthy + message: "Fencing healthy" + lastFenceEvent: + action: poweroff + status: success + lastUpdated: "2024-01-01T00:00:00Z" + completedTime: "2024-01-01T00:00:05Z" + resources: + - name: Kubelet + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + - name: Etcd + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + fencingAgents: + - name: master-0.example.com_redfish + method: Redfish + conditions: + - type: Healthy + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: ResourceHealthy + message: "Healthy" + - type: InService + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: InService + message: "In service" + - type: Managed + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Managed + message: "Managed" + - type: Enabled + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Enabled + message: "Enabled" + - type: Operational + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Operational + message: "Operational" + - type: Active + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Active + message: "Active" + - type: Started + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Started + message: "Started" + - type: Schedulable + status: "True" + lastTransitionTime: "2024-01-01T00:00:00Z" + reason: Schedulable + message: "Schedulable" + expectedStatusError: "Unsupported value" diff --git a/etcd/v1alpha1/types_pacemakercluster.go b/etcd/v1alpha1/types_pacemakercluster.go index b627413474b..9b40dd3e45e 100644 --- a/etcd/v1alpha1/types_pacemakercluster.go +++ b/etcd/v1alpha1/types_pacemakercluster.go @@ -14,6 +14,7 @@ const ( // Specifically, it aggregates the following conditions: // - ClusterInServiceConditionType // - ClusterNodeCountAsExpectedConditionType + // - ClusterFencingEnabledConditionType // - NodeHealthyConditionType (for each node) // When True, the cluster is healthy with reason "ClusterHealthy". // When False, the cluster is unhealthy with reason "ClusterUnhealthy". @@ -30,6 +31,16 @@ const ( // When True, the expected number of nodes are present with reason "AsExpected". // When False, the node count is incorrect with reason "InsufficientNodes" or "ExcessiveNodes". ClusterNodeCountAsExpectedConditionType = "NodeCountAsExpected" + + // ClusterFencingEnabledConditionType tracks whether STONITH (fencing) is enabled in the cluster. + // Fencing is the mechanism that isolates failed nodes by powering them off via BMC. In Two Node + // OpenShift with Fencing, fencing is mandatory — without it, a network partition leaves both nodes + // running and data diverges. If someone runs `pcs property set stonith-enabled=false`, fencing is + // globally disabled and the cluster cannot recover from node failures. + // When True, fencing is enabled with reason "FencingEnabled". This is the normal operating state. + // When False, fencing is globally disabled with reason "FencingDisabled". This is a critical state. + // When Unknown, the fencing-enabled property has not yet been observed with reason "Pending". + ClusterFencingEnabledConditionType = "FencingEnabled" ) // ClusterHealthy condition reasons @@ -72,6 +83,23 @@ const ( ClusterNodeCountAsExpectedReasonExcessiveNodes = "ExcessiveNodes" ) +// ClusterFencingEnabled condition reasons +const ( + // ClusterFencingEnabledReasonEnabled means fencing (STONITH) is enabled in the cluster. + // This is the normal and expected operating state for Two Node OpenShift with Fencing. + ClusterFencingEnabledReasonEnabled = "FencingEnabled" + + // ClusterFencingEnabledReasonDisabled means fencing (STONITH) is globally disabled in the cluster. + // Without fencing, the cluster cannot isolate failed nodes and etcd quorum recovery cannot happen. + // This is a critical state that should be investigated immediately. + ClusterFencingEnabledReasonDisabled = "FencingDisabled" + + // ClusterFencingEnabledReasonPending means the fencing-enabled property has not yet been observed + // by the status collector. This is expected to be temporary, for example immediately after upgrade + // or before the first successful status collection. + ClusterFencingEnabledReasonPending = "Pending" +) + // Node-level condition types for PacemakerCluster.status.nodes[].conditions const ( // NodeHealthyConditionType tracks the overall health of a node in the pacemaker cluster. @@ -462,6 +490,84 @@ const ( FencingMethodIPMI FencingMethod = "IPMI" ) +// PacemakerFenceEventAction represents the type of fencing action performed. +// Values are lowercase per Pacemaker's fence-event XML schema (fence-event-2.15.rng). +// +kubebuilder:validation:Enum=reboot;power-off;power-on +// +enum +type PacemakerFenceEventAction string + +const ( + // PacemakerFenceEventActionReboot is a fence action that power-cycles the target node. + PacemakerFenceEventActionReboot PacemakerFenceEventAction = "reboot" + + // PacemakerFenceEventActionPowerOff is a fence action that powers off the target node. + PacemakerFenceEventActionPowerOff PacemakerFenceEventAction = "power-off" + + // PacemakerFenceEventActionPowerOn is a fence action that powers on the target node. + PacemakerFenceEventActionPowerOn PacemakerFenceEventAction = "power-on" +) + +// PacemakerFenceEventStatus represents the outcome of a fencing operation. +// Values are lowercase per Pacemaker's fence-event XML schema (fence-event-2.15.rng). +// +kubebuilder:validation:Enum=success;failed;pending +// +enum +type PacemakerFenceEventStatus string + +const ( + // PacemakerFenceEventStatusSuccess means the fencing operation completed successfully. + PacemakerFenceEventStatusSuccess PacemakerFenceEventStatus = "success" + + // PacemakerFenceEventStatusFailed means the fencing operation failed. + PacemakerFenceEventStatusFailed PacemakerFenceEventStatus = "failed" + + // PacemakerFenceEventStatusPending means the fencing operation is currently in progress. + PacemakerFenceEventStatusPending PacemakerFenceEventStatus = "pending" +) + +// PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events +// are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during +// split-brain recovery or manually via stonith_admin. This struct captures the last fence event +// targeting a given node, providing visibility into what happened, its outcome, and whether it was +// triggered automatically by the cluster or manually by an operator. +type PacemakerFenceEvent struct { + // action is the type of fencing action that was performed or is in progress. + // Valid values are "reboot" (power-cycle the node), "power-off" (power off the node), + // and "power-on" (power on the node). + // +required + Action PacemakerFenceEventAction `json:"action,omitempty"` + + // status is the outcome of the fencing operation. + // Valid values are "success" (the operation completed successfully), "failed" (the operation + // failed), and "pending" (the operation is currently in progress). + // +required + Status PacemakerFenceEventStatus `json:"status,omitempty"` + + // client identifies the daemon or tool that requested the fencing operation. Typical values + // are "crmd" for automatic recovery initiated by the cluster resource manager, or + // "stonith_admin" for manual fencing initiated by an operator. This is useful for + // distinguishing "the cluster fenced itself for cause" from "someone fenced a node manually" + // during incident review. This field is optional and is omitted when the client is not + // reported by Pacemaker. When provided, the value must be between 1 and 256 characters. + // +kubebuilder:validation:MinLength=1 + // +kubebuilder:validation:MaxLength=256 + // +optional + Client string `json:"client,omitempty"` + + // lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. + // This timestamp is always present in the CIB fence history, including for pending operations + // where completedTime is not yet set. It must be a valid timestamp in RFC3339 format. + // +kubebuilder:validation:Format=date-time + // +required + LastUpdated *metav1.Time `json:"lastUpdated,omitempty"` + + // completedTime is the timestamp when the fencing operation completed. This field is optional + // and is omitted when the fencing operation is still in progress (status "pending") or when + // Pacemaker has not recorded a completion time. + // +kubebuilder:validation:Format=date-time + // +optional + CompletedTime *metav1.Time `json:"completedTime,omitempty"` +} + // +genclient // +genclient:nonNamespaced // +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object @@ -502,18 +608,20 @@ type PacemakerCluster struct { // +kubebuilder:validation:XValidation:rule="!has(oldSelf.lastUpdated) || self.lastUpdated >= oldSelf.lastUpdated",message="lastUpdated may not be set to an earlier timestamp" type PacemakerClusterStatus struct { // conditions represent the observations of the pacemaker cluster's current state. - // Known condition types are: "Healthy", "InService", "NodeCountAsExpected". + // Known condition types are: "Healthy", "InService", "NodeCountAsExpected", "FencingEnabled". // The "Healthy" condition is an aggregate that tracks the overall health of the cluster. // The "InService" condition tracks whether the cluster is in service (not in maintenance mode). // The "NodeCountAsExpected" condition tracks whether the expected number of nodes are present. - // Each of these conditions is required, so the array must contain at least 3 items. + // The "FencingEnabled" condition tracks whether STONITH (fencing) is enabled in the cluster. + // Each of these conditions is required, so the array must contain at least 4 items. // +listType=map // +listMapKey=type - // +kubebuilder:validation:MinItems=3 + // +kubebuilder:validation:MinItems=4 // +kubebuilder:validation:MaxItems=8 // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'Healthy')",message="conditions must contain a condition of type Healthy" // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'InService')",message="conditions must contain a condition of type InService" // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'NodeCountAsExpected')",message="conditions must contain a condition of type NodeCountAsExpected" + // +kubebuilder:validation:XValidation:rule="self.exists(c, c.type == 'FencingEnabled')",message="conditions must contain a condition of type FencingEnabled" // +required Conditions []metav1.Condition `json:"conditions,omitempty"` @@ -619,6 +727,15 @@ type PacemakerClusterNodeStatus struct { // +kubebuilder:validation:XValidation:rule="self.all(x, self.exists_one(y, x.name == y.name))",message="fencing agent names must be unique" // +required FencingAgents []PacemakerClusterFencingAgentStatus `json:"fencingAgents,omitempty"` + + // lastFenceEvent is the most recent fencing event targeting this node, as recorded in + // Pacemaker's fence history in the CIB. This captures the last time this node was fenced + // (or a fence attempt was made), including the action taken and its outcome. When the + // Clean condition is False, this field provides the concrete fencing context behind the + // unclean state. This field is optional and is omitted when no fencing event has been + // observed for this node. + // +optional + LastFenceEvent PacemakerFenceEvent `json:"lastFenceEvent,omitempty,omitzero"` } // PacemakerClusterFencingAgentStatus represents the status of a fencing agent that can fence a node. @@ -716,6 +833,28 @@ type PacemakerClusterResourceStatus struct { // Fencing agents are tracked separately in the node's fencingAgents field. // +required Name PacemakerClusterResourceName `json:"name,omitempty"` + + // failCount is the current failure count Pacemaker records for this resource on + // this node, as reported by the CIB. Pacemaker increments this count each time an + // operation for this resource fails, and resets it to zero when a `pcs resource + // cleanup` is performed. The value must be zero or greater. This field is optional + // and is omitted when the status collector has not yet observed a fail count for + // this resource, for example on a freshly bootstrapped cluster or for a resource + // that has never failed. + // +kubebuilder:validation:Minimum=0 + // +optional + FailCount *int32 `json:"failCount,omitempty"` + + // migrationThreshold is the configured number of failures after which Pacemaker + // will no longer attempt to run this resource on this node, as reported by the + // CIB. Without this value, failCount alone is uninterpretable — whether + // failCount 3 is alarming depends on whether the threshold is 5 or 1000000 + // (Pacemaker's default INFINITY). The value must be zero or greater. This field + // is optional and is omitted when the status collector has not yet observed a + // migration threshold for this resource. + // +kubebuilder:validation:Minimum=0 + // +optional + MigrationThreshold *int32 `json:"migrationThreshold,omitempty"` } // +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object diff --git a/etcd/v1alpha1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml b/etcd/v1alpha1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml index e26afefbb18..cfdeb64df4e 100644 --- a/etcd/v1alpha1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml +++ b/etcd/v1alpha1/zz_generated.crd-manifests/0000_25_etcd_01_pacemakerclusters.crd.yaml @@ -54,11 +54,12 @@ spec: conditions: description: |- conditions represent the observations of the pacemaker cluster's current state. - Known condition types are: "Healthy", "InService", "NodeCountAsExpected". + Known condition types are: "Healthy", "InService", "NodeCountAsExpected", "FencingEnabled". The "Healthy" condition is an aggregate that tracks the overall health of the cluster. The "InService" condition tracks whether the cluster is in service (not in maintenance mode). The "NodeCountAsExpected" condition tracks whether the expected number of nodes are present. - Each of these conditions is required, so the array must contain at least 3 items. + The "FencingEnabled" condition tracks whether STONITH (fencing) is enabled in the cluster. + Each of these conditions is required, so the array must contain at least 4 items. items: description: Condition contains details for one aspect of the current state of this API Resource. @@ -114,7 +115,7 @@ spec: - type type: object maxItems: 8 - minItems: 3 + minItems: 4 type: array x-kubernetes-list-map-keys: - type @@ -126,6 +127,8 @@ spec: rule: self.exists(c, c.type == 'InService') - message: conditions must contain a condition of type NodeCountAsExpected rule: self.exists(c, c.type == 'NodeCountAsExpected') + - message: conditions must contain a condition of type FencingEnabled + rule: self.exists(c, c.type == 'FencingEnabled') lastUpdated: description: |- lastUpdated is the timestamp when this status was last updated. This is useful for identifying @@ -437,6 +440,65 @@ spec: x-kubernetes-validations: - message: fencing agent names must be unique rule: self.all(x, self.exists_one(y, x.name == y.name)) + lastFenceEvent: + description: |- + lastFenceEvent is the most recent fencing event targeting this node, as recorded in + Pacemaker's fence history in the CIB. This captures the last time this node was fenced + (or a fence attempt was made), including the action taken and its outcome. When the + Clean condition is False, this field provides the concrete fencing context behind the + unclean state. This field is optional and is omitted when no fencing event has been + observed for this node. + properties: + action: + description: |- + action is the type of fencing action that was performed or is in progress. + Valid values are "reboot" (power-cycle the node), "power-off" (power off the node), + and "power-on" (power on the node). + enum: + - reboot + - power-off + - power-on + type: string + client: + description: |- + client identifies the daemon or tool that requested the fencing operation. Typical values + are "crmd" for automatic recovery initiated by the cluster resource manager, or + "stonith_admin" for manual fencing initiated by an operator. This is useful for + distinguishing "the cluster fenced itself for cause" from "someone fenced a node manually" + during incident review. This field is optional and is omitted when the client is not + reported by Pacemaker. When provided, the value must be between 1 and 256 characters. + maxLength: 256 + minLength: 1 + type: string + completedTime: + description: |- + completedTime is the timestamp when the fencing operation completed. This field is optional + and is omitted when the fencing operation is still in progress (status "pending") or when + Pacemaker has not recorded a completion time. + format: date-time + type: string + lastUpdated: + description: |- + lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. + This timestamp is always present in the CIB fence history, including for pending operations + where completedTime is not yet set. It must be a valid timestamp in RFC3339 format. + format: date-time + type: string + status: + description: |- + status is the outcome of the fencing operation. + Valid values are "success" (the operation completed successfully), "failed" (the operation + failed), and "pending" (the operation is currently in progress). + enum: + - success + - failed + - pending + type: string + required: + - action + - lastUpdated + - status + type: object nodeName: description: |- nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase @@ -571,6 +633,30 @@ spec: - message: conditions must contain a condition of type Schedulable rule: self.exists(c, c.type == 'Schedulable') + failCount: + description: |- + failCount is the current failure count Pacemaker records for this resource on + this node, as reported by the CIB. Pacemaker increments this count each time an + operation for this resource fails, and resets it to zero when a `pcs resource + cleanup` is performed. The value must be zero or greater. This field is optional + and is omitted when the status collector has not yet observed a fail count for + this resource, for example on a freshly bootstrapped cluster or for a resource + that has never failed. + format: int32 + minimum: 0 + type: integer + migrationThreshold: + description: |- + migrationThreshold is the configured number of failures after which Pacemaker + will no longer attempt to run this resource on this node, as reported by the + CIB. Without this value, failCount alone is uninterpretable — whether + failCount 3 is alarming depends on whether the threshold is 5 or 1000000 + (Pacemaker's default INFINITY). The value must be zero or greater. This field + is optional and is omitted when the status collector has not yet observed a + migration threshold for this resource. + format: int32 + minimum: 0 + type: integer name: description: |- name is the name of the pacemaker resource. diff --git a/etcd/v1alpha1/zz_generated.deepcopy.go b/etcd/v1alpha1/zz_generated.deepcopy.go index 17bf978510d..ded65796f35 100644 --- a/etcd/v1alpha1/zz_generated.deepcopy.go +++ b/etcd/v1alpha1/zz_generated.deepcopy.go @@ -122,6 +122,7 @@ func (in *PacemakerClusterNodeStatus) DeepCopyInto(out *PacemakerClusterNodeStat (*in)[i].DeepCopyInto(&(*out)[i]) } } + in.LastFenceEvent.DeepCopyInto(&out.LastFenceEvent) return } @@ -145,6 +146,16 @@ func (in *PacemakerClusterResourceStatus) DeepCopyInto(out *PacemakerClusterReso (*in)[i].DeepCopyInto(&(*out)[i]) } } + if in.FailCount != nil { + in, out := &in.FailCount, &out.FailCount + *out = new(int32) + **out = **in + } + if in.MigrationThreshold != nil { + in, out := &in.MigrationThreshold, &out.MigrationThreshold + *out = new(int32) + **out = **in + } return } @@ -193,6 +204,30 @@ func (in *PacemakerClusterStatus) DeepCopy() *PacemakerClusterStatus { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *PacemakerFenceEvent) DeepCopyInto(out *PacemakerFenceEvent) { + *out = *in + if in.LastUpdated != nil { + in, out := &in.LastUpdated, &out.LastUpdated + *out = (*in).DeepCopy() + } + if in.CompletedTime != nil { + in, out := &in.CompletedTime, &out.CompletedTime + *out = (*in).DeepCopy() + } + return +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PacemakerFenceEvent. +func (in *PacemakerFenceEvent) DeepCopy() *PacemakerFenceEvent { + if in == nil { + return nil + } + out := new(PacemakerFenceEvent) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *PacemakerNodeAddress) DeepCopyInto(out *PacemakerNodeAddress) { *out = *in diff --git a/etcd/v1alpha1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml b/etcd/v1alpha1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml index e867f16f114..fd17c01fc5a 100644 --- a/etcd/v1alpha1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml +++ b/etcd/v1alpha1/zz_generated.featuregated-crd-manifests/pacemakerclusters.etcd.openshift.io/DualReplica.yaml @@ -55,11 +55,12 @@ spec: conditions: description: |- conditions represent the observations of the pacemaker cluster's current state. - Known condition types are: "Healthy", "InService", "NodeCountAsExpected". + Known condition types are: "Healthy", "InService", "NodeCountAsExpected", "FencingEnabled". The "Healthy" condition is an aggregate that tracks the overall health of the cluster. The "InService" condition tracks whether the cluster is in service (not in maintenance mode). The "NodeCountAsExpected" condition tracks whether the expected number of nodes are present. - Each of these conditions is required, so the array must contain at least 3 items. + The "FencingEnabled" condition tracks whether STONITH (fencing) is enabled in the cluster. + Each of these conditions is required, so the array must contain at least 4 items. items: description: Condition contains details for one aspect of the current state of this API Resource. @@ -115,7 +116,7 @@ spec: - type type: object maxItems: 8 - minItems: 3 + minItems: 4 type: array x-kubernetes-list-map-keys: - type @@ -127,6 +128,8 @@ spec: rule: self.exists(c, c.type == 'InService') - message: conditions must contain a condition of type NodeCountAsExpected rule: self.exists(c, c.type == 'NodeCountAsExpected') + - message: conditions must contain a condition of type FencingEnabled + rule: self.exists(c, c.type == 'FencingEnabled') lastUpdated: description: |- lastUpdated is the timestamp when this status was last updated. This is useful for identifying @@ -438,6 +441,65 @@ spec: x-kubernetes-validations: - message: fencing agent names must be unique rule: self.all(x, self.exists_one(y, x.name == y.name)) + lastFenceEvent: + description: |- + lastFenceEvent is the most recent fencing event targeting this node, as recorded in + Pacemaker's fence history in the CIB. This captures the last time this node was fenced + (or a fence attempt was made), including the action taken and its outcome. When the + Clean condition is False, this field provides the concrete fencing context behind the + unclean state. This field is optional and is omitted when no fencing event has been + observed for this node. + properties: + action: + description: |- + action is the type of fencing action that was performed or is in progress. + Valid values are "reboot" (power-cycle the node), "power-off" (power off the node), + and "power-on" (power on the node). + enum: + - reboot + - power-off + - power-on + type: string + client: + description: |- + client identifies the daemon or tool that requested the fencing operation. Typical values + are "crmd" for automatic recovery initiated by the cluster resource manager, or + "stonith_admin" for manual fencing initiated by an operator. This is useful for + distinguishing "the cluster fenced itself for cause" from "someone fenced a node manually" + during incident review. This field is optional and is omitted when the client is not + reported by Pacemaker. When provided, the value must be between 1 and 256 characters. + maxLength: 256 + minLength: 1 + type: string + completedTime: + description: |- + completedTime is the timestamp when the fencing operation completed. This field is optional + and is omitted when the fencing operation is still in progress (status "pending") or when + Pacemaker has not recorded a completion time. + format: date-time + type: string + lastUpdated: + description: |- + lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. + This timestamp is always present in the CIB fence history, including for pending operations + where completedTime is not yet set. It must be a valid timestamp in RFC3339 format. + format: date-time + type: string + status: + description: |- + status is the outcome of the fencing operation. + Valid values are "success" (the operation completed successfully), "failed" (the operation + failed), and "pending" (the operation is currently in progress). + enum: + - success + - failed + - pending + type: string + required: + - action + - lastUpdated + - status + type: object nodeName: description: |- nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase @@ -572,6 +634,30 @@ spec: - message: conditions must contain a condition of type Schedulable rule: self.exists(c, c.type == 'Schedulable') + failCount: + description: |- + failCount is the current failure count Pacemaker records for this resource on + this node, as reported by the CIB. Pacemaker increments this count each time an + operation for this resource fails, and resets it to zero when a `pcs resource + cleanup` is performed. The value must be zero or greater. This field is optional + and is omitted when the status collector has not yet observed a fail count for + this resource, for example on a freshly bootstrapped cluster or for a resource + that has never failed. + format: int32 + minimum: 0 + type: integer + migrationThreshold: + description: |- + migrationThreshold is the configured number of failures after which Pacemaker + will no longer attempt to run this resource on this node, as reported by the + CIB. Without this value, failCount alone is uninterpretable — whether + failCount 3 is alarming depends on whether the threshold is 5 or 1000000 + (Pacemaker's default INFINITY). The value must be zero or greater. This field + is optional and is omitted when the status collector has not yet observed a + migration threshold for this resource. + format: int32 + minimum: 0 + type: integer name: description: |- name is the name of the pacemaker resource. diff --git a/etcd/v1alpha1/zz_generated.model_name.go b/etcd/v1alpha1/zz_generated.model_name.go index 11fac8dad4c..d534cda15be 100644 --- a/etcd/v1alpha1/zz_generated.model_name.go +++ b/etcd/v1alpha1/zz_generated.model_name.go @@ -35,6 +35,11 @@ func (in PacemakerClusterStatus) OpenAPIModelName() string { return "com.github.openshift.api.etcd.v1alpha1.PacemakerClusterStatus" } +// OpenAPIModelName returns the OpenAPI model name for this type. +func (in PacemakerFenceEvent) OpenAPIModelName() string { + return "com.github.openshift.api.etcd.v1alpha1.PacemakerFenceEvent" +} + // OpenAPIModelName returns the OpenAPI model name for this type. func (in PacemakerNodeAddress) OpenAPIModelName() string { return "com.github.openshift.api.etcd.v1alpha1.PacemakerNodeAddress" diff --git a/etcd/v1alpha1/zz_generated.swagger_doc_generated.go b/etcd/v1alpha1/zz_generated.swagger_doc_generated.go index dc6f224288e..994a0067c3b 100644 --- a/etcd/v1alpha1/zz_generated.swagger_doc_generated.go +++ b/etcd/v1alpha1/zz_generated.swagger_doc_generated.go @@ -43,12 +43,13 @@ func (PacemakerClusterList) SwaggerDoc() map[string]string { } var map_PacemakerClusterNodeStatus = map[string]string{ - "": "PacemakerClusterNodeStatus represents the status of a single node in the pacemaker cluster including the node's conditions and the health of critical resources running on that node.", - "conditions": "conditions represent the observations of the node's current state. Known condition types are: \"Healthy\", \"Online\", \"InService\", \"Active\", \"Ready\", \"Clean\", \"Member\", \"FencingAvailable\", \"FencingHealthy\". The \"Healthy\" condition is an aggregate that tracks the overall health of the node. The \"Online\" condition tracks whether the node is online. The \"InService\" condition tracks whether the node is in service (not in maintenance mode). The \"Active\" condition tracks whether the node is active (not in standby mode). The \"Ready\" condition tracks whether the node is ready (not in a pending state). The \"Clean\" condition tracks whether the node is in a clean (status known) state. The \"Member\" condition tracks whether the node is a member of the cluster. The \"FencingAvailable\" condition tracks whether this node can be fenced by at least one healthy agent. The \"FencingHealthy\" condition tracks whether all fencing agents for this node are healthy. Each of these conditions is required, so the array must contain at least 9 items.", - "nodeName": "nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase RFC 1123 subdomain consisting of lowercase alphanumeric characters, '-' or '.', starting and ending with an alphanumeric character, and be at most 253 characters in length.", - "addresses": "addresses is a list of IP addresses for the node. Pacemaker allows multiple IP addresses for Corosync communication between nodes. The first address in this list is used for IP-based peer URLs for etcd membership. Each address must be a valid global unicast IPv4 or IPv6 address in canonical form (e.g., \"192.168.1.1\" not \"192.168.001.001\", or \"2001:db8::1\" not \"2001:0db8::1\"). This excludes loopback, link-local, and multicast addresses.", - "resources": "resources contains the status of pacemaker resources scheduled on this node. Each resource entry includes the resource name and its health conditions. For Two Node OpenShift with Fencing, we track Kubelet and Etcd resources per node. Both resources are required to be present, so the array must contain at least 2 items. Valid resource names are \"Kubelet\" and \"Etcd\". Fencing agents are tracked separately in the fencingAgents field.", - "fencingAgents": "fencingAgents contains the status of fencing agents that can fence this node. Unlike resources (which are scheduled to run on this node), fencing agents are mapped to the node they can fence (their target), not the node where monitoring operations run. Each fencing agent entry includes a unique name, fencing type, target node, and health conditions. A node is considered fence-capable if at least one fencing agent is healthy. A healthy node is expected to have at least 1 fencing agent, but the list may be empty when fencing agent discovery fails. Names must be unique within this array.", + "": "PacemakerClusterNodeStatus represents the status of a single node in the pacemaker cluster including the node's conditions and the health of critical resources running on that node.", + "conditions": "conditions represent the observations of the node's current state. Known condition types are: \"Healthy\", \"Online\", \"InService\", \"Active\", \"Ready\", \"Clean\", \"Member\", \"FencingAvailable\", \"FencingHealthy\". The \"Healthy\" condition is an aggregate that tracks the overall health of the node. The \"Online\" condition tracks whether the node is online. The \"InService\" condition tracks whether the node is in service (not in maintenance mode). The \"Active\" condition tracks whether the node is active (not in standby mode). The \"Ready\" condition tracks whether the node is ready (not in a pending state). The \"Clean\" condition tracks whether the node is in a clean (status known) state. The \"Member\" condition tracks whether the node is a member of the cluster. The \"FencingAvailable\" condition tracks whether this node can be fenced by at least one healthy agent. The \"FencingHealthy\" condition tracks whether all fencing agents for this node are healthy. Each of these conditions is required, so the array must contain at least 9 items.", + "nodeName": "nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase RFC 1123 subdomain consisting of lowercase alphanumeric characters, '-' or '.', starting and ending with an alphanumeric character, and be at most 253 characters in length.", + "addresses": "addresses is a list of IP addresses for the node. Pacemaker allows multiple IP addresses for Corosync communication between nodes. The first address in this list is used for IP-based peer URLs for etcd membership. Each address must be a valid global unicast IPv4 or IPv6 address in canonical form (e.g., \"192.168.1.1\" not \"192.168.001.001\", or \"2001:db8::1\" not \"2001:0db8::1\"). This excludes loopback, link-local, and multicast addresses.", + "resources": "resources contains the status of pacemaker resources scheduled on this node. Each resource entry includes the resource name and its health conditions. For Two Node OpenShift with Fencing, we track Kubelet and Etcd resources per node. Both resources are required to be present, so the array must contain at least 2 items. Valid resource names are \"Kubelet\" and \"Etcd\". Fencing agents are tracked separately in the fencingAgents field.", + "fencingAgents": "fencingAgents contains the status of fencing agents that can fence this node. Unlike resources (which are scheduled to run on this node), fencing agents are mapped to the node they can fence (their target), not the node where monitoring operations run. Each fencing agent entry includes a unique name, fencing type, target node, and health conditions. A node is considered fence-capable if at least one fencing agent is healthy. A healthy node is expected to have at least 1 fencing agent, but the list may be empty when fencing agent discovery fails. Names must be unique within this array.", + "lastFenceEvent": "lastFenceEvent is the most recent fencing event targeting this node, as recorded in Pacemaker's fence history in the CIB. This captures the last time this node was fenced (or a fence attempt was made), including the action taken and its outcome. When the Clean condition is False, this field provides the concrete fencing context behind the unclean state. This field is optional and is omitted when no fencing event has been observed for this node.", } func (PacemakerClusterNodeStatus) SwaggerDoc() map[string]string { @@ -56,9 +57,11 @@ func (PacemakerClusterNodeStatus) SwaggerDoc() map[string]string { } var map_PacemakerClusterResourceStatus = map[string]string{ - "": "PacemakerClusterResourceStatus represents the status of a pacemaker resource scheduled on a node. A pacemaker resource is a unit of work managed by pacemaker. In pacemaker terminology, resources are services or applications that pacemaker monitors, starts, stops, and moves between nodes to maintain high availability. For Two Node OpenShift with Fencing, we track two resources per node:\n - Kubelet (the Kubernetes node agent and a prerequisite for etcd)\n - Etcd (the distributed key-value store)\n\nFencing agents are tracked separately in the fencingAgents field because they are mapped to their target node (the node they can fence), not the node where monitoring operations are scheduled.", - "conditions": "conditions represent the observations of the resource's current state. Known condition types are: \"Healthy\", \"InService\", \"Managed\", \"Enabled\", \"Operational\", \"Active\", \"Started\", \"Schedulable\". The \"Healthy\" condition is an aggregate that tracks the overall health of the resource. The \"InService\" condition tracks whether the resource is in service (not in maintenance mode). The \"Managed\" condition tracks whether the resource is managed by pacemaker. The \"Enabled\" condition tracks whether the resource is enabled. The \"Operational\" condition tracks whether the resource is operational (not failed). The \"Active\" condition tracks whether the resource is active (available to be used). The \"Started\" condition tracks whether the resource is started. The \"Schedulable\" condition tracks whether the resource is schedulable (not blocked). Each of these conditions is required, so the array must contain at least 8 items.", - "name": "name is the name of the pacemaker resource. Valid values are \"Kubelet\" and \"Etcd\". The Kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments. The Etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations. Fencing agents are tracked separately in the node's fencingAgents field.", + "": "PacemakerClusterResourceStatus represents the status of a pacemaker resource scheduled on a node. A pacemaker resource is a unit of work managed by pacemaker. In pacemaker terminology, resources are services or applications that pacemaker monitors, starts, stops, and moves between nodes to maintain high availability. For Two Node OpenShift with Fencing, we track two resources per node:\n - Kubelet (the Kubernetes node agent and a prerequisite for etcd)\n - Etcd (the distributed key-value store)\n\nFencing agents are tracked separately in the fencingAgents field because they are mapped to their target node (the node they can fence), not the node where monitoring operations are scheduled.", + "conditions": "conditions represent the observations of the resource's current state. Known condition types are: \"Healthy\", \"InService\", \"Managed\", \"Enabled\", \"Operational\", \"Active\", \"Started\", \"Schedulable\". The \"Healthy\" condition is an aggregate that tracks the overall health of the resource. The \"InService\" condition tracks whether the resource is in service (not in maintenance mode). The \"Managed\" condition tracks whether the resource is managed by pacemaker. The \"Enabled\" condition tracks whether the resource is enabled. The \"Operational\" condition tracks whether the resource is operational (not failed). The \"Active\" condition tracks whether the resource is active (available to be used). The \"Started\" condition tracks whether the resource is started. The \"Schedulable\" condition tracks whether the resource is schedulable (not blocked). Each of these conditions is required, so the array must contain at least 8 items.", + "name": "name is the name of the pacemaker resource. Valid values are \"Kubelet\" and \"Etcd\". The Kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments. The Etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations. Fencing agents are tracked separately in the node's fencingAgents field.", + "failCount": "failCount is the current failure count Pacemaker records for this resource on this node, as reported by the CIB. Pacemaker increments this count each time an operation for this resource fails, and resets it to zero when a `pcs resource cleanup` is performed. The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a fail count for this resource, for example on a freshly bootstrapped cluster or for a resource that has never failed.", + "migrationThreshold": "migrationThreshold is the configured number of failures after which Pacemaker will no longer attempt to run this resource on this node, as reported by the CIB. Without this value, failCount alone is uninterpretable — whether failCount 3 is alarming depends on whether the threshold is 5 or 1000000 (Pacemaker's default INFINITY). The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a migration threshold for this resource.", } func (PacemakerClusterResourceStatus) SwaggerDoc() map[string]string { @@ -67,7 +70,7 @@ func (PacemakerClusterResourceStatus) SwaggerDoc() map[string]string { var map_PacemakerClusterStatus = map[string]string{ "": "PacemakerClusterStatus contains the actual pacemaker cluster status information. As part of validating the status object, we need to ensure that the lastUpdated timestamp may not be set to an earlier timestamp than the current value. The validation rule checks if oldSelf has lastUpdated before comparing, to handle the initial status creation case.", - "conditions": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. Each of these conditions is required, so the array must contain at least 3 items.", + "conditions": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\", \"FencingEnabled\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. The \"FencingEnabled\" condition tracks whether STONITH (fencing) is enabled in the cluster. Each of these conditions is required, so the array must contain at least 4 items.", "lastUpdated": "lastUpdated is the timestamp when this status was last updated. This is useful for identifying stale status reports. It must be a valid timestamp in RFC3339 format. Once set, this field cannot be removed and cannot be set to an earlier timestamp than the current value.", "nodes": "nodes provides detailed status for each control-plane node in the Pacemaker cluster. While Pacemaker supports up to 32 nodes, the limit is set to 5 (max OpenShift control-plane nodes). For Two Node OpenShift with Fencing, exactly 2 nodes are expected in a healthy cluster. An empty list indicates a catastrophic failure where Pacemaker reports no nodes.", } @@ -76,6 +79,19 @@ func (PacemakerClusterStatus) SwaggerDoc() map[string]string { return map_PacemakerClusterStatus } +var map_PacemakerFenceEvent = map[string]string{ + "": "PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during split-brain recovery or manually via stonith_admin. This struct captures the last fence event targeting a given node, providing visibility into what happened, its outcome, and whether it was triggered automatically by the cluster or manually by an operator.", + "action": "action is the type of fencing action that was performed or is in progress. Valid values are \"reboot\" (power-cycle the node), \"power-off\" (power off the node), and \"power-on\" (power on the node).", + "status": "status is the outcome of the fencing operation. Valid values are \"success\" (the operation completed successfully), \"failed\" (the operation failed), and \"pending\" (the operation is currently in progress).", + "client": "client identifies the daemon or tool that requested the fencing operation. Typical values are \"crmd\" for automatic recovery initiated by the cluster resource manager, or \"stonith_admin\" for manual fencing initiated by an operator. This is useful for distinguishing \"the cluster fenced itself for cause\" from \"someone fenced a node manually\" during incident review. This field is optional and is omitted when the client is not reported by Pacemaker. When provided, the value must be between 1 and 256 characters.", + "lastUpdated": "lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. This timestamp is always present in the CIB fence history, including for pending operations where completedTime is not yet set. It must be a valid timestamp in RFC3339 format.", + "completedTime": "completedTime is the timestamp when the fencing operation completed. This field is optional and is omitted when the fencing operation is still in progress (status \"pending\") or when Pacemaker has not recorded a completion time.", +} + +func (PacemakerFenceEvent) SwaggerDoc() map[string]string { + return map_PacemakerFenceEvent +} + var map_PacemakerNodeAddress = map[string]string{ "": "PacemakerNodeAddress contains information for a node's address. This is similar to corev1.NodeAddress but adds validation for IP addresses.", "type": "type is the type of node address. Currently only \"InternalIP\" is supported.", diff --git a/openapi/generated_openapi/zz_generated.openapi.go b/openapi/generated_openapi/zz_generated.openapi.go index c60b07daff0..3fb809e946e 100644 --- a/openapi/generated_openapi/zz_generated.openapi.go +++ b/openapi/generated_openapi/zz_generated.openapi.go @@ -671,6 +671,7 @@ func GetOpenAPIDefinitions(ref common.ReferenceCallback) map[string]common.OpenA etcdv1.PacemakerClusterNodeStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1_PacemakerClusterNodeStatus(ref), etcdv1.PacemakerClusterResourceStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1_PacemakerClusterResourceStatus(ref), etcdv1.PacemakerClusterStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1_PacemakerClusterStatus(ref), + etcdv1.PacemakerFenceEvent{}.OpenAPIModelName(): schema_openshift_api_etcd_v1_PacemakerFenceEvent(ref), etcdv1.PacemakerNodeAddress{}.OpenAPIModelName(): schema_openshift_api_etcd_v1_PacemakerNodeAddress(ref), etcdv1alpha1.PacemakerCluster{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerCluster(ref), etcdv1alpha1.PacemakerClusterFencingAgentStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerClusterFencingAgentStatus(ref), @@ -678,6 +679,7 @@ func GetOpenAPIDefinitions(ref common.ReferenceCallback) map[string]common.OpenA etcdv1alpha1.PacemakerClusterNodeStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerClusterNodeStatus(ref), etcdv1alpha1.PacemakerClusterResourceStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerClusterResourceStatus(ref), etcdv1alpha1.PacemakerClusterStatus{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerClusterStatus(ref), + etcdv1alpha1.PacemakerFenceEvent{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerFenceEvent(ref), etcdv1alpha1.PacemakerNodeAddress{}.OpenAPIModelName(): schema_openshift_api_etcd_v1alpha1_PacemakerNodeAddress(ref), examplev1.CELUnion{}.OpenAPIModelName(): schema_openshift_api_example_v1_CELUnion(ref), examplev1.EvolvingUnion{}.OpenAPIModelName(): schema_openshift_api_example_v1_EvolvingUnion(ref), @@ -30337,12 +30339,19 @@ func schema_openshift_api_etcd_v1_PacemakerClusterNodeStatus(ref common.Referenc }, }, }, + "lastFenceEvent": { + SchemaProps: spec.SchemaProps{ + Description: "lastFenceEvent is the most recent fencing event targeting this node, as recorded in Pacemaker's fence history in the CIB. This captures the last time this node was fenced (or a fence attempt was made), including the action taken and its outcome. When the Clean condition is False, this field provides the concrete fencing context behind the unclean state. This field is optional and is omitted when no fencing event has been observed for this node.", + Default: map[string]interface{}{}, + Ref: ref(etcdv1.PacemakerFenceEvent{}.OpenAPIModelName()), + }, + }, }, Required: []string{"conditions", "nodeName", "addresses", "resources", "fencingAgents"}, }, }, Dependencies: []string{ - etcdv1.PacemakerClusterFencingAgentStatus{}.OpenAPIModelName(), etcdv1.PacemakerClusterResourceStatus{}.OpenAPIModelName(), etcdv1.PacemakerNodeAddress{}.OpenAPIModelName(), metav1.Condition{}.OpenAPIModelName()}, + etcdv1.PacemakerClusterFencingAgentStatus{}.OpenAPIModelName(), etcdv1.PacemakerClusterResourceStatus{}.OpenAPIModelName(), etcdv1.PacemakerFenceEvent{}.OpenAPIModelName(), etcdv1.PacemakerNodeAddress{}.OpenAPIModelName(), metav1.Condition{}.OpenAPIModelName()}, } } @@ -30383,6 +30392,20 @@ func schema_openshift_api_etcd_v1_PacemakerClusterResourceStatus(ref common.Refe Enum: []interface{}{"Etcd", "Kubelet"}, }, }, + "failCount": { + SchemaProps: spec.SchemaProps{ + Description: "failCount is the current failure count Pacemaker records for this resource on this node, as reported by the CIB. Pacemaker increments this count each time an operation for this resource fails, and resets it to zero when a `pcs resource cleanup` is performed. The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a fail count for this resource, for example on a freshly bootstrapped cluster or for a resource that has never failed.", + Type: []string{"integer"}, + Format: "int32", + }, + }, + "migrationThreshold": { + SchemaProps: spec.SchemaProps{ + Description: "migrationThreshold is the configured number of failures after which Pacemaker will no longer attempt to run this resource on this node, as reported by the CIB. Without this value, failCount alone is uninterpretable — whether failCount 3 is alarming depends on whether the threshold is 5 or 1000000 (Pacemaker's default INFINITY). The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a migration threshold for this resource.", + Type: []string{"integer"}, + Format: "int32", + }, + }, }, Required: []string{"conditions", "name"}, }, @@ -30409,7 +30432,7 @@ func schema_openshift_api_etcd_v1_PacemakerClusterStatus(ref common.ReferenceCal }, }, SchemaProps: spec.SchemaProps{ - Description: "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. Each of these conditions is required, so the array must contain at least 3 items.", + Description: "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\", \"FencingEnabled\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. The \"FencingEnabled\" condition tracks whether STONITH (fencing) is enabled in the cluster. Each of these conditions is required, so the array must contain at least 4 items.", Type: []string{"array"}, Items: &spec.SchemaOrArray{ Schema: &spec.Schema{ @@ -30458,6 +30481,57 @@ func schema_openshift_api_etcd_v1_PacemakerClusterStatus(ref common.ReferenceCal } } +func schema_openshift_api_etcd_v1_PacemakerFenceEvent(ref common.ReferenceCallback) common.OpenAPIDefinition { + return common.OpenAPIDefinition{ + Schema: spec.Schema{ + SchemaProps: spec.SchemaProps{ + Description: "PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during split-brain recovery or manually via stonith_admin. This struct captures the last fence event targeting a given node, providing visibility into what happened, its outcome, and whether it was triggered automatically by the cluster or manually by an operator.", + Type: []string{"object"}, + Properties: map[string]spec.Schema{ + "action": { + SchemaProps: spec.SchemaProps{ + Description: "action is the type of fencing action that was performed or is in progress. Valid values are \"reboot\" (power-cycle the node), \"power-off\" (power off the node), and \"power-on\" (power on the node).\n\nPossible enum values:\n - `\"power-off\"` is a fence action that powers off the target node.\n - `\"power-on\"` is a fence action that powers on the target node.\n - `\"reboot\"` is a fence action that power-cycles the target node.", + Type: []string{"string"}, + Format: "", + Enum: []interface{}{"power-off", "power-on", "reboot"}, + }, + }, + "status": { + SchemaProps: spec.SchemaProps{ + Description: "status is the outcome of the fencing operation. Valid values are \"success\" (the operation completed successfully), \"failed\" (the operation failed), and \"pending\" (the operation is currently in progress).\n\nPossible enum values:\n - `\"failed\"` means the fencing operation failed.\n - `\"pending\"` means the fencing operation is currently in progress.\n - `\"success\"` means the fencing operation completed successfully.", + Type: []string{"string"}, + Format: "", + Enum: []interface{}{"failed", "pending", "success"}, + }, + }, + "client": { + SchemaProps: spec.SchemaProps{ + Description: "client identifies the daemon or tool that requested the fencing operation. Typical values are \"crmd\" for automatic recovery initiated by the cluster resource manager, or \"stonith_admin\" for manual fencing initiated by an operator. This is useful for distinguishing \"the cluster fenced itself for cause\" from \"someone fenced a node manually\" during incident review. This field is optional and is omitted when the client is not reported by Pacemaker. When provided, the value must be between 1 and 256 characters.", + Type: []string{"string"}, + Format: "", + }, + }, + "lastUpdated": { + SchemaProps: spec.SchemaProps{ + Description: "lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. This timestamp is always present in the CIB fence history, including for pending operations where completedTime is not yet set. It must be a valid timestamp in RFC3339 format.", + Ref: ref(metav1.Time{}.OpenAPIModelName()), + }, + }, + "completedTime": { + SchemaProps: spec.SchemaProps{ + Description: "completedTime is the timestamp when the fencing operation completed. This field is optional and is omitted when the fencing operation is still in progress (status \"pending\") or when Pacemaker has not recorded a completion time.", + Ref: ref(metav1.Time{}.OpenAPIModelName()), + }, + }, + }, + Required: []string{"action", "status", "lastUpdated"}, + }, + }, + Dependencies: []string{ + metav1.Time{}.OpenAPIModelName()}, + } +} + func schema_openshift_api_etcd_v1_PacemakerNodeAddress(ref common.ReferenceCallback) common.OpenAPIDefinition { return common.OpenAPIDefinition{ Schema: spec.Schema{ @@ -30734,12 +30808,19 @@ func schema_openshift_api_etcd_v1alpha1_PacemakerClusterNodeStatus(ref common.Re }, }, }, + "lastFenceEvent": { + SchemaProps: spec.SchemaProps{ + Description: "lastFenceEvent is the most recent fencing event targeting this node, as recorded in Pacemaker's fence history in the CIB. This captures the last time this node was fenced (or a fence attempt was made), including the action taken and its outcome. When the Clean condition is False, this field provides the concrete fencing context behind the unclean state. This field is optional and is omitted when no fencing event has been observed for this node.", + Default: map[string]interface{}{}, + Ref: ref(etcdv1alpha1.PacemakerFenceEvent{}.OpenAPIModelName()), + }, + }, }, Required: []string{"conditions", "nodeName", "addresses", "resources", "fencingAgents"}, }, }, Dependencies: []string{ - etcdv1alpha1.PacemakerClusterFencingAgentStatus{}.OpenAPIModelName(), etcdv1alpha1.PacemakerClusterResourceStatus{}.OpenAPIModelName(), etcdv1alpha1.PacemakerNodeAddress{}.OpenAPIModelName(), metav1.Condition{}.OpenAPIModelName()}, + etcdv1alpha1.PacemakerClusterFencingAgentStatus{}.OpenAPIModelName(), etcdv1alpha1.PacemakerClusterResourceStatus{}.OpenAPIModelName(), etcdv1alpha1.PacemakerFenceEvent{}.OpenAPIModelName(), etcdv1alpha1.PacemakerNodeAddress{}.OpenAPIModelName(), metav1.Condition{}.OpenAPIModelName()}, } } @@ -30780,6 +30861,20 @@ func schema_openshift_api_etcd_v1alpha1_PacemakerClusterResourceStatus(ref commo Enum: []interface{}{"Etcd", "Kubelet"}, }, }, + "failCount": { + SchemaProps: spec.SchemaProps{ + Description: "failCount is the current failure count Pacemaker records for this resource on this node, as reported by the CIB. Pacemaker increments this count each time an operation for this resource fails, and resets it to zero when a `pcs resource cleanup` is performed. The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a fail count for this resource, for example on a freshly bootstrapped cluster or for a resource that has never failed.", + Type: []string{"integer"}, + Format: "int32", + }, + }, + "migrationThreshold": { + SchemaProps: spec.SchemaProps{ + Description: "migrationThreshold is the configured number of failures after which Pacemaker will no longer attempt to run this resource on this node, as reported by the CIB. Without this value, failCount alone is uninterpretable — whether failCount 3 is alarming depends on whether the threshold is 5 or 1000000 (Pacemaker's default INFINITY). The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a migration threshold for this resource.", + Type: []string{"integer"}, + Format: "int32", + }, + }, }, Required: []string{"conditions", "name"}, }, @@ -30806,7 +30901,7 @@ func schema_openshift_api_etcd_v1alpha1_PacemakerClusterStatus(ref common.Refere }, }, SchemaProps: spec.SchemaProps{ - Description: "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. Each of these conditions is required, so the array must contain at least 3 items.", + Description: "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\", \"FencingEnabled\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. The \"FencingEnabled\" condition tracks whether STONITH (fencing) is enabled in the cluster. Each of these conditions is required, so the array must contain at least 4 items.", Type: []string{"array"}, Items: &spec.SchemaOrArray{ Schema: &spec.Schema{ @@ -30855,6 +30950,57 @@ func schema_openshift_api_etcd_v1alpha1_PacemakerClusterStatus(ref common.Refere } } +func schema_openshift_api_etcd_v1alpha1_PacemakerFenceEvent(ref common.ReferenceCallback) common.OpenAPIDefinition { + return common.OpenAPIDefinition{ + Schema: spec.Schema{ + SchemaProps: spec.SchemaProps{ + Description: "PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during split-brain recovery or manually via stonith_admin. This struct captures the last fence event targeting a given node, providing visibility into what happened, its outcome, and whether it was triggered automatically by the cluster or manually by an operator.", + Type: []string{"object"}, + Properties: map[string]spec.Schema{ + "action": { + SchemaProps: spec.SchemaProps{ + Description: "action is the type of fencing action that was performed or is in progress. Valid values are \"reboot\" (power-cycle the node), \"power-off\" (power off the node), and \"power-on\" (power on the node).\n\nPossible enum values:\n - `\"power-off\"` is a fence action that powers off the target node.\n - `\"power-on\"` is a fence action that powers on the target node.\n - `\"reboot\"` is a fence action that power-cycles the target node.", + Type: []string{"string"}, + Format: "", + Enum: []interface{}{"power-off", "power-on", "reboot"}, + }, + }, + "status": { + SchemaProps: spec.SchemaProps{ + Description: "status is the outcome of the fencing operation. Valid values are \"success\" (the operation completed successfully), \"failed\" (the operation failed), and \"pending\" (the operation is currently in progress).\n\nPossible enum values:\n - `\"failed\"` means the fencing operation failed.\n - `\"pending\"` means the fencing operation is currently in progress.\n - `\"success\"` means the fencing operation completed successfully.", + Type: []string{"string"}, + Format: "", + Enum: []interface{}{"failed", "pending", "success"}, + }, + }, + "client": { + SchemaProps: spec.SchemaProps{ + Description: "client identifies the daemon or tool that requested the fencing operation. Typical values are \"crmd\" for automatic recovery initiated by the cluster resource manager, or \"stonith_admin\" for manual fencing initiated by an operator. This is useful for distinguishing \"the cluster fenced itself for cause\" from \"someone fenced a node manually\" during incident review. This field is optional and is omitted when the client is not reported by Pacemaker. When provided, the value must be between 1 and 256 characters.", + Type: []string{"string"}, + Format: "", + }, + }, + "lastUpdated": { + SchemaProps: spec.SchemaProps{ + Description: "lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. This timestamp is always present in the CIB fence history, including for pending operations where completedTime is not yet set. It must be a valid timestamp in RFC3339 format.", + Ref: ref(metav1.Time{}.OpenAPIModelName()), + }, + }, + "completedTime": { + SchemaProps: spec.SchemaProps{ + Description: "completedTime is the timestamp when the fencing operation completed. This field is optional and is omitted when the fencing operation is still in progress (status \"pending\") or when Pacemaker has not recorded a completion time.", + Ref: ref(metav1.Time{}.OpenAPIModelName()), + }, + }, + }, + Required: []string{"action", "status", "lastUpdated"}, + }, + }, + Dependencies: []string{ + metav1.Time{}.OpenAPIModelName()}, + } +} + func schema_openshift_api_etcd_v1alpha1_PacemakerNodeAddress(ref common.ReferenceCallback) common.OpenAPIDefinition { return common.OpenAPIDefinition{ Schema: spec.Schema{ diff --git a/openapi/openapi.json b/openapi/openapi.json index 832f0146d88..1f0da5d872c 100644 --- a/openapi/openapi.json +++ b/openapi/openapi.json @@ -16887,6 +16887,11 @@ ], "x-kubernetes-list-type": "map" }, + "lastFenceEvent": { + "description": "lastFenceEvent is the most recent fencing event targeting this node, as recorded in Pacemaker's fence history in the CIB. This captures the last time this node was fenced (or a fence attempt was made), including the action taken and its outcome. When the Clean condition is False, this field provides the concrete fencing context behind the unclean state. This field is optional and is omitted when no fencing event has been observed for this node.", + "default": {}, + "$ref": "#/definitions/com.github.openshift.api.etcd.v1.PacemakerFenceEvent" + }, "nodeName": { "description": "nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase RFC 1123 subdomain consisting of lowercase alphanumeric characters, '-' or '.', starting and ending with an alphanumeric character, and be at most 253 characters in length.", "type": "string" @@ -16925,6 +16930,16 @@ ], "x-kubernetes-list-type": "map" }, + "failCount": { + "description": "failCount is the current failure count Pacemaker records for this resource on this node, as reported by the CIB. Pacemaker increments this count each time an operation for this resource fails, and resets it to zero when a `pcs resource cleanup` is performed. The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a fail count for this resource, for example on a freshly bootstrapped cluster or for a resource that has never failed.", + "type": "integer", + "format": "int32" + }, + "migrationThreshold": { + "description": "migrationThreshold is the configured number of failures after which Pacemaker will no longer attempt to run this resource on this node, as reported by the CIB. Without this value, failCount alone is uninterpretable — whether failCount 3 is alarming depends on whether the threshold is 5 or 1000000 (Pacemaker's default INFINITY). The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a migration threshold for this resource.", + "type": "integer", + "format": "int32" + }, "name": { "description": "name is the name of the pacemaker resource. Valid values are \"Kubelet\" and \"Etcd\". The Kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments. The Etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations. Fencing agents are tracked separately in the node's fencingAgents field.\n\nPossible enum values:\n - `\"Etcd\"` is the etcd pacemaker resource. The etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations.\n - `\"Kubelet\"` is the kubelet pacemaker resource. The kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments.", "type": "string", @@ -16945,7 +16960,7 @@ ], "properties": { "conditions": { - "description": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. Each of these conditions is required, so the array must contain at least 3 items.", + "description": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\", \"FencingEnabled\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. The \"FencingEnabled\" condition tracks whether STONITH (fencing) is enabled in the cluster. Each of these conditions is required, so the array must contain at least 4 items.", "type": "array", "items": { "default": {}, @@ -16974,6 +16989,47 @@ } } }, + "com.github.openshift.api.etcd.v1.PacemakerFenceEvent": { + "description": "PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during split-brain recovery or manually via stonith_admin. This struct captures the last fence event targeting a given node, providing visibility into what happened, its outcome, and whether it was triggered automatically by the cluster or manually by an operator.", + "type": "object", + "required": [ + "action", + "status", + "lastUpdated" + ], + "properties": { + "action": { + "description": "action is the type of fencing action that was performed or is in progress. Valid values are \"reboot\" (power-cycle the node), \"power-off\" (power off the node), and \"power-on\" (power on the node).\n\nPossible enum values:\n - `\"power-off\"` is a fence action that powers off the target node.\n - `\"power-on\"` is a fence action that powers on the target node.\n - `\"reboot\"` is a fence action that power-cycles the target node.", + "type": "string", + "enum": [ + "power-off", + "power-on", + "reboot" + ] + }, + "client": { + "description": "client identifies the daemon or tool that requested the fencing operation. Typical values are \"crmd\" for automatic recovery initiated by the cluster resource manager, or \"stonith_admin\" for manual fencing initiated by an operator. This is useful for distinguishing \"the cluster fenced itself for cause\" from \"someone fenced a node manually\" during incident review. This field is optional and is omitted when the client is not reported by Pacemaker. When provided, the value must be between 1 and 256 characters.", + "type": "string" + }, + "completedTime": { + "description": "completedTime is the timestamp when the fencing operation completed. This field is optional and is omitted when the fencing operation is still in progress (status \"pending\") or when Pacemaker has not recorded a completion time.", + "$ref": "#/definitions/io.k8s.apimachinery.pkg.apis.meta.v1.Time" + }, + "lastUpdated": { + "description": "lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. This timestamp is always present in the CIB fence history, including for pending operations where completedTime is not yet set. It must be a valid timestamp in RFC3339 format.", + "$ref": "#/definitions/io.k8s.apimachinery.pkg.apis.meta.v1.Time" + }, + "status": { + "description": "status is the outcome of the fencing operation. Valid values are \"success\" (the operation completed successfully), \"failed\" (the operation failed), and \"pending\" (the operation is currently in progress).\n\nPossible enum values:\n - `\"failed\"` means the fencing operation failed.\n - `\"pending\"` means the fencing operation is currently in progress.\n - `\"success\"` means the fencing operation completed successfully.", + "type": "string", + "enum": [ + "failed", + "pending", + "success" + ] + } + } + }, "com.github.openshift.api.etcd.v1.PacemakerNodeAddress": { "description": "PacemakerNodeAddress contains information for a node's address. This is similar to corev1.NodeAddress but adds validation for IP addresses.", "type": "object", @@ -17131,6 +17187,11 @@ ], "x-kubernetes-list-type": "map" }, + "lastFenceEvent": { + "description": "lastFenceEvent is the most recent fencing event targeting this node, as recorded in Pacemaker's fence history in the CIB. This captures the last time this node was fenced (or a fence attempt was made), including the action taken and its outcome. When the Clean condition is False, this field provides the concrete fencing context behind the unclean state. This field is optional and is omitted when no fencing event has been observed for this node.", + "default": {}, + "$ref": "#/definitions/com.github.openshift.api.etcd.v1alpha1.PacemakerFenceEvent" + }, "nodeName": { "description": "nodeName is the name of the node. This is expected to match the Kubernetes node's name, which must be a lowercase RFC 1123 subdomain consisting of lowercase alphanumeric characters, '-' or '.', starting and ending with an alphanumeric character, and be at most 253 characters in length.", "type": "string" @@ -17169,6 +17230,16 @@ ], "x-kubernetes-list-type": "map" }, + "failCount": { + "description": "failCount is the current failure count Pacemaker records for this resource on this node, as reported by the CIB. Pacemaker increments this count each time an operation for this resource fails, and resets it to zero when a `pcs resource cleanup` is performed. The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a fail count for this resource, for example on a freshly bootstrapped cluster or for a resource that has never failed.", + "type": "integer", + "format": "int32" + }, + "migrationThreshold": { + "description": "migrationThreshold is the configured number of failures after which Pacemaker will no longer attempt to run this resource on this node, as reported by the CIB. Without this value, failCount alone is uninterpretable — whether failCount 3 is alarming depends on whether the threshold is 5 or 1000000 (Pacemaker's default INFINITY). The value must be zero or greater. This field is optional and is omitted when the status collector has not yet observed a migration threshold for this resource.", + "type": "integer", + "format": "int32" + }, "name": { "description": "name is the name of the pacemaker resource. Valid values are \"Kubelet\" and \"Etcd\". The Kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments. The Etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations. Fencing agents are tracked separately in the node's fencingAgents field.\n\nPossible enum values:\n - `\"Etcd\"` is the etcd pacemaker resource. The etcd resource may temporarily transition to stopped during pacemaker quorum-recovery operations.\n - `\"Kubelet\"` is the kubelet pacemaker resource. The kubelet resource is a prerequisite for etcd in Two Node OpenShift with Fencing deployments.", "type": "string", @@ -17189,7 +17260,7 @@ ], "properties": { "conditions": { - "description": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. Each of these conditions is required, so the array must contain at least 3 items.", + "description": "conditions represent the observations of the pacemaker cluster's current state. Known condition types are: \"Healthy\", \"InService\", \"NodeCountAsExpected\", \"FencingEnabled\". The \"Healthy\" condition is an aggregate that tracks the overall health of the cluster. The \"InService\" condition tracks whether the cluster is in service (not in maintenance mode). The \"NodeCountAsExpected\" condition tracks whether the expected number of nodes are present. The \"FencingEnabled\" condition tracks whether STONITH (fencing) is enabled in the cluster. Each of these conditions is required, so the array must contain at least 4 items.", "type": "array", "items": { "default": {}, @@ -17218,6 +17289,47 @@ } } }, + "com.github.openshift.api.etcd.v1alpha1.PacemakerFenceEvent": { + "description": "PacemakerFenceEvent represents the most recent fencing event observed for a node. Fencing events are recorded by Pacemaker in the CIB when STONITH operations occur — either automatically during split-brain recovery or manually via stonith_admin. This struct captures the last fence event targeting a given node, providing visibility into what happened, its outcome, and whether it was triggered automatically by the cluster or manually by an operator.", + "type": "object", + "required": [ + "action", + "status", + "lastUpdated" + ], + "properties": { + "action": { + "description": "action is the type of fencing action that was performed or is in progress. Valid values are \"reboot\" (power-cycle the node), \"power-off\" (power off the node), and \"power-on\" (power on the node).\n\nPossible enum values:\n - `\"power-off\"` is a fence action that powers off the target node.\n - `\"power-on\"` is a fence action that powers on the target node.\n - `\"reboot\"` is a fence action that power-cycles the target node.", + "type": "string", + "enum": [ + "power-off", + "power-on", + "reboot" + ] + }, + "client": { + "description": "client identifies the daemon or tool that requested the fencing operation. Typical values are \"crmd\" for automatic recovery initiated by the cluster resource manager, or \"stonith_admin\" for manual fencing initiated by an operator. This is useful for distinguishing \"the cluster fenced itself for cause\" from \"someone fenced a node manually\" during incident review. This field is optional and is omitted when the client is not reported by Pacemaker. When provided, the value must be between 1 and 256 characters.", + "type": "string" + }, + "completedTime": { + "description": "completedTime is the timestamp when the fencing operation completed. This field is optional and is omitted when the fencing operation is still in progress (status \"pending\") or when Pacemaker has not recorded a completion time.", + "$ref": "#/definitions/io.k8s.apimachinery.pkg.apis.meta.v1.Time" + }, + "lastUpdated": { + "description": "lastUpdated is the timestamp when Pacemaker last updated this fence event record in the CIB. This timestamp is always present in the CIB fence history, including for pending operations where completedTime is not yet set. It must be a valid timestamp in RFC3339 format.", + "$ref": "#/definitions/io.k8s.apimachinery.pkg.apis.meta.v1.Time" + }, + "status": { + "description": "status is the outcome of the fencing operation. Valid values are \"success\" (the operation completed successfully), \"failed\" (the operation failed), and \"pending\" (the operation is currently in progress).\n\nPossible enum values:\n - `\"failed\"` means the fencing operation failed.\n - `\"pending\"` means the fencing operation is currently in progress.\n - `\"success\"` means the fencing operation completed successfully.", + "type": "string", + "enum": [ + "failed", + "pending", + "success" + ] + } + } + }, "com.github.openshift.api.etcd.v1alpha1.PacemakerNodeAddress": { "description": "PacemakerNodeAddress contains information for a node's address. This is similar to corev1.NodeAddress but adds validation for IP addresses.", "type": "object",