Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 3 additions & 5 deletions api/v1alpha1/seinode_types.go
Original file line number Diff line number Diff line change
Expand Up @@ -88,12 +88,10 @@ import (
// +kubebuilder:validation:XValidation:rule="!has(self.nodeConfig) || !has(self.validator) || !has(self.validator.genesisCeremony)",message="spec.nodeConfig is not supported on a genesis-ceremony validator: the founding validator set is assembled during the ceremony and written into config.toml at run time"
// +kubebuilder:validation:XValidation:rule="!has(self.nodeConfig) || ((!has(self.fullNode) || !has(self.fullNode.snapshot) || !has(self.fullNode.snapshot.stateSync)) && (!has(self.validator) || !has(self.validator.snapshot) || !has(self.validator.snapshot.stateSync)) && (!has(self.replayer) || !has(self.replayer.snapshot.stateSync)))",message="spec.nodeConfig is not supported with a state-sync snapshot source: the trust height and hash are discovered from live witnesses and written into config.toml at run time"
// +kubebuilder:validation:XValidation:rule="!has(self.nodeConfig) || !has(self.consensus) || !has(self.consensus.engine) || self.consensus.engine != 'Autobahn'",message="spec.nodeConfig is not supported under consensus engine Autobahn: the engine's config.toml keys are controller-derived"
// nodeConfig may be added to an existing node but never removed (spec 013,
// temporary for the arctic-1 migration, PLT-1410). The StatefulSet's
// nodeConfig is fixed for the node's lifetime. The StatefulSet's
// podManagementPolicy follows it, and the API server refuses to change that
// field on an existing StatefulSet, so SyncStatefulSet recreates the
// StatefulSet when the field must change.
// +kubebuilder:validation:XValidation:rule="!has(oldSelf.nodeConfig) || has(self.nodeConfig)",message="spec.nodeConfig cannot be removed from an existing SeiNode: replace the node (dataVolume.import can carry its data over)"
// field on an existing StatefulSet.
// +kubebuilder:validation:XValidation:rule="has(self.nodeConfig) == has(oldSelf.nodeConfig)",message="spec.nodeConfig can be neither added to nor removed from an existing SeiNode: it is fixed at creation, so replace the node (dataVolume.import can carry its data over)"
// dataResetGeneration is a request counter: each increase asks for one data
// reset. It can only increase, so a git revert cannot lower it and rearm a wipe,
// and it needs nodeConfig, because only that node's plans run the reset.
Expand Down
13 changes: 6 additions & 7 deletions config/crd/sei.io_seinodes.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -107,11 +107,9 @@ spec:
The node shapes below reach config.toml through a task that writes
configuration seid can only learn at run time, so they are rejected beside
nodeConfig too.
nodeConfig may be added to an existing node but never removed (spec 013,
temporary for the arctic-1 migration, PLT-1410). The StatefulSet's
nodeConfig is fixed for the node's lifetime. The StatefulSet's
podManagementPolicy follows it, and the API server refuses to change that
field on an existing StatefulSet, so SyncStatefulSet recreates the
StatefulSet when the field must change.
field on an existing StatefulSet.
dataResetGeneration is a request counter: each increase asks for one data
reset. It can only increase, so a git revert cannot lower it and rearm a wipe,
and it needs nodeConfig, because only that node's plans run the reset.
Expand Down Expand Up @@ -1504,9 +1502,10 @@ spec:
the engine''s config.toml keys are controller-derived'
rule: '!has(self.nodeConfig) || !has(self.consensus) || !has(self.consensus.engine)
|| self.consensus.engine != ''Autobahn'''
- message: 'spec.nodeConfig cannot be removed from an existing SeiNode:
replace the node (dataVolume.import can carry its data over)'
rule: '!has(oldSelf.nodeConfig) || has(self.nodeConfig)'
- message: 'spec.nodeConfig can be neither added to nor removed from an
existing SeiNode: it is fixed at creation, so replace the node (dataVolume.import
can carry its data over)'
rule: has(self.nodeConfig) == has(oldSelf.nodeConfig)
- message: 'spec.dataResetGeneration needs spec.nodeConfig: only a node
that reads its config from ConfigMaps runs the declarative data reset'
rule: '!has(self.dataResetGeneration) || has(self.nodeConfig)'
Expand Down
4 changes: 3 additions & 1 deletion docs/specs/013-nodeconfig-switch/spec.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,9 @@

**Created**: 2026-10-08

**Status**: Draft. Temporary: revert after the arctic-1 migration.
**Status**: Reverted. The code shipped in `21ae61d` (#608) and moved 46 arctic-1 nodes in place on
2026-10-08. The revert restores the create-only rule `has(self.nodeConfig) == has(oldSelf.nodeConfig)`
and removes the StatefulSet recreation. A later in-place move needs this change again.

**Tracking**: PLT-1410 (atlantic-2 migration)

Expand Down
2 changes: 1 addition & 1 deletion docs/specs/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,4 +25,4 @@ not re-litigated later. See `001-configurable-node-resources/decisions.md`.
| [010](010-maintenance-hold/spec.md) | Maintenance hold for ConfigMap-configured nodes | Draft |
| [011](011-live-resize/spec.md) | Resize CPU, memory, and disk on a live ConfigMap-configured node | Draft |
| [012](012-drift-roll-budget/spec.md) | Drift-roll budget: pace pod-template drift per namespace | Draft |
| [013](013-nodeconfig-switch/spec.md) | Switch a running SeiNode to nodeConfig (temporary, arctic-1 migration) | Draft |
| [013](013-nodeconfig-switch/spec.md) | Switch a running SeiNode to nodeConfig (temporary, arctic-1 migration) | Reverted |
83 changes: 0 additions & 83 deletions internal/controller/node/envtest/nodeconfig_switch_test.go

This file was deleted.

17 changes: 10 additions & 7 deletions internal/controller/node/envtest/nodeconfig_validation_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -115,9 +115,10 @@ func TestNodeConfig_WithControllerManagedConfig_Rejected(t *testing.T) {
}
}

// Spec 013 Req 1.1: nodeConfig may be added to an existing node but never
// removed. Republishing under a new ConfigMap name stays allowed.
func TestNodeConfig_AddOnly(t *testing.T) {
// nodeConfig is fixed at creation in both directions: the StatefulSet's
// podManagementPolicy follows it, and that field cannot change on an existing
// StatefulSet. Republishing under a new ConfigMap name stays allowed.
func TestNodeConfig_CreateOnly(t *testing.T) {
g := NewWithT(t)
ns := makeNamespace(t)

Expand All @@ -133,18 +134,20 @@ func TestNodeConfig_AddOnly(t *testing.T) {
cur.Spec.NodeConfig = nil
})
g.Expect(err).To(HaveOccurred(), "removing nodeConfig must be rejected")
g.Expect(err.Error()).To(ContainSubstring("cannot be removed"))
g.Expect(err.Error()).To(ContainSubstring("fixed at creation"))

without := nodeConfigNode(ns, "nc-switch")
without := nodeConfigNode(ns, "nc-never")
without.Spec.NodeConfig = nil
g.Expect(testCli.Create(testCtx, without)).To(Succeed())

g.Expect(updateNodeWithRetry(t, client.ObjectKeyFromObject(without), func(cur *seiv1alpha1.SeiNode) {
err = updateNodeWithRetry(t, client.ObjectKeyFromObject(without), func(cur *seiv1alpha1.SeiNode) {
cur.Spec.NodeConfig = &seiv1alpha1.NodeConfig{
ConfigRef: seiv1alpha1.ConfigFileRef{Name: "rpc-config-v1"},
AppRef: seiv1alpha1.ConfigFileRef{Name: "rpc-app-v1"},
}
})).To(Succeed(), "adding nodeConfig must be accepted")
})
g.Expect(err).To(HaveOccurred(), "adding nodeConfig must be rejected")
g.Expect(err.Error()).To(ContainSubstring("fixed at creation"))
}

// These node shapes write config.toml at run time with values no ConfigMap
Expand Down
5 changes: 2 additions & 3 deletions internal/noderesource/noderesource.go
Original file line number Diff line number Diff line change
Expand Up @@ -718,9 +718,8 @@ func updateStrategy(node *seiv1alpha1.SeiNode) appsv1.StatefulSetUpdateStrategy
// podManagementPolicy is Parallel for a RollingUpdate node. Under OrderedReady
// the StatefulSet controller will not update a pod that is not Ready, so a seid
// halted at an upgrade height would never take the new image. The API server
// refuses to change this field on an existing StatefulSet, so SyncStatefulSet
// recreates the StatefulSet when a node gains spec.nodeConfig (spec 013).
// Empty leaves the API default.
// refuses to change this field on an existing StatefulSet, which is why
// spec.nodeConfig is fixed at creation. Empty leaves the API default.
func podManagementPolicy(node *seiv1alpha1.SeiNode) appsv1.PodManagementPolicyType {
if node.Spec.NodeConfig != nil {
return appsv1.ParallelPodManagement
Expand Down
39 changes: 0 additions & 39 deletions internal/noderesource/sync.go
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,6 @@ import (

appsv1 "k8s.io/api/apps/v1"
apierrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
Expand Down Expand Up @@ -103,40 +102,11 @@ func SyncStatefulSet(
node.Status.StatefulSet = nil
return nil, nil
}
// A node that gained spec.nodeConfig needs a different
// podManagementPolicy, which the API server refuses to change
// in place (spec 013). Delete the StatefulSet with orphan
// propagation: the running pod and the data PVC stay, the next
// reconcile applies the new StatefulSet, which adopts the pod
// and rolls it once onto the nodeConfig template.
if effectivePodManagementPolicy(existing.Spec.PodManagementPolicy) != effectivePodManagementPolicy(desired.Spec.PodManagementPolicy) {
if err := c.Delete(ctx, existing, client.PropagationPolicy(metav1.DeletePropagationOrphan)); err != nil && !apierrors.IsNotFound(err) {
return nil, fmt.Errorf("deleting statefulset to change its pod management policy: %w", err)
}
node.Status.StatefulSet = nil
return nil, nil
}
case apierrors.IsNotFound(err):
// Live object missing; the Apply below recreates it.
default:
return nil, fmt.Errorf("fetching tracked statefulset: %w", err)
}
} else {
// An orphan delete leaves the StatefulSet terminating until the
// garbage collector releases its pods. Wait for it to go: an Apply
// now would patch the terminating object, and the API server would
// reject the policy change. Its delete event triggers the next
// reconcile.
existing := &appsv1.StatefulSet{}
switch err := c.Get(ctx, key, existing); {
case err == nil:
if existing.DeletionTimestamp != nil {
return nil, nil
}
case apierrors.IsNotFound(err):
default:
return nil, fmt.Errorf("fetching statefulset: %w", err)
}
}

// client.Apply decodes the apiserver response into desired in-place,
Expand All @@ -150,12 +120,3 @@ func SyncStatefulSet(
}
return desired, nil
}

// effectivePodManagementPolicy resolves an empty policy to the API default, so
// a rendered "" compares equal to the live OrderedReady.
func effectivePodManagementPolicy(p appsv1.PodManagementPolicyType) appsv1.PodManagementPolicyType {
if p == "" {
return appsv1.OrderedReadyPodManagement
}
return p
}
141 changes: 0 additions & 141 deletions internal/noderesource/sync_switch_test.go

This file was deleted.

Loading
Loading