mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
V2 runtime packages: - sw-block/runtime/masterv2: identity authority (desired state, heartbeat handling, promotion arbitration via SelectPromotionCandidate) - sw-block/runtime/volumev2: per-volume micro-cluster shell (node, orchestrator, control session, iSCSI frontend, takeover gate, failover session + driver, replica summary reconstruction) - sw-block/runtime/purev2: RF1 execution shell (engine + store + dispatcher + local boundary observations) - sw-block/runtime/protocolv2: three-channel separation (heartbeat/assignment/query + replica summary) V2 binaries: - sw-block/cmd/v2singleblock: single-node RF1 block server - sw-block/cmd/purev2rf1: minimal RF1 runtime binary Milestone capabilities: - RF1 write/read/sync with engine-driven mode projection - masterv2 ↔ volumev2 heartbeat convergence + assignment reissue - Promotion query with fresh CommittedLSN/WALHeadLSN evidence - Replica summary for bounded takeover reconstruction - Primary-loss reconstruction from peer summaries (fail-closed gate) - In-process failover driver with session observability - Local boundary observations feed engine (Committed/Durable/Checkpoint) Design docs: - v2-two-loop-protocol.md: identity vs data-control separation - v2-automata-ownership-map.md: event/command ownership split - v2-loop1-surface-draft.md: heartbeat/query/assignment field spec - v2-volumev2-single-node-mvp.md: target layering - v2-kernel-closure-review.md: per-volume micro-cluster principle - v2-pure-runtime-rf1-bootstrap.md, v2-capability-map.md, v2-proof-and-retest-pyramid.md Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
116 lines
3.9 KiB
Go
116 lines
3.9 KiB
Go
package volumev2
|
|
|
|
import (
|
|
"fmt"
|
|
"slices"
|
|
|
|
"github.com/seaweedfs/seaweedfs/sw-block/runtime/masterv2"
|
|
"github.com/seaweedfs/seaweedfs/sw-block/runtime/protocolv2"
|
|
)
|
|
|
|
// ReplicaSummarySource is the bounded Loop 2 query surface a replacement
|
|
// primary uses to reconstruct takeover truth from peers.
|
|
type ReplicaSummarySource interface {
|
|
QueryReplicaSummary(protocolv2.ReplicaSummaryRequest) (protocolv2.ReplicaSummaryResponse, error)
|
|
}
|
|
|
|
// PrimaryTakeoverPlan is the minimal input needed for a selected replacement
|
|
// primary to prepare takeover locally.
|
|
type PrimaryTakeoverPlan struct {
|
|
Assignment masterv2.Assignment
|
|
Peers []ReplicaSummarySource
|
|
}
|
|
|
|
// ReconstructTakeoverTruth lets the selected replacement primary gather its own
|
|
// bounded summary plus peer summaries before it resumes data-control ownership.
|
|
// Peers should exclude the current node; duplicate node IDs are ignored.
|
|
func (n *Node) ReconstructTakeoverTruth(volumeName string, expectedEpoch uint64, peers []ReplicaSummarySource) (ReconstructedPrimaryTruth, error) {
|
|
if n == nil {
|
|
return ReconstructedPrimaryTruth{}, fmt.Errorf("volumev2: node is nil")
|
|
}
|
|
req := protocolv2.ReplicaSummaryRequest{
|
|
VolumeName: volumeName,
|
|
ExpectedEpoch: expectedEpoch,
|
|
}
|
|
self, err := n.QueryReplicaSummary(req)
|
|
if err != nil {
|
|
return ReconstructedPrimaryTruth{}, err
|
|
}
|
|
|
|
byNode := map[string]protocolv2.ReplicaSummaryResponse{
|
|
self.NodeID: self,
|
|
}
|
|
for _, peer := range peers {
|
|
if peer == nil {
|
|
continue
|
|
}
|
|
summary, err := peer.QueryReplicaSummary(req)
|
|
if err != nil {
|
|
return ReconstructedPrimaryTruth{}, fmt.Errorf("volumev2: peer replica summary %s: %w", volumeName, err)
|
|
}
|
|
if summary.NodeID == "" || summary.NodeID == n.id {
|
|
continue
|
|
}
|
|
byNode[summary.NodeID] = summary
|
|
}
|
|
|
|
nodeIDs := make([]string, 0, len(byNode))
|
|
for nodeID := range byNode {
|
|
nodeIDs = append(nodeIDs, nodeID)
|
|
}
|
|
slices.Sort(nodeIDs)
|
|
summaries := make([]protocolv2.ReplicaSummaryResponse, 0, len(nodeIDs))
|
|
for _, nodeID := range nodeIDs {
|
|
summaries = append(summaries, byNode[nodeID])
|
|
}
|
|
return ReconstructPrimaryTruth(n.id, summaries)
|
|
}
|
|
|
|
// PreparePrimaryTakeover applies the local primary assignment and reconstructs
|
|
// bounded takeover truth from self and peers. It does not decide whether the
|
|
// new primary is allowed to activate data control yet.
|
|
func (n *Node) PreparePrimaryTakeover(plan PrimaryTakeoverPlan) (ReconstructedPrimaryTruth, error) {
|
|
if n == nil {
|
|
return ReconstructedPrimaryTruth{}, fmt.Errorf("volumev2: node is nil")
|
|
}
|
|
a := plan.Assignment
|
|
if a.Role != "primary" {
|
|
return ReconstructedPrimaryTruth{}, fmt.Errorf("volumev2: unsupported takeover role %q", a.Role)
|
|
}
|
|
if a.NodeID != "" && a.NodeID != n.id {
|
|
return ReconstructedPrimaryTruth{}, fmt.Errorf("volumev2: takeover assignment targets %q, node is %q", a.NodeID, n.id)
|
|
}
|
|
if err := n.ApplyAssignments([]masterv2.Assignment{a}); err != nil {
|
|
return ReconstructedPrimaryTruth{}, err
|
|
}
|
|
return n.ReconstructTakeoverTruth(a.Name, a.Epoch, plan.Peers)
|
|
}
|
|
|
|
// GatePrimaryActivation fail-closes activation when the reconstructed truth
|
|
// says takeover is degraded, ambiguous, or rebuild-only.
|
|
func (n *Node) GatePrimaryActivation(volumeName string, truth ReconstructedPrimaryTruth) error {
|
|
if n == nil {
|
|
return fmt.Errorf("volumev2: node is nil")
|
|
}
|
|
if truth.NeedsRebuild {
|
|
return fmt.Errorf("volumev2: takeover gated for %s: needs rebuild", volumeName)
|
|
}
|
|
if truth.Degraded {
|
|
return fmt.Errorf("volumev2: takeover gated for %s: %s", volumeName, truth.Reason)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// ApplyPrimaryTakeover is a narrow compatibility wrapper that prepares
|
|
// takeover truth and then gates activation.
|
|
func (n *Node) ApplyPrimaryTakeover(plan PrimaryTakeoverPlan) (ReconstructedPrimaryTruth, error) {
|
|
truth, err := n.PreparePrimaryTakeover(plan)
|
|
if err != nil {
|
|
return ReconstructedPrimaryTruth{}, err
|
|
}
|
|
if err := n.GatePrimaryActivation(plan.Assignment.Name, truth); err != nil {
|
|
return truth, err
|
|
}
|
|
return truth, nil
|
|
}
|