mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
Add host-side protocol state seam that derives per-replica execution state from V2 sender/session snapshots and blocks live-tail WAL shipping while an active recovery session is in progress. New file: weed/server/block_protocol_state.go - replicaProtocolExecutionState derived from engine snapshots - LiveEligible=false during active catch-up/rebuild sessions - bindProtocolExecutionPolicy wires policy into BlockVol - syncProtocolExecutionState called after assignments + core events Data plane changes: - WALShipper.Ship() checks liveShippingPolicy before dial/send - BlockVol.SetLiveShippingPolicy persists across shipper group rebuilds - ShipperGroup propagates policy to all shippers Design contract: sw-block/design/v2-protocol-aware-execution.md Scope: WAL-first rollout only. Prevents illegal live-tail delivery during active recovery. Does not change snapshot/build behavior or move backlog. Next wave: bounded WAL catch-up under same contract. Tests: 4 unit/component tests for phase gate behavior, plus bootstrap seam tests that confirmed the two pre-existing bugs locally. 13 files changed, 900 insertions, 69 deletions. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
150 lines
4.1 KiB
Go
150 lines
4.1 KiB
Go
package blockvol
|
|
|
|
import (
|
|
"sync/atomic"
|
|
"testing"
|
|
)
|
|
|
|
func TestShipperGroup_ShipAll_Single(t *testing.T) {
|
|
s := newTestShipper()
|
|
sg := NewShipperGroup([]*WALShipper{s})
|
|
entry := &WALEntry{LSN: 1, Epoch: 1, Type: EntryTypeWrite}
|
|
sg.ShipAll(entry)
|
|
// Ship on a non-connected shipper degrades it but doesn't panic.
|
|
if sg.Len() != 1 {
|
|
t.Fatalf("Len: got %d, want 1", sg.Len())
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_ShipAll_Two(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s2 := newTestShipper()
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
entry := &WALEntry{LSN: 1, Epoch: 1, Type: EntryTypeWrite}
|
|
sg.ShipAll(entry)
|
|
if sg.Len() != 2 {
|
|
t.Fatalf("Len: got %d, want 2", sg.Len())
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_BarrierAll_AllSucceed(t *testing.T) {
|
|
// Create shippers that are stopped (Barrier returns ErrShipperStopped).
|
|
// We test the fan-out logic: all return errors, which is fine.
|
|
s1 := newTestShipper()
|
|
s2 := newTestShipper()
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
errs := sg.BarrierAll(10)
|
|
if len(errs) != 2 {
|
|
t.Fatalf("BarrierAll: got %d errors, want 2", len(errs))
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_BarrierAll_OneFail(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s1.state.Store(uint32(ReplicaDegraded))
|
|
s2 := newTestShipper()
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
errs := sg.BarrierAll(10)
|
|
if errs[0] == nil {
|
|
t.Fatalf("expected error for degraded shipper[0]")
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_BarrierAll_AllFail(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s1.state.Store(uint32(ReplicaDegraded))
|
|
s2 := newTestShipper()
|
|
s2.state.Store(uint32(ReplicaDegraded))
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
errs := sg.BarrierAll(10)
|
|
for i, e := range errs {
|
|
if e == nil {
|
|
t.Fatalf("expected error for shipper[%d]", i)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_AllDegraded_Empty(t *testing.T) {
|
|
sg := NewShipperGroup(nil)
|
|
if sg.AllDegraded() {
|
|
t.Fatal("empty group should not be AllDegraded")
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_AllDegraded_Mixed(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s1.state.Store(uint32(ReplicaDegraded))
|
|
s2 := newTestShipper()
|
|
s2.state.Store(uint32(ReplicaInSync)) // one healthy, one degraded
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
if sg.AllDegraded() {
|
|
t.Fatal("mixed group should not be AllDegraded")
|
|
}
|
|
if !sg.AnyDegraded() {
|
|
t.Fatal("mixed group should be AnyDegraded")
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_StopAll(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s2 := newTestShipper()
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
sg.StopAll()
|
|
if !s1.stopped.Load() || !s2.stopped.Load() {
|
|
t.Fatal("StopAll should stop all shippers")
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_DegradedCount(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s2 := newTestShipper()
|
|
s1.state.Store(uint32(ReplicaDegraded))
|
|
s2.state.Store(uint32(ReplicaInSync)) // one healthy, one degraded
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
if got := sg.DegradedCount(); got != 1 {
|
|
t.Fatalf("DegradedCount: got %d, want 1", got)
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_AllHaveTransportContact(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s2 := newTestShipper()
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
if sg.AllHaveTransportContact() {
|
|
t.Fatal("fresh disconnected shippers should not report transport contact")
|
|
}
|
|
|
|
s1.shippedLSN.Store(10)
|
|
if sg.AllHaveTransportContact() {
|
|
t.Fatal("partial transport contact should not satisfy full-set readiness")
|
|
}
|
|
|
|
s2.shippedLSN.Store(12)
|
|
if !sg.AllHaveTransportContact() {
|
|
t.Fatal("all shippers with shipped LSN should report transport contact")
|
|
}
|
|
}
|
|
|
|
func TestShipperGroup_AllHaveTransportContact_RejectsDegraded(t *testing.T) {
|
|
s1 := newTestShipper()
|
|
s2 := newTestShipper()
|
|
s1.shippedLSN.Store(10)
|
|
s2.shippedLSN.Store(12)
|
|
s2.state.Store(uint32(ReplicaDegraded))
|
|
sg := NewShipperGroup([]*WALShipper{s1, s2})
|
|
if sg.AllHaveTransportContact() {
|
|
t.Fatal("degraded shipper must not count as transport-connected")
|
|
}
|
|
}
|
|
|
|
// newTestShipper creates a WALShipper with a fixed epoch, not connected to anything.
|
|
func newTestShipper() *WALShipper {
|
|
var epoch atomic.Uint64
|
|
epoch.Store(1)
|
|
return &WALShipper{
|
|
dataAddr: "127.0.0.1:0",
|
|
controlAddr: "127.0.0.1:0",
|
|
epochFn: func() uint64 { return epoch.Load() },
|
|
}
|
|
}
|