From d2d57851b0151f6809e849d85cf0a6aeb00eb1a9 Mon Sep 17 00:00:00 2001 From: pingqiu Date: Tue, 7 Apr 2026 14:30:34 -0700 Subject: [PATCH] =?UTF-8?q?feat:=20rebuild=20MVP=20=E2=80=94=20dual-lane?= =?UTF-8?q?=20session=20with=20bitmap=20protection?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebuild session protocol implementation for v2-rebuild-mvp-session-protocol.md. New files: - rebuild_bitmap.go: RebuildBitmap — session-scoped dense bitset for WAL-applied LBA tracking. MarkApplied on local WAL write (not receive). ShouldApplyBase returns false for WAL-covered LBAs (WAL always wins). - rebuild_session.go: RebuildSession — replica-side two-line rebuild. WAL lane (ApplyWALEntry) + base lane (ApplyBaseBlock) with bitmap conflict resolution. TryComplete requires BOTH base_complete AND wal_applied_lsn >= target_lsn. Volume-level control surface: StartRebuildSession, ApplyRebuildSessionWALEntry/BaseBlock, MarkRebuildSessionBaseComplete, TryCompleteRebuildSession, CancelRebuildSession, ActiveRebuildSession. - rebuild_mvp_test.go: 4 correctness tests — base+WAL converge, WAL-applied never overwritten by base, bitmap set on applied not received, control surface start/supersede/complete. - rebuild_transport_test.go: 2 transport-level tests — two-line with real WAL shipping, live writes during base copy with bitmap conflict. Design docs: - v2-rebuild-mvp-session-protocol.md: MVP spec with message set, apply rules, completion/failure/crash rules, test matrix - v2-sync-recovery-protocol.md: full protocol context (keepup/catchup/ rebuild unified design, primary decision logic, two-line model) - v2-session-protocol-shape.md: protocol shape overview Protocol engine (reference, not production): - sw-block/protocol/: 7-event engine with ~300 lines, 13 tests 6 rebuild tests pass, all existing component tests pass. Co-Authored-By: Claude Opus 4.6 (1M context) --- .../design/v2-engine-maintainer-tutorial.md | 2 + .../design/v2-rebuild-mvp-session-protocol.md | 416 ++++++++++ sw-block/design/v2-session-protocol-shape.md | 298 +++++++ sw-block/design/v2-sync-recovery-protocol.md | 386 +++++++++ sw-block/engine/replication/command.go | 3 + sw-block/engine/replication/engine.go | 777 ++++++++++++++---- sw-block/engine/replication/event.go | 62 +- .../engine/replication/phase14_sync_test.go | 519 ++++++++++++ sw-block/engine/replication/projection.go | 5 +- sw-block/engine/replication/state.go | 92 ++- sw-block/protocol/engine.go | 398 +++++++++ sw-block/protocol/engine_test.go | 345 ++++++++ sw-block/protocol/types.go | 249 ++++++ weed/server/block_recovery.go | 32 +- weed/server/block_recovery_test.go | 69 ++ weed/server/volume_server_block.go | 11 + weed/server/volume_server_block_test.go | 51 ++ weed/storage/blockvol/blockvol.go | 2 + weed/storage/blockvol/rebuild_bitmap.go | 84 ++ weed/storage/blockvol/rebuild_session.go | 409 +++++++++ .../test/component/rebuild_mvp_test.go | 378 +++++++++ .../test/component/rebuild_transport_test.go | 294 +++++++ 22 files changed, 4692 insertions(+), 190 deletions(-) create mode 100644 sw-block/design/v2-rebuild-mvp-session-protocol.md create mode 100644 sw-block/design/v2-session-protocol-shape.md create mode 100644 sw-block/design/v2-sync-recovery-protocol.md create mode 100644 sw-block/engine/replication/phase14_sync_test.go create mode 100644 sw-block/protocol/engine.go create mode 100644 sw-block/protocol/engine_test.go create mode 100644 sw-block/protocol/types.go create mode 100644 weed/storage/blockvol/rebuild_bitmap.go create mode 100644 weed/storage/blockvol/rebuild_session.go create mode 100644 weed/storage/blockvol/test/component/rebuild_mvp_test.go create mode 100644 weed/storage/blockvol/test/component/rebuild_transport_test.go diff --git a/sw-block/design/v2-engine-maintainer-tutorial.md b/sw-block/design/v2-engine-maintainer-tutorial.md index 79ca097c4..7dcf7e5b0 100644 --- a/sw-block/design/v2-engine-maintainer-tutorial.md +++ b/sw-block/design/v2-engine-maintainer-tutorial.md @@ -133,6 +133,8 @@ See `v2-proof-and-retest-pyramid.md`. ## 10. Related documents - `v2-automata-ownership-map.md` — who owns which automaton. +- `v2-session-protocol-shape.md` — current VS-to-VS sync/session/data surface. +- `v2-rebuild-mvp-session-protocol.md` — implementation target for the first rebuild MVP. - `v2-protocol-aware-execution.md` — host-side execution gating. - `wal-replication-v2-state-machine.md` — replica FSM (design-level). - `engine/replication/doc.go` — source-level invariant list (always keep in sync when you change semantics). diff --git a/sw-block/design/v2-rebuild-mvp-session-protocol.md b/sw-block/design/v2-rebuild-mvp-session-protocol.md new file mode 100644 index 000000000..82462c687 --- /dev/null +++ b/sw-block/design/v2-rebuild-mvp-session-protocol.md @@ -0,0 +1,416 @@ +# V2 Rebuild MVP Session Protocol + +Date: 2026-04-07 +Status: active draft + +## Goal + +Define the smallest reliable VS-to-VS protocol that is sufficient to build a +working `rebuild` MVP on top of: + +1. trusted snapshot/base transfer +2. live WAL ingestion +3. primary-owned session control +4. replica-reported session progress + +This document is intentionally narrower than the long-term protocol. It is the +implementation target for the first rebuild MVP. + +## Non-Goals + +This MVP does not try to define: + +1. a full `catchup` protocol +2. `rangeBitmap` or delta-block rebuild execution +3. partial session resume from volatile in-memory rebuild state +4. advanced retransmit/window semantics beyond simple transport needs +5. broad product-ready transport or multi-host rollout guarantees + +## Control Model + +The protocol keeps one strict rule: + +1. `sync` asks for facts +2. the primary decides whether to issue `rebuild` +3. the replica executes the session +4. the replica reports session progress +5. only the primary decides when the session is complete + +The replica does not choose: + +1. the next session kind +2. the target boundary +3. whether it is quorum-eligible again + +## Message Set + +The MVP needs four semantic message families: + +1. `sync` +2. `syncAck` +3. `sessionControl` +4. `sessionAck` + +And two data-plane lanes: + +1. `walData` +2. `sessionData` + +Optional: + +1. `sessionDataAck` + transport/window control only + +## Message Definitions + +### `sync` + +Direction: + +1. primary -> replica + +Minimum fields: + +1. `volume_id` +2. `replica_id` +3. `epoch` +4. `sync_id` +5. `target_lsn` +6. `deadline_ms` + +Purpose: + +1. get bounded replica facts +2. observe whether a session is already active +3. decide whether to stay `keepup` or start `rebuild` + +### `syncAck` + +Direction: + +1. replica -> primary + +Minimum fields: + +1. `volume_id` +2. `replica_id` +3. `epoch` +4. `sync_id` +5. `ack_kind` +6. `applied_lsn` +7. `durable_lsn` +8. `session_active` +9. `session_id` optional +10. `session_kind` optional +11. `session_phase` optional +12. `reason` optional + +Allowed `ack_kind` values: + +1. `quorum` +2. `timed_out` +3. `transport_lost` +4. `epoch_mismatch` + +Rule: + +1. `syncAck` returns facts only +2. it does not recommend `keepup` / `catchup` / `rebuild` + +### `sessionControl` + +Direction: + +1. primary -> replica + +The MVP needs only: + +1. `start_rebuild` +2. `cancel_session` + +Minimum fields for `start_rebuild`: + +1. `volume_id` +2. `replica_id` +3. `epoch` +4. `session_id` +5. `session_kind = rebuild` +6. `base_kind = snapshot` +7. `base_lsn` +8. `target_lsn` +9. `snapshot_id` +10. `deadline_ms` + +Rules: + +1. `session_id` must be unique under the current primary authority +2. a new session may supersede an older one +3. epoch mismatch must be rejected + +### `sessionAck` + +Direction: + +1. replica -> primary + +Minimum fields: + +1. `volume_id` +2. `replica_id` +3. `epoch` +4. `session_id` +5. `session_kind = rebuild` +6. `phase` +7. `wal_applied_lsn` +8. `base_progress` +9. `base_complete` +10. `achieved_lsn` on completion +11. `reason` on failure + +Allowed `phase` values for the MVP: + +1. `accepted` +2. `running` +3. `base_complete` +4. `completed` +5. `failed` + +### `sessionData` + +Direction: + +1. primary -> replica + +Purpose: + +1. send trusted snapshot/base chunks + +Minimum fields: + +1. `volume_id` +2. `replica_id` +3. `epoch` +4. `session_id` +5. `snapshot_id` +6. `chunk_id` +7. `offset_or_lba_range` +8. `payload` +9. `is_last_chunk` + +### `walData` + +Direction: + +1. primary -> replica + +Purpose: + +1. continue live WAL ingestion during rebuild + +Minimum fields: + +1. `volume_id` +2. `replica_id` +3. `epoch` +4. `lsn` +5. `writes[]` + +Each write should carry: + +1. `lba_range` +2. `payload` + +## Replica Apply Rules + +### Base Rule + +Rebuild runs as two concurrent lanes: + +1. base lane from trusted snapshot/base +2. live WAL lane from `base_lsn` + +### Bitmap Rule + +The replica maintains a bitmap of LBAs covered by applied WAL. + +The bit is set when the WAL write is: + +1. applied into replica-local WAL/recovery truth +2. replayable after restart + +The bit is not set when data is only: + +1. received on the network +2. queued but not yet applied locally + +### Write Conflict Rule + +When a base chunk targets an LBA: + +1. if the bitmap bit is clear, base data may be written +2. if the bitmap bit is set, base data for that LBA must be skipped + +Short form: + +1. `WAL applied` wins over older base data + +### Flush Rule + +For bitmap protection, `applied` does not require: + +1. flushing the write into the final extent image + +Replica-local WAL durability and replay are sufficient for the MVP. + +## Completion Rule + +The primary may accept `rebuild completed` only when all are true: + +1. `base_complete = true` +2. `wal_applied_lsn >= target_lsn` +3. the session has not been cancelled or superseded +4. the replica reports one explicit `achieved_lsn` + +`base transfer finished` alone is not completion. + +Only after the primary accepts this completion may the replica become eligible +again for normal quorum-style sync closure. + +## Failure Rule + +Session failure does not decide the next semantic recovery path. + +`failed(reason)` means only: + +1. this rebuild session did not complete + +After failure: + +1. the replica reports fresh facts again through `syncAck` +2. the primary re-decides whether to issue a new rebuild session + +No local component may self-promote the failure into semantic `needs_rebuild`. + +## Crash Rule + +The MVP assumes bitmap may be session-local volatile state. + +Therefore after replica crash or session loss: + +1. do not resume a partially completed rebuild from volatile bitmap state +2. restart with a fresh `sync` +3. let the primary issue a fresh rebuild session + +This means the MVP supports: + +1. safe restart from durable WAL facts + +But does not support: + +1. arbitrary mid-session resume of partial base-copy progress + +## Primary Decision Rule + +The MVP decision rule should stay intentionally simple: + +1. if `syncAck.ack_kind = quorum`, remain `keepup` +2. otherwise, if the replica is not safely closed in normal sync semantics, issue + `rebuild` + +The first MVP does not need a full negotiated `catchup` protocol. + +## MVP Implementation Skeleton + +To reduce wiring ambiguity, the first implementation should expose one explicit +replica-side control surface in `blockvol`: + +1. `StartRebuildSession(config)` +2. `ApplyRebuildSessionWALEntry(session_id, entry)` +3. `ApplyRebuildSessionBaseBlock(session_id, lba, data)` +4. `MarkRebuildSessionBaseComplete(session_id, total_blocks)` +5. `TryCompleteRebuildSession(session_id)` +6. `CancelRebuildSession(session_id, reason)` +7. `ActiveRebuildSession()` + +Contract: + +1. `blockvol` owns only replica-local session state and dual-lane apply rules +2. host/server wiring owns transport routing and message decoding +3. stale packets must be rejected by `session_id` +4. supersede is explicit: a new `session_id` replaces the old active session +5. completion remains queryable until the host emits the matching + `SessionCompleted`-style event and clears the session + +Current MVP implementation choices: + +1. use a dedicated `RebuildBitmap`, not `DirtyMap` +2. use snapshot/trusted-base transfer for the base lane +3. reuse the existing rebuild TCP path for `sessionData` rather than inventing + a new transport first + +## Replica State Machine + +```mermaid +flowchart TD + idle[Idle] + accepted[Accepted] + running[Running] + baseComplete[BaseComplete] + completed[Completed] + failed[Failed] + + idle --> accepted + accepted --> running + running --> baseComplete + running --> failed + baseComplete --> completed + baseComplete --> failed +``` + +Interpretation: + +1. `accepted` + session contract is valid and epoch/session id are accepted +2. `running` + base lane and WAL lane are active +3. `baseComplete` + trusted base transfer is complete, WAL lane still determines final closure +4. `completed` + replica reports one achieved boundary at or beyond target +5. `failed` + session stopped without semantic completion + +## Test Matrix + +The rebuild MVP should not be considered ready until these tests exist. + +### Protocol + +1. `syncAck` returns facts only and never recommends an action +2. `start_rebuild` is rejected on epoch mismatch +3. a new `session_id` supersedes the previous session + +### Correctness + +1. base lane plus live WAL lane converge to target +2. WAL-applied LBA is never overwritten by later base-copy data +3. bitmap bit is set on `applied`, not on `received` + +### Crash / Failure + +1. crash after WAL receive but before apply leaves bitmap clear and base may + still cover the LBA safely +2. crash after WAL apply preserves correctness through local WAL replay +3. transport loss during rebuild yields `failed(reason)` and requires primary + re-decision +4. rebuild completion does not restore normal quorum eligibility until the + primary accepts completion + +## Follow-On Work + +After this MVP is working, the next candidates are: + +1. negotiated `catchup` +2. `rangeBitmap` / delta-block rebuild +3. durable rebuild checkpoints for safe mid-session resume +4. richer `sessionDataAck` flow control diff --git a/sw-block/design/v2-session-protocol-shape.md b/sw-block/design/v2-session-protocol-shape.md new file mode 100644 index 000000000..2769e9d6b --- /dev/null +++ b/sw-block/design/v2-session-protocol-shape.md @@ -0,0 +1,298 @@ +# V2 Session Protocol Shape + +Date: 2026-04-07 +Status: active draft + +Implementation-oriented companion: + +- `v2-rebuild-mvp-session-protocol.md` — concrete rebuild MVP protocol target + +## Purpose + +This note fixes the current protocol direction for VS-to-VS recovery control so +the engine can eventually shrink its semantic surface instead of re-explaining +transport/runtime details through many events. + +The goal is to keep one clear split: + +1. `sync` asks for facts +2. the primary decides the session +3. the replica executes and reports progress +4. data transport stays separate from semantic ack + +## Message Families + +The preferred bounded surface is: + +1. `walData` + - primary -> replica + - steady-state live WAL lane +2. `sync` + - primary -> replica + - bounded fact query +3. `syncAck` + - replica -> primary + - bounded facts only +4. `sessionControl` + - primary -> replica + - start/cancel/supersede one session contract +5. `sessionAck` + - replica -> primary + - accepted/progress/completed/failed +6. `sessionData` + - primary -> replica + - historical repair lane +7. `sessionDataAck` (optional) + - replica -> primary + - transport/window control only + +## Ack Separation Rule + +These meanings must stay separate: + +1. transport ack +2. session ack +3. sync ack + +`sessionDataAck` must never imply: + +1. quorum eligibility +2. recovery completion +3. return to `keepup` + +## Session Decision Rule + +The primary should decide from fresh sync facts: + +1. `keepup` if normal sync closure is still true +2. `catchup` if the replica is still within recoverable WAL history +3. `rebuild` if the replica is below recoverable retained history + +The replica does not choose the next session kind. + +## Recovery Paths + +### 1. Catch-up + +`catchup` is the narrow WAL-only recovery path. + +Expected role: + +1. network delay +2. short temporary gap +3. recoverable WAL-only replay + +It should not be treated as the main recovery framework. + +Catch-up uses two WAL lanes: + +1. replay lane from `pin_lsn` to frozen `current_lsn1` +2. live lane beyond `current_lsn1` + +No bitmap is needed because WAL is ordered by LSN. + +### 2. Rebuild + +`rebuild` is the formal primary recovery path. + +It should behave as one integrated contract with two concurrent lanes: + +1. base lane + - primary exposes a trusted snapshot/CoW view at `base_lsn` + - replica receives extent/base data from that frozen view +2. WAL lane + - replica accepts WAL from `base_lsn` + - replica applies WAL into its local recovery state while base transfer + continues + +This avoids a large delayed post-snapshot catch-up that would pin old WAL too +long. + +### Rebuild Variants + +All rebuild variants share the same semantic contract: + +1. trusted base +2. explicit target +3. live WAL lane +4. single completion boundary accepted by the primary + +The data source may vary: + +1. `full_copy` + - copy the full base image +2. `snapshot_or_cow` + - copy a trusted frozen snapshot/CoW view +3. `delta_blocks_since_base` + - copy only blocks known to have changed since a trusted base boundary + +This is an optimization choice, not a different session truth model. + +## Bitmap Rule For Rebuild + +The replica maintains a bitmap of LBAs already covered by applied WAL. + +The rule is: + +1. WAL-applied LBA => later base-copy data for that LBA must be skipped +2. WAL-received-but-not-applied LBA => not protected by bitmap + +So the bit is set on `applied`, not on `received`. + +### Meaning of Applied + +For this protocol, `applied` means: + +1. accepted into the replica's local WAL/recovery truth +2. replayable after replica restart + +It does not require the update to be flushed into the final extent image before +the bitmap may protect the LBA. + +## Range Bitmap Optimization + +### Purpose + +A persistent range bitmap can turn some rebuilds from "copy the full base" into +"copy only blocks changed since a trusted base boundary." + +This is a rebuild optimization, not a new engine-level recovery kind. + +### Trusted-Base Rule + +Range-bitmap optimization is only valid relative to a trusted base boundary. + +Valid anchors include: + +1. checkpoint/snapshot at `base_lsn` +2. previously accepted rebuild/session completion at `base_lsn` + +Invalid anchor: + +1. arbitrary replica-reported old `applied_lsn` with no trusted-base proof + +So the optimization rule is: + +1. choose trusted `base_lsn` +2. compute changed blocks for `(base_lsn, target_lsn]` +3. copy only that changed-block set as the base lane +4. keep live WAL lane running in parallel + +### Data Shape + +Conceptually: + +1. `rangeBitmap[lsn_range] -> changed_blocks` +2. planner computes `union(changed_blocks over requested range)` +3. rebuild sends only those blocks from the trusted base image + +This is similar in spirit to changed-block tracking or activity-log-assisted +resync, but it must remain anchored to one explicit trusted base point. + +### Layering Rule + +`rangeBitmap` belongs to: + +1. rebuild planner +2. storage/checkpoint metadata +3. execution optimization + +It does not belong to: + +1. engine projection truth +2. session semantic ownership +3. sync-decision semantics + +The engine still only needs to know: + +1. session kind +2. base boundary +3. target boundary +4. progress/completion/failure + +## Failure Rule + +Session failure must not silently decide the next semantic state. + +`SessionFailed` means only: + +1. this primary-issued contract did not complete + +After failure: + +1. the replica reports fresh facts again +2. the primary re-decides `keepup` / `catchup` / `rebuild` + +No local component may self-escalate to semantic `needs_rebuild`. + +## Rebuild-Time Ack Rule + +During rebuild: + +1. the replica may continue applying new WAL +2. the replica must continue reporting session progress +3. the replica must not be treated as normal quorum-eligible sync success until + the rebuild contract closes + +So `syncAck` during rebuild should carry: + +1. current facts +2. active session state +3. not-ready-for-quorum meaning + +Only after the primary accepts `SessionCompleted` may later `syncAck` regain +normal quorum semantics. + +## Minimal Session Shapes + +### `sessionControl` + +The minimum contract should carry: + +1. `session_id` +2. `epoch` +3. `replica_id` +4. `kind` +5. `base_lsn` +6. `target_lsn` +7. `deadline_ms` + +For rebuild it may also carry: + +1. `base_kind` +2. `snapshot_id` or `cow_view_id` +3. `reservation` + +### `sessionAck` + +The minimum replica response should carry: + +1. `session_id` +2. `epoch` +3. `kind` +4. `phase` +5. `accepted | progress | completed | failed` + +For progress reporting, the important facts are: + +1. `wal_applied_lsn` +2. `base_progress` +3. `base_complete` +4. `achieved_lsn` on completion + +`bitmap_coverage` may be added later if needed, but it is not required as the +first semantic surface. + +## Engine Consequence + +If this shape is preserved, the engine can eventually reduce its semantic +surface to a smaller set of facts: + +1. assignment truth +2. sync facts and session decision +3. session progress +4. session completion +5. session failure + +That reduction is only safe because transport ack, session ack, and sync ack are +kept separate at the protocol boundary. diff --git a/sw-block/design/v2-sync-recovery-protocol.md b/sw-block/design/v2-sync-recovery-protocol.md new file mode 100644 index 000000000..e079eb11e --- /dev/null +++ b/sw-block/design/v2-sync-recovery-protocol.md @@ -0,0 +1,386 @@ +# V2 Sync / Recovery Protocol + +Date: 2026-04-07 +Status: active design + +## Purpose + +Define the complete replication protocol for sw-block V2. This document +covers sync, keepup, catch-up, and rebuild as one unified protocol so +the sw agent has full context for implementation. + +`v2-rebuild-mvp-session-protocol.md` is the narrower first-slice spec. +This document is the surrounding context and long-term design. + +## Design Principles + +1. **Primary decides everything.** Replica only reports facts and executes + contracts. Replica never self-escalates to `needs_rebuild`. + +2. **One threshold.** `applied_lsn >= primary_wal_tail` → WAL-recoverable. + Otherwise → rebuild. Matches Ceph's `last_update >= log_tail`. + +3. **Deterministic engine.** Event in → state + commands + projection out. + No side effects inside the engine. Host executes commands. + +4. **Three authority layers:** + - Assignment: master → identity (who is primary/replica, epoch, replica set) + - Session: primary → per-replica recovery contract (keepup/catchup/rebuild) + - Projection: primary → derived volume mode/health + +5. **Failure never auto-escalates.** A failed session stays `failed`. The + primary re-decides from fresh `syncAck` facts. Only the primary can + issue a rebuild. + +## Reference Systems + +| System | Catch-up | Rebuild | Decision owner | Decision input | +|---|---|---|---|---| +| Ceph | PG log replay | Full backfill | Primary OSD (peering) | `last_update >= log_tail` | +| Mayastor | None | Segment copy | Control plane | Child sync state | +| Longhorn | None | Snapshot file sync | Controller | Revision counters | +| **sw-block V2** | WAL replay | Snapshot + live WAL (two-line) | **Primary** | `applied_lsn >= wal_tail` | + +## Protocol Overview + +### Normal Operation (keepup) + +``` +Primary Replica + │ │ + ├─ WriteLBA ────────────────────►│ (live WAL shipping via ShipAll) + │ │ apply to local WAL + │ │ + ├─ sync(target_lsn=N) ─────────►│ + │◄─ syncAck(durable=N, applied=N)│ + │ │ + │ decision: quorum → keepup │ + │ derive: publish_healthy │ +``` + +### Catch-up (replica behind but within retained WAL) + +``` +Primary Replica + │ │ + ├─ sync(target_lsn=1000) ──────►│ + │◄─ syncAck(applied=500) │ + │ │ + │ decision: 500 >= wal_tail(100)│ + │ → WAL catch-up │ + │ │ + ├─ sessionControl(start_catchup │ + │ start=500, target=1000, │ + │ pin=500) ─────────────────►│ + │ │ + │ LINE 1: WAL replay [500..1000]│ + ├─ walReplay(lsn=501...) ──────►│ apply, pin advances + │ │ + │ LINE 2: live WAL from 1001+ │ + ├─ walData(lsn=1001...) ───────►│ apply to local WAL + │ │ + │◄─ sessionAck(completed, │ + │ achieved=1050) ─────────────│ + │ │ + │ replica back in keepup │ +``` + +### Rebuild (replica beyond retained WAL, or fresh join) + +``` +Primary Replica + │ │ + ├─ sync(target_lsn=5000) ──────►│ + │◄─ syncAck(applied=0) │ + │ │ + │ decision: 0 < wal_tail(2000) │ + │ → rebuild │ + │ │ + ├─ sessionControl(start_rebuild │ + │ base_lsn=5000, │ + │ snapshot_id=snap1) ────────►│ + │ │ + │ LINE 1: snapshot extent blocks│ + ├─ sessionData(chunk...) ──────►│ apply if bitmap clear + │ │ + │ LINE 2: live WAL from 5001+ │ + ├─ walData(lsn=5001...) ───────►│ apply, set bitmap bit + │ │ + │ Bitmap: WAL-applied LBA wins │ + │ over later base block │ + │ │ + │◄─ sessionAck(base_complete) │ + │◄─ sessionAck(completed, │ + │ achieved=5200) ─────────────│ + │ │ + │ replica back in keepup │ +``` + +## Primary Decision Logic + +``` +func decide(ack SyncAck, walTail, walHead uint64) SessionKind { + replicaPos := max(ack.AppliedLSN, ack.DurableLSN) + + if replicaPos >= walHead && replicaPos > 0: + return keepup // fully caught up + + if replicaPos >= walTail && replicaPos > 0: + return catchup // behind but within retained WAL + + if replicaPos == 0 && walTail <= 1: + return catchup // fresh replica, WAL retained from beginning + + return rebuild // gap exceeds retained WAL +} +``` + +This is one function, one threshold. Matches Ceph's `last_update >= log_tail`. + +## Two-Line Recovery Model + +Both catch-up and rebuild use two concurrent data lines: + +### Catch-up: WAL replay + live WAL + +- **Line 1**: replay retained WAL entries from `pin_lsn` to `target_lsn` +- **Line 2**: forward live WAL entries from `target_lsn+1` onward +- **Pin movement**: as replay cursor advances, pin can advance (releases old WAL) +- **No bitmap needed**: WAL entries are strictly ordered by LSN, no LBA conflict +- **Completion**: replay cursor reaches target → lines merge → keepup + +### Rebuild: snapshot base + live WAL + +- **Line 1**: copy snapshot/CoW extent blocks to replica +- **Line 2**: forward live WAL entries from `base_lsn` onward +- **Bitmap required**: base blocks and WAL entries may target the same LBA +- **Bitmap rule**: bit set on WAL `applied` (not received). Base block skipped if bit set. +- **Completion**: all base blocks transferred AND `wal_applied_lsn >= target_lsn` + +### Why two lines instead of sequential (base → then catch-up) + +Sequential model: +1. Copy entire snapshot +2. Then replay WAL from snapshot LSN to current +3. Problem: must pin WAL for duration of snapshot copy (hours for large volumes) +4. Risk: WAL recycled before replay starts → must restart entire rebuild + +Two-line model: +1. Copy snapshot AND receive live WAL simultaneously +2. WAL pin pressure = only gap between current replay and live head (small) +3. If snapshot copy is slow, WAL line keeps replica current +4. Crash recovery is safe at any point (bitmap + local WAL) + +## Bitmap Rules (Rebuild Only) + +### When to set bit + +Set bitmap bit when WAL entry is **applied to replica's local WAL**: +- Entry has been written to local WAL file +- Entry is replayable after crash +- Does NOT require flush to final extent + +### When NOT to set bit + +Do not set on: +- Network receive (TCP buffer) +- Queue but not yet local WAL write + +### Conflict resolution + +When base lane sends a chunk for LBA range: +- Bitmap clear → write base data +- Bitmap set → skip (WAL-applied data is newer) + +Short form: **WAL always wins over base.** + +### Crash safety + +At any crash point: +- Bitmap can be volatile (session-local, in memory) +- Local WAL is durable → replay recovers all applied entries +- After crash: fresh sync → primary re-decides → new session if needed +- No need to persist bitmap across crashes in MVP + +## Replica State Machine + +``` +idle → accepted → running → base_complete → completed + │ │ + └──► failed ◄──┘ +``` + +- `idle`: no active session +- `accepted`: session contract valid, epoch/session_id accepted +- `running`: both base lane and WAL lane active +- `base_complete`: base transfer done, WAL lane still running +- `completed`: replica reports `achieved_lsn >= target_lsn` +- `failed`: session stopped without completion, reason reported + +Failure does NOT auto-escalate. Primary re-decides from next syncAck. + +## Mode Derivation (Projection) + +Primary derives volume mode from all replica states + boundaries: + +``` +any replica in rebuild session → needs_rebuild +any replica session failed → degraded +any replica in catch-up session → bootstrap_pending +no replicas assigned → allocated_only +replica role + receiver ready → replica_ready +primary + all readiness + durable → publish_healthy +assigned but not ready → bootstrap_pending +``` + +## Failure Handling + +### Principle + +No failure auto-escalates. All failures go through: +1. Session marked `failed` with reason +2. Primary waits for next `syncAck` from replica +3. Primary re-decides based on fresh facts + +### Failure scenarios + +| Scenario | Replica does | Primary does | +|---|---|---| +| Transport lost during session | Reports `failed(transport_lost)` or goes silent | Marks session failed, waits for reconnect | +| Replica crash | Restarts, recovers local WAL, reports facts via syncAck | Re-decides: if `applied_lsn >= wal_tail` → catch-up, else → new rebuild | +| Primary crash | Nothing (waits for new primary) | New primary elected, fresh epoch, all replicas report via syncAck | +| Slow progress / timeout | Continues trying | Can cancel session via `cancel_session`, then re-decide | +| WAL recycled during outage | Reports `applied_lsn` which is now < `wal_tail` | Decides rebuild (gap exceeds retained WAL) | + +### Failure reason vocabulary (stable) + +- `epoch_mismatch` +- `transport_lost` +- `progress_stalled` +- `deadline_exceeded` +- `pin_lost` +- `snapshot_unavailable` +- `local_wal_corrupt` +- `local_extent_corrupt` + +## Catch-up Details (Post-MVP) + +Catch-up uses the same session contract shape as rebuild, but without +snapshot/base copy: + +``` +sessionControl { + op: start_catchup + start_lsn: + target_lsn: + pin_lsn: +} +``` + +Replica receives: +- WAL replay entries [start_lsn .. target_lsn] (line 1) +- Live WAL entries [target_lsn+1 ..] (line 2) + +No bitmap needed because WAL entries are ordered by LSN — no LBA conflict +between replay and live. + +Pin advances as replay cursor moves forward, releasing old WAL entries. + +## Future: Range Bitmap / Delta Rebuild + +If primary maintains persistent per-checkpoint dirty block tracking: + +``` +modified_blocks[checkpoint_lsn_range] → set of dirty LBAs +``` + +Then rebuild can skip copying blocks that haven't changed since the +replica's last known position. Only modified blocks need to be sent. + +This turns rebuild from O(volume_size) to O(changed_blocks), similar to +VMware CBTT (Changed Block Tracking) or DRBD activity log. + +Implementation: persist DirtyMap snapshot at each flusher checkpoint along +with the checkpoint LSN. On rebuild, union of DirtyMap snapshots from +`replica_applied_lsn` to `current_checkpoint` = minimal copy set. + +## Component Test Requirements + +All tests use real BlockVol with real WAL, extent, and snapshot data. +No mocks for storage. Network can be in-process (localhost TCP or direct +function call). + +### Protocol tests (unit, engine only) + +1. `syncAck` returns facts only, never action +2. `sessionControl(start_rebuild)` rejected on epoch mismatch +3. New `session_id` supersedes old session +4. Primary decision: `applied >= wal_tail` → catch-up +5. Primary decision: `applied < wal_tail` → rebuild +6. Primary decision: `applied >= wal_head` → keepup +7. Session failure does not auto-escalate to needs_rebuild +8. Session failure then fresh syncAck → primary re-decides + +### Rebuild correctness (component, real BlockVol) + +These tests create real primary + replica volumes with actual WAL, +extent, and snapshot data: + +1. **Two-line convergence**: primary writes N blocks, takes snapshot, + starts rebuild session. Base lane copies snapshot extent. WAL lane + ships live entries. Verify replica has all N blocks correct at end. + +2. **WAL wins over base**: primary writes block A=1, snapshots, then + writes A=2. Rebuild sends base (A=1) and WAL (A=2). Verify replica + has A=2 (WAL-applied wins). + +3. **Bitmap on applied not received**: ship WAL entry but delay local + apply. Send base block for same LBA. Base block should land (bitmap + not set yet). Then apply WAL entry. Bitmap now set. Final state + should be WAL data. + +4. **Large rebuild**: 1000+ blocks, concurrent base + WAL, verify all + blocks correct at completion. + +5. **Writes during rebuild**: primary continues writing new blocks during + rebuild. Verify new blocks reach replica via WAL lane and are + preserved at rebuild completion. + +### Crash / failure (component, real BlockVol) + +1. **Crash after WAL receive before apply**: restart replica, verify + base can still cover the LBA (bitmap was not set). + +2. **Crash after WAL apply**: restart replica, verify local WAL replay + recovers the applied data correctly. + +3. **Transport lost during rebuild**: session reports `failed(transport_lost)`. + Primary re-decides from fresh syncAck. New rebuild session starts. + +4. **Rebuild completion does not restore quorum until primary accepts**: + verify replica reports `completed` but mode stays `needs_rebuild` until + primary processes the completion event. + +### Session lifecycle (unit, engine only) + +1. `idle → accepted → running → completed → keepup` +2. `idle → accepted → running → failed → (syncAck) → re-decide` +3. `running → cancel → idle` +4. Session supersede: new session_id replaces old + +### End-to-end (integration, hardware — after component tests pass) + +1. Fresh replica join on m01/m02: create RF=2, kill replica VS, restart, + verify rebuild completes and volume returns to `publish_healthy`. + +2. Sustained I/O during rebuild: fio running on primary while rebuild + progresses. Verify data continuity after rebuild completion. + +## Implementation Order + +1. Protocol engine (`sw-block/protocol/`) — already started, 7 events, 398 lines +2. Rebuild session on blockvol layer — two-line model, bitmap, completion +3. Session control wiring on volume server — sessionControl/sessionAck +4. Snapshot/CoW base transfer — extent read + chunk send +5. Component tests against real BlockVol +6. Integration test on hardware diff --git a/sw-block/engine/replication/command.go b/sw-block/engine/replication/command.go index 08649680a..5fc5b59be 100644 --- a/sw-block/engine/replication/command.go +++ b/sw-block/engine/replication/command.go @@ -33,6 +33,8 @@ type StartRecoveryTaskCommand struct { Kind SessionKind } +// StartRecoveryTaskCommand realizes the primary-owned session executor for one +// replica after assignment has established membership. func (StartRecoveryTaskCommand) commandName() string { return "start_recovery_task" } type DrainRecoveryTaskCommand struct { @@ -72,4 +74,5 @@ type PublishProjectionCommand struct { Projection PublicationProjection } +// PublishProjectionCommand emits the derived outward summary for the volume. func (PublishProjectionCommand) commandName() string { return "publish_projection" } diff --git a/sw-block/engine/replication/engine.go b/sw-block/engine/replication/engine.go index 6ebf5f0eb..66e757b57 100644 --- a/sw-block/engine/replication/engine.go +++ b/sw-block/engine/replication/engine.go @@ -8,6 +8,11 @@ import ( // CoreEngine is the first explicit Phase 14 V2 core shell. // It is deterministic and side-effect free: one event in, updated state and // commands/projection out. +// +// Ownership model: +// - master-owned assignment truth enters as assignment events +// - primary-owned per-replica session truth enters as sync/recovery events +// - outward mode/publication is always derived projection type CoreEngine struct { volumes map[string]*VolumeState } @@ -79,160 +84,69 @@ func (e *CoreEngine) ApplyEvent(ev Event) ApplyResult { st.commands.InvalidationReason = v.Reason } + case SyncAckObserved: + cmds = append(cmds, e.applySyncAckObserved(st, v)...) + case CheckpointAdvanced: if v.CheckpointLSN > st.Boundary.CheckpointLSN { st.Boundary.CheckpointLSN = v.CheckpointLSN } + case SessionStarted: + cmds = append(cmds, e.applySessionStarted(st, v)...) + + case SessionProgressObserved: + e.applySessionProgressObserved(st, v) + + case SessionCompleted: + e.applySessionCompleted(st, v) + + case SessionFailed: + cmds = append(cmds, e.applySessionFailed(st, v)...) + case CatchUpPlanned: - replicaID, ok := st.recoveryCommandReplicaIDFromEvent(v.ReplicaID) - if ok { - st.recordCatchUpPlan(replicaID, v.TargetLSN) - targetLSN, achievedLSN, _ := st.catchUpAggregate() - st.Recovery.Phase = RecoveryCatchingUp - st.Recovery.AchievedLSN = achievedLSN - st.Recovery.TargetLSN = targetLSN - st.Boundary.TargetLSN = targetLSN - st.Boundary.AchievedLSN = achievedLSN - } else { - st.Recovery.Phase = RecoveryCatchingUp - st.Recovery.AchievedLSN = 0 - if v.TargetLSN > st.Recovery.TargetLSN { - st.Recovery.TargetLSN = v.TargetLSN - } - if v.TargetLSN > st.Boundary.TargetLSN { - st.Boundary.TargetLSN = v.TargetLSN - } - } - st.Recovery.Reason = "" - if ok && st.shouldStartCatchUp(replicaID, v.TargetLSN) { - cmds = append(cmds, StartCatchUpCommand{ - VolumeID: st.VolumeID, - ReplicaID: replicaID, - TargetLSN: v.TargetLSN, - }) - if st.commands.CatchUpTargets == nil { - st.commands.CatchUpTargets = make(map[string]uint64) - } - st.commands.CatchUpTargets[replicaID] = v.TargetLSN - } + cmds = append(cmds, e.applySessionStarted(st, SessionStarted{ + ID: v.ID, + ReplicaID: v.ReplicaID, + Kind: SessionCatchUp, + TargetLSN: v.TargetLSN, + })...) case RecoveryProgressObserved: - if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(v.ReplicaID); ok && st.observeCatchUpProgress(replicaID, v.AchievedLSN) { - _, achievedLSN, _ := st.catchUpAggregate() - st.Recovery.AchievedLSN = achievedLSN - st.Boundary.AchievedLSN = achievedLSN - } else { - if v.AchievedLSN > st.Recovery.AchievedLSN { - st.Recovery.AchievedLSN = v.AchievedLSN - } - if v.AchievedLSN > st.Boundary.AchievedLSN { - st.Boundary.AchievedLSN = v.AchievedLSN - } - } + e.applySessionProgressObserved(st, SessionProgressObserved{ + ID: v.ID, + ReplicaID: v.ReplicaID, + AchievedLSN: v.AchievedLSN, + }) case CatchUpCompleted: - if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(v.ReplicaID); ok && st.completeCatchUp(replicaID, v.AchievedLSN) { - targetLSN, achievedLSN, allDone := st.catchUpAggregate() - st.Recovery.TargetLSN = targetLSN - st.Recovery.AchievedLSN = achievedLSN - st.Boundary.TargetLSN = targetLSN - st.Boundary.AchievedLSN = achievedLSN - if allDone { - if achievedLSN > st.Boundary.DurableLSN { - st.Boundary.DurableLSN = achievedLSN - } - st.Recovery.Phase = RecoveryIdle - st.Recovery.Reason = "" - st.commands.CatchUpTargets = nil - st.catchUps = nil - } else { - st.Recovery.Phase = RecoveryCatchingUp - st.Recovery.Reason = "" - } - } else { - if v.AchievedLSN > st.Recovery.AchievedLSN { - st.Recovery.AchievedLSN = v.AchievedLSN - } - if v.AchievedLSN > st.Boundary.AchievedLSN { - st.Boundary.AchievedLSN = v.AchievedLSN - } - if v.AchievedLSN > st.Boundary.DurableLSN { - st.Boundary.DurableLSN = v.AchievedLSN - } - st.Recovery.Phase = RecoveryIdle - st.Recovery.Reason = "" - st.commands.CatchUpTargets = nil - } + e.applySessionCompleted(st, SessionCompleted{ + ID: v.ID, + ReplicaID: v.ReplicaID, + Kind: SessionCatchUp, + AchievedLSN: v.AchievedLSN, + }) case NeedsRebuildObserved: - st.catchUps = nil - st.commands.CatchUpTargets = nil - st.needsRebuild = true - st.rebuildReason = v.Reason - st.degraded = false - st.degradeReason = "" - st.Recovery.Phase = RecoveryNeedsRebuild - st.Recovery.Reason = v.Reason - if st.shouldInvalidate(v.Reason) { - replicaID, _ := st.recoveryCommandReplicaIDFromEvent(v.ReplicaID) - cmds = append(cmds, InvalidateSessionCommand{ - VolumeID: st.VolumeID, - ReplicaID: replicaID, - Reason: v.Reason, - }) - st.commands.InvalidationIssued = true - st.commands.InvalidationReason = v.Reason - } + cmds = append(cmds, e.applyNeedsRebuildObserved(st, v)...) case RebuildStarted: - st.needsRebuild = true - st.Recovery.Phase = RecoveryRebuilding - st.Recovery.Reason = st.rebuildReason - st.Recovery.AchievedLSN = 0 - if v.TargetLSN > st.Recovery.TargetLSN { - st.Recovery.TargetLSN = v.TargetLSN - } - if v.TargetLSN > st.Boundary.TargetLSN { - st.Boundary.TargetLSN = v.TargetLSN - } - if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(v.ReplicaID); ok && st.shouldStartRebuild(replicaID, v.TargetLSN) { - cmds = append(cmds, StartRebuildCommand{ - VolumeID: st.VolumeID, - ReplicaID: replicaID, - TargetLSN: v.TargetLSN, - }) - if st.commands.RebuildTargets == nil { - st.commands.RebuildTargets = make(map[string]uint64) - } - st.commands.RebuildTargets[replicaID] = v.TargetLSN - } + cmds = append(cmds, e.applySessionStarted(st, SessionStarted{ + ID: v.ID, + ReplicaID: v.ReplicaID, + Kind: SessionRebuild, + TargetLSN: v.TargetLSN, + })...) case RebuildCommitted: - st.needsRebuild = false - st.rebuildReason = "" - st.degraded = false - st.degradeReason = "" - st.resetInvalidation() - st.Recovery.Phase = RecoveryIdle - st.Recovery.Reason = "" - if v.FlushedLSN > st.Boundary.DurableLSN { - st.Boundary.DurableLSN = v.FlushedLSN - } - if v.CheckpointLSN > st.Boundary.CheckpointLSN { - st.Boundary.CheckpointLSN = v.CheckpointLSN - } - achievedLSN := v.AchievedLSN - if achievedLSN == 0 { - achievedLSN = maxUint64(v.FlushedLSN, v.CheckpointLSN) - } - if achievedLSN > st.Recovery.AchievedLSN { - st.Recovery.AchievedLSN = achievedLSN - } - if achievedLSN > st.Boundary.AchievedLSN { - st.Boundary.AchievedLSN = achievedLSN - } - st.commands.RebuildTargets = nil + e.applySessionCompleted(st, SessionCompleted{ + ID: v.ID, + ReplicaID: v.ReplicaID, + Kind: SessionRebuild, + AchievedLSN: v.AchievedLSN, + FlushedLSN: v.FlushedLSN, + CheckpointLSN: v.CheckpointLSN, + }) } e.recompute(st) @@ -279,6 +193,7 @@ func (e *CoreEngine) mustState(volumeID string) *VolumeState { } func (e *CoreEngine) recompute(st *VolumeState) { + st.refreshRecoveryAggregate() st.Readiness.ReplicaReady = st.Role == RoleReplica && st.Readiness.RoleApplied && st.Readiness.ReceiverReady @@ -287,8 +202,8 @@ func (e *CoreEngine) recompute(st *VolumeState) { st.Publication = PublicationView{} switch { - case st.needsRebuild: - st.Publication.Reason = defaultReason(st.rebuildReason, "needs_rebuild") + case st.hasRebuilds(): + st.Publication.Reason = defaultReason(st.aggregateRebuildReason(), "needs_rebuild") st.Mode.Name = ModeNeedsRebuild st.Mode.Reason = st.Publication.Reason case st.degraded: @@ -360,13 +275,14 @@ func (e *CoreEngine) applyAssignment(st *VolumeState, ev AssignmentDelivered) [] st.degradeReason = "" st.resetInvalidation() st.Recovery = RecoveryView{Phase: RecoveryIdle} + st.Sync = SyncView{} + st.ReplicaSync = nil st.Boundary.TargetLSN = 0 st.Boundary.AchievedLSN = 0 - st.catchUps = nil + st.sessions = nil st.commands.RecoveryTaskEpoch = 0 st.commands.RecoveryTaskTargets = nil - st.commands.CatchUpTargets = nil - st.commands.RebuildTargets = nil + st.commands.SessionTargets = nil } var cmds []Command @@ -416,6 +332,182 @@ func (e *CoreEngine) applyAssignment(st *VolumeState, ev AssignmentDelivered) [] return cmds } +func (e *CoreEngine) applySessionStarted(st *VolumeState, ev SessionStarted) []Command { + replicaID, ok := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID) + switch ev.Kind { + case SessionRebuild: + reason := ev.Reason + if reason == "" { + reason = st.rebuildReasonForReplica(replicaID) + } + st.recordRebuild(replicaID, RecoveryRebuilding, reason, ev.TargetLSN) + st.Recovery.AchievedLSN = 0 + if ev.TargetLSN > st.Recovery.TargetLSN { + st.Recovery.TargetLSN = ev.TargetLSN + } + if ev.TargetLSN > st.Boundary.TargetLSN { + st.Boundary.TargetLSN = ev.TargetLSN + } + if !ok { + return nil + } + return st.startSessionCommand(st.VolumeID, replicaID, SessionRebuild, ev.TargetLSN) + + case SessionCatchUp: + default: + return nil + } + if ok { + st.recordCatchUpPlan(replicaID, ev.TargetLSN) + targetLSN, achievedLSN, _ := st.catchUpAggregate() + st.Recovery.Phase = RecoveryCatchingUp + st.Recovery.AchievedLSN = achievedLSN + st.Recovery.TargetLSN = targetLSN + st.Boundary.TargetLSN = targetLSN + st.Boundary.AchievedLSN = achievedLSN + } else { + st.Recovery.Phase = RecoveryCatchingUp + st.Recovery.AchievedLSN = 0 + if ev.TargetLSN > st.Recovery.TargetLSN { + st.Recovery.TargetLSN = ev.TargetLSN + } + if ev.TargetLSN > st.Boundary.TargetLSN { + st.Boundary.TargetLSN = ev.TargetLSN + } + } + st.Recovery.Reason = "" + if !ok { + return nil + } + return st.startSessionCommand(st.VolumeID, replicaID, SessionCatchUp, ev.TargetLSN) +} + +func (e *CoreEngine) applySessionProgressObserved(st *VolumeState, ev SessionProgressObserved) { + if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID); ok && st.observeCatchUpProgress(replicaID, ev.AchievedLSN) { + _, achievedLSN, _ := st.catchUpAggregate() + st.Recovery.AchievedLSN = achievedLSN + st.Boundary.AchievedLSN = achievedLSN + return + } + if ev.AchievedLSN > st.Recovery.AchievedLSN { + st.Recovery.AchievedLSN = ev.AchievedLSN + } + if ev.AchievedLSN > st.Boundary.AchievedLSN { + st.Boundary.AchievedLSN = ev.AchievedLSN + } +} + +func (e *CoreEngine) applySessionCompleted(st *VolumeState, ev SessionCompleted) { + if ev.Kind == SessionRebuild { + st.degraded = false + st.degradeReason = "" + replicaID, _ := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID) + st.clearRebuild(replicaID) + if !st.hasRebuilds() { + st.resetInvalidation() + } + st.Recovery.Phase = RecoveryIdle + st.Recovery.Reason = "" + if ev.FlushedLSN > st.Boundary.DurableLSN { + st.Boundary.DurableLSN = ev.FlushedLSN + } + if ev.CheckpointLSN > st.Boundary.CheckpointLSN { + st.Boundary.CheckpointLSN = ev.CheckpointLSN + } + achievedLSN := ev.AchievedLSN + if achievedLSN == 0 { + achievedLSN = maxUint64(ev.FlushedLSN, ev.CheckpointLSN) + } + if achievedLSN > st.Recovery.AchievedLSN { + st.Recovery.AchievedLSN = achievedLSN + } + if achievedLSN > st.Boundary.AchievedLSN { + st.Boundary.AchievedLSN = achievedLSN + } + st.clearSessionCommand(replicaID, SessionRebuild) + return + } + + if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID); ok && st.completeCatchUp(replicaID, ev.AchievedLSN) { + targetLSN, achievedLSN, allDone := st.catchUpAggregate() + st.Recovery.TargetLSN = targetLSN + st.Recovery.AchievedLSN = achievedLSN + st.Boundary.TargetLSN = targetLSN + st.Boundary.AchievedLSN = achievedLSN + if allDone { + if achievedLSN > st.Boundary.DurableLSN { + st.Boundary.DurableLSN = achievedLSN + } + st.Recovery.Phase = RecoveryIdle + st.Recovery.Reason = "" + } else { + st.Recovery.Phase = RecoveryCatchingUp + st.Recovery.Reason = "" + } + return + } + if ev.AchievedLSN > st.Recovery.AchievedLSN { + st.Recovery.AchievedLSN = ev.AchievedLSN + } + if ev.AchievedLSN > st.Boundary.AchievedLSN { + st.Boundary.AchievedLSN = ev.AchievedLSN + } + if ev.AchievedLSN > st.Boundary.DurableLSN { + st.Boundary.DurableLSN = ev.AchievedLSN + } + st.Recovery.Phase = RecoveryIdle + st.Recovery.Reason = "" + st.clearSessionCommand("", SessionCatchUp) +} + +func (e *CoreEngine) applyNeedsRebuildObserved(st *VolumeState, ev NeedsRebuildObserved) []Command { + replicaID, _ := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID) + st.clearCatchUp(replicaID) + st.recordRebuild(replicaID, RecoveryNeedsRebuild, ev.Reason, 0) + st.degraded = false + st.degradeReason = "" + if !st.shouldInvalidate(ev.Reason) { + return nil + } + st.commands.InvalidationIssued = true + st.commands.InvalidationReason = ev.Reason + return []Command{InvalidateSessionCommand{ + VolumeID: st.VolumeID, + ReplicaID: replicaID, + Reason: ev.Reason, + }} +} + +func (e *CoreEngine) applySessionFailed(st *VolumeState, ev SessionFailed) []Command { + replicaID, _ := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID) + switch ev.Kind { + case SessionRebuild: + st.clearRebuild(replicaID) + if !st.hasRebuilds() { + st.resetInvalidation() + } + case SessionCatchUp: + st.clearCatchUp(replicaID) + } + st.clearSessionCommand(replicaID, ev.Kind) + st.Recovery.Phase = RecoveryIdle + st.Recovery.Reason = ev.Reason + st.degraded = true + st.degradeReason = defaultReason(ev.Reason, "session_failed") + st.Boundary.LastBarrierOK = false + st.Boundary.LastBarrierReason = st.degradeReason + if !st.shouldInvalidate(st.degradeReason) { + return nil + } + st.commands.InvalidationIssued = true + st.commands.InvalidationReason = st.degradeReason + return []Command{InvalidateSessionCommand{ + VolumeID: st.VolumeID, + ReplicaID: replicaID, + Reason: st.degradeReason, + }} +} + func (e *CoreEngine) projectionFor(st *VolumeState) PublicationProjection { replicaIDs := make([]string, 0, len(st.DesiredReplicas)) for _, replica := range st.DesiredReplicas { @@ -428,6 +520,8 @@ func (e *CoreEngine) projectionFor(st *VolumeState) PublicationProjection { Mode: st.Mode, Publication: st.Publication, Recovery: st.Recovery, + Sync: st.Sync, + ReplicaSync: cloneReplicaSyncView(st.ReplicaSync), Readiness: st.Readiness, Boundary: st.Boundary, ReplicaIDs: replicaIDs, @@ -478,24 +572,72 @@ func (st *VolumeState) shouldInvalidate(reason string) bool { return !st.commands.InvalidationIssued || st.commands.InvalidationReason != reason } -func (st *VolumeState) shouldStartCatchUp(replicaID string, targetLSN uint64) bool { +func (st *VolumeState) shouldStartSessionCommand(replicaID string, kind SessionKind, targetLSN uint64) bool { if targetLSN == 0 || replicaID == "" { return false } - if st.commands.CatchUpTargets == nil { + if obs, ok := st.sessions[replicaID]; ok && + obs.Kind == kind && + obs.Completed && + obs.TargetLSN == targetLSN && + obs.AchievedLSN >= targetLSN { + return false + } + if st.commands.SessionTargets == nil { return true } - return st.commands.CatchUpTargets[replicaID] != targetLSN + target, ok := st.commands.SessionTargets[replicaID] + if !ok { + return true + } + return target.Kind != kind || target.TargetLSN != targetLSN } -func (st *VolumeState) shouldStartRebuild(replicaID string, targetLSN uint64) bool { - if targetLSN == 0 || replicaID == "" { - return false +func (st *VolumeState) startSessionCommand(volumeID, replicaID string, kind SessionKind, targetLSN uint64) []Command { + if !st.shouldStartSessionCommand(replicaID, kind, targetLSN) { + return nil } - if st.commands.RebuildTargets == nil { - return true + if st.commands.SessionTargets == nil { + st.commands.SessionTargets = make(map[string]sessionCommandTarget) + } + st.commands.SessionTargets[replicaID] = sessionCommandTarget{ + Kind: kind, + TargetLSN: targetLSN, + } + switch kind { + case SessionCatchUp: + return []Command{StartCatchUpCommand{ + VolumeID: volumeID, + ReplicaID: replicaID, + TargetLSN: targetLSN, + }} + case SessionRebuild: + return []Command{StartRebuildCommand{ + VolumeID: volumeID, + ReplicaID: replicaID, + TargetLSN: targetLSN, + }} + default: + return nil + } +} + +func (st *VolumeState) clearSessionCommand(replicaID string, kind SessionKind) { + if st.commands.SessionTargets == nil { + return + } + if replicaID == "" { + for id, target := range st.commands.SessionTargets { + if target.Kind == kind { + delete(st.commands.SessionTargets, id) + } + } + } else if target, ok := st.commands.SessionTargets[replicaID]; ok && target.Kind == kind { + delete(st.commands.SessionTargets, replicaID) + } + if len(st.commands.SessionTargets) == 0 { + st.commands.SessionTargets = nil } - return st.commands.RebuildTargets[replicaID] != targetLSN } func (st *VolumeState) resetInvalidation() { @@ -577,39 +719,62 @@ func (st *VolumeState) recordCatchUpPlan(replicaID string, targetLSN uint64) { if replicaID == "" || targetLSN == 0 { return } - if st.catchUps == nil { - st.catchUps = make(map[string]catchUpObservation) + if st.sessions == nil { + st.sessions = make(map[string]sessionObservation) + } + obs := st.sessions[replicaID] + if obs.Kind != SessionCatchUp { + obs = sessionObservation{Kind: SessionCatchUp, Phase: RecoveryCatchingUp} + } + if obs.Completed && obs.TargetLSN == targetLSN && obs.AchievedLSN >= targetLSN { + return } - obs := st.catchUps[replicaID] if obs.TargetLSN != targetLSN { obs.AchievedLSN = 0 } + obs.Kind = SessionCatchUp + obs.Phase = RecoveryCatchingUp obs.TargetLSN = targetLSN + obs.Reason = "" obs.Completed = false - st.catchUps[replicaID] = obs + st.sessions[replicaID] = obs +} + +func (st *VolumeState) clearCatchUp(replicaID string) { + if replicaID == "" || st.sessions == nil { + return + } + if obs, ok := st.sessions[replicaID]; ok && obs.Kind == SessionCatchUp { + delete(st.sessions, replicaID) + } + if len(st.sessions) == 0 { + st.sessions = nil + } + st.clearSessionCommand(replicaID, SessionCatchUp) } func (st *VolumeState) observeCatchUpProgress(replicaID string, achievedLSN uint64) bool { - if replicaID == "" || st.catchUps == nil { + if replicaID == "" || st.sessions == nil { return false } - obs, ok := st.catchUps[replicaID] - if !ok { + obs, ok := st.sessions[replicaID] + if !ok || obs.Kind != SessionCatchUp { return false } if achievedLSN > obs.AchievedLSN { obs.AchievedLSN = achievedLSN } - st.catchUps[replicaID] = obs + obs.Phase = RecoveryCatchingUp + st.sessions[replicaID] = obs return true } func (st *VolumeState) completeCatchUp(replicaID string, achievedLSN uint64) bool { - if replicaID == "" || st.catchUps == nil { + if replicaID == "" || st.sessions == nil { return false } - obs, ok := st.catchUps[replicaID] - if !ok { + obs, ok := st.sessions[replicaID] + if !ok || obs.Kind != SessionCatchUp { return false } if achievedLSN > obs.AchievedLSN { @@ -618,18 +783,24 @@ func (st *VolumeState) completeCatchUp(replicaID string, achievedLSN uint64) boo if obs.TargetLSN > obs.AchievedLSN { obs.AchievedLSN = obs.TargetLSN } + obs.Phase = RecoveryIdle obs.Completed = true - st.catchUps[replicaID] = obs + st.sessions[replicaID] = obs return true } func (st *VolumeState) catchUpAggregate() (targetLSN uint64, achievedLSN uint64, allDone bool) { - if len(st.catchUps) == 0 { + if len(st.sessions) == 0 { return 0, 0, true } allDone = true first := true - for _, obs := range st.catchUps { + found := false + for _, obs := range st.sessions { + if obs.Kind != SessionCatchUp { + continue + } + found = true if obs.TargetLSN > targetLSN { targetLSN = obs.TargetLSN } @@ -641,9 +812,122 @@ func (st *VolumeState) catchUpAggregate() (targetLSN uint64, achievedLSN uint64, allDone = false } } + if !found { + return 0, 0, true + } return targetLSN, achievedLSN, allDone } +func (st *VolumeState) recordRebuild(replicaID string, phase RecoveryPhase, reason string, targetLSN uint64) { + if replicaID == "" { + return + } + if st.sessions == nil { + st.sessions = make(map[string]sessionObservation) + } + if phase == "" { + phase = RecoveryNeedsRebuild + } + obs := st.sessions[replicaID] + obs.Kind = SessionRebuild + obs.Phase = phase + obs.Reason = reason + obs.TargetLSN = targetLSN + obs.Completed = false + st.sessions[replicaID] = obs +} + +func (st *VolumeState) clearRebuild(replicaID string) { + if replicaID == "" || st.sessions == nil { + return + } + if obs, ok := st.sessions[replicaID]; ok && obs.Kind == SessionRebuild { + delete(st.sessions, replicaID) + } + if len(st.sessions) == 0 { + st.sessions = nil + } +} + +func (st *VolumeState) hasRebuilds() bool { + for _, obs := range st.sessions { + if obs.Kind == SessionRebuild { + return true + } + } + return false +} + +func (st *VolumeState) rebuildReasonForReplica(replicaID string) string { + if replicaID == "" || st.sessions == nil { + return "" + } + obs, ok := st.sessions[replicaID] + if !ok || obs.Kind != SessionRebuild { + return "" + } + return obs.Reason +} + +func (st *VolumeState) aggregateRebuildReason() string { + for _, obs := range st.sessions { + if obs.Kind == SessionRebuild && obs.Reason != "" { + return obs.Reason + } + } + return "" +} + +func (st *VolumeState) refreshRecoveryAggregate() { + if st.hasRebuilds() { + phase := RecoveryNeedsRebuild + reason := "" + targetLSN := st.Recovery.TargetLSN + for _, obs := range st.sessions { + if obs.Kind != SessionRebuild { + continue + } + if obs.Phase == RecoveryRebuilding { + phase = RecoveryRebuilding + } + if reason == "" && obs.Reason != "" { + reason = obs.Reason + } + if obs.TargetLSN > targetLSN { + targetLSN = obs.TargetLSN + } + } + st.Recovery.Phase = phase + st.Recovery.Reason = reason + st.Recovery.TargetLSN = targetLSN + if targetLSN > st.Boundary.TargetLSN { + st.Boundary.TargetLSN = targetLSN + } + return + } + if targetLSN, achievedLSN, allDone := st.catchUpAggregate(); targetLSN > 0 { + if allDone { + st.Recovery.Phase = RecoveryIdle + } else { + st.Recovery.Phase = RecoveryCatchingUp + } + st.Recovery.Reason = "" + st.Recovery.TargetLSN = targetLSN + st.Recovery.AchievedLSN = achievedLSN + if targetLSN > st.Boundary.TargetLSN { + st.Boundary.TargetLSN = targetLSN + } + if achievedLSN > st.Boundary.AchievedLSN { + st.Boundary.AchievedLSN = achievedLSN + } + return + } + if st.Recovery.Phase == RecoveryNeedsRebuild || st.Recovery.Phase == RecoveryRebuilding || st.Recovery.Phase == RecoveryCatchingUp { + st.Recovery.Phase = RecoveryIdle + st.Recovery.Reason = "" + } +} + func (st *VolumeState) bootstrapReason() string { switch { case !st.Readiness.RoleApplied: @@ -667,6 +951,94 @@ func (st *VolumeState) recoveryActive() bool { return st.Recovery.Phase == RecoveryCatchingUp || st.Recovery.Phase == RecoveryRebuilding } +func (e *CoreEngine) applySyncAckObserved(st *VolumeState, ev SyncAckObserved) []Command { + syncView := SyncView{ + AckKind: ev.AckKind, + TargetLSN: ev.TargetLSN, + PrimaryTailLSN: ev.PrimaryTailLSN, + DurableLSN: ev.DurableLSN, + AppliedLSN: ev.AppliedLSN, + Reason: ev.Reason, + } + st.Sync = syncView + if ev.ReplicaID != "" { + if st.ReplicaSync == nil { + st.ReplicaSync = make(ReplicaSyncView) + } + st.ReplicaSync[ev.ReplicaID] = syncView + } + if ev.DurableLSN > st.Boundary.DurableLSN { + st.Boundary.DurableLSN = ev.DurableLSN + } + switch ev.AckKind { + case SyncAckEpochMismatch: + st.Boundary.LastBarrierOK = false + st.Boundary.LastBarrierReason = defaultReason(ev.Reason, "epoch_mismatch") + return nil + } + + action, achievedLSN, ok := decideSyncAction(ev) + if !ok { + if ev.AckKind == SyncAckTimedOut { + st.Boundary.LastBarrierOK = false + st.Boundary.LastBarrierReason = defaultReason(ev.Reason, string(ev.AckKind)) + } + return nil + } + + switch action { + case SyncActionKeepUp: + st.Boundary.LastBarrierOK = true + st.Boundary.LastBarrierReason = "" + st.degraded = false + st.degradeReason = "" + st.Sync.Action = SyncActionKeepUp + if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID); ok { + st.clearRebuild(replicaID) + } + if ev.ReplicaID != "" && st.ReplicaSync != nil { + st.ReplicaSync[ev.ReplicaID] = st.Sync + } + return nil + + case SyncActionCatchUp: + st.Boundary.LastBarrierOK = false + st.Boundary.LastBarrierReason = defaultReason(ev.Reason, string(ev.AckKind)) + st.degraded = false + st.degradeReason = "" + st.Sync.Action = SyncActionCatchUp + if ev.ReplicaID != "" && st.ReplicaSync != nil { + st.ReplicaSync[ev.ReplicaID] = st.Sync + } + if replicaID, ok := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID); ok { + st.recordCatchUpPlan(replicaID, ev.TargetLSN) + if achievedLSN > 0 { + st.observeCatchUpProgress(replicaID, achievedLSN) + } + targetLSN, aggregateAchievedLSN, _ := st.catchUpAggregate() + st.Recovery.Phase = RecoveryCatchingUp + st.Recovery.TargetLSN = targetLSN + st.Recovery.AchievedLSN = aggregateAchievedLSN + st.Recovery.Reason = "" + st.Boundary.TargetLSN = targetLSN + st.Boundary.AchievedLSN = aggregateAchievedLSN + return st.startSessionCommand(st.VolumeID, replicaID, SessionCatchUp, ev.TargetLSN) + } + st.Recovery.Phase = RecoveryCatchingUp + st.Recovery.TargetLSN = maxUint64(st.Recovery.TargetLSN, ev.TargetLSN) + st.Recovery.AchievedLSN = maxUint64(st.Recovery.AchievedLSN, achievedLSN) + st.Recovery.Reason = "" + st.Boundary.TargetLSN = maxUint64(st.Boundary.TargetLSN, ev.TargetLSN) + st.Boundary.AchievedLSN = maxUint64(st.Boundary.AchievedLSN, achievedLSN) + return nil + + case SyncActionRebuild: + return e.applySyncNeedsRebuild(st, ev) + } + + return nil +} + func defaultReason(reason, fallback string) string { if reason != "" { return reason @@ -674,6 +1046,79 @@ func defaultReason(reason, fallback string) string { return fallback } +func syncAckAchievedLSN(ev SyncAckObserved) uint64 { + return maxUint64(ev.DurableLSN, ev.AppliedLSN) +} + +func syncAckSupportsCatchUp(ev SyncAckObserved) bool { + if ev.TargetLSN == 0 { + return false + } + return maxUint64(ev.AppliedLSN, ev.DurableLSN) >= ev.PrimaryTailLSN +} + +func syncAckNeedsRebuild(ev SyncAckObserved) bool { + if ev.TargetLSN == 0 { + return false + } + return !syncAckSupportsCatchUp(ev) +} + +func decideSyncAction(ev SyncAckObserved) (SyncAction, uint64, bool) { + switch ev.AckKind { + case SyncAckQuorum: + return SyncActionKeepUp, 0, true + case SyncAckTimedOut: + if syncAckNeedsRebuild(ev) { + return SyncActionRebuild, 0, true + } + if syncAckSupportsCatchUp(ev) { + return SyncActionCatchUp, syncAckAchievedLSN(ev), true + } + return "", 0, false + case SyncAckTransportLost: + return SyncActionRebuild, 0, true + default: + return "", 0, false + } +} + +func (e *CoreEngine) applySyncNeedsRebuild(st *VolumeState, ev SyncAckObserved) []Command { + reason := defaultReason(ev.Reason, "needs_rebuild") + replicaID, _ := st.recoveryCommandReplicaIDFromEvent(ev.ReplicaID) + st.Sync.Action = SyncActionRebuild + if ev.ReplicaID != "" && st.ReplicaSync != nil { + st.ReplicaSync[ev.ReplicaID] = st.Sync + } + st.clearCatchUp(replicaID) + st.recordRebuild(replicaID, RecoveryNeedsRebuild, reason, ev.TargetLSN) + st.degraded = false + st.degradeReason = "" + st.Boundary.LastBarrierOK = false + st.Boundary.LastBarrierReason = reason + if st.shouldInvalidate(reason) { + st.commands.InvalidationIssued = true + st.commands.InvalidationReason = reason + return []Command{InvalidateSessionCommand{ + VolumeID: st.VolumeID, + ReplicaID: replicaID, + Reason: reason, + }} + } + return nil +} + +func cloneReplicaSyncView(in ReplicaSyncView) ReplicaSyncView { + if len(in) == 0 { + return nil + } + out := make(ReplicaSyncView, len(in)) + for replicaID, syncView := range in { + out[replicaID] = syncView + } + return out +} + func sameReplicaAssignments(left, right []ReplicaAssignment) bool { if len(left) != len(right) { return false diff --git a/sw-block/engine/replication/event.go b/sw-block/engine/replication/event.go index 4b0d6aba8..e46b37a1a 100644 --- a/sw-block/engine/replication/event.go +++ b/sw-block/engine/replication/event.go @@ -6,7 +6,8 @@ type Event interface { VolumeID() string } -// AssignmentDelivered carries the desired local role and replica set. +// AssignmentDelivered carries master-owned assignment truth: local role, epoch, +// and replica membership for one volume. type AssignmentDelivered struct { ID string Epoch uint64 @@ -78,6 +79,21 @@ type BarrierRejected struct { func (e BarrierRejected) VolumeID() string { return e.ID } +// SyncAckObserved records one sync ack plus the bounded facts the primary needs +// to decide the next session step (keep-up, catch-up, or rebuild). +type SyncAckObserved struct { + ReplicaID string + ID string + AckKind SyncAckKind + TargetLSN uint64 + PrimaryTailLSN uint64 + DurableLSN uint64 + AppliedLSN uint64 + Reason string +} + +func (e SyncAckObserved) VolumeID() string { return e.ID } + // CheckpointAdvanced updates the durable base-image boundary. type CheckpointAdvanced struct { ID string @@ -86,6 +102,50 @@ type CheckpointAdvanced struct { func (e CheckpointAdvanced) VolumeID() string { return e.ID } +// SessionStarted begins one primary-owned session contract for a replica. +type SessionStarted struct { + ReplicaID string + ID string + Kind SessionKind + TargetLSN uint64 + Reason string +} + +func (e SessionStarted) VolumeID() string { return e.ID } + +// SessionProgressObserved updates bounded progress for one running session. +type SessionProgressObserved struct { + ReplicaID string + ID string + Kind SessionKind + AchievedLSN uint64 +} + +func (e SessionProgressObserved) VolumeID() string { return e.ID } + +// SessionCompleted closes one session contract at an explicit achieved boundary. +type SessionCompleted struct { + ReplicaID string + ID string + Kind SessionKind + AchievedLSN uint64 + FlushedLSN uint64 + CheckpointLSN uint64 +} + +func (e SessionCompleted) VolumeID() string { return e.ID } + +// SessionFailed reports one failed session attempt without independently choosing +// the next semantic recovery path. +type SessionFailed struct { + ReplicaID string + ID string + Kind SessionKind + Reason string +} + +func (e SessionFailed) VolumeID() string { return e.ID } + // CatchUpPlanned freezes the current replay target as bounded recovery truth. type CatchUpPlanned struct { ReplicaID string diff --git a/sw-block/engine/replication/phase14_sync_test.go b/sw-block/engine/replication/phase14_sync_test.go new file mode 100644 index 000000000..a7c68eb08 --- /dev/null +++ b/sw-block/engine/replication/phase14_sync_test.go @@ -0,0 +1,519 @@ +package replication + +import "testing" + +func TestPhase14_SyncAckObserved_AckUpdatesViewAndDurability(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-ack", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.50:9333", CtrlAddr: "10.0.0.50:9334", Version: 1}}, + }, + }) + core.ApplyEvent(RoleApplied{ID: "vol-sync-ack"}) + core.ApplyEvent(ShipperConfiguredObserved{ID: "vol-sync-ack"}) + core.ApplyEvent(ShipperConnectedObserved{ID: "vol-sync-ack"}) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-ack", + ReplicaID: "replica-1", + AckKind: SyncAckQuorum, + TargetLSN: 120, + DurableLSN: 120, + AppliedLSN: 120, + }) + + if result.Projection.Sync.AckKind != SyncAckQuorum { + t.Fatalf("sync_ack_kind=%s", result.Projection.Sync.AckKind) + } + if result.Projection.Sync.Action != SyncActionKeepUp { + t.Fatalf("sync_action=%s", result.Projection.Sync.Action) + } + if result.Projection.Boundary.DurableLSN != 120 { + t.Fatalf("durable_lsn=%d", result.Projection.Boundary.DurableLSN) + } + if result.Projection.Mode.Name != ModePublishHealthy { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if !result.Projection.Publication.Healthy { + t.Fatal("ack should establish healthy publication on ready primary") + } +} + +func TestPhase14_SyncAckObserved_TimedOutFactsStartCatchUp(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-catchup", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.51:9333", CtrlAddr: "10.0.0.51:9334", Version: 1}}, + }, + }) + core.ApplyEvent(RoleApplied{ID: "vol-sync-catchup"}) + core.ApplyEvent(ShipperConfiguredObserved{ID: "vol-sync-catchup"}) + core.ApplyEvent(ShipperConnectedObserved{ID: "vol-sync-catchup"}) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-catchup", + ReplicaID: "replica-1", + AckKind: SyncAckTimedOut, + TargetLSN: 1000, + PrimaryTailLSN: 700, + DurableLSN: 500, + AppliedLSN: 900, + Reason: "gap_within_retention", + }) + + assertCommandNames(t, result.Commands, []string{ + "start_catchup", + "publish_projection", + }) + if result.State.Recovery.Phase != RecoveryCatchingUp { + t.Fatalf("recovery_phase=%s", result.State.Recovery.Phase) + } + if result.Projection.Recovery.TargetLSN != 1000 { + t.Fatalf("target_lsn=%d", result.Projection.Recovery.TargetLSN) + } + if result.Projection.Recovery.AchievedLSN != 900 { + t.Fatalf("achieved_lsn=%d", result.Projection.Recovery.AchievedLSN) + } + if result.Projection.Sync.AckKind != SyncAckTimedOut { + t.Fatalf("sync_ack_kind=%s", result.Projection.Sync.AckKind) + } + if result.Projection.Sync.Action != SyncActionCatchUp { + t.Fatalf("sync_action=%s", result.Projection.Sync.Action) + } + if result.Projection.Mode.Name != ModeBootstrapPending { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if result.Projection.Publication.Healthy { + t.Fatal("recoverable catch-up should not overclaim healthy publication") + } +} + +func TestPhase14_SyncAckObserved_TimedOutStillPreservesCatchUpContract(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-timeout", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.52:9333", CtrlAddr: "10.0.0.52:9334", Version: 1}}, + }, + }) + core.ApplyEvent(RoleApplied{ID: "vol-sync-timeout"}) + core.ApplyEvent(ShipperConfiguredObserved{ID: "vol-sync-timeout"}) + core.ApplyEvent(ShipperConnectedObserved{ID: "vol-sync-timeout"}) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-timeout", + ReplicaID: "replica-1", + AckKind: SyncAckTimedOut, + TargetLSN: 1000, + PrimaryTailLSN: 700, + DurableLSN: 500, + AppliedLSN: 880, + Reason: "deadline_exceeded", + }) + + assertCommandNames(t, result.Commands, []string{ + "start_catchup", + "publish_projection", + }) + if result.Projection.Mode.Name != ModeBootstrapPending { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if result.Projection.Recovery.Phase != RecoveryCatchingUp { + t.Fatalf("recovery_phase=%s", result.Projection.Recovery.Phase) + } + if result.Projection.Recovery.AchievedLSN != 880 { + t.Fatalf("achieved_lsn=%d", result.Projection.Recovery.AchievedLSN) + } + if result.Projection.Boundary.LastBarrierReason != "deadline_exceeded" { + t.Fatalf("last_barrier_reason=%q", result.Projection.Boundary.LastBarrierReason) + } +} + +func TestPhase14_SyncAckObserved_TransportLostEscalatesTargetedReplica(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-rebuild", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.53:9333", CtrlAddr: "10.0.0.53:9334", Version: 1}}, + {ReplicaID: "replica-2", Endpoint: Endpoint{DataAddr: "10.0.0.54:9333", CtrlAddr: "10.0.0.54:9334", Version: 1}}, + }, + }) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-rebuild", + ReplicaID: "replica-2", + AckKind: SyncAckTransportLost, + TargetLSN: 1200, + DurableLSN: 512, + AppliedLSN: 900, + Reason: "recoverability_lost", + }) + + assertCommandNames(t, result.Commands, []string{ + "invalidate_session", + "publish_projection", + }) + invalidate, ok := result.Commands[0].(InvalidateSessionCommand) + if !ok { + t.Fatalf("cmd0=%T", result.Commands[0]) + } + if invalidate.ReplicaID != "replica-2" { + t.Fatalf("invalidate_replica=%q", invalidate.ReplicaID) + } + if result.Projection.Mode.Name != ModeNeedsRebuild { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if result.Projection.Recovery.Phase != RecoveryNeedsRebuild { + t.Fatalf("recovery_phase=%s", result.Projection.Recovery.Phase) + } + if result.Projection.Sync.Action != SyncActionRebuild { + t.Fatalf("sync_action=%s", result.Projection.Sync.Action) + } +} + +func TestPhase14_SyncAckObserved_TracksPerReplicaSyncFacts(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-multi", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.55:9333", CtrlAddr: "10.0.0.55:9334", Version: 1}}, + {ReplicaID: "replica-2", Endpoint: Endpoint{DataAddr: "10.0.0.56:9333", CtrlAddr: "10.0.0.56:9334", Version: 1}}, + }, + }) + core.ApplyEvent(RoleApplied{ID: "vol-sync-multi"}) + core.ApplyEvent(ShipperConfiguredObserved{ID: "vol-sync-multi"}) + core.ApplyEvent(ShipperConnectedObserved{ID: "vol-sync-multi"}) + + core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-multi", + ReplicaID: "replica-1", + AckKind: SyncAckTimedOut, + TargetLSN: 1000, + PrimaryTailLSN: 700, + DurableLSN: 400, + AppliedLSN: 850, + Reason: "deadline_exceeded", + }) + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-multi", + ReplicaID: "replica-2", + AckKind: SyncAckQuorum, + TargetLSN: 1000, + DurableLSN: 1000, + AppliedLSN: 1000, + }) + + if len(result.Projection.ReplicaSync) != 2 { + t.Fatalf("replica_sync_len=%d", len(result.Projection.ReplicaSync)) + } + if got := result.Projection.ReplicaSync["replica-1"].AckKind; got != SyncAckTimedOut { + t.Fatalf("replica1_ack_kind=%s", got) + } + if got := result.Projection.ReplicaSync["replica-1"].Action; got != SyncActionCatchUp { + t.Fatalf("replica1_action=%s", got) + } + if got := result.Projection.ReplicaSync["replica-2"].AckKind; got != SyncAckQuorum { + t.Fatalf("replica2_ack_kind=%s", got) + } + if got := result.Projection.ReplicaSync["replica-2"].Action; got != SyncActionKeepUp { + t.Fatalf("replica2_action=%s", got) + } +} + +func TestPhase14_SyncAckObserved_ReplicaAckDoesNotClearOtherReplicaRebuild(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-rebuild-aggregate", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.61:9333", CtrlAddr: "10.0.0.61:9334", Version: 1}}, + {ReplicaID: "replica-2", Endpoint: Endpoint{DataAddr: "10.0.0.62:9333", CtrlAddr: "10.0.0.62:9334", Version: 1}}, + }, + }) + + core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-rebuild-aggregate", + ReplicaID: "replica-1", + AckKind: SyncAckTimedOut, + TargetLSN: 1000, + PrimaryTailLSN: 700, + DurableLSN: 500, + AppliedLSN: 600, + Reason: "gap_beyond_retention", + }) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-rebuild-aggregate", + ReplicaID: "replica-2", + AckKind: SyncAckQuorum, + TargetLSN: 1000, + DurableLSN: 1000, + AppliedLSN: 1000, + }) + + if result.Projection.Mode.Name != ModeNeedsRebuild { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if result.Projection.Recovery.Phase != RecoveryNeedsRebuild { + t.Fatalf("recovery_phase=%s", result.Projection.Recovery.Phase) + } + if got := result.Projection.ReplicaSync["replica-1"].Action; got != SyncActionRebuild { + t.Fatalf("replica1_action=%s", got) + } + if got := result.Projection.ReplicaSync["replica-2"].Action; got != SyncActionKeepUp { + t.Fatalf("replica2_action=%s", got) + } +} + +func TestPhase14_SyncAckObserved_AssignmentChangeClearsReplicaSyncFacts(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-reset", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.57:9333", CtrlAddr: "10.0.0.57:9334", Version: 1}}, + }, + }) + core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-reset", + ReplicaID: "replica-1", + AckKind: SyncAckQuorum, + TargetLSN: 30, + DurableLSN: 30, + AppliedLSN: 30, + }) + + result := core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-reset", + Epoch: 2, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.58:9333", CtrlAddr: "10.0.0.58:9334", Version: 2}}, + }, + }) + + if result.Projection.Sync.AckKind != SyncAckUnknown { + t.Fatalf("sync_ack_kind=%s", result.Projection.Sync.AckKind) + } + if len(result.Projection.ReplicaSync) != 0 { + t.Fatalf("replica_sync_len=%d", len(result.Projection.ReplicaSync)) + } +} + +func TestPhase14_SyncAckObserved_PrimaryTailFactsDriveCatchUp(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-facts-catchup", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.59:9333", CtrlAddr: "10.0.0.59:9334", Version: 1}}, + }, + }) + core.ApplyEvent(RoleApplied{ID: "vol-sync-facts-catchup"}) + core.ApplyEvent(ShipperConfiguredObserved{ID: "vol-sync-facts-catchup"}) + core.ApplyEvent(ShipperConnectedObserved{ID: "vol-sync-facts-catchup"}) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-facts-catchup", + ReplicaID: "replica-1", + AckKind: SyncAckTimedOut, + TargetLSN: 1000, + PrimaryTailLSN: 700, + DurableLSN: 500, + AppliedLSN: 900, + Reason: "gap_within_retention", + }) + + assertCommandNames(t, result.Commands, []string{ + "start_catchup", + "publish_projection", + }) + if result.Projection.Sync.Action != SyncActionCatchUp { + t.Fatalf("sync_action=%s", result.Projection.Sync.Action) + } + if result.Projection.Recovery.Phase != RecoveryCatchingUp { + t.Fatalf("recovery_phase=%s", result.Projection.Recovery.Phase) + } +} + +func TestPhase14_SyncAckObserved_PrimaryTailFactsDriveRebuild(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-sync-facts-rebuild", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.60:9333", CtrlAddr: "10.0.0.60:9334", Version: 1}}, + }, + }) + + result := core.ApplyEvent(SyncAckObserved{ + ID: "vol-sync-facts-rebuild", + ReplicaID: "replica-1", + AckKind: SyncAckTimedOut, + TargetLSN: 1000, + PrimaryTailLSN: 700, + DurableLSN: 500, + AppliedLSN: 600, + Reason: "gap_beyond_retention", + }) + + assertCommandNames(t, result.Commands, []string{ + "invalidate_session", + "publish_projection", + }) + if result.Projection.Sync.Action != SyncActionRebuild { + t.Fatalf("sync_action=%s", result.Projection.Sync.Action) + } + if result.Projection.Mode.Name != ModeNeedsRebuild { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } +} + +func TestPhase14_SessionStarted_CatchUpUsesUnifiedPath(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-session-catchup", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.63:9333", CtrlAddr: "10.0.0.63:9334", Version: 1}}, + }, + }) + + result := core.ApplyEvent(SessionStarted{ + ID: "vol-session-catchup", + ReplicaID: "replica-1", + Kind: SessionCatchUp, + TargetLSN: 77, + }) + + assertCommandNames(t, result.Commands, []string{ + "start_catchup", + "publish_projection", + }) + if result.Projection.Recovery.Phase != RecoveryCatchingUp { + t.Fatalf("recovery_phase=%s", result.Projection.Recovery.Phase) + } + if result.Projection.Recovery.TargetLSN != 77 { + t.Fatalf("target_lsn=%d", result.Projection.Recovery.TargetLSN) + } +} + +func TestPhase14_SessionCompleted_RebuildClearsNeedsRebuild(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-session-rebuild", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionRebuild, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.64:9333", CtrlAddr: "10.0.0.64:9334", Version: 1}}, + }, + }) + core.ApplyEvent(NeedsRebuildObserved{ + ID: "vol-session-rebuild", + ReplicaID: "replica-1", + Reason: "gap_too_large", + }) + core.ApplyEvent(SessionStarted{ + ID: "vol-session-rebuild", + ReplicaID: "replica-1", + Kind: SessionRebuild, + TargetLSN: 120, + }) + + result := core.ApplyEvent(SessionCompleted{ + ID: "vol-session-rebuild", + ReplicaID: "replica-1", + Kind: SessionRebuild, + AchievedLSN: 120, + FlushedLSN: 120, + CheckpointLSN: 120, + }) + + if result.Projection.Recovery.Phase != RecoveryIdle { + t.Fatalf("recovery_phase=%s", result.Projection.Recovery.Phase) + } + if result.Projection.Mode.Name == ModeNeedsRebuild { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if result.Projection.Boundary.DurableLSN != 120 { + t.Fatalf("durable_lsn=%d", result.Projection.Boundary.DurableLSN) + } +} + +func TestPhase14_SessionFailed_CatchUpFallsBackToDegraded(t *testing.T) { + core := NewCoreEngine() + + core.ApplyEvent(AssignmentDelivered{ + ID: "vol-session-failed", + Epoch: 1, + Role: RolePrimary, + RecoveryTarget: SessionCatchUp, + Replicas: []ReplicaAssignment{ + {ReplicaID: "replica-1", Endpoint: Endpoint{DataAddr: "10.0.0.65:9333", CtrlAddr: "10.0.0.65:9334", Version: 1}}, + }, + }) + core.ApplyEvent(SessionStarted{ + ID: "vol-session-failed", + ReplicaID: "replica-1", + Kind: SessionCatchUp, + TargetLSN: 80, + }) + + result := core.ApplyEvent(SessionFailed{ + ID: "vol-session-failed", + ReplicaID: "replica-1", + Kind: SessionCatchUp, + Reason: "transport_lost", + }) + + assertCommandNames(t, result.Commands, []string{ + "invalidate_session", + "publish_projection", + }) + if result.Projection.Mode.Name != ModeDegraded { + t.Fatalf("mode=%s", result.Projection.Mode.Name) + } + if result.Projection.Boundary.LastBarrierReason != "transport_lost" { + t.Fatalf("last_barrier_reason=%q", result.Projection.Boundary.LastBarrierReason) + } +} diff --git a/sw-block/engine/replication/projection.go b/sw-block/engine/replication/projection.go index 2d57afede..a966f0a79 100644 --- a/sw-block/engine/replication/projection.go +++ b/sw-block/engine/replication/projection.go @@ -1,7 +1,8 @@ package replication // PublicationProjection is the bounded outward projection derived from one -// VolumeState. It is intentionally detached from runtime internals. +// VolumeState. It is intentionally detached from runtime internals and is +// primary-derived projection, not assignment truth. type PublicationProjection struct { VolumeID string Epoch uint64 @@ -10,6 +11,8 @@ type PublicationProjection struct { Mode ModeView Publication PublicationView Recovery RecoveryView + Sync SyncView + ReplicaSync ReplicaSyncView Readiness ReadinessView Boundary BoundaryView diff --git a/sw-block/engine/replication/state.go b/sw-block/engine/replication/state.go index aa80325b5..13fd4c724 100644 --- a/sw-block/engine/replication/state.go +++ b/sw-block/engine/replication/state.go @@ -86,6 +86,45 @@ type RecoveryView struct { Reason string } +// SyncAckKind captures the transport/control result of one sync request. The +// primary derives recovery action from this ack plus the attached facts. +type SyncAckKind string + +const ( + SyncAckUnknown SyncAckKind = "" + SyncAckQuorum SyncAckKind = "quorum" + SyncAckTimedOut SyncAckKind = "timed_out" + SyncAckTransportLost SyncAckKind = "transport_lost" + SyncAckEpochMismatch SyncAckKind = "epoch_mismatch" +) + +// SyncAction captures the primary-owned session decision derived from sync ack +// facts. It is not replica-owned protocol input. +type SyncAction string + +const ( + SyncActionKeepUp SyncAction = "keepup" + SyncActionCatchUp SyncAction = "catchup" + SyncActionRebuild SyncAction = "rebuild" +) + +// SyncView keeps the latest sync ack facts plus the primary-owned session +// decision derived from those facts. It remains distinct from durable boundary +// truth and recovery execution progress. +type SyncView struct { + AckKind SyncAckKind + Action SyncAction + TargetLSN uint64 + PrimaryTailLSN uint64 + DurableLSN uint64 + AppliedLSN uint64 + Reason string +} + +// ReplicaSyncView stores the latest sync ack facts for each replica the primary +// is currently tracking. +type ReplicaSyncView map[string]SyncView + type commandState struct { RoleEpoch uint64 Role VolumeRole @@ -94,20 +133,33 @@ type commandState struct { ShipperConfigReplicas []ReplicaAssignment RecoveryTaskEpoch uint64 RecoveryTaskTargets map[string]SessionKind - CatchUpTargets map[string]uint64 - RebuildTargets map[string]uint64 + SessionTargets map[string]sessionCommandTarget InvalidationIssued bool InvalidationReason string } -type catchUpObservation struct { +type sessionCommandTarget struct { + Kind SessionKind + TargetLSN uint64 +} + +type sessionObservation struct { + Kind SessionKind + Phase RecoveryPhase TargetLSN uint64 AchievedLSN uint64 + Reason string Completed bool } -// VolumeState is the minimal V2-core-owned state for one volume on the bounded -// current path. +// VolumeState is the minimal V2-core-owned state for one volume. +// +// Ownership split: +// - Assignment fields normalize master-owned identity truth. +// - Sync/ReplicaSync plus catch-up observations normalize primary-owned +// session truth. +// - Mode/Publication are derived projection only; they are never assigned by +// master or replica. type VolumeState struct { VolumeID string Epoch uint64 @@ -119,14 +171,14 @@ type VolumeState struct { Mode ModeView Publication PublicationView Recovery RecoveryView + Sync SyncView + ReplicaSync ReplicaSyncView degraded bool degradeReason string - needsRebuild bool - rebuildReason string recoveryTarget SessionKind commands commandState - catchUps map[string]catchUpObservation + sessions map[string]sessionObservation } func newVolumeState(volumeID string) *VolumeState { @@ -158,22 +210,22 @@ func (s *VolumeState) Snapshot() VolumeState { out.commands.RecoveryTaskTargets[replicaID] = kind } } - if s.commands.CatchUpTargets != nil { - out.commands.CatchUpTargets = make(map[string]uint64, len(s.commands.CatchUpTargets)) - for replicaID, target := range s.commands.CatchUpTargets { - out.commands.CatchUpTargets[replicaID] = target + if s.commands.SessionTargets != nil { + out.commands.SessionTargets = make(map[string]sessionCommandTarget, len(s.commands.SessionTargets)) + for replicaID, target := range s.commands.SessionTargets { + out.commands.SessionTargets[replicaID] = target } } - if s.commands.RebuildTargets != nil { - out.commands.RebuildTargets = make(map[string]uint64, len(s.commands.RebuildTargets)) - for replicaID, target := range s.commands.RebuildTargets { - out.commands.RebuildTargets[replicaID] = target + if s.sessions != nil { + out.sessions = make(map[string]sessionObservation, len(s.sessions)) + for replicaID, obs := range s.sessions { + out.sessions[replicaID] = obs } } - if s.catchUps != nil { - out.catchUps = make(map[string]catchUpObservation, len(s.catchUps)) - for replicaID, obs := range s.catchUps { - out.catchUps[replicaID] = obs + if s.ReplicaSync != nil { + out.ReplicaSync = make(ReplicaSyncView, len(s.ReplicaSync)) + for replicaID, syncView := range s.ReplicaSync { + out.ReplicaSync[replicaID] = syncView } } return out diff --git a/sw-block/protocol/engine.go b/sw-block/protocol/engine.go new file mode 100644 index 000000000..40cac7ba6 --- /dev/null +++ b/sw-block/protocol/engine.go @@ -0,0 +1,398 @@ +package protocol + +// Engine is the v2 protocol engine. Deterministic, side-effect free. +// Event in → state mutation + commands + projection out. +// +// Compared to engine/replication (19 events, 979 lines): +// - 7 event types (Assignment, Readiness, SyncAck, SessionProgress, +// SessionCompleted, SessionFailed, BarrierConfirmed) +// - Primary decides catchup vs rebuild from SyncAck facts, not from +// autonomous shipper/budget logic +// - Session lifecycle is explicit (idle → issued → running → completed/failed) +// - No per-replica bookkeeping maps — ReplicaView holds everything +type Engine struct { + volumes map[string]*VolumeState +} + +type Result struct { + Commands []Command + Mode ModeName + ModeReason string + Healthy bool +} + +func NewEngine() *Engine { + return &Engine{volumes: make(map[string]*VolumeState)} +} + +func (e *Engine) ApplyEvent(ev Event) Result { + st := e.mustVolume(ev.volumeID()) + + var cmds []Command + + switch v := ev.(type) { + case AssignmentDelivered: + cmds = e.applyAssignment(st, v) + case ReadinessObserved: + cmds = e.applyReadiness(st, v) + case SyncAckReceived: + cmds = e.applySyncAck(st, v) + case SessionProgress: + e.applyProgress(st, v) + case SessionCompleted: + cmds = e.applyCompleted(st, v) + case SessionFailed: + cmds = e.applyFailed(st, v) + case BarrierConfirmed: + e.applyBarrier(st, v) + } + + e.deriveMode(st) + + return Result{ + Commands: cmds, + Mode: st.Mode, + ModeReason: st.ModeReason, + Healthy: st.Healthy, + } +} + +func (e *Engine) Volume(id string) (VolumeState, bool) { + st, ok := e.volumes[id] + if !ok { + return VolumeState{}, false + } + return *st, true +} + +// --- Assignment --- + +func (e *Engine) applyAssignment(st *VolumeState, ev AssignmentDelivered) []Command { + epochChanged := st.Epoch != ev.Epoch + roleChanged := st.Role != ev.Role + + st.Epoch = ev.Epoch + st.Role = ev.Role + st.Replicas = ev.Replicas + st.Readiness.Assigned = true + + if epochChanged || roleChanged { + st.Readiness.RoleApplied = false + st.Readiness.ShipperConfigured = false + st.Readiness.ShipperConnected = false + st.Readiness.ReceiverReady = false + // Clear all replica sessions on epoch/role change. + st.ReplicaStates = make(map[string]*ReplicaView) + } + + // Ensure ReplicaView exists for each assigned replica. + if st.ReplicaStates == nil { + st.ReplicaStates = make(map[string]*ReplicaView) + } + for _, r := range ev.Replicas { + if _, ok := st.ReplicaStates[r.ReplicaID]; !ok { + st.ReplicaStates[r.ReplicaID] = &ReplicaView{ + ReplicaID: r.ReplicaID, + Endpoint: r.Endpoint, + Session: ReplicaSession{Kind: SessionNone, State: SessionStateIdle}, + } + } + } + + var cmds []Command + cmds = append(cmds, ApplyRoleCommand{ + VolumeID: st.VolumeID, + Epoch: st.Epoch, + Role: st.Role, + }) + + if st.Role == RolePrimary && len(st.Replicas) > 0 { + cmds = append(cmds, ConfigureShipperCommand{ + VolumeID: st.VolumeID, + Replicas: st.Replicas, + }) + } + if st.Role == RoleReplica { + cmds = append(cmds, StartReceiverCommand{VolumeID: st.VolumeID}) + } + + return cmds +} + +// --- Readiness --- + +func (e *Engine) applyReadiness(st *VolumeState, ev ReadinessObserved) []Command { + if ev.RoleApplied != nil { + st.Readiness.RoleApplied = *ev.RoleApplied + } + if ev.ReceiverReady != nil { + st.Readiness.ReceiverReady = *ev.ReceiverReady + } + if ev.ShipperConfigured != nil { + st.Readiness.ShipperConfigured = *ev.ShipperConfigured + } + if ev.ShipperConnected != nil { + st.Readiness.ShipperConnected = *ev.ShipperConnected + } + return nil +} + +// --- SyncAck: the core decision point --- +// +// This is where the primary decides per-replica recovery mode. +// One threshold: applied_lsn >= primary_wal_tail → WAL catch-up, else → rebuild. +// Matches Ceph's last_update >= log_tail decision. + +func (e *Engine) applySyncAck(st *VolumeState, ev SyncAckReceived) []Command { + rv := st.replicaView(ev.ReplicaID) + if rv == nil { + return nil + } + rv.LastSyncAck = ev.Ack + st.WALTail = ev.PrimaryWALTail + st.WALHead = ev.PrimaryWALHead + + // If replica reports durable progress, advance volume boundary. + if ev.Ack.DurableLSN > st.DurableLSN { + st.DurableLSN = ev.Ack.DurableLSN + } + + // Already in an active session? Don't re-decide, just update ack. + if rv.Session.State == SessionStateRunning || rv.Session.State == SessionStateIssued { + return nil + } + + // --- Primary decision --- + decision := e.decide(ev.Ack, ev.PrimaryWALTail, ev.PrimaryWALHead) + + switch decision { + case SessionKeepUp: + rv.Session = ReplicaSession{Kind: SessionKeepUp, State: SessionStateIdle} + return nil + + case SessionCatchUp: + targetLSN := ev.PrimaryWALHead + startLSN := ev.Ack.AppliedLSN + if startLSN == 0 { + startLSN = ev.Ack.DurableLSN + } + rv.Session = ReplicaSession{ + Kind: SessionCatchUp, + State: SessionStateIssued, + StartLSN: startLSN, + TargetLSN: targetLSN, + PinLSN: startLSN, + } + return []Command{IssueCatchUpCommand{ + VolumeID: st.VolumeID, + ReplicaID: ev.ReplicaID, + StartLSN: startLSN, + TargetLSN: targetLSN, + PinLSN: startLSN, + }} + + case SessionRebuild: + rv.Session = ReplicaSession{ + Kind: SessionRebuild, + State: SessionStateIssued, + TargetLSN: ev.PrimaryWALHead, + Reason: "applied_lsn < primary_wal_tail", + } + return []Command{IssueRebuildCommand{ + VolumeID: st.VolumeID, + ReplicaID: ev.ReplicaID, + TargetLSN: ev.PrimaryWALHead, + }} + } + + return nil +} + +// decide is the one-threshold decision. +// Matches Ceph: last_update >= log_tail → log-recovery, else → backfill. +func (e *Engine) decide(ack SyncAck, primaryWALTail uint64, primaryWALHead uint64) SessionKind { + replicaPos := ack.AppliedLSN + if replicaPos == 0 { + replicaPos = ack.DurableLSN + } + + // Replica is fully caught up. + if replicaPos >= primaryWALHead && replicaPos > 0 { + return SessionKeepUp + } + + // Replica is behind but within retained WAL → catch-up. + if replicaPos >= primaryWALTail && replicaPos > 0 { + return SessionCatchUp + } + + // Fresh replica (pos=0) with retained WAL from the beginning. + if replicaPos == 0 && primaryWALTail <= 1 { + return SessionCatchUp + } + + // Gap exceeds retained WAL → rebuild. + return SessionRebuild +} + +// --- Session lifecycle --- + +func (e *Engine) applyProgress(st *VolumeState, ev SessionProgress) { + rv := st.replicaView(ev.ReplicaID) + if rv == nil { + return + } + if rv.Session.State == SessionStateIssued { + rv.Session.State = SessionStateRunning + } + rv.Session.Progress = ev.Progress +} + +func (e *Engine) applyCompleted(st *VolumeState, ev SessionCompleted) []Command { + rv := st.replicaView(ev.ReplicaID) + if rv == nil { + return nil + } + rv.Session = ReplicaSession{Kind: SessionKeepUp, State: SessionStateIdle} + if ev.DurableLSN > st.DurableLSN { + st.DurableLSN = ev.DurableLSN + } + return nil +} + +func (e *Engine) applyFailed(st *VolumeState, ev SessionFailed) []Command { + rv := st.replicaView(ev.ReplicaID) + if rv == nil { + return nil + } + rv.Session.State = SessionStateFailed + rv.Session.Reason = ev.Reason + // Don't auto-escalate. Wait for next SyncAck to re-decide. + // This is the key difference from v1: failure doesn't auto-become NeedsRebuild. + return nil +} + +// --- Barrier --- + +func (e *Engine) applyBarrier(st *VolumeState, ev BarrierConfirmed) { + if ev.DurableLSN > st.DurableLSN { + st.DurableLSN = ev.DurableLSN + } +} + +// --- Mode derivation (projection) --- +// +// This is the outward view. Derived from all replica states + boundaries. +// Matches the principle: projection is primary-derived, not assigned. + +func (e *Engine) deriveMode(st *VolumeState) { + st.Healthy = false + st.ModeReason = "" + + switch { + case st.anyReplicaInState(SessionRebuild): + st.Mode = ModeNeedsRebuild + st.ModeReason = "replica_needs_rebuild" + + case st.anyReplicaSessionFailed(): + st.Mode = ModeDegraded + st.ModeReason = st.failedReason() + + case st.anyReplicaInState(SessionCatchUp): + st.Mode = ModeBootstrapPending + st.ModeReason = "recovery_in_progress" + + case st.Role == RoleReplica && st.Readiness.ReceiverReady: + st.Mode = ModeReplicaReady + st.ModeReason = "replica_not_primary" + + case st.Role == RolePrimary && len(st.Replicas) == 0: + st.Mode = ModeAllocatedOnly + st.ModeReason = "allocated_only" + + case st.Role == RolePrimary && st.primaryEligible(): + st.Mode = ModePublishHealthy + st.Healthy = true + + case st.Readiness.Assigned: + st.Mode = ModeBootstrapPending + st.ModeReason = st.bootstrapReason() + + default: + st.Mode = ModeAllocatedOnly + st.ModeReason = "allocated_only" + } +} + +func (st *VolumeState) primaryEligible() bool { + return st.Readiness.RoleApplied && + st.Readiness.ShipperConfigured && + st.Readiness.ShipperConnected && + st.DurableLSN > 0 +} + +func (st *VolumeState) bootstrapReason() string { + switch { + case !st.Readiness.RoleApplied: + return "awaiting_role_apply" + case st.Role == RoleReplica && !st.Readiness.ReceiverReady: + return "awaiting_receiver_ready" + case st.Role == RolePrimary && !st.Readiness.ShipperConfigured: + return "awaiting_shipper_configured" + case st.Role == RolePrimary && !st.Readiness.ShipperConnected: + return "awaiting_shipper_connected" + case st.Role == RolePrimary && st.DurableLSN == 0: + return "awaiting_barrier_durability" + default: + return "bootstrap_pending" + } +} + +// --- Helpers --- + +func (e *Engine) mustVolume(id string) *VolumeState { + if st, ok := e.volumes[id]; ok { + return st + } + st := &VolumeState{ + VolumeID: id, + ReplicaStates: make(map[string]*ReplicaView), + } + e.volumes[id] = st + return st +} + +func (st *VolumeState) replicaView(replicaID string) *ReplicaView { + if replicaID == "" { + return nil + } + return st.ReplicaStates[replicaID] +} + +func (st *VolumeState) anyReplicaInState(kind SessionKind) bool { + for _, rv := range st.ReplicaStates { + if rv.Session.Kind == kind && + (rv.Session.State == SessionStateIssued || rv.Session.State == SessionStateRunning) { + return true + } + } + return false +} + +func (st *VolumeState) anyReplicaSessionFailed() bool { + for _, rv := range st.ReplicaStates { + if rv.Session.State == SessionStateFailed { + return true + } + } + return false +} + +func (st *VolumeState) failedReason() string { + for _, rv := range st.ReplicaStates { + if rv.Session.State == SessionStateFailed && rv.Session.Reason != "" { + return rv.Session.Reason + } + } + return "session_failed" +} diff --git a/sw-block/protocol/engine_test.go b/sw-block/protocol/engine_test.go new file mode 100644 index 000000000..ef2cccb0e --- /dev/null +++ b/sw-block/protocol/engine_test.go @@ -0,0 +1,345 @@ +package protocol + +import "testing" + +// --- Assignment --- + +func TestAssignment_SetsIdentity(t *testing.T) { + e := NewEngine() + r := e.ApplyEvent(AssignmentDelivered{ + VolumeID: "vol-1", + Epoch: 1, + Role: RolePrimary, + Replicas: []ReplicaAssignment{ + {ReplicaID: "vs-2", Endpoint: Endpoint{DataAddr: "10.0.0.2:4260", CtrlAddr: "10.0.0.2:4261"}}, + }, + }) + + st, ok := e.Volume("vol-1") + if !ok { + t.Fatal("volume not found") + } + if st.Epoch != 1 || st.Role != RolePrimary { + t.Fatalf("epoch=%d role=%s", st.Epoch, st.Role) + } + if len(r.Commands) < 2 { + t.Fatalf("commands=%d, want at least ApplyRole + ConfigureShipper", len(r.Commands)) + } + if r.Mode != ModeBootstrapPending { + t.Fatalf("mode=%s", r.Mode) + } +} + +// --- Readiness chain --- + +func TestReadiness_BootstrapChain(t *testing.T) { + e := NewEngine() + e.ApplyEvent(AssignmentDelivered{ + VolumeID: "vol-1", Epoch: 1, Role: RolePrimary, + Replicas: []ReplicaAssignment{{ReplicaID: "vs-2", Endpoint: Endpoint{DataAddr: "a", CtrlAddr: "b"}}}, + }) + + boolTrue := true + + r := e.ApplyEvent(ReadinessObserved{VolumeID: "vol-1", RoleApplied: &boolTrue}) + if r.ModeReason != "awaiting_shipper_configured" { + t.Fatalf("reason=%q", r.ModeReason) + } + + r = e.ApplyEvent(ReadinessObserved{VolumeID: "vol-1", ShipperConfigured: &boolTrue}) + if r.ModeReason != "awaiting_shipper_connected" { + t.Fatalf("reason=%q", r.ModeReason) + } + + r = e.ApplyEvent(ReadinessObserved{VolumeID: "vol-1", ShipperConnected: &boolTrue}) + if r.ModeReason != "awaiting_barrier_durability" { + t.Fatalf("reason=%q", r.ModeReason) + } + + r = e.ApplyEvent(BarrierConfirmed{VolumeID: "vol-1", DurableLSN: 1}) + if r.Mode != ModePublishHealthy { + t.Fatalf("mode=%s", r.Mode) + } + if !r.Healthy { + t.Fatal("expected healthy") + } +} + +// --- SyncAck: primary decision --- + +func TestSyncAck_ReplicaCaughtUp_KeepUp(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + r := e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", + ReplicaID: "vs-2", + Ack: SyncAck{DurableLSN: 100, AppliedLSN: 100}, + PrimaryWALTail: 50, + PrimaryWALHead: 100, + }) + + // Replica is caught up → no catch-up/rebuild command. + for _, cmd := range r.Commands { + switch cmd.(type) { + case IssueCatchUpCommand, IssueRebuildCommand: + t.Fatalf("unexpected command: %T", cmd) + } + } + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + if rv.Session.Kind != SessionKeepUp { + t.Fatalf("session=%s, want keepup", rv.Session.Kind) + } +} + +func TestSyncAck_ReplicaBehindWithinWAL_CatchUp(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + r := e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", + ReplicaID: "vs-2", + Ack: SyncAck{DurableLSN: 30, AppliedLSN: 50}, + PrimaryWALTail: 20, // replica at 50, tail at 20 → within WAL + PrimaryWALHead: 100, + }) + + var catchUp *IssueCatchUpCommand + for _, cmd := range r.Commands { + if c, ok := cmd.(IssueCatchUpCommand); ok { + catchUp = &c + } + } + if catchUp == nil { + t.Fatal("expected IssueCatchUpCommand") + } + if catchUp.StartLSN != 50 || catchUp.TargetLSN != 100 { + t.Fatalf("catchup start=%d target=%d", catchUp.StartLSN, catchUp.TargetLSN) + } + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + if rv.Session.Kind != SessionCatchUp { + t.Fatalf("session=%s, want catchup", rv.Session.Kind) + } +} + +func TestSyncAck_ReplicaBeyondWAL_Rebuild(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + r := e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", + ReplicaID: "vs-2", + Ack: SyncAck{DurableLSN: 5, AppliedLSN: 10}, + PrimaryWALTail: 500, // replica at 10, tail at 500 → gap beyond WAL + PrimaryWALHead: 1000, + }) + + var rebuild *IssueRebuildCommand + for _, cmd := range r.Commands { + if c, ok := cmd.(IssueRebuildCommand); ok { + rebuild = &c + } + } + if rebuild == nil { + t.Fatal("expected IssueRebuildCommand") + } + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + if rv.Session.Kind != SessionRebuild { + t.Fatalf("session=%s, want rebuild", rv.Session.Kind) + } + if r.Mode != ModeNeedsRebuild { + t.Fatalf("mode=%s, want needs_rebuild", r.Mode) + } +} + +func TestSyncAck_FreshReplica_WALRetained_CatchUp(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + r := e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", + ReplicaID: "vs-2", + Ack: SyncAck{DurableLSN: 0, AppliedLSN: 0}, + PrimaryWALTail: 1, // WAL retained from beginning + PrimaryWALHead: 50, + }) + + var catchUp *IssueCatchUpCommand + for _, cmd := range r.Commands { + if c, ok := cmd.(IssueCatchUpCommand); ok { + catchUp = &c + } + } + if catchUp == nil { + t.Fatal("fresh replica with WAL retained should get catch-up, not rebuild") + } +} + +func TestSyncAck_FreshReplica_WALNotRetained_Rebuild(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + r := e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", + ReplicaID: "vs-2", + Ack: SyncAck{DurableLSN: 0, AppliedLSN: 0}, + PrimaryWALTail: 500, // WAL starts at 500, replica at 0 + PrimaryWALHead: 1000, + }) + + var rebuild *IssueRebuildCommand + for _, cmd := range r.Commands { + if c, ok := cmd.(IssueRebuildCommand); ok { + rebuild = &c + } + } + if rebuild == nil { + t.Fatal("fresh replica with WAL not retained should get rebuild") + } +} + +// --- Session lifecycle --- + +func TestSession_ProgressDoesNotReDecide(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + // Issue catch-up. + e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", ReplicaID: "vs-2", + Ack: SyncAck{AppliedLSN: 50}, PrimaryWALTail: 20, PrimaryWALHead: 100, + }) + + // Progress during catch-up. + e.ApplyEvent(SessionProgress{VolumeID: "vol-1", ReplicaID: "vs-2", Progress: 75}) + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + if rv.Session.Kind != SessionCatchUp { + t.Fatalf("session=%s, should stay catchup during progress", rv.Session.Kind) + } + if rv.Session.State != SessionStateRunning { + t.Fatalf("state=%s, want running", rv.Session.State) + } + if rv.Session.Progress != 75 { + t.Fatalf("progress=%d", rv.Session.Progress) + } +} + +func TestSession_CompletedReturnsToKeepUp(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", ReplicaID: "vs-2", + Ack: SyncAck{AppliedLSN: 50}, PrimaryWALTail: 20, PrimaryWALHead: 100, + }) + + e.ApplyEvent(SessionCompleted{VolumeID: "vol-1", ReplicaID: "vs-2", DurableLSN: 100}) + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + if rv.Session.Kind != SessionKeepUp { + t.Fatalf("session=%s, want keepup after completion", rv.Session.Kind) + } +} + +func TestSession_FailedDoesNotAutoEscalate(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", ReplicaID: "vs-2", + Ack: SyncAck{AppliedLSN: 50}, PrimaryWALTail: 20, PrimaryWALHead: 100, + }) + + r := e.ApplyEvent(SessionFailed{VolumeID: "vol-1", ReplicaID: "vs-2", Reason: "transport_lost"}) + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + if rv.Session.Kind != SessionCatchUp { + t.Fatalf("session kind=%s, should preserve kind on failure", rv.Session.Kind) + } + if rv.Session.State != SessionStateFailed { + t.Fatalf("session state=%s, want failed", rv.Session.State) + } + // Key: failure doesn't auto-become NeedsRebuild. + if r.Mode == ModeNeedsRebuild { + t.Fatal("failed session must NOT auto-escalate to needs_rebuild") + } + if r.Mode != ModeDegraded { + t.Fatalf("mode=%s, want degraded after session failure", r.Mode) + } +} + +func TestSession_FailedThenSyncAck_ReDecides(t *testing.T) { + e := NewEngine() + setupPrimary(e, "vol-1", 1) + + e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", ReplicaID: "vs-2", + Ack: SyncAck{AppliedLSN: 50}, PrimaryWALTail: 20, PrimaryWALHead: 100, + }) + + e.ApplyEvent(SessionFailed{VolumeID: "vol-1", ReplicaID: "vs-2", Reason: "transport_lost"}) + + // Next sync ack triggers re-decision. Replica made progress to 80. + r := e.ApplyEvent(SyncAckReceived{ + VolumeID: "vol-1", ReplicaID: "vs-2", + Ack: SyncAck{AppliedLSN: 80}, PrimaryWALTail: 20, PrimaryWALHead: 120, + }) + + st, _ := e.Volume("vol-1") + rv := st.ReplicaStates["vs-2"] + // Should re-decide based on new facts, not stay in old failed state. + if rv.Session.State == SessionStateFailed { + t.Fatal("sync ack after failure should re-decide, not stay failed") + } + _ = r +} + +// --- Mode derivation --- + +func TestMode_NoReplicas_AllocatedOnly(t *testing.T) { + e := NewEngine() + r := e.ApplyEvent(AssignmentDelivered{ + VolumeID: "vol-1", Epoch: 1, Role: RolePrimary, + }) + if r.Mode != ModeAllocatedOnly { + t.Fatalf("mode=%s", r.Mode) + } +} + +func TestMode_ReplicaReady(t *testing.T) { + e := NewEngine() + e.ApplyEvent(AssignmentDelivered{ + VolumeID: "vol-1", Epoch: 1, Role: RoleReplica, + }) + boolTrue := true + e.ApplyEvent(ReadinessObserved{VolumeID: "vol-1", RoleApplied: &boolTrue}) + r := e.ApplyEvent(ReadinessObserved{VolumeID: "vol-1", ReceiverReady: &boolTrue}) + if r.Mode != ModeReplicaReady { + t.Fatalf("mode=%s", r.Mode) + } +} + +// --- Helpers --- + +func setupPrimary(e *Engine, volumeID string, epoch uint64) { + e.ApplyEvent(AssignmentDelivered{ + VolumeID: volumeID, Epoch: epoch, Role: RolePrimary, + Replicas: []ReplicaAssignment{ + {ReplicaID: "vs-2", Endpoint: Endpoint{DataAddr: "10.0.0.2:4260", CtrlAddr: "10.0.0.2:4261"}}, + }, + }) + boolTrue := true + e.ApplyEvent(ReadinessObserved{VolumeID: volumeID, RoleApplied: &boolTrue}) + e.ApplyEvent(ReadinessObserved{VolumeID: volumeID, ShipperConfigured: &boolTrue}) + e.ApplyEvent(ReadinessObserved{VolumeID: volumeID, ShipperConnected: &boolTrue}) +} diff --git a/sw-block/protocol/types.go b/sw-block/protocol/types.go new file mode 100644 index 000000000..e5e64ee89 --- /dev/null +++ b/sw-block/protocol/types.go @@ -0,0 +1,249 @@ +// Package protocol implements the v2 sync/recovery protocol engine. +// +// Design principles: +// - Deterministic, side-effect free: event in → state + commands + projection out +// - Primary decides everything: catchup vs rebuild based on replica-reported facts +// - Replica only reports facts and executes contracts +// - One threshold: applied_lsn >= wal_tail → WAL catch-up, otherwise → rebuild +// +// Three authority layers: +// - Assignment: master → identity (who is primary, replica set, epoch) +// - Session: primary → per-replica recovery contract (keepup/catchup/rebuild) +// - Projection: primary → derived volume mode/health +// +// Reference: Ceph peering (log-recovery vs backfill on last_update >= log_tail), +// Mayastor (control-plane-driven rebuild, nexus never self-escalates), +// Longhorn (controller-driven PrepareRebuild, replica is passive). +package protocol + +// --- Roles and Modes --- + +type Role string + +const ( + RolePrimary Role = "primary" + RoleReplica Role = "replica" + RoleNone Role = "" +) + +type ModeName string + +const ( + ModeAllocatedOnly ModeName = "allocated_only" + ModeBootstrapPending ModeName = "bootstrap_pending" + ModePublishHealthy ModeName = "publish_healthy" + ModeReplicaReady ModeName = "replica_ready" + ModeDegraded ModeName = "degraded" + ModeNeedsRebuild ModeName = "needs_rebuild" +) + +// --- Session --- + +type SessionKind string + +const ( + SessionNone SessionKind = "" + SessionKeepUp SessionKind = "keepup" + SessionCatchUp SessionKind = "catchup" + SessionRebuild SessionKind = "rebuild" +) + +type SessionState string + +const ( + SessionStateIdle SessionState = "idle" + SessionStateIssued SessionState = "issued" + SessionStateRunning SessionState = "running" + SessionStateCompleted SessionState = "completed" + SessionStateFailed SessionState = "failed" +) + +// --- Sync Ack --- + +// SyncAck is what the replica returns in response to a sync request. +// The replica only reports facts. The primary decides what to do. +type SyncAck struct { + DurableLSN uint64 // barrier-confirmed durable boundary + AppliedLSN uint64 // last WAL entry applied locally + ReceivedLSN uint64 // last WAL entry received (may not be applied yet) + WALTail uint64 // oldest retained WAL entry on replica + Recoverable bool // replica's self-assessment: can it still catch up? + Reason string // if not recoverable, why +} + +// --- Replica State (primary's view) --- + +type ReplicaView struct { + ReplicaID string + Endpoint Endpoint + Session ReplicaSession + LastSyncAck SyncAck +} + +type ReplicaSession struct { + Kind SessionKind + State SessionState + StartLSN uint64 + TargetLSN uint64 + PinLSN uint64 + Progress uint64 // last reported progress during session + Reason string // failure reason if failed +} + +type Endpoint struct { + DataAddr string + CtrlAddr string +} + +// --- Volume State --- + +type VolumeState struct { + VolumeID string + Epoch uint64 + Role Role + + // Assignment-level. + Replicas []ReplicaAssignment + + // Readiness (host-observed). + Readiness Readiness + + // Per-replica state (primary-owned). + ReplicaStates map[string]*ReplicaView + + // Boundaries. + DurableLSN uint64 // highest barrier-confirmed LSN + WALTail uint64 // primary's oldest retained WAL entry + WALHead uint64 // primary's newest WAL entry + + // Derived. + Mode ModeName + ModeReason string + Healthy bool +} + +type ReplicaAssignment struct { + ReplicaID string + Endpoint Endpoint +} + +type Readiness struct { + Assigned bool + RoleApplied bool + ReceiverReady bool + ShipperConfigured bool + ShipperConnected bool +} + +// --- Commands (emitted by engine, executed by host) --- + +type Command interface{ commandMarker() } + +type ApplyRoleCommand struct { + VolumeID string + Epoch uint64 + Role Role +} + +type ConfigureShipperCommand struct { + VolumeID string + Replicas []ReplicaAssignment +} + +type StartReceiverCommand struct { + VolumeID string +} + +type IssueCatchUpCommand struct { + VolumeID string + ReplicaID string + StartLSN uint64 + TargetLSN uint64 + PinLSN uint64 +} + +type IssueRebuildCommand struct { + VolumeID string + ReplicaID string + TargetLSN uint64 +} + +type PublishProjectionCommand struct { + VolumeID string + Mode ModeName + Reason string + Healthy bool +} + +func (ApplyRoleCommand) commandMarker() {} +func (ConfigureShipperCommand) commandMarker() {} +func (StartReceiverCommand) commandMarker() {} +func (IssueCatchUpCommand) commandMarker() {} +func (IssueRebuildCommand) commandMarker() {} +func (PublishProjectionCommand) commandMarker() {} + +// --- Events (fed into engine by host) --- + +type Event interface{ volumeID() string } + +// AssignmentDelivered: master assigned identity. +type AssignmentDelivered struct { + VolumeID string + Epoch uint64 + Role Role + Replicas []ReplicaAssignment +} + +// ReadinessObserved: host reports a readiness fact. +type ReadinessObserved struct { + VolumeID string + RoleApplied *bool + ReceiverReady *bool + ShipperConfigured *bool + ShipperConnected *bool +} + +// SyncAckReceived: replica responded to a sync request. +// Primary uses this to decide keepup/catchup/rebuild. +type SyncAckReceived struct { + VolumeID string + ReplicaID string + Ack SyncAck + PrimaryWALTail uint64 // primary's WAL tail at the time of sync + PrimaryWALHead uint64 // primary's WAL head at the time of sync +} + +// SessionProgress: replica reports progress during catchup/rebuild. +type SessionProgress struct { + VolumeID string + ReplicaID string + Progress uint64 +} + +// SessionCompleted: replica finished its recovery session. +type SessionCompleted struct { + VolumeID string + ReplicaID string + DurableLSN uint64 +} + +// SessionFailed: replica's recovery session failed. +type SessionFailed struct { + VolumeID string + ReplicaID string + Reason string +} + +// BarrierConfirmed: durability fence succeeded (from SyncCache path). +type BarrierConfirmed struct { + VolumeID string + DurableLSN uint64 +} + +func (e AssignmentDelivered) volumeID() string { return e.VolumeID } +func (e ReadinessObserved) volumeID() string { return e.VolumeID } +func (e SyncAckReceived) volumeID() string { return e.VolumeID } +func (e SessionProgress) volumeID() string { return e.VolumeID } +func (e SessionCompleted) volumeID() string { return e.VolumeID } +func (e SessionFailed) volumeID() string { return e.VolumeID } +func (e BarrierConfirmed) volumeID() string { return e.VolumeID } diff --git a/weed/server/block_recovery.go b/weed/server/block_recovery.go index edd7795e0..634e786fd 100644 --- a/weed/server/block_recovery.go +++ b/weed/server/block_recovery.go @@ -316,6 +316,10 @@ func (rm *RecoveryManager) runCatchUp(ctx context.Context, replicaID string, ass } switch plan.Outcome { case engine.OutcomeCatchUp: + if plan.Proof == nil { + glog.Warningf("recovery: missing recoverability proof for catch-up plan %s", replicaID) + return + } if bs.v2Core == nil { rm.executeLegacyCatchUp(ctx, rctx.volPath, replicaID, rctx.driver, plan, rctx.executor) return @@ -331,17 +335,40 @@ func (rm *RecoveryManager) runCatchUp(ctx context.Context, replicaID string, ass if rm.OnPendingExecution != nil { rm.OnPendingExecution(rctx.volPath, rm.coord.Peek(replicaID)) } + bs.applyCoreEvent(engine.SyncAckObserved{ + ID: rctx.volPath, + ReplicaID: replicaID, + AckKind: engine.SyncAckTimedOut, + TargetLSN: plan.CatchUpTarget, + PrimaryTailLSN: plan.Proof.TailLSN, + DurableLSN: rctx.replicaFlushedLSN, + AppliedLSN: plan.Proof.ReplicaFlushedLSN, + Reason: plan.Proof.Reason, + }) bs.applyCoreEvent(engine.CatchUpPlanned{ID: rctx.volPath, ReplicaID: replicaID, TargetLSN: plan.CatchUpTarget}) if rm.coord.Has(replicaID) { rm.coord.Cancel(replicaID, "start_catchup_not_emitted") return } case engine.OutcomeNeedsRebuild: + if plan.Proof == nil { + glog.Warningf("recovery: missing recoverability proof for rebuild plan %s", replicaID) + return + } reason := "needs_rebuild" if plan.Proof != nil && plan.Proof.Reason != "" { reason = plan.Proof.Reason } - bs.applyCoreEvent(engine.NeedsRebuildObserved{ID: rctx.volPath, ReplicaID: replicaID, Reason: reason}) + bs.applyCoreEvent(engine.SyncAckObserved{ + ID: rctx.volPath, + ReplicaID: replicaID, + AckKind: engine.SyncAckTimedOut, + TargetLSN: plan.Proof.CommittedLSN, + PrimaryTailLSN: plan.Proof.TailLSN, + DurableLSN: plan.Proof.ReplicaFlushedLSN, + AppliedLSN: plan.Proof.ReplicaFlushedLSN, + Reason: reason, + }) return } @@ -434,9 +461,10 @@ func (rm *RecoveryManager) OnCatchUpFailed(volumeID, replicaID, reason string) { return } glog.V(0).Infof("recovery: catch-up failed for %s via %s (%s)", volumeID, replicaID, reason) - rm.bs.applyCoreEvent(engine.NeedsRebuildObserved{ + rm.bs.applyCoreEvent(engine.SyncAckObserved{ ID: volumeID, ReplicaID: replicaID, + AckKind: engine.SyncAckTransportLost, Reason: reason, }) } diff --git a/weed/server/block_recovery_test.go b/weed/server/block_recovery_test.go index 59fb2227b..ca84c73b5 100644 --- a/weed/server/block_recovery_test.go +++ b/weed/server/block_recovery_test.go @@ -243,6 +243,16 @@ func TestP16B_RunCatchUp_UpdatesCoreProjectionFromLiveRecovery(t *testing.T) { if proj.Recovery.Phase != engine.RecoveryIdle { t.Fatalf("recovery_phase=%s", proj.Recovery.Phase) } + replicaSync, ok := proj.ReplicaSync[replicaID] + if !ok { + t.Fatalf("missing replica sync for %s", replicaID) + } + if replicaSync.AckKind != engine.SyncAckTimedOut { + t.Fatalf("sync_ack_kind=%s", replicaSync.AckKind) + } + if replicaSync.Action != engine.SyncActionCatchUp { + t.Fatalf("sync_action=%s", replicaSync.Action) + } if got := bs.ExecutedCoreCommands(volPath); len(got) == 0 || got[len(got)-1] != "start_catchup" { t.Fatalf("expected start_catchup execution, got %v", got) } @@ -296,11 +306,70 @@ func TestP16B_RunCatchUp_EscalatesNeedsRebuildIntoCoreProjection(t *testing.T) { if proj.Publication.Reason == "" { t.Fatal("expected needs_rebuild reason") } + replicaSync, ok := proj.ReplicaSync[replicaID] + if !ok { + t.Fatalf("missing replica sync for %s", replicaID) + } + if replicaSync.AckKind != engine.SyncAckTimedOut { + t.Fatalf("sync_ack_kind=%s", replicaSync.AckKind) + } + if replicaSync.Action != engine.SyncActionRebuild { + t.Fatalf("sync_action=%s", replicaSync.Action) + } if got := bs.ExecutedCoreCommands(volPath); len(got) != 3 { t.Fatalf("needs_rebuild path should not execute start_catchup, got %v", got) } } +func TestP16B_OnCatchUpFailed_UsesSyncAckForNeedsRebuild(t *testing.T) { + bs, volPath := createTestBlockServiceWithVolCoreNoRecovery(t) + + bs.ProcessAssignments([]blockvol.BlockVolumeAssignment{ + { + Path: volPath, + Epoch: 1, + Role: uint32(blockvol.RolePrimary), + ReplicaServerID: "vs2", + ReplicaDataAddr: "10.0.0.2:9333", + ReplicaCtrlAddr: "10.0.0.2:9334", + }, + }) + + replicaID := volPath + "/vs2" + sender := bs.v2Orchestrator.Registry.Sender(replicaID) + if sender == nil || !sender.HasActiveSession() { + t.Fatal("expected active sender session before catch-up failure") + } + + rm := NewRecoveryManager(bs) + bs.v2Recovery = rm + rm.OnCatchUpFailed(volPath, replicaID, "recoverability_lost") + + proj, ok := bs.CoreProjection(volPath) + if !ok { + t.Fatal("expected cached core projection after catch-up failure") + } + if proj.Mode.Name != engine.ModeNeedsRebuild { + t.Fatalf("mode=%s", proj.Mode.Name) + } + if proj.Recovery.Phase != engine.RecoveryNeedsRebuild { + t.Fatalf("recovery_phase=%s", proj.Recovery.Phase) + } + replicaSync, ok := proj.ReplicaSync[replicaID] + if !ok { + t.Fatalf("missing replica sync for %s", replicaID) + } + if replicaSync.AckKind != engine.SyncAckTransportLost { + t.Fatalf("sync_ack_kind=%s", replicaSync.AckKind) + } + if replicaSync.Reason != "recoverability_lost" { + t.Fatalf("sync_reason=%q", replicaSync.Reason) + } + if sender.HasActiveSession() { + t.Fatal("target replica session should be invalidated by sync-negotiated rebuild transition") + } +} + func TestP16B_RunRebuild_UsesCoreStartRebuildCommandOnLivePath(t *testing.T) { bs, volPath := createTestBlockServiceWithVolCoreNoRecovery(t) diff --git a/weed/server/volume_server_block.go b/weed/server/volume_server_block.go index 5762d8d2f..142b04395 100644 --- a/weed/server/volume_server_block.go +++ b/weed/server/volume_server_block.go @@ -269,6 +269,12 @@ func (bs *BlockService) handleBarrierAccepted(path string, flushedLSN uint64, ch return } } + bs.applyCoreEvent(engine.SyncAckObserved{ + ID: path, + AckKind: engine.SyncAckQuorum, + TargetLSN: flushedLSN, + DurableLSN: flushedLSN, + }) bs.applyCoreEvent(engine.BarrierAccepted{ID: path, FlushedLSN: flushedLSN}) } @@ -286,6 +292,11 @@ func (bs *BlockService) handleBarrierRejected(path string, reason string, ch cha if !ok || proj.Role != engine.RolePrimary { return } + bs.applyCoreEvent(engine.SyncAckObserved{ + ID: path, + AckKind: engine.SyncAckTimedOut, + Reason: reason, + }) bs.applyCoreEvent(engine.BarrierRejected{ID: path, Reason: reason}) } diff --git a/weed/server/volume_server_block_test.go b/weed/server/volume_server_block_test.go index 0d698c3c5..1215df5a6 100644 --- a/weed/server/volume_server_block_test.go +++ b/weed/server/volume_server_block_test.go @@ -1756,6 +1756,57 @@ func TestBlockService_BarrierRejectedCallback_UpdatesCoreProjection(t *testing.T if after.Boundary.LastBarrierReason != "barrier_timeout" { t.Fatalf("last_barrier_reason=%q, want %q", after.Boundary.LastBarrierReason, "barrier_timeout") } + if after.Sync.AckKind != engine.SyncAckTimedOut { + t.Fatalf("sync_ack_kind=%s, want %s", after.Sync.AckKind, engine.SyncAckTimedOut) + } + if after.Sync.Reason != "barrier_timeout" { + t.Fatalf("sync_reason=%q, want %q", after.Sync.Reason, "barrier_timeout") + } +} + +func TestBlockService_BarrierAcceptedCallback_UpdatesCoreSyncProjection(t *testing.T) { + bs := newTestBlockServiceDirect(t) + path := createTestVolDirect(t, bs, "vol-barrier-accepted-callback") + ch := make(chan bool, 1) + bs.WireStateChangeNotify(ch) + + errs := bs.ApplyAssignments([]blockvol.BlockVolumeAssignment{ + { + Path: path, + Epoch: 1, + Role: blockvol.RoleToWire(blockvol.RolePrimary), + LeaseTtlMs: 30000, + ReplicaServerID: "vs-2", + ReplicaDataAddr: "10.0.0.2:4260", + ReplicaCtrlAddr: "10.0.0.2:4261", + }, + }) + if len(errs) != 1 || errs[0] != nil { + t.Fatalf("apply assignment errs=%v", errs) + } + + bs.applyCoreEvent(engine.ShipperConnectedObserved{ID: path}) + bs.handleBarrierAccepted(path, 12, ch) + + select { + case <-ch: + default: + t.Fatal("expected immediate heartbeat notification") + } + + after, ok := bs.CoreProjection(path) + if !ok { + t.Fatal("expected core projection after barrier acceptance") + } + if after.Sync.AckKind != engine.SyncAckQuorum { + t.Fatalf("sync_ack_kind=%s, want %s", after.Sync.AckKind, engine.SyncAckQuorum) + } + if after.Sync.Action != engine.SyncActionKeepUp { + t.Fatalf("sync_action=%s, want %s", after.Sync.Action, engine.SyncActionKeepUp) + } + if after.Sync.DurableLSN != 12 { + t.Fatalf("sync_durable_lsn=%d, want 12", after.Sync.DurableLSN) + } } func TestBlockService_CreateBlockVol_WiresStateChangeCallbackForNewVolumes(t *testing.T) { diff --git a/weed/storage/blockvol/blockvol.go b/weed/storage/blockvol/blockvol.go index 73beb43c7..6cef234d2 100644 --- a/weed/storage/blockvol/blockvol.go +++ b/weed/storage/blockvol/blockvol.go @@ -78,6 +78,8 @@ type BlockVol struct { rebuildServer *RebuildServer assignMu sync.Mutex // serializes HandleAssignment calls drainTimeout time.Duration // default 10s, for demote drain + rebuildSessMu sync.RWMutex + rebuildSess *RebuildSession // Health score and scrub (CP8-2). healthScore *HealthScore diff --git a/weed/storage/blockvol/rebuild_bitmap.go b/weed/storage/blockvol/rebuild_bitmap.go new file mode 100644 index 000000000..60fec2661 --- /dev/null +++ b/weed/storage/blockvol/rebuild_bitmap.go @@ -0,0 +1,84 @@ +package blockvol + +// RebuildBitmap is a session-scoped dense bitset tracking which LBAs have +// been covered by applied WAL entries during a rebuild session. It is the +// overwrite protection for the two-line rebuild model: +// +// - Line 1 (base lane): snapshot/extent blocks copied to replica +// - Line 2 (WAL lane): live WAL entries applied to replica local WAL +// +// When a base chunk targets an LBA: +// - bitmap clear → write base data (no conflict) +// - bitmap set → skip (WAL-applied data is newer, WAL always wins) +// +// The bit is set when a WAL entry is APPLIED to the replica's local WAL +// (replayable after crash), NOT when it is merely received on the network. +// +// This bitmap is session-local volatile state. After crash, the rebuild +// session must restart from scratch with a fresh bitmap. Durable WAL +// entries survive crash and protect correctness via WAL replay. +// +// Implementation is a dense bitset modeled on SnapshotBitmap. No internal +// locking — callers must serialize access if needed. +type RebuildBitmap struct { + data []byte + totalLBAs uint64 // total number of LBAs (= volumeSize / blockSize) + blockSize uint32 + appliedCount uint64 // number of LBAs marked as WAL-applied +} + +// NewRebuildBitmap creates a zero-initialized rebuild bitmap. +// totalLBAs = volumeSize / blockSize. +func NewRebuildBitmap(totalLBAs uint64, blockSize uint32) *RebuildBitmap { + byteLen := (totalLBAs + 7) / 8 + return &RebuildBitmap{ + data: make([]byte, byteLen), + totalLBAs: totalLBAs, + blockSize: blockSize, + } +} + +// MarkApplied sets the bit for the given LBA, indicating that a WAL entry +// covering this LBA has been applied to the replica's local WAL. +func (b *RebuildBitmap) MarkApplied(lba uint64) { + if lba >= b.totalLBAs { + return + } + if !b.IsApplied(lba) { + b.appliedCount++ + } + b.data[lba/8] |= 1 << (lba % 8) +} + +// IsApplied returns true if the LBA has been covered by an applied WAL entry. +// When true, base lane data for this LBA must be skipped. +func (b *RebuildBitmap) IsApplied(lba uint64) bool { + if lba >= b.totalLBAs { + return false + } + return b.data[lba/8]&(1<<(lba%8)) != 0 +} + +// ShouldApplyBase returns true if the base lane may write data at this LBA. +// This is the conflict resolution rule: WAL-applied wins over base. +func (b *RebuildBitmap) ShouldApplyBase(lba uint64) bool { + return !b.IsApplied(lba) +} + +// AppliedCount returns the number of LBAs marked as WAL-applied. +func (b *RebuildBitmap) AppliedCount() uint64 { + return b.appliedCount +} + +// TotalLBAs returns the total number of trackable LBAs. +func (b *RebuildBitmap) TotalLBAs() uint64 { + return b.totalLBAs +} + +// Clear resets the bitmap to all-zero (no LBAs applied). +func (b *RebuildBitmap) Clear() { + for i := range b.data { + b.data[i] = 0 + } + b.appliedCount = 0 +} diff --git a/weed/storage/blockvol/rebuild_session.go b/weed/storage/blockvol/rebuild_session.go new file mode 100644 index 000000000..530146518 --- /dev/null +++ b/weed/storage/blockvol/rebuild_session.go @@ -0,0 +1,409 @@ +package blockvol + +import ( + "fmt" + "sync" +) + +// RebuildSessionPhase tracks the lifecycle of one rebuild session on the +// replica side. Matches the replica state machine in v2-rebuild-mvp-session-protocol.md. +type RebuildSessionPhase string + +const ( + RebuildPhaseIdle RebuildSessionPhase = "idle" + RebuildPhaseAccepted RebuildSessionPhase = "accepted" + RebuildPhaseRunning RebuildSessionPhase = "running" + RebuildPhaseBaseComplete RebuildSessionPhase = "base_complete" + RebuildPhaseCompleted RebuildSessionPhase = "completed" + RebuildPhaseFailed RebuildSessionPhase = "failed" +) + +// RebuildSessionConfig is the contract for starting one rebuild session. +// Issued by the primary via sessionControl(start_rebuild). +type RebuildSessionConfig struct { + SessionID uint64 + Epoch uint64 + BaseLSN uint64 // snapshot point-in-time LSN + TargetLSN uint64 // WAL must reach this before completion + SnapshotID uint32 // snapshot to use as base (0 = use current extent) +} + +// RebuildSession manages one replica-side rebuild session with two concurrent +// data lanes: +// +// - Base lane: trusted snapshot/extent blocks applied with bitmap protection +// - WAL lane: live WAL entries applied and marked in bitmap +// +// The bitmap ensures WAL-applied data always wins over base data. The session +// completes when both base is fully transferred AND WAL has reached the target. +// +// This is session-scoped volatile state. After crash, the session must restart +// from scratch. Durable WAL entries survive via local WAL replay. +type RebuildSession struct { + mu sync.Mutex + config RebuildSessionConfig + phase RebuildSessionPhase + bitmap *RebuildBitmap + vol *BlockVol + + // Progress tracking + walAppliedLSN uint64 // highest WAL LSN applied during this session + baseBlocksTotal uint64 // total base blocks to transfer + baseBlocksApplied uint64 // base blocks successfully applied (not skipped) + baseBlocksSkipped uint64 // base blocks skipped due to bitmap conflict + baseComplete bool // all base blocks have been processed + failReason string +} + +// NewRebuildSession creates a replica-side rebuild session. The session starts +// in Accepted phase. Call Start() to transition to Running. +func NewRebuildSession(vol *BlockVol, config RebuildSessionConfig) (*RebuildSession, error) { + if vol == nil { + return nil, fmt.Errorf("rebuild session: volume is nil") + } + if config.TargetLSN == 0 { + return nil, fmt.Errorf("rebuild session: target LSN is required") + } + if config.Epoch == 0 { + return nil, fmt.Errorf("rebuild session: epoch is required") + } + + info := vol.Info() + totalLBAs := info.VolumeSize / uint64(info.BlockSize) + bitmap := NewRebuildBitmap(totalLBAs, info.BlockSize) + + return &RebuildSession{ + config: config, + phase: RebuildPhaseAccepted, + bitmap: bitmap, + vol: vol, + }, nil +} + +// Start transitions the session from Accepted to Running. +func (s *RebuildSession) Start() error { + s.mu.Lock() + defer s.mu.Unlock() + if s.phase != RebuildPhaseAccepted { + return fmt.Errorf("rebuild session: cannot start from phase %s", s.phase) + } + s.phase = RebuildPhaseRunning + return nil +} + +// ApplyWALEntry applies one WAL entry through the WAL lane. The entry is +// applied to the replica's local WAL, and the bitmap bit is set for each +// LBA covered by the entry. This ensures base lane data for the same LBA +// will be skipped (WAL always wins). +// +// The bitmap bit is set AFTER successful WAL append (applied), not on +// receive. This is the key correctness invariant. +func (s *RebuildSession) ApplyWALEntry(entry *WALEntry) error { + s.mu.Lock() + defer s.mu.Unlock() + if s.phase != RebuildPhaseRunning && s.phase != RebuildPhaseBaseComplete { + return fmt.Errorf("rebuild session: WAL apply not allowed in phase %s", s.phase) + } + if entry.Epoch != s.config.Epoch { + return fmt.Errorf("rebuild session: epoch mismatch: entry=%d session=%d", entry.Epoch, s.config.Epoch) + } + + // Apply to local WAL via the volume's WAL writer. + if err := s.vol.applyRebuildWALEntry(entry); err != nil { + return fmt.Errorf("rebuild session: WAL apply LSN=%d: %w", entry.LSN, err) + } + + // AFTER successful apply: mark bitmap for each LBA covered by this entry. + if entry.Type == EntryTypeWrite && entry.Length > 0 { + blockSize := uint64(s.config.blockSize()) + if blockSize == 0 { + blockSize = uint64(s.vol.Info().BlockSize) + } + startLBA := entry.LBA + blocks := uint64(entry.Length) / blockSize + if blocks == 0 { + blocks = 1 + } + for i := uint64(0); i < blocks; i++ { + s.bitmap.MarkApplied(startLBA + i) + } + } + + if entry.LSN > s.walAppliedLSN { + s.walAppliedLSN = entry.LSN + } + return nil +} + +// ApplyBaseBlock applies one base (snapshot) block through the base lane. +// If the bitmap shows the LBA was already covered by a WAL entry, the base +// block is skipped (WAL always wins over older base data). +// +// Returns (applied bool, err error). applied=false means the block was +// skipped due to bitmap conflict, which is correct behavior. +func (s *RebuildSession) ApplyBaseBlock(lba uint64, data []byte) (bool, error) { + s.mu.Lock() + defer s.mu.Unlock() + if s.phase != RebuildPhaseRunning { + return false, fmt.Errorf("rebuild session: base apply not allowed in phase %s", s.phase) + } + + // Bitmap conflict check: WAL-applied LBA wins. + if !s.bitmap.ShouldApplyBase(lba) { + s.baseBlocksSkipped++ + return false, nil + } + + // Apply base block directly to the extent (not through WAL). + if err := s.vol.writeExtentDirect(lba, data); err != nil { + return false, fmt.Errorf("rebuild session: base apply LBA=%d: %w", lba, err) + } + + s.baseBlocksApplied++ + return true, nil +} + +// MarkBaseComplete marks the base lane as fully transferred. +// The session transitions to BaseComplete phase if currently Running. +func (s *RebuildSession) MarkBaseComplete(totalBlocks uint64) { + s.mu.Lock() + defer s.mu.Unlock() + s.baseBlocksTotal = totalBlocks + s.baseComplete = true + if s.phase == RebuildPhaseRunning { + s.phase = RebuildPhaseBaseComplete + } +} + +// TryComplete checks if both completion conditions are met: +// 1. base_complete = true +// 2. wal_applied_lsn >= target_lsn +// +// If both are true, transitions to Completed phase and returns the achieved LSN. +// If not ready, returns (0, false). +func (s *RebuildSession) TryComplete() (uint64, bool) { + s.mu.Lock() + defer s.mu.Unlock() + if !s.baseComplete { + return 0, false + } + if s.walAppliedLSN < s.config.TargetLSN { + return 0, false + } + if s.phase == RebuildPhaseCompleted || s.phase == RebuildPhaseFailed { + return 0, false + } + s.phase = RebuildPhaseCompleted + return s.walAppliedLSN, true +} + +// Fail marks the session as failed with a reason. +func (s *RebuildSession) Fail(reason string) { + s.mu.Lock() + defer s.mu.Unlock() + s.phase = RebuildPhaseFailed + s.failReason = reason +} + +// Phase returns the current session phase. +func (s *RebuildSession) Phase() RebuildSessionPhase { + s.mu.Lock() + defer s.mu.Unlock() + return s.phase +} + +// WALAppliedLSN returns the highest WAL LSN applied during this session. +func (s *RebuildSession) WALAppliedLSN() uint64 { + s.mu.Lock() + defer s.mu.Unlock() + return s.walAppliedLSN +} + +// Progress returns the current session progress for sessionAck reporting. +func (s *RebuildSession) Progress() RebuildSessionProgress { + s.mu.Lock() + defer s.mu.Unlock() + return RebuildSessionProgress{ + Phase: s.phase, + WALAppliedLSN: s.walAppliedLSN, + BaseBlocksTotal: s.baseBlocksTotal, + BaseBlocksApplied: s.baseBlocksApplied, + BaseBlocksSkipped: s.baseBlocksSkipped, + BaseComplete: s.baseComplete, + BitmapAppliedCount: s.bitmap.AppliedCount(), + FailReason: s.failReason, + } +} + +// Config returns the session configuration. +func (s *RebuildSession) Config() RebuildSessionConfig { + return s.config +} + +// RebuildSessionProgress is the read-only progress snapshot for sessionAck. +type RebuildSessionProgress struct { + Phase RebuildSessionPhase + WALAppliedLSN uint64 + BaseBlocksTotal uint64 + BaseBlocksApplied uint64 + BaseBlocksSkipped uint64 + BaseComplete bool + BitmapAppliedCount uint64 + FailReason string +} + +func (p RebuildSessionProgress) Completed() bool { + return p.Phase == RebuildPhaseCompleted +} + +// blockSize returns the block size from config, defaulting to 4096. +func (c RebuildSessionConfig) blockSize() uint32 { + // Config doesn't carry block size directly; callers use vol.Info().BlockSize. + return 0 +} + +func (s *RebuildSession) SessionID() uint64 { + return s.config.SessionID +} + +// StartRebuildSession installs and starts one active rebuild session on the +// replica. A new session supersedes any previous active rebuild session. +func (v *BlockVol) StartRebuildSession(config RebuildSessionConfig) error { + if config.SessionID == 0 { + return fmt.Errorf("rebuild session: session ID is required") + } + session, err := NewRebuildSession(v, config) + if err != nil { + return err + } + if err := session.Start(); err != nil { + return err + } + + v.rebuildSessMu.Lock() + defer v.rebuildSessMu.Unlock() + if v.rebuildSess != nil { + v.rebuildSess.Fail("superseded") + } + v.rebuildSess = session + return nil +} + +// CancelRebuildSession cancels and removes one active rebuild session. +func (v *BlockVol) CancelRebuildSession(sessionID uint64, reason string) error { + v.rebuildSessMu.Lock() + defer v.rebuildSessMu.Unlock() + if v.rebuildSess == nil { + return fmt.Errorf("rebuild session: no active session") + } + if sessionID != 0 && v.rebuildSess.SessionID() != sessionID { + return fmt.Errorf("rebuild session: session mismatch: have %d want %d", v.rebuildSess.SessionID(), sessionID) + } + if reason == "" { + reason = "cancelled" + } + v.rebuildSess.Fail(reason) + v.rebuildSess = nil + return nil +} + +// ActiveRebuildSession returns the current rebuild session snapshots. +func (v *BlockVol) ActiveRebuildSession() (RebuildSessionConfig, RebuildSessionProgress, bool) { + v.rebuildSessMu.RLock() + session := v.rebuildSess + v.rebuildSessMu.RUnlock() + if session == nil { + return RebuildSessionConfig{}, RebuildSessionProgress{}, false + } + return session.Config(), session.Progress(), true +} + +// ApplyRebuildSessionWALEntry routes one WAL entry into the active rebuild +// session after validating the session ID. +func (v *BlockVol) ApplyRebuildSessionWALEntry(sessionID uint64, entry *WALEntry) error { + session, err := v.activeRebuildSession(sessionID) + if err != nil { + return err + } + return session.ApplyWALEntry(entry) +} + +// ApplyRebuildSessionBaseBlock routes one base block into the active rebuild +// session after validating the session ID. +func (v *BlockVol) ApplyRebuildSessionBaseBlock(sessionID uint64, lba uint64, data []byte) (bool, error) { + session, err := v.activeRebuildSession(sessionID) + if err != nil { + return false, err + } + return session.ApplyBaseBlock(lba, data) +} + +// MarkRebuildSessionBaseComplete marks the active rebuild session's base lane as +// fully processed. +func (v *BlockVol) MarkRebuildSessionBaseComplete(sessionID uint64, totalBlocks uint64) error { + session, err := v.activeRebuildSession(sessionID) + if err != nil { + return err + } + session.MarkBaseComplete(totalBlocks) + return nil +} + +// TryCompleteRebuildSession evaluates whether the active rebuild session has +// reached its dual completion gate. +func (v *BlockVol) TryCompleteRebuildSession(sessionID uint64) (uint64, bool, error) { + session, err := v.activeRebuildSession(sessionID) + if err != nil { + return 0, false, err + } + achieved, completed := session.TryComplete() + return achieved, completed, nil +} + +func (v *BlockVol) activeRebuildSession(sessionID uint64) (*RebuildSession, error) { + v.rebuildSessMu.RLock() + session := v.rebuildSess + v.rebuildSessMu.RUnlock() + if session == nil { + return nil, fmt.Errorf("rebuild session: no active session") + } + if sessionID == 0 || session.SessionID() != sessionID { + return nil, fmt.Errorf("rebuild session: session mismatch: have %d want %d", session.SessionID(), sessionID) + } + return session, nil +} + +// applyRebuildWALEntry applies a WAL entry during rebuild without going +// through the normal write gate (epoch/role checks are session-level). +// The entry is appended to the local WAL and dirty map is updated. +func (v *BlockVol) applyRebuildWALEntry(entry *WALEntry) error { + if v == nil { + return fmt.Errorf("volume is nil") + } + v.ioMu.RLock() + defer v.ioMu.RUnlock() + + walOff, err := v.wal.Append(entry) + if err != nil { + return err + } + // Update dirty map so ReadLBA sees the WAL data. + v.dirtyMap.Put(entry.LBA, walOff, entry.LSN, entry.Length) + return nil +} + +// writeExtentDirect writes data directly to the extent file at the given LBA. +// Used by the base lane during rebuild when bitmap shows no WAL conflict. +// This bypasses the WAL — the data goes directly to the extent image. +func (v *BlockVol) writeExtentDirect(lba uint64, data []byte) error { + if v == nil { + return fmt.Errorf("volume is nil") + } + v.ioMu.RLock() + defer v.ioMu.RUnlock() + + extentStart := v.super.WALOffset + v.super.WALSize + offset := int64(extentStart) + int64(lba)*int64(v.super.BlockSize) + _, err := v.fd.WriteAt(data, offset) + return err +} diff --git a/weed/storage/blockvol/test/component/rebuild_mvp_test.go b/weed/storage/blockvol/test/component/rebuild_mvp_test.go new file mode 100644 index 000000000..c794ecd69 --- /dev/null +++ b/weed/storage/blockvol/test/component/rebuild_mvp_test.go @@ -0,0 +1,378 @@ +package component + +// Component tests for the rebuild MVP session protocol. +// +// These prove the three core correctness invariants from +// v2-rebuild-mvp-session-protocol.md: +// +// 1. Base lane + WAL lane converge to target +// 2. WAL-applied LBA is never overwritten by later base-copy data +// 3. Bitmap bit is set on "applied" (local WAL write), not "received" +// +// Each test uses real BlockVol (WAL, extent, dirty map) but no network. +// The rebuild session is exercised directly in-process. + +import ( + "bytes" + "path/filepath" + "testing" + "time" + + "github.com/seaweedfs/seaweedfs/weed/storage/blockvol" +) + +// Test 1: Base lane + WAL lane converge to target. +// +// Scenario: primary has 10 blocks of data. Rebuild session receives +// base blocks for all 10 LBAs AND WAL entries for some of them. +// After both lanes complete, the replica has all 10 blocks readable +// and the session reaches Completed phase. +func TestRebuild_BasePlusWAL_ConvergesToTarget(t *testing.T) { + primary, replica := createRebuildPair(t) + defer primary.Close() + defer replica.Close() + + // Write 10 blocks on primary at LBA 0-9. + blocks := make([][]byte, 10) + for i := 0; i < 10; i++ { + blocks[i] = bytes.Repeat([]byte{byte(0xA0 + i)}, 4096) + if err := primary.WriteLBA(uint64(i), blocks[i]); err != nil { + t.Fatalf("primary write LBA %d: %v", i, err) + } + } + primaryHead := primary.Status().WALHeadLSN + t.Logf("primary WALHeadLSN=%d after 10 writes", primaryHead) + + // Create rebuild session on replica targeting primary's head. + session, err := blockvol.NewRebuildSession(replica, blockvol.RebuildSessionConfig{ + SessionID: 1, + Epoch: 1, + BaseLSN: primaryHead, + TargetLSN: primaryHead, + }) + if err != nil { + t.Fatalf("new rebuild session: %v", err) + } + if err := session.Start(); err != nil { + t.Fatalf("start session: %v", err) + } + + // WAL lane: apply WAL entries for LBAs 0-4 (first half). + for lba := uint64(0); lba < 5; lba++ { + entry := &blockvol.WALEntry{ + LSN: lba + 1, // LSN 1-5 + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: lba, + Length: 4096, + Data: blocks[lba], + } + if err := session.ApplyWALEntry(entry); err != nil { + t.Fatalf("WAL apply LBA %d: %v", lba, err) + } + } + + // Base lane: apply base blocks for LBAs 0-9 (all). + // LBAs 0-4 should be SKIPPED (bitmap set by WAL lane). + // LBAs 5-9 should be APPLIED (bitmap clear). + for lba := uint64(0); lba < 10; lba++ { + applied, err := session.ApplyBaseBlock(lba, blocks[lba]) + if err != nil { + t.Fatalf("base apply LBA %d: %v", lba, err) + } + if lba < 5 && applied { + t.Fatalf("LBA %d: base should be skipped (WAL-applied), got applied=true", lba) + } + if lba >= 5 && !applied { + t.Fatalf("LBA %d: base should be applied (bitmap clear), got applied=false", lba) + } + } + + // Mark base complete + apply remaining WAL entries to reach target. + session.MarkBaseComplete(10) + for lba := uint64(5); lba < 10; lba++ { + entry := &blockvol.WALEntry{ + LSN: lba + 1, // LSN 6-10 + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: lba, + Length: 4096, + Data: blocks[lba], + } + if err := session.ApplyWALEntry(entry); err != nil { + t.Fatalf("WAL apply LBA %d: %v", lba, err) + } + } + + // Try completion: both conditions should be met. + achievedLSN, completed := session.TryComplete() + if !completed { + progress := session.Progress() + t.Fatalf("session did not complete: walApplied=%d target=%d baseComplete=%v", + progress.WALAppliedLSN, primaryHead, progress.BaseComplete) + } + t.Logf("session completed: achievedLSN=%d", achievedLSN) + + // Verify all 10 blocks are readable on replica. + for lba := uint64(0); lba < 10; lba++ { + data, err := replica.ReadLBA(lba, 4096) + if err != nil { + t.Fatalf("replica read LBA %d: %v", lba, err) + } + if !bytes.Equal(data, blocks[lba]) { + t.Fatalf("replica LBA %d mismatch: got[0]=0x%02x want[0]=0x%02x", + lba, data[0], blocks[lba][0]) + } + } + t.Log("all 10 blocks converged correctly on replica") +} + +// Test 2: WAL-applied LBA is never overwritten by later base-copy data. +// +// Scenario: WAL entry writes 0xBB to LBA 5, then base lane tries to +// write 0xAA to the same LBA. The base write must be skipped, and the +// replica must read 0xBB (WAL wins). +func TestRebuild_WALApplied_NeverOverwrittenByBase(t *testing.T) { + primary, replica := createRebuildPair(t) + defer primary.Close() + defer replica.Close() + + session, err := blockvol.NewRebuildSession(replica, blockvol.RebuildSessionConfig{ + SessionID: 2, + Epoch: 1, + BaseLSN: 100, + TargetLSN: 100, + }) + if err != nil { + t.Fatalf("new session: %v", err) + } + if err := session.Start(); err != nil { + t.Fatalf("start: %v", err) + } + + // WAL lane: apply 0xBB to LBA 5. + walData := bytes.Repeat([]byte{0xBB}, 4096) + walEntry := &blockvol.WALEntry{ + LSN: 1, + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: 5, + Length: 4096, + Data: walData, + } + if err := session.ApplyWALEntry(walEntry); err != nil { + t.Fatalf("WAL apply: %v", err) + } + + // Base lane: try to apply 0xAA to same LBA 5. + baseData := bytes.Repeat([]byte{0xAA}, 4096) + applied, err := session.ApplyBaseBlock(5, baseData) + if err != nil { + t.Fatalf("base apply: %v", err) + } + if applied { + t.Fatal("BUG: base block applied to WAL-covered LBA — bitmap conflict not enforced") + } + + // Read from replica: must be 0xBB (WAL wins), not 0xAA. + readBack, err := replica.ReadLBA(5, 4096) + if err != nil { + t.Fatalf("replica read: %v", err) + } + if readBack[0] != 0xBB { + t.Fatalf("BUG: replica LBA 5 = 0x%02x, want 0xBB (WAL must win over base)", readBack[0]) + } + t.Log("WAL-applied LBA correctly protected: base data skipped, WAL data preserved") +} + +// Test 3: Bitmap bit is set on "applied" (local WAL write), not "received". +// +// Scenario: We verify the bitmap state at precise points: +// - Before ApplyWALEntry: bitmap must be clear +// - After ApplyWALEntry succeeds: bitmap must be set +// +// This proves the bit is set AFTER successful local WAL append, which is +// the key correctness invariant for crash safety. +func TestRebuild_BitmapSetOnApplied_NotReceived(t *testing.T) { + primary, replica := createRebuildPair(t) + defer primary.Close() + defer replica.Close() + + session, err := blockvol.NewRebuildSession(replica, blockvol.RebuildSessionConfig{ + SessionID: 3, + Epoch: 1, + BaseLSN: 50, + TargetLSN: 50, + }) + if err != nil { + t.Fatalf("new session: %v", err) + } + if err := session.Start(); err != nil { + t.Fatalf("start: %v", err) + } + + // Before WAL apply: base lane should be allowed for LBA 7. + applied, err := session.ApplyBaseBlock(7, bytes.Repeat([]byte{0x11}, 4096)) + if err != nil { + t.Fatalf("pre-WAL base apply: %v", err) + } + if !applied { + t.Fatal("base block at LBA 7 should be applied before any WAL entry") + } + + // Apply WAL entry to LBA 7. + walEntry := &blockvol.WALEntry{ + LSN: 1, + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: 7, + Length: 4096, + Data: bytes.Repeat([]byte{0x22}, 4096), + } + if err := session.ApplyWALEntry(walEntry); err != nil { + t.Fatalf("WAL apply LBA 7: %v", err) + } + + // After WAL apply: base lane must be BLOCKED for LBA 7. + applied2, err := session.ApplyBaseBlock(7, bytes.Repeat([]byte{0x33}, 4096)) + if err != nil { + t.Fatalf("post-WAL base apply: %v", err) + } + if applied2 { + t.Fatal("BUG: base block applied after WAL entry — bitmap was not set on apply") + } + + // Verify replica reads WAL data (0x22), not base (0x11) or second base (0x33). + readBack, err := replica.ReadLBA(7, 4096) + if err != nil { + t.Fatalf("replica read: %v", err) + } + if readBack[0] != 0x22 { + t.Fatalf("replica LBA 7 = 0x%02x, want 0x22 (WAL-applied data)", readBack[0]) + } + + // Verify bitmap count: exactly 1 LBA should be marked. + progress := session.Progress() + if progress.BitmapAppliedCount != 1 { + t.Fatalf("bitmap applied count=%d, want 1", progress.BitmapAppliedCount) + } + t.Log("bitmap set on applied (after WAL append), not on received — correctness invariant holds") +} + +func TestRebuild_ControlSurface_StartSupersedeAndComplete(t *testing.T) { + primary, replica := createRebuildPair(t) + defer primary.Close() + defer replica.Close() + + if err := replica.StartRebuildSession(blockvol.RebuildSessionConfig{ + SessionID: 10, + Epoch: 1, + BaseLSN: 1, + TargetLSN: 1, + }); err != nil { + t.Fatalf("start session 10: %v", err) + } + + cfg, progress, ok := replica.ActiveRebuildSession() + if !ok { + t.Fatal("expected active rebuild session") + } + if cfg.SessionID != 10 { + t.Fatalf("active session ID=%d, want 10", cfg.SessionID) + } + if progress.Phase != blockvol.RebuildPhaseRunning { + t.Fatalf("active phase=%s, want running", progress.Phase) + } + + if err := replica.StartRebuildSession(blockvol.RebuildSessionConfig{ + SessionID: 11, + Epoch: 1, + BaseLSN: 1, + TargetLSN: 1, + }); err != nil { + t.Fatalf("start session 11: %v", err) + } + + cfg, _, ok = replica.ActiveRebuildSession() + if !ok || cfg.SessionID != 11 { + t.Fatalf("expected superseded active session 11, got ok=%v id=%d", ok, cfg.SessionID) + } + + err := replica.ApplyRebuildSessionWALEntry(10, &blockvol.WALEntry{ + LSN: 1, + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: 0, + Length: 4096, + Data: bytes.Repeat([]byte{0xAA}, 4096), + }) + if err == nil { + t.Fatal("expected stale session ID to be rejected") + } + + if err := replica.ApplyRebuildSessionWALEntry(11, &blockvol.WALEntry{ + LSN: 1, + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: 0, + Length: 4096, + Data: bytes.Repeat([]byte{0xBB}, 4096), + }); err != nil { + t.Fatalf("apply WAL through control surface: %v", err) + } + if err := replica.MarkRebuildSessionBaseComplete(11, 0); err != nil { + t.Fatalf("mark base complete: %v", err) + } + achieved, completed, err := replica.TryCompleteRebuildSession(11) + if err != nil { + t.Fatalf("try complete: %v", err) + } + if !completed || achieved != 1 { + t.Fatalf("completion result achieved=%d completed=%v, want achieved=1 completed=true", achieved, completed) + } + + _, progress, ok = replica.ActiveRebuildSession() + if !ok { + t.Fatal("expected completed session to remain queryable") + } + if !progress.Completed() { + t.Fatalf("progress phase=%s, want completed", progress.Phase) + } + + if err := replica.CancelRebuildSession(11, "test_done"); err != nil { + t.Fatalf("cancel session: %v", err) + } + if _, _, ok := replica.ActiveRebuildSession(); ok { + t.Fatal("expected no active session after cancel") + } +} + +// --- Helpers --- + +func createRebuildPair(t *testing.T) (primary, replica *blockvol.BlockVol) { + t.Helper() + opts := blockvol.CreateOptions{ + VolumeSize: 4 * 1024 * 1024, // 4MB = 1024 LBAs at 4K + BlockSize: 4096, + WALSize: 1 * 1024 * 1024, + } + p, err := blockvol.CreateBlockVol(filepath.Join(t.TempDir(), "primary.blk"), opts) + if err != nil { + t.Fatal(err) + } + if err := p.HandleAssignment(1, blockvol.RolePrimary, 30*time.Second); err != nil { + p.Close() + t.Fatal(err) + } + r, err := blockvol.CreateBlockVol(filepath.Join(t.TempDir(), "replica.blk"), opts) + if err != nil { + p.Close() + t.Fatal(err) + } + if err := r.HandleAssignment(1, blockvol.RoleReplica, 30*time.Second); err != nil { + p.Close() + r.Close() + t.Fatal(err) + } + return p, r +} diff --git a/weed/storage/blockvol/test/component/rebuild_transport_test.go b/weed/storage/blockvol/test/component/rebuild_transport_test.go new file mode 100644 index 000000000..f910c6e6e --- /dev/null +++ b/weed/storage/blockvol/test/component/rebuild_transport_test.go @@ -0,0 +1,294 @@ +package component + +// Component tests for rebuild with real WAL transport between two BlockVol +// instances. Unlike rebuild_mvp_test.go (direct function calls), these tests +// use real TCP shipping for the WAL lane and real extent read for the base lane. +// +// Architecture: +// Primary BlockVol (real WAL + extent + snapshot) +// ├─ WAL lane: ShipAll → TCP → ReplicaReceiver → replica WAL +// └─ Base lane: read primary extent → ApplyBaseBlock on session +// +// Replica BlockVol (real WAL + extent + RebuildSession + bitmap) + +import ( + "bytes" + "path/filepath" + "testing" + "time" + + "github.com/seaweedfs/seaweedfs/weed/storage/blockvol" +) + +// TestRebuild_Transport_TwoLineWithRealShipping exercises the full two-line +// rebuild with real TCP WAL shipping between primary and replica. +// +// Flow: +// 1. Primary writes 20 blocks (LBA 0-19), creating WAL entries +// 2. Replica starts receiver, primary wires shipper (real TCP) +// 3. Rebuild session on replica: base lane + WAL lane in parallel +// 4. Base lane reads primary extent directly, sends to replica session +// 5. WAL lane: primary continues writing, entries ship via TCP +// 6. Verify replica has all data correct after rebuild completes +func TestRebuild_Transport_TwoLineWithRealShipping(t *testing.T) { + primary, replica := createTransportRebuildPair(t) + defer primary.Close() + defer replica.Close() + + // Step 1: Write initial data on primary (pre-rebuild baseline). + initialBlocks := 20 + blockData := make(map[uint64][]byte) + for i := 0; i < initialBlocks; i++ { + data := bytes.Repeat([]byte{byte(0xA0 + i)}, 4096) + blockData[uint64(i)] = data + if err := primary.WriteLBA(uint64(i), data); err != nil { + t.Fatalf("primary write LBA %d: %v", i, err) + } + } + // Flush primary so extent has the data (needed for base lane read). + if err := primary.SyncCache(); err != nil { + t.Fatalf("primary SyncCache: %v", err) + } + if err := primary.ForceFlush(); err != nil { + t.Fatalf("primary ForceFlush: %v", err) + } + + baseLSN := primary.Status().WALHeadLSN + t.Logf("primary baseline: %d blocks, WALHeadLSN=%d", initialBlocks, baseLSN) + + // Step 2: Wire real TCP shipping (WAL lane transport). + if err := replica.StartReplicaReceiver(":0", ":0"); err != nil { + t.Fatal(err) + } + recvAddr := replica.ReplicaReceiverAddr() + primary.SetReplicaAddr(recvAddr.DataAddr, recvAddr.CtrlAddr) + t.Logf("WAL lane wired: primary → %s/%s", recvAddr.DataAddr, recvAddr.CtrlAddr) + + // Step 3: Create rebuild session on replica. + session, err := blockvol.NewRebuildSession(replica, blockvol.RebuildSessionConfig{ + SessionID: 1, + Epoch: 1, + BaseLSN: baseLSN, + TargetLSN: baseLSN + 5, // expect 5 more WAL entries during rebuild + }) + if err != nil { + t.Fatalf("new rebuild session: %v", err) + } + if err := session.Start(); err != nil { + t.Fatalf("start session: %v", err) + } + + // Step 4: Base lane — read primary extent, send to replica session. + // This simulates the snapshot/base block transfer. + info := primary.Info() + totalLBAs := info.VolumeSize / uint64(info.BlockSize) + baseApplied := 0 + baseSkipped := 0 + for lba := uint64(0); lba < totalLBAs && lba < uint64(initialBlocks); lba++ { + extentData, err := primary.ReadLBA(lba, uint32(info.BlockSize)) + if err != nil { + t.Fatalf("primary read LBA %d: %v", lba, err) + } + applied, err := session.ApplyBaseBlock(lba, extentData) + if err != nil { + t.Fatalf("base apply LBA %d: %v", lba, err) + } + if applied { + baseApplied++ + } else { + baseSkipped++ + } + } + session.MarkBaseComplete(uint64(initialBlocks)) + t.Logf("base lane: %d applied, %d skipped (WAL conflict)", baseApplied, baseSkipped) + + // Step 5: WAL lane — primary writes more blocks, shipping via real TCP. + // These writes go through ShipAll → TCP → ReplicaReceiver on replica. + liveBlocks := 5 + for i := 0; i < liveBlocks; i++ { + lba := uint64(initialBlocks + i) + data := bytes.Repeat([]byte{byte(0xF0 + i)}, 4096) + blockData[lba] = data + if err := primary.WriteLBA(lba, data); err != nil { + t.Fatalf("primary live write LBA %d: %v", lba, err) + } + } + + // Wait for WAL entries to arrive at replica via TCP. + time.Sleep(1 * time.Second) + + // Also apply the live WAL entries to the rebuild session. + // In production, the replica receiver would route these to the session. + // Here we manually apply them since the receiver doesn't know about + // the rebuild session yet (that wiring is a server-layer concern). + for i := 0; i < liveBlocks; i++ { + lba := uint64(initialBlocks + i) + entry := &blockvol.WALEntry{ + LSN: baseLSN + uint64(i) + 1, + Epoch: 1, + Type: blockvol.EntryTypeWrite, + LBA: lba, + Length: 4096, + Data: blockData[lba], + } + if err := session.ApplyWALEntry(entry); err != nil { + t.Fatalf("session WAL apply LBA %d: %v", lba, err) + } + } + + // Step 6: Try completion. + achievedLSN, completed := session.TryComplete() + if !completed { + progress := session.Progress() + t.Fatalf("session did not complete: walApplied=%d target=%d baseComplete=%v phase=%s", + progress.WALAppliedLSN, baseLSN+5, progress.BaseComplete, progress.Phase) + } + t.Logf("rebuild completed: achievedLSN=%d", achievedLSN) + + // Step 7: Verify ALL blocks on replica. + for lba, expected := range blockData { + got, err := replica.ReadLBA(lba, 4096) + if err != nil { + t.Fatalf("replica read LBA %d: %v", lba, err) + } + if !bytes.Equal(got, expected) { + t.Fatalf("replica LBA %d mismatch: got[0]=0x%02x want[0]=0x%02x", lba, got[0], expected[0]) + } + } + t.Logf("all %d blocks verified on replica", len(blockData)) +} + +// TestRebuild_Transport_LiveWritesDuringBaseCopy verifies that writes +// happening on the primary DURING base copy are correctly handled. +// The WAL lane ships them via TCP, and bitmap protects them from being +// overwritten by the base copy. +func TestRebuild_Transport_LiveWritesDuringBaseCopy(t *testing.T) { + primary, replica := createTransportRebuildPair(t) + defer primary.Close() + defer replica.Close() + + // Write initial data. + for i := 0; i < 10; i++ { + data := bytes.Repeat([]byte{byte(0x10 + i)}, 4096) + if err := primary.WriteLBA(uint64(i), data); err != nil { + t.Fatalf("primary write LBA %d: %v", i, err) + } + } + if err := primary.SyncCache(); err != nil { + t.Fatalf("SyncCache: %v", err) + } + if err := primary.ForceFlush(); err != nil { + t.Fatalf("ForceFlush: %v", err) + } + baseLSN := primary.Status().WALHeadLSN + + // Wire TCP shipping. + if err := replica.StartReplicaReceiver(":0", ":0"); err != nil { + t.Fatal(err) + } + recvAddr := replica.ReplicaReceiverAddr() + primary.SetReplicaAddr(recvAddr.DataAddr, recvAddr.CtrlAddr) + + session, err := blockvol.NewRebuildSession(replica, blockvol.RebuildSessionConfig{ + SessionID: 2, Epoch: 1, BaseLSN: baseLSN, TargetLSN: baseLSN + 3, + }) + if err != nil { + t.Fatal(err) + } + session.Start() + + // Simulate interleaved base copy + live writes: + // 1. Copy base blocks 0-4 + // 2. Primary writes NEW data to LBA 3 (live write during rebuild) + // 3. Apply that live write via WAL lane to session + // 4. Copy base blocks 5-9 (LBA 3 already covered by WAL) + + // Base blocks 0-4. + info := primary.Info() + for lba := uint64(0); lba < 5; lba++ { + data, _ := primary.ReadLBA(lba, uint32(info.BlockSize)) + session.ApplyBaseBlock(lba, data) + } + + // Live write to LBA 3 (overrides what base just wrote). + liveData := bytes.Repeat([]byte{0xFF}, 4096) + if err := primary.WriteLBA(3, liveData); err != nil { + t.Fatalf("primary live write LBA 3: %v", err) + } + + // Apply live WAL entry to session (WAL lane). + session.ApplyWALEntry(&blockvol.WALEntry{ + LSN: baseLSN + 1, Epoch: 1, Type: blockvol.EntryTypeWrite, + LBA: 3, Length: 4096, Data: liveData, + }) + + // Base blocks 5-9. + for lba := uint64(5); lba < 10; lba++ { + data, _ := primary.ReadLBA(lba, uint32(info.BlockSize)) + session.ApplyBaseBlock(lba, data) + } + + // Try to re-send base block for LBA 3 — should be SKIPPED (bitmap set). + oldData := bytes.Repeat([]byte{0x13}, 4096) // original data at LBA 3 + applied, _ := session.ApplyBaseBlock(3, oldData) + if applied { + t.Fatal("BUG: base block for LBA 3 applied AFTER live WAL write") + } + + // Apply remaining WAL entries to reach target. + for i := uint64(2); i <= 3; i++ { + session.ApplyWALEntry(&blockvol.WALEntry{ + LSN: baseLSN + i, Epoch: 1, Type: blockvol.EntryTypeWrite, + LBA: uint64(i + 5), Length: 4096, + Data: bytes.Repeat([]byte{byte(0xE0 + i)}, 4096), + }) + } + session.MarkBaseComplete(10) + + achievedLSN, completed := session.TryComplete() + if !completed { + t.Fatalf("session did not complete") + } + t.Logf("completed: achievedLSN=%d", achievedLSN) + + // Verify LBA 3 has the LIVE data (0xFF), not old base (0x13). + got, err := replica.ReadLBA(3, 4096) + if err != nil { + t.Fatalf("read LBA 3: %v", err) + } + if got[0] != 0xFF { + t.Fatalf("LBA 3 = 0x%02x, want 0xFF (live write during rebuild must win)", got[0]) + } + t.Log("live write during base copy correctly preserved via bitmap") +} + +// --- Helpers --- + +func createTransportRebuildPair(t *testing.T) (primary, replica *blockvol.BlockVol) { + t.Helper() + opts := blockvol.CreateOptions{ + VolumeSize: 4 * 1024 * 1024, // 4MB + BlockSize: 4096, + WALSize: 1 * 1024 * 1024, + DurabilityMode: blockvol.DurabilitySyncAll, + } + p, err := blockvol.CreateBlockVol(filepath.Join(t.TempDir(), "primary.blk"), opts) + if err != nil { + t.Fatal(err) + } + if err := p.HandleAssignment(1, blockvol.RolePrimary, 30*time.Second); err != nil { + p.Close() + t.Fatal(err) + } + r, err := blockvol.CreateBlockVol(filepath.Join(t.TempDir(), "replica.blk"), opts) + if err != nil { + p.Close() + t.Fatal(err) + } + if err := r.HandleAssignment(1, blockvol.RoleReplica, 30*time.Second); err != nil { + p.Close() + r.Close() + t.Fatal(err) + } + return p, r +}