s3api: no filer failover after the callback has consumed part of a response (#10902)

s3api: no filer failover after fn has consumed part of a response

withFilerClientFailover replays fn verbatim on the next filer, so a filer
that died mid-stream followed by a healthy peer returned success with the
callback's closure-captured accumulator holding the dead filer's prefix
twice; the per-attempt accumulator in listWithRetry could not close this,
because the replay happens inside a single attempt. Track delivery on the
connection handed to fn: once a unary reply or streamed message has reached
the callback, surface the transport error unwrapped instead of failing
over, and let callers replay from a clean slate. A filer that fails before
delivering anything fails over exactly as before.
This commit is contained in:
Chris Lu authored and GitHub committed 2026-08-23 11:30:43 -07:00
1 parent c167af541e
commit cf0dba334c
3 files changed
+188 -4

No files matched your search

+3 -3
View File
@@ -262,9 +262,9 @@ func (f *fakeTxnFiler) ObjectTransaction(ctx context.Context, req *filer_pb.Obje
return &filer_pb.ObjectTransactionResponse{}, nil
}
// startFakeTxnFiler serves impl on a random localhost port and returns the S3-style
// startFakeFiler serves impl on a random localhost port and returns the S3-style
// filer address whose ToGrpcAddress resolves back to that port.
func startFakeTxnFiler(t *testing.T, impl *fakeTxnFiler) pb.ServerAddress {
func startFakeFiler(t *testing.T, impl filer_pb.SeaweedFilerServer) pb.ServerAddress {
t.Helper()
lis, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
@@ -296,7 +296,7 @@ func closedFilerAddress(t *testing.T) pb.ServerAddress {
// so an object write survives a filer pod IP change without an S3 gateway restart.
func TestObjectTxnFailsOverStaleOwner(t *testing.T) {
live := &fakeTxnFiler{}
liveAddr := startFakeTxnFiler(t, live)
liveAddr := startFakeFiler(t, live)
deadOwner := closedFilerAddress(t)
dialOption := grpc.WithTransportCredentials(insecure.NewCredentials())