Files
seaweedfs/weed/pb/grpc_client_cascade_test.go
T
Chris Lu 50b388771a s3: stop one abandoned request from cancelling every concurrent upload (#10948)
* grpc: a non-cancellable context is no evidence of a stale channel

shouldInvalidateConnection only invalidates on Canceled/DeadlineExceeded
while the context handed to WithGrpcClient is still live, so that an RPC
timing out on its own does not close the shared cached ClientConn and
cancel every other in-flight RPC on it. context.Background()/TODO never
expire, so Err() stays nil forever and that guard always answered
"invalidate" - and Background is what almost every caller passes, the S3
gateway included.

One S3 request whose RPC rode an abandoned HTTP request context therefore
closed the shared filer connection, and every multipart part in flight
died with "the client connection is closing", surfacing to the client as
400 InvalidRequest.

Only a cancellable context bounds an RPC attempt, so require one before
reading it. A genuinely stale channel (a peer restart behind a stable L4
endpoint) surfaces as Unavailable, which invalidates on its own branch.

* grpc: a bystander of a connection teardown is not a stale-channel witness

gRPC raises ErrClientConnClosing locally, before an RPC reaches the wire,
when this process has already closed the ClientConn. Every caller that
touches a channel during another goroutine's teardown gets it, so reading
it as a stale-channel signal lets one teardown re-arm itself across the
whole herd of callers it just cancelled.

The cached-connection version check keeps those callers from closing a
replacement channel, but the streaming path invalidates by address alone
and has no such guard.

* grpc: end a stream without dropping the peer connection under it

A streaming caller gets its own ClientConn, but on any error it also drops
the cached non-streaming ClientConn every request handler shares with that
peer, to recover a peer restart hidden behind a stable L4 endpoint. Any
error includes the ordinary ones: a metadata subscription that reached its
stop point, a follow callback that refused an event, a caller that gave up.

The S3 gateway follows filer metadata on such a stream and reconnects
forever, so each ordinary end of it cancelled every S3 request in flight
against the filer. Drop the shared channel only for errors that say the
peer went away, which is what invalidation is for.

* test: close the connections the cascade tests leave cached

Each test swaps in a fresh connection cache and restores the previous one,
dropping its own entries without closing them, so the ClientConn's
transport and reconnect goroutines outlive the fake filer they dialed.

* grpc: say why ErrClientConnClosing's deprecation notice does not apply

It points at codes.Canceled, which is the code this function exists to
disambiguate. Only the message distinguishes a teardown a caller merely
walked into, so the sentinel stays.
2026-08-25 10:15:47 -07:00

148 lines
5.0 KiB
Go

package pb
import (
"context"
"net"
"sync"
"testing"
"google.golang.org/grpc"
"google.golang.org/grpc/credentials/insecure"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
)
// cascadeFilerServer answers a "slow" lookup only once its caller is known to be
// in flight, so a second caller can fail while the first still holds the shared
// cached ClientConn.
type cascadeFilerServer struct {
filer_pb.UnimplementedSeaweedFilerServer
inFlight chan struct{}
release chan struct{}
closeOnce sync.Once
}
func (s *cascadeFilerServer) LookupDirectoryEntry(ctx context.Context, req *filer_pb.LookupDirectoryEntryRequest) (*filer_pb.LookupDirectoryEntryResponse, error) {
if req.Name == "slow" {
s.closeOnce.Do(func() { close(s.inFlight) })
select {
case <-s.release:
case <-ctx.Done():
return nil, ctx.Err()
}
}
return &filer_pb.LookupDirectoryEntryResponse{}, nil
}
// startCascadeFiler serves a fake filer and resets the process-wide connection
// cache so the test owns the shared ClientConn under exercise.
func startCascadeFiler(t *testing.T) (*cascadeFilerServer, string, grpc.DialOption) {
t.Helper()
listener, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
fake := &cascadeFilerServer{inFlight: make(chan struct{}), release: make(chan struct{})}
server := grpc.NewServer()
filer_pb.RegisterSeaweedFilerServer(server, fake)
go server.Serve(listener)
grpcClientsLock.Lock()
previous := grpcClients
grpcClients = make(map[string]*versionedGrpcClient)
grpcClientsLock.Unlock()
t.Cleanup(func() {
grpcClientsLock.Lock()
cached := grpcClients
grpcClients = previous
grpcClientsLock.Unlock()
for _, connection := range cached {
connection.Close()
}
server.Stop()
})
return fake, listener.Addr().String(), grpc.WithTransportCredentials(insecure.NewCredentials())
}
// TestWithGrpcClient_AbandonedRequestKeepsSharedConnection reproduces
// seaweedfs#10947 end to end: an S3-style caller that hands WithGrpcClient a
// context.Background() while its RPC rides the (now abandoned) HTTP request
// context used to close the shared filer ClientConn, killing every concurrent
// upload on it with "the client connection is closing".
func TestWithGrpcClient_AbandonedRequestKeepsSharedConnection(t *testing.T) {
fake, address, dialOption := startCascadeFiler(t)
var concurrentErr error
var wg sync.WaitGroup
wg.Add(1)
go func() {
defer wg.Done()
concurrentErr = WithGrpcClient(context.Background(), false, 0, func(connection *grpc.ClientConn) error {
_, err := filer_pb.NewSeaweedFilerClient(connection).LookupDirectoryEntry(context.Background(),
&filer_pb.LookupDirectoryEntryRequest{Directory: "/buckets/b", Name: "slow"})
return err
}, address, false, dialOption)
}()
<-fake.inFlight
abandoned, cancel := context.WithCancel(context.Background())
cancel()
if err := WithGrpcClient(context.Background(), false, 0, func(connection *grpc.ClientConn) error {
_, err := filer_pb.NewSeaweedFilerClient(connection).LookupDirectoryEntry(abandoned,
&filer_pb.LookupDirectoryEntryRequest{Directory: "/buckets/b", Name: "fast"})
return err
}, address, false, dialOption); err == nil {
t.Fatal("the abandoned request should have failed")
}
close(fake.release)
wg.Wait()
if concurrentErr != nil {
t.Fatalf("concurrent caller must survive an unrelated abandoned request: %v", concurrentErr)
}
}
func (s *cascadeFilerServer) SubscribeMetadata(req *filer_pb.SubscribeMetadataRequest, stream filer_pb.SeaweedFiler_SubscribeMetadataServer) error {
return nil
}
// TestWithGrpcClient_EndedStreamKeepsSharedConnection covers the other half of
// seaweedfs#10947: the S3 gateway follows filer metadata on a long-lived
// SubscribeMetadata stream and reconnects forever. Every ordinary end of that
// stream used to drop the shared filer ClientConn the request handlers use,
// firing the same burst of "the client connection is closing" failures.
func TestWithGrpcClient_EndedStreamKeepsSharedConnection(t *testing.T) {
fake, address, dialOption := startCascadeFiler(t)
var concurrentErr error
var wg sync.WaitGroup
wg.Add(1)
go func() {
defer wg.Done()
concurrentErr = WithGrpcClient(context.Background(), false, 0, func(connection *grpc.ClientConn) error {
_, err := filer_pb.NewSeaweedFilerClient(connection).LookupDirectoryEntry(context.Background(),
&filer_pb.LookupDirectoryEntryRequest{Directory: "/buckets/b", Name: "slow"})
return err
}, address, false, dialOption)
}()
<-fake.inFlight
if err := WithGrpcClient(context.Background(), true, 0, func(connection *grpc.ClientConn) error {
stream, err := filer_pb.NewSeaweedFilerClient(connection).SubscribeMetadata(context.Background(), &filer_pb.SubscribeMetadataRequest{})
if err != nil {
return err
}
_, err = stream.Recv()
return err
}, address, false, dialOption); err == nil {
t.Fatal("the ended stream should have reported io.EOF")
}
close(fake.release)
wg.Wait()
if concurrentErr != nil {
t.Fatalf("concurrent caller must survive an unrelated stream ending: %v", concurrentErr)
}
}