mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-07 23:07:48 +02:00
* grpc: a non-cancellable context is no evidence of a stale channel shouldInvalidateConnection only invalidates on Canceled/DeadlineExceeded while the context handed to WithGrpcClient is still live, so that an RPC timing out on its own does not close the shared cached ClientConn and cancel every other in-flight RPC on it. context.Background()/TODO never expire, so Err() stays nil forever and that guard always answered "invalidate" - and Background is what almost every caller passes, the S3 gateway included. One S3 request whose RPC rode an abandoned HTTP request context therefore closed the shared filer connection, and every multipart part in flight died with "the client connection is closing", surfacing to the client as 400 InvalidRequest. Only a cancellable context bounds an RPC attempt, so require one before reading it. A genuinely stale channel (a peer restart behind a stable L4 endpoint) surfaces as Unavailable, which invalidates on its own branch. * grpc: a bystander of a connection teardown is not a stale-channel witness gRPC raises ErrClientConnClosing locally, before an RPC reaches the wire, when this process has already closed the ClientConn. Every caller that touches a channel during another goroutine's teardown gets it, so reading it as a stale-channel signal lets one teardown re-arm itself across the whole herd of callers it just cancelled. The cached-connection version check keeps those callers from closing a replacement channel, but the streaming path invalidates by address alone and has no such guard. * grpc: end a stream without dropping the peer connection under it A streaming caller gets its own ClientConn, but on any error it also drops the cached non-streaming ClientConn every request handler shares with that peer, to recover a peer restart hidden behind a stable L4 endpoint. Any error includes the ordinary ones: a metadata subscription that reached its stop point, a follow callback that refused an event, a caller that gave up. The S3 gateway follows filer metadata on such a stream and reconnects forever, so each ordinary end of it cancelled every S3 request in flight against the filer. Drop the shared channel only for errors that say the peer went away, which is what invalidation is for. * test: close the connections the cascade tests leave cached Each test swaps in a fresh connection cache and restores the previous one, dropping its own entries without closing them, so the ClientConn's transport and reconnect goroutines outlive the fake filer they dialed. * grpc: say why ErrClientConnClosing's deprecation notice does not apply It points at codes.Canceled, which is the code this function exists to disambiguate. Only the message distinguishes a teardown a caller merely walked into, so the sentinel stays.
148 lines
5.0 KiB
Go
148 lines
5.0 KiB
Go
package pb
|
|
|
|
import (
|
|
"context"
|
|
"net"
|
|
"sync"
|
|
"testing"
|
|
|
|
"google.golang.org/grpc"
|
|
"google.golang.org/grpc/credentials/insecure"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
)
|
|
|
|
// cascadeFilerServer answers a "slow" lookup only once its caller is known to be
|
|
// in flight, so a second caller can fail while the first still holds the shared
|
|
// cached ClientConn.
|
|
type cascadeFilerServer struct {
|
|
filer_pb.UnimplementedSeaweedFilerServer
|
|
inFlight chan struct{}
|
|
release chan struct{}
|
|
closeOnce sync.Once
|
|
}
|
|
|
|
func (s *cascadeFilerServer) LookupDirectoryEntry(ctx context.Context, req *filer_pb.LookupDirectoryEntryRequest) (*filer_pb.LookupDirectoryEntryResponse, error) {
|
|
if req.Name == "slow" {
|
|
s.closeOnce.Do(func() { close(s.inFlight) })
|
|
select {
|
|
case <-s.release:
|
|
case <-ctx.Done():
|
|
return nil, ctx.Err()
|
|
}
|
|
}
|
|
return &filer_pb.LookupDirectoryEntryResponse{}, nil
|
|
}
|
|
|
|
// startCascadeFiler serves a fake filer and resets the process-wide connection
|
|
// cache so the test owns the shared ClientConn under exercise.
|
|
func startCascadeFiler(t *testing.T) (*cascadeFilerServer, string, grpc.DialOption) {
|
|
t.Helper()
|
|
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
|
if err != nil {
|
|
t.Fatalf("listen: %v", err)
|
|
}
|
|
fake := &cascadeFilerServer{inFlight: make(chan struct{}), release: make(chan struct{})}
|
|
server := grpc.NewServer()
|
|
filer_pb.RegisterSeaweedFilerServer(server, fake)
|
|
go server.Serve(listener)
|
|
|
|
grpcClientsLock.Lock()
|
|
previous := grpcClients
|
|
grpcClients = make(map[string]*versionedGrpcClient)
|
|
grpcClientsLock.Unlock()
|
|
t.Cleanup(func() {
|
|
grpcClientsLock.Lock()
|
|
cached := grpcClients
|
|
grpcClients = previous
|
|
grpcClientsLock.Unlock()
|
|
for _, connection := range cached {
|
|
connection.Close()
|
|
}
|
|
server.Stop()
|
|
})
|
|
|
|
return fake, listener.Addr().String(), grpc.WithTransportCredentials(insecure.NewCredentials())
|
|
}
|
|
|
|
// TestWithGrpcClient_AbandonedRequestKeepsSharedConnection reproduces
|
|
// seaweedfs#10947 end to end: an S3-style caller that hands WithGrpcClient a
|
|
// context.Background() while its RPC rides the (now abandoned) HTTP request
|
|
// context used to close the shared filer ClientConn, killing every concurrent
|
|
// upload on it with "the client connection is closing".
|
|
func TestWithGrpcClient_AbandonedRequestKeepsSharedConnection(t *testing.T) {
|
|
fake, address, dialOption := startCascadeFiler(t)
|
|
|
|
var concurrentErr error
|
|
var wg sync.WaitGroup
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
concurrentErr = WithGrpcClient(context.Background(), false, 0, func(connection *grpc.ClientConn) error {
|
|
_, err := filer_pb.NewSeaweedFilerClient(connection).LookupDirectoryEntry(context.Background(),
|
|
&filer_pb.LookupDirectoryEntryRequest{Directory: "/buckets/b", Name: "slow"})
|
|
return err
|
|
}, address, false, dialOption)
|
|
}()
|
|
<-fake.inFlight
|
|
|
|
abandoned, cancel := context.WithCancel(context.Background())
|
|
cancel()
|
|
if err := WithGrpcClient(context.Background(), false, 0, func(connection *grpc.ClientConn) error {
|
|
_, err := filer_pb.NewSeaweedFilerClient(connection).LookupDirectoryEntry(abandoned,
|
|
&filer_pb.LookupDirectoryEntryRequest{Directory: "/buckets/b", Name: "fast"})
|
|
return err
|
|
}, address, false, dialOption); err == nil {
|
|
t.Fatal("the abandoned request should have failed")
|
|
}
|
|
|
|
close(fake.release)
|
|
wg.Wait()
|
|
if concurrentErr != nil {
|
|
t.Fatalf("concurrent caller must survive an unrelated abandoned request: %v", concurrentErr)
|
|
}
|
|
}
|
|
|
|
func (s *cascadeFilerServer) SubscribeMetadata(req *filer_pb.SubscribeMetadataRequest, stream filer_pb.SeaweedFiler_SubscribeMetadataServer) error {
|
|
return nil
|
|
}
|
|
|
|
// TestWithGrpcClient_EndedStreamKeepsSharedConnection covers the other half of
|
|
// seaweedfs#10947: the S3 gateway follows filer metadata on a long-lived
|
|
// SubscribeMetadata stream and reconnects forever. Every ordinary end of that
|
|
// stream used to drop the shared filer ClientConn the request handlers use,
|
|
// firing the same burst of "the client connection is closing" failures.
|
|
func TestWithGrpcClient_EndedStreamKeepsSharedConnection(t *testing.T) {
|
|
fake, address, dialOption := startCascadeFiler(t)
|
|
|
|
var concurrentErr error
|
|
var wg sync.WaitGroup
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
concurrentErr = WithGrpcClient(context.Background(), false, 0, func(connection *grpc.ClientConn) error {
|
|
_, err := filer_pb.NewSeaweedFilerClient(connection).LookupDirectoryEntry(context.Background(),
|
|
&filer_pb.LookupDirectoryEntryRequest{Directory: "/buckets/b", Name: "slow"})
|
|
return err
|
|
}, address, false, dialOption)
|
|
}()
|
|
<-fake.inFlight
|
|
|
|
if err := WithGrpcClient(context.Background(), true, 0, func(connection *grpc.ClientConn) error {
|
|
stream, err := filer_pb.NewSeaweedFilerClient(connection).SubscribeMetadata(context.Background(), &filer_pb.SubscribeMetadataRequest{})
|
|
if err != nil {
|
|
return err
|
|
}
|
|
_, err = stream.Recv()
|
|
return err
|
|
}, address, false, dialOption); err == nil {
|
|
t.Fatal("the ended stream should have reported io.EOF")
|
|
}
|
|
|
|
close(fake.release)
|
|
wg.Wait()
|
|
if concurrentErr != nil {
|
|
t.Fatalf("concurrent caller must survive an unrelated stream ending: %v", concurrentErr)
|
|
}
|
|
}
|