mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-06 06:22:05 +02:00
* filer sink: keep the gRPC status inside wrapped errors
%v stringifies the status, so a peer teardown reported as Canceled ("the
client connection is closing") reached IsTransientError as plain text and
matched nothing: the sync job failed on the first attempt and pinned the
offset. %w keeps the status reachable, so the retry runs on a fresh
connection once the target is back.
* pb: let a consumer drop the metadata stream to force a resubscribe
A MetadataProcessor job that exhausts its retries pins the processed
watermark so the event replays on the next subscribe — but nothing on the
source stream notices a target-side failure, so the replay waited for an
unrelated reconnect or a restart. The new Resubscribe channel cancels the
stream's context; the Recv loop answers it with ErrResubscribe so the
caller's retry loop resubscribes from GetResumeTsNs and replays the pinned
events in order.
* pb: stop the event retry loop once the stream context is done
RetryUntil ignores context, so a subscriber parked on a failing offset
write would keep retrying past a resubscribe signal until the sink came
back. Stop retrying when the stream is being dropped so the resubscribe
takes effect promptly.
* filer.sync: signal resubscribe when a job failure pins the offset
A job that exhausts its in-job retries leaves the event pinned behind oldestFailedTsNs, replayable only on a reconnect. Closing resubscribeCh on the first recorded failure lets the metadata follower drop the stream so the reconnect replays the pinned events instead of waiting for a process restart (#11572).
* filer.sync: wire the resubscribe signal into the follow options
filer.sync, filer.remote.sync, and the remote gateway bucket sync all run their subscription inside an outer retry loop, so ErrResubscribe resurfaces as a resubscribe from the persisted watermark.
* filer.sync: wait for in-flight jobs before signaling resubscribe
* remote sync: never resume past the saved offset when -timeAgo is set
* filer.sync: drop events that arrive after the drain signals resubscribe
* pb: interrupt the event retry backoff when the stream context ends
* filer.sync: stop admitting once a failure pins, and count jobs per timestamp
A pinned watermark only released once the processor went fully quiet, so a busy stream could starve the resubscribe — the failed event would wait for an unrelated reconnect anyway, the wait this mechanism exists to remove. The processor now latches stopped when a job fails: admission drops new events (they replay from the pinned watermark after the reconnect), a broadcast releases blocked waiters, and the resubscribe signals as soon as the jobs already in flight drain. A redelivery of an event still in the failure ledger may still run so its success shrinks the replay, but nothing starts once the signal has fired, or it would race the replay it asked for.
Dropped events no longer inflate the received counters — an event counts only once admitted, and the replay's own admission counts it.
While here: activeJobs keyed by TsNs collapsed events sharing a timestamp, so one completion could empty the map while a same-ts sibling was still running — letting the drain gate and the watermark outrun it. Jobs are now counted per timestamp, and the drain and lazy heap cleanup go through the counts.
283 lines
9.5 KiB
Go
283 lines
9.5 KiB
Go
package pb
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
"google.golang.org/grpc"
|
|
)
|
|
|
|
type EventErrorType int
|
|
|
|
const (
|
|
TrivialOnError EventErrorType = iota
|
|
FatalOnError
|
|
RetryForeverOnError
|
|
DontLogError
|
|
)
|
|
|
|
// MetadataFollowOption is used to control the behavior of the metadata following
|
|
// process. Part of it is used as a cursor to resume the following process.
|
|
type MetadataFollowOption struct {
|
|
ClientName string
|
|
ClientId int32
|
|
ClientEpoch int32
|
|
SelfSignature int32
|
|
PathPrefix string
|
|
AdditionalPathPrefixes []string
|
|
DirectoriesToWatch []string
|
|
StartTsNs int64
|
|
StopTsNs int64
|
|
EventErrorType EventErrorType
|
|
// LogFileReaderFn, when non-nil, enables metadata chunks mode:
|
|
// the server sends log file chunk fids instead of streaming events,
|
|
// and the client reads directly from volume servers.
|
|
LogFileReaderFn LogFileReaderFn
|
|
// OnIdleHeartbeat, when non-nil, opts in to idle heartbeats: while the
|
|
// subscriber is caught up the server periodically sends an empty response
|
|
// carrying the current time, and this is called with that timestamp. It is
|
|
// a freshness signal only and does not advance StartTsNs, so the resume
|
|
// checkpoint stays on the last real event.
|
|
OnIdleHeartbeat func(tsNs int64)
|
|
// GetResumeTsNs, when non-nil, supplies the reconnect position instead of
|
|
// StartTsNs. It is read on every subscribe, so a callback can return the
|
|
// durably processed watermark while StartTsNs keeps tracking positions the
|
|
// stream has merely seen.
|
|
GetResumeTsNs func() int64
|
|
// Resubscribe, when closed, drops the stream with ErrResubscribe so the
|
|
// caller's reconnect loop resubscribes from GetResumeTsNs and replays the
|
|
// events still pinning the processed watermark. Target-side job failures
|
|
// never surface on this stream, so without it a pinned event waits for an
|
|
// unrelated source-stream reconnect (or a restart) to be replayed.
|
|
Resubscribe <-chan struct{}
|
|
}
|
|
|
|
// ErrResubscribe ends a metadata follow when the consumer asked for the
|
|
// stream to drop so the reconnect replays what the resume watermark pins.
|
|
var ErrResubscribe = errors.New("resubscribe to replay events behind the failed offset")
|
|
|
|
type ProcessMetadataFunc func(resp *filer_pb.SubscribeMetadataResponse) error
|
|
|
|
func FollowMetadata(filerAddress ServerAddress, grpcDialOption grpc.DialOption, option *MetadataFollowOption, processEventFn ProcessMetadataFunc) error {
|
|
|
|
err := WithFilerClient(true, option.SelfSignature, filerAddress, grpcDialOption, makeSubscribeMetadataFunc(option, processEventFn))
|
|
if err != nil {
|
|
return fmt.Errorf("subscribing filer meta change: %w", err)
|
|
}
|
|
return err
|
|
}
|
|
|
|
func WithFilerClientFollowMetadata(filerClient filer_pb.FilerClient, option *MetadataFollowOption, processEventFn ProcessMetadataFunc) error {
|
|
|
|
err := filerClient.WithFilerClient(true, makeSubscribeMetadataFunc(option, processEventFn))
|
|
if err != nil {
|
|
return fmt.Errorf("subscribing filer meta change: %w", err)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func makeSubscribeMetadataFunc(option *MetadataFollowOption, processEventFn ProcessMetadataFunc) func(client filer_pb.SeaweedFilerClient) error {
|
|
return func(client filer_pb.SeaweedFilerClient) error {
|
|
ctx, cancel := context.WithCancel(context.Background())
|
|
defer cancel()
|
|
sinceNs := option.StartTsNs
|
|
if option.GetResumeTsNs != nil {
|
|
sinceNs = option.GetResumeTsNs()
|
|
}
|
|
stream, err := client.SubscribeMetadata(ctx, &filer_pb.SubscribeMetadataRequest{
|
|
ClientName: option.ClientName,
|
|
PathPrefix: option.PathPrefix,
|
|
PathPrefixes: option.AdditionalPathPrefixes,
|
|
Directories: option.DirectoriesToWatch,
|
|
SinceNs: sinceNs,
|
|
Signature: option.SelfSignature,
|
|
ClientId: option.ClientId,
|
|
ClientEpoch: option.ClientEpoch,
|
|
UntilNs: option.StopTsNs,
|
|
ClientSupportsBatching: true,
|
|
ClientSupportsMetadataChunks: option.LogFileReaderFn != nil,
|
|
ClientSupportsIdleHeartbeat: option.OnIdleHeartbeat != nil,
|
|
})
|
|
if err != nil {
|
|
return fmt.Errorf("subscribe: %w", err)
|
|
}
|
|
|
|
if option.Resubscribe != nil {
|
|
go func() {
|
|
select {
|
|
case <-option.Resubscribe:
|
|
cancel()
|
|
case <-ctx.Done():
|
|
}
|
|
}()
|
|
}
|
|
|
|
handleErr := func(resp *filer_pb.SubscribeMetadataResponse, err error) {
|
|
switch option.EventErrorType {
|
|
case TrivialOnError:
|
|
glog.Errorf("process %v: %v", resp, err)
|
|
case FatalOnError:
|
|
glog.Fatalf("process %v: %v", resp, err)
|
|
case RetryForeverOnError:
|
|
waitTime := time.Second
|
|
for ctx.Err() == nil {
|
|
if err := processEventFn(resp); err == nil {
|
|
break
|
|
} else {
|
|
glog.Errorf("process %v: %v", resp, err)
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
case <-time.After(waitTime):
|
|
}
|
|
if waitTime < util.RetryWaitTime {
|
|
waitTime += waitTime / 2
|
|
}
|
|
}
|
|
case DontLogError:
|
|
// pass
|
|
default:
|
|
glog.Errorf("process %v: %v", resp, err)
|
|
}
|
|
}
|
|
|
|
// handleOneEvent processes a single response, whether it arrived as the
|
|
// batch envelope or inside resp.Events. Freshness signals carry a timestamp
|
|
// but no entry: the idle heartbeat (nil EventNotification) and the
|
|
// MaxUnsyncedEvents marker (empty one). Both route to OnIdleHeartbeat,
|
|
// else the marker pins sync_offset to the stale processed watermark and a
|
|
// nil heartbeat folded into a batch nil-derefs in the sync job path.
|
|
handleOneEvent := func(resp *filer_pb.SubscribeMetadataResponse) {
|
|
if resp.EventNotification == nil || filer_pb.IsEmpty(resp) {
|
|
if resp.TsNs > 0 && option.OnIdleHeartbeat != nil {
|
|
option.OnIdleHeartbeat(resp.TsNs)
|
|
}
|
|
// The marker advances the resume cursor past the filtered range; the
|
|
// heartbeat leaves it put so a restart cannot outrun a straggler. A
|
|
// consumer with a resume callback keeps its cursor in its processed
|
|
// watermark, so it sees the marker instead.
|
|
if resp.EventNotification != nil && resp.TsNs > 0 {
|
|
if option.GetResumeTsNs != nil {
|
|
if err := processEventFn(resp); err != nil {
|
|
handleErr(resp, err)
|
|
}
|
|
} else {
|
|
option.StartTsNs = resp.TsNs
|
|
}
|
|
}
|
|
return
|
|
}
|
|
if err := processEventFn(resp); err != nil {
|
|
handleErr(resp, err)
|
|
// RetryForeverOnError only returns once the event was handled;
|
|
// other modes leave it failed and the cursor stays behind it.
|
|
if option.EventErrorType != RetryForeverOnError {
|
|
return
|
|
}
|
|
}
|
|
if option.GetResumeTsNs == nil {
|
|
option.StartTsNs = resp.TsNs
|
|
}
|
|
}
|
|
|
|
var pendingRefs []*filer_pb.LogFileChunkRef
|
|
|
|
// drainPendingRefs reads whatever chunk refs have accumulated. The server
|
|
// sends refs in their own responses, so they can only be read once
|
|
// something tells us the run of refs has ended — either a normal event, or
|
|
// the end of the stream. A bounded subscription (StopTsNs set, range
|
|
// already in the past) may get nothing but refs and then EOF, so draining
|
|
// only on the former silently returns no events at all.
|
|
drainPendingRefs := func() error {
|
|
if len(pendingRefs) == 0 || option.LogFileReaderFn == nil {
|
|
return nil
|
|
}
|
|
readFromNs := option.StartTsNs
|
|
if option.GetResumeTsNs != nil {
|
|
// The resume cursor lives in the processed watermark, so a
|
|
// resubscribed ref replay must not be filtered by positions the
|
|
// previous stream had only seen.
|
|
readFromNs = sinceNs
|
|
}
|
|
lastTs, readErr := ReadLogFileRefs(pendingRefs, option.LogFileReaderFn,
|
|
readFromNs, option.StopTsNs,
|
|
PathFilter{
|
|
PathPrefix: option.PathPrefix,
|
|
AdditionalPathPrefixes: option.AdditionalPathPrefixes,
|
|
DirectoriesToWatch: option.DirectoriesToWatch,
|
|
},
|
|
processEventFn)
|
|
if readErr != nil {
|
|
return fmt.Errorf("%w: %w", ErrLogFileRead, readErr)
|
|
}
|
|
if lastTs > 0 && option.GetResumeTsNs == nil {
|
|
option.StartTsNs = lastTs
|
|
}
|
|
pendingRefs = nil
|
|
return nil
|
|
}
|
|
|
|
for {
|
|
resp, listenErr := stream.Recv()
|
|
if listenErr == io.EOF {
|
|
return drainPendingRefs()
|
|
}
|
|
if listenErr != nil {
|
|
select {
|
|
case <-option.Resubscribe:
|
|
return ErrResubscribe
|
|
default:
|
|
}
|
|
return listenErr
|
|
}
|
|
|
|
// Accumulate log file chunk references (metadata chunks mode)
|
|
if len(resp.LogFileRefs) > 0 {
|
|
pendingRefs = append(pendingRefs, resp.LogFileRefs...)
|
|
continue
|
|
}
|
|
|
|
// Process accumulated refs before handling normal events (transition point)
|
|
if err := drainPendingRefs(); err != nil {
|
|
return err
|
|
}
|
|
|
|
// Process the envelope event (top-level fields) and any batched tail.
|
|
// The server folds a backlog into one response: the first event lives in
|
|
// the top-level fields, the rest in resp.Events. Either slot can hold a
|
|
// freshness signal, so both go through the same handler.
|
|
handleOneEvent(resp)
|
|
for _, batchedEvent := range resp.Events {
|
|
handleOneEvent(batchedEvent)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
func AddOffsetFunc(processEventFn ProcessMetadataFunc, offsetInterval time.Duration, offsetFunc func(counter int64, offset int64) error) ProcessMetadataFunc {
|
|
var counter int64
|
|
var lastWriteTime = time.Now()
|
|
return func(resp *filer_pb.SubscribeMetadataResponse) error {
|
|
if err := processEventFn(resp); err != nil {
|
|
return err
|
|
}
|
|
counter++
|
|
if lastWriteTime.Add(offsetInterval).Before(time.Now()) {
|
|
lastWriteTime = time.Now()
|
|
if err := offsetFunc(counter, resp.TsNs); err != nil {
|
|
return err
|
|
}
|
|
counter = 0
|
|
}
|
|
return nil
|
|
}
|
|
|
|
}
|