mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-20 13:30:46 +02:00
* s3: route object reads to the key's owner filer Writes already route by key to the owner filer on the lock ring, where the entry is created. Reads went to the gateway's local filer and treated its NotFound as authoritative, so a GET on one gateway could miss an object another gateway had just written until the filers' metadata replication caught up. Resolve an object's entry from the key's owner first, failing over to the gateway's filer set only on transport errors. An owner NotFound stays authoritative: no fan-out across filers, and no resurrecting a peer's not-yet-replicated tombstone, so a delete routed to the owner is visible at once and a genuine miss costs one lookup. Keys owned by the local filer are unchanged. Objects written through the non-routed lock path land on a gateway's local filer, so they can still read as absent on the owner until they replicate. withFilerClientFailover takes a preferred start filer; the object-entry reads pass the owner, every other caller passes "" and keeps the current-filer fast path. * s3: consult the prior owner on a rebalance-window read miss Owner-first reads route a key to its current ring owner. When a filer joins, ~1/N of keys reassign to it, and the new owner may not have replicated a just-moved key yet, so an owner NotFound would surface a transient 404 for an object that already exists elsewhere. Retain the previous ring on the gateway's LockClient for a cooling-off window (PriorOwnerForKey, mirroring the master's LockRing.PriorOwner) and, on the owner's NotFound, probe the key's previous owner once before treating the miss as final. The probe is scoped to keys whose ownership actually moved and only within the window, so steady-state reads are untouched. This trades the transient scale-up 404 for a transient stale read if a delete routed to the new owner races the same window — the same authoritative-NotFound tradeoff, narrowed to the rebalance. * s3: try healthy filers before unhealthy ones on failover The candidate list probed its first entry (usually the current filer) unconditionally, so a health-flagged current filer cost a transport timeout on every ordinary call before failover reached a replica. Partition candidates into healthy and unhealthy, keep priority within each, and fall back to unhealthy ones only when all healthy ones fail. * reduce comments on the routed read and lock client paths * s3: skip a recently-unreachable owner on route-by-key reads The gateway's filer health tracking no-ops for an owner outside the static -filer list, so during a sustained owner outage every route-by-key read re-dials the dead owner before failing over. Flag an owner whose owner-first read hit a transport error and skip it (read local-first) for a short TTL, so reads pay one dead dial per TTL instead of one per request; the flag expires so owner-first reads resume once the owner or the ring recovers. * s3: always try the preferred owner first, health-order only the rest The healthy/unhealthy partition also demoted a health-flagged preferred owner behind healthy replicas, so a replica's authoritative NotFound could mask a write that had only reached the owner — the read-after-write race this routing exists to close. Pull preferred out of the partition and keep it first; the recently-unreachable gate already steers reads away from a genuinely dead owner.
153 lines
5.1 KiB
Go
153 lines
5.1 KiB
Go
package s3api
|
|
|
|
import (
|
|
"context"
|
|
"encoding/base64"
|
|
"errors"
|
|
"fmt"
|
|
"net/http"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/glog"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
|
|
"google.golang.org/grpc"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
)
|
|
|
|
var _ = filer_pb.FilerClient(&S3ApiServer{})
|
|
|
|
func (s3a *S3ApiServer) WithFilerClient(streamingMode bool, fn func(filer_pb.SeaweedFilerClient) error) error {
|
|
// Use filerClient for proper connection management and failover
|
|
if s3a.filerClient != nil {
|
|
return s3a.withFilerClientFailover("", streamingMode, fn)
|
|
}
|
|
|
|
// Fallback to direct connection if filerClient not initialized
|
|
// This should only happen during initialization or testing
|
|
return pb.WithGrpcClient(context.Background(), streamingMode, s3a.randomClientId, func(grpcConnection *grpc.ClientConn) error {
|
|
client := filer_pb.NewSeaweedFilerClient(grpcConnection)
|
|
return fn(client)
|
|
}, s3a.getFilerAddress().ToGrpcAddress(), false, s3a.option.GrpcDialOption)
|
|
|
|
}
|
|
|
|
// withFilerClientFailover runs fn against preferred (if set) first, then the
|
|
// remaining filers (current, then the rest) with healthy ones before unhealthy.
|
|
// Failover is for transport errors only: a reached filer's ErrNotFound is
|
|
// authoritative (no
|
|
// fan-out, no resurrecting a peer's not-yet-replicated tombstone). preferred lets a
|
|
// caller route to a key's ring owner for read-after-write; it may be a filer outside
|
|
// the static list (the bookkeeping no-ops for untracked addresses). A failover
|
|
// updates the current filer; a preferred read does not, as its owner is per-key.
|
|
func (s3a *S3ApiServer) withFilerClientFailover(preferred pb.ServerAddress, streamingMode bool, fn func(filer_pb.SeaweedFilerClient) error) error {
|
|
currentFiler := s3a.filerClient.GetCurrentFiler()
|
|
|
|
candidates := make([]pb.ServerAddress, 0, 2+len(s3a.option.Filers))
|
|
seen := make(map[pb.ServerAddress]bool)
|
|
addCandidate := func(filer pb.ServerAddress) {
|
|
if filer == "" || seen[filer] {
|
|
return
|
|
}
|
|
seen[filer] = true
|
|
candidates = append(candidates, filer)
|
|
}
|
|
addCandidate(preferred)
|
|
addCandidate(currentFiler)
|
|
for _, filer := range s3a.filerClient.GetAllFilers() {
|
|
addCandidate(filer)
|
|
}
|
|
|
|
// An explicit preferred owner is tried first even if health-flagged: demoting it
|
|
// behind a healthy replica would let that replica's authoritative NotFound mask a
|
|
// write that has only reached the owner. Health-order the rest, unhealthy ones
|
|
// last so a request still progresses when all are flagged.
|
|
var healthy, unhealthy []pb.ServerAddress
|
|
for _, filer := range candidates {
|
|
if filer == preferred {
|
|
continue
|
|
}
|
|
if s3a.filerClient.ShouldSkipUnhealthyFiler(filer) {
|
|
unhealthy = append(unhealthy, filer)
|
|
} else {
|
|
healthy = append(healthy, filer)
|
|
}
|
|
}
|
|
ordered := make([]pb.ServerAddress, 0, len(candidates))
|
|
if preferred != "" {
|
|
ordered = append(ordered, preferred)
|
|
}
|
|
ordered = append(ordered, healthy...)
|
|
ordered = append(ordered, unhealthy...)
|
|
|
|
var lastErr error
|
|
for _, filer := range ordered {
|
|
err := pb.WithGrpcClient(context.Background(), streamingMode, s3a.randomClientId, func(grpcConnection *grpc.ClientConn) error {
|
|
return fn(filer_pb.NewSeaweedFilerClient(grpcConnection))
|
|
}, filer.ToGrpcAddress(), false, s3a.option.GrpcDialOption)
|
|
|
|
if err == nil {
|
|
s3a.filerClient.RecordFilerSuccess(filer)
|
|
if filer != currentFiler && filer != preferred {
|
|
s3a.filerClient.SetCurrentFiler(filer)
|
|
glog.V(1).Infof("WithFilerClient: failover from %s to %s succeeded", currentFiler, filer)
|
|
}
|
|
return nil
|
|
}
|
|
if errors.Is(err, filer_pb.ErrNotFound) {
|
|
return err
|
|
}
|
|
|
|
s3a.filerClient.RecordFilerFailure(filer)
|
|
// A preferred owner is often outside the static filer list, where the health
|
|
// tracking above no-ops; flag it so route-by-key reads skip it briefly.
|
|
if filer == preferred {
|
|
s3a.markOwnerUnreachable(filer)
|
|
}
|
|
glog.V(2).Infof("WithFilerClient: filer %s failed: %v", filer, err)
|
|
lastErr = err
|
|
}
|
|
|
|
if lastErr == nil {
|
|
lastErr = fmt.Errorf("no filer available")
|
|
}
|
|
return fmt.Errorf("all filers failed, last error: %w", lastErr)
|
|
}
|
|
|
|
func (s3a *S3ApiServer) AdjustedUrl(location *filer_pb.Location) string {
|
|
return location.Url
|
|
}
|
|
|
|
func (s3a *S3ApiServer) GetDataCenter() string {
|
|
return s3a.option.DataCenter
|
|
}
|
|
|
|
func writeSuccessResponseXML(w http.ResponseWriter, r *http.Request, response interface{}) {
|
|
s3err.WriteXMLResponse(w, r, http.StatusOK, response)
|
|
s3err.PostLog(r, http.StatusOK, s3err.ErrNone)
|
|
}
|
|
|
|
func writeSuccessResponseXMLBytes(w http.ResponseWriter, r *http.Request, response []byte) {
|
|
s3err.WriteResponse(w, r, http.StatusOK, response, s3err.MimeXML)
|
|
s3err.PostLog(r, http.StatusOK, s3err.ErrNone)
|
|
}
|
|
|
|
func writeSuccessResponseEmpty(w http.ResponseWriter, r *http.Request) {
|
|
s3err.WriteEmptyResponse(w, r, http.StatusOK)
|
|
}
|
|
|
|
func writeFailureResponse(w http.ResponseWriter, r *http.Request, errCode s3err.ErrorCode) {
|
|
s3err.WriteErrorResponse(w, r, errCode)
|
|
}
|
|
|
|
func validateContentMd5(h http.Header) ([]byte, error) {
|
|
md5B64, ok := h["Content-Md5"]
|
|
if ok {
|
|
if md5B64[0] == "" {
|
|
return nil, fmt.Errorf("Content-Md5 header set to empty value")
|
|
}
|
|
return base64.StdEncoding.DecodeString(md5B64[0])
|
|
}
|
|
return []byte{}, nil
|
|
}
|