mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-05 22:12:04 +02:00
* feat(filer): option to store system metadata logs in their own collection The filer's internal /topics/.system/log chunks are assigned to the filer's default collection (-collection). In a multi-filer deployment that default is often empty, so every restart flap, full-sync, or event-buffered flush grows the default collection with system chunks that are indistinguishable from user data in collection.list. This is a large part of what makes the default collection balloon and confuses orphan analysis. This keeps the internal log in a dedicated collection when the operator asks for one, without changing where user data goes: - New optional override, filer.options.metaLog.collection (and .replication), read in NewFiler so both `weed filer` and `weed server -filer` honour it. Default "" => exactly today's behaviour (log follows the filer default), fully backward compatible. - Resolution is a small helper: override first, then the filer default, then a storage rule matched on the log path. Kept separate from the user write path so the internal log targets itself. - bucketCollection() is hardened the same way it already protects the filer's default collection: a bucket that happens to resolve to the redirected meta-log collection must not drop it on delete, because it backs internal log volumes. - Scaffold filer.toml documents the new knobs under [filer.options]. Related to the persisted deletion ledger branch (fix/persist-deletion-queue): together they cut the two sources of post-flap junk in the default collection — that PR stops orphaned user-chunk leak on filer crash, this one stops the internal log from living in default at all. They are independent: no file overlap, no functional dependency; either can merge first. They are paired only in the narrative of cleaning up default. Adds unit tests for the collection/replication resolution chain, the viper keys, and the bucket-delete guard (run green under -race). Co-Authored-By: Athena 🏛️ <hermes-agent@local> (custom / Qwen3.8-Flash-Next-ROCmFP4) * filer: collect bucket chunks when its collection survives the delete bucketCollection returning "" preserves the collection, but the bucket path still skipped per-entry chunk collection and could skip listing the children entirely, so a bucket sharing the meta-log (or any preserved) collection left its object chunks orphaned with no entry pointing at them. Only the wholesale drop of a deleted collection skips those now. Note in filer.toml that the meta-log target should stay stable: chunks written under an older collection are not migrated. * filer: exercise the metaLog override wiring through NewFiler The viper test only echoed back the keys it set, so a wrong key in NewFiler would still pass. It now asserts the fields NewFiler fills from those keys. Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> * filer: tighten comments around the metaLog collection override Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Chris Lu <chris.lu@gmail.com> Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com> Co-authored-by: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com>
117 lines
3.7 KiB
Go
117 lines
3.7 KiB
Go
package filer
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"os"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/operation"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/util"
|
|
)
|
|
|
|
func (f *Filer) appendToFile(targetFile string, data []byte) error {
|
|
|
|
assignResult, uploadResult, err2 := f.assignAndUpload(targetFile, data)
|
|
if err2 != nil {
|
|
return err2
|
|
}
|
|
|
|
// find out existing entry
|
|
fullpath := util.FullPath(targetFile)
|
|
entry, err := f.FindEntry(context.Background(), fullpath)
|
|
var offset int64 = 0
|
|
if err == filer_pb.ErrNotFound {
|
|
entry = &Entry{
|
|
FullPath: fullpath,
|
|
Attr: Attr{
|
|
Crtime: time.Now(),
|
|
Mtime: time.Now(),
|
|
Mode: os.FileMode(0644),
|
|
Uid: OS_UID,
|
|
Gid: OS_GID,
|
|
},
|
|
}
|
|
} else if err != nil {
|
|
return fmt.Errorf("find %s: %v", fullpath, err)
|
|
} else {
|
|
offset = int64(TotalSize(entry.GetChunks()))
|
|
}
|
|
|
|
// append to existing chunks
|
|
entry.Chunks = append(entry.GetChunks(), uploadResult.ToPbFileChunk(assignResult.Fid, offset, time.Now().UnixNano()))
|
|
|
|
// update the entry
|
|
err = f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
|
|
|
return err
|
|
}
|
|
|
|
// resolveMetadataLogAssignDiskType returns the disk type and matched path rule for
|
|
// metadata log volume assigns. Disk type uses the rule when set, otherwise
|
|
// Filer.DefaultDiskType (from -filer.disk), mirroring resolveAssignStorageOption.
|
|
func (f *Filer) resolveMetadataLogAssignDiskType(targetFile string) (string, *filer_pb.FilerConf_PathConf) {
|
|
if f.FilerConf == nil {
|
|
return f.DefaultDiskType, &filer_pb.FilerConf_PathConf{}
|
|
}
|
|
rule := f.FilerConf.MatchStorageRule(targetFile)
|
|
return util.Nvl(rule.DiskType, f.DefaultDiskType), rule
|
|
}
|
|
|
|
// metaLogCollectionFor resolves the system metadata log's collection: the
|
|
// filer.options.metaLog.collection override, then the filer default, then the
|
|
// matched storage rule.
|
|
func (f *Filer) metaLogCollectionFor(ruleCollection string) string {
|
|
return util.Nvl(f.metaLogTargetCollection, f.metaLogCollection, ruleCollection)
|
|
}
|
|
|
|
func (f *Filer) metaLogReplicationFor(ruleReplication string) string {
|
|
return util.Nvl(f.metaLogTargetReplication, f.metaLogReplication, ruleReplication)
|
|
}
|
|
|
|
func (f *Filer) assignAndUpload(targetFile string, data []byte) (*operation.AssignResult, *operation.UploadResult, error) {
|
|
// assign a volume location
|
|
diskType, rule := f.resolveMetadataLogAssignDiskType(targetFile)
|
|
assignRequest := &operation.VolumeAssignRequest{
|
|
Count: 1,
|
|
Collection: f.metaLogCollectionFor(rule.Collection),
|
|
Replication: f.metaLogReplicationFor(rule.Replication),
|
|
DiskType: diskType,
|
|
WritableVolumeCount: rule.VolumeGrowthCount,
|
|
ExpectedDataSize: uint64(len(data)),
|
|
}
|
|
|
|
assignResult, err := operation.Assign(context.Background(), f.GetMaster, f.GrpcDialOption, assignRequest)
|
|
if err != nil {
|
|
return nil, nil, fmt.Errorf("AssignVolume: %w", err)
|
|
}
|
|
if assignResult.Error != "" {
|
|
return nil, nil, fmt.Errorf("AssignVolume error: %v", assignResult.Error)
|
|
}
|
|
|
|
// upload data
|
|
targetUrl := "http://" + assignResult.Url + "/" + assignResult.Fid
|
|
uploadOption := &operation.UploadOption{
|
|
UploadUrl: targetUrl,
|
|
Filename: "",
|
|
Cipher: f.Cipher,
|
|
IsInputCompressed: false,
|
|
MimeType: "",
|
|
PairMap: nil,
|
|
Jwt: assignResult.Auth,
|
|
}
|
|
|
|
uploader, err := operation.NewUploader()
|
|
if err != nil {
|
|
return nil, nil, fmt.Errorf("upload data %s: %v", targetUrl, err)
|
|
}
|
|
|
|
uploadResult, err := uploader.UploadData(context.Background(), data, uploadOption)
|
|
if err != nil {
|
|
return nil, nil, fmt.Errorf("upload data %s: %v", targetUrl, err)
|
|
}
|
|
// println("uploaded to", targetUrl)
|
|
return assignResult, uploadResult, nil
|
|
}
|