fix(ec): make ec.decode write-path crash-safe and atomic (#9949)

* fix(ec): check decode .idx writes and fsync decoded .dat/.idx

WriteIdxFileFromEcIndex silently dropped io.Copy and Write errors, so a
short or failed write of the reconstructed .idx went unnoticed and the
caller proceeded to delete the source EC shards. Propagate those errors.

Also fsync the decoded .dat and .idx before returning, so the bytes are
durable before the shards that produced them are removed cluster-wide.
Mirror the .idx fsync into the Rust volume server (its .dat already
syncs and its writes already propagate errors).

* fix(ec): publish decoded .dat/.idx atomically via temp file and rename

WriteDatFile and WriteIdxFileFromEcIndex wrote in place at the final
name with O_TRUNC. A crash mid-write left a truncated .dat/.idx at the
final name beside the still-present EC shards; on restart that partial
file could be mounted as the live volume even though the shards held the
real data. Write to a .tmp file, fsync it, then rename into place and
fsync the directory, so the final name is only ever absent or complete.
A failed decode removes its own temp file rather than leaking it.

Add util.FsyncDir as the shared directory-fsync primitive and reuse the
Rust volume server's fsync_dir for the mirrored change.

* fix(ec): propagate .ecj read errors in the Rust decoder

Path::exists returned false for any error (permission denied, transient
IO), silently skipping the deletion journal and resurrecting deleted
needles as live. Read the journal directly and treat only NotFound as
absent, propagating other errors. The Go decoder already behaves this
way (FileExists returns false only for IsNotExist, then the open
surfaces other errors).

* fix(ec): remove rename destination on Windows in the Rust decoder publish

std::fs::rename does not replace an existing file on every Windows
version. Remove the destination first under a Windows guard before the
atomic publish rename, matching the compaction commit path.
This commit is contained in:
Chris Lu authored and GitHub committed 2026-06-13 21:26:07 -07:00
1 parent 4fb3e22a01
commit 1e858d8af0
6 files changed
+259 -82

No files matched your search

+65 -11
View File
@@ -4,6 +4,7 @@ import (
"fmt"
"io"
"os"
"path/filepath"
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
"github.com/seaweedfs/seaweedfs/weed/storage/idx"
@@ -40,23 +41,53 @@ func WriteIdxFileFromEcIndex(baseFileName string) (err error) {
}
defer ecxFile.Close()
idxFile, openErr := os.OpenFile(baseFileName+".idx", os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0644)
// Write to a temp file and atomically rename into place, so a crash mid-write
// never leaves a partial .idx at the final name beside the source shards.
idxFileName := baseFileName + ".idx"
tmpFileName := idxFileName + ".tmp"
idxFile, openErr := os.OpenFile(tmpFileName, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0644)
if openErr != nil {
return fmt.Errorf("cannot open %s.idx: %v", baseFileName, openErr)
return fmt.Errorf("cannot open %s: %v", tmpFileName, openErr)
}
defer idxFile.Close()
committed := false
defer func() {
idxFile.Close()
if !committed {
os.Remove(tmpFileName)
}
}()
io.Copy(idxFile, ecxFile)
if _, err = io.Copy(idxFile, ecxFile); err != nil {
return fmt.Errorf("copy ecx to idx for %s: %v", baseFileName, err)
}
err = iterateEcjFile(baseFileName, func(key types.NeedleId) error {
bytes := needle_map.ToBytes(key, types.Offset{}, types.TombstoneFileSize)
idxFile.Write(bytes)
if _, writeErr := idxFile.Write(bytes); writeErr != nil {
return writeErr
}
return nil
})
if err != nil {
return err
}
return err
// fsync, rename, then fsync the dir so the decoded .idx is durable and
// atomically published before the caller deletes the source shards.
if err = idxFile.Sync(); err != nil {
return fmt.Errorf("sync idx for %s: %v", baseFileName, err)
}
if err = idxFile.Close(); err != nil {
return fmt.Errorf("close idx for %s: %v", baseFileName, err)
}
if err = os.Rename(tmpFileName, idxFileName); err != nil {
return fmt.Errorf("rename idx for %s: %v", baseFileName, err)
}
if err = util.FsyncDir(filepath.Dir(idxFileName)); err != nil {
return fmt.Errorf("fsync dir for %s: %v", baseFileName, err)
}
committed = true
return nil
}
// FindDatFileSize calculate .dat file size from max offset entry
@@ -175,23 +206,31 @@ func iterateEcjFile(baseFileName string, processNeedleFn func(key types.NeedleId
// WriteDatFile generates .dat from EC shard files (e.g., .ec00 ~ .ec09 for 10+4)
func WriteDatFile(baseFileName string, datFileSize int64, shardFileNames []string) error {
datFile, openErr := os.OpenFile(baseFileName+".dat", os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0644)
// Write to a temp file and atomically rename into place, so a crash mid-write
// never leaves a partial .dat at the final name beside the source shards.
datFileName := baseFileName + ".dat"
tmpFileName := datFileName + ".tmp"
datFile, openErr := os.OpenFile(tmpFileName, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0644)
if openErr != nil {
return fmt.Errorf("cannot write volume %s.dat: %v", baseFileName, openErr)
return fmt.Errorf("cannot write volume %s: %v", tmpFileName, openErr)
}
defer datFile.Close()
// Use the actual number of data shards passed in rather than the global
// constant, so the de-striping matches the caller's shard set.
dataShards := len(shardFileNames)
inputFiles := make([]*os.File, dataShards)
committed := false
defer func() {
datFile.Close()
for shardId := 0; shardId < dataShards; shardId++ {
if inputFiles[shardId] != nil {
inputFiles[shardId].Close()
}
}
if !committed {
os.Remove(tmpFileName)
}
}()
for shardId := 0; shardId < dataShards; shardId++ {
@@ -222,6 +261,21 @@ func WriteDatFile(baseFileName string, datFileSize int64, shardFileNames []strin
}
}
// fsync, rename, then fsync the dir so the decoded .dat is durable and
// atomically published before the caller deletes the source shards.
if err := datFile.Sync(); err != nil {
return fmt.Errorf("sync dat for %s: %v", baseFileName, err)
}
if err := datFile.Close(); err != nil {
return fmt.Errorf("close dat for %s: %v", baseFileName, err)
}
if err := os.Rename(tmpFileName, datFileName); err != nil {
return fmt.Errorf("rename dat for %s: %v", baseFileName, err)
}
if err := util.FsyncDir(filepath.Dir(datFileName)); err != nil {
return fmt.Errorf("fsync dir for %s: %v", baseFileName, err)
}
committed = true
return nil
}