diff --git a/seaweed-volume/proto/volume_server.proto b/seaweed-volume/proto/volume_server.proto
index 6b64ef529..961a800fe 100644
--- a/seaweed-volume/proto/volume_server.proto
+++ b/seaweed-volume/proto/volume_server.proto
@@ -539,6 +539,7 @@ message VolumeEcShardsInfoResponse {
uint64 volume_size = 2;
uint64 file_count = 3;
uint64 file_deleted_count = 4;
+ EcShardConfig ec_shard_config = 10; // the layout this holder serves reads through; a binary predating a field reports it as unset
}
message EcShardInfo {
@@ -612,6 +613,7 @@ message EcShardConfig {
uint32 data_shards = 1; // Number of data shards (e.g., 10)
uint32 parity_shards = 2; // Number of parity shards (e.g., 4)
int64 encode_ts_ns = 3; // encode time (unix nanos); a read served from a shard of a different encode run is rejected
+ int64 block_size = 4; // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout
}
// EcBitrotProtection is the entire content of a bitrot checksum sidecar
// ( .ecsum for the legacy generation, .ecsum.v for vacuum
diff --git a/seaweed-volume/src/server/grpc_server.rs b/seaweed-volume/src/server/grpc_server.rs
index e366413d5..dc956c414 100644
--- a/seaweed-volume/src/server/grpc_server.rs
+++ b/seaweed-volume/src/server/grpc_server.rs
@@ -92,6 +92,57 @@ pub fn load_state_file(
volume_server_pb::VolumeServerState::decode(data.as_slice()).ok()
}
+/// One disk location's stake in an EC volume, as seen by the rebuild handler.
+struct LocInfo {
+ dir: String,
+ idx_dir: String,
+ shard_count: usize,
+ has_ecx: bool,
+}
+
+/// Picks the location a rebuild should write into — the one holding an `.ecx`
+/// and the most shards — and returns every other directory it may have to read
+/// from. Shards are only half of what the rebuild needs: a split
+/// `-dir`/`-dir.idx` layout keeps `.ecx`/`.ecj`/`.vif` with the INDEX, and on a
+/// multi-disk server the chosen disk may hold nothing but shards while this
+/// volume's `.vif` or generation-0 `.ecsum` sits on a sibling. Miss those and
+/// the layout resolution falls back to 10+4 with the legacy striping and
+/// reconstructs through the wrong matrix, so both directories of every other
+/// location are listed. The rebuild's own two are passed separately by the
+/// caller and dropped here, along with empties and duplicates.
+///
+/// Returns `None` when no location holds an `.ecx`, i.e. there is nothing to
+/// rebuild from.
+fn select_rebuild_location(loc_infos: &[LocInfo]) -> Option<(usize, Vec)> {
+ let mut rebuild_loc_idx: Option = None;
+ let mut other_dirs: Vec = Vec::new();
+
+ for (i, info) in loc_infos.iter().enumerate() {
+ let better = info.has_ecx
+ && rebuild_loc_idx
+ .is_none_or(|prev| info.shard_count > loc_infos[prev].shard_count);
+ if better {
+ if let Some(prev) = rebuild_loc_idx {
+ other_dirs.push(loc_infos[prev].dir.clone());
+ other_dirs.push(loc_infos[prev].idx_dir.clone());
+ }
+ rebuild_loc_idx = Some(i);
+ } else {
+ other_dirs.push(info.dir.clone());
+ other_dirs.push(info.idx_dir.clone());
+ }
+ }
+
+ let rebuild_loc_idx = rebuild_loc_idx?;
+ let rebuild_dir = &loc_infos[rebuild_loc_idx].dir;
+ let rebuild_idx_dir = &loc_infos[rebuild_loc_idx].idx_dir;
+ other_dirs.retain(|d| !d.is_empty() && d != rebuild_dir && d != rebuild_idx_dir);
+ other_dirs.sort();
+ other_dirs.dedup();
+
+ Some((rebuild_loc_idx, other_dirs))
+}
+
struct WriteThrottler {
bytes_per_second: i64,
last_size_counter: i64,
@@ -2383,13 +2434,21 @@ impl VolumeServer for VolumeGrpcService {
)
};
- // Check existing .vif for EC shard config (matching Go's MaybeLoadVolumeInfo)
- let (data_shards, parity_shards) =
+ // Check existing .vif for EC shard config (matching Go's MaybeLoadVolumeInfo).
+ // The block size is recomputed by the encode for the current .dat, so
+ // only the ratio is carried over from a prior config.
+ let (data_shards, parity_shards, _) =
crate::storage::erasure_coding::ec_volume::read_ec_shard_config(
&dir, &idx_dir, collection, vid,
- );
+ )
+ .map_err(|e| {
+ tonic::Status::internal(format!(
+ "read ec shard config for volume {}: {}",
+ vid.0, e
+ ))
+ })?;
- if let Err(e) = crate::storage::erasure_coding::ec_encoder::write_ec_files(
+ let block_size = match crate::storage::erasure_coding::ec_encoder::write_ec_files(
&dir,
&idx_dir,
collection,
@@ -2397,16 +2456,19 @@ impl VolumeServer for VolumeGrpcService {
data_shards as usize,
parity_shards as usize,
) {
- // Cleanup partially-created .ecNN and .ecx files on failure (matching Go defer)
- let base = crate::storage::volume::volume_file_name(&dir, collection, vid);
- let total_shards = data_shards + parity_shards;
- for i in 0..total_shards {
- let shard_path = format!("{}.ec{:02}", base, i);
- let _ = std::fs::remove_file(&shard_path);
+ Ok(block_size) => block_size,
+ Err(e) => {
+ // Cleanup partially-created .ecNN and .ecx files on failure (matching Go defer)
+ let base = crate::storage::volume::volume_file_name(&dir, collection, vid);
+ let total_shards = data_shards + parity_shards;
+ for i in 0..total_shards {
+ let shard_path = format!("{}.ec{:02}", base, i);
+ let _ = std::fs::remove_file(&shard_path);
+ }
+ let _ = std::fs::remove_file(format!("{}.ecx", base));
+ return Err(Status::internal(e.to_string()));
}
- let _ = std::fs::remove_file(format!("{}.ecx", base));
- return Err(Status::internal(e.to_string()));
- }
+ };
// Write .vif file with EC shard metadata
{
@@ -2425,6 +2487,7 @@ impl VolumeServer for VolumeGrpcService {
.duration_since(std::time::UNIX_EPOCH)
.unwrap_or_default()
.as_nanos() as i64,
+ block_size,
}),
..Default::default()
};
@@ -2458,13 +2521,6 @@ impl VolumeServer for VolumeGrpcService {
format!("{}_{}", collection, vid.0)
};
- struct LocInfo {
- dir: String,
- idx_dir: String,
- shard_count: usize,
- has_ecx: bool,
- }
-
let store = self.state.store.read().unwrap();
let mut loc_infos: Vec = Vec::new();
@@ -2512,26 +2568,8 @@ impl VolumeServer for VolumeGrpcService {
));
}
- // Pick rebuild location: has .ecx and most shards
- let mut rebuild_loc_idx: Option = None;
- let mut other_dirs: Vec = Vec::new();
-
- for (i, info) in loc_infos.iter().enumerate() {
- if info.has_ecx
- && (rebuild_loc_idx.is_none()
- || info.shard_count > loc_infos[rebuild_loc_idx.unwrap()].shard_count)
- {
- if let Some(prev) = rebuild_loc_idx {
- other_dirs.push(loc_infos[prev].dir.clone());
- }
- rebuild_loc_idx = Some(i);
- } else {
- other_dirs.push(info.dir.clone());
- }
- }
-
- let rebuild_loc_idx = match rebuild_loc_idx {
- Some(i) => i,
+ let (rebuild_loc_idx, other_dirs) = match select_rebuild_location(&loc_infos) {
+ Some(picked) => picked,
None => {
return Ok(Response::new(
volume_server_pb::VolumeEcShardsRebuildResponse {
@@ -2545,13 +2583,37 @@ impl VolumeServer for VolumeGrpcService {
let rebuild_idx_dir = loc_infos[rebuild_loc_idx].idx_dir.clone();
// Determine data/parity shard config from rebuild dir
- let (data_shards, parity_shards) =
- crate::storage::erasure_coding::ec_volume::read_ec_shard_config(
+ // The encode-time .dat size resolves the row count the ecx rebuild
+ // de-stripes with; 0 leaves it to infer from the padded shard extent.
+ // Both lookups search the sibling disks too: the rebuild writes into one
+ // location, but a multi-disk server may keep this volume's .vif or its
+ // generation-0 .ecsum on another, and defaulting to 10+4 with the
+ // legacy layout would reconstruct through the wrong matrix.
+ let dat_file_size = crate::storage::erasure_coding::ec_volume::load_vif_info_across_dirs(
+ &rebuild_dir,
+ &rebuild_idx_dir,
+ &other_dirs,
+ collection,
+ vid,
+ )
+ .ok()
+ .flatten()
+ .map(|(v, _)| v.dat_file_size)
+ .unwrap_or(0);
+ let (data_shards, parity_shards, block_size) =
+ crate::storage::erasure_coding::ec_volume::read_ec_shard_config_across_dirs(
&rebuild_dir,
&rebuild_idx_dir,
+ &other_dirs,
collection,
vid,
- );
+ )
+ .map_err(|e| {
+ tonic::Status::internal(format!(
+ "read ec shard config for volume {}: {}",
+ vid.0, e
+ ))
+ })?;
let total_shards = data_shards + parity_shards;
// Check which shards are missing (check rebuild dir and all other dirs)
@@ -2590,8 +2652,17 @@ impl VolumeServer for VolumeGrpcService {
// Rebuild missing shards, searching all locations for input shards.
// Pass other_dirs so shards on sibling disks are found even when the
- // primary rebuild dir doesn't hold them.
- let other_dir_refs: Vec<&str> = other_dirs.iter().map(|s| s.as_str()).collect();
+ // primary rebuild dir doesn't hold them. This one takes a single flat
+ // list — the shape Go's RebuildEcFiles uses — so unlike the resolvers
+ // above it cannot be handed the rebuild's own index directory
+ // separately, and a split -dir/-dir.idx location keeps its .ecx and
+ // .vif there. Go's additionalDirs carries that directory for the same
+ // reason.
+ let mut rebuild_search_dirs: Vec = other_dirs.clone();
+ if !rebuild_idx_dir.is_empty() && rebuild_idx_dir != rebuild_dir {
+ rebuild_search_dirs.push(rebuild_idx_dir.clone());
+ }
+ let other_dir_refs: Vec<&str> = rebuild_search_dirs.iter().map(|s| s.as_str()).collect();
crate::storage::erasure_coding::ec_encoder::rebuild_ec_files(
&rebuild_dir,
collection,
@@ -2629,6 +2700,8 @@ impl VolumeServer for VolumeGrpcService {
collection,
vid,
data_shards as usize,
+ block_size,
+ dat_file_size,
&ecx_dir_refs,
)
.map_err(|e| Status::internal(format!("RebuildEcxFile: {}", e)))?;
@@ -3015,6 +3088,27 @@ impl VolumeServer for VolumeGrpcService {
Status::internal(format!("mount {}.{}: {}", req.volume_id, shard_id, e))
})?;
}
+ // A delivery can bring the checksum manifest alongside the shards, but
+ // the receive path only writes the file. When this server already had
+ // the volume mounted, the EcVolume in memory keeps whatever protection
+ // state it resolved at mount — off, for a volume whose sidecar arrives
+ // now — until a remount. Re-resolve it here, where the shards it
+ // describes have just been added.
+ //
+ // Every per-disk runtime, not just the first: a vid mounts as one
+ // EcVolume per disk, the delivery lands the .ecsum on one of them, and
+ // the first-match lookup would leave the siblings reporting no
+ // protection. Each re-resolves against its own data and index
+ // directories, so a shared -dir.idx reaches all of them.
+ // Resolving across every EC metadata directory is what makes that
+ // reload mean something: startup mirroring gives each shard-bearing
+ // disk its own .ecx/.ecj/.vif but deliberately not the sidecar, so a
+ // runtime restricted to its own two directories would find nothing
+ // however often it reloaded. One delivered copy, reachable from all.
+ let ec_metadata_dirs = store.ec_metadata_dirs();
+ for ec_vol in store.find_all_ec_volumes_mut(vid) {
+ ec_vol.reload_bitrot_sidecar(&ec_metadata_dirs);
+ }
drop(store);
self.state.volume_state_notify.notify_one();
@@ -3349,6 +3443,8 @@ impl VolumeServer for VolumeGrpcService {
let ecx_dir = ec_vol.ecx_actual_dir().to_string();
let collection = ec_vol.collection.clone();
let vif_dat_file_size = ec_vol.dat_file_size;
+ let (large_block_size, small_block_size) =
+ (ec_vol.large_block_size(), ec_vol.small_block_size());
// shard_dirs[i] is guaranteed Some for i in 0..data_shards by
// the check above; collect concrete dirs for the decoder.
let per_shard_dirs: Vec = shard_dirs[..data_shards]
@@ -3382,6 +3478,8 @@ impl VolumeServer for VolumeGrpcService {
vif_dat_file_size,
data_shards,
&per_shard_dirs,
+ large_block_size as usize,
+ small_block_size as usize,
)
.map_err(|e| Status::internal(format!("WriteDatFile: {}", e)))?;
@@ -3440,12 +3538,24 @@ impl VolumeServer for VolumeGrpcService {
.walk_ecx_stats()
.map_err(|e| Status::internal(e.to_string()))?;
+ // The layout this holder serves reads through, as Go reports it: a
+ // coordinator cannot otherwise tell a holder that understands the
+ // uniform block layout from one that dropped the unknown .vif field
+ // and mounted the volume as legacy.
+ let ec_shard_config = Some(volume_server_pb::EcShardConfig {
+ data_shards: ec_vol.data_shards,
+ parity_shards: ec_vol.parity_shards,
+ encode_ts_ns: 0,
+ block_size: ec_vol.block_size,
+ });
+
Ok(Response::new(
volume_server_pb::VolumeEcShardsInfoResponse {
ec_shard_infos: shard_infos,
volume_size,
file_count,
file_deleted_count,
+ ec_shard_config,
},
))
}
@@ -5086,6 +5196,110 @@ mod tests {
use tempfile::TempDir;
use tokio_stream::StreamExt;
+ fn loc(dir: &str, idx_dir: &str, shard_count: usize, has_ecx: bool) -> LocInfo {
+ LocInfo {
+ dir: dir.to_string(),
+ idx_dir: idx_dir.to_string(),
+ shard_count,
+ has_ecx,
+ }
+ }
+
+ // The rebuild reads its shards from one directory but resolves the volume's
+ // layout -- ratio and uniform block size -- from the .vif or the
+ // generation-0 .ecsum, which on a multi-disk server may sit anywhere. Every
+ // directory that could hold one has to be in the search list, or the
+ // resolution silently falls back to 10+4 with the legacy striping and
+ // reconstructs through the wrong matrix.
+ // The rebuild's own data and index directories are handed to the resolvers
+ // as their own arguments, so they are deliberately absent from this list --
+ // unlike Go, whose resolver takes a single directory list and therefore
+ // carries the rebuild's index directory inside it.
+ #[test]
+ fn select_rebuild_location_excludes_the_rebuilds_own_dirs() {
+ for infos in [
+ vec![loc("/data1", "/idx1", 3, true)],
+ vec![loc("/data1", "/data1", 3, true)],
+ ] {
+ let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx");
+ assert_eq!(idx, 0);
+ assert!(others.is_empty(), "got {:?}", others);
+ }
+ }
+
+ // The case two reviewers flagged: a sibling holding only shards while its
+ // index directory holds this volume's .vif.
+ #[test]
+ fn select_rebuild_location_searches_a_siblings_index_dir_not_just_its_data_dir() {
+ let infos = vec![
+ loc("/data1", "/data1", 5, true),
+ loc("/data2", "/idx2", 2, false),
+ ];
+ let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx");
+ assert_eq!(idx, 0);
+ assert_eq!(others, vec!["/data2".to_string(), "/idx2".to_string()]);
+ }
+
+ // Several disks pointed at one index directory is a normal -dir.idx
+ // deployment; the shared directory is worth searching but only once.
+ #[test]
+ fn select_rebuild_location_lists_a_shared_index_dir_once() {
+ let infos = vec![
+ loc("/data1", "/data1", 5, true),
+ loc("/data2", "/shared-idx", 2, false),
+ loc("/data3", "/shared-idx", 1, false),
+ ];
+ let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx");
+ assert_eq!(
+ others,
+ vec![
+ "/data2".to_string(),
+ "/data3".to_string(),
+ "/shared-idx".to_string()
+ ]
+ );
+ }
+
+ // When the shared index directory is the rebuild's own it drops out, since
+ // the caller passes it separately.
+ #[test]
+ fn select_rebuild_location_omits_a_shared_index_dir_it_rebuilds_into() {
+ let infos = vec![
+ loc("/data1", "/shared-idx", 5, true),
+ loc("/data2", "/shared-idx", 2, false),
+ ];
+ let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx");
+ assert_eq!(others, vec!["/data2".to_string()]);
+ }
+
+ // The winner moves as a fuller location turns up; the one it displaces
+ // still has to be searched, index directory included.
+ #[test]
+ fn select_rebuild_location_keeps_the_displaced_winners_dirs() {
+ let infos = vec![
+ loc("/data1", "/idx1", 2, true),
+ loc("/data2", "/idx2", 9, true),
+ ];
+ let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx");
+ assert_eq!(idx, 1, "the fuller location wins");
+ assert_eq!(others, vec!["/data1".to_string(), "/idx1".to_string()]);
+ }
+
+ #[test]
+ fn select_rebuild_location_drops_empty_dirs() {
+ let infos = vec![loc("/data1", "", 3, true), loc("/data2", "", 1, false)];
+ let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx");
+ assert_eq!(others, vec!["/data2".to_string()]);
+ }
+
+ // Nothing carries an .ecx: there is no index to rebuild the shards against,
+ // so the caller answers with an empty rebuild rather than guessing.
+ #[test]
+ fn select_rebuild_location_is_none_without_an_ecx() {
+ let infos = vec![loc("/data1", "/idx1", 3, false)];
+ assert!(select_rebuild_location(&infos).is_none());
+ }
+
#[test]
fn test_parse_grpc_address_with_explicit_grpc_port() {
// Format: "ip:port.grpcPort" — used by SeaweedFS for source_data_node
diff --git a/seaweed-volume/src/server/store_ec.rs b/seaweed-volume/src/server/store_ec.rs
index 82277c500..1a8715295 100644
--- a/seaweed-volume/src/server/store_ec.rs
+++ b/seaweed-volume/src/server/store_ec.rs
@@ -599,7 +599,7 @@ fn read_local_intervals(
) -> Vec {
let mut interval_results = Vec::with_capacity(intervals.len());
for interval in intervals {
- let (shard_id, shard_offset) = interval.to_shard_id_and_offset(ecv.data_shards);
+ let (shard_id, shard_offset) = ecv.interval_to_shard_id_and_offset(interval);
let buf_size = interval.size as usize;
let local = ecv.shards.get(shard_id as usize).and_then(|s| s.as_ref());
match local {
diff --git a/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs b/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs
index e1eab27b5..1dc9b42ee 100644
--- a/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs
+++ b/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs
@@ -465,6 +465,28 @@ pub fn resolve_status(
}
}
+/// Whether a generation-matching sidecar agrees with the geometry the volume is
+/// mounted with. Both files record the layout the generation was encoded with,
+/// so a disagreement means one of them is wrong and reads through the other
+/// would land at the wrong shard offsets — the caller fails the mount rather
+/// than merely dropping protection. A sidecar that records no EC config has
+/// nothing to contradict.
+pub fn geometry_matches(
+ prot: &EcBitrotProtection,
+ data_shards: usize,
+ parity_shards: usize,
+ block_size: i64,
+) -> bool {
+ match &prot.ec_shard_config {
+ None => true,
+ Some(cfg) => {
+ cfg.data_shards as usize == data_shards
+ && cfg.parity_shards as usize == parity_shards
+ && cfg.block_size == block_size
+ }
+ }
+}
+
/// Returns the [`EcShardChecksums`] entry for a shard id, or `None`.
pub fn shard_checksums(prot: &EcBitrotProtection, shard_id: u32) -> Option<&EcShardChecksums> {
prot.shards.iter().find(|s| s.shard_id == shard_id)
@@ -539,11 +561,12 @@ fn read_full_at(f: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
/// Builds the `EcShardConfig` proto for the given layout. The bitrot sidecar
/// carries its own top-level encode_uuid, so the nested config leaves it empty.
-pub fn ec_shard_config(data_shards: u32, parity_shards: u32) -> EcShardConfig {
+pub fn ec_shard_config(data_shards: u32, parity_shards: u32, block_size: i64) -> EcShardConfig {
EcShardConfig {
data_shards,
parity_shards,
encode_ts_ns: 0,
+ block_size,
}
}
@@ -567,6 +590,7 @@ mod tests {
data_shards: 10,
parity_shards: 4,
encode_ts_ns: 0,
+ block_size: 0,
}),
shards: vec![
EcShardChecksums {
@@ -712,7 +736,7 @@ mod tests {
algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
block_size: DEFAULT_BITROT_BLOCK_SIZE as u32,
generation: 0,
- ec_shard_config: Some(ec_shard_config(10, 4)),
+ ec_shard_config: Some(ec_shard_config(10, 4, 0)),
shards: vec![EcShardChecksums {
shard_id: 0,
covered_size: covered,
@@ -750,7 +774,7 @@ mod tests {
algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
block_size: DEFAULT_BITROT_BLOCK_SIZE as u32,
generation: 0,
- ec_shard_config: Some(ec_shard_config(10, 4)),
+ ec_shard_config: Some(ec_shard_config(10, 4, 0)),
shards: vec![EcShardChecksums {
shard_id: 0,
covered_size: 5,
@@ -794,7 +818,7 @@ mod tests {
algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
block_size: DEFAULT_BITROT_BLOCK_SIZE as u32,
generation: 0,
- ec_shard_config: Some(ec_shard_config(10, 4)),
+ ec_shard_config: Some(ec_shard_config(10, 4, 0)),
shards,
encode_uuid: vec![0u8; 16],
}
diff --git a/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs b/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs
index 38a7de561..81db2d602 100644
--- a/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs
+++ b/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs
@@ -80,6 +80,8 @@ pub fn write_dat_file_from_shards(
dat_file_size: i64,
encoded_dat_file_size: i64,
data_shards: usize,
+ large_block_size: usize,
+ small_block_size: usize,
) -> io::Result<()> {
let dirs: Vec = (0..data_shards).map(|_| dir.to_string()).collect();
write_dat_file_from_shards_with_dirs(
@@ -90,6 +92,8 @@ pub fn write_dat_file_from_shards(
encoded_dat_file_size,
data_shards,
&dirs,
+ large_block_size,
+ small_block_size,
)
}
@@ -113,7 +117,10 @@ pub fn write_dat_file_from_shards(
/// boundary, and deriving the layout from the shrunk extent would read
/// the shards in the wrong block order. Pass zero when the .vif does
/// not record the encode-time size to infer the layout from the shard
-/// size.
+/// size. `large_block_size`/`small_block_size` are the volume's shard
+/// block layout, e.g. `EcVolume::large_block_size()` /
+/// `small_block_size()` from its .vif EC config.
+#[allow(clippy::too_many_arguments)]
pub fn write_dat_file_from_shards_with_dirs(
dat_dir: &str,
collection: &str,
@@ -122,6 +129,8 @@ pub fn write_dat_file_from_shards_with_dirs(
encoded_dat_file_size: i64,
data_shards: usize,
shard_dirs: &[String],
+ large_block_size: usize,
+ small_block_size: usize,
) -> io::Result<()> {
write_dat_file(
dat_dir,
@@ -131,8 +140,8 @@ pub fn write_dat_file_from_shards_with_dirs(
encoded_dat_file_size,
data_shards,
shard_dirs,
- ERASURE_CODING_LARGE_BLOCK_SIZE,
- ERASURE_CODING_SMALL_BLOCK_SIZE,
+ large_block_size,
+ small_block_size,
)
}
@@ -412,7 +421,9 @@ mod tests {
// Encode to EC
let data_shards = 10;
let parity_shards = 4;
- ec_encoder::write_ec_files(dir, dir, "", VolumeId(1), data_shards, parity_shards).unwrap();
+ let block_size =
+ ec_encoder::write_ec_files(dir, dir, "", VolumeId(1), data_shards, parity_shards)
+ .unwrap();
// Delete original .dat and .idx
std::fs::remove_file(format!("{}/1.dat", dir)).unwrap();
@@ -426,6 +437,8 @@ mod tests {
original_dat_size as i64,
original_dat_size as i64,
data_shards,
+ block_size as usize,
+ block_size as usize,
)
.unwrap();
write_idx_file_from_ec_index(dir, "", VolumeId(1)).unwrap();
@@ -472,7 +485,16 @@ mod tests {
let dir = tmp.path().to_str().unwrap();
// No shard files exist, so de-striping must fail and publish nothing:
// neither the final .dat nor a partial .dat.tmp may remain.
- let res = write_dat_file_from_shards(dir, "", VolumeId(7), 100, 100, 10);
+ let res = write_dat_file_from_shards(
+ dir,
+ "",
+ VolumeId(7),
+ 100,
+ 100,
+ 10,
+ ERASURE_CODING_LARGE_BLOCK_SIZE,
+ ERASURE_CODING_SMALL_BLOCK_SIZE,
+ );
assert!(res.is_err());
assert!(!std::path::Path::new(&format!("{}/7.dat", dir)).exists());
assert!(!std::path::Path::new(&format!("{}/7.dat.tmp", dir)).exists());
@@ -525,6 +547,7 @@ mod tests {
&mut builders,
data_shards,
parity_shards,
+ SMALL,
LARGE,
SMALL,
)
@@ -616,6 +639,7 @@ mod tests {
&mut builders,
data_shards,
parity_shards,
+ SMALL,
LARGE,
SMALL,
)
diff --git a/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs b/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs
index 346b2fd1f..1470e85de 100644
--- a/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs
+++ b/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs
@@ -25,6 +25,10 @@ use crate::storage::volume::volume_file_name;
///
/// Creates .ec00-.ec13 files in the same directory.
/// Also creates a sorted .ecx index from the .idx file.
+///
+/// Always encodes with the uniform block layout, sized for this .dat, and
+/// returns the block size so the caller can persist it to .vif. Mirrors Go's
+/// WriteEcFiles.
pub fn write_ec_files(
dir: &str,
idx_dir: &str,
@@ -32,7 +36,7 @@ pub fn write_ec_files(
volume_id: VolumeId,
data_shards: usize,
parity_shards: usize,
-) -> io::Result<()> {
+) -> io::Result {
let base = volume_file_name(dir, collection, volume_id);
let dat_path = format!("{}.dat", base);
let idx_base = volume_file_name(idx_dir, collection, volume_id);
@@ -66,7 +70,7 @@ pub fn write_ec_files(
.map(|_| ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64))
.collect();
- // Encode in large blocks, then small blocks
+ let block_size = uniform_block_size(dat_size, data_shards);
encode_dat_file(
&dat_file,
dat_size,
@@ -75,8 +79,9 @@ pub fn write_ec_files(
&mut builders,
data_shards,
parity_shards,
- ERASURE_CODING_LARGE_BLOCK_SIZE,
- ERASURE_CODING_SMALL_BLOCK_SIZE,
+ ENCODE_BUFFER_SIZE,
+ block_size as usize,
+ block_size as usize,
)?;
// Close all shards
@@ -103,6 +108,7 @@ pub fn write_ec_files(
ec_shard_config: Some(ec_bitrot::ec_shard_config(
data_shards as u32,
parity_shards as u32,
+ block_size,
)),
shards: shard_checksums,
encode_uuid: ec_bitrot::new_encode_uuid(),
@@ -120,7 +126,19 @@ pub fn write_ec_files(
);
}
- Ok(())
+ Ok(block_size)
+}
+
+/// uniform_block_size returns the per-shard block size of the uniform layout
+/// for a .dat of the given size: ceil(dat_file_size/data_shards) rounded up to
+/// a whole small block. For every input this equals the legacy layout's padded
+/// shard size, so only the byte placement differs between the two layouts,
+/// never the shard length. Mirrors Go's UniformBlockSize.
+pub fn uniform_block_size(dat_file_size: i64, data_shards: usize) -> i64 {
+ let small = ERASURE_CODING_SMALL_BLOCK_SIZE as i64;
+ let per_shard = (dat_file_size + data_shards as i64 - 1) / data_shards as i64;
+ let blocks = ((per_shard + small - 1) / small).max(1);
+ blocks * small
}
/// Rebuild missing EC shard files from existing shards using Reed-Solomon reconstruct.
@@ -370,7 +388,7 @@ pub fn verify_ec_shards(
}
/// Write sorted .ecx index from .idx file.
-fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::Result<()> {
+pub(crate) fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::Result<()> {
if !std::path::Path::new(idx_path).exists() {
return Err(io::Error::new(
io::ErrorKind::NotFound,
@@ -421,6 +439,8 @@ pub fn rebuild_ecx_file(
collection: &str,
volume_id: VolumeId,
data_shards: usize,
+ block_size: i64,
+ dat_file_size: i64,
additional_dirs: &[&str],
) -> io::Result<()> {
use crate::storage::needle::needle::get_actual_size;
@@ -463,10 +483,42 @@ pub fn rebuild_ecx_file(
// Determine total logical data size from shard sizes
let shard_size = shards.iter().map(|s| s.file_size()).max().unwrap_or(0);
let total_data_size = shard_size as i64 * data_shards as i64;
+ // The volume's shard block layout: the .vif-recorded uniform block size,
+ // or the legacy two-tier sizes when 0. The row count comes from the shard
+ // length; -1 disambiguates a legacy shard that is an exact large-block
+ // multiple (mirrors the ecdFileSize-1 fallback in the read path).
+ let (large_block, small_block) = if block_size > 0 {
+ (block_size, block_size)
+ } else {
+ (
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ )
+ };
+ // The row count the de-stripe walks with. The encode-time .dat size is the
+ // authority — the same value the read path divides by data_shards — and the
+ // padded extent is only a fallback: under the legacy layout a shard that is
+ // an exact large-block multiple reads as one row too many, which
+ // re-interprets its last large row as small blocks and scrambles the
+ // recovered offsets. Subtracting one keeps that fallback on the safe side of
+ // the boundary, exactly as the read path's own fallback does.
+ let locate_shard_size = if dat_file_size > 0 {
+ dat_file_size / data_shards as i64
+ } else {
+ (shard_size as i64 - 1).max(0)
+ };
// Read version from superblock (first byte of logical data)
let mut sb_buf = [0u8; SUPER_BLOCK_SIZE];
- read_from_data_shards(&shards, &mut sb_buf, 0, data_shards)?;
+ read_from_data_shards(
+ &shards,
+ &mut sb_buf,
+ 0,
+ data_shards,
+ locate_shard_size,
+ large_block,
+ small_block,
+ )?;
let version = Version(sb_buf[0]);
// Walk needles starting after superblock
@@ -475,10 +527,30 @@ pub fn rebuild_ecx_file(
let mut entries: Vec<(NeedleId, Offset, Size)> = Vec::new();
while offset + header_size as i64 <= total_data_size {
- // Read needle header (cookie + needle_id + size = 16 bytes)
+ // Read needle header (cookie + needle_id + size = 16 bytes).
+ // A read failure is NOT the end of the data — every offset in
+ // range maps into the shards, so an error means a truncated or
+ // unreadable shard. Publishing the entries collected so far as
+ // a successful .ecx would hand out a silently incomplete
+ // recovery index; propagate instead. (The scan still ends
+ // normally on the zero-cookie tail below.)
let mut header_buf = [0u8; NEEDLE_HEADER_SIZE];
- if read_from_data_shards(&shards, &mut header_buf, offset as u64, data_shards).is_err() {
- break;
+ if let Err(e) = read_from_data_shards(
+ &shards,
+ &mut header_buf,
+ offset as u64,
+ data_shards,
+ locate_shard_size,
+ large_block,
+ small_block,
+ ) {
+ for s in &mut shards {
+ s.close();
+ }
+ return Err(io::Error::new(
+ e.kind(),
+ format!("scan needle header at offset {}: {}", offset, e),
+ ));
}
let cookie = Cookie::from_bytes(&header_buf[..COOKIE_SIZE]);
@@ -532,58 +604,83 @@ pub fn rebuild_ecx_file(
Ok(())
}
-/// Read bytes from EC data shards at a logical offset in the .dat file.
+/// Read bytes from EC data shards at a logical offset in the .dat file,
+/// resolving the shard/offset through the volume's block layout via
+/// locate_data — the same mapping the read path uses.
+#[allow(clippy::too_many_arguments)]
fn read_from_data_shards(
shards: &[EcVolumeShard],
buf: &mut [u8],
logical_offset: u64,
data_shards: usize,
+ locate_shard_size: i64,
+ large_block_size: i64,
+ small_block_size: i64,
) -> io::Result<()> {
- let small_block = ERASURE_CODING_SMALL_BLOCK_SIZE as u64;
- let data_shards_u64 = data_shards as u64;
-
- let mut bytes_read = 0u64;
- let mut remaining = buf.len() as u64;
- let mut current_offset = logical_offset;
-
- while remaining > 0 {
- // Determine which shard and at what shard-offset this logical offset maps to.
- // The data is interleaved: large blocks first, then small blocks.
- // For simplicity, use the small block size for all calculations since
- // large blocks are multiples of small blocks.
- let row_size = small_block * data_shards_u64;
- let row_index = current_offset / row_size;
- let row_offset = current_offset % row_size;
- let shard_index = (row_offset / small_block) as usize;
- let shard_offset = row_index * small_block + (row_offset % small_block);
-
- if shard_index >= data_shards {
+ let intervals = crate::storage::erasure_coding::ec_locate::locate_data(
+ logical_offset as i64,
+ Size(buf.len() as i32),
+ locate_shard_size,
+ data_shards as u32,
+ large_block_size,
+ small_block_size,
+ );
+ let mut bytes_read = 0usize;
+ for interval in &intervals {
+ let (shard_id, shard_offset) =
+ interval.to_shard_id_and_offset(data_shards as u32, large_block_size, small_block_size);
+ if shard_id as usize >= data_shards {
return Err(io::Error::new(
io::ErrorKind::InvalidInput,
"shard index out of range",
));
}
-
- // How many bytes can we read from this position in this shard block
- let bytes_left_in_block = small_block - (row_offset % small_block);
- let to_read = remaining.min(bytes_left_in_block) as usize;
-
- let dest = &mut buf[bytes_read as usize..bytes_read as usize + to_read];
- shards[shard_index].read_at(dest, shard_offset)?;
-
- bytes_read += to_read as u64;
- remaining -= to_read as u64;
- current_offset += to_read as u64;
+ let to_read = interval.size as usize;
+ let dest = &mut buf[bytes_read..bytes_read + to_read];
+ // Exact-read semantics: read_at may legally return fewer bytes
+ // than requested, and treating a short read as complete leaves
+ // the tail of `dest` as whatever the buffer held before. Loop
+ // until filled; zero bytes inside the mapped range means the
+ // shard is truncated — an error, not an end.
+ let mut filled = 0usize;
+ while filled < to_read {
+ let n = shards[shard_id as usize]
+ .read_at(&mut dest[filled..], shard_offset as u64 + filled as u64)?;
+ if n == 0 {
+ return Err(io::Error::new(
+ io::ErrorKind::UnexpectedEof,
+ format!(
+ "short read from data shard {}: {} of {} bytes at offset {}",
+ shard_id, filled, to_read, shard_offset
+ ),
+ ));
+ }
+ filled += n;
+ }
+ bytes_read += to_read;
+ }
+ if bytes_read != buf.len() {
+ return Err(io::Error::new(
+ io::ErrorKind::UnexpectedEof,
+ "short read from data shards",
+ ));
}
-
Ok(())
}
+/// Buffer size for one encode sub-batch per shard, mirroring Go's 256KB
+/// bufferSize in WriteEcFiles. A block is processed in block_size/buffer_size
+/// sub-batches, so memory stays at total_shards * 256KB no matter how large
+/// the uniform block is.
+const ENCODE_BUFFER_SIZE: usize = 256 * 1024;
+
/// Encode the .dat file data into shard files.
///
/// Uses a two-phase approach matching Go's ec_encoder.go:
-/// 1. Process as many large blocks (1GB) as possible
-/// 2. Process remaining data with small blocks (1MB)
+/// 1. Process as many large blocks as possible
+/// 2. Process remaining data with small blocks
+///
+/// `buffer_size` must divide both block sizes.
#[allow(clippy::too_many_arguments)]
pub(crate) fn encode_dat_file(
dat_file: &File,
@@ -593,44 +690,50 @@ pub(crate) fn encode_dat_file(
builders: &mut [ShardChecksumBuilder],
data_shards: usize,
parity_shards: usize,
+ buffer_size: usize,
large_block_size: usize,
small_block_size: usize,
) -> io::Result<()> {
+ let total_shards = data_shards + parity_shards;
+ let mut buffers: Vec> = (0..total_shards)
+ .map(|_| vec![0u8; buffer_size])
+ .collect();
+
let mut remaining = dat_size;
let mut offset: u64 = 0;
- // Phase 1: Process large blocks (1GB each) while enough data remains
+ // Phase 1: process whole large-block rows while enough data remains
let large_row_size = large_block_size * data_shards;
while remaining >= large_row_size as i64 {
- encode_one_batch(
+ encode_data(
dat_file,
offset,
large_block_size,
rs,
+ &mut buffers,
shards,
builders,
data_shards,
- parity_shards,
)?;
offset += large_row_size as u64;
remaining -= large_row_size as i64;
}
- // Phase 2: Process remaining data with small blocks (1MB each)
+ // Phase 2: process remaining data with small blocks
let small_row_size = small_block_size * data_shards;
while remaining > 0 {
let to_process = remaining.min(small_row_size as i64);
- encode_one_batch(
+ encode_data(
dat_file,
offset,
small_block_size,
rs,
+ &mut buffers,
shards,
builders,
data_shards,
- parity_shards,
)?;
offset += to_process as u64;
remaining -= to_process;
@@ -639,61 +742,71 @@ pub(crate) fn encode_dat_file(
Ok(())
}
-/// Encode one batch (row) of data.
+/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches so
+/// arbitrarily large blocks never require block-sized allocations. Mirrors
+/// Go's encodeData.
+#[allow(clippy::too_many_arguments)]
+fn encode_data(
+ dat_file: &File,
+ row_offset: u64,
+ block_size: usize,
+ rs: &ReedSolomon,
+ buffers: &mut [Vec],
+ shards: &mut [EcVolumeShard],
+ builders: &mut [ShardChecksumBuilder],
+ data_shards: usize,
+) -> io::Result<()> {
+ let buffer_size = buffers[0].len();
+ if block_size % buffer_size != 0 {
+ return Err(io::Error::new(
+ io::ErrorKind::InvalidInput,
+ format!(
+ "unexpected block size {} buffer size {}",
+ block_size, buffer_size
+ ),
+ ));
+ }
+ let batch_count = block_size / buffer_size;
+ for b in 0..batch_count {
+ encode_one_batch(
+ dat_file,
+ row_offset + (b * buffer_size) as u64,
+ block_size,
+ rs,
+ buffers,
+ shards,
+ builders,
+ data_shards,
+ )?;
+ }
+ Ok(())
+}
+
+/// Encode one sub-batch: the same buffer-sized slice of every shard's block in
+/// this row. Mirrors Go's encodeDataOneBatch.
+#[allow(clippy::too_many_arguments)]
fn encode_one_batch(
dat_file: &File,
offset: u64,
block_size: usize,
rs: &ReedSolomon,
+ buffers: &mut [Vec],
shards: &mut [EcVolumeShard],
builders: &mut [ShardChecksumBuilder],
data_shards: usize,
- parity_shards: usize,
) -> io::Result<()> {
- let total_shards = data_shards + parity_shards;
- // Each batch allocates block_size * total_shards bytes.
- // With large blocks (1 GiB) this is 14 GiB -- guard against OOM.
- let total_alloc = block_size.checked_mul(total_shards).ok_or_else(|| {
- io::Error::new(
- io::ErrorKind::InvalidInput,
- "block_size * shard count overflows usize",
- )
- })?;
- // Large-block encoding uses 1 GiB * 14 shards = 14 GiB; allow up to 16 GiB.
- const MAX_BATCH_ALLOC: usize = 16 * 1024 * 1024 * 1024; // 16 GiB safety limit
- if total_alloc > MAX_BATCH_ALLOC {
- return Err(io::Error::new(
- io::ErrorKind::InvalidInput,
- format!(
- "batch allocation too large ({} bytes, limit {} bytes); block_size={} shards={}",
- total_alloc, MAX_BATCH_ALLOC, block_size, total_shards,
- ),
- ));
- }
-
- // Allocate buffers for all shards
- let mut buffers: Vec> = (0..total_shards).map(|_| vec![0u8; block_size]).collect();
-
- // Read data shards from .dat file
+ // Read data shards from the .dat file, zero-filling past EOF — the buffers
+ // are reused across batches, so the tail must be cleared explicitly.
for i in 0..data_shards {
let read_offset = offset + (i * block_size) as u64;
-
- #[cfg(unix)]
- {
- use std::os::unix::fs::FileExt;
- dat_file.read_at(&mut buffers[i], read_offset)?;
- }
-
- #[cfg(not(unix))]
- {
- let mut f = dat_file.try_clone()?;
- f.seek(SeekFrom::Start(read_offset))?;
- f.read(&mut buffers[i])?;
+ let n = read_at_most(dat_file, &mut buffers[i], read_offset)?;
+ for b in buffers[i][n..].iter_mut() {
+ *b = 0;
}
}
// Encode parity shards
- rs.encode(&mut buffers).map_err(|e| {
+ rs.encode(&mut *buffers).map_err(|e| {
io::Error::new(
io::ErrorKind::Other,
format!("reed-solomon encode: {:?}", e),
@@ -710,6 +823,29 @@ fn encode_one_batch(
Ok(())
}
+/// Read into `buf` at `offset` until it is full or EOF; returns bytes read.
+fn read_at_most(dat_file: &File, buf: &mut [u8], offset: u64) -> io::Result {
+ let mut n = 0;
+ while n < buf.len() {
+ #[cfg(unix)]
+ let r = {
+ use std::os::unix::fs::FileExt;
+ dat_file.read_at(&mut buf[n..], offset + n as u64)?
+ };
+ #[cfg(not(unix))]
+ let r = {
+ let mut f = dat_file.try_clone()?;
+ f.seek(SeekFrom::Start(offset + n as u64))?;
+ f.read(&mut buf[n..])?
+ };
+ if r == 0 {
+ break;
+ }
+ n += r;
+ }
+ Ok(n)
+}
+
#[cfg(test)]
mod tests {
use super::*;
@@ -1020,7 +1156,7 @@ mod tests {
// Without additional_dirs, rebuild must fail: shards 1, 3, 6 are not
// in primary and the full logical .dat content can't be reconstructed.
- let res = rebuild_ecx_file(&primary, "", VolumeId(1), 10, &[]);
+ let res = rebuild_ecx_file(&primary, "", VolumeId(1), 10, 0, 0, &[]);
assert!(
res.is_err(),
"ecx rebuild without additional_dirs must fail when data shards are on another disk"
@@ -1031,7 +1167,7 @@ mod tests {
);
// With additional_dirs pointing at the secondary, the rebuild must succeed.
- rebuild_ecx_file(&primary, "", VolumeId(1), 10, &[secondary.as_str()]).unwrap();
+ rebuild_ecx_file(&primary, "", VolumeId(1), 10, 0, 0, &[secondary.as_str()]).unwrap();
assert!(
std::path::Path::new(&ecx_path).exists(),
@@ -1043,6 +1179,118 @@ mod tests {
);
}
+ // A uniform-layout volume (block size > 1MiB) must have its .ecx rebuilt
+ // through the recorded geometry; the legacy 1MiB mapping would scan
+ // garbage past the first block boundary.
+ #[test]
+ fn test_rebuild_ecx_file_uniform_layout() {
+ use crate::storage::needle_map::NeedleMapKind;
+ use crate::storage::volume::Volume;
+ let tmp = TempDir::new().unwrap();
+ let dir = tmp.path().to_str().unwrap().to_string();
+ let mut v = Volume::new(
+ &dir,
+ &dir,
+ "",
+ VolumeId(2),
+ NeedleMapKind::InMemory,
+ None,
+ None,
+ 0,
+ Version::current(),
+ )
+ .unwrap();
+ for i in 1u64..=12 {
+ let data: Vec = (0..2 << 20)
+ .map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8)
+ .collect();
+ let mut n = Needle {
+ id: NeedleId(i),
+ cookie: Cookie(i as u32),
+ data: data.clone(),
+ data_size: data.len() as u32,
+ ..Needle::default()
+ };
+ v.write_needle(&mut n, true, false).unwrap();
+ }
+ v.sync_to_disk().unwrap();
+ v.close();
+ let block_size = write_ec_files(&dir, &dir, "", VolumeId(2), 10, 4).unwrap();
+ assert!(
+ block_size > ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ "fixture must diverge from the legacy layout"
+ );
+
+ let ecx_path = format!("{}/2.ecx", dir);
+ let canonical = std::fs::read(&ecx_path).unwrap();
+ std::fs::remove_file(&ecx_path).unwrap();
+
+ rebuild_ecx_file(&dir, "", VolumeId(2), 10, block_size, 0, &[]).unwrap();
+ let rebuilt = std::fs::read(&ecx_path).unwrap();
+ assert_eq!(canonical, rebuilt, "rebuilt .ecx must match the encode-time .ecx");
+ }
+
+ // A truncated data shard must FAIL the .ecx rebuild, not publish the
+ // entries scanned so far as a successful (silently incomplete) index.
+ #[test]
+ fn test_rebuild_ecx_file_fails_on_truncated_shard() {
+ use crate::storage::needle_map::NeedleMapKind;
+ use crate::storage::volume::Volume;
+ let tmp = TempDir::new().unwrap();
+ let dir = tmp.path().to_str().unwrap().to_string();
+ let mut v = Volume::new(
+ &dir,
+ &dir,
+ "",
+ VolumeId(3),
+ NeedleMapKind::InMemory,
+ None,
+ None,
+ 0,
+ Version::current(),
+ )
+ .unwrap();
+ for i in 1u64..=12 {
+ let data: Vec = (0..2 << 20)
+ .map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8)
+ .collect();
+ let mut n = Needle {
+ id: NeedleId(i),
+ cookie: Cookie(i as u32),
+ data: data.clone(),
+ data_size: data.len() as u32,
+ ..Needle::default()
+ };
+ v.write_needle(&mut n, true, false).unwrap();
+ }
+ v.sync_to_disk().unwrap();
+ v.close();
+ let block_size = write_ec_files(&dir, &dir, "", VolumeId(3), 10, 4).unwrap();
+
+ let ecx_path = format!("{}/3.ecx", dir);
+ std::fs::remove_file(&ecx_path).unwrap();
+
+ // Truncate shard 0 to just the superblock: the scan's very first
+ // needle-header read (offset SUPER_BLOCK_SIZE, shard 0 under the
+ // uniform layout) lands in the missing region. The pre-fix code
+ // broke the scan there and published an EMPTY .ecx as success.
+ let shard_path = format!("{}/3.ec00", dir);
+ let f = std::fs::OpenOptions::new()
+ .write(true)
+ .open(&shard_path)
+ .unwrap();
+ f.set_len(crate::storage::super_block::SUPER_BLOCK_SIZE as u64)
+ .unwrap();
+ drop(f);
+
+ let res = rebuild_ecx_file(&dir, "", VolumeId(3), 10, block_size, 0, &[]);
+ assert!(res.is_err(), "rebuild over a truncated shard must fail");
+ assert!(
+ !std::path::Path::new(&ecx_path).exists(),
+ "a failed rebuild must not leave a partial .ecx behind"
+ );
+ }
+
#[test]
fn test_reed_solomon_basic() {
let data_shards = 10;
diff --git a/seaweed-volume/src/storage/erasure_coding/ec_locate.rs b/seaweed-volume/src/storage/erasure_coding/ec_locate.rs
index 4c1f06aa2..baaed9f3b 100644
--- a/seaweed-volume/src/storage/erasure_coding/ec_locate.rs
+++ b/seaweed-volume/src/storage/erasure_coding/ec_locate.rs
@@ -18,21 +18,26 @@ pub struct Interval {
}
impl Interval {
- pub fn to_shard_id_and_offset(&self, data_shards: u32) -> (ShardId, i64) {
+ pub fn to_shard_id_and_offset(
+ &self,
+ data_shards: u32,
+ large_block_size: i64,
+ small_block_size: i64,
+ ) -> (ShardId, i64) {
let data_shards_usize = data_shards as usize;
let shard_id = (self.block_index % data_shards_usize) as ShardId;
let row_index = self.block_index / data_shards_usize;
let block_size = if self.is_large_block {
- ERASURE_CODING_LARGE_BLOCK_SIZE as i64
+ large_block_size
} else {
- ERASURE_CODING_SMALL_BLOCK_SIZE as i64
+ small_block_size
};
let mut offset = row_index as i64 * block_size + self.inner_block_offset;
if !self.is_large_block {
// Small blocks come after large blocks in the shard file
- offset += self.large_block_rows_count as i64 * ERASURE_CODING_LARGE_BLOCK_SIZE as i64;
+ offset += self.large_block_rows_count as i64 * large_block_size;
}
(shard_id, offset)
@@ -42,7 +47,14 @@ impl Interval {
/// Locate the EC shard intervals needed to read data at the given offset and size.
///
/// `shard_size` is the size of a single shard file.
-pub fn locate_data(offset: i64, size: Size, shard_size: i64, data_shards: u32) -> Vec {
+pub fn locate_data(
+ offset: i64,
+ size: Size,
+ shard_size: i64,
+ data_shards: u32,
+ large_block_size: i64,
+ small_block_size: i64,
+) -> Vec {
let mut intervals = Vec::new();
let data_size = size.0 as i64;
@@ -50,17 +62,14 @@ pub fn locate_data(offset: i64, size: Size, shard_size: i64, data_shards: u32) -
return intervals;
}
- let large_block_size = ERASURE_CODING_LARGE_BLOCK_SIZE as i64;
- let small_block_size = ERASURE_CODING_SMALL_BLOCK_SIZE as i64;
let large_row_size = large_block_size * data_shards as i64;
let small_row_size = small_block_size * data_shards as i64;
- // Number of large block rows
- let n_large_block_rows = if shard_size > 0 {
- ((shard_size - 1) / large_block_size) as usize
- } else {
- 0
- };
+ // Number of large block rows. Mirrors Go's shardDatSize/largeBlockLength:
+ // the caller's ecd-size fallback already subtracts 1 to disambiguate the
+ // exact-multiple case, so no further -1 here — a shard size that IS an
+ // exact multiple (dat_file_size path) means real full large rows.
+ let n_large_block_rows = (shard_size / large_block_size) as usize;
let large_section_size = n_large_block_rows as i64 * large_row_size;
let mut remaining_offset = offset;
@@ -150,7 +159,11 @@ mod tests {
is_large_block: true,
large_block_rows_count: 1,
};
- let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards);
+ let (shard_id, offset) = interval.to_shard_id_and_offset(
+ data_shards,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
assert_eq!(shard_id, 0);
assert_eq!(offset, 100);
@@ -162,7 +175,11 @@ mod tests {
is_large_block: true,
large_block_rows_count: 1,
};
- let (shard_id, _offset) = interval.to_shard_id_and_offset(data_shards);
+ let (shard_id, _offset) = interval.to_shard_id_and_offset(
+ data_shards,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
assert_eq!(shard_id, 5);
// Block index 12 (data_shards=10) → row_index 1, shard_id 2
@@ -173,7 +190,11 @@ mod tests {
is_large_block: true,
large_block_rows_count: 5,
};
- let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards);
+ let (shard_id, offset) = interval.to_shard_id_and_offset(
+ data_shards,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
assert_eq!(shard_id, 2); // 12 % 10 = 2
assert_eq!(offset, large_block_size + 200); // row 1 offset + inner_block_offset
@@ -185,7 +206,11 @@ mod tests {
is_large_block: true,
large_block_rows_count: 2,
};
- let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards);
+ let (shard_id, offset) = interval.to_shard_id_and_offset(
+ data_shards,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
assert_eq!(shard_id, 0);
assert_eq!(offset, ERASURE_CODING_LARGE_BLOCK_SIZE as i64); // row 1 offset
}
@@ -193,7 +218,14 @@ mod tests {
#[test]
fn test_locate_data_small_file() {
// Small file: 100 bytes at offset 50, shard size = 1MB
- let intervals = locate_data(50, Size(100), 1024 * 1024, 10);
+ let intervals = locate_data(
+ 50,
+ Size(100),
+ 1024 * 1024,
+ 10,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
assert!(!intervals.is_empty());
// Should be a single small block interval (no large block rows for 1MB shard)
@@ -203,7 +235,14 @@ mod tests {
#[test]
fn test_locate_data_empty() {
- let intervals = locate_data(0, Size(0), 1024 * 1024, 10);
+ let intervals = locate_data(
+ 0,
+ Size(0),
+ 1024 * 1024,
+ 10,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
assert!(intervals.is_empty());
}
@@ -216,7 +255,11 @@ mod tests {
is_large_block: false,
large_block_rows_count: 2,
};
- let (_shard_id, offset) = interval.to_shard_id_and_offset(10);
+ let (_shard_id, offset) = interval.to_shard_id_and_offset(
+ 10,
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ );
// Should be after 2 large block rows
assert_eq!(offset, 2 * ERASURE_CODING_LARGE_BLOCK_SIZE as i64);
}
diff --git a/seaweed-volume/src/storage/erasure_coding/ec_volume.rs b/seaweed-volume/src/storage/erasure_coding/ec_volume.rs
index 709d8f485..34c12aaa9 100644
--- a/seaweed-volume/src/storage/erasure_coding/ec_volume.rs
+++ b/seaweed-volume/src/storage/erasure_coding/ec_volume.rs
@@ -26,6 +26,9 @@ pub struct EcVolume {
pub dat_file_size: i64,
pub data_shards: u32,
pub parity_shards: u32,
+ /// Uniform block layout: each shard is one contiguous block of this many
+ /// bytes. 0 = legacy 1GiB/1MiB two-tier layout. Loaded from .vif.
+ pub block_size: i64,
ecx_file: Option,
ecx_file_size: i64,
ecj_file: Option,
@@ -101,34 +104,249 @@ fn locate_vif_path(dir: &str, dir_idx: &str, collection: &str, volume_id: Volume
data_vif
}
-/// Read EC data/parity shard counts from `.vif`, defaulting to the
-/// build's standard ratio when no `.vif` is present or is malformed.
+/// Load the volume's `.vif` (data dir first, then idx dir, then the active
+/// generation's versioned sidecar — see [`locate_vif_path`]). Absent
+/// everywhere is legal — legacy volumes predate the sidecar — and returns
+/// `None`; so does a zero-byte stub (an ec.decode copy from a source
+/// without one), mirroring Go's MaybeLoadVolumeInfo. A
+/// present-but-unreadable or malformed `.vif` is an ERROR: silently
+/// defaulting would mount a uniform-layout volume with legacy offset math
+/// and serve wrong bytes with a straight face.
+pub fn load_vif_info(
+ dir: &str,
+ dir_idx: &str,
+ collection: &str,
+ volume_id: VolumeId,
+) -> io::Result> {
+ let vif_path = locate_vif_path(dir, dir_idx, collection, volume_id);
+ match std::fs::read_to_string(&vif_path) {
+ Ok(content) if content.trim().is_empty() => Ok(None),
+ Ok(content) => serde_json::from_str(&content).map(Some).map_err(|e| {
+ io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!("parse {}: {}", vif_path, e),
+ )
+ }),
+ Err(e) if e.kind() == io::ErrorKind::NotFound => Ok(None),
+ Err(e) => Err(io::Error::new(
+ e.kind(),
+ format!("read {}: {}", vif_path, e),
+ )),
+ }
+}
+
+/// Read the EC data/parity shard counts and shard block size from `.vif`,
+/// defaulting to the build's standard ratio and the legacy block layout when
+/// no `.vif` is present. A present-but-unreadable or malformed `.vif`
+/// fails instead of defaulting (see [`load_vif_info`]).
/// Looks at the data dir first, then the idx dir — see [`locate_vif_path`].
pub fn read_ec_shard_config(
dir: &str,
dir_idx: &str,
collection: &str,
volume_id: VolumeId,
-) -> (u32, u32) {
- let mut data_shards = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT as u32;
- let mut parity_shards = crate::storage::erasure_coding::ec_shard::PARITY_SHARDS_COUNT as u32;
- let vif_path = locate_vif_path(dir, dir_idx, collection, volume_id);
- if let Ok(vif_content) = std::fs::read_to_string(&vif_path) {
- if let Ok(vif_info) =
- serde_json::from_str::(&vif_content)
- {
- if let Some(ec) = vif_info.ec_shard_config {
- if ec.data_shards > 0
- && ec.parity_shards > 0
- && (ec.data_shards + ec.parity_shards) <= MAX_SHARD_COUNT as u32
- {
- data_shards = ec.data_shards;
- parity_shards = ec.parity_shards;
- }
+) -> io::Result<(u32, u32, i64)> {
+ let vif = load_vif_info(dir, dir_idx, collection, volume_id)?;
+ ec_shard_config_from(vif.as_ref(), &[dir, dir_idx], collection, volume_id)
+}
+
+/// Load the volume's `.vif` from the selected location or, failing that, from
+/// any sibling disk — returning the directory it came from so the caller can
+/// resolve the rest of the volume's metadata against the same place. A rebuild
+/// picks one disk to write into, but a multi-disk server may hold that
+/// volume's metadata on another.
+pub fn load_vif_info_across_dirs(
+ dir: &str,
+ dir_idx: &str,
+ other_dirs: &[String],
+ collection: &str,
+ volume_id: VolumeId,
+) -> io::Result> {
+ if let Some(vif) = load_vif_info(dir, dir_idx, collection, volume_id)? {
+ // Report where it actually came from. load_vif_info probes the data
+ // directory then the index directory, so naming `dir` unconditionally
+ // pointed callers at the wrong disk whenever the vif lived with the
+ // index — and a caller resolving the rest of the volume's metadata
+ // against that answer would look in a directory holding none of it.
+ let data_vif = format!(
+ "{}.vif",
+ crate::storage::volume::volume_file_name(dir, collection, volume_id)
+ );
+ let found_in = if std::path::Path::new(&data_vif).exists() {
+ dir.to_string()
+ } else {
+ dir_idx.to_string()
+ };
+ return Ok(Some((vif, found_in)));
+ }
+ for other in other_dirs {
+ if let Some(vif) = load_vif_info(other, other, collection, volume_id)? {
+ return Ok(Some((vif, other.clone())));
+ }
+ }
+ Ok(None)
+}
+
+/// [`read_ec_shard_config`] for a caller that may find the volume's metadata on
+/// any of the server's disks. Without this a rebuild that lands on a disk
+/// holding only shards reads the default 10+4 and the legacy block layout,
+/// then reconstructs a custom-ratio or uniform volume through the wrong matrix
+/// and de-striping geometry.
+pub fn read_ec_shard_config_across_dirs(
+ dir: &str,
+ dir_idx: &str,
+ other_dirs: &[String],
+ collection: &str,
+ volume_id: VolumeId,
+) -> io::Result<(u32, u32, i64)> {
+ // Every directory this volume's metadata could be in. A split -dir/-dir.idx
+ // location keeps .vif and .ecsum with the INDEX, and callers exclude their
+ // own index directory from other_dirs on the understanding that it is
+ // passed here separately — so it is chained explicitly, exactly as the .vif
+ // lookup and Go's findBitrotSidecar both do.
+ let mut candidates: Vec<&str> = Vec::with_capacity(2 + other_dirs.len());
+ candidates.push(dir);
+ if !dir_idx.is_empty() && dir_idx != dir {
+ candidates.push(dir_idx);
+ }
+ candidates.extend(other_dirs.iter().map(|s| s.as_str()));
+
+ // A vif that carries no ecShardConfig answers nothing about the layout, so
+ // it must NOT short-circuit the sidecar search: a legacy config-free vif
+ // and the generation-0 sidecar can sit in different directories, and
+ // stopping at the vif resolved a 12+4 uniform volume as 10+4 legacy. Pass
+ // the whole candidate list so the fallback covers the same ground the vif
+ // lookup did.
+ let vif = load_vif_info_across_dirs(dir, dir_idx, other_dirs, collection, volume_id)?;
+ ec_shard_config_from(
+ vif.as_ref().map(|(v, _)| v),
+ &candidates,
+ collection,
+ volume_id,
+ )
+}
+
+/// Resolve (data_shards, parity_shards, block_size) from an already-loaded
+/// `.vif`. With no vif — or one carrying no EC config — the bitrot sidecar
+/// records the same config at encode time and answers the layout question the
+/// vif cannot: defaulting a uniform-layout volume to the legacy block sizes
+/// maps every read to the wrong shard offset. `weed fix -ecx` reads the
+/// sidecar for the same reason. Absent both, the build's standard ratio and
+/// the legacy layout.
+pub fn ec_shard_config_from(
+ vif: Option<&crate::storage::volume::VifVolumeInfo>,
+ dirs: &[&str],
+ collection: &str,
+ volume_id: VolumeId,
+) -> io::Result<(u32, u32, i64)> {
+ let default_ds = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT as u32;
+ let default_ps = crate::storage::erasure_coding::ec_shard::PARITY_SHARDS_COUNT as u32;
+ // Sum as u64: counts near the u32 ceiling would wrap and pass the bound.
+ let usable = |ds: u32, ps: u32| {
+ ds > 0 && ps > 0 && (ds as u64 + ps as u64) <= MAX_SHARD_COUNT as u64
+ };
+
+ if let Some(ec) = vif.and_then(|v| v.ec_shard_config.as_ref()) {
+ // A config that is PRESENT but records an impossible ratio is not a
+ // volume to fall back on: dropping through to the sidecar or the
+ // defaults would read uniform shards with the legacy offset math and
+ // answer with the wrong bytes. Only an ENTIRELY absent config means
+ // "this predates the record".
+ if !usable(ec.data_shards, ec.parity_shards) {
+ return Err(io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!(
+ "vif for volume {} records invalid shard counts {}+{}",
+ volume_id.0, ec.data_shards, ec.parity_shards
+ ),
+ ));
+ }
+ // A recorded block size that no encoder could have produced maps
+ // every read to the wrong shard offset. Refuse the mount rather
+ // than serve those bytes or silently pick a layout.
+ validate_block_size(ec.block_size)?;
+ return Ok((ec.data_shards, ec.parity_shards, ec.block_size));
+ }
+ // With no usable vif config the sidecar is the ONLY record of this volume's
+ // layout, so the three answers stay distinct: absent EVERYWHERE means the
+ // volume may genuinely predate the sidecar and legacy is the right guess;
+ // present-but-unusable means the record is corrupt, and answering reads
+ // from a guessed layout returns wrong bytes rather than none.
+ //
+ // Every candidate directory is searched, not just the first: a split
+ // -dir/-dir.idx location keeps the sidecar with the INDEX, and a multi-disk
+ // server may keep it on a sibling. Stopping at the data directory resolved
+ // a 12+4 uniform volume as 10+4 legacy.
+ let mut path = String::new();
+ for candidate in dirs.iter().filter(|d| !d.is_empty()) {
+ let base = crate::storage::volume::volume_file_name(candidate, collection, volume_id);
+ let probe = crate::storage::erasure_coding::ec_bitrot::bitrot_sidecar_path(&base, 0);
+ match std::fs::metadata(&probe) {
+ Err(e) if e.kind() == io::ErrorKind::NotFound => continue,
+ Err(e) => {
+ return Err(io::Error::new(e.kind(), format!("stat {}: {}", probe, e)));
+ }
+ Ok(_) => {
+ path = probe;
+ break;
}
}
}
- (data_shards, parity_shards)
+ if path.is_empty() {
+ return Ok((default_ds, default_ps, 0));
+ }
+ let prot = crate::storage::erasure_coding::ec_bitrot::load_bitrot_sidecar(&path)
+ .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, format!("read {}: {}", path, e)))?;
+ // The un-suffixed sidecar describes generation 0 and nothing else.
+ if prot.generation != 0 {
+ return Err(io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!("{} records generation {}, not generation 0", path, prot.generation),
+ ));
+ }
+ let ec = prot.ec_shard_config.as_ref().ok_or_else(|| {
+ io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!("{} records no EC config", path),
+ )
+ })?;
+ if !usable(ec.data_shards, ec.parity_shards) {
+ return Err(io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!(
+ "{} records invalid shard counts {}+{}",
+ path, ec.data_shards, ec.parity_shards
+ ),
+ ));
+ }
+ validate_block_size(ec.block_size)
+ .map_err(|e| io::Error::new(e.kind(), format!("{}: {}", path, e)))?;
+ Ok((ec.data_shards, ec.parity_shards, ec.block_size))
+}
+
+/// Reports whether a `.vif`-recorded shard block size is one an encoder could
+/// have produced. 0 means the legacy two-tier layout, which is always valid;
+/// anything positive must be a whole number of small blocks, because that is
+/// what `uniform_block_size` rounds to. Mirrors Go's `ValidateBlockSize`.
+pub fn validate_block_size(block_size: i64) -> io::Result<()> {
+ if block_size == 0 {
+ return Ok(());
+ }
+ if block_size < 0
+ || block_size
+ % crate::storage::erasure_coding::ec_shard::ERASURE_CODING_SMALL_BLOCK_SIZE as i64
+ != 0
+ {
+ return Err(io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!(
+ "invalid shard block size {}: expected 0 (legacy) or a multiple of {}",
+ block_size,
+ crate::storage::erasure_coding::ec_shard::ERASURE_CODING_SMALL_BLOCK_SIZE
+ ),
+ ));
+ }
+ Ok(())
}
impl EcVolume {
@@ -139,7 +357,14 @@ impl EcVolume {
collection: &str,
volume_id: VolumeId,
) -> io::Result {
- let (data_shards, parity_shards) = read_ec_shard_config(dir, dir_idx, collection, volume_id);
+ // One load of the volume's `.vif`, used for both the shard config and
+ // the version / dat-size fields below.
+ let vif = load_vif_info(dir, dir_idx, collection, volume_id)?;
+ // Both directories: a split -dir/-dir.idx layout keeps the .ecsum with
+ // the INDEX, and it is the only layout record a volume whose vif is
+ // absent or config-free still has.
+ let (data_shards, parity_shards, block_size) =
+ ec_shard_config_from(vif.as_ref(), &[dir, dir_idx], collection, volume_id)?;
let total_shards = (data_shards + parity_shards) as usize;
let mut shards = Vec::with_capacity(total_shards);
@@ -155,12 +380,11 @@ impl EcVolume {
// for shard-size math and by the Store-level prune in
// `store_ec_reconcile.rs` to verify a sibling-disk .dat is
// plausibly the encoding source (#9478).
- let (expire_at_sec, vif_version, vif_dat_file_size, encode_ts_ns) = {
- let vif_path = locate_vif_path(dir, dir_idx, collection, volume_id);
- if let Ok(vif_content) = std::fs::read_to_string(&vif_path) {
- if let Ok(vif_info) =
- serde_json::from_str::(&vif_content)
- {
+ let (expire_at_sec, vif_version, vif_dat_file_size, encode_ts_ns) =
+ // Absent everywhere = legacy defaults; a present-but-unreadable or
+ // malformed vif already failed the mount in load_vif_info above.
+ match vif.as_ref() {
+ Some(vif_info) => {
let ver = if vif_info.version > 0 {
Version(vif_info.version as u8)
} else {
@@ -176,13 +400,9 @@ impl EcVolume {
vif_info.dat_file_size,
cfg_encode_ts_ns,
)
- } else {
- (0, Version::current(), 0, 0)
}
- } else {
- (0, Version::current(), 0, 0)
- }
- };
+ None => (0, Version::current(), 0, 0),
+ };
let mut vol = EcVolume {
volume_id,
@@ -194,6 +414,7 @@ impl EcVolume {
dat_file_size: vif_dat_file_size,
data_shards,
parity_shards,
+ block_size,
ecx_file: None,
ecx_file_size: 0,
ecj_file: None,
@@ -254,17 +475,33 @@ impl EcVolume {
// Seed the in-memory deleted set from the journal.
vol.load_deleted_needles_from_ecj()?;
- // Load the generation-0 EC bitrot checksum sidecar (optional; best-effort).
- vol.load_active_bitrot_sidecar();
+ // Load the generation-0 EC bitrot checksum sidecar. Optional, except
+ // when it contradicts the volume's own geometry — see
+ // load_bitrot_for_generation.
+ vol.load_active_bitrot_sidecar(&[])?;
Ok(vol)
}
+ /// Re-resolve the checksum sidecar for a volume that is already mounted. A
+ /// shard delivery can bring the manifest with it, and the receive path only
+ /// writes the file, so without this the in-memory volume keeps the
+ /// protection state it resolved at mount (off) until a remount.
+ pub fn reload_bitrot_sidecar(&mut self, additional_dirs: &[String]) {
+ if let Err(e) = self.load_active_bitrot_sidecar(additional_dirs) {
+ tracing::warn!(
+ volume_id = self.volume_id.0,
+ error = %e,
+ "reload bitrot sidecar",
+ );
+ }
+ }
+
/// Load the generation-0 checksum sidecar into `self.bitrot`/`self.bitrot_status`.
/// OSS only produces generation-0 (fresh-encode) sidecars, mirroring Go's
/// `loadActiveBitrotSidecar`.
- fn load_active_bitrot_sidecar(&mut self) {
- self.load_bitrot_for_generation(0);
+ fn load_active_bitrot_sidecar(&mut self, additional_dirs: &[String]) -> io::Result<()> {
+ self.load_bitrot_for_generation(0, additional_dirs)
}
/// Load and validate the sidecar describing `generation`, setting
@@ -272,11 +509,60 @@ impl EcVolume {
/// => `Off` (protection off, not corruption); self-integrity or manifest
/// failure => `Invalid` with a warning (protection off pending repair); usable
/// => `On`. Mirrors Go's `loadBitrotForGeneration`.
- fn load_bitrot_for_generation(&mut self, generation: u32) {
+ fn load_bitrot_for_generation(&mut self, generation: u32, additional_dirs: &[String]) -> io::Result<()> {
use crate::storage::erasure_coding::ec_bitrot;
- let base = self.base_name();
- let path = ec_bitrot::bitrot_sidecar_path(&base, generation);
+ // Data base then index base, matching Go's findBitrotSidecar. A split
+ // -dir/-dir.idx location keeps the sidecar with the INDEX, and on a
+ // multi-disk server sharing one index directory that is the only copy
+ // the per-disk runtimes other than the first can see — searching the
+ // data base alone left them reporting no protection at all.
+ let data_path = ec_bitrot::bitrot_sidecar_path(&self.base_name(), generation);
+ let idx_path = ec_bitrot::bitrot_sidecar_path(&self.idx_base_name(), generation);
+ // Sibling disks last: startup mirroring gives each shard-bearing disk
+ // its own .ecx/.ecj/.vif but not the sidecar, and a delivery lands
+ // exactly one copy, so a runtime restricted to its own two directories
+ // would report no protection however often it reloaded.
+ let path = std::iter::once(data_path.clone())
+ .chain(std::iter::once(idx_path))
+ .chain(additional_dirs.iter().filter(|d| !d.is_empty()).map(|d| {
+ ec_bitrot::bitrot_sidecar_path(
+ &crate::storage::volume::volume_file_name(d, &self.collection, self.volume_id),
+ generation,
+ )
+ }))
+ .find(|p| std::path::Path::new(p).exists())
+ .unwrap_or(data_path);
let loaded = ec_bitrot::load_bitrot_sidecar(&path);
+ // A sidecar written for THIS generation that contradicts the volume's
+ // geometry is not "no protection" — it says the layout the volume is
+ // about to serve reads with is wrong. Fail the mount.
+ if let Ok(prot) = &loaded {
+ if prot.generation == generation
+ && !ec_bitrot::geometry_matches(
+ prot,
+ self.data_shards as usize,
+ self.parity_shards as usize,
+ self.block_size,
+ )
+ {
+ let cfg = prot.ec_shard_config.as_ref();
+ return Err(io::Error::new(
+ io::ErrorKind::InvalidData,
+ format!(
+ "ec volume {} generation {}: {} records layout {}+{} block {} but the volume is mounted as {}+{} block {}; refusing to serve one of the two layouts",
+ self.volume_id.0,
+ generation,
+ path,
+ cfg.map(|c| c.data_shards).unwrap_or(0),
+ cfg.map(|c| c.parity_shards).unwrap_or(0),
+ cfg.map(|c| c.block_size).unwrap_or(0),
+ self.data_shards,
+ self.parity_shards,
+ self.block_size,
+ ),
+ ));
+ }
+ }
let status = ec_bitrot::resolve_status(
&loaded,
generation,
@@ -297,6 +583,7 @@ impl EcVolume {
);
}
}
+ Ok(())
}
/// The active-generation bitrot protection AND its status (cached at mount),
@@ -765,6 +1052,26 @@ impl EcVolume {
Ok(None)
}
+ /// Large-block length of this volume's shard layout (the uniform block
+ /// size when set, the legacy 1GiB otherwise). Mirrors Go's
+ /// ECContext.LargeBlockSize.
+ pub fn large_block_size(&self) -> i64 {
+ if self.block_size > 0 {
+ self.block_size
+ } else {
+ ERASURE_CODING_LARGE_BLOCK_SIZE as i64
+ }
+ }
+
+ /// Small-block length of this volume's shard layout.
+ pub fn small_block_size(&self) -> i64 {
+ if self.block_size > 0 {
+ self.block_size
+ } else {
+ ERASURE_CODING_SMALL_BLOCK_SIZE as i64
+ }
+ }
+
/// Locate the EC shard intervals needed to read a needle.
/// Locate the EC shard intervals covering a needle at `actual_offset` whose
/// index size is `size`. Mirrors Go's EcVolume.LocateEcShardNeedleInterval.
@@ -783,7 +1090,27 @@ impl EcVolume {
};
// locate_data wants the on-disk size (header+body+checksum+timestamp+padding).
let actual = get_actual_size(size, self.version);
- ec_locate::locate_data(actual_offset, Size(actual as i32), shard_size, self.data_shards)
+ ec_locate::locate_data(
+ actual_offset,
+ Size(actual as i32),
+ shard_size,
+ self.data_shards,
+ self.large_block_size(),
+ self.small_block_size(),
+ )
+ }
+
+ /// Resolve an interval against this volume's shard block layout. Mirrors
+ /// Go's EcVolume.IntervalToShardIdAndOffset.
+ pub fn interval_to_shard_id_and_offset(
+ &self,
+ interval: &ec_locate::Interval,
+ ) -> (ShardId, i64) {
+ interval.to_shard_id_and_offset(
+ self.data_shards,
+ self.large_block_size(),
+ self.small_block_size(),
+ )
}
pub fn locate_needle(
@@ -829,7 +1156,7 @@ impl EcVolume {
let mut bytes = Vec::with_capacity(actual_size);
for interval in &intervals {
- let (shard_id, shard_offset) = interval.to_shard_id_and_offset(self.data_shards);
+ let (shard_id, shard_offset) = self.interval_to_shard_id_and_offset(interval);
let shard = self
.shards
.get(shard_id as usize)
@@ -989,7 +1316,7 @@ impl EcVolume {
// A needle is verifiable locally only if every shard it spans is local;
// when any is remote, skip the reassembly buffer entirely.
let has_remote_chunks = locations.iter().any(|iv| {
- let (sid, _) = iv.to_shard_id_and_offset(self.data_shards);
+ let (sid, _) = self.interval_to_shard_id_and_offset(iv);
self.shards.get(sid as usize).and_then(|s| s.as_ref()).is_none()
});
let mut read: i64 = 0;
@@ -1001,7 +1328,7 @@ impl EcVolume {
let mut local_shard_ids: Vec = Vec::new();
for (i, iv) in locations.iter().enumerate() {
- let (sid, soffset) = iv.to_shard_id_and_offset(self.data_shards);
+ let (sid, soffset) = self.interval_to_shard_id_and_offset(iv);
let ssize = iv.size;
let shard = match self.shards.get(sid as usize).and_then(|s| s.as_ref()) {
Some(s) => s,
@@ -1800,8 +2127,14 @@ mod tests {
assert_eq!(vol.parity_shards, 3);
}
+ /// A vif that RECORDS a ratio no volume could have must fail the mount.
+ /// Substituting the default 10+4 (and with it the legacy block layout)
+ /// would read a uniform volume's shards at the wrong offsets and answer
+ /// with the wrong bytes; only an entirely absent config means "this
+ /// predates the record", which
+ /// `test_ec_volume_absent_vif_config_uses_defaults` covers.
#[test]
- fn test_ec_volume_invalid_vif_config_falls_back_to_defaults() {
+ fn test_ec_volume_invalid_vif_config_fails_the_mount() {
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
write_ecx_file(dir, "pics", VolumeId(1), &[]);
@@ -1822,9 +2155,27 @@ mod tests {
)
.unwrap();
+ // EcVolume has no Debug impl, so match rather than expect_err.
+ match EcVolume::new(dir, dir, "pics", VolumeId(1)) {
+ Ok(_) => panic!("an impossible ratio must fail the mount"),
+ Err(e) => assert_eq!(e.kind(), io::ErrorKind::InvalidData, "got {e}"),
+ }
+ }
+
+ /// The compatibility case the check above must not swallow: a vif with no
+ /// EC config at all predates the record, and mounts on the defaults.
+ #[test]
+ fn test_ec_volume_absent_vif_config_uses_defaults() {
+ let tmp = TempDir::new().unwrap();
+ let dir = tmp.path().to_str().unwrap();
+ write_ecx_file(dir, "pics", VolumeId(1), &[]);
+ let base = crate::storage::volume::volume_file_name(dir, "pics", VolumeId(1));
+ std::fs::write(format!("{}.vif", base), r#"{"version":3}"#).unwrap();
+
let vol = EcVolume::new(dir, dir, "pics", VolumeId(1)).unwrap();
assert_eq!(vol.data_shards, DATA_SHARDS_COUNT as u32);
assert_eq!(vol.parity_shards, PARITY_SHARDS_COUNT as u32);
+ assert_eq!(vol.block_size, 0, "legacy layout for a pre-record volume");
}
#[test]
@@ -1973,3 +2324,370 @@ mod tests {
);
}
}
+
+#[cfg(test)]
+mod uniform_layout_tests {
+ use super::*;
+ use crate::storage::needle_map::NeedleMapKind;
+ use crate::storage::volume::{VifEcShardConfig, VifVolumeInfo, Volume};
+ use tempfile::TempDir;
+
+ // Write ~26MB of needles so the uniform block size (3MB) diverges from the
+ // legacy layout, encode, and verify EcVolume reads every needle back
+ // through the .vif-recorded geometry. A legacy-encoded fixture (no .vif
+ // block size) must keep reading through the legacy interpretation.
+ #[test]
+ fn test_read_needles_uniform_and_legacy_layouts() {
+ for legacy in [false, true] {
+ let tmp = TempDir::new().unwrap();
+ let dir = tmp.path().to_str().unwrap();
+ let vid = VolumeId(8);
+
+ let mut v = Volume::new(
+ dir,
+ dir,
+ "",
+ vid,
+ NeedleMapKind::InMemory,
+ None,
+ None,
+ 0,
+ Version::current(),
+ )
+ .unwrap();
+ let mut expected: Vec<(NeedleId, Vec)> = Vec::new();
+ for i in 1u64..=11 {
+ let size = if i <= 5 { 4 << 20 } else { 1 << 20 };
+ let data: Vec = (0..size)
+ .map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8)
+ .collect();
+ let mut n = Needle {
+ id: NeedleId(i),
+ cookie: Cookie(i as u32),
+ data: data.clone(),
+ data_size: data.len() as u32,
+ ..Needle::default()
+ };
+ v.write_needle(&mut n, true, false).unwrap();
+ expected.push((NeedleId(i), data));
+ }
+ v.sync_to_disk().unwrap();
+ let dat_size = v.dat_file_size().unwrap() as i64;
+ v.close();
+
+ let block_size = if legacy {
+ // Legacy fixture: two-tier encode plus a .vif without a block
+ // size, the state every pre-upgrade EC volume is in.
+ use crate::storage::erasure_coding::ec_bitrot::{
+ ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
+ };
+ use reed_solomon_erasure::galois_8::ReedSolomon;
+ let base = crate::storage::volume::volume_file_name(dir, "", vid);
+ crate::storage::erasure_coding::ec_encoder::write_sorted_ecx_from_idx(
+ &format!("{}.idx", base),
+ &format!("{}.ecx", base),
+ )
+ .unwrap();
+ let dat_file = std::fs::File::open(format!("{}.dat", base)).unwrap();
+ let rs = ReedSolomon::new(10, 4).unwrap();
+ let mut shards: Vec = (0..14u8)
+ .map(|i| EcVolumeShard::new(dir, "", vid, i))
+ .collect();
+ for shard in &mut shards {
+ shard.create().unwrap();
+ }
+ let mut builders: Vec = (0..14)
+ .map(|_| ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64))
+ .collect();
+ crate::storage::erasure_coding::ec_encoder::encode_dat_file(
+ &dat_file,
+ dat_size,
+ &rs,
+ &mut shards,
+ &mut builders,
+ 10,
+ 4,
+ 256 * 1024,
+ ERASURE_CODING_LARGE_BLOCK_SIZE,
+ ERASURE_CODING_SMALL_BLOCK_SIZE,
+ )
+ .unwrap();
+ for shard in &mut shards {
+ shard.close();
+ }
+ 0
+ } else {
+ let bs = crate::storage::erasure_coding::ec_encoder::write_ec_files(
+ dir, dir, "", vid, 10, 4,
+ )
+ .unwrap();
+ assert!(
+ bs > ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
+ "block size {} does not diverge from the legacy layout",
+ bs
+ );
+ bs
+ };
+
+ let base = crate::storage::volume::volume_file_name(dir, "", vid);
+ let vif = VifVolumeInfo {
+ version: Version::current().0 as u32,
+ dat_file_size: dat_size,
+ ec_shard_config: Some(VifEcShardConfig {
+ data_shards: 10,
+ parity_shards: 4,
+ block_size,
+ ..Default::default()
+ }),
+ ..Default::default()
+ };
+ std::fs::write(
+ format!("{}.vif", base),
+ serde_json::to_string_pretty(&vif).unwrap(),
+ )
+ .unwrap();
+ std::fs::remove_file(format!("{}.dat", base)).unwrap();
+ std::fs::remove_file(format!("{}.idx", base)).unwrap();
+
+ let mut vol = EcVolume::new(dir, dir, "", vid).unwrap();
+ assert_eq!(vol.block_size, block_size, "block size not loaded from .vif");
+ for i in 0..10u8 {
+ vol.add_shard(EcVolumeShard::new(dir, "", vid, i)).unwrap();
+ }
+
+ for (id, data) in &expected {
+ let n = vol
+ .read_ec_shard_needle(*id)
+ .unwrap()
+ .unwrap_or_else(|| panic!("needle {} not found (legacy={})", id.0, legacy));
+ assert_eq!(
+ n.data, *data,
+ "needle {} data mismatch (legacy={})",
+ id.0, legacy
+ );
+ }
+ }
+ }
+
+ // A present-but-malformed .vif must FAIL the mount: every new encode
+ // records a positive uniform block size there, and silently defaulting
+ // to the legacy layout would serve those shards with the wrong offset
+ // math. Absence stays legal — legacy volumes predate the sidecar.
+ #[test]
+ fn new_fails_on_malformed_vif() {
+ let tmp = tempfile::TempDir::new().unwrap();
+ let dir = tmp.path().to_str().unwrap();
+ let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(1));
+ std::fs::write(format!("{}.ecx", base), b"").unwrap();
+ std::fs::write(format!("{}.vif", base), b"not json").unwrap();
+ let res = EcVolume::new(dir, dir, "", VolumeId(1));
+ assert!(res.is_err(), "mount over a malformed .vif must fail");
+ }
+
+ #[test]
+ fn new_absent_vif_mounts_with_defaults() {
+ let tmp = tempfile::TempDir::new().unwrap();
+ let dir = tmp.path().to_str().unwrap();
+ let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(1));
+ std::fs::write(format!("{}.ecx", base), b"").unwrap();
+ let ev = EcVolume::new(dir, dir, "", VolumeId(1)).unwrap();
+ assert_eq!(ev.block_size, 0, "legacy mount must use the legacy layout");
+ }
+
+ /// A rebuild lands on one disk, but the volume's metadata may live on
+ /// another. Reading only the selected directory returns the default 10+4
+ /// and the legacy block layout, which reconstructs a custom-ratio or
+ /// uniform volume through the wrong matrix.
+ fn seed_uniform_sidecar(base: &str, ds: u32, ps: u32, block: i64) {
+ use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums};
+ use crate::storage::erasure_coding::ec_bitrot;
+ let prot = EcBitrotProtection {
+ algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
+ block_size: ec_bitrot::DEFAULT_BITROT_BLOCK_SIZE as u32,
+ generation: 0,
+ ec_shard_config: Some(ec_bitrot::ec_shard_config(ds, ps, block)),
+ shards: (0..(ds + ps))
+ .map(|shard_id| EcShardChecksums {
+ shard_id,
+ covered_size: 4,
+ block_crc32c: vec![0u8; 4],
+ })
+ .collect(),
+ encode_uuid: vec![0u8; 16],
+ };
+ ec_bitrot::save_bitrot_sidecar(&ec_bitrot::bitrot_sidecar_path(base, 0), &prot).unwrap();
+ }
+
+ fn seed_config_free_vif(base: &str) {
+ std::fs::write(
+ format!("{}.vif", base),
+ r#"{"version":3,"datFileSize":0}"#,
+ )
+ .unwrap();
+ }
+
+ // A .vif that omits ecShardConfig answers nothing about the layout, so it
+ // must not short-circuit the sidecar search. Returning as soon as any
+ // parseable vif turned up resolved a 12+4 uniform volume as 10+4 legacy.
+ #[test]
+ fn read_ec_shard_config_falls_back_when_the_vif_has_no_config() {
+ let d = tempfile::TempDir::new().unwrap();
+ let dir = d.path().to_str().unwrap();
+ let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(11));
+ seed_config_free_vif(&base);
+ seed_uniform_sidecar(&base, 12, 4, 3 * 1024 * 1024);
+
+ let got = read_ec_shard_config(dir, dir, "", VolumeId(11)).unwrap();
+ assert_eq!(got, (12, 4, 3 * 1024 * 1024));
+ }
+
+ // Split -dir/-dir.idx: the vif and the sidecar both live with the INDEX,
+ // and the resolver was told to report the DATA directory, so the fallback
+ // looked in a directory holding neither.
+ #[test]
+ fn read_ec_shard_config_finds_a_config_free_vifs_sidecar_in_the_index_dir() {
+ let data = tempfile::TempDir::new().unwrap();
+ let idx = tempfile::TempDir::new().unwrap();
+ let (dir, dir_idx) = (data.path().to_str().unwrap(), idx.path().to_str().unwrap());
+ let idx_base = crate::storage::volume::volume_file_name(dir_idx, "", VolumeId(12));
+ seed_config_free_vif(&idx_base);
+ seed_uniform_sidecar(&idx_base, 12, 4, 3 * 1024 * 1024);
+
+ let got = read_ec_shard_config(dir, dir_idx, "", VolumeId(12)).unwrap();
+ assert_eq!(got, (12, 4, 3 * 1024 * 1024));
+ }
+
+ // Same gap in the cross-disk resolver the rebuild uses.
+ #[test]
+ fn read_ec_shard_config_across_dirs_falls_back_when_the_vif_has_no_config() {
+ let data = tempfile::TempDir::new().unwrap();
+ let sib = tempfile::TempDir::new().unwrap();
+ let (dir, sibling) = (data.path().to_str().unwrap(), sib.path().to_str().unwrap());
+ seed_config_free_vif(&crate::storage::volume::volume_file_name(dir, "", VolumeId(13)));
+ seed_uniform_sidecar(
+ &crate::storage::volume::volume_file_name(sibling, "", VolumeId(13)),
+ 12,
+ 4,
+ 3 * 1024 * 1024,
+ );
+
+ let got = read_ec_shard_config_across_dirs(
+ dir,
+ dir,
+ &[sibling.to_string()],
+ "",
+ VolumeId(13),
+ )
+ .unwrap();
+ assert_eq!(got, (12, 4, 3 * 1024 * 1024));
+ }
+
+ // Absence stays legal: a volume predating both records is genuinely legacy.
+ #[test]
+ fn read_ec_shard_config_config_free_vif_without_a_sidecar_is_legacy() {
+ let d = tempfile::TempDir::new().unwrap();
+ let dir = d.path().to_str().unwrap();
+ seed_config_free_vif(&crate::storage::volume::volume_file_name(dir, "", VolumeId(14)));
+ let got = read_ec_shard_config(dir, dir, "", VolumeId(14)).unwrap();
+ assert_eq!(got, (10, 4, 0));
+ }
+
+ #[test]
+ fn read_ec_shard_config_finds_a_sibling_disks_vif() {
+ let a = tempfile::TempDir::new().unwrap();
+ let b = tempfile::TempDir::new().unwrap();
+ let (rebuild, sibling) = (a.path().to_str().unwrap(), b.path().to_str().unwrap());
+ let base = crate::storage::volume::volume_file_name(sibling, "", VolumeId(3));
+ std::fs::write(
+ format!("{}.vif", base),
+ r#"{"version":3,"ecShardConfig":{"dataShards":12,"parityShards":4,"blockSize":3145728}}"#,
+ )
+ .unwrap();
+
+ let (ds, ps, bs) = read_ec_shard_config_across_dirs(
+ rebuild,
+ rebuild,
+ &[sibling.to_string()],
+ "",
+ VolumeId(3),
+ )
+ .unwrap();
+ assert_eq!((ds, ps, bs), (12, 4, 3 * 1024 * 1024));
+ }
+
+ /// Same, for the sidecar: with no .vif anywhere it is the surviving record
+ /// of the geometry, wherever it sits.
+ // A split -dir/-dir.idx location keeps its metadata with the INDEX, and
+ // callers leave their own index directory out of other_dirs because it is
+ // passed separately. With no .vif anywhere the generation-0 .ecsum is the
+ // only record of the layout, so missing that directory resolves a 12+4
+ // uniform volume to 10+4 legacy and reconstructs through the wrong matrix.
+ #[test]
+ fn read_ec_shard_config_finds_the_sidecar_in_its_own_index_dir() {
+ use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums};
+ use crate::storage::erasure_coding::ec_bitrot;
+
+ let data = tempfile::TempDir::new().unwrap();
+ let idx = tempfile::TempDir::new().unwrap();
+ let (dir, dir_idx) = (
+ data.path().to_str().unwrap(),
+ idx.path().to_str().unwrap(),
+ );
+ let base = crate::storage::volume::volume_file_name(dir_idx, "", VolumeId(7));
+ let prot = EcBitrotProtection {
+ algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
+ block_size: ec_bitrot::DEFAULT_BITROT_BLOCK_SIZE as u32,
+ generation: 0,
+ ec_shard_config: Some(ec_bitrot::ec_shard_config(12, 4, 3 * 1024 * 1024)),
+ shards: (0..16u32)
+ .map(|shard_id| EcShardChecksums {
+ shard_id,
+ covered_size: 4,
+ block_crc32c: vec![0u8; 4],
+ })
+ .collect(),
+ encode_uuid: vec![0u8; 16],
+ };
+ ec_bitrot::save_bitrot_sidecar(&ec_bitrot::bitrot_sidecar_path(&base, 0), &prot).unwrap();
+
+ let (ds, ps, bs) =
+ read_ec_shard_config_across_dirs(dir, dir_idx, &[], "", VolumeId(7)).unwrap();
+ assert_eq!((ds, ps, bs), (12, 4, 3 * 1024 * 1024));
+ }
+
+ #[test]
+ fn read_ec_shard_config_finds_a_sibling_disks_sidecar() {
+ use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums};
+ use crate::storage::erasure_coding::ec_bitrot;
+
+ let a = tempfile::TempDir::new().unwrap();
+ let b = tempfile::TempDir::new().unwrap();
+ let (rebuild, sibling) = (a.path().to_str().unwrap(), b.path().to_str().unwrap());
+ let base = crate::storage::volume::volume_file_name(sibling, "", VolumeId(4));
+ let prot = EcBitrotProtection {
+ algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
+ block_size: ec_bitrot::DEFAULT_BITROT_BLOCK_SIZE as u32,
+ generation: 0,
+ ec_shard_config: Some(ec_bitrot::ec_shard_config(12, 4, 3 * 1024 * 1024)),
+ shards: (0..16u32)
+ .map(|shard_id| EcShardChecksums {
+ shard_id,
+ covered_size: 4,
+ block_crc32c: vec![0u8; 4],
+ })
+ .collect(),
+ encode_uuid: vec![0u8; 16],
+ };
+ ec_bitrot::save_bitrot_sidecar(&ec_bitrot::bitrot_sidecar_path(&base, 0), &prot).unwrap();
+
+ let (ds, ps, bs) = read_ec_shard_config_across_dirs(
+ rebuild,
+ rebuild,
+ &[sibling.to_string()],
+ "",
+ VolumeId(4),
+ )
+ .unwrap();
+ assert_eq!((ds, ps, bs), (12, 4, 3 * 1024 * 1024));
+ }
+}
diff --git a/seaweed-volume/src/storage/store.rs b/seaweed-volume/src/storage/store.rs
index 644c6b76d..b458efb47 100644
--- a/seaweed-volume/src/storage/store.rs
+++ b/seaweed-volume/src/storage/store.rs
@@ -804,6 +804,37 @@ impl Store {
}
/// Find an EC volume across all locations (mutable).
+ /// Every per-disk `EcVolume` this store maps for `vid`. A vid can mount on
+ /// N disks as N distinct runtimes, and the first-match `find_ec_volume_mut`
+ /// hides the siblings — so anything that has to reach the whole volume,
+ /// rather than any one runtime of it, iterates this instead.
+ /// Every directory on this server that could hold an EC volume's metadata —
+ /// each disk's data and index directory. Startup mirroring gives each
+ /// shard-bearing disk its own .ecx/.ecj/.vif, but the checksum sidecar is
+ /// not mirrored and a repair delivers exactly one copy, so a runtime looking
+ /// only at its own two directories cannot see it. Handing this list to the
+ /// sidecar resolution keeps one authoritative copy reachable from every
+ /// runtime rather than duplicating a file that is rewritten as shards are
+ /// repaired and generations published.
+ pub fn ec_metadata_dirs(&self) -> Vec {
+ let mut dirs: Vec = Vec::with_capacity(self.locations.len() * 2);
+ for loc in &self.locations {
+ for dir in [&loc.directory, &loc.idx_directory] {
+ if !dir.is_empty() && !dirs.iter().any(|d| d == dir) {
+ dirs.push(dir.clone());
+ }
+ }
+ }
+ dirs
+ }
+
+ pub fn find_all_ec_volumes_mut(&mut self, vid: VolumeId) -> Vec<&mut EcVolume> {
+ self.locations
+ .iter_mut()
+ .filter_map(|loc| loc.find_ec_volume_mut(vid))
+ .collect()
+ }
+
pub fn find_ec_volume_mut(&mut self, vid: VolumeId) -> Option<&mut EcVolume> {
for loc in &mut self.locations {
if let Some(ecv) = loc.find_ec_volume_mut(vid) {
diff --git a/seaweed-volume/src/storage/volume.rs b/seaweed-volume/src/storage/volume.rs
index 64572c345..6373c9eda 100644
--- a/seaweed-volume/src/storage/volume.rs
+++ b/seaweed-volume/src/storage/volume.rs
@@ -194,6 +194,10 @@ pub struct VifEcShardConfig {
/// so a read served from a different run's shard is rejected.
#[serde(default, rename = "encodeTsNs", with = "string_or_i64")]
pub encode_ts_ns: i64,
+ /// Uniform block layout: each shard is a single contiguous block of this
+ /// many bytes. 0 = legacy 1GiB/1MiB two-tier layout.
+ #[serde(default, rename = "blockSize", with = "string_or_i64")]
+ pub block_size: i64,
}
/// Serde-compatible representation of OldVersionVolumeInfo for legacy .vif JSON deserialization.
@@ -288,6 +292,7 @@ impl VifVolumeInfo {
data_shards: c.data_shards,
parity_shards: c.parity_shards,
encode_ts_ns: c.encode_ts_ns,
+ block_size: c.block_size,
}),
read_only_can_delete: pb.read_only_can_delete,
}
@@ -320,6 +325,7 @@ impl VifVolumeInfo {
data_shards: c.data_shards,
parity_shards: c.parity_shards,
encode_ts_ns: c.encode_ts_ns,
+ block_size: c.block_size,
}
}),
read_only_can_delete: self.read_only_can_delete,
diff --git a/test/plugin_workers/fake_volume_server.go b/test/plugin_workers/fake_volume_server.go
index 103515ce6..1ea58276e 100644
--- a/test/plugin_workers/fake_volume_server.go
+++ b/test/plugin_workers/fake_volume_server.go
@@ -15,6 +15,7 @@ import (
"github.com/seaweedfs/seaweedfs/weed/operation"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
"google.golang.org/grpc"
"google.golang.org/grpc/credentials/insecure"
)
@@ -339,7 +340,14 @@ func (v *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serve
}
}
- resp := &volume_server_pb.VolumeEcShardsInfoResponse{}
+ // Answer with the layout out of the .vif that was actually delivered here,
+ // the way a real holder answers from the context it mounted the shards
+ // with. A coordinator uses this to tell a server that understands the
+ // shard block layout from one that never knew the field, so a fake that
+ // always reported "unset" would look like a pre-upgrade server.
+ resp := &volume_server_pb.VolumeEcShardsInfoResponse{
+ EcShardConfig: v.ecShardConfigFromVif(req.VolumeId),
+ }
prefix := fmt.Sprintf("%d.ec", req.VolumeId)
entries, _ := os.ReadDir(v.baseDir)
for _, entry := range entries {
@@ -372,6 +380,16 @@ func (v *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serve
return resp, nil
}
+// ecShardConfigFromVif reads the EC layout out of the .vif this server was
+// given, which distribution ships to every holder alongside its shards.
+func (v *VolumeServer) ecShardConfigFromVif(volumeID uint32) *volume_server_pb.EcShardConfig {
+ vi, _, found, err := volume_info.MaybeLoadVolumeInfo(v.filePath(volumeID, ".vif"))
+ if err != nil || !found {
+ return nil
+ }
+ return vi.GetEcShardConfig()
+}
+
func (v *VolumeServer) VolumeDelete(ctx context.Context, req *volume_server_pb.VolumeDeleteRequest) (*volume_server_pb.VolumeDeleteResponse, error) {
v.mu.Lock()
v.deleteRequests = append(v.deleteRequests, req)
diff --git a/weed/command/fix.go b/weed/command/fix.go
index 6c24dd320..301bfd090 100644
--- a/weed/command/fix.go
+++ b/weed/command/fix.go
@@ -327,9 +327,15 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6
// Priority: explicit flags > existing .vif > defaults (10+4).
vifName := base + ".vif"
vifExists := util.FileExists(vifName)
+ // Whether the on-disk .vif actually answers the layout question. An empty
+ // stub (MaybeLoadVolumeInfo reads it as absent) or one with no EC config
+ // exists but tells us nothing, and the recovered layout must still be
+ // written back over it.
+ vifUsable := false
dataShards := erasure_coding.DataShardsCount
parityShards := erasure_coding.ParityShardsCount
var datFileSize int64
+ blockSize := int64(-1) // the shard block layout; 0 legacy, >0 uniform, <0 unknown
if vifExists {
// MaybeLoadVolumeInfo returns a non-nil error when the .vif exists but
// cannot be read or unmarshalled; fail loudly rather than silently
@@ -338,13 +344,66 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6
fail(fmt.Errorf("volume %d: read %s: %w", volumeId, vifName, loadErr))
return
} else if found && vi != nil {
- if cfg := vi.GetEcShardConfig(); cfg != nil && cfg.GetDataShards() > 0 {
+ // A partial config (parity 0, or counts past the shard ceiling)
+ // describes no layout: leave the sentinel unknown and let the
+ // sidecar / dual scan below answer, and rewrite the .vif at the end
+ // rather than trusting it.
+ cfg := vi.GetEcShardConfig()
+ if cfg != nil && erasure_coding.ValidEcShardCounts(cfg.GetDataShards(), cfg.GetParityShards()) {
dataShards = int(cfg.GetDataShards())
parityShards = int(cfg.GetParityShards())
+ // Only an EC config answers the layout question. Reading 0 off a
+ // vif with no config would assert "legacy" and suppress both the
+ // .ecsum fallback and the dual-layout scan below; leave the
+ // sentinel at -1 (unknown) instead.
+ // A recorded block size no encoder could have produced is not
+ // an answer: a positive one would pin the scan to a geometry
+ // that de-stripes to garbage, and a negative one would leave
+ // the invalid .vif in place after the dual scan recovers the
+ // real layout. Leave the sentinel unknown and rewrite the file
+ // at the end.
+ if bs := cfg.GetBlockSize(); erasure_coding.ValidateBlockSize(bs) == nil {
+ blockSize = bs
+ vifUsable = true
+ } else {
+ glog.Warningf("volume %d: %s records an invalid shard block size %d; recovering the layout by scan and rewriting it",
+ volumeId, vifName, bs)
+ }
}
datFileSize = vi.GetDatFileSize()
}
}
+ // The bitrot sidecar records the same EC config at encode time; use it when
+ // the .vif is gone. A uniform-layout volume cannot be de-striped correctly
+ // without its block size.
+ if blockSize < 0 {
+ sidecarPath := erasure_coding.BitrotSidecarPath(base, 0)
+ if prot, serr := erasure_coding.LoadBitrotSidecar(sidecarPath); serr == nil {
+ // The sidecar is about to pin the reconstruction to one geometry
+ // instead of letting the dual scan decide, so every field it
+ // contributes has to hold up: it must describe generation 0 (the
+ // shards being read), a complete in-range ratio, and a block size
+ // an encoder could have produced. Anything less leaves the layout
+ // unknown, which is the answer that still recovers by scanning.
+ cfg := prot.GetEcShardConfig()
+ ds, ps := int(cfg.GetDataShards()), int(cfg.GetParityShards())
+ switch {
+ case prot.GetGeneration() != 0:
+ glog.Warningf("volume %d: %s records generation %d, not the generation-0 shards; ignoring it",
+ volumeId, sidecarPath, prot.GetGeneration())
+ case !erasure_coding.ValidEcShardCounts(cfg.GetDataShards(), cfg.GetParityShards()):
+ glog.Warningf("volume %d: %s records invalid shard counts %d+%d; ignoring it",
+ volumeId, sidecarPath, ds, ps)
+ case erasure_coding.ValidateBlockSize(cfg.GetBlockSize()) != nil:
+ glog.Warningf("volume %d: %s records an invalid shard block size %d; ignoring it",
+ volumeId, sidecarPath, cfg.GetBlockSize())
+ default:
+ dataShards = ds
+ parityShards = ps
+ blockSize = cfg.GetBlockSize()
+ }
+ }
+ }
if *fixEcDataShards > 0 {
dataShards = *fixEcDataShards
}
@@ -381,6 +440,9 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6
}
if !dataComplete {
ctx := &erasure_coding.ECContext{DataShards: dataShards, ParityShards: parityShards}
+ if blockSize > 0 {
+ ctx.BlockSize = blockSize
+ }
glog.Infof("volume %d: %d/%d shards present; reconstructing missing shards (%s) before index rebuild", volumeId, presentCount, dataShards+parityShards, ctx.String())
if _, err := erasure_coding.RebuildEcFiles(base, ctx, *fixEcUnsafeIgnoreSidecar); err != nil {
fail(fmt.Errorf("volume %d: reconstruct missing shards from %d survivors: %w", volumeId, presentCount, err))
@@ -407,37 +469,114 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6
glog.V(0).Infof("volume %d: no .dat size in .vif; reconstructing padded .dat (%d bytes) from %d data shards", volumeId, reconstructSize, dataShards)
}
- // De-stripe the data shards into a temporary .dat next to the shards.
+ // De-stripe the data shards into a temporary .dat next to the shards and
+ // scan it into a fresh .ecx. With the layout unknown, de-stripe under both
+ // candidate layouts and keep the one whose needle chain scans furthest.
+ type layoutCandidate struct {
+ name string
+ large, small int64
+ }
+ var candidates []layoutCandidate
+ switch {
+ case blockSize > 0:
+ candidates = []layoutCandidate{{"uniform", blockSize, blockSize}}
+ case blockSize == 0:
+ candidates = []layoutCandidate{{"legacy", erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize}}
+ default:
+ candidates = []layoutCandidate{
+ {"legacy", erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize},
+ }
+ // A uniform-layout shard is exactly one block long, and every block
+ // size an encoder can produce is a whole number of small blocks. An
+ // extent that is not — a truncated or partially copied shard — could
+ // not have come from a uniform encode, and recording it would write a
+ // .vif that ValidateBlockSize refuses on the next mount: the volume
+ // this tool was run to rescue would never open again.
+ if erasure_coding.ValidateBlockSize(shardSize) == nil && shardSize > 0 {
+ candidates = append(candidates, layoutCandidate{"uniform", shardSize, shardSize})
+ glog.Infof("volume %d: no .vif or .ecsum records the shard block layout; trying both", volumeId)
+ } else {
+ glog.Infof("volume %d: no .vif or .ecsum records the shard block layout, and the %d-byte shard extent is not a whole number of %d-byte blocks; trying the legacy layout only",
+ volumeId, shardSize, erasure_coding.ErasureCodingSmallBlockSize)
+ }
+ }
+
tmpBase := base + ".ecxrecover"
tmpDat := tmpBase + ".dat"
- if err := erasure_coding.WriteDatFile(tmpBase, reconstructSize, reconstructSize, shardFileNames); err != nil {
- os.Remove(tmpDat)
- fail(fmt.Errorf("volume %d: reconstruct .dat from data shards: %w", volumeId, err))
+ defer os.Remove(tmpDat)
+ bestIdx := -1
+ var realDatSize, bestNeedles int64
+ var version needle.Version
+ var candErr error
+ for i, cand := range candidates {
+ if err := erasure_coding.WriteDatFile(tmpBase, reconstructSize, reconstructSize, shardFileNames, cand.large, cand.small); err != nil {
+ candErr = fmt.Errorf("volume %d: reconstruct .dat (%s layout): %w", volumeId, cand.name, err)
+ continue
+ }
+ size, needles, ver, err := writeEcxFromDat(tmpDat, ecxName+"."+cand.name)
+ if err != nil {
+ os.Remove(ecxName + "." + cand.name)
+ candErr = fmt.Errorf("volume %d: build .ecx from reconstructed .dat (%s layout): %w", volumeId, cand.name, err)
+ continue
+ }
+ // The wrong layout de-stripes to garbage past the first block
+ // boundary, so the layout that indexes more valid needles wins.
+ // On a tie — garbage bytes do sometimes parse as plausible sizes — the
+ // layout whose needle chain reached further into the .dat wins, which is
+ // the distance the scan actually validated.
+ if bestIdx < 0 || needles > bestNeedles || (needles == bestNeedles && size > realDatSize) {
+ bestIdx = i
+ bestNeedles = needles
+ realDatSize = size
+ version = ver
+ }
+ }
+ if bestIdx < 0 {
+ fail(candErr)
return
}
- defer os.Remove(tmpDat)
-
- realDatSize, version, err := writeEcxFromDat(tmpDat, ecxName)
- if err != nil {
- os.Remove(ecxName)
- fail(fmt.Errorf("volume %d: build .ecx from reconstructed .dat: %w", volumeId, err))
+ for i := range candidates {
+ if i != bestIdx {
+ os.Remove(ecxName + "." + candidates[i].name)
+ }
+ }
+ if err := os.Rename(ecxName+"."+candidates[bestIdx].name, ecxName); err != nil {
+ fail(fmt.Errorf("volume %d: publish %s: %w", volumeId, ecxName, err))
return
}
+ if blockSize < 0 {
+ if candidates[bestIdx].name == "uniform" {
+ blockSize = shardSize
+ } else {
+ blockSize = 0
+ }
+ glog.Infof("volume %d: %s layout indexes %d needles over %d bytes; keeping it", volumeId, candidates[bestIdx].name, bestNeedles, realDatSize)
+ }
glog.Infof("volume %d: wrote %s from %d data shards", volumeId, ecxName, dataShards)
- // Regenerate the .vif when missing so the volume can mount and future
- // rebuilds know the EC ratio and original .dat size.
- if !vifExists {
+ // Regenerate the .vif when it is missing OR unusable (an empty stub, or one
+ // with no EC config): the layout just recovered by the dual scan is the only
+ // record of it, and leaving the stub in place would mount the volume as
+ // legacy on the next start and serve the wrong offsets.
+ if !vifUsable {
size := datFileSize
if size <= 0 {
size = realDatSize
}
+ // Never publish a layout the mount path will refuse: NewEcVolume runs
+ // the same check and fails closed, so an unvalidated write here would
+ // trade a recoverable volume for one that can no longer be opened.
+ if bsErr := erasure_coding.ValidateBlockSize(blockSize); bsErr != nil {
+ fail(fmt.Errorf("volume %d: refusing to write %s: %w", volumeId, vifName, bsErr))
+ return
+ }
volumeInfo := &volume_server_pb.VolumeInfo{
Version: uint32(version),
DatFileSize: size,
EcShardConfig: &volume_server_pb.EcShardConfig{
DataShards: uint32(dataShards),
ParityShards: uint32(parityShards),
+ BlockSize: blockSize,
},
}
if err := volume_info.SaveVolumeInfo(vifName, volumeInfo); err != nil {
@@ -451,25 +590,26 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6
// writeEcxFromDat scans a (reconstructed) .dat and writes an ascending-sorted
// .ecx containing only live needles — the same on-disk shape
// WriteSortedFileFromIdx produces when an EC volume is first encoded. It returns
-// the physical .dat size (the offset where the EC zero padding begins) and the
-// volume version read from the superblock.
-func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle.Version, err error) {
+// the physical .dat size (the offset where the EC zero padding begins), the
+// number of live needles indexed, and the volume version read from the
+// superblock.
+func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, liveNeedles int64, version needle.Version, err error) {
f, err := os.OpenFile(datPath, os.O_RDONLY, 0644)
if err != nil {
- return 0, 0, fmt.Errorf("open %s: %w", datPath, err)
+ return 0, 0, 0, fmt.Errorf("open %s: %w", datPath, err)
}
datBackend := backend.NewDiskFile(f)
defer datBackend.Close()
superBlock, err := super_block.ReadSuperBlock(datBackend)
if err != nil {
- return 0, 0, fmt.Errorf("read superblock: %w", err)
+ return 0, 0, 0, fmt.Errorf("read superblock: %w", err)
}
version = superBlock.Version
fileSize, _, err := datBackend.GetStat()
if err != nil {
- return 0, version, fmt.Errorf("stat %s: %w", datPath, err)
+ return 0, 0, version, fmt.Errorf("stat %s: %w", datPath, err)
}
nm := needle_map.NewMemDb()
@@ -482,7 +622,7 @@ func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle
if readErr == io.EOF {
break
}
- return 0, version, fmt.Errorf("read needle header at offset %d: %w", offset, readErr)
+ return 0, 0, version, fmt.Errorf("read needle header at offset %d: %w", offset, readErr)
}
// EC encoding zero-pads the tail of the last block row. An all-zero
// header marks the start of that padding, i.e. the end of real needles.
@@ -491,13 +631,13 @@ func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle
}
if n.Size.IsValid() {
if pe := nm.Set(n.Id, types.ToOffset(offset), n.Size); pe != nil {
- return 0, version, fmt.Errorf("set needle %d: %w", n.Id, pe)
+ return 0, 0, version, fmt.Errorf("set needle %d: %w", n.Id, pe)
}
} else {
// Deleted/invalid: drop it so the .ecx carries only live entries,
// matching the encode-time WriteSortedFileFromIdx behavior.
if pe := nm.Delete(n.Id); pe != nil {
- return 0, version, fmt.Errorf("delete needle %d: %w", n.Id, pe)
+ return 0, 0, version, fmt.Errorf("delete needle %d: %w", n.Id, pe)
}
}
offset += types.NeedleHeaderSize + rest
@@ -506,16 +646,17 @@ func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle
ecxFile, err := os.OpenFile(ecxPath, os.O_TRUNC|os.O_CREATE|os.O_WRONLY, 0644)
if err != nil {
- return 0, version, fmt.Errorf("open %s: %w", ecxPath, err)
+ return 0, 0, version, fmt.Errorf("open %s: %w", ecxPath, err)
}
defer ecxFile.Close()
if err := nm.AscendingVisit(func(value needle_map.NeedleValue) error {
+ liveNeedles++
_, writeErr := ecxFile.Write(value.ToBytes())
return writeErr
}); err != nil {
- return 0, version, fmt.Errorf("write %s: %w", ecxPath, err)
+ return 0, 0, version, fmt.Errorf("write %s: %w", ecxPath, err)
}
- return datFileSize, version, nil
+ return datFileSize, liveNeedles, version, nil
}
diff --git a/weed/command/fix_ecx_test.go b/weed/command/fix_ecx_test.go
index d99f82694..c9509b0eb 100644
--- a/weed/command/fix_ecx_test.go
+++ b/weed/command/fix_ecx_test.go
@@ -17,11 +17,12 @@ import (
"github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
)
-// buildAndEncodeTestEcVolume writes a small volume (.dat + .idx), EC-encodes it
-// into .ec00..ec13, and produces the canonical sorted .ecx. It returns the base
-// path, the canonical .ecx bytes, and the original .dat size. A couple of
-// needles are deleted so the .ecx must exclude them (live entries only).
-func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string, canonicalEcx []byte, origDatSize int64) {
+// buildAndEncodeTestEcVolume writes a volume of needleCount needles of about
+// needleSize bytes (.dat + .idx), EC-encodes it into .ec00..ec13, and produces
+// the canonical sorted .ecx. It returns the base path, the canonical .ecx
+// bytes, the original .dat size, and the encode's bitrot protection. A couple
+// of needles are deleted so the .ecx must exclude them (live entries only).
+func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string, needleCount, needleSize int) (base string, canonicalEcx []byte, origDatSize int64, prot *volume_server_pb.EcBitrotProtection) {
t.Helper()
base = filepath.Join(dir, baseName)
version := needle.GetCurrentVersion()
@@ -41,10 +42,10 @@ func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string
}
nm := needle_map.NewMemDb()
- for i := uint64(1); i <= 12; i++ {
+ for i := uint64(1); i <= uint64(needleCount); i++ {
n := new(needle.Needle)
n.Id = types.Uint64ToNeedleId(i)
- n.Data = make([]byte, 200+int(i))
+ n.Data = make([]byte, needleSize+int(i))
rand.Read(n.Data)
n.Checksum = needle.NewCRC(n.Data)
offset, _, _, err := n.Append(datBackend, version)
@@ -92,7 +93,8 @@ func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string
idxFile.Close()
nm.Close()
- if _, err := erasure_coding.WriteEcFiles(base, erasure_coding.BackgroundECContext()); err != nil {
+ prot, err = erasure_coding.WriteEcFiles(base, erasure_coding.BackgroundECContext())
+ if err != nil {
t.Fatalf("WriteEcFiles: %v", err)
}
if err := erasure_coding.WriteSortedFileFromIdx(base, ".ecx"); err != nil {
@@ -105,7 +107,7 @@ func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string
if len(canonicalEcx) == 0 {
t.Fatal("canonical .ecx is empty")
}
- return base, canonicalEcx, origDatSize
+ return base, canonicalEcx, origDatSize, prot
}
// TestFixEcxFromShards verifies the .ecx and .vif are rebuilt purely from the
@@ -121,7 +123,7 @@ func TestFixEcxFromShards(t *testing.T) {
dir := t.TempDir()
const volumeId = 7
- base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "7")
+ base, canonical, origDatSize, _ := buildAndEncodeTestEcVolume(t, dir, "7", 12, 200)
// Disaster: keep only the shards.
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif"} {
@@ -175,7 +177,7 @@ func TestFixEcxFromShardsWithVif(t *testing.T) {
dir := t.TempDir()
const volumeId = 9
- base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "9")
+ base, canonical, origDatSize, _ := buildAndEncodeTestEcVolume(t, dir, "9", 12, 200)
// Write a .vif as the volume server would after encoding.
if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
@@ -234,7 +236,7 @@ func TestFixEcxFromShardsMissingShards(t *testing.T) {
dir := t.TempDir()
const volumeId = 11
- base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "11")
+ base, canonical, origDatSize, _ := buildAndEncodeTestEcVolume(t, dir, "11", 12, 200)
// Lose every index/metadata file plus three shards (two data: .ec02, .ec05;
// one parity: .ec11), keeping 11 of 14 — enough to reconstruct. The highest
@@ -271,3 +273,60 @@ func TestFixEcxFromShardsMissingShards(t *testing.T) {
t.Fatalf("dat size = %d, want %d", vi.GetDatFileSize(), origDatSize)
}
}
+
+// TestFixEcxFromShardsUniformLayout loses the .vif of a volume big enough that
+// the uniform and legacy layouts place bytes differently, and verifies the
+// layout is recovered — from the bitrot sidecar when it survives, and by
+// de-striping under both candidate layouts when it does not.
+func TestFixEcxFromShardsUniformLayout(t *testing.T) {
+ oldData, oldParity := *fixEcDataShards, *fixEcParityShards
+ *fixEcDataShards, *fixEcParityShards = 0, 0
+ *fixIgnoreError = true
+ t.Cleanup(func() {
+ *fixIgnoreError = false
+ *fixEcDataShards, *fixEcParityShards = oldData, oldParity
+ })
+
+ for _, withSidecar := range []bool{true, false} {
+ name := "from-sidecar"
+ if !withSidecar {
+ name = "dual-scan"
+ }
+ t.Run(name, func(t *testing.T) {
+ dir := t.TempDir()
+ const volumeId = 13
+ base, canonical, _, prot := buildAndEncodeTestEcVolume(t, dir, "13", 13, 2*1024*1024)
+ if got := prot.GetEcShardConfig().GetBlockSize(); got <= erasure_coding.ErasureCodingSmallBlockSize {
+ t.Fatalf("fixture block size %d does not diverge from the legacy layout", got)
+ }
+ if withSidecar {
+ if err := erasure_coding.SaveBitrotSidecar(erasure_coding.BitrotSidecarPath(base, 0), prot); err != nil {
+ t.Fatal(err)
+ }
+ }
+
+ for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif"} {
+ if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
+ t.Fatalf("remove %s: %v", base+ext, err)
+ }
+ }
+
+ doFixEcxFromShards(dir, "13", "", volumeId)
+
+ recovered, err := os.ReadFile(base + ".ecx")
+ if err != nil {
+ t.Fatalf("recovered .ecx not written: %v", err)
+ }
+ if !bytes.Equal(canonical, recovered) {
+ t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
+ }
+ vi, _, found, err := volume_info.MaybeLoadVolumeInfo(base + ".vif")
+ if err != nil || !found {
+ t.Fatalf(".vif not regenerated: found=%v err=%v", found, err)
+ }
+ if got, want := vi.GetEcShardConfig().GetBlockSize(), prot.GetEcShardConfig().GetBlockSize(); got != want {
+ t.Fatalf("regenerated .vif block size = %d, want %d", got, want)
+ }
+ })
+ }
+}
diff --git a/weed/pb/volume_server.proto b/weed/pb/volume_server.proto
index d44712c3f..4962a4e9f 100644
--- a/weed/pb/volume_server.proto
+++ b/weed/pb/volume_server.proto
@@ -541,6 +541,7 @@ message VolumeEcShardsInfoResponse {
uint64 volume_size = 2;
uint64 file_count = 3;
uint64 file_deleted_count = 4;
+ EcShardConfig ec_shard_config = 10; // the layout this holder serves reads through; a binary predating a field reports it as unset
}
message EcShardInfo {
@@ -615,6 +616,7 @@ message EcShardConfig {
uint32 data_shards = 1; // Number of data shards (e.g., 10)
uint32 parity_shards = 2; // Number of parity shards (e.g., 4)
int64 encode_ts_ns = 3; // encode time (unix nanos); a read served from a shard of a different encode run is rejected
+ int64 block_size = 4; // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout
}
// EcBitrotProtection is the entire content of a bitrot checksum sidecar
diff --git a/weed/pb/volume_server_pb/volume_server.pb.go b/weed/pb/volume_server_pb/volume_server.pb.go
index afe59bc08..24ad97778 100644
--- a/weed/pb/volume_server_pb/volume_server.pb.go
+++ b/weed/pb/volume_server_pb/volume_server.pb.go
@@ -4448,6 +4448,7 @@ type VolumeEcShardsInfoResponse struct {
VolumeSize uint64 `protobuf:"varint,2,opt,name=volume_size,json=volumeSize,proto3" json:"volume_size,omitempty"`
FileCount uint64 `protobuf:"varint,3,opt,name=file_count,json=fileCount,proto3" json:"file_count,omitempty"`
FileDeletedCount uint64 `protobuf:"varint,4,opt,name=file_deleted_count,json=fileDeletedCount,proto3" json:"file_deleted_count,omitempty"`
+ EcShardConfig *EcShardConfig `protobuf:"bytes,10,opt,name=ec_shard_config,json=ecShardConfig,proto3" json:"ec_shard_config,omitempty"` // the layout this holder serves reads through; a binary predating a field reports it as unset
unknownFields protoimpl.UnknownFields
sizeCache protoimpl.SizeCache
}
@@ -4510,6 +4511,13 @@ func (x *VolumeEcShardsInfoResponse) GetFileDeletedCount() uint64 {
return 0
}
+func (x *VolumeEcShardsInfoResponse) GetEcShardConfig() *EcShardConfig {
+ if x != nil {
+ return x.EcShardConfig
+ }
+ return nil
+}
+
type EcShardInfo struct {
state protoimpl.MessageState `protogen:"open.v1"`
ShardId uint32 `protobuf:"varint,1,opt,name=shard_id,json=shardId,proto3" json:"shard_id,omitempty"`
@@ -5145,6 +5153,7 @@ type EcShardConfig struct {
DataShards uint32 `protobuf:"varint,1,opt,name=data_shards,json=dataShards,proto3" json:"data_shards,omitempty"` // Number of data shards (e.g., 10)
ParityShards uint32 `protobuf:"varint,2,opt,name=parity_shards,json=parityShards,proto3" json:"parity_shards,omitempty"` // Number of parity shards (e.g., 4)
EncodeTsNs int64 `protobuf:"varint,3,opt,name=encode_ts_ns,json=encodeTsNs,proto3" json:"encode_ts_ns,omitempty"` // encode time (unix nanos); a read served from a shard of a different encode run is rejected
+ BlockSize int64 `protobuf:"varint,4,opt,name=block_size,json=blockSize,proto3" json:"block_size,omitempty"` // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout
unknownFields protoimpl.UnknownFields
sizeCache protoimpl.SizeCache
}
@@ -5200,6 +5209,13 @@ func (x *EcShardConfig) GetEncodeTsNs() int64 {
return 0
}
+func (x *EcShardConfig) GetBlockSize() int64 {
+ if x != nil {
+ return x.BlockSize
+ }
+ return 0
+}
+
// EcBitrotProtection is the entire content of a bitrot checksum sidecar
// ( .ecsum for the legacy generation, .ecsum.v for vacuum
// generation N). On disk it is wrapped in a fixed header carrying a CRC32C
@@ -7509,14 +7525,16 @@ const file_volume_server_proto_rawDesc = "" +
"\tdisk_type\x18\x04 \x01(\tR\bdiskType\" \n" +
"\x1eVolumeEcShardsToVolumeResponse\"8\n" +
"\x19VolumeEcShardsInfoRequest\x12\x1b\n" +
- "\tvolume_id\x18\x01 \x01(\rR\bvolumeId\"\xcf\x01\n" +
+ "\tvolume_id\x18\x01 \x01(\rR\bvolumeId\"\x98\x02\n" +
"\x1aVolumeEcShardsInfoResponse\x12C\n" +
"\x0eec_shard_infos\x18\x01 \x03(\v2\x1d.volume_server_pb.EcShardInfoR\fecShardInfos\x12\x1f\n" +
"\vvolume_size\x18\x02 \x01(\x04R\n" +
"volumeSize\x12\x1d\n" +
"\n" +
"file_count\x18\x03 \x01(\x04R\tfileCount\x12,\n" +
- "\x12file_deleted_count\x18\x04 \x01(\x04R\x10fileDeletedCount\"y\n" +
+ "\x12file_deleted_count\x18\x04 \x01(\x04R\x10fileDeletedCount\x12G\n" +
+ "\x0fec_shard_config\x18\n" +
+ " \x01(\v2\x1f.volume_server_pb.EcShardConfigR\recShardConfig\"y\n" +
"\vEcShardInfo\x12\x19\n" +
"\bshard_id\x18\x01 \x01(\rR\ashardId\x12\x12\n" +
"\x04size\x18\x02 \x01(\x03R\x04size\x12\x1e\n" +
@@ -7583,13 +7601,15 @@ const file_volume_server_proto_rawDesc = "" +
"\rexpire_at_sec\x18\x06 \x01(\x04R\vexpireAtSec\x12\x1b\n" +
"\tread_only\x18\a \x01(\bR\breadOnly\x12G\n" +
"\x0fec_shard_config\x18\b \x01(\v2\x1f.volume_server_pb.EcShardConfigR\recShardConfig\x12/\n" +
- "\x14read_only_can_delete\x18\t \x01(\bR\x11readOnlyCanDelete\"w\n" +
+ "\x14read_only_can_delete\x18\t \x01(\bR\x11readOnlyCanDelete\"\x96\x01\n" +
"\rEcShardConfig\x12\x1f\n" +
"\vdata_shards\x18\x01 \x01(\rR\n" +
"dataShards\x12#\n" +
"\rparity_shards\x18\x02 \x01(\rR\fparityShards\x12 \n" +
"\fencode_ts_ns\x18\x03 \x01(\x03R\n" +
- "encodeTsNs\"\xbc\x02\n" +
+ "encodeTsNs\x12\x1d\n" +
+ "\n" +
+ "block_size\x18\x04 \x01(\x03R\tblockSize\"\xbc\x02\n" +
"\x12EcBitrotProtection\x12A\n" +
"\talgorithm\x18\x01 \x01(\x0e2#.volume_server_pb.ChecksumAlgorithmR\talgorithm\x12\x1d\n" +
"\n" +
@@ -7958,133 +7978,134 @@ var file_volume_server_proto_depIdxs = []int32{
2, // 3: volume_server_pb.SetStateResponse.state:type_name -> volume_server_pb.VolumeServerState
48, // 4: volume_server_pb.ReceiveFileRequest.info:type_name -> volume_server_pb.ReceiveFileInfo
82, // 5: volume_server_pb.VolumeEcShardsInfoResponse.ec_shard_infos:type_name -> volume_server_pb.EcShardInfo
- 88, // 6: volume_server_pb.ReadVolumeFileStatusResponse.volume_info:type_name -> volume_server_pb.VolumeInfo
- 87, // 7: volume_server_pb.VolumeInfo.files:type_name -> volume_server_pb.RemoteFile
- 89, // 8: volume_server_pb.VolumeInfo.ec_shard_config:type_name -> volume_server_pb.EcShardConfig
- 0, // 9: volume_server_pb.EcBitrotProtection.algorithm:type_name -> volume_server_pb.ChecksumAlgorithm
- 89, // 10: volume_server_pb.EcBitrotProtection.ec_shard_config:type_name -> volume_server_pb.EcShardConfig
- 91, // 11: volume_server_pb.EcBitrotProtection.shards:type_name -> volume_server_pb.EcShardChecksums
- 87, // 12: volume_server_pb.OldVersionVolumeInfo.files:type_name -> volume_server_pb.RemoteFile
- 85, // 13: volume_server_pb.VolumeServerStatusResponse.disk_statuses:type_name -> volume_server_pb.DiskStatus
- 86, // 14: volume_server_pb.VolumeServerStatusResponse.memory_status:type_name -> volume_server_pb.MemStatus
- 2, // 15: volume_server_pb.VolumeServerStatusResponse.state:type_name -> volume_server_pb.VolumeServerState
- 113, // 16: volume_server_pb.FetchAndWriteNeedleRequest.replicas:type_name -> volume_server_pb.FetchAndWriteNeedleRequest.Replica
- 122, // 17: volume_server_pb.FetchAndWriteNeedleRequest.remote_conf:type_name -> remote_pb.RemoteConf
- 123, // 18: volume_server_pb.FetchAndWriteNeedleRequest.remote_location:type_name -> remote_pb.RemoteStorageLocation
- 1, // 19: volume_server_pb.ScrubVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode
- 1, // 20: volume_server_pb.ScrubEcVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode
- 82, // 21: volume_server_pb.ScrubEcVolumeResponse.broken_shard_infos:type_name -> volume_server_pb.EcShardInfo
- 114, // 22: volume_server_pb.QueryRequest.filter:type_name -> volume_server_pb.QueryRequest.Filter
- 115, // 23: volume_server_pb.QueryRequest.input_serialization:type_name -> volume_server_pb.QueryRequest.InputSerialization
- 116, // 24: volume_server_pb.QueryRequest.output_serialization:type_name -> volume_server_pb.QueryRequest.OutputSerialization
- 117, // 25: volume_server_pb.QueryRequest.InputSerialization.csv_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.CSVInput
- 118, // 26: volume_server_pb.QueryRequest.InputSerialization.json_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.JSONInput
- 119, // 27: volume_server_pb.QueryRequest.InputSerialization.parquet_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.ParquetInput
- 120, // 28: volume_server_pb.QueryRequest.OutputSerialization.csv_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.CSVOutput
- 121, // 29: volume_server_pb.QueryRequest.OutputSerialization.json_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.JSONOutput
- 3, // 30: volume_server_pb.VolumeServer.BatchDelete:input_type -> volume_server_pb.BatchDeleteRequest
- 7, // 31: volume_server_pb.VolumeServer.VacuumVolumeCheck:input_type -> volume_server_pb.VacuumVolumeCheckRequest
- 9, // 32: volume_server_pb.VolumeServer.VacuumVolumeCompact:input_type -> volume_server_pb.VacuumVolumeCompactRequest
- 11, // 33: volume_server_pb.VolumeServer.VacuumVolumeCommit:input_type -> volume_server_pb.VacuumVolumeCommitRequest
- 13, // 34: volume_server_pb.VolumeServer.VacuumVolumeCleanup:input_type -> volume_server_pb.VacuumVolumeCleanupRequest
- 15, // 35: volume_server_pb.VolumeServer.DeleteCollection:input_type -> volume_server_pb.DeleteCollectionRequest
- 17, // 36: volume_server_pb.VolumeServer.AllocateVolume:input_type -> volume_server_pb.AllocateVolumeRequest
- 19, // 37: volume_server_pb.VolumeServer.VolumeSyncStatus:input_type -> volume_server_pb.VolumeSyncStatusRequest
- 21, // 38: volume_server_pb.VolumeServer.VolumeIncrementalCopy:input_type -> volume_server_pb.VolumeIncrementalCopyRequest
- 23, // 39: volume_server_pb.VolumeServer.VolumeMount:input_type -> volume_server_pb.VolumeMountRequest
- 25, // 40: volume_server_pb.VolumeServer.VolumeUnmount:input_type -> volume_server_pb.VolumeUnmountRequest
- 27, // 41: volume_server_pb.VolumeServer.VolumeConsolidateIndex:input_type -> volume_server_pb.VolumeConsolidateIndexRequest
- 29, // 42: volume_server_pb.VolumeServer.VolumeDelete:input_type -> volume_server_pb.VolumeDeleteRequest
- 31, // 43: volume_server_pb.VolumeServer.VolumeMarkReadonly:input_type -> volume_server_pb.VolumeMarkReadonlyRequest
- 33, // 44: volume_server_pb.VolumeServer.VolumeMarkWritable:input_type -> volume_server_pb.VolumeMarkWritableRequest
- 35, // 45: volume_server_pb.VolumeServer.VolumeConfigure:input_type -> volume_server_pb.VolumeConfigureRequest
- 37, // 46: volume_server_pb.VolumeServer.VolumeStatus:input_type -> volume_server_pb.VolumeStatusRequest
- 39, // 47: volume_server_pb.VolumeServer.GetState:input_type -> volume_server_pb.GetStateRequest
- 41, // 48: volume_server_pb.VolumeServer.SetState:input_type -> volume_server_pb.SetStateRequest
- 43, // 49: volume_server_pb.VolumeServer.VolumeCopy:input_type -> volume_server_pb.VolumeCopyRequest
- 83, // 50: volume_server_pb.VolumeServer.ReadVolumeFileStatus:input_type -> volume_server_pb.ReadVolumeFileStatusRequest
- 45, // 51: volume_server_pb.VolumeServer.CopyFile:input_type -> volume_server_pb.CopyFileRequest
- 47, // 52: volume_server_pb.VolumeServer.ReceiveFile:input_type -> volume_server_pb.ReceiveFileRequest
- 50, // 53: volume_server_pb.VolumeServer.ReadNeedleBlob:input_type -> volume_server_pb.ReadNeedleBlobRequest
- 52, // 54: volume_server_pb.VolumeServer.ReadNeedleMeta:input_type -> volume_server_pb.ReadNeedleMetaRequest
- 54, // 55: volume_server_pb.VolumeServer.WriteNeedleBlob:input_type -> volume_server_pb.WriteNeedleBlobRequest
- 56, // 56: volume_server_pb.VolumeServer.ReadAllNeedles:input_type -> volume_server_pb.ReadAllNeedlesRequest
- 58, // 57: volume_server_pb.VolumeServer.VolumeTailSender:input_type -> volume_server_pb.VolumeTailSenderRequest
- 60, // 58: volume_server_pb.VolumeServer.VolumeTailReceiver:input_type -> volume_server_pb.VolumeTailReceiverRequest
- 62, // 59: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:input_type -> volume_server_pb.VolumeEcShardsGenerateRequest
- 64, // 60: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:input_type -> volume_server_pb.VolumeEcShardsRebuildRequest
- 66, // 61: volume_server_pb.VolumeServer.VolumeEcShardsCopy:input_type -> volume_server_pb.VolumeEcShardsCopyRequest
- 68, // 62: volume_server_pb.VolumeServer.VolumeEcShardsDelete:input_type -> volume_server_pb.VolumeEcShardsDeleteRequest
- 70, // 63: volume_server_pb.VolumeServer.VolumeEcShardsMount:input_type -> volume_server_pb.VolumeEcShardsMountRequest
- 72, // 64: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:input_type -> volume_server_pb.VolumeEcShardsUnmountRequest
- 74, // 65: volume_server_pb.VolumeServer.VolumeEcShardRead:input_type -> volume_server_pb.VolumeEcShardReadRequest
- 76, // 66: volume_server_pb.VolumeServer.VolumeEcBlobDelete:input_type -> volume_server_pb.VolumeEcBlobDeleteRequest
- 78, // 67: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:input_type -> volume_server_pb.VolumeEcShardsToVolumeRequest
- 80, // 68: volume_server_pb.VolumeServer.VolumeEcShardsInfo:input_type -> volume_server_pb.VolumeEcShardsInfoRequest
- 93, // 69: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:input_type -> volume_server_pb.VolumeTierMoveDatToRemoteRequest
- 95, // 70: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:input_type -> volume_server_pb.VolumeTierMoveDatFromRemoteRequest
- 97, // 71: volume_server_pb.VolumeServer.VolumeServerStatus:input_type -> volume_server_pb.VolumeServerStatusRequest
- 99, // 72: volume_server_pb.VolumeServer.VolumeServerLeave:input_type -> volume_server_pb.VolumeServerLeaveRequest
- 101, // 73: volume_server_pb.VolumeServer.FetchAndWriteNeedle:input_type -> volume_server_pb.FetchAndWriteNeedleRequest
- 103, // 74: volume_server_pb.VolumeServer.ScrubVolume:input_type -> volume_server_pb.ScrubVolumeRequest
- 105, // 75: volume_server_pb.VolumeServer.ScrubEcVolume:input_type -> volume_server_pb.ScrubEcVolumeRequest
- 107, // 76: volume_server_pb.VolumeServer.Query:input_type -> volume_server_pb.QueryRequest
- 109, // 77: volume_server_pb.VolumeServer.VolumeNeedleStatus:input_type -> volume_server_pb.VolumeNeedleStatusRequest
- 111, // 78: volume_server_pb.VolumeServer.Ping:input_type -> volume_server_pb.PingRequest
- 4, // 79: volume_server_pb.VolumeServer.BatchDelete:output_type -> volume_server_pb.BatchDeleteResponse
- 8, // 80: volume_server_pb.VolumeServer.VacuumVolumeCheck:output_type -> volume_server_pb.VacuumVolumeCheckResponse
- 10, // 81: volume_server_pb.VolumeServer.VacuumVolumeCompact:output_type -> volume_server_pb.VacuumVolumeCompactResponse
- 12, // 82: volume_server_pb.VolumeServer.VacuumVolumeCommit:output_type -> volume_server_pb.VacuumVolumeCommitResponse
- 14, // 83: volume_server_pb.VolumeServer.VacuumVolumeCleanup:output_type -> volume_server_pb.VacuumVolumeCleanupResponse
- 16, // 84: volume_server_pb.VolumeServer.DeleteCollection:output_type -> volume_server_pb.DeleteCollectionResponse
- 18, // 85: volume_server_pb.VolumeServer.AllocateVolume:output_type -> volume_server_pb.AllocateVolumeResponse
- 20, // 86: volume_server_pb.VolumeServer.VolumeSyncStatus:output_type -> volume_server_pb.VolumeSyncStatusResponse
- 22, // 87: volume_server_pb.VolumeServer.VolumeIncrementalCopy:output_type -> volume_server_pb.VolumeIncrementalCopyResponse
- 24, // 88: volume_server_pb.VolumeServer.VolumeMount:output_type -> volume_server_pb.VolumeMountResponse
- 26, // 89: volume_server_pb.VolumeServer.VolumeUnmount:output_type -> volume_server_pb.VolumeUnmountResponse
- 28, // 90: volume_server_pb.VolumeServer.VolumeConsolidateIndex:output_type -> volume_server_pb.VolumeConsolidateIndexResponse
- 30, // 91: volume_server_pb.VolumeServer.VolumeDelete:output_type -> volume_server_pb.VolumeDeleteResponse
- 32, // 92: volume_server_pb.VolumeServer.VolumeMarkReadonly:output_type -> volume_server_pb.VolumeMarkReadonlyResponse
- 34, // 93: volume_server_pb.VolumeServer.VolumeMarkWritable:output_type -> volume_server_pb.VolumeMarkWritableResponse
- 36, // 94: volume_server_pb.VolumeServer.VolumeConfigure:output_type -> volume_server_pb.VolumeConfigureResponse
- 38, // 95: volume_server_pb.VolumeServer.VolumeStatus:output_type -> volume_server_pb.VolumeStatusResponse
- 40, // 96: volume_server_pb.VolumeServer.GetState:output_type -> volume_server_pb.GetStateResponse
- 42, // 97: volume_server_pb.VolumeServer.SetState:output_type -> volume_server_pb.SetStateResponse
- 44, // 98: volume_server_pb.VolumeServer.VolumeCopy:output_type -> volume_server_pb.VolumeCopyResponse
- 84, // 99: volume_server_pb.VolumeServer.ReadVolumeFileStatus:output_type -> volume_server_pb.ReadVolumeFileStatusResponse
- 46, // 100: volume_server_pb.VolumeServer.CopyFile:output_type -> volume_server_pb.CopyFileResponse
- 49, // 101: volume_server_pb.VolumeServer.ReceiveFile:output_type -> volume_server_pb.ReceiveFileResponse
- 51, // 102: volume_server_pb.VolumeServer.ReadNeedleBlob:output_type -> volume_server_pb.ReadNeedleBlobResponse
- 53, // 103: volume_server_pb.VolumeServer.ReadNeedleMeta:output_type -> volume_server_pb.ReadNeedleMetaResponse
- 55, // 104: volume_server_pb.VolumeServer.WriteNeedleBlob:output_type -> volume_server_pb.WriteNeedleBlobResponse
- 57, // 105: volume_server_pb.VolumeServer.ReadAllNeedles:output_type -> volume_server_pb.ReadAllNeedlesResponse
- 59, // 106: volume_server_pb.VolumeServer.VolumeTailSender:output_type -> volume_server_pb.VolumeTailSenderResponse
- 61, // 107: volume_server_pb.VolumeServer.VolumeTailReceiver:output_type -> volume_server_pb.VolumeTailReceiverResponse
- 63, // 108: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:output_type -> volume_server_pb.VolumeEcShardsGenerateResponse
- 65, // 109: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:output_type -> volume_server_pb.VolumeEcShardsRebuildResponse
- 67, // 110: volume_server_pb.VolumeServer.VolumeEcShardsCopy:output_type -> volume_server_pb.VolumeEcShardsCopyResponse
- 69, // 111: volume_server_pb.VolumeServer.VolumeEcShardsDelete:output_type -> volume_server_pb.VolumeEcShardsDeleteResponse
- 71, // 112: volume_server_pb.VolumeServer.VolumeEcShardsMount:output_type -> volume_server_pb.VolumeEcShardsMountResponse
- 73, // 113: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:output_type -> volume_server_pb.VolumeEcShardsUnmountResponse
- 75, // 114: volume_server_pb.VolumeServer.VolumeEcShardRead:output_type -> volume_server_pb.VolumeEcShardReadResponse
- 77, // 115: volume_server_pb.VolumeServer.VolumeEcBlobDelete:output_type -> volume_server_pb.VolumeEcBlobDeleteResponse
- 79, // 116: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:output_type -> volume_server_pb.VolumeEcShardsToVolumeResponse
- 81, // 117: volume_server_pb.VolumeServer.VolumeEcShardsInfo:output_type -> volume_server_pb.VolumeEcShardsInfoResponse
- 94, // 118: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:output_type -> volume_server_pb.VolumeTierMoveDatToRemoteResponse
- 96, // 119: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:output_type -> volume_server_pb.VolumeTierMoveDatFromRemoteResponse
- 98, // 120: volume_server_pb.VolumeServer.VolumeServerStatus:output_type -> volume_server_pb.VolumeServerStatusResponse
- 100, // 121: volume_server_pb.VolumeServer.VolumeServerLeave:output_type -> volume_server_pb.VolumeServerLeaveResponse
- 102, // 122: volume_server_pb.VolumeServer.FetchAndWriteNeedle:output_type -> volume_server_pb.FetchAndWriteNeedleResponse
- 104, // 123: volume_server_pb.VolumeServer.ScrubVolume:output_type -> volume_server_pb.ScrubVolumeResponse
- 106, // 124: volume_server_pb.VolumeServer.ScrubEcVolume:output_type -> volume_server_pb.ScrubEcVolumeResponse
- 108, // 125: volume_server_pb.VolumeServer.Query:output_type -> volume_server_pb.QueriedStripe
- 110, // 126: volume_server_pb.VolumeServer.VolumeNeedleStatus:output_type -> volume_server_pb.VolumeNeedleStatusResponse
- 112, // 127: volume_server_pb.VolumeServer.Ping:output_type -> volume_server_pb.PingResponse
- 79, // [79:128] is the sub-list for method output_type
- 30, // [30:79] is the sub-list for method input_type
- 30, // [30:30] is the sub-list for extension type_name
- 30, // [30:30] is the sub-list for extension extendee
- 0, // [0:30] is the sub-list for field type_name
+ 89, // 6: volume_server_pb.VolumeEcShardsInfoResponse.ec_shard_config:type_name -> volume_server_pb.EcShardConfig
+ 88, // 7: volume_server_pb.ReadVolumeFileStatusResponse.volume_info:type_name -> volume_server_pb.VolumeInfo
+ 87, // 8: volume_server_pb.VolumeInfo.files:type_name -> volume_server_pb.RemoteFile
+ 89, // 9: volume_server_pb.VolumeInfo.ec_shard_config:type_name -> volume_server_pb.EcShardConfig
+ 0, // 10: volume_server_pb.EcBitrotProtection.algorithm:type_name -> volume_server_pb.ChecksumAlgorithm
+ 89, // 11: volume_server_pb.EcBitrotProtection.ec_shard_config:type_name -> volume_server_pb.EcShardConfig
+ 91, // 12: volume_server_pb.EcBitrotProtection.shards:type_name -> volume_server_pb.EcShardChecksums
+ 87, // 13: volume_server_pb.OldVersionVolumeInfo.files:type_name -> volume_server_pb.RemoteFile
+ 85, // 14: volume_server_pb.VolumeServerStatusResponse.disk_statuses:type_name -> volume_server_pb.DiskStatus
+ 86, // 15: volume_server_pb.VolumeServerStatusResponse.memory_status:type_name -> volume_server_pb.MemStatus
+ 2, // 16: volume_server_pb.VolumeServerStatusResponse.state:type_name -> volume_server_pb.VolumeServerState
+ 113, // 17: volume_server_pb.FetchAndWriteNeedleRequest.replicas:type_name -> volume_server_pb.FetchAndWriteNeedleRequest.Replica
+ 122, // 18: volume_server_pb.FetchAndWriteNeedleRequest.remote_conf:type_name -> remote_pb.RemoteConf
+ 123, // 19: volume_server_pb.FetchAndWriteNeedleRequest.remote_location:type_name -> remote_pb.RemoteStorageLocation
+ 1, // 20: volume_server_pb.ScrubVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode
+ 1, // 21: volume_server_pb.ScrubEcVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode
+ 82, // 22: volume_server_pb.ScrubEcVolumeResponse.broken_shard_infos:type_name -> volume_server_pb.EcShardInfo
+ 114, // 23: volume_server_pb.QueryRequest.filter:type_name -> volume_server_pb.QueryRequest.Filter
+ 115, // 24: volume_server_pb.QueryRequest.input_serialization:type_name -> volume_server_pb.QueryRequest.InputSerialization
+ 116, // 25: volume_server_pb.QueryRequest.output_serialization:type_name -> volume_server_pb.QueryRequest.OutputSerialization
+ 117, // 26: volume_server_pb.QueryRequest.InputSerialization.csv_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.CSVInput
+ 118, // 27: volume_server_pb.QueryRequest.InputSerialization.json_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.JSONInput
+ 119, // 28: volume_server_pb.QueryRequest.InputSerialization.parquet_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.ParquetInput
+ 120, // 29: volume_server_pb.QueryRequest.OutputSerialization.csv_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.CSVOutput
+ 121, // 30: volume_server_pb.QueryRequest.OutputSerialization.json_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.JSONOutput
+ 3, // 31: volume_server_pb.VolumeServer.BatchDelete:input_type -> volume_server_pb.BatchDeleteRequest
+ 7, // 32: volume_server_pb.VolumeServer.VacuumVolumeCheck:input_type -> volume_server_pb.VacuumVolumeCheckRequest
+ 9, // 33: volume_server_pb.VolumeServer.VacuumVolumeCompact:input_type -> volume_server_pb.VacuumVolumeCompactRequest
+ 11, // 34: volume_server_pb.VolumeServer.VacuumVolumeCommit:input_type -> volume_server_pb.VacuumVolumeCommitRequest
+ 13, // 35: volume_server_pb.VolumeServer.VacuumVolumeCleanup:input_type -> volume_server_pb.VacuumVolumeCleanupRequest
+ 15, // 36: volume_server_pb.VolumeServer.DeleteCollection:input_type -> volume_server_pb.DeleteCollectionRequest
+ 17, // 37: volume_server_pb.VolumeServer.AllocateVolume:input_type -> volume_server_pb.AllocateVolumeRequest
+ 19, // 38: volume_server_pb.VolumeServer.VolumeSyncStatus:input_type -> volume_server_pb.VolumeSyncStatusRequest
+ 21, // 39: volume_server_pb.VolumeServer.VolumeIncrementalCopy:input_type -> volume_server_pb.VolumeIncrementalCopyRequest
+ 23, // 40: volume_server_pb.VolumeServer.VolumeMount:input_type -> volume_server_pb.VolumeMountRequest
+ 25, // 41: volume_server_pb.VolumeServer.VolumeUnmount:input_type -> volume_server_pb.VolumeUnmountRequest
+ 27, // 42: volume_server_pb.VolumeServer.VolumeConsolidateIndex:input_type -> volume_server_pb.VolumeConsolidateIndexRequest
+ 29, // 43: volume_server_pb.VolumeServer.VolumeDelete:input_type -> volume_server_pb.VolumeDeleteRequest
+ 31, // 44: volume_server_pb.VolumeServer.VolumeMarkReadonly:input_type -> volume_server_pb.VolumeMarkReadonlyRequest
+ 33, // 45: volume_server_pb.VolumeServer.VolumeMarkWritable:input_type -> volume_server_pb.VolumeMarkWritableRequest
+ 35, // 46: volume_server_pb.VolumeServer.VolumeConfigure:input_type -> volume_server_pb.VolumeConfigureRequest
+ 37, // 47: volume_server_pb.VolumeServer.VolumeStatus:input_type -> volume_server_pb.VolumeStatusRequest
+ 39, // 48: volume_server_pb.VolumeServer.GetState:input_type -> volume_server_pb.GetStateRequest
+ 41, // 49: volume_server_pb.VolumeServer.SetState:input_type -> volume_server_pb.SetStateRequest
+ 43, // 50: volume_server_pb.VolumeServer.VolumeCopy:input_type -> volume_server_pb.VolumeCopyRequest
+ 83, // 51: volume_server_pb.VolumeServer.ReadVolumeFileStatus:input_type -> volume_server_pb.ReadVolumeFileStatusRequest
+ 45, // 52: volume_server_pb.VolumeServer.CopyFile:input_type -> volume_server_pb.CopyFileRequest
+ 47, // 53: volume_server_pb.VolumeServer.ReceiveFile:input_type -> volume_server_pb.ReceiveFileRequest
+ 50, // 54: volume_server_pb.VolumeServer.ReadNeedleBlob:input_type -> volume_server_pb.ReadNeedleBlobRequest
+ 52, // 55: volume_server_pb.VolumeServer.ReadNeedleMeta:input_type -> volume_server_pb.ReadNeedleMetaRequest
+ 54, // 56: volume_server_pb.VolumeServer.WriteNeedleBlob:input_type -> volume_server_pb.WriteNeedleBlobRequest
+ 56, // 57: volume_server_pb.VolumeServer.ReadAllNeedles:input_type -> volume_server_pb.ReadAllNeedlesRequest
+ 58, // 58: volume_server_pb.VolumeServer.VolumeTailSender:input_type -> volume_server_pb.VolumeTailSenderRequest
+ 60, // 59: volume_server_pb.VolumeServer.VolumeTailReceiver:input_type -> volume_server_pb.VolumeTailReceiverRequest
+ 62, // 60: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:input_type -> volume_server_pb.VolumeEcShardsGenerateRequest
+ 64, // 61: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:input_type -> volume_server_pb.VolumeEcShardsRebuildRequest
+ 66, // 62: volume_server_pb.VolumeServer.VolumeEcShardsCopy:input_type -> volume_server_pb.VolumeEcShardsCopyRequest
+ 68, // 63: volume_server_pb.VolumeServer.VolumeEcShardsDelete:input_type -> volume_server_pb.VolumeEcShardsDeleteRequest
+ 70, // 64: volume_server_pb.VolumeServer.VolumeEcShardsMount:input_type -> volume_server_pb.VolumeEcShardsMountRequest
+ 72, // 65: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:input_type -> volume_server_pb.VolumeEcShardsUnmountRequest
+ 74, // 66: volume_server_pb.VolumeServer.VolumeEcShardRead:input_type -> volume_server_pb.VolumeEcShardReadRequest
+ 76, // 67: volume_server_pb.VolumeServer.VolumeEcBlobDelete:input_type -> volume_server_pb.VolumeEcBlobDeleteRequest
+ 78, // 68: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:input_type -> volume_server_pb.VolumeEcShardsToVolumeRequest
+ 80, // 69: volume_server_pb.VolumeServer.VolumeEcShardsInfo:input_type -> volume_server_pb.VolumeEcShardsInfoRequest
+ 93, // 70: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:input_type -> volume_server_pb.VolumeTierMoveDatToRemoteRequest
+ 95, // 71: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:input_type -> volume_server_pb.VolumeTierMoveDatFromRemoteRequest
+ 97, // 72: volume_server_pb.VolumeServer.VolumeServerStatus:input_type -> volume_server_pb.VolumeServerStatusRequest
+ 99, // 73: volume_server_pb.VolumeServer.VolumeServerLeave:input_type -> volume_server_pb.VolumeServerLeaveRequest
+ 101, // 74: volume_server_pb.VolumeServer.FetchAndWriteNeedle:input_type -> volume_server_pb.FetchAndWriteNeedleRequest
+ 103, // 75: volume_server_pb.VolumeServer.ScrubVolume:input_type -> volume_server_pb.ScrubVolumeRequest
+ 105, // 76: volume_server_pb.VolumeServer.ScrubEcVolume:input_type -> volume_server_pb.ScrubEcVolumeRequest
+ 107, // 77: volume_server_pb.VolumeServer.Query:input_type -> volume_server_pb.QueryRequest
+ 109, // 78: volume_server_pb.VolumeServer.VolumeNeedleStatus:input_type -> volume_server_pb.VolumeNeedleStatusRequest
+ 111, // 79: volume_server_pb.VolumeServer.Ping:input_type -> volume_server_pb.PingRequest
+ 4, // 80: volume_server_pb.VolumeServer.BatchDelete:output_type -> volume_server_pb.BatchDeleteResponse
+ 8, // 81: volume_server_pb.VolumeServer.VacuumVolumeCheck:output_type -> volume_server_pb.VacuumVolumeCheckResponse
+ 10, // 82: volume_server_pb.VolumeServer.VacuumVolumeCompact:output_type -> volume_server_pb.VacuumVolumeCompactResponse
+ 12, // 83: volume_server_pb.VolumeServer.VacuumVolumeCommit:output_type -> volume_server_pb.VacuumVolumeCommitResponse
+ 14, // 84: volume_server_pb.VolumeServer.VacuumVolumeCleanup:output_type -> volume_server_pb.VacuumVolumeCleanupResponse
+ 16, // 85: volume_server_pb.VolumeServer.DeleteCollection:output_type -> volume_server_pb.DeleteCollectionResponse
+ 18, // 86: volume_server_pb.VolumeServer.AllocateVolume:output_type -> volume_server_pb.AllocateVolumeResponse
+ 20, // 87: volume_server_pb.VolumeServer.VolumeSyncStatus:output_type -> volume_server_pb.VolumeSyncStatusResponse
+ 22, // 88: volume_server_pb.VolumeServer.VolumeIncrementalCopy:output_type -> volume_server_pb.VolumeIncrementalCopyResponse
+ 24, // 89: volume_server_pb.VolumeServer.VolumeMount:output_type -> volume_server_pb.VolumeMountResponse
+ 26, // 90: volume_server_pb.VolumeServer.VolumeUnmount:output_type -> volume_server_pb.VolumeUnmountResponse
+ 28, // 91: volume_server_pb.VolumeServer.VolumeConsolidateIndex:output_type -> volume_server_pb.VolumeConsolidateIndexResponse
+ 30, // 92: volume_server_pb.VolumeServer.VolumeDelete:output_type -> volume_server_pb.VolumeDeleteResponse
+ 32, // 93: volume_server_pb.VolumeServer.VolumeMarkReadonly:output_type -> volume_server_pb.VolumeMarkReadonlyResponse
+ 34, // 94: volume_server_pb.VolumeServer.VolumeMarkWritable:output_type -> volume_server_pb.VolumeMarkWritableResponse
+ 36, // 95: volume_server_pb.VolumeServer.VolumeConfigure:output_type -> volume_server_pb.VolumeConfigureResponse
+ 38, // 96: volume_server_pb.VolumeServer.VolumeStatus:output_type -> volume_server_pb.VolumeStatusResponse
+ 40, // 97: volume_server_pb.VolumeServer.GetState:output_type -> volume_server_pb.GetStateResponse
+ 42, // 98: volume_server_pb.VolumeServer.SetState:output_type -> volume_server_pb.SetStateResponse
+ 44, // 99: volume_server_pb.VolumeServer.VolumeCopy:output_type -> volume_server_pb.VolumeCopyResponse
+ 84, // 100: volume_server_pb.VolumeServer.ReadVolumeFileStatus:output_type -> volume_server_pb.ReadVolumeFileStatusResponse
+ 46, // 101: volume_server_pb.VolumeServer.CopyFile:output_type -> volume_server_pb.CopyFileResponse
+ 49, // 102: volume_server_pb.VolumeServer.ReceiveFile:output_type -> volume_server_pb.ReceiveFileResponse
+ 51, // 103: volume_server_pb.VolumeServer.ReadNeedleBlob:output_type -> volume_server_pb.ReadNeedleBlobResponse
+ 53, // 104: volume_server_pb.VolumeServer.ReadNeedleMeta:output_type -> volume_server_pb.ReadNeedleMetaResponse
+ 55, // 105: volume_server_pb.VolumeServer.WriteNeedleBlob:output_type -> volume_server_pb.WriteNeedleBlobResponse
+ 57, // 106: volume_server_pb.VolumeServer.ReadAllNeedles:output_type -> volume_server_pb.ReadAllNeedlesResponse
+ 59, // 107: volume_server_pb.VolumeServer.VolumeTailSender:output_type -> volume_server_pb.VolumeTailSenderResponse
+ 61, // 108: volume_server_pb.VolumeServer.VolumeTailReceiver:output_type -> volume_server_pb.VolumeTailReceiverResponse
+ 63, // 109: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:output_type -> volume_server_pb.VolumeEcShardsGenerateResponse
+ 65, // 110: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:output_type -> volume_server_pb.VolumeEcShardsRebuildResponse
+ 67, // 111: volume_server_pb.VolumeServer.VolumeEcShardsCopy:output_type -> volume_server_pb.VolumeEcShardsCopyResponse
+ 69, // 112: volume_server_pb.VolumeServer.VolumeEcShardsDelete:output_type -> volume_server_pb.VolumeEcShardsDeleteResponse
+ 71, // 113: volume_server_pb.VolumeServer.VolumeEcShardsMount:output_type -> volume_server_pb.VolumeEcShardsMountResponse
+ 73, // 114: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:output_type -> volume_server_pb.VolumeEcShardsUnmountResponse
+ 75, // 115: volume_server_pb.VolumeServer.VolumeEcShardRead:output_type -> volume_server_pb.VolumeEcShardReadResponse
+ 77, // 116: volume_server_pb.VolumeServer.VolumeEcBlobDelete:output_type -> volume_server_pb.VolumeEcBlobDeleteResponse
+ 79, // 117: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:output_type -> volume_server_pb.VolumeEcShardsToVolumeResponse
+ 81, // 118: volume_server_pb.VolumeServer.VolumeEcShardsInfo:output_type -> volume_server_pb.VolumeEcShardsInfoResponse
+ 94, // 119: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:output_type -> volume_server_pb.VolumeTierMoveDatToRemoteResponse
+ 96, // 120: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:output_type -> volume_server_pb.VolumeTierMoveDatFromRemoteResponse
+ 98, // 121: volume_server_pb.VolumeServer.VolumeServerStatus:output_type -> volume_server_pb.VolumeServerStatusResponse
+ 100, // 122: volume_server_pb.VolumeServer.VolumeServerLeave:output_type -> volume_server_pb.VolumeServerLeaveResponse
+ 102, // 123: volume_server_pb.VolumeServer.FetchAndWriteNeedle:output_type -> volume_server_pb.FetchAndWriteNeedleResponse
+ 104, // 124: volume_server_pb.VolumeServer.ScrubVolume:output_type -> volume_server_pb.ScrubVolumeResponse
+ 106, // 125: volume_server_pb.VolumeServer.ScrubEcVolume:output_type -> volume_server_pb.ScrubEcVolumeResponse
+ 108, // 126: volume_server_pb.VolumeServer.Query:output_type -> volume_server_pb.QueriedStripe
+ 110, // 127: volume_server_pb.VolumeServer.VolumeNeedleStatus:output_type -> volume_server_pb.VolumeNeedleStatusResponse
+ 112, // 128: volume_server_pb.VolumeServer.Ping:output_type -> volume_server_pb.PingResponse
+ 80, // [80:129] is the sub-list for method output_type
+ 31, // [31:80] is the sub-list for method input_type
+ 31, // [31:31] is the sub-list for extension type_name
+ 31, // [31:31] is the sub-list for extension extendee
+ 0, // [0:31] is the sub-list for field type_name
}
func init() { file_volume_server_proto_init() }
diff --git a/weed/server/volume_grpc_erasure_coding.go b/weed/server/volume_grpc_erasure_coding.go
index f4a1d15ca..14c1a4ee2 100644
--- a/weed/server/volume_grpc_erasure_coding.go
+++ b/weed/server/volume_grpc_erasure_coding.go
@@ -122,9 +122,6 @@ func (vs *VolumeServer) VolumeEcShardsGenerate(ctx context.Context, req *volume_
return nil, fmt.Errorf("WriteSortedFileFromIdx %s: %v", v.IndexFileName(), err)
}
- // snapshot .dat file size before encoding — must match what .ecx references
- datSize, _, _ := v.FileStat()
-
// write .ec00 ~ .ec[TotalShards-1] files using context
ecBitrot, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx)
if err != nil {
@@ -151,7 +148,10 @@ func (vs *VolumeServer) VolumeEcShardsGenerate(ctx context.Context, req *volume_
}
volumeInfo := &volume_server_pb.VolumeInfo{Version: uint32(v.Version())}
volumeInfo.ExpireAtSec = expireAtSec
- volumeInfo.DatFileSize = int64(datSize)
+ // The size the encode actually read, not a separate stat: a replica-sync
+ // write can land between two stats of a live .dat, and the .vif would then
+ // record a DatFileSize and a BlockSize describing different files.
+ volumeInfo.DatFileSize = ecCtx.DatFileSize
// Validate EC configuration before saving to .vif
if ecCtx.DataShards <= 0 || ecCtx.ParityShards <= 0 || ecCtx.Total() > erasure_coding.MaxShardCount {
@@ -165,6 +165,7 @@ func (vs *VolumeServer) VolumeEcShardsGenerate(ctx context.Context, req *volume_
DataShards: uint32(ecCtx.DataShards),
ParityShards: uint32(ecCtx.ParityShards),
EncodeTsNs: time.Now().UnixNano(),
+ BlockSize: ecCtx.BlockSize,
}
glog.V(1).Infof("Saving EC config to .vif for volume %d: %d+%d (total: %d)",
req.VolumeId, ecCtx.DataShards, ecCtx.ParityShards, ecCtx.Total())
@@ -239,17 +240,22 @@ func (vs *VolumeServer) VolumeEcShardsRebuild(ctx context.Context, req *volume_s
// On multi-disk servers, existing local shards may be on a different disk
// than where copied shards were placed during ec.rebuild.
rebuildDataDir := rebuildLocation.Directory
- var additionalDirs []string
- for _, otherLocation := range otherLocationsWithShards {
- additionalDirs = append(additionalDirs, otherLocation.Directory)
- }
+ additionalDirs := rebuildSearchDirs(rebuildLocation, otherLocationsWithShards)
// Rebuild missing EC files, searching all disk locations for input shards.
// Present input shards are verified against the bitrot sidecar (when present)
// and corrupt ones are regenerated; unsafe_ignore_sidecar bypasses the guard.
start := time.Now()
dataBaseFileName := path.Join(rebuildDataDir, baseFileName)
- generatedShardIds, err := erasure_coding.RebuildEcFiles(dataBaseFileName, erasure_coding.BackgroundECContext(), req.UnsafeIgnoreSidecar, additionalDirs...)
+ // Resolve the layout ONCE and use that same answer for the rebuild and for
+ // the backfill below: a manifest describing these shards has to record the
+ // geometry they were actually reconstructed with.
+ rebuildCtx, resolveErr := erasure_coding.ResolveRebuildECContext(dataBaseFileName, erasure_coding.BackgroundECContext(), additionalDirs)
+ if resolveErr != nil {
+ recordEcRebuild("failure", time.Since(start))
+ return nil, fmt.Errorf("resolve rebuild layout for %s: %v", dataBaseFileName, resolveErr)
+ }
+ generatedShardIds, err := erasure_coding.RebuildEcFiles(dataBaseFileName, rebuildCtx, req.UnsafeIgnoreSidecar, additionalDirs...)
if err != nil {
recordEcRebuild("failure", time.Since(start))
return nil, fmt.Errorf("RebuildEcFiles %s: %v", dataBaseFileName, err)
@@ -273,14 +279,19 @@ func (vs *VolumeServer) VolumeEcShardsRebuild(ctx context.Context, req *volume_s
// blesses current bytes; ComputeProtectionFromShards refuses a partial
// manifest, so a multi-server rebuild that cannot reach all shards just skips.
if erasure_coding.BitrotProtectionEnabled {
- sidecarPath := erasure_coding.BitrotSidecarPath(dataBaseFileName, 0)
- if _, statErr := os.Stat(sidecarPath); os.IsNotExist(statErr) {
- ctx := erasure_coding.NewDefaultECContext("", 0)
- if vi, _, found, _ := volume_info.MaybeLoadVolumeInfo(dataBaseFileName + ".vif"); found && vi.EcShardConfig != nil {
- if ds, ps := int(vi.EcShardConfig.DataShards), int(vi.EcShardConfig.ParityShards); ds > 0 && ps > 0 && ds+ps <= erasure_coding.MaxShardCount {
- ctx = &erasure_coding.ECContext{DataShards: ds, ParityShards: ps}
- }
- }
+ // "No sidecar yet" has to be asked of every place one could be, not
+ // just this directory. A split -dir/-dir.idx layout keeps it with the
+ // index and a sibling disk may hold it, and answering from the data
+ // base alone would write a fresh TOFU baseline over a volume that
+ // already has a manifest — blessing whatever the shards currently say
+ // and shadowing the real record, since the data base is searched first.
+ if erasure_coding.FindBitrotSidecar(0, dataBaseFileName, indexBaseFileName, additionalDirs...) == "" {
+ sidecarPath := erasure_coding.BitrotSidecarPath(dataBaseFileName, 0)
+ // The manifest must describe the shards as rebuilt, so it takes the
+ // context the rebuild resolved — not a narrower re-derivation that
+ // reads only this directory's .vif and drops the block size, which
+ // records the legacy layout for shards written with a uniform one.
+ ctx := rebuildCtx
if prot, berr := erasure_coding.ComputeProtectionFromShards(dataBaseFileName, ctx, 0, additionalDirs); berr != nil {
glog.V(2).Infof("bitrot backfill skipped for %s: %v", dataBaseFileName, berr)
} else if werr := erasure_coding.SaveBitrotSidecar(sidecarPath, prot); werr != nil {
@@ -710,6 +721,36 @@ func removeBitrotSidecars(baseFilename string) error {
return firstErr
}
+// rebuildSearchDirs lists every directory besides the rebuild's own data
+// directory that a rebuild may have to read from. Shards are only half of it:
+// a split -dir/-dir.idx layout keeps .ecx/.ecj/.vif with the index, and on a
+// multi-disk server the chosen disk may hold nothing but shards while the
+// volume's .vif or generation-0 .ecsum sits on a sibling. Miss those and the
+// layout resolution falls back to the default ratio and the legacy block
+// size, reconstructing through the wrong matrix. The rebuild's own INDEX
+// directory belongs here too — callers pass the data-directory base name, so
+// it is not otherwise searched. Empty and duplicate entries are dropped.
+func rebuildSearchDirs(rebuildLocation *storage.DiskLocation, otherLocations []*storage.DiskLocation) []string {
+ var dirs []string
+ appendDir := func(dir string) {
+ if dir == "" || dir == rebuildLocation.Directory {
+ return
+ }
+ for _, existing := range dirs {
+ if existing == dir {
+ return
+ }
+ }
+ dirs = append(dirs, dir)
+ }
+ appendDir(rebuildLocation.IdxDirectory)
+ for _, otherLocation := range otherLocations {
+ appendDir(otherLocation.Directory)
+ appendDir(otherLocation.IdxDirectory)
+ }
+ return dirs
+}
+
func checkEcVolumeStatus(bName string, location *storage.DiskLocation) (hasEcxFile bool, hasIdxFile bool, existingShardCount int, err error) {
// check whether to delete the .ecx and .ecj file also
fileInfos, err := os.ReadDir(location.Directory)
@@ -782,6 +823,27 @@ func (vs *VolumeServer) VolumeEcShardsMount(ctx context.Context, req *volume_ser
}
}
+ // A shard delivery can bring the checksum manifest with it, but the receive
+ // path only writes the file. When this server already had the volume
+ // mounted, the EcVolume in memory keeps whatever protection state it
+ // resolved at mount — off, for a volume whose sidecar arrives now — until a
+ // remount. Re-resolve it here, where the shards it describes were added.
+ // Every per-disk runtime, not just the first: a vid mounts as one EcVolume
+ // per disk, the delivery lands the .ecsum on one of them, and the
+ // first-match FindEcVolume would leave the siblings reporting no protection
+ // until a remount. Each re-resolves against its own data and index base, so
+ // a shared -dir.idx reaches all of them.
+ //
+ // Resolving across every EC metadata directory is what makes that reload
+ // mean something. Startup mirroring gives each shard-bearing disk its own
+ // .ecx/.ecj/.vif but deliberately not the sidecar, so a runtime restricted
+ // to its own two directories would find nothing however often it reloaded.
+ // One delivered copy, reachable from all of them.
+ ecMetadataDirs := vs.store.EcMetadataDirs()
+ for _, v := range vs.store.FindAllEcVolumes(needle.VolumeId(req.VolumeId)) {
+ v.ReloadBitrotSidecar(ecMetadataDirs...)
+ }
+
return &volume_server_pb.VolumeEcShardsMountResponse{}, nil
}
@@ -1003,7 +1065,7 @@ func (vs *VolumeServer) VolumeEcShardsToVolume(ctx context.Context, req *volume_
// boundary, so the layout must not be derived from datFileSize. WriteDatFile
// infers the layout from the shard size when .vif does not record it.
// write .dat file from .ec00 ~ .ec09 files
- if err := erasure_coding.WriteDatFile(dataBaseFileName, datFileSize, v.DatFileSize(), shardFileNames); err != nil {
+ if err := erasure_coding.WriteDatFile(dataBaseFileName, datFileSize, v.DatFileSize(), shardFileNames, v.ECContext.LargeBlockSize(), v.ECContext.SmallBlockSize()); err != nil {
return nil, fmt.Errorf("WriteDatFile %s: %v", dataBaseFileName, err)
}
@@ -1179,11 +1241,25 @@ func (vs *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serv
return nil, err
}
+ // Report the layout this holder will actually serve reads through. It is
+ // the only way a coordinator can tell a holder that understands the
+ // uniform block layout from one that dropped the unknown .vif field on the
+ // floor and mounted the volume as legacy.
+ var ecShardConfig *volume_server_pb.EcShardConfig
+ if primary.ECContext != nil {
+ ecShardConfig = &volume_server_pb.EcShardConfig{
+ DataShards: uint32(primary.ECContext.DataShards),
+ ParityShards: uint32(primary.ECContext.ParityShards),
+ BlockSize: primary.ECContext.BlockSize,
+ }
+ }
+
res := &volume_server_pb.VolumeEcShardsInfoResponse{
EcShardInfos: shardInfos,
FileCount: files,
FileDeletedCount: filesDeleted,
VolumeSize: totalSize,
+ EcShardConfig: ecShardConfig,
}
return res, nil
diff --git a/weed/server/volume_grpc_erasure_coding_dirs_test.go b/weed/server/volume_grpc_erasure_coding_dirs_test.go
new file mode 100644
index 000000000..4516fca6f
--- /dev/null
+++ b/weed/server/volume_grpc_erasure_coding_dirs_test.go
@@ -0,0 +1,73 @@
+package weed_server
+
+import (
+ "reflect"
+ "testing"
+
+ "github.com/seaweedfs/seaweedfs/weed/storage"
+)
+
+func loc(dir, idxDir string) *storage.DiskLocation {
+ return &storage.DiskLocation{Directory: dir, IdxDirectory: idxDir}
+}
+
+// The rebuild reads its shards from one directory but resolves the volume's
+// layout — ratio and uniform block size — from the .vif or the generation-0
+// .ecsum, which on a multi-disk server may sit anywhere. Every directory that
+// could hold one has to be in the search list, or the resolution silently
+// falls back to 10+4 with the legacy striping and reconstructs through the
+// wrong matrix.
+func TestRebuildSearchDirs(t *testing.T) {
+ tests := []struct {
+ name string
+ rebuild *storage.DiskLocation
+ others []*storage.DiskLocation
+ want []string
+ }{
+ {
+ name: "the rebuild's own index directory is searched",
+ rebuild: loc("/data1", "/idx1"),
+ want: []string{"/idx1"},
+ },
+ {
+ name: "an unsplit rebuild location contributes nothing",
+ rebuild: loc("/data1", "/data1"),
+ want: nil,
+ },
+ {
+ // The case two reviewers flagged: a sibling holding only shards
+ // while its index directory holds this volume's .vif.
+ name: "a sibling's index directory is searched, not just its data directory",
+ rebuild: loc("/data1", "/data1"),
+ others: []*storage.DiskLocation{loc("/data2", "/idx2")},
+ want: []string{"/data2", "/idx2"},
+ },
+ {
+ name: "shared index directories are listed once",
+ rebuild: loc("/data1", "/shared-idx"),
+ others: []*storage.DiskLocation{loc("/data2", "/shared-idx"), loc("/data3", "/shared-idx")},
+ want: []string{"/shared-idx", "/data2", "/data3"},
+ },
+ {
+ name: "the rebuild's data directory is never repeated",
+ rebuild: loc("/data1", "/idx1"),
+ others: []*storage.DiskLocation{loc("/data1", "/idx1"), loc("/data2", "/data1")},
+ want: []string{"/idx1", "/data2"},
+ },
+ {
+ name: "empty directories are dropped",
+ rebuild: loc("/data1", ""),
+ others: []*storage.DiskLocation{loc("/data2", "")},
+ want: []string{"/data2"},
+ },
+ }
+
+ for _, tc := range tests {
+ t.Run(tc.name, func(t *testing.T) {
+ got := rebuildSearchDirs(tc.rebuild, tc.others)
+ if !reflect.DeepEqual(got, tc.want) {
+ t.Errorf("rebuildSearchDirs() = %v, want %v", got, tc.want)
+ }
+ })
+ }
+}
diff --git a/weed/server/volume_grpc_erasure_coding_recover_test.go b/weed/server/volume_grpc_erasure_coding_recover_test.go
index 9f21edc94..fc0d0d1ab 100644
--- a/weed/server/volume_grpc_erasure_coding_recover_test.go
+++ b/weed/server/volume_grpc_erasure_coding_recover_test.go
@@ -18,6 +18,7 @@ import (
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
"github.com/seaweedfs/seaweedfs/weed/storage/types"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
"github.com/seaweedfs/seaweedfs/weed/util"
)
@@ -111,7 +112,12 @@ func TestFetchEcIndexFromPeers_CopiesIndexOverGrpc(t *testing.T) {
if err := os.WriteFile(srcBase+".ecj", []byte("journal"), 0o644); err != nil {
t.Fatalf("write source .ecj: %v", err)
}
- if err := os.WriteFile(srcBase+".vif", []byte("volinfo"), 0o644); err != nil {
+ // A real .vif: the receiver MOUNTS the volume from the copied files, and a
+ // mount refuses a .vif it cannot parse rather than guessing the layout.
+ if err := volume_info.SaveVolumeInfo(srcBase+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10, ParityShards: 4},
+ }); err != nil {
t.Fatalf("write source .vif: %v", err)
}
@@ -223,7 +229,12 @@ func TestVolumeEcShardsMount_RecoverMissingIndex(t *testing.T) {
if err := os.WriteFile(srcBase+".ecj", nil, 0o644); err != nil {
t.Fatalf("write source .ecj: %v", err)
}
- if err := os.WriteFile(srcBase+".vif", []byte("volinfo"), 0o644); err != nil {
+ // A real .vif: the receiver MOUNTS the volume from the copied files, and a
+ // mount refuses a .vif it cannot parse rather than guessing the layout.
+ if err := volume_info.SaveVolumeInfo(srcBase+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10, ParityShards: 4},
+ }); err != nil {
t.Fatalf("write source .vif: %v", err)
}
srcGrpcPort := serveGrpc(t, func(s *grpc.Server) {
diff --git a/weed/storage/disk_location_ec.go b/weed/storage/disk_location_ec.go
index 218abc8d6..71987b550 100644
--- a/weed/storage/disk_location_ec.go
+++ b/weed/storage/disk_location_ec.go
@@ -484,26 +484,17 @@ func (l *DiskLocation) checkOrphanedShards(shards []string, collection string, v
// erasure_coding.DataShardsCount so that tests writing a custom layout
// to .vif compute the matching shard size, and so custom-ratio builds
// (e.g. enterprise) can swap the default without touching this helper.
+// The padded shard length is the same number under both layouts —
+// TestUniformBlockSizeMatchesLegacyShardSize asserts the equivalence for every
+// input — so defer to the encoder's own helper instead of keeping a second copy
+// of the padding rule that a future change would have to be made in twice. An
+// empty .dat keeps its historic answer: the legacy encoder emits no block for
+// it, where UniformBlockSize floors at one.
func calculateExpectedShardSize(datFileSize int64, dataShardCount int) int64 {
- if dataShardCount <= 0 {
+ if dataShardCount <= 0 || datFileSize <= 0 {
return 0
}
- var shardSize int64
-
- // Process large blocks (1GB * dataShardCount per batch)
- largeBatchSize := int64(erasure_coding.ErasureCodingLargeBlockSize) * int64(dataShardCount)
- numLargeBatches := datFileSize / largeBatchSize
- shardSize = numLargeBatches * int64(erasure_coding.ErasureCodingLargeBlockSize)
- remainingSize := datFileSize - (numLargeBatches * largeBatchSize)
-
- // Process remaining data in small blocks (1MB * dataShardCount per batch)
- if remainingSize > 0 {
- smallBatchSize := int64(erasure_coding.ErasureCodingSmallBlockSize) * int64(dataShardCount)
- numSmallBatches := (remainingSize + smallBatchSize - 1) / smallBatchSize // Ceiling division
- shardSize += numSmallBatches * int64(erasure_coding.ErasureCodingSmallBlockSize)
- }
-
- return shardSize
+ return erasure_coding.UniformBlockSize(datFileSize, dataShardCount)
}
// validateEcVolume reports whether the EC files for (collection, vid) on this
diff --git a/weed/storage/erasure_coding/ec_bitrot.go b/weed/storage/erasure_coding/ec_bitrot.go
index 5aef042c7..dfe78a5f2 100644
--- a/weed/storage/erasure_coding/ec_bitrot.go
+++ b/weed/storage/erasure_coding/ec_bitrot.go
@@ -195,6 +195,7 @@ func buildProtectionFromBuilders(ctx *ECContext, builders []*shardChecksumBuilde
EcShardConfig: &volume_server_pb.EcShardConfig{
DataShards: uint32(ctx.DataShards),
ParityShards: uint32(ctx.ParityShards),
+ BlockSize: ctx.BlockSize,
},
Shards: shards,
EncodeUuid: NewEncodeUUID(),
@@ -431,6 +432,7 @@ func ComputeProtectionFromShards(baseFileName string, ctx *ECContext, generation
EcShardConfig: &volume_server_pb.EcShardConfig{
DataShards: uint32(ctx.DataShards),
ParityShards: uint32(ctx.ParityShards),
+ BlockSize: ctx.BlockSize,
},
Shards: shards,
EncodeUuid: NewEncodeUUID(),
@@ -474,53 +476,141 @@ func (ev *EcVolume) BitrotProtection() (*volume_server_pb.EcBitrotProtection, Bi
return ev.bitrot, ev.bitrotStatus
}
+// ReloadBitrotSidecar re-resolves the checksum sidecar for a volume that is
+// already mounted — a shard delivery can bring the manifest with it, and the
+// receive path only writes the file, so without this the in-memory volume
+// keeps the protection state it resolved at mount (off) until a remount.
+func (ev *EcVolume) ReloadBitrotSidecar(additionalDirs ...string) {
+ if err := ev.loadActiveBitrotSidecar(additionalDirs...); err != nil {
+ glog.Warningf("ec volume %d: reload bitrot sidecar: %v", ev.VolumeId, err)
+ }
+}
+
// loadActiveBitrotSidecar loads the generation-0 checksum sidecar into the
// volume. Best-effort: any failure leaves protection off/invalid without
// failing the mount. OSS only produces generation-0 (fresh-encode) sidecars.
-func (ev *EcVolume) loadActiveBitrotSidecar() {
- ev.loadBitrotForGeneration(0)
+func (ev *EcVolume) loadActiveBitrotSidecar(additionalDirs ...string) error {
+ return ev.loadBitrotForGeneration(0, additionalDirs...)
}
// loadBitrotForGeneration loads and validates the sidecar describing generation
// `generation`, setting ev.bitrot/ev.bitrotStatus. Called at mount. Absent or
-// generation/config-mismatched => BitrotOff; self-integrity/manifest failure =>
+// generation-mismatched => BitrotOff; self-integrity/manifest failure =>
// BitrotInvalid; usable => BitrotOn.
-func (ev *EcVolume) loadBitrotForGeneration(generation uint32) {
+//
+// A sidecar written FOR THIS generation that disagrees with the volume's shard
+// geometry is different in kind from those: both files record the layout the
+// generation was encoded with, so a disagreement means one of them is wrong and
+// reads through the other would land at the wrong shard offsets. That returns
+// an error and fails the mount instead of quietly dropping to unprotected
+// reads.
+func (ev *EcVolume) loadBitrotForGeneration(generation uint32, additionalDirs ...string) error {
ev.bitrotLock.Lock()
defer ev.bitrotLock.Unlock()
ev.bitrot = nil
ev.bitrotStatus = BitrotOff
if ev.ECContext == nil {
- return
+ return nil
}
- path := findBitrotSidecar(generation, ev.DataBaseFileName(), ev.IndexBaseFileName())
+ path := findBitrotSidecar(generation, ev.DataBaseFileName(), ev.IndexBaseFileName(), additionalDirs...)
if path == "" {
- return
+ return nil
}
prot, err := LoadBitrotSidecar(path)
if err != nil {
glog.Warningf("ec volume %d: bitrot sidecar %s self-integrity failed: %v", ev.VolumeId, path, err)
ev.bitrotStatus = BitrotInvalid
- return
+ return nil
}
if prot.Generation != generation {
- return // not for this generation -> off, not corruption
+ return nil // not for this generation -> off, not corruption
}
- if prot.EcShardConfig == nil ||
- int(prot.EcShardConfig.DataShards) != ev.ECContext.DataShards ||
- int(prot.EcShardConfig.ParityShards) != ev.ECContext.ParityShards {
- return
+ if prot.EcShardConfig == nil {
+ return nil // records no geometry -> nothing to contradict
+ }
+ if int(prot.EcShardConfig.DataShards) != ev.ECContext.DataShards ||
+ int(prot.EcShardConfig.ParityShards) != ev.ECContext.ParityShards ||
+ prot.EcShardConfig.BlockSize != ev.ECContext.BlockSize {
+ return fmt.Errorf("ec volume %d generation %d: %s records layout %d+%d block %d but the volume is mounted as %d+%d block %d; refusing to serve one of the two layouts",
+ ev.VolumeId, generation, path,
+ prot.EcShardConfig.DataShards, prot.EcShardConfig.ParityShards, prot.EcShardConfig.BlockSize,
+ ev.ECContext.DataShards, ev.ECContext.ParityShards, ev.ECContext.BlockSize)
}
if err := ValidateBitrotManifest(prot, ev.ECContext.DataShards, ev.ECContext.ParityShards); err != nil {
glog.Warningf("ec volume %d: bitrot sidecar %s manifest invalid: %v", ev.VolumeId, path, err)
ev.bitrotStatus = BitrotInvalid
- return
+ return nil
}
ev.bitrot = prot
ev.bitrotStatus = BitrotOn
glog.V(1).Infof("ec volume %d: loaded bitrot protection generation %d (%d shards, block_size %d)",
ev.VolumeId, generation, len(prot.Shards), prot.BlockSize)
+ return nil
+}
+
+// layoutFromSidecar resolves a volume's EC layout from its generation-0
+// checksum sidecar, searching the data base and then the index base. A split
+// -dir/-dir.idx layout keeps the sidecar with the index, so probing the data
+// base alone reports "absent" for a volume that has the record right there —
+// and absent is the one answer that selects the legacy 10+4 layout.
+//
+// The three answers stay distinct: (nil, false, nil) means genuinely absent
+// and the caller may fall back to the defaults; a non-nil error means the
+// record exists but cannot establish the layout, which must fail rather than
+// default; otherwise the config is usable.
+func layoutFromSidecar(dataBaseFileName, indexBaseFileName string) (cfg *volume_server_pb.EcShardConfig, found bool, err error) {
+ path := findBitrotSidecar(0, dataBaseFileName, indexBaseFileName)
+ if path == "" {
+ return nil, false, nil
+ }
+ cfg, err = EcShardConfigFromSidecarPath(path)
+ return cfg, true, err
+}
+
+// EcShardConfigFromSidecar reads the generation-0 bitrot sidecar's record of a
+// volume's EC config. The three answers are distinct on purpose: absent means
+// the volume may genuinely predate the sidecar, which is the only case a caller
+// may answer with the legacy layout; present-but-unusable means the one
+// surviving record of the geometry is corrupt, and guessing from there returns
+// wrong bytes rather than no bytes.
+func EcShardConfigFromSidecar(baseFileName string) (cfg *volume_server_pb.EcShardConfig, found bool, err error) {
+ path := BitrotSidecarPath(baseFileName, 0)
+ if _, statErr := os.Stat(path); statErr != nil {
+ if os.IsNotExist(statErr) {
+ return nil, false, nil
+ }
+ return nil, false, fmt.Errorf("stat %s: %w", path, statErr)
+ }
+ cfg, err = EcShardConfigFromSidecarPath(path)
+ return cfg, true, err
+}
+
+// EcShardConfigFromSidecarPath is EcShardConfigFromSidecar for a sidecar whose
+// path the caller already resolved — the rebuild finds it across a multi-disk
+// server's directories, not only next to the base name.
+func EcShardConfigFromSidecarPath(path string) (*volume_server_pb.EcShardConfig, error) {
+ prot, loadErr := LoadBitrotSidecar(path)
+ if loadErr != nil {
+ return nil, fmt.Errorf("read %s: %w", path, loadErr)
+ }
+ // The un-suffixed sidecar describes generation 0 and nothing else; one
+ // stamped for another generation is not a record of these shards.
+ if prot.GetGeneration() != 0 {
+ return nil, fmt.Errorf("%s records generation %d, not generation 0", path, prot.GetGeneration())
+ }
+ cfg := prot.GetEcShardConfig()
+ if cfg == nil {
+ return nil, fmt.Errorf("%s records no EC config", path)
+ }
+ if !ValidEcShardCounts(cfg.GetDataShards(), cfg.GetParityShards()) {
+ return nil, fmt.Errorf("%s records invalid shard counts %d+%d",
+ path, cfg.GetDataShards(), cfg.GetParityShards())
+ }
+ if bsErr := ValidateBlockSize(cfg.GetBlockSize()); bsErr != nil {
+ return nil, fmt.Errorf("%s: %w", path, bsErr)
+ }
+ return cfg, nil
}
// RemoveBitrotSidecars removes the legacy .ecsum and any versioned
@@ -534,6 +624,15 @@ func RemoveBitrotSidecars(base string) {
}
}
+// FindBitrotSidecar is findBitrotSidecar for callers outside this package. It
+// answers "does this volume already have a manifest, and where" — a question
+// that cannot be settled from one base name, because a split -dir/-dir.idx
+// layout keeps the sidecar with the index and a multi-disk server may keep it
+// on a sibling. Returns "" when no candidate exists.
+func FindBitrotSidecar(generation uint32, dataBase, indexBase string, additionalDirs ...string) string {
+ return findBitrotSidecar(generation, dataBase, indexBase, additionalDirs...)
+}
+
// findBitrotSidecar resolves the sidecar path for a generation, searching the
// data base, the index base, and any additional directories — mirroring how
// shard/.vif lookups handle split data/idx layouts and per-disk mirrors.
diff --git a/weed/storage/erasure_coding/ec_context.go b/weed/storage/erasure_coding/ec_context.go
index 9fc352d83..ac77f688b 100644
--- a/weed/storage/erasure_coding/ec_context.go
+++ b/weed/storage/erasure_coding/ec_context.go
@@ -13,6 +13,16 @@ type ECContext struct {
ParityShards int
Collection string
VolumeId needle.VolumeId
+ // BlockSize > 0 selects the uniform block layout: every shard is a single
+ // contiguous block of this many bytes, so consecutive .dat ranges stay on
+ // one shard instead of striping across all of them at 1MiB granularity.
+ // 0 is the legacy two-tier 1GiB/1MiB layout.
+ BlockSize int64
+ // DatFileSize is the length of the .dat the encode actually read, set by
+ // WriteEcFiles from the same measurement it derives BlockSize from. The
+ // .vif must record the two together: they describe one file, and a live
+ // volume can grow between two separate stats.
+ DatFileSize int64
}
// Total returns the total number of shards (data + parity)
@@ -20,6 +30,33 @@ func (ctx *ECContext) Total() int {
return ctx.DataShards + ctx.ParityShards
}
+// ValidEcShardCounts reports whether a recorded (data, parity) pair could
+// describe a real EC volume. The sum is taken in uint64 deliberately: on a
+// 32-bit build `int` is 32 bits, so converting each count first and adding
+// them wraps for values near the uint32 ceiling — 0x7fffffff + 0x7fffffff
+// lands at -2, which slips under the MaxShardCount bound.
+func ValidEcShardCounts(dataShards, parityShards uint32) bool {
+ return dataShards > 0 && parityShards > 0 &&
+ uint64(dataShards)+uint64(parityShards) <= uint64(MaxShardCount)
+}
+
+// LargeBlockSize returns the large-block length of this context's shard
+// layout; nil-safe so callers can pass an unset context for the legacy layout.
+func (ctx *ECContext) LargeBlockSize() int64 {
+ if ctx != nil && ctx.BlockSize > 0 {
+ return ctx.BlockSize
+ }
+ return ErasureCodingLargeBlockSize
+}
+
+// SmallBlockSize returns the small-block length of this context's shard layout.
+func (ctx *ECContext) SmallBlockSize() int64 {
+ if ctx != nil && ctx.BlockSize > 0 {
+ return ctx.BlockSize
+ }
+ return ErasureCodingSmallBlockSize
+}
+
// NewDefaultECContext creates a context with default 10+4 shard configuration
func NewDefaultECContext(collection string, volumeId needle.VolumeId) *ECContext {
return &ECContext{
diff --git a/weed/storage/erasure_coding/ec_decoder.go b/weed/storage/erasure_coding/ec_decoder.go
index 5bf4342d3..96494c4a5 100644
--- a/weed/storage/erasure_coding/ec_decoder.go
+++ b/weed/storage/erasure_coding/ec_decoder.go
@@ -233,8 +233,10 @@ func iterateEcjFile(baseFileName string, processNeedleFn func(key types.NeedleId
// large-block row boundary, and deriving the layout from the shrunk extent
// would read the shards in the wrong block order. Pass zero when the .vif does
// not record the encode-time size to infer the layout from the shard size.
-func WriteDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int64, shardFileNames []string) error {
- return writeDatFile(baseFileName, datFileSize, encodedDatFileSize, shardFileNames, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize)
+// largeBlockSize/smallBlockSize are the volume's shard block layout, e.g.
+// ctx.LargeBlockSize()/ctx.SmallBlockSize() from its .vif EC config.
+func WriteDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int64, shardFileNames []string, largeBlockSize int64, smallBlockSize int64) error {
+ return writeDatFile(baseFileName, datFileSize, encodedDatFileSize, shardFileNames, largeBlockSize, smallBlockSize)
}
func writeDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int64, shardFileNames []string, largeBlockSize int64, smallBlockSize int64) error {
@@ -288,7 +290,11 @@ func writeDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int
// A shard size that is an exact multiple of the large block size is
// ambiguous: N large rows, or N-1 large rows plus a full small-block
// region. The two layouts only agree below the last large row.
- if shardSize%largeBlockSize == 0 && datFileSize > (shardSize/largeBlockSize-1)*largeBlockSize*int64(dataShards) {
+ // A uniform layout has no such ambiguity — its large and small blocks
+ // are the same size, so every reading of the shard is the same one, and
+ // without this precondition the check fires on every volume.
+ if largeBlockSize != smallBlockSize &&
+ shardSize%largeBlockSize == 0 && datFileSize > (shardSize/largeBlockSize-1)*largeBlockSize*int64(dataShards) {
return fmt.Errorf("shard size %d of %s does not identify the block layout; re-encode to record the dat size in .vif", shardSize, baseFileName)
}
encodedDatFileSize = int64(dataShards) * shardSize
diff --git a/weed/storage/erasure_coding/ec_decoder_test.go b/weed/storage/erasure_coding/ec_decoder_test.go
index 8c655d7fa..b09fed947 100644
--- a/weed/storage/erasure_coding/ec_decoder_test.go
+++ b/weed/storage/erasure_coding/ec_decoder_test.go
@@ -217,7 +217,7 @@ func TestDecodeAtomicPublish(t *testing.T) {
// final .dat nor a partial .dat.tmp behind.
datBase := filepath.Join(dir, "bar_2")
missingShards := []string{filepath.Join(dir, "does_not_exist.ec00")}
- if err := erasure_coding.WriteDatFile(datBase, 100, 100, missingShards); err == nil {
+ if err := erasure_coding.WriteDatFile(datBase, 100, 100, missingShards, erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize); err == nil {
t.Fatalf("expected WriteDatFile to fail on missing shard")
}
if _, err := os.Stat(datBase + ".dat"); !os.IsNotExist(err) {
diff --git a/weed/storage/erasure_coding/ec_encoder.go b/weed/storage/erasure_coding/ec_encoder.go
index 927a50d86..1b28fddf1 100644
--- a/weed/storage/erasure_coding/ec_encoder.go
+++ b/weed/storage/erasure_coding/ec_encoder.go
@@ -62,12 +62,126 @@ func WriteSortedFileFromIdx(baseFileName string, ext string) (e error) {
// BackgroundECContext for the default ratio, or an explicit ctx for a configured
// (e.g. custom-ratio) layout. It returns the bitrot protection (per-shard block
// CRC32C) computed during the single encode pass; the caller persists it as a
-// .ecsum sidecar.
+// .ecsum sidecar, and persists ctx.BlockSize and ctx.DatFileSize (both
+// set here, from one measurement of the .dat) to the .vif so readers resolve
+// the shard block layout.
func WriteEcFiles(baseFileName string, ctx *ECContext) (*volume_server_pb.EcBitrotProtection, error) {
- if ctx == nil || ctx.Total() == 0 {
+ if ctx == nil {
ctx = NewDefaultECContext("", 0)
+ } else if ctx.Total() == 0 {
+ // Fill the placeholder in place rather than swapping the pointer: the
+ // caller reads BlockSize and DatFileSize back off the context it
+ // passed, and a replacement leaves it holding the zero values.
+ ctx.DataShards, ctx.ParityShards = DataShardsCount, ParityShardsCount
}
- return generateEcFiles(baseFileName, 256*1024, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize, ctx)
+ // Always encode with the uniform block layout, sized for this .dat. Both
+ // the block size and the .dat length it was derived from are left on ctx,
+ // so the caller persists a .vif whose two fields describe one measurement
+ // — a second stat could see a different size on a volume still taking
+ // writes.
+ fi, err := os.Stat(baseFileName + ".dat")
+ if err != nil {
+ return nil, fmt.Errorf("failed to stat dat file: %w", err)
+ }
+ ctx.DatFileSize = fi.Size()
+ ctx.BlockSize = UniformBlockSize(fi.Size(), ctx.DataShards)
+ return generateEcFiles(baseFileName, 256*1024, ctx.BlockSize, ctx.BlockSize, ctx)
+}
+
+// ValidateBlockSize reports whether a `.vif`-recorded shard block size is one
+// an encoder could have produced. 0 means the legacy two-tier layout, which is
+// always valid; anything positive must be a whole number of small blocks,
+// because that is what UniformBlockSize rounds to. A negative or unaligned
+// value is corruption, and using it would map every read to the wrong shard
+// offset.
+func ValidateBlockSize(blockSize int64) error {
+ if blockSize == 0 {
+ return nil
+ }
+ if blockSize < 0 || blockSize%ErasureCodingSmallBlockSize != 0 {
+ return fmt.Errorf("invalid shard block size %d: expected 0 (legacy) or a multiple of %d",
+ blockSize, ErasureCodingSmallBlockSize)
+ }
+ return nil
+}
+
+// UniformBlockSize returns the per-shard block size of the uniform layout for
+// a .dat of the given size: ceil(datFileSize/dataShards) rounded up to a whole
+// small block. For every input this equals the legacy layout's padded shard
+// size, so only the byte placement differs between the two layouts, never the
+// shard length.
+func UniformBlockSize(datFileSize int64, dataShards int) int64 {
+ perShard := (datFileSize + int64(dataShards) - 1) / int64(dataShards)
+ blocks := (perShard + ErasureCodingSmallBlockSize - 1) / ErasureCodingSmallBlockSize
+ if blocks < 1 {
+ blocks = 1
+ }
+ return blocks * ErasureCodingSmallBlockSize
+}
+
+// ResolveRebuildECContext answers which shard layout a rebuild of baseFileName
+// will use: the caller's context when it already states one, else the volume's
+// own metadata — its `.vif` in any of the directories the rebuild can read,
+// then the generation-0 bitrot sidecar, and only then the build defaults.
+// Exported so a caller that must agree with the rebuild (the post-rebuild
+// bitrot backfill writes a manifest describing these very shards) resolves
+// once and uses the same answer, instead of re-deriving it from a narrower
+// search and recording a layout the rebuild did not use.
+func ResolveRebuildECContext(baseFileName string, ctx *ECContext, additionalDirs []string) (*ECContext, error) {
+ if ctx == nil || ctx.Total() == 0 {
+ // Resolve the layout from the .vif to preserve the original configuration.
+ vifPath := findVifPath(baseFileName, additionalDirs)
+ volumeInfo, _, foundVif, vifErr := volume_info.MaybeLoadVolumeInfo(vifPath)
+ if vifErr != nil {
+ // The .vif exists but cannot be read or parsed. Fail closed rather
+ // than silently falling back to the default ratio, which would
+ // rebuild a custom-ratio volume with the wrong layout. Pass an
+ // explicit ctx to override.
+ return nil, fmt.Errorf("RebuildEcFiles %s: cannot load .vif: %w", baseFileName, vifErr)
+ }
+ switch {
+ case foundVif && volumeInfo.EcShardConfig != nil &&
+ ValidEcShardCounts(volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards):
+ if bsErr := ValidateBlockSize(volumeInfo.EcShardConfig.GetBlockSize()); bsErr != nil {
+ return nil, fmt.Errorf("RebuildEcFiles %s: %s: %w", baseFileName, vifPath, bsErr)
+ }
+ ctx = &ECContext{
+ DataShards: int(volumeInfo.EcShardConfig.DataShards),
+ ParityShards: int(volumeInfo.EcShardConfig.ParityShards),
+ BlockSize: volumeInfo.EcShardConfig.GetBlockSize(),
+ }
+ glog.V(0).Infof("Rebuilding EC files for %s with config from .vif: %s", baseFileName, ctx.String())
+ case foundVif && volumeInfo.EcShardConfig != nil:
+ // A recorded-but-impossible ratio is corruption, not a reason to
+ // substitute the default one: a 12+4 volume rebuilt as 10+4
+ // reconstructs from the wrong matrix and never regenerates shards
+ // 14-15. Pass an explicit ctx to override.
+ return nil, fmt.Errorf("RebuildEcFiles %s: %s records invalid shard counts %d+%d",
+ baseFileName, vifPath, volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards)
+ default:
+ // No usable .vif: the bitrot sidecar records the same config at
+ // encode time and is then the surviving authority. Reading the
+ // default ratio and the legacy block size instead would rebuild
+ // from the wrong geometry AND make the sidecar look like it
+ // disagrees, which silently skips every checksum check below.
+ if sidecarPath := findBitrotSidecar(0, baseFileName, baseFileName, additionalDirs...); sidecarPath != "" {
+ cfg, cfgErr := EcShardConfigFromSidecarPath(sidecarPath)
+ if cfgErr != nil {
+ return nil, fmt.Errorf("RebuildEcFiles %s: no usable .vif and %w", baseFileName, cfgErr)
+ }
+ ctx = &ECContext{
+ DataShards: int(cfg.GetDataShards()),
+ ParityShards: int(cfg.GetParityShards()),
+ BlockSize: cfg.GetBlockSize(),
+ }
+ glog.V(0).Infof("Rebuilding EC files for %s with config from the bitrot sidecar: %s", baseFileName, ctx.String())
+ break
+ }
+ glog.V(0).Infof("Rebuilding EC files for %s with default config", baseFileName)
+ ctx = NewDefaultECContext("", 0)
+ }
+ }
+ return ctx, nil
}
// RebuildEcFiles rebuilds missing EC shard files. Pass BackgroundECContext to
@@ -79,38 +193,11 @@ func WriteEcFiles(baseFileName string, ctx *ECContext) (*volume_server_pb.EcBitr
// present input shards are verified against it and corrupt ones are excluded
// from Reed-Solomon and regenerated; unsafeIgnoreSidecar bypasses that guard.
func RebuildEcFiles(baseFileName string, ctx *ECContext, unsafeIgnoreSidecar bool, additionalDirs ...string) ([]uint32, error) {
- if ctx == nil || ctx.Total() == 0 {
- // Resolve the layout from the .vif to preserve the original configuration.
- volumeInfo, _, foundVif, vifErr := volume_info.MaybeLoadVolumeInfo(baseFileName + ".vif")
- if vifErr != nil {
- // The .vif exists but cannot be read or parsed. Fail closed rather
- // than silently falling back to the default ratio, which would
- // rebuild a custom-ratio volume with the wrong layout. Pass an
- // explicit ctx to override.
- return nil, fmt.Errorf("RebuildEcFiles %s: cannot load .vif: %w", baseFileName, vifErr)
- }
- if foundVif && volumeInfo.EcShardConfig != nil {
- ds := int(volumeInfo.EcShardConfig.DataShards)
- ps := int(volumeInfo.EcShardConfig.ParityShards)
-
- // Validate EC config before using it
- if ds > 0 && ps > 0 && ds+ps <= MaxShardCount {
- ctx = &ECContext{
- DataShards: ds,
- ParityShards: ps,
- }
- glog.V(0).Infof("Rebuilding EC files for %s with config from .vif: %s", baseFileName, ctx.String())
- } else {
- glog.Warningf("Invalid EC config in .vif for %s (data=%d, parity=%d), using default", baseFileName, ds, ps)
- ctx = NewDefaultECContext("", 0)
- }
- } else {
- glog.V(0).Infof("Rebuilding EC files for %s with default config", baseFileName)
- ctx = NewDefaultECContext("", 0)
- }
+ ctx, err := ResolveRebuildECContext(baseFileName, ctx, additionalDirs)
+ if err != nil {
+ return nil, err
}
-
- return generateMissingEcFiles(baseFileName, 256*1024, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize, ctx, unsafeIgnoreSidecar, additionalDirs)
+ return generateMissingEcFiles(baseFileName, 256*1024, ctx, unsafeIgnoreSidecar, additionalDirs)
}
func ToExt(ecIndex int) string {
@@ -159,7 +246,10 @@ func findShardFile(baseFileName string, ext string, additionalDirs []string) str
return ""
}
-func generateMissingEcFiles(baseFileName string, bufferSize int, largeBlockSize int64, smallBlockSize int64, ctx *ECContext, unsafeIgnoreSidecar bool, additionalDirs []string) (generatedShardIds []uint32, err error) {
+// generateMissingEcFiles takes no block sizes: Reed-Solomon reconstruction is
+// layout-agnostic — it rebuilds a missing shard from the same offsets of the
+// survivors — so the shard layout only ever reaches it through ctx.
+func generateMissingEcFiles(baseFileName string, bufferSize int, ctx *ECContext, unsafeIgnoreSidecar bool, additionalDirs []string) (generatedShardIds []uint32, err error) {
// Pass 1: discover which shards exist and which are missing,
// opening input files but NOT creating output files yet.
@@ -363,6 +453,30 @@ func cleanupRebuildOutputs(outputFiles []*os.File, writePaths []string) {
}
}
+// findVifPath locates the volume's `.vif` for a rebuild: next to the shards
+// first, then in every directory the caller also handed us — the index
+// directory of a split `-dir`/`-dir.idx` layout, and the sibling disks of a
+// multi-disk server, where `.ecx`/`.ecj`/`.vif` may live while this disk holds
+// only shards. Returns the data-base path when nothing exists, so the caller's
+// "not found" handling stays on the canonical name.
+func findVifPath(baseFileName string, additionalDirs []string) string {
+ dataPath := baseFileName + ".vif"
+ candidates := []string{dataPath}
+ base := filepath.Base(baseFileName)
+ for _, dir := range additionalDirs {
+ if dir == "" {
+ continue
+ }
+ candidates = append(candidates, filepath.Join(dir, base)+".vif")
+ }
+ for _, c := range candidates {
+ if _, err := os.Stat(c); err == nil {
+ return c
+ }
+ }
+ return dataPath
+}
+
// loadRebuildSidecar loads and validates the generation-0 checksum sidecar for a
// rebuild. RebuildEcFiles operates on the un-suffixed (generation 0) shard
// names, so only the legacy sidecar is relevant here. Returns BitrotOff when
@@ -381,10 +495,20 @@ func loadRebuildSidecar(baseFileName string, ctx *ECContext, additionalDirs []st
if prot.Generation != 0 {
return nil, BitrotOff
}
- if prot.EcShardConfig == nil ||
- int(prot.EcShardConfig.DataShards) != ctx.DataShards ||
- int(prot.EcShardConfig.ParityShards) != ctx.ParityShards {
- return nil, BitrotOff
+ if prot.EcShardConfig == nil {
+ return nil, BitrotOff // records no geometry -> nothing to contradict
+ }
+ if int(prot.EcShardConfig.DataShards) != ctx.DataShards ||
+ int(prot.EcShardConfig.ParityShards) != ctx.ParityShards ||
+ prot.EcShardConfig.BlockSize != ctx.BlockSize {
+ // Both records describe the same encode, so a disagreement means the
+ // rebuild is about to reconstruct under a geometry the checksums do not
+ // cover. Treating that as "no protection" skipped every input and
+ // output check exactly when they matter most.
+ glog.Warningf("bitrot: sidecar %s records layout %d+%d block %d but the rebuild uses %d+%d block %d",
+ path, prot.EcShardConfig.DataShards, prot.EcShardConfig.ParityShards, prot.EcShardConfig.BlockSize,
+ ctx.DataShards, ctx.ParityShards, ctx.BlockSize)
+ return nil, BitrotInvalid
}
if err := ValidateBitrotManifest(prot, ctx.DataShards, ctx.ParityShards); err != nil {
glog.Warningf("bitrot: sidecar %s manifest invalid: %v", path, err)
diff --git a/weed/storage/erasure_coding/ec_rebuild_safety_test.go b/weed/storage/erasure_coding/ec_rebuild_safety_test.go
index 99b912470..3ccb4a77b 100644
--- a/weed/storage/erasure_coding/ec_rebuild_safety_test.go
+++ b/weed/storage/erasure_coding/ec_rebuild_safety_test.go
@@ -5,7 +5,10 @@ import (
"path/filepath"
"testing"
+ "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
+ "github.com/seaweedfs/seaweedfs/weed/storage/needle"
"github.com/seaweedfs/seaweedfs/weed/storage/types"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
)
// readEcxSizeField returns the size field of the .ecx entry for needleId, or
@@ -295,3 +298,101 @@ func TestRebuildEcFiles_CustomRatioRebuildsByteIdentical(t *testing.T) {
t.Errorf("unexpected shard .ec12 for a 9+3 volume")
}
}
+
+// With no .vif, the bitrot sidecar is the surviving record of the volume's
+// geometry. Defaulting to 10+4 and the legacy block layout instead would
+// reconstruct a custom-ratio volume from the wrong matrix — and would make the
+// sidecar look like it disagrees, silently skipping every checksum check.
+func TestRebuildEcFiles_ResolvesGeometryFromSidecarWithoutVif(t *testing.T) {
+ if !BitrotProtectionEnabled {
+ t.Skip("bitrot protection is compiled out")
+ }
+ dir := t.TempDir()
+ base := filepath.Join(dir, "vol")
+ ctx := &ECContext{DataShards: 12, ParityShards: 4}
+ writeRandomDat(t, base, 7000)
+
+ prot, err := WriteEcFiles(base, ctx)
+ if err != nil {
+ t.Fatalf("WriteEcFiles: %v", err)
+ }
+ if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil {
+ t.Fatalf("save sidecar: %v", err)
+ }
+ // No .vif at all: the sidecar has to answer.
+ const dropped = 11
+ want, err := os.ReadFile(base + ToExt(dropped))
+ if err != nil {
+ t.Fatalf("read shard: %v", err)
+ }
+ if err := os.Remove(base + ToExt(dropped)); err != nil {
+ t.Fatalf("remove shard: %v", err)
+ }
+
+ // nil ctx: the production RPC's BackgroundECContext path.
+ if _, err := RebuildEcFiles(base, nil, false); err != nil {
+ t.Fatalf("RebuildEcFiles: %v", err)
+ }
+ got, err := os.ReadFile(base + ToExt(dropped))
+ if err != nil {
+ t.Fatalf("read rebuilt shard: %v", err)
+ }
+ if string(got) != string(want) {
+ t.Errorf("rebuilt shard %d differs; the rebuild used the wrong geometry", dropped)
+ }
+ // The 12+4 shards past the default ratio must still be there.
+ for _, id := range []int{14, 15} {
+ if _, err := os.Stat(base + ToExt(id)); err != nil {
+ t.Errorf("shard %d missing after rebuild: %v", id, err)
+ }
+ }
+}
+
+// A split -dir/-dir.idx layout keeps the .vif with the index, and a multi-disk
+// server may leave this disk holding only shards. Probing just the data base
+// resolves such a volume to the default ratio and the legacy block layout.
+func TestRebuildEcFiles_FindsVifInAnAdditionalDir(t *testing.T) {
+ dataDir := t.TempDir()
+ idxDir := t.TempDir()
+ base := filepath.Join(dataDir, "vol")
+ ctx := &ECContext{DataShards: 12, ParityShards: 4}
+ writeRandomDat(t, base, 7000)
+
+ if _, err := WriteEcFiles(base, ctx); err != nil {
+ t.Fatalf("WriteEcFiles: %v", err)
+ }
+ // The volume's .vif lives with the index, not the shards.
+ if err := volume_info.SaveVolumeInfo(filepath.Join(idxDir, "vol")+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.GetCurrentVersion()),
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 12, ParityShards: 4, BlockSize: ctx.BlockSize,
+ },
+ }); err != nil {
+ t.Fatalf("save .vif: %v", err)
+ }
+
+ const dropped = 11
+ want, err := os.ReadFile(base + ToExt(dropped))
+ if err != nil {
+ t.Fatalf("read shard: %v", err)
+ }
+ if err := os.Remove(base + ToExt(dropped)); err != nil {
+ t.Fatalf("remove shard: %v", err)
+ }
+
+ if _, err := RebuildEcFiles(base, nil, true, idxDir); err != nil {
+ t.Fatalf("RebuildEcFiles: %v", err)
+ }
+ got, err := os.ReadFile(base + ToExt(dropped))
+ if err != nil {
+ t.Fatalf("read rebuilt shard: %v", err)
+ }
+ if string(got) != string(want) {
+ t.Errorf("rebuilt shard %d differs; the rebuild used the wrong geometry", dropped)
+ }
+ for _, id := range []int{14, 15} {
+ if _, err := os.Stat(base + ToExt(id)); err != nil {
+ t.Errorf("shard %d missing after rebuild: %v", id, err)
+ }
+ }
+}
diff --git a/weed/storage/erasure_coding/ec_roundtrip_test.go b/weed/storage/erasure_coding/ec_roundtrip_test.go
index 8c07a338d..f70c23970 100644
--- a/weed/storage/erasure_coding/ec_roundtrip_test.go
+++ b/weed/storage/erasure_coding/ec_roundtrip_test.go
@@ -326,7 +326,7 @@ func testDecodeDat(t *testing.T, datSize int64) {
shardFileNames[i] = fmt.Sprintf("%s%s", baseFileName, ctx.ToExt(i))
}
- err = WriteDatFile(decodedBase, datSize, datSize, shardFileNames)
+ err = WriteDatFile(decodedBase, datSize, datSize, shardFileNames, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize)
require.NoError(t, err, "WriteDatFile")
// The atomic publish must rename the temp file away, never leaving it behind.
diff --git a/weed/storage/erasure_coding/ec_uniform_layout_test.go b/weed/storage/erasure_coding/ec_uniform_layout_test.go
new file mode 100644
index 000000000..215fd7bfa
--- /dev/null
+++ b/weed/storage/erasure_coding/ec_uniform_layout_test.go
@@ -0,0 +1,250 @@
+package erasure_coding
+
+import (
+ "bytes"
+ "fmt"
+ "math/rand"
+ "os"
+ "testing"
+
+ "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
+ "github.com/seaweedfs/seaweedfs/weed/storage/needle"
+ "github.com/seaweedfs/seaweedfs/weed/storage/types"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
+)
+
+// legacyShardSize is the padded shard length the two-tier layout produces:
+// whole 1GiB rows, then the remainder in 1MiB rows.
+func legacyShardSize(datFileSize int64, dataShards int) int64 {
+ largeRow := int64(ErasureCodingLargeBlockSize) * int64(dataShards)
+ nLarge := datFileSize / largeRow
+ size := nLarge * ErasureCodingLargeBlockSize
+ if rem := datFileSize - nLarge*largeRow; rem > 0 {
+ smallRow := int64(ErasureCodingSmallBlockSize) * int64(dataShards)
+ size += (rem + smallRow - 1) / smallRow * ErasureCodingSmallBlockSize
+ }
+ return size
+}
+
+// The uniform layout must never change a shard's length, only the byte
+// placement — capacity math and shard-size credibility checks depend on it.
+func TestUniformBlockSizeMatchesLegacyShardSize(t *testing.T) {
+ sizes := []int64{
+ 8, 1000, ErasureCodingSmallBlockSize,
+ 10 * ErasureCodingSmallBlockSize,
+ 10*ErasureCodingSmallBlockSize + 1,
+ 25 * 1024 * 1024,
+ 10 * ErasureCodingLargeBlockSize,
+ 10*ErasureCodingLargeBlockSize + 1,
+ 10*ErasureCodingLargeBlockSize - ErasureCodingSmallBlockSize,
+ 30 * 1024 * 1024 * 1024,
+ 30*1024*1024*1024 + 12345,
+ }
+ for _, datSize := range sizes {
+ for _, ds := range []int{10, 9, 5, 1} {
+ if got, want := UniformBlockSize(datSize, ds), legacyShardSize(datSize, ds); got != want {
+ t.Errorf("UniformBlockSize(%d, %d) = %d, legacy shard size %d", datSize, ds, got, want)
+ }
+ }
+ }
+ if got := UniformBlockSize(0, 10); got != ErasureCodingSmallBlockSize {
+ t.Errorf("UniformBlockSize(0, 10) = %d, want one small block", got)
+ }
+}
+
+// encodeLayoutFixture writes a random .dat and encodes it with the given block
+// sizes, returning the .dat content.
+func encodeLayoutFixture(t *testing.T, baseFileName string, datSize int64, large, small int64, ctx *ECContext) []byte {
+ t.Helper()
+ data := make([]byte, datSize)
+ rand.New(rand.NewSource(datSize)).Read(data)
+ if err := os.WriteFile(baseFileName+".dat", data, 0o644); err != nil {
+ t.Fatalf("write .dat: %v", err)
+ }
+ if _, err := generateEcFiles(baseFileName, 256*1024, large, small, ctx); err != nil {
+ t.Fatalf("generateEcFiles: %v", err)
+ }
+ return data
+}
+
+// readIntervals assembles a byte range from the shard files the way the read
+// path does.
+func readIntervals(t *testing.T, baseFileName string, ctx *ECContext, intervals []Interval) []byte {
+ t.Helper()
+ var assembled []byte
+ for _, iv := range intervals {
+ shardId, shardOffset := iv.ToShardIdAndOffset(ctx.LargeBlockSize(), ctx.SmallBlockSize())
+ f, err := os.Open(baseFileName + ctx.ToExt(int(shardId)))
+ if err != nil {
+ t.Fatalf("open shard %d: %v", shardId, err)
+ }
+ buf := make([]byte, iv.Size)
+ if _, err := f.ReadAt(buf, shardOffset); err != nil {
+ t.Fatalf("read shard %d at %d: %v", shardId, shardOffset, err)
+ }
+ f.Close()
+ assembled = append(assembled, buf...)
+ }
+ return assembled
+}
+
+// Encode 25MB — large enough that the uniform and legacy layouts place bytes
+// differently — under both layouts and verify LocateData maps every probed
+// range back to the original bytes.
+func TestLocateDataMatchesEncoderForBothLayouts(t *testing.T) {
+ const datSize = 25*1024*1024 + 12345
+ dir := t.TempDir()
+
+ uniformCtx := NewDefaultECContext("", 1)
+ uniformCtx.BlockSize = UniformBlockSize(datSize, uniformCtx.DataShards)
+ legacyCtx := NewDefaultECContext("", 2)
+
+ for name, ctx := range map[string]*ECContext{"uniform": uniformCtx, "legacy": legacyCtx} {
+ t.Run(name, func(t *testing.T) {
+ base := fmt.Sprintf("%s/%s", dir, name)
+ data := encodeLayoutFixture(t, base, datSize, ctx.LargeBlockSize(), ctx.SmallBlockSize(), ctx)
+
+ shardSize := datSize / int64(ctx.DataShards)
+ for _, probe := range []struct{ offset, size int64 }{
+ {0, 1000},
+ {ctx.SmallBlockSize() - 100, 200}, // straddles a block boundary
+ {5 * 1024 * 1024, 4 * 1024 * 1024},
+ {datSize - 2000, 2000},
+ } {
+ intervals := LocateData(ctx.LargeBlockSize(), ctx.SmallBlockSize(), shardSize, probe.offset, types.Size(probe.size))
+ got := readIntervals(t, base, ctx, intervals)
+ if !bytes.Equal(got, data[probe.offset:probe.offset+probe.size]) {
+ t.Fatalf("%s layout: bytes at [%d,+%d) do not round-trip", name, probe.offset, probe.size)
+ }
+ }
+
+ // The point of the uniform layout: a 2MB range inside one 3MB
+ // block is a single interval instead of three 1MB stripes.
+ intervals := LocateData(ctx.LargeBlockSize(), ctx.SmallBlockSize(), shardSize, 6*1024*1024+512*1024, types.Size(2*1024*1024))
+ if name == "uniform" && len(intervals) != 1 {
+ t.Errorf("uniform layout reads 2MB in %d intervals, want 1", len(intervals))
+ }
+ if name == "legacy" && len(intervals) != 3 {
+ t.Errorf("legacy layout reads 2MB in %d intervals, want 3", len(intervals))
+ }
+ })
+ }
+}
+
+// Encode under both layouts and verify WriteDatFile reconstructs the original
+// .dat byte for byte — and that decoding a uniform volume with the legacy
+// geometry does NOT, i.e. the recorded block size is load-bearing.
+func TestUniformDecodeRoundTrip(t *testing.T) {
+ const datSize = 25 * 1024 * 1024
+ dir := t.TempDir()
+ base := dir + "/1"
+
+ ctx := NewDefaultECContext("", 1)
+ ctx.BlockSize = UniformBlockSize(datSize, ctx.DataShards)
+ data := encodeLayoutFixture(t, base, datSize, ctx.BlockSize, ctx.BlockSize, ctx)
+
+ shardFileNames := make([]string, ctx.DataShards)
+ for i := range shardFileNames {
+ shardFileNames[i] = base + ctx.ToExt(i)
+ }
+
+ if err := WriteDatFile(dir+"/decoded", datSize, datSize, shardFileNames, ctx.LargeBlockSize(), ctx.SmallBlockSize()); err != nil {
+ t.Fatalf("WriteDatFile: %v", err)
+ }
+ decoded, err := os.ReadFile(dir + "/decoded.dat")
+ if err != nil {
+ t.Fatalf("read decoded: %v", err)
+ }
+ if !bytes.Equal(decoded, data) {
+ t.Fatal("uniform decode with recorded geometry does not round-trip")
+ }
+
+ if err := WriteDatFile(dir+"/wrong", datSize, datSize, shardFileNames, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize); err != nil {
+ t.Fatalf("WriteDatFile legacy geometry: %v", err)
+ }
+ wrong, err := os.ReadFile(dir + "/wrong.dat")
+ if err != nil {
+ t.Fatalf("read wrong: %v", err)
+ }
+ if bytes.Equal(wrong, data) {
+ t.Fatal("decoding a uniform volume with legacy geometry should scramble it; the layouts did not diverge")
+ }
+}
+
+// Mount fixtures through NewEcVolume and read a large extent back through
+// LocateEcShardNeedleInterval, proving the .vif block size steers the read
+// path: a legacy .vif (no block size) keeps the legacy interpretation, a
+// uniform .vif reads uniform shards.
+func TestEcVolumeGeometryFromVif(t *testing.T) {
+ const datSize = 25 * 1024 * 1024
+
+ for _, layout := range []string{"legacy", "uniform"} {
+ t.Run(layout, func(t *testing.T) {
+ dir := t.TempDir()
+ vid := needle.VolumeId(7)
+ base := EcShardFileName("", dir, int(vid))
+
+ ctx := NewDefaultECContext("", vid)
+ if layout == "uniform" {
+ ctx.BlockSize = UniformBlockSize(datSize, ctx.DataShards)
+ }
+ data := encodeLayoutFixture(t, base, datSize, ctx.LargeBlockSize(), ctx.SmallBlockSize(), ctx)
+
+ if err := os.WriteFile(base+".ecx", nil, 0o644); err != nil {
+ t.Fatalf("write .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.GetCurrentVersion()),
+ DatFileSize: datSize,
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: uint32(ctx.DataShards),
+ ParityShards: uint32(ctx.ParityShards),
+ BlockSize: ctx.BlockSize,
+ },
+ }); err != nil {
+ t.Fatalf("save .vif: %v", err)
+ }
+
+ ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", vid)
+ if err != nil {
+ t.Fatalf("NewEcVolume: %v", err)
+ }
+ defer ev.Close()
+ if ev.ECContext.BlockSize != ctx.BlockSize {
+ t.Fatalf("loaded BlockSize = %d, want %d", ev.ECContext.BlockSize, ctx.BlockSize)
+ }
+ for i := 0; i < ctx.DataShards; i++ {
+ shard, err := NewEcVolumeShard(types.HardDriveType, dir, "", vid, ShardId(i))
+ if err != nil {
+ t.Fatalf("NewEcVolumeShard %d: %v", i, err)
+ }
+ ev.AddEcVolumeShard(shard)
+ }
+
+ // Sweep a large extent through the volume's own interval mapping.
+ // LocateEcShardNeedleInterval expands the size by the needle
+ // overhead, so compare only the probed prefix.
+ const probeSize = datSize - 64*1024
+ intervals := ev.LocateEcShardNeedleInterval(needle.GetCurrentVersion(), 0, types.Size(probeSize))
+ var assembled []byte
+ for _, iv := range intervals {
+ shardId, shardOffset := ev.IntervalToShardIdAndOffset(iv)
+ shard, found := ev.FindEcVolumeShard(shardId)
+ if !found {
+ t.Fatalf("shard %d not mounted", shardId)
+ }
+ buf := make([]byte, iv.Size)
+ if _, err := shard.ReadAt(buf, shardOffset); err != nil {
+ t.Fatalf("read shard %d at %d: %v", shardId, shardOffset, err)
+ }
+ assembled = append(assembled, buf...)
+ }
+ if len(assembled) < probeSize {
+ t.Fatalf("assembled %d bytes, want at least %d", len(assembled), probeSize)
+ }
+ if !bytes.Equal(assembled[:probeSize], data[:probeSize]) {
+ t.Fatalf("%s layout: full-extent read through the %s .vif does not match the .dat", layout, layout)
+ }
+ })
+ }
+}
diff --git a/weed/storage/erasure_coding/ec_volume.go b/weed/storage/erasure_coding/ec_volume.go
index 8bb4a4e8f..cb6463948 100644
--- a/weed/storage/erasure_coding/ec_volume.go
+++ b/weed/storage/erasure_coding/ec_volume.go
@@ -173,7 +173,16 @@ func NewEcVolume(diskType types.DiskType, dir string, dirIdx string, collection
}
}
ev.Version = needle.Version3
- if volumeInfo, _, found, _ := volume_info.MaybeLoadVolumeInfo(vifFileName); found {
+ // A present-but-unreadable or malformed .vif FAILS the mount: every new
+ // encode records a positive uniform block size there, and defaulting to
+ // the legacy layout would serve those shards with the wrong offset math.
+ // Absent stays legal — legacy volumes predate the sidecar.
+ volumeInfo, _, found, vifErr := volume_info.MaybeLoadVolumeInfo(vifFileName)
+ if vifErr != nil {
+ ev.Close()
+ return nil, fmt.Errorf("ec volume %d: load %s: %w", vid, vifFileName, vifErr)
+ }
+ if found {
ev.Version = needle.Version(volumeInfo.Version)
ev.datFileSize = volumeInfo.DatFileSize
ev.ExpireAtSec = volumeInfo.ExpireAtSec
@@ -184,21 +193,56 @@ func NewEcVolume(diskType types.DiskType, dir string, dirIdx string, collection
ps := int(volumeInfo.EcShardConfig.ParityShards)
ev.EncodeTsNs = volumeInfo.EcShardConfig.GetEncodeTsNs()
- // Validate shard counts to prevent zero or invalid values
- if ds <= 0 || ps <= 0 || ds+ps > MaxShardCount {
- glog.Warningf("Invalid EC config in VolumeInfo for volume %d (data=%d, parity=%d), using defaults", vid, ds, ps)
- ev.ECContext = NewDefaultECContext(collection, vid)
+ // A config that is PRESENT but records an impossible ratio is not
+ // a volume to fall back on: substituting the default 10+4 with the
+ // legacy layout would read uniform shards with the wrong offset
+ // math and answer with the wrong bytes. Only an ENTIRELY absent
+ // config means "this predates the record", which the else-branch
+ // below serves with the legacy defaults.
+ if !ValidEcShardCounts(volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards) {
+ ev.Close()
+ return nil, fmt.Errorf("ec volume %d: %s records invalid shard counts %d+%d",
+ vid, vifFileName, volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards)
+ } else if blockErr := ValidateBlockSize(volumeInfo.EcShardConfig.GetBlockSize()); blockErr != nil {
+ // A recorded block size that no encoder could have produced maps
+ // every read to the wrong shard offset. Refuse the mount rather
+ // than serve those bytes or silently pick a layout.
+ ev.Close()
+ return nil, fmt.Errorf("ec volume %d: %s: %w", vid, vifFileName, blockErr)
} else {
ev.ECContext = &ECContext{
Collection: collection,
VolumeId: vid,
DataShards: ds,
ParityShards: ps,
+ BlockSize: volumeInfo.EcShardConfig.GetBlockSize(),
}
glog.V(1).Infof("Loaded EC config from VolumeInfo for volume %d: %s", vid, ev.ECContext.String())
}
} else {
- ev.ECContext = NewDefaultECContext(collection, vid)
+ // A vif that carries no ecShardConfig answers nothing about the
+ // layout — it is no more informative than an absent one, so it
+ // must not skip the sidecar. Going straight to the defaults here
+ // read a uniform volume with the legacy offset math.
+ cfg, sidecarFound, sidecarErr := layoutFromSidecar(dataBaseFileName, indexBaseFileName)
+ if sidecarErr != nil {
+ ev.Close()
+ return nil, fmt.Errorf("ec volume %d: %s records no EC config and the bitrot sidecar cannot establish the layout: %w", vid, vifFileName, sidecarErr)
+ }
+ if sidecarFound {
+ ev.ECContext = &ECContext{
+ Collection: collection,
+ VolumeId: vid,
+ DataShards: int(cfg.GetDataShards()),
+ ParityShards: int(cfg.GetParityShards()),
+ BlockSize: cfg.GetBlockSize(),
+ }
+ ev.EncodeTsNs = cfg.GetEncodeTsNs()
+ glog.V(0).Infof("ec volume %d: .vif records no EC config; took it from the bitrot sidecar: %s",
+ vid, ev.ECContext.String())
+ } else {
+ ev.ECContext = NewDefaultECContext(collection, vid)
+ }
}
} else {
// Don't fabricate a stub .vif here: a version-only stub implies the
@@ -207,14 +251,48 @@ func NewEcVolume(diskType types.DiskType, dir string, dirIdx string, collection
// mistake for an authoritative config. Mount with in-memory defaults and
// leave the real .vif to the encoder or a recovery tool (the Rust volume
// server already behaves this way).
- glog.Warningf("vif file not found, using defaults, volumeId:%d, filename:%s", vid, vifFileName)
+ //
+ // The bitrot sidecar records the same EC config at encode time, so when
+ // it is present it answers the layout question the missing .vif cannot:
+ // defaulting a uniform-layout volume to the legacy block sizes maps
+ // every read to the wrong shard offset. `weed fix -ecx` reads the
+ // sidecar for the same reason.
ev.ECContext = NewDefaultECContext(collection, vid)
+ cfg, sidecarFound, sidecarErr := layoutFromSidecar(dataBaseFileName, indexBaseFileName)
+ if sidecarErr != nil {
+ // With no .vif the sidecar is the ONLY record of this volume's
+ // layout. Present but unusable is not "assume legacy" — that
+ // answers reads with the wrong shard offsets, which is worse than
+ // not answering at all.
+ ev.Close()
+ return nil, fmt.Errorf("ec volume %d: no .vif and the bitrot sidecar cannot establish the layout: %w", vid, sidecarErr)
+ }
+ if sidecarFound {
+ ev.ECContext = &ECContext{
+ Collection: collection,
+ VolumeId: vid,
+ DataShards: int(cfg.GetDataShards()),
+ ParityShards: int(cfg.GetParityShards()),
+ BlockSize: cfg.GetBlockSize(),
+ }
+ ev.EncodeTsNs = cfg.GetEncodeTsNs()
+ glog.V(0).Infof("ec volume %d: .vif missing; took EC config from the bitrot sidecar: %s",
+ vid, ev.ECContext.String())
+ } else {
+ // Only now are the defaults what the volume actually mounted on;
+ // logging this after the sidecar answered would send an operator
+ // triaging wrong bytes after the legacy layout instead.
+ glog.Warningf("vif file not found, using defaults, volumeId:%d, filename:%s", vid, vifFileName)
+ }
}
ev.ShardLocations = make(map[ShardId][]pb.ServerAddress)
// Load the active-generation bitrot checksum sidecar (optional).
- ev.loadActiveBitrotSidecar()
+ if err := ev.loadActiveBitrotSidecar(); err != nil {
+ ev.Close()
+ return nil, err
+ }
return
}
@@ -528,11 +606,17 @@ func (ev *EcVolume) LocateEcShardNeedleInterval(version needle.Version, offset i
shardSize = shard.ecdFileSize - 1
}
// calculate the locations in the ec shards
- intervals = LocateData(ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize, shardSize, offset, types.Size(needle.GetActualSize(size, version)))
+ intervals = LocateData(ev.ECContext.LargeBlockSize(), ev.ECContext.SmallBlockSize(), shardSize, offset, types.Size(needle.GetActualSize(size, version)))
return
}
+// IntervalToShardIdAndOffset resolves an interval against this volume's shard
+// block layout.
+func (ev *EcVolume) IntervalToShardIdAndOffset(interval Interval) (ShardId, int64) {
+ return interval.ToShardIdAndOffset(ev.ECContext.LargeBlockSize(), ev.ECContext.SmallBlockSize())
+}
+
func (ev *EcVolume) FindNeedleFromEcx(needleId types.NeedleId) (offset types.Offset, size types.Size, err error) {
offset, size, err = SearchNeedleFromSortedIndex(ev.ecxFile, ev.ecxFileSize, needleId, nil)
if err != nil {
diff --git a/weed/storage/erasure_coding/ec_volume_scrub.go b/weed/storage/erasure_coding/ec_volume_scrub.go
index 8cb33297e..40457f58e 100644
--- a/weed/storage/erasure_coding/ec_volume_scrub.go
+++ b/weed/storage/erasure_coding/ec_volume_scrub.go
@@ -238,7 +238,7 @@ func (ecv *EcVolume) ScrubLocal() (int64, []*volume_server_pb.EcShardInfo, []err
localShardIds := []ShardId{}
for i, iv := range locations {
- sid, soffset := iv.ToShardIdAndOffset(ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize)
+ sid, soffset := ecv.IntervalToShardIdAndOffset(iv)
ssize := int64(iv.Size.Raw())
shard, found := ecv.FindEcVolumeShard(sid)
diff --git a/weed/storage/erasure_coding/ec_volume_sidecar_layout_test.go b/weed/storage/erasure_coding/ec_volume_sidecar_layout_test.go
new file mode 100644
index 000000000..34e992416
--- /dev/null
+++ b/weed/storage/erasure_coding/ec_volume_sidecar_layout_test.go
@@ -0,0 +1,246 @@
+package erasure_coding
+
+import (
+ "os"
+ "path/filepath"
+ "testing"
+
+ "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
+ "github.com/seaweedfs/seaweedfs/weed/storage/needle"
+ "github.com/seaweedfs/seaweedfs/weed/storage/types"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
+)
+
+func uniformSidecar(t *testing.T, base string, ds, ps uint32, blockSize int64) {
+ t.Helper()
+ prot := &volume_server_pb.EcBitrotProtection{
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 0,
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: ds, ParityShards: ps, BlockSize: blockSize,
+ },
+ }
+ if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil {
+ t.Fatalf("seed .ecsum: %v", err)
+ }
+}
+
+// The sidecar records the layout at encode time, so it answers what a .vif
+// cannot. A .vif that merely omits ecShardConfig is no more informative than an
+// absent one — going straight to the defaults reads a uniform volume with the
+// legacy offset math and returns the wrong bytes.
+func TestNewEcVolumeTakesLayoutFromSidecarWhenVifHasNoConfig(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ // A config-free .vif: version and dat size only, as a legacy encode left it.
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ uniformSidecar(t, base, 12, 4, 3*1024*1024)
+
+ ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1))
+ if err != nil {
+ t.Fatalf("mount: %v", err)
+ }
+ defer ev.Close()
+ if got := ev.ECContext; got.DataShards != 12 || got.ParityShards != 4 || got.BlockSize != 3*1024*1024 {
+ t.Errorf("layout = %d+%d block %d, want 12+4 block 3MiB", got.DataShards, got.ParityShards, got.BlockSize)
+ }
+}
+
+// A split -dir/-dir.idx layout keeps the sidecar with the INDEX. Probing only
+// the data base reports "absent", and absent is the one answer that selects the
+// legacy layout.
+func TestNewEcVolumeFindsSidecarInTheIndexDirectory(t *testing.T) {
+ for _, tc := range []struct {
+ name string
+ writeVif bool
+ }{
+ {name: "no vif at all"},
+ {name: "vif without ec config", writeVif: true},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ dataDir, idxDir := t.TempDir(), t.TempDir()
+ idxBase := filepath.Join(idxDir, "2")
+ if err := os.WriteFile(idxBase+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if tc.writeVif {
+ if err := volume_info.SaveVolumeInfo(idxBase+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ }
+ uniformSidecar(t, idxBase, 12, 4, 3*1024*1024)
+
+ ev, err := NewEcVolume(types.HardDriveType, dataDir, idxDir, "", needle.VolumeId(2))
+ if err != nil {
+ t.Fatalf("mount: %v", err)
+ }
+ defer ev.Close()
+ if got := ev.ECContext; got.DataShards != 12 || got.ParityShards != 4 || got.BlockSize != 3*1024*1024 {
+ t.Errorf("layout = %d+%d block %d, want 12+4 block 3MiB", got.DataShards, got.ParityShards, got.BlockSize)
+ }
+ })
+ }
+}
+
+// Absence stays legal — a volume encoded before either record exists is a
+// genuine legacy volume, and only that case may select the legacy layout.
+func TestNewEcVolumeConfigFreeVifWithNoSidecarUsesDefaults(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "3")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(3))
+ if err != nil {
+ t.Fatalf("mount: %v", err)
+ }
+ defer ev.Close()
+ if got := ev.ECContext; got.DataShards != DataShardsCount || got.ParityShards != ParityShardsCount || got.BlockSize != 0 {
+ t.Errorf("layout = %d+%d block %d, want the legacy defaults", got.DataShards, got.ParityShards, got.BlockSize)
+ }
+}
+
+// Present but unusable must fail the mount rather than default, in the
+// config-free-vif branch exactly as in the absent-vif one.
+func TestNewEcVolumeConfigFreeVifWithUnusableSidecarFails(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "4")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ // Stamped for another generation: not a record of these shards.
+ if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), &volume_server_pb.EcBitrotProtection{
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 5,
+ EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 12, ParityShards: 4},
+ }); err != nil {
+ t.Fatalf("seed .ecsum: %v", err)
+ }
+ if _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(4)); err == nil {
+ t.Fatal("an unusable sole layout record must fail the mount")
+ }
+}
+
+// Startup mirroring gives each shard-bearing disk its own .ecx/.ecj/.vif but
+// deliberately not the .ecsum, and a repair delivers exactly one copy. A
+// runtime restricted to its own two directories therefore reports no
+// protection no matter how often it reloads — so the resolution has to reach
+// the sibling disk that actually received the manifest.
+func TestReloadBitrotSidecarFindsItOnASiblingDisk(t *testing.T) {
+ mine, sibling := t.TempDir(), t.TempDir()
+ myBase := filepath.Join(mine, "7")
+ if err := os.WriteFile(myBase+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(myBase+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: 0,
+ },
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+
+ ev, err := NewEcVolume(types.HardDriveType, mine, mine, "", needle.VolumeId(7))
+ if err != nil {
+ t.Fatalf("mount: %v", err)
+ }
+ defer ev.Close()
+ if got := ev.bitrotStatus; got != BitrotOff {
+ t.Fatalf("precondition: status = %v, want BitrotOff before the delivery", got)
+ }
+
+ // The delivery lands the manifest on the sibling disk only.
+ prot := &volume_server_pb.EcBitrotProtection{
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 0,
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: 0,
+ },
+ }
+ for i := 0; i < 14; i++ {
+ prot.Shards = append(prot.Shards, &volume_server_pb.EcShardChecksums{
+ ShardId: uint32(i), CoveredSize: 1, BlockCrc32C: make([]byte, 4),
+ })
+ }
+ if err := SaveBitrotSidecar(BitrotSidecarPath(filepath.Join(sibling, "7"), 0), prot); err != nil {
+ t.Fatalf("seed sibling .ecsum: %v", err)
+ }
+
+ // Reloading against only its own directories still finds nothing.
+ ev.ReloadBitrotSidecar()
+ if got := ev.bitrotStatus; got != BitrotOff {
+ t.Errorf("own-directories reload: status = %v, want BitrotOff", got)
+ }
+
+ ev.ReloadBitrotSidecar(mine, sibling)
+ if got := ev.bitrotStatus; got != BitrotOn {
+ t.Errorf("cross-disk reload: status = %v, want BitrotOn", got)
+ }
+}
+
+// "Does this volume already have a manifest" cannot be answered from one base
+// name. The rebuild's opportunistic backfill asks it before writing a TOFU
+// baseline, and a false "no" there writes a second sidecar at the data base
+// that then shadows the real one — blessing whatever the shards currently say.
+func TestFindBitrotSidecarSearchesEveryCandidate(t *testing.T) {
+ dataDir, idxDir, sibling := t.TempDir(), t.TempDir(), t.TempDir()
+ dataBase := filepath.Join(dataDir, "8")
+ idxBase := filepath.Join(idxDir, "8")
+ siblingBase := filepath.Join(sibling, "8")
+
+ if got := FindBitrotSidecar(0, dataBase, idxBase, sibling); got != "" {
+ t.Errorf("no sidecar anywhere should answer empty, got %q", got)
+ }
+
+ // Each location on its own: any one of them is enough to answer "yes".
+ for _, tc := range []struct{ name, base string }{
+ {"data directory", dataBase},
+ {"index directory", idxBase},
+ {"sibling disk", siblingBase},
+ } {
+ want := BitrotSidecarPath(tc.base, 0)
+ if err := os.WriteFile(want, []byte("x"), 0644); err != nil {
+ t.Fatalf("%s: seed: %v", tc.name, err)
+ }
+ if got := FindBitrotSidecar(0, dataBase, idxBase, sibling); got != want {
+ t.Errorf("%s: got %q, want %q", tc.name, got, want)
+ }
+ if err := os.Remove(want); err != nil {
+ t.Fatalf("%s: cleanup: %v", tc.name, err)
+ }
+ }
+
+ // With copies everywhere the data base wins, which is the order every
+ // resolver uses — so a stray copy there shadows the others.
+ for _, base := range []string{siblingBase, idxBase, dataBase} {
+ if err := os.WriteFile(BitrotSidecarPath(base, 0), []byte("x"), 0644); err != nil {
+ t.Fatalf("seed %s: %v", base, err)
+ }
+ }
+ if got, want := FindBitrotSidecar(0, dataBase, idxBase, sibling), BitrotSidecarPath(dataBase, 0); got != want {
+ t.Errorf("precedence: got %q, want the data base %q", got, want)
+ }
+}
diff --git a/weed/storage/erasure_coding/ec_volume_vif_test.go b/weed/storage/erasure_coding/ec_volume_vif_test.go
new file mode 100644
index 000000000..e054d7612
--- /dev/null
+++ b/weed/storage/erasure_coding/ec_volume_vif_test.go
@@ -0,0 +1,187 @@
+package erasure_coding
+
+import (
+ "os"
+ "path/filepath"
+ "strings"
+ "testing"
+
+ "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
+ "github.com/seaweedfs/seaweedfs/weed/storage/needle"
+ "github.com/seaweedfs/seaweedfs/weed/storage/types"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
+)
+
+// A present-but-malformed .vif must FAIL the mount: every new encode
+// records a positive uniform block size there, and silently defaulting
+// to the legacy layout would serve those shards with the wrong offset
+// math. Absence stays legal — legacy volumes predate the sidecar.
+func TestNewEcVolumeFailsOnMalformedVif(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := os.WriteFile(base+".vif", []byte("not json"), 0644); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1))
+ if err == nil {
+ t.Fatal("mount over a malformed .vif must fail")
+ }
+ if !strings.Contains(err.Error(), ".vif") {
+ t.Errorf("error should name the vif: %v", err)
+ }
+}
+
+func TestNewEcVolumeAbsentVifMountsWithDefaults(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1))
+ if err != nil {
+ t.Fatalf("legacy mount without .vif must succeed: %v", err)
+ }
+ defer ev.Close()
+ if ev.ECContext.BlockSize != 0 {
+ t.Errorf("legacy mount must use the legacy layout: BlockSize=%d", ev.ECContext.BlockSize)
+ }
+}
+
+// A block size no encoder could have produced (negative, or not a whole
+// number of small blocks) maps every read to the wrong shard offset, so the
+// mount refuses it rather than serving those bytes.
+func TestNewEcVolumeRejectsImplausibleBlockSize(t *testing.T) {
+ for _, blockSize := range []int64{1, -1, 3*1024*1024 + 1} {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: blockSize,
+ },
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ if _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)); err == nil {
+ t.Errorf("block size %d must fail the mount", blockSize)
+ }
+ }
+}
+
+func TestNewEcVolumeAcceptsAlignedBlockSize(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: 3 * 1024 * 1024,
+ },
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1))
+ if err != nil {
+ t.Fatalf("aligned block size must mount: %v", err)
+ }
+ defer ev.Close()
+ if ev.ECContext.BlockSize != 3*1024*1024 {
+ t.Errorf("BlockSize = %d, want 3MiB", ev.ECContext.BlockSize)
+ }
+}
+
+// The .vif and the .ecsum both record the layout their generation was encoded
+// with. When a generation-matching sidecar disagrees, one of them is wrong and
+// reads through the other land at the wrong shard offsets — so the mount must
+// refuse rather than quietly drop to unprotected reads.
+func TestNewEcVolumeRejectsSidecarGeometryDisagreement(t *testing.T) {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("seed .ecx: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.Version3),
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: 0, // says legacy
+ },
+ }); err != nil {
+ t.Fatalf("seed .vif: %v", err)
+ }
+ // The sidecar for the same generation says uniform.
+ prot := &volume_server_pb.EcBitrotProtection{
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 0,
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: 3 * 1024 * 1024,
+ },
+ }
+ if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil {
+ t.Fatalf("seed .ecsum: %v", err)
+ }
+
+ _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1))
+ if err == nil {
+ t.Fatal("a generation-matching layout disagreement must fail the mount")
+ }
+ if !strings.Contains(err.Error(), "block") {
+ t.Errorf("the error should name the disagreement: %v", err)
+ }
+}
+
+// With no .vif the bitrot sidecar is the ONLY record of the volume's layout.
+// Present but unusable — unreadable, stamped for another generation, or
+// carrying an incomplete ratio — must fail the mount: answering reads from a
+// guessed layout returns wrong bytes rather than none. Genuine absence stays
+// legal, and is covered by TestNewEcVolumeAbsentVifMountsWithDefaults.
+func TestNewEcVolumeRejectsUnusableSoleSidecar(t *testing.T) {
+ seed := map[string]*volume_server_pb.EcBitrotProtection{
+ "another generation": {
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 5,
+ EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10, ParityShards: 4},
+ },
+ "no ec config": {
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 0,
+ },
+ "incomplete ratio": {
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 0,
+ EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10},
+ },
+ "invalid block size": {
+ Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C,
+ BlockSize: uint32(BitrotBlockSize),
+ Generation: 0,
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: 10, ParityShards: 4, BlockSize: 3*1024*1024 + 1,
+ },
+ },
+ }
+ for name, prot := range seed {
+ dir := t.TempDir()
+ base := filepath.Join(dir, "1")
+ if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil {
+ t.Fatalf("%s: seed .ecx: %v", name, err)
+ }
+ if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil {
+ t.Fatalf("%s: seed .ecsum: %v", name, err)
+ }
+ if _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)); err == nil {
+ t.Errorf("%s: an unusable sole layout record must fail the mount", name)
+ }
+ }
+}
diff --git a/weed/storage/erasure_coding/verification.go b/weed/storage/erasure_coding/verification.go
index ac6fa6ecb..76c52a2e1 100644
--- a/weed/storage/erasure_coding/verification.go
+++ b/weed/storage/erasure_coding/verification.go
@@ -14,6 +14,11 @@ import (
type ServerShardInventory struct {
Bits ShardBits
QueryError error
+ // BlockSize is the shard block layout this holder reports for the volume —
+ // the one it will serve reads through. A binary predating the uniform
+ // layout drops the field off the wire and reports 0, which is how
+ // RequireAgreedBlockLayout tells such a holder from one that agrees.
+ BlockSize int64
}
// Query errors are recorded per-server and treated as zero shards rather
@@ -50,6 +55,7 @@ func VerifyShardsAcrossServers(ctx context.Context, volumeID uint32,
}
inv.Bits = inv.Bits.Set(ShardId(s.ShardId))
}
+ inv.BlockSize = resp.GetEcShardConfig().GetBlockSize()
return nil
})
if callErr != nil {
@@ -99,6 +105,39 @@ func RequireRecoverableShardSet(volumeID uint32, shardsPresent ShardBits, dataSh
volumeID, totalShards-len(missing), totalShards, dataShards, missing)
}
+// RequireAgreedBlockLayout gates source-volume deletion on every holder that
+// answered agreeing with the layout the shards were encoded under.
+//
+// The uniform block layout lives in a `.vif` field that older volume servers
+// have never heard of: such a server parses the file, silently discards the
+// unknown field, and mounts the volume on the legacy 1GiB/1MiB striping. Its
+// reads then land at the wrong shard offsets and return wrong bytes, and
+// nothing else in the encode path notices — the shard files are the same
+// length under either layout. Asking each holder which layout it is serving is
+// the one question that separates the two, and asking it here means the answer
+// arrives while the source .dat is still on disk.
+//
+// Holders that could not be reached, or that report no shard of this volume,
+// are skipped: RequireRecoverableShardSet already covers a missing holder, and
+// one that answered nothing has promised nothing.
+func RequireAgreedBlockLayout(volumeID uint32, encodedBlockSize int64, perServer map[string]ServerShardInventory) error {
+ disagreeing := make([]string, 0, len(perServer))
+ for server, inv := range perServer {
+ if inv.QueryError != nil || inv.Bits.Count() == 0 {
+ continue
+ }
+ if inv.BlockSize != encodedBlockSize {
+ disagreeing = append(disagreeing, fmt.Sprintf("%s serves block size %d", server, inv.BlockSize))
+ }
+ }
+ if len(disagreeing) == 0 {
+ return nil
+ }
+ sort.Strings(disagreeing)
+ return fmt.Errorf("volume %d was encoded with shard block size %d but %v; upgrade every volume server to a build that understands the uniform shard block layout before encoding",
+ volumeID, encodedBlockSize, disagreeing)
+}
+
func SummarizeShardInventory(perServer map[string]ServerShardInventory) string {
servers := make([]string, 0, len(perServer))
for s := range perServer {
diff --git a/weed/storage/store_ec.go b/weed/storage/store_ec.go
index 0abd9fc43..9ab91b735 100644
--- a/weed/storage/store_ec.go
+++ b/weed/storage/store_ec.go
@@ -345,6 +345,49 @@ func (s *Store) FindEcVolumeWithShard(vid needle.VolumeId, shardId erasure_codin
return nil, nil, false
}
+// EcMetadataDirs lists every directory on this server that could hold an EC
+// volume's metadata — each disk's data and index directory. Startup mirroring
+// gives each shard-bearing disk its own .ecx/.ecj/.vif, but the checksum
+// sidecar is not mirrored and a repair delivers exactly one copy, so a runtime
+// looking only at its own two directories cannot see it. Handing this list to
+// the sidecar resolution keeps one authoritative copy reachable from every
+// runtime rather than duplicating a file that is rewritten as shards are
+// repaired and generations published.
+func (s *Store) EcMetadataDirs() []string {
+ var dirs []string
+ appendDir := func(dir string) {
+ if dir == "" {
+ return
+ }
+ for _, existing := range dirs {
+ if existing == dir {
+ return
+ }
+ }
+ dirs = append(dirs, dir)
+ }
+ for _, location := range s.Locations {
+ appendDir(location.Directory)
+ appendDir(location.IdxDirectory)
+ }
+ return dirs
+}
+
+// FindAllEcVolumes returns every per-disk *EcVolume the store maps for vid. A vid
+// can mount on N disks as N distinct runtimes, and the first-match FindEcVolume
+// hides the siblings — so anything that has to reach the whole volume, rather
+// than any one runtime of it, iterates this instead. Order mirrors the
+// deterministic s.Locations order; nil when no disk holds the vid.
+func (s *Store) FindAllEcVolumes(vid needle.VolumeId) []*erasure_coding.EcVolume {
+ var evs []*erasure_coding.EcVolume
+ for _, location := range s.Locations {
+ if ev, found := location.FindEcVolume(vid); found {
+ evs = append(evs, ev)
+ }
+ }
+ return evs
+}
+
func (s *Store) FindEcVolume(vid needle.VolumeId) (*erasure_coding.EcVolume, bool) {
for _, location := range s.Locations {
if s, found := location.FindEcVolume(vid); found {
@@ -436,10 +479,6 @@ func (s *Store) ReadEcShardNeedle(vid needle.VolumeId, n *needle.Needle, onReadS
return 0, fmt.Errorf("ec shard %d not found", vid)
}
-func (s *Store) IntervalToShardIdAndOffset(iv erasure_coding.Interval) (erasure_coding.ShardId, int64) {
- return iv.ToShardIdAndOffset(erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize)
-}
-
var (
// ecRecoverBudget bounds the interval-sized buffers EC recovery holds across
// all concurrent reads: a peer that is slow to fail keeps a whole fan-out of
@@ -507,7 +546,7 @@ func (s *Store) readEcShardIntervals(needleId types.NeedleId, ecVolume *erasure_
// readOneEcShardInterval fills data, which must be interval.Size long.
func (s *Store) readOneEcShardInterval(needleId types.NeedleId, ecVolume *erasure_coding.EcVolume, interval erasure_coding.Interval, data []byte) (is_deleted bool, err error) {
- shardId, actualOffset := s.IntervalToShardIdAndOffset(interval)
+ shardId, actualOffset := ecVolume.IntervalToShardIdAndOffset(interval)
// try local read
err = s.readLocalEcShardInterval(ecVolume, shardId, data, actualOffset)
diff --git a/weed/storage/store_ec_delete.go b/weed/storage/store_ec_delete.go
index da20ffd52..970967cb0 100644
--- a/weed/storage/store_ec_delete.go
+++ b/weed/storage/store_ec_delete.go
@@ -52,7 +52,7 @@ func (s *Store) doDeleteNeedleFromAtLeastOneRemoteEcShards(ecVolume *erasure_cod
return erasure_coding.NotFoundError
}
- primaryShardId, _ := intervals[0].ToShardIdAndOffset(erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize)
+ primaryShardId, _ := ecVolume.IntervalToShardIdAndOffset(intervals[0])
// Normal path: delete on exactly one node holding the primary data shard.
err = s.doDeleteNeedleFromRemoteEcShardServers(primaryShardId, ecVolume, needleId)
diff --git a/weed/storage/store_ec_interval_read_test.go b/weed/storage/store_ec_interval_read_test.go
index 8dba1bb03..dedb57a33 100644
--- a/weed/storage/store_ec_interval_read_test.go
+++ b/weed/storage/store_ec_interval_read_test.go
@@ -2,17 +2,112 @@ package storage
import (
"bytes"
+ "math/rand"
+ "os"
"testing"
+ "time"
+ "github.com/seaweedfs/seaweedfs/weed/pb"
+ "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
+ "github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
+ "github.com/seaweedfs/seaweedfs/weed/storage/super_block"
+ "github.com/seaweedfs/seaweedfs/weed/storage/types"
+ "github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
)
+// encodeAndMountEcVolume writes needles into a fresh volume, EC-encodes it
+// (uniform layout), stamps the .vif the way VolumeEcShardsGenerate does, and
+// mounts every shard on the store.
+func encodeAndMountEcVolume(t *testing.T, store *Store, vid needle.VolumeId, needles []*needle.Needle) *erasure_coding.EcVolume {
+ t.Helper()
+ dir := store.Locations[0].Directory
+
+ v, err := NewVolume(dir, dir, "", vid, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0)
+ if err != nil {
+ t.Fatalf("new volume: %v", err)
+ }
+ for _, n := range needles {
+ if _, _, _, err := v.writeNeedle2(n, true, false, false); err != nil {
+ t.Fatalf("write needle: %v", err)
+ }
+ }
+ baseFileName := v.DataFileName()
+ v.Close()
+
+ datSize, err := os.Stat(baseFileName + ".dat")
+ if err != nil {
+ t.Fatalf("stat .dat: %v", err)
+ }
+ ecCtx := erasure_coding.NewDefaultECContext("", vid)
+ if _, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx); err != nil {
+ t.Fatalf("write ec files: %v", err)
+ }
+ if err := erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"); err != nil {
+ t.Fatalf("write .ecx: %v", err)
+ }
+ if err := os.WriteFile(baseFileName+".ecj", nil, 0o644); err != nil {
+ t.Fatalf("write .ecj: %v", err)
+ }
+ if err := volume_info.SaveVolumeInfo(baseFileName+".vif", &volume_server_pb.VolumeInfo{
+ Version: uint32(needle.GetCurrentVersion()),
+ DatFileSize: datSize.Size(),
+ EcShardConfig: &volume_server_pb.EcShardConfig{
+ DataShards: uint32(ecCtx.DataShards),
+ ParityShards: uint32(ecCtx.ParityShards),
+ BlockSize: ecCtx.BlockSize,
+ },
+ }); err != nil {
+ t.Fatalf("save .vif: %v", err)
+ }
+ for _, ext := range []string{".dat", ".idx"} {
+ if err := os.Remove(baseFileName + ext); err != nil {
+ t.Fatalf("remove %s: %v", ext, err)
+ }
+ }
+
+ for shardId := 0; shardId < erasure_coding.TotalShardsCount; shardId++ {
+ if err := store.MountEcShards("", vid, erasure_coding.ShardId(shardId), ""); err != nil {
+ t.Fatalf("mount shard %d: %v", shardId, err)
+ }
+ }
+ ecVolume, found := store.Locations[0].FindEcVolume(vid)
+ if !found {
+ t.Fatal("ec volume not mounted")
+ }
+ if ecVolume.ECContext.BlockSize != ecCtx.BlockSize {
+ t.Fatalf("mounted BlockSize = %d, want %d", ecVolume.ECContext.BlockSize, ecCtx.BlockSize)
+ }
+
+ // Every shard is local, so seed the location cache to keep the read off the
+ // master this test does not have.
+ ecVolume.ShardLocationsLock.Lock()
+ for shardId := 0; shardId < erasure_coding.TotalShardsCount; shardId++ {
+ ecVolume.ShardLocations[erasure_coding.ShardId(shardId)] = []pb.ServerAddress{"localhost:8080"}
+ }
+ ecVolume.ShardLocationsRefreshTime = time.Now()
+ ecVolume.ShardLocationsLock.Unlock()
+ return ecVolume
+}
+
+func randomNeedleOfSize(id uint64, size int) *needle.Needle {
+ n := new(needle.Needle)
+ n.Id = types.Uint64ToNeedleId(id)
+ n.Data = make([]byte, size)
+ rand.New(rand.NewSource(int64(id))).Read(n.Data)
+ n.Checksum = needle.NewCRC(n.Data)
+ return n
+}
+
// A needle larger than one EC block is split over consecutive blocks, which
// live on different shards. The intervals are read concurrently, so check they
// still come back in order.
func TestReadEcShardNeedleSpanningBlocks(t *testing.T) {
+ store := newTestStore(t, 1)
const vid = needle.VolumeId(7)
- store, ecVolume, n := newLocalEcVolume(t, vid)
+
+ n := randomNeedleOfSize(42, 3*erasure_coding.ErasureCodingSmallBlockSize+1234)
+ ecVolume := encodeAndMountEcVolume(t, store, vid, []*needle.Needle{n})
_, _, intervals, err := ecVolume.LocateEcShardNeedle(n.Id, ecVolume.Version)
if err != nil {
@@ -31,3 +126,52 @@ func TestReadEcShardNeedleSpanningBlocks(t *testing.T) {
t.Fatalf("read back %d bytes, want the %d written", len(got.Data), len(n.Data))
}
}
+
+// On a volume big enough for multi-MB blocks, the uniform layout keeps a
+// needle smaller than the block in one interval on one shard, and reads every
+// needle back intact.
+func TestReadEcShardNeedleUniformLayout(t *testing.T) {
+ store := newTestStore(t, 1)
+ const vid = needle.VolumeId(8)
+
+ // ~26MB of needles: block size becomes 3MB, so the layouts diverge and a
+ // 1MB needle would stripe across shards under the legacy layout.
+ var needles []*needle.Needle
+ for i := uint64(1); i <= 5; i++ {
+ needles = append(needles, randomNeedleOfSize(i, 4*1024*1024))
+ }
+ for i := uint64(6); i <= 11; i++ {
+ needles = append(needles, randomNeedleOfSize(i, 1024*1024))
+ }
+ ecVolume := encodeAndMountEcVolume(t, store, vid, needles)
+ if ecVolume.ECContext.BlockSize <= erasure_coding.ErasureCodingSmallBlockSize {
+ t.Fatalf("block size %d does not diverge from the legacy layout", ecVolume.ECContext.BlockSize)
+ }
+
+ // With 3MB blocks a 1MB needle sits inside one block (bar a boundary
+ // straddle), where the legacy layout would stripe it at 1MB granularity.
+ singleInterval := false
+ for _, n := range needles[5:] {
+ _, _, intervals, err := ecVolume.LocateEcShardNeedle(n.Id, ecVolume.Version)
+ if err != nil {
+ t.Fatalf("locate needle %v: %v", n.Id, err)
+ }
+ if len(intervals) == 1 {
+ singleInterval = true
+ }
+ }
+ if !singleInterval {
+ t.Fatal("no 1MB needle mapped to a single interval; uniform layout not in effect")
+ }
+
+ for _, n := range needles {
+ got := new(needle.Needle)
+ got.Id = n.Id
+ if _, err := store.ReadEcShardNeedle(vid, got, nil); err != nil {
+ t.Fatalf("read needle %v: %v", n.Id, err)
+ }
+ if !bytes.Equal(got.Data, n.Data) {
+ t.Fatalf("needle %v: read back %d bytes that do not match the %d written", n.Id, len(got.Data), len(n.Data))
+ }
+ }
+}
diff --git a/weed/storage/store_ec_recover_fanout_test.go b/weed/storage/store_ec_recover_fanout_test.go
index c30a3f443..0bd75286e 100644
--- a/weed/storage/store_ec_recover_fanout_test.go
+++ b/weed/storage/store_ec_recover_fanout_test.go
@@ -95,7 +95,8 @@ func writeEcVolumeFiles(t *testing.T, dir string, vid needle.VolumeId) (baseFile
if err != nil {
t.Fatalf("stat .dat: %v", err)
}
- if _, err := erasure_coding.WriteEcFiles(baseFileName, erasure_coding.BackgroundECContext()); err != nil {
+ ecCtx := erasure_coding.BackgroundECContext()
+ if _, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx); err != nil {
t.Fatalf("write ec files: %v", err)
}
if err := erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"); err != nil {
@@ -110,6 +111,10 @@ func writeEcVolumeFiles(t *testing.T, dir string, vid needle.VolumeId) (baseFile
EcShardConfig: &volume_server_pb.EcShardConfig{
DataShards: erasure_coding.DataShardsCount,
ParityShards: erasure_coding.ParityShardsCount,
+ // The encode writes the uniform layout; a .vif that omitted its
+ // block size would mount these shards as legacy and read every
+ // interval at the wrong offset.
+ BlockSize: ecCtx.BlockSize,
},
}); err != nil {
t.Fatalf("save .vif: %v", err)
@@ -162,7 +167,7 @@ func TestRecoverOneRemoteEcShardIntervalUsesLocalShards(t *testing.T) {
if err != nil {
t.Fatalf("locate needle: %v", err)
}
- shardIdToRecover, actualOffset := store.IntervalToShardIdAndOffset(intervals[0])
+ shardIdToRecover, actualOffset := ecVolume.IntervalToShardIdAndOffset(intervals[0])
got := make([]byte, intervals[0].Size)
nRead, _, err := store.recoverOneRemoteEcShardInterval(n.Id, ecVolume, shardIdToRecover, got, actualOffset)
diff --git a/weed/storage/store_ec_scrub.go b/weed/storage/store_ec_scrub.go
index 08c002682..4a5c11169 100644
--- a/weed/storage/store_ec_scrub.go
+++ b/weed/storage/store_ec_scrub.go
@@ -47,14 +47,14 @@ func (s *Store) ScrubEcVolume(vid needle.VolumeId, mode volume_server_pb.VolumeS
shardIds := make([]erasure_coding.ShardId, len(intervals))
for i, iv := range intervals {
- sid, _ := s.IntervalToShardIdAndOffset(iv)
+ sid, _ := ecv.IntervalToShardIdAndOffset(iv)
shardIds[i] = sid
}
slices.Sort(shardIds)
for i, iv := range intervals {
chunk := make([]byte, iv.Size)
- shardId, offset := s.IntervalToShardIdAndOffset(iv)
+ shardId, offset := ecv.IntervalToShardIdAndOffset(iv)
// try a local shard read first...
if err := s.readLocalEcShardInterval(ecv, shardId, chunk, offset); err == nil {
diff --git a/weed/storage/store_ec_scrub_reads_test.go b/weed/storage/store_ec_scrub_reads_test.go
index d8a1758b3..f30ff853a 100644
--- a/weed/storage/store_ec_scrub_reads_test.go
+++ b/weed/storage/store_ec_scrub_reads_test.go
@@ -45,7 +45,8 @@ func newLocalEcVolume(t *testing.T, vid needle.VolumeId) (*Store, *erasure_codin
if err != nil {
t.Fatalf("stat .dat: %v", err)
}
- if _, err := erasure_coding.WriteEcFiles(baseFileName, erasure_coding.BackgroundECContext()); err != nil {
+ ecCtx := erasure_coding.BackgroundECContext()
+ if _, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx); err != nil {
t.Fatalf("write ec files: %v", err)
}
if err := erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"); err != nil {
@@ -60,6 +61,10 @@ func newLocalEcVolume(t *testing.T, vid needle.VolumeId) (*Store, *erasure_codin
EcShardConfig: &volume_server_pb.EcShardConfig{
DataShards: erasure_coding.DataShardsCount,
ParityShards: erasure_coding.ParityShardsCount,
+ // The encode writes the uniform layout; a .vif that omitted its
+ // block size would mount these shards as legacy and read every
+ // interval at the wrong offset.
+ BlockSize: ecCtx.BlockSize,
},
}); err != nil {
t.Fatalf("save .vif: %v", err)
diff --git a/weed/worker/tasks/erasure_coding/ec_task.go b/weed/worker/tasks/erasure_coding/ec_task.go
index ac8cbe787..b4ba0f126 100644
--- a/weed/worker/tasks/erasure_coding/ec_task.go
+++ b/weed/worker/tasks/erasure_coding/ec_task.go
@@ -51,6 +51,11 @@ type ErasureCodingTask struct {
// pre-distribute sweep. deleteOriginalVolume skips these so it does not
// re-delete and remove the now-EC .vif those servers share.
emptyReplicasDeleted map[string]bool
+
+ // encodedBlockSize is the shard block layout WriteEcFiles actually encoded
+ // with, read back off the EC context. Every holder must report serving the
+ // same one before the source volume may be deleted.
+ encodedBlockSize int64
}
// NewErasureCodingTask creates a new unified EC task instance
@@ -583,16 +588,23 @@ func (t *ErasureCodingTask) generateEcShardsLocally(localFiles map[string]string
}
// Generate EC shard files (.ec00 ~ .ec13)
- ecBitrot, err := erasure_coding.WriteEcFiles(baseName, erasure_coding.BackgroundECContext())
+ ecCtx := erasure_coding.BackgroundECContext()
+ ecBitrot, err := erasure_coding.WriteEcFiles(baseName, ecCtx)
if err != nil {
return nil, fmt.Errorf("failed to generate EC shard files: %w", err)
}
+ // The layout the shards were actually written in, for the holder agreement
+ // check before the source is deleted.
+ t.encodedBlockSize = ecCtx.BlockSize
// Persist the bitrot checksum sidecar (generation 0) alongside the shards so
- // it travels with them during distribution. Best-effort: a failed sidecar
- // write leaves the generation unprotected rather than failing the encode.
+ // it travels with them during distribution. Protection was asked for, and
+ // this write is orders of magnitude smaller than the shards that just
+ // landed: if it fails the disk is in trouble, and continuing would delete
+ // the source replicas in exchange for a generation that is both unprotected
+ // and missing the geometry record the .vif fallback reads.
if erasure_coding.BitrotProtectionEnabled && ecBitrot != nil {
if serr := erasure_coding.SaveBitrotSidecar(erasure_coding.BitrotSidecarPath(baseName, 0), ecBitrot); serr != nil {
- glog.Warningf("failed to write EC bitrot sidecar for %s: %v", baseName, serr)
+ return nil, fmt.Errorf("write EC bitrot sidecar for %s: %w", baseName, serr)
}
}
@@ -662,6 +674,7 @@ func (t *ErasureCodingTask) generateEcShardsLocally(localFiles map[string]string
if ecBitrot != nil && ecBitrot.EcShardConfig != nil {
ecShardConfig.DataShards = ecBitrot.EcShardConfig.DataShards
ecShardConfig.ParityShards = ecBitrot.EcShardConfig.ParityShards
+ ecShardConfig.BlockSize = ecBitrot.EcShardConfig.BlockSize
}
volumeInfo := &volume_server_pb.VolumeInfo{
Version: uint32(needle.GetCurrentVersion()),
@@ -675,30 +688,40 @@ func (t *ErasureCodingTask) generateEcShardsLocally(localFiles map[string]string
} else {
glog.Warningf("stat %s for .vif dat file size: %v", datFile, err)
}
+ // The .vif carries the shard block layout the holders will read through, so
+ // neither the write nor the inclusion below is optional: an encode that
+ // distributed shards without it would leave every reader falling back to
+ // the legacy layout, and the task deletes the source replicas afterwards.
if err := volume_info.SaveVolumeInfo(vifFile, volumeInfo); err != nil {
- glog.Warningf("Failed to create .vif file: %v", err)
- } else {
- shardFiles["vif"] = vifFile
- if info, err := os.Stat(vifFile); err == nil {
- t.GetLogger().WithFields(map[string]interface{}{
- "file_type": "vif",
- "file_path": vifFile,
- "size_bytes": info.Size(),
- }).Info("Volume info file generated")
- }
+ return nil, fmt.Errorf("write %s: %w", vifFile, err)
}
+ vifInfo, err := os.Stat(vifFile)
+ if err != nil {
+ return nil, fmt.Errorf("stat %s for distribution: %w", vifFile, err)
+ }
+ shardFiles["vif"] = vifFile
+ t.GetLogger().WithFields(map[string]interface{}{
+ "file_type": "vif",
+ "file_path": vifFile,
+ "size_bytes": vifInfo.Size(),
+ }).Info("Volume info file generated")
// Add the generation-0 bitrot checksum sidecar so it is distributed with
// the shards (DistributeEcShards only ships files present in shardFiles).
- // Best-effort like the sidecar write above: if it is absent the holders
- // are simply unprotected rather than failing the encode.
+ // Strict when protection is enabled and the encoder produced a manifest —
+ // the write above already failed the encode otherwise — so the holders
+ // cannot end up with shards whose checksums stayed behind on the worker.
ecsumFile := erasure_coding.BitrotSidecarPath(baseName, 0)
- if info, err := os.Stat(ecsumFile); err == nil {
+ ecsumInfo, ecsumErr := os.Stat(ecsumFile)
+ if ecsumErr != nil && erasure_coding.BitrotProtectionEnabled && ecBitrot != nil {
+ return nil, fmt.Errorf("stat %s for distribution: %w", ecsumFile, ecsumErr)
+ }
+ if ecsumErr == nil {
shardFiles["ecsum"] = ecsumFile
t.GetLogger().WithFields(map[string]interface{}{
"file_type": "ecsum",
"file_path": ecsumFile,
- "size_bytes": info.Size(),
+ "size_bytes": ecsumInfo.Size(),
}).Info("EC bitrot checksum sidecar generated")
}
@@ -768,6 +791,20 @@ func (t *ErasureCodingTask) verifyEcShardsBeforeDelete(ctx context.Context) erro
"per_server": summary,
}).Warning("EC shard set incomplete but recoverable; proceeding with source deletion")
}
+
+ // Before anything irreversible: every holder that answered must report
+ // serving the layout these shards were encoded in. A holder too old to
+ // know the uniform layout mounts them as legacy and returns wrong bytes,
+ // and the source volume is the only remaining correct copy.
+ if err := erasure_coding.RequireAgreedBlockLayout(t.volumeID, t.encodedBlockSize, perServer); err != nil {
+ t.GetLogger().WithFields(map[string]interface{}{
+ "volume_id": t.volumeID,
+ "per_server": summary,
+ "error": err.Error(),
+ }).Error("EC holders disagree on the shard block layout — source volume will be kept")
+ return err
+ }
+
return nil
}