diff --git a/seaweed-volume/proto/volume_server.proto b/seaweed-volume/proto/volume_server.proto index 6b64ef529..961a800fe 100644 --- a/seaweed-volume/proto/volume_server.proto +++ b/seaweed-volume/proto/volume_server.proto @@ -539,6 +539,7 @@ message VolumeEcShardsInfoResponse { uint64 volume_size = 2; uint64 file_count = 3; uint64 file_deleted_count = 4; + EcShardConfig ec_shard_config = 10; // the layout this holder serves reads through; a binary predating a field reports it as unset } message EcShardInfo { @@ -612,6 +613,7 @@ message EcShardConfig { uint32 data_shards = 1; // Number of data shards (e.g., 10) uint32 parity_shards = 2; // Number of parity shards (e.g., 4) int64 encode_ts_ns = 3; // encode time (unix nanos); a read served from a shard of a different encode run is rejected + int64 block_size = 4; // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout } // EcBitrotProtection is the entire content of a bitrot checksum sidecar // (.ecsum for the legacy generation, .ecsum.v for vacuum diff --git a/seaweed-volume/src/server/grpc_server.rs b/seaweed-volume/src/server/grpc_server.rs index e366413d5..dc956c414 100644 --- a/seaweed-volume/src/server/grpc_server.rs +++ b/seaweed-volume/src/server/grpc_server.rs @@ -92,6 +92,57 @@ pub fn load_state_file( volume_server_pb::VolumeServerState::decode(data.as_slice()).ok() } +/// One disk location's stake in an EC volume, as seen by the rebuild handler. +struct LocInfo { + dir: String, + idx_dir: String, + shard_count: usize, + has_ecx: bool, +} + +/// Picks the location a rebuild should write into — the one holding an `.ecx` +/// and the most shards — and returns every other directory it may have to read +/// from. Shards are only half of what the rebuild needs: a split +/// `-dir`/`-dir.idx` layout keeps `.ecx`/`.ecj`/`.vif` with the INDEX, and on a +/// multi-disk server the chosen disk may hold nothing but shards while this +/// volume's `.vif` or generation-0 `.ecsum` sits on a sibling. Miss those and +/// the layout resolution falls back to 10+4 with the legacy striping and +/// reconstructs through the wrong matrix, so both directories of every other +/// location are listed. The rebuild's own two are passed separately by the +/// caller and dropped here, along with empties and duplicates. +/// +/// Returns `None` when no location holds an `.ecx`, i.e. there is nothing to +/// rebuild from. +fn select_rebuild_location(loc_infos: &[LocInfo]) -> Option<(usize, Vec)> { + let mut rebuild_loc_idx: Option = None; + let mut other_dirs: Vec = Vec::new(); + + for (i, info) in loc_infos.iter().enumerate() { + let better = info.has_ecx + && rebuild_loc_idx + .is_none_or(|prev| info.shard_count > loc_infos[prev].shard_count); + if better { + if let Some(prev) = rebuild_loc_idx { + other_dirs.push(loc_infos[prev].dir.clone()); + other_dirs.push(loc_infos[prev].idx_dir.clone()); + } + rebuild_loc_idx = Some(i); + } else { + other_dirs.push(info.dir.clone()); + other_dirs.push(info.idx_dir.clone()); + } + } + + let rebuild_loc_idx = rebuild_loc_idx?; + let rebuild_dir = &loc_infos[rebuild_loc_idx].dir; + let rebuild_idx_dir = &loc_infos[rebuild_loc_idx].idx_dir; + other_dirs.retain(|d| !d.is_empty() && d != rebuild_dir && d != rebuild_idx_dir); + other_dirs.sort(); + other_dirs.dedup(); + + Some((rebuild_loc_idx, other_dirs)) +} + struct WriteThrottler { bytes_per_second: i64, last_size_counter: i64, @@ -2383,13 +2434,21 @@ impl VolumeServer for VolumeGrpcService { ) }; - // Check existing .vif for EC shard config (matching Go's MaybeLoadVolumeInfo) - let (data_shards, parity_shards) = + // Check existing .vif for EC shard config (matching Go's MaybeLoadVolumeInfo). + // The block size is recomputed by the encode for the current .dat, so + // only the ratio is carried over from a prior config. + let (data_shards, parity_shards, _) = crate::storage::erasure_coding::ec_volume::read_ec_shard_config( &dir, &idx_dir, collection, vid, - ); + ) + .map_err(|e| { + tonic::Status::internal(format!( + "read ec shard config for volume {}: {}", + vid.0, e + )) + })?; - if let Err(e) = crate::storage::erasure_coding::ec_encoder::write_ec_files( + let block_size = match crate::storage::erasure_coding::ec_encoder::write_ec_files( &dir, &idx_dir, collection, @@ -2397,16 +2456,19 @@ impl VolumeServer for VolumeGrpcService { data_shards as usize, parity_shards as usize, ) { - // Cleanup partially-created .ecNN and .ecx files on failure (matching Go defer) - let base = crate::storage::volume::volume_file_name(&dir, collection, vid); - let total_shards = data_shards + parity_shards; - for i in 0..total_shards { - let shard_path = format!("{}.ec{:02}", base, i); - let _ = std::fs::remove_file(&shard_path); + Ok(block_size) => block_size, + Err(e) => { + // Cleanup partially-created .ecNN and .ecx files on failure (matching Go defer) + let base = crate::storage::volume::volume_file_name(&dir, collection, vid); + let total_shards = data_shards + parity_shards; + for i in 0..total_shards { + let shard_path = format!("{}.ec{:02}", base, i); + let _ = std::fs::remove_file(&shard_path); + } + let _ = std::fs::remove_file(format!("{}.ecx", base)); + return Err(Status::internal(e.to_string())); } - let _ = std::fs::remove_file(format!("{}.ecx", base)); - return Err(Status::internal(e.to_string())); - } + }; // Write .vif file with EC shard metadata { @@ -2425,6 +2487,7 @@ impl VolumeServer for VolumeGrpcService { .duration_since(std::time::UNIX_EPOCH) .unwrap_or_default() .as_nanos() as i64, + block_size, }), ..Default::default() }; @@ -2458,13 +2521,6 @@ impl VolumeServer for VolumeGrpcService { format!("{}_{}", collection, vid.0) }; - struct LocInfo { - dir: String, - idx_dir: String, - shard_count: usize, - has_ecx: bool, - } - let store = self.state.store.read().unwrap(); let mut loc_infos: Vec = Vec::new(); @@ -2512,26 +2568,8 @@ impl VolumeServer for VolumeGrpcService { )); } - // Pick rebuild location: has .ecx and most shards - let mut rebuild_loc_idx: Option = None; - let mut other_dirs: Vec = Vec::new(); - - for (i, info) in loc_infos.iter().enumerate() { - if info.has_ecx - && (rebuild_loc_idx.is_none() - || info.shard_count > loc_infos[rebuild_loc_idx.unwrap()].shard_count) - { - if let Some(prev) = rebuild_loc_idx { - other_dirs.push(loc_infos[prev].dir.clone()); - } - rebuild_loc_idx = Some(i); - } else { - other_dirs.push(info.dir.clone()); - } - } - - let rebuild_loc_idx = match rebuild_loc_idx { - Some(i) => i, + let (rebuild_loc_idx, other_dirs) = match select_rebuild_location(&loc_infos) { + Some(picked) => picked, None => { return Ok(Response::new( volume_server_pb::VolumeEcShardsRebuildResponse { @@ -2545,13 +2583,37 @@ impl VolumeServer for VolumeGrpcService { let rebuild_idx_dir = loc_infos[rebuild_loc_idx].idx_dir.clone(); // Determine data/parity shard config from rebuild dir - let (data_shards, parity_shards) = - crate::storage::erasure_coding::ec_volume::read_ec_shard_config( + // The encode-time .dat size resolves the row count the ecx rebuild + // de-stripes with; 0 leaves it to infer from the padded shard extent. + // Both lookups search the sibling disks too: the rebuild writes into one + // location, but a multi-disk server may keep this volume's .vif or its + // generation-0 .ecsum on another, and defaulting to 10+4 with the + // legacy layout would reconstruct through the wrong matrix. + let dat_file_size = crate::storage::erasure_coding::ec_volume::load_vif_info_across_dirs( + &rebuild_dir, + &rebuild_idx_dir, + &other_dirs, + collection, + vid, + ) + .ok() + .flatten() + .map(|(v, _)| v.dat_file_size) + .unwrap_or(0); + let (data_shards, parity_shards, block_size) = + crate::storage::erasure_coding::ec_volume::read_ec_shard_config_across_dirs( &rebuild_dir, &rebuild_idx_dir, + &other_dirs, collection, vid, - ); + ) + .map_err(|e| { + tonic::Status::internal(format!( + "read ec shard config for volume {}: {}", + vid.0, e + )) + })?; let total_shards = data_shards + parity_shards; // Check which shards are missing (check rebuild dir and all other dirs) @@ -2590,8 +2652,17 @@ impl VolumeServer for VolumeGrpcService { // Rebuild missing shards, searching all locations for input shards. // Pass other_dirs so shards on sibling disks are found even when the - // primary rebuild dir doesn't hold them. - let other_dir_refs: Vec<&str> = other_dirs.iter().map(|s| s.as_str()).collect(); + // primary rebuild dir doesn't hold them. This one takes a single flat + // list — the shape Go's RebuildEcFiles uses — so unlike the resolvers + // above it cannot be handed the rebuild's own index directory + // separately, and a split -dir/-dir.idx location keeps its .ecx and + // .vif there. Go's additionalDirs carries that directory for the same + // reason. + let mut rebuild_search_dirs: Vec = other_dirs.clone(); + if !rebuild_idx_dir.is_empty() && rebuild_idx_dir != rebuild_dir { + rebuild_search_dirs.push(rebuild_idx_dir.clone()); + } + let other_dir_refs: Vec<&str> = rebuild_search_dirs.iter().map(|s| s.as_str()).collect(); crate::storage::erasure_coding::ec_encoder::rebuild_ec_files( &rebuild_dir, collection, @@ -2629,6 +2700,8 @@ impl VolumeServer for VolumeGrpcService { collection, vid, data_shards as usize, + block_size, + dat_file_size, &ecx_dir_refs, ) .map_err(|e| Status::internal(format!("RebuildEcxFile: {}", e)))?; @@ -3015,6 +3088,27 @@ impl VolumeServer for VolumeGrpcService { Status::internal(format!("mount {}.{}: {}", req.volume_id, shard_id, e)) })?; } + // A delivery can bring the checksum manifest alongside the shards, but + // the receive path only writes the file. When this server already had + // the volume mounted, the EcVolume in memory keeps whatever protection + // state it resolved at mount — off, for a volume whose sidecar arrives + // now — until a remount. Re-resolve it here, where the shards it + // describes have just been added. + // + // Every per-disk runtime, not just the first: a vid mounts as one + // EcVolume per disk, the delivery lands the .ecsum on one of them, and + // the first-match lookup would leave the siblings reporting no + // protection. Each re-resolves against its own data and index + // directories, so a shared -dir.idx reaches all of them. + // Resolving across every EC metadata directory is what makes that + // reload mean something: startup mirroring gives each shard-bearing + // disk its own .ecx/.ecj/.vif but deliberately not the sidecar, so a + // runtime restricted to its own two directories would find nothing + // however often it reloaded. One delivered copy, reachable from all. + let ec_metadata_dirs = store.ec_metadata_dirs(); + for ec_vol in store.find_all_ec_volumes_mut(vid) { + ec_vol.reload_bitrot_sidecar(&ec_metadata_dirs); + } drop(store); self.state.volume_state_notify.notify_one(); @@ -3349,6 +3443,8 @@ impl VolumeServer for VolumeGrpcService { let ecx_dir = ec_vol.ecx_actual_dir().to_string(); let collection = ec_vol.collection.clone(); let vif_dat_file_size = ec_vol.dat_file_size; + let (large_block_size, small_block_size) = + (ec_vol.large_block_size(), ec_vol.small_block_size()); // shard_dirs[i] is guaranteed Some for i in 0..data_shards by // the check above; collect concrete dirs for the decoder. let per_shard_dirs: Vec = shard_dirs[..data_shards] @@ -3382,6 +3478,8 @@ impl VolumeServer for VolumeGrpcService { vif_dat_file_size, data_shards, &per_shard_dirs, + large_block_size as usize, + small_block_size as usize, ) .map_err(|e| Status::internal(format!("WriteDatFile: {}", e)))?; @@ -3440,12 +3538,24 @@ impl VolumeServer for VolumeGrpcService { .walk_ecx_stats() .map_err(|e| Status::internal(e.to_string()))?; + // The layout this holder serves reads through, as Go reports it: a + // coordinator cannot otherwise tell a holder that understands the + // uniform block layout from one that dropped the unknown .vif field + // and mounted the volume as legacy. + let ec_shard_config = Some(volume_server_pb::EcShardConfig { + data_shards: ec_vol.data_shards, + parity_shards: ec_vol.parity_shards, + encode_ts_ns: 0, + block_size: ec_vol.block_size, + }); + Ok(Response::new( volume_server_pb::VolumeEcShardsInfoResponse { ec_shard_infos: shard_infos, volume_size, file_count, file_deleted_count, + ec_shard_config, }, )) } @@ -5086,6 +5196,110 @@ mod tests { use tempfile::TempDir; use tokio_stream::StreamExt; + fn loc(dir: &str, idx_dir: &str, shard_count: usize, has_ecx: bool) -> LocInfo { + LocInfo { + dir: dir.to_string(), + idx_dir: idx_dir.to_string(), + shard_count, + has_ecx, + } + } + + // The rebuild reads its shards from one directory but resolves the volume's + // layout -- ratio and uniform block size -- from the .vif or the + // generation-0 .ecsum, which on a multi-disk server may sit anywhere. Every + // directory that could hold one has to be in the search list, or the + // resolution silently falls back to 10+4 with the legacy striping and + // reconstructs through the wrong matrix. + // The rebuild's own data and index directories are handed to the resolvers + // as their own arguments, so they are deliberately absent from this list -- + // unlike Go, whose resolver takes a single directory list and therefore + // carries the rebuild's index directory inside it. + #[test] + fn select_rebuild_location_excludes_the_rebuilds_own_dirs() { + for infos in [ + vec![loc("/data1", "/idx1", 3, true)], + vec![loc("/data1", "/data1", 3, true)], + ] { + let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx"); + assert_eq!(idx, 0); + assert!(others.is_empty(), "got {:?}", others); + } + } + + // The case two reviewers flagged: a sibling holding only shards while its + // index directory holds this volume's .vif. + #[test] + fn select_rebuild_location_searches_a_siblings_index_dir_not_just_its_data_dir() { + let infos = vec![ + loc("/data1", "/data1", 5, true), + loc("/data2", "/idx2", 2, false), + ]; + let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx"); + assert_eq!(idx, 0); + assert_eq!(others, vec!["/data2".to_string(), "/idx2".to_string()]); + } + + // Several disks pointed at one index directory is a normal -dir.idx + // deployment; the shared directory is worth searching but only once. + #[test] + fn select_rebuild_location_lists_a_shared_index_dir_once() { + let infos = vec![ + loc("/data1", "/data1", 5, true), + loc("/data2", "/shared-idx", 2, false), + loc("/data3", "/shared-idx", 1, false), + ]; + let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx"); + assert_eq!( + others, + vec![ + "/data2".to_string(), + "/data3".to_string(), + "/shared-idx".to_string() + ] + ); + } + + // When the shared index directory is the rebuild's own it drops out, since + // the caller passes it separately. + #[test] + fn select_rebuild_location_omits_a_shared_index_dir_it_rebuilds_into() { + let infos = vec![ + loc("/data1", "/shared-idx", 5, true), + loc("/data2", "/shared-idx", 2, false), + ]; + let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx"); + assert_eq!(others, vec!["/data2".to_string()]); + } + + // The winner moves as a fuller location turns up; the one it displaces + // still has to be searched, index directory included. + #[test] + fn select_rebuild_location_keeps_the_displaced_winners_dirs() { + let infos = vec![ + loc("/data1", "/idx1", 2, true), + loc("/data2", "/idx2", 9, true), + ]; + let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx"); + assert_eq!(idx, 1, "the fuller location wins"); + assert_eq!(others, vec!["/data1".to_string(), "/idx1".to_string()]); + } + + #[test] + fn select_rebuild_location_drops_empty_dirs() { + let infos = vec![loc("/data1", "", 3, true), loc("/data2", "", 1, false)]; + let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx"); + assert_eq!(others, vec!["/data2".to_string()]); + } + + // Nothing carries an .ecx: there is no index to rebuild the shards against, + // so the caller answers with an empty rebuild rather than guessing. + #[test] + fn select_rebuild_location_is_none_without_an_ecx() { + let infos = vec![loc("/data1", "/idx1", 3, false)]; + assert!(select_rebuild_location(&infos).is_none()); + } + #[test] fn test_parse_grpc_address_with_explicit_grpc_port() { // Format: "ip:port.grpcPort" — used by SeaweedFS for source_data_node diff --git a/seaweed-volume/src/server/store_ec.rs b/seaweed-volume/src/server/store_ec.rs index 82277c500..1a8715295 100644 --- a/seaweed-volume/src/server/store_ec.rs +++ b/seaweed-volume/src/server/store_ec.rs @@ -599,7 +599,7 @@ fn read_local_intervals( ) -> Vec { let mut interval_results = Vec::with_capacity(intervals.len()); for interval in intervals { - let (shard_id, shard_offset) = interval.to_shard_id_and_offset(ecv.data_shards); + let (shard_id, shard_offset) = ecv.interval_to_shard_id_and_offset(interval); let buf_size = interval.size as usize; let local = ecv.shards.get(shard_id as usize).and_then(|s| s.as_ref()); match local { diff --git a/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs b/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs index e1eab27b5..1dc9b42ee 100644 --- a/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs +++ b/seaweed-volume/src/storage/erasure_coding/ec_bitrot.rs @@ -465,6 +465,28 @@ pub fn resolve_status( } } +/// Whether a generation-matching sidecar agrees with the geometry the volume is +/// mounted with. Both files record the layout the generation was encoded with, +/// so a disagreement means one of them is wrong and reads through the other +/// would land at the wrong shard offsets — the caller fails the mount rather +/// than merely dropping protection. A sidecar that records no EC config has +/// nothing to contradict. +pub fn geometry_matches( + prot: &EcBitrotProtection, + data_shards: usize, + parity_shards: usize, + block_size: i64, +) -> bool { + match &prot.ec_shard_config { + None => true, + Some(cfg) => { + cfg.data_shards as usize == data_shards + && cfg.parity_shards as usize == parity_shards + && cfg.block_size == block_size + } + } +} + /// Returns the [`EcShardChecksums`] entry for a shard id, or `None`. pub fn shard_checksums(prot: &EcBitrotProtection, shard_id: u32) -> Option<&EcShardChecksums> { prot.shards.iter().find(|s| s.shard_id == shard_id) @@ -539,11 +561,12 @@ fn read_full_at(f: &File, buf: &mut [u8], offset: u64) -> io::Result<()> { /// Builds the `EcShardConfig` proto for the given layout. The bitrot sidecar /// carries its own top-level encode_uuid, so the nested config leaves it empty. -pub fn ec_shard_config(data_shards: u32, parity_shards: u32) -> EcShardConfig { +pub fn ec_shard_config(data_shards: u32, parity_shards: u32, block_size: i64) -> EcShardConfig { EcShardConfig { data_shards, parity_shards, encode_ts_ns: 0, + block_size, } } @@ -567,6 +590,7 @@ mod tests { data_shards: 10, parity_shards: 4, encode_ts_ns: 0, + block_size: 0, }), shards: vec![ EcShardChecksums { @@ -712,7 +736,7 @@ mod tests { algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32, block_size: DEFAULT_BITROT_BLOCK_SIZE as u32, generation: 0, - ec_shard_config: Some(ec_shard_config(10, 4)), + ec_shard_config: Some(ec_shard_config(10, 4, 0)), shards: vec![EcShardChecksums { shard_id: 0, covered_size: covered, @@ -750,7 +774,7 @@ mod tests { algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32, block_size: DEFAULT_BITROT_BLOCK_SIZE as u32, generation: 0, - ec_shard_config: Some(ec_shard_config(10, 4)), + ec_shard_config: Some(ec_shard_config(10, 4, 0)), shards: vec![EcShardChecksums { shard_id: 0, covered_size: 5, @@ -794,7 +818,7 @@ mod tests { algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32, block_size: DEFAULT_BITROT_BLOCK_SIZE as u32, generation: 0, - ec_shard_config: Some(ec_shard_config(10, 4)), + ec_shard_config: Some(ec_shard_config(10, 4, 0)), shards, encode_uuid: vec![0u8; 16], } diff --git a/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs b/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs index 38a7de561..81db2d602 100644 --- a/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs +++ b/seaweed-volume/src/storage/erasure_coding/ec_decoder.rs @@ -80,6 +80,8 @@ pub fn write_dat_file_from_shards( dat_file_size: i64, encoded_dat_file_size: i64, data_shards: usize, + large_block_size: usize, + small_block_size: usize, ) -> io::Result<()> { let dirs: Vec = (0..data_shards).map(|_| dir.to_string()).collect(); write_dat_file_from_shards_with_dirs( @@ -90,6 +92,8 @@ pub fn write_dat_file_from_shards( encoded_dat_file_size, data_shards, &dirs, + large_block_size, + small_block_size, ) } @@ -113,7 +117,10 @@ pub fn write_dat_file_from_shards( /// boundary, and deriving the layout from the shrunk extent would read /// the shards in the wrong block order. Pass zero when the .vif does /// not record the encode-time size to infer the layout from the shard -/// size. +/// size. `large_block_size`/`small_block_size` are the volume's shard +/// block layout, e.g. `EcVolume::large_block_size()` / +/// `small_block_size()` from its .vif EC config. +#[allow(clippy::too_many_arguments)] pub fn write_dat_file_from_shards_with_dirs( dat_dir: &str, collection: &str, @@ -122,6 +129,8 @@ pub fn write_dat_file_from_shards_with_dirs( encoded_dat_file_size: i64, data_shards: usize, shard_dirs: &[String], + large_block_size: usize, + small_block_size: usize, ) -> io::Result<()> { write_dat_file( dat_dir, @@ -131,8 +140,8 @@ pub fn write_dat_file_from_shards_with_dirs( encoded_dat_file_size, data_shards, shard_dirs, - ERASURE_CODING_LARGE_BLOCK_SIZE, - ERASURE_CODING_SMALL_BLOCK_SIZE, + large_block_size, + small_block_size, ) } @@ -412,7 +421,9 @@ mod tests { // Encode to EC let data_shards = 10; let parity_shards = 4; - ec_encoder::write_ec_files(dir, dir, "", VolumeId(1), data_shards, parity_shards).unwrap(); + let block_size = + ec_encoder::write_ec_files(dir, dir, "", VolumeId(1), data_shards, parity_shards) + .unwrap(); // Delete original .dat and .idx std::fs::remove_file(format!("{}/1.dat", dir)).unwrap(); @@ -426,6 +437,8 @@ mod tests { original_dat_size as i64, original_dat_size as i64, data_shards, + block_size as usize, + block_size as usize, ) .unwrap(); write_idx_file_from_ec_index(dir, "", VolumeId(1)).unwrap(); @@ -472,7 +485,16 @@ mod tests { let dir = tmp.path().to_str().unwrap(); // No shard files exist, so de-striping must fail and publish nothing: // neither the final .dat nor a partial .dat.tmp may remain. - let res = write_dat_file_from_shards(dir, "", VolumeId(7), 100, 100, 10); + let res = write_dat_file_from_shards( + dir, + "", + VolumeId(7), + 100, + 100, + 10, + ERASURE_CODING_LARGE_BLOCK_SIZE, + ERASURE_CODING_SMALL_BLOCK_SIZE, + ); assert!(res.is_err()); assert!(!std::path::Path::new(&format!("{}/7.dat", dir)).exists()); assert!(!std::path::Path::new(&format!("{}/7.dat.tmp", dir)).exists()); @@ -525,6 +547,7 @@ mod tests { &mut builders, data_shards, parity_shards, + SMALL, LARGE, SMALL, ) @@ -616,6 +639,7 @@ mod tests { &mut builders, data_shards, parity_shards, + SMALL, LARGE, SMALL, ) diff --git a/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs b/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs index 346b2fd1f..1470e85de 100644 --- a/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs +++ b/seaweed-volume/src/storage/erasure_coding/ec_encoder.rs @@ -25,6 +25,10 @@ use crate::storage::volume::volume_file_name; /// /// Creates .ec00-.ec13 files in the same directory. /// Also creates a sorted .ecx index from the .idx file. +/// +/// Always encodes with the uniform block layout, sized for this .dat, and +/// returns the block size so the caller can persist it to .vif. Mirrors Go's +/// WriteEcFiles. pub fn write_ec_files( dir: &str, idx_dir: &str, @@ -32,7 +36,7 @@ pub fn write_ec_files( volume_id: VolumeId, data_shards: usize, parity_shards: usize, -) -> io::Result<()> { +) -> io::Result { let base = volume_file_name(dir, collection, volume_id); let dat_path = format!("{}.dat", base); let idx_base = volume_file_name(idx_dir, collection, volume_id); @@ -66,7 +70,7 @@ pub fn write_ec_files( .map(|_| ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64)) .collect(); - // Encode in large blocks, then small blocks + let block_size = uniform_block_size(dat_size, data_shards); encode_dat_file( &dat_file, dat_size, @@ -75,8 +79,9 @@ pub fn write_ec_files( &mut builders, data_shards, parity_shards, - ERASURE_CODING_LARGE_BLOCK_SIZE, - ERASURE_CODING_SMALL_BLOCK_SIZE, + ENCODE_BUFFER_SIZE, + block_size as usize, + block_size as usize, )?; // Close all shards @@ -103,6 +108,7 @@ pub fn write_ec_files( ec_shard_config: Some(ec_bitrot::ec_shard_config( data_shards as u32, parity_shards as u32, + block_size, )), shards: shard_checksums, encode_uuid: ec_bitrot::new_encode_uuid(), @@ -120,7 +126,19 @@ pub fn write_ec_files( ); } - Ok(()) + Ok(block_size) +} + +/// uniform_block_size returns the per-shard block size of the uniform layout +/// for a .dat of the given size: ceil(dat_file_size/data_shards) rounded up to +/// a whole small block. For every input this equals the legacy layout's padded +/// shard size, so only the byte placement differs between the two layouts, +/// never the shard length. Mirrors Go's UniformBlockSize. +pub fn uniform_block_size(dat_file_size: i64, data_shards: usize) -> i64 { + let small = ERASURE_CODING_SMALL_BLOCK_SIZE as i64; + let per_shard = (dat_file_size + data_shards as i64 - 1) / data_shards as i64; + let blocks = ((per_shard + small - 1) / small).max(1); + blocks * small } /// Rebuild missing EC shard files from existing shards using Reed-Solomon reconstruct. @@ -370,7 +388,7 @@ pub fn verify_ec_shards( } /// Write sorted .ecx index from .idx file. -fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::Result<()> { +pub(crate) fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::Result<()> { if !std::path::Path::new(idx_path).exists() { return Err(io::Error::new( io::ErrorKind::NotFound, @@ -421,6 +439,8 @@ pub fn rebuild_ecx_file( collection: &str, volume_id: VolumeId, data_shards: usize, + block_size: i64, + dat_file_size: i64, additional_dirs: &[&str], ) -> io::Result<()> { use crate::storage::needle::needle::get_actual_size; @@ -463,10 +483,42 @@ pub fn rebuild_ecx_file( // Determine total logical data size from shard sizes let shard_size = shards.iter().map(|s| s.file_size()).max().unwrap_or(0); let total_data_size = shard_size as i64 * data_shards as i64; + // The volume's shard block layout: the .vif-recorded uniform block size, + // or the legacy two-tier sizes when 0. The row count comes from the shard + // length; -1 disambiguates a legacy shard that is an exact large-block + // multiple (mirrors the ecdFileSize-1 fallback in the read path). + let (large_block, small_block) = if block_size > 0 { + (block_size, block_size) + } else { + ( + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ) + }; + // The row count the de-stripe walks with. The encode-time .dat size is the + // authority — the same value the read path divides by data_shards — and the + // padded extent is only a fallback: under the legacy layout a shard that is + // an exact large-block multiple reads as one row too many, which + // re-interprets its last large row as small blocks and scrambles the + // recovered offsets. Subtracting one keeps that fallback on the safe side of + // the boundary, exactly as the read path's own fallback does. + let locate_shard_size = if dat_file_size > 0 { + dat_file_size / data_shards as i64 + } else { + (shard_size as i64 - 1).max(0) + }; // Read version from superblock (first byte of logical data) let mut sb_buf = [0u8; SUPER_BLOCK_SIZE]; - read_from_data_shards(&shards, &mut sb_buf, 0, data_shards)?; + read_from_data_shards( + &shards, + &mut sb_buf, + 0, + data_shards, + locate_shard_size, + large_block, + small_block, + )?; let version = Version(sb_buf[0]); // Walk needles starting after superblock @@ -475,10 +527,30 @@ pub fn rebuild_ecx_file( let mut entries: Vec<(NeedleId, Offset, Size)> = Vec::new(); while offset + header_size as i64 <= total_data_size { - // Read needle header (cookie + needle_id + size = 16 bytes) + // Read needle header (cookie + needle_id + size = 16 bytes). + // A read failure is NOT the end of the data — every offset in + // range maps into the shards, so an error means a truncated or + // unreadable shard. Publishing the entries collected so far as + // a successful .ecx would hand out a silently incomplete + // recovery index; propagate instead. (The scan still ends + // normally on the zero-cookie tail below.) let mut header_buf = [0u8; NEEDLE_HEADER_SIZE]; - if read_from_data_shards(&shards, &mut header_buf, offset as u64, data_shards).is_err() { - break; + if let Err(e) = read_from_data_shards( + &shards, + &mut header_buf, + offset as u64, + data_shards, + locate_shard_size, + large_block, + small_block, + ) { + for s in &mut shards { + s.close(); + } + return Err(io::Error::new( + e.kind(), + format!("scan needle header at offset {}: {}", offset, e), + )); } let cookie = Cookie::from_bytes(&header_buf[..COOKIE_SIZE]); @@ -532,58 +604,83 @@ pub fn rebuild_ecx_file( Ok(()) } -/// Read bytes from EC data shards at a logical offset in the .dat file. +/// Read bytes from EC data shards at a logical offset in the .dat file, +/// resolving the shard/offset through the volume's block layout via +/// locate_data — the same mapping the read path uses. +#[allow(clippy::too_many_arguments)] fn read_from_data_shards( shards: &[EcVolumeShard], buf: &mut [u8], logical_offset: u64, data_shards: usize, + locate_shard_size: i64, + large_block_size: i64, + small_block_size: i64, ) -> io::Result<()> { - let small_block = ERASURE_CODING_SMALL_BLOCK_SIZE as u64; - let data_shards_u64 = data_shards as u64; - - let mut bytes_read = 0u64; - let mut remaining = buf.len() as u64; - let mut current_offset = logical_offset; - - while remaining > 0 { - // Determine which shard and at what shard-offset this logical offset maps to. - // The data is interleaved: large blocks first, then small blocks. - // For simplicity, use the small block size for all calculations since - // large blocks are multiples of small blocks. - let row_size = small_block * data_shards_u64; - let row_index = current_offset / row_size; - let row_offset = current_offset % row_size; - let shard_index = (row_offset / small_block) as usize; - let shard_offset = row_index * small_block + (row_offset % small_block); - - if shard_index >= data_shards { + let intervals = crate::storage::erasure_coding::ec_locate::locate_data( + logical_offset as i64, + Size(buf.len() as i32), + locate_shard_size, + data_shards as u32, + large_block_size, + small_block_size, + ); + let mut bytes_read = 0usize; + for interval in &intervals { + let (shard_id, shard_offset) = + interval.to_shard_id_and_offset(data_shards as u32, large_block_size, small_block_size); + if shard_id as usize >= data_shards { return Err(io::Error::new( io::ErrorKind::InvalidInput, "shard index out of range", )); } - - // How many bytes can we read from this position in this shard block - let bytes_left_in_block = small_block - (row_offset % small_block); - let to_read = remaining.min(bytes_left_in_block) as usize; - - let dest = &mut buf[bytes_read as usize..bytes_read as usize + to_read]; - shards[shard_index].read_at(dest, shard_offset)?; - - bytes_read += to_read as u64; - remaining -= to_read as u64; - current_offset += to_read as u64; + let to_read = interval.size as usize; + let dest = &mut buf[bytes_read..bytes_read + to_read]; + // Exact-read semantics: read_at may legally return fewer bytes + // than requested, and treating a short read as complete leaves + // the tail of `dest` as whatever the buffer held before. Loop + // until filled; zero bytes inside the mapped range means the + // shard is truncated — an error, not an end. + let mut filled = 0usize; + while filled < to_read { + let n = shards[shard_id as usize] + .read_at(&mut dest[filled..], shard_offset as u64 + filled as u64)?; + if n == 0 { + return Err(io::Error::new( + io::ErrorKind::UnexpectedEof, + format!( + "short read from data shard {}: {} of {} bytes at offset {}", + shard_id, filled, to_read, shard_offset + ), + )); + } + filled += n; + } + bytes_read += to_read; + } + if bytes_read != buf.len() { + return Err(io::Error::new( + io::ErrorKind::UnexpectedEof, + "short read from data shards", + )); } - Ok(()) } +/// Buffer size for one encode sub-batch per shard, mirroring Go's 256KB +/// bufferSize in WriteEcFiles. A block is processed in block_size/buffer_size +/// sub-batches, so memory stays at total_shards * 256KB no matter how large +/// the uniform block is. +const ENCODE_BUFFER_SIZE: usize = 256 * 1024; + /// Encode the .dat file data into shard files. /// /// Uses a two-phase approach matching Go's ec_encoder.go: -/// 1. Process as many large blocks (1GB) as possible -/// 2. Process remaining data with small blocks (1MB) +/// 1. Process as many large blocks as possible +/// 2. Process remaining data with small blocks +/// +/// `buffer_size` must divide both block sizes. #[allow(clippy::too_many_arguments)] pub(crate) fn encode_dat_file( dat_file: &File, @@ -593,44 +690,50 @@ pub(crate) fn encode_dat_file( builders: &mut [ShardChecksumBuilder], data_shards: usize, parity_shards: usize, + buffer_size: usize, large_block_size: usize, small_block_size: usize, ) -> io::Result<()> { + let total_shards = data_shards + parity_shards; + let mut buffers: Vec> = (0..total_shards) + .map(|_| vec![0u8; buffer_size]) + .collect(); + let mut remaining = dat_size; let mut offset: u64 = 0; - // Phase 1: Process large blocks (1GB each) while enough data remains + // Phase 1: process whole large-block rows while enough data remains let large_row_size = large_block_size * data_shards; while remaining >= large_row_size as i64 { - encode_one_batch( + encode_data( dat_file, offset, large_block_size, rs, + &mut buffers, shards, builders, data_shards, - parity_shards, )?; offset += large_row_size as u64; remaining -= large_row_size as i64; } - // Phase 2: Process remaining data with small blocks (1MB each) + // Phase 2: process remaining data with small blocks let small_row_size = small_block_size * data_shards; while remaining > 0 { let to_process = remaining.min(small_row_size as i64); - encode_one_batch( + encode_data( dat_file, offset, small_block_size, rs, + &mut buffers, shards, builders, data_shards, - parity_shards, )?; offset += to_process as u64; remaining -= to_process; @@ -639,61 +742,71 @@ pub(crate) fn encode_dat_file( Ok(()) } -/// Encode one batch (row) of data. +/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches so +/// arbitrarily large blocks never require block-sized allocations. Mirrors +/// Go's encodeData. +#[allow(clippy::too_many_arguments)] +fn encode_data( + dat_file: &File, + row_offset: u64, + block_size: usize, + rs: &ReedSolomon, + buffers: &mut [Vec], + shards: &mut [EcVolumeShard], + builders: &mut [ShardChecksumBuilder], + data_shards: usize, +) -> io::Result<()> { + let buffer_size = buffers[0].len(); + if block_size % buffer_size != 0 { + return Err(io::Error::new( + io::ErrorKind::InvalidInput, + format!( + "unexpected block size {} buffer size {}", + block_size, buffer_size + ), + )); + } + let batch_count = block_size / buffer_size; + for b in 0..batch_count { + encode_one_batch( + dat_file, + row_offset + (b * buffer_size) as u64, + block_size, + rs, + buffers, + shards, + builders, + data_shards, + )?; + } + Ok(()) +} + +/// Encode one sub-batch: the same buffer-sized slice of every shard's block in +/// this row. Mirrors Go's encodeDataOneBatch. +#[allow(clippy::too_many_arguments)] fn encode_one_batch( dat_file: &File, offset: u64, block_size: usize, rs: &ReedSolomon, + buffers: &mut [Vec], shards: &mut [EcVolumeShard], builders: &mut [ShardChecksumBuilder], data_shards: usize, - parity_shards: usize, ) -> io::Result<()> { - let total_shards = data_shards + parity_shards; - // Each batch allocates block_size * total_shards bytes. - // With large blocks (1 GiB) this is 14 GiB -- guard against OOM. - let total_alloc = block_size.checked_mul(total_shards).ok_or_else(|| { - io::Error::new( - io::ErrorKind::InvalidInput, - "block_size * shard count overflows usize", - ) - })?; - // Large-block encoding uses 1 GiB * 14 shards = 14 GiB; allow up to 16 GiB. - const MAX_BATCH_ALLOC: usize = 16 * 1024 * 1024 * 1024; // 16 GiB safety limit - if total_alloc > MAX_BATCH_ALLOC { - return Err(io::Error::new( - io::ErrorKind::InvalidInput, - format!( - "batch allocation too large ({} bytes, limit {} bytes); block_size={} shards={}", - total_alloc, MAX_BATCH_ALLOC, block_size, total_shards, - ), - )); - } - - // Allocate buffers for all shards - let mut buffers: Vec> = (0..total_shards).map(|_| vec![0u8; block_size]).collect(); - - // Read data shards from .dat file + // Read data shards from the .dat file, zero-filling past EOF — the buffers + // are reused across batches, so the tail must be cleared explicitly. for i in 0..data_shards { let read_offset = offset + (i * block_size) as u64; - - #[cfg(unix)] - { - use std::os::unix::fs::FileExt; - dat_file.read_at(&mut buffers[i], read_offset)?; - } - - #[cfg(not(unix))] - { - let mut f = dat_file.try_clone()?; - f.seek(SeekFrom::Start(read_offset))?; - f.read(&mut buffers[i])?; + let n = read_at_most(dat_file, &mut buffers[i], read_offset)?; + for b in buffers[i][n..].iter_mut() { + *b = 0; } } // Encode parity shards - rs.encode(&mut buffers).map_err(|e| { + rs.encode(&mut *buffers).map_err(|e| { io::Error::new( io::ErrorKind::Other, format!("reed-solomon encode: {:?}", e), @@ -710,6 +823,29 @@ fn encode_one_batch( Ok(()) } +/// Read into `buf` at `offset` until it is full or EOF; returns bytes read. +fn read_at_most(dat_file: &File, buf: &mut [u8], offset: u64) -> io::Result { + let mut n = 0; + while n < buf.len() { + #[cfg(unix)] + let r = { + use std::os::unix::fs::FileExt; + dat_file.read_at(&mut buf[n..], offset + n as u64)? + }; + #[cfg(not(unix))] + let r = { + let mut f = dat_file.try_clone()?; + f.seek(SeekFrom::Start(offset + n as u64))?; + f.read(&mut buf[n..])? + }; + if r == 0 { + break; + } + n += r; + } + Ok(n) +} + #[cfg(test)] mod tests { use super::*; @@ -1020,7 +1156,7 @@ mod tests { // Without additional_dirs, rebuild must fail: shards 1, 3, 6 are not // in primary and the full logical .dat content can't be reconstructed. - let res = rebuild_ecx_file(&primary, "", VolumeId(1), 10, &[]); + let res = rebuild_ecx_file(&primary, "", VolumeId(1), 10, 0, 0, &[]); assert!( res.is_err(), "ecx rebuild without additional_dirs must fail when data shards are on another disk" @@ -1031,7 +1167,7 @@ mod tests { ); // With additional_dirs pointing at the secondary, the rebuild must succeed. - rebuild_ecx_file(&primary, "", VolumeId(1), 10, &[secondary.as_str()]).unwrap(); + rebuild_ecx_file(&primary, "", VolumeId(1), 10, 0, 0, &[secondary.as_str()]).unwrap(); assert!( std::path::Path::new(&ecx_path).exists(), @@ -1043,6 +1179,118 @@ mod tests { ); } + // A uniform-layout volume (block size > 1MiB) must have its .ecx rebuilt + // through the recorded geometry; the legacy 1MiB mapping would scan + // garbage past the first block boundary. + #[test] + fn test_rebuild_ecx_file_uniform_layout() { + use crate::storage::needle_map::NeedleMapKind; + use crate::storage::volume::Volume; + let tmp = TempDir::new().unwrap(); + let dir = tmp.path().to_str().unwrap().to_string(); + let mut v = Volume::new( + &dir, + &dir, + "", + VolumeId(2), + NeedleMapKind::InMemory, + None, + None, + 0, + Version::current(), + ) + .unwrap(); + for i in 1u64..=12 { + let data: Vec = (0..2 << 20) + .map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8) + .collect(); + let mut n = Needle { + id: NeedleId(i), + cookie: Cookie(i as u32), + data: data.clone(), + data_size: data.len() as u32, + ..Needle::default() + }; + v.write_needle(&mut n, true, false).unwrap(); + } + v.sync_to_disk().unwrap(); + v.close(); + let block_size = write_ec_files(&dir, &dir, "", VolumeId(2), 10, 4).unwrap(); + assert!( + block_size > ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + "fixture must diverge from the legacy layout" + ); + + let ecx_path = format!("{}/2.ecx", dir); + let canonical = std::fs::read(&ecx_path).unwrap(); + std::fs::remove_file(&ecx_path).unwrap(); + + rebuild_ecx_file(&dir, "", VolumeId(2), 10, block_size, 0, &[]).unwrap(); + let rebuilt = std::fs::read(&ecx_path).unwrap(); + assert_eq!(canonical, rebuilt, "rebuilt .ecx must match the encode-time .ecx"); + } + + // A truncated data shard must FAIL the .ecx rebuild, not publish the + // entries scanned so far as a successful (silently incomplete) index. + #[test] + fn test_rebuild_ecx_file_fails_on_truncated_shard() { + use crate::storage::needle_map::NeedleMapKind; + use crate::storage::volume::Volume; + let tmp = TempDir::new().unwrap(); + let dir = tmp.path().to_str().unwrap().to_string(); + let mut v = Volume::new( + &dir, + &dir, + "", + VolumeId(3), + NeedleMapKind::InMemory, + None, + None, + 0, + Version::current(), + ) + .unwrap(); + for i in 1u64..=12 { + let data: Vec = (0..2 << 20) + .map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8) + .collect(); + let mut n = Needle { + id: NeedleId(i), + cookie: Cookie(i as u32), + data: data.clone(), + data_size: data.len() as u32, + ..Needle::default() + }; + v.write_needle(&mut n, true, false).unwrap(); + } + v.sync_to_disk().unwrap(); + v.close(); + let block_size = write_ec_files(&dir, &dir, "", VolumeId(3), 10, 4).unwrap(); + + let ecx_path = format!("{}/3.ecx", dir); + std::fs::remove_file(&ecx_path).unwrap(); + + // Truncate shard 0 to just the superblock: the scan's very first + // needle-header read (offset SUPER_BLOCK_SIZE, shard 0 under the + // uniform layout) lands in the missing region. The pre-fix code + // broke the scan there and published an EMPTY .ecx as success. + let shard_path = format!("{}/3.ec00", dir); + let f = std::fs::OpenOptions::new() + .write(true) + .open(&shard_path) + .unwrap(); + f.set_len(crate::storage::super_block::SUPER_BLOCK_SIZE as u64) + .unwrap(); + drop(f); + + let res = rebuild_ecx_file(&dir, "", VolumeId(3), 10, block_size, 0, &[]); + assert!(res.is_err(), "rebuild over a truncated shard must fail"); + assert!( + !std::path::Path::new(&ecx_path).exists(), + "a failed rebuild must not leave a partial .ecx behind" + ); + } + #[test] fn test_reed_solomon_basic() { let data_shards = 10; diff --git a/seaweed-volume/src/storage/erasure_coding/ec_locate.rs b/seaweed-volume/src/storage/erasure_coding/ec_locate.rs index 4c1f06aa2..baaed9f3b 100644 --- a/seaweed-volume/src/storage/erasure_coding/ec_locate.rs +++ b/seaweed-volume/src/storage/erasure_coding/ec_locate.rs @@ -18,21 +18,26 @@ pub struct Interval { } impl Interval { - pub fn to_shard_id_and_offset(&self, data_shards: u32) -> (ShardId, i64) { + pub fn to_shard_id_and_offset( + &self, + data_shards: u32, + large_block_size: i64, + small_block_size: i64, + ) -> (ShardId, i64) { let data_shards_usize = data_shards as usize; let shard_id = (self.block_index % data_shards_usize) as ShardId; let row_index = self.block_index / data_shards_usize; let block_size = if self.is_large_block { - ERASURE_CODING_LARGE_BLOCK_SIZE as i64 + large_block_size } else { - ERASURE_CODING_SMALL_BLOCK_SIZE as i64 + small_block_size }; let mut offset = row_index as i64 * block_size + self.inner_block_offset; if !self.is_large_block { // Small blocks come after large blocks in the shard file - offset += self.large_block_rows_count as i64 * ERASURE_CODING_LARGE_BLOCK_SIZE as i64; + offset += self.large_block_rows_count as i64 * large_block_size; } (shard_id, offset) @@ -42,7 +47,14 @@ impl Interval { /// Locate the EC shard intervals needed to read data at the given offset and size. /// /// `shard_size` is the size of a single shard file. -pub fn locate_data(offset: i64, size: Size, shard_size: i64, data_shards: u32) -> Vec { +pub fn locate_data( + offset: i64, + size: Size, + shard_size: i64, + data_shards: u32, + large_block_size: i64, + small_block_size: i64, +) -> Vec { let mut intervals = Vec::new(); let data_size = size.0 as i64; @@ -50,17 +62,14 @@ pub fn locate_data(offset: i64, size: Size, shard_size: i64, data_shards: u32) - return intervals; } - let large_block_size = ERASURE_CODING_LARGE_BLOCK_SIZE as i64; - let small_block_size = ERASURE_CODING_SMALL_BLOCK_SIZE as i64; let large_row_size = large_block_size * data_shards as i64; let small_row_size = small_block_size * data_shards as i64; - // Number of large block rows - let n_large_block_rows = if shard_size > 0 { - ((shard_size - 1) / large_block_size) as usize - } else { - 0 - }; + // Number of large block rows. Mirrors Go's shardDatSize/largeBlockLength: + // the caller's ecd-size fallback already subtracts 1 to disambiguate the + // exact-multiple case, so no further -1 here — a shard size that IS an + // exact multiple (dat_file_size path) means real full large rows. + let n_large_block_rows = (shard_size / large_block_size) as usize; let large_section_size = n_large_block_rows as i64 * large_row_size; let mut remaining_offset = offset; @@ -150,7 +159,11 @@ mod tests { is_large_block: true, large_block_rows_count: 1, }; - let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards); + let (shard_id, offset) = interval.to_shard_id_and_offset( + data_shards, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); assert_eq!(shard_id, 0); assert_eq!(offset, 100); @@ -162,7 +175,11 @@ mod tests { is_large_block: true, large_block_rows_count: 1, }; - let (shard_id, _offset) = interval.to_shard_id_and_offset(data_shards); + let (shard_id, _offset) = interval.to_shard_id_and_offset( + data_shards, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); assert_eq!(shard_id, 5); // Block index 12 (data_shards=10) → row_index 1, shard_id 2 @@ -173,7 +190,11 @@ mod tests { is_large_block: true, large_block_rows_count: 5, }; - let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards); + let (shard_id, offset) = interval.to_shard_id_and_offset( + data_shards, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); assert_eq!(shard_id, 2); // 12 % 10 = 2 assert_eq!(offset, large_block_size + 200); // row 1 offset + inner_block_offset @@ -185,7 +206,11 @@ mod tests { is_large_block: true, large_block_rows_count: 2, }; - let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards); + let (shard_id, offset) = interval.to_shard_id_and_offset( + data_shards, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); assert_eq!(shard_id, 0); assert_eq!(offset, ERASURE_CODING_LARGE_BLOCK_SIZE as i64); // row 1 offset } @@ -193,7 +218,14 @@ mod tests { #[test] fn test_locate_data_small_file() { // Small file: 100 bytes at offset 50, shard size = 1MB - let intervals = locate_data(50, Size(100), 1024 * 1024, 10); + let intervals = locate_data( + 50, + Size(100), + 1024 * 1024, + 10, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); assert!(!intervals.is_empty()); // Should be a single small block interval (no large block rows for 1MB shard) @@ -203,7 +235,14 @@ mod tests { #[test] fn test_locate_data_empty() { - let intervals = locate_data(0, Size(0), 1024 * 1024, 10); + let intervals = locate_data( + 0, + Size(0), + 1024 * 1024, + 10, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); assert!(intervals.is_empty()); } @@ -216,7 +255,11 @@ mod tests { is_large_block: false, large_block_rows_count: 2, }; - let (_shard_id, offset) = interval.to_shard_id_and_offset(10); + let (_shard_id, offset) = interval.to_shard_id_and_offset( + 10, + ERASURE_CODING_LARGE_BLOCK_SIZE as i64, + ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + ); // Should be after 2 large block rows assert_eq!(offset, 2 * ERASURE_CODING_LARGE_BLOCK_SIZE as i64); } diff --git a/seaweed-volume/src/storage/erasure_coding/ec_volume.rs b/seaweed-volume/src/storage/erasure_coding/ec_volume.rs index 709d8f485..34c12aaa9 100644 --- a/seaweed-volume/src/storage/erasure_coding/ec_volume.rs +++ b/seaweed-volume/src/storage/erasure_coding/ec_volume.rs @@ -26,6 +26,9 @@ pub struct EcVolume { pub dat_file_size: i64, pub data_shards: u32, pub parity_shards: u32, + /// Uniform block layout: each shard is one contiguous block of this many + /// bytes. 0 = legacy 1GiB/1MiB two-tier layout. Loaded from .vif. + pub block_size: i64, ecx_file: Option, ecx_file_size: i64, ecj_file: Option, @@ -101,34 +104,249 @@ fn locate_vif_path(dir: &str, dir_idx: &str, collection: &str, volume_id: Volume data_vif } -/// Read EC data/parity shard counts from `.vif`, defaulting to the -/// build's standard ratio when no `.vif` is present or is malformed. +/// Load the volume's `.vif` (data dir first, then idx dir, then the active +/// generation's versioned sidecar — see [`locate_vif_path`]). Absent +/// everywhere is legal — legacy volumes predate the sidecar — and returns +/// `None`; so does a zero-byte stub (an ec.decode copy from a source +/// without one), mirroring Go's MaybeLoadVolumeInfo. A +/// present-but-unreadable or malformed `.vif` is an ERROR: silently +/// defaulting would mount a uniform-layout volume with legacy offset math +/// and serve wrong bytes with a straight face. +pub fn load_vif_info( + dir: &str, + dir_idx: &str, + collection: &str, + volume_id: VolumeId, +) -> io::Result> { + let vif_path = locate_vif_path(dir, dir_idx, collection, volume_id); + match std::fs::read_to_string(&vif_path) { + Ok(content) if content.trim().is_empty() => Ok(None), + Ok(content) => serde_json::from_str(&content).map(Some).map_err(|e| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("parse {}: {}", vif_path, e), + ) + }), + Err(e) if e.kind() == io::ErrorKind::NotFound => Ok(None), + Err(e) => Err(io::Error::new( + e.kind(), + format!("read {}: {}", vif_path, e), + )), + } +} + +/// Read the EC data/parity shard counts and shard block size from `.vif`, +/// defaulting to the build's standard ratio and the legacy block layout when +/// no `.vif` is present. A present-but-unreadable or malformed `.vif` +/// fails instead of defaulting (see [`load_vif_info`]). /// Looks at the data dir first, then the idx dir — see [`locate_vif_path`]. pub fn read_ec_shard_config( dir: &str, dir_idx: &str, collection: &str, volume_id: VolumeId, -) -> (u32, u32) { - let mut data_shards = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT as u32; - let mut parity_shards = crate::storage::erasure_coding::ec_shard::PARITY_SHARDS_COUNT as u32; - let vif_path = locate_vif_path(dir, dir_idx, collection, volume_id); - if let Ok(vif_content) = std::fs::read_to_string(&vif_path) { - if let Ok(vif_info) = - serde_json::from_str::(&vif_content) - { - if let Some(ec) = vif_info.ec_shard_config { - if ec.data_shards > 0 - && ec.parity_shards > 0 - && (ec.data_shards + ec.parity_shards) <= MAX_SHARD_COUNT as u32 - { - data_shards = ec.data_shards; - parity_shards = ec.parity_shards; - } +) -> io::Result<(u32, u32, i64)> { + let vif = load_vif_info(dir, dir_idx, collection, volume_id)?; + ec_shard_config_from(vif.as_ref(), &[dir, dir_idx], collection, volume_id) +} + +/// Load the volume's `.vif` from the selected location or, failing that, from +/// any sibling disk — returning the directory it came from so the caller can +/// resolve the rest of the volume's metadata against the same place. A rebuild +/// picks one disk to write into, but a multi-disk server may hold that +/// volume's metadata on another. +pub fn load_vif_info_across_dirs( + dir: &str, + dir_idx: &str, + other_dirs: &[String], + collection: &str, + volume_id: VolumeId, +) -> io::Result> { + if let Some(vif) = load_vif_info(dir, dir_idx, collection, volume_id)? { + // Report where it actually came from. load_vif_info probes the data + // directory then the index directory, so naming `dir` unconditionally + // pointed callers at the wrong disk whenever the vif lived with the + // index — and a caller resolving the rest of the volume's metadata + // against that answer would look in a directory holding none of it. + let data_vif = format!( + "{}.vif", + crate::storage::volume::volume_file_name(dir, collection, volume_id) + ); + let found_in = if std::path::Path::new(&data_vif).exists() { + dir.to_string() + } else { + dir_idx.to_string() + }; + return Ok(Some((vif, found_in))); + } + for other in other_dirs { + if let Some(vif) = load_vif_info(other, other, collection, volume_id)? { + return Ok(Some((vif, other.clone()))); + } + } + Ok(None) +} + +/// [`read_ec_shard_config`] for a caller that may find the volume's metadata on +/// any of the server's disks. Without this a rebuild that lands on a disk +/// holding only shards reads the default 10+4 and the legacy block layout, +/// then reconstructs a custom-ratio or uniform volume through the wrong matrix +/// and de-striping geometry. +pub fn read_ec_shard_config_across_dirs( + dir: &str, + dir_idx: &str, + other_dirs: &[String], + collection: &str, + volume_id: VolumeId, +) -> io::Result<(u32, u32, i64)> { + // Every directory this volume's metadata could be in. A split -dir/-dir.idx + // location keeps .vif and .ecsum with the INDEX, and callers exclude their + // own index directory from other_dirs on the understanding that it is + // passed here separately — so it is chained explicitly, exactly as the .vif + // lookup and Go's findBitrotSidecar both do. + let mut candidates: Vec<&str> = Vec::with_capacity(2 + other_dirs.len()); + candidates.push(dir); + if !dir_idx.is_empty() && dir_idx != dir { + candidates.push(dir_idx); + } + candidates.extend(other_dirs.iter().map(|s| s.as_str())); + + // A vif that carries no ecShardConfig answers nothing about the layout, so + // it must NOT short-circuit the sidecar search: a legacy config-free vif + // and the generation-0 sidecar can sit in different directories, and + // stopping at the vif resolved a 12+4 uniform volume as 10+4 legacy. Pass + // the whole candidate list so the fallback covers the same ground the vif + // lookup did. + let vif = load_vif_info_across_dirs(dir, dir_idx, other_dirs, collection, volume_id)?; + ec_shard_config_from( + vif.as_ref().map(|(v, _)| v), + &candidates, + collection, + volume_id, + ) +} + +/// Resolve (data_shards, parity_shards, block_size) from an already-loaded +/// `.vif`. With no vif — or one carrying no EC config — the bitrot sidecar +/// records the same config at encode time and answers the layout question the +/// vif cannot: defaulting a uniform-layout volume to the legacy block sizes +/// maps every read to the wrong shard offset. `weed fix -ecx` reads the +/// sidecar for the same reason. Absent both, the build's standard ratio and +/// the legacy layout. +pub fn ec_shard_config_from( + vif: Option<&crate::storage::volume::VifVolumeInfo>, + dirs: &[&str], + collection: &str, + volume_id: VolumeId, +) -> io::Result<(u32, u32, i64)> { + let default_ds = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT as u32; + let default_ps = crate::storage::erasure_coding::ec_shard::PARITY_SHARDS_COUNT as u32; + // Sum as u64: counts near the u32 ceiling would wrap and pass the bound. + let usable = |ds: u32, ps: u32| { + ds > 0 && ps > 0 && (ds as u64 + ps as u64) <= MAX_SHARD_COUNT as u64 + }; + + if let Some(ec) = vif.and_then(|v| v.ec_shard_config.as_ref()) { + // A config that is PRESENT but records an impossible ratio is not a + // volume to fall back on: dropping through to the sidecar or the + // defaults would read uniform shards with the legacy offset math and + // answer with the wrong bytes. Only an ENTIRELY absent config means + // "this predates the record". + if !usable(ec.data_shards, ec.parity_shards) { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "vif for volume {} records invalid shard counts {}+{}", + volume_id.0, ec.data_shards, ec.parity_shards + ), + )); + } + // A recorded block size that no encoder could have produced maps + // every read to the wrong shard offset. Refuse the mount rather + // than serve those bytes or silently pick a layout. + validate_block_size(ec.block_size)?; + return Ok((ec.data_shards, ec.parity_shards, ec.block_size)); + } + // With no usable vif config the sidecar is the ONLY record of this volume's + // layout, so the three answers stay distinct: absent EVERYWHERE means the + // volume may genuinely predate the sidecar and legacy is the right guess; + // present-but-unusable means the record is corrupt, and answering reads + // from a guessed layout returns wrong bytes rather than none. + // + // Every candidate directory is searched, not just the first: a split + // -dir/-dir.idx location keeps the sidecar with the INDEX, and a multi-disk + // server may keep it on a sibling. Stopping at the data directory resolved + // a 12+4 uniform volume as 10+4 legacy. + let mut path = String::new(); + for candidate in dirs.iter().filter(|d| !d.is_empty()) { + let base = crate::storage::volume::volume_file_name(candidate, collection, volume_id); + let probe = crate::storage::erasure_coding::ec_bitrot::bitrot_sidecar_path(&base, 0); + match std::fs::metadata(&probe) { + Err(e) if e.kind() == io::ErrorKind::NotFound => continue, + Err(e) => { + return Err(io::Error::new(e.kind(), format!("stat {}: {}", probe, e))); + } + Ok(_) => { + path = probe; + break; } } } - (data_shards, parity_shards) + if path.is_empty() { + return Ok((default_ds, default_ps, 0)); + } + let prot = crate::storage::erasure_coding::ec_bitrot::load_bitrot_sidecar(&path) + .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, format!("read {}: {}", path, e)))?; + // The un-suffixed sidecar describes generation 0 and nothing else. + if prot.generation != 0 { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!("{} records generation {}, not generation 0", path, prot.generation), + )); + } + let ec = prot.ec_shard_config.as_ref().ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("{} records no EC config", path), + ) + })?; + if !usable(ec.data_shards, ec.parity_shards) { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "{} records invalid shard counts {}+{}", + path, ec.data_shards, ec.parity_shards + ), + )); + } + validate_block_size(ec.block_size) + .map_err(|e| io::Error::new(e.kind(), format!("{}: {}", path, e)))?; + Ok((ec.data_shards, ec.parity_shards, ec.block_size)) +} + +/// Reports whether a `.vif`-recorded shard block size is one an encoder could +/// have produced. 0 means the legacy two-tier layout, which is always valid; +/// anything positive must be a whole number of small blocks, because that is +/// what `uniform_block_size` rounds to. Mirrors Go's `ValidateBlockSize`. +pub fn validate_block_size(block_size: i64) -> io::Result<()> { + if block_size == 0 { + return Ok(()); + } + if block_size < 0 + || block_size + % crate::storage::erasure_coding::ec_shard::ERASURE_CODING_SMALL_BLOCK_SIZE as i64 + != 0 + { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "invalid shard block size {}: expected 0 (legacy) or a multiple of {}", + block_size, + crate::storage::erasure_coding::ec_shard::ERASURE_CODING_SMALL_BLOCK_SIZE + ), + )); + } + Ok(()) } impl EcVolume { @@ -139,7 +357,14 @@ impl EcVolume { collection: &str, volume_id: VolumeId, ) -> io::Result { - let (data_shards, parity_shards) = read_ec_shard_config(dir, dir_idx, collection, volume_id); + // One load of the volume's `.vif`, used for both the shard config and + // the version / dat-size fields below. + let vif = load_vif_info(dir, dir_idx, collection, volume_id)?; + // Both directories: a split -dir/-dir.idx layout keeps the .ecsum with + // the INDEX, and it is the only layout record a volume whose vif is + // absent or config-free still has. + let (data_shards, parity_shards, block_size) = + ec_shard_config_from(vif.as_ref(), &[dir, dir_idx], collection, volume_id)?; let total_shards = (data_shards + parity_shards) as usize; let mut shards = Vec::with_capacity(total_shards); @@ -155,12 +380,11 @@ impl EcVolume { // for shard-size math and by the Store-level prune in // `store_ec_reconcile.rs` to verify a sibling-disk .dat is // plausibly the encoding source (#9478). - let (expire_at_sec, vif_version, vif_dat_file_size, encode_ts_ns) = { - let vif_path = locate_vif_path(dir, dir_idx, collection, volume_id); - if let Ok(vif_content) = std::fs::read_to_string(&vif_path) { - if let Ok(vif_info) = - serde_json::from_str::(&vif_content) - { + let (expire_at_sec, vif_version, vif_dat_file_size, encode_ts_ns) = + // Absent everywhere = legacy defaults; a present-but-unreadable or + // malformed vif already failed the mount in load_vif_info above. + match vif.as_ref() { + Some(vif_info) => { let ver = if vif_info.version > 0 { Version(vif_info.version as u8) } else { @@ -176,13 +400,9 @@ impl EcVolume { vif_info.dat_file_size, cfg_encode_ts_ns, ) - } else { - (0, Version::current(), 0, 0) } - } else { - (0, Version::current(), 0, 0) - } - }; + None => (0, Version::current(), 0, 0), + }; let mut vol = EcVolume { volume_id, @@ -194,6 +414,7 @@ impl EcVolume { dat_file_size: vif_dat_file_size, data_shards, parity_shards, + block_size, ecx_file: None, ecx_file_size: 0, ecj_file: None, @@ -254,17 +475,33 @@ impl EcVolume { // Seed the in-memory deleted set from the journal. vol.load_deleted_needles_from_ecj()?; - // Load the generation-0 EC bitrot checksum sidecar (optional; best-effort). - vol.load_active_bitrot_sidecar(); + // Load the generation-0 EC bitrot checksum sidecar. Optional, except + // when it contradicts the volume's own geometry — see + // load_bitrot_for_generation. + vol.load_active_bitrot_sidecar(&[])?; Ok(vol) } + /// Re-resolve the checksum sidecar for a volume that is already mounted. A + /// shard delivery can bring the manifest with it, and the receive path only + /// writes the file, so without this the in-memory volume keeps the + /// protection state it resolved at mount (off) until a remount. + pub fn reload_bitrot_sidecar(&mut self, additional_dirs: &[String]) { + if let Err(e) = self.load_active_bitrot_sidecar(additional_dirs) { + tracing::warn!( + volume_id = self.volume_id.0, + error = %e, + "reload bitrot sidecar", + ); + } + } + /// Load the generation-0 checksum sidecar into `self.bitrot`/`self.bitrot_status`. /// OSS only produces generation-0 (fresh-encode) sidecars, mirroring Go's /// `loadActiveBitrotSidecar`. - fn load_active_bitrot_sidecar(&mut self) { - self.load_bitrot_for_generation(0); + fn load_active_bitrot_sidecar(&mut self, additional_dirs: &[String]) -> io::Result<()> { + self.load_bitrot_for_generation(0, additional_dirs) } /// Load and validate the sidecar describing `generation`, setting @@ -272,11 +509,60 @@ impl EcVolume { /// => `Off` (protection off, not corruption); self-integrity or manifest /// failure => `Invalid` with a warning (protection off pending repair); usable /// => `On`. Mirrors Go's `loadBitrotForGeneration`. - fn load_bitrot_for_generation(&mut self, generation: u32) { + fn load_bitrot_for_generation(&mut self, generation: u32, additional_dirs: &[String]) -> io::Result<()> { use crate::storage::erasure_coding::ec_bitrot; - let base = self.base_name(); - let path = ec_bitrot::bitrot_sidecar_path(&base, generation); + // Data base then index base, matching Go's findBitrotSidecar. A split + // -dir/-dir.idx location keeps the sidecar with the INDEX, and on a + // multi-disk server sharing one index directory that is the only copy + // the per-disk runtimes other than the first can see — searching the + // data base alone left them reporting no protection at all. + let data_path = ec_bitrot::bitrot_sidecar_path(&self.base_name(), generation); + let idx_path = ec_bitrot::bitrot_sidecar_path(&self.idx_base_name(), generation); + // Sibling disks last: startup mirroring gives each shard-bearing disk + // its own .ecx/.ecj/.vif but not the sidecar, and a delivery lands + // exactly one copy, so a runtime restricted to its own two directories + // would report no protection however often it reloaded. + let path = std::iter::once(data_path.clone()) + .chain(std::iter::once(idx_path)) + .chain(additional_dirs.iter().filter(|d| !d.is_empty()).map(|d| { + ec_bitrot::bitrot_sidecar_path( + &crate::storage::volume::volume_file_name(d, &self.collection, self.volume_id), + generation, + ) + })) + .find(|p| std::path::Path::new(p).exists()) + .unwrap_or(data_path); let loaded = ec_bitrot::load_bitrot_sidecar(&path); + // A sidecar written for THIS generation that contradicts the volume's + // geometry is not "no protection" — it says the layout the volume is + // about to serve reads with is wrong. Fail the mount. + if let Ok(prot) = &loaded { + if prot.generation == generation + && !ec_bitrot::geometry_matches( + prot, + self.data_shards as usize, + self.parity_shards as usize, + self.block_size, + ) + { + let cfg = prot.ec_shard_config.as_ref(); + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "ec volume {} generation {}: {} records layout {}+{} block {} but the volume is mounted as {}+{} block {}; refusing to serve one of the two layouts", + self.volume_id.0, + generation, + path, + cfg.map(|c| c.data_shards).unwrap_or(0), + cfg.map(|c| c.parity_shards).unwrap_or(0), + cfg.map(|c| c.block_size).unwrap_or(0), + self.data_shards, + self.parity_shards, + self.block_size, + ), + )); + } + } let status = ec_bitrot::resolve_status( &loaded, generation, @@ -297,6 +583,7 @@ impl EcVolume { ); } } + Ok(()) } /// The active-generation bitrot protection AND its status (cached at mount), @@ -765,6 +1052,26 @@ impl EcVolume { Ok(None) } + /// Large-block length of this volume's shard layout (the uniform block + /// size when set, the legacy 1GiB otherwise). Mirrors Go's + /// ECContext.LargeBlockSize. + pub fn large_block_size(&self) -> i64 { + if self.block_size > 0 { + self.block_size + } else { + ERASURE_CODING_LARGE_BLOCK_SIZE as i64 + } + } + + /// Small-block length of this volume's shard layout. + pub fn small_block_size(&self) -> i64 { + if self.block_size > 0 { + self.block_size + } else { + ERASURE_CODING_SMALL_BLOCK_SIZE as i64 + } + } + /// Locate the EC shard intervals needed to read a needle. /// Locate the EC shard intervals covering a needle at `actual_offset` whose /// index size is `size`. Mirrors Go's EcVolume.LocateEcShardNeedleInterval. @@ -783,7 +1090,27 @@ impl EcVolume { }; // locate_data wants the on-disk size (header+body+checksum+timestamp+padding). let actual = get_actual_size(size, self.version); - ec_locate::locate_data(actual_offset, Size(actual as i32), shard_size, self.data_shards) + ec_locate::locate_data( + actual_offset, + Size(actual as i32), + shard_size, + self.data_shards, + self.large_block_size(), + self.small_block_size(), + ) + } + + /// Resolve an interval against this volume's shard block layout. Mirrors + /// Go's EcVolume.IntervalToShardIdAndOffset. + pub fn interval_to_shard_id_and_offset( + &self, + interval: &ec_locate::Interval, + ) -> (ShardId, i64) { + interval.to_shard_id_and_offset( + self.data_shards, + self.large_block_size(), + self.small_block_size(), + ) } pub fn locate_needle( @@ -829,7 +1156,7 @@ impl EcVolume { let mut bytes = Vec::with_capacity(actual_size); for interval in &intervals { - let (shard_id, shard_offset) = interval.to_shard_id_and_offset(self.data_shards); + let (shard_id, shard_offset) = self.interval_to_shard_id_and_offset(interval); let shard = self .shards .get(shard_id as usize) @@ -989,7 +1316,7 @@ impl EcVolume { // A needle is verifiable locally only if every shard it spans is local; // when any is remote, skip the reassembly buffer entirely. let has_remote_chunks = locations.iter().any(|iv| { - let (sid, _) = iv.to_shard_id_and_offset(self.data_shards); + let (sid, _) = self.interval_to_shard_id_and_offset(iv); self.shards.get(sid as usize).and_then(|s| s.as_ref()).is_none() }); let mut read: i64 = 0; @@ -1001,7 +1328,7 @@ impl EcVolume { let mut local_shard_ids: Vec = Vec::new(); for (i, iv) in locations.iter().enumerate() { - let (sid, soffset) = iv.to_shard_id_and_offset(self.data_shards); + let (sid, soffset) = self.interval_to_shard_id_and_offset(iv); let ssize = iv.size; let shard = match self.shards.get(sid as usize).and_then(|s| s.as_ref()) { Some(s) => s, @@ -1800,8 +2127,14 @@ mod tests { assert_eq!(vol.parity_shards, 3); } + /// A vif that RECORDS a ratio no volume could have must fail the mount. + /// Substituting the default 10+4 (and with it the legacy block layout) + /// would read a uniform volume's shards at the wrong offsets and answer + /// with the wrong bytes; only an entirely absent config means "this + /// predates the record", which + /// `test_ec_volume_absent_vif_config_uses_defaults` covers. #[test] - fn test_ec_volume_invalid_vif_config_falls_back_to_defaults() { + fn test_ec_volume_invalid_vif_config_fails_the_mount() { let tmp = TempDir::new().unwrap(); let dir = tmp.path().to_str().unwrap(); write_ecx_file(dir, "pics", VolumeId(1), &[]); @@ -1822,9 +2155,27 @@ mod tests { ) .unwrap(); + // EcVolume has no Debug impl, so match rather than expect_err. + match EcVolume::new(dir, dir, "pics", VolumeId(1)) { + Ok(_) => panic!("an impossible ratio must fail the mount"), + Err(e) => assert_eq!(e.kind(), io::ErrorKind::InvalidData, "got {e}"), + } + } + + /// The compatibility case the check above must not swallow: a vif with no + /// EC config at all predates the record, and mounts on the defaults. + #[test] + fn test_ec_volume_absent_vif_config_uses_defaults() { + let tmp = TempDir::new().unwrap(); + let dir = tmp.path().to_str().unwrap(); + write_ecx_file(dir, "pics", VolumeId(1), &[]); + let base = crate::storage::volume::volume_file_name(dir, "pics", VolumeId(1)); + std::fs::write(format!("{}.vif", base), r#"{"version":3}"#).unwrap(); + let vol = EcVolume::new(dir, dir, "pics", VolumeId(1)).unwrap(); assert_eq!(vol.data_shards, DATA_SHARDS_COUNT as u32); assert_eq!(vol.parity_shards, PARITY_SHARDS_COUNT as u32); + assert_eq!(vol.block_size, 0, "legacy layout for a pre-record volume"); } #[test] @@ -1973,3 +2324,370 @@ mod tests { ); } } + +#[cfg(test)] +mod uniform_layout_tests { + use super::*; + use crate::storage::needle_map::NeedleMapKind; + use crate::storage::volume::{VifEcShardConfig, VifVolumeInfo, Volume}; + use tempfile::TempDir; + + // Write ~26MB of needles so the uniform block size (3MB) diverges from the + // legacy layout, encode, and verify EcVolume reads every needle back + // through the .vif-recorded geometry. A legacy-encoded fixture (no .vif + // block size) must keep reading through the legacy interpretation. + #[test] + fn test_read_needles_uniform_and_legacy_layouts() { + for legacy in [false, true] { + let tmp = TempDir::new().unwrap(); + let dir = tmp.path().to_str().unwrap(); + let vid = VolumeId(8); + + let mut v = Volume::new( + dir, + dir, + "", + vid, + NeedleMapKind::InMemory, + None, + None, + 0, + Version::current(), + ) + .unwrap(); + let mut expected: Vec<(NeedleId, Vec)> = Vec::new(); + for i in 1u64..=11 { + let size = if i <= 5 { 4 << 20 } else { 1 << 20 }; + let data: Vec = (0..size) + .map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8) + .collect(); + let mut n = Needle { + id: NeedleId(i), + cookie: Cookie(i as u32), + data: data.clone(), + data_size: data.len() as u32, + ..Needle::default() + }; + v.write_needle(&mut n, true, false).unwrap(); + expected.push((NeedleId(i), data)); + } + v.sync_to_disk().unwrap(); + let dat_size = v.dat_file_size().unwrap() as i64; + v.close(); + + let block_size = if legacy { + // Legacy fixture: two-tier encode plus a .vif without a block + // size, the state every pre-upgrade EC volume is in. + use crate::storage::erasure_coding::ec_bitrot::{ + ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE, + }; + use reed_solomon_erasure::galois_8::ReedSolomon; + let base = crate::storage::volume::volume_file_name(dir, "", vid); + crate::storage::erasure_coding::ec_encoder::write_sorted_ecx_from_idx( + &format!("{}.idx", base), + &format!("{}.ecx", base), + ) + .unwrap(); + let dat_file = std::fs::File::open(format!("{}.dat", base)).unwrap(); + let rs = ReedSolomon::new(10, 4).unwrap(); + let mut shards: Vec = (0..14u8) + .map(|i| EcVolumeShard::new(dir, "", vid, i)) + .collect(); + for shard in &mut shards { + shard.create().unwrap(); + } + let mut builders: Vec = (0..14) + .map(|_| ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64)) + .collect(); + crate::storage::erasure_coding::ec_encoder::encode_dat_file( + &dat_file, + dat_size, + &rs, + &mut shards, + &mut builders, + 10, + 4, + 256 * 1024, + ERASURE_CODING_LARGE_BLOCK_SIZE, + ERASURE_CODING_SMALL_BLOCK_SIZE, + ) + .unwrap(); + for shard in &mut shards { + shard.close(); + } + 0 + } else { + let bs = crate::storage::erasure_coding::ec_encoder::write_ec_files( + dir, dir, "", vid, 10, 4, + ) + .unwrap(); + assert!( + bs > ERASURE_CODING_SMALL_BLOCK_SIZE as i64, + "block size {} does not diverge from the legacy layout", + bs + ); + bs + }; + + let base = crate::storage::volume::volume_file_name(dir, "", vid); + let vif = VifVolumeInfo { + version: Version::current().0 as u32, + dat_file_size: dat_size, + ec_shard_config: Some(VifEcShardConfig { + data_shards: 10, + parity_shards: 4, + block_size, + ..Default::default() + }), + ..Default::default() + }; + std::fs::write( + format!("{}.vif", base), + serde_json::to_string_pretty(&vif).unwrap(), + ) + .unwrap(); + std::fs::remove_file(format!("{}.dat", base)).unwrap(); + std::fs::remove_file(format!("{}.idx", base)).unwrap(); + + let mut vol = EcVolume::new(dir, dir, "", vid).unwrap(); + assert_eq!(vol.block_size, block_size, "block size not loaded from .vif"); + for i in 0..10u8 { + vol.add_shard(EcVolumeShard::new(dir, "", vid, i)).unwrap(); + } + + for (id, data) in &expected { + let n = vol + .read_ec_shard_needle(*id) + .unwrap() + .unwrap_or_else(|| panic!("needle {} not found (legacy={})", id.0, legacy)); + assert_eq!( + n.data, *data, + "needle {} data mismatch (legacy={})", + id.0, legacy + ); + } + } + } + + // A present-but-malformed .vif must FAIL the mount: every new encode + // records a positive uniform block size there, and silently defaulting + // to the legacy layout would serve those shards with the wrong offset + // math. Absence stays legal — legacy volumes predate the sidecar. + #[test] + fn new_fails_on_malformed_vif() { + let tmp = tempfile::TempDir::new().unwrap(); + let dir = tmp.path().to_str().unwrap(); + let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(1)); + std::fs::write(format!("{}.ecx", base), b"").unwrap(); + std::fs::write(format!("{}.vif", base), b"not json").unwrap(); + let res = EcVolume::new(dir, dir, "", VolumeId(1)); + assert!(res.is_err(), "mount over a malformed .vif must fail"); + } + + #[test] + fn new_absent_vif_mounts_with_defaults() { + let tmp = tempfile::TempDir::new().unwrap(); + let dir = tmp.path().to_str().unwrap(); + let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(1)); + std::fs::write(format!("{}.ecx", base), b"").unwrap(); + let ev = EcVolume::new(dir, dir, "", VolumeId(1)).unwrap(); + assert_eq!(ev.block_size, 0, "legacy mount must use the legacy layout"); + } + + /// A rebuild lands on one disk, but the volume's metadata may live on + /// another. Reading only the selected directory returns the default 10+4 + /// and the legacy block layout, which reconstructs a custom-ratio or + /// uniform volume through the wrong matrix. + fn seed_uniform_sidecar(base: &str, ds: u32, ps: u32, block: i64) { + use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums}; + use crate::storage::erasure_coding::ec_bitrot; + let prot = EcBitrotProtection { + algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32, + block_size: ec_bitrot::DEFAULT_BITROT_BLOCK_SIZE as u32, + generation: 0, + ec_shard_config: Some(ec_bitrot::ec_shard_config(ds, ps, block)), + shards: (0..(ds + ps)) + .map(|shard_id| EcShardChecksums { + shard_id, + covered_size: 4, + block_crc32c: vec![0u8; 4], + }) + .collect(), + encode_uuid: vec![0u8; 16], + }; + ec_bitrot::save_bitrot_sidecar(&ec_bitrot::bitrot_sidecar_path(base, 0), &prot).unwrap(); + } + + fn seed_config_free_vif(base: &str) { + std::fs::write( + format!("{}.vif", base), + r#"{"version":3,"datFileSize":0}"#, + ) + .unwrap(); + } + + // A .vif that omits ecShardConfig answers nothing about the layout, so it + // must not short-circuit the sidecar search. Returning as soon as any + // parseable vif turned up resolved a 12+4 uniform volume as 10+4 legacy. + #[test] + fn read_ec_shard_config_falls_back_when_the_vif_has_no_config() { + let d = tempfile::TempDir::new().unwrap(); + let dir = d.path().to_str().unwrap(); + let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(11)); + seed_config_free_vif(&base); + seed_uniform_sidecar(&base, 12, 4, 3 * 1024 * 1024); + + let got = read_ec_shard_config(dir, dir, "", VolumeId(11)).unwrap(); + assert_eq!(got, (12, 4, 3 * 1024 * 1024)); + } + + // Split -dir/-dir.idx: the vif and the sidecar both live with the INDEX, + // and the resolver was told to report the DATA directory, so the fallback + // looked in a directory holding neither. + #[test] + fn read_ec_shard_config_finds_a_config_free_vifs_sidecar_in_the_index_dir() { + let data = tempfile::TempDir::new().unwrap(); + let idx = tempfile::TempDir::new().unwrap(); + let (dir, dir_idx) = (data.path().to_str().unwrap(), idx.path().to_str().unwrap()); + let idx_base = crate::storage::volume::volume_file_name(dir_idx, "", VolumeId(12)); + seed_config_free_vif(&idx_base); + seed_uniform_sidecar(&idx_base, 12, 4, 3 * 1024 * 1024); + + let got = read_ec_shard_config(dir, dir_idx, "", VolumeId(12)).unwrap(); + assert_eq!(got, (12, 4, 3 * 1024 * 1024)); + } + + // Same gap in the cross-disk resolver the rebuild uses. + #[test] + fn read_ec_shard_config_across_dirs_falls_back_when_the_vif_has_no_config() { + let data = tempfile::TempDir::new().unwrap(); + let sib = tempfile::TempDir::new().unwrap(); + let (dir, sibling) = (data.path().to_str().unwrap(), sib.path().to_str().unwrap()); + seed_config_free_vif(&crate::storage::volume::volume_file_name(dir, "", VolumeId(13))); + seed_uniform_sidecar( + &crate::storage::volume::volume_file_name(sibling, "", VolumeId(13)), + 12, + 4, + 3 * 1024 * 1024, + ); + + let got = read_ec_shard_config_across_dirs( + dir, + dir, + &[sibling.to_string()], + "", + VolumeId(13), + ) + .unwrap(); + assert_eq!(got, (12, 4, 3 * 1024 * 1024)); + } + + // Absence stays legal: a volume predating both records is genuinely legacy. + #[test] + fn read_ec_shard_config_config_free_vif_without_a_sidecar_is_legacy() { + let d = tempfile::TempDir::new().unwrap(); + let dir = d.path().to_str().unwrap(); + seed_config_free_vif(&crate::storage::volume::volume_file_name(dir, "", VolumeId(14))); + let got = read_ec_shard_config(dir, dir, "", VolumeId(14)).unwrap(); + assert_eq!(got, (10, 4, 0)); + } + + #[test] + fn read_ec_shard_config_finds_a_sibling_disks_vif() { + let a = tempfile::TempDir::new().unwrap(); + let b = tempfile::TempDir::new().unwrap(); + let (rebuild, sibling) = (a.path().to_str().unwrap(), b.path().to_str().unwrap()); + let base = crate::storage::volume::volume_file_name(sibling, "", VolumeId(3)); + std::fs::write( + format!("{}.vif", base), + r#"{"version":3,"ecShardConfig":{"dataShards":12,"parityShards":4,"blockSize":3145728}}"#, + ) + .unwrap(); + + let (ds, ps, bs) = read_ec_shard_config_across_dirs( + rebuild, + rebuild, + &[sibling.to_string()], + "", + VolumeId(3), + ) + .unwrap(); + assert_eq!((ds, ps, bs), (12, 4, 3 * 1024 * 1024)); + } + + /// Same, for the sidecar: with no .vif anywhere it is the surviving record + /// of the geometry, wherever it sits. + // A split -dir/-dir.idx location keeps its metadata with the INDEX, and + // callers leave their own index directory out of other_dirs because it is + // passed separately. With no .vif anywhere the generation-0 .ecsum is the + // only record of the layout, so missing that directory resolves a 12+4 + // uniform volume to 10+4 legacy and reconstructs through the wrong matrix. + #[test] + fn read_ec_shard_config_finds_the_sidecar_in_its_own_index_dir() { + use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums}; + use crate::storage::erasure_coding::ec_bitrot; + + let data = tempfile::TempDir::new().unwrap(); + let idx = tempfile::TempDir::new().unwrap(); + let (dir, dir_idx) = ( + data.path().to_str().unwrap(), + idx.path().to_str().unwrap(), + ); + let base = crate::storage::volume::volume_file_name(dir_idx, "", VolumeId(7)); + let prot = EcBitrotProtection { + algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32, + block_size: ec_bitrot::DEFAULT_BITROT_BLOCK_SIZE as u32, + generation: 0, + ec_shard_config: Some(ec_bitrot::ec_shard_config(12, 4, 3 * 1024 * 1024)), + shards: (0..16u32) + .map(|shard_id| EcShardChecksums { + shard_id, + covered_size: 4, + block_crc32c: vec![0u8; 4], + }) + .collect(), + encode_uuid: vec![0u8; 16], + }; + ec_bitrot::save_bitrot_sidecar(&ec_bitrot::bitrot_sidecar_path(&base, 0), &prot).unwrap(); + + let (ds, ps, bs) = + read_ec_shard_config_across_dirs(dir, dir_idx, &[], "", VolumeId(7)).unwrap(); + assert_eq!((ds, ps, bs), (12, 4, 3 * 1024 * 1024)); + } + + #[test] + fn read_ec_shard_config_finds_a_sibling_disks_sidecar() { + use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums}; + use crate::storage::erasure_coding::ec_bitrot; + + let a = tempfile::TempDir::new().unwrap(); + let b = tempfile::TempDir::new().unwrap(); + let (rebuild, sibling) = (a.path().to_str().unwrap(), b.path().to_str().unwrap()); + let base = crate::storage::volume::volume_file_name(sibling, "", VolumeId(4)); + let prot = EcBitrotProtection { + algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32, + block_size: ec_bitrot::DEFAULT_BITROT_BLOCK_SIZE as u32, + generation: 0, + ec_shard_config: Some(ec_bitrot::ec_shard_config(12, 4, 3 * 1024 * 1024)), + shards: (0..16u32) + .map(|shard_id| EcShardChecksums { + shard_id, + covered_size: 4, + block_crc32c: vec![0u8; 4], + }) + .collect(), + encode_uuid: vec![0u8; 16], + }; + ec_bitrot::save_bitrot_sidecar(&ec_bitrot::bitrot_sidecar_path(&base, 0), &prot).unwrap(); + + let (ds, ps, bs) = read_ec_shard_config_across_dirs( + rebuild, + rebuild, + &[sibling.to_string()], + "", + VolumeId(4), + ) + .unwrap(); + assert_eq!((ds, ps, bs), (12, 4, 3 * 1024 * 1024)); + } +} diff --git a/seaweed-volume/src/storage/store.rs b/seaweed-volume/src/storage/store.rs index 644c6b76d..b458efb47 100644 --- a/seaweed-volume/src/storage/store.rs +++ b/seaweed-volume/src/storage/store.rs @@ -804,6 +804,37 @@ impl Store { } /// Find an EC volume across all locations (mutable). + /// Every per-disk `EcVolume` this store maps for `vid`. A vid can mount on + /// N disks as N distinct runtimes, and the first-match `find_ec_volume_mut` + /// hides the siblings — so anything that has to reach the whole volume, + /// rather than any one runtime of it, iterates this instead. + /// Every directory on this server that could hold an EC volume's metadata — + /// each disk's data and index directory. Startup mirroring gives each + /// shard-bearing disk its own .ecx/.ecj/.vif, but the checksum sidecar is + /// not mirrored and a repair delivers exactly one copy, so a runtime looking + /// only at its own two directories cannot see it. Handing this list to the + /// sidecar resolution keeps one authoritative copy reachable from every + /// runtime rather than duplicating a file that is rewritten as shards are + /// repaired and generations published. + pub fn ec_metadata_dirs(&self) -> Vec { + let mut dirs: Vec = Vec::with_capacity(self.locations.len() * 2); + for loc in &self.locations { + for dir in [&loc.directory, &loc.idx_directory] { + if !dir.is_empty() && !dirs.iter().any(|d| d == dir) { + dirs.push(dir.clone()); + } + } + } + dirs + } + + pub fn find_all_ec_volumes_mut(&mut self, vid: VolumeId) -> Vec<&mut EcVolume> { + self.locations + .iter_mut() + .filter_map(|loc| loc.find_ec_volume_mut(vid)) + .collect() + } + pub fn find_ec_volume_mut(&mut self, vid: VolumeId) -> Option<&mut EcVolume> { for loc in &mut self.locations { if let Some(ecv) = loc.find_ec_volume_mut(vid) { diff --git a/seaweed-volume/src/storage/volume.rs b/seaweed-volume/src/storage/volume.rs index 64572c345..6373c9eda 100644 --- a/seaweed-volume/src/storage/volume.rs +++ b/seaweed-volume/src/storage/volume.rs @@ -194,6 +194,10 @@ pub struct VifEcShardConfig { /// so a read served from a different run's shard is rejected. #[serde(default, rename = "encodeTsNs", with = "string_or_i64")] pub encode_ts_ns: i64, + /// Uniform block layout: each shard is a single contiguous block of this + /// many bytes. 0 = legacy 1GiB/1MiB two-tier layout. + #[serde(default, rename = "blockSize", with = "string_or_i64")] + pub block_size: i64, } /// Serde-compatible representation of OldVersionVolumeInfo for legacy .vif JSON deserialization. @@ -288,6 +292,7 @@ impl VifVolumeInfo { data_shards: c.data_shards, parity_shards: c.parity_shards, encode_ts_ns: c.encode_ts_ns, + block_size: c.block_size, }), read_only_can_delete: pb.read_only_can_delete, } @@ -320,6 +325,7 @@ impl VifVolumeInfo { data_shards: c.data_shards, parity_shards: c.parity_shards, encode_ts_ns: c.encode_ts_ns, + block_size: c.block_size, } }), read_only_can_delete: self.read_only_can_delete, diff --git a/test/plugin_workers/fake_volume_server.go b/test/plugin_workers/fake_volume_server.go index 103515ce6..1ea58276e 100644 --- a/test/plugin_workers/fake_volume_server.go +++ b/test/plugin_workers/fake_volume_server.go @@ -15,6 +15,7 @@ import ( "github.com/seaweedfs/seaweedfs/weed/operation" "github.com/seaweedfs/seaweedfs/weed/pb" "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" "google.golang.org/grpc" "google.golang.org/grpc/credentials/insecure" ) @@ -339,7 +340,14 @@ func (v *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serve } } - resp := &volume_server_pb.VolumeEcShardsInfoResponse{} + // Answer with the layout out of the .vif that was actually delivered here, + // the way a real holder answers from the context it mounted the shards + // with. A coordinator uses this to tell a server that understands the + // shard block layout from one that never knew the field, so a fake that + // always reported "unset" would look like a pre-upgrade server. + resp := &volume_server_pb.VolumeEcShardsInfoResponse{ + EcShardConfig: v.ecShardConfigFromVif(req.VolumeId), + } prefix := fmt.Sprintf("%d.ec", req.VolumeId) entries, _ := os.ReadDir(v.baseDir) for _, entry := range entries { @@ -372,6 +380,16 @@ func (v *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serve return resp, nil } +// ecShardConfigFromVif reads the EC layout out of the .vif this server was +// given, which distribution ships to every holder alongside its shards. +func (v *VolumeServer) ecShardConfigFromVif(volumeID uint32) *volume_server_pb.EcShardConfig { + vi, _, found, err := volume_info.MaybeLoadVolumeInfo(v.filePath(volumeID, ".vif")) + if err != nil || !found { + return nil + } + return vi.GetEcShardConfig() +} + func (v *VolumeServer) VolumeDelete(ctx context.Context, req *volume_server_pb.VolumeDeleteRequest) (*volume_server_pb.VolumeDeleteResponse, error) { v.mu.Lock() v.deleteRequests = append(v.deleteRequests, req) diff --git a/weed/command/fix.go b/weed/command/fix.go index 6c24dd320..301bfd090 100644 --- a/weed/command/fix.go +++ b/weed/command/fix.go @@ -327,9 +327,15 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6 // Priority: explicit flags > existing .vif > defaults (10+4). vifName := base + ".vif" vifExists := util.FileExists(vifName) + // Whether the on-disk .vif actually answers the layout question. An empty + // stub (MaybeLoadVolumeInfo reads it as absent) or one with no EC config + // exists but tells us nothing, and the recovered layout must still be + // written back over it. + vifUsable := false dataShards := erasure_coding.DataShardsCount parityShards := erasure_coding.ParityShardsCount var datFileSize int64 + blockSize := int64(-1) // the shard block layout; 0 legacy, >0 uniform, <0 unknown if vifExists { // MaybeLoadVolumeInfo returns a non-nil error when the .vif exists but // cannot be read or unmarshalled; fail loudly rather than silently @@ -338,13 +344,66 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6 fail(fmt.Errorf("volume %d: read %s: %w", volumeId, vifName, loadErr)) return } else if found && vi != nil { - if cfg := vi.GetEcShardConfig(); cfg != nil && cfg.GetDataShards() > 0 { + // A partial config (parity 0, or counts past the shard ceiling) + // describes no layout: leave the sentinel unknown and let the + // sidecar / dual scan below answer, and rewrite the .vif at the end + // rather than trusting it. + cfg := vi.GetEcShardConfig() + if cfg != nil && erasure_coding.ValidEcShardCounts(cfg.GetDataShards(), cfg.GetParityShards()) { dataShards = int(cfg.GetDataShards()) parityShards = int(cfg.GetParityShards()) + // Only an EC config answers the layout question. Reading 0 off a + // vif with no config would assert "legacy" and suppress both the + // .ecsum fallback and the dual-layout scan below; leave the + // sentinel at -1 (unknown) instead. + // A recorded block size no encoder could have produced is not + // an answer: a positive one would pin the scan to a geometry + // that de-stripes to garbage, and a negative one would leave + // the invalid .vif in place after the dual scan recovers the + // real layout. Leave the sentinel unknown and rewrite the file + // at the end. + if bs := cfg.GetBlockSize(); erasure_coding.ValidateBlockSize(bs) == nil { + blockSize = bs + vifUsable = true + } else { + glog.Warningf("volume %d: %s records an invalid shard block size %d; recovering the layout by scan and rewriting it", + volumeId, vifName, bs) + } } datFileSize = vi.GetDatFileSize() } } + // The bitrot sidecar records the same EC config at encode time; use it when + // the .vif is gone. A uniform-layout volume cannot be de-striped correctly + // without its block size. + if blockSize < 0 { + sidecarPath := erasure_coding.BitrotSidecarPath(base, 0) + if prot, serr := erasure_coding.LoadBitrotSidecar(sidecarPath); serr == nil { + // The sidecar is about to pin the reconstruction to one geometry + // instead of letting the dual scan decide, so every field it + // contributes has to hold up: it must describe generation 0 (the + // shards being read), a complete in-range ratio, and a block size + // an encoder could have produced. Anything less leaves the layout + // unknown, which is the answer that still recovers by scanning. + cfg := prot.GetEcShardConfig() + ds, ps := int(cfg.GetDataShards()), int(cfg.GetParityShards()) + switch { + case prot.GetGeneration() != 0: + glog.Warningf("volume %d: %s records generation %d, not the generation-0 shards; ignoring it", + volumeId, sidecarPath, prot.GetGeneration()) + case !erasure_coding.ValidEcShardCounts(cfg.GetDataShards(), cfg.GetParityShards()): + glog.Warningf("volume %d: %s records invalid shard counts %d+%d; ignoring it", + volumeId, sidecarPath, ds, ps) + case erasure_coding.ValidateBlockSize(cfg.GetBlockSize()) != nil: + glog.Warningf("volume %d: %s records an invalid shard block size %d; ignoring it", + volumeId, sidecarPath, cfg.GetBlockSize()) + default: + dataShards = ds + parityShards = ps + blockSize = cfg.GetBlockSize() + } + } + } if *fixEcDataShards > 0 { dataShards = *fixEcDataShards } @@ -381,6 +440,9 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6 } if !dataComplete { ctx := &erasure_coding.ECContext{DataShards: dataShards, ParityShards: parityShards} + if blockSize > 0 { + ctx.BlockSize = blockSize + } glog.Infof("volume %d: %d/%d shards present; reconstructing missing shards (%s) before index rebuild", volumeId, presentCount, dataShards+parityShards, ctx.String()) if _, err := erasure_coding.RebuildEcFiles(base, ctx, *fixEcUnsafeIgnoreSidecar); err != nil { fail(fmt.Errorf("volume %d: reconstruct missing shards from %d survivors: %w", volumeId, presentCount, err)) @@ -407,37 +469,114 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6 glog.V(0).Infof("volume %d: no .dat size in .vif; reconstructing padded .dat (%d bytes) from %d data shards", volumeId, reconstructSize, dataShards) } - // De-stripe the data shards into a temporary .dat next to the shards. + // De-stripe the data shards into a temporary .dat next to the shards and + // scan it into a fresh .ecx. With the layout unknown, de-stripe under both + // candidate layouts and keep the one whose needle chain scans furthest. + type layoutCandidate struct { + name string + large, small int64 + } + var candidates []layoutCandidate + switch { + case blockSize > 0: + candidates = []layoutCandidate{{"uniform", blockSize, blockSize}} + case blockSize == 0: + candidates = []layoutCandidate{{"legacy", erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize}} + default: + candidates = []layoutCandidate{ + {"legacy", erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize}, + } + // A uniform-layout shard is exactly one block long, and every block + // size an encoder can produce is a whole number of small blocks. An + // extent that is not — a truncated or partially copied shard — could + // not have come from a uniform encode, and recording it would write a + // .vif that ValidateBlockSize refuses on the next mount: the volume + // this tool was run to rescue would never open again. + if erasure_coding.ValidateBlockSize(shardSize) == nil && shardSize > 0 { + candidates = append(candidates, layoutCandidate{"uniform", shardSize, shardSize}) + glog.Infof("volume %d: no .vif or .ecsum records the shard block layout; trying both", volumeId) + } else { + glog.Infof("volume %d: no .vif or .ecsum records the shard block layout, and the %d-byte shard extent is not a whole number of %d-byte blocks; trying the legacy layout only", + volumeId, shardSize, erasure_coding.ErasureCodingSmallBlockSize) + } + } + tmpBase := base + ".ecxrecover" tmpDat := tmpBase + ".dat" - if err := erasure_coding.WriteDatFile(tmpBase, reconstructSize, reconstructSize, shardFileNames); err != nil { - os.Remove(tmpDat) - fail(fmt.Errorf("volume %d: reconstruct .dat from data shards: %w", volumeId, err)) + defer os.Remove(tmpDat) + bestIdx := -1 + var realDatSize, bestNeedles int64 + var version needle.Version + var candErr error + for i, cand := range candidates { + if err := erasure_coding.WriteDatFile(tmpBase, reconstructSize, reconstructSize, shardFileNames, cand.large, cand.small); err != nil { + candErr = fmt.Errorf("volume %d: reconstruct .dat (%s layout): %w", volumeId, cand.name, err) + continue + } + size, needles, ver, err := writeEcxFromDat(tmpDat, ecxName+"."+cand.name) + if err != nil { + os.Remove(ecxName + "." + cand.name) + candErr = fmt.Errorf("volume %d: build .ecx from reconstructed .dat (%s layout): %w", volumeId, cand.name, err) + continue + } + // The wrong layout de-stripes to garbage past the first block + // boundary, so the layout that indexes more valid needles wins. + // On a tie — garbage bytes do sometimes parse as plausible sizes — the + // layout whose needle chain reached further into the .dat wins, which is + // the distance the scan actually validated. + if bestIdx < 0 || needles > bestNeedles || (needles == bestNeedles && size > realDatSize) { + bestIdx = i + bestNeedles = needles + realDatSize = size + version = ver + } + } + if bestIdx < 0 { + fail(candErr) return } - defer os.Remove(tmpDat) - - realDatSize, version, err := writeEcxFromDat(tmpDat, ecxName) - if err != nil { - os.Remove(ecxName) - fail(fmt.Errorf("volume %d: build .ecx from reconstructed .dat: %w", volumeId, err)) + for i := range candidates { + if i != bestIdx { + os.Remove(ecxName + "." + candidates[i].name) + } + } + if err := os.Rename(ecxName+"."+candidates[bestIdx].name, ecxName); err != nil { + fail(fmt.Errorf("volume %d: publish %s: %w", volumeId, ecxName, err)) return } + if blockSize < 0 { + if candidates[bestIdx].name == "uniform" { + blockSize = shardSize + } else { + blockSize = 0 + } + glog.Infof("volume %d: %s layout indexes %d needles over %d bytes; keeping it", volumeId, candidates[bestIdx].name, bestNeedles, realDatSize) + } glog.Infof("volume %d: wrote %s from %d data shards", volumeId, ecxName, dataShards) - // Regenerate the .vif when missing so the volume can mount and future - // rebuilds know the EC ratio and original .dat size. - if !vifExists { + // Regenerate the .vif when it is missing OR unusable (an empty stub, or one + // with no EC config): the layout just recovered by the dual scan is the only + // record of it, and leaving the stub in place would mount the volume as + // legacy on the next start and serve the wrong offsets. + if !vifUsable { size := datFileSize if size <= 0 { size = realDatSize } + // Never publish a layout the mount path will refuse: NewEcVolume runs + // the same check and fails closed, so an unvalidated write here would + // trade a recoverable volume for one that can no longer be opened. + if bsErr := erasure_coding.ValidateBlockSize(blockSize); bsErr != nil { + fail(fmt.Errorf("volume %d: refusing to write %s: %w", volumeId, vifName, bsErr)) + return + } volumeInfo := &volume_server_pb.VolumeInfo{ Version: uint32(version), DatFileSize: size, EcShardConfig: &volume_server_pb.EcShardConfig{ DataShards: uint32(dataShards), ParityShards: uint32(parityShards), + BlockSize: blockSize, }, } if err := volume_info.SaveVolumeInfo(vifName, volumeInfo); err != nil { @@ -451,25 +590,26 @@ func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int6 // writeEcxFromDat scans a (reconstructed) .dat and writes an ascending-sorted // .ecx containing only live needles — the same on-disk shape // WriteSortedFileFromIdx produces when an EC volume is first encoded. It returns -// the physical .dat size (the offset where the EC zero padding begins) and the -// volume version read from the superblock. -func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle.Version, err error) { +// the physical .dat size (the offset where the EC zero padding begins), the +// number of live needles indexed, and the volume version read from the +// superblock. +func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, liveNeedles int64, version needle.Version, err error) { f, err := os.OpenFile(datPath, os.O_RDONLY, 0644) if err != nil { - return 0, 0, fmt.Errorf("open %s: %w", datPath, err) + return 0, 0, 0, fmt.Errorf("open %s: %w", datPath, err) } datBackend := backend.NewDiskFile(f) defer datBackend.Close() superBlock, err := super_block.ReadSuperBlock(datBackend) if err != nil { - return 0, 0, fmt.Errorf("read superblock: %w", err) + return 0, 0, 0, fmt.Errorf("read superblock: %w", err) } version = superBlock.Version fileSize, _, err := datBackend.GetStat() if err != nil { - return 0, version, fmt.Errorf("stat %s: %w", datPath, err) + return 0, 0, version, fmt.Errorf("stat %s: %w", datPath, err) } nm := needle_map.NewMemDb() @@ -482,7 +622,7 @@ func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle if readErr == io.EOF { break } - return 0, version, fmt.Errorf("read needle header at offset %d: %w", offset, readErr) + return 0, 0, version, fmt.Errorf("read needle header at offset %d: %w", offset, readErr) } // EC encoding zero-pads the tail of the last block row. An all-zero // header marks the start of that padding, i.e. the end of real needles. @@ -491,13 +631,13 @@ func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle } if n.Size.IsValid() { if pe := nm.Set(n.Id, types.ToOffset(offset), n.Size); pe != nil { - return 0, version, fmt.Errorf("set needle %d: %w", n.Id, pe) + return 0, 0, version, fmt.Errorf("set needle %d: %w", n.Id, pe) } } else { // Deleted/invalid: drop it so the .ecx carries only live entries, // matching the encode-time WriteSortedFileFromIdx behavior. if pe := nm.Delete(n.Id); pe != nil { - return 0, version, fmt.Errorf("delete needle %d: %w", n.Id, pe) + return 0, 0, version, fmt.Errorf("delete needle %d: %w", n.Id, pe) } } offset += types.NeedleHeaderSize + rest @@ -506,16 +646,17 @@ func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle ecxFile, err := os.OpenFile(ecxPath, os.O_TRUNC|os.O_CREATE|os.O_WRONLY, 0644) if err != nil { - return 0, version, fmt.Errorf("open %s: %w", ecxPath, err) + return 0, 0, version, fmt.Errorf("open %s: %w", ecxPath, err) } defer ecxFile.Close() if err := nm.AscendingVisit(func(value needle_map.NeedleValue) error { + liveNeedles++ _, writeErr := ecxFile.Write(value.ToBytes()) return writeErr }); err != nil { - return 0, version, fmt.Errorf("write %s: %w", ecxPath, err) + return 0, 0, version, fmt.Errorf("write %s: %w", ecxPath, err) } - return datFileSize, version, nil + return datFileSize, liveNeedles, version, nil } diff --git a/weed/command/fix_ecx_test.go b/weed/command/fix_ecx_test.go index d99f82694..c9509b0eb 100644 --- a/weed/command/fix_ecx_test.go +++ b/weed/command/fix_ecx_test.go @@ -17,11 +17,12 @@ import ( "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" ) -// buildAndEncodeTestEcVolume writes a small volume (.dat + .idx), EC-encodes it -// into .ec00..ec13, and produces the canonical sorted .ecx. It returns the base -// path, the canonical .ecx bytes, and the original .dat size. A couple of -// needles are deleted so the .ecx must exclude them (live entries only). -func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string, canonicalEcx []byte, origDatSize int64) { +// buildAndEncodeTestEcVolume writes a volume of needleCount needles of about +// needleSize bytes (.dat + .idx), EC-encodes it into .ec00..ec13, and produces +// the canonical sorted .ecx. It returns the base path, the canonical .ecx +// bytes, the original .dat size, and the encode's bitrot protection. A couple +// of needles are deleted so the .ecx must exclude them (live entries only). +func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string, needleCount, needleSize int) (base string, canonicalEcx []byte, origDatSize int64, prot *volume_server_pb.EcBitrotProtection) { t.Helper() base = filepath.Join(dir, baseName) version := needle.GetCurrentVersion() @@ -41,10 +42,10 @@ func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string } nm := needle_map.NewMemDb() - for i := uint64(1); i <= 12; i++ { + for i := uint64(1); i <= uint64(needleCount); i++ { n := new(needle.Needle) n.Id = types.Uint64ToNeedleId(i) - n.Data = make([]byte, 200+int(i)) + n.Data = make([]byte, needleSize+int(i)) rand.Read(n.Data) n.Checksum = needle.NewCRC(n.Data) offset, _, _, err := n.Append(datBackend, version) @@ -92,7 +93,8 @@ func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string idxFile.Close() nm.Close() - if _, err := erasure_coding.WriteEcFiles(base, erasure_coding.BackgroundECContext()); err != nil { + prot, err = erasure_coding.WriteEcFiles(base, erasure_coding.BackgroundECContext()) + if err != nil { t.Fatalf("WriteEcFiles: %v", err) } if err := erasure_coding.WriteSortedFileFromIdx(base, ".ecx"); err != nil { @@ -105,7 +107,7 @@ func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string if len(canonicalEcx) == 0 { t.Fatal("canonical .ecx is empty") } - return base, canonicalEcx, origDatSize + return base, canonicalEcx, origDatSize, prot } // TestFixEcxFromShards verifies the .ecx and .vif are rebuilt purely from the @@ -121,7 +123,7 @@ func TestFixEcxFromShards(t *testing.T) { dir := t.TempDir() const volumeId = 7 - base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "7") + base, canonical, origDatSize, _ := buildAndEncodeTestEcVolume(t, dir, "7", 12, 200) // Disaster: keep only the shards. for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif"} { @@ -175,7 +177,7 @@ func TestFixEcxFromShardsWithVif(t *testing.T) { dir := t.TempDir() const volumeId = 9 - base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "9") + base, canonical, origDatSize, _ := buildAndEncodeTestEcVolume(t, dir, "9", 12, 200) // Write a .vif as the volume server would after encoding. if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ @@ -234,7 +236,7 @@ func TestFixEcxFromShardsMissingShards(t *testing.T) { dir := t.TempDir() const volumeId = 11 - base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "11") + base, canonical, origDatSize, _ := buildAndEncodeTestEcVolume(t, dir, "11", 12, 200) // Lose every index/metadata file plus three shards (two data: .ec02, .ec05; // one parity: .ec11), keeping 11 of 14 — enough to reconstruct. The highest @@ -271,3 +273,60 @@ func TestFixEcxFromShardsMissingShards(t *testing.T) { t.Fatalf("dat size = %d, want %d", vi.GetDatFileSize(), origDatSize) } } + +// TestFixEcxFromShardsUniformLayout loses the .vif of a volume big enough that +// the uniform and legacy layouts place bytes differently, and verifies the +// layout is recovered — from the bitrot sidecar when it survives, and by +// de-striping under both candidate layouts when it does not. +func TestFixEcxFromShardsUniformLayout(t *testing.T) { + oldData, oldParity := *fixEcDataShards, *fixEcParityShards + *fixEcDataShards, *fixEcParityShards = 0, 0 + *fixIgnoreError = true + t.Cleanup(func() { + *fixIgnoreError = false + *fixEcDataShards, *fixEcParityShards = oldData, oldParity + }) + + for _, withSidecar := range []bool{true, false} { + name := "from-sidecar" + if !withSidecar { + name = "dual-scan" + } + t.Run(name, func(t *testing.T) { + dir := t.TempDir() + const volumeId = 13 + base, canonical, _, prot := buildAndEncodeTestEcVolume(t, dir, "13", 13, 2*1024*1024) + if got := prot.GetEcShardConfig().GetBlockSize(); got <= erasure_coding.ErasureCodingSmallBlockSize { + t.Fatalf("fixture block size %d does not diverge from the legacy layout", got) + } + if withSidecar { + if err := erasure_coding.SaveBitrotSidecar(erasure_coding.BitrotSidecarPath(base, 0), prot); err != nil { + t.Fatal(err) + } + } + + for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif"} { + if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) { + t.Fatalf("remove %s: %v", base+ext, err) + } + } + + doFixEcxFromShards(dir, "13", "", volumeId) + + recovered, err := os.ReadFile(base + ".ecx") + if err != nil { + t.Fatalf("recovered .ecx not written: %v", err) + } + if !bytes.Equal(canonical, recovered) { + t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered)) + } + vi, _, found, err := volume_info.MaybeLoadVolumeInfo(base + ".vif") + if err != nil || !found { + t.Fatalf(".vif not regenerated: found=%v err=%v", found, err) + } + if got, want := vi.GetEcShardConfig().GetBlockSize(), prot.GetEcShardConfig().GetBlockSize(); got != want { + t.Fatalf("regenerated .vif block size = %d, want %d", got, want) + } + }) + } +} diff --git a/weed/pb/volume_server.proto b/weed/pb/volume_server.proto index d44712c3f..4962a4e9f 100644 --- a/weed/pb/volume_server.proto +++ b/weed/pb/volume_server.proto @@ -541,6 +541,7 @@ message VolumeEcShardsInfoResponse { uint64 volume_size = 2; uint64 file_count = 3; uint64 file_deleted_count = 4; + EcShardConfig ec_shard_config = 10; // the layout this holder serves reads through; a binary predating a field reports it as unset } message EcShardInfo { @@ -615,6 +616,7 @@ message EcShardConfig { uint32 data_shards = 1; // Number of data shards (e.g., 10) uint32 parity_shards = 2; // Number of parity shards (e.g., 4) int64 encode_ts_ns = 3; // encode time (unix nanos); a read served from a shard of a different encode run is rejected + int64 block_size = 4; // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout } // EcBitrotProtection is the entire content of a bitrot checksum sidecar diff --git a/weed/pb/volume_server_pb/volume_server.pb.go b/weed/pb/volume_server_pb/volume_server.pb.go index afe59bc08..24ad97778 100644 --- a/weed/pb/volume_server_pb/volume_server.pb.go +++ b/weed/pb/volume_server_pb/volume_server.pb.go @@ -4448,6 +4448,7 @@ type VolumeEcShardsInfoResponse struct { VolumeSize uint64 `protobuf:"varint,2,opt,name=volume_size,json=volumeSize,proto3" json:"volume_size,omitempty"` FileCount uint64 `protobuf:"varint,3,opt,name=file_count,json=fileCount,proto3" json:"file_count,omitempty"` FileDeletedCount uint64 `protobuf:"varint,4,opt,name=file_deleted_count,json=fileDeletedCount,proto3" json:"file_deleted_count,omitempty"` + EcShardConfig *EcShardConfig `protobuf:"bytes,10,opt,name=ec_shard_config,json=ecShardConfig,proto3" json:"ec_shard_config,omitempty"` // the layout this holder serves reads through; a binary predating a field reports it as unset unknownFields protoimpl.UnknownFields sizeCache protoimpl.SizeCache } @@ -4510,6 +4511,13 @@ func (x *VolumeEcShardsInfoResponse) GetFileDeletedCount() uint64 { return 0 } +func (x *VolumeEcShardsInfoResponse) GetEcShardConfig() *EcShardConfig { + if x != nil { + return x.EcShardConfig + } + return nil +} + type EcShardInfo struct { state protoimpl.MessageState `protogen:"open.v1"` ShardId uint32 `protobuf:"varint,1,opt,name=shard_id,json=shardId,proto3" json:"shard_id,omitempty"` @@ -5145,6 +5153,7 @@ type EcShardConfig struct { DataShards uint32 `protobuf:"varint,1,opt,name=data_shards,json=dataShards,proto3" json:"data_shards,omitempty"` // Number of data shards (e.g., 10) ParityShards uint32 `protobuf:"varint,2,opt,name=parity_shards,json=parityShards,proto3" json:"parity_shards,omitempty"` // Number of parity shards (e.g., 4) EncodeTsNs int64 `protobuf:"varint,3,opt,name=encode_ts_ns,json=encodeTsNs,proto3" json:"encode_ts_ns,omitempty"` // encode time (unix nanos); a read served from a shard of a different encode run is rejected + BlockSize int64 `protobuf:"varint,4,opt,name=block_size,json=blockSize,proto3" json:"block_size,omitempty"` // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout unknownFields protoimpl.UnknownFields sizeCache protoimpl.SizeCache } @@ -5200,6 +5209,13 @@ func (x *EcShardConfig) GetEncodeTsNs() int64 { return 0 } +func (x *EcShardConfig) GetBlockSize() int64 { + if x != nil { + return x.BlockSize + } + return 0 +} + // EcBitrotProtection is the entire content of a bitrot checksum sidecar // (.ecsum for the legacy generation, .ecsum.v for vacuum // generation N). On disk it is wrapped in a fixed header carrying a CRC32C @@ -7509,14 +7525,16 @@ const file_volume_server_proto_rawDesc = "" + "\tdisk_type\x18\x04 \x01(\tR\bdiskType\" \n" + "\x1eVolumeEcShardsToVolumeResponse\"8\n" + "\x19VolumeEcShardsInfoRequest\x12\x1b\n" + - "\tvolume_id\x18\x01 \x01(\rR\bvolumeId\"\xcf\x01\n" + + "\tvolume_id\x18\x01 \x01(\rR\bvolumeId\"\x98\x02\n" + "\x1aVolumeEcShardsInfoResponse\x12C\n" + "\x0eec_shard_infos\x18\x01 \x03(\v2\x1d.volume_server_pb.EcShardInfoR\fecShardInfos\x12\x1f\n" + "\vvolume_size\x18\x02 \x01(\x04R\n" + "volumeSize\x12\x1d\n" + "\n" + "file_count\x18\x03 \x01(\x04R\tfileCount\x12,\n" + - "\x12file_deleted_count\x18\x04 \x01(\x04R\x10fileDeletedCount\"y\n" + + "\x12file_deleted_count\x18\x04 \x01(\x04R\x10fileDeletedCount\x12G\n" + + "\x0fec_shard_config\x18\n" + + " \x01(\v2\x1f.volume_server_pb.EcShardConfigR\recShardConfig\"y\n" + "\vEcShardInfo\x12\x19\n" + "\bshard_id\x18\x01 \x01(\rR\ashardId\x12\x12\n" + "\x04size\x18\x02 \x01(\x03R\x04size\x12\x1e\n" + @@ -7583,13 +7601,15 @@ const file_volume_server_proto_rawDesc = "" + "\rexpire_at_sec\x18\x06 \x01(\x04R\vexpireAtSec\x12\x1b\n" + "\tread_only\x18\a \x01(\bR\breadOnly\x12G\n" + "\x0fec_shard_config\x18\b \x01(\v2\x1f.volume_server_pb.EcShardConfigR\recShardConfig\x12/\n" + - "\x14read_only_can_delete\x18\t \x01(\bR\x11readOnlyCanDelete\"w\n" + + "\x14read_only_can_delete\x18\t \x01(\bR\x11readOnlyCanDelete\"\x96\x01\n" + "\rEcShardConfig\x12\x1f\n" + "\vdata_shards\x18\x01 \x01(\rR\n" + "dataShards\x12#\n" + "\rparity_shards\x18\x02 \x01(\rR\fparityShards\x12 \n" + "\fencode_ts_ns\x18\x03 \x01(\x03R\n" + - "encodeTsNs\"\xbc\x02\n" + + "encodeTsNs\x12\x1d\n" + + "\n" + + "block_size\x18\x04 \x01(\x03R\tblockSize\"\xbc\x02\n" + "\x12EcBitrotProtection\x12A\n" + "\talgorithm\x18\x01 \x01(\x0e2#.volume_server_pb.ChecksumAlgorithmR\talgorithm\x12\x1d\n" + "\n" + @@ -7958,133 +7978,134 @@ var file_volume_server_proto_depIdxs = []int32{ 2, // 3: volume_server_pb.SetStateResponse.state:type_name -> volume_server_pb.VolumeServerState 48, // 4: volume_server_pb.ReceiveFileRequest.info:type_name -> volume_server_pb.ReceiveFileInfo 82, // 5: volume_server_pb.VolumeEcShardsInfoResponse.ec_shard_infos:type_name -> volume_server_pb.EcShardInfo - 88, // 6: volume_server_pb.ReadVolumeFileStatusResponse.volume_info:type_name -> volume_server_pb.VolumeInfo - 87, // 7: volume_server_pb.VolumeInfo.files:type_name -> volume_server_pb.RemoteFile - 89, // 8: volume_server_pb.VolumeInfo.ec_shard_config:type_name -> volume_server_pb.EcShardConfig - 0, // 9: volume_server_pb.EcBitrotProtection.algorithm:type_name -> volume_server_pb.ChecksumAlgorithm - 89, // 10: volume_server_pb.EcBitrotProtection.ec_shard_config:type_name -> volume_server_pb.EcShardConfig - 91, // 11: volume_server_pb.EcBitrotProtection.shards:type_name -> volume_server_pb.EcShardChecksums - 87, // 12: volume_server_pb.OldVersionVolumeInfo.files:type_name -> volume_server_pb.RemoteFile - 85, // 13: volume_server_pb.VolumeServerStatusResponse.disk_statuses:type_name -> volume_server_pb.DiskStatus - 86, // 14: volume_server_pb.VolumeServerStatusResponse.memory_status:type_name -> volume_server_pb.MemStatus - 2, // 15: volume_server_pb.VolumeServerStatusResponse.state:type_name -> volume_server_pb.VolumeServerState - 113, // 16: volume_server_pb.FetchAndWriteNeedleRequest.replicas:type_name -> volume_server_pb.FetchAndWriteNeedleRequest.Replica - 122, // 17: volume_server_pb.FetchAndWriteNeedleRequest.remote_conf:type_name -> remote_pb.RemoteConf - 123, // 18: volume_server_pb.FetchAndWriteNeedleRequest.remote_location:type_name -> remote_pb.RemoteStorageLocation - 1, // 19: volume_server_pb.ScrubVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode - 1, // 20: volume_server_pb.ScrubEcVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode - 82, // 21: volume_server_pb.ScrubEcVolumeResponse.broken_shard_infos:type_name -> volume_server_pb.EcShardInfo - 114, // 22: volume_server_pb.QueryRequest.filter:type_name -> volume_server_pb.QueryRequest.Filter - 115, // 23: volume_server_pb.QueryRequest.input_serialization:type_name -> volume_server_pb.QueryRequest.InputSerialization - 116, // 24: volume_server_pb.QueryRequest.output_serialization:type_name -> volume_server_pb.QueryRequest.OutputSerialization - 117, // 25: volume_server_pb.QueryRequest.InputSerialization.csv_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.CSVInput - 118, // 26: volume_server_pb.QueryRequest.InputSerialization.json_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.JSONInput - 119, // 27: volume_server_pb.QueryRequest.InputSerialization.parquet_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.ParquetInput - 120, // 28: volume_server_pb.QueryRequest.OutputSerialization.csv_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.CSVOutput - 121, // 29: volume_server_pb.QueryRequest.OutputSerialization.json_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.JSONOutput - 3, // 30: volume_server_pb.VolumeServer.BatchDelete:input_type -> volume_server_pb.BatchDeleteRequest - 7, // 31: volume_server_pb.VolumeServer.VacuumVolumeCheck:input_type -> volume_server_pb.VacuumVolumeCheckRequest - 9, // 32: volume_server_pb.VolumeServer.VacuumVolumeCompact:input_type -> volume_server_pb.VacuumVolumeCompactRequest - 11, // 33: volume_server_pb.VolumeServer.VacuumVolumeCommit:input_type -> volume_server_pb.VacuumVolumeCommitRequest - 13, // 34: volume_server_pb.VolumeServer.VacuumVolumeCleanup:input_type -> volume_server_pb.VacuumVolumeCleanupRequest - 15, // 35: volume_server_pb.VolumeServer.DeleteCollection:input_type -> volume_server_pb.DeleteCollectionRequest - 17, // 36: volume_server_pb.VolumeServer.AllocateVolume:input_type -> volume_server_pb.AllocateVolumeRequest - 19, // 37: volume_server_pb.VolumeServer.VolumeSyncStatus:input_type -> volume_server_pb.VolumeSyncStatusRequest - 21, // 38: volume_server_pb.VolumeServer.VolumeIncrementalCopy:input_type -> volume_server_pb.VolumeIncrementalCopyRequest - 23, // 39: volume_server_pb.VolumeServer.VolumeMount:input_type -> volume_server_pb.VolumeMountRequest - 25, // 40: volume_server_pb.VolumeServer.VolumeUnmount:input_type -> volume_server_pb.VolumeUnmountRequest - 27, // 41: volume_server_pb.VolumeServer.VolumeConsolidateIndex:input_type -> volume_server_pb.VolumeConsolidateIndexRequest - 29, // 42: volume_server_pb.VolumeServer.VolumeDelete:input_type -> volume_server_pb.VolumeDeleteRequest - 31, // 43: volume_server_pb.VolumeServer.VolumeMarkReadonly:input_type -> volume_server_pb.VolumeMarkReadonlyRequest - 33, // 44: volume_server_pb.VolumeServer.VolumeMarkWritable:input_type -> volume_server_pb.VolumeMarkWritableRequest - 35, // 45: volume_server_pb.VolumeServer.VolumeConfigure:input_type -> volume_server_pb.VolumeConfigureRequest - 37, // 46: volume_server_pb.VolumeServer.VolumeStatus:input_type -> volume_server_pb.VolumeStatusRequest - 39, // 47: volume_server_pb.VolumeServer.GetState:input_type -> volume_server_pb.GetStateRequest - 41, // 48: volume_server_pb.VolumeServer.SetState:input_type -> volume_server_pb.SetStateRequest - 43, // 49: volume_server_pb.VolumeServer.VolumeCopy:input_type -> volume_server_pb.VolumeCopyRequest - 83, // 50: volume_server_pb.VolumeServer.ReadVolumeFileStatus:input_type -> volume_server_pb.ReadVolumeFileStatusRequest - 45, // 51: volume_server_pb.VolumeServer.CopyFile:input_type -> volume_server_pb.CopyFileRequest - 47, // 52: volume_server_pb.VolumeServer.ReceiveFile:input_type -> volume_server_pb.ReceiveFileRequest - 50, // 53: volume_server_pb.VolumeServer.ReadNeedleBlob:input_type -> volume_server_pb.ReadNeedleBlobRequest - 52, // 54: volume_server_pb.VolumeServer.ReadNeedleMeta:input_type -> volume_server_pb.ReadNeedleMetaRequest - 54, // 55: volume_server_pb.VolumeServer.WriteNeedleBlob:input_type -> volume_server_pb.WriteNeedleBlobRequest - 56, // 56: volume_server_pb.VolumeServer.ReadAllNeedles:input_type -> volume_server_pb.ReadAllNeedlesRequest - 58, // 57: volume_server_pb.VolumeServer.VolumeTailSender:input_type -> volume_server_pb.VolumeTailSenderRequest - 60, // 58: volume_server_pb.VolumeServer.VolumeTailReceiver:input_type -> volume_server_pb.VolumeTailReceiverRequest - 62, // 59: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:input_type -> volume_server_pb.VolumeEcShardsGenerateRequest - 64, // 60: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:input_type -> volume_server_pb.VolumeEcShardsRebuildRequest - 66, // 61: volume_server_pb.VolumeServer.VolumeEcShardsCopy:input_type -> volume_server_pb.VolumeEcShardsCopyRequest - 68, // 62: volume_server_pb.VolumeServer.VolumeEcShardsDelete:input_type -> volume_server_pb.VolumeEcShardsDeleteRequest - 70, // 63: volume_server_pb.VolumeServer.VolumeEcShardsMount:input_type -> volume_server_pb.VolumeEcShardsMountRequest - 72, // 64: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:input_type -> volume_server_pb.VolumeEcShardsUnmountRequest - 74, // 65: volume_server_pb.VolumeServer.VolumeEcShardRead:input_type -> volume_server_pb.VolumeEcShardReadRequest - 76, // 66: volume_server_pb.VolumeServer.VolumeEcBlobDelete:input_type -> volume_server_pb.VolumeEcBlobDeleteRequest - 78, // 67: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:input_type -> volume_server_pb.VolumeEcShardsToVolumeRequest - 80, // 68: volume_server_pb.VolumeServer.VolumeEcShardsInfo:input_type -> volume_server_pb.VolumeEcShardsInfoRequest - 93, // 69: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:input_type -> volume_server_pb.VolumeTierMoveDatToRemoteRequest - 95, // 70: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:input_type -> volume_server_pb.VolumeTierMoveDatFromRemoteRequest - 97, // 71: volume_server_pb.VolumeServer.VolumeServerStatus:input_type -> volume_server_pb.VolumeServerStatusRequest - 99, // 72: volume_server_pb.VolumeServer.VolumeServerLeave:input_type -> volume_server_pb.VolumeServerLeaveRequest - 101, // 73: volume_server_pb.VolumeServer.FetchAndWriteNeedle:input_type -> volume_server_pb.FetchAndWriteNeedleRequest - 103, // 74: volume_server_pb.VolumeServer.ScrubVolume:input_type -> volume_server_pb.ScrubVolumeRequest - 105, // 75: volume_server_pb.VolumeServer.ScrubEcVolume:input_type -> volume_server_pb.ScrubEcVolumeRequest - 107, // 76: volume_server_pb.VolumeServer.Query:input_type -> volume_server_pb.QueryRequest - 109, // 77: volume_server_pb.VolumeServer.VolumeNeedleStatus:input_type -> volume_server_pb.VolumeNeedleStatusRequest - 111, // 78: volume_server_pb.VolumeServer.Ping:input_type -> volume_server_pb.PingRequest - 4, // 79: volume_server_pb.VolumeServer.BatchDelete:output_type -> volume_server_pb.BatchDeleteResponse - 8, // 80: volume_server_pb.VolumeServer.VacuumVolumeCheck:output_type -> volume_server_pb.VacuumVolumeCheckResponse - 10, // 81: volume_server_pb.VolumeServer.VacuumVolumeCompact:output_type -> volume_server_pb.VacuumVolumeCompactResponse - 12, // 82: volume_server_pb.VolumeServer.VacuumVolumeCommit:output_type -> volume_server_pb.VacuumVolumeCommitResponse - 14, // 83: volume_server_pb.VolumeServer.VacuumVolumeCleanup:output_type -> volume_server_pb.VacuumVolumeCleanupResponse - 16, // 84: volume_server_pb.VolumeServer.DeleteCollection:output_type -> volume_server_pb.DeleteCollectionResponse - 18, // 85: volume_server_pb.VolumeServer.AllocateVolume:output_type -> volume_server_pb.AllocateVolumeResponse - 20, // 86: volume_server_pb.VolumeServer.VolumeSyncStatus:output_type -> volume_server_pb.VolumeSyncStatusResponse - 22, // 87: volume_server_pb.VolumeServer.VolumeIncrementalCopy:output_type -> volume_server_pb.VolumeIncrementalCopyResponse - 24, // 88: volume_server_pb.VolumeServer.VolumeMount:output_type -> volume_server_pb.VolumeMountResponse - 26, // 89: volume_server_pb.VolumeServer.VolumeUnmount:output_type -> volume_server_pb.VolumeUnmountResponse - 28, // 90: volume_server_pb.VolumeServer.VolumeConsolidateIndex:output_type -> volume_server_pb.VolumeConsolidateIndexResponse - 30, // 91: volume_server_pb.VolumeServer.VolumeDelete:output_type -> volume_server_pb.VolumeDeleteResponse - 32, // 92: volume_server_pb.VolumeServer.VolumeMarkReadonly:output_type -> volume_server_pb.VolumeMarkReadonlyResponse - 34, // 93: volume_server_pb.VolumeServer.VolumeMarkWritable:output_type -> volume_server_pb.VolumeMarkWritableResponse - 36, // 94: volume_server_pb.VolumeServer.VolumeConfigure:output_type -> volume_server_pb.VolumeConfigureResponse - 38, // 95: volume_server_pb.VolumeServer.VolumeStatus:output_type -> volume_server_pb.VolumeStatusResponse - 40, // 96: volume_server_pb.VolumeServer.GetState:output_type -> volume_server_pb.GetStateResponse - 42, // 97: volume_server_pb.VolumeServer.SetState:output_type -> volume_server_pb.SetStateResponse - 44, // 98: volume_server_pb.VolumeServer.VolumeCopy:output_type -> volume_server_pb.VolumeCopyResponse - 84, // 99: volume_server_pb.VolumeServer.ReadVolumeFileStatus:output_type -> volume_server_pb.ReadVolumeFileStatusResponse - 46, // 100: volume_server_pb.VolumeServer.CopyFile:output_type -> volume_server_pb.CopyFileResponse - 49, // 101: volume_server_pb.VolumeServer.ReceiveFile:output_type -> volume_server_pb.ReceiveFileResponse - 51, // 102: volume_server_pb.VolumeServer.ReadNeedleBlob:output_type -> volume_server_pb.ReadNeedleBlobResponse - 53, // 103: volume_server_pb.VolumeServer.ReadNeedleMeta:output_type -> volume_server_pb.ReadNeedleMetaResponse - 55, // 104: volume_server_pb.VolumeServer.WriteNeedleBlob:output_type -> volume_server_pb.WriteNeedleBlobResponse - 57, // 105: volume_server_pb.VolumeServer.ReadAllNeedles:output_type -> volume_server_pb.ReadAllNeedlesResponse - 59, // 106: volume_server_pb.VolumeServer.VolumeTailSender:output_type -> volume_server_pb.VolumeTailSenderResponse - 61, // 107: volume_server_pb.VolumeServer.VolumeTailReceiver:output_type -> volume_server_pb.VolumeTailReceiverResponse - 63, // 108: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:output_type -> volume_server_pb.VolumeEcShardsGenerateResponse - 65, // 109: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:output_type -> volume_server_pb.VolumeEcShardsRebuildResponse - 67, // 110: volume_server_pb.VolumeServer.VolumeEcShardsCopy:output_type -> volume_server_pb.VolumeEcShardsCopyResponse - 69, // 111: volume_server_pb.VolumeServer.VolumeEcShardsDelete:output_type -> volume_server_pb.VolumeEcShardsDeleteResponse - 71, // 112: volume_server_pb.VolumeServer.VolumeEcShardsMount:output_type -> volume_server_pb.VolumeEcShardsMountResponse - 73, // 113: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:output_type -> volume_server_pb.VolumeEcShardsUnmountResponse - 75, // 114: volume_server_pb.VolumeServer.VolumeEcShardRead:output_type -> volume_server_pb.VolumeEcShardReadResponse - 77, // 115: volume_server_pb.VolumeServer.VolumeEcBlobDelete:output_type -> volume_server_pb.VolumeEcBlobDeleteResponse - 79, // 116: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:output_type -> volume_server_pb.VolumeEcShardsToVolumeResponse - 81, // 117: volume_server_pb.VolumeServer.VolumeEcShardsInfo:output_type -> volume_server_pb.VolumeEcShardsInfoResponse - 94, // 118: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:output_type -> volume_server_pb.VolumeTierMoveDatToRemoteResponse - 96, // 119: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:output_type -> volume_server_pb.VolumeTierMoveDatFromRemoteResponse - 98, // 120: volume_server_pb.VolumeServer.VolumeServerStatus:output_type -> volume_server_pb.VolumeServerStatusResponse - 100, // 121: volume_server_pb.VolumeServer.VolumeServerLeave:output_type -> volume_server_pb.VolumeServerLeaveResponse - 102, // 122: volume_server_pb.VolumeServer.FetchAndWriteNeedle:output_type -> volume_server_pb.FetchAndWriteNeedleResponse - 104, // 123: volume_server_pb.VolumeServer.ScrubVolume:output_type -> volume_server_pb.ScrubVolumeResponse - 106, // 124: volume_server_pb.VolumeServer.ScrubEcVolume:output_type -> volume_server_pb.ScrubEcVolumeResponse - 108, // 125: volume_server_pb.VolumeServer.Query:output_type -> volume_server_pb.QueriedStripe - 110, // 126: volume_server_pb.VolumeServer.VolumeNeedleStatus:output_type -> volume_server_pb.VolumeNeedleStatusResponse - 112, // 127: volume_server_pb.VolumeServer.Ping:output_type -> volume_server_pb.PingResponse - 79, // [79:128] is the sub-list for method output_type - 30, // [30:79] is the sub-list for method input_type - 30, // [30:30] is the sub-list for extension type_name - 30, // [30:30] is the sub-list for extension extendee - 0, // [0:30] is the sub-list for field type_name + 89, // 6: volume_server_pb.VolumeEcShardsInfoResponse.ec_shard_config:type_name -> volume_server_pb.EcShardConfig + 88, // 7: volume_server_pb.ReadVolumeFileStatusResponse.volume_info:type_name -> volume_server_pb.VolumeInfo + 87, // 8: volume_server_pb.VolumeInfo.files:type_name -> volume_server_pb.RemoteFile + 89, // 9: volume_server_pb.VolumeInfo.ec_shard_config:type_name -> volume_server_pb.EcShardConfig + 0, // 10: volume_server_pb.EcBitrotProtection.algorithm:type_name -> volume_server_pb.ChecksumAlgorithm + 89, // 11: volume_server_pb.EcBitrotProtection.ec_shard_config:type_name -> volume_server_pb.EcShardConfig + 91, // 12: volume_server_pb.EcBitrotProtection.shards:type_name -> volume_server_pb.EcShardChecksums + 87, // 13: volume_server_pb.OldVersionVolumeInfo.files:type_name -> volume_server_pb.RemoteFile + 85, // 14: volume_server_pb.VolumeServerStatusResponse.disk_statuses:type_name -> volume_server_pb.DiskStatus + 86, // 15: volume_server_pb.VolumeServerStatusResponse.memory_status:type_name -> volume_server_pb.MemStatus + 2, // 16: volume_server_pb.VolumeServerStatusResponse.state:type_name -> volume_server_pb.VolumeServerState + 113, // 17: volume_server_pb.FetchAndWriteNeedleRequest.replicas:type_name -> volume_server_pb.FetchAndWriteNeedleRequest.Replica + 122, // 18: volume_server_pb.FetchAndWriteNeedleRequest.remote_conf:type_name -> remote_pb.RemoteConf + 123, // 19: volume_server_pb.FetchAndWriteNeedleRequest.remote_location:type_name -> remote_pb.RemoteStorageLocation + 1, // 20: volume_server_pb.ScrubVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode + 1, // 21: volume_server_pb.ScrubEcVolumeRequest.mode:type_name -> volume_server_pb.VolumeScrubMode + 82, // 22: volume_server_pb.ScrubEcVolumeResponse.broken_shard_infos:type_name -> volume_server_pb.EcShardInfo + 114, // 23: volume_server_pb.QueryRequest.filter:type_name -> volume_server_pb.QueryRequest.Filter + 115, // 24: volume_server_pb.QueryRequest.input_serialization:type_name -> volume_server_pb.QueryRequest.InputSerialization + 116, // 25: volume_server_pb.QueryRequest.output_serialization:type_name -> volume_server_pb.QueryRequest.OutputSerialization + 117, // 26: volume_server_pb.QueryRequest.InputSerialization.csv_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.CSVInput + 118, // 27: volume_server_pb.QueryRequest.InputSerialization.json_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.JSONInput + 119, // 28: volume_server_pb.QueryRequest.InputSerialization.parquet_input:type_name -> volume_server_pb.QueryRequest.InputSerialization.ParquetInput + 120, // 29: volume_server_pb.QueryRequest.OutputSerialization.csv_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.CSVOutput + 121, // 30: volume_server_pb.QueryRequest.OutputSerialization.json_output:type_name -> volume_server_pb.QueryRequest.OutputSerialization.JSONOutput + 3, // 31: volume_server_pb.VolumeServer.BatchDelete:input_type -> volume_server_pb.BatchDeleteRequest + 7, // 32: volume_server_pb.VolumeServer.VacuumVolumeCheck:input_type -> volume_server_pb.VacuumVolumeCheckRequest + 9, // 33: volume_server_pb.VolumeServer.VacuumVolumeCompact:input_type -> volume_server_pb.VacuumVolumeCompactRequest + 11, // 34: volume_server_pb.VolumeServer.VacuumVolumeCommit:input_type -> volume_server_pb.VacuumVolumeCommitRequest + 13, // 35: volume_server_pb.VolumeServer.VacuumVolumeCleanup:input_type -> volume_server_pb.VacuumVolumeCleanupRequest + 15, // 36: volume_server_pb.VolumeServer.DeleteCollection:input_type -> volume_server_pb.DeleteCollectionRequest + 17, // 37: volume_server_pb.VolumeServer.AllocateVolume:input_type -> volume_server_pb.AllocateVolumeRequest + 19, // 38: volume_server_pb.VolumeServer.VolumeSyncStatus:input_type -> volume_server_pb.VolumeSyncStatusRequest + 21, // 39: volume_server_pb.VolumeServer.VolumeIncrementalCopy:input_type -> volume_server_pb.VolumeIncrementalCopyRequest + 23, // 40: volume_server_pb.VolumeServer.VolumeMount:input_type -> volume_server_pb.VolumeMountRequest + 25, // 41: volume_server_pb.VolumeServer.VolumeUnmount:input_type -> volume_server_pb.VolumeUnmountRequest + 27, // 42: volume_server_pb.VolumeServer.VolumeConsolidateIndex:input_type -> volume_server_pb.VolumeConsolidateIndexRequest + 29, // 43: volume_server_pb.VolumeServer.VolumeDelete:input_type -> volume_server_pb.VolumeDeleteRequest + 31, // 44: volume_server_pb.VolumeServer.VolumeMarkReadonly:input_type -> volume_server_pb.VolumeMarkReadonlyRequest + 33, // 45: volume_server_pb.VolumeServer.VolumeMarkWritable:input_type -> volume_server_pb.VolumeMarkWritableRequest + 35, // 46: volume_server_pb.VolumeServer.VolumeConfigure:input_type -> volume_server_pb.VolumeConfigureRequest + 37, // 47: volume_server_pb.VolumeServer.VolumeStatus:input_type -> volume_server_pb.VolumeStatusRequest + 39, // 48: volume_server_pb.VolumeServer.GetState:input_type -> volume_server_pb.GetStateRequest + 41, // 49: volume_server_pb.VolumeServer.SetState:input_type -> volume_server_pb.SetStateRequest + 43, // 50: volume_server_pb.VolumeServer.VolumeCopy:input_type -> volume_server_pb.VolumeCopyRequest + 83, // 51: volume_server_pb.VolumeServer.ReadVolumeFileStatus:input_type -> volume_server_pb.ReadVolumeFileStatusRequest + 45, // 52: volume_server_pb.VolumeServer.CopyFile:input_type -> volume_server_pb.CopyFileRequest + 47, // 53: volume_server_pb.VolumeServer.ReceiveFile:input_type -> volume_server_pb.ReceiveFileRequest + 50, // 54: volume_server_pb.VolumeServer.ReadNeedleBlob:input_type -> volume_server_pb.ReadNeedleBlobRequest + 52, // 55: volume_server_pb.VolumeServer.ReadNeedleMeta:input_type -> volume_server_pb.ReadNeedleMetaRequest + 54, // 56: volume_server_pb.VolumeServer.WriteNeedleBlob:input_type -> volume_server_pb.WriteNeedleBlobRequest + 56, // 57: volume_server_pb.VolumeServer.ReadAllNeedles:input_type -> volume_server_pb.ReadAllNeedlesRequest + 58, // 58: volume_server_pb.VolumeServer.VolumeTailSender:input_type -> volume_server_pb.VolumeTailSenderRequest + 60, // 59: volume_server_pb.VolumeServer.VolumeTailReceiver:input_type -> volume_server_pb.VolumeTailReceiverRequest + 62, // 60: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:input_type -> volume_server_pb.VolumeEcShardsGenerateRequest + 64, // 61: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:input_type -> volume_server_pb.VolumeEcShardsRebuildRequest + 66, // 62: volume_server_pb.VolumeServer.VolumeEcShardsCopy:input_type -> volume_server_pb.VolumeEcShardsCopyRequest + 68, // 63: volume_server_pb.VolumeServer.VolumeEcShardsDelete:input_type -> volume_server_pb.VolumeEcShardsDeleteRequest + 70, // 64: volume_server_pb.VolumeServer.VolumeEcShardsMount:input_type -> volume_server_pb.VolumeEcShardsMountRequest + 72, // 65: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:input_type -> volume_server_pb.VolumeEcShardsUnmountRequest + 74, // 66: volume_server_pb.VolumeServer.VolumeEcShardRead:input_type -> volume_server_pb.VolumeEcShardReadRequest + 76, // 67: volume_server_pb.VolumeServer.VolumeEcBlobDelete:input_type -> volume_server_pb.VolumeEcBlobDeleteRequest + 78, // 68: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:input_type -> volume_server_pb.VolumeEcShardsToVolumeRequest + 80, // 69: volume_server_pb.VolumeServer.VolumeEcShardsInfo:input_type -> volume_server_pb.VolumeEcShardsInfoRequest + 93, // 70: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:input_type -> volume_server_pb.VolumeTierMoveDatToRemoteRequest + 95, // 71: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:input_type -> volume_server_pb.VolumeTierMoveDatFromRemoteRequest + 97, // 72: volume_server_pb.VolumeServer.VolumeServerStatus:input_type -> volume_server_pb.VolumeServerStatusRequest + 99, // 73: volume_server_pb.VolumeServer.VolumeServerLeave:input_type -> volume_server_pb.VolumeServerLeaveRequest + 101, // 74: volume_server_pb.VolumeServer.FetchAndWriteNeedle:input_type -> volume_server_pb.FetchAndWriteNeedleRequest + 103, // 75: volume_server_pb.VolumeServer.ScrubVolume:input_type -> volume_server_pb.ScrubVolumeRequest + 105, // 76: volume_server_pb.VolumeServer.ScrubEcVolume:input_type -> volume_server_pb.ScrubEcVolumeRequest + 107, // 77: volume_server_pb.VolumeServer.Query:input_type -> volume_server_pb.QueryRequest + 109, // 78: volume_server_pb.VolumeServer.VolumeNeedleStatus:input_type -> volume_server_pb.VolumeNeedleStatusRequest + 111, // 79: volume_server_pb.VolumeServer.Ping:input_type -> volume_server_pb.PingRequest + 4, // 80: volume_server_pb.VolumeServer.BatchDelete:output_type -> volume_server_pb.BatchDeleteResponse + 8, // 81: volume_server_pb.VolumeServer.VacuumVolumeCheck:output_type -> volume_server_pb.VacuumVolumeCheckResponse + 10, // 82: volume_server_pb.VolumeServer.VacuumVolumeCompact:output_type -> volume_server_pb.VacuumVolumeCompactResponse + 12, // 83: volume_server_pb.VolumeServer.VacuumVolumeCommit:output_type -> volume_server_pb.VacuumVolumeCommitResponse + 14, // 84: volume_server_pb.VolumeServer.VacuumVolumeCleanup:output_type -> volume_server_pb.VacuumVolumeCleanupResponse + 16, // 85: volume_server_pb.VolumeServer.DeleteCollection:output_type -> volume_server_pb.DeleteCollectionResponse + 18, // 86: volume_server_pb.VolumeServer.AllocateVolume:output_type -> volume_server_pb.AllocateVolumeResponse + 20, // 87: volume_server_pb.VolumeServer.VolumeSyncStatus:output_type -> volume_server_pb.VolumeSyncStatusResponse + 22, // 88: volume_server_pb.VolumeServer.VolumeIncrementalCopy:output_type -> volume_server_pb.VolumeIncrementalCopyResponse + 24, // 89: volume_server_pb.VolumeServer.VolumeMount:output_type -> volume_server_pb.VolumeMountResponse + 26, // 90: volume_server_pb.VolumeServer.VolumeUnmount:output_type -> volume_server_pb.VolumeUnmountResponse + 28, // 91: volume_server_pb.VolumeServer.VolumeConsolidateIndex:output_type -> volume_server_pb.VolumeConsolidateIndexResponse + 30, // 92: volume_server_pb.VolumeServer.VolumeDelete:output_type -> volume_server_pb.VolumeDeleteResponse + 32, // 93: volume_server_pb.VolumeServer.VolumeMarkReadonly:output_type -> volume_server_pb.VolumeMarkReadonlyResponse + 34, // 94: volume_server_pb.VolumeServer.VolumeMarkWritable:output_type -> volume_server_pb.VolumeMarkWritableResponse + 36, // 95: volume_server_pb.VolumeServer.VolumeConfigure:output_type -> volume_server_pb.VolumeConfigureResponse + 38, // 96: volume_server_pb.VolumeServer.VolumeStatus:output_type -> volume_server_pb.VolumeStatusResponse + 40, // 97: volume_server_pb.VolumeServer.GetState:output_type -> volume_server_pb.GetStateResponse + 42, // 98: volume_server_pb.VolumeServer.SetState:output_type -> volume_server_pb.SetStateResponse + 44, // 99: volume_server_pb.VolumeServer.VolumeCopy:output_type -> volume_server_pb.VolumeCopyResponse + 84, // 100: volume_server_pb.VolumeServer.ReadVolumeFileStatus:output_type -> volume_server_pb.ReadVolumeFileStatusResponse + 46, // 101: volume_server_pb.VolumeServer.CopyFile:output_type -> volume_server_pb.CopyFileResponse + 49, // 102: volume_server_pb.VolumeServer.ReceiveFile:output_type -> volume_server_pb.ReceiveFileResponse + 51, // 103: volume_server_pb.VolumeServer.ReadNeedleBlob:output_type -> volume_server_pb.ReadNeedleBlobResponse + 53, // 104: volume_server_pb.VolumeServer.ReadNeedleMeta:output_type -> volume_server_pb.ReadNeedleMetaResponse + 55, // 105: volume_server_pb.VolumeServer.WriteNeedleBlob:output_type -> volume_server_pb.WriteNeedleBlobResponse + 57, // 106: volume_server_pb.VolumeServer.ReadAllNeedles:output_type -> volume_server_pb.ReadAllNeedlesResponse + 59, // 107: volume_server_pb.VolumeServer.VolumeTailSender:output_type -> volume_server_pb.VolumeTailSenderResponse + 61, // 108: volume_server_pb.VolumeServer.VolumeTailReceiver:output_type -> volume_server_pb.VolumeTailReceiverResponse + 63, // 109: volume_server_pb.VolumeServer.VolumeEcShardsGenerate:output_type -> volume_server_pb.VolumeEcShardsGenerateResponse + 65, // 110: volume_server_pb.VolumeServer.VolumeEcShardsRebuild:output_type -> volume_server_pb.VolumeEcShardsRebuildResponse + 67, // 111: volume_server_pb.VolumeServer.VolumeEcShardsCopy:output_type -> volume_server_pb.VolumeEcShardsCopyResponse + 69, // 112: volume_server_pb.VolumeServer.VolumeEcShardsDelete:output_type -> volume_server_pb.VolumeEcShardsDeleteResponse + 71, // 113: volume_server_pb.VolumeServer.VolumeEcShardsMount:output_type -> volume_server_pb.VolumeEcShardsMountResponse + 73, // 114: volume_server_pb.VolumeServer.VolumeEcShardsUnmount:output_type -> volume_server_pb.VolumeEcShardsUnmountResponse + 75, // 115: volume_server_pb.VolumeServer.VolumeEcShardRead:output_type -> volume_server_pb.VolumeEcShardReadResponse + 77, // 116: volume_server_pb.VolumeServer.VolumeEcBlobDelete:output_type -> volume_server_pb.VolumeEcBlobDeleteResponse + 79, // 117: volume_server_pb.VolumeServer.VolumeEcShardsToVolume:output_type -> volume_server_pb.VolumeEcShardsToVolumeResponse + 81, // 118: volume_server_pb.VolumeServer.VolumeEcShardsInfo:output_type -> volume_server_pb.VolumeEcShardsInfoResponse + 94, // 119: volume_server_pb.VolumeServer.VolumeTierMoveDatToRemote:output_type -> volume_server_pb.VolumeTierMoveDatToRemoteResponse + 96, // 120: volume_server_pb.VolumeServer.VolumeTierMoveDatFromRemote:output_type -> volume_server_pb.VolumeTierMoveDatFromRemoteResponse + 98, // 121: volume_server_pb.VolumeServer.VolumeServerStatus:output_type -> volume_server_pb.VolumeServerStatusResponse + 100, // 122: volume_server_pb.VolumeServer.VolumeServerLeave:output_type -> volume_server_pb.VolumeServerLeaveResponse + 102, // 123: volume_server_pb.VolumeServer.FetchAndWriteNeedle:output_type -> volume_server_pb.FetchAndWriteNeedleResponse + 104, // 124: volume_server_pb.VolumeServer.ScrubVolume:output_type -> volume_server_pb.ScrubVolumeResponse + 106, // 125: volume_server_pb.VolumeServer.ScrubEcVolume:output_type -> volume_server_pb.ScrubEcVolumeResponse + 108, // 126: volume_server_pb.VolumeServer.Query:output_type -> volume_server_pb.QueriedStripe + 110, // 127: volume_server_pb.VolumeServer.VolumeNeedleStatus:output_type -> volume_server_pb.VolumeNeedleStatusResponse + 112, // 128: volume_server_pb.VolumeServer.Ping:output_type -> volume_server_pb.PingResponse + 80, // [80:129] is the sub-list for method output_type + 31, // [31:80] is the sub-list for method input_type + 31, // [31:31] is the sub-list for extension type_name + 31, // [31:31] is the sub-list for extension extendee + 0, // [0:31] is the sub-list for field type_name } func init() { file_volume_server_proto_init() } diff --git a/weed/server/volume_grpc_erasure_coding.go b/weed/server/volume_grpc_erasure_coding.go index f4a1d15ca..14c1a4ee2 100644 --- a/weed/server/volume_grpc_erasure_coding.go +++ b/weed/server/volume_grpc_erasure_coding.go @@ -122,9 +122,6 @@ func (vs *VolumeServer) VolumeEcShardsGenerate(ctx context.Context, req *volume_ return nil, fmt.Errorf("WriteSortedFileFromIdx %s: %v", v.IndexFileName(), err) } - // snapshot .dat file size before encoding — must match what .ecx references - datSize, _, _ := v.FileStat() - // write .ec00 ~ .ec[TotalShards-1] files using context ecBitrot, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx) if err != nil { @@ -151,7 +148,10 @@ func (vs *VolumeServer) VolumeEcShardsGenerate(ctx context.Context, req *volume_ } volumeInfo := &volume_server_pb.VolumeInfo{Version: uint32(v.Version())} volumeInfo.ExpireAtSec = expireAtSec - volumeInfo.DatFileSize = int64(datSize) + // The size the encode actually read, not a separate stat: a replica-sync + // write can land between two stats of a live .dat, and the .vif would then + // record a DatFileSize and a BlockSize describing different files. + volumeInfo.DatFileSize = ecCtx.DatFileSize // Validate EC configuration before saving to .vif if ecCtx.DataShards <= 0 || ecCtx.ParityShards <= 0 || ecCtx.Total() > erasure_coding.MaxShardCount { @@ -165,6 +165,7 @@ func (vs *VolumeServer) VolumeEcShardsGenerate(ctx context.Context, req *volume_ DataShards: uint32(ecCtx.DataShards), ParityShards: uint32(ecCtx.ParityShards), EncodeTsNs: time.Now().UnixNano(), + BlockSize: ecCtx.BlockSize, } glog.V(1).Infof("Saving EC config to .vif for volume %d: %d+%d (total: %d)", req.VolumeId, ecCtx.DataShards, ecCtx.ParityShards, ecCtx.Total()) @@ -239,17 +240,22 @@ func (vs *VolumeServer) VolumeEcShardsRebuild(ctx context.Context, req *volume_s // On multi-disk servers, existing local shards may be on a different disk // than where copied shards were placed during ec.rebuild. rebuildDataDir := rebuildLocation.Directory - var additionalDirs []string - for _, otherLocation := range otherLocationsWithShards { - additionalDirs = append(additionalDirs, otherLocation.Directory) - } + additionalDirs := rebuildSearchDirs(rebuildLocation, otherLocationsWithShards) // Rebuild missing EC files, searching all disk locations for input shards. // Present input shards are verified against the bitrot sidecar (when present) // and corrupt ones are regenerated; unsafe_ignore_sidecar bypasses the guard. start := time.Now() dataBaseFileName := path.Join(rebuildDataDir, baseFileName) - generatedShardIds, err := erasure_coding.RebuildEcFiles(dataBaseFileName, erasure_coding.BackgroundECContext(), req.UnsafeIgnoreSidecar, additionalDirs...) + // Resolve the layout ONCE and use that same answer for the rebuild and for + // the backfill below: a manifest describing these shards has to record the + // geometry they were actually reconstructed with. + rebuildCtx, resolveErr := erasure_coding.ResolveRebuildECContext(dataBaseFileName, erasure_coding.BackgroundECContext(), additionalDirs) + if resolveErr != nil { + recordEcRebuild("failure", time.Since(start)) + return nil, fmt.Errorf("resolve rebuild layout for %s: %v", dataBaseFileName, resolveErr) + } + generatedShardIds, err := erasure_coding.RebuildEcFiles(dataBaseFileName, rebuildCtx, req.UnsafeIgnoreSidecar, additionalDirs...) if err != nil { recordEcRebuild("failure", time.Since(start)) return nil, fmt.Errorf("RebuildEcFiles %s: %v", dataBaseFileName, err) @@ -273,14 +279,19 @@ func (vs *VolumeServer) VolumeEcShardsRebuild(ctx context.Context, req *volume_s // blesses current bytes; ComputeProtectionFromShards refuses a partial // manifest, so a multi-server rebuild that cannot reach all shards just skips. if erasure_coding.BitrotProtectionEnabled { - sidecarPath := erasure_coding.BitrotSidecarPath(dataBaseFileName, 0) - if _, statErr := os.Stat(sidecarPath); os.IsNotExist(statErr) { - ctx := erasure_coding.NewDefaultECContext("", 0) - if vi, _, found, _ := volume_info.MaybeLoadVolumeInfo(dataBaseFileName + ".vif"); found && vi.EcShardConfig != nil { - if ds, ps := int(vi.EcShardConfig.DataShards), int(vi.EcShardConfig.ParityShards); ds > 0 && ps > 0 && ds+ps <= erasure_coding.MaxShardCount { - ctx = &erasure_coding.ECContext{DataShards: ds, ParityShards: ps} - } - } + // "No sidecar yet" has to be asked of every place one could be, not + // just this directory. A split -dir/-dir.idx layout keeps it with the + // index and a sibling disk may hold it, and answering from the data + // base alone would write a fresh TOFU baseline over a volume that + // already has a manifest — blessing whatever the shards currently say + // and shadowing the real record, since the data base is searched first. + if erasure_coding.FindBitrotSidecar(0, dataBaseFileName, indexBaseFileName, additionalDirs...) == "" { + sidecarPath := erasure_coding.BitrotSidecarPath(dataBaseFileName, 0) + // The manifest must describe the shards as rebuilt, so it takes the + // context the rebuild resolved — not a narrower re-derivation that + // reads only this directory's .vif and drops the block size, which + // records the legacy layout for shards written with a uniform one. + ctx := rebuildCtx if prot, berr := erasure_coding.ComputeProtectionFromShards(dataBaseFileName, ctx, 0, additionalDirs); berr != nil { glog.V(2).Infof("bitrot backfill skipped for %s: %v", dataBaseFileName, berr) } else if werr := erasure_coding.SaveBitrotSidecar(sidecarPath, prot); werr != nil { @@ -710,6 +721,36 @@ func removeBitrotSidecars(baseFilename string) error { return firstErr } +// rebuildSearchDirs lists every directory besides the rebuild's own data +// directory that a rebuild may have to read from. Shards are only half of it: +// a split -dir/-dir.idx layout keeps .ecx/.ecj/.vif with the index, and on a +// multi-disk server the chosen disk may hold nothing but shards while the +// volume's .vif or generation-0 .ecsum sits on a sibling. Miss those and the +// layout resolution falls back to the default ratio and the legacy block +// size, reconstructing through the wrong matrix. The rebuild's own INDEX +// directory belongs here too — callers pass the data-directory base name, so +// it is not otherwise searched. Empty and duplicate entries are dropped. +func rebuildSearchDirs(rebuildLocation *storage.DiskLocation, otherLocations []*storage.DiskLocation) []string { + var dirs []string + appendDir := func(dir string) { + if dir == "" || dir == rebuildLocation.Directory { + return + } + for _, existing := range dirs { + if existing == dir { + return + } + } + dirs = append(dirs, dir) + } + appendDir(rebuildLocation.IdxDirectory) + for _, otherLocation := range otherLocations { + appendDir(otherLocation.Directory) + appendDir(otherLocation.IdxDirectory) + } + return dirs +} + func checkEcVolumeStatus(bName string, location *storage.DiskLocation) (hasEcxFile bool, hasIdxFile bool, existingShardCount int, err error) { // check whether to delete the .ecx and .ecj file also fileInfos, err := os.ReadDir(location.Directory) @@ -782,6 +823,27 @@ func (vs *VolumeServer) VolumeEcShardsMount(ctx context.Context, req *volume_ser } } + // A shard delivery can bring the checksum manifest with it, but the receive + // path only writes the file. When this server already had the volume + // mounted, the EcVolume in memory keeps whatever protection state it + // resolved at mount — off, for a volume whose sidecar arrives now — until a + // remount. Re-resolve it here, where the shards it describes were added. + // Every per-disk runtime, not just the first: a vid mounts as one EcVolume + // per disk, the delivery lands the .ecsum on one of them, and the + // first-match FindEcVolume would leave the siblings reporting no protection + // until a remount. Each re-resolves against its own data and index base, so + // a shared -dir.idx reaches all of them. + // + // Resolving across every EC metadata directory is what makes that reload + // mean something. Startup mirroring gives each shard-bearing disk its own + // .ecx/.ecj/.vif but deliberately not the sidecar, so a runtime restricted + // to its own two directories would find nothing however often it reloaded. + // One delivered copy, reachable from all of them. + ecMetadataDirs := vs.store.EcMetadataDirs() + for _, v := range vs.store.FindAllEcVolumes(needle.VolumeId(req.VolumeId)) { + v.ReloadBitrotSidecar(ecMetadataDirs...) + } + return &volume_server_pb.VolumeEcShardsMountResponse{}, nil } @@ -1003,7 +1065,7 @@ func (vs *VolumeServer) VolumeEcShardsToVolume(ctx context.Context, req *volume_ // boundary, so the layout must not be derived from datFileSize. WriteDatFile // infers the layout from the shard size when .vif does not record it. // write .dat file from .ec00 ~ .ec09 files - if err := erasure_coding.WriteDatFile(dataBaseFileName, datFileSize, v.DatFileSize(), shardFileNames); err != nil { + if err := erasure_coding.WriteDatFile(dataBaseFileName, datFileSize, v.DatFileSize(), shardFileNames, v.ECContext.LargeBlockSize(), v.ECContext.SmallBlockSize()); err != nil { return nil, fmt.Errorf("WriteDatFile %s: %v", dataBaseFileName, err) } @@ -1179,11 +1241,25 @@ func (vs *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serv return nil, err } + // Report the layout this holder will actually serve reads through. It is + // the only way a coordinator can tell a holder that understands the + // uniform block layout from one that dropped the unknown .vif field on the + // floor and mounted the volume as legacy. + var ecShardConfig *volume_server_pb.EcShardConfig + if primary.ECContext != nil { + ecShardConfig = &volume_server_pb.EcShardConfig{ + DataShards: uint32(primary.ECContext.DataShards), + ParityShards: uint32(primary.ECContext.ParityShards), + BlockSize: primary.ECContext.BlockSize, + } + } + res := &volume_server_pb.VolumeEcShardsInfoResponse{ EcShardInfos: shardInfos, FileCount: files, FileDeletedCount: filesDeleted, VolumeSize: totalSize, + EcShardConfig: ecShardConfig, } return res, nil diff --git a/weed/server/volume_grpc_erasure_coding_dirs_test.go b/weed/server/volume_grpc_erasure_coding_dirs_test.go new file mode 100644 index 000000000..4516fca6f --- /dev/null +++ b/weed/server/volume_grpc_erasure_coding_dirs_test.go @@ -0,0 +1,73 @@ +package weed_server + +import ( + "reflect" + "testing" + + "github.com/seaweedfs/seaweedfs/weed/storage" +) + +func loc(dir, idxDir string) *storage.DiskLocation { + return &storage.DiskLocation{Directory: dir, IdxDirectory: idxDir} +} + +// The rebuild reads its shards from one directory but resolves the volume's +// layout — ratio and uniform block size — from the .vif or the generation-0 +// .ecsum, which on a multi-disk server may sit anywhere. Every directory that +// could hold one has to be in the search list, or the resolution silently +// falls back to 10+4 with the legacy striping and reconstructs through the +// wrong matrix. +func TestRebuildSearchDirs(t *testing.T) { + tests := []struct { + name string + rebuild *storage.DiskLocation + others []*storage.DiskLocation + want []string + }{ + { + name: "the rebuild's own index directory is searched", + rebuild: loc("/data1", "/idx1"), + want: []string{"/idx1"}, + }, + { + name: "an unsplit rebuild location contributes nothing", + rebuild: loc("/data1", "/data1"), + want: nil, + }, + { + // The case two reviewers flagged: a sibling holding only shards + // while its index directory holds this volume's .vif. + name: "a sibling's index directory is searched, not just its data directory", + rebuild: loc("/data1", "/data1"), + others: []*storage.DiskLocation{loc("/data2", "/idx2")}, + want: []string{"/data2", "/idx2"}, + }, + { + name: "shared index directories are listed once", + rebuild: loc("/data1", "/shared-idx"), + others: []*storage.DiskLocation{loc("/data2", "/shared-idx"), loc("/data3", "/shared-idx")}, + want: []string{"/shared-idx", "/data2", "/data3"}, + }, + { + name: "the rebuild's data directory is never repeated", + rebuild: loc("/data1", "/idx1"), + others: []*storage.DiskLocation{loc("/data1", "/idx1"), loc("/data2", "/data1")}, + want: []string{"/idx1", "/data2"}, + }, + { + name: "empty directories are dropped", + rebuild: loc("/data1", ""), + others: []*storage.DiskLocation{loc("/data2", "")}, + want: []string{"/data2"}, + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + got := rebuildSearchDirs(tc.rebuild, tc.others) + if !reflect.DeepEqual(got, tc.want) { + t.Errorf("rebuildSearchDirs() = %v, want %v", got, tc.want) + } + }) + } +} diff --git a/weed/server/volume_grpc_erasure_coding_recover_test.go b/weed/server/volume_grpc_erasure_coding_recover_test.go index 9f21edc94..fc0d0d1ab 100644 --- a/weed/server/volume_grpc_erasure_coding_recover_test.go +++ b/weed/server/volume_grpc_erasure_coding_recover_test.go @@ -18,6 +18,7 @@ import ( "github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding" "github.com/seaweedfs/seaweedfs/weed/storage/needle" "github.com/seaweedfs/seaweedfs/weed/storage/types" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" "github.com/seaweedfs/seaweedfs/weed/util" ) @@ -111,7 +112,12 @@ func TestFetchEcIndexFromPeers_CopiesIndexOverGrpc(t *testing.T) { if err := os.WriteFile(srcBase+".ecj", []byte("journal"), 0o644); err != nil { t.Fatalf("write source .ecj: %v", err) } - if err := os.WriteFile(srcBase+".vif", []byte("volinfo"), 0o644); err != nil { + // A real .vif: the receiver MOUNTS the volume from the copied files, and a + // mount refuses a .vif it cannot parse rather than guessing the layout. + if err := volume_info.SaveVolumeInfo(srcBase+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10, ParityShards: 4}, + }); err != nil { t.Fatalf("write source .vif: %v", err) } @@ -223,7 +229,12 @@ func TestVolumeEcShardsMount_RecoverMissingIndex(t *testing.T) { if err := os.WriteFile(srcBase+".ecj", nil, 0o644); err != nil { t.Fatalf("write source .ecj: %v", err) } - if err := os.WriteFile(srcBase+".vif", []byte("volinfo"), 0o644); err != nil { + // A real .vif: the receiver MOUNTS the volume from the copied files, and a + // mount refuses a .vif it cannot parse rather than guessing the layout. + if err := volume_info.SaveVolumeInfo(srcBase+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10, ParityShards: 4}, + }); err != nil { t.Fatalf("write source .vif: %v", err) } srcGrpcPort := serveGrpc(t, func(s *grpc.Server) { diff --git a/weed/storage/disk_location_ec.go b/weed/storage/disk_location_ec.go index 218abc8d6..71987b550 100644 --- a/weed/storage/disk_location_ec.go +++ b/weed/storage/disk_location_ec.go @@ -484,26 +484,17 @@ func (l *DiskLocation) checkOrphanedShards(shards []string, collection string, v // erasure_coding.DataShardsCount so that tests writing a custom layout // to .vif compute the matching shard size, and so custom-ratio builds // (e.g. enterprise) can swap the default without touching this helper. +// The padded shard length is the same number under both layouts — +// TestUniformBlockSizeMatchesLegacyShardSize asserts the equivalence for every +// input — so defer to the encoder's own helper instead of keeping a second copy +// of the padding rule that a future change would have to be made in twice. An +// empty .dat keeps its historic answer: the legacy encoder emits no block for +// it, where UniformBlockSize floors at one. func calculateExpectedShardSize(datFileSize int64, dataShardCount int) int64 { - if dataShardCount <= 0 { + if dataShardCount <= 0 || datFileSize <= 0 { return 0 } - var shardSize int64 - - // Process large blocks (1GB * dataShardCount per batch) - largeBatchSize := int64(erasure_coding.ErasureCodingLargeBlockSize) * int64(dataShardCount) - numLargeBatches := datFileSize / largeBatchSize - shardSize = numLargeBatches * int64(erasure_coding.ErasureCodingLargeBlockSize) - remainingSize := datFileSize - (numLargeBatches * largeBatchSize) - - // Process remaining data in small blocks (1MB * dataShardCount per batch) - if remainingSize > 0 { - smallBatchSize := int64(erasure_coding.ErasureCodingSmallBlockSize) * int64(dataShardCount) - numSmallBatches := (remainingSize + smallBatchSize - 1) / smallBatchSize // Ceiling division - shardSize += numSmallBatches * int64(erasure_coding.ErasureCodingSmallBlockSize) - } - - return shardSize + return erasure_coding.UniformBlockSize(datFileSize, dataShardCount) } // validateEcVolume reports whether the EC files for (collection, vid) on this diff --git a/weed/storage/erasure_coding/ec_bitrot.go b/weed/storage/erasure_coding/ec_bitrot.go index 5aef042c7..dfe78a5f2 100644 --- a/weed/storage/erasure_coding/ec_bitrot.go +++ b/weed/storage/erasure_coding/ec_bitrot.go @@ -195,6 +195,7 @@ func buildProtectionFromBuilders(ctx *ECContext, builders []*shardChecksumBuilde EcShardConfig: &volume_server_pb.EcShardConfig{ DataShards: uint32(ctx.DataShards), ParityShards: uint32(ctx.ParityShards), + BlockSize: ctx.BlockSize, }, Shards: shards, EncodeUuid: NewEncodeUUID(), @@ -431,6 +432,7 @@ func ComputeProtectionFromShards(baseFileName string, ctx *ECContext, generation EcShardConfig: &volume_server_pb.EcShardConfig{ DataShards: uint32(ctx.DataShards), ParityShards: uint32(ctx.ParityShards), + BlockSize: ctx.BlockSize, }, Shards: shards, EncodeUuid: NewEncodeUUID(), @@ -474,53 +476,141 @@ func (ev *EcVolume) BitrotProtection() (*volume_server_pb.EcBitrotProtection, Bi return ev.bitrot, ev.bitrotStatus } +// ReloadBitrotSidecar re-resolves the checksum sidecar for a volume that is +// already mounted — a shard delivery can bring the manifest with it, and the +// receive path only writes the file, so without this the in-memory volume +// keeps the protection state it resolved at mount (off) until a remount. +func (ev *EcVolume) ReloadBitrotSidecar(additionalDirs ...string) { + if err := ev.loadActiveBitrotSidecar(additionalDirs...); err != nil { + glog.Warningf("ec volume %d: reload bitrot sidecar: %v", ev.VolumeId, err) + } +} + // loadActiveBitrotSidecar loads the generation-0 checksum sidecar into the // volume. Best-effort: any failure leaves protection off/invalid without // failing the mount. OSS only produces generation-0 (fresh-encode) sidecars. -func (ev *EcVolume) loadActiveBitrotSidecar() { - ev.loadBitrotForGeneration(0) +func (ev *EcVolume) loadActiveBitrotSidecar(additionalDirs ...string) error { + return ev.loadBitrotForGeneration(0, additionalDirs...) } // loadBitrotForGeneration loads and validates the sidecar describing generation // `generation`, setting ev.bitrot/ev.bitrotStatus. Called at mount. Absent or -// generation/config-mismatched => BitrotOff; self-integrity/manifest failure => +// generation-mismatched => BitrotOff; self-integrity/manifest failure => // BitrotInvalid; usable => BitrotOn. -func (ev *EcVolume) loadBitrotForGeneration(generation uint32) { +// +// A sidecar written FOR THIS generation that disagrees with the volume's shard +// geometry is different in kind from those: both files record the layout the +// generation was encoded with, so a disagreement means one of them is wrong and +// reads through the other would land at the wrong shard offsets. That returns +// an error and fails the mount instead of quietly dropping to unprotected +// reads. +func (ev *EcVolume) loadBitrotForGeneration(generation uint32, additionalDirs ...string) error { ev.bitrotLock.Lock() defer ev.bitrotLock.Unlock() ev.bitrot = nil ev.bitrotStatus = BitrotOff if ev.ECContext == nil { - return + return nil } - path := findBitrotSidecar(generation, ev.DataBaseFileName(), ev.IndexBaseFileName()) + path := findBitrotSidecar(generation, ev.DataBaseFileName(), ev.IndexBaseFileName(), additionalDirs...) if path == "" { - return + return nil } prot, err := LoadBitrotSidecar(path) if err != nil { glog.Warningf("ec volume %d: bitrot sidecar %s self-integrity failed: %v", ev.VolumeId, path, err) ev.bitrotStatus = BitrotInvalid - return + return nil } if prot.Generation != generation { - return // not for this generation -> off, not corruption + return nil // not for this generation -> off, not corruption } - if prot.EcShardConfig == nil || - int(prot.EcShardConfig.DataShards) != ev.ECContext.DataShards || - int(prot.EcShardConfig.ParityShards) != ev.ECContext.ParityShards { - return + if prot.EcShardConfig == nil { + return nil // records no geometry -> nothing to contradict + } + if int(prot.EcShardConfig.DataShards) != ev.ECContext.DataShards || + int(prot.EcShardConfig.ParityShards) != ev.ECContext.ParityShards || + prot.EcShardConfig.BlockSize != ev.ECContext.BlockSize { + return fmt.Errorf("ec volume %d generation %d: %s records layout %d+%d block %d but the volume is mounted as %d+%d block %d; refusing to serve one of the two layouts", + ev.VolumeId, generation, path, + prot.EcShardConfig.DataShards, prot.EcShardConfig.ParityShards, prot.EcShardConfig.BlockSize, + ev.ECContext.DataShards, ev.ECContext.ParityShards, ev.ECContext.BlockSize) } if err := ValidateBitrotManifest(prot, ev.ECContext.DataShards, ev.ECContext.ParityShards); err != nil { glog.Warningf("ec volume %d: bitrot sidecar %s manifest invalid: %v", ev.VolumeId, path, err) ev.bitrotStatus = BitrotInvalid - return + return nil } ev.bitrot = prot ev.bitrotStatus = BitrotOn glog.V(1).Infof("ec volume %d: loaded bitrot protection generation %d (%d shards, block_size %d)", ev.VolumeId, generation, len(prot.Shards), prot.BlockSize) + return nil +} + +// layoutFromSidecar resolves a volume's EC layout from its generation-0 +// checksum sidecar, searching the data base and then the index base. A split +// -dir/-dir.idx layout keeps the sidecar with the index, so probing the data +// base alone reports "absent" for a volume that has the record right there — +// and absent is the one answer that selects the legacy 10+4 layout. +// +// The three answers stay distinct: (nil, false, nil) means genuinely absent +// and the caller may fall back to the defaults; a non-nil error means the +// record exists but cannot establish the layout, which must fail rather than +// default; otherwise the config is usable. +func layoutFromSidecar(dataBaseFileName, indexBaseFileName string) (cfg *volume_server_pb.EcShardConfig, found bool, err error) { + path := findBitrotSidecar(0, dataBaseFileName, indexBaseFileName) + if path == "" { + return nil, false, nil + } + cfg, err = EcShardConfigFromSidecarPath(path) + return cfg, true, err +} + +// EcShardConfigFromSidecar reads the generation-0 bitrot sidecar's record of a +// volume's EC config. The three answers are distinct on purpose: absent means +// the volume may genuinely predate the sidecar, which is the only case a caller +// may answer with the legacy layout; present-but-unusable means the one +// surviving record of the geometry is corrupt, and guessing from there returns +// wrong bytes rather than no bytes. +func EcShardConfigFromSidecar(baseFileName string) (cfg *volume_server_pb.EcShardConfig, found bool, err error) { + path := BitrotSidecarPath(baseFileName, 0) + if _, statErr := os.Stat(path); statErr != nil { + if os.IsNotExist(statErr) { + return nil, false, nil + } + return nil, false, fmt.Errorf("stat %s: %w", path, statErr) + } + cfg, err = EcShardConfigFromSidecarPath(path) + return cfg, true, err +} + +// EcShardConfigFromSidecarPath is EcShardConfigFromSidecar for a sidecar whose +// path the caller already resolved — the rebuild finds it across a multi-disk +// server's directories, not only next to the base name. +func EcShardConfigFromSidecarPath(path string) (*volume_server_pb.EcShardConfig, error) { + prot, loadErr := LoadBitrotSidecar(path) + if loadErr != nil { + return nil, fmt.Errorf("read %s: %w", path, loadErr) + } + // The un-suffixed sidecar describes generation 0 and nothing else; one + // stamped for another generation is not a record of these shards. + if prot.GetGeneration() != 0 { + return nil, fmt.Errorf("%s records generation %d, not generation 0", path, prot.GetGeneration()) + } + cfg := prot.GetEcShardConfig() + if cfg == nil { + return nil, fmt.Errorf("%s records no EC config", path) + } + if !ValidEcShardCounts(cfg.GetDataShards(), cfg.GetParityShards()) { + return nil, fmt.Errorf("%s records invalid shard counts %d+%d", + path, cfg.GetDataShards(), cfg.GetParityShards()) + } + if bsErr := ValidateBlockSize(cfg.GetBlockSize()); bsErr != nil { + return nil, fmt.Errorf("%s: %w", path, bsErr) + } + return cfg, nil } // RemoveBitrotSidecars removes the legacy .ecsum and any versioned @@ -534,6 +624,15 @@ func RemoveBitrotSidecars(base string) { } } +// FindBitrotSidecar is findBitrotSidecar for callers outside this package. It +// answers "does this volume already have a manifest, and where" — a question +// that cannot be settled from one base name, because a split -dir/-dir.idx +// layout keeps the sidecar with the index and a multi-disk server may keep it +// on a sibling. Returns "" when no candidate exists. +func FindBitrotSidecar(generation uint32, dataBase, indexBase string, additionalDirs ...string) string { + return findBitrotSidecar(generation, dataBase, indexBase, additionalDirs...) +} + // findBitrotSidecar resolves the sidecar path for a generation, searching the // data base, the index base, and any additional directories — mirroring how // shard/.vif lookups handle split data/idx layouts and per-disk mirrors. diff --git a/weed/storage/erasure_coding/ec_context.go b/weed/storage/erasure_coding/ec_context.go index 9fc352d83..ac77f688b 100644 --- a/weed/storage/erasure_coding/ec_context.go +++ b/weed/storage/erasure_coding/ec_context.go @@ -13,6 +13,16 @@ type ECContext struct { ParityShards int Collection string VolumeId needle.VolumeId + // BlockSize > 0 selects the uniform block layout: every shard is a single + // contiguous block of this many bytes, so consecutive .dat ranges stay on + // one shard instead of striping across all of them at 1MiB granularity. + // 0 is the legacy two-tier 1GiB/1MiB layout. + BlockSize int64 + // DatFileSize is the length of the .dat the encode actually read, set by + // WriteEcFiles from the same measurement it derives BlockSize from. The + // .vif must record the two together: they describe one file, and a live + // volume can grow between two separate stats. + DatFileSize int64 } // Total returns the total number of shards (data + parity) @@ -20,6 +30,33 @@ func (ctx *ECContext) Total() int { return ctx.DataShards + ctx.ParityShards } +// ValidEcShardCounts reports whether a recorded (data, parity) pair could +// describe a real EC volume. The sum is taken in uint64 deliberately: on a +// 32-bit build `int` is 32 bits, so converting each count first and adding +// them wraps for values near the uint32 ceiling — 0x7fffffff + 0x7fffffff +// lands at -2, which slips under the MaxShardCount bound. +func ValidEcShardCounts(dataShards, parityShards uint32) bool { + return dataShards > 0 && parityShards > 0 && + uint64(dataShards)+uint64(parityShards) <= uint64(MaxShardCount) +} + +// LargeBlockSize returns the large-block length of this context's shard +// layout; nil-safe so callers can pass an unset context for the legacy layout. +func (ctx *ECContext) LargeBlockSize() int64 { + if ctx != nil && ctx.BlockSize > 0 { + return ctx.BlockSize + } + return ErasureCodingLargeBlockSize +} + +// SmallBlockSize returns the small-block length of this context's shard layout. +func (ctx *ECContext) SmallBlockSize() int64 { + if ctx != nil && ctx.BlockSize > 0 { + return ctx.BlockSize + } + return ErasureCodingSmallBlockSize +} + // NewDefaultECContext creates a context with default 10+4 shard configuration func NewDefaultECContext(collection string, volumeId needle.VolumeId) *ECContext { return &ECContext{ diff --git a/weed/storage/erasure_coding/ec_decoder.go b/weed/storage/erasure_coding/ec_decoder.go index 5bf4342d3..96494c4a5 100644 --- a/weed/storage/erasure_coding/ec_decoder.go +++ b/weed/storage/erasure_coding/ec_decoder.go @@ -233,8 +233,10 @@ func iterateEcjFile(baseFileName string, processNeedleFn func(key types.NeedleId // large-block row boundary, and deriving the layout from the shrunk extent // would read the shards in the wrong block order. Pass zero when the .vif does // not record the encode-time size to infer the layout from the shard size. -func WriteDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int64, shardFileNames []string) error { - return writeDatFile(baseFileName, datFileSize, encodedDatFileSize, shardFileNames, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize) +// largeBlockSize/smallBlockSize are the volume's shard block layout, e.g. +// ctx.LargeBlockSize()/ctx.SmallBlockSize() from its .vif EC config. +func WriteDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int64, shardFileNames []string, largeBlockSize int64, smallBlockSize int64) error { + return writeDatFile(baseFileName, datFileSize, encodedDatFileSize, shardFileNames, largeBlockSize, smallBlockSize) } func writeDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int64, shardFileNames []string, largeBlockSize int64, smallBlockSize int64) error { @@ -288,7 +290,11 @@ func writeDatFile(baseFileName string, datFileSize int64, encodedDatFileSize int // A shard size that is an exact multiple of the large block size is // ambiguous: N large rows, or N-1 large rows plus a full small-block // region. The two layouts only agree below the last large row. - if shardSize%largeBlockSize == 0 && datFileSize > (shardSize/largeBlockSize-1)*largeBlockSize*int64(dataShards) { + // A uniform layout has no such ambiguity — its large and small blocks + // are the same size, so every reading of the shard is the same one, and + // without this precondition the check fires on every volume. + if largeBlockSize != smallBlockSize && + shardSize%largeBlockSize == 0 && datFileSize > (shardSize/largeBlockSize-1)*largeBlockSize*int64(dataShards) { return fmt.Errorf("shard size %d of %s does not identify the block layout; re-encode to record the dat size in .vif", shardSize, baseFileName) } encodedDatFileSize = int64(dataShards) * shardSize diff --git a/weed/storage/erasure_coding/ec_decoder_test.go b/weed/storage/erasure_coding/ec_decoder_test.go index 8c655d7fa..b09fed947 100644 --- a/weed/storage/erasure_coding/ec_decoder_test.go +++ b/weed/storage/erasure_coding/ec_decoder_test.go @@ -217,7 +217,7 @@ func TestDecodeAtomicPublish(t *testing.T) { // final .dat nor a partial .dat.tmp behind. datBase := filepath.Join(dir, "bar_2") missingShards := []string{filepath.Join(dir, "does_not_exist.ec00")} - if err := erasure_coding.WriteDatFile(datBase, 100, 100, missingShards); err == nil { + if err := erasure_coding.WriteDatFile(datBase, 100, 100, missingShards, erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize); err == nil { t.Fatalf("expected WriteDatFile to fail on missing shard") } if _, err := os.Stat(datBase + ".dat"); !os.IsNotExist(err) { diff --git a/weed/storage/erasure_coding/ec_encoder.go b/weed/storage/erasure_coding/ec_encoder.go index 927a50d86..1b28fddf1 100644 --- a/weed/storage/erasure_coding/ec_encoder.go +++ b/weed/storage/erasure_coding/ec_encoder.go @@ -62,12 +62,126 @@ func WriteSortedFileFromIdx(baseFileName string, ext string) (e error) { // BackgroundECContext for the default ratio, or an explicit ctx for a configured // (e.g. custom-ratio) layout. It returns the bitrot protection (per-shard block // CRC32C) computed during the single encode pass; the caller persists it as a -// .ecsum sidecar. +// .ecsum sidecar, and persists ctx.BlockSize and ctx.DatFileSize (both +// set here, from one measurement of the .dat) to the .vif so readers resolve +// the shard block layout. func WriteEcFiles(baseFileName string, ctx *ECContext) (*volume_server_pb.EcBitrotProtection, error) { - if ctx == nil || ctx.Total() == 0 { + if ctx == nil { ctx = NewDefaultECContext("", 0) + } else if ctx.Total() == 0 { + // Fill the placeholder in place rather than swapping the pointer: the + // caller reads BlockSize and DatFileSize back off the context it + // passed, and a replacement leaves it holding the zero values. + ctx.DataShards, ctx.ParityShards = DataShardsCount, ParityShardsCount } - return generateEcFiles(baseFileName, 256*1024, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize, ctx) + // Always encode with the uniform block layout, sized for this .dat. Both + // the block size and the .dat length it was derived from are left on ctx, + // so the caller persists a .vif whose two fields describe one measurement + // — a second stat could see a different size on a volume still taking + // writes. + fi, err := os.Stat(baseFileName + ".dat") + if err != nil { + return nil, fmt.Errorf("failed to stat dat file: %w", err) + } + ctx.DatFileSize = fi.Size() + ctx.BlockSize = UniformBlockSize(fi.Size(), ctx.DataShards) + return generateEcFiles(baseFileName, 256*1024, ctx.BlockSize, ctx.BlockSize, ctx) +} + +// ValidateBlockSize reports whether a `.vif`-recorded shard block size is one +// an encoder could have produced. 0 means the legacy two-tier layout, which is +// always valid; anything positive must be a whole number of small blocks, +// because that is what UniformBlockSize rounds to. A negative or unaligned +// value is corruption, and using it would map every read to the wrong shard +// offset. +func ValidateBlockSize(blockSize int64) error { + if blockSize == 0 { + return nil + } + if blockSize < 0 || blockSize%ErasureCodingSmallBlockSize != 0 { + return fmt.Errorf("invalid shard block size %d: expected 0 (legacy) or a multiple of %d", + blockSize, ErasureCodingSmallBlockSize) + } + return nil +} + +// UniformBlockSize returns the per-shard block size of the uniform layout for +// a .dat of the given size: ceil(datFileSize/dataShards) rounded up to a whole +// small block. For every input this equals the legacy layout's padded shard +// size, so only the byte placement differs between the two layouts, never the +// shard length. +func UniformBlockSize(datFileSize int64, dataShards int) int64 { + perShard := (datFileSize + int64(dataShards) - 1) / int64(dataShards) + blocks := (perShard + ErasureCodingSmallBlockSize - 1) / ErasureCodingSmallBlockSize + if blocks < 1 { + blocks = 1 + } + return blocks * ErasureCodingSmallBlockSize +} + +// ResolveRebuildECContext answers which shard layout a rebuild of baseFileName +// will use: the caller's context when it already states one, else the volume's +// own metadata — its `.vif` in any of the directories the rebuild can read, +// then the generation-0 bitrot sidecar, and only then the build defaults. +// Exported so a caller that must agree with the rebuild (the post-rebuild +// bitrot backfill writes a manifest describing these very shards) resolves +// once and uses the same answer, instead of re-deriving it from a narrower +// search and recording a layout the rebuild did not use. +func ResolveRebuildECContext(baseFileName string, ctx *ECContext, additionalDirs []string) (*ECContext, error) { + if ctx == nil || ctx.Total() == 0 { + // Resolve the layout from the .vif to preserve the original configuration. + vifPath := findVifPath(baseFileName, additionalDirs) + volumeInfo, _, foundVif, vifErr := volume_info.MaybeLoadVolumeInfo(vifPath) + if vifErr != nil { + // The .vif exists but cannot be read or parsed. Fail closed rather + // than silently falling back to the default ratio, which would + // rebuild a custom-ratio volume with the wrong layout. Pass an + // explicit ctx to override. + return nil, fmt.Errorf("RebuildEcFiles %s: cannot load .vif: %w", baseFileName, vifErr) + } + switch { + case foundVif && volumeInfo.EcShardConfig != nil && + ValidEcShardCounts(volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards): + if bsErr := ValidateBlockSize(volumeInfo.EcShardConfig.GetBlockSize()); bsErr != nil { + return nil, fmt.Errorf("RebuildEcFiles %s: %s: %w", baseFileName, vifPath, bsErr) + } + ctx = &ECContext{ + DataShards: int(volumeInfo.EcShardConfig.DataShards), + ParityShards: int(volumeInfo.EcShardConfig.ParityShards), + BlockSize: volumeInfo.EcShardConfig.GetBlockSize(), + } + glog.V(0).Infof("Rebuilding EC files for %s with config from .vif: %s", baseFileName, ctx.String()) + case foundVif && volumeInfo.EcShardConfig != nil: + // A recorded-but-impossible ratio is corruption, not a reason to + // substitute the default one: a 12+4 volume rebuilt as 10+4 + // reconstructs from the wrong matrix and never regenerates shards + // 14-15. Pass an explicit ctx to override. + return nil, fmt.Errorf("RebuildEcFiles %s: %s records invalid shard counts %d+%d", + baseFileName, vifPath, volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards) + default: + // No usable .vif: the bitrot sidecar records the same config at + // encode time and is then the surviving authority. Reading the + // default ratio and the legacy block size instead would rebuild + // from the wrong geometry AND make the sidecar look like it + // disagrees, which silently skips every checksum check below. + if sidecarPath := findBitrotSidecar(0, baseFileName, baseFileName, additionalDirs...); sidecarPath != "" { + cfg, cfgErr := EcShardConfigFromSidecarPath(sidecarPath) + if cfgErr != nil { + return nil, fmt.Errorf("RebuildEcFiles %s: no usable .vif and %w", baseFileName, cfgErr) + } + ctx = &ECContext{ + DataShards: int(cfg.GetDataShards()), + ParityShards: int(cfg.GetParityShards()), + BlockSize: cfg.GetBlockSize(), + } + glog.V(0).Infof("Rebuilding EC files for %s with config from the bitrot sidecar: %s", baseFileName, ctx.String()) + break + } + glog.V(0).Infof("Rebuilding EC files for %s with default config", baseFileName) + ctx = NewDefaultECContext("", 0) + } + } + return ctx, nil } // RebuildEcFiles rebuilds missing EC shard files. Pass BackgroundECContext to @@ -79,38 +193,11 @@ func WriteEcFiles(baseFileName string, ctx *ECContext) (*volume_server_pb.EcBitr // present input shards are verified against it and corrupt ones are excluded // from Reed-Solomon and regenerated; unsafeIgnoreSidecar bypasses that guard. func RebuildEcFiles(baseFileName string, ctx *ECContext, unsafeIgnoreSidecar bool, additionalDirs ...string) ([]uint32, error) { - if ctx == nil || ctx.Total() == 0 { - // Resolve the layout from the .vif to preserve the original configuration. - volumeInfo, _, foundVif, vifErr := volume_info.MaybeLoadVolumeInfo(baseFileName + ".vif") - if vifErr != nil { - // The .vif exists but cannot be read or parsed. Fail closed rather - // than silently falling back to the default ratio, which would - // rebuild a custom-ratio volume with the wrong layout. Pass an - // explicit ctx to override. - return nil, fmt.Errorf("RebuildEcFiles %s: cannot load .vif: %w", baseFileName, vifErr) - } - if foundVif && volumeInfo.EcShardConfig != nil { - ds := int(volumeInfo.EcShardConfig.DataShards) - ps := int(volumeInfo.EcShardConfig.ParityShards) - - // Validate EC config before using it - if ds > 0 && ps > 0 && ds+ps <= MaxShardCount { - ctx = &ECContext{ - DataShards: ds, - ParityShards: ps, - } - glog.V(0).Infof("Rebuilding EC files for %s with config from .vif: %s", baseFileName, ctx.String()) - } else { - glog.Warningf("Invalid EC config in .vif for %s (data=%d, parity=%d), using default", baseFileName, ds, ps) - ctx = NewDefaultECContext("", 0) - } - } else { - glog.V(0).Infof("Rebuilding EC files for %s with default config", baseFileName) - ctx = NewDefaultECContext("", 0) - } + ctx, err := ResolveRebuildECContext(baseFileName, ctx, additionalDirs) + if err != nil { + return nil, err } - - return generateMissingEcFiles(baseFileName, 256*1024, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize, ctx, unsafeIgnoreSidecar, additionalDirs) + return generateMissingEcFiles(baseFileName, 256*1024, ctx, unsafeIgnoreSidecar, additionalDirs) } func ToExt(ecIndex int) string { @@ -159,7 +246,10 @@ func findShardFile(baseFileName string, ext string, additionalDirs []string) str return "" } -func generateMissingEcFiles(baseFileName string, bufferSize int, largeBlockSize int64, smallBlockSize int64, ctx *ECContext, unsafeIgnoreSidecar bool, additionalDirs []string) (generatedShardIds []uint32, err error) { +// generateMissingEcFiles takes no block sizes: Reed-Solomon reconstruction is +// layout-agnostic — it rebuilds a missing shard from the same offsets of the +// survivors — so the shard layout only ever reaches it through ctx. +func generateMissingEcFiles(baseFileName string, bufferSize int, ctx *ECContext, unsafeIgnoreSidecar bool, additionalDirs []string) (generatedShardIds []uint32, err error) { // Pass 1: discover which shards exist and which are missing, // opening input files but NOT creating output files yet. @@ -363,6 +453,30 @@ func cleanupRebuildOutputs(outputFiles []*os.File, writePaths []string) { } } +// findVifPath locates the volume's `.vif` for a rebuild: next to the shards +// first, then in every directory the caller also handed us — the index +// directory of a split `-dir`/`-dir.idx` layout, and the sibling disks of a +// multi-disk server, where `.ecx`/`.ecj`/`.vif` may live while this disk holds +// only shards. Returns the data-base path when nothing exists, so the caller's +// "not found" handling stays on the canonical name. +func findVifPath(baseFileName string, additionalDirs []string) string { + dataPath := baseFileName + ".vif" + candidates := []string{dataPath} + base := filepath.Base(baseFileName) + for _, dir := range additionalDirs { + if dir == "" { + continue + } + candidates = append(candidates, filepath.Join(dir, base)+".vif") + } + for _, c := range candidates { + if _, err := os.Stat(c); err == nil { + return c + } + } + return dataPath +} + // loadRebuildSidecar loads and validates the generation-0 checksum sidecar for a // rebuild. RebuildEcFiles operates on the un-suffixed (generation 0) shard // names, so only the legacy sidecar is relevant here. Returns BitrotOff when @@ -381,10 +495,20 @@ func loadRebuildSidecar(baseFileName string, ctx *ECContext, additionalDirs []st if prot.Generation != 0 { return nil, BitrotOff } - if prot.EcShardConfig == nil || - int(prot.EcShardConfig.DataShards) != ctx.DataShards || - int(prot.EcShardConfig.ParityShards) != ctx.ParityShards { - return nil, BitrotOff + if prot.EcShardConfig == nil { + return nil, BitrotOff // records no geometry -> nothing to contradict + } + if int(prot.EcShardConfig.DataShards) != ctx.DataShards || + int(prot.EcShardConfig.ParityShards) != ctx.ParityShards || + prot.EcShardConfig.BlockSize != ctx.BlockSize { + // Both records describe the same encode, so a disagreement means the + // rebuild is about to reconstruct under a geometry the checksums do not + // cover. Treating that as "no protection" skipped every input and + // output check exactly when they matter most. + glog.Warningf("bitrot: sidecar %s records layout %d+%d block %d but the rebuild uses %d+%d block %d", + path, prot.EcShardConfig.DataShards, prot.EcShardConfig.ParityShards, prot.EcShardConfig.BlockSize, + ctx.DataShards, ctx.ParityShards, ctx.BlockSize) + return nil, BitrotInvalid } if err := ValidateBitrotManifest(prot, ctx.DataShards, ctx.ParityShards); err != nil { glog.Warningf("bitrot: sidecar %s manifest invalid: %v", path, err) diff --git a/weed/storage/erasure_coding/ec_rebuild_safety_test.go b/weed/storage/erasure_coding/ec_rebuild_safety_test.go index 99b912470..3ccb4a77b 100644 --- a/weed/storage/erasure_coding/ec_rebuild_safety_test.go +++ b/weed/storage/erasure_coding/ec_rebuild_safety_test.go @@ -5,7 +5,10 @@ import ( "path/filepath" "testing" + "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb" + "github.com/seaweedfs/seaweedfs/weed/storage/needle" "github.com/seaweedfs/seaweedfs/weed/storage/types" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" ) // readEcxSizeField returns the size field of the .ecx entry for needleId, or @@ -295,3 +298,101 @@ func TestRebuildEcFiles_CustomRatioRebuildsByteIdentical(t *testing.T) { t.Errorf("unexpected shard .ec12 for a 9+3 volume") } } + +// With no .vif, the bitrot sidecar is the surviving record of the volume's +// geometry. Defaulting to 10+4 and the legacy block layout instead would +// reconstruct a custom-ratio volume from the wrong matrix — and would make the +// sidecar look like it disagrees, silently skipping every checksum check. +func TestRebuildEcFiles_ResolvesGeometryFromSidecarWithoutVif(t *testing.T) { + if !BitrotProtectionEnabled { + t.Skip("bitrot protection is compiled out") + } + dir := t.TempDir() + base := filepath.Join(dir, "vol") + ctx := &ECContext{DataShards: 12, ParityShards: 4} + writeRandomDat(t, base, 7000) + + prot, err := WriteEcFiles(base, ctx) + if err != nil { + t.Fatalf("WriteEcFiles: %v", err) + } + if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil { + t.Fatalf("save sidecar: %v", err) + } + // No .vif at all: the sidecar has to answer. + const dropped = 11 + want, err := os.ReadFile(base + ToExt(dropped)) + if err != nil { + t.Fatalf("read shard: %v", err) + } + if err := os.Remove(base + ToExt(dropped)); err != nil { + t.Fatalf("remove shard: %v", err) + } + + // nil ctx: the production RPC's BackgroundECContext path. + if _, err := RebuildEcFiles(base, nil, false); err != nil { + t.Fatalf("RebuildEcFiles: %v", err) + } + got, err := os.ReadFile(base + ToExt(dropped)) + if err != nil { + t.Fatalf("read rebuilt shard: %v", err) + } + if string(got) != string(want) { + t.Errorf("rebuilt shard %d differs; the rebuild used the wrong geometry", dropped) + } + // The 12+4 shards past the default ratio must still be there. + for _, id := range []int{14, 15} { + if _, err := os.Stat(base + ToExt(id)); err != nil { + t.Errorf("shard %d missing after rebuild: %v", id, err) + } + } +} + +// A split -dir/-dir.idx layout keeps the .vif with the index, and a multi-disk +// server may leave this disk holding only shards. Probing just the data base +// resolves such a volume to the default ratio and the legacy block layout. +func TestRebuildEcFiles_FindsVifInAnAdditionalDir(t *testing.T) { + dataDir := t.TempDir() + idxDir := t.TempDir() + base := filepath.Join(dataDir, "vol") + ctx := &ECContext{DataShards: 12, ParityShards: 4} + writeRandomDat(t, base, 7000) + + if _, err := WriteEcFiles(base, ctx); err != nil { + t.Fatalf("WriteEcFiles: %v", err) + } + // The volume's .vif lives with the index, not the shards. + if err := volume_info.SaveVolumeInfo(filepath.Join(idxDir, "vol")+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.GetCurrentVersion()), + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 12, ParityShards: 4, BlockSize: ctx.BlockSize, + }, + }); err != nil { + t.Fatalf("save .vif: %v", err) + } + + const dropped = 11 + want, err := os.ReadFile(base + ToExt(dropped)) + if err != nil { + t.Fatalf("read shard: %v", err) + } + if err := os.Remove(base + ToExt(dropped)); err != nil { + t.Fatalf("remove shard: %v", err) + } + + if _, err := RebuildEcFiles(base, nil, true, idxDir); err != nil { + t.Fatalf("RebuildEcFiles: %v", err) + } + got, err := os.ReadFile(base + ToExt(dropped)) + if err != nil { + t.Fatalf("read rebuilt shard: %v", err) + } + if string(got) != string(want) { + t.Errorf("rebuilt shard %d differs; the rebuild used the wrong geometry", dropped) + } + for _, id := range []int{14, 15} { + if _, err := os.Stat(base + ToExt(id)); err != nil { + t.Errorf("shard %d missing after rebuild: %v", id, err) + } + } +} diff --git a/weed/storage/erasure_coding/ec_roundtrip_test.go b/weed/storage/erasure_coding/ec_roundtrip_test.go index 8c07a338d..f70c23970 100644 --- a/weed/storage/erasure_coding/ec_roundtrip_test.go +++ b/weed/storage/erasure_coding/ec_roundtrip_test.go @@ -326,7 +326,7 @@ func testDecodeDat(t *testing.T, datSize int64) { shardFileNames[i] = fmt.Sprintf("%s%s", baseFileName, ctx.ToExt(i)) } - err = WriteDatFile(decodedBase, datSize, datSize, shardFileNames) + err = WriteDatFile(decodedBase, datSize, datSize, shardFileNames, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize) require.NoError(t, err, "WriteDatFile") // The atomic publish must rename the temp file away, never leaving it behind. diff --git a/weed/storage/erasure_coding/ec_uniform_layout_test.go b/weed/storage/erasure_coding/ec_uniform_layout_test.go new file mode 100644 index 000000000..215fd7bfa --- /dev/null +++ b/weed/storage/erasure_coding/ec_uniform_layout_test.go @@ -0,0 +1,250 @@ +package erasure_coding + +import ( + "bytes" + "fmt" + "math/rand" + "os" + "testing" + + "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb" + "github.com/seaweedfs/seaweedfs/weed/storage/needle" + "github.com/seaweedfs/seaweedfs/weed/storage/types" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" +) + +// legacyShardSize is the padded shard length the two-tier layout produces: +// whole 1GiB rows, then the remainder in 1MiB rows. +func legacyShardSize(datFileSize int64, dataShards int) int64 { + largeRow := int64(ErasureCodingLargeBlockSize) * int64(dataShards) + nLarge := datFileSize / largeRow + size := nLarge * ErasureCodingLargeBlockSize + if rem := datFileSize - nLarge*largeRow; rem > 0 { + smallRow := int64(ErasureCodingSmallBlockSize) * int64(dataShards) + size += (rem + smallRow - 1) / smallRow * ErasureCodingSmallBlockSize + } + return size +} + +// The uniform layout must never change a shard's length, only the byte +// placement — capacity math and shard-size credibility checks depend on it. +func TestUniformBlockSizeMatchesLegacyShardSize(t *testing.T) { + sizes := []int64{ + 8, 1000, ErasureCodingSmallBlockSize, + 10 * ErasureCodingSmallBlockSize, + 10*ErasureCodingSmallBlockSize + 1, + 25 * 1024 * 1024, + 10 * ErasureCodingLargeBlockSize, + 10*ErasureCodingLargeBlockSize + 1, + 10*ErasureCodingLargeBlockSize - ErasureCodingSmallBlockSize, + 30 * 1024 * 1024 * 1024, + 30*1024*1024*1024 + 12345, + } + for _, datSize := range sizes { + for _, ds := range []int{10, 9, 5, 1} { + if got, want := UniformBlockSize(datSize, ds), legacyShardSize(datSize, ds); got != want { + t.Errorf("UniformBlockSize(%d, %d) = %d, legacy shard size %d", datSize, ds, got, want) + } + } + } + if got := UniformBlockSize(0, 10); got != ErasureCodingSmallBlockSize { + t.Errorf("UniformBlockSize(0, 10) = %d, want one small block", got) + } +} + +// encodeLayoutFixture writes a random .dat and encodes it with the given block +// sizes, returning the .dat content. +func encodeLayoutFixture(t *testing.T, baseFileName string, datSize int64, large, small int64, ctx *ECContext) []byte { + t.Helper() + data := make([]byte, datSize) + rand.New(rand.NewSource(datSize)).Read(data) + if err := os.WriteFile(baseFileName+".dat", data, 0o644); err != nil { + t.Fatalf("write .dat: %v", err) + } + if _, err := generateEcFiles(baseFileName, 256*1024, large, small, ctx); err != nil { + t.Fatalf("generateEcFiles: %v", err) + } + return data +} + +// readIntervals assembles a byte range from the shard files the way the read +// path does. +func readIntervals(t *testing.T, baseFileName string, ctx *ECContext, intervals []Interval) []byte { + t.Helper() + var assembled []byte + for _, iv := range intervals { + shardId, shardOffset := iv.ToShardIdAndOffset(ctx.LargeBlockSize(), ctx.SmallBlockSize()) + f, err := os.Open(baseFileName + ctx.ToExt(int(shardId))) + if err != nil { + t.Fatalf("open shard %d: %v", shardId, err) + } + buf := make([]byte, iv.Size) + if _, err := f.ReadAt(buf, shardOffset); err != nil { + t.Fatalf("read shard %d at %d: %v", shardId, shardOffset, err) + } + f.Close() + assembled = append(assembled, buf...) + } + return assembled +} + +// Encode 25MB — large enough that the uniform and legacy layouts place bytes +// differently — under both layouts and verify LocateData maps every probed +// range back to the original bytes. +func TestLocateDataMatchesEncoderForBothLayouts(t *testing.T) { + const datSize = 25*1024*1024 + 12345 + dir := t.TempDir() + + uniformCtx := NewDefaultECContext("", 1) + uniformCtx.BlockSize = UniformBlockSize(datSize, uniformCtx.DataShards) + legacyCtx := NewDefaultECContext("", 2) + + for name, ctx := range map[string]*ECContext{"uniform": uniformCtx, "legacy": legacyCtx} { + t.Run(name, func(t *testing.T) { + base := fmt.Sprintf("%s/%s", dir, name) + data := encodeLayoutFixture(t, base, datSize, ctx.LargeBlockSize(), ctx.SmallBlockSize(), ctx) + + shardSize := datSize / int64(ctx.DataShards) + for _, probe := range []struct{ offset, size int64 }{ + {0, 1000}, + {ctx.SmallBlockSize() - 100, 200}, // straddles a block boundary + {5 * 1024 * 1024, 4 * 1024 * 1024}, + {datSize - 2000, 2000}, + } { + intervals := LocateData(ctx.LargeBlockSize(), ctx.SmallBlockSize(), shardSize, probe.offset, types.Size(probe.size)) + got := readIntervals(t, base, ctx, intervals) + if !bytes.Equal(got, data[probe.offset:probe.offset+probe.size]) { + t.Fatalf("%s layout: bytes at [%d,+%d) do not round-trip", name, probe.offset, probe.size) + } + } + + // The point of the uniform layout: a 2MB range inside one 3MB + // block is a single interval instead of three 1MB stripes. + intervals := LocateData(ctx.LargeBlockSize(), ctx.SmallBlockSize(), shardSize, 6*1024*1024+512*1024, types.Size(2*1024*1024)) + if name == "uniform" && len(intervals) != 1 { + t.Errorf("uniform layout reads 2MB in %d intervals, want 1", len(intervals)) + } + if name == "legacy" && len(intervals) != 3 { + t.Errorf("legacy layout reads 2MB in %d intervals, want 3", len(intervals)) + } + }) + } +} + +// Encode under both layouts and verify WriteDatFile reconstructs the original +// .dat byte for byte — and that decoding a uniform volume with the legacy +// geometry does NOT, i.e. the recorded block size is load-bearing. +func TestUniformDecodeRoundTrip(t *testing.T) { + const datSize = 25 * 1024 * 1024 + dir := t.TempDir() + base := dir + "/1" + + ctx := NewDefaultECContext("", 1) + ctx.BlockSize = UniformBlockSize(datSize, ctx.DataShards) + data := encodeLayoutFixture(t, base, datSize, ctx.BlockSize, ctx.BlockSize, ctx) + + shardFileNames := make([]string, ctx.DataShards) + for i := range shardFileNames { + shardFileNames[i] = base + ctx.ToExt(i) + } + + if err := WriteDatFile(dir+"/decoded", datSize, datSize, shardFileNames, ctx.LargeBlockSize(), ctx.SmallBlockSize()); err != nil { + t.Fatalf("WriteDatFile: %v", err) + } + decoded, err := os.ReadFile(dir + "/decoded.dat") + if err != nil { + t.Fatalf("read decoded: %v", err) + } + if !bytes.Equal(decoded, data) { + t.Fatal("uniform decode with recorded geometry does not round-trip") + } + + if err := WriteDatFile(dir+"/wrong", datSize, datSize, shardFileNames, ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize); err != nil { + t.Fatalf("WriteDatFile legacy geometry: %v", err) + } + wrong, err := os.ReadFile(dir + "/wrong.dat") + if err != nil { + t.Fatalf("read wrong: %v", err) + } + if bytes.Equal(wrong, data) { + t.Fatal("decoding a uniform volume with legacy geometry should scramble it; the layouts did not diverge") + } +} + +// Mount fixtures through NewEcVolume and read a large extent back through +// LocateEcShardNeedleInterval, proving the .vif block size steers the read +// path: a legacy .vif (no block size) keeps the legacy interpretation, a +// uniform .vif reads uniform shards. +func TestEcVolumeGeometryFromVif(t *testing.T) { + const datSize = 25 * 1024 * 1024 + + for _, layout := range []string{"legacy", "uniform"} { + t.Run(layout, func(t *testing.T) { + dir := t.TempDir() + vid := needle.VolumeId(7) + base := EcShardFileName("", dir, int(vid)) + + ctx := NewDefaultECContext("", vid) + if layout == "uniform" { + ctx.BlockSize = UniformBlockSize(datSize, ctx.DataShards) + } + data := encodeLayoutFixture(t, base, datSize, ctx.LargeBlockSize(), ctx.SmallBlockSize(), ctx) + + if err := os.WriteFile(base+".ecx", nil, 0o644); err != nil { + t.Fatalf("write .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.GetCurrentVersion()), + DatFileSize: datSize, + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: uint32(ctx.DataShards), + ParityShards: uint32(ctx.ParityShards), + BlockSize: ctx.BlockSize, + }, + }); err != nil { + t.Fatalf("save .vif: %v", err) + } + + ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", vid) + if err != nil { + t.Fatalf("NewEcVolume: %v", err) + } + defer ev.Close() + if ev.ECContext.BlockSize != ctx.BlockSize { + t.Fatalf("loaded BlockSize = %d, want %d", ev.ECContext.BlockSize, ctx.BlockSize) + } + for i := 0; i < ctx.DataShards; i++ { + shard, err := NewEcVolumeShard(types.HardDriveType, dir, "", vid, ShardId(i)) + if err != nil { + t.Fatalf("NewEcVolumeShard %d: %v", i, err) + } + ev.AddEcVolumeShard(shard) + } + + // Sweep a large extent through the volume's own interval mapping. + // LocateEcShardNeedleInterval expands the size by the needle + // overhead, so compare only the probed prefix. + const probeSize = datSize - 64*1024 + intervals := ev.LocateEcShardNeedleInterval(needle.GetCurrentVersion(), 0, types.Size(probeSize)) + var assembled []byte + for _, iv := range intervals { + shardId, shardOffset := ev.IntervalToShardIdAndOffset(iv) + shard, found := ev.FindEcVolumeShard(shardId) + if !found { + t.Fatalf("shard %d not mounted", shardId) + } + buf := make([]byte, iv.Size) + if _, err := shard.ReadAt(buf, shardOffset); err != nil { + t.Fatalf("read shard %d at %d: %v", shardId, shardOffset, err) + } + assembled = append(assembled, buf...) + } + if len(assembled) < probeSize { + t.Fatalf("assembled %d bytes, want at least %d", len(assembled), probeSize) + } + if !bytes.Equal(assembled[:probeSize], data[:probeSize]) { + t.Fatalf("%s layout: full-extent read through the %s .vif does not match the .dat", layout, layout) + } + }) + } +} diff --git a/weed/storage/erasure_coding/ec_volume.go b/weed/storage/erasure_coding/ec_volume.go index 8bb4a4e8f..cb6463948 100644 --- a/weed/storage/erasure_coding/ec_volume.go +++ b/weed/storage/erasure_coding/ec_volume.go @@ -173,7 +173,16 @@ func NewEcVolume(diskType types.DiskType, dir string, dirIdx string, collection } } ev.Version = needle.Version3 - if volumeInfo, _, found, _ := volume_info.MaybeLoadVolumeInfo(vifFileName); found { + // A present-but-unreadable or malformed .vif FAILS the mount: every new + // encode records a positive uniform block size there, and defaulting to + // the legacy layout would serve those shards with the wrong offset math. + // Absent stays legal — legacy volumes predate the sidecar. + volumeInfo, _, found, vifErr := volume_info.MaybeLoadVolumeInfo(vifFileName) + if vifErr != nil { + ev.Close() + return nil, fmt.Errorf("ec volume %d: load %s: %w", vid, vifFileName, vifErr) + } + if found { ev.Version = needle.Version(volumeInfo.Version) ev.datFileSize = volumeInfo.DatFileSize ev.ExpireAtSec = volumeInfo.ExpireAtSec @@ -184,21 +193,56 @@ func NewEcVolume(diskType types.DiskType, dir string, dirIdx string, collection ps := int(volumeInfo.EcShardConfig.ParityShards) ev.EncodeTsNs = volumeInfo.EcShardConfig.GetEncodeTsNs() - // Validate shard counts to prevent zero or invalid values - if ds <= 0 || ps <= 0 || ds+ps > MaxShardCount { - glog.Warningf("Invalid EC config in VolumeInfo for volume %d (data=%d, parity=%d), using defaults", vid, ds, ps) - ev.ECContext = NewDefaultECContext(collection, vid) + // A config that is PRESENT but records an impossible ratio is not + // a volume to fall back on: substituting the default 10+4 with the + // legacy layout would read uniform shards with the wrong offset + // math and answer with the wrong bytes. Only an ENTIRELY absent + // config means "this predates the record", which the else-branch + // below serves with the legacy defaults. + if !ValidEcShardCounts(volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards) { + ev.Close() + return nil, fmt.Errorf("ec volume %d: %s records invalid shard counts %d+%d", + vid, vifFileName, volumeInfo.EcShardConfig.DataShards, volumeInfo.EcShardConfig.ParityShards) + } else if blockErr := ValidateBlockSize(volumeInfo.EcShardConfig.GetBlockSize()); blockErr != nil { + // A recorded block size that no encoder could have produced maps + // every read to the wrong shard offset. Refuse the mount rather + // than serve those bytes or silently pick a layout. + ev.Close() + return nil, fmt.Errorf("ec volume %d: %s: %w", vid, vifFileName, blockErr) } else { ev.ECContext = &ECContext{ Collection: collection, VolumeId: vid, DataShards: ds, ParityShards: ps, + BlockSize: volumeInfo.EcShardConfig.GetBlockSize(), } glog.V(1).Infof("Loaded EC config from VolumeInfo for volume %d: %s", vid, ev.ECContext.String()) } } else { - ev.ECContext = NewDefaultECContext(collection, vid) + // A vif that carries no ecShardConfig answers nothing about the + // layout — it is no more informative than an absent one, so it + // must not skip the sidecar. Going straight to the defaults here + // read a uniform volume with the legacy offset math. + cfg, sidecarFound, sidecarErr := layoutFromSidecar(dataBaseFileName, indexBaseFileName) + if sidecarErr != nil { + ev.Close() + return nil, fmt.Errorf("ec volume %d: %s records no EC config and the bitrot sidecar cannot establish the layout: %w", vid, vifFileName, sidecarErr) + } + if sidecarFound { + ev.ECContext = &ECContext{ + Collection: collection, + VolumeId: vid, + DataShards: int(cfg.GetDataShards()), + ParityShards: int(cfg.GetParityShards()), + BlockSize: cfg.GetBlockSize(), + } + ev.EncodeTsNs = cfg.GetEncodeTsNs() + glog.V(0).Infof("ec volume %d: .vif records no EC config; took it from the bitrot sidecar: %s", + vid, ev.ECContext.String()) + } else { + ev.ECContext = NewDefaultECContext(collection, vid) + } } } else { // Don't fabricate a stub .vif here: a version-only stub implies the @@ -207,14 +251,48 @@ func NewEcVolume(diskType types.DiskType, dir string, dirIdx string, collection // mistake for an authoritative config. Mount with in-memory defaults and // leave the real .vif to the encoder or a recovery tool (the Rust volume // server already behaves this way). - glog.Warningf("vif file not found, using defaults, volumeId:%d, filename:%s", vid, vifFileName) + // + // The bitrot sidecar records the same EC config at encode time, so when + // it is present it answers the layout question the missing .vif cannot: + // defaulting a uniform-layout volume to the legacy block sizes maps + // every read to the wrong shard offset. `weed fix -ecx` reads the + // sidecar for the same reason. ev.ECContext = NewDefaultECContext(collection, vid) + cfg, sidecarFound, sidecarErr := layoutFromSidecar(dataBaseFileName, indexBaseFileName) + if sidecarErr != nil { + // With no .vif the sidecar is the ONLY record of this volume's + // layout. Present but unusable is not "assume legacy" — that + // answers reads with the wrong shard offsets, which is worse than + // not answering at all. + ev.Close() + return nil, fmt.Errorf("ec volume %d: no .vif and the bitrot sidecar cannot establish the layout: %w", vid, sidecarErr) + } + if sidecarFound { + ev.ECContext = &ECContext{ + Collection: collection, + VolumeId: vid, + DataShards: int(cfg.GetDataShards()), + ParityShards: int(cfg.GetParityShards()), + BlockSize: cfg.GetBlockSize(), + } + ev.EncodeTsNs = cfg.GetEncodeTsNs() + glog.V(0).Infof("ec volume %d: .vif missing; took EC config from the bitrot sidecar: %s", + vid, ev.ECContext.String()) + } else { + // Only now are the defaults what the volume actually mounted on; + // logging this after the sidecar answered would send an operator + // triaging wrong bytes after the legacy layout instead. + glog.Warningf("vif file not found, using defaults, volumeId:%d, filename:%s", vid, vifFileName) + } } ev.ShardLocations = make(map[ShardId][]pb.ServerAddress) // Load the active-generation bitrot checksum sidecar (optional). - ev.loadActiveBitrotSidecar() + if err := ev.loadActiveBitrotSidecar(); err != nil { + ev.Close() + return nil, err + } return } @@ -528,11 +606,17 @@ func (ev *EcVolume) LocateEcShardNeedleInterval(version needle.Version, offset i shardSize = shard.ecdFileSize - 1 } // calculate the locations in the ec shards - intervals = LocateData(ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize, shardSize, offset, types.Size(needle.GetActualSize(size, version))) + intervals = LocateData(ev.ECContext.LargeBlockSize(), ev.ECContext.SmallBlockSize(), shardSize, offset, types.Size(needle.GetActualSize(size, version))) return } +// IntervalToShardIdAndOffset resolves an interval against this volume's shard +// block layout. +func (ev *EcVolume) IntervalToShardIdAndOffset(interval Interval) (ShardId, int64) { + return interval.ToShardIdAndOffset(ev.ECContext.LargeBlockSize(), ev.ECContext.SmallBlockSize()) +} + func (ev *EcVolume) FindNeedleFromEcx(needleId types.NeedleId) (offset types.Offset, size types.Size, err error) { offset, size, err = SearchNeedleFromSortedIndex(ev.ecxFile, ev.ecxFileSize, needleId, nil) if err != nil { diff --git a/weed/storage/erasure_coding/ec_volume_scrub.go b/weed/storage/erasure_coding/ec_volume_scrub.go index 8cb33297e..40457f58e 100644 --- a/weed/storage/erasure_coding/ec_volume_scrub.go +++ b/weed/storage/erasure_coding/ec_volume_scrub.go @@ -238,7 +238,7 @@ func (ecv *EcVolume) ScrubLocal() (int64, []*volume_server_pb.EcShardInfo, []err localShardIds := []ShardId{} for i, iv := range locations { - sid, soffset := iv.ToShardIdAndOffset(ErasureCodingLargeBlockSize, ErasureCodingSmallBlockSize) + sid, soffset := ecv.IntervalToShardIdAndOffset(iv) ssize := int64(iv.Size.Raw()) shard, found := ecv.FindEcVolumeShard(sid) diff --git a/weed/storage/erasure_coding/ec_volume_sidecar_layout_test.go b/weed/storage/erasure_coding/ec_volume_sidecar_layout_test.go new file mode 100644 index 000000000..34e992416 --- /dev/null +++ b/weed/storage/erasure_coding/ec_volume_sidecar_layout_test.go @@ -0,0 +1,246 @@ +package erasure_coding + +import ( + "os" + "path/filepath" + "testing" + + "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb" + "github.com/seaweedfs/seaweedfs/weed/storage/needle" + "github.com/seaweedfs/seaweedfs/weed/storage/types" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" +) + +func uniformSidecar(t *testing.T, base string, ds, ps uint32, blockSize int64) { + t.Helper() + prot := &volume_server_pb.EcBitrotProtection{ + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 0, + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: ds, ParityShards: ps, BlockSize: blockSize, + }, + } + if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil { + t.Fatalf("seed .ecsum: %v", err) + } +} + +// The sidecar records the layout at encode time, so it answers what a .vif +// cannot. A .vif that merely omits ecShardConfig is no more informative than an +// absent one — going straight to the defaults reads a uniform volume with the +// legacy offset math and returns the wrong bytes. +func TestNewEcVolumeTakesLayoutFromSidecarWhenVifHasNoConfig(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + // A config-free .vif: version and dat size only, as a legacy encode left it. + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + uniformSidecar(t, base, 12, 4, 3*1024*1024) + + ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)) + if err != nil { + t.Fatalf("mount: %v", err) + } + defer ev.Close() + if got := ev.ECContext; got.DataShards != 12 || got.ParityShards != 4 || got.BlockSize != 3*1024*1024 { + t.Errorf("layout = %d+%d block %d, want 12+4 block 3MiB", got.DataShards, got.ParityShards, got.BlockSize) + } +} + +// A split -dir/-dir.idx layout keeps the sidecar with the INDEX. Probing only +// the data base reports "absent", and absent is the one answer that selects the +// legacy layout. +func TestNewEcVolumeFindsSidecarInTheIndexDirectory(t *testing.T) { + for _, tc := range []struct { + name string + writeVif bool + }{ + {name: "no vif at all"}, + {name: "vif without ec config", writeVif: true}, + } { + t.Run(tc.name, func(t *testing.T) { + dataDir, idxDir := t.TempDir(), t.TempDir() + idxBase := filepath.Join(idxDir, "2") + if err := os.WriteFile(idxBase+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if tc.writeVif { + if err := volume_info.SaveVolumeInfo(idxBase+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + } + uniformSidecar(t, idxBase, 12, 4, 3*1024*1024) + + ev, err := NewEcVolume(types.HardDriveType, dataDir, idxDir, "", needle.VolumeId(2)) + if err != nil { + t.Fatalf("mount: %v", err) + } + defer ev.Close() + if got := ev.ECContext; got.DataShards != 12 || got.ParityShards != 4 || got.BlockSize != 3*1024*1024 { + t.Errorf("layout = %d+%d block %d, want 12+4 block 3MiB", got.DataShards, got.ParityShards, got.BlockSize) + } + }) + } +} + +// Absence stays legal — a volume encoded before either record exists is a +// genuine legacy volume, and only that case may select the legacy layout. +func TestNewEcVolumeConfigFreeVifWithNoSidecarUsesDefaults(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "3") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(3)) + if err != nil { + t.Fatalf("mount: %v", err) + } + defer ev.Close() + if got := ev.ECContext; got.DataShards != DataShardsCount || got.ParityShards != ParityShardsCount || got.BlockSize != 0 { + t.Errorf("layout = %d+%d block %d, want the legacy defaults", got.DataShards, got.ParityShards, got.BlockSize) + } +} + +// Present but unusable must fail the mount rather than default, in the +// config-free-vif branch exactly as in the absent-vif one. +func TestNewEcVolumeConfigFreeVifWithUnusableSidecarFails(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "4") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + // Stamped for another generation: not a record of these shards. + if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), &volume_server_pb.EcBitrotProtection{ + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 5, + EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 12, ParityShards: 4}, + }); err != nil { + t.Fatalf("seed .ecsum: %v", err) + } + if _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(4)); err == nil { + t.Fatal("an unusable sole layout record must fail the mount") + } +} + +// Startup mirroring gives each shard-bearing disk its own .ecx/.ecj/.vif but +// deliberately not the .ecsum, and a repair delivers exactly one copy. A +// runtime restricted to its own two directories therefore reports no +// protection no matter how often it reloads — so the resolution has to reach +// the sibling disk that actually received the manifest. +func TestReloadBitrotSidecarFindsItOnASiblingDisk(t *testing.T) { + mine, sibling := t.TempDir(), t.TempDir() + myBase := filepath.Join(mine, "7") + if err := os.WriteFile(myBase+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(myBase+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: 0, + }, + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + + ev, err := NewEcVolume(types.HardDriveType, mine, mine, "", needle.VolumeId(7)) + if err != nil { + t.Fatalf("mount: %v", err) + } + defer ev.Close() + if got := ev.bitrotStatus; got != BitrotOff { + t.Fatalf("precondition: status = %v, want BitrotOff before the delivery", got) + } + + // The delivery lands the manifest on the sibling disk only. + prot := &volume_server_pb.EcBitrotProtection{ + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 0, + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: 0, + }, + } + for i := 0; i < 14; i++ { + prot.Shards = append(prot.Shards, &volume_server_pb.EcShardChecksums{ + ShardId: uint32(i), CoveredSize: 1, BlockCrc32C: make([]byte, 4), + }) + } + if err := SaveBitrotSidecar(BitrotSidecarPath(filepath.Join(sibling, "7"), 0), prot); err != nil { + t.Fatalf("seed sibling .ecsum: %v", err) + } + + // Reloading against only its own directories still finds nothing. + ev.ReloadBitrotSidecar() + if got := ev.bitrotStatus; got != BitrotOff { + t.Errorf("own-directories reload: status = %v, want BitrotOff", got) + } + + ev.ReloadBitrotSidecar(mine, sibling) + if got := ev.bitrotStatus; got != BitrotOn { + t.Errorf("cross-disk reload: status = %v, want BitrotOn", got) + } +} + +// "Does this volume already have a manifest" cannot be answered from one base +// name. The rebuild's opportunistic backfill asks it before writing a TOFU +// baseline, and a false "no" there writes a second sidecar at the data base +// that then shadows the real one — blessing whatever the shards currently say. +func TestFindBitrotSidecarSearchesEveryCandidate(t *testing.T) { + dataDir, idxDir, sibling := t.TempDir(), t.TempDir(), t.TempDir() + dataBase := filepath.Join(dataDir, "8") + idxBase := filepath.Join(idxDir, "8") + siblingBase := filepath.Join(sibling, "8") + + if got := FindBitrotSidecar(0, dataBase, idxBase, sibling); got != "" { + t.Errorf("no sidecar anywhere should answer empty, got %q", got) + } + + // Each location on its own: any one of them is enough to answer "yes". + for _, tc := range []struct{ name, base string }{ + {"data directory", dataBase}, + {"index directory", idxBase}, + {"sibling disk", siblingBase}, + } { + want := BitrotSidecarPath(tc.base, 0) + if err := os.WriteFile(want, []byte("x"), 0644); err != nil { + t.Fatalf("%s: seed: %v", tc.name, err) + } + if got := FindBitrotSidecar(0, dataBase, idxBase, sibling); got != want { + t.Errorf("%s: got %q, want %q", tc.name, got, want) + } + if err := os.Remove(want); err != nil { + t.Fatalf("%s: cleanup: %v", tc.name, err) + } + } + + // With copies everywhere the data base wins, which is the order every + // resolver uses — so a stray copy there shadows the others. + for _, base := range []string{siblingBase, idxBase, dataBase} { + if err := os.WriteFile(BitrotSidecarPath(base, 0), []byte("x"), 0644); err != nil { + t.Fatalf("seed %s: %v", base, err) + } + } + if got, want := FindBitrotSidecar(0, dataBase, idxBase, sibling), BitrotSidecarPath(dataBase, 0); got != want { + t.Errorf("precedence: got %q, want the data base %q", got, want) + } +} diff --git a/weed/storage/erasure_coding/ec_volume_vif_test.go b/weed/storage/erasure_coding/ec_volume_vif_test.go new file mode 100644 index 000000000..e054d7612 --- /dev/null +++ b/weed/storage/erasure_coding/ec_volume_vif_test.go @@ -0,0 +1,187 @@ +package erasure_coding + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb" + "github.com/seaweedfs/seaweedfs/weed/storage/needle" + "github.com/seaweedfs/seaweedfs/weed/storage/types" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" +) + +// A present-but-malformed .vif must FAIL the mount: every new encode +// records a positive uniform block size there, and silently defaulting +// to the legacy layout would serve those shards with the wrong offset +// math. Absence stays legal — legacy volumes predate the sidecar. +func TestNewEcVolumeFailsOnMalformedVif(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := os.WriteFile(base+".vif", []byte("not json"), 0644); err != nil { + t.Fatalf("seed .vif: %v", err) + } + _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)) + if err == nil { + t.Fatal("mount over a malformed .vif must fail") + } + if !strings.Contains(err.Error(), ".vif") { + t.Errorf("error should name the vif: %v", err) + } +} + +func TestNewEcVolumeAbsentVifMountsWithDefaults(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)) + if err != nil { + t.Fatalf("legacy mount without .vif must succeed: %v", err) + } + defer ev.Close() + if ev.ECContext.BlockSize != 0 { + t.Errorf("legacy mount must use the legacy layout: BlockSize=%d", ev.ECContext.BlockSize) + } +} + +// A block size no encoder could have produced (negative, or not a whole +// number of small blocks) maps every read to the wrong shard offset, so the +// mount refuses it rather than serving those bytes. +func TestNewEcVolumeRejectsImplausibleBlockSize(t *testing.T) { + for _, blockSize := range []int64{1, -1, 3*1024*1024 + 1} { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: blockSize, + }, + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + if _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)); err == nil { + t.Errorf("block size %d must fail the mount", blockSize) + } + } +} + +func TestNewEcVolumeAcceptsAlignedBlockSize(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: 3 * 1024 * 1024, + }, + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + ev, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)) + if err != nil { + t.Fatalf("aligned block size must mount: %v", err) + } + defer ev.Close() + if ev.ECContext.BlockSize != 3*1024*1024 { + t.Errorf("BlockSize = %d, want 3MiB", ev.ECContext.BlockSize) + } +} + +// The .vif and the .ecsum both record the layout their generation was encoded +// with. When a generation-matching sidecar disagrees, one of them is wrong and +// reads through the other land at the wrong shard offsets — so the mount must +// refuse rather than quietly drop to unprotected reads. +func TestNewEcVolumeRejectsSidecarGeometryDisagreement(t *testing.T) { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("seed .ecx: %v", err) + } + if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.Version3), + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: 0, // says legacy + }, + }); err != nil { + t.Fatalf("seed .vif: %v", err) + } + // The sidecar for the same generation says uniform. + prot := &volume_server_pb.EcBitrotProtection{ + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 0, + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: 3 * 1024 * 1024, + }, + } + if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil { + t.Fatalf("seed .ecsum: %v", err) + } + + _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)) + if err == nil { + t.Fatal("a generation-matching layout disagreement must fail the mount") + } + if !strings.Contains(err.Error(), "block") { + t.Errorf("the error should name the disagreement: %v", err) + } +} + +// With no .vif the bitrot sidecar is the ONLY record of the volume's layout. +// Present but unusable — unreadable, stamped for another generation, or +// carrying an incomplete ratio — must fail the mount: answering reads from a +// guessed layout returns wrong bytes rather than none. Genuine absence stays +// legal, and is covered by TestNewEcVolumeAbsentVifMountsWithDefaults. +func TestNewEcVolumeRejectsUnusableSoleSidecar(t *testing.T) { + seed := map[string]*volume_server_pb.EcBitrotProtection{ + "another generation": { + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 5, + EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10, ParityShards: 4}, + }, + "no ec config": { + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 0, + }, + "incomplete ratio": { + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 0, + EcShardConfig: &volume_server_pb.EcShardConfig{DataShards: 10}, + }, + "invalid block size": { + Algorithm: volume_server_pb.ChecksumAlgorithm_CHECKSUM_CRC32C, + BlockSize: uint32(BitrotBlockSize), + Generation: 0, + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: 10, ParityShards: 4, BlockSize: 3*1024*1024 + 1, + }, + }, + } + for name, prot := range seed { + dir := t.TempDir() + base := filepath.Join(dir, "1") + if err := os.WriteFile(base+".ecx", []byte{}, 0644); err != nil { + t.Fatalf("%s: seed .ecx: %v", name, err) + } + if err := SaveBitrotSidecar(BitrotSidecarPath(base, 0), prot); err != nil { + t.Fatalf("%s: seed .ecsum: %v", name, err) + } + if _, err := NewEcVolume(types.HardDriveType, dir, dir, "", needle.VolumeId(1)); err == nil { + t.Errorf("%s: an unusable sole layout record must fail the mount", name) + } + } +} diff --git a/weed/storage/erasure_coding/verification.go b/weed/storage/erasure_coding/verification.go index ac6fa6ecb..76c52a2e1 100644 --- a/weed/storage/erasure_coding/verification.go +++ b/weed/storage/erasure_coding/verification.go @@ -14,6 +14,11 @@ import ( type ServerShardInventory struct { Bits ShardBits QueryError error + // BlockSize is the shard block layout this holder reports for the volume — + // the one it will serve reads through. A binary predating the uniform + // layout drops the field off the wire and reports 0, which is how + // RequireAgreedBlockLayout tells such a holder from one that agrees. + BlockSize int64 } // Query errors are recorded per-server and treated as zero shards rather @@ -50,6 +55,7 @@ func VerifyShardsAcrossServers(ctx context.Context, volumeID uint32, } inv.Bits = inv.Bits.Set(ShardId(s.ShardId)) } + inv.BlockSize = resp.GetEcShardConfig().GetBlockSize() return nil }) if callErr != nil { @@ -99,6 +105,39 @@ func RequireRecoverableShardSet(volumeID uint32, shardsPresent ShardBits, dataSh volumeID, totalShards-len(missing), totalShards, dataShards, missing) } +// RequireAgreedBlockLayout gates source-volume deletion on every holder that +// answered agreeing with the layout the shards were encoded under. +// +// The uniform block layout lives in a `.vif` field that older volume servers +// have never heard of: such a server parses the file, silently discards the +// unknown field, and mounts the volume on the legacy 1GiB/1MiB striping. Its +// reads then land at the wrong shard offsets and return wrong bytes, and +// nothing else in the encode path notices — the shard files are the same +// length under either layout. Asking each holder which layout it is serving is +// the one question that separates the two, and asking it here means the answer +// arrives while the source .dat is still on disk. +// +// Holders that could not be reached, or that report no shard of this volume, +// are skipped: RequireRecoverableShardSet already covers a missing holder, and +// one that answered nothing has promised nothing. +func RequireAgreedBlockLayout(volumeID uint32, encodedBlockSize int64, perServer map[string]ServerShardInventory) error { + disagreeing := make([]string, 0, len(perServer)) + for server, inv := range perServer { + if inv.QueryError != nil || inv.Bits.Count() == 0 { + continue + } + if inv.BlockSize != encodedBlockSize { + disagreeing = append(disagreeing, fmt.Sprintf("%s serves block size %d", server, inv.BlockSize)) + } + } + if len(disagreeing) == 0 { + return nil + } + sort.Strings(disagreeing) + return fmt.Errorf("volume %d was encoded with shard block size %d but %v; upgrade every volume server to a build that understands the uniform shard block layout before encoding", + volumeID, encodedBlockSize, disagreeing) +} + func SummarizeShardInventory(perServer map[string]ServerShardInventory) string { servers := make([]string, 0, len(perServer)) for s := range perServer { diff --git a/weed/storage/store_ec.go b/weed/storage/store_ec.go index 0abd9fc43..9ab91b735 100644 --- a/weed/storage/store_ec.go +++ b/weed/storage/store_ec.go @@ -345,6 +345,49 @@ func (s *Store) FindEcVolumeWithShard(vid needle.VolumeId, shardId erasure_codin return nil, nil, false } +// EcMetadataDirs lists every directory on this server that could hold an EC +// volume's metadata — each disk's data and index directory. Startup mirroring +// gives each shard-bearing disk its own .ecx/.ecj/.vif, but the checksum +// sidecar is not mirrored and a repair delivers exactly one copy, so a runtime +// looking only at its own two directories cannot see it. Handing this list to +// the sidecar resolution keeps one authoritative copy reachable from every +// runtime rather than duplicating a file that is rewritten as shards are +// repaired and generations published. +func (s *Store) EcMetadataDirs() []string { + var dirs []string + appendDir := func(dir string) { + if dir == "" { + return + } + for _, existing := range dirs { + if existing == dir { + return + } + } + dirs = append(dirs, dir) + } + for _, location := range s.Locations { + appendDir(location.Directory) + appendDir(location.IdxDirectory) + } + return dirs +} + +// FindAllEcVolumes returns every per-disk *EcVolume the store maps for vid. A vid +// can mount on N disks as N distinct runtimes, and the first-match FindEcVolume +// hides the siblings — so anything that has to reach the whole volume, rather +// than any one runtime of it, iterates this instead. Order mirrors the +// deterministic s.Locations order; nil when no disk holds the vid. +func (s *Store) FindAllEcVolumes(vid needle.VolumeId) []*erasure_coding.EcVolume { + var evs []*erasure_coding.EcVolume + for _, location := range s.Locations { + if ev, found := location.FindEcVolume(vid); found { + evs = append(evs, ev) + } + } + return evs +} + func (s *Store) FindEcVolume(vid needle.VolumeId) (*erasure_coding.EcVolume, bool) { for _, location := range s.Locations { if s, found := location.FindEcVolume(vid); found { @@ -436,10 +479,6 @@ func (s *Store) ReadEcShardNeedle(vid needle.VolumeId, n *needle.Needle, onReadS return 0, fmt.Errorf("ec shard %d not found", vid) } -func (s *Store) IntervalToShardIdAndOffset(iv erasure_coding.Interval) (erasure_coding.ShardId, int64) { - return iv.ToShardIdAndOffset(erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize) -} - var ( // ecRecoverBudget bounds the interval-sized buffers EC recovery holds across // all concurrent reads: a peer that is slow to fail keeps a whole fan-out of @@ -507,7 +546,7 @@ func (s *Store) readEcShardIntervals(needleId types.NeedleId, ecVolume *erasure_ // readOneEcShardInterval fills data, which must be interval.Size long. func (s *Store) readOneEcShardInterval(needleId types.NeedleId, ecVolume *erasure_coding.EcVolume, interval erasure_coding.Interval, data []byte) (is_deleted bool, err error) { - shardId, actualOffset := s.IntervalToShardIdAndOffset(interval) + shardId, actualOffset := ecVolume.IntervalToShardIdAndOffset(interval) // try local read err = s.readLocalEcShardInterval(ecVolume, shardId, data, actualOffset) diff --git a/weed/storage/store_ec_delete.go b/weed/storage/store_ec_delete.go index da20ffd52..970967cb0 100644 --- a/weed/storage/store_ec_delete.go +++ b/weed/storage/store_ec_delete.go @@ -52,7 +52,7 @@ func (s *Store) doDeleteNeedleFromAtLeastOneRemoteEcShards(ecVolume *erasure_cod return erasure_coding.NotFoundError } - primaryShardId, _ := intervals[0].ToShardIdAndOffset(erasure_coding.ErasureCodingLargeBlockSize, erasure_coding.ErasureCodingSmallBlockSize) + primaryShardId, _ := ecVolume.IntervalToShardIdAndOffset(intervals[0]) // Normal path: delete on exactly one node holding the primary data shard. err = s.doDeleteNeedleFromRemoteEcShardServers(primaryShardId, ecVolume, needleId) diff --git a/weed/storage/store_ec_interval_read_test.go b/weed/storage/store_ec_interval_read_test.go index 8dba1bb03..dedb57a33 100644 --- a/weed/storage/store_ec_interval_read_test.go +++ b/weed/storage/store_ec_interval_read_test.go @@ -2,17 +2,112 @@ package storage import ( "bytes" + "math/rand" + "os" "testing" + "time" + "github.com/seaweedfs/seaweedfs/weed/pb" + "github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb" + "github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding" "github.com/seaweedfs/seaweedfs/weed/storage/needle" + "github.com/seaweedfs/seaweedfs/weed/storage/super_block" + "github.com/seaweedfs/seaweedfs/weed/storage/types" + "github.com/seaweedfs/seaweedfs/weed/storage/volume_info" ) +// encodeAndMountEcVolume writes needles into a fresh volume, EC-encodes it +// (uniform layout), stamps the .vif the way VolumeEcShardsGenerate does, and +// mounts every shard on the store. +func encodeAndMountEcVolume(t *testing.T, store *Store, vid needle.VolumeId, needles []*needle.Needle) *erasure_coding.EcVolume { + t.Helper() + dir := store.Locations[0].Directory + + v, err := NewVolume(dir, dir, "", vid, NeedleMapInMemory, &super_block.ReplicaPlacement{}, &needle.TTL{}, 0, needle.GetCurrentVersion(), 0, 0) + if err != nil { + t.Fatalf("new volume: %v", err) + } + for _, n := range needles { + if _, _, _, err := v.writeNeedle2(n, true, false, false); err != nil { + t.Fatalf("write needle: %v", err) + } + } + baseFileName := v.DataFileName() + v.Close() + + datSize, err := os.Stat(baseFileName + ".dat") + if err != nil { + t.Fatalf("stat .dat: %v", err) + } + ecCtx := erasure_coding.NewDefaultECContext("", vid) + if _, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx); err != nil { + t.Fatalf("write ec files: %v", err) + } + if err := erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"); err != nil { + t.Fatalf("write .ecx: %v", err) + } + if err := os.WriteFile(baseFileName+".ecj", nil, 0o644); err != nil { + t.Fatalf("write .ecj: %v", err) + } + if err := volume_info.SaveVolumeInfo(baseFileName+".vif", &volume_server_pb.VolumeInfo{ + Version: uint32(needle.GetCurrentVersion()), + DatFileSize: datSize.Size(), + EcShardConfig: &volume_server_pb.EcShardConfig{ + DataShards: uint32(ecCtx.DataShards), + ParityShards: uint32(ecCtx.ParityShards), + BlockSize: ecCtx.BlockSize, + }, + }); err != nil { + t.Fatalf("save .vif: %v", err) + } + for _, ext := range []string{".dat", ".idx"} { + if err := os.Remove(baseFileName + ext); err != nil { + t.Fatalf("remove %s: %v", ext, err) + } + } + + for shardId := 0; shardId < erasure_coding.TotalShardsCount; shardId++ { + if err := store.MountEcShards("", vid, erasure_coding.ShardId(shardId), ""); err != nil { + t.Fatalf("mount shard %d: %v", shardId, err) + } + } + ecVolume, found := store.Locations[0].FindEcVolume(vid) + if !found { + t.Fatal("ec volume not mounted") + } + if ecVolume.ECContext.BlockSize != ecCtx.BlockSize { + t.Fatalf("mounted BlockSize = %d, want %d", ecVolume.ECContext.BlockSize, ecCtx.BlockSize) + } + + // Every shard is local, so seed the location cache to keep the read off the + // master this test does not have. + ecVolume.ShardLocationsLock.Lock() + for shardId := 0; shardId < erasure_coding.TotalShardsCount; shardId++ { + ecVolume.ShardLocations[erasure_coding.ShardId(shardId)] = []pb.ServerAddress{"localhost:8080"} + } + ecVolume.ShardLocationsRefreshTime = time.Now() + ecVolume.ShardLocationsLock.Unlock() + return ecVolume +} + +func randomNeedleOfSize(id uint64, size int) *needle.Needle { + n := new(needle.Needle) + n.Id = types.Uint64ToNeedleId(id) + n.Data = make([]byte, size) + rand.New(rand.NewSource(int64(id))).Read(n.Data) + n.Checksum = needle.NewCRC(n.Data) + return n +} + // A needle larger than one EC block is split over consecutive blocks, which // live on different shards. The intervals are read concurrently, so check they // still come back in order. func TestReadEcShardNeedleSpanningBlocks(t *testing.T) { + store := newTestStore(t, 1) const vid = needle.VolumeId(7) - store, ecVolume, n := newLocalEcVolume(t, vid) + + n := randomNeedleOfSize(42, 3*erasure_coding.ErasureCodingSmallBlockSize+1234) + ecVolume := encodeAndMountEcVolume(t, store, vid, []*needle.Needle{n}) _, _, intervals, err := ecVolume.LocateEcShardNeedle(n.Id, ecVolume.Version) if err != nil { @@ -31,3 +126,52 @@ func TestReadEcShardNeedleSpanningBlocks(t *testing.T) { t.Fatalf("read back %d bytes, want the %d written", len(got.Data), len(n.Data)) } } + +// On a volume big enough for multi-MB blocks, the uniform layout keeps a +// needle smaller than the block in one interval on one shard, and reads every +// needle back intact. +func TestReadEcShardNeedleUniformLayout(t *testing.T) { + store := newTestStore(t, 1) + const vid = needle.VolumeId(8) + + // ~26MB of needles: block size becomes 3MB, so the layouts diverge and a + // 1MB needle would stripe across shards under the legacy layout. + var needles []*needle.Needle + for i := uint64(1); i <= 5; i++ { + needles = append(needles, randomNeedleOfSize(i, 4*1024*1024)) + } + for i := uint64(6); i <= 11; i++ { + needles = append(needles, randomNeedleOfSize(i, 1024*1024)) + } + ecVolume := encodeAndMountEcVolume(t, store, vid, needles) + if ecVolume.ECContext.BlockSize <= erasure_coding.ErasureCodingSmallBlockSize { + t.Fatalf("block size %d does not diverge from the legacy layout", ecVolume.ECContext.BlockSize) + } + + // With 3MB blocks a 1MB needle sits inside one block (bar a boundary + // straddle), where the legacy layout would stripe it at 1MB granularity. + singleInterval := false + for _, n := range needles[5:] { + _, _, intervals, err := ecVolume.LocateEcShardNeedle(n.Id, ecVolume.Version) + if err != nil { + t.Fatalf("locate needle %v: %v", n.Id, err) + } + if len(intervals) == 1 { + singleInterval = true + } + } + if !singleInterval { + t.Fatal("no 1MB needle mapped to a single interval; uniform layout not in effect") + } + + for _, n := range needles { + got := new(needle.Needle) + got.Id = n.Id + if _, err := store.ReadEcShardNeedle(vid, got, nil); err != nil { + t.Fatalf("read needle %v: %v", n.Id, err) + } + if !bytes.Equal(got.Data, n.Data) { + t.Fatalf("needle %v: read back %d bytes that do not match the %d written", n.Id, len(got.Data), len(n.Data)) + } + } +} diff --git a/weed/storage/store_ec_recover_fanout_test.go b/weed/storage/store_ec_recover_fanout_test.go index c30a3f443..0bd75286e 100644 --- a/weed/storage/store_ec_recover_fanout_test.go +++ b/weed/storage/store_ec_recover_fanout_test.go @@ -95,7 +95,8 @@ func writeEcVolumeFiles(t *testing.T, dir string, vid needle.VolumeId) (baseFile if err != nil { t.Fatalf("stat .dat: %v", err) } - if _, err := erasure_coding.WriteEcFiles(baseFileName, erasure_coding.BackgroundECContext()); err != nil { + ecCtx := erasure_coding.BackgroundECContext() + if _, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx); err != nil { t.Fatalf("write ec files: %v", err) } if err := erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"); err != nil { @@ -110,6 +111,10 @@ func writeEcVolumeFiles(t *testing.T, dir string, vid needle.VolumeId) (baseFile EcShardConfig: &volume_server_pb.EcShardConfig{ DataShards: erasure_coding.DataShardsCount, ParityShards: erasure_coding.ParityShardsCount, + // The encode writes the uniform layout; a .vif that omitted its + // block size would mount these shards as legacy and read every + // interval at the wrong offset. + BlockSize: ecCtx.BlockSize, }, }); err != nil { t.Fatalf("save .vif: %v", err) @@ -162,7 +167,7 @@ func TestRecoverOneRemoteEcShardIntervalUsesLocalShards(t *testing.T) { if err != nil { t.Fatalf("locate needle: %v", err) } - shardIdToRecover, actualOffset := store.IntervalToShardIdAndOffset(intervals[0]) + shardIdToRecover, actualOffset := ecVolume.IntervalToShardIdAndOffset(intervals[0]) got := make([]byte, intervals[0].Size) nRead, _, err := store.recoverOneRemoteEcShardInterval(n.Id, ecVolume, shardIdToRecover, got, actualOffset) diff --git a/weed/storage/store_ec_scrub.go b/weed/storage/store_ec_scrub.go index 08c002682..4a5c11169 100644 --- a/weed/storage/store_ec_scrub.go +++ b/weed/storage/store_ec_scrub.go @@ -47,14 +47,14 @@ func (s *Store) ScrubEcVolume(vid needle.VolumeId, mode volume_server_pb.VolumeS shardIds := make([]erasure_coding.ShardId, len(intervals)) for i, iv := range intervals { - sid, _ := s.IntervalToShardIdAndOffset(iv) + sid, _ := ecv.IntervalToShardIdAndOffset(iv) shardIds[i] = sid } slices.Sort(shardIds) for i, iv := range intervals { chunk := make([]byte, iv.Size) - shardId, offset := s.IntervalToShardIdAndOffset(iv) + shardId, offset := ecv.IntervalToShardIdAndOffset(iv) // try a local shard read first... if err := s.readLocalEcShardInterval(ecv, shardId, chunk, offset); err == nil { diff --git a/weed/storage/store_ec_scrub_reads_test.go b/weed/storage/store_ec_scrub_reads_test.go index d8a1758b3..f30ff853a 100644 --- a/weed/storage/store_ec_scrub_reads_test.go +++ b/weed/storage/store_ec_scrub_reads_test.go @@ -45,7 +45,8 @@ func newLocalEcVolume(t *testing.T, vid needle.VolumeId) (*Store, *erasure_codin if err != nil { t.Fatalf("stat .dat: %v", err) } - if _, err := erasure_coding.WriteEcFiles(baseFileName, erasure_coding.BackgroundECContext()); err != nil { + ecCtx := erasure_coding.BackgroundECContext() + if _, err := erasure_coding.WriteEcFiles(baseFileName, ecCtx); err != nil { t.Fatalf("write ec files: %v", err) } if err := erasure_coding.WriteSortedFileFromIdx(baseFileName, ".ecx"); err != nil { @@ -60,6 +61,10 @@ func newLocalEcVolume(t *testing.T, vid needle.VolumeId) (*Store, *erasure_codin EcShardConfig: &volume_server_pb.EcShardConfig{ DataShards: erasure_coding.DataShardsCount, ParityShards: erasure_coding.ParityShardsCount, + // The encode writes the uniform layout; a .vif that omitted its + // block size would mount these shards as legacy and read every + // interval at the wrong offset. + BlockSize: ecCtx.BlockSize, }, }); err != nil { t.Fatalf("save .vif: %v", err) diff --git a/weed/worker/tasks/erasure_coding/ec_task.go b/weed/worker/tasks/erasure_coding/ec_task.go index ac8cbe787..b4ba0f126 100644 --- a/weed/worker/tasks/erasure_coding/ec_task.go +++ b/weed/worker/tasks/erasure_coding/ec_task.go @@ -51,6 +51,11 @@ type ErasureCodingTask struct { // pre-distribute sweep. deleteOriginalVolume skips these so it does not // re-delete and remove the now-EC .vif those servers share. emptyReplicasDeleted map[string]bool + + // encodedBlockSize is the shard block layout WriteEcFiles actually encoded + // with, read back off the EC context. Every holder must report serving the + // same one before the source volume may be deleted. + encodedBlockSize int64 } // NewErasureCodingTask creates a new unified EC task instance @@ -583,16 +588,23 @@ func (t *ErasureCodingTask) generateEcShardsLocally(localFiles map[string]string } // Generate EC shard files (.ec00 ~ .ec13) - ecBitrot, err := erasure_coding.WriteEcFiles(baseName, erasure_coding.BackgroundECContext()) + ecCtx := erasure_coding.BackgroundECContext() + ecBitrot, err := erasure_coding.WriteEcFiles(baseName, ecCtx) if err != nil { return nil, fmt.Errorf("failed to generate EC shard files: %w", err) } + // The layout the shards were actually written in, for the holder agreement + // check before the source is deleted. + t.encodedBlockSize = ecCtx.BlockSize // Persist the bitrot checksum sidecar (generation 0) alongside the shards so - // it travels with them during distribution. Best-effort: a failed sidecar - // write leaves the generation unprotected rather than failing the encode. + // it travels with them during distribution. Protection was asked for, and + // this write is orders of magnitude smaller than the shards that just + // landed: if it fails the disk is in trouble, and continuing would delete + // the source replicas in exchange for a generation that is both unprotected + // and missing the geometry record the .vif fallback reads. if erasure_coding.BitrotProtectionEnabled && ecBitrot != nil { if serr := erasure_coding.SaveBitrotSidecar(erasure_coding.BitrotSidecarPath(baseName, 0), ecBitrot); serr != nil { - glog.Warningf("failed to write EC bitrot sidecar for %s: %v", baseName, serr) + return nil, fmt.Errorf("write EC bitrot sidecar for %s: %w", baseName, serr) } } @@ -662,6 +674,7 @@ func (t *ErasureCodingTask) generateEcShardsLocally(localFiles map[string]string if ecBitrot != nil && ecBitrot.EcShardConfig != nil { ecShardConfig.DataShards = ecBitrot.EcShardConfig.DataShards ecShardConfig.ParityShards = ecBitrot.EcShardConfig.ParityShards + ecShardConfig.BlockSize = ecBitrot.EcShardConfig.BlockSize } volumeInfo := &volume_server_pb.VolumeInfo{ Version: uint32(needle.GetCurrentVersion()), @@ -675,30 +688,40 @@ func (t *ErasureCodingTask) generateEcShardsLocally(localFiles map[string]string } else { glog.Warningf("stat %s for .vif dat file size: %v", datFile, err) } + // The .vif carries the shard block layout the holders will read through, so + // neither the write nor the inclusion below is optional: an encode that + // distributed shards without it would leave every reader falling back to + // the legacy layout, and the task deletes the source replicas afterwards. if err := volume_info.SaveVolumeInfo(vifFile, volumeInfo); err != nil { - glog.Warningf("Failed to create .vif file: %v", err) - } else { - shardFiles["vif"] = vifFile - if info, err := os.Stat(vifFile); err == nil { - t.GetLogger().WithFields(map[string]interface{}{ - "file_type": "vif", - "file_path": vifFile, - "size_bytes": info.Size(), - }).Info("Volume info file generated") - } + return nil, fmt.Errorf("write %s: %w", vifFile, err) } + vifInfo, err := os.Stat(vifFile) + if err != nil { + return nil, fmt.Errorf("stat %s for distribution: %w", vifFile, err) + } + shardFiles["vif"] = vifFile + t.GetLogger().WithFields(map[string]interface{}{ + "file_type": "vif", + "file_path": vifFile, + "size_bytes": vifInfo.Size(), + }).Info("Volume info file generated") // Add the generation-0 bitrot checksum sidecar so it is distributed with // the shards (DistributeEcShards only ships files present in shardFiles). - // Best-effort like the sidecar write above: if it is absent the holders - // are simply unprotected rather than failing the encode. + // Strict when protection is enabled and the encoder produced a manifest — + // the write above already failed the encode otherwise — so the holders + // cannot end up with shards whose checksums stayed behind on the worker. ecsumFile := erasure_coding.BitrotSidecarPath(baseName, 0) - if info, err := os.Stat(ecsumFile); err == nil { + ecsumInfo, ecsumErr := os.Stat(ecsumFile) + if ecsumErr != nil && erasure_coding.BitrotProtectionEnabled && ecBitrot != nil { + return nil, fmt.Errorf("stat %s for distribution: %w", ecsumFile, ecsumErr) + } + if ecsumErr == nil { shardFiles["ecsum"] = ecsumFile t.GetLogger().WithFields(map[string]interface{}{ "file_type": "ecsum", "file_path": ecsumFile, - "size_bytes": info.Size(), + "size_bytes": ecsumInfo.Size(), }).Info("EC bitrot checksum sidecar generated") } @@ -768,6 +791,20 @@ func (t *ErasureCodingTask) verifyEcShardsBeforeDelete(ctx context.Context) erro "per_server": summary, }).Warning("EC shard set incomplete but recoverable; proceeding with source deletion") } + + // Before anything irreversible: every holder that answered must report + // serving the layout these shards were encoded in. A holder too old to + // know the uniform layout mounts them as legacy and returns wrong bytes, + // and the source volume is the only remaining correct copy. + if err := erasure_coding.RequireAgreedBlockLayout(t.volumeID, t.encodedBlockSize, perServer); err != nil { + t.GetLogger().WithFields(map[string]interface{}{ + "volume_id": t.volumeID, + "per_server": summary, + "error": err.Error(), + }).Error("EC holders disagree on the shard block layout — source volume will be kept") + return err + } + return nil }