oxedyne/fe2o3/fe2o3_o3db_sync/tests/blind_index.rs
14.8 KiB, 12 runs
created by r1870400018:21932, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! An index file that does not account for its data file must not read as an empty store. |
| 2 | //! |
| 3 | //! A zone is scanned by walking its index files. A data file that holds records beside an |
| 4 | //! index file of zero bytes therefore contributes nothing to a scan -- successfully, with no |
| 5 | //! error and no warning the caller can see -- while `get()` by key over those same records |
| 6 | //! goes on working perfectly. That combination is the signature: reads fine, scans empty. |
| 7 | //! |
| 8 | //! It is not a rare state. A store was left in it by an ordinary gate run: `zone_001` with a |
| 9 | //! `.dat` of 5,805 bytes holding two passcodes and three settings writes, and a `.ind` of |
| 10 | //! zero. One of those passcodes was successfully redeemed by key during the same run, while |
| 11 | //! the console answered `200` and `minted: 0` over the records that held the answer. The |
| 12 | //! gateway had said so at start-up, once, to the log: |
| 13 | //! |
| 14 | //! ```ignore |
| 15 | //! WARN bot_initgc.rs: InitGarbageBot:1:1: The index file 1 is empty, trying data file... |
| 16 | //! ``` |
| 17 | //! |
| 18 | //! Two defects meet there, and this test holds each of them separately. |
| 19 | //! |
| 20 | //! 1. **A scan cannot report an under-count.** The walk had no way to tell the caller that a |
| 21 | //! file it was asked to read accounted for none of what it held, so "nothing there" and "I |
| 22 | //! could not look" arrived as the same answer. |
| 23 | //! |
| 24 | //! 2. **The repair detached the writer.** Start-up does notice the empty index and rebuild it |
| 25 | //! from the data file -- but it used to rename a freshly written file over the old one, and |
| 26 | //! `ZoneBot::survey_files` hands the wbot its live `(data, index)` pair before |
| 27 | //! `init_caches` asks for the rebuild. So a wbot was already holding an append handle on |
| 28 | //! that index file, the rename replaced the inode underneath it, and every index record |
| 29 | //! written for the rest of the process went into an unlinked file. The data file is never |
| 30 | //! renamed and kept everything, which is why keyed reads stayed correct. Every restart |
| 31 | //! reproduced the state, and a restart is every deploy. |
| 32 | //! |
| 33 | //! The sharpest assertion here is the last: after the recovery, the index file on disk must |
| 34 | //! GROW when a record is written. That one does not go through the scan at all, so it cannot |
| 35 | //! be satisfied by the scan being fixed. |
| 36 | //! |
| 37 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 38 | //! Anthropic Claude |
| 39 | |
| 40 | use oxedyne_fe2o3_core::{ |
| 41 | prelude::*, |
| 42 | alt::Override, |
| 43 | }; |
| 44 | use oxedyne_fe2o3_crypto::enc::EncryptionScheme; |
| 45 | use oxedyne_fe2o3_hash::{ |
| 46 | csum::ChecksumScheme, |
| 47 | hash::HashScheme, |
| 48 | }; |
| 49 | use oxedyne_fe2o3_iop_db::api::{ |
| 50 | Database, |
| 51 | RestSchemesOverride, |
| 52 | ScanOpts, |
| 53 | }; |
| 54 | use oxedyne_fe2o3_jdat::prelude::*; |
| 55 | use oxedyne_fe2o3_o3db_sync::{ |
| 56 | base::index::ZoneInd, |
| 57 | data::core::RestSchemesInput, |
| 58 | file::{ |
| 59 | core::FileType, |
| 60 | zdir::ZoneDir, |
| 61 | }, |
| 62 | test::setup, |
| 63 | }; |
| 64 | |
| 65 | use std::{ |
| 66 | collections::BTreeMap, |
| 67 | fs::OpenOptions, |
| 68 | path::Path, |
| 69 | thread, |
| 70 | time::Duration, |
| 71 | }; |
| 72 | |
| 73 | // Records written before the index is emptied. |
| 74 | const KEYS: usize = 40; |
| 75 | |
| 76 | // The key written after the recovery, whose index record is the one that used |
| 77 | // to go into an unlinked file. |
| 78 | const AFTER: &str = "blind:after-recovery"; |
| 79 | |
| 80 | fn key(i: usize) -> Dat { |
| 81 | dat!(fmt!("blind:{:04}", i)) |
| 82 | } |
| 83 | |
| 84 | fn val(i: usize) -> Dat { |
| 85 | dat!(fmt!("value of blind record {:04}", i)) |
| 86 | } |
| 87 | |
| 88 | /// Total bytes held by the data files and by the index files of the given zones. |
| 89 | fn zone_bytes(zdirs: &BTreeMap<ZoneInd, ZoneDir>) -> Outcome<(u64, u64)> { |
| 90 | let mut dat_bytes = 0u64; |
| 91 | let mut ind_bytes = 0u64; |
| 92 | for (_zind, zdir) in zdirs { |
| 93 | for entry in res!(std::fs::read_dir(&zdir.dir)) { |
| 94 | let entry = res!(entry); |
| 95 | let path = entry.path(); |
| 96 | if !path.is_file() || ZoneDir::is_gc_temp_file(&path) { |
| 97 | continue; |
| 98 | } |
| 99 | let (_fnum, ftyp) = match ZoneDir::ozone_file_number_and_type(&path) { |
| 100 | Ok(pair) => pair, |
| 101 | Err(_) => continue, |
| 102 | }; |
| 103 | let len = res!(entry.metadata()).len(); |
| 104 | match ftyp { |
| 105 | FileType::Data => dat_bytes += len, |
| 106 | FileType::Index => ind_bytes += len, |
| 107 | } |
| 108 | } |
| 109 | } |
| 110 | Ok((dat_bytes, ind_bytes)) |
| 111 | } |
| 112 | |
| 113 | /// Truncate every index file in the given zones to zero bytes, leaving the data files alone. |
| 114 | /// |
| 115 | /// This is the on-disk state the gate run left, produced deliberately rather than waited for. |
| 116 | /// The files are truncated rather than deleted, because a missing index file and an empty one |
| 117 | /// are surveyed differently: a missing one is `Present::Solo(Data)` and an empty one is |
| 118 | /// `Present::Pair`, and it is the second that occurred in production. |
| 119 | fn empty_all_index_files(zdirs: &BTreeMap<ZoneInd, ZoneDir>) -> Outcome<usize> { |
| 120 | let mut n = 0; |
| 121 | for (_zind, zdir) in zdirs { |
| 122 | for entry in res!(std::fs::read_dir(&zdir.dir)) { |
| 123 | let entry = res!(entry); |
| 124 | let path = entry.path(); |
| 125 | if !path.is_file() || ZoneDir::is_gc_temp_file(&path) { |
| 126 | continue; |
| 127 | } |
| 128 | let (_fnum, ftyp) = match ZoneDir::ozone_file_number_and_type(&path) { |
| 129 | Ok(pair) => pair, |
| 130 | Err(_) => continue, |
| 131 | }; |
| 132 | if ftyp == FileType::Index { |
| 133 | let file = res!(OpenOptions::new().write(true).open(&path)); |
| 134 | res!(file.set_len(0)); |
| 135 | test!(sync_log::stream(), "Emptied index file {:?}.", path); |
| 136 | n += 1; |
| 137 | } |
| 138 | } |
| 139 | } |
| 140 | Ok(n) |
| 141 | } |
| 142 | |
| 143 | /// Runs as its own integration binary, because it restarts the database and wants its own |
| 144 | /// directory while it does: |
| 145 | /// |
| 146 | /// ```ignore |
| 147 | /// cargo test -p oxedyne_fe2o3_o3db_sync --test blind_index -- --nocapture |
| 148 | /// ``` |
| 149 | #[test] |
| 150 | fn main() -> Outcome<()> { |
| 151 | |
| 152 | log_set_level!("test"); |
| 153 | |
| 154 | let outcome = test_blind_index(); |
| 155 | |
| 156 | log_finish_wait!(); |
| 157 | |
| 158 | outcome |
| 159 | } |
| 160 | |
| 161 | pub fn test_blind_index() -> Outcome<()> { |
| 162 | |
| 163 | let dir = "./test_db_blind_index"; |
| 164 | let db_root = res!(Path::new(dir).canonicalize().or_else(|_| { |
| 165 | ok!(std::fs::create_dir_all(dir)); |
| 166 | Path::new(dir).canonicalize() |
| 167 | })); |
| 168 | |
| 169 | // Fixed key so the test is deterministic. |
| 170 | let enckey = [0x5au8; 32]; |
| 171 | let aes_gcm = res!(EncryptionScheme::new_aes_256_gcm_with_key(&enckey[..])); |
| 172 | let crc32 = ChecksumScheme::new_crc32(); |
| 173 | let schms2: RestSchemesOverride<EncryptionScheme, HashScheme> = |
| 174 | RestSchemesOverride::default() |
| 175 | .set_encrypter(Override::Default(aes_gcm.clone())); |
| 176 | let schms2 = Some(&schms2); |
| 177 | let user = setup::Uid::default(); |
| 178 | |
| 179 | let schms_input = RestSchemesInput::new( |
| 180 | Some(aes_gcm.clone()), |
| 181 | None::<HashScheme>, |
| 182 | None::<HashScheme>, |
| 183 | Some(crc32.clone()), |
| 184 | ); |
| 185 | |
| 186 | let mut cfg = res!(setup::default_cfg()); |
| 187 | // One zone and one bot of each kind: there is then exactly one live file, and the file |
| 188 | // whose index is emptied is the same one the writer holds open. That is the production |
| 189 | // case, and with more files the detached-handle defect merely becomes intermittent. |
| 190 | cfg.num_zones = 1; |
| 191 | cfg.num_cbots_per_zone = 1; |
| 192 | cfg.num_fbots_per_zone = 1; |
| 193 | cfg.num_wbots_per_zone = 1; |
| 194 | cfg.num_igbots_per_zone = 1; |
| 195 | cfg.num_scbots_per_zone = 1; |
| 196 | // Far larger than this test writes, so every record lands in file 1 and file 1 is still |
| 197 | // "incomplete" on the restart -- which is what makes the writer take it as its live file |
| 198 | // rather than starting a fresh one. |
| 199 | cfg.data_file_max_bytes = 1_000_000; |
| 200 | // Values are tiny; keep chunking out of the way entirely. |
| 201 | cfg.rest_chunk_threshold = 100_000; |
| 202 | cfg.rest_chunk_bytes = 10_000; |
| 203 | // Keep the zone inside this test's own root. The default configuration sends zone 1 to a |
| 204 | // container shared with the other tests. |
| 205 | cfg.zone_overrides = DaticleMap::new(); |
| 206 | |
| 207 | test!(sync_log::stream(), "+---------------------------------------------+"); |
| 208 | test!(sync_log::stream(), "| BLIND INDEX TEST |"); |
| 209 | test!(sync_log::stream(), "+---------------------------------------------+"); |
| 210 | |
| 211 | let zdirs: BTreeMap<ZoneInd, ZoneDir>; |
| 212 | |
| 213 | // ── Session 1: an emptied index must not read as an empty store ── |
| 214 | { |
| 215 | let mut db = res!(setup::start_db( |
| 216 | db_root.clone(), |
| 217 | Some(cfg.clone()), |
| 218 | schms_input.clone(), |
| 219 | None, |
| 220 | false, // gc off: nothing here is about collection |
| 221 | true, // wipe: start from nothing |
| 222 | )); |
| 223 | |
| 224 | for i in 0..KEYS { |
| 225 | res!(db.insert(key(i), val(i), user, schms2)); |
| 226 | } |
| 227 | thread::sleep(Duration::from_millis(500)); |
| 228 | |
| 229 | zdirs = res!(db.api().get_zone_dirs()); |
| 230 | |
| 231 | // The instrument works before the fault is introduced. Without this, a scan that |
| 232 | // failed for any other reason would satisfy the assertion below. |
| 233 | let entries = res!(db.scan(&ScanOpts::all(), schms2)); |
| 234 | if entries.len() != KEYS { |
| 235 | return Err(err!( |
| 236 | "Before the index was touched, a scan returned {} entries, expected {}. \ |
| 237 | The rest of this test cannot mean anything.", entries.len(), KEYS; |
| 238 | Test, Mismatch)); |
| 239 | } |
| 240 | |
| 241 | let (dat_bytes, ind_bytes) = res!(zone_bytes(&zdirs)); |
| 242 | test!(sync_log::stream(), "Zone holds {} bytes of data and {} bytes of index.", |
| 243 | dat_bytes, ind_bytes); |
| 244 | if dat_bytes == 0 { |
| 245 | return Err(err!( |
| 246 | "The zone holds no data at all, so emptying its index proves nothing."; |
| 247 | Test, Missing, Data)); |
| 248 | } |
| 249 | |
| 250 | // 1. The fault: index files at zero, data files untouched. |
| 251 | let emptied = res!(empty_all_index_files(&zdirs)); |
| 252 | if emptied == 0 { |
| 253 | return Err(err!( |
| 254 | "No index files were found to empty, so the fault was never introduced."; |
| 255 | Test, Missing, File)); |
| 256 | } |
| 257 | let (dat_after, ind_after) = res!(zone_bytes(&zdirs)); |
| 258 | if ind_after != 0 || dat_after != dat_bytes { |
| 259 | return Err(err!( |
| 260 | "After emptying, the zone should hold {} bytes of data and no index; it \ |
| 261 | holds {} and {}.", dat_bytes, dat_after, ind_after; |
| 262 | Test, Mismatch, Data)); |
| 263 | } |
| 264 | |
| 265 | // A scan over a store whose records are all still there must not answer as though |
| 266 | // they are not. This is the assertion the old walk could not make: it returned an |
| 267 | // empty list and `Ok`. |
| 268 | match db.scan(&ScanOpts::all(), schms2) { |
| 269 | Ok(entries) => return Err(err!( |
| 270 | "A scan over {} bytes of records whose index files are empty answered \ |
| 271 | successfully with {} entries. A caller cannot tell that from an empty \ |
| 272 | store, and this is exactly how an operator's new limit was accepted and \ |
| 273 | then never enforced.", dat_bytes, entries.len(); |
| 274 | Test, Invalid, Data)), |
| 275 | Err(e) => test!(sync_log::stream(), |
| 276 | "Scan refused to answer over a blind zone, as it must: {}", e), |
| 277 | } |
| 278 | |
| 279 | // And the signature that makes it dangerous: every record is still readable by key |
| 280 | // throughout. If this half fails, the store really was damaged and the test is |
| 281 | // about something else. |
| 282 | for i in 0..KEYS { |
| 283 | match res!(db.get(&key(i), schms2)) { |
| 284 | Some((v, _)) => if v != val(i) { |
| 285 | return Err(err!( |
| 286 | "Key {:?} read back as {:?}, expected {:?}.", key(i), v, val(i); |
| 287 | Test, Mismatch, Data)); |
| 288 | }, |
| 289 | None => return Err(err!( |
| 290 | "Key {:?} is unreadable, so the store is damaged rather than merely \ |
| 291 | unscannable, and this test is measuring the wrong thing.", key(i); |
| 292 | Test, Missing, Data)), |
| 293 | } |
| 294 | } |
| 295 | |
| 296 | res!(db.shutdown()); |
| 297 | } |
| 298 | |
| 299 | thread::sleep(Duration::from_secs(1)); |
| 300 | |
| 301 | // ── Session 2: the recovery repairs the index, and writes reach it ── |
| 302 | { |
| 303 | let mut db = res!(setup::start_db( |
| 304 | db_root.clone(), |
| 305 | Some(cfg.clone()), |
| 306 | schms_input.clone(), |
| 307 | None, |
| 308 | false, // gc off |
| 309 | false, // do not wipe: read what session 1 left |
| 310 | )); |
| 311 | |
| 312 | // 2. Start-up rebuilt the index from the data file, so the zone scans again. |
| 313 | let entries = res!(db.scan(&ScanOpts::all(), schms2)); |
| 314 | if entries.len() != KEYS { |
| 315 | return Err(err!( |
| 316 | "After the empty-index recovery, a scan returned {} entries, expected {}. \ |
| 317 | The index was not rebuilt from the data file.", entries.len(), KEYS; |
| 318 | Test, Mismatch, Data)); |
| 319 | } |
| 320 | |
| 321 | let (_dat_rebuilt, ind_rebuilt) = res!(zone_bytes(&zdirs)); |
| 322 | if ind_rebuilt == 0 { |
| 323 | return Err(err!( |
| 324 | "After the empty-index recovery, the index files still hold no bytes."; |
| 325 | Test, Missing, Data)); |
| 326 | } |
| 327 | test!(sync_log::stream(), "Index rebuilt to {} bytes.", ind_rebuilt); |
| 328 | |
| 329 | // 3. THE SHARP ONE. A record written after the recovery must reach the index file |
| 330 | // on disk. When the rebuild renamed a fresh file over the index, the writer's |
| 331 | // append handle kept pointing at the replaced inode and this file never grew |
| 332 | // again for the life of the process -- while the data file took every record, so |
| 333 | // nothing else looked wrong. |
| 334 | res!(db.insert( |
| 335 | dat!(AFTER), |
| 336 | dat!("written after the index was rebuilt"), |
| 337 | user, |
| 338 | schms2, |
| 339 | )); |
| 340 | thread::sleep(Duration::from_millis(500)); |
| 341 | |
| 342 | let (_dat_after, ind_after) = res!(zone_bytes(&zdirs)); |
| 343 | if ind_after <= ind_rebuilt { |
| 344 | return Err(err!( |
| 345 | "A record was written after the empty-index recovery and the index files \ |
| 346 | went from {} bytes to {}. The write did not reach the index: it went to a \ |
| 347 | file handle the rebuild had detached, so it is on no index at all and no \ |
| 348 | scan will ever see it.", ind_rebuilt, ind_after; |
| 349 | Test, Missing, Data)); |
| 350 | } |
| 351 | |
| 352 | // And the same thing said through the scan, which is where a caller would notice. |
| 353 | let entries = res!(db.scan(&ScanOpts::all(), schms2)); |
| 354 | if entries.len() != KEYS + 1 { |
| 355 | return Err(err!( |
| 356 | "After the recovery and one further write, a scan returned {} entries, \ |
| 357 | expected {}.", entries.len(), KEYS + 1; |
| 358 | Test, Mismatch, Data)); |
| 359 | } |
| 360 | let after = dat!(AFTER); |
| 361 | if !entries.iter().any(|(k, _v, _m)| *k == after) { |
| 362 | return Err(err!( |
| 363 | "The record written after the recovery is missing from the scan, though \ |
| 364 | the count came out right."; |
| 365 | Test, Missing, Data)); |
| 366 | } |
| 367 | |
| 368 | res!(db.shutdown()); |
| 369 | } |
| 370 | |
| 371 | Ok(()) |
| 372 | } |