Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_o3db_sync/tests/blind_index.rs

14.8 KiB, 12 runs

created by r1870400018:21932, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! An index file that does not account for its data file must not read as an empty store.
2//!
3//! A zone is scanned by walking its index files. A data file that holds records beside an
4//! index file of zero bytes therefore contributes nothing to a scan -- successfully, with no
5//! error and no warning the caller can see -- while `get()` by key over those same records
6//! goes on working perfectly. That combination is the signature: reads fine, scans empty.
7//!
8//! It is not a rare state. A store was left in it by an ordinary gate run: `zone_001` with a
9//! `.dat` of 5,805 bytes holding two passcodes and three settings writes, and a `.ind` of
10//! zero. One of those passcodes was successfully redeemed by key during the same run, while
11//! the console answered `200` and `minted: 0` over the records that held the answer. The
12//! gateway had said so at start-up, once, to the log:
13//!
14//! ```ignore
15//! WARN bot_initgc.rs: InitGarbageBot:1:1: The index file 1 is empty, trying data file...
16//! ```
17//!
18//! Two defects meet there, and this test holds each of them separately.
19//!
20//! 1. **A scan cannot report an under-count.** The walk had no way to tell the caller that a
21//! file it was asked to read accounted for none of what it held, so "nothing there" and "I
22//! could not look" arrived as the same answer.
23//!
24//! 2. **The repair detached the writer.** Start-up does notice the empty index and rebuild it
25//! from the data file -- but it used to rename a freshly written file over the old one, and
26//! `ZoneBot::survey_files` hands the wbot its live `(data, index)` pair before
27//! `init_caches` asks for the rebuild. So a wbot was already holding an append handle on
28//! that index file, the rename replaced the inode underneath it, and every index record
29//! written for the rest of the process went into an unlinked file. The data file is never
30//! renamed and kept everything, which is why keyed reads stayed correct. Every restart
31//! reproduced the state, and a restart is every deploy.
32//!
33//! The sharpest assertion here is the last: after the recovery, the index file on disk must
34//! GROW when a record is written. That one does not go through the scan at all, so it cannot
35//! be satisfied by the scan being fixed.
36//!
37//! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\
38//! Anthropic Claude
39
40use oxedyne_fe2o3_core::{
41 prelude::*,
42 alt::Override,
43};
44use oxedyne_fe2o3_crypto::enc::EncryptionScheme;
45use oxedyne_fe2o3_hash::{
46 csum::ChecksumScheme,
47 hash::HashScheme,
48};
49use oxedyne_fe2o3_iop_db::api::{
50 Database,
51 RestSchemesOverride,
52 ScanOpts,
53};
54use oxedyne_fe2o3_jdat::prelude::*;
55use oxedyne_fe2o3_o3db_sync::{
56 base::index::ZoneInd,
57 data::core::RestSchemesInput,
58 file::{
59 core::FileType,
60 zdir::ZoneDir,
61 },
62 test::setup,
63};
64
65use std::{
66 collections::BTreeMap,
67 fs::OpenOptions,
68 path::Path,
69 thread,
70 time::Duration,
71};
72
73// Records written before the index is emptied.
74const KEYS: usize = 40;
75
76// The key written after the recovery, whose index record is the one that used
77// to go into an unlinked file.
78const AFTER: &str = "blind:after-recovery";
79
80fn key(i: usize) -> Dat {
81 dat!(fmt!("blind:{:04}", i))
82}
83
84fn val(i: usize) -> Dat {
85 dat!(fmt!("value of blind record {:04}", i))
86}
87
88/// Total bytes held by the data files and by the index files of the given zones.
89fn zone_bytes(zdirs: &BTreeMap<ZoneInd, ZoneDir>) -> Outcome<(u64, u64)> {
90 let mut dat_bytes = 0u64;
91 let mut ind_bytes = 0u64;
92 for (_zind, zdir) in zdirs {
93 for entry in res!(std::fs::read_dir(&zdir.dir)) {
94 let entry = res!(entry);
95 let path = entry.path();
96 if !path.is_file() || ZoneDir::is_gc_temp_file(&path) {
97 continue;
98 }
99 let (_fnum, ftyp) = match ZoneDir::ozone_file_number_and_type(&path) {
100 Ok(pair) => pair,
101 Err(_) => continue,
102 };
103 let len = res!(entry.metadata()).len();
104 match ftyp {
105 FileType::Data => dat_bytes += len,
106 FileType::Index => ind_bytes += len,
107 }
108 }
109 }
110 Ok((dat_bytes, ind_bytes))
111}
112
113/// Truncate every index file in the given zones to zero bytes, leaving the data files alone.
114///
115/// This is the on-disk state the gate run left, produced deliberately rather than waited for.
116/// The files are truncated rather than deleted, because a missing index file and an empty one
117/// are surveyed differently: a missing one is `Present::Solo(Data)` and an empty one is
118/// `Present::Pair`, and it is the second that occurred in production.
119fn empty_all_index_files(zdirs: &BTreeMap<ZoneInd, ZoneDir>) -> Outcome<usize> {
120 let mut n = 0;
121 for (_zind, zdir) in zdirs {
122 for entry in res!(std::fs::read_dir(&zdir.dir)) {
123 let entry = res!(entry);
124 let path = entry.path();
125 if !path.is_file() || ZoneDir::is_gc_temp_file(&path) {
126 continue;
127 }
128 let (_fnum, ftyp) = match ZoneDir::ozone_file_number_and_type(&path) {
129 Ok(pair) => pair,
130 Err(_) => continue,
131 };
132 if ftyp == FileType::Index {
133 let file = res!(OpenOptions::new().write(true).open(&path));
134 res!(file.set_len(0));
135 test!(sync_log::stream(), "Emptied index file {:?}.", path);
136 n += 1;
137 }
138 }
139 }
140 Ok(n)
141}
142
143/// Runs as its own integration binary, because it restarts the database and wants its own
144/// directory while it does:
145///
146/// ```ignore
147/// cargo test -p oxedyne_fe2o3_o3db_sync --test blind_index -- --nocapture
148/// ```
149#[test]
150fn main() -> Outcome<()> {
151
152 log_set_level!("test");
153
154 let outcome = test_blind_index();
155
156 log_finish_wait!();
157
158 outcome
159}
160
161pub fn test_blind_index() -> Outcome<()> {
162
163 let dir = "./test_db_blind_index";
164 let db_root = res!(Path::new(dir).canonicalize().or_else(|_| {
165 ok!(std::fs::create_dir_all(dir));
166 Path::new(dir).canonicalize()
167 }));
168
169 // Fixed key so the test is deterministic.
170 let enckey = [0x5au8; 32];
171 let aes_gcm = res!(EncryptionScheme::new_aes_256_gcm_with_key(&enckey[..]));
172 let crc32 = ChecksumScheme::new_crc32();
173 let schms2: RestSchemesOverride<EncryptionScheme, HashScheme> =
174 RestSchemesOverride::default()
175 .set_encrypter(Override::Default(aes_gcm.clone()));
176 let schms2 = Some(&schms2);
177 let user = setup::Uid::default();
178
179 let schms_input = RestSchemesInput::new(
180 Some(aes_gcm.clone()),
181 None::<HashScheme>,
182 None::<HashScheme>,
183 Some(crc32.clone()),
184 );
185
186 let mut cfg = res!(setup::default_cfg());
187 // One zone and one bot of each kind: there is then exactly one live file, and the file
188 // whose index is emptied is the same one the writer holds open. That is the production
189 // case, and with more files the detached-handle defect merely becomes intermittent.
190 cfg.num_zones = 1;
191 cfg.num_cbots_per_zone = 1;
192 cfg.num_fbots_per_zone = 1;
193 cfg.num_wbots_per_zone = 1;
194 cfg.num_igbots_per_zone = 1;
195 cfg.num_scbots_per_zone = 1;
196 // Far larger than this test writes, so every record lands in file 1 and file 1 is still
197 // "incomplete" on the restart -- which is what makes the writer take it as its live file
198 // rather than starting a fresh one.
199 cfg.data_file_max_bytes = 1_000_000;
200 // Values are tiny; keep chunking out of the way entirely.
201 cfg.rest_chunk_threshold = 100_000;
202 cfg.rest_chunk_bytes = 10_000;
203 // Keep the zone inside this test's own root. The default configuration sends zone 1 to a
204 // container shared with the other tests.
205 cfg.zone_overrides = DaticleMap::new();
206
207 test!(sync_log::stream(), "+---------------------------------------------+");
208 test!(sync_log::stream(), "| BLIND INDEX TEST |");
209 test!(sync_log::stream(), "+---------------------------------------------+");
210
211 let zdirs: BTreeMap<ZoneInd, ZoneDir>;
212
213 // ── Session 1: an emptied index must not read as an empty store ──
214 {
215 let mut db = res!(setup::start_db(
216 db_root.clone(),
217 Some(cfg.clone()),
218 schms_input.clone(),
219 None,
220 false, // gc off: nothing here is about collection
221 true, // wipe: start from nothing
222 ));
223
224 for i in 0..KEYS {
225 res!(db.insert(key(i), val(i), user, schms2));
226 }
227 thread::sleep(Duration::from_millis(500));
228
229 zdirs = res!(db.api().get_zone_dirs());
230
231 // The instrument works before the fault is introduced. Without this, a scan that
232 // failed for any other reason would satisfy the assertion below.
233 let entries = res!(db.scan(&ScanOpts::all(), schms2));
234 if entries.len() != KEYS {
235 return Err(err!(
236 "Before the index was touched, a scan returned {} entries, expected {}. \
237 The rest of this test cannot mean anything.", entries.len(), KEYS;
238 Test, Mismatch));
239 }
240
241 let (dat_bytes, ind_bytes) = res!(zone_bytes(&zdirs));
242 test!(sync_log::stream(), "Zone holds {} bytes of data and {} bytes of index.",
243 dat_bytes, ind_bytes);
244 if dat_bytes == 0 {
245 return Err(err!(
246 "The zone holds no data at all, so emptying its index proves nothing.";
247 Test, Missing, Data));
248 }
249
250 // 1. The fault: index files at zero, data files untouched.
251 let emptied = res!(empty_all_index_files(&zdirs));
252 if emptied == 0 {
253 return Err(err!(
254 "No index files were found to empty, so the fault was never introduced.";
255 Test, Missing, File));
256 }
257 let (dat_after, ind_after) = res!(zone_bytes(&zdirs));
258 if ind_after != 0 || dat_after != dat_bytes {
259 return Err(err!(
260 "After emptying, the zone should hold {} bytes of data and no index; it \
261 holds {} and {}.", dat_bytes, dat_after, ind_after;
262 Test, Mismatch, Data));
263 }
264
265 // A scan over a store whose records are all still there must not answer as though
266 // they are not. This is the assertion the old walk could not make: it returned an
267 // empty list and `Ok`.
268 match db.scan(&ScanOpts::all(), schms2) {
269 Ok(entries) => return Err(err!(
270 "A scan over {} bytes of records whose index files are empty answered \
271 successfully with {} entries. A caller cannot tell that from an empty \
272 store, and this is exactly how an operator's new limit was accepted and \
273 then never enforced.", dat_bytes, entries.len();
274 Test, Invalid, Data)),
275 Err(e) => test!(sync_log::stream(),
276 "Scan refused to answer over a blind zone, as it must: {}", e),
277 }
278
279 // And the signature that makes it dangerous: every record is still readable by key
280 // throughout. If this half fails, the store really was damaged and the test is
281 // about something else.
282 for i in 0..KEYS {
283 match res!(db.get(&key(i), schms2)) {
284 Some((v, _)) => if v != val(i) {
285 return Err(err!(
286 "Key {:?} read back as {:?}, expected {:?}.", key(i), v, val(i);
287 Test, Mismatch, Data));
288 },
289 None => return Err(err!(
290 "Key {:?} is unreadable, so the store is damaged rather than merely \
291 unscannable, and this test is measuring the wrong thing.", key(i);
292 Test, Missing, Data)),
293 }
294 }
295
296 res!(db.shutdown());
297 }
298
299 thread::sleep(Duration::from_secs(1));
300
301 // ── Session 2: the recovery repairs the index, and writes reach it ──
302 {
303 let mut db = res!(setup::start_db(
304 db_root.clone(),
305 Some(cfg.clone()),
306 schms_input.clone(),
307 None,
308 false, // gc off
309 false, // do not wipe: read what session 1 left
310 ));
311
312 // 2. Start-up rebuilt the index from the data file, so the zone scans again.
313 let entries = res!(db.scan(&ScanOpts::all(), schms2));
314 if entries.len() != KEYS {
315 return Err(err!(
316 "After the empty-index recovery, a scan returned {} entries, expected {}. \
317 The index was not rebuilt from the data file.", entries.len(), KEYS;
318 Test, Mismatch, Data));
319 }
320
321 let (_dat_rebuilt, ind_rebuilt) = res!(zone_bytes(&zdirs));
322 if ind_rebuilt == 0 {
323 return Err(err!(
324 "After the empty-index recovery, the index files still hold no bytes.";
325 Test, Missing, Data));
326 }
327 test!(sync_log::stream(), "Index rebuilt to {} bytes.", ind_rebuilt);
328
329 // 3. THE SHARP ONE. A record written after the recovery must reach the index file
330 // on disk. When the rebuild renamed a fresh file over the index, the writer's
331 // append handle kept pointing at the replaced inode and this file never grew
332 // again for the life of the process -- while the data file took every record, so
333 // nothing else looked wrong.
334 res!(db.insert(
335 dat!(AFTER),
336 dat!("written after the index was rebuilt"),
337 user,
338 schms2,
339 ));
340 thread::sleep(Duration::from_millis(500));
341
342 let (_dat_after, ind_after) = res!(zone_bytes(&zdirs));
343 if ind_after <= ind_rebuilt {
344 return Err(err!(
345 "A record was written after the empty-index recovery and the index files \
346 went from {} bytes to {}. The write did not reach the index: it went to a \
347 file handle the rebuild had detached, so it is on no index at all and no \
348 scan will ever see it.", ind_rebuilt, ind_after;
349 Test, Missing, Data));
350 }
351
352 // And the same thing said through the scan, which is where a caller would notice.
353 let entries = res!(db.scan(&ScanOpts::all(), schms2));
354 if entries.len() != KEYS + 1 {
355 return Err(err!(
356 "After the recovery and one further write, a scan returned {} entries, \
357 expected {}.", entries.len(), KEYS + 1;
358 Test, Mismatch, Data));
359 }
360 let after = dat!(AFTER);
361 if !entries.iter().any(|(k, _v, _m)| *k == after) {
362 return Err(err!(
363 "The record written after the recovery is missing from the scan, though \
364 the count came out right.";
365 Test, Missing, Data));
366 }
367
368 res!(db.shutdown());
369 }
370
371 Ok(())
372}