oxedyne/fe2o3/fe2o3_o3db_sync/src/lib.rs
6.1 KiB, 31 runs
created by r1870400018:805, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! A log-structured key-value database with robust error handling and high throughput. |
| 2 | //! |
| 3 | //! # Overview |
| 4 | //! |
| 5 | //! O3db (Ozone Database) is a log-structured key-value store inspired by BitCask. It provides: |
| 6 | //! |
| 7 | //! - Fast writes through log-structured append operations |
| 8 | //! - High parallelism using operating system threads |
| 9 | //! - Automatic garbage collection |
| 10 | //! - Configurable caching |
| 11 | //! - Comprehensive error handling |
| 12 | //! |
| 13 | //! # Examples |
| 14 | //! |
| 15 | //! Basic usage with error handling (illustrative sketch): |
| 16 | //! |
| 17 | //! ```ignore |
| 18 | //! use oxedyne_fe2o3_core::prelude::*; |
| 19 | //! use oxedyne_fe2o3_o3db_sync::prelude::*; |
| 20 | //! |
| 21 | //! fn store_value() -> Outcome<()> { |
| 22 | //! // Configure and start the database |
| 23 | //! let mut db = res!(O3db::new( |
| 24 | //! "my_db", |
| 25 | //! None, // Use default config |
| 26 | //! RestSchemesInput::default(), |
| 27 | //! Uid::default(), |
| 28 | //! )); |
| 29 | //! // Returns once every zone has surveyed its files and loaded its caches. |
| 30 | //! res!(db.start("my_db")); |
| 31 | //! |
| 32 | //! // Store a value |
| 33 | //! let resp = res!(db.api().store( |
| 34 | //! dat!("my_key"), |
| 35 | //! dat!(42), |
| 36 | //! user_id, |
| 37 | //! )); |
| 38 | //! |
| 39 | //! // Each record is answered written, within the user request deadline, and then durable, |
| 40 | //! // within the durability deadline. A writer's error arrives as the error it was. |
| 41 | //! let (_existed, _records) = res!(resp.recv_store_ack()); |
| 42 | //! Ok(()) |
| 43 | //! } |
| 44 | //! ``` |
| 45 | //! |
| 46 | //! Fetching values with chunking support (illustrative sketch): |
| 47 | //! |
| 48 | //! ```ignore |
| 49 | //! fn fetch_large_value() -> Outcome<Vec<u8>> { |
| 50 | //! let resp = res!(db.api().fetch_using_schemes( |
| 51 | //! &dat!("large_key"), |
| 52 | //! None, // Use default schemes |
| 53 | //! )); |
| 54 | //! |
| 55 | //! match res!(resp.recv_daticle(db.api().schemes().encrypter(), None)) { |
| 56 | //! (None, _) => Err(err!("Value not found"; Data, Missing)), |
| 57 | //! (Some((Dat::Tup5u64(tup), _)), _) => { |
| 58 | //! // Handle chunked data |
| 59 | //! res!(db.api().fetch_chunks(&Dat::Tup5u64(tup), None)) |
| 60 | //! }, |
| 61 | //! (Some((dat, _)), _) => Ok(res!(dat.as_bytes())), |
| 62 | //! } |
| 63 | //! } |
| 64 | //! ``` |
| 65 | //! |
| 66 | //! Garbage collection (illustrative sketch): |
| 67 | //! |
| 68 | //! ```ignore |
| 69 | //! fn manage_gc() -> Outcome<()> { |
| 70 | //! // Enable garbage collection |
| 71 | //! res!(db.api().activate_gc(true)); |
| 72 | //! |
| 73 | //! // Wait for GC to complete |
| 74 | //! thread::sleep(Duration::from_secs(1)); |
| 75 | //! |
| 76 | //! // Verify file states |
| 77 | //! res!(db.api().dump_file_states(constant::USER_REQUEST_WAIT)); |
| 78 | //! |
| 79 | //! Ok(()) |
| 80 | //! } |
| 81 | //! ``` |
| 82 | //! |
| 83 | //! # Error Handling |
| 84 | //! |
| 85 | //! The database uses the `Outcome<T>` type for error handling, which provides: |
| 86 | //! |
| 87 | //! - Error tags for categorisation |
| 88 | //! - Error chaining for context |
| 89 | //! - Panic catching through the `res!` macro |
| 90 | //! - Detailed error messages |
| 91 | //! |
| 92 | //! All errors include file and line information for debugging. |
| 93 | //! |
| 94 | //! # Implementation |
| 95 | //! |
| 96 | //! This Rust implementation uses operating system threads. |
| 97 | //! |
| 98 | //! ## Persistent file side |
| 99 | //! |
| 100 | //! Values are appended to live data files, and their location is appended to live index files which |
| 101 | //! speed up initialisation. Upon reaching their maximum allowed size, data files are closed |
| 102 | //! (archived) and a new file in the sequence is created. Garbage collection works in the background |
| 103 | //! to remove old values from archive files, some of which may ultimately be completely deleted. Files |
| 104 | //! can be allocated across multiple zone directories. |
| 105 | //! |
| 106 | //! ## Volatile memory side |
| 107 | //! |
| 108 | //! Each zone is associated with a cache map that contains file locations for all values, but also |
| 109 | //! retains the values themselves when they are retrieved while the cache stays within the specified |
| 110 | //! size limit. Caches can be configured to have the same fixed number of readers and a differing |
| 111 | //! fixed number of writers. Caches are initialised by reading zone index files. Each writer controls |
| 112 | //! precisely one live file. |
| 113 | //! |
| 114 | //! # Initial Roadmap |
| 115 | //! |
| 116 | //! 1. [✘] Reliable persistent data storage and retrieval. |
| 117 | //! 1.1 [✔] Robust cache initialisation. |
| 118 | //! 1.2 [✔] Server. |
| 119 | //! 1.3 [✘] Recaching. |
| 120 | //! 1.4 [✘] Rezoning. |
| 121 | //! 2. [✘] Multiple users. |
| 122 | //! 2.1. [✔] Timestamps -> Metadata including user identification. |
| 123 | //! 2.2. [✔] Encryption. |
| 124 | //! 2.3. [✔] Digital signatures. |
| 125 | //! 2.4. [✔] Multiple user login. |
| 126 | //! 2.5. [✘] Recording user access. |
| 127 | //! 2.6. [✘] User access control. |
| 128 | //! 3. [✔] Reliable resource management. |
| 129 | //! 3.1. [✔] Cache size reporting. |
| 130 | //! 3.2. [✔] Zone directory size reporting. |
| 131 | //! 3.3. [✔] Cache size limitation with automated jettison of oldest cached values. |
| 132 | //! 3.4. [✔] File garbage collection. |
| 133 | //! 4. [✘] Documentation. |
| 134 | //! 4.1. [✘] Documentation of intentions and architecture with diagrams. |
| 135 | //! 4.2. [✘] Source code thoroughly and extensively documented. |
| 136 | //! 5. [✘] Testing. |
| 137 | //! 5.1. [✔] Basic integration tests. |
| 138 | //! 5.2. [✘] Extensive scenario tests. |
| 139 | //! 6. [✘] Performance. |
| 140 | //! 6.1. [✔] Basic performance measurement. |
| 141 | //! 6.2. [✘] Extensive performance measurement varying all relevant configuration parameters. |
| 142 | //! 6.3. [✘] Peer benchmarks. |
| 143 | //! |
| 144 | #![forbid(unsafe_code)] |
| 145 | #![allow(dead_code)] |
| 146 | pub mod api; |
| 147 | pub mod base; |
| 148 | pub mod bots; |
| 149 | pub mod cas; // Content-addressed storage: SHA-256 chunk addressing for large syncable payloads. |
| 150 | pub mod cas_o3db; // Cas trait backed by a local O3db instance. |
| 151 | pub mod comm; |
| 152 | pub mod dal; // Data Abstraction Layer. |
| 153 | pub mod data; |
| 154 | pub mod file; |
| 155 | pub mod test; |
| 156 | |
| 157 | pub mod db; |
| 158 | pub mod migrate; // Live-set store migration and compaction (drops orphaned chunk records). |
| 159 | pub mod sweep; // In-place, online orphan sweep (reclaims orphaned chunk records via supersession). |
| 160 | pub mod prelude; |
| 161 | |
| 162 | /// Peer-to-peer distributed Ozone: OAM placement + IBLT anti-entropy + |
| 163 | /// HotStuff consensus. Gated behind the `dist` cargo feature. |
| 164 | #[cfg(feature = "dist")] |
| 165 | pub mod kademlia; // Kademlia DHT routing-table primitive for the distributed layer. |
| 166 | #[cfg(feature = "dist")] |
| 167 | pub mod oam; // Oxegen Allocation Mechanism placement primitive for the distributed layer. |
| 168 | #[cfg(feature = "dist")] |
| 169 | pub mod dist; |
| 170 | |
| 171 | pub use crate::db::O3db; |