oxedyne/fe2o3/fe2o3_steel/src/srv/health.rs
25.1 KiB, 23 runs
created by r1870400018:59440, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! The token-gated health body and the counters that feed it. |
| 2 | //! |
| 3 | //! Each Steel serves a small health body at a fixed path (see |
| 4 | //! [`crate::srv::cfg::ServerConfig::health_path`]): a flat map of integers read |
| 5 | //! from the host sampler's last sample, the address guard's live counts, and the |
| 6 | //! rolling admission counters below. It is the surface a peer [`Watcher`] reads |
| 7 | //! to tell *distress* -- a box that is answering but unwell -- from plain |
| 8 | //! liveness, which a `200` alone cannot show. |
| 9 | //! |
| 10 | //! # Why integers, and why a shared format |
| 11 | //! |
| 12 | //! The body is emitted here and parsed by the watcher there, both in this crate, |
| 13 | //! so the format is settled by agreement rather than by a schema: a flat JSON |
| 14 | //! object whose every value is an integer. Percentages are whole per cent; the |
| 15 | //! one-minute load average is carried times a hundred (a `load1` of `250` means |
| 16 | //! `2.50`), so a fractional load survives the integer contract without a float on |
| 17 | //! the wire. |
| 18 | //! |
| 19 | //! # Stamp ages, read at request time |
| 20 | //! |
| 21 | //! A job elsewhere on the box -- a backup pull, say -- proves each good run by |
| 22 | //! touching a stamp file. The body reports each configured stamp's age in whole |
| 23 | //! seconds (see [`HealthStamp`]), measured when the body is asked for rather than |
| 24 | //! sampled, so a job whose timer has died shows an age that keeps growing instead |
| 25 | //! of a figure frozen at its last run. Nothing has to stay alive for the alarm to |
| 26 | //! hold, which is the property a freshness file in a web root lacks. |
| 27 | //! |
| 28 | //! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\ |
| 29 | //! Anthropic Claude |
| 30 | |
| 31 | use crate::srv::admin::host_sampler::HealthHostMetrics; |
| 32 | |
| 33 | use oxedyne_fe2o3_core::prelude::*; |
| 34 | use oxedyne_fe2o3_sys::resident::Resident; |
| 35 | |
| 36 | use std::{ |
| 37 | collections::{ |
| 38 | BTreeMap, |
| 39 | VecDeque, |
| 40 | }, |
| 41 | path::{ |
| 42 | Path, |
| 43 | PathBuf, |
| 44 | }, |
| 45 | sync::{ |
| 46 | Arc, |
| 47 | Mutex, |
| 48 | }, |
| 49 | time::{ |
| 50 | SystemTime, |
| 51 | UNIX_EPOCH, |
| 52 | }, |
| 53 | }; |
| 54 | |
| 55 | // The health-body field names, named once so the emitter, the parser and a |
| 56 | // watcher's threshold map all agree on the spelling. |
| 57 | pub const F_MEM_PCT: &str = "mem_pct"; |
| 58 | pub const F_SWAP_PCT: &str = "swap_pct"; |
| 59 | pub const F_DISK_IOPS: &str = "disk_iops"; |
| 60 | pub const F_LOAD1: &str = "load1"; // 1-minute load average x100 |
| 61 | pub const F_CONNS: &str = "conns"; |
| 62 | pub const F_R429_1M: &str = "r429_1m"; |
| 63 | pub const F_DROPPED_1M: &str = "dropped_1m"; |
| 64 | pub const F_GUARD_FAILED: &str = "guard_failed"; // 1 when the guard's self-test failed |
| 65 | pub const F_UPTIME_S: &str = "uptime_s"; |
| 66 | pub const F_SEALED: &str = "sealed"; |
| 67 | |
| 68 | // How long a rolling counter keeps its per-second buckets. Two minutes so a |
| 69 | // one-minute query is always fully covered without an off-by-one at the boundary. |
| 70 | const ROLL_KEEP_SECS: u64 = 120; |
| 71 | |
| 72 | // ── Fleet fields ───────────────────────────────────────────────────────────── |
| 73 | // |
| 74 | // The fields the Fleet view added (2026-09-23), beside the originals above. |
| 75 | pub const F_DISK_PCT: &str = "disk_pct"; // the app root's filesystem, as `df` puts it |
| 76 | pub const F_SEALED_DBS: &str = "sealed_dbs"; // databases the seal holds shut |
| 77 | pub const F_MAIL_DOWN: &str = "mail_down"; // mail listeners asked for, not bound |
| 78 | // Written by the watcher into the body it read, never served by a peer: the time |
| 79 | // the probe took, measured at the watching end. |
| 80 | pub const F_PROBE_MS: &str = "probe_ms"; |
| 81 | |
| 82 | // Every field the body carries of its own accord: the ten `assemble` makes, the |
| 83 | // three `AdminState::health_body` sets beside them, and the probe time a watcher |
| 84 | // writes into what it read. A stamp may not take one of these names: it would |
| 85 | // replace the reading a watcher's threshold was written for. |
| 86 | pub const BUILTIN_FIELDS: [&str; 14] = [ |
| 87 | F_MEM_PCT, |
| 88 | F_SWAP_PCT, |
| 89 | F_DISK_IOPS, |
| 90 | F_LOAD1, |
| 91 | F_CONNS, |
| 92 | F_R429_1M, |
| 93 | F_DROPPED_1M, |
| 94 | F_GUARD_FAILED, |
| 95 | F_UPTIME_S, |
| 96 | F_SEALED, |
| 97 | F_DISK_PCT, |
| 98 | F_SEALED_DBS, |
| 99 | F_MAIL_DOWN, |
| 100 | F_PROBE_MS, |
| 101 | ]; |
| 102 | |
| 103 | // The per-process figures for the services named in `health_residents` are a list |
| 104 | // in spirit, and the format has no lists. Each resident is therefore flattened |
| 105 | // into keys of the form `res.<name>.<figure>` -- `res.steel.rss_kb`, |
| 106 | // `res.daimond_gateway.cap_pct` -- so a resident is an ordinary field to the |
| 107 | // parser and an ordinary threshold to a watcher's `distress` map. A name is |
| 108 | // letters, digits, `_`, `-` and `.`, so none can break the object, and the split |
| 109 | // back into name and figure is taken at the last dot. `procs` is always present, |
| 110 | // so a service that is not running reads as zero processes rather than as a |
| 111 | // service nobody asked about; `rss_kb` is in the kernel's kB, which are KiB; |
| 112 | // `cap_kb` and `cap_pct` appear only when the service's cgroup sets a limit. |
| 113 | pub const RES_PREFIX: &str = "res."; |
| 114 | pub const RES_PROCS: &str = "procs"; |
| 115 | pub const RES_RSS_KB: &str = "rss_kb"; |
| 116 | pub const RES_CAP_KB: &str = "cap_kb"; |
| 117 | pub const RES_CAP_PCT: &str = "cap_pct"; |
| 118 | pub const RES_NAME_MAX: usize = 64; |
| 119 | |
| 120 | /// Can this name ride in a flattened resident key? |
| 121 | pub fn is_resident_name(name: &str) -> bool { |
| 122 | !name.is_empty() |
| 123 | && name.len() <= RES_NAME_MAX |
| 124 | && name.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'_' || b == b'-' || b == b'.') |
| 125 | } |
| 126 | |
| 127 | pub fn resident_key(name: &str, figure: &str) -> String { |
| 128 | fmt!("{}{}.{}", RES_PREFIX, name, figure) |
| 129 | } |
| 130 | |
| 131 | /// The resident name and figure a flattened key carries, or `None` for any other |
| 132 | /// key. The split is at the last dot, since a name may hold dots and a figure |
| 133 | /// never does. |
| 134 | pub fn split_resident_key(key: &str) -> Option<(&str, &str)> { |
| 135 | let rest = ok!(key.strip_prefix(RES_PREFIX)); |
| 136 | let (name, figure) = ok!(rest.rsplit_once('.')); |
| 137 | if name.is_empty() || figure.is_empty() { |
| 138 | return None; |
| 139 | } |
| 140 | Some((name, figure)) |
| 141 | } |
| 142 | |
| 143 | /// One resident's figures as a body carries them, keyed by figure. |
| 144 | #[derive(Clone, Debug, Default, Eq, PartialEq)] |
| 145 | pub struct ResidentFigures { |
| 146 | pub name: String, |
| 147 | pub figures: BTreeMap<String, i64>, |
| 148 | } |
| 149 | |
| 150 | impl ResidentFigures { |
| 151 | pub fn get(&self, figure: &str) -> Option<i64> { |
| 152 | self.figures.get(figure).copied() |
| 153 | } |
| 154 | } |
| 155 | |
| 156 | // Stamp limits |
| 157 | pub const STAMP_NAME_MAX: usize = 64; |
| 158 | pub const STAMPS_MAX: usize = 16; // each is an `lstat` on every health request |
| 159 | |
| 160 | /// Can this name be a stamp's key in the body? |
| 161 | /// |
| 162 | /// Lower-case letters, digits and `_` only, so it can neither break the object |
| 163 | /// nor be mistaken for a flattened resident key, and none of [`BUILTIN_FIELDS`]. |
| 164 | pub fn is_stamp_name(name: &str) -> bool { |
| 165 | !name.is_empty() |
| 166 | && name.len() <= STAMP_NAME_MAX |
| 167 | && name.bytes().all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'_') |
| 168 | && !BUILTIN_FIELDS.contains(&name) |
| 169 | } |
| 170 | |
| 171 | /// A job's freshness stamp, reported in the body as its age. |
| 172 | #[derive(Clone, Debug, Eq, PartialEq)] |
| 173 | pub struct HealthStamp { |
| 174 | pub field: String, // the body key, e.g. `forge_state_age_s` |
| 175 | pub path: PathBuf, // absolute, and never followed through a symlink |
| 176 | } |
| 177 | |
| 178 | /// Whole seconds since a stamp was last written, as the body reports it. |
| 179 | /// |
| 180 | /// The rule the forge copy's own `freshness.sh` applies: the mtime alone, never |
| 181 | /// the content, and never through a symlink, so a link planted where the stamp |
| 182 | /// should be cannot lend it another file's freshness. Anything that is not a |
| 183 | /// regular file read by `lstat` -- missing, unreachable, a symlink, a directory |
| 184 | /// -- reads as written at the epoch, an age of `now` that trips any threshold. |
| 185 | /// An mtime in the future reads as `0` rather than as a negative age. |
| 186 | pub fn stamp_age_secs(path: &Path, now: SystemTime) -> i64 { |
| 187 | let written = match std::fs::symlink_metadata(path) { |
| 188 | Ok(m) if m.file_type().is_file() => m.modified().unwrap_or(UNIX_EPOCH), |
| 189 | _ => UNIX_EPOCH, |
| 190 | }; |
| 191 | let age = now.duration_since(written).map(|d| d.as_secs()).unwrap_or(0); |
| 192 | age.min(i64::MAX as u64) as i64 |
| 193 | } |
| 194 | |
| 195 | fn unix_secs() -> u64 { |
| 196 | SystemTime::now() |
| 197 | .duration_since(UNIX_EPOCH) |
| 198 | .map(|d| d.as_secs()) |
| 199 | .unwrap_or(0) |
| 200 | } |
| 201 | |
| 202 | /// A count of events over a recent sliding window. |
| 203 | /// |
| 204 | /// One bucket per unix second, trimmed to [`ROLL_KEEP_SECS`]. Increments take a |
| 205 | /// short mutex; the events counted -- a `429` emitted, a connection dropped at |
| 206 | /// admission -- are rare in normal operation and hot only under attack, so the |
| 207 | /// lock is not on any ordinary request's path. Cheaply cloneable via `Arc`, so |
| 208 | /// the increment site and the health reader share one counter. |
| 209 | #[derive(Debug)] |
| 210 | pub struct RollingCounter { |
| 211 | buckets: Mutex<VecDeque<(u64, u64)>>, // (unix second, count), newest last |
| 212 | } |
| 213 | |
| 214 | impl RollingCounter { |
| 215 | pub fn new() -> Self { |
| 216 | Self { buckets: Mutex::new(VecDeque::new()) } |
| 217 | } |
| 218 | |
| 219 | pub fn new_shared() -> Arc<Self> { |
| 220 | Arc::new(Self::new()) |
| 221 | } |
| 222 | |
| 223 | pub fn incr(&self) { |
| 224 | self.add(1); |
| 225 | } |
| 226 | |
| 227 | pub fn add(&self, n: u64) { |
| 228 | let now = unix_secs(); |
| 229 | // A counter is not worth poisoning a process over: recover the buckets |
| 230 | // from a poisoned lock rather than propagate, since a lost count is a far |
| 231 | // smaller harm than a health route that stops answering. |
| 232 | let mut b = match self.buckets.lock() { |
| 233 | Ok(g) => g, |
| 234 | Err(poisoned) => poisoned.into_inner(), |
| 235 | }; |
| 236 | match b.back_mut() { |
| 237 | Some((sec, cnt)) if *sec == now => *cnt = cnt.saturating_add(n), |
| 238 | _ => b.push_back((now, n)), |
| 239 | } |
| 240 | while let Some((sec, _)) = b.front() { |
| 241 | if now.saturating_sub(*sec) > ROLL_KEEP_SECS { |
| 242 | b.pop_front(); |
| 243 | } else { |
| 244 | break; |
| 245 | } |
| 246 | } |
| 247 | } |
| 248 | |
| 249 | /// Events counted in the last `window_secs` seconds. |
| 250 | pub fn last(&self, window_secs: u64) -> u64 { |
| 251 | let cutoff = unix_secs().saturating_sub(window_secs); |
| 252 | let b = match self.buckets.lock() { |
| 253 | Ok(g) => g, |
| 254 | Err(poisoned) => poisoned.into_inner(), |
| 255 | }; |
| 256 | b.iter().filter(|(sec, _)| *sec >= cutoff).map(|(_, cnt)| cnt).sum() |
| 257 | } |
| 258 | } |
| 259 | |
| 260 | impl Default for RollingCounter { |
| 261 | fn default() -> Self { |
| 262 | Self::new() |
| 263 | } |
| 264 | } |
| 265 | |
| 266 | /// A flat map of integer health fields, emitted as JSON and parsed back the same. |
| 267 | #[derive(Clone, Debug, Default, Eq, PartialEq)] |
| 268 | pub struct HealthBody { |
| 269 | pub fields: BTreeMap<String, i64>, |
| 270 | } |
| 271 | |
| 272 | impl HealthBody { |
| 273 | pub fn new() -> Self { |
| 274 | Self { fields: BTreeMap::new() } |
| 275 | } |
| 276 | |
| 277 | pub fn set(&mut self, key: &str, value: i64) { |
| 278 | self.fields.insert(key.to_string(), value); |
| 279 | } |
| 280 | |
| 281 | pub fn get(&self, key: &str) -> Option<i64> { |
| 282 | self.fields.get(key).copied() |
| 283 | } |
| 284 | |
| 285 | /// A flat JSON object of integers, keys in sorted order. |
| 286 | pub fn to_json(&self) -> String { |
| 287 | let mut out = String::from("{"); |
| 288 | for (i, (k, v)) in self.fields.iter().enumerate() { |
| 289 | if i > 0 { |
| 290 | out.push(','); |
| 291 | } |
| 292 | out.push_str(&fmt!("\"{}\":{}", k, v)); |
| 293 | } |
| 294 | out.push('}'); |
| 295 | out |
| 296 | } |
| 297 | |
| 298 | /// Assemble the health body from the pieces already kept elsewhere: the |
| 299 | /// host sampler's last reading, the guard's live count and self-test, the |
| 300 | /// rolling admission counters, and the two liveness scalars. Every value is |
| 301 | /// an integer; a boolean rides as `0` or `1`. |
| 302 | /// |
| 303 | /// `host` is the sampler's reduced figures, `None` while the ring is still |
| 304 | /// empty (the level fields are then simply absent rather than zeroed, so a |
| 305 | /// watcher does not read a not-yet-sampled box as one at 0% memory). |
| 306 | pub fn assemble( |
| 307 | host: Option<HealthHostMetrics>, |
| 308 | conns: usize, |
| 309 | r429_1m: u64, |
| 310 | dropped_1m: u64, |
| 311 | guard_selftest: bool, |
| 312 | uptime_s: u64, |
| 313 | sealed: bool, |
| 314 | ) |
| 315 | -> Self |
| 316 | { |
| 317 | let mut b = Self::new(); |
| 318 | if let Some(h) = host { |
| 319 | b.set(F_MEM_PCT, h.mem_pct); |
| 320 | b.set(F_SWAP_PCT, h.swap_pct); |
| 321 | b.set(F_DISK_IOPS, h.disk_iops); |
| 322 | b.set(F_LOAD1, h.load1); |
| 323 | } |
| 324 | b.set(F_CONNS, conns as i64); |
| 325 | b.set(F_R429_1M, r429_1m as i64); |
| 326 | b.set(F_DROPPED_1M, dropped_1m as i64); |
| 327 | // Carried as a failure, as `mail_down` and `sealed_dbs` are, so an ordinary |
| 328 | // `distress` threshold of 1 alarms it (D-06 audit D2). |
| 329 | b.set(F_GUARD_FAILED, if guard_selftest { 0 } else { 1 }); |
| 330 | b.set(F_UPTIME_S, uptime_s as i64); |
| 331 | b.set(F_SEALED, if sealed { 1 } else { 0 }); |
| 332 | b |
| 333 | } |
| 334 | |
| 335 | /// Each stamp's age in whole seconds, under its own field. See |
| 336 | /// [`stamp_age_secs`] for what a missing or suspect stamp reads as. |
| 337 | pub fn set_stamps(&mut self, stamps: &[HealthStamp], now: SystemTime) { |
| 338 | for s in stamps { |
| 339 | self.set(&s.field, stamp_age_secs(&s.path, now)); |
| 340 | } |
| 341 | } |
| 342 | |
| 343 | /// Parse the flat integer object this crate emits. Deliberately narrow: it |
| 344 | /// reads `{"key":int,...}` and nothing nested, because that is the whole of |
| 345 | /// the format both ends agreed on, and a lenient parser would only hide a |
| 346 | /// drift between them. |
| 347 | pub fn parse(s: &str) -> Outcome<Self> { |
| 348 | let trimmed = s.trim(); |
| 349 | let inner = res!(trimmed.strip_prefix('{') |
| 350 | .and_then(|r| r.strip_suffix('}')) |
| 351 | .ok_or_else(|| err!( |
| 352 | "A health body must be a JSON object; got {:?}.", trimmed; |
| 353 | Invalid, Input, Decode))); |
| 354 | let mut out = Self::new(); |
| 355 | for part in inner.split(',') { |
| 356 | let part = part.trim(); |
| 357 | if part.is_empty() { |
| 358 | continue; |
| 359 | } |
| 360 | let (raw_key, raw_val) = res!(part.split_once(':').ok_or_else(|| err!( |
| 361 | "A health-body field {:?} is not 'key:value'.", part; |
| 362 | Invalid, Input, Decode))); |
| 363 | let key = raw_key.trim().trim_matches('"').to_string(); |
| 364 | let val: i64 = res!(raw_val.trim().parse().map_err(|_| err!( |
| 365 | "The health-body field '{}' has a non-integer value {:?}.", |
| 366 | key, raw_val.trim(); |
| 367 | Invalid, Input, Decode))); |
| 368 | out.fields.insert(key, val); |
| 369 | } |
| 370 | Ok(out) |
| 371 | } |
| 372 | } |
| 373 | |
| 374 | impl HealthBody { |
| 375 | /// Flatten one resident into its `res.<name>.*` keys. A name that cannot |
| 376 | /// ride in a key is skipped: configuration refuses one at start-up, so |
| 377 | /// reaching here with one is a caller bypassing that check. |
| 378 | pub fn set_resident(&mut self, r: &Resident) { |
| 379 | if !is_resident_name(&r.name) { |
| 380 | return; |
| 381 | } |
| 382 | let clamp = |v: u64| -> i64 { v.min(i64::MAX as u64) as i64 }; |
| 383 | self.set(&resident_key(&r.name, RES_PROCS), r.procs as i64); |
| 384 | self.set(&resident_key(&r.name, RES_RSS_KB), clamp(r.rss_kib)); |
| 385 | if let Some(cap) = r.cap_kib { |
| 386 | self.set(&resident_key(&r.name, RES_CAP_KB), clamp(cap)); |
| 387 | } |
| 388 | if let Some(pct) = r.cap_pct() { |
| 389 | self.set(&resident_key(&r.name, RES_CAP_PCT), clamp(pct)); |
| 390 | } |
| 391 | } |
| 392 | |
| 393 | /// The residents a body carries, gathered back from their flattened keys, in |
| 394 | /// name order. |
| 395 | pub fn residents(&self) -> Vec<ResidentFigures> { |
| 396 | let mut by_name: BTreeMap<&str, BTreeMap<String, i64>> = BTreeMap::new(); |
| 397 | for (k, v) in &self.fields { |
| 398 | if let Some((name, figure)) = split_resident_key(k) { |
| 399 | by_name.entry(name).or_default().insert(figure.to_string(), *v); |
| 400 | } |
| 401 | } |
| 402 | by_name.into_iter() |
| 403 | .map(|(name, figures)| ResidentFigures { name: name.to_string(), figures }) |
| 404 | .collect() |
| 405 | } |
| 406 | } |
| 407 | |
| 408 | |
| 409 | #[cfg(test)] |
| 410 | mod tests { |
| 411 | use super::*; |
| 412 | |
| 413 | /// Residents flatten into ordinary integer keys, survive the wire, and gather |
| 414 | /// back into the same figures -- including a name that holds a dot, and a |
| 415 | /// service that is not running, which must still say so. |
| 416 | #[test] |
| 417 | fn residents_flatten_and_gather_back_across_the_wire() -> Outcome<()> { |
| 418 | let running = Resident { |
| 419 | name: fmt!("daimond_gateway"), |
| 420 | procs: 1, |
| 421 | rss_kib: 412_000, |
| 422 | cap_kib: Some(524_288), |
| 423 | }; |
| 424 | let uncapped = Resident { |
| 425 | name: fmt!("python3.11"), |
| 426 | procs: 2, |
| 427 | rss_kib: 90_000, |
| 428 | cap_kib: None, |
| 429 | }; |
| 430 | let absent = Resident { name: fmt!("steel"), ..Resident::default() }; |
| 431 | let mut body = HealthBody::assemble(None, 3, 0, 0, true, 60, false); |
| 432 | for r in [&running, &uncapped, &absent] { |
| 433 | body.set_resident(r); |
| 434 | } |
| 435 | |
| 436 | let json = body.to_json(); |
| 437 | assert!(json.contains("\"res.daimond_gateway.rss_kb\":412000"), "got {}", json); |
| 438 | assert!(json.contains("\"res.daimond_gateway.cap_pct\":78"), "got {}", json); |
| 439 | let back = res!(HealthBody::parse(&json)); |
| 440 | assert_eq!(back, body, "the flattened body must round-trip exactly"); |
| 441 | |
| 442 | let gathered = back.residents(); |
| 443 | let names: Vec<&str> = gathered.iter().map(|r| r.name.as_str()).collect(); |
| 444 | assert_eq!(names, vec!["daimond_gateway", "python3.11", "steel"]); |
| 445 | assert_eq!(gathered[0].get(RES_CAP_KB), Some(524_288)); |
| 446 | assert_eq!(gathered[0].get(RES_CAP_PCT), Some(78)); |
| 447 | assert_eq!(gathered[1].get(RES_RSS_KB), Some(90_000), |
| 448 | "a dot in the name must not split it: the split is at the last dot"); |
| 449 | assert_eq!(gathered[1].get(RES_CAP_PCT), None, "no cap, no percentage"); |
| 450 | assert_eq!(gathered[2].get(RES_PROCS), Some(0), |
| 451 | "a service that is not running still reports, as zero processes"); |
| 452 | Ok(()) |
| 453 | } |
| 454 | |
| 455 | /// A resident key is recognised only in its full shape, and a name that could |
| 456 | /// break the object is refused before it can become one. |
| 457 | #[test] |
| 458 | fn resident_keys_split_only_in_their_own_shape() { |
| 459 | assert_eq!(split_resident_key("res.steel.rss_kb"), Some(("steel", "rss_kb"))); |
| 460 | assert_eq!(split_resident_key("res.a.b.cap_pct"), Some(("a.b", "cap_pct"))); |
| 461 | assert_eq!(split_resident_key("mem_pct"), None); |
| 462 | assert_eq!(split_resident_key("res.steel"), None); |
| 463 | assert_eq!(split_resident_key("res..rss_kb"), None); |
| 464 | assert!(is_resident_name("daimond_gateway")); |
| 465 | assert!(!is_resident_name("")); |
| 466 | assert!(!is_resident_name("a\"b"), "a quote would end the JSON key"); |
| 467 | assert!(!is_resident_name("a:b"), "a colon would split the field"); |
| 468 | assert!(!is_resident_name("a,b"), "a comma would split the object"); |
| 469 | let mut b = HealthBody::new(); |
| 470 | b.set_resident(&Resident { name: fmt!("bad,name"), procs: 1, ..Resident::default() }); |
| 471 | assert!(b.fields.is_empty(), "an unsafe name must never reach the body"); |
| 472 | } |
| 473 | |
| 474 | #[test] |
| 475 | fn json_round_trips() { |
| 476 | let mut b = HealthBody::new(); |
| 477 | b.set(F_MEM_PCT, 94); |
| 478 | b.set(F_SWAP_PCT, 71); |
| 479 | b.set(F_SEALED, 0); |
| 480 | b.set(F_LOAD1, 250); |
| 481 | let json = b.to_json(); |
| 482 | let back = HealthBody::parse(&json).expect("parse"); |
| 483 | assert_eq!(b, back); |
| 484 | assert_eq!(back.get(F_MEM_PCT), Some(94)); |
| 485 | assert_eq!(back.get(F_LOAD1), Some(250)); |
| 486 | } |
| 487 | |
| 488 | #[test] |
| 489 | fn parse_tolerates_whitespace() { |
| 490 | let b = HealthBody::parse("{ \"mem_pct\" : 40 , \"conns\": 3 }").expect("parse"); |
| 491 | assert_eq!(b.get(F_MEM_PCT), Some(40)); |
| 492 | assert_eq!(b.get(F_CONNS), Some(3)); |
| 493 | } |
| 494 | |
| 495 | #[test] |
| 496 | fn parse_rejects_a_non_object() { |
| 497 | assert!(HealthBody::parse("not json").is_err()); |
| 498 | assert!(HealthBody::parse("{\"a\":notint}").is_err()); |
| 499 | } |
| 500 | |
| 501 | /// A fresh scratch directory for stamp files, removed by the caller. |
| 502 | fn stamp_dir(tag: &str) -> Outcome<PathBuf> { |
| 503 | let nanos = SystemTime::now().duration_since(UNIX_EPOCH).map(|d| d.as_nanos()).unwrap_or(0); |
| 504 | let dir = std::env::temp_dir().join(fmt!( |
| 505 | "fe2o3_steel_stamps_{}_{}_{}", tag, std::process::id(), nanos)); |
| 506 | res!(std::fs::create_dir_all(&dir), IO, File); |
| 507 | Ok(dir) |
| 508 | } |
| 509 | |
| 510 | /// Write a stamp and set its mtime to `when`, the way a job's `touch` would. |
| 511 | fn stamp_at(path: &Path, when: SystemTime) -> Outcome<()> { |
| 512 | res!(std::fs::write(path, b"ok\n"), IO, File); |
| 513 | let f = res!(std::fs::OpenOptions::new().write(true).open(path), IO, File); |
| 514 | res!(f.set_modified(when), IO, File); |
| 515 | Ok(()) |
| 516 | } |
| 517 | |
| 518 | /// The age is the mtime's, to the second; a stamp that is not there, or that |
| 519 | /// is not a plain file, reads as never written, and so trips any threshold. |
| 520 | #[test] |
| 521 | fn a_stamp_reads_its_age_and_anything_suspect_reads_as_never() -> Outcome<()> { |
| 522 | let dir = res!(stamp_dir("age")); |
| 523 | let now = SystemTime::now(); |
| 524 | let never = res!(now.duration_since(UNIX_EPOCH).map_err(|e| err!(e, |
| 525 | "The clock is before the epoch."; Test))).as_secs() as i64; |
| 526 | |
| 527 | let fresh = dir.join("state.ok"); |
| 528 | res!(stamp_at(&fresh, now - std::time::Duration::from_secs(3_700))); |
| 529 | let age = stamp_age_secs(&fresh, now); |
| 530 | assert!((3_699..=3_701).contains(&age), "a stamp written 3700 s ago read {}", age); |
| 531 | |
| 532 | assert_eq!(stamp_age_secs(&dir.join("absent.ok"), now), never, |
| 533 | "a missing stamp must read as never written, not as fresh or absent"); |
| 534 | |
| 535 | // A symlink to a perfectly fresh stamp still reads as never: a link must not |
| 536 | // lend the stamp another file's freshness. |
| 537 | #[cfg(unix)] |
| 538 | { |
| 539 | let link = dir.join("link.ok"); |
| 540 | res!(std::os::unix::fs::symlink(&fresh, &link), IO, File); |
| 541 | res!(stamp_at(&fresh, now)); |
| 542 | assert_eq!(stamp_age_secs(&fresh, now), 0); |
| 543 | assert_eq!(stamp_age_secs(&link, now), never, |
| 544 | "a symlinked stamp was followed to its target"); |
| 545 | } |
| 546 | |
| 547 | let subdir = dir.join("a_directory"); |
| 548 | res!(std::fs::create_dir_all(&subdir), IO, File); |
| 549 | assert_eq!(stamp_age_secs(&subdir, now), never, "a directory is not a stamp"); |
| 550 | |
| 551 | let future = dir.join("future.ok"); |
| 552 | res!(stamp_at(&future, now + std::time::Duration::from_secs(600))); |
| 553 | assert_eq!(stamp_age_secs(&future, now), 0, "a future mtime reads as 0, not negative"); |
| 554 | |
| 555 | let _ = std::fs::remove_dir_all(&dir); |
| 556 | Ok(()) |
| 557 | } |
| 558 | |
| 559 | /// Stamp ages are ordinary integer fields: they survive the wire beside the |
| 560 | /// built-in fields and are read back by the parser a watcher uses. |
| 561 | #[test] |
| 562 | fn stamp_ages_ride_the_body_and_round_trip() -> Outcome<()> { |
| 563 | let dir = res!(stamp_dir("body")); |
| 564 | let now = SystemTime::now(); |
| 565 | let state = dir.join("state.ok"); |
| 566 | res!(stamp_at(&state, now - std::time::Duration::from_secs(120))); |
| 567 | let stamps = vec![ |
| 568 | HealthStamp { field: fmt!("forge_state_age_s"), path: state }, |
| 569 | HealthStamp { field: fmt!("forge_repos_age_s"), path: dir.join("repos.ok") }, |
| 570 | ]; |
| 571 | let mut body = HealthBody::assemble(None, 3, 0, 0, true, 60, false); |
| 572 | body.set_stamps(&stamps, now); |
| 573 | |
| 574 | let back = res!(HealthBody::parse(&body.to_json())); |
| 575 | assert_eq!(back, body, "the body with stamps must round-trip exactly"); |
| 576 | let state_age = back.get("forge_state_age_s").unwrap_or(-1); |
| 577 | assert!((119..=121).contains(&state_age), "state stamp read {}", state_age); |
| 578 | assert!(back.get("forge_repos_age_s").unwrap_or(0) > 1_000_000_000, |
| 579 | "the missing repos stamp must read as never, a very large age"); |
| 580 | assert_eq!(back.get(F_CONNS), Some(3), "the built-in fields are untouched"); |
| 581 | let _ = std::fs::remove_dir_all(&dir); |
| 582 | Ok(()) |
| 583 | } |
| 584 | |
| 585 | #[test] |
| 586 | fn a_stamp_name_cannot_break_the_body_or_take_a_builtin() { |
| 587 | assert!(is_stamp_name("forge_state_age_s")); |
| 588 | assert!(is_stamp_name("backup2_age_s")); |
| 589 | for bad in ["", "Forge_age", "forge-age", "forge.age", "a:b", "a,b", "a\"b", "a b"] { |
| 590 | assert!(!is_stamp_name(bad), "'{}' was accepted as a stamp name", bad); |
| 591 | } |
| 592 | assert!(!is_stamp_name(&"a".repeat(STAMP_NAME_MAX + 1))); |
| 593 | for builtin in BUILTIN_FIELDS { |
| 594 | assert!(!is_stamp_name(builtin), "the built-in '{}' was accepted", builtin); |
| 595 | } |
| 596 | } |
| 597 | |
| 598 | /// The reserved list is the body's own output, so a field added to `assemble` |
| 599 | /// without joining the list is a failing test rather than a name a stamp can |
| 600 | /// quietly take. The four set outside `assemble` are named here; the served |
| 601 | /// body is held to the list by `AdminState::health_body`'s own test. |
| 602 | #[test] |
| 603 | fn the_builtin_list_is_what_assemble_emits_and_the_four_set_beside_it() { |
| 604 | let host = HealthHostMetrics { mem_pct: 40, swap_pct: 1, disk_iops: 2, load1: 50 }; |
| 605 | let body = HealthBody::assemble(Some(host), 1, 0, 0, true, 60, false); |
| 606 | let mut emitted: Vec<&str> = body.fields.keys().map(|k| k.as_str()).collect(); |
| 607 | emitted.extend([F_DISK_PCT, F_SEALED_DBS, F_MAIL_DOWN, F_PROBE_MS]); |
| 608 | emitted.sort(); |
| 609 | let mut reserved: Vec<&str> = BUILTIN_FIELDS.to_vec(); |
| 610 | reserved.sort(); |
| 611 | assert_eq!(emitted, reserved); |
| 612 | } |
| 613 | |
| 614 | #[test] |
| 615 | fn rolling_counter_sums_the_window() { |
| 616 | let c = RollingCounter::new(); |
| 617 | c.incr(); |
| 618 | c.incr(); |
| 619 | c.add(5); |
| 620 | // All within the same second, so a one-minute window sees all seven. |
| 621 | assert_eq!(c.last(60), 7); |
| 622 | // A zero-width window still sees the current second's bucket. |
| 623 | assert_eq!(c.last(0), 7); |
| 624 | } |
| 625 | } |