oxedyne/daimond/src/compact.rs
122 KiB, 1 run
created by r2519314175:937, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Folding a conversation that has outgrown the model's context window. |
| 2 | //! |
| 3 | //! A working session grows without limit: every round re-sends the whole history, and a |
| 4 | //! coding session's history is mostly tool results -- a 60 KB file read is 60 KB in every |
| 5 | //! request from then until the chat is deleted. Left alone it ends one way. The provider |
| 6 | //! refuses the request for being longer than the window, the turn dies, and the NEXT turn |
| 7 | //! sends the same oversized history and dies the same way. The chat is then permanently |
| 8 | //! unusable, with no recovery but folding it into a Diamond or throwing it away. |
| 9 | //! |
| 10 | //! ## What a fold is |
| 11 | //! |
| 12 | //! Everything before a cut is replaced by ONE message; everything after it is kept exactly |
| 13 | //! as it was. The replacement carries two things, produced two different ways: |
| 14 | //! |
| 15 | //! * **A ledger**, read from the tool calls themselves by [`ledger_of`]. Which files were |
| 16 | //! read, which were written, which commands ran, and -- separately -- which of those were |
| 17 | //! REFUSED or FAILED. This is arithmetic, not judgement: no model is asked, so none can |
| 18 | //! lose it or invent it. A fold that drops which files were written leaves the model |
| 19 | //! claiming work it can no longer verify, which is worse than the overflow it was avoiding |
| 20 | //! -- and a fold that lists a REFUSED write as a write is that same claim, made by the app |
| 21 | //! itself, in the one part of the note the model is told to trust. |
| 22 | //! * **A summary**, written by a model from a bounded rendering of the folded part (see |
| 23 | //! [`render_for_fold`]). Goal, decisions, findings, open threads -- the part that needs |
| 24 | //! judgement. |
| 25 | //! |
| 26 | //! The split is what makes failure survivable. When the summarising call fails -- refused |
| 27 | //! key, no network, the user pressed Stop -- the ledger alone still folds the conversation, |
| 28 | //! truthfully, and the turn goes on. Falling back to "send the oversized history" would |
| 29 | //! reproduce the exact bug this module exists to remove. |
| 30 | //! |
| 31 | //! ## The rule that breaks a compactor |
| 32 | //! |
| 33 | //! An assistant message bearing `tool_calls` MUST be followed by one tool reply per call. |
| 34 | //! A cut that lands between them produces a conversation the provider rejects outright -- |
| 35 | //! and it rejects it on every subsequent turn, which is the very failure being fixed, now |
| 36 | //! caused by the fix. So the cut is chosen by [`tail_start`], which cannot land inside a |
| 37 | //! block, and [`fold`] then REFUSES to return a conversation with more orphans than it was |
| 38 | //! given ([`orphan_count`]). A refused fold is not a dead end: [`elide_bulk`] shrinks the |
| 39 | //! same conversation without adding or removing a single message, so pairing cannot be |
| 40 | //! touched at all. |
| 41 | //! |
| 42 | //! ## The other fold |
| 43 | //! |
| 44 | //! [`crystal_proposal`] belongs to the reducer rather than to any of the above: it is the |
| 45 | //! check a proposed crystal must pass before a user is offered it to accept. It is here |
| 46 | //! because it is the same rule -- a fold is refused before it is offered, never after -- |
| 47 | //! and because the two would otherwise drift apart, which is how one of them ends up with |
| 48 | //! the reasoning and the other with the bug. |
| 49 | |
| 50 | use crate::llm::{OpenSet, extract_json_string, extract_json_string_array}; |
| 51 | use crate::protocol::{ChatMessage, ImagePart, MessageContent, ToolCall}; |
| 52 | use crate::tools::{CallOutcome, call_outcome}; |
| 53 | |
| 54 | use oxedyne_fe2o3_core::prelude::*; |
| 55 | use oxedyne_fe2o3_jdat::Dat; |
| 56 | use oxedyne_fe2o3_jdat::bdat::DecodeLimits; |
| 57 | use oxedyne_fe2o3_jdat::string::dec::DecoderConfig; |
| 58 | use oxedyne_fe2o3_jdat::usr::{UsrKind, UsrKindCode, UsrKindId}; |
| 59 | |
| 60 | use std::cell::Cell; |
| 61 | use std::collections::BTreeMap; |
| 62 | |
| 63 | |
| 64 | // ┌───────────────────────────────────────────────────────────────┐ |
| 65 | // │ The numbers │ |
| 66 | // └───────────────────────────────────────────────────────────────┘ |
| 67 | |
| 68 | /// Tool-call rounds a single turn may take before the app stops it. |
| 69 | /// |
| 70 | /// A real task -- find the wiring, read three files, make the change, build, fix two errors, |
| 71 | /// run the tests -- is comfortably thirty to fifty calls. The old ceiling of 25 stopped |
| 72 | /// such a turn in the middle and wrote its own surrender into the conversation. Runaway |
| 73 | /// cost is not what this defends against; the spend governor does that, with the user's own |
| 74 | /// figure and their own decision. |
| 75 | pub const DEFAULT_MAX_ROUNDS: usize = 150; |
| 76 | |
| 77 | /// What window to assume when nobody has said what the model's is. |
| 78 | /// |
| 79 | /// The smallest window in Daimond's own price table, so assuming it is the assumption that |
| 80 | /// fails safe: on a bigger model the conversation is folded earlier than it needed to be, |
| 81 | /// which costs one summary; on a smaller one the reactive path in the agent still catches |
| 82 | /// the refusal. |
| 83 | pub const DEFAULT_WINDOW: u64 = 131_072; |
| 84 | |
| 85 | // Where a conversation folds |
| 86 | // |
| 87 | // It was 0.8 until 2026-08-28, and the owner moved it down after a tester read a long chat |
| 88 | // as losing its history. Two things go wrong close to the ceiling and both are silent: the |
| 89 | // budget is computed against a window this app only GUESSED at, so an estimate a fifth too |
| 90 | // generous puts the request over the real line before anything here notices; and the fold's |
| 91 | // own summary is paid for out of the same window, so a fold begun at 0.8 can leave the |
| 92 | // folded conversation still over it. Folding earlier costs one summary sooner and buys the |
| 93 | // headroom both of those want. |
| 94 | // |
| 95 | // The band exists because `Limits::budget` clamped whatever it was handed anyway, and the |
| 96 | // figures now have to be read by the control that offers the choice: one offering 0.05 would |
| 97 | // be offering a conversation folded to nothing, and one offering 0.99 no fold at all until |
| 98 | // the provider refuses. |
| 99 | |
| 100 | pub const FOLD_AT: f64 = 0.65; // default, overridden by the user's own figure |
| 101 | pub const FOLD_AT_MIN: f64 = 0.1; // below this a fold leaves nothing standing |
| 102 | pub const FOLD_AT_MAX: f64 = 0.95; // above it the provider refuses before the fold runs |
| 103 | |
| 104 | /// Fraction of the budget kept verbatim at the end of the conversation. |
| 105 | /// |
| 106 | /// The recent exchanges are the ones that must survive intact: a model that has just been |
| 107 | /// told the answer, or has just read the file it is editing, must not find that summarised |
| 108 | /// away. Two fifths of the budget is several full rounds of tool calls at typical sizes. |
| 109 | pub const KEEP: f64 = 0.4; |
| 110 | |
| 111 | /// Messages always kept verbatim however big they are, so a fold cannot leave a turn with |
| 112 | /// nothing but a summary and the user's own sentence. |
| 113 | pub const MIN_KEEP_MESSAGES: usize = 6; |
| 114 | |
| 115 | /// Tokens held back from the budget for the reply itself, on top of the model's own cap. |
| 116 | const RESERVE_TOKENS: u64 = 1_024; |
| 117 | |
| 118 | /// The smallest budget that will ever be computed, so a mis-reported tiny window cannot |
| 119 | /// produce a budget nothing fits in. |
| 120 | const MIN_BUDGET_TOKENS: u64 = 1_000; |
| 121 | |
| 122 | /// The smallest window a refusal will ever teach, so one wild size estimate cannot leave a |
| 123 | /// chat folding itself to nothing on every turn. |
| 124 | pub const MIN_LEARNED_WINDOW: u64 = 4_096; |
| 125 | |
| 126 | /// The prompt size below which a bare refusal is not read as an overflow. |
| 127 | /// |
| 128 | /// Two things have to be told apart with no help from the provider: a request refused for |
| 129 | /// being too long, and a request refused for being wrong. Size is the only signal, and it |
| 130 | /// has to be an ABSOLUTE floor rather than a fraction of the budget, because the case that |
| 131 | /// matters most is the one where the budget is wrong -- a model with a 16k window that |
| 132 | /// nobody published, which the app is treating as 131k. Judged against that budget the |
| 133 | /// refused prompt looks small; judged against this it plainly is not. |
| 134 | pub const OVERFLOW_FLOOR_TOKENS: u64 = 4_000; |
| 135 | |
| 136 | /// Tokens per byte before any provider has said otherwise -- English prose and source code |
| 137 | /// both sit near this. |
| 138 | const DEFAULT_TOKENS_PER_BYTE: f64 = 0.27; |
| 139 | |
| 140 | /// The band a calibrated ratio is held inside, so one odd `usage` block cannot make the |
| 141 | /// gauge useless. |
| 142 | const MIN_TOKENS_PER_BYTE: f64 = 0.15; |
| 143 | |
| 144 | /// See [`MIN_TOKENS_PER_BYTE`]. |
| 145 | const MAX_TOKENS_PER_BYTE: f64 = 0.60; |
| 146 | |
| 147 | /// Bytes of rendered history handed to the summarising call. |
| 148 | /// |
| 149 | /// Bounded on purpose, and it is the whole reason the fold cannot fail the way the request |
| 150 | /// it is fixing failed: the part being folded is by definition near the window, so a |
| 151 | /// summarising call carrying it whole would be refused for the same reason. It is also |
| 152 | /// what makes the cost of a fold a fixed number rather than a fraction of the session. |
| 153 | /// |
| 154 | /// It has been cited as the reason a crystal carrying a page would be dear to fold. That |
| 155 | /// reasoning was for a design where the crystal was one markdown file with the markup |
| 156 | /// inside it, and it does not survive the split: presentation lives in `crystal.html`, |
| 157 | /// which the reducer is never shown and which never enters a system prompt, so no page |
| 158 | /// reaches this budget by any route. The crystal's data half does, since it rides in the |
| 159 | /// standing context of every request -- which is what its own cap is for, and why that cap |
| 160 | /// bounds only the half a model reads. |
| 161 | pub const FOLD_INPUT_CAP: u64 = 48_000; |
| 162 | |
| 163 | /// Bytes of a tool result kept when it is elided in place. |
| 164 | pub const TOOL_ELISION_CAP: usize = 400; |
| 165 | |
| 166 | /// The side of the square block of pixels a vision model charges one token for. |
| 167 | /// |
| 168 | /// From Anthropic's vision documentation: "Claude views images in patches instead of pixels. Each |
| 169 | /// patch is a 28x28-pixel block of the image, referred to as a visual token. An image, therefore, |
| 170 | /// costs ceil(width / 28) x ceil(height / 28) visual tokens." OpenAI tiles differently -- 85 |
| 171 | /// tokens plus 170 per 512-pixel tile after fitting the image inside 2048x2048 with a 768-pixel |
| 172 | /// short edge -- and comes out LOWER on every size worth sending, so the one formula here is the |
| 173 | /// dearer of the two and therefore the safe one to budget against. |
| 174 | pub const IMAGE_PATCH_PX: u64 = 28; |
| 175 | |
| 176 | /// The most visual tokens one image can cost, whatever its size. |
| 177 | /// |
| 178 | /// The provider downscales anything bigger before it charges for it, so the patch count above is |
| 179 | /// bounded: 4,784 on the high-resolution tier (long edge 2,576 px), 1,568 on the standard tier |
| 180 | /// (long edge 1,568 px). The higher figure is the one used, because nothing in the browser knows |
| 181 | /// which tier a configured model sits in, and a budget that overstates folds a turn early where |
| 182 | /// one that understates has the provider refuse it. |
| 183 | pub const IMAGE_TOKEN_CAP: u64 = 4_784; |
| 184 | |
| 185 | /// Tokens per byte for an image whose pixel dimensions this build cannot read. |
| 186 | /// |
| 187 | /// PNG and JPEG headers are read directly, so this covers GIF and WebP and a file whose header is |
| 188 | /// damaged. Measured against this repository's own screenshots -- 1500x950 in 199,280 bytes is |
| 189 | /// 1,836 visual tokens (0.0092 tokens per byte) and 390x844 in 52,783 bytes is 434 (0.0082) -- and |
| 190 | /// then set several times higher, because the two formats it actually covers compress harder for |
| 191 | /// the same pixel count and because overstating is the direction to err in. It is bounded above |
| 192 | /// by [`IMAGE_TOKEN_CAP`] regardless, so a large file is charged the ceiling rather than a |
| 193 | /// runaway figure. |
| 194 | const FALLBACK_IMAGE_TOKENS_PER_BYTE: f64 = 0.05; |
| 195 | |
| 196 | /// Tokens the summarising call may generate. |
| 197 | /// |
| 198 | /// The CONTEXT fold's, and only that one -- it is set on the client where the summarising |
| 199 | /// call is made in `agent.rs` and nowhere else. Nothing caps what the crystal reducer |
| 200 | /// emits, which is a different fold under a different prompt, so a figure derived from this |
| 201 | /// one is not a statement about how large a crystal a fold can produce. The crystal's own |
| 202 | /// ceiling is [`crate::tools::crystal_cap`], enforced at the write. |
| 203 | pub const FOLD_MAX_TOKENS: u32 = 1_400; |
| 204 | |
| 205 | |
| 206 | // ┌───────────────────────────────────────────────────────────────┐ |
| 207 | // │ Limits │ |
| 208 | // └───────────────────────────────────────────────────────────────┘ |
| 209 | |
| 210 | /// What bounds a turn: how long it may run, and how big its conversation may get. |
| 211 | /// |
| 212 | /// Held as data rather than as constants because both are per-model and per-user: the |
| 213 | /// window is whatever the provider published for the model in the chat's header, and a user |
| 214 | /// who wants a longer leash than 150 rounds should be able to have one without a rebuild. |
| 215 | #[derive(Clone, Debug)] |
| 216 | pub struct Limits { |
| 217 | /// Tool-call rounds a single turn may take. |
| 218 | pub max_rounds: usize, |
| 219 | /// The model's context window in tokens; zero when nobody has said. |
| 220 | pub window: u64, |
| 221 | /// Fraction of the window at which the conversation is folded. |
| 222 | pub fold_at: f64, |
| 223 | /// Fraction of the budget kept verbatim at the end. |
| 224 | pub keep: f64, |
| 225 | /// Model to fold with; empty means the chat's own. |
| 226 | pub fold_model: String, |
| 227 | } |
| 228 | |
| 229 | impl Default for Limits { |
| 230 | fn default() -> Self { |
| 231 | Self { |
| 232 | max_rounds: DEFAULT_MAX_ROUNDS, |
| 233 | window: 0, |
| 234 | fold_at: FOLD_AT, |
| 235 | keep: KEEP, |
| 236 | fold_model: String::new(), |
| 237 | } |
| 238 | } |
| 239 | } |
| 240 | |
| 241 | impl Limits { |
| 242 | |
| 243 | /// The largest prompt this turn should send, in tokens. |
| 244 | /// |
| 245 | /// Two ceilings, and the lower wins. The first is the fraction of the window a fold is |
| 246 | /// meant to trigger at. The second is what is left of the window once the reply the |
| 247 | /// model is allowed to generate has been subtracted -- on a small window the reply is |
| 248 | /// the bigger share, and a budget that ignored it would leave the prompt legal and the |
| 249 | /// call still refused. |
| 250 | /// |
| 251 | /// # Arguments |
| 252 | /// * `max_completion` - The cap the client puts on generated tokens. |
| 253 | pub fn budget(&self, max_completion: u32) -> u64 { |
| 254 | let w = if self.window == 0 { DEFAULT_WINDOW } else { self.window }; |
| 255 | let by_fraction = (w as f64 * self.fold_at.clamp(FOLD_AT_MIN, FOLD_AT_MAX)) as u64; |
| 256 | // Never more than half the window to the reply, however big the client's cap is. |
| 257 | // A cap larger than the whole window is not a reason to leave no budget for the |
| 258 | // conversation; it is a reason to ignore most of the cap. |
| 259 | let reserved = (max_completion as u64).min(w / 2) + RESERVE_TOKENS; |
| 260 | let by_reserve = w.saturating_sub(reserved); |
| 261 | by_fraction.min(by_reserve).max(MIN_BUDGET_TOKENS) |
| 262 | } |
| 263 | |
| 264 | /// Take a provider's refusal as the fact it is: a prompt this big did not fit. |
| 265 | /// |
| 266 | /// The one authority on the window is the provider, and this is the only occasion it |
| 267 | /// speaks -- so a chat against a model nobody published a window for learns its size the |
| 268 | /// hard way, once, rather than dying of it repeatedly. The window is set BELOW the |
| 269 | /// prompt that was refused, because that prompt is known not to have fitted, and never |
| 270 | /// upward: a bigger figure than the one already held would undo what an earlier refusal |
| 271 | /// taught. |
| 272 | /// |
| 273 | /// Returns whether the figure moved. |
| 274 | /// |
| 275 | /// # Arguments |
| 276 | /// * `refused_tokens` - Estimated size of the prompt the provider would not take. |
| 277 | pub fn learn_from_refusal(&mut self, refused_tokens: u64) -> bool { |
| 278 | let held = if self.window == 0 { DEFAULT_WINDOW } else { self.window }; |
| 279 | let learnt = (refused_tokens * 3 / 4).max(MIN_LEARNED_WINDOW); |
| 280 | if learnt >= held { |
| 281 | return false; |
| 282 | } |
| 283 | self.window = learnt; |
| 284 | true |
| 285 | } |
| 286 | |
| 287 | /// How much of the budget is kept verbatim, in tokens. |
| 288 | pub fn tail_budget(&self, max_completion: u32) -> u64 { |
| 289 | ((self.budget(max_completion) as f64) * self.keep.clamp(0.1, 0.8)) as u64 |
| 290 | } |
| 291 | } |
| 292 | |
| 293 | |
| 294 | // ┌───────────────────────────────────────────────────────────────┐ |
| 295 | // │ Gauge │ |
| 296 | // └───────────────────────────────────────────────────────────────┘ |
| 297 | |
| 298 | /// How many tokens a byte of this conversation costs, learned from the provider. |
| 299 | /// |
| 300 | /// Nothing in the browser can tokenise the way a given model does, and shipping a tokeniser |
| 301 | /// per model is not on offer. But the provider reports `prompt_tokens` for every call, and |
| 302 | /// the bytes that produced that figure are known here -- so the ratio is measured rather |
| 303 | /// than assumed, against the only authority that matters. It is held inside a band so that |
| 304 | /// one odd `usage` block, or a provider that reports nothing, cannot leave the gauge unable |
| 305 | /// to see an overflow coming. |
| 306 | #[derive(Debug, Default)] |
| 307 | pub struct Gauge { |
| 308 | /// Bytes of the last request whose token count came back. |
| 309 | bytes: Cell<u64>, |
| 310 | /// What the provider charged for those bytes. |
| 311 | tokens: Cell<u64>, |
| 312 | } |
| 313 | |
| 314 | impl Gauge { |
| 315 | |
| 316 | /// Record what a request of `bytes` actually cost in prompt tokens. |
| 317 | /// |
| 318 | /// A call that reported nothing is ignored rather than recorded as free. |
| 319 | /// |
| 320 | /// # Arguments |
| 321 | /// * `bytes` - Bytes of conversation and tool schema sent. |
| 322 | /// * `tokens` - Prompt tokens the provider says it charged. |
| 323 | pub fn observe(&self, bytes: u64, tokens: u64) { |
| 324 | if bytes == 0 || tokens == 0 { |
| 325 | return; |
| 326 | } |
| 327 | self.bytes.set(bytes); |
| 328 | self.tokens.set(tokens); |
| 329 | } |
| 330 | |
| 331 | /// Tokens per byte, as last measured or the default. |
| 332 | pub fn ratio(&self) -> f64 { |
| 333 | let b = self.bytes.get(); |
| 334 | if b == 0 { |
| 335 | return DEFAULT_TOKENS_PER_BYTE; |
| 336 | } |
| 337 | (self.tokens.get() as f64 / b as f64).clamp(MIN_TOKENS_PER_BYTE, MAX_TOKENS_PER_BYTE) |
| 338 | } |
| 339 | |
| 340 | /// What `bytes` will cost in tokens. |
| 341 | pub fn tokens(&self, bytes: u64) -> u64 { |
| 342 | (bytes as f64 * self.ratio()).ceil() as u64 |
| 343 | } |
| 344 | |
| 345 | /// How many bytes fit in `tokens`. |
| 346 | pub fn bytes(&self, tokens: u64) -> u64 { |
| 347 | (tokens as f64 / self.ratio()) as u64 |
| 348 | } |
| 349 | } |
| 350 | |
| 351 | |
| 352 | // ┌───────────────────────────────────────────────────────────────┐ |
| 353 | // │ Measuring │ |
| 354 | // └───────────────────────────────────────────────────────────────┘ |
| 355 | |
| 356 | /// What one image costs the model, in tokens. |
| 357 | /// |
| 358 | /// Pixels, not bytes. A screenshot is the clearest case: 1500x950 weighs 199,280 bytes on disk, |
| 359 | /// which the text ratio of 0.27 prices at 53,806 tokens; the provider charges 1,836. Left alone, |
| 360 | /// that thirty-fold overstatement puts a single screenshot most of the way through a 131k budget |
| 361 | /// and folds the conversation on the turn it was read -- and on every turn after it, since the |
| 362 | /// fold cannot shrink what it is mis-measuring. A feature that cannot survive its own first use |
| 363 | /// is not a feature, which is why this function exists. |
| 364 | /// |
| 365 | /// See [`IMAGE_PATCH_PX`] for the formula and where it comes from, [`IMAGE_TOKEN_CAP`] for the |
| 366 | /// ceiling, and [`FALLBACK_IMAGE_TOKENS_PER_BYTE`] for the formats whose header this build cannot |
| 367 | /// read. |
| 368 | /// |
| 369 | /// # Arguments |
| 370 | /// * `img` - The image part being measured. |
| 371 | pub fn image_tokens(img: &ImagePart) -> u64 { |
| 372 | let raw = match img.dims() { |
| 373 | Some((w, h)) => { |
| 374 | let cols = (w as u64).div_ceil(IMAGE_PATCH_PX); |
| 375 | let rows = (h as u64).div_ceil(IMAGE_PATCH_PX); |
| 376 | cols.saturating_mul(rows) |
| 377 | }, |
| 378 | None => (img.data.len() as f64 * FALLBACK_IMAGE_TOKENS_PER_BYTE).ceil() as u64, |
| 379 | }; |
| 380 | raw.min(IMAGE_TOKEN_CAP).max(1) |
| 381 | } |
| 382 | |
| 383 | /// What one image costs in the currency the rest of this module counts in: bytes. |
| 384 | /// |
| 385 | /// Everything here -- the tail budget, the elision target, the fold trigger -- is measured in |
| 386 | /// bytes and converted to tokens once, by [`Gauge`]. An image has no honest byte count in that |
| 387 | /// sense, so it is given the byte count that WOULD produce its token count at the default ratio. |
| 388 | /// Two things follow, and both are the point: every existing byte-denominated calculation keeps |
| 389 | /// working untouched, and the gauge's learned ratio stays near the ratio for text, because the |
| 390 | /// image is no longer being fed to it as thirty times its own weight. |
| 391 | /// |
| 392 | /// # Arguments |
| 393 | /// * `img` - The image part being measured. |
| 394 | pub fn image_bytes(img: &ImagePart) -> u64 { |
| 395 | (image_tokens(img) as f64 / DEFAULT_TOKENS_PER_BYTE).ceil() as u64 |
| 396 | } |
| 397 | |
| 398 | /// Bytes one message costs on the wire, its role framing and any tool calls included. |
| 399 | /// |
| 400 | /// Bytes rather than characters: multi-byte text overstates slightly, and overstating is |
| 401 | /// the direction a size estimate should err in. |
| 402 | /// |
| 403 | /// An image is counted by [`image_bytes`], NOT by its own length -- see there for why the |
| 404 | /// difference is the whole of this feature's viability. |
| 405 | /// |
| 406 | /// **A fold is sized by what SERIALISATION will send, which is not what the model wrote.** |
| 407 | /// [`crate::llm::sent_args_len`] and [`crate::llm::sent_text_len`] are the serialiser's own |
| 408 | /// functions, asked here rather than imitated: a closed fold's body leaves the payload on the way |
| 409 | /// out, so counting it kept the trigger measuring a conversation that was never going to be sent. |
| 410 | /// Whether it leaves depends on the fold state, which is why `open` is a parameter of every size |
| 411 | /// in this module. |
| 412 | /// |
| 413 | /// Both depths of answer are covered. A STORED `say` call -- the tool is gone, but a conversation |
| 414 | /// saved before it went still carries them -- folds through tool arguments and is sized by them; |
| 415 | /// an inline `<details>` folds through the assistant's own prose and is sized by its text. Sizing |
| 416 | /// one and not the other would leave the same defect standing, just moved. |
| 417 | /// |
| 418 | /// # Arguments |
| 419 | /// * `m` - The message to size. |
| 420 | /// * `open` - The folds the user has open, from [`crate::llm::LlmClient::open_folds`] -- stored |
| 421 | /// `say` call ids and inline fold keys in the one set. |
| 422 | pub fn msg_bytes(m: &ChatMessage, open: &OpenSet) -> u64 { |
| 423 | // The JSON around every message: the role, the two keys, the braces and the commas. |
| 424 | let mut n = 32u64; |
| 425 | for img in m.content().images() { |
| 426 | n += image_bytes(img); |
| 427 | } |
| 428 | match m { |
| 429 | ChatMessage::Assistant { content, tool_calls } => { |
| 430 | n += crate::llm::sent_text_len(&content.as_text(), open) as u64; |
| 431 | for tc in tool_calls { |
| 432 | let args = crate::llm::sent_args_len( |
| 433 | &tc.name, &tc.arguments, open.contains(&tc.id)); |
| 434 | n += (tc.id.len() + tc.name.len() + args) as u64 + 64; |
| 435 | } |
| 436 | }, |
| 437 | ChatMessage::Tool { tool_call_id, .. } => { |
| 438 | n += m.content().text_len() as u64 + tool_call_id.len() as u64; |
| 439 | }, |
| 440 | _ => n += m.content().text_len() as u64, |
| 441 | } |
| 442 | n |
| 443 | } |
| 444 | |
| 445 | /// Bytes a whole conversation costs on the wire. |
| 446 | /// |
| 447 | /// # Arguments |
| 448 | /// * `msgs` - The conversation, oldest first. |
| 449 | /// * `open` - The folds the user has open. See [`msg_bytes`]. |
| 450 | pub fn conversation_bytes(msgs: &[ChatMessage], open: &OpenSet) -> u64 { |
| 451 | msgs.iter().map(|m| msg_bytes(m, open)).sum() |
| 452 | } |
| 453 | |
| 454 | |
| 455 | // ┌───────────────────────────────────────────────────────────────┐ |
| 456 | // │ Pairing │ |
| 457 | // └───────────────────────────────────────────────────────────────┘ |
| 458 | |
| 459 | /// Tool calls with no reply, plus tool replies with no call. |
| 460 | /// |
| 461 | /// The number a fold is not allowed to increase. It is counted rather than merely detected |
| 462 | /// because a conversation can arrive already broken -- a session read back from storage |
| 463 | /// loses the `tool_calls` off its assistant turns -- and a fold that refused to run on such |
| 464 | /// a conversation would leave it with no way out of an overflow at all. What matters is |
| 465 | /// that the fold adds none of its own. |
| 466 | pub fn orphan_count(msgs: &[ChatMessage]) -> usize { |
| 467 | let mut n = 0; |
| 468 | let mut i = 0; |
| 469 | while i < msgs.len() { |
| 470 | match &msgs[i] { |
| 471 | ChatMessage::Assistant { tool_calls, .. } if !tool_calls.is_empty() => { |
| 472 | let mut k = 0; |
| 473 | while k < tool_calls.len() { |
| 474 | match msgs.get(i + 1 + k) { |
| 475 | Some(ChatMessage::Tool { tool_call_id, .. }) |
| 476 | if *tool_call_id == tool_calls[k].id => k += 1, |
| 477 | _ => break, |
| 478 | } |
| 479 | } |
| 480 | n += tool_calls.len() - k; // calls nobody answered |
| 481 | i += 1 + k; |
| 482 | }, |
| 483 | ChatMessage::Tool { .. } => { n += 1; i += 1; }, // a reply to no call |
| 484 | _ => i += 1, |
| 485 | } |
| 486 | } |
| 487 | n |
| 488 | } |
| 489 | |
| 490 | /// Whether every tool call is answered and every tool reply was asked for. |
| 491 | pub fn pairing_is_whole(msgs: &[ChatMessage]) -> bool { |
| 492 | orphan_count(msgs) == 0 |
| 493 | } |
| 494 | |
| 495 | |
| 496 | // ┌───────────────────────────────────────────────────────────────┐ |
| 497 | // │ Choosing the cut │ |
| 498 | // └───────────────────────────────────────────────────────────────┘ |
| 499 | |
| 500 | /// Where the verbatim tail begins: the index of the first message to keep. |
| 501 | /// |
| 502 | /// Walks back from the end taking messages until `keep_bytes` is spent, then walks back |
| 503 | /// further over any tool replies, because a reply may never be the first kept message: its |
| 504 | /// call would have been folded away and the provider would reject the whole request. That |
| 505 | /// backward step keeps a WHOLE block rather than dropping the rest of it, so a fold errs |
| 506 | /// towards keeping too much, never towards keeping something unanswerable. |
| 507 | /// |
| 508 | /// `min_keep` is a preference, not a promise, and `hard_bytes` is why. Six messages is the |
| 509 | /// right number when messages are ordinary; when two of them are twenty-kilobyte replies it |
| 510 | /// is a tail bigger than the whole budget, and a fold that honoured it would shrink the |
| 511 | /// conversation by two messages and leave it just as unsendable as before. That is not |
| 512 | /// hypothetical -- it is what a mock provider with a real context ceiling did to the first |
| 513 | /// version of this function. So the ceiling wins, and the last message is kept whatever |
| 514 | /// its size, because it is the request being answered. |
| 515 | /// |
| 516 | /// Zero means there is nothing to fold. |
| 517 | /// |
| 518 | /// # Arguments |
| 519 | /// * `msgs` - The conversation, oldest first. |
| 520 | /// * `keep_bytes` - Bytes of tail to keep verbatim. |
| 521 | /// * `min_keep` - Messages to keep where they fit inside `hard_bytes`. |
| 522 | /// * `hard_bytes` - Bytes the tail may not exceed, whatever `min_keep` asks for. |
| 523 | /// * `open` - The folds the user has open, so the tail is measured as it will be sent. See |
| 524 | /// [`msg_bytes`]. |
| 525 | pub fn tail_start( |
| 526 | msgs: &[ChatMessage], |
| 527 | keep_bytes: u64, |
| 528 | min_keep: usize, |
| 529 | hard_bytes: u64, |
| 530 | open: &OpenSet, |
| 531 | ) |
| 532 | -> usize |
| 533 | { |
| 534 | let n = msgs.len(); |
| 535 | if n <= min_keep { |
| 536 | return 0; |
| 537 | } |
| 538 | let mut i = n; |
| 539 | let mut acc = 0u64; |
| 540 | while i > 0 { |
| 541 | let c = msg_bytes(&msgs[i - 1], open); |
| 542 | let kept = n - i; |
| 543 | if kept >= 1 && acc + c > hard_bytes { |
| 544 | break; |
| 545 | } |
| 546 | if kept >= min_keep && acc + c > keep_bytes { |
| 547 | break; |
| 548 | } |
| 549 | acc += c; |
| 550 | i -= 1; |
| 551 | } |
| 552 | // A tool reply cannot open the kept tail; step back to the assistant turn that asked |
| 553 | // for it, taking the rest of its block along. |
| 554 | while i > 0 && matches!(msgs[i], ChatMessage::Tool { .. }) { |
| 555 | i -= 1; |
| 556 | } |
| 557 | i |
| 558 | } |
| 559 | |
| 560 | |
| 561 | // ┌───────────────────────────────────────────────────────────────┐ |
| 562 | // │ The ledger │ |
| 563 | // └───────────────────────────────────────────────────────────────┘ |
| 564 | |
| 565 | /// What the folded part of a conversation actually did. |
| 566 | /// |
| 567 | /// Derived from the tool calls and their replies, so it is a record rather than a claim. |
| 568 | /// The distinction between this and the prose summary is the whole point of the module: a |
| 569 | /// file read twenty rounds ago is worth one line, but WHICH files were read and written is |
| 570 | /// worth keeping exactly, because a model that has lost it will describe work it cannot |
| 571 | /// check. |
| 572 | #[derive(Clone, Debug, Default)] |
| 573 | pub struct Ledger { |
| 574 | /// Files and directories read or searched. |
| 575 | pub read: Vec<String>, |
| 576 | /// Files created, changed, moved or deleted. |
| 577 | pub wrote: Vec<String>, |
| 578 | /// Commands run. |
| 579 | pub ran: Vec<String>, |
| 580 | /// Pages fetched or driven. |
| 581 | pub fetched: Vec<String>, |
| 582 | /// Agents dispatched, by name. |
| 583 | pub spawned: Vec<String>, |
| 584 | /// Calls a door refused, as `tool path`. Kept apart from [`Ledger::failed`] because they |
| 585 | /// are different news for the model: a refusal is a rule it met and can work around, an |
| 586 | /// error is a tool that broke. Told the wrong one, it retries the wrong thing. |
| 587 | pub refused: Vec<String>, |
| 588 | /// Calls that came back an error, as `tool path`. Kept apart from the rest so a fold |
| 589 | /// never reports an attempted write as a write. |
| 590 | pub failed: Vec<String>, |
| 591 | } |
| 592 | |
| 593 | impl Ledger { |
| 594 | |
| 595 | /// Whether nothing at all was recorded. |
| 596 | pub fn is_empty(&self) -> bool { |
| 597 | self.read.is_empty() && self.wrote.is_empty() && self.ran.is_empty() |
| 598 | && self.fetched.is_empty() && self.spawned.is_empty() && self.refused.is_empty() |
| 599 | && self.failed.is_empty() |
| 600 | } |
| 601 | |
| 602 | /// The ledger as lines for the fold notice, each capped so one pathological session |
| 603 | /// cannot fill the window with its own history. |
| 604 | pub fn lines(&self) -> Vec<String> { |
| 605 | let mut out = Vec::new(); |
| 606 | let mut put = |label: &str, items: &Vec<String>| { |
| 607 | if items.is_empty() { |
| 608 | return; |
| 609 | } |
| 610 | let shown = items.len().min(40); |
| 611 | let mut s = fmt!("{}: {}", label, items[..shown].join(", ")); |
| 612 | if items.len() > shown { |
| 613 | s.push_str(&fmt!(" (and {} more)", items.len() - shown)); |
| 614 | } |
| 615 | out.push(s); |
| 616 | }; |
| 617 | put("Files read", &self.read); |
| 618 | put("Files written", &self.wrote); |
| 619 | put("Commands run", &self.ran); |
| 620 | put("Pages fetched", &self.fetched); |
| 621 | put("Agents dispatched", &self.spawned); |
| 622 | // Said in full rather than as one word: "REFUSED" alone invites the model to read a |
| 623 | // closed door as a broken tool and try it again. |
| 624 | put("REFUSED, so nothing was touched", &self.refused); |
| 625 | put("FAILED", &self.failed); |
| 626 | out |
| 627 | } |
| 628 | |
| 629 | /// Add `item` to `into` unless it is already there or empty. |
| 630 | fn push(into: &mut Vec<String>, item: String) { |
| 631 | let t = item.trim(); |
| 632 | if t.is_empty() || into.iter().any(|x| x == t) { |
| 633 | return; |
| 634 | } |
| 635 | into.push(t.to_string()); |
| 636 | } |
| 637 | } |
| 638 | |
| 639 | /// Read the ledger out of a conversation's tool calls. |
| 640 | /// |
| 641 | /// **What a reply MEANS is the tool layer's to say, and [`call_outcome`] is where it says it.** |
| 642 | /// This module used to decide for itself, with `!reply.trim_start().starts_with("Error")` -- and |
| 643 | /// every refusal in the build is returned as an `Ok` sentence that opens "Refused", so a write the |
| 644 | /// scope fence had just stopped was booked as a write. A live run refused two writes by the chat |
| 645 | /// fence and the fold then told the model `Files written: alpha.txt, beta.txt`, directly above the |
| 646 | /// line instructing it to claim nothing that is not listed. The ledger is the half of a fold that |
| 647 | /// is supposed to be arithmetic rather than judgement, and it was inventing files. |
| 648 | /// |
| 649 | /// Adding "Refused" to the sniff would have been the same defect one layer along: a list of |
| 650 | /// openings held in this file, against wording held in another, kept in step by nobody. So the |
| 651 | /// question is asked of the layer that composes the wording, and answered there. |
| 652 | pub fn ledger_of(msgs: &[ChatMessage]) -> Ledger { |
| 653 | let mut l = Ledger::default(); |
| 654 | let mut i = 0; |
| 655 | while i < msgs.len() { |
| 656 | let calls = match &msgs[i] { |
| 657 | ChatMessage::Assistant { tool_calls, .. } if !tool_calls.is_empty() => tool_calls, |
| 658 | _ => { i += 1; continue; }, |
| 659 | }; |
| 660 | for (k, tc) in calls.iter().enumerate() { |
| 661 | let reply = match msgs.get(i + 1 + k) { |
| 662 | Some(ChatMessage::Tool { content, .. }) => content.as_text(), |
| 663 | _ => std::borrow::Cow::Borrowed(""), |
| 664 | }; |
| 665 | record(&mut l, tc, call_outcome(&reply)); |
| 666 | } |
| 667 | i += 1 + calls.len(); |
| 668 | } |
| 669 | l |
| 670 | } |
| 671 | |
| 672 | /// Put one call in the right column of the ledger. |
| 673 | /// |
| 674 | /// Whether it did anything is NOT decided here: [`crate::tools::call_outcome`] decides it, beside |
| 675 | /// the two constructors that compose every refusal and every error, and the answer arrives as |
| 676 | /// `outcome`. |
| 677 | /// |
| 678 | /// Which column a call that DID happen belongs in is still worked out from its name. That |
| 679 | /// duplicates what `Tool::write_targets` knows, and the standing note here said a visibility |
| 680 | /// change would let this call it instead -- it would not: `write_targets` needs a `ToolContext`, |
| 681 | /// which is the turn's, and a fold reads a conversation long after the turn that made it is gone. |
| 682 | /// The mapping below is over tool NAMES, which is the stable wire contract; a tool added without |
| 683 | /// a line here is listed in no column, which is a fold that says less than it could and never one |
| 684 | /// that says something untrue. |
| 685 | /// |
| 686 | /// # Arguments |
| 687 | /// * `l` - The ledger being built. |
| 688 | /// * `tc` - The call the model made. |
| 689 | /// * `outcome` - What the tool layer says became of it. |
| 690 | fn record(l: &mut Ledger, tc: &ToolCall, outcome: CallOutcome) { |
| 691 | let arg = |k: &str| extract_json_string(&tc.arguments, k).unwrap_or_default(); |
| 692 | let path = arg("path"); |
| 693 | // A call that did not happen is recorded as what it was and in no other column. A refusal |
| 694 | // and an error are held apart because they are different news: one is a rule the model met, |
| 695 | // the other is a tool that broke. |
| 696 | if !matches!(outcome, CallOutcome::Done) { |
| 697 | let what = if path.is_empty() { arg("url") } else { path.clone() }; |
| 698 | let entry = fmt!("{} {}", tc.name, what).trim().to_string(); |
| 699 | match outcome { |
| 700 | CallOutcome::Refused => Ledger::push(&mut l.refused, entry), |
| 701 | CallOutcome::Failed => Ledger::push(&mut l.failed, entry), |
| 702 | CallOutcome::Done => {}, // answered above |
| 703 | } |
| 704 | return; |
| 705 | } |
| 706 | match tc.name.as_str() { |
| 707 | "file_read" => Ledger::push(&mut l.read, path), |
| 708 | "file_list" | "file_search" => Ledger::push(&mut l.read, |
| 709 | if path.is_empty() { fmt!(".") } else { path }), |
| 710 | "file_write" | "file_edit" | "file_delete" |
| 711 | | "dir_create" | "file_fetch" | "artefact_add" => Ledger::push(&mut l.wrote, path), |
| 712 | "typst_compile" => Ledger::push(&mut l.wrote, path), |
| 713 | "file_move" => Ledger::push(&mut l.wrote, |
| 714 | fmt!("{} -> {}", path, arg("to"))), |
| 715 | "shell" => Ledger::push(&mut l.ran, arg("command")), |
| 716 | "run" => { |
| 717 | let argv = extract_json_string_array(&tc.arguments, "argv").unwrap_or_default(); |
| 718 | Ledger::push(&mut l.ran, argv.join(" ")); |
| 719 | }, |
| 720 | "web_fetch" | "web_open" | "web_read" => Ledger::push(&mut l.fetched, arg("url")), |
| 721 | "spawn_agent" => Ledger::push(&mut l.spawned, arg("name")), |
| 722 | _ => {}, |
| 723 | } |
| 724 | } |
| 725 | |
| 726 | |
| 727 | // ┌───────────────────────────────────────────────────────────────┐ |
| 728 | // │ Rendering what is being folded │ |
| 729 | // └───────────────────────────────────────────────────────────────┘ |
| 730 | |
| 731 | /// One message as a line of transcript for the summarising call. |
| 732 | fn render_one(m: &ChatMessage) -> String { |
| 733 | match m { |
| 734 | ChatMessage::System { content } => fmt!("[system] {}", clip(&content.as_text(), 600)), |
| 735 | ChatMessage::User { content } => fmt!("user: {}", clip(&content.as_text(), 2_000)), |
| 736 | ChatMessage::Assistant { content, tool_calls } => { |
| 737 | let mut s = fmt!("assistant: {}", clip(&content.as_text(), 2_000)); |
| 738 | for tc in tool_calls { |
| 739 | s.push_str(&fmt!("\n calls {}({})", tc.name, clip(&tc.arguments, 240))); |
| 740 | } |
| 741 | s |
| 742 | }, |
| 743 | ChatMessage::Tool { content, .. } => |
| 744 | fmt!(" result ({} bytes): {}", content.text_len(), clip(&content.as_text(), 300)), |
| 745 | } |
| 746 | } |
| 747 | |
| 748 | /// `s` cut to `n` bytes on a character boundary, with a marker when anything went. |
| 749 | fn clip(s: &str, n: usize) -> String { |
| 750 | if s.len() <= n { |
| 751 | return s.to_string(); |
| 752 | } |
| 753 | let mut end = n; |
| 754 | while end > 0 && !s.is_char_boundary(end) { |
| 755 | end -= 1; |
| 756 | } |
| 757 | fmt!("{}… (+{} bytes)", &s[..end], s.len() - end) |
| 758 | } |
| 759 | |
| 760 | /// The part being folded, rendered for the model that will summarise it, never larger than |
| 761 | /// `cap` bytes. |
| 762 | /// |
| 763 | /// The cap is the point. What is being folded is by definition most of a context window, |
| 764 | /// so handing it over whole would fail exactly as the request that triggered the fold |
| 765 | /// failed. When it does not fit, the OPENING quarter is kept and the rest of the budget |
| 766 | /// goes to the most recent part: the opening is where the user said what they wanted, which |
| 767 | /// nothing later restates, and the recent part is where the work actually is. What is |
| 768 | /// dropped is the middle, and the notice says how much. |
| 769 | /// |
| 770 | /// # Arguments |
| 771 | /// * `msgs` - The messages being folded, oldest first. |
| 772 | /// * `cap` - Bytes the rendering may occupy. |
| 773 | pub fn render_for_fold(msgs: &[ChatMessage], cap: u64) -> String { |
| 774 | let lines: Vec<String> = msgs.iter().map(render_one).collect(); |
| 775 | let total: u64 = lines.iter().map(|l| l.len() as u64 + 1).sum(); |
| 776 | if total <= cap { |
| 777 | return lines.join("\n"); |
| 778 | } |
| 779 | let head_cap = cap / 4; |
| 780 | let mut used = 0u64; |
| 781 | let mut i = 0; |
| 782 | while i < lines.len() { |
| 783 | let c = lines[i].len() as u64 + 1; |
| 784 | if used + c > head_cap { |
| 785 | break; |
| 786 | } |
| 787 | used += c; |
| 788 | i += 1; |
| 789 | } |
| 790 | let mut j = lines.len(); |
| 791 | let mut tused = 0u64; |
| 792 | while j > i { |
| 793 | let c = lines[j - 1].len() as u64 + 1; |
| 794 | if used + tused + c > cap { |
| 795 | break; |
| 796 | } |
| 797 | tused += c; |
| 798 | j -= 1; |
| 799 | } |
| 800 | let mut out: Vec<String> = lines[..i].to_vec(); |
| 801 | if j > i { |
| 802 | out.push(fmt!("[{} messages in the middle were too long to include here]", j - i)); |
| 803 | } |
| 804 | out.extend_from_slice(&lines[j..]); |
| 805 | out.join("\n") |
| 806 | } |
| 807 | |
| 808 | |
| 809 | // ┌───────────────────────────────────────────────────────────────┐ |
| 810 | // │ The notice │ |
| 811 | // └───────────────────────────────────────────────────────────────┘ |
| 812 | |
| 813 | /// The one message a fold leaves in place of everything it folded. |
| 814 | /// |
| 815 | /// A user message, and deliberately not an assistant one: a summary in the assistant's own |
| 816 | /// voice is the model reading invented memories as things it said, and the round-limit note |
| 817 | /// this codebase used to write proved how badly that reads. Not a system message either, |
| 818 | /// for a duller reason that decides it -- the browser rehydrates a reloaded chat from |
| 819 | /// `user` and `assistant` messages only, so a system-role fold would silently vanish on |
| 820 | /// reload and the conversation would spring back to full size. |
| 821 | /// |
| 822 | /// # Arguments |
| 823 | /// * `folded` - How many messages went. |
| 824 | /// * `summary` - The model's prose, or empty when the call could not be made. |
| 825 | /// * `ledger` - What the folded part did. |
| 826 | /// * `why` - Why there is no prose, when there is none. |
| 827 | pub fn notice(folded: usize, summary: &str, ledger: &Ledger, why: Option<&str>) -> String { |
| 828 | let mut s = fmt!( |
| 829 | "[Daimond folded the earlier part of this conversation to keep it inside the model's \ |
| 830 | context window. {} messages were replaced by this note; everything after it is \ |
| 831 | exactly as it was. This note is the only record of them that the model now has.]\n", |
| 832 | folded); |
| 833 | if !summary.trim().is_empty() { |
| 834 | s.push_str("\n## What happened\n\n"); |
| 835 | s.push_str(summary.trim()); |
| 836 | s.push('\n'); |
| 837 | } else if let Some(reason) = why { |
| 838 | s.push_str(&fmt!( |
| 839 | "\n## What happened\n\nNo summary could be written ({}), so what follows is the \ |
| 840 | record of the tool calls themselves and nothing more. Ask before assuming \ |
| 841 | anything not listed here was done.\n", reason)); |
| 842 | } |
| 843 | let lines = ledger.lines(); |
| 844 | if !lines.is_empty() { |
| 845 | s.push_str("\n## What was touched\n\n"); |
| 846 | for l in lines { |
| 847 | s.push_str(&fmt!("- {}\n", l)); |
| 848 | } |
| 849 | } |
| 850 | s.push_str( |
| 851 | "\nFile contents from before this note are gone; read a file again rather than \ |
| 852 | recalling it. Do not claim to have done anything that is not listed above."); |
| 853 | s |
| 854 | } |
| 855 | |
| 856 | |
| 857 | /// What goes into the conversation when a turn is stopped at the round limit. |
| 858 | /// |
| 859 | /// A SYSTEM message, because the limit is the app's and so is the sentence. It used to be |
| 860 | /// written as an assistant message reading `[Reached the tool-call round limit (25).]`, so |
| 861 | /// the next turn began with the model reading its own surrender back as something it had |
| 862 | /// said -- a turn stopped from outside became, in the record, a turn that gave up. It also |
| 863 | /// says the work may be unfinished, because a boundary that reads like a conclusion invites |
| 864 | /// the model to write one. |
| 865 | /// |
| 866 | /// # Arguments |
| 867 | /// * `max_rounds` - The limit that was reached. |
| 868 | /// Said to the model when the turn before it produced no words at all. |
| 869 | /// |
| 870 | /// A turn can finish having said nothing: the provider returns an empty final |
| 871 | /// message, and on a reasoning model the whole answer sometimes goes to a channel |
| 872 | /// the app does not print. Nothing then appears on screen, the spinner clears, |
| 873 | /// and the user cannot tell a finished turn from a hung one -- which is what was |
| 874 | /// reported against a DeepSeek model after it had read five files and stopped. |
| 875 | /// |
| 876 | /// It is said in the app's voice, for the reason [`round_limit_note`] gives: a |
| 877 | /// silence the app noticed must not read, on the next turn, as something the |
| 878 | /// assistant chose to say. |
| 879 | pub fn empty_turn_note() -> ChatMessage { |
| 880 | ChatMessage::system( |
| 881 | "[The previous turn ended without producing any text. Daimond noticed and \ |
| 882 | said so; the assistant did not choose to stop. Say what was found and \ |
| 883 | carry on from there.]".to_string()) |
| 884 | } |
| 885 | |
| 886 | pub fn round_limit_note(max_rounds: usize) -> ChatMessage { |
| 887 | ChatMessage::system(fmt!( |
| 888 | "[Daimond stopped the previous turn after {} tool-call rounds, which is its \ |
| 889 | limit. The assistant did not choose to stop and the task may be unfinished; \ |
| 890 | say where it had got to before carrying on.]", max_rounds)) |
| 891 | } |
| 892 | |
| 893 | |
| 894 | // ┌───────────────────────────────────────────────────────────────┐ |
| 895 | // │ Folding │ |
| 896 | // └───────────────────────────────────────────────────────────────┘ |
| 897 | |
| 898 | /// Replace everything before `cut` with one notice, keeping the rest exactly as it was. |
| 899 | /// |
| 900 | /// Refuses -- returning the reason rather than a conversation -- when the result would |
| 901 | /// carry more orphaned tool calls or replies than it was given. That refusal is the point: |
| 902 | /// a fold that orphans a call produces a request every provider rejects, on this turn and |
| 903 | /// on every turn after it, which is the failure this module exists to prevent. The caller |
| 904 | /// falls back to [`elide_tool_results`], which cannot orphan anything because it adds and |
| 905 | /// removes nothing. |
| 906 | /// |
| 907 | /// # Arguments |
| 908 | /// * `msgs` - The conversation, oldest first. |
| 909 | /// * `cut` - Index of the first message to keep. |
| 910 | /// * `notice` - What replaces the folded part. |
| 911 | pub fn fold(msgs: &[ChatMessage], cut: usize, notice: String) -> Outcome<Vec<ChatMessage>> { |
| 912 | if cut == 0 || cut >= msgs.len() { |
| 913 | return Err(err!( |
| 914 | "Fold: a cut at {} of {} messages folds nothing or everything.", cut, msgs.len(); |
| 915 | Invalid, Input)); |
| 916 | } |
| 917 | let mut out = Vec::with_capacity(msgs.len() - cut + 1); |
| 918 | out.push(ChatMessage::user(notice)); |
| 919 | out.extend_from_slice(&msgs[cut..]); |
| 920 | let before = orphan_count(msgs); |
| 921 | let after = orphan_count(&out); |
| 922 | if after > before { |
| 923 | return Err(err!( |
| 924 | "Fold: cutting at {} would leave {} unpaired tool calls where the conversation \ |
| 925 | had {}; a provider rejects that outright.", cut, after, before; |
| 926 | Invalid, Data)); |
| 927 | } |
| 928 | Ok(out) |
| 929 | } |
| 930 | |
| 931 | /// Shrink the oldest bulky messages in place until the conversation fits. |
| 932 | /// |
| 933 | /// The safe half of compaction, and the fallback for every case the folding half refuses: |
| 934 | /// no message is added, removed or reordered and no `tool_call_id` changes, so pairing is |
| 935 | /// untouched by construction. It is also the only thing that helps when the conversation |
| 936 | /// is one turn long and already too big -- there is no earlier part to fold. |
| 937 | /// |
| 938 | /// Two kinds of message are shrunk and two are not. A tool result and a long assistant |
| 939 | /// turn are both machine output and can be read or asked for again; the user's own words |
| 940 | /// and the system prompt are neither, and are left exactly as they are. An assistant turn |
| 941 | /// keeps its `tool_calls` untouched -- only its prose is shortened -- so the block it opens |
| 942 | /// stays answerable. |
| 943 | /// |
| 944 | /// IMAGES GO FIRST, in a pass of their own before any prose is touched. Three reasons, in the |
| 945 | /// order they matter: an image is the largest single thing a transcript can hold, so dropping one |
| 946 | /// buys more room than shortening every tool result in the conversation; it is the least |
| 947 | /// re-readable, because nothing in the text can stand in for what it showed; and it is the most |
| 948 | /// cheaply recovered, because the ledger already records which file was read and `file_read` will |
| 949 | /// fetch it again. The line left in its place names that file -- see [`ImagePart::elision`] -- |
| 950 | /// so an elided image is a pointer, not a hole. A user's own image is dropped too, unlike a |
| 951 | /// user's own words: the words cannot be recovered and the file can. |
| 952 | /// |
| 953 | /// Returns how many messages were shrunk. |
| 954 | /// |
| 955 | /// # Arguments |
| 956 | /// * `msgs` - The conversation, edited in place. |
| 957 | /// * `target_bytes` - The size to get under. |
| 958 | /// * `keep_last` - Trailing messages never touched, so the newest work stays whole. |
| 959 | /// * `open` - The folds the user has open, so what is elided is measured as it will be sent. See |
| 960 | /// [`msg_bytes`]. |
| 961 | pub fn elide_bulk( |
| 962 | msgs: &mut [ChatMessage], |
| 963 | target_bytes: u64, |
| 964 | keep_last: usize, |
| 965 | open: &OpenSet, |
| 966 | ) |
| 967 | -> usize |
| 968 | { |
| 969 | let mut total = conversation_bytes(msgs, open); |
| 970 | if total <= target_bytes { |
| 971 | return 0; |
| 972 | } |
| 973 | let last = msgs.len().saturating_sub(keep_last); |
| 974 | let mut n = 0; |
| 975 | |
| 976 | // Pass one: the images, oldest first. |
| 977 | for i in 0..last { |
| 978 | if total <= target_bytes { |
| 979 | return n; |
| 980 | } |
| 981 | if !msgs[i].content().has_image() { |
| 982 | continue; |
| 983 | } |
| 984 | let before = msg_bytes(&msgs[i], open); |
| 985 | msgs[i] = msgs[i].with_content(msgs[i].content().without_images(crate::protocol::Dropped::ToFit)); |
| 986 | total = total.saturating_sub(before.saturating_sub(msg_bytes(&msgs[i], open))); |
| 987 | n += 1; |
| 988 | } |
| 989 | |
| 990 | // Pass two: the prose, as before. |
| 991 | for i in 0..last { |
| 992 | if total <= target_bytes { |
| 993 | break; |
| 994 | } |
| 995 | let len = match &msgs[i] { |
| 996 | ChatMessage::Tool { content, .. } if content.text_len() > TOOL_ELISION_CAP => |
| 997 | content.text_len(), |
| 998 | ChatMessage::Assistant { content, .. } if content.text_len() > TOOL_ELISION_CAP => |
| 999 | content.text_len(), |
| 1000 | _ => continue, |
| 1001 | }; |
| 1002 | let shrunk = fmt!( |
| 1003 | "{}\n[the remaining {} bytes were folded away to fit the context window; read it \ |
| 1004 | again if you need it]", |
| 1005 | clip(&msgs[i].text(), TOOL_ELISION_CAP), len - TOOL_ELISION_CAP.min(len)); |
| 1006 | msgs[i] = msgs[i].with_content(MessageContent::text(shrunk.clone())); |
| 1007 | total = total.saturating_sub((len - shrunk.len().min(len)) as u64); |
| 1008 | n += 1; |
| 1009 | } |
| 1010 | n |
| 1011 | } |
| 1012 | |
| 1013 | |
| 1014 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1015 | // │ The crystal fold's gate │ |
| 1016 | // └───────────────────────────────────────────────────────────────┘ |
| 1017 | // |
| 1018 | // A different fold from the rest of this module -- the reducer's, which folds one delta |
| 1019 | // into a Diamond's memory rather than a conversation into a notice -- and it sits here |
| 1020 | // because it is the same refusal. [`fold`] will not hand back a conversation a provider |
| 1021 | // would reject; this will not hand back a crystal nothing can read. A fold is checked |
| 1022 | // BEFORE it is offered, or the checking is left to a user who has already pressed accept. |
| 1023 | // |
| 1024 | // The check earns its place at the moment the crystal stopped being markdown. Markdown |
| 1025 | // has no parse failure, so anything the reducer said was a crystal and the only sensible |
| 1026 | // gate was "is it empty". JSON does have one, and the failure it admits is total: an |
| 1027 | // accepted proposal REPLACES the file, so one stray sentence of preamble costs the user |
| 1028 | // everything the Diamond remembered. The prompt asks for bare JSON |
| 1029 | // ([`crate::prompts::CRYSTAL_SCHEMA_NOTE`]); this is what happens when the model does not |
| 1030 | // listen. |
| 1031 | |
| 1032 | /// Greatest nesting the crystal check will descend to. |
| 1033 | /// |
| 1034 | /// A crystal is an object of lists of small objects, so honest text reaches five or six |
| 1035 | /// levels and never more. The decoder's own default for text is 512, which is chosen so |
| 1036 | /// that a 2 MiB native thread stack survives the worst case -- and this runs in a browser, |
| 1037 | /// where the stack is a quarter of that. Thirty-two levels is generous for the shape and |
| 1038 | /// costs well under a tenth of what is there, so a crystal nested to provoke a crash is |
| 1039 | /// refused with a sentence rather than taking the wasm module down. |
| 1040 | const CRYSTAL_MAX_DEPTH: usize = 32; |
| 1041 | |
| 1042 | /// Greatest proposal the crystal check will parse at all. |
| 1043 | /// |
| 1044 | /// Far above [`crate::tools::CRYSTAL_CAP_DEFAULT`] on purpose: the cap is a judgement |
| 1045 | /// about what a summary should weigh and belongs at the write, whereas this is only a |
| 1046 | /// bound on how much text is worth handing a parser before refusing outright. A proposal |
| 1047 | /// this large has gone wrong in some way the cap will also catch. |
| 1048 | const CRYSTAL_MAX_BYTES: usize = 1024 * 1024; |
| 1049 | |
| 1050 | /// The reducer's raw output as a crystal, or the reason it is not one. |
| 1051 | /// |
| 1052 | /// Three refusals, in the order they cost the user. An EMPTY proposal is nothing to fold |
| 1053 | /// in. One that is not a single whole object -- truncated, or trailing prose, or a bare |
| 1054 | /// list -- carries no crystal keys the app, the page or the next reducer could name; and |
| 1055 | /// the next reducer matters most, because it is handed the crystal as its input and would |
| 1056 | /// fold the following delta into wreckage. One that is shaped right and still will not |
| 1057 | /// parse is the same loss arriving a step later. |
| 1058 | /// |
| 1059 | /// **Two checks and not one, and the structural one is not the belt.** The daticle |
| 1060 | /// decoder is what says whether the text is JSON, but its text form is forgiving where a |
| 1061 | /// browser is not: a map whose input simply stops is returned as a map of what arrived, so |
| 1062 | /// `{"title": "half` decodes happily here and is refused by `JSON.parse` -- and a reply cut |
| 1063 | /// at the model's output limit is the commonest way a fold goes wrong. It also stops |
| 1064 | /// reading at the close, so a second object after the first is neither read nor complained |
| 1065 | /// about. [`one_whole_object`] answers exactly the question the decoder does not: one |
| 1066 | /// object, opened at the first byte and closed at the last. |
| 1067 | /// |
| 1068 | /// What comes back is the model's OWN text, unfenced and trimmed, never a re-encoding of |
| 1069 | /// what was parsed. Re-encoding would sort the keys and normalise the numbers, and the |
| 1070 | /// contract turns on carrying a key through exactly as it arrived -- including one this |
| 1071 | /// build has never heard of. The parse is a question asked of the text, not a stage the |
| 1072 | /// text passes through. |
| 1073 | /// |
| 1074 | /// # Arguments |
| 1075 | /// * `raw` - Everything the reducer emitted, exactly as it emitted it. |
| 1076 | pub fn crystal_proposal(raw: &str) -> Outcome<String> { |
| 1077 | let text = unfence(raw); |
| 1078 | if text.is_empty() { |
| 1079 | return Err(err!( |
| 1080 | "The reducer returned an empty proposal, so there is nothing to fold in. A fold \ |
| 1081 | never empties a crystal; try again, or steer the Diamond instead."; |
| 1082 | Invalid, Data)); |
| 1083 | } |
| 1084 | if !one_whole_object(text) { |
| 1085 | return Err(err!( |
| 1086 | "The reducer's proposal is not one whole JSON object, so accepting it would \ |
| 1087 | replace this Diamond's crystal with something nothing can read. A crystal opens \ |
| 1088 | with a brace and closes with the matching one, and carries nothing on either \ |
| 1089 | side of it."; |
| 1090 | Invalid, Data)); |
| 1091 | } |
| 1092 | let cfg = json_cfg(); |
| 1093 | let dat = match Dat::decode_string_with_config(text, &cfg) { |
| 1094 | Ok(d) => d, |
| 1095 | Err(e) => return Err(err!(e, |
| 1096 | "The reducer's proposal is not JSON, so accepting it would replace this \ |
| 1097 | Diamond's crystal with text nothing can read."; |
| 1098 | Invalid, Data)), |
| 1099 | }; |
| 1100 | match dat { |
| 1101 | Dat::Map(_) | Dat::OrdMap(_) => Ok(text.to_string()), |
| 1102 | // Unreachable through the check above, and kept because it is the assertion that |
| 1103 | // makes the check above load-bearing rather than decorative. |
| 1104 | _ => Err(err!( |
| 1105 | "The reducer's proposal is JSON but not an object, so it carries no crystal keys \ |
| 1106 | at all. A crystal is one object: title, summary, sections, facts, open, links."; |
| 1107 | Invalid, Data)), |
| 1108 | } |
| 1109 | } |
| 1110 | |
| 1111 | /// How a crystal is read, wherever it is read: strictly, as JSON, and bounded. |
| 1112 | /// |
| 1113 | /// One function so that the two questions asked of a crystal in this module cannot come to |
| 1114 | /// disagree about what a crystal is. The decoder's JSON configuration rather than its own |
| 1115 | /// -- no comments and no trailing comma -- because being laxer here than the browser is the |
| 1116 | /// one thing this must not be: a proposal Rust waves through and `JSON.parse` then rejects |
| 1117 | /// is a crystal the user accepted and cannot open, which is worse than a refusal, since a |
| 1118 | /// refusal at least leaves the old crystal standing. |
| 1119 | /// |
| 1120 | /// `use_ordmaps` stays off, as it is by default, so a decoded object is always a |
| 1121 | /// [`Dat::Map`] and never a [`Dat::OrdMap`]. |
| 1122 | fn json_cfg() -> DecoderConfig<BTreeMap<UsrKindCode, UsrKind>, BTreeMap<String, UsrKindId>> { |
| 1123 | DecoderConfig::json(None) |
| 1124 | .with_limits(DecodeLimits::new(CRYSTAL_MAX_DEPTH, CRYSTAL_MAX_BYTES)) |
| 1125 | } |
| 1126 | |
| 1127 | /// Top-level keys that carried something in `old` and are simply not in `new`. |
| 1128 | /// |
| 1129 | /// The rule the whole crystal design turns on is that nothing may ever drop a key it does |
| 1130 | /// not recognise, and the fold is the one place that rule can be broken wholesale: an |
| 1131 | /// accepted proposal REPLACES the file, on a single click, by a user who is being invited |
| 1132 | /// to click. No parse check can catch it, because a crystal that has lost half its keys is |
| 1133 | /// as valid a document as one that has not -- `{}` itself is legal, since every core key is |
| 1134 | /// optional. Only the crystal it replaces knows what went. |
| 1135 | /// |
| 1136 | /// So this is not a refusal and must not become one. A key may leave for good reasons: the |
| 1137 | /// user asked for it to go, or the delta superseded the only thing it held. What it must |
| 1138 | /// not do is leave WITHOUT ANYONE SEEING, which is the Home Assistant failure the schema |
| 1139 | /// exists to prevent, and a form editor at least knows which fields it understands. The |
| 1140 | /// answer is handed back as the keys themselves rather than as a sentence, so the app can |
| 1141 | /// put them in front of the user in the user's own language and let them decide. |
| 1142 | /// |
| 1143 | /// **Absence only.** A key emptied in place -- `open` going from three threads to none -- |
| 1144 | /// is not reported, because closing the last open thread is exactly what a good fold does |
| 1145 | /// and flagging it would train the user to wave the warning through. That is the known |
| 1146 | /// limitation: an unknown key emptied rather than removed passes unremarked. |
| 1147 | /// |
| 1148 | /// Silent, deliberately, where it cannot be sure: either text failing to parse, or either |
| 1149 | /// one not being an object, yields nothing rather than a claim. A crystal still in its |
| 1150 | /// legacy markdown is the ordinary case of that, and the migration owns it -- reporting |
| 1151 | /// every key in the world as lost there would be noise at exactly the moment the user is |
| 1152 | /// least able to judge it. |
| 1153 | /// |
| 1154 | /// The keys come back in the decoder's own order, which is alphabetical rather than the |
| 1155 | /// order they sit in the file. Nothing is re-encoded and no crystal text is produced here: |
| 1156 | /// this reads two documents and names a difference between them. |
| 1157 | /// |
| 1158 | /// # Arguments |
| 1159 | /// * `old` - The crystal as it stands, from disk. |
| 1160 | /// * `new` - The proposal, as [`crystal_proposal`] returned it. |
| 1161 | pub fn crystal_keys_lost(old: &str, new: &str) -> Vec<String> { |
| 1162 | let cfg = json_cfg(); |
| 1163 | let (before, after) = match ( |
| 1164 | Dat::decode_string_with_config(unfence(old), &cfg), |
| 1165 | Dat::decode_string_with_config(unfence(new), &cfg), |
| 1166 | ) { |
| 1167 | (Ok(b), Ok(a)) => (b, a), |
| 1168 | _ => return Vec::new(), |
| 1169 | }; |
| 1170 | let (before, after) = match (before, after) { |
| 1171 | (Dat::Map(b), Dat::Map(a)) => (b, a), |
| 1172 | _ => return Vec::new(), |
| 1173 | }; |
| 1174 | let mut lost = Vec::new(); |
| 1175 | for (key, val) in &before { |
| 1176 | let name = match key { |
| 1177 | Dat::Str(s) => s, |
| 1178 | // A crystal's keys are strings. Anything else is not a key a page or a form |
| 1179 | // could name, so there is nothing to tell the user about it. |
| 1180 | _ => continue, |
| 1181 | }; |
| 1182 | if !carries_content(val) { |
| 1183 | continue; |
| 1184 | } |
| 1185 | if !after.contains_key(key) { |
| 1186 | lost.push(name.clone()); |
| 1187 | } |
| 1188 | } |
| 1189 | lost |
| 1190 | } |
| 1191 | |
| 1192 | /// Whether a value holds anything a user would miss. |
| 1193 | /// |
| 1194 | /// The point is not tidiness, it is noise: a crystal often carries a key standing empty -- |
| 1195 | /// `open` with no threads left, a `summary` not yet written -- and reporting one of those as |
| 1196 | /// lost when the reducer drops it would be a warning about nothing, which is how a warning |
| 1197 | /// stops being read. Anything this build does not recognise counts as content, because the |
| 1198 | /// keys most worth protecting are exactly the ones it has never heard of. |
| 1199 | /// |
| 1200 | /// # Arguments |
| 1201 | /// * `d` - A top-level value from a crystal. |
| 1202 | fn carries_content(d: &Dat) -> bool { |
| 1203 | match d { |
| 1204 | Dat::Empty => false, |
| 1205 | Dat::Str(s) => !s.trim().is_empty(), |
| 1206 | Dat::List(v) => !v.is_empty(), |
| 1207 | Dat::Map(m) => !m.is_empty(), |
| 1208 | // JSON `null`, which the decoder reads as an absent option. |
| 1209 | Dat::Opt(o) => matches!(**o, Some(_)), |
| 1210 | _ => true, |
| 1211 | } |
| 1212 | } |
| 1213 | |
| 1214 | /// Whether `s` is one JSON object and nothing else: it opens with `{`, every brace, bracket |
| 1215 | /// and quote closes, and the closing brace is the last byte. |
| 1216 | /// |
| 1217 | /// A near relation of [`crate::agent::json_object_is_whole`] and deliberately not a call to |
| 1218 | /// it, because the two answer different questions. That one asks whether a tool call's |
| 1219 | /// arguments arrived whole and is right to ignore whatever follows the close -- a provider |
| 1220 | /// may append anything and the call is still dispatchable. A crystal is a FILE: text after |
| 1221 | /// the close is written into it and makes it unreadable, so "and nothing else" is half of |
| 1222 | /// what is being asked here. |
| 1223 | /// |
| 1224 | /// String contents are skipped, so a brace inside a body of markdown is not counted as |
| 1225 | /// structure, and a string the reply stopped inside runs to the end of the input and is |
| 1226 | /// reported as the truncation it is. |
| 1227 | /// |
| 1228 | /// # Arguments |
| 1229 | /// * `s` - The proposal, already unfenced and trimmed. |
| 1230 | fn one_whole_object(s: &str) -> bool { |
| 1231 | let b = s.as_bytes(); |
| 1232 | if b.first() != Some(&b'{') { |
| 1233 | return false; |
| 1234 | } |
| 1235 | let mut depth = 0i32; |
| 1236 | let mut i = 0usize; |
| 1237 | // The last byte that mattered, outside any string. It is here for one job: a comma |
| 1238 | // immediately before a closer. `DecoderConfig::json` tolerates a trailing comma and |
| 1239 | // `JSON.parse` refuses one, and being laxer than the browser that reads the file is the |
| 1240 | // one thing this gate must never be -- a crystal accepted here and refused there is |
| 1241 | // written to disk and then unreadable, which is worse than a fold that was never offered. |
| 1242 | let mut last = 0u8; |
| 1243 | while i < b.len() { |
| 1244 | match b[i] { |
| 1245 | b'{' | b'[' => depth += 1, |
| 1246 | b'}' | b']' => { |
| 1247 | if last == b',' { |
| 1248 | return false; |
| 1249 | } |
| 1250 | depth -= 1; |
| 1251 | if depth == 0 { |
| 1252 | return i + 1 == b.len(); |
| 1253 | } |
| 1254 | if depth < 0 { |
| 1255 | return false; |
| 1256 | } |
| 1257 | }, |
| 1258 | b'"' => { |
| 1259 | i += 1; |
| 1260 | while i < b.len() { |
| 1261 | if b[i] == b'\\' { |
| 1262 | i += 2; |
| 1263 | continue; |
| 1264 | } |
| 1265 | if b[i] == b'"' { |
| 1266 | break; |
| 1267 | } |
| 1268 | i += 1; |
| 1269 | } |
| 1270 | if i >= b.len() { |
| 1271 | // The reply stopped inside a string. |
| 1272 | return false; |
| 1273 | } |
| 1274 | }, |
| 1275 | _ => {}, |
| 1276 | } |
| 1277 | // Whitespace between a comma and its closer must not hide the comma, so it is the |
| 1278 | // last SIGNIFICANT byte that is remembered. After a string the cursor sits on the |
| 1279 | // closing quote, which is significant and is what lands here. |
| 1280 | if !b[i].is_ascii_whitespace() { |
| 1281 | last = b[i]; |
| 1282 | } |
| 1283 | i += 1; |
| 1284 | } |
| 1285 | false |
| 1286 | } |
| 1287 | |
| 1288 | /// `raw` trimmed, with one markdown code fence taken off it if it is wearing one. |
| 1289 | /// |
| 1290 | /// Told and stripped both, deliberately. A prompt is a request, and this is the request a |
| 1291 | /// model ignores most reliably; the cost of losing it here is the whole crystal, so the |
| 1292 | /// cheap defence is taken as well as asked for. |
| 1293 | /// |
| 1294 | /// It strips a fence the text OPENS with, and nothing else. Prose wrapped around the |
| 1295 | /// object is a reducer that has ignored a plain instruction, and digging the JSON out of it |
| 1296 | /// would make that invisible -- the fold would quietly work, the prompt would stay |
| 1297 | /// unheeded, and nobody would learn. A fence is different: it is punctuation the model |
| 1298 | /// adds without meaning anything by it. |
| 1299 | /// |
| 1300 | /// # Arguments |
| 1301 | /// * `raw` - Everything the reducer emitted. |
| 1302 | fn unfence(raw: &str) -> &str { |
| 1303 | let t = raw.trim(); |
| 1304 | if !t.starts_with("```") { |
| 1305 | return t; |
| 1306 | } |
| 1307 | // The opening line carries the fence and whatever language tag rides on it, and a fence |
| 1308 | // with no newline after it is a fence and nothing else. |
| 1309 | let body = match t.find('\n') { |
| 1310 | Some(i) => t[i + 1..].trim_end(), |
| 1311 | None => return "", |
| 1312 | }; |
| 1313 | match body.strip_suffix("```") { |
| 1314 | Some(b) => b.trim(), |
| 1315 | None => body.trim(), |
| 1316 | } |
| 1317 | } |
| 1318 | |
| 1319 | |
| 1320 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1321 | // │ Recognising the failure │ |
| 1322 | // └───────────────────────────────────────────────────────────────┘ |
| 1323 | |
| 1324 | /// Whether a failed round looks like the context window being exceeded. |
| 1325 | /// |
| 1326 | /// Two ways of telling, and the first is now the ordinary one. The provider's own words |
| 1327 | /// are decisive, and BOTH transports carry them: the native one always did, and the |
| 1328 | /// browser one does since `LlmClient::body_detail` -- until then it could report no more |
| 1329 | /// than `LLM: HTTP error: 400 Bad Request.`, and the size test below was the only thing |
| 1330 | /// standing between a browser chat and permanent death by overflow. |
| 1331 | /// |
| 1332 | /// The size test is therefore a FALLBACK rather than the browser's whole answer. It |
| 1333 | /// still earns its place: a provider can refuse with an empty body, a CDN can answer 413 |
| 1334 | /// with an HTML page of its own that says nothing about tokens, and a body can fail to |
| 1335 | /// read at all. In each of those the status plus a prompt big enough to explain it is |
| 1336 | /// all there is. Both halves are needed even then -- acting on the status alone would |
| 1337 | /// fold the conversation every time a key was mistyped. |
| 1338 | /// |
| 1339 | /// # Arguments |
| 1340 | /// * `err` - The error text from the failed call. |
| 1341 | /// * `prompt_tokens` - The estimated size of the prompt that was refused. |
| 1342 | /// * `budget` - The turn's token budget. |
| 1343 | pub fn looks_like_overflow(err: &str, prompt_tokens: u64, budget: u64) -> bool { |
| 1344 | let low = err.to_lowercase(); |
| 1345 | for m in [ |
| 1346 | "context length", |
| 1347 | "context_length", |
| 1348 | "maximum context", |
| 1349 | "context window", |
| 1350 | "too many tokens", |
| 1351 | "prompt is too long", |
| 1352 | "reduce the length", |
| 1353 | "input length", |
| 1354 | "request too large", |
| 1355 | "payload too large", |
| 1356 | ] { |
| 1357 | if low.contains(m) { |
| 1358 | return true; |
| 1359 | } |
| 1360 | } |
| 1361 | // The fallback: no words that say so, so the status and the size are all there is. |
| 1362 | let refused = low.contains("400") || low.contains("413") || low.contains("422"); |
| 1363 | refused && prompt_tokens >= OVERFLOW_FLOOR_TOKENS.min(budget / 2) |
| 1364 | } |
| 1365 | |
| 1366 | |
| 1367 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1368 | // │ The prompt the fold runs under │ |
| 1369 | // └───────────────────────────────────────────────────────────────┘ |
| 1370 | // |
| 1371 | // It lives in `crate::prompts` as [`crate::prompts::Role::Compactor`], not here. A prompt |
| 1372 | // held as a private constant beside the code that sends it is a prompt the user cannot |
| 1373 | // read, and this was the last one in the app: as a `Role` it is backed by |
| 1374 | // `prompts/compactor.md` like every other, so it can be read, edited and put back by |
| 1375 | // deleting the file. [`crate::agent::Agent::fold_prompt`] composes it. |
| 1376 | |
| 1377 | |
| 1378 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1379 | // │ Tests │ |
| 1380 | // └───────────────────────────────────────────────────────────────┘ |
| 1381 | |
| 1382 | #[cfg(test)] |
| 1383 | mod tests { |
| 1384 | |
| 1385 | #[test] |
| 1386 | fn test_the_shipped_fold_fraction_is_the_owners_figure_00() { |
| 1387 | // Pinned as a FIGURE, not as "whatever the constant says". The whole point of moving it |
| 1388 | // was that 0.8 arrived too late to be told apart from a fault, and a test written as |
| 1389 | // `assert_eq!(FOLD_AT, Limits::default().fold_at)` would have passed on 0.8, on 0.65 and |
| 1390 | // on anything a later hand put there. |
| 1391 | assert_eq!(0.65, FOLD_AT); |
| 1392 | assert_eq!(FOLD_AT, Limits::default().fold_at); |
| 1393 | // Two thirds of the window and not four fifths of it, measured where it is spent: a |
| 1394 | // 100,000 window with nothing reserved for the reply. |
| 1395 | let mut l = Limits::default(); |
| 1396 | l.window = 100_000; |
| 1397 | assert_eq!(65_000, l.budget(0)); |
| 1398 | } |
| 1399 | |
| 1400 | #[test] |
| 1401 | fn test_the_fold_band_is_the_one_the_budget_uses_00() { |
| 1402 | // The clamp was written twice -- once as literals inside `budget` and once wherever a |
| 1403 | // caller decided what to offer -- and a band held in two places is a band that drifts. |
| 1404 | let mut l = Limits::default(); |
| 1405 | l.window = 100_000; |
| 1406 | l.fold_at = 9.0; |
| 1407 | assert_eq!((100_000.0 * FOLD_AT_MAX) as u64, l.budget(0)); |
| 1408 | l.fold_at = 0.0; |
| 1409 | assert_eq!((100_000.0 * FOLD_AT_MIN) as u64, l.budget(0)); |
| 1410 | } |
| 1411 | use super::*; |
| 1412 | |
| 1413 | /// No fold open, which is what every size in these tests is measured against unless the test |
| 1414 | /// is about the fold state itself. |
| 1415 | fn shut() -> OpenSet { OpenSet::new() } |
| 1416 | |
| 1417 | // The tool layer as the app runs it. The ledger's tests take their tool replies from the |
| 1418 | // dispatcher rather than composing any, so what they assert about is what a user's |
| 1419 | // conversation would actually contain. |
| 1420 | use crate::executor::Executor; |
| 1421 | use crate::tools::{ |
| 1422 | FileRoot, |
| 1423 | PACK_DROP01, |
| 1424 | Tool, |
| 1425 | ToolContext, |
| 1426 | ToolRegistry, |
| 1427 | chat_bounds, |
| 1428 | new_read_cache, |
| 1429 | set_locked_packs, |
| 1430 | }; |
| 1431 | use crate::workspace::Workspace; |
| 1432 | |
| 1433 | /// An assistant turn asking for one tool call. |
| 1434 | fn asks(id: &str, name: &str, args: &str) -> ChatMessage { |
| 1435 | ChatMessage::Assistant { |
| 1436 | content: MessageContent::text(""), |
| 1437 | tool_calls: vec![ToolCall { |
| 1438 | id: id.to_string(), name: name.to_string(), arguments: args.to_string(), |
| 1439 | }], |
| 1440 | } |
| 1441 | } |
| 1442 | |
| 1443 | /// The reply to one. |
| 1444 | fn replies(id: &str, body: &str) -> ChatMessage { |
| 1445 | ChatMessage::tool(id.to_string(), body.to_string()) |
| 1446 | } |
| 1447 | |
| 1448 | fn user(s: &str) -> ChatMessage { ChatMessage::user(s.to_string()) } |
| 1449 | fn says(s: &str) -> ChatMessage { |
| 1450 | ChatMessage::assistant(s.to_string()) |
| 1451 | } |
| 1452 | |
| 1453 | // ── Images ─────────────────────────────────────────────────────────────── |
| 1454 | |
| 1455 | /// One of this repository's own screenshots, read off disk, carrying the path a |
| 1456 | /// conversation would have read it from. |
| 1457 | /// |
| 1458 | /// A real file rather than a synthesised one, because the numbers in [`image_tokens`]'s |
| 1459 | /// documentation were measured against real screenshots: a test built on a fabricated |
| 1460 | /// header would confirm the arithmetic and say nothing about whether the arithmetic |
| 1461 | /// describes a screenshot. |
| 1462 | /// |
| 1463 | /// THE BYTES AND THE LABEL ARE NOT THE SAME THING, and separating them is the whole of |
| 1464 | /// this. The label is each test's own fiction -- `shots/mobile-desktop-after.png` is a |
| 1465 | /// path a model asked `file_read` for, and three tests below assert that the elision |
| 1466 | /// notice names it -- so it need not be a file that exists. The BYTES must be real, and |
| 1467 | /// must therefore come from somewhere nothing rewrites. |
| 1468 | /// |
| 1469 | /// They used to be one file, and `shots/` is neither committed (`.gitignore` line 52) |
| 1470 | /// nor stable (`dev/verify_mobile.mjs` rewrites it on every mobile sweep). So five |
| 1471 | /// tests here and three in [`crate::tools`] could pass only in a tree that happened to |
| 1472 | /// hold four untracked screenshots: not in a clone, and not in `dev/gate.sh`'s own |
| 1473 | /// worktree, where they were eight standing reds that every brief had learnt to excuse. |
| 1474 | /// `src/testdata/` is committed, crosses into the public mirror with the rest of `src/`, |
| 1475 | /// and is written by no sweep and no harness -- see [`owned_shot`], which settled this |
| 1476 | /// for one test and was never carried to the rest. |
| 1477 | /// |
| 1478 | /// # Arguments |
| 1479 | /// * `name` - The path the conversation names. Each maps to the committed fixture of |
| 1480 | /// the same dimensions, since it is the dimensions the arithmetic is about. |
| 1481 | fn shot(name: &str) -> ImagePart { |
| 1482 | let file = match name { |
| 1483 | "mobile-desktop-after.png" => "screenshot-1500x950.png", |
| 1484 | "mobile-sheet-web.png" => "screenshot-390x844.png", |
| 1485 | other => panic!("no committed fixture stands in for {}", other), |
| 1486 | }; |
| 1487 | let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) |
| 1488 | .join("src").join("testdata").join(file); |
| 1489 | let data = std::fs::read(&path) |
| 1490 | .unwrap_or_else(|e| panic!("the fixture {} must be readable: {}", path.display(), e)); |
| 1491 | let media = crate::protocol::ImageMedia::sniff(&data).expect("the fixture must be an image"); |
| 1492 | ImagePart::new(media, data, fmt!("shots/{}", name)) |
| 1493 | } |
| 1494 | |
| 1495 | /// The screenshot this module owns, read off disk. |
| 1496 | /// |
| 1497 | /// [`shot`] USED TO READ `shots/`, which is a SWEPT directory: `dev/verify_mobile.mjs` |
| 1498 | /// rewrites those files on every mobile sweep, and `.gitignore` excludes them, so they are |
| 1499 | /// neither stable across a run nor present in a clone at all. On 2026-08-12 the sweep re-took |
| 1500 | /// `mobile-desktop-after.png`, it compressed to half the size, and the overstatement in |
| 1501 | /// [`test_an_image_is_not_priced_as_if_it_were_text`] fell from 29x to 15.3x. |
| 1502 | /// |
| 1503 | /// Nothing about pricing had changed. [`image_tokens`] still reads the header and still |
| 1504 | /// never looks at `data.len()`, and the test beside this one still pins 54x34 against the |
| 1505 | /// published formula. What moved was the EVIDENCE: that assertion is a check that its own |
| 1506 | /// fixture still demonstrates the problem, which is the opposite of a check that cannot |
| 1507 | /// fail, and it is worth keeping exactly as it is. Restoring the screenshot would only |
| 1508 | /// wait for the next sweep, and relaxing the multiplier would trade the property away to |
| 1509 | /// accommodate an accident. |
| 1510 | /// |
| 1511 | /// So the test owns its bytes. `src/testdata/` is committed, crosses into the public |
| 1512 | /// mirror with the rest of `src/`, and is written by nothing: no sweep, no shot script, no |
| 1513 | /// harness. Re-taking a screenshot cannot reach it, and a cloner has it. |
| 1514 | /// |
| 1515 | /// It is still a REAL screenshot of this app, for the reason given on [`shot`] -- 390x844, |
| 1516 | /// the phone viewport, as `dev/verify_mobile.mjs` took it once. A fabricated header would |
| 1517 | /// confirm the arithmetic and say nothing about whether the arithmetic describes a |
| 1518 | /// screenshot. |
| 1519 | fn owned_shot() -> ImagePart { |
| 1520 | let rel = "src/testdata/screenshot-390x844.png"; |
| 1521 | let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join(rel); |
| 1522 | let data = std::fs::read(&path) |
| 1523 | .unwrap_or_else(|e| panic!("the fixture {} must be readable: {}", path.display(), e)); |
| 1524 | let media = crate::protocol::ImageMedia::sniff(&data).expect("the fixture must be an image"); |
| 1525 | ImagePart::new(media, data, rel.to_string()) |
| 1526 | } |
| 1527 | |
| 1528 | /// A tool reply carrying a screenshot, as `file_read` produces one. |
| 1529 | fn replies_with_image(id: &str, name: &str) -> ChatMessage { |
| 1530 | ChatMessage::tool(id.to_string(), MessageContent::parts(vec![ |
| 1531 | crate::protocol::ContentPart::Text(fmt!("Read the image shots/{}.", name)), |
| 1532 | crate::protocol::ContentPart::Image(shot(name)), |
| 1533 | ])) |
| 1534 | } |
| 1535 | |
| 1536 | /// An image is charged by its area, and the figures in the documentation are the figures a |
| 1537 | /// real screenshot produces. |
| 1538 | /// |
| 1539 | /// The formula is Anthropic's published one -- ceil(w/28) x ceil(h/28) visual tokens -- so |
| 1540 | /// this is checking our arithmetic against a provider's rule and two files on disk, not |
| 1541 | /// against another function of ours. |
| 1542 | #[test] |
| 1543 | fn test_a_screenshot_costs_what_the_provider_charges_for_its_area() { |
| 1544 | let big = shot("mobile-desktop-after.png"); |
| 1545 | assert_eq!(Some((1500, 950)), big.dims(), "the fixture is not the size the docs assume"); |
| 1546 | // ceil(1500/28) = 54 columns, ceil(950/28) = 34 rows. |
| 1547 | assert_eq!(54 * 34, image_tokens(&big)); |
| 1548 | |
| 1549 | let small = shot("mobile-sheet-web.png"); |
| 1550 | assert_eq!(Some((390, 844)), small.dims()); |
| 1551 | assert_eq!(14 * 31, image_tokens(&small)); |
| 1552 | } |
| 1553 | |
| 1554 | /// The catastrophe this whole change exists to prevent: a screenshot priced as if it were |
| 1555 | /// text. |
| 1556 | /// |
| 1557 | /// 62,456 bytes at the text ratio of 0.27 is 16,863 tokens, on a file the provider charges |
| 1558 | /// 434 for -- a thirty-nine-fold overstatement, and this is a PHONE screenshot. The one |
| 1559 | /// that provoked the change was a desktop capture of 199,280 bytes: 53,806 tokens against |
| 1560 | /// 1,836 charged, most of a 131k budget. Priced that way, the conversation folds on the |
| 1561 | /// turn the screenshot is read and on every turn after it. |
| 1562 | /// |
| 1563 | /// The fixture is [`owned_shot`] and not [`shot`] on purpose; the reason is written there, |
| 1564 | /// and it is that the first assertion below is a check that the fixture still demonstrates |
| 1565 | /// the problem, so the fixture must be something nothing else regenerates. |
| 1566 | #[test] |
| 1567 | fn test_an_image_is_not_priced_as_if_it_were_text() { |
| 1568 | let img = owned_shot(); |
| 1569 | let as_text = (img.data.len() as f64 * DEFAULT_TOKENS_PER_BYTE) as u64; |
| 1570 | let actual = image_tokens(&img); |
| 1571 | assert!(as_text > actual * 20, |
| 1572 | "the fixture no longer demonstrates the overstatement: {} vs {}", as_text, actual); |
| 1573 | |
| 1574 | // What the module actually counts. Bounded against `actual` rather than against a |
| 1575 | // figure written down here, so the bound stays as tight for a fixture of any size as |
| 1576 | // it was for the one it was written against: an image charged by its BYTES would be |
| 1577 | // the thirty-nine-fold figure above, not twice. |
| 1578 | let msg = ChatMessage::user(MessageContent::parts(vec![ |
| 1579 | crate::protocol::ContentPart::Image(img.clone()), |
| 1580 | ])); |
| 1581 | let charged = msg_bytes(&msg, &shut()); |
| 1582 | let gauge = Gauge::default(); |
| 1583 | assert!(gauge.tokens(charged) < actual * 2, |
| 1584 | "a screenshot is being charged {} tokens against the {} the provider will", |
| 1585 | gauge.tokens(charged), actual); |
| 1586 | assert!(gauge.tokens(charged) >= actual, |
| 1587 | "a screenshot is being charged less than the provider will"); |
| 1588 | |
| 1589 | // And it does not, on its own, trip the fold. |
| 1590 | let limits = Limits { window: 131_072, ..Limits::default() }; |
| 1591 | assert!(charged < limits.budget(4_096) / 4, |
| 1592 | "one screenshot took a quarter of the budget: {} of {}", |
| 1593 | charged, limits.budget(4_096)); |
| 1594 | } |
| 1595 | |
| 1596 | /// A format whose header this build cannot read is estimated from its bytes and bounded by the |
| 1597 | /// same ceiling, rather than falling through to the text ratio. |
| 1598 | #[test] |
| 1599 | fn test_an_unreadable_header_falls_back_to_a_bounded_estimate() { |
| 1600 | // A RIFF/WEBP wrapper with nothing decodable inside it. |
| 1601 | let mut data = b"RIFF\x00\x00\x00\x00WEBPVP8 ".to_vec(); |
| 1602 | data.resize(300_000, 0x5A); |
| 1603 | let img = ImagePart::new(crate::protocol::ImageMedia::WebP, data, "big.webp".to_string()); |
| 1604 | assert_eq!(None, img.dims(), "this build should not claim to read a WebP header"); |
| 1605 | assert_eq!(IMAGE_TOKEN_CAP, image_tokens(&img), "the fallback must be capped"); |
| 1606 | |
| 1607 | let small = ImagePart::new( |
| 1608 | crate::protocol::ImageMedia::WebP, |
| 1609 | b"RIFF\x00\x00\x00\x00WEBPVP8 ".to_vec(), |
| 1610 | "tiny.webp".to_string()); |
| 1611 | assert!(image_tokens(&small) >= 1, "an image never costs nothing"); |
| 1612 | assert!(image_tokens(&small) < 100, "a tiny file must not be charged the ceiling"); |
| 1613 | } |
| 1614 | |
| 1615 | /// Elision drops the images before it shortens any prose. |
| 1616 | /// |
| 1617 | /// Ordered that way because an image is the largest thing in the transcript and the least |
| 1618 | /// re-readable; the assertion is that the prose is still whole once the images have gone. |
| 1619 | #[test] |
| 1620 | fn test_elision_drops_images_before_it_touches_prose() { |
| 1621 | let mut v = vec![ |
| 1622 | user("look at these"), |
| 1623 | asks("call_0", "file_read", r#"{"path":"shots/mobile-desktop-after.png"}"#), |
| 1624 | replies_with_image("call_0", "mobile-desktop-after.png"), |
| 1625 | asks("call_1", "file_read", r#"{"path":"src/main.rs"}"#), |
| 1626 | replies("call_1", &"z".repeat(4_000)), |
| 1627 | says("done"), |
| 1628 | ]; |
| 1629 | let before = conversation_bytes(&v, &shut()); |
| 1630 | // A target that the images alone can meet. |
| 1631 | let target = before - msg_bytes(&v[2], &shut()) / 2; |
| 1632 | let n = elide_bulk(&mut v, target, 1, &shut()); |
| 1633 | assert_eq!(1, n, "exactly the image message should have been touched"); |
| 1634 | assert!(!v[2].content().has_image(), "the image survived"); |
| 1635 | assert_eq!(4_000, v[4].content().text_len(), "the prose was shortened before it had to be"); |
| 1636 | } |
| 1637 | |
| 1638 | /// An elided image leaves the file's name behind, so the model can read it again. |
| 1639 | #[test] |
| 1640 | fn test_an_elided_image_says_which_file_it_was() { |
| 1641 | let mut v = vec![ |
| 1642 | user("look"), |
| 1643 | asks("call_0", "file_read", r#"{"path":"shots/mobile-desktop-after.png"}"#), |
| 1644 | replies_with_image("call_0", "mobile-desktop-after.png"), |
| 1645 | says("done"), |
| 1646 | ]; |
| 1647 | elide_bulk(&mut v, 100, 1, &shut()); |
| 1648 | let left = v[2].text(); |
| 1649 | assert!(!v[2].content().has_image(), "the image should have gone"); |
| 1650 | assert!(left.contains("shots/mobile-desktop-after.png"), |
| 1651 | "the elision must name the file: {}", left); |
| 1652 | assert!(left.contains("read it again"), "it must say what to do: {}", left); |
| 1653 | // The pairing the provider checks is untouched. |
| 1654 | assert_eq!(0, orphan_count(&v)); |
| 1655 | assert!(matches!(v[2], ChatMessage::Tool { .. }), "the role changed"); |
| 1656 | } |
| 1657 | |
| 1658 | /// Elision still gets a conversation under the target when the images alone are not enough. |
| 1659 | #[test] |
| 1660 | fn test_images_then_prose_reaches_the_target() { |
| 1661 | let mut v = vec![user("go")]; |
| 1662 | for r in 0..4 { |
| 1663 | let id = fmt!("call_{}", r); |
| 1664 | v.push(asks(&id, "file_read", r#"{"path":"x"}"#)); |
| 1665 | v.push(if r % 2 == 0 { |
| 1666 | replies_with_image(&id, "mobile-sheet-web.png") |
| 1667 | } else { |
| 1668 | replies(&id, &"q".repeat(8_000)) |
| 1669 | }); |
| 1670 | } |
| 1671 | v.push(says("done")); |
| 1672 | let target = 4_000; |
| 1673 | elide_bulk(&mut v, target, 1, &shut()); |
| 1674 | assert!(conversation_bytes(&v, &shut()) <= target + 2_000, |
| 1675 | "elision left {} bytes against a target of {}", conversation_bytes(&v, &shut()), target); |
| 1676 | assert!(!v.iter().any(|m| m.content().has_image()), "an image survived"); |
| 1677 | assert_eq!(0, orphan_count(&v)); |
| 1678 | } |
| 1679 | |
| 1680 | /// The fold's own input cap counts an image at its token cost too, so a folded conversation |
| 1681 | /// carrying screenshots does not produce a summarising call that is itself refused. |
| 1682 | #[test] |
| 1683 | fn test_the_fold_input_cap_measures_an_image_by_its_tokens() { |
| 1684 | let v = vec![ |
| 1685 | user("go"), |
| 1686 | asks("call_0", "file_read", r#"{"path":"shots/mobile-desktop-after.png"}"#), |
| 1687 | replies_with_image("call_0", "mobile-desktop-after.png"), |
| 1688 | ]; |
| 1689 | assert!(conversation_bytes(&v, &shut()) < FOLD_INPUT_CAP, |
| 1690 | "one screenshot exceeded the whole fold input cap: {} of {}", |
| 1691 | conversation_bytes(&v, &shut()), FOLD_INPUT_CAP); |
| 1692 | // And what the summariser is shown names the file rather than carrying it. |
| 1693 | let r = render_for_fold(&v, FOLD_INPUT_CAP); |
| 1694 | assert!(r.contains("shots/mobile-desktop-after.png"), "{}", r); |
| 1695 | assert!(!r.contains("iVBORw0KGgo"), "base64 reached the summarising call: {}", r); |
| 1696 | } |
| 1697 | |
| 1698 | /// A conversation of `rounds` complete tool blocks, each carrying a fat result. |
| 1699 | fn session(rounds: usize, result_bytes: usize) -> Vec<ChatMessage> { |
| 1700 | let mut v = vec![user("please refactor the parser")]; |
| 1701 | for r in 0..rounds { |
| 1702 | let id = fmt!("call_{}", r); |
| 1703 | v.push(asks(&id, "file_read", &fmt!("{{\"path\":\"src/f{}.rs\"}}", r))); |
| 1704 | v.push(replies(&id, &"x".repeat(result_bytes))); |
| 1705 | } |
| 1706 | v.push(says("done")); |
| 1707 | v |
| 1708 | } |
| 1709 | |
| 1710 | // ── Pairing: the rule that breaks a compactor ──────────────────────────── |
| 1711 | |
| 1712 | #[test] |
| 1713 | fn test_a_whole_conversation_has_no_orphans_00() { |
| 1714 | assert_eq!(orphan_count(&session(3, 10)), 0); |
| 1715 | assert!(pairing_is_whole(&session(3, 10))); |
| 1716 | } |
| 1717 | |
| 1718 | #[test] |
| 1719 | fn test_a_reply_whose_call_was_folded_away_is_an_orphan_00() { |
| 1720 | // The shape a careless cut produces, and the one a provider rejects outright. |
| 1721 | let v = session(2, 10); |
| 1722 | let orphaned = &v[2..]; // starts at the tool reply, not the assistant turn |
| 1723 | assert!(matches!(orphaned[0], ChatMessage::Tool { .. })); |
| 1724 | assert_eq!(orphan_count(orphaned), 1); |
| 1725 | assert!(!pairing_is_whole(orphaned)); |
| 1726 | } |
| 1727 | |
| 1728 | #[test] |
| 1729 | fn test_a_call_whose_reply_was_dropped_is_an_orphan_00() { |
| 1730 | let mut v = session(1, 10); |
| 1731 | v.remove(2); // the reply |
| 1732 | assert_eq!(orphan_count(&v), 1); |
| 1733 | } |
| 1734 | |
| 1735 | #[test] |
| 1736 | fn test_two_calls_in_one_turn_need_two_replies_00() { |
| 1737 | let both = ChatMessage::Assistant { |
| 1738 | content: MessageContent::text(""), |
| 1739 | tool_calls: vec![ |
| 1740 | ToolCall { id: fmt!("a"), name: fmt!("file_read"), arguments: fmt!("{{}}") }, |
| 1741 | ToolCall { id: fmt!("b"), name: fmt!("file_list"), arguments: fmt!("{{}}") }, |
| 1742 | ], |
| 1743 | }; |
| 1744 | let whole = vec![user("go"), both.clone(), replies("a", "1"), replies("b", "2")]; |
| 1745 | assert_eq!(orphan_count(&whole), 0); |
| 1746 | let half = vec![user("go"), both, replies("a", "1")]; |
| 1747 | assert_eq!(orphan_count(&half), 1); |
| 1748 | } |
| 1749 | |
| 1750 | #[test] |
| 1751 | fn test_replies_out_of_order_do_not_count_as_paired_00() { |
| 1752 | let both = ChatMessage::Assistant { |
| 1753 | content: MessageContent::text(""), |
| 1754 | tool_calls: vec![ |
| 1755 | ToolCall { id: fmt!("a"), name: fmt!("file_read"), arguments: fmt!("{{}}") }, |
| 1756 | ToolCall { id: fmt!("b"), name: fmt!("file_list"), arguments: fmt!("{{}}") }, |
| 1757 | ], |
| 1758 | }; |
| 1759 | let swapped = vec![user("go"), both, replies("b", "2"), replies("a", "1")]; |
| 1760 | assert!(orphan_count(&swapped) > 0); |
| 1761 | } |
| 1762 | |
| 1763 | // ── The cut ────────────────────────────────────────────────────────────── |
| 1764 | |
| 1765 | #[test] |
| 1766 | fn test_the_cut_never_lands_inside_a_tool_block_00() { |
| 1767 | // Every budget, on a conversation made entirely of blocks. The cut must never be a |
| 1768 | // tool reply, whatever the arithmetic wanted. |
| 1769 | let v = session(12, 500); |
| 1770 | for keep in (0..conversation_bytes(&v, &shut())).step_by(37) { |
| 1771 | let cut = tail_start(&v, keep, MIN_KEEP_MESSAGES, u64::MAX, &shut()); |
| 1772 | assert!(!matches!(v.get(cut), Some(ChatMessage::Tool { .. })), |
| 1773 | "a cut at {} opens the tail with a reply to a call that was folded away", cut); |
| 1774 | } |
| 1775 | } |
| 1776 | |
| 1777 | #[test] |
| 1778 | fn test_folding_at_any_cut_leaves_a_whole_conversation_00() { |
| 1779 | // The property that actually matters, checked across the whole range rather than at |
| 1780 | // one convenient point. |
| 1781 | let v = session(12, 500); |
| 1782 | for keep in (0..conversation_bytes(&v, &shut())).step_by(37) { |
| 1783 | let cut = tail_start(&v, keep, MIN_KEEP_MESSAGES, u64::MAX, &shut()); |
| 1784 | if cut == 0 { |
| 1785 | continue; |
| 1786 | } |
| 1787 | let out = match fold(&v, cut, fmt!("folded")) { |
| 1788 | Ok(o) => o, |
| 1789 | Err(e) => panic!("a fold at {} was refused: {}", cut, e), |
| 1790 | }; |
| 1791 | assert!(pairing_is_whole(&out), "cut {} orphaned something", cut); |
| 1792 | } |
| 1793 | } |
| 1794 | |
| 1795 | #[test] |
| 1796 | fn test_a_cut_that_would_orphan_a_call_is_refused_00() { |
| 1797 | // Constructed by hand, because `tail_start` will not produce one: the fold itself |
| 1798 | // must refuse it, so that no future caller computing its own cut can reintroduce the |
| 1799 | // bug this module exists to remove. |
| 1800 | let v = session(3, 100); |
| 1801 | let bad = v.iter().position(|m| matches!(m, ChatMessage::Tool { .. })) |
| 1802 | .expect("a tool reply"); |
| 1803 | assert!(bad > 0); |
| 1804 | let r = fold(&v, bad, fmt!("folded")); |
| 1805 | assert!(r.is_err(), "a cut opening on a tool reply must be refused, not returned"); |
| 1806 | let msg = match r { Ok(_) => fmt!(""), Err(e) => fmt!("{}", e) }; |
| 1807 | assert!(msg.contains("unpaired"), "{}", msg); |
| 1808 | } |
| 1809 | |
| 1810 | #[test] |
| 1811 | fn test_a_conversation_already_broken_can_still_be_folded_00() { |
| 1812 | // A session read back from storage loses the tool_calls off its assistant turns, so |
| 1813 | // it arrives with orphans through no fault of the fold. Refusing to fold it would |
| 1814 | // leave it with no way out of an overflow at all; what the fold must not do is add |
| 1815 | // orphans of its own. |
| 1816 | let mut v = session(4, 100); |
| 1817 | for m in v.iter_mut() { |
| 1818 | if let ChatMessage::Assistant { tool_calls, .. } = m { |
| 1819 | tool_calls.clear(); |
| 1820 | } |
| 1821 | } |
| 1822 | let before = orphan_count(&v); |
| 1823 | assert!(before > 0, "the fixture is meant to arrive broken"); |
| 1824 | let out = match fold(&v, 3, fmt!("folded")) { |
| 1825 | Ok(o) => o, |
| 1826 | Err(e) => panic!("a fold on an already-broken conversation was refused: {}", e), |
| 1827 | }; |
| 1828 | assert!(orphan_count(&out) <= before); |
| 1829 | } |
| 1830 | |
| 1831 | #[test] |
| 1832 | fn test_the_newest_messages_are_always_kept_00() { |
| 1833 | let v = session(20, 800); |
| 1834 | let cut = tail_start(&v, 1_000, MIN_KEEP_MESSAGES, u64::MAX, &shut()); |
| 1835 | assert!(v.len() - cut >= MIN_KEEP_MESSAGES, |
| 1836 | "only {} messages kept; a model just told the answer must not lose it", |
| 1837 | v.len() - cut); |
| 1838 | // And the very last message is always in the tail. |
| 1839 | assert_eq!(v[v.len() - 1], v[v.len() - 1]); |
| 1840 | assert!(cut < v.len()); |
| 1841 | } |
| 1842 | |
| 1843 | #[test] |
| 1844 | fn test_a_huge_recent_message_does_not_defeat_the_fold_00() { |
| 1845 | // Found by a mock provider with a real context ceiling, not by reasoning: two |
| 1846 | // twenty-kilobyte replies among the six messages `min_keep` asks for made a tail |
| 1847 | // bigger than the whole budget, so the fold shrank the conversation by two messages |
| 1848 | // and it was refused again. A preference for six messages cannot outrank the size |
| 1849 | // that may actually be sent. |
| 1850 | let mut v = vec![user("go")]; |
| 1851 | for i in 0..8 { |
| 1852 | v.push(user(&fmt!("more {}", i))); |
| 1853 | v.push(says(&"z".repeat(20_000))); |
| 1854 | } |
| 1855 | let ceiling = 8_000u64; |
| 1856 | let cut = tail_start(&v, ceiling, MIN_KEEP_MESSAGES, ceiling, &shut()); |
| 1857 | let kept: u64 = v[cut..].iter().map(|m| msg_bytes(m, &shut())).sum(); |
| 1858 | assert!(kept <= ceiling + 20_100, |
| 1859 | "the tail is {} bytes against a ceiling of {}", kept, ceiling); |
| 1860 | assert!(v.len() - cut < MIN_KEEP_MESSAGES, |
| 1861 | "six messages were kept anyway, which is the bug"); |
| 1862 | assert!(cut < v.len(), "the message being answered is always kept"); |
| 1863 | } |
| 1864 | |
| 1865 | #[test] |
| 1866 | fn test_a_long_assistant_turn_can_be_shortened_too_00() { |
| 1867 | // A tool result is not the only bulky thing in a conversation. Machine output can be |
| 1868 | // shortened; the user's own words cannot. |
| 1869 | let mut v = vec![ |
| 1870 | user(&"u".repeat(5_000)), |
| 1871 | says(&"a".repeat(5_000)), |
| 1872 | user("what now?"), |
| 1873 | ]; |
| 1874 | let n = elide_bulk(&mut v, 3_000, 1, &shut()); |
| 1875 | assert_eq!(n, 1, "the assistant's own turn should have been shortened"); |
| 1876 | assert!(v[1].content().text_len() < 1_000, "{}", v[1].content().text_len()); |
| 1877 | assert_eq!(v[0].content().text_len(), 5_000, "the user's own words were shortened"); |
| 1878 | } |
| 1879 | |
| 1880 | #[test] |
| 1881 | fn test_shortening_an_assistant_turn_keeps_its_tool_calls_00() { |
| 1882 | // Shrinking the prose must not orphan the block the turn opens. |
| 1883 | let mut v = vec![ |
| 1884 | user("go"), |
| 1885 | ChatMessage::Assistant { |
| 1886 | content: MessageContent::text("z".repeat(5_000)), |
| 1887 | tool_calls: vec![ToolCall { |
| 1888 | id: fmt!("a"), name: fmt!("file_read"), arguments: fmt!("{{}}") }], |
| 1889 | }, |
| 1890 | replies("a", "1"), |
| 1891 | user("next"), |
| 1892 | ]; |
| 1893 | elide_bulk(&mut v, 500, 1, &shut()); |
| 1894 | assert_eq!(orphan_count(&v), 0, "the block lost its call"); |
| 1895 | assert!(v[1].content().text_len() < 1_000); |
| 1896 | } |
| 1897 | |
| 1898 | #[test] |
| 1899 | fn test_a_short_conversation_is_not_folded_00() { |
| 1900 | let v = vec![user("hello"), says("hi")]; |
| 1901 | assert_eq!(tail_start(&v, 0, MIN_KEEP_MESSAGES, u64::MAX, &shut()), 0); |
| 1902 | assert!(fold(&v, 0, fmt!("x")).is_err(), "a fold of nothing is not a fold"); |
| 1903 | // Not even under a ceiling it cannot meet. There is nothing worth folding in three |
| 1904 | // messages, and replacing one of them with a note explaining the fold is pure loss. |
| 1905 | let three = vec![user("hello"), says(&"x".repeat(50_000)), user("and?")]; |
| 1906 | assert_eq!(tail_start(&three, 100, MIN_KEEP_MESSAGES, 100, &shut()), 0); |
| 1907 | } |
| 1908 | |
| 1909 | // ── The ledger ─────────────────────────────────────────────────────────── |
| 1910 | |
| 1911 | #[test] |
| 1912 | fn test_the_ledger_keeps_which_files_were_read_and_written_00() { |
| 1913 | let v = vec![ |
| 1914 | user("go"), |
| 1915 | asks("a", "file_read", r#"{"path":"src/lib.rs"}"#), |
| 1916 | replies("a", "fn main() {}"), |
| 1917 | asks("b", "file_write", r#"{"path":"src/new.rs","content":"x"}"#), |
| 1918 | replies("b", "Wrote 1 byte."), |
| 1919 | asks("c", "run", r#"{"argv":["cargo","test"]}"#), |
| 1920 | replies("c", "ok"), |
| 1921 | ]; |
| 1922 | let l = ledger_of(&v); |
| 1923 | assert_eq!(l.read, vec![fmt!("src/lib.rs")]); |
| 1924 | assert_eq!(l.wrote, vec![fmt!("src/new.rs")]); |
| 1925 | assert_eq!(l.ran, vec![fmt!("cargo test")]); |
| 1926 | assert!(l.failed.is_empty()); |
| 1927 | } |
| 1928 | |
| 1929 | // ── A call that did not happen ─────────────────────────────────────────── |
| 1930 | // |
| 1931 | // Every reply below is taken from `ToolRegistry::dispatch`, which is the only route by which |
| 1932 | // a tool result reaches a conversation. Nothing here writes a refusal by hand, and that is |
| 1933 | // the whole lesson of the defect these replace: the test that shipped fed the ledger the |
| 1934 | // sentence "Error: path is outside the workspace.", which no part of the product has ever |
| 1935 | // emitted. It agreed with itself, stayed green from the day it was written, and every |
| 1936 | // refused write in the build was folded in as a write underneath it. |
| 1937 | |
| 1938 | /// A scratch workspace bounded as a chat's is: `notes` to work in, `book` to consult. |
| 1939 | /// |
| 1940 | /// A REAL directory, because one case below has to be a write that actually lands. A fixture |
| 1941 | /// in which nothing can succeed would be satisfied by a classifier that called every call |
| 1942 | /// refused, which would empty the ledger of the one thing it exists to carry. |
| 1943 | fn chat_ctx() -> ToolContext { |
| 1944 | let dir = match oxedyne_fe2o3_test::scratch::scratch_dir("daimond_compact_test") { |
| 1945 | Ok(d) => d, |
| 1946 | Err(e) => panic!("a scratch directory: {}", e), |
| 1947 | }; |
| 1948 | let ws = match Workspace::new(dir) { |
| 1949 | Ok(w) => w, |
| 1950 | Err(e) => panic!("a workspace: {}", e), |
| 1951 | }; |
| 1952 | ToolContext { |
| 1953 | workspace: ws, |
| 1954 | executor: Executor::local_default(), |
| 1955 | cwd: String::new(), |
| 1956 | path_prefix: String::new(), |
| 1957 | root: FileRoot::Workspace, |
| 1958 | read_seen: new_read_cache(), |
| 1959 | no_write: chat_bounds("chats/c1", &[fmt!("notes")], &[fmt!("book")]), |
| 1960 | daimon_of: String::new(), |
| 1961 | } |
| 1962 | } |
| 1963 | |
| 1964 | /// What the app itself puts in the conversation for one call. |
| 1965 | async fn dispatched(tool: Tool, args: &str, ctx: &ToolContext) -> String { |
| 1966 | let reg = ToolRegistry::new(vec![tool], ctx.clone()); |
| 1967 | let out = reg.dispatch(tool.name(), args).await; |
| 1968 | out.as_text().to_string() |
| 1969 | } |
| 1970 | |
| 1971 | /// A conversation of one call and the reply the product gave it. |
| 1972 | fn one_call(tool: Tool, args: &str, reply: &str) -> Vec<ChatMessage> { |
| 1973 | vec![user("go"), asks("a", tool.name(), args), replies("a", reply)] |
| 1974 | } |
| 1975 | |
| 1976 | #[tokio::test] |
| 1977 | async fn test_a_refused_write_is_never_reported_as_a_write_00() { |
| 1978 | // The failure that matters most: a fold claiming a file was written, when the write was |
| 1979 | // refused, leaves the model describing work that was never done -- and the notice ends by |
| 1980 | // telling it to trust exactly that list. |
| 1981 | let ctx = chat_ctx(); |
| 1982 | // One case per door a write meets, each phrased by a different function in `tools`, and |
| 1983 | // not one of them phrased here. |
| 1984 | let cases: Vec<(&str, &str)> = vec![ |
| 1985 | ("the scope fence", r#"{"path":"secrets/keys.txt","content":"x"}"#), |
| 1986 | ("the read-only mark", r#"{"path":"book/ch1.md","content":"x"}"#), |
| 1987 | ("an absolute path", r#"{"path":"/etc/passwd","content":"x"}"#), |
| 1988 | ]; |
| 1989 | for (door, args) in cases { |
| 1990 | let reply = dispatched(Tool::FileWrite, args, &ctx).await; |
| 1991 | let l = ledger_of(&one_call(Tool::FileWrite, args, &reply)); |
| 1992 | assert!(l.wrote.is_empty(), |
| 1993 | "{} booked a refused write as a write: {:?} -- the reply was: {}", |
| 1994 | door, l.wrote, reply); |
| 1995 | assert_eq!(l.refused.len(), 1, |
| 1996 | "{} was not recorded as a refusal: {:?} -- the reply was: {}", door, l, reply); |
| 1997 | assert!(l.failed.is_empty(), |
| 1998 | "{} was recorded as a broken tool rather than a closed door: {:?}", door, l); |
| 1999 | // And the notice must not name the file under what was written, which is the part of |
| 2000 | // it the model is instructed to trust. |
| 2001 | let n = notice(9, "", &l, None); |
| 2002 | assert!(!n.contains("Files written"), "{}: {}", door, n); |
| 2003 | } |
| 2004 | } |
| 2005 | |
| 2006 | #[tokio::test] |
| 2007 | async fn test_a_refusal_that_opens_with_the_tools_own_name_is_still_a_refusal_00() { |
| 2008 | // The locked-pack refusal is the case a list of openings kept in this module would have |
| 2009 | // missed: its author wrote it opening with the tool's name and no marker at all. It is |
| 2010 | // recognisable because `Tool::guard` composes every answer it gives with |
| 2011 | // `tools::refusal_line` -- which is the fix, stated as a test. |
| 2012 | let ctx = chat_ctx(); |
| 2013 | let args = r#"{"path":"paper.typ"}"#; |
| 2014 | set_locked_packs(PACK_DROP01); |
| 2015 | let reply = dispatched(Tool::TypstCompile, args, &ctx).await; |
| 2016 | set_locked_packs(""); // A global left changed decides whether the next test passes. |
| 2017 | assert!(reply.contains(PACK_DROP01), "not the pack refusal: {}", reply); |
| 2018 | |
| 2019 | let l = ledger_of(&one_call(Tool::TypstCompile, args, &reply)); |
| 2020 | assert!(l.wrote.is_empty(), |
| 2021 | "a compile that never ran was booked as a write: {:?} -- the reply was: {}", |
| 2022 | l.wrote, reply); |
| 2023 | assert_eq!(l.refused.len(), 1, "{:?} -- the reply was: {}", l, reply); |
| 2024 | } |
| 2025 | |
| 2026 | #[tokio::test] |
| 2027 | async fn test_a_tool_that_broke_is_held_apart_from_one_that_was_refused_00() { |
| 2028 | // A GENUINE error, wrapped by the dispatcher exactly as it wraps every other: `run` |
| 2029 | // reaches the machine hand, which the native build does not have. The two must not be |
| 2030 | // merged -- a model told a closed door is a broken tool retries it, and a model told a |
| 2031 | // broken tool is a closed door goes looking for permission it already has. |
| 2032 | let ctx = chat_ctx(); |
| 2033 | let args = r#"{"argv":["cargo","test"]}"#; |
| 2034 | let reply = dispatched(Tool::Run, args, &ctx).await; |
| 2035 | let l = ledger_of(&one_call(Tool::Run, args, &reply)); |
| 2036 | assert!(l.ran.is_empty(), |
| 2037 | "a command that never ran was booked as run: {:?} -- the reply was: {}", l.ran, reply); |
| 2038 | assert!(l.refused.is_empty(), |
| 2039 | "an error was booked as a refusal: {:?} -- the reply was: {}", l.refused, reply); |
| 2040 | assert_eq!(l.failed.len(), 1, "{:?} -- the reply was: {}", l, reply); |
| 2041 | } |
| 2042 | |
| 2043 | #[tokio::test] |
| 2044 | async fn test_a_write_that_landed_is_still_reported_as_a_write_00() { |
| 2045 | // The control every test above needs. Three refusals proved nothing on their own: a |
| 2046 | // classifier that answered "refused" to everything would satisfy all three, and would |
| 2047 | // hand the model a fold that mentions none of the work it actually did. |
| 2048 | let ctx = chat_ctx(); |
| 2049 | let args = r#"{"path":"notes/new.rs","content":"x"}"#; |
| 2050 | let reply = dispatched(Tool::FileWrite, args, &ctx).await; |
| 2051 | let l = ledger_of(&one_call(Tool::FileWrite, args, &reply)); |
| 2052 | assert_eq!(l.wrote, vec![fmt!("notes/new.rs")], "the reply was: {}", reply); |
| 2053 | assert!(l.refused.is_empty() && l.failed.is_empty(), |
| 2054 | "a write that landed was booked as something else: {:?}", l); |
| 2055 | } |
| 2056 | |
| 2057 | #[tokio::test] |
| 2058 | async fn test_a_refused_call_is_named_in_the_notice_as_one_that_did_nothing_00() { |
| 2059 | // What the model is finally shown. The columns are not interchangeable: the line has to |
| 2060 | // say the call touched nothing, or a fold reporting a refusal is a fold reporting work. |
| 2061 | // |
| 2062 | // THIS TEST BUILT ITS LEDGER BY HAND AND THEREFORE PROVED ONLY THE RENDERING. Restoring |
| 2063 | // the shipped defect -- judging a call by whether its reply opens with "Error" -- left it |
| 2064 | // green, because a hand-made ledger never meets the classifier at all. Its name promises |
| 2065 | // something about a refused CALL, so it now starts from one: a real refusal, from the |
| 2066 | // dispatcher, through `ledger_of`, and only then to the notice. Classification and |
| 2067 | // rendering are one claim here, and a break in either must reach it. |
| 2068 | let ctx = chat_ctx(); |
| 2069 | let args = r#"{"path":"secrets/keys.txt","content":"x"}"#; |
| 2070 | let reply = dispatched(Tool::FileWrite, args, &ctx).await; |
| 2071 | let l = ledger_of(&one_call(Tool::FileWrite, args, &reply)); |
| 2072 | let n = notice(9, "", &l, None); |
| 2073 | assert!(n.contains("secrets/keys.txt"), "the refusal was not named at all: {}", n); |
| 2074 | assert!(!n.contains("Files written"), |
| 2075 | "a refusal must not be listed as a write: {} -- the reply was: {}", n, reply); |
| 2076 | assert!(n.contains("nothing was touched"), |
| 2077 | "the notice did not say the call touched nothing: {}", n); |
| 2078 | } |
| 2079 | |
| 2080 | #[test] |
| 2081 | fn test_the_same_file_read_twice_is_listed_once_00() { |
| 2082 | let v = vec![ |
| 2083 | user("go"), |
| 2084 | asks("a", "file_read", r#"{"path":"a.rs"}"#), replies("a", "1"), |
| 2085 | asks("b", "file_read", r#"{"path":"a.rs"}"#), replies("b", "1"), |
| 2086 | ]; |
| 2087 | assert_eq!(ledger_of(&v).read, vec![fmt!("a.rs")]); |
| 2088 | } |
| 2089 | |
| 2090 | #[test] |
| 2091 | fn test_a_move_records_both_ends_00() { |
| 2092 | let v = vec![ |
| 2093 | user("go"), |
| 2094 | asks("a", "file_move", r#"{"path":"old.rs","to":"new.rs"}"#), |
| 2095 | replies("a", "Moved."), |
| 2096 | ]; |
| 2097 | assert_eq!(ledger_of(&v).wrote, vec![fmt!("old.rs -> new.rs")]); |
| 2098 | } |
| 2099 | |
| 2100 | #[test] |
| 2101 | fn test_the_notice_carries_the_ledger_even_with_no_summary_00() { |
| 2102 | // What a failed summarising call leaves behind. It must still be a truthful record, |
| 2103 | // and it must say that the prose is missing rather than quietly omitting it. |
| 2104 | let l = ledger_of(&vec![ |
| 2105 | user("go"), |
| 2106 | asks("a", "file_write", r#"{"path":"src/new.rs"}"#), |
| 2107 | replies("a", "Wrote."), |
| 2108 | ]); |
| 2109 | let n = notice(9, "", &l, Some("the key was refused")); |
| 2110 | assert!(n.contains("src/new.rs"), "{}", n); |
| 2111 | assert!(n.contains("No summary could be written"), "{}", n); |
| 2112 | assert!(n.contains("the key was refused"), "{}", n); |
| 2113 | assert!(n.contains("9 messages"), "{}", n); |
| 2114 | } |
| 2115 | |
| 2116 | #[test] |
| 2117 | fn test_the_notice_is_not_in_the_assistants_voice_00() { |
| 2118 | // It is a user message on purpose. An assistant-voiced summary is the model reading |
| 2119 | // invented memories back as things it said. |
| 2120 | let out = match fold(&session(4, 100), 3, notice(3, "did things", &Ledger::default(), None)) { |
| 2121 | Ok(o) => o, |
| 2122 | Err(e) => panic!("{}", e), |
| 2123 | }; |
| 2124 | assert_eq!(out[0].role(), "user"); |
| 2125 | assert!(out[0].text().contains("Daimond folded")); |
| 2126 | } |
| 2127 | |
| 2128 | // ── The round limit ────────────────────────────────────────────────────── |
| 2129 | |
| 2130 | #[test] |
| 2131 | fn test_the_round_limit_is_not_recorded_in_the_assistants_voice_00() { |
| 2132 | // The bug this replaces: an assistant message reading "[Reached the tool-call round |
| 2133 | // limit (25).]", which the next turn read back as its own words. |
| 2134 | let n = round_limit_note(150); |
| 2135 | assert_eq!(n.role(), "system", |
| 2136 | "a limit the app imposed must not be said in the model's own voice"); |
| 2137 | assert!(n.text().contains("Daimond stopped"), "{}", n.text()); |
| 2138 | assert!(n.text().contains("did not choose to stop"), "{}", n.text()); |
| 2139 | assert!(n.text().contains("may be unfinished"), "{}", n.text()); |
| 2140 | assert!(n.text().contains("150"), "{}", n.text()); |
| 2141 | } |
| 2142 | |
| 2143 | #[test] |
| 2144 | fn test_a_turn_is_allowed_a_real_tasks_worth_of_rounds_00() { |
| 2145 | // Find the wiring, read three files, change one, build, fix two errors, run the |
| 2146 | // tests: comfortably thirty to fifty calls. Twenty-five stopped that in the middle. |
| 2147 | assert!(DEFAULT_MAX_ROUNDS >= 100, "{} rounds", DEFAULT_MAX_ROUNDS); |
| 2148 | assert_eq!(Limits::default().max_rounds, DEFAULT_MAX_ROUNDS); |
| 2149 | } |
| 2150 | |
| 2151 | // ── Rendering ──────────────────────────────────────────────────────────── |
| 2152 | |
| 2153 | #[test] |
| 2154 | fn test_what_is_handed_to_the_summariser_is_bounded_00() { |
| 2155 | // The fold must not fail the way the request that triggered it failed. Whatever the |
| 2156 | // history's size, the summarising call's input is capped. |
| 2157 | for rounds in [4usize, 40, 400] { |
| 2158 | let v = session(rounds, 4_000); |
| 2159 | let r = render_for_fold(&v, FOLD_INPUT_CAP); |
| 2160 | assert!(r.len() as u64 <= FOLD_INPUT_CAP + 200, |
| 2161 | "{} rounds rendered to {} bytes", rounds, r.len()); |
| 2162 | } |
| 2163 | } |
| 2164 | |
| 2165 | #[test] |
| 2166 | fn test_the_opening_of_the_conversation_survives_rendering_00() { |
| 2167 | // What the user asked for is said once, at the start, and nothing later restates it. |
| 2168 | let mut v = session(200, 2_000); |
| 2169 | v[0] = user("SENTINEL: port the parser to the new lexer"); |
| 2170 | let r = render_for_fold(&v, FOLD_INPUT_CAP); |
| 2171 | assert!(r.contains("SENTINEL"), "the original request was dropped from the rendering"); |
| 2172 | assert!(r.contains("too long to include"), "the gap must be declared"); |
| 2173 | } |
| 2174 | |
| 2175 | // ── Eliding ────────────────────────────────────────────────────────────── |
| 2176 | |
| 2177 | #[test] |
| 2178 | fn test_eliding_shrinks_without_touching_pairing_00() { |
| 2179 | let mut v = session(10, 5_000); |
| 2180 | let before = conversation_bytes(&v, &shut()); |
| 2181 | let n = elide_bulk(&mut v, before / 4, 4, &shut()); |
| 2182 | assert!(n > 0); |
| 2183 | assert!(conversation_bytes(&v, &shut()) < before); |
| 2184 | assert_eq!(orphan_count(&v), 0, "eliding must not add or remove a message"); |
| 2185 | assert_eq!(v.len(), session(10, 5_000).len()); |
| 2186 | } |
| 2187 | |
| 2188 | #[test] |
| 2189 | fn test_eliding_leaves_the_newest_results_whole_00() { |
| 2190 | let mut v = session(10, 5_000); |
| 2191 | elide_bulk(&mut v, 100, 4, &shut()); |
| 2192 | let last_tool = v.iter().rposition(|m| matches!(m, ChatMessage::Tool { .. })) |
| 2193 | .expect("a reply"); |
| 2194 | assert_eq!(v[last_tool].content().text_len(), 5_000, |
| 2195 | "the newest result was elided; that is the one the model is working from"); |
| 2196 | } |
| 2197 | |
| 2198 | #[test] |
| 2199 | fn test_eliding_says_what_it_took_00() { |
| 2200 | let mut v = session(3, 5_000); |
| 2201 | elide_bulk(&mut v, 100, 0, &shut()); |
| 2202 | let t = v.iter().find(|m| matches!(m, ChatMessage::Tool { .. })).expect("a reply"); |
| 2203 | assert!(t.text().contains("folded away"), "{}", t.text()); |
| 2204 | assert!(t.text().contains("read it again"), "{}", t.text()); |
| 2205 | } |
| 2206 | |
| 2207 | // ── The budget ─────────────────────────────────────────────────────────── |
| 2208 | |
| 2209 | #[test] |
| 2210 | fn test_an_unknown_window_still_yields_a_budget_00() { |
| 2211 | let l = Limits::default(); |
| 2212 | assert_eq!(l.window, 0); |
| 2213 | assert!(l.budget(4_096) > MIN_BUDGET_TOKENS); |
| 2214 | assert!(l.budget(4_096) < DEFAULT_WINDOW); |
| 2215 | } |
| 2216 | |
| 2217 | #[test] |
| 2218 | fn test_the_reply_is_subtracted_from_a_small_window_00() { |
| 2219 | // On an 8k window with a 4k reply cap, eighty per cent of the window is not a legal |
| 2220 | // prompt: the reply has to fit too. |
| 2221 | let l = Limits { window: 8_192, ..Limits::default() }; |
| 2222 | assert!(l.budget(4_096) <= 8_192 - 4_096, |
| 2223 | "budget {} leaves no room for the reply", l.budget(4_096)); |
| 2224 | } |
| 2225 | |
| 2226 | #[test] |
| 2227 | fn test_the_budget_tracks_the_window_00() { |
| 2228 | let small = Limits { window: 32_768, ..Limits::default() }; |
| 2229 | let big = Limits { window: 1_048_576, ..Limits::default() }; |
| 2230 | assert!(big.budget(8_192) > small.budget(8_192)); |
| 2231 | } |
| 2232 | |
| 2233 | // ── The gauge ──────────────────────────────────────────────────────────── |
| 2234 | |
| 2235 | #[test] |
| 2236 | fn test_the_gauge_learns_from_what_the_provider_charged_00() { |
| 2237 | let g = Gauge::default(); |
| 2238 | assert_eq!(g.ratio(), DEFAULT_TOKENS_PER_BYTE); |
| 2239 | // A provider that charged 1 token per 2 bytes -- dense code, say. |
| 2240 | g.observe(20_000, 10_000); |
| 2241 | assert!((g.ratio() - 0.5).abs() < 1e-9, "{}", g.ratio()); |
| 2242 | assert_eq!(g.tokens(1_000), 500); |
| 2243 | } |
| 2244 | |
| 2245 | #[test] |
| 2246 | fn test_a_provider_that_reports_nothing_leaves_the_gauge_alone_00() { |
| 2247 | let g = Gauge::default(); |
| 2248 | g.observe(20_000, 0); |
| 2249 | assert_eq!(g.ratio(), DEFAULT_TOKENS_PER_BYTE); |
| 2250 | } |
| 2251 | |
| 2252 | #[test] |
| 2253 | fn test_an_absurd_ratio_is_held_inside_the_band_00() { |
| 2254 | let g = Gauge::default(); |
| 2255 | g.observe(10, 10_000); |
| 2256 | assert!(g.ratio() <= MAX_TOKENS_PER_BYTE); |
| 2257 | g.observe(10_000, 1); |
| 2258 | assert!(g.ratio() >= MIN_TOKENS_PER_BYTE); |
| 2259 | } |
| 2260 | |
| 2261 | // ── Recognising the failure ────────────────────────────────────────────── |
| 2262 | |
| 2263 | #[test] |
| 2264 | fn test_a_providers_own_words_are_enough_00() { |
| 2265 | for msg in [ |
| 2266 | "LLM: HTTP error: 400 | {\"error\":{\"message\":\"This model's maximum context \ |
| 2267 | length is 131072 tokens, however you requested 140000\"}}", |
| 2268 | "input length exceeds the model limit", |
| 2269 | "Request too large for model", |
| 2270 | ] { |
| 2271 | assert!(looks_like_overflow(msg, 0, 100_000), "{}", msg); |
| 2272 | } |
| 2273 | } |
| 2274 | |
| 2275 | #[test] |
| 2276 | fn test_both_dialects_say_it_in_words_the_test_knows_00() { |
| 2277 | // The browser transport now carries the refusal body, so the words are available |
| 2278 | // wherever the app runs -- and the two wire dialects word it differently. Each of |
| 2279 | // these is a real refusal shape, in the form the browser now reports it. |
| 2280 | for msg in [ |
| 2281 | // OpenAI-compatible. |
| 2282 | "LLM: HTTP error: 400 Bad Request | {\"error\":{\"message\":\"This model's \ |
| 2283 | maximum context length is 131072 tokens, however you requested 174233 \ |
| 2284 | tokens\",\"code\":\"context_length_exceeded\"}}", |
| 2285 | // Anthropic's Messages API, which says it two ways. |
| 2286 | "LLM: HTTP error: 400 Bad Request | {\"type\":\"error\",\"error\":{\"type\":\ |
| 2287 | \"invalid_request_error\",\"message\":\"prompt is too long: 213412 tokens > \ |
| 2288 | 200000 maximum\"}}", |
| 2289 | "LLM: HTTP error: 400 Bad Request | {\"type\":\"error\",\"error\":{\"type\":\ |
| 2290 | \"invalid_request_error\",\"message\":\"input length and `max_tokens` exceed \ |
| 2291 | context limit: 198000 + 8192 > 200000\"}}", |
| 2292 | ] { |
| 2293 | // Recognised on the words alone: the prompt size is given as zero, so |
| 2294 | // nothing here can be passing on the fallback below. |
| 2295 | assert!(looks_like_overflow(msg, 0, 100_000), "{}", msg); |
| 2296 | } |
| 2297 | } |
| 2298 | |
| 2299 | #[test] |
| 2300 | fn test_the_words_decide_even_when_the_prompt_looks_small_00() { |
| 2301 | // The case the size fallback cannot reach: a model with a window nobody |
| 2302 | // published, refusing a prompt that is small against the ASSUMED budget. Before |
| 2303 | // the browser carried the body this was indistinguishable from a mistyped |
| 2304 | // request unless the prompt cleared an absolute floor. |
| 2305 | let body = "LLM: HTTP error: 400 Bad Request | {\"error\":{\"message\":\"This \ |
| 2306 | model's maximum context length is 8192 tokens\"}}"; |
| 2307 | assert!(looks_like_overflow(body, 100, Limits::default().budget(4_096)), |
| 2308 | "the provider said so and it was not believed"); |
| 2309 | } |
| 2310 | |
| 2311 | #[test] |
| 2312 | fn test_a_refusal_with_no_words_still_falls_back_to_the_size_00() { |
| 2313 | // The fallback earns its place: a provider can refuse with an empty body, a CDN |
| 2314 | // can answer 413 with a page of its own that says nothing about tokens, and a |
| 2315 | // body can fail to read at all. Then the status and the size are all there is. |
| 2316 | for bare in [ |
| 2317 | "LLM: HTTP error: 400 Bad Request | ", |
| 2318 | "LLM: HTTP error: 413 Request Entity Too Large | <html><head><title>413 \ |
| 2319 | Request Entity Too Large</title></head><body><center><h1>413 Request \ |
| 2320 | Entity Too Large</h1></center><hr><center>nginx</center></body></html>", |
| 2321 | ] { |
| 2322 | assert!(looks_like_overflow(bare, 90_000, 100_000), "{}", bare); |
| 2323 | assert!(!looks_like_overflow(bare, 500, 100_000), |
| 2324 | "a small prompt refused with no explanation is a bad request: {}", bare); |
| 2325 | } |
| 2326 | } |
| 2327 | |
| 2328 | #[test] |
| 2329 | fn test_a_bare_status_needs_a_big_prompt_behind_it_00() { |
| 2330 | // What the browser transport reports, which carries no body at all. A 400 with a |
| 2331 | // small prompt is a mistyped request, not an overflow, and folding on it would throw |
| 2332 | // away the user's history for nothing. |
| 2333 | let bare = "LLM: HTTP error: 400 Bad Request."; |
| 2334 | assert!(!looks_like_overflow(bare, 500, 100_000)); |
| 2335 | assert!(looks_like_overflow(bare, 90_000, 100_000)); |
| 2336 | } |
| 2337 | |
| 2338 | #[test] |
| 2339 | fn test_a_window_smaller_than_the_one_assumed_is_still_recognised_00() { |
| 2340 | // The case that decides the whole reactive path. A model with a 16k window that |
| 2341 | // nobody published is being treated as 131k, so its refusal arrives with a prompt |
| 2342 | // that looks small against the assumed budget. Judged as a fraction of that budget |
| 2343 | // it would be dismissed as a malformed request, and the chat would die exactly as |
| 2344 | // it did before. |
| 2345 | // THE GUARD IS A RATIO, NOT A FIGURE. It was `assumed > 90_000`, which is the budget |
| 2346 | // FOLD_AT 0.8 produced and nothing else -- so moving the fold fraction to 0.65 reddened a |
| 2347 | // test about a refusal. What it actually needs is that the refused prompt looks small |
| 2348 | // against the assumed budget, which is what a fraction-of-budget test would dismiss it on. |
| 2349 | let refused = 12_000u64; |
| 2350 | let assumed = Limits::default().budget(4_096); |
| 2351 | assert!(assumed > refused * 5, "the assumed budget is {}", assumed); |
| 2352 | let bare = "LLM: HTTP error: 400 Bad Request."; |
| 2353 | assert!(looks_like_overflow(bare, refused, assumed), |
| 2354 | "a 12k-token prompt refused by a 16k-window model was not recognised"); |
| 2355 | } |
| 2356 | |
| 2357 | #[test] |
| 2358 | fn test_a_refusal_teaches_the_window_and_only_downwards_00() { |
| 2359 | let mut l = Limits::default(); |
| 2360 | assert_eq!(l.window, 0); |
| 2361 | assert!(l.learn_from_refusal(16_000), "a refusal at 16k tokens says something"); |
| 2362 | assert_eq!(l.window, 12_000); |
| 2363 | // A later, larger refusal must not undo it: what was learned is a ceiling. |
| 2364 | assert!(!l.learn_from_refusal(100_000)); |
| 2365 | assert_eq!(l.window, 12_000); |
| 2366 | // And it never learns its way down to nothing. |
| 2367 | let mut tiny = Limits::default(); |
| 2368 | tiny.learn_from_refusal(10); |
| 2369 | assert_eq!(tiny.window, MIN_LEARNED_WINDOW); |
| 2370 | } |
| 2371 | |
| 2372 | #[test] |
| 2373 | fn test_what_a_refusal_teaches_makes_the_next_prompt_smaller_00() { |
| 2374 | // The property the recovery rests on: after learning, the budget is BELOW the |
| 2375 | // prompt that was refused. A fold that targeted the old budget would change nothing |
| 2376 | // and the retry would be refused again. |
| 2377 | let mut l = Limits::default(); |
| 2378 | let refused = 16_000; |
| 2379 | l.learn_from_refusal(refused); |
| 2380 | assert!(l.budget(4_096) < refused, |
| 2381 | "budget {} is not below the {} tokens the provider refused", l.budget(4_096), refused); |
| 2382 | } |
| 2383 | |
| 2384 | #[test] |
| 2385 | fn test_an_ordinary_failure_is_not_an_overflow_00() { |
| 2386 | for msg in [ |
| 2387 | "LLM: HTTP error: 401 Unauthorized.", |
| 2388 | "LLM: fetch failed: NetworkError.", |
| 2389 | "LLM: HTTP error: 429 Too Many Requests.", |
| 2390 | "LLM: HTTP error: 500 Internal Server Error.", |
| 2391 | ] { |
| 2392 | assert!(!looks_like_overflow(msg, 200_000, 100_000), "{}", msg); |
| 2393 | } |
| 2394 | } |
| 2395 | |
| 2396 | // ── End to end, on the conversation the bug actually killed ────────────── |
| 2397 | |
| 2398 | #[test] |
| 2399 | fn test_a_session_too_big_for_its_window_folds_to_something_that_fits_00() { |
| 2400 | // Forty rounds of six-kilobyte file reads: a quarter of a megabyte of history, which |
| 2401 | // is what an afternoon's work looks like and what used to make a chat unusable |
| 2402 | // forever. |
| 2403 | let v = session(40, 6_000); |
| 2404 | let limits = Limits { window: 32_768, ..Limits::default() }; |
| 2405 | let gauge = Gauge::default(); |
| 2406 | let budget = limits.budget(4_096); |
| 2407 | assert!(gauge.tokens(conversation_bytes(&v, &shut())) > budget, "the fixture must be too big"); |
| 2408 | |
| 2409 | let keep = gauge.bytes(limits.tail_budget(4_096)); |
| 2410 | let cut = tail_start(&v, keep, MIN_KEEP_MESSAGES, u64::MAX, &shut()); |
| 2411 | assert!(cut > 0); |
| 2412 | let l = ledger_of(&v[..cut]); |
| 2413 | let mut out = match fold(&v, cut, notice(cut, "read forty files", &l, None)) { |
| 2414 | Ok(o) => o, |
| 2415 | Err(e) => panic!("{}", e), |
| 2416 | }; |
| 2417 | elide_bulk(&mut out, gauge.bytes(budget), MIN_KEEP_MESSAGES, &shut()); |
| 2418 | |
| 2419 | assert!(out.len() < v.len(), |
| 2420 | "the fold kept {} of {} messages, so nothing was folded at all", |
| 2421 | out.len(), v.len()); |
| 2422 | assert!(gauge.tokens(conversation_bytes(&out, &shut())) <= budget, |
| 2423 | "still {} tokens against a budget of {}", |
| 2424 | gauge.tokens(conversation_bytes(&out, &shut())), budget); |
| 2425 | assert!(pairing_is_whole(&out), "the folded conversation would be rejected"); |
| 2426 | // And it still knows what it did: the earliest file it read is named in the note, |
| 2427 | // and the latest is still in the conversation verbatim. |
| 2428 | assert!(out[0].text().contains("src/f0.rs"), |
| 2429 | "the fold forgot the first file it read: {}", out[0].content()); |
| 2430 | let tail_mentions = out.iter().skip(1).any(|m| match m { |
| 2431 | ChatMessage::Assistant { tool_calls, .. } => |
| 2432 | tool_calls.iter().any(|tc| tc.arguments.contains("src/f39.rs")), |
| 2433 | _ => m.text().contains("src/f39.rs"), |
| 2434 | }); |
| 2435 | assert!(tail_mentions, "the newest work was folded away instead of kept"); |
| 2436 | } |
| 2437 | |
| 2438 | #[test] |
| 2439 | fn test_folding_the_same_conversation_twice_is_stable_00() { |
| 2440 | // A folded conversation that folds again must keep shrinking, or a long session ends |
| 2441 | // up folding on every single turn and paying for a summary each time. |
| 2442 | let v = session(40, 6_000); |
| 2443 | let g = Gauge::default(); |
| 2444 | let cut = tail_start(&v, g.bytes(4_000), MIN_KEEP_MESSAGES, u64::MAX, &shut()); |
| 2445 | let once = match fold(&v, cut, notice(cut, "", &ledger_of(&v[..cut]), None)) { |
| 2446 | Ok(o) => o, Err(e) => panic!("{}", e), |
| 2447 | }; |
| 2448 | let cut2 = tail_start(&once, g.bytes(2_000), MIN_KEEP_MESSAGES, u64::MAX, &shut()); |
| 2449 | if cut2 > 0 { |
| 2450 | let twice = match fold(&once, cut2, notice(cut2, "", &ledger_of(&once[..cut2]), None)) { |
| 2451 | Ok(o) => o, Err(e) => panic!("{}", e), |
| 2452 | }; |
| 2453 | assert!(conversation_bytes(&twice, &shut()) < conversation_bytes(&once, &shut())); |
| 2454 | assert!(pairing_is_whole(&twice)); |
| 2455 | } |
| 2456 | } |
| 2457 | |
| 2458 | // ── The crystal fold's gate ────────────────────────────────────────────── |
| 2459 | // |
| 2460 | // Each is written as the crystal being lost: a proposal accepted that nothing can read, |
| 2461 | // one accepted that holds no keys, one refused for wearing a fence the model added |
| 2462 | // without meaning anything by it, and one whose unknown key was tidied away by the check |
| 2463 | // itself. |
| 2464 | |
| 2465 | /// A crystal with a key this build has never heard of, which is the case the whole |
| 2466 | /// contract turns on. |
| 2467 | fn crystal() -> &'static str { |
| 2468 | "{\"title\":\"Ship the parser\",\"summary\":\"Half done.\",\ |
| 2469 | \"sections\":[{\"heading\":\"State\",\"body\":\"Lexer lands.\"}],\ |
| 2470 | \"facts\":[{\"k\":\"crate\",\"v\":\"csv\"}],\"open\":[\"quoting\"],\ |
| 2471 | \"links\":[{\"label\":\"RFC\",\"href\":\"https://example.invalid/rfc\"}],\ |
| 2472 | \"mood\":{\"colour\":\"amber\"}}" |
| 2473 | } |
| 2474 | |
| 2475 | #[test] |
| 2476 | fn test_a_proposal_that_is_not_json_never_reaches_the_user() { |
| 2477 | // The gate that did not exist. Accepting a proposal REPLACES the crystal, so prose |
| 2478 | // offered as a fold is the whole memory of a Diamond traded for an apology. |
| 2479 | for bad in [ |
| 2480 | "I have folded the delta in. Here is the new crystal: {\"title\":\"x\"}", |
| 2481 | "{\"title\": \"unterminated", |
| 2482 | "{\"title\": \"x\",}", |
| 2483 | "# The old pursuit\n\nA crystal from before the migration.\n", |
| 2484 | ] { |
| 2485 | assert!(crystal_proposal(bad).is_err(), |
| 2486 | "a proposal nothing can parse was offered as a fold: {:?}", bad); |
| 2487 | } |
| 2488 | } |
| 2489 | |
| 2490 | #[test] |
| 2491 | fn test_a_proposal_that_is_json_but_not_an_object_is_refused_too() { |
| 2492 | // Valid JSON is not the question; a crystal is. A list or a string carries no key |
| 2493 | // the app, the page or the next reducer can name, which is the same total loss. |
| 2494 | for bad in ["[]", "[{\"title\":\"x\"}]", "\"the crystal\"", "42", "null", "true"] { |
| 2495 | assert!(crystal_proposal(bad).is_err(), |
| 2496 | "{:?} was offered as a crystal", bad); |
| 2497 | } |
| 2498 | } |
| 2499 | |
| 2500 | #[test] |
| 2501 | fn test_an_empty_proposal_is_still_refused() { |
| 2502 | // The one check there used to be, kept: a fold never empties a crystal. |
| 2503 | for bad in ["", " \n\t ", "```json\n```", "```"] { |
| 2504 | assert!(crystal_proposal(bad).is_err(), "{:?} emptied a crystal", bad); |
| 2505 | } |
| 2506 | } |
| 2507 | |
| 2508 | #[test] |
| 2509 | fn test_a_fenced_proposal_is_taken_rather_than_refused() { |
| 2510 | // The prompt forbids a fence and models add one anyway. Refusing here would spend a |
| 2511 | // paid round trip and the user's patience on punctuation. |
| 2512 | let inner = crystal(); |
| 2513 | for wrapped in [ |
| 2514 | fmt!("```json\n{}\n```", inner), |
| 2515 | fmt!("```\n{}\n```", inner), |
| 2516 | fmt!("```JSON\n{}\n```\n\n", inner), |
| 2517 | // A reply cut at the output limit loses its closing fence, and what it has is |
| 2518 | // still a whole object. |
| 2519 | fmt!("```json\n{}", inner), |
| 2520 | ] { |
| 2521 | let out = match crystal_proposal(&wrapped) { |
| 2522 | Ok(o) => o, |
| 2523 | Err(e) => panic!("a fenced crystal was refused: {}\n{}", e, wrapped), |
| 2524 | }; |
| 2525 | assert!(out.starts_with('{'), "the fence survived: {}", out); |
| 2526 | assert!(!out.contains("```"), "the fence survived: {}", out); |
| 2527 | } |
| 2528 | } |
| 2529 | |
| 2530 | #[test] |
| 2531 | fn test_the_check_hands_back_the_model_s_own_text_unaltered() { |
| 2532 | // It asks a question of the text; it is not a stage the text passes through. A |
| 2533 | // re-encoding would sort the keys and could not carry `mood` through, and carrying an |
| 2534 | // unrecognised key through EXACTLY is what the crystal's open schema is. |
| 2535 | let out = match crystal_proposal(crystal()) { |
| 2536 | Ok(o) => o, |
| 2537 | Err(e) => panic!("{}", e), |
| 2538 | }; |
| 2539 | assert_eq!(out, crystal()); |
| 2540 | assert!(out.contains("\"mood\""), "the check dropped a key it did not know: {}", out); |
| 2541 | // And the key order the model chose survives, which a decode-and-re-encode would |
| 2542 | // have sorted into alphabetical nonsense. |
| 2543 | let title = match out.find("\"title\"") { Some(i) => i, None => panic!("{}", out) }; |
| 2544 | let open = match out.find("\"open\"") { Some(i) => i, None => panic!("{}", out) }; |
| 2545 | assert!(title < open, "the keys were reordered: {}", out); |
| 2546 | } |
| 2547 | |
| 2548 | #[test] |
| 2549 | fn test_the_check_is_not_laxer_than_the_browser_that_reads_the_file() { |
| 2550 | // The one thing it must never be. A proposal Rust waves through and `JSON.parse` |
| 2551 | // then rejects is a crystal the user accepted and cannot open -- worse than a |
| 2552 | // refusal, because the refusal at least leaves the old crystal standing. |
| 2553 | // |
| 2554 | // A trailing comma is legal JDAT and is not JSON, which is why the decoder is asked |
| 2555 | // under its JSON configuration rather than its own. |
| 2556 | assert!(crystal_proposal("{\"title\":\"x\",}").is_err(), |
| 2557 | "a trailing comma was accepted"); |
| 2558 | } |
| 2559 | |
| 2560 | #[test] |
| 2561 | fn test_a_reply_cut_at_the_output_limit_is_refused_rather_than_written() { |
| 2562 | // The commonest way a fold goes wrong, and the one the decoder alone gets wrong: its |
| 2563 | // text form returns a map of whatever arrived when the input simply stops, so every |
| 2564 | // one of these decodes happily and `JSON.parse` rejects every one of them. |
| 2565 | for cut in [ |
| 2566 | "{\"title\": \"half", |
| 2567 | "{\"title\": \"Ship it\", \"sections\": [{\"heading\": \"State\"", |
| 2568 | "{\"title\": \"Ship it\",", |
| 2569 | "{\"title\": \"Ship it\"", |
| 2570 | ] { |
| 2571 | assert!(crystal_proposal(cut).is_err(), |
| 2572 | "a truncated proposal was offered as a whole crystal: {:?}", cut); |
| 2573 | } |
| 2574 | } |
| 2575 | |
| 2576 | #[test] |
| 2577 | fn test_anything_after_the_closing_brace_is_refused() { |
| 2578 | // The decoder stops reading at the close and says nothing about what follows, so a |
| 2579 | // reducer that emits the old crystal and then the new one would have had BOTH |
| 2580 | // written to the file. Nothing then reads it. |
| 2581 | for trailing in [ |
| 2582 | "{\"title\":\"old\"}\n{\"title\":\"new\"}", |
| 2583 | "{\"title\":\"x\"} — I kept the mood key.", |
| 2584 | "{\"title\":\"x\"}}", |
| 2585 | ] { |
| 2586 | assert!(crystal_proposal(trailing).is_err(), |
| 2587 | "a proposal with a second thing after the object was accepted: {:?}", |
| 2588 | trailing); |
| 2589 | } |
| 2590 | } |
| 2591 | |
| 2592 | #[test] |
| 2593 | fn test_a_brace_inside_the_prose_of_a_crystal_is_not_read_as_structure() { |
| 2594 | // A crystal's bodies are markdown and a Diamond's work is often code, so braces |
| 2595 | // inside strings are ordinary. Counting them would refuse the most useful crystals |
| 2596 | // there are -- and an escaped quote must not end the string either. |
| 2597 | let code = "{\"title\":\"Parser\",\"sections\":[{\"heading\":\"Snippet\",\ |
| 2598 | \"body\":\"```rust\\nfn main() { let s = \\\"}\\\"; }\\n```\"}]}"; |
| 2599 | match crystal_proposal(code) { |
| 2600 | Ok(o) => assert_eq!(o, code), |
| 2601 | Err(e) => panic!("a crystal carrying code was refused: {}\n{}", e, code), |
| 2602 | } |
| 2603 | } |
| 2604 | |
| 2605 | // ── What a fold took away ──────────────────────────────────────────────── |
| 2606 | // |
| 2607 | // The gate above cannot see any of this: a crystal that has lost half its keys parses |
| 2608 | // exactly as well as one that has not, and `{}` is a legal crystal because every core |
| 2609 | // key is optional. Each of these is written as the key going without anyone seeing. |
| 2610 | |
| 2611 | #[test] |
| 2612 | fn test_a_key_this_build_has_never_heard_of_is_named_when_it_goes() { |
| 2613 | // The whole rule, and the one key no schema, no form and no fallback view can miss |
| 2614 | // on the user's behalf, because nothing here knows what `mood` is for. |
| 2615 | let after = "{\"title\":\"Ship the parser\",\"summary\":\"Half done.\"}"; |
| 2616 | let lost = crystal_keys_lost(crystal(), after); |
| 2617 | assert!(lost.iter().any(|k| k == "mood"), |
| 2618 | "the unknown key went unremarked: {:?}", lost); |
| 2619 | } |
| 2620 | |
| 2621 | #[test] |
| 2622 | fn test_every_key_that_carried_something_and_went_is_named() { |
| 2623 | // Not just the unknown one. A fold that keeps the title and drops the rest is a |
| 2624 | // Diamond's memory gone on one click. |
| 2625 | let lost = crystal_keys_lost(crystal(), "{\"title\":\"Ship the parser\"}"); |
| 2626 | for k in ["summary", "sections", "facts", "open", "links", "mood"] { |
| 2627 | assert!(lost.iter().any(|l| l == k), "{} went unremarked: {:?}", k, lost); |
| 2628 | } |
| 2629 | assert!(!lost.iter().any(|l| l == "title"), "a key that stayed was named: {:?}", lost); |
| 2630 | } |
| 2631 | |
| 2632 | #[test] |
| 2633 | fn test_a_fold_that_took_nothing_away_says_nothing() { |
| 2634 | // It must be quiet in the ordinary case or it will be waved through in the one that |
| 2635 | // matters. Reordered, reworded, and with a key ADDED, is still nothing lost. |
| 2636 | let after = "{\"mood\":{\"colour\":\"amber\"},\"open\":[\"quoting\",\"CRLF\"],\ |
| 2637 | \"title\":\"Ship the parser\",\"summary\":\"Nearly there.\",\ |
| 2638 | \"sections\":[{\"heading\":\"State\",\"body\":\"Lexer lands.\"}],\ |
| 2639 | \"facts\":[{\"k\":\"crate\",\"v\":\"csv\"}],\ |
| 2640 | \"links\":[{\"label\":\"RFC\",\"href\":\"https://example.invalid/rfc\"}],\ |
| 2641 | \"owner\":\"jason\"}"; |
| 2642 | assert_eq!(Vec::<String>::new(), crystal_keys_lost(crystal(), after)); |
| 2643 | } |
| 2644 | |
| 2645 | #[test] |
| 2646 | fn test_a_key_emptied_rather_than_removed_is_not_reported_as_lost() { |
| 2647 | // Closing the last open thread is what a good fold DOES. Flagging it would teach the |
| 2648 | // user that the warning means nothing, which costs more than the case it catches. |
| 2649 | let after = "{\"title\":\"Ship the parser\",\"summary\":\"Half done.\",\ |
| 2650 | \"sections\":[{\"heading\":\"State\",\"body\":\"Lexer lands.\"}],\ |
| 2651 | \"facts\":[{\"k\":\"crate\",\"v\":\"csv\"}],\"open\":[],\ |
| 2652 | \"links\":[{\"label\":\"RFC\",\"href\":\"https://example.invalid/rfc\"}],\ |
| 2653 | \"mood\":{\"colour\":\"amber\"}}"; |
| 2654 | assert_eq!(Vec::<String>::new(), crystal_keys_lost(crystal(), after)); |
| 2655 | } |
| 2656 | |
| 2657 | #[test] |
| 2658 | fn test_a_key_that_was_already_empty_is_not_mourned() { |
| 2659 | // Same reasoning from the other end: a key standing empty in the old crystal held |
| 2660 | // nothing to lose, so its removal is tidying rather than damage. |
| 2661 | let before = "{\"title\":\"x\",\"summary\":\"\",\"open\":[],\"aside\":null,\ |
| 2662 | \"extra\":{},\"keeps\":\"something\"}"; |
| 2663 | // Four keys gone and not one of them held anything, so there is nothing to say. |
| 2664 | assert_eq!(Vec::<String>::new(), |
| 2665 | crystal_keys_lost(before, "{\"title\":\"x\",\"keeps\":\"something\"}")); |
| 2666 | // And the one that did hold something is named, alone. |
| 2667 | assert_eq!(vec![fmt!("keeps")], crystal_keys_lost(before, "{\"title\":\"x\"}")); |
| 2668 | } |
| 2669 | |
| 2670 | #[test] |
| 2671 | fn test_a_crystal_that_cannot_be_read_produces_a_claim_about_nothing() { |
| 2672 | // A Diamond still holding legacy markdown is the ordinary case, and the migration |
| 2673 | // owns it. Announcing that every key in the world has been lost, at the moment the |
| 2674 | // user is least able to judge it, would be noise standing where a real warning goes. |
| 2675 | assert_eq!(Vec::<String>::new(), |
| 2676 | crystal_keys_lost("# The old pursuit\n\nWritten before the migration.\n", |
| 2677 | crystal())); |
| 2678 | assert_eq!(Vec::<String>::new(), crystal_keys_lost(crystal(), "not json either")); |
| 2679 | assert_eq!(Vec::<String>::new(), crystal_keys_lost("[1,2,3]", crystal())); |
| 2680 | } |
| 2681 | |
| 2682 | #[test] |
| 2683 | fn test_the_comparison_reads_a_fenced_proposal_as_the_crystal_it_is() { |
| 2684 | // It is meant to be handed what `crystal_proposal` returned, which is already |
| 2685 | // unfenced -- but a caller reaching for the raw text must not be told that every key |
| 2686 | // survived because neither side parsed. |
| 2687 | let after = fmt!("```json\n{{\"title\":\"Ship the parser\"}}\n```"); |
| 2688 | assert!(crystal_keys_lost(crystal(), &after).iter().any(|k| k == "mood"), |
| 2689 | "a fenced proposal read as no loss at all: {:?}", |
| 2690 | crystal_keys_lost(crystal(), &after)); |
| 2691 | } |
| 2692 | |
| 2693 | #[test] |
| 2694 | fn test_the_emptiest_legal_crystal_is_the_case_no_parse_check_can_catch() { |
| 2695 | // `{}` is a valid crystal -- every core key is optional -- so the gate accepts it and |
| 2696 | // must. This is the only thing standing between that and a Diamond's whole memory. |
| 2697 | match crystal_proposal("{}") { |
| 2698 | Ok(o) => assert_eq!(o, "{}"), |
| 2699 | Err(e) => panic!("an empty object is a legal crystal: {}", e), |
| 2700 | } |
| 2701 | let lost = crystal_keys_lost(crystal(), "{}"); |
| 2702 | for k in ["title", "summary", "sections", "facts", "open", "links", "mood"] { |
| 2703 | assert!(lost.iter().any(|l| l == k), |
| 2704 | "{} vanished into an empty crystal unremarked: {:?}", k, lost); |
| 2705 | } |
| 2706 | } |
| 2707 | |
| 2708 | #[test] |
| 2709 | fn test_a_crystal_nested_to_provoke_a_crash_is_refused_with_a_sentence() { |
| 2710 | // The decoder recurses, and this one runs in a browser where the stack is a quarter |
| 2711 | // of what a native thread gets. A depth bound is the difference between a refusal |
| 2712 | // the user can read and the wasm module going down mid-fold. |
| 2713 | let deep = fmt!("{}{}{}", "{\"a\":", "[".repeat(400), "]".repeat(400)); |
| 2714 | assert!(crystal_proposal(&fmt!("{}}}", deep)).is_err(), |
| 2715 | "a proposal nested past the bound was parsed rather than refused"); |
| 2716 | } |
| 2717 | } |