342 KiB, 1 run
created by r2519314175:953, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! LLM client — chat completions with SSE streaming, in two wire dialects. |
| 2 | //! |
| 3 | //! Uses `fe2o3_net` for the underlying TLS connection. Parses the |
| 4 | //! `text/event-stream` response line-by-line, extracting `data:` lines |
| 5 | //! containing JSON objects with `delta` content. |
| 6 | //! |
| 7 | //! No `serde` or `reqwest` — neither API's JSON is complicated enough to |
| 8 | //! need one, and both are parsed by string scanning. This keeps the |
| 9 | //! dependency surface minimal and stays within the fe2o3 ecosystem. |
| 10 | //! |
| 11 | //! Two dialects share every public entry point, the retry policy and the |
| 12 | //! prompt-cache placement: the OpenAI-compatible `/chat/completions` that |
| 13 | //! every router speaks, and Anthropic's own `/v1/messages`. See |
| 14 | //! [`Dialect`] for why the second one could not simply be bent into the |
| 15 | //! first. |
| 16 | |
| 17 | use oxedyne_fe2o3_core::prelude::*; |
| 18 | use oxedyne_fe2o3_core::rand::Rand; |
| 19 | use oxedyne_fe2o3_jdat::prelude::*; |
| 20 | |
| 21 | use crate::protocol::{ChatMessage, ContentPart, Dropped, ImagePart, MessageContent, ToolCall}; |
| 22 | |
| 23 | // Native transport imports — the hand-rolled TLS client lives behind |
| 24 | // tokio + rustls, which do not target wasm32. |
| 25 | #[cfg(not(target_arch = "wasm32"))] |
| 26 | use std::sync::Arc; |
| 27 | #[cfg(not(target_arch = "wasm32"))] |
| 28 | use tokio::io::{AsyncReadExt, AsyncWriteExt}; |
| 29 | #[cfg(not(target_arch = "wasm32"))] |
| 30 | use tokio_rustls::rustls::ClientConfig; |
| 31 | |
| 32 | |
| 33 | // ┌───────────────────────────────────────────────────────────────┐ |
| 34 | // │ Dialect │ |
| 35 | // └───────────────────────────────────────────────────────────────┘ |
| 36 | |
| 37 | /// Which wire protocol an endpoint speaks. |
| 38 | /// |
| 39 | /// The OpenAI-compatible shape carried every provider Daimond had, so the |
| 40 | /// client was written as if there were only one. Anthropic's own Messages |
| 41 | /// API is not that shape and cannot be made into it: the system prompt is a |
| 42 | /// top-level field rather than a message, content is an array of typed |
| 43 | /// blocks rather than a string, a tool call is a `tool_use` block and its |
| 44 | /// result a `tool_result` block inside the *user* turn, the streamed events |
| 45 | /// are named rather than deltas of one object, and the usage counts have |
| 46 | /// different names and a different meaning. Bending one into the other |
| 47 | /// would have meant a translation layer that silently dropped whatever it |
| 48 | /// did not understand -- thinking blocks above all -- so the seam is |
| 49 | /// explicit instead, and every branch that needs it says which side it is on. |
| 50 | /// |
| 51 | /// The dialect is a property of the *endpoint*, not of the model: the same |
| 52 | /// Claude model is reachable through a router's `/chat/completions` (where |
| 53 | /// it speaks OpenAI) and through Anthropic's `/v1/messages` (where it does |
| 54 | /// not). Prompt caching gates on the model id for exactly the same reason |
| 55 | /// in reverse -- see [`model_caches_on_request`]. |
| 56 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 57 | pub enum Dialect { |
| 58 | /// OpenAI-compatible chat completions. |
| 59 | OpenAi, |
| 60 | /// Anthropic's Messages API. |
| 61 | Anthropic, |
| 62 | } |
| 63 | |
| 64 | impl Dialect { |
| 65 | |
| 66 | /// Which dialect the endpoint at `host``path` speaks. |
| 67 | /// |
| 68 | /// Two independent signals, either of which is conclusive: Anthropic's |
| 69 | /// own host, and the `/v1/messages` path that only the Messages API |
| 70 | /// serves. Everything else is OpenAI-compatible, which is the right |
| 71 | /// default -- a router serving `anthropic/claude-opus-5` is still |
| 72 | /// speaking OpenAI. |
| 73 | /// |
| 74 | /// # Arguments |
| 75 | /// * `host` - The request host, without scheme or port. |
| 76 | /// * `path` - The request path. |
| 77 | pub fn for_endpoint(host: &str, path: &str) -> Self { |
| 78 | let h = host.to_ascii_lowercase(); |
| 79 | let p = path.trim_end_matches('/').to_ascii_lowercase(); |
| 80 | if h == "api.anthropic.com" || h.ends_with(".anthropic.com") || p.ends_with("/v1/messages") { |
| 81 | Self::Anthropic |
| 82 | } else { |
| 83 | Self::OpenAi |
| 84 | } |
| 85 | } |
| 86 | } |
| 87 | |
| 88 | |
| 89 | // ┌───────────────────────────────────────────────────────────────┐ |
| 90 | // │ Thinking carry │ |
| 91 | // └───────────────────────────────────────────────────────────────┘ |
| 92 | |
| 93 | /// How many assistant turns of signed reasoning to hold at once. |
| 94 | /// |
| 95 | /// A round of an agentic loop adds one entry, so this bounds the memory a very |
| 96 | /// long loop can hold while covering more rounds than any single tool loop runs. |
| 97 | const CARRY_MAX_TURNS: usize = 32; |
| 98 | |
| 99 | /// The signed thinking blocks of recent assistant turns, held until their tool |
| 100 | /// results come back. |
| 101 | /// |
| 102 | /// Anthropic requires that a thinking-enabled assistant turn which asked for |
| 103 | /// tools be handed back *complete and unmodified* alongside the tool results: |
| 104 | /// "within a tool-use turn, pass thinking blocks back". A block the caller |
| 105 | /// edited is rejected with a 400; a block the caller dropped makes the API |
| 106 | /// silently disable thinking for the request, which is the same defect wearing |
| 107 | /// a quieter coat. Passing every turn's blocks back is the documented |
| 108 | /// recommendation beyond that: on the models that keep them, the reasoning |
| 109 | /// stays in context and caches incrementally with the tool results, so dropping |
| 110 | /// it costs both continuity and money on every round after the first. |
| 111 | /// |
| 112 | /// The conversation type this client is given ([`ChatMessage`]) has nowhere to |
| 113 | /// put a thinking block -- it is OpenAI-shaped, and OpenAI has no such thing -- |
| 114 | /// so the blocks are held here instead, keyed by the tool-call id they were |
| 115 | /// generated beside. That id is what makes the association safe: the very next |
| 116 | /// request carries the same id in its assistant turn, so a turn's reasoning can |
| 117 | /// only ever be handed back with the call it actually produced. A turn that |
| 118 | /// asked for no tools stores nothing, because it is already over. |
| 119 | #[derive(Clone, Debug, Default)] |
| 120 | struct ThinkCarry { |
| 121 | /// `(first tool-call id, blocks)`, oldest first. The blocks are already |
| 122 | /// serialised as JSON objects, in the order the model produced them. |
| 123 | turns: Vec<(String, Vec<String>)>, |
| 124 | } |
| 125 | |
| 126 | /// The `say` calls whose fold the user currently has OPEN, by tool-call id. |
| 127 | /// |
| 128 | /// **THE FOLD IS THE CONTEXT CONTROL, and this is what makes that true.** A folded detail is |
| 129 | /// stripped from the payload, which is right when the user has closed it: they are done with it, |
| 130 | /// and re-sending it on every later turn buys nothing. But a fold they have OPENED is a fold they |
| 131 | /// are reading, and the next thing they say is likely to be about it -- so the model should be |
| 132 | /// holding what the user is looking at. |
| 133 | /// |
| 134 | /// The user's own gesture therefore decides the model's working set, with no second control to |
| 135 | /// learn and no decision to make twice. What is on their screen and what is in its context are |
| 136 | /// the same set, which is the only arrangement where "why does it not remember that?" has an |
| 137 | /// answer they can see. |
| 138 | /// |
| 139 | /// It is rebuilt from the page before every request rather than accumulated here, because a fold |
| 140 | /// can be opened and closed between two turns and the payload has to follow. |
| 141 | /// |
| 142 | /// **It costs a cache miss on the turn it changes.** Opening a fold rewrites a message that was |
| 143 | /// already in the prefix, so everything from that point is re-read once. Stable again afterwards. |
| 144 | type OpenFolds = std::rc::Rc<std::cell::RefCell<OpenSet>>; |
| 145 | |
| 146 | /// The call ids of the `say` folds the user has open, as [`LlmClient::open_folds`] hands them |
| 147 | /// over and as the sizing path in [`crate::agent::compact`] reads them. |
| 148 | pub type OpenSet = std::collections::HashSet<String>; |
| 149 | |
| 150 | /// A [`ThinkCarry`] shared across clones of a client. |
| 151 | #[cfg(not(target_arch = "wasm32"))] |
| 152 | type Carry = std::sync::Arc<std::sync::Mutex<ThinkCarry>>; |
| 153 | |
| 154 | /// A [`ThinkCarry`] shared across clones of a client. |
| 155 | #[cfg(target_arch = "wasm32")] |
| 156 | type Carry = std::rc::Rc<std::cell::RefCell<ThinkCarry>>; |
| 157 | |
| 158 | /// Whether an endpoint has been caught refusing pictures, shared across clones of a client. |
| 159 | /// |
| 160 | /// Learned rather than declared. [`model_can_see`] is a list of eight model ids known to be |
| 161 | /// blind, so every model it has not heard of is assumed sighted -- which is the right default |
| 162 | /// (a new sighted model works at once) and is wrong for exactly as long as it takes one turn |
| 163 | /// to fail. This is the other half: once a request carrying pictures comes back refused and the |
| 164 | /// same request without them succeeds, the endpoint is marked and no later turn pays for the |
| 165 | /// discovery twice. |
| 166 | #[cfg(not(target_arch = "wasm32"))] |
| 167 | type Blind = std::sync::Arc<std::sync::atomic::AtomicBool>; |
| 168 | |
| 169 | /// Whether an endpoint has been caught refusing pictures, shared across clones of a client. |
| 170 | #[cfg(target_arch = "wasm32")] |
| 171 | type Blind = std::rc::Rc<std::cell::Cell<bool>>; |
| 172 | |
| 173 | /// A fresh flag, unset: nothing has been refused yet. |
| 174 | fn new_blind() -> Blind { |
| 175 | #[cfg(not(target_arch = "wasm32"))] |
| 176 | { std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)) } |
| 177 | #[cfg(target_arch = "wasm32")] |
| 178 | { std::rc::Rc::new(std::cell::Cell::new(false)) } |
| 179 | } |
| 180 | |
| 181 | /// A fresh, empty carry. |
| 182 | fn new_carry() -> Carry { |
| 183 | #[cfg(not(target_arch = "wasm32"))] |
| 184 | { std::sync::Arc::new(std::sync::Mutex::new(ThinkCarry::default())) } |
| 185 | #[cfg(target_arch = "wasm32")] |
| 186 | { std::rc::Rc::new(std::cell::RefCell::new(ThinkCarry::default())) } |
| 187 | } |
| 188 | |
| 189 | |
| 190 | /// One piece of a streamed turn, labelled with what kind of thing it is. |
| 191 | /// |
| 192 | /// The two are different KINDS of content and not two shades of one. Text is the |
| 193 | /// answer: it is accumulated, persisted, and sent back to the model next turn as |
| 194 | /// what the assistant said. Reasoning is the model's working, which the user pays |
| 195 | /// for and which decides the answer, but which is not the answer and must never be |
| 196 | /// stored as one -- see `AgentEvent::Thinking` in src/protocol.rs. |
| 197 | /// |
| 198 | /// One sink and not two, because a caller holds ONE `&mut` to whatever it is |
| 199 | /// forwarding into. Two closures over the same event sink is a borrow the compiler |
| 200 | /// refuses, and the ways round it (a `RefCell`, a channel) buy nothing: the stream |
| 201 | /// is serial, so exactly one delta is in flight at a time. |
| 202 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 203 | pub enum Delta<'a> { |
| 204 | Text(&'a str), |
| 205 | Reasoning(&'a str), |
| 206 | } |
| 207 | |
| 208 | // ┌───────────────────────────────────────────────────────────────┐ |
| 209 | // │ LlmClient │ |
| 210 | // └───────────────────────────────────────────────────────────────┘ |
| 211 | |
| 212 | /// Async client for a chat completions API, in either [`Dialect`]. |
| 213 | /// |
| 214 | /// Connects via TLS to the configured host, POSTs a chat completion |
| 215 | /// request with `stream: true`, and parses the SSE response |
| 216 | /// incrementally — calling `on_token` for each chunk as it arrives, |
| 217 | /// saying whether it is the answer or the model's working ([`Delta`]). |
| 218 | #[derive(Clone, Debug)] |
| 219 | pub struct LlmClient { |
| 220 | pub host: String, |
| 221 | pub port: u16, |
| 222 | pub path: String, |
| 223 | pub api_key: String, |
| 224 | pub model: String, |
| 225 | /// Upper bound on generated tokens per turn. Prevents runaway |
| 226 | /// reasoning loops (e.g. GLM-5.2 without a cap). |
| 227 | pub max_tokens: u32, |
| 228 | /// Which wire protocol the endpoint speaks, derived from the host and |
| 229 | /// path at construction. See [`Dialect`]. |
| 230 | pub dialect: Dialect, |
| 231 | /// How transient provider failures are retried. Shared by both transports. |
| 232 | pub retry: RetryPolicy, |
| 233 | /// The signed thinking blocks of the assistant turn now awaiting tool |
| 234 | /// results, so they can be handed back on the next request. Shared |
| 235 | /// across clones, because a sub-agent built from a cloned client is |
| 236 | /// continuing the same turn. See [`ThinkCarry`]. |
| 237 | think: Carry, |
| 238 | /// The `say` folds the user has open. See [`OpenFolds`]. |
| 239 | open_folds: OpenFolds, |
| 240 | /// Set once this endpoint has been caught refusing a request that carried pictures. |
| 241 | /// See [`Blind`]; read by [`LlmClient::vision_guard`] and set by the strip-and-retry in |
| 242 | /// [`LlmClient::stream_turn`] and [`LlmClient::chat_once`]. |
| 243 | blind: Blind, |
| 244 | /// Root-trust TLS configuration for the native transport. The wasm |
| 245 | /// transport delegates trust to the browser's `fetch`, so this field |
| 246 | /// is native-only. |
| 247 | #[cfg(not(target_arch = "wasm32"))] |
| 248 | pub tls_config: Arc<ClientConfig>, |
| 249 | /// Wasm transport URL scheme selector: `true` builds `https://…`, |
| 250 | /// `false` builds `http://…`. Defaults to `https` (all real |
| 251 | /// providers are TLS-only); an `http` client targets a local mock |
| 252 | /// over `127.0.0.1` for headless testing, where the browser still |
| 253 | /// treats the origin as a secure context. |
| 254 | #[cfg(target_arch = "wasm32")] |
| 255 | pub secure: bool, |
| 256 | /// Shared abort slot for the browser transport. Each `fetch` installs |
| 257 | /// a fresh [`web_sys::AbortController`] here and wires its signal into |
| 258 | /// the request; [`abort`](Self::abort) fires it to cancel the in-flight |
| 259 | /// turn. An `Rc<RefCell<…>>` (never `unsafe`), shared across clones so |
| 260 | /// a sub-agent built from a cloned client aborts on the same signal. |
| 261 | #[cfg(target_arch = "wasm32")] |
| 262 | abort: std::rc::Rc<std::cell::RefCell<Option<web_sys::AbortController>>>, |
| 263 | } |
| 264 | |
| 265 | /// The `usage` block a provider reports for a call. |
| 266 | /// |
| 267 | /// The token counts were always read; the other two are what the provider |
| 268 | /// says about its own billing, and are worth strictly more than any estimate |
| 269 | /// made from them. A router charges its own negotiated rate, and a prompt |
| 270 | /// cache read is a fraction of a fresh one -- neither is visible in a token |
| 271 | /// count, so pricing from tokens alone overstated spend several-fold. |
| 272 | #[derive(Clone, Copy, Debug, Default)] |
| 273 | pub struct Usage { |
| 274 | pub prompt: u64, |
| 275 | pub completion: u64, |
| 276 | /// Prompt tokens served from the provider's cache, a subset of `prompt`. |
| 277 | pub cached: u64, |
| 278 | /// What the provider says the call actually cost, in USD. Zero means it |
| 279 | /// said nothing, never that the call was free. |
| 280 | pub cost_usd: f64, |
| 281 | } |
| 282 | |
| 283 | /// The response from a completed streaming chat call. |
| 284 | #[derive(Clone, Debug, Default)] |
| 285 | pub struct ChatResponse { |
| 286 | pub content: String, |
| 287 | pub prompt_tokens: u64, |
| 288 | pub completion_tokens: u64, |
| 289 | /// Prompt tokens the provider served from its cache. |
| 290 | pub cached_tokens: u64, |
| 291 | /// What the provider says this call cost, in USD; `0.0` when it did not |
| 292 | /// say. An aborted stream may never deliver the usage chunk at all. |
| 293 | pub cost_usd: f64, |
| 294 | /// Set when the turn was cancelled mid-stream (browser abort). The |
| 295 | /// `content` then holds whatever streamed before the cancellation, so |
| 296 | /// the caller keeps the partial answer rather than reporting an error. |
| 297 | pub aborted: bool, |
| 298 | /// How many times this call was retried before it succeeded; see |
| 299 | /// [`ChatOnceResponse::retries`]. |
| 300 | pub retries: u32, |
| 301 | /// The model's own reasoning; see [`ChatOnceResponse::thinking`]. |
| 302 | pub thinking: String, |
| 303 | /// Set when the provider stopped because the reply hit `max_tokens` -- |
| 304 | /// `finish_reason: "length"`, or Anthropic's `stop_reason: "max_tokens"`. |
| 305 | /// |
| 306 | /// A tool call cut here arrives as MALFORMED JSON, so the caller needs to |
| 307 | /// tell "the model wrote bad JSON" from "the reply ran out of room": the |
| 308 | /// first is the model's mistake, the second is a setting, and only one of |
| 309 | /// them is worth retrying. |
| 310 | /// |
| 311 | /// It is NOT an error and is never retried. A reply that hit the cap is a |
| 312 | /// complete HTTP 200, and sending the same request again costs money and |
| 313 | /// produces the same cut. |
| 314 | pub truncated: bool, |
| 315 | } |
| 316 | |
| 317 | /// The response from a chat call that may include tool calls the model |
| 318 | /// wants executed. Whether it was produced by a streaming or a |
| 319 | /// non-streaming request, the accumulated shape is the same. |
| 320 | #[derive(Clone, Debug, Default)] |
| 321 | pub struct ChatOnceResponse { |
| 322 | pub content: String, |
| 323 | pub tool_calls: Vec<ToolCall>, |
| 324 | pub prompt_tokens: u64, |
| 325 | pub completion_tokens: u64, |
| 326 | /// Prompt tokens the provider served from its cache. |
| 327 | pub cached_tokens: u64, |
| 328 | /// What the provider says this call cost, in USD; see |
| 329 | /// [`ChatResponse::cost_usd`]. |
| 330 | pub cost_usd: f64, |
| 331 | /// Set when the turn was cancelled mid-stream (browser abort); see |
| 332 | /// [`ChatResponse::aborted`]. |
| 333 | pub aborted: bool, |
| 334 | /// How many times this call was retried before it succeeded. Zero is the |
| 335 | /// ordinary case; anything else is time the user waited for a provider that |
| 336 | /// was not ready, and is worth showing rather than hiding. |
| 337 | pub retries: u32, |
| 338 | /// The model's whole reasoning for this turn, gathered as it streamed. Anthropic |
| 339 | /// direct returns it from `thinking_delta`; an OpenAI-dialect endpoint returns it |
| 340 | /// on `reasoning` (OpenRouter's spelling) or `reasoning_content` (DeepSeek's own). |
| 341 | /// Empty for a model that does not reason, which is most of them. |
| 342 | /// |
| 343 | /// NEVER part of `content`: reasoning is not the answer, and a caller that |
| 344 | /// persisted it as one would be putting the model's working out where its reply |
| 345 | /// should be -- and handing it back next turn as something the model said. The |
| 346 | /// tokens are already counted in `completion_tokens`, because thinking is billed |
| 347 | /// as output whether or not its text comes back. |
| 348 | /// |
| 349 | /// A streaming caller does not need this: the same words reached it as |
| 350 | /// [`Delta::Reasoning`] while the round ran, which is the only time showing them |
| 351 | /// does any good. It is here for the callers that take a turn whole. |
| 352 | pub thinking: String, |
| 353 | /// Set when the provider stopped because the reply hit `max_tokens` -- |
| 354 | /// `finish_reason: "length"`, or Anthropic's `stop_reason: "max_tokens"`. |
| 355 | /// |
| 356 | /// A tool call cut here arrives as MALFORMED JSON, so the caller needs to |
| 357 | /// tell "the model wrote bad JSON" from "the reply ran out of room": the |
| 358 | /// first is the model's mistake, the second is a setting, and only one of |
| 359 | /// them is worth retrying. |
| 360 | /// |
| 361 | /// It is NOT an error and is never retried. A reply that hit the cap is a |
| 362 | /// complete HTTP 200, and sending the same request again costs money and |
| 363 | /// produces the same cut. |
| 364 | pub truncated: bool, |
| 365 | } |
| 366 | |
| 367 | |
| 368 | // ┌───────────────────────────────────────────────────────────────┐ |
| 369 | // │ Retry │ |
| 370 | // └───────────────────────────────────────────────────────────────┘ |
| 371 | |
| 372 | /// Extra milliseconds added on top of a provider's `Retry-After`, so a fan-out |
| 373 | /// of workers told the same thing does not all come back at the same instant. |
| 374 | const RETRY_AFTER_JITTER_MS: u64 = 250; |
| 375 | |
| 376 | /// Bytes of a refusal's body carried into the error. |
| 377 | /// |
| 378 | /// Enough for what every provider puts first -- the message, the type and the code -- |
| 379 | /// and short enough that a provider answering an oversized request with an echo of it |
| 380 | /// cannot put the whole thing in a user's message pane. Both transports use it, so the |
| 381 | /// browser and the native path say the same thing about the same failure. |
| 382 | const ERR_BODY_BYTES: usize = 300; |
| 383 | |
| 384 | /// `s` cut to at most `n` bytes, never through the middle of a character. |
| 385 | /// |
| 386 | /// `&s[..n]` panics on a multi-byte boundary, and the one place this is used is an error |
| 387 | /// path handed arbitrary bytes from a provider -- exactly where a panic is least welcome |
| 388 | /// and least likely to be noticed in testing. |
| 389 | /// |
| 390 | /// # Arguments |
| 391 | /// * `s` - The text to cut. |
| 392 | /// * `n` - The most bytes the result may occupy. |
| 393 | fn clip_bytes(s: &str, n: usize) -> &str { |
| 394 | let mut cut = s.len().min(n); |
| 395 | while cut > 0 && !s.is_char_boundary(cut) { |
| 396 | cut -= 1; |
| 397 | } |
| 398 | &s[..cut] |
| 399 | } |
| 400 | |
| 401 | /// Bounded exponential backoff for transient provider failures. |
| 402 | /// |
| 403 | /// A 429, a 5xx or a dropped connection is the provider saying "not now"; every |
| 404 | /// other 4xx is the request itself being wrong, and sending it again only costs |
| 405 | /// money and time. Only the former is retried. |
| 406 | #[derive(Clone, Copy, Debug)] |
| 407 | pub struct RetryPolicy { |
| 408 | /// Total attempts including the first. One disables retrying. |
| 409 | pub max_attempts: u32, |
| 410 | /// Backoff before the first retry, doubling for each one after it. |
| 411 | pub base_ms: u64, |
| 412 | /// Ceiling on any single backoff. |
| 413 | pub max_backoff_ms: u64, |
| 414 | /// Ceiling on the sum of every backoff within one call, so a turn ends |
| 415 | /// while the user is still watching it. |
| 416 | pub max_total_wait_ms: u64, |
| 417 | } |
| 418 | |
| 419 | impl Default for RetryPolicy { |
| 420 | fn default() -> Self { |
| 421 | // Widened 2026-08-19: a laptop moving between locations routinely drops |
| 422 | // the network for tens of seconds while it wakes, reconnects and DNS |
| 423 | // resolves. The previous budget (4 attempts, 20s total) survived a flaky |
| 424 | // access point but not a 30-second gap, so a turn that could have completed |
| 425 | // once the machine settled died instead. Eight attempts over up to two |
| 426 | // minutes gives the reconnect time to happen, while still bounded so a |
| 427 | // genuinely down provider ends the turn rather than hanging. The stub test |
| 428 | // client overrides this with a fast policy, so the suite is unaffected. |
| 429 | Self { |
| 430 | max_attempts: 8, |
| 431 | base_ms: 1_000, |
| 432 | max_backoff_ms: 30_000, |
| 433 | max_total_wait_ms: 120_000, |
| 434 | } |
| 435 | } |
| 436 | } |
| 437 | |
| 438 | impl RetryPolicy { |
| 439 | |
| 440 | /// The delay before retry number `retry`, counting the first retry as one. |
| 441 | /// |
| 442 | /// A provider's own `Retry-After` wins over the computed backoff and is |
| 443 | /// never shortened -- it is the one party that knows when it will be ready. |
| 444 | /// Jitter is added either way: eight workers that hit the same 429 must not |
| 445 | /// retry in lockstep. |
| 446 | pub fn delay_ms(&self, retry: u32, after_ms: Option<u64>) -> u64 { |
| 447 | if let Some(ms) = after_ms { |
| 448 | return ms.saturating_add(Rand::in_range(0u64, RETRY_AFTER_JITTER_MS)); |
| 449 | } |
| 450 | let shift = retry.saturating_sub(1).min(16); |
| 451 | let nominal = self.base_ms |
| 452 | .saturating_mul(1u64 << shift) |
| 453 | .min(self.max_backoff_ms); |
| 454 | // Equal jitter: half the nominal delay, plus a random part of the rest. |
| 455 | let half = nominal / 2; |
| 456 | half + Rand::in_range(0u64, nominal - half) |
| 457 | } |
| 458 | |
| 459 | /// Whether another attempt is allowed, and what it must wait first. |
| 460 | /// |
| 461 | /// `None` ends the attempt: either the budget of attempts is spent, or the |
| 462 | /// next backoff would push the total wait past its bound. |
| 463 | /// |
| 464 | /// # Arguments |
| 465 | /// * `retries` - Retries already made. |
| 466 | /// * `waited` - Milliseconds already slept within this call. |
| 467 | /// * `after_ms` - What the provider asked for, if it asked. |
| 468 | pub fn next_delay(&self, retries: u32, waited: u64, after_ms: Option<u64>) -> Option<u64> { |
| 469 | if retries + 1 >= self.max_attempts { |
| 470 | return None; |
| 471 | } |
| 472 | let delay = self.delay_ms(retries + 1, after_ms); |
| 473 | if waited.saturating_add(delay) > self.max_total_wait_ms { |
| 474 | return None; |
| 475 | } |
| 476 | Some(delay) |
| 477 | } |
| 478 | } |
| 479 | |
| 480 | /// A transport failure, and whether trying again could plausibly succeed. |
| 481 | /// |
| 482 | /// Retryability is decided where the status code is still in hand, rather than |
| 483 | /// by reading it back out of an error message later. |
| 484 | struct TransportErr { |
| 485 | retryable: bool, // is another attempt worth making? |
| 486 | after_ms: Option<u64>, // what the provider asked us to wait, if it said |
| 487 | // A few plain words for the retry notice, and -- since 2026-08-28 -- for the |
| 488 | // caller. The error itself carries file, line and ANSI colouring, none of |
| 489 | // which belongs in a user's message pane. See [`TransportErr::crossed`]. |
| 490 | reason: String, |
| 491 | err: Error<ErrTag>, |
| 492 | } |
| 493 | |
| 494 | impl TransportErr { |
| 495 | |
| 496 | /// A failure worth another attempt: a 429, a 5xx, or a broken connection. |
| 497 | fn transient(reason: String, err: Error<ErrTag>) -> Self { |
| 498 | Self { retryable: true, after_ms: None, reason, err } |
| 499 | } |
| 500 | |
| 501 | /// A failure that will fail again the same way: a malformed request, a bad |
| 502 | /// key, an unknown model. |
| 503 | fn fatal(reason: String, err: Error<ErrTag>) -> Self { |
| 504 | Self { retryable: false, after_ms: None, reason, err } |
| 505 | } |
| 506 | |
| 507 | /// Attach the provider's requested delay. |
| 508 | fn after(mut self, after_ms: Option<u64>) -> Self { |
| 509 | self.after_ms = after_ms; |
| 510 | self |
| 511 | } |
| 512 | |
| 513 | /// The failure as it LEAVES this module, with the reason in front of it. |
| 514 | /// |
| 515 | /// WHY THE REASON HAS TO TRAVEL. Until this existed only `err` was returned, |
| 516 | /// and `err` on the browser path is the browser's own sentence -- Chromium |
| 517 | /// says `Failed to fetch` and WebKit says `Load failed` for the identical |
| 518 | /// event. The app's offline classifier (`isUnreachable`, www/js/daimond.js) |
| 519 | /// was therefore reduced to matching one vendor's prose, and on iOS it matched |
| 520 | /// nothing: a turn that died before the first token was written off as a |
| 521 | /// provider refusal, and the recovery built for exactly that case never ran. |
| 522 | /// |
| 523 | /// `reason` is this client's own wording and is the same on every browser, so |
| 524 | /// putting it in front of the error makes the classification a property of |
| 525 | /// Daimond rather than of Safari. It is first because the reader -- person or |
| 526 | /// regex -- should meet the plain sentence before the framing. |
| 527 | fn crossed(self) -> Error<ErrTag> { |
| 528 | err!(self.err, "{}", self.reason; IO, Network, Wire) |
| 529 | } |
| 530 | } |
| 531 | |
| 532 | /// Whether an HTTP status is worth another attempt. |
| 533 | /// |
| 534 | /// 429 is rate limiting and 5xx is the provider's own trouble; every other |
| 535 | /// status is about this request and will not change by being sent twice. |
| 536 | pub(crate) fn status_retryable(status: u16) -> bool { |
| 537 | status == 429 || (500..600).contains(&status) |
| 538 | } |
| 539 | |
| 540 | /// Read a `Retry-After` header value as milliseconds. |
| 541 | /// |
| 542 | /// Only the delta-seconds form is understood. The HTTP-date form reads as |
| 543 | /// absent, which falls back to the client's own backoff rather than guessing. |
| 544 | pub(crate) fn parse_retry_after(value: &str) -> Option<u64> { |
| 545 | value.trim().parse::<u64>().ok().map(|s| s.saturating_mul(1_000)) |
| 546 | } |
| 547 | |
| 548 | /// Read the status code out of an HTTP status line. |
| 549 | #[cfg(not(target_arch = "wasm32"))] |
| 550 | pub(crate) fn status_code(line: &str) -> Option<u16> { |
| 551 | line.split_whitespace().nth(1).and_then(|c| c.parse::<u16>().ok()) |
| 552 | } |
| 553 | |
| 554 | /// Find a header's value in a raw HTTP header block, case-insensitively. |
| 555 | #[cfg(not(target_arch = "wasm32"))] |
| 556 | pub(crate) fn header_value(headers: &str, name: &str) -> Option<String> { |
| 557 | for line in headers.lines() { |
| 558 | let (key, value) = match line.split_once(':') { |
| 559 | Some(kv) => kv, |
| 560 | None => continue, |
| 561 | }; |
| 562 | if key.trim().eq_ignore_ascii_case(name) { |
| 563 | return Some(value.trim().to_string()); |
| 564 | } |
| 565 | } |
| 566 | None |
| 567 | } |
| 568 | |
| 569 | /// Sleep for `ms` milliseconds on the native transport. |
| 570 | #[cfg(not(target_arch = "wasm32"))] |
| 571 | pub(crate) async fn sleep_ms(ms: u64) { |
| 572 | tokio::time::sleep(std::time::Duration::from_millis(ms)).await; |
| 573 | } |
| 574 | |
| 575 | /// Sleep for `ms` milliseconds in the browser, via `setTimeout`. |
| 576 | /// |
| 577 | /// A scope with no timer resolves immediately, so a retry still happens -- just |
| 578 | /// without the pause. |
| 579 | #[cfg(target_arch = "wasm32")] |
| 580 | pub(crate) async fn sleep_ms(ms: u64) { |
| 581 | use wasm_bindgen::JsCast; |
| 582 | use wasm_bindgen::JsValue; |
| 583 | use wasm_bindgen_futures::JsFuture; |
| 584 | |
| 585 | let ms = ms.min(i32::MAX as u64) as i32; |
| 586 | let promise = js_sys::Promise::new(&mut |resolve: js_sys::Function, _reject| { |
| 587 | let scheduled = if let Some(win) = web_sys::window() { |
| 588 | win.set_timeout_with_callback_and_timeout_and_arguments_0(&resolve, ms) |
| 589 | } else { |
| 590 | match js_sys::global().dyn_into::<web_sys::WorkerGlobalScope>() { |
| 591 | Ok(scope) => scope |
| 592 | .set_timeout_with_callback_and_timeout_and_arguments_0(&resolve, ms), |
| 593 | Err(_) => Err(JsValue::NULL), |
| 594 | } |
| 595 | }; |
| 596 | if scheduled.is_err() { |
| 597 | let _ = resolve.call0(&JsValue::NULL); |
| 598 | } |
| 599 | }); |
| 600 | let _ = JsFuture::from(promise).await; |
| 601 | } |
| 602 | |
| 603 | |
| 604 | impl LlmClient { |
| 605 | |
| 606 | /// Construct a client for the native transport (tokio + rustls). |
| 607 | #[cfg(not(target_arch = "wasm32"))] |
| 608 | pub fn new( |
| 609 | host: &str, |
| 610 | port: u16, |
| 611 | path: &str, |
| 612 | api_key: &str, |
| 613 | model: &str, |
| 614 | max_tokens: u32, |
| 615 | tls_config: Arc<ClientConfig>, |
| 616 | ) -> Self { |
| 617 | Self { |
| 618 | dialect: Dialect::for_endpoint(host, path), |
| 619 | host: host.to_string(), |
| 620 | port, |
| 621 | path: path.to_string(), |
| 622 | api_key: api_key.to_string(), |
| 623 | model: model.to_string(), |
| 624 | max_tokens, |
| 625 | retry: RetryPolicy::default(), |
| 626 | think: new_carry(), |
| 627 | open_folds: std::rc::Rc::new(std::cell::RefCell::new(std::collections::HashSet::new())), |
| 628 | blind: new_blind(), |
| 629 | tls_config, |
| 630 | } |
| 631 | } |
| 632 | |
| 633 | /// Construct a client for the wasm transport (browser `fetch`). |
| 634 | /// |
| 635 | /// TLS trust is handled by the browser, so no `tls_config` is |
| 636 | /// required — the streaming API (`chat_stream` / `chat_once`) is |
| 637 | /// otherwise identical to the native client. |
| 638 | #[cfg(target_arch = "wasm32")] |
| 639 | pub fn new( |
| 640 | host: &str, |
| 641 | port: u16, |
| 642 | path: &str, |
| 643 | api_key: &str, |
| 644 | model: &str, |
| 645 | max_tokens: u32, |
| 646 | ) -> Self { |
| 647 | Self::new_with_scheme(host, port, path, api_key, model, max_tokens, true) |
| 648 | } |
| 649 | |
| 650 | /// Construct a wasm client with an explicit URL scheme. |
| 651 | /// |
| 652 | /// `secure` selects `https` (`true`) or `http` (`false`). Real |
| 653 | /// providers always use `https`; the `http` form exists so a local |
| 654 | /// mock over `127.0.0.1` can be driven in a headless test. |
| 655 | #[cfg(target_arch = "wasm32")] |
| 656 | pub fn new_with_scheme( |
| 657 | host: &str, |
| 658 | port: u16, |
| 659 | path: &str, |
| 660 | api_key: &str, |
| 661 | model: &str, |
| 662 | max_tokens: u32, |
| 663 | secure: bool, |
| 664 | ) -> Self { |
| 665 | Self { |
| 666 | dialect: Dialect::for_endpoint(host, path), |
| 667 | host: host.to_string(), |
| 668 | port, |
| 669 | path: path.to_string(), |
| 670 | api_key: api_key.to_string(), |
| 671 | model: model.to_string(), |
| 672 | max_tokens, |
| 673 | retry: RetryPolicy::default(), |
| 674 | think: new_carry(), |
| 675 | open_folds: std::rc::Rc::new(std::cell::RefCell::new(std::collections::HashSet::new())), |
| 676 | blind: new_blind(), |
| 677 | secure, |
| 678 | abort: std::rc::Rc::new(std::cell::RefCell::new(None)), |
| 679 | } |
| 680 | } |
| 681 | |
| 682 | /// Send a streaming chat completion request. |
| 683 | /// |
| 684 | /// Reads the SSE response line-by-line from the TLS stream, calling `on_token` for |
| 685 | /// each delta *as it arrives* -- [`Delta::Text`] for the answer, [`Delta::Reasoning`] |
| 686 | /// for the model's own working, which is never part of it. |
| 687 | /// Returns the full accumulated response and token usage when |
| 688 | /// the stream completes. |
| 689 | /// A stream that failed before emitting a token is retried; one that failed |
| 690 | /// after is not, because the caller has already been handed those tokens |
| 691 | /// and a fresh attempt would hand them over a second time. |
| 692 | pub async fn chat_stream( |
| 693 | &self, |
| 694 | messages: &[ChatMessage], |
| 695 | on_token: &mut impl FnMut(Delta<'_>), |
| 696 | ) -> Outcome<ChatResponse> { |
| 697 | let resp = res!(self.stream_turn(messages, None, on_token, false).await); |
| 698 | Ok(ChatResponse { |
| 699 | content: resp.content, |
| 700 | prompt_tokens: resp.prompt_tokens, |
| 701 | completion_tokens: resp.completion_tokens, |
| 702 | cached_tokens: resp.cached_tokens, |
| 703 | cost_usd: resp.cost_usd, |
| 704 | aborted: resp.aborted, |
| 705 | retries: resp.retries, |
| 706 | thinking: resp.thinking, |
| 707 | truncated: resp.truncated, |
| 708 | }) |
| 709 | } |
| 710 | |
| 711 | /// Streaming chat completion with tools enabled. |
| 712 | /// |
| 713 | /// Issues the request with `stream: true` and reconstructs the |
| 714 | /// assistant turn from the SSE deltas: text and reasoning are forwarded |
| 715 | /// to `on_token` as they arrive, each labelled (so the answer streams even |
| 716 | /// while tools are active, and so does the working that precedes it), and |
| 717 | /// any `tool_calls` fragments are accumulated across chunks into whole |
| 718 | /// calls (see [`StreamAcc`]). Returns the same |
| 719 | /// [`ChatOnceResponse`] shape as [`chat_once`](Self::chat_once). |
| 720 | /// |
| 721 | /// A 429, a 5xx or a dropped connection is retried with bounded exponential |
| 722 | /// backoff -- but only while the turn has produced nothing. Once a token, |
| 723 | /// or a fragment of a tool call, has reached the caller, a retry would |
| 724 | /// deliver it twice, so the partial and the error are surfaced instead. |
| 725 | /// Each retry announces itself through `on_token`, because a thirty-second |
| 726 | /// turn that silently becomes ninety is its own defect. |
| 727 | pub async fn chat_stream_tools( |
| 728 | &self, |
| 729 | messages: &[ChatMessage], |
| 730 | tools: Option<&str>, |
| 731 | on_token: &mut impl FnMut(Delta<'_>), |
| 732 | ) -> Outcome<ChatOnceResponse> { |
| 733 | self.stream_turn(messages, tools, on_token, true).await |
| 734 | } |
| 735 | |
| 736 | /// The one streamed turn both public streaming entry points run. |
| 737 | /// |
| 738 | /// Builds the request in whichever [`Dialect`] the endpoint speaks, drives |
| 739 | /// the SSE response through the matching accumulator, and applies the retry |
| 740 | /// policy. `notify` decides whether a retry announces itself through |
| 741 | /// `on_token`: the tool path does (a thirty-second turn that silently |
| 742 | /// becomes ninety is its own defect), the plain-chat path does not, because |
| 743 | /// its caller treats every token as answer text. |
| 744 | /// |
| 745 | /// # Arguments |
| 746 | /// * `messages` - The conversation so far. |
| 747 | /// * `tools` - A ready-made OpenAI-shaped tool array, translated for the |
| 748 | /// Anthropic dialect; `None` disables tools. |
| 749 | /// * `on_token` - Called with each delta as it arrives, labelled by kind. |
| 750 | /// * `notify` - Whether to announce a retry through `on_token`, as text. |
| 751 | async fn stream_turn( |
| 752 | &self, |
| 753 | messages: &[ChatMessage], |
| 754 | tools: Option<&str>, |
| 755 | on_token: &mut impl FnMut(Delta<'_>), |
| 756 | notify: bool, |
| 757 | ) -> Outcome<ChatOnceResponse> { |
| 758 | let images = res!(self.vision_guard(messages)); |
| 759 | let stripped = self.sighted(messages, images); |
| 760 | let mut body = self.build_body(stripped.as_deref().unwrap_or(messages), tools, true); |
| 761 | // Set once the pictures have been taken out and the turn tried again, so the retry |
| 762 | // happens at most once and a second failure is reported as itself. |
| 763 | let mut retried_blind = stripped.is_some(); |
| 764 | let mut waited = 0u64; |
| 765 | let mut retries = 0u32; |
| 766 | loop { |
| 767 | let mut acc = Acc::new(self.dialect); |
| 768 | let mut emitted = false; |
| 769 | let outcome = { |
| 770 | let mut sink = |data: &str| { |
| 771 | acc.ingest(data, &mut |d: Delta<'_>| { |
| 772 | // ONLY TEXT MAKES A TURN UNREPEATABLE. Reasoning already shown and |
| 773 | // then shown again reads as the model thinking twice, which is odd; |
| 774 | // an answer delivered twice is wrong. So a turn that has only |
| 775 | // reasoned so far is still safe to start over. |
| 776 | if matches!(d, Delta::Text(_)) { emitted = true; } |
| 777 | on_token(d); |
| 778 | }); |
| 779 | }; |
| 780 | self.stream_sse(&body, &mut sink).await |
| 781 | }; |
| 782 | // An `error` event on a 200 stream is the provider's own trouble |
| 783 | // arriving after the headers, so it is classified like a status code |
| 784 | // rather than read as a short answer. |
| 785 | let outcome = match outcome { |
| 786 | Ok(aborted) => match acc.stream_error() { |
| 787 | Some(e) if !emitted && !acc.has_output() => Err(e), |
| 788 | _ => Ok(aborted), |
| 789 | }, |
| 790 | Err(e) => Err(e), |
| 791 | }; |
| 792 | match outcome { |
| 793 | Ok(aborted) => { |
| 794 | let thinking = acc.take_thinking(); |
| 795 | let resp = acc.into_response(aborted, retries); |
| 796 | // The signed reasoning of a turn that asked for tools is held |
| 797 | // for the request that returns their results; see [`ThinkCarry`]. |
| 798 | if let Some(tc) = resp.tool_calls.first() { |
| 799 | self.carry_put(&tc.id, thinking); |
| 800 | } |
| 801 | return Ok(resp); |
| 802 | } |
| 803 | Err(e) => { |
| 804 | // Anything the caller has already seen -- streamed text, or a |
| 805 | // tool-call fragment that will become one -- makes this turn |
| 806 | // unrepeatable. |
| 807 | let started = emitted || acc.has_output(); |
| 808 | // A REFUSED PICTURE IS NOT A DEAD TURN. The provider would not take this |
| 809 | // request and it carried images, so the likeliest reason is the one thing |
| 810 | // in it a text model cannot read. Take them out, say so in their place, and |
| 811 | // send it again -- once. Only where nothing has been emitted: a turn the |
| 812 | // user has already seen tokens from cannot be started over. |
| 813 | if !started && !retried_blind && images > 0 { |
| 814 | retried_blind = true; |
| 815 | self.mark_blind(); |
| 816 | let text_only: Vec<ChatMessage> = |
| 817 | messages.iter() |
| 818 | .map(|m| m.with_content(m.content().without_images(Dropped::Unseeable))) |
| 819 | .collect(); |
| 820 | body = self.build_body(&text_only, tools, true); |
| 821 | if notify { |
| 822 | on_token(Delta::Text(&fmt!( |
| 823 | "\n[daimond: the model would not take {} image{}; asking again \ |
| 824 | without {} -- it cannot see]\n", |
| 825 | images, |
| 826 | if images == 1 { "" } else { "s" }, |
| 827 | if images == 1 { "it" } else { "them" }))); |
| 828 | } |
| 829 | continue; |
| 830 | } |
| 831 | if started || !e.retryable { |
| 832 | return Err(self.vision_error(e.crossed(), images)); |
| 833 | } |
| 834 | let delay = match self.retry.next_delay(retries, waited, e.after_ms) { |
| 835 | Some(d) => d, |
| 836 | None => return Err(self.vision_error(e.crossed(), images)), |
| 837 | }; |
| 838 | waited += delay; |
| 839 | retries += 1; |
| 840 | if notify { |
| 841 | on_token(Delta::Text(&fmt!( |
| 842 | "\n[daimond: {}; retrying in {}.{:01}s -- attempt {} of {}]\n", |
| 843 | e.reason, |
| 844 | delay / 1_000, |
| 845 | (delay % 1_000) / 100, |
| 846 | retries + 1, |
| 847 | self.retry.max_attempts))); |
| 848 | } |
| 849 | sleep_ms(delay).await; |
| 850 | } |
| 851 | } |
| 852 | } |
| 853 | } |
| 854 | |
| 855 | /// Non-streaming chat completion, optionally with tools. |
| 856 | /// |
| 857 | /// Returns the assistant content and any `tool_calls` the model |
| 858 | /// wants executed, plus token usage. Retained for callers that |
| 859 | /// prefer a single whole-response parse over streamed fragments. |
| 860 | pub async fn chat_once( |
| 861 | &self, |
| 862 | messages: &[ChatMessage], |
| 863 | tools: Option<&str>, |
| 864 | ) -> Outcome<ChatOnceResponse> { |
| 865 | let images = res!(self.vision_guard(messages)); |
| 866 | let stripped = self.sighted(messages, images); |
| 867 | let mut body = self.build_body(stripped.as_deref().unwrap_or(messages), tools, false); |
| 868 | let mut retried_blind = stripped.is_some(); |
| 869 | let mut waited = 0u64; |
| 870 | let mut retries = 0u32; |
| 871 | let raw = loop { |
| 872 | match self.do_request_full(&body).await { |
| 873 | Ok(r) => break r, |
| 874 | Err(e) => { |
| 875 | // Nothing streams on this path, so there is never a partial |
| 876 | // to protect -- only the classification matters. |
| 877 | // The picture retry, exactly as `stream_turn` does it and for the same |
| 878 | // reason; there is no emitted-tokens condition here because nothing has |
| 879 | // been shown to anybody yet. |
| 880 | if !retried_blind && images > 0 { |
| 881 | retried_blind = true; |
| 882 | self.mark_blind(); |
| 883 | let text_only: Vec<ChatMessage> = |
| 884 | messages.iter() |
| 885 | .map(|m| m.with_content(m.content().without_images(Dropped::Unseeable))) |
| 886 | .collect(); |
| 887 | body = self.build_body(&text_only, tools, false); |
| 888 | continue; |
| 889 | } |
| 890 | if !e.retryable { |
| 891 | return Err(self.vision_error(e.crossed(), images)); |
| 892 | } |
| 893 | let delay = match self.retry.next_delay(retries, waited, e.after_ms) { |
| 894 | Some(d) => d, |
| 895 | None => return Err(self.vision_error(e.crossed(), images)), |
| 896 | }; |
| 897 | waited += delay; |
| 898 | retries += 1; |
| 899 | sleep_ms(delay).await; |
| 900 | } |
| 901 | } |
| 902 | }; |
| 903 | let (content, tool_calls, use_, thinking) = match self.dialect { |
| 904 | Dialect::OpenAi => { |
| 905 | let (c, t, u) = parse_full_response(&raw); |
| 906 | (c, t, u, Vec::new()) |
| 907 | } |
| 908 | Dialect::Anthropic => parse_anthropic_response(&raw), |
| 909 | }; |
| 910 | let thinking_text = thinking.iter() |
| 911 | .filter_map(|b| extract_json_string(b, "thinking")) |
| 912 | .filter(|s| !s.is_empty()) |
| 913 | .collect::<Vec<String>>() |
| 914 | .join("\n"); |
| 915 | if let Some(tc) = tool_calls.first() { |
| 916 | self.carry_put(&tc.id, thinking); |
| 917 | } |
| 918 | // Read from the whole body, in whichever dialect it came back in. |
| 919 | let truncated = match self.dialect { |
| 920 | Dialect::OpenAi => openai_truncated(&raw), |
| 921 | Dialect::Anthropic => anthropic_truncated(&raw), |
| 922 | }; |
| 923 | Ok(ChatOnceResponse { |
| 924 | content, |
| 925 | tool_calls, |
| 926 | prompt_tokens: use_.prompt, |
| 927 | completion_tokens: use_.completion, |
| 928 | cached_tokens: use_.cached, |
| 929 | cost_usd: use_.cost_usd, |
| 930 | aborted: false, |
| 931 | retries, |
| 932 | thinking: thinking_text, |
| 933 | truncated, |
| 934 | }) |
| 935 | } |
| 936 | |
| 937 | /// Refuse, before the request is built, to send an image to a model known not to see. |
| 938 | /// |
| 939 | /// Returns how many images the conversation carries, which is zero on nearly every turn and |
| 940 | /// is what [`vision_error`](Self::vision_error) needs afterwards. |
| 941 | /// |
| 942 | /// The refusal names the model, because that is the fact the user has to act on: the app |
| 943 | /// cannot tell them which model to pick, but it can tell them the one they picked is the |
| 944 | /// reason nothing was looked at. A provider's own 400 says none of that -- at best it names |
| 945 | /// a content type. |
| 946 | /// |
| 947 | /// # Arguments |
| 948 | /// * `messages` - The conversation about to be sent. |
| 949 | /// Whether this endpoint has already been caught refusing pictures. |
| 950 | fn is_blind(&self) -> bool { |
| 951 | #[cfg(not(target_arch = "wasm32"))] |
| 952 | { self.blind.load(std::sync::atomic::Ordering::Relaxed) } |
| 953 | #[cfg(target_arch = "wasm32")] |
| 954 | { self.blind.get() } |
| 955 | } |
| 956 | |
| 957 | /// Record that it does, so no later turn pays to find out again. |
| 958 | fn mark_blind(&self) { |
| 959 | #[cfg(not(target_arch = "wasm32"))] |
| 960 | { self.blind.store(true, std::sync::atomic::Ordering::Relaxed) } |
| 961 | #[cfg(target_arch = "wasm32")] |
| 962 | { self.blind.set(true) } |
| 963 | } |
| 964 | |
| 965 | /// May a picture be put in front of this endpoint? |
| 966 | /// |
| 967 | /// Both halves of what is known, and nothing else: the deny-list [`model_can_see`] before any |
| 968 | /// request has been sent, and the refusal [`is_blind`](Self::is_blind) records after one has |
| 969 | /// been turned away. There is no third source -- no `vision` flag is published by anybody -- |
| 970 | /// so a model released after this line was written is taken to see until it says otherwise. |
| 971 | pub fn can_take_images(&self) -> bool { |
| 972 | model_can_see(&self.model) && !self.is_blind() |
| 973 | } |
| 974 | |
| 975 | /// The conversation as it must be sent: whole, or with the pictures turned into words when |
| 976 | /// this endpoint has been caught refusing them. |
| 977 | /// |
| 978 | /// Returns `None` when nothing needs changing, so the ordinary turn copies no messages. |
| 979 | fn sighted<'m>(&self, messages: &'m [ChatMessage], images: usize) |
| 980 | -> Option<Vec<ChatMessage>> |
| 981 | { |
| 982 | if images == 0 || !self.is_blind() { |
| 983 | return None; |
| 984 | } |
| 985 | let _ = messages.len(); |
| 986 | Some(messages.iter() |
| 987 | .map(|m| m.with_content(m.content().without_images(Dropped::Unseeable))) |
| 988 | .collect()) |
| 989 | } |
| 990 | |
| 991 | fn vision_guard(&self, messages: &[ChatMessage]) -> Outcome<usize> { |
| 992 | let images: usize = messages.iter().map(|m| m.content().images().count()).sum(); |
| 993 | // A refusal already seen is not an error any more: the pictures come out and the turn |
| 994 | // goes ahead. Refusing here instead would leave a conversation that carries one image |
| 995 | // permanently unable to take a turn -- which is what happened to a real Diamond on |
| 996 | // 2026-08-13, where a cover read into the daimon's history bricked every later steer. |
| 997 | if images == 0 || model_can_see(&self.model) || self.is_blind() { |
| 998 | return Ok(images); |
| 999 | } |
| 1000 | Err(err!( |
| 1001 | "The model '{}' cannot see. This turn carries {} image{} and that model takes text \ |
| 1002 | only, so it would answer as though nothing had been shown to it. Choose a model with \ |
| 1003 | vision and read the file again.", |
| 1004 | self.model, images, if images == 1 { "" } else { "s" }; |
| 1005 | Invalid, Input, Unimplemented)) |
| 1006 | } |
| 1007 | |
| 1008 | /// Say what a failed request that carried images most likely failed for. |
| 1009 | /// |
| 1010 | /// [`vision_guard`](Self::vision_guard) can only refuse a model it has been told about, and no |
| 1011 | /// list of those is ever complete -- Daimond takes an arbitrary endpoint and an arbitrary |
| 1012 | /// model id. So the second half of the answer is here: when a turn that carried images comes |
| 1013 | /// back refused, and the provider's words are about images, the model is named and the reason |
| 1014 | /// is said plainly, with the provider's own sentence kept after it rather than replaced. |
| 1015 | /// |
| 1016 | /// A failure with no images in the turn, or whose text says nothing about them, is returned |
| 1017 | /// exactly as it arrived. Guessing at an unrelated failure would be worse than saying nothing. |
| 1018 | /// |
| 1019 | /// # Arguments |
| 1020 | /// * `e` - The error the provider produced. |
| 1021 | /// * `images` - How many images the refused turn carried. |
| 1022 | fn vision_error(&self, e: Error<ErrTag>, images: usize) -> Error<ErrTag> { |
| 1023 | if images == 0 { |
| 1024 | return e; |
| 1025 | } |
| 1026 | let low = fmt!("{}", e).to_lowercase(); |
| 1027 | let about_images = [ |
| 1028 | "image", "vision", "multimodal", "media_type", "media type", "image_url", |
| 1029 | ].iter().any(|m| low.contains(m)); |
| 1030 | if !about_images { |
| 1031 | return e; |
| 1032 | } |
| 1033 | err!( |
| 1034 | "The model '{}' could not be shown the {} image{} in this turn -- it appears not to \ |
| 1035 | see. Choose a model with vision. The provider said: {}", |
| 1036 | self.model, images, if images == 1 { "" } else { "s" }, e; |
| 1037 | Invalid, Input, Unimplemented) |
| 1038 | } |
| 1039 | |
| 1040 | /// Build the JSON request body for the OpenAI-compatible API. |
| 1041 | /// |
| 1042 | /// `tools` (if present) is a ready-made JSON array injected as the |
| 1043 | /// `tools` field with `tool_choice: auto`. `stream` toggles SSE |
| 1044 | /// streaming and usage reporting. |
| 1045 | /// |
| 1046 | /// Messages chosen by [`cache_breakpoints`](Self::cache_breakpoints) carry an |
| 1047 | /// Anthropic `cache_control` marker. Providers that cache automatically |
| 1048 | /// ignore it; Claude models, which do not, need it or an agentic session |
| 1049 | /// re-pays full price for the same prompt on every round. |
| 1050 | fn build_body(&self, messages: &[ChatMessage], tools: Option<&str>, stream: bool) -> String { |
| 1051 | match self.dialect { |
| 1052 | Dialect::OpenAi => self.build_openai_body(messages, tools, stream), |
| 1053 | Dialect::Anthropic => self.build_anthropic_body(messages, tools, stream), |
| 1054 | } |
| 1055 | } |
| 1056 | |
| 1057 | /// The OpenAI-compatible request body. |
| 1058 | /// |
| 1059 | /// See [`build_body`](Self::build_body) for the shared contract. |
| 1060 | /// |
| 1061 | /// One thing here is not a straight translation of the message list. A `tool` message on this |
| 1062 | /// side may hold text and nothing else -- the content-part union for that role has no image |
| 1063 | /// member -- so an image returned by a tool cannot ride in the reply that returned it. It is |
| 1064 | /// re-homed instead: the tool reply carries its text, and the images from a whole RUN of tool |
| 1065 | /// replies are emitted together in one `user` message directly after the run. After the run |
| 1066 | /// and not between the replies, because a run of `tool` messages answers one assistant turn |
| 1067 | /// and a message of another role wedged inside it is a conversation the API rejects. |
| 1068 | fn build_openai_body(&self, messages: &[ChatMessage], tools: Option<&str>, stream: bool) |
| 1069 | -> String |
| 1070 | { |
| 1071 | let marks = self.cache_breakpoints(messages, tools); |
| 1072 | let mut out = String::with_capacity(1024); |
| 1073 | out.push('{'); |
| 1074 | out.push_str(&fmt!("\"model\":\"{}\",", self.model)); |
| 1075 | out.push_str("\"messages\":["); |
| 1076 | let mut first = true; |
| 1077 | // Images lifted out of the tool replies of the run now being emitted. |
| 1078 | let mut carried: Vec<String> = Vec::new(); |
| 1079 | for (i, msg) in messages.iter().enumerate() { |
| 1080 | if !matches!(msg, ChatMessage::Tool { .. }) && !carried.is_empty() { |
| 1081 | if !first { out.push(','); } |
| 1082 | out.push_str(&tool_image_message(&carried)); |
| 1083 | carried.clear(); |
| 1084 | first = false; |
| 1085 | } |
| 1086 | if let ChatMessage::Tool { content, .. } = msg { |
| 1087 | for img in content.images() { |
| 1088 | carried.push(fmt!( |
| 1089 | "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:{};base64,{}\"}}}}", |
| 1090 | img.media.mime(), img.base64())); |
| 1091 | } |
| 1092 | } |
| 1093 | if !first { out.push(','); } |
| 1094 | first = false; |
| 1095 | if marks.contains(&i) { |
| 1096 | out.push_str(&message_to_json_cached(msg, &self.open_folds.borrow())); |
| 1097 | } else { |
| 1098 | out.push_str(&message_to_json(msg, &self.open_folds.borrow())); |
| 1099 | } |
| 1100 | } |
| 1101 | if !carried.is_empty() { |
| 1102 | if !first { out.push(','); } |
| 1103 | out.push_str(&tool_image_message(&carried)); |
| 1104 | } |
| 1105 | out.push_str("],"); |
| 1106 | if let Some(t) = tools { |
| 1107 | out.push_str(&fmt!("\"tools\":{},", t)); |
| 1108 | out.push_str("\"tool_choice\":\"auto\","); |
| 1109 | } |
| 1110 | if stream { |
| 1111 | out.push_str("\"stream\":true,"); |
| 1112 | out.push_str("\"stream_options\":{\"include_usage\":true},"); |
| 1113 | } else { |
| 1114 | out.push_str("\"stream\":false,"); |
| 1115 | } |
| 1116 | out.push_str(&fmt!("\"max_tokens\":{}", self.max_tokens)); |
| 1117 | out.push('}'); |
| 1118 | out |
| 1119 | } |
| 1120 | |
| 1121 | /// Streaming body (no tools). Kept for the pure-chat path's unit test, |
| 1122 | /// which is the only caller now that both paths share [`stream_turn`](Self::stream_turn). |
| 1123 | #[cfg(test)] |
| 1124 | fn build_request_body(&self, messages: &[ChatMessage]) -> String { |
| 1125 | self.build_body(messages, None, true) |
| 1126 | } |
| 1127 | |
| 1128 | /// The Anthropic Messages API request body. |
| 1129 | /// |
| 1130 | /// Four things differ from the OpenAI shape, and each one is why this |
| 1131 | /// could not be a couple of extra fields on the other builder: |
| 1132 | /// |
| 1133 | /// * the system prompt is a top-level `system`, not a message, so every |
| 1134 | /// system message is hoisted out and joined; |
| 1135 | /// * content is an array of typed blocks, so a `cache_control` marker has |
| 1136 | /// somewhere to live without changing the message's shape; |
| 1137 | /// * a tool call is a `tool_use` block on the assistant turn and its result |
| 1138 | /// a `tool_result` block on the *user* turn, so a run of tool results |
| 1139 | /// coalesces into one user message rather than becoming several; |
| 1140 | /// * thinking blocks precede the `tool_use` blocks they were generated |
| 1141 | /// beside, and must be handed back unmodified -- see [`ThinkCarry`]. |
| 1142 | /// |
| 1143 | /// Thinking is requested only for the models that take the adaptive form; |
| 1144 | /// see [`model_takes_adaptive_thinking`]. |
| 1145 | fn build_anthropic_body(&self, messages: &[ChatMessage], tools: Option<&str>, stream: bool) |
| 1146 | -> String |
| 1147 | { |
| 1148 | let marks = self.cache_breakpoints(messages, tools); |
| 1149 | let thinks = model_takes_adaptive_thinking(&self.model); |
| 1150 | let mut out = String::with_capacity(1024); |
| 1151 | out.push('{'); |
| 1152 | out.push_str(&fmt!("\"model\":\"{}\",", self.model)); |
| 1153 | out.push_str(&fmt!("\"max_tokens\":{},", self.anthropic_max_tokens(thinks, stream))); |
| 1154 | |
| 1155 | // The system prompt, hoisted. Several system messages become one |
| 1156 | // block: the API takes a single system field, and the model reads a |
| 1157 | // joined prompt exactly as it read separate messages. |
| 1158 | let sys: Vec<String> = messages.iter().filter_map(|m| match m { |
| 1159 | ChatMessage::System { content } => Some(content.as_text().into_owned()), |
| 1160 | _ => None, |
| 1161 | }).collect(); |
| 1162 | if !sys.is_empty() { |
| 1163 | let mark = messages.iter().enumerate().any(|(i, m)| |
| 1164 | matches!(m, ChatMessage::System { .. }) && marks.contains(&i)); |
| 1165 | out.push_str("\"system\":[{\"type\":\"text\",\"text\":\""); |
| 1166 | out.push_str(&json_escape(&sys.join("\n\n"))); |
| 1167 | out.push('"'); |
| 1168 | if mark { out.push_str(",\"cache_control\":{\"type\":\"ephemeral\"}"); } |
| 1169 | out.push_str("}],"); |
| 1170 | } |
| 1171 | |
| 1172 | // The conversation. `pending` holds the content blocks of the user |
| 1173 | // message being assembled, so consecutive tool results land in one |
| 1174 | // message rather than in several the API would reject. |
| 1175 | let mut msgs: Vec<String> = Vec::new(); |
| 1176 | let mut pending: Vec<String> = Vec::new(); |
| 1177 | for (i, msg) in messages.iter().enumerate() { |
| 1178 | match msg { |
| 1179 | ChatMessage::System { .. } => {} |
| 1180 | ChatMessage::User { content } => { |
| 1181 | // An empty text block is rejected outright, where the |
| 1182 | // OpenAI side simply carries the empty string through. |
| 1183 | pending.extend(anthropic_blocks(content, marks.contains(&i))); |
| 1184 | } |
| 1185 | ChatMessage::Tool { tool_call_id, content } => { |
| 1186 | // A `tool_result` takes either a string or an array of blocks, and this side |
| 1187 | // -- unlike OpenAI's -- takes an image among them. So a screenshot stays |
| 1188 | // attached to the call that produced it rather than being re-homed. |
| 1189 | if content.has_image() { |
| 1190 | let blocks = anthropic_blocks(content, false); |
| 1191 | pending.push(fmt!( |
| 1192 | "{{\"type\":\"tool_result\",\"tool_use_id\":\"{}\",\"content\":[{}]}}", |
| 1193 | json_escape(tool_call_id), blocks.join(","))); |
| 1194 | } else { |
| 1195 | pending.push(fmt!( |
| 1196 | "{{\"type\":\"tool_result\",\"tool_use_id\":\"{}\",\"content\":\"{}\"}}", |
| 1197 | json_escape(tool_call_id), json_escape(&content.as_text()))); |
| 1198 | } |
| 1199 | } |
| 1200 | ChatMessage::Assistant { content, tool_calls } => { |
| 1201 | if !pending.is_empty() { |
| 1202 | msgs.push(fmt!("{{\"role\":\"user\",\"content\":[{}]}}", pending.join(","))); |
| 1203 | pending.clear(); |
| 1204 | } |
| 1205 | let mut blocks: Vec<String> = Vec::new(); |
| 1206 | // The reasoning that led to these tool calls, first and |
| 1207 | // verbatim. Absent for a turn that asked for nothing. |
| 1208 | if let Some(tc) = tool_calls.first() { |
| 1209 | blocks.extend(self.carry_get(&tc.id)); |
| 1210 | } |
| 1211 | // Assistant turns are the model's own words; an image cannot appear in one. |
| 1212 | let said = content.as_text(); |
| 1213 | if !said.is_empty() { |
| 1214 | // The same fold strip the OpenAI side applies. Applied at one site and |
| 1215 | // not the other, the same conversation would cost different amounts |
| 1216 | // through different endpoints, silently. |
| 1217 | let folded = strip_folds(&said, &self.open_folds.borrow()); |
| 1218 | blocks.push(text_block(folded.as_deref().unwrap_or(&said), false)); |
| 1219 | } |
| 1220 | for tc in tool_calls { |
| 1221 | let stripped = strip_said(&tc.name, &tc.arguments, self.fold_open(&tc.id)); |
| 1222 | let raw = stripped.as_deref().unwrap_or(&tc.arguments); |
| 1223 | let args = if raw.trim_start().starts_with('{') { |
| 1224 | raw |
| 1225 | } else { |
| 1226 | "{}" |
| 1227 | }; |
| 1228 | blocks.push(fmt!( |
| 1229 | "{{\"type\":\"tool_use\",\"id\":\"{}\",\"name\":\"{}\",\"input\":{}}}", |
| 1230 | json_escape(&tc.id), json_escape(&tc.name), args)); |
| 1231 | } |
| 1232 | // An assistant turn with no content at all is not a message |
| 1233 | // the API will take, and it says nothing the model needs. |
| 1234 | if !blocks.is_empty() { |
| 1235 | msgs.push(fmt!("{{\"role\":\"assistant\",\"content\":[{}]}}", blocks.join(","))); |
| 1236 | } |
| 1237 | } |
| 1238 | } |
| 1239 | } |
| 1240 | if !pending.is_empty() { |
| 1241 | msgs.push(fmt!("{{\"role\":\"user\",\"content\":[{}]}}", pending.join(","))); |
| 1242 | } |
| 1243 | out.push_str(&fmt!("\"messages\":[{}],", msgs.join(","))); |
| 1244 | |
| 1245 | if let Some(t) = tools { |
| 1246 | out.push_str(&fmt!("\"tools\":{},", openai_tools_to_anthropic(t))); |
| 1247 | out.push_str("\"tool_choice\":{\"type\":\"auto\"},"); |
| 1248 | } |
| 1249 | if thinks { |
| 1250 | // `display` defaults to `omitted` on every current model, which |
| 1251 | // streams thinking blocks whose text is empty. Summarised costs |
| 1252 | // the same -- the billed thinking is the full reasoning either way |
| 1253 | // -- and is the difference between a visible pause and a silent one. |
| 1254 | out.push_str("\"thinking\":{\"type\":\"adaptive\",\"display\":\"summarized\"},"); |
| 1255 | } |
| 1256 | out.push_str(&fmt!("\"stream\":{}", if stream { "true" } else { "false" })); |
| 1257 | out.push('}'); |
| 1258 | out |
| 1259 | } |
| 1260 | |
| 1261 | /// The output cap for a Messages API request. |
| 1262 | /// |
| 1263 | /// On the OpenAI side `max_tokens` bounds the answer. On this side it |
| 1264 | /// bounds the reasoning *and* the answer together -- thinking is billed as |
| 1265 | /// output and counts against the same cap -- so a figure chosen for the |
| 1266 | /// first meaning truncates under the second, and the app's is 4096: enough |
| 1267 | /// for an answer, not enough for a hard problem thought through first. A |
| 1268 | /// floor is applied rather than the configured value being used, because |
| 1269 | /// that value is an internal default and not something a user chose. |
| 1270 | /// |
| 1271 | /// It applies only where both halves of the reason hold: a model that |
| 1272 | /// actually thinks, and a streamed request. The one-shot path keeps the |
| 1273 | /// configured cap, since a large one there risks an HTTP timeout on a |
| 1274 | /// connection with nothing arriving on it. |
| 1275 | /// |
| 1276 | /// # Arguments |
| 1277 | /// * `thinks` - Whether this request asks for thinking. |
| 1278 | /// * `stream` - Whether the response is streamed. |
| 1279 | fn anthropic_max_tokens(&self, thinks: bool, stream: bool) -> u32 { |
| 1280 | if thinks && stream { |
| 1281 | self.max_tokens.max(THINKING_MIN_MAX_TOKENS) |
| 1282 | } else { |
| 1283 | self.max_tokens |
| 1284 | } |
| 1285 | } |
| 1286 | |
| 1287 | /// The headers this request needs beyond `Host` and `Content-Length`. |
| 1288 | /// |
| 1289 | /// The two dialects do not merely differ in the name of the auth header: |
| 1290 | /// Anthropic wants `x-api-key` plus a pinned API version, and refuses a |
| 1291 | /// bearer token. `browser` adds the header that makes Anthropic's edge |
| 1292 | /// answer a cross-origin `fetch` at all -- the same one the official |
| 1293 | /// TypeScript SDK sends for `dangerouslyAllowBrowser`. It is sent only |
| 1294 | /// from the browser transport, where it is the difference between the app |
| 1295 | /// working and CORS refusing it. |
| 1296 | /// |
| 1297 | /// # Arguments |
| 1298 | /// * `browser` - Whether the request is being made from a browser. |
| 1299 | fn auth_headers(&self, browser: bool) -> Vec<(&'static str, String)> { |
| 1300 | let mut out = vec![("Content-Type", "application/json".to_string())]; |
| 1301 | match self.dialect { |
| 1302 | Dialect::OpenAi => { |
| 1303 | out.push(("Authorization", fmt!("Bearer {}", self.api_key))); |
| 1304 | } |
| 1305 | Dialect::Anthropic => { |
| 1306 | out.push(("x-api-key", self.api_key.clone())); |
| 1307 | out.push(("anthropic-version", ANTHROPIC_VERSION.to_string())); |
| 1308 | if browser { |
| 1309 | out.push(("anthropic-dangerous-direct-browser-access", "true".to_string())); |
| 1310 | } |
| 1311 | } |
| 1312 | } |
| 1313 | out |
| 1314 | } |
| 1315 | |
| 1316 | /// Hold this turn's thinking blocks against the tool call they accompany. |
| 1317 | /// |
| 1318 | /// A poisoned lock loses the carry rather than the turn: the next request |
| 1319 | /// then goes without thinking blocks, which the API answers by quietly |
| 1320 | /// disabling thinking for it. That is a worse answer, not a broken one, |
| 1321 | /// and it is the right trade against failing a turn the user is watching. |
| 1322 | /// |
| 1323 | /// # Arguments |
| 1324 | /// * `id` - The first tool-call id of the turn the blocks came from. |
| 1325 | /// * `blocks` - The serialised blocks, in the order the model produced them. |
| 1326 | fn carry_put(&self, id: &str, blocks: Vec<String>) { |
| 1327 | if id.is_empty() || blocks.is_empty() { |
| 1328 | return; |
| 1329 | } |
| 1330 | let go = |c: &mut ThinkCarry| { |
| 1331 | // A retried round re-reports the same id; the newer blocks replace |
| 1332 | // the older rather than sitting beside them. |
| 1333 | c.turns.retain(|(k, _)| k != id); |
| 1334 | c.turns.push((id.to_string(), blocks.clone())); |
| 1335 | if c.turns.len() > CARRY_MAX_TURNS { |
| 1336 | let drop = c.turns.len() - CARRY_MAX_TURNS; |
| 1337 | c.turns.drain(..drop); |
| 1338 | } |
| 1339 | }; |
| 1340 | #[cfg(not(target_arch = "wasm32"))] |
| 1341 | { if let Ok(mut g) = self.think.lock() { go(&mut g); } } |
| 1342 | #[cfg(target_arch = "wasm32")] |
| 1343 | { go(&mut self.think.borrow_mut()); } |
| 1344 | } |
| 1345 | |
| 1346 | /// Is this `say` call's fold open on screen? |
| 1347 | fn fold_open(&self, id: &str) -> bool { |
| 1348 | !id.is_empty() && self.open_folds.borrow().contains(id) |
| 1349 | } |
| 1350 | |
| 1351 | /// The open folds, copied out. |
| 1352 | /// |
| 1353 | /// A COPY and not a borrow: the sizing path holds this across the awaits of a fold, and the |
| 1354 | /// page may set the folds again at any point in between -- a `RefCell` borrow still live at |
| 1355 | /// that moment would panic. The set holds one short id per fold on screen. |
| 1356 | pub fn open_folds(&self) -> OpenSet { |
| 1357 | self.open_folds.borrow().clone() |
| 1358 | } |
| 1359 | |
| 1360 | /// Replace the set of open folds, from the page, before a request goes out. |
| 1361 | /// |
| 1362 | /// REPLACED and not added to: a fold the user has since closed must leave the payload, and an |
| 1363 | /// accumulating set could only ever grow. |
| 1364 | pub fn set_open_folds(&self, ids: Vec<String>) { |
| 1365 | let mut f = self.open_folds.borrow_mut(); |
| 1366 | f.clear(); |
| 1367 | for id in ids { |
| 1368 | f.insert(id); |
| 1369 | } |
| 1370 | } |
| 1371 | |
| 1372 | /// The thinking blocks held for `id`, or none when no held turn produced |
| 1373 | /// that call (or a lock could not be taken; see |
| 1374 | /// [`carry_put`](Self::carry_put)). |
| 1375 | /// |
| 1376 | /// # Arguments |
| 1377 | /// * `id` - The first tool-call id of the assistant turn being serialised. |
| 1378 | fn carry_get(&self, id: &str) -> Vec<String> { |
| 1379 | if id.is_empty() { |
| 1380 | return Vec::new(); |
| 1381 | } |
| 1382 | let find = |c: &ThinkCarry| c.turns.iter() |
| 1383 | .find(|(k, _)| k == id) |
| 1384 | .map(|(_, b)| b.clone()) |
| 1385 | .unwrap_or_default(); |
| 1386 | #[cfg(not(target_arch = "wasm32"))] |
| 1387 | { |
| 1388 | match self.think.lock() { |
| 1389 | Ok(g) => find(&g), |
| 1390 | Err(_) => Vec::new(), |
| 1391 | } |
| 1392 | } |
| 1393 | #[cfg(target_arch = "wasm32")] |
| 1394 | { find(&self.think.borrow()) } |
| 1395 | } |
| 1396 | |
| 1397 | /// Which message indices get an Anthropic prompt-cache breakpoint. |
| 1398 | /// |
| 1399 | /// Two at most, both placed at a boundary between what stays the same and |
| 1400 | /// what changes: |
| 1401 | /// |
| 1402 | /// * the last system message, which with the tool definitions rendered ahead |
| 1403 | /// of it is the largest block that never varies within a session; |
| 1404 | /// * the last user message, which is the tip of the settled conversation -- |
| 1405 | /// the next turn reads everything before it back out of the cache. |
| 1406 | /// |
| 1407 | /// Nothing is marked for a model that does not honour the marker, and |
| 1408 | /// nothing is marked when the prefix is too short to be cacheable at all. |
| 1409 | /// Assistant and tool messages are deliberately left unmarked: the array |
| 1410 | /// content form they would need is the one an OpenAI-compatible router is |
| 1411 | /// least certain to carry through, and a rejected body loses the whole turn. |
| 1412 | fn cache_breakpoints(&self, messages: &[ChatMessage], tools: Option<&str>) -> Vec<usize> { |
| 1413 | let mut marks = Vec::new(); |
| 1414 | if !model_caches_on_request(&self.model) { |
| 1415 | return marks; |
| 1416 | } |
| 1417 | // The prefix at each message, in characters, standing in for tokens. |
| 1418 | let mut prefix = tools.map(|t| t.len()).unwrap_or(0); |
| 1419 | let mut sys = None; |
| 1420 | let mut usr = None; |
| 1421 | for (i, msg) in messages.iter().enumerate() { |
| 1422 | prefix += message_len(msg); |
| 1423 | if prefix < CACHE_MIN_PREFIX_CHARS { |
| 1424 | continue; |
| 1425 | } |
| 1426 | match msg { |
| 1427 | ChatMessage::System { .. } => sys = Some(i), |
| 1428 | ChatMessage::User { .. } => usr = Some(i), |
| 1429 | _ => {} |
| 1430 | } |
| 1431 | } |
| 1432 | if let Some(i) = sys { marks.push(i); } |
| 1433 | if let Some(i) = usr { |
| 1434 | if Some(i) != sys { marks.push(i); } |
| 1435 | } |
| 1436 | marks |
| 1437 | } |
| 1438 | |
| 1439 | /// Connect, TLS-handshake, send the request, and consume the |
| 1440 | /// response headers. Returns the stream positioned at the body |
| 1441 | /// start plus whether the body uses chunked transfer encoding. |
| 1442 | /// Errors on a non-200 status (with body detail). |
| 1443 | /// |
| 1444 | /// Every failure here is classified but none is retried: retrying belongs to |
| 1445 | /// the public call, which is the only layer that knows whether anything has |
| 1446 | /// already reached the caller and is the only one that can say so. |
| 1447 | #[cfg(not(target_arch = "wasm32"))] |
| 1448 | async fn open( |
| 1449 | &self, |
| 1450 | body: &str, |
| 1451 | ) |
| 1452 | -> Result<(tokio_rustls::client::TlsStream<tokio::net::TcpStream>, bool), TransportErr> |
| 1453 | { |
| 1454 | use tokio_rustls::TlsConnector; |
| 1455 | use tokio::net::TcpStream; |
| 1456 | |
| 1457 | let body_bytes = body.as_bytes(); |
| 1458 | |
| 1459 | let mut request = String::with_capacity(512 + body_bytes.len()); |
| 1460 | request.push_str(&fmt!("POST {} HTTP/1.1\r\n", self.path)); |
| 1461 | request.push_str(&fmt!("Host: {}\r\n", self.host)); |
| 1462 | for (name, value) in self.auth_headers(false) { |
| 1463 | request.push_str(&fmt!("{}: {}\r\n", name, value)); |
| 1464 | } |
| 1465 | request.push_str(&fmt!("Content-Length: {}\r\n", body_bytes.len())); |
| 1466 | request.push_str("Connection: close\r\n"); |
| 1467 | request.push_str("\r\n"); |
| 1468 | |
| 1469 | // A connection that never came up carries no partial answer, so every |
| 1470 | // failure from here to the status line is worth another attempt. |
| 1471 | let tcp = match TcpStream::connect((self.host.as_str(), self.port)).await { |
| 1472 | Ok(s) => s, |
| 1473 | Err(e) => return Err(TransportErr::transient(fmt!("could not reach {}", self.host), err!(e, |
| 1474 | "LLM: TCP connect to {}:{} failed.", self.host, self.port; |
| 1475 | IO, Network, Init))), |
| 1476 | }; |
| 1477 | let server_name = match tokio_rustls::rustls::pki_types::ServerName::try_from(self.host.clone()) { |
| 1478 | Ok(n) => n, |
| 1479 | // A name that will not parse will not parse next time either. |
| 1480 | Err(e) => return Err(TransportErr::fatal(fmt!("invalid server name '{}'", self.host), err!(e, |
| 1481 | "LLM: invalid server name '{}'.", self.host; |
| 1482 | IO, Network, Invalid, Input))), |
| 1483 | }; |
| 1484 | let connector = TlsConnector::from(self.tls_config.clone()); |
| 1485 | let mut stream = match connector.connect(server_name, tcp).await { |
| 1486 | Ok(s) => s, |
| 1487 | Err(e) => return Err(TransportErr::transient(fmt!("TLS handshake with {} failed", self.host), err!(e, |
| 1488 | "LLM: TLS handshake to {} failed.", self.host; |
| 1489 | IO, Network, Init))), |
| 1490 | }; |
| 1491 | |
| 1492 | let mut req = Vec::with_capacity(request.as_bytes().len() + body_bytes.len()); |
| 1493 | req.extend_from_slice(request.as_bytes()); |
| 1494 | req.extend_from_slice(body_bytes); |
| 1495 | if let Err(e) = stream.write_all(&req).await { |
| 1496 | return Err(TransportErr::transient("could not send the request".to_string(), err!(e, |
| 1497 | "LLM: write request failed."; IO, Network, Wire, Write))); |
| 1498 | } |
| 1499 | if let Err(e) = stream.flush().await { |
| 1500 | return Err(TransportErr::transient("could not send the request".to_string(), err!(e, |
| 1501 | "LLM: flush failed."; IO, Network, Wire, Write))); |
| 1502 | } |
| 1503 | |
| 1504 | // Read headers byte-by-byte until \r\n\r\n. |
| 1505 | let mut hdr_buf = Vec::with_capacity(2048); |
| 1506 | let mut byte = [0u8; 1]; |
| 1507 | loop { |
| 1508 | match stream.read(&mut byte).await { |
| 1509 | Ok(0) => break, |
| 1510 | Ok(_) => { |
| 1511 | hdr_buf.push(byte[0]); |
| 1512 | if hdr_buf.ends_with(b"\r\n\r\n") { break; } |
| 1513 | } |
| 1514 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => break, |
| 1515 | Err(e) => return Err(TransportErr::transient("the provider closed before replying".to_string(), err!(e, |
| 1516 | "LLM: read headers failed."; IO, Network, Wire, Read))), |
| 1517 | } |
| 1518 | } |
| 1519 | |
| 1520 | let headers_str = String::from_utf8_lossy(&hdr_buf); |
| 1521 | let is_chunked = headers_str |
| 1522 | .to_ascii_lowercase() |
| 1523 | .contains("transfer-encoding: chunked"); |
| 1524 | |
| 1525 | let status_line = headers_str.lines().next().unwrap_or(""); |
| 1526 | let status = status_code(status_line).unwrap_or(0); |
| 1527 | if status != 200 { |
| 1528 | let mut err_body = Vec::new(); |
| 1529 | let mut chunk = [0u8; 4096]; |
| 1530 | loop { |
| 1531 | match stream.read(&mut chunk).await { |
| 1532 | Ok(0) => break, |
| 1533 | Ok(n) => err_body.extend_from_slice(&chunk[..n]), |
| 1534 | Err(_) => break, |
| 1535 | } |
| 1536 | } |
| 1537 | let err_msg = String::from_utf8_lossy(&err_body); |
| 1538 | let err = err!( |
| 1539 | "LLM: HTTP error: {} | {}", status_line, clip_bytes(&err_msg, ERR_BODY_BYTES); |
| 1540 | IO, Network, Wire, Read); |
| 1541 | // A 429 or a 5xx is the provider saying "not now"; a 400 is this |
| 1542 | // request being wrong, and sending it again only costs money. |
| 1543 | let reason = fmt!("the provider returned HTTP {}", status); |
| 1544 | return Err(if status_retryable(status) { |
| 1545 | let after = header_value(&headers_str, "retry-after") |
| 1546 | .and_then(|v| parse_retry_after(&v)); |
| 1547 | TransportErr::transient(reason, err).after(after) |
| 1548 | } else { |
| 1549 | TransportErr::fatal(reason, err) |
| 1550 | }); |
| 1551 | } |
| 1552 | |
| 1553 | Ok((stream, is_chunked)) |
| 1554 | } |
| 1555 | |
| 1556 | /// Perform a non-streaming request and return the full response |
| 1557 | /// body as one string. Lines are concatenated (JSON does not need |
| 1558 | /// the newlines), dechunking transparently. |
| 1559 | #[cfg(not(target_arch = "wasm32"))] |
| 1560 | async fn do_request_full( |
| 1561 | &self, |
| 1562 | body: &str, |
| 1563 | ) -> Result<String, TransportErr> { |
| 1564 | let (stream, is_chunked) = match self.open(body).await { |
| 1565 | Ok(v) => v, |
| 1566 | Err(e) => return Err(e), |
| 1567 | }; |
| 1568 | let mut reader = LineReader::new(stream, is_chunked); |
| 1569 | let mut full = String::new(); |
| 1570 | loop { |
| 1571 | match reader.read_line().await { |
| 1572 | Ok(Some(l)) => full.push_str(&l), |
| 1573 | Ok(None) => break, |
| 1574 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => break, |
| 1575 | Err(e) => return Err(TransportErr::transient("the reply was cut short".to_string(), err!(e, |
| 1576 | "LLM: read response body failed."; IO, Network, Wire, Read))), |
| 1577 | } |
| 1578 | } |
| 1579 | Ok(full) |
| 1580 | } |
| 1581 | |
| 1582 | /// Send the HTTP request and stream the SSE response line-by-line, |
| 1583 | /// calling `on_data` with each `data:` payload (the JSON after the |
| 1584 | /// `data: ` prefix) as it arrives, stopping at `[DONE]`. Handles |
| 1585 | /// both chunked and identity transfer encoding via [`LineReader`]. |
| 1586 | /// |
| 1587 | /// Returns whether the stream was aborted. The native transport has |
| 1588 | /// no cancellation path, so it always returns `false`; the wasm |
| 1589 | /// transport returns `true` when the browser fired the abort signal. |
| 1590 | #[cfg(not(target_arch = "wasm32"))] |
| 1591 | async fn stream_sse( |
| 1592 | &self, |
| 1593 | body: &str, |
| 1594 | on_data: &mut impl FnMut(&str), |
| 1595 | ) -> Result<bool, TransportErr> |
| 1596 | { |
| 1597 | let (stream, is_chunked) = match self.open(body).await { |
| 1598 | Ok(v) => v, |
| 1599 | Err(e) => return Err(e), |
| 1600 | }; |
| 1601 | let mut reader = LineReader::new(stream, is_chunked); |
| 1602 | loop { |
| 1603 | let line = match reader.read_line().await { |
| 1604 | Ok(Some(l)) => l, |
| 1605 | Ok(None) => break, |
| 1606 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => break, |
| 1607 | Err(e) => return Err(TransportErr::transient("the stream broke".to_string(), err!(e, |
| 1608 | "LLM: read SSE line failed."; IO, Network, Wire, Read))), |
| 1609 | }; |
| 1610 | let line = line.trim(); |
| 1611 | if !line.starts_with("data: ") { |
| 1612 | continue; |
| 1613 | } |
| 1614 | let data = &line[6..]; |
| 1615 | if data == "[DONE]" { |
| 1616 | break; |
| 1617 | } |
| 1618 | on_data(data); |
| 1619 | } |
| 1620 | Ok(false) |
| 1621 | } |
| 1622 | } |
| 1623 | |
| 1624 | |
| 1625 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1626 | // │ Wasm transport — browser `fetch` + `ReadableStream` │ |
| 1627 | // └───────────────────────────────────────────────────────────────┘ |
| 1628 | // |
| 1629 | // The wasm build has no TCP sockets or TLS stack; the browser owns |
| 1630 | // both. These methods mirror the native transport's private contract |
| 1631 | // (`do_request_full` / `stream_sse`) using `fetch`, so the |
| 1632 | // `chat_stream` / `chat_stream_tools` / `chat_once` API above is |
| 1633 | // target-agnostic. |
| 1634 | |
| 1635 | #[cfg(target_arch = "wasm32")] |
| 1636 | impl LlmClient { |
| 1637 | |
| 1638 | /// The absolute request URL for the browser transport. |
| 1639 | /// |
| 1640 | /// The scheme follows [`secure`](Self::secure); the port is elided |
| 1641 | /// only when it is the scheme's default (443 for `https`, 80 for |
| 1642 | /// `http`), so a mock on a custom port is addressed explicitly. |
| 1643 | fn wasm_url(&self) -> String { |
| 1644 | let (scheme, default_port) = if self.secure { ("https", 443u16) } else { ("http", 80u16) }; |
| 1645 | if self.port == default_port { |
| 1646 | fmt!("{}://{}{}", scheme, self.host, self.path) |
| 1647 | } else { |
| 1648 | fmt!("{}://{}:{}{}", scheme, self.host, self.port, self.path) |
| 1649 | } |
| 1650 | } |
| 1651 | |
| 1652 | /// Issue a lightweight transport probe and return the raw HTTP |
| 1653 | /// status the provider replies with. |
| 1654 | /// |
| 1655 | /// Unlike [`wasm_fetch`](Self::wasm_fetch), a non-2xx status is *not* |
| 1656 | /// treated as an error — the status number is the whole point. A |
| 1657 | /// `401` from a real provider with a dummy key proves the full |
| 1658 | /// `fetch` + CORS + transport path end-to-end without a valid key. |
| 1659 | pub async fn probe_status(&self) -> Outcome<u16> { |
| 1660 | let messages = [crate::protocol::ChatMessage::User { |
| 1661 | content: MessageContent::text("ping"), |
| 1662 | }]; |
| 1663 | let body = self.build_body(&messages, None, false); |
| 1664 | let resp = res!(self.wasm_fetch_raw(&body).await); |
| 1665 | Ok(resp.status()) |
| 1666 | } |
| 1667 | |
| 1668 | /// POST `body` via `fetch`, retrying a transient failure with bounded |
| 1669 | /// backoff, and await the `Response`. |
| 1670 | /// |
| 1671 | /// Nothing has reached the caller at this point, so a dropped `fetch`, a 429 |
| 1672 | /// or a 5xx is simply tried again; every other non-2xx is this request being |
| 1673 | /// wrong and is returned as-is. `waited` is the shared backoff budget; see |
| 1674 | /// the native [`open`](LlmClient::open). |
| 1675 | /// |
| 1676 | /// A refusal carries the provider's OWN WORDS, as the native transport has always |
| 1677 | /// done. Without them the browser could say no more than |
| 1678 | /// `LLM: HTTP error: 400 Bad Request.`, which tells the user nothing they can act on |
| 1679 | /// and tells [`compact::looks_like_overflow`](crate::agent::compact::looks_like_overflow) |
| 1680 | /// nothing at all -- so a request refused for being too long and one refused for |
| 1681 | /// being wrong had to be told apart by size alone. Both dialects are covered, |
| 1682 | /// because it is the raw body that is carried and neither is parsed. |
| 1683 | async fn wasm_fetch(&self, body: &str) -> Result<web_sys::Response, TransportErr> { |
| 1684 | let resp = match self.wasm_fetch_raw(body).await { |
| 1685 | Ok(r) => r, |
| 1686 | // A rejected `fetch` is a network or CORS failure; an armed abort is |
| 1687 | // the caller cancelling, and must not be retried. |
| 1688 | Err(e) => return Err(if self.abort_signalled() { |
| 1689 | TransportErr::fatal("the turn was cancelled".to_string(), e) |
| 1690 | } else { |
| 1691 | TransportErr::transient("could not reach the provider".to_string(), e) |
| 1692 | }), |
| 1693 | }; |
| 1694 | if !resp.ok() { |
| 1695 | let status = resp.status(); |
| 1696 | let status_text = resp.status_text(); |
| 1697 | // Read BEFORE the body: consuming the stream cannot then cost the retry its |
| 1698 | // requested delay, and a `Retry-After` is the one thing on a 429 worth more |
| 1699 | // than the message. |
| 1700 | let after = if status_retryable(status) { |
| 1701 | resp.headers().get("retry-after").ok().flatten() |
| 1702 | .and_then(|v| parse_retry_after(&v)) |
| 1703 | } else { |
| 1704 | None |
| 1705 | }; |
| 1706 | let detail = self.body_detail(&resp).await; |
| 1707 | let err = err!( |
| 1708 | "LLM: HTTP error: {} {} | {}", status, status_text, detail; |
| 1709 | IO, Network, Wire, Read); |
| 1710 | let reason = fmt!("the provider returned HTTP {}", status); |
| 1711 | return Err(if status_retryable(status) { |
| 1712 | TransportErr::transient(reason, err).after(after) |
| 1713 | } else { |
| 1714 | TransportErr::fatal(reason, err) |
| 1715 | }); |
| 1716 | } |
| 1717 | Ok(resp) |
| 1718 | } |
| 1719 | |
| 1720 | /// The first [`ERR_BODY_BYTES`] of a refusal's body, or nothing when it cannot be read. |
| 1721 | /// |
| 1722 | /// Consumes the response, which is why it is called only on the failing path. A body |
| 1723 | /// that will not resolve -- an abort landing between the headers and the text, a |
| 1724 | /// provider that sent none -- yields an empty string rather than turning a refusal |
| 1725 | /// with a known status into a failure of a different kind. |
| 1726 | /// |
| 1727 | /// # Arguments |
| 1728 | /// * `resp` - The non-2xx response, whose body is read to exhaustion. |
| 1729 | async fn body_detail(&self, resp: &web_sys::Response) -> String { |
| 1730 | use wasm_bindgen_futures::JsFuture; |
| 1731 | |
| 1732 | let text = match resp.text() { |
| 1733 | Ok(p) => match JsFuture::from(p).await { |
| 1734 | Ok(v) => v.as_string().unwrap_or_default(), |
| 1735 | Err(_) => String::new(), |
| 1736 | }, |
| 1737 | Err(_) => String::new(), |
| 1738 | }; |
| 1739 | clip_bytes(&text, ERR_BODY_BYTES).to_string() |
| 1740 | } |
| 1741 | |
| 1742 | /// POST `body` via `fetch` and await the `Response` without checking |
| 1743 | /// the status, mapping any JS error into an `Outcome`. TLS trust is |
| 1744 | /// the browser's. Callers that need a 2xx guarantee go through |
| 1745 | /// [`wasm_fetch`](Self::wasm_fetch). |
| 1746 | async fn wasm_fetch_raw(&self, body: &str) -> Outcome<web_sys::Response> { |
| 1747 | use wasm_bindgen::JsCast; |
| 1748 | use wasm_bindgen::JsValue; |
| 1749 | use wasm_bindgen_futures::JsFuture; |
| 1750 | use web_sys::{Headers, Request, RequestInit, RequestMode, Response}; |
| 1751 | |
| 1752 | let headers = res!(Headers::new() |
| 1753 | .map_err(|e| err!("LLM: create headers failed: {}.", js_str(&e); IO, Network, Init))); |
| 1754 | // `true`: this is the browser transport, so an Anthropic endpoint also |
| 1755 | // gets the header that makes its edge answer a cross-origin request. |
| 1756 | for (name, value) in self.auth_headers(true) { |
| 1757 | res!(headers.append(name, &value) |
| 1758 | .map_err(|e| err!("LLM: set header {} failed: {}.", name, js_str(&e); |
| 1759 | IO, Network, Init))); |
| 1760 | } |
| 1761 | |
| 1762 | let opts = RequestInit::new(); |
| 1763 | opts.set_method("POST"); |
| 1764 | opts.set_mode(RequestMode::Cors); |
| 1765 | opts.set_headers(&headers); |
| 1766 | opts.set_body(&JsValue::from_str(body)); |
| 1767 | |
| 1768 | // Install a fresh abort controller for this request and wire its |
| 1769 | // signal in, so `abort` can cancel the in-flight fetch/stream. A |
| 1770 | // controller that fails to construct simply leaves the request |
| 1771 | // uncancellable rather than failing the turn. |
| 1772 | if let Ok(ctrl) = web_sys::AbortController::new() { |
| 1773 | opts.set_signal(Some(&ctrl.signal())); |
| 1774 | *self.abort.borrow_mut() = Some(ctrl); |
| 1775 | } |
| 1776 | |
| 1777 | let url = self.wasm_url(); |
| 1778 | let request = res!(Request::new_with_str_and_init(&url, &opts) |
| 1779 | .map_err(|e| err!("LLM: build request failed: {}.", js_str(&e); IO, Network, Init))); |
| 1780 | |
| 1781 | // `fetch` lives on the window in a document context and on the |
| 1782 | // global scope in a worker; support both. |
| 1783 | let promise = if let Some(win) = web_sys::window() { |
| 1784 | win.fetch_with_request(&request) |
| 1785 | } else { |
| 1786 | let scope = res!(js_sys::global() |
| 1787 | .dyn_into::<web_sys::WorkerGlobalScope>() |
| 1788 | .map_err(|_| err!( |
| 1789 | "LLM: no window or worker scope for fetch."; IO, Network, Init))); |
| 1790 | scope.fetch_with_request(&request) |
| 1791 | }; |
| 1792 | |
| 1793 | let resp_val = res!(JsFuture::from(promise).await |
| 1794 | .map_err(|e| err!("LLM: fetch failed: {}.", js_str(&e); IO, Network, Wire))); |
| 1795 | let resp: Response = res!(resp_val.dyn_into() |
| 1796 | .map_err(|_| err!("LLM: fetch did not return a Response."; IO, Network, Wire))); |
| 1797 | Ok(resp) |
| 1798 | } |
| 1799 | |
| 1800 | /// Non-streaming request — await the full response body as text. |
| 1801 | async fn do_request_full(&self, body: &str) -> Result<String, TransportErr> { |
| 1802 | use wasm_bindgen_futures::JsFuture; |
| 1803 | |
| 1804 | let resp = match self.wasm_fetch(body).await { |
| 1805 | Ok(r) => r, |
| 1806 | Err(e) => return Err(e), |
| 1807 | }; |
| 1808 | let text_promise = match resp.text() { |
| 1809 | Ok(p) => p, |
| 1810 | Err(e) => return Err(TransportErr::transient("the reply was cut short".to_string(), err!( |
| 1811 | "LLM: read response text failed: {}.", js_str(&e); IO, Network, Wire, Read))), |
| 1812 | }; |
| 1813 | let text_val = match JsFuture::from(text_promise).await { |
| 1814 | Ok(v) => v, |
| 1815 | Err(e) => return Err(TransportErr::transient("the reply was cut short".to_string(), err!( |
| 1816 | "LLM: await response text failed: {}.", js_str(&e); IO, Network, Wire, Read))), |
| 1817 | }; |
| 1818 | Ok(text_val.as_string().unwrap_or_default()) |
| 1819 | } |
| 1820 | |
| 1821 | /// Streaming request — read the SSE body incrementally from the |
| 1822 | /// response's `ReadableStream`, calling `on_data` with each `data:` |
| 1823 | /// payload as it arrives, stopping at `[DONE]`. |
| 1824 | /// |
| 1825 | /// Returns whether the browser fired the abort signal. When the |
| 1826 | /// initial `fetch` or a stream read rejects, an armed abort is |
| 1827 | /// distinguished from a genuine transport failure: an abort resolves |
| 1828 | /// to `Ok(true)` (the caller keeps whatever streamed and ends the |
| 1829 | /// turn cleanly), any other rejection is a real error. |
| 1830 | async fn stream_sse( |
| 1831 | &self, |
| 1832 | body: &str, |
| 1833 | on_data: &mut impl FnMut(&str), |
| 1834 | ) -> Result<bool, TransportErr> |
| 1835 | { |
| 1836 | use wasm_bindgen::JsValue; |
| 1837 | use wasm_bindgen_futures::JsFuture; |
| 1838 | use web_sys::{ReadableStream, ReadableStreamDefaultReader}; |
| 1839 | |
| 1840 | let resp = match self.wasm_fetch(body).await { |
| 1841 | Ok(r) => r, |
| 1842 | Err(e) => { |
| 1843 | if self.abort_signalled() { return Ok(true); } |
| 1844 | return Err(e); |
| 1845 | } |
| 1846 | }; |
| 1847 | let stream: ReadableStream = match resp.body() { |
| 1848 | Some(s) => s, |
| 1849 | None => return Err(TransportErr::transient("the reply carried no stream".to_string(), err!( |
| 1850 | "LLM: response has no body stream."; IO, Network, Wire, Read))), |
| 1851 | }; |
| 1852 | let reader = match ReadableStreamDefaultReader::new(&stream) { |
| 1853 | Ok(r) => r, |
| 1854 | Err(e) => return Err(TransportErr::fatal("the stream could not be read".to_string(), err!( |
| 1855 | "LLM: acquire stream reader failed: {}.", js_str(&e); IO, Network, Wire, Read))), |
| 1856 | }; |
| 1857 | |
| 1858 | // Accumulate raw bytes and extract complete SSE lines as they |
| 1859 | // arrive, mirroring the native `LineReader` line discipline. |
| 1860 | let mut buf: Vec<u8> = Vec::with_capacity(8192); |
| 1861 | |
| 1862 | loop { |
| 1863 | let result = match JsFuture::from(reader.read()).await { |
| 1864 | Ok(r) => r, |
| 1865 | Err(e) => { |
| 1866 | if self.abort_signalled() { return Ok(true); } |
| 1867 | // A stream that broke mid-flight; whether it is safe to try |
| 1868 | // again is the caller's judgement, not this layer's. |
| 1869 | return Err(TransportErr::transient("the stream broke".to_string(), err!( |
| 1870 | "LLM: read stream chunk failed: {}.", js_str(&e); |
| 1871 | IO, Network, Wire, Read))); |
| 1872 | } |
| 1873 | }; |
| 1874 | let done = match js_sys::Reflect::get(&result, &JsValue::from_str("done")) { |
| 1875 | Ok(v) => v.as_bool().unwrap_or(true), |
| 1876 | Err(e) => return Err(TransportErr::fatal("the stream was malformed".to_string(), err!( |
| 1877 | "LLM: read 'done' failed: {}.", js_str(&e); IO, Network, Wire, Read))), |
| 1878 | }; |
| 1879 | if done { |
| 1880 | break; |
| 1881 | } |
| 1882 | let value = match js_sys::Reflect::get(&result, &JsValue::from_str("value")) { |
| 1883 | Ok(v) => v, |
| 1884 | Err(e) => return Err(TransportErr::fatal("the stream was malformed".to_string(), err!( |
| 1885 | "LLM: read 'value' failed: {}.", js_str(&e); IO, Network, Wire, Read))), |
| 1886 | }; |
| 1887 | let chunk = js_sys::Uint8Array::new(&value).to_vec(); |
| 1888 | buf.extend_from_slice(&chunk); |
| 1889 | |
| 1890 | // Drain complete lines (terminated by `\n`) from the buffer. |
| 1891 | loop { |
| 1892 | let nl = match buf.iter().position(|&b| b == b'\n') { |
| 1893 | Some(p) => p, |
| 1894 | None => break, |
| 1895 | }; |
| 1896 | let line_bytes: Vec<u8> = buf.drain(..=nl).collect(); |
| 1897 | let line = String::from_utf8_lossy(&line_bytes[..line_bytes.len() - 1]); |
| 1898 | let line = line.trim(); |
| 1899 | if !line.starts_with("data: ") { |
| 1900 | continue; |
| 1901 | } |
| 1902 | let data = &line[6..]; |
| 1903 | if data == "[DONE]" { |
| 1904 | return Ok(false); |
| 1905 | } |
| 1906 | on_data(data); |
| 1907 | } |
| 1908 | } |
| 1909 | |
| 1910 | Ok(false) |
| 1911 | } |
| 1912 | |
| 1913 | /// Fire the abort signal for the in-flight request, if any. Safe to |
| 1914 | /// call when idle: with no armed controller it is a no-op. |
| 1915 | pub fn abort(&self) { |
| 1916 | if let Some(ctrl) = self.abort.borrow().as_ref() { |
| 1917 | ctrl.abort(); |
| 1918 | } |
| 1919 | } |
| 1920 | |
| 1921 | /// Whether the armed abort controller's signal has fired. Used to |
| 1922 | /// tell a cancelled fetch/stream apart from a genuine failure. |
| 1923 | fn abort_signalled(&self) -> bool { |
| 1924 | self.abort |
| 1925 | .borrow() |
| 1926 | .as_ref() |
| 1927 | .map(|ctrl| ctrl.signal().aborted()) |
| 1928 | .unwrap_or(false) |
| 1929 | } |
| 1930 | } |
| 1931 | |
| 1932 | /// Render a JS error value as a human-readable string for error tags. |
| 1933 | #[cfg(target_arch = "wasm32")] |
| 1934 | fn js_str(v: &wasm_bindgen::JsValue) -> String { |
| 1935 | v.as_string().unwrap_or_else(|| fmt!("{:?}", v)) |
| 1936 | } |
| 1937 | |
| 1938 | |
| 1939 | // ┌───────────────────────────────────────────────────────────────┐ |
| 1940 | // │ LineReader — incremental line reader for TLS streams │ |
| 1941 | // └───────────────────────────────────────────────────────────────┘ |
| 1942 | |
| 1943 | /// Reads lines from a TLS stream, handling HTTP chunked transfer |
| 1944 | /// encoding transparently. |
| 1945 | /// |
| 1946 | /// For identity (Content-Length) encoding, lines are read directly |
| 1947 | /// from the stream. For chunked encoding, chunk headers are parsed |
| 1948 | /// and chunk data is dechunked on the fly, so the caller sees a |
| 1949 | /// continuous stream of lines. |
| 1950 | /// |
| 1951 | /// A line is terminated by `\n` (with or without a preceding `\r`). |
| 1952 | #[cfg(not(target_arch = "wasm32"))] |
| 1953 | struct LineReader<S: tokio::io::AsyncRead + Unpin> { |
| 1954 | stream: S, |
| 1955 | buf: Vec<u8>, |
| 1956 | buf_pos: usize, |
| 1957 | is_chunked: bool, |
| 1958 | // For chunked encoding: remaining bytes in the current chunk. |
| 1959 | // None means we need to read the next chunk header. |
| 1960 | chunk_remaining: Option<usize>, |
| 1961 | eof: bool, |
| 1962 | } |
| 1963 | |
| 1964 | #[cfg(not(target_arch = "wasm32"))] |
| 1965 | impl<S: tokio::io::AsyncRead + Unpin> LineReader<S> { |
| 1966 | |
| 1967 | fn new(stream: S, is_chunked: bool) -> Self { |
| 1968 | Self { |
| 1969 | stream, |
| 1970 | buf: Vec::with_capacity(8192), |
| 1971 | buf_pos: 0, |
| 1972 | is_chunked, |
| 1973 | chunk_remaining: None, |
| 1974 | eof: false, |
| 1975 | } |
| 1976 | } |
| 1977 | |
| 1978 | /// Read the next line (without the trailing newline). |
| 1979 | /// |
| 1980 | /// Returns `Ok(None)` at end of stream. |
| 1981 | async fn read_line(&mut self) -> std::io::Result<Option<String>> { |
| 1982 | loop { |
| 1983 | // Try to find a complete line in the buffer. |
| 1984 | if let Some(line) = self.try_extract_line() { |
| 1985 | return Ok(Some(line)); |
| 1986 | } |
| 1987 | if self.eof { |
| 1988 | // If there's remaining data without a newline, |
| 1989 | // return it as the last line. |
| 1990 | if self.buf_pos < self.buf.len() { |
| 1991 | let rest = String::from_utf8_lossy( |
| 1992 | &self.buf[self.buf_pos..] |
| 1993 | ).to_string(); |
| 1994 | self.buf_pos = self.buf.len(); |
| 1995 | return Ok(Some(rest)); |
| 1996 | } |
| 1997 | return Ok(None); |
| 1998 | } |
| 1999 | // Need more data. |
| 2000 | match self.fill_buf().await { |
| 2001 | Ok(()) => {}, |
| 2002 | Err(e) => return Err(e), |
| 2003 | } |
| 2004 | } |
| 2005 | } |
| 2006 | |
| 2007 | /// Try to extract a complete line from the buffer. |
| 2008 | fn try_extract_line(&mut self) -> Option<String> { |
| 2009 | let search_start = self.buf_pos; |
| 2010 | let rest = &self.buf[search_start..]; |
| 2011 | if let Some(pos) = rest.iter().position(|&b| b == b'\n') { |
| 2012 | let end = search_start + pos; |
| 2013 | let line = &self.buf[self.buf_pos..end]; |
| 2014 | // Strip trailing \r if present. |
| 2015 | let line = if line.ends_with(b"\r") { &line[..line.len()-1] } else { line }; |
| 2016 | let s = String::from_utf8_lossy(line).to_string(); |
| 2017 | self.buf_pos = end + 1; // skip the \n |
| 2018 | // Compact buffer periodically. |
| 2019 | if self.buf_pos > 16384 { |
| 2020 | self.buf.drain(..self.buf_pos); |
| 2021 | self.buf_pos = 0; |
| 2022 | } |
| 2023 | return Some(s); |
| 2024 | } |
| 2025 | None |
| 2026 | } |
| 2027 | |
| 2028 | /// Read more data into the buffer. |
| 2029 | async fn fill_buf(&mut self) -> std::io::Result<()> { |
| 2030 | let mut tmp = [0u8; 4096]; |
| 2031 | |
| 2032 | if self.is_chunked { |
| 2033 | // For chunked encoding, we need to be careful about |
| 2034 | // chunk boundaries. However, SSE lines are always |
| 2035 | // within a single chunk in practice (servers don't |
| 2036 | // split a data: line across chunks). We read raw |
| 2037 | // bytes and handle chunk boundaries in the line |
| 2038 | // buffer. This is simpler than tracking exact chunk |
| 2039 | // positions and works because we only need lines. |
| 2040 | // |
| 2041 | // For correctness, we parse chunk headers when we |
| 2042 | // run out of chunk data. |
| 2043 | if self.chunk_remaining == Some(0) { |
| 2044 | // Read and discard the trailing \r\n after a chunk, |
| 2045 | // then read the next chunk header. |
| 2046 | let mut crlf = [0u8; 2]; |
| 2047 | match self.stream.read_exact(&mut crlf).await { |
| 2048 | Ok(_) => {} |
| 2049 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => { |
| 2050 | self.eof = true; |
| 2051 | return Ok(()); |
| 2052 | } |
| 2053 | Err(e) => return Err(e), |
| 2054 | } |
| 2055 | self.chunk_remaining = None; |
| 2056 | } |
| 2057 | |
| 2058 | if self.chunk_remaining.is_none() { |
| 2059 | // Read chunk size line. |
| 2060 | let mut size_line = Vec::new(); |
| 2061 | let mut byte = [0u8; 1]; |
| 2062 | loop { |
| 2063 | match self.stream.read(&mut byte).await { |
| 2064 | Ok(0) => { self.eof = true; return Ok(()); } |
| 2065 | Ok(_) => { |
| 2066 | size_line.push(byte[0]); |
| 2067 | if size_line.ends_with(b"\r\n") { |
| 2068 | break; |
| 2069 | } |
| 2070 | // Some servers include chunk extensions |
| 2071 | // after the size: 1a;ext=val\r\n |
| 2072 | if size_line.ends_with(b"\n") { |
| 2073 | break; |
| 2074 | } |
| 2075 | } |
| 2076 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => { |
| 2077 | self.eof = true; |
| 2078 | return Ok(()); |
| 2079 | } |
| 2080 | Err(e) => return Err(e), |
| 2081 | } |
| 2082 | } |
| 2083 | let size_str = String::from_utf8_lossy(&size_line); |
| 2084 | let size_str = size_str.trim(); |
| 2085 | // Strip chunk extensions (everything after ;). |
| 2086 | let size_str = size_str.split(';').next().unwrap_or("0").trim(); |
| 2087 | let size = match usize::from_str_radix(size_str, 16) { |
| 2088 | Ok(n) => n, |
| 2089 | Err(_) => { self.eof = true; return Ok(()); } |
| 2090 | }; |
| 2091 | if size == 0 { |
| 2092 | // Last chunk — end of body. |
| 2093 | self.eof = true; |
| 2094 | return Ok(()); |
| 2095 | } |
| 2096 | self.chunk_remaining = Some(size); |
| 2097 | } |
| 2098 | |
| 2099 | // Read up to chunk_remaining bytes or tmp.len(), whichever is smaller. |
| 2100 | let remaining = match self.chunk_remaining { |
| 2101 | Some(r) => r, |
| 2102 | None => return Err(std::io::Error::new( |
| 2103 | std::io::ErrorKind::Other, |
| 2104 | "chunk_remaining unexpectedly unset")), |
| 2105 | }; |
| 2106 | let to_read = remaining.min(tmp.len()); |
| 2107 | match self.stream.read(&mut tmp[..to_read]).await { |
| 2108 | Ok(0) => { self.eof = true; return Ok(()); } |
| 2109 | Ok(n) => { |
| 2110 | self.buf.extend_from_slice(&tmp[..n]); |
| 2111 | self.chunk_remaining = Some(remaining - n); |
| 2112 | } |
| 2113 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => { |
| 2114 | self.eof = true; |
| 2115 | return Ok(()); |
| 2116 | } |
| 2117 | Err(e) => return Err(e), |
| 2118 | } |
| 2119 | } else { |
| 2120 | // Identity encoding — read directly. |
| 2121 | match self.stream.read(&mut tmp).await { |
| 2122 | Ok(0) => { self.eof = true; return Ok(()); } |
| 2123 | Ok(n) => self.buf.extend_from_slice(&tmp[..n]), |
| 2124 | Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => { |
| 2125 | self.eof = true; |
| 2126 | return Ok(()); |
| 2127 | } |
| 2128 | Err(e) => return Err(e), |
| 2129 | } |
| 2130 | } |
| 2131 | Ok(()) |
| 2132 | } |
| 2133 | } |
| 2134 | |
| 2135 | /// Parse an SSE response body, calling `on_token` for each text delta. |
| 2136 | /// |
| 2137 | /// SSE format: |
| 2138 | /// ```text |
| 2139 | /// data: {"choices":[{"delta":{"content":"Hello"}}]} |
| 2140 | /// |
| 2141 | /// data: {"choices":[{"delta":{"content":" world"}}]} |
| 2142 | /// |
| 2143 | /// data: [DONE] |
| 2144 | /// ``` |
| 2145 | /// |
| 2146 | /// We scan for `"content":"..."` in each `data:` line. This is a |
| 2147 | /// deliberately simple parser — it handles the common case without |
| 2148 | /// needing a full JSON parser. Escaped quotes inside content are |
| 2149 | /// handled by scanning for the matching unescaped quote. |
| 2150 | pub fn parse_sse_stream(body: &[u8], on_token: &mut impl FnMut(&str)) |
| 2151 | -> (String, Usage) |
| 2152 | { |
| 2153 | let text = String::from_utf8_lossy(body); |
| 2154 | let mut full = String::new(); |
| 2155 | let mut use_ = Usage::default(); |
| 2156 | |
| 2157 | for line in text.lines() { |
| 2158 | let line = line.trim(); |
| 2159 | if !line.starts_with("data: ") { |
| 2160 | continue; |
| 2161 | } |
| 2162 | let data = &line[6..]; |
| 2163 | if data == "[DONE]" { |
| 2164 | break; |
| 2165 | } |
| 2166 | // Extract content from: {"choices":[{"delta":{"content":"..."}}]} |
| 2167 | if let Some(content) = extract_json_string(data, "content") { |
| 2168 | on_token(&content); |
| 2169 | full.push_str(&content); |
| 2170 | } |
| 2171 | // Extract usage from the final chunk: |
| 2172 | // {"choices":[],"usage":{"prompt_tokens":13,"completion_tokens":200}} |
| 2173 | if let Some(u) = parse_usage(data) { |
| 2174 | use_ = u; |
| 2175 | } |
| 2176 | } |
| 2177 | |
| 2178 | (full, use_) |
| 2179 | } |
| 2180 | |
| 2181 | /// Read a `usage` object out of a whole response body or one SSE chunk, |
| 2182 | /// returning `None` when the chunk carries none. |
| 2183 | /// |
| 2184 | /// Intermediate streamed chunks send `"usage":null`, which is not an object |
| 2185 | /// and so reads as absent rather than as a zeroed usage -- otherwise the last |
| 2186 | /// chunk before `[DONE]` would erase what the usage chunk reported. |
| 2187 | /// |
| 2188 | /// # Arguments |
| 2189 | /// * `json` - A response body, or one SSE `data:` payload. |
| 2190 | pub(crate) fn parse_usage(json: &str) -> Option<Usage> { |
| 2191 | let usage = match find_json_object(json, "usage") { |
| 2192 | Some(u) => u, |
| 2193 | None => return None, |
| 2194 | }; |
| 2195 | let mut u = Usage::default(); |
| 2196 | if let Some(p) = extract_json_number(&usage, "prompt_tokens") { u.prompt = p; } |
| 2197 | if let Some(c) = extract_json_number(&usage, "completion_tokens") { u.completion = c; } |
| 2198 | // Cache reads live in a nested `prompt_tokens_details`; a provider that |
| 2199 | // flattens the field is read too, so neither shape is missed. A cache read |
| 2200 | // bills at a fraction of a fresh prompt token, and in an agentic tool loop |
| 2201 | // -- where every round's prompt is the last round's plus a little -- it is |
| 2202 | // most of the prompt, so counting it at the full input rate was the single |
| 2203 | // largest source of overstatement. |
| 2204 | // Three spellings, because three providers report the same figure three |
| 2205 | // ways: the OpenAI-compatible nesting, a flattened copy of it, and |
| 2206 | // Anthropic's own `cache_read_input_tokens`. Missing the last one would |
| 2207 | // read a working prompt cache as no cache at all. |
| 2208 | u.cached = match find_json_object(&usage, "prompt_tokens_details") { |
| 2209 | Some(d) => extract_json_number(&d, "cached_tokens").unwrap_or(0), |
| 2210 | None => extract_json_number(&usage, "cached_tokens") |
| 2211 | .or_else(|| extract_json_number(&usage, "cache_read_input_tokens")) |
| 2212 | .unwrap_or(0), |
| 2213 | }; |
| 2214 | // What the provider actually drew. This is money, not an estimate, and it |
| 2215 | // supersedes anything the price table would have guessed. |
| 2216 | if let Some(c) = extract_json_f64(&usage, "cost") { u.cost_usd = c; } |
| 2217 | Some(u) |
| 2218 | } |
| 2219 | |
| 2220 | /// Extract a JSON object value for a key from a JSON string. |
| 2221 | /// |
| 2222 | /// Scans for `"key":{...}` and returns the inner object string |
| 2223 | /// (including the braces). Used to extract the `usage` object |
| 2224 | /// from the final SSE chunk. |
| 2225 | fn find_json_object(json: &str, key: &str) -> Option<String> { |
| 2226 | let needle = fmt!("\"{}\":", key); |
| 2227 | let pos = match json.find(&needle) { |
| 2228 | Some(p) => p, |
| 2229 | None => return None, |
| 2230 | }; |
| 2231 | let bytes = json.as_bytes(); |
| 2232 | // Skip whitespace after the colon to the opening brace. |
| 2233 | let mut start = pos + needle.len(); |
| 2234 | while start < bytes.len() && bytes[start].is_ascii_whitespace() { start += 1; } |
| 2235 | if start >= bytes.len() || bytes[start] != b'{' { return None; } |
| 2236 | let mut depth = 0i32; |
| 2237 | let mut i = start; |
| 2238 | while i < bytes.len() { |
| 2239 | match bytes[i] { |
| 2240 | b'{' => depth += 1, |
| 2241 | b'}' => { |
| 2242 | depth -= 1; |
| 2243 | if depth == 0 { |
| 2244 | return Some(json[start..=i].to_string()); |
| 2245 | } |
| 2246 | } |
| 2247 | b'"' => { |
| 2248 | // Skip string contents. |
| 2249 | i += 1; |
| 2250 | while i < bytes.len() { |
| 2251 | if bytes[i] == b'\\' { i += 2; continue; } |
| 2252 | if bytes[i] == b'"' { break; } |
| 2253 | i += 1; |
| 2254 | } |
| 2255 | } |
| 2256 | _ => (), |
| 2257 | } |
| 2258 | i += 1; |
| 2259 | } |
| 2260 | None |
| 2261 | } |
| 2262 | |
| 2263 | /// Extract a numeric value for a key from a JSON string. |
| 2264 | /// |
| 2265 | /// Scans for `"key":number` and returns the parsed value. |
| 2266 | pub(crate) fn extract_json_number(json: &str, key: &str) -> Option<u64> { |
| 2267 | let needle = fmt!("\"{}\":", key); |
| 2268 | let pos = match json.find(&needle) { |
| 2269 | Some(p) => p, |
| 2270 | None => return None, |
| 2271 | }; |
| 2272 | let mut start = pos + needle.len(); |
| 2273 | let bytes = json.as_bytes(); |
| 2274 | // Skip whitespace. |
| 2275 | while start < bytes.len() && bytes[start].is_ascii_whitespace() { |
| 2276 | start += 1; |
| 2277 | } |
| 2278 | let mut end = start; |
| 2279 | while end < bytes.len() && (bytes[end].is_ascii_digit() || bytes[end] == b'-') { |
| 2280 | end += 1; |
| 2281 | } |
| 2282 | json[start..end].parse::<u64>().ok() |
| 2283 | } |
| 2284 | |
| 2285 | /// Extract a SIGNED integer value for a key from a JSON string. |
| 2286 | /// |
| 2287 | /// [`extract_json_number`] parses a `u64`, so a negative number does not merely come back wrong -- |
| 2288 | /// it comes back as `None`, and every caller that reached for `unwrap_or(0)` then read a negative |
| 2289 | /// value as zero. For a process exit status that is the difference between "the command was |
| 2290 | /// killed" and "the command succeeded", so the signed reader exists separately rather than as a |
| 2291 | /// cast at the call site. |
| 2292 | /// |
| 2293 | /// # Arguments |
| 2294 | /// * `json` - The JSON text to read. |
| 2295 | /// * `key` - The key whose value is wanted. |
| 2296 | pub(crate) fn extract_json_i64(json: &str, key: &str) -> Option<i64> { |
| 2297 | let needle = fmt!("\"{}\":", key); |
| 2298 | let pos = match json.find(&needle) { |
| 2299 | Some(p) => p, |
| 2300 | None => return None, |
| 2301 | }; |
| 2302 | let mut start = pos + needle.len(); |
| 2303 | let bytes = json.as_bytes(); |
| 2304 | while start < bytes.len() && bytes[start].is_ascii_whitespace() { |
| 2305 | start += 1; |
| 2306 | } |
| 2307 | let mut end = start; |
| 2308 | while end < bytes.len() && (bytes[end].is_ascii_digit() || bytes[end] == b'-') { |
| 2309 | end += 1; |
| 2310 | } |
| 2311 | json[start..end].parse::<i64>().ok() |
| 2312 | } |
| 2313 | |
| 2314 | /// Extract a fractional numeric value for a key from a JSON string. |
| 2315 | /// |
| 2316 | /// [`extract_json_number`] stops at the first non-digit, so it reads `0.0021` |
| 2317 | /// as `0` -- which silently priced every reported cost at nothing. This scans |
| 2318 | /// the whole JSON number grammar: sign, digits, decimal point and exponent. |
| 2319 | /// |
| 2320 | /// # Arguments |
| 2321 | /// * `json` - The JSON text to scan. |
| 2322 | /// * `key` - The key whose value is wanted. |
| 2323 | pub(crate) fn extract_json_f64(json: &str, key: &str) -> Option<f64> { |
| 2324 | let needle = fmt!("\"{}\":", key); |
| 2325 | let pos = match json.find(&needle) { |
| 2326 | Some(p) => p, |
| 2327 | None => return None, |
| 2328 | }; |
| 2329 | let bytes = json.as_bytes(); |
| 2330 | let mut start = pos + needle.len(); |
| 2331 | // Skip whitespace, and an opening quote for a provider that sends the |
| 2332 | // figure as a string. |
| 2333 | while start < bytes.len() && bytes[start].is_ascii_whitespace() { |
| 2334 | start += 1; |
| 2335 | } |
| 2336 | if start < bytes.len() && bytes[start] == b'"' { |
| 2337 | start += 1; |
| 2338 | } |
| 2339 | let mut end = start; |
| 2340 | while end < bytes.len() { |
| 2341 | let b = bytes[end]; |
| 2342 | let numeric = b.is_ascii_digit() |
| 2343 | || b == b'.' |
| 2344 | || b == b'-' |
| 2345 | || b == b'+' |
| 2346 | || b == b'e' |
| 2347 | || b == b'E'; |
| 2348 | if !numeric { break; } |
| 2349 | end += 1; |
| 2350 | } |
| 2351 | json[start..end].parse::<f64>().ok() |
| 2352 | } |
| 2353 | |
| 2354 | /// Extract a boolean value for a key from a JSON string. |
| 2355 | /// |
| 2356 | /// Scans for `"key":true`/`false` and accepts the quoted forms too, since |
| 2357 | /// models routinely send a boolean argument as the string `"true"`. |
| 2358 | pub fn extract_json_bool(json: &str, key: &str) -> Option<bool> { |
| 2359 | let needle = fmt!("\"{}\":", key); |
| 2360 | let pos = match json.find(&needle) { |
| 2361 | Some(p) => p, |
| 2362 | None => return None, |
| 2363 | }; |
| 2364 | let bytes = json.as_bytes(); |
| 2365 | let mut i = pos + needle.len(); |
| 2366 | // Skip whitespace, then an optional opening quote. |
| 2367 | while i < bytes.len() && bytes[i].is_ascii_whitespace() { |
| 2368 | i += 1; |
| 2369 | } |
| 2370 | if i < bytes.len() && bytes[i] == b'"' { |
| 2371 | i += 1; |
| 2372 | } |
| 2373 | let rest = &json[i..]; |
| 2374 | if rest.starts_with("true") { |
| 2375 | Some(true) |
| 2376 | } else if rest.starts_with("false") { |
| 2377 | Some(false) |
| 2378 | } else { |
| 2379 | None |
| 2380 | } |
| 2381 | } |
| 2382 | |
| 2383 | /// Extract an array of strings for a key from a JSON string. |
| 2384 | /// |
| 2385 | /// `None` when the key is absent or its value is not an array, which is what |
| 2386 | /// lets a caller tell a field that was never written from one written empty. |
| 2387 | pub(crate) fn extract_json_string_array(json: &str, key: &str) -> Option<Vec<String>> { |
| 2388 | let arr = match find_json_array(json, key) { |
| 2389 | Some(a) => a, |
| 2390 | None => return None, |
| 2391 | }; |
| 2392 | Some(parse_json_string_array(&arr)) |
| 2393 | } |
| 2394 | |
| 2395 | /// Extract an array of objects for a key from a JSON string, each as its own text. |
| 2396 | /// |
| 2397 | /// The sibling of [`extract_json_string_array`] for the shape a tool argument takes when one call |
| 2398 | /// carries several of a thing -- `"edits":[{...},{...}]`. Each element comes back whole, for |
| 2399 | /// [`extract_json_string`] and its siblings to read the fields out of. |
| 2400 | /// |
| 2401 | /// `None` when the key is absent or its value is not an array, which is what lets a caller tell a |
| 2402 | /// field that was never written from one written empty. |
| 2403 | pub(crate) fn extract_json_objects(json: &str, key: &str) -> Option<Vec<String>> { |
| 2404 | find_json_array(json, key).map(|arr| split_top_level_objects(&arr)) |
| 2405 | } |
| 2406 | |
| 2407 | /// Parse a JSON array's text into its string elements, ignoring any element |
| 2408 | /// that is not a string. |
| 2409 | /// |
| 2410 | /// Handles the escapes [`json_escape`] emits, `\uXXXX` among them, so a value |
| 2411 | /// survives the round trip out to storage and back. |
| 2412 | pub(crate) fn parse_json_string_array(arr: &str) -> Vec<String> { |
| 2413 | let bytes = arr.as_bytes(); |
| 2414 | let mut out: Vec<String> = Vec::new(); |
| 2415 | let mut i = 0; |
| 2416 | while i < bytes.len() { |
| 2417 | // Anything outside a quoted element -- brackets, commas, a number -- |
| 2418 | // is not a string, and is stepped over. |
| 2419 | if bytes[i] != b'"' { |
| 2420 | i += 1; |
| 2421 | continue; |
| 2422 | } |
| 2423 | i += 1; // past the opening quote |
| 2424 | // Collect as bytes, then decode as UTF-8, so multi-byte characters survive. |
| 2425 | let mut buf: Vec<u8> = Vec::new(); |
| 2426 | while i < bytes.len() { |
| 2427 | let b = bytes[i]; |
| 2428 | if b == b'\\' && i + 1 < bytes.len() { |
| 2429 | match bytes[i + 1] { |
| 2430 | b'"' => { buf.push(b'"'); i += 2; } |
| 2431 | b'\\' => { buf.push(b'\\'); i += 2; } |
| 2432 | b'n' => { buf.push(b'\n'); i += 2; } |
| 2433 | b't' => { buf.push(b'\t'); i += 2; } |
| 2434 | b'r' => { buf.push(b'\r'); i += 2; } |
| 2435 | b'/' => { buf.push(b'/'); i += 2; } |
| 2436 | b'b' => { buf.push(0x08); i += 2; } |
| 2437 | b'f' => { buf.push(0x0c); i += 2; } |
| 2438 | b'u' => match decode_json_unicode(bytes, i + 2) { |
| 2439 | Some((c, next)) => { |
| 2440 | let mut enc = [0u8; 4]; |
| 2441 | buf.extend_from_slice(c.encode_utf8(&mut enc).as_bytes()); |
| 2442 | i = next; |
| 2443 | } |
| 2444 | // Not a well-formed escape; keep it as written rather |
| 2445 | // than lose the characters. |
| 2446 | None => { buf.push(b'\\'); buf.push(b'u'); i += 2; } |
| 2447 | }, |
| 2448 | other => { buf.push(b'\\'); buf.push(other); i += 2; } |
| 2449 | } |
| 2450 | } else if b == b'"' { |
| 2451 | i += 1; // past the closing quote |
| 2452 | break; |
| 2453 | } else { |
| 2454 | buf.push(b); |
| 2455 | i += 1; |
| 2456 | } |
| 2457 | } |
| 2458 | out.push(String::from_utf8_lossy(&buf).to_string()); |
| 2459 | } |
| 2460 | out |
| 2461 | } |
| 2462 | |
| 2463 | /// Decode a `\uXXXX` escape whose first hex digit is at `i`, pairing a leading |
| 2464 | /// surrogate with the trailing one that follows it. |
| 2465 | /// |
| 2466 | /// Returns the character and the index just past the escape, or `None` when the |
| 2467 | /// escape is malformed or a surrogate is left unpaired. |
| 2468 | fn decode_json_unicode(bytes: &[u8], i: usize) -> Option<(char, usize)> { |
| 2469 | // Four hex digits at `s`, as a code unit. |
| 2470 | let unit = |s: usize| -> Option<u32> { |
| 2471 | if s + 4 > bytes.len() { |
| 2472 | return None; |
| 2473 | } |
| 2474 | match std::str::from_utf8(&bytes[s..s + 4]) { |
| 2475 | Ok(txt) => u32::from_str_radix(txt, 16).ok(), |
| 2476 | Err(_) => None, |
| 2477 | } |
| 2478 | }; |
| 2479 | let first = match unit(i) { |
| 2480 | Some(v) => v, |
| 2481 | None => return None, |
| 2482 | }; |
| 2483 | // A leading surrogate is only half a character: its pair follows as a |
| 2484 | // second `\uXXXX`, and the two combine into one code point. |
| 2485 | if (0xD800..0xDC00).contains(&first) { |
| 2486 | let j = i + 4; |
| 2487 | if j + 6 <= bytes.len() && bytes[j] == b'\\' && bytes[j + 1] == b'u' { |
| 2488 | if let Some(second) = unit(j + 2) { |
| 2489 | if (0xDC00..0xE000).contains(&second) { |
| 2490 | let cp = 0x10000 + ((first - 0xD800) << 10) + (second - 0xDC00); |
| 2491 | return char::from_u32(cp).map(|c| (c, j + 6)); |
| 2492 | } |
| 2493 | } |
| 2494 | } |
| 2495 | return None; |
| 2496 | } |
| 2497 | char::from_u32(first).map(|c| (c, i + 4)) |
| 2498 | } |
| 2499 | |
| 2500 | |
| 2501 | /// Handles `\"`, `\\`, `\n`, `\t` escapes. |
| 2502 | /// |
| 2503 | /// The search ensures `key` is a complete JSON key, not a suffix of |
| 2504 | /// a longer key (e.g. `"content"` must not match inside |
| 2505 | /// `"reasoning_content"`). This is done by requiring the character |
| 2506 | /// before the opening quote to be `{` or `,` (whitespace-tolerant). |
| 2507 | pub(crate) fn extract_json_string(json: &str, key: &str) -> Option<String> { |
| 2508 | let needle = fmt!("\"{}\":", key); |
| 2509 | let bytes = json.as_bytes(); |
| 2510 | let mut search_from = 0; |
| 2511 | loop { |
| 2512 | let pos = match json[search_from..].find(&needle) { |
| 2513 | Some(p) => search_from + p, |
| 2514 | None => return None, |
| 2515 | }; |
| 2516 | // Reject suffix matches (e.g. "content" inside |
| 2517 | // "reasoning_content") by checking the character before the |
| 2518 | // key's opening quote. |
| 2519 | let valid_prefix = pos == 0 || { |
| 2520 | let prev = bytes[pos - 1]; |
| 2521 | prev == b'{' || prev == b',' || prev.is_ascii_whitespace() |
| 2522 | }; |
| 2523 | if !valid_prefix { |
| 2524 | search_from = pos + needle.len(); |
| 2525 | continue; |
| 2526 | } |
| 2527 | // Skip whitespace between the colon and the value — real API |
| 2528 | // output uses `"key": "value"` with a space. |
| 2529 | let mut i = pos + needle.len(); |
| 2530 | while i < bytes.len() && bytes[i].is_ascii_whitespace() { i += 1; } |
| 2531 | if i >= bytes.len() || bytes[i] != b'"' { |
| 2532 | // Value is not a string (null / number / object); keep |
| 2533 | // searching in case the key appears again. |
| 2534 | search_from = pos + needle.len(); |
| 2535 | continue; |
| 2536 | } |
| 2537 | i += 1; // past the opening quote |
| 2538 | // Collect the string value as bytes, then decode as UTF-8, so |
| 2539 | // multi-byte characters survive. |
| 2540 | let mut out: Vec<u8> = Vec::new(); |
| 2541 | while i < bytes.len() { |
| 2542 | let b = bytes[i]; |
| 2543 | if b == b'\\' && i + 1 < bytes.len() { |
| 2544 | match bytes[i + 1] { |
| 2545 | b'"' => out.push(b'"'), |
| 2546 | b'\\' => out.push(b'\\'), |
| 2547 | b'n' => out.push(b'\n'), |
| 2548 | b't' => out.push(b'\t'), |
| 2549 | b'r' => out.push(b'\r'), |
| 2550 | b'/' => out.push(b'/'), |
| 2551 | other => { out.push(b'\\'); out.push(other); } |
| 2552 | } |
| 2553 | i += 2; |
| 2554 | } else if b == b'"' { |
| 2555 | return Some(String::from_utf8_lossy(&out).to_string()); |
| 2556 | } else { |
| 2557 | out.push(b); |
| 2558 | i += 1; |
| 2559 | } |
| 2560 | } |
| 2561 | return None; |
| 2562 | } |
| 2563 | } |
| 2564 | |
| 2565 | /// Convert a JDAT DaticleMap to a minimal JSON string. |
| 2566 | /// |
| 2567 | /// This is used to build the LLM API request body without `serde`. |
| 2568 | /// Only handles the types we need: String, U64, Bool, Map, List. |
| 2569 | /// The shortest prefix worth a cache breakpoint, in characters. |
| 2570 | /// |
| 2571 | /// Anthropic will not cache a prefix below a per-model minimum, and says nothing |
| 2572 | /// when it declines -- the request simply reports no cache write. 512 tokens is |
| 2573 | /// the lowest of those minimums (Claude Opus 5; other models want 1024 or more), |
| 2574 | /// and four characters per token is the usual rough conversion, so a prefix |
| 2575 | /// shorter than this cannot cache on any model and the marker is not sent. |
| 2576 | pub(crate) const CACHE_MIN_PREFIX_CHARS: usize = 2048; |
| 2577 | |
| 2578 | /// Whether this model honours an explicit `cache_control` breakpoint. |
| 2579 | /// |
| 2580 | /// Claude is the case that needs one: Fireworks, DeepSeek and OpenAI cache |
| 2581 | /// automatically, and Anthropic does not. The model id is what selects the |
| 2582 | /// upstream model -- the host varies (direct, a router, Daimond's own gateway |
| 2583 | /// proxy) while the same Claude model behind any of them reads the same marker, |
| 2584 | /// so the id is what this gates on. Every id form is covered: the OpenRouter |
| 2585 | /// `anthropic/claude-...`, the Bedrock `anthropic.claude-...`, and the bare |
| 2586 | /// `claude-...`. |
| 2587 | pub(crate) fn model_caches_on_request(model: &str) -> bool { |
| 2588 | let m = model.to_ascii_lowercase(); |
| 2589 | m.contains("claude") || m.starts_with("anthropic/") || m.starts_with("anthropic.") |
| 2590 | } |
| 2591 | |
| 2592 | /// The Anthropic API version this client is written against. |
| 2593 | /// |
| 2594 | /// Pinned rather than tracking latest: the version header is what stops a |
| 2595 | /// breaking change to the wire shape arriving without a code change. |
| 2596 | pub(crate) const ANTHROPIC_VERSION: &str = "2023-06-01"; |
| 2597 | |
| 2598 | /// The smallest output cap a streamed thinking request is sent with. |
| 2599 | /// |
| 2600 | /// Well under the 128k ceiling every thinking-capable model offers, and enough |
| 2601 | /// room for the reasoning and the answer that follows it. See |
| 2602 | /// [`anthropic_max_tokens`](LlmClient::anthropic_max_tokens). |
| 2603 | pub(crate) const THINKING_MIN_MAX_TOKENS: u32 = 32_000; |
| 2604 | |
| 2605 | /// An Anthropic `text` content block, optionally carrying a cache breakpoint. |
| 2606 | /// |
| 2607 | /// # Arguments |
| 2608 | /// * `text` - The block's text, escaped here. |
| 2609 | /// * `cached` - Whether to attach an ephemeral `cache_control` marker. |
| 2610 | fn text_block(text: &str, cached: bool) -> String { |
| 2611 | if cached { |
| 2612 | fmt!("{{\"type\":\"text\",\"text\":\"{}\",\"cache_control\":{{\"type\":\"ephemeral\"}}}}", |
| 2613 | json_escape(text)) |
| 2614 | } else { |
| 2615 | fmt!("{{\"type\":\"text\",\"text\":\"{}\"}}", json_escape(text)) |
| 2616 | } |
| 2617 | } |
| 2618 | |
| 2619 | /// An Anthropic `image` content block, base64 source. |
| 2620 | /// |
| 2621 | /// The shape is the one the Messages API publishes: |
| 2622 | /// `{"type":"image","source":{"type":"base64","media_type":…,"data":…}}`. Key order matters to |
| 2623 | /// nothing but the fixture that pins it, and it is the documentation's order. |
| 2624 | /// |
| 2625 | /// A cache breakpoint may sit on an image block as on any other, and it has to be able to: the |
| 2626 | /// breakpoint caches everything up to the block it is on, so a message whose last block is the |
| 2627 | /// image would otherwise have no legal place to put one and would silently lose the cache. |
| 2628 | /// |
| 2629 | /// # Arguments |
| 2630 | /// * `img` - The image; its bytes are base64-encoded here, once per request. |
| 2631 | /// * `cached` - Whether to attach an ephemeral `cache_control` marker. |
| 2632 | fn image_block(img: &ImagePart, cached: bool) -> String { |
| 2633 | let mark = if cached { ",\"cache_control\":{\"type\":\"ephemeral\"}" } else { "" }; |
| 2634 | fmt!( |
| 2635 | "{{\"type\":\"image\",\"source\":{{\"type\":\"base64\",\"media_type\":\"{}\",\ |
| 2636 | \"data\":\"{}\"}}{}}}", |
| 2637 | img.media.mime(), img.base64(), mark) |
| 2638 | } |
| 2639 | |
| 2640 | /// A message's content as Anthropic content blocks, in order. |
| 2641 | /// |
| 2642 | /// The cache marker goes on the LAST block, because a breakpoint caches the prefix up to and |
| 2643 | /// including the block it sits on -- putting it on the first of several would leave the rest of |
| 2644 | /// the message re-billed on every turn, which is the opposite of what marking it was for. |
| 2645 | /// |
| 2646 | /// # Arguments |
| 2647 | /// * `content` - What the message carries. |
| 2648 | /// * `cached` - Whether this message is a cache breakpoint. |
| 2649 | fn anthropic_blocks(content: &MessageContent, cached: bool) -> Vec<String> { |
| 2650 | let parts: Vec<&ContentPart> = match content { |
| 2651 | MessageContent::Text(s) => { |
| 2652 | // An empty text block is rejected outright, where the OpenAI side simply carries the |
| 2653 | // empty string through. |
| 2654 | return if s.is_empty() { Vec::new() } else { vec![text_block(s, cached)] }; |
| 2655 | }, |
| 2656 | MessageContent::Parts(parts) => parts.iter().collect(), |
| 2657 | }; |
| 2658 | let last = parts.len().saturating_sub(1); |
| 2659 | let mut out = Vec::with_capacity(parts.len()); |
| 2660 | for (i, p) in parts.iter().enumerate() { |
| 2661 | match p { |
| 2662 | ContentPart::Text(t) if t.is_empty() => {}, |
| 2663 | ContentPart::Text(t) => out.push(text_block(t, cached && i == last)), |
| 2664 | ContentPart::Image(m) => out.push(image_block(m, cached && i == last)), |
| 2665 | } |
| 2666 | } |
| 2667 | out |
| 2668 | } |
| 2669 | |
| 2670 | /// The `user` message that carries images lifted out of a run of OpenAI tool replies. |
| 2671 | /// |
| 2672 | /// The leading sentence is not decoration: without it the model receives images with no statement |
| 2673 | /// of where they came from, and the turn reads as the user having pasted them. |
| 2674 | /// |
| 2675 | /// # Arguments |
| 2676 | /// * `blocks` - Ready-made `image_url` parts, in the order the tools returned them. |
| 2677 | fn tool_image_message(blocks: &[String]) -> String { |
| 2678 | fmt!( |
| 2679 | "{{\"role\":\"user\",\"content\":[{{\"type\":\"text\",\"text\":\"{}\"}},{}]}}", |
| 2680 | json_escape("[The images returned by the tool calls above, in order.]"), |
| 2681 | blocks.join(",")) |
| 2682 | } |
| 2683 | |
| 2684 | /// A message's content as the OpenAI `content` field: a JSON string, or an array of parts. |
| 2685 | /// |
| 2686 | /// A bare string whenever there is no image, because that is what every OpenAI-compatible router |
| 2687 | /// has always been sent and the array form buys nothing. With an image it becomes the documented |
| 2688 | /// parts array, where an image is `{"type":"image_url","image_url":{"url":"data:…;base64,…"}}` -- |
| 2689 | /// the `url` field takes "a URL or a base64 encoded data URL", so the bytes ride in an RFC 2397 |
| 2690 | /// data URL rather than in a field of their own. |
| 2691 | /// |
| 2692 | /// # Arguments |
| 2693 | /// * `content` - What the message carries. |
| 2694 | fn openai_content(content: &MessageContent) -> String { |
| 2695 | match content { |
| 2696 | MessageContent::Text(s) => fmt!("\"{}\"", json_escape(s)), |
| 2697 | MessageContent::Parts(parts) => { |
| 2698 | let items: Vec<String> = parts.iter().map(|p| match p { |
| 2699 | ContentPart::Text(t) => |
| 2700 | fmt!("{{\"type\":\"text\",\"text\":\"{}\"}}", json_escape(t)), |
| 2701 | ContentPart::Image(m) => fmt!( |
| 2702 | "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:{};base64,{}\"}}}}", |
| 2703 | m.media.mime(), m.base64()), |
| 2704 | }).collect(); |
| 2705 | fmt!("[{}]", items.join(",")) |
| 2706 | }, |
| 2707 | } |
| 2708 | } |
| 2709 | |
| 2710 | /// Whether `model` takes Anthropic's adaptive thinking configuration. |
| 2711 | /// |
| 2712 | /// Adaptive is the only form worth sending: `budget_tokens` is removed on every |
| 2713 | /// model released since Opus 4.7 and returns a 400 there, and deprecated on the |
| 2714 | /// two before it. So a model that does not take adaptive is sent no thinking |
| 2715 | /// configuration at all rather than a guessed budget -- an older model then |
| 2716 | /// simply answers without thinking, which is what it did before this existed. |
| 2717 | /// |
| 2718 | /// The list is explicit rather than a `claude-` prefix test, because the whole |
| 2719 | /// point of the gate is that the newer families and the older ones disagree |
| 2720 | /// about what the parameter even means. |
| 2721 | /// |
| 2722 | /// # Arguments |
| 2723 | /// * `model` - The model id, in any of the forms a caller can configure. |
| 2724 | pub(crate) fn model_takes_adaptive_thinking(model: &str) -> bool { |
| 2725 | let m = model.to_ascii_lowercase(); |
| 2726 | const ADAPTIVE: [&str; 8] = [ |
| 2727 | "claude-fable-5", |
| 2728 | "claude-mythos-5", |
| 2729 | "claude-opus-5", |
| 2730 | "claude-opus-4-8", |
| 2731 | "claude-opus-4-7", |
| 2732 | "claude-opus-4-6", |
| 2733 | "claude-sonnet-5", |
| 2734 | "claude-sonnet-4-6", |
| 2735 | ]; |
| 2736 | ADAPTIVE.iter().any(|id| m.contains(id)) |
| 2737 | } |
| 2738 | |
| 2739 | /// Whether `model` can be shown an image. |
| 2740 | /// |
| 2741 | /// The test is a list of what is KNOWN BLIND, not a list of what is known to see, and the default |
| 2742 | /// is that a model sees. That direction is chosen deliberately: nearly every model a user would |
| 2743 | /// configure today is multimodal, an allow-list would refuse every model released after this line |
| 2744 | /// was written, and the cost of getting it wrong in this direction is one clear error from |
| 2745 | /// [`LlmClient::vision_error`] rather than a refusal to try. The names are matched as substrings |
| 2746 | /// because a router spells the same model half a dozen ways (`openai/gpt-3.5-turbo`, |
| 2747 | /// `gpt-3.5-turbo-0125`), and the family is what is blind, not the spelling. |
| 2748 | /// |
| 2749 | /// # Arguments |
| 2750 | /// * `model` - The model id, in whatever form the user configured it. |
| 2751 | pub(crate) fn model_can_see(model: &str) -> bool { |
| 2752 | let m = model.to_ascii_lowercase(); |
| 2753 | const BLIND: [&str; 8] = [ |
| 2754 | "gpt-3.5", |
| 2755 | "text-davinci", |
| 2756 | "o1-mini", |
| 2757 | "o1-preview", |
| 2758 | "claude-instant", |
| 2759 | "claude-1", |
| 2760 | "claude-2", |
| 2761 | "embedding", |
| 2762 | ]; |
| 2763 | !BLIND.iter().any(|id| m.contains(id)) |
| 2764 | } |
| 2765 | |
| 2766 | /// Translate an OpenAI-shaped tool array into the Anthropic one. |
| 2767 | /// |
| 2768 | /// `[{"type":"function","function":{"name":…,"description":…,"parameters":{…}}}]` |
| 2769 | /// becomes `[{"name":…,"description":…,"input_schema":{…}}]`. The schema |
| 2770 | /// itself is JSON Schema in both, so it is carried through verbatim; only the |
| 2771 | /// wrapper differs. A definition missing a name or a schema is dropped rather |
| 2772 | /// than sent half-built, since the API would reject the whole request for it. |
| 2773 | /// |
| 2774 | /// # Arguments |
| 2775 | /// * `tools` - The OpenAI-shaped tool array, as JSON text. |
| 2776 | fn openai_tools_to_anthropic(tools: &str) -> String { |
| 2777 | let mut out: Vec<String> = Vec::new(); |
| 2778 | for elem in split_top_level_objects(tools) { |
| 2779 | // The function object, so `name` and `description` are read from it |
| 2780 | // rather than from a property of the schema that happens to share a key. |
| 2781 | let f = match find_json_object(&elem, "function") { |
| 2782 | Some(f) => f, |
| 2783 | None => elem.clone(), |
| 2784 | }; |
| 2785 | let name = match extract_json_string(&f, "name") { |
| 2786 | Some(n) if !n.is_empty() => n, |
| 2787 | _ => continue, |
| 2788 | }; |
| 2789 | let schema = match find_json_object(&f, "parameters") |
| 2790 | .or_else(|| find_json_object(&f, "input_schema")) |
| 2791 | { |
| 2792 | Some(s) => s, |
| 2793 | None => continue, |
| 2794 | }; |
| 2795 | let desc = extract_json_string(&f, "description").unwrap_or_default(); |
| 2796 | out.push(fmt!( |
| 2797 | "{{\"name\":\"{}\",\"description\":\"{}\",\"input_schema\":{}}}", |
| 2798 | json_escape(&name), json_escape(&desc), schema)); |
| 2799 | } |
| 2800 | fmt!("[{}]", out.join(",")) |
| 2801 | } |
| 2802 | |
| 2803 | /// The character length of a message's payload, as a stand-in for its tokens. |
| 2804 | /// |
| 2805 | /// An image counts its own bytes here rather than its token cost, because this figure decides |
| 2806 | /// only WHERE the prompt-cache breakpoints go, and what matters for that is what the message |
| 2807 | /// weighs on the wire -- which for an image is its bytes. |
| 2808 | fn message_len(msg: &ChatMessage) -> usize { |
| 2809 | let content = msg.content(); |
| 2810 | let mut n = content.text_len() + content.images().map(|i| i.data.len()).sum::<usize>(); |
| 2811 | if let ChatMessage::Assistant { tool_calls, .. } = msg { |
| 2812 | n += tool_calls.iter().map(|tc| tc.name.len() + tc.arguments.len()).sum::<usize>(); |
| 2813 | } |
| 2814 | n |
| 2815 | } |
| 2816 | |
| 2817 | /// Serialise a `ChatMessage` with an Anthropic prompt-cache breakpoint on it. |
| 2818 | /// |
| 2819 | /// The marker only exists on a content *block*, so the content becomes a |
| 2820 | /// one-element array rather than a bare string. Only the system and user roles |
| 2821 | /// are given this form; anything else falls back to the plain serialisation, so |
| 2822 | /// a caller that marks the wrong message loses the cache rather than the turn. |
| 2823 | fn message_to_json_cached(msg: &ChatMessage, open: &std::collections::HashSet<String>) -> String { |
| 2824 | let (role, content) = match msg { |
| 2825 | ChatMessage::System { content } => ("system", content), |
| 2826 | ChatMessage::User { content } => ("user", content), |
| 2827 | _ => return message_to_json(msg, open), |
| 2828 | }; |
| 2829 | // With an image in it the content is already an array, and the marker goes on the last block |
| 2830 | // rather than replacing the whole thing with one text block -- which would drop the image. |
| 2831 | // |
| 2832 | // On this side the marker goes on the last TEXT part and nowhere else. `cache_control` is an |
| 2833 | // Anthropic field that routers pass through; putting it inside an `image_url` part would put |
| 2834 | // an unrecognised key somewhere every OpenAI-compatible server validates strictly, to buy a |
| 2835 | // cache hit on a request that is mostly image bytes anyway. A message ending in an image |
| 2836 | // simply is not a breakpoint. |
| 2837 | if let MessageContent::Parts(parts) = content { |
| 2838 | let last = parts.iter().rposition(|p| matches!(p, ContentPart::Text(_))) |
| 2839 | .unwrap_or(usize::MAX); |
| 2840 | let items: Vec<String> = parts.iter().enumerate().map(|(i, p)| match p { |
| 2841 | ContentPart::Text(t) if i == last => fmt!( |
| 2842 | "{{\"type\":\"text\",\"text\":\"{}\",\"cache_control\":{{\"type\":\"ephemeral\"}}}}", |
| 2843 | json_escape(t)), |
| 2844 | ContentPart::Text(t) => fmt!("{{\"type\":\"text\",\"text\":\"{}\"}}", json_escape(t)), |
| 2845 | ContentPart::Image(m) => fmt!( |
| 2846 | "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:{};base64,{}\"}}}}", |
| 2847 | m.media.mime(), m.base64()), |
| 2848 | }).collect(); |
| 2849 | return fmt!("{{\"role\":\"{}\",\"content\":[{}]}}", role, items.join(",")); |
| 2850 | } |
| 2851 | fmt!( |
| 2852 | "{{\"role\":\"{}\",\"content\":[{{\"type\":\"text\",\"text\":\"{}\",\ |
| 2853 | \"cache_control\":{{\"type\":\"ephemeral\"}}}}]}}", |
| 2854 | role, json_escape(&content.as_text())) |
| 2855 | } |
| 2856 | |
| 2857 | /// The arguments a `say` call is REPLAYED with, or `None` for every other tool. |
| 2858 | /// |
| 2859 | /// **`say` was a tool and is not one any more.** Folding is written into the model's own prose as |
| 2860 | /// a `<details>` element now -- see [`sent_text_len`] -- and no model is offered `say` or can call |
| 2861 | /// it. This stays because STORED CONVERSATIONS still carry `say` tool_calls: every one of them |
| 2862 | /// travels on every request for the life of that conversation, so a stripper deleted here is not a |
| 2863 | /// tidying, it is every old folded answer going out in full again, on exactly the conversations |
| 2864 | /// the feature existed to make cheap. |
| 2865 | /// |
| 2866 | /// `say` answered at two depths: a summary the user read at once and a detail behind a fold. The |
| 2867 | /// detail is for a person, once. Left in the transcript it would be re-sent on every later request |
| 2868 | /// for the life of the conversation -- so a model that explained at length would charge for that |
| 2869 | /// explanation again on every turn, whether or not anybody looked at it twice. |
| 2870 | /// |
| 2871 | /// So the wire carries the summary and a note in the detail's place. Nothing is lost: the browser |
| 2872 | /// keeps the whole call in its own transcript record, which is what the fold opens and what |
| 2873 | /// survives a reload. The local record and the payload simply stop being the same thing. |
| 2874 | /// |
| 2875 | /// ONE FUNCTION FOR BOTH DIALECTS. The OpenAI body escapes these arguments into a string and the |
| 2876 | /// Anthropic body embeds them as JSON, so the two sites look nothing alike -- and a rule applied at |
| 2877 | /// one and not the other would mean the same conversation cost different amounts through different |
| 2878 | /// endpoints, silently. |
| 2879 | /// |
| 2880 | /// It costs ONE cache miss, at the request after the call: the prefix changes once where the |
| 2881 | /// arguments shrink, and is stable from then on. |
| 2882 | /// |
| 2883 | /// # Arguments |
| 2884 | /// * `name` - The tool the call names. |
| 2885 | /// * `arguments` - Its arguments, as the model wrote them. |
| 2886 | fn strip_said(name: &str, arguments: &str, open: bool) -> Option<String> { |
| 2887 | if name != "say" { |
| 2888 | return None; |
| 2889 | } |
| 2890 | // OPEN MEANS THE USER IS READING IT, so the model holds it too. Their own gesture decides the |
| 2891 | // working set, and the two things that ought to agree -- what is on their screen and what it |
| 2892 | // knows -- do. |
| 2893 | if open { |
| 2894 | return None; |
| 2895 | } |
| 2896 | // A call whose summary cannot be read is left alone. It is malformed, and rewriting a |
| 2897 | // malformed call would replace one problem the model can see with one it cannot. |
| 2898 | let summary = extract_json_string(arguments, "summary")?; |
| 2899 | let n = extract_json_string(arguments, "detail") |
| 2900 | .map(|d| d.chars().count()) |
| 2901 | .unwrap_or(0); |
| 2902 | Some(fmt!( |
| 2903 | "{{\"summary\":\"{}\",\"detail\":\"{}\"}}", |
| 2904 | json_escape(&summary), |
| 2905 | json_escape(&fmt!( |
| 2906 | "[folded to the user, {} characters. They have it on screen; you no longer carry it. \ |
| 2907 | Ask them, or read the file you wrote it from, if you need it again.]", n)))) |
| 2908 | } |
| 2909 | |
| 2910 | /// How many bytes of `arguments` this call will actually put on the wire. |
| 2911 | /// |
| 2912 | /// **The compaction trigger's half of [`strip_said`], and it calls that function rather than |
| 2913 | /// restating its rule.** [`crate::agent::compact::msg_bytes`] sized a `say` with the length the |
| 2914 | /// model wrote, so the trigger measured a conversation nobody was going to send: every closed |
| 2915 | /// fold in it was counted at full length, the budget was spent on bytes that leave at |
| 2916 | /// serialisation, and a conversation was folded earlier than it needed to be. The |
| 2917 | /// [`Gauge`](crate::agent::compact::Gauge) absorbed part of the error by recalibrating tokens-per-byte against the provider's real |
| 2918 | /// `prompt_tokens` -- but that ratio is one number for the whole conversation, so the correction |
| 2919 | /// was paid for by every other message's estimate. |
| 2920 | /// |
| 2921 | /// Whether the detail travels depends on the fold state, which is why `open` is asked for here |
| 2922 | /// and not decided here: an OPEN fold is one the user is reading and its detail goes out in full. |
| 2923 | /// |
| 2924 | /// # Arguments |
| 2925 | /// * `name` - The tool the call names. |
| 2926 | /// * `arguments` - Its arguments, as the model wrote them. |
| 2927 | /// * `open` - Whether this call's fold is open on screen. |
| 2928 | pub fn sent_args_len(name: &str, arguments: &str, open: bool) -> usize { |
| 2929 | match strip_said(name, arguments, open) { |
| 2930 | Some(replayed) => replayed.len(), |
| 2931 | None => arguments.len(), |
| 2932 | } |
| 2933 | } |
| 2934 | |
| 2935 | |
| 2936 | // ┌───────────────────────────────────────────────────────────────┐ |
| 2937 | // │ The two-depth answer, written inline │ |
| 2938 | // └───────────────────────────────────────────────────────────────┘ |
| 2939 | // |
| 2940 | // `say` answered at two depths through a tool call. The model now writes the same two depths |
| 2941 | // into its own prose as a `<details>` element, and this is the reading half of it: the same |
| 2942 | // economy applied to text rather than to tool arguments. `strip_said` above stays exactly as it |
| 2943 | // is -- stored conversations carry `say` tool_logs, and a reader deleted is an answer that |
| 2944 | // renders as nothing. The shared shape is pinned in `dev/CONTRACT_FOLD.md`; the fixture both |
| 2945 | // languages are tested against is `dev/fixtures/fold_keys.json`. |
| 2946 | |
| 2947 | /// The note a stripped fold leaves in place of its body. |
| 2948 | /// |
| 2949 | /// [`strip_said`]'s wording, near enough, so the two paths read alike to a model that may hold |
| 2950 | /// both in one conversation. The exact string is pinned by the fixture. |
| 2951 | /// |
| 2952 | /// # Arguments |
| 2953 | /// * `n` - Characters of the fold's trimmed body, as the user still has it on screen. |
| 2954 | fn fold_note(n: usize) -> String { |
| 2955 | fmt!("[folded to the user, {} characters. They have it on screen; you no longer carry it. \ |
| 2956 | Ask them if you need it again.]", n) |
| 2957 | } |
| 2958 | |
| 2959 | /// One `<details>` fold found in one assistant message's text. |
| 2960 | /// |
| 2961 | /// Byte offsets rather than copied strings, so the strip rewrites in place and every character |
| 2962 | /// outside the body stays exactly where the model put it -- including the blank lines the |
| 2963 | /// renderer needs, which a reconstructed element would have to get right a second time. |
| 2964 | struct Fold { |
| 2965 | ord: usize, // 0-based over the real folds of this message |
| 2966 | summary: String, // tags removed, whitespace runs collapsed to one space |
| 2967 | body: std::ops::Range<usize>, // the TRIMMED body, as byte offsets into the text |
| 2968 | chars: usize, // characters of that trimmed body |
| 2969 | closed: bool, // whether a `</details>` was found for it |
| 2970 | } |
| 2971 | |
| 2972 | impl Fold { |
| 2973 | /// The one name the Rust and JS halves must agree on: `"<ordinal>:<summary>"`. |
| 2974 | /// |
| 2975 | /// No hash and no message identity, deliberately -- see `dev/CONTRACT_FOLD.md` §2. Two |
| 2976 | /// messages can therefore share a key, and the only consequence is that opening one fold |
| 2977 | /// sends the body of a same-labelled, same-ordinal fold in another. Nothing renders wrong. |
| 2978 | fn key(&self) -> String { |
| 2979 | fmt!("{}:{}", self.ord, self.summary) |
| 2980 | } |
| 2981 | } |
| 2982 | |
| 2983 | /// The fence a line opens or closes, as `(character, run length, whether an info string follows)`. |
| 2984 | /// |
| 2985 | /// CommonMark's rule, to the part that matters here: up to three leading spaces, then three or |
| 2986 | /// more backticks or tildes. A run with an info string after it can only OPEN a fence, never |
| 2987 | /// close one, which is what keeps ```` ```html ```` from closing the fence it opened. |
| 2988 | fn fence_run(line: &str) -> Option<(u8, usize, bool)> { |
| 2989 | let t = line.trim_end_matches(['\n', '\r']); |
| 2990 | let lead = t.len() - t.trim_start_matches(' ').len(); |
| 2991 | if lead > 3 { |
| 2992 | return None; |
| 2993 | } |
| 2994 | let rest = &t[lead..]; |
| 2995 | let ch = match rest.as_bytes().first() { |
| 2996 | Some(&c) if c == b'`' || c == b'~' => c, |
| 2997 | _ => return None, |
| 2998 | }; |
| 2999 | let n = rest.bytes().take_while(|&b| b == ch).count(); |
| 3000 | if n < 3 { |
| 3001 | return None; |
| 3002 | } |
| 3003 | Some((ch, n, !rest[n..].trim().is_empty())) |
| 3004 | } |
| 3005 | |
| 3006 | /// The byte ranges of every line inside a fenced code region, the fence lines included. |
| 3007 | /// |
| 3008 | /// **This is the single most likely source of a wrong strip.** A `<details>` inside a fence is |
| 3009 | /// literal text the model is SHOWING the user -- the markup itself, quoted. It is not a fold, it |
| 3010 | /// takes no ordinal, and rewriting it would edit the example out from under the person reading |
| 3011 | /// it. The renderer is already safe there because `marked` escapes it; nothing but this function |
| 3012 | /// makes the stripper safe. |
| 3013 | fn fenced_spans(text: &str) -> Vec<(usize, usize)> { |
| 3014 | let mut out: Vec<(usize, usize)> = Vec::new(); |
| 3015 | let mut open: Option<(u8, usize)> = None; |
| 3016 | let mut at = 0usize; |
| 3017 | for line in text.split_inclusive('\n') { |
| 3018 | let end = at + line.len(); |
| 3019 | match (fence_run(line), open) { |
| 3020 | // A close must match the character it opened with, be at least as long, and carry no |
| 3021 | // info string. |
| 3022 | (Some((ch, n, info)), Some((c, k))) if ch == c && n >= k && !info => { |
| 3023 | out.push((at, end)); |
| 3024 | open = None; |
| 3025 | }, |
| 3026 | (Some((ch, n, _)), None) => { |
| 3027 | out.push((at, end)); |
| 3028 | open = Some((ch, n)); |
| 3029 | }, |
| 3030 | // Anything else inside a fence is fenced; anything else outside one is prose. |
| 3031 | _ => if open.is_some() { out.push((at, end)); }, |
| 3032 | } |
| 3033 | at = end; |
| 3034 | } |
| 3035 | out |
| 3036 | } |
| 3037 | |
| 3038 | // ── The seam the APP places ───────────────────────────────────────────────────────────────── |
| 3039 | // |
| 3040 | // Three wordings by three authors asked the model to write the `<details>` itself, and 5 answers |
| 3041 | // in 76 carried one -- `dev/PROMPT_NOTES.md` §5 and `dev/REGISTER_NOTES.md` §11, across two |
| 3042 | // register kinds on the shape of question the note exists for. So the markup is the app's now |
| 3043 | // and the model writes one line: `Fold:` and a sentence or two of what the working below it |
| 3044 | // concludes. [`seam_text`] turns that line into exactly the element a model used to write, and |
| 3045 | // [`folds`] below never learns that anything is different. |
| 3046 | // |
| 3047 | // **The refusals are the point, not the fold.** A length threshold applied blindly produces |
| 3048 | // FOLD-ALL, which `dev/CONTRACT_FOLD.md` §5 calls worse than no control at all. Here the two |
| 3049 | // failures no wording could prevent are unreachable instead: too little above the line, too few |
| 3050 | // words in the summary, or too little below it, and there is no fold -- the sentence is left as |
| 3051 | // prose and nothing is hidden. |
| 3052 | // |
| 3053 | // `www/js/render.js`'s `seamText` is the other half and the two must agree character for |
| 3054 | // character, for `Fold::key`'s reason: the page names the fold the reader opened and this side |
| 3055 | // matches the name. `dev/fixtures/fold_seam.json` is what both are tested against. |
| 3056 | |
| 3057 | /// Enough above the seam to be an answer. The same forty characters `dev/probe_notes.mjs` calls |
| 3058 | /// FOLD-ALL below, counted the same way -- whitespace collapsed, ends trimmed -- because it is |
| 3059 | /// the same rule and not a second opinion about it. |
| 3060 | const SEAM_LEAD_MIN: usize = 40; |
| 3061 | /// Enough below it to be worth a control, which is `dev/CONTRACT_FOLD.md` §5's carve-out. |
| 3062 | const SEAM_BODY_MIN: usize = 240; |
| 3063 | /// A summary is a summary and not a label -- §13. |
| 3064 | const SEAM_WORDS_MIN: usize = 6; |
| 3065 | |
| 3066 | /// The summary a `Fold:` line carries, or `None` if the line is not one. |
| 3067 | /// |
| 3068 | /// Generous in what it accepts, because a model that has understood the note and reached for a |
| 3069 | /// heading or for bold should not lose its fold over the decoration: `Fold:`, `**Fold:**`, |
| 3070 | /// `## Fold:` and `_Fold:_` all seam. What it will not accept is an indented line, which is a |
| 3071 | /// code block, or one whose summary is empty. |
| 3072 | fn seam_line(line: &str) -> Option<String> { |
| 3073 | let raw = line.trim_end_matches(['\n', '\r']); |
| 3074 | if raw.starts_with(" ") || raw.starts_with('\t') { |
| 3075 | return None; |
| 3076 | } |
| 3077 | let mut s = raw.trim_start_matches(' '); |
| 3078 | // A heading marker, then at least one space, or `#Fold:` would seam as a heading. |
| 3079 | let hashes = s.bytes().take_while(|&b| b == b'#').count(); |
| 3080 | if (1..=6).contains(&hashes) && s[hashes..].starts_with(' ') { |
| 3081 | s = s[hashes..].trim_start_matches(' '); |
| 3082 | } |
| 3083 | let lead = emphasis_at(s); |
| 3084 | s = &s[lead.len()..]; |
| 3085 | if s.len() < 4 || !s[..4].eq_ignore_ascii_case("fold") { |
| 3086 | return None; |
| 3087 | } |
| 3088 | let after = s[4..].trim_start_matches([' ', '\t']); |
| 3089 | let mut rest = match after.strip_prefix(':') { |
| 3090 | Some(r) => r, |
| 3091 | None => return None, |
| 3092 | }; |
| 3093 | let shut = emphasis_at(rest); |
| 3094 | if !shut.is_empty() { |
| 3095 | rest = &rest[shut.len()..]; |
| 3096 | } else if !lead.is_empty() { |
| 3097 | let t = rest.trim_end(); |
| 3098 | let tail = emphasis_end(t); |
| 3099 | if !tail.is_empty() { |
| 3100 | rest = &t[..t.len() - tail.len()]; |
| 3101 | } |
| 3102 | } |
| 3103 | let sum = rest.split_whitespace().collect::<Vec<_>>().join(" "); |
| 3104 | if sum.is_empty() { None } else { Some(sum) } |
| 3105 | } |
| 3106 | |
| 3107 | /// The emphasis run a markdown span opens with, longest first. |
| 3108 | fn emphasis_at(s: &str) -> &str { |
| 3109 | for m in ["**", "__", "*", "_"] { |
| 3110 | if s.starts_with(m) { |
| 3111 | return &s[..m.len()]; |
| 3112 | } |
| 3113 | } |
| 3114 | "" |
| 3115 | } |
| 3116 | |
| 3117 | /// The emphasis run a markdown span closes with, longest first. |
| 3118 | fn emphasis_end(s: &str) -> &str { |
| 3119 | for m in ["**", "__", "*", "_"] { |
| 3120 | if s.ends_with(m) { |
| 3121 | return &s[s.len() - m.len()..]; |
| 3122 | } |
| 3123 | } |
| 3124 | "" |
| 3125 | } |
| 3126 | |
| 3127 | /// How much of `t` a reader would actually see, counted as the page counts it. |
| 3128 | /// |
| 3129 | /// UTF-16 units rather than characters, because the other half of this is JavaScript and a |
| 3130 | /// boundary the two halves disagreed on would fold on one side and not the other. |
| 3131 | fn seam_visible(t: &str) -> usize { |
| 3132 | t.split_whitespace().collect::<Vec<_>>().join(" ").encode_utf16().count() |
| 3133 | } |
| 3134 | |
| 3135 | /// `text` with the model's `Fold:` line turned into the element it stands for, or `None` where |
| 3136 | /// there is nothing to do. |
| 3137 | /// |
| 3138 | /// Four outcomes and only one of them is a fold: no line at all, or a model that wrote its own |
| 3139 | /// `<details>`, leaves the text alone; a line on a qualifying answer becomes one top-level fold, |
| 3140 | /// blank lines and all; a line on an answer that does not qualify loses its `Fold:` and stays as |
| 3141 | /// prose, so nothing is hidden and nothing is lost. The page has a fourth, which this side |
| 3142 | /// cannot have: mid-stream it holds the line back rather than showing a fold that might unwind. |
| 3143 | pub fn seam_text(text: &str) -> Option<String> { |
| 3144 | if !text.to_ascii_lowercase().contains("fold") { |
| 3145 | return None; |
| 3146 | } |
| 3147 | // A model that wrote the markup itself has already placed its seam. |
| 3148 | if text.contains("<details") && !folds(text).is_empty() { |
| 3149 | return None; |
| 3150 | } |
| 3151 | let fenced = fenced_spans(text); |
| 3152 | let hidden = |p: usize| fenced.iter().any(|&(a, b)| p >= a && p < b); |
| 3153 | // Every `Fold:` line outside a fence: where it starts, how long it is, and what it says. |
| 3154 | let mut marks: Vec<(usize, usize, String)> = Vec::new(); |
| 3155 | let mut at = 0usize; |
| 3156 | for line in text.split_inclusive('\n') { |
| 3157 | let bare = line.trim_end_matches(['\n', '\r']); |
| 3158 | if !hidden(at) { |
| 3159 | if let Some(sum) = seam_line(bare) { |
| 3160 | marks.push((at, bare.len(), sum)); |
| 3161 | } |
| 3162 | } |
| 3163 | at += line.len(); |
| 3164 | } |
| 3165 | let (start, len, sum) = match marks.first() { |
| 3166 | Some(m) => (m.0, m.1, m.2.clone()), |
| 3167 | None => return None, |
| 3168 | }; |
| 3169 | // Only the first line is the seam. A second would otherwise reach the reader with its |
| 3170 | // marker still on it, so every later one loses the marker and stays where it is. |
| 3171 | let bare_from = |from: usize| -> String { |
| 3172 | let mut out = String::with_capacity(text.len()); |
| 3173 | let mut cut = from; |
| 3174 | for &(a, n, ref s) in &marks { |
| 3175 | if a < from { |
| 3176 | continue; |
| 3177 | } |
| 3178 | out.push_str(&text[cut..a]); |
| 3179 | out.push_str(s); |
| 3180 | cut = a + n; |
| 3181 | } |
| 3182 | out.push_str(&text[cut..]); |
| 3183 | out |
| 3184 | }; |
| 3185 | let above = &text[..start]; |
| 3186 | // The line's own newline goes with the line. |
| 3187 | let rest_at = (start + len + 1).min(text.len()); |
| 3188 | let body = &text[rest_at..]; |
| 3189 | // A summary carrying the very tags this builds would close the element early and leave the |
| 3190 | // rest of the answer outside it. |
| 3191 | let tagged = sum.to_ascii_lowercase(); |
| 3192 | if tagged.contains("<summary") || tagged.contains("</summary") |
| 3193 | || tagged.contains("<details") || tagged.contains("</details") |
| 3194 | || sum.split_whitespace().count() < SEAM_WORDS_MIN |
| 3195 | || seam_visible(above) < SEAM_LEAD_MIN |
| 3196 | || seam_visible(body) < SEAM_BODY_MIN |
| 3197 | { |
| 3198 | return Some(bare_from(0)); |
| 3199 | } |
| 3200 | Some(fmt!("{}\n\n<details>\n<summary>{}</summary>\n\n{}\n\n</details>", |
| 3201 | above.trim_end(), sum, bare_from(rest_at).trim())) |
| 3202 | } |
| 3203 | |
| 3204 | /// [`seam_text`] applied, for a caller that only wants the text back. |
| 3205 | /// |
| 3206 | /// The answer is stored seamed rather than seamed on the way out, so the element exists exactly |
| 3207 | /// once and everything downstream -- the strip, the compactor, a reload of the thread a year |
| 3208 | /// later -- meets an ordinary fold and nothing has to know about a marker line. |
| 3209 | pub fn seamed(text: String) -> String { |
| 3210 | match seam_text(&text) { |
| 3211 | Some(t) => t, |
| 3212 | None => text, |
| 3213 | } |
| 3214 | } |
| 3215 | |
| 3216 | /// Every real fold in one assistant message's text, in document order. |
| 3217 | /// |
| 3218 | /// Four shapes are deliberately NOT folds, and each one is a case in the fixture. A `<details>` |
| 3219 | /// in a fence is quoted markup. A `<details>` with no `<summary>` has no label, so it has no key |
| 3220 | /// and could not be matched against the open set anyway. A `<details>` NESTED inside another is |
| 3221 | /// carried away by its parent's strip, so a key of its own would name a body that no longer |
| 3222 | /// exists. A `<details>` with no `</details>` is a fold still being written: it keys, because the |
| 3223 | /// ordinal it takes is settled the moment its summary is, but it is left alone by the strip -- |
| 3224 | /// see [`strip_folds`]. |
| 3225 | /// |
| 3226 | /// **Elements are paired by DEPTH, not by the first closing tag that turns up.** The browser's |
| 3227 | /// parser nests correctly, so a scanner that closed an outer fold at its child's `</details>` |
| 3228 | /// would disagree with the renderer about where the fold ENDS -- and then replace the wrong span |
| 3229 | /// of text, cutting the body short and leaving the remainder of the element dangling in the |
| 3230 | /// payload. Measured: the shared fixture's nested case gives a 48-character body under |
| 3231 | /// first-close pairing and the correct 67 under this one. |
| 3232 | fn folds(text: &str) -> Vec<Fold> { |
| 3233 | let fenced = fenced_spans(text); |
| 3234 | let hidden = |p: usize| fenced.iter().any(|&(a, b)| p >= a && p < b); |
| 3235 | // The next occurrence of `tag` at or after `from` that is not inside a fence. |
| 3236 | let next = |from: usize, tag: &str| -> Option<usize> { |
| 3237 | let mut at = from; |
| 3238 | while at < text.len() { |
| 3239 | match text[at..].find(tag) { |
| 3240 | Some(p) => { |
| 3241 | let abs = at + p; |
| 3242 | if !hidden(abs) { |
| 3243 | return Some(abs); |
| 3244 | } |
| 3245 | at = abs + tag.len(); |
| 3246 | }, |
| 3247 | None => return None, |
| 3248 | } |
| 3249 | } |
| 3250 | None |
| 3251 | }; |
| 3252 | let mut out: Vec<Fold> = Vec::new(); |
| 3253 | let mut at = 0usize; |
| 3254 | while at < text.len() { |
| 3255 | let open = match next(at, "<details") { |
| 3256 | Some(p) => p, |
| 3257 | None => break, |
| 3258 | }; |
| 3259 | // Walk to the MATCHING close, counting depth. `inner` is where the first child element |
| 3260 | // starts, which is the bound on how far the label may be looked for. |
| 3261 | let mut depth = 1usize; |
| 3262 | let mut scan = open + "<details".len(); |
| 3263 | let mut close: Option<usize> = None; |
| 3264 | let mut inner: Option<usize> = None; |
| 3265 | while depth > 0 { |
| 3266 | // `"<details"` cannot match inside `"</details>"`, so the two searches never see the |
| 3267 | // same tag twice. |
| 3268 | match (next(scan, "<details"), next(scan, "</details>")) { |
| 3269 | (Some(o), Some(c)) if o < c => { |
| 3270 | depth += 1; |
| 3271 | if inner.is_none() { |
| 3272 | inner = Some(o); |
| 3273 | } |
| 3274 | scan = o + "<details".len(); |
| 3275 | }, |
| 3276 | (_, Some(c)) => { |
| 3277 | depth -= 1; |
| 3278 | scan = c + "</details>".len(); |
| 3279 | if depth == 0 { |
| 3280 | close = Some(c); |
| 3281 | } |
| 3282 | }, |
| 3283 | // Nothing closes it: the model is still writing. |
| 3284 | (_, None) => break, |
| 3285 | } |
| 3286 | } |
| 3287 | // Without a close the element runs to the end of what has been written so far, which is |
| 3288 | // what a fold looks like part way through a stream. |
| 3289 | let limit = close.unwrap_or(text.len()); |
| 3290 | // Where scanning resumes whether or not this element turns out to be a fold. Past the |
| 3291 | // WHOLE element, which is what keeps a nested fold from taking an ordinal of its own. |
| 3292 | let after = match close { |
| 3293 | Some(c) => c + "</details>".len(), |
| 3294 | None => text.len(), |
| 3295 | }; |
| 3296 | // The label must belong to THIS element: a `<summary>` after the first child is the |
| 3297 | // child's, and borrowing it would put a nested fold's name on its parent's body. |
| 3298 | let labelled = |p: usize| p < limit && inner.map_or(true, |i| p < i); |
| 3299 | let sum_open = match next(open, "<summary").filter(|&p| labelled(p)) { |
| 3300 | Some(p) => p, |
| 3301 | None => { at = after; continue; }, |
| 3302 | }; |
| 3303 | // AND IT MUST BE THE FIRST THING INSIDE, whitespace aside. This is stricter than the |
| 3304 | // browser, which takes the first `<summary>` child as the control wherever it sits, and |
| 3305 | // the strictness is the point: it is the only shape in which "the body is what lies |
| 3306 | // between `</summary>` and `</details>`" is unambiguous, and it is exactly what the |
| 3307 | // prompt asks the model to write. Prose before the label makes the element not a fold, |
| 3308 | // so it is left alone -- the reader still gets a native disclosure widget and the body |
| 3309 | // still travels, which costs tokens. Keying it instead would strip a fold whose open |
| 3310 | // state the page never tracked, and lose the reader's gesture rather than some money. |
| 3311 | let head_gt = match text[open..limit].find('>') { |
| 3312 | Some(p) => open + p + 1, |
| 3313 | None => { at = after; continue; }, |
| 3314 | }; |
| 3315 | if !text[head_gt..sum_open].trim().is_empty() { |
| 3316 | at = after; |
| 3317 | continue; |
| 3318 | } |
| 3319 | let sum_gt = match text[sum_open..limit].find('>') { |
| 3320 | Some(p) => sum_open + p + 1, |
| 3321 | None => { at = after; continue; }, |
| 3322 | }; |
| 3323 | let sum_close = match next(sum_gt, "</summary>").filter(|&p| p < limit) { |
| 3324 | Some(p) => p, |
| 3325 | None => { at = after; continue; }, |
| 3326 | }; |
| 3327 | let body_from = sum_close + "</summary>".len(); |
| 3328 | let raw = &text[body_from..limit]; |
| 3329 | let lead = raw.len() - raw.trim_start().len(); |
| 3330 | let trimmed = raw.trim(); |
| 3331 | out.push(Fold { |
| 3332 | ord: out.len(), |
| 3333 | summary: collapse_ws(&strip_tags(&text[sum_gt..sum_close])), |
| 3334 | body: (body_from + lead)..(body_from + lead + trimmed.len()), |
| 3335 | chars: trimmed.chars().count(), |
| 3336 | closed: close.is_some(), |
| 3337 | }); |
| 3338 | at = after; |
| 3339 | } |
| 3340 | out |
| 3341 | } |
| 3342 | |
| 3343 | /// Everything outside `<...>`, which is the `<summary>` element's own text. |
| 3344 | /// |
| 3345 | /// A `<` with no `>` after it is NOT a tag and is kept, because the browser keeps it too: the JS |
| 3346 | /// half reads this same summary with `textContent`, and a model writing `a < b` in a label would |
| 3347 | /// otherwise key it as `a` here and as `a < b` there -- one label, two keys, and a fold that |
| 3348 | /// never opens. |
| 3349 | fn strip_tags(s: &str) -> String { |
| 3350 | let mut out = String::with_capacity(s.len()); |
| 3351 | let mut rest = s; |
| 3352 | loop { |
| 3353 | let lt = match rest.find('<') { |
| 3354 | Some(p) => p, |
| 3355 | None => { out.push_str(rest); return out; }, |
| 3356 | }; |
| 3357 | match rest[lt..].find('>') { |
| 3358 | Some(gt) => { |
| 3359 | out.push_str(&rest[..lt]); |
| 3360 | rest = &rest[lt + gt + 1..]; |
| 3361 | }, |
| 3362 | None => { out.push_str(rest); return out; }, |
| 3363 | } |
| 3364 | } |
| 3365 | } |
| 3366 | |
| 3367 | /// Trimmed, with every internal whitespace run collapsed to one space. |
| 3368 | fn collapse_ws(s: &str) -> String { |
| 3369 | s.split_whitespace().collect::<Vec<_>>().join(" ") |
| 3370 | } |
| 3371 | |
| 3372 | /// The text an assistant message is REPLAYED with, or `None` when nothing in it changes. |
| 3373 | /// |
| 3374 | /// The text sibling of [`strip_said`], and it exists for the same reason: a fold's body is for a |
| 3375 | /// person, once. Left in the transcript it is re-sent on every later request for the life of the |
| 3376 | /// conversation, so a model that explains at length charges for the explanation again on every |
| 3377 | /// turn whether or not anybody looked at it twice. |
| 3378 | /// |
| 3379 | /// **The element is kept and only the body is replaced.** A model that sees the fold it wrote |
| 3380 | /// still knows it folded something and what it called it, which a wholesale deletion would take |
| 3381 | /// away along with the bytes. |
| 3382 | /// |
| 3383 | /// Two folds are passed over. An OPEN one is on the user's screen, so the model holds it too -- |
| 3384 | /// their own gesture decides the working set. An UNCLOSED one is malformed, and [`strip_said`]'s |
| 3385 | /// rule applies: rewriting it replaces a problem the model can see with one it cannot. |
| 3386 | /// |
| 3387 | /// # Arguments |
| 3388 | /// * `text` - The assistant's own words, as the model wrote them. |
| 3389 | /// * `open` - The keys of the folds the user has open. See [`Fold::key`]. |
| 3390 | fn strip_folds(text: &str, open: &OpenSet) -> Option<String> { |
| 3391 | let found = folds(text); |
| 3392 | if found.is_empty() { |
| 3393 | return None; |
| 3394 | } |
| 3395 | let mut out = String::with_capacity(text.len()); |
| 3396 | let mut at = 0usize; |
| 3397 | for f in &found { |
| 3398 | if !f.closed || open.contains(&f.key()) { |
| 3399 | continue; |
| 3400 | } |
| 3401 | // An empty body would be REPLACED by a hundred characters of note, so the one case where |
| 3402 | // stripping costs tokens rather than saving them is not stripped. |
| 3403 | if f.chars == 0 { |
| 3404 | continue; |
| 3405 | } |
| 3406 | out.push_str(&text[at..f.body.start]); |
| 3407 | out.push_str(&fold_note(f.chars)); |
| 3408 | at = f.body.end; |
| 3409 | } |
| 3410 | if at == 0 { |
| 3411 | return None; |
| 3412 | } |
| 3413 | out.push_str(&text[at..]); |
| 3414 | Some(out) |
| 3415 | } |
| 3416 | |
| 3417 | /// How many bytes of an assistant message's text will actually go on the wire. |
| 3418 | /// |
| 3419 | /// The sibling of [`sent_args_len`] for the inline fold, and it exists for the same defect: the |
| 3420 | /// compaction trigger in [`crate::agent::compact::msg_bytes`] sized a message by what the model |
| 3421 | /// wrote, while serialisation takes every closed fold's body out. So the trigger measured a |
| 3422 | /// conversation nobody was going to send, spent the budget on bytes that leave on the way out, |
| 3423 | /// and folded a conversation earlier than it needed to. Asked here rather than restated there, |
| 3424 | /// because a rule written twice is a rule that eventually disagrees with itself. |
| 3425 | /// |
| 3426 | /// # Arguments |
| 3427 | /// * `text` - The assistant's own words. |
| 3428 | /// * `open` - The keys of the folds the user has open. |
| 3429 | pub fn sent_text_len(text: &str, open: &OpenSet) -> usize { |
| 3430 | match strip_folds(text, open) { |
| 3431 | Some(replayed) => replayed.len(), |
| 3432 | None => text.len(), |
| 3433 | } |
| 3434 | } |
| 3435 | |
| 3436 | /// Serialise a `ChatMessage` to an OpenAI-API JSON object, including |
| 3437 | /// assistant `tool_calls` and the `tool` role — which `datmap_to_json` |
| 3438 | /// does not carry. |
| 3439 | /// |
| 3440 | /// Only the `user` role may carry an image on this side; `system`, `assistant` and `tool` take a |
| 3441 | /// string or text parts and nothing else. A message of another role that somehow holds one is |
| 3442 | /// flattened to the `[image …]` stand-in rather than sent as a part the API would reject; the |
| 3443 | /// tool results that legitimately produce images are re-homed by |
| 3444 | /// [`build_openai_body`](LlmClient::build_openai_body) instead. |
| 3445 | /// |
| 3446 | /// # Arguments |
| 3447 | /// * `msg` - The message to serialise. |
| 3448 | /// * `open` - The `say` folds the user has open, whose detail therefore travels. See |
| 3449 | /// [`OpenFolds`]. |
| 3450 | fn message_to_json(msg: &ChatMessage, open: &std::collections::HashSet<String>) -> String { |
| 3451 | match msg { |
| 3452 | ChatMessage::System { content } => |
| 3453 | fmt!("{{\"role\":\"system\",\"content\":\"{}\"}}", json_escape(&content.as_text())), |
| 3454 | ChatMessage::User { content } => |
| 3455 | fmt!("{{\"role\":\"user\",\"content\":{}}}", openai_content(content)), |
| 3456 | ChatMessage::Assistant { content, tool_calls } => { |
| 3457 | // The assistant's own words are the one role's text that serialisation rewrites: a |
| 3458 | // closed `<details>` fold travels as a note in its body's place. Both branches below |
| 3459 | // read this, so neither can be given the fold and the other the raw text. |
| 3460 | let said = content.as_text(); |
| 3461 | let folded = strip_folds(&said, open); |
| 3462 | let text = json_escape(folded.as_deref().unwrap_or(&said)); |
| 3463 | if tool_calls.is_empty() { |
| 3464 | fmt!("{{\"role\":\"assistant\",\"content\":\"{}\"}}", text) |
| 3465 | } else { |
| 3466 | let calls: Vec<String> = tool_calls.iter().map(|tc| { |
| 3467 | let stripped = strip_said(&tc.name, &tc.arguments, open.contains(&tc.id)); |
| 3468 | let args = stripped.as_deref().unwrap_or(&tc.arguments); |
| 3469 | fmt!( |
| 3470 | "{{\"id\":\"{}\",\"type\":\"function\",\"function\":{{\"name\":\"{}\",\"arguments\":\"{}\"}}}}", |
| 3471 | json_escape(&tc.id), json_escape(&tc.name), json_escape(args)) |
| 3472 | }).collect(); |
| 3473 | fmt!("{{\"role\":\"assistant\",\"content\":\"{}\",\"tool_calls\":[{}]}}", |
| 3474 | text, calls.join(",")) |
| 3475 | } |
| 3476 | } |
| 3477 | ChatMessage::Tool { tool_call_id, content } => |
| 3478 | fmt!("{{\"role\":\"tool\",\"tool_call_id\":\"{}\",\"content\":\"{}\"}}", |
| 3479 | json_escape(tool_call_id), json_escape(&content.as_text())), |
| 3480 | } |
| 3481 | } |
| 3482 | |
| 3483 | /// Whether an OpenAI-shaped payload says the reply was cut at the output limit. |
| 3484 | /// |
| 3485 | /// `finish_reason` is `null` on every delta but the last, and `"stop"` on an answer |
| 3486 | /// that finished; `"length"` is the one value that means the model was still writing. |
| 3487 | /// Read from the raw payload rather than inferred from malformed arguments, which is |
| 3488 | /// what the browser had to do and which cannot see a plain text reply cut short. |
| 3489 | fn openai_truncated(json: &str) -> bool { |
| 3490 | matches!(extract_json_string(json, "finish_reason").as_deref(), Some("length")) |
| 3491 | } |
| 3492 | |
| 3493 | /// Whether an Anthropic payload says the same thing. |
| 3494 | fn anthropic_truncated(json: &str) -> bool { |
| 3495 | matches!(extract_json_string(json, "stop_reason").as_deref(), Some("max_tokens")) |
| 3496 | } |
| 3497 | |
| 3498 | /// Parse a non-streaming chat completion body into |
| 3499 | /// `(content, tool_calls, usage)`. |
| 3500 | fn parse_full_response(body: &str) -> (String, Vec<ToolCall>, Usage) { |
| 3501 | // Scope content extraction to before "tool_calls" so we don't pick |
| 3502 | // up a "content" key inside a tool call's arguments. |
| 3503 | let scope_end = body.find("\"tool_calls\"").unwrap_or(body.len()); |
| 3504 | let content = extract_json_string(&body[..scope_end], "content").unwrap_or_default(); |
| 3505 | |
| 3506 | let mut tool_calls = Vec::new(); |
| 3507 | if let Some(arr) = find_json_array(body, "tool_calls") { |
| 3508 | for elem in split_top_level_objects(&arr) { |
| 3509 | let name = match extract_json_string(&elem, "name") { |
| 3510 | Some(n) if !n.is_empty() => n, |
| 3511 | _ => continue, |
| 3512 | }; |
| 3513 | let id = extract_json_string(&elem, "id").unwrap_or_default(); |
| 3514 | let arguments = extract_json_string(&elem, "arguments") |
| 3515 | .unwrap_or_else(|| "{}".to_string()); |
| 3516 | tool_calls.push(ToolCall { id, name, arguments }); |
| 3517 | } |
| 3518 | } |
| 3519 | |
| 3520 | (content, tool_calls, parse_usage(body).unwrap_or_default()) |
| 3521 | } |
| 3522 | |
| 3523 | |
| 3524 | // ┌───────────────────────────────────────────────────────────────┐ |
| 3525 | // │ StreamAcc — streamed delta accumulator │ |
| 3526 | // └───────────────────────────────────────────────────────────────┘ |
| 3527 | |
| 3528 | /// One tool call being reconstructed from streamed fragments. |
| 3529 | /// |
| 3530 | /// A streamed `tool_calls` delta arrives in pieces keyed by `index`: the |
| 3531 | /// first fragment usually carries the `id` and function `name` with an |
| 3532 | /// empty `arguments`, and later fragments append `arguments` text until |
| 3533 | /// the call is whole. |
| 3534 | struct StreamCall { |
| 3535 | /// Position of this call within the assistant turn. |
| 3536 | index: i64, |
| 3537 | id: String, |
| 3538 | name: String, |
| 3539 | /// Accumulated raw JSON arguments, concatenated across fragments. |
| 3540 | arguments: String, |
| 3541 | } |
| 3542 | |
| 3543 | /// Accumulates OpenAI-style streaming chat deltas across SSE chunks: |
| 3544 | /// text content, incrementally-built tool calls, and usage. |
| 3545 | /// |
| 3546 | /// Each `data:` payload is fed to [`ingest`](StreamAcc::ingest); when the |
| 3547 | /// stream ends, [`into_response`](StreamAcc::into_response) yields the |
| 3548 | /// assembled [`ChatOnceResponse`]. |
| 3549 | #[derive(Default)] |
| 3550 | struct StreamAcc { |
| 3551 | content: String, |
| 3552 | // The model's own working, kept apart from the answer it produced. |
| 3553 | reasoning: String, |
| 3554 | /// The last usage block the stream reported. An aborted stream may never |
| 3555 | /// deliver one, which leaves this at its default rather than erroring. |
| 3556 | usage: Usage, |
| 3557 | calls: Vec<StreamCall>, |
| 3558 | /// Whether a chunk said the reply stopped at the output limit. |
| 3559 | truncated: bool, |
| 3560 | } |
| 3561 | |
| 3562 | impl StreamAcc { |
| 3563 | |
| 3564 | /// Fold one SSE `data:` payload into the accumulator, forwarding each delta to |
| 3565 | /// `on_token` as it arrives, labelled as answer or as working. |
| 3566 | fn ingest(&mut self, data: &str, on_token: &mut impl FnMut(Delta<'_>)) { |
| 3567 | // Text delta — scoped to before any `tool_calls` so a `content` |
| 3568 | // key inside a tool call's arguments is never mistaken for it. |
| 3569 | let scope_end = data.find("\"tool_calls\"").unwrap_or(data.len()); |
| 3570 | if let Some(content) = extract_json_string(&data[..scope_end], "content") { |
| 3571 | if !content.is_empty() { |
| 3572 | on_token(Delta::Text(&content)); |
| 3573 | self.content.push_str(&content); |
| 3574 | } |
| 3575 | } |
| 3576 | |
| 3577 | // THE MODEL'S OWN WORKING, which every reasoning model on this dialect streams |
| 3578 | // and which this client read none of until 2026-08-28. A measured DeepSeek round |
| 3579 | // pulled 1.8 MB down the wire over 84 seconds and put fifty characters on the |
| 3580 | // screen; all the rest was this field, discarded delta by delta, and the user was |
| 3581 | // billed for it while watching a spinner. |
| 3582 | // |
| 3583 | // Two spellings, and never both in one delta: `reasoning` is what OpenRouter |
| 3584 | // sends, `reasoning_content` is what DeepSeek's own endpoint calls it. So one is |
| 3585 | // read and then the other, rather than both concatenated. |
| 3586 | // |
| 3587 | // `reasoning_details` is NOT read. OpenRouter sends it alongside `reasoning` with |
| 3588 | // the same words in it, verbatim, so a reader that took both would put every |
| 3589 | // token on the page twice. |
| 3590 | // |
| 3591 | // `null` is the value on the deltas that carry no reasoning, and |
| 3592 | // `extract_json_string` answers None for a value that is not a string -- so the |
| 3593 | // absence needs no test of its own here. |
| 3594 | let think = extract_json_string(&data[..scope_end], "reasoning") |
| 3595 | .or_else(|| extract_json_string(&data[..scope_end], "reasoning_content")); |
| 3596 | if let Some(t) = think { |
| 3597 | if !t.is_empty() { |
| 3598 | on_token(Delta::Reasoning(&t)); |
| 3599 | self.reasoning.push_str(&t); |
| 3600 | } |
| 3601 | } |
| 3602 | |
| 3603 | // Tool-call fragments — merge each into its slot by `index`. |
| 3604 | if let Some(arr) = find_json_array(data, "tool_calls") { |
| 3605 | for elem in split_top_level_objects(&arr) { |
| 3606 | let index = extract_json_number(&elem, "index") |
| 3607 | .map(|n| n as i64) |
| 3608 | .unwrap_or(0); |
| 3609 | // Locate an existing slot by index before borrowing |
| 3610 | // mutably, so a new slot can be pushed without an |
| 3611 | // overlapping borrow. |
| 3612 | let pos = self.calls.iter().position(|c| c.index == index); |
| 3613 | let slot = match pos { |
| 3614 | Some(p) => &mut self.calls[p], |
| 3615 | None => { |
| 3616 | self.calls.push(StreamCall { |
| 3617 | index, |
| 3618 | id: String::new(), |
| 3619 | name: String::new(), |
| 3620 | arguments: String::new(), |
| 3621 | }); |
| 3622 | let last = self.calls.len() - 1; |
| 3623 | &mut self.calls[last] |
| 3624 | } |
| 3625 | }; |
| 3626 | if let Some(id) = extract_json_string(&elem, "id") { |
| 3627 | if !id.is_empty() { slot.id = id; } |
| 3628 | } |
| 3629 | if let Some(name) = extract_json_string(&elem, "name") { |
| 3630 | if !name.is_empty() { slot.name = name; } |
| 3631 | } |
| 3632 | if let Some(args) = extract_json_string(&elem, "arguments") { |
| 3633 | slot.arguments.push_str(&args); |
| 3634 | } |
| 3635 | } |
| 3636 | } |
| 3637 | |
| 3638 | // Usage — present on the final chunk when include_usage is set. |
| 3639 | if let Some(u) = parse_usage(data) { |
| 3640 | self.usage = u; |
| 3641 | } |
| 3642 | |
| 3643 | // Why the model stopped, which arrives on the last delta and nowhere else. |
| 3644 | // Sticky: a later chunk carrying only usage must not unsay it. |
| 3645 | if openai_truncated(data) { |
| 3646 | self.truncated = true; |
| 3647 | } |
| 3648 | } |
| 3649 | |
| 3650 | /// Whether this turn has produced anything yet. |
| 3651 | /// |
| 3652 | /// A retry is only safe while this is false: text has already been handed to |
| 3653 | /// the caller, and a tool-call fragment is a partial the next attempt would |
| 3654 | /// duplicate rather than replace. |
| 3655 | fn has_output(&self) -> bool { |
| 3656 | !self.content.is_empty() || !self.calls.is_empty() |
| 3657 | } |
| 3658 | |
| 3659 | /// Consume the accumulator into a [`ChatOnceResponse`]. Calls with no |
| 3660 | /// name are dropped (a stray fragment), and an empty arguments string |
| 3661 | /// becomes `{}` so tool dispatch always sees a valid JSON object. |
| 3662 | fn into_response(self, aborted: bool, retries: u32) -> ChatOnceResponse { |
| 3663 | let tool_calls = self.calls.into_iter() |
| 3664 | .filter(|c| !c.name.is_empty()) |
| 3665 | .map(|c| ToolCall { |
| 3666 | id: c.id, |
| 3667 | name: c.name, |
| 3668 | arguments: if c.arguments.is_empty() { "{}".to_string() } else { c.arguments }, |
| 3669 | }) |
| 3670 | .collect(); |
| 3671 | ChatOnceResponse { |
| 3672 | content: self.content, |
| 3673 | tool_calls, |
| 3674 | prompt_tokens: self.usage.prompt, |
| 3675 | completion_tokens: self.usage.completion, |
| 3676 | cached_tokens: self.usage.cached, |
| 3677 | cost_usd: self.usage.cost_usd, |
| 3678 | aborted, |
| 3679 | retries, |
| 3680 | thinking: self.reasoning, |
| 3681 | truncated: self.truncated, |
| 3682 | } |
| 3683 | } |
| 3684 | } |
| 3685 | |
| 3686 | // ┌───────────────────────────────────────────────────────────────┐ |
| 3687 | // │ AnthropicAcc — Messages API event accumulator │ |
| 3688 | // └───────────────────────────────────────────────────────────────┘ |
| 3689 | |
| 3690 | /// What one Anthropic content block is. |
| 3691 | #[derive(Clone, Copy, Debug, Eq, PartialEq)] |
| 3692 | enum AnthKind { |
| 3693 | Text, |
| 3694 | Thinking, |
| 3695 | /// A `redacted_thinking` block, which is opaque and replayed verbatim. |
| 3696 | Redacted, |
| 3697 | ToolUse, |
| 3698 | /// A block this client does not act on (a server tool, a fallback marker). |
| 3699 | Other, |
| 3700 | } |
| 3701 | |
| 3702 | /// One content block being rebuilt from the event stream. |
| 3703 | struct AnthBlock { |
| 3704 | /// Position in the message's `content` array; the streamed events key on it. |
| 3705 | index: i64, |
| 3706 | kind: AnthKind, |
| 3707 | id: String, |
| 3708 | name: String, |
| 3709 | /// `input_json_delta` fragments, concatenated into the tool's arguments. |
| 3710 | args: String, |
| 3711 | /// `thinking_delta` fragments, concatenated. |
| 3712 | think: String, |
| 3713 | /// The block's signature, which the API verifies when it is handed back. |
| 3714 | sig: String, |
| 3715 | /// A block replayed verbatim rather than rebuilt, as JSON. |
| 3716 | raw: String, |
| 3717 | } |
| 3718 | |
| 3719 | /// The Anthropic `usage` counts, kept as reported. |
| 3720 | /// |
| 3721 | /// Held raw rather than folded into [`Usage`] on arrival because they arrive |
| 3722 | /// twice -- once on `message_start` and again, cumulatively, on `message_delta` |
| 3723 | /// -- and the second report names only the fields that changed. Overwriting |
| 3724 | /// [`Usage`] wholesale from the second would zero the input counts. |
| 3725 | #[derive(Clone, Copy, Debug, Default)] |
| 3726 | struct AnthUsage { |
| 3727 | input: u64, |
| 3728 | output: u64, |
| 3729 | /// Prompt tokens served from the cache, billed at a tenth of a fresh read. |
| 3730 | read: u64, |
| 3731 | /// Prompt tokens written to the cache, billed at 1.25x a fresh read. |
| 3732 | write: u64, |
| 3733 | } |
| 3734 | |
| 3735 | impl AnthUsage { |
| 3736 | |
| 3737 | /// Fold in one `usage` object, taking only the fields it actually carries. |
| 3738 | fn merge(&mut self, usage: &str) { |
| 3739 | if let Some(v) = extract_json_number(usage, "input_tokens") { self.input = v; } |
| 3740 | if let Some(v) = extract_json_number(usage, "output_tokens") { self.output = v; } |
| 3741 | if let Some(v) = extract_json_number(usage, "cache_read_input_tokens") { self.read = v; } |
| 3742 | if let Some(v) = extract_json_number(usage, "cache_creation_input_tokens") { self.write = v; } |
| 3743 | } |
| 3744 | |
| 3745 | /// The client's own usage shape. |
| 3746 | /// |
| 3747 | /// Anthropic's `input_tokens` counts only what was neither read from nor |
| 3748 | /// written to the cache, where this client's `prompt` means every prompt |
| 3749 | /// token processed -- so the three are added, and `cached` is the read. |
| 3750 | /// The 1.25x premium on a cache *write* is not modelled: the price table |
| 3751 | /// carries one cached rate, not two, so a write is priced as a fresh read. |
| 3752 | /// That understates the first request of a session slightly and nothing |
| 3753 | /// afterwards, which is the smaller of the two errors available. |
| 3754 | fn into_usage(self) -> Usage { |
| 3755 | Usage { |
| 3756 | prompt: self.input.saturating_add(self.read).saturating_add(self.write), |
| 3757 | completion: self.output, |
| 3758 | cached: self.read, |
| 3759 | // Anthropic bills against an account and reports no per-call cost, |
| 3760 | // so this stays zero and the price table answers instead. |
| 3761 | cost_usd: 0.0, |
| 3762 | } |
| 3763 | } |
| 3764 | } |
| 3765 | |
| 3766 | /// Accumulates the Anthropic Messages API event stream: text, thinking, |
| 3767 | /// incrementally-built tool calls, and usage. |
| 3768 | /// |
| 3769 | /// The events are named (`content_block_start`, `content_block_delta`, …) and |
| 3770 | /// keyed by block index, rather than being deltas of one growing object, so |
| 3771 | /// this is a different machine from [`StreamAcc`] rather than a variation of it. |
| 3772 | #[derive(Default)] |
| 3773 | struct AnthropicAcc { |
| 3774 | content: String, |
| 3775 | usage: AnthUsage, |
| 3776 | blocks: Vec<AnthBlock>, |
| 3777 | /// An `error` event delivered on an otherwise-successful stream, as |
| 3778 | /// `(type, message)`. |
| 3779 | error: Option<(String, String)>, |
| 3780 | /// Whether a `message_delta` said the reply stopped at the output limit. |
| 3781 | truncated: bool, |
| 3782 | } |
| 3783 | |
| 3784 | impl AnthropicAcc { |
| 3785 | |
| 3786 | /// The slot for `index`, created if this is the first event for it. |
| 3787 | fn slot(&mut self, index: i64, kind: AnthKind) -> &mut AnthBlock { |
| 3788 | match self.blocks.iter().position(|b| b.index == index) { |
| 3789 | Some(p) => &mut self.blocks[p], |
| 3790 | None => { |
| 3791 | self.blocks.push(AnthBlock { |
| 3792 | index, |
| 3793 | kind, |
| 3794 | id: String::new(), |
| 3795 | name: String::new(), |
| 3796 | args: String::new(), |
| 3797 | think: String::new(), |
| 3798 | sig: String::new(), |
| 3799 | raw: String::new(), |
| 3800 | }); |
| 3801 | let last = self.blocks.len() - 1; |
| 3802 | &mut self.blocks[last] |
| 3803 | } |
| 3804 | } |
| 3805 | } |
| 3806 | |
| 3807 | /// Fold one SSE `data:` payload in, forwarding both kinds of delta to `on_token`. |
| 3808 | /// |
| 3809 | /// Thinking goes out as [`Delta::Reasoning`] and never as [`Delta::Text`], which is |
| 3810 | /// the whole of what keeps it out of the answer. It used not to be forwarded at |
| 3811 | /// all, because the sink took a bare `&str` and a caller that was handed one had no |
| 3812 | /// way to tell working from reply -- so the reasoning was held back until the round |
| 3813 | /// ended. Now the sink says which is which, so it can be shown as it arrives, and a |
| 3814 | /// round that thinks for a minute stops looking like a round that has hung. |
| 3815 | fn ingest(&mut self, data: &str, on_token: &mut impl FnMut(Delta<'_>)) { |
| 3816 | let ty = match extract_json_string(data, "type") { |
| 3817 | Some(t) => t, |
| 3818 | None => return, |
| 3819 | }; |
| 3820 | match ty.as_str() { |
| 3821 | "message_start" | "message_delta" => { |
| 3822 | if let Some(u) = find_json_object(data, "usage") { self.usage.merge(&u); } |
| 3823 | // `message_start` carries `stop_reason: null`, so only a real one sets |
| 3824 | // this; and once set, nothing later unsets it. |
| 3825 | if anthropic_truncated(data) { self.truncated = true; } |
| 3826 | } |
| 3827 | "content_block_start" => { |
| 3828 | let index = extract_json_number(data, "index").map(|n| n as i64).unwrap_or(0); |
| 3829 | let cb = match find_json_object(data, "content_block") { |
| 3830 | Some(c) => c, |
| 3831 | None => return, |
| 3832 | }; |
| 3833 | match extract_json_string(&cb, "type").unwrap_or_default().as_str() { |
| 3834 | "text" => { self.slot(index, AnthKind::Text); } |
| 3835 | "thinking" => { self.slot(index, AnthKind::Thinking); } |
| 3836 | "redacted_thinking" => { |
| 3837 | let slot = self.slot(index, AnthKind::Redacted); |
| 3838 | slot.kind = AnthKind::Redacted; |
| 3839 | slot.raw = cb.clone(); |
| 3840 | } |
| 3841 | "tool_use" => { |
| 3842 | let id = extract_json_string(&cb, "id").unwrap_or_default(); |
| 3843 | let name = extract_json_string(&cb, "name").unwrap_or_default(); |
| 3844 | let slot = self.slot(index, AnthKind::ToolUse); |
| 3845 | slot.kind = AnthKind::ToolUse; |
| 3846 | slot.id = id; |
| 3847 | slot.name = name; |
| 3848 | } |
| 3849 | // A server tool, or a block type added after this was |
| 3850 | // written: recorded so its deltas land somewhere harmless. |
| 3851 | _ => { self.slot(index, AnthKind::Other); } |
| 3852 | } |
| 3853 | } |
| 3854 | "content_block_delta" => { |
| 3855 | let index = extract_json_number(data, "index").map(|n| n as i64).unwrap_or(0); |
| 3856 | let d = match find_json_object(data, "delta") { |
| 3857 | Some(d) => d, |
| 3858 | None => return, |
| 3859 | }; |
| 3860 | match extract_json_string(&d, "type").unwrap_or_default().as_str() { |
| 3861 | "text_delta" => { |
| 3862 | if let Some(t) = extract_json_string(&d, "text") { |
| 3863 | if !t.is_empty() { |
| 3864 | on_token(Delta::Text(&t)); |
| 3865 | self.content.push_str(&t); |
| 3866 | } |
| 3867 | } |
| 3868 | } |
| 3869 | "thinking_delta" => { |
| 3870 | if let Some(t) = extract_json_string(&d, "thinking") { |
| 3871 | if !t.is_empty() { on_token(Delta::Reasoning(&t)); } |
| 3872 | self.slot(index, AnthKind::Thinking).think.push_str(&t); |
| 3873 | } |
| 3874 | } |
| 3875 | "signature_delta" => { |
| 3876 | if let Some(s) = extract_json_string(&d, "signature") { |
| 3877 | self.slot(index, AnthKind::Thinking).sig.push_str(&s); |
| 3878 | } |
| 3879 | } |
| 3880 | "input_json_delta" => { |
| 3881 | if let Some(p) = extract_json_string(&d, "partial_json") { |
| 3882 | self.slot(index, AnthKind::ToolUse).args.push_str(&p); |
| 3883 | } |
| 3884 | } |
| 3885 | _ => {} |
| 3886 | } |
| 3887 | } |
| 3888 | "error" => { |
| 3889 | let e = find_json_object(data, "error").unwrap_or_default(); |
| 3890 | self.error = Some(( |
| 3891 | extract_json_string(&e, "type").unwrap_or_else(|| "api_error".to_string()), |
| 3892 | extract_json_string(&e, "message").unwrap_or_default())); |
| 3893 | } |
| 3894 | _ => {} |
| 3895 | } |
| 3896 | } |
| 3897 | |
| 3898 | /// Whether this turn has produced anything the caller now holds. |
| 3899 | /// |
| 3900 | /// Thinking does not count: it is never handed to the caller, and a retry |
| 3901 | /// would simply produce a fresh block rather than a duplicate one. |
| 3902 | fn has_output(&self) -> bool { |
| 3903 | !self.content.is_empty() |
| 3904 | || self.blocks.iter().any(|b| b.kind == AnthKind::ToolUse) |
| 3905 | } |
| 3906 | |
| 3907 | /// The signed thinking blocks of this turn, serialised for replay. |
| 3908 | /// |
| 3909 | /// Empty when any block of the run is unsigned -- a stream cut before its |
| 3910 | /// `signature_delta`, say. The API requires the run to match what the model |
| 3911 | /// generated, so half of it is worse than none: an unsigned block is a 400, |
| 3912 | /// and a run with one block quietly dropped is a rearrangement. |
| 3913 | fn thinking_blocks(&self) -> Vec<String> { |
| 3914 | let mut out = Vec::new(); |
| 3915 | for b in &self.blocks { |
| 3916 | match b.kind { |
| 3917 | AnthKind::Thinking => { |
| 3918 | if b.sig.is_empty() { return Vec::new(); } |
| 3919 | out.push(fmt!( |
| 3920 | "{{\"type\":\"thinking\",\"thinking\":\"{}\",\"signature\":\"{}\"}}", |
| 3921 | json_escape(&b.think), json_escape(&b.sig))); |
| 3922 | } |
| 3923 | AnthKind::Redacted => { |
| 3924 | if b.raw.is_empty() { return Vec::new(); } |
| 3925 | out.push(b.raw.clone()); |
| 3926 | } |
| 3927 | _ => {} |
| 3928 | } |
| 3929 | } |
| 3930 | out |
| 3931 | } |
| 3932 | |
| 3933 | /// The summarised reasoning of this turn, for a caller that wants to show it. |
| 3934 | fn thinking_text(&self) -> String { |
| 3935 | let parts: Vec<&str> = self.blocks.iter() |
| 3936 | .filter(|b| b.kind == AnthKind::Thinking && !b.think.is_empty()) |
| 3937 | .map(|b| b.think.as_str()) |
| 3938 | .collect(); |
| 3939 | parts.join("\n") |
| 3940 | } |
| 3941 | |
| 3942 | /// Consume the accumulator into a [`ChatOnceResponse`]. |
| 3943 | fn into_response(self, aborted: bool, retries: u32) -> ChatOnceResponse { |
| 3944 | let thinking = self.thinking_text(); |
| 3945 | let tool_calls = self.blocks.iter() |
| 3946 | .filter(|b| b.kind == AnthKind::ToolUse && !b.name.is_empty()) |
| 3947 | .map(|b| ToolCall { |
| 3948 | id: b.id.clone(), |
| 3949 | name: b.name.clone(), |
| 3950 | arguments: if b.args.is_empty() { "{}".to_string() } else { b.args.clone() }, |
| 3951 | }) |
| 3952 | .collect(); |
| 3953 | ChatOnceResponse { |
| 3954 | content: self.content, |
| 3955 | tool_calls, |
| 3956 | prompt_tokens: self.usage.into_usage().prompt, |
| 3957 | completion_tokens: self.usage.output, |
| 3958 | cached_tokens: self.usage.read, |
| 3959 | cost_usd: 0.0, |
| 3960 | aborted, |
| 3961 | retries, |
| 3962 | thinking, |
| 3963 | truncated: self.truncated, |
| 3964 | } |
| 3965 | } |
| 3966 | } |
| 3967 | |
| 3968 | |
| 3969 | // ┌───────────────────────────────────────────────────────────────┐ |
| 3970 | // │ Acc — whichever accumulator the dialect needs │ |
| 3971 | // └───────────────────────────────────────────────────────────────┘ |
| 3972 | |
| 3973 | /// The stream accumulator for a [`Dialect`]. |
| 3974 | /// |
| 3975 | /// An enum rather than a trait object: there are exactly two wire shapes, both |
| 3976 | /// known here, and the retry loop wants them by value. |
| 3977 | enum Acc { |
| 3978 | OpenAi(StreamAcc), |
| 3979 | Anthropic(AnthropicAcc), |
| 3980 | } |
| 3981 | |
| 3982 | impl Acc { |
| 3983 | |
| 3984 | /// A fresh accumulator for `dialect`. |
| 3985 | fn new(dialect: Dialect) -> Self { |
| 3986 | match dialect { |
| 3987 | Dialect::OpenAi => Self::OpenAi(StreamAcc::default()), |
| 3988 | Dialect::Anthropic => Self::Anthropic(AnthropicAcc::default()), |
| 3989 | } |
| 3990 | } |
| 3991 | |
| 3992 | /// Fold one SSE `data:` payload in, forwarding text deltas to `on_token`. |
| 3993 | fn ingest(&mut self, data: &str, on_token: &mut impl FnMut(Delta<'_>)) { |
| 3994 | match self { |
| 3995 | Self::OpenAi(a) => a.ingest(data, on_token), |
| 3996 | Self::Anthropic(a) => a.ingest(data, on_token), |
| 3997 | } |
| 3998 | } |
| 3999 | |
| 4000 | /// Whether this turn has produced anything the caller now holds. |
| 4001 | fn has_output(&self) -> bool { |
| 4002 | match self { |
| 4003 | Self::OpenAi(a) => a.has_output(), |
| 4004 | Self::Anthropic(a) => a.has_output(), |
| 4005 | } |
| 4006 | } |
| 4007 | |
| 4008 | /// An error the provider delivered inside an otherwise-successful stream. |
| 4009 | /// |
| 4010 | /// Only Anthropic sends one: an OpenAI-compatible endpoint that is |
| 4011 | /// overloaded says so with a status code, before the body starts. |
| 4012 | fn stream_error(&self) -> Option<TransportErr> { |
| 4013 | let (kind, msg) = match self { |
| 4014 | Self::OpenAi(_) => return None, |
| 4015 | Self::Anthropic(a) => match &a.error { |
| 4016 | Some(e) => e.clone(), |
| 4017 | None => return None, |
| 4018 | }, |
| 4019 | }; |
| 4020 | let err = err!( |
| 4021 | "LLM: stream error: {} | {}", kind, msg; IO, Network, Wire, Read); |
| 4022 | let reason = fmt!("the provider reported {}", kind); |
| 4023 | // The same split as the status codes: the provider's own trouble is |
| 4024 | // worth another attempt, a complaint about this request is not. |
| 4025 | let transient = kind == "overloaded_error" |
| 4026 | || kind == "api_error" |
| 4027 | || kind == "rate_limit_error"; |
| 4028 | Some(if transient { |
| 4029 | TransportErr::transient(reason, err) |
| 4030 | } else { |
| 4031 | TransportErr::fatal(reason, err) |
| 4032 | }) |
| 4033 | } |
| 4034 | |
| 4035 | /// The signed thinking blocks to hold against this turn's tool calls. |
| 4036 | fn take_thinking(&self) -> Vec<String> { |
| 4037 | match self { |
| 4038 | Self::OpenAi(_) => Vec::new(), |
| 4039 | Self::Anthropic(a) => a.thinking_blocks(), |
| 4040 | } |
| 4041 | } |
| 4042 | |
| 4043 | /// Consume the accumulator into a [`ChatOnceResponse`]. |
| 4044 | fn into_response(self, aborted: bool, retries: u32) -> ChatOnceResponse { |
| 4045 | match self { |
| 4046 | Self::OpenAi(a) => a.into_response(aborted, retries), |
| 4047 | Self::Anthropic(a) => a.into_response(aborted, retries), |
| 4048 | } |
| 4049 | } |
| 4050 | } |
| 4051 | |
| 4052 | /// Parse a whole (non-streamed) Anthropic Messages response into |
| 4053 | /// `(content, tool_calls, usage, thinking blocks)`. |
| 4054 | /// |
| 4055 | /// The thinking blocks come back serialised for replay, exactly as the streamed |
| 4056 | /// path produces them -- see [`AnthropicAcc::thinking_blocks`]. |
| 4057 | /// |
| 4058 | /// # Arguments |
| 4059 | /// * `body` - The response body, as JSON text. |
| 4060 | fn parse_anthropic_response(body: &str) -> (String, Vec<ToolCall>, Usage, Vec<String>) { |
| 4061 | let mut content = String::new(); |
| 4062 | let mut tool_calls = Vec::new(); |
| 4063 | let mut thinking: Vec<String> = Vec::new(); |
| 4064 | let mut signed = true; |
| 4065 | if let Some(arr) = find_json_array(body, "content") { |
| 4066 | for elem in split_top_level_objects(&arr) { |
| 4067 | match extract_json_string(&elem, "type").unwrap_or_default().as_str() { |
| 4068 | "text" => { |
| 4069 | if let Some(t) = extract_json_string(&elem, "text") { content.push_str(&t); } |
| 4070 | } |
| 4071 | "thinking" => { |
| 4072 | match extract_json_string(&elem, "signature") { |
| 4073 | Some(s) if !s.is_empty() => thinking.push(elem.clone()), |
| 4074 | _ => signed = false, |
| 4075 | } |
| 4076 | } |
| 4077 | "redacted_thinking" => thinking.push(elem.clone()), |
| 4078 | "tool_use" => { |
| 4079 | let name = match extract_json_string(&elem, "name") { |
| 4080 | Some(n) if !n.is_empty() => n, |
| 4081 | _ => continue, |
| 4082 | }; |
| 4083 | let id = extract_json_string(&elem, "id").unwrap_or_default(); |
| 4084 | let input = find_json_object(&elem, "input") |
| 4085 | .unwrap_or_else(|| "{}".to_string()); |
| 4086 | tool_calls.push(ToolCall { id, name, arguments: input }); |
| 4087 | } |
| 4088 | _ => {} |
| 4089 | } |
| 4090 | } |
| 4091 | } |
| 4092 | // A run with an unsigned block in it cannot be replayed; see |
| 4093 | // [`AnthropicAcc::thinking_blocks`]. |
| 4094 | if !signed { thinking.clear(); } |
| 4095 | let mut usage = AnthUsage::default(); |
| 4096 | if let Some(u) = find_json_object(body, "usage") { usage.merge(&u); } |
| 4097 | (content, tool_calls, usage.into_usage(), thinking) |
| 4098 | } |
| 4099 | |
| 4100 | /// Extract a JSON array value for a key, returning the inner text |
| 4101 | /// including the surrounding brackets. String contents are skipped so |
| 4102 | /// brackets inside strings don't confuse the depth count. |
| 4103 | fn find_json_array(json: &str, key: &str) -> Option<String> { |
| 4104 | let needle = fmt!("\"{}\":", key); |
| 4105 | let pos = match json.find(&needle) { |
| 4106 | Some(p) => p, |
| 4107 | None => return None, |
| 4108 | }; |
| 4109 | let bytes = json.as_bytes(); |
| 4110 | // Skip whitespace after the colon to the opening bracket. |
| 4111 | let mut start = pos + needle.len(); |
| 4112 | while start < bytes.len() && bytes[start].is_ascii_whitespace() { start += 1; } |
| 4113 | if start >= bytes.len() || bytes[start] != b'[' { return None; } |
| 4114 | let mut depth = 0i32; |
| 4115 | let mut in_str = false; |
| 4116 | let mut i = start; |
| 4117 | while i < bytes.len() { |
| 4118 | let b = bytes[i]; |
| 4119 | if in_str { |
| 4120 | if b == b'\\' { i += 2; continue; } |
| 4121 | if b == b'"' { in_str = false; } |
| 4122 | } else { |
| 4123 | match b { |
| 4124 | b'"' => in_str = true, |
| 4125 | b'[' => depth += 1, |
| 4126 | b']' => { |
| 4127 | depth -= 1; |
| 4128 | if depth == 0 { return Some(json[start..=i].to_string()); } |
| 4129 | } |
| 4130 | _ => {} |
| 4131 | } |
| 4132 | } |
| 4133 | i += 1; |
| 4134 | } |
| 4135 | None |
| 4136 | } |
| 4137 | |
| 4138 | /// Split a JSON array's text into its top-level `{...}` object elements. |
| 4139 | fn split_top_level_objects(arr: &str) -> Vec<String> { |
| 4140 | let bytes = arr.as_bytes(); |
| 4141 | let mut out = Vec::new(); |
| 4142 | let mut depth = 0i32; |
| 4143 | let mut start = 0usize; |
| 4144 | let mut in_str = false; |
| 4145 | let mut i = 0usize; |
| 4146 | while i < bytes.len() { |
| 4147 | let b = bytes[i]; |
| 4148 | if in_str { |
| 4149 | if b == b'\\' { i += 2; continue; } |
| 4150 | if b == b'"' { in_str = false; } |
| 4151 | } else { |
| 4152 | match b { |
| 4153 | b'"' => in_str = true, |
| 4154 | b'{' => { if depth == 0 { start = i; } depth += 1; } |
| 4155 | b'}' => { |
| 4156 | depth -= 1; |
| 4157 | if depth == 0 { out.push(arr[start..=i].to_string()); } |
| 4158 | } |
| 4159 | _ => {} |
| 4160 | } |
| 4161 | } |
| 4162 | i += 1; |
| 4163 | } |
| 4164 | out |
| 4165 | } |
| 4166 | |
| 4167 | pub fn datmap_to_json(m: &DaticleMap) -> String { |
| 4168 | let mut out = String::with_capacity(256); |
| 4169 | out.push('{'); |
| 4170 | let mut first = true; |
| 4171 | // DaticleMap iteration is not ordered — we sort keys for |
| 4172 | // deterministic output (not required by the API but cleaner). |
| 4173 | let mut entries: Vec<(&Dat, &Dat)> = m.iter().collect(); |
| 4174 | entries.sort_by(|a, b| { |
| 4175 | match (a.0, b.0) { |
| 4176 | (Dat::Str(a_s), Dat::Str(b_s)) => a_s.cmp(b_s), |
| 4177 | _ => std::cmp::Ordering::Equal, |
| 4178 | } |
| 4179 | }); |
| 4180 | for (k, v) in entries { |
| 4181 | if !first { out.push(','); } |
| 4182 | first = false; |
| 4183 | if let Dat::Str(k_s) = k { |
| 4184 | out.push('"'); |
| 4185 | out.push_str(k_s); |
| 4186 | out.push_str("\":"); |
| 4187 | out.push_str(&dat_to_json(v)); |
| 4188 | } |
| 4189 | } |
| 4190 | out.push('}'); |
| 4191 | out |
| 4192 | } |
| 4193 | |
| 4194 | /// Escape a string for embedding inside a JSON string literal (no |
| 4195 | /// surrounding quotes). Shared with the tool-definition builder. |
| 4196 | pub(crate) fn json_escape(s: &str) -> String { |
| 4197 | let mut out = String::with_capacity(s.len()); |
| 4198 | for c in s.chars() { |
| 4199 | match c { |
| 4200 | '"' => out.push_str("\\\""), |
| 4201 | '\\' => out.push_str("\\\\"), |
| 4202 | '\n' => out.push_str("\\n"), |
| 4203 | '\t' => out.push_str("\\t"), |
| 4204 | '\r' => out.push_str("\\r"), |
| 4205 | c if (c as u32) < 0x20 => out.push_str(&fmt!("\\u{:04x}", c as u32)), |
| 4206 | c => out.push(c), |
| 4207 | } |
| 4208 | } |
| 4209 | out |
| 4210 | } |
| 4211 | |
| 4212 | /// Convert a JDAT Dat value to JSON. |
| 4213 | fn dat_to_json(d: &Dat) -> String { |
| 4214 | match d { |
| 4215 | Dat::Str(s) => { |
| 4216 | let mut out = String::with_capacity(s.len() + 2); |
| 4217 | out.push('"'); |
| 4218 | for c in s.chars() { |
| 4219 | match c { |
| 4220 | '"' => out.push_str("\\\""), |
| 4221 | '\\' => out.push_str("\\\\"), |
| 4222 | '\n' => out.push_str("\\n"), |
| 4223 | '\t' => out.push_str("\\t"), |
| 4224 | '\r' => out.push_str("\\r"), |
| 4225 | c if (c as u32) < 0x20 => { |
| 4226 | out.push_str(&fmt!("\\u{:04x}", c as u32)); |
| 4227 | } |
| 4228 | c => out.push(c), |
| 4229 | } |
| 4230 | } |
| 4231 | out.push('"'); |
| 4232 | out |
| 4233 | } |
| 4234 | Dat::U64(n) => fmt!("{}", n), |
| 4235 | Dat::Bool(b) => fmt!("{}", b), |
| 4236 | Dat::List(list) => { |
| 4237 | let items: Vec<String> = list.iter().map(dat_to_json).collect(); |
| 4238 | fmt!("[{}]", items.join(",")) |
| 4239 | } |
| 4240 | Dat::Map(m) => datmap_to_json(m), |
| 4241 | Dat::Empty => "null".to_string(), |
| 4242 | _ => "null".to_string(), |
| 4243 | } |
| 4244 | } |
| 4245 | |
| 4246 | |
| 4247 | // ┌───────────────────────────────────────────────────────────────┐ |
| 4248 | // │ Tests │ |
| 4249 | // └───────────────────────────────────────────────────────────────┘ |
| 4250 | |
| 4251 | #[cfg(test)] |
| 4252 | pub mod tests { |
| 4253 | use super::*; |
| 4254 | |
| 4255 | use crate::protocol::ImageMedia; |
| 4256 | |
| 4257 | // ── The two-depth answer, written inline ───────────────────────────────── |
| 4258 | |
| 4259 | /// The fixture the Rust and JS halves are BOTH tested against. |
| 4260 | /// |
| 4261 | /// Authored by the orchestrator and read from disk rather than transcribed into this file, |
| 4262 | /// so a case added or corrected there cannot silently stop being checked here. |
| 4263 | const FOLD_FIXTURE: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/dev/fixtures/fold_keys.json"); |
| 4264 | |
| 4265 | /// The integers in a JSON array, for the fixture's `body_chars`. |
| 4266 | fn json_numbers(json: &str, key: &str) -> Vec<usize> { |
| 4267 | let arr = match find_json_array(json, key) { |
| 4268 | Some(a) => a, |
| 4269 | None => return Vec::new(), |
| 4270 | }; |
| 4271 | arr.split(|c: char| !c.is_ascii_digit()) |
| 4272 | .filter(|s| !s.is_empty()) |
| 4273 | .filter_map(|s| s.parse::<usize>().ok()) |
| 4274 | .collect() |
| 4275 | } |
| 4276 | |
| 4277 | /// **Every case in `dev/fixtures/fold_keys.json`, driven from the file itself.** |
| 4278 | /// |
| 4279 | /// The key is the one name the two languages must agree on, and the fixture is where they |
| 4280 | /// agree. Three things are checked per case, and the third is the one that carries the risk: |
| 4281 | /// the keys, the body character counts, and the stripped text -- where a `null` means the |
| 4282 | /// stripper must leave the input EXACTLY as it found it, which is the assertion a stripper |
| 4283 | /// that rewrites too eagerly fails. |
| 4284 | #[test] |
| 4285 | fn test_every_case_in_the_shared_fold_fixture() { |
| 4286 | let json = match std::fs::read_to_string(FOLD_FIXTURE) { |
| 4287 | Ok(s) => s, |
| 4288 | Err(e) => panic!("the shared fixture must be readable at {}: {}", FOLD_FIXTURE, e), |
| 4289 | }; |
| 4290 | // The wording of the note, pinned by the fixture rather than by this file: the JS half |
| 4291 | // renders the same sentence and the two must not drift apart. |
| 4292 | let want_note = extract_json_string(&json, "_placeholder").unwrap_or_default(); |
| 4293 | assert_eq!(fold_note(7), want_note.replace("N characters", "7 characters"), |
| 4294 | "the placeholder wording has drifted from the fixture"); |
| 4295 | |
| 4296 | let cases = match extract_json_objects(&json, "cases") { |
| 4297 | Some(c) => c, |
| 4298 | None => panic!("the fixture has no `cases` array: {}", FOLD_FIXTURE), |
| 4299 | }; |
| 4300 | // A fixture that stopped being read would pass every case it no longer had, and one that |
| 4301 | // LOST a case would pass just as quietly. The floor rises with the file: 11 at first |
| 4302 | // writing, 15 once nesting, the empty body and the lone angle bracket were pinned. |
| 4303 | assert!(cases.len() >= 15, "only {} cases read from {}", cases.len(), FOLD_FIXTURE); |
| 4304 | |
| 4305 | let shut = OpenSet::new(); |
| 4306 | for c in &cases { |
| 4307 | let name = extract_json_string(c, "name").unwrap_or_default(); |
| 4308 | let input = match extract_json_string(c, "input") { |
| 4309 | Some(i) => i, |
| 4310 | None => panic!("case '{}' has no input", name), |
| 4311 | }; |
| 4312 | let found = folds(&input); |
| 4313 | |
| 4314 | let keys: Vec<String> = found.iter().map(|f| f.key()).collect(); |
| 4315 | let want: Vec<String> = extract_json_string_array(c, "keys").unwrap_or_default(); |
| 4316 | assert_eq!(keys, want, "case '{}': keys, from {:?}", name, input); |
| 4317 | |
| 4318 | let chars: Vec<usize> = found.iter().map(|f| f.chars).collect(); |
| 4319 | assert_eq!(chars, json_numbers(c, "body_chars"), |
| 4320 | "case '{}': body characters, from {:?}", name, input); |
| 4321 | |
| 4322 | let got = strip_folds(&input, &shut); |
| 4323 | match extract_json_string(c, "stripped_all_closed") { |
| 4324 | Some(w) => assert_eq!(got.as_deref(), Some(w.as_str()), |
| 4325 | "case '{}': the strip", name), |
| 4326 | None => assert!(got.is_none(), |
| 4327 | "case '{}': the stripper rewrote a text it must leave exactly as it found \ |
| 4328 | it.\n in: {:?}\n out: {:?}", name, input, got), |
| 4329 | } |
| 4330 | } |
| 4331 | } |
| 4332 | |
| 4333 | /// The fixture the two halves of the SEAM are both tested against. |
| 4334 | const SEAM_FIXTURE: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/dev/fixtures/fold_seam.json"); |
| 4335 | |
| 4336 | /// **Every case in `dev/fixtures/fold_seam.json`, driven from the file itself.** |
| 4337 | /// |
| 4338 | /// Three assertions a case, and the second is the one with the risk in it. The expansion has |
| 4339 | /// to be exactly the text `www/js/render.js` builds, because a character between the two is a |
| 4340 | /// key the page and the engine disagree about and a fold the reader opens that never leaves |
| 4341 | /// the payload. `null` means the seam must leave the text exactly as it found it, which is |
| 4342 | /// what a seam that fires on a fenced line or on `Folder:` fails. And the keys are read off |
| 4343 | /// the RESULT, so a case proves the fold that comes out of the expansion and not just the |
| 4344 | /// string. |
| 4345 | #[test] |
| 4346 | fn test_every_case_in_the_shared_seam_fixture() { |
| 4347 | let json = match std::fs::read_to_string(SEAM_FIXTURE) { |
| 4348 | Ok(s) => s, |
| 4349 | Err(e) => panic!("the shared fixture must be readable at {}: {}", SEAM_FIXTURE, e), |
| 4350 | }; |
| 4351 | let cases = match extract_json_objects(&json, "cases") { |
| 4352 | Some(c) => c, |
| 4353 | None => panic!("the fixture has no `cases` array: {}", SEAM_FIXTURE), |
| 4354 | }; |
| 4355 | // A fixture that stopped being read would pass every case it no longer had. |
| 4356 | assert!(cases.len() >= 14, "only {} cases read from {}", cases.len(), SEAM_FIXTURE); |
| 4357 | for c in &cases { |
| 4358 | let name = extract_json_string(c, "name").unwrap_or_default(); |
| 4359 | let input = match extract_json_string(c, "input") { |
| 4360 | Some(i) => i, |
| 4361 | None => panic!("case '{}' has no input", name), |
| 4362 | }; |
| 4363 | let got = seam_text(&input); |
| 4364 | match extract_json_string(c, "seamed") { |
| 4365 | Some(w) => assert_eq!(got.as_deref(), Some(w.as_str()), |
| 4366 | "case '{}': the seam", name), |
| 4367 | None => assert!(got.is_none(), |
| 4368 | "case '{}': the seam rewrote a text it must leave exactly as it found \ |
| 4369 | it.\n in: {:?}\n out: {:?}", name, input, got), |
| 4370 | } |
| 4371 | let after = got.unwrap_or(input.clone()); |
| 4372 | let keys: Vec<String> = folds(&after).iter().map(|f| f.key()).collect(); |
| 4373 | let want: Vec<String> = extract_json_string_array(c, "keys").unwrap_or_default(); |
| 4374 | assert_eq!(keys, want, "case '{}': the keys of what came out", name); |
| 4375 | } |
| 4376 | } |
| 4377 | |
| 4378 | /// **The seam refuses the two failures no wording could prevent.** |
| 4379 | /// |
| 4380 | /// `dev/PROMPT_NOTES.md` §5 measured a candidate wording that folded 8 answers in 8 and put |
| 4381 | /// nothing above the fold in 8 of 8 -- FOLD-ALL, which `dev/CONTRACT_FOLD.md` §5 calls worse |
| 4382 | /// than no control at all. A length threshold applied blindly does the same thing by another |
| 4383 | /// route. So the refusals are asserted as behaviour rather than left to the fixture's |
| 4384 | /// examples: below the lead, below the summary's words, below the body, nothing folds, and |
| 4385 | /// the reader still gets every word the model wrote. |
| 4386 | #[test] |
| 4387 | fn test_the_seam_refuses_rather_than_folding_everything() { |
| 4388 | let body = "x. ".repeat(120); |
| 4389 | let sum = "The store wins on scans and loses on isolation, so I take the file."; |
| 4390 | let lead = "Take one file per Diamond: isolation is worth more here than scan speed."; |
| 4391 | for (what, text) in [ |
| 4392 | ("nothing above the seam", fmt!("Fold: {}\n\n{}", sum, body)), |
| 4393 | ("a lead of a few words", fmt!("Short.\n\nFold: {}\n\n{}", sum, body)), |
| 4394 | ("a label, not a summary", fmt!("{}\n\nFold: Reasoning\n\n{}", lead, body)), |
| 4395 | ("nothing below the seam", fmt!("{}\n\nFold: {}\n\nTiny.", lead, sum)), |
| 4396 | ] { |
| 4397 | let got = seam_text(&text).unwrap_or_else(|| text.clone()); |
| 4398 | assert!(folds(&got).is_empty(), "{}: it folded anyway: {:?}", what, got); |
| 4399 | assert!(!got.contains("<details"), "{}: it built an element: {:?}", what, got); |
| 4400 | // Refused is not lost: every word the model wrote is still there, marker aside. |
| 4401 | for line in text.lines().filter(|l| !l.trim().is_empty()) { |
| 4402 | let want = line.trim_start_matches("Fold: "); |
| 4403 | assert!(got.contains(want), "{}: {:?} went missing from {:?}", what, want, got); |
| 4404 | } |
| 4405 | } |
| 4406 | // And the one that does qualify folds, or the four above would pass on a seam that never |
| 4407 | // fires at all. |
| 4408 | let good = fmt!("{}\n\nFold: {}\n\n{}", lead, sum, body); |
| 4409 | let got = match seam_text(&good) { |
| 4410 | Some(g) => g, |
| 4411 | None => panic!("a qualifying answer did not seam"), |
| 4412 | }; |
| 4413 | let found = folds(&got); |
| 4414 | assert_eq!(found.len(), 1, "one fold, from {:?}", got); |
| 4415 | assert_eq!(found[0].key(), fmt!("0:{}", sum)); |
| 4416 | // The blank lines CONTRACT_FOLD.md §1 calls mandatory, without which `marked` never |
| 4417 | // parses the markdown inside the element. |
| 4418 | assert!(got.contains(&fmt!("<summary>{}</summary>\n\n", sum)), "no blank line after the \ |
| 4419 | summary: {:?}", got); |
| 4420 | assert!(got.contains("\n\n</details>"), "no blank line before the close: {:?}", got); |
| 4421 | } |
| 4422 | |
| 4423 | /// **A `<details>` inside a fenced region is markup being SHOWN, not a fold.** |
| 4424 | /// |
| 4425 | /// Named rather than incidental because it is the likeliest wrong strip in the feature: a |
| 4426 | /// model quoting the convention to the user -- which the prompt now invites, since the prompt |
| 4427 | /// itself contains the markup -- would have its example silently edited out from under the |
| 4428 | /// person reading it, and the fake would steal the ordinal of the real fold below. |
| 4429 | /// |
| 4430 | /// Four shapes, and each has failed a stripper written without one of them: a fence with an |
| 4431 | /// info string, a tilde fence, a fence indented up to three spaces, and a fence INSIDE a |
| 4432 | /// real fold whose contents include a literal `</details>` that must not end the element. |
| 4433 | #[test] |
| 4434 | fn test_a_fold_inside_a_fenced_region_is_not_a_fold() { |
| 4435 | let shut = OpenSet::new(); |
| 4436 | for (what, text) in [ |
| 4437 | ("a backtick fence with an info string", |
| 4438 | "```html\n<details>\n<summary>Not a fold</summary>\n\nLiteral.\n\n</details>\n```\n"), |
| 4439 | ("a tilde fence", |
| 4440 | "~~~\n<details>\n<summary>Not a fold</summary>\n\nLiteral.\n\n</details>\n~~~\n"), |
| 4441 | ("a fence indented three spaces", |
| 4442 | " ```\n <details>\n <summary>Not a fold</summary>\n\n Literal.\n\n </details>\n ```\n"), |
| 4443 | ] { |
| 4444 | assert!(folds(text).is_empty(), "{} produced folds: {:?}", what, folds(text).len()); |
| 4445 | assert!(strip_folds(text, &shut).is_none(), "{} was rewritten", what); |
| 4446 | } |
| 4447 | |
| 4448 | // The fake takes no ordinal, so the real fold below it is fold ZERO. |
| 4449 | let mixed = "```\n<details>\n<summary>Fake</summary>\nx\n</details>\n```\n\n\ |
| 4450 | <details>\n<summary>Real</summary>\n\nYes.\n\n</details>\n"; |
| 4451 | let f = folds(mixed); |
| 4452 | assert_eq!(1, f.len(), "the fenced fake was counted as a fold"); |
| 4453 | assert_eq!("0:Real", f[0].key(), "the fenced fake consumed an ordinal"); |
| 4454 | |
| 4455 | // A run carrying an INFO STRING opens a fence and can never close one, so a `\u{60}\u{60}\u{60}rust` |
| 4456 | // line part way through a code block does not end it and hand the rest of the answer back |
| 4457 | // to the scanner as prose. |
| 4458 | let info = "```\n<details>\n<summary>Fake</summary>\nx\n```rust\nstill fenced\n```\n\n\ |
| 4459 | <details>\n<summary>Real</summary>\n\nYes.\n\n</details>\n"; |
| 4460 | assert_eq!(vec![fmt!("0:Real")], |
| 4461 | folds(info).iter().map(|f| f.key()).collect::<Vec<_>>(), |
| 4462 | "an info string closed a fence it can only open"); |
| 4463 | |
| 4464 | // And a SHORTER run does not close a longer fence, which is how a model shows a fenced |
| 4465 | // block inside a fenced block. |
| 4466 | let longer = "````\n```\n<details>\n<summary>Fake</summary>\nx\n</details>\n```\n````\n\n\ |
| 4467 | <details>\n<summary>Real</summary>\n\nYes.\n\n</details>\n"; |
| 4468 | assert_eq!(vec![fmt!("0:Real")], |
| 4469 | folds(longer).iter().map(|f| f.key()).collect::<Vec<_>>(), |
| 4470 | "a three-backtick run closed a four-backtick fence"); |
| 4471 | |
| 4472 | // A fence INSIDE a fold: the literal `</details>` in the code block must not be taken for |
| 4473 | // this element's close, or the body is cut short and the remainder left dangling. |
| 4474 | let inner = "<details>\n<summary>How to write one</summary>\n\n\ |
| 4475 | ```\n</details>\n```\n\nand that is the shape.\n\n</details>\n"; |
| 4476 | let g = folds(inner); |
| 4477 | assert_eq!(1, g.len(), "the fold with a fence in it was lost"); |
| 4478 | assert_eq!("0:How to write one", g[0].key()); |
| 4479 | assert!(inner[g[0].body.clone()].ends_with("and that is the shape."), |
| 4480 | "the body stopped at the fenced `</details>`: {:?}", &inner[g[0].body.clone()]); |
| 4481 | } |
| 4482 | |
| 4483 | /// **A nested fold belongs to its parent: no ordinal, no key, no strip of its own.** |
| 4484 | /// |
| 4485 | /// Pairing by the FIRST `</details>` rather than the matching one is the failure this guards. |
| 4486 | /// The browser's parser nests, so the renderer's idea of where the outer fold ends is the |
| 4487 | /// last tag and the scanner's was the first -- and the strip then replaced the wrong span, |
| 4488 | /// cutting the body short and leaving the tail of the element sitting in the payload it was |
| 4489 | /// meant to remove. The shared fixture measures it at 48 characters against the correct 67. |
| 4490 | /// |
| 4491 | /// The inner fold needs no key because the outer strip carries it away entirely: a key for a |
| 4492 | /// body that no longer exists is a fold the user can open to no effect. |
| 4493 | #[test] |
| 4494 | fn test_a_nested_fold_is_carried_by_its_parent() { |
| 4495 | let shut = OpenSet::new(); |
| 4496 | let text = "<details>\n<summary>Outer</summary>\n\nbefore\n\n\ |
| 4497 | <details>\n<summary>Inner</summary>\n\ndeep\n\n</details>\n\n\ |
| 4498 | after\n\n</details>\n"; |
| 4499 | let f = folds(text); |
| 4500 | assert_eq!(vec![fmt!("0:Outer")], f.iter().map(|x| x.key()).collect::<Vec<_>>(), |
| 4501 | "the inner fold took an ordinal of its own"); |
| 4502 | // The body runs to the LAST closing tag, so it holds the whole inner element. |
| 4503 | let body = &text[f[0].body.clone()]; |
| 4504 | assert!(body.starts_with("before") && body.ends_with("after"), |
| 4505 | "the outer body was cut at the inner fold's close: {:?}", body); |
| 4506 | assert!(body.contains("<summary>Inner</summary>"), |
| 4507 | "the inner element fell outside its parent's body: {:?}", body); |
| 4508 | // And the strip takes the whole of it, leaving one element where there were two. |
| 4509 | let out = match strip_folds(text, &shut) { |
| 4510 | Some(o) => o, |
| 4511 | None => panic!("the outer fold was not stripped"), |
| 4512 | }; |
| 4513 | assert_eq!(1, out.matches("<details>").count(), |
| 4514 | "the inner element survived its parent's strip: {}", out); |
| 4515 | assert!(out.contains("folded to the user, 67 characters"), "{}", out); |
| 4516 | |
| 4517 | // A nested pair consumes NOTHING, so the next top-level fold is ordinal one. |
| 4518 | let after = fmt!("{}\n<details>\n<summary>Sibling</summary>\n\nYes.\n\n</details>\n", text); |
| 4519 | assert_eq!(vec![fmt!("0:Outer"), fmt!("1:Sibling")], |
| 4520 | folds(&after).iter().map(|x| x.key()).collect::<Vec<_>>(), |
| 4521 | "the nested fold shifted the ordinal of the one after it"); |
| 4522 | |
| 4523 | // An unlabelled parent does NOT borrow its child's summary. It is not a fold, and neither |
| 4524 | // is the child, which is inside it -- so the answer is no folds rather than a fold whose |
| 4525 | // label names something else. |
| 4526 | let borrowed = "<details>\n\n<details>\n<summary>Inner</summary>\n\ndeep\n\n</details>\n\n</details>\n"; |
| 4527 | assert!(folds(borrowed).is_empty(), |
| 4528 | "an unlabelled parent wore its child's label: {:?}", |
| 4529 | folds(borrowed).iter().map(|x| x.key()).collect::<Vec<_>>()); |
| 4530 | assert!(strip_folds(borrowed, &shut).is_none()); |
| 4531 | } |
| 4532 | |
| 4533 | /// **A malformed fold is left exactly as the model wrote it.** |
| 4534 | /// |
| 4535 | /// [`strip_said`]'s rule, and the reason is the same: rewriting a malformed element replaces |
| 4536 | /// a problem the model can SEE -- its own broken markup, in its own transcript -- with one it |
| 4537 | /// cannot. An unclosed fold still keys, because its ordinal is settled the moment its summary |
| 4538 | /// is and the user may already have opened it; it is only the rewrite that stands off. |
| 4539 | #[test] |
| 4540 | fn test_a_malformed_fold_is_left_alone() { |
| 4541 | let shut = OpenSet::new(); |
| 4542 | let unclosed = "<details>\n<summary>Partial</summary>\n\nStill being written"; |
| 4543 | assert_eq!(vec![fmt!("0:Partial")], |
| 4544 | folds(unclosed).iter().map(|f| f.key()).collect::<Vec<_>>()); |
| 4545 | assert!(strip_folds(unclosed, &shut).is_none(), "an unclosed fold was rewritten"); |
| 4546 | |
| 4547 | let unlabelled = "<details>\n\nNo summary here.\n\n</details>\n"; |
| 4548 | assert!(folds(unlabelled).is_empty(), "a `<details>` with no `<summary>` keyed"); |
| 4549 | assert!(strip_folds(unlabelled, &shut).is_none(), "an unlabelled fold was rewritten"); |
| 4550 | |
| 4551 | // A well-formed fold BESIDE a malformed one is still stripped: the leniency is per fold, |
| 4552 | // not per message, or one broken element would keep a whole answer on the wire forever. |
| 4553 | let both = fmt!("{}\n\nand then\n\n{}", unlabelled.trim_end(), unclosed); |
| 4554 | assert!(strip_folds(&both, &shut).is_none()); |
| 4555 | let good = fmt!("<details>\n<summary>Good</summary>\n\nkept short.\n\n</details>\n\n{}", |
| 4556 | unclosed); |
| 4557 | let out = match strip_folds(&good, &shut) { |
| 4558 | Some(s) => s, |
| 4559 | None => panic!("the sound fold beside a broken one was not stripped"), |
| 4560 | }; |
| 4561 | assert!(out.contains("folded to the user, 11 characters"), "{}", out); |
| 4562 | assert!(out.ends_with("Still being written"), "the broken fold was touched: {}", out); |
| 4563 | |
| 4564 | // A fold with an EMPTY body is left alone too, and for the opposite reason: there is |
| 4565 | // nothing to save, and replacing nothing with a hundred characters of note would cost |
| 4566 | // tokens rather than save them. |
| 4567 | let hollow = "<details>\n<summary>Nothing in here</summary>\n\n</details>\n"; |
| 4568 | assert_eq!(vec![fmt!("0:Nothing in here")], |
| 4569 | folds(hollow).iter().map(|f| f.key()).collect::<Vec<_>>()); |
| 4570 | assert!(strip_folds(hollow, &shut).is_none(), "an empty fold grew a note: {:?}", |
| 4571 | strip_folds(hollow, &shut)); |
| 4572 | } |
| 4573 | |
| 4574 | /// **A `<` that opens no tag stays in the label, because the browser keeps it too.** |
| 4575 | /// |
| 4576 | /// The JS half reads the summary with `textContent`, which returns `a < b` unchanged. A |
| 4577 | /// stripper that treated every `<` as a tag opener would key that label `0:a` while the page |
| 4578 | /// keyed it `0:a < b` -- one label, two keys, and a fold the user opens that never travels. |
| 4579 | /// The contract says HTML TAGS removed; a `<` with no `>` after it is not one. |
| 4580 | #[test] |
| 4581 | fn test_a_lone_angle_bracket_in_a_summary_is_not_a_tag() { |
| 4582 | let text = "<details>\n<summary>when a < b</summary>\n\nBody.\n\n</details>\n"; |
| 4583 | assert_eq!(vec![fmt!("0:when a < b")], |
| 4584 | folds(text).iter().map(|f| f.key()).collect::<Vec<_>>()); |
| 4585 | // And a real tag is still removed, which is the half the fixture already pins. |
| 4586 | let tagged = "<details>\n<summary><em>when</em> a < b</summary>\n\nBody.\n\n</details>\n"; |
| 4587 | assert_eq!(vec![fmt!("0:when a < b")], |
| 4588 | folds(tagged).iter().map(|f| f.key()).collect::<Vec<_>>()); |
| 4589 | } |
| 4590 | |
| 4591 | /// **The strip reaches ALL THREE serialisation sites.** |
| 4592 | /// |
| 4593 | /// `message_to_json` has two assistant branches -- with tool calls and without -- and |
| 4594 | /// `build_anthropic_body` has an assistant text path of its own. Missing one means the same |
| 4595 | /// conversation costs different amounts through different endpoints, silently, which is |
| 4596 | /// exactly the failure [`sent_args_len`]'s doc comment records for `say`. |
| 4597 | /// |
| 4598 | /// Asserted on the finished bodies of both dialects, not on the stripper: a stripper that |
| 4599 | /// works and is never called is the shape this defect takes. |
| 4600 | #[test] |
| 4601 | fn test_the_folded_body_leaves_by_every_serialisation_path() { |
| 4602 | use rustls::crypto::ring; |
| 4603 | let _ = ring::default_provider().install_default(); |
| 4604 | let tls = Arc::new(ClientConfig::builder().dangerous() |
| 4605 | .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth()); |
| 4606 | const BODY: &str = "THE-WORKING-BEHIND-THE-FOLD"; |
| 4607 | let said = fmt!("Yes, it terminates.\n\n<details>\n<summary>the working</summary>\n\n\ |
| 4608 | {}\n\n</details>\n", BODY); |
| 4609 | let msgs = vec![ |
| 4610 | ChatMessage::user(fmt!("does it terminate?")), |
| 4611 | // The assistant branch with NO tool calls. |
| 4612 | ChatMessage::Assistant { |
| 4613 | content: MessageContent::text(said.clone()), |
| 4614 | tool_calls: Vec::new(), |
| 4615 | }, |
| 4616 | ChatMessage::user(fmt!("and again?")), |
| 4617 | // The assistant branch WITH tool calls, which formats its text separately. |
| 4618 | ChatMessage::Assistant { |
| 4619 | content: MessageContent::text(said.clone()), |
| 4620 | tool_calls: vec![crate::protocol::ToolCall { |
| 4621 | id: fmt!("call_3"), |
| 4622 | name: fmt!("file_read"), |
| 4623 | arguments: fmt!("{{\"path\":\"a.txt\"}}"), |
| 4624 | }], |
| 4625 | }, |
| 4626 | ChatMessage::tool(fmt!("call_3"), MessageContent::text(fmt!("ok"))), |
| 4627 | ]; |
| 4628 | for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] { |
| 4629 | let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone()); |
| 4630 | c.set_open_folds(Vec::new()); |
| 4631 | let body = c.build_body(&msgs, None, false); |
| 4632 | assert_eq!(0, body.matches(BODY).count(), |
| 4633 | "a closed fold's body is still on the wire via {}: {}", path, body); |
| 4634 | assert_eq!(2, body.matches("folded to the user").count(), |
| 4635 | "via {} only {} of the two assistant messages left a note, so one \ |
| 4636 | serialisation path does not strip: {}", |
| 4637 | path, body.matches("folded to the user").count(), body); |
| 4638 | // The element is KEPT: a model that sees the fold it wrote still knows it folded |
| 4639 | // something and what it called it. |
| 4640 | assert_eq!(2, body.matches("<summary>the working</summary>").count(), |
| 4641 | "the element was deleted rather than emptied, via {}: {}", path, body); |
| 4642 | assert!(body.contains("Yes, it terminates."), |
| 4643 | "the short answer above the fold was stripped too, via {}: {}", path, body); |
| 4644 | |
| 4645 | // And the key JS sends is the key Rust matches: open it and the body travels. |
| 4646 | c.set_open_folds(vec![fmt!("0:the working")]); |
| 4647 | assert_eq!(2, c.build_body(&msgs, None, false).matches(BODY).count(), |
| 4648 | "an OPEN fold was withheld via {}, so the model cannot see what the user is \ |
| 4649 | reading", path); |
| 4650 | } |
| 4651 | |
| 4652 | // The cache-marked serialisation delegates the assistant role rather than repeating it, |
| 4653 | // and this is the assertion that keeps that true. |
| 4654 | let shut = OpenSet::new(); |
| 4655 | assert_eq!(message_to_json(&msgs[1], &shut), message_to_json_cached(&msgs[1], &shut), |
| 4656 | "the cached path grew an assistant branch of its own"); |
| 4657 | } |
| 4658 | |
| 4659 | /// **What the compaction trigger measures is the text the wire will carry.** |
| 4660 | /// |
| 4661 | /// Defect F, one depth further in. [`crate::agent::compact::msg_bytes`] was taught to ask |
| 4662 | /// [`sent_args_len`] what a `say` costs; an inline fold folds the assistant's own prose |
| 4663 | /// instead, and a sizer that knew about one and not the other would leave the same defect |
| 4664 | /// standing with a new name. |
| 4665 | /// |
| 4666 | /// The second assertion is the one that matters. A closed fold sizing smaller than an open |
| 4667 | /// one is satisfied by any discount at all; that the discount is exactly what serialisation |
| 4668 | /// saves is satisfied only by asking the serialiser. |
| 4669 | #[test] |
| 4670 | fn test_the_trigger_sizes_the_folded_text_the_wire_will_carry() { |
| 4671 | use crate::agent::compact::conversation_bytes; |
| 4672 | use rustls::crypto::ring; |
| 4673 | let _ = ring::default_provider().install_default(); |
| 4674 | let tls = Arc::new(ClientConfig::builder().dangerous() |
| 4675 | .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth()); |
| 4676 | // Plain letters and spaces on both sides of the swap, so the two bodies differ by the |
| 4677 | // fold's body and by nothing an escape would change. |
| 4678 | let working = "the long working behind the fold ".repeat(40); |
| 4679 | let said = fmt!("Yes.\n\n<details>\n<summary>the working</summary>\n\n{}\n\n</details>\n", |
| 4680 | working.trim()); |
| 4681 | let msgs = vec![ |
| 4682 | ChatMessage::user(fmt!("explain")), |
| 4683 | ChatMessage::Assistant { |
| 4684 | content: MessageContent::text(said), |
| 4685 | tool_calls: Vec::new(), |
| 4686 | }, |
| 4687 | ]; |
| 4688 | let shut = OpenSet::new(); |
| 4689 | let open: OpenSet = [fmt!("0:the working")].into_iter().collect(); |
| 4690 | let sized_shut = conversation_bytes(&msgs, &shut); |
| 4691 | let sized_open = conversation_bytes(&msgs, &open); |
| 4692 | |
| 4693 | for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] { |
| 4694 | let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone()); |
| 4695 | c.set_open_folds(Vec::new()); |
| 4696 | let wire_shut = c.build_body(&msgs, None, false).len() as u64; |
| 4697 | c.set_open_folds(vec![fmt!("0:the working")]); |
| 4698 | let wire_open = c.build_body(&msgs, None, false).len() as u64; |
| 4699 | |
| 4700 | assert!(wire_shut < wire_open, "the fixture proves nothing via {}: closing the fold \ |
| 4701 | did not shrink the payload", path); |
| 4702 | assert!(sized_shut < sized_open, |
| 4703 | "a closed fold is sized as though its body were still sent, so the trigger folds \ |
| 4704 | a conversation of {} bytes that goes out as {} (via {})", |
| 4705 | sized_shut, wire_shut, path); |
| 4706 | assert_eq!(sized_open - sized_shut, wire_open - wire_shut, |
| 4707 | "the sizer books {} bytes for closing the fold and the wire saves {} (via {}), \ |
| 4708 | so the trigger is measuring a rule of its own rather than the serialiser's", |
| 4709 | sized_open - sized_shut, wire_open - wire_shut, path); |
| 4710 | } |
| 4711 | } |
| 4712 | |
| 4713 | #[test] |
| 4714 | fn test_extract_json_string() { |
| 4715 | let json = r#"{"choices":[{"delta":{"content":"hello"}}]}"#; |
| 4716 | assert_eq!(extract_json_string(json, "content"), Some("hello".to_string())); |
| 4717 | } |
| 4718 | |
| 4719 | #[test] |
| 4720 | fn test_extract_json_bool() { |
| 4721 | assert_eq!(extract_json_bool(r#"{"submit":true}"#, "submit"), Some(true)); |
| 4722 | assert_eq!(extract_json_bool(r#"{"submit": false}"#, "submit"), Some(false)); |
| 4723 | // A model that quotes the boolean is still understood. |
| 4724 | assert_eq!(extract_json_bool(r#"{"submit":"true"}"#, "submit"), Some(true)); |
| 4725 | assert_eq!(extract_json_bool(r#"{"ref":3}"#, "submit"), None); |
| 4726 | } |
| 4727 | |
| 4728 | #[test] |
| 4729 | fn test_extract_json_f64() { |
| 4730 | // The case that made a reported cost read as free: `extract_json_number` |
| 4731 | // stops at the '.', so `0.0021` was 0. |
| 4732 | assert_eq!(extract_json_number(r#"{"cost":0.0021}"#, "cost"), Some(0)); |
| 4733 | assert_eq!(extract_json_f64(r#"{"cost":0.0021}"#, "cost"), Some(0.0021)); |
| 4734 | // Whitespace, exponents both ways, a negative, and a quoted figure. |
| 4735 | assert_eq!(extract_json_f64(r#"{"cost": 1.5}"#, "cost"), Some(1.5)); |
| 4736 | assert_eq!(extract_json_f64(r#"{"cost":2.1e-5}"#, "cost"), Some(2.1e-5)); |
| 4737 | assert_eq!(extract_json_f64(r#"{"cost":3E+2}"#, "cost"), Some(300.0)); |
| 4738 | assert_eq!(extract_json_f64(r#"{"cost":-0.5,"x":1}"#, "cost"), Some(-0.5)); |
| 4739 | assert_eq!(extract_json_f64(r#"{"cost":"0.0021"}"#, "cost"), Some(0.0021)); |
| 4740 | // A whole number is still a number, and an absent key is still absent. |
| 4741 | assert_eq!(extract_json_f64(r#"{"cost":0}"#, "cost"), Some(0.0)); |
| 4742 | assert_eq!(extract_json_f64(r#"{"total":1.0}"#, "cost"), None); |
| 4743 | // A longer key that merely ends in the wanted one is not it. |
| 4744 | assert_eq!(extract_json_f64(r#"{"upstream_inference_cost":9.0}"#, "cost"), None); |
| 4745 | } |
| 4746 | |
| 4747 | #[test] |
| 4748 | fn test_parse_usage_openrouter() { |
| 4749 | // The shape OpenRouter actually returns: authoritative cost, and the |
| 4750 | // cache read nested under `prompt_tokens_details`. |
| 4751 | let body = r#"{"id":"gen-1","choices":[{"message":{"content":"hi"}}],"usage":{"prompt_tokens":10240,"completion_tokens":128,"total_tokens":10368,"cost":0.0021,"cost_details":{"upstream_inference_cost":null},"prompt_tokens_details":{"cached_tokens":9216},"completion_tokens_details":{"reasoning_tokens":0}}}"#; |
| 4752 | let u = match parse_usage(body) { |
| 4753 | Some(u) => u, |
| 4754 | None => panic!("usage not found"), |
| 4755 | }; |
| 4756 | assert_eq!(u.prompt, 10240); |
| 4757 | assert_eq!(u.completion, 128); |
| 4758 | assert_eq!(u.cached, 9216); |
| 4759 | assert_eq!(u.cost_usd, 0.0021); |
| 4760 | } |
| 4761 | |
| 4762 | #[test] |
| 4763 | fn test_parse_usage_absent_and_null() { |
| 4764 | // No usage at all, and the `"usage":null` every intermediate streamed |
| 4765 | // chunk carries: both must read as absent, so the usage chunk that came |
| 4766 | // before is not erased by the chunk that follows it. |
| 4767 | assert!(parse_usage(r#"{"choices":[{"delta":{"content":"x"}}]}"#).is_none()); |
| 4768 | assert!(parse_usage(r#"{"choices":[{"delta":{}}],"usage":null}"#).is_none()); |
| 4769 | // A provider reporting only tokens leaves cost and cache at zero, which |
| 4770 | // is "it did not say", never "it was free". |
| 4771 | let u = match parse_usage(r#"{"usage":{"prompt_tokens":4,"completion_tokens":2}}"#) { |
| 4772 | Some(u) => u, |
| 4773 | None => panic!("usage not found"), |
| 4774 | }; |
| 4775 | assert_eq!(u.cached, 0); |
| 4776 | assert_eq!(u.cost_usd, 0.0); |
| 4777 | } |
| 4778 | |
| 4779 | #[test] |
| 4780 | fn test_parse_usage_anthropic_native_cache_read() { |
| 4781 | // Anthropic's own name for the figure. A prompt cache that is working |
| 4782 | // must not read as one that is not, or the breakpoint looks inert. |
| 4783 | let u = match parse_usage( |
| 4784 | r#"{"usage":{"prompt_tokens":100,"cache_read_input_tokens":80}}"#) { |
| 4785 | Some(u) => u, |
| 4786 | None => panic!("usage not found"), |
| 4787 | }; |
| 4788 | assert_eq!(u.cached, 80); |
| 4789 | } |
| 4790 | |
| 4791 | #[test] |
| 4792 | fn test_parse_usage_flat_cached() { |
| 4793 | // A provider that flattens the cache read onto `usage` is read too. |
| 4794 | let u = match parse_usage(r#"{"usage":{"prompt_tokens":100,"cached_tokens":80}}"#) { |
| 4795 | Some(u) => u, |
| 4796 | None => panic!("usage not found"), |
| 4797 | }; |
| 4798 | assert_eq!(u.cached, 80); |
| 4799 | } |
| 4800 | |
| 4801 | #[test] |
| 4802 | fn test_extract_json_string_escaped() { |
| 4803 | let json = r#"{"choices":[{"delta":{"content":"hello \"world\""}}]}"#; |
| 4804 | assert_eq!(extract_json_string(json, "content"), Some("hello \"world\"".to_string())); |
| 4805 | } |
| 4806 | |
| 4807 | #[test] |
| 4808 | fn test_extract_json_string_newline() { |
| 4809 | let json = r#"{"choices":[{"delta":{"content":"line1\nline2"}}]}"#; |
| 4810 | assert_eq!(extract_json_string(json, "content"), Some("line1\nline2".to_string())); |
| 4811 | } |
| 4812 | |
| 4813 | #[test] |
| 4814 | fn test_parse_sse_simple() { |
| 4815 | let sse = "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\ndata: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\ndata: [DONE]\n"; |
| 4816 | let mut tokens = Vec::new(); |
| 4817 | let (full, _use) = parse_sse_stream(sse.as_bytes(), &mut |t| tokens.push(t.to_string())); |
| 4818 | assert_eq!(tokens, vec!["Hello", " world"]); |
| 4819 | assert_eq!(full, "Hello world"); |
| 4820 | } |
| 4821 | |
| 4822 | #[test] |
| 4823 | fn test_parse_sse_empty_lines() { |
| 4824 | let sse = "\r\ndata: {\"choices\":[{\"delta\":{\"content\":\"Hi\"}}]}\r\n\r\ndata: [DONE]\r\n"; |
| 4825 | let mut tokens = Vec::new(); |
| 4826 | let (full, _use) = parse_sse_stream(sse.as_bytes(), &mut |t| tokens.push(t.to_string())); |
| 4827 | assert_eq!(tokens, vec!["Hi"]); |
| 4828 | assert_eq!(full, "Hi"); |
| 4829 | } |
| 4830 | |
| 4831 | // Chunked transfer decoding is now handled inline by `LineReader`; |
| 4832 | // the standalone `dechunk` helper and its tests were removed. |
| 4833 | |
| 4834 | #[test] |
| 4835 | fn test_parse_full_response_tool_calls() { |
| 4836 | let body = r#"{"choices":[{"index":0,"message":{"role":"assistant","content":null,"tool_calls":[{"id":"call_1","type":"function","function":{"name":"file_read","arguments":"{\"path\":\"a.txt\"}"}}]},"finish_reason":"tool_calls"}],"usage":{"prompt_tokens":12,"completion_tokens":8}}"#; |
| 4837 | let (content, calls, use_) = parse_full_response(body); |
| 4838 | assert_eq!(content, ""); |
| 4839 | assert_eq!(calls.len(), 1); |
| 4840 | assert_eq!(calls[0].id, "call_1"); |
| 4841 | assert_eq!(calls[0].name, "file_read"); |
| 4842 | assert_eq!(calls[0].arguments, r#"{"path":"a.txt"}"#); |
| 4843 | assert_eq!(use_.prompt, 12); |
| 4844 | assert_eq!(use_.completion, 8); |
| 4845 | } |
| 4846 | |
| 4847 | #[test] |
| 4848 | fn test_extract_json_string_whitespace() { |
| 4849 | // Real model output has a space after the colon. |
| 4850 | assert_eq!(extract_json_string(r#"{"path": "a.txt"}"#, "path"), Some("a.txt".to_string())); |
| 4851 | assert_eq!(extract_json_string(r#"{ "content": "hi" }"#, "content"), Some("hi".to_string())); |
| 4852 | // A null value is not a string. |
| 4853 | assert_eq!(extract_json_string(r#"{"content": null, "x":"y"}"#, "content"), None); |
| 4854 | } |
| 4855 | |
| 4856 | #[test] |
| 4857 | fn test_parse_full_response_spaced() { |
| 4858 | // Whitespace after colons, as real APIs emit. |
| 4859 | let body = r#"{"choices": [{"message": {"content": null, "tool_calls": [{"id": "c1", "type": "function", "function": {"name": "file_write", "arguments": "{\"path\": \"a.txt\", \"content\": \"hi\"}"}}]}}], "usage": {"prompt_tokens": 4, "completion_tokens": 2}}"#; |
| 4860 | let (content, calls, use_) = parse_full_response(body); |
| 4861 | assert_eq!(content, ""); |
| 4862 | assert_eq!(calls.len(), 1); |
| 4863 | assert_eq!(calls[0].name, "file_write"); |
| 4864 | assert_eq!(calls[0].arguments, r#"{"path": "a.txt", "content": "hi"}"#); |
| 4865 | assert_eq!(use_.prompt, 4); |
| 4866 | assert_eq!(use_.completion, 2); |
| 4867 | // And the tool can extract the spaced args. |
| 4868 | assert_eq!(extract_json_string(&calls[0].arguments, "path"), Some("a.txt".to_string())); |
| 4869 | } |
| 4870 | |
| 4871 | #[test] |
| 4872 | fn test_parse_full_response_text() { |
| 4873 | let body = r#"{"choices":[{"message":{"role":"assistant","content":"Hello there."},"finish_reason":"stop"}],"usage":{"prompt_tokens":5,"completion_tokens":3}}"#; |
| 4874 | let (content, calls, use_) = parse_full_response(body); |
| 4875 | assert_eq!(content, "Hello there."); |
| 4876 | assert!(calls.is_empty()); |
| 4877 | assert_eq!(use_.prompt, 5); |
| 4878 | assert_eq!(use_.completion, 3); |
| 4879 | } |
| 4880 | |
| 4881 | #[test] |
| 4882 | fn test_parse_full_response_two_calls() { |
| 4883 | let body = r#"{"choices":[{"message":{"content":null,"tool_calls":[{"id":"c1","type":"function","function":{"name":"file_list","arguments":"{}"}},{"id":"c2","type":"function","function":{"name":"shell","arguments":"{\"command\":\"ls\"}"}}]}}]}"#; |
| 4884 | let (_c, calls, _use) = parse_full_response(body); |
| 4885 | assert_eq!(calls.len(), 2); |
| 4886 | assert_eq!(calls[0].name, "file_list"); |
| 4887 | assert_eq!(calls[1].name, "shell"); |
| 4888 | assert_eq!(calls[1].arguments, r#"{"command":"ls"}"#); |
| 4889 | } |
| 4890 | |
| 4891 | /// A sink that keeps the ANSWER and throws the working away. |
| 4892 | /// |
| 4893 | /// What nearly every check in this module is about: the answer is what gets |
| 4894 | /// persisted and sent back next turn, so a check that let reasoning into the |
| 4895 | /// same vector would pass on a client that confused the two. |
| 4896 | fn text_sink(out: &mut Vec<String>) -> impl FnMut(Delta<'_>) + '_ { |
| 4897 | move |d| if let Delta::Text(t) = d { out.push(t.to_string()); } |
| 4898 | } |
| 4899 | |
| 4900 | /// The other half: the working, kept and the answer thrown away. |
| 4901 | fn think_sink(out: &mut Vec<String>) -> impl FnMut(Delta<'_>) + '_ { |
| 4902 | move |d| if let Delta::Reasoning(t) = d { out.push(t.to_string()); } |
| 4903 | } |
| 4904 | |
| 4905 | /// Drive a sequence of SSE `data:` payloads through a fresh |
| 4906 | /// [`StreamAcc`], collecting the forwarded text tokens. |
| 4907 | fn run_stream(chunks: &[&str]) -> (ChatOnceResponse, Vec<String>) { |
| 4908 | let mut acc = StreamAcc::default(); |
| 4909 | let mut tokens = Vec::new(); |
| 4910 | for c in chunks { |
| 4911 | acc.ingest(c, &mut text_sink(&mut tokens)); |
| 4912 | } |
| 4913 | (acc.into_response(false, 0), tokens) |
| 4914 | } |
| 4915 | |
| 4916 | /// Drive the same payloads and keep BOTH sides, so a check can say which sink |
| 4917 | /// each piece reached rather than only that it arrived somewhere. |
| 4918 | fn run_stream_both(chunks: &[&str]) -> (ChatOnceResponse, Vec<String>, Vec<String>) { |
| 4919 | let mut acc = StreamAcc::default(); |
| 4920 | let mut tokens = Vec::new(); |
| 4921 | let mut thought = Vec::new(); |
| 4922 | for c in chunks { |
| 4923 | acc.ingest(c, &mut |d: Delta<'_>| match d { |
| 4924 | Delta::Text(t) => tokens.push(t.to_string()), |
| 4925 | Delta::Reasoning(t) => thought.push(t.to_string()), |
| 4926 | }); |
| 4927 | } |
| 4928 | (acc.into_response(false, 0), tokens, thought) |
| 4929 | } |
| 4930 | |
| 4931 | #[test] |
| 4932 | fn test_the_working_of_an_openai_dialect_model_reaches_the_page_as_it_arrives() { |
| 4933 | // Captured from OpenRouter on 2026-08-28, `z-ai/glm-4.6` by way of DeepInfra: |
| 4934 | // the reasoning is on `delta.reasoning` and repeated VERBATIM inside |
| 4935 | // `delta.reasoning_details`, and `content` is an empty string throughout it. |
| 4936 | // A round of this model spent 230 of its 300 output tokens here, and the app |
| 4937 | // showed a spinner for all of them. |
| 4938 | let (resp, tokens, thought) = run_stream_both(&[ |
| 4939 | r#"{"choices":[{"delta":{"content":"","role":"assistant","reasoning":"1","reasoning_details":[{"type":"reasoning.text","text":"1","format":"unknown","index":0}]}}]}"#, |
| 4940 | r#"{"choices":[{"delta":{"content":"","role":"assistant","reasoning":"7 x 23","reasoning_details":[{"type":"reasoning.text","text":"7 x 23","format":"unknown","index":0}]}}]}"#, |
| 4941 | r#"{"choices":[{"delta":{"content":"391","role":"assistant","reasoning":null}}]}"#, |
| 4942 | r#"{"choices":[{"delta":{},"finish_reason":"stop"}]}"#, |
| 4943 | ]); |
| 4944 | // ONE copy of each piece. `reasoning_details` says the same words again, so a |
| 4945 | // reader that took both would show every token twice. |
| 4946 | assert_eq!(thought, vec!["1", "7 x 23"], |
| 4947 | "the model's working was dropped, or doubled by reasoning_details: {:?}", thought); |
| 4948 | // A `null` reasoning field is the provider saying there is none this chunk. |
| 4949 | assert_eq!(tokens, vec!["391"], "reasoning reached the answer sink: {:?}", tokens); |
| 4950 | assert_eq!(resp.content, "391", "reasoning was accumulated as the reply"); |
| 4951 | assert_eq!(resp.thinking, "17 x 23", |
| 4952 | "the round's working was not kept on the response"); |
| 4953 | } |
| 4954 | |
| 4955 | #[test] |
| 4956 | fn test_deepseeks_own_spelling_of_its_working_is_read_too() { |
| 4957 | // `reasoning_content` is what DeepSeek's own endpoint calls it; `reasoning` is |
| 4958 | // OpenRouter's. Both are wanted -- Daimond reaches DeepSeek both ways -- and |
| 4959 | // never both in one delta, which is why one is read and then the other. |
| 4960 | let (resp, tokens, thought) = run_stream_both(&[ |
| 4961 | r#"{"choices":[{"delta":{"role":"assistant","content":null,"reasoning_content":"So the"}}]}"#, |
| 4962 | r#"{"choices":[{"delta":{"content":null,"reasoning_content":" answer is"}}]}"#, |
| 4963 | r#"{"choices":[{"delta":{"content":"391","reasoning_content":null}}]}"#, |
| 4964 | ]); |
| 4965 | assert_eq!(thought, vec!["So the", " answer is"], "{:?}", thought); |
| 4966 | assert_eq!(tokens, vec!["391"], "{:?}", tokens); |
| 4967 | assert_eq!(resp.thinking, "So the answer is"); |
| 4968 | assert_eq!(resp.content, "391"); |
| 4969 | } |
| 4970 | |
| 4971 | #[test] |
| 4972 | fn test_a_models_working_is_never_stored_as_what_it_said() { |
| 4973 | // THE DEFECT THIS WHOLE PATH IS ONE MISTAKE AWAY FROM. The reply is what gets |
| 4974 | // written into the transcript and sent back to the model next turn as its own |
| 4975 | // words. Reasoning put there is the model's working out quoted back to it as |
| 4976 | // its answer -- and the working of a tool round is mostly wrong turns. |
| 4977 | let (resp, tokens, thought) = run_stream_both(&[ |
| 4978 | r#"{"choices":[{"delta":{"content":"","reasoning":"Maybe I should delete it."}}]}"#, |
| 4979 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"call_1","function":{"name":"file_read","arguments":"{}"}}]}}]}"#, |
| 4980 | ]); |
| 4981 | assert!(resp.content.is_empty(), |
| 4982 | "the working was accumulated as the reply: {:?}", resp.content); |
| 4983 | assert!(tokens.is_empty(), "the working reached the answer sink: {:?}", tokens); |
| 4984 | assert_eq!(thought, vec!["Maybe I should delete it."]); |
| 4985 | assert_eq!(resp.tool_calls.len(), 1, "the tool call was lost"); |
| 4986 | } |
| 4987 | |
| 4988 | #[test] |
| 4989 | fn test_a_reasoning_key_inside_tool_arguments_is_not_the_models_working() { |
| 4990 | // A model writing JSON about reasoning is not reasoning. The argument text is |
| 4991 | // an escaped string, so the key form never matches -- asserted rather than |
| 4992 | // assumed, because the same trap already caught `content` once. |
| 4993 | let (resp, tokens, thought) = run_stream_both(&[ |
| 4994 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c1","function":{"name":"note_write","arguments":"{\"reasoning\":\"not mine\",\"content\":\"nor this\"}"}}]}}]}"#, |
| 4995 | ]); |
| 4996 | assert!(thought.is_empty(), "a tool argument was read as reasoning: {:?}", thought); |
| 4997 | assert!(tokens.is_empty(), "a tool argument was read as text: {:?}", tokens); |
| 4998 | assert_eq!(resp.tool_calls[0].arguments, |
| 4999 | r#"{"reasoning":"not mine","content":"nor this"}"#); |
| 5000 | } |
| 5001 | |
| 5002 | #[test] |
| 5003 | fn test_stream_acc_text_only() { |
| 5004 | let (resp, tokens) = run_stream(&[ |
| 5005 | r#"{"choices":[{"delta":{"role":"assistant","content":"Hel"}}]}"#, |
| 5006 | r#"{"choices":[{"delta":{"content":"lo!"}}]}"#, |
| 5007 | r#"{"choices":[{"delta":{}}],"usage":{"prompt_tokens":7,"completion_tokens":3}}"#, |
| 5008 | ]); |
| 5009 | assert_eq!(tokens, vec!["Hel", "lo!"]); |
| 5010 | assert_eq!(resp.content, "Hello!"); |
| 5011 | assert!(resp.tool_calls.is_empty()); |
| 5012 | assert_eq!(resp.prompt_tokens, 7); |
| 5013 | assert_eq!(resp.completion_tokens, 3); |
| 5014 | assert!(!resp.aborted); |
| 5015 | // Nothing said about cost or caching, so nothing is claimed. |
| 5016 | assert_eq!(resp.cached_tokens, 0); |
| 5017 | assert_eq!(resp.cost_usd, 0.0); |
| 5018 | } |
| 5019 | |
| 5020 | #[test] |
| 5021 | fn test_stream_acc_reported_cost_survives_later_chunks() { |
| 5022 | // The usage chunk arrives, and a `"usage":null` chunk follows it before |
| 5023 | // `[DONE]`. The reported figures must survive that. |
| 5024 | let (resp, _tokens) = run_stream(&[ |
| 5025 | r#"{"choices":[{"delta":{"content":"ok"}}],"usage":null}"#, |
| 5026 | r#"{"choices":[],"usage":{"prompt_tokens":8192,"completion_tokens":64,"cost":0.0021,"prompt_tokens_details":{"cached_tokens":7168}}}"#, |
| 5027 | r#"{"choices":[{"delta":{}}],"usage":null}"#, |
| 5028 | ]); |
| 5029 | assert_eq!(resp.prompt_tokens, 8192); |
| 5030 | assert_eq!(resp.cached_tokens, 7168); |
| 5031 | assert_eq!(resp.cost_usd, 0.0021); |
| 5032 | } |
| 5033 | |
| 5034 | #[test] |
| 5035 | fn test_stream_acc_tool_call_fragments() { |
| 5036 | // The name arrives with the first fragment; the arguments are split |
| 5037 | // across two later fragments and must be concatenated verbatim. |
| 5038 | let (resp, tokens) = run_stream(&[ |
| 5039 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"call_1","type":"function","function":{"name":"file_read","arguments":""}}]}}]}"#, |
| 5040 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"{\"path\":\""}}]}}]}"#, |
| 5041 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"a.txt\"}"}}]}}]}"#, |
| 5042 | r#"{"choices":[{"delta":{},"finish_reason":"tool_calls"}],"usage":{"prompt_tokens":12,"completion_tokens":8}}"#, |
| 5043 | ]); |
| 5044 | assert!(tokens.is_empty()); |
| 5045 | assert_eq!(resp.tool_calls.len(), 1); |
| 5046 | assert_eq!(resp.tool_calls[0].id, "call_1"); |
| 5047 | assert_eq!(resp.tool_calls[0].name, "file_read"); |
| 5048 | assert_eq!(resp.tool_calls[0].arguments, r#"{"path":"a.txt"}"#); |
| 5049 | assert_eq!(resp.prompt_tokens, 12); |
| 5050 | assert_eq!(resp.completion_tokens, 8); |
| 5051 | } |
| 5052 | |
| 5053 | #[test] |
| 5054 | fn test_stream_acc_two_parallel_calls() { |
| 5055 | // Two calls interleaved by index across chunks. |
| 5056 | let (resp, _t) = run_stream(&[ |
| 5057 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c0","function":{"name":"file_list","arguments":"{}"}}]}}]}"#, |
| 5058 | r#"{"choices":[{"delta":{"tool_calls":[{"index":1,"id":"c1","function":{"name":"file_read","arguments":"{\"path\":"}}]}}]}"#, |
| 5059 | r#"{"choices":[{"delta":{"tool_calls":[{"index":1,"function":{"arguments":"\"b.txt\"}"}}]}}]}"#, |
| 5060 | ]); |
| 5061 | assert_eq!(resp.tool_calls.len(), 2); |
| 5062 | assert_eq!(resp.tool_calls[0].name, "file_list"); |
| 5063 | assert_eq!(resp.tool_calls[0].arguments, "{}"); |
| 5064 | assert_eq!(resp.tool_calls[1].name, "file_read"); |
| 5065 | assert_eq!(resp.tool_calls[1].arguments, r#"{"path":"b.txt"}"#); |
| 5066 | } |
| 5067 | |
| 5068 | #[test] |
| 5069 | fn test_stream_acc_text_then_tool_call() { |
| 5070 | // Interim assistant text streams, then a tool call is requested. |
| 5071 | let (resp, tokens) = run_stream(&[ |
| 5072 | r#"{"choices":[{"delta":{"content":"Let me check. "}}]}"#, |
| 5073 | r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c0","function":{"name":"file_list","arguments":"{}"}}]}}]}"#, |
| 5074 | ]); |
| 5075 | assert_eq!(tokens, vec!["Let me check. "]); |
| 5076 | assert_eq!(resp.content, "Let me check. "); |
| 5077 | assert_eq!(resp.tool_calls.len(), 1); |
| 5078 | assert_eq!(resp.tool_calls[0].name, "file_list"); |
| 5079 | } |
| 5080 | |
| 5081 | #[test] |
| 5082 | fn test_message_to_json_assistant_tool_calls() { |
| 5083 | let msg = ChatMessage::Assistant { |
| 5084 | content: MessageContent::text(""), |
| 5085 | tool_calls: vec![ToolCall { |
| 5086 | id: "c1".to_string(), |
| 5087 | name: "shell".to_string(), |
| 5088 | arguments: r#"{"command":"ls"}"#.to_string(), |
| 5089 | }], |
| 5090 | }; |
| 5091 | let j = message_to_json(&msg, &std::collections::HashSet::new()); |
| 5092 | assert!(j.contains(r#""role":"assistant""#)); |
| 5093 | assert!(j.contains(r#""tool_calls""#)); |
| 5094 | assert!(j.contains(r#""name":"shell""#)); |
| 5095 | // Arguments must be re-escaped as a JSON string literal. |
| 5096 | assert!(j.contains(r#""arguments":"{\"command\":\"ls\"}""#)); |
| 5097 | } |
| 5098 | |
| 5099 | #[test] |
| 5100 | fn test_datmap_to_json() { |
| 5101 | let mut m = DaticleMap::new(); |
| 5102 | m.insert(dat!("role"), dat!("user")); |
| 5103 | m.insert(dat!("content"), dat!("hello")); |
| 5104 | let json = datmap_to_json(&m); |
| 5105 | // Keys are sorted. |
| 5106 | assert!(json.contains("\"content\":\"hello\"")); |
| 5107 | assert!(json.contains("\"role\":\"user\"")); |
| 5108 | } |
| 5109 | |
| 5110 | #[test] |
| 5111 | fn test_datmap_to_json_escaped() { |
| 5112 | let mut m = DaticleMap::new(); |
| 5113 | m.insert(dat!("content"), dat!("hello \"world\"\n")); |
| 5114 | let json = datmap_to_json(&m); |
| 5115 | assert!(json.contains("\\\"world\\\"")); |
| 5116 | assert!(json.contains("\\n")); |
| 5117 | } |
| 5118 | |
| 5119 | /// **The detail a stored `say` folds does not go over the wire, and the summary does.** |
| 5120 | /// |
| 5121 | /// This was the whole point of the tool, and the tool is gone — so this is now the guard on |
| 5122 | /// what remains of it. A conversation saved before the `<details>` convention still carries |
| 5123 | /// `say` tool_calls, and every one of them is re-sent on every later request for the life of |
| 5124 | /// that conversation. Delete the stripper with the tool and nothing on screen changes: those |
| 5125 | /// answers simply start travelling in full again, and the bill goes up on the conversations |
| 5126 | /// the feature existed to make cheap. The message below is exactly that — an assistant turn |
| 5127 | /// out of an old transcript, built by NAME rather than through any `Tool`, because there is no |
| 5128 | /// longer a variant to build it from. |
| 5129 | /// |
| 5130 | /// BOTH DIALECTS, because they serialise a call in ways that look nothing alike: one escapes |
| 5131 | /// the arguments into a JSON string, the other embeds them as an object. A rule applied at one |
| 5132 | /// site and not the other means the same conversation costs different amounts through |
| 5133 | /// different endpoints, and nothing on screen would say so. |
| 5134 | /// |
| 5135 | /// And a NON-`say` call is asserted to keep its arguments, which is what stops this from being |
| 5136 | /// a stripper aimed at everything: `file_write`'s content has to survive, or a write replayed |
| 5137 | /// to the model becomes a write of a placeholder. |
| 5138 | #[test] |
| 5139 | fn test_a_folded_detail_never_reaches_the_wire() { |
| 5140 | use rustls::crypto::ring; |
| 5141 | let _ = ring::default_provider().install_default(); |
| 5142 | let tls = Arc::new( |
| 5143 | ClientConfig::builder() |
| 5144 | .dangerous() |
| 5145 | .with_custom_certificate_verifier(Arc::new(NoVerify)) |
| 5146 | .with_no_client_auth() |
| 5147 | ); |
| 5148 | const DETAIL: &str = "THE-LONG-EXPLANATION-NOBODY-SHOULD-RESEND"; |
| 5149 | const GIST: &str = "the fence is a path allow-list"; |
| 5150 | let msgs = vec![ |
| 5151 | ChatMessage::user("explain the fence".to_string()), |
| 5152 | ChatMessage::Assistant { |
| 5153 | content: MessageContent::text(String::new()), |
| 5154 | tool_calls: vec![ |
| 5155 | crate::protocol::ToolCall { |
| 5156 | id: fmt!("c1"), |
| 5157 | name: fmt!("say"), |
| 5158 | arguments: fmt!("{{\"summary\":\"{}\",\"detail\":\"{}\"}}", GIST, DETAIL), |
| 5159 | }, |
| 5160 | crate::protocol::ToolCall { |
| 5161 | id: fmt!("c2"), |
| 5162 | name: fmt!("file_write"), |
| 5163 | arguments: fmt!("{{\"path\":\"a.md\",\"content\":\"{}\"}}", DETAIL), |
| 5164 | }, |
| 5165 | ], |
| 5166 | }, |
| 5167 | ]; |
| 5168 | for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] { |
| 5169 | let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone()); |
| 5170 | let body = c.build_body(&msgs, None, false); |
| 5171 | assert!(!body.contains(DETAIL) || body.matches(DETAIL).count() == 1, |
| 5172 | "the folded detail is still on the wire via {}: {}", path, body); |
| 5173 | // Exactly once — carried by `file_write`, never by `say`. |
| 5174 | assert_eq!(1, body.matches(DETAIL).count(), |
| 5175 | "via {} the detail appears {} times; it must survive file_write and never say", |
| 5176 | path, body.matches(DETAIL).count()); |
| 5177 | assert!(body.contains(GIST), "the summary was stripped too, via {}: {}", path, body); |
| 5178 | assert!(body.contains("folded to the user"), |
| 5179 | "nothing tells the model what became of the detail, via {}: {}", path, body); |
| 5180 | } |
| 5181 | } |
| 5182 | |
| 5183 | /// **An OPEN fold travels; a closed one does not.** |
| 5184 | /// |
| 5185 | /// The user's own gesture decides the model's working set. A fold they have closed is one they |
| 5186 | /// are done with, and re-sending it every turn buys nothing; a fold they have OPEN is one they |
| 5187 | /// are reading, and the next thing they say is likely to be about it — so the model holds what |
| 5188 | /// they are looking at. Two controls for one idea would be one control too many. |
| 5189 | /// |
| 5190 | /// Asserted BOTH WAYS from the same message, because either half alone is satisfied by a |
| 5191 | /// stripper that is simply broken: always-strip passes the closed case, never-strip passes the |
| 5192 | /// open one. |
| 5193 | #[test] |
| 5194 | fn test_an_open_fold_travels_and_a_closed_one_does_not() { |
| 5195 | use rustls::crypto::ring; |
| 5196 | let _ = ring::default_provider().install_default(); |
| 5197 | let tls = Arc::new(ClientConfig::builder().dangerous() |
| 5198 | .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth()); |
| 5199 | const DETAIL: &str = "THE-DETAIL-BEHIND-THE-FOLD"; |
| 5200 | let msgs = vec![ |
| 5201 | ChatMessage::user("explain".to_string()), |
| 5202 | ChatMessage::Assistant { |
| 5203 | content: MessageContent::text(String::new()), |
| 5204 | tool_calls: vec![crate::protocol::ToolCall { |
| 5205 | id: fmt!("call_7"), |
| 5206 | name: fmt!("say"), |
| 5207 | arguments: fmt!("{{\"summary\":\"the gist\",\"detail\":\"{}\"}}", DETAIL), |
| 5208 | }], |
| 5209 | }, |
| 5210 | ]; |
| 5211 | for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] { |
| 5212 | let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone()); |
| 5213 | |
| 5214 | c.set_open_folds(Vec::new()); |
| 5215 | assert!(!c.build_body(&msgs, None, false).contains(DETAIL), |
| 5216 | "a CLOSED fold was sent via {}", path); |
| 5217 | |
| 5218 | c.set_open_folds(vec![fmt!("call_7")]); |
| 5219 | assert!(c.build_body(&msgs, None, false).contains(DETAIL), |
| 5220 | "an OPEN fold was withheld via {}, so the model cannot see what the user is \ |
| 5221 | reading", path); |
| 5222 | |
| 5223 | // And closing it again takes it back out, which is what makes this a control rather |
| 5224 | // than a one-way door. |
| 5225 | c.set_open_folds(vec![fmt!("some_other_call")]); |
| 5226 | assert!(!c.build_body(&msgs, None, false).contains(DETAIL), |
| 5227 | "closing a fold did not take it back out of the payload, via {}", path); |
| 5228 | } |
| 5229 | } |
| 5230 | |
| 5231 | /// **What the compaction trigger measures is what the wire will carry.** |
| 5232 | /// |
| 5233 | /// [`crate::agent::compact::msg_bytes`] sized a `say` with `tc.arguments.len()` -- the full |
| 5234 | /// call as the model wrote it -- while [`strip_said`] takes a closed fold's detail out at |
| 5235 | /// serialisation. So the trigger measured a conversation nobody was going to send, spent the |
| 5236 | /// budget on bytes that leave on the way out, and folded earlier than it needed to. The |
| 5237 | /// [`Gauge`](crate::agent::compact::Gauge) absorbed part of that by recalibrating |
| 5238 | /// tokens-per-byte against the provider's real `prompt_tokens`, but the ratio is one number |
| 5239 | /// for the whole conversation, so the correction was paid for by distorting every other |
| 5240 | /// message's estimate. |
| 5241 | /// |
| 5242 | /// Two things are asserted and the second is the one that matters. A closed fold sizing |
| 5243 | /// smaller than an open one is satisfied by ANY discount, arbitrary or not; that the discount |
| 5244 | /// is exactly the number of bytes serialisation actually saves is satisfied only by asking |
| 5245 | /// the serialiser, which is what [`sent_args_len`] does. |
| 5246 | /// |
| 5247 | /// Both dialects, because the strip applies to both and a sizing that matched one of them |
| 5248 | /// would mean the same conversation folded at different lengths through different endpoints. |
| 5249 | #[test] |
| 5250 | fn test_the_fold_trigger_sizes_what_the_wire_will_carry() { |
| 5251 | use crate::agent::compact::conversation_bytes; |
| 5252 | use rustls::crypto::ring; |
| 5253 | let _ = ring::default_provider().install_default(); |
| 5254 | let tls = Arc::new(ClientConfig::builder().dangerous() |
| 5255 | .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth()); |
| 5256 | // Plain letters and spaces: nothing here is escaped differently from the note that |
| 5257 | // replaces it, so the two bodies differ by the detail and by nothing else. |
| 5258 | let detail = "the long explanation behind the fold ".repeat(40); |
| 5259 | let msgs = vec![ |
| 5260 | ChatMessage::user(fmt!("explain")), |
| 5261 | ChatMessage::Assistant { |
| 5262 | content: MessageContent::text(String::new()), |
| 5263 | tool_calls: vec![crate::protocol::ToolCall { |
| 5264 | id: fmt!("call_9"), |
| 5265 | name: fmt!("say"), |
| 5266 | arguments: fmt!("{{\"summary\":\"the gist\",\"detail\":\"{}\"}}", detail), |
| 5267 | }], |
| 5268 | }, |
| 5269 | ChatMessage::tool(fmt!("call_9"), MessageContent::text(fmt!("Shown."))), |
| 5270 | ]; |
| 5271 | let shut = OpenSet::new(); |
| 5272 | let open: OpenSet = [fmt!("call_9")].into_iter().collect(); |
| 5273 | let sized_shut = conversation_bytes(&msgs, &shut); |
| 5274 | let sized_open = conversation_bytes(&msgs, &open); |
| 5275 | |
| 5276 | for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] { |
| 5277 | let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone()); |
| 5278 | c.set_open_folds(Vec::new()); |
| 5279 | let wire_shut = c.build_body(&msgs, None, false).len() as u64; |
| 5280 | c.set_open_folds(vec![fmt!("call_9")]); |
| 5281 | let wire_open = c.build_body(&msgs, None, false).len() as u64; |
| 5282 | |
| 5283 | assert!(wire_shut < wire_open, "the fixture proves nothing via {}: closing the fold \ |
| 5284 | did not shrink the payload", path); |
| 5285 | assert!(sized_shut < sized_open, |
| 5286 | "a closed fold is sized as though its detail were still sent, so the trigger \ |
| 5287 | folds a conversation of {} bytes that goes out as {} (via {})", |
| 5288 | sized_shut, wire_shut, path); |
| 5289 | assert_eq!(sized_open - sized_shut, wire_open - wire_shut, |
| 5290 | "the sizer books {} bytes for closing the fold and the wire saves {} (via {}), \ |
| 5291 | so the trigger is measuring a rule of its own rather than the serialiser's", |
| 5292 | sized_open - sized_shut, wire_open - wire_shut, path); |
| 5293 | } |
| 5294 | } |
| 5295 | |
| 5296 | #[test] |
| 5297 | fn test_build_request_body() { |
| 5298 | use rustls::crypto::ring; |
| 5299 | let _ = ring::default_provider().install_default(); |
| 5300 | let tls = Arc::new( |
| 5301 | ClientConfig::builder() |
| 5302 | .dangerous() |
| 5303 | .with_custom_certificate_verifier(Arc::new(NoVerify)) |
| 5304 | .with_no_client_auth() |
| 5305 | ); |
| 5306 | let client = LlmClient::new("api.test.com", 443, "/v1/chat", "key", "model", 4096, tls); |
| 5307 | let messages = vec![ |
| 5308 | ChatMessage::system("You are helpful".to_string()), |
| 5309 | ChatMessage::user("Hello".to_string()), |
| 5310 | ]; |
| 5311 | let body = client.build_request_body(&messages); |
| 5312 | assert!(body.contains("\"model\":\"model\"")); |
| 5313 | assert!(body.contains("\"stream\":true")); |
| 5314 | assert!(body.contains("\"role\":\"system\"")); |
| 5315 | assert!(body.contains("\"role\":\"user\"")); |
| 5316 | assert!(body.contains("\"content\":\"You are helpful\"")); |
| 5317 | assert!(body.contains("\"content\":\"Hello\"")); |
| 5318 | } |
| 5319 | |
| 5320 | // ┌───────────────────────────────────────────────────────────────┐ |
| 5321 | // │ Retry — pure parts │ |
| 5322 | // └───────────────────────────────────────────────────────────────┘ |
| 5323 | |
| 5324 | #[test] |
| 5325 | fn test_status_retryable() { |
| 5326 | // Which statuses mean "not now" and which mean "not ever" is HTTP's |
| 5327 | // answer, not ours: 429 carries Retry-After and 5xx is the server's own |
| 5328 | // trouble, while every other 4xx describes this request. |
| 5329 | for code in [429u16, 500, 502, 503, 504, 529] { |
| 5330 | assert!(status_retryable(code), "{} should be retryable", code); |
| 5331 | } |
| 5332 | for code in [400u16, 401, 403, 404, 413, 422] { |
| 5333 | assert!(!status_retryable(code), "{} must NOT be retried", code); |
| 5334 | } |
| 5335 | } |
| 5336 | |
| 5337 | #[test] |
| 5338 | fn test_parse_retry_after() { |
| 5339 | // The delta-seconds form, which is what a provider sends. |
| 5340 | assert_eq!(parse_retry_after("2"), Some(2_000)); |
| 5341 | assert_eq!(parse_retry_after(" 30 "), Some(30_000)); |
| 5342 | assert_eq!(parse_retry_after("0"), Some(0)); |
| 5343 | // The HTTP-date form is not understood, and reads as absent rather than |
| 5344 | // as zero -- a zero would retry instantly against a provider that asked |
| 5345 | // for a minute. |
| 5346 | assert_eq!(parse_retry_after("Wed, 21 Oct 2026 07:28:00 GMT"), None); |
| 5347 | assert_eq!(parse_retry_after(""), None); |
| 5348 | } |
| 5349 | |
| 5350 | #[test] |
| 5351 | fn test_status_code_and_header_value() { |
| 5352 | assert_eq!(status_code("HTTP/1.1 429 Too Many Requests"), Some(429)); |
| 5353 | assert_eq!(status_code("HTTP/1.1 200 OK"), Some(200)); |
| 5354 | assert_eq!(status_code("garbage"), None); |
| 5355 | let head = "HTTP/1.1 429 Too Many Requests\r\nRetry-After: 3\r\nContent-Length: 0\r\n"; |
| 5356 | assert_eq!(header_value(head, "retry-after"), Some("3".to_string())); |
| 5357 | assert_eq!(header_value(head, "RETRY-AFTER"), Some("3".to_string())); |
| 5358 | assert_eq!(header_value(head, "x-absent"), None); |
| 5359 | } |
| 5360 | |
| 5361 | #[test] |
| 5362 | fn test_backoff_grows_jitters_and_is_capped() { |
| 5363 | let p = RetryPolicy { max_attempts: 6, base_ms: 100, max_backoff_ms: 400, |
| 5364 | max_total_wait_ms: 10_000 }; |
| 5365 | // Equal jitter: every delay sits in the top half of its nominal window, |
| 5366 | // so it is neither instant nor in lockstep with another worker's. |
| 5367 | let mut spread = std::collections::BTreeSet::new(); |
| 5368 | for _ in 0..64 { |
| 5369 | let d = p.delay_ms(1, None); |
| 5370 | assert!((50..=100).contains(&d), "first backoff out of band: {}", d); |
| 5371 | spread.insert(d); |
| 5372 | } |
| 5373 | assert!(spread.len() > 1, "no jitter: eight workers would retry in lockstep"); |
| 5374 | for _ in 0..16 { |
| 5375 | assert!((100..=200).contains(&p.delay_ms(2, None))); |
| 5376 | assert!((200..=400).contains(&p.delay_ms(3, None))); |
| 5377 | // Capped, not doubled forever. |
| 5378 | assert!((200..=400).contains(&p.delay_ms(9, None))); |
| 5379 | } |
| 5380 | } |
| 5381 | |
| 5382 | #[test] |
| 5383 | fn test_retry_after_is_honoured_and_never_shortened() { |
| 5384 | let p = RetryPolicy::default(); |
| 5385 | for _ in 0..32 { |
| 5386 | let d = p.delay_ms(1, Some(3_000)); |
| 5387 | // The provider is the one party that knows when it will be ready, so |
| 5388 | // its figure is a floor -- jitter is only ever added to it. |
| 5389 | assert!(d >= 3_000, "Retry-After was shortened to {}", d); |
| 5390 | assert!(d <= 3_000 + RETRY_AFTER_JITTER_MS); |
| 5391 | } |
| 5392 | } |
| 5393 | |
| 5394 | #[test] |
| 5395 | fn test_attempts_and_total_wait_are_both_bounded() { |
| 5396 | let p = RetryPolicy { max_attempts: 3, base_ms: 100, max_backoff_ms: 100, |
| 5397 | max_total_wait_ms: 10_000 }; |
| 5398 | assert!(p.next_delay(0, 0, None).is_some()); |
| 5399 | assert!(p.next_delay(1, 0, None).is_some()); |
| 5400 | // Three attempts means two retries. |
| 5401 | assert!(p.next_delay(2, 0, None).is_none()); |
| 5402 | // And a backoff that would push the total past its bound ends the |
| 5403 | // attempt, however many are left -- the user is watching a spinner. |
| 5404 | assert!(p.next_delay(0, 9_990, None).is_none()); |
| 5405 | assert!(p.next_delay(0, 0, Some(60_000)).is_none()); |
| 5406 | } |
| 5407 | |
| 5408 | // ┌───────────────────────────────────────────────────────────────┐ |
| 5409 | // │ Prompt caching │ |
| 5410 | // └───────────────────────────────────────────────────────────────┘ |
| 5411 | |
| 5412 | #[test] |
| 5413 | fn test_model_caches_on_request() { |
| 5414 | // Claude is the model family that needs an explicit breakpoint, in every |
| 5415 | // id form a caller can configure. |
| 5416 | assert!(model_caches_on_request("anthropic/claude-opus-5")); |
| 5417 | assert!(model_caches_on_request("claude-sonnet-5")); |
| 5418 | assert!(model_caches_on_request("anthropic.claude-opus-5")); |
| 5419 | assert!(model_caches_on_request("us.anthropic.claude-haiku-4.5")); |
| 5420 | assert!(model_caches_on_request("ANTHROPIC/CLAUDE-OPUS-5")); |
| 5421 | // Everything else caches automatically or not at all, and must not be |
| 5422 | // sent a marker it did not ask for. |
| 5423 | assert!(!model_caches_on_request("accounts/fireworks/models/glm-5p2")); |
| 5424 | assert!(!model_caches_on_request("openai/gpt-5.4")); |
| 5425 | assert!(!model_caches_on_request("deepseek/deepseek-v3")); |
| 5426 | assert!(!model_caches_on_request("google/gemini-3.1-pro-preview")); |
| 5427 | assert!(!model_caches_on_request("x-ai/grok-4.5")); |
| 5428 | } |
| 5429 | |
| 5430 | /// A system prompt long enough to be worth caching. |
| 5431 | fn long_system() -> String { |
| 5432 | "You are a careful assistant. ".repeat(120) |
| 5433 | } |
| 5434 | |
| 5435 | #[test] |
| 5436 | fn test_cache_breakpoints_for_a_claude_model() { |
| 5437 | let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 5438 | let messages = vec![ |
| 5439 | ChatMessage::system(long_system()), |
| 5440 | ChatMessage::user("Hello".to_string()), |
| 5441 | ]; |
| 5442 | let body = client.build_body(&messages, None, true); |
| 5443 | // The system message carries a breakpoint, in the content-block form the |
| 5444 | // marker can only live on. |
| 5445 | assert!(body.contains("\"role\":\"system\",\"content\":[{\"type\":\"text\""), |
| 5446 | "system message did not become a content block: {}", body); |
| 5447 | // Two breakpoints: the stable system prefix, and the tip of the settled |
| 5448 | // conversation for the next turn to read back. |
| 5449 | assert_eq!(body.matches("\"cache_control\":{\"type\":\"ephemeral\"}").count(), 2, |
| 5450 | "expected a system and a user breakpoint: {}", body); |
| 5451 | assert!(body.contains("\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"")); |
| 5452 | } |
| 5453 | |
| 5454 | #[test] |
| 5455 | fn test_no_cache_control_for_a_model_that_does_not_take_it() { |
| 5456 | let client = test_client("api.fireworks.ai", 443, "accounts/fireworks/models/glm-5p2"); |
| 5457 | let messages = vec![ |
| 5458 | ChatMessage::system(long_system()), |
| 5459 | ChatMessage::user("Hello".to_string()), |
| 5460 | ]; |
| 5461 | let body = client.build_body(&messages, None, true); |
| 5462 | assert!(!body.contains("cache_control"), |
| 5463 | "a marker reached a provider that never asked for one: {}", body); |
| 5464 | // And the message shape is untouched: plain string content, as before. |
| 5465 | assert!(body.contains("\"role\":\"user\",\"content\":\"Hello\"")); |
| 5466 | } |
| 5467 | |
| 5468 | #[test] |
| 5469 | fn test_a_prefix_too_short_to_cache_gets_no_breakpoint() { |
| 5470 | // Below Anthropic's minimum cacheable prefix nothing is stored, and the |
| 5471 | // provider says nothing about having declined -- so the marker is simply |
| 5472 | // not sent. |
| 5473 | let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 5474 | let messages = vec![ |
| 5475 | ChatMessage::system("Be brief.".to_string()), |
| 5476 | ChatMessage::user("Hi".to_string()), |
| 5477 | ]; |
| 5478 | assert!(!client.build_body(&messages, None, true).contains("cache_control")); |
| 5479 | } |
| 5480 | |
| 5481 | #[test] |
| 5482 | fn test_tool_definitions_count_towards_the_cacheable_prefix() { |
| 5483 | // The tools render ahead of the system message, so a large tool array is |
| 5484 | // itself most of what the breakpoint caches. |
| 5485 | let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 5486 | let messages = vec![ChatMessage::system("Be brief.".to_string())]; |
| 5487 | assert!(!client.build_body(&messages, None, true).contains("cache_control")); |
| 5488 | let tools = "[".to_string() + &"x".repeat(CACHE_MIN_PREFIX_CHARS) + "]"; |
| 5489 | assert!(client.build_body(&messages, Some(&tools), true).contains("cache_control")); |
| 5490 | } |
| 5491 | |
| 5492 | #[test] |
| 5493 | fn test_the_second_breakpoint_follows_the_conversation() { |
| 5494 | // Several turns in, the second breakpoint sits on the LAST user message, |
| 5495 | // so everything settled before it is read from the cache next round. |
| 5496 | let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 5497 | let messages = vec![ |
| 5498 | ChatMessage::system(long_system()), |
| 5499 | ChatMessage::user("first".to_string()), |
| 5500 | ChatMessage::assistant("ok".to_string()), |
| 5501 | ChatMessage::user("second".to_string()), |
| 5502 | ]; |
| 5503 | let body = client.build_body(&messages, None, true); |
| 5504 | assert!(body.contains("\"text\":\"second\",\"cache_control\""), |
| 5505 | "breakpoint is not on the latest user turn: {}", body); |
| 5506 | assert!(!body.contains("\"text\":\"first\",\"cache_control\""), |
| 5507 | "a stale breakpoint was left on an earlier turn: {}", body); |
| 5508 | assert_eq!(body.matches("cache_control").count(), 2); |
| 5509 | } |
| 5510 | |
| 5511 | #[test] |
| 5512 | fn test_a_marked_message_still_round_trips_its_escapes() { |
| 5513 | let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 5514 | let messages = vec![ |
| 5515 | ChatMessage::system(long_system() + "say \"hi\"\n"), |
| 5516 | ]; |
| 5517 | let body = client.build_body(&messages, None, true); |
| 5518 | assert!(body.contains("say \\\"hi\\\"\\n"), "escapes broke: {}", body); |
| 5519 | } |
| 5520 | |
| 5521 | // ┌───────────────────────────────────────────────────────────────┐ |
| 5522 | // │ Anthropic — dialect, request shape, headers │ |
| 5523 | // └───────────────────────────────────────────────────────────────┘ |
| 5524 | |
| 5525 | #[test] |
| 5526 | fn test_the_dialect_is_chosen_by_the_endpoint_not_the_model() { |
| 5527 | // The same Claude model is reachable both ways, so the model id cannot |
| 5528 | // decide this; the endpoint can, and does. |
| 5529 | assert_eq!(Dialect::for_endpoint("api.anthropic.com", "/v1/messages"), |
| 5530 | Dialect::Anthropic); |
| 5531 | assert_eq!(Dialect::for_endpoint("API.Anthropic.Com", "/v1/messages/"), |
| 5532 | Dialect::Anthropic); |
| 5533 | // A proxy in front of the Messages API is still speaking it. |
| 5534 | assert_eq!(Dialect::for_endpoint("gateway.example.com", "/proxy/v1/messages"), |
| 5535 | Dialect::Anthropic); |
| 5536 | // And a router serving a Claude model over chat completions is not. |
| 5537 | assert_eq!(Dialect::for_endpoint("openrouter.ai", "/api/v1/chat/completions"), |
| 5538 | Dialect::OpenAi); |
| 5539 | assert_eq!(Dialect::for_endpoint("api.fireworks.ai", "/inference/v1/chat/completions"), |
| 5540 | Dialect::OpenAi); |
| 5541 | } |
| 5542 | |
| 5543 | #[test] |
| 5544 | fn test_the_auth_headers_differ_by_dialect() { |
| 5545 | // Anthropic refuses a bearer token, wants a pinned version, and answers |
| 5546 | // a browser only when asked to. |
| 5547 | let anth = test_client_at("api.anthropic.com", 443, "/v1/messages", "claude-opus-5"); |
| 5548 | let native: Vec<String> = anth.auth_headers(false).iter() |
| 5549 | .map(|(k, v)| fmt!("{}: {}", k, v)).collect(); |
| 5550 | assert!(native.iter().any(|h| h == "x-api-key: key"), "{:?}", native); |
| 5551 | assert!(native.iter().any(|h| h == &fmt!("anthropic-version: {}", ANTHROPIC_VERSION)), |
| 5552 | "{:?}", native); |
| 5553 | assert!(!native.iter().any(|h| h.starts_with("Authorization")), |
| 5554 | "a bearer token reached the Messages API: {:?}", native); |
| 5555 | assert!(!native.iter().any(|h| h.contains("dangerous-direct-browser-access")), |
| 5556 | "the browser header was sent from a transport that is not one: {:?}", native); |
| 5557 | let browser: Vec<String> = anth.auth_headers(true).iter() |
| 5558 | .map(|(k, v)| fmt!("{}: {}", k, v)).collect(); |
| 5559 | assert!(browser.iter().any(|h| h == "anthropic-dangerous-direct-browser-access: true"), |
| 5560 | "without this header the browser call never leaves CORS: {:?}", browser); |
| 5561 | |
| 5562 | // And the OpenAI side is untouched, in either transport. |
| 5563 | let oai = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 5564 | for browser in [false, true] { |
| 5565 | let hs: Vec<String> = oai.auth_headers(browser).iter() |
| 5566 | .map(|(k, v)| fmt!("{}: {}", k, v)).collect(); |
| 5567 | assert!(hs.iter().any(|h| h == "Authorization: Bearer key"), "{:?}", hs); |
| 5568 | assert!(!hs.iter().any(|h| h.starts_with("anthropic-")), |
| 5569 | "an Anthropic header reached an OpenAI endpoint: {:?}", hs); |
| 5570 | } |
| 5571 | } |
| 5572 | |
| 5573 | /// A client speaking the Messages API to Anthropic. |
| 5574 | fn anth_client(model: &str) -> LlmClient { |
| 5575 | test_client_at("api.anthropic.com", 443, "/v1/messages", model) |
| 5576 | } |
| 5577 | |
| 5578 | // ── Images on the wire ─────────────────────────────────────────────────── |
| 5579 | // |
| 5580 | // The fixtures below are NOT what this code produces; they are what the two providers publish, |
| 5581 | // copied out of their own documents, and every one of them says where it came from. A |
| 5582 | // serialisation test written the other way round -- build with our encoder, read with our |
| 5583 | // parser -- proves only that the two halves agree with each other, which they would go on |
| 5584 | // doing while both were wrong. |
| 5585 | |
| 5586 | /// The one-pixel PNG from Anthropic's vision documentation, base64 exactly as printed there. |
| 5587 | /// |
| 5588 | /// Source: `platform.claude.com/docs/en/build-with-claude/vision`, the "Multiple images" |
| 5589 | /// example, `image1_data`. Using the provider's own bytes rather than bytes of this test's |
| 5590 | /// invention means the encoder is checked against a string a provider published, not against |
| 5591 | /// itself. |
| 5592 | const DOC_PNG_B64: &str = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nG\ |
| 5593 | P4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC"; |
| 5594 | |
| 5595 | /// Those bytes, decoded. |
| 5596 | fn doc_png() -> Vec<u8> { |
| 5597 | oxedyne_fe2o3_text::base64::decode(DOC_PNG_B64).expect("the documented base64 must decode") |
| 5598 | } |
| 5599 | |
| 5600 | /// An image part holding the documented PNG. |
| 5601 | fn doc_image(source: &str) -> ImagePart { |
| 5602 | ImagePart::new(ImageMedia::Png, doc_png(), source.to_string()) |
| 5603 | } |
| 5604 | |
| 5605 | /// The base64 encoder agrees with the provider on the provider's own bytes. |
| 5606 | /// |
| 5607 | /// The fixtures below all embed [`DOC_PNG_B64`]; if the encoder disagreed with Anthropic about |
| 5608 | /// how those bytes are spelled, every one of them would fail for a reason that had nothing to |
| 5609 | /// do with the shape being tested. This isolates that. |
| 5610 | #[test] |
| 5611 | fn test_the_base64_encoding_matches_the_providers_own_string() { |
| 5612 | let bytes = doc_png(); |
| 5613 | assert!(!bytes.is_empty(), "the documented base64 decoded to nothing"); |
| 5614 | assert_eq!(DOC_PNG_B64, oxedyne_fe2o3_text::base64::encode(&bytes), |
| 5615 | "our base64 disagrees with the string Anthropic published for these bytes"); |
| 5616 | } |
| 5617 | |
| 5618 | /// An Anthropic image block is the block Anthropic documents. |
| 5619 | /// |
| 5620 | /// Fixture source: `platform.claude.com/docs/en/build-with-claude/vision`, "Base64-encoded |
| 5621 | /// image example", the cURL request body -- `{"type":"image","source":{"type":"base64", |
| 5622 | /// "media_type":…,"data":…}}`, in that key order. |
| 5623 | #[test] |
| 5624 | fn test_an_anthropic_image_block_is_the_documented_shape() { |
| 5625 | let want = fmt!( |
| 5626 | "{{\"type\":\"image\",\"source\":{{\"type\":\"base64\",\"media_type\":\"image/png\",\ |
| 5627 | \"data\":\"{}\"}}}}", DOC_PNG_B64); |
| 5628 | let client = anth_client("claude-opus-5"); |
| 5629 | let msgs = vec![ChatMessage::user(MessageContent::parts(vec![ |
| 5630 | ContentPart::Image(doc_image("shots/after.png")), |
| 5631 | ContentPart::Text("Describe this image.".to_string()), |
| 5632 | ]))]; |
| 5633 | let body = client.build_anthropic_body(&msgs, None, true); |
| 5634 | assert!(body.contains(&want), "the image block is not the documented one.\nwant: {}\ngot: {}", |
| 5635 | want, body); |
| 5636 | // The image precedes the text, as the documentation recommends and as the part order says. |
| 5637 | let img = body.find("\"type\":\"image\"").expect("no image block"); |
| 5638 | let txt = body.find("Describe this image.").expect("no text block"); |
| 5639 | assert!(img < txt, "the parts were reordered"); |
| 5640 | } |
| 5641 | |
| 5642 | /// An OpenAI image part is the part OpenAI documents. |
| 5643 | /// |
| 5644 | /// Fixture source: OpenAI's own OpenAPI specification, schema |
| 5645 | /// `ChatCompletionRequestMessageContentPartImage` -- `type` is the constant `"image_url"`, and |
| 5646 | /// `image_url.url` is documented as "URL of the image. This can be a URL or a base64 encoded |
| 5647 | /// data URL". The data URL itself is RFC 2397 syntax, `data:<media-type>;base64,<data>`. |
| 5648 | /// `detail` is optional and defaults to `"auto"`, so it is not sent. |
| 5649 | #[test] |
| 5650 | fn test_an_openai_image_part_is_the_documented_shape() { |
| 5651 | let want = fmt!( |
| 5652 | "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:image/png;base64,{}\"}}}}", |
| 5653 | DOC_PNG_B64); |
| 5654 | let client = test_client("api.example.com", 443, "gpt-5.6"); |
| 5655 | let msgs = vec![ChatMessage::user(MessageContent::parts(vec![ |
| 5656 | ContentPart::Text("What is in this image?".to_string()), |
| 5657 | ContentPart::Image(doc_image("shots/after.png")), |
| 5658 | ]))]; |
| 5659 | let body = client.build_openai_body(&msgs, None, true); |
| 5660 | assert!(body.contains(&want), "the image part is not the documented one.\nwant: {}\ngot: {}", |
| 5661 | want, body); |
| 5662 | assert!(body.contains("\"content\":[{\"type\":\"text\",\"text\":\"What is in this image?\"}"), |
| 5663 | "an image turns the content into the documented parts array: {}", body); |
| 5664 | } |
| 5665 | |
| 5666 | /// A message with no image keeps the bare-string content it always had. |
| 5667 | /// |
| 5668 | /// The parts array is legal for text too, and switching every message to it would have been |
| 5669 | /// simpler -- and would have changed the bytes of every request every router has ever been |
| 5670 | /// sent, for nothing. |
| 5671 | #[test] |
| 5672 | fn test_text_only_content_stays_a_bare_string_on_both_sides() { |
| 5673 | let msgs = vec![ChatMessage::user("Hello".to_string())]; |
| 5674 | let openai = test_client("api.example.com", 443, "gpt-5.6") |
| 5675 | .build_openai_body(&msgs, None, true); |
| 5676 | assert!(openai.contains("{\"role\":\"user\",\"content\":\"Hello\"}"), |
| 5677 | "text content grew an array: {}", openai); |
| 5678 | let anth = anth_client("claude-opus-5").build_anthropic_body(&msgs, None, true); |
| 5679 | assert!(anth.contains("{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]}"), |
| 5680 | "the Anthropic user turn is not the block form it always was: {}", anth); |
| 5681 | } |
| 5682 | |
| 5683 | /// Anthropic takes an image inside a `tool_result`; OpenAI does not, and the image is re-homed |
| 5684 | /// into a `user` turn after the run of tool replies rather than dropped. |
| 5685 | /// |
| 5686 | /// Source for the asymmetry: Anthropic's `tool_result` content is documented as a string or an |
| 5687 | /// array of text and image blocks; OpenAI's tool-message content part union |
| 5688 | /// (`ChatCompletionRequestToolMessageContentPart`) has a text member and no image member. |
| 5689 | #[test] |
| 5690 | fn test_a_tool_result_image_rides_the_reply_on_one_side_and_a_user_turn_on_the_other() { |
| 5691 | let msgs = vec![ |
| 5692 | ChatMessage::user("look at the page".to_string()), |
| 5693 | ChatMessage::assistant_calling("", vec![ToolCall { |
| 5694 | id: "call_1".to_string(), |
| 5695 | name: "file_read".to_string(), |
| 5696 | arguments: r#"{"path":"shots/after.png"}"#.to_string(), |
| 5697 | }]), |
| 5698 | ChatMessage::tool("call_1".to_string(), MessageContent::parts(vec![ |
| 5699 | ContentPart::Text("Read the image shots/after.png.".to_string()), |
| 5700 | ContentPart::Image(doc_image("shots/after.png")), |
| 5701 | ])), |
| 5702 | ]; |
| 5703 | |
| 5704 | let anth = anth_client("claude-opus-5").build_anthropic_body(&msgs, None, true); |
| 5705 | assert!(anth.contains("\"type\":\"tool_result\",\"tool_use_id\":\"call_1\",\"content\":["), |
| 5706 | "the Anthropic tool result should carry blocks: {}", anth); |
| 5707 | let result_at = anth.find("tool_result").expect("no tool_result"); |
| 5708 | let image_at = anth.find("\"type\":\"image\"").expect("no image block"); |
| 5709 | assert!(image_at > result_at, "the image left the tool result it belongs to"); |
| 5710 | |
| 5711 | let openai = test_client("api.example.com", 443, "gpt-5.6") |
| 5712 | .build_openai_body(&msgs, None, true); |
| 5713 | // The tool reply itself is text only -- the API has nowhere else to put an image. |
| 5714 | let tool_msg = openai.find("\"role\":\"tool\"").expect("no tool message"); |
| 5715 | let img_at = openai.find("image_url").expect("the image was dropped"); |
| 5716 | assert!(img_at > tool_msg, "an image_url was put inside the tool reply"); |
| 5717 | assert!(openai[tool_msg..img_at].contains("\"role\":\"user\""), |
| 5718 | "the image should be re-homed into a user turn after the run: {}", openai); |
| 5719 | } |
| 5720 | |
| 5721 | /// A cache breakpoint survives a message that ends in an image. |
| 5722 | /// |
| 5723 | /// The marker caches everything up to the block it sits on. If it could only go on a text |
| 5724 | /// block, a user turn whose last part is the screenshot would carry no marker at all and the |
| 5725 | /// whole prefix would be re-billed on every round of the turn -- silently, since nothing |
| 5726 | /// fails. |
| 5727 | #[test] |
| 5728 | fn test_a_cache_breakpoint_survives_a_message_that_ends_in_an_image() { |
| 5729 | let ends_in_image = MessageContent::parts(vec![ |
| 5730 | ContentPart::Text("here".to_string()), |
| 5731 | ContentPart::Image(doc_image("shots/after.png")), |
| 5732 | ]); |
| 5733 | let blocks = anthropic_blocks(&ends_in_image, true); |
| 5734 | assert_eq!(2, blocks.len()); |
| 5735 | assert!(!blocks[0].contains("cache_control"), |
| 5736 | "the marker must be on the LAST block, not the first: {}", blocks[0]); |
| 5737 | assert!(blocks[1].contains("\"cache_control\":{\"type\":\"ephemeral\"}"), |
| 5738 | "a message ending in an image lost its cache breakpoint: {}", blocks[1]); |
| 5739 | |
| 5740 | // And the marker is not attached when the message is not a breakpoint. |
| 5741 | let plain = anthropic_blocks(&ends_in_image, false); |
| 5742 | assert!(!plain.iter().any(|b| b.contains("cache_control"))); |
| 5743 | } |
| 5744 | |
| 5745 | /// A model on the known-blind list is refused before the request is built, by name. |
| 5746 | #[test] |
| 5747 | fn test_a_model_that_cannot_see_is_refused_by_name() { |
| 5748 | let client = test_client("api.example.com", 443, "openai/gpt-3.5-turbo-0125"); |
| 5749 | let msgs = vec![ChatMessage::user(MessageContent::parts(vec![ |
| 5750 | ContentPart::Image(doc_image("shots/after.png")), |
| 5751 | ]))]; |
| 5752 | let e = client.vision_guard(&msgs).expect_err("a blind model must be refused"); |
| 5753 | let msg = fmt!("{}", e); |
| 5754 | assert!(msg.contains("gpt-3.5-turbo-0125"), "the refusal must name the model: {}", msg); |
| 5755 | assert!(msg.contains("cannot see"), "the refusal must say what is wrong: {}", msg); |
| 5756 | // And a turn with no image goes through on the same model, because the model is only |
| 5757 | // unusable for the thing it cannot do. |
| 5758 | assert_eq!(0, client.vision_guard(&[ChatMessage::user("hi".to_string())]) |
| 5759 | .expect("text must still be allowed")); |
| 5760 | } |
| 5761 | |
| 5762 | /// THE CLIENT'S OWN REASON MUST LEAVE THIS MODULE, because the browser's does not survive |
| 5763 | /// the crossing intact and is not the same on two browsers. |
| 5764 | /// |
| 5765 | /// A failed `fetch` reads `TypeError: Failed to fetch` in Chromium and `TypeError: Load |
| 5766 | /// failed` in WebKit for the identical event. `www/js/daimond.js` decides from that string |
| 5767 | /// whether to hand a turn back with a Continue button or write it off, so while only `err` |
| 5768 | /// crossed, that decision was a property of the browser. `crossed` puts `reason` -- which is |
| 5769 | /// this file's wording and is the same everywhere -- in front of it. |
| 5770 | #[test] |
| 5771 | fn test_a_transport_failure_carries_its_reason_out_of_this_module() { |
| 5772 | // The exact shape of the iOS case: the fetch never got a response, and the browser's |
| 5773 | // own sentence is the only thing in the error. |
| 5774 | let e = TransportErr::transient( |
| 5775 | "could not reach the provider".to_string(), |
| 5776 | err!("LLM: fetch failed: TypeError: Load failed."; IO, Network, Wire)); |
| 5777 | let out = fmt!("{}", e.crossed()); |
| 5778 | assert!(out.contains("could not reach the provider"), |
| 5779 | "the client's own reason did not cross: {}", out); |
| 5780 | assert!(out.contains("Load failed"), |
| 5781 | "the provider's -- or the browser's -- own words must survive with it: {}", out); |
| 5782 | // And a failure that is the PROVIDER answering carries a reason that says so, which is |
| 5783 | // what keeps the app from reading a 429 as a dead road. |
| 5784 | let e = TransportErr::fatal( |
| 5785 | "the provider returned HTTP 400".to_string(), |
| 5786 | err!("LLM: HTTP error: 400 Bad Request | context length exceeded"; IO, Network, Wire)); |
| 5787 | let out = fmt!("{}", e.crossed()); |
| 5788 | assert!(out.contains("the provider returned HTTP 400"), "{}", out); |
| 5789 | // `compact::looks_like_overflow` reads this text, so the body detail must still be in it. |
| 5790 | assert!(out.contains("context length"), |
| 5791 | "the refusal's own body was lost, and overflow detection reads it: {}", out); |
| 5792 | } |
| 5793 | |
| 5794 | /// A model NOT on the list is allowed through -- the list is of what is known blind, not of |
| 5795 | /// what is known to see, so a model released tomorrow is not refused today. |
| 5796 | #[test] |
| 5797 | fn test_an_unknown_model_is_assumed_to_see() { |
| 5798 | assert!(model_can_see("some-vendor/brand-new-model-9")); |
| 5799 | assert!(model_can_see("claude-opus-5")); |
| 5800 | assert!(!model_can_see("gpt-3.5-turbo")); |
| 5801 | assert!(!model_can_see("anthropic/claude-2.1")); |
| 5802 | } |
| 5803 | |
| 5804 | /// When the provider refuses a turn that carried images and its words are about images, the |
| 5805 | /// error names the model and says it cannot see -- with the provider's own sentence kept. |
| 5806 | #[test] |
| 5807 | fn test_a_provider_refusal_about_images_is_rewritten_to_name_the_model() { |
| 5808 | let client = test_client("api.example.com", 443, "some-router/mystery-model"); |
| 5809 | let raw = err!("HTTP error: 400 Bad Request: invalid_request_error: \ |
| 5810 | this model does not support image_url content"; Invalid, Input); |
| 5811 | let out = fmt!("{}", client.vision_error(raw, 1)); |
| 5812 | assert!(out.contains("some-router/mystery-model"), "the model must be named: {}", out); |
| 5813 | assert!(out.contains("not to see"), "it must say what is wrong: {}", out); |
| 5814 | assert!(out.contains("400 Bad Request"), "the provider's own words must survive: {}", out); |
| 5815 | } |
| 5816 | |
| 5817 | /// A failure unrelated to images is handed back untouched, even on a turn that carried one. |
| 5818 | #[test] |
| 5819 | fn test_an_unrelated_failure_is_not_blamed_on_the_images() { |
| 5820 | let client = test_client("api.example.com", 443, "some-router/mystery-model"); |
| 5821 | let raw = err!("HTTP error: 401 Unauthorized"; Invalid, Input); |
| 5822 | let out = fmt!("{}", client.vision_error(raw, 1)); |
| 5823 | assert!(out.contains("401 Unauthorized"), "the provider's words were lost: {}", out); |
| 5824 | assert!(!out.contains("not to see"), |
| 5825 | "an unrelated failure was rewritten as a vision failure: {}", out); |
| 5826 | assert!(!out.contains("mystery-model"), |
| 5827 | "an unrelated failure was rewritten as a vision failure: {}", out); |
| 5828 | } |
| 5829 | |
| 5830 | #[test] |
| 5831 | fn test_the_system_prompt_is_hoisted_out_of_the_messages() { |
| 5832 | // The Messages API has no system role: a system message left in the |
| 5833 | // array is a 400, and one silently dropped is an agent with no rules. |
| 5834 | let client = anth_client("claude-opus-5"); |
| 5835 | let msgs = vec![ |
| 5836 | ChatMessage::system(long_system()), |
| 5837 | ChatMessage::system("And be brief.".to_string()), |
| 5838 | ChatMessage::user("Hello".to_string()), |
| 5839 | ]; |
| 5840 | let body = client.build_anthropic_body(&msgs, None, true); |
| 5841 | assert!(body.contains("\"system\":[{\"type\":\"text\""), |
| 5842 | "no top-level system field: {}", body); |
| 5843 | assert!(!body.contains("\"role\":\"system\""), |
| 5844 | "a system message was left in the array: {}", body); |
| 5845 | // Both of them, joined, rather than only the last. |
| 5846 | assert!(body.contains("And be brief."), "the second system message was lost: {}", body); |
| 5847 | assert!(body.contains("You are a careful assistant."), "{}", body); |
| 5848 | } |
| 5849 | |
| 5850 | #[test] |
| 5851 | fn test_the_breakpoints_land_on_the_anthropic_blocks() { |
| 5852 | // The marker only exists on a content block, and the Messages API's |
| 5853 | // blocks are in different places from the OpenAI ones. |
| 5854 | let client = anth_client("claude-opus-5"); |
| 5855 | let msgs = vec![ |
| 5856 | ChatMessage::system(long_system()), |
| 5857 | ChatMessage::user("first".to_string()), |
| 5858 | ChatMessage::assistant("ok".to_string()), |
| 5859 | ChatMessage::user("second".to_string()), |
| 5860 | ]; |
| 5861 | let body = client.build_anthropic_body(&msgs, None, true); |
| 5862 | assert_eq!(body.matches("\"cache_control\":{\"type\":\"ephemeral\"}").count(), 2, |
| 5863 | "expected a system and a user breakpoint: {}", body); |
| 5864 | assert!(body.contains("\"text\":\"second\",\"cache_control\""), |
| 5865 | "the second breakpoint is not on the latest user turn: {}", body); |
| 5866 | assert!(!body.contains("\"text\":\"first\",\"cache_control\""), |
| 5867 | "a stale breakpoint was left on an earlier turn: {}", body); |
| 5868 | // The system block carries the other one. |
| 5869 | let sys_end = match body.find("}],\"messages\"") { |
| 5870 | Some(p) => p, |
| 5871 | None => panic!("no system block: {}", body), |
| 5872 | }; |
| 5873 | assert!(body[..sys_end].contains("cache_control"), |
| 5874 | "the system prefix -- the largest stable block there is -- is uncached: {}", body); |
| 5875 | } |
| 5876 | |
| 5877 | #[test] |
| 5878 | fn test_a_model_that_does_not_cache_gets_no_marker_on_this_path_either() { |
| 5879 | // The gate is the model id, and it must still be the model id here. |
| 5880 | let client = test_client_at("api.example.com", 443, "/v1/messages", "some-other-model"); |
| 5881 | let msgs = vec![ |
| 5882 | ChatMessage::system(long_system()), |
| 5883 | ChatMessage::user("Hello".to_string()), |
| 5884 | ]; |
| 5885 | let body = client.build_anthropic_body(&msgs, None, true); |
| 5886 | assert!(!body.contains("cache_control"), |
| 5887 | "a marker reached a model that never asked for one: {}", body); |
| 5888 | } |
| 5889 | |
| 5890 | #[test] |
| 5891 | fn test_a_run_of_tool_results_becomes_one_user_message() { |
| 5892 | // Two parallel tool calls produce two `Tool` messages in a row. The |
| 5893 | // Messages API wants both results as blocks of a SINGLE user turn; |
| 5894 | // sending two consecutive user messages is a different conversation. |
| 5895 | let client = anth_client("claude-opus-5"); |
| 5896 | let msgs = vec![ |
| 5897 | ChatMessage::user("list and read".to_string()), |
| 5898 | ChatMessage::Assistant { |
| 5899 | content: MessageContent::text(""), |
| 5900 | tool_calls: vec![ |
| 5901 | ToolCall { id: "t1".to_string(), name: "file_list".to_string(), |
| 5902 | arguments: "{}".to_string() }, |
| 5903 | ToolCall { id: "t2".to_string(), name: "file_read".to_string(), |
| 5904 | arguments: r#"{"path":"a.txt"}"#.to_string() }, |
| 5905 | ], |
| 5906 | }, |
| 5907 | ChatMessage::tool("t1".to_string(), "a.txt".to_string()), |
| 5908 | ChatMessage::tool("t2".to_string(), "hello".to_string()), |
| 5909 | ]; |
| 5910 | let body = client.build_anthropic_body(&msgs, None, false); |
| 5911 | assert_eq!(body.matches("\"role\":\"user\"").count(), 2, |
| 5912 | "the two tool results did not coalesce into one turn: {}", body); |
| 5913 | assert_eq!(body.matches("\"type\":\"tool_result\"").count(), 2, "{}", body); |
| 5914 | assert!(body.contains("\"tool_use_id\":\"t1\""), "{}", body); |
| 5915 | assert!(body.contains("\"tool_use_id\":\"t2\""), "{}", body); |
| 5916 | // And the assistant turn's calls are `tool_use` blocks whose input is a |
| 5917 | // JSON OBJECT -- the OpenAI form is a string, and sending that is a 400. |
| 5918 | assert!(body.contains("\"type\":\"tool_use\",\"id\":\"t2\",\"name\":\"file_read\",\ |
| 5919 | \"input\":{\"path\":\"a.txt\"}"), |
| 5920 | "the arguments were not carried as an object: {}", body); |
| 5921 | } |
| 5922 | |
| 5923 | #[test] |
| 5924 | fn test_tool_definitions_are_translated_to_the_anthropic_shape() { |
| 5925 | let tools = r#"[{"type":"function","function":{"name":"file_read", |
| 5926 | "description":"Read a file","parameters":{"type":"object","properties":{ |
| 5927 | "path":{"type":"string","description":"name"}},"required":["path"]}}}]"#; |
| 5928 | let out = openai_tools_to_anthropic(tools); |
| 5929 | assert!(out.contains("\"name\":\"file_read\""), "{}", out); |
| 5930 | assert!(out.contains("\"description\":\"Read a file\""), |
| 5931 | "the description was read from the schema instead of the function: {}", out); |
| 5932 | assert!(out.contains("\"input_schema\":{\"type\":\"object\""), |
| 5933 | "the schema is not under input_schema: {}", out); |
| 5934 | assert!(!out.contains("\"parameters\""), "the OpenAI wrapper survived: {}", out); |
| 5935 | assert!(!out.contains("\"type\":\"function\""), "{}", out); |
| 5936 | // A definition with no schema is dropped rather than sent half-built. |
| 5937 | assert_eq!(openai_tools_to_anthropic(r#"[{"type":"function","function":{"name":"x"}}]"#), |
| 5938 | "[]"); |
| 5939 | } |
| 5940 | |
| 5941 | #[test] |
| 5942 | fn test_thinking_is_asked_for_only_where_it_is_taken() { |
| 5943 | // `budget_tokens` is a 400 on every model since Opus 4.7, and adaptive |
| 5944 | // is a 400 on the ones before Opus 4.6 -- so the gate is a list, not a |
| 5945 | // family test. |
| 5946 | for id in ["claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", |
| 5947 | "claude-sonnet-5", "claude-sonnet-4-6", "claude-fable-5", "claude-mythos-5", |
| 5948 | "anthropic/claude-opus-5", "us.anthropic.claude-sonnet-5-v1"] { |
| 5949 | assert!(model_takes_adaptive_thinking(id), "{} takes adaptive thinking", id); |
| 5950 | } |
| 5951 | for id in ["claude-haiku-4-5", "claude-sonnet-4-5", "claude-opus-4-5", "claude-3-opus", |
| 5952 | "accounts/fireworks/models/glm-5p2", "openai/gpt-5.4"] { |
| 5953 | assert!(!model_takes_adaptive_thinking(id), "{} must not be sent adaptive", id); |
| 5954 | } |
| 5955 | // And the request follows the gate. |
| 5956 | let msgs = [ChatMessage::user("Hi".to_string())]; |
| 5957 | let on = anth_client("claude-opus-5").build_anthropic_body(&msgs, None, true); |
| 5958 | assert!(on.contains("\"thinking\":{\"type\":\"adaptive\",\"display\":\"summarized\"}"), |
| 5959 | "{}", on); |
| 5960 | assert!(!on.contains("budget_tokens"), "a removed parameter was sent: {}", on); |
| 5961 | let off = anth_client("claude-haiku-4-5").build_anthropic_body(&msgs, None, true); |
| 5962 | assert!(!off.contains("thinking"), "{}", off); |
| 5963 | } |
| 5964 | |
| 5965 | #[test] |
| 5966 | fn test_an_empty_message_does_not_become_an_empty_block() { |
| 5967 | // The Messages API rejects a text block with no text, where the OpenAI |
| 5968 | // side carries the empty string through without comment. One stray |
| 5969 | // empty user message would then fail every turn of the conversation. |
| 5970 | let client = anth_client("claude-opus-5"); |
| 5971 | let msgs = vec![ |
| 5972 | ChatMessage::user("hello".to_string()), |
| 5973 | ChatMessage::assistant(String::new()), |
| 5974 | ChatMessage::user(String::new()), |
| 5975 | ]; |
| 5976 | let body = client.build_anthropic_body(&msgs, None, true); |
| 5977 | assert!(!body.contains("\"text\":\"\""), "an empty text block was sent: {}", body); |
| 5978 | // And the assistant turn that says nothing and asks for nothing is left |
| 5979 | // out entirely rather than sent as a message with no content. |
| 5980 | assert_eq!(body.matches("\"role\":\"assistant\"").count(), 0, "{}", body); |
| 5981 | assert!(body.contains("\"text\":\"hello\""), "{}", body); |
| 5982 | } |
| 5983 | |
| 5984 | #[test] |
| 5985 | fn test_a_thinking_turn_is_given_room_for_the_reasoning_and_the_answer() { |
| 5986 | // `max_tokens` caps thinking AND the reply together here, and the app's |
| 5987 | // internal default is 4096 -- chosen when it only ever meant the reply. |
| 5988 | // Left alone, a hard question is answered with a truncated sentence. |
| 5989 | let msgs = [ChatMessage::user("Hi".to_string())]; |
| 5990 | let c = anth_client("claude-opus-5"); |
| 5991 | assert_eq!(c.max_tokens, 4096, "the fixture no longer reflects the app's default"); |
| 5992 | let streamed = c.build_anthropic_body(&msgs, None, true); |
| 5993 | assert!(streamed.contains(&fmt!("\"max_tokens\":{}", THINKING_MIN_MAX_TOKENS)), |
| 5994 | "a streamed thinking turn was capped at the answer-only figure: {}", streamed); |
| 5995 | // The one-shot path keeps the configured cap: a big one there is a long |
| 5996 | // silence on an open connection, which is how a request times out. |
| 5997 | let once = c.build_anthropic_body(&msgs, None, false); |
| 5998 | assert!(once.contains("\"max_tokens\":4096"), "{}", once); |
| 5999 | // And a model that does not think is not given the extra room either. |
| 6000 | let plain = anth_client("claude-haiku-4-5").build_anthropic_body(&msgs, None, true); |
| 6001 | assert!(plain.contains("\"max_tokens\":4096"), "{}", plain); |
| 6002 | } |
| 6003 | |
| 6004 | #[test] |
| 6005 | fn test_the_openai_body_is_unchanged_by_all_this() { |
| 6006 | // The regression that matters most: five providers already work through |
| 6007 | // the other dialect, and none of them may notice this. |
| 6008 | let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5"); |
| 6009 | let msgs = vec![ |
| 6010 | ChatMessage::system(long_system()), |
| 6011 | ChatMessage::user("Hello".to_string()), |
| 6012 | ]; |
| 6013 | let body = client.build_body(&msgs, None, true); |
| 6014 | assert!(body.contains("\"stream_options\":{\"include_usage\":true}"), "{}", body); |
| 6015 | assert!(body.contains("\"role\":\"system\""), "{}", body); |
| 6016 | assert!(!body.contains("\"system\":["), "{}", body); |
| 6017 | assert!(!body.contains("\"thinking\""), "{}", body); |
| 6018 | assert!(!body.contains("input_schema"), "{}", body); |
| 6019 | } |
| 6020 | |
| 6021 | // ┌───────────────────────────────────────────────────────────────┐ |
| 6022 | // │ Anthropic — the event stream │ |
| 6023 | // └───────────────────────────────────────────────────────────────┘ |
| 6024 | |
| 6025 | /// Drive a sequence of Anthropic SSE payloads through a fresh accumulator. |
| 6026 | fn run_anth(chunks: &[&str]) -> (AnthropicAcc, Vec<String>) { |
| 6027 | let mut acc = AnthropicAcc::default(); |
| 6028 | let mut tokens = Vec::new(); |
| 6029 | for c in chunks { |
| 6030 | acc.ingest(c, &mut text_sink(&mut tokens)); |
| 6031 | } |
| 6032 | (acc, tokens) |
| 6033 | } |
| 6034 | |
| 6035 | /// The same, keeping the working rather than the answer. |
| 6036 | fn run_anth_thinking(chunks: &[&str]) -> (AnthropicAcc, Vec<String>) { |
| 6037 | let mut acc = AnthropicAcc::default(); |
| 6038 | let mut thought = Vec::new(); |
| 6039 | for c in chunks { |
| 6040 | acc.ingest(c, &mut think_sink(&mut thought)); |
| 6041 | } |
| 6042 | (acc, thought) |
| 6043 | } |
| 6044 | |
| 6045 | #[test] |
| 6046 | fn test_the_anthropic_stream_rebuilds_text_and_tool_calls() { |
| 6047 | // The documented event sequence, verbatim from the streaming reference. |
| 6048 | let (acc, tokens) = run_anth(&[ |
| 6049 | r#"{"type":"message_start","message":{"id":"msg_1","usage":{"input_tokens":472,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":2}}}"#, |
| 6050 | r#"{"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}"#, |
| 6051 | r#"{"type":"ping"}"#, |
| 6052 | r#"{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Okay"}}"#, |
| 6053 | r#"{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":", checking"}}"#, |
| 6054 | r#"{"type":"content_block_stop","index":0}"#, |
| 6055 | r#"{"type":"content_block_start","index":1,"content_block":{"type":"tool_use","id":"toolu_1","name":"get_weather","input":{}}}"#, |
| 6056 | r#"{"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"{\"location\":"}}"#, |
| 6057 | r#"{"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" \"Paris\"}"}}"#, |
| 6058 | r#"{"type":"content_block_stop","index":1}"#, |
| 6059 | r#"{"type":"message_delta","delta":{"stop_reason":"tool_use"},"usage":{"output_tokens":89}}"#, |
| 6060 | r#"{"type":"message_stop"}"#, |
| 6061 | ]); |
| 6062 | assert_eq!(tokens, vec!["Okay", ", checking"]); |
| 6063 | let resp = acc.into_response(false, 0); |
| 6064 | assert_eq!(resp.content, "Okay, checking"); |
| 6065 | assert_eq!(resp.tool_calls.len(), 1); |
| 6066 | assert_eq!(resp.tool_calls[0].id, "toolu_1"); |
| 6067 | assert_eq!(resp.tool_calls[0].name, "get_weather"); |
| 6068 | assert_eq!(resp.tool_calls[0].arguments, r#"{"location": "Paris"}"#); |
| 6069 | // The `message_delta` counts are CUMULATIVE and name only what changed: |
| 6070 | // taking them wholesale would zero the input side of the bill. |
| 6071 | assert_eq!(resp.prompt_tokens, 472); |
| 6072 | assert_eq!(resp.completion_tokens, 89); |
| 6073 | } |
| 6074 | |
| 6075 | #[test] |
| 6076 | fn test_the_whole_prompt_is_counted_and_the_cache_read_named() { |
| 6077 | // Anthropic's `input_tokens` EXCLUDES what it read from and wrote to the |
| 6078 | // cache; this client's `prompt` means every prompt token processed, and |
| 6079 | // the ledger prices `prompt - cached` at the fresh rate. Reading |
| 6080 | // `input_tokens` straight across would bill a 90%-cached turn as if the |
| 6081 | // cache were not there at all. |
| 6082 | let (acc, _t) = run_anth(&[ |
| 6083 | r#"{"type":"message_start","message":{"usage":{"input_tokens":120,"cache_creation_input_tokens":40,"cache_read_input_tokens":9000,"output_tokens":1}}}"#, |
| 6084 | r#"{"type":"message_delta","delta":{},"usage":{"output_tokens":64}}"#, |
| 6085 | ]); |
| 6086 | let resp = acc.into_response(false, 0); |
| 6087 | assert_eq!(resp.prompt_tokens, 9160, "the cached prefix is part of the prompt"); |
| 6088 | assert_eq!(resp.cached_tokens, 9000); |
| 6089 | assert_eq!(resp.completion_tokens, 64); |
| 6090 | // Anthropic reports no money, so nothing is claimed about it. |
| 6091 | assert_eq!(resp.cost_usd, 0.0); |
| 6092 | } |
| 6093 | |
| 6094 | #[test] |
| 6095 | fn test_thinking_streams_are_kept_but_never_handed_over_as_the_answer() { |
| 6096 | let (acc, tokens) = run_anth(&[ |
| 6097 | r#"{"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}"#, |
| 6098 | r#"{"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Euclid: 1071 = 2 x 462 + 147"}}"#, |
| 6099 | r#"{"type":"content_block_delta","index":0,"delta":{"type":"signature_delta","signature":"EqQBCgIYAhIM"}}"#, |
| 6100 | r#"{"type":"content_block_stop","index":0}"#, |
| 6101 | r#"{"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}"#, |
| 6102 | r#"{"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"21."}}"#, |
| 6103 | ]); |
| 6104 | // The reasoning is not the reply: a caller that streamed it into the |
| 6105 | // message would persist the model's working out as its answer. |
| 6106 | assert_eq!(tokens, vec!["21."], "thinking reached the token sink: {:?}", tokens); |
| 6107 | // And it IS handed over, as its own kind, while the round is still running. |
| 6108 | // Held back until the round ended, a model that thinks for a minute and a half |
| 6109 | // is a minute and a half of blank spinner. |
| 6110 | let (_a2, thought) = run_anth_thinking(&[ |
| 6111 | r#"{"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}"#, |
| 6112 | r#"{"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Euclid: 1071 = 2 x 462 + 147"}}"#, |
| 6113 | r#"{"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"21."}}"#, |
| 6114 | ]); |
| 6115 | assert_eq!(thought, vec!["Euclid: 1071 = 2 x 462 + 147"], |
| 6116 | "the working did not reach the reasoning sink: {:?}", thought); |
| 6117 | let blocks = acc.thinking_blocks(); |
| 6118 | assert_eq!(blocks.len(), 1, "the signed block was not kept for replay"); |
| 6119 | assert!(blocks[0].contains("\"signature\":\"EqQBCgIYAhIM\""), "{}", blocks[0]); |
| 6120 | assert!(blocks[0].contains("1071 = 2 x 462 + 147"), "{}", blocks[0]); |
| 6121 | let resp = acc.into_response(false, 0); |
| 6122 | assert_eq!(resp.content, "21."); |
| 6123 | assert_eq!(resp.thinking, "Euclid: 1071 = 2 x 462 + 147", |
| 6124 | "the reasoning was neither shown nor accounted for"); |
| 6125 | } |
| 6126 | |
| 6127 | #[test] |
| 6128 | fn test_an_unsigned_thinking_run_is_not_replayed() { |
| 6129 | // A stream cut before its `signature_delta` leaves a block the API will |
| 6130 | // not verify. The run must match what the model generated, so half of |
| 6131 | // it is worse than none: sending it is a 400 on every following turn. |
| 6132 | let (acc, _t) = run_anth(&[ |
| 6133 | r#"{"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}"#, |
| 6134 | r#"{"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"half a thought"}}"#, |
| 6135 | ]); |
| 6136 | assert!(acc.thinking_blocks().is_empty(), |
| 6137 | "an unsigned block was queued for replay"); |
| 6138 | } |
| 6139 | |
| 6140 | #[test] |
| 6141 | fn test_a_redacted_thinking_block_is_replayed_verbatim() { |
| 6142 | let (acc, _t) = run_anth(&[ |
| 6143 | r#"{"type":"content_block_start","index":0,"content_block":{"type":"redacted_thinking","data":"EroBCkYIAxgCKkB"}}"#, |
| 6144 | ]); |
| 6145 | let blocks = acc.thinking_blocks(); |
| 6146 | assert_eq!(blocks.len(), 1); |
| 6147 | assert!(blocks[0].contains("\"data\":\"EroBCkYIAxgCKkB\""), |
| 6148 | "an opaque block was rebuilt rather than replayed: {}", blocks[0]); |
| 6149 | } |
| 6150 | |
| 6151 | #[test] |
| 6152 | fn test_a_stream_error_event_is_not_read_as_an_answer() { |
| 6153 | // An overload arrives INSIDE a 200 stream here, not as a status code. |
| 6154 | // Read as a short answer it would end the turn silently and wrongly. |
| 6155 | let (acc, _t) = run_anth(&[ |
| 6156 | r#"{"type":"message_start","message":{"usage":{"input_tokens":10}}}"#, |
| 6157 | r#"{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}"#, |
| 6158 | ]); |
| 6159 | let wrapped = Acc::Anthropic(acc); |
| 6160 | let e = match wrapped.stream_error() { |
| 6161 | Some(e) => e, |
| 6162 | None => panic!("the error event was swallowed"), |
| 6163 | }; |
| 6164 | assert!(e.retryable, "an overload is the provider saying 'not now'"); |
| 6165 | assert!(e.reason.contains("overloaded_error"), "{}", e.reason); |
| 6166 | // A complaint about the request is not retried, exactly as for a 400. |
| 6167 | let (bad, _t) = run_anth(&[ |
| 6168 | r#"{"type":"error","error":{"type":"invalid_request_error","message":"bad"}}"#, |
| 6169 | ]); |
| 6170 | let e = match Acc::Anthropic(bad).stream_error() { |
| 6171 | Some(e) => e, |
| 6172 | None => panic!("the error event was swallowed"), |
| 6173 | }; |
| 6174 | assert!(!e.retryable, "a malformed request was queued for another attempt"); |
| 6175 | } |
| 6176 | |
| 6177 | #[test] |
| 6178 | fn test_a_whole_anthropic_response_parses() { |
| 6179 | let body = r#"{"id":"msg_1","type":"message","role":"assistant","model":"claude-opus-5", |
| 6180 | "content":[{"type":"thinking","thinking":"work","signature":"sig1"}, |
| 6181 | {"type":"text","text":"Here you are."}, |
| 6182 | {"type":"tool_use","id":"toolu_9","name":"file_read","input":{"path":"a.txt"}}], |
| 6183 | "stop_reason":"tool_use", |
| 6184 | "usage":{"input_tokens":100,"cache_read_input_tokens":900,"output_tokens":12}}"#; |
| 6185 | let (content, calls, use_, thinking) = parse_anthropic_response(body); |
| 6186 | assert_eq!(content, "Here you are."); |
| 6187 | assert_eq!(calls.len(), 1); |
| 6188 | assert_eq!(calls[0].name, "file_read"); |
| 6189 | assert_eq!(calls[0].arguments, r#"{"path":"a.txt"}"#); |
| 6190 | assert_eq!(use_.prompt, 1000); |
| 6191 | assert_eq!(use_.cached, 900); |
| 6192 | assert_eq!(use_.completion, 12); |
| 6193 | assert_eq!(thinking.len(), 1); |
| 6194 | assert!(thinking[0].contains("\"signature\":\"sig1\""), "{}", thinking[0]); |
| 6195 | } |
| 6196 | |
| 6197 | // ┌───────────────────────────────────────────────────────────────┐ |
| 6198 | // │ A real HTTPS server to retry against │ |
| 6199 | // └───────────────────────────────────────────────────────────────┘ |
| 6200 | // |
| 6201 | // Not a mock of the client's own idea of a provider: a TCP listener, a TLS |
| 6202 | // handshake, an HTTP/1.1 status line and a chunked SSE body. What the |
| 6203 | // client does with a 429 is then observed rather than asserted about. |
| 6204 | |
| 6205 | /// One scripted reply from the stub provider. |
| 6206 | #[derive(Clone)] |
| 6207 | pub enum Reply { |
| 6208 | /// A complete response: status line, headers, body. |
| 6209 | Http { |
| 6210 | status: u16, |
| 6211 | reason: &'static str, |
| 6212 | headers: Vec<(&'static str, String)>, |
| 6213 | body: String, |
| 6214 | }, |
| 6215 | /// A chunked `text/event-stream` body. `reset_after` cuts the |
| 6216 | /// connection with an RST once that many chunks have gone out, which is |
| 6217 | /// what a provider dropping mid-answer looks like on the wire. |
| 6218 | Sse { |
| 6219 | chunks: Vec<String>, |
| 6220 | reset_after: Option<usize>, |
| 6221 | }, |
| 6222 | } |
| 6223 | |
| 6224 | impl Reply { |
| 6225 | /// A 429, optionally with the provider's own `Retry-After`. |
| 6226 | fn too_many(retry_after: Option<u64>) -> Self { |
| 6227 | let mut headers = vec![("Content-Type", "application/json".to_string())]; |
| 6228 | if let Some(s) = retry_after { |
| 6229 | headers.push(("Retry-After", fmt!("{}", s))); |
| 6230 | } |
| 6231 | Self::Http { |
| 6232 | status: 429, reason: "Too Many Requests", headers, |
| 6233 | body: "{\"error\":{\"message\":\"rate limited\"}}".to_string(), |
| 6234 | } |
| 6235 | } |
| 6236 | |
| 6237 | /// A 500, the provider's own trouble. |
| 6238 | fn server_error() -> Self { |
| 6239 | Self::Http { |
| 6240 | status: 500, reason: "Internal Server Error", headers: Vec::new(), |
| 6241 | body: "{\"error\":{\"message\":\"upstream fell over\"}}".to_string(), |
| 6242 | } |
| 6243 | } |
| 6244 | |
| 6245 | /// A 400, this request being wrong. |
| 6246 | fn bad_request() -> Self { |
| 6247 | Self::Http { |
| 6248 | status: 400, reason: "Bad Request", headers: Vec::new(), |
| 6249 | body: "{\"error\":{\"message\":\"unknown field\"}}".to_string(), |
| 6250 | } |
| 6251 | } |
| 6252 | |
| 6253 | /// A 404 with a body that says nothing about images -- which is the case that mattered: |
| 6254 | /// `vision_error` can only rewrite a refusal whose words mention pictures, and a bare |
| 6255 | /// 404 gives it nothing to work with. |
| 6256 | fn not_found() -> Self { |
| 6257 | Self::Http { |
| 6258 | status: 404, reason: "Not Found", headers: Vec::new(), |
| 6259 | body: "{\"error\":{\"message\":\"No endpoint found\"}}".to_string(), |
| 6260 | } |
| 6261 | } |
| 6262 | |
| 6263 | /// A whole answer, streamed as two deltas and a usage chunk. |
| 6264 | fn answer() -> Self { |
| 6265 | Self::Sse { |
| 6266 | chunks: vec![ |
| 6267 | "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\n".to_string(), |
| 6268 | "data: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\n".to_string(), |
| 6269 | "data: {\"choices\":[],\"usage\":{\"prompt_tokens\":11,\"completion_tokens\":2,\ |
| 6270 | \"cost\":0.0003,\"prompt_tokens_details\":{\"cached_tokens\":9}}}\n\n".to_string(), |
| 6271 | "data: [DONE]\n\n".to_string(), |
| 6272 | ], |
| 6273 | reset_after: None, |
| 6274 | } |
| 6275 | } |
| 6276 | |
| 6277 | /// An Anthropic turn that thinks, then asks for a tool. |
| 6278 | fn anth_thinks_then_calls() -> Self { |
| 6279 | Self::Sse { |
| 6280 | chunks: vec![ |
| 6281 | "event: message_start\ndata: {\"type\":\"message_start\",\"message\":\ |
| 6282 | {\"id\":\"msg_1\",\"usage\":{\"input_tokens\":30,\ |
| 6283 | \"cache_read_input_tokens\":900,\"output_tokens\":1}}}\n\n".to_string(), |
| 6284 | "event: content_block_start\ndata: {\"type\":\"content_block_start\",\ |
| 6285 | \"index\":0,\"content_block\":{\"type\":\"thinking\",\"thinking\":\"\",\ |
| 6286 | \"signature\":\"\"}}\n\n".to_string(), |
| 6287 | "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\ |
| 6288 | \"index\":0,\"delta\":{\"type\":\"thinking_delta\",\ |
| 6289 | \"thinking\":\"I should read the file.\"}}\n\n".to_string(), |
| 6290 | "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\ |
| 6291 | \"index\":0,\"delta\":{\"type\":\"signature_delta\",\ |
| 6292 | \"signature\":\"SIGNATURE-1\"}}\n\n".to_string(), |
| 6293 | "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\ |
| 6294 | \"index\":0}\n\n".to_string(), |
| 6295 | "event: content_block_start\ndata: {\"type\":\"content_block_start\",\ |
| 6296 | \"index\":1,\"content_block\":{\"type\":\"tool_use\",\"id\":\"toolu_1\",\ |
| 6297 | \"name\":\"file_read\",\"input\":{}}}\n\n".to_string(), |
| 6298 | "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\ |
| 6299 | \"index\":1,\"delta\":{\"type\":\"input_json_delta\",\ |
| 6300 | \"partial_json\":\"{\\\"path\\\":\\\"a.txt\\\"}\"}}\n\n".to_string(), |
| 6301 | "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\ |
| 6302 | \"index\":1}\n\n".to_string(), |
| 6303 | "event: message_delta\ndata: {\"type\":\"message_delta\",\ |
| 6304 | \"delta\":{\"stop_reason\":\"tool_use\"},\ |
| 6305 | \"usage\":{\"output_tokens\":40}}\n\n".to_string(), |
| 6306 | "event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n".to_string(), |
| 6307 | ], |
| 6308 | reset_after: None, |
| 6309 | } |
| 6310 | } |
| 6311 | |
| 6312 | /// A second thinking-plus-tool round, with its own signature and call id. |
| 6313 | fn anth_thinks_then_calls_again() -> Self { |
| 6314 | match Self::anth_thinks_then_calls() { |
| 6315 | Self::Sse { chunks, reset_after } => Self::Sse { |
| 6316 | chunks: chunks.iter() |
| 6317 | .map(|c| c.replace("SIGNATURE-1", "SIGNATURE-2") |
| 6318 | .replace("toolu_1", "toolu_2")) |
| 6319 | .collect(), |
| 6320 | reset_after, |
| 6321 | }, |
| 6322 | other => other, |
| 6323 | } |
| 6324 | } |
| 6325 | |
| 6326 | /// An Anthropic turn that just answers. |
| 6327 | fn anth_answer() -> Self { |
| 6328 | Self::Sse { |
| 6329 | chunks: vec![ |
| 6330 | "event: message_start\ndata: {\"type\":\"message_start\",\"message\":\ |
| 6331 | {\"id\":\"msg_2\",\"usage\":{\"input_tokens\":60,\ |
| 6332 | \"output_tokens\":1}}}\n\n".to_string(), |
| 6333 | "event: content_block_start\ndata: {\"type\":\"content_block_start\",\ |
| 6334 | \"index\":0,\"content_block\":{\"type\":\"text\",\"text\":\"\"}}\n\n" |
| 6335 | .to_string(), |
| 6336 | "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\ |
| 6337 | \"index\":0,\"delta\":{\"type\":\"text_delta\",\ |
| 6338 | \"text\":\"It says hello.\"}}\n\n".to_string(), |
| 6339 | "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\ |
| 6340 | \"index\":0}\n\n".to_string(), |
| 6341 | "event: message_delta\ndata: {\"type\":\"message_delta\",\ |
| 6342 | \"delta\":{\"stop_reason\":\"end_turn\"},\ |
| 6343 | \"usage\":{\"output_tokens\":8}}\n\n".to_string(), |
| 6344 | "event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n".to_string(), |
| 6345 | ], |
| 6346 | reset_after: None, |
| 6347 | } |
| 6348 | } |
| 6349 | } |
| 6350 | |
| 6351 | /// What the stub provider saw, readable once the turn is over. |
| 6352 | #[derive(Default)] |
| 6353 | pub struct Seen { |
| 6354 | /// One entry per accepted connection, holding the request body. |
| 6355 | pub bodies: Vec<String>, |
| 6356 | } |
| 6357 | |
| 6358 | /// A self-signed certificate and key for the stub, generated once. |
| 6359 | /// |
| 6360 | /// Real TLS, because the native transport has no other mode -- the client is |
| 6361 | /// exercised through exactly the path a provider gets. |
| 6362 | fn stub_cert() -> &'static (Vec<u8>, Vec<u8>) { |
| 6363 | static CERT: std::sync::OnceLock<(Vec<u8>, Vec<u8>)> = std::sync::OnceLock::new(); |
| 6364 | CERT.get_or_init(|| { |
| 6365 | // Under the user cache, not the tmpfs at `/tmp`. The key is written to |
| 6366 | // disk for as long as openssl takes to write it, and a private key in a |
| 6367 | // tmpfs is a private key in the machine's memory. |
| 6368 | let dir = match oxedyne_fe2o3_test::scratch::scratch_dir("daimond_llm_cert") { |
| 6369 | Ok(d) => d, |
| 6370 | Err(e) => panic!("could not make a cert directory: {}", e), |
| 6371 | }; |
| 6372 | let cert = dir.join("cert.pem"); |
| 6373 | let key = dir.join("key.pem"); |
| 6374 | let out = std::process::Command::new("openssl") |
| 6375 | // P-256, because the test verifier below advertises |
| 6376 | // `ECDSA_NISTP256_SHA256` and TLS 1.3 will not sign an RSA |
| 6377 | // certificate with any scheme it also advertises. |
| 6378 | .args(["req", "-x509", "-newkey", "ec", |
| 6379 | "-pkeyopt", "ec_paramgen_curve:prime256v1", |
| 6380 | "-nodes", "-days", "1", "-subj", "/CN=localhost"]) |
| 6381 | .arg("-keyout").arg(&key) |
| 6382 | .arg("-out").arg(&cert) |
| 6383 | .output(); |
| 6384 | let out = match out { |
| 6385 | Ok(o) => o, |
| 6386 | // Loudly, rather than skipping: a check that quietly does not run |
| 6387 | // is a check that proves nothing. |
| 6388 | Err(e) => panic!("openssl is required for the stub provider: {}", e), |
| 6389 | }; |
| 6390 | assert!(out.status.success(), "openssl failed: {}", |
| 6391 | String::from_utf8_lossy(&out.stderr)); |
| 6392 | let pair = match (std::fs::read(&cert), std::fs::read(&key)) { |
| 6393 | (Ok(c), Ok(k)) => (c, k), |
| 6394 | _ => panic!("openssl wrote no certificate"), |
| 6395 | }; |
| 6396 | let _ = std::fs::remove_dir_all(&dir); |
| 6397 | pair |
| 6398 | }) |
| 6399 | } |
| 6400 | |
| 6401 | /// Start the stub provider on an ephemeral port. |
| 6402 | /// |
| 6403 | /// Each connection is served the next reply in `script`; the last one repeats |
| 6404 | /// for as long as the client keeps trying, so "gives up" is observable as a |
| 6405 | /// connection count rather than as a hang. |
| 6406 | pub async fn start_stub(script: Vec<Reply>) -> (u16, Arc<std::sync::Mutex<Seen>>) { |
| 6407 | // THE STUB IS A TLS SERVER AND NEEDS A PROVIDER TOO. Every client helper here installs |
| 6408 | // one, so in a whole-suite run some earlier test has always installed it process-wide by |
| 6409 | // the time a stub starts, and every stub test passed. Run one of them ALONE and the |
| 6410 | // server is built first, with nothing installed, and rustls panics -- so |
| 6411 | // `cargo test -- one_test_name` failed for a reason that had nothing to do with the test. |
| 6412 | // |
| 6413 | // That is not a hypothetical: a daimon changed the retry policy, wrote a test for it, and |
| 6414 | // told the user to prove it with exactly that command. Both it and the test beside it |
| 6415 | // would have failed, and the change would have looked broken. Idempotent, so installing |
| 6416 | // it here costs nothing where a client got there first. |
| 6417 | let _ = rustls::crypto::ring::default_provider().install_default(); |
| 6418 | use tokio_rustls::rustls::ServerConfig; |
| 6419 | use tokio_rustls::rustls::pki_types::CertificateDer; |
| 6420 | use tokio_rustls::TlsAcceptor; |
| 6421 | |
| 6422 | let (cert_pem, key_pem) = stub_cert(); |
| 6423 | let certs: Vec<CertificateDer<'static>> = rustls_pemfile::certs(&mut &cert_pem[..]) |
| 6424 | .filter_map(|c| c.ok()) |
| 6425 | .collect(); |
| 6426 | let key = match rustls_pemfile::private_key(&mut &key_pem[..]) { |
| 6427 | Ok(Some(k)) => k, |
| 6428 | _ => panic!("no private key in the stub's PEM"), |
| 6429 | }; |
| 6430 | let cfg = match ServerConfig::builder().with_no_client_auth().with_single_cert(certs, key) { |
| 6431 | Ok(c) => c, |
| 6432 | Err(e) => panic!("stub TLS config: {}", e), |
| 6433 | }; |
| 6434 | let acceptor = TlsAcceptor::from(Arc::new(cfg)); |
| 6435 | |
| 6436 | let listener = match tokio::net::TcpListener::bind(("127.0.0.1", 0)).await { |
| 6437 | Ok(l) => l, |
| 6438 | Err(e) => panic!("stub listen: {}", e), |
| 6439 | }; |
| 6440 | let port = match listener.local_addr() { |
| 6441 | Ok(a) => a.port(), |
| 6442 | Err(e) => panic!("stub addr: {}", e), |
| 6443 | }; |
| 6444 | let seen = Arc::new(std::sync::Mutex::new(Seen::default())); |
| 6445 | let seen_task = seen.clone(); |
| 6446 | |
| 6447 | tokio::spawn(async move { |
| 6448 | let mut n = 0usize; |
| 6449 | loop { |
| 6450 | let (tcp, _) = match listener.accept().await { |
| 6451 | Ok(v) => v, |
| 6452 | Err(_) => return, |
| 6453 | }; |
| 6454 | let reply = script[n.min(script.len() - 1)].clone(); |
| 6455 | n += 1; |
| 6456 | let acceptor = acceptor.clone(); |
| 6457 | let seen = seen_task.clone(); |
| 6458 | tokio::spawn(async move { |
| 6459 | let mut tls = match acceptor.accept(tcp).await { |
| 6460 | Ok(s) => s, |
| 6461 | Err(_) => return, |
| 6462 | }; |
| 6463 | let body = read_request(&mut tls).await; |
| 6464 | if let Ok(mut g) = seen.lock() { |
| 6465 | g.bodies.push(body); |
| 6466 | } |
| 6467 | write_reply(&mut tls, &reply).await; |
| 6468 | }); |
| 6469 | } |
| 6470 | }); |
| 6471 | (port, seen) |
| 6472 | } |
| 6473 | |
| 6474 | /// Read one HTTP request off the stream and return its body. |
| 6475 | async fn read_request( |
| 6476 | tls: &mut tokio_rustls::server::TlsStream<tokio::net::TcpStream>, |
| 6477 | ) -> String { |
| 6478 | let mut head = Vec::new(); |
| 6479 | let mut byte = [0u8; 1]; |
| 6480 | loop { |
| 6481 | match tls.read(&mut byte).await { |
| 6482 | Ok(0) => return String::new(), |
| 6483 | Ok(_) => { |
| 6484 | head.push(byte[0]); |
| 6485 | if head.ends_with(b"\r\n\r\n") { break; } |
| 6486 | } |
| 6487 | Err(_) => return String::new(), |
| 6488 | } |
| 6489 | } |
| 6490 | let head_str = String::from_utf8_lossy(&head).to_string(); |
| 6491 | let len: usize = header_value(&head_str, "content-length") |
| 6492 | .and_then(|v| v.parse().ok()) |
| 6493 | .unwrap_or(0); |
| 6494 | let mut body = vec![0u8; len]; |
| 6495 | let mut got = 0usize; |
| 6496 | while got < len { |
| 6497 | match tls.read(&mut body[got..]).await { |
| 6498 | Ok(0) => break, |
| 6499 | Ok(n) => got += n, |
| 6500 | Err(_) => break, |
| 6501 | } |
| 6502 | } |
| 6503 | String::from_utf8_lossy(&body[..got]).to_string() |
| 6504 | } |
| 6505 | |
| 6506 | /// Serve one scripted reply. |
| 6507 | async fn write_reply( |
| 6508 | tls: &mut tokio_rustls::server::TlsStream<tokio::net::TcpStream>, |
| 6509 | reply: &Reply, |
| 6510 | ) { |
| 6511 | match reply { |
| 6512 | Reply::Http { status, reason, headers, body } => { |
| 6513 | let mut out = fmt!("HTTP/1.1 {} {}\r\n", status, reason); |
| 6514 | for (k, v) in headers { |
| 6515 | out.push_str(&fmt!("{}: {}\r\n", k, v)); |
| 6516 | } |
| 6517 | out.push_str(&fmt!("Content-Length: {}\r\n", body.len())); |
| 6518 | out.push_str("Connection: close\r\n\r\n"); |
| 6519 | out.push_str(body); |
| 6520 | let _ = tls.write_all(out.as_bytes()).await; |
| 6521 | let _ = tls.flush().await; |
| 6522 | } |
| 6523 | Reply::Sse { chunks, reset_after } => { |
| 6524 | let head = "HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\n\ |
| 6525 | Transfer-Encoding: chunked\r\nConnection: close\r\n\r\n"; |
| 6526 | let _ = tls.write_all(head.as_bytes()).await; |
| 6527 | let _ = tls.flush().await; |
| 6528 | for (i, chunk) in chunks.iter().enumerate() { |
| 6529 | if Some(i) == *reset_after { |
| 6530 | // Abrupt reset: no close_notify, no final chunk -- the |
| 6531 | // provider vanishing mid-answer. A reset discards |
| 6532 | // anything still unacknowledged, so give what has |
| 6533 | // already gone out time to land first; otherwise the |
| 6534 | // client never sees the partial and the test proves |
| 6535 | // nothing about replaying it. |
| 6536 | tokio::time::sleep(std::time::Duration::from_millis(200)).await; |
| 6537 | let _ = tls.get_ref().0.set_linger(Some(std::time::Duration::ZERO)); |
| 6538 | return; |
| 6539 | } |
| 6540 | let framed = fmt!("{:x}\r\n{}\r\n", chunk.len(), chunk); |
| 6541 | let _ = tls.write_all(framed.as_bytes()).await; |
| 6542 | let _ = tls.flush().await; |
| 6543 | } |
| 6544 | let _ = tls.write_all(b"0\r\n\r\n").await; |
| 6545 | let _ = tls.flush().await; |
| 6546 | } |
| 6547 | } |
| 6548 | } |
| 6549 | |
| 6550 | /// A client pointed at the stub, with a fast retry policy so the suite does |
| 6551 | /// not spend its time asleep. |
| 6552 | pub fn stub_client(port: u16) -> LlmClient { |
| 6553 | let mut client = test_client("localhost", port, "anthropic/claude-opus-5"); |
| 6554 | client.retry = RetryPolicy { |
| 6555 | max_attempts: 4, |
| 6556 | base_ms: 20, |
| 6557 | max_backoff_ms: 40, |
| 6558 | max_total_wait_ms: 5_000, |
| 6559 | }; |
| 6560 | client |
| 6561 | } |
| 6562 | |
| 6563 | /// A client with a certificate verifier that accepts the stub's self-signed |
| 6564 | /// certificate, at the OpenAI-compatible path. |
| 6565 | fn test_client(host: &str, port: u16, model: &str) -> LlmClient { |
| 6566 | test_client_at(host, port, "/v1/chat/completions", model) |
| 6567 | } |
| 6568 | |
| 6569 | /// The same, at an explicit path -- which is what selects the [`Dialect`]. |
| 6570 | fn test_client_at(host: &str, port: u16, path: &str, model: &str) -> LlmClient { |
| 6571 | use rustls::crypto::ring; |
| 6572 | let _ = ring::default_provider().install_default(); |
| 6573 | let tls = Arc::new( |
| 6574 | ClientConfig::builder() |
| 6575 | .dangerous() |
| 6576 | .with_custom_certificate_verifier(Arc::new(NoVerify)) |
| 6577 | .with_no_client_auth() |
| 6578 | ); |
| 6579 | LlmClient::new(host, port, path, "key", model, 4096, tls) |
| 6580 | } |
| 6581 | |
| 6582 | /// How many connections the stub accepted. |
| 6583 | pub fn connections(seen: &Arc<std::sync::Mutex<Seen>>) -> usize { |
| 6584 | match seen.lock() { |
| 6585 | Ok(g) => g.bodies.len(), |
| 6586 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 6587 | } |
| 6588 | } |
| 6589 | |
| 6590 | /// A picture the endpoint will not take costs the pictures, not the turn. |
| 6591 | /// |
| 6592 | /// THE DEFECT. A daimon read a book's cover, the request went to a text-only model, and the |
| 6593 | /// provider answered a bare 404. The turn died -- and the picture stayed in the daimon's |
| 6594 | /// stored conversation, so every later turn re-sent it and died the same way. The Diamond's |
| 6595 | /// daimon was unusable until its whole conversation was thrown away. |
| 6596 | /// |
| 6597 | /// Neither existing guard could have caught it. `model_can_see` is a list of eight ids known |
| 6598 | /// to be blind, so an unheard-of model is assumed sighted; `vision_error` only rewrites a |
| 6599 | /// refusal whose text mentions images, and this one said "No endpoint found". |
| 6600 | #[tokio::test] |
| 6601 | async fn test_a_refused_picture_costs_the_pictures_and_not_the_turn() { |
| 6602 | let (port, seen) = start_stub(vec![ |
| 6603 | Reply::not_found(), |
| 6604 | Reply::answer(), |
| 6605 | ]).await; |
| 6606 | let client = stub_client(port); |
| 6607 | let msgs = [ChatMessage::user(MessageContent::parts(vec![ |
| 6608 | ContentPart::Text("what is on this cover".to_string()), |
| 6609 | ContentPart::Image(doc_image("cover.png")), |
| 6610 | ]))]; |
| 6611 | let mut tokens = Vec::new(); |
| 6612 | let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await { |
| 6613 | Ok(r) => r, |
| 6614 | Err(e) => panic!("a refused picture must not kill the turn: {}", e), |
| 6615 | }; |
| 6616 | |
| 6617 | assert_eq!(connections(&seen), 2, "the turn was not tried again without the picture"); |
| 6618 | assert_eq!(resp.content, "Hello world", "the second attempt did not produce the answer"); |
| 6619 | |
| 6620 | let bodies = match seen.lock() { |
| 6621 | Ok(g) => g.bodies.clone(), |
| 6622 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 6623 | }; |
| 6624 | // The first attempt carried it, so the failure being recovered from is the real one. |
| 6625 | assert!(bodies[0].contains(DOC_PNG_B64), "the first request did not carry the picture"); |
| 6626 | // The second did not, and says why in its place -- a silently dropped image would leave |
| 6627 | // the model describing a cover nobody showed it. |
| 6628 | assert!(!bodies[1].contains(DOC_PNG_B64), "the picture was sent a second time"); |
| 6629 | assert!(bodies[1].contains("cannot be shown"), |
| 6630 | "the model was not told the picture was left out: {}", bodies[1]); |
| 6631 | assert!(bodies[1].contains("cover.png"), "the file was not named in its place"); |
| 6632 | assert!(bodies[1].contains("what is on this cover"), "the prose beside it was lost"); |
| 6633 | // And the user is told, because a turn that quietly stops seeing is its own defect. |
| 6634 | assert!(tokens.iter().any(|t| t.contains("cannot see")), |
| 6635 | "nothing said the model had turned out to be blind: {:?}", tokens); |
| 6636 | } |
| 6637 | |
| 6638 | /// And once it is known, no later turn pays to discover it again. |
| 6639 | #[tokio::test] |
| 6640 | async fn test_an_endpoint_caught_refusing_pictures_is_not_asked_twice() { |
| 6641 | let (port, seen) = start_stub(vec![ |
| 6642 | Reply::not_found(), |
| 6643 | Reply::answer(), |
| 6644 | Reply::answer(), |
| 6645 | ]).await; |
| 6646 | let client = stub_client(port); |
| 6647 | let msgs = [ChatMessage::user(MessageContent::parts(vec![ |
| 6648 | ContentPart::Text("and this one".to_string()), |
| 6649 | ContentPart::Image(doc_image("cover.png")), |
| 6650 | ]))]; |
| 6651 | let mut sink = |_: Delta<'_>| {}; |
| 6652 | let _ = client.chat_stream_tools(&msgs, None, &mut sink).await |
| 6653 | .expect("the first turn recovers"); |
| 6654 | let _ = client.chat_stream_tools(&msgs, None, &mut sink).await |
| 6655 | .expect("the second turn goes straight through"); |
| 6656 | |
| 6657 | // Three replies were queued and only three connections may have been made: two for the |
| 6658 | // first turn, ONE for the second. A fourth would mean the client had forgotten. |
| 6659 | assert_eq!(connections(&seen), 3, "the second turn re-sent a picture already refused"); |
| 6660 | let bodies = match seen.lock() { |
| 6661 | Ok(g) => g.bodies.clone(), |
| 6662 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 6663 | }; |
| 6664 | assert!(!bodies[2].contains(DOC_PNG_B64), |
| 6665 | "the second turn sent the picture the endpoint had already refused"); |
| 6666 | } |
| 6667 | |
| 6668 | #[tokio::test] |
| 6669 | async fn test_a_429_is_retried_and_the_turn_completes() { |
| 6670 | let (port, seen) = start_stub(vec![ |
| 6671 | Reply::too_many(Some(1)), |
| 6672 | Reply::answer(), |
| 6673 | ]).await; |
| 6674 | let client = stub_client(port); |
| 6675 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6676 | let mut tokens = Vec::new(); |
| 6677 | |
| 6678 | let started = std::time::Instant::now(); |
| 6679 | let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await { |
| 6680 | Ok(r) => r, |
| 6681 | Err(e) => panic!("a 429 followed by a 200 should complete: {}", e), |
| 6682 | }; |
| 6683 | let elapsed = started.elapsed(); |
| 6684 | |
| 6685 | assert_eq!(connections(&seen), 2, "the stub was not asked a second time"); |
| 6686 | assert_eq!(resp.content, "Hello world"); |
| 6687 | assert_eq!(resp.retries, 1, "the retry was not counted for the user to see"); |
| 6688 | // The provider asked for a second and got one: its own figure beat the |
| 6689 | // client's 20ms backoff. |
| 6690 | assert!(elapsed >= std::time::Duration::from_millis(1_000), |
| 6691 | "Retry-After was ignored; waited only {:?}", elapsed); |
| 6692 | // The answer streamed once, and the retry announced itself. |
| 6693 | let text: String = tokens.iter().filter(|t| !t.starts_with("\n[daimond")).cloned().collect(); |
| 6694 | assert_eq!(text, "Hello world"); |
| 6695 | let notice = match tokens.iter().find(|t| t.contains("[daimond")) { |
| 6696 | Some(n) => n.clone(), |
| 6697 | None => panic!("a retry the user cannot see is its own defect: {:?}", tokens), |
| 6698 | }; |
| 6699 | assert!(notice.contains("HTTP 429"), "the notice does not say what happened: {}", notice); |
| 6700 | assert!(notice.contains("attempt 2 of 4"), "the notice does not say where we are: {}", notice); |
| 6701 | // The error's own rendering carries file, line and terminal colouring. |
| 6702 | assert!(!notice.contains('\u{1b}'), |
| 6703 | "ANSI escapes reached the user's message pane: {:?}", notice); |
| 6704 | // Provider-reported figures survive the retry. |
| 6705 | assert_eq!(resp.prompt_tokens, 11); |
| 6706 | assert_eq!(resp.cached_tokens, 9); |
| 6707 | assert_eq!(resp.cost_usd, 0.0003); |
| 6708 | } |
| 6709 | |
| 6710 | #[tokio::test] |
| 6711 | async fn test_a_400_is_never_retried() { |
| 6712 | let (port, seen) = start_stub(vec![Reply::bad_request()]).await; |
| 6713 | let client = stub_client(port); |
| 6714 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6715 | let mut tokens = Vec::new(); |
| 6716 | |
| 6717 | let result = client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await; |
| 6718 | assert!(result.is_err(), "a malformed request must not be reported as success"); |
| 6719 | // The whole point: a 400 will fail the same way next time, and retrying |
| 6720 | // it only costs the user money and time. |
| 6721 | assert_eq!(connections(&seen), 1, "a 400 was sent again"); |
| 6722 | assert!(tokens.is_empty(), "nothing should have streamed: {:?}", tokens); |
| 6723 | } |
| 6724 | |
| 6725 | #[tokio::test] |
| 6726 | async fn test_a_5xx_is_retried_until_it_clears() { |
| 6727 | let (port, seen) = start_stub(vec![ |
| 6728 | Reply::server_error(), |
| 6729 | Reply::server_error(), |
| 6730 | Reply::answer(), |
| 6731 | ]).await; |
| 6732 | let client = stub_client(port); |
| 6733 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6734 | let mut tokens = Vec::new(); |
| 6735 | |
| 6736 | let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await { |
| 6737 | Ok(r) => r, |
| 6738 | Err(e) => panic!("two 500s then a 200 should complete: {}", e), |
| 6739 | }; |
| 6740 | assert_eq!(connections(&seen), 3); |
| 6741 | assert_eq!(resp.content, "Hello world"); |
| 6742 | assert_eq!(resp.retries, 2); |
| 6743 | } |
| 6744 | |
| 6745 | #[tokio::test] |
| 6746 | async fn test_a_refusal_carries_the_providers_own_words() { |
| 6747 | // What the compactor reads to tell an oversized prompt from a malformed one. |
| 6748 | // Without the body it has only the status and a size estimate to go on, and a |
| 6749 | // provider that publishes no window can then kill a chat permanently. |
| 6750 | let over = "{\"error\":{\"message\":\"This model's maximum context length is \ |
| 6751 | 131072 tokens, however you requested 174233 tokens.\",\ |
| 6752 | \"code\":\"context_length_exceeded\"}}"; |
| 6753 | let (port, _seen) = start_stub(vec![Reply::Http { |
| 6754 | status: 400, reason: "Bad Request", headers: Vec::new(), body: over.to_string(), |
| 6755 | }]).await; |
| 6756 | let client = stub_client(port); |
| 6757 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6758 | |
| 6759 | let e = match client.chat_stream_tools(&msgs, None, &mut |_| {}).await { |
| 6760 | Ok(_) => panic!("a 400 must not be reported as success"), |
| 6761 | Err(e) => fmt!("{}", e), |
| 6762 | }; |
| 6763 | assert!(e.contains("maximum context length"), |
| 6764 | "the provider said why and the error does not: {}", e); |
| 6765 | assert!(e.contains("400"), "{}", e); |
| 6766 | // And that is enough on its own -- no size estimate needed. |
| 6767 | assert!(crate::agent::compact::looks_like_overflow(&e, 0, 100_000), |
| 6768 | "the words the provider used were not recognised: {}", e); |
| 6769 | } |
| 6770 | |
| 6771 | // ── A reply that hit the output limit ─────────────────────────────── |
| 6772 | |
| 6773 | #[test] |
| 6774 | fn test_both_dialects_say_when_a_reply_ran_out_of_room() { |
| 6775 | // Neither was read anywhere outside a test, so the browser had to infer |
| 6776 | // truncation from tool arguments that would not parse -- which cannot see a |
| 6777 | // plain text reply cut short, and cannot tell the model anything at all. |
| 6778 | assert!(openai_truncated( |
| 6779 | "{\"choices\":[{\"index\":0,\"delta\":{},\"finish_reason\":\"length\"}]}")); |
| 6780 | assert!(anthropic_truncated( |
| 6781 | "{\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"max_tokens\"}}")); |
| 6782 | // And a reply that simply finished is not truncated, in either dialect. |
| 6783 | assert!(!openai_truncated( |
| 6784 | "{\"choices\":[{\"delta\":{},\"finish_reason\":\"stop\"}]}")); |
| 6785 | assert!(!openai_truncated( |
| 6786 | "{\"choices\":[{\"delta\":{\"content\":\"hi\"},\"finish_reason\":null}]}")); |
| 6787 | assert!(!openai_truncated( |
| 6788 | "{\"choices\":[{\"delta\":{},\"finish_reason\":\"tool_calls\"}]}")); |
| 6789 | assert!(!anthropic_truncated( |
| 6790 | "{\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"end_turn\"}}")); |
| 6791 | assert!(!anthropic_truncated( |
| 6792 | "{\"type\":\"message_start\",\"message\":{\"stop_reason\":null}}")); |
| 6793 | } |
| 6794 | |
| 6795 | #[test] |
| 6796 | fn test_a_stream_cut_at_the_limit_says_so_on_the_response() { |
| 6797 | // Through the accumulator, which is where the app reads it: the flag is sticky, |
| 6798 | // because the usage chunk arrives AFTER the finish reason and must not unsay it. |
| 6799 | let mut acc = StreamAcc::default(); |
| 6800 | acc.ingest("{\"choices\":[{\"delta\":{\"content\":\"fn main\"}}]}", &mut |_| {}); |
| 6801 | assert!(!acc.into_response(false, 0).truncated); |
| 6802 | |
| 6803 | let mut acc = StreamAcc::default(); |
| 6804 | acc.ingest("{\"choices\":[{\"delta\":{\"content\":\"fn main\"}}]}", &mut |_| {}); |
| 6805 | acc.ingest("{\"choices\":[{\"delta\":{},\"finish_reason\":\"length\"}]}", &mut |_| {}); |
| 6806 | acc.ingest("{\"choices\":[],\"usage\":{\"prompt_tokens\":9,\"completion_tokens\":8192}}", |
| 6807 | &mut |_| {}); |
| 6808 | let r = acc.into_response(false, 0); |
| 6809 | assert!(r.truncated, "the usage chunk unsaid the finish reason"); |
| 6810 | assert_eq!(r.completion_tokens, 8192); |
| 6811 | } |
| 6812 | |
| 6813 | #[test] |
| 6814 | fn test_an_anthropic_stream_cut_at_the_limit_says_so_too() { |
| 6815 | let mut acc = AnthropicAcc::default(); |
| 6816 | acc.ingest("{\"type\":\"message_start\",\"message\":{\"usage\":{\"input_tokens\":9}}}", |
| 6817 | &mut |_| {}); |
| 6818 | acc.ingest("{\"type\":\"content_block_delta\",\"index\":0,\ |
| 6819 | \"delta\":{\"type\":\"text_delta\",\"text\":\"fn main\"}}", &mut |_| {}); |
| 6820 | assert!(!acc.truncated, "nothing has said the reply was cut"); |
| 6821 | acc.ingest("{\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"max_tokens\"},\ |
| 6822 | \"usage\":{\"output_tokens\":8192}}", &mut |_| {}); |
| 6823 | assert!(acc.into_response(false, 0).truncated); |
| 6824 | } |
| 6825 | |
| 6826 | #[tokio::test] |
| 6827 | async fn test_a_truncated_reply_is_not_an_error_and_is_not_retried() { |
| 6828 | // The interaction that matters. A reply that hit `max_tokens` is a complete |
| 6829 | // HTTP 200: sending it again costs money and produces the same cut, and treating |
| 6830 | // it as a failure would throw away text the user has already been shown. |
| 6831 | let (port, seen) = start_stub(vec![Reply::Sse { chunks: vec![ |
| 6832 | "data: {\"choices\":[{\"delta\":{\"content\":\"fn main() {\"}}]}\n\n".to_string(), |
| 6833 | "data: {\"choices\":[{\"delta\":{},\"finish_reason\":\"length\"}]}\n\n".to_string(), |
| 6834 | "data: [DONE]\n\n".to_string(), |
| 6835 | ], reset_after: None }]).await; |
| 6836 | let client = stub_client(port); |
| 6837 | let msgs = [ChatMessage::user("write the file".to_string())]; |
| 6838 | |
| 6839 | let r = match client.chat_stream_tools(&msgs, None, &mut |_| {}).await { |
| 6840 | Ok(r) => r, |
| 6841 | Err(e) => panic!("a reply that hit the cap is not a failed call: {}", e), |
| 6842 | }; |
| 6843 | assert!(r.truncated, "the cap was reached and the response does not say so"); |
| 6844 | assert_eq!(r.content, "fn main() {", "the partial answer was thrown away"); |
| 6845 | assert_eq!(connections(&seen), 1, "a complete 200 was sent again"); |
| 6846 | assert_eq!(r.retries, 0); |
| 6847 | } |
| 6848 | |
| 6849 | #[test] |
| 6850 | fn test_a_refusal_body_is_cut_without_splitting_a_character() { |
| 6851 | // A provider's body is arbitrary bytes on an error path, which is exactly where |
| 6852 | // a panic is least welcome and least likely to be found in testing. `&s[..300]` |
| 6853 | // on a multi-byte boundary is a panic, not a truncation. |
| 6854 | // One ASCII byte in front, so the two-byte characters after it sit on ODD |
| 6855 | // offsets and the cut at 300 lands in the middle of one. Without the offset the |
| 6856 | // boundaries happen to line up and a broken clip passes. |
| 6857 | let s = fmt!("a{}", "é".repeat(400)); |
| 6858 | assert!(!s.is_char_boundary(ERR_BODY_BYTES), "the fixture must actually straddle"); |
| 6859 | let cut = clip_bytes(&s, ERR_BODY_BYTES); |
| 6860 | assert!(cut.len() <= ERR_BODY_BYTES); |
| 6861 | assert!(cut.chars().skip(1).all(|c| c == 'é'), "a character was split"); |
| 6862 | // A short body is untouched, and an empty one is not a special case. |
| 6863 | assert_eq!(clip_bytes("short", ERR_BODY_BYTES), "short"); |
| 6864 | assert_eq!(clip_bytes("", ERR_BODY_BYTES), ""); |
| 6865 | } |
| 6866 | |
| 6867 | #[tokio::test] |
| 6868 | async fn test_retrying_stops_at_the_attempt_budget() { |
| 6869 | // A provider that is never ready: the attempt must end, not loop. |
| 6870 | let (port, seen) = start_stub(vec![Reply::too_many(None)]).await; |
| 6871 | let mut client = stub_client(port); |
| 6872 | client.retry.max_attempts = 3; |
| 6873 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6874 | |
| 6875 | let result = client.chat_stream_tools(&msgs, None, &mut |_| {}).await; |
| 6876 | assert!(result.is_err()); |
| 6877 | assert_eq!(connections(&seen), 3, |
| 6878 | "the attempt budget was not the bound on how many requests went out"); |
| 6879 | } |
| 6880 | |
| 6881 | #[tokio::test] |
| 6882 | async fn test_a_retry_after_beyond_the_wait_bound_ends_the_attempt() { |
| 6883 | // The provider asks for a minute; the user is watching a spinner. The |
| 6884 | // turn ends rather than honouring it. |
| 6885 | let (port, seen) = start_stub(vec![Reply::too_many(Some(60))]).await; |
| 6886 | let mut client = stub_client(port); |
| 6887 | client.retry.max_total_wait_ms = 2_000; |
| 6888 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6889 | |
| 6890 | let started = std::time::Instant::now(); |
| 6891 | let result = client.chat_stream_tools(&msgs, None, &mut |_| {}).await; |
| 6892 | assert!(result.is_err()); |
| 6893 | assert_eq!(connections(&seen), 1); |
| 6894 | assert!(started.elapsed() < std::time::Duration::from_secs(5), |
| 6895 | "the client slept through a Retry-After it had no budget for"); |
| 6896 | } |
| 6897 | |
| 6898 | #[tokio::test] |
| 6899 | async fn test_a_stream_that_breaks_after_tokens_is_not_replayed() { |
| 6900 | // THE streaming hazard. The provider streams one delta and then |
| 6901 | // vanishes; a retry here would hand the caller "Hello" a second time. |
| 6902 | let (port, seen) = start_stub(vec![ |
| 6903 | Reply::Sse { |
| 6904 | chunks: vec![ |
| 6905 | "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\n".to_string(), |
| 6906 | "data: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\n".to_string(), |
| 6907 | ], |
| 6908 | reset_after: Some(1), |
| 6909 | }, |
| 6910 | Reply::answer(), |
| 6911 | ]).await; |
| 6912 | let client = stub_client(port); |
| 6913 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6914 | let mut tokens = Vec::new(); |
| 6915 | |
| 6916 | let _ = client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await; |
| 6917 | |
| 6918 | let text: String = tokens.iter().filter(|t| !t.starts_with("\n[daimond")).cloned().collect(); |
| 6919 | assert_eq!(text, "Hello", |
| 6920 | "the partial was replayed or lost -- got {:?}", tokens); |
| 6921 | assert_eq!(connections(&seen), 1, |
| 6922 | "the turn was restarted after tokens had already reached the caller"); |
| 6923 | } |
| 6924 | |
| 6925 | #[tokio::test] |
| 6926 | async fn test_a_stream_that_breaks_after_a_tool_call_fragment_is_not_replayed() { |
| 6927 | // No text has streamed, so `emitted` is false -- but a half-built tool |
| 6928 | // call is output all the same, and starting over would either duplicate |
| 6929 | // the call or splice two halves of different ones together. |
| 6930 | let (port, seen) = start_stub(vec![ |
| 6931 | Reply::Sse { |
| 6932 | chunks: vec![ |
| 6933 | "data: {\"choices\":[{\"delta\":{\"tool_calls\":[{\"index\":0,\"id\":\"c0\",\ |
| 6934 | \"function\":{\"name\":\"file_read\",\"arguments\":\"{\\\"path\\\":\\\"\"}}]}}]}\n\n" |
| 6935 | .to_string(), |
| 6936 | "data: {\"choices\":[{\"delta\":{\"tool_calls\":[{\"index\":0,\ |
| 6937 | \"function\":{\"arguments\":\"a.txt\\\"}\"}}]}}]}\n\n".to_string(), |
| 6938 | ], |
| 6939 | reset_after: Some(1), |
| 6940 | }, |
| 6941 | Reply::answer(), |
| 6942 | ]).await; |
| 6943 | let client = stub_client(port); |
| 6944 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6945 | let mut tokens = Vec::new(); |
| 6946 | |
| 6947 | let _ = client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await; |
| 6948 | |
| 6949 | assert_eq!(connections(&seen), 1, |
| 6950 | "the turn was restarted on top of a partial tool call"); |
| 6951 | assert!(tokens.iter().all(|t| t.starts_with("\n[daimond")), |
| 6952 | "text streamed from a replayed turn: {:?}", tokens); |
| 6953 | } |
| 6954 | |
| 6955 | #[tokio::test] |
| 6956 | async fn test_a_stream_that_breaks_before_any_token_is_retried() { |
| 6957 | // A network drop before the provider has streamed anything — the laptop |
| 6958 | // moving between locations, the wifi handing off — is the failure the |
| 6959 | // widened retry policy is for. The stream resets on chunk 0, nothing has |
| 6960 | // been emitted, and the turn should start over and complete. This is the |
| 6961 | // exact shape of "the stream broke" the user sees on the road. |
| 6962 | let (port, seen) = start_stub(vec![ |
| 6963 | Reply::Sse { |
| 6964 | chunks: vec![ |
| 6965 | "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\n".to_string(), |
| 6966 | "data: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\n".to_string(), |
| 6967 | "data: [DONE]\n\n".to_string(), |
| 6968 | ], |
| 6969 | reset_after: Some(0), |
| 6970 | }, |
| 6971 | Reply::answer(), |
| 6972 | ]).await; |
| 6973 | let client = stub_client(port); |
| 6974 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 6975 | let mut tokens = Vec::new(); |
| 6976 | |
| 6977 | let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await { |
| 6978 | Ok(r) => r, |
| 6979 | Err(e) => panic!("a stream that breaks before tokens should recover: {}", e), |
| 6980 | }; |
| 6981 | |
| 6982 | assert_eq!(connections(&seen), 2, |
| 6983 | "the broken stream was not retried"); |
| 6984 | assert_eq!(resp.content, "Hello world", |
| 6985 | "the retry did not produce the answer"); |
| 6986 | assert_eq!(resp.retries, 1, |
| 6987 | "the retry was not counted"); |
| 6988 | let text: String = tokens.iter().filter(|t| !t.starts_with("\n[daimond")).cloned().collect(); |
| 6989 | assert_eq!(text, "Hello world", |
| 6990 | "the answer was not streamed cleanly after the retry: {:?}", tokens); |
| 6991 | } |
| 6992 | |
| 6993 | #[tokio::test] |
| 6994 | async fn test_the_breakpoint_reaches_the_wire() { |
| 6995 | // What the provider actually receives, read back off its own socket. |
| 6996 | let (port, seen) = start_stub(vec![Reply::answer()]).await; |
| 6997 | let client = stub_client(port); |
| 6998 | let msgs = [ |
| 6999 | ChatMessage::system(long_system()), |
| 7000 | ChatMessage::user("hello".to_string()), |
| 7001 | ]; |
| 7002 | let _ = client.chat_stream_tools(&msgs, None, &mut |_| {}).await; |
| 7003 | |
| 7004 | let body = match seen.lock() { |
| 7005 | Ok(g) => g.bodies[0].clone(), |
| 7006 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 7007 | }; |
| 7008 | assert!(body.contains("\"cache_control\":{\"type\":\"ephemeral\"}"), |
| 7009 | "no breakpoint reached the provider: {}", body); |
| 7010 | assert!(body.contains("\"role\":\"system\",\"content\":[{\"type\":\"text\"")); |
| 7011 | } |
| 7012 | |
| 7013 | /// A client pointed at the stub, speaking the Messages API. |
| 7014 | fn anth_stub_client(port: u16) -> LlmClient { |
| 7015 | let mut client = test_client_at("localhost", port, "/v1/messages", "claude-opus-5"); |
| 7016 | client.retry = RetryPolicy { |
| 7017 | max_attempts: 4, |
| 7018 | base_ms: 20, |
| 7019 | max_backoff_ms: 40, |
| 7020 | max_total_wait_ms: 5_000, |
| 7021 | }; |
| 7022 | client |
| 7023 | } |
| 7024 | |
| 7025 | #[tokio::test] |
| 7026 | async fn test_the_messages_api_request_reaches_the_wire_in_its_own_shape() { |
| 7027 | // What the provider actually receives, read back off its own socket -- |
| 7028 | // not what this file believes it sent. |
| 7029 | let (port, seen) = start_stub(vec![Reply::anth_answer()]).await; |
| 7030 | let client = anth_stub_client(port); |
| 7031 | let tools = r#"[{"type":"function","function":{"name":"file_read", |
| 7032 | "description":"Read a file","parameters":{"type":"object","properties":{}}}}]"#; |
| 7033 | let msgs = [ |
| 7034 | ChatMessage::system(long_system()), |
| 7035 | ChatMessage::user("hello".to_string()), |
| 7036 | ]; |
| 7037 | let resp = match client.chat_stream_tools(&msgs, Some(tools), &mut |_| {}).await { |
| 7038 | Ok(r) => r, |
| 7039 | Err(e) => panic!("the Messages API turn failed: {}", e), |
| 7040 | }; |
| 7041 | assert_eq!(resp.content, "It says hello."); |
| 7042 | assert_eq!(resp.prompt_tokens, 60); |
| 7043 | assert_eq!(resp.completion_tokens, 8); |
| 7044 | |
| 7045 | let body = match seen.lock() { |
| 7046 | Ok(g) => g.bodies[0].clone(), |
| 7047 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 7048 | }; |
| 7049 | assert!(body.contains("\"system\":[{\"type\":\"text\""), |
| 7050 | "the system prompt did not reach the wire hoisted: {}", body); |
| 7051 | assert!(!body.contains("\"role\":\"system\""), "{}", body); |
| 7052 | assert!(body.contains("\"input_schema\""), |
| 7053 | "the tools reached the wire in the OpenAI shape: {}", body); |
| 7054 | assert!(body.contains("\"thinking\":{\"type\":\"adaptive\""), "{}", body); |
| 7055 | assert!(body.contains("\"cache_control\":{\"type\":\"ephemeral\"}"), |
| 7056 | "no breakpoint reached the provider: {}", body); |
| 7057 | assert!(!body.contains("\"stream_options\""), |
| 7058 | "an OpenAI-only field reached the Messages API: {}", body); |
| 7059 | } |
| 7060 | |
| 7061 | #[tokio::test] |
| 7062 | async fn test_thinking_blocks_are_handed_back_with_the_tool_results() { |
| 7063 | // The constraint that produces an error on EVERY following turn when it |
| 7064 | // is missed: within a tool-use turn, the signed thinking blocks must go |
| 7065 | // back complete and unmodified, ahead of the tool_use block they |
| 7066 | // accompanied. Two real requests, and the second one is read off the |
| 7067 | // provider's socket. |
| 7068 | let (port, seen) = start_stub(vec![ |
| 7069 | Reply::anth_thinks_then_calls(), |
| 7070 | Reply::anth_answer(), |
| 7071 | ]).await; |
| 7072 | let client = anth_stub_client(port); |
| 7073 | let tools = r#"[{"type":"function","function":{"name":"file_read", |
| 7074 | "description":"Read a file","parameters":{"type":"object","properties":{}}}}]"#; |
| 7075 | |
| 7076 | // Round one: the model thinks, then asks for a tool. |
| 7077 | let first = match client.chat_stream_tools( |
| 7078 | &[ChatMessage::user("read a.txt".to_string())], |
| 7079 | Some(tools), &mut |_| {}).await |
| 7080 | { |
| 7081 | Ok(r) => r, |
| 7082 | Err(e) => panic!("round one failed: {}", e), |
| 7083 | }; |
| 7084 | assert_eq!(first.tool_calls.len(), 1); |
| 7085 | assert_eq!(first.tool_calls[0].id, "toolu_1"); |
| 7086 | assert_eq!(first.thinking, "I should read the file."); |
| 7087 | assert_eq!(first.cached_tokens, 900); |
| 7088 | assert_eq!(first.prompt_tokens, 930, "the cached prefix is part of the prompt"); |
| 7089 | |
| 7090 | // Round two: the agent loop's shape -- the assistant turn that asked, |
| 7091 | // then the result. |
| 7092 | let round_two = vec![ |
| 7093 | ChatMessage::user("read a.txt".to_string()), |
| 7094 | ChatMessage::Assistant { |
| 7095 | content: MessageContent::text(""), |
| 7096 | tool_calls: first.tool_calls.clone(), |
| 7097 | }, |
| 7098 | ChatMessage::tool("toolu_1".to_string(), "hello".to_string()), |
| 7099 | ]; |
| 7100 | if let Err(e) = client.chat_stream_tools(&round_two, Some(tools), &mut |_| {}).await { |
| 7101 | panic!("round two failed: {}", e); |
| 7102 | } |
| 7103 | |
| 7104 | let body = match seen.lock() { |
| 7105 | Ok(g) => g.bodies[1].clone(), |
| 7106 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 7107 | }; |
| 7108 | assert!(body.contains("\"signature\":\"SIGNATURE-1\""), |
| 7109 | "the signed thinking block never went back: {}", body); |
| 7110 | assert!(body.contains("I should read the file."), |
| 7111 | "the thinking text was dropped, which the API reads as a modified block: {}", body); |
| 7112 | // Order matters: the reasoning precedes the call it produced. |
| 7113 | let think_at = match body.find("\"type\":\"thinking\"") { |
| 7114 | Some(p) => p, |
| 7115 | None => panic!("no thinking block in the assistant turn: {}", body), |
| 7116 | }; |
| 7117 | let call_at = match body.find("\"type\":\"tool_use\"") { |
| 7118 | Some(p) => p, |
| 7119 | None => panic!("no tool_use block: {}", body), |
| 7120 | }; |
| 7121 | assert!(think_at < call_at, |
| 7122 | "the thinking block was placed after the call it led to: {}", body); |
| 7123 | } |
| 7124 | |
| 7125 | #[tokio::test] |
| 7126 | async fn test_reasoning_only_ever_goes_back_with_the_call_that_produced_it() { |
| 7127 | // Blocks are kept across a whole tool loop -- that is the documented |
| 7128 | // recommendation, and on the models that keep them it is what makes the |
| 7129 | // round-by-round cache hits happen. What must NOT happen is one turn's |
| 7130 | // reasoning being glued to a different turn's call: within an assistant |
| 7131 | // message the run has to match what the model generated there, so a |
| 7132 | // block from elsewhere is a rearrangement and a 400. |
| 7133 | let (port, seen) = start_stub(vec![ |
| 7134 | Reply::anth_thinks_then_calls(), |
| 7135 | Reply::anth_answer(), |
| 7136 | ]).await; |
| 7137 | let client = anth_stub_client(port); |
| 7138 | let msgs = [ChatMessage::user("hello".to_string())]; |
| 7139 | let _ = client.chat_stream_tools(&msgs, None, &mut |_| {}).await; |
| 7140 | |
| 7141 | // A later turn quoting a DIFFERENT call: the held reasoning belongs to |
| 7142 | // `toolu_1`, and nothing may hand it to `toolu_other`. |
| 7143 | let elsewhere = vec![ |
| 7144 | ChatMessage::user("hello".to_string()), |
| 7145 | ChatMessage::Assistant { |
| 7146 | content: MessageContent::text(""), |
| 7147 | tool_calls: vec![ToolCall { id: "toolu_other".to_string(), |
| 7148 | name: "file_read".to_string(), arguments: "{}".to_string() }], |
| 7149 | }, |
| 7150 | ChatMessage::Tool { tool_call_id: "toolu_other".to_string(), |
| 7151 | content: MessageContent::text("x") }, |
| 7152 | ]; |
| 7153 | let _ = client.chat_stream_tools(&elsewhere, None, &mut |_| {}).await; |
| 7154 | let body = match seen.lock() { |
| 7155 | Ok(g) => g.bodies[1].clone(), |
| 7156 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 7157 | }; |
| 7158 | assert!(!body.contains("SIGNATURE-1"), |
| 7159 | "one turn's reasoning was handed to another turn's call: {}", body); |
| 7160 | assert!(!body.contains("\"type\":\"thinking\""), "{}", body); |
| 7161 | } |
| 7162 | |
| 7163 | #[tokio::test] |
| 7164 | async fn test_every_round_of_a_tool_loop_keeps_its_own_reasoning() { |
| 7165 | // Round three carries BOTH earlier rounds' blocks, each beside its own |
| 7166 | // call. A client that held only the latest would drop the first |
| 7167 | // round's reasoning from a loop that is, to the model, one turn -- and |
| 7168 | // with it the cache hit that the tool results were supposed to earn. |
| 7169 | let (port, seen) = start_stub(vec![ |
| 7170 | Reply::anth_thinks_then_calls(), |
| 7171 | Reply::anth_thinks_then_calls_again(), |
| 7172 | Reply::anth_answer(), |
| 7173 | ]).await; |
| 7174 | let client = anth_stub_client(port); |
| 7175 | let call = |id: &str| ToolCall { |
| 7176 | id: id.to_string(), name: "file_read".to_string(), arguments: "{}".to_string() }; |
| 7177 | |
| 7178 | let mut working = vec![ChatMessage::user("read them".to_string())]; |
| 7179 | for _ in 0..2 { |
| 7180 | let r = match client.chat_stream_tools(&working, None, &mut |_| {}).await { |
| 7181 | Ok(r) => r, |
| 7182 | Err(e) => panic!("round failed: {}", e), |
| 7183 | }; |
| 7184 | let id = r.tool_calls[0].id.clone(); |
| 7185 | working.push(ChatMessage::Assistant { |
| 7186 | content: MessageContent::text(""), tool_calls: vec![call(&id)] }); |
| 7187 | working.push(ChatMessage::tool(id, "ok".to_string())); |
| 7188 | } |
| 7189 | let _ = client.chat_stream_tools(&working, None, &mut |_| {}).await; |
| 7190 | |
| 7191 | let body = match seen.lock() { |
| 7192 | Ok(g) => g.bodies[2].clone(), |
| 7193 | Err(e) => panic!("stub bookkeeping poisoned: {}", e), |
| 7194 | }; |
| 7195 | assert!(body.contains("SIGNATURE-1"), |
| 7196 | "the first round's reasoning was dropped from the loop: {}", body); |
| 7197 | assert!(body.contains("SIGNATURE-2"), |
| 7198 | "the second round's reasoning was dropped: {}", body); |
| 7199 | assert_eq!(body.matches("\"type\":\"thinking\"").count(), 2, "{}", body); |
| 7200 | } |
| 7201 | |
| 7202 | // Test verifier that accepts any certificate (for unit tests only). |
| 7203 | use tokio_rustls::rustls::client::danger::{HandshakeSignatureValid, ServerCertVerified, ServerCertVerifier}; |
| 7204 | use std::sync::Arc; |
| 7205 | |
| 7206 | #[derive(Debug)] |
| 7207 | pub struct NoVerify; |
| 7208 | |
| 7209 | impl ServerCertVerifier for NoVerify { |
| 7210 | fn verify_server_cert( |
| 7211 | &self, |
| 7212 | _end_entity: &tokio_rustls::rustls::pki_types::CertificateDer<'_>, |
| 7213 | _intermediates: &[tokio_rustls::rustls::pki_types::CertificateDer<'_>], |
| 7214 | _server_name: &tokio_rustls::rustls::pki_types::ServerName<'_>, |
| 7215 | _ocsp_response: &[u8], |
| 7216 | _now: tokio_rustls::rustls::pki_types::UnixTime, |
| 7217 | ) -> Result<ServerCertVerified, tokio_rustls::rustls::Error> { |
| 7218 | Ok(ServerCertVerified::assertion()) |
| 7219 | } |
| 7220 | fn verify_tls12_signature( |
| 7221 | &self, |
| 7222 | _message: &[u8], |
| 7223 | _cert: &tokio_rustls::rustls::pki_types::CertificateDer<'_>, |
| 7224 | _dss: &tokio_rustls::rustls::DigitallySignedStruct, |
| 7225 | ) -> Result<HandshakeSignatureValid, tokio_rustls::rustls::Error> { |
| 7226 | Ok(HandshakeSignatureValid::assertion()) |
| 7227 | } |
| 7228 | fn verify_tls13_signature( |
| 7229 | &self, |
| 7230 | _message: &[u8], |
| 7231 | _cert: &tokio_rustls::rustls::pki_types::CertificateDer<'_>, |
| 7232 | _dss: &tokio_rustls::rustls::DigitallySignedStruct, |
| 7233 | ) -> Result<HandshakeSignatureValid, tokio_rustls::rustls::Error> { |
| 7234 | Ok(HandshakeSignatureValid::assertion()) |
| 7235 | } |
| 7236 | fn supported_verify_schemes(&self) -> Vec<tokio_rustls::rustls::SignatureScheme> { |
| 7237 | vec![ |
| 7238 | tokio_rustls::rustls::SignatureScheme::RSA_PKCS1_SHA256, |
| 7239 | tokio_rustls::rustls::SignatureScheme::ECDSA_NISTP256_SHA256, |
| 7240 | tokio_rustls::rustls::SignatureScheme::ED25519, |
| 7241 | ] |
| 7242 | } |
| 7243 | } |
| 7244 | } |