Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/src/llm.rs

342 KiB, 1 run

created by r2519314175:953, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! LLM client — chat completions with SSE streaming, in two wire dialects.
2//!
3//! Uses `fe2o3_net` for the underlying TLS connection. Parses the
4//! `text/event-stream` response line-by-line, extracting `data:` lines
5//! containing JSON objects with `delta` content.
6//!
7//! No `serde` or `reqwest` — neither API's JSON is complicated enough to
8//! need one, and both are parsed by string scanning. This keeps the
9//! dependency surface minimal and stays within the fe2o3 ecosystem.
10//!
11//! Two dialects share every public entry point, the retry policy and the
12//! prompt-cache placement: the OpenAI-compatible `/chat/completions` that
13//! every router speaks, and Anthropic's own `/v1/messages`. See
14//! [`Dialect`] for why the second one could not simply be bent into the
15//! first.
16
17use oxedyne_fe2o3_core::prelude::*;
18use oxedyne_fe2o3_core::rand::Rand;
19use oxedyne_fe2o3_jdat::prelude::*;
20
21use crate::protocol::{ChatMessage, ContentPart, Dropped, ImagePart, MessageContent, ToolCall};
22
23// Native transport imports — the hand-rolled TLS client lives behind
24// tokio + rustls, which do not target wasm32.
25#[cfg(not(target_arch = "wasm32"))]
26use std::sync::Arc;
27#[cfg(not(target_arch = "wasm32"))]
28use tokio::io::{AsyncReadExt, AsyncWriteExt};
29#[cfg(not(target_arch = "wasm32"))]
30use tokio_rustls::rustls::ClientConfig;
31
32
33// ┌───────────────────────────────────────────────────────────────┐
34// │ Dialect │
35// └───────────────────────────────────────────────────────────────┘
36
37/// Which wire protocol an endpoint speaks.
38///
39/// The OpenAI-compatible shape carried every provider Daimond had, so the
40/// client was written as if there were only one. Anthropic's own Messages
41/// API is not that shape and cannot be made into it: the system prompt is a
42/// top-level field rather than a message, content is an array of typed
43/// blocks rather than a string, a tool call is a `tool_use` block and its
44/// result a `tool_result` block inside the *user* turn, the streamed events
45/// are named rather than deltas of one object, and the usage counts have
46/// different names and a different meaning. Bending one into the other
47/// would have meant a translation layer that silently dropped whatever it
48/// did not understand -- thinking blocks above all -- so the seam is
49/// explicit instead, and every branch that needs it says which side it is on.
50///
51/// The dialect is a property of the *endpoint*, not of the model: the same
52/// Claude model is reachable through a router's `/chat/completions` (where
53/// it speaks OpenAI) and through Anthropic's `/v1/messages` (where it does
54/// not). Prompt caching gates on the model id for exactly the same reason
55/// in reverse -- see [`model_caches_on_request`].
56#[derive(Clone, Copy, Debug, Eq, PartialEq)]
57pub enum Dialect {
58 /// OpenAI-compatible chat completions.
59 OpenAi,
60 /// Anthropic's Messages API.
61 Anthropic,
62}
63
64impl Dialect {
65
66 /// Which dialect the endpoint at `host``path` speaks.
67 ///
68 /// Two independent signals, either of which is conclusive: Anthropic's
69 /// own host, and the `/v1/messages` path that only the Messages API
70 /// serves. Everything else is OpenAI-compatible, which is the right
71 /// default -- a router serving `anthropic/claude-opus-5` is still
72 /// speaking OpenAI.
73 ///
74 /// # Arguments
75 /// * `host` - The request host, without scheme or port.
76 /// * `path` - The request path.
77 pub fn for_endpoint(host: &str, path: &str) -> Self {
78 let h = host.to_ascii_lowercase();
79 let p = path.trim_end_matches('/').to_ascii_lowercase();
80 if h == "api.anthropic.com" || h.ends_with(".anthropic.com") || p.ends_with("/v1/messages") {
81 Self::Anthropic
82 } else {
83 Self::OpenAi
84 }
85 }
86}
87
88
89// ┌───────────────────────────────────────────────────────────────┐
90// │ Thinking carry │
91// └───────────────────────────────────────────────────────────────┘
92
93/// How many assistant turns of signed reasoning to hold at once.
94///
95/// A round of an agentic loop adds one entry, so this bounds the memory a very
96/// long loop can hold while covering more rounds than any single tool loop runs.
97const CARRY_MAX_TURNS: usize = 32;
98
99/// The signed thinking blocks of recent assistant turns, held until their tool
100/// results come back.
101///
102/// Anthropic requires that a thinking-enabled assistant turn which asked for
103/// tools be handed back *complete and unmodified* alongside the tool results:
104/// "within a tool-use turn, pass thinking blocks back". A block the caller
105/// edited is rejected with a 400; a block the caller dropped makes the API
106/// silently disable thinking for the request, which is the same defect wearing
107/// a quieter coat. Passing every turn's blocks back is the documented
108/// recommendation beyond that: on the models that keep them, the reasoning
109/// stays in context and caches incrementally with the tool results, so dropping
110/// it costs both continuity and money on every round after the first.
111///
112/// The conversation type this client is given ([`ChatMessage`]) has nowhere to
113/// put a thinking block -- it is OpenAI-shaped, and OpenAI has no such thing --
114/// so the blocks are held here instead, keyed by the tool-call id they were
115/// generated beside. That id is what makes the association safe: the very next
116/// request carries the same id in its assistant turn, so a turn's reasoning can
117/// only ever be handed back with the call it actually produced. A turn that
118/// asked for no tools stores nothing, because it is already over.
119#[derive(Clone, Debug, Default)]
120struct ThinkCarry {
121 /// `(first tool-call id, blocks)`, oldest first. The blocks are already
122 /// serialised as JSON objects, in the order the model produced them.
123 turns: Vec<(String, Vec<String>)>,
124}
125
126/// The `say` calls whose fold the user currently has OPEN, by tool-call id.
127///
128/// **THE FOLD IS THE CONTEXT CONTROL, and this is what makes that true.** A folded detail is
129/// stripped from the payload, which is right when the user has closed it: they are done with it,
130/// and re-sending it on every later turn buys nothing. But a fold they have OPENED is a fold they
131/// are reading, and the next thing they say is likely to be about it -- so the model should be
132/// holding what the user is looking at.
133///
134/// The user's own gesture therefore decides the model's working set, with no second control to
135/// learn and no decision to make twice. What is on their screen and what is in its context are
136/// the same set, which is the only arrangement where "why does it not remember that?" has an
137/// answer they can see.
138///
139/// It is rebuilt from the page before every request rather than accumulated here, because a fold
140/// can be opened and closed between two turns and the payload has to follow.
141///
142/// **It costs a cache miss on the turn it changes.** Opening a fold rewrites a message that was
143/// already in the prefix, so everything from that point is re-read once. Stable again afterwards.
144type OpenFolds = std::rc::Rc<std::cell::RefCell<OpenSet>>;
145
146/// The call ids of the `say` folds the user has open, as [`LlmClient::open_folds`] hands them
147/// over and as the sizing path in [`crate::agent::compact`] reads them.
148pub type OpenSet = std::collections::HashSet<String>;
149
150/// A [`ThinkCarry`] shared across clones of a client.
151#[cfg(not(target_arch = "wasm32"))]
152type Carry = std::sync::Arc<std::sync::Mutex<ThinkCarry>>;
153
154/// A [`ThinkCarry`] shared across clones of a client.
155#[cfg(target_arch = "wasm32")]
156type Carry = std::rc::Rc<std::cell::RefCell<ThinkCarry>>;
157
158/// Whether an endpoint has been caught refusing pictures, shared across clones of a client.
159///
160/// Learned rather than declared. [`model_can_see`] is a list of eight model ids known to be
161/// blind, so every model it has not heard of is assumed sighted -- which is the right default
162/// (a new sighted model works at once) and is wrong for exactly as long as it takes one turn
163/// to fail. This is the other half: once a request carrying pictures comes back refused and the
164/// same request without them succeeds, the endpoint is marked and no later turn pays for the
165/// discovery twice.
166#[cfg(not(target_arch = "wasm32"))]
167type Blind = std::sync::Arc<std::sync::atomic::AtomicBool>;
168
169/// Whether an endpoint has been caught refusing pictures, shared across clones of a client.
170#[cfg(target_arch = "wasm32")]
171type Blind = std::rc::Rc<std::cell::Cell<bool>>;
172
173/// A fresh flag, unset: nothing has been refused yet.
174fn new_blind() -> Blind {
175 #[cfg(not(target_arch = "wasm32"))]
176 { std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)) }
177 #[cfg(target_arch = "wasm32")]
178 { std::rc::Rc::new(std::cell::Cell::new(false)) }
179}
180
181/// A fresh, empty carry.
182fn new_carry() -> Carry {
183 #[cfg(not(target_arch = "wasm32"))]
184 { std::sync::Arc::new(std::sync::Mutex::new(ThinkCarry::default())) }
185 #[cfg(target_arch = "wasm32")]
186 { std::rc::Rc::new(std::cell::RefCell::new(ThinkCarry::default())) }
187}
188
189
190/// One piece of a streamed turn, labelled with what kind of thing it is.
191///
192/// The two are different KINDS of content and not two shades of one. Text is the
193/// answer: it is accumulated, persisted, and sent back to the model next turn as
194/// what the assistant said. Reasoning is the model's working, which the user pays
195/// for and which decides the answer, but which is not the answer and must never be
196/// stored as one -- see `AgentEvent::Thinking` in src/protocol.rs.
197///
198/// One sink and not two, because a caller holds ONE `&mut` to whatever it is
199/// forwarding into. Two closures over the same event sink is a borrow the compiler
200/// refuses, and the ways round it (a `RefCell`, a channel) buy nothing: the stream
201/// is serial, so exactly one delta is in flight at a time.
202#[derive(Clone, Copy, Debug, Eq, PartialEq)]
203pub enum Delta<'a> {
204 Text(&'a str),
205 Reasoning(&'a str),
206}
207
208// ┌───────────────────────────────────────────────────────────────┐
209// │ LlmClient │
210// └───────────────────────────────────────────────────────────────┘
211
212/// Async client for a chat completions API, in either [`Dialect`].
213///
214/// Connects via TLS to the configured host, POSTs a chat completion
215/// request with `stream: true`, and parses the SSE response
216/// incrementally — calling `on_token` for each chunk as it arrives,
217/// saying whether it is the answer or the model's working ([`Delta`]).
218#[derive(Clone, Debug)]
219pub struct LlmClient {
220 pub host: String,
221 pub port: u16,
222 pub path: String,
223 pub api_key: String,
224 pub model: String,
225 /// Upper bound on generated tokens per turn. Prevents runaway
226 /// reasoning loops (e.g. GLM-5.2 without a cap).
227 pub max_tokens: u32,
228 /// Which wire protocol the endpoint speaks, derived from the host and
229 /// path at construction. See [`Dialect`].
230 pub dialect: Dialect,
231 /// How transient provider failures are retried. Shared by both transports.
232 pub retry: RetryPolicy,
233 /// The signed thinking blocks of the assistant turn now awaiting tool
234 /// results, so they can be handed back on the next request. Shared
235 /// across clones, because a sub-agent built from a cloned client is
236 /// continuing the same turn. See [`ThinkCarry`].
237 think: Carry,
238 /// The `say` folds the user has open. See [`OpenFolds`].
239 open_folds: OpenFolds,
240 /// Set once this endpoint has been caught refusing a request that carried pictures.
241 /// See [`Blind`]; read by [`LlmClient::vision_guard`] and set by the strip-and-retry in
242 /// [`LlmClient::stream_turn`] and [`LlmClient::chat_once`].
243 blind: Blind,
244 /// Root-trust TLS configuration for the native transport. The wasm
245 /// transport delegates trust to the browser's `fetch`, so this field
246 /// is native-only.
247 #[cfg(not(target_arch = "wasm32"))]
248 pub tls_config: Arc<ClientConfig>,
249 /// Wasm transport URL scheme selector: `true` builds `https://…`,
250 /// `false` builds `http://…`. Defaults to `https` (all real
251 /// providers are TLS-only); an `http` client targets a local mock
252 /// over `127.0.0.1` for headless testing, where the browser still
253 /// treats the origin as a secure context.
254 #[cfg(target_arch = "wasm32")]
255 pub secure: bool,
256 /// Shared abort slot for the browser transport. Each `fetch` installs
257 /// a fresh [`web_sys::AbortController`] here and wires its signal into
258 /// the request; [`abort`](Self::abort) fires it to cancel the in-flight
259 /// turn. An `Rc<RefCell<…>>` (never `unsafe`), shared across clones so
260 /// a sub-agent built from a cloned client aborts on the same signal.
261 #[cfg(target_arch = "wasm32")]
262 abort: std::rc::Rc<std::cell::RefCell<Option<web_sys::AbortController>>>,
263}
264
265/// The `usage` block a provider reports for a call.
266///
267/// The token counts were always read; the other two are what the provider
268/// says about its own billing, and are worth strictly more than any estimate
269/// made from them. A router charges its own negotiated rate, and a prompt
270/// cache read is a fraction of a fresh one -- neither is visible in a token
271/// count, so pricing from tokens alone overstated spend several-fold.
272#[derive(Clone, Copy, Debug, Default)]
273pub struct Usage {
274 pub prompt: u64,
275 pub completion: u64,
276 /// Prompt tokens served from the provider's cache, a subset of `prompt`.
277 pub cached: u64,
278 /// What the provider says the call actually cost, in USD. Zero means it
279 /// said nothing, never that the call was free.
280 pub cost_usd: f64,
281}
282
283/// The response from a completed streaming chat call.
284#[derive(Clone, Debug, Default)]
285pub struct ChatResponse {
286 pub content: String,
287 pub prompt_tokens: u64,
288 pub completion_tokens: u64,
289 /// Prompt tokens the provider served from its cache.
290 pub cached_tokens: u64,
291 /// What the provider says this call cost, in USD; `0.0` when it did not
292 /// say. An aborted stream may never deliver the usage chunk at all.
293 pub cost_usd: f64,
294 /// Set when the turn was cancelled mid-stream (browser abort). The
295 /// `content` then holds whatever streamed before the cancellation, so
296 /// the caller keeps the partial answer rather than reporting an error.
297 pub aborted: bool,
298 /// How many times this call was retried before it succeeded; see
299 /// [`ChatOnceResponse::retries`].
300 pub retries: u32,
301 /// The model's own reasoning; see [`ChatOnceResponse::thinking`].
302 pub thinking: String,
303 /// Set when the provider stopped because the reply hit `max_tokens` --
304 /// `finish_reason: "length"`, or Anthropic's `stop_reason: "max_tokens"`.
305 ///
306 /// A tool call cut here arrives as MALFORMED JSON, so the caller needs to
307 /// tell "the model wrote bad JSON" from "the reply ran out of room": the
308 /// first is the model's mistake, the second is a setting, and only one of
309 /// them is worth retrying.
310 ///
311 /// It is NOT an error and is never retried. A reply that hit the cap is a
312 /// complete HTTP 200, and sending the same request again costs money and
313 /// produces the same cut.
314 pub truncated: bool,
315}
316
317/// The response from a chat call that may include tool calls the model
318/// wants executed. Whether it was produced by a streaming or a
319/// non-streaming request, the accumulated shape is the same.
320#[derive(Clone, Debug, Default)]
321pub struct ChatOnceResponse {
322 pub content: String,
323 pub tool_calls: Vec<ToolCall>,
324 pub prompt_tokens: u64,
325 pub completion_tokens: u64,
326 /// Prompt tokens the provider served from its cache.
327 pub cached_tokens: u64,
328 /// What the provider says this call cost, in USD; see
329 /// [`ChatResponse::cost_usd`].
330 pub cost_usd: f64,
331 /// Set when the turn was cancelled mid-stream (browser abort); see
332 /// [`ChatResponse::aborted`].
333 pub aborted: bool,
334 /// How many times this call was retried before it succeeded. Zero is the
335 /// ordinary case; anything else is time the user waited for a provider that
336 /// was not ready, and is worth showing rather than hiding.
337 pub retries: u32,
338 /// The model's whole reasoning for this turn, gathered as it streamed. Anthropic
339 /// direct returns it from `thinking_delta`; an OpenAI-dialect endpoint returns it
340 /// on `reasoning` (OpenRouter's spelling) or `reasoning_content` (DeepSeek's own).
341 /// Empty for a model that does not reason, which is most of them.
342 ///
343 /// NEVER part of `content`: reasoning is not the answer, and a caller that
344 /// persisted it as one would be putting the model's working out where its reply
345 /// should be -- and handing it back next turn as something the model said. The
346 /// tokens are already counted in `completion_tokens`, because thinking is billed
347 /// as output whether or not its text comes back.
348 ///
349 /// A streaming caller does not need this: the same words reached it as
350 /// [`Delta::Reasoning`] while the round ran, which is the only time showing them
351 /// does any good. It is here for the callers that take a turn whole.
352 pub thinking: String,
353 /// Set when the provider stopped because the reply hit `max_tokens` --
354 /// `finish_reason: "length"`, or Anthropic's `stop_reason: "max_tokens"`.
355 ///
356 /// A tool call cut here arrives as MALFORMED JSON, so the caller needs to
357 /// tell "the model wrote bad JSON" from "the reply ran out of room": the
358 /// first is the model's mistake, the second is a setting, and only one of
359 /// them is worth retrying.
360 ///
361 /// It is NOT an error and is never retried. A reply that hit the cap is a
362 /// complete HTTP 200, and sending the same request again costs money and
363 /// produces the same cut.
364 pub truncated: bool,
365}
366
367
368// ┌───────────────────────────────────────────────────────────────┐
369// │ Retry │
370// └───────────────────────────────────────────────────────────────┘
371
372/// Extra milliseconds added on top of a provider's `Retry-After`, so a fan-out
373/// of workers told the same thing does not all come back at the same instant.
374const RETRY_AFTER_JITTER_MS: u64 = 250;
375
376/// Bytes of a refusal's body carried into the error.
377///
378/// Enough for what every provider puts first -- the message, the type and the code --
379/// and short enough that a provider answering an oversized request with an echo of it
380/// cannot put the whole thing in a user's message pane. Both transports use it, so the
381/// browser and the native path say the same thing about the same failure.
382const ERR_BODY_BYTES: usize = 300;
383
384/// `s` cut to at most `n` bytes, never through the middle of a character.
385///
386/// `&s[..n]` panics on a multi-byte boundary, and the one place this is used is an error
387/// path handed arbitrary bytes from a provider -- exactly where a panic is least welcome
388/// and least likely to be noticed in testing.
389///
390/// # Arguments
391/// * `s` - The text to cut.
392/// * `n` - The most bytes the result may occupy.
393fn clip_bytes(s: &str, n: usize) -> &str {
394 let mut cut = s.len().min(n);
395 while cut > 0 && !s.is_char_boundary(cut) {
396 cut -= 1;
397 }
398 &s[..cut]
399}
400
401/// Bounded exponential backoff for transient provider failures.
402///
403/// A 429, a 5xx or a dropped connection is the provider saying "not now"; every
404/// other 4xx is the request itself being wrong, and sending it again only costs
405/// money and time. Only the former is retried.
406#[derive(Clone, Copy, Debug)]
407pub struct RetryPolicy {
408 /// Total attempts including the first. One disables retrying.
409 pub max_attempts: u32,
410 /// Backoff before the first retry, doubling for each one after it.
411 pub base_ms: u64,
412 /// Ceiling on any single backoff.
413 pub max_backoff_ms: u64,
414 /// Ceiling on the sum of every backoff within one call, so a turn ends
415 /// while the user is still watching it.
416 pub max_total_wait_ms: u64,
417}
418
419impl Default for RetryPolicy {
420 fn default() -> Self {
421 // Widened 2026-08-19: a laptop moving between locations routinely drops
422 // the network for tens of seconds while it wakes, reconnects and DNS
423 // resolves. The previous budget (4 attempts, 20s total) survived a flaky
424 // access point but not a 30-second gap, so a turn that could have completed
425 // once the machine settled died instead. Eight attempts over up to two
426 // minutes gives the reconnect time to happen, while still bounded so a
427 // genuinely down provider ends the turn rather than hanging. The stub test
428 // client overrides this with a fast policy, so the suite is unaffected.
429 Self {
430 max_attempts: 8,
431 base_ms: 1_000,
432 max_backoff_ms: 30_000,
433 max_total_wait_ms: 120_000,
434 }
435 }
436}
437
438impl RetryPolicy {
439
440 /// The delay before retry number `retry`, counting the first retry as one.
441 ///
442 /// A provider's own `Retry-After` wins over the computed backoff and is
443 /// never shortened -- it is the one party that knows when it will be ready.
444 /// Jitter is added either way: eight workers that hit the same 429 must not
445 /// retry in lockstep.
446 pub fn delay_ms(&self, retry: u32, after_ms: Option<u64>) -> u64 {
447 if let Some(ms) = after_ms {
448 return ms.saturating_add(Rand::in_range(0u64, RETRY_AFTER_JITTER_MS));
449 }
450 let shift = retry.saturating_sub(1).min(16);
451 let nominal = self.base_ms
452 .saturating_mul(1u64 << shift)
453 .min(self.max_backoff_ms);
454 // Equal jitter: half the nominal delay, plus a random part of the rest.
455 let half = nominal / 2;
456 half + Rand::in_range(0u64, nominal - half)
457 }
458
459 /// Whether another attempt is allowed, and what it must wait first.
460 ///
461 /// `None` ends the attempt: either the budget of attempts is spent, or the
462 /// next backoff would push the total wait past its bound.
463 ///
464 /// # Arguments
465 /// * `retries` - Retries already made.
466 /// * `waited` - Milliseconds already slept within this call.
467 /// * `after_ms` - What the provider asked for, if it asked.
468 pub fn next_delay(&self, retries: u32, waited: u64, after_ms: Option<u64>) -> Option<u64> {
469 if retries + 1 >= self.max_attempts {
470 return None;
471 }
472 let delay = self.delay_ms(retries + 1, after_ms);
473 if waited.saturating_add(delay) > self.max_total_wait_ms {
474 return None;
475 }
476 Some(delay)
477 }
478}
479
480/// A transport failure, and whether trying again could plausibly succeed.
481///
482/// Retryability is decided where the status code is still in hand, rather than
483/// by reading it back out of an error message later.
484struct TransportErr {
485 retryable: bool, // is another attempt worth making?
486 after_ms: Option<u64>, // what the provider asked us to wait, if it said
487 // A few plain words for the retry notice, and -- since 2026-08-28 -- for the
488 // caller. The error itself carries file, line and ANSI colouring, none of
489 // which belongs in a user's message pane. See [`TransportErr::crossed`].
490 reason: String,
491 err: Error<ErrTag>,
492}
493
494impl TransportErr {
495
496 /// A failure worth another attempt: a 429, a 5xx, or a broken connection.
497 fn transient(reason: String, err: Error<ErrTag>) -> Self {
498 Self { retryable: true, after_ms: None, reason, err }
499 }
500
501 /// A failure that will fail again the same way: a malformed request, a bad
502 /// key, an unknown model.
503 fn fatal(reason: String, err: Error<ErrTag>) -> Self {
504 Self { retryable: false, after_ms: None, reason, err }
505 }
506
507 /// Attach the provider's requested delay.
508 fn after(mut self, after_ms: Option<u64>) -> Self {
509 self.after_ms = after_ms;
510 self
511 }
512
513 /// The failure as it LEAVES this module, with the reason in front of it.
514 ///
515 /// WHY THE REASON HAS TO TRAVEL. Until this existed only `err` was returned,
516 /// and `err` on the browser path is the browser's own sentence -- Chromium
517 /// says `Failed to fetch` and WebKit says `Load failed` for the identical
518 /// event. The app's offline classifier (`isUnreachable`, www/js/daimond.js)
519 /// was therefore reduced to matching one vendor's prose, and on iOS it matched
520 /// nothing: a turn that died before the first token was written off as a
521 /// provider refusal, and the recovery built for exactly that case never ran.
522 ///
523 /// `reason` is this client's own wording and is the same on every browser, so
524 /// putting it in front of the error makes the classification a property of
525 /// Daimond rather than of Safari. It is first because the reader -- person or
526 /// regex -- should meet the plain sentence before the framing.
527 fn crossed(self) -> Error<ErrTag> {
528 err!(self.err, "{}", self.reason; IO, Network, Wire)
529 }
530}
531
532/// Whether an HTTP status is worth another attempt.
533///
534/// 429 is rate limiting and 5xx is the provider's own trouble; every other
535/// status is about this request and will not change by being sent twice.
536pub(crate) fn status_retryable(status: u16) -> bool {
537 status == 429 || (500..600).contains(&status)
538}
539
540/// Read a `Retry-After` header value as milliseconds.
541///
542/// Only the delta-seconds form is understood. The HTTP-date form reads as
543/// absent, which falls back to the client's own backoff rather than guessing.
544pub(crate) fn parse_retry_after(value: &str) -> Option<u64> {
545 value.trim().parse::<u64>().ok().map(|s| s.saturating_mul(1_000))
546}
547
548/// Read the status code out of an HTTP status line.
549#[cfg(not(target_arch = "wasm32"))]
550pub(crate) fn status_code(line: &str) -> Option<u16> {
551 line.split_whitespace().nth(1).and_then(|c| c.parse::<u16>().ok())
552}
553
554/// Find a header's value in a raw HTTP header block, case-insensitively.
555#[cfg(not(target_arch = "wasm32"))]
556pub(crate) fn header_value(headers: &str, name: &str) -> Option<String> {
557 for line in headers.lines() {
558 let (key, value) = match line.split_once(':') {
559 Some(kv) => kv,
560 None => continue,
561 };
562 if key.trim().eq_ignore_ascii_case(name) {
563 return Some(value.trim().to_string());
564 }
565 }
566 None
567}
568
569/// Sleep for `ms` milliseconds on the native transport.
570#[cfg(not(target_arch = "wasm32"))]
571pub(crate) async fn sleep_ms(ms: u64) {
572 tokio::time::sleep(std::time::Duration::from_millis(ms)).await;
573}
574
575/// Sleep for `ms` milliseconds in the browser, via `setTimeout`.
576///
577/// A scope with no timer resolves immediately, so a retry still happens -- just
578/// without the pause.
579#[cfg(target_arch = "wasm32")]
580pub(crate) async fn sleep_ms(ms: u64) {
581 use wasm_bindgen::JsCast;
582 use wasm_bindgen::JsValue;
583 use wasm_bindgen_futures::JsFuture;
584
585 let ms = ms.min(i32::MAX as u64) as i32;
586 let promise = js_sys::Promise::new(&mut |resolve: js_sys::Function, _reject| {
587 let scheduled = if let Some(win) = web_sys::window() {
588 win.set_timeout_with_callback_and_timeout_and_arguments_0(&resolve, ms)
589 } else {
590 match js_sys::global().dyn_into::<web_sys::WorkerGlobalScope>() {
591 Ok(scope) => scope
592 .set_timeout_with_callback_and_timeout_and_arguments_0(&resolve, ms),
593 Err(_) => Err(JsValue::NULL),
594 }
595 };
596 if scheduled.is_err() {
597 let _ = resolve.call0(&JsValue::NULL);
598 }
599 });
600 let _ = JsFuture::from(promise).await;
601}
602
603
604impl LlmClient {
605
606 /// Construct a client for the native transport (tokio + rustls).
607 #[cfg(not(target_arch = "wasm32"))]
608 pub fn new(
609 host: &str,
610 port: u16,
611 path: &str,
612 api_key: &str,
613 model: &str,
614 max_tokens: u32,
615 tls_config: Arc<ClientConfig>,
616 ) -> Self {
617 Self {
618 dialect: Dialect::for_endpoint(host, path),
619 host: host.to_string(),
620 port,
621 path: path.to_string(),
622 api_key: api_key.to_string(),
623 model: model.to_string(),
624 max_tokens,
625 retry: RetryPolicy::default(),
626 think: new_carry(),
627 open_folds: std::rc::Rc::new(std::cell::RefCell::new(std::collections::HashSet::new())),
628 blind: new_blind(),
629 tls_config,
630 }
631 }
632
633 /// Construct a client for the wasm transport (browser `fetch`).
634 ///
635 /// TLS trust is handled by the browser, so no `tls_config` is
636 /// required — the streaming API (`chat_stream` / `chat_once`) is
637 /// otherwise identical to the native client.
638 #[cfg(target_arch = "wasm32")]
639 pub fn new(
640 host: &str,
641 port: u16,
642 path: &str,
643 api_key: &str,
644 model: &str,
645 max_tokens: u32,
646 ) -> Self {
647 Self::new_with_scheme(host, port, path, api_key, model, max_tokens, true)
648 }
649
650 /// Construct a wasm client with an explicit URL scheme.
651 ///
652 /// `secure` selects `https` (`true`) or `http` (`false`). Real
653 /// providers always use `https`; the `http` form exists so a local
654 /// mock over `127.0.0.1` can be driven in a headless test.
655 #[cfg(target_arch = "wasm32")]
656 pub fn new_with_scheme(
657 host: &str,
658 port: u16,
659 path: &str,
660 api_key: &str,
661 model: &str,
662 max_tokens: u32,
663 secure: bool,
664 ) -> Self {
665 Self {
666 dialect: Dialect::for_endpoint(host, path),
667 host: host.to_string(),
668 port,
669 path: path.to_string(),
670 api_key: api_key.to_string(),
671 model: model.to_string(),
672 max_tokens,
673 retry: RetryPolicy::default(),
674 think: new_carry(),
675 open_folds: std::rc::Rc::new(std::cell::RefCell::new(std::collections::HashSet::new())),
676 blind: new_blind(),
677 secure,
678 abort: std::rc::Rc::new(std::cell::RefCell::new(None)),
679 }
680 }
681
682 /// Send a streaming chat completion request.
683 ///
684 /// Reads the SSE response line-by-line from the TLS stream, calling `on_token` for
685 /// each delta *as it arrives* -- [`Delta::Text`] for the answer, [`Delta::Reasoning`]
686 /// for the model's own working, which is never part of it.
687 /// Returns the full accumulated response and token usage when
688 /// the stream completes.
689 /// A stream that failed before emitting a token is retried; one that failed
690 /// after is not, because the caller has already been handed those tokens
691 /// and a fresh attempt would hand them over a second time.
692 pub async fn chat_stream(
693 &self,
694 messages: &[ChatMessage],
695 on_token: &mut impl FnMut(Delta<'_>),
696 ) -> Outcome<ChatResponse> {
697 let resp = res!(self.stream_turn(messages, None, on_token, false).await);
698 Ok(ChatResponse {
699 content: resp.content,
700 prompt_tokens: resp.prompt_tokens,
701 completion_tokens: resp.completion_tokens,
702 cached_tokens: resp.cached_tokens,
703 cost_usd: resp.cost_usd,
704 aborted: resp.aborted,
705 retries: resp.retries,
706 thinking: resp.thinking,
707 truncated: resp.truncated,
708 })
709 }
710
711 /// Streaming chat completion with tools enabled.
712 ///
713 /// Issues the request with `stream: true` and reconstructs the
714 /// assistant turn from the SSE deltas: text and reasoning are forwarded
715 /// to `on_token` as they arrive, each labelled (so the answer streams even
716 /// while tools are active, and so does the working that precedes it), and
717 /// any `tool_calls` fragments are accumulated across chunks into whole
718 /// calls (see [`StreamAcc`]). Returns the same
719 /// [`ChatOnceResponse`] shape as [`chat_once`](Self::chat_once).
720 ///
721 /// A 429, a 5xx or a dropped connection is retried with bounded exponential
722 /// backoff -- but only while the turn has produced nothing. Once a token,
723 /// or a fragment of a tool call, has reached the caller, a retry would
724 /// deliver it twice, so the partial and the error are surfaced instead.
725 /// Each retry announces itself through `on_token`, because a thirty-second
726 /// turn that silently becomes ninety is its own defect.
727 pub async fn chat_stream_tools(
728 &self,
729 messages: &[ChatMessage],
730 tools: Option<&str>,
731 on_token: &mut impl FnMut(Delta<'_>),
732 ) -> Outcome<ChatOnceResponse> {
733 self.stream_turn(messages, tools, on_token, true).await
734 }
735
736 /// The one streamed turn both public streaming entry points run.
737 ///
738 /// Builds the request in whichever [`Dialect`] the endpoint speaks, drives
739 /// the SSE response through the matching accumulator, and applies the retry
740 /// policy. `notify` decides whether a retry announces itself through
741 /// `on_token`: the tool path does (a thirty-second turn that silently
742 /// becomes ninety is its own defect), the plain-chat path does not, because
743 /// its caller treats every token as answer text.
744 ///
745 /// # Arguments
746 /// * `messages` - The conversation so far.
747 /// * `tools` - A ready-made OpenAI-shaped tool array, translated for the
748 /// Anthropic dialect; `None` disables tools.
749 /// * `on_token` - Called with each delta as it arrives, labelled by kind.
750 /// * `notify` - Whether to announce a retry through `on_token`, as text.
751 async fn stream_turn(
752 &self,
753 messages: &[ChatMessage],
754 tools: Option<&str>,
755 on_token: &mut impl FnMut(Delta<'_>),
756 notify: bool,
757 ) -> Outcome<ChatOnceResponse> {
758 let images = res!(self.vision_guard(messages));
759 let stripped = self.sighted(messages, images);
760 let mut body = self.build_body(stripped.as_deref().unwrap_or(messages), tools, true);
761 // Set once the pictures have been taken out and the turn tried again, so the retry
762 // happens at most once and a second failure is reported as itself.
763 let mut retried_blind = stripped.is_some();
764 let mut waited = 0u64;
765 let mut retries = 0u32;
766 loop {
767 let mut acc = Acc::new(self.dialect);
768 let mut emitted = false;
769 let outcome = {
770 let mut sink = |data: &str| {
771 acc.ingest(data, &mut |d: Delta<'_>| {
772 // ONLY TEXT MAKES A TURN UNREPEATABLE. Reasoning already shown and
773 // then shown again reads as the model thinking twice, which is odd;
774 // an answer delivered twice is wrong. So a turn that has only
775 // reasoned so far is still safe to start over.
776 if matches!(d, Delta::Text(_)) { emitted = true; }
777 on_token(d);
778 });
779 };
780 self.stream_sse(&body, &mut sink).await
781 };
782 // An `error` event on a 200 stream is the provider's own trouble
783 // arriving after the headers, so it is classified like a status code
784 // rather than read as a short answer.
785 let outcome = match outcome {
786 Ok(aborted) => match acc.stream_error() {
787 Some(e) if !emitted && !acc.has_output() => Err(e),
788 _ => Ok(aborted),
789 },
790 Err(e) => Err(e),
791 };
792 match outcome {
793 Ok(aborted) => {
794 let thinking = acc.take_thinking();
795 let resp = acc.into_response(aborted, retries);
796 // The signed reasoning of a turn that asked for tools is held
797 // for the request that returns their results; see [`ThinkCarry`].
798 if let Some(tc) = resp.tool_calls.first() {
799 self.carry_put(&tc.id, thinking);
800 }
801 return Ok(resp);
802 }
803 Err(e) => {
804 // Anything the caller has already seen -- streamed text, or a
805 // tool-call fragment that will become one -- makes this turn
806 // unrepeatable.
807 let started = emitted || acc.has_output();
808 // A REFUSED PICTURE IS NOT A DEAD TURN. The provider would not take this
809 // request and it carried images, so the likeliest reason is the one thing
810 // in it a text model cannot read. Take them out, say so in their place, and
811 // send it again -- once. Only where nothing has been emitted: a turn the
812 // user has already seen tokens from cannot be started over.
813 if !started && !retried_blind && images > 0 {
814 retried_blind = true;
815 self.mark_blind();
816 let text_only: Vec<ChatMessage> =
817 messages.iter()
818 .map(|m| m.with_content(m.content().without_images(Dropped::Unseeable)))
819 .collect();
820 body = self.build_body(&text_only, tools, true);
821 if notify {
822 on_token(Delta::Text(&fmt!(
823 "\n[daimond: the model would not take {} image{}; asking again \
824 without {} -- it cannot see]\n",
825 images,
826 if images == 1 { "" } else { "s" },
827 if images == 1 { "it" } else { "them" })));
828 }
829 continue;
830 }
831 if started || !e.retryable {
832 return Err(self.vision_error(e.crossed(), images));
833 }
834 let delay = match self.retry.next_delay(retries, waited, e.after_ms) {
835 Some(d) => d,
836 None => return Err(self.vision_error(e.crossed(), images)),
837 };
838 waited += delay;
839 retries += 1;
840 if notify {
841 on_token(Delta::Text(&fmt!(
842 "\n[daimond: {}; retrying in {}.{:01}s -- attempt {} of {}]\n",
843 e.reason,
844 delay / 1_000,
845 (delay % 1_000) / 100,
846 retries + 1,
847 self.retry.max_attempts)));
848 }
849 sleep_ms(delay).await;
850 }
851 }
852 }
853 }
854
855 /// Non-streaming chat completion, optionally with tools.
856 ///
857 /// Returns the assistant content and any `tool_calls` the model
858 /// wants executed, plus token usage. Retained for callers that
859 /// prefer a single whole-response parse over streamed fragments.
860 pub async fn chat_once(
861 &self,
862 messages: &[ChatMessage],
863 tools: Option<&str>,
864 ) -> Outcome<ChatOnceResponse> {
865 let images = res!(self.vision_guard(messages));
866 let stripped = self.sighted(messages, images);
867 let mut body = self.build_body(stripped.as_deref().unwrap_or(messages), tools, false);
868 let mut retried_blind = stripped.is_some();
869 let mut waited = 0u64;
870 let mut retries = 0u32;
871 let raw = loop {
872 match self.do_request_full(&body).await {
873 Ok(r) => break r,
874 Err(e) => {
875 // Nothing streams on this path, so there is never a partial
876 // to protect -- only the classification matters.
877 // The picture retry, exactly as `stream_turn` does it and for the same
878 // reason; there is no emitted-tokens condition here because nothing has
879 // been shown to anybody yet.
880 if !retried_blind && images > 0 {
881 retried_blind = true;
882 self.mark_blind();
883 let text_only: Vec<ChatMessage> =
884 messages.iter()
885 .map(|m| m.with_content(m.content().without_images(Dropped::Unseeable)))
886 .collect();
887 body = self.build_body(&text_only, tools, false);
888 continue;
889 }
890 if !e.retryable {
891 return Err(self.vision_error(e.crossed(), images));
892 }
893 let delay = match self.retry.next_delay(retries, waited, e.after_ms) {
894 Some(d) => d,
895 None => return Err(self.vision_error(e.crossed(), images)),
896 };
897 waited += delay;
898 retries += 1;
899 sleep_ms(delay).await;
900 }
901 }
902 };
903 let (content, tool_calls, use_, thinking) = match self.dialect {
904 Dialect::OpenAi => {
905 let (c, t, u) = parse_full_response(&raw);
906 (c, t, u, Vec::new())
907 }
908 Dialect::Anthropic => parse_anthropic_response(&raw),
909 };
910 let thinking_text = thinking.iter()
911 .filter_map(|b| extract_json_string(b, "thinking"))
912 .filter(|s| !s.is_empty())
913 .collect::<Vec<String>>()
914 .join("\n");
915 if let Some(tc) = tool_calls.first() {
916 self.carry_put(&tc.id, thinking);
917 }
918 // Read from the whole body, in whichever dialect it came back in.
919 let truncated = match self.dialect {
920 Dialect::OpenAi => openai_truncated(&raw),
921 Dialect::Anthropic => anthropic_truncated(&raw),
922 };
923 Ok(ChatOnceResponse {
924 content,
925 tool_calls,
926 prompt_tokens: use_.prompt,
927 completion_tokens: use_.completion,
928 cached_tokens: use_.cached,
929 cost_usd: use_.cost_usd,
930 aborted: false,
931 retries,
932 thinking: thinking_text,
933 truncated,
934 })
935 }
936
937 /// Refuse, before the request is built, to send an image to a model known not to see.
938 ///
939 /// Returns how many images the conversation carries, which is zero on nearly every turn and
940 /// is what [`vision_error`](Self::vision_error) needs afterwards.
941 ///
942 /// The refusal names the model, because that is the fact the user has to act on: the app
943 /// cannot tell them which model to pick, but it can tell them the one they picked is the
944 /// reason nothing was looked at. A provider's own 400 says none of that -- at best it names
945 /// a content type.
946 ///
947 /// # Arguments
948 /// * `messages` - The conversation about to be sent.
949 /// Whether this endpoint has already been caught refusing pictures.
950 fn is_blind(&self) -> bool {
951 #[cfg(not(target_arch = "wasm32"))]
952 { self.blind.load(std::sync::atomic::Ordering::Relaxed) }
953 #[cfg(target_arch = "wasm32")]
954 { self.blind.get() }
955 }
956
957 /// Record that it does, so no later turn pays to find out again.
958 fn mark_blind(&self) {
959 #[cfg(not(target_arch = "wasm32"))]
960 { self.blind.store(true, std::sync::atomic::Ordering::Relaxed) }
961 #[cfg(target_arch = "wasm32")]
962 { self.blind.set(true) }
963 }
964
965 /// May a picture be put in front of this endpoint?
966 ///
967 /// Both halves of what is known, and nothing else: the deny-list [`model_can_see`] before any
968 /// request has been sent, and the refusal [`is_blind`](Self::is_blind) records after one has
969 /// been turned away. There is no third source -- no `vision` flag is published by anybody --
970 /// so a model released after this line was written is taken to see until it says otherwise.
971 pub fn can_take_images(&self) -> bool {
972 model_can_see(&self.model) && !self.is_blind()
973 }
974
975 /// The conversation as it must be sent: whole, or with the pictures turned into words when
976 /// this endpoint has been caught refusing them.
977 ///
978 /// Returns `None` when nothing needs changing, so the ordinary turn copies no messages.
979 fn sighted<'m>(&self, messages: &'m [ChatMessage], images: usize)
980 -> Option<Vec<ChatMessage>>
981 {
982 if images == 0 || !self.is_blind() {
983 return None;
984 }
985 let _ = messages.len();
986 Some(messages.iter()
987 .map(|m| m.with_content(m.content().without_images(Dropped::Unseeable)))
988 .collect())
989 }
990
991 fn vision_guard(&self, messages: &[ChatMessage]) -> Outcome<usize> {
992 let images: usize = messages.iter().map(|m| m.content().images().count()).sum();
993 // A refusal already seen is not an error any more: the pictures come out and the turn
994 // goes ahead. Refusing here instead would leave a conversation that carries one image
995 // permanently unable to take a turn -- which is what happened to a real Diamond on
996 // 2026-08-13, where a cover read into the daimon's history bricked every later steer.
997 if images == 0 || model_can_see(&self.model) || self.is_blind() {
998 return Ok(images);
999 }
1000 Err(err!(
1001 "The model '{}' cannot see. This turn carries {} image{} and that model takes text \
1002 only, so it would answer as though nothing had been shown to it. Choose a model with \
1003 vision and read the file again.",
1004 self.model, images, if images == 1 { "" } else { "s" };
1005 Invalid, Input, Unimplemented))
1006 }
1007
1008 /// Say what a failed request that carried images most likely failed for.
1009 ///
1010 /// [`vision_guard`](Self::vision_guard) can only refuse a model it has been told about, and no
1011 /// list of those is ever complete -- Daimond takes an arbitrary endpoint and an arbitrary
1012 /// model id. So the second half of the answer is here: when a turn that carried images comes
1013 /// back refused, and the provider's words are about images, the model is named and the reason
1014 /// is said plainly, with the provider's own sentence kept after it rather than replaced.
1015 ///
1016 /// A failure with no images in the turn, or whose text says nothing about them, is returned
1017 /// exactly as it arrived. Guessing at an unrelated failure would be worse than saying nothing.
1018 ///
1019 /// # Arguments
1020 /// * `e` - The error the provider produced.
1021 /// * `images` - How many images the refused turn carried.
1022 fn vision_error(&self, e: Error<ErrTag>, images: usize) -> Error<ErrTag> {
1023 if images == 0 {
1024 return e;
1025 }
1026 let low = fmt!("{}", e).to_lowercase();
1027 let about_images = [
1028 "image", "vision", "multimodal", "media_type", "media type", "image_url",
1029 ].iter().any(|m| low.contains(m));
1030 if !about_images {
1031 return e;
1032 }
1033 err!(
1034 "The model '{}' could not be shown the {} image{} in this turn -- it appears not to \
1035 see. Choose a model with vision. The provider said: {}",
1036 self.model, images, if images == 1 { "" } else { "s" }, e;
1037 Invalid, Input, Unimplemented)
1038 }
1039
1040 /// Build the JSON request body for the OpenAI-compatible API.
1041 ///
1042 /// `tools` (if present) is a ready-made JSON array injected as the
1043 /// `tools` field with `tool_choice: auto`. `stream` toggles SSE
1044 /// streaming and usage reporting.
1045 ///
1046 /// Messages chosen by [`cache_breakpoints`](Self::cache_breakpoints) carry an
1047 /// Anthropic `cache_control` marker. Providers that cache automatically
1048 /// ignore it; Claude models, which do not, need it or an agentic session
1049 /// re-pays full price for the same prompt on every round.
1050 fn build_body(&self, messages: &[ChatMessage], tools: Option<&str>, stream: bool) -> String {
1051 match self.dialect {
1052 Dialect::OpenAi => self.build_openai_body(messages, tools, stream),
1053 Dialect::Anthropic => self.build_anthropic_body(messages, tools, stream),
1054 }
1055 }
1056
1057 /// The OpenAI-compatible request body.
1058 ///
1059 /// See [`build_body`](Self::build_body) for the shared contract.
1060 ///
1061 /// One thing here is not a straight translation of the message list. A `tool` message on this
1062 /// side may hold text and nothing else -- the content-part union for that role has no image
1063 /// member -- so an image returned by a tool cannot ride in the reply that returned it. It is
1064 /// re-homed instead: the tool reply carries its text, and the images from a whole RUN of tool
1065 /// replies are emitted together in one `user` message directly after the run. After the run
1066 /// and not between the replies, because a run of `tool` messages answers one assistant turn
1067 /// and a message of another role wedged inside it is a conversation the API rejects.
1068 fn build_openai_body(&self, messages: &[ChatMessage], tools: Option<&str>, stream: bool)
1069 -> String
1070 {
1071 let marks = self.cache_breakpoints(messages, tools);
1072 let mut out = String::with_capacity(1024);
1073 out.push('{');
1074 out.push_str(&fmt!("\"model\":\"{}\",", self.model));
1075 out.push_str("\"messages\":[");
1076 let mut first = true;
1077 // Images lifted out of the tool replies of the run now being emitted.
1078 let mut carried: Vec<String> = Vec::new();
1079 for (i, msg) in messages.iter().enumerate() {
1080 if !matches!(msg, ChatMessage::Tool { .. }) && !carried.is_empty() {
1081 if !first { out.push(','); }
1082 out.push_str(&tool_image_message(&carried));
1083 carried.clear();
1084 first = false;
1085 }
1086 if let ChatMessage::Tool { content, .. } = msg {
1087 for img in content.images() {
1088 carried.push(fmt!(
1089 "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:{};base64,{}\"}}}}",
1090 img.media.mime(), img.base64()));
1091 }
1092 }
1093 if !first { out.push(','); }
1094 first = false;
1095 if marks.contains(&i) {
1096 out.push_str(&message_to_json_cached(msg, &self.open_folds.borrow()));
1097 } else {
1098 out.push_str(&message_to_json(msg, &self.open_folds.borrow()));
1099 }
1100 }
1101 if !carried.is_empty() {
1102 if !first { out.push(','); }
1103 out.push_str(&tool_image_message(&carried));
1104 }
1105 out.push_str("],");
1106 if let Some(t) = tools {
1107 out.push_str(&fmt!("\"tools\":{},", t));
1108 out.push_str("\"tool_choice\":\"auto\",");
1109 }
1110 if stream {
1111 out.push_str("\"stream\":true,");
1112 out.push_str("\"stream_options\":{\"include_usage\":true},");
1113 } else {
1114 out.push_str("\"stream\":false,");
1115 }
1116 out.push_str(&fmt!("\"max_tokens\":{}", self.max_tokens));
1117 out.push('}');
1118 out
1119 }
1120
1121 /// Streaming body (no tools). Kept for the pure-chat path's unit test,
1122 /// which is the only caller now that both paths share [`stream_turn`](Self::stream_turn).
1123 #[cfg(test)]
1124 fn build_request_body(&self, messages: &[ChatMessage]) -> String {
1125 self.build_body(messages, None, true)
1126 }
1127
1128 /// The Anthropic Messages API request body.
1129 ///
1130 /// Four things differ from the OpenAI shape, and each one is why this
1131 /// could not be a couple of extra fields on the other builder:
1132 ///
1133 /// * the system prompt is a top-level `system`, not a message, so every
1134 /// system message is hoisted out and joined;
1135 /// * content is an array of typed blocks, so a `cache_control` marker has
1136 /// somewhere to live without changing the message's shape;
1137 /// * a tool call is a `tool_use` block on the assistant turn and its result
1138 /// a `tool_result` block on the *user* turn, so a run of tool results
1139 /// coalesces into one user message rather than becoming several;
1140 /// * thinking blocks precede the `tool_use` blocks they were generated
1141 /// beside, and must be handed back unmodified -- see [`ThinkCarry`].
1142 ///
1143 /// Thinking is requested only for the models that take the adaptive form;
1144 /// see [`model_takes_adaptive_thinking`].
1145 fn build_anthropic_body(&self, messages: &[ChatMessage], tools: Option<&str>, stream: bool)
1146 -> String
1147 {
1148 let marks = self.cache_breakpoints(messages, tools);
1149 let thinks = model_takes_adaptive_thinking(&self.model);
1150 let mut out = String::with_capacity(1024);
1151 out.push('{');
1152 out.push_str(&fmt!("\"model\":\"{}\",", self.model));
1153 out.push_str(&fmt!("\"max_tokens\":{},", self.anthropic_max_tokens(thinks, stream)));
1154
1155 // The system prompt, hoisted. Several system messages become one
1156 // block: the API takes a single system field, and the model reads a
1157 // joined prompt exactly as it read separate messages.
1158 let sys: Vec<String> = messages.iter().filter_map(|m| match m {
1159 ChatMessage::System { content } => Some(content.as_text().into_owned()),
1160 _ => None,
1161 }).collect();
1162 if !sys.is_empty() {
1163 let mark = messages.iter().enumerate().any(|(i, m)|
1164 matches!(m, ChatMessage::System { .. }) && marks.contains(&i));
1165 out.push_str("\"system\":[{\"type\":\"text\",\"text\":\"");
1166 out.push_str(&json_escape(&sys.join("\n\n")));
1167 out.push('"');
1168 if mark { out.push_str(",\"cache_control\":{\"type\":\"ephemeral\"}"); }
1169 out.push_str("}],");
1170 }
1171
1172 // The conversation. `pending` holds the content blocks of the user
1173 // message being assembled, so consecutive tool results land in one
1174 // message rather than in several the API would reject.
1175 let mut msgs: Vec<String> = Vec::new();
1176 let mut pending: Vec<String> = Vec::new();
1177 for (i, msg) in messages.iter().enumerate() {
1178 match msg {
1179 ChatMessage::System { .. } => {}
1180 ChatMessage::User { content } => {
1181 // An empty text block is rejected outright, where the
1182 // OpenAI side simply carries the empty string through.
1183 pending.extend(anthropic_blocks(content, marks.contains(&i)));
1184 }
1185 ChatMessage::Tool { tool_call_id, content } => {
1186 // A `tool_result` takes either a string or an array of blocks, and this side
1187 // -- unlike OpenAI's -- takes an image among them. So a screenshot stays
1188 // attached to the call that produced it rather than being re-homed.
1189 if content.has_image() {
1190 let blocks = anthropic_blocks(content, false);
1191 pending.push(fmt!(
1192 "{{\"type\":\"tool_result\",\"tool_use_id\":\"{}\",\"content\":[{}]}}",
1193 json_escape(tool_call_id), blocks.join(",")));
1194 } else {
1195 pending.push(fmt!(
1196 "{{\"type\":\"tool_result\",\"tool_use_id\":\"{}\",\"content\":\"{}\"}}",
1197 json_escape(tool_call_id), json_escape(&content.as_text())));
1198 }
1199 }
1200 ChatMessage::Assistant { content, tool_calls } => {
1201 if !pending.is_empty() {
1202 msgs.push(fmt!("{{\"role\":\"user\",\"content\":[{}]}}", pending.join(",")));
1203 pending.clear();
1204 }
1205 let mut blocks: Vec<String> = Vec::new();
1206 // The reasoning that led to these tool calls, first and
1207 // verbatim. Absent for a turn that asked for nothing.
1208 if let Some(tc) = tool_calls.first() {
1209 blocks.extend(self.carry_get(&tc.id));
1210 }
1211 // Assistant turns are the model's own words; an image cannot appear in one.
1212 let said = content.as_text();
1213 if !said.is_empty() {
1214 // The same fold strip the OpenAI side applies. Applied at one site and
1215 // not the other, the same conversation would cost different amounts
1216 // through different endpoints, silently.
1217 let folded = strip_folds(&said, &self.open_folds.borrow());
1218 blocks.push(text_block(folded.as_deref().unwrap_or(&said), false));
1219 }
1220 for tc in tool_calls {
1221 let stripped = strip_said(&tc.name, &tc.arguments, self.fold_open(&tc.id));
1222 let raw = stripped.as_deref().unwrap_or(&tc.arguments);
1223 let args = if raw.trim_start().starts_with('{') {
1224 raw
1225 } else {
1226 "{}"
1227 };
1228 blocks.push(fmt!(
1229 "{{\"type\":\"tool_use\",\"id\":\"{}\",\"name\":\"{}\",\"input\":{}}}",
1230 json_escape(&tc.id), json_escape(&tc.name), args));
1231 }
1232 // An assistant turn with no content at all is not a message
1233 // the API will take, and it says nothing the model needs.
1234 if !blocks.is_empty() {
1235 msgs.push(fmt!("{{\"role\":\"assistant\",\"content\":[{}]}}", blocks.join(",")));
1236 }
1237 }
1238 }
1239 }
1240 if !pending.is_empty() {
1241 msgs.push(fmt!("{{\"role\":\"user\",\"content\":[{}]}}", pending.join(",")));
1242 }
1243 out.push_str(&fmt!("\"messages\":[{}],", msgs.join(",")));
1244
1245 if let Some(t) = tools {
1246 out.push_str(&fmt!("\"tools\":{},", openai_tools_to_anthropic(t)));
1247 out.push_str("\"tool_choice\":{\"type\":\"auto\"},");
1248 }
1249 if thinks {
1250 // `display` defaults to `omitted` on every current model, which
1251 // streams thinking blocks whose text is empty. Summarised costs
1252 // the same -- the billed thinking is the full reasoning either way
1253 // -- and is the difference between a visible pause and a silent one.
1254 out.push_str("\"thinking\":{\"type\":\"adaptive\",\"display\":\"summarized\"},");
1255 }
1256 out.push_str(&fmt!("\"stream\":{}", if stream { "true" } else { "false" }));
1257 out.push('}');
1258 out
1259 }
1260
1261 /// The output cap for a Messages API request.
1262 ///
1263 /// On the OpenAI side `max_tokens` bounds the answer. On this side it
1264 /// bounds the reasoning *and* the answer together -- thinking is billed as
1265 /// output and counts against the same cap -- so a figure chosen for the
1266 /// first meaning truncates under the second, and the app's is 4096: enough
1267 /// for an answer, not enough for a hard problem thought through first. A
1268 /// floor is applied rather than the configured value being used, because
1269 /// that value is an internal default and not something a user chose.
1270 ///
1271 /// It applies only where both halves of the reason hold: a model that
1272 /// actually thinks, and a streamed request. The one-shot path keeps the
1273 /// configured cap, since a large one there risks an HTTP timeout on a
1274 /// connection with nothing arriving on it.
1275 ///
1276 /// # Arguments
1277 /// * `thinks` - Whether this request asks for thinking.
1278 /// * `stream` - Whether the response is streamed.
1279 fn anthropic_max_tokens(&self, thinks: bool, stream: bool) -> u32 {
1280 if thinks && stream {
1281 self.max_tokens.max(THINKING_MIN_MAX_TOKENS)
1282 } else {
1283 self.max_tokens
1284 }
1285 }
1286
1287 /// The headers this request needs beyond `Host` and `Content-Length`.
1288 ///
1289 /// The two dialects do not merely differ in the name of the auth header:
1290 /// Anthropic wants `x-api-key` plus a pinned API version, and refuses a
1291 /// bearer token. `browser` adds the header that makes Anthropic's edge
1292 /// answer a cross-origin `fetch` at all -- the same one the official
1293 /// TypeScript SDK sends for `dangerouslyAllowBrowser`. It is sent only
1294 /// from the browser transport, where it is the difference between the app
1295 /// working and CORS refusing it.
1296 ///
1297 /// # Arguments
1298 /// * `browser` - Whether the request is being made from a browser.
1299 fn auth_headers(&self, browser: bool) -> Vec<(&'static str, String)> {
1300 let mut out = vec![("Content-Type", "application/json".to_string())];
1301 match self.dialect {
1302 Dialect::OpenAi => {
1303 out.push(("Authorization", fmt!("Bearer {}", self.api_key)));
1304 }
1305 Dialect::Anthropic => {
1306 out.push(("x-api-key", self.api_key.clone()));
1307 out.push(("anthropic-version", ANTHROPIC_VERSION.to_string()));
1308 if browser {
1309 out.push(("anthropic-dangerous-direct-browser-access", "true".to_string()));
1310 }
1311 }
1312 }
1313 out
1314 }
1315
1316 /// Hold this turn's thinking blocks against the tool call they accompany.
1317 ///
1318 /// A poisoned lock loses the carry rather than the turn: the next request
1319 /// then goes without thinking blocks, which the API answers by quietly
1320 /// disabling thinking for it. That is a worse answer, not a broken one,
1321 /// and it is the right trade against failing a turn the user is watching.
1322 ///
1323 /// # Arguments
1324 /// * `id` - The first tool-call id of the turn the blocks came from.
1325 /// * `blocks` - The serialised blocks, in the order the model produced them.
1326 fn carry_put(&self, id: &str, blocks: Vec<String>) {
1327 if id.is_empty() || blocks.is_empty() {
1328 return;
1329 }
1330 let go = |c: &mut ThinkCarry| {
1331 // A retried round re-reports the same id; the newer blocks replace
1332 // the older rather than sitting beside them.
1333 c.turns.retain(|(k, _)| k != id);
1334 c.turns.push((id.to_string(), blocks.clone()));
1335 if c.turns.len() > CARRY_MAX_TURNS {
1336 let drop = c.turns.len() - CARRY_MAX_TURNS;
1337 c.turns.drain(..drop);
1338 }
1339 };
1340 #[cfg(not(target_arch = "wasm32"))]
1341 { if let Ok(mut g) = self.think.lock() { go(&mut g); } }
1342 #[cfg(target_arch = "wasm32")]
1343 { go(&mut self.think.borrow_mut()); }
1344 }
1345
1346 /// Is this `say` call's fold open on screen?
1347 fn fold_open(&self, id: &str) -> bool {
1348 !id.is_empty() && self.open_folds.borrow().contains(id)
1349 }
1350
1351 /// The open folds, copied out.
1352 ///
1353 /// A COPY and not a borrow: the sizing path holds this across the awaits of a fold, and the
1354 /// page may set the folds again at any point in between -- a `RefCell` borrow still live at
1355 /// that moment would panic. The set holds one short id per fold on screen.
1356 pub fn open_folds(&self) -> OpenSet {
1357 self.open_folds.borrow().clone()
1358 }
1359
1360 /// Replace the set of open folds, from the page, before a request goes out.
1361 ///
1362 /// REPLACED and not added to: a fold the user has since closed must leave the payload, and an
1363 /// accumulating set could only ever grow.
1364 pub fn set_open_folds(&self, ids: Vec<String>) {
1365 let mut f = self.open_folds.borrow_mut();
1366 f.clear();
1367 for id in ids {
1368 f.insert(id);
1369 }
1370 }
1371
1372 /// The thinking blocks held for `id`, or none when no held turn produced
1373 /// that call (or a lock could not be taken; see
1374 /// [`carry_put`](Self::carry_put)).
1375 ///
1376 /// # Arguments
1377 /// * `id` - The first tool-call id of the assistant turn being serialised.
1378 fn carry_get(&self, id: &str) -> Vec<String> {
1379 if id.is_empty() {
1380 return Vec::new();
1381 }
1382 let find = |c: &ThinkCarry| c.turns.iter()
1383 .find(|(k, _)| k == id)
1384 .map(|(_, b)| b.clone())
1385 .unwrap_or_default();
1386 #[cfg(not(target_arch = "wasm32"))]
1387 {
1388 match self.think.lock() {
1389 Ok(g) => find(&g),
1390 Err(_) => Vec::new(),
1391 }
1392 }
1393 #[cfg(target_arch = "wasm32")]
1394 { find(&self.think.borrow()) }
1395 }
1396
1397 /// Which message indices get an Anthropic prompt-cache breakpoint.
1398 ///
1399 /// Two at most, both placed at a boundary between what stays the same and
1400 /// what changes:
1401 ///
1402 /// * the last system message, which with the tool definitions rendered ahead
1403 /// of it is the largest block that never varies within a session;
1404 /// * the last user message, which is the tip of the settled conversation --
1405 /// the next turn reads everything before it back out of the cache.
1406 ///
1407 /// Nothing is marked for a model that does not honour the marker, and
1408 /// nothing is marked when the prefix is too short to be cacheable at all.
1409 /// Assistant and tool messages are deliberately left unmarked: the array
1410 /// content form they would need is the one an OpenAI-compatible router is
1411 /// least certain to carry through, and a rejected body loses the whole turn.
1412 fn cache_breakpoints(&self, messages: &[ChatMessage], tools: Option<&str>) -> Vec<usize> {
1413 let mut marks = Vec::new();
1414 if !model_caches_on_request(&self.model) {
1415 return marks;
1416 }
1417 // The prefix at each message, in characters, standing in for tokens.
1418 let mut prefix = tools.map(|t| t.len()).unwrap_or(0);
1419 let mut sys = None;
1420 let mut usr = None;
1421 for (i, msg) in messages.iter().enumerate() {
1422 prefix += message_len(msg);
1423 if prefix < CACHE_MIN_PREFIX_CHARS {
1424 continue;
1425 }
1426 match msg {
1427 ChatMessage::System { .. } => sys = Some(i),
1428 ChatMessage::User { .. } => usr = Some(i),
1429 _ => {}
1430 }
1431 }
1432 if let Some(i) = sys { marks.push(i); }
1433 if let Some(i) = usr {
1434 if Some(i) != sys { marks.push(i); }
1435 }
1436 marks
1437 }
1438
1439 /// Connect, TLS-handshake, send the request, and consume the
1440 /// response headers. Returns the stream positioned at the body
1441 /// start plus whether the body uses chunked transfer encoding.
1442 /// Errors on a non-200 status (with body detail).
1443 ///
1444 /// Every failure here is classified but none is retried: retrying belongs to
1445 /// the public call, which is the only layer that knows whether anything has
1446 /// already reached the caller and is the only one that can say so.
1447 #[cfg(not(target_arch = "wasm32"))]
1448 async fn open(
1449 &self,
1450 body: &str,
1451 )
1452 -> Result<(tokio_rustls::client::TlsStream<tokio::net::TcpStream>, bool), TransportErr>
1453 {
1454 use tokio_rustls::TlsConnector;
1455 use tokio::net::TcpStream;
1456
1457 let body_bytes = body.as_bytes();
1458
1459 let mut request = String::with_capacity(512 + body_bytes.len());
1460 request.push_str(&fmt!("POST {} HTTP/1.1\r\n", self.path));
1461 request.push_str(&fmt!("Host: {}\r\n", self.host));
1462 for (name, value) in self.auth_headers(false) {
1463 request.push_str(&fmt!("{}: {}\r\n", name, value));
1464 }
1465 request.push_str(&fmt!("Content-Length: {}\r\n", body_bytes.len()));
1466 request.push_str("Connection: close\r\n");
1467 request.push_str("\r\n");
1468
1469 // A connection that never came up carries no partial answer, so every
1470 // failure from here to the status line is worth another attempt.
1471 let tcp = match TcpStream::connect((self.host.as_str(), self.port)).await {
1472 Ok(s) => s,
1473 Err(e) => return Err(TransportErr::transient(fmt!("could not reach {}", self.host), err!(e,
1474 "LLM: TCP connect to {}:{} failed.", self.host, self.port;
1475 IO, Network, Init))),
1476 };
1477 let server_name = match tokio_rustls::rustls::pki_types::ServerName::try_from(self.host.clone()) {
1478 Ok(n) => n,
1479 // A name that will not parse will not parse next time either.
1480 Err(e) => return Err(TransportErr::fatal(fmt!("invalid server name '{}'", self.host), err!(e,
1481 "LLM: invalid server name '{}'.", self.host;
1482 IO, Network, Invalid, Input))),
1483 };
1484 let connector = TlsConnector::from(self.tls_config.clone());
1485 let mut stream = match connector.connect(server_name, tcp).await {
1486 Ok(s) => s,
1487 Err(e) => return Err(TransportErr::transient(fmt!("TLS handshake with {} failed", self.host), err!(e,
1488 "LLM: TLS handshake to {} failed.", self.host;
1489 IO, Network, Init))),
1490 };
1491
1492 let mut req = Vec::with_capacity(request.as_bytes().len() + body_bytes.len());
1493 req.extend_from_slice(request.as_bytes());
1494 req.extend_from_slice(body_bytes);
1495 if let Err(e) = stream.write_all(&req).await {
1496 return Err(TransportErr::transient("could not send the request".to_string(), err!(e,
1497 "LLM: write request failed."; IO, Network, Wire, Write)));
1498 }
1499 if let Err(e) = stream.flush().await {
1500 return Err(TransportErr::transient("could not send the request".to_string(), err!(e,
1501 "LLM: flush failed."; IO, Network, Wire, Write)));
1502 }
1503
1504 // Read headers byte-by-byte until \r\n\r\n.
1505 let mut hdr_buf = Vec::with_capacity(2048);
1506 let mut byte = [0u8; 1];
1507 loop {
1508 match stream.read(&mut byte).await {
1509 Ok(0) => break,
1510 Ok(_) => {
1511 hdr_buf.push(byte[0]);
1512 if hdr_buf.ends_with(b"\r\n\r\n") { break; }
1513 }
1514 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => break,
1515 Err(e) => return Err(TransportErr::transient("the provider closed before replying".to_string(), err!(e,
1516 "LLM: read headers failed."; IO, Network, Wire, Read))),
1517 }
1518 }
1519
1520 let headers_str = String::from_utf8_lossy(&hdr_buf);
1521 let is_chunked = headers_str
1522 .to_ascii_lowercase()
1523 .contains("transfer-encoding: chunked");
1524
1525 let status_line = headers_str.lines().next().unwrap_or("");
1526 let status = status_code(status_line).unwrap_or(0);
1527 if status != 200 {
1528 let mut err_body = Vec::new();
1529 let mut chunk = [0u8; 4096];
1530 loop {
1531 match stream.read(&mut chunk).await {
1532 Ok(0) => break,
1533 Ok(n) => err_body.extend_from_slice(&chunk[..n]),
1534 Err(_) => break,
1535 }
1536 }
1537 let err_msg = String::from_utf8_lossy(&err_body);
1538 let err = err!(
1539 "LLM: HTTP error: {} | {}", status_line, clip_bytes(&err_msg, ERR_BODY_BYTES);
1540 IO, Network, Wire, Read);
1541 // A 429 or a 5xx is the provider saying "not now"; a 400 is this
1542 // request being wrong, and sending it again only costs money.
1543 let reason = fmt!("the provider returned HTTP {}", status);
1544 return Err(if status_retryable(status) {
1545 let after = header_value(&headers_str, "retry-after")
1546 .and_then(|v| parse_retry_after(&v));
1547 TransportErr::transient(reason, err).after(after)
1548 } else {
1549 TransportErr::fatal(reason, err)
1550 });
1551 }
1552
1553 Ok((stream, is_chunked))
1554 }
1555
1556 /// Perform a non-streaming request and return the full response
1557 /// body as one string. Lines are concatenated (JSON does not need
1558 /// the newlines), dechunking transparently.
1559 #[cfg(not(target_arch = "wasm32"))]
1560 async fn do_request_full(
1561 &self,
1562 body: &str,
1563 ) -> Result<String, TransportErr> {
1564 let (stream, is_chunked) = match self.open(body).await {
1565 Ok(v) => v,
1566 Err(e) => return Err(e),
1567 };
1568 let mut reader = LineReader::new(stream, is_chunked);
1569 let mut full = String::new();
1570 loop {
1571 match reader.read_line().await {
1572 Ok(Some(l)) => full.push_str(&l),
1573 Ok(None) => break,
1574 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => break,
1575 Err(e) => return Err(TransportErr::transient("the reply was cut short".to_string(), err!(e,
1576 "LLM: read response body failed."; IO, Network, Wire, Read))),
1577 }
1578 }
1579 Ok(full)
1580 }
1581
1582 /// Send the HTTP request and stream the SSE response line-by-line,
1583 /// calling `on_data` with each `data:` payload (the JSON after the
1584 /// `data: ` prefix) as it arrives, stopping at `[DONE]`. Handles
1585 /// both chunked and identity transfer encoding via [`LineReader`].
1586 ///
1587 /// Returns whether the stream was aborted. The native transport has
1588 /// no cancellation path, so it always returns `false`; the wasm
1589 /// transport returns `true` when the browser fired the abort signal.
1590 #[cfg(not(target_arch = "wasm32"))]
1591 async fn stream_sse(
1592 &self,
1593 body: &str,
1594 on_data: &mut impl FnMut(&str),
1595 ) -> Result<bool, TransportErr>
1596 {
1597 let (stream, is_chunked) = match self.open(body).await {
1598 Ok(v) => v,
1599 Err(e) => return Err(e),
1600 };
1601 let mut reader = LineReader::new(stream, is_chunked);
1602 loop {
1603 let line = match reader.read_line().await {
1604 Ok(Some(l)) => l,
1605 Ok(None) => break,
1606 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => break,
1607 Err(e) => return Err(TransportErr::transient("the stream broke".to_string(), err!(e,
1608 "LLM: read SSE line failed."; IO, Network, Wire, Read))),
1609 };
1610 let line = line.trim();
1611 if !line.starts_with("data: ") {
1612 continue;
1613 }
1614 let data = &line[6..];
1615 if data == "[DONE]" {
1616 break;
1617 }
1618 on_data(data);
1619 }
1620 Ok(false)
1621 }
1622}
1623
1624
1625// ┌───────────────────────────────────────────────────────────────┐
1626// │ Wasm transport — browser `fetch` + `ReadableStream` │
1627// └───────────────────────────────────────────────────────────────┘
1628//
1629// The wasm build has no TCP sockets or TLS stack; the browser owns
1630// both. These methods mirror the native transport's private contract
1631// (`do_request_full` / `stream_sse`) using `fetch`, so the
1632// `chat_stream` / `chat_stream_tools` / `chat_once` API above is
1633// target-agnostic.
1634
1635#[cfg(target_arch = "wasm32")]
1636impl LlmClient {
1637
1638 /// The absolute request URL for the browser transport.
1639 ///
1640 /// The scheme follows [`secure`](Self::secure); the port is elided
1641 /// only when it is the scheme's default (443 for `https`, 80 for
1642 /// `http`), so a mock on a custom port is addressed explicitly.
1643 fn wasm_url(&self) -> String {
1644 let (scheme, default_port) = if self.secure { ("https", 443u16) } else { ("http", 80u16) };
1645 if self.port == default_port {
1646 fmt!("{}://{}{}", scheme, self.host, self.path)
1647 } else {
1648 fmt!("{}://{}:{}{}", scheme, self.host, self.port, self.path)
1649 }
1650 }
1651
1652 /// Issue a lightweight transport probe and return the raw HTTP
1653 /// status the provider replies with.
1654 ///
1655 /// Unlike [`wasm_fetch`](Self::wasm_fetch), a non-2xx status is *not*
1656 /// treated as an error — the status number is the whole point. A
1657 /// `401` from a real provider with a dummy key proves the full
1658 /// `fetch` + CORS + transport path end-to-end without a valid key.
1659 pub async fn probe_status(&self) -> Outcome<u16> {
1660 let messages = [crate::protocol::ChatMessage::User {
1661 content: MessageContent::text("ping"),
1662 }];
1663 let body = self.build_body(&messages, None, false);
1664 let resp = res!(self.wasm_fetch_raw(&body).await);
1665 Ok(resp.status())
1666 }
1667
1668 /// POST `body` via `fetch`, retrying a transient failure with bounded
1669 /// backoff, and await the `Response`.
1670 ///
1671 /// Nothing has reached the caller at this point, so a dropped `fetch`, a 429
1672 /// or a 5xx is simply tried again; every other non-2xx is this request being
1673 /// wrong and is returned as-is. `waited` is the shared backoff budget; see
1674 /// the native [`open`](LlmClient::open).
1675 ///
1676 /// A refusal carries the provider's OWN WORDS, as the native transport has always
1677 /// done. Without them the browser could say no more than
1678 /// `LLM: HTTP error: 400 Bad Request.`, which tells the user nothing they can act on
1679 /// and tells [`compact::looks_like_overflow`](crate::agent::compact::looks_like_overflow)
1680 /// nothing at all -- so a request refused for being too long and one refused for
1681 /// being wrong had to be told apart by size alone. Both dialects are covered,
1682 /// because it is the raw body that is carried and neither is parsed.
1683 async fn wasm_fetch(&self, body: &str) -> Result<web_sys::Response, TransportErr> {
1684 let resp = match self.wasm_fetch_raw(body).await {
1685 Ok(r) => r,
1686 // A rejected `fetch` is a network or CORS failure; an armed abort is
1687 // the caller cancelling, and must not be retried.
1688 Err(e) => return Err(if self.abort_signalled() {
1689 TransportErr::fatal("the turn was cancelled".to_string(), e)
1690 } else {
1691 TransportErr::transient("could not reach the provider".to_string(), e)
1692 }),
1693 };
1694 if !resp.ok() {
1695 let status = resp.status();
1696 let status_text = resp.status_text();
1697 // Read BEFORE the body: consuming the stream cannot then cost the retry its
1698 // requested delay, and a `Retry-After` is the one thing on a 429 worth more
1699 // than the message.
1700 let after = if status_retryable(status) {
1701 resp.headers().get("retry-after").ok().flatten()
1702 .and_then(|v| parse_retry_after(&v))
1703 } else {
1704 None
1705 };
1706 let detail = self.body_detail(&resp).await;
1707 let err = err!(
1708 "LLM: HTTP error: {} {} | {}", status, status_text, detail;
1709 IO, Network, Wire, Read);
1710 let reason = fmt!("the provider returned HTTP {}", status);
1711 return Err(if status_retryable(status) {
1712 TransportErr::transient(reason, err).after(after)
1713 } else {
1714 TransportErr::fatal(reason, err)
1715 });
1716 }
1717 Ok(resp)
1718 }
1719
1720 /// The first [`ERR_BODY_BYTES`] of a refusal's body, or nothing when it cannot be read.
1721 ///
1722 /// Consumes the response, which is why it is called only on the failing path. A body
1723 /// that will not resolve -- an abort landing between the headers and the text, a
1724 /// provider that sent none -- yields an empty string rather than turning a refusal
1725 /// with a known status into a failure of a different kind.
1726 ///
1727 /// # Arguments
1728 /// * `resp` - The non-2xx response, whose body is read to exhaustion.
1729 async fn body_detail(&self, resp: &web_sys::Response) -> String {
1730 use wasm_bindgen_futures::JsFuture;
1731
1732 let text = match resp.text() {
1733 Ok(p) => match JsFuture::from(p).await {
1734 Ok(v) => v.as_string().unwrap_or_default(),
1735 Err(_) => String::new(),
1736 },
1737 Err(_) => String::new(),
1738 };
1739 clip_bytes(&text, ERR_BODY_BYTES).to_string()
1740 }
1741
1742 /// POST `body` via `fetch` and await the `Response` without checking
1743 /// the status, mapping any JS error into an `Outcome`. TLS trust is
1744 /// the browser's. Callers that need a 2xx guarantee go through
1745 /// [`wasm_fetch`](Self::wasm_fetch).
1746 async fn wasm_fetch_raw(&self, body: &str) -> Outcome<web_sys::Response> {
1747 use wasm_bindgen::JsCast;
1748 use wasm_bindgen::JsValue;
1749 use wasm_bindgen_futures::JsFuture;
1750 use web_sys::{Headers, Request, RequestInit, RequestMode, Response};
1751
1752 let headers = res!(Headers::new()
1753 .map_err(|e| err!("LLM: create headers failed: {}.", js_str(&e); IO, Network, Init)));
1754 // `true`: this is the browser transport, so an Anthropic endpoint also
1755 // gets the header that makes its edge answer a cross-origin request.
1756 for (name, value) in self.auth_headers(true) {
1757 res!(headers.append(name, &value)
1758 .map_err(|e| err!("LLM: set header {} failed: {}.", name, js_str(&e);
1759 IO, Network, Init)));
1760 }
1761
1762 let opts = RequestInit::new();
1763 opts.set_method("POST");
1764 opts.set_mode(RequestMode::Cors);
1765 opts.set_headers(&headers);
1766 opts.set_body(&JsValue::from_str(body));
1767
1768 // Install a fresh abort controller for this request and wire its
1769 // signal in, so `abort` can cancel the in-flight fetch/stream. A
1770 // controller that fails to construct simply leaves the request
1771 // uncancellable rather than failing the turn.
1772 if let Ok(ctrl) = web_sys::AbortController::new() {
1773 opts.set_signal(Some(&ctrl.signal()));
1774 *self.abort.borrow_mut() = Some(ctrl);
1775 }
1776
1777 let url = self.wasm_url();
1778 let request = res!(Request::new_with_str_and_init(&url, &opts)
1779 .map_err(|e| err!("LLM: build request failed: {}.", js_str(&e); IO, Network, Init)));
1780
1781 // `fetch` lives on the window in a document context and on the
1782 // global scope in a worker; support both.
1783 let promise = if let Some(win) = web_sys::window() {
1784 win.fetch_with_request(&request)
1785 } else {
1786 let scope = res!(js_sys::global()
1787 .dyn_into::<web_sys::WorkerGlobalScope>()
1788 .map_err(|_| err!(
1789 "LLM: no window or worker scope for fetch."; IO, Network, Init)));
1790 scope.fetch_with_request(&request)
1791 };
1792
1793 let resp_val = res!(JsFuture::from(promise).await
1794 .map_err(|e| err!("LLM: fetch failed: {}.", js_str(&e); IO, Network, Wire)));
1795 let resp: Response = res!(resp_val.dyn_into()
1796 .map_err(|_| err!("LLM: fetch did not return a Response."; IO, Network, Wire)));
1797 Ok(resp)
1798 }
1799
1800 /// Non-streaming request — await the full response body as text.
1801 async fn do_request_full(&self, body: &str) -> Result<String, TransportErr> {
1802 use wasm_bindgen_futures::JsFuture;
1803
1804 let resp = match self.wasm_fetch(body).await {
1805 Ok(r) => r,
1806 Err(e) => return Err(e),
1807 };
1808 let text_promise = match resp.text() {
1809 Ok(p) => p,
1810 Err(e) => return Err(TransportErr::transient("the reply was cut short".to_string(), err!(
1811 "LLM: read response text failed: {}.", js_str(&e); IO, Network, Wire, Read))),
1812 };
1813 let text_val = match JsFuture::from(text_promise).await {
1814 Ok(v) => v,
1815 Err(e) => return Err(TransportErr::transient("the reply was cut short".to_string(), err!(
1816 "LLM: await response text failed: {}.", js_str(&e); IO, Network, Wire, Read))),
1817 };
1818 Ok(text_val.as_string().unwrap_or_default())
1819 }
1820
1821 /// Streaming request — read the SSE body incrementally from the
1822 /// response's `ReadableStream`, calling `on_data` with each `data:`
1823 /// payload as it arrives, stopping at `[DONE]`.
1824 ///
1825 /// Returns whether the browser fired the abort signal. When the
1826 /// initial `fetch` or a stream read rejects, an armed abort is
1827 /// distinguished from a genuine transport failure: an abort resolves
1828 /// to `Ok(true)` (the caller keeps whatever streamed and ends the
1829 /// turn cleanly), any other rejection is a real error.
1830 async fn stream_sse(
1831 &self,
1832 body: &str,
1833 on_data: &mut impl FnMut(&str),
1834 ) -> Result<bool, TransportErr>
1835 {
1836 use wasm_bindgen::JsValue;
1837 use wasm_bindgen_futures::JsFuture;
1838 use web_sys::{ReadableStream, ReadableStreamDefaultReader};
1839
1840 let resp = match self.wasm_fetch(body).await {
1841 Ok(r) => r,
1842 Err(e) => {
1843 if self.abort_signalled() { return Ok(true); }
1844 return Err(e);
1845 }
1846 };
1847 let stream: ReadableStream = match resp.body() {
1848 Some(s) => s,
1849 None => return Err(TransportErr::transient("the reply carried no stream".to_string(), err!(
1850 "LLM: response has no body stream."; IO, Network, Wire, Read))),
1851 };
1852 let reader = match ReadableStreamDefaultReader::new(&stream) {
1853 Ok(r) => r,
1854 Err(e) => return Err(TransportErr::fatal("the stream could not be read".to_string(), err!(
1855 "LLM: acquire stream reader failed: {}.", js_str(&e); IO, Network, Wire, Read))),
1856 };
1857
1858 // Accumulate raw bytes and extract complete SSE lines as they
1859 // arrive, mirroring the native `LineReader` line discipline.
1860 let mut buf: Vec<u8> = Vec::with_capacity(8192);
1861
1862 loop {
1863 let result = match JsFuture::from(reader.read()).await {
1864 Ok(r) => r,
1865 Err(e) => {
1866 if self.abort_signalled() { return Ok(true); }
1867 // A stream that broke mid-flight; whether it is safe to try
1868 // again is the caller's judgement, not this layer's.
1869 return Err(TransportErr::transient("the stream broke".to_string(), err!(
1870 "LLM: read stream chunk failed: {}.", js_str(&e);
1871 IO, Network, Wire, Read)));
1872 }
1873 };
1874 let done = match js_sys::Reflect::get(&result, &JsValue::from_str("done")) {
1875 Ok(v) => v.as_bool().unwrap_or(true),
1876 Err(e) => return Err(TransportErr::fatal("the stream was malformed".to_string(), err!(
1877 "LLM: read 'done' failed: {}.", js_str(&e); IO, Network, Wire, Read))),
1878 };
1879 if done {
1880 break;
1881 }
1882 let value = match js_sys::Reflect::get(&result, &JsValue::from_str("value")) {
1883 Ok(v) => v,
1884 Err(e) => return Err(TransportErr::fatal("the stream was malformed".to_string(), err!(
1885 "LLM: read 'value' failed: {}.", js_str(&e); IO, Network, Wire, Read))),
1886 };
1887 let chunk = js_sys::Uint8Array::new(&value).to_vec();
1888 buf.extend_from_slice(&chunk);
1889
1890 // Drain complete lines (terminated by `\n`) from the buffer.
1891 loop {
1892 let nl = match buf.iter().position(|&b| b == b'\n') {
1893 Some(p) => p,
1894 None => break,
1895 };
1896 let line_bytes: Vec<u8> = buf.drain(..=nl).collect();
1897 let line = String::from_utf8_lossy(&line_bytes[..line_bytes.len() - 1]);
1898 let line = line.trim();
1899 if !line.starts_with("data: ") {
1900 continue;
1901 }
1902 let data = &line[6..];
1903 if data == "[DONE]" {
1904 return Ok(false);
1905 }
1906 on_data(data);
1907 }
1908 }
1909
1910 Ok(false)
1911 }
1912
1913 /// Fire the abort signal for the in-flight request, if any. Safe to
1914 /// call when idle: with no armed controller it is a no-op.
1915 pub fn abort(&self) {
1916 if let Some(ctrl) = self.abort.borrow().as_ref() {
1917 ctrl.abort();
1918 }
1919 }
1920
1921 /// Whether the armed abort controller's signal has fired. Used to
1922 /// tell a cancelled fetch/stream apart from a genuine failure.
1923 fn abort_signalled(&self) -> bool {
1924 self.abort
1925 .borrow()
1926 .as_ref()
1927 .map(|ctrl| ctrl.signal().aborted())
1928 .unwrap_or(false)
1929 }
1930}
1931
1932/// Render a JS error value as a human-readable string for error tags.
1933#[cfg(target_arch = "wasm32")]
1934fn js_str(v: &wasm_bindgen::JsValue) -> String {
1935 v.as_string().unwrap_or_else(|| fmt!("{:?}", v))
1936}
1937
1938
1939// ┌───────────────────────────────────────────────────────────────┐
1940// │ LineReader — incremental line reader for TLS streams │
1941// └───────────────────────────────────────────────────────────────┘
1942
1943/// Reads lines from a TLS stream, handling HTTP chunked transfer
1944/// encoding transparently.
1945///
1946/// For identity (Content-Length) encoding, lines are read directly
1947/// from the stream. For chunked encoding, chunk headers are parsed
1948/// and chunk data is dechunked on the fly, so the caller sees a
1949/// continuous stream of lines.
1950///
1951/// A line is terminated by `\n` (with or without a preceding `\r`).
1952#[cfg(not(target_arch = "wasm32"))]
1953struct LineReader<S: tokio::io::AsyncRead + Unpin> {
1954 stream: S,
1955 buf: Vec<u8>,
1956 buf_pos: usize,
1957 is_chunked: bool,
1958 // For chunked encoding: remaining bytes in the current chunk.
1959 // None means we need to read the next chunk header.
1960 chunk_remaining: Option<usize>,
1961 eof: bool,
1962}
1963
1964#[cfg(not(target_arch = "wasm32"))]
1965impl<S: tokio::io::AsyncRead + Unpin> LineReader<S> {
1966
1967 fn new(stream: S, is_chunked: bool) -> Self {
1968 Self {
1969 stream,
1970 buf: Vec::with_capacity(8192),
1971 buf_pos: 0,
1972 is_chunked,
1973 chunk_remaining: None,
1974 eof: false,
1975 }
1976 }
1977
1978 /// Read the next line (without the trailing newline).
1979 ///
1980 /// Returns `Ok(None)` at end of stream.
1981 async fn read_line(&mut self) -> std::io::Result<Option<String>> {
1982 loop {
1983 // Try to find a complete line in the buffer.
1984 if let Some(line) = self.try_extract_line() {
1985 return Ok(Some(line));
1986 }
1987 if self.eof {
1988 // If there's remaining data without a newline,
1989 // return it as the last line.
1990 if self.buf_pos < self.buf.len() {
1991 let rest = String::from_utf8_lossy(
1992 &self.buf[self.buf_pos..]
1993 ).to_string();
1994 self.buf_pos = self.buf.len();
1995 return Ok(Some(rest));
1996 }
1997 return Ok(None);
1998 }
1999 // Need more data.
2000 match self.fill_buf().await {
2001 Ok(()) => {},
2002 Err(e) => return Err(e),
2003 }
2004 }
2005 }
2006
2007 /// Try to extract a complete line from the buffer.
2008 fn try_extract_line(&mut self) -> Option<String> {
2009 let search_start = self.buf_pos;
2010 let rest = &self.buf[search_start..];
2011 if let Some(pos) = rest.iter().position(|&b| b == b'\n') {
2012 let end = search_start + pos;
2013 let line = &self.buf[self.buf_pos..end];
2014 // Strip trailing \r if present.
2015 let line = if line.ends_with(b"\r") { &line[..line.len()-1] } else { line };
2016 let s = String::from_utf8_lossy(line).to_string();
2017 self.buf_pos = end + 1; // skip the \n
2018 // Compact buffer periodically.
2019 if self.buf_pos > 16384 {
2020 self.buf.drain(..self.buf_pos);
2021 self.buf_pos = 0;
2022 }
2023 return Some(s);
2024 }
2025 None
2026 }
2027
2028 /// Read more data into the buffer.
2029 async fn fill_buf(&mut self) -> std::io::Result<()> {
2030 let mut tmp = [0u8; 4096];
2031
2032 if self.is_chunked {
2033 // For chunked encoding, we need to be careful about
2034 // chunk boundaries. However, SSE lines are always
2035 // within a single chunk in practice (servers don't
2036 // split a data: line across chunks). We read raw
2037 // bytes and handle chunk boundaries in the line
2038 // buffer. This is simpler than tracking exact chunk
2039 // positions and works because we only need lines.
2040 //
2041 // For correctness, we parse chunk headers when we
2042 // run out of chunk data.
2043 if self.chunk_remaining == Some(0) {
2044 // Read and discard the trailing \r\n after a chunk,
2045 // then read the next chunk header.
2046 let mut crlf = [0u8; 2];
2047 match self.stream.read_exact(&mut crlf).await {
2048 Ok(_) => {}
2049 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => {
2050 self.eof = true;
2051 return Ok(());
2052 }
2053 Err(e) => return Err(e),
2054 }
2055 self.chunk_remaining = None;
2056 }
2057
2058 if self.chunk_remaining.is_none() {
2059 // Read chunk size line.
2060 let mut size_line = Vec::new();
2061 let mut byte = [0u8; 1];
2062 loop {
2063 match self.stream.read(&mut byte).await {
2064 Ok(0) => { self.eof = true; return Ok(()); }
2065 Ok(_) => {
2066 size_line.push(byte[0]);
2067 if size_line.ends_with(b"\r\n") {
2068 break;
2069 }
2070 // Some servers include chunk extensions
2071 // after the size: 1a;ext=val\r\n
2072 if size_line.ends_with(b"\n") {
2073 break;
2074 }
2075 }
2076 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => {
2077 self.eof = true;
2078 return Ok(());
2079 }
2080 Err(e) => return Err(e),
2081 }
2082 }
2083 let size_str = String::from_utf8_lossy(&size_line);
2084 let size_str = size_str.trim();
2085 // Strip chunk extensions (everything after ;).
2086 let size_str = size_str.split(';').next().unwrap_or("0").trim();
2087 let size = match usize::from_str_radix(size_str, 16) {
2088 Ok(n) => n,
2089 Err(_) => { self.eof = true; return Ok(()); }
2090 };
2091 if size == 0 {
2092 // Last chunk — end of body.
2093 self.eof = true;
2094 return Ok(());
2095 }
2096 self.chunk_remaining = Some(size);
2097 }
2098
2099 // Read up to chunk_remaining bytes or tmp.len(), whichever is smaller.
2100 let remaining = match self.chunk_remaining {
2101 Some(r) => r,
2102 None => return Err(std::io::Error::new(
2103 std::io::ErrorKind::Other,
2104 "chunk_remaining unexpectedly unset")),
2105 };
2106 let to_read = remaining.min(tmp.len());
2107 match self.stream.read(&mut tmp[..to_read]).await {
2108 Ok(0) => { self.eof = true; return Ok(()); }
2109 Ok(n) => {
2110 self.buf.extend_from_slice(&tmp[..n]);
2111 self.chunk_remaining = Some(remaining - n);
2112 }
2113 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => {
2114 self.eof = true;
2115 return Ok(());
2116 }
2117 Err(e) => return Err(e),
2118 }
2119 } else {
2120 // Identity encoding — read directly.
2121 match self.stream.read(&mut tmp).await {
2122 Ok(0) => { self.eof = true; return Ok(()); }
2123 Ok(n) => self.buf.extend_from_slice(&tmp[..n]),
2124 Err(e) if e.kind() == tokio::io::ErrorKind::UnexpectedEof => {
2125 self.eof = true;
2126 return Ok(());
2127 }
2128 Err(e) => return Err(e),
2129 }
2130 }
2131 Ok(())
2132 }
2133}
2134
2135/// Parse an SSE response body, calling `on_token` for each text delta.
2136///
2137/// SSE format:
2138/// ```text
2139/// data: {"choices":[{"delta":{"content":"Hello"}}]}
2140///
2141/// data: {"choices":[{"delta":{"content":" world"}}]}
2142///
2143/// data: [DONE]
2144/// ```
2145///
2146/// We scan for `"content":"..."` in each `data:` line. This is a
2147/// deliberately simple parser — it handles the common case without
2148/// needing a full JSON parser. Escaped quotes inside content are
2149/// handled by scanning for the matching unescaped quote.
2150pub fn parse_sse_stream(body: &[u8], on_token: &mut impl FnMut(&str))
2151 -> (String, Usage)
2152{
2153 let text = String::from_utf8_lossy(body);
2154 let mut full = String::new();
2155 let mut use_ = Usage::default();
2156
2157 for line in text.lines() {
2158 let line = line.trim();
2159 if !line.starts_with("data: ") {
2160 continue;
2161 }
2162 let data = &line[6..];
2163 if data == "[DONE]" {
2164 break;
2165 }
2166 // Extract content from: {"choices":[{"delta":{"content":"..."}}]}
2167 if let Some(content) = extract_json_string(data, "content") {
2168 on_token(&content);
2169 full.push_str(&content);
2170 }
2171 // Extract usage from the final chunk:
2172 // {"choices":[],"usage":{"prompt_tokens":13,"completion_tokens":200}}
2173 if let Some(u) = parse_usage(data) {
2174 use_ = u;
2175 }
2176 }
2177
2178 (full, use_)
2179}
2180
2181/// Read a `usage` object out of a whole response body or one SSE chunk,
2182/// returning `None` when the chunk carries none.
2183///
2184/// Intermediate streamed chunks send `"usage":null`, which is not an object
2185/// and so reads as absent rather than as a zeroed usage -- otherwise the last
2186/// chunk before `[DONE]` would erase what the usage chunk reported.
2187///
2188/// # Arguments
2189/// * `json` - A response body, or one SSE `data:` payload.
2190pub(crate) fn parse_usage(json: &str) -> Option<Usage> {
2191 let usage = match find_json_object(json, "usage") {
2192 Some(u) => u,
2193 None => return None,
2194 };
2195 let mut u = Usage::default();
2196 if let Some(p) = extract_json_number(&usage, "prompt_tokens") { u.prompt = p; }
2197 if let Some(c) = extract_json_number(&usage, "completion_tokens") { u.completion = c; }
2198 // Cache reads live in a nested `prompt_tokens_details`; a provider that
2199 // flattens the field is read too, so neither shape is missed. A cache read
2200 // bills at a fraction of a fresh prompt token, and in an agentic tool loop
2201 // -- where every round's prompt is the last round's plus a little -- it is
2202 // most of the prompt, so counting it at the full input rate was the single
2203 // largest source of overstatement.
2204 // Three spellings, because three providers report the same figure three
2205 // ways: the OpenAI-compatible nesting, a flattened copy of it, and
2206 // Anthropic's own `cache_read_input_tokens`. Missing the last one would
2207 // read a working prompt cache as no cache at all.
2208 u.cached = match find_json_object(&usage, "prompt_tokens_details") {
2209 Some(d) => extract_json_number(&d, "cached_tokens").unwrap_or(0),
2210 None => extract_json_number(&usage, "cached_tokens")
2211 .or_else(|| extract_json_number(&usage, "cache_read_input_tokens"))
2212 .unwrap_or(0),
2213 };
2214 // What the provider actually drew. This is money, not an estimate, and it
2215 // supersedes anything the price table would have guessed.
2216 if let Some(c) = extract_json_f64(&usage, "cost") { u.cost_usd = c; }
2217 Some(u)
2218}
2219
2220/// Extract a JSON object value for a key from a JSON string.
2221///
2222/// Scans for `"key":{...}` and returns the inner object string
2223/// (including the braces). Used to extract the `usage` object
2224/// from the final SSE chunk.
2225fn find_json_object(json: &str, key: &str) -> Option<String> {
2226 let needle = fmt!("\"{}\":", key);
2227 let pos = match json.find(&needle) {
2228 Some(p) => p,
2229 None => return None,
2230 };
2231 let bytes = json.as_bytes();
2232 // Skip whitespace after the colon to the opening brace.
2233 let mut start = pos + needle.len();
2234 while start < bytes.len() && bytes[start].is_ascii_whitespace() { start += 1; }
2235 if start >= bytes.len() || bytes[start] != b'{' { return None; }
2236 let mut depth = 0i32;
2237 let mut i = start;
2238 while i < bytes.len() {
2239 match bytes[i] {
2240 b'{' => depth += 1,
2241 b'}' => {
2242 depth -= 1;
2243 if depth == 0 {
2244 return Some(json[start..=i].to_string());
2245 }
2246 }
2247 b'"' => {
2248 // Skip string contents.
2249 i += 1;
2250 while i < bytes.len() {
2251 if bytes[i] == b'\\' { i += 2; continue; }
2252 if bytes[i] == b'"' { break; }
2253 i += 1;
2254 }
2255 }
2256 _ => (),
2257 }
2258 i += 1;
2259 }
2260 None
2261}
2262
2263/// Extract a numeric value for a key from a JSON string.
2264///
2265/// Scans for `"key":number` and returns the parsed value.
2266pub(crate) fn extract_json_number(json: &str, key: &str) -> Option<u64> {
2267 let needle = fmt!("\"{}\":", key);
2268 let pos = match json.find(&needle) {
2269 Some(p) => p,
2270 None => return None,
2271 };
2272 let mut start = pos + needle.len();
2273 let bytes = json.as_bytes();
2274 // Skip whitespace.
2275 while start < bytes.len() && bytes[start].is_ascii_whitespace() {
2276 start += 1;
2277 }
2278 let mut end = start;
2279 while end < bytes.len() && (bytes[end].is_ascii_digit() || bytes[end] == b'-') {
2280 end += 1;
2281 }
2282 json[start..end].parse::<u64>().ok()
2283}
2284
2285/// Extract a SIGNED integer value for a key from a JSON string.
2286///
2287/// [`extract_json_number`] parses a `u64`, so a negative number does not merely come back wrong --
2288/// it comes back as `None`, and every caller that reached for `unwrap_or(0)` then read a negative
2289/// value as zero. For a process exit status that is the difference between "the command was
2290/// killed" and "the command succeeded", so the signed reader exists separately rather than as a
2291/// cast at the call site.
2292///
2293/// # Arguments
2294/// * `json` - The JSON text to read.
2295/// * `key` - The key whose value is wanted.
2296pub(crate) fn extract_json_i64(json: &str, key: &str) -> Option<i64> {
2297 let needle = fmt!("\"{}\":", key);
2298 let pos = match json.find(&needle) {
2299 Some(p) => p,
2300 None => return None,
2301 };
2302 let mut start = pos + needle.len();
2303 let bytes = json.as_bytes();
2304 while start < bytes.len() && bytes[start].is_ascii_whitespace() {
2305 start += 1;
2306 }
2307 let mut end = start;
2308 while end < bytes.len() && (bytes[end].is_ascii_digit() || bytes[end] == b'-') {
2309 end += 1;
2310 }
2311 json[start..end].parse::<i64>().ok()
2312}
2313
2314/// Extract a fractional numeric value for a key from a JSON string.
2315///
2316/// [`extract_json_number`] stops at the first non-digit, so it reads `0.0021`
2317/// as `0` -- which silently priced every reported cost at nothing. This scans
2318/// the whole JSON number grammar: sign, digits, decimal point and exponent.
2319///
2320/// # Arguments
2321/// * `json` - The JSON text to scan.
2322/// * `key` - The key whose value is wanted.
2323pub(crate) fn extract_json_f64(json: &str, key: &str) -> Option<f64> {
2324 let needle = fmt!("\"{}\":", key);
2325 let pos = match json.find(&needle) {
2326 Some(p) => p,
2327 None => return None,
2328 };
2329 let bytes = json.as_bytes();
2330 let mut start = pos + needle.len();
2331 // Skip whitespace, and an opening quote for a provider that sends the
2332 // figure as a string.
2333 while start < bytes.len() && bytes[start].is_ascii_whitespace() {
2334 start += 1;
2335 }
2336 if start < bytes.len() && bytes[start] == b'"' {
2337 start += 1;
2338 }
2339 let mut end = start;
2340 while end < bytes.len() {
2341 let b = bytes[end];
2342 let numeric = b.is_ascii_digit()
2343 || b == b'.'
2344 || b == b'-'
2345 || b == b'+'
2346 || b == b'e'
2347 || b == b'E';
2348 if !numeric { break; }
2349 end += 1;
2350 }
2351 json[start..end].parse::<f64>().ok()
2352}
2353
2354/// Extract a boolean value for a key from a JSON string.
2355///
2356/// Scans for `"key":true`/`false` and accepts the quoted forms too, since
2357/// models routinely send a boolean argument as the string `"true"`.
2358pub fn extract_json_bool(json: &str, key: &str) -> Option<bool> {
2359 let needle = fmt!("\"{}\":", key);
2360 let pos = match json.find(&needle) {
2361 Some(p) => p,
2362 None => return None,
2363 };
2364 let bytes = json.as_bytes();
2365 let mut i = pos + needle.len();
2366 // Skip whitespace, then an optional opening quote.
2367 while i < bytes.len() && bytes[i].is_ascii_whitespace() {
2368 i += 1;
2369 }
2370 if i < bytes.len() && bytes[i] == b'"' {
2371 i += 1;
2372 }
2373 let rest = &json[i..];
2374 if rest.starts_with("true") {
2375 Some(true)
2376 } else if rest.starts_with("false") {
2377 Some(false)
2378 } else {
2379 None
2380 }
2381}
2382
2383/// Extract an array of strings for a key from a JSON string.
2384///
2385/// `None` when the key is absent or its value is not an array, which is what
2386/// lets a caller tell a field that was never written from one written empty.
2387pub(crate) fn extract_json_string_array(json: &str, key: &str) -> Option<Vec<String>> {
2388 let arr = match find_json_array(json, key) {
2389 Some(a) => a,
2390 None => return None,
2391 };
2392 Some(parse_json_string_array(&arr))
2393}
2394
2395/// Extract an array of objects for a key from a JSON string, each as its own text.
2396///
2397/// The sibling of [`extract_json_string_array`] for the shape a tool argument takes when one call
2398/// carries several of a thing -- `"edits":[{...},{...}]`. Each element comes back whole, for
2399/// [`extract_json_string`] and its siblings to read the fields out of.
2400///
2401/// `None` when the key is absent or its value is not an array, which is what lets a caller tell a
2402/// field that was never written from one written empty.
2403pub(crate) fn extract_json_objects(json: &str, key: &str) -> Option<Vec<String>> {
2404 find_json_array(json, key).map(|arr| split_top_level_objects(&arr))
2405}
2406
2407/// Parse a JSON array's text into its string elements, ignoring any element
2408/// that is not a string.
2409///
2410/// Handles the escapes [`json_escape`] emits, `\uXXXX` among them, so a value
2411/// survives the round trip out to storage and back.
2412pub(crate) fn parse_json_string_array(arr: &str) -> Vec<String> {
2413 let bytes = arr.as_bytes();
2414 let mut out: Vec<String> = Vec::new();
2415 let mut i = 0;
2416 while i < bytes.len() {
2417 // Anything outside a quoted element -- brackets, commas, a number --
2418 // is not a string, and is stepped over.
2419 if bytes[i] != b'"' {
2420 i += 1;
2421 continue;
2422 }
2423 i += 1; // past the opening quote
2424 // Collect as bytes, then decode as UTF-8, so multi-byte characters survive.
2425 let mut buf: Vec<u8> = Vec::new();
2426 while i < bytes.len() {
2427 let b = bytes[i];
2428 if b == b'\\' && i + 1 < bytes.len() {
2429 match bytes[i + 1] {
2430 b'"' => { buf.push(b'"'); i += 2; }
2431 b'\\' => { buf.push(b'\\'); i += 2; }
2432 b'n' => { buf.push(b'\n'); i += 2; }
2433 b't' => { buf.push(b'\t'); i += 2; }
2434 b'r' => { buf.push(b'\r'); i += 2; }
2435 b'/' => { buf.push(b'/'); i += 2; }
2436 b'b' => { buf.push(0x08); i += 2; }
2437 b'f' => { buf.push(0x0c); i += 2; }
2438 b'u' => match decode_json_unicode(bytes, i + 2) {
2439 Some((c, next)) => {
2440 let mut enc = [0u8; 4];
2441 buf.extend_from_slice(c.encode_utf8(&mut enc).as_bytes());
2442 i = next;
2443 }
2444 // Not a well-formed escape; keep it as written rather
2445 // than lose the characters.
2446 None => { buf.push(b'\\'); buf.push(b'u'); i += 2; }
2447 },
2448 other => { buf.push(b'\\'); buf.push(other); i += 2; }
2449 }
2450 } else if b == b'"' {
2451 i += 1; // past the closing quote
2452 break;
2453 } else {
2454 buf.push(b);
2455 i += 1;
2456 }
2457 }
2458 out.push(String::from_utf8_lossy(&buf).to_string());
2459 }
2460 out
2461}
2462
2463/// Decode a `\uXXXX` escape whose first hex digit is at `i`, pairing a leading
2464/// surrogate with the trailing one that follows it.
2465///
2466/// Returns the character and the index just past the escape, or `None` when the
2467/// escape is malformed or a surrogate is left unpaired.
2468fn decode_json_unicode(bytes: &[u8], i: usize) -> Option<(char, usize)> {
2469 // Four hex digits at `s`, as a code unit.
2470 let unit = |s: usize| -> Option<u32> {
2471 if s + 4 > bytes.len() {
2472 return None;
2473 }
2474 match std::str::from_utf8(&bytes[s..s + 4]) {
2475 Ok(txt) => u32::from_str_radix(txt, 16).ok(),
2476 Err(_) => None,
2477 }
2478 };
2479 let first = match unit(i) {
2480 Some(v) => v,
2481 None => return None,
2482 };
2483 // A leading surrogate is only half a character: its pair follows as a
2484 // second `\uXXXX`, and the two combine into one code point.
2485 if (0xD800..0xDC00).contains(&first) {
2486 let j = i + 4;
2487 if j + 6 <= bytes.len() && bytes[j] == b'\\' && bytes[j + 1] == b'u' {
2488 if let Some(second) = unit(j + 2) {
2489 if (0xDC00..0xE000).contains(&second) {
2490 let cp = 0x10000 + ((first - 0xD800) << 10) + (second - 0xDC00);
2491 return char::from_u32(cp).map(|c| (c, j + 6));
2492 }
2493 }
2494 }
2495 return None;
2496 }
2497 char::from_u32(first).map(|c| (c, i + 4))
2498}
2499
2500
2501/// Handles `\"`, `\\`, `\n`, `\t` escapes.
2502///
2503/// The search ensures `key` is a complete JSON key, not a suffix of
2504/// a longer key (e.g. `"content"` must not match inside
2505/// `"reasoning_content"`). This is done by requiring the character
2506/// before the opening quote to be `{` or `,` (whitespace-tolerant).
2507pub(crate) fn extract_json_string(json: &str, key: &str) -> Option<String> {
2508 let needle = fmt!("\"{}\":", key);
2509 let bytes = json.as_bytes();
2510 let mut search_from = 0;
2511 loop {
2512 let pos = match json[search_from..].find(&needle) {
2513 Some(p) => search_from + p,
2514 None => return None,
2515 };
2516 // Reject suffix matches (e.g. "content" inside
2517 // "reasoning_content") by checking the character before the
2518 // key's opening quote.
2519 let valid_prefix = pos == 0 || {
2520 let prev = bytes[pos - 1];
2521 prev == b'{' || prev == b',' || prev.is_ascii_whitespace()
2522 };
2523 if !valid_prefix {
2524 search_from = pos + needle.len();
2525 continue;
2526 }
2527 // Skip whitespace between the colon and the value — real API
2528 // output uses `"key": "value"` with a space.
2529 let mut i = pos + needle.len();
2530 while i < bytes.len() && bytes[i].is_ascii_whitespace() { i += 1; }
2531 if i >= bytes.len() || bytes[i] != b'"' {
2532 // Value is not a string (null / number / object); keep
2533 // searching in case the key appears again.
2534 search_from = pos + needle.len();
2535 continue;
2536 }
2537 i += 1; // past the opening quote
2538 // Collect the string value as bytes, then decode as UTF-8, so
2539 // multi-byte characters survive.
2540 let mut out: Vec<u8> = Vec::new();
2541 while i < bytes.len() {
2542 let b = bytes[i];
2543 if b == b'\\' && i + 1 < bytes.len() {
2544 match bytes[i + 1] {
2545 b'"' => out.push(b'"'),
2546 b'\\' => out.push(b'\\'),
2547 b'n' => out.push(b'\n'),
2548 b't' => out.push(b'\t'),
2549 b'r' => out.push(b'\r'),
2550 b'/' => out.push(b'/'),
2551 other => { out.push(b'\\'); out.push(other); }
2552 }
2553 i += 2;
2554 } else if b == b'"' {
2555 return Some(String::from_utf8_lossy(&out).to_string());
2556 } else {
2557 out.push(b);
2558 i += 1;
2559 }
2560 }
2561 return None;
2562 }
2563}
2564
2565/// Convert a JDAT DaticleMap to a minimal JSON string.
2566///
2567/// This is used to build the LLM API request body without `serde`.
2568/// Only handles the types we need: String, U64, Bool, Map, List.
2569/// The shortest prefix worth a cache breakpoint, in characters.
2570///
2571/// Anthropic will not cache a prefix below a per-model minimum, and says nothing
2572/// when it declines -- the request simply reports no cache write. 512 tokens is
2573/// the lowest of those minimums (Claude Opus 5; other models want 1024 or more),
2574/// and four characters per token is the usual rough conversion, so a prefix
2575/// shorter than this cannot cache on any model and the marker is not sent.
2576pub(crate) const CACHE_MIN_PREFIX_CHARS: usize = 2048;
2577
2578/// Whether this model honours an explicit `cache_control` breakpoint.
2579///
2580/// Claude is the case that needs one: Fireworks, DeepSeek and OpenAI cache
2581/// automatically, and Anthropic does not. The model id is what selects the
2582/// upstream model -- the host varies (direct, a router, Daimond's own gateway
2583/// proxy) while the same Claude model behind any of them reads the same marker,
2584/// so the id is what this gates on. Every id form is covered: the OpenRouter
2585/// `anthropic/claude-...`, the Bedrock `anthropic.claude-...`, and the bare
2586/// `claude-...`.
2587pub(crate) fn model_caches_on_request(model: &str) -> bool {
2588 let m = model.to_ascii_lowercase();
2589 m.contains("claude") || m.starts_with("anthropic/") || m.starts_with("anthropic.")
2590}
2591
2592/// The Anthropic API version this client is written against.
2593///
2594/// Pinned rather than tracking latest: the version header is what stops a
2595/// breaking change to the wire shape arriving without a code change.
2596pub(crate) const ANTHROPIC_VERSION: &str = "2023-06-01";
2597
2598/// The smallest output cap a streamed thinking request is sent with.
2599///
2600/// Well under the 128k ceiling every thinking-capable model offers, and enough
2601/// room for the reasoning and the answer that follows it. See
2602/// [`anthropic_max_tokens`](LlmClient::anthropic_max_tokens).
2603pub(crate) const THINKING_MIN_MAX_TOKENS: u32 = 32_000;
2604
2605/// An Anthropic `text` content block, optionally carrying a cache breakpoint.
2606///
2607/// # Arguments
2608/// * `text` - The block's text, escaped here.
2609/// * `cached` - Whether to attach an ephemeral `cache_control` marker.
2610fn text_block(text: &str, cached: bool) -> String {
2611 if cached {
2612 fmt!("{{\"type\":\"text\",\"text\":\"{}\",\"cache_control\":{{\"type\":\"ephemeral\"}}}}",
2613 json_escape(text))
2614 } else {
2615 fmt!("{{\"type\":\"text\",\"text\":\"{}\"}}", json_escape(text))
2616 }
2617}
2618
2619/// An Anthropic `image` content block, base64 source.
2620///
2621/// The shape is the one the Messages API publishes:
2622/// `{"type":"image","source":{"type":"base64","media_type":…,"data":…}}`. Key order matters to
2623/// nothing but the fixture that pins it, and it is the documentation's order.
2624///
2625/// A cache breakpoint may sit on an image block as on any other, and it has to be able to: the
2626/// breakpoint caches everything up to the block it is on, so a message whose last block is the
2627/// image would otherwise have no legal place to put one and would silently lose the cache.
2628///
2629/// # Arguments
2630/// * `img` - The image; its bytes are base64-encoded here, once per request.
2631/// * `cached` - Whether to attach an ephemeral `cache_control` marker.
2632fn image_block(img: &ImagePart, cached: bool) -> String {
2633 let mark = if cached { ",\"cache_control\":{\"type\":\"ephemeral\"}" } else { "" };
2634 fmt!(
2635 "{{\"type\":\"image\",\"source\":{{\"type\":\"base64\",\"media_type\":\"{}\",\
2636 \"data\":\"{}\"}}{}}}",
2637 img.media.mime(), img.base64(), mark)
2638}
2639
2640/// A message's content as Anthropic content blocks, in order.
2641///
2642/// The cache marker goes on the LAST block, because a breakpoint caches the prefix up to and
2643/// including the block it sits on -- putting it on the first of several would leave the rest of
2644/// the message re-billed on every turn, which is the opposite of what marking it was for.
2645///
2646/// # Arguments
2647/// * `content` - What the message carries.
2648/// * `cached` - Whether this message is a cache breakpoint.
2649fn anthropic_blocks(content: &MessageContent, cached: bool) -> Vec<String> {
2650 let parts: Vec<&ContentPart> = match content {
2651 MessageContent::Text(s) => {
2652 // An empty text block is rejected outright, where the OpenAI side simply carries the
2653 // empty string through.
2654 return if s.is_empty() { Vec::new() } else { vec![text_block(s, cached)] };
2655 },
2656 MessageContent::Parts(parts) => parts.iter().collect(),
2657 };
2658 let last = parts.len().saturating_sub(1);
2659 let mut out = Vec::with_capacity(parts.len());
2660 for (i, p) in parts.iter().enumerate() {
2661 match p {
2662 ContentPart::Text(t) if t.is_empty() => {},
2663 ContentPart::Text(t) => out.push(text_block(t, cached && i == last)),
2664 ContentPart::Image(m) => out.push(image_block(m, cached && i == last)),
2665 }
2666 }
2667 out
2668}
2669
2670/// The `user` message that carries images lifted out of a run of OpenAI tool replies.
2671///
2672/// The leading sentence is not decoration: without it the model receives images with no statement
2673/// of where they came from, and the turn reads as the user having pasted them.
2674///
2675/// # Arguments
2676/// * `blocks` - Ready-made `image_url` parts, in the order the tools returned them.
2677fn tool_image_message(blocks: &[String]) -> String {
2678 fmt!(
2679 "{{\"role\":\"user\",\"content\":[{{\"type\":\"text\",\"text\":\"{}\"}},{}]}}",
2680 json_escape("[The images returned by the tool calls above, in order.]"),
2681 blocks.join(","))
2682}
2683
2684/// A message's content as the OpenAI `content` field: a JSON string, or an array of parts.
2685///
2686/// A bare string whenever there is no image, because that is what every OpenAI-compatible router
2687/// has always been sent and the array form buys nothing. With an image it becomes the documented
2688/// parts array, where an image is `{"type":"image_url","image_url":{"url":"data:…;base64,…"}}` --
2689/// the `url` field takes "a URL or a base64 encoded data URL", so the bytes ride in an RFC 2397
2690/// data URL rather than in a field of their own.
2691///
2692/// # Arguments
2693/// * `content` - What the message carries.
2694fn openai_content(content: &MessageContent) -> String {
2695 match content {
2696 MessageContent::Text(s) => fmt!("\"{}\"", json_escape(s)),
2697 MessageContent::Parts(parts) => {
2698 let items: Vec<String> = parts.iter().map(|p| match p {
2699 ContentPart::Text(t) =>
2700 fmt!("{{\"type\":\"text\",\"text\":\"{}\"}}", json_escape(t)),
2701 ContentPart::Image(m) => fmt!(
2702 "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:{};base64,{}\"}}}}",
2703 m.media.mime(), m.base64()),
2704 }).collect();
2705 fmt!("[{}]", items.join(","))
2706 },
2707 }
2708}
2709
2710/// Whether `model` takes Anthropic's adaptive thinking configuration.
2711///
2712/// Adaptive is the only form worth sending: `budget_tokens` is removed on every
2713/// model released since Opus 4.7 and returns a 400 there, and deprecated on the
2714/// two before it. So a model that does not take adaptive is sent no thinking
2715/// configuration at all rather than a guessed budget -- an older model then
2716/// simply answers without thinking, which is what it did before this existed.
2717///
2718/// The list is explicit rather than a `claude-` prefix test, because the whole
2719/// point of the gate is that the newer families and the older ones disagree
2720/// about what the parameter even means.
2721///
2722/// # Arguments
2723/// * `model` - The model id, in any of the forms a caller can configure.
2724pub(crate) fn model_takes_adaptive_thinking(model: &str) -> bool {
2725 let m = model.to_ascii_lowercase();
2726 const ADAPTIVE: [&str; 8] = [
2727 "claude-fable-5",
2728 "claude-mythos-5",
2729 "claude-opus-5",
2730 "claude-opus-4-8",
2731 "claude-opus-4-7",
2732 "claude-opus-4-6",
2733 "claude-sonnet-5",
2734 "claude-sonnet-4-6",
2735 ];
2736 ADAPTIVE.iter().any(|id| m.contains(id))
2737}
2738
2739/// Whether `model` can be shown an image.
2740///
2741/// The test is a list of what is KNOWN BLIND, not a list of what is known to see, and the default
2742/// is that a model sees. That direction is chosen deliberately: nearly every model a user would
2743/// configure today is multimodal, an allow-list would refuse every model released after this line
2744/// was written, and the cost of getting it wrong in this direction is one clear error from
2745/// [`LlmClient::vision_error`] rather than a refusal to try. The names are matched as substrings
2746/// because a router spells the same model half a dozen ways (`openai/gpt-3.5-turbo`,
2747/// `gpt-3.5-turbo-0125`), and the family is what is blind, not the spelling.
2748///
2749/// # Arguments
2750/// * `model` - The model id, in whatever form the user configured it.
2751pub(crate) fn model_can_see(model: &str) -> bool {
2752 let m = model.to_ascii_lowercase();
2753 const BLIND: [&str; 8] = [
2754 "gpt-3.5",
2755 "text-davinci",
2756 "o1-mini",
2757 "o1-preview",
2758 "claude-instant",
2759 "claude-1",
2760 "claude-2",
2761 "embedding",
2762 ];
2763 !BLIND.iter().any(|id| m.contains(id))
2764}
2765
2766/// Translate an OpenAI-shaped tool array into the Anthropic one.
2767///
2768/// `[{"type":"function","function":{"name":…,"description":…,"parameters":{…}}}]`
2769/// becomes `[{"name":…,"description":…,"input_schema":{…}}]`. The schema
2770/// itself is JSON Schema in both, so it is carried through verbatim; only the
2771/// wrapper differs. A definition missing a name or a schema is dropped rather
2772/// than sent half-built, since the API would reject the whole request for it.
2773///
2774/// # Arguments
2775/// * `tools` - The OpenAI-shaped tool array, as JSON text.
2776fn openai_tools_to_anthropic(tools: &str) -> String {
2777 let mut out: Vec<String> = Vec::new();
2778 for elem in split_top_level_objects(tools) {
2779 // The function object, so `name` and `description` are read from it
2780 // rather than from a property of the schema that happens to share a key.
2781 let f = match find_json_object(&elem, "function") {
2782 Some(f) => f,
2783 None => elem.clone(),
2784 };
2785 let name = match extract_json_string(&f, "name") {
2786 Some(n) if !n.is_empty() => n,
2787 _ => continue,
2788 };
2789 let schema = match find_json_object(&f, "parameters")
2790 .or_else(|| find_json_object(&f, "input_schema"))
2791 {
2792 Some(s) => s,
2793 None => continue,
2794 };
2795 let desc = extract_json_string(&f, "description").unwrap_or_default();
2796 out.push(fmt!(
2797 "{{\"name\":\"{}\",\"description\":\"{}\",\"input_schema\":{}}}",
2798 json_escape(&name), json_escape(&desc), schema));
2799 }
2800 fmt!("[{}]", out.join(","))
2801}
2802
2803/// The character length of a message's payload, as a stand-in for its tokens.
2804///
2805/// An image counts its own bytes here rather than its token cost, because this figure decides
2806/// only WHERE the prompt-cache breakpoints go, and what matters for that is what the message
2807/// weighs on the wire -- which for an image is its bytes.
2808fn message_len(msg: &ChatMessage) -> usize {
2809 let content = msg.content();
2810 let mut n = content.text_len() + content.images().map(|i| i.data.len()).sum::<usize>();
2811 if let ChatMessage::Assistant { tool_calls, .. } = msg {
2812 n += tool_calls.iter().map(|tc| tc.name.len() + tc.arguments.len()).sum::<usize>();
2813 }
2814 n
2815}
2816
2817/// Serialise a `ChatMessage` with an Anthropic prompt-cache breakpoint on it.
2818///
2819/// The marker only exists on a content *block*, so the content becomes a
2820/// one-element array rather than a bare string. Only the system and user roles
2821/// are given this form; anything else falls back to the plain serialisation, so
2822/// a caller that marks the wrong message loses the cache rather than the turn.
2823fn message_to_json_cached(msg: &ChatMessage, open: &std::collections::HashSet<String>) -> String {
2824 let (role, content) = match msg {
2825 ChatMessage::System { content } => ("system", content),
2826 ChatMessage::User { content } => ("user", content),
2827 _ => return message_to_json(msg, open),
2828 };
2829 // With an image in it the content is already an array, and the marker goes on the last block
2830 // rather than replacing the whole thing with one text block -- which would drop the image.
2831 //
2832 // On this side the marker goes on the last TEXT part and nowhere else. `cache_control` is an
2833 // Anthropic field that routers pass through; putting it inside an `image_url` part would put
2834 // an unrecognised key somewhere every OpenAI-compatible server validates strictly, to buy a
2835 // cache hit on a request that is mostly image bytes anyway. A message ending in an image
2836 // simply is not a breakpoint.
2837 if let MessageContent::Parts(parts) = content {
2838 let last = parts.iter().rposition(|p| matches!(p, ContentPart::Text(_)))
2839 .unwrap_or(usize::MAX);
2840 let items: Vec<String> = parts.iter().enumerate().map(|(i, p)| match p {
2841 ContentPart::Text(t) if i == last => fmt!(
2842 "{{\"type\":\"text\",\"text\":\"{}\",\"cache_control\":{{\"type\":\"ephemeral\"}}}}",
2843 json_escape(t)),
2844 ContentPart::Text(t) => fmt!("{{\"type\":\"text\",\"text\":\"{}\"}}", json_escape(t)),
2845 ContentPart::Image(m) => fmt!(
2846 "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:{};base64,{}\"}}}}",
2847 m.media.mime(), m.base64()),
2848 }).collect();
2849 return fmt!("{{\"role\":\"{}\",\"content\":[{}]}}", role, items.join(","));
2850 }
2851 fmt!(
2852 "{{\"role\":\"{}\",\"content\":[{{\"type\":\"text\",\"text\":\"{}\",\
2853 \"cache_control\":{{\"type\":\"ephemeral\"}}}}]}}",
2854 role, json_escape(&content.as_text()))
2855}
2856
2857/// The arguments a `say` call is REPLAYED with, or `None` for every other tool.
2858///
2859/// **`say` was a tool and is not one any more.** Folding is written into the model's own prose as
2860/// a `<details>` element now -- see [`sent_text_len`] -- and no model is offered `say` or can call
2861/// it. This stays because STORED CONVERSATIONS still carry `say` tool_calls: every one of them
2862/// travels on every request for the life of that conversation, so a stripper deleted here is not a
2863/// tidying, it is every old folded answer going out in full again, on exactly the conversations
2864/// the feature existed to make cheap.
2865///
2866/// `say` answered at two depths: a summary the user read at once and a detail behind a fold. The
2867/// detail is for a person, once. Left in the transcript it would be re-sent on every later request
2868/// for the life of the conversation -- so a model that explained at length would charge for that
2869/// explanation again on every turn, whether or not anybody looked at it twice.
2870///
2871/// So the wire carries the summary and a note in the detail's place. Nothing is lost: the browser
2872/// keeps the whole call in its own transcript record, which is what the fold opens and what
2873/// survives a reload. The local record and the payload simply stop being the same thing.
2874///
2875/// ONE FUNCTION FOR BOTH DIALECTS. The OpenAI body escapes these arguments into a string and the
2876/// Anthropic body embeds them as JSON, so the two sites look nothing alike -- and a rule applied at
2877/// one and not the other would mean the same conversation cost different amounts through different
2878/// endpoints, silently.
2879///
2880/// It costs ONE cache miss, at the request after the call: the prefix changes once where the
2881/// arguments shrink, and is stable from then on.
2882///
2883/// # Arguments
2884/// * `name` - The tool the call names.
2885/// * `arguments` - Its arguments, as the model wrote them.
2886fn strip_said(name: &str, arguments: &str, open: bool) -> Option<String> {
2887 if name != "say" {
2888 return None;
2889 }
2890 // OPEN MEANS THE USER IS READING IT, so the model holds it too. Their own gesture decides the
2891 // working set, and the two things that ought to agree -- what is on their screen and what it
2892 // knows -- do.
2893 if open {
2894 return None;
2895 }
2896 // A call whose summary cannot be read is left alone. It is malformed, and rewriting a
2897 // malformed call would replace one problem the model can see with one it cannot.
2898 let summary = extract_json_string(arguments, "summary")?;
2899 let n = extract_json_string(arguments, "detail")
2900 .map(|d| d.chars().count())
2901 .unwrap_or(0);
2902 Some(fmt!(
2903 "{{\"summary\":\"{}\",\"detail\":\"{}\"}}",
2904 json_escape(&summary),
2905 json_escape(&fmt!(
2906 "[folded to the user, {} characters. They have it on screen; you no longer carry it. \
2907 Ask them, or read the file you wrote it from, if you need it again.]", n))))
2908}
2909
2910/// How many bytes of `arguments` this call will actually put on the wire.
2911///
2912/// **The compaction trigger's half of [`strip_said`], and it calls that function rather than
2913/// restating its rule.** [`crate::agent::compact::msg_bytes`] sized a `say` with the length the
2914/// model wrote, so the trigger measured a conversation nobody was going to send: every closed
2915/// fold in it was counted at full length, the budget was spent on bytes that leave at
2916/// serialisation, and a conversation was folded earlier than it needed to be. The
2917/// [`Gauge`](crate::agent::compact::Gauge) absorbed part of the error by recalibrating tokens-per-byte against the provider's real
2918/// `prompt_tokens` -- but that ratio is one number for the whole conversation, so the correction
2919/// was paid for by every other message's estimate.
2920///
2921/// Whether the detail travels depends on the fold state, which is why `open` is asked for here
2922/// and not decided here: an OPEN fold is one the user is reading and its detail goes out in full.
2923///
2924/// # Arguments
2925/// * `name` - The tool the call names.
2926/// * `arguments` - Its arguments, as the model wrote them.
2927/// * `open` - Whether this call's fold is open on screen.
2928pub fn sent_args_len(name: &str, arguments: &str, open: bool) -> usize {
2929 match strip_said(name, arguments, open) {
2930 Some(replayed) => replayed.len(),
2931 None => arguments.len(),
2932 }
2933}
2934
2935
2936// ┌───────────────────────────────────────────────────────────────┐
2937// │ The two-depth answer, written inline │
2938// └───────────────────────────────────────────────────────────────┘
2939//
2940// `say` answered at two depths through a tool call. The model now writes the same two depths
2941// into its own prose as a `<details>` element, and this is the reading half of it: the same
2942// economy applied to text rather than to tool arguments. `strip_said` above stays exactly as it
2943// is -- stored conversations carry `say` tool_logs, and a reader deleted is an answer that
2944// renders as nothing. The shared shape is pinned in `dev/CONTRACT_FOLD.md`; the fixture both
2945// languages are tested against is `dev/fixtures/fold_keys.json`.
2946
2947/// The note a stripped fold leaves in place of its body.
2948///
2949/// [`strip_said`]'s wording, near enough, so the two paths read alike to a model that may hold
2950/// both in one conversation. The exact string is pinned by the fixture.
2951///
2952/// # Arguments
2953/// * `n` - Characters of the fold's trimmed body, as the user still has it on screen.
2954fn fold_note(n: usize) -> String {
2955 fmt!("[folded to the user, {} characters. They have it on screen; you no longer carry it. \
2956 Ask them if you need it again.]", n)
2957}
2958
2959/// One `<details>` fold found in one assistant message's text.
2960///
2961/// Byte offsets rather than copied strings, so the strip rewrites in place and every character
2962/// outside the body stays exactly where the model put it -- including the blank lines the
2963/// renderer needs, which a reconstructed element would have to get right a second time.
2964struct Fold {
2965 ord: usize, // 0-based over the real folds of this message
2966 summary: String, // tags removed, whitespace runs collapsed to one space
2967 body: std::ops::Range<usize>, // the TRIMMED body, as byte offsets into the text
2968 chars: usize, // characters of that trimmed body
2969 closed: bool, // whether a `</details>` was found for it
2970}
2971
2972impl Fold {
2973 /// The one name the Rust and JS halves must agree on: `"<ordinal>:<summary>"`.
2974 ///
2975 /// No hash and no message identity, deliberately -- see `dev/CONTRACT_FOLD.md` §2. Two
2976 /// messages can therefore share a key, and the only consequence is that opening one fold
2977 /// sends the body of a same-labelled, same-ordinal fold in another. Nothing renders wrong.
2978 fn key(&self) -> String {
2979 fmt!("{}:{}", self.ord, self.summary)
2980 }
2981}
2982
2983/// The fence a line opens or closes, as `(character, run length, whether an info string follows)`.
2984///
2985/// CommonMark's rule, to the part that matters here: up to three leading spaces, then three or
2986/// more backticks or tildes. A run with an info string after it can only OPEN a fence, never
2987/// close one, which is what keeps ```` ```html ```` from closing the fence it opened.
2988fn fence_run(line: &str) -> Option<(u8, usize, bool)> {
2989 let t = line.trim_end_matches(['\n', '\r']);
2990 let lead = t.len() - t.trim_start_matches(' ').len();
2991 if lead > 3 {
2992 return None;
2993 }
2994 let rest = &t[lead..];
2995 let ch = match rest.as_bytes().first() {
2996 Some(&c) if c == b'`' || c == b'~' => c,
2997 _ => return None,
2998 };
2999 let n = rest.bytes().take_while(|&b| b == ch).count();
3000 if n < 3 {
3001 return None;
3002 }
3003 Some((ch, n, !rest[n..].trim().is_empty()))
3004}
3005
3006/// The byte ranges of every line inside a fenced code region, the fence lines included.
3007///
3008/// **This is the single most likely source of a wrong strip.** A `<details>` inside a fence is
3009/// literal text the model is SHOWING the user -- the markup itself, quoted. It is not a fold, it
3010/// takes no ordinal, and rewriting it would edit the example out from under the person reading
3011/// it. The renderer is already safe there because `marked` escapes it; nothing but this function
3012/// makes the stripper safe.
3013fn fenced_spans(text: &str) -> Vec<(usize, usize)> {
3014 let mut out: Vec<(usize, usize)> = Vec::new();
3015 let mut open: Option<(u8, usize)> = None;
3016 let mut at = 0usize;
3017 for line in text.split_inclusive('\n') {
3018 let end = at + line.len();
3019 match (fence_run(line), open) {
3020 // A close must match the character it opened with, be at least as long, and carry no
3021 // info string.
3022 (Some((ch, n, info)), Some((c, k))) if ch == c && n >= k && !info => {
3023 out.push((at, end));
3024 open = None;
3025 },
3026 (Some((ch, n, _)), None) => {
3027 out.push((at, end));
3028 open = Some((ch, n));
3029 },
3030 // Anything else inside a fence is fenced; anything else outside one is prose.
3031 _ => if open.is_some() { out.push((at, end)); },
3032 }
3033 at = end;
3034 }
3035 out
3036}
3037
3038// ── The seam the APP places ─────────────────────────────────────────────────────────────────
3039//
3040// Three wordings by three authors asked the model to write the `<details>` itself, and 5 answers
3041// in 76 carried one -- `dev/PROMPT_NOTES.md` §5 and `dev/REGISTER_NOTES.md` §11, across two
3042// register kinds on the shape of question the note exists for. So the markup is the app's now
3043// and the model writes one line: `Fold:` and a sentence or two of what the working below it
3044// concludes. [`seam_text`] turns that line into exactly the element a model used to write, and
3045// [`folds`] below never learns that anything is different.
3046//
3047// **The refusals are the point, not the fold.** A length threshold applied blindly produces
3048// FOLD-ALL, which `dev/CONTRACT_FOLD.md` §5 calls worse than no control at all. Here the two
3049// failures no wording could prevent are unreachable instead: too little above the line, too few
3050// words in the summary, or too little below it, and there is no fold -- the sentence is left as
3051// prose and nothing is hidden.
3052//
3053// `www/js/render.js`'s `seamText` is the other half and the two must agree character for
3054// character, for `Fold::key`'s reason: the page names the fold the reader opened and this side
3055// matches the name. `dev/fixtures/fold_seam.json` is what both are tested against.
3056
3057/// Enough above the seam to be an answer. The same forty characters `dev/probe_notes.mjs` calls
3058/// FOLD-ALL below, counted the same way -- whitespace collapsed, ends trimmed -- because it is
3059/// the same rule and not a second opinion about it.
3060const SEAM_LEAD_MIN: usize = 40;
3061/// Enough below it to be worth a control, which is `dev/CONTRACT_FOLD.md` §5's carve-out.
3062const SEAM_BODY_MIN: usize = 240;
3063/// A summary is a summary and not a label -- §13.
3064const SEAM_WORDS_MIN: usize = 6;
3065
3066/// The summary a `Fold:` line carries, or `None` if the line is not one.
3067///
3068/// Generous in what it accepts, because a model that has understood the note and reached for a
3069/// heading or for bold should not lose its fold over the decoration: `Fold:`, `**Fold:**`,
3070/// `## Fold:` and `_Fold:_` all seam. What it will not accept is an indented line, which is a
3071/// code block, or one whose summary is empty.
3072fn seam_line(line: &str) -> Option<String> {
3073 let raw = line.trim_end_matches(['\n', '\r']);
3074 if raw.starts_with(" ") || raw.starts_with('\t') {
3075 return None;
3076 }
3077 let mut s = raw.trim_start_matches(' ');
3078 // A heading marker, then at least one space, or `#Fold:` would seam as a heading.
3079 let hashes = s.bytes().take_while(|&b| b == b'#').count();
3080 if (1..=6).contains(&hashes) && s[hashes..].starts_with(' ') {
3081 s = s[hashes..].trim_start_matches(' ');
3082 }
3083 let lead = emphasis_at(s);
3084 s = &s[lead.len()..];
3085 if s.len() < 4 || !s[..4].eq_ignore_ascii_case("fold") {
3086 return None;
3087 }
3088 let after = s[4..].trim_start_matches([' ', '\t']);
3089 let mut rest = match after.strip_prefix(':') {
3090 Some(r) => r,
3091 None => return None,
3092 };
3093 let shut = emphasis_at(rest);
3094 if !shut.is_empty() {
3095 rest = &rest[shut.len()..];
3096 } else if !lead.is_empty() {
3097 let t = rest.trim_end();
3098 let tail = emphasis_end(t);
3099 if !tail.is_empty() {
3100 rest = &t[..t.len() - tail.len()];
3101 }
3102 }
3103 let sum = rest.split_whitespace().collect::<Vec<_>>().join(" ");
3104 if sum.is_empty() { None } else { Some(sum) }
3105}
3106
3107/// The emphasis run a markdown span opens with, longest first.
3108fn emphasis_at(s: &str) -> &str {
3109 for m in ["**", "__", "*", "_"] {
3110 if s.starts_with(m) {
3111 return &s[..m.len()];
3112 }
3113 }
3114 ""
3115}
3116
3117/// The emphasis run a markdown span closes with, longest first.
3118fn emphasis_end(s: &str) -> &str {
3119 for m in ["**", "__", "*", "_"] {
3120 if s.ends_with(m) {
3121 return &s[s.len() - m.len()..];
3122 }
3123 }
3124 ""
3125}
3126
3127/// How much of `t` a reader would actually see, counted as the page counts it.
3128///
3129/// UTF-16 units rather than characters, because the other half of this is JavaScript and a
3130/// boundary the two halves disagreed on would fold on one side and not the other.
3131fn seam_visible(t: &str) -> usize {
3132 t.split_whitespace().collect::<Vec<_>>().join(" ").encode_utf16().count()
3133}
3134
3135/// `text` with the model's `Fold:` line turned into the element it stands for, or `None` where
3136/// there is nothing to do.
3137///
3138/// Four outcomes and only one of them is a fold: no line at all, or a model that wrote its own
3139/// `<details>`, leaves the text alone; a line on a qualifying answer becomes one top-level fold,
3140/// blank lines and all; a line on an answer that does not qualify loses its `Fold:` and stays as
3141/// prose, so nothing is hidden and nothing is lost. The page has a fourth, which this side
3142/// cannot have: mid-stream it holds the line back rather than showing a fold that might unwind.
3143pub fn seam_text(text: &str) -> Option<String> {
3144 if !text.to_ascii_lowercase().contains("fold") {
3145 return None;
3146 }
3147 // A model that wrote the markup itself has already placed its seam.
3148 if text.contains("<details") && !folds(text).is_empty() {
3149 return None;
3150 }
3151 let fenced = fenced_spans(text);
3152 let hidden = |p: usize| fenced.iter().any(|&(a, b)| p >= a && p < b);
3153 // Every `Fold:` line outside a fence: where it starts, how long it is, and what it says.
3154 let mut marks: Vec<(usize, usize, String)> = Vec::new();
3155 let mut at = 0usize;
3156 for line in text.split_inclusive('\n') {
3157 let bare = line.trim_end_matches(['\n', '\r']);
3158 if !hidden(at) {
3159 if let Some(sum) = seam_line(bare) {
3160 marks.push((at, bare.len(), sum));
3161 }
3162 }
3163 at += line.len();
3164 }
3165 let (start, len, sum) = match marks.first() {
3166 Some(m) => (m.0, m.1, m.2.clone()),
3167 None => return None,
3168 };
3169 // Only the first line is the seam. A second would otherwise reach the reader with its
3170 // marker still on it, so every later one loses the marker and stays where it is.
3171 let bare_from = |from: usize| -> String {
3172 let mut out = String::with_capacity(text.len());
3173 let mut cut = from;
3174 for &(a, n, ref s) in &marks {
3175 if a < from {
3176 continue;
3177 }
3178 out.push_str(&text[cut..a]);
3179 out.push_str(s);
3180 cut = a + n;
3181 }
3182 out.push_str(&text[cut..]);
3183 out
3184 };
3185 let above = &text[..start];
3186 // The line's own newline goes with the line.
3187 let rest_at = (start + len + 1).min(text.len());
3188 let body = &text[rest_at..];
3189 // A summary carrying the very tags this builds would close the element early and leave the
3190 // rest of the answer outside it.
3191 let tagged = sum.to_ascii_lowercase();
3192 if tagged.contains("<summary") || tagged.contains("</summary")
3193 || tagged.contains("<details") || tagged.contains("</details")
3194 || sum.split_whitespace().count() < SEAM_WORDS_MIN
3195 || seam_visible(above) < SEAM_LEAD_MIN
3196 || seam_visible(body) < SEAM_BODY_MIN
3197 {
3198 return Some(bare_from(0));
3199 }
3200 Some(fmt!("{}\n\n<details>\n<summary>{}</summary>\n\n{}\n\n</details>",
3201 above.trim_end(), sum, bare_from(rest_at).trim()))
3202}
3203
3204/// [`seam_text`] applied, for a caller that only wants the text back.
3205///
3206/// The answer is stored seamed rather than seamed on the way out, so the element exists exactly
3207/// once and everything downstream -- the strip, the compactor, a reload of the thread a year
3208/// later -- meets an ordinary fold and nothing has to know about a marker line.
3209pub fn seamed(text: String) -> String {
3210 match seam_text(&text) {
3211 Some(t) => t,
3212 None => text,
3213 }
3214}
3215
3216/// Every real fold in one assistant message's text, in document order.
3217///
3218/// Four shapes are deliberately NOT folds, and each one is a case in the fixture. A `<details>`
3219/// in a fence is quoted markup. A `<details>` with no `<summary>` has no label, so it has no key
3220/// and could not be matched against the open set anyway. A `<details>` NESTED inside another is
3221/// carried away by its parent's strip, so a key of its own would name a body that no longer
3222/// exists. A `<details>` with no `</details>` is a fold still being written: it keys, because the
3223/// ordinal it takes is settled the moment its summary is, but it is left alone by the strip --
3224/// see [`strip_folds`].
3225///
3226/// **Elements are paired by DEPTH, not by the first closing tag that turns up.** The browser's
3227/// parser nests correctly, so a scanner that closed an outer fold at its child's `</details>`
3228/// would disagree with the renderer about where the fold ENDS -- and then replace the wrong span
3229/// of text, cutting the body short and leaving the remainder of the element dangling in the
3230/// payload. Measured: the shared fixture's nested case gives a 48-character body under
3231/// first-close pairing and the correct 67 under this one.
3232fn folds(text: &str) -> Vec<Fold> {
3233 let fenced = fenced_spans(text);
3234 let hidden = |p: usize| fenced.iter().any(|&(a, b)| p >= a && p < b);
3235 // The next occurrence of `tag` at or after `from` that is not inside a fence.
3236 let next = |from: usize, tag: &str| -> Option<usize> {
3237 let mut at = from;
3238 while at < text.len() {
3239 match text[at..].find(tag) {
3240 Some(p) => {
3241 let abs = at + p;
3242 if !hidden(abs) {
3243 return Some(abs);
3244 }
3245 at = abs + tag.len();
3246 },
3247 None => return None,
3248 }
3249 }
3250 None
3251 };
3252 let mut out: Vec<Fold> = Vec::new();
3253 let mut at = 0usize;
3254 while at < text.len() {
3255 let open = match next(at, "<details") {
3256 Some(p) => p,
3257 None => break,
3258 };
3259 // Walk to the MATCHING close, counting depth. `inner` is where the first child element
3260 // starts, which is the bound on how far the label may be looked for.
3261 let mut depth = 1usize;
3262 let mut scan = open + "<details".len();
3263 let mut close: Option<usize> = None;
3264 let mut inner: Option<usize> = None;
3265 while depth > 0 {
3266 // `"<details"` cannot match inside `"</details>"`, so the two searches never see the
3267 // same tag twice.
3268 match (next(scan, "<details"), next(scan, "</details>")) {
3269 (Some(o), Some(c)) if o < c => {
3270 depth += 1;
3271 if inner.is_none() {
3272 inner = Some(o);
3273 }
3274 scan = o + "<details".len();
3275 },
3276 (_, Some(c)) => {
3277 depth -= 1;
3278 scan = c + "</details>".len();
3279 if depth == 0 {
3280 close = Some(c);
3281 }
3282 },
3283 // Nothing closes it: the model is still writing.
3284 (_, None) => break,
3285 }
3286 }
3287 // Without a close the element runs to the end of what has been written so far, which is
3288 // what a fold looks like part way through a stream.
3289 let limit = close.unwrap_or(text.len());
3290 // Where scanning resumes whether or not this element turns out to be a fold. Past the
3291 // WHOLE element, which is what keeps a nested fold from taking an ordinal of its own.
3292 let after = match close {
3293 Some(c) => c + "</details>".len(),
3294 None => text.len(),
3295 };
3296 // The label must belong to THIS element: a `<summary>` after the first child is the
3297 // child's, and borrowing it would put a nested fold's name on its parent's body.
3298 let labelled = |p: usize| p < limit && inner.map_or(true, |i| p < i);
3299 let sum_open = match next(open, "<summary").filter(|&p| labelled(p)) {
3300 Some(p) => p,
3301 None => { at = after; continue; },
3302 };
3303 // AND IT MUST BE THE FIRST THING INSIDE, whitespace aside. This is stricter than the
3304 // browser, which takes the first `<summary>` child as the control wherever it sits, and
3305 // the strictness is the point: it is the only shape in which "the body is what lies
3306 // between `</summary>` and `</details>`" is unambiguous, and it is exactly what the
3307 // prompt asks the model to write. Prose before the label makes the element not a fold,
3308 // so it is left alone -- the reader still gets a native disclosure widget and the body
3309 // still travels, which costs tokens. Keying it instead would strip a fold whose open
3310 // state the page never tracked, and lose the reader's gesture rather than some money.
3311 let head_gt = match text[open..limit].find('>') {
3312 Some(p) => open + p + 1,
3313 None => { at = after; continue; },
3314 };
3315 if !text[head_gt..sum_open].trim().is_empty() {
3316 at = after;
3317 continue;
3318 }
3319 let sum_gt = match text[sum_open..limit].find('>') {
3320 Some(p) => sum_open + p + 1,
3321 None => { at = after; continue; },
3322 };
3323 let sum_close = match next(sum_gt, "</summary>").filter(|&p| p < limit) {
3324 Some(p) => p,
3325 None => { at = after; continue; },
3326 };
3327 let body_from = sum_close + "</summary>".len();
3328 let raw = &text[body_from..limit];
3329 let lead = raw.len() - raw.trim_start().len();
3330 let trimmed = raw.trim();
3331 out.push(Fold {
3332 ord: out.len(),
3333 summary: collapse_ws(&strip_tags(&text[sum_gt..sum_close])),
3334 body: (body_from + lead)..(body_from + lead + trimmed.len()),
3335 chars: trimmed.chars().count(),
3336 closed: close.is_some(),
3337 });
3338 at = after;
3339 }
3340 out
3341}
3342
3343/// Everything outside `<...>`, which is the `<summary>` element's own text.
3344///
3345/// A `<` with no `>` after it is NOT a tag and is kept, because the browser keeps it too: the JS
3346/// half reads this same summary with `textContent`, and a model writing `a < b` in a label would
3347/// otherwise key it as `a` here and as `a < b` there -- one label, two keys, and a fold that
3348/// never opens.
3349fn strip_tags(s: &str) -> String {
3350 let mut out = String::with_capacity(s.len());
3351 let mut rest = s;
3352 loop {
3353 let lt = match rest.find('<') {
3354 Some(p) => p,
3355 None => { out.push_str(rest); return out; },
3356 };
3357 match rest[lt..].find('>') {
3358 Some(gt) => {
3359 out.push_str(&rest[..lt]);
3360 rest = &rest[lt + gt + 1..];
3361 },
3362 None => { out.push_str(rest); return out; },
3363 }
3364 }
3365}
3366
3367/// Trimmed, with every internal whitespace run collapsed to one space.
3368fn collapse_ws(s: &str) -> String {
3369 s.split_whitespace().collect::<Vec<_>>().join(" ")
3370}
3371
3372/// The text an assistant message is REPLAYED with, or `None` when nothing in it changes.
3373///
3374/// The text sibling of [`strip_said`], and it exists for the same reason: a fold's body is for a
3375/// person, once. Left in the transcript it is re-sent on every later request for the life of the
3376/// conversation, so a model that explains at length charges for the explanation again on every
3377/// turn whether or not anybody looked at it twice.
3378///
3379/// **The element is kept and only the body is replaced.** A model that sees the fold it wrote
3380/// still knows it folded something and what it called it, which a wholesale deletion would take
3381/// away along with the bytes.
3382///
3383/// Two folds are passed over. An OPEN one is on the user's screen, so the model holds it too --
3384/// their own gesture decides the working set. An UNCLOSED one is malformed, and [`strip_said`]'s
3385/// rule applies: rewriting it replaces a problem the model can see with one it cannot.
3386///
3387/// # Arguments
3388/// * `text` - The assistant's own words, as the model wrote them.
3389/// * `open` - The keys of the folds the user has open. See [`Fold::key`].
3390fn strip_folds(text: &str, open: &OpenSet) -> Option<String> {
3391 let found = folds(text);
3392 if found.is_empty() {
3393 return None;
3394 }
3395 let mut out = String::with_capacity(text.len());
3396 let mut at = 0usize;
3397 for f in &found {
3398 if !f.closed || open.contains(&f.key()) {
3399 continue;
3400 }
3401 // An empty body would be REPLACED by a hundred characters of note, so the one case where
3402 // stripping costs tokens rather than saving them is not stripped.
3403 if f.chars == 0 {
3404 continue;
3405 }
3406 out.push_str(&text[at..f.body.start]);
3407 out.push_str(&fold_note(f.chars));
3408 at = f.body.end;
3409 }
3410 if at == 0 {
3411 return None;
3412 }
3413 out.push_str(&text[at..]);
3414 Some(out)
3415}
3416
3417/// How many bytes of an assistant message's text will actually go on the wire.
3418///
3419/// The sibling of [`sent_args_len`] for the inline fold, and it exists for the same defect: the
3420/// compaction trigger in [`crate::agent::compact::msg_bytes`] sized a message by what the model
3421/// wrote, while serialisation takes every closed fold's body out. So the trigger measured a
3422/// conversation nobody was going to send, spent the budget on bytes that leave on the way out,
3423/// and folded a conversation earlier than it needed to. Asked here rather than restated there,
3424/// because a rule written twice is a rule that eventually disagrees with itself.
3425///
3426/// # Arguments
3427/// * `text` - The assistant's own words.
3428/// * `open` - The keys of the folds the user has open.
3429pub fn sent_text_len(text: &str, open: &OpenSet) -> usize {
3430 match strip_folds(text, open) {
3431 Some(replayed) => replayed.len(),
3432 None => text.len(),
3433 }
3434}
3435
3436/// Serialise a `ChatMessage` to an OpenAI-API JSON object, including
3437/// assistant `tool_calls` and the `tool` role — which `datmap_to_json`
3438/// does not carry.
3439///
3440/// Only the `user` role may carry an image on this side; `system`, `assistant` and `tool` take a
3441/// string or text parts and nothing else. A message of another role that somehow holds one is
3442/// flattened to the `[image …]` stand-in rather than sent as a part the API would reject; the
3443/// tool results that legitimately produce images are re-homed by
3444/// [`build_openai_body`](LlmClient::build_openai_body) instead.
3445///
3446/// # Arguments
3447/// * `msg` - The message to serialise.
3448/// * `open` - The `say` folds the user has open, whose detail therefore travels. See
3449/// [`OpenFolds`].
3450fn message_to_json(msg: &ChatMessage, open: &std::collections::HashSet<String>) -> String {
3451 match msg {
3452 ChatMessage::System { content } =>
3453 fmt!("{{\"role\":\"system\",\"content\":\"{}\"}}", json_escape(&content.as_text())),
3454 ChatMessage::User { content } =>
3455 fmt!("{{\"role\":\"user\",\"content\":{}}}", openai_content(content)),
3456 ChatMessage::Assistant { content, tool_calls } => {
3457 // The assistant's own words are the one role's text that serialisation rewrites: a
3458 // closed `<details>` fold travels as a note in its body's place. Both branches below
3459 // read this, so neither can be given the fold and the other the raw text.
3460 let said = content.as_text();
3461 let folded = strip_folds(&said, open);
3462 let text = json_escape(folded.as_deref().unwrap_or(&said));
3463 if tool_calls.is_empty() {
3464 fmt!("{{\"role\":\"assistant\",\"content\":\"{}\"}}", text)
3465 } else {
3466 let calls: Vec<String> = tool_calls.iter().map(|tc| {
3467 let stripped = strip_said(&tc.name, &tc.arguments, open.contains(&tc.id));
3468 let args = stripped.as_deref().unwrap_or(&tc.arguments);
3469 fmt!(
3470 "{{\"id\":\"{}\",\"type\":\"function\",\"function\":{{\"name\":\"{}\",\"arguments\":\"{}\"}}}}",
3471 json_escape(&tc.id), json_escape(&tc.name), json_escape(args))
3472 }).collect();
3473 fmt!("{{\"role\":\"assistant\",\"content\":\"{}\",\"tool_calls\":[{}]}}",
3474 text, calls.join(","))
3475 }
3476 }
3477 ChatMessage::Tool { tool_call_id, content } =>
3478 fmt!("{{\"role\":\"tool\",\"tool_call_id\":\"{}\",\"content\":\"{}\"}}",
3479 json_escape(tool_call_id), json_escape(&content.as_text())),
3480 }
3481}
3482
3483/// Whether an OpenAI-shaped payload says the reply was cut at the output limit.
3484///
3485/// `finish_reason` is `null` on every delta but the last, and `"stop"` on an answer
3486/// that finished; `"length"` is the one value that means the model was still writing.
3487/// Read from the raw payload rather than inferred from malformed arguments, which is
3488/// what the browser had to do and which cannot see a plain text reply cut short.
3489fn openai_truncated(json: &str) -> bool {
3490 matches!(extract_json_string(json, "finish_reason").as_deref(), Some("length"))
3491}
3492
3493/// Whether an Anthropic payload says the same thing.
3494fn anthropic_truncated(json: &str) -> bool {
3495 matches!(extract_json_string(json, "stop_reason").as_deref(), Some("max_tokens"))
3496}
3497
3498/// Parse a non-streaming chat completion body into
3499/// `(content, tool_calls, usage)`.
3500fn parse_full_response(body: &str) -> (String, Vec<ToolCall>, Usage) {
3501 // Scope content extraction to before "tool_calls" so we don't pick
3502 // up a "content" key inside a tool call's arguments.
3503 let scope_end = body.find("\"tool_calls\"").unwrap_or(body.len());
3504 let content = extract_json_string(&body[..scope_end], "content").unwrap_or_default();
3505
3506 let mut tool_calls = Vec::new();
3507 if let Some(arr) = find_json_array(body, "tool_calls") {
3508 for elem in split_top_level_objects(&arr) {
3509 let name = match extract_json_string(&elem, "name") {
3510 Some(n) if !n.is_empty() => n,
3511 _ => continue,
3512 };
3513 let id = extract_json_string(&elem, "id").unwrap_or_default();
3514 let arguments = extract_json_string(&elem, "arguments")
3515 .unwrap_or_else(|| "{}".to_string());
3516 tool_calls.push(ToolCall { id, name, arguments });
3517 }
3518 }
3519
3520 (content, tool_calls, parse_usage(body).unwrap_or_default())
3521}
3522
3523
3524// ┌───────────────────────────────────────────────────────────────┐
3525// │ StreamAcc — streamed delta accumulator │
3526// └───────────────────────────────────────────────────────────────┘
3527
3528/// One tool call being reconstructed from streamed fragments.
3529///
3530/// A streamed `tool_calls` delta arrives in pieces keyed by `index`: the
3531/// first fragment usually carries the `id` and function `name` with an
3532/// empty `arguments`, and later fragments append `arguments` text until
3533/// the call is whole.
3534struct StreamCall {
3535 /// Position of this call within the assistant turn.
3536 index: i64,
3537 id: String,
3538 name: String,
3539 /// Accumulated raw JSON arguments, concatenated across fragments.
3540 arguments: String,
3541}
3542
3543/// Accumulates OpenAI-style streaming chat deltas across SSE chunks:
3544/// text content, incrementally-built tool calls, and usage.
3545///
3546/// Each `data:` payload is fed to [`ingest`](StreamAcc::ingest); when the
3547/// stream ends, [`into_response`](StreamAcc::into_response) yields the
3548/// assembled [`ChatOnceResponse`].
3549#[derive(Default)]
3550struct StreamAcc {
3551 content: String,
3552 // The model's own working, kept apart from the answer it produced.
3553 reasoning: String,
3554 /// The last usage block the stream reported. An aborted stream may never
3555 /// deliver one, which leaves this at its default rather than erroring.
3556 usage: Usage,
3557 calls: Vec<StreamCall>,
3558 /// Whether a chunk said the reply stopped at the output limit.
3559 truncated: bool,
3560}
3561
3562impl StreamAcc {
3563
3564 /// Fold one SSE `data:` payload into the accumulator, forwarding each delta to
3565 /// `on_token` as it arrives, labelled as answer or as working.
3566 fn ingest(&mut self, data: &str, on_token: &mut impl FnMut(Delta<'_>)) {
3567 // Text delta — scoped to before any `tool_calls` so a `content`
3568 // key inside a tool call's arguments is never mistaken for it.
3569 let scope_end = data.find("\"tool_calls\"").unwrap_or(data.len());
3570 if let Some(content) = extract_json_string(&data[..scope_end], "content") {
3571 if !content.is_empty() {
3572 on_token(Delta::Text(&content));
3573 self.content.push_str(&content);
3574 }
3575 }
3576
3577 // THE MODEL'S OWN WORKING, which every reasoning model on this dialect streams
3578 // and which this client read none of until 2026-08-28. A measured DeepSeek round
3579 // pulled 1.8 MB down the wire over 84 seconds and put fifty characters on the
3580 // screen; all the rest was this field, discarded delta by delta, and the user was
3581 // billed for it while watching a spinner.
3582 //
3583 // Two spellings, and never both in one delta: `reasoning` is what OpenRouter
3584 // sends, `reasoning_content` is what DeepSeek's own endpoint calls it. So one is
3585 // read and then the other, rather than both concatenated.
3586 //
3587 // `reasoning_details` is NOT read. OpenRouter sends it alongside `reasoning` with
3588 // the same words in it, verbatim, so a reader that took both would put every
3589 // token on the page twice.
3590 //
3591 // `null` is the value on the deltas that carry no reasoning, and
3592 // `extract_json_string` answers None for a value that is not a string -- so the
3593 // absence needs no test of its own here.
3594 let think = extract_json_string(&data[..scope_end], "reasoning")
3595 .or_else(|| extract_json_string(&data[..scope_end], "reasoning_content"));
3596 if let Some(t) = think {
3597 if !t.is_empty() {
3598 on_token(Delta::Reasoning(&t));
3599 self.reasoning.push_str(&t);
3600 }
3601 }
3602
3603 // Tool-call fragments — merge each into its slot by `index`.
3604 if let Some(arr) = find_json_array(data, "tool_calls") {
3605 for elem in split_top_level_objects(&arr) {
3606 let index = extract_json_number(&elem, "index")
3607 .map(|n| n as i64)
3608 .unwrap_or(0);
3609 // Locate an existing slot by index before borrowing
3610 // mutably, so a new slot can be pushed without an
3611 // overlapping borrow.
3612 let pos = self.calls.iter().position(|c| c.index == index);
3613 let slot = match pos {
3614 Some(p) => &mut self.calls[p],
3615 None => {
3616 self.calls.push(StreamCall {
3617 index,
3618 id: String::new(),
3619 name: String::new(),
3620 arguments: String::new(),
3621 });
3622 let last = self.calls.len() - 1;
3623 &mut self.calls[last]
3624 }
3625 };
3626 if let Some(id) = extract_json_string(&elem, "id") {
3627 if !id.is_empty() { slot.id = id; }
3628 }
3629 if let Some(name) = extract_json_string(&elem, "name") {
3630 if !name.is_empty() { slot.name = name; }
3631 }
3632 if let Some(args) = extract_json_string(&elem, "arguments") {
3633 slot.arguments.push_str(&args);
3634 }
3635 }
3636 }
3637
3638 // Usage — present on the final chunk when include_usage is set.
3639 if let Some(u) = parse_usage(data) {
3640 self.usage = u;
3641 }
3642
3643 // Why the model stopped, which arrives on the last delta and nowhere else.
3644 // Sticky: a later chunk carrying only usage must not unsay it.
3645 if openai_truncated(data) {
3646 self.truncated = true;
3647 }
3648 }
3649
3650 /// Whether this turn has produced anything yet.
3651 ///
3652 /// A retry is only safe while this is false: text has already been handed to
3653 /// the caller, and a tool-call fragment is a partial the next attempt would
3654 /// duplicate rather than replace.
3655 fn has_output(&self) -> bool {
3656 !self.content.is_empty() || !self.calls.is_empty()
3657 }
3658
3659 /// Consume the accumulator into a [`ChatOnceResponse`]. Calls with no
3660 /// name are dropped (a stray fragment), and an empty arguments string
3661 /// becomes `{}` so tool dispatch always sees a valid JSON object.
3662 fn into_response(self, aborted: bool, retries: u32) -> ChatOnceResponse {
3663 let tool_calls = self.calls.into_iter()
3664 .filter(|c| !c.name.is_empty())
3665 .map(|c| ToolCall {
3666 id: c.id,
3667 name: c.name,
3668 arguments: if c.arguments.is_empty() { "{}".to_string() } else { c.arguments },
3669 })
3670 .collect();
3671 ChatOnceResponse {
3672 content: self.content,
3673 tool_calls,
3674 prompt_tokens: self.usage.prompt,
3675 completion_tokens: self.usage.completion,
3676 cached_tokens: self.usage.cached,
3677 cost_usd: self.usage.cost_usd,
3678 aborted,
3679 retries,
3680 thinking: self.reasoning,
3681 truncated: self.truncated,
3682 }
3683 }
3684}
3685
3686// ┌───────────────────────────────────────────────────────────────┐
3687// │ AnthropicAcc — Messages API event accumulator │
3688// └───────────────────────────────────────────────────────────────┘
3689
3690/// What one Anthropic content block is.
3691#[derive(Clone, Copy, Debug, Eq, PartialEq)]
3692enum AnthKind {
3693 Text,
3694 Thinking,
3695 /// A `redacted_thinking` block, which is opaque and replayed verbatim.
3696 Redacted,
3697 ToolUse,
3698 /// A block this client does not act on (a server tool, a fallback marker).
3699 Other,
3700}
3701
3702/// One content block being rebuilt from the event stream.
3703struct AnthBlock {
3704 /// Position in the message's `content` array; the streamed events key on it.
3705 index: i64,
3706 kind: AnthKind,
3707 id: String,
3708 name: String,
3709 /// `input_json_delta` fragments, concatenated into the tool's arguments.
3710 args: String,
3711 /// `thinking_delta` fragments, concatenated.
3712 think: String,
3713 /// The block's signature, which the API verifies when it is handed back.
3714 sig: String,
3715 /// A block replayed verbatim rather than rebuilt, as JSON.
3716 raw: String,
3717}
3718
3719/// The Anthropic `usage` counts, kept as reported.
3720///
3721/// Held raw rather than folded into [`Usage`] on arrival because they arrive
3722/// twice -- once on `message_start` and again, cumulatively, on `message_delta`
3723/// -- and the second report names only the fields that changed. Overwriting
3724/// [`Usage`] wholesale from the second would zero the input counts.
3725#[derive(Clone, Copy, Debug, Default)]
3726struct AnthUsage {
3727 input: u64,
3728 output: u64,
3729 /// Prompt tokens served from the cache, billed at a tenth of a fresh read.
3730 read: u64,
3731 /// Prompt tokens written to the cache, billed at 1.25x a fresh read.
3732 write: u64,
3733}
3734
3735impl AnthUsage {
3736
3737 /// Fold in one `usage` object, taking only the fields it actually carries.
3738 fn merge(&mut self, usage: &str) {
3739 if let Some(v) = extract_json_number(usage, "input_tokens") { self.input = v; }
3740 if let Some(v) = extract_json_number(usage, "output_tokens") { self.output = v; }
3741 if let Some(v) = extract_json_number(usage, "cache_read_input_tokens") { self.read = v; }
3742 if let Some(v) = extract_json_number(usage, "cache_creation_input_tokens") { self.write = v; }
3743 }
3744
3745 /// The client's own usage shape.
3746 ///
3747 /// Anthropic's `input_tokens` counts only what was neither read from nor
3748 /// written to the cache, where this client's `prompt` means every prompt
3749 /// token processed -- so the three are added, and `cached` is the read.
3750 /// The 1.25x premium on a cache *write* is not modelled: the price table
3751 /// carries one cached rate, not two, so a write is priced as a fresh read.
3752 /// That understates the first request of a session slightly and nothing
3753 /// afterwards, which is the smaller of the two errors available.
3754 fn into_usage(self) -> Usage {
3755 Usage {
3756 prompt: self.input.saturating_add(self.read).saturating_add(self.write),
3757 completion: self.output,
3758 cached: self.read,
3759 // Anthropic bills against an account and reports no per-call cost,
3760 // so this stays zero and the price table answers instead.
3761 cost_usd: 0.0,
3762 }
3763 }
3764}
3765
3766/// Accumulates the Anthropic Messages API event stream: text, thinking,
3767/// incrementally-built tool calls, and usage.
3768///
3769/// The events are named (`content_block_start`, `content_block_delta`, …) and
3770/// keyed by block index, rather than being deltas of one growing object, so
3771/// this is a different machine from [`StreamAcc`] rather than a variation of it.
3772#[derive(Default)]
3773struct AnthropicAcc {
3774 content: String,
3775 usage: AnthUsage,
3776 blocks: Vec<AnthBlock>,
3777 /// An `error` event delivered on an otherwise-successful stream, as
3778 /// `(type, message)`.
3779 error: Option<(String, String)>,
3780 /// Whether a `message_delta` said the reply stopped at the output limit.
3781 truncated: bool,
3782}
3783
3784impl AnthropicAcc {
3785
3786 /// The slot for `index`, created if this is the first event for it.
3787 fn slot(&mut self, index: i64, kind: AnthKind) -> &mut AnthBlock {
3788 match self.blocks.iter().position(|b| b.index == index) {
3789 Some(p) => &mut self.blocks[p],
3790 None => {
3791 self.blocks.push(AnthBlock {
3792 index,
3793 kind,
3794 id: String::new(),
3795 name: String::new(),
3796 args: String::new(),
3797 think: String::new(),
3798 sig: String::new(),
3799 raw: String::new(),
3800 });
3801 let last = self.blocks.len() - 1;
3802 &mut self.blocks[last]
3803 }
3804 }
3805 }
3806
3807 /// Fold one SSE `data:` payload in, forwarding both kinds of delta to `on_token`.
3808 ///
3809 /// Thinking goes out as [`Delta::Reasoning`] and never as [`Delta::Text`], which is
3810 /// the whole of what keeps it out of the answer. It used not to be forwarded at
3811 /// all, because the sink took a bare `&str` and a caller that was handed one had no
3812 /// way to tell working from reply -- so the reasoning was held back until the round
3813 /// ended. Now the sink says which is which, so it can be shown as it arrives, and a
3814 /// round that thinks for a minute stops looking like a round that has hung.
3815 fn ingest(&mut self, data: &str, on_token: &mut impl FnMut(Delta<'_>)) {
3816 let ty = match extract_json_string(data, "type") {
3817 Some(t) => t,
3818 None => return,
3819 };
3820 match ty.as_str() {
3821 "message_start" | "message_delta" => {
3822 if let Some(u) = find_json_object(data, "usage") { self.usage.merge(&u); }
3823 // `message_start` carries `stop_reason: null`, so only a real one sets
3824 // this; and once set, nothing later unsets it.
3825 if anthropic_truncated(data) { self.truncated = true; }
3826 }
3827 "content_block_start" => {
3828 let index = extract_json_number(data, "index").map(|n| n as i64).unwrap_or(0);
3829 let cb = match find_json_object(data, "content_block") {
3830 Some(c) => c,
3831 None => return,
3832 };
3833 match extract_json_string(&cb, "type").unwrap_or_default().as_str() {
3834 "text" => { self.slot(index, AnthKind::Text); }
3835 "thinking" => { self.slot(index, AnthKind::Thinking); }
3836 "redacted_thinking" => {
3837 let slot = self.slot(index, AnthKind::Redacted);
3838 slot.kind = AnthKind::Redacted;
3839 slot.raw = cb.clone();
3840 }
3841 "tool_use" => {
3842 let id = extract_json_string(&cb, "id").unwrap_or_default();
3843 let name = extract_json_string(&cb, "name").unwrap_or_default();
3844 let slot = self.slot(index, AnthKind::ToolUse);
3845 slot.kind = AnthKind::ToolUse;
3846 slot.id = id;
3847 slot.name = name;
3848 }
3849 // A server tool, or a block type added after this was
3850 // written: recorded so its deltas land somewhere harmless.
3851 _ => { self.slot(index, AnthKind::Other); }
3852 }
3853 }
3854 "content_block_delta" => {
3855 let index = extract_json_number(data, "index").map(|n| n as i64).unwrap_or(0);
3856 let d = match find_json_object(data, "delta") {
3857 Some(d) => d,
3858 None => return,
3859 };
3860 match extract_json_string(&d, "type").unwrap_or_default().as_str() {
3861 "text_delta" => {
3862 if let Some(t) = extract_json_string(&d, "text") {
3863 if !t.is_empty() {
3864 on_token(Delta::Text(&t));
3865 self.content.push_str(&t);
3866 }
3867 }
3868 }
3869 "thinking_delta" => {
3870 if let Some(t) = extract_json_string(&d, "thinking") {
3871 if !t.is_empty() { on_token(Delta::Reasoning(&t)); }
3872 self.slot(index, AnthKind::Thinking).think.push_str(&t);
3873 }
3874 }
3875 "signature_delta" => {
3876 if let Some(s) = extract_json_string(&d, "signature") {
3877 self.slot(index, AnthKind::Thinking).sig.push_str(&s);
3878 }
3879 }
3880 "input_json_delta" => {
3881 if let Some(p) = extract_json_string(&d, "partial_json") {
3882 self.slot(index, AnthKind::ToolUse).args.push_str(&p);
3883 }
3884 }
3885 _ => {}
3886 }
3887 }
3888 "error" => {
3889 let e = find_json_object(data, "error").unwrap_or_default();
3890 self.error = Some((
3891 extract_json_string(&e, "type").unwrap_or_else(|| "api_error".to_string()),
3892 extract_json_string(&e, "message").unwrap_or_default()));
3893 }
3894 _ => {}
3895 }
3896 }
3897
3898 /// Whether this turn has produced anything the caller now holds.
3899 ///
3900 /// Thinking does not count: it is never handed to the caller, and a retry
3901 /// would simply produce a fresh block rather than a duplicate one.
3902 fn has_output(&self) -> bool {
3903 !self.content.is_empty()
3904 || self.blocks.iter().any(|b| b.kind == AnthKind::ToolUse)
3905 }
3906
3907 /// The signed thinking blocks of this turn, serialised for replay.
3908 ///
3909 /// Empty when any block of the run is unsigned -- a stream cut before its
3910 /// `signature_delta`, say. The API requires the run to match what the model
3911 /// generated, so half of it is worse than none: an unsigned block is a 400,
3912 /// and a run with one block quietly dropped is a rearrangement.
3913 fn thinking_blocks(&self) -> Vec<String> {
3914 let mut out = Vec::new();
3915 for b in &self.blocks {
3916 match b.kind {
3917 AnthKind::Thinking => {
3918 if b.sig.is_empty() { return Vec::new(); }
3919 out.push(fmt!(
3920 "{{\"type\":\"thinking\",\"thinking\":\"{}\",\"signature\":\"{}\"}}",
3921 json_escape(&b.think), json_escape(&b.sig)));
3922 }
3923 AnthKind::Redacted => {
3924 if b.raw.is_empty() { return Vec::new(); }
3925 out.push(b.raw.clone());
3926 }
3927 _ => {}
3928 }
3929 }
3930 out
3931 }
3932
3933 /// The summarised reasoning of this turn, for a caller that wants to show it.
3934 fn thinking_text(&self) -> String {
3935 let parts: Vec<&str> = self.blocks.iter()
3936 .filter(|b| b.kind == AnthKind::Thinking && !b.think.is_empty())
3937 .map(|b| b.think.as_str())
3938 .collect();
3939 parts.join("\n")
3940 }
3941
3942 /// Consume the accumulator into a [`ChatOnceResponse`].
3943 fn into_response(self, aborted: bool, retries: u32) -> ChatOnceResponse {
3944 let thinking = self.thinking_text();
3945 let tool_calls = self.blocks.iter()
3946 .filter(|b| b.kind == AnthKind::ToolUse && !b.name.is_empty())
3947 .map(|b| ToolCall {
3948 id: b.id.clone(),
3949 name: b.name.clone(),
3950 arguments: if b.args.is_empty() { "{}".to_string() } else { b.args.clone() },
3951 })
3952 .collect();
3953 ChatOnceResponse {
3954 content: self.content,
3955 tool_calls,
3956 prompt_tokens: self.usage.into_usage().prompt,
3957 completion_tokens: self.usage.output,
3958 cached_tokens: self.usage.read,
3959 cost_usd: 0.0,
3960 aborted,
3961 retries,
3962 thinking,
3963 truncated: self.truncated,
3964 }
3965 }
3966}
3967
3968
3969// ┌───────────────────────────────────────────────────────────────┐
3970// │ Acc — whichever accumulator the dialect needs │
3971// └───────────────────────────────────────────────────────────────┘
3972
3973/// The stream accumulator for a [`Dialect`].
3974///
3975/// An enum rather than a trait object: there are exactly two wire shapes, both
3976/// known here, and the retry loop wants them by value.
3977enum Acc {
3978 OpenAi(StreamAcc),
3979 Anthropic(AnthropicAcc),
3980}
3981
3982impl Acc {
3983
3984 /// A fresh accumulator for `dialect`.
3985 fn new(dialect: Dialect) -> Self {
3986 match dialect {
3987 Dialect::OpenAi => Self::OpenAi(StreamAcc::default()),
3988 Dialect::Anthropic => Self::Anthropic(AnthropicAcc::default()),
3989 }
3990 }
3991
3992 /// Fold one SSE `data:` payload in, forwarding text deltas to `on_token`.
3993 fn ingest(&mut self, data: &str, on_token: &mut impl FnMut(Delta<'_>)) {
3994 match self {
3995 Self::OpenAi(a) => a.ingest(data, on_token),
3996 Self::Anthropic(a) => a.ingest(data, on_token),
3997 }
3998 }
3999
4000 /// Whether this turn has produced anything the caller now holds.
4001 fn has_output(&self) -> bool {
4002 match self {
4003 Self::OpenAi(a) => a.has_output(),
4004 Self::Anthropic(a) => a.has_output(),
4005 }
4006 }
4007
4008 /// An error the provider delivered inside an otherwise-successful stream.
4009 ///
4010 /// Only Anthropic sends one: an OpenAI-compatible endpoint that is
4011 /// overloaded says so with a status code, before the body starts.
4012 fn stream_error(&self) -> Option<TransportErr> {
4013 let (kind, msg) = match self {
4014 Self::OpenAi(_) => return None,
4015 Self::Anthropic(a) => match &a.error {
4016 Some(e) => e.clone(),
4017 None => return None,
4018 },
4019 };
4020 let err = err!(
4021 "LLM: stream error: {} | {}", kind, msg; IO, Network, Wire, Read);
4022 let reason = fmt!("the provider reported {}", kind);
4023 // The same split as the status codes: the provider's own trouble is
4024 // worth another attempt, a complaint about this request is not.
4025 let transient = kind == "overloaded_error"
4026 || kind == "api_error"
4027 || kind == "rate_limit_error";
4028 Some(if transient {
4029 TransportErr::transient(reason, err)
4030 } else {
4031 TransportErr::fatal(reason, err)
4032 })
4033 }
4034
4035 /// The signed thinking blocks to hold against this turn's tool calls.
4036 fn take_thinking(&self) -> Vec<String> {
4037 match self {
4038 Self::OpenAi(_) => Vec::new(),
4039 Self::Anthropic(a) => a.thinking_blocks(),
4040 }
4041 }
4042
4043 /// Consume the accumulator into a [`ChatOnceResponse`].
4044 fn into_response(self, aborted: bool, retries: u32) -> ChatOnceResponse {
4045 match self {
4046 Self::OpenAi(a) => a.into_response(aborted, retries),
4047 Self::Anthropic(a) => a.into_response(aborted, retries),
4048 }
4049 }
4050}
4051
4052/// Parse a whole (non-streamed) Anthropic Messages response into
4053/// `(content, tool_calls, usage, thinking blocks)`.
4054///
4055/// The thinking blocks come back serialised for replay, exactly as the streamed
4056/// path produces them -- see [`AnthropicAcc::thinking_blocks`].
4057///
4058/// # Arguments
4059/// * `body` - The response body, as JSON text.
4060fn parse_anthropic_response(body: &str) -> (String, Vec<ToolCall>, Usage, Vec<String>) {
4061 let mut content = String::new();
4062 let mut tool_calls = Vec::new();
4063 let mut thinking: Vec<String> = Vec::new();
4064 let mut signed = true;
4065 if let Some(arr) = find_json_array(body, "content") {
4066 for elem in split_top_level_objects(&arr) {
4067 match extract_json_string(&elem, "type").unwrap_or_default().as_str() {
4068 "text" => {
4069 if let Some(t) = extract_json_string(&elem, "text") { content.push_str(&t); }
4070 }
4071 "thinking" => {
4072 match extract_json_string(&elem, "signature") {
4073 Some(s) if !s.is_empty() => thinking.push(elem.clone()),
4074 _ => signed = false,
4075 }
4076 }
4077 "redacted_thinking" => thinking.push(elem.clone()),
4078 "tool_use" => {
4079 let name = match extract_json_string(&elem, "name") {
4080 Some(n) if !n.is_empty() => n,
4081 _ => continue,
4082 };
4083 let id = extract_json_string(&elem, "id").unwrap_or_default();
4084 let input = find_json_object(&elem, "input")
4085 .unwrap_or_else(|| "{}".to_string());
4086 tool_calls.push(ToolCall { id, name, arguments: input });
4087 }
4088 _ => {}
4089 }
4090 }
4091 }
4092 // A run with an unsigned block in it cannot be replayed; see
4093 // [`AnthropicAcc::thinking_blocks`].
4094 if !signed { thinking.clear(); }
4095 let mut usage = AnthUsage::default();
4096 if let Some(u) = find_json_object(body, "usage") { usage.merge(&u); }
4097 (content, tool_calls, usage.into_usage(), thinking)
4098}
4099
4100/// Extract a JSON array value for a key, returning the inner text
4101/// including the surrounding brackets. String contents are skipped so
4102/// brackets inside strings don't confuse the depth count.
4103fn find_json_array(json: &str, key: &str) -> Option<String> {
4104 let needle = fmt!("\"{}\":", key);
4105 let pos = match json.find(&needle) {
4106 Some(p) => p,
4107 None => return None,
4108 };
4109 let bytes = json.as_bytes();
4110 // Skip whitespace after the colon to the opening bracket.
4111 let mut start = pos + needle.len();
4112 while start < bytes.len() && bytes[start].is_ascii_whitespace() { start += 1; }
4113 if start >= bytes.len() || bytes[start] != b'[' { return None; }
4114 let mut depth = 0i32;
4115 let mut in_str = false;
4116 let mut i = start;
4117 while i < bytes.len() {
4118 let b = bytes[i];
4119 if in_str {
4120 if b == b'\\' { i += 2; continue; }
4121 if b == b'"' { in_str = false; }
4122 } else {
4123 match b {
4124 b'"' => in_str = true,
4125 b'[' => depth += 1,
4126 b']' => {
4127 depth -= 1;
4128 if depth == 0 { return Some(json[start..=i].to_string()); }
4129 }
4130 _ => {}
4131 }
4132 }
4133 i += 1;
4134 }
4135 None
4136}
4137
4138/// Split a JSON array's text into its top-level `{...}` object elements.
4139fn split_top_level_objects(arr: &str) -> Vec<String> {
4140 let bytes = arr.as_bytes();
4141 let mut out = Vec::new();
4142 let mut depth = 0i32;
4143 let mut start = 0usize;
4144 let mut in_str = false;
4145 let mut i = 0usize;
4146 while i < bytes.len() {
4147 let b = bytes[i];
4148 if in_str {
4149 if b == b'\\' { i += 2; continue; }
4150 if b == b'"' { in_str = false; }
4151 } else {
4152 match b {
4153 b'"' => in_str = true,
4154 b'{' => { if depth == 0 { start = i; } depth += 1; }
4155 b'}' => {
4156 depth -= 1;
4157 if depth == 0 { out.push(arr[start..=i].to_string()); }
4158 }
4159 _ => {}
4160 }
4161 }
4162 i += 1;
4163 }
4164 out
4165}
4166
4167pub fn datmap_to_json(m: &DaticleMap) -> String {
4168 let mut out = String::with_capacity(256);
4169 out.push('{');
4170 let mut first = true;
4171 // DaticleMap iteration is not ordered — we sort keys for
4172 // deterministic output (not required by the API but cleaner).
4173 let mut entries: Vec<(&Dat, &Dat)> = m.iter().collect();
4174 entries.sort_by(|a, b| {
4175 match (a.0, b.0) {
4176 (Dat::Str(a_s), Dat::Str(b_s)) => a_s.cmp(b_s),
4177 _ => std::cmp::Ordering::Equal,
4178 }
4179 });
4180 for (k, v) in entries {
4181 if !first { out.push(','); }
4182 first = false;
4183 if let Dat::Str(k_s) = k {
4184 out.push('"');
4185 out.push_str(k_s);
4186 out.push_str("\":");
4187 out.push_str(&dat_to_json(v));
4188 }
4189 }
4190 out.push('}');
4191 out
4192}
4193
4194/// Escape a string for embedding inside a JSON string literal (no
4195/// surrounding quotes). Shared with the tool-definition builder.
4196pub(crate) fn json_escape(s: &str) -> String {
4197 let mut out = String::with_capacity(s.len());
4198 for c in s.chars() {
4199 match c {
4200 '"' => out.push_str("\\\""),
4201 '\\' => out.push_str("\\\\"),
4202 '\n' => out.push_str("\\n"),
4203 '\t' => out.push_str("\\t"),
4204 '\r' => out.push_str("\\r"),
4205 c if (c as u32) < 0x20 => out.push_str(&fmt!("\\u{:04x}", c as u32)),
4206 c => out.push(c),
4207 }
4208 }
4209 out
4210}
4211
4212/// Convert a JDAT Dat value to JSON.
4213fn dat_to_json(d: &Dat) -> String {
4214 match d {
4215 Dat::Str(s) => {
4216 let mut out = String::with_capacity(s.len() + 2);
4217 out.push('"');
4218 for c in s.chars() {
4219 match c {
4220 '"' => out.push_str("\\\""),
4221 '\\' => out.push_str("\\\\"),
4222 '\n' => out.push_str("\\n"),
4223 '\t' => out.push_str("\\t"),
4224 '\r' => out.push_str("\\r"),
4225 c if (c as u32) < 0x20 => {
4226 out.push_str(&fmt!("\\u{:04x}", c as u32));
4227 }
4228 c => out.push(c),
4229 }
4230 }
4231 out.push('"');
4232 out
4233 }
4234 Dat::U64(n) => fmt!("{}", n),
4235 Dat::Bool(b) => fmt!("{}", b),
4236 Dat::List(list) => {
4237 let items: Vec<String> = list.iter().map(dat_to_json).collect();
4238 fmt!("[{}]", items.join(","))
4239 }
4240 Dat::Map(m) => datmap_to_json(m),
4241 Dat::Empty => "null".to_string(),
4242 _ => "null".to_string(),
4243 }
4244}
4245
4246
4247// ┌───────────────────────────────────────────────────────────────┐
4248// │ Tests │
4249// └───────────────────────────────────────────────────────────────┘
4250
4251#[cfg(test)]
4252pub mod tests {
4253 use super::*;
4254
4255 use crate::protocol::ImageMedia;
4256
4257 // ── The two-depth answer, written inline ─────────────────────────────────
4258
4259 /// The fixture the Rust and JS halves are BOTH tested against.
4260 ///
4261 /// Authored by the orchestrator and read from disk rather than transcribed into this file,
4262 /// so a case added or corrected there cannot silently stop being checked here.
4263 const FOLD_FIXTURE: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/dev/fixtures/fold_keys.json");
4264
4265 /// The integers in a JSON array, for the fixture's `body_chars`.
4266 fn json_numbers(json: &str, key: &str) -> Vec<usize> {
4267 let arr = match find_json_array(json, key) {
4268 Some(a) => a,
4269 None => return Vec::new(),
4270 };
4271 arr.split(|c: char| !c.is_ascii_digit())
4272 .filter(|s| !s.is_empty())
4273 .filter_map(|s| s.parse::<usize>().ok())
4274 .collect()
4275 }
4276
4277 /// **Every case in `dev/fixtures/fold_keys.json`, driven from the file itself.**
4278 ///
4279 /// The key is the one name the two languages must agree on, and the fixture is where they
4280 /// agree. Three things are checked per case, and the third is the one that carries the risk:
4281 /// the keys, the body character counts, and the stripped text -- where a `null` means the
4282 /// stripper must leave the input EXACTLY as it found it, which is the assertion a stripper
4283 /// that rewrites too eagerly fails.
4284 #[test]
4285 fn test_every_case_in_the_shared_fold_fixture() {
4286 let json = match std::fs::read_to_string(FOLD_FIXTURE) {
4287 Ok(s) => s,
4288 Err(e) => panic!("the shared fixture must be readable at {}: {}", FOLD_FIXTURE, e),
4289 };
4290 // The wording of the note, pinned by the fixture rather than by this file: the JS half
4291 // renders the same sentence and the two must not drift apart.
4292 let want_note = extract_json_string(&json, "_placeholder").unwrap_or_default();
4293 assert_eq!(fold_note(7), want_note.replace("N characters", "7 characters"),
4294 "the placeholder wording has drifted from the fixture");
4295
4296 let cases = match extract_json_objects(&json, "cases") {
4297 Some(c) => c,
4298 None => panic!("the fixture has no `cases` array: {}", FOLD_FIXTURE),
4299 };
4300 // A fixture that stopped being read would pass every case it no longer had, and one that
4301 // LOST a case would pass just as quietly. The floor rises with the file: 11 at first
4302 // writing, 15 once nesting, the empty body and the lone angle bracket were pinned.
4303 assert!(cases.len() >= 15, "only {} cases read from {}", cases.len(), FOLD_FIXTURE);
4304
4305 let shut = OpenSet::new();
4306 for c in &cases {
4307 let name = extract_json_string(c, "name").unwrap_or_default();
4308 let input = match extract_json_string(c, "input") {
4309 Some(i) => i,
4310 None => panic!("case '{}' has no input", name),
4311 };
4312 let found = folds(&input);
4313
4314 let keys: Vec<String> = found.iter().map(|f| f.key()).collect();
4315 let want: Vec<String> = extract_json_string_array(c, "keys").unwrap_or_default();
4316 assert_eq!(keys, want, "case '{}': keys, from {:?}", name, input);
4317
4318 let chars: Vec<usize> = found.iter().map(|f| f.chars).collect();
4319 assert_eq!(chars, json_numbers(c, "body_chars"),
4320 "case '{}': body characters, from {:?}", name, input);
4321
4322 let got = strip_folds(&input, &shut);
4323 match extract_json_string(c, "stripped_all_closed") {
4324 Some(w) => assert_eq!(got.as_deref(), Some(w.as_str()),
4325 "case '{}': the strip", name),
4326 None => assert!(got.is_none(),
4327 "case '{}': the stripper rewrote a text it must leave exactly as it found \
4328 it.\n in: {:?}\n out: {:?}", name, input, got),
4329 }
4330 }
4331 }
4332
4333 /// The fixture the two halves of the SEAM are both tested against.
4334 const SEAM_FIXTURE: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/dev/fixtures/fold_seam.json");
4335
4336 /// **Every case in `dev/fixtures/fold_seam.json`, driven from the file itself.**
4337 ///
4338 /// Three assertions a case, and the second is the one with the risk in it. The expansion has
4339 /// to be exactly the text `www/js/render.js` builds, because a character between the two is a
4340 /// key the page and the engine disagree about and a fold the reader opens that never leaves
4341 /// the payload. `null` means the seam must leave the text exactly as it found it, which is
4342 /// what a seam that fires on a fenced line or on `Folder:` fails. And the keys are read off
4343 /// the RESULT, so a case proves the fold that comes out of the expansion and not just the
4344 /// string.
4345 #[test]
4346 fn test_every_case_in_the_shared_seam_fixture() {
4347 let json = match std::fs::read_to_string(SEAM_FIXTURE) {
4348 Ok(s) => s,
4349 Err(e) => panic!("the shared fixture must be readable at {}: {}", SEAM_FIXTURE, e),
4350 };
4351 let cases = match extract_json_objects(&json, "cases") {
4352 Some(c) => c,
4353 None => panic!("the fixture has no `cases` array: {}", SEAM_FIXTURE),
4354 };
4355 // A fixture that stopped being read would pass every case it no longer had.
4356 assert!(cases.len() >= 14, "only {} cases read from {}", cases.len(), SEAM_FIXTURE);
4357 for c in &cases {
4358 let name = extract_json_string(c, "name").unwrap_or_default();
4359 let input = match extract_json_string(c, "input") {
4360 Some(i) => i,
4361 None => panic!("case '{}' has no input", name),
4362 };
4363 let got = seam_text(&input);
4364 match extract_json_string(c, "seamed") {
4365 Some(w) => assert_eq!(got.as_deref(), Some(w.as_str()),
4366 "case '{}': the seam", name),
4367 None => assert!(got.is_none(),
4368 "case '{}': the seam rewrote a text it must leave exactly as it found \
4369 it.\n in: {:?}\n out: {:?}", name, input, got),
4370 }
4371 let after = got.unwrap_or(input.clone());
4372 let keys: Vec<String> = folds(&after).iter().map(|f| f.key()).collect();
4373 let want: Vec<String> = extract_json_string_array(c, "keys").unwrap_or_default();
4374 assert_eq!(keys, want, "case '{}': the keys of what came out", name);
4375 }
4376 }
4377
4378 /// **The seam refuses the two failures no wording could prevent.**
4379 ///
4380 /// `dev/PROMPT_NOTES.md` §5 measured a candidate wording that folded 8 answers in 8 and put
4381 /// nothing above the fold in 8 of 8 -- FOLD-ALL, which `dev/CONTRACT_FOLD.md` §5 calls worse
4382 /// than no control at all. A length threshold applied blindly does the same thing by another
4383 /// route. So the refusals are asserted as behaviour rather than left to the fixture's
4384 /// examples: below the lead, below the summary's words, below the body, nothing folds, and
4385 /// the reader still gets every word the model wrote.
4386 #[test]
4387 fn test_the_seam_refuses_rather_than_folding_everything() {
4388 let body = "x. ".repeat(120);
4389 let sum = "The store wins on scans and loses on isolation, so I take the file.";
4390 let lead = "Take one file per Diamond: isolation is worth more here than scan speed.";
4391 for (what, text) in [
4392 ("nothing above the seam", fmt!("Fold: {}\n\n{}", sum, body)),
4393 ("a lead of a few words", fmt!("Short.\n\nFold: {}\n\n{}", sum, body)),
4394 ("a label, not a summary", fmt!("{}\n\nFold: Reasoning\n\n{}", lead, body)),
4395 ("nothing below the seam", fmt!("{}\n\nFold: {}\n\nTiny.", lead, sum)),
4396 ] {
4397 let got = seam_text(&text).unwrap_or_else(|| text.clone());
4398 assert!(folds(&got).is_empty(), "{}: it folded anyway: {:?}", what, got);
4399 assert!(!got.contains("<details"), "{}: it built an element: {:?}", what, got);
4400 // Refused is not lost: every word the model wrote is still there, marker aside.
4401 for line in text.lines().filter(|l| !l.trim().is_empty()) {
4402 let want = line.trim_start_matches("Fold: ");
4403 assert!(got.contains(want), "{}: {:?} went missing from {:?}", what, want, got);
4404 }
4405 }
4406 // And the one that does qualify folds, or the four above would pass on a seam that never
4407 // fires at all.
4408 let good = fmt!("{}\n\nFold: {}\n\n{}", lead, sum, body);
4409 let got = match seam_text(&good) {
4410 Some(g) => g,
4411 None => panic!("a qualifying answer did not seam"),
4412 };
4413 let found = folds(&got);
4414 assert_eq!(found.len(), 1, "one fold, from {:?}", got);
4415 assert_eq!(found[0].key(), fmt!("0:{}", sum));
4416 // The blank lines CONTRACT_FOLD.md §1 calls mandatory, without which `marked` never
4417 // parses the markdown inside the element.
4418 assert!(got.contains(&fmt!("<summary>{}</summary>\n\n", sum)), "no blank line after the \
4419 summary: {:?}", got);
4420 assert!(got.contains("\n\n</details>"), "no blank line before the close: {:?}", got);
4421 }
4422
4423 /// **A `<details>` inside a fenced region is markup being SHOWN, not a fold.**
4424 ///
4425 /// Named rather than incidental because it is the likeliest wrong strip in the feature: a
4426 /// model quoting the convention to the user -- which the prompt now invites, since the prompt
4427 /// itself contains the markup -- would have its example silently edited out from under the
4428 /// person reading it, and the fake would steal the ordinal of the real fold below.
4429 ///
4430 /// Four shapes, and each has failed a stripper written without one of them: a fence with an
4431 /// info string, a tilde fence, a fence indented up to three spaces, and a fence INSIDE a
4432 /// real fold whose contents include a literal `</details>` that must not end the element.
4433 #[test]
4434 fn test_a_fold_inside_a_fenced_region_is_not_a_fold() {
4435 let shut = OpenSet::new();
4436 for (what, text) in [
4437 ("a backtick fence with an info string",
4438 "```html\n<details>\n<summary>Not a fold</summary>\n\nLiteral.\n\n</details>\n```\n"),
4439 ("a tilde fence",
4440 "~~~\n<details>\n<summary>Not a fold</summary>\n\nLiteral.\n\n</details>\n~~~\n"),
4441 ("a fence indented three spaces",
4442 " ```\n <details>\n <summary>Not a fold</summary>\n\n Literal.\n\n </details>\n ```\n"),
4443 ] {
4444 assert!(folds(text).is_empty(), "{} produced folds: {:?}", what, folds(text).len());
4445 assert!(strip_folds(text, &shut).is_none(), "{} was rewritten", what);
4446 }
4447
4448 // The fake takes no ordinal, so the real fold below it is fold ZERO.
4449 let mixed = "```\n<details>\n<summary>Fake</summary>\nx\n</details>\n```\n\n\
4450 <details>\n<summary>Real</summary>\n\nYes.\n\n</details>\n";
4451 let f = folds(mixed);
4452 assert_eq!(1, f.len(), "the fenced fake was counted as a fold");
4453 assert_eq!("0:Real", f[0].key(), "the fenced fake consumed an ordinal");
4454
4455 // A run carrying an INFO STRING opens a fence and can never close one, so a `\u{60}\u{60}\u{60}rust`
4456 // line part way through a code block does not end it and hand the rest of the answer back
4457 // to the scanner as prose.
4458 let info = "```\n<details>\n<summary>Fake</summary>\nx\n```rust\nstill fenced\n```\n\n\
4459 <details>\n<summary>Real</summary>\n\nYes.\n\n</details>\n";
4460 assert_eq!(vec![fmt!("0:Real")],
4461 folds(info).iter().map(|f| f.key()).collect::<Vec<_>>(),
4462 "an info string closed a fence it can only open");
4463
4464 // And a SHORTER run does not close a longer fence, which is how a model shows a fenced
4465 // block inside a fenced block.
4466 let longer = "````\n```\n<details>\n<summary>Fake</summary>\nx\n</details>\n```\n````\n\n\
4467 <details>\n<summary>Real</summary>\n\nYes.\n\n</details>\n";
4468 assert_eq!(vec![fmt!("0:Real")],
4469 folds(longer).iter().map(|f| f.key()).collect::<Vec<_>>(),
4470 "a three-backtick run closed a four-backtick fence");
4471
4472 // A fence INSIDE a fold: the literal `</details>` in the code block must not be taken for
4473 // this element's close, or the body is cut short and the remainder left dangling.
4474 let inner = "<details>\n<summary>How to write one</summary>\n\n\
4475 ```\n</details>\n```\n\nand that is the shape.\n\n</details>\n";
4476 let g = folds(inner);
4477 assert_eq!(1, g.len(), "the fold with a fence in it was lost");
4478 assert_eq!("0:How to write one", g[0].key());
4479 assert!(inner[g[0].body.clone()].ends_with("and that is the shape."),
4480 "the body stopped at the fenced `</details>`: {:?}", &inner[g[0].body.clone()]);
4481 }
4482
4483 /// **A nested fold belongs to its parent: no ordinal, no key, no strip of its own.**
4484 ///
4485 /// Pairing by the FIRST `</details>` rather than the matching one is the failure this guards.
4486 /// The browser's parser nests, so the renderer's idea of where the outer fold ends is the
4487 /// last tag and the scanner's was the first -- and the strip then replaced the wrong span,
4488 /// cutting the body short and leaving the tail of the element sitting in the payload it was
4489 /// meant to remove. The shared fixture measures it at 48 characters against the correct 67.
4490 ///
4491 /// The inner fold needs no key because the outer strip carries it away entirely: a key for a
4492 /// body that no longer exists is a fold the user can open to no effect.
4493 #[test]
4494 fn test_a_nested_fold_is_carried_by_its_parent() {
4495 let shut = OpenSet::new();
4496 let text = "<details>\n<summary>Outer</summary>\n\nbefore\n\n\
4497 <details>\n<summary>Inner</summary>\n\ndeep\n\n</details>\n\n\
4498 after\n\n</details>\n";
4499 let f = folds(text);
4500 assert_eq!(vec![fmt!("0:Outer")], f.iter().map(|x| x.key()).collect::<Vec<_>>(),
4501 "the inner fold took an ordinal of its own");
4502 // The body runs to the LAST closing tag, so it holds the whole inner element.
4503 let body = &text[f[0].body.clone()];
4504 assert!(body.starts_with("before") && body.ends_with("after"),
4505 "the outer body was cut at the inner fold's close: {:?}", body);
4506 assert!(body.contains("<summary>Inner</summary>"),
4507 "the inner element fell outside its parent's body: {:?}", body);
4508 // And the strip takes the whole of it, leaving one element where there were two.
4509 let out = match strip_folds(text, &shut) {
4510 Some(o) => o,
4511 None => panic!("the outer fold was not stripped"),
4512 };
4513 assert_eq!(1, out.matches("<details>").count(),
4514 "the inner element survived its parent's strip: {}", out);
4515 assert!(out.contains("folded to the user, 67 characters"), "{}", out);
4516
4517 // A nested pair consumes NOTHING, so the next top-level fold is ordinal one.
4518 let after = fmt!("{}\n<details>\n<summary>Sibling</summary>\n\nYes.\n\n</details>\n", text);
4519 assert_eq!(vec![fmt!("0:Outer"), fmt!("1:Sibling")],
4520 folds(&after).iter().map(|x| x.key()).collect::<Vec<_>>(),
4521 "the nested fold shifted the ordinal of the one after it");
4522
4523 // An unlabelled parent does NOT borrow its child's summary. It is not a fold, and neither
4524 // is the child, which is inside it -- so the answer is no folds rather than a fold whose
4525 // label names something else.
4526 let borrowed = "<details>\n\n<details>\n<summary>Inner</summary>\n\ndeep\n\n</details>\n\n</details>\n";
4527 assert!(folds(borrowed).is_empty(),
4528 "an unlabelled parent wore its child's label: {:?}",
4529 folds(borrowed).iter().map(|x| x.key()).collect::<Vec<_>>());
4530 assert!(strip_folds(borrowed, &shut).is_none());
4531 }
4532
4533 /// **A malformed fold is left exactly as the model wrote it.**
4534 ///
4535 /// [`strip_said`]'s rule, and the reason is the same: rewriting a malformed element replaces
4536 /// a problem the model can SEE -- its own broken markup, in its own transcript -- with one it
4537 /// cannot. An unclosed fold still keys, because its ordinal is settled the moment its summary
4538 /// is and the user may already have opened it; it is only the rewrite that stands off.
4539 #[test]
4540 fn test_a_malformed_fold_is_left_alone() {
4541 let shut = OpenSet::new();
4542 let unclosed = "<details>\n<summary>Partial</summary>\n\nStill being written";
4543 assert_eq!(vec![fmt!("0:Partial")],
4544 folds(unclosed).iter().map(|f| f.key()).collect::<Vec<_>>());
4545 assert!(strip_folds(unclosed, &shut).is_none(), "an unclosed fold was rewritten");
4546
4547 let unlabelled = "<details>\n\nNo summary here.\n\n</details>\n";
4548 assert!(folds(unlabelled).is_empty(), "a `<details>` with no `<summary>` keyed");
4549 assert!(strip_folds(unlabelled, &shut).is_none(), "an unlabelled fold was rewritten");
4550
4551 // A well-formed fold BESIDE a malformed one is still stripped: the leniency is per fold,
4552 // not per message, or one broken element would keep a whole answer on the wire forever.
4553 let both = fmt!("{}\n\nand then\n\n{}", unlabelled.trim_end(), unclosed);
4554 assert!(strip_folds(&both, &shut).is_none());
4555 let good = fmt!("<details>\n<summary>Good</summary>\n\nkept short.\n\n</details>\n\n{}",
4556 unclosed);
4557 let out = match strip_folds(&good, &shut) {
4558 Some(s) => s,
4559 None => panic!("the sound fold beside a broken one was not stripped"),
4560 };
4561 assert!(out.contains("folded to the user, 11 characters"), "{}", out);
4562 assert!(out.ends_with("Still being written"), "the broken fold was touched: {}", out);
4563
4564 // A fold with an EMPTY body is left alone too, and for the opposite reason: there is
4565 // nothing to save, and replacing nothing with a hundred characters of note would cost
4566 // tokens rather than save them.
4567 let hollow = "<details>\n<summary>Nothing in here</summary>\n\n</details>\n";
4568 assert_eq!(vec![fmt!("0:Nothing in here")],
4569 folds(hollow).iter().map(|f| f.key()).collect::<Vec<_>>());
4570 assert!(strip_folds(hollow, &shut).is_none(), "an empty fold grew a note: {:?}",
4571 strip_folds(hollow, &shut));
4572 }
4573
4574 /// **A `<` that opens no tag stays in the label, because the browser keeps it too.**
4575 ///
4576 /// The JS half reads the summary with `textContent`, which returns `a < b` unchanged. A
4577 /// stripper that treated every `<` as a tag opener would key that label `0:a` while the page
4578 /// keyed it `0:a < b` -- one label, two keys, and a fold the user opens that never travels.
4579 /// The contract says HTML TAGS removed; a `<` with no `>` after it is not one.
4580 #[test]
4581 fn test_a_lone_angle_bracket_in_a_summary_is_not_a_tag() {
4582 let text = "<details>\n<summary>when a < b</summary>\n\nBody.\n\n</details>\n";
4583 assert_eq!(vec![fmt!("0:when a < b")],
4584 folds(text).iter().map(|f| f.key()).collect::<Vec<_>>());
4585 // And a real tag is still removed, which is the half the fixture already pins.
4586 let tagged = "<details>\n<summary><em>when</em> a < b</summary>\n\nBody.\n\n</details>\n";
4587 assert_eq!(vec![fmt!("0:when a < b")],
4588 folds(tagged).iter().map(|f| f.key()).collect::<Vec<_>>());
4589 }
4590
4591 /// **The strip reaches ALL THREE serialisation sites.**
4592 ///
4593 /// `message_to_json` has two assistant branches -- with tool calls and without -- and
4594 /// `build_anthropic_body` has an assistant text path of its own. Missing one means the same
4595 /// conversation costs different amounts through different endpoints, silently, which is
4596 /// exactly the failure [`sent_args_len`]'s doc comment records for `say`.
4597 ///
4598 /// Asserted on the finished bodies of both dialects, not on the stripper: a stripper that
4599 /// works and is never called is the shape this defect takes.
4600 #[test]
4601 fn test_the_folded_body_leaves_by_every_serialisation_path() {
4602 use rustls::crypto::ring;
4603 let _ = ring::default_provider().install_default();
4604 let tls = Arc::new(ClientConfig::builder().dangerous()
4605 .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth());
4606 const BODY: &str = "THE-WORKING-BEHIND-THE-FOLD";
4607 let said = fmt!("Yes, it terminates.\n\n<details>\n<summary>the working</summary>\n\n\
4608 {}\n\n</details>\n", BODY);
4609 let msgs = vec![
4610 ChatMessage::user(fmt!("does it terminate?")),
4611 // The assistant branch with NO tool calls.
4612 ChatMessage::Assistant {
4613 content: MessageContent::text(said.clone()),
4614 tool_calls: Vec::new(),
4615 },
4616 ChatMessage::user(fmt!("and again?")),
4617 // The assistant branch WITH tool calls, which formats its text separately.
4618 ChatMessage::Assistant {
4619 content: MessageContent::text(said.clone()),
4620 tool_calls: vec![crate::protocol::ToolCall {
4621 id: fmt!("call_3"),
4622 name: fmt!("file_read"),
4623 arguments: fmt!("{{\"path\":\"a.txt\"}}"),
4624 }],
4625 },
4626 ChatMessage::tool(fmt!("call_3"), MessageContent::text(fmt!("ok"))),
4627 ];
4628 for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] {
4629 let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone());
4630 c.set_open_folds(Vec::new());
4631 let body = c.build_body(&msgs, None, false);
4632 assert_eq!(0, body.matches(BODY).count(),
4633 "a closed fold's body is still on the wire via {}: {}", path, body);
4634 assert_eq!(2, body.matches("folded to the user").count(),
4635 "via {} only {} of the two assistant messages left a note, so one \
4636 serialisation path does not strip: {}",
4637 path, body.matches("folded to the user").count(), body);
4638 // The element is KEPT: a model that sees the fold it wrote still knows it folded
4639 // something and what it called it.
4640 assert_eq!(2, body.matches("<summary>the working</summary>").count(),
4641 "the element was deleted rather than emptied, via {}: {}", path, body);
4642 assert!(body.contains("Yes, it terminates."),
4643 "the short answer above the fold was stripped too, via {}: {}", path, body);
4644
4645 // And the key JS sends is the key Rust matches: open it and the body travels.
4646 c.set_open_folds(vec![fmt!("0:the working")]);
4647 assert_eq!(2, c.build_body(&msgs, None, false).matches(BODY).count(),
4648 "an OPEN fold was withheld via {}, so the model cannot see what the user is \
4649 reading", path);
4650 }
4651
4652 // The cache-marked serialisation delegates the assistant role rather than repeating it,
4653 // and this is the assertion that keeps that true.
4654 let shut = OpenSet::new();
4655 assert_eq!(message_to_json(&msgs[1], &shut), message_to_json_cached(&msgs[1], &shut),
4656 "the cached path grew an assistant branch of its own");
4657 }
4658
4659 /// **What the compaction trigger measures is the text the wire will carry.**
4660 ///
4661 /// Defect F, one depth further in. [`crate::agent::compact::msg_bytes`] was taught to ask
4662 /// [`sent_args_len`] what a `say` costs; an inline fold folds the assistant's own prose
4663 /// instead, and a sizer that knew about one and not the other would leave the same defect
4664 /// standing with a new name.
4665 ///
4666 /// The second assertion is the one that matters. A closed fold sizing smaller than an open
4667 /// one is satisfied by any discount at all; that the discount is exactly what serialisation
4668 /// saves is satisfied only by asking the serialiser.
4669 #[test]
4670 fn test_the_trigger_sizes_the_folded_text_the_wire_will_carry() {
4671 use crate::agent::compact::conversation_bytes;
4672 use rustls::crypto::ring;
4673 let _ = ring::default_provider().install_default();
4674 let tls = Arc::new(ClientConfig::builder().dangerous()
4675 .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth());
4676 // Plain letters and spaces on both sides of the swap, so the two bodies differ by the
4677 // fold's body and by nothing an escape would change.
4678 let working = "the long working behind the fold ".repeat(40);
4679 let said = fmt!("Yes.\n\n<details>\n<summary>the working</summary>\n\n{}\n\n</details>\n",
4680 working.trim());
4681 let msgs = vec![
4682 ChatMessage::user(fmt!("explain")),
4683 ChatMessage::Assistant {
4684 content: MessageContent::text(said),
4685 tool_calls: Vec::new(),
4686 },
4687 ];
4688 let shut = OpenSet::new();
4689 let open: OpenSet = [fmt!("0:the working")].into_iter().collect();
4690 let sized_shut = conversation_bytes(&msgs, &shut);
4691 let sized_open = conversation_bytes(&msgs, &open);
4692
4693 for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] {
4694 let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone());
4695 c.set_open_folds(Vec::new());
4696 let wire_shut = c.build_body(&msgs, None, false).len() as u64;
4697 c.set_open_folds(vec![fmt!("0:the working")]);
4698 let wire_open = c.build_body(&msgs, None, false).len() as u64;
4699
4700 assert!(wire_shut < wire_open, "the fixture proves nothing via {}: closing the fold \
4701 did not shrink the payload", path);
4702 assert!(sized_shut < sized_open,
4703 "a closed fold is sized as though its body were still sent, so the trigger folds \
4704 a conversation of {} bytes that goes out as {} (via {})",
4705 sized_shut, wire_shut, path);
4706 assert_eq!(sized_open - sized_shut, wire_open - wire_shut,
4707 "the sizer books {} bytes for closing the fold and the wire saves {} (via {}), \
4708 so the trigger is measuring a rule of its own rather than the serialiser's",
4709 sized_open - sized_shut, wire_open - wire_shut, path);
4710 }
4711 }
4712
4713 #[test]
4714 fn test_extract_json_string() {
4715 let json = r#"{"choices":[{"delta":{"content":"hello"}}]}"#;
4716 assert_eq!(extract_json_string(json, "content"), Some("hello".to_string()));
4717 }
4718
4719 #[test]
4720 fn test_extract_json_bool() {
4721 assert_eq!(extract_json_bool(r#"{"submit":true}"#, "submit"), Some(true));
4722 assert_eq!(extract_json_bool(r#"{"submit": false}"#, "submit"), Some(false));
4723 // A model that quotes the boolean is still understood.
4724 assert_eq!(extract_json_bool(r#"{"submit":"true"}"#, "submit"), Some(true));
4725 assert_eq!(extract_json_bool(r#"{"ref":3}"#, "submit"), None);
4726 }
4727
4728 #[test]
4729 fn test_extract_json_f64() {
4730 // The case that made a reported cost read as free: `extract_json_number`
4731 // stops at the '.', so `0.0021` was 0.
4732 assert_eq!(extract_json_number(r#"{"cost":0.0021}"#, "cost"), Some(0));
4733 assert_eq!(extract_json_f64(r#"{"cost":0.0021}"#, "cost"), Some(0.0021));
4734 // Whitespace, exponents both ways, a negative, and a quoted figure.
4735 assert_eq!(extract_json_f64(r#"{"cost": 1.5}"#, "cost"), Some(1.5));
4736 assert_eq!(extract_json_f64(r#"{"cost":2.1e-5}"#, "cost"), Some(2.1e-5));
4737 assert_eq!(extract_json_f64(r#"{"cost":3E+2}"#, "cost"), Some(300.0));
4738 assert_eq!(extract_json_f64(r#"{"cost":-0.5,"x":1}"#, "cost"), Some(-0.5));
4739 assert_eq!(extract_json_f64(r#"{"cost":"0.0021"}"#, "cost"), Some(0.0021));
4740 // A whole number is still a number, and an absent key is still absent.
4741 assert_eq!(extract_json_f64(r#"{"cost":0}"#, "cost"), Some(0.0));
4742 assert_eq!(extract_json_f64(r#"{"total":1.0}"#, "cost"), None);
4743 // A longer key that merely ends in the wanted one is not it.
4744 assert_eq!(extract_json_f64(r#"{"upstream_inference_cost":9.0}"#, "cost"), None);
4745 }
4746
4747 #[test]
4748 fn test_parse_usage_openrouter() {
4749 // The shape OpenRouter actually returns: authoritative cost, and the
4750 // cache read nested under `prompt_tokens_details`.
4751 let body = r#"{"id":"gen-1","choices":[{"message":{"content":"hi"}}],"usage":{"prompt_tokens":10240,"completion_tokens":128,"total_tokens":10368,"cost":0.0021,"cost_details":{"upstream_inference_cost":null},"prompt_tokens_details":{"cached_tokens":9216},"completion_tokens_details":{"reasoning_tokens":0}}}"#;
4752 let u = match parse_usage(body) {
4753 Some(u) => u,
4754 None => panic!("usage not found"),
4755 };
4756 assert_eq!(u.prompt, 10240);
4757 assert_eq!(u.completion, 128);
4758 assert_eq!(u.cached, 9216);
4759 assert_eq!(u.cost_usd, 0.0021);
4760 }
4761
4762 #[test]
4763 fn test_parse_usage_absent_and_null() {
4764 // No usage at all, and the `"usage":null` every intermediate streamed
4765 // chunk carries: both must read as absent, so the usage chunk that came
4766 // before is not erased by the chunk that follows it.
4767 assert!(parse_usage(r#"{"choices":[{"delta":{"content":"x"}}]}"#).is_none());
4768 assert!(parse_usage(r#"{"choices":[{"delta":{}}],"usage":null}"#).is_none());
4769 // A provider reporting only tokens leaves cost and cache at zero, which
4770 // is "it did not say", never "it was free".
4771 let u = match parse_usage(r#"{"usage":{"prompt_tokens":4,"completion_tokens":2}}"#) {
4772 Some(u) => u,
4773 None => panic!("usage not found"),
4774 };
4775 assert_eq!(u.cached, 0);
4776 assert_eq!(u.cost_usd, 0.0);
4777 }
4778
4779 #[test]
4780 fn test_parse_usage_anthropic_native_cache_read() {
4781 // Anthropic's own name for the figure. A prompt cache that is working
4782 // must not read as one that is not, or the breakpoint looks inert.
4783 let u = match parse_usage(
4784 r#"{"usage":{"prompt_tokens":100,"cache_read_input_tokens":80}}"#) {
4785 Some(u) => u,
4786 None => panic!("usage not found"),
4787 };
4788 assert_eq!(u.cached, 80);
4789 }
4790
4791 #[test]
4792 fn test_parse_usage_flat_cached() {
4793 // A provider that flattens the cache read onto `usage` is read too.
4794 let u = match parse_usage(r#"{"usage":{"prompt_tokens":100,"cached_tokens":80}}"#) {
4795 Some(u) => u,
4796 None => panic!("usage not found"),
4797 };
4798 assert_eq!(u.cached, 80);
4799 }
4800
4801 #[test]
4802 fn test_extract_json_string_escaped() {
4803 let json = r#"{"choices":[{"delta":{"content":"hello \"world\""}}]}"#;
4804 assert_eq!(extract_json_string(json, "content"), Some("hello \"world\"".to_string()));
4805 }
4806
4807 #[test]
4808 fn test_extract_json_string_newline() {
4809 let json = r#"{"choices":[{"delta":{"content":"line1\nline2"}}]}"#;
4810 assert_eq!(extract_json_string(json, "content"), Some("line1\nline2".to_string()));
4811 }
4812
4813 #[test]
4814 fn test_parse_sse_simple() {
4815 let sse = "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\ndata: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\ndata: [DONE]\n";
4816 let mut tokens = Vec::new();
4817 let (full, _use) = parse_sse_stream(sse.as_bytes(), &mut |t| tokens.push(t.to_string()));
4818 assert_eq!(tokens, vec!["Hello", " world"]);
4819 assert_eq!(full, "Hello world");
4820 }
4821
4822 #[test]
4823 fn test_parse_sse_empty_lines() {
4824 let sse = "\r\ndata: {\"choices\":[{\"delta\":{\"content\":\"Hi\"}}]}\r\n\r\ndata: [DONE]\r\n";
4825 let mut tokens = Vec::new();
4826 let (full, _use) = parse_sse_stream(sse.as_bytes(), &mut |t| tokens.push(t.to_string()));
4827 assert_eq!(tokens, vec!["Hi"]);
4828 assert_eq!(full, "Hi");
4829 }
4830
4831 // Chunked transfer decoding is now handled inline by `LineReader`;
4832 // the standalone `dechunk` helper and its tests were removed.
4833
4834 #[test]
4835 fn test_parse_full_response_tool_calls() {
4836 let body = r#"{"choices":[{"index":0,"message":{"role":"assistant","content":null,"tool_calls":[{"id":"call_1","type":"function","function":{"name":"file_read","arguments":"{\"path\":\"a.txt\"}"}}]},"finish_reason":"tool_calls"}],"usage":{"prompt_tokens":12,"completion_tokens":8}}"#;
4837 let (content, calls, use_) = parse_full_response(body);
4838 assert_eq!(content, "");
4839 assert_eq!(calls.len(), 1);
4840 assert_eq!(calls[0].id, "call_1");
4841 assert_eq!(calls[0].name, "file_read");
4842 assert_eq!(calls[0].arguments, r#"{"path":"a.txt"}"#);
4843 assert_eq!(use_.prompt, 12);
4844 assert_eq!(use_.completion, 8);
4845 }
4846
4847 #[test]
4848 fn test_extract_json_string_whitespace() {
4849 // Real model output has a space after the colon.
4850 assert_eq!(extract_json_string(r#"{"path": "a.txt"}"#, "path"), Some("a.txt".to_string()));
4851 assert_eq!(extract_json_string(r#"{ "content": "hi" }"#, "content"), Some("hi".to_string()));
4852 // A null value is not a string.
4853 assert_eq!(extract_json_string(r#"{"content": null, "x":"y"}"#, "content"), None);
4854 }
4855
4856 #[test]
4857 fn test_parse_full_response_spaced() {
4858 // Whitespace after colons, as real APIs emit.
4859 let body = r#"{"choices": [{"message": {"content": null, "tool_calls": [{"id": "c1", "type": "function", "function": {"name": "file_write", "arguments": "{\"path\": \"a.txt\", \"content\": \"hi\"}"}}]}}], "usage": {"prompt_tokens": 4, "completion_tokens": 2}}"#;
4860 let (content, calls, use_) = parse_full_response(body);
4861 assert_eq!(content, "");
4862 assert_eq!(calls.len(), 1);
4863 assert_eq!(calls[0].name, "file_write");
4864 assert_eq!(calls[0].arguments, r#"{"path": "a.txt", "content": "hi"}"#);
4865 assert_eq!(use_.prompt, 4);
4866 assert_eq!(use_.completion, 2);
4867 // And the tool can extract the spaced args.
4868 assert_eq!(extract_json_string(&calls[0].arguments, "path"), Some("a.txt".to_string()));
4869 }
4870
4871 #[test]
4872 fn test_parse_full_response_text() {
4873 let body = r#"{"choices":[{"message":{"role":"assistant","content":"Hello there."},"finish_reason":"stop"}],"usage":{"prompt_tokens":5,"completion_tokens":3}}"#;
4874 let (content, calls, use_) = parse_full_response(body);
4875 assert_eq!(content, "Hello there.");
4876 assert!(calls.is_empty());
4877 assert_eq!(use_.prompt, 5);
4878 assert_eq!(use_.completion, 3);
4879 }
4880
4881 #[test]
4882 fn test_parse_full_response_two_calls() {
4883 let body = r#"{"choices":[{"message":{"content":null,"tool_calls":[{"id":"c1","type":"function","function":{"name":"file_list","arguments":"{}"}},{"id":"c2","type":"function","function":{"name":"shell","arguments":"{\"command\":\"ls\"}"}}]}}]}"#;
4884 let (_c, calls, _use) = parse_full_response(body);
4885 assert_eq!(calls.len(), 2);
4886 assert_eq!(calls[0].name, "file_list");
4887 assert_eq!(calls[1].name, "shell");
4888 assert_eq!(calls[1].arguments, r#"{"command":"ls"}"#);
4889 }
4890
4891 /// A sink that keeps the ANSWER and throws the working away.
4892 ///
4893 /// What nearly every check in this module is about: the answer is what gets
4894 /// persisted and sent back next turn, so a check that let reasoning into the
4895 /// same vector would pass on a client that confused the two.
4896 fn text_sink(out: &mut Vec<String>) -> impl FnMut(Delta<'_>) + '_ {
4897 move |d| if let Delta::Text(t) = d { out.push(t.to_string()); }
4898 }
4899
4900 /// The other half: the working, kept and the answer thrown away.
4901 fn think_sink(out: &mut Vec<String>) -> impl FnMut(Delta<'_>) + '_ {
4902 move |d| if let Delta::Reasoning(t) = d { out.push(t.to_string()); }
4903 }
4904
4905 /// Drive a sequence of SSE `data:` payloads through a fresh
4906 /// [`StreamAcc`], collecting the forwarded text tokens.
4907 fn run_stream(chunks: &[&str]) -> (ChatOnceResponse, Vec<String>) {
4908 let mut acc = StreamAcc::default();
4909 let mut tokens = Vec::new();
4910 for c in chunks {
4911 acc.ingest(c, &mut text_sink(&mut tokens));
4912 }
4913 (acc.into_response(false, 0), tokens)
4914 }
4915
4916 /// Drive the same payloads and keep BOTH sides, so a check can say which sink
4917 /// each piece reached rather than only that it arrived somewhere.
4918 fn run_stream_both(chunks: &[&str]) -> (ChatOnceResponse, Vec<String>, Vec<String>) {
4919 let mut acc = StreamAcc::default();
4920 let mut tokens = Vec::new();
4921 let mut thought = Vec::new();
4922 for c in chunks {
4923 acc.ingest(c, &mut |d: Delta<'_>| match d {
4924 Delta::Text(t) => tokens.push(t.to_string()),
4925 Delta::Reasoning(t) => thought.push(t.to_string()),
4926 });
4927 }
4928 (acc.into_response(false, 0), tokens, thought)
4929 }
4930
4931 #[test]
4932 fn test_the_working_of_an_openai_dialect_model_reaches_the_page_as_it_arrives() {
4933 // Captured from OpenRouter on 2026-08-28, `z-ai/glm-4.6` by way of DeepInfra:
4934 // the reasoning is on `delta.reasoning` and repeated VERBATIM inside
4935 // `delta.reasoning_details`, and `content` is an empty string throughout it.
4936 // A round of this model spent 230 of its 300 output tokens here, and the app
4937 // showed a spinner for all of them.
4938 let (resp, tokens, thought) = run_stream_both(&[
4939 r#"{"choices":[{"delta":{"content":"","role":"assistant","reasoning":"1","reasoning_details":[{"type":"reasoning.text","text":"1","format":"unknown","index":0}]}}]}"#,
4940 r#"{"choices":[{"delta":{"content":"","role":"assistant","reasoning":"7 x 23","reasoning_details":[{"type":"reasoning.text","text":"7 x 23","format":"unknown","index":0}]}}]}"#,
4941 r#"{"choices":[{"delta":{"content":"391","role":"assistant","reasoning":null}}]}"#,
4942 r#"{"choices":[{"delta":{},"finish_reason":"stop"}]}"#,
4943 ]);
4944 // ONE copy of each piece. `reasoning_details` says the same words again, so a
4945 // reader that took both would show every token twice.
4946 assert_eq!(thought, vec!["1", "7 x 23"],
4947 "the model's working was dropped, or doubled by reasoning_details: {:?}", thought);
4948 // A `null` reasoning field is the provider saying there is none this chunk.
4949 assert_eq!(tokens, vec!["391"], "reasoning reached the answer sink: {:?}", tokens);
4950 assert_eq!(resp.content, "391", "reasoning was accumulated as the reply");
4951 assert_eq!(resp.thinking, "17 x 23",
4952 "the round's working was not kept on the response");
4953 }
4954
4955 #[test]
4956 fn test_deepseeks_own_spelling_of_its_working_is_read_too() {
4957 // `reasoning_content` is what DeepSeek's own endpoint calls it; `reasoning` is
4958 // OpenRouter's. Both are wanted -- Daimond reaches DeepSeek both ways -- and
4959 // never both in one delta, which is why one is read and then the other.
4960 let (resp, tokens, thought) = run_stream_both(&[
4961 r#"{"choices":[{"delta":{"role":"assistant","content":null,"reasoning_content":"So the"}}]}"#,
4962 r#"{"choices":[{"delta":{"content":null,"reasoning_content":" answer is"}}]}"#,
4963 r#"{"choices":[{"delta":{"content":"391","reasoning_content":null}}]}"#,
4964 ]);
4965 assert_eq!(thought, vec!["So the", " answer is"], "{:?}", thought);
4966 assert_eq!(tokens, vec!["391"], "{:?}", tokens);
4967 assert_eq!(resp.thinking, "So the answer is");
4968 assert_eq!(resp.content, "391");
4969 }
4970
4971 #[test]
4972 fn test_a_models_working_is_never_stored_as_what_it_said() {
4973 // THE DEFECT THIS WHOLE PATH IS ONE MISTAKE AWAY FROM. The reply is what gets
4974 // written into the transcript and sent back to the model next turn as its own
4975 // words. Reasoning put there is the model's working out quoted back to it as
4976 // its answer -- and the working of a tool round is mostly wrong turns.
4977 let (resp, tokens, thought) = run_stream_both(&[
4978 r#"{"choices":[{"delta":{"content":"","reasoning":"Maybe I should delete it."}}]}"#,
4979 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"call_1","function":{"name":"file_read","arguments":"{}"}}]}}]}"#,
4980 ]);
4981 assert!(resp.content.is_empty(),
4982 "the working was accumulated as the reply: {:?}", resp.content);
4983 assert!(tokens.is_empty(), "the working reached the answer sink: {:?}", tokens);
4984 assert_eq!(thought, vec!["Maybe I should delete it."]);
4985 assert_eq!(resp.tool_calls.len(), 1, "the tool call was lost");
4986 }
4987
4988 #[test]
4989 fn test_a_reasoning_key_inside_tool_arguments_is_not_the_models_working() {
4990 // A model writing JSON about reasoning is not reasoning. The argument text is
4991 // an escaped string, so the key form never matches -- asserted rather than
4992 // assumed, because the same trap already caught `content` once.
4993 let (resp, tokens, thought) = run_stream_both(&[
4994 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c1","function":{"name":"note_write","arguments":"{\"reasoning\":\"not mine\",\"content\":\"nor this\"}"}}]}}]}"#,
4995 ]);
4996 assert!(thought.is_empty(), "a tool argument was read as reasoning: {:?}", thought);
4997 assert!(tokens.is_empty(), "a tool argument was read as text: {:?}", tokens);
4998 assert_eq!(resp.tool_calls[0].arguments,
4999 r#"{"reasoning":"not mine","content":"nor this"}"#);
5000 }
5001
5002 #[test]
5003 fn test_stream_acc_text_only() {
5004 let (resp, tokens) = run_stream(&[
5005 r#"{"choices":[{"delta":{"role":"assistant","content":"Hel"}}]}"#,
5006 r#"{"choices":[{"delta":{"content":"lo!"}}]}"#,
5007 r#"{"choices":[{"delta":{}}],"usage":{"prompt_tokens":7,"completion_tokens":3}}"#,
5008 ]);
5009 assert_eq!(tokens, vec!["Hel", "lo!"]);
5010 assert_eq!(resp.content, "Hello!");
5011 assert!(resp.tool_calls.is_empty());
5012 assert_eq!(resp.prompt_tokens, 7);
5013 assert_eq!(resp.completion_tokens, 3);
5014 assert!(!resp.aborted);
5015 // Nothing said about cost or caching, so nothing is claimed.
5016 assert_eq!(resp.cached_tokens, 0);
5017 assert_eq!(resp.cost_usd, 0.0);
5018 }
5019
5020 #[test]
5021 fn test_stream_acc_reported_cost_survives_later_chunks() {
5022 // The usage chunk arrives, and a `"usage":null` chunk follows it before
5023 // `[DONE]`. The reported figures must survive that.
5024 let (resp, _tokens) = run_stream(&[
5025 r#"{"choices":[{"delta":{"content":"ok"}}],"usage":null}"#,
5026 r#"{"choices":[],"usage":{"prompt_tokens":8192,"completion_tokens":64,"cost":0.0021,"prompt_tokens_details":{"cached_tokens":7168}}}"#,
5027 r#"{"choices":[{"delta":{}}],"usage":null}"#,
5028 ]);
5029 assert_eq!(resp.prompt_tokens, 8192);
5030 assert_eq!(resp.cached_tokens, 7168);
5031 assert_eq!(resp.cost_usd, 0.0021);
5032 }
5033
5034 #[test]
5035 fn test_stream_acc_tool_call_fragments() {
5036 // The name arrives with the first fragment; the arguments are split
5037 // across two later fragments and must be concatenated verbatim.
5038 let (resp, tokens) = run_stream(&[
5039 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"call_1","type":"function","function":{"name":"file_read","arguments":""}}]}}]}"#,
5040 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"{\"path\":\""}}]}}]}"#,
5041 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"a.txt\"}"}}]}}]}"#,
5042 r#"{"choices":[{"delta":{},"finish_reason":"tool_calls"}],"usage":{"prompt_tokens":12,"completion_tokens":8}}"#,
5043 ]);
5044 assert!(tokens.is_empty());
5045 assert_eq!(resp.tool_calls.len(), 1);
5046 assert_eq!(resp.tool_calls[0].id, "call_1");
5047 assert_eq!(resp.tool_calls[0].name, "file_read");
5048 assert_eq!(resp.tool_calls[0].arguments, r#"{"path":"a.txt"}"#);
5049 assert_eq!(resp.prompt_tokens, 12);
5050 assert_eq!(resp.completion_tokens, 8);
5051 }
5052
5053 #[test]
5054 fn test_stream_acc_two_parallel_calls() {
5055 // Two calls interleaved by index across chunks.
5056 let (resp, _t) = run_stream(&[
5057 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c0","function":{"name":"file_list","arguments":"{}"}}]}}]}"#,
5058 r#"{"choices":[{"delta":{"tool_calls":[{"index":1,"id":"c1","function":{"name":"file_read","arguments":"{\"path\":"}}]}}]}"#,
5059 r#"{"choices":[{"delta":{"tool_calls":[{"index":1,"function":{"arguments":"\"b.txt\"}"}}]}}]}"#,
5060 ]);
5061 assert_eq!(resp.tool_calls.len(), 2);
5062 assert_eq!(resp.tool_calls[0].name, "file_list");
5063 assert_eq!(resp.tool_calls[0].arguments, "{}");
5064 assert_eq!(resp.tool_calls[1].name, "file_read");
5065 assert_eq!(resp.tool_calls[1].arguments, r#"{"path":"b.txt"}"#);
5066 }
5067
5068 #[test]
5069 fn test_stream_acc_text_then_tool_call() {
5070 // Interim assistant text streams, then a tool call is requested.
5071 let (resp, tokens) = run_stream(&[
5072 r#"{"choices":[{"delta":{"content":"Let me check. "}}]}"#,
5073 r#"{"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c0","function":{"name":"file_list","arguments":"{}"}}]}}]}"#,
5074 ]);
5075 assert_eq!(tokens, vec!["Let me check. "]);
5076 assert_eq!(resp.content, "Let me check. ");
5077 assert_eq!(resp.tool_calls.len(), 1);
5078 assert_eq!(resp.tool_calls[0].name, "file_list");
5079 }
5080
5081 #[test]
5082 fn test_message_to_json_assistant_tool_calls() {
5083 let msg = ChatMessage::Assistant {
5084 content: MessageContent::text(""),
5085 tool_calls: vec![ToolCall {
5086 id: "c1".to_string(),
5087 name: "shell".to_string(),
5088 arguments: r#"{"command":"ls"}"#.to_string(),
5089 }],
5090 };
5091 let j = message_to_json(&msg, &std::collections::HashSet::new());
5092 assert!(j.contains(r#""role":"assistant""#));
5093 assert!(j.contains(r#""tool_calls""#));
5094 assert!(j.contains(r#""name":"shell""#));
5095 // Arguments must be re-escaped as a JSON string literal.
5096 assert!(j.contains(r#""arguments":"{\"command\":\"ls\"}""#));
5097 }
5098
5099 #[test]
5100 fn test_datmap_to_json() {
5101 let mut m = DaticleMap::new();
5102 m.insert(dat!("role"), dat!("user"));
5103 m.insert(dat!("content"), dat!("hello"));
5104 let json = datmap_to_json(&m);
5105 // Keys are sorted.
5106 assert!(json.contains("\"content\":\"hello\""));
5107 assert!(json.contains("\"role\":\"user\""));
5108 }
5109
5110 #[test]
5111 fn test_datmap_to_json_escaped() {
5112 let mut m = DaticleMap::new();
5113 m.insert(dat!("content"), dat!("hello \"world\"\n"));
5114 let json = datmap_to_json(&m);
5115 assert!(json.contains("\\\"world\\\""));
5116 assert!(json.contains("\\n"));
5117 }
5118
5119 /// **The detail a stored `say` folds does not go over the wire, and the summary does.**
5120 ///
5121 /// This was the whole point of the tool, and the tool is gone — so this is now the guard on
5122 /// what remains of it. A conversation saved before the `<details>` convention still carries
5123 /// `say` tool_calls, and every one of them is re-sent on every later request for the life of
5124 /// that conversation. Delete the stripper with the tool and nothing on screen changes: those
5125 /// answers simply start travelling in full again, and the bill goes up on the conversations
5126 /// the feature existed to make cheap. The message below is exactly that — an assistant turn
5127 /// out of an old transcript, built by NAME rather than through any `Tool`, because there is no
5128 /// longer a variant to build it from.
5129 ///
5130 /// BOTH DIALECTS, because they serialise a call in ways that look nothing alike: one escapes
5131 /// the arguments into a JSON string, the other embeds them as an object. A rule applied at one
5132 /// site and not the other means the same conversation costs different amounts through
5133 /// different endpoints, and nothing on screen would say so.
5134 ///
5135 /// And a NON-`say` call is asserted to keep its arguments, which is what stops this from being
5136 /// a stripper aimed at everything: `file_write`'s content has to survive, or a write replayed
5137 /// to the model becomes a write of a placeholder.
5138 #[test]
5139 fn test_a_folded_detail_never_reaches_the_wire() {
5140 use rustls::crypto::ring;
5141 let _ = ring::default_provider().install_default();
5142 let tls = Arc::new(
5143 ClientConfig::builder()
5144 .dangerous()
5145 .with_custom_certificate_verifier(Arc::new(NoVerify))
5146 .with_no_client_auth()
5147 );
5148 const DETAIL: &str = "THE-LONG-EXPLANATION-NOBODY-SHOULD-RESEND";
5149 const GIST: &str = "the fence is a path allow-list";
5150 let msgs = vec![
5151 ChatMessage::user("explain the fence".to_string()),
5152 ChatMessage::Assistant {
5153 content: MessageContent::text(String::new()),
5154 tool_calls: vec![
5155 crate::protocol::ToolCall {
5156 id: fmt!("c1"),
5157 name: fmt!("say"),
5158 arguments: fmt!("{{\"summary\":\"{}\",\"detail\":\"{}\"}}", GIST, DETAIL),
5159 },
5160 crate::protocol::ToolCall {
5161 id: fmt!("c2"),
5162 name: fmt!("file_write"),
5163 arguments: fmt!("{{\"path\":\"a.md\",\"content\":\"{}\"}}", DETAIL),
5164 },
5165 ],
5166 },
5167 ];
5168 for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] {
5169 let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone());
5170 let body = c.build_body(&msgs, None, false);
5171 assert!(!body.contains(DETAIL) || body.matches(DETAIL).count() == 1,
5172 "the folded detail is still on the wire via {}: {}", path, body);
5173 // Exactly once — carried by `file_write`, never by `say`.
5174 assert_eq!(1, body.matches(DETAIL).count(),
5175 "via {} the detail appears {} times; it must survive file_write and never say",
5176 path, body.matches(DETAIL).count());
5177 assert!(body.contains(GIST), "the summary was stripped too, via {}: {}", path, body);
5178 assert!(body.contains("folded to the user"),
5179 "nothing tells the model what became of the detail, via {}: {}", path, body);
5180 }
5181 }
5182
5183 /// **An OPEN fold travels; a closed one does not.**
5184 ///
5185 /// The user's own gesture decides the model's working set. A fold they have closed is one they
5186 /// are done with, and re-sending it every turn buys nothing; a fold they have OPEN is one they
5187 /// are reading, and the next thing they say is likely to be about it — so the model holds what
5188 /// they are looking at. Two controls for one idea would be one control too many.
5189 ///
5190 /// Asserted BOTH WAYS from the same message, because either half alone is satisfied by a
5191 /// stripper that is simply broken: always-strip passes the closed case, never-strip passes the
5192 /// open one.
5193 #[test]
5194 fn test_an_open_fold_travels_and_a_closed_one_does_not() {
5195 use rustls::crypto::ring;
5196 let _ = ring::default_provider().install_default();
5197 let tls = Arc::new(ClientConfig::builder().dangerous()
5198 .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth());
5199 const DETAIL: &str = "THE-DETAIL-BEHIND-THE-FOLD";
5200 let msgs = vec![
5201 ChatMessage::user("explain".to_string()),
5202 ChatMessage::Assistant {
5203 content: MessageContent::text(String::new()),
5204 tool_calls: vec![crate::protocol::ToolCall {
5205 id: fmt!("call_7"),
5206 name: fmt!("say"),
5207 arguments: fmt!("{{\"summary\":\"the gist\",\"detail\":\"{}\"}}", DETAIL),
5208 }],
5209 },
5210 ];
5211 for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] {
5212 let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone());
5213
5214 c.set_open_folds(Vec::new());
5215 assert!(!c.build_body(&msgs, None, false).contains(DETAIL),
5216 "a CLOSED fold was sent via {}", path);
5217
5218 c.set_open_folds(vec![fmt!("call_7")]);
5219 assert!(c.build_body(&msgs, None, false).contains(DETAIL),
5220 "an OPEN fold was withheld via {}, so the model cannot see what the user is \
5221 reading", path);
5222
5223 // And closing it again takes it back out, which is what makes this a control rather
5224 // than a one-way door.
5225 c.set_open_folds(vec![fmt!("some_other_call")]);
5226 assert!(!c.build_body(&msgs, None, false).contains(DETAIL),
5227 "closing a fold did not take it back out of the payload, via {}", path);
5228 }
5229 }
5230
5231 /// **What the compaction trigger measures is what the wire will carry.**
5232 ///
5233 /// [`crate::agent::compact::msg_bytes`] sized a `say` with `tc.arguments.len()` -- the full
5234 /// call as the model wrote it -- while [`strip_said`] takes a closed fold's detail out at
5235 /// serialisation. So the trigger measured a conversation nobody was going to send, spent the
5236 /// budget on bytes that leave on the way out, and folded earlier than it needed to. The
5237 /// [`Gauge`](crate::agent::compact::Gauge) absorbed part of that by recalibrating
5238 /// tokens-per-byte against the provider's real `prompt_tokens`, but the ratio is one number
5239 /// for the whole conversation, so the correction was paid for by distorting every other
5240 /// message's estimate.
5241 ///
5242 /// Two things are asserted and the second is the one that matters. A closed fold sizing
5243 /// smaller than an open one is satisfied by ANY discount, arbitrary or not; that the discount
5244 /// is exactly the number of bytes serialisation actually saves is satisfied only by asking
5245 /// the serialiser, which is what [`sent_args_len`] does.
5246 ///
5247 /// Both dialects, because the strip applies to both and a sizing that matched one of them
5248 /// would mean the same conversation folded at different lengths through different endpoints.
5249 #[test]
5250 fn test_the_fold_trigger_sizes_what_the_wire_will_carry() {
5251 use crate::agent::compact::conversation_bytes;
5252 use rustls::crypto::ring;
5253 let _ = ring::default_provider().install_default();
5254 let tls = Arc::new(ClientConfig::builder().dangerous()
5255 .with_custom_certificate_verifier(Arc::new(NoVerify)).with_no_client_auth());
5256 // Plain letters and spaces: nothing here is escaped differently from the note that
5257 // replaces it, so the two bodies differ by the detail and by nothing else.
5258 let detail = "the long explanation behind the fold ".repeat(40);
5259 let msgs = vec![
5260 ChatMessage::user(fmt!("explain")),
5261 ChatMessage::Assistant {
5262 content: MessageContent::text(String::new()),
5263 tool_calls: vec![crate::protocol::ToolCall {
5264 id: fmt!("call_9"),
5265 name: fmt!("say"),
5266 arguments: fmt!("{{\"summary\":\"the gist\",\"detail\":\"{}\"}}", detail),
5267 }],
5268 },
5269 ChatMessage::tool(fmt!("call_9"), MessageContent::text(fmt!("Shown."))),
5270 ];
5271 let shut = OpenSet::new();
5272 let open: OpenSet = [fmt!("call_9")].into_iter().collect();
5273 let sized_shut = conversation_bytes(&msgs, &shut);
5274 let sized_open = conversation_bytes(&msgs, &open);
5275
5276 for (host, path) in [("api.test.com", "/v1/chat"), ("api.anthropic.com", "/v1/messages")] {
5277 let c = LlmClient::new(host, 443, path, "key", "claude-opus-5", 4096, tls.clone());
5278 c.set_open_folds(Vec::new());
5279 let wire_shut = c.build_body(&msgs, None, false).len() as u64;
5280 c.set_open_folds(vec![fmt!("call_9")]);
5281 let wire_open = c.build_body(&msgs, None, false).len() as u64;
5282
5283 assert!(wire_shut < wire_open, "the fixture proves nothing via {}: closing the fold \
5284 did not shrink the payload", path);
5285 assert!(sized_shut < sized_open,
5286 "a closed fold is sized as though its detail were still sent, so the trigger \
5287 folds a conversation of {} bytes that goes out as {} (via {})",
5288 sized_shut, wire_shut, path);
5289 assert_eq!(sized_open - sized_shut, wire_open - wire_shut,
5290 "the sizer books {} bytes for closing the fold and the wire saves {} (via {}), \
5291 so the trigger is measuring a rule of its own rather than the serialiser's",
5292 sized_open - sized_shut, wire_open - wire_shut, path);
5293 }
5294 }
5295
5296 #[test]
5297 fn test_build_request_body() {
5298 use rustls::crypto::ring;
5299 let _ = ring::default_provider().install_default();
5300 let tls = Arc::new(
5301 ClientConfig::builder()
5302 .dangerous()
5303 .with_custom_certificate_verifier(Arc::new(NoVerify))
5304 .with_no_client_auth()
5305 );
5306 let client = LlmClient::new("api.test.com", 443, "/v1/chat", "key", "model", 4096, tls);
5307 let messages = vec![
5308 ChatMessage::system("You are helpful".to_string()),
5309 ChatMessage::user("Hello".to_string()),
5310 ];
5311 let body = client.build_request_body(&messages);
5312 assert!(body.contains("\"model\":\"model\""));
5313 assert!(body.contains("\"stream\":true"));
5314 assert!(body.contains("\"role\":\"system\""));
5315 assert!(body.contains("\"role\":\"user\""));
5316 assert!(body.contains("\"content\":\"You are helpful\""));
5317 assert!(body.contains("\"content\":\"Hello\""));
5318 }
5319
5320 // ┌───────────────────────────────────────────────────────────────┐
5321 // │ Retry — pure parts │
5322 // └───────────────────────────────────────────────────────────────┘
5323
5324 #[test]
5325 fn test_status_retryable() {
5326 // Which statuses mean "not now" and which mean "not ever" is HTTP's
5327 // answer, not ours: 429 carries Retry-After and 5xx is the server's own
5328 // trouble, while every other 4xx describes this request.
5329 for code in [429u16, 500, 502, 503, 504, 529] {
5330 assert!(status_retryable(code), "{} should be retryable", code);
5331 }
5332 for code in [400u16, 401, 403, 404, 413, 422] {
5333 assert!(!status_retryable(code), "{} must NOT be retried", code);
5334 }
5335 }
5336
5337 #[test]
5338 fn test_parse_retry_after() {
5339 // The delta-seconds form, which is what a provider sends.
5340 assert_eq!(parse_retry_after("2"), Some(2_000));
5341 assert_eq!(parse_retry_after(" 30 "), Some(30_000));
5342 assert_eq!(parse_retry_after("0"), Some(0));
5343 // The HTTP-date form is not understood, and reads as absent rather than
5344 // as zero -- a zero would retry instantly against a provider that asked
5345 // for a minute.
5346 assert_eq!(parse_retry_after("Wed, 21 Oct 2026 07:28:00 GMT"), None);
5347 assert_eq!(parse_retry_after(""), None);
5348 }
5349
5350 #[test]
5351 fn test_status_code_and_header_value() {
5352 assert_eq!(status_code("HTTP/1.1 429 Too Many Requests"), Some(429));
5353 assert_eq!(status_code("HTTP/1.1 200 OK"), Some(200));
5354 assert_eq!(status_code("garbage"), None);
5355 let head = "HTTP/1.1 429 Too Many Requests\r\nRetry-After: 3\r\nContent-Length: 0\r\n";
5356 assert_eq!(header_value(head, "retry-after"), Some("3".to_string()));
5357 assert_eq!(header_value(head, "RETRY-AFTER"), Some("3".to_string()));
5358 assert_eq!(header_value(head, "x-absent"), None);
5359 }
5360
5361 #[test]
5362 fn test_backoff_grows_jitters_and_is_capped() {
5363 let p = RetryPolicy { max_attempts: 6, base_ms: 100, max_backoff_ms: 400,
5364 max_total_wait_ms: 10_000 };
5365 // Equal jitter: every delay sits in the top half of its nominal window,
5366 // so it is neither instant nor in lockstep with another worker's.
5367 let mut spread = std::collections::BTreeSet::new();
5368 for _ in 0..64 {
5369 let d = p.delay_ms(1, None);
5370 assert!((50..=100).contains(&d), "first backoff out of band: {}", d);
5371 spread.insert(d);
5372 }
5373 assert!(spread.len() > 1, "no jitter: eight workers would retry in lockstep");
5374 for _ in 0..16 {
5375 assert!((100..=200).contains(&p.delay_ms(2, None)));
5376 assert!((200..=400).contains(&p.delay_ms(3, None)));
5377 // Capped, not doubled forever.
5378 assert!((200..=400).contains(&p.delay_ms(9, None)));
5379 }
5380 }
5381
5382 #[test]
5383 fn test_retry_after_is_honoured_and_never_shortened() {
5384 let p = RetryPolicy::default();
5385 for _ in 0..32 {
5386 let d = p.delay_ms(1, Some(3_000));
5387 // The provider is the one party that knows when it will be ready, so
5388 // its figure is a floor -- jitter is only ever added to it.
5389 assert!(d >= 3_000, "Retry-After was shortened to {}", d);
5390 assert!(d <= 3_000 + RETRY_AFTER_JITTER_MS);
5391 }
5392 }
5393
5394 #[test]
5395 fn test_attempts_and_total_wait_are_both_bounded() {
5396 let p = RetryPolicy { max_attempts: 3, base_ms: 100, max_backoff_ms: 100,
5397 max_total_wait_ms: 10_000 };
5398 assert!(p.next_delay(0, 0, None).is_some());
5399 assert!(p.next_delay(1, 0, None).is_some());
5400 // Three attempts means two retries.
5401 assert!(p.next_delay(2, 0, None).is_none());
5402 // And a backoff that would push the total past its bound ends the
5403 // attempt, however many are left -- the user is watching a spinner.
5404 assert!(p.next_delay(0, 9_990, None).is_none());
5405 assert!(p.next_delay(0, 0, Some(60_000)).is_none());
5406 }
5407
5408 // ┌───────────────────────────────────────────────────────────────┐
5409 // │ Prompt caching │
5410 // └───────────────────────────────────────────────────────────────┘
5411
5412 #[test]
5413 fn test_model_caches_on_request() {
5414 // Claude is the model family that needs an explicit breakpoint, in every
5415 // id form a caller can configure.
5416 assert!(model_caches_on_request("anthropic/claude-opus-5"));
5417 assert!(model_caches_on_request("claude-sonnet-5"));
5418 assert!(model_caches_on_request("anthropic.claude-opus-5"));
5419 assert!(model_caches_on_request("us.anthropic.claude-haiku-4.5"));
5420 assert!(model_caches_on_request("ANTHROPIC/CLAUDE-OPUS-5"));
5421 // Everything else caches automatically or not at all, and must not be
5422 // sent a marker it did not ask for.
5423 assert!(!model_caches_on_request("accounts/fireworks/models/glm-5p2"));
5424 assert!(!model_caches_on_request("openai/gpt-5.4"));
5425 assert!(!model_caches_on_request("deepseek/deepseek-v3"));
5426 assert!(!model_caches_on_request("google/gemini-3.1-pro-preview"));
5427 assert!(!model_caches_on_request("x-ai/grok-4.5"));
5428 }
5429
5430 /// A system prompt long enough to be worth caching.
5431 fn long_system() -> String {
5432 "You are a careful assistant. ".repeat(120)
5433 }
5434
5435 #[test]
5436 fn test_cache_breakpoints_for_a_claude_model() {
5437 let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
5438 let messages = vec![
5439 ChatMessage::system(long_system()),
5440 ChatMessage::user("Hello".to_string()),
5441 ];
5442 let body = client.build_body(&messages, None, true);
5443 // The system message carries a breakpoint, in the content-block form the
5444 // marker can only live on.
5445 assert!(body.contains("\"role\":\"system\",\"content\":[{\"type\":\"text\""),
5446 "system message did not become a content block: {}", body);
5447 // Two breakpoints: the stable system prefix, and the tip of the settled
5448 // conversation for the next turn to read back.
5449 assert_eq!(body.matches("\"cache_control\":{\"type\":\"ephemeral\"}").count(), 2,
5450 "expected a system and a user breakpoint: {}", body);
5451 assert!(body.contains("\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\""));
5452 }
5453
5454 #[test]
5455 fn test_no_cache_control_for_a_model_that_does_not_take_it() {
5456 let client = test_client("api.fireworks.ai", 443, "accounts/fireworks/models/glm-5p2");
5457 let messages = vec![
5458 ChatMessage::system(long_system()),
5459 ChatMessage::user("Hello".to_string()),
5460 ];
5461 let body = client.build_body(&messages, None, true);
5462 assert!(!body.contains("cache_control"),
5463 "a marker reached a provider that never asked for one: {}", body);
5464 // And the message shape is untouched: plain string content, as before.
5465 assert!(body.contains("\"role\":\"user\",\"content\":\"Hello\""));
5466 }
5467
5468 #[test]
5469 fn test_a_prefix_too_short_to_cache_gets_no_breakpoint() {
5470 // Below Anthropic's minimum cacheable prefix nothing is stored, and the
5471 // provider says nothing about having declined -- so the marker is simply
5472 // not sent.
5473 let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
5474 let messages = vec![
5475 ChatMessage::system("Be brief.".to_string()),
5476 ChatMessage::user("Hi".to_string()),
5477 ];
5478 assert!(!client.build_body(&messages, None, true).contains("cache_control"));
5479 }
5480
5481 #[test]
5482 fn test_tool_definitions_count_towards_the_cacheable_prefix() {
5483 // The tools render ahead of the system message, so a large tool array is
5484 // itself most of what the breakpoint caches.
5485 let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
5486 let messages = vec![ChatMessage::system("Be brief.".to_string())];
5487 assert!(!client.build_body(&messages, None, true).contains("cache_control"));
5488 let tools = "[".to_string() + &"x".repeat(CACHE_MIN_PREFIX_CHARS) + "]";
5489 assert!(client.build_body(&messages, Some(&tools), true).contains("cache_control"));
5490 }
5491
5492 #[test]
5493 fn test_the_second_breakpoint_follows_the_conversation() {
5494 // Several turns in, the second breakpoint sits on the LAST user message,
5495 // so everything settled before it is read from the cache next round.
5496 let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
5497 let messages = vec![
5498 ChatMessage::system(long_system()),
5499 ChatMessage::user("first".to_string()),
5500 ChatMessage::assistant("ok".to_string()),
5501 ChatMessage::user("second".to_string()),
5502 ];
5503 let body = client.build_body(&messages, None, true);
5504 assert!(body.contains("\"text\":\"second\",\"cache_control\""),
5505 "breakpoint is not on the latest user turn: {}", body);
5506 assert!(!body.contains("\"text\":\"first\",\"cache_control\""),
5507 "a stale breakpoint was left on an earlier turn: {}", body);
5508 assert_eq!(body.matches("cache_control").count(), 2);
5509 }
5510
5511 #[test]
5512 fn test_a_marked_message_still_round_trips_its_escapes() {
5513 let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
5514 let messages = vec![
5515 ChatMessage::system(long_system() + "say \"hi\"\n"),
5516 ];
5517 let body = client.build_body(&messages, None, true);
5518 assert!(body.contains("say \\\"hi\\\"\\n"), "escapes broke: {}", body);
5519 }
5520
5521 // ┌───────────────────────────────────────────────────────────────┐
5522 // │ Anthropic — dialect, request shape, headers │
5523 // └───────────────────────────────────────────────────────────────┘
5524
5525 #[test]
5526 fn test_the_dialect_is_chosen_by_the_endpoint_not_the_model() {
5527 // The same Claude model is reachable both ways, so the model id cannot
5528 // decide this; the endpoint can, and does.
5529 assert_eq!(Dialect::for_endpoint("api.anthropic.com", "/v1/messages"),
5530 Dialect::Anthropic);
5531 assert_eq!(Dialect::for_endpoint("API.Anthropic.Com", "/v1/messages/"),
5532 Dialect::Anthropic);
5533 // A proxy in front of the Messages API is still speaking it.
5534 assert_eq!(Dialect::for_endpoint("gateway.example.com", "/proxy/v1/messages"),
5535 Dialect::Anthropic);
5536 // And a router serving a Claude model over chat completions is not.
5537 assert_eq!(Dialect::for_endpoint("openrouter.ai", "/api/v1/chat/completions"),
5538 Dialect::OpenAi);
5539 assert_eq!(Dialect::for_endpoint("api.fireworks.ai", "/inference/v1/chat/completions"),
5540 Dialect::OpenAi);
5541 }
5542
5543 #[test]
5544 fn test_the_auth_headers_differ_by_dialect() {
5545 // Anthropic refuses a bearer token, wants a pinned version, and answers
5546 // a browser only when asked to.
5547 let anth = test_client_at("api.anthropic.com", 443, "/v1/messages", "claude-opus-5");
5548 let native: Vec<String> = anth.auth_headers(false).iter()
5549 .map(|(k, v)| fmt!("{}: {}", k, v)).collect();
5550 assert!(native.iter().any(|h| h == "x-api-key: key"), "{:?}", native);
5551 assert!(native.iter().any(|h| h == &fmt!("anthropic-version: {}", ANTHROPIC_VERSION)),
5552 "{:?}", native);
5553 assert!(!native.iter().any(|h| h.starts_with("Authorization")),
5554 "a bearer token reached the Messages API: {:?}", native);
5555 assert!(!native.iter().any(|h| h.contains("dangerous-direct-browser-access")),
5556 "the browser header was sent from a transport that is not one: {:?}", native);
5557 let browser: Vec<String> = anth.auth_headers(true).iter()
5558 .map(|(k, v)| fmt!("{}: {}", k, v)).collect();
5559 assert!(browser.iter().any(|h| h == "anthropic-dangerous-direct-browser-access: true"),
5560 "without this header the browser call never leaves CORS: {:?}", browser);
5561
5562 // And the OpenAI side is untouched, in either transport.
5563 let oai = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
5564 for browser in [false, true] {
5565 let hs: Vec<String> = oai.auth_headers(browser).iter()
5566 .map(|(k, v)| fmt!("{}: {}", k, v)).collect();
5567 assert!(hs.iter().any(|h| h == "Authorization: Bearer key"), "{:?}", hs);
5568 assert!(!hs.iter().any(|h| h.starts_with("anthropic-")),
5569 "an Anthropic header reached an OpenAI endpoint: {:?}", hs);
5570 }
5571 }
5572
5573 /// A client speaking the Messages API to Anthropic.
5574 fn anth_client(model: &str) -> LlmClient {
5575 test_client_at("api.anthropic.com", 443, "/v1/messages", model)
5576 }
5577
5578 // ── Images on the wire ───────────────────────────────────────────────────
5579 //
5580 // The fixtures below are NOT what this code produces; they are what the two providers publish,
5581 // copied out of their own documents, and every one of them says where it came from. A
5582 // serialisation test written the other way round -- build with our encoder, read with our
5583 // parser -- proves only that the two halves agree with each other, which they would go on
5584 // doing while both were wrong.
5585
5586 /// The one-pixel PNG from Anthropic's vision documentation, base64 exactly as printed there.
5587 ///
5588 /// Source: `platform.claude.com/docs/en/build-with-claude/vision`, the "Multiple images"
5589 /// example, `image1_data`. Using the provider's own bytes rather than bytes of this test's
5590 /// invention means the encoder is checked against a string a provider published, not against
5591 /// itself.
5592 const DOC_PNG_B64: &str = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nG\
5593 P4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC";
5594
5595 /// Those bytes, decoded.
5596 fn doc_png() -> Vec<u8> {
5597 oxedyne_fe2o3_text::base64::decode(DOC_PNG_B64).expect("the documented base64 must decode")
5598 }
5599
5600 /// An image part holding the documented PNG.
5601 fn doc_image(source: &str) -> ImagePart {
5602 ImagePart::new(ImageMedia::Png, doc_png(), source.to_string())
5603 }
5604
5605 /// The base64 encoder agrees with the provider on the provider's own bytes.
5606 ///
5607 /// The fixtures below all embed [`DOC_PNG_B64`]; if the encoder disagreed with Anthropic about
5608 /// how those bytes are spelled, every one of them would fail for a reason that had nothing to
5609 /// do with the shape being tested. This isolates that.
5610 #[test]
5611 fn test_the_base64_encoding_matches_the_providers_own_string() {
5612 let bytes = doc_png();
5613 assert!(!bytes.is_empty(), "the documented base64 decoded to nothing");
5614 assert_eq!(DOC_PNG_B64, oxedyne_fe2o3_text::base64::encode(&bytes),
5615 "our base64 disagrees with the string Anthropic published for these bytes");
5616 }
5617
5618 /// An Anthropic image block is the block Anthropic documents.
5619 ///
5620 /// Fixture source: `platform.claude.com/docs/en/build-with-claude/vision`, "Base64-encoded
5621 /// image example", the cURL request body -- `{"type":"image","source":{"type":"base64",
5622 /// "media_type":…,"data":…}}`, in that key order.
5623 #[test]
5624 fn test_an_anthropic_image_block_is_the_documented_shape() {
5625 let want = fmt!(
5626 "{{\"type\":\"image\",\"source\":{{\"type\":\"base64\",\"media_type\":\"image/png\",\
5627 \"data\":\"{}\"}}}}", DOC_PNG_B64);
5628 let client = anth_client("claude-opus-5");
5629 let msgs = vec![ChatMessage::user(MessageContent::parts(vec![
5630 ContentPart::Image(doc_image("shots/after.png")),
5631 ContentPart::Text("Describe this image.".to_string()),
5632 ]))];
5633 let body = client.build_anthropic_body(&msgs, None, true);
5634 assert!(body.contains(&want), "the image block is not the documented one.\nwant: {}\ngot: {}",
5635 want, body);
5636 // The image precedes the text, as the documentation recommends and as the part order says.
5637 let img = body.find("\"type\":\"image\"").expect("no image block");
5638 let txt = body.find("Describe this image.").expect("no text block");
5639 assert!(img < txt, "the parts were reordered");
5640 }
5641
5642 /// An OpenAI image part is the part OpenAI documents.
5643 ///
5644 /// Fixture source: OpenAI's own OpenAPI specification, schema
5645 /// `ChatCompletionRequestMessageContentPartImage` -- `type` is the constant `"image_url"`, and
5646 /// `image_url.url` is documented as "URL of the image. This can be a URL or a base64 encoded
5647 /// data URL". The data URL itself is RFC 2397 syntax, `data:<media-type>;base64,<data>`.
5648 /// `detail` is optional and defaults to `"auto"`, so it is not sent.
5649 #[test]
5650 fn test_an_openai_image_part_is_the_documented_shape() {
5651 let want = fmt!(
5652 "{{\"type\":\"image_url\",\"image_url\":{{\"url\":\"data:image/png;base64,{}\"}}}}",
5653 DOC_PNG_B64);
5654 let client = test_client("api.example.com", 443, "gpt-5.6");
5655 let msgs = vec![ChatMessage::user(MessageContent::parts(vec![
5656 ContentPart::Text("What is in this image?".to_string()),
5657 ContentPart::Image(doc_image("shots/after.png")),
5658 ]))];
5659 let body = client.build_openai_body(&msgs, None, true);
5660 assert!(body.contains(&want), "the image part is not the documented one.\nwant: {}\ngot: {}",
5661 want, body);
5662 assert!(body.contains("\"content\":[{\"type\":\"text\",\"text\":\"What is in this image?\"}"),
5663 "an image turns the content into the documented parts array: {}", body);
5664 }
5665
5666 /// A message with no image keeps the bare-string content it always had.
5667 ///
5668 /// The parts array is legal for text too, and switching every message to it would have been
5669 /// simpler -- and would have changed the bytes of every request every router has ever been
5670 /// sent, for nothing.
5671 #[test]
5672 fn test_text_only_content_stays_a_bare_string_on_both_sides() {
5673 let msgs = vec![ChatMessage::user("Hello".to_string())];
5674 let openai = test_client("api.example.com", 443, "gpt-5.6")
5675 .build_openai_body(&msgs, None, true);
5676 assert!(openai.contains("{\"role\":\"user\",\"content\":\"Hello\"}"),
5677 "text content grew an array: {}", openai);
5678 let anth = anth_client("claude-opus-5").build_anthropic_body(&msgs, None, true);
5679 assert!(anth.contains("{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Hello\"}]}"),
5680 "the Anthropic user turn is not the block form it always was: {}", anth);
5681 }
5682
5683 /// Anthropic takes an image inside a `tool_result`; OpenAI does not, and the image is re-homed
5684 /// into a `user` turn after the run of tool replies rather than dropped.
5685 ///
5686 /// Source for the asymmetry: Anthropic's `tool_result` content is documented as a string or an
5687 /// array of text and image blocks; OpenAI's tool-message content part union
5688 /// (`ChatCompletionRequestToolMessageContentPart`) has a text member and no image member.
5689 #[test]
5690 fn test_a_tool_result_image_rides_the_reply_on_one_side_and_a_user_turn_on_the_other() {
5691 let msgs = vec![
5692 ChatMessage::user("look at the page".to_string()),
5693 ChatMessage::assistant_calling("", vec![ToolCall {
5694 id: "call_1".to_string(),
5695 name: "file_read".to_string(),
5696 arguments: r#"{"path":"shots/after.png"}"#.to_string(),
5697 }]),
5698 ChatMessage::tool("call_1".to_string(), MessageContent::parts(vec![
5699 ContentPart::Text("Read the image shots/after.png.".to_string()),
5700 ContentPart::Image(doc_image("shots/after.png")),
5701 ])),
5702 ];
5703
5704 let anth = anth_client("claude-opus-5").build_anthropic_body(&msgs, None, true);
5705 assert!(anth.contains("\"type\":\"tool_result\",\"tool_use_id\":\"call_1\",\"content\":["),
5706 "the Anthropic tool result should carry blocks: {}", anth);
5707 let result_at = anth.find("tool_result").expect("no tool_result");
5708 let image_at = anth.find("\"type\":\"image\"").expect("no image block");
5709 assert!(image_at > result_at, "the image left the tool result it belongs to");
5710
5711 let openai = test_client("api.example.com", 443, "gpt-5.6")
5712 .build_openai_body(&msgs, None, true);
5713 // The tool reply itself is text only -- the API has nowhere else to put an image.
5714 let tool_msg = openai.find("\"role\":\"tool\"").expect("no tool message");
5715 let img_at = openai.find("image_url").expect("the image was dropped");
5716 assert!(img_at > tool_msg, "an image_url was put inside the tool reply");
5717 assert!(openai[tool_msg..img_at].contains("\"role\":\"user\""),
5718 "the image should be re-homed into a user turn after the run: {}", openai);
5719 }
5720
5721 /// A cache breakpoint survives a message that ends in an image.
5722 ///
5723 /// The marker caches everything up to the block it sits on. If it could only go on a text
5724 /// block, a user turn whose last part is the screenshot would carry no marker at all and the
5725 /// whole prefix would be re-billed on every round of the turn -- silently, since nothing
5726 /// fails.
5727 #[test]
5728 fn test_a_cache_breakpoint_survives_a_message_that_ends_in_an_image() {
5729 let ends_in_image = MessageContent::parts(vec![
5730 ContentPart::Text("here".to_string()),
5731 ContentPart::Image(doc_image("shots/after.png")),
5732 ]);
5733 let blocks = anthropic_blocks(&ends_in_image, true);
5734 assert_eq!(2, blocks.len());
5735 assert!(!blocks[0].contains("cache_control"),
5736 "the marker must be on the LAST block, not the first: {}", blocks[0]);
5737 assert!(blocks[1].contains("\"cache_control\":{\"type\":\"ephemeral\"}"),
5738 "a message ending in an image lost its cache breakpoint: {}", blocks[1]);
5739
5740 // And the marker is not attached when the message is not a breakpoint.
5741 let plain = anthropic_blocks(&ends_in_image, false);
5742 assert!(!plain.iter().any(|b| b.contains("cache_control")));
5743 }
5744
5745 /// A model on the known-blind list is refused before the request is built, by name.
5746 #[test]
5747 fn test_a_model_that_cannot_see_is_refused_by_name() {
5748 let client = test_client("api.example.com", 443, "openai/gpt-3.5-turbo-0125");
5749 let msgs = vec![ChatMessage::user(MessageContent::parts(vec![
5750 ContentPart::Image(doc_image("shots/after.png")),
5751 ]))];
5752 let e = client.vision_guard(&msgs).expect_err("a blind model must be refused");
5753 let msg = fmt!("{}", e);
5754 assert!(msg.contains("gpt-3.5-turbo-0125"), "the refusal must name the model: {}", msg);
5755 assert!(msg.contains("cannot see"), "the refusal must say what is wrong: {}", msg);
5756 // And a turn with no image goes through on the same model, because the model is only
5757 // unusable for the thing it cannot do.
5758 assert_eq!(0, client.vision_guard(&[ChatMessage::user("hi".to_string())])
5759 .expect("text must still be allowed"));
5760 }
5761
5762 /// THE CLIENT'S OWN REASON MUST LEAVE THIS MODULE, because the browser's does not survive
5763 /// the crossing intact and is not the same on two browsers.
5764 ///
5765 /// A failed `fetch` reads `TypeError: Failed to fetch` in Chromium and `TypeError: Load
5766 /// failed` in WebKit for the identical event. `www/js/daimond.js` decides from that string
5767 /// whether to hand a turn back with a Continue button or write it off, so while only `err`
5768 /// crossed, that decision was a property of the browser. `crossed` puts `reason` -- which is
5769 /// this file's wording and is the same everywhere -- in front of it.
5770 #[test]
5771 fn test_a_transport_failure_carries_its_reason_out_of_this_module() {
5772 // The exact shape of the iOS case: the fetch never got a response, and the browser's
5773 // own sentence is the only thing in the error.
5774 let e = TransportErr::transient(
5775 "could not reach the provider".to_string(),
5776 err!("LLM: fetch failed: TypeError: Load failed."; IO, Network, Wire));
5777 let out = fmt!("{}", e.crossed());
5778 assert!(out.contains("could not reach the provider"),
5779 "the client's own reason did not cross: {}", out);
5780 assert!(out.contains("Load failed"),
5781 "the provider's -- or the browser's -- own words must survive with it: {}", out);
5782 // And a failure that is the PROVIDER answering carries a reason that says so, which is
5783 // what keeps the app from reading a 429 as a dead road.
5784 let e = TransportErr::fatal(
5785 "the provider returned HTTP 400".to_string(),
5786 err!("LLM: HTTP error: 400 Bad Request | context length exceeded"; IO, Network, Wire));
5787 let out = fmt!("{}", e.crossed());
5788 assert!(out.contains("the provider returned HTTP 400"), "{}", out);
5789 // `compact::looks_like_overflow` reads this text, so the body detail must still be in it.
5790 assert!(out.contains("context length"),
5791 "the refusal's own body was lost, and overflow detection reads it: {}", out);
5792 }
5793
5794 /// A model NOT on the list is allowed through -- the list is of what is known blind, not of
5795 /// what is known to see, so a model released tomorrow is not refused today.
5796 #[test]
5797 fn test_an_unknown_model_is_assumed_to_see() {
5798 assert!(model_can_see("some-vendor/brand-new-model-9"));
5799 assert!(model_can_see("claude-opus-5"));
5800 assert!(!model_can_see("gpt-3.5-turbo"));
5801 assert!(!model_can_see("anthropic/claude-2.1"));
5802 }
5803
5804 /// When the provider refuses a turn that carried images and its words are about images, the
5805 /// error names the model and says it cannot see -- with the provider's own sentence kept.
5806 #[test]
5807 fn test_a_provider_refusal_about_images_is_rewritten_to_name_the_model() {
5808 let client = test_client("api.example.com", 443, "some-router/mystery-model");
5809 let raw = err!("HTTP error: 400 Bad Request: invalid_request_error: \
5810 this model does not support image_url content"; Invalid, Input);
5811 let out = fmt!("{}", client.vision_error(raw, 1));
5812 assert!(out.contains("some-router/mystery-model"), "the model must be named: {}", out);
5813 assert!(out.contains("not to see"), "it must say what is wrong: {}", out);
5814 assert!(out.contains("400 Bad Request"), "the provider's own words must survive: {}", out);
5815 }
5816
5817 /// A failure unrelated to images is handed back untouched, even on a turn that carried one.
5818 #[test]
5819 fn test_an_unrelated_failure_is_not_blamed_on_the_images() {
5820 let client = test_client("api.example.com", 443, "some-router/mystery-model");
5821 let raw = err!("HTTP error: 401 Unauthorized"; Invalid, Input);
5822 let out = fmt!("{}", client.vision_error(raw, 1));
5823 assert!(out.contains("401 Unauthorized"), "the provider's words were lost: {}", out);
5824 assert!(!out.contains("not to see"),
5825 "an unrelated failure was rewritten as a vision failure: {}", out);
5826 assert!(!out.contains("mystery-model"),
5827 "an unrelated failure was rewritten as a vision failure: {}", out);
5828 }
5829
5830 #[test]
5831 fn test_the_system_prompt_is_hoisted_out_of_the_messages() {
5832 // The Messages API has no system role: a system message left in the
5833 // array is a 400, and one silently dropped is an agent with no rules.
5834 let client = anth_client("claude-opus-5");
5835 let msgs = vec![
5836 ChatMessage::system(long_system()),
5837 ChatMessage::system("And be brief.".to_string()),
5838 ChatMessage::user("Hello".to_string()),
5839 ];
5840 let body = client.build_anthropic_body(&msgs, None, true);
5841 assert!(body.contains("\"system\":[{\"type\":\"text\""),
5842 "no top-level system field: {}", body);
5843 assert!(!body.contains("\"role\":\"system\""),
5844 "a system message was left in the array: {}", body);
5845 // Both of them, joined, rather than only the last.
5846 assert!(body.contains("And be brief."), "the second system message was lost: {}", body);
5847 assert!(body.contains("You are a careful assistant."), "{}", body);
5848 }
5849
5850 #[test]
5851 fn test_the_breakpoints_land_on_the_anthropic_blocks() {
5852 // The marker only exists on a content block, and the Messages API's
5853 // blocks are in different places from the OpenAI ones.
5854 let client = anth_client("claude-opus-5");
5855 let msgs = vec![
5856 ChatMessage::system(long_system()),
5857 ChatMessage::user("first".to_string()),
5858 ChatMessage::assistant("ok".to_string()),
5859 ChatMessage::user("second".to_string()),
5860 ];
5861 let body = client.build_anthropic_body(&msgs, None, true);
5862 assert_eq!(body.matches("\"cache_control\":{\"type\":\"ephemeral\"}").count(), 2,
5863 "expected a system and a user breakpoint: {}", body);
5864 assert!(body.contains("\"text\":\"second\",\"cache_control\""),
5865 "the second breakpoint is not on the latest user turn: {}", body);
5866 assert!(!body.contains("\"text\":\"first\",\"cache_control\""),
5867 "a stale breakpoint was left on an earlier turn: {}", body);
5868 // The system block carries the other one.
5869 let sys_end = match body.find("}],\"messages\"") {
5870 Some(p) => p,
5871 None => panic!("no system block: {}", body),
5872 };
5873 assert!(body[..sys_end].contains("cache_control"),
5874 "the system prefix -- the largest stable block there is -- is uncached: {}", body);
5875 }
5876
5877 #[test]
5878 fn test_a_model_that_does_not_cache_gets_no_marker_on_this_path_either() {
5879 // The gate is the model id, and it must still be the model id here.
5880 let client = test_client_at("api.example.com", 443, "/v1/messages", "some-other-model");
5881 let msgs = vec![
5882 ChatMessage::system(long_system()),
5883 ChatMessage::user("Hello".to_string()),
5884 ];
5885 let body = client.build_anthropic_body(&msgs, None, true);
5886 assert!(!body.contains("cache_control"),
5887 "a marker reached a model that never asked for one: {}", body);
5888 }
5889
5890 #[test]
5891 fn test_a_run_of_tool_results_becomes_one_user_message() {
5892 // Two parallel tool calls produce two `Tool` messages in a row. The
5893 // Messages API wants both results as blocks of a SINGLE user turn;
5894 // sending two consecutive user messages is a different conversation.
5895 let client = anth_client("claude-opus-5");
5896 let msgs = vec![
5897 ChatMessage::user("list and read".to_string()),
5898 ChatMessage::Assistant {
5899 content: MessageContent::text(""),
5900 tool_calls: vec![
5901 ToolCall { id: "t1".to_string(), name: "file_list".to_string(),
5902 arguments: "{}".to_string() },
5903 ToolCall { id: "t2".to_string(), name: "file_read".to_string(),
5904 arguments: r#"{"path":"a.txt"}"#.to_string() },
5905 ],
5906 },
5907 ChatMessage::tool("t1".to_string(), "a.txt".to_string()),
5908 ChatMessage::tool("t2".to_string(), "hello".to_string()),
5909 ];
5910 let body = client.build_anthropic_body(&msgs, None, false);
5911 assert_eq!(body.matches("\"role\":\"user\"").count(), 2,
5912 "the two tool results did not coalesce into one turn: {}", body);
5913 assert_eq!(body.matches("\"type\":\"tool_result\"").count(), 2, "{}", body);
5914 assert!(body.contains("\"tool_use_id\":\"t1\""), "{}", body);
5915 assert!(body.contains("\"tool_use_id\":\"t2\""), "{}", body);
5916 // And the assistant turn's calls are `tool_use` blocks whose input is a
5917 // JSON OBJECT -- the OpenAI form is a string, and sending that is a 400.
5918 assert!(body.contains("\"type\":\"tool_use\",\"id\":\"t2\",\"name\":\"file_read\",\
5919 \"input\":{\"path\":\"a.txt\"}"),
5920 "the arguments were not carried as an object: {}", body);
5921 }
5922
5923 #[test]
5924 fn test_tool_definitions_are_translated_to_the_anthropic_shape() {
5925 let tools = r#"[{"type":"function","function":{"name":"file_read",
5926 "description":"Read a file","parameters":{"type":"object","properties":{
5927 "path":{"type":"string","description":"name"}},"required":["path"]}}}]"#;
5928 let out = openai_tools_to_anthropic(tools);
5929 assert!(out.contains("\"name\":\"file_read\""), "{}", out);
5930 assert!(out.contains("\"description\":\"Read a file\""),
5931 "the description was read from the schema instead of the function: {}", out);
5932 assert!(out.contains("\"input_schema\":{\"type\":\"object\""),
5933 "the schema is not under input_schema: {}", out);
5934 assert!(!out.contains("\"parameters\""), "the OpenAI wrapper survived: {}", out);
5935 assert!(!out.contains("\"type\":\"function\""), "{}", out);
5936 // A definition with no schema is dropped rather than sent half-built.
5937 assert_eq!(openai_tools_to_anthropic(r#"[{"type":"function","function":{"name":"x"}}]"#),
5938 "[]");
5939 }
5940
5941 #[test]
5942 fn test_thinking_is_asked_for_only_where_it_is_taken() {
5943 // `budget_tokens` is a 400 on every model since Opus 4.7, and adaptive
5944 // is a 400 on the ones before Opus 4.6 -- so the gate is a list, not a
5945 // family test.
5946 for id in ["claude-opus-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6",
5947 "claude-sonnet-5", "claude-sonnet-4-6", "claude-fable-5", "claude-mythos-5",
5948 "anthropic/claude-opus-5", "us.anthropic.claude-sonnet-5-v1"] {
5949 assert!(model_takes_adaptive_thinking(id), "{} takes adaptive thinking", id);
5950 }
5951 for id in ["claude-haiku-4-5", "claude-sonnet-4-5", "claude-opus-4-5", "claude-3-opus",
5952 "accounts/fireworks/models/glm-5p2", "openai/gpt-5.4"] {
5953 assert!(!model_takes_adaptive_thinking(id), "{} must not be sent adaptive", id);
5954 }
5955 // And the request follows the gate.
5956 let msgs = [ChatMessage::user("Hi".to_string())];
5957 let on = anth_client("claude-opus-5").build_anthropic_body(&msgs, None, true);
5958 assert!(on.contains("\"thinking\":{\"type\":\"adaptive\",\"display\":\"summarized\"}"),
5959 "{}", on);
5960 assert!(!on.contains("budget_tokens"), "a removed parameter was sent: {}", on);
5961 let off = anth_client("claude-haiku-4-5").build_anthropic_body(&msgs, None, true);
5962 assert!(!off.contains("thinking"), "{}", off);
5963 }
5964
5965 #[test]
5966 fn test_an_empty_message_does_not_become_an_empty_block() {
5967 // The Messages API rejects a text block with no text, where the OpenAI
5968 // side carries the empty string through without comment. One stray
5969 // empty user message would then fail every turn of the conversation.
5970 let client = anth_client("claude-opus-5");
5971 let msgs = vec![
5972 ChatMessage::user("hello".to_string()),
5973 ChatMessage::assistant(String::new()),
5974 ChatMessage::user(String::new()),
5975 ];
5976 let body = client.build_anthropic_body(&msgs, None, true);
5977 assert!(!body.contains("\"text\":\"\""), "an empty text block was sent: {}", body);
5978 // And the assistant turn that says nothing and asks for nothing is left
5979 // out entirely rather than sent as a message with no content.
5980 assert_eq!(body.matches("\"role\":\"assistant\"").count(), 0, "{}", body);
5981 assert!(body.contains("\"text\":\"hello\""), "{}", body);
5982 }
5983
5984 #[test]
5985 fn test_a_thinking_turn_is_given_room_for_the_reasoning_and_the_answer() {
5986 // `max_tokens` caps thinking AND the reply together here, and the app's
5987 // internal default is 4096 -- chosen when it only ever meant the reply.
5988 // Left alone, a hard question is answered with a truncated sentence.
5989 let msgs = [ChatMessage::user("Hi".to_string())];
5990 let c = anth_client("claude-opus-5");
5991 assert_eq!(c.max_tokens, 4096, "the fixture no longer reflects the app's default");
5992 let streamed = c.build_anthropic_body(&msgs, None, true);
5993 assert!(streamed.contains(&fmt!("\"max_tokens\":{}", THINKING_MIN_MAX_TOKENS)),
5994 "a streamed thinking turn was capped at the answer-only figure: {}", streamed);
5995 // The one-shot path keeps the configured cap: a big one there is a long
5996 // silence on an open connection, which is how a request times out.
5997 let once = c.build_anthropic_body(&msgs, None, false);
5998 assert!(once.contains("\"max_tokens\":4096"), "{}", once);
5999 // And a model that does not think is not given the extra room either.
6000 let plain = anth_client("claude-haiku-4-5").build_anthropic_body(&msgs, None, true);
6001 assert!(plain.contains("\"max_tokens\":4096"), "{}", plain);
6002 }
6003
6004 #[test]
6005 fn test_the_openai_body_is_unchanged_by_all_this() {
6006 // The regression that matters most: five providers already work through
6007 // the other dialect, and none of them may notice this.
6008 let client = test_client("openrouter.ai", 443, "anthropic/claude-opus-5");
6009 let msgs = vec![
6010 ChatMessage::system(long_system()),
6011 ChatMessage::user("Hello".to_string()),
6012 ];
6013 let body = client.build_body(&msgs, None, true);
6014 assert!(body.contains("\"stream_options\":{\"include_usage\":true}"), "{}", body);
6015 assert!(body.contains("\"role\":\"system\""), "{}", body);
6016 assert!(!body.contains("\"system\":["), "{}", body);
6017 assert!(!body.contains("\"thinking\""), "{}", body);
6018 assert!(!body.contains("input_schema"), "{}", body);
6019 }
6020
6021 // ┌───────────────────────────────────────────────────────────────┐
6022 // │ Anthropic — the event stream │
6023 // └───────────────────────────────────────────────────────────────┘
6024
6025 /// Drive a sequence of Anthropic SSE payloads through a fresh accumulator.
6026 fn run_anth(chunks: &[&str]) -> (AnthropicAcc, Vec<String>) {
6027 let mut acc = AnthropicAcc::default();
6028 let mut tokens = Vec::new();
6029 for c in chunks {
6030 acc.ingest(c, &mut text_sink(&mut tokens));
6031 }
6032 (acc, tokens)
6033 }
6034
6035 /// The same, keeping the working rather than the answer.
6036 fn run_anth_thinking(chunks: &[&str]) -> (AnthropicAcc, Vec<String>) {
6037 let mut acc = AnthropicAcc::default();
6038 let mut thought = Vec::new();
6039 for c in chunks {
6040 acc.ingest(c, &mut think_sink(&mut thought));
6041 }
6042 (acc, thought)
6043 }
6044
6045 #[test]
6046 fn test_the_anthropic_stream_rebuilds_text_and_tool_calls() {
6047 // The documented event sequence, verbatim from the streaming reference.
6048 let (acc, tokens) = run_anth(&[
6049 r#"{"type":"message_start","message":{"id":"msg_1","usage":{"input_tokens":472,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":2}}}"#,
6050 r#"{"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}"#,
6051 r#"{"type":"ping"}"#,
6052 r#"{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Okay"}}"#,
6053 r#"{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":", checking"}}"#,
6054 r#"{"type":"content_block_stop","index":0}"#,
6055 r#"{"type":"content_block_start","index":1,"content_block":{"type":"tool_use","id":"toolu_1","name":"get_weather","input":{}}}"#,
6056 r#"{"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"{\"location\":"}}"#,
6057 r#"{"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" \"Paris\"}"}}"#,
6058 r#"{"type":"content_block_stop","index":1}"#,
6059 r#"{"type":"message_delta","delta":{"stop_reason":"tool_use"},"usage":{"output_tokens":89}}"#,
6060 r#"{"type":"message_stop"}"#,
6061 ]);
6062 assert_eq!(tokens, vec!["Okay", ", checking"]);
6063 let resp = acc.into_response(false, 0);
6064 assert_eq!(resp.content, "Okay, checking");
6065 assert_eq!(resp.tool_calls.len(), 1);
6066 assert_eq!(resp.tool_calls[0].id, "toolu_1");
6067 assert_eq!(resp.tool_calls[0].name, "get_weather");
6068 assert_eq!(resp.tool_calls[0].arguments, r#"{"location": "Paris"}"#);
6069 // The `message_delta` counts are CUMULATIVE and name only what changed:
6070 // taking them wholesale would zero the input side of the bill.
6071 assert_eq!(resp.prompt_tokens, 472);
6072 assert_eq!(resp.completion_tokens, 89);
6073 }
6074
6075 #[test]
6076 fn test_the_whole_prompt_is_counted_and_the_cache_read_named() {
6077 // Anthropic's `input_tokens` EXCLUDES what it read from and wrote to the
6078 // cache; this client's `prompt` means every prompt token processed, and
6079 // the ledger prices `prompt - cached` at the fresh rate. Reading
6080 // `input_tokens` straight across would bill a 90%-cached turn as if the
6081 // cache were not there at all.
6082 let (acc, _t) = run_anth(&[
6083 r#"{"type":"message_start","message":{"usage":{"input_tokens":120,"cache_creation_input_tokens":40,"cache_read_input_tokens":9000,"output_tokens":1}}}"#,
6084 r#"{"type":"message_delta","delta":{},"usage":{"output_tokens":64}}"#,
6085 ]);
6086 let resp = acc.into_response(false, 0);
6087 assert_eq!(resp.prompt_tokens, 9160, "the cached prefix is part of the prompt");
6088 assert_eq!(resp.cached_tokens, 9000);
6089 assert_eq!(resp.completion_tokens, 64);
6090 // Anthropic reports no money, so nothing is claimed about it.
6091 assert_eq!(resp.cost_usd, 0.0);
6092 }
6093
6094 #[test]
6095 fn test_thinking_streams_are_kept_but_never_handed_over_as_the_answer() {
6096 let (acc, tokens) = run_anth(&[
6097 r#"{"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}"#,
6098 r#"{"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Euclid: 1071 = 2 x 462 + 147"}}"#,
6099 r#"{"type":"content_block_delta","index":0,"delta":{"type":"signature_delta","signature":"EqQBCgIYAhIM"}}"#,
6100 r#"{"type":"content_block_stop","index":0}"#,
6101 r#"{"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}"#,
6102 r#"{"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"21."}}"#,
6103 ]);
6104 // The reasoning is not the reply: a caller that streamed it into the
6105 // message would persist the model's working out as its answer.
6106 assert_eq!(tokens, vec!["21."], "thinking reached the token sink: {:?}", tokens);
6107 // And it IS handed over, as its own kind, while the round is still running.
6108 // Held back until the round ended, a model that thinks for a minute and a half
6109 // is a minute and a half of blank spinner.
6110 let (_a2, thought) = run_anth_thinking(&[
6111 r#"{"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}"#,
6112 r#"{"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Euclid: 1071 = 2 x 462 + 147"}}"#,
6113 r#"{"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"21."}}"#,
6114 ]);
6115 assert_eq!(thought, vec!["Euclid: 1071 = 2 x 462 + 147"],
6116 "the working did not reach the reasoning sink: {:?}", thought);
6117 let blocks = acc.thinking_blocks();
6118 assert_eq!(blocks.len(), 1, "the signed block was not kept for replay");
6119 assert!(blocks[0].contains("\"signature\":\"EqQBCgIYAhIM\""), "{}", blocks[0]);
6120 assert!(blocks[0].contains("1071 = 2 x 462 + 147"), "{}", blocks[0]);
6121 let resp = acc.into_response(false, 0);
6122 assert_eq!(resp.content, "21.");
6123 assert_eq!(resp.thinking, "Euclid: 1071 = 2 x 462 + 147",
6124 "the reasoning was neither shown nor accounted for");
6125 }
6126
6127 #[test]
6128 fn test_an_unsigned_thinking_run_is_not_replayed() {
6129 // A stream cut before its `signature_delta` leaves a block the API will
6130 // not verify. The run must match what the model generated, so half of
6131 // it is worse than none: sending it is a 400 on every following turn.
6132 let (acc, _t) = run_anth(&[
6133 r#"{"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}"#,
6134 r#"{"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"half a thought"}}"#,
6135 ]);
6136 assert!(acc.thinking_blocks().is_empty(),
6137 "an unsigned block was queued for replay");
6138 }
6139
6140 #[test]
6141 fn test_a_redacted_thinking_block_is_replayed_verbatim() {
6142 let (acc, _t) = run_anth(&[
6143 r#"{"type":"content_block_start","index":0,"content_block":{"type":"redacted_thinking","data":"EroBCkYIAxgCKkB"}}"#,
6144 ]);
6145 let blocks = acc.thinking_blocks();
6146 assert_eq!(blocks.len(), 1);
6147 assert!(blocks[0].contains("\"data\":\"EroBCkYIAxgCKkB\""),
6148 "an opaque block was rebuilt rather than replayed: {}", blocks[0]);
6149 }
6150
6151 #[test]
6152 fn test_a_stream_error_event_is_not_read_as_an_answer() {
6153 // An overload arrives INSIDE a 200 stream here, not as a status code.
6154 // Read as a short answer it would end the turn silently and wrongly.
6155 let (acc, _t) = run_anth(&[
6156 r#"{"type":"message_start","message":{"usage":{"input_tokens":10}}}"#,
6157 r#"{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}"#,
6158 ]);
6159 let wrapped = Acc::Anthropic(acc);
6160 let e = match wrapped.stream_error() {
6161 Some(e) => e,
6162 None => panic!("the error event was swallowed"),
6163 };
6164 assert!(e.retryable, "an overload is the provider saying 'not now'");
6165 assert!(e.reason.contains("overloaded_error"), "{}", e.reason);
6166 // A complaint about the request is not retried, exactly as for a 400.
6167 let (bad, _t) = run_anth(&[
6168 r#"{"type":"error","error":{"type":"invalid_request_error","message":"bad"}}"#,
6169 ]);
6170 let e = match Acc::Anthropic(bad).stream_error() {
6171 Some(e) => e,
6172 None => panic!("the error event was swallowed"),
6173 };
6174 assert!(!e.retryable, "a malformed request was queued for another attempt");
6175 }
6176
6177 #[test]
6178 fn test_a_whole_anthropic_response_parses() {
6179 let body = r#"{"id":"msg_1","type":"message","role":"assistant","model":"claude-opus-5",
6180 "content":[{"type":"thinking","thinking":"work","signature":"sig1"},
6181 {"type":"text","text":"Here you are."},
6182 {"type":"tool_use","id":"toolu_9","name":"file_read","input":{"path":"a.txt"}}],
6183 "stop_reason":"tool_use",
6184 "usage":{"input_tokens":100,"cache_read_input_tokens":900,"output_tokens":12}}"#;
6185 let (content, calls, use_, thinking) = parse_anthropic_response(body);
6186 assert_eq!(content, "Here you are.");
6187 assert_eq!(calls.len(), 1);
6188 assert_eq!(calls[0].name, "file_read");
6189 assert_eq!(calls[0].arguments, r#"{"path":"a.txt"}"#);
6190 assert_eq!(use_.prompt, 1000);
6191 assert_eq!(use_.cached, 900);
6192 assert_eq!(use_.completion, 12);
6193 assert_eq!(thinking.len(), 1);
6194 assert!(thinking[0].contains("\"signature\":\"sig1\""), "{}", thinking[0]);
6195 }
6196
6197 // ┌───────────────────────────────────────────────────────────────┐
6198 // │ A real HTTPS server to retry against │
6199 // └───────────────────────────────────────────────────────────────┘
6200 //
6201 // Not a mock of the client's own idea of a provider: a TCP listener, a TLS
6202 // handshake, an HTTP/1.1 status line and a chunked SSE body. What the
6203 // client does with a 429 is then observed rather than asserted about.
6204
6205 /// One scripted reply from the stub provider.
6206 #[derive(Clone)]
6207 pub enum Reply {
6208 /// A complete response: status line, headers, body.
6209 Http {
6210 status: u16,
6211 reason: &'static str,
6212 headers: Vec<(&'static str, String)>,
6213 body: String,
6214 },
6215 /// A chunked `text/event-stream` body. `reset_after` cuts the
6216 /// connection with an RST once that many chunks have gone out, which is
6217 /// what a provider dropping mid-answer looks like on the wire.
6218 Sse {
6219 chunks: Vec<String>,
6220 reset_after: Option<usize>,
6221 },
6222 }
6223
6224 impl Reply {
6225 /// A 429, optionally with the provider's own `Retry-After`.
6226 fn too_many(retry_after: Option<u64>) -> Self {
6227 let mut headers = vec![("Content-Type", "application/json".to_string())];
6228 if let Some(s) = retry_after {
6229 headers.push(("Retry-After", fmt!("{}", s)));
6230 }
6231 Self::Http {
6232 status: 429, reason: "Too Many Requests", headers,
6233 body: "{\"error\":{\"message\":\"rate limited\"}}".to_string(),
6234 }
6235 }
6236
6237 /// A 500, the provider's own trouble.
6238 fn server_error() -> Self {
6239 Self::Http {
6240 status: 500, reason: "Internal Server Error", headers: Vec::new(),
6241 body: "{\"error\":{\"message\":\"upstream fell over\"}}".to_string(),
6242 }
6243 }
6244
6245 /// A 400, this request being wrong.
6246 fn bad_request() -> Self {
6247 Self::Http {
6248 status: 400, reason: "Bad Request", headers: Vec::new(),
6249 body: "{\"error\":{\"message\":\"unknown field\"}}".to_string(),
6250 }
6251 }
6252
6253 /// A 404 with a body that says nothing about images -- which is the case that mattered:
6254 /// `vision_error` can only rewrite a refusal whose words mention pictures, and a bare
6255 /// 404 gives it nothing to work with.
6256 fn not_found() -> Self {
6257 Self::Http {
6258 status: 404, reason: "Not Found", headers: Vec::new(),
6259 body: "{\"error\":{\"message\":\"No endpoint found\"}}".to_string(),
6260 }
6261 }
6262
6263 /// A whole answer, streamed as two deltas and a usage chunk.
6264 fn answer() -> Self {
6265 Self::Sse {
6266 chunks: vec![
6267 "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\n".to_string(),
6268 "data: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\n".to_string(),
6269 "data: {\"choices\":[],\"usage\":{\"prompt_tokens\":11,\"completion_tokens\":2,\
6270 \"cost\":0.0003,\"prompt_tokens_details\":{\"cached_tokens\":9}}}\n\n".to_string(),
6271 "data: [DONE]\n\n".to_string(),
6272 ],
6273 reset_after: None,
6274 }
6275 }
6276
6277 /// An Anthropic turn that thinks, then asks for a tool.
6278 fn anth_thinks_then_calls() -> Self {
6279 Self::Sse {
6280 chunks: vec![
6281 "event: message_start\ndata: {\"type\":\"message_start\",\"message\":\
6282 {\"id\":\"msg_1\",\"usage\":{\"input_tokens\":30,\
6283 \"cache_read_input_tokens\":900,\"output_tokens\":1}}}\n\n".to_string(),
6284 "event: content_block_start\ndata: {\"type\":\"content_block_start\",\
6285 \"index\":0,\"content_block\":{\"type\":\"thinking\",\"thinking\":\"\",\
6286 \"signature\":\"\"}}\n\n".to_string(),
6287 "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\
6288 \"index\":0,\"delta\":{\"type\":\"thinking_delta\",\
6289 \"thinking\":\"I should read the file.\"}}\n\n".to_string(),
6290 "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\
6291 \"index\":0,\"delta\":{\"type\":\"signature_delta\",\
6292 \"signature\":\"SIGNATURE-1\"}}\n\n".to_string(),
6293 "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\
6294 \"index\":0}\n\n".to_string(),
6295 "event: content_block_start\ndata: {\"type\":\"content_block_start\",\
6296 \"index\":1,\"content_block\":{\"type\":\"tool_use\",\"id\":\"toolu_1\",\
6297 \"name\":\"file_read\",\"input\":{}}}\n\n".to_string(),
6298 "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\
6299 \"index\":1,\"delta\":{\"type\":\"input_json_delta\",\
6300 \"partial_json\":\"{\\\"path\\\":\\\"a.txt\\\"}\"}}\n\n".to_string(),
6301 "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\
6302 \"index\":1}\n\n".to_string(),
6303 "event: message_delta\ndata: {\"type\":\"message_delta\",\
6304 \"delta\":{\"stop_reason\":\"tool_use\"},\
6305 \"usage\":{\"output_tokens\":40}}\n\n".to_string(),
6306 "event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n".to_string(),
6307 ],
6308 reset_after: None,
6309 }
6310 }
6311
6312 /// A second thinking-plus-tool round, with its own signature and call id.
6313 fn anth_thinks_then_calls_again() -> Self {
6314 match Self::anth_thinks_then_calls() {
6315 Self::Sse { chunks, reset_after } => Self::Sse {
6316 chunks: chunks.iter()
6317 .map(|c| c.replace("SIGNATURE-1", "SIGNATURE-2")
6318 .replace("toolu_1", "toolu_2"))
6319 .collect(),
6320 reset_after,
6321 },
6322 other => other,
6323 }
6324 }
6325
6326 /// An Anthropic turn that just answers.
6327 fn anth_answer() -> Self {
6328 Self::Sse {
6329 chunks: vec![
6330 "event: message_start\ndata: {\"type\":\"message_start\",\"message\":\
6331 {\"id\":\"msg_2\",\"usage\":{\"input_tokens\":60,\
6332 \"output_tokens\":1}}}\n\n".to_string(),
6333 "event: content_block_start\ndata: {\"type\":\"content_block_start\",\
6334 \"index\":0,\"content_block\":{\"type\":\"text\",\"text\":\"\"}}\n\n"
6335 .to_string(),
6336 "event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\
6337 \"index\":0,\"delta\":{\"type\":\"text_delta\",\
6338 \"text\":\"It says hello.\"}}\n\n".to_string(),
6339 "event: content_block_stop\ndata: {\"type\":\"content_block_stop\",\
6340 \"index\":0}\n\n".to_string(),
6341 "event: message_delta\ndata: {\"type\":\"message_delta\",\
6342 \"delta\":{\"stop_reason\":\"end_turn\"},\
6343 \"usage\":{\"output_tokens\":8}}\n\n".to_string(),
6344 "event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n".to_string(),
6345 ],
6346 reset_after: None,
6347 }
6348 }
6349 }
6350
6351 /// What the stub provider saw, readable once the turn is over.
6352 #[derive(Default)]
6353 pub struct Seen {
6354 /// One entry per accepted connection, holding the request body.
6355 pub bodies: Vec<String>,
6356 }
6357
6358 /// A self-signed certificate and key for the stub, generated once.
6359 ///
6360 /// Real TLS, because the native transport has no other mode -- the client is
6361 /// exercised through exactly the path a provider gets.
6362 fn stub_cert() -> &'static (Vec<u8>, Vec<u8>) {
6363 static CERT: std::sync::OnceLock<(Vec<u8>, Vec<u8>)> = std::sync::OnceLock::new();
6364 CERT.get_or_init(|| {
6365 // Under the user cache, not the tmpfs at `/tmp`. The key is written to
6366 // disk for as long as openssl takes to write it, and a private key in a
6367 // tmpfs is a private key in the machine's memory.
6368 let dir = match oxedyne_fe2o3_test::scratch::scratch_dir("daimond_llm_cert") {
6369 Ok(d) => d,
6370 Err(e) => panic!("could not make a cert directory: {}", e),
6371 };
6372 let cert = dir.join("cert.pem");
6373 let key = dir.join("key.pem");
6374 let out = std::process::Command::new("openssl")
6375 // P-256, because the test verifier below advertises
6376 // `ECDSA_NISTP256_SHA256` and TLS 1.3 will not sign an RSA
6377 // certificate with any scheme it also advertises.
6378 .args(["req", "-x509", "-newkey", "ec",
6379 "-pkeyopt", "ec_paramgen_curve:prime256v1",
6380 "-nodes", "-days", "1", "-subj", "/CN=localhost"])
6381 .arg("-keyout").arg(&key)
6382 .arg("-out").arg(&cert)
6383 .output();
6384 let out = match out {
6385 Ok(o) => o,
6386 // Loudly, rather than skipping: a check that quietly does not run
6387 // is a check that proves nothing.
6388 Err(e) => panic!("openssl is required for the stub provider: {}", e),
6389 };
6390 assert!(out.status.success(), "openssl failed: {}",
6391 String::from_utf8_lossy(&out.stderr));
6392 let pair = match (std::fs::read(&cert), std::fs::read(&key)) {
6393 (Ok(c), Ok(k)) => (c, k),
6394 _ => panic!("openssl wrote no certificate"),
6395 };
6396 let _ = std::fs::remove_dir_all(&dir);
6397 pair
6398 })
6399 }
6400
6401 /// Start the stub provider on an ephemeral port.
6402 ///
6403 /// Each connection is served the next reply in `script`; the last one repeats
6404 /// for as long as the client keeps trying, so "gives up" is observable as a
6405 /// connection count rather than as a hang.
6406 pub async fn start_stub(script: Vec<Reply>) -> (u16, Arc<std::sync::Mutex<Seen>>) {
6407 // THE STUB IS A TLS SERVER AND NEEDS A PROVIDER TOO. Every client helper here installs
6408 // one, so in a whole-suite run some earlier test has always installed it process-wide by
6409 // the time a stub starts, and every stub test passed. Run one of them ALONE and the
6410 // server is built first, with nothing installed, and rustls panics -- so
6411 // `cargo test -- one_test_name` failed for a reason that had nothing to do with the test.
6412 //
6413 // That is not a hypothetical: a daimon changed the retry policy, wrote a test for it, and
6414 // told the user to prove it with exactly that command. Both it and the test beside it
6415 // would have failed, and the change would have looked broken. Idempotent, so installing
6416 // it here costs nothing where a client got there first.
6417 let _ = rustls::crypto::ring::default_provider().install_default();
6418 use tokio_rustls::rustls::ServerConfig;
6419 use tokio_rustls::rustls::pki_types::CertificateDer;
6420 use tokio_rustls::TlsAcceptor;
6421
6422 let (cert_pem, key_pem) = stub_cert();
6423 let certs: Vec<CertificateDer<'static>> = rustls_pemfile::certs(&mut &cert_pem[..])
6424 .filter_map(|c| c.ok())
6425 .collect();
6426 let key = match rustls_pemfile::private_key(&mut &key_pem[..]) {
6427 Ok(Some(k)) => k,
6428 _ => panic!("no private key in the stub's PEM"),
6429 };
6430 let cfg = match ServerConfig::builder().with_no_client_auth().with_single_cert(certs, key) {
6431 Ok(c) => c,
6432 Err(e) => panic!("stub TLS config: {}", e),
6433 };
6434 let acceptor = TlsAcceptor::from(Arc::new(cfg));
6435
6436 let listener = match tokio::net::TcpListener::bind(("127.0.0.1", 0)).await {
6437 Ok(l) => l,
6438 Err(e) => panic!("stub listen: {}", e),
6439 };
6440 let port = match listener.local_addr() {
6441 Ok(a) => a.port(),
6442 Err(e) => panic!("stub addr: {}", e),
6443 };
6444 let seen = Arc::new(std::sync::Mutex::new(Seen::default()));
6445 let seen_task = seen.clone();
6446
6447 tokio::spawn(async move {
6448 let mut n = 0usize;
6449 loop {
6450 let (tcp, _) = match listener.accept().await {
6451 Ok(v) => v,
6452 Err(_) => return,
6453 };
6454 let reply = script[n.min(script.len() - 1)].clone();
6455 n += 1;
6456 let acceptor = acceptor.clone();
6457 let seen = seen_task.clone();
6458 tokio::spawn(async move {
6459 let mut tls = match acceptor.accept(tcp).await {
6460 Ok(s) => s,
6461 Err(_) => return,
6462 };
6463 let body = read_request(&mut tls).await;
6464 if let Ok(mut g) = seen.lock() {
6465 g.bodies.push(body);
6466 }
6467 write_reply(&mut tls, &reply).await;
6468 });
6469 }
6470 });
6471 (port, seen)
6472 }
6473
6474 /// Read one HTTP request off the stream and return its body.
6475 async fn read_request(
6476 tls: &mut tokio_rustls::server::TlsStream<tokio::net::TcpStream>,
6477 ) -> String {
6478 let mut head = Vec::new();
6479 let mut byte = [0u8; 1];
6480 loop {
6481 match tls.read(&mut byte).await {
6482 Ok(0) => return String::new(),
6483 Ok(_) => {
6484 head.push(byte[0]);
6485 if head.ends_with(b"\r\n\r\n") { break; }
6486 }
6487 Err(_) => return String::new(),
6488 }
6489 }
6490 let head_str = String::from_utf8_lossy(&head).to_string();
6491 let len: usize = header_value(&head_str, "content-length")
6492 .and_then(|v| v.parse().ok())
6493 .unwrap_or(0);
6494 let mut body = vec![0u8; len];
6495 let mut got = 0usize;
6496 while got < len {
6497 match tls.read(&mut body[got..]).await {
6498 Ok(0) => break,
6499 Ok(n) => got += n,
6500 Err(_) => break,
6501 }
6502 }
6503 String::from_utf8_lossy(&body[..got]).to_string()
6504 }
6505
6506 /// Serve one scripted reply.
6507 async fn write_reply(
6508 tls: &mut tokio_rustls::server::TlsStream<tokio::net::TcpStream>,
6509 reply: &Reply,
6510 ) {
6511 match reply {
6512 Reply::Http { status, reason, headers, body } => {
6513 let mut out = fmt!("HTTP/1.1 {} {}\r\n", status, reason);
6514 for (k, v) in headers {
6515 out.push_str(&fmt!("{}: {}\r\n", k, v));
6516 }
6517 out.push_str(&fmt!("Content-Length: {}\r\n", body.len()));
6518 out.push_str("Connection: close\r\n\r\n");
6519 out.push_str(body);
6520 let _ = tls.write_all(out.as_bytes()).await;
6521 let _ = tls.flush().await;
6522 }
6523 Reply::Sse { chunks, reset_after } => {
6524 let head = "HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\n\
6525 Transfer-Encoding: chunked\r\nConnection: close\r\n\r\n";
6526 let _ = tls.write_all(head.as_bytes()).await;
6527 let _ = tls.flush().await;
6528 for (i, chunk) in chunks.iter().enumerate() {
6529 if Some(i) == *reset_after {
6530 // Abrupt reset: no close_notify, no final chunk -- the
6531 // provider vanishing mid-answer. A reset discards
6532 // anything still unacknowledged, so give what has
6533 // already gone out time to land first; otherwise the
6534 // client never sees the partial and the test proves
6535 // nothing about replaying it.
6536 tokio::time::sleep(std::time::Duration::from_millis(200)).await;
6537 let _ = tls.get_ref().0.set_linger(Some(std::time::Duration::ZERO));
6538 return;
6539 }
6540 let framed = fmt!("{:x}\r\n{}\r\n", chunk.len(), chunk);
6541 let _ = tls.write_all(framed.as_bytes()).await;
6542 let _ = tls.flush().await;
6543 }
6544 let _ = tls.write_all(b"0\r\n\r\n").await;
6545 let _ = tls.flush().await;
6546 }
6547 }
6548 }
6549
6550 /// A client pointed at the stub, with a fast retry policy so the suite does
6551 /// not spend its time asleep.
6552 pub fn stub_client(port: u16) -> LlmClient {
6553 let mut client = test_client("localhost", port, "anthropic/claude-opus-5");
6554 client.retry = RetryPolicy {
6555 max_attempts: 4,
6556 base_ms: 20,
6557 max_backoff_ms: 40,
6558 max_total_wait_ms: 5_000,
6559 };
6560 client
6561 }
6562
6563 /// A client with a certificate verifier that accepts the stub's self-signed
6564 /// certificate, at the OpenAI-compatible path.
6565 fn test_client(host: &str, port: u16, model: &str) -> LlmClient {
6566 test_client_at(host, port, "/v1/chat/completions", model)
6567 }
6568
6569 /// The same, at an explicit path -- which is what selects the [`Dialect`].
6570 fn test_client_at(host: &str, port: u16, path: &str, model: &str) -> LlmClient {
6571 use rustls::crypto::ring;
6572 let _ = ring::default_provider().install_default();
6573 let tls = Arc::new(
6574 ClientConfig::builder()
6575 .dangerous()
6576 .with_custom_certificate_verifier(Arc::new(NoVerify))
6577 .with_no_client_auth()
6578 );
6579 LlmClient::new(host, port, path, "key", model, 4096, tls)
6580 }
6581
6582 /// How many connections the stub accepted.
6583 pub fn connections(seen: &Arc<std::sync::Mutex<Seen>>) -> usize {
6584 match seen.lock() {
6585 Ok(g) => g.bodies.len(),
6586 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
6587 }
6588 }
6589
6590 /// A picture the endpoint will not take costs the pictures, not the turn.
6591 ///
6592 /// THE DEFECT. A daimon read a book's cover, the request went to a text-only model, and the
6593 /// provider answered a bare 404. The turn died -- and the picture stayed in the daimon's
6594 /// stored conversation, so every later turn re-sent it and died the same way. The Diamond's
6595 /// daimon was unusable until its whole conversation was thrown away.
6596 ///
6597 /// Neither existing guard could have caught it. `model_can_see` is a list of eight ids known
6598 /// to be blind, so an unheard-of model is assumed sighted; `vision_error` only rewrites a
6599 /// refusal whose text mentions images, and this one said "No endpoint found".
6600 #[tokio::test]
6601 async fn test_a_refused_picture_costs_the_pictures_and_not_the_turn() {
6602 let (port, seen) = start_stub(vec![
6603 Reply::not_found(),
6604 Reply::answer(),
6605 ]).await;
6606 let client = stub_client(port);
6607 let msgs = [ChatMessage::user(MessageContent::parts(vec![
6608 ContentPart::Text("what is on this cover".to_string()),
6609 ContentPart::Image(doc_image("cover.png")),
6610 ]))];
6611 let mut tokens = Vec::new();
6612 let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await {
6613 Ok(r) => r,
6614 Err(e) => panic!("a refused picture must not kill the turn: {}", e),
6615 };
6616
6617 assert_eq!(connections(&seen), 2, "the turn was not tried again without the picture");
6618 assert_eq!(resp.content, "Hello world", "the second attempt did not produce the answer");
6619
6620 let bodies = match seen.lock() {
6621 Ok(g) => g.bodies.clone(),
6622 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
6623 };
6624 // The first attempt carried it, so the failure being recovered from is the real one.
6625 assert!(bodies[0].contains(DOC_PNG_B64), "the first request did not carry the picture");
6626 // The second did not, and says why in its place -- a silently dropped image would leave
6627 // the model describing a cover nobody showed it.
6628 assert!(!bodies[1].contains(DOC_PNG_B64), "the picture was sent a second time");
6629 assert!(bodies[1].contains("cannot be shown"),
6630 "the model was not told the picture was left out: {}", bodies[1]);
6631 assert!(bodies[1].contains("cover.png"), "the file was not named in its place");
6632 assert!(bodies[1].contains("what is on this cover"), "the prose beside it was lost");
6633 // And the user is told, because a turn that quietly stops seeing is its own defect.
6634 assert!(tokens.iter().any(|t| t.contains("cannot see")),
6635 "nothing said the model had turned out to be blind: {:?}", tokens);
6636 }
6637
6638 /// And once it is known, no later turn pays to discover it again.
6639 #[tokio::test]
6640 async fn test_an_endpoint_caught_refusing_pictures_is_not_asked_twice() {
6641 let (port, seen) = start_stub(vec![
6642 Reply::not_found(),
6643 Reply::answer(),
6644 Reply::answer(),
6645 ]).await;
6646 let client = stub_client(port);
6647 let msgs = [ChatMessage::user(MessageContent::parts(vec![
6648 ContentPart::Text("and this one".to_string()),
6649 ContentPart::Image(doc_image("cover.png")),
6650 ]))];
6651 let mut sink = |_: Delta<'_>| {};
6652 let _ = client.chat_stream_tools(&msgs, None, &mut sink).await
6653 .expect("the first turn recovers");
6654 let _ = client.chat_stream_tools(&msgs, None, &mut sink).await
6655 .expect("the second turn goes straight through");
6656
6657 // Three replies were queued and only three connections may have been made: two for the
6658 // first turn, ONE for the second. A fourth would mean the client had forgotten.
6659 assert_eq!(connections(&seen), 3, "the second turn re-sent a picture already refused");
6660 let bodies = match seen.lock() {
6661 Ok(g) => g.bodies.clone(),
6662 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
6663 };
6664 assert!(!bodies[2].contains(DOC_PNG_B64),
6665 "the second turn sent the picture the endpoint had already refused");
6666 }
6667
6668 #[tokio::test]
6669 async fn test_a_429_is_retried_and_the_turn_completes() {
6670 let (port, seen) = start_stub(vec![
6671 Reply::too_many(Some(1)),
6672 Reply::answer(),
6673 ]).await;
6674 let client = stub_client(port);
6675 let msgs = [ChatMessage::user("hello".to_string())];
6676 let mut tokens = Vec::new();
6677
6678 let started = std::time::Instant::now();
6679 let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await {
6680 Ok(r) => r,
6681 Err(e) => panic!("a 429 followed by a 200 should complete: {}", e),
6682 };
6683 let elapsed = started.elapsed();
6684
6685 assert_eq!(connections(&seen), 2, "the stub was not asked a second time");
6686 assert_eq!(resp.content, "Hello world");
6687 assert_eq!(resp.retries, 1, "the retry was not counted for the user to see");
6688 // The provider asked for a second and got one: its own figure beat the
6689 // client's 20ms backoff.
6690 assert!(elapsed >= std::time::Duration::from_millis(1_000),
6691 "Retry-After was ignored; waited only {:?}", elapsed);
6692 // The answer streamed once, and the retry announced itself.
6693 let text: String = tokens.iter().filter(|t| !t.starts_with("\n[daimond")).cloned().collect();
6694 assert_eq!(text, "Hello world");
6695 let notice = match tokens.iter().find(|t| t.contains("[daimond")) {
6696 Some(n) => n.clone(),
6697 None => panic!("a retry the user cannot see is its own defect: {:?}", tokens),
6698 };
6699 assert!(notice.contains("HTTP 429"), "the notice does not say what happened: {}", notice);
6700 assert!(notice.contains("attempt 2 of 4"), "the notice does not say where we are: {}", notice);
6701 // The error's own rendering carries file, line and terminal colouring.
6702 assert!(!notice.contains('\u{1b}'),
6703 "ANSI escapes reached the user's message pane: {:?}", notice);
6704 // Provider-reported figures survive the retry.
6705 assert_eq!(resp.prompt_tokens, 11);
6706 assert_eq!(resp.cached_tokens, 9);
6707 assert_eq!(resp.cost_usd, 0.0003);
6708 }
6709
6710 #[tokio::test]
6711 async fn test_a_400_is_never_retried() {
6712 let (port, seen) = start_stub(vec![Reply::bad_request()]).await;
6713 let client = stub_client(port);
6714 let msgs = [ChatMessage::user("hello".to_string())];
6715 let mut tokens = Vec::new();
6716
6717 let result = client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await;
6718 assert!(result.is_err(), "a malformed request must not be reported as success");
6719 // The whole point: a 400 will fail the same way next time, and retrying
6720 // it only costs the user money and time.
6721 assert_eq!(connections(&seen), 1, "a 400 was sent again");
6722 assert!(tokens.is_empty(), "nothing should have streamed: {:?}", tokens);
6723 }
6724
6725 #[tokio::test]
6726 async fn test_a_5xx_is_retried_until_it_clears() {
6727 let (port, seen) = start_stub(vec![
6728 Reply::server_error(),
6729 Reply::server_error(),
6730 Reply::answer(),
6731 ]).await;
6732 let client = stub_client(port);
6733 let msgs = [ChatMessage::user("hello".to_string())];
6734 let mut tokens = Vec::new();
6735
6736 let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await {
6737 Ok(r) => r,
6738 Err(e) => panic!("two 500s then a 200 should complete: {}", e),
6739 };
6740 assert_eq!(connections(&seen), 3);
6741 assert_eq!(resp.content, "Hello world");
6742 assert_eq!(resp.retries, 2);
6743 }
6744
6745 #[tokio::test]
6746 async fn test_a_refusal_carries_the_providers_own_words() {
6747 // What the compactor reads to tell an oversized prompt from a malformed one.
6748 // Without the body it has only the status and a size estimate to go on, and a
6749 // provider that publishes no window can then kill a chat permanently.
6750 let over = "{\"error\":{\"message\":\"This model's maximum context length is \
6751 131072 tokens, however you requested 174233 tokens.\",\
6752 \"code\":\"context_length_exceeded\"}}";
6753 let (port, _seen) = start_stub(vec![Reply::Http {
6754 status: 400, reason: "Bad Request", headers: Vec::new(), body: over.to_string(),
6755 }]).await;
6756 let client = stub_client(port);
6757 let msgs = [ChatMessage::user("hello".to_string())];
6758
6759 let e = match client.chat_stream_tools(&msgs, None, &mut |_| {}).await {
6760 Ok(_) => panic!("a 400 must not be reported as success"),
6761 Err(e) => fmt!("{}", e),
6762 };
6763 assert!(e.contains("maximum context length"),
6764 "the provider said why and the error does not: {}", e);
6765 assert!(e.contains("400"), "{}", e);
6766 // And that is enough on its own -- no size estimate needed.
6767 assert!(crate::agent::compact::looks_like_overflow(&e, 0, 100_000),
6768 "the words the provider used were not recognised: {}", e);
6769 }
6770
6771 // ── A reply that hit the output limit ───────────────────────────────
6772
6773 #[test]
6774 fn test_both_dialects_say_when_a_reply_ran_out_of_room() {
6775 // Neither was read anywhere outside a test, so the browser had to infer
6776 // truncation from tool arguments that would not parse -- which cannot see a
6777 // plain text reply cut short, and cannot tell the model anything at all.
6778 assert!(openai_truncated(
6779 "{\"choices\":[{\"index\":0,\"delta\":{},\"finish_reason\":\"length\"}]}"));
6780 assert!(anthropic_truncated(
6781 "{\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"max_tokens\"}}"));
6782 // And a reply that simply finished is not truncated, in either dialect.
6783 assert!(!openai_truncated(
6784 "{\"choices\":[{\"delta\":{},\"finish_reason\":\"stop\"}]}"));
6785 assert!(!openai_truncated(
6786 "{\"choices\":[{\"delta\":{\"content\":\"hi\"},\"finish_reason\":null}]}"));
6787 assert!(!openai_truncated(
6788 "{\"choices\":[{\"delta\":{},\"finish_reason\":\"tool_calls\"}]}"));
6789 assert!(!anthropic_truncated(
6790 "{\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"end_turn\"}}"));
6791 assert!(!anthropic_truncated(
6792 "{\"type\":\"message_start\",\"message\":{\"stop_reason\":null}}"));
6793 }
6794
6795 #[test]
6796 fn test_a_stream_cut_at_the_limit_says_so_on_the_response() {
6797 // Through the accumulator, which is where the app reads it: the flag is sticky,
6798 // because the usage chunk arrives AFTER the finish reason and must not unsay it.
6799 let mut acc = StreamAcc::default();
6800 acc.ingest("{\"choices\":[{\"delta\":{\"content\":\"fn main\"}}]}", &mut |_| {});
6801 assert!(!acc.into_response(false, 0).truncated);
6802
6803 let mut acc = StreamAcc::default();
6804 acc.ingest("{\"choices\":[{\"delta\":{\"content\":\"fn main\"}}]}", &mut |_| {});
6805 acc.ingest("{\"choices\":[{\"delta\":{},\"finish_reason\":\"length\"}]}", &mut |_| {});
6806 acc.ingest("{\"choices\":[],\"usage\":{\"prompt_tokens\":9,\"completion_tokens\":8192}}",
6807 &mut |_| {});
6808 let r = acc.into_response(false, 0);
6809 assert!(r.truncated, "the usage chunk unsaid the finish reason");
6810 assert_eq!(r.completion_tokens, 8192);
6811 }
6812
6813 #[test]
6814 fn test_an_anthropic_stream_cut_at_the_limit_says_so_too() {
6815 let mut acc = AnthropicAcc::default();
6816 acc.ingest("{\"type\":\"message_start\",\"message\":{\"usage\":{\"input_tokens\":9}}}",
6817 &mut |_| {});
6818 acc.ingest("{\"type\":\"content_block_delta\",\"index\":0,\
6819 \"delta\":{\"type\":\"text_delta\",\"text\":\"fn main\"}}", &mut |_| {});
6820 assert!(!acc.truncated, "nothing has said the reply was cut");
6821 acc.ingest("{\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"max_tokens\"},\
6822 \"usage\":{\"output_tokens\":8192}}", &mut |_| {});
6823 assert!(acc.into_response(false, 0).truncated);
6824 }
6825
6826 #[tokio::test]
6827 async fn test_a_truncated_reply_is_not_an_error_and_is_not_retried() {
6828 // The interaction that matters. A reply that hit `max_tokens` is a complete
6829 // HTTP 200: sending it again costs money and produces the same cut, and treating
6830 // it as a failure would throw away text the user has already been shown.
6831 let (port, seen) = start_stub(vec![Reply::Sse { chunks: vec![
6832 "data: {\"choices\":[{\"delta\":{\"content\":\"fn main() {\"}}]}\n\n".to_string(),
6833 "data: {\"choices\":[{\"delta\":{},\"finish_reason\":\"length\"}]}\n\n".to_string(),
6834 "data: [DONE]\n\n".to_string(),
6835 ], reset_after: None }]).await;
6836 let client = stub_client(port);
6837 let msgs = [ChatMessage::user("write the file".to_string())];
6838
6839 let r = match client.chat_stream_tools(&msgs, None, &mut |_| {}).await {
6840 Ok(r) => r,
6841 Err(e) => panic!("a reply that hit the cap is not a failed call: {}", e),
6842 };
6843 assert!(r.truncated, "the cap was reached and the response does not say so");
6844 assert_eq!(r.content, "fn main() {", "the partial answer was thrown away");
6845 assert_eq!(connections(&seen), 1, "a complete 200 was sent again");
6846 assert_eq!(r.retries, 0);
6847 }
6848
6849 #[test]
6850 fn test_a_refusal_body_is_cut_without_splitting_a_character() {
6851 // A provider's body is arbitrary bytes on an error path, which is exactly where
6852 // a panic is least welcome and least likely to be found in testing. `&s[..300]`
6853 // on a multi-byte boundary is a panic, not a truncation.
6854 // One ASCII byte in front, so the two-byte characters after it sit on ODD
6855 // offsets and the cut at 300 lands in the middle of one. Without the offset the
6856 // boundaries happen to line up and a broken clip passes.
6857 let s = fmt!("a{}", "é".repeat(400));
6858 assert!(!s.is_char_boundary(ERR_BODY_BYTES), "the fixture must actually straddle");
6859 let cut = clip_bytes(&s, ERR_BODY_BYTES);
6860 assert!(cut.len() <= ERR_BODY_BYTES);
6861 assert!(cut.chars().skip(1).all(|c| c == 'é'), "a character was split");
6862 // A short body is untouched, and an empty one is not a special case.
6863 assert_eq!(clip_bytes("short", ERR_BODY_BYTES), "short");
6864 assert_eq!(clip_bytes("", ERR_BODY_BYTES), "");
6865 }
6866
6867 #[tokio::test]
6868 async fn test_retrying_stops_at_the_attempt_budget() {
6869 // A provider that is never ready: the attempt must end, not loop.
6870 let (port, seen) = start_stub(vec![Reply::too_many(None)]).await;
6871 let mut client = stub_client(port);
6872 client.retry.max_attempts = 3;
6873 let msgs = [ChatMessage::user("hello".to_string())];
6874
6875 let result = client.chat_stream_tools(&msgs, None, &mut |_| {}).await;
6876 assert!(result.is_err());
6877 assert_eq!(connections(&seen), 3,
6878 "the attempt budget was not the bound on how many requests went out");
6879 }
6880
6881 #[tokio::test]
6882 async fn test_a_retry_after_beyond_the_wait_bound_ends_the_attempt() {
6883 // The provider asks for a minute; the user is watching a spinner. The
6884 // turn ends rather than honouring it.
6885 let (port, seen) = start_stub(vec![Reply::too_many(Some(60))]).await;
6886 let mut client = stub_client(port);
6887 client.retry.max_total_wait_ms = 2_000;
6888 let msgs = [ChatMessage::user("hello".to_string())];
6889
6890 let started = std::time::Instant::now();
6891 let result = client.chat_stream_tools(&msgs, None, &mut |_| {}).await;
6892 assert!(result.is_err());
6893 assert_eq!(connections(&seen), 1);
6894 assert!(started.elapsed() < std::time::Duration::from_secs(5),
6895 "the client slept through a Retry-After it had no budget for");
6896 }
6897
6898 #[tokio::test]
6899 async fn test_a_stream_that_breaks_after_tokens_is_not_replayed() {
6900 // THE streaming hazard. The provider streams one delta and then
6901 // vanishes; a retry here would hand the caller "Hello" a second time.
6902 let (port, seen) = start_stub(vec![
6903 Reply::Sse {
6904 chunks: vec![
6905 "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\n".to_string(),
6906 "data: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\n".to_string(),
6907 ],
6908 reset_after: Some(1),
6909 },
6910 Reply::answer(),
6911 ]).await;
6912 let client = stub_client(port);
6913 let msgs = [ChatMessage::user("hello".to_string())];
6914 let mut tokens = Vec::new();
6915
6916 let _ = client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await;
6917
6918 let text: String = tokens.iter().filter(|t| !t.starts_with("\n[daimond")).cloned().collect();
6919 assert_eq!(text, "Hello",
6920 "the partial was replayed or lost -- got {:?}", tokens);
6921 assert_eq!(connections(&seen), 1,
6922 "the turn was restarted after tokens had already reached the caller");
6923 }
6924
6925 #[tokio::test]
6926 async fn test_a_stream_that_breaks_after_a_tool_call_fragment_is_not_replayed() {
6927 // No text has streamed, so `emitted` is false -- but a half-built tool
6928 // call is output all the same, and starting over would either duplicate
6929 // the call or splice two halves of different ones together.
6930 let (port, seen) = start_stub(vec![
6931 Reply::Sse {
6932 chunks: vec![
6933 "data: {\"choices\":[{\"delta\":{\"tool_calls\":[{\"index\":0,\"id\":\"c0\",\
6934 \"function\":{\"name\":\"file_read\",\"arguments\":\"{\\\"path\\\":\\\"\"}}]}}]}\n\n"
6935 .to_string(),
6936 "data: {\"choices\":[{\"delta\":{\"tool_calls\":[{\"index\":0,\
6937 \"function\":{\"arguments\":\"a.txt\\\"}\"}}]}}]}\n\n".to_string(),
6938 ],
6939 reset_after: Some(1),
6940 },
6941 Reply::answer(),
6942 ]).await;
6943 let client = stub_client(port);
6944 let msgs = [ChatMessage::user("hello".to_string())];
6945 let mut tokens = Vec::new();
6946
6947 let _ = client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await;
6948
6949 assert_eq!(connections(&seen), 1,
6950 "the turn was restarted on top of a partial tool call");
6951 assert!(tokens.iter().all(|t| t.starts_with("\n[daimond")),
6952 "text streamed from a replayed turn: {:?}", tokens);
6953 }
6954
6955 #[tokio::test]
6956 async fn test_a_stream_that_breaks_before_any_token_is_retried() {
6957 // A network drop before the provider has streamed anything — the laptop
6958 // moving between locations, the wifi handing off — is the failure the
6959 // widened retry policy is for. The stream resets on chunk 0, nothing has
6960 // been emitted, and the turn should start over and complete. This is the
6961 // exact shape of "the stream broke" the user sees on the road.
6962 let (port, seen) = start_stub(vec![
6963 Reply::Sse {
6964 chunks: vec![
6965 "data: {\"choices\":[{\"delta\":{\"content\":\"Hello\"}}]}\n\n".to_string(),
6966 "data: {\"choices\":[{\"delta\":{\"content\":\" world\"}}]}\n\n".to_string(),
6967 "data: [DONE]\n\n".to_string(),
6968 ],
6969 reset_after: Some(0),
6970 },
6971 Reply::answer(),
6972 ]).await;
6973 let client = stub_client(port);
6974 let msgs = [ChatMessage::user("hello".to_string())];
6975 let mut tokens = Vec::new();
6976
6977 let resp = match client.chat_stream_tools(&msgs, None, &mut text_sink(&mut tokens)).await {
6978 Ok(r) => r,
6979 Err(e) => panic!("a stream that breaks before tokens should recover: {}", e),
6980 };
6981
6982 assert_eq!(connections(&seen), 2,
6983 "the broken stream was not retried");
6984 assert_eq!(resp.content, "Hello world",
6985 "the retry did not produce the answer");
6986 assert_eq!(resp.retries, 1,
6987 "the retry was not counted");
6988 let text: String = tokens.iter().filter(|t| !t.starts_with("\n[daimond")).cloned().collect();
6989 assert_eq!(text, "Hello world",
6990 "the answer was not streamed cleanly after the retry: {:?}", tokens);
6991 }
6992
6993 #[tokio::test]
6994 async fn test_the_breakpoint_reaches_the_wire() {
6995 // What the provider actually receives, read back off its own socket.
6996 let (port, seen) = start_stub(vec![Reply::answer()]).await;
6997 let client = stub_client(port);
6998 let msgs = [
6999 ChatMessage::system(long_system()),
7000 ChatMessage::user("hello".to_string()),
7001 ];
7002 let _ = client.chat_stream_tools(&msgs, None, &mut |_| {}).await;
7003
7004 let body = match seen.lock() {
7005 Ok(g) => g.bodies[0].clone(),
7006 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
7007 };
7008 assert!(body.contains("\"cache_control\":{\"type\":\"ephemeral\"}"),
7009 "no breakpoint reached the provider: {}", body);
7010 assert!(body.contains("\"role\":\"system\",\"content\":[{\"type\":\"text\""));
7011 }
7012
7013 /// A client pointed at the stub, speaking the Messages API.
7014 fn anth_stub_client(port: u16) -> LlmClient {
7015 let mut client = test_client_at("localhost", port, "/v1/messages", "claude-opus-5");
7016 client.retry = RetryPolicy {
7017 max_attempts: 4,
7018 base_ms: 20,
7019 max_backoff_ms: 40,
7020 max_total_wait_ms: 5_000,
7021 };
7022 client
7023 }
7024
7025 #[tokio::test]
7026 async fn test_the_messages_api_request_reaches_the_wire_in_its_own_shape() {
7027 // What the provider actually receives, read back off its own socket --
7028 // not what this file believes it sent.
7029 let (port, seen) = start_stub(vec![Reply::anth_answer()]).await;
7030 let client = anth_stub_client(port);
7031 let tools = r#"[{"type":"function","function":{"name":"file_read",
7032 "description":"Read a file","parameters":{"type":"object","properties":{}}}}]"#;
7033 let msgs = [
7034 ChatMessage::system(long_system()),
7035 ChatMessage::user("hello".to_string()),
7036 ];
7037 let resp = match client.chat_stream_tools(&msgs, Some(tools), &mut |_| {}).await {
7038 Ok(r) => r,
7039 Err(e) => panic!("the Messages API turn failed: {}", e),
7040 };
7041 assert_eq!(resp.content, "It says hello.");
7042 assert_eq!(resp.prompt_tokens, 60);
7043 assert_eq!(resp.completion_tokens, 8);
7044
7045 let body = match seen.lock() {
7046 Ok(g) => g.bodies[0].clone(),
7047 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
7048 };
7049 assert!(body.contains("\"system\":[{\"type\":\"text\""),
7050 "the system prompt did not reach the wire hoisted: {}", body);
7051 assert!(!body.contains("\"role\":\"system\""), "{}", body);
7052 assert!(body.contains("\"input_schema\""),
7053 "the tools reached the wire in the OpenAI shape: {}", body);
7054 assert!(body.contains("\"thinking\":{\"type\":\"adaptive\""), "{}", body);
7055 assert!(body.contains("\"cache_control\":{\"type\":\"ephemeral\"}"),
7056 "no breakpoint reached the provider: {}", body);
7057 assert!(!body.contains("\"stream_options\""),
7058 "an OpenAI-only field reached the Messages API: {}", body);
7059 }
7060
7061 #[tokio::test]
7062 async fn test_thinking_blocks_are_handed_back_with_the_tool_results() {
7063 // The constraint that produces an error on EVERY following turn when it
7064 // is missed: within a tool-use turn, the signed thinking blocks must go
7065 // back complete and unmodified, ahead of the tool_use block they
7066 // accompanied. Two real requests, and the second one is read off the
7067 // provider's socket.
7068 let (port, seen) = start_stub(vec![
7069 Reply::anth_thinks_then_calls(),
7070 Reply::anth_answer(),
7071 ]).await;
7072 let client = anth_stub_client(port);
7073 let tools = r#"[{"type":"function","function":{"name":"file_read",
7074 "description":"Read a file","parameters":{"type":"object","properties":{}}}}]"#;
7075
7076 // Round one: the model thinks, then asks for a tool.
7077 let first = match client.chat_stream_tools(
7078 &[ChatMessage::user("read a.txt".to_string())],
7079 Some(tools), &mut |_| {}).await
7080 {
7081 Ok(r) => r,
7082 Err(e) => panic!("round one failed: {}", e),
7083 };
7084 assert_eq!(first.tool_calls.len(), 1);
7085 assert_eq!(first.tool_calls[0].id, "toolu_1");
7086 assert_eq!(first.thinking, "I should read the file.");
7087 assert_eq!(first.cached_tokens, 900);
7088 assert_eq!(first.prompt_tokens, 930, "the cached prefix is part of the prompt");
7089
7090 // Round two: the agent loop's shape -- the assistant turn that asked,
7091 // then the result.
7092 let round_two = vec![
7093 ChatMessage::user("read a.txt".to_string()),
7094 ChatMessage::Assistant {
7095 content: MessageContent::text(""),
7096 tool_calls: first.tool_calls.clone(),
7097 },
7098 ChatMessage::tool("toolu_1".to_string(), "hello".to_string()),
7099 ];
7100 if let Err(e) = client.chat_stream_tools(&round_two, Some(tools), &mut |_| {}).await {
7101 panic!("round two failed: {}", e);
7102 }
7103
7104 let body = match seen.lock() {
7105 Ok(g) => g.bodies[1].clone(),
7106 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
7107 };
7108 assert!(body.contains("\"signature\":\"SIGNATURE-1\""),
7109 "the signed thinking block never went back: {}", body);
7110 assert!(body.contains("I should read the file."),
7111 "the thinking text was dropped, which the API reads as a modified block: {}", body);
7112 // Order matters: the reasoning precedes the call it produced.
7113 let think_at = match body.find("\"type\":\"thinking\"") {
7114 Some(p) => p,
7115 None => panic!("no thinking block in the assistant turn: {}", body),
7116 };
7117 let call_at = match body.find("\"type\":\"tool_use\"") {
7118 Some(p) => p,
7119 None => panic!("no tool_use block: {}", body),
7120 };
7121 assert!(think_at < call_at,
7122 "the thinking block was placed after the call it led to: {}", body);
7123 }
7124
7125 #[tokio::test]
7126 async fn test_reasoning_only_ever_goes_back_with_the_call_that_produced_it() {
7127 // Blocks are kept across a whole tool loop -- that is the documented
7128 // recommendation, and on the models that keep them it is what makes the
7129 // round-by-round cache hits happen. What must NOT happen is one turn's
7130 // reasoning being glued to a different turn's call: within an assistant
7131 // message the run has to match what the model generated there, so a
7132 // block from elsewhere is a rearrangement and a 400.
7133 let (port, seen) = start_stub(vec![
7134 Reply::anth_thinks_then_calls(),
7135 Reply::anth_answer(),
7136 ]).await;
7137 let client = anth_stub_client(port);
7138 let msgs = [ChatMessage::user("hello".to_string())];
7139 let _ = client.chat_stream_tools(&msgs, None, &mut |_| {}).await;
7140
7141 // A later turn quoting a DIFFERENT call: the held reasoning belongs to
7142 // `toolu_1`, and nothing may hand it to `toolu_other`.
7143 let elsewhere = vec![
7144 ChatMessage::user("hello".to_string()),
7145 ChatMessage::Assistant {
7146 content: MessageContent::text(""),
7147 tool_calls: vec![ToolCall { id: "toolu_other".to_string(),
7148 name: "file_read".to_string(), arguments: "{}".to_string() }],
7149 },
7150 ChatMessage::Tool { tool_call_id: "toolu_other".to_string(),
7151 content: MessageContent::text("x") },
7152 ];
7153 let _ = client.chat_stream_tools(&elsewhere, None, &mut |_| {}).await;
7154 let body = match seen.lock() {
7155 Ok(g) => g.bodies[1].clone(),
7156 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
7157 };
7158 assert!(!body.contains("SIGNATURE-1"),
7159 "one turn's reasoning was handed to another turn's call: {}", body);
7160 assert!(!body.contains("\"type\":\"thinking\""), "{}", body);
7161 }
7162
7163 #[tokio::test]
7164 async fn test_every_round_of_a_tool_loop_keeps_its_own_reasoning() {
7165 // Round three carries BOTH earlier rounds' blocks, each beside its own
7166 // call. A client that held only the latest would drop the first
7167 // round's reasoning from a loop that is, to the model, one turn -- and
7168 // with it the cache hit that the tool results were supposed to earn.
7169 let (port, seen) = start_stub(vec![
7170 Reply::anth_thinks_then_calls(),
7171 Reply::anth_thinks_then_calls_again(),
7172 Reply::anth_answer(),
7173 ]).await;
7174 let client = anth_stub_client(port);
7175 let call = |id: &str| ToolCall {
7176 id: id.to_string(), name: "file_read".to_string(), arguments: "{}".to_string() };
7177
7178 let mut working = vec![ChatMessage::user("read them".to_string())];
7179 for _ in 0..2 {
7180 let r = match client.chat_stream_tools(&working, None, &mut |_| {}).await {
7181 Ok(r) => r,
7182 Err(e) => panic!("round failed: {}", e),
7183 };
7184 let id = r.tool_calls[0].id.clone();
7185 working.push(ChatMessage::Assistant {
7186 content: MessageContent::text(""), tool_calls: vec![call(&id)] });
7187 working.push(ChatMessage::tool(id, "ok".to_string()));
7188 }
7189 let _ = client.chat_stream_tools(&working, None, &mut |_| {}).await;
7190
7191 let body = match seen.lock() {
7192 Ok(g) => g.bodies[2].clone(),
7193 Err(e) => panic!("stub bookkeeping poisoned: {}", e),
7194 };
7195 assert!(body.contains("SIGNATURE-1"),
7196 "the first round's reasoning was dropped from the loop: {}", body);
7197 assert!(body.contains("SIGNATURE-2"),
7198 "the second round's reasoning was dropped: {}", body);
7199 assert_eq!(body.matches("\"type\":\"thinking\"").count(), 2, "{}", body);
7200 }
7201
7202 // Test verifier that accepts any certificate (for unit tests only).
7203 use tokio_rustls::rustls::client::danger::{HandshakeSignatureValid, ServerCertVerified, ServerCertVerifier};
7204 use std::sync::Arc;
7205
7206 #[derive(Debug)]
7207 pub struct NoVerify;
7208
7209 impl ServerCertVerifier for NoVerify {
7210 fn verify_server_cert(
7211 &self,
7212 _end_entity: &tokio_rustls::rustls::pki_types::CertificateDer<'_>,
7213 _intermediates: &[tokio_rustls::rustls::pki_types::CertificateDer<'_>],
7214 _server_name: &tokio_rustls::rustls::pki_types::ServerName<'_>,
7215 _ocsp_response: &[u8],
7216 _now: tokio_rustls::rustls::pki_types::UnixTime,
7217 ) -> Result<ServerCertVerified, tokio_rustls::rustls::Error> {
7218 Ok(ServerCertVerified::assertion())
7219 }
7220 fn verify_tls12_signature(
7221 &self,
7222 _message: &[u8],
7223 _cert: &tokio_rustls::rustls::pki_types::CertificateDer<'_>,
7224 _dss: &tokio_rustls::rustls::DigitallySignedStruct,
7225 ) -> Result<HandshakeSignatureValid, tokio_rustls::rustls::Error> {
7226 Ok(HandshakeSignatureValid::assertion())
7227 }
7228 fn verify_tls13_signature(
7229 &self,
7230 _message: &[u8],
7231 _cert: &tokio_rustls::rustls::pki_types::CertificateDer<'_>,
7232 _dss: &tokio_rustls::rustls::DigitallySignedStruct,
7233 ) -> Result<HandshakeSignatureValid, tokio_rustls::rustls::Error> {
7234 Ok(HandshakeSignatureValid::assertion())
7235 }
7236 fn supported_verify_schemes(&self) -> Vec<tokio_rustls::rustls::SignatureScheme> {
7237 vec![
7238 tokio_rustls::rustls::SignatureScheme::RSA_PKCS1_SHA256,
7239 tokio_rustls::rustls::SignatureScheme::ECDSA_NISTP256_SHA256,
7240 tokio_rustls::rustls::SignatureScheme::ED25519,
7241 ]
7242 }
7243 }
7244}