oxedyne/daimond/dev/mockllm.mjs
35.6 KiB, 1 run
created by r2519314175:29, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // Mock LLM provider — an OpenAI-compatible endpoint that answers to a script. |
| 2 | // |
| 3 | // The agent loop is the one part of Daimond that could never be driven in a |
| 4 | // test, because it needs a real provider and a real key. This stands in for |
| 5 | // one: it speaks the same wire format (streaming and not, tool calls and not), |
| 6 | // but what it says is dictated by a directive in the user's own message, so a |
| 7 | // test can ask for exactly the reply it wants to exercise. |
| 8 | // |
| 9 | // node dev/mockllm.mjs [port] # default 9099 |
| 10 | // |
| 11 | // Point Daimond at it with provider "Custom" and base URL |
| 12 | // http://127.0.0.1:9099/v1/chat/completions |
| 13 | // Any key is accepted. |
| 14 | // |
| 15 | // ── The directive language ──────────────────────────────────────────────── |
| 16 | // A user message beginning with `@` is a directive to the mock, not a prompt. |
| 17 | // |
| 18 | // @text <words> plain assistant reply |
| 19 | // @long <n> stream <n> chunks slowly (exercises Stop/abort) |
| 20 | // @tool <name> <json> one tool call, then a text reply once it returns |
| 21 | // @tools <name> <json> ;; <name> <json> several tool calls in one turn |
| 22 | // @chain <name> <json> tool call, then a second call, then text |
| 23 | // @narrate <words> ;; <name> <json> |
| 24 | // @reason <working> ;; <answer> |
| 25 | // the model THINKS before it answers, on the wire the way |
| 26 | // OpenRouter sends it: `delta.reasoning` with the same words |
| 27 | // repeated in `delta.reasoning_details`, and `content` empty |
| 28 | // until the working is done. A client that reads both fields |
| 29 | // shows every word twice, which is what this is for. |
| 30 | // @reasonslow <working> ;; <answer> |
| 31 | // the same as @reason, paced the way a real reasoning round |
| 32 | // is paced, so a check can look at the page while the model |
| 33 | // is still thinking rather than only after. |
| 34 | // @reasonc <working> ;; <answer> |
| 35 | // the same, spelled `delta.reasoning_content` -- which is |
| 36 | // what DeepSeek's own endpoint calls it, and the reason the |
| 37 | // accumulator reads two field names. |
| 38 | // @reasontool <working> ;; <name> <json> |
| 39 | // working, then a tool call and no text at all: the round |
| 40 | // that spends a minute and a half thinking and then says one |
| 41 | // word, which is the round this was all built for. |
| 42 | // PROSE AND THEN A TOOL CALL, in one assistant message, |
| 43 | // then a text reply once the tool returns. Every other |
| 44 | // directive here emits `content: null` beside its calls, |
| 45 | // so until 2026-08-23 NO fixture in this repository |
| 46 | // produced the shape a real provider produces constantly |
| 47 | // -- the model saying "let me check the line numbers" |
| 48 | // and then checking them. That is the shape the owner |
| 49 | // read twenty of in one turn while asking where the |
| 50 | // folding was, and nothing could have caught it, because |
| 51 | // nothing could make it happen. |
| 52 | // @usage <in> <out> [cost] [cached] |
| 53 | // reply reporting those token counts, and -- when the |
| 54 | // trailing two are given -- the USD the provider says |
| 55 | // the call cost and the prompt tokens it served from |
| 56 | // its cache. The trailing pair is what a router |
| 57 | // actually sends (`cost`, and `cached_tokens` nested |
| 58 | // in `prompt_tokens_details`), and it is the only way |
| 59 | // a test can prove the app bills the REPORTED figure |
| 60 | // rather than its own table's guess. |
| 61 | // @err <code> fail with that HTTP status (the error path) |
| 62 | // @drop <n> stream n words, then DESTROY the socket -- the road |
| 63 | // failing part way through an answer, which is not the |
| 64 | // same event as the provider refusing and must not be |
| 65 | // treated as one. There was no way to produce it, so |
| 66 | // the branch that tells them apart could not be tested. |
| 67 | // @slow <ms> reply after a delay |
| 68 | // @look <path> the model that keeps trying to LOOK. It answers with |
| 69 | // `file_read {"path":…,"as":"image"}` until a tool result |
| 70 | // has come back, and with words afterwards. Unlike every |
| 71 | // directive above it is read from the WHOLE transcript |
| 72 | // rather than from the last user message, because a worker |
| 73 | // moved to another model is RESUMED -- its new session is |
| 74 | // seeded with its own task and text and the last thing in |
| 75 | // it is the app's nudge. A mock that only read the last |
| 76 | // message would answer that nudge with prose, the second |
| 77 | // leg would carry no picture, and a verifier could not |
| 78 | // tell a correct re-route from a broken one. |
| 79 | // |
| 80 | // ── Two roles are answered by their SYSTEM prompt, not by a directive ───── |
| 81 | // |
| 82 | // A reducer and a triage both have a fixed ANSWER SHAPE that the app parses, so |
| 83 | // a mock replying with prose could not drive either feature at all. Both are |
| 84 | // recognised the way a real one is -- by the role they were given -- and both |
| 85 | // still lose to `@text`, so a test that wants a malformed answer can ask for one. |
| 86 | // |
| 87 | // reducer system prompt mentions "crystal"; answered with a crystal whose |
| 88 | // summary is the delta's own words. |
| 89 | // triage system prompt opens "triaging one person's notes"; answered with a |
| 90 | // plan for www/js/triage.js. `MOCK_TRIAGE_PLAN=<file>` answers with |
| 91 | // that file, so a test can replay a real model's clustering; without |
| 92 | // it, one draft per note id found in the brief. |
| 93 | // |
| 94 | // Anything else gets a short generic reply. Every request is appended to |
| 95 | // dev/mockllm.log as JSON lines, so a test can assert on what the model was |
| 96 | // actually shown — the system prompt, the tool results, the whole transcript. |
| 97 | // |
| 98 | // ── Blindness is a property of the MODEL NAME ───────────────────────────── |
| 99 | // |
| 100 | // A model whose id contains `blind` refuses any request that carries a picture, with |
| 101 | // the 400 and the words a real provider uses; `mock/eyes` takes it. Both answer at |
| 102 | // THIS endpoint, with this key, at the same time, which is the whole point: a test of |
| 103 | // vision routing has to watch the work move from one model to another, and if |
| 104 | // blindness were a directive or a flag it would be a property of the REQUEST — the |
| 105 | // very thing the app under test is being asked to change. So it is a property of the |
| 106 | // name, which is what it is in the real world. |
| 107 | // |
| 108 | // The refusal's shape is not decorative. `LlmClient::stream_turn` (src/llm.rs, the |
| 109 | // `!started && !retried_blind && images > 0` arm) learns that an endpoint is blind by |
| 110 | // being refused ONCE: it calls `mark_blind`, takes the pictures out, and sends the same |
| 111 | // turn again. A 200 saying no, or a refusal before the pictures are logged, teaches it |
| 112 | // nothing. Hence: log first, then refuse, with `image_url` in the provider's own words. |
| 113 | // |
| 114 | // Every logged request carries `images` — how many picture parts it held, in either |
| 115 | // dialect — and a refused one carries `refusedImages`. Those two fields are what let a |
| 116 | // verifier say "the second leg named the other model AND carried the picture", which is |
| 117 | // what a re-route looks like from outside. |
| 118 | |
| 119 | import http from 'node:http'; |
| 120 | import crypto from 'node:crypto'; |
| 121 | import fs from 'node:fs'; |
| 122 | import path from 'node:path'; |
| 123 | import { fileURLToPath } from 'node:url'; |
| 124 | |
| 125 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 126 | // The log is per-world, not per-repo: two mocks appending to one file make every |
| 127 | // assertion that reads it see another agent's traffic. See dev/world.sh. |
| 128 | const LOG = process.env.DAIMOND_MOCK_LOG || path.join(HERE, 'mockllm.log'); |
| 129 | const PORT = Number(process.argv[2] || process.env.DAIMOND_MOCK_PORT || 9099); |
| 130 | |
| 131 | // ── WHICH REVISION OF THIS FILE IS ANSWERING ──────────────────────────────── |
| 132 | // |
| 133 | // A port is not evidence of anything, and neither is a log path. `world.sh --up` |
| 134 | // deliberately ADOPTS a mock it finds already listening, and `requireOwnMock` in |
| 135 | // dev/harness.mjs then asks it which file it writes -- which settles whether it is |
| 136 | // this WORLD'S mock and says nothing about whether it is this TREE'S. |
| 137 | // |
| 138 | // Those two questions came apart on 2026-08-28. A world was brought up, the tree |
| 139 | // under it was then merged forward, and the mock from before the merge kept |
| 140 | // answering: same port, same log path, so every existing guard passed. It did not |
| 141 | // know the directives the merge had added, so it fell through to `Mock reply to: |
| 142 | // <the whole directive line>` -- and `dev/verify_thinking.mjs` went red on twelve |
| 143 | // checks, one of which reported that the model's reasoning had been stored as its |
| 144 | // answer. Nothing of the sort had happened. The mock had echoed the prompt, the |
| 145 | // prompt contained the markers the check looks for, and the most alarming message |
| 146 | // the suite can print was produced by a fixture rather than by the product. |
| 147 | // |
| 148 | // The exact-source question is the one that catches it, and it catches every other |
| 149 | // version of it at the same time -- a directive whose meaning changed, a bug fixed |
| 150 | // in the fixture, an argument parsed differently -- where a capability list would |
| 151 | // only catch a name that was missing. |
| 152 | // |
| 153 | // `directives` rides along because a hash cannot say WHAT is different. It is read |
| 154 | // out of this file's own `case` labels rather than written down beside them, so it |
| 155 | // cannot drift from what the switch actually answers to. |
| 156 | const SELF = fs.readFileSync(fileURLToPath(import.meta.url), 'utf8'); |
| 157 | const SHA = crypto.createHash('sha256').update(SELF).digest('hex'); |
| 158 | const DIRECTIVES = [...new Set( |
| 159 | [...SELF.matchAll(/^\t\tcase '([a-z0-9]+)':/gm)].map(m => m[1]))].sort(); |
| 160 | |
| 161 | const MODELS = [ |
| 162 | 'mock/fast', |
| 163 | 'mock/thinker', |
| 164 | 'accounts/fireworks/models/glm-5p2', |
| 165 | // Appended, never inserted: dev/verify_diamondmodels.mjs picks "the first model |
| 166 | // in the pulldown that is not the one already chosen", so the order of the three |
| 167 | // above is load-bearing for a file that is not about vision at all. |
| 168 | 'mock/blind', |
| 169 | 'mock/eyes', |
| 170 | ]; |
| 171 | |
| 172 | /// Is this a model that cannot be shown pictures? |
| 173 | /// |
| 174 | /// Substring, like [`model_can_see`]'s deny-list in src/llm.rs, so a test can mint a |
| 175 | /// second blind model — `mock/blind-too` — without touching this file. That is what |
| 176 | /// the no-ping-pong check needs: a Diamond whose IMAGE model is itself blind. |
| 177 | const isBlindModel = (model) => /blind/i.test(String(model || '')); |
| 178 | |
| 179 | /// How many picture parts a request carries, in either wire dialect. |
| 180 | /// |
| 181 | /// OpenAI puts them in the content array as `{"type":"image_url",…}` and the Messages |
| 182 | /// API as `{"type":"image","source":{…}}`; both are counted because the mock does not |
| 183 | /// know which dialect the caller configured. |
| 184 | const imageCount = (messages) => (messages || []).reduce((n, m) => { |
| 185 | const c = m && m.content; |
| 186 | if (!Array.isArray(c)) return n; |
| 187 | return n + c.filter((p) => p && (p.type === 'image_url' || p.type === 'image')).length; |
| 188 | }, 0); |
| 189 | |
| 190 | // Requests are logged for assertion, newest last. A test truncates the file |
| 191 | // first, then reads it back to see what the model saw. |
| 192 | const log = (entry) => { |
| 193 | try { |
| 194 | fs.appendFileSync(LOG, JSON.stringify(entry) + '\n'); |
| 195 | } catch (e) { |
| 196 | console.error('mockllm: could not write log:', e.message); |
| 197 | } |
| 198 | }; |
| 199 | |
| 200 | const cors = (res) => { |
| 201 | res.setHeader('Access-Control-Allow-Origin', '*'); |
| 202 | res.setHeader('Access-Control-Allow-Headers', '*'); |
| 203 | res.setHeader('Access-Control-Allow-Methods', 'GET,POST,OPTIONS'); |
| 204 | }; |
| 205 | |
| 206 | // The last thing the user actually typed, which is where a directive lives. |
| 207 | const lastUser = (messages) => { |
| 208 | for (let i = messages.length - 1; i >= 0; i--) { |
| 209 | if (messages[i].role === 'user') { |
| 210 | const c = messages[i].content; |
| 211 | return typeof c === 'string' ? c |
| 212 | : Array.isArray(c) ? c.map(p => p.text || '').join(' ') |
| 213 | : ''; |
| 214 | } |
| 215 | } |
| 216 | return ''; |
| 217 | }; |
| 218 | |
| 219 | // How many tool results have come back WITHIN the current turn — that is, since |
| 220 | // the last user message. Counting the whole conversation would be wrong now that |
| 221 | // tool calls persist across turns: a later @tool directive would see an earlier |
| 222 | // turn's results and wrongly believe its own round had already happened. |
| 223 | const toolRounds = (messages) => { |
| 224 | let lastUserIdx = -1; |
| 225 | for (let i = messages.length - 1; i >= 0; i--) { |
| 226 | if (messages[i].role === 'user') { lastUserIdx = i; break; } |
| 227 | } |
| 228 | return messages.slice(lastUserIdx + 1).filter(m => m.role === 'tool').length; |
| 229 | }; |
| 230 | |
| 231 | // The path a `@look` named, from ANY user message in the conversation, or ''. |
| 232 | // |
| 233 | // The whole transcript rather than the last message, and see the header for why: a |
| 234 | // resumed worker's last user message is the app's own nudge, not its task. The last |
| 235 | // occurrence wins, so a second `@look` in a later turn replaces the first. |
| 236 | const lookPath = (messages) => { |
| 237 | let found = ''; |
| 238 | for (const m of messages || []) { |
| 239 | if (!m || m.role !== 'user') continue; |
| 240 | const c = typeof m.content === 'string' ? m.content |
| 241 | : Array.isArray(m.content) ? m.content.map(p => (p && p.text) || '').join(' ') |
| 242 | : ''; |
| 243 | const hit = /(^|\s)@look\s+(\S+)/.exec(c); |
| 244 | if (hit) found = hit[2]; |
| 245 | } |
| 246 | return found; |
| 247 | }; |
| 248 | |
| 249 | // Has this session already been handed a tool result? Anywhere in the transcript, |
| 250 | // not just since the last user message, because a look that already happened must not |
| 251 | // happen again inside the same session however many nudges follow it. |
| 252 | const hasLooked = (messages) => (messages || []).some(m => m && m.role === 'tool'); |
| 253 | |
| 254 | const parseDirective = (text) => { |
| 255 | const t = (text || '').trim(); |
| 256 | if (!t.startsWith('@')) return { kind: 'plain', text: t }; |
| 257 | const sp = t.indexOf(' '); |
| 258 | const verb = (sp === -1 ? t : t.slice(0, sp)).slice(1); |
| 259 | const rest = sp === -1 ? '' : t.slice(sp + 1).trim(); |
| 260 | return { kind: verb, rest }; |
| 261 | }; |
| 262 | |
| 263 | // A directive argument that cannot be read is REFUSED, never defaulted. |
| 264 | // |
| 265 | // This was `Number(d.rest)`, and `d.rest` is the whole of the line after the verb. |
| 266 | // So `@slow 9000 A-ANSWER` was `Number('9000 A-ANSWER')` -- NaN -- and the `|| 2000` |
| 267 | // beside it turned that into the two-second default. A check that asked for a nine |
| 268 | // second delay got two, silently, and went green for a reason unconnected to what it |
| 269 | // was testing. `dev/verify_daimonconc.mjs` lost an afternoon to it and left a comment |
| 270 | // warning the next reader off the label rather than fixing the mock. |
| 271 | // |
| 272 | // A trailing label is legitimate and stays legal: it is how a caller makes one prompt |
| 273 | // distinguishable from the next in the mock's log. So the FIRST token is the number, |
| 274 | // and the rest is the caller's business. |
| 275 | // |
| 276 | // What is not legal is an argument that is not a number. That throws, becomes a 400, |
| 277 | // and reddens the run. A harness that fails quietly is worse than one that fails |
| 278 | // loudly, because it launders a broken check into a green result. |
| 279 | class DirectiveError extends Error {} |
| 280 | |
| 281 | const numArg = (rest, dflt, verb) => { |
| 282 | const head = String(rest == null ? '' : rest).trim().split(/\s+/)[0]; |
| 283 | if (head === '') return dflt; |
| 284 | const n = Number(head); |
| 285 | if (!Number.isFinite(n)) { |
| 286 | throw new DirectiveError( |
| 287 | `@${verb} wants a number as its first word, got ${JSON.stringify(head)}`); |
| 288 | } |
| 289 | return n; |
| 290 | }; |
| 291 | |
| 292 | // A fresh tool-call id, unique for the life of this mock. |
| 293 | // |
| 294 | // It used to mint `call_1` on every turn, which meant two separate turns of one |
| 295 | // conversation both carried a call with the SAME id. No real provider does that, |
| 296 | // and a test asserting that a reloaded conversation is well formed cannot tell a |
| 297 | // duplicate the app caused from a duplicate the fixture caused. So the fixture |
| 298 | // stopped causing them. Nothing asserts on the literal value. |
| 299 | let callSeq = 0; |
| 300 | const nextCallId = () => `call_${++callSeq}`; |
| 301 | |
| 302 | // A tool call as the wire format wants it: the arguments are a JSON *string*, |
| 303 | // which is the detail most hand-rolled clients get wrong. |
| 304 | const toolCall = (id, name, args) => ({ |
| 305 | id, |
| 306 | type: 'function', |
| 307 | function: { name, arguments: typeof args === 'string' ? args : JSON.stringify(args) }, |
| 308 | }); |
| 309 | |
| 310 | // Split "<name> <json>" — the JSON may itself contain spaces. |
| 311 | const splitCall = (s) => { |
| 312 | const i = s.indexOf(' '); |
| 313 | if (i === -1) return { name: s.trim(), args: {} }; |
| 314 | const name = s.slice(0, i).trim(); |
| 315 | const raw = s.slice(i + 1).trim(); |
| 316 | try { |
| 317 | return { name, args: JSON.parse(raw) }; |
| 318 | } catch { |
| 319 | return { name, args: {} }; |
| 320 | } |
| 321 | }; |
| 322 | |
| 323 | // Decide the turn: text, or calls, or a failure — from the directive and how |
| 324 | // many tool rounds have already come back. |
| 325 | /// Whether this request is the crystal reducer's. |
| 326 | /// |
| 327 | /// The reducer is a role of its own, with its own system prompt, and since the |
| 328 | /// crystal became `crystal.json` it must emit ONE WHOLE JSON OBJECT and nothing |
| 329 | /// else -- `crystal_proposal` refuses anything else before the user can accept it, |
| 330 | /// because accepting an unparseable proposal replaces the Diamond's memory with |
| 331 | /// something no reader downstream can open. |
| 332 | /// |
| 333 | /// So a mock that answers every request with prose cannot drive a fold at all. It |
| 334 | /// is recognised here rather than in each verifier because a real reducer is |
| 335 | /// recognisable the same way -- by the role it was given -- and because every test |
| 336 | /// that folds needs the same answer. |
| 337 | const isReducer = (messages) => (messages || []).some((m) => |
| 338 | m && m.role === 'system' && /crystal/i.test(String(m.content || ''))); |
| 339 | |
| 340 | /// A crystal carrying `words`, as the core schema wants it. |
| 341 | /// |
| 342 | /// The delta's own text goes in, because what the fold verifiers assert is that |
| 343 | /// the words a user typed reached the crystal -- a fixed reply would pass the |
| 344 | /// parse gate and prove nothing. |
| 345 | const crystalReply = (words) => JSON.stringify({ |
| 346 | title: 'Mock crystal', |
| 347 | summary: String(words || '').slice(0, 400), |
| 348 | }, null, 2); |
| 349 | |
| 350 | /// Whether this request is a triage of somebody's notes into proposals. |
| 351 | /// |
| 352 | /// The Social panel's drafting (`www/js/triage.js`) has a fixed answer shape -- |
| 353 | /// one JSON object holding `drafts` and `left` -- which the panel then parses, |
| 354 | /// so a mock answering it with prose could not drive the feature at all. Matched |
| 355 | /// on the system prompt's own opening sentence, which is the role. |
| 356 | const isTriage = (messages) => (messages || []).some((m) => |
| 357 | m && m.role === 'system' && /triaging one person's notes/i.test(String(m.content || ''))); |
| 358 | |
| 359 | /// A plan, as `www/js/triage.js` parses one. |
| 360 | /// |
| 361 | /// `MOCK_TRIAGE_PLAN` names a file holding the plan to answer with, and a test |
| 362 | /// that wants to assert on a REAL model's clustering points it at one that a |
| 363 | /// real model produced. Without it, a draft per note id found in the brief -- |
| 364 | /// which asserts nothing about clustering and everything about the notes having |
| 365 | /// reached the model at all, the way `crystalReply` echoes the delta's words |
| 366 | /// rather than answering a fixed string. |
| 367 | const triageReply = (brief) => { |
| 368 | const named = process.env.MOCK_TRIAGE_PLAN; |
| 369 | if (named) { |
| 370 | try { return fs.readFileSync(named, 'utf8'); } |
| 371 | catch { /* fall through to the derived plan, which says so in its titles */ } |
| 372 | } |
| 373 | const ids = [...String(brief || '').matchAll(/^id (\S+)$/gm)].map(m => m[1]); |
| 374 | return JSON.stringify({ |
| 375 | drafts: ids.map((id) => ({ |
| 376 | kind: 'new', |
| 377 | title: `Mock draft from ${id}`, |
| 378 | body: 'Derived by dev/mockllm.mjs from the brief it was given.', |
| 379 | from: [id], |
| 380 | why: 'One draft per note, because no plan file was named.', |
| 381 | })), |
| 382 | left: [], |
| 383 | }, null, 2); |
| 384 | }; |
| 385 | |
| 386 | const plan = (messages) => { |
| 387 | const d = parseDirective(lastUser(messages)); |
| 388 | const rounds = toolRounds(messages); |
| 389 | |
| 390 | // A worker that was told to look keeps trying until it has looked. Before the |
| 391 | // switch, so a RESUMED session -- whose last user message is the app's nudge and |
| 392 | // whose task is buried two messages up -- still asks for the picture. Reached only |
| 393 | // by a conversation that actually carries `@look`: every other transcript takes the |
| 394 | // path it took yesterday, byte for byte. |
| 395 | // |
| 396 | // AND BEFORE THE REDUCER, which is not where it was first put. `isReducer` matches |
| 397 | // any conversation whose SYSTEM prompt says "crystal", and a worker's does: its role |
| 398 | // prompt is composed with the Diamond's crystal. So every worker request whose last |
| 399 | // user message is not a directive -- the tool-result round, and the app's own nudge |
| 400 | // on a resumed leg -- was being answered with a crystal proposal. Harmless while the |
| 401 | // re-route does not exist, and fatal the moment it does: the second leg's first |
| 402 | // request would have been answered with JSON instead of a `file_read`, no picture |
| 403 | // would ever have reached the sighted model, and dev/verify_vision.mjs's first check |
| 404 | // would have failed against a working fix. A worker is not a reducer, and a |
| 405 | // conversation carrying `@look` is never a fold. |
| 406 | if (d.kind === 'look' || d.kind === 'plain') { |
| 407 | const look = lookPath(messages); |
| 408 | if (look) { |
| 409 | if (!hasLooked(messages)) { |
| 410 | return { calls: [toolCall(nextCallId(), 'file_read', { path: look, as: 'image' })] }; |
| 411 | } |
| 412 | // Names the file, so a resumed session's seeded assistant message is |
| 413 | // recognisable as THIS worker's own earlier words and not a generic reply. |
| 414 | return { text: `Looked at ${look}.` }; |
| 415 | } |
| 416 | } |
| 417 | |
| 418 | // Before the directives: a reducer answered with prose is a fold that cannot |
| 419 | // land. `@text` and the rest still win, so a test that wants to exercise a |
| 420 | // MALFORMED proposal -- and one does -- can still ask for one. |
| 421 | if (isReducer(messages) && d.kind === 'plain') { |
| 422 | return { text: crystalReply(d.text || d.rest || '') }; |
| 423 | } |
| 424 | |
| 425 | // And a triage answered with prose is a plan the Social panel cannot read. |
| 426 | // Recognised by ROLE, exactly as the reducer above is and for the same |
| 427 | // reason: a real triage is recognisable the same way, and every test that |
| 428 | // drafts from a list of notes needs the same answer. `@text` still wins, so |
| 429 | // a test that wants an unreadable plan -- and one does -- can ask for one. |
| 430 | if (isTriage(messages) && d.kind === 'plain') { |
| 431 | return { text: triageReply(lastUser(messages)) }; |
| 432 | } |
| 433 | |
| 434 | switch (d.kind) { |
| 435 | case 'text': |
| 436 | return { text: d.rest || 'Right.' }; |
| 437 | |
| 438 | case 'long': { |
| 439 | const n = Math.max(1, numArg(d.rest, 40, 'long')); |
| 440 | return { text: Array.from({ length: n }, (_, i) => `chunk-${i + 1}`).join(' ') , slowChunks: true }; |
| 441 | } |
| 442 | |
| 443 | case 'usage': { |
| 444 | const [i, o, cost, cached] = d.rest.split(/\s+/).map(Number); |
| 445 | const usage = { prompt_tokens: i || 100, completion_tokens: o || 50 }; |
| 446 | // Only when asked for. A `cost` of zero means "nobody said", and an |
| 447 | // unconditional `cost: 0` would make every @usage turn claim the |
| 448 | // provider had reported the call as free. |
| 449 | if (isFinite(cost) && cost > 0) usage.cost = cost; |
| 450 | // Where a router puts it: nested, not alongside the token counts. |
| 451 | if (isFinite(cached) && cached > 0) usage.prompt_tokens_details = { cached_tokens: cached }; |
| 452 | return { text: 'Counted.', usage }; |
| 453 | } |
| 454 | |
| 455 | case 'err': |
| 456 | return { httpError: numArg(d.rest, 500, 'err') }; |
| 457 | |
| 458 | // A connection that dies mid-answer. `n` words arrive first so the partial |
| 459 | // reply is a real one -- a drop before the first token is a different and |
| 460 | // much easier case, and the app retries THAT one on its own. |
| 461 | case 'drop': { |
| 462 | const n = Math.max(1, numArg(d.rest, 3, 'drop')); |
| 463 | return { text: Array.from({ length: n }, (_, i) => `word-${i + 1}`).join(' '), |
| 464 | dropAfter: n }; |
| 465 | } |
| 466 | |
| 467 | case 'slow': |
| 468 | return { text: 'Eventually.', delayMs: numArg(d.rest, 2000, 'slow') }; |
| 469 | |
| 470 | case 'tool': { |
| 471 | if (rounds > 0) return { text: 'Tool done.' }; |
| 472 | const { name, args } = splitCall(d.rest); |
| 473 | return { calls: [toolCall(nextCallId(), name, args)] }; |
| 474 | } |
| 475 | |
| 476 | case 'tools': { |
| 477 | if (rounds > 0) return { text: 'Tools done.' }; |
| 478 | const calls = d.rest.split(';;').map((part, i) => { |
| 479 | const { name, args } = splitCall(part.trim()); |
| 480 | return toolCall(nextCallId(), name, args); |
| 481 | }); |
| 482 | return { calls }; |
| 483 | } |
| 484 | |
| 485 | // ── The model's own working ────────────────────────────────────── |
| 486 | // |
| 487 | // `reasoning` is streamed BEFORE any content, which is the order every |
| 488 | // reasoning provider sends it in and the order that makes a round look like |
| 489 | // a hang: the wire is busy for the whole of it and the page has nothing. |
| 490 | case 'reason': |
| 491 | case 'reasonc': |
| 492 | case 'reasonslow': { |
| 493 | const [think, said] = d.rest.split(';;'); |
| 494 | return { |
| 495 | think: (think || 'Let me work this out.').trim(), |
| 496 | thinkKey: d.kind === 'reasonc' ? 'reasoning_content' : 'reasoning', |
| 497 | text: (said || 'Done thinking.').trim(), |
| 498 | // A REAL ROUND SPENDS MINUTES HERE. A check that wants to look at the |
| 499 | // page WHILE the model is thinking cannot do it at five milliseconds a |
| 500 | // word, so the slow form exists to be looked at. |
| 501 | slowChunks: d.kind === 'reasonslow', |
| 502 | }; |
| 503 | } |
| 504 | |
| 505 | case 'reasontool': { |
| 506 | if (rounds > 0) return { text: 'Reasoned and done.' }; |
| 507 | const [think, call] = d.rest.split(';;'); |
| 508 | const { name, args } = splitCall((call || 'file_list {"path":"."}').trim()); |
| 509 | return { |
| 510 | think: (think || 'I should look first.').trim(), |
| 511 | thinkKey: 'reasoning', |
| 512 | text: '', |
| 513 | calls: [toolCall(nextCallId(), name, args)], |
| 514 | }; |
| 515 | } |
| 516 | |
| 517 | // Prose, then a call, in ONE message. See the directive list. |
| 518 | case 'narrate': { |
| 519 | if (rounds > 0) return { text: 'Narration done.' }; |
| 520 | const [say, call] = d.rest.split(';;'); |
| 521 | const { name, args } = splitCall((call || 'file_list {"path":"."}').trim()); |
| 522 | return { text: (say || '').trim(), calls: [toolCall(nextCallId(), name, args)] }; |
| 523 | } |
| 524 | |
| 525 | case 'chain': { |
| 526 | // Two rounds of one call each, then a text reply — the shape a real |
| 527 | // agentic turn takes, and the one the UI has to keep up with. |
| 528 | if (rounds === 0) { |
| 529 | const { name, args } = splitCall(d.rest); |
| 530 | return { calls: [toolCall(nextCallId(), name, args)] }; |
| 531 | } |
| 532 | if (rounds === 1) { |
| 533 | return { calls: [toolCall(nextCallId(), 'file_list', { path: '.' })] }; |
| 534 | } |
| 535 | return { text: 'Chain done.' }; |
| 536 | } |
| 537 | |
| 538 | case 'toolslow': { |
| 539 | // One tool call, then a slow stream — so a running tile has booked |
| 540 | // usage from round one (the meter) while still streaming round two. |
| 541 | // Exercises live per-tile cost on a worker that is still running. |
| 542 | if (rounds === 0) return { calls: [toolCall(nextCallId(), 'file_list', { path: '.' })] }; |
| 543 | return { text: Array.from({ length: 60 }, (_, i) => `chunk-${i + 1}`).join(' '), slowChunks: true }; |
| 544 | } |
| 545 | |
| 546 | default: |
| 547 | if (rounds > 0) return { text: 'Done.' }; |
| 548 | return { text: `Mock reply to: ${d.text || d.rest || '(empty)'}` }; |
| 549 | } |
| 550 | }; |
| 551 | |
| 552 | const sleep = (ms) => new Promise(r => setTimeout(r, ms)); |
| 553 | |
| 554 | const sendJson = (res, obj, code = 200) => { |
| 555 | const body = JSON.stringify(obj); |
| 556 | cors(res); |
| 557 | res.writeHead(code, { 'content-type': 'application/json', 'content-length': Buffer.byteLength(body) }); |
| 558 | res.end(body); |
| 559 | }; |
| 560 | |
| 561 | const completion = (model, { text, calls, usage }) => ({ |
| 562 | id: 'chatcmpl-mock', |
| 563 | object: 'chat.completion', |
| 564 | created: 1700000000, |
| 565 | model, |
| 566 | choices: [{ |
| 567 | index: 0, |
| 568 | // BOTH, when a turn has both. `content: null` beside `tool_calls` was the only |
| 569 | // shape this mock could make, and a real provider sends prose alongside a call |
| 570 | // all the time -- which is how the narration fault stayed invisible to every |
| 571 | // fixture here. `text` stays null when there is none, so nothing else moves. |
| 572 | message: calls |
| 573 | ? { role: 'assistant', content: text || null, tool_calls: calls } |
| 574 | : { role: 'assistant', content: text }, |
| 575 | finish_reason: calls ? 'tool_calls' : 'stop', |
| 576 | }], |
| 577 | usage: usage || { prompt_tokens: 42, completion_tokens: 17, total_tokens: 59 }, |
| 578 | }); |
| 579 | |
| 580 | // Stream the same turn as SSE deltas. Tool calls stream as fragments of their |
| 581 | // argument JSON, because that is how the providers do it and it is where an |
| 582 | // accumulator breaks. |
| 583 | const stream = async (res, model, p) => { |
| 584 | cors(res); |
| 585 | res.writeHead(200, { |
| 586 | 'content-type': 'text/event-stream', |
| 587 | 'cache-control': 'no-cache', |
| 588 | 'connection': 'keep-alive', |
| 589 | }); |
| 590 | const send = (o) => res.write(`data: ${JSON.stringify(o)}\n\n`); |
| 591 | const frame = (delta, finish = null) => ({ |
| 592 | id: 'chatcmpl-mock', object: 'chat.completion.chunk', created: 1700000000, model, |
| 593 | choices: [{ index: 0, delta, finish_reason: finish }], |
| 594 | }); |
| 595 | |
| 596 | send(frame({ role: 'assistant', content: '' })); |
| 597 | |
| 598 | // The working first, word by word, with `content` empty throughout -- which is |
| 599 | // exactly what a real provider sends and exactly what used to reach the page as |
| 600 | // nothing at all. `reasoning_details` carries the same words a second time, as |
| 601 | // OpenRouter's does, so a client reading both is caught here rather than in front |
| 602 | // of a user. |
| 603 | if (p.think) { |
| 604 | const key = p.thinkKey || 'reasoning'; |
| 605 | let at = 0; |
| 606 | for (const w of p.think.split(' ').filter(Boolean)) { |
| 607 | if (res.writableEnded || res.destroyed) return; |
| 608 | const piece = w + ' '; |
| 609 | const delta = { content: '', role: 'assistant' }; |
| 610 | delta[key] = piece; |
| 611 | if (key === 'reasoning') { |
| 612 | delta.reasoning_details = [ |
| 613 | { type: 'reasoning.text', text: piece, format: 'unknown', index: at++ }]; |
| 614 | } |
| 615 | send(frame(delta)); |
| 616 | await sleep(p.slowChunks ? 120 : 5); |
| 617 | } |
| 618 | // And the null the providers send once the working is over. |
| 619 | send(frame({ content: '', role: 'assistant', [key]: null })); |
| 620 | } |
| 621 | |
| 622 | if (p.calls) { |
| 623 | // The preamble, word by word, BEFORE the call frames -- which is the order a |
| 624 | // provider sends it in and the order the app has to cope with. |
| 625 | for (const w of (p.text || '').split(' ').filter(Boolean)) { |
| 626 | if (res.writableEnded || res.destroyed) return; |
| 627 | send(frame({ content: w + ' ' })); |
| 628 | await sleep(5); |
| 629 | } |
| 630 | p.calls.forEach((c, i) => { |
| 631 | send(frame({ tool_calls: [{ index: i, id: c.id, type: 'function', |
| 632 | function: { name: c.function.name, arguments: '' } }] })); |
| 633 | }); |
| 634 | // Dribble the arguments out in two pieces, so an accumulator that only |
| 635 | // keeps the last fragment is caught. |
| 636 | for (const [i, c] of p.calls.entries()) { |
| 637 | const a = c.function.arguments; |
| 638 | const cut = Math.max(1, Math.floor(a.length / 2)); |
| 639 | send(frame({ tool_calls: [{ index: i, function: { arguments: a.slice(0, cut) } }] })); |
| 640 | await sleep(10); |
| 641 | send(frame({ tool_calls: [{ index: i, function: { arguments: a.slice(cut) } }] })); |
| 642 | } |
| 643 | send(frame({}, 'tool_calls')); |
| 644 | } else { |
| 645 | const words = (p.text || '').split(' '); |
| 646 | let sent = 0; |
| 647 | for (const w of words) { |
| 648 | if (res.writableEnded || res.destroyed) return; // the client aborted |
| 649 | send(frame({ content: w + ' ' })); |
| 650 | sent++; |
| 651 | // THE SOCKET IS DESTROYED, not ended: an `end()` is a well-formed stream |
| 652 | // that stops, which the client reads as a finished turn. A destroy is the |
| 653 | // road going away mid-sentence, with no `[DONE]` and no finish reason, |
| 654 | // which is what a laptop lid or a lost access point actually does. |
| 655 | if (p.dropAfter && sent >= p.dropAfter) { |
| 656 | res.destroy(); |
| 657 | return; |
| 658 | } |
| 659 | await sleep(p.slowChunks ? 120 : 5); |
| 660 | } |
| 661 | send(frame({}, 'stop')); |
| 662 | } |
| 663 | |
| 664 | send({ id: 'chatcmpl-mock', object: 'chat.completion.chunk', model, choices: [], |
| 665 | usage: p.usage || { prompt_tokens: 42, completion_tokens: 17, total_tokens: 59 } }); |
| 666 | res.write('data: [DONE]\n\n'); |
| 667 | res.end(); |
| 668 | }; |
| 669 | |
| 670 | const server = http.createServer((req, res) => { |
| 671 | if (req.method === 'OPTIONS') { cors(res); res.writeHead(204); return res.end(); } |
| 672 | |
| 673 | // WHICH LOG THIS MOCK WRITES TO, so a caller can identify the process holding |
| 674 | // the port instead of trusting that it is the one it started. |
| 675 | // |
| 676 | // The log path is the whole of a world's mock identity: a verifier asserts on |
| 677 | // what the model was sent by READING THE FILE, so a mock answering this port |
| 678 | // while appending somewhere else makes every such assertion read an empty |
| 679 | // file. That is not hypothetical. On 2026-08-17 a gate found :9108 already |
| 680 | // held by an earlier gate's mock, left it alone, and set DAIMOND_MOCK_LOG to |
| 681 | // its own worktree's copy -- which stayed 0 bytes for two hours. Eighteen or |
| 682 | // more verifiers then reported "the provider was reached: no", "0 requests", |
| 683 | // "nothing in the mock log", about turns that had in fact been answered |
| 684 | // perfectly well by a mock writing to a path nobody was reading. |
| 685 | if (req.method === 'GET' && req.url.startsWith('/__world')) { |
| 686 | // `sha` and `directives` say which REVISION is answering; see the note beside |
| 687 | // them. `log` says which world. A caller needs both and they are different |
| 688 | // questions. |
| 689 | return sendJson(res, { log: LOG, port: PORT, pid: process.pid, |
| 690 | sha: SHA, directives: DIRECTIVES }); |
| 691 | } |
| 692 | |
| 693 | if (req.method === 'GET' && req.url.startsWith('/v1/models')) { |
| 694 | // A test can drive the rejected-key path with the sentinel key "reject". |
| 695 | const auth = req.headers.authorization || ''; |
| 696 | if (/\breject\b/.test(auth)) { |
| 697 | return sendJson(res, { error: { message: 'mock: invalid api key' } }, 401); |
| 698 | } |
| 699 | return sendJson(res, { object: 'list', data: MODELS.map(id => ({ id, object: 'model' })) }); |
| 700 | } |
| 701 | |
| 702 | if (req.method !== 'POST') { cors(res); res.writeHead(404); return res.end(); } |
| 703 | |
| 704 | let body = ''; |
| 705 | req.on('data', c => { body += c; }); |
| 706 | req.on('end', async () => { |
| 707 | let payload; |
| 708 | try { |
| 709 | payload = JSON.parse(body); |
| 710 | } catch { |
| 711 | return sendJson(res, { error: { message: 'mock: body was not JSON' } }, 400); |
| 712 | } |
| 713 | |
| 714 | const messages = payload.messages || []; |
| 715 | // A blind model refuses a picture; the count is needed either way, because a |
| 716 | // verifier reads it back to say which leg carried one. |
| 717 | const images = imageCount(messages); |
| 718 | const refused = images > 0 && isBlindModel(payload.model); |
| 719 | log({ |
| 720 | at: new Date().toISOString(), |
| 721 | model: payload.model, |
| 722 | stream: !!payload.stream, |
| 723 | tools: (payload.tools || []).map(t => t.function?.name).filter(Boolean), |
| 724 | auth: !!(req.headers.authorization), |
| 725 | images, // picture parts in this request, either dialect |
| 726 | ...(refused ? { refusedImages: true } : {}), |
| 727 | messages, // the whole transcript, so a test can assert what was sent |
| 728 | }); |
| 729 | |
| 730 | // THE REFUSAL A REAL PROVIDER SENDS, and it is logged before it is sent: the |
| 731 | // request that was turned away is the evidence that the app was on the wrong |
| 732 | // model, so a mock that refused before logging would hide the very thing. |
| 733 | // |
| 734 | // 400 with the provider's own words about `image_url`. `stream_turn` does not |
| 735 | // read them -- it retries on any refusal of a turn that carried pictures -- but |
| 736 | // `vision_error` (src/llm.rs) does, and it only names the model when the |
| 737 | // provider's sentence is about images. So the wording is part of the fixture. |
| 738 | if (refused) { |
| 739 | return sendJson(res, { error: { |
| 740 | message: 'this model does not support image_url content', |
| 741 | type: 'invalid_request_error', |
| 742 | param: 'messages', |
| 743 | code: 'unsupported_content', |
| 744 | } }, 400); |
| 745 | } |
| 746 | |
| 747 | // A directive the mock cannot read is a fault in the CHECK, not in the app, |
| 748 | // and it is reported where a lane will actually see it: on stderr, which |
| 749 | // world.sh keeps in the world's `mock.out`, and as a 400 that fails the turn. |
| 750 | // The old behaviour -- substitute the default and carry on -- is what made |
| 751 | // `@slow 9000 A-ANSWER` a two-second delay for weeks. |
| 752 | let p; |
| 753 | try { |
| 754 | p = plan(messages); |
| 755 | } catch (e) { |
| 756 | if (e instanceof DirectiveError) { |
| 757 | console.error(`mockllm: REFUSED a directive it could not read -- ${e.message}`); |
| 758 | return sendJson(res, { error: { |
| 759 | message: `mockllm: ${e.message}`, |
| 760 | type: 'invalid_request_error', |
| 761 | } }, 400); |
| 762 | } |
| 763 | throw e; |
| 764 | } |
| 765 | |
| 766 | if (p.httpError) { |
| 767 | return sendJson(res, { error: { message: 'mock: as requested' } }, p.httpError); |
| 768 | } |
| 769 | if (p.delayMs) await sleep(p.delayMs); |
| 770 | |
| 771 | if (payload.stream) return stream(res, payload.model || 'mock/fast', p); |
| 772 | return sendJson(res, completion(payload.model || 'mock/fast', p)); |
| 773 | }); |
| 774 | }); |
| 775 | |
| 776 | server.listen(PORT, '127.0.0.1', () => { |
| 777 | console.log(`mockllm: http://127.0.0.1:${PORT}/v1/chat/completions (log: ${LOG})`); |
| 778 | }); |