oxedyne/daimond/dev/verify_replylen.mjs
22.4 KiB, 1 run
created by r2519314175:651, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // verify_replylen.mjs — the reply-length cap, and the context meter. |
| 2 | // |
| 3 | // Two defects, one file, and both of them are about a number the app believed |
| 4 | // that was not true. |
| 5 | // |
| 6 | // ── The 4096-token output cap ──────────────────────────────────────────── |
| 7 | // Every request carried `max_tokens: 4096`, whatever the model. 4096 OUTPUT |
| 8 | // tokens is about 250 lines of code, so a `file_write` of a 400-line module ran |
| 9 | // out of room part way — and because a tool call's arguments ARE a JSON string, |
| 10 | // what arrived was not a truncated file but a MALFORMED TOOL CALL. Nothing was |
| 11 | // written, and the model was told only that its JSON was bad. |
| 12 | // |
| 13 | // That is not a claim a unit test on a constant can settle, so it is not tested |
| 14 | // that way. `dev/mockcap.mjs` is a provider that HONOURS `max_tokens` and cuts |
| 15 | // the reply when it does not fit, exactly as a real one does, and logs the |
| 16 | // `max_tokens` of every request it is sent. The failure is demonstrated, then |
| 17 | // the fix is demonstrated, against the same 400-line write. |
| 18 | // |
| 19 | // ── The context meter reading ~12x high ────────────────────────────────── |
| 20 | // The tile meter drew `lastPrompt / contextWindow`, and `lastPrompt` was set to |
| 21 | // the turn's CUMULATIVE prompt tokens — the sum of every round. An agentic turn |
| 22 | // sends the whole conversation once per round, so a five-round turn read five |
| 23 | // times high. The per-round figure was tracked in Rust all along |
| 24 | // (`session.last_prompt_tokens`); there was no getter to read it back. |
| 25 | // |
| 26 | // The mock reports a KNOWN, DIFFERENT prompt figure per round — 5000, 10000, |
| 27 | // 15000, … — so "the last one" and "the sum of them" cannot be confused, and |
| 28 | // the number the meter draws is compared against what the provider actually |
| 29 | // sent, not against another part of the app. |
| 30 | // |
| 31 | // Needs `dev/serve.mjs` (`DAIMOND_PORT`, default 8777) and `dev/mockcap.mjs` |
| 32 | // (`DAIMOND_CAP_PORT`, default 9250 + the world number), the latter started here |
| 33 | // if it is not up -- and checked to BE mockcap, see the port note below. |
| 34 | // |
| 35 | // Every write it asks for goes into the chat's own scratch folder. A workspace- |
| 36 | // ROOT path is refused by the chat fence as an ordinary tool result, which reads |
| 37 | // as neither a write nor a cut: see `freshChat`. |
| 38 | |
| 39 | import fs from 'node:fs'; |
| 40 | import path from 'node:path'; |
| 41 | import http from 'node:http'; |
| 42 | import { spawn } from 'node:child_process'; |
| 43 | import { fileURLToPath } from 'node:url'; |
| 44 | import { open, signInAs, connectMock, chat, newChat, shot } from './harness.mjs'; |
| 45 | |
| 46 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 47 | // The cap mock is a world fixture like the LLM mock, so its port and log follow |
| 48 | // the world's. `dev/world.sh` numbers a world by its offset from port 8777. |
| 49 | const WORLD = Number(process.env.DAIMOND_PORT || 8777) - 8777; |
| 50 | // 9250, NOT 9098. `9098 + WORLD` is `9099 + (WORLD - 1)`, which is the MOCK LLM |
| 51 | // PORT OF THE WORLD NEXT DOOR: in world 18 the cap mock's port was world 17's |
| 52 | // mockllm. The guard below found something answering /v1/models there, so no |
| 53 | // mockcap was started, and every request in this file went to another agent's |
| 54 | // mock -- which ignores `max_tokens` and does not log, so all eleven checks that |
| 55 | // read `lastAsk()` reported `max_tokens=null` and the two write checks measured |
| 56 | // mockllm's plain reply. Measured 2026-08-14 in world 18, with world 17 live. |
| 57 | // 9250 + WORLD is clear of every other fixture in this tree (compact 9188+N, |
| 58 | // pickers 9160+2N, applications 9400+N). |
| 59 | const CAP_PORT = Number(process.env.DAIMOND_CAP_PORT || 9250 + WORLD); |
| 60 | const CAP_LOG = process.env.DAIMOND_CAP_LOG |
| 61 | || path.join(HERE, WORLD ? `mockcap-${WORLD}.log` : 'mockcap.log'); |
| 62 | const CAP_URL = `http://127.0.0.1:${CAP_PORT}/v1/chat/completions`; |
| 63 | const MODEL = 'cap/plain'; |
| 64 | |
| 65 | /// What the mock reports as round 0's prompt; round k reports (k+1) times it. |
| 66 | const ROUND_PROMPT = 5000; |
| 67 | /// The default the app should now send when nothing published a smaller ceiling. |
| 68 | const AUTO_MAX = 32768; |
| 69 | |
| 70 | const ok = [], bad = []; |
| 71 | const check = (name, pass, detail) => { |
| 72 | (pass ? ok : bad).push(name + (detail ? ' — ' + detail : '')); |
| 73 | console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : '')); |
| 74 | }; |
| 75 | |
| 76 | // ── The mock ──────────────────────────────────────────────────────────── |
| 77 | |
| 78 | const up = (url) => new Promise((res) => { |
| 79 | const r = http.get(url, () => res(true)); |
| 80 | r.on('error', () => res(false)); |
| 81 | r.setTimeout(700, () => { r.destroy(); res(false); }); |
| 82 | }); |
| 83 | |
| 84 | const CAP_MODELS = `http://127.0.0.1:${CAP_PORT}/v1/models`; |
| 85 | let spawned = null; |
| 86 | if (!(await up(CAP_MODELS))) { |
| 87 | spawned = spawn('node', [path.join(HERE, 'mockcap.mjs'), String(CAP_PORT)], |
| 88 | { stdio: 'ignore', detached: false, env: { ...process.env, DAIMOND_CAP_LOG: CAP_LOG } }); |
| 89 | for (let i = 0; i < 20 && !(await up(CAP_MODELS)); i++) { |
| 90 | await new Promise(r => setTimeout(r, 200)); |
| 91 | } |
| 92 | } |
| 93 | |
| 94 | // WHAT ANSWERS THERE HAS TO BE MOCKCAP. "Something is listening" was the whole |
| 95 | // test, and any OpenAI-compatible mock answers /v1/models -- which is how this |
| 96 | // file spent a run driven by another world's mockllm (see CAP_PORT above). |
| 97 | // `cap/plain` is mockcap's own model id, so asking for it by name is the cheapest |
| 98 | // question that only mockcap can answer. |
| 99 | const capModels = await new Promise((res) => { |
| 100 | const r = http.get(CAP_MODELS, (m) => { |
| 101 | let raw = ''; |
| 102 | m.on('data', (c) => { raw += c; }); |
| 103 | m.on('end', () => { try { res(JSON.parse(raw)); } catch { res(null); } }); |
| 104 | }); |
| 105 | r.on('error', () => res(null)); |
| 106 | r.setTimeout(1500, () => { r.destroy(); res(null); }); |
| 107 | }); |
| 108 | if (!(capModels && (capModels.data || []).some((m) => m.id === MODEL))) { |
| 109 | console.log(`\nCANNOT START: :${CAP_PORT} does not answer as dev/mockcap.mjs.`); |
| 110 | console.log(` It offers ${JSON.stringify((capModels && capModels.data || []).map((m) => m.id))},`); |
| 111 | console.log(` and this whole file measures what a provider that HONOURS max_tokens does.`); |
| 112 | console.log(' Something else is on the port — set DAIMOND_CAP_PORT, or free it.'); |
| 113 | if (spawned) spawned.kill(); |
| 114 | process.exit(2); |
| 115 | } |
| 116 | |
| 117 | const clearLog = () => { try { fs.writeFileSync(CAP_LOG, ''); } catch {} }; |
| 118 | const capLog = () => { |
| 119 | if (!fs.existsSync(CAP_LOG)) return []; |
| 120 | return fs.readFileSync(CAP_LOG, 'utf8').split('\n').filter(Boolean) |
| 121 | .map(l => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean); |
| 122 | }; |
| 123 | /// The `max_tokens` of the last request the mock was sent. |
| 124 | const lastAsk = () => { const l = capLog(); return l.length ? l[l.length - 1].max_tokens : null; }; |
| 125 | |
| 126 | // ── The app ───────────────────────────────────────────────────────────── |
| 127 | |
| 128 | const s = await open({ name: 'replylen', connect: false }); |
| 129 | const p = s.page; |
| 130 | await connectMock(s, { baseUrl: CAP_URL, model: MODEL }); |
| 131 | await p.waitForTimeout(600); |
| 132 | |
| 133 | /// Force a reply-length setting through the real control, or through the store |
| 134 | /// when the control has not been built (the panel may be closed). |
| 135 | async function setReplyLength(n) { |
| 136 | await p.evaluate(async (n) => { |
| 137 | const sel = document.getElementById('cfg-max-tokens'); |
| 138 | if (sel) { |
| 139 | // The value may not be on the offered ladder; add it so `change` takes. |
| 140 | if (![...sel.options].some(o => o.value === String(n))) { |
| 141 | const o = document.createElement('option'); |
| 142 | o.value = String(n); o.textContent = String(n); |
| 143 | sel.appendChild(o); |
| 144 | } |
| 145 | sel.value = String(n); |
| 146 | sel.dispatchEvent(new Event('change', { bubbles: true })); |
| 147 | return; |
| 148 | } |
| 149 | const raw = JSON.parse(localStorage.getItem('daimond-byok') || '{}'); |
| 150 | raw.maxOut = n; |
| 151 | localStorage.setItem('daimond-byok', JSON.stringify(raw)); |
| 152 | }, n); |
| 153 | await p.waitForTimeout(200); |
| 154 | // A DaimondApp freezes its max_tokens at construction, so a chat built under |
| 155 | // the old setting must be rebuilt. Reloading is what a user would see anyway. |
| 156 | await p.reload({ waitUntil: 'domcontentloaded' }); |
| 157 | await signInAs(s, 'replylen'); |
| 158 | await p.waitForTimeout(900); |
| 159 | } |
| 160 | |
| 161 | /// The stored chats, as the app persists them. |
| 162 | /// |
| 163 | /// IndexedDB, not localStorage: transcripts moved there when a day of tool results |
| 164 | /// stopped fitting in the origin's five megabytes. Read from outside the app, so |
| 165 | /// what is asserted is what is on disk rather than what the page believes. |
| 166 | const storedChats = () => p.evaluate(() => new Promise((res) => { |
| 167 | const req = indexedDB.open('daimond-chats', 1); |
| 168 | req.onsuccess = () => { |
| 169 | const db = req.result; |
| 170 | let t; |
| 171 | try { t = db.transaction('chats', 'readonly'); } catch (e) { res([]); return; } |
| 172 | const all = t.objectStore('chats').getAll(); |
| 173 | all.onsuccess = () => res(all.result || []); |
| 174 | all.onerror = () => res([]); |
| 175 | }; |
| 176 | req.onerror = () => res([]); |
| 177 | })); |
| 178 | |
| 179 | /// Empty the chat store, so a turn is never metered against a previous one. Both |
| 180 | /// places: the store itself, and the old localStorage key, which would otherwise be |
| 181 | /// migrated straight back in on the next boot. |
| 182 | const clearChats = () => p.evaluate(() => new Promise((res) => { |
| 183 | try { localStorage.removeItem('daimond-chats'); localStorage.removeItem('daimond-chats-legacy'); } catch (e) { /* full */ } |
| 184 | const req = indexedDB.open('daimond-chats', 1); |
| 185 | req.onsuccess = () => { |
| 186 | const db = req.result; |
| 187 | let t; |
| 188 | try { t = db.transaction('chats', 'readwrite'); } catch (e) { res(); return; } |
| 189 | t.objectStore('chats').clear(); |
| 190 | t.oncomplete = () => res(); |
| 191 | t.onerror = () => res(); |
| 192 | }; |
| 193 | req.onerror = () => res(); |
| 194 | })); |
| 195 | |
| 196 | /// Start a fresh chat so a turn is never metered against a previous one, and |
| 197 | /// answer where that chat may write. |
| 198 | /// |
| 199 | /// THE PATH IS NOT OPTIONAL. §2 below asked the model for `@write 400 big.js` -- |
| 200 | /// a workspace-ROOT path -- and since the chat fence landed on 2026-08-12 |
| 201 | /// (5389864) every chat is confined to `chats/<id>/work`: `Tool::guard` |
| 202 | /// (src/tools.rs) refuses anything outside it BEFORE the tool runs, and the |
| 203 | /// refusal comes back as an ORDINARY TOOL RESULT. Nothing throws, the turn |
| 204 | /// finishes, and the transcript holds `Refused: 'big.js' is not in this chat's |
| 205 | /// workspace` -- which contains neither "Wrote N bytes" nor "ran out of room", |
| 206 | /// so 2a passed for the wrong reason and 2b, the only check that says the raised |
| 207 | /// default FIXES a long write, had been measuring an apology. The commit that |
| 208 | /// repaired six other verifiers this way (see dev/harness.mjs) missed this one. |
| 209 | async function freshChat() { |
| 210 | await clearChats(); |
| 211 | await p.reload({ waitUntil: 'domcontentloaded' }); |
| 212 | await signInAs(s, 'replylen'); |
| 213 | await p.waitForTimeout(900); |
| 214 | await newChat(s); |
| 215 | return await scratchDir(); |
| 216 | } |
| 217 | |
| 218 | /// `chats/<id>/work` for the chat in focus -- what `scopeChatTo` hands the |
| 219 | /// engine, asked of the app rather than spelled out here, so a change to the |
| 220 | /// fence's shape moves this with it. |
| 221 | async function scratchDir() { |
| 222 | return await p.evaluate(() => { |
| 223 | const f = window.DaimondAttach && window.DaimondAttach.focus(); |
| 224 | if (!f || f.kind !== 'chat') return ''; |
| 225 | return window.DaimondAttach.chatScratch(f.id) || ''; |
| 226 | }); |
| 227 | } |
| 228 | |
| 229 | console.log('\n── 1. The cap that reaches the provider ──'); |
| 230 | |
| 231 | // ── 1a. The default is no longer 4096 ──────────────────────────────────── |
| 232 | |
| 233 | clearLog(); |
| 234 | await freshChat(); |
| 235 | await chat(s, '@text hello'); |
| 236 | const ask1 = lastAsk(); |
| 237 | check('the request carries the raised default, not 4096', |
| 238 | ask1 === AUTO_MAX, `max_tokens=${ask1}`); |
| 239 | |
| 240 | // ── 1b. A per-model ceiling binds below the default ────────────────────── |
| 241 | // |
| 242 | // claude-opus-4-1 accepts at most 32,000 output tokens, which is BELOW the |
| 243 | // 32,768 default: a request above a model's maximum is an error, not a clamp, |
| 244 | // so the app must ask for the model's figure and not its own. |
| 245 | |
| 246 | await clearChats(); |
| 247 | await connectMock(s, { baseUrl: CAP_URL, model: 'claude-opus-4-1' }); |
| 248 | await p.waitForTimeout(400); |
| 249 | clearLog(); |
| 250 | await freshChat(); |
| 251 | await chat(s, '@text hello'); |
| 252 | const askOpus41 = lastAsk(); |
| 253 | check('a model whose published ceiling is lower gets its own figure', |
| 254 | askOpus41 === 32000, `claude-opus-4-1 → max_tokens=${askOpus41}`); |
| 255 | |
| 256 | // ── 1c. A model with a larger ceiling still gets the default ───────────── |
| 257 | |
| 258 | await connectMock(s, { baseUrl: CAP_URL, model: 'claude-opus-5' }); |
| 259 | await p.waitForTimeout(400); |
| 260 | clearLog(); |
| 261 | await freshChat(); |
| 262 | await chat(s, '@text hello'); |
| 263 | const askOpus5 = lastAsk(); |
| 264 | check('a model with a larger ceiling keeps the default', |
| 265 | askOpus5 === AUTO_MAX, `claude-opus-5 (128k ceiling) → max_tokens=${askOpus5}`); |
| 266 | |
| 267 | await connectMock(s, { baseUrl: CAP_URL, model: MODEL }); |
| 268 | await p.waitForTimeout(400); |
| 269 | |
| 270 | console.log('\n── 2. What the cap does to a 400-line write ──'); |
| 271 | |
| 272 | // ── 2a. Under the OLD cap the write fails as a malformed tool call ─────── |
| 273 | |
| 274 | await setReplyLength(4096); |
| 275 | clearLog(); |
| 276 | const cutDir = await freshChat(); |
| 277 | |
| 278 | // A CAN'T-START CHECK, and not one of the checks. |
| 279 | // |
| 280 | // Everything below this line writes into the chat's scratch folder, and a path |
| 281 | // outside it is refused before the tool runs -- silently, as an ordinary tool |
| 282 | // result. If the app cannot say where that folder is, the fixture cannot be |
| 283 | // seeded, and a run that carried on would be measuring the refusal again. It |
| 284 | // exits 2 rather than failing a check, because nothing here was tested. |
| 285 | if (!/^chats\/[^/]+\/work$/.test(cutDir)) { |
| 286 | console.log(`\nCANNOT START: the chat's scratch folder did not answer — got '${cutDir}'.`); |
| 287 | console.log(' §2 writes a 400-line file, and only `chats/<id>/work` will take it.'); |
| 288 | await s.close(); |
| 289 | if (spawned) spawned.kill(); |
| 290 | process.exit(2); |
| 291 | } |
| 292 | /// The file §2 asks for, inside the fence. `big.js` on its own is refused. |
| 293 | const BIG = `${cutDir}/big.js`; |
| 294 | const wrote = (out, file) => new RegExp('Wrote (\\d+) bytes to ' + file.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).exec(out); |
| 295 | |
| 296 | const cut = await chat(s, `@write 400 ${BIG}`, { timeout: 60000 }); |
| 297 | const askCut = lastAsk(); |
| 298 | check('the forced 4096 setting is what is sent', askCut === 4096, `max_tokens=${askCut}`); |
| 299 | // The refusal that used to satisfy this check by accident is now itself a |
| 300 | // failure: it is neither a write nor a length cut, and it must not be either. |
| 301 | check('the fixture reached the tool at all — it was not refused by the chat fence', |
| 302 | !/is not in this chat's workspace/i.test(cut), |
| 303 | (cut.match(/Refused:[^\n]*/) || ['no refusal'])[0]); |
| 304 | check('under 4096 the write does NOT complete', |
| 305 | !wrote(cut, BIG), |
| 306 | cut.match(/Wrote \d+ bytes[^\n]*/)?.[0] || 'no write recorded'); |
| 307 | check('under 4096 the failure is REPORTED as a length cut, not left as bad JSON', |
| 308 | /ran out of room/i.test(cut), (cut.match(/ran out of room[^\n]*/) || ['(no notice)'])[0]); |
| 309 | await shot(s, 'replylen-cut'); |
| 310 | |
| 311 | // ── 2b. Under the new default the same write completes ─────────────────── |
| 312 | |
| 313 | await setReplyLength(0); // 0 = Automatic |
| 314 | clearLog(); |
| 315 | const wholeDir = await freshChat(); |
| 316 | const BIG2 = `${wholeDir}/big.js`; |
| 317 | const whole = await chat(s, `@write 400 ${BIG2}`, { timeout: 60000 }); |
| 318 | const askWhole = lastAsk(); |
| 319 | const said = (wrote(whole, BIG2) || [])[1]; |
| 320 | check('Automatic sends the raised default', askWhole === AUTO_MAX, `max_tokens=${askWhole}`); |
| 321 | // Said here as well as in 2a, and it is here that it bites: at 4096 the reply is |
| 322 | // cut before any tool call is made, so there is nothing for the fence to refuse. |
| 323 | // This is the turn that DOES reach the tool, and a refusal is what it used to |
| 324 | // come back with. |
| 325 | check('the write reached the tool at all — it was not refused by the chat fence', |
| 326 | !/is not in this chat's workspace/i.test(whole), |
| 327 | (whole.match(/Refused:[^\n]*/) || ['no refusal'])[0]); |
| 328 | check('under the new default the same 400-line write completes', |
| 329 | !!said, said ? `${said} bytes written` : 'no write recorded'); |
| 330 | // The transcript is what the MODEL was told. The file is the thing the user |
| 331 | // keeps, so it is read back through the engine's own door -- not fenced, |
| 332 | // because it is not a chat -- and its size compared with what was claimed. |
| 333 | const onDisk = await p.evaluate(async (f) => { |
| 334 | try { return (await (await import('/pkg/oxedyne_daimond.js')).read_file(f)).length; } |
| 335 | catch (e) { return -1; } |
| 336 | }, BIG2); |
| 337 | check('and the file is really there, the size it said', |
| 338 | said && onDisk === Number(said), `${onDisk} bytes at ${BIG2}`); |
| 339 | check('and nothing is reported as cut', !/ran out of room/i.test(whole)); |
| 340 | await shot(s, 'replylen-whole'); |
| 341 | |
| 342 | // ── 2c. The setting is real: a chosen figure is what is sent ───────────── |
| 343 | |
| 344 | await setReplyLength(8192); |
| 345 | clearLog(); |
| 346 | await freshChat(); |
| 347 | await chat(s, '@text hello'); |
| 348 | const askSet = lastAsk(); |
| 349 | check('a chosen reply length reaches the provider', askSet === 8192, `max_tokens=${askSet}`); |
| 350 | const persisted = await p.evaluate(() => { |
| 351 | try { return JSON.parse(localStorage.getItem('daimond-byok') || '{}').maxOut; } catch { return null; } |
| 352 | }); |
| 353 | check('and it survives a reload', persisted === 8192, `stored maxOut=${persisted}`); |
| 354 | |
| 355 | // The control itself, on screen: a row that says which model it is reading and |
| 356 | // what that model will accept. Read off the DOM and shot, because a setting |
| 357 | // nobody can find is not a setting. |
| 358 | await p.evaluate(() => { document.getElementById('settings-btn')?.click(); }); |
| 359 | await p.waitForTimeout(400); |
| 360 | await p.evaluate(() => { |
| 361 | const row = document.querySelector('.astat-btn#astat-model') || document.getElementById('astat-model'); |
| 362 | if (row) row.click(); |
| 363 | }); |
| 364 | await p.waitForTimeout(500); |
| 365 | const knob = await p.evaluate(() => { |
| 366 | const sel = document.getElementById('cfg-max-tokens'); |
| 367 | if (!sel) return null; |
| 368 | const note = document.getElementById('cfg-max-tokens-note'); |
| 369 | const r = sel.getBoundingClientRect(); |
| 370 | return { |
| 371 | visible: r.width > 0 && r.height > 0, |
| 372 | options: [...sel.options].map(o => o.textContent), |
| 373 | value: sel.value, |
| 374 | note: (note || {}).textContent || '', |
| 375 | }; |
| 376 | }); |
| 377 | check('the reply-length control is on screen in the Models panel', |
| 378 | !!(knob && knob.visible), knob ? `value=${knob.value}` : 'not built'); |
| 379 | check('it offers Automatic and a ladder bounded by the model', |
| 380 | !!(knob && /Automatic/.test(knob.options[0]) && knob.options.length > 2), |
| 381 | knob ? knob.options.join(' | ') : ''); |
| 382 | check('and it names what the model will accept', |
| 383 | !!(knob && knob.note.length > 20), knob ? knob.note : ''); |
| 384 | await shot(s, 'replylen-setting'); |
| 385 | |
| 386 | // ── 2d. A provider that refuses the length is answered, not reported ───── |
| 387 | |
| 388 | await setReplyLength(0); |
| 389 | clearLog(); |
| 390 | await freshChat(); |
| 391 | // Refuses above 20,000: the first ask (32,768) is rejected, the halved one |
| 392 | // (16,384) is not — so the backoff is shown to LAND, not merely to happen. |
| 393 | const refused = await chat(s, '@refuse-cap 20000', { timeout: 60000 }); |
| 394 | const asks = capLog().map(e => e.max_tokens); |
| 395 | // The provider's own words never reach the browser -- `wasm_fetch` in src/llm.rs |
| 396 | // builds the error from the status line and drops the body -- so what the app |
| 397 | // has to work from is a bare 400. It answers by asking for half as much, once, |
| 398 | // and believes the smaller figure only because the smaller ask then succeeds. |
| 399 | check('a refused length is retried at half, not surfaced', |
| 400 | asks.length >= 2 && asks[0] === AUTO_MAX && asks[1] === AUTO_MAX / 2, |
| 401 | `asked ${asks.join(' then ')}`); |
| 402 | check('and the turn then succeeds', |
| 403 | /Fine at this length/.test(refused), (refused.split('\n').pop() || '').slice(0, 60)); |
| 404 | |
| 405 | console.log('\n── 3. The context meter ──'); |
| 406 | |
| 407 | // A five-round tool loop. The mock reports 5000, 10000, 15000, 20000, 25000 |
| 408 | // prompt tokens across the rounds: the LAST is 25000, the SUM is 75000. |
| 409 | // |
| 410 | // On `claude-opus-5`, because a meter needs a DENOMINATOR: a model nobody |
| 411 | // publishes a context window for draws no meter at all, which is the honest |
| 412 | // behaviour and the reason the mock answers to a real model id here. Anthropic |
| 413 | // publishes 1,000,000 tokens for this model, which is what the app's table says |
| 414 | // -- so the denominator is checked against the published figure, not against |
| 415 | // another part of the app. |
| 416 | const METER_MODEL = 'claude-opus-5'; |
| 417 | const METER_CTX = 1000000; |
| 418 | const ROUNDS = 5; |
| 419 | const LAST_ROUND = ROUND_PROMPT * ROUNDS; |
| 420 | const SUM_ROUNDS = ROUND_PROMPT * (ROUNDS * (ROUNDS + 1) / 2); |
| 421 | |
| 422 | await setReplyLength(0); |
| 423 | await connectMock(s, { baseUrl: CAP_URL, model: METER_MODEL }); |
| 424 | await p.waitForTimeout(400); |
| 425 | clearLog(); |
| 426 | await freshChat(); |
| 427 | await chat(s, `@rounds ${ROUNDS}`, { timeout: 60000 }); |
| 428 | await p.waitForTimeout(600); |
| 429 | |
| 430 | const rounds = capLog().length; |
| 431 | check(`the turn really ran ${ROUNDS} rounds`, rounds === ROUNDS, `${rounds} requests`); |
| 432 | |
| 433 | const chatRow = (await storedChats())[0] || {}; |
| 434 | check('the meter reads the LAST round, not the sum of the rounds', |
| 435 | chatRow.lastPrompt === LAST_ROUND, |
| 436 | `lastPrompt=${chatRow.lastPrompt}, last round sent ${LAST_ROUND}, sum is ${SUM_ROUNDS}`); |
| 437 | check('the cumulative counter is still the sum (nothing else was broken to fix it)', |
| 438 | chatRow.promptTokens === SUM_ROUNDS, |
| 439 | `promptTokens=${chatRow.promptTokens}`); |
| 440 | |
| 441 | // What the tile actually DRAWS, read off the DOM rather than recomputed. |
| 442 | const meter = await p.evaluate(() => { |
| 443 | const el = document.querySelector('.session-box .tile-ctx'); |
| 444 | if (!el) return null; |
| 445 | const pct = (el.querySelector('.tile-ctx-pct') || {}).textContent || ''; |
| 446 | const fill = (el.querySelector('.tile-ctx-fill') || {}).style || {}; |
| 447 | return { title: el.getAttribute('title') || '', pct, width: fill.width || '' }; |
| 448 | }); |
| 449 | check('the tile draws a context meter at all', !!meter, meter ? meter.pct : 'no .tile-ctx'); |
| 450 | |
| 451 | if (meter) { |
| 452 | // The denominator: whatever the app believes this model's window is. |
| 453 | const ctx = await p.evaluate((m) => |
| 454 | (window.DaimondPricing ? DaimondPricing.contextWindow(m, '') : null), METER_MODEL); |
| 455 | check('the denominator is the model\'s published context window', |
| 456 | ctx === METER_CTX, `table says ${ctx}, Anthropic publishes ${METER_CTX}`); |
| 457 | const want = ctx ? Math.min(100, Math.round(LAST_ROUND / ctx * 100)) : null; |
| 458 | const wrong = ctx ? Math.min(100, Math.round(SUM_ROUNDS / ctx * 100)) : null; |
| 459 | check('the percentage drawn is the last round over the window', |
| 460 | ctx ? meter.pct === want + '%' : false, |
| 461 | `drew ${meter.pct}; truth ${want}%; the old sum would have drawn ${wrong}%`); |
| 462 | check('and the bar width agrees with the percentage', |
| 463 | ctx ? meter.width === want + '%' : false, `width=${meter.width}`); |
| 464 | } |
| 465 | await shot(s, 'replylen-meter'); |
| 466 | |
| 467 | // ── Report ────────────────────────────────────────────────────────────── |
| 468 | |
| 469 | console.log(`\n${ok.length} ok, ${bad.length} failed`); |
| 470 | if (bad.length) bad.forEach(b => console.log(' FAIL ' + b)); |
| 471 | await s.close(); |
| 472 | if (spawned) spawned.kill(); |
| 473 | process.exit(bad.length ? 1 : 0); |