Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/verify_replylen.mjs

22.4 KiB, 1 run

created by r2519314175:651, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1// verify_replylen.mjs — the reply-length cap, and the context meter.
2//
3// Two defects, one file, and both of them are about a number the app believed
4// that was not true.
5//
6// ── The 4096-token output cap ────────────────────────────────────────────
7// Every request carried `max_tokens: 4096`, whatever the model. 4096 OUTPUT
8// tokens is about 250 lines of code, so a `file_write` of a 400-line module ran
9// out of room part way — and because a tool call's arguments ARE a JSON string,
10// what arrived was not a truncated file but a MALFORMED TOOL CALL. Nothing was
11// written, and the model was told only that its JSON was bad.
12//
13// That is not a claim a unit test on a constant can settle, so it is not tested
14// that way. `dev/mockcap.mjs` is a provider that HONOURS `max_tokens` and cuts
15// the reply when it does not fit, exactly as a real one does, and logs the
16// `max_tokens` of every request it is sent. The failure is demonstrated, then
17// the fix is demonstrated, against the same 400-line write.
18//
19// ── The context meter reading ~12x high ──────────────────────────────────
20// The tile meter drew `lastPrompt / contextWindow`, and `lastPrompt` was set to
21// the turn's CUMULATIVE prompt tokens — the sum of every round. An agentic turn
22// sends the whole conversation once per round, so a five-round turn read five
23// times high. The per-round figure was tracked in Rust all along
24// (`session.last_prompt_tokens`); there was no getter to read it back.
25//
26// The mock reports a KNOWN, DIFFERENT prompt figure per round — 5000, 10000,
27// 15000, … — so "the last one" and "the sum of them" cannot be confused, and
28// the number the meter draws is compared against what the provider actually
29// sent, not against another part of the app.
30//
31// Needs `dev/serve.mjs` (`DAIMOND_PORT`, default 8777) and `dev/mockcap.mjs`
32// (`DAIMOND_CAP_PORT`, default 9250 + the world number), the latter started here
33// if it is not up -- and checked to BE mockcap, see the port note below.
34//
35// Every write it asks for goes into the chat's own scratch folder. A workspace-
36// ROOT path is refused by the chat fence as an ordinary tool result, which reads
37// as neither a write nor a cut: see `freshChat`.
38
39import fs from 'node:fs';
40import path from 'node:path';
41import http from 'node:http';
42import { spawn } from 'node:child_process';
43import { fileURLToPath } from 'node:url';
44import { open, signInAs, connectMock, chat, newChat, shot } from './harness.mjs';
45
46const HERE = path.dirname(fileURLToPath(import.meta.url));
47// The cap mock is a world fixture like the LLM mock, so its port and log follow
48// the world's. `dev/world.sh` numbers a world by its offset from port 8777.
49const WORLD = Number(process.env.DAIMOND_PORT || 8777) - 8777;
50// 9250, NOT 9098. `9098 + WORLD` is `9099 + (WORLD - 1)`, which is the MOCK LLM
51// PORT OF THE WORLD NEXT DOOR: in world 18 the cap mock's port was world 17's
52// mockllm. The guard below found something answering /v1/models there, so no
53// mockcap was started, and every request in this file went to another agent's
54// mock -- which ignores `max_tokens` and does not log, so all eleven checks that
55// read `lastAsk()` reported `max_tokens=null` and the two write checks measured
56// mockllm's plain reply. Measured 2026-08-14 in world 18, with world 17 live.
57// 9250 + WORLD is clear of every other fixture in this tree (compact 9188+N,
58// pickers 9160+2N, applications 9400+N).
59const CAP_PORT = Number(process.env.DAIMOND_CAP_PORT || 9250 + WORLD);
60const CAP_LOG = process.env.DAIMOND_CAP_LOG
61 || path.join(HERE, WORLD ? `mockcap-${WORLD}.log` : 'mockcap.log');
62const CAP_URL = `http://127.0.0.1:${CAP_PORT}/v1/chat/completions`;
63const MODEL = 'cap/plain';
64
65/// What the mock reports as round 0's prompt; round k reports (k+1) times it.
66const ROUND_PROMPT = 5000;
67/// The default the app should now send when nothing published a smaller ceiling.
68const AUTO_MAX = 32768;
69
70const ok = [], bad = [];
71const check = (name, pass, detail) => {
72 (pass ? ok : bad).push(name + (detail ? ' — ' + detail : ''));
73 console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : ''));
74};
75
76// ── The mock ────────────────────────────────────────────────────────────
77
78const up = (url) => new Promise((res) => {
79 const r = http.get(url, () => res(true));
80 r.on('error', () => res(false));
81 r.setTimeout(700, () => { r.destroy(); res(false); });
82});
83
84const CAP_MODELS = `http://127.0.0.1:${CAP_PORT}/v1/models`;
85let spawned = null;
86if (!(await up(CAP_MODELS))) {
87 spawned = spawn('node', [path.join(HERE, 'mockcap.mjs'), String(CAP_PORT)],
88 { stdio: 'ignore', detached: false, env: { ...process.env, DAIMOND_CAP_LOG: CAP_LOG } });
89 for (let i = 0; i < 20 && !(await up(CAP_MODELS)); i++) {
90 await new Promise(r => setTimeout(r, 200));
91 }
92}
93
94// WHAT ANSWERS THERE HAS TO BE MOCKCAP. "Something is listening" was the whole
95// test, and any OpenAI-compatible mock answers /v1/models -- which is how this
96// file spent a run driven by another world's mockllm (see CAP_PORT above).
97// `cap/plain` is mockcap's own model id, so asking for it by name is the cheapest
98// question that only mockcap can answer.
99const capModels = await new Promise((res) => {
100 const r = http.get(CAP_MODELS, (m) => {
101 let raw = '';
102 m.on('data', (c) => { raw += c; });
103 m.on('end', () => { try { res(JSON.parse(raw)); } catch { res(null); } });
104 });
105 r.on('error', () => res(null));
106 r.setTimeout(1500, () => { r.destroy(); res(null); });
107});
108if (!(capModels && (capModels.data || []).some((m) => m.id === MODEL))) {
109 console.log(`\nCANNOT START: :${CAP_PORT} does not answer as dev/mockcap.mjs.`);
110 console.log(` It offers ${JSON.stringify((capModels && capModels.data || []).map((m) => m.id))},`);
111 console.log(` and this whole file measures what a provider that HONOURS max_tokens does.`);
112 console.log(' Something else is on the port — set DAIMOND_CAP_PORT, or free it.');
113 if (spawned) spawned.kill();
114 process.exit(2);
115}
116
117const clearLog = () => { try { fs.writeFileSync(CAP_LOG, ''); } catch {} };
118const capLog = () => {
119 if (!fs.existsSync(CAP_LOG)) return [];
120 return fs.readFileSync(CAP_LOG, 'utf8').split('\n').filter(Boolean)
121 .map(l => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
122};
123/// The `max_tokens` of the last request the mock was sent.
124const lastAsk = () => { const l = capLog(); return l.length ? l[l.length - 1].max_tokens : null; };
125
126// ── The app ─────────────────────────────────────────────────────────────
127
128const s = await open({ name: 'replylen', connect: false });
129const p = s.page;
130await connectMock(s, { baseUrl: CAP_URL, model: MODEL });
131await p.waitForTimeout(600);
132
133/// Force a reply-length setting through the real control, or through the store
134/// when the control has not been built (the panel may be closed).
135async function setReplyLength(n) {
136 await p.evaluate(async (n) => {
137 const sel = document.getElementById('cfg-max-tokens');
138 if (sel) {
139 // The value may not be on the offered ladder; add it so `change` takes.
140 if (![...sel.options].some(o => o.value === String(n))) {
141 const o = document.createElement('option');
142 o.value = String(n); o.textContent = String(n);
143 sel.appendChild(o);
144 }
145 sel.value = String(n);
146 sel.dispatchEvent(new Event('change', { bubbles: true }));
147 return;
148 }
149 const raw = JSON.parse(localStorage.getItem('daimond-byok') || '{}');
150 raw.maxOut = n;
151 localStorage.setItem('daimond-byok', JSON.stringify(raw));
152 }, n);
153 await p.waitForTimeout(200);
154 // A DaimondApp freezes its max_tokens at construction, so a chat built under
155 // the old setting must be rebuilt. Reloading is what a user would see anyway.
156 await p.reload({ waitUntil: 'domcontentloaded' });
157 await signInAs(s, 'replylen');
158 await p.waitForTimeout(900);
159}
160
161/// The stored chats, as the app persists them.
162///
163/// IndexedDB, not localStorage: transcripts moved there when a day of tool results
164/// stopped fitting in the origin's five megabytes. Read from outside the app, so
165/// what is asserted is what is on disk rather than what the page believes.
166const storedChats = () => p.evaluate(() => new Promise((res) => {
167 const req = indexedDB.open('daimond-chats', 1);
168 req.onsuccess = () => {
169 const db = req.result;
170 let t;
171 try { t = db.transaction('chats', 'readonly'); } catch (e) { res([]); return; }
172 const all = t.objectStore('chats').getAll();
173 all.onsuccess = () => res(all.result || []);
174 all.onerror = () => res([]);
175 };
176 req.onerror = () => res([]);
177}));
178
179/// Empty the chat store, so a turn is never metered against a previous one. Both
180/// places: the store itself, and the old localStorage key, which would otherwise be
181/// migrated straight back in on the next boot.
182const clearChats = () => p.evaluate(() => new Promise((res) => {
183 try { localStorage.removeItem('daimond-chats'); localStorage.removeItem('daimond-chats-legacy'); } catch (e) { /* full */ }
184 const req = indexedDB.open('daimond-chats', 1);
185 req.onsuccess = () => {
186 const db = req.result;
187 let t;
188 try { t = db.transaction('chats', 'readwrite'); } catch (e) { res(); return; }
189 t.objectStore('chats').clear();
190 t.oncomplete = () => res();
191 t.onerror = () => res();
192 };
193 req.onerror = () => res();
194}));
195
196/// Start a fresh chat so a turn is never metered against a previous one, and
197/// answer where that chat may write.
198///
199/// THE PATH IS NOT OPTIONAL. §2 below asked the model for `@write 400 big.js` --
200/// a workspace-ROOT path -- and since the chat fence landed on 2026-08-12
201/// (5389864) every chat is confined to `chats/<id>/work`: `Tool::guard`
202/// (src/tools.rs) refuses anything outside it BEFORE the tool runs, and the
203/// refusal comes back as an ORDINARY TOOL RESULT. Nothing throws, the turn
204/// finishes, and the transcript holds `Refused: 'big.js' is not in this chat's
205/// workspace` -- which contains neither "Wrote N bytes" nor "ran out of room",
206/// so 2a passed for the wrong reason and 2b, the only check that says the raised
207/// default FIXES a long write, had been measuring an apology. The commit that
208/// repaired six other verifiers this way (see dev/harness.mjs) missed this one.
209async function freshChat() {
210 await clearChats();
211 await p.reload({ waitUntil: 'domcontentloaded' });
212 await signInAs(s, 'replylen');
213 await p.waitForTimeout(900);
214 await newChat(s);
215 return await scratchDir();
216}
217
218/// `chats/<id>/work` for the chat in focus -- what `scopeChatTo` hands the
219/// engine, asked of the app rather than spelled out here, so a change to the
220/// fence's shape moves this with it.
221async function scratchDir() {
222 return await p.evaluate(() => {
223 const f = window.DaimondAttach && window.DaimondAttach.focus();
224 if (!f || f.kind !== 'chat') return '';
225 return window.DaimondAttach.chatScratch(f.id) || '';
226 });
227}
228
229console.log('\n── 1. The cap that reaches the provider ──');
230
231// ── 1a. The default is no longer 4096 ────────────────────────────────────
232
233clearLog();
234await freshChat();
235await chat(s, '@text hello');
236const ask1 = lastAsk();
237check('the request carries the raised default, not 4096',
238 ask1 === AUTO_MAX, `max_tokens=${ask1}`);
239
240// ── 1b. A per-model ceiling binds below the default ──────────────────────
241//
242// claude-opus-4-1 accepts at most 32,000 output tokens, which is BELOW the
243// 32,768 default: a request above a model's maximum is an error, not a clamp,
244// so the app must ask for the model's figure and not its own.
245
246await clearChats();
247await connectMock(s, { baseUrl: CAP_URL, model: 'claude-opus-4-1' });
248await p.waitForTimeout(400);
249clearLog();
250await freshChat();
251await chat(s, '@text hello');
252const askOpus41 = lastAsk();
253check('a model whose published ceiling is lower gets its own figure',
254 askOpus41 === 32000, `claude-opus-4-1 → max_tokens=${askOpus41}`);
255
256// ── 1c. A model with a larger ceiling still gets the default ─────────────
257
258await connectMock(s, { baseUrl: CAP_URL, model: 'claude-opus-5' });
259await p.waitForTimeout(400);
260clearLog();
261await freshChat();
262await chat(s, '@text hello');
263const askOpus5 = lastAsk();
264check('a model with a larger ceiling keeps the default',
265 askOpus5 === AUTO_MAX, `claude-opus-5 (128k ceiling) → max_tokens=${askOpus5}`);
266
267await connectMock(s, { baseUrl: CAP_URL, model: MODEL });
268await p.waitForTimeout(400);
269
270console.log('\n── 2. What the cap does to a 400-line write ──');
271
272// ── 2a. Under the OLD cap the write fails as a malformed tool call ───────
273
274await setReplyLength(4096);
275clearLog();
276const cutDir = await freshChat();
277
278// A CAN'T-START CHECK, and not one of the checks.
279//
280// Everything below this line writes into the chat's scratch folder, and a path
281// outside it is refused before the tool runs -- silently, as an ordinary tool
282// result. If the app cannot say where that folder is, the fixture cannot be
283// seeded, and a run that carried on would be measuring the refusal again. It
284// exits 2 rather than failing a check, because nothing here was tested.
285if (!/^chats\/[^/]+\/work$/.test(cutDir)) {
286 console.log(`\nCANNOT START: the chat's scratch folder did not answer — got '${cutDir}'.`);
287 console.log(' §2 writes a 400-line file, and only `chats/<id>/work` will take it.');
288 await s.close();
289 if (spawned) spawned.kill();
290 process.exit(2);
291}
292/// The file §2 asks for, inside the fence. `big.js` on its own is refused.
293const BIG = `${cutDir}/big.js`;
294const wrote = (out, file) => new RegExp('Wrote (\\d+) bytes to ' + file.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).exec(out);
295
296const cut = await chat(s, `@write 400 ${BIG}`, { timeout: 60000 });
297const askCut = lastAsk();
298check('the forced 4096 setting is what is sent', askCut === 4096, `max_tokens=${askCut}`);
299// The refusal that used to satisfy this check by accident is now itself a
300// failure: it is neither a write nor a length cut, and it must not be either.
301check('the fixture reached the tool at all — it was not refused by the chat fence',
302 !/is not in this chat's workspace/i.test(cut),
303 (cut.match(/Refused:[^\n]*/) || ['no refusal'])[0]);
304check('under 4096 the write does NOT complete',
305 !wrote(cut, BIG),
306 cut.match(/Wrote \d+ bytes[^\n]*/)?.[0] || 'no write recorded');
307check('under 4096 the failure is REPORTED as a length cut, not left as bad JSON',
308 /ran out of room/i.test(cut), (cut.match(/ran out of room[^\n]*/) || ['(no notice)'])[0]);
309await shot(s, 'replylen-cut');
310
311// ── 2b. Under the new default the same write completes ───────────────────
312
313await setReplyLength(0); // 0 = Automatic
314clearLog();
315const wholeDir = await freshChat();
316const BIG2 = `${wholeDir}/big.js`;
317const whole = await chat(s, `@write 400 ${BIG2}`, { timeout: 60000 });
318const askWhole = lastAsk();
319const said = (wrote(whole, BIG2) || [])[1];
320check('Automatic sends the raised default', askWhole === AUTO_MAX, `max_tokens=${askWhole}`);
321// Said here as well as in 2a, and it is here that it bites: at 4096 the reply is
322// cut before any tool call is made, so there is nothing for the fence to refuse.
323// This is the turn that DOES reach the tool, and a refusal is what it used to
324// come back with.
325check('the write reached the tool at all — it was not refused by the chat fence',
326 !/is not in this chat's workspace/i.test(whole),
327 (whole.match(/Refused:[^\n]*/) || ['no refusal'])[0]);
328check('under the new default the same 400-line write completes',
329 !!said, said ? `${said} bytes written` : 'no write recorded');
330// The transcript is what the MODEL was told. The file is the thing the user
331// keeps, so it is read back through the engine's own door -- not fenced,
332// because it is not a chat -- and its size compared with what was claimed.
333const onDisk = await p.evaluate(async (f) => {
334 try { return (await (await import('/pkg/oxedyne_daimond.js')).read_file(f)).length; }
335 catch (e) { return -1; }
336}, BIG2);
337check('and the file is really there, the size it said',
338 said && onDisk === Number(said), `${onDisk} bytes at ${BIG2}`);
339check('and nothing is reported as cut', !/ran out of room/i.test(whole));
340await shot(s, 'replylen-whole');
341
342// ── 2c. The setting is real: a chosen figure is what is sent ─────────────
343
344await setReplyLength(8192);
345clearLog();
346await freshChat();
347await chat(s, '@text hello');
348const askSet = lastAsk();
349check('a chosen reply length reaches the provider', askSet === 8192, `max_tokens=${askSet}`);
350const persisted = await p.evaluate(() => {
351 try { return JSON.parse(localStorage.getItem('daimond-byok') || '{}').maxOut; } catch { return null; }
352});
353check('and it survives a reload', persisted === 8192, `stored maxOut=${persisted}`);
354
355// The control itself, on screen: a row that says which model it is reading and
356// what that model will accept. Read off the DOM and shot, because a setting
357// nobody can find is not a setting.
358await p.evaluate(() => { document.getElementById('settings-btn')?.click(); });
359await p.waitForTimeout(400);
360await p.evaluate(() => {
361 const row = document.querySelector('.astat-btn#astat-model') || document.getElementById('astat-model');
362 if (row) row.click();
363});
364await p.waitForTimeout(500);
365const knob = await p.evaluate(() => {
366 const sel = document.getElementById('cfg-max-tokens');
367 if (!sel) return null;
368 const note = document.getElementById('cfg-max-tokens-note');
369 const r = sel.getBoundingClientRect();
370 return {
371 visible: r.width > 0 && r.height > 0,
372 options: [...sel.options].map(o => o.textContent),
373 value: sel.value,
374 note: (note || {}).textContent || '',
375 };
376});
377check('the reply-length control is on screen in the Models panel',
378 !!(knob && knob.visible), knob ? `value=${knob.value}` : 'not built');
379check('it offers Automatic and a ladder bounded by the model',
380 !!(knob && /Automatic/.test(knob.options[0]) && knob.options.length > 2),
381 knob ? knob.options.join(' | ') : '');
382check('and it names what the model will accept',
383 !!(knob && knob.note.length > 20), knob ? knob.note : '');
384await shot(s, 'replylen-setting');
385
386// ── 2d. A provider that refuses the length is answered, not reported ─────
387
388await setReplyLength(0);
389clearLog();
390await freshChat();
391// Refuses above 20,000: the first ask (32,768) is rejected, the halved one
392// (16,384) is not — so the backoff is shown to LAND, not merely to happen.
393const refused = await chat(s, '@refuse-cap 20000', { timeout: 60000 });
394const asks = capLog().map(e => e.max_tokens);
395// The provider's own words never reach the browser -- `wasm_fetch` in src/llm.rs
396// builds the error from the status line and drops the body -- so what the app
397// has to work from is a bare 400. It answers by asking for half as much, once,
398// and believes the smaller figure only because the smaller ask then succeeds.
399check('a refused length is retried at half, not surfaced',
400 asks.length >= 2 && asks[0] === AUTO_MAX && asks[1] === AUTO_MAX / 2,
401 `asked ${asks.join(' then ')}`);
402check('and the turn then succeeds',
403 /Fine at this length/.test(refused), (refused.split('\n').pop() || '').slice(0, 60));
404
405console.log('\n── 3. The context meter ──');
406
407// A five-round tool loop. The mock reports 5000, 10000, 15000, 20000, 25000
408// prompt tokens across the rounds: the LAST is 25000, the SUM is 75000.
409//
410// On `claude-opus-5`, because a meter needs a DENOMINATOR: a model nobody
411// publishes a context window for draws no meter at all, which is the honest
412// behaviour and the reason the mock answers to a real model id here. Anthropic
413// publishes 1,000,000 tokens for this model, which is what the app's table says
414// -- so the denominator is checked against the published figure, not against
415// another part of the app.
416const METER_MODEL = 'claude-opus-5';
417const METER_CTX = 1000000;
418const ROUNDS = 5;
419const LAST_ROUND = ROUND_PROMPT * ROUNDS;
420const SUM_ROUNDS = ROUND_PROMPT * (ROUNDS * (ROUNDS + 1) / 2);
421
422await setReplyLength(0);
423await connectMock(s, { baseUrl: CAP_URL, model: METER_MODEL });
424await p.waitForTimeout(400);
425clearLog();
426await freshChat();
427await chat(s, `@rounds ${ROUNDS}`, { timeout: 60000 });
428await p.waitForTimeout(600);
429
430const rounds = capLog().length;
431check(`the turn really ran ${ROUNDS} rounds`, rounds === ROUNDS, `${rounds} requests`);
432
433const chatRow = (await storedChats())[0] || {};
434check('the meter reads the LAST round, not the sum of the rounds',
435 chatRow.lastPrompt === LAST_ROUND,
436 `lastPrompt=${chatRow.lastPrompt}, last round sent ${LAST_ROUND}, sum is ${SUM_ROUNDS}`);
437check('the cumulative counter is still the sum (nothing else was broken to fix it)',
438 chatRow.promptTokens === SUM_ROUNDS,
439 `promptTokens=${chatRow.promptTokens}`);
440
441// What the tile actually DRAWS, read off the DOM rather than recomputed.
442const meter = await p.evaluate(() => {
443 const el = document.querySelector('.session-box .tile-ctx');
444 if (!el) return null;
445 const pct = (el.querySelector('.tile-ctx-pct') || {}).textContent || '';
446 const fill = (el.querySelector('.tile-ctx-fill') || {}).style || {};
447 return { title: el.getAttribute('title') || '', pct, width: fill.width || '' };
448});
449check('the tile draws a context meter at all', !!meter, meter ? meter.pct : 'no .tile-ctx');
450
451if (meter) {
452 // The denominator: whatever the app believes this model's window is.
453 const ctx = await p.evaluate((m) =>
454 (window.DaimondPricing ? DaimondPricing.contextWindow(m, '') : null), METER_MODEL);
455 check('the denominator is the model\'s published context window',
456 ctx === METER_CTX, `table says ${ctx}, Anthropic publishes ${METER_CTX}`);
457 const want = ctx ? Math.min(100, Math.round(LAST_ROUND / ctx * 100)) : null;
458 const wrong = ctx ? Math.min(100, Math.round(SUM_ROUNDS / ctx * 100)) : null;
459 check('the percentage drawn is the last round over the window',
460 ctx ? meter.pct === want + '%' : false,
461 `drew ${meter.pct}; truth ${want}%; the old sum would have drawn ${wrong}%`);
462 check('and the bar width agrees with the percentage',
463 ctx ? meter.width === want + '%' : false, `width=${meter.width}`);
464}
465await shot(s, 'replylen-meter');
466
467// ── Report ──────────────────────────────────────────────────────────────
468
469console.log(`\n${ok.length} ok, ${bad.length} failed`);
470if (bad.length) bad.forEach(b => console.log(' FAIL ' + b));
471await s.close();
472if (spawned) spawned.kill();
473process.exit(bad.length ? 1 : 0);