oxedyne/daimond/dev/verify_compact.mjs
17.8 KiB, 1 run
created by r2519314175:301, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // Does a long session die? |
| 2 | // |
| 3 | // The failure being fixed is not subtle: the provider refuses the request for being |
| 4 | // longer than its window, the turn dies, and every turn after it sends the same |
| 5 | // oversized history and dies the same way. The chat is then unusable forever. |
| 6 | // |
| 7 | // So this drives the real app against a provider that really does refuse -- see |
| 8 | // ctxmock.mjs, which counts the request the way a provider does and answers 400 with |
| 9 | // `context_length_exceeded` -- and then asks the only question that matters: after the |
| 10 | // refusal, does the chat still work, and does it still know what it did? |
| 11 | // |
| 12 | // dev/mockllm.mjs accepts a request of any size, which is why nothing else in the suite |
| 13 | // has ever exercised this. The mock is started and stopped here rather than by hand, so |
| 14 | // this runs from `run_all.sh` like everything else. |
| 15 | // |
| 16 | // Every check is made against the mock's own log, which is what the model was really |
| 17 | // sent, not against anything the compactor says about itself. |
| 18 | // |
| 19 | // node dev/verify_compact.mjs |
| 20 | |
| 21 | import fs from 'node:fs'; |
| 22 | import path from 'node:path'; |
| 23 | import { spawn } from 'node:child_process'; |
| 24 | import { fileURLToPath } from 'node:url'; |
| 25 | |
| 26 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 27 | const ROOT = path.dirname(HERE); |
| 28 | const { open, newChat, transcript, shot, connectMock } = await import(path.join(HERE, 'harness.mjs')); |
| 29 | |
| 30 | // The refusing provider follows the world, so two worlds neither fight over the |
| 31 | // port nor read one another's log. A world is numbered by its offset from 8777. |
| 32 | const WORLD = Number(process.env.DAIMOND_PORT || 8777) - 8777; |
| 33 | const LOG = process.env.DAIMOND_CTX_LOG |
| 34 | || path.join(HERE, WORLD ? `ctxmock-${WORLD}.log` : 'ctxmock.log'); |
| 35 | const PORT = Number(process.env.DAIMOND_CTX_PORT || 9188 + WORLD); |
| 36 | // THE WINDOW IS MEASURED, NOT WRITTEN DOWN, and that is the whole of this file's |
| 37 | // calibration. It was `const LIMIT = 12000`, chosen when the fixed cost of a request -- |
| 38 | // the system prompt plus every tool's schema -- sat just under it, so a short |
| 39 | // conversation crossed the line and had to be folded back. |
| 40 | // |
| 41 | // That cost is not a constant and nobody owns it. It was 11,719 tokens on 2026-08-24 |
| 42 | // with 29 tools, and 12,652 with 31 by the end of the same day: a lane adding a tool or |
| 43 | // lengthening a description moves it, and none of them is doing anything wrong. The day |
| 44 | // it passed 12,000 this file stopped being able to pass AT ALL -- an EMPTY conversation |
| 45 | // was already over the window, so there was nothing left to fold, and seven checks went |
| 46 | // red saying things like "the conversation was actually made smaller -- peak 12714 -> |
| 47 | // last 12714". None of that was about compaction. |
| 48 | // |
| 49 | // So the floor is probed first, against a mock that never refuses, and the real window is |
| 50 | // set just above what was measured. The test then always asks its own question: a |
| 51 | // conversation that grows past the window is folded back under it. |
| 52 | // |
| 53 | // AND THE FLOOR IS NOT THE WINDOW. `Limits::budget` folds at a FRACTION of the window |
| 54 | // (`compact::FOLD_AT`), so a window set to the floor plus a little leaves a budget a third |
| 55 | // BELOW the floor and the same seven checks go red for the same non-reason. That is what |
| 56 | // happened on 2026-08-28, when the owner moved the fraction from 0.8 to 0.65 -- so the |
| 57 | // fraction is read from `compact.rs` and divided out, and this file stops caring what it is. |
| 58 | const PROBE_LIMIT = 10_000_000; // a mock that refuses nothing, for the probe |
| 59 | // Tokens above the floor. It has to clear the FOLD NOTICE, not just a message or two: |
| 60 | // folding replaces the conversation with a summary that is itself sent every turn, so a |
| 61 | // window only a hair above the floor leaves the folded conversation still over it -- |
| 62 | // measured at 120, where the fold shrank a turn from 12,799 to 12,797 and refused |
| 63 | // anyway. The equivalent figure was 281 on 2026-08-24, when this last worked. |
| 64 | const HEADROOM = 400; |
| 65 | /// Where the engine folds, as a fraction of the window: the one authority is `compact.rs`. |
| 66 | const FOLD_AT = (() => { |
| 67 | const m = /pub const FOLD_AT:\s*f64\s*=\s*([0-9.]+)/.exec( |
| 68 | fs.readFileSync(path.join(ROOT, 'src', 'compact.rs'), 'utf8')); |
| 69 | if (!m) throw new Error('compact.rs no longer declares FOLD_AT; this file cannot calibrate'); |
| 70 | return Number(m[1]); |
| 71 | })(); |
| 72 | let LIMIT = 0; // set from the probe, below |
| 73 | const MOCK = `http://127.0.0.1:${PORT}/v1/chat/completions`; |
| 74 | const MODEL = 'mock/fast'; |
| 75 | |
| 76 | const log = (...a) => console.log(...a); |
| 77 | const line = (t) => log('\n════════ ' + t + ' ════════'); |
| 78 | |
| 79 | let failures = 0, pending = 0; |
| 80 | const check = (ok, what, detail = '') => { |
| 81 | log((ok ? ' PASS ' : ' FAIL ') + what + (detail ? ' -- ' + detail : '')); |
| 82 | if (!ok) failures++; |
| 83 | }; |
| 84 | |
| 85 | // Whether the browser half draws the fold at all yet. The fold stopped borrowing the |
| 86 | // tool surface and became its own event; until `www/js/daimond.js` handles |
| 87 | // `ev.type === 'compacted'` there is nothing in the thread to look for. |
| 88 | let rendersCompacted = false; |
| 89 | try { |
| 90 | rendersCompacted = /'compacted'|"compacted"/.test( |
| 91 | fs.readFileSync(path.join(ROOT, 'www', 'js', 'daimond.js'), 'utf8')); |
| 92 | } catch (e) { /* no www tree: treat as not rendered */ } |
| 93 | |
| 94 | // A check whose subject is that browser-side edit. It becomes a real check the moment |
| 95 | // the edit lands, and says so plainly until then -- rather than passing quietly, which |
| 96 | // would hide the gap, or failing, which would report another lane's unfinished work as |
| 97 | // a defect in this one. |
| 98 | const whenRendered = (ok, what, detail = '') => { |
| 99 | if (rendersCompacted) return check(ok, what, detail); |
| 100 | log(' PEND ' + what + ' -- www/js/daimond.js does not handle ev.type === \'compacted\' yet'); |
| 101 | pending++; |
| 102 | }; |
| 103 | |
| 104 | const requests = () => fs.readFileSync(LOG, 'utf8').split('\n').filter(Boolean) |
| 105 | .map(l => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean); |
| 106 | |
| 107 | /// An independent reading of the rule a provider enforces: an assistant message with |
| 108 | /// tool_calls must be followed by one tool reply per call, in order, and a tool reply |
| 109 | /// must answer a call. Written here rather than imported so it agrees with nothing. |
| 110 | function orphans(msgs) { |
| 111 | let n = 0, i = 0; |
| 112 | while (i < msgs.length) { |
| 113 | const m = msgs[i]; |
| 114 | if (m.role === 'assistant' && Array.isArray(m.tool_calls) && m.tool_calls.length) { |
| 115 | let k = 0; |
| 116 | while (k < m.tool_calls.length) { |
| 117 | const r = msgs[i + 1 + k]; |
| 118 | if (r && r.role === 'tool' && r.tool_call_id === m.tool_calls[k].id) k++; |
| 119 | else break; |
| 120 | } |
| 121 | n += m.tool_calls.length - k; |
| 122 | i += 1 + k; |
| 123 | } else if (m.role === 'tool') { n++; i++; } |
| 124 | else i++; |
| 125 | } |
| 126 | return n; |
| 127 | } |
| 128 | |
| 129 | const say = async (s, text, waitMs = 6000) => { |
| 130 | await s.page.fill('#chat-input', text); |
| 131 | await s.page.click('#chat-send'); |
| 132 | await s.page.waitForTimeout(waitMs); |
| 133 | }; |
| 134 | |
| 135 | /// Start the mock and wait for it to answer, so a slow start is not read as a refusal. |
| 136 | async function startMock(limit) { |
| 137 | const child = spawn('node', [path.join(HERE, 'ctxmock.mjs'), String(PORT), String(limit)], |
| 138 | { stdio: ['ignore', 'ignore', 'inherit'], env: { ...process.env, DAIMOND_CTX_LOG: LOG } }); |
| 139 | for (let i = 0; i < 50; i++) { |
| 140 | try { |
| 141 | const r = await fetch(`http://127.0.0.1:${PORT}/v1/models`); |
| 142 | if (r.ok) return child; |
| 143 | } catch (e) { /* not up yet */ } |
| 144 | await new Promise(r => setTimeout(r, 100)); |
| 145 | } |
| 146 | child.kill(); |
| 147 | throw new Error(`ctxmock did not come up on ${PORT}; is something else already listening?`); |
| 148 | } |
| 149 | |
| 150 | // ── go ─────────────────────────────────────────────────────────────────────── |
| 151 | try { fs.writeFileSync(LOG, ''); } catch {} |
| 152 | let mock = await startMock(PROBE_LIMIT); |
| 153 | |
| 154 | const s = await open({ name: 'compact', connect: false }); |
| 155 | const cfg = await connectMock(s, { baseUrl: MOCK, model: MODEL }); |
| 156 | log('connected:', JSON.stringify(cfg)); |
| 157 | await newChat(s); |
| 158 | |
| 159 | // ── The probe: what does a conversation cost before anybody has said anything? ── |
| 160 | // |
| 161 | // Read off the mock's own log, which is what the model was really sent -- the same rule |
| 162 | // every check below follows. One short message is enough: the floor is the system prompt |
| 163 | // and the tool schemas, and neither depends on what was said. |
| 164 | await say(s, 'hello', 4000); |
| 165 | const probed = (() => { |
| 166 | try { |
| 167 | const rows = fs.readFileSync(LOG, 'utf8').trim().split('\n').filter(Boolean).map(JSON.parse); |
| 168 | return rows.length ? rows[0].used : 0; |
| 169 | } catch (e) { return 0; } |
| 170 | })(); |
| 171 | if (!probed) { |
| 172 | log('FAIL the probe never reached the mock, so the window cannot be calibrated'); |
| 173 | process.exit(1); |
| 174 | } |
| 175 | // The window whose BUDGET clears the floor, not the window that clears it itself. |
| 176 | LIMIT = Math.ceil((probed + HEADROOM) / FOLD_AT); |
| 177 | log(`floor ${probed} tokens (system prompt + ${(() => { |
| 178 | try { |
| 179 | const rows = fs.readFileSync(LOG, 'utf8').trim().split('\n').filter(Boolean).map(JSON.parse); |
| 180 | return rows[0].tools; |
| 181 | } catch (e) { return '?'; } |
| 182 | })()} tool schemas), folding at ${FOLD_AT}, so the window for this run is ${LIMIT}`); |
| 183 | |
| 184 | // The real mock, at the measured window, and a fresh log so the probe's own rows are not |
| 185 | // read as refusals that never happened. |
| 186 | mock.kill(); |
| 187 | await new Promise((r) => setTimeout(r, 300)); |
| 188 | try { fs.writeFileSync(LOG, ''); } catch {} |
| 189 | mock = await startMock(LIMIT); |
| 190 | await newChat(s); |
| 191 | |
| 192 | line('1. do some work worth remembering'); |
| 193 | // TWO WRITES THAT END DIFFERENTLY, and that is the fixture, not an accident. |
| 194 | // |
| 195 | // ALPHA goes in the chat's own scratch and really lands. BETA names a workspace-ROOT |
| 196 | // path and is REFUSED: since the chat fence landed on 2026-08-12 a chat is confined to |
| 197 | // `chats/<id>/work` (`scopeChatTo`, www/js/daimond.js) and `Tool::guard` |
| 198 | // (src/tools.rs:5490) turns a root path back before the write, returning the refusal |
| 199 | // as an ordinary tool result -- nothing written, nothing thrown. |
| 200 | // |
| 201 | // Both used to be root paths, so by the time the fence arrived NEITHER file existed |
| 202 | // and the check below still passed on a fold notice naming both. That is the shape |
| 203 | // of the product defect this is re-aimed at: a fold that lists files that were never |
| 204 | // created, under a closing line telling the model to trust the list. |
| 205 | const SCRATCH = await s.page.evaluate(() => { |
| 206 | const f = window.DaimondAttach.focus(); |
| 207 | return f && f.id ? window.DaimondAttach.chatScratch(f.id) : ''; |
| 208 | }); |
| 209 | check(!!SCRATCH, 'the chat has a scratch folder, so one of the two writes can succeed', SCRATCH); |
| 210 | const ALPHA = SCRATCH + '/alpha.txt'; // written |
| 211 | const BETA = 'beta.txt'; // refused: outside the fence |
| 212 | await say(s, `@tool file_write {"path":"${ALPHA}","content":"the first file"}`, 7000); |
| 213 | await say(s, `@tool file_write {"path":"${BETA}","content":"the second file"}`, 7000); |
| 214 | log('after two writes:', (await transcript(s)).slice(-120).replace(/\n/g, ' | ')); |
| 215 | |
| 216 | /// Is this path really on disk? Asked of OPFS, outside the app, so what the fold |
| 217 | /// CLAIMS is measured against the store and not against another claim. |
| 218 | const onDisk = (p) => s.page.evaluate(async (p) => { |
| 219 | const parts = p.split('/'); |
| 220 | let d = await navigator.storage.getDirectory(); |
| 221 | for (const seg of parts.slice(0, -1)) d = await d.getDirectoryHandle(seg); |
| 222 | await d.getFileHandle(parts[parts.length - 1]); |
| 223 | return true; |
| 224 | }, p).catch(() => false); |
| 225 | const alphaThere = await onDisk(ALPHA); |
| 226 | const betaThere = await onDisk(BETA); |
| 227 | check(alphaThere, 'the fixture: one write really landed', ALPHA); |
| 228 | check(!betaThere, 'and the other really did not', BETA + ' is ' + (betaThere ? 'on disk' : 'absent, as the fence intends')); |
| 229 | |
| 230 | line('2. fill the window'); |
| 231 | // Each of these is about 20 KB of assistant text, so the conversation crosses the |
| 232 | // mock's 12,000-token ceiling within a few turns. |
| 233 | for (let i = 0; i < 5; i++) { |
| 234 | await say(s, `@big 20`, 9000); |
| 235 | const r = requests(); |
| 236 | const last = r[r.length - 1] || {}; |
| 237 | log(` turn ${i + 1}: last request ${last.used} tokens, refused=${!!last.refused}, ` + |
| 238 | `${(last.messages || []).length} messages`); |
| 239 | if (r.some(x => x.refused)) { log(' the provider has started refusing'); break; } |
| 240 | } |
| 241 | |
| 242 | line('3. keep going -- this is where the old build died'); |
| 243 | await say(s, '@text REPLY-AFTER-FOLD-ONE', 12000); |
| 244 | await say(s, '@text REPLY-AFTER-FOLD-TWO', 12000); |
| 245 | const tail = await transcript(s); |
| 246 | |
| 247 | // ── what the mock actually saw ─────────────────────────────────────────────── |
| 248 | line('what the model was really sent'); |
| 249 | const reqs = requests(); |
| 250 | const refused = reqs.filter(r => r.refused); |
| 251 | const after = reqs.slice(reqs.findIndex(r => r.refused) + 1); |
| 252 | const peak = Math.max(...reqs.map(r => r.used)); |
| 253 | log(`requests: ${reqs.length}, refused: ${refused.length}, peak: ${peak} tokens, ` + |
| 254 | `last: ${reqs[reqs.length - 1].used} tokens`); |
| 255 | |
| 256 | check(refused.length > 0, |
| 257 | 'the provider really did refuse an oversized request', |
| 258 | `${refused.length} refusals, first at ${(refused[0] || {}).used} tokens`); |
| 259 | |
| 260 | check(after.length > 0 && after.every(r => !r.refused), |
| 261 | 'every request after the first refusal was accepted', |
| 262 | `${after.filter(r => r.refused).length} of ${after.length} still refused`); |
| 263 | |
| 264 | // The mock's own log, not the transcript: the transcript echoes what the user typed, |
| 265 | // so a dead chat still "contains" the words. What proves the chat is alive is that the |
| 266 | // provider ACCEPTED a request carrying them and the reply came back. |
| 267 | const answered = reqs.some(r => !r.refused && (r.messages || []).some( |
| 268 | m => m.role === 'user' && /REPLY-AFTER-FOLD-TWO/.test(String(m.content || '')))); |
| 269 | check(answered && /REPLY-AFTER-FOLD-TWO/.test(tail), |
| 270 | 'the chat still answers after the refusal', |
| 271 | tail.slice(-160).replace(/\n/g, ' | ')); |
| 272 | |
| 273 | const last = reqs[reqs.length - 1]; |
| 274 | check(last.used < peak, |
| 275 | 'the conversation was actually made smaller', |
| 276 | `peak ${peak} -> last ${last.used} tokens`); |
| 277 | |
| 278 | const folded = reqs.find(r => (r.messages || []).some(m => /Daimond folded/.test(String(m.content || '')))); |
| 279 | check(!!folded, 'a fold notice reached the model'); |
| 280 | |
| 281 | if (folded) { |
| 282 | const note = folded.messages.find(m => /Daimond folded/.test(String(m.content || ''))); |
| 283 | check(note.role === 'user', |
| 284 | 'the fold notice is not in the assistant\'s own voice', `role=${note.role}`); |
| 285 | // THE LEDGER MUST MATCH THE DISK, IN BOTH DIRECTIONS. A fold's "Files written" |
| 286 | // is the only record of the work that survives the fold, and the model is told |
| 287 | // to trust it -- so it has to name every file that WAS written and no file that |
| 288 | // was not. One direction alone is satisfiable by a ledger that lists every |
| 289 | // attempt (naming the refused one too) or by one that lists nothing at all. |
| 290 | // READ THE COLUMN, NOT THE WHOLE NOTICE. This was a blanket substring test |
| 291 | // over the entire notice, which forbade `beta.txt` appearing anywhere in it -- |
| 292 | // and the fix it was written for names refusals in a column of their own |
| 293 | // (`REFUSED, so nothing was touched:`, src/compact.rs) precisely so the model |
| 294 | // is told the door was shut rather than left to infer it from silence. So the |
| 295 | // check forbade the very output its own fix produces. |
| 296 | // |
| 297 | // What the ledger has to get right is which COLUMN a name lands in, and both |
| 298 | // directions of that are asserted: a refused file absent from "Files written" |
| 299 | // but present on the refusal line, and the written file present on "Files |
| 300 | // written". One direction alone is satisfiable by a ledger that lists nothing. |
| 301 | // Each ledger line arrives as a bullet under "## What was touched" (`notice`, |
| 302 | // src/compact.rs), so the leading "- " comes off before the label is read. |
| 303 | const line = label => (String(note.content).split('\n') |
| 304 | .map(l => l.trim().replace(/^-\s*/, '')) |
| 305 | .find(l => l.startsWith(label + ':')) || ''); |
| 306 | const wroteLine = line('Files written'); |
| 307 | const refusedLine = line('REFUSED, so nothing was touched'); |
| 308 | check(alphaThere && /alpha\.txt/.test(wroteLine), |
| 309 | 'the fold names the file that WAS written on "Files written"', |
| 310 | wroteLine.trim().slice(0, 200) || '(no "Files written" line)'); |
| 311 | check(!/beta\.txt/.test(wroteLine), |
| 312 | 'and does NOT claim the refused write there — nothing was created, so nothing may be claimed', |
| 313 | wroteLine.trim().slice(0, 240) || '(no "Files written" line)'); |
| 314 | check(betaThere === false && /beta\.txt/.test(refusedLine), |
| 315 | 'the refused write is named on the REFUSED line, so the model is told the door was shut', |
| 316 | refusedLine.trim().slice(0, 240) || '(no REFUSED line)'); |
| 317 | check(/SUMMARY-FROM-MODEL/.test(note.content), |
| 318 | 'the summarising call\'s answer is in the notice'); |
| 319 | } |
| 320 | |
| 321 | const broken = reqs.filter(r => orphans(r.messages || []) > 0); |
| 322 | check(broken.length === 0, |
| 323 | 'every request the browser sent was a whole conversation (no orphaned tool calls)', |
| 324 | `${broken.length} of ${reqs.length} were not`); |
| 325 | |
| 326 | check(!/HTTP error/.test(tail), |
| 327 | 'the user was never shown the refusal', |
| 328 | (tail.match(/LLM: HTTP error[^|]*/g) || []).join(' | ')); |
| 329 | |
| 330 | // The fold used to borrow the tool surface and appear as a `context_compaction` row. |
| 331 | // It is its own event now, so what the user sees is the notice itself -- and until the |
| 332 | // browser handles `ev.type === 'compacted'` there is nothing to see at all. |
| 333 | whenRendered(/[Ff]olded \d+ earlier messages/.test(tail), |
| 334 | 'the user WAS shown that the conversation had been folded', |
| 335 | tail.slice(-200).replace(/\n/g, ' | ')); |
| 336 | check(!/context_compaction/.test(tail), |
| 337 | 'the fold no longer masquerades as a tool the model called'); |
| 338 | |
| 339 | // 502s are the local gateway proxy (/api) not running. A 400 is the deliberate |
| 340 | // refusal this whole audit is built around: the browser logs the failed fetch, and |
| 341 | // what matters is that the app recovered from it, which the checks above establish. |
| 342 | const errs = s.errs.filter(e => !/favicon|manifest|502|Bad Gateway|400 \(Bad Request\)/i.test(e)); |
| 343 | check(errs.length === 0, 'no unexpected console errors', errs.slice(0, 3).join(' | ')); |
| 344 | const four00 = s.errs.filter(e => /400 \(Bad Request\)/.test(e)); |
| 345 | check(four00.length === refused.length, |
| 346 | 'the browser saw exactly the refusals the provider issued and no more', |
| 347 | `${four00.length} console 400s against ${refused.length} refusals`); |
| 348 | |
| 349 | await shot(s, 'compact-final'); |
| 350 | await s.close(); |
| 351 | mock.kill(); |
| 352 | |
| 353 | log(`\n${failures === 0 ? 'ALL CHECKS PASSED' : failures + ' CHECK(S) FAILED'}` + |
| 354 | (pending ? ` (${pending} pending a browser-side edit)` : '')); |
| 355 | process.exit(failures === 0 ? 0 : 1); |