oxedyne/daimond/dev/verify_toolreload.mjs
12.0 KiB, 1 run
created by r2519314175:749, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // verify_toolreload.mjs — does the model still know what it did, after a reload? |
| 2 | // |
| 3 | // A page reload used to amputate the agent's memory of its own tool use. |
| 4 | // `DaimondApp::restore` rebuilt a session from the transcript on SCREEN, which |
| 5 | // carries prose and nothing else: the assistant turns came back bare |
| 6 | // (`tool_calls: Vec::new()`) and the tool results were dropped outright. So a |
| 7 | // reloaded daimon saw what it had SAID and had no record of what it had read, |
| 8 | // written or run. It re-read files it had already read, and could claim work it |
| 9 | // could no longer check. |
| 10 | // |
| 11 | // It could not have carried them. An assistant turn bearing `tool_calls` must be |
| 12 | // followed by a `tool` reply for each one or the provider rejects the whole |
| 13 | // request — and the browser never sees the provider's call ids. It mints its own |
| 14 | // (`t1`, `t2`) for drawing the thread, and those pair with nothing. |
| 15 | // |
| 16 | // The fix keeps the ids in Rust: `export_session` / `restore_session` carry the |
| 17 | // session's own message list across the boundary, ids and all. |
| 18 | // |
| 19 | // WHAT IS ASSERTED, and against what. Not the app's belief about what it sent — |
| 20 | // the MOCK PROVIDER'S OWN LOG of what actually arrived (DAIMOND_MOCK_LOG). The |
| 21 | // pairing rule is re-implemented here, independently of the Rust that repairs |
| 22 | // it, and applied to every request in the log rather than to the one this test |
| 23 | // is about: a fix that makes turn two legal by making turn three illegal has |
| 24 | // fixed nothing. |
| 25 | // |
| 26 | // Needs dev/serve.mjs (DAIMOND_PORT, default 8777) and dev/mockllm.mjs |
| 27 | // (DAIMOND_MOCK_PORT, default 9099). |
| 28 | import { open, chat, newChat, signInAs, clearMockLog, mockLog, errors } from './harness.mjs'; |
| 29 | |
| 30 | const ok = [], bad = []; |
| 31 | const check = (name, pass, detail) => { |
| 32 | (pass ? ok : bad).push(name + (detail ? ' — ' + detail : '')); |
| 33 | console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : '')); |
| 34 | }; |
| 35 | |
| 36 | // ── The provider's rule, written out here rather than trusted ────────────── |
| 37 | // |
| 38 | // An assistant message carrying `tool_calls` must be followed IMMEDIATELY by one |
| 39 | // `tool` message per call, each naming a call it made; and a `tool` message must |
| 40 | // answer a call in the assistant message before it. Anything else is a request an |
| 41 | // OpenAI-compatible provider rejects outright — not degrades, rejects. |
| 42 | // |
| 43 | // Returns the faults, so a failure says which message broke which half. |
| 44 | function faultsIn(messages) { |
| 45 | const faults = []; |
| 46 | const seen = new Set(); |
| 47 | for (let i = 0; i < messages.length; i++) { |
| 48 | const m = messages[i]; |
| 49 | if (m.role === 'assistant' && m.tool_calls && m.tool_calls.length) { |
| 50 | const want = m.tool_calls.map(tc => (tc.id != null ? String(tc.id) : '')); |
| 51 | for (const id of want) { |
| 52 | if (!id) faults.push(`msg ${i}: a tool_call with no id`); |
| 53 | if (seen.has(id)) faults.push(`msg ${i}: tool_call id ${id} used twice`); |
| 54 | seen.add(id); |
| 55 | } |
| 56 | const answered = []; |
| 57 | let j = i + 1; |
| 58 | while (j < messages.length && messages[j].role === 'tool') { |
| 59 | answered.push(String(messages[j].tool_call_id == null ? '' : messages[j].tool_call_id)); |
| 60 | j++; |
| 61 | } |
| 62 | for (const id of want) { |
| 63 | if (!answered.includes(id)) faults.push(`msg ${i}: tool_call ${id} was never answered`); |
| 64 | } |
| 65 | for (const id of answered) { |
| 66 | if (!want.includes(id)) faults.push(`msg ${i}: a tool reply answers ${id}, which was not asked for`); |
| 67 | } |
| 68 | i = j - 1; |
| 69 | } else if (m.role === 'tool') { |
| 70 | faults.push(`msg ${i}: a tool reply with no assistant turn asking for it`); |
| 71 | } |
| 72 | } |
| 73 | return faults; |
| 74 | } |
| 75 | |
| 76 | /// Reload the page for real, get back in, and open the first chat in the rail. |
| 77 | async function reloadAndOpen(s, name) { |
| 78 | await s.page.reload({ waitUntil: 'domcontentloaded' }); |
| 79 | await s.page.waitForTimeout(1200); |
| 80 | await signInAs(s, name); |
| 81 | await s.page.waitForTimeout(900); |
| 82 | const box = s.page.locator('#session-list .chat-box').first(); |
| 83 | await box.click({ force: true }); |
| 84 | await s.page.waitForSelector('#chat-input', { state: 'visible', timeout: 10000 }); |
| 85 | await s.page.waitForTimeout(400); |
| 86 | } |
| 87 | |
| 88 | const NAME = 'toolreload'; |
| 89 | const s = await open({ name: NAME }); |
| 90 | |
| 91 | // ── Turn one: read a file and write a file ───────────────────────────────── |
| 92 | // |
| 93 | // IN THE CHAT'S OWN SCRATCH. Until 2026-08-14 this named `reload-note.txt` at the |
| 94 | // workspace ROOT; since the chat fence landed on 2026-08-12 a chat is confined to |
| 95 | // `chats/<id>/work` (`scopeChatTo`, www/js/daimond.js) and `Tool::guard` |
| 96 | // (src/tools.rs:5490) refuses a root path before the write. Both turns then carried |
| 97 | // a refusal instead of a write and a read, and what the reload was asked to |
| 98 | // remember was an apology. |
| 99 | clearMockLog(); |
| 100 | await newChat(s); |
| 101 | const NOTE = await s.page.evaluate(() => { |
| 102 | const f = window.DaimondAttach.focus(); |
| 103 | return f && f.id ? window.DaimondAttach.chatScratch(f.id) + '/reload-note.txt' : ''; |
| 104 | }); |
| 105 | check('the chat has a scratch folder to work in', !!NOTE, NOTE || '(no chat in focus)'); |
| 106 | const wrote = await chat(s, `@tool file_write {"path":"${NOTE}","content":"the number is 4711"}`); |
| 107 | check('the file was really written, so there is a tool call worth remembering', |
| 108 | !/Refused/.test(wrote) && /Wrote \d+ bytes/.test(wrote), wrote.slice(-140).replace(/\n/g, ' | ')); |
| 109 | await chat(s, `@tool file_read {"path":"${NOTE}"}`); |
| 110 | |
| 111 | const before = mockLog(); |
| 112 | const beforeLast = before[before.length - 1] || { messages: [] }; |
| 113 | check('before the reload the model is shown its own tool calls', |
| 114 | beforeLast.messages.some(m => m.tool_calls && m.tool_calls.length), |
| 115 | `${beforeLast.messages.length} messages`); |
| 116 | |
| 117 | // What the browser stored as the model's own conversation. |
| 118 | const storedSession = await s.page.evaluate(async () => { |
| 119 | const rows = await new Promise((res) => { |
| 120 | const req = indexedDB.open('daimond-chats', 1); |
| 121 | req.onsuccess = () => { |
| 122 | const db = req.result; |
| 123 | const t = db.transaction('chats', 'readonly'); |
| 124 | const out = []; |
| 125 | const cur = t.objectStore('chats').openCursor(); |
| 126 | cur.onsuccess = () => { const c = cur.result; if (c) { out.push(c.value); c.continue(); } else res(out); }; |
| 127 | cur.onerror = () => res([]); |
| 128 | }; |
| 129 | req.onerror = () => res([]); |
| 130 | }); |
| 131 | const c = rows[0] || {}; |
| 132 | return c.session ? { n: (c.session.msgs || []).length, msgs: c.session.msgs } : null; |
| 133 | }); |
| 134 | check('the model\'s own conversation is stored, not only the screen transcript', |
| 135 | !!(storedSession && storedSession.n), storedSession ? `${storedSession.n} messages` : 'nothing stored'); |
| 136 | check('and it carries the provider\'s own call ids, which the screen cannot', |
| 137 | !!(storedSession && storedSession.msgs.some(m => m.role === 'assistant' && m.tool_calls && m.tool_calls.length |
| 138 | && String(m.tool_calls[0].id || '').length > 0)), |
| 139 | storedSession ? JSON.stringify((storedSession.msgs.find(m => m.tool_calls && m.tool_calls.length) || {}).tool_calls || null) : ''); |
| 140 | |
| 141 | // ── Reload, then ask a follow-up ─────────────────────────────────────────── |
| 142 | await reloadAndOpen(s, NAME); |
| 143 | clearMockLog(); |
| 144 | await chat(s, '@text What did you write, and what did you read?'); |
| 145 | |
| 146 | const after = mockLog(); |
| 147 | check('the reloaded chat sent something at all', after.length > 0, `${after.length} requests`); |
| 148 | const last = after[after.length - 1] || { messages: [] }; |
| 149 | |
| 150 | const sawCall = last.messages.some(m => m.tool_calls && m.tool_calls.length); |
| 151 | const sawResult = last.messages.some(m => m.role === 'tool'); |
| 152 | check('after a reload the request still carries the assistant turn that asked for a tool', sawCall); |
| 153 | check('and the tool replies that answered it', sawResult); |
| 154 | // In a TOOL REPLY, not merely somewhere in the request: the same number is in the |
| 155 | // directive the user typed, so a looser search would pass on prose alone — which |
| 156 | // is exactly the amputated state this is meant to catch. |
| 157 | check('and the result text itself, so the model can see what it read', |
| 158 | last.messages.some(m => m.role === 'tool' && String(m.content || '').includes('4711'))); |
| 159 | |
| 160 | // Every request in the log, not only the one this test is about. |
| 161 | let allFaults = []; |
| 162 | for (const req of after) { |
| 163 | const f = faultsIn(req.messages || []); |
| 164 | if (f.length) allFaults = allFaults.concat(f); |
| 165 | } |
| 166 | check('every request the provider was sent is whole by the pairing rule', |
| 167 | allFaults.length === 0, allFaults.slice(0, 4).join('; ')); |
| 168 | |
| 169 | // ── A second reload: it must still be durable, not merely survive once ───── |
| 170 | await reloadAndOpen(s, NAME); |
| 171 | clearMockLog(); |
| 172 | await chat(s, '@text And again?'); |
| 173 | const twice = mockLog(); |
| 174 | const lastTwice = twice[twice.length - 1] || { messages: [] }; |
| 175 | check('a second reload keeps it too', |
| 176 | lastTwice.messages.some(m => m.tool_calls && m.tool_calls.length) |
| 177 | && lastTwice.messages.some(m => m.role === 'tool')); |
| 178 | let twiceFaults = []; |
| 179 | for (const req of twice) twiceFaults = twiceFaults.concat(faultsIn(req.messages || [])); |
| 180 | check('and every request is still whole', twiceFaults.length === 0, twiceFaults.slice(0, 4).join('; ')); |
| 181 | |
| 182 | // ── Switching model mid-session drops the app; the memory must not go ────── |
| 183 | await s.page.evaluate(() => { |
| 184 | const sel = document.getElementById('chat-model-select'); |
| 185 | if (!sel) return; |
| 186 | const other = [...sel.options].map(o => o.value).find(v => v && v !== sel.value); |
| 187 | if (!other) return; |
| 188 | sel.value = other; |
| 189 | sel.dispatchEvent(new Event('change', { bubbles: true })); |
| 190 | }); |
| 191 | await s.page.waitForTimeout(500); |
| 192 | clearMockLog(); |
| 193 | await chat(s, '@text after the switch'); |
| 194 | const sw = mockLog(); |
| 195 | const lastSw = sw[sw.length - 1] || { messages: [] }; |
| 196 | check('a mid-session model switch keeps the tool history too', |
| 197 | lastSw.messages.some(m => m.tool_calls && m.tool_calls.length)); |
| 198 | let swFaults = []; |
| 199 | for (const req of sw) swFaults = swFaults.concat(faultsIn(req.messages || [])); |
| 200 | check('and its requests are whole', swFaults.length === 0, swFaults.slice(0, 4).join('; ')); |
| 201 | |
| 202 | // ── A store that has LOST a tool reply must cost that call, not every turn ── |
| 203 | // |
| 204 | // The store is merged across tabs, synced between devices and restored from |
| 205 | // backups, so a session list arriving with a reply missing is a thing that will |
| 206 | // happen. An assistant turn whose call is unanswered is a request the provider |
| 207 | // rejects WHOLE — so one lost reply from last Tuesday would take every turn after |
| 208 | // it, for ever. `pair_up` in src/wasm/app.rs repairs it on the way in; this is |
| 209 | // what proves the repair exists, by breaking the store on purpose. |
| 210 | const damage = await s.page.evaluate(() => new Promise((res) => { |
| 211 | const req = indexedDB.open('daimond-chats', 1); |
| 212 | req.onsuccess = () => { |
| 213 | const db = req.result; |
| 214 | const t = db.transaction('chats', 'readwrite'); |
| 215 | const st = t.objectStore('chats'); |
| 216 | const all = st.getAll(); |
| 217 | all.onsuccess = () => { |
| 218 | const rows = all.result || []; |
| 219 | let dropped = 0; |
| 220 | rows.forEach((c) => { |
| 221 | if (!c.session || !c.session.msgs) return; |
| 222 | const before = c.session.msgs.length; |
| 223 | // Take out the FIRST tool reply, leaving the call that asked for it. |
| 224 | const i = c.session.msgs.findIndex(m => m.role === 'tool'); |
| 225 | if (i >= 0) { c.session.msgs.splice(i, 1); dropped += before - c.session.msgs.length; } |
| 226 | st.put(c); |
| 227 | }); |
| 228 | t.oncomplete = () => res(dropped); |
| 229 | t.onerror = () => res(-1); |
| 230 | }; |
| 231 | all.onerror = () => res(-1); |
| 232 | }; |
| 233 | req.onerror = () => res(-1); |
| 234 | })); |
| 235 | check('a tool reply was taken out of the store, to see what a reload makes of it', |
| 236 | damage > 0, `${damage} removed`); |
| 237 | |
| 238 | await reloadAndOpen(s, NAME); |
| 239 | clearMockLog(); |
| 240 | await chat(s, '@text after the damage'); |
| 241 | const dmg = mockLog(); |
| 242 | let dmgFaults = []; |
| 243 | for (const req of dmg) dmgFaults = dmgFaults.concat(faultsIn(req.messages || [])); |
| 244 | check('a session missing a tool reply is repaired, not sent as it stands', |
| 245 | dmg.length > 0 && dmgFaults.length === 0, dmgFaults.slice(0, 4).join('; ') || `${dmg.length} requests`); |
| 246 | const dmgLast = dmg[dmg.length - 1] || { messages: [] }; |
| 247 | check('and the turn still happened rather than being refused', |
| 248 | (dmgLast.messages || []).length > 0); |
| 249 | |
| 250 | const errs = errors(s).filter(e => !/502|Bad Gateway/.test(e)); |
| 251 | check('nothing threw while all that happened', errs.length === 0, errs.slice(0, 2).join(' | ')); |
| 252 | |
| 253 | await s.close(); |
| 254 | console.log(`\n${ok.length} passed, ${bad.length} failed`); |
| 255 | if (bad.length) { bad.forEach(b => console.log(' FAILED: ' + b)); process.exit(1); } |