oxedyne/daimond/dev/verify_toolmemory.mjs
2.9 KiB, 1 run
created by r2519314175:747, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // Does the agent remember its own tool calls across turns? |
| 2 | // Turn 1 makes a tool call. Turn 2 asks a follow-up. What does the model see |
| 3 | // on turn 2 — the earlier assistant tool_call + tool result, or nothing? |
| 4 | // |
| 5 | // THE ROOT PATH IS LEFT ALONE HERE, DELIBERATELY. Since the chat fence landed on |
| 6 | // 2026-08-12 a chat is confined to `chats/<id>/work` (`scopeChatTo`, |
| 7 | // www/js/daimond.js), so `note.txt` at the workspace root is REFUSED by |
| 8 | // `Tool::guard` (src/tools.rs:5490) and nothing is written. Everywhere else in |
| 9 | // this suite that broke the check outright; here it does not, because what is |
| 10 | // asserted is the SHAPE of the conversation and not the file: a refusal is still a |
| 11 | // tool call, and it still comes back as a tool result. The turn below carries both |
| 12 | // either way, which is exactly the memory this file is about. Anything that starts |
| 13 | // asserting on the FILE has to move to the scratch first — see verify_writeguard. |
| 14 | import { open, chat, clearMockLog, mockLog } from './harness.mjs'; |
| 15 | |
| 16 | const ok = [], bad = []; |
| 17 | const check = (name, pass, detail) => { |
| 18 | (pass ? ok : bad).push(name + (detail ? ' — ' + detail : '')); |
| 19 | console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : '')); |
| 20 | }; |
| 21 | |
| 22 | clearMockLog(); |
| 23 | const s = await open({ name: 'toolmem' }); |
| 24 | |
| 25 | await chat(s, '@tool file_write {"path":"note.txt","content":"remember me"}'); |
| 26 | await chat(s, '@text What did you just write?'); // a second, separate turn |
| 27 | |
| 28 | const reqs = mockLog(); |
| 29 | // The model was asked TWICE, or there is no "turn 2" to be reading. A single |
| 30 | // request would satisfy every check below by being its own history. |
| 31 | check('two turns really reached the model', reqs.length >= 2, `${reqs.length} request(s)`); |
| 32 | const last = reqs[reqs.length - 1] || { messages: [] }; |
| 33 | console.log('turn-2 request carried', (last.messages || []).length, 'messages:'); |
| 34 | for (const m of last.messages || []) { |
| 35 | const role = m.role; |
| 36 | const hasToolCalls = !!(m.tool_calls && m.tool_calls.length); |
| 37 | const isToolResult = role === 'tool'; |
| 38 | const preview = typeof m.content === 'string' ? m.content.slice(0, 50) : ''; |
| 39 | console.log(` ${role}${hasToolCalls ? ' [+tool_calls]' : ''}${isToolResult ? ' [tool result]' : ''} ${preview}`); |
| 40 | } |
| 41 | const msgs = last.messages || []; |
| 42 | const sawToolCall = msgs.some(m => m.tool_calls && m.tool_calls.length); |
| 43 | const sawToolResult = msgs.some(m => m.role === 'tool'); |
| 44 | console.log('\nVERDICT: on turn 2 the model', |
| 45 | (sawToolCall && sawToolResult) ? 'REMEMBERS its turn-1 tool call+result (GOOD)' |
| 46 | : 'FORGOT its turn-1 tool call/result (BUG CONFIRMED)'); |
| 47 | check('on turn 2 the model is shown the tool call it made on turn 1', sawToolCall, |
| 48 | msgs.map(m => m.role).join(',')); |
| 49 | check('and the result that came back from it', sawToolResult, |
| 50 | msgs.map(m => m.role).join(',')); |
| 51 | |
| 52 | await s.close(); |
| 53 | console.log(`\n${ok.length} passed, ${bad.length} failed`); |
| 54 | if (bad.length) { bad.forEach(b => console.log(' FAILED: ' + b)); process.exit(1); } |