oxedyne/daimond/dev/verify_agentroute.mjs
13.0 KiB, 1 run
created by r2519314175:227, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // verify_agentroute.mjs — where agents are dispatched from, and whether two of |
| 2 | // them really run at once. |
| 3 | // |
| 4 | // Written against a real session. In an ORDINARY CHAT the user asked for two |
| 5 | // agents to be given the same list and each to sort it, saying afterwards |
| 6 | // *"this is a test of our ability to start and watch agents and have them |
| 7 | // complete their task"*. The chat did the sorting itself and then said: |
| 8 | // |
| 9 | // "There aren't two independent agents in this session to hand the list to; |
| 10 | // there's only me." |
| 11 | // "There's no way to spawn two independent agents in parallel, hand each the |
| 12 | // list, and watch their outputs come back concurrently." |
| 13 | // |
| 14 | // The first sentence is true of a chat. The second is false of the app, and it |
| 15 | // is the one the user reads. A Diamond's daimon holds `spawn_agent`, several |
| 16 | // calls in one turn become several workers, and the workers run at the same |
| 17 | // time — so the model reasoned from the one toolbelt it could see and reported |
| 18 | // the app as incapable. |
| 19 | // |
| 20 | // Four properties, and the first two are measured ON THE WIRE, in the request |
| 21 | // the provider actually received, because a claim about what an agent was |
| 22 | // offered cannot be read off the screen: |
| 23 | // |
| 24 | // 1. THE TOOLBELTS ARE WHAT THE CODE SAYS. A chat's request offers no |
| 25 | // `spawn_agent`; a daimon's request does. This pair is asserted together |
| 26 | // on purpose: `Tool::browser()` (src/tools.rs) is also the list the Tools |
| 27 | // panel shows a PERSON, and `runTurn` in www/js/daimond.js has no arm that |
| 28 | // turns a `spawn_agent` call into a worker — so a build that offered a chat |
| 29 | // the tool without wiring the dispatch would have it announce |
| 30 | // "Dispatched agent 'alpha'" (src/tools.rs, Tool::spawn_agent) with nothing |
| 31 | // whatsoever dispatched. That is a worse lie than the one being fixed. |
| 32 | // |
| 33 | // 2. THE CHAT KNOWS WHERE IT IS DONE. Its system prompt names the tool it |
| 34 | // lacks, names the Diamond as the surface that has it, says the workers run |
| 35 | // at the same time, and rules out the sentence above. Read from the wire |
| 36 | // rather than from the Rust constant, because what the model is told is the |
| 37 | // COMPOSED prompt and a user may have rewritten the body. |
| 38 | // |
| 39 | // 3. TWO DISPATCHED IN ONE TURN RUN AT THE SAME TIME. Not "two tiles exist" — |
| 40 | // the highest number of workers observed simultaneously in `running`, polled |
| 41 | // while the batch is in flight. `verify_agents.mjs` counts three cards and |
| 42 | // would pass just as happily against a pool that ran them one after another. |
| 43 | // |
| 44 | // 4. EACH REPORTS, AND REPORTS ITS OWN. The two tasks are `@long 21` and |
| 45 | // `@long 34`, so each worker's report ends at a different chunk: a report |
| 46 | // read off the wrong tile, or one tile read twice, is caught. Asserting |
| 47 | // "two non-empty reports" would not catch either. |
| 48 | // |
| 49 | // node dev/verify_agentroute.mjs |
| 50 | // node dev/verify_agentroute.mjs --break oldprompt # the chat's paragraph removed |
| 51 | // node dev/verify_agentroute.mjs --break serial # the worker pool narrowed to one |
| 52 | // |
| 53 | // The two break modes are how this file earns its keep. `oldprompt` puts the |
| 54 | // chat back on the text it shipped with before this was written, so check 2 must |
| 55 | // fail. `serial` sets `Workers.MAX` to 1, which makes the pump start the second |
| 56 | // worker only once the first has finished, so check 3 must fail while 4 still |
| 57 | // passes — a run that reports "concurrent" under `--break serial` is measuring |
| 58 | // nothing, and the number it prints is the proof either way. |
| 59 | // |
| 60 | // Needs dev/serve.mjs and dev/mockllm.mjs (dev/world.sh N --up gives both). |
| 61 | import { open, connectMock, steerDiamond, scratch, shot, chat, |
| 62 | mockLog, clearMockLog, contentText, errors } from './harness.mjs'; |
| 63 | |
| 64 | const BI = process.argv.indexOf('--break'); |
| 65 | const BEQ = process.argv.find(a => a.startsWith('--break=')); |
| 66 | const BREAK = BEQ ? BEQ.split('=')[1] : (BI >= 0 ? (process.argv[BI + 1] || '') : ''); |
| 67 | |
| 68 | let failures = 0; |
| 69 | const check = (cond, msg, detail) => { |
| 70 | console.log((cond ? ' ok ' : ' FAIL ') + msg + (detail != null ? ' — ' + detail : '')); |
| 71 | if (!cond) failures++; |
| 72 | }; |
| 73 | |
| 74 | /// The last request the mock received, or null. |
| 75 | const lastRequest = () => { |
| 76 | const rows = mockLog(); |
| 77 | return rows.length ? rows[rows.length - 1] : null; |
| 78 | }; |
| 79 | |
| 80 | /// The system prompt out of a logged request. |
| 81 | const systemOf = (row) => { |
| 82 | const m = (row && row.messages || []).find(x => x && x.role === 'system'); |
| 83 | return m ? contentText(m.content) : ''; |
| 84 | }; |
| 85 | |
| 86 | /// Make a Diamond through its own dialog, the way a person does. |
| 87 | async function create(p, name) { |
| 88 | await p.evaluate(() => document.getElementById('new-diamond-btn').click()); |
| 89 | await p.waitForSelector('.dlg-card', { timeout: 8000 }); |
| 90 | await p.evaluate((nm) => { |
| 91 | const card = [...document.querySelectorAll('.dlg-card')] |
| 92 | .filter(c => c.getClientRects().length).pop(); |
| 93 | const inp = card.querySelector('input.dlg-input'); |
| 94 | inp.value = nm; |
| 95 | inp.dispatchEvent(new Event('input', { bubbles: true })); |
| 96 | card.querySelector('.dlg-ok').click(); |
| 97 | }, name); |
| 98 | await p.waitForTimeout(1200); |
| 99 | } |
| 100 | |
| 101 | const s = await open({ name: 'agentroute', profile: scratch('pw', 'agentroute-' + process.pid) }); |
| 102 | const { page: p } = s; |
| 103 | try { |
| 104 | await connectMock(s); |
| 105 | if (BREAK) console.log(` .. running with --break ${BREAK}`); |
| 106 | |
| 107 | // ══ The chat: what it holds, and what it has been told ════════════ |
| 108 | // |
| 109 | // `oldprompt` is applied through the ordinary override, which is the same |
| 110 | // path a user's `prompts/chat.md` takes — so the break exercises the real |
| 111 | // mechanism rather than a hook that exists for this test. The text is the |
| 112 | // shipped default with the new paragraph cut off, i.e. exactly what the chat |
| 113 | // in the reported session was running on. |
| 114 | if (BREAK === 'oldprompt') { |
| 115 | await p.evaluate(() => { |
| 116 | const full = window.DaimondPrompts.defaultFor('chat') || ''; |
| 117 | const cut = full.indexOf('You can dispatch workers'); |
| 118 | window.DaimondPrompts.md.chat = cut > 0 ? full.slice(0, cut).trim() : full; |
| 119 | }); |
| 120 | } |
| 121 | |
| 122 | clearMockLog(); |
| 123 | await chat(s, '@text right'); |
| 124 | const chatReq = lastRequest(); |
| 125 | check(!!chatReq, 'the chat turn reached the provider', |
| 126 | chatReq ? String((chatReq.tools || []).length) + ' tools offered' : 'no request logged'); |
| 127 | |
| 128 | // A CHAT CAN NOW DISPATCH, and these three checks assert the reverse of what |
| 129 | // they first did. That is a decision, not a regression: an ordinary chat was |
| 130 | // refused the tool and told in its own prompt that dispatching "belongs to a |
| 131 | // Diamond", so a user who asked for two agents in a chat was told the app |
| 132 | // could not do it — on a surface where it now can. `Tool::browser` still omits |
| 133 | // `spawn_agent`, because a WORKER is built from that same list and a worker |
| 134 | // that could dispatch workers is a fan-out with no bottom; the chat is granted |
| 135 | // it afterwards, by the one caller that builds a chat. |
| 136 | const chatTools = (chatReq && chatReq.tools) || []; |
| 137 | check(chatTools.includes('spawn_agent'), |
| 138 | 'a chat is offered spawn_agent', |
| 139 | chatTools.length ? chatTools.join(', ') : '(none)'); |
| 140 | const wired = await p.evaluate(() => typeof window.DaimondWorkers === 'object'); |
| 141 | check(wired, 'and the dispatch machinery it reaches is present in the page'); |
| 142 | |
| 143 | const sys = systemOf(chatReq); |
| 144 | check(/spawn_agent/.test(sys), |
| 145 | 'the chat is told which tool it has', sys ? 'system prompt read' : 'no system prompt'); |
| 146 | check(/several times in the SAME turn|at once/.test(sys), |
| 147 | 'and that calling it twice in one turn runs two agents at once'); |
| 148 | check(/in parallel/.test(sys), |
| 149 | 'and is told not to say the app cannot run agents in parallel'); |
| 150 | |
| 151 | // ══ The Diamond: two agents in one turn ═══════════════════════════ |
| 152 | await create(p, 'Dispatch'); |
| 153 | |
| 154 | // The pool narrowed to one worker: the pump then starts the second only when |
| 155 | // the first has finished, which is what "no parallelism" would look like. |
| 156 | if (BREAK === 'serial') { |
| 157 | await p.evaluate(() => { window.DaimondWorkers.MAX = 1; }); |
| 158 | } |
| 159 | |
| 160 | clearMockLog(); |
| 161 | // Two `spawn_agent` calls in ONE turn — the shape the daimon's own prompt |
| 162 | // describes. Different chunk counts so each report is traceable to its task. |
| 163 | await steerDiamond(s, '@tools spawn_agent {"name":"alpha","task":"@long 21"}' |
| 164 | + ' ;; spawn_agent {"name":"beta","task":"@long 34"}'); |
| 165 | await p.waitForTimeout(1500); |
| 166 | |
| 167 | // The DAIMON's request, found by the directive it was sent — not the last row |
| 168 | // in the log. By the time the steer returns, the workers it started have |
| 169 | // already sent requests of their own, and a worker holds `Tool::browser()` |
| 170 | // just as a chat does: reading the last row reports the chat's 22 tools and |
| 171 | // calls it the daimon's belt. |
| 172 | const daimonRow = mockLog().find((row) => (row.messages || []) |
| 173 | .some(m => m && m.role === 'user' && /@tools spawn_agent/.test(contentText(m.content)))); |
| 174 | const daimonTools = (daimonRow || {}).tools || []; |
| 175 | check(daimonTools.includes('spawn_agent'), |
| 176 | 'a daimon IS offered spawn_agent, and a chat is the only surface without it', |
| 177 | daimonTools.join(', ') || '(the daimon request was not found)'); |
| 178 | |
| 179 | // Watched while it happens, which is what the user asked for. `running` is |
| 180 | // set in Workers.start and cleared when the turn ends, so the highest count |
| 181 | // seen is the number that were genuinely in flight together. |
| 182 | let peak = 0, saw = []; |
| 183 | const t0 = Date.now(); |
| 184 | while (Date.now() - t0 < 40000) { |
| 185 | const now = await p.evaluate(() => (window.DaimondWorkers.runs || []) |
| 186 | .map(r => ({ name: r.name, status: r.status }))); |
| 187 | const running = now.filter(r => r.status === 'running').length; |
| 188 | if (running > peak) peak = running; |
| 189 | saw = now; |
| 190 | const terminal = now.filter(r => ['done', 'error', 'stopped'].includes(r.status)).length; |
| 191 | if (now.length >= 2 && terminal >= 2) break; |
| 192 | await p.waitForTimeout(100); |
| 193 | } |
| 194 | await shot(s, 'agentroute-dispatched'); |
| 195 | |
| 196 | check(saw.length === 2, 'two workers were started', saw.map(r => r.name).join(', ') || '(none)'); |
| 197 | check(peak >= 2, 'and both were running at the same moment', |
| 198 | 'peak concurrent = ' + peak); |
| 199 | |
| 200 | // The same claim again, from OUTSIDE the app, because the poll above reads a |
| 201 | // status the page sets before it has sent anything: a build that marked both |
| 202 | // `running` and then queued the requests behind one another would satisfy it. |
| 203 | // |
| 204 | // `@long 21` streams 21 words at 120ms each, so the mock cannot have finished |
| 205 | // answering alpha until at least 2520ms after alpha's request arrived. If |
| 206 | // beta's request arrived inside that window, the two were on the wire together. |
| 207 | // Only arrival times are recorded, and that is enough for this direction. |
| 208 | const arrivals = {}; |
| 209 | for (const row of mockLog()) { |
| 210 | const us = (row.messages || []).filter(m => m && m.role === 'user'); |
| 211 | const last = us.length ? contentText(us[us.length - 1].content).trim() : ''; |
| 212 | if (/^@long 21\b/.test(last) && !arrivals.alpha) arrivals.alpha = Date.parse(row.at); |
| 213 | if (/^@long 34\b/.test(last) && !arrivals.beta) arrivals.beta = Date.parse(row.at); |
| 214 | } |
| 215 | const gap = (arrivals.alpha && arrivals.beta) |
| 216 | ? Math.abs(arrivals.beta - arrivals.alpha) : null; |
| 217 | check(gap !== null && gap < 21 * 120, |
| 218 | 'and the provider had both requests in hand at once', |
| 219 | gap === null ? 'one of the two never reached the mock' : gap + 'ms apart, alpha needs 2520ms'); |
| 220 | |
| 221 | const reports = await p.evaluate(() => (window.DaimondWorkers.runs || []) |
| 222 | .map(r => ({ name: r.name, status: r.status, text: r.text || '' }))); |
| 223 | const alpha = reports.find(r => r.name === 'alpha'); |
| 224 | const beta = reports.find(r => r.name === 'beta'); |
| 225 | check(!!alpha && alpha.status === 'done', 'alpha finished', alpha ? alpha.status : 'absent'); |
| 226 | check(!!beta && beta.status === 'done', 'beta finished', beta ? beta.status : 'absent'); |
| 227 | // Each report has to be the one its OWN task produced. Two reports that are |
| 228 | // merely non-empty would pass with both tiles showing the same worker's words. |
| 229 | check(!!alpha && /chunk-21\b/.test(alpha.text) && !/chunk-34\b/.test(alpha.text), |
| 230 | 'alpha reported its own task and only its own', |
| 231 | alpha ? alpha.text.slice(-24).trim() : 'absent'); |
| 232 | // `chunk-22` is the discriminator: only the 34-chunk task ever emits one, so a |
| 233 | // beta tile carrying alpha's words cannot satisfy this. |
| 234 | check(!!beta && /chunk-34\b/.test(beta.text) && /chunk-22\b/.test(beta.text), |
| 235 | 'beta reported its own task, which ran longer', |
| 236 | beta ? beta.text.slice(-24).trim() : 'absent'); |
| 237 | |
| 238 | // A world is the dev server and the mock, and deliberately no gateway -- it |
| 239 | // owns a gateway PORT (9700 + N) and starts nothing on it, so the browser-only |
| 240 | // tiers run without one and the page's account poll answers 502. That is the |
| 241 | // world being a world, not this flow going wrong, so it is named and skipped |
| 242 | // rather than left to fail every run and train the eye past the line. |
| 243 | const thrown = errors(s).filter(e => !/502 \(Bad Gateway\)/.test(e)); |
| 244 | check(thrown.length === 0, 'and nothing threw on the way', |
| 245 | thrown.slice(0, 2).join(' | ') || 'clean (gateway 502s aside: no gateway in a world)'); |
| 246 | |
| 247 | } catch (e) { |
| 248 | check(false, 'the run finished', String(e && e.message || e)); |
| 249 | try { await shot(s, 'agentroute-threw'); } catch {} |
| 250 | } finally { |
| 251 | await s.close(); |
| 252 | } |
| 253 | |
| 254 | console.log(failures === 0 |
| 255 | ? `\nverify_agentroute: all checks pass.` |
| 256 | : `\nverify_agentroute: ${failures} failed.`); |
| 257 | process.exit(failures === 0 ? 0 : 1); |