Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/verify_agentroute.mjs

13.0 KiB, 1 run

created by r2519314175:227, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1// verify_agentroute.mjs — where agents are dispatched from, and whether two of
2// them really run at once.
3//
4// Written against a real session. In an ORDINARY CHAT the user asked for two
5// agents to be given the same list and each to sort it, saying afterwards
6// *"this is a test of our ability to start and watch agents and have them
7// complete their task"*. The chat did the sorting itself and then said:
8//
9// "There aren't two independent agents in this session to hand the list to;
10// there's only me."
11// "There's no way to spawn two independent agents in parallel, hand each the
12// list, and watch their outputs come back concurrently."
13//
14// The first sentence is true of a chat. The second is false of the app, and it
15// is the one the user reads. A Diamond's daimon holds `spawn_agent`, several
16// calls in one turn become several workers, and the workers run at the same
17// time — so the model reasoned from the one toolbelt it could see and reported
18// the app as incapable.
19//
20// Four properties, and the first two are measured ON THE WIRE, in the request
21// the provider actually received, because a claim about what an agent was
22// offered cannot be read off the screen:
23//
24// 1. THE TOOLBELTS ARE WHAT THE CODE SAYS. A chat's request offers no
25// `spawn_agent`; a daimon's request does. This pair is asserted together
26// on purpose: `Tool::browser()` (src/tools.rs) is also the list the Tools
27// panel shows a PERSON, and `runTurn` in www/js/daimond.js has no arm that
28// turns a `spawn_agent` call into a worker — so a build that offered a chat
29// the tool without wiring the dispatch would have it announce
30// "Dispatched agent 'alpha'" (src/tools.rs, Tool::spawn_agent) with nothing
31// whatsoever dispatched. That is a worse lie than the one being fixed.
32//
33// 2. THE CHAT KNOWS WHERE IT IS DONE. Its system prompt names the tool it
34// lacks, names the Diamond as the surface that has it, says the workers run
35// at the same time, and rules out the sentence above. Read from the wire
36// rather than from the Rust constant, because what the model is told is the
37// COMPOSED prompt and a user may have rewritten the body.
38//
39// 3. TWO DISPATCHED IN ONE TURN RUN AT THE SAME TIME. Not "two tiles exist" —
40// the highest number of workers observed simultaneously in `running`, polled
41// while the batch is in flight. `verify_agents.mjs` counts three cards and
42// would pass just as happily against a pool that ran them one after another.
43//
44// 4. EACH REPORTS, AND REPORTS ITS OWN. The two tasks are `@long 21` and
45// `@long 34`, so each worker's report ends at a different chunk: a report
46// read off the wrong tile, or one tile read twice, is caught. Asserting
47// "two non-empty reports" would not catch either.
48//
49// node dev/verify_agentroute.mjs
50// node dev/verify_agentroute.mjs --break oldprompt # the chat's paragraph removed
51// node dev/verify_agentroute.mjs --break serial # the worker pool narrowed to one
52//
53// The two break modes are how this file earns its keep. `oldprompt` puts the
54// chat back on the text it shipped with before this was written, so check 2 must
55// fail. `serial` sets `Workers.MAX` to 1, which makes the pump start the second
56// worker only once the first has finished, so check 3 must fail while 4 still
57// passes — a run that reports "concurrent" under `--break serial` is measuring
58// nothing, and the number it prints is the proof either way.
59//
60// Needs dev/serve.mjs and dev/mockllm.mjs (dev/world.sh N --up gives both).
61import { open, connectMock, steerDiamond, scratch, shot, chat,
62 mockLog, clearMockLog, contentText, errors } from './harness.mjs';
63
64const BI = process.argv.indexOf('--break');
65const BEQ = process.argv.find(a => a.startsWith('--break='));
66const BREAK = BEQ ? BEQ.split('=')[1] : (BI >= 0 ? (process.argv[BI + 1] || '') : '');
67
68let failures = 0;
69const check = (cond, msg, detail) => {
70 console.log((cond ? ' ok ' : ' FAIL ') + msg + (detail != null ? ' — ' + detail : ''));
71 if (!cond) failures++;
72};
73
74/// The last request the mock received, or null.
75const lastRequest = () => {
76 const rows = mockLog();
77 return rows.length ? rows[rows.length - 1] : null;
78};
79
80/// The system prompt out of a logged request.
81const systemOf = (row) => {
82 const m = (row && row.messages || []).find(x => x && x.role === 'system');
83 return m ? contentText(m.content) : '';
84};
85
86/// Make a Diamond through its own dialog, the way a person does.
87async function create(p, name) {
88 await p.evaluate(() => document.getElementById('new-diamond-btn').click());
89 await p.waitForSelector('.dlg-card', { timeout: 8000 });
90 await p.evaluate((nm) => {
91 const card = [...document.querySelectorAll('.dlg-card')]
92 .filter(c => c.getClientRects().length).pop();
93 const inp = card.querySelector('input.dlg-input');
94 inp.value = nm;
95 inp.dispatchEvent(new Event('input', { bubbles: true }));
96 card.querySelector('.dlg-ok').click();
97 }, name);
98 await p.waitForTimeout(1200);
99}
100
101const s = await open({ name: 'agentroute', profile: scratch('pw', 'agentroute-' + process.pid) });
102const { page: p } = s;
103try {
104 await connectMock(s);
105 if (BREAK) console.log(` .. running with --break ${BREAK}`);
106
107 // ══ The chat: what it holds, and what it has been told ════════════
108 //
109 // `oldprompt` is applied through the ordinary override, which is the same
110 // path a user's `prompts/chat.md` takes — so the break exercises the real
111 // mechanism rather than a hook that exists for this test. The text is the
112 // shipped default with the new paragraph cut off, i.e. exactly what the chat
113 // in the reported session was running on.
114 if (BREAK === 'oldprompt') {
115 await p.evaluate(() => {
116 const full = window.DaimondPrompts.defaultFor('chat') || '';
117 const cut = full.indexOf('You can dispatch workers');
118 window.DaimondPrompts.md.chat = cut > 0 ? full.slice(0, cut).trim() : full;
119 });
120 }
121
122 clearMockLog();
123 await chat(s, '@text right');
124 const chatReq = lastRequest();
125 check(!!chatReq, 'the chat turn reached the provider',
126 chatReq ? String((chatReq.tools || []).length) + ' tools offered' : 'no request logged');
127
128 // A CHAT CAN NOW DISPATCH, and these three checks assert the reverse of what
129 // they first did. That is a decision, not a regression: an ordinary chat was
130 // refused the tool and told in its own prompt that dispatching "belongs to a
131 // Diamond", so a user who asked for two agents in a chat was told the app
132 // could not do it — on a surface where it now can. `Tool::browser` still omits
133 // `spawn_agent`, because a WORKER is built from that same list and a worker
134 // that could dispatch workers is a fan-out with no bottom; the chat is granted
135 // it afterwards, by the one caller that builds a chat.
136 const chatTools = (chatReq && chatReq.tools) || [];
137 check(chatTools.includes('spawn_agent'),
138 'a chat is offered spawn_agent',
139 chatTools.length ? chatTools.join(', ') : '(none)');
140 const wired = await p.evaluate(() => typeof window.DaimondWorkers === 'object');
141 check(wired, 'and the dispatch machinery it reaches is present in the page');
142
143 const sys = systemOf(chatReq);
144 check(/spawn_agent/.test(sys),
145 'the chat is told which tool it has', sys ? 'system prompt read' : 'no system prompt');
146 check(/several times in the SAME turn|at once/.test(sys),
147 'and that calling it twice in one turn runs two agents at once');
148 check(/in parallel/.test(sys),
149 'and is told not to say the app cannot run agents in parallel');
150
151 // ══ The Diamond: two agents in one turn ═══════════════════════════
152 await create(p, 'Dispatch');
153
154 // The pool narrowed to one worker: the pump then starts the second only when
155 // the first has finished, which is what "no parallelism" would look like.
156 if (BREAK === 'serial') {
157 await p.evaluate(() => { window.DaimondWorkers.MAX = 1; });
158 }
159
160 clearMockLog();
161 // Two `spawn_agent` calls in ONE turn — the shape the daimon's own prompt
162 // describes. Different chunk counts so each report is traceable to its task.
163 await steerDiamond(s, '@tools spawn_agent {"name":"alpha","task":"@long 21"}'
164 + ' ;; spawn_agent {"name":"beta","task":"@long 34"}');
165 await p.waitForTimeout(1500);
166
167 // The DAIMON's request, found by the directive it was sent — not the last row
168 // in the log. By the time the steer returns, the workers it started have
169 // already sent requests of their own, and a worker holds `Tool::browser()`
170 // just as a chat does: reading the last row reports the chat's 22 tools and
171 // calls it the daimon's belt.
172 const daimonRow = mockLog().find((row) => (row.messages || [])
173 .some(m => m && m.role === 'user' && /@tools spawn_agent/.test(contentText(m.content))));
174 const daimonTools = (daimonRow || {}).tools || [];
175 check(daimonTools.includes('spawn_agent'),
176 'a daimon IS offered spawn_agent, and a chat is the only surface without it',
177 daimonTools.join(', ') || '(the daimon request was not found)');
178
179 // Watched while it happens, which is what the user asked for. `running` is
180 // set in Workers.start and cleared when the turn ends, so the highest count
181 // seen is the number that were genuinely in flight together.
182 let peak = 0, saw = [];
183 const t0 = Date.now();
184 while (Date.now() - t0 < 40000) {
185 const now = await p.evaluate(() => (window.DaimondWorkers.runs || [])
186 .map(r => ({ name: r.name, status: r.status })));
187 const running = now.filter(r => r.status === 'running').length;
188 if (running > peak) peak = running;
189 saw = now;
190 const terminal = now.filter(r => ['done', 'error', 'stopped'].includes(r.status)).length;
191 if (now.length >= 2 && terminal >= 2) break;
192 await p.waitForTimeout(100);
193 }
194 await shot(s, 'agentroute-dispatched');
195
196 check(saw.length === 2, 'two workers were started', saw.map(r => r.name).join(', ') || '(none)');
197 check(peak >= 2, 'and both were running at the same moment',
198 'peak concurrent = ' + peak);
199
200 // The same claim again, from OUTSIDE the app, because the poll above reads a
201 // status the page sets before it has sent anything: a build that marked both
202 // `running` and then queued the requests behind one another would satisfy it.
203 //
204 // `@long 21` streams 21 words at 120ms each, so the mock cannot have finished
205 // answering alpha until at least 2520ms after alpha's request arrived. If
206 // beta's request arrived inside that window, the two were on the wire together.
207 // Only arrival times are recorded, and that is enough for this direction.
208 const arrivals = {};
209 for (const row of mockLog()) {
210 const us = (row.messages || []).filter(m => m && m.role === 'user');
211 const last = us.length ? contentText(us[us.length - 1].content).trim() : '';
212 if (/^@long 21\b/.test(last) && !arrivals.alpha) arrivals.alpha = Date.parse(row.at);
213 if (/^@long 34\b/.test(last) && !arrivals.beta) arrivals.beta = Date.parse(row.at);
214 }
215 const gap = (arrivals.alpha && arrivals.beta)
216 ? Math.abs(arrivals.beta - arrivals.alpha) : null;
217 check(gap !== null && gap < 21 * 120,
218 'and the provider had both requests in hand at once',
219 gap === null ? 'one of the two never reached the mock' : gap + 'ms apart, alpha needs 2520ms');
220
221 const reports = await p.evaluate(() => (window.DaimondWorkers.runs || [])
222 .map(r => ({ name: r.name, status: r.status, text: r.text || '' })));
223 const alpha = reports.find(r => r.name === 'alpha');
224 const beta = reports.find(r => r.name === 'beta');
225 check(!!alpha && alpha.status === 'done', 'alpha finished', alpha ? alpha.status : 'absent');
226 check(!!beta && beta.status === 'done', 'beta finished', beta ? beta.status : 'absent');
227 // Each report has to be the one its OWN task produced. Two reports that are
228 // merely non-empty would pass with both tiles showing the same worker's words.
229 check(!!alpha && /chunk-21\b/.test(alpha.text) && !/chunk-34\b/.test(alpha.text),
230 'alpha reported its own task and only its own',
231 alpha ? alpha.text.slice(-24).trim() : 'absent');
232 // `chunk-22` is the discriminator: only the 34-chunk task ever emits one, so a
233 // beta tile carrying alpha's words cannot satisfy this.
234 check(!!beta && /chunk-34\b/.test(beta.text) && /chunk-22\b/.test(beta.text),
235 'beta reported its own task, which ran longer',
236 beta ? beta.text.slice(-24).trim() : 'absent');
237
238 // A world is the dev server and the mock, and deliberately no gateway -- it
239 // owns a gateway PORT (9700 + N) and starts nothing on it, so the browser-only
240 // tiers run without one and the page's account poll answers 502. That is the
241 // world being a world, not this flow going wrong, so it is named and skipped
242 // rather than left to fail every run and train the eye past the line.
243 const thrown = errors(s).filter(e => !/502 \(Bad Gateway\)/.test(e));
244 check(thrown.length === 0, 'and nothing threw on the way',
245 thrown.slice(0, 2).join(' | ') || 'clean (gateway 502s aside: no gateway in a world)');
246
247} catch (e) {
248 check(false, 'the run finished', String(e && e.message || e));
249 try { await shot(s, 'agentroute-threw'); } catch {}
250} finally {
251 await s.close();
252}
253
254console.log(failures === 0
255 ? `\nverify_agentroute: all checks pass.`
256 : `\nverify_agentroute: ${failures} failed.`);
257process.exit(failures === 0 ? 0 : 1);