oxedyne/daimond/dev/verify_chatagents.mjs
11.4 KiB, 1 run
created by r2519314175:267, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // verify_chatagents.mjs — the user's own scenario, end to end. |
| 2 | // |
| 3 | // "From an ordinary chat: give two agents this list of 100 words and have |
| 4 | // each sort it." |
| 5 | // |
| 6 | // Four things have to be true, and the app got three of them wrong: |
| 7 | // |
| 8 | // 1. TWO AGENTS START, FROM A CHAT. An ordinary chat could not dispatch at |
| 9 | // all: `spawn_agent` was absent from its toolbelt and its prompt told it |
| 10 | // so in as many words. |
| 11 | // 2. THEY RUN AT THE SAME TIME. Asserted on the REQUESTS, at the mock — two |
| 12 | // workers whose provider calls overlap in time. A pair that merely both |
| 13 | // finished proves nothing: they could have run one after the other. |
| 14 | // 3. THEY ARE VISIBLE WHILE RUNNING. The Agents panel is where somebody |
| 15 | // watches work they cannot see happening, and half the requirement is the |
| 16 | // watching. |
| 17 | // 4. BOTH ANSWERS ARRIVE IN THE CHAT. This is the one that used to vanish |
| 18 | // silently: `gather` reported through `doSteer`, which writes a crystal, |
| 19 | // and it gave up at `currentDiamond.id !== b.diamondId` — false for every |
| 20 | // chat there has ever been. Two agents finished, and the person who asked |
| 21 | // got nothing. |
| 22 | // |
| 23 | // Asserted by MEANING, never by arity: each report is found by the WORDS THAT |
| 24 | // WORKER PRODUCED, so a transcript carrying one answer twice fails where a |
| 25 | // count of two would pass. |
| 26 | import { open, mockLog, clearMockLog, contentText, shot, errors } from './harness.mjs'; |
| 27 | |
| 28 | const ok = [], bad = []; |
| 29 | const check = (name, pass, detail) => { |
| 30 | (pass ? ok : bad).push(name + (detail ? ' — ' + detail : '')); |
| 31 | console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : '')); |
| 32 | }; |
| 33 | |
| 34 | const s = await open({ name: 'chatagents' }); |
| 35 | const { page } = s; |
| 36 | |
| 37 | try { |
| 38 | clearMockLog(); |
| 39 | |
| 40 | // A chat has to be open. A fresh profile shows the empty state, whose own |
| 41 | // button is the ordinary way in — driven rather than reached around, so this |
| 42 | // exercises the same path a person takes. |
| 43 | await page.evaluate(() => document.getElementById('new-session-btn').click()); |
| 44 | // A new chat starts PENDING and shows a Start button rather than a composer — |
| 45 | // nothing is spent until somebody asks for it. Pressed here, because that is |
| 46 | // what a person does, and because the composer does not exist until it is. |
| 47 | await page.waitForSelector('.pending-centre .empty-new-session', { timeout: 15000 }); |
| 48 | await page.evaluate(() => document.querySelector('.pending-centre .empty-new-session').click()); |
| 49 | await page.waitForSelector('#chat-input', { state: 'visible', timeout: 15000 }); |
| 50 | |
| 51 | // The panel first: a tile cannot be watched in a panel that is shut. |
| 52 | await page.evaluate(() => window.DaimondPanels.show('agents')); |
| 53 | |
| 54 | // One turn, two dispatches. The mock answers the directive with two |
| 55 | // `spawn_agent` calls in a single turn, which is exactly what a model does |
| 56 | // when it is asked for two agents. |
| 57 | await page.fill('#chat-input', |
| 58 | '@tools spawn_agent {"name":"sorter-a","task":"Say exactly: ALPHA-SORTED-LIST"} ' |
| 59 | + ';; spawn_agent {"name":"sorter-b","task":"Say exactly: BETA-SORTED-LIST"}'); |
| 60 | await page.keyboard.press('Enter'); |
| 61 | |
| 62 | // Watch WHILE they run, not after. Sampled from INSIDE the page on a fine |
| 63 | // interval, because against a mock provider a worker can be born and finished |
| 64 | // between two round-trips of the driver — and "no live tile was ever seen" |
| 65 | // would then be reported as the panel failing to draw one. The recorder keeps |
| 66 | // the high-water mark, which is the property: how many were on screen and |
| 67 | // live at once. |
| 68 | await page.evaluate(() => { |
| 69 | window.__watch = { maxLive: 0, sample: '' }; |
| 70 | window.__watchTimer = setInterval(() => { |
| 71 | const cards = [...document.querySelectorAll('#panel-agents .acard')].map(c => ({ |
| 72 | name: (c.querySelector('.an') || {}).textContent || '', |
| 73 | pill: (c.querySelector('.pill') || {}).textContent || '', |
| 74 | chip: (c.querySelector('.diamond-chip') || {}).textContent || '', |
| 75 | })); |
| 76 | const live = cards.filter(c => c.pill === 'running' || c.pill === 'queued'); |
| 77 | if (live.length > window.__watch.maxLive) { |
| 78 | window.__watch.maxLive = live.length; |
| 79 | window.__watch.sample = JSON.stringify(cards); |
| 80 | } |
| 81 | }, 30); |
| 82 | }); |
| 83 | |
| 84 | // Wait for both to finish and for the reports to be delivered. |
| 85 | await page.waitForFunction(() => { |
| 86 | const cards = [...document.querySelectorAll('#panel-agents .acard')]; |
| 87 | return cards.length >= 2 && cards.every(c => |
| 88 | ['done', 'error', 'stopped'].includes((c.querySelector('.pill') || {}).textContent || '')); |
| 89 | }, null, { timeout: 45000 }).catch(() => {}); |
| 90 | |
| 91 | const watch = await page.evaluate(() => { |
| 92 | clearInterval(window.__watchTimer); |
| 93 | return window.__watch; |
| 94 | }); |
| 95 | check('two agents dispatched from an ordinary chat are shown while they run', |
| 96 | watch.maxLive >= 2, watch.maxLive + ' live at once — ' + (watch.sample || 'no tiles appeared')); |
| 97 | await shot(s, 'chatagents-1-both-live'); |
| 98 | |
| 99 | const tiles = await page.evaluate(() => [...document.querySelectorAll('#panel-agents .acard')].map(c => ({ |
| 100 | name: (c.querySelector('.an') || {}).textContent || '', |
| 101 | pill: (c.querySelector('.pill') || {}).textContent || '', |
| 102 | chip: (c.querySelector('.diamond-chip') || {}).textContent || '', |
| 103 | fold: [...c.querySelectorAll('.abtn')].map(b => b.textContent).join('|'), |
| 104 | }))); |
| 105 | check('both finished', tiles.length === 2 && tiles.every(t => t.pill === 'done'), |
| 106 | JSON.stringify(tiles)); |
| 107 | |
| 108 | // The chip names the CONVERSATION. `agents.no_diamond` — "not from a Diamond" — |
| 109 | // is true and useless once two chats have agents running at once. |
| 110 | check('each tile says which chat sent it', |
| 111 | tiles.length === 2 && tiles.every(t => /↳/.test(t.chip) && !/^↳\s*$/.test(t.chip)), |
| 112 | JSON.stringify(tiles.map(t => t.chip))); |
| 113 | |
| 114 | // Fold in belongs to a crystal, and a chat has none. Offering it would open a |
| 115 | // dialog that says a Diamond is gone — about a Diamond that never existed. |
| 116 | check('and is not offered a fold into a Diamond it never had', |
| 117 | tiles.every(t => !/Fold in/.test(t.fold)), JSON.stringify(tiles.map(t => t.fold))); |
| 118 | |
| 119 | // ── Concurrency, at the wire ── |
| 120 | // Two worker requests whose spans overlap. The mock records when each request |
| 121 | // arrived; workers are network-bound, so overlap is the only honest evidence |
| 122 | // that they ran together rather than in turn. |
| 123 | const log = mockLog(); |
| 124 | const workerReqs = log.filter(e => { |
| 125 | const msgs = e.messages || []; |
| 126 | // A WORKER's request, not the chat's: the task is its user message, and the |
| 127 | // chat's own turn carries the directive that asked for it instead. |
| 128 | return msgs.some(m => m.role === 'user' && /SORTED-LIST/.test(contentText(m.content)) |
| 129 | && !/^@tools/.test(contentText(m.content))); |
| 130 | }); |
| 131 | const stamps = workerReqs.map(e => Date.parse(e.at || '')).filter(n => !isNaN(n)).sort((a, b) => a - b); |
| 132 | const gap = stamps.length >= 2 ? stamps[stamps.length - 1] - stamps[0] : Infinity; |
| 133 | check('the two workers issued their provider requests together', |
| 134 | workerReqs.length >= 2 && gap < 2000, |
| 135 | workerReqs.length + ' worker request(s), ' + gap + 'ms apart'); |
| 136 | |
| 137 | // ── Both answers in the chat ── |
| 138 | // By the words each worker produced. A transcript holding one answer twice |
| 139 | // fails this, where "two messages arrived" would pass. |
| 140 | const transcript = await page.evaluate(() => { |
| 141 | const el = document.getElementById('chat-output'); |
| 142 | return el ? el.textContent : ''; |
| 143 | }); |
| 144 | check('the first agent\'s answer is in the chat', /ALPHA-SORTED-LIST/.test(transcript)); |
| 145 | check('the second agent\'s answer is in the chat', /BETA-SORTED-LIST/.test(transcript)); |
| 146 | |
| 147 | // ── A fan-out the app decides NOT to start ────────────────────────── |
| 148 | // |
| 149 | // The chat's half of the property `dev/verify_gather.mjs` asserts for a daimon. |
| 150 | // `spawn_agent` tells the model its workers begin when the turn ends; the spend |
| 151 | // gate is asked AFTER the turn has ended, and a chat whose fan-out it declines |
| 152 | // used to get a line of red text on screen while the MODEL was told nothing. Read |
| 153 | // off the wire, because what matters is what the model is sent. |
| 154 | await page.evaluate(() => window.DaimondGovernor.observe({ t: Date.now(), u: 9 })); |
| 155 | clearMockLog(); |
| 156 | await page.fill('#chat-input', |
| 157 | '@tools spawn_agent {"name":"sorter-c","task":"Say exactly: GAMMA-SORTED-LIST"} ' |
| 158 | + ';; spawn_agent {"name":"sorter-d","task":"Say exactly: DELTA-SORTED-LIST"}'); |
| 159 | await page.keyboard.press('Enter'); |
| 160 | await page.waitForSelector('.dlg-card .dlg-cancel', { timeout: 20000 }); |
| 161 | await page.click('.dlg-card .dlg-cancel'); |
| 162 | await page.waitForTimeout(2500); |
| 163 | const declined = mockLog(); |
| 164 | check('a declined fan-out from a chat starts no worker', |
| 165 | !declined.some(m => { |
| 166 | const j = JSON.stringify(m.messages || []); |
| 167 | return j.includes('GAMMA-SORTED-LIST') && !j.includes('spawn_agent'); |
| 168 | }), `${declined.length} request(s)`); |
| 169 | |
| 170 | clearMockLog(); |
| 171 | await page.fill('#chat-input', 'carry on then'); |
| 172 | await page.keyboard.press('Enter'); |
| 173 | await page.waitForTimeout(5000); |
| 174 | const next = mockLog(); |
| 175 | check('and the chat\'s own agent is told they were NOT started', |
| 176 | next.some(m => JSON.stringify(m.messages || []).includes('WERE NOT STARTED')), |
| 177 | `${next.length} request(s) since`); |
| 178 | check('and told why, and that nothing was spent', |
| 179 | next.some(m => { |
| 180 | const j = JSON.stringify(m.messages || []); |
| 181 | return j.includes('told no') && j.includes('nothing was spent'); |
| 182 | })); |
| 183 | |
| 184 | // And it survives a reload, because it was written to the record rather than |
| 185 | // only painted. A report that lives in the DOM is a report a refresh loses. |
| 186 | await page.reload(); |
| 187 | await page.waitForFunction(() => !!window.DaimondCore, null, { timeout: 15000 }).catch(() => {}); |
| 188 | await page.waitForTimeout(1500); |
| 189 | // Asked of the STORE, not of the screen. A reload lands on whatever the app |
| 190 | // chooses to show, so reading the DOM would be testing which chat got selected; |
| 191 | // the property is that the reports were WRITTEN, and the store is where that is |
| 192 | // true or false. |
| 193 | // Read from the STORE the app itself reads — IndexedDB, namespaced per account. |
| 194 | // Reading the DOM instead would be testing which chat the app chose to select |
| 195 | // after a reload; the property is that the reports were WRITTEN. |
| 196 | const after = await page.evaluate(async () => { |
| 197 | const dbs = await indexedDB.databases(); |
| 198 | const name = (dbs.find(d => /chat/i.test(d.name || '')) || {}).name; |
| 199 | if (!name) return ''; |
| 200 | const db = await new Promise((res, rej) => { |
| 201 | const r = indexedDB.open(name); |
| 202 | r.onsuccess = () => res(r.result); r.onerror = () => rej(r.error); |
| 203 | }); |
| 204 | const store = db.objectStoreNames[0]; |
| 205 | const rows = await new Promise((res) => { |
| 206 | const r = db.transaction(store, 'readonly').objectStore(store).getAll(); |
| 207 | r.onsuccess = () => res(r.result || []); r.onerror = () => res([]); |
| 208 | }); |
| 209 | return JSON.stringify(rows.map(c => (c.messages || []).map(m => m.content).join(' '))); |
| 210 | }); |
| 211 | check('and both are still there after a reload', |
| 212 | /ALPHA-SORTED-LIST/.test(after) && /BETA-SORTED-LIST/.test(after), |
| 213 | after.slice(0, 160)); |
| 214 | await shot(s, 'chatagents-2-reports'); |
| 215 | |
| 216 | // The gateway is not part of a world (see dev/world.sh), so its 502s are the |
| 217 | // fixture and not the app. Filtered by what they ARE rather than by a count, |
| 218 | // so a real error appearing alongside them is still caught. |
| 219 | const errs = errors(s).filter(e => !/502|Bad Gateway|api\/(gw|account)/i.test(String(e))); |
| 220 | check('no console errors during the run', errs.length === 0, JSON.stringify(errs).slice(0, 200)); |
| 221 | } catch (e) { |
| 222 | check('no exception during the run', false, String(e && e.message || e)); |
| 223 | } finally { |
| 224 | try { await s.browser.close(); } catch (e) { /* ignore */ } |
| 225 | } |
| 226 | |
| 227 | console.log('\n' + ok.length + ' ok, ' + bad.length + ' failed'); |
| 228 | process.exit(bad.length ? 1 : 0); |