oxedyne/daimond/dev/verify_toolbatch.mjs
12.1 KiB, 1 run
created by r2519314175:745, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // verify_toolbatch.mjs — a round of reads takes ONE read's time, and still answers |
| 2 | // in the order the model asked. |
| 3 | // |
| 4 | // Daimond ran a round's tool calls one after another: `for tc in &resp.tool_calls` |
| 5 | // in src/agent.rs awaited each call before starting the next, so a round's wall |
| 6 | // clock was the SUM of its calls where it could be the MAXIMUM. A frontier model |
| 7 | // routinely emits several tool-use blocks in one reply, and doing so is a large |
| 8 | // part of why one feels fast. |
| 9 | // |
| 10 | // WHAT MAY RUN TOGETHER IS NOT "ANYTHING": src/batch.rs holds the rule and the |
| 11 | // reasons, and the short version is that only a call which reads, asks nothing and |
| 12 | // changes nothing may share a batch. Everything else is a batch of one, exactly as |
| 13 | // before. That is why §3 exists — a check that only measured speed would go green |
| 14 | // on a build that batched the writes too. |
| 15 | // |
| 16 | // §1 a round of reads answers all of them, in the order the model gave |
| 17 | // §2 the round takes less than a queue of the same calls |
| 18 | // §3 a write among the reads keeps its place |
| 19 | // §4 the page still sees one call announced, then its result, then the next |
| 20 | // |
| 21 | // §4 is the one that looks like tidiness and is not. `www/js/daimond.js` keeps ONE |
| 22 | // `pendingTool` and ONE `pendingCallId` — in the chat, in the worker dock and in |
| 23 | // the daimon alike — and files each result against whichever call was announced |
| 24 | // last. Announce a batch up front and the first result is filed under the last |
| 25 | // call while every result after it is dropped, out of the transcript and out of |
| 26 | // the write-ahead journal both. The engine therefore keeps the wire strictly |
| 27 | // alternating and a batch is invisible on it. |
| 28 | // |
| 29 | // ── HOW THE TIMING IS MADE HONEST ──────────────────────────────────────────── |
| 30 | // |
| 31 | // A figure from one engine is not a measurement of a change; it is a measurement |
| 32 | // of a machine. So `--against <dir>` loads a SECOND engine into the same page and |
| 33 | // runs the same rounds on both, alternating, and reports both medians. The two |
| 34 | // numbers then differ by the change and by nothing else — same browser, same OPFS, |
| 35 | // same fixture, same minute. `dev/breakproof_toolbatch.sh` builds that second |
| 36 | // engine, with `batch::may_run_beside` answering no to everything, which is the |
| 37 | // loop exactly as it stood. |
| 38 | // |
| 39 | // IT MUST BE A RELEASE BUILD. `wasm-pack build --dev` is three or four times |
| 40 | // slower at everything, so a dev-built comparison engine would report the change |
| 41 | // as an enormous win and the number would be about the optimiser. |
| 42 | // |
| 43 | // node dev/verify_toolbatch.mjs |
| 44 | // node dev/verify_toolbatch.mjs --against pkg-serial |
| 45 | import path from 'node:path'; |
| 46 | import { fileURLToPath } from 'node:url'; |
| 47 | import { open as openApp, MOCK } from './harness.mjs'; |
| 48 | import { whyStaleWasm, refuse } from './staleguard.mjs'; |
| 49 | |
| 50 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 51 | const ROOT = path.join(HERE, '..'); |
| 52 | |
| 53 | const ok = [], bad = []; |
| 54 | const check = (name, pass, detail) => { |
| 55 | (pass ? ok : bad).push(name); |
| 56 | console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : '')); |
| 57 | }; |
| 58 | const note = (s) => console.log(' · ' + s); |
| 59 | |
| 60 | const argOf = (flag, fallback) => { |
| 61 | const i = process.argv.indexOf(flag); |
| 62 | return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback; |
| 63 | }; |
| 64 | /// Which directory under `www/` the page imports the engine under test from. |
| 65 | const ENGINE = argOf('--engine', 'pkg'); |
| 66 | /// A second engine to run the same rounds against, or none. |
| 67 | const AGAINST = argOf('--against', ''); |
| 68 | |
| 69 | // Six, because that is what a model asking for a batch of reads actually asks for, |
| 70 | // and because it is under the engine's own cap of eight — a round split into two |
| 71 | // batches would be measuring the cap rather than the batching. |
| 72 | const READS = 6; |
| 73 | // Big enough that a read is a real read, small enough that six stay inside the |
| 74 | // turn's output budget: past it the results are cut, and the two engines would then |
| 75 | // be doing different amounts of work. |
| 76 | const BYTES = 8 * 1024; |
| 77 | // A tree for the searches to walk. A search is the slowest thing in the batch rule |
| 78 | // and the one a model most often asks for several of at once. |
| 79 | const TREE = 300; |
| 80 | const ROUNDS = 5; |
| 81 | |
| 82 | const readArgs = (n) => JSON.stringify({ path: `src/file${n}.txt` }); |
| 83 | const searchArgs = (n) => JSON.stringify({ path: 'tree', query: `needle${n % 6}`, fixed: true }); |
| 84 | |
| 85 | /// The prompt that asks for a round of `n` calls of one tool. |
| 86 | const roundOf = (tool, args, n) => |
| 87 | '@tools ' + Array.from({ length: n }, (_, i) => `${tool} ${args(i)}`).join(' ;; '); |
| 88 | |
| 89 | // The whole of this file is a claim about the ENGINE, and it drives the engine |
| 90 | // module directly rather than through the page -- so a stale bundle would not show |
| 91 | // up as a broken app, it would show up as yesterday's dispatch reporting today's |
| 92 | // timing. |
| 93 | refuse(whyStaleWasm(path.join(ROOT, `www/${ENGINE}/oxedyne_daimond_bg.wasm`), |
| 94 | path.join(ROOT, 'src'), { |
| 95 | subject: 'What the round dispatches together', |
| 96 | holds: '`agent::batch` and the loop that uses it', |
| 97 | })); |
| 98 | |
| 99 | async function main() { |
| 100 | const s = await openApp({ name: 'toolbatch' }); |
| 101 | const page = s.page; |
| 102 | try { |
| 103 | if (ENGINE !== 'pkg') { |
| 104 | console.log(`\n !!!! the engine under test is www/${ENGINE}, not the app's own bundle.\n`); |
| 105 | } |
| 106 | // The engine module, whichever package it comes from. `www/pkg` is the one |
| 107 | // the app booted, so it is already initialised; any other package has never |
| 108 | // been, and wasm-bindgen's exports are inert until it is. |
| 109 | await page.evaluate(() => { |
| 110 | const done = {}; |
| 111 | window.__mod = async (name) => { |
| 112 | const mod = await import(`../${name}/oxedyne_daimond.js`); |
| 113 | if (name !== 'pkg' && !done[name]) { await mod.default(); done[name] = true; } |
| 114 | return mod; |
| 115 | }; |
| 116 | window.__app = async (engine) => { |
| 117 | const mod = await window.__mod(engine); |
| 118 | const app = new mod.DaimondApp(window.__mock, 'k', 'mock/fast', 4096, '', true); |
| 119 | app.set_max_rounds(3); |
| 120 | return app; |
| 121 | }; |
| 122 | // The span of the ROUND, from the first call being announced to the last |
| 123 | // result landing -- not of the turn, which also carries two provider |
| 124 | // requests that have nothing to do with what is being measured. |
| 125 | window.__round = async (app, prompt) => { |
| 126 | const ev = []; |
| 127 | const began = performance.now(); |
| 128 | await app.run_turn(prompt, (e) => ev.push({ |
| 129 | type: e.type, name: e.name, content: e.content, at: performance.now() - began })); |
| 130 | const first = ev.find((e) => e.type === 'tool_call'); |
| 131 | const last = ev.filter((e) => e.type === 'tool_result').pop(); |
| 132 | return { took: (first && last) ? last.at - first.at : NaN, events: ev }; |
| 133 | }; |
| 134 | window.__median = (xs) => { |
| 135 | const v = xs.slice().sort((a, b) => a - b); |
| 136 | return v[Math.floor(v.length / 2)]; |
| 137 | }; |
| 138 | }); |
| 139 | await page.evaluate((m) => { window.__mock = m; }, MOCK); |
| 140 | |
| 141 | // The fixture, written through the engine's own file tools, so the reads come |
| 142 | // out of the same OPFS everything else uses. |
| 143 | const built = await page.evaluate(async ({ engine, reads, bytes, tree }) => { |
| 144 | const app = await window.__app(engine); |
| 145 | window.__fixture = app; |
| 146 | const body = 'x'.repeat(bytes); |
| 147 | for (let i = 0; i < reads; i++) { |
| 148 | // Each file opens with its own word, so a result in the wrong place is |
| 149 | // caught by what it says and not merely by where it is. |
| 150 | await app.run_tool('file_write', JSON.stringify({ |
| 151 | path: `src/file${i}.txt`, content: `MARK-${i}\n` + body })); |
| 152 | } |
| 153 | for (let i = 0; i < tree; i++) { |
| 154 | await app.run_tool('file_write', JSON.stringify({ |
| 155 | path: `tree/d${i % 12}/leaf${i}.txt`, |
| 156 | content: `needle${i % 6} in a haystack of ${i}\n` })); |
| 157 | } |
| 158 | return reads; |
| 159 | }, { engine: ENGINE, reads: READS, bytes: BYTES, tree: TREE }); |
| 160 | check(`${READS} files and a tree of ${TREE} are in the workspace`, built === READS, |
| 161 | `built ${built}`); |
| 162 | |
| 163 | // ── §1: the round answers, and answers in order ───────────── |
| 164 | const batch = await page.evaluate(async ({ engine, prompt }) => { |
| 165 | const app = await window.__app(engine); |
| 166 | return await window.__round(app, prompt); |
| 167 | }, { engine: ENGINE, prompt: roundOf('file_read', readArgs, READS) }); |
| 168 | |
| 169 | const results = batch.events.filter((e) => e.type === 'tool_result'); |
| 170 | check(`§1 all ${READS} reads answered`, results.length === READS, |
| 171 | `${results.length} result(s)`); |
| 172 | const order = results.map((r, i) => (r.content || '').includes(`MARK-${i}`)); |
| 173 | check('§1 and each answer is the answer to its own call, in the order the model gave', |
| 174 | order.every(Boolean), |
| 175 | order.map((v, i) => (v ? '' : `call ${i}`)).filter(Boolean).join(', ') || 'all in place'); |
| 176 | |
| 177 | // ── §2: the wall clock ────────────────────────────────────── |
| 178 | // |
| 179 | // Both engines, alternating, so a machine that gets busy half way through |
| 180 | // spoils both medians equally instead of one of them. |
| 181 | const engines = AGAINST ? [ENGINE, AGAINST] : [ENGINE]; |
| 182 | const work = [ |
| 183 | [`${READS} reads of ${BYTES / 1024}KB`, roundOf('file_read', readArgs, READS)], |
| 184 | [`4 searches over ${TREE} files`, roundOf('file_search', searchArgs, 4)], |
| 185 | ]; |
| 186 | for (const [what, prompt] of work) { |
| 187 | const took = await page.evaluate(async ({ engines, prompt, rounds }) => { |
| 188 | const apps = {}; |
| 189 | for (const e of engines) apps[e] = await window.__app(e); |
| 190 | const runs = {}; |
| 191 | for (const e of engines) runs[e] = []; |
| 192 | // One warm round each before anything is recorded: the first walk of a |
| 193 | // tree pays for handles nothing else pays for. |
| 194 | for (const e of engines) await window.__round(apps[e], prompt); |
| 195 | for (let i = 0; i < rounds; i++) { |
| 196 | for (const e of engines) { |
| 197 | runs[e].push((await window.__round(apps[e], prompt)).took); |
| 198 | } |
| 199 | } |
| 200 | const out = {}; |
| 201 | for (const e of engines) out[e] = window.__median(runs[e]); |
| 202 | return out; |
| 203 | }, { engines, prompt, rounds: ROUNDS }); |
| 204 | const mine = took[ENGINE]; |
| 205 | if (AGAINST) { |
| 206 | const theirs = took[AGAINST]; |
| 207 | // Named by the engine each figure came from rather than "batched" and |
| 208 | // "serial": the two can be given the other way round -- which is exactly |
| 209 | // what the breakproof does -- and a fixed label would then read as a lie. |
| 210 | note(`${what}: ${mine.toFixed(1)} ms on ${ENGINE}, ${theirs.toFixed(1)} ms on ` |
| 211 | + `${AGAINST} — ${(theirs - mine).toFixed(1)} ms saved, ` |
| 212 | + `${(mine / theirs * 100).toFixed(0)}% of the round on ${AGAINST}`); |
| 213 | check(`§2 ${what} is quicker on ${ENGINE} than on ${AGAINST}`, mine < theirs, |
| 214 | `${mine.toFixed(1)} ms against ${theirs.toFixed(1)} ms`); |
| 215 | } else { |
| 216 | note(`${what}: ${mine.toFixed(1)} ms per round`); |
| 217 | } |
| 218 | } |
| 219 | if (!AGAINST) { |
| 220 | note('no --against engine, so §2 reports figures and asserts nothing;' |
| 221 | + ' run dev/breakproof_toolbatch.sh for the comparison'); |
| 222 | } |
| 223 | |
| 224 | // ── §3: a write keeps its place ───────────────────────────── |
| 225 | const mixed = await page.evaluate(async ({ engine, prompt }) => { |
| 226 | const app = await window.__app(engine); |
| 227 | return (await window.__round(app, prompt)).events; |
| 228 | }, { engine: ENGINE, prompt: '@tools file_read ' + readArgs(0) |
| 229 | + ' ;; file_read ' + readArgs(1) |
| 230 | + ' ;; file_write ' + JSON.stringify({ path: 'src/out.txt', content: 'written' }) |
| 231 | + ' ;; file_read ' + readArgs(2) }); |
| 232 | const names = mixed.filter((e) => e.type === 'tool_call').map((e) => e.name); |
| 233 | check('§3 a write among the reads keeps the place the model gave it', |
| 234 | JSON.stringify(names) === JSON.stringify( |
| 235 | ['file_read', 'file_read', 'file_write', 'file_read']), |
| 236 | names.join(', ')); |
| 237 | |
| 238 | // ── §4: the page's contract ───────────────────────────────── |
| 239 | const wire = batch.events.filter((e) => e.type === 'tool_call' || e.type === 'tool_result'); |
| 240 | const alternates = wire.length === READS * 2 |
| 241 | && wire.every((e, i) => e.type === (i % 2 === 0 ? 'tool_call' : 'tool_result')); |
| 242 | check('§4 one call is announced, then its result, then the next', alternates, |
| 243 | wire.map((e) => (e.type === 'tool_call' ? 'C' : 'R')).join('')); |
| 244 | } finally { |
| 245 | await s.close(); |
| 246 | } |
| 247 | } |
| 248 | |
| 249 | main().then(() => { |
| 250 | console.log(`\n ${ok.length} ok, ${bad.length} failed`); |
| 251 | process.exit(bad.length ? 1 : 0); |
| 252 | }).catch((e) => { |
| 253 | console.error('ABORT: ' + (e && e.stack || e)); |
| 254 | process.exit(2); |
| 255 | }); |