oxedyne/daimond/dev/verify_wholerecord.mjs
10.4 KiB, 1 run
created by r2519314175:811, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // An answer shortened to fit the model's window stays whole in the user's record. |
| 2 | // |
| 3 | // The owner's ruling of 2026-08-28: *the model gets the shortened version, his transcript |
| 4 | // keeps every word.* `compact::elide_bulk` used to rewrite the older assistant turns of |
| 5 | // `session.messages` down to `TOOL_ELISION_CAP` -- 400 characters and a note -- and |
| 6 | // `session.messages` is the list `DaimondApp::export_session` hands the browser to store, to |
| 7 | // back up and to put in the sync parcel. So a thousand-word answer became two sentences in |
| 8 | // the STORE as well as on the wire, permanently, and nothing anywhere said so. |
| 9 | // |
| 10 | // The separation is made at the point of derivation: the elision is done to the turn's own |
| 11 | // message list on its way out and never to the session. This asks the two questions that |
| 12 | // distinguishes those, and it asks them of the two artefacts rather than of the app -- |
| 13 | // `dev/ctxmock.log`, which is what the provider was really sent, and IndexedDB, which is what |
| 14 | // is really on disk. |
| 15 | // |
| 16 | // The window is calibrated the way `dev/verify_compact.mjs` calibrates it, and for the reason |
| 17 | // written there at length: the floor is the system prompt plus the tool schemas, nobody owns |
| 18 | // it, and a written-down window stops being a window the day a lane adds a tool. |
| 19 | // |
| 20 | // node dev/verify_wholerecord.mjs |
| 21 | |
| 22 | import fs from 'node:fs'; |
| 23 | import path from 'node:path'; |
| 24 | import { spawn } from 'node:child_process'; |
| 25 | import { fileURLToPath } from 'node:url'; |
| 26 | |
| 27 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 28 | const ROOT = path.dirname(HERE); |
| 29 | const H = await import(path.join(HERE, 'harness.mjs')); |
| 30 | const { open, newChat, transcript, shot, connectMock, storedChats, signInAs } = H; |
| 31 | |
| 32 | const WORLD = Number(process.env.DAIMOND_PORT || 8777) - 8777; |
| 33 | const LOG = process.env.DAIMOND_WHOLE_LOG |
| 34 | || path.join(HERE, WORLD ? `wholemock-${WORLD}.log` : 'wholemock.log'); |
| 35 | const PORT = Number(process.env.DAIMOND_WHOLE_PORT || 9800 + WORLD); |
| 36 | const MOCK = `http://127.0.0.1:${PORT}/v1/chat/completions`; |
| 37 | const MODEL = 'mock/fast'; |
| 38 | const NAME = 'wholerecord'; |
| 39 | const MARK = 'LONGANSWER-K4W9'; |
| 40 | // The sentence the engine writes over what it shortened. Read from `compact.rs` rather than |
| 41 | // copied, so a change to the wording fails here loudly instead of turning every check below |
| 42 | // into one that cannot see its own subject. |
| 43 | const CLIP = (() => { |
| 44 | const src = fs.readFileSync(path.join(ROOT, 'src', 'compact.rs'), 'utf8'); |
| 45 | const m = /were folded away to fit the context window/.exec(src); |
| 46 | if (!m) throw new Error('compact.rs no longer writes that sentence; this file cannot see its subject'); |
| 47 | return m[0]; |
| 48 | })(); |
| 49 | const PROBE_LIMIT = 10_000_000; |
| 50 | const HEADROOM = 400; |
| 51 | const FOLD_AT = (() => { |
| 52 | const m = /pub const FOLD_AT:\s*f64\s*=\s*([0-9.]+)/.exec( |
| 53 | fs.readFileSync(path.join(ROOT, 'src', 'compact.rs'), 'utf8')); |
| 54 | if (!m) throw new Error('compact.rs no longer declares FOLD_AT; this file cannot calibrate'); |
| 55 | return Number(m[1]); |
| 56 | })(); |
| 57 | |
| 58 | let failures = 0; |
| 59 | const log = (...a) => console.log(...a); |
| 60 | const line = (t) => log('\n════════ ' + t + ' ════════'); |
| 61 | const check = (ok, what, detail = '') => { |
| 62 | log((ok ? ' PASS ' : ' FAIL ') + what + (detail ? ' -- ' + detail : '')); |
| 63 | if (!ok) failures++; |
| 64 | }; |
| 65 | const requests = () => fs.readFileSync(LOG, 'utf8').split('\n').filter(Boolean) |
| 66 | .map(l => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean); |
| 67 | const bodyOf = (r) => (r.messages || []).map(m => { |
| 68 | const c = m.content; |
| 69 | if (typeof c === 'string') return c; |
| 70 | if (Array.isArray(c)) return c.map(p => (p && p.text) || '').join(' '); |
| 71 | return ''; |
| 72 | }).join('\n'); |
| 73 | |
| 74 | async function startMock(limit) { |
| 75 | const child = spawn('node', [path.join(HERE, 'ctxmock.mjs'), String(PORT), String(limit)], |
| 76 | { stdio: ['ignore', 'ignore', 'inherit'], env: { ...process.env, DAIMOND_CTX_LOG: LOG } }); |
| 77 | for (let i = 0; i < 50; i++) { |
| 78 | try { |
| 79 | const r = await fetch(`http://127.0.0.1:${PORT}/v1/models`); |
| 80 | if (r.ok) return child; |
| 81 | } catch (e) { /* not up yet */ } |
| 82 | await new Promise(r => setTimeout(r, 100)); |
| 83 | } |
| 84 | child.kill(); |
| 85 | throw new Error(`ctxmock did not come up on ${PORT}; is something else already listening?`); |
| 86 | } |
| 87 | |
| 88 | const say = async (s, text, waitMs = 7000) => { |
| 89 | await s.page.fill('#chat-input', text); |
| 90 | await s.page.click('#chat-send'); |
| 91 | await s.page.waitForTimeout(waitMs); |
| 92 | }; |
| 93 | |
| 94 | try { fs.writeFileSync(LOG, ''); } catch {} |
| 95 | let mock = await startMock(PROBE_LIMIT); |
| 96 | const s = await open({ name: NAME, connect: false }); |
| 97 | await connectMock(s, { baseUrl: MOCK, model: MODEL }); |
| 98 | await newChat(s); |
| 99 | |
| 100 | await say(s, 'hello', 5000); |
| 101 | const probed = (() => { const r = requests(); return r.length ? r[0].used : 0; })(); |
| 102 | if (!probed) { log('FAIL the probe never reached the mock, so the window cannot be calibrated'); process.exit(1); } |
| 103 | const LIMIT = Math.ceil((probed + HEADROOM) / FOLD_AT); |
| 104 | log(`floor ${probed} tokens, folding at ${FOLD_AT}, so the window for this run is ${LIMIT}`); |
| 105 | mock.kill(); |
| 106 | await new Promise((r) => setTimeout(r, 300)); |
| 107 | try { fs.writeFileSync(LOG, ''); } catch {} |
| 108 | mock = await startMock(LIMIT); |
| 109 | await newChat(s); |
| 110 | |
| 111 | // ── The fixture: one long answer, then enough bulk to outgrow the window ───── |
| 112 | line('1. a long answer, and then a conversation too big to send'); |
| 113 | // `@text` echoes what follows it, so the answer is built here and its marker sits well past |
| 114 | // TOOL_ELISION_CAP -- the whole point is a word that only survives if the record is whole. |
| 115 | const FILL = 'the answer goes on at length about what it found and why it matters. '; |
| 116 | const LONG = FILL.repeat(12) + MARK + ' ' + FILL.repeat(40); |
| 117 | await say(s, `@text ${LONG}`, 8000); |
| 118 | let reached = false; |
| 119 | for (let i = 0; i < 8 && !reached; i++) { |
| 120 | await say(s, '@big 20', 10000); |
| 121 | reached = requests().some(r => bodyOf(r).includes(CLIP)); |
| 122 | const last = requests().slice(-1)[0] || {}; |
| 123 | log(` turn ${i + 1}: ${last.used} tokens, refused=${!!last.refused}, shortened=${reached}`); |
| 124 | } |
| 125 | // A fixture that never reached the branch proves nothing, so it stops rather than passing. |
| 126 | if (!reached) { |
| 127 | console.error('the conversation never had an answer shortened on the way out, so there is ' |
| 128 | + 'nothing here to be whole or not whole. The fixture no longer reaches elide_bulk.'); |
| 129 | await s.close(); |
| 130 | mock.kill(); |
| 131 | process.exit(2); |
| 132 | } |
| 133 | check(true, 'the wire really did carry a shortened answer, which is the state this is about'); |
| 134 | |
| 135 | line('2. what the model was sent, and what the store kept'); |
| 136 | const stored = await storedChats(s); |
| 137 | const chat = stored.slice().sort((a, b) => (b.updatedAt || 0) - (a.updatedAt || 0))[0] || {}; |
| 138 | const sess = (chat.session && chat.session.msgs) || []; |
| 139 | const sessText = sess.map(m => String((m && m.content) || '')).join('\n'); |
| 140 | const seenText = (chat.messages || []).map(m => String((m && m.content) || '')).join('\n'); |
| 141 | |
| 142 | check(sess.length > 0, |
| 143 | 'the model\'s own conversation was stored, which is the artefact under test', |
| 144 | `${sess.length} messages`); |
| 145 | // THE RULING, and the check that was red before it. The store used to hold exactly what went on |
| 146 | // the wire, sentence for sentence, because the two were one list. |
| 147 | // |
| 148 | // A FOLD IS NOT THIS. A conversation this size folds as well as shortening, and a fold really |
| 149 | // does replace what it folded -- deliberately, announced, and drawn as a line across the thread |
| 150 | // by `appendCompacted`. So the marker is not asked of the stored session, which may legitimately |
| 151 | // no longer hold the message it was in; what is asked is that nothing in that store was CLIPPED, |
| 152 | // which is the operation that was never announced and never had a way back. |
| 153 | check(!sessText.includes(CLIP), |
| 154 | 'AND NOTHING IN IT WAS SHORTENED — the record is not the wire', |
| 155 | sessText.includes(CLIP) ? 'the store holds the engine\'s own elision note' : 'no elision note in the store'); |
| 156 | check(!seenText.includes(CLIP), |
| 157 | 'nor was anything in the transcript on screen'); |
| 158 | check(seenText.includes(MARK), |
| 159 | 'and the long answer is still there in full, marker and all — which is the user\'s record'); |
| 160 | |
| 161 | line('3. and the reader is told, once and quietly'); |
| 162 | const said = await transcript(s); |
| 163 | // The notice that actually shortened something, not merely a notice. A fold that shortened |
| 164 | // nothing writes "shortened 0", and a check that matched it would pass on a run where the |
| 165 | // sentence under test never described anything. |
| 166 | const shortNote = (said.match(/[^\n]*[Ss]hortened [1-9][0-9]* long[^\n]*/) || [''])[0]; |
| 167 | check(/on the way to the model/.test(shortNote), |
| 168 | 'the notice says where the shortening applies, rather than leaving it to be guessed', |
| 169 | shortNote.slice(0, 200) || '(no notice reported shortening anything)'); |
| 170 | // TOLD ONCE. The shortening is redone from the whole record on every turn from here on, so the |
| 171 | // engine says so every turn; a thread with a grey notice per turn is the noise `worthSaying` |
| 172 | // exists to stop. |
| 173 | const shortLines = (said.match(/[Ss]hortened [1-9][0-9]* long/g) || []).length; |
| 174 | check(shortLines <= 1, |
| 175 | 'and it is said once rather than on every turn from then on', |
| 176 | `${shortLines} such notice(s) in the thread`); |
| 177 | |
| 178 | // SCROLLED TO THE NOTICE BEFORE IT IS SHOT. A conversation this size ends in several screens of |
| 179 | // bulk, so a screenshot taken where the thread happens to be shows lorem ipsum and settles |
| 180 | // nothing about the one element this file is about. |
| 181 | await s.page.evaluate(() => { |
| 182 | const notes = [...document.querySelectorAll('.chat-msg-compacted')]; |
| 183 | const last = notes[notes.length - 1]; |
| 184 | if (last) last.scrollIntoView({ block: 'center' }); |
| 185 | }); |
| 186 | await s.page.waitForTimeout(400); |
| 187 | await shot(s, 'wholerecord-notice'); |
| 188 | |
| 189 | line('4. after a reload'); |
| 190 | await s.page.reload({ waitUntil: 'domcontentloaded' }); |
| 191 | await s.page.waitForTimeout(1200); |
| 192 | await signInAs(s, NAME); |
| 193 | await s.page.waitForTimeout(1500); |
| 194 | await s.page.evaluate(() => { |
| 195 | const b = document.querySelector('#session-list .chat-box.active') |
| 196 | || document.querySelector('#session-list .chat-box'); |
| 197 | if (b) b.click(); |
| 198 | }); |
| 199 | await s.page.waitForTimeout(1200); |
| 200 | const after = await transcript(s); |
| 201 | check(after.includes(MARK), 'a reader who comes back tomorrow still has every word of it'); |
| 202 | |
| 203 | await shot(s, 'wholerecord-final'); |
| 204 | // The 400 is the fixture: `ctxmock` refuses a request past the window it was started with, and |
| 205 | // driving the app into exactly that refusal is how this file gets a conversation shortened at |
| 206 | // all. Filtering it is not looking away from an error; it is naming the one this run causes. |
| 207 | const errs = s.errs.filter(e => !/favicon|manifest|502|Bad Gateway|\b400\b/i.test(e)); |
| 208 | check(errs.length === 0, 'no unexpected console errors', errs.slice(0, 3).join(' | ')); |
| 209 | await s.close(); |
| 210 | mock.kill(); |
| 211 | |
| 212 | log(`\n${failures === 0 ? 'ALL CHECKS PASSED' : failures + ' CHECK(S) FAILED'}`); |
| 213 | process.exit(failures === 0 ? 0 : 1); |