oxedyne/daimond/dev/probe_askav.mjs
11.3 KiB, 1 run
created by r2519314175:33, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // probe_askav.mjs — a model asks the owner a question, and he answers with one tap. |
| 2 | // |
| 3 | // The whole path in the real page: the model calls `ask`, the card is drawn in the thread, |
| 4 | // a tap sends the answer, and the turn that asked has ALREADY ENDED. Then a reload, to prove |
| 5 | // the question survives the tab and that an answered one comes back answered. |
| 6 | import { open, chat, mockLog, clearMockLog, contentText, connectMock, transcript, shot, signInAs } from './harness.mjs'; |
| 7 | |
| 8 | const Q = { |
| 9 | question: 'Which store should the drafts live in?', |
| 10 | options: [ |
| 11 | { label: 'OPFS', means: 'Drafts stay on this device only. Nothing to pay, and a lost laptop is a lost draft.' }, |
| 12 | { label: 'Cloud', means: 'Drafts sync to your other devices. Needs Pro, and the bytes are billed.' }, |
| 13 | ], |
| 14 | recommend: 'OPFS', |
| 15 | why: 'You write on one machine and have said twice that you do not want another bill.', |
| 16 | if_silent: 'I will use OPFS and say so in the commit message.', |
| 17 | n: 1, of: 3, |
| 18 | }; |
| 19 | |
| 20 | const s = await open({ name: 'askav', defaults: false }); |
| 21 | const { page } = s; |
| 22 | const errs = []; |
| 23 | page.on('pageerror', e => errs.push('PAGEERROR ' + String(e).slice(0, 200))); |
| 24 | await page.waitForTimeout(1500); |
| 25 | await connectMock(s); |
| 26 | await chat(s, 'hello'); |
| 27 | await page.waitForTimeout(400); |
| 28 | |
| 29 | const out = {}; |
| 30 | |
| 31 | // ── 1. The model asks ──────────────────────────────────────────────── |
| 32 | clearMockLog(); |
| 33 | await chat(s, '@tool ask ' + JSON.stringify(Q)); |
| 34 | await page.waitForTimeout(600); |
| 35 | |
| 36 | out.card = await page.evaluate(() => { |
| 37 | const c = document.querySelector('.ask-card'); |
| 38 | if (!c) return null; |
| 39 | return { |
| 40 | role: c.getAttribute('role'), |
| 41 | count: (c.querySelector('.ask-count') || {}).textContent || '', |
| 42 | q: (c.querySelector('.ask-q') || {}).textContent || '', |
| 43 | opts: [...c.querySelectorAll('.ask-opt')].map(b => ({ |
| 44 | label: (b.querySelector('.ask-label') || {}).textContent || '', |
| 45 | rec: b.classList.contains('recommended'), |
| 46 | badge: (b.querySelector('.ask-rec') || {}).textContent || '', |
| 47 | means: (b.querySelector('.ask-means') || {}).textContent || '', |
| 48 | })), |
| 49 | why: (c.querySelector('.ask-why') || {}).textContent || '', |
| 50 | silent: (c.querySelector('.ask-silent') || {}).textContent || '', |
| 51 | other: !!c.querySelector('.ask-other-open'), |
| 52 | // A permission prompt is a modal over a scrim. A question must not be one. |
| 53 | modal: !!document.querySelector('.modal.dlg'), |
| 54 | }; |
| 55 | }); |
| 56 | // ROUNDS. `@tool` normally makes two requests: the one that returns the call, and the one |
| 57 | // that returns the text afterwards. A turn that ends on the question makes ONE. |
| 58 | out.rounds = mockLog().length; |
| 59 | // And the result the model was handed, which it never got to read because the turn ended -- |
| 60 | // but which is in the record, and is what the NEXT turn reads. |
| 61 | out.result = await page.evaluate(() => { |
| 62 | const c = window.DaimondChats && DaimondChats.current ? DaimondChats.current() : null; |
| 63 | return null; |
| 64 | }); |
| 65 | await shot(s, 'askav-1-question'); |
| 66 | |
| 67 | // THE CONTROL on that number. A tool that does NOT end the turn makes two requests, so |
| 68 | // `rounds: 1` above is a measurement rather than a coincidence of this harness. |
| 69 | clearMockLog(); |
| 70 | await chat(s, '@tool file_list {"path":"."}'); |
| 71 | await page.waitForTimeout(400); |
| 72 | out.roundsControl = mockLog().length; |
| 73 | |
| 74 | // ── 2. One tap ─────────────────────────────────────────────────────── |
| 75 | clearMockLog(); |
| 76 | await page.evaluate(() => document.querySelector('.ask-card .ask-opt.recommended').click()); |
| 77 | await page.waitForTimeout(2500); |
| 78 | out.afterTap = await page.evaluate(() => { |
| 79 | const c = document.querySelector('.ask-card'); |
| 80 | return { |
| 81 | answered: !!(c && c.dataset.answered), |
| 82 | done: c ? ((c.querySelector('.ask-done') || {}).textContent || '') : '', |
| 83 | buttons: c ? c.querySelectorAll('.ask-opt').length : -1, |
| 84 | live: c ? c.querySelectorAll('.ask-opt:not(:disabled)').length : -1, |
| 85 | chosen: c ? ((c.querySelector('.ask-opt.chosen .ask-label') || {}).textContent || '') : '', |
| 86 | }; |
| 87 | }); |
| 88 | // What the model got. The user message on the wire is the answer. |
| 89 | out.sent = mockLog().flatMap(r => (r.messages || []) |
| 90 | .filter(m => m.role === 'user') |
| 91 | .map(m => contentText(m.content))) |
| 92 | .filter(t => /^(Chose|Other): /.test(t)); |
| 93 | // And what the tool handed the model, which the NEXT turn reads: it rides in the same |
| 94 | // request as the answer. |
| 95 | out.toolResult = mockLog().flatMap(r => (r.messages || []) |
| 96 | .filter(m => m.role === 'tool').map(m => contentText(m.content))) |
| 97 | .filter(t => /^Asked\./.test(t)); |
| 98 | await shot(s, 'askav-2-answered'); |
| 99 | |
| 100 | // ── 3. A reload ────────────────────────────────────────────────────── |
| 101 | await page.reload(); |
| 102 | await page.waitForFunction(() => !!window.DaimondCore, null, { timeout: 20000 }).catch(() => {}); |
| 103 | await page.waitForTimeout(1500); |
| 104 | // A reload lands on the passphrase gate: the tab really did close, as far as the app |
| 105 | // is concerned, which is the case being tested. |
| 106 | if (await page.$('#id-primary')) { await signInAs(s, 'askav'); } |
| 107 | await page.waitForTimeout(3000); |
| 108 | // And back into the chat that holds the question, the way a person gets there. |
| 109 | await page.evaluate(() => { |
| 110 | const row = document.querySelector('#session-list .session-item, .rail-item'); |
| 111 | if (row) row.click(); |
| 112 | }); |
| 113 | await page.waitForTimeout(2000); |
| 114 | out.afterReload = await page.evaluate(() => { |
| 115 | const c = document.querySelector('.ask-card'); |
| 116 | return c ? { |
| 117 | present: true, |
| 118 | answered: !!c.dataset.answered, |
| 119 | done: (c.querySelector('.ask-done') || {}).textContent || '', |
| 120 | buttons: c.querySelectorAll('.ask-opt').length, |
| 121 | live: c.querySelectorAll('.ask-opt:not(:disabled)').length, |
| 122 | q: (c.querySelector('.ask-q') || {}).textContent || '', |
| 123 | } : { present: false }; |
| 124 | }); |
| 125 | await shot(s, 'askav-3-reloaded'); |
| 126 | |
| 127 | // ── 4. A second question, rejected in his own words ────────────────── |
| 128 | clearMockLog(); |
| 129 | const Q2 = Object.assign({}, Q, { question: 'Shall I rename the field?', n: 2, |
| 130 | options: [{ label: 'Rename it', means: 'Every caller changes today.' }, |
| 131 | { label: 'Leave it', means: 'Nothing changes and the name stays wrong.' }], |
| 132 | recommend: 'Leave it' }); |
| 133 | await chat(s, '@tool ask ' + JSON.stringify(Q2)); |
| 134 | await page.waitForTimeout(600); |
| 135 | out.second = await page.evaluate(() => { |
| 136 | const cs = [...document.querySelectorAll('.ask-card')]; |
| 137 | return { cards: cs.length, live: cs.filter(c => !c.dataset.answered).length }; |
| 138 | }); |
| 139 | clearMockLog(); |
| 140 | await page.evaluate(() => { |
| 141 | const c = [...document.querySelectorAll('.ask-card')].filter(x => !x.dataset.answered).pop(); |
| 142 | c.querySelector('.ask-other-open').click(); |
| 143 | }); |
| 144 | await page.waitForTimeout(300); |
| 145 | await page.evaluate(() => { |
| 146 | const c = [...document.querySelectorAll('.ask-card')].filter(x => !x.dataset.answered).pop(); |
| 147 | const box = c.querySelector('.ask-other-box'); |
| 148 | box.value = 'Neither — split it in two and keep both names.'; |
| 149 | box.dispatchEvent(new Event('input', { bubbles: true })); |
| 150 | c.querySelector('.ask-other-go').click(); |
| 151 | }); |
| 152 | await page.waitForTimeout(2500); |
| 153 | out.other = mockLog().flatMap(r => (r.messages || []) |
| 154 | .filter(m => m.role === 'user').map(m => contentText(m.content))) |
| 155 | .filter(t => /^Other: /.test(t)); |
| 156 | await shot(s, 'askav-4-other'); |
| 157 | |
| 158 | // ── 5. Two questions in one turn: the second is refused ────────────── |
| 159 | clearMockLog(); |
| 160 | await chat(s, '@tools ask ' + JSON.stringify(Q) + ' ;; ask ' + JSON.stringify(Q2)); |
| 161 | await page.waitForTimeout(800); |
| 162 | out.twoInOneTurn = await page.evaluate(() => { |
| 163 | const cs = [...document.querySelectorAll('.ask-card')].filter(x => !x.dataset.answered); |
| 164 | return { live: cs.length, last: cs.length ? (cs[cs.length-1].querySelector('.ask-q')||{}).textContent : '' }; |
| 165 | }); |
| 166 | out.tail = (await transcript(s)).slice(-500); |
| 167 | await shot(s, 'askav-5-two'); |
| 168 | |
| 169 | out.errs = errs; |
| 170 | |
| 171 | // ── What has to be true ────────────────────────────────────────────── |
| 172 | // |
| 173 | // Written as checks with an exit code rather than as a dump to read, because a probe |
| 174 | // whose result is a human squinting at JSON is a probe that passes the day it stops |
| 175 | // working. |
| 176 | const bad = []; |
| 177 | const ok = (cond, why) => { if (!cond) bad.push(why); }; |
| 178 | ok(out.card, 'no question card was drawn at all'); |
| 179 | if (out.card) { |
| 180 | ok(out.card.role === 'group' && !out.card.modal, |
| 181 | 'the question is a modal dialog, which is what a PERMISSION prompt is'); |
| 182 | ok(out.card.count === 'Decision 1 of 3', 'the card does not say how many decisions follow'); |
| 183 | ok(out.card.opts.length === 2, 'the options were not drawn as buttons'); |
| 184 | ok(out.card.opts.filter(o => o.rec).length === 1 && out.card.opts[0].rec, |
| 185 | 'the recommendation is not marked, so it is implied by ordering or not at all'); |
| 186 | ok(/RECOMMENDED/i.test(out.card.opts[0].badge), |
| 187 | 'the recommendation is marked by colour alone, which a red-green reader cannot see'); |
| 188 | ok(out.card.opts.every(o => o.means.length > 20), |
| 189 | 'an option is a bare label, so it names a category rather than a consequence'); |
| 190 | ok(/another bill/.test(out.card.why), 'the reason for the recommendation is not on the card'); |
| 191 | ok(/commit message/.test(out.card.silent), |
| 192 | 'what happens if he says nothing is not on the card, so silence is not a knowing answer'); |
| 193 | ok(out.card.other, 'there is no way to reject every option'); |
| 194 | } |
| 195 | // THE MEASUREMENT AND ITS CONTROL. A tool that does not end the turn costs two requests; |
| 196 | // this one costs one. Both halves, so the number means something. |
| 197 | ok(out.rounds === 1, `the turn went round again after asking: ${out.rounds} request(s)`); |
| 198 | ok(out.roundsControl === 2, |
| 199 | `the control is wrong: an ordinary tool took ${out.roundsControl} request(s), not 2, so ` + |
| 200 | 'the figure above proves nothing'); |
| 201 | ok(out.afterTap.answered && out.afterTap.buttons === 2 && out.afterTap.live === 0, |
| 202 | `an answered card still offers its buttons, or has thrown the alternatives away: ${JSON.stringify(out.afterTap)}`); |
| 203 | ok(out.afterTap.chosen === 'OPFS', |
| 204 | `the option he chose is not marked on the card: ${out.afterTap.chosen}`); |
| 205 | ok(out.sent.length === 1 && out.sent[0] === 'Chose: OPFS', |
| 206 | `the tap did not reach the model as an answer: ${JSON.stringify(out.sent)}`); |
| 207 | ok(out.toolResult.length === 1 && /Chose:/.test(out.toolResult[0]) && /Other:/.test(out.toolResult[0]), |
| 208 | 'the model was not told the two shapes its answer arrives in'); |
| 209 | // THE RELOAD, which is the whole durability claim: the tab really did go. |
| 210 | ok(out.afterReload.present && /drafts live in/.test(out.afterReload.q), |
| 211 | 'the question did not survive the reload'); |
| 212 | ok(out.afterReload.answered && out.afterReload.live === 0 && out.afterReload.buttons === 2, |
| 213 | `a question already answered came back offering its buttons, or without them: ${JSON.stringify(out.afterReload)}`); |
| 214 | ok(out.other.length === 1 && /^Other: Neither/.test(out.other[0]), |
| 215 | `an answer in his own words did not reach the model as the answer: ${JSON.stringify(out.other)}`); |
| 216 | ok(out.twoInOneTurn.live === 1, |
| 217 | `two questions were put in one turn: ${out.twoInOneTurn.live} card(s) live`); |
| 218 | ok(errs.length === 0, `the page threw: ${errs.join(' | ')}`); |
| 219 | |
| 220 | console.log(JSON.stringify(out, null, 1)); |
| 221 | if (bad.length) { |
| 222 | console.log('\nprobe_askav: FAILED'); |
| 223 | bad.forEach(b => console.log(' - ' + b)); |
| 224 | } else { |
| 225 | console.log('\nprobe_askav: the model asked, he tapped once, and the answer reached the model.'); |
| 226 | } |
| 227 | await s.close(); |
| 228 | process.exit(bad.length ? 1 : 0); |