Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/verify_credits.mjs

41.6 KiB, 1 run

created by r2519314175:323, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1// verify_credits.mjs — credits buy inference: a balance is a provider row, and its key is minted.
2//
3// Daimond had two halves that did not touch. BYOK inference ran in the browser against a key the
4// user pasted; credits paid for web fetches, mail, sync and tools through the gateway. A user with
5// $10 in their account and no key of their own could do everything the app does EXCEPT think, and
6// the model picker told them so: "no model connected". This is the seam.
7//
8// The shape of the fix is the thing to keep honest, so it is what is checked hardest:
9//
10// * The gateway MINTS a provider key; it does not proxy the request. The browser still calls the
11// provider directly, which is the entire product. A test that let a relay creep in would be
12// testing a different application.
13// * That minted key is a bearer credential for money and has NO at-rest story: it must not reach
14// localStorage, sealed or otherwise. Asserted by dumping storage and grepping for the key —
15// and, first, by proving the key was really there to leak, so the grep cannot pass vacuously.
16// * Two economies now share one picker. Some rows spend the user's Daimond balance; the rest are
17// billed by a provider the user holds their own account with. A user must be able to see which
18// BEFORE they pick, because the same model sits on both.
19//
20// The gateway is not running locally (/api/* is a 502 from dev/serve.mjs), so its four calls and
21// the provider behind the minted key are FETCH-stubbed — the store is never touched, and every
22// assertion below is against the real wasm, the real models.js and the real DOM.
23import { open, signInAs, connectMock, shot, errors, mockLog, clearMockLog, contentText, APP, MOCK } from './harness.mjs';
24
25const ok = [], bad = [];
26const check = (name, pass, detail) => {
27 (pass ? ok : bad).push(name + (detail ? ' — ' + detail : ''));
28 console.log((pass ? ' ok ' : ' FAIL ') + name + (detail ? ' — ' + detail : ''));
29};
30
31// A key shaped like the real thing and unmistakable in a haystack: the storage check greps for it.
32const keyN = (n) => `sk-or-v1-MINT${n}-zqxwmarker0000000000000000000000000000000000`;
33// What the gateway actually hands back is a BASE url -- its `DEFAULT_INFERENCE_URL`, which is what
34// its operator configures and what the host documents -- and NOT the endpoint a turn is posted to.
35// The client has to make the second out of the first. Stubbing the endpoint here instead would
36// have tested a contract nobody implements, and passed while every real turn 404'd.
37const OR_BASE = 'https://openrouter.ai/api/v1';
38const OR_URL = `${OR_BASE}/chat/completions`;
39// The gateway caps a minted key at min(float, balance), and its float is $2.00 -- so a key is
40// routinely spent while the account still holds credits. That gap is the whole reason a refusal
41// mid-session is answered with a fresh key rather than reported as a bad key.
42const FLOAT = 200;
43const CORS = { 'access-control-allow-origin': '*', 'access-control-allow-headers': '*' };
44const json = (body, status = 200) => ({
45 status, contentType: 'application/json', headers: CORS, body: JSON.stringify(body),
46});
47
48// What OpenRouter answers /models with, in miniature but in its real shape: vendor-prefixed ids,
49// the frontier models a credits user is buying access to, and — deliberately — `z-ai/glm-5p2`,
50// which the BYOK mock provider also serves as `accounts/fireworks/models/glm-5p2`. One model, two
51// hosts, two prefixes, one bare name. That collision is the case the picker must not fudge.
52const CATALOGUE = [
53 'anthropic/claude-opus-4.5',
54 'openai/gpt-5.2',
55 'deepseek/deepseek-v3',
56 'meta-llama/llama-3.3-70b-instruct',
57 'qwen/qwen3-235b-a22b',
58 'z-ai/glm-5p2',
59];
60
61// The gateway's answers, and the levers a test pulls on them.
62const gw = {
63 bal: 840, // minor units: $8.40
64 mints: 0, // POSTs to /api/inference-key
65 refuse: false, // the account has run dry: every mint is refused
66 chatHits: 0, // POSTs the browser made to the provider, direct
67 spent: [], // keys the provider now refuses, as a minted key's cap is reached
68 // Which requests a spent key is refused for. A conductor's turn and the workers it dispatches
69 // run on the SAME key, and the conductor must survive to dispatch them at all -- so the worker
70 // test spends the key only for the workers. A worker's system prompt says "You are a worker
71 // agent dispatched...", which is the request identifying itself and needs no bookkeeping here.
72 onlyFor: null,
73};
74/// The marker in a worker's system prompt (src/wasm/diamond.rs: WORKER_PROMPT).
75const WORKER_SAYS = /You are a worker agent dispatched/;
76
77async function stubGateway(page) {
78 // bootstrap(): register the device key, prove possession, take a session.
79 await page.route('**/api/account', r => r.fulfill(json({ ok: true })));
80 await page.route('**/api/auth/challenge', r => r.fulfill(json({ ok: true, challenge: 'chal-zqxw', challenge_id: 'cid-1' })));
81 await page.route('**/api/auth/verify', r => r.fulfill(json({ ok: true })));
82 await page.route('**/api/balance', r => r.fulfill(json({ ok: true, credits_minor: gw.bal, currency: 'usd', entries: [] })));
83
84 // The contract under test: session-authed, empty body, a key capped at the balance.
85 await page.route('**/api/inference-key', r => {
86 gw.mints++;
87 gw.mintHead = r.request().headers();
88 // "Your tab is too old to serve." Only reachable because the client names its version.
89 if (gw.stale426) return r.fulfill(json({ ok: false, error: 'Client too old.', min_api: 99, api: 99 }, 426));
90 // A dry account is `402 Payment Required` with the gateway's own copy, not a 200 with a
91 // flag: the client has to read the status and the body, as the real one sends both.
92 if (gw.refuse) return r.fulfill(json({
93 ok: false, credits_minor: 0,
94 error: 'The account has no credits, so Daimond cannot buy inference for it. '
95 + 'Add credits, or bring your own model key.',
96 }, 402));
97 return r.fulfill(json({
98 ok: true, key: keyN(gw.mints), url: OR_BASE,
99 limit_minor: Math.min(FLOAT, gw.bal), credits_minor: gw.bal, currency: 'usd',
100 }));
101 });
102
103 // The provider, which the BROWSER talks to and the gateway never sees.
104 await page.route('https://openrouter.ai/api/v1/models',
105 r => r.fulfill(json({ data: CATALOGUE.map(id => ({ id })) })));
106
107 // A minted key is capped at the balance behind it, so the provider refuses it the moment that
108 // cap is spent — indistinguishable, from out here, from a bad key. Otherwise the request is
109 // handed to the mock LLM, so a turn really completes and the retry can be seen to have worked.
110 await page.route(OR_URL, async (r) => {
111 gw.chatHits++;
112 const auth = r.request().headers()['authorization'] || '';
113 const sent = r.request().postData() || '';
114 const spent = gw.spent.some(k => auth.includes(k));
115 if (spent && (!gw.onlyFor || gw.onlyFor.test(sent))) {
116 return r.fulfill(json({ error: { message: 'Key limit exceeded' } }, 401));
117 }
118 const res = await fetch(MOCK, {
119 method: 'POST', headers: { 'content-type': 'application/json' }, body: r.request().postData(),
120 });
121 const body = await res.text();
122 return r.fulfill({
123 status: res.status, headers: CORS,
124 contentType: res.headers.get('content-type') || 'application/json',
125 body,
126 });
127 });
128}
129
130// ── A session that has credits and no key of its own ──────────────────
131const s = await open({ name: 'credits', signIn: false, connect: false });
132const { page } = s;
133await stubGateway(page);
134await signInAs(s, 'credits');
135await page.waitForTimeout(2500); // unlock → bootstrap → mint → list the catalogue
136
137check('unlocking mints a key (the balance is asked for models, not just shown)', gw.mints === 1, `${gw.mints} mint(s)`);
138
139// ── The contract version rides on the mint, and is not a second copy ──
140// The gateway only refuses a client that NAMES a version it knows is too old, so a mint that sent
141// no version could never be refused -- the 426 branch would be unreachable code that reads as
142// though it works.
143const clientApi = await page.evaluate(() => window.DaimondGateway.clientApi());
144check('the mint sends x-daimond-api', (gw.mintHead || {})['x-daimond-api'] === String(clientApi),
145 `sent ${JSON.stringify((gw.mintHead || {})['x-daimond-api'])}, gateway.js says ${clientApi}`);
146const modelsSrc = await (await fetch(`${APP}/js/models.js`)).text();
147check('...read from gateway.js rather than copied — one copy of that number exists in the client',
148 /DaimondGateway\.clientApi\(\)/.test(modelsSrc) && !/CLIENT_API\s*=/.test(modelsSrc),
149 'models.js reads the version and declares none of its own');
150
151// The 426 itself is exercised further down, once the assertions about the FIRST minted key are
152// out of the way -- proving it mints a fresh key, which would renumber them.
153
154const row = () => page.evaluate(() => {
155 const M = window.DaimondModels;
156 return (M.providers() || []).find(p => p.id === 'credits') || null;
157});
158
159const p0 = await row();
160check('a credits provider row exists', !!p0, p0 ? p0.id : 'absent');
161check('hasKey() counts a live in-memory key (the bug: a minted key reported hasKey:false)',
162 !!p0 && p0.hasKey === true, `hasKey=${p0 && p0.hasKey}`);
163check('the row is ready, so the picker will not disable it', !!p0 && p0.ready === true, `ready=${p0 && p0.ready}`);
164check('a minted key is never "sealed" — there is nothing stored to seal', !!p0 && p0.sealed === false);
165check('the catalogue behind the key is listed', !!p0 && p0.count === CATALOGUE.length, `${p0 && p0.count} models`);
166check('the frontier models a credits user is buying are in it',
167 !!p0 && p0.models.includes('anthropic/claude-opus-4.5') && p0.models.includes('openai/gpt-5.2'));
168
169// ── The key must have no at-rest story at all ─────────────────────────
170// First prove it is really there, or the grep below passes for the wrong reason.
171const live = await page.evaluate(() => window.DaimondModels.keyFor('credits'));
172check('the minted key IS live in memory (so the storage check cannot pass vacuously)',
173 live === keyN(1), live ? live.slice(0, 18) + '…' : '(empty)');
174
175const dump = await page.evaluate(() => {
176 let all = '';
177 for (let i = 0; i < localStorage.length; i++) all += localStorage.key(i) + '=' + localStorage.getItem(localStorage.key(i)) + '\n';
178 for (let i = 0; i < sessionStorage.length; i++) all += sessionStorage.key(i) + '=' + sessionStorage.getItem(sessionStorage.key(i)) + '\n';
179 return all;
180});
181check('THE MINTED KEY IS NOWHERE IN localStorage OR sessionStorage', !dump.includes(keyN(1)),
182 dump.includes(keyN(1)) ? 'LEAKED' : `${dump.length} bytes of storage searched`);
183check('nothing key-shaped from the mint leaked under another name', !dump.includes('zqxwmarker'));
184
185const stored = await page.evaluate(() => {
186 const st = JSON.parse(localStorage.getItem('daimond-models-v2') || '{}');
187 const c = (st.providers || {}).credits;
188 return c ? { has: true, key: c.key, keyEnc: c.keyEnc, url: c.url, models: (c.models || []).length } : { has: false };
189});
190check('the row IS stored (its name, host and models are not secrets)', stored.has === true);
191check('...with an empty `key` AND an empty `keyEnc` — not sealed, absent',
192 stored.key === '' && stored.keyEnc === '', `key=${JSON.stringify(stored.key)} keyEnc=${JSON.stringify(stored.keyEnc)}`);
193check('the gateway\'s BASE url is turned into the endpoint a turn is actually posted to',
194 stored.url === OR_URL, `${OR_BASE} → ${stored.url}`);
195check('...and it is the gateway\'s host, not one hardcoded in the client', stored.url.startsWith(OR_BASE));
196check('the key\'s cap is NOT the balance: a key can be spent while credits remain',
197 (await page.evaluate(() => window.DaimondModels.creditsState())).limit === FLOAT,
198 `limit=${(await page.evaluate(() => window.DaimondModels.creditsState())).limit} of ${gw.bal} balance`);
199
200// ── The rail's credits row: drawn from what the app KNOWS ─────────────
201//
202// The row that carries the balance sits in the status block at the foot of the rail, and it is
203// the only place in the app that says how much money is left. It used to be gated on
204// `navigator.onLine` ahead of everything else, and that value is a GUESS: a browser reports
205// offline whenever its own connectivity probe fails, network or no network. One machine
206// answering wrongly was enough to replace a signed-in account's balance with "Offline" while
207// the same tab went on fetching, syncing and spending -- the app knew the figure and refused
208// to show it.
209//
210// So: a session and a balance decide the row. The browser's guess may NAME a connection that
211// has actually failed; it may not overrule one that plainly has not.
212const railRow = await page.evaluate(() => {
213 const read = () => {
214 const n = document.getElementById('astat-account');
215 return n ? n.textContent : 'MISSING';
216 };
217 const out = {};
218 window.DaimondAdmin.status();
219 out.online = read();
220 // Make the browser lie, the way a misreporting one does, and repaint.
221 const real = Object.getOwnPropertyDescriptor(Navigator.prototype, 'onLine');
222 Object.defineProperty(Navigator.prototype, 'onLine', { configurable: true, get: () => false });
223 window.DaimondAdmin.status();
224 out.lying = read();
225 Object.defineProperty(Navigator.prototype, 'onLine', real);
226 window.DaimondAdmin.status();
227 out.restored = read();
228 return out;
229});
230// `/credits/i`, not `/Credits/`: the row names whose money it is now
231// ("Daimond credits"), because a bare "Credits" beside a funded key of the
232// user's own was read as their own balance being empty.
233check('the rail\'s status row shows the balance the app holds',
234 /credits/i.test(railRow.online) && /8\.40/.test(railRow.online), JSON.stringify(railRow.online));
235check('...and still shows it when the browser wrongly claims to be offline',
236 /credits/i.test(railRow.lying) && /8\.40/.test(railRow.lying), JSON.stringify(railRow.lying));
237check('...and is unchanged once the browser tells the truth again',
238 railRow.restored === railRow.online, JSON.stringify(railRow.restored));
239
240// The row is painted from `DaimondGateway.state()` at render time, not by the `daimond:credits`
241// event: that event fires only when the figure MOVES (an unconditional dispatch closed a
242// request loop -- see verify_spend), so anything that waited for it would show nothing on a
243// balance that never changes. Re-read the same balance: nothing is announced, and the row is
244// right anyway.
245const railQuiet = await page.evaluate(async () => {
246 let fired = 0;
247 const on = () => fired++;
248 window.addEventListener('daimond:credits', on);
249 await window.DaimondGateway.refreshBalance();
250 window.removeEventListener('daimond:credits', on);
251 const n = document.getElementById('astat-account');
252 return { fired, text: n ? n.textContent : 'MISSING' };
253});
254check('a re-read that does not move the balance announces nothing', railQuiet.fired === 0,
255 railQuiet.fired + ' dispatches');
256check('...and the row carries the balance regardless, because it reads the state itself',
257 /8\.40/.test(railQuiet.text), JSON.stringify(railQuiet.text));
258
259// ── The row, on screen ────────────────────────────────────────────────
260await page.click('#astat-model', { force: true });
261await page.waitForTimeout(600);
262const head = await page.$eval('.models-prov.paid .models-prov-head', e => ({
263 name: e.querySelector('.models-nm').textContent.trim(),
264 bal: (e.querySelector('.models-bal') || {}).textContent || '',
265 via: (e.querySelector('.models-via') || {}).textContent || '',
266 count:(e.querySelector('.models-prov-count') || {}).textContent || '',
267 all: e.textContent,
268}));
269check('the row is named for what the user BOUGHT, not for the vendor', head.name === 'Daimond credits', JSON.stringify(head.name));
270check('the vendor is NOT the headline', !head.name.includes('OpenRouter'), JSON.stringify(head.name));
271check('the balance is inline on the row', head.bal === '$8.40 left', JSON.stringify(head.bal));
272check('the host is named anyway — honesty about whose machine it lands on', head.via === 'via OpenRouter', JSON.stringify(head.via));
273check('...read from the gateway\'s URL, so it is whoever was actually minted against',
274 head.all.includes('OpenRouter') && stored.url.includes('openrouter.ai'));
275check('the row counts its models', head.count.trim() === `${CATALOGUE.length} models`, head.count);
276
277// "via" must be secondary, not a second headline: smaller and dimmer than the name it sits beside.
278const rank = await page.$eval('.models-prov.paid .models-prov-head', (e) => {
279 const px = (el) => parseFloat(getComputedStyle(el).fontSize);
280 const nm = e.querySelector('.models-nm'), via = e.querySelector('.models-via'), bal = e.querySelector('.models-bal');
281 // Each of the three must sit whole on one line: a break through any of them is the bug.
282 const lines = (el) => el.getClientRects().length;
283 return {
284 name: px(nm), via: px(via), bal: px(bal),
285 viaCol: getComputedStyle(via).color, nameCol: getComputedStyle(nm).color,
286 wrapped: [nm, bal, via].filter(el => lines(el) > 1).length,
287 };
288});
289check('in a 220px rail, name / balance / host each stay whole — the break falls BETWEEN facts',
290 rank.wrapped === 0, `${rank.wrapped} of 3 broken across lines`);
291check('"via OpenRouter" is SUBDUED: smaller than the name…', rank.via < rank.name, `${rank.via}px vs ${rank.name}px`);
292check('…and dimmer than it', rank.viaCol !== rank.nameCol, `${rank.viaCol} vs ${rank.nameCol}`);
293check('the balance is legible, not fine print', rank.bal >= rank.via, `${rank.bal}px vs ${rank.via}px`);
294await shot(s, 'credits-row');
295
296// ── The second economy: a key of the user's own ───────────────────────
297await connectMock(s);
298await page.waitForTimeout(800);
299await page.click('#astat-model', { force: true });
300await page.waitForTimeout(600);
301const rows = await page.$$eval('.models-prov', els => els.map(e => ({
302 name: e.querySelector('.models-prov-name').firstChild.textContent.trim(),
303 paid: e.classList.contains('paid'),
304})));
305check('both economies are in the one list', rows.length === 2, JSON.stringify(rows.map(r => r.name)));
306check('exactly one row is marked as spending the Daimond balance',
307 rows.filter(r => r.paid).length === 1, JSON.stringify(rows));
308check('the BYOK row is NOT marked — so a row that costs no credits is visibly not this',
309 rows.some(r => !r.paid && r.name !== 'Daimond credits'));
310
311// The rows are told apart by more than a class: the tint must actually differ.
312const tint = await page.$$eval('.models-prov .models-prov-head',
313 els => els.map(e => getComputedStyle(e).backgroundColor));
314check('the paid row is visibly tinted, not just semantically flagged',
315 new Set(tint).size === 2, JSON.stringify(tint));
316
317// The tint is theme variables, not a hardcoded brown: which economy a row belongs to is not a
318// fact about the dark theme, and a user in the light one is owed the same warning.
319for (const theme of ['light', 'lollypop']) {
320 await page.evaluate((t) => window.DaimondTheme.set(t), theme);
321 await page.waitForTimeout(500);
322 const t = await page.$$eval('.models-prov .models-prov-head', els => els.map(e => getComputedStyle(e).backgroundColor));
323 check(`the paid row is still told apart in the ${theme} theme`, new Set(t).size === 2, JSON.stringify(t));
324 const c = await page.$eval('.models-prov.paid .models-nm', e => ({
325 fg: getComputedStyle(e).color,
326 bg: getComputedStyle(e.closest('.models-prov-head')).backgroundColor,
327 }));
328 check(`...and its name is legible against that tint`, c.fg !== c.bg, JSON.stringify(c));
329 await shot(s, `credits-${theme}`);
330}
331await page.evaluate(() => window.DaimondTheme.set('dark'));
332await page.waitForTimeout(400);
333
334// ── The picker: where the two economies actually collide ──────────────
335await page.evaluate(() => {
336 const sel = document.createElement('select');
337 sel.id = 'zz-probe';
338 document.body.appendChild(sel);
339 window.DaimondModels.fillSelect(sel, '', '');
340});
341const opts = await page.$$eval('#zz-probe option', els => els.map(o => ({
342 text: o.textContent, value: o.value, provider: o.dataset.provider,
343 paid: o.dataset.paid, group: o.parentElement.label || '', disabled: o.disabled,
344})));
345const groups = [...new Set(opts.map(o => o.group))];
346check('the picker groups by provider', groups.length === 2, JSON.stringify(groups));
347check('the credits group names the balance and the host in its heading',
348 groups.some(g => g.includes('Daimond credits') && g.includes('$8.40 left') && g.includes('via OpenRouter')),
349 JSON.stringify(groups));
350// The BYOK group keeps the label it has always had. The mark has to be the exception to read as
351// one: relabel every group and the credits row stops standing out from the list it sits in.
352check('the BYOK group is left alone — only the paid row is relabelled',
353 groups.some(g => g === 'Custom provider'), JSON.stringify(groups));
354
355const credOpts = opts.filter(o => o.provider === 'credits');
356check('every credits model is offered', credOpts.length === CATALOGUE.length, `${credOpts.length}`);
357check('EVERY credits option is marked on the option itself, not only in the group heading',
358 credOpts.every(o => o.text.includes('· credits')),
359 JSON.stringify(credOpts.slice(0, 2).map(o => o.text)));
360check('a credits option is not disabled', credOpts.every(o => !o.disabled));
361check('the option carries the economy for a reader that is not a human',
362 credOpts.every(o => o.paid === '1'));
363
364// The collision: one model, two rows, two economies.
365const twins = opts.filter(o => o.value.toLowerCase().endsWith('glm-5p2'));
366check('the same model on two providers appears TWICE — never silently deduped',
367 twins.length === 2, JSON.stringify(twins.map(t => t.value)));
368check('...on different providers, so the key sent with it differs',
369 new Set(twins.map(t => t.provider)).size === 2, JSON.stringify(twins.map(t => t.provider)));
370check('...and the two are distinguishable by their text alone',
371 twins.length === 2 && twins[0].text !== twins[1].text, JSON.stringify(twins.map(t => t.text)));
372check('the credits twin says "credits"', twins.some(t => t.provider === 'credits' && t.text.includes('· credits')));
373check('the BYOK twin says "your key" — the pair is marked on BOTH sides',
374 twins.some(t => t.provider !== 'credits' && t.text.includes('· your key')),
375 JSON.stringify(twins.map(t => t.text)));
376
377// A BYOK model NOT served by the credits row stays unmarked: the mark means something.
378const lone = opts.find(o => o.value === 'mock/thinker');
379check('a BYOK model with no twin carries no mark — an unmarked row reads as "not credits"',
380 !!lone && !lone.text.includes('·'), lone && JSON.stringify(lone.text));
381await page.evaluate(() => document.getElementById('zz-probe').remove());
382
383/// Open the credits row's body, whatever it was showing before. The panel remembers which rows are
384/// expanded, so a blind click is as likely to shut it as to open it.
385async function expandCredits() {
386 if (await page.$('.models-prov.paid .models-prov-body')) return;
387 await page.click('.models-prov.paid .models-prov-head', { force: true });
388 await page.waitForSelector('.models-prov.paid .models-prov-body', { timeout: 5000 });
389 await page.waitForTimeout(300);
390}
391
392// ── A turn: the browser calls the provider, and Daimond is not in the path ──
393await page.click('#astat-model', { force: true });
394await page.waitForTimeout(500);
395await expandCredits();
396await page.$$eval('.models-prov.paid .models-model', (els) => {
397 for (const e of els) if (e.textContent.includes('mock')) { e.click(); return; }
398 els[0].click(); // the catalogue is stubbed; any of it will do
399});
400await page.waitForTimeout(500);
401const def = await page.evaluate(() => window.DaimondModels.getDefault());
402check('a credits model can be starred as the default', def.provider === 'credits', JSON.stringify(def));
403
404const mintsBefore = gw.mints;
405gw.spent = [keyN(1)]; // the key minted at unlock has now spent its cap
406gw.chatHits = 0;
407clearMockLog(); // so the retry's request can be read back off the wire
408
409// The Admin drawer is open from the models work above and overlays the rail
410// (the + button sits under it since the rail gained its divider) — a force
411// click would land on the drawer. Close it the way a user would.
412if (await page.locator('#admin-close').isVisible().catch(() => false)) {
413 await page.click('#admin-close', { force: true });
414 await page.waitForTimeout(200);
415}
416await page.click('#new-session-btn', { force: true });
417await page.waitForTimeout(600);
418// `.tile-start` by class, NOT `button:has-text("Start")` — has-text is a
419// substring match that also hits the BYOK form's hidden "Save & start"
420// button, and clicking that fails as "not visible" (see harness.mjs newChat).
421const start = page.locator('.tile-start').first();
422if (await start.count()) await start.click({ force: true });
423await page.waitForSelector('#chat-input', { state: 'visible', timeout: 10000 });
424await page.fill('#chat-input', '@text Hello from a credits chat.');
425await page.click('#chat-send', { force: true });
426await page.waitForTimeout(6000);
427
428const t1 = await page.evaluate(() => (document.getElementById('chat-output') || {}).innerText || '');
429check('a 401 mid-session re-mints EXACTLY ONCE', gw.mints === mintsBefore + 1, `${gw.mints - mintsBefore} re-mint(s)`);
430check('the turn was retried on the fresh key and completed', gw.chatHits === 2, `${gw.chatHits} provider call(s)`);
431check('the answer is on screen', /hello|mock/i.test(t1), JSON.stringify(t1.slice(-120)));
432check('the recovered 401 was never written into the conversation',
433 !/401|rejected|refused/i.test(t1), JSON.stringify(t1.slice(-160)));
434
435// The retried turn is a REBUILT agent, seeded from the persisted history — which by then already
436// holds the message being sent, because runTurn persists before a token comes back. Read what the
437// model was actually asked: seeding it AND passing it to run_turn would ask the same question
438// twice, and that is invisible from the transcript, which shows the user's own bubble either way.
439const sent = mockLog();
440const asked = sent.length
441 ? sent[sent.length - 1].messages.filter(m => m.role === 'user' && /Hello from a credits chat/.test(contentText(m.content)))
442 : [];
443check('the mock was reached on the fresh key (so the next check can mean something)', sent.length === 1, `${sent.length} request(s)`);
444check('the retry asked the model the question ONCE, not twice (it is excluded from the restore)',
445 asked.length === 1, `${asked.length} copies of the user message on the wire`);
446check('the retried request carried the NEW key, not the spent one',
447 gw.chatHits === 2 && !gw.spent.includes(keyN(2)));
448await shot(s, 'credits-turn');
449
450// ── Out of credits: a thing to do, not an HTTP error to read ──────────
451gw.spent = [keyN(1), keyN(2), keyN(3)]; // every key is refused…
452gw.refuse = true; // …and the account cannot mint another
453gw.bal = 0;
454const mintsBefore2 = gw.mints;
455await page.fill('#chat-input', '@text And again.');
456await page.click('#chat-send', { force: true });
457await page.waitForTimeout(6000);
458const t2 = await page.evaluate(() => (document.getElementById('chat-output') || {}).innerText || '');
459check('a dry account is still only ONE re-mint attempt, not a retry storm',
460 gw.mints === mintsBefore2 + 1, `${gw.mints - mintsBefore2} attempt(s)`);
461check('the user is told to TOP UP, not shown a raw HTTP error',
462 /run out|top up/i.test(t2) && !/\b401\b/.test(t2), JSON.stringify(t2.slice(-200)));
463check('...and is offered the other way out: their own key',
464 /provider key of your own/i.test(t2), JSON.stringify(t2.slice(-140)));
465
466// The row must now say so itself, rather than offering models it cannot run.
467await page.click('#astat-model', { force: true });
468await page.waitForTimeout(600);
469await expandCredits();
470const dry = await page.$eval('.models-prov.paid', e => ({
471 key: (e.querySelector('.models-prov-key') || {}).textContent || '',
472 bal: (e.querySelector('.models-bal') || {}).textContent || '',
473 why: (e.querySelector('.models-why') || {}).textContent || '',
474}));
475check('the row says "no credits" once the balance is gone', dry.key.includes('no credits'), JSON.stringify(dry.key));
476check('a stale balance is not left on screen', dry.bal === '', JSON.stringify(dry.bal));
477check('the GATEWAY\'s own words reach the user, not a flattened one-word state',
478 dry.why.includes('bring your own model key'), JSON.stringify(dry.why.slice(0, 60)));
479const dryRow = await row();
480check('a dry credits row is not ready, so its models are disabled in the picker', dryRow.ready === false);
481check('the row is NOT removed — a user who has just run out must see what happened', !!dryRow);
482await expandCredits();
483const topup = await page.$$eval('.models-prov.paid .models-refetch', els => els.map(e => e.textContent));
484check('and it offers the fix, which is the only thing here a button can fix',
485 topup.some(t => /top up/i.test(t)), JSON.stringify(topup));
486check('a credits row offers no "Remove" — a balance is not the user\'s to delete from a panel',
487 (await page.$$eval('.models-prov.paid .models-remove', els => els.length)) === 0);
488await shot(s, 'credits-dry');
489
490// ── Locking forgets it, because that is the whole of forgetting it ────
491gw.refuse = false; gw.bal = 840; gw.spent = [];
492await page.evaluate(() => window.DaimondModels.lock());
493await page.waitForTimeout(300);
494const afterLock = await page.evaluate(() => ({
495 key: window.DaimondModels.keyFor('credits'),
496 row: (window.DaimondModels.providers() || []).find(p => p.id === 'credits'),
497 st: window.DaimondModels.creditsState(),
498}));
499check('locking forgets the minted key outright', !afterLock.key, JSON.stringify(afterLock.key));
500check('a locked credits row cannot run', afterLock.row.ready === false && afterLock.row.hasKey === false);
501check('a locked row shows no balance', afterLock.st.credits === 0 && afterLock.row.balance === '');
502
503// ── Dispatched agents survive a spent key, and do NOT storm the mint ──
504//
505// A worker's key is frozen when its agent is built, exactly as a chat's is, and it spends the same
506// minted key. Daimond's claim is a team rather than a chat, so a credits user whose chat heals
507// while their agents die on that same exhausted key has the worst of both halves.
508//
509// The dangerous part is the healing, not the failing. The gateway keeps at most ONE live key per
510// account -- minting revokes the last -- so N agents that hit a spent key together must produce
511// ONE mint. N mints would each revoke the one before, and the agents would chase each other round
512// spending real money per lap. That is what is measured here: not "did it recover" but "how many
513// keys did it buy to recover".
514gw.refuse = false; gw.bal = 840; gw.spent = [];
515
516// ── A 426 on the mint: the tab is too old to serve ────────────────────
517// Reachable only because the client names its version, which is the point of sending it. It must
518// reach the updater -- the one thing that can fix a tab that is behind the gateway.
519//
520// `daimond:stale` is a real force-reload, so the loop guard is pre-set to this build (as
521// verify_version.mjs does): otherwise the proof reloads the page out from under the test. That the
522// event turns INTO a reload is verify_updates.mjs's subject; that the mint fires it is this one's.
523await page.evaluate(() => {
524 var b = window.DaimondUpdater && window.DaimondUpdater.booted();
525 try { if (b) sessionStorage.setItem('daimond-forced-from', b); } catch (e) {}
526});
527gw.stale426 = true;
528const stale = await page.evaluate(() => new Promise(resolve => {
529 window.addEventListener('daimond:stale', () => resolve(true), { once: true });
530 setTimeout(() => resolve(false), 3000);
531 window.DaimondModels.remint().catch(() => {});
532}));
533gw.stale426 = false;
534check('a 426 on the mint tells the updater this tab is out of date', stale === true);
535
536await page.reload({ waitUntil: 'domcontentloaded' });
537await signInAs(s, 'credits');
538await page.waitForTimeout(2500);
539
540const liveKey = await page.evaluate(() => window.DaimondModels.keyFor('credits'));
541const genBefore = await page.evaluate(() => window.DaimondModels.creditsGen());
542check('the session is back on a live minted key', !!liveKey && liveKey === keyN(gw.mints));
543
544// Star a credits model as the default: a worker runs on whatever the default is.
545await page.evaluate(() => window.DaimondModels.setDefault('credits', 'anthropic/claude-opus-4.5'));
546await page.waitForTimeout(400);
547
548// A Diamond, whose conductor dispatches three agents in one turn -- the concurrent case.
549await page.click('#new-diamond-btn', { force: true });
550await page.waitForSelector('.dlg-input', { timeout: 10000 });
551await page.fill('.dlg-input', 'Credits worker test');
552await page.click('.dlg-ok', { force: true });
553await page.waitForTimeout(900);
554await page.$$eval('.diamond-box', els => els[0].click());
555await page.waitForTimeout(600);
556
557// The key is spent for WORKERS only: the conductor's own turn runs on the same key, and if it
558// cannot finish it never dispatches anything and the case under test never happens.
559gw.spent = [liveKey];
560gw.onlyFor = WORKER_SAYS;
561const mintsBefore4 = gw.mints;
562clearMockLog();
563await page.fill('#chat-input',
564 '@tools spawn_agent {"name":"w1","task":"@text one"} ;; spawn_agent {"name":"w2","task":"@text two"} ;; spawn_agent {"name":"w3","task":"@text three"}');
565await page.keyboard.press('Enter');
566await page.waitForTimeout(14000);
567
568// One worker = one `.acard`, whose class carries its status (running/done/error).
569const agents = () => page.$$eval('#agents-list .acard', els => els.map(e => ({
570 text: e.innerText || '', cls: e.className || '',
571})));
572let runs = await agents();
573// Vacuously-true checks are worse than no checks: every assertion below is about agents, so
574// nothing below may pass on a pane that has none.
575check('the conductor dispatched three agents (nothing below is true of an empty pane)',
576 runs.length === 3, `${runs.length} agent tile(s)`);
577// The measurement that matters. Not "did it recover" -- "how many keys did it buy to recover".
578//
579// Each worker now holds its OWN slot key (6da5d0b, 2026-07-18): a shared key is
580// exactly what let concurrent workers race the host's stale cap check into an
581// overspend, so three workers refused on their three keys legitimately mint
582// three. "No storm" is therefore ONE mint per worker, not one between them --
583// a fourth mint would be the retry ring this check exists to catch.
584check('THREE refused agents mint ONE key EACH — a slot apiece, and no retry ring',
585 gw.mints === mintsBefore4 + 3, `${gw.mints - mintsBefore4} mint(s) for 3 agents`);
586// Slot 0 is the chat's own key. A worker's refusal must not disturb it, or a
587// fan-out would re-key the conversation that dispatched it.
588check('the chat’s own key is untouched by a worker’s re-mint',
589 (await page.evaluate(() => window.DaimondModels.creditsGen())) === genBefore,
590 `slot 0 generation ${await page.evaluate(() => window.DaimondModels.creditsGen())}, was ${genBefore}`);
591check('every agent recovered on the fresh key rather than dying on the spent one',
592 runs.length === 3 && runs.every(r => !/401|rejected|error/i.test(r.text)),
593 JSON.stringify(runs.map(r => r.text.slice(0, 40))));
594check('every agent finished its task',
595 runs.length === 3 && runs.filter(r => /one|two|three/i.test(r.text)).length === 3,
596 JSON.stringify(runs.map(r => r.text.slice(0, 60))));
597// The workers really did meet the spent key: otherwise this proves nothing about healing.
598const workerReqs = mockLog().filter(m => (m.messages || []).some(x => WORKER_SAYS.test(contentText(x.content))));
599check('the workers did reach the model on the new key (they were really refused first)',
600 workerReqs.length === 3, `${workerReqs.length} worker request(s) landed`);
601await shot(s, 'credits-workers');
602
603// A worker that genuinely cannot heal says "top up", not 401.
604gw.spent = [keyN(gw.mints)]; gw.refuse = true; gw.bal = 0;
605const mintsBefore5 = gw.mints;
606await page.fill('#chat-input', '@tool spawn_agent {"name":"w4","task":"@text four"}');
607await page.keyboard.press('Enter');
608await page.waitForTimeout(12000);
609runs = await agents();
610check('a worker that cannot heal tries ONE mint, not a retry storm',
611 gw.mints === mintsBefore5 + 1, `${gw.mints - mintsBefore5} attempt(s)`);
612check('the failed agent is marked failed, not quietly done',
613 runs.some(r => /\berror\b/.test(r.cls) && /w4/.test(r.text)), JSON.stringify(runs.map(r => r.cls)));
614
615// A worker's message lives behind its own "Read", which is the only way a user sees it -- so it
616// is the way the message is read here too, rather than reaching into the object behind the tile.
617const w4text = await page.evaluate(async () => {
618 const cards = [...document.querySelectorAll('#agents-list .acard')];
619 const card = cards.find(c => /w4/.test(c.innerText));
620 if (!card) return '(no w4 tile)';
621 const read = [...card.querySelectorAll('.abtn')].find(b => b.textContent === 'Read');
622 if (!read) return '(no Read button)';
623 read.click();
624 await new Promise(r => setTimeout(r, 400));
625 const dlg = document.querySelector('.dlg-body, .dlg-text, .dlg');
626 return dlg ? dlg.innerText : '(no dialog)';
627});
628// The words are the GATEWAY's, not a phrase of the client's own -- the same rule
629// the credits row is held to above ("the GATEWAY's own words reach the user").
630// What is asserted is therefore the SUBSTANCE the user must be given: that there
631// are no credits, and the two ways out. Not a raw status code, and not a wording
632// the client would have to keep in step with the host.
633check('...and fails by saying there are no credits, not by showing a raw 401',
634 /credit/i.test(w4text) && !/\b401\b/.test(w4text), JSON.stringify(w4text.slice(-220)));
635check('...and names both ways out: add credits, or bring a key of your own',
636 /add credits/i.test(w4text) && /own (model|provider) key|key of your own/i.test(w4text),
637 JSON.stringify(w4text.slice(-120)));
638await shot(s, 'credits-workers-dry');
639gw.refuse = false; gw.bal = 840; gw.spent = []; gw.onlyFor = null;
640
641// ── A reload holds no key, and mints a new one ────────────────────────
642const mintsBefore3 = gw.mints;
643await page.reload({ waitUntil: 'domcontentloaded' });
644await signInAs(s, 'credits');
645await page.waitForTimeout(2500);
646const after = await page.evaluate(() => ({
647 key: window.DaimondModels.keyFor('credits'),
648 row: (window.DaimondModels.providers() || []).find(p => p.id === 'credits'),
649}));
650check('a reload re-mints rather than reloading a key from disk', gw.mints === mintsBefore3 + 1, `${gw.mints - mintsBefore3}`);
651check('the key after a reload is a NEW one, not the old one back', after.key === keyN(gw.mints) && after.key !== keyN(1),
652 after.key ? after.key.slice(0, 18) + '…' : '(empty)');
653check('the models survive the reload (the row is stored; only the key is not)', after.row.count === CATALOGUE.length);
654
655// The 502s are the gateway this test deliberately does not run. The 401, 402 and 426 are this
656// test's own injections -- a spent key, a dry account, and a tab too old. Everything else is a
657// real complaint.
658const errs = errors(s).filter(e => !/502 \(Bad Gateway\)/.test(e) && !/401 \(Unauthorized\)/.test(e)
659 && !/402 \(Payment Required\)/.test(e) && !/426 \(Upgrade Required\)/.test(e));
660check('no console errors beyond the offline gateway and the injected 401/402/426',
661 errs.length === 0, JSON.stringify(errs.slice(0, 3)));
662await s.close();
663
664// ── A user with no credits gets no row at all ─────────────────────────
665// The panel of somebody who only ever wanted their own key must not grow a row advertising models
666// they cannot run — and `providers().length` decides whether a first run opens the add-a-key form.
667gw.bal = 0; gw.refuse = true; gw.mints = 0;
668const b = await open({ name: 'creditsB', signIn: false, connect: false });
669await stubGateway(b.page);
670await signInAs(b, 'creditsB');
671await b.page.waitForTimeout(2000);
672const none = await b.page.evaluate(() => window.DaimondModels.providers().map(p => p.id));
673check('a zero balance mints nothing at all', gw.mints === 0, `${gw.mints} mint(s)`);
674check('and adds no row: a BYOK-only user sees no credits row they never asked for',
675 none.length === 0, JSON.stringify(none));
676
677// A balance of nothing is still a balance: $0.00 is what the account holds, and saying so is the
678// difference between "you have none" and "we cannot tell you".
679const zero = await b.page.evaluate(() => {
680 window.DaimondAdmin.status();
681 const n = document.getElementById('astat-account');
682 return n ? n.textContent : 'MISSING';
683});
684// By MEANING, not by the old label. The row used to read "Credits"; it now names
685// whose money it is ("Daimond credits"), because a user funding their work with
686// their own key read a bare "Credits $0.00" as being broke. What must still be
687// true is that a zero balance is stated as a figure rather than left blank.
688check('a zero balance is still reported as a figure',
689 /credits/i.test(zero) && /0\.00/.test(zero),
690 JSON.stringify(zero));
691
692// And with no session there is no figure to report, so the row says which of the three reasons
693// it is -- the browser reporting no network, the service unreachable, or no account at all. The
694// point of the row is that it is never blank.
695const noSession = await b.page.evaluate(async () => {
696 await window.DaimondGateway.logout();
697 window.DaimondAdmin.status();
698 const n = document.getElementById('astat-account');
699 const out = { away: n ? n.textContent : 'MISSING' };
700 const real = Object.getOwnPropertyDescriptor(Navigator.prototype, 'onLine');
701 Object.defineProperty(Navigator.prototype, 'onLine', { configurable: true, get: () => false });
702 window.DaimondAdmin.status();
703 out.offline = n ? n.textContent : 'MISSING';
704 Object.defineProperty(Navigator.prototype, 'onLine', real);
705 return out;
706});
707check('with no session the row says there is no account', /No credits account/.test(noSession.away),
708 JSON.stringify(noSession.away));
709check('...and names the browser\'s own offline claim when there is nothing to contradict it',
710 /Offline/.test(noSession.offline), JSON.stringify(noSession.offline));
711await b.close();
712
713console.log(`\n${ok.length} passed, ${bad.length} failed`);
714if (bad.length) { console.log('FAILED:\n - ' + bad.join('\n - ')); process.exit(1); }