Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/probe_shape.mjs

3.7 KiB, 1 run

created by r2519314175:73, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1// What does the SIZE of Daimond's request cost, at a real model?
2//
3// dev/probe_provider.mjs replays the loop as it is. This asks the counterfactual:
4// the same model, the same moment, with the request cut down. Four shapes, in
5// order of what they take away:
6//
7// bare a two-line conversation and no tools -- the model at its fastest
8// convo Daimond's conversation, no tool schemas
9// half Daimond's conversation with half the schemas
10// full what Daimond actually sends
11//
12// The gap between `bare` and `full` is what the app's own prompt costs in time,
13// separated from anything the provider or the network does.
14//
15// OPENROUTER_API_KEY=... node dev/probe_shape.mjs <captured.log> <model> [--reps 3]
16
17import fs from 'node:fs';
18
19const LOG = process.argv[2];
20const MODEL = process.argv[3];
21const arg = (n, d) => { const i = process.argv.indexOf('--' + n); return i === -1 ? d : process.argv[i + 1]; };
22const REPS = Number(arg('reps', 3));
23const URL_ = arg('url', 'https://openrouter.ai/api/v1/chat/completions');
24const KEY = process.env.OPENROUTER_API_KEY;
25if (!LOG || !MODEL || !KEY) {
26 console.error('usage: OPENROUTER_API_KEY=... probe_shape.mjs <captured.log> <model>');
27 process.exit(2);
28}
29const first = fs.readFileSync(LOG, 'utf8').trim().split('\n')
30 .filter(Boolean).map(l => JSON.parse(l)).find(l => l.raw);
31const body = JSON.parse(first.raw);
32const now = () => Number(process.hrtime.bigint() / 1000n) / 1000;
33
34const shapes = {
35 bare: { messages: [{ role: 'user', content: 'Reply with the single word: ok.' }] },
36 convo: { messages: body.messages },
37 half: { messages: body.messages, tools: body.tools.slice(0, Math.ceil(body.tools.length / 2)),
38 tool_choice: 'auto' },
39 full: { messages: body.messages, tools: body.tools, tool_choice: 'auto' },
40};
41
42async function once(shape) {
43 const send = { model: MODEL, max_tokens: 32, stream: true,
44 stream_options: { include_usage: true }, usage: { include: true }, ...shape };
45 const raw = JSON.stringify(send);
46 const t0 = now();
47 const r = await fetch(URL_, { method: 'POST',
48 headers: { 'Content-Type': 'application/json', 'Authorization': `Bearer ${KEY}` }, body: raw });
49 if (!r.ok) return { err: `${r.status} ${(await r.text()).slice(0, 120)}` };
50 const reader = r.body.getReader();
51 const dec = new TextDecoder();
52 let ttft = 0, buf = '', usage = null;
53 for (;;) {
54 const { done, value } = await reader.read();
55 if (done) break;
56 if (!ttft) ttft = now();
57 buf += dec.decode(value, { stream: true });
58 let nl;
59 while ((nl = buf.indexOf('\n')) !== -1) {
60 const line = buf.slice(0, nl).trim(); buf = buf.slice(nl + 1);
61 if (!line.startsWith('data: ')) continue;
62 const d = line.slice(6); if (d === '[DONE]') continue;
63 try { const j = JSON.parse(d); if (j.usage) usage = j.usage; } catch {}
64 }
65 }
66 const u = usage || {}, det = u.prompt_tokens_details || {};
67 return { kb: +(Buffer.byteLength(raw, 'utf8') / 1024).toFixed(1), ttft: Math.round(ttft - t0),
68 total: Math.round(now() - t0), ptok: u.prompt_tokens || 0, cached: det.cached_tokens || 0,
69 cost: u.cost || 0 };
70}
71
72console.log(`model ${MODEL}`);
73console.log('shape reqKB promptTok cached ttft(ms, each rep)');
74let spend = 0;
75for (const [name, shape] of Object.entries(shapes)) {
76 const runs = [];
77 let last = {};
78 for (let i = 0; i < REPS; i++) {
79 const r = await once(shape);
80 if (r.err) { console.log(`${name.padEnd(7)} ERROR ${r.err}`); break; }
81 spend += r.cost; runs.push(r.ttft); last = r;
82 }
83 if (!runs.length) continue;
84 console.log([name.padEnd(7), String(last.kb).padStart(6), String(last.ptok).padStart(10),
85 String(last.cached).padStart(8), ' ' + runs.join(', ')].join(''));
86}
87console.log(`\nspent on ${MODEL}: $${spend.toFixed(4)}`);