Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/mockcap.mjs

10.2 KiB, 1 run

created by r2519314175:27, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1// mockcap.mjs — a mock provider that HONOURS max_tokens, and says what it saw.
2//
3// dev/mockllm.mjs ignores `max_tokens` entirely, so nothing driven through it
4// can show what the reply-length cap actually does to a turn. This one obeys
5// it, which is the whole point: a reply longer than the cap is CUT, and when
6// the thing being cut is a tool call's arguments — themselves a JSON string —
7// what the client receives is a malformed tool call, not a short file. That is
8// the failure a 4096-token cap produced on any real file_write, and it can only
9// be demonstrated against a provider that enforces the cap.
10//
11// node dev/mockcap.mjs [port] # default 9098
12//
13// The port is `DAIMOND_CAP_PORT` (or argv[2]) and the log `DAIMOND_CAP_LOG`, so
14// this can be one fixture of a numbered world -- see dev/world.sh. A shared log
15// is the trap the ports alone do not close: two suites appending to one file make
16// every assertion read another agent's traffic.
17//
18// Every request is appended to the log as one JSON line carrying the
19// `max_tokens` the client asked for, so a test can assert on what Daimond SENT
20// as well as on what it did with the reply.
21//
22// ── Directives (in the user's message, as with mockllm) ───────────────────
23//
24// @write <lines> [path] one `file_write` call whose content is <lines>
25// numbered lines. Cut at the cap if it does not fit.
26// @rounds <n> an n-round tool loop: n-1 `file_list` calls then a
27// text reply. Round k reports prompt_tokens of
28// ROUND_PROMPT*(k+1), so the rounds are all different
29// and a meter that sums them is obvious.
30// @refuse-cap <n> refuse any request asking for more than <n>, with
31// the wording a real provider uses; answer the rest.
32// @text <words> a plain reply.
33//
34// Tokens are counted at CHARS_PER_TOKEN characters each. That is a stand-in for
35// a real tokeniser, and it is the right kind of stand-in: the property under
36// test is "the cap is enforced and the cut lands mid-string", which does not
37// depend on the tokeniser being exact.
38
39import http from 'node:http';
40import fs from 'node:fs';
41import path from 'node:path';
42import { fileURLToPath } from 'node:url';
43
44const HERE = path.dirname(fileURLToPath(import.meta.url));
45const LOG = process.env.DAIMOND_CAP_LOG || path.join(HERE, 'mockcap.log');
46const PORT = Number(process.argv[2] || process.env.DAIMOND_CAP_PORT || 9098);
47
48/// Characters per token, for the cap arithmetic.
49const CHARS_PER_TOKEN = 4;
50
51/// Prompt tokens round 0 reports; round k reports this times (k+1).
52export const ROUND_PROMPT = 5000;
53
54const MODELS = ['cap/plain', 'claude-opus-4-1', 'claude-opus-5'];
55
56const log = (entry) => {
57 try { fs.appendFileSync(LOG, JSON.stringify(entry) + '\n'); }
58 catch (e) { console.error('mockcap: could not write log:', e.message); }
59};
60
61const cors = (res) => {
62 res.setHeader('Access-Control-Allow-Origin', '*');
63 res.setHeader('Access-Control-Allow-Headers', '*');
64 res.setHeader('Access-Control-Allow-Methods', 'GET,POST,OPTIONS');
65};
66
67const lastUser = (messages) => {
68 for (let i = messages.length - 1; i >= 0; i--) {
69 if (messages[i].role === 'user') {
70 const c = messages[i].content;
71 return typeof c === 'string' ? c
72 : Array.isArray(c) ? c.map(p => p.text || '').join(' ') : '';
73 }
74 }
75 return '';
76};
77
78// Tool replies since the last user message — the round number within this turn.
79const toolRounds = (messages) => {
80 let at = -1;
81 for (let i = messages.length - 1; i >= 0; i--) {
82 if (messages[i].role === 'user') { at = i; break; }
83 }
84 return messages.slice(at + 1).filter(m => m.role === 'tool').length;
85};
86
87const parseDirective = (text) => {
88 const t = (text || '').trim();
89 if (!t.startsWith('@')) return { kind: 'plain', rest: t };
90 const sp = t.indexOf(' ');
91 return {
92 kind: (sp === -1 ? t : t.slice(0, sp)).slice(1),
93 rest: sp === -1 ? '' : t.slice(sp + 1).trim(),
94 };
95};
96
97/// A file body of `n` numbered lines — long enough that its own length, not the
98/// wrapper, is what decides whether the reply fits.
99const body = (n) => Array.from({ length: n },
100 (_, i) => `\tconst line_${i + 1} = compute(${i + 1}); // generated line ${i + 1}`).join('\n');
101
102const sendJson = (res, obj, code = 200) => {
103 const s = JSON.stringify(obj);
104 cors(res);
105 res.writeHead(code, { 'content-type': 'application/json', 'content-length': Buffer.byteLength(s) });
106 res.end(s);
107};
108
109/// Decide the turn from the directive, the round, and the cap the client asked
110/// for. `cut` is true when the cap bit — the client should see `length`.
111const plan = (messages, maxTokens) => {
112 const d = parseDirective(lastUser(messages));
113 const rounds = toolRounds(messages);
114 const budget = Math.max(1, (maxTokens || 4096)) * CHARS_PER_TOKEN;
115
116 switch (d.kind) {
117 case 'write': {
118 if (rounds > 0) return { text: 'Written.', prompt: ROUND_PROMPT };
119 const [linesRaw, pathRaw] = d.rest.split(/\s+/);
120 const lines = Math.max(1, Number(linesRaw) || 400);
121 const file = pathRaw || 'generated.js';
122 const args = JSON.stringify({ path: file, content: body(lines) });
123 const cut = args.length > budget;
124 return {
125 calls: [{ id: 'call_1', type: 'function',
126 function: { name: 'file_write', arguments: cut ? args.slice(0, budget) : args } }],
127 cut,
128 prompt: ROUND_PROMPT,
129 };
130 }
131
132 case 'rounds': {
133 const n = Math.max(1, Number(d.rest) || 3);
134 // Round k's prompt is (k+1) times the base, so every round differs and
135 // the sum is unmistakably not the last one.
136 const prompt = ROUND_PROMPT * (rounds + 1);
137 if (rounds < n - 1) {
138 return {
139 calls: [{ id: `call_${rounds + 1}`, type: 'function',
140 function: { name: 'file_list', arguments: JSON.stringify({ path: '.' }) } }],
141 prompt,
142 };
143 }
144 return { text: `Done after ${n} rounds.`, prompt };
145 }
146
147 case 'refuse-cap': {
148 const limit = Math.max(1, Number(d.rest) || 8192);
149 if ((maxTokens || 0) > limit) return { refuse: limit };
150 if (rounds > 0) return { text: 'Fine at this length.', prompt: ROUND_PROMPT };
151 return { text: 'Fine at this length.', prompt: ROUND_PROMPT };
152 }
153
154 case 'text':
155 default:
156 if (rounds > 0) return { text: 'Done.', prompt: ROUND_PROMPT };
157 return { text: d.rest || 'Right.', prompt: ROUND_PROMPT };
158 }
159};
160
161const usageOf = (p, outChars) => ({
162 prompt_tokens: p.prompt || ROUND_PROMPT,
163 completion_tokens: Math.ceil(outChars / CHARS_PER_TOKEN),
164 total_tokens: (p.prompt || ROUND_PROMPT) + Math.ceil(outChars / CHARS_PER_TOKEN),
165});
166
167/// Stream the turn as SSE, in the shape the OpenAI-compatible wire uses.
168const stream = async (res, model, p, maxTokens) => {
169 cors(res);
170 res.writeHead(200, {
171 'content-type': 'text/event-stream', 'cache-control': 'no-cache', 'connection': 'keep-alive',
172 });
173 const send = (o) => res.write(`data: ${JSON.stringify(o)}\n\n`);
174 const frame = (delta, finish = null) => ({
175 id: 'chatcmpl-cap', object: 'chat.completion.chunk', created: 1700000000, model,
176 choices: [{ index: 0, delta, finish_reason: finish }],
177 });
178
179 send(frame({ role: 'assistant', content: '' }));
180 let outChars = 0;
181
182 if (p.calls) {
183 p.calls.forEach((c, i) => {
184 send(frame({ tool_calls: [{ index: i, id: c.id, type: 'function',
185 function: { name: c.function.name, arguments: '' } }] }));
186 });
187 for (const [i, c] of p.calls.entries()) {
188 const a = c.function.arguments;
189 outChars += a.length;
190 // Dribbled in pieces, as a provider does, so an accumulator that keeps
191 // only the last fragment is caught here too.
192 const step = Math.max(1, Math.ceil(a.length / 6));
193 for (let at = 0; at < a.length; at += step) {
194 send(frame({ tool_calls: [{ index: i, function: { arguments: a.slice(at, at + step) } }] }));
195 }
196 }
197 // `length` when the cap bit — the same finish reason a real provider sends,
198 // and the one nothing in the client currently reads.
199 send(frame({}, p.cut ? 'length' : 'tool_calls'));
200 } else {
201 const words = (p.text || '').split(' ');
202 for (const w of words) {
203 if (res.writableEnded || res.destroyed) return;
204 outChars += w.length + 1;
205 send(frame({ content: w + ' ' }));
206 }
207 send(frame({}, 'stop'));
208 }
209
210 send({ id: 'chatcmpl-cap', object: 'chat.completion.chunk', model, choices: [],
211 usage: usageOf(p, outChars) });
212 res.write('data: [DONE]\n\n');
213 res.end();
214};
215
216const server = http.createServer((req, res) => {
217 if (req.method === 'OPTIONS') { cors(res); res.writeHead(204); return res.end(); }
218
219 if (req.method === 'GET' && req.url.startsWith('/v1/models')) {
220 return sendJson(res, { object: 'list', data: MODELS.map(id => ({ id, object: 'model' })) });
221 }
222 if (req.method !== 'POST') { cors(res); res.writeHead(404); return res.end(); }
223
224 let raw = '';
225 req.on('data', c => { raw += c; });
226 req.on('end', async () => {
227 let payload;
228 try { payload = JSON.parse(raw); }
229 catch { return sendJson(res, { error: { message: 'mockcap: body was not JSON' } }, 400); }
230
231 const messages = payload.messages || [];
232 const maxTokens = payload.max_tokens;
233 // The whole point of the log: what the client ASKED FOR, per request.
234 log({ at: new Date().toISOString(), model: payload.model, max_tokens: maxTokens,
235 stream: !!payload.stream, rounds: toolRounds(messages) });
236
237 const p = plan(messages, maxTokens);
238
239 if (p.refuse) {
240 // The wording a real OpenAI-compatible provider uses when the reply
241 // length asked for is above what the model will generate.
242 return sendJson(res, { error: {
243 message: `max_tokens is too large: ${maxTokens}. This model supports at most ${p.refuse} completion tokens.`,
244 type: 'invalid_request_error', param: 'max_tokens',
245 } }, 400);
246 }
247 if (payload.stream) return stream(res, payload.model || 'cap/plain', p, maxTokens);
248 return sendJson(res, {
249 id: 'chatcmpl-cap', object: 'chat.completion', created: 1700000000,
250 model: payload.model || 'cap/plain',
251 choices: [{ index: 0,
252 message: p.calls ? { role: 'assistant', content: null, tool_calls: p.calls }
253 : { role: 'assistant', content: p.text },
254 finish_reason: p.cut ? 'length' : (p.calls ? 'tool_calls' : 'stop') }],
255 usage: usageOf(p, 0),
256 });
257 });
258});
259
260server.listen(PORT, '127.0.0.1', () => {
261 console.log(`mockcap: http://127.0.0.1:${PORT}/v1/chat/completions (log: ${LOG})`);
262});