oxedyne/daimond/dev/mockcap.mjs
10.2 KiB, 1 run
created by r2519314175:27, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // mockcap.mjs — a mock provider that HONOURS max_tokens, and says what it saw. |
| 2 | // |
| 3 | // dev/mockllm.mjs ignores `max_tokens` entirely, so nothing driven through it |
| 4 | // can show what the reply-length cap actually does to a turn. This one obeys |
| 5 | // it, which is the whole point: a reply longer than the cap is CUT, and when |
| 6 | // the thing being cut is a tool call's arguments — themselves a JSON string — |
| 7 | // what the client receives is a malformed tool call, not a short file. That is |
| 8 | // the failure a 4096-token cap produced on any real file_write, and it can only |
| 9 | // be demonstrated against a provider that enforces the cap. |
| 10 | // |
| 11 | // node dev/mockcap.mjs [port] # default 9098 |
| 12 | // |
| 13 | // The port is `DAIMOND_CAP_PORT` (or argv[2]) and the log `DAIMOND_CAP_LOG`, so |
| 14 | // this can be one fixture of a numbered world -- see dev/world.sh. A shared log |
| 15 | // is the trap the ports alone do not close: two suites appending to one file make |
| 16 | // every assertion read another agent's traffic. |
| 17 | // |
| 18 | // Every request is appended to the log as one JSON line carrying the |
| 19 | // `max_tokens` the client asked for, so a test can assert on what Daimond SENT |
| 20 | // as well as on what it did with the reply. |
| 21 | // |
| 22 | // ── Directives (in the user's message, as with mockllm) ─────────────────── |
| 23 | // |
| 24 | // @write <lines> [path] one `file_write` call whose content is <lines> |
| 25 | // numbered lines. Cut at the cap if it does not fit. |
| 26 | // @rounds <n> an n-round tool loop: n-1 `file_list` calls then a |
| 27 | // text reply. Round k reports prompt_tokens of |
| 28 | // ROUND_PROMPT*(k+1), so the rounds are all different |
| 29 | // and a meter that sums them is obvious. |
| 30 | // @refuse-cap <n> refuse any request asking for more than <n>, with |
| 31 | // the wording a real provider uses; answer the rest. |
| 32 | // @text <words> a plain reply. |
| 33 | // |
| 34 | // Tokens are counted at CHARS_PER_TOKEN characters each. That is a stand-in for |
| 35 | // a real tokeniser, and it is the right kind of stand-in: the property under |
| 36 | // test is "the cap is enforced and the cut lands mid-string", which does not |
| 37 | // depend on the tokeniser being exact. |
| 38 | |
| 39 | import http from 'node:http'; |
| 40 | import fs from 'node:fs'; |
| 41 | import path from 'node:path'; |
| 42 | import { fileURLToPath } from 'node:url'; |
| 43 | |
| 44 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 45 | const LOG = process.env.DAIMOND_CAP_LOG || path.join(HERE, 'mockcap.log'); |
| 46 | const PORT = Number(process.argv[2] || process.env.DAIMOND_CAP_PORT || 9098); |
| 47 | |
| 48 | /// Characters per token, for the cap arithmetic. |
| 49 | const CHARS_PER_TOKEN = 4; |
| 50 | |
| 51 | /// Prompt tokens round 0 reports; round k reports this times (k+1). |
| 52 | export const ROUND_PROMPT = 5000; |
| 53 | |
| 54 | const MODELS = ['cap/plain', 'claude-opus-4-1', 'claude-opus-5']; |
| 55 | |
| 56 | const log = (entry) => { |
| 57 | try { fs.appendFileSync(LOG, JSON.stringify(entry) + '\n'); } |
| 58 | catch (e) { console.error('mockcap: could not write log:', e.message); } |
| 59 | }; |
| 60 | |
| 61 | const cors = (res) => { |
| 62 | res.setHeader('Access-Control-Allow-Origin', '*'); |
| 63 | res.setHeader('Access-Control-Allow-Headers', '*'); |
| 64 | res.setHeader('Access-Control-Allow-Methods', 'GET,POST,OPTIONS'); |
| 65 | }; |
| 66 | |
| 67 | const lastUser = (messages) => { |
| 68 | for (let i = messages.length - 1; i >= 0; i--) { |
| 69 | if (messages[i].role === 'user') { |
| 70 | const c = messages[i].content; |
| 71 | return typeof c === 'string' ? c |
| 72 | : Array.isArray(c) ? c.map(p => p.text || '').join(' ') : ''; |
| 73 | } |
| 74 | } |
| 75 | return ''; |
| 76 | }; |
| 77 | |
| 78 | // Tool replies since the last user message — the round number within this turn. |
| 79 | const toolRounds = (messages) => { |
| 80 | let at = -1; |
| 81 | for (let i = messages.length - 1; i >= 0; i--) { |
| 82 | if (messages[i].role === 'user') { at = i; break; } |
| 83 | } |
| 84 | return messages.slice(at + 1).filter(m => m.role === 'tool').length; |
| 85 | }; |
| 86 | |
| 87 | const parseDirective = (text) => { |
| 88 | const t = (text || '').trim(); |
| 89 | if (!t.startsWith('@')) return { kind: 'plain', rest: t }; |
| 90 | const sp = t.indexOf(' '); |
| 91 | return { |
| 92 | kind: (sp === -1 ? t : t.slice(0, sp)).slice(1), |
| 93 | rest: sp === -1 ? '' : t.slice(sp + 1).trim(), |
| 94 | }; |
| 95 | }; |
| 96 | |
| 97 | /// A file body of `n` numbered lines — long enough that its own length, not the |
| 98 | /// wrapper, is what decides whether the reply fits. |
| 99 | const body = (n) => Array.from({ length: n }, |
| 100 | (_, i) => `\tconst line_${i + 1} = compute(${i + 1}); // generated line ${i + 1}`).join('\n'); |
| 101 | |
| 102 | const sendJson = (res, obj, code = 200) => { |
| 103 | const s = JSON.stringify(obj); |
| 104 | cors(res); |
| 105 | res.writeHead(code, { 'content-type': 'application/json', 'content-length': Buffer.byteLength(s) }); |
| 106 | res.end(s); |
| 107 | }; |
| 108 | |
| 109 | /// Decide the turn from the directive, the round, and the cap the client asked |
| 110 | /// for. `cut` is true when the cap bit — the client should see `length`. |
| 111 | const plan = (messages, maxTokens) => { |
| 112 | const d = parseDirective(lastUser(messages)); |
| 113 | const rounds = toolRounds(messages); |
| 114 | const budget = Math.max(1, (maxTokens || 4096)) * CHARS_PER_TOKEN; |
| 115 | |
| 116 | switch (d.kind) { |
| 117 | case 'write': { |
| 118 | if (rounds > 0) return { text: 'Written.', prompt: ROUND_PROMPT }; |
| 119 | const [linesRaw, pathRaw] = d.rest.split(/\s+/); |
| 120 | const lines = Math.max(1, Number(linesRaw) || 400); |
| 121 | const file = pathRaw || 'generated.js'; |
| 122 | const args = JSON.stringify({ path: file, content: body(lines) }); |
| 123 | const cut = args.length > budget; |
| 124 | return { |
| 125 | calls: [{ id: 'call_1', type: 'function', |
| 126 | function: { name: 'file_write', arguments: cut ? args.slice(0, budget) : args } }], |
| 127 | cut, |
| 128 | prompt: ROUND_PROMPT, |
| 129 | }; |
| 130 | } |
| 131 | |
| 132 | case 'rounds': { |
| 133 | const n = Math.max(1, Number(d.rest) || 3); |
| 134 | // Round k's prompt is (k+1) times the base, so every round differs and |
| 135 | // the sum is unmistakably not the last one. |
| 136 | const prompt = ROUND_PROMPT * (rounds + 1); |
| 137 | if (rounds < n - 1) { |
| 138 | return { |
| 139 | calls: [{ id: `call_${rounds + 1}`, type: 'function', |
| 140 | function: { name: 'file_list', arguments: JSON.stringify({ path: '.' }) } }], |
| 141 | prompt, |
| 142 | }; |
| 143 | } |
| 144 | return { text: `Done after ${n} rounds.`, prompt }; |
| 145 | } |
| 146 | |
| 147 | case 'refuse-cap': { |
| 148 | const limit = Math.max(1, Number(d.rest) || 8192); |
| 149 | if ((maxTokens || 0) > limit) return { refuse: limit }; |
| 150 | if (rounds > 0) return { text: 'Fine at this length.', prompt: ROUND_PROMPT }; |
| 151 | return { text: 'Fine at this length.', prompt: ROUND_PROMPT }; |
| 152 | } |
| 153 | |
| 154 | case 'text': |
| 155 | default: |
| 156 | if (rounds > 0) return { text: 'Done.', prompt: ROUND_PROMPT }; |
| 157 | return { text: d.rest || 'Right.', prompt: ROUND_PROMPT }; |
| 158 | } |
| 159 | }; |
| 160 | |
| 161 | const usageOf = (p, outChars) => ({ |
| 162 | prompt_tokens: p.prompt || ROUND_PROMPT, |
| 163 | completion_tokens: Math.ceil(outChars / CHARS_PER_TOKEN), |
| 164 | total_tokens: (p.prompt || ROUND_PROMPT) + Math.ceil(outChars / CHARS_PER_TOKEN), |
| 165 | }); |
| 166 | |
| 167 | /// Stream the turn as SSE, in the shape the OpenAI-compatible wire uses. |
| 168 | const stream = async (res, model, p, maxTokens) => { |
| 169 | cors(res); |
| 170 | res.writeHead(200, { |
| 171 | 'content-type': 'text/event-stream', 'cache-control': 'no-cache', 'connection': 'keep-alive', |
| 172 | }); |
| 173 | const send = (o) => res.write(`data: ${JSON.stringify(o)}\n\n`); |
| 174 | const frame = (delta, finish = null) => ({ |
| 175 | id: 'chatcmpl-cap', object: 'chat.completion.chunk', created: 1700000000, model, |
| 176 | choices: [{ index: 0, delta, finish_reason: finish }], |
| 177 | }); |
| 178 | |
| 179 | send(frame({ role: 'assistant', content: '' })); |
| 180 | let outChars = 0; |
| 181 | |
| 182 | if (p.calls) { |
| 183 | p.calls.forEach((c, i) => { |
| 184 | send(frame({ tool_calls: [{ index: i, id: c.id, type: 'function', |
| 185 | function: { name: c.function.name, arguments: '' } }] })); |
| 186 | }); |
| 187 | for (const [i, c] of p.calls.entries()) { |
| 188 | const a = c.function.arguments; |
| 189 | outChars += a.length; |
| 190 | // Dribbled in pieces, as a provider does, so an accumulator that keeps |
| 191 | // only the last fragment is caught here too. |
| 192 | const step = Math.max(1, Math.ceil(a.length / 6)); |
| 193 | for (let at = 0; at < a.length; at += step) { |
| 194 | send(frame({ tool_calls: [{ index: i, function: { arguments: a.slice(at, at + step) } }] })); |
| 195 | } |
| 196 | } |
| 197 | // `length` when the cap bit — the same finish reason a real provider sends, |
| 198 | // and the one nothing in the client currently reads. |
| 199 | send(frame({}, p.cut ? 'length' : 'tool_calls')); |
| 200 | } else { |
| 201 | const words = (p.text || '').split(' '); |
| 202 | for (const w of words) { |
| 203 | if (res.writableEnded || res.destroyed) return; |
| 204 | outChars += w.length + 1; |
| 205 | send(frame({ content: w + ' ' })); |
| 206 | } |
| 207 | send(frame({}, 'stop')); |
| 208 | } |
| 209 | |
| 210 | send({ id: 'chatcmpl-cap', object: 'chat.completion.chunk', model, choices: [], |
| 211 | usage: usageOf(p, outChars) }); |
| 212 | res.write('data: [DONE]\n\n'); |
| 213 | res.end(); |
| 214 | }; |
| 215 | |
| 216 | const server = http.createServer((req, res) => { |
| 217 | if (req.method === 'OPTIONS') { cors(res); res.writeHead(204); return res.end(); } |
| 218 | |
| 219 | if (req.method === 'GET' && req.url.startsWith('/v1/models')) { |
| 220 | return sendJson(res, { object: 'list', data: MODELS.map(id => ({ id, object: 'model' })) }); |
| 221 | } |
| 222 | if (req.method !== 'POST') { cors(res); res.writeHead(404); return res.end(); } |
| 223 | |
| 224 | let raw = ''; |
| 225 | req.on('data', c => { raw += c; }); |
| 226 | req.on('end', async () => { |
| 227 | let payload; |
| 228 | try { payload = JSON.parse(raw); } |
| 229 | catch { return sendJson(res, { error: { message: 'mockcap: body was not JSON' } }, 400); } |
| 230 | |
| 231 | const messages = payload.messages || []; |
| 232 | const maxTokens = payload.max_tokens; |
| 233 | // The whole point of the log: what the client ASKED FOR, per request. |
| 234 | log({ at: new Date().toISOString(), model: payload.model, max_tokens: maxTokens, |
| 235 | stream: !!payload.stream, rounds: toolRounds(messages) }); |
| 236 | |
| 237 | const p = plan(messages, maxTokens); |
| 238 | |
| 239 | if (p.refuse) { |
| 240 | // The wording a real OpenAI-compatible provider uses when the reply |
| 241 | // length asked for is above what the model will generate. |
| 242 | return sendJson(res, { error: { |
| 243 | message: `max_tokens is too large: ${maxTokens}. This model supports at most ${p.refuse} completion tokens.`, |
| 244 | type: 'invalid_request_error', param: 'max_tokens', |
| 245 | } }, 400); |
| 246 | } |
| 247 | if (payload.stream) return stream(res, payload.model || 'cap/plain', p, maxTokens); |
| 248 | return sendJson(res, { |
| 249 | id: 'chatcmpl-cap', object: 'chat.completion', created: 1700000000, |
| 250 | model: payload.model || 'cap/plain', |
| 251 | choices: [{ index: 0, |
| 252 | message: p.calls ? { role: 'assistant', content: null, tool_calls: p.calls } |
| 253 | : { role: 'assistant', content: p.text }, |
| 254 | finish_reason: p.cut ? 'length' : (p.calls ? 'tool_calls' : 'stop') }], |
| 255 | usage: usageOf(p, 0), |
| 256 | }); |
| 257 | }); |
| 258 | }); |
| 259 | |
| 260 | server.listen(PORT, '127.0.0.1', () => { |
| 261 | console.log(`mockcap: http://127.0.0.1:${PORT}/v1/chat/completions (log: ${LOG})`); |
| 262 | }); |