oxedyne/daimond/dev/ctxmock.mjs
7.2 KiB, 1 run
created by r2519314175:15, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // A mock provider that behaves like a real one about its context window. |
| 2 | // |
| 3 | // dev/mockllm.mjs accepts a request of any size, so nothing in the suite has ever |
| 4 | // exercised the failure this compactor exists to fix. This one refuses, with the |
| 5 | // status and the wording an OpenAI-compatible provider actually uses: |
| 6 | // |
| 7 | // 400 {"error":{"message":"This model's maximum context length is N tokens, |
| 8 | // however you requested M tokens", "code":"context_length_exceeded"}} |
| 9 | // |
| 10 | // node ctxmock.mjs [port] [limit-tokens] |
| 11 | // |
| 12 | // Directives, as dev/mockllm.mjs spells them, plus: |
| 13 | // @big <kb> a reply of roughly <kb> kilobytes, streamed fast |
| 14 | // |
| 15 | // Every request is logged as a JSON line to ctxmock.log, whole, so a test can check |
| 16 | // what the model was really shown -- including whether the tool calls in it were paired. |
| 17 | |
| 18 | import http from 'node:http'; |
| 19 | import fs from 'node:fs'; |
| 20 | import path from 'node:path'; |
| 21 | import { fileURLToPath } from 'node:url'; |
| 22 | |
| 23 | const HERE = path.dirname(fileURLToPath(import.meta.url)); |
| 24 | // The port and the log are settable so this can belong to a numbered world (see |
| 25 | // dev/world.sh); a shared log makes every assertion read another agent's traffic. |
| 26 | const LOG = process.env.DAIMOND_CTX_LOG || path.join(HERE, 'ctxmock.log'); |
| 27 | const PORT = Number(process.argv[2] || process.env.DAIMOND_CTX_PORT || 9188); |
| 28 | const LIMIT = Number(process.argv[3] || 12000); // tokens |
| 29 | |
| 30 | const MODELS = ['mock/fast', 'mock/thinker']; |
| 31 | |
| 32 | const log = (e) => { try { fs.appendFileSync(LOG, JSON.stringify(e) + '\n'); } catch {} }; |
| 33 | |
| 34 | const cors = (res) => { |
| 35 | res.setHeader('Access-Control-Allow-Origin', '*'); |
| 36 | res.setHeader('Access-Control-Allow-Headers', '*'); |
| 37 | res.setHeader('Access-Control-Allow-Methods', 'GET,POST,OPTIONS'); |
| 38 | }; |
| 39 | |
| 40 | // The provider's own tokeniser, standing in: four bytes to a token, applied to the |
| 41 | // whole request body exactly as a provider would count what it was sent. |
| 42 | const countTokens = (payload) => Math.ceil(JSON.stringify(payload).length / 4); |
| 43 | |
| 44 | const lastUser = (messages) => { |
| 45 | for (let i = messages.length - 1; i >= 0; i--) { |
| 46 | if (messages[i].role === 'user') return String(messages[i].content || ''); |
| 47 | } |
| 48 | return ''; |
| 49 | }; |
| 50 | |
| 51 | const toolRounds = (messages) => { |
| 52 | let at = -1; |
| 53 | for (let i = messages.length - 1; i >= 0; i--) { |
| 54 | if (messages[i].role === 'user') { at = i; break; } |
| 55 | } |
| 56 | return messages.slice(at + 1).filter(m => m.role === 'tool').length; |
| 57 | }; |
| 58 | |
| 59 | const toolCall = (id, name, args) => ({ |
| 60 | id, type: 'function', |
| 61 | function: { name, arguments: typeof args === 'string' ? args : JSON.stringify(args) }, |
| 62 | }); |
| 63 | |
| 64 | const splitCall = (s) => { |
| 65 | const i = s.indexOf(' '); |
| 66 | if (i === -1) return { name: s.trim(), args: {} }; |
| 67 | try { return { name: s.slice(0, i).trim(), args: JSON.parse(s.slice(i + 1).trim()) }; } |
| 68 | catch { return { name: s.slice(0, i).trim(), args: {} }; } |
| 69 | }; |
| 70 | |
| 71 | const plan = (payload) => { |
| 72 | const messages = payload.messages || []; |
| 73 | const system = (messages[0] && messages[0].role === 'system') ? String(messages[0].content) : ''; |
| 74 | |
| 75 | // The compaction call, answered as a summariser would. Recognised by its own |
| 76 | // prompt, so a test can prove the summary reached the conversation. |
| 77 | if (system.includes('folding the earlier part')) { |
| 78 | return { text: 'SUMMARY-FROM-MODEL: the user asked for two files; both were written.' }; |
| 79 | } |
| 80 | |
| 81 | const t = lastUser(messages).trim(); |
| 82 | const rounds = toolRounds(messages); |
| 83 | if (!t.startsWith('@')) return { text: rounds > 0 ? 'Done.' : `Mock reply to: ${t.slice(0, 60)}` }; |
| 84 | const sp = t.indexOf(' '); |
| 85 | const verb = (sp === -1 ? t : t.slice(0, sp)).slice(1); |
| 86 | const rest = sp === -1 ? '' : t.slice(sp + 1).trim(); |
| 87 | |
| 88 | switch (verb) { |
| 89 | case 'text': return { text: rest || 'Right.' }; |
| 90 | case 'big': { |
| 91 | const kb = Math.max(1, Number(rest) || 8); |
| 92 | const chunk = 'lorem ipsum dolor sit amet consectetur adipiscing elit sed do ' |
| 93 | + 'eiusmod tempor incididunt ut labore et dolore magna aliqua. '; |
| 94 | return { text: chunk.repeat(Math.ceil(kb * 1024 / chunk.length)).slice(0, kb * 1024) }; |
| 95 | } |
| 96 | case 'tool': { |
| 97 | if (rounds > 0) return { text: 'Tool done.' }; |
| 98 | const { name, args } = splitCall(rest); |
| 99 | return { calls: [toolCall('call_' + Date.now(), name, args)] }; |
| 100 | } |
| 101 | default: return { text: rounds > 0 ? 'Done.' : `Mock reply to: ${t.slice(0, 60)}` }; |
| 102 | } |
| 103 | }; |
| 104 | |
| 105 | const sendJson = (res, obj, code = 200) => { |
| 106 | const body = JSON.stringify(obj); |
| 107 | cors(res); |
| 108 | res.writeHead(code, { 'content-type': 'application/json', 'content-length': Buffer.byteLength(body) }); |
| 109 | res.end(body); |
| 110 | }; |
| 111 | |
| 112 | const stream = (res, model, p, used) => { |
| 113 | cors(res); |
| 114 | res.writeHead(200, { 'content-type': 'text/event-stream', 'cache-control': 'no-cache' }); |
| 115 | const send = (o) => res.write(`data: ${JSON.stringify(o)}\n\n`); |
| 116 | const frame = (delta, finish = null) => ({ |
| 117 | id: 'chatcmpl-ctxmock', object: 'chat.completion.chunk', model, |
| 118 | choices: [{ index: 0, delta, finish_reason: finish }], |
| 119 | }); |
| 120 | send(frame({ role: 'assistant', content: '' })); |
| 121 | if (p.calls) { |
| 122 | p.calls.forEach((c, i) => send(frame({ tool_calls: [{ index: i, id: c.id, type: 'function', |
| 123 | function: { name: c.function.name, arguments: c.function.arguments } }] }))); |
| 124 | send(frame({}, 'tool_calls')); |
| 125 | } else { |
| 126 | // Big replies in a handful of frames rather than word by word: this mock exists to |
| 127 | // make conversations long, not to make tests slow. |
| 128 | const text = p.text || ''; |
| 129 | for (let i = 0; i < text.length; i += 2048) send(frame({ content: text.slice(i, i + 2048) })); |
| 130 | send(frame({}, 'stop')); |
| 131 | } |
| 132 | send({ id: 'chatcmpl-ctxmock', object: 'chat.completion.chunk', model, choices: [], |
| 133 | usage: { prompt_tokens: used, completion_tokens: 20, total_tokens: used + 20 } }); |
| 134 | res.write('data: [DONE]\n\n'); |
| 135 | res.end(); |
| 136 | }; |
| 137 | |
| 138 | const server = http.createServer((req, res) => { |
| 139 | if (req.method === 'OPTIONS') { cors(res); res.writeHead(204); return res.end(); } |
| 140 | if (req.method === 'GET' && req.url.startsWith('/v1/models')) { |
| 141 | return sendJson(res, { object: 'list', data: MODELS.map(id => ({ id, object: 'model' })) }); |
| 142 | } |
| 143 | if (req.method !== 'POST') { cors(res); res.writeHead(404); return res.end(); } |
| 144 | |
| 145 | let body = ''; |
| 146 | req.on('data', c => { body += c; }); |
| 147 | req.on('end', () => { |
| 148 | let payload; |
| 149 | try { payload = JSON.parse(body); } |
| 150 | catch { return sendJson(res, { error: { message: 'not JSON' } }, 400); } |
| 151 | |
| 152 | const used = countTokens(payload); |
| 153 | const over = used > LIMIT; |
| 154 | log({ at: new Date().toISOString(), used, limit: LIMIT, refused: over, |
| 155 | messages: payload.messages || [], tools: (payload.tools || []).length }); |
| 156 | if (over) { |
| 157 | return sendJson(res, { error: { |
| 158 | message: `This model's maximum context length is ${LIMIT} tokens, however you ` |
| 159 | + `requested ${used} tokens. Please reduce the length of the messages.`, |
| 160 | type: 'invalid_request_error', code: 'context_length_exceeded', |
| 161 | } }, 400); |
| 162 | } |
| 163 | const p = plan(payload); |
| 164 | if (payload.stream) return stream(res, payload.model || 'mock/fast', p, used); |
| 165 | return sendJson(res, { |
| 166 | id: 'chatcmpl-ctxmock', object: 'chat.completion', model: payload.model || 'mock/fast', |
| 167 | choices: [{ index: 0, message: p.calls |
| 168 | ? { role: 'assistant', content: null, tool_calls: p.calls } |
| 169 | : { role: 'assistant', content: p.text }, |
| 170 | finish_reason: p.calls ? 'tool_calls' : 'stop' }], |
| 171 | usage: { prompt_tokens: used, completion_tokens: 20, total_tokens: used + 20 }, |
| 172 | }); |
| 173 | }); |
| 174 | }); |
| 175 | |
| 176 | server.listen(PORT, '127.0.0.1', () => { |
| 177 | console.log(`ctxmock: http://127.0.0.1:${PORT}/v1/chat/completions limit=${LIMIT} tokens log=${LOG}`); |
| 178 | }); |