Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/ctxmock.mjs

7.2 KiB, 1 run

created by r2519314175:15, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1// A mock provider that behaves like a real one about its context window.
2//
3// dev/mockllm.mjs accepts a request of any size, so nothing in the suite has ever
4// exercised the failure this compactor exists to fix. This one refuses, with the
5// status and the wording an OpenAI-compatible provider actually uses:
6//
7// 400 {"error":{"message":"This model's maximum context length is N tokens,
8// however you requested M tokens", "code":"context_length_exceeded"}}
9//
10// node ctxmock.mjs [port] [limit-tokens]
11//
12// Directives, as dev/mockllm.mjs spells them, plus:
13// @big <kb> a reply of roughly <kb> kilobytes, streamed fast
14//
15// Every request is logged as a JSON line to ctxmock.log, whole, so a test can check
16// what the model was really shown -- including whether the tool calls in it were paired.
17
18import http from 'node:http';
19import fs from 'node:fs';
20import path from 'node:path';
21import { fileURLToPath } from 'node:url';
22
23const HERE = path.dirname(fileURLToPath(import.meta.url));
24// The port and the log are settable so this can belong to a numbered world (see
25// dev/world.sh); a shared log makes every assertion read another agent's traffic.
26const LOG = process.env.DAIMOND_CTX_LOG || path.join(HERE, 'ctxmock.log');
27const PORT = Number(process.argv[2] || process.env.DAIMOND_CTX_PORT || 9188);
28const LIMIT = Number(process.argv[3] || 12000); // tokens
29
30const MODELS = ['mock/fast', 'mock/thinker'];
31
32const log = (e) => { try { fs.appendFileSync(LOG, JSON.stringify(e) + '\n'); } catch {} };
33
34const cors = (res) => {
35 res.setHeader('Access-Control-Allow-Origin', '*');
36 res.setHeader('Access-Control-Allow-Headers', '*');
37 res.setHeader('Access-Control-Allow-Methods', 'GET,POST,OPTIONS');
38};
39
40// The provider's own tokeniser, standing in: four bytes to a token, applied to the
41// whole request body exactly as a provider would count what it was sent.
42const countTokens = (payload) => Math.ceil(JSON.stringify(payload).length / 4);
43
44const lastUser = (messages) => {
45 for (let i = messages.length - 1; i >= 0; i--) {
46 if (messages[i].role === 'user') return String(messages[i].content || '');
47 }
48 return '';
49};
50
51const toolRounds = (messages) => {
52 let at = -1;
53 for (let i = messages.length - 1; i >= 0; i--) {
54 if (messages[i].role === 'user') { at = i; break; }
55 }
56 return messages.slice(at + 1).filter(m => m.role === 'tool').length;
57};
58
59const toolCall = (id, name, args) => ({
60 id, type: 'function',
61 function: { name, arguments: typeof args === 'string' ? args : JSON.stringify(args) },
62});
63
64const splitCall = (s) => {
65 const i = s.indexOf(' ');
66 if (i === -1) return { name: s.trim(), args: {} };
67 try { return { name: s.slice(0, i).trim(), args: JSON.parse(s.slice(i + 1).trim()) }; }
68 catch { return { name: s.slice(0, i).trim(), args: {} }; }
69};
70
71const plan = (payload) => {
72 const messages = payload.messages || [];
73 const system = (messages[0] && messages[0].role === 'system') ? String(messages[0].content) : '';
74
75 // The compaction call, answered as a summariser would. Recognised by its own
76 // prompt, so a test can prove the summary reached the conversation.
77 if (system.includes('folding the earlier part')) {
78 return { text: 'SUMMARY-FROM-MODEL: the user asked for two files; both were written.' };
79 }
80
81 const t = lastUser(messages).trim();
82 const rounds = toolRounds(messages);
83 if (!t.startsWith('@')) return { text: rounds > 0 ? 'Done.' : `Mock reply to: ${t.slice(0, 60)}` };
84 const sp = t.indexOf(' ');
85 const verb = (sp === -1 ? t : t.slice(0, sp)).slice(1);
86 const rest = sp === -1 ? '' : t.slice(sp + 1).trim();
87
88 switch (verb) {
89 case 'text': return { text: rest || 'Right.' };
90 case 'big': {
91 const kb = Math.max(1, Number(rest) || 8);
92 const chunk = 'lorem ipsum dolor sit amet consectetur adipiscing elit sed do '
93 + 'eiusmod tempor incididunt ut labore et dolore magna aliqua. ';
94 return { text: chunk.repeat(Math.ceil(kb * 1024 / chunk.length)).slice(0, kb * 1024) };
95 }
96 case 'tool': {
97 if (rounds > 0) return { text: 'Tool done.' };
98 const { name, args } = splitCall(rest);
99 return { calls: [toolCall('call_' + Date.now(), name, args)] };
100 }
101 default: return { text: rounds > 0 ? 'Done.' : `Mock reply to: ${t.slice(0, 60)}` };
102 }
103};
104
105const sendJson = (res, obj, code = 200) => {
106 const body = JSON.stringify(obj);
107 cors(res);
108 res.writeHead(code, { 'content-type': 'application/json', 'content-length': Buffer.byteLength(body) });
109 res.end(body);
110};
111
112const stream = (res, model, p, used) => {
113 cors(res);
114 res.writeHead(200, { 'content-type': 'text/event-stream', 'cache-control': 'no-cache' });
115 const send = (o) => res.write(`data: ${JSON.stringify(o)}\n\n`);
116 const frame = (delta, finish = null) => ({
117 id: 'chatcmpl-ctxmock', object: 'chat.completion.chunk', model,
118 choices: [{ index: 0, delta, finish_reason: finish }],
119 });
120 send(frame({ role: 'assistant', content: '' }));
121 if (p.calls) {
122 p.calls.forEach((c, i) => send(frame({ tool_calls: [{ index: i, id: c.id, type: 'function',
123 function: { name: c.function.name, arguments: c.function.arguments } }] })));
124 send(frame({}, 'tool_calls'));
125 } else {
126 // Big replies in a handful of frames rather than word by word: this mock exists to
127 // make conversations long, not to make tests slow.
128 const text = p.text || '';
129 for (let i = 0; i < text.length; i += 2048) send(frame({ content: text.slice(i, i + 2048) }));
130 send(frame({}, 'stop'));
131 }
132 send({ id: 'chatcmpl-ctxmock', object: 'chat.completion.chunk', model, choices: [],
133 usage: { prompt_tokens: used, completion_tokens: 20, total_tokens: used + 20 } });
134 res.write('data: [DONE]\n\n');
135 res.end();
136};
137
138const server = http.createServer((req, res) => {
139 if (req.method === 'OPTIONS') { cors(res); res.writeHead(204); return res.end(); }
140 if (req.method === 'GET' && req.url.startsWith('/v1/models')) {
141 return sendJson(res, { object: 'list', data: MODELS.map(id => ({ id, object: 'model' })) });
142 }
143 if (req.method !== 'POST') { cors(res); res.writeHead(404); return res.end(); }
144
145 let body = '';
146 req.on('data', c => { body += c; });
147 req.on('end', () => {
148 let payload;
149 try { payload = JSON.parse(body); }
150 catch { return sendJson(res, { error: { message: 'not JSON' } }, 400); }
151
152 const used = countTokens(payload);
153 const over = used > LIMIT;
154 log({ at: new Date().toISOString(), used, limit: LIMIT, refused: over,
155 messages: payload.messages || [], tools: (payload.tools || []).length });
156 if (over) {
157 return sendJson(res, { error: {
158 message: `This model's maximum context length is ${LIMIT} tokens, however you `
159 + `requested ${used} tokens. Please reduce the length of the messages.`,
160 type: 'invalid_request_error', code: 'context_length_exceeded',
161 } }, 400);
162 }
163 const p = plan(payload);
164 if (payload.stream) return stream(res, payload.model || 'mock/fast', p, used);
165 return sendJson(res, {
166 id: 'chatcmpl-ctxmock', object: 'chat.completion', model: payload.model || 'mock/fast',
167 choices: [{ index: 0, message: p.calls
168 ? { role: 'assistant', content: null, tool_calls: p.calls }
169 : { role: 'assistant', content: p.text },
170 finish_reason: p.calls ? 'tool_calls' : 'stop' }],
171 usage: { prompt_tokens: used, completion_tokens: 20, total_tokens: used + 20 },
172 });
173 });
174});
175
176server.listen(PORT, '127.0.0.1', () => {
177 console.log(`ctxmock: http://127.0.0.1:${PORT}/v1/chat/completions limit=${LIMIT} tokens log=${LOG}`);
178});