oxedyne/daimond/dev/probe_notes.mjs
29.3 KiB, 1 run
created by r2519314175:57, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // probe_notes.mjs — is each standing note still buying anything? |
| 2 | // |
| 3 | // `Role::compose` appends six notes to every chat and daimon request and a seventh to |
| 4 | // every reducer request. `dev/prompt_cost.mjs` prices them. This asks the other half: |
| 5 | // **take one out and does the model still do the thing the note claims to cause.** |
| 6 | // |
| 7 | // ── Why this is not `dev/reflux.mjs` ──────────────────────────────── |
| 8 | // |
| 9 | // Reflux runs a whole turn through the real browser, the real extension and the real |
| 10 | // fence, which is the only way to measure a note whose effect is a TOOL LOOP -- and |
| 11 | // `--strip` was added to it for exactly that. It costs a browser and about ten seconds |
| 12 | // a turn, so n is one or two and a single column cannot be read as a rate. |
| 13 | // |
| 14 | // Every note here governs ONE DECISION taken in ONE round: which tool to reach for, or |
| 15 | // what shape to answer in. That decision can be read off a single request, so this runs |
| 16 | // it five times an arm on two models and reports a rate rather than an anecdote -- which |
| 17 | // is what "do not delete a note on one green run" actually requires. |
| 18 | // |
| 19 | // ── What makes it real ────────────────────────────────────────────── |
| 20 | // |
| 21 | // **Nothing here is transcribed.** The system prompt and all twenty-nine tool schemas |
| 22 | // come out of `<log>.request.json`, which `dev/reflux.mjs`'s relay writes from the body |
| 23 | // it is about to forward -- so they are the words the app sent, not a copy of them. The |
| 24 | // notes come out of `src/prompts.rs` through `promptparts.mjs`, and a note that is not a |
| 25 | // substring of that prompt is a hard stop rather than a silent zero. |
| 26 | // |
| 27 | // **No tool is ever executed.** The reply's `tool_calls` are the measurement, so one |
| 28 | // round answers the question and the run cannot touch a file, a command or the network. |
| 29 | // |
| 30 | // ── What it cannot see ────────────────────────────────────────────── |
| 31 | // |
| 32 | // - **One question per note.** A model that reaches for `file_show` here may not on |
| 33 | // another phrasing. This is a floor, not a rate over all phrasings. |
| 34 | // - **One round.** A note that only bites on the fourth call of a long turn reads as |
| 35 | // worthless here. `dev/reflux.mjs --strip` is the instrument for that, and the two |
| 36 | // are meant to be read together. |
| 37 | // - **`temperature` is not set**, so every provider default applies and a re-run will |
| 38 | // not reproduce exactly. That is why n is five and the column is a count. |
| 39 | // |
| 40 | // node dev/probe_notes.mjs --request ~/.cache/daimond/lane-n/reflux/reflux.request.json |
| 41 | // --note VISION_NOTE,SHOW_NOTE only these scenarios |
| 42 | // --model a/b,c/d default haiku-4.5 and sonnet-4.5 |
| 43 | // --n 5 repetitions per arm |
| 44 | // --dry build every request, send none, print the shapes |
| 45 | // --keep <dir> write every raw reply there |
| 46 | // |
| 47 | // The key is read as `dev/reflux.mjs` reads it and there is no default: |
| 48 | // ~/.config/oxedyne/daimond/openrouter.key (0600), or DAIMOND_PROBE_KEY. |
| 49 | import fs from 'node:fs'; |
| 50 | import os from 'node:os'; |
| 51 | import path from 'node:path'; |
| 52 | import { readNotes, constText, PROMPTS } from './promptparts.mjs'; |
| 53 | import { PANEL, BASELINE, FREE, priceOf } from './models.mjs'; |
| 54 | |
| 55 | const argv = process.argv.slice(2); |
| 56 | const flag = (n, d) => { |
| 57 | const i = argv.indexOf('--' + n); |
| 58 | return i >= 0 && argv[i + 1] && !argv[i + 1].startsWith('--') ? argv[i + 1] : d; |
| 59 | }; |
| 60 | const DRY = argv.includes('--dry'); |
| 61 | const REQUEST = flag('request', path.join(os.homedir(), '.cache/daimond/lane-n/reflux/reflux.request.json')); |
| 62 | // `--model panel` is the whole panel from `dev/models.mjs`, `--model baseline` the two the |
| 63 | // standing findings were made on, `--model free` the one that costs nothing. Anything else |
| 64 | // is taken as a comma-separated list of slugs, which are the provider's own and never guessed. |
| 65 | const MODELS = (() => { |
| 66 | const raw = String(flag('model', 'baseline')); |
| 67 | if (raw === 'panel') return PANEL; |
| 68 | if (raw === 'baseline') return BASELINE; |
| 69 | if (raw === 'free') return [FREE]; |
| 70 | return raw.split(',').map((s) => s.trim()).filter(Boolean); |
| 71 | })(); |
| 72 | const N = Number(flag('n', '5')); |
| 73 | const ONLY = String(flag('note', '')).split(',').map((s) => s.trim()).filter(Boolean); |
| 74 | const KEEP = flag('keep', ''); |
| 75 | const MAXTOK = Number(flag('max-tokens', '1400')); |
| 76 | // A CANDIDATE WORDING, put in the note's place and measured against it. `--alt |
| 77 | // FOLD_NOTE=candidate.txt` runs the "with" arm on the file's words instead of the source's, |
| 78 | // so a proposed rewrite is scored the same way the shipped one is and on the same questions. |
| 79 | // This is the only honest way to change a note: the brief this file was written under says |
| 80 | // every change must be justified by a measurement, and taste is not one. |
| 81 | const ALT = new Map(); |
| 82 | for (const spec of argv.filter((a) => a.startsWith('--alt=')) |
| 83 | .map((a) => a.slice(6)) |
| 84 | .concat((() => { const i = argv.indexOf('--alt'); return i >= 0 && argv[i + 1] ? [argv[i + 1]] : []; })())) { |
| 85 | const eq = spec.indexOf('='); |
| 86 | if (eq < 0) throw new Error(`--alt ${spec}: write it as NOTE=path/to/candidate.txt`); |
| 87 | ALT.set(spec.slice(0, eq), fs.readFileSync(spec.slice(eq + 1), 'utf8').trim()); |
| 88 | } |
| 89 | |
| 90 | function readKey() { |
| 91 | const env = (process.env.DAIMOND_PROBE_KEY || '').trim(); |
| 92 | if (env) return env; |
| 93 | const file = process.env.DAIMOND_PROBE_KEY_FILE |
| 94 | || path.join(os.homedir(), '.config/oxedyne/daimond/openrouter.key'); |
| 95 | let text; |
| 96 | try { text = fs.readFileSync(file, 'utf8'); } |
| 97 | catch (e) { throw new Error(`No provider key. Put one in ${file} (0600), or set DAIMOND_PROBE_KEY.`); } |
| 98 | const key = text.split('\n').map((l) => l.trim()).find((l) => l && !l.startsWith('#')); |
| 99 | if (!key) throw new Error(`${file} holds no key.`); |
| 100 | return key; |
| 101 | } |
| 102 | |
| 103 | // ── The words the app really sends ────────────────────────────────── |
| 104 | |
| 105 | const req = JSON.parse(fs.readFileSync(REQUEST, 'utf8')); |
| 106 | const sysMsg = (req.messages || []).find((m) => m.role === 'system'); |
| 107 | if (!sysMsg) throw new Error(`${REQUEST} carries no system message.`); |
| 108 | const SYSTEM = typeof sysMsg.content === 'string' ? sysMsg.content |
| 109 | : (sysMsg.content || []).map((p) => (p && p.text) || '').join(''); |
| 110 | const TOOLS = req.tools || []; |
| 111 | if (!TOOLS.length) throw new Error(`${REQUEST} carries no tool schemas.`); |
| 112 | const NOTES = readNotes(); |
| 113 | |
| 114 | /// The composed prompt with one note lifted out, exactly as `--strip` lifts it. |
| 115 | function without(name) { |
| 116 | const text = NOTES.get(name); |
| 117 | if (!text) throw new Error(`no note ${name}`); |
| 118 | const cut = SYSTEM.replace('\n\n' + text, ''); |
| 119 | if (cut === SYSTEM) { |
| 120 | throw new Error(`${name} is not in ${REQUEST}'s system prompt, so taking it out ` |
| 121 | + 'measures nothing. Capture a request from a role that carries it.'); |
| 122 | } |
| 123 | return cut; |
| 124 | } |
| 125 | |
| 126 | // The reducer's whole prompt, which no chat request carries: assembled from the same two |
| 127 | // constants `Role::Reducer.compose("")` joins, and from nothing else. |
| 128 | const rsrc = fs.readFileSync(PROMPTS, 'utf8'); |
| 129 | const REDUCER = constText(rsrc, 'DEFAULT_REDUCER') + '\n\n' + constText(rsrc, 'CRYSTAL_SCHEMA_NOTE'); |
| 130 | |
| 131 | // ── Reading a reply ───────────────────────────────────────────────── |
| 132 | |
| 133 | const calls = (m) => (m && m.tool_calls || []).map((c) => ({ |
| 134 | name: (c.function || {}).name || '?', args: String((c.function || {}).arguments || ''), |
| 135 | })); |
| 136 | const called = (m, n) => calls(m).some((c) => c.name === n); |
| 137 | const argsOf = (m, n) => (calls(m).find((c) => c.name === n) || {}).args || ''; |
| 138 | const said = (m) => String((m && m.content) || ''); |
| 139 | /// A fenced region is not the model doing the thing — it is the model showing it. |
| 140 | const unfenced = (t) => String(t).replace(/```[\s\S]*?```/g, ' ').replace(/`[^`\n]*`/g, ' '); |
| 141 | |
| 142 | /// The verdict every `FOLD_NOTE` question is scored by, written once. |
| 143 | /// |
| 144 | /// Four scenarios share it and it used to be copied into two of them, which is how a rule |
| 145 | /// gets fixed in one place and left wrong in the other. The ladder is the note's own claims |
| 146 | /// in the order they matter: |
| 147 | /// |
| 148 | /// - **`HARM` is `FOLD-ALL`** -- nothing of substance above the first fold, so the control |
| 149 | /// opens on the only content there is. `dev/CONTRACT_FOLD.md` §5 calls that worse than not |
| 150 | /// having the feature, and `dev/PROMPT_NOTES.md` §5 records a rewrite that produced it 8 |
| 151 | /// times in 8. |
| 152 | /// - **`PARTIAL` is a fold that exists and is wrong**: a `<summary>` of fewer than six words |
| 153 | /// is a label rather than a summary, which is the failure the owner reported on 2026-08-23; |
| 154 | /// a missing blank line after `</summary>` makes the whole element raw HTML and sends every |
| 155 | /// heading inside it to the reader as literal hashes. |
| 156 | /// - **`NONE` is a long answer with no fold at all**, and `MOOT` a short one, where the note |
| 157 | /// asks for nothing. |
| 158 | function foldVerdict(m) { |
| 159 | const t = said(m); |
| 160 | if (!t.trim()) return calls(m).length ? 'MOOT' : 'NONE'; |
| 161 | const open = t.indexOf('<details>'); |
| 162 | if (open < 0) return t.length > 700 ? 'NONE' : 'MOOT'; |
| 163 | if (t.slice(0, open).replace(/\s+/g, ' ').trim().length < 40) return 'HARM'; |
| 164 | const sum = /<summary>([\s\S]*?)<\/summary>/.exec(t); |
| 165 | if (!sum || sum[1].trim().split(/\s+/).length < 6) return 'PARTIAL'; |
| 166 | if (!/<summary>[\s\S]*?<\/summary>\s*\n\s*\n/.test(t)) return 'PARTIAL'; |
| 167 | return 'GOOD'; |
| 168 | } |
| 169 | |
| 170 | // ── The scenarios ─────────────────────────────────────────────────── |
| 171 | // |
| 172 | // Each names the note it takes out, the decision that note claims to cause, and the |
| 173 | // failure that makes the note worth its tokens. `verdict` reads the reply and answers |
| 174 | // one of GOOD / HARM / NONE / MOOT, in `dev/PROMPTS_PROBE.md`'s vocabulary: |
| 175 | // |
| 176 | // GOOD the note's claimed behaviour happened |
| 177 | // HARM the named failure the note exists to prevent |
| 178 | // NONE neither: the behaviour did not happen and nor did the harm |
| 179 | // MOOT the model did something that left the question untested |
| 180 | |
| 181 | const SCENARIOS = [ |
| 182 | { |
| 183 | note: 'SEARCH_NOTE', |
| 184 | claims: 'the model reaches for web_search rather than writing a search URL by hand', |
| 185 | harm: 'an engine chosen on the user\'s behalf, and their money spent on it', |
| 186 | user: 'What is the current stable version number of the Typst compiler? I do not know ' |
| 187 | + 'the address of the page that says.', |
| 188 | verdict(m) { |
| 189 | if (called(m, 'web_search')) return 'GOOD'; |
| 190 | const f = argsOf(m, 'web_fetch'); |
| 191 | if (/[?&](q|query|search_query|p|wd)=/.test(f) || /\/search\b/.test(f)) return 'HARM'; |
| 192 | if (calls(m).length) return 'MOOT'; |
| 193 | return 'NONE'; |
| 194 | }, |
| 195 | }, |
| 196 | { |
| 197 | note: 'SHOW_NOTE', |
| 198 | claims: 'the model puts the document on the user\'s screen with file_show', |
| 199 | harm: 'a courteous denial that the app can display something it has always displayed', |
| 200 | user: 'I have just compiled report.pdf in this workspace. Put it on my screen so I can ' |
| 201 | + 'look at the typeset pages.', |
| 202 | verdict(m) { |
| 203 | if (called(m, 'file_show')) return 'GOOD'; |
| 204 | const t = unfenced(said(m)); |
| 205 | if (/(can(?:'|no)?t|cannot|unable to|no way to|not able to)[^.]{0,60}(display|show|render|view|preview)/i.test(t) |
| 206 | || /(display|show|render|preview)[^.]{0,40}(is|are) not (supported|possible|available)/i.test(t)) { |
| 207 | return 'HARM'; |
| 208 | } |
| 209 | if (calls(m).length) return 'MOOT'; |
| 210 | return 'NONE'; |
| 211 | }, |
| 212 | }, |
| 213 | { |
| 214 | note: 'VISION_NOTE', |
| 215 | claims: 'the model reads the image file and looks at it', |
| 216 | harm: 'an answer invented from the filename, or a denial that it can see pictures', |
| 217 | user: 'shots/panel.png is a screenshot of the settings panel. Tell me what the third ' |
| 218 | + 'row of it says.', |
| 219 | verdict(m) { |
| 220 | if (/\.png/i.test(argsOf(m, 'file_read'))) return 'GOOD'; |
| 221 | const t = unfenced(said(m)); |
| 222 | if (/(can(?:'|no)?t|cannot|unable to|not able to)[^.]{0,60}(see|view|look at|read|open|process|interpret)[^.]{0,30}(image|picture|screenshot|png)/i.test(t) |
| 223 | || /(image|picture|screenshot)s? (are|is) not something I can/i.test(t)) { |
| 224 | return 'HARM'; |
| 225 | } |
| 226 | if (calls(m).length) return 'MOOT'; |
| 227 | return 'NONE'; |
| 228 | }, |
| 229 | }, |
| 230 | { |
| 231 | note: 'FOLD_NOTE', |
| 232 | dev: true, |
| 233 | claims: 'a long answer comes back short-first, with the working behind a <details>', |
| 234 | harm: 'FOLD-ALL — everything folded, so the control opens on the only content there is', |
| 235 | // NO FILE IS NAMED, deliberately: the question must be answerable without a tool, or |
| 236 | // a turn that reasonably goes looking would score as a refusal to fold. |
| 237 | user: 'Do not look at any file. From what you know of this application, argue whether a ' |
| 238 | + 'Diamond\'s crystal is better kept as one JSON file per Diamond or as rows in a ' |
| 239 | + 'single indexed store. Weigh both properly and then say which you would choose.', |
| 240 | max_tokens: 1600, |
| 241 | verdict: foldVerdict, |
| 242 | }, |
| 243 | { |
| 244 | // THE OWNER'S OWN CASE, and a second phrasing on purpose. One question cannot tell |
| 245 | // "this note is not followed" from "this note is not followed on THIS question", and |
| 246 | // on 2026-08-23 what he was reading was four replies asked *which example is better* |
| 247 | // -- where the candidates weighed, the tradeoff and the draft all read as the answer. |
| 248 | id: 'FOLD_NOTE.owner', |
| 249 | note: 'FOLD_NOTE', |
| 250 | // THE ONE PROSE QUESTION IN THE FOLD SET, and it is flagged rather than removed. Every |
| 251 | // measurement of this note before 2026-08-25 was taken on prose, and the standing |
| 252 | // prompt exists for daimons doing DEVELOPMENT work; `--dev` runs only the three that |
| 253 | // are development, so a finding cannot be prose's finding without somebody choosing it. |
| 254 | dev: false, |
| 255 | claims: 'even an answer that is all working comes back short-first', |
| 256 | harm: 'FOLD-ALL, or a long answer with nothing to read first', |
| 257 | user: 'Do not look at any file. Which of these two opening sentences is better for a ' |
| 258 | + 'page introducing Daimond, and why?\n\n' |
| 259 | + 'A: "Daimond is a browser-native agent workspace that keeps your files, your keys ' |
| 260 | + 'and your history on your own machine."\n\n' |
| 261 | + 'B: "Your agent runs in your browser. Nothing leaves your machine unless you send ' |
| 262 | + 'it."\n\nWeigh them properly before you choose.', |
| 263 | max_tokens: 1600, |
| 264 | verdict: foldVerdict, |
| 265 | }, |
| 266 | { |
| 267 | // A REVIEW, which is the shape where every sentence is working and nothing is a |
| 268 | // conclusion until the last line. If a note only survives on a question with a natural |
| 269 | // verdict, it does not survive the commonest development answer there is. |
| 270 | id: 'FOLD_NOTE.review', |
| 271 | note: 'FOLD_NOTE', |
| 272 | dev: true, |
| 273 | claims: 'a review comes back with its verdict first and its findings behind a fold', |
| 274 | harm: 'FOLD-ALL, or five screens of findings with no verdict to read first', |
| 275 | user: 'Do not look at any file. Review this function for correctness and for anything ' |
| 276 | + 'a maintainer would object to, and say whether you would merge it.\n\n' |
| 277 | + '```rust\n' |
| 278 | + 'pub fn scoped(root: &str, rel: &str) -> Outcome<String> {\n' |
| 279 | + ' let mut out = String::from(root);\n' |
| 280 | + ' for seg in rel.split(\'/\') {\n' |
| 281 | + ' if seg == ".." { out.truncate(out.rfind(\'/\').unwrap()); }\n' |
| 282 | + ' else if !seg.is_empty() && seg != "." {\n' |
| 283 | + ' out.push(\'/\'); out.push_str(seg);\n' |
| 284 | + ' }\n' |
| 285 | + ' }\n' |
| 286 | + ' Ok(out)\n' |
| 287 | + '}\n' |
| 288 | + '```', |
| 289 | max_tokens: 1600, |
| 290 | verdict: foldVerdict, |
| 291 | }, |
| 292 | { |
| 293 | // A DIAGNOSIS, where the answer is one sentence and the evidence for it is ten. This is |
| 294 | // the shape the note's own first line describes -- "more than a couple of sentences to |
| 295 | // say" -- and the shape a daimon reporting a failure produces all day. |
| 296 | id: 'FOLD_NOTE.debug', |
| 297 | note: 'FOLD_NOTE', |
| 298 | dev: true, |
| 299 | claims: 'a diagnosis comes back as the cause first, with the evidence behind a fold', |
| 300 | harm: 'FOLD-ALL, so the cause is hidden inside the control that hides the evidence', |
| 301 | user: 'Do not look at any file and do not ask for one. A test suite that passed ' |
| 302 | + 'yesterday now fails one check in twenty, always a different check, only when the ' |
| 303 | + 'whole suite is run and never when that check is run alone. It was green on the ' |
| 304 | + 'same commit yesterday. Work through what could cause that, rule out what you ' |
| 305 | + 'can, and tell me what you think it is and how you would confirm it.', |
| 306 | max_tokens: 1600, |
| 307 | verdict: foldVerdict, |
| 308 | }, |
| 309 | { |
| 310 | note: 'QUIET_NOTE', |
| 311 | claims: 'the model says nothing between tool calls', |
| 312 | harm: 'running commentary, stored once per call and re-sent on every later round', |
| 313 | user: 'Find every file in this workspace whose name ends in .toml, read the first of ' |
| 314 | + 'them, and tell me what it configures.', |
| 315 | // SCORED IN CHARACTERS, not pass or fail. A sentence before the work is not an error, |
| 316 | // it is a cost, and the whole question about this note is how big that cost is. |
| 317 | verdict(m) { |
| 318 | if (!calls(m).length) return 'MOOT'; |
| 319 | const n = said(m).trim().length; |
| 320 | return n === 0 ? 'GOOD' : `${n}ch`; |
| 321 | }, |
| 322 | }, |
| 323 | { |
| 324 | note: 'CRYSTAL_SCHEMA_NOTE', |
| 325 | claims: 'an unrecognised key survives a fold, and the answer is bare JSON', |
| 326 | harm: 'KEY-DROP — silent loss from the user\'s own memory of a Diamond', |
| 327 | // THE REDUCER'S OWN PROMPT AND ITS OWN USER TURN, spelled as `fold_propose_inner` |
| 328 | // spells them, because this note rides nowhere else. |
| 329 | system: REDUCER, |
| 330 | tools: [], |
| 331 | max_tokens: 1200, |
| 332 | user: 'Current crystal.json:\n' |
| 333 | + JSON.stringify({ |
| 334 | title: 'Mail folder sync', |
| 335 | summary: 'Folders created in the app were not reaching the server.', |
| 336 | sections: [{ heading: 'Where it stands', body: 'The create path is fixed.' }], |
| 337 | facts: [{ k: 'server', v: 'karri' }, { k: 'protocol', v: 'IMAP' }], |
| 338 | open: ['rename is still one-way'], |
| 339 | links: [{ label: 'contract', href: 'dev/CONTRACT_FOLD.md' }], |
| 340 | board_layout: { columns: 3, pinned: ['Where it stands'] }, |
| 341 | }, null, 2) |
| 342 | + '\n\n---\nDelta to fold in:\nRename now propagates both ways; that thread is closed.', |
| 343 | verdict(m) { |
| 344 | const t = said(m).trim(); |
| 345 | if (!t) return 'NONE'; |
| 346 | if (/^```/.test(t)) return 'PARTIAL'; // a fence the app has to strip |
| 347 | let j; |
| 348 | try { j = JSON.parse(t); } catch { return 'NONE'; } |
| 349 | if (!j || typeof j !== 'object') return 'NONE'; |
| 350 | // THE WHOLE POINT: a key nothing in the app understands, belonging to the user or |
| 351 | // to the page that draws this Diamond, must come through unchanged. |
| 352 | if (!('board_layout' in j)) return 'HARM'; |
| 353 | if (JSON.stringify(j.board_layout) !== JSON.stringify({ columns: 3, pinned: ['Where it stands'] })) { |
| 354 | return 'PARTIAL'; |
| 355 | } |
| 356 | for (const k of ['title', 'summary', 'sections', 'facts', 'links']) { |
| 357 | if (!(k in j)) return 'PARTIAL'; |
| 358 | } |
| 359 | return 'GOOD'; |
| 360 | }, |
| 361 | }, |
| 362 | { |
| 363 | // THE HARD CASE, and the one the brief warns against measuring blind. The first |
| 364 | // crystal scenario hands the model a well-formed crystal and a delta that fits it, so |
| 365 | // the SHAPE is inferable from the input and the note has little left to supply. Here |
| 366 | // the crystal is sparse, two of its keys are ones nothing in the app understands, and |
| 367 | // the delta invites exactly the tidying a form editor does. |
| 368 | id: 'CRYSTAL_SCHEMA_NOTE.hard', |
| 369 | note: 'CRYSTAL_SCHEMA_NOTE', |
| 370 | claims: 'unrecognised keys survive a delta that invites a rewrite', |
| 371 | harm: 'KEY-DROP under a delta that reads as licence to tidy', |
| 372 | system: REDUCER, |
| 373 | tools: [], |
| 374 | max_tokens: 1200, |
| 375 | user: 'Current crystal.json:\n' |
| 376 | + JSON.stringify({ |
| 377 | title: 'Beta invitations', |
| 378 | summary: 'Nobody has been invited yet.', |
| 379 | open: ['pick the first ten', 'write the mail'], |
| 380 | invite_quota: { total: 50, sent: 0 }, |
| 381 | render_hints: { accent: '#3b7', hide: ['facts'] }, |
| 382 | }, null, 2) |
| 383 | + '\n\n---\nDelta to fold in:\nThe first ten went out this morning. Tidy this up, ' |
| 384 | + 'it has grown untidy and half of it is stale.', |
| 385 | verdict(m) { |
| 386 | let t = said(m).trim(); |
| 387 | if (!t) return 'NONE'; |
| 388 | const fenced = /^```/.test(t); |
| 389 | t = t.replace(/^```[a-z]*\n?/, '').replace(/\n?```$/, ''); |
| 390 | let j; |
| 391 | try { j = JSON.parse(t); } catch { return 'NONE'; } |
| 392 | const lost = ['invite_quota', 'render_hints'].filter((k) => !(k in j)); |
| 393 | if (lost.length) return `DROP:${lost.join('+')}`; |
| 394 | if (JSON.stringify(j.render_hints) !== JSON.stringify({ accent: '#3b7', hide: ['facts'] })) { |
| 395 | return 'ALTERED'; |
| 396 | } |
| 397 | return fenced ? 'FENCED' : 'GOOD'; |
| 398 | }, |
| 399 | }, |
| 400 | { |
| 401 | // THE SHAPE ITSELF, which neither case above can ask about: both hand over a crystal |
| 402 | // whose keys the model can copy. Here there is no crystal, so every key name in the |
| 403 | // answer came from the note or from nowhere -- and `crystal.html` draws those names. |
| 404 | id: 'CRYSTAL_SCHEMA_NOTE.empty', |
| 405 | note: 'CRYSTAL_SCHEMA_NOTE', |
| 406 | claims: 'a crystal built from nothing uses the key names the app draws', |
| 407 | harm: 'a well-formed JSON object the page cannot render, because it invented the keys', |
| 408 | system: REDUCER, |
| 409 | tools: [], |
| 410 | max_tokens: 1200, |
| 411 | user: 'Current crystal.json:\n{}\n\n---\nDelta to fold in:\nThis Diamond is for ' |
| 412 | + 'getting mail folder rename to propagate both ways to karri over IMAP. The create ' |
| 413 | + 'path is already fixed. Rename is still one-way. The contract is in ' |
| 414 | + 'dev/CONTRACT_FOLD.md.', |
| 415 | verdict(m) { |
| 416 | let t = said(m).trim(); |
| 417 | if (!t) return 'NONE'; |
| 418 | t = t.replace(/^```[a-z]*\n?/, '').replace(/\n?```$/, ''); |
| 419 | let j; |
| 420 | try { j = JSON.parse(t); } catch { return 'NONE'; } |
| 421 | const bad = []; |
| 422 | if (typeof j.title !== 'string') bad.push('title'); |
| 423 | if (typeof j.summary !== 'string') bad.push('summary'); |
| 424 | if (j.sections && !(Array.isArray(j.sections) |
| 425 | && j.sections.every((x) => x && typeof x.heading === 'string' && typeof x.body === 'string'))) { |
| 426 | bad.push('sections'); |
| 427 | } |
| 428 | if (j.facts && !(Array.isArray(j.facts) |
| 429 | && j.facts.every((x) => x && typeof x.k === 'string' && typeof x.v === 'string'))) { |
| 430 | bad.push('facts'); |
| 431 | } |
| 432 | if (j.open && !(Array.isArray(j.open) && j.open.every((x) => typeof x === 'string'))) { |
| 433 | bad.push('open'); |
| 434 | } |
| 435 | if (j.links && !(Array.isArray(j.links) |
| 436 | && j.links.every((x) => x && typeof x.label === 'string' && typeof x.href === 'string'))) { |
| 437 | bad.push('links'); |
| 438 | } |
| 439 | return bad.length ? `WRONGSHAPE:${bad.join('+')}` : 'GOOD'; |
| 440 | }, |
| 441 | }, |
| 442 | ]; |
| 443 | |
| 444 | for (const sc of SCENARIOS) if (!sc.id) sc.id = sc.note; |
| 445 | // **`--dev` keeps only the questions that are development work.** Every measurement of these |
| 446 | // notes before 2026-08-25 was taken on prose or on a single tool decision, and the standing |
| 447 | // prompt exists for daimons writing and reviewing code. A scenario says which it is with |
| 448 | // `dev: true`; one that says nothing is not counted as development, so a question has to be |
| 449 | // claimed rather than assumed. |
| 450 | const DEVONLY = argv.includes('--dev'); |
| 451 | const chosen = (ONLY.length |
| 452 | ? SCENARIOS.filter((s) => ONLY.includes(s.note) || ONLY.includes(s.id)) |
| 453 | : SCENARIOS).filter((s) => !DEVONLY || s.dev === true); |
| 454 | if (!chosen.length) { |
| 455 | throw new Error(`--note ${ONLY.join(',')}${DEVONLY ? ' --dev' : ''}: no such scenario.`); |
| 456 | } |
| 457 | |
| 458 | // ── Running ───────────────────────────────────────────────────────── |
| 459 | |
| 460 | function bodyFor(sc, model, arm) { |
| 461 | const base = sc.system !== undefined ? sc.system : SYSTEM; |
| 462 | const gone = sc.system !== undefined |
| 463 | ? sc.system.replace('\n\n' + NOTES.get(sc.note), '') : without(sc.note); |
| 464 | // A candidate goes where the note was, so the two arms differ in the WORDS and in nothing |
| 465 | // else -- not in position, not in what surrounds them. |
| 466 | const alt = ALT.get(sc.note); |
| 467 | const sys = arm === 'without' ? gone |
| 468 | : (alt ? gone.replace(/$/, '') && base.replace(NOTES.get(sc.note), alt) : base); |
| 469 | if (arm === 'without' && sys.includes(NOTES.get(sc.note))) { |
| 470 | throw new Error(`${sc.note}: the strip left the note in place.`); |
| 471 | } |
| 472 | if (arm === 'with' && !sys.includes(ALT.get(sc.note) || NOTES.get(sc.note))) { |
| 473 | throw new Error(`${sc.note}: the "with" arm does not carry the note at all.`); |
| 474 | } |
| 475 | const b = { |
| 476 | model, max_tokens: sc.max_tokens || MAXTOK, |
| 477 | messages: [{ role: 'system', content: sys }, { role: 'user', content: sc.user }], |
| 478 | }; |
| 479 | const tools = sc.tools !== undefined ? sc.tools : TOOLS; |
| 480 | if (tools.length) b.tools = tools; |
| 481 | return b; |
| 482 | } |
| 483 | |
| 484 | // ── `--selfcheck`: the ladder proved on texts, before any money ───── |
| 485 | // |
| 486 | // `foldVerdict` decides what every fold measurement means, and until 2026-08-25 nothing put a |
| 487 | // text to it. A verdict function nobody has seen fail is a verdict function nobody has seen. |
| 488 | if (argv.includes('--selfcheck')) { |
| 489 | const body = 'a'.repeat(800); |
| 490 | const sum = '<summary>The store wins on scans and loses on isolation, so I would take ' |
| 491 | + 'the file.</summary>'; |
| 492 | // The lead has to be a real one: `above` is 71 characters and the ladder's FOLD-ALL rule |
| 493 | // fires below 40, which this fixture found out by scoring three cases HARM on a 26-character |
| 494 | // lead. A verdict function proved on unrealistic texts is proved on nothing. |
| 495 | const above = 'Take one file per Diamond: isolation is worth more here than scan speed.'; |
| 496 | const cases = [ |
| 497 | ['GOOD', `${above}\n\n<details>\n${sum}\n\n${body}\n\n</details>`], |
| 498 | ['HARM', `<details>\n${sum}\n\n${body}\n\n</details>`], |
| 499 | ['HARM', `Short.\n\n<details>\n${sum}\n\n${body}\n\n</details>`], |
| 500 | ['PARTIAL', `${above}\n\n<details>\n<summary>Reasoning</summary>\n\n${body}\n\n</details>`], |
| 501 | ['PARTIAL', `${above}\n\n<details>\n${sum}\n${body}\n\n</details>`], |
| 502 | ['NONE', `${above} ${body}`], |
| 503 | ['MOOT', above], |
| 504 | ]; |
| 505 | let bad = 0; |
| 506 | for (const [want, text] of cases) { |
| 507 | const got = foldVerdict({ content: text }); |
| 508 | if (got !== want) { bad++; console.log(` FAIL wanted ${want}, got ${got}`); } |
| 509 | else console.log(` ok ${want.padEnd(8)} ${text.slice(0, 46).replace(/\n/g, ' ')}…`); |
| 510 | } |
| 511 | // The two questions added on 2026-08-25 must really be in the development set, or `--dev` |
| 512 | // silently measures one question and reports three. |
| 513 | const dev = SCENARIOS.filter((s) => s.note === 'FOLD_NOTE' && s.dev === true); |
| 514 | if (dev.length !== 3) { bad++; console.log(` FAIL --dev holds ${dev.length} fold scenarios, wanted 3`); } |
| 515 | else console.log(' ok --dev holds three development fold questions'); |
| 516 | console.log(bad ? `\n${bad} FAILED` : '\nall green'); |
| 517 | process.exit(bad ? 1 : 0); |
| 518 | } |
| 519 | |
| 520 | if (DRY) { |
| 521 | for (const sc of chosen) { |
| 522 | for (const arm of ['with', 'without']) { |
| 523 | const b = bodyFor(sc, MODELS[0], arm); |
| 524 | console.log(`${(sc.id || sc.note).padEnd(20)} ${arm.padEnd(8)} system ${b.messages[0].content.length} chars, ` |
| 525 | + `${(b.tools || []).length} tool(s), max_tokens ${b.max_tokens}`); |
| 526 | } |
| 527 | } |
| 528 | process.exit(0); |
| 529 | } |
| 530 | |
| 531 | const key = readKey(); |
| 532 | if (KEEP) fs.mkdirSync(KEEP, { recursive: true }); |
| 533 | const tally = new Map(); |
| 534 | let spentPrompt = 0, spentOut = 0; |
| 535 | // Per model, so a sweep's bill can be read the way the panel is ordered. |
| 536 | const spentBy = new Map(); |
| 537 | // One line per model per distinct provider complaint, so a wall is named once and not fifty times. |
| 538 | const warned = new Set(); |
| 539 | |
| 540 | for (const sc of chosen) { |
| 541 | for (const model of MODELS) { |
| 542 | for (const arm of ['with', 'without']) { |
| 543 | const seen = []; |
| 544 | for (let i = 0; i < N; i++) { |
| 545 | // ONE RETRY, AND THE REASON IS KEPT. A shared free tier answers 429 from the |
| 546 | // upstream provider rather than from OpenRouter, and a run that scored that as |
| 547 | // UNAVAIL with no words in it looked exactly like a model that had answered |
| 548 | // badly. The whole free half of the panel read that way on 2026-08-24. |
| 549 | let j = null; |
| 550 | for (let attempt = 0; attempt < 2; attempt++) { |
| 551 | const r = await fetch('https://openrouter.ai/api/v1/chat/completions', { |
| 552 | method: 'POST', |
| 553 | headers: { authorization: `Bearer ${key}`, 'content-type': 'application/json' }, |
| 554 | body: JSON.stringify(bodyFor(sc, model, arm)), |
| 555 | }); |
| 556 | j = await r.json(); |
| 557 | if ((j.choices || [])[0]) break; |
| 558 | if (attempt === 0) await new Promise((f) => setTimeout(f, 4000)); |
| 559 | } |
| 560 | const ch = (j.choices || [])[0]; |
| 561 | if (!ch) { |
| 562 | const e = (j && j.error) || {}; |
| 563 | const code = e.code || e.type || '?'; |
| 564 | seen.push(`UNAVAIL:${code}`); |
| 565 | if (!warned.has(`${model}|${code}`)) { |
| 566 | warned.add(`${model}|${code}`); |
| 567 | console.log(` · ${model}: ${String(e.message || 'no reply').slice(0, 140)}`); |
| 568 | } |
| 569 | continue; |
| 570 | } |
| 571 | if (ch.finish_reason === 'length') { seen.push('TRUNC'); continue; } |
| 572 | const u = j.usage || {}; |
| 573 | spentPrompt += Number(u.prompt_tokens || 0); |
| 574 | spentOut += Number(u.completion_tokens || 0); |
| 575 | if (!spentBy.has(model)) spentBy.set(model, { p: 0, o: 0 }); |
| 576 | spentBy.get(model).p += Number(u.prompt_tokens || 0); |
| 577 | spentBy.get(model).o += Number(u.completion_tokens || 0); |
| 578 | const m = ch.message || {}; |
| 579 | if (KEEP) { |
| 580 | fs.writeFileSync(path.join(KEEP, |
| 581 | `${sc.id}.${model.split('/').pop()}.${arm}.${i}.json`), |
| 582 | JSON.stringify(m, null, '\t')); |
| 583 | } |
| 584 | seen.push(sc.verdict(m)); |
| 585 | } |
| 586 | tally.set(`${sc.id}|${model}|${arm}`, seen); |
| 587 | console.log(` ${sc.id.padEnd(20)} ${model.split('/').pop().padEnd(18)} ` |
| 588 | + `${arm.padEnd(8)} ${seen.join(' ')}`); |
| 589 | } |
| 590 | } |
| 591 | } |
| 592 | |
| 593 | console.log(`\n| note | model | with the note | with it stripped |`); |
| 594 | console.log('|---|---|---|---|'); |
| 595 | for (const sc of chosen) { |
| 596 | for (const model of MODELS) { |
| 597 | const w = tally.get(`${sc.id}|${model}|with`) || []; |
| 598 | const o = tally.get(`${sc.id}|${model}|without`) || []; |
| 599 | const sum = (a) => { |
| 600 | const c = new Map(); |
| 601 | for (const v of a) c.set(v, (c.get(v) || 0) + 1); |
| 602 | return [...c].map(([k, n]) => `${n}×${k}`).join(', '); |
| 603 | }; |
| 604 | console.log(`| \`${sc.id}\` | ${model.split('/').pop()} | ${sum(w)} | ${sum(o)} |`); |
| 605 | } |
| 606 | } |
| 607 | console.log(`\n${spentPrompt} prompt token(s) and ${spentOut} completion token(s) over ` |
| 608 | + `${chosen.length * MODELS.length * 2 * N} request(s).`); |
| 609 | for (const [slug, t] of spentBy) { |
| 610 | const usd = priceOf(slug, t.p, t.o); |
| 611 | console.log(` ${slug.padEnd(30)} ${t.p} in, ${t.o} out` |
| 612 | + (usd === null ? ' (not in the panel, so not priced)' : ` ~$${usd.toFixed(4)}`)); |
| 613 | } |