oxedyne/daimond/www/js/pricing.js
21.4 KiB, 1 run
created by r2519314175:1419, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | /* ============================================================ |
| 2 | Daimond — per-model pricing table (DaimondPricing) |
| 3 | ------------------------------------------------------------ |
| 4 | A static, offline lookup of per-token prices, used when |
| 5 | nothing better is available. It is the LAST resort, not the |
| 6 | first: a turn is costed from what the provider itself |
| 7 | reported, failing that from the live rates `DaimondModels` |
| 8 | captured when it listed the provider's models, and only |
| 9 | failing both from the figures baked in here. |
| 10 | |
| 11 | All rates are USD per 1,000,000 tokens. Where a provider |
| 12 | publishes a cached-input rate (prompt reuse, a small |
| 13 | fraction of a fresh read) it is recorded too; otherwise |
| 14 | cached tokens fall back to the ordinary input rate. |
| 15 | |
| 16 | Pricing source: openrouter.ai/api/v1/models, surveyed |
| 17 | 2026-07-30, converted from per-token to per-million. These |
| 18 | are the ROUTED prices actually charged, which is the point: |
| 19 | the table previously held direct-provider list prices 2.3x |
| 20 | to 4.5x above them (glm-5.2 at 1.40/4.40 against a real |
| 21 | 0.62/1.93, deepseek-r1 at 3.00/7.00 against 0.70/2.50), and |
| 22 | an "estimate" built from those overstated a session's spend |
| 23 | roughly six-fold. Prices move; re-verify before relying on |
| 24 | the absolute figures. |
| 25 | |
| 26 | Attaches a single global, `window.DaimondPricing`. |
| 27 | ============================================================ */ |
| 28 | (function () { |
| 29 | 'use strict'; |
| 30 | |
| 31 | // ── Fallback for unknown models ──────────────────────────── |
| 32 | // When nothing knows a model we still want a plausible, |
| 33 | // non-zero cost so the ledger never silently reads $0. It sits |
| 34 | // near the middle of the open-model spread rather than at the |
| 35 | // pricier end: the old 1.00/3.00 caught MOST current router ids |
| 36 | // -- every anthropic/*, openai/gpt-5*, google/gemini-*, grok, |
| 37 | // minimax and qwen3-coder id missed the table -- at about four |
| 38 | // times reality, so the "conservative" figure was the single |
| 39 | // biggest source of overstatement rather than a safety margin. |
| 40 | var FALLBACK = { inUsdPerM: 0.40, outUsdPerM: 1.20, cachedInUsdPerM: 0.10 }; |
| 41 | |
| 42 | // ── Price table ──────────────────────────────────────────── |
| 43 | // Keyed by a canonical model id. Each entry carries: |
| 44 | // in — input (prompt) USD per 1M tokens. |
| 45 | // out — output (completion) USD per 1M tokens. |
| 46 | // cached — cached-input USD per 1M tokens, when published. |
| 47 | // ctx — context window in tokens, or null if unknown. |
| 48 | // alias — extra ids/spellings that map to this entry. |
| 49 | // Grouped by vendor family. A router may serve the same weights |
| 50 | // from several hosts at slightly different rates, so a match |
| 51 | // here is a fair estimate rather than a quote. |
| 52 | var TABLE = { |
| 53 | |
| 54 | // ── Z.ai (GLM) ───────────────────────────────────────── |
| 55 | 'glm-5.2': { |
| 56 | in: 0.6153, out: 1.9338, cached: 0.11427, ctx: 1048576, |
| 57 | alias: ['z-ai/glm-5.2', 'accounts/fireworks/models/glm-5p2', 'glm-5p2', 'glm5.2'], |
| 58 | }, |
| 59 | 'glm-5.1': { |
| 60 | in: 0.966, out: 3.036, cached: 0.1794, ctx: 204800, |
| 61 | alias: ['z-ai/glm-5.1', 'accounts/fireworks/models/glm-5p1', 'glm-5p1', 'glm5.1'], |
| 62 | }, |
| 63 | 'glm-5': { |
| 64 | in: 0.95, out: 2.55, cached: 0.20, ctx: 204800, |
| 65 | alias: ['z-ai/glm-5'], |
| 66 | }, |
| 67 | 'glm-4.7': { |
| 68 | in: 0.40, out: 1.75, cached: 0.08, ctx: 204800, |
| 69 | alias: ['z-ai/glm-4.7'], |
| 70 | }, |
| 71 | 'glm-4.7-flash': { |
| 72 | in: 0.06, out: 0.40, cached: 0.01, ctx: 202752, |
| 73 | alias: ['z-ai/glm-4.7-flash'], |
| 74 | }, |
| 75 | 'glm-4.6': { |
| 76 | in: 0.50, out: 2.00, cached: 0.10, ctx: 204800, |
| 77 | alias: ['z-ai/glm-4.6'], |
| 78 | }, |
| 79 | |
| 80 | // ── OpenAI open weights ──────────────────────────────── |
| 81 | 'gpt-oss-120b': { |
| 82 | in: 0.037, out: 0.17, cached: null, ctx: 131072, |
| 83 | alias: ['accounts/fireworks/models/gpt-oss-120b', 'openai/gpt-oss-120b'], |
| 84 | }, |
| 85 | 'gpt-oss-20b': { |
| 86 | in: 0.03, out: 0.13, cached: 0.03, ctx: 131072, |
| 87 | alias: ['openai/gpt-oss-20b'], |
| 88 | }, |
| 89 | |
| 90 | // ── DeepSeek ─────────────────────────────────────────── |
| 91 | 'deepseek-v4-pro': { |
| 92 | in: 0.435, out: 0.87, cached: 0.003625, ctx: 1048576, |
| 93 | alias: ['deepseek/deepseek-v4-pro', 'accounts/fireworks/models/deepseek-v4-pro', |
| 94 | 'deepseek-ai/deepseek-v4-pro'], |
| 95 | }, |
| 96 | 'deepseek-v4-flash': { |
| 97 | in: 0.14, out: 0.28, cached: 0.028, ctx: 1048576, |
| 98 | alias: ['deepseek/deepseek-v4-flash'], |
| 99 | }, |
| 100 | 'deepseek-r1': { |
| 101 | in: 0.70, out: 2.50, cached: null, ctx: 163840, |
| 102 | alias: ['deepseek/deepseek-r1', 'deepseek-ai/deepseek-r1', 'deepseek-reasoner'], |
| 103 | }, |
| 104 | // Three near-identical v3 ids at three different prices. Each |
| 105 | // keeps its own entry: the substring resolver used to fold |
| 106 | // v3.2-exp onto v3.1 and bill it at more than twice its rate. |
| 107 | 'deepseek-v3.1': { |
| 108 | in: 0.25, out: 0.95, cached: 0.13, ctx: 163840, |
| 109 | alias: ['deepseek/deepseek-chat-v3.1', 'deepseek-chat-v3.1', |
| 110 | 'deepseek-ai/deepseek-v3.1', 'deepseek-v3p1'], |
| 111 | }, |
| 112 | 'deepseek-v3.1-terminus': { |
| 113 | in: 0.27, out: 1.00, cached: 0.135, ctx: 163840, |
| 114 | alias: ['deepseek/deepseek-v3.1-terminus'], |
| 115 | }, |
| 116 | 'deepseek-v3.2': { |
| 117 | in: 0.269, out: 0.40, cached: 0.1345, ctx: 163840, |
| 118 | alias: ['deepseek/deepseek-v3.2', 'deepseek-ai/deepseek-v3.2', 'deepseek-v3p2'], |
| 119 | }, |
| 120 | 'deepseek-v3.2-exp': { |
| 121 | in: 0.27, out: 0.41, cached: null, ctx: 163840, |
| 122 | alias: ['deepseek/deepseek-v3.2-exp'], |
| 123 | }, |
| 124 | |
| 125 | // ── Moonshot (Kimi) ──────────────────────────────────── |
| 126 | 'kimi-k2': { |
| 127 | in: 0.57, out: 2.30, cached: null, ctx: 131072, |
| 128 | alias: ['moonshotai/kimi-k2', 'moonshotai/kimi-k2-instruct', 'kimi-k2-instruct'], |
| 129 | }, |
| 130 | 'kimi-k2.5': { |
| 131 | in: 0.57, out: 2.85, cached: 0.095, ctx: 262144, |
| 132 | alias: ['moonshotai/kimi-k2.5'], |
| 133 | }, |
| 134 | 'kimi-k2.6': { |
| 135 | in: 0.646, out: 2.72, cached: 0.1088, ctx: 262144, |
| 136 | alias: ['moonshotai/kimi-k2.6', 'accounts/fireworks/models/kimi-k2p6', 'kimi-k2p6', |
| 137 | 'kimi-k2-6'], |
| 138 | }, |
| 139 | 'kimi-k2-thinking': { |
| 140 | in: 0.60, out: 2.50, cached: 0.15, ctx: 262144, |
| 141 | alias: ['moonshotai/kimi-k2-thinking'], |
| 142 | }, |
| 143 | 'kimi-k3': { |
| 144 | in: 3.00, out: 15.00, cached: 0.30, ctx: 1048576, |
| 145 | alias: ['moonshotai/kimi-k3'], |
| 146 | }, |
| 147 | |
| 148 | // ── Meta (Llama) ─────────────────────────────────────── |
| 149 | 'llama-3.3-70b': { |
| 150 | in: 0.13, out: 0.40, cached: null, ctx: 131072, |
| 151 | alias: ['llama-3.3-70b-versatile', 'meta-llama/llama-3.3-70b-instruct', 'llama3.3-70b'], |
| 152 | }, |
| 153 | 'llama-4-scout': { |
| 154 | in: 0.10, out: 0.30, cached: null, ctx: 1310720, |
| 155 | alias: ['meta-llama/llama-4-scout', 'meta-llama/llama-4-scout-17b-16e-instruct'], |
| 156 | }, |
| 157 | 'llama-4-maverick': { |
| 158 | in: 0.20, out: 0.80, cached: null, ctx: 1048576, |
| 159 | alias: ['meta-llama/llama-4-maverick', 'meta-llama/llama-4-maverick-17b-128e-instruct'], |
| 160 | }, |
| 161 | |
| 162 | // ── Qwen ─────────────────────────────────────────────── |
| 163 | 'qwen3-235b': { |
| 164 | in: 0.455, out: 1.82, cached: null, ctx: 131072, |
| 165 | alias: ['qwen/qwen3-235b-a22b', 'qwen3-235b-a22b', 'qwen/qwen3-235b'], |
| 166 | }, |
| 167 | 'qwen3-coder': { |
| 168 | in: 0.30, out: 1.00, cached: 0.10, ctx: 262144, |
| 169 | alias: ['qwen/qwen3-coder'], |
| 170 | }, |
| 171 | 'qwen3-max': { |
| 172 | in: 0.78, out: 3.90, cached: 0.156, ctx: 262144, |
| 173 | alias: ['qwen/qwen3-max'], |
| 174 | }, |
| 175 | 'qwen3.7-plus': { |
| 176 | in: 0.32, out: 1.28, cached: 0.064, ctx: 1000000, |
| 177 | alias: ['qwen/qwen3.7-plus', 'accounts/fireworks/models/qwen3p7-plus', 'qwen3p7-plus'], |
| 178 | }, |
| 179 | |
| 180 | // ── MiniMax ──────────────────────────────────────────── |
| 181 | 'minimax-m2': { |
| 182 | in: 0.255, out: 1.02, cached: null, ctx: 204800, |
| 183 | alias: ['minimax/minimax-m2'], |
| 184 | }, |
| 185 | 'minimax-m2.5': { |
| 186 | in: 0.15, out: 0.90, cached: 0.05, ctx: 204800, |
| 187 | alias: ['minimax/minimax-m2.5'], |
| 188 | }, |
| 189 | 'minimax-m3': { |
| 190 | in: 0.30, out: 1.20, cached: 0.06, ctx: 1048576, |
| 191 | alias: ['minimax/minimax-m3'], |
| 192 | }, |
| 193 | |
| 194 | // ── xAI ──────────────────────────────────────────────── |
| 195 | 'grok-4.5': { |
| 196 | in: 2.00, out: 6.00, cached: 0.30, ctx: 500000, |
| 197 | alias: ['x-ai/grok-4.5'], |
| 198 | }, |
| 199 | 'grok-4.3': { |
| 200 | in: 1.25, out: 2.50, cached: 0.20, ctx: 1000000, |
| 201 | alias: ['x-ai/grok-4.3'], |
| 202 | }, |
| 203 | |
| 204 | // ── Anthropic ────────────────────────────────────────── |
| 205 | // Reached two ways: through a router, and — since the client |
| 206 | // learned the Messages API — straight from the browser. The |
| 207 | // second way is the one these figures have to be right for. |
| 208 | // A router reports `cost` on every call and that verbatim |
| 209 | // figure wins; Anthropic reports NO cost at all, so for a |
| 210 | // direct call this table IS the number the user is shown. |
| 211 | // |
| 212 | // Source: the Anthropic pricing table, USD per million tokens, |
| 213 | // read 2026-08-02. `cached` is the cache-READ rate, a tenth of |
| 214 | // the input rate on every model here. The cache-WRITE premium |
| 215 | // (1.25x input for the 5-minute TTL) is not modelled: there is |
| 216 | // one cached rate per entry, and a write happens once per |
| 217 | // prefix where a read happens every turn after it. See |
| 218 | // `AnthUsage::into_usage` in src/llm.rs, which folds writes in |
| 219 | // with the fresh prompt tokens for the same reason. |
| 220 | 'claude-fable-5': { |
| 221 | in: 10.00, out: 50.00, cached: 1.00, ctx: 1000000, |
| 222 | alias: ['anthropic/claude-fable-5'], |
| 223 | }, |
| 224 | 'claude-mythos-5': { |
| 225 | in: 10.00, out: 50.00, cached: 1.00, ctx: 1000000, |
| 226 | alias: ['anthropic/claude-mythos-5'], |
| 227 | }, |
| 228 | 'claude-opus-5': { |
| 229 | in: 5.00, out: 25.00, cached: 0.50, ctx: 1000000, |
| 230 | alias: ['anthropic/claude-opus-5'], |
| 231 | }, |
| 232 | 'claude-opus-4-8': { |
| 233 | in: 5.00, out: 25.00, cached: 0.50, ctx: 1000000, |
| 234 | alias: ['anthropic/claude-opus-4-8'], |
| 235 | }, |
| 236 | 'claude-opus-4-7': { |
| 237 | in: 5.00, out: 25.00, cached: 0.50, ctx: 1000000, |
| 238 | alias: ['anthropic/claude-opus-4-7'], |
| 239 | }, |
| 240 | 'claude-opus-4-6': { |
| 241 | in: 5.00, out: 25.00, cached: 0.50, ctx: 1000000, |
| 242 | alias: ['anthropic/claude-opus-4-6'], |
| 243 | }, |
| 244 | // INTRODUCTORY pricing, in force through 2026-08-31. It |
| 245 | // reverts to 3.00 / 15.00 / 0.30 on 2026-09-01 — update this |
| 246 | // line then. The intro figure is used rather than the standard |
| 247 | // one because it is what the user is charged today, and a |
| 248 | // table that quotes list price against a discounted invoice is |
| 249 | // the overstatement this file was rewritten to stop. |
| 250 | 'claude-sonnet-5': { |
| 251 | in: 2.00, out: 10.00, cached: 0.20, ctx: 1000000, |
| 252 | alias: ['anthropic/claude-sonnet-5'], |
| 253 | }, |
| 254 | 'claude-sonnet-4-6': { |
| 255 | in: 3.00, out: 15.00, cached: 0.30, ctx: 1000000, |
| 256 | alias: ['anthropic/claude-sonnet-4-6'], |
| 257 | }, |
| 258 | 'claude-haiku-4.5': { |
| 259 | in: 1.00, out: 5.00, cached: 0.10, ctx: 200000, |
| 260 | alias: ['anthropic/claude-haiku-4.5', 'claude-haiku-4-5'], |
| 261 | }, |
| 262 | // Still served, and still one click away in the picker. Left out, |
| 263 | // they fell to the unknown-model fallback — which prices an Opus |
| 264 | // turn at a twelfth of what it costs, and an Opus 4.1 turn at a |
| 265 | // fortieth. Understating a bill is not the safe direction either. |
| 266 | 'claude-opus-4-5': { |
| 267 | in: 5.00, out: 25.00, cached: 0.50, ctx: 200000, |
| 268 | alias: ['anthropic/claude-opus-4-5', 'claude-opus-4-5-20251101'], |
| 269 | }, |
| 270 | 'claude-sonnet-4-5': { |
| 271 | in: 3.00, out: 15.00, cached: 0.30, ctx: 200000, |
| 272 | alias: ['anthropic/claude-sonnet-4-5', 'claude-sonnet-4-5-20250929'], |
| 273 | }, |
| 274 | 'claude-opus-4-1': { |
| 275 | in: 15.00, out: 75.00, cached: 1.50, ctx: 200000, |
| 276 | alias: ['anthropic/claude-opus-4-1', 'claude-opus-4-1-20250805'], |
| 277 | }, |
| 278 | |
| 279 | // ── Other closed models a router also serves ─────────── |
| 280 | // Not open weights, but one line away in the picker, and the |
| 281 | // fallback priced them at a quarter of what a turn costs. |
| 282 | 'gpt-5.4': { |
| 283 | in: 2.50, out: 15.00, cached: 0.25, ctx: 1050000, |
| 284 | alias: ['openai/gpt-5.4'], |
| 285 | }, |
| 286 | 'gpt-5.4-mini': { |
| 287 | in: 0.75, out: 4.50, cached: 0.075, ctx: 400000, |
| 288 | alias: ['openai/gpt-5.4-mini'], |
| 289 | }, |
| 290 | 'gpt-5.2': { |
| 291 | in: 1.75, out: 14.00, cached: 0.175, ctx: 400000, |
| 292 | alias: ['openai/gpt-5.2'], |
| 293 | }, |
| 294 | 'gemini-3.1-pro': { |
| 295 | in: 2.00, out: 12.00, cached: 0.20, ctx: 1048576, |
| 296 | alias: ['google/gemini-3.1-pro-preview'], |
| 297 | }, |
| 298 | 'gemini-3.6-flash': { |
| 299 | in: 1.50, out: 7.50, cached: 0.15, ctx: 1048576, |
| 300 | alias: ['google/gemini-3.6-flash'], |
| 301 | }, |
| 302 | 'gemini-2.5-flash': { |
| 303 | in: 0.30, out: 2.50, cached: 0.03, ctx: 1048576, |
| 304 | alias: ['google/gemini-2.5-flash'], |
| 305 | }, |
| 306 | }; |
| 307 | |
| 308 | // ── Normalisation + index ────────────────────────────────── |
| 309 | // Fold an id to a comparison key: take the last path segment, |
| 310 | // rewrite a provider's `NpM` spelling to `N.M` (so `glm-5p2` |
| 311 | // meets `glm-5.2`), then drop every non-alphanumeric. Thus |
| 312 | // `accounts/fireworks/models/glm-5p2` and `GLM-5.2` both |
| 313 | // become `glm52`. |
| 314 | function norm(id) { |
| 315 | var s = String(id == null ? '' : id).toLowerCase(); |
| 316 | var seg = s.split('/').pop(); // last path segment |
| 317 | seg = seg.replace(/(\d)p(\d)/g, '$1.$2'); // 5p2 → 5.2 |
| 318 | return seg.replace(/[^a-z0-9]/g, ''); // glm-5.2 → glm52 |
| 319 | } |
| 320 | |
| 321 | // Build a normalised-key → canonical-id index once, covering |
| 322 | // each canonical id and all its aliases. Keys are also held in |
| 323 | // longest-first order, which is what `resolve` walks. |
| 324 | var INDEX = {}; |
| 325 | var KEYS = []; |
| 326 | (function () { |
| 327 | for (var id in TABLE) { |
| 328 | INDEX[norm(id)] = id; |
| 329 | var aliases = TABLE[id].alias || []; |
| 330 | for (var i = 0; i < aliases.length; i++) INDEX[norm(aliases[i])] = id; |
| 331 | } |
| 332 | KEYS = Object.keys(INDEX).sort(function (a, b) { return b.length - a.length; }); |
| 333 | })(); |
| 334 | |
| 335 | // The entry the caller actually named: exact key or alias, and |
| 336 | // nothing else. A hit here IS the model, so it is the only kind |
| 337 | // of hit allowed to state a context window. |
| 338 | function resolveExact(model) { |
| 339 | var key = norm(model); |
| 340 | if (!key) return null; |
| 341 | return INDEX[key] ? TABLE[INDEX[key]] : null; |
| 342 | } |
| 343 | |
| 344 | // Resolve a caller's model string to a table entry, or null. |
| 345 | // |
| 346 | // Exact key or alias first; then the LONGEST table key that the |
| 347 | // caller's id CONTAINS. The direction matters. This used to test |
| 348 | // both ways and return the first hit in object order, so |
| 349 | // `deepseek/deepseek-v3.2-exp` met the earlier `deepseekv3` alias |
| 350 | // and was billed at v3.1's rate -- more than twice its own. A |
| 351 | // shorter key can never outrank a longer one now, and a key that |
| 352 | // merely contains the caller's id (`glm52` offered for a bare |
| 353 | // `glm5`) is not a match at all. |
| 354 | // |
| 355 | // That last guard used to be written here as though it settled |
| 356 | // the glm-5 / glm-5.2 confusion. It settles ONE HALF of it. The |
| 357 | // same pair collides the other way round and is not caught: |
| 358 | // `glm-5.3` normalises to `glm53`, which contains `glm5`, so it |
| 359 | // lands on the glm-5 entry -- a fifth of glm-5.2's window and a |
| 360 | // price nobody published. The stated reason ("ids arrive whole") |
| 361 | // is true and does not help, because a whole id can be a whole |
| 362 | // DIFFERENT model whose name starts with a table key. |
| 363 | // |
| 364 | // Nor can containment be tightened to tell the two cases apart: |
| 365 | // `norm` has already dropped the separator that carried the |
| 366 | // difference, so `claudeopus5` + `20251001` (the same model, |
| 367 | // dated, which should match) and `glm5` + `3` (a version bump, |
| 368 | // which should not) are one shape by the time this sees them. |
| 369 | // So the fallback is kept for RATES and marked `near` -- see |
| 370 | // `entryFor` -- and supplies no context window at all. A wrong |
| 371 | // price is visible in the spend readout; a wrong window silently |
| 372 | // clips a conversation. |
| 373 | function resolve(model) { |
| 374 | var key = norm(model); |
| 375 | if (!key) return null; |
| 376 | if (INDEX[key]) return TABLE[INDEX[key]]; // exact or alias |
| 377 | for (var i = 0; i < KEYS.length; i++) { |
| 378 | if (key.indexOf(KEYS[i]) !== -1) return TABLE[INDEX[KEYS[i]]]; |
| 379 | } |
| 380 | return null; |
| 381 | } |
| 382 | |
| 383 | // The live rates a provider published when `DaimondModels` last |
| 384 | // listed its models, in this table's shape. A router's own |
| 385 | // figures beat anything baked in here, so they are asked first. |
| 386 | // Absent module, absent provider or absent model: null, and the |
| 387 | // table answers instead. |
| 388 | function liveEntry(provider, model) { |
| 389 | if (!provider || !window.DaimondModels |
| 390 | || typeof window.DaimondModels.rateFor !== 'function') return null; |
| 391 | var r = null; |
| 392 | try { r = window.DaimondModels.rateFor(provider, model); } catch (e) { return null; } |
| 393 | if (!r || typeof r.inPerM !== 'number' || typeof r.outPerM !== 'number') return null; |
| 394 | return { |
| 395 | in: r.inPerM, |
| 396 | out: r.outPerM, |
| 397 | cached: (typeof r.cachedPerM === 'number') ? r.cachedPerM : null, |
| 398 | ctx: (typeof r.ctx === 'number') ? r.ctx : null, |
| 399 | }; |
| 400 | } |
| 401 | |
| 402 | // The entry to price with, and how far it is from the model asked |
| 403 | // about. `live` is a quote from the provider, `table` a surveyed |
| 404 | // figure for this very model, `near` a surveyed figure for a |
| 405 | // NEIGHBOUR the id merely contains, `fallback` no figure at all. |
| 406 | // The last two are estimates: neither is a number anybody |
| 407 | // published about this model. |
| 408 | function entryFor(model, provider) { |
| 409 | var live = liveEntry(provider, model); |
| 410 | if (live) return { entry: live, source: 'live' }; |
| 411 | var exact = resolveExact(model); |
| 412 | if (exact) return { entry: exact, source: 'table' }; |
| 413 | // `resolve` has already tried exact, so what is left here is a |
| 414 | // containment hit and only that. |
| 415 | var nearby = resolve(model); |
| 416 | if (nearby) return { entry: nearby, source: 'near' }; |
| 417 | return { entry: FALLBACK_ENTRY(), source: 'fallback' }; |
| 418 | } |
| 419 | |
| 420 | // Is this a figure nobody published about the model asked about? |
| 421 | function guessed(source) { |
| 422 | return source === 'fallback' || source === 'near'; |
| 423 | } |
| 424 | |
| 425 | // ── Public API ───────────────────────────────────────────── |
| 426 | |
| 427 | /// Cost a turn for `model` given its token counts. |
| 428 | /// |
| 429 | /// `cachedTokens` is the portion of `promptTokens` served from |
| 430 | /// the provider's prompt cache; it is billed at the cached rate |
| 431 | /// where one is published, and at the ordinary input rate |
| 432 | /// otherwise. The remaining prompt tokens bill at the input |
| 433 | /// rate and completion tokens at the output rate. |
| 434 | /// |
| 435 | /// `provider` is optional and, when given, lets the live rates |
| 436 | /// captured from that provider's own model list take precedence |
| 437 | /// over the baked-in table. |
| 438 | /// |
| 439 | /// Returns `{ usd, estimated, source }`. `estimated` is true |
| 440 | /// whenever no rate was published for this model itself: the |
| 441 | /// fallback was applied (`source: 'fallback'`), or the figure was |
| 442 | /// borrowed from a table entry the id merely contains |
| 443 | /// (`source: 'near'`). |
| 444 | function priceFor(model, promptTokens, completionTokens, cachedTokens, provider) { |
| 445 | var got = entryFor(model, provider); |
| 446 | var r = got.entry; |
| 447 | |
| 448 | var prompt = Math.max(0, promptTokens || 0); |
| 449 | var completion = Math.max(0, completionTokens || 0); |
| 450 | var cached = Math.max(0, Math.min(cachedTokens || 0, prompt)); // a subset of the prompt |
| 451 | var fresh = prompt - cached; // prompt tokens billed at input rate |
| 452 | |
| 453 | var cachedRate = (r.cached != null) ? r.cached : r.in; |
| 454 | var usd = (fresh * r.in + cached * cachedRate + completion * r.out) / 1e6; |
| 455 | return { usd: usd, estimated: guessed(got.source), source: got.source }; |
| 456 | } |
| 457 | |
| 458 | // The fallback expressed in the same shape as a table entry. |
| 459 | function FALLBACK_ENTRY() { |
| 460 | return { in: FALLBACK.inUsdPerM, out: FALLBACK.outUsdPerM, cached: FALLBACK.cachedInUsdPerM, ctx: null }; |
| 461 | } |
| 462 | |
| 463 | /// Context window for `model` in tokens, or null when unknown |
| 464 | /// (or when the model itself is unknown). A provider's own |
| 465 | /// figure, where one was captured, beats the table's. |
| 466 | /// |
| 467 | /// `resolveExact`, not `resolve`: a neighbour's window is not |
| 468 | /// this model's window, and `glm-5.3` borrowing glm-5's 204,800 |
| 469 | /// clipped a conversation to a fifth of what it could have held, |
| 470 | /// silently. A null is left alone by every caller -- the agent's |
| 471 | /// own default assumption is a better guess than a number |
| 472 | /// invented here, and the reactive fold still catches a refusal. |
| 473 | function contextWindow(model, provider) { |
| 474 | var live = liveEntry(provider, model); |
| 475 | if (live && live.ctx != null) return live.ctx; |
| 476 | var entry = resolveExact(model); |
| 477 | return (entry && entry.ctx != null) ? entry.ctx : null; |
| 478 | } |
| 479 | |
| 480 | /// Display rates for `model` as `{ inUsdPerM, outUsdPerM, |
| 481 | /// cachedInUsdPerM, source }`, or null when nothing knows the |
| 482 | /// model. A null `cachedInUsdPerM` means no separate cached rate |
| 483 | /// is published. `source` is named exactly as `priceFor` names |
| 484 | /// it, so a display and a charge cannot disagree about how well |
| 485 | /// the figure is known. |
| 486 | function rate(model, provider) { |
| 487 | var live = liveEntry(provider, model); |
| 488 | var entry = live || resolve(model); |
| 489 | if (!entry) return null; |
| 490 | return { |
| 491 | inUsdPerM: entry.in, |
| 492 | outUsdPerM: entry.out, |
| 493 | cachedInUsdPerM: (entry.cached != null) ? entry.cached : null, |
| 494 | source: live ? 'live' : (resolveExact(model) ? 'table' : 'near'), |
| 495 | }; |
| 496 | } |
| 497 | |
| 498 | window.DaimondPricing = { |
| 499 | priceFor: priceFor, |
| 500 | contextWindow: contextWindow, |
| 501 | rate: rate, |
| 502 | /// The fallback rate applied to unknown models, for display. |
| 503 | fallback: { inUsdPerM: FALLBACK.inUsdPerM, outUsdPerM: FALLBACK.outUsdPerM, cachedInUsdPerM: FALLBACK.cachedInUsdPerM }, |
| 504 | /// The table and both resolvers, for the pure-node hygiene check |
| 505 | /// in `dev/verify_pricing.mjs`: every canonical id must resolve |
| 506 | /// to itself, no id may resolve to an entry other than the |
| 507 | /// owner of its longest matching key, and an id that only |
| 508 | /// `resolve` can place must carry no window. Read-only by |
| 509 | /// convention. |
| 510 | _core: { TABLE: TABLE, INDEX: INDEX, KEYS: KEYS, norm: norm, |
| 511 | resolve: resolve, resolveExact: resolveExact }, |
| 512 | }; |
| 513 | })(); |