Oregami
Repositories/oxedyne/daimond

oxedyne/daimond/dev/run_all.sh

63.2 KiB, 1 run

created by r2519314175:127, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1#!/bin/bash
2# run_all.sh -- run every functional verifier, each in the environment it needs.
3#
4# The verifiers do NOT all want the same world, and a flat loop over them cannot
5# be right for all three groups at once:
6#
7# A. Most want NO gateway at all. `verify_credits` fetch-stubs the gateway on
8# purpose; and with a real one up, a non-Pro test account gets a 402 from
9# sync, which raises the "Sync is part of Pro" dialog OVER the app, so every
10# later click in that test times out against it.
11# B. Some start their OWN gateway and REFUSE to run if one is already there,
12# because they pin an owner in configuration (operators, releases, logout).
13# They need the port free, same as A.
14# C. A few need a gateway ALREADY running, and two of those need an account
15# holding the `email` unlock and some credits (tools, compose), plus the
16# mail fixtures (compose).
17#
18# So: phase 1 runs A and B with the gateway port clear; phase 2 brings a gateway
19# up, grants what C needs, and runs C. Anything that cannot be provisioned SKIPS
20# loudly -- a verifier that silently did not run is worse than one that failed.
21#
22# ALL THREE PORTS ARE THIS RUN'S OWN. They were :9002, :1143 and :1587, one
23# instance of each on the machine, and that is what made "one gate at a time" a
24# rule rather than a preference. `dev/world.sh` gives each world its own
25# (9700 + N, 1143 + N, 1587 + N) and exports them; nothing here derives a port,
26# it reads one.
27#
28# Needs `dev/serve.mjs` (`DAIMOND_PORT`, default 8777) and `dev/mockllm.mjs`
29# (`DAIMOND_MOCK_PORT`, default 9099) -- without the mock every `@tool` call quietly
30# does nothing. Start those yourself; this script does not, so a suite run never
31# kills a server you were using. `bash dev/world.sh N --env` prints a matching set,
32# and $DAIMOND_SCRATCH below then keeps this run's logs and profiles to itself.
33#
34# LOG=/tmp/suite.log bash dev/run_all.sh # everything
35# bash dev/run_all.sh verify_tags verify_doc # just these
36cd "$(dirname "$0")/.."
37ROOT=$(pwd)
38
39# Scratch root for logs, profiles and artefacts. NOT /tmp: it is a tmpfs, so
40# leftovers there are RAM charged to the agent fleet's cgroup, reclaimable only
41# by swapping -- three OOM incidents have come out of it. See harness.mjs.
42SCRATCH=${DAIMOND_SCRATCH:-$HOME/.cache/daimond}
43mkdir -p "$SCRATCH"
44
45LOG=${LOG:-$SCRATCH/suite.log} # override with LOG=... ; session-independent
46GW_BIN=${GW_BIN:-gateway/target/release/daimond_gateway}
47CTL_BIN=${CTL_BIN:-gateway/target/release/daimond_ctl}
48# THE GATEWAY'S PORT IS THIS RUN'S, NOT THE FLEET'S.
49#
50# It was the literal 9002 in five places, and `dev/world.sh` says why that is not
51# good enough any more: every world-numbered port is a lane's own, but the gateway,
52# the IMAP and SMTP fixtures and the forge are "fixed and shared by every world".
53# In a fleet of one that is a rule; in a fleet of six it is a queue, and a lane that
54# wants the two mail verifiers cannot have them for as long as anybody else holds
55# the port.
56#
57# `serve.mjs` has always read `DAIMOND_GW_PORT` (default 9002) to decide where to
58# proxy `/api`, so the app side needed nothing. What was missing was the other two
59# halves: the gateway's own `listen_port`, which `dev/devgw.sh` now rewrites, and
60# this script's health checks, which asked :9002 whatever the run was told.
61GW_PORT=${DAIMOND_GW_PORT:-9002}
62export DAIMOND_GW_PORT=$GW_PORT
63# The mail fixtures, on the same reasoning and for the same reason: one IMAP
64# server and one submission stand-in on the machine meant one lane could run
65# verify_compose at a time, and the other lanes' runs of it SKIPPED with
66# "mail fixtures absent" -- which reads as a missing build, not as a queue.
67IMAP_PORT=${DAIMOND_IMAP_PORT:-1143}
68SMTP_PORT=${SMTPD_PORT:-1587}
69export DAIMOND_IMAP_PORT=$IMAP_PORT
70export SMTPD_PORT=$SMTP_PORT
71# An entitled identity's browser profile is $SCRATCH/<identity>-profile; see ident_for.
72IMAP_FIXTURE=${IMAP_FIXTURE:-$HOME/usr/code/rust/fe2o3/target/debug/examples/imap_test_server}
73SMTPD=${SMTPD:-dev/smtpd.mjs}
74
75# Which verifiers need a gateway ALREADY up (group C) -- ASKED OF THE FILES.
76#
77# This was a hand-kept list of thirteen names, and a hand-kept list is the thing
78# that goes stale: `verify_look` and `verify_wakerearm` each STATE the
79# requirement in their own header ("Needs dev/serve.mjs (DAIMOND_PORT) AND the
80# gateway on :9002") and neither was ever added to it. Both landed on 2026-08-12
81# and both have failed identically in every gate since -- `webhook 0, pro=false`,
82# which is a fetch that never connected -- while every browser-only check in
83# verify_look passed. Two days of red that said nothing about the app. The same
84# accident had already cost verify_gwretry and verify_sessionrenew, which exit 0
85# with a SKIP line when :9002 is clear and so were reported as PASSING for every
86# run in which they refused to do anything at all.
87#
88# So it is derived, on the same reasoning `wants_gateway` already applies below:
89# ask the verifier, not a list somebody has to remember to edit. What is asked is
90# the header comment, because that is where every one of these declares itself,
91# and the answer is checked in two directions:
92#
93# IT MUST ASK FOR ONE. A sentence in the header naming the gateway together
94# with the thing that has to be there -- :9002, or the `dev_insecure` mail
95# config phase 2 generates. A mention is not a requirement: "No gateway on
96# :9002" and "It does NOT need ... the gateway on :9002" are NINE other files
97# saying the opposite, and are excluded by the words that make them opposite.
98#
99# AND IT MUST NOT BRING ITS OWN. Group B spawns a gateway and REFUSES to run
100# with one already up, so it belongs in phase 1. `procLog` is gwbin's "I am
101# about to spawn a server and need somewhere to log it" helper, and importing
102# it is what tells the two groups apart -- verify_passkey_blob talks to a live
103# gateway and imports SUITE_GW_LOG instead, saying in its own comment that it
104# "starts no gateway of its own".
105#
106# Checked against the list it replaces (2026-08-14): it reproduces all thirteen
107# names exactly, and adds exactly verify_look and verify_wakerearm. The phase
108# lines this script prints are the audit -- every run says which verifiers it put
109# in which phase, which is what a list in a file never did.
110needs_live_gateway() { # name -> 0 if it needs a gateway ALREADY up
111 local f="dev/$1.mjs"
112 [ -f "$f" ] || return 1
113 grep -q 'procLog(' "$f" && return 1 # it starts one of its own
114 awk '
115 /^[[:space:]]*\/\// { sub(/^[[:space:]]*\/\/[[:space:]]?/, ""); h = h " " $0; next }
116 /^[[:space:]]*$/ { next }
117 { exit } # the header ends at the first code
118 END {
119 n = split(h, s, /\. /)
120 for (i = 1; i <= n; i++) {
121 t = tolower(s[i])
122 if (t !~ /gateway/) continue
123 # The port token, in either spelling. `:9002` is what every header
124 # in the tree says today; `daimond_gw_port` is what a header
125 # written since the gateway became per-world should say, and
126 # accepting it now means the first file to be reworded does not
127 # silently change phase.
128 if (t !~ /:9002|daimond_gw_port|dev_insecure/) continue
129 if (t ~ /no gateway|not need|free :9002|starts its own|start its own|spawns/) continue
130 exit 0
131 }
132 exit 1
133 }
134 ' "$f"
135}
136# Of those, the ones that also need an entitled account (and, for compose, mail).
137# Still a list: an entitlement is not something a header declares, and these are
138# named again by the provisioning block below in any case.
139#
140# `verify_sync` JOINED THIS LIST ON 2026-08-24, and what it cost to be missing is
141# the argument for reading the next paragraph rather than skipping it. It drives
142# an identity of its OWN -- `sync` -- and this block provisioned exactly one,
143# `compose`. So its account never held Pro, sync is Pro-gated, and the file came
144# out of every gate at 37 failed / 68 passed with `pro=false` on its second line
145# and all thirty-five others downstream of it. Nothing was wrong with the engine:
146# provisioned by hand it answers 177/177. That is the same defect this file's own
147# header records against `verify_look` and `verify_wakerearm` -- a hand-kept list
148# that nobody remembered to edit -- reappearing one field along.
149NEEDS_GRANT="verify_compose verify_mailfolders verify_sync verify_pausesync verify_sessionrenew"
150
151# WHICH IDENTITY EACH OF THEM DRIVES, because they do not all drive one.
152#
153# An account belongs to an identity, and an identity is a name plus the harness's
154# fixed passphrase; two verifiers signing in under different names are two
155# accounts, however much profile they share. `provision.mjs <profile> <name>` is
156# what turns one into an account id.
157ident_for() {
158 case "$1" in
159 verify_compose|verify_mailfolders) echo "compose" ;;
160 verify_sync) echo "sync" ;;
161 # Two more that drive identities of their own, found the same way verify_sync was:
162 # `pro=false` and a 401 on their second line, and every check below downstream of it.
163 #
164 # THESE TWO STILL SKIP, and the skip is the honest answer rather than the fix.
165 # `provision.mjs` mints an account for an identity that already has a browser profile;
166 # for a NAME THAT HAS NEVER RUN it answers "the gateway did not name an account" even
167 # with the gateway up, so `pausesync` and `sessrenew` are named here, are provisioned
168 # for, fail to be, and SKIP loudly saying so. That is better than the three reds they
169 # gave before -- a verifier that did not run should never read as a verifier that
170 # failed -- and it is not the same thing as working. Whoever takes it: the first run
171 # of a new identity has to create the profile before `/api/account` is asked.
172 verify_pausesync) echo "pausesync" ;;
173 verify_sessionrenew) echo "sessrenew" ;;
174 *) echo "" ;;
175 esac
176}
177
178# Verifiers that need something this suite cannot invent, and that say so by
179# exiting 2 rather than pretending. `verify_droots_real` proves the Diamond
180# migration against a backup exported from a REAL install: without one it prints
181# "SKIPPED -- and this is NOT a pass" and exits 2, deliberately, so that nobody
182# can read a suite line as a claim about a corpus that was never opened. Point
183# DAIMOND_BACKUP at such a file and it runs; leave it unset and it is skipped
184# here, loudly, and never counted as a pass.
185#
186# DAIMOND_BACKUP=~/Downloads/daimond-backup-2026-08-01.json bash dev/run_all.sh
187args_for() {
188 case "$1" in
189 verify_droots_real) echo "--backup $DAIMOND_BACKUP" ;;
190 # The free half: the journeys, no model, no key, no spend. A suite cannot
191 # carry the paid run, and the wire is the half that can be unwired silently.
192 refluxduo) echo "--wire --journey tools" ;;
193 *) echo "" ;;
194 esac
195}
196needs_input() { # name -> prints why it cannot run, or nothing
197 case "$1" in
198 verify_droots_real)
199 [ -n "${DAIMOND_BACKUP:-}" ] || {
200 echo "needs a backup from a real install: DAIMOND_BACKUP=<file.json> (it exits 2 rather than pass)"
201 return
202 }
203 [ -f "$DAIMOND_BACKUP" ] || echo "DAIMOND_BACKUP=$DAIMOND_BACKUP is not a file" ;;
204 # ASKED OF THE VERIFIER, not answered here. `verify_conformance` measures a
205 # LIVE Oregami forge, and where that forge is meant to be is resolved from
206 # ORE_FORGE or gateway/app.jdat by the same lines that will do the asking.
207 # Deriving the address a second time in this file is the staleness this
208 # script's header spends four paragraphs arguing against, so `--why-not`
209 # answers instead: silence if it can run, one sentence if it cannot.
210 #
211 # It was a FAIL on the 2026-08-17 gate -- one line among fifty-six reds,
212 # indistinguishable from a forge that had answered wrongly, when no forge was
213 # running at all. Pointing it at `dev/mock_forge.mjs` was the other way out
214 # and is wrong: the mock was written FROM the contract, so a conformance run
215 # against it proves the contract agrees with itself. That file's own header
216 # names this -- "31 passed" was a true sentence about the wrong repository --
217 # and a suite that answered it with a second wrong repository would have
218 # learned nothing from the day that produced the file.
219 verify_conformance)
220 node dev/verify_conformance.mjs --why-not 2>/dev/null ;;
221 # The ONE verifier in this tree that spends real money at a real provider.
222 # It reaches `dev/reflux.mjs` for a daimon (see its header, and BLOCKERS
223 # B13), and reflux drives a real model through a real browser. Its own
224 # checks are free and it exits 2 by itself where there is no key -- but
225 # once the owner puts a key in place, an unguarded suite would spend from
226 # it on every gate, several times a day, and nobody would notice until the
227 # key was empty. So the SUITE refuses it and a daimon asking for it by name
228 # through `verify` still gets the run.
229 verify_reflux)
230 [ -n "${DAIMOND_REFLUX_PAID:-}" ] || echo "spends real money at a real provider; set DAIMOND_REFLUX_PAID=1 to let a suite run it (it exits 2 rather than pass)" ;;
231 esac
232}
233
234# The extension flows load a real unpacked extension, which Chromium will only
235# do HEADED -- and a headed browser needs a display. Never the user's: an X
236# forward that has gone quiet (a sleeping laptop at the other end) fails these
237# with "Missing X server", and a live one throws windows in their face. Xvfb
238# gives them a display of their own.
239# verify_handreal is here too, and is the only one that also builds a Rust binary
240# and runs a real `cargo test` through the browser -- see its own header.
241#
242# verify_scope, verify_kitfence, verify_pty, verify_ptyedge and verify_sweep_mobile
243# were MISSING from this list and each asks for a real window -- the first four
244# load an unpacked extension, the last drives device emulation. Run without a
245# display they fail on "Missing X server", which reads on the summary as the
246# fence being broken rather than the suite being wrong about how to start them;
247# run WITH the user's display they throw windows into whatever they were doing.
248#
249# FOUR MORE JOINED THE LIST ON 2026-08-25 and the accident is the one the paragraph
250# above describes, happening a second time: `verify_consolenav`,
251# `verify_interfacediagram`, `verify_search_console` and `verify_vocabulary` each
252# call `chromium.launch({ headless: false })` in their own source and none was
253# named here. Without `xvfb-run` a headed launch has two outcomes and both are bad:
254# "Missing X server" on a box with no display, which reads on a summary as the app
255# being broken; or, on this box, a real window thrown onto the owner's desktop --
256# `dev/display.mjs` strips `WAYLAND_DISPLAY` so `DISPLAY` is honoured, and if that
257# is the seat's own it is the seat's own screen.
258HEADED="verify_ext verify_grant verify_hand verify_ext_i18n verify_handrun verify_handreal \
259verify_scope verify_kitfence verify_pty verify_ptyedge verify_sweep_mobile \
260verify_consolenav verify_interfacediagram verify_search_console verify_vocabulary"
261
262# AND THE LIST IS ASKED OF THE FILES, at the one thing a file can be asked.
263#
264# This is a hand-kept list twice caught stale, and the rest of this script long ago
265# stopped keeping those -- `wants_gateway` and `needs_live_gateway` each ask the
266# verifier instead. This one cannot be fully derived: eleven of the fifteen launch
267# through `dev/extdev.mjs` or the harness rather than saying `headless: false`
268# themselves, so a grep is a floor and not a ceiling. It is exact in the direction
269# that matters, which is the direction the list has failed in both times: a file
270# that says `headless: false` in its own source and is NOT named above stops the
271# suite here, before two hours of browsers, rather than after.
272#
273# ONE FILE STARTS ITS OWN DISPLAY AND MUST NOT BE GIVEN ONE. `verify_reflux`
274# reaches `dev/reflux.mjs` for a daimon, and the guard it was built around
275# refuses any display it did not start itself -- an inherited one and `:0`
276# alike, because the hand is started by the browser and so carries the seat of
277# whoever is sitting at the machine. Wrapping it in the suite's `xvfb-run` hands
278# it exactly the inherited display that guard exists to refuse, so it would fail
279# on its own protection. It is named here rather than added to HEADED because
280# the two lists mean opposite things: HEADED is "needs a display from us", this
281# is "brings its own". The suite refuses to run it at all for money reasons; the
282# list above must still not stop the whole suite on its account.
283OWN_DISPLAY="verify_reflux"
284
285missing_headed=""
286for f in dev/verify_*.mjs; do
287 grep -q 'headless: *false' "$f" || continue
288 n=$(basename "$f" .mjs)
289 case " $OWN_DISPLAY " in *" $n "*) continue ;; esac
290 case " $HEADED " in *" $n "*) ;; *) missing_headed="$missing_headed $n" ;; esac
291done
292if [ -n "$missing_headed" ]; then
293 echo "FATAL these verifiers launch \`headless: false\` and are not in HEADED:$missing_headed"
294 echo " Run without xvfb-run they either die on \"Missing X server\", which reads as a"
295 echo " product failure, or open a window on whoever's screen DISPLAY names. Add them to"
296 echo " HEADED in dev/run_all.sh. Nothing has been run."
297 exit 2
298fi
299# verify_style walks 3 themes x 3 device sizes and is simply slower than the rest.
300# verify_handreal builds the hand from source before it can drive it, and a cold
301# release build of the hand is minutes rather than seconds. verify_ptyedge builds
302# a whole wasm package per property it proves against broken code.
303#
304# verify_reversible and verify_sweep_desktop were never listed here, so both drew
305# the 180s default and both were killed by it -- reported as exit 124 and carried
306# for days as unexplained reds. Neither is broken: reversible passes in 209s
307# (measured 2026-08-07) because it opens every control on every surface in its own
308# isolated session, and sweep_desktop walks every skin against both spacings, the
309# same shape of matrix verify_sweep_mobile already gets 900s for. A budget is a
310# claim about how long a thing takes; an unlisted verifier makes that claim by
311# accident.
312slow_for() {
313 case "$1" in
314 verify_style) echo 600 ;;
315 verify_scope|verify_kitfence) echo 600 ;;
316 verify_reversible) echo 420 ;;
317 verify_sweep_mobile) echo 900 ;;
318 verify_sweep_desktop) echo 900 ;;
319 verify_handreal) echo 900 ;;
320 verify_ptyedge) echo 2400 ;;
321 # A real model, a real browser and a hand built from source before either
322 # of them -- verify_handreal's 900 plus a turn's worth of a provider.
323 verify_reflux) echo 1800 ;;
324 # Two devices, a gateway and a full parcel round trip each way. It has
325 # been over the default for a while and nobody noticed, because a killed
326 # verifier does not say it was killed: `timeout` cuts the browser out
327 # from under it and Playwright reports "Target page, context or browser
328 # has been closed" as six ordinary-looking sync failures. The 2026-08-10
329 # gate spent its whole red budget on those six, all of which were this.
330 verify_sync) echo 1200 ;;
331 # Two real browser profiles, a gateway, and two waits measured in the
332 # engine's own constants: the catch-up (20s) and the focus throttle. It
333 # spends most of its time NOT touching the second device, which is the
334 # whole point of it -- a check that hurried would be testing something
335 # else. MEASURED at 220s on a quiet box (2026-08-28); 480 on the same
336 # reasoning as verify_raildialogs, since those waits stretch when the
337 # box is busy and a killed verifier does not say it was killed.
338 verify_syncviews) echo 480 ;;
339 # Eight reloads, each waited out past the push debounce so that a push
340 # which is coming has come -- and a check that hurried one of them would
341 # report a push as absent when it was merely late, which is the exact
342 # false pass this file is written to avoid. Two of those reloads carry a
343 # second real device. MEASURED at 470s on a quiet box (2026-08-28), so
344 # 900 on the same reasoning as the row above it.
345 verify_reloadpush) echo 900 ;;
346 # Sixteen dialogs at four skin/theme/width cells, and it was killed by the
347 # 180s default on the 2026-08-11 gate -- the same accident as the two above,
348 # read as an unexplained exit 124. MEASURED at 250s on a quiet box
349 # (2026-08-12): 11s to seed the rail, then 41s, 59s, 58s and 82s for the
350 # four cells. Most of that is not work but PROOF OF ABSENCE: 19 of the 64
351 # dialog attempts do not open at their width -- Fold with no active chat,
352 # the tile cogs on a phone -- and each costs a 10s selector timeout to
353 # establish, the phone cell alone spending 8 of them. So 480: not quite
354 # twice the measurement, on the same reasoning as verify_reversible, and
355 # those 10s waits are exactly what stretches when the box is busy.
356 verify_raildialogs) echo 480 ;;
357 # Eleven palettes, each opened, focused through and measured for ink. Killed
358 # by the 180s default and reported as exit 124 -- it dies part way through
359 # the fifth palette, which reads on a summary as the app's focus ring being
360 # broken. MEASURED three times to completion in a world: 326s and ~330s
361 # (2026-08-13, the second ALL PASS) and 379s (2026-08-14, world 18, which
362 # found one real ink shortfall in the dark palettes). So 600, near twice the
363 # measurement, on the same reasoning as verify_reversible and
364 # verify_raildialogs -- and the spread across those three is why not 400.
365 verify_focus_and_ink) echo 600 ;;
366 # Two devices, an account look pushed each way, and a THIRD sign-in on the
367 # second device to prove the migration case. It has no measurement to be
368 # sized by -- it has never once been run to completion, because it was left
369 # out of the gateway group above and so has spent every gate failing on
370 # fetches that never connected, then being killed at 180s. Sized by its own
371 # declared waits instead, which are what it spends when the answer does not
372 # come: four 25s pushes, a 30s and two 20s settles, four 20s readiness waits
373 # and two browser launches -- a little over 300s if every one of them runs
374 # long. 600 is twice that, and the first run under a real gateway is the
375 # measurement this should be replaced by.
376 verify_look) echo 600 ;;
377 # Parks a request at the gateway for three quarters of a minute, twice, and
378 # the whole point is that the second park is HELD OPEN. Its own waits come to
379 # about 170s before two browser launches, so the 180s default would kill it
380 # the first time it ever reaches a gateway (see verify_look above: it has
381 # never run either). Unmeasured, for the same reason.
382 verify_wakerearm) echo 420 ;;
383 # Twelve turns of a mock provider, three chats, two fan-outs and a reload,
384 # then the same audit again. MEASURED at 41s on a quiet box (2026-08-28),
385 # which the 180s default covers comfortably -- and the default is not what
386 # it would be killed by. Every turn carries a 40s timeout, so a mock that
387 # has gone slow turns 41s into eight minutes without anything being wrong
388 # with the app, and `timeout` cuts the browser out from under it and reports
389 # the result as six ordinary-looking failures (see verify_sync above). 480
390 # is the sum of those timeouts, which is what the file can honestly cost.
391 verify_sweep_used) echo 480 ;;
392 *) echo 180 ;;
393 esac
394}
395
396# Can this verifier report a failure AT ALL?
397#
398# The gate decided PASS on the exit code and nothing else, and six verifiers had no
399# way of setting one: verify_backup, verify_writeguard, verify_viewer,
400# verify_localpage, verify_normalwrite and verify_toolmemory each printed
401# `SOMETHING: false` and exited 0. The summary quotes the LAST line of output, and
402# for three of them that line was `errors: [...]`, so the printed red never even
403# reached the summary. A PRINTED RED WAS A GATE GREEN, for months, over the
404# stale-write guard and over whether a backup restores a user's files.
405#
406# Asked of the SOURCE, because it is the only question with a certain answer: a file
407# containing no `process.exit`, no `throw` and no assertion cannot fail whatever it
408# prints, and no amount of reading its output will tell you that. It is a floor and
409# not a ceiling -- a file that counts reds and then exits 0 anyway still gets past
410# this -- but it is exact in the direction that matters: nothing healthy is ever
411# flagged, because a healthy verifier has to be able to say no somehow.
412#
413# The verifier is still RUN and its output still kept: the reason for the red is that
414# it asserts nothing, and that reason is worth reading beside whatever it printed.
415can_fail() { # name -> 0 if it can express a failure
416 grep -qE 'process\.exit|process\.exitCode|throw new |\bassert[.(]' "dev/$1.mjs" 2>/dev/null
417}
418
419pass=0; fail=0; skip=0; flight=0; failed=""; skipped=""; inflight=""
420
421# ── Verifiers whose failures are KNOWN, ASSIGNED and IN FLIGHT ──────────
422#
423# One entry per line: the verifier, HOW MANY of its checks are expected to fail,
424# and who owns the fixes. Nothing else may be in here, and nothing goes in without
425# a name attached to the work.
426#
427# THE COUNT IS WHAT RETIRES THE ENTRY, and it is the only reason this list is
428# allowed to exist at all. This file's own header argues against hand-kept lists
429# because they go stale silently -- and every mechanism above it therefore asks the
430# verifier instead. This one cannot: whether a red is known work in flight or a
431# regression is a fact about a decision somebody made this week, not a property a
432# verifier can declare about itself, and a marker living inside the verifier would
433# be one an author could quietly widen to keep their own suite green.
434#
435# So the staleness is closed from the other end. A declared verifier whose failing
436# count does not EXACTLY match is a hard FAIL, in both directions:
437#
438# MORE failing than declared -> a new defect has landed behind the known ones,
439# and the entry must not absorb it;
440# FEWER failing than declared -> fixes have landed, so the entry is now hiding
441# nothing and must shrink or go.
442#
443# That makes an entry self-retiring: the gate goes red the moment the work it
444# describes is done, which is the correct pressure and the opposite of the usual
445# failure mode. An entry can only stay quiet while it is telling the exact truth.
446#
447# It is NOT a pass. It is counted apart, printed apart, and named in the summary,
448# because a suite in which a known red and a green look the same has given up the
449# only thing it was for.
450IN_FLIGHT="
451"
452
453# How many failures are declared in flight for a verifier, or '' if none are.
454declared_flight() { # name -> prints the count, or nothing
455 echo "$IN_FLIGHT" | awk -v n="$1" '$1 == n { print $2; exit }'
456}
457
458# The failing count a verifier reported about ITSELF, from its own summary line.
459#
460# Read from the verifier's output rather than from its exit code, because an exit
461# code is one bit and the whole point here is the number. A verifier that prints no
462# such line cannot be declared in flight, and `run_one` fails it rather than
463# guessing -- an unparseable declaration is an unchecked one.
464reported_failures() { # log file -> prints the count, or nothing
465 grep -oE '[0-9]+ failed' "$1" | tail -1 | grep -oE '^[0-9]+'
466}
467: > "$LOG"
468# Truncated ONCE here, appended to thereafter: phase 2 stops and restarts the
469# gateway to take the store lock for a grant, and what the first process said on
470# its way out is exactly the part a `>` on each start would erase. Named
471# SUITE_GW_LOG in dev/gwbin.mjs, which is where the verifier that starts no
472# gateway of its own goes looking for it.
473: > "$SCRATCH/suite-gw.log"
474say() { echo "$1" | tee -a "$LOG"; }
475
476run_one() {
477 local name=$1 out code tail why extra
478 # A verifier that cannot be given what it needs is not run at all. It is NOT
479 # run and then forgiven: `verify_droots_real` exits 2 on purpose in that
480 # state, and a suite that turned an exit 2 into a pass would be the same
481 # defect it exists to prevent.
482 why=$(needs_input "$name")
483 if [ -n "$why" ]; then skip_one "$name" "$why"; return; fi
484 extra=$(args_for "$name")
485 [ "$name" = "verify_durability" ] && rm -rf "$SCRATCH/durability-profile"
486 case " $HEADED " in
487 *" $name "*)
488 if command -v xvfb-run >/dev/null 2>&1; then
489 out=$(timeout "$(slow_for "$name")" xvfb-run -a -s "-screen 0 1400x900x24" \
490 node "dev/$name.mjs" $extra 2>&1)
491 code=$?
492 else
493 skip_one "$name" "needs a headed browser and xvfb-run is not installed"
494 return
495 fi ;;
496 *)
497 out=$(timeout "$(slow_for "$name")" node "dev/$name.mjs" $extra 2>&1)
498 code=$? ;;
499 esac
500 # Keep the WHOLE output, not just the line the summary quotes.
501 #
502 # Only the last line survived here, so diagnosing any red meant running the
503 # verifier again by hand -- and for the ones that need a gateway, a grant and
504 # mail fixtures, that means reproducing the provisioning this script already
505 # did. Every red chased in this session cost a second full run for want of a
506 # file that had already been captured and thrown away.
507 mkdir -p "$SCRATCH/out"
508 printf '%s\n' "$out" > "$SCRATCH/out/$name.log"
509 tail=$(echo "$out" | grep -vE "Skipping host" | tail -1)
510 # Two spellings, because both are in the tree: `SKIPPED: <why>` and
511 # `SKIP <name> — <why>`. Only the first was recognised, so verify_gwretry,
512 # verify_sessionrenew and verify_chunkgw -- each of which printed the second
513 # and then exited 0 -- were counted as PASSES for runs in which they had
514 # refused to do anything at all. verify_chunkgw can no longer skip: its
515 # "no binary built" branch was unreachable once gwbin.mjs began refusing
516 # that case outright, and it has gone.
517 if echo "$out" | grep -qE '^SKIPPED:|^SKIP '; then
518 skip=$((skip+1)); skipped="$skipped $name"
519 say "SKIP $name — $(echo "$out" | grep -E '^SKIPPED:|^SKIP ' | head -1)"
520 elif [ $code -eq 0 ] && ! can_fail "$name"; then
521 # Green, and worth nothing: see `can_fail`. Counted as a failure because a
522 # file that cannot say no is not evidence of anything, and a suite that
523 # reports it as a pass is making a claim on its behalf that it never made.
524 fail=$((fail+1)); failed="$failed $name"
525 say "FAIL $name (asserts nothing) — no process.exit, no throw, no assertion: it exits 0"
526 say " whatever it prints, so its green line means only that it ran. Last line was: $tail"
527 say " full output: $SCRATCH/out/$name.log"
528 elif [ $code -eq 0 ]; then
529 # A verifier declared in flight that has gone GREEN is an entry to delete,
530 # and it is said here rather than at the end: the work is done and the
531 # declaration is now a lie about the tree in the quiet direction.
532 local want_ok; want_ok=$(declared_flight "$name")
533 if [ -n "$want_ok" ]; then
534 fail=$((fail+1)); failed="$failed $name"
535 say "FAIL $name — IN_FLIGHT declares $want_ok failing and it now passes CLEAN."
536 say " The fixes have landed. Delete its line from IN_FLIGHT in dev/run_all.sh;"
537 say " until then this red is the entry, not the verifier."
538 else
539 pass=$((pass+1)); say "PASS $name — $tail"
540 fi
541 else
542 local want; want=$(declared_flight "$name")
543 local got; got=$(reported_failures "$SCRATCH/out/$name.log")
544 if [ -z "$want" ]; then
545 fail=$((fail+1)); failed="$failed $name"
546 say "FAIL $name (exit $code) — $tail"
547 say " full output: $SCRATCH/out/$name.log"
548 elif [ -z "$got" ]; then
549 # Declared, and the declaration cannot be checked. That is worse than an
550 # undeclared red, because it is a red somebody has arranged to be quiet
551 # about on the strength of a number nobody can read.
552 fail=$((fail+1)); failed="$failed $name"
553 say "FAIL $name (exit $code) — declared in flight with $want failing, but it printed"
554 say " no '<n> failed' line, so the declaration cannot be checked. A red that is"
555 say " excused by an unreadable number is not excused."
556 say " full output: $SCRATCH/out/$name.log"
557 elif [ "$got" != "$want" ]; then
558 fail=$((fail+1)); failed="$failed $name"
559 if [ "$got" -gt "$want" ]; then
560 say "FAIL $name (exit $code) — $got failing, but only $want are declared in flight."
561 say " $((got - want)) more than the known work. A new defect has landed behind it,"
562 say " and the IN_FLIGHT entry must not absorb it."
563 else
564 say "FAIL $name (exit $code) — $got failing, and $want are declared in flight."
565 say " $((want - got)) of them are fixed. Shrink the IN_FLIGHT entry in"
566 say " dev/run_all.sh to $got, or delete it."
567 fi
568 say " full output: $SCRATCH/out/$name.log"
569 else
570 flight=$((flight+1)); inflight="$inflight $name"
571 say "FLIGHT $name — $got failing, exactly the $want declared in flight. NOT A PASS."
572 say " $(echo "$IN_FLIGHT" | awk -v n="$name" '$1 == n { $1=""; $2=""; sub(/^ /, ""); print }')"
573 say " full output: $SCRATCH/out/$name.log"
574 fi
575 fi
576}
577
578skip_one() { # name, why
579 skip=$((skip+1)); skipped="$skipped $1"
580 say "SKIP $1 — $2"
581}
582
583gateway_up() { ss -ltn 2>/dev/null | grep -q ":$GW_PORT "; }
584wait_gateway() { # tries
585 local i=0
586 while [ $i -lt "${1:-20}" ]; do
587 curl -sf -m 2 "http://127.0.0.1:$GW_PORT/api/health" >/dev/null 2>&1 && return 0
588 i=$((i+1)); sleep 1
589 done
590 return 1
591}
592# Where the gateway is run FROM decides which app.jdat it reads. `gateway/` is
593# the shipped config; `dev/devgw/` is a generated copy carrying `dev_insecure`
594# on the mail routes so the local IMAP/SMTP fixtures can be reached at all.
595GW_CWD=gateway
596# A GATEWAY ALREADY ON THE PORT IS NOT THIS RUN'S, AND `wait_gateway` CANNOT TELL.
597#
598# This used to spawn regardless and then ask the PORT whether a gateway was up.
599# Any gateway answers that -- another lane's, or the one this run had just failed
600# to stop -- so `start_gateway` reported success while the process it started was
601# dying on a bind it could never win, and everything after it addressed a stranger.
602#
603# What that cost, on 2026-08-25 at `25d9e51`: `verify_mailfolders` and
604# `verify_compose` went red on `the server is asked what folders it has -- 0:`,
605# with seven folders on the fixture's own wire and `entitled accounts ready: yes`
606# above them. The gateway's log named the hop -- `mail_folders` failed at
607# `handlers/mail.rs:405`, the Pro check, under a chain ending
608# `csum.rs:139 [Checksum] Mismatch detected`. The store had been written by two
609# processes at once: the previous gateway had not gone in the fifteen seconds
610# `stop_gateway` allows (a 3.3 GB store takes longer), four `daimond_ctl` calls
611# then wrote entitlements underneath it, and a second gateway opened the same
612# files while the first was still appending. The licence record read back with
613# somebody else's bytes at the offset the index remembered, so every route that
614# reads Pro -- `mail_folders`, `mail_accounts`, `mail_sync`, `sync`, `licence` --
615# answered a 500, and the client showed a mailbox with no folders in it.
616#
617# So the port is checked BEFORE spawning, and the pid this run started is checked
618# AFTER: "something is answering" was never the question.
619start_gateway() {
620 [ -x "$GW_BIN" ] || return 1
621 if gateway_up; then
622 say " :$GW_PORT is already answering and this run did not start it."
623 say " Refusing to put a second gateway over the same store -- that is what"
624 say " corrupted one on 2026-08-25. Find it with \`ss -ltnp | grep :$GW_PORT\`,"
625 say " or give this run a port of its own with DAIMOND_GW_PORT."
626 return 1
627 fi
628 # The pid is written down because the only safe way to stop a process is to stop
629 # the one you started. See `stop_gateway`.
630 ( cd "$GW_CWD" && APP_MODE=sandbox nohup "$ROOT/$GW_BIN" >>"$SCRATCH/suite-gw.log" 2>&1 &
631 echo $! > "$SCRATCH/suite-gw.pid" )
632 wait_gateway 25 || return 1
633 # AND THE PID THAT SERVES IS NOT THE PID THAT WAS SPAWNED, which is the whole
634 # fault and took a day to see because everything about it reads right.
635 #
636 # `$!` is what bash forked. Measured on 2026-08-25 with this exact construct:
637 # recorded pid 740408, port held by 740409. `kill 740408` returned in 258 ms
638 # and the gateway was STILL BOUND AND STILL SERVING SIXTY SECONDS LATER --
639 # `stop_gateway` allows fifteen, so it reported failure while the process it
640 # meant to stop went on writing the store. Then `daimond_ctl` wrote
641 # entitlements underneath it and a second gateway opened the same files, and
642 # the licence record came back with the wrong bytes at the offset the index
643 # held: `csum.rs:139 [Checksum] Mismatch detected`, nineteen times in one
644 # gate, and `verify_mailfolders` red on seven folders it could not see.
645 #
646 # So the port is asked who is on it. That is only safe because the guard above
647 # has already refused a port that was not free: whoever holds it now can only
648 # be this run's. Both pids are kept, newest first, and `stop_gateway` stops
649 # every one it finds alive -- a wrapper that is already gone costs nothing.
650 local held; held=$(ss -ltnp 2>/dev/null | grep ":$GW_PORT " \
651 | grep -oE 'pid=[0-9]+' | head -1 | cut -d= -f2)
652 local spawned; spawned=$(cat "$SCRATCH/suite-gw.pid" 2>/dev/null | head -1)
653 { [ -n "$held" ] && echo "$held"; [ -n "$spawned" ] && [ "$held" != "$spawned" ] \
654 && echo "$spawned"; } > "$SCRATCH/suite-gw.pid.new"
655 mv "$SCRATCH/suite-gw.pid.new" "$SCRATCH/suite-gw.pid"
656 [ -s "$SCRATCH/suite-gw.pid" ]
657}
658# WHAT THIS USED TO BE, AND WHY IT WAS THE WORST LINE IN THE SUITE. It was
659# `pkill -f "$(basename "$GW_BIN")"`, which signals every process on the machine
660# whose command line contains the string `daimond_gateway`. That is not the
661# suite's gateway. It is also, at any moment on a machine running more than one
662# lane:
663#
664# * every other worktree's `cargo test --bin daimond_gateway`,
665# * every other worktree's libtest harness, `…/deps/daimond_gateway-<hash>`,
666# * the `rustc --crate-name daimond_gateway …` of a build in flight,
667# * every other worktree's release gateway,
668# * and the SHELL of anybody whose command line happens to mention the name.
669#
670# Measured on 2026-08-24 with three lanes at work: one `pkill -f` would have
671# signalled nine processes, of which exactly one was this suite's. Two builds in
672# this lane died on `(signal: 15, SIGTERM)` from it while it ran elsewhere.
673# `dev/world.sh` and `dev/attribute.sh` both already say, in as many words, "do NOT
674# pkill by command line: it is not scoped to a world." The lesson had been learnt
675# for `serve.mjs` and never carried across to the gateway.
676#
677# So: the pid this suite started, or nothing. A gateway somebody else started is
678# not ours to kill, and saying so is more use than killing it.
679stop_gateway() {
680 # EVERY pid this run wrote down, because there is more than one: see the note in
681 # `start_gateway` about the process that serves not being the process that was
682 # spawned. A pid that has already gone is skipped, so an ordinary run stops one
683 # process and says nothing about the wrapper that is no longer there.
684 local pid stopped=""
685 while read -r pid; do
686 [ -n "$pid" ] || continue
687 if kill -0 "$pid" 2>/dev/null; then kill "$pid" 2>/dev/null; stopped="$stopped $pid"; fi
688 done < <(cat "$SCRATCH/suite-gw.pid" 2>/dev/null)
689 if [ -n "$stopped" ]; then
690 :
691 elif gateway_up; then
692 say " :$GW_PORT is held by a gateway this suite did not start. It is being left"
693 say " alone: find its pid with \`ss -ltnp | grep :$GW_PORT\` and stop that one."
694 return 1
695 fi
696 rm -f "$SCRATCH/suite-gw.pid"
697 local i=0
698 while gateway_up && [ $i -lt 15 ]; do sleep 1; i=$((i+1)); done
699 ! gateway_up
700}
701
702# A check may want these functions and none of the run. `dev/breakproof_stopgateway.sh`
703# and `dev/breakproof_startgateway.sh` source THIS file -- not a copy of it -- to certify
704# that `stop_gateway` stops the pid it started and leaves every other process on the machine
705# alone, and that `start_gateway` refuses a port it did not put a gateway on. Sourcing the
706# real file is
707# the whole point: a copy would drift from what actually runs, and the fault being guarded
708# against is precisely one that looked harmless for months.
709if [ "${RUN_ALL_FUNCTIONS_ONLY:-}" = 1 ]; then return 0; fi
710
711# ── Which verifiers to run, and in which phase ──────────────────────────
712# `refluxduo` IS IN A DEFAULT RUN, and it is not a `verify_*.mjs`, so it is named.
713#
714# THE RELEASE GATE FOR SOCIAL, and the only thing that defends it. `--wire --journey tools`
715# drives the journey with no provider, no key and no spend, which is why it can sit in a suite
716# at all. Without it the Social tools can be unwired by any later change and nothing goes red:
717# the next daimon simply reports that the feature does not exist, which is exactly how this
718# whole thread began. It stands up its OWN gateway and forge, so it belongs in phase 1 with
719# the other verifiers that need :9002 clear.
720if [ $# -gt 0 ]; then
721 ALL="$*"
722else
723 ALL=$(cd dev && ls verify_*.mjs | sed 's/\.mjs$//')
724 ALL="$ALL refluxduo"
725fi
726PHASE1=""; PHASE2=""
727for name in $ALL; do
728 if needs_live_gateway "$name"; then PHASE2="$PHASE2 $name"; else PHASE1="$PHASE1 $name"; fi
729done
730
731# ── The gateway binary, built once for the whole run ────────────────────
732#
733# `dev/gwbin.mjs` refuses to measure a gateway older than the code under test,
734# and it is right to: on 2026-08-10 a three-day-old binary answered "Unknown
735# admin view 'secrets'" and that read as a console defect rather than a stale
736# build. But an mtime is a coarse authority. Another agent touching any of the
737# twelve crates the gateway path-depends on under rust/fe2o3 voids the run, and
738# that happened mid-gate the same day. Cargo knows precisely whether a rebuild
739# is needed -- including whether a new module is even referenced -- so ask it,
740# once, and leave the mtime guard as what it should be: a cheap net for someone
741# running a single verifier by hand.
742#
743# Before PHASE 1, not merely before phase 2. Nine of the ten verifiers that
744# spawn a gateway are in phase 1; only verify_passkey_blob is in phase 2, and
745# start_gateway needs $GW_BIN there in any case. ONCE, so that every verifier
746# in the run measures the same artefact.
747#
748# CARGO_TARGET_DIR is unset on purpose. Agents point it at their own slot
749# directory, which is what leaves gateway/target/release/ behind in the first
750# place, and that path is the one the verifiers spawn.
751wants_gateway() { # does anything in this run touch the binary?
752 [ -n "$PHASE2" ] && return 0
753 # Asked of the verifiers themselves rather than of a list kept here: a list
754 # of names is the thing that goes stale, as HEADED and NEEDS_GATEWAY both
755 # have, and each time it did the suite drew a wrong conclusion quietly.
756 local n
757 for n in $ALL; do
758 grep -q 'gwbin\.mjs\|daimond_gateway' "dev/$n.mjs" 2>/dev/null && return 0
759 done
760 return 1
761}
762if wants_gateway; then
763 say "── Building the gateway, so every verifier measures one artefact ──"
764 if ( cd gateway && env -u CARGO_TARGET_DIR cargo build --release ) >>"$LOG" 2>&1; then
765 [ -f "$GW_BIN" ] && say " $GW_BIN ($(date -r "$GW_BIN" '+%Y-%m-%d %H:%M:%S'))"
766 else
767 say "FATAL the gateway did not build, so nothing below could be measured against"
768 say " the current source: every verifier that spawns one would either run a"
769 say " stale binary or refuse outright. The cargo output is at the end of $LOG"
770 exit 2
771 fi
772fi
773
774# ── Phase 0: the static checks, which need no browser and take a second ─────
775#
776# Neither of these had ever run under this suite, and on 2026-08-11 the second one
777# would have caught 39 English fallbacks left behind by a catalogue rewrite -- text
778# that ships in all eight locales whenever the catalogue fails to load, and that
779# nothing else looks at. They cost about a second between them, so they run first:
780# a red here explains reds later, and a suite that finds it after two hours of
781# browsers has learnt the same thing far too late.
782#
783# A named subset is honoured, so `run_all.sh verify_tags` still means just that.
784static_one() { # name, command…
785 local name="$1"; shift
786 local out code
787 out=$("$@" 2>&1); code=$?
788 mkdir -p "$SCRATCH/out"
789 printf '%s\n' "$out" > "$SCRATCH/out/$name.log"
790 if [ $code -eq 0 ]; then
791 pass=$((pass+1)); say "PASS $name — $(echo "$out" | tail -1)"
792 else
793 fail=$((fail+1)); failed="$failed $name"
794 say "FAIL $name (exit $code) — $(echo "$out" | tail -1)"
795 say " full output: $SCRATCH/out/$name.log"
796 fi
797}
798# `--frozen`, because a suite ASSERTS and does not write. Without it `i18ncheck`
799# rewrites `dev/results/i18n-coverage.json` as it runs, and under `dev/gate.sh`
800# it would rewrite the copy in the gate's own worktree -- a file nobody reads,
801# leaving the committed map exactly as stale as it was while the run went green.
802# The map is what tells a runtime reporter a by-design hole from a real one, so a
803# stale one retires findings. Frozen, it fails and says which half moved.
804#
805# THE COST, so nobody meets it as a surprise: any edit that adds or removes a
806# `t()` call site moves a count in the map, and the gate then goes red until
807# `node dev/i18ncheck.mjs --write-map` is run in the main tree and the map committed. That is
808# one command, it is named in the failure, and it is the price of the map being
809# true rather than merely present.
810# ── WHAT THIS RUN RAN UNDER, before anything it says about the app ──────
811#
812# Two numbers, because both have already made one commit answer two different
813# things and neither output said which it was under.
814#
815# THE PORTS. The gateway, the IMAP fixture and the submission stand-in were
816# fixed for the whole machine until 2026-08-25, so a suite could be reading
817# another lane's gateway and its log would look identical either way.
818#
819# THE DESCRIPTOR CEILING. `systemd-run --user --unit=…` gives a service a NOFILE
820# soft limit of 1024 where an interactive shell has 524288, and the gateway
821# suite holds a descriptor per store it opens. The same commit answered
822# "634 passed, 0 failed" under a shell and "630 passed, 4 failed" under a unit,
823# and nothing in either run named the ceiling. `-p LimitNOFILE=524288` is the
824# fix at the launch; this is the line that lets a reader tell afterwards.
825say "run: gateway :$GW_PORT mail :$IMAP_PORT/:$SMTP_PORT app :${DAIMOND_PORT:-8777}"
826say " descriptor ceiling $(ulimit -Sn) soft / $(ulimit -Hn) hard$([ "$(ulimit -Sn)" -lt 65536 ] && echo " — LOW: a suite that leaves databases open fails on it")"
827say ""
828
829if [ $# -eq 0 ]; then
830 say "── Phase 0 (static, no browser)"
831 static_one i18ncheck node dev/i18ncheck.mjs --frozen
832 static_one i18nfallback node dev/i18nfallback.mjs --quiet
833 # `dev/jscheck.sh` had never run in a gate -- it is a `.sh`, and the work list two
834 # blocks down is `ls verify_*.mjs`, so nothing could see it and phase 0's list is
835 # hand-written and did not name it. Confirmed against forty `suite.log` files: not
836 # one mentions it. It exists because `node --check` EXITS 0 on a `.js` file holding
837 # a syntax error, which is proved in its own header, so until now nothing in any
838 # gate parsed the browser JavaScript at all.
839 #
840 # UNDER TWO SECONDS FOR 93 FILES, and it was 61 until 2026-08-25: its own list was
841 # `ls www/js/*.js`, which could not see the eight locale tables, the service worker,
842 # the operator console, the guide, or either browser extension. The one gate against
843 # false greens was giving a confident number about two thirds of the tree. It asks
844 # git now, so a directory added later is in the list without anybody widening a glob.
845 static_one jscheck bash dev/jscheck.sh
846 # Four assertions, no browser, no port, a fraction of a second: that a terminal
847 # request composed in Rust reaches the wire whole, and that a fence root outside the
848 # grant travels with the toolkit it belongs to. Written on 2026-08-24 for the day the
849 # owner could not open a terminal at all, and never wired into anything -- it is not a
850 # `verify_*.mjs` and phase 0's list is hand-written, which is the same accident that
851 # hid jscheck. Found by `dev/verify_checkreach.mjs`, which is what that file is for.
852 static_one ptyfields node dev/prove_ptyfields.mjs
853 # Five seconds, and it guards the one line in this file that could reach off the
854 # machine and into another lane's work. It signals nothing but its own stand-ins.
855 static_one stopgateway bash dev/breakproof_stopgateway.sh
856 # Its other half. `stop_gateway` guards what this run kills; this guards what it
857 # reports as started, which is how a suite came to drive a stranger's gateway
858 # over a store its own processes were writing.
859 static_one startgateway bash dev/breakproof_startgateway.sh
860fi
861
862# ── Phase 0b: the Rust tests, counted ───────────────────────────────────────
863#
864# THIS IS THE FIRST RUST COVERAGE THE GATE HAS EVER HAD, and about ten minutes is
865# the price of having any. Read that before trimming it.
866#
867# The suite never ran a single Rust test. Not one, in any release this project
868# has made. It built the gateway and stopped there, and the work list below is
869# `ls verify_*.mjs`, so what a gate measured was browser verifiers exclusively --
870# the "269 passed of 277" that seq 150 shipped on was browser verifiers and
871# nothing else. Every Rust number anybody has quoted came from a run somebody
872# did by hand in their own worktree, and nothing checked that the run finished.
873# A Rust regression could have shipped in any release ever made and nothing would
874# have said a word.
875#
876# `dev/testcount.mjs` is what makes that checkable: it asks each harness `--list`
877# for the number of tests compiled into it, runs the suite, and refuses to call it
878# a pass unless the number executed matches. A `cargo test` alone cannot do that
879# -- it exits 0 on a filtered run (`cargo test -- sweeper::` is 12 of 645 and a
880# cheerful "ok"), and it stops at the first failing harness, so the gateway's
881# integration tests never run at all when its unit tests are red and their 3 are
882# silently not in anybody's total. `--no-fail-fast`, and then every harness is
883# counted against what it was compiled with.
884#
885# THE COST, measured on 2026-08-24 so nobody has to take it again: about half a
886# minute for the library's 737, and the rest for the gateway's 645 and its three
887# integration tests. Two things moved it and they are not the same thing --
888# `Store::open_temp` now closes its database, which took the gateway harness from
889# 312 s to 565 s and is what stopped the suite growing without bound; and the
890# `daimond_ctl` test harness is gone, which gave back the 89 s it took before that
891# change and rather more after it, for nine tests that moved into the gateway
892# harness rather than being dropped. Only on a whole run; a named subset skips
893# it, since a subset is not a total anyway.
894if [ $# -eq 0 ]; then
895 say "── Phase 0b (Rust, counted)"
896 # THE WASM ARMS FIRST, because a green test run is not evidence about them.
897 # Every tool's DECISION is a pure native function so it can be tested; every
898 # tool's ACTION is `#[cfg(target_arch = "wasm32")]` because it touches the
899 # browser. So the native test build never compiles the acting half, and on
900 # 2026-08-25 a `Tool::runs` arm was missing entirely while 790 of 790 tests
901 # ran and passed -- both numbers honest, both about a build that did not
902 # contain the code. `testcount.mjs` closes "a test was displaced"; nothing
903 # closed "an arm was never written". About a minute warm.
904 static_one wasm_arms cargo check --target wasm32-unknown-unknown --lib
905 static_one rust_lib node dev/testcount.mjs .
906 static_one rust_gateway node dev/testcount.mjs gateway
907 # AND THE HAND, added 2026-08-25, because this block was itself the blocker it
908 # closed. `.` and `gateway` were a hand-kept list of two directories, and this
909 # tree holds THREE `Cargo.toml`s: `hand/` is its own workspace, so it is not a
910 # member of the root one and `testcount.mjs .` never reaches it. The fence, the
911 # seccomp filter, the journal and the codec -- the most security-critical
912 # component there is -- and not one of their tests had ever run in a gate. The
913 # fix for "the gate has never run a single Rust test" had the same shape as the
914 # fault, one directory along, and nothing said so.
915 #
916 # `verify_handreal` does not cover it: it builds the hand and then runs a `cargo
917 # test` in a FIXTURE project through the daimon, which proves the hand can run
918 # cargo and proves nothing about the hand's own tests.
919 #
920 # RUN FOR THE FIRST TIME 2026-08-25, and the two lines below are what that run
921 # cost. It is 268 tests and not the 194 the entry above used to claim: that
922 # number came from grepping `#[test]`, and `#[tokio::test]` is a different
923 # spelling of the same thing. 242 in the library harness, 26 in the binary's.
924 #
925 # ELEVEN OF THEM FAILED, every one for a single reason and none of it about the
926 # fence. The launcher tests exec the SHIPPING binary, and `cargo test` never
927 # builds it -- this crate has no integration test, so cargo has no reason to --
928 # so `shipping_hand` in hand/src/exec.rs refuses rather than skipping, on the
929 # rule that a fence test which cannot say which code it measured must not report
930 # success. `verify_handreal` does not supply it either: that builds `--release`
931 # with `CARGO_TARGET_DIR` deleted from the environment on purpose, and these
932 # tests look for a DEBUG binary beside themselves, in whatever target directory
933 # they were compiled into. So the build belongs HERE, in this environment, where
934 # it lands where the tests will look for it -- and it is a check in its own
935 # right, because a hand that does not build is a finding on its own.
936 #
937 # With it: 268 compiled, 268 executed, 268 passed. Measured from an EMPTIED
938 # target directory, on a 16-core machine with another lane building beside it:
939 # 25 s for the cold build and the tests together, and 12 s warm, of which the
940 # build is 4. The dearest part is fe2o3, which the hand takes four crates of.
941 # 1.1 GB of artefacts, which is why `hand/` needs a target directory of its own
942 # rather than sharing one with the root workspace.
943 static_one handbin cargo build --manifest-path hand/Cargo.toml
944 static_one rust_hand node dev/testcount.mjs hand
945fi
946
947# ── Phase 1: the gateway port clear ─────────────────────────────────────
948if [ -n "$PHASE1" ]; then
949 if gateway_up; then
950 say "Stopping the gateway on :$GW_PORT — phase 1 needs it clear."
951 stop_gateway || { say "Could not free :$GW_PORT; stop it by hand and re-run."; exit 2; }
952 fi
953 say "── Phase 1 (no gateway):$PHASE1"
954 for name in $PHASE1; do run_one "$name"; done
955fi
956
957# ── Phase 2: a gateway, and what the entitled tests need ────────────────
958if [ -n "$PHASE2" ]; then
959 say ""
960 # compose and mailfolders talk to loopback mail fixtures, which the shipped config refuses.
961 # Run the gateway from the generated dev CWD for the whole of phase 2: it is
962 # the same binary over the same store, one flag different.
963 # A run on a port of its own needs the generated CWD too, whatever it is running:
964 # `gateway/app.jdat` is the deployed config and holds :9002, and moving a port in
965 # it is the temporary edit `devgw.sh`'s own header refuses to make.
966 NEED_DEVGW=no
967 case " $PHASE2 " in *" verify_compose "*|*" verify_mailfolders "*) NEED_DEVGW=yes ;; esac
968 [ "$GW_PORT" = 9002 ] || NEED_DEVGW=yes
969 case " $NEED_DEVGW " in *" yes "*)
970 # Said rather than swallowed: without the generated CWD the mail routes
971 # refuse loopback, and compose then fails for a reason this script chose.
972 if bash dev/devgw.sh >>"$SCRATCH/suite-devgw.log" 2>&1; then
973 GW_CWD=dev/devgw
974 else
975 say " dev/devgw.sh failed — running from $GW_CWD, whose config refuses the"
976 say " loopback mail fixtures: $SCRATCH/suite-devgw.log"
977 fi ;;
978 esac
979 if ! start_gateway; then
980 for name in $PHASE2; do
981 # Not "build it" any more: the build happened above, so a gateway that
982 # will not start has a reason, and the reason is in its own log.
983 skip_one "$name" "the gateway would not start on :$GW_PORT — $SCRATCH/suite-gw.log"
984 done
985 else
986 say "── Phase 2 (gateway up on :$GW_PORT):$PHASE2"
987
988 # WHICH IDENTITIES THIS RUN HAS TO PROVISION -- a set, not one name.
989 # `verify_compose` and `verify_mailfolders` share `compose`; `verify_sync`
990 # drives `sync`. Deduplicated, so two verifiers on one identity cost one
991 # grant, which is what the single hard-wired pair used to give for free.
992 WANT_GRANT=no; GRANTED=no; IDENTS=""
993 for name in $PHASE2; do
994 case " $NEEDS_GRANT " in *" $name "*) ;; *) continue ;; esac
995 WANT_GRANT=yes
996 id=$(ident_for "$name")
997 [ -n "$id" ] || { say " $name is in NEEDS_GRANT and ident_for names no identity for it"; continue; }
998 case " $IDENTS " in *" $id "*) ;; *) IDENTS="$IDENTS $id" ;; esac
999 done
1000 if [ "$WANT_GRANT" = yes ] && [ -x "$CTL_BIN" ]; then
1001 # All of this used to go to /dev/null, exit codes included. A grant
1002 # that failed silently is worse than no grant at all: GRANTED stayed
1003 # yes, the three entitled verifiers ran without the entitlement, and
1004 # went red for a reason this script already knew and had discarded.
1005 PROV_LOG=$SCRATCH/suite-provision.log
1006 : > "$PROV_LOG"
1007 # Every identity's account id is read FIRST, with the gateway up, because
1008 # `/api/account` is what turns an identity into one and it needs a gateway
1009 # to ask. The grants come after, together, behind a single restart.
1010 ACCTS=""; MISSING=""
1011 for id in $IDENTS; do
1012 a=$(node dev/provision.mjs "$SCRATCH/$id-profile" "$id" 2>>"$PROV_LOG" | tail -1)
1013 if [ -n "$a" ]; then ACCTS="$ACCTS $id:$a"; else MISSING="$MISSING $id"; fi
1014 done
1015 if [ -n "$MISSING" ]; then
1016 say " could not read an account id for:$MISSING — the tests that drive them will skip"
1017 say " what went wrong: $PROV_LOG"
1018 fi
1019 if [ -n "$ACCTS" ]; then
1020 # The gateway stands down for the grants, and NOT because of a
1021 # lock: there is no cross-process locking in o3db. One was added
1022 # to fe2o3 on 2026-08-16 and reverted three hours later, the
1023 # diagnosis behind it having been wrong -- data files are opened
1024 # for append and a live file number is now claimed with
1025 # `create_new`, so two processes writing one store is the design
1026 # rather than a hazard.
1027 #
1028 # The real reason is VISIBILITY. o3db holds its key index in
1029 # memory, per process, built when the store is opened, and a
1030 # lookup that misses it answers "not found" without going to disk
1031 # (`bot_cache.rs::read`). So an entitlement `daimond_ctl` appends
1032 # underneath a running gateway is in the files and in nobody's
1033 # index: every `has_entitlement` in the live process would answer
1034 # no, exactly as though the grant had failed, and the entitled
1035 # verifiers would go red for a reason that is not theirs.
1036 # Restarting is what rebuilds the index, so it is how the grant
1037 # reaches the gateway. ONE restart for all of them, which is the
1038 # whole reason the ids are read before any grant is made.
1039 # AND ITS FAILURE IS FATAL TO THE GRANTS, WHICH IT WAS NOT.
1040 #
1041 # `stop_gateway` answers whether the port actually went quiet, and
1042 # phase 1 above has always acted on that answer (`|| exit 2`). Here
1043 # the answer was dropped on the floor, so a gateway that outlived the
1044 # fifteen-second wait was still serving -- and still WRITING -- while
1045 # the four `daimond_ctl` calls below opened the same store to append
1046 # entitlements to it. That is the second writer whose bytes the next
1047 # gateway's index could not verify; see `start_gateway`.
1048 #
1049 # Nothing is granted rather than granted into a store somebody else
1050 # holds. The verifiers that need the entitlement then skip by name,
1051 # which is the outcome this script already prefers to a silent one.
1052 if stop_gateway; then
1053 GRANT_OK=yes
1054 else
1055 GRANT_OK=no
1056 ACCTS=""
1057 say " the gateway would not stand down, so NO grant is being made:"
1058 say " writing entitlements underneath a live gateway is what corrupted"
1059 say " the store on 2026-08-25. The entitled verifiers below will skip."
1060 fi
1061 for pair in $ACCTS; do
1062 id=${pair%%:*}; a=${pair#*:}
1063 # `email` ONLY where the identity is used to read mail. Granting an
1064 # entitlement a verifier does not need would hide the day it starts
1065 # needing one, which is the failure this whole block exists against.
1066 case "$id" in
1067 compose) ( cd "$GW_CWD" && "$ROOT/$CTL_BIN" grant "$a" email ) >>"$PROV_LOG" 2>&1 || GRANT_OK=no ;;
1068 esac
1069 ( cd "$GW_CWD" && "$ROOT/$CTL_BIN" topup "$a" 5000 ) >>"$PROV_LOG" 2>&1 || GRANT_OK=no
1070 done
1071 # The gateway comes back either way -- the rest of phase 2 needs
1072 # it whether or not the grants landed.
1073 if start_gateway && [ "$GRANT_OK" = yes ]; then
1074 # Pro as well: Email, sync and cloud storage are all behind it
1075 # since 2026-07-24, so without it the app raises the "Sync is
1076 # part of Pro" dialog OVER the page mid-run and the clicks that
1077 # follow land on the dialog. It is bought the way a user buys
1078 # it -- a signed checkout event the gateway verifies.
1079 GRANTED=yes
1080 for pair in $ACCTS; do
1081 id=${pair%%:*}; a=${pair#*:}
1082 PROST=$(node dev/pro.mjs "$a" "$ROOT/gateway" 2>>"$PROV_LOG" | tail -1)
1083 say " provisioned $id ($a): credits + Pro webhook ${PROST:-?}"
1084 case "$PROST" in 200) ;; *) GRANTED=no ;; esac
1085 done
1086 fi
1087 say " entitled accounts ready: $GRANTED"
1088 [ "$GRANTED" = yes ] || say " what went wrong: $PROV_LOG"
1089 fi
1090 elif [ "$WANT_GRANT" = yes ]; then
1091 # Said, because the skip below reads "no entitled account (see above)"
1092 # and nothing above said anything at all when this was the reason.
1093 say " $CTL_BIN is not there or not executable, so no grant can be made"
1094 fi
1095
1096 # compose and mailfolders need the mail fixtures: an IMAP server to read,
1097 # and a submission stand-in to catch what is sent.
1098 MAIL=no
1099 case " $PHASE2 " in *" verify_compose "*|*" verify_mailfolders "*)
1100 if [ -x "$IMAP_FIXTURE" ]; then
1101 # Pids kept, for `stop_gateway`'s reason: these are killed by NAME otherwise,
1102 # and another lane's IMAP fixture has the same name as this one's.
1103 # The fixture takes its port as its first argument and the
1104 # stand-in reads SMTPD_PORT; both default to the historical
1105 # numbers, so a hand run in no world is unchanged.
1106 nohup "$IMAP_FIXTURE" "$IMAP_PORT" >"$SCRATCH/suite-imap.log" 2>&1 &
1107 IMAP_PID=$!
1108 nohup node "$SMTPD" >"$SCRATCH/suite-smtpd.log" 2>&1 &
1109 SMTPD_PID=$!
1110 sleep 2
1111 ss -ltn 2>/dev/null | grep -q ":$IMAP_PORT " \
1112 && ss -ltn 2>/dev/null | grep -q ":$SMTP_PORT " && MAIL=yes
1113 fi
1114 say " mail fixtures on :$IMAP_PORT/:$SMTP_PORT: $MAIL"
1115 ;; esac
1116
1117 for name in $PHASE2; do
1118 case " $NEEDS_GRANT " in
1119 *" $name "*) [ "$GRANTED" = yes ] || { skip_one "$name" "no entitled account (see above)"; continue; } ;;
1120 esac
1121 if { [ "$name" = verify_compose ] || [ "$name" = verify_mailfolders ]; } && [ "$MAIL" != yes ]; then
1122 skip_one "$name" "mail fixtures absent (build fe2o3's imap_test_server example)"
1123 continue
1124 fi
1125 run_one "$name"
1126 done
1127
1128 # Only the fixtures THIS run started, for the reason written over `stop_gateway`.
1129 for pid in "${IMAP_PID:-}" "${SMTPD_PID:-}"; do
1130 [ -n "$pid" ] && kill "$pid" >/dev/null 2>&1
1131 done
1132 stop_gateway
1133 fi
1134fi
1135
1136say ""
1137# In flight is named on the SUITE line itself, not tucked underneath it. A reader
1138# who takes in one line has to see that some of this run was red on purpose;
1139# putting the figure only in a detail line below is how "42 passed" comes to stand
1140# for a suite that never went green.
1141if [ $flight -gt 0 ]; then
1142 say "SUITE: $pass passed, $fail failed, $skip skipped, $flight IN FLIGHT (red, known, assigned)."
1143else
1144 say "SUITE: $pass passed, $fail failed, $skip skipped."
1145fi
1146[ -n "$failed" ] && say " failed: $failed"
1147[ -n "$skipped" ] && say " skipped:$skipped"
1148[ -n "$inflight" ] && {
1149 say " in flight:$inflight"
1150 say " These are NOT passes. Each is red by declaration in dev/run_all.sh's IN_FLIGHT,"
1151 say " with the exact count of its failures; the entry fails the suite the moment that"
1152 say " count changes in either direction, so none of them can outlive its defects."
1153}
1154[ $fail -eq 0 ]