oxedyne/fe2o3/fe2o3_austenite/web/pearl-reader/jdat.js
7.8 KiB, 7 runs
created by r1870400018:38498, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | // jdat.js -- a minimal browser parser for the text-jdat subset a Pearl (.prl) document uses. |
| 2 | // |
| 3 | // A .prl is text jdat: an ordered map of ordered maps, lists, strings and typed scalar atoms. This |
| 4 | // parses exactly the subset the Pearl v1 writer emits (see fe2o3_austenite/src/emit/pearl.rs) and |
| 5 | // returns plain JavaScript values -- objects, arrays, strings and numbers -- so the renderer consumes |
| 6 | // the real document with no JSON projection in between. |
| 7 | // |
| 8 | // The subset, confirmed against a real .prl: |
| 9 | // (omap|{ "k": v, ... }) an ordered map -> a plain object, insertion order preserved |
| 10 | // [ v, v, ... ] a list -> an array |
| 11 | // "..." a string (RFC 8259 escapes) -> a string |
| 12 | // (u32|N) (i32|N) (u8|N) integer atoms -> a Number |
| 13 | // (f32|X.YeZ) a float atom (scientific) -> a Number |
| 14 | // |
| 15 | // There are no byte-strings, bools or nulls in a v1 .prl: a raster's PNG rides as a base64 *string* |
| 16 | // (the writer stores `base64::encode(png)`), so the whole file is these five shapes. Unknown type tags |
| 17 | // and unknown map keys are tolerated -- a parallel lane may add fields -- by stripping the tag and |
| 18 | // keeping the value, and by never assuming a fixed key set. |
| 19 | |
| 20 | "use strict"; |
| 21 | |
| 22 | // A recursive-descent parser over the source string, tracking one cursor. |
| 23 | function parseJdat(src) { |
| 24 | let i = 0; |
| 25 | const n = src.length; |
| 26 | |
| 27 | function isWs(c) { return c === " " || c === "\t" || c === "\n" || c === "\r"; } |
| 28 | |
| 29 | function skipWs() { |
| 30 | while (i < n && isWs(src[i])) i++; |
| 31 | } |
| 32 | |
| 33 | function fail(msg) { |
| 34 | const near = src.slice(Math.max(0, i - 20), i + 20); |
| 35 | throw new Error(`jdat parse error at ${i}: ${msg} -- near "${near}"`); |
| 36 | } |
| 37 | |
| 38 | // A value is a typed atom, a map, a list, a string, or a bare number. |
| 39 | function value() { |
| 40 | skipWs(); |
| 41 | if (i >= n) fail("unexpected end of input"); |
| 42 | const c = src[i]; |
| 43 | if (c === "(") return typed(); |
| 44 | if (c === "{") return map(); |
| 45 | if (c === "[") return list(); |
| 46 | if (c === '"') return str(); |
| 47 | return number(); |
| 48 | } |
| 49 | |
| 50 | // A typed atom: '(' TAG '|' inner ')'. The tag is stripped -- an omap's inner is a map, a scalar's |
| 51 | // inner is a bare number -- so the caller sees the plain value either way. |
| 52 | function typed() { |
| 53 | i++; // ( |
| 54 | const tagStart = i; |
| 55 | while (i < n && src[i] !== "|" && src[i] !== ")") i++; |
| 56 | if (i >= n) fail("unterminated typed atom"); |
| 57 | // A tag with no '|' (no payload separator) is not a shape this subset uses. |
| 58 | if (src[i] === ")") fail("typed atom without a '|' payload"); |
| 59 | i++; // | |
| 60 | const inner = value(); |
| 61 | skipWs(); |
| 62 | if (src[i] !== ")") fail("typed atom not closed with ')'"); |
| 63 | i++; // ) |
| 64 | return inner; |
| 65 | } |
| 66 | |
| 67 | // A map body: '{' ( "key" : value (',' ...)* )? '}'. Keys are always quoted strings here. |
| 68 | function map() { |
| 69 | i++; // { |
| 70 | const obj = {}; |
| 71 | skipWs(); |
| 72 | if (src[i] === "}") { i++; return obj; } |
| 73 | for (;;) { |
| 74 | skipWs(); |
| 75 | if (src[i] !== '"') fail("map key is not a string"); |
| 76 | const key = str(); |
| 77 | skipWs(); |
| 78 | if (src[i] !== ":") fail("map key not followed by ':'"); |
| 79 | i++; // : |
| 80 | obj[key] = value(); |
| 81 | skipWs(); |
| 82 | if (src[i] === ",") { i++; continue; } |
| 83 | if (src[i] === "}") { i++; break; } |
| 84 | fail("expected ',' or '}' in map"); |
| 85 | } |
| 86 | return obj; |
| 87 | } |
| 88 | |
| 89 | // A list: '[' ( value (',' ...)* )? ']'. |
| 90 | function list() { |
| 91 | i++; // [ |
| 92 | const arr = []; |
| 93 | skipWs(); |
| 94 | if (src[i] === "]") { i++; return arr; } |
| 95 | for (;;) { |
| 96 | arr.push(value()); |
| 97 | skipWs(); |
| 98 | if (src[i] === ",") { i++; continue; } |
| 99 | if (src[i] === "]") { i++; break; } |
| 100 | fail("expected ',' or ']' in list"); |
| 101 | } |
| 102 | return arr; |
| 103 | } |
| 104 | |
| 105 | // A quoted string with RFC 8259 escapes. The .prl's own strings (hex keys, slug keys, SVG path data |
| 106 | // and base64) carry no escapes, but the full set is handled so any legal jdat string round-trips. |
| 107 | function str() { |
| 108 | i++; // opening quote |
| 109 | let out = ""; |
| 110 | while (i < n) { |
| 111 | const c = src[i++]; |
| 112 | if (c === '"') return out; |
| 113 | if (c === "\\") { |
| 114 | const e = src[i++]; |
| 115 | switch (e) { |
| 116 | case '"': out += '"'; break; |
| 117 | case "\\": out += "\\"; break; |
| 118 | case "/": out += "/"; break; |
| 119 | case "b": out += "\b"; break; |
| 120 | case "f": out += "\f"; break; |
| 121 | case "n": out += "\n"; break; |
| 122 | case "r": out += "\r"; break; |
| 123 | case "t": out += "\t"; break; |
| 124 | case "u": { |
| 125 | const hex = src.slice(i, i + 4); |
| 126 | i += 4; |
| 127 | out += String.fromCharCode(parseInt(hex, 16)); |
| 128 | break; |
| 129 | } |
| 130 | default: out += e; // Tolerate an unknown escape by keeping the char. |
| 131 | } |
| 132 | } else { |
| 133 | out += c; |
| 134 | } |
| 135 | } |
| 136 | fail("unterminated string"); |
| 137 | } |
| 138 | |
| 139 | // A bare number: integer or float, with an optional sign and scientific exponent (jdat writes f32 as |
| 140 | // e.g. "5.665e0"). parseFloat covers every form this subset emits. |
| 141 | function number() { |
| 142 | const start = i; |
| 143 | while (i < n && "+-0123456789.eE".includes(src[i])) i++; |
| 144 | if (i === start) fail("expected a value"); |
| 145 | const num = parseFloat(src.slice(start, i)); |
| 146 | if (Number.isNaN(num)) fail("not a number"); |
| 147 | return num; |
| 148 | } |
| 149 | |
| 150 | const result = value(); |
| 151 | skipWs(); |
| 152 | // Trailing content after the top value means the file was not the single document map expected. |
| 153 | if (i < n) fail("trailing content after document"); |
| 154 | return result; |
| 155 | } |
| 156 | |
| 157 | // --------------------------------------------------------------------------------------------------- |
| 158 | // Serialising back to text jdat -- the inverse of the parser above, enough to re-emit an annotation and |
| 159 | // the list that holds it. The parser strips a scalar's type tag on the way in, so a re-serialisation of |
| 160 | // the *whole* document from the parsed model could not know a number's original tag (u32 vs i32 vs u8 vs |
| 161 | // f32), and the Rust decoder rejects the wrong one (`try_extract_dat!(_, U8)` and its kin). Annotations |
| 162 | // are the one section the authoring layer builds itself, with every atom's tag known, so this serialiser |
| 163 | // carries the tag explicitly through the small wrappers below and emits a section the Rust reader accepts. |
| 164 | // --------------------------------------------------------------------------------------------------- |
| 165 | |
| 166 | // A scalar with its jdat tag preserved, so `encodeJdat` writes `(<tag>|<n>)` rather than a bare number. |
| 167 | function atom(tag, n) { return { __atom: tag, value: n }; } |
| 168 | |
| 169 | // An ordered map with its tag, so `encodeJdat` writes `(omap|{ ... })` rather than a bare `{ ... }`. |
| 170 | function omap(obj) { return { __omap: obj }; } |
| 171 | |
| 172 | // A jdat string with RFC 8259 escapes, matching the escapes the parser's `str()` accepts. |
| 173 | function encodeString(s) { |
| 174 | let out = '"'; |
| 175 | for (const ch of s) { |
| 176 | switch (ch) { |
| 177 | case '"': out += '\\"'; break; |
| 178 | case "\\": out += "\\\\"; break; |
| 179 | case "\b": out += "\\b"; break; |
| 180 | case "\f": out += "\\f"; break; |
| 181 | case "\n": out += "\\n"; break; |
| 182 | case "\r": out += "\\r"; break; |
| 183 | case "\t": out += "\\t"; break; |
| 184 | default: |
| 185 | const code = ch.codePointAt(0); |
| 186 | if (code < 0x20) out += "\\u" + code.toString(16).padStart(4, "0"); |
| 187 | else out += ch; |
| 188 | } |
| 189 | } |
| 190 | return out + '"'; |
| 191 | } |
| 192 | |
| 193 | // Emits a value as text jdat, reproducing the spacing the Rust encoder uses: a space after `{`, `[`, `:` |
| 194 | // and `,`, and none before the closing bracket. An `omap`/`atom` wrapper carries its tag; a plain string, |
| 195 | // array or number falls through to the bare forms. |
| 196 | function encodeJdat(v) { |
| 197 | if (v && v.__atom !== undefined) return `(${v.__atom}|${v.value})`; |
| 198 | if (v && v.__omap !== undefined) { |
| 199 | const obj = v.__omap; |
| 200 | const keys = Object.keys(obj); |
| 201 | if (keys.length === 0) return "(omap|{})"; |
| 202 | const body = keys.map(k => `${encodeString(k)}: ${encodeJdat(obj[k])}`).join(", "); |
| 203 | return `(omap|{ ${body}})`; |
| 204 | } |
| 205 | if (Array.isArray(v)) { |
| 206 | if (v.length === 0) return "[]"; |
| 207 | return `[ ${v.map(encodeJdat).join(", ")}]`; |
| 208 | } |
| 209 | if (typeof v === "string") return encodeString(v); |
| 210 | if (typeof v === "number") return String(v); |
| 211 | throw new Error("encodeJdat: cannot serialise " + typeof v); |
| 212 | } |
| 213 | |
| 214 | window.Jdat = { |
| 215 | parse: parseJdat, |
| 216 | encode: encodeJdat, |
| 217 | atom, |
| 218 | omap, |
| 219 | i32: n => atom("i32", Math.round(n)), |
| 220 | u32: n => atom("u32", Math.round(n)), |
| 221 | u8: n => atom("u8", Math.round(n)), |
| 222 | f32: n => atom("f32", n), |
| 223 | }; |