]z<@Gx,!&hFPCRlUhFPCRlVcqvPw'?[)a0a4?XgLgGw`Rh#axGw>!(jdw^QPCRfd~qsJ~pRfe5Zzjg8QReKHvDA[)]yi1@bk~OMv?r]Gvy[gHvKRecg>v`ReHdl#!&!.a,a3!a7aQUaTawbEbAbM!bObabf~r4p}Sekx8W?2p~t~Gx0^Dt{29p4Gw:|>!Zz=Nw}4p|xpf?vbKv&sX=5ZzlGuX[]ykMx5uAg@x5u=df#!'U.a'a+!a.Va1`a6gDsFxdd6tCGx6W~vrku#]z?NuZ;rl2Du/2DuE=5Zzm2]yle3#0Ua)Ua-___a1b,bGbTbkcJcPcR!cU!cVcXc[cfcr!cscxc|d'd(HQAW&?E|hE{tKvJRlMKwCRlDxVhKRo&1Xl}de#[!*V!a%0!a/a1a4!ab~wrniowfx*Rm_?1bq|Xg8s:t46Sl.vE^RmEbh|ZRn|BW/a&Mverrd+#V%89rpbo|Ube~_u%Iw+wSW)a*1Rl^BQRm&hWs>Xlh7Rfly*hI#|[_!ag!aqAcqu|w>A!)a,a3aTNPCxbfL1bl~Fj>w(vs;Gwr[gCvgRft=Rfsjmw^QPwWRf~hNv!AHu|@!'.NPCxkfP|P1?=<<;bu}*>B861??Gw/WAE|R6AE~/AE|S8bp|TdN#U&V!*gFw`Rm9<[45F~(hNv!YfgHul9W'Xm=ZznB@Rn8hNua@Rn>c=#&@Yg`bd}3xmkZ#}HgBu[RkZ1?xYa=YmGdi#[V+V0VUa'?E|?;4E}Z1?xSg&Rm,ReF?8RiCdi#!&V|K*V`a(Ua1gEvuRe_]ym:SNux^Ro(Ro*Gwp}}!Xnp?4bh~+gAwJRiMKwCRm5hYwJYnrGv,[1XmJxmkY|=}G?Gwr!&hWs>Xlgg>QRm%Gv`!(jGvfw->ThM~!bS~ne-#___V&!aRa[a`ah!alaoUb(!b/b3|FUb:!bEf{weRh!da#_a#`/!a9?p@Gw]^RkoxTk~@4Rk~xafbA8Xoy!1?bd~N:xTh*xih-|11E{r@BGvy^bh|%bh|&Hsm=!a(a-xphiuS@!|(a#To[J4A{.hgen>xihm~m2xThfbt|'xphjuS@!|)a#To]J4A{.hhen>xihn~mdY#WU'V!)bo{qLuYq%bj{yjAv]vH4Gw>!'g;w^xTilbt{vhNv!AxTimbt{}xZd8Sa8wJ[?@[)-0a/a4aIaVKv%Ri3KwCRl@=Th6enKv%Ri2Gw]WThJenTaSenhSv1;RmuHuRA[(?Xl>ThIen{1aQen84Ti9enGwX[?Xl?Mu^Ti:en8:Th%enHru;[0?Xg1?GwReY]zDdB#U[!)xdd=t?:1F{I;4Rkmxrd9#!(j?uqw@axRiXt@Gwo!-;Gv>RoK8:Rij8bp{l:4xkeZ{hHry@~g!~ibs~cHun={j!a'HO@[)LQRj:g?ubRj/gHusRj0xbgcbr{iMvFRi6d@#V!&]ypt;hSwH=Re@dc#U}RV!&+U/!a)]zE=5ZzsNvU4Ref2]yqHuVW/a&Mves/d+#V%89s-bo{cbe~_u*IvDwRW)a%1Rl_hIs>Xm(Bwi!)a,a@aHaRNPCxbfN1bl~:j>w(vs;Gwr[B861??ZztB@Rn9hNua@Rn?Gvx!'?xVa>YmHiqv4wdRn*1?E~MIslw0[{[)gEvuRe`]yrGwp~1!Bbr}Vy!h`ryuXvbwQD^,.a(a+a:a?aBRoJca#WRoN<;s7Bbh{bxSoH89s5?2s3Hry@^&RoL=RoP8:Rikiqv4wdRn+8bp{`u+dA#V!*:1Sd;x!Ut7y+h%#&!)V-!0!a(!a,xihk~[4>xiho~`:Ro^Gvy^RohRof2xwh`slv`@!'a%/~*==?bt~#6Rkty(hhp*qaFaJpIpCpDRo]GwPWYoTBavRokxShlxihl~W4>xihp~b:Ro_Gvy^RogRoiHru;[/?Xg4?Gw<^bn}0x]fUbxyLgIQRlZLuypodq#,.a0a3a8!a<8V!b!b;!b@br`b|cp#U&g=w]Rj2t8?bn{!HuVDW'Mves;~vs9u,8:Rh%@bs{:GwF!~mbs{=?;q1c`#%}13bhy:4@xDbZ#!'xPi'Xn!*45A1??^RfvRfu9axRkDGw[!/Gw@!(?;xTj3Xj3<=Rj-?8RkeGOW2?sAp:Gv~W<;sI5Zzwdl#U'`/`a5VUa8!aD??w(vs;Gw>[45F~Q867F~PBbu{+8xxd<#U^t><;t9j^QPwWRg!Hv.A!/a'Gw@!(?;xTj4Xj4<=Rj,;6sE?8Rkf2]yuHv/?[({0ho~!To`~!jHw]ub>{0hp~!Toa~!Gw<[4Abd}{jAv]vH4Gw>[g;w^Ri7hNv!Yi8DtxKv%Ri)HvF?!-0SgnuW[1Xi@>RhBLw!RipGwZ!{*1E{)]zJA?bl~Z@BGvyWThgenThhen=5Zzx?Gw+!{?]yxGwf[9Bbu{E?bl{Adl#!-!a)a+Ua/V!a2`a7!a<2d%#U[A4q0u9Gx8W?2sMu5;p7]zM2Du@=5Zz{2]yyGvR^Du89q2dq#!&a%0a/!a3a6V!a;`a@`aE~r4sPGx4WMvesTt} no value)
- * 13 FLAG13. If valueLength>0: semicolon required flag (implicit ';').
- * If valueLength==0: compact run flag.
- * 12..7 BRANCH_LENGTH Branch length (0 => single branch in 6..0 if jumpOffset==char) OR run length (when compact run)
- * 6..0 JUMP_TABLE Jump offset (jump table) OR single-branch char code OR first run char
+ * The trie is a flat `Uint16Array`. Every node starts with one header word:
+ *
+ * 15..14 VALUE_LENGTH Number of words the value occupies, +1.
+ * 0 = no value; 1 = value inline in bits 13..0;
+ * 2/3 = value in the 1/2 words after the header.
+ * 13 FLAG13 If VALUE_LENGTH > 0: semicolon required ("strict"
+ * entity; `;` is never stored as a branch).
+ * If VALUE_LENGTH == 0: this node is a compact run.
+ * 12..7 BRANCH_LENGTH Number of branches (or run length for runs).
+ * 6..0 JUMP_TABLE Jump-table offset / single-branch char / first
+ * run char (see below).
+ *
+ * Branch data follows the header and any value words. Its shape is selected
+ * by (JUMP_TABLE, BRANCH_LENGTH) in the header:
+ *
+ * Single branch JUMP_TABLE = the only child's char, BRANCH_LENGTH = 0.
+ * No branch words; the child node follows immediately.
+ * Jump table JUMP_TABLE = first covered char (> 0), BRANCH_LENGTH =
+ * table length. One word per covered char: 0 = no branch,
+ * otherwise the child's offset from the END of the table,
+ * +1 (so 0 stays the no-branch sentinel).
+ * Dictionary JUMP_TABLE = 0, BRANCH_LENGTH = number of branches.
+ * ceil(n/2) words of sorted keys packed two per word
+ * (low byte first), then n pointer words storing the
+ * child's offset from the END of the branch data.
+ * Compact run VALUE_LENGTH = 0, FLAG13 set. BRANCH_LENGTH = run
+ * length (3..63), JUMP_TABLE = first char; remaining run
+ * chars packed two per word after the header. The target
+ * node follows the packed words immediately.
+ *
+ * Pointers are end-relative (rather than relative to the pointer's own
+ * position) because that makes the common "child encoded right after the
+ * branch data" case a small constant, which compresses far better. Offsets
+ * to already-encoded (shared) nodes wrap via uint16 modulo arithmetic; the
+ * decoder masks navigation results with `& 0xff_ff` to match.
*/
export const enum BinTrieFlags {
VALUE_LENGTH = 0b1100_0000_0000_0000,
diff --git a/src/internal/decode-shared.ts b/src/internal/decode-shared.ts
index f4ed62cf..ba0aeef5 100644
--- a/src/internal/decode-shared.ts
+++ b/src/internal/decode-shared.ts
@@ -1,107 +1,222 @@
-/** Number of most-frequent values assigned to 1-char codes. */
-const DICT_SIZE = 61;
+/*
+ * Inverse of the encoder's SAFE alphabet (0x21..0x7E minus 0x22, 0x24, 0x5C),
+ * precomputed once at module load. A flat table lookup per input char is
+ * measurably cheaper during import than recomputing the three exclusion
+ * comparisons inline; entries for excluded chars stay 0 but are never read.
+ */
+const BASE91_INVERSE = /* #__PURE__ */ (() => {
+ const table = new Uint8Array(127);
+ let code = 0;
+ for (let char = 0x21; char <= 0x7e; char++) {
+ if (char !== 0x22 && char !== 0x24 && char !== 0x5c) {
+ table[char] = code++;
+ }
+ }
+ return table;
+})();
/**
- * Decode a dictionary-encoded trie string into a Uint16Array.
+ * Decode a dictionary-encoded trie string back into its Uint16Array.
+ *
+ * Stream layout (consumed in this order):
+ * 1. dict1 atoms — `dict1AtomCount` uint16 values, delta+RLE encoded.
+ * 2. dict2 atoms — `atomCount - dict1AtomCount` values, delta+RLE.
+ * 3. dict2 ngrams — `ngramCount - (dictSize - dict1AtomCount)` entries,
+ * each a pair of slot codes that resolve to earlier slots.
+ * 4. dict1 ngrams — `dictSize - dict1AtomCount` entries, same shape.
+ * 5. data — slot codes, each expanding to one or more uint16 values.
*
- * Format: [dict1: D values delta+RLE][dict2: remaining delta+RLE][data]
+ * Codes use a 91-char base (printable ASCII minus `"`, `$`, `\`):
+ * - char1 < dictSize → 1-char code, slot = char1
+ * - char1 ≥ dictSize → 2-char code, slot = dictSize + (char1 - dictSize)*91 + char2
*
- * - dict1: D most-frequent values, delta-encoded from 0 → 1-char codes.
- * - dict2: remaining unique values, delta-encoded from 0 → 2-char codes.
- * - data: each trie entry as 1 char (dict1) or 2 chars (dict2).
+ * Slot index → token kind:
+ * [0, A) dict1 atoms (1-char codes)
+ * [A, dictSize) dict1 ngrams (1-char codes)
+ * [dictSize, dictSize+D) dict2 atoms (2-char codes)
+ * [dictSize+D, end) dict2 ngrams (2-char codes)
+ *
+ * Both atom dicts decode before any ngram, and dict2 ngrams decode before
+ * dict1 ngrams. So every ngram entry references slots whose contents are
+ * already filled — no forward references to handle.
+ *
+ * This runs on library import, so it is written for a cold VM: flat typed
+ * arrays instead of per-slot number[]s, indexed loops instead of iterators
+ * or spreads, and slot contents stored as either a plain value (`single`,
+ * covering every atom) or a range in a shared `pool` (ngrams).
* @param input Packed trie string.
* @param resultLength Expected number of uint16 values in the output.
- * @param headerLength Number of chars occupied by the dict1+dict2 header.
+ * @param atomCount Total number of distinct uint16 values in the trie.
+ * @param dict1AtomCount Atoms in the 1-char range (`A` above).
+ * @param ngramCount Total number of ngram entries (dict1 + dict2).
+ * @param dictSize Number of 1-char code slots; the rest of `BASE - dictSize`
+ * first-byte values are 2-char codes.
*/
export function decodeTrieDict(
input: string,
resultLength: number,
- headerLength: number,
+ atomCount: number,
+ dict1AtomCount: number,
+ ngramCount: number,
+ dictSize: number,
): Uint16Array {
const base = 91;
-
- // Build base-91 lookup table inline (91 printable ASCII chars, excluding `"`, `$`, `\`).
- const lookup = new Uint8Array(0x7f);
- for (let codePoint = 0x21, index = 0; codePoint <= 0x7e; codePoint++) {
- if (codePoint !== 0x22 && codePoint !== 0x24 && codePoint !== 0x5c) {
- lookup[codePoint] = index++;
- }
- }
+ const inputLength = input.length;
+ // For 2-char codes, slot = char1 * base - twoCharBias + char2.
+ const twoCharBias = dictSize * (base - 1);
let pos = 0;
+ /** Read one slot code at `pos` and return its slot index, advancing pos. */
+ const readSlotCode = (): number => {
+ const c1 = BASE91_INVERSE[input.charCodeAt(pos++)];
+ return c1 < dictSize
+ ? c1
+ : c1 * base - twoCharBias + BASE91_INVERSE[input.charCodeAt(pos++)];
+ };
+
+ const dict2AtomCount = atomCount - dict1AtomCount;
+ const slotCount = atomCount + ngramCount;
+
/*
- * Delta-decode helper: reads `count` values (0 = read until `endPos`).
+ * Per-slot contents: atoms (always a single value) live directly in
+ * `single`; ngram slots hold -1 there and expand to
+ * `pool[start[slot] .. start[slot] + length[slot])`.
+ */
+ const single = new Int32Array(slotCount);
+ single.fill(-1, dict1AtomCount, dictSize);
+ single.fill(-1, dictSize + dict2AtomCount, slotCount);
+ const start = new Int32Array(slotCount);
+ const length = new Int32Array(slotCount);
+
+ /**
+ * Decode `count` ascending uint16 values from a delta+RLE stream into
+ * `single[off..off+count)`.
*
- * Encoding per delta:
- * code < 89 → delta = code
- * code == 89 → RLE: next char = N-2, emit N consecutive +1 values
- * code == 90 → escape: next chars encode delta-89 (2 or 3 chars)
+ * code < 89 → delta = code
+ * code == 89 → run-length: next char encodes runLength-2; emit `runLength` consecutive +1 values
+ * code == 90, next < 90 → escape: delta = 89 + next * BASE + after-next
+ * code == 90, next == 90 → double-escape: extra char for very large deltas
+ * @param count
+ * @param off
*/
- function decodeDelta(count: number, endPos: number): number[] {
- const result: number[] = [];
+ function decodeDelta(count: number, off: number): void {
let previous = 0;
- while (count > 0 ? result.length < count : pos < endPos) {
- const code = lookup[input.charCodeAt(pos)];
+ let slot = off;
+ const end = off + count;
+ while (slot < end) {
+ const code = BASE91_INVERSE[input.charCodeAt(pos++)];
if (code < 89) {
previous += code;
- pos += 1;
- result.push(previous);
+ single[slot++] = previous;
} else if (code === 89) {
- // RLE: next char encodes count-2, emit count consecutive +1 values
- pos += 1;
- const runLength = lookup[input.charCodeAt(pos)] + 2;
- pos += 1;
- for (let r = 0; r < runLength; r++) {
- result.push(++previous);
- }
+ let runLength = BASE91_INVERSE[input.charCodeAt(pos++)] + 2;
+ while (runLength--) single[slot++] = ++previous;
} else {
- // Escape: next char(s) encode a larger delta
- pos += 1;
- const next = lookup[input.charCodeAt(pos)];
- if (next < 90) {
- previous +=
- 89 + next * base + lookup[input.charCodeAt(pos + 1)];
- pos += 2;
- } else {
- // Double escape
- pos += 1;
- previous +=
- 89 +
- lookup[input.charCodeAt(pos)] * 8281 +
- lookup[input.charCodeAt(pos + 1)] * base +
- lookup[input.charCodeAt(pos + 2)];
- pos += 3;
- }
- result.push(previous);
+ const next = BASE91_INVERSE[input.charCodeAt(pos++)];
+ previous +=
+ 89 +
+ // eslint-disable-next-line unicorn/prefer-minimal-ternary -- branches read a different number of side-effecting input bytes
+ (next < 90
+ ? next * base + BASE91_INVERSE[input.charCodeAt(pos++)]
+ : BASE91_INVERSE[input.charCodeAt(pos++)] * 8281 +
+ BASE91_INVERSE[input.charCodeAt(pos++)] * base +
+ BASE91_INVERSE[input.charCodeAt(pos++)]);
+ single[slot++] = previous;
}
}
- return result;
}
- // Decode dict1: DICT_SIZE values, delta-encoded from 0
- const dict1 = new Uint16Array(decodeDelta(DICT_SIZE, 0));
+ // Streams 1 & 2: atoms decoded into their slot ranges.
+ decodeDelta(dict1AtomCount, 0);
+ decodeDelta(dict2AtomCount, dictSize);
+
+ /*
+ * Streams 3 & 4 are read in two passes: first collect every ngram's two
+ * references and derive its expanded length (each ref resolves to an earlier
+ * slot, so lengths are already known), which sizes the shared pool.
+ * `order` records the slot each entry fills, since stream 3 (dict2)
+ * decodes before stream 4 (dict1) but occupies higher slots.
+ */
+ const references = new Int32Array(ngramCount * 2);
+ const order = new Int32Array(ngramCount);
+ let poolSize = 0;
+ let ngramIndex = 0;
- // Decode dict2: remaining values until header ends, delta-encoded from 0
- const dict2 = decodeDelta(0, headerLength);
+ /**
+ * Read `count` ngram entries (each = 2 slot-code references) for the slots
+ * starting at `startSlot`, recording references and assigning pool ranges.
+ * @param count
+ * @param startSlot
+ */
+ function readNgramReferences(count: number, startSlot: number): void {
+ for (let index = 0; index < count; index++) {
+ const slot = startSlot + index;
+ const a = readSlotCode();
+ const b = readSlotCode();
+ references[ngramIndex * 2] = a;
+ references[ngramIndex * 2 + 1] = b;
+ order[ngramIndex++] = slot;
+ start[slot] = poolSize;
+ const entryLength =
+ (single[a] < 0 ? length[a] : 1) +
+ (single[b] < 0 ? length[b] : 1);
+ length[slot] = entryLength;
+ poolSize += entryLength;
+ }
+ }
+ readNgramReferences(
+ ngramCount - dictSize + dict1AtomCount,
+ dictSize + dict2AtomCount,
+ );
+ readNgramReferences(dictSize - dict1AtomCount, dict1AtomCount);
+
+ // Second pass: concatenate each ngram's two halves into the pool.
+ const pool = new Uint16Array(poolSize);
+ for (let index = 0; index < ngramIndex; index++) {
+ let write = start[order[index]];
+ for (let half = 0; half < 2; half++) {
+ const source = references[index * 2 + half];
+ const value = single[source];
+ if (value < 0) {
+ let read = start[source];
+ const readEnd = read + length[source];
+ while (read < readEnd) pool[write++] = pool[read++];
+ } else {
+ pool[write++] = value;
+ }
+ }
+ }
- // Decode data
+ // Stream 5: data. Each code expands to its slot's stored values.
const out = new Uint16Array(resultLength);
let outIndex = 0;
- while (pos < input.length) {
- const code = lookup[input.charCodeAt(pos)];
- if (code < DICT_SIZE) {
- out[outIndex++] = dict1[code];
- pos += 1;
+ while (pos < inputLength) {
+ let slot = BASE91_INVERSE[input.charCodeAt(pos++)];
+ if (slot >= dictSize) {
+ slot =
+ slot * base -
+ twoCharBias +
+ BASE91_INVERSE[input.charCodeAt(pos++)];
+ }
+ const value = single[slot];
+ if (value < 0) {
+ let read = start[slot];
+ const readEnd = read + length[slot];
+ while (read < readEnd) out[outIndex++] = pool[read++];
} else {
- out[outIndex++] =
- dict2[
- (code - DICT_SIZE) * base +
- lookup[input.charCodeAt(pos + 1)]
- ];
- pos += 2;
+ out[outIndex++] = value;
}
}
-
return out;
}
+
+/*
+ * Warm-up: decode a minimal trie (one atom with value 5, one data token) at
+ * module load. The real call decodes ~18k chars in a cold VM; this dummy
+ * call makes V8 baseline-compile `decodeTrieDict` first, which speeds the
+ * real decode up by ~15% (measured on fresh-process import). Intentionally
+ * not marked #__PURE__ — dropping it would silently undo the effect.
+ */
+// eslint-disable-next-line unicorn/no-top-level-side-effects -- deliberate warm-up, see above
+decodeTrieDict("(!", 1, 1, 1, 0, 1);