fix(code-runtime-python): meter escaped string bytes without allocating, bind frame roster to the unions

Two review findings on checkDoneValue's metering and the wire-mirror binding:

- The string/key byte check used a decoded-length lower bound and then called
  JSON.stringify, which materializes the ~6x escaped copy before the over-budget
  check — the hundreds-of-MB spike the metered walk exists to avoid. Add
  jsonStringBytesUpTo, a non-allocating scan that computes the exact escaped
  UTF-8 size (matching JSON.stringify byte for byte, including surrogate pairs
  vs lone surrogates) and bails the instant it crosses the remaining budget; use
  it for both string values and object keys.
- The frame roster in WIRE_FRAME_FIELD_ROLES was hand-written, so a frame added
  to ChildToHost/ReplyMessage without a roster entry slipped past. Introduce
  WireFrameShapes (name -> interface) as the canonical roster the roles map is
  bound against, plus a WireFrameShapesCoverUnions compile-time assertion that
  every message-union member appears in it (verified: adding a frame to a union
  without a WireFrameShapes entry fails typecheck).
This commit is contained in:
Chinesezjc
2026-08-07 13:27:54 +08:00
parent 4de8c914c9
commit f9ab1edc68
2 changed files with 124 additions and 43 deletions
@@ -170,15 +170,44 @@ type OptionalKeys<T> = { [K in keyof T]-?: object extends Pick<T, K> ? K : never
*/
type FrameFieldRoles<T> = Record<RequiredKeys<T>, 'required'> & Record<OptionalKeys<T>, 'optional'>
interface WireFrameShapes {
BootMessage: BootMessage
Namespace: Namespace
RunMessage: RunMessage
BootAckMessage: BootAckMessage
CallMessage: CallMessage
LogMessage: LogMessage
DoneErrorField: DoneErrorField
DoneMessage: DoneMessage
ErrorClass: ErrorClass
ReplyOk: ReplyOk
ReplyErr: ReplyErr
}
/**
* Compile-time proof that {@link WireFrameShapes} lists every frame carried on a
* message union: the union of the frame types (`ChildToHost`, the reply
* variants, and the host-to-child boot/run frames) must be assignable to the
* union of the roster's value types. Adding a frame to a union without a
* `WireFrameShapes` entry makes this alias `false`, so the assignment below
* fails to compile — closing the whole-frame drift the field-level binding
* alone could not see. Nested shapes (`Namespace`, `ErrorClass`,
* `DoneErrorField`) are not union members; they are covered by the roles
* `satisfies` and the mirror e2e's roster comparison.
*/
type WireFrameShapesCoverUnions =
[ChildToHost | ReplyMessage | BootMessage | RunMessage] extends [WireFrameShapes[keyof WireFrameShapes]] ? true : false
const _wireFrameShapesCoverUnions: WireFrameShapesCoverUnions = true
void _wireFrameShapesCoverUnions
/**
* Each frame's wire fields tagged by required/optional, keyed by field name so
* the mapping is exhaustive over the frame interface (see
* {@link FrameFieldRoles}). Bound to the interfaces by `satisfies` below, this
* is the single source of truth the cross-language mirror test derives its
* expectations from; {@link WIRE_FRAME_FIELDS} projects it to sorted
* required/optional arrays for the comparison. `global` is the JSON key
* {@link CallMessage} and {@link Namespace} send (a reserved word the Python
* side carries via a functional `TypedDict`).
* the mapping is exhaustive over the frame interface (see {@link FrameFieldRoles})
* across the whole {@link WireFrameShapes} roster. Bound to the interfaces by
* `satisfies` below; {@link WIRE_FRAME_FIELDS} projects it to sorted
* required/optional arrays for the cross-language mirror comparison. `global` is
* the JSON key {@link CallMessage} and {@link Namespace} send (a reserved word
* the Python side carries via a functional `TypedDict`).
*/
const WIRE_FRAME_FIELD_ROLES = {
BootMessage: { type: 'required', cpuSeconds: 'required', addressSpaceBytes: 'required', maxLogBytes: 'required', maxValueBytes: 'required', namespaces: 'required' },
@@ -192,19 +221,7 @@ const WIRE_FRAME_FIELD_ROLES = {
ErrorClass: { name: 'required', memberNameProperty: 'required' },
ReplyOk: { type: 'required', id: 'required', ok: 'required', value: 'required' },
ReplyErr: { type: 'required', id: 'required', ok: 'required', message: 'required' },
} as const satisfies {
BootMessage: FrameFieldRoles<BootMessage>
Namespace: FrameFieldRoles<Namespace>
RunMessage: FrameFieldRoles<RunMessage>
BootAckMessage: FrameFieldRoles<BootAckMessage>
CallMessage: FrameFieldRoles<CallMessage>
LogMessage: FrameFieldRoles<LogMessage>
DoneErrorField: FrameFieldRoles<DoneErrorField>
DoneMessage: FrameFieldRoles<DoneMessage>
ErrorClass: FrameFieldRoles<ErrorClass>
ReplyOk: FrameFieldRoles<ReplyOk>
ReplyErr: FrameFieldRoles<ReplyErr>
}
} as const satisfies { [K in keyof WireFrameShapes]: FrameFieldRoles<WireFrameShapes[K]> }
/**
* The wire field names of each frame, split into sorted required and optional
@@ -307,6 +324,53 @@ function scalarJson(current: unknown): string {
return String(current)
}
/**
* Exact UTF-8 byte length of one string's compact JSON form (quotes + escapes),
* computed by a single non-allocating scan that stops the instant the running
* total exceeds `maxBytes`. Used instead of `Buffer.byteLength(JSON.stringify(s))`
* so a control-heavy forged string — whose escaped copy expands up to ~6x — is
* rejected BEFORE that copy is materialized: `JSON.stringify` would allocate the
* full escaped form first, the very hundreds-of-MB spike the metered traversal
* exists to avoid. Mirrors `JSON.stringify`'s escaping byte-for-byte: `"` and
* `\` and the five short C0 escapes cost 2, other C0 controls `\uXXXX` cost 6, a
* valid surrogate pair is one astral code point emitted as raw 4-byte UTF-8, a
* LONE surrogate becomes `\uXXXX` at 6, and any other code point costs its raw
* UTF-8 width.
* @param text - the string to meter.
* @param maxBytes - largest serialized size the caller can still admit.
* @returns the exact serialized byte length, or `undefined` once it exceeds `maxBytes`.
*/
function jsonStringBytesUpTo(text: string, maxBytes: number): number | undefined {
let bytes = 2 // the two quotes
if (bytes > maxBytes) return undefined
for (let index = 0; index < text.length; index++) {
const code = text.charCodeAt(index)
if (code === 0x22 || code === 0x5c || code === 0x08 || code === 0x09 || code === 0x0a || code === 0x0c || code === 0x0d) {
bytes += 2 // `\"` `\\` `\b` `\t` `\n` `\f` `\r`
} else if (code < 0x20) {
bytes += 6 // other C0 controls: `\uXXXX`
} else if (code < 0x80) {
bytes += 1
} else if (code < 0x800) {
bytes += 2
} else if (code >= 0xd800 && code <= 0xdbff && index + 1 < text.length) {
const next = text.charCodeAt(index + 1)
if (next >= 0xdc00 && next <= 0xdfff) {
bytes += 4 // valid high+low pair: one astral code point, raw 4-byte UTF-8
index++
} else {
bytes += 6 // lone high surrogate: `\uXXXX`
}
} else if (code >= 0xd800 && code <= 0xdfff) {
bytes += 6 // lone surrogate (unpaired high at end, or any low): `\uXXXX`
} else {
bytes += 3 // other BMP code point
}
if (bytes > maxBytes) return undefined
}
return bytes
}
/**
* Meter a `JSON.parse`-produced done value's compact-JSON byte length AND its
* number losslessness in one traversal, stopping the instant `maxBytes` is
@@ -325,10 +389,11 @@ function scalarJson(current: unknown): string {
* enqueue loop. A non-lossless number (non-finite, negative zero) is caught only
* when the value fits the budget — an over-budget value is rejected regardless,
* so the distinction is moot. Same JSON-plain precondition and traversal shape
* as {@link encodeJsonPlain}; per-scalar byte length is measured through
* as {@link encodeJsonPlain}; a number's byte length is measured through
* {@link scalarJson} (matching the encoder, so a beyond-safe-range integer
* meters its exact BigInt digits, not `JSON.stringify`'s rounded spelling) and
* `JSON.stringify` for strings.
* a string's/key's through {@link jsonStringBytesUpTo} (the exact escaped size,
* scanned without allocating the escaped copy).
* @param value - a JSON-plain value (e.g. straight from `JSON.parse`).
* @param maxBytes - the completion-value budget in bytes.
* @returns `{ ok: true, bytes }` with the exact serialized size, or
@@ -356,12 +421,13 @@ export function checkDoneValue(value: unknown, maxBytes: number): { ok: true; by
if (!Number.isFinite(current) || Object.is(current, -0)) nonLossless = true
bytes += Buffer.byteLength(scalarJson(current), 'utf8')
} else if (typeof current === 'string') {
// Lower-bound BEFORE materializing the escaped form: every UTF-16 code
// unit is at least one UTF-8 byte plus the two quotes, so a huge or
// control-heavy forged string (whose escaped copy expands severalfold)
// is rejected without allocating that copy.
if (bytes + current.length + 2 > maxBytes) return { ok: false, reason: 'over-budget' }
bytes += Buffer.byteLength(JSON.stringify(current), 'utf8')
// Meter the escaped form WITHOUT allocating it: jsonStringBytesUpTo scans
// and bails the instant the running cost crosses the remaining budget, so
// a control-heavy forgery (escaped copy up to ~6x) never materializes that
// copy the way `JSON.stringify` would.
const stringBytes = jsonStringBytesUpTo(current, maxBytes - bytes)
if (stringBytes === undefined) return { ok: false, reason: 'over-budget' }
bytes += stringBytes
} else if (Array.isArray(current)) {
// Brackets plus one comma per gap; elements add themselves. Reject
// BEFORE enqueuing children: every element serializes to at least one
@@ -384,9 +450,11 @@ export function checkDoneValue(value: unknown, maxBytes: number): { ok: true; by
if (bytes + count * 4 > maxBytes) return { ok: false, reason: 'over-budget' }
for (const key in record) {
if (!Object.hasOwn(record, key)) continue
// The same string lower bound, before escaping the key.
if (bytes + key.length + 3 > maxBytes) return { ok: false, reason: 'over-budget' }
bytes += Buffer.byteLength(JSON.stringify(key), 'utf8') + 1
// Meter the key's escaped form without allocating it (same reason as the
// string branch), then add the colon separator. `+ 1` for the `:`.
const keyBytes = jsonStringBytesUpTo(key, maxBytes - bytes)
if (keyBytes === undefined) return { ok: false, reason: 'over-budget' }
bytes += keyBytes + 1
stack.push(record[key])
}
} else {
@@ -213,20 +213,33 @@ describe('checkDoneValue', () => {
expect(checkDoneValue(wide, 12)).toEqual({ ok: false, reason: 'over-budget' })
})
it('rejects an over-budget string on its length before escaping it', () => {
// A control-heavy forged string escapes to ~6x its length (each NUL becomes
// the 6-character `\u0000`); the walk must refuse it on the cheap
// `length + 2` lower bound so the escaped copy is never allocated. Observable
// through the boundary: a string whose LENGTH already exceeds the cap fails
// even though every source character is one UTF-16 code unit.
expect(checkDoneValue('\0'.repeat(4096), 1024)).toEqual({ ok: false, reason: 'over-budget' })
// The bound is a lower bound, never a false rejection: a string that fits
// exactly still passes with its exact escaped size — one NUL serializes to
// `"\u0000"`, i.e. two quotes plus the 6-character escape = 8 bytes.
it('meters a string\'s exact escaped size without allocating it', () => {
// A control-heavy string that fits by DECODED length but not once escaped
// must still reject: 200 NULs are 200 UTF-16 units (would pass a naive
// length bound against cap 1024) but escape to 200*6 + 2 = 1202 bytes.
// jsonStringBytesUpTo scans and bails before the escaped copy is built.
expect(checkDoneValue('\0'.repeat(200), 1024)).toEqual({ ok: false, reason: 'over-budget' })
// Exact-size acceptance, no false rejection: one NUL serializes to a
// 6-char \\uXXXX escape, so with the two quotes = 8 bytes.
expect(checkDoneValue('\0', 8)).toEqual({ ok: true, bytes: 8 })
expect(checkDoneValue('\0', 7)).toEqual({ ok: false, reason: 'over-budget' })
// Same lower bound for keys, checked before the key is escaped.
expect(checkDoneValue({ ['\0'.repeat(4096)]: 1 }, 1024)).toEqual({ ok: false, reason: 'over-budget' })
// Multi-byte and astral characters meter at their raw UTF-8 width (a valid
// surrogate pair is 4 bytes, matching JSON.stringify), not a 6-byte escape.
expect(checkDoneValue('\u00e9', 4)).toEqual({ ok: true, bytes: 4 }) // 2 quotes + 2-byte UTF-8
expect(checkDoneValue('\u{1f600}', 6)).toEqual({ ok: true, bytes: 6 }) // 2 quotes + 4-byte UTF-8
expect(checkDoneValue('\u{1f600}', 5)).toEqual({ ok: false, reason: 'over-budget' })
// A lone surrogate escapes to \\uXXXX = 6, so with quotes = 8.
expect(checkDoneValue('\ud800', 8)).toEqual({ ok: true, bytes: 8 })
// A high surrogate followed by a NON-low character is a lone surrogate (6-byte
// escape) plus that character: `\ud800` + `a` = 2 quotes + 6 + 1 = 9.
expect(checkDoneValue('\ud800a', 9)).toEqual({ ok: true, bytes: 9 })
// A BMP 3-byte code point (CJK) meters at its raw UTF-8 width: 2 quotes + 3.
expect(checkDoneValue('中', 5)).toEqual({ ok: true, bytes: 5 })
// Same non-allocating meter for object keys, before the value is enqueued.
expect(checkDoneValue({ ['\0'.repeat(200)]: 1 }, 1024)).toEqual({ ok: false, reason: 'over-budget' })
// A string reached with less than the two quotes' worth of budget is refused
// immediately (even the empty escaped form does not fit).
expect(checkDoneValue('x', 1)).toEqual({ ok: false, reason: 'over-budget' })
})
it('meters only own enumerable keys', () => {