From 94b4116a102488a94c9a45d3f269346e3bfefe22 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:38:02 -0700 Subject: [PATCH] Bound AI Vault cache loading and keep atomic saves responsive (#24789) * Bound AI Vault cache loading and cooperative atomic saves Preserve schema 3 caches across compatible releases while limiting bytes, JSON structure, and newest unique rows. Keep in-process entries authoritative and retain a valid prior snapshot when the newest row cannot fit. Credits @AmethystLiang for the original PR10708 cache bounds and cooperative persistence intent. * Use checked cache JSON properties in cooperative serialization Preserves lazy own-property access and all serializer bounds, yields and errors. --- .../session-parse-cache-bounds.test.ts | 192 ++++++++++++++++ .../session-parse-cache-persistence.ts | 86 +++++--- ...sion-parse-cache-snapshot-serialization.ts | 206 ++++++++++++++++++ .../ai-vault/session-parse-cache-store.ts | 27 ++- src/shared/json-text-structure-limit.ts | 114 ++++++---- 5 files changed, 544 insertions(+), 81 deletions(-) create mode 100644 src/main/ai-vault/session-parse-cache-bounds.test.ts create mode 100644 src/main/ai-vault/session-parse-cache-snapshot-serialization.ts diff --git a/src/main/ai-vault/session-parse-cache-bounds.test.ts b/src/main/ai-vault/session-parse-cache-bounds.test.ts new file mode 100644 index 00000000000..6ad56a3e4f5 --- /dev/null +++ b/src/main/ai-vault/session-parse-cache-bounds.test.ts @@ -0,0 +1,192 @@ +import { mkdtemp, open, readFile, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import type { AiVaultSession } from '../../shared/ai-vault-types' +import { + assertJsonTextStructureWithinLimits, + JsonTextStructureValidator +} from '../../shared/json-text-structure-limit' +import { + ensureSessionParseCacheLoaded, + flushSessionParseCachePersist, + initSessionParseCachePersistence, + resetSessionParseCachePersistenceForTests, + scheduleSessionParseCachePersist +} from './session-parse-cache-persistence' +import { + getSessionParseCacheEntry, + resetSessionParseCacheForTests, + seedSessionParseCache, + snapshotSessionParseCacheForPersistence, + storeSessionParseCacheEntry, + type PersistedSessionParseCacheEntry +} from './session-parse-cache-store' +import { + SESSION_PARSE_CACHE_JSON_LIMITS, + SESSION_PARSE_CACHE_MAX_BYTES, + serializeSessionParseCacheSnapshotPiecesCooperatively +} from './session-parse-cache-snapshot-serialization' + +let root: string +let file: string +const entry = (mtimeMs: number): PersistedSessionParseCacheEntry => ({ + mtimeMs, + sizeBytes: null, + platform: 'darwin', + session: null +}) +const row = (path: string, mtimeMs: number): [string, PersistedSessionParseCacheEntry] => [ + path, + entry(mtimeMs) +] +const stats = { reused: 0, fullParses: 1, incremental: 0, earlyStopped: 0, bytesRead: 0 } + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-cache-bounds-')) + file = join(root, 'cache.json') + resetSessionParseCacheForTests() + resetSessionParseCachePersistenceForTests() + initSessionParseCachePersistence({ filePath: file, appVersion: 'new-release' }) +}) +afterEach(async () => { + resetSessionParseCachePersistenceForTests() + await rm(root, { recursive: true, force: true }) +}) + +it('loads the newest 4096 unique rows across duplicate-heavy input and keeps in-process entries', async () => { + const entries = Array.from({ length: 4200 }, (_, index) => row(`p${index}`, index)) + entries.push(...Array.from({ length: 5000 }, (_, index) => row('p4199', 5000 + index))) + entries.push(row('in-process', 1)) + seedSessionParseCache([row('in-process', 20_000)]) + await writeFile(file, JSON.stringify({ schemaVersion: 3, appVersion: 'old-release', entries })) + await ensureSessionParseCacheLoaded() + expect(snapshotSessionParseCacheForPersistence()).toHaveLength(4096) + expect(getSessionParseCacheEntry('p4199')?.mtimeMs).toBe(9999) + expect(getSessionParseCacheEntry('p104')).toBeUndefined() + expect(getSessionParseCacheEntry('p105')?.mtimeMs).toBe(105) + expect(getSessionParseCacheEntry('in-process')?.mtimeMs).toBe(20_000) +}) + +it('seeds newest unique rows directly without retaining an unbounded iterable', () => { + function* entries() { + for (let index = 0; index < 5000; index++) { + yield row(`p${index}`, index) + } + yield row('p4999', 6000) + } + seedSessionParseCache(entries()) + expect(snapshotSessionParseCacheForPersistence()).toHaveLength(4096) + expect(getSessionParseCacheEntry('p903')).toBeUndefined() + expect(getSessionParseCacheEntry('p904')?.mtimeMs).toBe(904) + expect(getSessionParseCacheEntry('p4999')?.mtimeMs).toBe(6000) +}) + +it('rejects an oversized sparse file before decoding and preserves resident work', async () => { + seedSessionParseCache([row('resident', 3)]) + const handle = await open(file, 'w') + await handle.truncate(SESSION_PARSE_CACHE_MAX_BYTES + 1) + await handle.close() + await ensureSessionParseCacheLoaded() + expect(snapshotSessionParseCacheForPersistence().map(([path]) => path)).toEqual(['resident']) +}) + +it('rejects excessive tokens, depth, and malformed older rows before seeding any tail', async () => { + for (const raw of [ + `{"schemaVersion":3,"appVersion":"old","entries":[],"wide":[${'0,'.repeat(1_000_000)}0]}`, + `{"schemaVersion":3,"appVersion":"old","entries":[],"deep":${'['.repeat(33)}0${']'.repeat(33)}}`, + JSON.stringify({ + schemaVersion: 3, + appVersion: 'old', + entries: [['bad', {}], ...Array.from({ length: 4200 }, (_, index) => row(`p${index}`, index))] + }) + ]) { + await writeFile(file, raw) + resetSessionParseCachePersistenceForTests() + initSessionParseCachePersistence({ filePath: file, appVersion: 'new' }) + await ensureSessionParseCacheLoaded() + expect(snapshotSessionParseCacheForPersistence()).toEqual([]) + } +}) + +it('retains a valid newest suffix under both byte and structural capacity', async () => { + const entries = Array.from({ length: 40 }, (_, index) => row(`p${index}`, index)) + const snapshot = await serializeSessionParseCacheSnapshotPiecesCooperatively( + entries, + 'release', + 700, + { structuralTokens: 120, nestingDepth: 32 } + ) + expect(snapshot).not.toBeNull() + const serialized = snapshot!.pieces.join('') + expect(Buffer.byteLength(serialized)).toBeLessThanOrEqual(700) + assertJsonTextStructureWithinLimits(serialized, { structuralTokens: 120, nestingDepth: 32 }) + expect(JSON.parse(serialized).entries).toEqual(entries.slice(-snapshot!.retainedEntries)) + expect(snapshot!.retainedEntries).toBeGreaterThan(0) + expect(snapshot!.retainedEntries).toBeLessThan(entries.length) +}) + +it('keeps the previous atomic snapshot when its newest row cannot fit', async () => { + seedSessionParseCache([row('valid', 1)]) + scheduleSessionParseCachePersist(stats) + await flushSessionParseCachePersist() + const previous = await readFile(file, 'utf8') + const session: AiVaultSession = { + id: 'local:claude:newest', + executionHostId: 'local', + agent: 'claude', + sessionId: 'newest', + title: 'newest', + cwd: null, + branch: null, + model: null, + filePath: 'newest', + codexHome: null, + createdAt: null, + updatedAt: null, + modifiedAt: new Date(0).toISOString(), + messageCount: 1, + totalTokens: 0, + previewMessages: [], + queuedMessageCount: 0, + subagentTranscriptCount: 0, + resumeCommand: '', + subagent: null + } + let deep: unknown = null + for (let index = 0; index < 35; index++) { + deep = { nested: deep } + } + Reflect.set(session, 'syntheticDeepField', deep) + storeSessionParseCacheEntry('newest', { ...entry(2), session, resume: null }) + scheduleSessionParseCachePersist(stats) + await flushSessionParseCachePersist() + expect(await readFile(file, 'utf8')).toBe(previous) +}) + +it('preserves escape state at every chunk boundary while ignoring punctuation inside strings', () => { + const content = JSON.stringify({ + nested: ['[\\\\\"{}:,]', '\\', 'end\\', { value: 'x'.repeat(300_000) }] + }) + for (const size of [1, 2, 3, 17, 256 * 1024]) { + const validator = new JsonTextStructureValidator(SESSION_PARSE_CACHE_JSON_LIMITS) + for (let start = 0; start < content.length; start += size) { + validator.consume(content.slice(start, start + size)) + } + expect(validator.usage()).toEqual({ structuralTokens: 11, nestingDepth: 3 }) + } +}) + +it('yields while escaping a large Unicode string and preserves surrogate pairs', async () => { + const path = '😀"\\\n'.repeat(100_000) + let progressed = false + setImmediate(() => { + progressed = true + }) + const snapshot = await serializeSessionParseCacheSnapshotPiecesCooperatively( + [row(path, 1)], + 'release' + ) + expect(progressed).toBe(true) + expect(JSON.parse(snapshot!.pieces.join('')).entries).toEqual([row(path, 1)]) +}) diff --git a/src/main/ai-vault/session-parse-cache-persistence.ts b/src/main/ai-vault/session-parse-cache-persistence.ts index 7e7475e6a5b..59008259154 100644 --- a/src/main/ai-vault/session-parse-cache-persistence.ts +++ b/src/main/ai-vault/session-parse-cache-persistence.ts @@ -3,8 +3,17 @@ // of re-reading the whole transcript corpus (issue #9210: 6.7 GB / 109 s cold // scans). Disabled unless the composition root calls init; every failure mode // degrades to today's cold-scan behavior. -import { mkdir, readdir, readFile, rename, rm, writeFile } from 'node:fs/promises' +import { mkdir, readdir, rename, rm, writeFile } from 'node:fs/promises' import { dirname, join } from 'node:path' +import { readNodeFileWithinLimit } from '../../shared/node-bounded-file-reader' +import { readStreamedSessionDocument } from './session-document-stream' +import { MAX_CACHE_ENTRIES } from './session-parse-cache-store' +import { + assertSessionParseCacheJsonWithinLimitsCooperatively, + serializeSessionParseCacheSnapshotPiecesCooperatively, + SESSION_PARSE_CACHE_SCHEMA_VERSION, + SESSION_PARSE_CACHE_MAX_BYTES +} from './session-parse-cache-snapshot-serialization' import { seedSessionParseCache, snapshotSessionParseCacheForPersistence, @@ -15,7 +24,7 @@ import type { SessionSidecarObservation } from './session-sidecar-stat' // Bump when the persisted entry layout or cached session semantics change; a // mismatched file is discarded whole. -const SCHEMA_VERSION = 3 +const SCHEMA_VERSION = SESSION_PARSE_CACHE_SCHEMA_VERSION // Debounce so back-to-back scans (desktop IPC + runtime RPC) collapse into one write. const SAVE_DEBOUNCE_MS = 1_500 // The payload contains transcript-derived preview text; keep it user-only @@ -108,8 +117,12 @@ export const flushSessionParseCachePersistForTests = flushSessionParseCachePersi async function loadPersistedEntries(current: SessionParseCachePersistenceOptions): Promise { await sweepOrphanedTempFiles(current.filePath) try { - const raw = await readFile(current.filePath, 'utf-8') - const entries = parsePersistedFile(JSON.parse(raw)) + const { buffer } = await readNodeFileWithinLimit( + current.filePath, + SESSION_PARSE_CACHE_MAX_BYTES + ) + await assertSessionParseCacheJsonWithinLimitsCooperatively(buffer) + const entries = await parsePersistedFile(buffer) if (entries) { seedSessionParseCache(entries) } @@ -136,29 +149,44 @@ async function sweepOrphanedTempFiles(filePath: string): Promise { } } -function parsePersistedFile(parsed: unknown): [string, PersistedSessionParseCacheEntry][] | null { - if (typeof parsed !== 'object' || parsed === null) { - return null - } - const file = parsed as Record +async function parsePersistedFile( + buffer: Buffer +): Promise<[string, PersistedSessionParseCacheEntry][] | null> { + const parsed = await readStreamedSessionDocument({ + bytes: cacheFileChunks(buffer), + arrayKey: 'entries', + fields: ['schemaVersion', 'appVersion'], + create: () => new Map(), + consume(entries, item) { + const entry = parsePersistedEntry(item) + if (entry === null) { + throw new Error('Malformed session parse cache row') + } + entries.delete(entry[0]) + entries.set(entry[0], entry[1]) + if (entries.size > MAX_CACHE_ENTRIES) { + const oldest = entries.keys().next() + if (!oldest.done) { + entries.delete(oldest.value) + } + } + } + }) // Why: application releases that keep this schema promise compatible cached // session semantics, so an update does not force a multi-gigabyte cold scan. - if (file.schemaVersion !== SCHEMA_VERSION || typeof file.appVersion !== 'string') { + if ( + parsed?.record.schemaVersion !== SCHEMA_VERSION || + typeof parsed.record.appVersion !== 'string' + ) { return null } - if (!Array.isArray(file.entries)) { - return null + return [...parsed.state] +} + +async function* cacheFileChunks(buffer: Buffer): AsyncGenerator { + for (let start = 0; start < buffer.length; start += 64 * 1024) { + yield buffer.subarray(start, start + 64 * 1024) } - const entries: [string, PersistedSessionParseCacheEntry][] = [] - for (const item of file.entries) { - const entry = parsePersistedEntry(item) - if (entry === null) { - // One malformed entry means the file can't be trusted; discard it whole. - return null - } - entries.push(entry) - } - return entries } function parsePersistedEntry(item: unknown): [string, PersistedSessionParseCacheEntry] | null { @@ -217,13 +245,15 @@ async function persistSnapshot(current: SessionParseCachePersistenceOptions): Pr const directory = dirname(current.filePath) const tempPath = join(directory, `session-parse-cache-${process.pid}-${Date.now()}.tmp`) try { - const payload = JSON.stringify({ - schemaVersion: SCHEMA_VERSION, - appVersion: current.appVersion, - entries: snapshotSessionParseCacheForPersistence() - }) + const payload = await serializeSessionParseCacheSnapshotPiecesCooperatively( + snapshotSessionParseCacheForPersistence(), + current.appVersion + ) + if (payload === null) { + return + } await mkdir(directory, { recursive: true, mode: PRIVATE_DIRECTORY_MODE }) - await writeFile(tempPath, payload, { mode: PRIVATE_FILE_MODE }) + await writeFile(tempPath, payload.pieces, { mode: PRIVATE_FILE_MODE }) // Atomic on POSIX; on Windows a rename racing an open handle fails and is // caught below (save lost, never a torn file). await rename(tempPath, current.filePath) diff --git a/src/main/ai-vault/session-parse-cache-snapshot-serialization.ts b/src/main/ai-vault/session-parse-cache-snapshot-serialization.ts new file mode 100644 index 00000000000..9900dcfa8de --- /dev/null +++ b/src/main/ai-vault/session-parse-cache-snapshot-serialization.ts @@ -0,0 +1,206 @@ +import { setImmediate as yieldToEventLoop } from 'node:timers/promises' +import { StringDecoder } from 'node:string_decoder' +import { + JsonTextStructureCapacityError, + JsonTextStructureValidator, + type JsonTextStructureLimits +} from '../../shared/json-text-structure-limit' +import { + JsonStringifyByteLimitError, + stringifyJsonWithinByteLimit +} from '../../shared/node-bounded-json-stringify' +import type { PersistedSessionParseCacheEntry } from './session-parse-cache-store' + +export const SESSION_PARSE_CACHE_SCHEMA_VERSION = 3 +export const SESSION_PARSE_CACHE_MAX_BYTES = 64 * 1024 * 1024 +export const SESSION_PARSE_CACHE_JSON_LIMITS = { + structuralTokens: 1_000_000, + nestingDepth: 32 +} as const + +const TEXT_CHUNK_CHARACTERS = 16 * 1024 +const VALIDATE_CHUNK_CHARACTERS = 256 * 1024 +const SERIALIZE_YIELD_STEPS = 1024 +type CacheEntry = [string, PersistedSessionParseCacheEntry] +type JsonPiece = { text: string; tokens: number } +type CacheJsonObject = Record + +function isCacheJsonObject(value: unknown): value is CacheJsonObject { + return typeof value === 'object' && value !== null +} + +export async function assertSessionParseCacheJsonWithinLimitsCooperatively( + content: string | Buffer, + limits: JsonTextStructureLimits = SESSION_PARSE_CACHE_JSON_LIMITS +): Promise { + const validator = new JsonTextStructureValidator(limits) + const decoder = new StringDecoder('utf8') + for (let start = 0; start < content.length; start += VALIDATE_CHUNK_CHARACTERS) { + const chunk = content.slice(start, start + VALIDATE_CHUNK_CHARACTERS) + validator.consume(typeof chunk === 'string' ? chunk : decoder.write(chunk)) + if (start + VALIDATE_CHUNK_CHARACTERS < content.length) { + await yieldToEventLoop() + } + } + validator.consume(decoder.end()) +} + +/** Retain the newest complete suffix; a newest row that cannot fit preserves the prior file. */ +export async function serializeSessionParseCacheSnapshotPiecesCooperatively( + entries: readonly CacheEntry[], + appVersion: string, + maxBytes = SESSION_PARSE_CACHE_MAX_BYTES, + limits: JsonTextStructureLimits = SESSION_PARSE_CACHE_JSON_LIMITS +): Promise<{ pieces: string[]; byteLength: number; retainedEntries: number } | null> { + const header = stringifyJsonWithinByteLimit( + { schemaVersion: SESSION_PARSE_CACHE_SCHEMA_VERSION, appVersion, entries: [] }, + maxBytes + ).serialized + const outer = new JsonTextStructureValidator(limits) + outer.consume(header) + let byteLength = Buffer.byteLength(header) + let tokens = outer.usage().structuralTokens + const rows: string[][] = [] + for (let index = entries.length - 1; index >= 0; index--) { + const comma = rows.length === 0 ? 0 : 1 + try { + const row = await serializeCacheRow(entries[index], maxBytes - byteLength - comma, { + structuralTokens: limits.structuralTokens - tokens - comma, + nestingDepth: limits.nestingDepth - 2 + }) + rows.push(row.pieces) + byteLength += row.byteLength + comma + tokens += row.tokens + comma + } catch (error) { + if ( + !( + error instanceof JsonStringifyByteLimitError || + error instanceof JsonTextStructureCapacityError + ) + ) { + throw error + } + break + } + if (rows.length % 16 === 0) { + await yieldToEventLoop() + } + } + if (entries.length > 0 && rows.length === 0) { + return null + } + const pieces = [header.slice(0, -2)] + for (let index = rows.length - 1; index >= 0; index--) { + if (index < rows.length - 1) { + pieces.push(',') + } + pieces.push(...rows[index]!) + } + pieces.push(']}') + return { pieces, byteLength, retainedEntries: rows.length } +} + +async function serializeCacheRow( + value: unknown, + maxBytes: number, + limits: JsonTextStructureLimits +) { + const pieces: string[] = [] + let block = '' + let byteLength = 0 + let tokens = 0 + let steps = 0 + for (const piece of cacheJsonPieces(value, 0, limits.nestingDepth, new Set())) { + byteLength += Buffer.byteLength(piece.text) + tokens += piece.tokens + if (byteLength > maxBytes) { + throw new JsonStringifyByteLimitError(byteLength, maxBytes) + } + if (tokens > limits.structuralTokens) { + throw new JsonTextStructureCapacityError('structuralTokens', limits.structuralTokens) + } + block += piece.text + if (block.length >= TEXT_CHUNK_CHARACTERS) { + pieces.push(block) + block = '' + } + // Large strings and wide objects both yield, including omitted properties. + if (++steps % SERIALIZE_YIELD_STEPS === 0 || piece.text.length >= TEXT_CHUNK_CHARACTERS / 2) { + await yieldToEventLoop() + } + } + if (block) { + pieces.push(block) + } + return { pieces, byteLength, tokens } +} + +// Native encoding allocates at most six bytes per character of a bounded string chunk. +function* cacheJsonPieces( + value: unknown, + depth: number, + maxDepth: number, + ancestors: Set +): Generator { + if (typeof value === 'string') { + yield { text: '"', tokens: 0 } + for (let start = 0; start < value.length;) { + let end = Math.min(value.length, start + TEXT_CHUNK_CHARACTERS) + const last = value.charCodeAt(end - 1) + if (end < value.length && last >= 0xd800 && last <= 0xdbff) { + end-- + } + const serialized = JSON.stringify(value.slice(start, end)) + yield { text: serialized.slice(1, -1), tokens: 0 } + start = end + } + yield { text: '"', tokens: 0 } + return + } + if (!isCacheJsonObject(value)) { + const serialized = JSON.stringify(value) + if (serialized === undefined) { + throw new TypeError('Session parse cache value is not serializable') + } + yield { text: serialized, tokens: 0 } + return + } + if (depth + 1 > maxDepth) { + throw new JsonTextStructureCapacityError('nestingDepth', maxDepth) + } + if (ancestors.has(value)) { + throw new TypeError('Circular session parse cache row') + } + ancestors.add(value) + let count = 0 + if (Array.isArray(value)) { + yield { text: '[', tokens: 1 } + for (const item of value) { + if (count++) { + yield { text: ',', tokens: 1 } + } + yield* cacheJsonPieces(item === undefined ? null : item, depth + 1, maxDepth, ancestors) + } + yield { text: ']', tokens: 1 } + } else { + yield { text: '{', tokens: 1 } + for (const key in value) { + if (!Object.hasOwn(value, key)) { + continue + } + const item: unknown = value[key] + if (item === undefined) { + yield { text: '', tokens: 0 } + continue + } + if (count++) { + yield { text: ',', tokens: 1 } + } + yield* cacheJsonPieces(key, depth + 1, maxDepth, ancestors) + yield { text: ':', tokens: 1 } + yield* cacheJsonPieces(item, depth + 1, maxDepth, ancestors) + } + yield { text: '}', tokens: 1 } + } + ancestors.delete(value) +} diff --git a/src/main/ai-vault/session-parse-cache-store.ts b/src/main/ai-vault/session-parse-cache-store.ts index 569a8b94d5d..f08dc4917b7 100644 --- a/src/main/ai-vault/session-parse-cache-store.ts +++ b/src/main/ai-vault/session-parse-cache-store.ts @@ -6,7 +6,7 @@ import type { SkippedTranscriptRecord } from './session-transcript-record-budget // Sized past the default recency cap (1000) plus the in-scope cap (2000) so a // full steady-state result set stays resident between forced rescans. -const MAX_CACHE_ENTRIES = 4096 +export const MAX_CACHE_ENTRIES = 4096 export type SessionParseResumePoint = { state: ResumableSessionParseState @@ -77,17 +77,26 @@ export function snapshotSessionParseCacheForPersistence(): [ export function seedSessionParseCache( entries: Iterable<[string, PersistedSessionParseCacheEntry]> ): void { - const list = [...entries] - // Snapshot order is oldest→newest (LRU); an over-cap list keeps the newest - // tail rather than seeding the oldest entries and dropping the tail. - for (const [path, entry] of list.slice(Math.max(0, list.length - MAX_CACHE_ENTRIES))) { - if (cache.size >= MAX_CACHE_ENTRIES) { - return - } - // In-process entries are always fresher than persisted ones; never clobber. + const newest = new Map() + for (const [path, entry] of entries) { if (cache.has(path)) { continue } + newest.delete(path) + newest.set(path, entry) + if (newest.size > MAX_CACHE_ENTRIES) { + const oldest = newest.keys().next() + if (!oldest.done) { + newest.delete(oldest.value) + } + } + } + const available = MAX_CACHE_ENTRIES - cache.size + if (available <= 0) { + return + } + // In-process entries win; among persisted duplicates the newest row wins. + for (const [path, entry] of [...newest].slice(-available)) { cache.set(path, { mtimeMs: entry.mtimeMs, sizeBytes: entry.sizeBytes, diff --git a/src/shared/json-text-structure-limit.ts b/src/shared/json-text-structure-limit.ts index f33ecd1e2ca..22a778d3fa4 100644 --- a/src/shared/json-text-structure-limit.ts +++ b/src/shared/json-text-structure-limit.ts @@ -21,59 +21,85 @@ export function assertJsonTextStructureWithinLimits( content: string, limits: JsonTextStructureLimits ): void { - assertLimit(limits.structuralTokens) - assertLimit(limits.nestingDepth) - let structuralTokens = 0 - let depth = 0 - for (let index = 0; index < content.length; index += 1) { - const character = content[index] - if (character === '"') { - let quote = content.indexOf('"', index + 1) - if (quote !== -1) { - // Only an odd backslash run escapes the quote. - let backslashes = 0 - for (let at = quote - 1; at > index && content[at] === '\\'; at -= 1) { - backslashes += 1 + new JsonTextStructureValidator(limits).consume(content) +} + +/** Carries string/escape and structure state across bounded chunks. */ +export class JsonTextStructureValidator { + private structuralTokens = 0 + private depth = 0 + private maximumDepth = 0 + private inString = false + private escaped = false + + constructor(private readonly limits: JsonTextStructureLimits) { + assertLimit(limits.structuralTokens) + assertLimit(limits.nestingDepth) + } + + consume(content: string): void { + let linearString = false + for (let index = 0; index < content.length; index += 1) { + const character = content[index] + if (this.inString) { + if (this.escaped) { + this.escaped = false + continue } - if (backslashes % 2 !== 0) { - // Escape-heavy strings use the linear scan to avoid repeated native searches. - let escaped = false - for (quote += 1; quote < content.length; quote += 1) { - if (escaped) { - escaped = false - } else if (content[quote] === '\\') { - escaped = true - } else if (content[quote] === '"') { - break - } + if (!linearString) { + const quote = content.indexOf('"', index) + const end = quote === -1 ? content.length : quote + let backslashes = 0 + for (let at = end - 1; at >= index && content[at] === '\\'; at -= 1) { + backslashes++ } - if (quote === content.length) { - quote = -1 + if (quote === -1) { + this.escaped = backslashes % 2 !== 0 + return } + index = quote + if (backslashes % 2 === 0) { + this.inString = false + } else { + // Escape-heavy strings scan linearly instead of repeating native searches. + linearString = true + } + continue } + if (character === '\\') { + this.escaped = true + } else if (character === '"') { + this.inString = false + linearString = false + } + continue } - if (quote === -1) { - return + if (character === '"') { + this.inString = true + continue } - index = quote - continue - } - if (!isStructuralToken(character)) { - continue - } - structuralTokens += 1 - if (structuralTokens > limits.structuralTokens) { - throw new JsonTextStructureCapacityError('structuralTokens', limits.structuralTokens) - } - if (character === '{' || character === '[') { - depth += 1 - if (depth > limits.nestingDepth) { - throw new JsonTextStructureCapacityError('nestingDepth', limits.nestingDepth) + if (!isStructuralToken(character)) { + continue + } + this.structuralTokens++ + if (this.structuralTokens > this.limits.structuralTokens) { + throw new JsonTextStructureCapacityError('structuralTokens', this.limits.structuralTokens) + } + if (character === '{' || character === '[') { + this.depth++ + this.maximumDepth = Math.max(this.maximumDepth, this.depth) + if (this.depth > this.limits.nestingDepth) { + throw new JsonTextStructureCapacityError('nestingDepth', this.limits.nestingDepth) + } + } else if (character === '}' || character === ']') { + this.depth = Math.max(0, this.depth - 1) } - } else if (character === '}' || character === ']') { - depth = Math.max(0, depth - 1) } } + + usage(): { structuralTokens: number; nestingDepth: number } { + return { structuralTokens: this.structuralTokens, nestingDepth: this.maximumDepth } + } } function assertLimit(value: number): void {