mirror of
https://github.com/stablyai/orca.git
synced 2026-10-07 00:02:29 +00:00
Bound AI Vault cache loading and keep atomic saves responsive (#24789)
* Bound AI Vault cache loading and cooperative atomic saves Preserve schema 3 caches across compatible releases while limiting bytes, JSON structure, and newest unique rows. Keep in-process entries authoritative and retain a valid prior snapshot when the newest row cannot fit. Credits @AmethystLiang for the original PR10708 cache bounds and cooperative persistence intent. * Use checked cache JSON properties in cooperative serialization Preserves lazy own-property access and all serializer bounds, yields and errors.
This commit is contained in:
@@ -0,0 +1,192 @@
|
||||
import { mkdtemp, open, readFile, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, beforeEach, expect, it } from 'vitest'
|
||||
import type { AiVaultSession } from '../../shared/ai-vault-types'
|
||||
import {
|
||||
assertJsonTextStructureWithinLimits,
|
||||
JsonTextStructureValidator
|
||||
} from '../../shared/json-text-structure-limit'
|
||||
import {
|
||||
ensureSessionParseCacheLoaded,
|
||||
flushSessionParseCachePersist,
|
||||
initSessionParseCachePersistence,
|
||||
resetSessionParseCachePersistenceForTests,
|
||||
scheduleSessionParseCachePersist
|
||||
} from './session-parse-cache-persistence'
|
||||
import {
|
||||
getSessionParseCacheEntry,
|
||||
resetSessionParseCacheForTests,
|
||||
seedSessionParseCache,
|
||||
snapshotSessionParseCacheForPersistence,
|
||||
storeSessionParseCacheEntry,
|
||||
type PersistedSessionParseCacheEntry
|
||||
} from './session-parse-cache-store'
|
||||
import {
|
||||
SESSION_PARSE_CACHE_JSON_LIMITS,
|
||||
SESSION_PARSE_CACHE_MAX_BYTES,
|
||||
serializeSessionParseCacheSnapshotPiecesCooperatively
|
||||
} from './session-parse-cache-snapshot-serialization'
|
||||
|
||||
let root: string
|
||||
let file: string
|
||||
const entry = (mtimeMs: number): PersistedSessionParseCacheEntry => ({
|
||||
mtimeMs,
|
||||
sizeBytes: null,
|
||||
platform: 'darwin',
|
||||
session: null
|
||||
})
|
||||
const row = (path: string, mtimeMs: number): [string, PersistedSessionParseCacheEntry] => [
|
||||
path,
|
||||
entry(mtimeMs)
|
||||
]
|
||||
const stats = { reused: 0, fullParses: 1, incremental: 0, earlyStopped: 0, bytesRead: 0 }
|
||||
|
||||
beforeEach(async () => {
|
||||
root = await mkdtemp(join(tmpdir(), 'orca-cache-bounds-'))
|
||||
file = join(root, 'cache.json')
|
||||
resetSessionParseCacheForTests()
|
||||
resetSessionParseCachePersistenceForTests()
|
||||
initSessionParseCachePersistence({ filePath: file, appVersion: 'new-release' })
|
||||
})
|
||||
afterEach(async () => {
|
||||
resetSessionParseCachePersistenceForTests()
|
||||
await rm(root, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
it('loads the newest 4096 unique rows across duplicate-heavy input and keeps in-process entries', async () => {
|
||||
const entries = Array.from({ length: 4200 }, (_, index) => row(`p${index}`, index))
|
||||
entries.push(...Array.from({ length: 5000 }, (_, index) => row('p4199', 5000 + index)))
|
||||
entries.push(row('in-process', 1))
|
||||
seedSessionParseCache([row('in-process', 20_000)])
|
||||
await writeFile(file, JSON.stringify({ schemaVersion: 3, appVersion: 'old-release', entries }))
|
||||
await ensureSessionParseCacheLoaded()
|
||||
expect(snapshotSessionParseCacheForPersistence()).toHaveLength(4096)
|
||||
expect(getSessionParseCacheEntry('p4199')?.mtimeMs).toBe(9999)
|
||||
expect(getSessionParseCacheEntry('p104')).toBeUndefined()
|
||||
expect(getSessionParseCacheEntry('p105')?.mtimeMs).toBe(105)
|
||||
expect(getSessionParseCacheEntry('in-process')?.mtimeMs).toBe(20_000)
|
||||
})
|
||||
|
||||
it('seeds newest unique rows directly without retaining an unbounded iterable', () => {
|
||||
function* entries() {
|
||||
for (let index = 0; index < 5000; index++) {
|
||||
yield row(`p${index}`, index)
|
||||
}
|
||||
yield row('p4999', 6000)
|
||||
}
|
||||
seedSessionParseCache(entries())
|
||||
expect(snapshotSessionParseCacheForPersistence()).toHaveLength(4096)
|
||||
expect(getSessionParseCacheEntry('p903')).toBeUndefined()
|
||||
expect(getSessionParseCacheEntry('p904')?.mtimeMs).toBe(904)
|
||||
expect(getSessionParseCacheEntry('p4999')?.mtimeMs).toBe(6000)
|
||||
})
|
||||
|
||||
it('rejects an oversized sparse file before decoding and preserves resident work', async () => {
|
||||
seedSessionParseCache([row('resident', 3)])
|
||||
const handle = await open(file, 'w')
|
||||
await handle.truncate(SESSION_PARSE_CACHE_MAX_BYTES + 1)
|
||||
await handle.close()
|
||||
await ensureSessionParseCacheLoaded()
|
||||
expect(snapshotSessionParseCacheForPersistence().map(([path]) => path)).toEqual(['resident'])
|
||||
})
|
||||
|
||||
it('rejects excessive tokens, depth, and malformed older rows before seeding any tail', async () => {
|
||||
for (const raw of [
|
||||
`{"schemaVersion":3,"appVersion":"old","entries":[],"wide":[${'0,'.repeat(1_000_000)}0]}`,
|
||||
`{"schemaVersion":3,"appVersion":"old","entries":[],"deep":${'['.repeat(33)}0${']'.repeat(33)}}`,
|
||||
JSON.stringify({
|
||||
schemaVersion: 3,
|
||||
appVersion: 'old',
|
||||
entries: [['bad', {}], ...Array.from({ length: 4200 }, (_, index) => row(`p${index}`, index))]
|
||||
})
|
||||
]) {
|
||||
await writeFile(file, raw)
|
||||
resetSessionParseCachePersistenceForTests()
|
||||
initSessionParseCachePersistence({ filePath: file, appVersion: 'new' })
|
||||
await ensureSessionParseCacheLoaded()
|
||||
expect(snapshotSessionParseCacheForPersistence()).toEqual([])
|
||||
}
|
||||
})
|
||||
|
||||
it('retains a valid newest suffix under both byte and structural capacity', async () => {
|
||||
const entries = Array.from({ length: 40 }, (_, index) => row(`p${index}`, index))
|
||||
const snapshot = await serializeSessionParseCacheSnapshotPiecesCooperatively(
|
||||
entries,
|
||||
'release',
|
||||
700,
|
||||
{ structuralTokens: 120, nestingDepth: 32 }
|
||||
)
|
||||
expect(snapshot).not.toBeNull()
|
||||
const serialized = snapshot!.pieces.join('')
|
||||
expect(Buffer.byteLength(serialized)).toBeLessThanOrEqual(700)
|
||||
assertJsonTextStructureWithinLimits(serialized, { structuralTokens: 120, nestingDepth: 32 })
|
||||
expect(JSON.parse(serialized).entries).toEqual(entries.slice(-snapshot!.retainedEntries))
|
||||
expect(snapshot!.retainedEntries).toBeGreaterThan(0)
|
||||
expect(snapshot!.retainedEntries).toBeLessThan(entries.length)
|
||||
})
|
||||
|
||||
it('keeps the previous atomic snapshot when its newest row cannot fit', async () => {
|
||||
seedSessionParseCache([row('valid', 1)])
|
||||
scheduleSessionParseCachePersist(stats)
|
||||
await flushSessionParseCachePersist()
|
||||
const previous = await readFile(file, 'utf8')
|
||||
const session: AiVaultSession = {
|
||||
id: 'local:claude:newest',
|
||||
executionHostId: 'local',
|
||||
agent: 'claude',
|
||||
sessionId: 'newest',
|
||||
title: 'newest',
|
||||
cwd: null,
|
||||
branch: null,
|
||||
model: null,
|
||||
filePath: 'newest',
|
||||
codexHome: null,
|
||||
createdAt: null,
|
||||
updatedAt: null,
|
||||
modifiedAt: new Date(0).toISOString(),
|
||||
messageCount: 1,
|
||||
totalTokens: 0,
|
||||
previewMessages: [],
|
||||
queuedMessageCount: 0,
|
||||
subagentTranscriptCount: 0,
|
||||
resumeCommand: '',
|
||||
subagent: null
|
||||
}
|
||||
let deep: unknown = null
|
||||
for (let index = 0; index < 35; index++) {
|
||||
deep = { nested: deep }
|
||||
}
|
||||
Reflect.set(session, 'syntheticDeepField', deep)
|
||||
storeSessionParseCacheEntry('newest', { ...entry(2), session, resume: null })
|
||||
scheduleSessionParseCachePersist(stats)
|
||||
await flushSessionParseCachePersist()
|
||||
expect(await readFile(file, 'utf8')).toBe(previous)
|
||||
})
|
||||
|
||||
it('preserves escape state at every chunk boundary while ignoring punctuation inside strings', () => {
|
||||
const content = JSON.stringify({
|
||||
nested: ['[\\\\\"{}:,]', '\\', 'end\\', { value: 'x'.repeat(300_000) }]
|
||||
})
|
||||
for (const size of [1, 2, 3, 17, 256 * 1024]) {
|
||||
const validator = new JsonTextStructureValidator(SESSION_PARSE_CACHE_JSON_LIMITS)
|
||||
for (let start = 0; start < content.length; start += size) {
|
||||
validator.consume(content.slice(start, start + size))
|
||||
}
|
||||
expect(validator.usage()).toEqual({ structuralTokens: 11, nestingDepth: 3 })
|
||||
}
|
||||
})
|
||||
|
||||
it('yields while escaping a large Unicode string and preserves surrogate pairs', async () => {
|
||||
const path = '😀"\\\n'.repeat(100_000)
|
||||
let progressed = false
|
||||
setImmediate(() => {
|
||||
progressed = true
|
||||
})
|
||||
const snapshot = await serializeSessionParseCacheSnapshotPiecesCooperatively(
|
||||
[row(path, 1)],
|
||||
'release'
|
||||
)
|
||||
expect(progressed).toBe(true)
|
||||
expect(JSON.parse(snapshot!.pieces.join('')).entries).toEqual([row(path, 1)])
|
||||
})
|
||||
@@ -3,8 +3,17 @@
|
||||
// of re-reading the whole transcript corpus (issue #9210: 6.7 GB / 109 s cold
|
||||
// scans). Disabled unless the composition root calls init; every failure mode
|
||||
// degrades to today's cold-scan behavior.
|
||||
import { mkdir, readdir, readFile, rename, rm, writeFile } from 'node:fs/promises'
|
||||
import { mkdir, readdir, rename, rm, writeFile } from 'node:fs/promises'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { readNodeFileWithinLimit } from '../../shared/node-bounded-file-reader'
|
||||
import { readStreamedSessionDocument } from './session-document-stream'
|
||||
import { MAX_CACHE_ENTRIES } from './session-parse-cache-store'
|
||||
import {
|
||||
assertSessionParseCacheJsonWithinLimitsCooperatively,
|
||||
serializeSessionParseCacheSnapshotPiecesCooperatively,
|
||||
SESSION_PARSE_CACHE_SCHEMA_VERSION,
|
||||
SESSION_PARSE_CACHE_MAX_BYTES
|
||||
} from './session-parse-cache-snapshot-serialization'
|
||||
import {
|
||||
seedSessionParseCache,
|
||||
snapshotSessionParseCacheForPersistence,
|
||||
@@ -15,7 +24,7 @@ import type { SessionSidecarObservation } from './session-sidecar-stat'
|
||||
|
||||
// Bump when the persisted entry layout or cached session semantics change; a
|
||||
// mismatched file is discarded whole.
|
||||
const SCHEMA_VERSION = 3
|
||||
const SCHEMA_VERSION = SESSION_PARSE_CACHE_SCHEMA_VERSION
|
||||
// Debounce so back-to-back scans (desktop IPC + runtime RPC) collapse into one write.
|
||||
const SAVE_DEBOUNCE_MS = 1_500
|
||||
// The payload contains transcript-derived preview text; keep it user-only
|
||||
@@ -108,8 +117,12 @@ export const flushSessionParseCachePersistForTests = flushSessionParseCachePersi
|
||||
async function loadPersistedEntries(current: SessionParseCachePersistenceOptions): Promise<void> {
|
||||
await sweepOrphanedTempFiles(current.filePath)
|
||||
try {
|
||||
const raw = await readFile(current.filePath, 'utf-8')
|
||||
const entries = parsePersistedFile(JSON.parse(raw))
|
||||
const { buffer } = await readNodeFileWithinLimit(
|
||||
current.filePath,
|
||||
SESSION_PARSE_CACHE_MAX_BYTES
|
||||
)
|
||||
await assertSessionParseCacheJsonWithinLimitsCooperatively(buffer)
|
||||
const entries = await parsePersistedFile(buffer)
|
||||
if (entries) {
|
||||
seedSessionParseCache(entries)
|
||||
}
|
||||
@@ -136,29 +149,44 @@ async function sweepOrphanedTempFiles(filePath: string): Promise<void> {
|
||||
}
|
||||
}
|
||||
|
||||
function parsePersistedFile(parsed: unknown): [string, PersistedSessionParseCacheEntry][] | null {
|
||||
if (typeof parsed !== 'object' || parsed === null) {
|
||||
return null
|
||||
}
|
||||
const file = parsed as Record<string, unknown>
|
||||
async function parsePersistedFile(
|
||||
buffer: Buffer
|
||||
): Promise<[string, PersistedSessionParseCacheEntry][] | null> {
|
||||
const parsed = await readStreamedSessionDocument({
|
||||
bytes: cacheFileChunks(buffer),
|
||||
arrayKey: 'entries',
|
||||
fields: ['schemaVersion', 'appVersion'],
|
||||
create: () => new Map<string, PersistedSessionParseCacheEntry>(),
|
||||
consume(entries, item) {
|
||||
const entry = parsePersistedEntry(item)
|
||||
if (entry === null) {
|
||||
throw new Error('Malformed session parse cache row')
|
||||
}
|
||||
entries.delete(entry[0])
|
||||
entries.set(entry[0], entry[1])
|
||||
if (entries.size > MAX_CACHE_ENTRIES) {
|
||||
const oldest = entries.keys().next()
|
||||
if (!oldest.done) {
|
||||
entries.delete(oldest.value)
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
// Why: application releases that keep this schema promise compatible cached
|
||||
// session semantics, so an update does not force a multi-gigabyte cold scan.
|
||||
if (file.schemaVersion !== SCHEMA_VERSION || typeof file.appVersion !== 'string') {
|
||||
if (
|
||||
parsed?.record.schemaVersion !== SCHEMA_VERSION ||
|
||||
typeof parsed.record.appVersion !== 'string'
|
||||
) {
|
||||
return null
|
||||
}
|
||||
if (!Array.isArray(file.entries)) {
|
||||
return null
|
||||
return [...parsed.state]
|
||||
}
|
||||
|
||||
async function* cacheFileChunks(buffer: Buffer): AsyncGenerator<Buffer> {
|
||||
for (let start = 0; start < buffer.length; start += 64 * 1024) {
|
||||
yield buffer.subarray(start, start + 64 * 1024)
|
||||
}
|
||||
const entries: [string, PersistedSessionParseCacheEntry][] = []
|
||||
for (const item of file.entries) {
|
||||
const entry = parsePersistedEntry(item)
|
||||
if (entry === null) {
|
||||
// One malformed entry means the file can't be trusted; discard it whole.
|
||||
return null
|
||||
}
|
||||
entries.push(entry)
|
||||
}
|
||||
return entries
|
||||
}
|
||||
|
||||
function parsePersistedEntry(item: unknown): [string, PersistedSessionParseCacheEntry] | null {
|
||||
@@ -217,13 +245,15 @@ async function persistSnapshot(current: SessionParseCachePersistenceOptions): Pr
|
||||
const directory = dirname(current.filePath)
|
||||
const tempPath = join(directory, `session-parse-cache-${process.pid}-${Date.now()}.tmp`)
|
||||
try {
|
||||
const payload = JSON.stringify({
|
||||
schemaVersion: SCHEMA_VERSION,
|
||||
appVersion: current.appVersion,
|
||||
entries: snapshotSessionParseCacheForPersistence()
|
||||
})
|
||||
const payload = await serializeSessionParseCacheSnapshotPiecesCooperatively(
|
||||
snapshotSessionParseCacheForPersistence(),
|
||||
current.appVersion
|
||||
)
|
||||
if (payload === null) {
|
||||
return
|
||||
}
|
||||
await mkdir(directory, { recursive: true, mode: PRIVATE_DIRECTORY_MODE })
|
||||
await writeFile(tempPath, payload, { mode: PRIVATE_FILE_MODE })
|
||||
await writeFile(tempPath, payload.pieces, { mode: PRIVATE_FILE_MODE })
|
||||
// Atomic on POSIX; on Windows a rename racing an open handle fails and is
|
||||
// caught below (save lost, never a torn file).
|
||||
await rename(tempPath, current.filePath)
|
||||
|
||||
@@ -0,0 +1,206 @@
|
||||
import { setImmediate as yieldToEventLoop } from 'node:timers/promises'
|
||||
import { StringDecoder } from 'node:string_decoder'
|
||||
import {
|
||||
JsonTextStructureCapacityError,
|
||||
JsonTextStructureValidator,
|
||||
type JsonTextStructureLimits
|
||||
} from '../../shared/json-text-structure-limit'
|
||||
import {
|
||||
JsonStringifyByteLimitError,
|
||||
stringifyJsonWithinByteLimit
|
||||
} from '../../shared/node-bounded-json-stringify'
|
||||
import type { PersistedSessionParseCacheEntry } from './session-parse-cache-store'
|
||||
|
||||
export const SESSION_PARSE_CACHE_SCHEMA_VERSION = 3
|
||||
export const SESSION_PARSE_CACHE_MAX_BYTES = 64 * 1024 * 1024
|
||||
export const SESSION_PARSE_CACHE_JSON_LIMITS = {
|
||||
structuralTokens: 1_000_000,
|
||||
nestingDepth: 32
|
||||
} as const
|
||||
|
||||
const TEXT_CHUNK_CHARACTERS = 16 * 1024
|
||||
const VALIDATE_CHUNK_CHARACTERS = 256 * 1024
|
||||
const SERIALIZE_YIELD_STEPS = 1024
|
||||
type CacheEntry = [string, PersistedSessionParseCacheEntry]
|
||||
type JsonPiece = { text: string; tokens: number }
|
||||
type CacheJsonObject = Record<string, unknown>
|
||||
|
||||
function isCacheJsonObject(value: unknown): value is CacheJsonObject {
|
||||
return typeof value === 'object' && value !== null
|
||||
}
|
||||
|
||||
export async function assertSessionParseCacheJsonWithinLimitsCooperatively(
|
||||
content: string | Buffer,
|
||||
limits: JsonTextStructureLimits = SESSION_PARSE_CACHE_JSON_LIMITS
|
||||
): Promise<void> {
|
||||
const validator = new JsonTextStructureValidator(limits)
|
||||
const decoder = new StringDecoder('utf8')
|
||||
for (let start = 0; start < content.length; start += VALIDATE_CHUNK_CHARACTERS) {
|
||||
const chunk = content.slice(start, start + VALIDATE_CHUNK_CHARACTERS)
|
||||
validator.consume(typeof chunk === 'string' ? chunk : decoder.write(chunk))
|
||||
if (start + VALIDATE_CHUNK_CHARACTERS < content.length) {
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
}
|
||||
validator.consume(decoder.end())
|
||||
}
|
||||
|
||||
/** Retain the newest complete suffix; a newest row that cannot fit preserves the prior file. */
|
||||
export async function serializeSessionParseCacheSnapshotPiecesCooperatively(
|
||||
entries: readonly CacheEntry[],
|
||||
appVersion: string,
|
||||
maxBytes = SESSION_PARSE_CACHE_MAX_BYTES,
|
||||
limits: JsonTextStructureLimits = SESSION_PARSE_CACHE_JSON_LIMITS
|
||||
): Promise<{ pieces: string[]; byteLength: number; retainedEntries: number } | null> {
|
||||
const header = stringifyJsonWithinByteLimit(
|
||||
{ schemaVersion: SESSION_PARSE_CACHE_SCHEMA_VERSION, appVersion, entries: [] },
|
||||
maxBytes
|
||||
).serialized
|
||||
const outer = new JsonTextStructureValidator(limits)
|
||||
outer.consume(header)
|
||||
let byteLength = Buffer.byteLength(header)
|
||||
let tokens = outer.usage().structuralTokens
|
||||
const rows: string[][] = []
|
||||
for (let index = entries.length - 1; index >= 0; index--) {
|
||||
const comma = rows.length === 0 ? 0 : 1
|
||||
try {
|
||||
const row = await serializeCacheRow(entries[index], maxBytes - byteLength - comma, {
|
||||
structuralTokens: limits.structuralTokens - tokens - comma,
|
||||
nestingDepth: limits.nestingDepth - 2
|
||||
})
|
||||
rows.push(row.pieces)
|
||||
byteLength += row.byteLength + comma
|
||||
tokens += row.tokens + comma
|
||||
} catch (error) {
|
||||
if (
|
||||
!(
|
||||
error instanceof JsonStringifyByteLimitError ||
|
||||
error instanceof JsonTextStructureCapacityError
|
||||
)
|
||||
) {
|
||||
throw error
|
||||
}
|
||||
break
|
||||
}
|
||||
if (rows.length % 16 === 0) {
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
}
|
||||
if (entries.length > 0 && rows.length === 0) {
|
||||
return null
|
||||
}
|
||||
const pieces = [header.slice(0, -2)]
|
||||
for (let index = rows.length - 1; index >= 0; index--) {
|
||||
if (index < rows.length - 1) {
|
||||
pieces.push(',')
|
||||
}
|
||||
pieces.push(...rows[index]!)
|
||||
}
|
||||
pieces.push(']}')
|
||||
return { pieces, byteLength, retainedEntries: rows.length }
|
||||
}
|
||||
|
||||
async function serializeCacheRow(
|
||||
value: unknown,
|
||||
maxBytes: number,
|
||||
limits: JsonTextStructureLimits
|
||||
) {
|
||||
const pieces: string[] = []
|
||||
let block = ''
|
||||
let byteLength = 0
|
||||
let tokens = 0
|
||||
let steps = 0
|
||||
for (const piece of cacheJsonPieces(value, 0, limits.nestingDepth, new Set())) {
|
||||
byteLength += Buffer.byteLength(piece.text)
|
||||
tokens += piece.tokens
|
||||
if (byteLength > maxBytes) {
|
||||
throw new JsonStringifyByteLimitError(byteLength, maxBytes)
|
||||
}
|
||||
if (tokens > limits.structuralTokens) {
|
||||
throw new JsonTextStructureCapacityError('structuralTokens', limits.structuralTokens)
|
||||
}
|
||||
block += piece.text
|
||||
if (block.length >= TEXT_CHUNK_CHARACTERS) {
|
||||
pieces.push(block)
|
||||
block = ''
|
||||
}
|
||||
// Large strings and wide objects both yield, including omitted properties.
|
||||
if (++steps % SERIALIZE_YIELD_STEPS === 0 || piece.text.length >= TEXT_CHUNK_CHARACTERS / 2) {
|
||||
await yieldToEventLoop()
|
||||
}
|
||||
}
|
||||
if (block) {
|
||||
pieces.push(block)
|
||||
}
|
||||
return { pieces, byteLength, tokens }
|
||||
}
|
||||
|
||||
// Native encoding allocates at most six bytes per character of a bounded string chunk.
|
||||
function* cacheJsonPieces(
|
||||
value: unknown,
|
||||
depth: number,
|
||||
maxDepth: number,
|
||||
ancestors: Set<object>
|
||||
): Generator<JsonPiece> {
|
||||
if (typeof value === 'string') {
|
||||
yield { text: '"', tokens: 0 }
|
||||
for (let start = 0; start < value.length;) {
|
||||
let end = Math.min(value.length, start + TEXT_CHUNK_CHARACTERS)
|
||||
const last = value.charCodeAt(end - 1)
|
||||
if (end < value.length && last >= 0xd800 && last <= 0xdbff) {
|
||||
end--
|
||||
}
|
||||
const serialized = JSON.stringify(value.slice(start, end))
|
||||
yield { text: serialized.slice(1, -1), tokens: 0 }
|
||||
start = end
|
||||
}
|
||||
yield { text: '"', tokens: 0 }
|
||||
return
|
||||
}
|
||||
if (!isCacheJsonObject(value)) {
|
||||
const serialized = JSON.stringify(value)
|
||||
if (serialized === undefined) {
|
||||
throw new TypeError('Session parse cache value is not serializable')
|
||||
}
|
||||
yield { text: serialized, tokens: 0 }
|
||||
return
|
||||
}
|
||||
if (depth + 1 > maxDepth) {
|
||||
throw new JsonTextStructureCapacityError('nestingDepth', maxDepth)
|
||||
}
|
||||
if (ancestors.has(value)) {
|
||||
throw new TypeError('Circular session parse cache row')
|
||||
}
|
||||
ancestors.add(value)
|
||||
let count = 0
|
||||
if (Array.isArray(value)) {
|
||||
yield { text: '[', tokens: 1 }
|
||||
for (const item of value) {
|
||||
if (count++) {
|
||||
yield { text: ',', tokens: 1 }
|
||||
}
|
||||
yield* cacheJsonPieces(item === undefined ? null : item, depth + 1, maxDepth, ancestors)
|
||||
}
|
||||
yield { text: ']', tokens: 1 }
|
||||
} else {
|
||||
yield { text: '{', tokens: 1 }
|
||||
for (const key in value) {
|
||||
if (!Object.hasOwn(value, key)) {
|
||||
continue
|
||||
}
|
||||
const item: unknown = value[key]
|
||||
if (item === undefined) {
|
||||
yield { text: '', tokens: 0 }
|
||||
continue
|
||||
}
|
||||
if (count++) {
|
||||
yield { text: ',', tokens: 1 }
|
||||
}
|
||||
yield* cacheJsonPieces(key, depth + 1, maxDepth, ancestors)
|
||||
yield { text: ':', tokens: 1 }
|
||||
yield* cacheJsonPieces(item, depth + 1, maxDepth, ancestors)
|
||||
}
|
||||
yield { text: '}', tokens: 1 }
|
||||
}
|
||||
ancestors.delete(value)
|
||||
}
|
||||
@@ -6,7 +6,7 @@ import type { SkippedTranscriptRecord } from './session-transcript-record-budget
|
||||
|
||||
// Sized past the default recency cap (1000) plus the in-scope cap (2000) so a
|
||||
// full steady-state result set stays resident between forced rescans.
|
||||
const MAX_CACHE_ENTRIES = 4096
|
||||
export const MAX_CACHE_ENTRIES = 4096
|
||||
|
||||
export type SessionParseResumePoint = {
|
||||
state: ResumableSessionParseState
|
||||
@@ -77,17 +77,26 @@ export function snapshotSessionParseCacheForPersistence(): [
|
||||
export function seedSessionParseCache(
|
||||
entries: Iterable<[string, PersistedSessionParseCacheEntry]>
|
||||
): void {
|
||||
const list = [...entries]
|
||||
// Snapshot order is oldest→newest (LRU); an over-cap list keeps the newest
|
||||
// tail rather than seeding the oldest entries and dropping the tail.
|
||||
for (const [path, entry] of list.slice(Math.max(0, list.length - MAX_CACHE_ENTRIES))) {
|
||||
if (cache.size >= MAX_CACHE_ENTRIES) {
|
||||
return
|
||||
}
|
||||
// In-process entries are always fresher than persisted ones; never clobber.
|
||||
const newest = new Map<string, PersistedSessionParseCacheEntry>()
|
||||
for (const [path, entry] of entries) {
|
||||
if (cache.has(path)) {
|
||||
continue
|
||||
}
|
||||
newest.delete(path)
|
||||
newest.set(path, entry)
|
||||
if (newest.size > MAX_CACHE_ENTRIES) {
|
||||
const oldest = newest.keys().next()
|
||||
if (!oldest.done) {
|
||||
newest.delete(oldest.value)
|
||||
}
|
||||
}
|
||||
}
|
||||
const available = MAX_CACHE_ENTRIES - cache.size
|
||||
if (available <= 0) {
|
||||
return
|
||||
}
|
||||
// In-process entries win; among persisted duplicates the newest row wins.
|
||||
for (const [path, entry] of [...newest].slice(-available)) {
|
||||
cache.set(path, {
|
||||
mtimeMs: entry.mtimeMs,
|
||||
sizeBytes: entry.sizeBytes,
|
||||
|
||||
@@ -21,59 +21,85 @@ export function assertJsonTextStructureWithinLimits(
|
||||
content: string,
|
||||
limits: JsonTextStructureLimits
|
||||
): void {
|
||||
assertLimit(limits.structuralTokens)
|
||||
assertLimit(limits.nestingDepth)
|
||||
let structuralTokens = 0
|
||||
let depth = 0
|
||||
for (let index = 0; index < content.length; index += 1) {
|
||||
const character = content[index]
|
||||
if (character === '"') {
|
||||
let quote = content.indexOf('"', index + 1)
|
||||
if (quote !== -1) {
|
||||
// Only an odd backslash run escapes the quote.
|
||||
let backslashes = 0
|
||||
for (let at = quote - 1; at > index && content[at] === '\\'; at -= 1) {
|
||||
backslashes += 1
|
||||
new JsonTextStructureValidator(limits).consume(content)
|
||||
}
|
||||
|
||||
/** Carries string/escape and structure state across bounded chunks. */
|
||||
export class JsonTextStructureValidator {
|
||||
private structuralTokens = 0
|
||||
private depth = 0
|
||||
private maximumDepth = 0
|
||||
private inString = false
|
||||
private escaped = false
|
||||
|
||||
constructor(private readonly limits: JsonTextStructureLimits) {
|
||||
assertLimit(limits.structuralTokens)
|
||||
assertLimit(limits.nestingDepth)
|
||||
}
|
||||
|
||||
consume(content: string): void {
|
||||
let linearString = false
|
||||
for (let index = 0; index < content.length; index += 1) {
|
||||
const character = content[index]
|
||||
if (this.inString) {
|
||||
if (this.escaped) {
|
||||
this.escaped = false
|
||||
continue
|
||||
}
|
||||
if (backslashes % 2 !== 0) {
|
||||
// Escape-heavy strings use the linear scan to avoid repeated native searches.
|
||||
let escaped = false
|
||||
for (quote += 1; quote < content.length; quote += 1) {
|
||||
if (escaped) {
|
||||
escaped = false
|
||||
} else if (content[quote] === '\\') {
|
||||
escaped = true
|
||||
} else if (content[quote] === '"') {
|
||||
break
|
||||
}
|
||||
if (!linearString) {
|
||||
const quote = content.indexOf('"', index)
|
||||
const end = quote === -1 ? content.length : quote
|
||||
let backslashes = 0
|
||||
for (let at = end - 1; at >= index && content[at] === '\\'; at -= 1) {
|
||||
backslashes++
|
||||
}
|
||||
if (quote === content.length) {
|
||||
quote = -1
|
||||
if (quote === -1) {
|
||||
this.escaped = backslashes % 2 !== 0
|
||||
return
|
||||
}
|
||||
index = quote
|
||||
if (backslashes % 2 === 0) {
|
||||
this.inString = false
|
||||
} else {
|
||||
// Escape-heavy strings scan linearly instead of repeating native searches.
|
||||
linearString = true
|
||||
}
|
||||
continue
|
||||
}
|
||||
if (character === '\\') {
|
||||
this.escaped = true
|
||||
} else if (character === '"') {
|
||||
this.inString = false
|
||||
linearString = false
|
||||
}
|
||||
continue
|
||||
}
|
||||
if (quote === -1) {
|
||||
return
|
||||
if (character === '"') {
|
||||
this.inString = true
|
||||
continue
|
||||
}
|
||||
index = quote
|
||||
continue
|
||||
}
|
||||
if (!isStructuralToken(character)) {
|
||||
continue
|
||||
}
|
||||
structuralTokens += 1
|
||||
if (structuralTokens > limits.structuralTokens) {
|
||||
throw new JsonTextStructureCapacityError('structuralTokens', limits.structuralTokens)
|
||||
}
|
||||
if (character === '{' || character === '[') {
|
||||
depth += 1
|
||||
if (depth > limits.nestingDepth) {
|
||||
throw new JsonTextStructureCapacityError('nestingDepth', limits.nestingDepth)
|
||||
if (!isStructuralToken(character)) {
|
||||
continue
|
||||
}
|
||||
this.structuralTokens++
|
||||
if (this.structuralTokens > this.limits.structuralTokens) {
|
||||
throw new JsonTextStructureCapacityError('structuralTokens', this.limits.structuralTokens)
|
||||
}
|
||||
if (character === '{' || character === '[') {
|
||||
this.depth++
|
||||
this.maximumDepth = Math.max(this.maximumDepth, this.depth)
|
||||
if (this.depth > this.limits.nestingDepth) {
|
||||
throw new JsonTextStructureCapacityError('nestingDepth', this.limits.nestingDepth)
|
||||
}
|
||||
} else if (character === '}' || character === ']') {
|
||||
this.depth = Math.max(0, this.depth - 1)
|
||||
}
|
||||
} else if (character === '}' || character === ']') {
|
||||
depth = Math.max(0, depth - 1)
|
||||
}
|
||||
}
|
||||
|
||||
usage(): { structuralTokens: number; nestingDepth: number } {
|
||||
return { structuralTokens: this.structuralTokens, nestingDepth: this.maximumDepth }
|
||||
}
|
||||
}
|
||||
|
||||
function assertLimit(value: number): void {
|
||||
|
||||
Reference in New Issue
Block a user