mirror of
https://github.com/stablyai/orca.git
synced 2026-10-03 16:02:11 +00:00
* perf: avoid repeatedly encoding retained VM recipe output
* perf: capture retained VM recipe output as raw bytes in the shared byte buffer
The previous commit added a third byte-retention buffer to the repo. This
replaces it with the one that already existed and removes the remaining
encoding work.
`runRecipeCommand` no longer calls `setEncoding('utf8')` on the child's stdout
and stderr. It keeps the raw `Buffer` chunks and runs one `StringDecoder` per
stream to feed the existing string callbacks, which is exactly how
`setEncoding` is implemented, so callbacks see the same characters at the same
boundaries. With the bytes already in hand the capture encodes nothing: the
4,194,304 bytes the ring still encoded for 4 MiB of output drop to 0, and the
UTF-8 continuation trim collapses from one scan per chunk to a single scan when
the tail is decoded.
Retention is now `GrowingByteBuffer.appendRetainedSuffix`, which had no
production consumer. It gained an O(1) head offset, so `discardPrefix` and
`retainSuffix` mark bytes dead instead of moving the whole tail and `append`
slides or grows only when the head offset runs out of room. Quick Open path
accumulation and the SOCKS handshake buffer get that win too. Without the
offset the per-chunk memmove costs 12.36 ms for 4 MiB; with it, 0.25 ms against
the ring's 0.53 ms and the old per-chunk re-encode's 265.78 ms.
Two behaviour notes. Odd capture limits are clamped once at entry instead of
carrying a per-chunk coercion path no production caller could reach, so an
infinite or NaN cap is now bounded at 1 MiB rather than retaining everything.
And malformed UTF-8 yields a different tail: replacement characters no longer
inflate the byte count, so a malformed tail keeps more of what the recipe
actually wrote.
The encoding-budget assertions no longer spy on `Buffer` itself, where any
unrelated allocation in the same tick could flip them. They count bytes through
the capture's own buffer class and still assert the deterministic oracle: at
most 5 MiB moved for 4 MiB of output, exactly 4 MiB appended, 1 MiB decoded,
and the stored chunks identical to the Buffers the stream delivered.
Co-Authored-By: Claude <noreply@anthropic.com>
* test(vm-recipe): emit Buffers from the doctor stream doubles
Dropping setEncoding('utf8') means stdout and stderr now deliver Buffers, so
the hand-rolled EventEmitter doubles emitting strings threw inside the data
listener — the capture retained nothing and the exit path never settled.
---------
Co-authored-by: Claude <noreply@anthropic.com>
96 lines
3.2 KiB
TypeScript
96 lines
3.2 KiB
TypeScript
import { describe, expect, it } from 'vitest'
|
|
import { GrowingByteBuffer } from './growing-byte-buffer'
|
|
|
|
describe('GrowingByteBuffer', () => {
|
|
it('transfers binary bytes without changing them on clear or reuse', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
const bytes = Buffer.from([0, 255, 128, 10, 0])
|
|
buffer.append(bytes.subarray(0, 2))
|
|
buffer.append(bytes.subarray(2))
|
|
const taken = buffer.takeBuffer()
|
|
expect(taken).toEqual(bytes)
|
|
expect(buffer.byteLength).toBe(0)
|
|
expect(buffer.takeBuffer()).toEqual(Buffer.alloc(0))
|
|
buffer.append(Buffer.alloc(1024, 7))
|
|
buffer.clear()
|
|
expect(taken).toEqual(bytes)
|
|
})
|
|
|
|
it('retains 100,000 one-byte fragments in one growable allocation', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
const expected = Buffer.alloc(100_000)
|
|
|
|
for (let index = 0; index < expected.byteLength; index += 1) {
|
|
const value = index % 251
|
|
expected[index] = value
|
|
buffer.append(Uint8Array.of(value))
|
|
}
|
|
|
|
expect(buffer.byteLength).toBe(expected.byteLength)
|
|
expect(buffer.takeString('latin1')).toBe(expected.toString('latin1'))
|
|
expect(buffer.byteLength).toBe(0)
|
|
})
|
|
|
|
it('consumes delimited prefixes and retains a bounded suffix', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
for (const byte of Buffer.from('first\nsecond-tail')) {
|
|
buffer.append(Uint8Array.of(byte))
|
|
}
|
|
|
|
const newline = buffer.indexOfByte(0x0a)
|
|
expect(buffer.takePrefixString(newline)).toBe('first')
|
|
buffer.discardPrefix(1)
|
|
buffer.retainSuffix(4)
|
|
|
|
expect(buffer.toString()).toBe('tail')
|
|
})
|
|
|
|
it('appends only a bounded copy from an oversized source chunk', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
buffer.append(Buffer.from('old'))
|
|
const source = Buffer.from('discard-prefix-tail')
|
|
|
|
buffer.appendRetainedSuffix(source, 4)
|
|
source.fill(0)
|
|
|
|
expect(buffer.byteLength).toBe(4)
|
|
expect(buffer.toString()).toBe('tail')
|
|
})
|
|
|
|
it('keeps the newest bytes across bounded suffix appends', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
|
|
buffer.appendRetainedSuffix(Buffer.from('1234'), 6)
|
|
buffer.appendRetainedSuffix(Buffer.from('5678'), 6)
|
|
|
|
expect(buffer.toString()).toBe('345678')
|
|
})
|
|
|
|
it('reads and writes correctly after the head offset has advanced', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
buffer.append(Buffer.from('aaa\nbbb\nccc'))
|
|
|
|
expect(buffer.takePrefixString(buffer.indexOfByte(0x0a))).toBe('aaa')
|
|
buffer.discardPrefix(1)
|
|
expect(buffer.indexOfByte(0x0a)).toBe(3)
|
|
expect(buffer.byteLength).toBe(7)
|
|
buffer.append(Buffer.from('!'))
|
|
expect(buffer.toString()).toBe('bbb\nccc!')
|
|
expect(buffer.takeBuffer()).toEqual(Buffer.from('bbb\nccc!'))
|
|
expect(buffer.byteLength).toBe(0)
|
|
})
|
|
|
|
it('keeps a bounded suffix stable across many small appends without unbounded growth', () => {
|
|
const buffer = new GrowingByteBuffer()
|
|
let expected = ''
|
|
for (let index = 0; index < 5_000; index += 1) {
|
|
const chunk = String(index % 10).repeat(3)
|
|
buffer.appendRetainedSuffix(Buffer.from(chunk), 64)
|
|
expected = (expected + chunk).slice(-64)
|
|
}
|
|
|
|
expect(buffer.byteLength).toBe(64)
|
|
expect(buffer.takeString()).toBe(expected)
|
|
})
|
|
})
|