Files
orca/src/main/sqlite/sqlite-read-failure.test.ts
T
Neil 9062494f9b fix(ai-vault): stop a whole opencode.db failure reading as one skipped transcript (#16587)
* fix(ai-vault): stop a whole opencode.db failure reading as one skipped transcript

#15036 reported "1 transcript skipped / database is locked" with both Agent
Session History scopes empty. Two separate defects.

The panel counts every unkinded scan issue as a skipped transcript, so a
failure that lost an entire *source* was reported as one lost *file*. The
whole-database failure is now kinded `scope`, and an unknown `kind` from a
newer host degrades to `scope` instead of failing validation and coming back
unkinded — a mixed-version remote host previously turned a source-level
failure into a phantom skipped transcript.

The read also inherited sqlite3's 0 ms busy timeout, so a genuinely contended
open failed in ~1 ms. It now opens once with a bounded timeout. No retry loop:
sqlite's own busy handler already blocks and retries internally for the whole
timeout, and WAL readers do not block on a writer at all (measured: 547/547
cross-process reads at timeout=0 while a writer held open transactions).

Measured against a real Ubuntu-24.04 distro, Windows cannot take SQLite's file
locks over \\wsl.localhost at all: an idle, never-WAL, nothing-attached
database still answers SQLITE_BUSY, a 5 s busy timeout does not change it, and
the identical bytes open fine once copied to local disk. So a lock-family error
on that share never means "a writer holds it" and no timeout can help. The copy
says so rather than sending the user after a write-ahead log that is not the
problem. Restoring those sessions needs an in-distro read; that is a follow-up,
and this PR no longer pretends a timeout will do it.

immutable=1 is deliberately not used as a workaround: over the same share it
opens and returns 100 of 150 rows, silently dropping everything still in the
uncheckpointed -wal — in a history panel, exactly the newest sessions.

* skip the provably futile busy wait on \\wsl.localhost paths
2026-08-26 23:37:16 -07:00

114 lines
3.9 KiB
TypeScript

import { mkdtempSync, rmSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { afterEach, describe, expect, it } from 'vitest'
import SyncDatabase from './sync-database'
import { classifySqliteReadFailure, isTransientSqliteContention } from './sqlite-read-failure'
// Error codes here are the ones a real node:sqlite open produces: a contended
// database reports errcode 5 ("database is locked"), while a read-only WAL open
// with no usable -shm reports errcode 14 ("unable to open database file"). The
// two need opposite responses, so the classifier must never conflate them.
let tempDirs: string[] = []
afterEach(() => {
for (const dir of tempDirs) {
rmSync(dir, { recursive: true, force: true })
}
tempDirs = []
})
function contendedDatabase(): { path: string; release: () => void } {
const dir = mkdtempSync(join(tmpdir(), 'orca-sqlite-failure-'))
tempDirs.push(dir)
const path = join(dir, 'contended.db')
const writer = new SyncDatabase(path)
writer.exec('PRAGMA journal_mode=DELETE')
writer.exec('CREATE TABLE session (id TEXT PRIMARY KEY)')
writer.exec('BEGIN EXCLUSIVE')
writer.exec("INSERT INTO session VALUES ('a')")
return {
path,
release: () => {
writer.exec('COMMIT')
writer.close()
}
}
}
describe('isTransientSqliteContention', () => {
it('recognizes a real SQLITE_BUSY thrown by a read-only open', () => {
const contended = contendedDatabase()
let thrown: unknown
try {
new SyncDatabase(contended.path, { readonly: true, timeout: 0 })
.prepare('SELECT id FROM session')
.all()
} catch (error) {
thrown = error
} finally {
contended.release()
}
expect((thrown as { errcode?: number }).errcode).toBe(5)
expect(isTransientSqliteContention(thrown)).toBe(true)
})
it('recognizes a relayed message with no errcode, as the Codex heal pass sees it', () => {
expect(
isTransientSqliteContention('codex app-server thread/read failed: database is locked')
).toBe(true)
expect(isTransientSqliteContention(new Error('SQLITE_LOCKED: table is locked'))).toBe(true)
})
it('reads extended result codes through their primary code', () => {
// SQLITE_BUSY_SNAPSHOT (517) and SQLITE_BUSY_RECOVERY (261) both pack 5.
expect(isTransientSqliteContention({ errcode: 517, message: 'busy snapshot' })).toBe(true)
expect(isTransientSqliteContention({ errcode: 261, message: 'recovery' })).toBe(true)
})
it('does not treat an unreadable or absent database as contention', () => {
expect(isTransientSqliteContention(new Error('file is not a database'))).toBe(false)
expect(
isTransientSqliteContention({ errcode: 14, message: 'unable to open database file' })
).toBe(false)
})
})
describe('classifySqliteReadFailure', () => {
it('classifies contention as retryable regardless of the file evidence', () => {
const error = { errcode: 5, message: 'database is locked' }
expect(classifySqliteReadFailure({ error, databaseFileExists: true })).toBe('contended')
expect(classifySqliteReadFailure({ error, databaseFileExists: false })).toBe('contended')
})
it('classifies SQLITE_CANTOPEN against a present database as an unreachable wal-index', () => {
expect(
classifySqliteReadFailure({
error: { errcode: 14, message: 'unable to open database file' },
databaseFileExists: true
})
).toBe('wal-index-unavailable')
})
it('does not blame the wal-index when the database file itself is gone', () => {
expect(
classifySqliteReadFailure({
error: { errcode: 14, message: 'unable to open database file' },
databaseFileExists: false
})
).toBe('unreadable')
})
it('falls back to unreadable for anything else', () => {
expect(
classifySqliteReadFailure({
error: { errcode: 11, message: 'database disk image is malformed' },
databaseFileExists: true
})
).toBe('unreadable')
})
})