mirror of
https://github.com/stablyai/orca.git
synced 2026-10-05 08:02:33 +00:00
* fix(push): bound the delivery claim and stop keeping finished batches * fix(push): bound claim scans to the notification TTL and document the queue * test(push): boot waits out a table writer for the queue indexes; queue stays correct without them * fix(push): share the claim lock so a previous-revision claim cannot re-lease a delivery During a deploy overlap the previous revision's claim scans under an exclusive push-worker-claim lock and re-reads the row without checking lease_until, so it could overwrite a lease this revision had just committed and send twice. The new claim now takes the same key shared: new claimers never block each other, and the previous claim waits until their leases commit before it scans. Droppable one release after every worker runs this revision. * fix(push): install one queue index at boot, not three push_batches_leased_device indexed lease_until, so every lease, renew and finish UPDATE lost heap-only eligibility and rewrote every index. push_batches_pending_due was unused: on a synthetic 902k-row table the candidate scan plans onto the existing (state, due_at) and expiry indexes with or without it. The per-device pending index stays; the head check and the busy anti-join use it. Fewer boot-time builds also shorten the SHARE lock the first boot takes on the table. * test(push): pin the claim's TTL scan bound and the server's worker connection cap Removing either guard left the suite green. The claim test captures every row the candidate scan returns and plants one row that only the TTL term excludes; the server test drives the real worker through createPushServer and fails when the request-connection reservation is unwired (peak 4 instead of 2). * fix(push): renew delivery leases outside the background connection cap Renew shared the single background slot with claim retries and prune batches, so a heartbeat could wait long enough for a lease to lapse and the delivery to be re-leased mid-send. It is a keyed one-row UPDATE, so request traffic cannot starve it on the ungated pool. * docs(push): describe the shared claim lock for mixed-revision deploys
57 lines
2.5 KiB
TypeScript
57 lines
2.5 KiB
TypeScript
import { randomUUID } from 'node:crypto'
|
|
import pg from 'pg'
|
|
import { afterEach, expect, it, vi } from 'vitest'
|
|
import { openPushDatabase } from './push-database.js'
|
|
import { durablePushTestDatabaseUrl } from './durable-push-store.test-fixture.js'
|
|
|
|
const QUEUE_INDEXES = ['push_batches_pending_device']
|
|
const cleanups: (() => Promise<void>)[] = []
|
|
afterEach(async () => {
|
|
vi.restoreAllMocks()
|
|
for (const cleanup of cleanups.splice(0).reverse()) await cleanup()
|
|
})
|
|
|
|
// Push applies its schema with retryLockTimeout, so a lock timeout is retried, never deferred:
|
|
// the new revision either boots with every index or does not boot at all.
|
|
it.skipIf(!durablePushTestDatabaseUrl)(
|
|
'creates the queue index after waiting out a writer that holds the table',
|
|
async () => {
|
|
const admin = new pg.Client({ connectionString: durablePushTestDatabaseUrl })
|
|
await admin.connect()
|
|
const schema = `boot_${randomUUID().replaceAll('-', '')}`
|
|
cleanups.push(async () => {
|
|
await admin.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`)
|
|
await admin.end()
|
|
})
|
|
await admin.query(`CREATE SCHEMA ${schema}`)
|
|
const url = new URL(durablePushTestDatabaseUrl!)
|
|
url.searchParams.set('options', `-c search_path=${schema}`)
|
|
await (await openPushDatabase({ databaseUrl: url.toString(), dataDir: '' })).close()
|
|
await admin.query(
|
|
`DROP INDEX ${QUEUE_INDEXES.map((name) => `${schema}.${name}`).join(', ')}`
|
|
)
|
|
|
|
const writer = new pg.Client({ connectionString: url.toString() })
|
|
await writer.connect()
|
|
cleanups.push(() => writer.end())
|
|
await writer.query('BEGIN')
|
|
await writer.query('LOCK TABLE push_delivery_batches IN ROW EXCLUSIVE MODE')
|
|
const warnings: string[] = []
|
|
vi.spyOn(console, 'warn').mockImplementation((line: string) => warnings.push(line))
|
|
vi.spyOn(console, 'log').mockImplementation(() => undefined)
|
|
const held = setTimeout(() => void writer.query('COMMIT'), 2_500)
|
|
cleanups.push(async () => clearTimeout(held))
|
|
|
|
const booted = await openPushDatabase({ databaseUrl: url.toString(), dataDir: '' })
|
|
await booted.close()
|
|
const { rows } = await admin.query(
|
|
'SELECT indexname FROM pg_indexes WHERE schemaname = $1 AND indexname = ANY($2) ORDER BY 1',
|
|
[schema, QUEUE_INDEXES]
|
|
)
|
|
expect(rows.map((row) => row.indexname)).toEqual(QUEUE_INDEXES)
|
|
const events = warnings.map((line) => JSON.parse(line).event)
|
|
expect(events).toContain('orca_push_postgres_schema_retry')
|
|
expect(events).not.toContain('orca_push_postgres_schema_object_deferred')
|
|
}
|
|
)
|