fix(relay): name the LAN flip that lands mid-mint, not relay_control_not_active

withTransientDemand checks the host pairing policy once, at entry. But
RelayDemandLedger.hasDemand filters the in-flight operation's OWN transient
ref through the live policy, so a flip to local-only while createPairingRelay
or provisionRelay is awaiting withdraws the demand that operation is holding.
The coordinator then reaches its no-demand branch and publishes 'standby',
publish() clears offlineReason to null for any non-offline status, and the
broker wait returns with no cause at all — relay_control_not_active, the
generic code this stack exists to remove.

The flip is exactly the cause and relay_disabled_for_device is already its
word; the entry gate just could not see a flip that had not happened yet.
Re-ask the policy on the failure path rather than trusting the thrown code.

Reproduced first: the added test failed with "expected ... to throw error
including 'relay_disabled_for_device' but got 'relay_control_not_active'".
Controls cover the two ways this could over-reach — a failure the policy had
nothing to do with is not rewritten, and a grant that succeeded under a flip
is left alone.

Top follow-up found alongside this and deliberately not fixed here:
flushRevoke swallows every error with a bare catch, no attempt cap and no
item expiry, while the revoke check in hasDemand is deliberately not filtered
through the policy. A revoke that fails permanently server-side therefore
holds demand open forever and defeats the LAN pick entirely — the same
symptom this change is part of closing, by a different route, with no test
coverage.
This commit is contained in:
Neil
2026-09-11 01:08:04 -07:00
parent 806272e9c5
commit 13eee7eacf
2 changed files with 98 additions and 0 deletions
@@ -0,0 +1,89 @@
import { describe, expect, it, vi } from 'vitest'
import { DesktopRelayService } from './desktop-relay-service'
import type { MobilePairingConnectionMode } from '../../../shared/mobile-pairing-connection-mode'
/**
* A host LAN flip lands mid-mint.
*
* `hasDemand` filters the in-flight operation's OWN transient ref through the live policy
* (relay-demand-ledger.ts:50-54), so the flip withdraws the demand that operation is holding.
* The coordinator then publishes `standby`, which clears `offlineReason` to null, and the broker
* wait returns with no cause — `relay_control_not_active`, the generic code this stack exists to
* remove. The flip is the cause and `relay_disabled_for_device` is already its word.
*/
type TransientDemandHost = {
withTransientDemand: (
kind: string,
deviceId: string,
operation: () => Promise<unknown>
) => Promise<unknown>
}
function serviceWithMode(mode: { current: MobilePairingConnectionMode }): {
run: (operation: () => Promise<unknown>) => Promise<unknown>
release: ReturnType<typeof vi.fn>
} {
const release = vi.fn()
const service = Object.create(DesktopRelayService.prototype) as DesktopRelayService
Object.assign(service, {
hostMobilePairingConnectionMode: () => mode.current,
runtimeRpc: {
getDeviceRegistry: () => ({ getMobilePairingConnectionMode: () => 'automatic' })
},
demandLedger: { acquireTransient: () => release },
refreshDemand: () => {}
})
const host = service as unknown as TransientDemandHost
return {
run: (operation) => host.withTransientDemand.call(service, 'pairing', 'device-1', operation),
release
}
}
describe('DesktopRelayService policy flip during an in-flight grant', () => {
it('names the LAN flip rather than the generic control-not-active code', async () => {
const mode = { current: 'automatic' as MobilePairingConnectionMode }
const { run, release } = serviceWithMode(mode)
await expect(
run(async () => {
mode.current = 'local-only'
// What requireActiveBroker throws once the flip cleared the offline reason.
throw new Error('relay_control_not_active')
})
).rejects.toThrow('relay_disabled_for_device')
expect(release).toHaveBeenCalled()
})
it('does not rewrite a failure the policy had nothing to do with', async () => {
const mode = { current: 'automatic' as MobilePairingConnectionMode }
const { run } = serviceWithMode(mode)
await expect(
run(async () => {
throw new Error('relay_broker_unavailable')
})
).rejects.toThrow('relay_broker_unavailable')
})
it('still refuses at the entry gate when the policy already excludes the device', async () => {
const mode = { current: 'local-only' as MobilePairingConnectionMode }
const { run } = serviceWithMode(mode)
const operation = vi.fn(async () => 'unreachable')
await expect(run(operation)).rejects.toThrow('relay_disabled_for_device')
expect(operation).not.toHaveBeenCalled()
})
it('leaves a successful grant alone even if the policy flips under it', async () => {
const mode = { current: 'automatic' as MobilePairingConnectionMode }
const { run } = serviceWithMode(mode)
await expect(
run(async () => {
mode.current = 'local-only'
return 'minted'
})
).resolves.toBe('minted')
})
})
@@ -295,6 +295,15 @@ export class DesktopRelayService {
this.refreshDemand()
try {
return await operation()
} catch (error) {
// Why re-ask instead of trusting the thrown code: a flip lands mid-operation, and
// `hasDemand` filters this ref's own demand through the live policy, so the coordinator
// reaches `standby` and clears the offline reason. The wait then ends with no cause at all
// — the generic `relay_control_not_active` — when the flip is exactly the cause.
if (!this.isRelayAllowedForDevice(deviceId)) {
throw new Error('relay_disabled_for_device')
}
throw error
} finally {
release()
this.refreshDemand()