Merge remote-tracking branch 'origin/main' into brennanb2025/native-chat-turn-lifecycle-durable

This commit is contained in:
Merge Sim
2026-09-09 21:08:11 -07:00
59 changed files with 3044 additions and 534 deletions
+240 -38
View File
@@ -16,10 +16,9 @@ permissions:
contents: read
id-token: write
# The gateway applies its own schema at startup against the shared Cloud SQL instance, so a
# deploy is a connection-budget rollout and belongs in the same serialized group as the relay.
# Serialize push traffic changes independently of Relay and the shared database.
concurrency:
group: production-cloud-sql-rollout
group: production-push-rollout
cancel-in-progress: false
defaults:
@@ -75,8 +74,8 @@ jobs:
- uses: google-github-actions/auth@v2
with:
workload_identity_provider: ${{ vars.PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER }}
service_account: ${{ vars.PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT }}
workload_identity_provider: ${{ vars.PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER }}
service_account: ${{ vars.PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT }}
- uses: google-github-actions/setup-gcloud@v2
@@ -85,31 +84,41 @@ jobs:
- name: Configure Docker auth
run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet
# Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance,
# and a multi-minute image build inside the lease blocks every relay deploy and rehome for
# its duration. The lease below covers exactly the connection-budget window: deploy, probe,
# shift.
# Building an image does not need the deployment lease.
- name: Build and publish the immutable gateway image
shell: bash
run: |
set -euo pipefail
image_tag="${GCP_REGION}-docker.pkg.dev/${GCP_PROJECT_ID}/${REPOSITORY_ID}/${IMAGE_NAME}:sha-${SOURCE_SHA}"
docker build -f "${RUNNER_TEMP}/push-source/cloud/apps/push/Dockerfile" \
docker buildx build --push --platform linux/amd64 --provenance=false --metadata-file "${RUNNER_TEMP}/push-image.json" \
-f "${RUNNER_TEMP}/push-source/cloud/apps/push/Dockerfile" \
-t "${image_tag}" "${RUNNER_TEMP}/push-source/cloud"
docker push "${image_tag}"
digest="$(gcloud artifacts docker images describe "${image_tag}" \
--format='value(image_summary.digest)')"
digest="$(jq -er '."containerimage.digest"' "${RUNNER_TEMP}/push-image.json")"
[[ "${digest}" =~ ^sha256:[a-f0-9]{64}$ ]]
echo "IMAGE=${GCP_REGION}-docker.pkg.dev/${GCP_PROJECT_ID}/${REPOSITORY_ID}/${IMAGE_NAME}@${digest}" \
>> "${GITHUB_ENV}"
echo "IMAGE_DIGEST=${digest}" >> "${GITHUB_ENV}"
# Refuse older images before they ever boot against production.
- name: Require image support for inert validation
shell: bash
run: |
set -euo pipefail
docker run --rm --network none --entrypoint node "${IMAGE}" --input-type=module -e '
import { loadPushConfig } from "./apps/push/dist/config.js";
const env = { ORCA_PUSH_PUBLIC_URL: "https://push.onorca.dev", ORCA_PUSH_MODE: "validation" };
if (loadPushConfig(env).mode !== "validation") throw new Error("validation_mode_unsupported");
let rejected = false;
try { loadPushConfig({ ...env, ORCA_PUSH_MODE: "invalid" }); } catch { rejected = true; }
if (!rejected) throw new Error("validation_mode_not_fail_closed");
'
# Held across the deploy, not just a separate schema step: the gateway opens its pool and
# applies its schema while the new revision starts, so the revision is the schema step.
- uses: ./.github/actions/cloud-sql-rollout-lease
with:
bucket: onorca-cloud-terraform-state
object: terraform/state/cloud-sql-rollout/production.lock
object: terraform/state/push-rollout/production.lock
# Why: the candidate inherits the serving revision's scaling. A serving revision that has
# drifted below the floor would hand the candidate a cold start on every notification, and
@@ -124,6 +133,12 @@ jobs:
| jq -r '[.status.traffic[] | select((.percent // 0) > 0)]
| if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
test -n "${serving}"
revisions="$(gcloud run revisions list --service "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format='value(metadata.name)')"
if test "${revisions}" != "${serving}"; then
echo 'Retire leftover revisions under the rollout lease before deploying; three pools are the limit.' >&2
exit 1
fi
floor="$(gcloud run revisions describe "${serving}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--format="value(metadata.annotations['autoscaling.knative.dev/minScale'])")"
@@ -140,16 +155,29 @@ jobs:
test "${ceiling}" = "${PUSH_MAX_INSTANCES}"
echo "serving revision ${serving} holds ${floor} minimum and ${ceiling} maximum instances"
echo "ROLLBACK_REVISION=${serving}" >> "${GITHUB_ENV}"
gcloud run revisions describe "${serving}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
> "${RUNNER_TEMP}/push-rollback-revision.json"
image="$(jq -er '.status.imageDigest' "${RUNNER_TEMP}/push-rollback-revision.json")"
[[ "${image}" =~ @sha256:[a-f0-9]{64}$ ]]
echo "ROLLBACK_IMAGE=${image}" >> "${GITHUB_ENV}"
jq -e 'all(.spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE" or .value == "active")' \
"${RUNNER_TEMP}/push-rollback-revision.json" > /dev/null
# No traffic and a per-revision tag: the candidate boots, applies schema, and is probed on
# its own URL while every phone and desktop still reaches the previous revision.
# Validation has no schema writes, HTTP mutations, worker, or pruners; tags alone do not
# isolate background consumers from production.
- name: Deploy the candidate revision with no traffic
shell: bash
run: |
set -euo pipefail
tag="c${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
echo "CANDIDATE_TAG=${tag}" >> "${GITHUB_ENV}"
echo "CANDIDATE_REVISION=${SERVICE_NAME}-${tag}" >> "${GITHUB_ENV}"
{
echo "VALIDATION_DEPLOY_ATTEMPTED=true"
echo "VALIDATION_REVISION=${SERVICE_NAME}-${tag}"
echo "VALIDATION_TAG=${tag}"
echo "CANDIDATE_TAG=${tag}"
echo "CANDIDATE_REVISION=${SERVICE_NAME}-${tag}"
} >> "${GITHUB_ENV}"
gcloud run deploy "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" \
--region "${GCP_REGION}" \
@@ -157,6 +185,7 @@ jobs:
--tag "${tag}" \
--revision-suffix "${tag}" \
--no-traffic \
--update-env-vars ORCA_PUSH_MODE=validation \
--quiet
candidate="$(gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
@@ -168,8 +197,7 @@ jobs:
# A tagged revision is directly addressable and sits outside the service-wide cap, so the
# candidate and the serving revision each draw up to the ceiling during the probe window.
# The lease is taken for exactly that doubling; a candidate that inherited a wider ceiling
# would exceed it, so the inherited scaling is asserted here too.
# Successor creation later requires three revision pools; assert the inherited ceiling.
- name: Require the candidate to serve the exact image and inherited scaling
shell: bash
run: |
@@ -195,7 +223,7 @@ jobs:
if test "${code}" = 200; then
jq -e . < "${RUNNER_TEMP}/push-ready.json" > /dev/null
curl --fail --silent --show-error --max-time 10 "${CANDIDATE_URL}/health" \
| jq -e '.ok == true and .deliveryProtocol == 2' > /dev/null
| jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "validation"' > /dev/null
echo "candidate ${CANDIDATE_REVISION} is ready after ${attempt} attempt(s)"
exit 0
fi
@@ -242,6 +270,53 @@ jobs:
fi
test "${status}" = INVALID_ARGUMENT
# Cloud Run requires a successor before the latest revision can be deleted.
- name: Retire inert validation and activate the verified image
shell: bash
run: |
set -euo pipefail
tag="a${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
{
echo "CANDIDATE_TAG=${tag}"
echo "CANDIDATE_REVISION=${SERVICE_NAME}-${tag}"
echo "ACTIVATION_ATTEMPTED=true"
} >> "${GITHUB_ENV}"
# This is the production-effect boundary: schema, pruners and workers start here.
gcloud run deploy "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--image "${IMAGE}" --tag "${tag}" --revision-suffix "${tag}" \
--remove-env-vars ORCA_PUSH_MODE --no-traffic --quiet
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--remove-tags "${VALIDATION_TAG}" --quiet
gcloud run revisions delete "${VALIDATION_REVISION}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet
echo "VALIDATION_RETIRED=true" >> "${GITHUB_ENV}"
revision="$(gcloud run revisions describe "${SERVICE_NAME}-${tag}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json)"
jq -e --arg image "${IMAGE}" --arg account "${PUSH_RUNTIME_SERVICE_ACCOUNT}" \
--arg ceiling "${PUSH_MAX_INSTANCES}" --arg floor "${PUSH_MIN_INSTANCES}" \
--slurpfile prior "${RUNNER_TEMP}/push-rollback-revision.json" '
def shape: del(.containers[0].image) |
.containers[0].env = ((.containers[0].env // []) |
map(select(.name != "ORCA_PUSH_MODE")) | sort_by(.name));
.spec.containers[0].image == $image and .spec.serviceAccountName == $account and
all(.spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE") and
(.spec | shape) == ($prior[0].spec | shape) and
.metadata.annotations["autoscaling.knative.dev/maxScale"] == $ceiling and
(.metadata.annotations["autoscaling.knative.dev/minScale"] | tonumber) >= ($floor | tonumber)' \
<<< "${revision}" > /dev/null
candidate="$(gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
| jq -er --arg tag "${tag}" '[.status.traffic[] | select(.tag == $tag)]
| if length == 1 then .[0] else error("active candidate is not unique") end')"
test "$(jq -r '.revisionName' <<< "${candidate}")" = "${SERVICE_NAME}-${tag}"
url="$(jq -er '.url' <<< "${candidate}")"
[[ "${url}" =~ ^https://[^/]+$ ]]
curl --fail --silent --show-error --max-time 10 "${url}/ready" | jq -e '.ok == true'
curl --fail --silent --show-error --max-time 10 "${url}/health" \
| jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"'
- name: Shift all traffic to the verified candidate
shell: bash
run: |
@@ -275,9 +350,14 @@ jobs:
echo "Revision: \`${CANDIDATE_REVISION}\`"
echo
echo "Image: \`${IMAGE_DIGEST}\`"
echo "Known-good image: \`${ROLLBACK_IMAGE}\`"
echo
echo "Rollback: \`gcloud run services update-traffic ${SERVICE_NAME}" \
"--region ${GCP_REGION} --to-revisions ${ROLLBACK_REVISION}=100\`"
echo "Recovery: deploy \`${ROLLBACK_IMAGE}\` as a new revision with" \
"\`--remove-env-vars ORCA_PUSH_MODE --no-traffic --tag <unique-recovery-tag> --revision-suffix <unique-recovery-suffix>\`."
echo 'Verify its exact digest, configuration, readiness and active mode, then promote and check the public origin.'
echo 'Only then remove obsolete tags and delete rejected/previous revisions; never delete the latest revision.'
echo 'The previous revision is retired after public checks; retain this immutable image for recovery under the rollout lease.'
echo "Activation attempted: ${ACTIVATION_ATTEMPTED:-false}; traffic rollback cannot undo schema or deliveries."
} >> "${GITHUB_STEP_SUMMARY}"
- name: Verify the public origin after the shift
@@ -289,7 +369,8 @@ jobs:
"${PUSH_ORIGIN}/ready" || true)"
if test "${code}" = 200; then
curl --fail --silent --show-error --max-time 10 "${PUSH_ORIGIN}/health" \
| jq -e '.ok == true and .deliveryProtocol == 2' > /dev/null
| jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"' > /dev/null
echo "ROLLOUT_VERIFIED=true" >> "${GITHUB_ENV}"
echo "${PUSH_ORIGIN} is ready after ${attempt} attempt(s)"
exit 0
fi
@@ -303,7 +384,7 @@ jobs:
# is not a failure to deploy, it is a live gateway that has to go back, so the traffic move
# is undone here rather than left to whoever reads the run.
- name: Roll traffic back to the previous revision
if: ${{ (failure() || cancelled()) && env.TRAFFIC_SHIFT_ATTEMPTED == 'true' }}
if: ${{ (failure() || cancelled()) && env.TRAFFIC_SHIFT_ATTEMPTED == 'true' && env.ROLLOUT_VERIFIED != 'true' }}
shell: bash
run: |
set -euo pipefail
@@ -324,19 +405,107 @@ jobs:
echo '### Push gateway rolled back'
echo
echo "Traffic returned to \`${ROLLBACK_REVISION}\`; the candidate" \
"\`${CANDIDATE_REVISION}\` no longer serves."
"\`${CANDIDATE_REVISION}\` no longer serves HTTP; deletion below must stop its workers."
} >> "${GITHUB_STEP_SUMMARY}"
# Why: a candidate that never took traffic is a revision holding a warm floor and a Cloud
# SQL pool for nothing. Its tag comes off first, because Cloud Run refuses to delete a
# revision a traffic target still names, and clearing CANDIDATE_TAG makes the always() tag
# step below a no-op rather than a second failure.
- name: Delete the rejected candidate revision
if: ${{ (failure() || cancelled()) && (env.TRAFFIC_SHIFT_ATTEMPTED != 'true' || env.TRAFFIC_ROLLED_BACK == 'true') }}
# Deleting a revision does not restore the service template that Terraform reconciles.
- name: Restore the known-good service template
if: ${{ (failure() || cancelled()) && env.VALIDATION_DEPLOY_ATTEMPTED == 'true' && env.ROLLOUT_VERIFIED != 'true' }}
shell: bash
run: |
set -euo pipefail
test -n "${CANDIDATE_REVISION:-}" || exit 0
test "${TRAFFIC_SHIFT_ATTEMPTED:-false}" != true || test "${TRAFFIC_ROLLED_BACK:-false}" = true
test -n "${ROLLBACK_IMAGE}"
# A partial activation may leave validation plus active; free one slot before recovery.
latest="$(gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--format='value(status.latestCreatedRevisionName)')"
test -n "${latest}"
if test "${VALIDATION_RETIRED:-false}" != true && test "${latest}" != "${VALIDATION_REVISION}"; then
existing="$(gcloud run revisions list --service "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--filter "metadata.name=${VALIDATION_REVISION}" --format='value(metadata.name)')"
if test -n "${existing}"; then
test "${existing}" = "${VALIDATION_REVISION}"
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--remove-tags "${VALIDATION_TAG}" --quiet
gcloud run revisions delete "${VALIDATION_REVISION}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet
fi
fi
tag="r${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
echo "RECOVERY_TAG=${tag}" >> "${GITHUB_ENV}"
echo "TEMPLATE_RECOVERY_REVISION=${SERVICE_NAME}-${tag}" >> "${GITHUB_ENV}"
echo "Template recovery attempted: ${SERVICE_NAME}-${tag}, image ${ROLLBACK_IMAGE}." \
>> "${GITHUB_STEP_SUMMARY}"
# Known-good schema/workers can run here even though HTTP stays on the old revision.
gcloud run deploy "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--image "${ROLLBACK_IMAGE}" --revision-suffix "${tag}" --tag "${tag}" \
--remove-env-vars ORCA_PUSH_MODE --no-traffic --quiet
gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
> "${RUNNER_TEMP}/push-recovered-service.json"
gcloud run revisions describe "${SERVICE_NAME}-${tag}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
> "${RUNNER_TEMP}/push-recovery-revision.json"
jq -e --arg image "${ROLLBACK_IMAGE}" --arg serving "${ROLLBACK_REVISION}" \
--arg floor "${PUSH_MIN_INSTANCES}" --arg ceiling "${PUSH_MAX_INSTANCES}" \
--slurpfile prior "${RUNNER_TEMP}/push-rollback-revision.json" \
--slurpfile recovered "${RUNNER_TEMP}/push-recovery-revision.json" '
def shape: del(.containers[0].image) |
.containers[0].env = ((.containers[0].env // []) |
map(select(.name != "ORCA_PUSH_MODE")) | sort_by(.name));
.spec.template.spec.containers[0].image == $image and
all(.spec.template.spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE") and
$recovered[0].spec.containers[0].image == $image and
all($recovered[0].spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE") and
($recovered[0].spec | shape) == ($prior[0].spec | shape) and
.spec.template.metadata.annotations["autoscaling.knative.dev/maxScale"] == $ceiling and
(.spec.template.metadata.annotations["autoscaling.knative.dev/minScale"] | tonumber) >= ($floor | tonumber) and
([.status.traffic[] | select((.percent // 0) > 0)] |
length == 1 and .[0].revisionName == $serving and .[0].percent == 100)
' "${RUNNER_TEMP}/push-recovered-service.json" > /dev/null
echo "TEMPLATE_RESTORED=true" >> "${GITHUB_ENV}"
echo 'Known-good image and normal mode restored; previous revision still serves HTTP.' \
>> "${GITHUB_STEP_SUMMARY}"
- name: Promote and verify the known-good recovery revision
if: ${{ always() && env.TEMPLATE_RESTORED == 'true' }}
shell: bash
run: |
set -euo pipefail
url="$(jq -er --arg tag "${RECOVERY_TAG}" \
'.status.traffic[] | select(.tag == $tag) | .url' "${RUNNER_TEMP}/push-recovered-service.json")"
[[ "${url}" =~ ^https://[^/]+$ ]]
curl --fail --silent --show-error --max-time 10 "${url}/ready" | jq -e '.ok == true'
curl --fail --silent --show-error --max-time 10 "${url}/health" \
| jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"'
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--to-revisions "${TEMPLATE_RECOVERY_REVISION}=100" --quiet
serving="$(gcloud run services describe "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \
| jq -er '[.status.traffic[] | select((.percent // 0) > 0)] |
if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')"
test "${serving}" = "${TEMPLATE_RECOVERY_REVISION}"
curl --fail --silent --show-error --max-time 10 "${PUSH_ORIGIN}/ready" | jq -e '.ok == true'
curl --fail --silent --show-error --max-time 10 "${PUSH_ORIGIN}/health" \
| jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"'
echo "RECOVERY_VERIFIED=true" >> "${GITHUB_ENV}"
echo "Recovery revision ${TEMPLATE_RECOVERY_REVISION} now serves the known-good image." \
>> "${GITHUB_STEP_SUMMARY}"
- name: Delete the rejected candidate revision
if: ${{ always() && env.RECOVERY_VERIFIED == 'true' }}
shell: bash
run: |
set -euo pipefail
if test -z "${CANDIDATE_REVISION:-}"; then
echo "CANDIDATE_DELETED=true" >> "${GITHUB_ENV}"
exit 0
fi
if test -n "${CANDIDATE_TAG:-}"; then
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" \
@@ -345,11 +514,44 @@ jobs:
--quiet
echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}"
fi
gcloud run revisions delete "${CANDIDATE_REVISION}" \
--project "${GCP_PROJECT_ID}" \
--region "${GCP_REGION}" \
--quiet
echo "deleted the candidate revision ${CANDIDATE_REVISION}"
existing="$(gcloud run revisions list --service "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \
--filter "metadata.name=${CANDIDATE_REVISION}" --format='value(metadata.name)')"
if test -n "${existing}"; then
test "${existing}" = "${CANDIDATE_REVISION}"
gcloud run revisions delete "${CANDIDATE_REVISION}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet
fi
echo "CANDIDATE_DELETED=true" >> "${GITHUB_ENV}"
echo "candidate revision ${CANDIDATE_REVISION} is absent"
# Public checks commit the serving revision; cleanup failures must not roll it back.
- name: Retire previous consumers after public checks
if: ${{ always() && (env.ROLLOUT_VERIFIED == 'true' || (env.RECOVERY_VERIFIED == 'true' && env.CANDIDATE_DELETED == 'true')) }}
shell: bash
run: |
set -euo pipefail
serving="${CANDIDATE_REVISION}"
if test "${RECOVERY_VERIFIED:-false}" = true; then
serving="${TEMPLATE_RECOVERY_REVISION}"
fi
gcloud run services update-traffic "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --clear-tags --quiet
revisions="$(gcloud run revisions list --service "${SERVICE_NAME}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format='value(metadata.name)')"
while IFS= read -r revision; do
test -n "${revision}" || continue
test "${revision}" != "${serving}" || continue
case "${revision}" in
"${ROLLBACK_REVISION}"|"${VALIDATION_REVISION}"|"${CANDIDATE_REVISION}") ;;
*) echo "Unexpected revision ${revision}; manual retirement required." >&2; exit 1 ;;
esac
gcloud run revisions delete "${revision}" \
--project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet
done <<< "${revisions}"
echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}"
echo 'Obsolete revision resources retired; verify actual SQL session drain operationally.' \
>> "${GITHUB_STEP_SUMMARY}"
- name: Drop the candidate traffic tag
if: always()
@@ -283,8 +283,6 @@ export const LEASED_WORKFLOWS = named([
'operate-relay-production-rehome.yml',
production({ leaseFiles: ['operate-relay-production-rehome-job.yml'] })
],
// The gateway applies its schema at startup, so its deploy revision is the schema step.
['push-deploy.yml', production()],
['deploy-relay-asia-topology.yml', eitherEnvironment()],
['operate-relay-asia-admission.yml', eitherEnvironment()],
['deploy-relay-staging.yml', staging()],
@@ -300,6 +298,7 @@ export const LEASED_WORKFLOWS = named([
])
export const NOT_A_CLOUD_SQL_CANDIDATE = named([
['push-deploy.yml', 'Push uses dedicated SQL and its own production-push-rollout group and durable push-rollout lease.'],
[
'monitor-relay-production.yml',
'Read-only. Its identity holds monitoring, logging, Cloud SQL and compute viewer roles only, and it runs `gcloud sql instances describe`, never a mutation. It consumes no connection budget, so the durable lease would only let monitoring block a rollout and a rollout block monitoring.'
@@ -32,8 +32,7 @@ test('no workflow names the retired generic production deploy identity', async (
'deploy-relay-production.yml',
'operate-relay-asia-admission.yml',
'operate-relay-production-rehome-job.yml',
'publish-relay-production.yml',
'push-deploy.yml'
'publish-relay-production.yml'
].map((name) => relayWorkflowFile(name)).sort())
})
@@ -168,3 +167,12 @@ test('fence broker pins the production-proven Terraform planner', async () => {
const dockerfile = await source('apps/relay-fence-broker/Dockerfile')
assert.match(dockerfile, /FROM hashicorp\/terraform:1\.15\.8 AS terraform/)
})
test('push uses its dedicated identity and rollout lease', () => {
const workflow = readRelayWorkflow('push-deploy.yml')
assert.match(workflow, /PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER/)
assert.match(workflow, /PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT/)
assert.doesNotMatch(workflow, /PRODUCTION_GCP_RELAY_DEPLOY_/)
assert.match(workflow, /group: production-push-rollout/)
assert.match(workflow, /object: terraform\/state\/push-rollout\/production.lock/)
})
+505
View File
@@ -83,6 +83,91 @@
],
"demotionRule": "Keep experimental until CI soak; investigate fidelity or count failures without relaxing the row budget."
},
{
"id": "agent-session.hibernation-runtime-inventory-budget",
"title": "Hibernation skips irrelevant remote inventories and preserves fresh host evidence",
"maturity": "experimental",
"protection": "partial",
"owner": "agent-session-runtime",
"layer": "shared-and-renderer-unit",
"surfaces": ["automatic agent hibernation", "remote terminal liveness"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "remote-runtime"],
"coverageNotes": "Local hibernation, remote inventory authority and folder-workspace identity are covered by coordinator/model contracts on macOS. Daemon, SSH, WSL, relay and native platform execution boundaries and existing RPC payloads are unchanged.",
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/src/renderer/src/lib/agent-hibernation-coordinator.test.ts"
],
"invariant": "Automatic hibernation must retain two stable confirmations and fresh execution-host evidence before shutdown. Skipping a workspace with no completed agent must not authorize a newly completed pane using stale client PTYs.",
"oracle": "100 remote workspaces containing working/waiting agents issue zero runtime calls. A skipped workspace completing while another inventory awaits remains ineligible until two later host-confirmed passes, and so does a workspace that only becomes runtime-owned while an inventory is outstanding — a pass carrying no host evidence for a workspace never counts as one of its two confirmations. Folder and git workspaces hibernate the exact host PTY; rejected/truncated inventories and intervening input/output or state changes block shutdown.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/agent-hibernation-coordinator.test.ts src/renderer/src/lib/agent-hibernation-planner.test.ts src/renderer/src/lib/agent-hibernation-confirmation.test.ts src/renderer/src/lib/agent-hibernation-pane-age.test.ts src/renderer/src/lib/agent-hibernation-output-activity.test.ts src/renderer/src/lib/agent-hibernation-visibility.test.ts src/renderer/src/lib/foreground-terminal-tabs.test.ts src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts"
],
"testFiles": [
"src/renderer/src/lib/agent-hibernation-coordinator.test.ts",
"src/renderer/src/lib/agent-hibernation-planner.test.ts",
"src/renderer/src/lib/agent-hibernation-confirmation.test.ts",
"src/renderer/src/lib/agent-hibernation-pane-age.test.ts",
"src/renderer/src/lib/agent-hibernation-output-activity.test.ts",
"src/renderer/src/lib/agent-hibernation-visibility.test.ts",
"src/renderer/src/lib/foreground-terminal-tabs.test.ts",
"src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts"
],
"assertionRefs": [
{
"file": "src/renderer/src/lib/agent-hibernation-coordinator.test.ts",
"assertions": [
"does not request runtime inventories for 100 workspaces without completed agents",
"requires host evidence after a skipped workspace completes during another inventory request",
"hibernates a runtime-backed candidate in %s with fresh liveness and exact PTYs",
"fails closed on truncated runtime liveness samples",
"fails closed when fresh runtime liveness rejects after an earlier good sample"
]
},
{
"file": "src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts",
"assertions": [
"requires host evidence when a workspace becomes runtime-owned during an inventory request"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/agent-hibernation-coordinator.test.ts src/renderer/src/lib/agent-hibernation-planner.test.ts src/renderer/src/lib/agent-hibernation-confirmation.test.ts src/renderer/src/lib/agent-hibernation-pane-age.test.ts src/renderer/src/lib/agent-hibernation-output-activity.test.ts src/renderer/src/lib/agent-hibernation-visibility.test.ts src/renderer/src/lib/foreground-terminal-tabs.test.ts src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts",
"result": "passed",
"durationSeconds": 48.18,
"summary": "93 tests passed across eight coordinator, liveness-race, planner, confirmation, age, activity and visibility files."
}
],
"runtimeBudget": {
"p95Seconds": 60,
"scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95."
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial local validation; no CI soak history."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "Baseline made 101 runtime calls in the new no-completed-agent fixture; candidate makes zero. Existing confirmation, final recheck and failure-path assertions continue to pass, including completion during an outstanding request. Separately, baseline hibernated a workspace that became runtime-owned mid-inventory one tick early, taking its first confirmation from a pass planned entirely from client PTYs; candidate withholds that pass and requires two host-confirmed ones."
},
"performanceBudget": {
"required": true,
"evidence": "The 100-workspace fixture falls from 100 terminal.list calls plus one compatibility handshake to zero calls. Two-workspace confirmation/recheck fixture falls from five inventories to three. A single status scan and tab membership lookups precede existing requests. Recomputing the required-worktree set from the post-await state adds one in-memory owner resolution per workspace per tick and no RPC. No new timer, concurrency, subprocess or retry; only active when experimental automatic hibernation is enabled."
},
"knownGaps": [
"Remote contracts use mocked runtime replies; no live SSH/network-fault, Windows/Linux/WSL or relay run.",
"This improves the experimental automatic hibernation path only; it does not remove inventories for workspaces that contain completed agents."
],
"promotionCriteria": [
"Complete CI soak with zero unexplained flakes and preserve the observable oracle."
],
"demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions."
},
{
"id": "terminal-performance.osc-status-scan-budget",
"title": "OSC 9999 status bursts reuse forward terminator searches",
@@ -165,6 +250,426 @@
],
"demotionRule": "Keep experimental until CI soak; investigate output, offset, carry or search-budget failures without relaxing the oracle."
},
{
"id": "workspace-performance.ai-vault-title-input-guard",
"title": "Title synchronization skips unchanged session collections on live heartbeats",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-runtime",
"layer": "renderer-unit",
"surfaces": ["workspace title synchronization", "terminal heartbeat store subscribers"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local"],
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.ts"
],
"flakeHistory": {
"status": "not-started",
"evidence": "Initial local validation; no CI soak history."
},
"promotionCriteria": [
"Complete CI soak with zero unexplained flakes and preserve the observable oracle."
],
"demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity or resource-count assertions.",
"coverageNotes": "Actual title-sync subscriber and Zustand publications are covered on macOS. Existing fixtures include SSH/runtime owner invalidation and folder workspaces; no live remote transport was launched. The guard has no platform branches, remote execution, liveness verdict or wire change.",
"invariant": "Title-sync publications must not enumerate unchanged session collections, while title identity, provider availability, active pane, stored title and effective execution-host changes retain their previous invalidation decisions.",
"oracle": "Fifty live heartbeat publications with 500 retained and 500 sleeping records cause zero unchanged-map enumerations and no extra scheduled reconciliations. Provider identity, additions/removals, agent/pane ownership, effective host, active pane and stored title changes still invalidate.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts src/renderer/src/lib/ai-vault-tab-title-sync.test.ts src/renderer/src/store/slices/agent-status-batch.test.ts src/renderer/src/store/slices/agent-status-provider-session.test.ts src/renderer/src/store/slices/agent-status-retained-leak.test.ts src/renderer/src/store/slices/agent-status-manual-sleep-capture.test.ts"
],
"testFiles": [
"src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts",
"src/renderer/src/lib/ai-vault-tab-title-sync.test.ts",
"src/renderer/src/store/slices/agent-status-batch.test.ts",
"src/renderer/src/store/slices/agent-status-provider-session.test.ts",
"src/renderer/src/store/slices/agent-status-retained-leak.test.ts",
"src/renderer/src/store/slices/agent-status-manual-sleep-capture.test.ts"
],
"assertionRefs": [
{
"file": "src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts",
"assertions": [
"does not enumerate unchanged retained and sleeping maps during live status writes",
"still detects provider changes in %s with other maps reused",
"still checks workspace ownership after unchanged record collections",
"still checks active panes and stored titles after unchanged record collections"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts src/renderer/src/lib/ai-vault-tab-title-sync.test.ts src/renderer/src/store/slices/agent-status-batch.test.ts src/renderer/src/store/slices/agent-status-provider-session.test.ts src/renderer/src/store/slices/agent-status-retained-leak.test.ts src/renderer/src/store/slices/agent-status-manual-sleep-capture.test.ts",
"result": "passed",
"durationSeconds": 5.35,
"summary": "69 tests passed across six files, including actual subscriber scheduling and producer identity contracts."
}
],
"runtimeBudget": {
"p95Seconds": 30,
"scope": "Focused unit/subscriber suite; local runtime budget, not an established p95."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "The actual subscriber regression fails baseline with 200 unchanged-map enumerations and passes with zero. Frozen source differential matched 13,002 decisions over 6,501 transitions. Four actual producer checks passed against deeply frozen inputs."
},
"performanceBudget": {
"required": true,
"evidence": "Actual bundled production title subscriber plus Zustand, CPU per 1,000 writes, baseline to candidate: 25 live/100 retained/100 sleeping 41.642 to 0.699 ms; 100/500/500 251.076 to 5.350 ms; 500/500/500 343.025 to 59.897 ms. Zero extra title reads or schedules. Only same-reference collections skip comparisons; no new allocation, cache, timer, IO or scheduling."
},
"knownGaps": [
"No launched Electron/native-focus latency test or live external title lookup.",
"No live SSH/WSL/Linux/Windows session; production heartbeat incidence and whole-renderer latency are not established by the fixture."
]
},
{
"id": "terminal-performance.status-heartbeat-projection-budget",
"title": "Unchanged status heartbeats reuse the aggregate workspace projection",
"maturity": "experimental",
"protection": "partial",
"owner": "workspace-runtime",
"layer": "shared-and-renderer-unit",
"surfaces": ["desktop runtime graph subscriber", "terminal status heartbeat projection"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "daemon", "ssh", "remote-runtime"],
"coverageNotes": "The always-mounted desktop subscriber is covered through pure projection/reference and graph-sync contracts. All providers use the same serialized fields. WSL, platform launch, liveness, transport and mobile rendering are unaffected.",
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts"
],
"invariant": "The desktop agent-status comparison projection remains identical for every serialized field, key membership and freshness bucket while avoiding a full aggregate join when per-entry serialized content is unchanged.",
"oracle": "500 statuses with accumulated assistant previews receive 50 same-bucket heartbeat replacements with zero aggregate joins. A bucket boundary and assistant-detail change invalidate the projection. Existing reference comparisons cover field content, ordering, insertion/removal and subscriber decisions.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts src/renderer/src/runtime/sync-runtime-graph.test.ts"
],
"testFiles": [
"src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts",
"src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts",
"src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts",
"src/renderer/src/runtime/sync-runtime-graph.test.ts"
],
"assertionRefs": [
{
"file": "src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts",
"assertions": [
"does not rejoin accumulated previews for timestamp-only heartbeats",
"still rebuilds when an entry changes",
"still rebuilds when a pane is removed, even though every survivor is reused",
"still rebuilds when key membership swaps at a constant entry count",
"still rebuilds when a pane is added"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts src/renderer/src/runtime/sync-runtime-graph.test.ts",
"result": "passed",
"summary": "38 tests passed across four projection and graph-sync files; renderer typecheck passed.",
"durationSeconds": 2.7
}
],
"runtimeBudget": {
"p95Seconds": 60,
"scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95."
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial local validation; no CI soak history."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "The new heartbeat fixture failed on baseline with 50 aggregate joins; candidate performs zero and still invalidates on the next freshness bucket and new assistant detail."
},
"performanceBudget": {
"required": true,
"evidence": "50 redundant aggregate joins fall to zero. Paired production projection/equality benchmark: 500 agents with 8 KB previews improves 1.716 to 0.089 ms/update; 1000 improves 3.601 to 0.123 ms/update. Avoids rebuilding 5.06/10.11 million-character aggregate strings. Entry serialization and sorting remain bounded by the current status map; no new cache, polling or asynchronous work."
},
"knownGaps": [
"Measured production projection self-time excludes store updates and rendering; no whole-app input-latency claim.",
"Native platforms, live SSH/relay and older clients were not launched. This changes only equality-preserving cache reuse, with no wire or persisted schema change."
],
"promotionCriteria": [
"Complete CI soak with zero unexplained flakes and preserve the observable oracle."
],
"demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions."
},
{
"id": "terminal-performance.partial-escape-ground-scan",
"title": "Terminal snapshots preserve partial escapes with bounded ground-state scan work",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-runtime",
"layer": "shared-and-daemon-unit",
"surfaces": ["headless terminal output ingestion", "terminal snapshot continuity"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "daemon", "ssh", "remote-runtime"],
"coverageNotes": "Shared parser, production emulator and Session contracts cover local/daemon paths; remote snapshot semantics use the existing corruption reproduction without live transport. The scanner has no workspace or host branching; folder/git workspaces, WSL, mobile/relay and platform execution/wire boundaries are unchanged.",
"motivatingLinks": ["https://github.com/stablyai/orca/issues/7329"],
"invariant": "Partial escape tracking preserves exact pending UTF-16 and completion across chunk/snapshot boundaries while skipping ordinary ground-state text without a JavaScript per-code-unit walk.",
"oracle": "Colored ASCII and UTF-16 output retains the exact incomplete CSI tail using fewer than 32 code-unit inspections. Every split of CSI/OSC/DCS/ESC-intermediate controls and CAN/SUB/ESC aborts preserve completion; one-character echo performs zero inspections or native ESC searches. Seeded headless snapshot restore matches a renderer terminal twin.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/shared/terminal-partial-escape-tail-ground-scan.test.ts src/shared/terminal-partial-escape-tail.test.ts src/shared/terminal-partial-escape-tail.fuzz.test.ts src/main/daemon/headless-emulator.test.ts",
"ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/repro-7329-remote-snapshot-corruption.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/headless-emulator-fidelity.fuzz.test.ts"
],
"testFiles": [
"src/shared/terminal-partial-escape-tail-ground-scan.test.ts",
"src/shared/terminal-partial-escape-tail.test.ts",
"src/shared/terminal-partial-escape-tail.fuzz.test.ts",
"src/main/daemon/headless-emulator.test.ts",
"src/main/daemon/repro-7329-remote-snapshot-corruption.test.ts",
"src/main/daemon/session-shell-recovery.test.ts",
"src/main/daemon/headless-emulator-fidelity.fuzz.test.ts"
],
"assertionRefs": [
{
"file": "src/shared/terminal-partial-escape-tail-ground-scan.test.ts",
"assertions": [
"skips ordinary %s text between completed escapes",
"preserves %j through every split after ordinary text",
"handles aborts before returning to ordinary text",
"keeps one-character echo free of per-code-unit scans and escape searches"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/shared/terminal-partial-escape-tail-ground-scan.test.ts src/shared/terminal-partial-escape-tail.test.ts src/shared/terminal-partial-escape-tail.fuzz.test.ts src/main/daemon/headless-emulator.test.ts",
"result": "passed",
"durationSeconds": 2.17,
"summary": "82 tests passed across four files including 593,468 fold-fuzz cases."
},
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/repro-7329-remote-snapshot-corruption.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/headless-emulator-fidelity.fuzz.test.ts",
"result": "passed",
"durationSeconds": 36.24,
"summary": "15 tests passed across three files including 300 headless-to-renderer snapshot fidelity streams."
}
],
"runtimeBudget": {
"p95Seconds": 60,
"scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95."
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial local validation; no CI soak history."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "New ASCII/UTF-16 work budgets failed baseline at 262,160/196,624 inspected code units; candidate inspects 16 code units and 2 native ESC searches in each with exact tails and completion. External differential matched 2,710,126 cases (plus 388,416 hybrid-vs-baseline cases over the full VT alphabet with lone/split surrogates, 0 mismatches); existing fold fuzz covers 593,468 cases."
},
"performanceBudget": {
"required": true,
"evidence": "Production HeadlessEmulator.writeSync CPU for 2 MiB colored logs improves 32.321 to 26.638 ms; sparse ANSI improves 26.757 to 21.982 ms. Agent-redraw CPU improves 402.308 to 394.633 ms. Rotated 100,000-write baseline/identical-control/candidate medians: plain echo 27.340/27.015/27.822 ms, colored echo 36.009/35.556/35.981 ms. Native ESC searches advance only in ground, and only when the current code unit is not already ESC, so dense back-to-back CSI streams do not pay a search per sequence: 0.9 MiB dense SGR/CSI medians are 2.04 ms baseline, 2.56 ms search-always, 1.92 ms shipped. Existing plain-output fast path is unchanged. No new state, allocation, timer, provider call or transport behavior."
},
"knownGaps": [
"No launched Electron or live SSH/WSL/Windows/Linux process, native input or end-to-end UI-latency measurement.",
"Existing 4,096-code-unit tracking abandonment and idle-death partial-tail behavior are unchanged."
],
"promotionCriteria": [
"Complete CI soak with zero unexplained flakes and preserve the observable oracle."
],
"demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions."
},
{
"id": "terminal-performance.pending-control-storage",
"title": "Pending terminal controls release consumed output backing storage",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-runtime",
"layer": "shared-and-renderer-unit",
"surfaces": [
"terminal output ingestion",
"pending agent status frames",
"terminal preview normalization",
"terminal title tracking"
],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "daemon", "ssh", "remote-runtime"],
"coverageNotes": "Shared code-unit and production normalizer/parser tests cover provider-independent behavior on macOS; existing buffer contracts cover normalized output. Native execution, process ownership, wire formats and mobile UI are unchanged. Folder/git identity is not inspected by these string functions.",
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/src/shared/agent-status-osc.test.ts"
],
"invariant": "Small incomplete status, ANSI and title controls must not retain the larger consumed output strings they were sliced from. Every retained control fragment is routed through ownRetainedString, which preserves exact UTF-16, current caps and trimming, payload order and clean-output offsets on both the Buffer and the code-unit fallback copier.",
"oracle": "One forced-GC fixture proves the primitive itself: 32 x 1 Mi parents pinned by 32 x 4 Ki slices retain over 16 MiB, and the same tails owned retain under 4 MiB and under an eighth of the sliced figure. Each retention site is then covered deterministically by spying on ownRetainedString: the retained value is routed through it on every chunk, including growing fragments. Large trimmed ANSI/title tails preserve their introducers and newest payload units; lone surrogates and split pairs survive BEL/ST completion; the Buffer-free fallback is byte-identical.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/own-retained-string.test.ts src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-pending-retention.test.ts src/shared/agent-status-types.test.ts src/main/runtime/terminal-ansi-pending-retention.test.ts src/shared/osc-title-scan-tail-retention.test.ts src/shared/osc-title-scan-tail.test.ts src/main/runtime/terminal-tail-buffer.test.ts src/main/runtime/terminal-tail-whitespace.test.ts"
],
"testFiles": [
"src/shared/own-retained-string.test.ts",
"src/shared/agent-status-osc.test.ts",
"src/shared/agent-status-osc-pending-retention.test.ts",
"src/shared/agent-status-types.test.ts",
"src/main/runtime/terminal-ansi-pending-retention.test.ts",
"src/shared/osc-title-scan-tail-retention.test.ts",
"src/shared/osc-title-scan-tail.test.ts",
"src/main/runtime/terminal-tail-buffer.test.ts",
"src/main/runtime/terminal-tail-whitespace.test.ts"
],
"assertionRefs": [
{
"file": "src/shared/own-retained-string.test.ts",
"assertions": [
"releases the parent chunk that a retained tail was sliced from",
"round-trips %s exactly",
"leaves already-flat short strings alone",
"matches the block copier when Buffer is unavailable"
]
},
{
"file": "src/shared/agent-status-osc-pending-retention.test.ts",
"assertions": [
"routes every retained pending frame through ownRetainedString",
"keeps %i-character chunks with %i incomplete statuses byte-exact",
"preserves raw UTF-16 across an owned suffix and %j",
"drops an owned frame that grows past the pending cap"
]
},
{
"file": "src/main/runtime/terminal-ansi-pending-retention.test.ts",
"assertions": [
"routes every retained pending control through ownRetainedString",
"keeps %i-character chunks with %i incomplete statuses byte-exact",
"preserves trimming and code units for %j",
"preserves split UTF-16 through %j termination",
"keeps trimming an owned fragment that grows past the cap"
]
},
{
"file": "src/shared/osc-title-scan-tail-retention.test.ts",
"assertions": [
"routes every retained title tail through ownRetainedString",
"keeps %i-character chunks with %i incomplete titles byte-exact",
"preserves title %s introducer and exact UTF-16 at the cap",
"keeps trimming an owned title that grows past the cap"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/own-retained-string.test.ts src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-pending-retention.test.ts src/shared/agent-status-types.test.ts src/main/runtime/terminal-ansi-pending-retention.test.ts src/shared/osc-title-scan-tail-retention.test.ts src/shared/osc-title-scan-tail.test.ts src/main/runtime/terminal-tail-buffer.test.ts src/main/runtime/terminal-tail-whitespace.test.ts",
"result": "passed",
"durationSeconds": 11.4,
"summary": "114 tests passed across nine primitive, parser, payload, normalization, title and terminal-buffer files."
}
],
"runtimeBudget": {
"p95Seconds": 60,
"scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95."
},
"flakeHistory": {
"status": "not-started",
"evidence": "Initial local validation; no CI soak history. Only one fixture depends on forced GC; the three per-site retention checks are deterministic spy assertions."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "The primitive fixture measures 32.00 MB retained for plain slices against 1.13 MB owned, so its 16 MiB/4 MiB bounds fail without ownership. Removing ownRetainedString from terminal-ansi-normalization.ts reproducibly fails 'routes every retained pending control through ownRetainedString' while the ten fidelity cases still pass, confirming the spy assertions carry the retention contract rather than the fidelity ones. Exact-output equivalence against the pre-change parsers was re-checked in the differential fixtures: byte-exact clean output, trimmed tails, payload order and clean offsets are unchanged."
},
"performanceBudget": {
"required": true,
"evidence": "ownRetainedString is a Buffer utf16le round trip, measured at 0.57 us for 4 Ki code units and 21.9 us for 64 Ki, against 10.0 us and 170.3 us for the code-unit block copier it replaces (min of 12 rounds x 500, isolated processes). Strings below V8 SlicedString::kMinLength (13) are returned unchanged. Interleaved min-of-15 comparisons against the un-owned parsers: 16 Ki-char ANSI chunks with a 4 Ki tail 48.7 to 50.1 us/chunk (+2.8%); 4 Ki-char ANSI chunks with a 2 Ki tail 14.7 to 12.9 us/chunk (-12.0%); 194 Ki-char status chunks with a 64 Ki pending frame 298.1 to 360.5 us/chunk (+21.0%). Ordinary streams with no pending control never call the primitive. Plain-output paths, scheduling, provider calls, limits and trimming are unchanged."
},
"knownGaps": [
"No live Electron input-latency or Linux/Windows/WSL/SSH execution measurement; exact string and tail behavior is covered on macOS.",
"The primitive's GC fixture is experimental with no CI soak. A provider input already sliced from an unseen larger ancestor can retain that ancestor; no universal heap cap or whole-process memory reduction is claimed.",
"The renderer and mobile fallback copier is covered by a Buffer-free equivalence test, not by a real renderer bundle run.",
"Tail-buffer row retention is a larger instance of the same pattern and is not covered by this gate."
],
"promotionCriteria": [
"Complete CI soak with zero unexplained flakes and preserve the observable oracle."
],
"demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions."
},
{
"id": "terminal-performance.vertical-control-scan",
"title": "Main terminal preview scanning skips ordinary output between controls",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-runtime",
"layer": "main-runtime-unit",
"surfaces": ["terminal output ingestion", "terminal preview and read tails"],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "daemon", "ssh", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local"],
"motivatingLinks": [
"https://github.com/stablyai/orca/blob/main/.agents/skills/perf/SKILL.md"
],
"flakeHistory": {
"status": "not-started",
"evidence": "Initial local validation; no CI soak history."
},
"promotionCriteria": [
"Complete CI soak with zero unexplained flakes and preserve the observable oracle."
],
"demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity or resource-count assertions.",
"coverageNotes": "Pure shared-host string semantics and main runtime tests cover the production scanner on macOS. Folder/git workspaces, local/daemon/SSH/remote-runtime authority, wire content and mobile/relay behavior are unchanged. No live native remote session was launched.",
"invariant": "Main terminal preview/read tails choose the same append/redraw path while ordinary text is skipped without a JavaScript code-unit walk. Complete string-control payloads are not interpreted as vertical controls; incomplete controls stop at the same point.",
"oracle": "Plain, colored and vertical-CSI output uses fewer than 16 code-unit inspections with the same recognition result. Canonical and noncanonical CSI, embedded CSI in OSC/DCS/SOS/PM/APC, ESC inside CSI parameter bytes, back-to-back controls and incomplete sequences preserve decisions. Production normalization and tail append retain exact cursor-up redraw rows.",
"commands": [
"ORCA_BACKGROUND_LAUNCH=1 pnpm test src/main/runtime/terminal-vertical-control-scan.test.ts"
],
"testFiles": ["src/main/runtime/terminal-vertical-control-scan.test.ts"],
"assertionRefs": [
{
"file": "src/main/runtime/terminal-vertical-control-scan.test.ts",
"assertions": [
"bounds code-unit inspections on %s output",
"preserves numeric CSI A recognition for %j",
"skips embedded CSI and stops at an incomplete %s",
"resumes scanning after a parsed control for %j",
"preserves tail rows when ordinary output is followed by a cursor-up redraw"
]
}
],
"evidenceRuns": [
{
"date": "2026-09-07",
"runner": "local",
"platform": "macos",
"command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/main/runtime/terminal-vertical-control-scan.test.ts",
"result": "passed",
"durationSeconds": 0.325,
"summary": "29 scanner work-budget, control recognition, scan-resumption and complete-tail integration tests passed."
}
],
"runtimeBudget": {
"p95Seconds": 15,
"scope": "Focused scanner and tail-contract suite; local runtime budget, not an established p95."
},
"redGreenEvidence": {
"status": "complete",
"evidence": "Baseline inspections 90,112/90,119/98,307 become 0/5/2. Frozen differential matched 204,925 predicate cases and 3,000 complete tail transitions. Broader integrated validation passed 1,308 tests with one existing skip across six files. Independent original-base normalizer with only this scanner change passed 1,298 tests with one existing skip across five files (20.55 seconds), using a temporary Vite source override; this confirms independence from the pending-storage PR."
},
"performanceBudget": {
"required": true,
"evidence": "Actual normalize, append-tail, transcript and preview pipeline: 4 MiB/64 KiB-chunk CPU medians plain 43.418 to 37.932 ms, colored 49.321 to 42.570, wide 41.022 to 32.883, dense newlines 37.112 to 31.601, no-newline 32.009 to 25.100. 10,000 sequential single-character writes into a growing partial line 190.374 to 126.820 ms. Equal-bundle rotated controls show no material tiny-echo/redraw regression. Forward native ESC search skips only ground text; no new retained state, scheduling, timer or provider calls."
},
"knownGaps": [
"No live end-to-end input latency, many-pane frame measurement or launched Electron/SSH/WSL/Linux/Windows journey.",
"Existing preview fidelity and incomplete-control handling remain unchanged; broader runtime suite has one existing skipped case."
]
},
{
"id": "terminal-performance.padded-fullscreen-redraw",
"title": "Fullscreen redraw padding does not stall terminal delivery",
+4 -4
View File
@@ -1,5 +1,5 @@
<svg xmlns="http://www.w3.org/2000/svg" width="106" height="20" role="img" aria-label="downloads: 44m">
<title>downloads: 44m</title>
<svg xmlns="http://www.w3.org/2000/svg" width="106" height="20" role="img" aria-label="downloads: 45m">
<title>downloads: 45m</title>
<linearGradient id="s" x2="0" y2="100%">
<stop offset="0" stop-color="#bbb" stop-opacity=".1"/>
<stop offset="1" stop-opacity=".1"/>
@@ -15,7 +15,7 @@
<g fill="#fff" text-anchor="middle" font-family="Verdana,Geneva,DejaVu Sans,sans-serif" text-rendering="geometricPrecision" font-size="11">
<text x="37" y="15" fill="#010101" fill-opacity=".3">downloads</text>
<text x="37" y="14">downloads</text>
<text x="90" y="15" fill="#010101" fill-opacity=".3">44m</text>
<text x="90" y="14">44m</text>
<text x="90" y="15" fill="#010101" fill-opacity=".3">45m</text>
<text x="90" y="14">45m</text>
</g>
</svg>

Before

Width:  |  Height:  |  Size: 935 B

After

Width:  |  Height:  |  Size: 935 B

@@ -0,0 +1,103 @@
import { setImmediate } from 'node:timers/promises'
import { beforeEach, describe, expect, it, vi } from 'vitest'
const { statMock } = vi.hoisted(() => ({ statMock: vi.fn() }))
vi.mock('node:fs/promises', () => ({ stat: statMock }))
import {
createWatcherProcessEventDeliveryQueue,
prepareWatcherProcessEvents
} from './parcel-watcher-event-delivery'
const delivery = { includeDirectoryMetadata: true, maxEventsPerBatch: 100 }
const directory = { isDirectory: () => true }
const events = (prefix: string, count: number) =>
Array.from({ length: count }, (_, index) => ({
type: 'update' as const,
path: `/${prefix}/${index}`
}))
beforeEach(() => {
statMock.mockReset()
})
describe('closed watcher metadata work', () => {
it('removes queued lookups before a new live subscription needs their slots', async () => {
const gate = Promise.withResolvers<void>()
statMock.mockImplementation(async (path: string) => {
if (path.startsWith('/held/')) {
await gate.promise
}
return directory
})
const held = prepareWatcherProcessEvents(events('held', 8), delivery)
await setImmediate()
expect(statMock.mock.calls.length).toBe(8)
const deliver = vi.fn(async () => undefined)
const onError = vi.fn()
const queue = createWatcherProcessEventDeliveryQueue(delivery, deliver, onError)
let live: ReturnType<typeof prepareWatcherProcessEvents> | undefined
try {
queue.enqueue(events('closed', 32))
queue.close()
live = prepareWatcherProcessEvents(events('live', 1), delivery)
gate.resolve()
expect(await live).toEqual([{ type: 'update', path: '/live/0', isDirectory: true }])
await held
await setImmediate()
expect(statMock.mock.calls.map(([path]) => path)).toEqual([
...events('held', 8).map((event) => event.path),
'/live/0'
])
expect(deliver).not.toHaveBeenCalled()
expect(onError).not.toHaveBeenCalled()
} finally {
gate.resolve()
queue.close()
await held
await live
}
})
it('lets active stats settle without starting the rest of a closed batch', async () => {
const gate = Promise.withResolvers<void>()
statMock.mockImplementation(async () => {
await gate.promise
return directory
})
const deliver = vi.fn(async () => undefined)
const onError = vi.fn()
const queue = createWatcherProcessEventDeliveryQueue(delivery, deliver, onError)
try {
queue.enqueue(events('active', 32))
await setImmediate()
expect(statMock.mock.calls.length).toBe(8)
queue.close()
gate.resolve()
await setImmediate()
expect(statMock.mock.calls.length).toBe(8)
expect(deliver).not.toHaveBeenCalled()
expect(onError).not.toHaveBeenCalled()
await prepareWatcherProcessEvents(events('next', 8), delivery)
expect(statMock.mock.calls.length).toBe(16)
} finally {
gate.resolve()
queue.close()
await setImmediate()
}
})
it('still delivers file-like invalidations for live stat failures and skips deleted paths', async () => {
statMock.mockRejectedValue(new Error('file vanished'))
expect(
await prepareWatcherProcessEvents(
[
{ type: 'update', path: '/vanished' },
{ type: 'delete', path: '/deleted' }
],
delivery
)
).toEqual([
{ type: 'update', path: '/vanished', isDirectory: false },
{ type: 'delete', path: '/deleted' }
])
expect(statMock).toHaveBeenCalledTimes(1)
})
})
+19 -46
View File
@@ -1,4 +1,6 @@
import { stat } from 'node:fs/promises'
import { PrioritySemaphore } from '../../shared/priority-semaphore'
import { mapWithConcurrency } from '../../shared/map-with-concurrency'
import type { Event as ParcelWatcherEvent } from '@parcel/watcher'
import { MAX_BATCHED_WATCHER_EVENTS } from './filesystem-watcher-event-batch'
import type {
@@ -7,70 +9,37 @@ import type {
} from './parcel-watcher-process-protocol'
const DIRECTORY_STAT_CONCURRENCY = 8
let activeDirectoryStats = 0
const directoryStatWaiters: (() => void)[] = []
const directoryStatSlots = new PrioritySemaphore(DIRECTORY_STAT_CONCURRENCY)
export type WatcherProcessEventDeliveryQueue = {
enqueue(events: readonly ParcelWatcherEvent[]): void
close(): void
}
async function acquireDirectoryStatSlot(): Promise<void> {
if (activeDirectoryStats < DIRECTORY_STAT_CONCURRENCY) {
activeDirectoryStats++
return
}
await new Promise<void>((resolve) => directoryStatWaiters.push(resolve))
}
function releaseDirectoryStatSlot(): void {
const next = directoryStatWaiters.shift()
if (next) {
// Transfer the existing slot directly so a newly arriving task cannot
// overtake this waiter and temporarily exceed the global budget.
next()
return
}
activeDirectoryStats--
}
async function statWatcherEventPath(eventPath: string): Promise<boolean> {
await acquireDirectoryStatSlot()
async function statWatcherEventPath(eventPath: string, signal?: AbortSignal): Promise<boolean> {
const release = await directoryStatSlots.acquire(0, signal)
try {
signal?.throwIfAborted()
return (await stat(eventPath)).isDirectory()
} finally {
releaseDirectoryStatSlot()
release()
}
}
async function mapWithConcurrency<T, R>(
items: readonly T[],
limit: number,
mapper: (item: T) => Promise<R>
): Promise<R[]> {
const results = Array.from<R>({ length: items.length })
let cursor = 0
const lane = async (): Promise<void> => {
while (cursor < items.length) {
const index = cursor++
results[index] = await mapper(items[index])
}
}
await Promise.all(Array.from({ length: Math.min(limit, items.length) }, lane))
return results
}
async function mapWatcherEvent(
event: ParcelWatcherEvent,
includeDirectoryMetadata: boolean
includeDirectoryMetadata: boolean,
signal?: AbortSignal
): Promise<WatcherProcessEvent> {
signal?.throwIfAborted()
if (!includeDirectoryMetadata || event.type === 'delete') {
return { type: event.type, path: event.path }
}
let isDirectory = false
try {
isDirectory = await statWatcherEventPath(event.path)
isDirectory = await statWatcherEventPath(event.path, signal)
} catch {
signal?.throwIfAborted()
// Why: a path can vanish between the native event and metadata lookup.
// Treat unknown metadata as a file-like event so parent invalidation still runs.
}
@@ -79,8 +48,10 @@ async function mapWatcherEvent(
export async function prepareWatcherProcessEvents(
events: readonly ParcelWatcherEvent[],
delivery: WatcherProcessDeliveryOptions | undefined
delivery: WatcherProcessDeliveryOptions | undefined,
signal?: AbortSignal
): Promise<WatcherProcessEvent[] | null> {
signal?.throwIfAborted()
if (delivery?.maxEventsPerBatch !== undefined && events.length > delivery.maxEventsPerBatch) {
return null
}
@@ -88,7 +59,7 @@ export async function prepareWatcherProcessEvents(
return events.map((event) => ({ type: event.type, path: event.path }))
}
return mapWithConcurrency(events, DIRECTORY_STAT_CONCURRENCY, (event) =>
mapWatcherEvent(event, true)
mapWatcherEvent(event, true, signal)
)
}
@@ -99,6 +70,7 @@ export function createWatcherProcessEventDeliveryQueue(
onError: (error: unknown) => void
): WatcherProcessEventDeliveryQueue {
const eventLimit = delivery?.maxEventsPerBatch ?? MAX_BATCHED_WATCHER_EVENTS
const controller = new AbortController()
let active = true
let draining = false
let pendingOverflow = false
@@ -119,7 +91,7 @@ export function createWatcherProcessEventDeliveryQueue(
await deliver(null)
continue
}
const prepared = await prepareWatcherProcessEvents(events, delivery)
const prepared = await prepareWatcherProcessEvents(events, delivery, controller.signal)
if (active) {
await deliver(prepared)
}
@@ -153,6 +125,7 @@ export function createWatcherProcessEventDeliveryQueue(
},
close(): void {
active = false
controller.abort()
pendingEvents = []
pendingOverflow = false
}
@@ -167,12 +167,14 @@ describe('Claude structured session handoff options', () => {
).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } })
expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true })
await vi.waitFor(async () =>
expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' })
await vi.waitFor(
async () => expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }),
{ timeout: 10_000 }
)
expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true })
await vi.waitFor(async () =>
expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' })
await vi.waitFor(
async () => expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }),
{ timeout: 10_000 }
)
expect(acquire.mock.calls[1]?.[0].options).toEqual({ model: PICKED_MODEL })
@@ -0,0 +1,115 @@
import { expect, it, vi } from 'vitest'
import { PluginAuditLog } from './plugin-audit-log'
const source = vi.hoisted(() => ({ content: '' }))
vi.mock('node:fs/promises', () => ({
readFile: async (path: string) => (path.endsWith('.1') ? '' : source.content)
}))
it('extracts the recent window without splitting all historical log records', async () => {
source.content = `${Array.from({ length: 10000 }, (_, ts) => JSON.stringify({ ts })).join('\n')}\n`
const split = vi.spyOn(String.prototype, 'split')
let entries: Awaited<ReturnType<PluginAuditLog['readRecent']>>
try {
entries = await new PluginAuditLog('/logs').readRecent(200)
expect(split.mock.calls.length).toBe(0)
} finally {
split.mockRestore()
}
expect(entries!.map((entry) => entry.ts)).toEqual(
Array.from({ length: 200 }, (_, index) => 9800 + index)
)
})
it('preserves blank, malformed, unterminated and unusual-limit selection', async () => {
source.content = '\n{"ts":1}\n\ninvalid\n \n{"ts":2}'
const original = (limit: number) =>
source.content
.split('\n')
.filter((line) => line.length > 0)
.slice(-limit)
.flatMap((line) => {
try {
return [JSON.parse(line)]
} catch {
return []
}
})
const log = new PluginAuditLog('/logs')
for (const limit of [1, 2, 3, 4, 5, 0, -1, 0.5, 1.5, Infinity, Number.NaN]) {
expect(await log.readRecent(limit)).toEqual(original(limit))
}
})
/** The pre-scan `split/filter/slice` selection, as the differential oracle. */
function referenceRecentLines(text: string, limit: number): string[] {
return text
.split('\n')
.filter((line) => line.length > 0)
.slice(-limit)
}
function makeRandom(seed: number): () => number {
let state = seed >>> 0
return () => {
state = (state * 1664525 + 1013904223) >>> 0
return state / 0x100000000
}
}
it('selects the identical line set as the full split over randomized log shapes', async () => {
const log = new PluginAuditLog('/logs')
let nonEmptyCases = 0
for (let seed = 1; seed <= 1500; seed += 1) {
const random = makeRandom(seed)
const pieces: string[] = []
const lineCount = Math.floor(random() * 12)
for (let i = 0; i < lineCount; i += 1) {
const kind = random()
// Blank lines, whitespace-only lines, CRLF rows, non-JSON rows and multi-byte
// payloads all have to land on the same boundaries the split-based scan found.
if (kind < 0.15) {
pieces.push('')
} else if (kind < 0.25) {
pieces.push(' ')
} else if (kind < 0.35) {
pieces.push('not json')
} else if (kind < 0.45) {
pieces.push(`${JSON.stringify({ ts: i, summary: 'ünïcøde ✅' })}\r`)
} else {
pieces.push(JSON.stringify({ ts: i, actor: 'plugin:x' }))
}
}
// Half the corpora end without a trailing newline (a torn final append).
source.content = pieces.join('\n') + (random() < 0.5 ? '\n' : '')
for (const limit of [1, 2, 3, 5, 200]) {
const expected = referenceRecentLines(source.content, limit).flatMap((line) => {
try {
return [JSON.parse(line)]
} catch {
return []
}
})
expect(await log.readRecent(limit), `seed ${seed} limit ${limit}`).toEqual(expected)
if (expected.length > 0) {
nonEmptyCases += 1
}
}
}
expect(nonEmptyCases).toBeGreaterThan(1000)
})
it('never yields a record split across a line boundary', async () => {
// A 200-record window from the tail of a file whose records are long and carry
// escaped newlines: every returned line must still parse to one whole record.
source.content = `${Array.from({ length: 3000 }, (_, ts) =>
JSON.stringify({ ts, summary: `line\\nwith escapes ${'x'.repeat(200)}` })
).join('\n')}\n`
const entries = await new PluginAuditLog('/logs').readRecent(200)
expect(entries).toHaveLength(200)
expect(entries.map((entry) => entry.ts)).toEqual(
Array.from({ length: 200 }, (_, index) => 2800 + index)
)
expect(entries.every((entry) => entry.summary.endsWith('x'.repeat(200)))).toBe(true)
})
+25 -2
View File
@@ -71,8 +71,8 @@ export class PluginAuditLog {
[this.rotatedFilePath, this.filePath].map((path) => readFile(path, 'utf8').catch(() => ''))
)
const text = rotated + current
const lines = text.split('\n').filter((line) => line.length > 0)
return lines.slice(-limit).flatMap((line) => {
const lines = recentAuditLines(text, limit)
return lines.flatMap((line) => {
try {
return [JSON.parse(line) as PluginAuditEntry]
} catch {
@@ -84,3 +84,26 @@ export class PluginAuditLog {
}
}
}
function recentAuditLines(text: string, limit: number): string[] {
if (!Number.isFinite(limit) || limit < 1) {
return text
.split('\n')
.filter((line) => line.length > 0)
.slice(-limit)
}
const lines: string[] = []
let end = text.length
const count = Math.trunc(limit)
while (end > 0 && lines.length < count) {
const start = text.lastIndexOf('\n', end - 1) + 1
if (start < end) {
lines.push(text.slice(start, end))
}
if (start === 0) {
break
}
end = start - 1
}
return lines.toReversed()
}
@@ -16,6 +16,7 @@ import {
computeTerminalTailWaitState,
tailGainedNewerBlockedReason
} from './terminal-wait-tail-state'
import { ownRetainedString } from '../../shared/own-retained-string'
import type { ProcessedAgentStatusChunk } from '../../shared/agent-status-osc'
import { createAgentStatusOscProcessor } from '../../shared/agent-status-osc'
import type { RuntimePtyTitleTrackerEntry } from './runtime-terminal-state-records'
@@ -32,7 +33,9 @@ export class OrcaRuntimeWithScheduleWaitBlockedCheck extends OrcaRuntimeWithOnPt
// plus the concatenation the pattern flattens anyway.
const keywordWindow = `${state.keywordCarry}${appendedText}`.toLowerCase()
const keywordHit = WAIT_BLOCKED_KEYWORD_PATTERN.test(keywordWindow)
state.keywordCarry = keywordWindow.slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS)
// Why own: this 31-char carry lives per PTY until the next chunk, and un-owned it pins the
// whole lowercased window — a full copy of the chunk.
state.keywordCarry = ownRetainedString(keywordWindow.slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS))
appendWaitBlockedCarry(state.appended, appendedText)
const elapsed = at - state.lastAt
if (keywordHit || elapsed >= WAIT_BLOCKED_CHECK_MIN_INTERVAL_MS || elapsed < 0) {
@@ -1,4 +1,5 @@
import { MAX_TAIL_PENDING_ANSI_CHARS } from './terminal-tail-limits'
import { ownRetainedString } from '../../shared/own-retained-string'
export function parseAnsiControlSequence(
value: string,
@@ -59,14 +60,9 @@ export function hasCanonicalNumericCsiParams(params: string): boolean {
return /^[0-9;]*$/.test(params)
}
const ESCAPE_CHAR_CODE = 0x1b
export function containsTerminalVerticalLineControl(value: string): boolean {
for (let index = 0; index < value.length; index += 1) {
// Why charCodeAt: `value[index]` mints a one-char string per position on every chunk.
if (value.charCodeAt(index) !== ESCAPE_CHAR_CODE) {
continue
}
// Only ESC can introduce a vertical control; ordinary output needs no code-unit walk.
for (let index = value.indexOf('\x1b'); index !== -1; index = value.indexOf('\x1b', index + 1)) {
const parsed = parseAnsiControlSequence(value, index)
if (!parsed) {
return false
@@ -105,7 +101,8 @@ export function normalizeTerminalChunk(
if (!parsed) {
return {
text: parts.join(''),
pendingAnsi: trimPendingAnsiControl(combined.slice(index))
// Own the tail so it stops pinning the consumed chunk it was sliced from.
pendingAnsi: ownRetainedString(trimPendingAnsiControl(combined.slice(index)))
}
}
if (parsed.kind === 'csi' && isTerminalPreviewLineControl(parsed)) {
@@ -0,0 +1,98 @@
import { describe, expect, it, vi } from 'vitest'
import * as ownership from '../../shared/own-retained-string'
import { normalizeTerminalChunk } from './terminal-ansi-normalization'
import { MAX_TAIL_PENDING_ANSI_CHARS } from './terminal-tail-limits'
const INCOMPLETE_STATUS = '\x1b]9999;{"state":"working","prompt":"fragment'
describe('terminal preview pending ANSI storage', () => {
it('routes every retained pending control through ownRetainedString', () => {
const own = vi.spyOn(ownership, 'ownRetainedString')
try {
const first = normalizeTerminalChunk('x'.repeat(32 * 1024) + INCOMPLETE_STATUS)
expect(own).toHaveBeenCalledTimes(1)
expect(own).toHaveBeenLastCalledWith(INCOMPLETE_STATUS)
expect(first.pendingAnsi).toBe(INCOMPLETE_STATUS)
// Ownership is unconditional: a growing fragment is re-owned on every chunk.
let pending = first.pendingAnsi
for (let index = 0; index < 8; index += 1) {
pending = normalizeTerminalChunk('x'.repeat(512), pending).pendingAnsi
}
expect(own).toHaveBeenCalledTimes(9)
expect(own).toHaveBeenLastCalledWith(pending)
} finally {
own.mockRestore()
}
})
it.each([
[16 * 1024, 256, INCOMPLETE_STATUS],
[64 * 1024, 64, INCOMPLETE_STATUS],
[1024 * 1024, 16, INCOMPLETE_STATUS],
[16 * 1024, 128, INCOMPLETE_STATUS + 'x'.repeat(16 * 1024)]
])('keeps %i-character chunks with %i incomplete statuses byte-exact', (size, count, control) => {
const tails: string[] = []
let cleanChars = 0
for (let index = 0; index < count; index++) {
const result = normalizeTerminalChunk(
String.fromCharCode(65 + (index % 26)).repeat(size) + control
)
cleanChars += result.text.length
tails.push(result.pendingAnsi)
}
expect(cleanChars).toBe(size * count)
const expected =
control.length <= MAX_TAIL_PENDING_ANSI_CHARS
? control
: control.slice(0, 2) + control.slice(-(MAX_TAIL_PENDING_ANSI_CHARS - 2))
for (const pending of tails) {
expect(pending).toBe(expected)
expect(normalizeTerminalChunk('"}\x07after', pending)).toEqual({
text: 'after',
pendingAnsi: ''
})
}
})
it.each(['\x1b]', '\x1bP', '\x1b['])('preserves trimming and code units for %j', (prefix) => {
for (const length of [4095, 4096, 4097, 16 * 1024]) {
const value = `${prefix}${'x'.repeat(length - 7)}漢\ud8001\udc00\ud83d`
const expected =
value.length <= MAX_TAIL_PENDING_ANSI_CHARS
? value
: prefix + value.slice(-(MAX_TAIL_PENDING_ANSI_CHARS - prefix.length))
// CSI parameters must stay below its final-byte range until the suffix is retained.
const input = prefix === '\x1b[' ? value.replaceAll('x', '1') : value
const expectedInput = prefix === '\x1b[' ? expected.replaceAll('x', '1') : expected
const result = normalizeTerminalChunk('a'.repeat(32 * 1024) + input)
expect(result).toEqual({
text: 'a'.repeat(32 * 1024),
pendingAnsi: expectedInput
})
}
})
it.each(['\x07', '\x1b\\'])('preserves split UTF-16 through %j termination', (terminator) => {
const pending = '\x1b]2;漢\ud800|\udc00|\ud83d'
const first = normalizeTerminalChunk('x'.repeat(16 * 1024) + pending)
expect(first.pendingAnsi).toBe(pending)
expect(normalizeTerminalChunk(`\ude00${terminator}after`, first.pendingAnsi)).toEqual({
text: 'after',
pendingAnsi: ''
})
})
it('keeps trimming an owned fragment that grows past the cap', () => {
let pending = '\x1b]2;'
for (let index = 0; index < 128; index++) {
pending = normalizeTerminalChunk('x'.repeat(512), pending).pendingAnsi
}
expect(pending).toBe(`\x1b]${'x'.repeat(MAX_TAIL_PENDING_ANSI_CHARS - 2)}`)
expect(normalizeTerminalChunk('\x07after', pending)).toEqual({
text: 'after',
pendingAnsi: ''
})
})
})
+5 -2
View File
@@ -1,4 +1,5 @@
import { containsTerminalVerticalLineControl } from './terminal-ansi-normalization'
import { ownRetainedString } from '../../shared/own-retained-string'
import { carryTerminalTailSentinelMatches } from './terminal-tail-sentinel-index'
import {
applyTerminalLineControls,
@@ -171,11 +172,13 @@ export function appendNormalizedToTailBuffer(
const pieces = processTerminalTailCompleteSegments(segments.completeSegments)
const newlyCompletedLines: string[] = []
for (const piece of pieces) {
newlyCompletedLines.push(trimTerminalLineRight(piece))
// Why own: every retained row is sliced from this chunk but outlives it by up to
// MAX_TAIL_LINES chunks, so an un-owned row pins its whole 64 KiB chunk.
newlyCompletedLines.push(ownRetainedString(trimTerminalLineRight(piece)))
}
const partialResult = applyTerminalLineControls(segments.partialSegment)
const nextPartialLine = trimTerminalLineRight(partialResult.text)
const retainedPartialLine = nextPartialLine.slice(-MAX_TAIL_PARTIAL_CHARS)
const retainedPartialLine = ownRetainedString(nextPartialLine.slice(-MAX_TAIL_PARTIAL_CHARS))
const newCompleteLines = segments.completeLineCount
const omittedNewCompleteLines = newCompleteLines - pieces.length
@@ -2,6 +2,7 @@ import {
hasCanonicalNumericCsiParams,
parseAnsiControlSequence
} from './terminal-ansi-normalization'
import { ownRetainedString } from '../../shared/own-retained-string'
import { clampTerminalPreviewCursor, trimTerminalLineRight } from './terminal-tail-line-controls'
import { MAX_TAIL_CHARS, MAX_TAIL_LINES, MAX_TAIL_PARTIAL_CHARS } from './terminal-tail-limits'
@@ -241,7 +242,11 @@ function finalizeRetainedTerminalRows(
return {
lines,
partialLine,
// Why only the partial: redraw rows are built character by character and never sliced from
// the chunk, but the partial is re-sliced from its own row on every chunk, so it alone can
// accumulate a backing string across frames. Owning the rows too costs 20-36% on TUI floods
// for no measured retention.
partialLine: ownRetainedString(partialLine),
redrawCursor,
truncated,
newCompleteLines,
@@ -0,0 +1,112 @@
import { describe, expect, it, vi } from 'vitest'
import * as ownership from '../../shared/own-retained-string'
import { appendNormalizedToTailBuffer } from './terminal-tail-buffer'
const SPINNER = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴']
/** A Claude Code / Codex spinner chunk: many CR-redraw frames, then one completed line. */
function spinnerChunk(index: number, chunkChars: number): string {
const parts: string[] = []
let length = 0
let frame = 0
while (length < chunkChars - 64) {
const piece = `\r${SPINNER[frame % SPINNER.length]} Thinking... (${frame}s) esc to interrupt`
parts.push(piece)
length += piece.length
frame += 1
}
parts.push(`\rstep ${index} done\n`)
return parts.join('')
}
function collectHeap(): number {
const gc = (globalThis as { gc?: () => void }).gc
if (!gc) {
throw new Error('global.gc unavailable - config/vitest.config.ts must pass --expose-gc')
}
void /reset/.test('reset')
gc()
gc()
return process.memoryUsage().heapUsed
}
describe('retained terminal tail row storage', () => {
it('does not pin one chunk per retained row across a spinner workload', () => {
let lines: string[] = []
let partialLine = ''
for (let index = 0; index < 4; index += 1) {
const warm = appendNormalizedToTailBuffer(lines, partialLine, spinnerChunk(index, 4096))
lines = warm.lines
partialLine = warm.partialLine
}
lines = []
partialLine = ''
const chunkChars = 64 * 1024
const chunkCount = 200
const before = collectHeap()
// Why build each chunk inside the loop: a pre-built array would pin every chunk itself.
for (let index = 0; index < chunkCount; index += 1) {
const next = appendNormalizedToTailBuffer(lines, partialLine, spinnerChunk(index, chunkChars))
lines = next.lines
partialLine = next.partialLine
}
const retained = collectHeap() - before
expect(lines).toHaveLength(chunkCount)
const tailChars = lines.reduce((sum, line) => sum + line.length, 0) + partialLine.length
expect(tailChars).toBeLessThan(8 * 1024)
// Un-owned, each of the 200 rows pins its own 64 Ki chunk: about 25 MB.
expect(retained).toBeLessThan(4 * 1024 * 1024)
expect(lines.at(-1)).toBe(`step ${chunkCount - 1} done`)
})
it('routes every retained row and partial line through ownRetainedString', () => {
const own = vi.spyOn(ownership, 'ownRetainedString')
try {
const chunk = `${'x'.repeat(16 * 1024)}\nsecond line\ntrailing partial`
const result = appendNormalizedToTailBuffer([], '', chunk)
expect(result.lines).toEqual(['x'.repeat(16 * 1024), 'second line'])
expect(result.partialLine).toBe('trailing partial')
expect(own.mock.calls.map(([value]) => value)).toEqual([
'x'.repeat(16 * 1024),
'second line',
'trailing partial'
])
} finally {
own.mockRestore()
}
})
it('owns the redraw partial line without re-owning carried rows', () => {
const own = vi.spyOn(ownership, 'ownRetainedString')
try {
const seeded = appendNormalizedToTailBuffer([], '', 'row one\nrow two\nrow three\n')
own.mockClear()
// \x1b[2A drives the multiline redraw builder rather than the plain path.
const redrawn = appendNormalizedToTailBuffer(
seeded.lines,
seeded.partialLine,
'\x1b[2A\x1b[Krewritten two'
)
expect(redrawn.lines).toEqual(['row one'])
expect(redrawn.partialLine).toBe('rewritten two')
// One call for the partial only: redraw rows are built character by character.
expect(own).toHaveBeenCalledTimes(1)
expect(own).toHaveBeenLastCalledWith('rewritten two')
} finally {
own.mockRestore()
}
})
it('keeps completed lines and the tail transcript sharing the same owned rows', () => {
const result = appendNormalizedToTailBuffer([], '', 'alpha line\nbeta line\n')
expect(result.newlyCompletedLines).toEqual(['alpha line', 'beta line'])
// The transcript keeps these exact strings, so owning them covers it transitively.
result.newlyCompletedLines.forEach((line, index) => {
expect(result.lines[index]).toBe(line)
})
})
})
@@ -0,0 +1,85 @@
import { describe, expect, it, vi } from 'vitest'
import {
containsTerminalVerticalLineControl,
normalizeTerminalChunk
} from './terminal-ansi-normalization'
import { appendNormalizedToTailBuffer } from './terminal-tail-buffer'
describe('terminal vertical-control scanning', () => {
it.each([
['plain', 'log output '.repeat(8192), false],
['nonvertical CSI', `\x1b[31m${'log output '.repeat(8192)}\x1b[0m`, false],
['vertical CSI', `${'漢字😀 output '.repeat(8192)}\x1b[2A`, true]
] as const)('bounds code-unit inspections on %s output', (_name, input, expected) => {
const charCodeAt = vi.spyOn(String.prototype, 'charCodeAt')
let actual: boolean
let inspections: number
try {
actual = containsTerminalVerticalLineControl(input)
inspections = charCodeAt.mock.calls.length
} finally {
charCodeAt.mockRestore()
}
expect(actual).toBe(expected)
expect(inspections).toBeLessThan(16)
})
it.each([
['\x1b[A', true],
['\x1b[0A', true],
['\x1b[;A', true],
['\x1b[12;34A', true],
['\x1b[?1A', false],
['\x1b[1:2A', false],
['\x1b[1 A', false],
['\x1b[1\nA', false],
['\x1b[1B', false],
['\x9b1A', false],
['\x1b', false],
['\x1b[123', false]
] as const)('preserves numeric CSI A recognition for %j', (input, expected) => {
expect(containsTerminalVerticalLineControl(input)).toBe(expected)
})
it.each([
['OSC BEL', '\x1b]2;title', '\x07'],
['OSC ST', '\x1b]2;title', '\x1b\\'],
['DCS', '\x1bPpayload', '\x1b\\'],
['SOS', '\x1bXpayload', '\x1b\\'],
['PM', '\x1b^payload', '\x1b\\'],
['APC', '\x1b_payload', '\x1b\\']
])('skips embedded CSI and stops at an incomplete %s', (_name, prefix, terminator) => {
const incomplete = `${prefix}\x1b[2A`
expect(containsTerminalVerticalLineControl(incomplete)).toBe(false)
expect(containsTerminalVerticalLineControl(`${incomplete}${terminator}ordinary`)).toBe(false)
expect(containsTerminalVerticalLineControl(`${incomplete}${terminator}\x1b[3A`)).toBe(true)
})
it.each([
// An ESC inside CSI parameter bytes is consumed by that control, not treated as a new introducer.
['\x1b[\x1b[A', false],
['\x1b[\x1b[2A\x1b[1A', true],
['\x1b[31m\x1b[1A', true],
['\x1b]0;t\x07\x1b[1A', true],
['\x1b[1A\x1b', true],
['ordinary\x1b', false],
['ordinary\x1b[0m more\x1b[1;A', true]
] as const)('resumes scanning after a parsed control for %j', (input, expected) => {
expect(containsTerminalVerticalLineControl(input)).toBe(expected)
})
it('preserves tail rows when ordinary output is followed by a cursor-up redraw', () => {
const first = appendNormalizedToTailBuffer([], '', 'first\nold\n')
const normalized = normalizeTerminalChunk('\x1b[1A\x1b[2K\x1b[32mnew\x1b[0m\n')
const next = appendNormalizedToTailBuffer(
first.lines,
first.partialLine,
normalized.text,
first.redrawCursor
)
expect(next.lines).toEqual(['first', 'new'])
expect(next.partialLine).toBe('')
})
})
@@ -0,0 +1,50 @@
import { describe, expect, it, vi } from 'vitest'
import * as ownership from '../../shared/own-retained-string'
import { OrcaRuntimeWithScheduleWaitBlockedCheck } from './orca-runtime-schedule-wait-blocked-check'
import { WAIT_BLOCKED_KEYWORD_CARRY_CHARS } from './orca-runtime-postlude'
import type { createWaitBlockedCheckState } from './wait-blocked-check-state'
type ScheduleHost = {
waitBlockedCheckStateByPtyId: Map<string, ReturnType<typeof createWaitBlockedCheckState>>
runWaitBlockedCheck: () => void
scheduleWaitBlockedCheck: (ptyId: string, appendedText: string, at: number) => void
}
function createScheduleHost(): ScheduleHost {
const prototype = OrcaRuntimeWithScheduleWaitBlockedCheck.prototype as unknown as ScheduleHost
return {
waitBlockedCheckStateByPtyId: new Map(),
runWaitBlockedCheck: () => {},
scheduleWaitBlockedCheck: prototype.scheduleWaitBlockedCheck
}
}
describe('wait-blocked keyword carry storage', () => {
it('owns the carry so it stops pinning the lowercased chunk window', () => {
const own = vi.spyOn(ownership, 'ownRetainedString')
try {
const host = createScheduleHost()
const chunk = `${'Building Project '.repeat(4096)}tail-marker-text`
host.scheduleWaitBlockedCheck('pty-1', chunk, 0)
const carry = host.waitBlockedCheckStateByPtyId.get('pty-1')?.keywordCarry
expect(carry).toBe(chunk.toLowerCase().slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS))
expect(carry).toHaveLength(WAIT_BLOCKED_KEYWORD_CARRY_CHARS)
expect(own).toHaveBeenCalledTimes(1)
expect(own).toHaveBeenLastCalledWith(carry)
} finally {
own.mockRestore()
}
})
it('keeps the carry joined to the next chunk so split keywords still match', () => {
const host = createScheduleHost()
host.scheduleWaitBlockedCheck('pty-2', `${'x'.repeat(8 * 1024)}press`, 0)
const carry = host.waitBlockedCheckStateByPtyId.get('pty-2')?.keywordCarry
expect(carry?.endsWith('press')).toBe(true)
host.scheduleWaitBlockedCheck('pty-2', ' ENTER to continue', 1)
expect(host.waitBlockedCheckStateByPtyId.get('pty-2')?.keywordCarry).toBe(
`${carry} enter to continue`.slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS)
)
})
})
@@ -304,3 +304,84 @@ describe('resolveOpenedMarkdownDocuments', () => {
expect(authorizeExternalPath).not.toHaveBeenCalled()
})
})
describe('OsOpenedMarkdownFileState delivery cap', () => {
it('stops merging an OS file batch at the pending delivery cap', () => {
const state = new OsOpenedMarkdownFileState()
const includes = vi.spyOn(Array.prototype, 'includes')
const warn = vi.spyOn(console, 'warn').mockImplementation(() => {})
let probes: number
try {
state.captureFilePaths(
Array.from({ length: 10000 }, (_, index) => resolve(`/notes/${index}.md`))
)
probes = includes.mock.calls.length
} finally {
includes.mockRestore()
warn.mockRestore()
}
expect(probes).toBeLessThan(100)
expect(state.consume()).toEqual(
Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) =>
resolve(`/notes/${index}.md`)
)
)
})
// Why: the cap drops files the user explicitly asked to open. Pin which end is
// dropped (the tail, in shell order) and that the loss is reported, not silent.
it('keeps the first paths in shell order and reports the dropped tail', () => {
const state = new OsOpenedMarkdownFileState()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => {})
const total = MAX_PENDING_OS_OPENED_MARKDOWN_FILES + 8
const paths = Array.from({ length: total }, (_, index) =>
resolve(`/notes/${String(index).padStart(3, '0')}.md`)
)
try {
expect(state.captureFilePaths(paths)).toBe(true)
expect(warn).toHaveBeenCalledWith(
expect.stringContaining(`Dropped 8 of ${total} OS-opened markdown files`)
)
} finally {
warn.mockRestore()
}
expect(state.consume()).toEqual(paths.slice(0, MAX_PENDING_OS_OPENED_MARKDOWN_FILES))
})
it('stays silent for a batch that fits under the cap', () => {
const state = new OsOpenedMarkdownFileState()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => {})
try {
state.captureFilePaths(
Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) =>
resolve(`/notes/${index}.md`)
)
)
expect(warn).not.toHaveBeenCalled()
} finally {
warn.mockRestore()
}
})
it('reports a drop when an already-full queue rejects a later batch', () => {
const state = new OsOpenedMarkdownFileState()
const warn = vi.spyOn(console, 'warn').mockImplementation(() => {})
try {
state.captureFilePaths(
Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) =>
resolve(`/first/${index}.md`)
)
)
expect(warn).not.toHaveBeenCalled()
state.captureFilePaths([resolve('/second/a.md'), resolve('/second/b.md')])
expect(warn).toHaveBeenCalledWith(expect.stringContaining('Dropped 2 of 2'))
} finally {
warn.mockRestore()
}
expect(state.consume()).toEqual(
Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) =>
resolve(`/first/${index}.md`)
)
)
})
})
+13 -1
View File
@@ -105,11 +105,23 @@ export class OsOpenedMarkdownFileState {
return false
}
const merged = [...this.pending]
for (const filePath of filePaths) {
let index = 0
for (; index < filePaths.length; index++) {
if (merged.length >= MAX_PENDING_OS_OPENED_MARKDOWN_FILES) {
break
}
const filePath = filePaths[index]!
if (!merged.includes(filePath)) {
merged.push(filePath)
}
}
if (index < filePaths.length) {
// Why logged: the cap drops the tail of an oversized selection, and a file the
// user explicitly asked to open must not vanish without leaving a trace.
console.warn(
`[os-open] Dropped ${filePaths.length - index} of ${filePaths.length} OS-opened markdown files; the pending queue is capped at ${MAX_PENDING_OS_OPENED_MARKDOWN_FILES}.`
)
}
this.pending = merged.slice(0, MAX_PENDING_OS_OPENED_MARKDOWN_FILES)
publish?.()
return true
+10 -1
View File
@@ -34,6 +34,14 @@ function isAlive(pid: number): boolean {
}
}
// Teardown is asynchronous; poll instead of guessing how long the job takes.
async function waitUntilDead(pid: number, timeoutMs = 30_000): Promise<void> {
const deadline = Date.now() + timeoutMs
while (isAlive(pid) && Date.now() < deadline) {
await sleep(50)
}
}
describeOnWindows('ConPTY job ownership', () => {
const spawned: IPty[] = []
@@ -108,7 +116,8 @@ describeOnWindows('ConPTY job ownership', () => {
expect(isAlive(grandchildPid)).toBe(true)
expect(terminatePtyJob(proc)).toBe('terminated')
await sleep(1_500)
await waitUntilDead(proc.pid)
await waitUntilDead(grandchildPid)
expect(isAlive(proc.pid)).toBe(false)
expect(isAlive(grandchildPid)).toBe(false)
@@ -93,31 +93,16 @@ vi.mock('@/components/ui/popover', () => ({
let container: HTMLDivElement
let root: Root
function createButton(): HTMLButtonElement {
const button = container.querySelector<HTMLButtonElement>('[aria-label="Create"]')
function headerButton(label: string): HTMLButtonElement {
const button = container.querySelector<HTMLButtonElement>(`[aria-label="${label}"]`)
if (!button) {
throw new Error('Create button not rendered')
throw new Error(`Header button not rendered: ${label}`)
}
return button
}
async function openCreateMenu(): Promise<void> {
await act(async () => {
// Why not click(): the Radix trigger opens on pointerdown, which happy-dom does not synthesize.
createButton().dispatchEvent(
new window.PointerEvent('pointerdown', { bubbles: true, button: 0 })
)
})
}
function createMenuItem(label: string): HTMLElement {
const item = [...document.querySelectorAll<HTMLElement>('[role="menuitem"]')].find((candidate) =>
candidate.textContent?.includes(label)
)
if (!item) {
throw new Error(`Create menu item not rendered: ${label}`)
}
return item
function createButton(): HTMLButtonElement {
return headerButton('New workspace')
}
beforeEach(() => {
@@ -152,10 +137,9 @@ describe('SidebarHeader', () => {
})
expect(createButton().disabled).toBe(false)
await openCreateMenu()
await act(async () => {
createMenuItem('New workspace').click()
createButton().click()
})
expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1)
@@ -167,42 +151,53 @@ describe('SidebarHeader', () => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
await openCreateMenu()
await act(async () => {
createMenuItem('New workspace').click()
createButton().click()
})
expect(createButton().disabled).toBe(false)
expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1)
})
it('offers Add project beside New workspace under the create button', async () => {
it('reaches Add project and New workspace in one click each, with no menu', async () => {
act(() => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
await openCreateMenu()
expect(createMenuItem('New workspace')).toBeTruthy()
expect(createMenuItem('Add project')).toBeTruthy()
expect(headerButton('New workspace')).toBeTruthy()
expect(headerButton('Add project')).toBeTruthy()
expect(container.querySelector('[data-slot="dropdown-menu-trigger"]')).toBeNull()
await act(async () => {
createMenuItem('Add project').click()
headerButton('Add project').click()
})
expect(mockState.openModal).toHaveBeenCalledWith('add-repo')
expect(mocks.openWorkspaceCreationComposerWithTourHandoff).not.toHaveBeenCalled()
})
it('omits the shortcut hint when workspace creation is unassigned', async () => {
mocks.shortcutLabel.current = null
it('keeps the create button rightmost so the frequent action stays where it was', () => {
act(() => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
await openCreateMenu()
const labels = [...container.querySelectorAll<HTMLElement>('[aria-label]')]
.map((node) => node.getAttribute('aria-label'))
.filter((label): label is string => label === 'Add project' || label === 'New workspace')
expect(labels).toEqual(['Add project', 'New workspace'])
})
expect(document.querySelector('[data-slot="dropdown-menu-shortcut"]')).toBeNull()
it('advertises the workspace shortcut on the create tooltip, and omits it when unassigned', () => {
act(() => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
expect(container.textContent).toContain('⌘N')
mocks.shortcutLabel.current = null
act(() => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
expect(container.textContent).not.toContain('⌘N')
})
it('opens agent activity from the bell button', () => {
@@ -275,16 +270,16 @@ describe('SidebarHeader', () => {
)
})
it('keeps the workspace filter alongside the active bell without Add Project', () => {
it('drops both project actions in the agents view, which lists activity, not projects', () => {
mockState.sidebarBody = 'agents'
act(() => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
expect(container.querySelector('[aria-label="Turn off activity view"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Create"]')).toBeTruthy()
expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Workspace options"]')).toBeNull()
expect(container.querySelector('[aria-label="Add Project"]')).toBeNull()
expect(container.querySelector('[aria-label="Add project"]')).toBeNull()
})
it('keeps the activity bell and actions on one row at the default sidebar width', () => {
@@ -297,8 +292,8 @@ describe('SidebarHeader', () => {
expect(headerClasses.has('flex-wrap')).toBe(false)
expect(headerClasses.has('h-8')).toBe(true)
expect(container.querySelector('[aria-label="View activity"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Add Project"]')).toBeNull()
expect(container.querySelector('[aria-label="Create"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Add project"]')).toBeTruthy()
expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy()
})
it('keeps the same actions on one row at compact width', async () => {
@@ -307,15 +302,14 @@ describe('SidebarHeader', () => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
expect(container.querySelector('[aria-label="Add Project"]')).toBeNull()
expect(container.querySelector('[aria-label="Add project"]')).toBeTruthy()
expect(container.querySelector('[aria-label="View activity"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Create"]')).toBeTruthy()
expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Workspace options"]')).toBeTruthy()
expect(container.querySelector('[aria-label="More workspace actions"]')).toBeNull()
await openCreateMenu()
await act(async () => {
createMenuItem('New workspace').click()
createButton().click()
})
expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1)
})
@@ -340,8 +334,8 @@ describe('SidebarHeader', () => {
expect(container.querySelector('[aria-label="Open full Agents view"]')).toBeNull()
})
// Why: the compact overflow existed only to carry Add Project, which now lives
// under the create button, so both widths render one identical header.
// Why: the compact overflow existed only to carry Add Project, which now sits
// beside the create button, so both widths render one identical header.
it('renders the same actions on both sides of the old wide-layout breakpoint', () => {
for (const width of [234, 235]) {
mockState.sidebarWidth = width
@@ -349,8 +343,8 @@ describe('SidebarHeader', () => {
root.render(<SidebarHeader onWorkspaceBoardMenuOpenChange={vi.fn()} />)
})
expect(container.querySelector('[aria-label="More workspace actions"]')).toBeNull()
expect(container.querySelector('[aria-label="Add Project"]')).toBeNull()
expect(container.querySelector('[aria-label="Create"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Add project"]')).toBeTruthy()
expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy()
expect(container.querySelector('[aria-label="Workspace options"]')).toBeTruthy()
}
})
@@ -49,7 +49,9 @@ const SidebarHeader = React.memo(function SidebarHeader({
<div className="mt-2 flex h-8 min-w-0 items-center justify-between gap-1.5 px-2">
<div className="flex min-w-0 items-center gap-1">
<span
className="select-none pl-2 pr-0.5 text-xs font-semibold text-muted-foreground/80"
// Why truncate: the action cluster is shrink-0, so a long localized title
// (es "Espacios de trabajo") otherwise wraps out of the h-8 row.
className="min-w-0 truncate select-none pl-2 pr-0.5 text-xs font-semibold text-muted-foreground/80"
data-sidebar-section-title={groupBy === 'repo' ? 'projects' : 'workspaces'}
>
{sidebarTitle}
@@ -125,7 +127,7 @@ const SidebarHeader = React.memo(function SidebarHeader({
) : null}
<SidebarHeaderActions
onWorkspaceBoardMenuOpenChange={onWorkspaceBoardMenuOpenChange}
hideWorkspaceOptions={agentsViewActive}
agentsViewActive={agentsViewActive}
/>
</div>
</div>
@@ -1,131 +1,104 @@
import React, { useCallback, useEffect, useRef, useState } from 'react'
import { FolderPlus, GitBranchPlus, Plus } from 'lucide-react'
import React, { useCallback } from 'react'
import { FolderPlus, Plus } from 'lucide-react'
import { useAppStore } from '@/store'
import { Button } from '@/components/ui/button'
import {
DropdownMenu,
DropdownMenuContent,
DropdownMenuItem,
DropdownMenuShortcut,
DropdownMenuTrigger
} from '@/components/ui/dropdown-menu'
import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip'
import { formatOptionalPrimaryShortcutLabel } from '@/hooks/useShortcutLabel'
import { translate } from '@/i18n/i18n'
import { openWorkspaceCreationComposerWithTourHandoff } from '../contextual-tours/workspace-creation-tour-handoff'
import SidebarWorkspaceOptionsMenu from './SidebarWorkspaceOptionsMenu'
function SidebarCreateMenu({
function AddProjectButton({
preserveWorkspaceBoardOpen
}: {
preserveWorkspaceBoardOpen: boolean
}): React.JSX.Element {
const openModal = useAppStore((s) => s.openModal)
const keybindings = useAppStore((s) => s.keybindings)
const [open, setOpen] = useState(false)
const menuContentRef = useRef<HTMLDivElement | null>(null)
// Why primary: workspace.create binds both Mod+N and Mod+Shift+N, and listing
// every alias in a two-row menu reads as noise rather than help.
const newWorktreeShortcutLabel = formatOptionalPrimaryShortcutLabel(
'workspace.create',
keybindings
const label = translate('auto.components.sidebar.SidebarHeader.addProject', 'Add project')
return (
<Tooltip>
<TooltipTrigger asChild>
<Button
variant="ghost"
size="icon-xs"
type="button"
className="text-muted-foreground"
aria-label={label}
data-workspace-board-preserve-open={preserveWorkspaceBoardOpen ? '' : undefined}
onClick={() => openModal('add-repo')}
>
<FolderPlus className="size-3.5" strokeWidth={2.25} />
</Button>
</TooltipTrigger>
<TooltipContent side="bottom" sideOffset={6}>
{label}
</TooltipContent>
</Tooltip>
)
const boardAttr = preserveWorkspaceBoardOpen ? '' : undefined
}
// Why query, not a ref on the item: Radix wraps each item in a roving-focus Slot,
// and a second ref on that child conflicts with the one the Slot already owns.
// Why at all: Radix highlights the first item only when opened by keyboard, so a
// mouse click would otherwise leave Enter with nothing to activate.
useEffect(() => {
if (!open) {
return
}
const frame = requestAnimationFrame(() =>
menuContentRef.current?.querySelector<HTMLElement>('[role="menuitem"]')?.focus()
)
return () => cancelAnimationFrame(frame)
}, [open])
function NewWorkspaceButton({
preserveWorkspaceBoardOpen
}: {
preserveWorkspaceBoardOpen: boolean
}): React.JSX.Element {
const keybindings = useAppStore((s) => s.keybindings)
// Why primary: workspace.create binds both Mod+N and Mod+Shift+N, and listing
// every alias in a one-line tooltip reads as noise rather than help.
const shortcutLabel = formatOptionalPrimaryShortcutLabel('workspace.create', keybindings)
const label = translate('auto.components.sidebar.SidebarHeader.92154beb7e', 'New workspace')
// Why: the tour highlights this trigger, so the handoff has to fire from the
// menu item rather than the button that now only opens the menu.
// Why the tour handoff here: the tour highlights this button, and it is now
// the control that performs the action rather than one that opens a menu.
const handleCreateWorkspace = useCallback(() => {
// Why: opening after Radix tears down the menu prevents its focus restoration
// from treating the new dialog as an outside interaction.
window.setTimeout(openWorkspaceCreationComposerWithTourHandoff, 0)
openWorkspaceCreationComposerWithTourHandoff()
}, [])
return (
<DropdownMenu modal={false} open={open} onOpenChange={setOpen}>
<Tooltip>
<TooltipTrigger asChild>
<DropdownMenuTrigger asChild>
<Button
variant="ghost"
size="icon-xs"
type="button"
className="text-muted-foreground"
aria-label={translate('auto.components.sidebar.SidebarHeader.createMenu', 'Create')}
data-workspace-board-preserve-open={boardAttr}
data-contextual-tour-target="workspace-create-control"
>
<Plus className="size-3.5" strokeWidth={2.25} />
</Button>
</DropdownMenuTrigger>
</TooltipTrigger>
<TooltipContent side="bottom" sideOffset={6}>
{translate('auto.components.sidebar.SidebarHeader.createMenu', 'Create')}
</TooltipContent>
</Tooltip>
<DropdownMenuContent
side="bottom"
// Why start: the menu hangs from the button's left edge and opens rightward.
align="start"
// Keep the panel clear of the trigger: Radix opens on pointerdown, and an
// overlapping first item activates on the same click's pointerup.
sideOffset={4}
ref={menuContentRef}
className="w-52 p-1.5"
data-workspace-board-preserve-open={boardAttr}
>
<DropdownMenuItem
className="cursor-pointer gap-2.5 py-1.5"
onSelect={handleCreateWorkspace}
<Tooltip>
<TooltipTrigger asChild>
<Button
variant="ghost"
size="icon-xs"
type="button"
className="text-muted-foreground"
aria-label={label}
data-workspace-board-preserve-open={preserveWorkspaceBoardOpen ? '' : undefined}
data-contextual-tour-target="workspace-create-control"
onClick={handleCreateWorkspace}
>
{/* GitBranchPlus matches the create-workspace button on the landing screen. */}
<GitBranchPlus className="size-3.5" strokeWidth={2.25} />
{translate('auto.components.sidebar.SidebarHeader.92154beb7e', 'New workspace')}
{newWorktreeShortcutLabel ? (
<DropdownMenuShortcut>{newWorktreeShortcutLabel}</DropdownMenuShortcut>
) : null}
</DropdownMenuItem>
<DropdownMenuItem
className="cursor-pointer gap-2.5 py-1.5"
onSelect={() => openModal('add-repo')}
>
<FolderPlus className="size-3.5" strokeWidth={2.25} />
{translate('auto.components.sidebar.SidebarHeader.addProject', 'Add project')}
</DropdownMenuItem>
</DropdownMenuContent>
</DropdownMenu>
<Plus className="size-3.5" strokeWidth={2.25} />
</Button>
</TooltipTrigger>
<TooltipContent side="bottom" sideOffset={6}>
{label}
{shortcutLabel ? <span className="ml-1.5 text-background/60">{shortcutLabel}</span> : null}
</TooltipContent>
</Tooltip>
)
}
export function SidebarHeaderActions({
onWorkspaceBoardMenuOpenChange,
hideWorkspaceOptions = false
agentsViewActive = false
}: {
onWorkspaceBoardMenuOpenChange: (open: boolean) => void
hideWorkspaceOptions?: boolean
agentsViewActive?: boolean
}): React.JSX.Element {
return (
<div className="flex shrink-0 items-center gap-1">
{hideWorkspaceOptions ? null : (
<SidebarWorkspaceOptionsMenu
preserveWorkspaceBoardOpen
onMenuOpenChange={onWorkspaceBoardMenuOpenChange}
/>
<div className="flex shrink-0 items-center gap-1" data-sidebar-header-actions="">
{/* Why both hidden in the agents view: it lists activity, not projects. */}
{agentsViewActive ? null : (
<>
<SidebarWorkspaceOptionsMenu
preserveWorkspaceBoardOpen
onMenuOpenChange={onWorkspaceBoardMenuOpenChange}
/>
<AddProjectButton preserveWorkspaceBoardOpen />
</>
)}
<SidebarCreateMenu preserveWorkspaceBoardOpen />
<NewWorkspaceButton preserveWorkspaceBoardOpen />
</div>
)
}
@@ -4,7 +4,7 @@ import {
Ellipsis,
Eye,
FolderInput,
FolderPlus,
FolderTree,
Plus,
Shapes,
SlidersHorizontal,
@@ -135,7 +135,8 @@ export function RepoHeaderProjectActionsMenu({
</DropdownMenuItem>
) : null}
<DropdownMenuItem onSelect={() => actions.onCreateGroupFromRepo(repo)}>
<FolderPlus className="size-3.5" />
{/* Not FolderPlus: that now means "Add project" in the sidebar header above. */}
<FolderTree className="size-3.5" />
{translate('auto.components.sidebar.WorktreeList.cbfd565f83', 'New group from project')}
</DropdownMenuItem>
{projectGroups.length > 0 ? (
-1
View File
@@ -5298,7 +5298,6 @@
"5c9c7c16aa": "Add a project to create workspaces",
"a30e34eb5c": "Close workspace board",
"views": "Sidebar view",
"createMenu": "Create",
"addProject": "Add project"
},
"SidebarNav": {
-1
View File
@@ -4412,7 +4412,6 @@
"a30e34eb5c": "Cerrar tablero del espacio de trabajo",
"spaces": "Espacios",
"views": "Vista de la barra lateral",
"createMenu": "Crear",
"addProject": "Agregar proyecto"
},
"SidebarNav": {
-1
View File
@@ -5044,7 +5044,6 @@
"49f62c5665": "Tableau des espaces de travail",
"5c9c7c16aa": "Ajoutez un projet pour créer des espaces de travail",
"a30e34eb5c": "Fermer le tableau des espaces de travail",
"createMenu": "Créer",
"addProject": "Ajouter un projet"
},
"SidebarNav": {
-1
View File
@@ -4393,7 +4393,6 @@
"a30e34eb5c": "ワークスペースボードを閉じる",
"spaces": "スペース",
"views": "サイドバービュー",
"createMenu": "作成",
"addProject": "プロジェクトを追加"
},
"SidebarNav": {
-1
View File
@@ -4398,7 +4398,6 @@
"a30e34eb5c": "워크스페이스 보드 닫기",
"spaces": "스페이스",
"views": "사이드바 보기",
"createMenu": "생성",
"addProject": "프로젝트 추가"
},
"SidebarNav": {
-2
View File
@@ -4441,7 +4441,6 @@
"a30e34eb5c": "关闭工作区板",
"spaces": "空间",
"views": "侧边栏视图",
"createMenu": "创建",
"addProject": "添加项目"
},
"SidebarNav": {
@@ -14287,7 +14286,6 @@
"filtersSection": "筛选",
"viewSection": "视图"
},
"ActivityScopeFilterControls": {
"resetScope": "显示所有主机和项目"
},
@@ -0,0 +1,197 @@
import { vi, type Mock } from 'vitest'
import type { AgentStatusEntry } from '../../../shared/agent-status-types'
import type { TerminalLayoutSnapshot, TerminalTab } from '../../../shared/terminal-tab-types'
import { useAppStore } from '@/store'
import type { AppState } from '@/store/types'
import { DEFAULT_AGENT_HIBERNATION_IDLE_MS } from './agent-hibernation-planner'
import { resetAgentHibernationCoordinatorForTests } from './agent-hibernation-coordinator'
import { hydrateDrivers } from './pane-manager/mobile-driver-state'
import { resetForegroundTerminalTabIdsForTests } from './foreground-terminal-tabs'
import { resetAgentHibernationOutputActivityForTests } from './agent-hibernation-output-activity'
import {
observeHibernationPtyBindings,
resetHibernationPaneAgeForTests
} from './agent-hibernation-pane-age'
import {
createCompatibleRuntimeStatusResponseIfNeeded,
type RuntimeEnvironmentCallRequest
} from '../runtime/runtime-compatibility-test-fixture'
import { clearRuntimeCompatibilityCacheForTests } from '../runtime/runtime-rpc-client'
export const NOW = 10_000_000
export const LEAF = '11111111-1111-4111-8111-111111111111'
export type RuntimeEnvironmentCallStub = Mock<(args: RuntimeEnvironmentCallRequest) => unknown>
export const mockRuntimeEnvironmentCall: RuntimeEnvironmentCallStub = vi.fn()
vi.stubGlobal('window', {
api: {
runtimeEnvironments: {
call: mockRuntimeEnvironmentCall
}
}
})
export function tab(): TerminalTab {
return {
id: 'tab-1',
ptyId: null,
worktreeId: 'wt-bg',
title: 'Agent',
customTitle: null,
color: null,
sortOrder: 0,
createdAt: 1
}
}
export function layout(): TerminalLayoutSnapshot {
return {
root: { type: 'leaf', leafId: LEAF },
activeLeafId: LEAF,
expandedLeafId: null,
ptyIdsByLeafId: { [LEAF]: 'pty-1' }
}
}
export function entry(): AgentStatusEntry {
return {
state: 'done',
prompt: 'ship it',
updatedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1,
stateStartedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1,
paneKey: `tab-1:${LEAF}`,
tabId: 'tab-1',
worktreeId: 'wt-bg',
agentType: 'claude',
providerSession: { key: 'session_id', id: 'session-1' },
stateHistory: []
}
}
export type HibernationShutdownStub = Mock<AppState['shutdownCompletedAgentPaneForHibernation']>
export function installEligibleState(
shutdownCompletedAgentPaneForHibernation: HibernationShutdownStub = vi.fn(),
overrides: Partial<AppState> = {}
): HibernationShutdownStub {
const e = entry()
const runtimeOwnerEnvironmentId = overrides.settings?.activeRuntimeEnvironmentId ?? undefined
useAppStore.setState({
settings: {
experimentalAgentHibernation: true,
agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS
} as never,
activeWorktreeId: 'wt-active',
repos: [],
worktreesByRepo: {
'fixture-repo': [
{
id: 'wt-bg',
repoId: 'fixture-repo',
hostId: 'local',
runtimeOwnerEnvironmentId
}
]
} as never,
detectedWorktreesByRepo: {},
tabsByWorktree: { 'wt-bg': [tab()] },
terminalLayoutsByTabId: { 'tab-1': layout() },
ptyIdsByTabId: { 'tab-1': ['pty-1'] },
agentStatusByPaneKey: { [e.paneKey]: e },
sleepingAgentSessionsByPaneKey: {},
lastTerminalInputAtByPaneKey: {},
shutdownCompletedAgentPaneForHibernation: shutdownCompletedAgentPaneForHibernation as never,
shutdownWorktreeTerminals: vi.fn() as never,
...overrides
})
// Why: a pane idle long enough to hibernate has necessarily been observed by earlier
// coordinator passes, so its PTY binding is old. Seed that here — otherwise the
// binding-age floor (which exists to stop a wake or app restart sleeping the whole
// backlog immediately) would defer every candidate on its first observed tick.
const state = useAppStore.getState()
observeHibernationPtyBindings({
tabsByWorktree: state.tabsByWorktree,
terminalLayoutsByTabId: state.terminalLayoutsByTabId,
now: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 60_000,
idleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS
})
return shutdownCompletedAgentPaneForHibernation
}
export function runtimeListResult(ptyIds: string[], truncated = false) {
return {
terminals: ptyIds.map((ptyId) => ({
handle: `handle-${ptyId}`,
ptyId,
worktreeId: 'wt-bg',
worktreePath: '/tmp/wt-bg',
branch: 'feature',
tabId: `pty:${ptyId}`,
leafId: `pty:${ptyId}`,
title: 'Agent',
connected: true,
writable: true,
lastOutputAt: null,
preview: ''
})),
totalCount: ptyIds.length,
truncated
}
}
export function installRuntimeListResponses(
...responses: (ReturnType<typeof runtimeListResult> | Error)[]
): void {
const queue = [...responses]
mockRuntimeEnvironmentCall.mockImplementation((args: RuntimeEnvironmentCallRequest) => {
const compatible = createCompatibleRuntimeStatusResponseIfNeeded(args)
if (compatible) {
return Promise.resolve(compatible)
}
if (args.method === 'terminal.list') {
const response = queue.shift() ?? runtimeListResult(['pty-1'])
if (response instanceof Error) {
return Promise.reject(response)
}
return Promise.resolve({
id: 'terminal-list',
ok: true,
result: response,
_meta: { runtimeId: 'runtime-1' }
})
}
return Promise.resolve({
id: 'default',
ok: true,
result: {},
_meta: { runtimeId: 'runtime-1' }
})
})
}
export function deferred<T>(): {
promise: Promise<T>
resolve: (value: T) => void
reject: (error: Error) => void
} {
let resolve!: (value: T) => void
let reject!: (error: Error) => void
const promise = new Promise<T>((res, rej) => {
resolve = res
reject = rej
})
return { promise, resolve, reject }
}
export function resetAgentHibernationCoordinatorFixture(): void {
resetAgentHibernationCoordinatorForTests()
clearRuntimeCompatibilityCacheForTests()
resetForegroundTerminalTabIdsForTests()
resetAgentHibernationOutputActivityForTests()
resetHibernationPaneAgeForTests()
hydrateDrivers([])
mockRuntimeEnvironmentCall.mockReset()
vi.useRealTimers()
}
@@ -1,202 +1,33 @@
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { afterEach, describe, expect, it, vi } from 'vitest'
import type { AgentStatusEntry } from '../../../shared/agent-status-types'
import type { TerminalLayoutSnapshot, TerminalTab } from '../../../shared/terminal-tab-types'
import { useAppStore } from '@/store'
import { DEFAULT_AGENT_HIBERNATION_IDLE_MS } from './agent-hibernation-planner'
import {
resetAgentHibernationCoordinatorForTests,
runAgentHibernationTick,
startAgentHibernationCoordinator
} from './agent-hibernation-coordinator'
import { hydrateDrivers, setDriverForPty } from './pane-manager/mobile-driver-state'
import {
registerVisibleTerminalTab,
resetForegroundTerminalTabIdsForTests,
setForegroundTerminalTabIds
} from './foreground-terminal-tabs'
import {
recordAgentHibernationPaneOutput,
resetAgentHibernationOutputActivityForTests
} from './agent-hibernation-output-activity'
import {
observeHibernationPtyBindings,
resetHibernationPaneAgeForTests
} from './agent-hibernation-pane-age'
import { setDriverForPty } from './pane-manager/mobile-driver-state'
import { registerVisibleTerminalTab, setForegroundTerminalTabIds } from './foreground-terminal-tabs'
import { recordAgentHibernationPaneOutput } from './agent-hibernation-output-activity'
import { createCompatibleRuntimeStatusResponseIfNeeded } from '../runtime/runtime-compatibility-test-fixture'
import { clearRuntimeCompatibilityCacheForTests } from '../runtime/runtime-rpc-client'
import type { AppState } from '@/store/types'
import {
deferred,
entry,
installEligibleState,
installRuntimeListResponses,
layout,
LEAF,
mockRuntimeEnvironmentCall,
NOW,
resetAgentHibernationCoordinatorFixture,
runtimeListResult,
tab
} from './agent-hibernation-coordinator-test-fixture'
const NOW = 10_000_000
const LEAF = '11111111-1111-4111-8111-111111111111'
const PI_TRANSCRIPT_PATH = join(tmpdir(), 'pi-session-1.jsonl')
const mockRuntimeEnvironmentCall = vi.fn()
vi.stubGlobal('window', {
api: {
runtimeEnvironments: {
call: mockRuntimeEnvironmentCall
}
}
})
function tab(): TerminalTab {
return {
id: 'tab-1',
ptyId: null,
worktreeId: 'wt-bg',
title: 'Agent',
customTitle: null,
color: null,
sortOrder: 0,
createdAt: 1
}
}
function layout(): TerminalLayoutSnapshot {
return {
root: { type: 'leaf', leafId: LEAF },
activeLeafId: LEAF,
expandedLeafId: null,
ptyIdsByLeafId: { [LEAF]: 'pty-1' }
}
}
function entry(): AgentStatusEntry {
return {
state: 'done',
prompt: 'ship it',
updatedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1,
stateStartedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1,
paneKey: `tab-1:${LEAF}`,
tabId: 'tab-1',
worktreeId: 'wt-bg',
agentType: 'claude',
providerSession: { key: 'session_id', id: 'session-1' },
stateHistory: []
}
}
function installEligibleState(
shutdownCompletedAgentPaneForHibernation = vi.fn(),
overrides: Partial<AppState> = {}
): typeof shutdownCompletedAgentPaneForHibernation {
const e = entry()
const runtimeOwnerEnvironmentId = overrides.settings?.activeRuntimeEnvironmentId ?? undefined
useAppStore.setState({
settings: {
experimentalAgentHibernation: true,
agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS
} as never,
activeWorktreeId: 'wt-active',
repos: [],
worktreesByRepo: {
'fixture-repo': [
{ id: 'wt-bg', repoId: 'fixture-repo', hostId: 'local', runtimeOwnerEnvironmentId }
]
} as never,
detectedWorktreesByRepo: {},
tabsByWorktree: { 'wt-bg': [tab()] },
terminalLayoutsByTabId: { 'tab-1': layout() },
ptyIdsByTabId: { 'tab-1': ['pty-1'] },
agentStatusByPaneKey: { [e.paneKey]: e },
sleepingAgentSessionsByPaneKey: {},
lastTerminalInputAtByPaneKey: {},
shutdownCompletedAgentPaneForHibernation: shutdownCompletedAgentPaneForHibernation as never,
shutdownWorktreeTerminals: vi.fn() as never,
...overrides
})
// Why: a pane idle long enough to hibernate has necessarily been observed by earlier
// coordinator passes, so its PTY binding is old. Seed that here — otherwise the
// binding-age floor (which exists to stop a wake or app restart sleeping the whole
// backlog immediately) would defer every candidate on its first observed tick.
const state = useAppStore.getState()
observeHibernationPtyBindings({
tabsByWorktree: state.tabsByWorktree,
terminalLayoutsByTabId: state.terminalLayoutsByTabId,
now: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 60_000,
idleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS
})
return shutdownCompletedAgentPaneForHibernation
}
function runtimeListResult(ptyIds: string[], truncated = false) {
return {
terminals: ptyIds.map((ptyId) => ({
handle: `handle-${ptyId}`,
ptyId,
worktreeId: 'wt-bg',
worktreePath: '/tmp/wt-bg',
branch: 'feature',
tabId: `pty:${ptyId}`,
leafId: `pty:${ptyId}`,
title: 'Agent',
connected: true,
writable: true,
lastOutputAt: null,
preview: ''
})),
totalCount: ptyIds.length,
truncated
}
}
function installRuntimeListResponses(
...responses: (ReturnType<typeof runtimeListResult> | Error)[]
): void {
const queue = [...responses]
mockRuntimeEnvironmentCall.mockImplementation((args: { method: string }) => {
const compatible = createCompatibleRuntimeStatusResponseIfNeeded(args)
if (compatible) {
return Promise.resolve(compatible)
}
if (args.method === 'terminal.list') {
const response = queue.shift() ?? runtimeListResult(['pty-1'])
if (response instanceof Error) {
return Promise.reject(response)
}
return Promise.resolve({
id: 'terminal-list',
ok: true,
result: response,
_meta: { runtimeId: 'runtime-1' }
})
}
return Promise.resolve({
id: 'default',
ok: true,
result: {},
_meta: { runtimeId: 'runtime-1' }
})
})
}
function deferred<T>(): {
promise: Promise<T>
resolve: (value: T) => void
reject: (error: Error) => void
} {
let resolve!: (value: T) => void
let reject!: (error: Error) => void
const promise = new Promise<T>((res, rej) => {
resolve = res
reject = rej
})
return { promise, resolve, reject }
}
afterEach(() => {
resetAgentHibernationCoordinatorForTests()
clearRuntimeCompatibilityCacheForTests()
resetForegroundTerminalTabIdsForTests()
resetAgentHibernationOutputActivityForTests()
resetHibernationPaneAgeForTests()
hydrateDrivers([])
mockRuntimeEnvironmentCall.mockReset()
vi.useRealTimers()
})
afterEach(resetAgentHibernationCoordinatorFixture)
describe('agent sleep coordinator', () => {
it('hibernates an eligible background worktree after two stable ticks', async () => {
@@ -484,40 +315,50 @@ describe('agent sleep coordinator', () => {
expect(shutdown).not.toHaveBeenCalled()
})
it('hibernates a runtime-backed candidate with fresh liveness and exact PTYs', async () => {
vi.useFakeTimers()
installRuntimeListResponses(
runtimeListResult(['pty-1']),
runtimeListResult(['pty-1']),
runtimeListResult(['pty-1'])
)
const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), {
settings: {
experimentalAgentHibernation: true,
agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS,
activeRuntimeEnvironmentId: 'runtime-1'
} as never,
ptyIdsByTabId: { 'tab-1': [] }
})
startAgentHibernationCoordinator({ intervalMs: 1000, now: () => NOW })
await vi.advanceTimersByTimeAsync(1000)
await vi.advanceTimersByTimeAsync(1000)
expect(shutdown).toHaveBeenCalledWith('wt-bg', {
paneKey: `tab-1:${LEAF}`,
tabId: 'tab-1',
leafId: LEAF,
ptyId: 'pty-1',
expectedRuntimePtyId: 'pty-1'
})
expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith(
expect.objectContaining({
method: 'terminal.list',
params: expect.objectContaining({ requireFreshPtyLiveness: true })
it.each(['wt-bg', 'folder:folder-1'])(
'hibernates a runtime-backed candidate in %s with fresh liveness and exact PTYs',
async (worktreeId) => {
vi.useFakeTimers()
const result = runtimeListResult(['pty-1'])
result.terminals[0].worktreeId = worktreeId
installRuntimeListResponses(result, result, result)
const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), {
settings: {
experimentalAgentHibernation: true,
agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS,
activeRuntimeEnvironmentId: 'runtime-1'
} as never,
folderWorkspaces: [
{
id: 'folder-1',
folderPath: tmpdir(),
executionHostId: 'runtime:runtime-1'
}
] as never,
tabsByWorktree: { [worktreeId]: [{ ...tab(), worktreeId }] },
agentStatusByPaneKey: { [entry().paneKey]: { ...entry(), worktreeId } },
ptyIdsByTabId: { 'tab-1': [] }
})
)
})
startAgentHibernationCoordinator({ intervalMs: 1000, now: () => NOW })
await vi.advanceTimersByTimeAsync(1000)
await vi.advanceTimersByTimeAsync(1000)
expect(shutdown).toHaveBeenCalledWith(worktreeId, {
paneKey: `tab-1:${LEAF}`,
tabId: 'tab-1',
leafId: LEAF,
ptyId: 'pty-1',
expectedRuntimePtyId: 'pty-1'
})
expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith(
expect.objectContaining({
method: 'terminal.list',
params: expect.objectContaining({ requireFreshPtyLiveness: true })
})
)
}
)
it('requires fresh runtime liveness for confirmation and pre-shutdown recheck', async () => {
vi.useFakeTimers()
@@ -546,7 +387,7 @@ describe('agent sleep coordinator', () => {
})
it('revalidates a confirmed pane without listing unrelated runtime worktrees', async () => {
installRuntimeListResponses(...Array.from({ length: 5 }, () => runtimeListResult(['pty-1'])))
installRuntimeListResponses(...Array.from({ length: 3 }, () => runtimeListResult(['pty-1'])))
const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), {
settings: {
experimentalAgentHibernation: true,
@@ -583,15 +424,104 @@ describe('agent sleep coordinator', () => {
const listCalls = mockRuntimeEnvironmentCall.mock.calls.filter(
([args]) => args.method === 'terminal.list'
)
// Two global confirmation samples list both worktrees; the destructive
// recheck lists only the candidate's owner: 2W + C, not 2W + C×W.
expect(listCalls).toHaveLength(5)
// Both confirmation samples and the destructive recheck query only the completed agent's owner.
expect(listCalls).toHaveLength(3)
expect(listCalls.at(-1)?.[0]).toMatchObject({
selector: 'runtime-1',
params: { worktree: expect.anything() }
})
})
it('does not request runtime inventories for 100 workspaces without completed agents', async () => {
installRuntimeListResponses()
const tabs = Array.from({ length: 100 }, (_, index) => ({
...tab(),
id: `tab-${index}`,
worktreeId: `wt-${index}`
}))
const shutdown = installEligibleState(vi.fn(), {
worktreesByRepo: {
'fixture-repo': tabs.map((t) => ({
id: t.worktreeId,
repoId: 'fixture-repo',
hostId: 'runtime:runtime-1',
runtimeOwnerEnvironmentId: 'runtime-1'
}))
} as never,
tabsByWorktree: Object.fromEntries(tabs.map((t) => [t.worktreeId, [t]])),
agentStatusByPaneKey: Object.fromEntries(
tabs.map((t, index) => [
`${t.id}:${LEAF}`,
{
...entry(),
tabId: t.id,
worktreeId: t.worktreeId,
paneKey: `${t.id}:${LEAF}`,
state: index % 2 === 0 ? 'working' : 'waiting'
}
])
)
})
await runAgentHibernationTick()
expect(mockRuntimeEnvironmentCall).not.toHaveBeenCalled()
expect(shutdown).not.toHaveBeenCalled()
})
it('requires host evidence after a skipped workspace completes during another inventory request', async () => {
const delayed = deferred<ReturnType<typeof runtimeListResult>>()
installRuntimeListResponses()
const respond = mockRuntimeEnvironmentCall.getMockImplementation()!
mockRuntimeEnvironmentCall.mockImplementation((args: { method: string }) =>
args.method === 'terminal.list'
? delayed.promise.then((result) => ({ id: 'delayed', ok: true, result }))
: respond(args)
)
const first = entry()
const second = { ...entry(), tabId: 'tab-2', paneKey: `tab-2:${LEAF}`, worktreeId: 'wt-other' }
const shutdown = installEligibleState(vi.fn(), {
worktreesByRepo: {
'fixture-repo': ['wt-bg', 'wt-other'].map((id) => ({
id,
repoId: 'fixture-repo',
hostId: 'runtime:runtime-1',
runtimeOwnerEnvironmentId: 'runtime-1'
}))
} as never,
tabsByWorktree: {
'wt-bg': [tab()],
'wt-other': [{ ...tab(), id: 'tab-2', worktreeId: 'wt-other' }]
},
agentStatusByPaneKey: {
[first.paneKey]: { ...first, state: 'working' },
[second.paneKey]: second
}
})
const tick = runAgentHibernationTick()
await vi.waitFor(() =>
expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith(
expect.objectContaining({ method: 'terminal.list' })
)
)
useAppStore.setState({ agentStatusByPaneKey: { [first.paneKey]: first } })
delayed.resolve(runtimeListResult([]))
await tick
installRuntimeListResponses()
await runAgentHibernationTick()
expect(shutdown).not.toHaveBeenCalled()
await runAgentHibernationTick()
expect(shutdown).toHaveBeenCalledTimes(1)
expect(shutdown).toHaveBeenCalledWith(
'wt-bg',
expect.objectContaining({
expectedRuntimePtyId: 'pty-1'
})
)
})
it('uses fresh store state after awaiting runtime liveness before shutdown', async () => {
vi.useFakeTimers()
const delayed = deferred<ReturnType<typeof runtimeListResult>>()
@@ -30,6 +30,7 @@ import type {
RuntimeTerminalSummary
} from '../../../shared/runtime-types'
import { getWindowParkVisible, subscribeWindowParkVisibility } from './window-park-visibility'
import { getEntryTabId } from './agent-hibernation-pane-eligibility'
export const AGENT_HIBERNATION_TICK_MS = 60 * 1000
@@ -79,7 +80,17 @@ function snapshotFromState(
terminalLayoutsByTabId: state.terminalLayoutsByTabId,
ptyIdsByTabId: state.ptyIdsByTabId,
runtimeLivePtyIdsByWorktreeId: runtimeLiveness.runtimeLivePtyIdsByWorktreeId,
runtimeLivenessRequiredWorktreeIds: runtimeLiveness.runtimeLivenessRequiredWorktreeIds,
// Why: a workspace can gain tabs or resolve its runtime owner while the inventory above
// is in flight, and the plan is built from this later state. Union the fresh targets in
// so such a workspace is required-but-absent and the planner skips it, rather than
// answering for the execution host from client PTYs. Union, never replace: dropping a
// pre-await target would narrow the fail-closed set instead of widening it.
runtimeLivenessRequiredWorktreeIds: [
...new Set([
...runtimeLiveness.runtimeLivenessRequiredWorktreeIds,
...getRuntimeLivenessTargetWorktrees(state, targetWorktreeId).keys()
])
],
mobileLockedPtyIds: [...getAllDrivers()]
.filter(([, driver]) => driver.kind === 'mobile')
.map(([ptyId]) => ptyId),
@@ -132,8 +143,23 @@ async function collectRuntimePtyLiveness(
const targets = getRuntimeLivenessTargetWorktrees(state, targetWorktreeId)
const runtimeLivePtyIdsByWorktreeId: Record<string, string[]> = {}
const runtimeLivenessRequiredWorktreeIds = [...targets.keys()]
if (targets.size === 0) {
// Why: an all-local install has nothing to ask, so it must not pay the status scan below.
return { runtimeLivePtyIdsByWorktreeId, runtimeLivenessRequiredWorktreeIds }
}
const completedTabIds = new Set<string>()
for (const entry of Object.values(state.agentStatusByPaneKey)) {
const tabId = entry?.state === 'done' ? getEntryTabId(entry) : null
if (tabId) {
completedTabIds.add(tabId)
}
}
await Promise.all(
[...targets].map(async ([worktreeId, runtimeEnvironmentId]) => {
if (!state.tabsByWorktree[worktreeId]?.some((tab) => completedTabIds.has(tab.id))) {
// Skipped owners still require host evidence if an agent completes during this pass.
return
}
try {
const result = await callRuntimeRpc<RuntimeTerminalListResult>(
{ kind: 'environment', environmentId: runtimeEnvironmentId },
@@ -0,0 +1,127 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import { useAppStore } from '@/store'
import { runAgentHibernationTick } from './agent-hibernation-coordinator'
import {
createCompatibleRuntimeStatusResponseIfNeeded,
type RuntimeEnvironmentCallRequest
} from '../runtime/runtime-compatibility-test-fixture'
import {
deferred,
entry,
installEligibleState,
layout,
LEAF,
mockRuntimeEnvironmentCall,
resetAgentHibernationCoordinatorFixture,
runtimeListResult,
tab
} from './agent-hibernation-coordinator-test-fixture'
afterEach(resetAgentHibernationCoordinatorFixture)
describe('agent sleep coordinator runtime-liveness races', () => {
it('requires host evidence when a workspace becomes runtime-owned during an inventory request', async () => {
const delayed = deferred<ReturnType<typeof runtimeListResult>>()
const lateTab = { ...tab(), id: 'tab-late', worktreeId: 'wt-late' }
const lateEntry = {
...entry(),
tabId: 'tab-late',
worktreeId: 'wt-late',
paneKey: `tab-late:${LEAF}`
}
const lateList = {
...runtimeListResult(['pty-late']),
terminals: runtimeListResult(['pty-late']).terminals.map((terminal) => ({
...terminal,
worktreeId: 'wt-late'
}))
}
let firstListPending = true
mockRuntimeEnvironmentCall.mockImplementation((args: RuntimeEnvironmentCallRequest) => {
const compatible = createCompatibleRuntimeStatusResponseIfNeeded(args)
if (compatible) {
return Promise.resolve(compatible)
}
if (args.method !== 'terminal.list') {
return Promise.resolve({ id: 'default', ok: true, result: {} })
}
const isLate = args.params?.worktree === 'id:wt-late'
if (!isLate && firstListPending) {
firstListPending = false
return delayed.promise.then((result) => ({
id: 'delayed',
ok: true,
result
}))
}
return Promise.resolve({
id: 'terminal-list',
ok: true,
result: isLate ? lateList : runtimeListResult(['pty-1'])
})
})
// Why: `wt-late` starts local-owned, so the pre-await target sample never lists it.
const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), {
worktreesByRepo: {
'fixture-repo': [
{
id: 'wt-bg',
repoId: 'fixture-repo',
hostId: 'runtime:runtime-1',
runtimeOwnerEnvironmentId: 'runtime-1'
},
{ id: 'wt-late', repoId: 'fixture-repo', hostId: 'local' }
]
} as never,
tabsByWorktree: { 'wt-bg': [tab()], 'wt-late': [lateTab] },
terminalLayoutsByTabId: {
'tab-1': layout(),
'tab-late': {
root: { type: 'leaf', leafId: LEAF },
activeLeafId: LEAF,
expandedLeafId: null,
ptyIdsByLeafId: { [LEAF]: 'pty-late' }
}
},
ptyIdsByTabId: { 'tab-1': ['pty-1'], 'tab-late': ['pty-late'] },
agentStatusByPaneKey: {
[entry().paneKey]: entry(),
[lateEntry.paneKey]: lateEntry
}
})
const tick = runAgentHibernationTick()
await vi.waitFor(() =>
expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith(
expect.objectContaining({ method: 'terminal.list' })
)
)
// Runtime ownership resolves while the `wt-bg` inventory is still outstanding.
useAppStore.setState({
worktreesByRepo: {
'fixture-repo': ['wt-bg', 'wt-late'].map((id) => ({
id,
repoId: 'fixture-repo',
hostId: 'runtime:runtime-1',
runtimeOwnerEnvironmentId: 'runtime-1'
}))
} as never
})
delayed.resolve(runtimeListResult(['pty-1']))
await tick
// The racing pass has no host evidence for `wt-late`, so it must not count as one of
// the two confirmations; hibernating on the next tick would rest on client PTYs alone.
await runAgentHibernationTick()
expect(shutdown).not.toHaveBeenCalledWith('wt-late', expect.anything())
await runAgentHibernationTick()
expect(shutdown).toHaveBeenCalledWith(
'wt-late',
expect.objectContaining({
ptyId: 'pty-late',
expectedRuntimePtyId: 'pty-late'
})
)
})
})
@@ -0,0 +1,252 @@
import { createStore } from 'zustand/vanilla'
import { describe, expect, it, vi } from 'vitest'
import type { AppState } from '@/store/types'
import type { AgentProviderSessionMetadata } from '../../../shared/agent-session-resume'
import type { TerminalTab } from '../../../shared/terminal-tab-types'
import { aiVaultTitleSyncInputsChanged } from './ai-vault-tab-title-sync-inputs'
import { startAiVaultTabTitleSync } from './ai-vault-tab-title-sync'
const COLLECTIONS = [
'agentStatusByPaneKey',
'retainedAgentsByPaneKey',
'sleepingAgentSessionsByPaneKey'
] as const
type Collection = (typeof COLLECTIONS)[number]
function makeState(count = 1): AppState {
const live: AppState['agentStatusByPaneKey'] = {}
const retained: AppState['retainedAgentsByPaneKey'] = {}
const sleeping: AppState['sleepingAgentSessionsByPaneKey'] = {}
const tabs: TerminalTab[] = []
for (let index = 0; index < count * 3; index++) {
const tab: TerminalTab = {
id: `tab-${index}`,
worktreeId: 'wt-title',
ptyId: null,
title: 'Agent',
customTitle: null,
color: null,
sortOrder: index,
createdAt: 1
}
const entry = {
paneKey: `${tab.id}:00000000-0000-4000-8000-000000000001`,
tabId: tab.id,
worktreeId: tab.worktreeId,
agentType: 'codex' as const,
providerSession: { key: 'session_id' as const, id: `session-${index}` },
state: 'done' as const,
prompt: '',
updatedAt: 1,
stateStartedAt: 1,
stateHistory: []
}
tabs.push(tab)
if (index < count) {
live[entry.paneKey] = entry
} else if (index < count * 2) {
retained[entry.paneKey] = {
entry,
tab,
worktreeId: tab.worktreeId,
agentType: 'codex',
startedAt: 1
}
} else {
sleeping[entry.paneKey] = {
...entry,
agent: 'codex',
capturedAt: 1,
origin: 'worktree-sleep'
}
}
}
return {
agentStatusByPaneKey: live,
retainedAgentsByPaneKey: retained,
sleepingAgentSessionsByPaneKey: sleeping,
tabsByWorktree: { 'wt-title': tabs },
terminalLayoutsByTabId: {},
activeWorktreeId: 'wt-title',
activeWorkspaceExecutionHostId: 'local',
worktreesByRepo: { fixture: [{ id: 'wt-title', repoId: 'fixture', hostId: 'local' }] },
detectedWorktreesByRepo: {},
folderWorkspaces: [],
repos: [],
settings: {}
} as unknown as AppState
}
function replaceProvider(
state: AppState,
collection: Collection,
providerSession: AgentProviderSessionMetadata
): AppState {
const paneKey = Object.keys(state[collection])[0]
const existing = state[collection][paneKey]
const next =
collection === 'retainedAgentsByPaneKey'
? { ...existing, entry: { ...state.retainedAgentsByPaneKey[paneKey].entry, providerSession } }
: { ...existing, providerSession }
return { ...state, [collection]: { ...state[collection], [paneKey]: next } }
}
describe('AI Vault title subscription inputs', () => {
it('does not enumerate unchanged retained and sleeping maps during live status writes', () => {
const state = makeState(500)
let unchangedEnumerations = 0
const observeEnumerations = <T extends object>(records: T): T =>
new Proxy(records, {
ownKeys(target) {
unchangedEnumerations++
return Reflect.ownKeys(target)
}
})
state.retainedAgentsByPaneKey = observeEnumerations(state.retainedAgentsByPaneKey)
state.sleepingAgentSessionsByPaneKey = observeEnumerations(state.sleepingAgentSessionsByPaneKey)
const store = createStore<AppState>(() => state)
const scheduleReconcile = vi.fn(() => () => {})
const stop = startAiVaultTabTitleSync({
getState: store.getState,
subscribe: store.subscribe,
scheduleReconcile,
resolveSessionTitles: vi.fn()
})
try {
const paneKey = Object.keys(state.agentStatusByPaneKey)[0]
for (let index = 0; index < 50; index++) {
store.setState((current) => ({
agentStatusByPaneKey: {
...current.agentStatusByPaneKey,
[paneKey]: { ...current.agentStatusByPaneKey[paneKey], updatedAt: index + 2 }
}
}))
}
expect(unchangedEnumerations).toBe(0)
expect(scheduleReconcile).toHaveBeenCalledTimes(1)
} finally {
stop()
}
})
it.each(COLLECTIONS)(
'still detects provider changes in %s with other maps reused',
(collection) => {
const state = makeState()
const paneKey = Object.keys(state[collection])[0]
const provider =
collection === 'retainedAgentsByPaneKey'
? state.retainedAgentsByPaneKey[paneKey].entry.providerSession!
: (state[collection][paneKey] as { providerSession: AgentProviderSessionMetadata })
.providerSession
for (const patch of [
{ id: 'changed' },
{ key: 'conversation_id' as const },
{ transcriptPath: '/changed/session' }
]) {
expect(
aiVaultTitleSyncInputsChanged(
replaceProvider(state, collection, { ...provider, ...patch }),
state
)
).toBe(true)
}
expect(
aiVaultTitleSyncInputsChanged(replaceProvider(state, collection, { ...provider }), state)
).toBe(false)
}
)
it.each(COLLECTIONS)('detects additions and removals in %s', (collection) => {
const state = makeState()
const records = state[collection]
const empty = { ...state, [collection]: {} }
expect(aiVaultTitleSyncInputsChanged(empty, state)).toBe(true)
expect(aiVaultTitleSyncInputsChanged(state, empty)).toBe(true)
expect(aiVaultTitleSyncInputsChanged({ ...state, [collection]: { ...records } }, state)).toBe(
false
)
})
it.each(COLLECTIONS)('detects agent and pane ownership changes in %s', (collection) => {
const state = makeState()
const records = state[collection]
const paneKey = Object.keys(records)[0]
const record = records[paneKey]
const entry =
collection === 'retainedAgentsByPaneKey'
? state.retainedAgentsByPaneKey[paneKey].entry
: record
const agentField = collection === 'sleepingAgentSessionsByPaneKey' ? 'agent' : 'agentType'
const changedRecords = [
{ ...record, [agentField]: 'claude' },
{ ...record, [agentField]: 'gemini' },
{ ...record, worktreeId: 'other' },
...['paneKey', 'tabId'].map((field) =>
collection === 'retainedAgentsByPaneKey'
? { ...record, entry: { ...entry, [field]: 'other' } }
: { ...record, [field]: 'other' }
)
]
for (const changed of changedRecords) {
expect(
aiVaultTitleSyncInputsChanged(
{ ...state, [collection]: { ...records, [paneKey]: changed } },
state
)
).toBe(true)
}
})
it('still checks workspace ownership after unchanged record collections', () => {
const state = makeState()
for (const host of ['ssh:host-1', 'runtime:server-1'] as const) {
expect(
aiVaultTitleSyncInputsChanged(
{
...state,
agentStatusByPaneKey: { ...state.agentStatusByPaneKey },
activeWorkspaceExecutionHostId: host
},
state
)
).toBe(true)
}
})
it('still checks active panes and stored titles after unchanged record collections', () => {
const state = makeState()
const agentStatusByPaneKey = { ...state.agentStatusByPaneKey }
expect(
aiVaultTitleSyncInputsChanged(
{
...state,
agentStatusByPaneKey,
terminalLayoutsByTabId: {
'tab-0': { root: null, activeLeafId: 'other', expandedLeafId: null }
}
},
state
)
).toBe(true)
const tabs = state.tabsByWorktree['wt-title']
expect(
aiVaultTitleSyncInputsChanged(
{
...state,
agentStatusByPaneKey,
tabsByWorktree: {
'wt-title': [
{
...tabs[0],
aiVaultTitle: { agent: 'codex', sessionId: 'session-0', title: 'New title' }
},
...tabs.slice(1)
]
}
},
state
)
).toBe(true)
})
})
@@ -22,6 +22,9 @@ function relevantRecordEqual<T>(
relevant: (entry: T) => boolean,
equal: (current: T, previous: T) => boolean
): boolean {
if (current === previous) {
return true
}
let currentCount = 0
let previousCount = 0
for (const [key, entry] of Object.entries(current)) {
@@ -144,4 +144,51 @@ describe('resolveWorkspaceDocAddressTarget', () => {
resolveWorkspaceDocAddressTarget(makeState(), CURRENT, '/home/alice/wt1/x.html')
).toEqual({ status: 'unsupported', message: 'no channel' })
})
it('resolves a current-workspace address without enumerating unrelated workspace roots', () => {
const state = makeState()
state.allWorktrees = vi.fn(() => {
throw new Error('unnecessary global workspace enumeration')
})
expect(
resolveWorkspaceDocAddressTarget(state, CURRENT, '/home/alice/wt1/docs/index.html').status
).toBe('workspace-doc')
expect(state.allWorktrees).not.toHaveBeenCalled()
})
// The current worktree outranks every other root, including a more specific one nested inside it.
it('keeps the current worktree ahead of a workspace nested inside it', () => {
const state = makeState()
const nestedInsideCurrent = {
...state,
allWorktrees: () => [
...state.allWorktrees(),
{ id: 'repo3::/home/alice/wt1/vendor', path: '/home/alice/wt1/vendor' }
]
} as typeof state
expect(
resolveWorkspaceDocAddressTarget(
nestedInsideCurrent,
CURRENT,
'/home/alice/wt1/vendor/index.html'
)
).toMatchObject({ status: 'workspace-doc', docLocation: { worktreeId: CURRENT } })
})
it('short-circuits for a folder workspace that is the current workspace', () => {
const state = makeState()
const folderCurrent = {
...state,
getKnownWorktreeById: (id: string) =>
id === 'folder:folder-1' ? { id: 'folder:folder-1', path: '/srv/site' } : undefined,
allWorktrees: vi.fn(() => {
throw new Error('unnecessary global workspace enumeration')
})
} as unknown as AppState
expect(
resolveWorkspaceDocAddressTarget(folderCurrent, 'folder:folder-1', '/srv/site/index.html')
).toMatchObject({ status: 'workspace-doc', docLocation: { worktreeId: 'folder:folder-1' } })
})
})
@@ -19,7 +19,6 @@ function ownedWorktreeRoots(
currentWorktreeId: string
): { worktreeId: string; root: string }[] {
const roots: { worktreeId: string; root: string }[] = []
const currentPath = state.getKnownWorktreeById(currentWorktreeId)?.path ?? null
for (const worktree of state.allWorktrees()) {
if (worktree.id !== currentWorktreeId && worktree.path) {
roots.push({ worktreeId: worktree.id, root: worktree.path })
@@ -34,7 +33,7 @@ function ownedWorktreeRoots(
// Most-specific root first (after the current worktree), so a file inside a nested workspace is
// attributed to that workspace and not to the outer one that lexically contains it too.
roots.sort((a, b) => b.root.length - a.root.length)
return currentPath ? [{ worktreeId: currentWorktreeId, root: currentPath }, ...roots] : roots
return roots
}
function planToTarget(
@@ -80,6 +79,10 @@ export function resolveWorkspaceDocAddressTarget(
if (input.split(/[\\/]+/).some((segment) => segment === '..' || segment === '.')) {
return { status: 'not-a-workspace-doc' }
}
const currentPath = state.getKnownWorktreeById(currentWorktreeId)?.path
if (currentPath && getRelativePathInsideRoot(input, currentPath)) {
return planToTarget(state, currentWorktreeId, input)
}
for (const { worktreeId, root } of ownedWorktreeRoots(state, currentWorktreeId)) {
if (getRelativePathInsideRoot(input, root)) {
return planToTarget(state, worktreeId, input)
@@ -17,7 +17,7 @@ const SURFACE_PROVIDING_CALLERS = [
'src/renderer/src/hooks/composer-state/full-creation-execution.ts',
'src/renderer/src/lib/fix-checks-agent-launch.ts',
'src/renderer/src/lib/launch-work-item-direct.ts',
'src/renderer/src/lib/worktree-creation-flow-execute.ts',
'src/renderer/src/lib/worktree-creation-structured-session.ts',
'src/renderer/src/lib/workspace-port-actions.ts',
'src/renderer/src/store/repos/repo-add-actions.ts'
]
@@ -173,21 +173,18 @@ export async function executeWorktreeCreation(
let activation: ActivateAndRevealResult | false = false
let primaryTabId: string | null
if (shouldActivateOnCompletion) {
if (shouldActivateOnCompletion && !structuredLaunch) {
activation = activateAndRevealWorktree(worktree.id, {
sidebarRevealBehavior: 'auto',
...(result.setup ? { setup: result.setup } : {}),
...(result.defaultTabs ? { defaultTabs: result.defaultTabs } : {}),
...(startupOpt ? { startup: startupOpt } : {}),
...(preparedRequest.issueCommand ? { issueCommand: preparedRequest.issueCommand } : {}),
...(backendSpawned ? { backendStartupTerminalSpawned: true } : {}),
...(structuredLaunch ? { providesInitialSurface: true } : {})
...(backendSpawned ? { backendStartupTerminalSpawned: true } : {})
})
primaryTabId = activation === false ? null : activation.primaryTabId
} else {
// The user moved on. Seed the worktree's terminal + setup in the background
// (setActiveTab only writes global focus for the active worktree, so this is
// safe) without yanking them back to it.
// Keep chat creation on its pending surface until the session is ready.
const hasExplicitTerminalWork = Boolean(
startupOpt || result.setup || preparedRequest.issueCommand || result.defaultTabs
)
@@ -10,7 +10,8 @@ const mocks = vi.hoisted(() => ({
cancelStructuredAgentLaunch: vi.fn(),
closeStructuredAgentSession: vi.fn(),
callRuntimeRpc: vi.fn(),
activateStructuredAgentSessionById: vi.fn()
activateStructuredAgentSessionById: vi.fn(),
activateAndRevealWorktree: vi.fn()
}))
vi.mock('@/store', () => ({
@@ -51,7 +52,7 @@ vi.mock('@/lib/worktree-initial-terminal-seeding', () => ({
}))
vi.mock('@/lib/worktree-activation', () => ({
activateAndRevealWorktree: vi.fn()
activateAndRevealWorktree: mocks.activateAndRevealWorktree
}))
vi.mock('@/lib/agent-trust-preflight', () => ({
@@ -170,4 +171,41 @@ describe('launchStructuredWorktreeSession', () => {
expect(releaseCallerAfterUnknownOutcome).toHaveBeenCalledOnce()
expect(mocks.unsubscribe).toHaveBeenCalledOnce()
})
it('activates the workspace before selecting a chat when creation deferred activation', async () => {
mocks.startStructuredAgentLaunch.mockReturnValue({
sessionId: 'session-1',
launchResult: Promise.resolve({ sessionId: 'session-1', fence: 1 }),
claimDefinitiveRefusalFallback: vi.fn(() => Promise.resolve(false))
})
mocks.activateAndRevealWorktree.mockReturnValue({ primaryTabId: null })
const result = await launchStructuredWorktreeSession({
creationId: 'creation-1',
request: {
repoId: 'repo-1',
name: 'routing-recovery',
setupDecision: 'run',
agent: 'codex',
pendingFirstAgentMessageRename: false,
note: '',
startupPlan: null,
quickPrompt: 'Fix the route',
quickTelemetry: null
},
worktreeId: 'worktree-1',
shouldActivateOnCompletion: true,
fallbackStartupOpt: undefined,
activation: false,
primaryTabId: null
})
expect(mocks.activateAndRevealWorktree).toHaveBeenCalledWith('worktree-1', {
providesInitialSurface: true
})
expect(mocks.activateAndRevealWorktree.mock.invocationCallOrder[0]).toBeLessThan(
mocks.activateStructuredAgentSessionById.mock.invocationCallOrder[0]
)
expect(result.activation).toEqual({ primaryTabId: null })
})
})
@@ -139,6 +139,11 @@ export async function launchStructuredWorktreeSession(args: {
return { accepted, cancelled, visibilityUnknown, activation, primaryTabId }
}
if (args.shouldActivateOnCompletion) {
// Chat selection requires its workspace to be active.
if (!activation) {
activation = activateAndRevealWorktree(args.worktreeId, { providesInitialSurface: true })
primaryTabId = activation === false ? null : activation.primaryTabId
}
activateStructuredAgentSessionById({
worktreeId: args.worktreeId,
sessionId: receipt.sessionId
@@ -30,7 +30,7 @@ function mapOf(indices: readonly number[]): AppState['agentStatusByPaneKey'] {
return map
}
/** Re-spread with the same entry objects, as a status ping does. */
/** A collection refresh can replace its container while preserving the rows. */
function respread(map: AppState['agentStatusByPaneKey']): AppState['agentStatusByPaneKey'] {
return { ...map }
}
@@ -76,7 +76,7 @@ describe('agent-status projection join short circuit', () => {
const map = mapOf([0, 1, 2])
const first = buildRuntimeMobileAgentStatusProjectionForTests(map)
// A new map identity with identical entry references — the common ping shape.
// A new map identity with identical entry references.
const { result, joins, sorts } = countJoins(() =>
buildRuntimeMobileAgentStatusProjectionForTests(respread(map))
)
@@ -101,6 +101,41 @@ describe('agent-status projection join short circuit', () => {
expect(sorts).toBe(0)
})
it('does not rejoin accumulated previews for timestamp-only heartbeats', () => {
let map = mapOf(Array.from({ length: 500 }, (_, index) => index))
for (const paneKey of Object.keys(map)) {
map[paneKey] = {
...map[paneKey],
updatedAt: 30_000_000,
lastAssistantMessage: 'answer '.repeat(1_000)
}
}
const first = buildRuntimeMobileAgentStatusProjectionForTests(map)
const { result, joins } = countJoins(() => {
let projection = first
for (let index = 0; index < 50; index++) {
const paneKey = `tab-${index}:leaf-0`
map = { ...map, [paneKey]: { ...map[paneKey], updatedAt: 30_000_001 + index } }
projection = buildRuntimeMobileAgentStatusProjectionForTests(map)
}
return projection
})
expect(result).toBe(first)
expect(joins).toBe(0)
const paneKey = 'tab-0:leaf-0'
const nextBucket = { ...map, [paneKey]: { ...map[paneKey], updatedAt: 30_030_000 } }
expect(buildRuntimeMobileAgentStatusProjectionForTests(nextBucket)).not.toBe(first)
expect(
buildRuntimeMobileAgentStatusProjectionForTests({
...map,
[paneKey]: { ...map[paneKey], lastAssistantMessage: 'A new answer' }
})
).not.toBe(first)
})
it('still rebuilds when an entry changes', () => {
resetRuntimeMobileAgentStatusProjectionCacheForTests()
const map = mapOf([0, 1])
@@ -132,6 +167,20 @@ describe('agent-status projection join short circuit', () => {
)
})
it('still rebuilds when key membership swaps at a constant entry count', () => {
// The entry-count check cannot see a same-size swap; only the per-key lookup does.
resetRuntimeMobileAgentStatusProjectionCacheForTests()
const map = mapOf([0, 1])
const first = buildRuntimeMobileAgentStatusProjectionForTests(map)
const swapped = { 'tab-0:leaf-0': map['tab-0:leaf-0'], 'tab-9:leaf-0': makeEntry(9) }
const result = buildRuntimeMobileAgentStatusProjectionForTests(swapped)
expect(result).not.toBe(first)
expect(result).toContain('tab-9:leaf-0')
expect(result).not.toContain('tab-1:leaf-0')
})
it('still rebuilds when a pane is added', () => {
resetRuntimeMobileAgentStatusProjectionCacheForTests()
const map = mapOf([0])
@@ -60,6 +60,7 @@ export function buildRuntimeMobileAgentStatusProjection(
// A status ping replaces one entry and re-spreads the map; reuse every other entry.
const entries = new Map<string, AgentStatusProjectionCacheEntry>()
const parts: string[] = []
let projectionUnchanged = cached != null && nextEntries.length === cached.entries.size
// Code-unit order, not `localeCompare`: this projection is only ever compared with `===`, so it
// must be deterministic, not locale-correct — and an ICU collator per comparison is ~4.5k calls
// per ping at the 500-entry cap.
@@ -71,8 +72,10 @@ export function buildRuntimeMobileAgentStatusProjection(
: { entry, projection: serializeAgentStatusEntry(paneKey, entry) }
entries.set(paneKey, entryCache)
parts.push(entryCache.projection)
projectionUnchanged &&= previous?.projection === entryCache.projection
}
const projection = `[${parts.join(',')}]`
// Same-bucket heartbeats must not rejoin every pane's accumulated preview text.
const projection = projectionUnchanged && cached ? cached.projection : `[${parts.join(',')}]`
graphState.cachedAgentStatusProjection = {
source: agentStatusByPaneKey,
entries,
@@ -3,7 +3,8 @@ import {
RECENTLY_CLOSED_AGENT_STATUS_TAB_IDS_MAX,
RECENTLY_RETIRED_AGENT_STATUS_PANE_KEYS_MAX,
boundRecentlyClosedAgentStatusTabIds,
boundRecentlyRetiredAgentStatusPaneKeys
boundRecentlyRetiredAgentStatusPaneKeys,
removePaneKeys
} from './agent-status-pane-keyed-records'
function keyRecord(keys: readonly string[]): Record<string, true> {
@@ -108,3 +109,49 @@ describe('boundRecentlyClosedAgentStatusTabIds', () => {
expect(next.fresh).toBe(true)
})
})
it('does not enumerate unrelated pane records when removing absent keys', () => {
let enumerations = 0
const record = new Proxy(
Object.fromEntries(Array.from({ length: 1000 }, (_, index) => [`tab-${index}:leaf`, index])),
{
ownKeys(target) {
enumerations += 1
return Reflect.ownKeys(target)
}
}
)
for (let index = 0; index < 200; index += 1) {
expect(removePaneKeys(record, new Set(['absent:leaf']))).toBe(record)
}
expect(enumerations).toBe(0)
})
it('removes enumerable undefined values without removing inherited or hidden keys', () => {
const record = { visible: undefined }
Object.defineProperty(record, 'hidden', { value: 1, enumerable: false })
expect(removePaneKeys(record, new Set(['hidden', 'toString']))).toBe(record)
expect(Object.hasOwn(removePaneKeys(record, new Set(['visible'])), 'visible')).toBe(false)
})
// Probing the requested keys only matches the old record-enumeration when the two sets
// disagree in both directions, and the store relies on the unchanged case staying identical.
it('drops every requested key that is present and keeps the reference when none are', () => {
const base = { a: 1, b: 2, c: 3 }
expect(removePaneKeys({ ...base }, new Set(['a', 'b', 'c', 'd', 'e']))).toEqual({})
expect(removePaneKeys({ ...base }, new Set(['b', 'missing']))).toEqual({ a: 1, c: 3 })
const disjoint = { ...base }
expect(removePaneKeys(disjoint, new Set(['x', 'y']))).toBe(disjoint)
const noKeys = { ...base }
expect(removePaneKeys(noKeys, new Set())).toBe(noKeys)
const empty = {}
expect(removePaneKeys(empty, new Set(['a']))).toBe(empty)
})
it('leaves surviving keys in their original order and does not pollute the prototype', () => {
const record: Record<string, number> = { z: 1, y: 2, x: 3, w: 4 }
expect(Object.keys(removePaneKeys(record, new Set(['x', 'z'])))).toEqual(['y', 'w'])
const plain = { safe: 1 }
expect(removePaneKeys(plain, new Set(['__proto__', 'constructor']))).toBe(plain)
expect(Object.getPrototypeOf(plain)).toBe(Object.prototype)
})
@@ -83,15 +83,20 @@ export function removePaneKeys<T>(
record: Record<string, T>,
paneKeys: ReadonlySet<string>
): Record<string, T> {
const matchingKeys = Object.keys(record).filter((key) => paneKeys.has(key))
if (matchingKeys.length === 0) {
return record
}
const next = { ...record }
for (const key of matchingKeys) {
// Probe the requested keys instead of enumerating the record, and copy only once a key
// actually matches: pane retirement calls this across a dozen records that usually hold
// none of the retired keys, so the no-op path must stay allocation-free and keep returning
// the same reference. `propertyIsEnumerable` is own-only, so inherited keys such as
// `toString` or `__proto__` are never deletable.
let next: Record<string, T> | null = null
for (const key of paneKeys) {
if (!Object.prototype.propertyIsEnumerable.call(record, key)) {
continue
}
next ??= { ...record }
delete next[key]
}
return next
return next ?? record
}
export function removePaneKeysByTabPrefix<T>(
@@ -0,0 +1,86 @@
import { describe, expect, it, vi } from 'vitest'
import * as ownership from './own-retained-string'
import { createAgentStatusOscProcessor } from './agent-status-osc'
const INCOMPLETE_STATUS = '\x1b]9999;{"state":"working","prompt":"fragment'
describe('OSC 9999 pending storage', () => {
it('routes every retained pending frame through ownRetainedString', () => {
const own = vi.spyOn(ownership, 'ownRetainedString')
try {
const process = createAgentStatusOscProcessor()
expect(process('x'.repeat(32 * 1024) + INCOMPLETE_STATUS).cleanData).toHaveLength(32 * 1024)
expect(own).toHaveBeenCalledTimes(1)
expect(own).toHaveBeenLastCalledWith(INCOMPLETE_STATUS)
// Ownership is unconditional: a growing frame is re-owned on every chunk.
for (let index = 0; index < 8; index += 1) {
process('x'.repeat(512))
}
expect(own).toHaveBeenCalledTimes(9)
expect(own).toHaveBeenLastCalledWith(INCOMPLETE_STATUS + 'x'.repeat(8 * 512))
} finally {
own.mockRestore()
}
})
it.each([
[16 * 1024, 256],
[64 * 1024, 64],
[1024 * 1024, 16]
])('keeps %i-character chunks with %i incomplete statuses byte-exact', (size, count) => {
const parsers: ReturnType<typeof createAgentStatusOscProcessor>[] = []
let cleanChars = 0
for (let index = 0; index < count; index += 1) {
const process = createAgentStatusOscProcessor()
cleanChars += process(String.fromCharCode(65 + (index % 26)).repeat(size) + INCOMPLETE_STATUS)
.cleanData.length
parsers.push(process)
}
expect(cleanChars).toBe(size * count)
for (const process of parsers) {
expect(process('"}\x07after')).toEqual({
cleanData: 'after',
payloads: [{ state: 'working', prompt: 'fragment' }],
lastPayloadCleanOffset: 0
})
}
})
it.each(['\x07', '\x1b\\'])(
'preserves raw UTF-16 across an owned suffix and %j',
(terminator) => {
const process = createAgentStatusOscProcessor()
const ordinary = '😀'.repeat(8 * 1024)
const promptStart = '漢字\ud800|\udc00|\ud83d'
expect(process(`${ordinary}\x1b]9999;{"state":"working","prompt":"${promptStart}`)).toEqual({
cleanData: ordinary,
payloads: [],
lastPayloadCleanOffset: null
})
expect(process(`\ude00"}${terminator}after`)).toEqual({
cleanData: 'after',
payloads: [{ state: 'working', prompt: '漢字\ud800|\udc00|😀' }],
lastPayloadCleanOffset: 0
})
}
)
it('drops an owned frame that grows past the pending cap', () => {
const process = createAgentStatusOscProcessor()
const marker = '\x1b]9999;{"state":"working"}'
const pending = marker + ' '.repeat(64 * 1024 - marker.length)
for (let offset = 0; offset < pending.length; offset += 512) {
process(pending.slice(offset, offset + 512))
}
expect(process('\x07after')).toEqual({
cleanData: 'after',
payloads: [{ state: 'working', prompt: '' }],
lastPayloadCleanOffset: 0
})
})
})
+3 -1
View File
@@ -1,5 +1,6 @@
import type { ParsedAgentStatusPayload } from './agent-status-types'
import { parseAgentStatusPayload } from './agent-status-types'
import { ownRetainedString } from './own-retained-string'
const OSC_AGENT_STATUS_PREFIX = '\x1b]9999;'
@@ -103,7 +104,8 @@ export function createAgentStatusOscProcessor(): (data: string) => ProcessedAgen
if (terminator === null) {
const candidate = combined.slice(start)
pending = candidate.length > MAX_PENDING ? '' : candidate
// Own the frame so it stops pinning the consumed chunk it was sliced from.
pending = candidate.length > MAX_PENDING ? '' : ownRetainedString(candidate)
break
}
@@ -171,4 +171,45 @@ describe('browser network tunnel stream framing', () => {
expect(writer.queuedBytes).toBe(0)
expect(onError).not.toHaveBeenCalled()
})
it('rejects saturated writer frames before encoding and copying their payloads', () => {
const writer = new BrowserNetworkTunnelStreamFrameWriter(
() => {},
() => {},
{ maxQueuedFrames: 1 }
)
const frame = new Uint8Array(65536)
expect(writer.send(frame)).toBe(true)
const set = vi.spyOn(Uint8Array.prototype, 'set')
try {
for (let index = 0; index < 1000; index += 1) {
expect(writer.send(frame)).toBe(false)
}
expect(set.mock.calls.length).toBe(0)
expect(writer.queuedBytes).toBe(65540)
} finally {
set.mockRestore()
writer.close()
}
})
it('admits a frame that exactly fills the byte cap and rejects one byte past it', () => {
const writer = new BrowserNetworkTunnelStreamFrameWriter(
() => {},
() => {},
{ maxQueuedBytes: 158 }
)
expect(writer.send(new Uint8Array(100))).toBe(true)
expect(writer.queuedBytes).toBe(104)
expect(writer.send(new Uint8Array(51))).toBe(false)
expect(writer.send(new Uint8Array(50))).toBe(true)
expect(writer.queuedBytes).toBe(158)
writer.close()
})
it('encodes exactly the byte count writer admission reserves', () => {
for (const size of [1, 2, 255, 4096, 65536, 64 * 1024 + 16]) {
expect(encodeBrowserNetworkTunnelStreamFrame(new Uint8Array(size)).byteLength).toBe(4 + size)
}
})
})
@@ -4,11 +4,16 @@ const DEFAULT_MAX_RETAINED_BYTES = 2 * 1024 * 1024
const DEFAULT_MAX_QUEUED_BYTES = 1024 * 1024
const DEFAULT_MAX_QUEUED_FRAMES = 512
// Single source of truth so writer admission reserves exactly what encoding allocates.
function encodedFrameByteLength(frame: Uint8Array): number {
return LENGTH_BYTES + frame.byteLength
}
export function encodeBrowserNetworkTunnelStreamFrame(frame: Uint8Array): Uint8Array {
if (frame.byteLength === 0 || frame.byteLength > DEFAULT_MAX_FRAME_BYTES) {
throw new Error('browser_tunnel_stream_frame_invalid')
}
const encoded = new Uint8Array(LENGTH_BYTES + frame.byteLength)
const encoded = new Uint8Array(encodedFrameByteLength(frame))
new DataView(encoded.buffer).setUint32(0, frame.byteLength, false)
encoded.set(frame, LENGTH_BYTES)
return encoded
@@ -134,18 +139,18 @@ export class BrowserNetworkTunnelStreamFrameWriter {
if (this.closed) {
return false
}
if (
this.retainedBytes + encodedFrameByteLength(frame) > this.maxQueuedBytes ||
this.frames.length + (this.writing ? 1 : 0) >= this.maxQueuedFrames
) {
return false
}
let encoded: Uint8Array
try {
encoded = encodeBrowserNetworkTunnelStreamFrame(frame)
} catch {
return false
}
if (
this.retainedBytes + encoded.byteLength > this.maxQueuedBytes ||
this.frames.length + (this.writing ? 1 : 0) >= this.maxQueuedFrames
) {
return false
}
this.frames.push(encoded)
this.retainedBytes += encoded.byteLength
this.pump()
@@ -0,0 +1,65 @@
import { describe, expect, it, vi } from 'vitest'
import * as ownership from './own-retained-string'
import { extractOscTitleScanTail } from './osc-title-scan-tail'
const INCOMPLETE_TITLE = '\x1b]2;Working on a terminal title'
describe('OSC title scan tail storage', () => {
it('routes every retained title tail through ownRetainedString', () => {
const own = vi.spyOn(ownership, 'ownRetainedString')
try {
const tail = extractOscTitleScanTail('x'.repeat(32 * 1024) + INCOMPLETE_TITLE)
expect(own).toHaveBeenCalledTimes(1)
expect(own).toHaveBeenLastCalledWith(INCOMPLETE_TITLE)
expect(tail).toBe(INCOMPLETE_TITLE)
// Ownership is unconditional: a growing title is re-owned on every chunk.
let pending = tail
for (let index = 0; index < 8; index += 1) {
pending = extractOscTitleScanTail(pending + 'x'.repeat(512))
}
expect(own).toHaveBeenCalledTimes(9)
expect(own).toHaveBeenLastCalledWith(pending)
// An unrelated incomplete OSC still yields nothing to retain.
expect(extractOscTitleScanTail(`${'x'.repeat(32 * 1024)}\x1b]9999;incomplete`)).toBe('')
} finally {
own.mockRestore()
}
})
it.each([
[16 * 1024, 256, INCOMPLETE_TITLE],
[64 * 1024, 64, INCOMPLETE_TITLE],
[1024 * 1024, 16, INCOMPLETE_TITLE],
[16 * 1024, 128, INCOMPLETE_TITLE + 'x'.repeat(16 * 1024)]
])('keeps %i-character chunks with %i incomplete titles byte-exact', (size, count, title) => {
const tails: string[] = []
for (let index = 0; index < count; index++) {
tails.push(
extractOscTitleScanTail(String.fromCharCode(65 + (index % 26)).repeat(size) + title)
)
}
const expected = title.length <= 4096 ? title : title.slice(0, 4) + title.slice(-4092)
expect(tails).toEqual(Array.from({ length: count }, () => expected))
})
it.each(['0', '1', '2'])('preserves title %s introducer and exact UTF-16 at the cap', (code) => {
const prefix = `\x1b]${code};`
for (const length of [4095, 4096, 4097, 16 * 1024]) {
const value = `${prefix}${'x'.repeat(length - 9)}漢\ud800|\udc00\ud83d`
const expected = value.length <= 4096 ? value : prefix + value.slice(-4092)
const tail = extractOscTitleScanTail('a'.repeat(32 * 1024) + value)
expect(tail).toBe(expected)
expect(extractOscTitleScanTail(`${tail}\ude00\x1b\\`)).toBe('')
}
})
it('keeps trimming an owned title that grows past the cap', () => {
let pending = '\x1b]2;'
for (let index = 0; index < 128; index++) {
pending = extractOscTitleScanTail(pending + 'x'.repeat(512))
}
expect(pending).toBe(`\x1b]2;${'x'.repeat(4092)}`)
})
})
+4 -1
View File
@@ -1,3 +1,5 @@
import { ownRetainedString } from './own-retained-string'
const OSC_TITLE_SCAN_TAIL_LIMIT = 4096
const OSC_TITLE_PREFIX_LENGTH = 4
const OSC_TITLE_CODES = new Set(['0', '1', '2'])
@@ -7,7 +9,8 @@ export function extractOscTitleScanTail(input: string): string {
if (lastOsc !== -1) {
const suffix = input.slice(lastOsc)
if (!suffix.includes('\x07') && !suffix.includes('\x1b\\')) {
return extractIncompleteTitleOscTail(suffix)
// Own the tail so it stops pinning the consumed chunk it was sliced from.
return ownRetainedString(extractIncompleteTitleOscTail(suffix))
}
return input.endsWith('\x1b') ? '\x1b' : ''
}
+97
View File
@@ -0,0 +1,97 @@
import { afterEach, describe, expect, it } from 'vitest'
import { ownRetainedString, resetOwnRetainedStringCopier } from './own-retained-string'
import { copyUtf16SuffixToOwnedString } from './owned-utf16-suffix'
const PARENT_CHARS = 1024 * 1024
const PARENTS = 32
const TAIL_CHARS = 4096
function collectHeap(): number {
const gc = (globalThis as { gc?: () => void }).gc
if (!gc) {
throw new Error('global.gc unavailable - config/vitest.config.ts must pass --expose-gc')
}
void /reset/.test('reset')
gc()
gc()
return process.memoryUsage().heapUsed
}
function retainedBytes(take: (parent: string) => string): number {
const kept: string[] = []
for (let index = 0; index < 8; index += 1) {
take('warmup'.repeat(PARENT_CHARS / 6))
}
const before = collectHeap()
for (let index = 0; index < PARENTS; index += 1) {
kept.push(take(String.fromCharCode(65 + (index % 26)).repeat(PARENT_CHARS)))
}
const used = collectHeap() - before
expect(kept).toHaveLength(PARENTS)
expect(kept[0]).toHaveLength(TAIL_CHARS)
return used
}
describe('ownRetainedString', () => {
afterEach(() => {
resetOwnRetainedStringCopier()
})
it('releases the parent chunk that a retained tail was sliced from', () => {
const sliced = retainedBytes((parent) => parent.slice(parent.length - TAIL_CHARS))
const owned = retainedBytes((parent) =>
ownRetainedString(parent.slice(parent.length - TAIL_CHARS))
)
// 32 x 1 Mi parents pinned by 32 x 4 Ki tails, versus the tails alone.
expect(sliced).toBeGreaterThan(16 * 1024 * 1024)
expect(owned).toBeLessThan(4 * 1024 * 1024)
expect(owned).toBeLessThan(sliced / 8)
})
it.each([
['ascii', 'x'.repeat(4096)],
['two-byte', '漢'.repeat(4096)],
['lone lead surrogate', `\ud800${'a'.repeat(64)}`],
['lone trail surrogate', `${'a'.repeat(64)}\udfff`],
['split pair around padding', `\ud83d${'a'.repeat(64)}\ude00`],
['well-formed astral', '😀'.repeat(2048)],
['mixed controls', `\x1b]2;${'漢\ud800|\udc00\ud83d'.repeat(512)}`]
])('round-trips %s exactly', (_label, value) => {
expect(ownRetainedString(value)).toBe(value)
expect(ownRetainedString(value).length).toBe(value.length)
})
it('leaves already-flat short strings alone', () => {
// Below V8 SlicedString::kMinLength a slice is already a standalone copy.
for (const value of ['', '\x1b', '\ud800', 'x'.repeat(12)]) {
const parent = `${'y'.repeat(64 * 1024)}${value}`
const short = parent.slice(parent.length - value.length)
expect(ownRetainedString(short)).toBe(short)
}
expect(ownRetainedString('x'.repeat(13))).toBe('x'.repeat(13))
})
it('matches the block copier when Buffer is unavailable', () => {
const original = (globalThis as { Buffer?: unknown }).Buffer
const values = [
'漢\ud800|\udc00|\ud83d'.repeat(512),
`\x1b]9999;${'x'.repeat(4096)}\ude00`,
'😀'.repeat(2048)
]
const withBuffer = values.map((value) => ownRetainedString(value))
resetOwnRetainedStringCopier()
// Renderer and mobile bundles have no Buffer; the fallback must be byte-identical.
delete (globalThis as { Buffer?: unknown }).Buffer
try {
values.forEach((value, index) => {
expect(ownRetainedString(value)).toBe(value)
expect(ownRetainedString(value)).toBe(withBuffer[index])
expect(ownRetainedString(value)).toBe(copyUtf16SuffixToOwnedString(value, value.length))
})
} finally {
;(globalThis as { Buffer?: unknown }).Buffer = original
}
})
})
+50
View File
@@ -0,0 +1,50 @@
import { copyUtf16SuffixToOwnedString } from './owned-utf16-suffix'
// V8 SlicedString::kMinLength. Shorter slices are already flat, so copying them is pure waste.
const MIN_SLICED_STRING_LENGTH = 13
/** Lone surrogates and an unpaired trail must survive the round trip byte-for-byte. */
const ROUND_TRIP_PROBE = '\ud800a\udfff\u0000\u6f22'
let copyString: ((value: string) => string) | null = null
function detectBufferCopier(): ((value: string) => string) | null {
const buffer = (globalThis as { Buffer?: typeof Buffer }).Buffer
if (typeof buffer?.from !== 'function') {
return null
}
try {
if (buffer.from(ROUND_TRIP_PROBE, 'utf16le').toString('utf16le') !== ROUND_TRIP_PROBE) {
return null
}
} catch {
return null
}
return (value) => buffer.from(value, 'utf16le').toString('utf16le')
}
function resolveCopyString(): (value: string) => string {
if (!copyString) {
// Buffer is absent in the renderer and on mobile; fall back to the code-unit block copier.
copyString =
detectBufferCopier() ?? ((value: string) => copyUtf16SuffixToOwnedString(value, value.length))
}
return copyString
}
/**
* Return a standalone copy of a string that outlives the chunk it came from.
* Why: `bigChunk.slice(a, b)` is a V8 SlicedString that pins the whole parent,
* so retaining a few KiB of tail can pin megabytes of already-consumed output.
*/
export function ownRetainedString(value: string): string {
if (value.length < MIN_SLICED_STRING_LENGTH) {
return value
}
return resolveCopyString()(value)
}
/** Test-only: drop the memoized copier so a fallback path can be exercised. */
export function resetOwnRetainedStringCopier(): void {
copyString = null
}
@@ -0,0 +1,72 @@
import { describe, expect, it, vi } from 'vitest'
import { advancePartialEscapeTail, extractPartialEscapeTail } from './terminal-partial-escape-tail'
describe('partial escape scanning between sequences', () => {
it.each([
['ASCII', 'ordinary output '.repeat(8192)],
['UTF-16', '漢字😀 output '.repeat(8192)]
])('skips ordinary %s text between completed escapes', (_name, text) => {
const pending = '\x1b[38;5;'
const stream = `\x1b[32m${text}\x1b[0m${text}${pending}`
const charCodeAt = vi.spyOn(String.prototype, 'charCodeAt')
let actual: string
let inspectedCodeUnits: number
try {
actual = extractPartialEscapeTail(stream)
inspectedCodeUnits = charCodeAt.mock.calls.length
} finally {
charCodeAt.mockRestore()
}
expect(actual).toBe(pending)
expect(inspectedCodeUnits).toBeLessThan(32)
expect(advancePartialEscapeTail(actual, '196mready')).toBe('')
})
it.each([
['\x1b[38;5;', '196m'],
['\x1b]2;building', '\x07'],
['\x1b]2;building\x1b', '\\'],
['\x1bPqpayload', '\x1b\\'],
['\x1bPqpayload\x1b', '\\'],
['\x1b(', 'B']
])('preserves %j through every split after ordinary text', (pending, completion) => {
const prefix = `\x1b[32m${'build output '.repeat(128)}\x1b[0m`
for (let split = 0; split <= pending.length; split += 1) {
const first = advancePartialEscapeTail('', prefix + pending.slice(0, split))
const continued = advancePartialEscapeTail(first, pending.slice(split))
expect(continued).toBe(pending)
expect(advancePartialEscapeTail(continued, `${completion}ready`)).toBe('')
}
})
it('handles aborts before returning to ordinary text', () => {
const text = 'ordinary text '.repeat(128)
for (const abort of ['\x18', '\x1a']) {
for (const pending of ['\x1b[38;', '\x1b]2;title', '\x1bPdata', '\x1b(']) {
expect(extractPartialEscapeTail(`${pending}${abort}${text}\x1b[1;`)).toBe('\x1b[1;')
}
}
expect(extractPartialEscapeTail(`\x1b]2;title\x1b[32m${text}\x1b[1;`)).toBe('\x1b[1;')
})
it('keeps one-character echo free of per-code-unit scans and escape searches', () => {
const charCodeAt = vi.spyOn(String.prototype, 'charCodeAt')
const indexOf = vi.spyOn(String.prototype, 'indexOf')
let actual: string
let inspections: number
let searches: number
try {
actual = advancePartialEscapeTail('', 'x')
inspections = charCodeAt.mock.calls.length
searches = indexOf.mock.calls.length
} finally {
charCodeAt.mockRestore()
indexOf.mockRestore()
}
expect(actual).toBe('')
expect(inspections).toBe(0)
expect(searches).toBe(0)
})
})
+10 -4
View File
@@ -59,14 +59,20 @@ export function extractPartialEscapeTail(stream: string): string {
let state: ScanState = 'ground'
let start = 0
for (let i = 0; i < stream.length; i++) {
const code = stream.charCodeAt(i)
if (state === 'ground') {
if (code === ESC) {
start = i
state = 'esc'
// Only ESC leaves ground; skip ordinary text without a per-code-unit walk.
// Check the current unit first: on dense escape streams it is usually the
// ESC itself, and indexOf's call + SIMD setup costs more than the compare.
const escape = stream.charCodeAt(i) === ESC ? i : stream.indexOf('\x1b', i)
if (escape === -1) {
return ''
}
start = escape
i = escape
state = 'esc'
continue
}
const code = stream.charCodeAt(i)
if (
code === ESC &&
state !== 'osc' &&
+1 -2
View File
@@ -471,12 +471,11 @@ test.describe('New-user golden core flow', () => {
.locator('[data-contextual-tour-target="workspace-create-control"]')
.first()
await expect(createControl).toBeVisible()
await expect(createControl).toHaveAttribute('aria-label', 'Create')
await expect(createControl).toHaveAttribute('aria-label', 'New workspace')
const createControlBox = await createControl.boundingBox()
expect(createControlBox?.width ?? 0).toBeGreaterThan(0)
expect(createControlBox?.height ?? 0).toBeGreaterThan(0)
await createControl.click()
await orcaPage.getByRole('menuitem', { name: /^New workspace/ }).click()
const workspaceName = `golden-new-${Date.now()}`
await completeWorkspaceCreationTour(orcaPage, workspaceName)
+13 -6
View File
@@ -1,14 +1,21 @@
import { expect, type Page } from '@stablyai/playwright-test'
// Why scoped: the Landing screen renders its own "Add project" button whenever no
// workspace is open, which is exactly the state these helpers run in.
function sidebarHeaderActions(page: Page) {
return page.locator('[data-sidebar-header-actions]')
}
export async function openSidebarProjectDialog(page: Page): Promise<void> {
await page.getByRole('button', { name: 'Create', exact: true }).click()
await page.getByRole('menuitem', { name: 'Add project', exact: true }).click()
await sidebarHeaderActions(page).getByRole('button', { name: 'Add project', exact: true }).click()
await expect(page.getByRole('dialog', { name: /Add a project/i })).toBeVisible()
}
export async function openSidebarWorkspaceComposer(page: Page): Promise<void> {
await page.getByRole('button', { name: 'Create', exact: true }).click()
const newWorkspaceItem = page.getByRole('menuitem', { name: /^New workspace/ })
await expect(newWorkspaceItem).toBeVisible()
await newWorkspaceItem.click()
const createButton = sidebarHeaderActions(page).getByRole('button', {
name: 'New workspace',
exact: true
})
await expect(createButton).toBeVisible()
await createButton.click()
}