diff --git a/.github/workflows/cloud-push-deploy.yml b/.github/workflows/cloud-push-deploy.yml index ea19a589bb2..6014681372d 100644 --- a/.github/workflows/cloud-push-deploy.yml +++ b/.github/workflows/cloud-push-deploy.yml @@ -16,10 +16,9 @@ permissions: contents: read id-token: write -# The gateway applies its own schema at startup against the shared Cloud SQL instance, so a -# deploy is a connection-budget rollout and belongs in the same serialized group as the relay. +# Serialize push traffic changes independently of Relay and the shared database. concurrency: - group: production-cloud-sql-rollout + group: production-push-rollout cancel-in-progress: false defaults: @@ -75,8 +74,8 @@ jobs: - uses: google-github-actions/auth@v2 with: - workload_identity_provider: ${{ vars.PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER }} - service_account: ${{ vars.PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT }} + workload_identity_provider: ${{ vars.PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER }} + service_account: ${{ vars.PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT }} - uses: google-github-actions/setup-gcloud@v2 @@ -85,31 +84,41 @@ jobs: - name: Configure Docker auth run: gcloud auth configure-docker "${GCP_REGION}-docker.pkg.dev" --quiet - # Why: the build runs before the lease. Artifact Registry is not the Cloud SQL instance, - # and a multi-minute image build inside the lease blocks every relay deploy and rehome for - # its duration. The lease below covers exactly the connection-budget window: deploy, probe, - # shift. + # Building an image does not need the deployment lease. - name: Build and publish the immutable gateway image shell: bash run: | set -euo pipefail image_tag="${GCP_REGION}-docker.pkg.dev/${GCP_PROJECT_ID}/${REPOSITORY_ID}/${IMAGE_NAME}:sha-${SOURCE_SHA}" - docker build -f "${RUNNER_TEMP}/push-source/cloud/apps/push/Dockerfile" \ + docker buildx build --push --platform linux/amd64 --provenance=false --metadata-file "${RUNNER_TEMP}/push-image.json" \ + -f "${RUNNER_TEMP}/push-source/cloud/apps/push/Dockerfile" \ -t "${image_tag}" "${RUNNER_TEMP}/push-source/cloud" - docker push "${image_tag}" - digest="$(gcloud artifacts docker images describe "${image_tag}" \ - --format='value(image_summary.digest)')" + digest="$(jq -er '."containerimage.digest"' "${RUNNER_TEMP}/push-image.json")" [[ "${digest}" =~ ^sha256:[a-f0-9]{64}$ ]] echo "IMAGE=${GCP_REGION}-docker.pkg.dev/${GCP_PROJECT_ID}/${REPOSITORY_ID}/${IMAGE_NAME}@${digest}" \ >> "${GITHUB_ENV}" echo "IMAGE_DIGEST=${digest}" >> "${GITHUB_ENV}" + # Refuse older images before they ever boot against production. + - name: Require image support for inert validation + shell: bash + run: | + set -euo pipefail + docker run --rm --network none --entrypoint node "${IMAGE}" --input-type=module -e ' + import { loadPushConfig } from "./apps/push/dist/config.js"; + const env = { ORCA_PUSH_PUBLIC_URL: "https://push.onorca.dev", ORCA_PUSH_MODE: "validation" }; + if (loadPushConfig(env).mode !== "validation") throw new Error("validation_mode_unsupported"); + let rejected = false; + try { loadPushConfig({ ...env, ORCA_PUSH_MODE: "invalid" }); } catch { rejected = true; } + if (!rejected) throw new Error("validation_mode_not_fail_closed"); + ' + # Held across the deploy, not just a separate schema step: the gateway opens its pool and # applies its schema while the new revision starts, so the revision is the schema step. - uses: ./.github/actions/cloud-sql-rollout-lease with: bucket: onorca-cloud-terraform-state - object: terraform/state/cloud-sql-rollout/production.lock + object: terraform/state/push-rollout/production.lock # Why: the candidate inherits the serving revision's scaling. A serving revision that has # drifted below the floor would hand the candidate a cold start on every notification, and @@ -124,6 +133,12 @@ jobs: | jq -r '[.status.traffic[] | select((.percent // 0) > 0)] | if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')" test -n "${serving}" + revisions="$(gcloud run revisions list --service "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format='value(metadata.name)')" + if test "${revisions}" != "${serving}"; then + echo 'Retire leftover revisions under the rollout lease before deploying; three pools are the limit.' >&2 + exit 1 + fi floor="$(gcloud run revisions describe "${serving}" \ --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ --format="value(metadata.annotations['autoscaling.knative.dev/minScale'])")" @@ -140,16 +155,29 @@ jobs: test "${ceiling}" = "${PUSH_MAX_INSTANCES}" echo "serving revision ${serving} holds ${floor} minimum and ${ceiling} maximum instances" echo "ROLLBACK_REVISION=${serving}" >> "${GITHUB_ENV}" + gcloud run revisions describe "${serving}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ + > "${RUNNER_TEMP}/push-rollback-revision.json" + image="$(jq -er '.status.imageDigest' "${RUNNER_TEMP}/push-rollback-revision.json")" + [[ "${image}" =~ @sha256:[a-f0-9]{64}$ ]] + echo "ROLLBACK_IMAGE=${image}" >> "${GITHUB_ENV}" + jq -e 'all(.spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE" or .value == "active")' \ + "${RUNNER_TEMP}/push-rollback-revision.json" > /dev/null - # No traffic and a per-revision tag: the candidate boots, applies schema, and is probed on - # its own URL while every phone and desktop still reaches the previous revision. + # Validation has no schema writes, HTTP mutations, worker, or pruners; tags alone do not + # isolate background consumers from production. - name: Deploy the candidate revision with no traffic shell: bash run: | set -euo pipefail tag="c${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" - echo "CANDIDATE_TAG=${tag}" >> "${GITHUB_ENV}" - echo "CANDIDATE_REVISION=${SERVICE_NAME}-${tag}" >> "${GITHUB_ENV}" + { + echo "VALIDATION_DEPLOY_ATTEMPTED=true" + echo "VALIDATION_REVISION=${SERVICE_NAME}-${tag}" + echo "VALIDATION_TAG=${tag}" + echo "CANDIDATE_TAG=${tag}" + echo "CANDIDATE_REVISION=${SERVICE_NAME}-${tag}" + } >> "${GITHUB_ENV}" gcloud run deploy "${SERVICE_NAME}" \ --project "${GCP_PROJECT_ID}" \ --region "${GCP_REGION}" \ @@ -157,6 +185,7 @@ jobs: --tag "${tag}" \ --revision-suffix "${tag}" \ --no-traffic \ + --update-env-vars ORCA_PUSH_MODE=validation \ --quiet candidate="$(gcloud run services describe "${SERVICE_NAME}" \ --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ @@ -168,8 +197,7 @@ jobs: # A tagged revision is directly addressable and sits outside the service-wide cap, so the # candidate and the serving revision each draw up to the ceiling during the probe window. - # The lease is taken for exactly that doubling; a candidate that inherited a wider ceiling - # would exceed it, so the inherited scaling is asserted here too. + # Successor creation later requires three revision pools; assert the inherited ceiling. - name: Require the candidate to serve the exact image and inherited scaling shell: bash run: | @@ -195,7 +223,7 @@ jobs: if test "${code}" = 200; then jq -e . < "${RUNNER_TEMP}/push-ready.json" > /dev/null curl --fail --silent --show-error --max-time 10 "${CANDIDATE_URL}/health" \ - | jq -e '.ok == true and .deliveryProtocol == 2' > /dev/null + | jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "validation"' > /dev/null echo "candidate ${CANDIDATE_REVISION} is ready after ${attempt} attempt(s)" exit 0 fi @@ -242,6 +270,53 @@ jobs: fi test "${status}" = INVALID_ARGUMENT + # Cloud Run requires a successor before the latest revision can be deleted. + - name: Retire inert validation and activate the verified image + shell: bash + run: | + set -euo pipefail + tag="a${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" + { + echo "CANDIDATE_TAG=${tag}" + echo "CANDIDATE_REVISION=${SERVICE_NAME}-${tag}" + echo "ACTIVATION_ATTEMPTED=true" + } >> "${GITHUB_ENV}" + # This is the production-effect boundary: schema, pruners and workers start here. + gcloud run deploy "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --image "${IMAGE}" --tag "${tag}" --revision-suffix "${tag}" \ + --remove-env-vars ORCA_PUSH_MODE --no-traffic --quiet + gcloud run services update-traffic "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --remove-tags "${VALIDATION_TAG}" --quiet + gcloud run revisions delete "${VALIDATION_REVISION}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet + echo "VALIDATION_RETIRED=true" >> "${GITHUB_ENV}" + revision="$(gcloud run revisions describe "${SERVICE_NAME}-${tag}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json)" + jq -e --arg image "${IMAGE}" --arg account "${PUSH_RUNTIME_SERVICE_ACCOUNT}" \ + --arg ceiling "${PUSH_MAX_INSTANCES}" --arg floor "${PUSH_MIN_INSTANCES}" \ + --slurpfile prior "${RUNNER_TEMP}/push-rollback-revision.json" ' + def shape: del(.containers[0].image) | + .containers[0].env = ((.containers[0].env // []) | + map(select(.name != "ORCA_PUSH_MODE")) | sort_by(.name)); + .spec.containers[0].image == $image and .spec.serviceAccountName == $account and + all(.spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE") and + (.spec | shape) == ($prior[0].spec | shape) and + .metadata.annotations["autoscaling.knative.dev/maxScale"] == $ceiling and + (.metadata.annotations["autoscaling.knative.dev/minScale"] | tonumber) >= ($floor | tonumber)' \ + <<< "${revision}" > /dev/null + candidate="$(gcloud run services describe "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ + | jq -er --arg tag "${tag}" '[.status.traffic[] | select(.tag == $tag)] + | if length == 1 then .[0] else error("active candidate is not unique") end')" + test "$(jq -r '.revisionName' <<< "${candidate}")" = "${SERVICE_NAME}-${tag}" + url="$(jq -er '.url' <<< "${candidate}")" + [[ "${url}" =~ ^https://[^/]+$ ]] + curl --fail --silent --show-error --max-time 10 "${url}/ready" | jq -e '.ok == true' + curl --fail --silent --show-error --max-time 10 "${url}/health" \ + | jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"' + - name: Shift all traffic to the verified candidate shell: bash run: | @@ -275,9 +350,14 @@ jobs: echo "Revision: \`${CANDIDATE_REVISION}\`" echo echo "Image: \`${IMAGE_DIGEST}\`" + echo "Known-good image: \`${ROLLBACK_IMAGE}\`" echo - echo "Rollback: \`gcloud run services update-traffic ${SERVICE_NAME}" \ - "--region ${GCP_REGION} --to-revisions ${ROLLBACK_REVISION}=100\`" + echo "Recovery: deploy \`${ROLLBACK_IMAGE}\` as a new revision with" \ + "\`--remove-env-vars ORCA_PUSH_MODE --no-traffic --tag --revision-suffix \`." + echo 'Verify its exact digest, configuration, readiness and active mode, then promote and check the public origin.' + echo 'Only then remove obsolete tags and delete rejected/previous revisions; never delete the latest revision.' + echo 'The previous revision is retired after public checks; retain this immutable image for recovery under the rollout lease.' + echo "Activation attempted: ${ACTIVATION_ATTEMPTED:-false}; traffic rollback cannot undo schema or deliveries." } >> "${GITHUB_STEP_SUMMARY}" - name: Verify the public origin after the shift @@ -289,7 +369,8 @@ jobs: "${PUSH_ORIGIN}/ready" || true)" if test "${code}" = 200; then curl --fail --silent --show-error --max-time 10 "${PUSH_ORIGIN}/health" \ - | jq -e '.ok == true and .deliveryProtocol == 2' > /dev/null + | jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"' > /dev/null + echo "ROLLOUT_VERIFIED=true" >> "${GITHUB_ENV}" echo "${PUSH_ORIGIN} is ready after ${attempt} attempt(s)" exit 0 fi @@ -303,7 +384,7 @@ jobs: # is not a failure to deploy, it is a live gateway that has to go back, so the traffic move # is undone here rather than left to whoever reads the run. - name: Roll traffic back to the previous revision - if: ${{ (failure() || cancelled()) && env.TRAFFIC_SHIFT_ATTEMPTED == 'true' }} + if: ${{ (failure() || cancelled()) && env.TRAFFIC_SHIFT_ATTEMPTED == 'true' && env.ROLLOUT_VERIFIED != 'true' }} shell: bash run: | set -euo pipefail @@ -324,19 +405,107 @@ jobs: echo '### Push gateway rolled back' echo echo "Traffic returned to \`${ROLLBACK_REVISION}\`; the candidate" \ - "\`${CANDIDATE_REVISION}\` no longer serves." + "\`${CANDIDATE_REVISION}\` no longer serves HTTP; deletion below must stop its workers." } >> "${GITHUB_STEP_SUMMARY}" - # Why: a candidate that never took traffic is a revision holding a warm floor and a Cloud - # SQL pool for nothing. Its tag comes off first, because Cloud Run refuses to delete a - # revision a traffic target still names, and clearing CANDIDATE_TAG makes the always() tag - # step below a no-op rather than a second failure. - - name: Delete the rejected candidate revision - if: ${{ (failure() || cancelled()) && (env.TRAFFIC_SHIFT_ATTEMPTED != 'true' || env.TRAFFIC_ROLLED_BACK == 'true') }} + # Deleting a revision does not restore the service template that Terraform reconciles. + - name: Restore the known-good service template + if: ${{ (failure() || cancelled()) && env.VALIDATION_DEPLOY_ATTEMPTED == 'true' && env.ROLLOUT_VERIFIED != 'true' }} shell: bash run: | set -euo pipefail - test -n "${CANDIDATE_REVISION:-}" || exit 0 + test "${TRAFFIC_SHIFT_ATTEMPTED:-false}" != true || test "${TRAFFIC_ROLLED_BACK:-false}" = true + test -n "${ROLLBACK_IMAGE}" + # A partial activation may leave validation plus active; free one slot before recovery. + latest="$(gcloud run services describe "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --format='value(status.latestCreatedRevisionName)')" + test -n "${latest}" + if test "${VALIDATION_RETIRED:-false}" != true && test "${latest}" != "${VALIDATION_REVISION}"; then + existing="$(gcloud run revisions list --service "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --filter "metadata.name=${VALIDATION_REVISION}" --format='value(metadata.name)')" + if test -n "${existing}"; then + test "${existing}" = "${VALIDATION_REVISION}" + gcloud run services update-traffic "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --remove-tags "${VALIDATION_TAG}" --quiet + gcloud run revisions delete "${VALIDATION_REVISION}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet + fi + fi + tag="r${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" + echo "RECOVERY_TAG=${tag}" >> "${GITHUB_ENV}" + echo "TEMPLATE_RECOVERY_REVISION=${SERVICE_NAME}-${tag}" >> "${GITHUB_ENV}" + echo "Template recovery attempted: ${SERVICE_NAME}-${tag}, image ${ROLLBACK_IMAGE}." \ + >> "${GITHUB_STEP_SUMMARY}" + # Known-good schema/workers can run here even though HTTP stays on the old revision. + gcloud run deploy "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --image "${ROLLBACK_IMAGE}" --revision-suffix "${tag}" --tag "${tag}" \ + --remove-env-vars ORCA_PUSH_MODE --no-traffic --quiet + gcloud run services describe "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ + > "${RUNNER_TEMP}/push-recovered-service.json" + gcloud run revisions describe "${SERVICE_NAME}-${tag}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ + > "${RUNNER_TEMP}/push-recovery-revision.json" + jq -e --arg image "${ROLLBACK_IMAGE}" --arg serving "${ROLLBACK_REVISION}" \ + --arg floor "${PUSH_MIN_INSTANCES}" --arg ceiling "${PUSH_MAX_INSTANCES}" \ + --slurpfile prior "${RUNNER_TEMP}/push-rollback-revision.json" \ + --slurpfile recovered "${RUNNER_TEMP}/push-recovery-revision.json" ' + def shape: del(.containers[0].image) | + .containers[0].env = ((.containers[0].env // []) | + map(select(.name != "ORCA_PUSH_MODE")) | sort_by(.name)); + .spec.template.spec.containers[0].image == $image and + all(.spec.template.spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE") and + $recovered[0].spec.containers[0].image == $image and + all($recovered[0].spec.containers[0].env[]?; .name != "ORCA_PUSH_MODE") and + ($recovered[0].spec | shape) == ($prior[0].spec | shape) and + .spec.template.metadata.annotations["autoscaling.knative.dev/maxScale"] == $ceiling and + (.spec.template.metadata.annotations["autoscaling.knative.dev/minScale"] | tonumber) >= ($floor | tonumber) and + ([.status.traffic[] | select((.percent // 0) > 0)] | + length == 1 and .[0].revisionName == $serving and .[0].percent == 100) + ' "${RUNNER_TEMP}/push-recovered-service.json" > /dev/null + echo "TEMPLATE_RESTORED=true" >> "${GITHUB_ENV}" + echo 'Known-good image and normal mode restored; previous revision still serves HTTP.' \ + >> "${GITHUB_STEP_SUMMARY}" + + - name: Promote and verify the known-good recovery revision + if: ${{ always() && env.TEMPLATE_RESTORED == 'true' }} + shell: bash + run: | + set -euo pipefail + url="$(jq -er --arg tag "${RECOVERY_TAG}" \ + '.status.traffic[] | select(.tag == $tag) | .url' "${RUNNER_TEMP}/push-recovered-service.json")" + [[ "${url}" =~ ^https://[^/]+$ ]] + curl --fail --silent --show-error --max-time 10 "${url}/ready" | jq -e '.ok == true' + curl --fail --silent --show-error --max-time 10 "${url}/health" \ + | jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"' + gcloud run services update-traffic "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --to-revisions "${TEMPLATE_RECOVERY_REVISION}=100" --quiet + serving="$(gcloud run services describe "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format=json \ + | jq -er '[.status.traffic[] | select((.percent // 0) > 0)] | + if length == 1 and .[0].percent == 100 then .[0].revisionName else empty end')" + test "${serving}" = "${TEMPLATE_RECOVERY_REVISION}" + curl --fail --silent --show-error --max-time 10 "${PUSH_ORIGIN}/ready" | jq -e '.ok == true' + curl --fail --silent --show-error --max-time 10 "${PUSH_ORIGIN}/health" \ + | jq -e '.ok == true and .deliveryProtocol == 2 and .mode == "active"' + echo "RECOVERY_VERIFIED=true" >> "${GITHUB_ENV}" + echo "Recovery revision ${TEMPLATE_RECOVERY_REVISION} now serves the known-good image." \ + >> "${GITHUB_STEP_SUMMARY}" + + - name: Delete the rejected candidate revision + if: ${{ always() && env.RECOVERY_VERIFIED == 'true' }} + shell: bash + run: | + set -euo pipefail + if test -z "${CANDIDATE_REVISION:-}"; then + echo "CANDIDATE_DELETED=true" >> "${GITHUB_ENV}" + exit 0 + fi if test -n "${CANDIDATE_TAG:-}"; then gcloud run services update-traffic "${SERVICE_NAME}" \ --project "${GCP_PROJECT_ID}" \ @@ -345,11 +514,44 @@ jobs: --quiet echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}" fi - gcloud run revisions delete "${CANDIDATE_REVISION}" \ - --project "${GCP_PROJECT_ID}" \ - --region "${GCP_REGION}" \ - --quiet - echo "deleted the candidate revision ${CANDIDATE_REVISION}" + existing="$(gcloud run revisions list --service "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" \ + --filter "metadata.name=${CANDIDATE_REVISION}" --format='value(metadata.name)')" + if test -n "${existing}"; then + test "${existing}" = "${CANDIDATE_REVISION}" + gcloud run revisions delete "${CANDIDATE_REVISION}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet + fi + echo "CANDIDATE_DELETED=true" >> "${GITHUB_ENV}" + echo "candidate revision ${CANDIDATE_REVISION} is absent" + + # Public checks commit the serving revision; cleanup failures must not roll it back. + - name: Retire previous consumers after public checks + if: ${{ always() && (env.ROLLOUT_VERIFIED == 'true' || (env.RECOVERY_VERIFIED == 'true' && env.CANDIDATE_DELETED == 'true')) }} + shell: bash + run: | + set -euo pipefail + serving="${CANDIDATE_REVISION}" + if test "${RECOVERY_VERIFIED:-false}" = true; then + serving="${TEMPLATE_RECOVERY_REVISION}" + fi + gcloud run services update-traffic "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --clear-tags --quiet + revisions="$(gcloud run revisions list --service "${SERVICE_NAME}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --format='value(metadata.name)')" + while IFS= read -r revision; do + test -n "${revision}" || continue + test "${revision}" != "${serving}" || continue + case "${revision}" in + "${ROLLBACK_REVISION}"|"${VALIDATION_REVISION}"|"${CANDIDATE_REVISION}") ;; + *) echo "Unexpected revision ${revision}; manual retirement required." >&2; exit 1 ;; + esac + gcloud run revisions delete "${revision}" \ + --project "${GCP_PROJECT_ID}" --region "${GCP_REGION}" --quiet + done <<< "${revisions}" + echo "CANDIDATE_TAG=" >> "${GITHUB_ENV}" + echo 'Obsolete revision resources retired; verify actual SQL session drain operationally.' \ + >> "${GITHUB_STEP_SUMMARY}" - name: Drop the candidate traffic tag if: always() diff --git a/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs b/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs index 2f7157d823c..bbfe72ec2b4 100644 --- a/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs +++ b/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs @@ -283,8 +283,6 @@ export const LEASED_WORKFLOWS = named([ 'operate-relay-production-rehome.yml', production({ leaseFiles: ['operate-relay-production-rehome-job.yml'] }) ], - // The gateway applies its schema at startup, so its deploy revision is the schema step. - ['push-deploy.yml', production()], ['deploy-relay-asia-topology.yml', eitherEnvironment()], ['operate-relay-asia-admission.yml', eitherEnvironment()], ['deploy-relay-staging.yml', staging()], @@ -300,6 +298,7 @@ export const LEASED_WORKFLOWS = named([ ]) export const NOT_A_CLOUD_SQL_CANDIDATE = named([ + ['push-deploy.yml', 'Push uses dedicated SQL and its own production-push-rollout group and durable push-rollout lease.'], [ 'monitor-relay-production.yml', 'Read-only. Its identity holds monitoring, logging, Cloud SQL and compute viewer roles only, and it runs `gcloud sql instances describe`, never a mutation. It consumes no connection budget, so the durable lease would only let monitoring block a rollout and a rollout block monitoring.' diff --git a/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs b/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs index f97e742215b..3845cb90061 100644 --- a/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs +++ b/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs @@ -32,8 +32,7 @@ test('no workflow names the retired generic production deploy identity', async ( 'deploy-relay-production.yml', 'operate-relay-asia-admission.yml', 'operate-relay-production-rehome-job.yml', - 'publish-relay-production.yml', - 'push-deploy.yml' + 'publish-relay-production.yml' ].map((name) => relayWorkflowFile(name)).sort()) }) @@ -168,3 +167,12 @@ test('fence broker pins the production-proven Terraform planner', async () => { const dockerfile = await source('apps/relay-fence-broker/Dockerfile') assert.match(dockerfile, /FROM hashicorp\/terraform:1\.15\.8 AS terraform/) }) + + test('push uses its dedicated identity and rollout lease', () => { + const workflow = readRelayWorkflow('push-deploy.yml') + assert.match(workflow, /PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(workflow, /PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT/) + assert.doesNotMatch(workflow, /PRODUCTION_GCP_RELAY_DEPLOY_/) + assert.match(workflow, /group: production-push-rollout/) + assert.match(workflow, /object: terraform\/state\/push-rollout\/production.lock/) +}) diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index c56dd9941e8..bd1b3c20ad6 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -83,6 +83,91 @@ ], "demotionRule": "Keep experimental until CI soak; investigate fidelity or count failures without relaxing the row budget." }, + { + "id": "agent-session.hibernation-runtime-inventory-budget", + "title": "Hibernation skips irrelevant remote inventories and preserves fresh host evidence", + "maturity": "experimental", + "protection": "partial", + "owner": "agent-session-runtime", + "layer": "shared-and-renderer-unit", + "surfaces": ["automatic agent hibernation", "remote terminal liveness"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "remote-runtime"], + "coverageNotes": "Local hibernation, remote inventory authority and folder-workspace identity are covered by coordinator/model contracts on macOS. Daemon, SSH, WSL, relay and native platform execution boundaries and existing RPC payloads are unchanged.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/src/renderer/src/lib/agent-hibernation-coordinator.test.ts" + ], + "invariant": "Automatic hibernation must retain two stable confirmations and fresh execution-host evidence before shutdown. Skipping a workspace with no completed agent must not authorize a newly completed pane using stale client PTYs.", + "oracle": "100 remote workspaces containing working/waiting agents issue zero runtime calls. A skipped workspace completing while another inventory awaits remains ineligible until two later host-confirmed passes, and so does a workspace that only becomes runtime-owned while an inventory is outstanding — a pass carrying no host evidence for a workspace never counts as one of its two confirmations. Folder and git workspaces hibernate the exact host PTY; rejected/truncated inventories and intervening input/output or state changes block shutdown.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/agent-hibernation-coordinator.test.ts src/renderer/src/lib/agent-hibernation-planner.test.ts src/renderer/src/lib/agent-hibernation-confirmation.test.ts src/renderer/src/lib/agent-hibernation-pane-age.test.ts src/renderer/src/lib/agent-hibernation-output-activity.test.ts src/renderer/src/lib/agent-hibernation-visibility.test.ts src/renderer/src/lib/foreground-terminal-tabs.test.ts src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts" + ], + "testFiles": [ + "src/renderer/src/lib/agent-hibernation-coordinator.test.ts", + "src/renderer/src/lib/agent-hibernation-planner.test.ts", + "src/renderer/src/lib/agent-hibernation-confirmation.test.ts", + "src/renderer/src/lib/agent-hibernation-pane-age.test.ts", + "src/renderer/src/lib/agent-hibernation-output-activity.test.ts", + "src/renderer/src/lib/agent-hibernation-visibility.test.ts", + "src/renderer/src/lib/foreground-terminal-tabs.test.ts", + "src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts" + ], + "assertionRefs": [ + { + "file": "src/renderer/src/lib/agent-hibernation-coordinator.test.ts", + "assertions": [ + "does not request runtime inventories for 100 workspaces without completed agents", + "requires host evidence after a skipped workspace completes during another inventory request", + "hibernates a runtime-backed candidate in %s with fresh liveness and exact PTYs", + "fails closed on truncated runtime liveness samples", + "fails closed when fresh runtime liveness rejects after an earlier good sample" + ] + }, + { + "file": "src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts", + "assertions": [ + "requires host evidence when a workspace becomes runtime-owned during an inventory request" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/agent-hibernation-coordinator.test.ts src/renderer/src/lib/agent-hibernation-planner.test.ts src/renderer/src/lib/agent-hibernation-confirmation.test.ts src/renderer/src/lib/agent-hibernation-pane-age.test.ts src/renderer/src/lib/agent-hibernation-output-activity.test.ts src/renderer/src/lib/agent-hibernation-visibility.test.ts src/renderer/src/lib/foreground-terminal-tabs.test.ts src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts", + "result": "passed", + "durationSeconds": 48.18, + "summary": "93 tests passed across eight coordinator, liveness-race, planner, confirmation, age, activity and visibility files." + } + ], + "runtimeBudget": { + "p95Seconds": 60, + "scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial local validation; no CI soak history." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "Baseline made 101 runtime calls in the new no-completed-agent fixture; candidate makes zero. Existing confirmation, final recheck and failure-path assertions continue to pass, including completion during an outstanding request. Separately, baseline hibernated a workspace that became runtime-owned mid-inventory one tick early, taking its first confirmation from a pass planned entirely from client PTYs; candidate withholds that pass and requires two host-confirmed ones." + }, + "performanceBudget": { + "required": true, + "evidence": "The 100-workspace fixture falls from 100 terminal.list calls plus one compatibility handshake to zero calls. Two-workspace confirmation/recheck fixture falls from five inventories to three. A single status scan and tab membership lookups precede existing requests. Recomputing the required-worktree set from the post-await state adds one in-memory owner resolution per workspace per tick and no RPC. No new timer, concurrency, subprocess or retry; only active when experimental automatic hibernation is enabled." + }, + "knownGaps": [ + "Remote contracts use mocked runtime replies; no live SSH/network-fault, Windows/Linux/WSL or relay run.", + "This improves the experimental automatic hibernation path only; it does not remove inventories for workspaces that contain completed agents." + ], + "promotionCriteria": [ + "Complete CI soak with zero unexplained flakes and preserve the observable oracle." + ], + "demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions." + }, { "id": "terminal-performance.osc-status-scan-budget", "title": "OSC 9999 status bursts reuse forward terminator searches", @@ -165,6 +250,426 @@ ], "demotionRule": "Keep experimental until CI soak; investigate output, offset, carry or search-budget failures without relaxing the oracle." }, + { + "id": "workspace-performance.ai-vault-title-input-guard", + "title": "Title synchronization skips unchanged session collections on live heartbeats", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "renderer-unit", + "surfaces": ["workspace title synchronization", "terminal heartbeat store subscribers"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local"], + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.ts" + ], + "flakeHistory": { + "status": "not-started", + "evidence": "Initial local validation; no CI soak history." + }, + "promotionCriteria": [ + "Complete CI soak with zero unexplained flakes and preserve the observable oracle." + ], + "demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity or resource-count assertions.", + "coverageNotes": "Actual title-sync subscriber and Zustand publications are covered on macOS. Existing fixtures include SSH/runtime owner invalidation and folder workspaces; no live remote transport was launched. The guard has no platform branches, remote execution, liveness verdict or wire change.", + "invariant": "Title-sync publications must not enumerate unchanged session collections, while title identity, provider availability, active pane, stored title and effective execution-host changes retain their previous invalidation decisions.", + "oracle": "Fifty live heartbeat publications with 500 retained and 500 sleeping records cause zero unchanged-map enumerations and no extra scheduled reconciliations. Provider identity, additions/removals, agent/pane ownership, effective host, active pane and stored title changes still invalidate.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts src/renderer/src/lib/ai-vault-tab-title-sync.test.ts src/renderer/src/store/slices/agent-status-batch.test.ts src/renderer/src/store/slices/agent-status-provider-session.test.ts src/renderer/src/store/slices/agent-status-retained-leak.test.ts src/renderer/src/store/slices/agent-status-manual-sleep-capture.test.ts" + ], + "testFiles": [ + "src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts", + "src/renderer/src/lib/ai-vault-tab-title-sync.test.ts", + "src/renderer/src/store/slices/agent-status-batch.test.ts", + "src/renderer/src/store/slices/agent-status-provider-session.test.ts", + "src/renderer/src/store/slices/agent-status-retained-leak.test.ts", + "src/renderer/src/store/slices/agent-status-manual-sleep-capture.test.ts" + ], + "assertionRefs": [ + { + "file": "src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts", + "assertions": [ + "does not enumerate unchanged retained and sleeping maps during live status writes", + "still detects provider changes in %s with other maps reused", + "still checks workspace ownership after unchanged record collections", + "still checks active panes and stored titles after unchanged record collections" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts src/renderer/src/lib/ai-vault-tab-title-sync.test.ts src/renderer/src/store/slices/agent-status-batch.test.ts src/renderer/src/store/slices/agent-status-provider-session.test.ts src/renderer/src/store/slices/agent-status-retained-leak.test.ts src/renderer/src/store/slices/agent-status-manual-sleep-capture.test.ts", + "result": "passed", + "durationSeconds": 5.35, + "summary": "69 tests passed across six files, including actual subscriber scheduling and producer identity contracts." + } + ], + "runtimeBudget": { + "p95Seconds": 30, + "scope": "Focused unit/subscriber suite; local runtime budget, not an established p95." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "The actual subscriber regression fails baseline with 200 unchanged-map enumerations and passes with zero. Frozen source differential matched 13,002 decisions over 6,501 transitions. Four actual producer checks passed against deeply frozen inputs." + }, + "performanceBudget": { + "required": true, + "evidence": "Actual bundled production title subscriber plus Zustand, CPU per 1,000 writes, baseline to candidate: 25 live/100 retained/100 sleeping 41.642 to 0.699 ms; 100/500/500 251.076 to 5.350 ms; 500/500/500 343.025 to 59.897 ms. Zero extra title reads or schedules. Only same-reference collections skip comparisons; no new allocation, cache, timer, IO or scheduling." + }, + "knownGaps": [ + "No launched Electron/native-focus latency test or live external title lookup.", + "No live SSH/WSL/Linux/Windows session; production heartbeat incidence and whole-renderer latency are not established by the fixture." + ] + }, + { + "id": "terminal-performance.status-heartbeat-projection-budget", + "title": "Unchanged status heartbeats reuse the aggregate workspace projection", + "maturity": "experimental", + "protection": "partial", + "owner": "workspace-runtime", + "layer": "shared-and-renderer-unit", + "surfaces": ["desktop runtime graph subscriber", "terminal status heartbeat projection"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "daemon", "ssh", "remote-runtime"], + "coverageNotes": "The always-mounted desktop subscriber is covered through pure projection/reference and graph-sync contracts. All providers use the same serialized fields. WSL, platform launch, liveness, transport and mobile rendering are unaffected.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts" + ], + "invariant": "The desktop agent-status comparison projection remains identical for every serialized field, key membership and freshness bucket while avoiding a full aggregate join when per-entry serialized content is unchanged.", + "oracle": "500 statuses with accumulated assistant previews receive 50 same-bucket heartbeat replacements with zero aggregate joins. A bucket boundary and assistant-detail change invalidate the projection. Existing reference comparisons cover field content, ordering, insertion/removal and subscriber decisions.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts src/renderer/src/runtime/sync-runtime-graph.test.ts" + ], + "testFiles": [ + "src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts", + "src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts", + "src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts", + "src/renderer/src/runtime/sync-runtime-graph.test.ts" + ], + "assertionRefs": [ + { + "file": "src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts", + "assertions": [ + "does not rejoin accumulated previews for timestamp-only heartbeats", + "still rebuilds when an entry changes", + "still rebuilds when a pane is removed, even though every survivor is reused", + "still rebuilds when key membership swaps at a constant entry count", + "still rebuilds when a pane is added" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts src/renderer/src/runtime/sync-runtime-graph.test.ts", + "result": "passed", + "summary": "38 tests passed across four projection and graph-sync files; renderer typecheck passed.", + "durationSeconds": 2.7 + } + ], + "runtimeBudget": { + "p95Seconds": 60, + "scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial local validation; no CI soak history." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "The new heartbeat fixture failed on baseline with 50 aggregate joins; candidate performs zero and still invalidates on the next freshness bucket and new assistant detail." + }, + "performanceBudget": { + "required": true, + "evidence": "50 redundant aggregate joins fall to zero. Paired production projection/equality benchmark: 500 agents with 8 KB previews improves 1.716 to 0.089 ms/update; 1000 improves 3.601 to 0.123 ms/update. Avoids rebuilding 5.06/10.11 million-character aggregate strings. Entry serialization and sorting remain bounded by the current status map; no new cache, polling or asynchronous work." + }, + "knownGaps": [ + "Measured production projection self-time excludes store updates and rendering; no whole-app input-latency claim.", + "Native platforms, live SSH/relay and older clients were not launched. This changes only equality-preserving cache reuse, with no wire or persisted schema change." + ], + "promotionCriteria": [ + "Complete CI soak with zero unexplained flakes and preserve the observable oracle." + ], + "demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions." + }, + { + "id": "terminal-performance.partial-escape-ground-scan", + "title": "Terminal snapshots preserve partial escapes with bounded ground-state scan work", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "shared-and-daemon-unit", + "surfaces": ["headless terminal output ingestion", "terminal snapshot continuity"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "daemon", "ssh", "remote-runtime"], + "coverageNotes": "Shared parser, production emulator and Session contracts cover local/daemon paths; remote snapshot semantics use the existing corruption reproduction without live transport. The scanner has no workspace or host branching; folder/git workspaces, WSL, mobile/relay and platform execution/wire boundaries are unchanged.", + "motivatingLinks": ["https://github.com/stablyai/orca/issues/7329"], + "invariant": "Partial escape tracking preserves exact pending UTF-16 and completion across chunk/snapshot boundaries while skipping ordinary ground-state text without a JavaScript per-code-unit walk.", + "oracle": "Colored ASCII and UTF-16 output retains the exact incomplete CSI tail using fewer than 32 code-unit inspections. Every split of CSI/OSC/DCS/ESC-intermediate controls and CAN/SUB/ESC aborts preserve completion; one-character echo performs zero inspections or native ESC searches. Seeded headless snapshot restore matches a renderer terminal twin.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/shared/terminal-partial-escape-tail-ground-scan.test.ts src/shared/terminal-partial-escape-tail.test.ts src/shared/terminal-partial-escape-tail.fuzz.test.ts src/main/daemon/headless-emulator.test.ts", + "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/repro-7329-remote-snapshot-corruption.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/headless-emulator-fidelity.fuzz.test.ts" + ], + "testFiles": [ + "src/shared/terminal-partial-escape-tail-ground-scan.test.ts", + "src/shared/terminal-partial-escape-tail.test.ts", + "src/shared/terminal-partial-escape-tail.fuzz.test.ts", + "src/main/daemon/headless-emulator.test.ts", + "src/main/daemon/repro-7329-remote-snapshot-corruption.test.ts", + "src/main/daemon/session-shell-recovery.test.ts", + "src/main/daemon/headless-emulator-fidelity.fuzz.test.ts" + ], + "assertionRefs": [ + { + "file": "src/shared/terminal-partial-escape-tail-ground-scan.test.ts", + "assertions": [ + "skips ordinary %s text between completed escapes", + "preserves %j through every split after ordinary text", + "handles aborts before returning to ordinary text", + "keeps one-character echo free of per-code-unit scans and escape searches" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/shared/terminal-partial-escape-tail-ground-scan.test.ts src/shared/terminal-partial-escape-tail.test.ts src/shared/terminal-partial-escape-tail.fuzz.test.ts src/main/daemon/headless-emulator.test.ts", + "result": "passed", + "durationSeconds": 2.17, + "summary": "82 tests passed across four files including 593,468 fold-fuzz cases." + }, + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/daemon/repro-7329-remote-snapshot-corruption.test.ts src/main/daemon/session-shell-recovery.test.ts src/main/daemon/headless-emulator-fidelity.fuzz.test.ts", + "result": "passed", + "durationSeconds": 36.24, + "summary": "15 tests passed across three files including 300 headless-to-renderer snapshot fidelity streams." + } + ], + "runtimeBudget": { + "p95Seconds": 60, + "scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial local validation; no CI soak history." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "New ASCII/UTF-16 work budgets failed baseline at 262,160/196,624 inspected code units; candidate inspects 16 code units and 2 native ESC searches in each with exact tails and completion. External differential matched 2,710,126 cases (plus 388,416 hybrid-vs-baseline cases over the full VT alphabet with lone/split surrogates, 0 mismatches); existing fold fuzz covers 593,468 cases." + }, + "performanceBudget": { + "required": true, + "evidence": "Production HeadlessEmulator.writeSync CPU for 2 MiB colored logs improves 32.321 to 26.638 ms; sparse ANSI improves 26.757 to 21.982 ms. Agent-redraw CPU improves 402.308 to 394.633 ms. Rotated 100,000-write baseline/identical-control/candidate medians: plain echo 27.340/27.015/27.822 ms, colored echo 36.009/35.556/35.981 ms. Native ESC searches advance only in ground, and only when the current code unit is not already ESC, so dense back-to-back CSI streams do not pay a search per sequence: 0.9 MiB dense SGR/CSI medians are 2.04 ms baseline, 2.56 ms search-always, 1.92 ms shipped. Existing plain-output fast path is unchanged. No new state, allocation, timer, provider call or transport behavior." + }, + "knownGaps": [ + "No launched Electron or live SSH/WSL/Windows/Linux process, native input or end-to-end UI-latency measurement.", + "Existing 4,096-code-unit tracking abandonment and idle-death partial-tail behavior are unchanged." + ], + "promotionCriteria": [ + "Complete CI soak with zero unexplained flakes and preserve the observable oracle." + ], + "demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions." + }, + { + "id": "terminal-performance.pending-control-storage", + "title": "Pending terminal controls release consumed output backing storage", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "shared-and-renderer-unit", + "surfaces": [ + "terminal output ingestion", + "pending agent status frames", + "terminal preview normalization", + "terminal title tracking" + ], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "daemon", "ssh", "remote-runtime"], + "coverageNotes": "Shared code-unit and production normalizer/parser tests cover provider-independent behavior on macOS; existing buffer contracts cover normalized output. Native execution, process ownership, wire formats and mobile UI are unchanged. Folder/git identity is not inspected by these string functions.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/src/shared/agent-status-osc.test.ts" + ], + "invariant": "Small incomplete status, ANSI and title controls must not retain the larger consumed output strings they were sliced from. Every retained control fragment is routed through ownRetainedString, which preserves exact UTF-16, current caps and trimming, payload order and clean-output offsets on both the Buffer and the code-unit fallback copier.", + "oracle": "One forced-GC fixture proves the primitive itself: 32 x 1 Mi parents pinned by 32 x 4 Ki slices retain over 16 MiB, and the same tails owned retain under 4 MiB and under an eighth of the sliced figure. Each retention site is then covered deterministically by spying on ownRetainedString: the retained value is routed through it on every chunk, including growing fragments. Large trimmed ANSI/title tails preserve their introducers and newest payload units; lone surrogates and split pairs survive BEL/ST completion; the Buffer-free fallback is byte-identical.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/own-retained-string.test.ts src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-pending-retention.test.ts src/shared/agent-status-types.test.ts src/main/runtime/terminal-ansi-pending-retention.test.ts src/shared/osc-title-scan-tail-retention.test.ts src/shared/osc-title-scan-tail.test.ts src/main/runtime/terminal-tail-buffer.test.ts src/main/runtime/terminal-tail-whitespace.test.ts" + ], + "testFiles": [ + "src/shared/own-retained-string.test.ts", + "src/shared/agent-status-osc.test.ts", + "src/shared/agent-status-osc-pending-retention.test.ts", + "src/shared/agent-status-types.test.ts", + "src/main/runtime/terminal-ansi-pending-retention.test.ts", + "src/shared/osc-title-scan-tail-retention.test.ts", + "src/shared/osc-title-scan-tail.test.ts", + "src/main/runtime/terminal-tail-buffer.test.ts", + "src/main/runtime/terminal-tail-whitespace.test.ts" + ], + "assertionRefs": [ + { + "file": "src/shared/own-retained-string.test.ts", + "assertions": [ + "releases the parent chunk that a retained tail was sliced from", + "round-trips %s exactly", + "leaves already-flat short strings alone", + "matches the block copier when Buffer is unavailable" + ] + }, + { + "file": "src/shared/agent-status-osc-pending-retention.test.ts", + "assertions": [ + "routes every retained pending frame through ownRetainedString", + "keeps %i-character chunks with %i incomplete statuses byte-exact", + "preserves raw UTF-16 across an owned suffix and %j", + "drops an owned frame that grows past the pending cap" + ] + }, + { + "file": "src/main/runtime/terminal-ansi-pending-retention.test.ts", + "assertions": [ + "routes every retained pending control through ownRetainedString", + "keeps %i-character chunks with %i incomplete statuses byte-exact", + "preserves trimming and code units for %j", + "preserves split UTF-16 through %j termination", + "keeps trimming an owned fragment that grows past the cap" + ] + }, + { + "file": "src/shared/osc-title-scan-tail-retention.test.ts", + "assertions": [ + "routes every retained title tail through ownRetainedString", + "keeps %i-character chunks with %i incomplete titles byte-exact", + "preserves title %s introducer and exact UTF-16 at the cap", + "keeps trimming an owned title that grows past the cap" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/own-retained-string.test.ts src/shared/agent-status-osc.test.ts src/shared/agent-status-osc-pending-retention.test.ts src/shared/agent-status-types.test.ts src/main/runtime/terminal-ansi-pending-retention.test.ts src/shared/osc-title-scan-tail-retention.test.ts src/shared/osc-title-scan-tail.test.ts src/main/runtime/terminal-tail-buffer.test.ts src/main/runtime/terminal-tail-whitespace.test.ts", + "result": "passed", + "durationSeconds": 11.4, + "summary": "114 tests passed across nine primitive, parser, payload, normalization, title and terminal-buffer files." + } + ], + "runtimeBudget": { + "p95Seconds": 60, + "scope": "Focused unit/provider-contract suite; local runtime budget, not an established p95." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "Initial local validation; no CI soak history. Only one fixture depends on forced GC; the three per-site retention checks are deterministic spy assertions." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "The primitive fixture measures 32.00 MB retained for plain slices against 1.13 MB owned, so its 16 MiB/4 MiB bounds fail without ownership. Removing ownRetainedString from terminal-ansi-normalization.ts reproducibly fails 'routes every retained pending control through ownRetainedString' while the ten fidelity cases still pass, confirming the spy assertions carry the retention contract rather than the fidelity ones. Exact-output equivalence against the pre-change parsers was re-checked in the differential fixtures: byte-exact clean output, trimmed tails, payload order and clean offsets are unchanged." + }, + "performanceBudget": { + "required": true, + "evidence": "ownRetainedString is a Buffer utf16le round trip, measured at 0.57 us for 4 Ki code units and 21.9 us for 64 Ki, against 10.0 us and 170.3 us for the code-unit block copier it replaces (min of 12 rounds x 500, isolated processes). Strings below V8 SlicedString::kMinLength (13) are returned unchanged. Interleaved min-of-15 comparisons against the un-owned parsers: 16 Ki-char ANSI chunks with a 4 Ki tail 48.7 to 50.1 us/chunk (+2.8%); 4 Ki-char ANSI chunks with a 2 Ki tail 14.7 to 12.9 us/chunk (-12.0%); 194 Ki-char status chunks with a 64 Ki pending frame 298.1 to 360.5 us/chunk (+21.0%). Ordinary streams with no pending control never call the primitive. Plain-output paths, scheduling, provider calls, limits and trimming are unchanged." + }, + "knownGaps": [ + "No live Electron input-latency or Linux/Windows/WSL/SSH execution measurement; exact string and tail behavior is covered on macOS.", + "The primitive's GC fixture is experimental with no CI soak. A provider input already sliced from an unseen larger ancestor can retain that ancestor; no universal heap cap or whole-process memory reduction is claimed.", + "The renderer and mobile fallback copier is covered by a Buffer-free equivalence test, not by a real renderer bundle run.", + "Tail-buffer row retention is a larger instance of the same pattern and is not covered by this gate." + ], + "promotionCriteria": [ + "Complete CI soak with zero unexplained flakes and preserve the observable oracle." + ], + "demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity, liveness or resource-count assertions." + }, + { + "id": "terminal-performance.vertical-control-scan", + "title": "Main terminal preview scanning skips ordinary output between controls", + "maturity": "experimental", + "protection": "partial", + "owner": "terminal-runtime", + "layer": "main-runtime-unit", + "surfaces": ["terminal output ingestion", "terminal preview and read tails"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "daemon", "ssh", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local"], + "motivatingLinks": [ + "https://github.com/stablyai/orca/blob/main/.agents/skills/perf/SKILL.md" + ], + "flakeHistory": { + "status": "not-started", + "evidence": "Initial local validation; no CI soak history." + }, + "promotionCriteria": [ + "Complete CI soak with zero unexplained flakes and preserve the observable oracle." + ], + "demotionRule": "Keep experimental until CI soak; investigate failures without relaxing fidelity or resource-count assertions.", + "coverageNotes": "Pure shared-host string semantics and main runtime tests cover the production scanner on macOS. Folder/git workspaces, local/daemon/SSH/remote-runtime authority, wire content and mobile/relay behavior are unchanged. No live native remote session was launched.", + "invariant": "Main terminal preview/read tails choose the same append/redraw path while ordinary text is skipped without a JavaScript code-unit walk. Complete string-control payloads are not interpreted as vertical controls; incomplete controls stop at the same point.", + "oracle": "Plain, colored and vertical-CSI output uses fewer than 16 code-unit inspections with the same recognition result. Canonical and noncanonical CSI, embedded CSI in OSC/DCS/SOS/PM/APC, ESC inside CSI parameter bytes, back-to-back controls and incomplete sequences preserve decisions. Production normalization and tail append retain exact cursor-up redraw rows.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/main/runtime/terminal-vertical-control-scan.test.ts" + ], + "testFiles": ["src/main/runtime/terminal-vertical-control-scan.test.ts"], + "assertionRefs": [ + { + "file": "src/main/runtime/terminal-vertical-control-scan.test.ts", + "assertions": [ + "bounds code-unit inspections on %s output", + "preserves numeric CSI A recognition for %j", + "skips embedded CSI and stops at an incomplete %s", + "resumes scanning after a parsed control for %j", + "preserves tail rows when ordinary output is followed by a cursor-up redraw" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/main/runtime/terminal-vertical-control-scan.test.ts", + "result": "passed", + "durationSeconds": 0.325, + "summary": "29 scanner work-budget, control recognition, scan-resumption and complete-tail integration tests passed." + } + ], + "runtimeBudget": { + "p95Seconds": 15, + "scope": "Focused scanner and tail-contract suite; local runtime budget, not an established p95." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "Baseline inspections 90,112/90,119/98,307 become 0/5/2. Frozen differential matched 204,925 predicate cases and 3,000 complete tail transitions. Broader integrated validation passed 1,308 tests with one existing skip across six files. Independent original-base normalizer with only this scanner change passed 1,298 tests with one existing skip across five files (20.55 seconds), using a temporary Vite source override; this confirms independence from the pending-storage PR." + }, + "performanceBudget": { + "required": true, + "evidence": "Actual normalize, append-tail, transcript and preview pipeline: 4 MiB/64 KiB-chunk CPU medians plain 43.418 to 37.932 ms, colored 49.321 to 42.570, wide 41.022 to 32.883, dense newlines 37.112 to 31.601, no-newline 32.009 to 25.100. 10,000 sequential single-character writes into a growing partial line 190.374 to 126.820 ms. Equal-bundle rotated controls show no material tiny-echo/redraw regression. Forward native ESC search skips only ground text; no new retained state, scheduling, timer or provider calls." + }, + "knownGaps": [ + "No live end-to-end input latency, many-pane frame measurement or launched Electron/SSH/WSL/Linux/Windows journey.", + "Existing preview fidelity and incomplete-control handling remain unchanged; broader runtime suite has one existing skipped case." + ] + }, { "id": "terminal-performance.padded-fullscreen-redraw", "title": "Fullscreen redraw padding does not stall terminal delivery", diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index 724be685ea7..39008eaa963 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 44m + + downloads: 45m @@ -15,7 +15,7 @@ downloads downloads - 44m - 44m + 45m + 45m diff --git a/src/main/ipc/parcel-watcher-event-cancellation.test.ts b/src/main/ipc/parcel-watcher-event-cancellation.test.ts new file mode 100644 index 00000000000..f7cb759cf2e --- /dev/null +++ b/src/main/ipc/parcel-watcher-event-cancellation.test.ts @@ -0,0 +1,103 @@ +import { setImmediate } from 'node:timers/promises' +import { beforeEach, describe, expect, it, vi } from 'vitest' +const { statMock } = vi.hoisted(() => ({ statMock: vi.fn() })) +vi.mock('node:fs/promises', () => ({ stat: statMock })) +import { + createWatcherProcessEventDeliveryQueue, + prepareWatcherProcessEvents +} from './parcel-watcher-event-delivery' + +const delivery = { includeDirectoryMetadata: true, maxEventsPerBatch: 100 } +const directory = { isDirectory: () => true } +const events = (prefix: string, count: number) => + Array.from({ length: count }, (_, index) => ({ + type: 'update' as const, + path: `/${prefix}/${index}` + })) +beforeEach(() => { + statMock.mockReset() +}) + +describe('closed watcher metadata work', () => { + it('removes queued lookups before a new live subscription needs their slots', async () => { + const gate = Promise.withResolvers() + statMock.mockImplementation(async (path: string) => { + if (path.startsWith('/held/')) { + await gate.promise + } + return directory + }) + const held = prepareWatcherProcessEvents(events('held', 8), delivery) + await setImmediate() + expect(statMock.mock.calls.length).toBe(8) + const deliver = vi.fn(async () => undefined) + const onError = vi.fn() + const queue = createWatcherProcessEventDeliveryQueue(delivery, deliver, onError) + let live: ReturnType | undefined + try { + queue.enqueue(events('closed', 32)) + queue.close() + live = prepareWatcherProcessEvents(events('live', 1), delivery) + gate.resolve() + expect(await live).toEqual([{ type: 'update', path: '/live/0', isDirectory: true }]) + await held + await setImmediate() + expect(statMock.mock.calls.map(([path]) => path)).toEqual([ + ...events('held', 8).map((event) => event.path), + '/live/0' + ]) + expect(deliver).not.toHaveBeenCalled() + expect(onError).not.toHaveBeenCalled() + } finally { + gate.resolve() + queue.close() + await held + await live + } + }) + + it('lets active stats settle without starting the rest of a closed batch', async () => { + const gate = Promise.withResolvers() + statMock.mockImplementation(async () => { + await gate.promise + return directory + }) + const deliver = vi.fn(async () => undefined) + const onError = vi.fn() + const queue = createWatcherProcessEventDeliveryQueue(delivery, deliver, onError) + try { + queue.enqueue(events('active', 32)) + await setImmediate() + expect(statMock.mock.calls.length).toBe(8) + queue.close() + gate.resolve() + await setImmediate() + expect(statMock.mock.calls.length).toBe(8) + expect(deliver).not.toHaveBeenCalled() + expect(onError).not.toHaveBeenCalled() + await prepareWatcherProcessEvents(events('next', 8), delivery) + expect(statMock.mock.calls.length).toBe(16) + } finally { + gate.resolve() + queue.close() + await setImmediate() + } + }) + + it('still delivers file-like invalidations for live stat failures and skips deleted paths', async () => { + statMock.mockRejectedValue(new Error('file vanished')) + expect( + await prepareWatcherProcessEvents( + [ + { type: 'update', path: '/vanished' }, + { type: 'delete', path: '/deleted' } + ], + delivery + ) + ).toEqual([ + { type: 'update', path: '/vanished', isDirectory: false }, + { type: 'delete', path: '/deleted' } + ]) + expect(statMock).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/ipc/parcel-watcher-event-delivery.ts b/src/main/ipc/parcel-watcher-event-delivery.ts index 54e8448801c..ed208b964f8 100644 --- a/src/main/ipc/parcel-watcher-event-delivery.ts +++ b/src/main/ipc/parcel-watcher-event-delivery.ts @@ -1,4 +1,6 @@ import { stat } from 'node:fs/promises' +import { PrioritySemaphore } from '../../shared/priority-semaphore' +import { mapWithConcurrency } from '../../shared/map-with-concurrency' import type { Event as ParcelWatcherEvent } from '@parcel/watcher' import { MAX_BATCHED_WATCHER_EVENTS } from './filesystem-watcher-event-batch' import type { @@ -7,70 +9,37 @@ import type { } from './parcel-watcher-process-protocol' const DIRECTORY_STAT_CONCURRENCY = 8 -let activeDirectoryStats = 0 -const directoryStatWaiters: (() => void)[] = [] +const directoryStatSlots = new PrioritySemaphore(DIRECTORY_STAT_CONCURRENCY) export type WatcherProcessEventDeliveryQueue = { enqueue(events: readonly ParcelWatcherEvent[]): void close(): void } -async function acquireDirectoryStatSlot(): Promise { - if (activeDirectoryStats < DIRECTORY_STAT_CONCURRENCY) { - activeDirectoryStats++ - return - } - await new Promise((resolve) => directoryStatWaiters.push(resolve)) -} - -function releaseDirectoryStatSlot(): void { - const next = directoryStatWaiters.shift() - if (next) { - // Transfer the existing slot directly so a newly arriving task cannot - // overtake this waiter and temporarily exceed the global budget. - next() - return - } - activeDirectoryStats-- -} - -async function statWatcherEventPath(eventPath: string): Promise { - await acquireDirectoryStatSlot() +async function statWatcherEventPath(eventPath: string, signal?: AbortSignal): Promise { + const release = await directoryStatSlots.acquire(0, signal) try { + signal?.throwIfAborted() return (await stat(eventPath)).isDirectory() } finally { - releaseDirectoryStatSlot() + release() } } -async function mapWithConcurrency( - items: readonly T[], - limit: number, - mapper: (item: T) => Promise -): Promise { - const results = Array.from({ length: items.length }) - let cursor = 0 - const lane = async (): Promise => { - while (cursor < items.length) { - const index = cursor++ - results[index] = await mapper(items[index]) - } - } - await Promise.all(Array.from({ length: Math.min(limit, items.length) }, lane)) - return results -} - async function mapWatcherEvent( event: ParcelWatcherEvent, - includeDirectoryMetadata: boolean + includeDirectoryMetadata: boolean, + signal?: AbortSignal ): Promise { + signal?.throwIfAborted() if (!includeDirectoryMetadata || event.type === 'delete') { return { type: event.type, path: event.path } } let isDirectory = false try { - isDirectory = await statWatcherEventPath(event.path) + isDirectory = await statWatcherEventPath(event.path, signal) } catch { + signal?.throwIfAborted() // Why: a path can vanish between the native event and metadata lookup. // Treat unknown metadata as a file-like event so parent invalidation still runs. } @@ -79,8 +48,10 @@ async function mapWatcherEvent( export async function prepareWatcherProcessEvents( events: readonly ParcelWatcherEvent[], - delivery: WatcherProcessDeliveryOptions | undefined + delivery: WatcherProcessDeliveryOptions | undefined, + signal?: AbortSignal ): Promise { + signal?.throwIfAborted() if (delivery?.maxEventsPerBatch !== undefined && events.length > delivery.maxEventsPerBatch) { return null } @@ -88,7 +59,7 @@ export async function prepareWatcherProcessEvents( return events.map((event) => ({ type: event.type, path: event.path })) } return mapWithConcurrency(events, DIRECTORY_STAT_CONCURRENCY, (event) => - mapWatcherEvent(event, true) + mapWatcherEvent(event, true, signal) ) } @@ -99,6 +70,7 @@ export function createWatcherProcessEventDeliveryQueue( onError: (error: unknown) => void ): WatcherProcessEventDeliveryQueue { const eventLimit = delivery?.maxEventsPerBatch ?? MAX_BATCHED_WATCHER_EVENTS + const controller = new AbortController() let active = true let draining = false let pendingOverflow = false @@ -119,7 +91,7 @@ export function createWatcherProcessEventDeliveryQueue( await deliver(null) continue } - const prepared = await prepareWatcherProcessEvents(events, delivery) + const prepared = await prepareWatcherProcessEvents(events, delivery, controller.signal) if (active) { await deliver(prepared) } @@ -153,6 +125,7 @@ export function createWatcherProcessEventDeliveryQueue( }, close(): void { active = false + controller.abort() pendingEvents = [] pendingOverflow = false } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts index 87bc33bc4b9..2d1b3d248f8 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts @@ -167,12 +167,14 @@ describe('Claude structured session handoff options', () => { ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) - await vi.waitFor(async () => - expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + await vi.waitFor( + async () => expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }), + { timeout: 10_000 } ) expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) - await vi.waitFor(async () => - expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + await vi.waitFor( + async () => expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }), + { timeout: 10_000 } ) expect(acquire.mock.calls[1]?.[0].options).toEqual({ model: PICKED_MODEL }) diff --git a/src/main/plugins/plugin-audit-log-scaling.test.ts b/src/main/plugins/plugin-audit-log-scaling.test.ts new file mode 100644 index 00000000000..45abd9b6470 --- /dev/null +++ b/src/main/plugins/plugin-audit-log-scaling.test.ts @@ -0,0 +1,115 @@ +import { expect, it, vi } from 'vitest' +import { PluginAuditLog } from './plugin-audit-log' + +const source = vi.hoisted(() => ({ content: '' })) +vi.mock('node:fs/promises', () => ({ + readFile: async (path: string) => (path.endsWith('.1') ? '' : source.content) +})) + +it('extracts the recent window without splitting all historical log records', async () => { + source.content = `${Array.from({ length: 10000 }, (_, ts) => JSON.stringify({ ts })).join('\n')}\n` + const split = vi.spyOn(String.prototype, 'split') + let entries: Awaited> + try { + entries = await new PluginAuditLog('/logs').readRecent(200) + expect(split.mock.calls.length).toBe(0) + } finally { + split.mockRestore() + } + expect(entries!.map((entry) => entry.ts)).toEqual( + Array.from({ length: 200 }, (_, index) => 9800 + index) + ) +}) + +it('preserves blank, malformed, unterminated and unusual-limit selection', async () => { + source.content = '\n{"ts":1}\n\ninvalid\n \n{"ts":2}' + const original = (limit: number) => + source.content + .split('\n') + .filter((line) => line.length > 0) + .slice(-limit) + .flatMap((line) => { + try { + return [JSON.parse(line)] + } catch { + return [] + } + }) + const log = new PluginAuditLog('/logs') + for (const limit of [1, 2, 3, 4, 5, 0, -1, 0.5, 1.5, Infinity, Number.NaN]) { + expect(await log.readRecent(limit)).toEqual(original(limit)) + } +}) + +/** The pre-scan `split/filter/slice` selection, as the differential oracle. */ +function referenceRecentLines(text: string, limit: number): string[] { + return text + .split('\n') + .filter((line) => line.length > 0) + .slice(-limit) +} + +function makeRandom(seed: number): () => number { + let state = seed >>> 0 + return () => { + state = (state * 1664525 + 1013904223) >>> 0 + return state / 0x100000000 + } +} + +it('selects the identical line set as the full split over randomized log shapes', async () => { + const log = new PluginAuditLog('/logs') + let nonEmptyCases = 0 + for (let seed = 1; seed <= 1500; seed += 1) { + const random = makeRandom(seed) + const pieces: string[] = [] + const lineCount = Math.floor(random() * 12) + for (let i = 0; i < lineCount; i += 1) { + const kind = random() + // Blank lines, whitespace-only lines, CRLF rows, non-JSON rows and multi-byte + // payloads all have to land on the same boundaries the split-based scan found. + if (kind < 0.15) { + pieces.push('') + } else if (kind < 0.25) { + pieces.push(' ') + } else if (kind < 0.35) { + pieces.push('not json') + } else if (kind < 0.45) { + pieces.push(`${JSON.stringify({ ts: i, summary: 'ünïcøde ✅' })}\r`) + } else { + pieces.push(JSON.stringify({ ts: i, actor: 'plugin:x' })) + } + } + // Half the corpora end without a trailing newline (a torn final append). + source.content = pieces.join('\n') + (random() < 0.5 ? '\n' : '') + + for (const limit of [1, 2, 3, 5, 200]) { + const expected = referenceRecentLines(source.content, limit).flatMap((line) => { + try { + return [JSON.parse(line)] + } catch { + return [] + } + }) + expect(await log.readRecent(limit), `seed ${seed} limit ${limit}`).toEqual(expected) + if (expected.length > 0) { + nonEmptyCases += 1 + } + } + } + expect(nonEmptyCases).toBeGreaterThan(1000) +}) + +it('never yields a record split across a line boundary', async () => { + // A 200-record window from the tail of a file whose records are long and carry + // escaped newlines: every returned line must still parse to one whole record. + source.content = `${Array.from({ length: 3000 }, (_, ts) => + JSON.stringify({ ts, summary: `line\\nwith escapes ${'x'.repeat(200)}` }) + ).join('\n')}\n` + const entries = await new PluginAuditLog('/logs').readRecent(200) + expect(entries).toHaveLength(200) + expect(entries.map((entry) => entry.ts)).toEqual( + Array.from({ length: 200 }, (_, index) => 2800 + index) + ) + expect(entries.every((entry) => entry.summary.endsWith('x'.repeat(200)))).toBe(true) +}) diff --git a/src/main/plugins/plugin-audit-log.ts b/src/main/plugins/plugin-audit-log.ts index 0b5f4cd3056..fd5baaf3126 100644 --- a/src/main/plugins/plugin-audit-log.ts +++ b/src/main/plugins/plugin-audit-log.ts @@ -71,8 +71,8 @@ export class PluginAuditLog { [this.rotatedFilePath, this.filePath].map((path) => readFile(path, 'utf8').catch(() => '')) ) const text = rotated + current - const lines = text.split('\n').filter((line) => line.length > 0) - return lines.slice(-limit).flatMap((line) => { + const lines = recentAuditLines(text, limit) + return lines.flatMap((line) => { try { return [JSON.parse(line) as PluginAuditEntry] } catch { @@ -84,3 +84,26 @@ export class PluginAuditLog { } } } + +function recentAuditLines(text: string, limit: number): string[] { + if (!Number.isFinite(limit) || limit < 1) { + return text + .split('\n') + .filter((line) => line.length > 0) + .slice(-limit) + } + const lines: string[] = [] + let end = text.length + const count = Math.trunc(limit) + while (end > 0 && lines.length < count) { + const start = text.lastIndexOf('\n', end - 1) + 1 + if (start < end) { + lines.push(text.slice(start, end)) + } + if (start === 0) { + break + } + end = start - 1 + } + return lines.toReversed() +} diff --git a/src/main/runtime/orca-runtime-schedule-wait-blocked-check.ts b/src/main/runtime/orca-runtime-schedule-wait-blocked-check.ts index cc3bd4f6d3c..f6219217925 100644 --- a/src/main/runtime/orca-runtime-schedule-wait-blocked-check.ts +++ b/src/main/runtime/orca-runtime-schedule-wait-blocked-check.ts @@ -16,6 +16,7 @@ import { computeTerminalTailWaitState, tailGainedNewerBlockedReason } from './terminal-wait-tail-state' +import { ownRetainedString } from '../../shared/own-retained-string' import type { ProcessedAgentStatusChunk } from '../../shared/agent-status-osc' import { createAgentStatusOscProcessor } from '../../shared/agent-status-osc' import type { RuntimePtyTitleTrackerEntry } from './runtime-terminal-state-records' @@ -32,7 +33,9 @@ export class OrcaRuntimeWithScheduleWaitBlockedCheck extends OrcaRuntimeWithOnPt // plus the concatenation the pattern flattens anyway. const keywordWindow = `${state.keywordCarry}${appendedText}`.toLowerCase() const keywordHit = WAIT_BLOCKED_KEYWORD_PATTERN.test(keywordWindow) - state.keywordCarry = keywordWindow.slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS) + // Why own: this 31-char carry lives per PTY until the next chunk, and un-owned it pins the + // whole lowercased window — a full copy of the chunk. + state.keywordCarry = ownRetainedString(keywordWindow.slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS)) appendWaitBlockedCarry(state.appended, appendedText) const elapsed = at - state.lastAt if (keywordHit || elapsed >= WAIT_BLOCKED_CHECK_MIN_INTERVAL_MS || elapsed < 0) { diff --git a/src/main/runtime/terminal-ansi-normalization.ts b/src/main/runtime/terminal-ansi-normalization.ts index e6be4dac859..df76da3c35e 100644 --- a/src/main/runtime/terminal-ansi-normalization.ts +++ b/src/main/runtime/terminal-ansi-normalization.ts @@ -1,4 +1,5 @@ import { MAX_TAIL_PENDING_ANSI_CHARS } from './terminal-tail-limits' +import { ownRetainedString } from '../../shared/own-retained-string' export function parseAnsiControlSequence( value: string, @@ -59,14 +60,9 @@ export function hasCanonicalNumericCsiParams(params: string): boolean { return /^[0-9;]*$/.test(params) } -const ESCAPE_CHAR_CODE = 0x1b - export function containsTerminalVerticalLineControl(value: string): boolean { - for (let index = 0; index < value.length; index += 1) { - // Why charCodeAt: `value[index]` mints a one-char string per position on every chunk. - if (value.charCodeAt(index) !== ESCAPE_CHAR_CODE) { - continue - } + // Only ESC can introduce a vertical control; ordinary output needs no code-unit walk. + for (let index = value.indexOf('\x1b'); index !== -1; index = value.indexOf('\x1b', index + 1)) { const parsed = parseAnsiControlSequence(value, index) if (!parsed) { return false @@ -105,7 +101,8 @@ export function normalizeTerminalChunk( if (!parsed) { return { text: parts.join(''), - pendingAnsi: trimPendingAnsiControl(combined.slice(index)) + // Own the tail so it stops pinning the consumed chunk it was sliced from. + pendingAnsi: ownRetainedString(trimPendingAnsiControl(combined.slice(index))) } } if (parsed.kind === 'csi' && isTerminalPreviewLineControl(parsed)) { diff --git a/src/main/runtime/terminal-ansi-pending-retention.test.ts b/src/main/runtime/terminal-ansi-pending-retention.test.ts new file mode 100644 index 00000000000..342a99e098e --- /dev/null +++ b/src/main/runtime/terminal-ansi-pending-retention.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import * as ownership from '../../shared/own-retained-string' +import { normalizeTerminalChunk } from './terminal-ansi-normalization' +import { MAX_TAIL_PENDING_ANSI_CHARS } from './terminal-tail-limits' + +const INCOMPLETE_STATUS = '\x1b]9999;{"state":"working","prompt":"fragment' + +describe('terminal preview pending ANSI storage', () => { + it('routes every retained pending control through ownRetainedString', () => { + const own = vi.spyOn(ownership, 'ownRetainedString') + try { + const first = normalizeTerminalChunk('x'.repeat(32 * 1024) + INCOMPLETE_STATUS) + expect(own).toHaveBeenCalledTimes(1) + expect(own).toHaveBeenLastCalledWith(INCOMPLETE_STATUS) + expect(first.pendingAnsi).toBe(INCOMPLETE_STATUS) + + // Ownership is unconditional: a growing fragment is re-owned on every chunk. + let pending = first.pendingAnsi + for (let index = 0; index < 8; index += 1) { + pending = normalizeTerminalChunk('x'.repeat(512), pending).pendingAnsi + } + expect(own).toHaveBeenCalledTimes(9) + expect(own).toHaveBeenLastCalledWith(pending) + } finally { + own.mockRestore() + } + }) + + it.each([ + [16 * 1024, 256, INCOMPLETE_STATUS], + [64 * 1024, 64, INCOMPLETE_STATUS], + [1024 * 1024, 16, INCOMPLETE_STATUS], + [16 * 1024, 128, INCOMPLETE_STATUS + 'x'.repeat(16 * 1024)] + ])('keeps %i-character chunks with %i incomplete statuses byte-exact', (size, count, control) => { + const tails: string[] = [] + let cleanChars = 0 + for (let index = 0; index < count; index++) { + const result = normalizeTerminalChunk( + String.fromCharCode(65 + (index % 26)).repeat(size) + control + ) + cleanChars += result.text.length + tails.push(result.pendingAnsi) + } + + expect(cleanChars).toBe(size * count) + const expected = + control.length <= MAX_TAIL_PENDING_ANSI_CHARS + ? control + : control.slice(0, 2) + control.slice(-(MAX_TAIL_PENDING_ANSI_CHARS - 2)) + for (const pending of tails) { + expect(pending).toBe(expected) + expect(normalizeTerminalChunk('"}\x07after', pending)).toEqual({ + text: 'after', + pendingAnsi: '' + }) + } + }) + + it.each(['\x1b]', '\x1bP', '\x1b['])('preserves trimming and code units for %j', (prefix) => { + for (const length of [4095, 4096, 4097, 16 * 1024]) { + const value = `${prefix}${'x'.repeat(length - 7)}漢\ud8001\udc00\ud83d` + const expected = + value.length <= MAX_TAIL_PENDING_ANSI_CHARS + ? value + : prefix + value.slice(-(MAX_TAIL_PENDING_ANSI_CHARS - prefix.length)) + // CSI parameters must stay below its final-byte range until the suffix is retained. + const input = prefix === '\x1b[' ? value.replaceAll('x', '1') : value + const expectedInput = prefix === '\x1b[' ? expected.replaceAll('x', '1') : expected + const result = normalizeTerminalChunk('a'.repeat(32 * 1024) + input) + expect(result).toEqual({ + text: 'a'.repeat(32 * 1024), + pendingAnsi: expectedInput + }) + } + }) + + it.each(['\x07', '\x1b\\'])('preserves split UTF-16 through %j termination', (terminator) => { + const pending = '\x1b]2;漢\ud800|\udc00|\ud83d' + const first = normalizeTerminalChunk('x'.repeat(16 * 1024) + pending) + expect(first.pendingAnsi).toBe(pending) + expect(normalizeTerminalChunk(`\ude00${terminator}after`, first.pendingAnsi)).toEqual({ + text: 'after', + pendingAnsi: '' + }) + }) + + it('keeps trimming an owned fragment that grows past the cap', () => { + let pending = '\x1b]2;' + for (let index = 0; index < 128; index++) { + pending = normalizeTerminalChunk('x'.repeat(512), pending).pendingAnsi + } + expect(pending).toBe(`\x1b]${'x'.repeat(MAX_TAIL_PENDING_ANSI_CHARS - 2)}`) + expect(normalizeTerminalChunk('\x07after', pending)).toEqual({ + text: 'after', + pendingAnsi: '' + }) + }) +}) diff --git a/src/main/runtime/terminal-tail-buffer.ts b/src/main/runtime/terminal-tail-buffer.ts index b3e15d1f375..de9a36331fb 100644 --- a/src/main/runtime/terminal-tail-buffer.ts +++ b/src/main/runtime/terminal-tail-buffer.ts @@ -1,4 +1,5 @@ import { containsTerminalVerticalLineControl } from './terminal-ansi-normalization' +import { ownRetainedString } from '../../shared/own-retained-string' import { carryTerminalTailSentinelMatches } from './terminal-tail-sentinel-index' import { applyTerminalLineControls, @@ -171,11 +172,13 @@ export function appendNormalizedToTailBuffer( const pieces = processTerminalTailCompleteSegments(segments.completeSegments) const newlyCompletedLines: string[] = [] for (const piece of pieces) { - newlyCompletedLines.push(trimTerminalLineRight(piece)) + // Why own: every retained row is sliced from this chunk but outlives it by up to + // MAX_TAIL_LINES chunks, so an un-owned row pins its whole 64 KiB chunk. + newlyCompletedLines.push(ownRetainedString(trimTerminalLineRight(piece))) } const partialResult = applyTerminalLineControls(segments.partialSegment) const nextPartialLine = trimTerminalLineRight(partialResult.text) - const retainedPartialLine = nextPartialLine.slice(-MAX_TAIL_PARTIAL_CHARS) + const retainedPartialLine = ownRetainedString(nextPartialLine.slice(-MAX_TAIL_PARTIAL_CHARS)) const newCompleteLines = segments.completeLineCount const omittedNewCompleteLines = newCompleteLines - pieces.length diff --git a/src/main/runtime/terminal-tail-redraw-buffer.ts b/src/main/runtime/terminal-tail-redraw-buffer.ts index 7ebb06753dd..d8971f7ea8b 100644 --- a/src/main/runtime/terminal-tail-redraw-buffer.ts +++ b/src/main/runtime/terminal-tail-redraw-buffer.ts @@ -2,6 +2,7 @@ import { hasCanonicalNumericCsiParams, parseAnsiControlSequence } from './terminal-ansi-normalization' +import { ownRetainedString } from '../../shared/own-retained-string' import { clampTerminalPreviewCursor, trimTerminalLineRight } from './terminal-tail-line-controls' import { MAX_TAIL_CHARS, MAX_TAIL_LINES, MAX_TAIL_PARTIAL_CHARS } from './terminal-tail-limits' @@ -241,7 +242,11 @@ function finalizeRetainedTerminalRows( return { lines, - partialLine, + // Why only the partial: redraw rows are built character by character and never sliced from + // the chunk, but the partial is re-sliced from its own row on every chunk, so it alone can + // accumulate a backing string across frames. Owning the rows too costs 20-36% on TUI floods + // for no measured retention. + partialLine: ownRetainedString(partialLine), redrawCursor, truncated, newCompleteLines, diff --git a/src/main/runtime/terminal-tail-row-retention.test.ts b/src/main/runtime/terminal-tail-row-retention.test.ts new file mode 100644 index 00000000000..0e4effbf968 --- /dev/null +++ b/src/main/runtime/terminal-tail-row-retention.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it, vi } from 'vitest' +import * as ownership from '../../shared/own-retained-string' +import { appendNormalizedToTailBuffer } from './terminal-tail-buffer' + +const SPINNER = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴'] + +/** A Claude Code / Codex spinner chunk: many CR-redraw frames, then one completed line. */ +function spinnerChunk(index: number, chunkChars: number): string { + const parts: string[] = [] + let length = 0 + let frame = 0 + while (length < chunkChars - 64) { + const piece = `\r${SPINNER[frame % SPINNER.length]} Thinking... (${frame}s) esc to interrupt` + parts.push(piece) + length += piece.length + frame += 1 + } + parts.push(`\rstep ${index} done\n`) + return parts.join('') +} + +function collectHeap(): number { + const gc = (globalThis as { gc?: () => void }).gc + if (!gc) { + throw new Error('global.gc unavailable - config/vitest.config.ts must pass --expose-gc') + } + void /reset/.test('reset') + gc() + gc() + return process.memoryUsage().heapUsed +} + +describe('retained terminal tail row storage', () => { + it('does not pin one chunk per retained row across a spinner workload', () => { + let lines: string[] = [] + let partialLine = '' + for (let index = 0; index < 4; index += 1) { + const warm = appendNormalizedToTailBuffer(lines, partialLine, spinnerChunk(index, 4096)) + lines = warm.lines + partialLine = warm.partialLine + } + + lines = [] + partialLine = '' + const chunkChars = 64 * 1024 + const chunkCount = 200 + const before = collectHeap() + // Why build each chunk inside the loop: a pre-built array would pin every chunk itself. + for (let index = 0; index < chunkCount; index += 1) { + const next = appendNormalizedToTailBuffer(lines, partialLine, spinnerChunk(index, chunkChars)) + lines = next.lines + partialLine = next.partialLine + } + const retained = collectHeap() - before + + expect(lines).toHaveLength(chunkCount) + const tailChars = lines.reduce((sum, line) => sum + line.length, 0) + partialLine.length + expect(tailChars).toBeLessThan(8 * 1024) + // Un-owned, each of the 200 rows pins its own 64 Ki chunk: about 25 MB. + expect(retained).toBeLessThan(4 * 1024 * 1024) + expect(lines.at(-1)).toBe(`step ${chunkCount - 1} done`) + }) + + it('routes every retained row and partial line through ownRetainedString', () => { + const own = vi.spyOn(ownership, 'ownRetainedString') + try { + const chunk = `${'x'.repeat(16 * 1024)}\nsecond line\ntrailing partial` + const result = appendNormalizedToTailBuffer([], '', chunk) + + expect(result.lines).toEqual(['x'.repeat(16 * 1024), 'second line']) + expect(result.partialLine).toBe('trailing partial') + expect(own.mock.calls.map(([value]) => value)).toEqual([ + 'x'.repeat(16 * 1024), + 'second line', + 'trailing partial' + ]) + } finally { + own.mockRestore() + } + }) + + it('owns the redraw partial line without re-owning carried rows', () => { + const own = vi.spyOn(ownership, 'ownRetainedString') + try { + const seeded = appendNormalizedToTailBuffer([], '', 'row one\nrow two\nrow three\n') + own.mockClear() + // \x1b[2A drives the multiline redraw builder rather than the plain path. + const redrawn = appendNormalizedToTailBuffer( + seeded.lines, + seeded.partialLine, + '\x1b[2A\x1b[Krewritten two' + ) + + expect(redrawn.lines).toEqual(['row one']) + expect(redrawn.partialLine).toBe('rewritten two') + // One call for the partial only: redraw rows are built character by character. + expect(own).toHaveBeenCalledTimes(1) + expect(own).toHaveBeenLastCalledWith('rewritten two') + } finally { + own.mockRestore() + } + }) + + it('keeps completed lines and the tail transcript sharing the same owned rows', () => { + const result = appendNormalizedToTailBuffer([], '', 'alpha line\nbeta line\n') + expect(result.newlyCompletedLines).toEqual(['alpha line', 'beta line']) + // The transcript keeps these exact strings, so owning them covers it transitively. + result.newlyCompletedLines.forEach((line, index) => { + expect(result.lines[index]).toBe(line) + }) + }) +}) diff --git a/src/main/runtime/terminal-vertical-control-scan.test.ts b/src/main/runtime/terminal-vertical-control-scan.test.ts new file mode 100644 index 00000000000..b672e28c921 --- /dev/null +++ b/src/main/runtime/terminal-vertical-control-scan.test.ts @@ -0,0 +1,85 @@ +import { describe, expect, it, vi } from 'vitest' +import { + containsTerminalVerticalLineControl, + normalizeTerminalChunk +} from './terminal-ansi-normalization' +import { appendNormalizedToTailBuffer } from './terminal-tail-buffer' + +describe('terminal vertical-control scanning', () => { + it.each([ + ['plain', 'log output '.repeat(8192), false], + ['nonvertical CSI', `\x1b[31m${'log output '.repeat(8192)}\x1b[0m`, false], + ['vertical CSI', `${'漢字😀 output '.repeat(8192)}\x1b[2A`, true] + ] as const)('bounds code-unit inspections on %s output', (_name, input, expected) => { + const charCodeAt = vi.spyOn(String.prototype, 'charCodeAt') + let actual: boolean + let inspections: number + try { + actual = containsTerminalVerticalLineControl(input) + inspections = charCodeAt.mock.calls.length + } finally { + charCodeAt.mockRestore() + } + + expect(actual).toBe(expected) + expect(inspections).toBeLessThan(16) + }) + + it.each([ + ['\x1b[A', true], + ['\x1b[0A', true], + ['\x1b[;A', true], + ['\x1b[12;34A', true], + ['\x1b[?1A', false], + ['\x1b[1:2A', false], + ['\x1b[1 A', false], + ['\x1b[1\nA', false], + ['\x1b[1B', false], + ['\x9b1A', false], + ['\x1b', false], + ['\x1b[123', false] + ] as const)('preserves numeric CSI A recognition for %j', (input, expected) => { + expect(containsTerminalVerticalLineControl(input)).toBe(expected) + }) + + it.each([ + ['OSC BEL', '\x1b]2;title', '\x07'], + ['OSC ST', '\x1b]2;title', '\x1b\\'], + ['DCS', '\x1bPpayload', '\x1b\\'], + ['SOS', '\x1bXpayload', '\x1b\\'], + ['PM', '\x1b^payload', '\x1b\\'], + ['APC', '\x1b_payload', '\x1b\\'] + ])('skips embedded CSI and stops at an incomplete %s', (_name, prefix, terminator) => { + const incomplete = `${prefix}\x1b[2A` + expect(containsTerminalVerticalLineControl(incomplete)).toBe(false) + expect(containsTerminalVerticalLineControl(`${incomplete}${terminator}ordinary`)).toBe(false) + expect(containsTerminalVerticalLineControl(`${incomplete}${terminator}\x1b[3A`)).toBe(true) + }) + + it.each([ + // An ESC inside CSI parameter bytes is consumed by that control, not treated as a new introducer. + ['\x1b[\x1b[A', false], + ['\x1b[\x1b[2A\x1b[1A', true], + ['\x1b[31m\x1b[1A', true], + ['\x1b]0;t\x07\x1b[1A', true], + ['\x1b[1A\x1b', true], + ['ordinary\x1b', false], + ['ordinary\x1b[0m more\x1b[1;A', true] + ] as const)('resumes scanning after a parsed control for %j', (input, expected) => { + expect(containsTerminalVerticalLineControl(input)).toBe(expected) + }) + + it('preserves tail rows when ordinary output is followed by a cursor-up redraw', () => { + const first = appendNormalizedToTailBuffer([], '', 'first\nold\n') + const normalized = normalizeTerminalChunk('\x1b[1A\x1b[2K\x1b[32mnew\x1b[0m\n') + const next = appendNormalizedToTailBuffer( + first.lines, + first.partialLine, + normalized.text, + first.redrawCursor + ) + + expect(next.lines).toEqual(['first', 'new']) + expect(next.partialLine).toBe('') + }) +}) diff --git a/src/main/runtime/wait-blocked-keyword-carry-retention.test.ts b/src/main/runtime/wait-blocked-keyword-carry-retention.test.ts new file mode 100644 index 00000000000..3d4c0542750 --- /dev/null +++ b/src/main/runtime/wait-blocked-keyword-carry-retention.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, it, vi } from 'vitest' +import * as ownership from '../../shared/own-retained-string' +import { OrcaRuntimeWithScheduleWaitBlockedCheck } from './orca-runtime-schedule-wait-blocked-check' +import { WAIT_BLOCKED_KEYWORD_CARRY_CHARS } from './orca-runtime-postlude' +import type { createWaitBlockedCheckState } from './wait-blocked-check-state' + +type ScheduleHost = { + waitBlockedCheckStateByPtyId: Map> + runWaitBlockedCheck: () => void + scheduleWaitBlockedCheck: (ptyId: string, appendedText: string, at: number) => void +} + +function createScheduleHost(): ScheduleHost { + const prototype = OrcaRuntimeWithScheduleWaitBlockedCheck.prototype as unknown as ScheduleHost + return { + waitBlockedCheckStateByPtyId: new Map(), + runWaitBlockedCheck: () => {}, + scheduleWaitBlockedCheck: prototype.scheduleWaitBlockedCheck + } +} + +describe('wait-blocked keyword carry storage', () => { + it('owns the carry so it stops pinning the lowercased chunk window', () => { + const own = vi.spyOn(ownership, 'ownRetainedString') + try { + const host = createScheduleHost() + const chunk = `${'Building Project '.repeat(4096)}tail-marker-text` + host.scheduleWaitBlockedCheck('pty-1', chunk, 0) + + const carry = host.waitBlockedCheckStateByPtyId.get('pty-1')?.keywordCarry + expect(carry).toBe(chunk.toLowerCase().slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS)) + expect(carry).toHaveLength(WAIT_BLOCKED_KEYWORD_CARRY_CHARS) + expect(own).toHaveBeenCalledTimes(1) + expect(own).toHaveBeenLastCalledWith(carry) + } finally { + own.mockRestore() + } + }) + + it('keeps the carry joined to the next chunk so split keywords still match', () => { + const host = createScheduleHost() + host.scheduleWaitBlockedCheck('pty-2', `${'x'.repeat(8 * 1024)}press`, 0) + const carry = host.waitBlockedCheckStateByPtyId.get('pty-2')?.keywordCarry + expect(carry?.endsWith('press')).toBe(true) + host.scheduleWaitBlockedCheck('pty-2', ' ENTER to continue', 1) + expect(host.waitBlockedCheckStateByPtyId.get('pty-2')?.keywordCarry).toBe( + `${carry} enter to continue`.slice(-WAIT_BLOCKED_KEYWORD_CARRY_CHARS) + ) + }) +}) diff --git a/src/main/startup/os-opened-markdown-files.test.ts b/src/main/startup/os-opened-markdown-files.test.ts index f174e10d834..c980b6258d2 100644 --- a/src/main/startup/os-opened-markdown-files.test.ts +++ b/src/main/startup/os-opened-markdown-files.test.ts @@ -304,3 +304,84 @@ describe('resolveOpenedMarkdownDocuments', () => { expect(authorizeExternalPath).not.toHaveBeenCalled() }) }) + +describe('OsOpenedMarkdownFileState delivery cap', () => { + it('stops merging an OS file batch at the pending delivery cap', () => { + const state = new OsOpenedMarkdownFileState() + const includes = vi.spyOn(Array.prototype, 'includes') + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + let probes: number + try { + state.captureFilePaths( + Array.from({ length: 10000 }, (_, index) => resolve(`/notes/${index}.md`)) + ) + probes = includes.mock.calls.length + } finally { + includes.mockRestore() + warn.mockRestore() + } + expect(probes).toBeLessThan(100) + expect(state.consume()).toEqual( + Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) => + resolve(`/notes/${index}.md`) + ) + ) + }) + + // Why: the cap drops files the user explicitly asked to open. Pin which end is + // dropped (the tail, in shell order) and that the loss is reported, not silent. + it('keeps the first paths in shell order and reports the dropped tail', () => { + const state = new OsOpenedMarkdownFileState() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const total = MAX_PENDING_OS_OPENED_MARKDOWN_FILES + 8 + const paths = Array.from({ length: total }, (_, index) => + resolve(`/notes/${String(index).padStart(3, '0')}.md`) + ) + try { + expect(state.captureFilePaths(paths)).toBe(true) + expect(warn).toHaveBeenCalledWith( + expect.stringContaining(`Dropped 8 of ${total} OS-opened markdown files`) + ) + } finally { + warn.mockRestore() + } + expect(state.consume()).toEqual(paths.slice(0, MAX_PENDING_OS_OPENED_MARKDOWN_FILES)) + }) + + it('stays silent for a batch that fits under the cap', () => { + const state = new OsOpenedMarkdownFileState() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + state.captureFilePaths( + Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) => + resolve(`/notes/${index}.md`) + ) + ) + expect(warn).not.toHaveBeenCalled() + } finally { + warn.mockRestore() + } + }) + + it('reports a drop when an already-full queue rejects a later batch', () => { + const state = new OsOpenedMarkdownFileState() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + state.captureFilePaths( + Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) => + resolve(`/first/${index}.md`) + ) + ) + expect(warn).not.toHaveBeenCalled() + state.captureFilePaths([resolve('/second/a.md'), resolve('/second/b.md')]) + expect(warn).toHaveBeenCalledWith(expect.stringContaining('Dropped 2 of 2')) + } finally { + warn.mockRestore() + } + expect(state.consume()).toEqual( + Array.from({ length: MAX_PENDING_OS_OPENED_MARKDOWN_FILES }, (_, index) => + resolve(`/first/${index}.md`) + ) + ) + }) +}) diff --git a/src/main/startup/os-opened-markdown-files.ts b/src/main/startup/os-opened-markdown-files.ts index dae27fb7a78..144f59fe578 100644 --- a/src/main/startup/os-opened-markdown-files.ts +++ b/src/main/startup/os-opened-markdown-files.ts @@ -105,11 +105,23 @@ export class OsOpenedMarkdownFileState { return false } const merged = [...this.pending] - for (const filePath of filePaths) { + let index = 0 + for (; index < filePaths.length; index++) { + if (merged.length >= MAX_PENDING_OS_OPENED_MARKDOWN_FILES) { + break + } + const filePath = filePaths[index]! if (!merged.includes(filePath)) { merged.push(filePath) } } + if (index < filePaths.length) { + // Why logged: the cap drops the tail of an oversized selection, and a file the + // user explicitly asked to open must not vanish without leaving a trace. + console.warn( + `[os-open] Dropped ${filePaths.length - index} of ${filePaths.length} OS-opened markdown files; the pending queue is capped at ${MAX_PENDING_OS_OPENED_MARKDOWN_FILES}.` + ) + } this.pending = merged.slice(0, MAX_PENDING_OS_OPENED_MARKDOWN_FILES) publish?.() return true diff --git a/src/main/windows/windows-pty-job.win32.test.ts b/src/main/windows/windows-pty-job.win32.test.ts index ee5b915f916..0ffa55e988e 100644 --- a/src/main/windows/windows-pty-job.win32.test.ts +++ b/src/main/windows/windows-pty-job.win32.test.ts @@ -34,6 +34,14 @@ function isAlive(pid: number): boolean { } } +// Teardown is asynchronous; poll instead of guessing how long the job takes. +async function waitUntilDead(pid: number, timeoutMs = 30_000): Promise { + const deadline = Date.now() + timeoutMs + while (isAlive(pid) && Date.now() < deadline) { + await sleep(50) + } +} + describeOnWindows('ConPTY job ownership', () => { const spawned: IPty[] = [] @@ -108,7 +116,8 @@ describeOnWindows('ConPTY job ownership', () => { expect(isAlive(grandchildPid)).toBe(true) expect(terminatePtyJob(proc)).toBe('terminated') - await sleep(1_500) + await waitUntilDead(proc.pid) + await waitUntilDead(grandchildPid) expect(isAlive(proc.pid)).toBe(false) expect(isAlive(grandchildPid)).toBe(false) diff --git a/src/renderer/src/components/sidebar/SidebarHeader.test.tsx b/src/renderer/src/components/sidebar/SidebarHeader.test.tsx index 3371cddec58..40e62c6ce46 100644 --- a/src/renderer/src/components/sidebar/SidebarHeader.test.tsx +++ b/src/renderer/src/components/sidebar/SidebarHeader.test.tsx @@ -93,31 +93,16 @@ vi.mock('@/components/ui/popover', () => ({ let container: HTMLDivElement let root: Root -function createButton(): HTMLButtonElement { - const button = container.querySelector('[aria-label="Create"]') +function headerButton(label: string): HTMLButtonElement { + const button = container.querySelector(`[aria-label="${label}"]`) if (!button) { - throw new Error('Create button not rendered') + throw new Error(`Header button not rendered: ${label}`) } return button } -async function openCreateMenu(): Promise { - await act(async () => { - // Why not click(): the Radix trigger opens on pointerdown, which happy-dom does not synthesize. - createButton().dispatchEvent( - new window.PointerEvent('pointerdown', { bubbles: true, button: 0 }) - ) - }) -} - -function createMenuItem(label: string): HTMLElement { - const item = [...document.querySelectorAll('[role="menuitem"]')].find((candidate) => - candidate.textContent?.includes(label) - ) - if (!item) { - throw new Error(`Create menu item not rendered: ${label}`) - } - return item +function createButton(): HTMLButtonElement { + return headerButton('New workspace') } beforeEach(() => { @@ -152,10 +137,9 @@ describe('SidebarHeader', () => { }) expect(createButton().disabled).toBe(false) - await openCreateMenu() await act(async () => { - createMenuItem('New workspace').click() + createButton().click() }) expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1) @@ -167,42 +151,53 @@ describe('SidebarHeader', () => { root.render() }) - await openCreateMenu() await act(async () => { - createMenuItem('New workspace').click() + createButton().click() }) expect(createButton().disabled).toBe(false) expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1) }) - it('offers Add project beside New workspace under the create button', async () => { + it('reaches Add project and New workspace in one click each, with no menu', async () => { act(() => { root.render() }) - await openCreateMenu() - - expect(createMenuItem('New workspace')).toBeTruthy() - expect(createMenuItem('Add project')).toBeTruthy() + expect(headerButton('New workspace')).toBeTruthy() + expect(headerButton('Add project')).toBeTruthy() + expect(container.querySelector('[data-slot="dropdown-menu-trigger"]')).toBeNull() await act(async () => { - createMenuItem('Add project').click() + headerButton('Add project').click() }) expect(mockState.openModal).toHaveBeenCalledWith('add-repo') expect(mocks.openWorkspaceCreationComposerWithTourHandoff).not.toHaveBeenCalled() }) - it('omits the shortcut hint when workspace creation is unassigned', async () => { - mocks.shortcutLabel.current = null + it('keeps the create button rightmost so the frequent action stays where it was', () => { act(() => { root.render() }) - await openCreateMenu() + const labels = [...container.querySelectorAll('[aria-label]')] + .map((node) => node.getAttribute('aria-label')) + .filter((label): label is string => label === 'Add project' || label === 'New workspace') + expect(labels).toEqual(['Add project', 'New workspace']) + }) - expect(document.querySelector('[data-slot="dropdown-menu-shortcut"]')).toBeNull() + it('advertises the workspace shortcut on the create tooltip, and omits it when unassigned', () => { + act(() => { + root.render() + }) + expect(container.textContent).toContain('⌘N') + + mocks.shortcutLabel.current = null + act(() => { + root.render() + }) + expect(container.textContent).not.toContain('⌘N') }) it('opens agent activity from the bell button', () => { @@ -275,16 +270,16 @@ describe('SidebarHeader', () => { ) }) - it('keeps the workspace filter alongside the active bell without Add Project', () => { + it('drops both project actions in the agents view, which lists activity, not projects', () => { mockState.sidebarBody = 'agents' act(() => { root.render() }) expect(container.querySelector('[aria-label="Turn off activity view"]')).toBeTruthy() - expect(container.querySelector('[aria-label="Create"]')).toBeTruthy() + expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy() expect(container.querySelector('[aria-label="Workspace options"]')).toBeNull() - expect(container.querySelector('[aria-label="Add Project"]')).toBeNull() + expect(container.querySelector('[aria-label="Add project"]')).toBeNull() }) it('keeps the activity bell and actions on one row at the default sidebar width', () => { @@ -297,8 +292,8 @@ describe('SidebarHeader', () => { expect(headerClasses.has('flex-wrap')).toBe(false) expect(headerClasses.has('h-8')).toBe(true) expect(container.querySelector('[aria-label="View activity"]')).toBeTruthy() - expect(container.querySelector('[aria-label="Add Project"]')).toBeNull() - expect(container.querySelector('[aria-label="Create"]')).toBeTruthy() + expect(container.querySelector('[aria-label="Add project"]')).toBeTruthy() + expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy() }) it('keeps the same actions on one row at compact width', async () => { @@ -307,15 +302,14 @@ describe('SidebarHeader', () => { root.render() }) - expect(container.querySelector('[aria-label="Add Project"]')).toBeNull() + expect(container.querySelector('[aria-label="Add project"]')).toBeTruthy() expect(container.querySelector('[aria-label="View activity"]')).toBeTruthy() - expect(container.querySelector('[aria-label="Create"]')).toBeTruthy() + expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy() expect(container.querySelector('[aria-label="Workspace options"]')).toBeTruthy() expect(container.querySelector('[aria-label="More workspace actions"]')).toBeNull() - await openCreateMenu() await act(async () => { - createMenuItem('New workspace').click() + createButton().click() }) expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1) }) @@ -340,8 +334,8 @@ describe('SidebarHeader', () => { expect(container.querySelector('[aria-label="Open full Agents view"]')).toBeNull() }) - // Why: the compact overflow existed only to carry Add Project, which now lives - // under the create button, so both widths render one identical header. + // Why: the compact overflow existed only to carry Add Project, which now sits + // beside the create button, so both widths render one identical header. it('renders the same actions on both sides of the old wide-layout breakpoint', () => { for (const width of [234, 235]) { mockState.sidebarWidth = width @@ -349,8 +343,8 @@ describe('SidebarHeader', () => { root.render() }) expect(container.querySelector('[aria-label="More workspace actions"]')).toBeNull() - expect(container.querySelector('[aria-label="Add Project"]')).toBeNull() - expect(container.querySelector('[aria-label="Create"]')).toBeTruthy() + expect(container.querySelector('[aria-label="Add project"]')).toBeTruthy() + expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy() expect(container.querySelector('[aria-label="Workspace options"]')).toBeTruthy() } }) diff --git a/src/renderer/src/components/sidebar/SidebarHeader.tsx b/src/renderer/src/components/sidebar/SidebarHeader.tsx index 413558ff9c7..b9642b94a04 100644 --- a/src/renderer/src/components/sidebar/SidebarHeader.tsx +++ b/src/renderer/src/components/sidebar/SidebarHeader.tsx @@ -49,7 +49,9 @@ const SidebarHeader = React.memo(function SidebarHeader({
{sidebarTitle} @@ -125,7 +127,7 @@ const SidebarHeader = React.memo(function SidebarHeader({ ) : null}
diff --git a/src/renderer/src/components/sidebar/sidebar-header-actions.tsx b/src/renderer/src/components/sidebar/sidebar-header-actions.tsx index d25f98700d1..994c26eb88d 100644 --- a/src/renderer/src/components/sidebar/sidebar-header-actions.tsx +++ b/src/renderer/src/components/sidebar/sidebar-header-actions.tsx @@ -1,131 +1,104 @@ -import React, { useCallback, useEffect, useRef, useState } from 'react' -import { FolderPlus, GitBranchPlus, Plus } from 'lucide-react' +import React, { useCallback } from 'react' +import { FolderPlus, Plus } from 'lucide-react' import { useAppStore } from '@/store' import { Button } from '@/components/ui/button' -import { - DropdownMenu, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuShortcut, - DropdownMenuTrigger -} from '@/components/ui/dropdown-menu' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { formatOptionalPrimaryShortcutLabel } from '@/hooks/useShortcutLabel' import { translate } from '@/i18n/i18n' import { openWorkspaceCreationComposerWithTourHandoff } from '../contextual-tours/workspace-creation-tour-handoff' import SidebarWorkspaceOptionsMenu from './SidebarWorkspaceOptionsMenu' -function SidebarCreateMenu({ +function AddProjectButton({ preserveWorkspaceBoardOpen }: { preserveWorkspaceBoardOpen: boolean }): React.JSX.Element { const openModal = useAppStore((s) => s.openModal) - const keybindings = useAppStore((s) => s.keybindings) - const [open, setOpen] = useState(false) - const menuContentRef = useRef(null) - // Why primary: workspace.create binds both Mod+N and Mod+Shift+N, and listing - // every alias in a two-row menu reads as noise rather than help. - const newWorktreeShortcutLabel = formatOptionalPrimaryShortcutLabel( - 'workspace.create', - keybindings + const label = translate('auto.components.sidebar.SidebarHeader.addProject', 'Add project') + + return ( + + + + + + {label} + + ) - const boardAttr = preserveWorkspaceBoardOpen ? '' : undefined +} - // Why query, not a ref on the item: Radix wraps each item in a roving-focus Slot, - // and a second ref on that child conflicts with the one the Slot already owns. - // Why at all: Radix highlights the first item only when opened by keyboard, so a - // mouse click would otherwise leave Enter with nothing to activate. - useEffect(() => { - if (!open) { - return - } - const frame = requestAnimationFrame(() => - menuContentRef.current?.querySelector('[role="menuitem"]')?.focus() - ) - return () => cancelAnimationFrame(frame) - }, [open]) +function NewWorkspaceButton({ + preserveWorkspaceBoardOpen +}: { + preserveWorkspaceBoardOpen: boolean +}): React.JSX.Element { + const keybindings = useAppStore((s) => s.keybindings) + // Why primary: workspace.create binds both Mod+N and Mod+Shift+N, and listing + // every alias in a one-line tooltip reads as noise rather than help. + const shortcutLabel = formatOptionalPrimaryShortcutLabel('workspace.create', keybindings) + const label = translate('auto.components.sidebar.SidebarHeader.92154beb7e', 'New workspace') - // Why: the tour highlights this trigger, so the handoff has to fire from the - // menu item rather than the button that now only opens the menu. + // Why the tour handoff here: the tour highlights this button, and it is now + // the control that performs the action rather than one that opens a menu. const handleCreateWorkspace = useCallback(() => { - // Why: opening after Radix tears down the menu prevents its focus restoration - // from treating the new dialog as an outside interaction. - window.setTimeout(openWorkspaceCreationComposerWithTourHandoff, 0) + openWorkspaceCreationComposerWithTourHandoff() }, []) return ( - - - - - - - - - {translate('auto.components.sidebar.SidebarHeader.createMenu', 'Create')} - - - - + + + + + {label} + {shortcutLabel ? {shortcutLabel} : null} + + ) } export function SidebarHeaderActions({ onWorkspaceBoardMenuOpenChange, - hideWorkspaceOptions = false + agentsViewActive = false }: { onWorkspaceBoardMenuOpenChange: (open: boolean) => void - hideWorkspaceOptions?: boolean + agentsViewActive?: boolean }): React.JSX.Element { return ( -
- {hideWorkspaceOptions ? null : ( - +
+ {/* Why both hidden in the agents view: it lists activity, not projects. */} + {agentsViewActive ? null : ( + <> + + + )} - +
) } diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/repo-header-project-actions.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/repo-header-project-actions.tsx index 379a8d8e614..ba4751b597c 100644 --- a/src/renderer/src/components/sidebar/worktree-list/rows/repo-header-project-actions.tsx +++ b/src/renderer/src/components/sidebar/worktree-list/rows/repo-header-project-actions.tsx @@ -4,7 +4,7 @@ import { Ellipsis, Eye, FolderInput, - FolderPlus, + FolderTree, Plus, Shapes, SlidersHorizontal, @@ -135,7 +135,8 @@ export function RepoHeaderProjectActionsMenu({ ) : null} actions.onCreateGroupFromRepo(repo)}> - + {/* Not FolderPlus: that now means "Add project" in the sidebar header above. */} + {translate('auto.components.sidebar.WorktreeList.cbfd565f83', 'New group from project')} {projectGroups.length > 0 ? ( diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 4f283dc4f47..ad5d2e296b0 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -5298,7 +5298,6 @@ "5c9c7c16aa": "Add a project to create workspaces", "a30e34eb5c": "Close workspace board", "views": "Sidebar view", - "createMenu": "Create", "addProject": "Add project" }, "SidebarNav": { diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 2ee1716ea20..56695c14755 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -4412,7 +4412,6 @@ "a30e34eb5c": "Cerrar tablero del espacio de trabajo", "spaces": "Espacios", "views": "Vista de la barra lateral", - "createMenu": "Crear", "addProject": "Agregar proyecto" }, "SidebarNav": { diff --git a/src/renderer/src/i18n/locales/fr.json b/src/renderer/src/i18n/locales/fr.json index 6113d606c90..fe8eaf19756 100644 --- a/src/renderer/src/i18n/locales/fr.json +++ b/src/renderer/src/i18n/locales/fr.json @@ -5044,7 +5044,6 @@ "49f62c5665": "Tableau des espaces de travail", "5c9c7c16aa": "Ajoutez un projet pour créer des espaces de travail", "a30e34eb5c": "Fermer le tableau des espaces de travail", - "createMenu": "Créer", "addProject": "Ajouter un projet" }, "SidebarNav": { diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index f0c1666d911..2224e322951 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -4393,7 +4393,6 @@ "a30e34eb5c": "ワークスペースボードを閉じる", "spaces": "スペース", "views": "サイドバービュー", - "createMenu": "作成", "addProject": "プロジェクトを追加" }, "SidebarNav": { diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 42983088062..25359a26785 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -4398,7 +4398,6 @@ "a30e34eb5c": "워크스페이스 보드 닫기", "spaces": "스페이스", "views": "사이드바 보기", - "createMenu": "생성", "addProject": "프로젝트 추가" }, "SidebarNav": { diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 09ca1689ea3..f00d3ea0d90 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -4441,7 +4441,6 @@ "a30e34eb5c": "关闭工作区板", "spaces": "空间", "views": "侧边栏视图", - "createMenu": "创建", "addProject": "添加项目" }, "SidebarNav": { @@ -14287,7 +14286,6 @@ "filtersSection": "筛选", "viewSection": "视图" }, - "ActivityScopeFilterControls": { "resetScope": "显示所有主机和项目" }, diff --git a/src/renderer/src/lib/agent-hibernation-coordinator-test-fixture.ts b/src/renderer/src/lib/agent-hibernation-coordinator-test-fixture.ts new file mode 100644 index 00000000000..837a8f3a59d --- /dev/null +++ b/src/renderer/src/lib/agent-hibernation-coordinator-test-fixture.ts @@ -0,0 +1,197 @@ +import { vi, type Mock } from 'vitest' +import type { AgentStatusEntry } from '../../../shared/agent-status-types' +import type { TerminalLayoutSnapshot, TerminalTab } from '../../../shared/terminal-tab-types' +import { useAppStore } from '@/store' +import type { AppState } from '@/store/types' +import { DEFAULT_AGENT_HIBERNATION_IDLE_MS } from './agent-hibernation-planner' +import { resetAgentHibernationCoordinatorForTests } from './agent-hibernation-coordinator' +import { hydrateDrivers } from './pane-manager/mobile-driver-state' +import { resetForegroundTerminalTabIdsForTests } from './foreground-terminal-tabs' +import { resetAgentHibernationOutputActivityForTests } from './agent-hibernation-output-activity' +import { + observeHibernationPtyBindings, + resetHibernationPaneAgeForTests +} from './agent-hibernation-pane-age' +import { + createCompatibleRuntimeStatusResponseIfNeeded, + type RuntimeEnvironmentCallRequest +} from '../runtime/runtime-compatibility-test-fixture' +import { clearRuntimeCompatibilityCacheForTests } from '../runtime/runtime-rpc-client' + +export const NOW = 10_000_000 +export const LEAF = '11111111-1111-4111-8111-111111111111' + +export type RuntimeEnvironmentCallStub = Mock<(args: RuntimeEnvironmentCallRequest) => unknown> + +export const mockRuntimeEnvironmentCall: RuntimeEnvironmentCallStub = vi.fn() + +vi.stubGlobal('window', { + api: { + runtimeEnvironments: { + call: mockRuntimeEnvironmentCall + } + } +}) + +export function tab(): TerminalTab { + return { + id: 'tab-1', + ptyId: null, + worktreeId: 'wt-bg', + title: 'Agent', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 1 + } +} + +export function layout(): TerminalLayoutSnapshot { + return { + root: { type: 'leaf', leafId: LEAF }, + activeLeafId: LEAF, + expandedLeafId: null, + ptyIdsByLeafId: { [LEAF]: 'pty-1' } + } +} + +export function entry(): AgentStatusEntry { + return { + state: 'done', + prompt: 'ship it', + updatedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1, + stateStartedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1, + paneKey: `tab-1:${LEAF}`, + tabId: 'tab-1', + worktreeId: 'wt-bg', + agentType: 'claude', + providerSession: { key: 'session_id', id: 'session-1' }, + stateHistory: [] + } +} + +export type HibernationShutdownStub = Mock + +export function installEligibleState( + shutdownCompletedAgentPaneForHibernation: HibernationShutdownStub = vi.fn(), + overrides: Partial = {} +): HibernationShutdownStub { + const e = entry() + const runtimeOwnerEnvironmentId = overrides.settings?.activeRuntimeEnvironmentId ?? undefined + useAppStore.setState({ + settings: { + experimentalAgentHibernation: true, + agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS + } as never, + activeWorktreeId: 'wt-active', + repos: [], + worktreesByRepo: { + 'fixture-repo': [ + { + id: 'wt-bg', + repoId: 'fixture-repo', + hostId: 'local', + runtimeOwnerEnvironmentId + } + ] + } as never, + detectedWorktreesByRepo: {}, + tabsByWorktree: { 'wt-bg': [tab()] }, + terminalLayoutsByTabId: { 'tab-1': layout() }, + ptyIdsByTabId: { 'tab-1': ['pty-1'] }, + agentStatusByPaneKey: { [e.paneKey]: e }, + sleepingAgentSessionsByPaneKey: {}, + lastTerminalInputAtByPaneKey: {}, + shutdownCompletedAgentPaneForHibernation: shutdownCompletedAgentPaneForHibernation as never, + shutdownWorktreeTerminals: vi.fn() as never, + ...overrides + }) + // Why: a pane idle long enough to hibernate has necessarily been observed by earlier + // coordinator passes, so its PTY binding is old. Seed that here — otherwise the + // binding-age floor (which exists to stop a wake or app restart sleeping the whole + // backlog immediately) would defer every candidate on its first observed tick. + const state = useAppStore.getState() + observeHibernationPtyBindings({ + tabsByWorktree: state.tabsByWorktree, + terminalLayoutsByTabId: state.terminalLayoutsByTabId, + now: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 60_000, + idleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS + }) + return shutdownCompletedAgentPaneForHibernation +} + +export function runtimeListResult(ptyIds: string[], truncated = false) { + return { + terminals: ptyIds.map((ptyId) => ({ + handle: `handle-${ptyId}`, + ptyId, + worktreeId: 'wt-bg', + worktreePath: '/tmp/wt-bg', + branch: 'feature', + tabId: `pty:${ptyId}`, + leafId: `pty:${ptyId}`, + title: 'Agent', + connected: true, + writable: true, + lastOutputAt: null, + preview: '' + })), + totalCount: ptyIds.length, + truncated + } +} + +export function installRuntimeListResponses( + ...responses: (ReturnType | Error)[] +): void { + const queue = [...responses] + mockRuntimeEnvironmentCall.mockImplementation((args: RuntimeEnvironmentCallRequest) => { + const compatible = createCompatibleRuntimeStatusResponseIfNeeded(args) + if (compatible) { + return Promise.resolve(compatible) + } + if (args.method === 'terminal.list') { + const response = queue.shift() ?? runtimeListResult(['pty-1']) + if (response instanceof Error) { + return Promise.reject(response) + } + return Promise.resolve({ + id: 'terminal-list', + ok: true, + result: response, + _meta: { runtimeId: 'runtime-1' } + }) + } + return Promise.resolve({ + id: 'default', + ok: true, + result: {}, + _meta: { runtimeId: 'runtime-1' } + }) + }) +} + +export function deferred(): { + promise: Promise + resolve: (value: T) => void + reject: (error: Error) => void +} { + let resolve!: (value: T) => void + let reject!: (error: Error) => void + const promise = new Promise((res, rej) => { + resolve = res + reject = rej + }) + return { promise, resolve, reject } +} + +export function resetAgentHibernationCoordinatorFixture(): void { + resetAgentHibernationCoordinatorForTests() + clearRuntimeCompatibilityCacheForTests() + resetForegroundTerminalTabIdsForTests() + resetAgentHibernationOutputActivityForTests() + resetHibernationPaneAgeForTests() + hydrateDrivers([]) + mockRuntimeEnvironmentCall.mockReset() + vi.useRealTimers() +} diff --git a/src/renderer/src/lib/agent-hibernation-coordinator.test.ts b/src/renderer/src/lib/agent-hibernation-coordinator.test.ts index 75d253980a6..e73beb50d10 100644 --- a/src/renderer/src/lib/agent-hibernation-coordinator.test.ts +++ b/src/renderer/src/lib/agent-hibernation-coordinator.test.ts @@ -1,202 +1,33 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' -import type { AgentStatusEntry } from '../../../shared/agent-status-types' -import type { TerminalLayoutSnapshot, TerminalTab } from '../../../shared/terminal-tab-types' import { useAppStore } from '@/store' import { DEFAULT_AGENT_HIBERNATION_IDLE_MS } from './agent-hibernation-planner' import { - resetAgentHibernationCoordinatorForTests, runAgentHibernationTick, startAgentHibernationCoordinator } from './agent-hibernation-coordinator' -import { hydrateDrivers, setDriverForPty } from './pane-manager/mobile-driver-state' -import { - registerVisibleTerminalTab, - resetForegroundTerminalTabIdsForTests, - setForegroundTerminalTabIds -} from './foreground-terminal-tabs' -import { - recordAgentHibernationPaneOutput, - resetAgentHibernationOutputActivityForTests -} from './agent-hibernation-output-activity' -import { - observeHibernationPtyBindings, - resetHibernationPaneAgeForTests -} from './agent-hibernation-pane-age' +import { setDriverForPty } from './pane-manager/mobile-driver-state' +import { registerVisibleTerminalTab, setForegroundTerminalTabIds } from './foreground-terminal-tabs' +import { recordAgentHibernationPaneOutput } from './agent-hibernation-output-activity' import { createCompatibleRuntimeStatusResponseIfNeeded } from '../runtime/runtime-compatibility-test-fixture' -import { clearRuntimeCompatibilityCacheForTests } from '../runtime/runtime-rpc-client' -import type { AppState } from '@/store/types' +import { + deferred, + entry, + installEligibleState, + installRuntimeListResponses, + layout, + LEAF, + mockRuntimeEnvironmentCall, + NOW, + resetAgentHibernationCoordinatorFixture, + runtimeListResult, + tab +} from './agent-hibernation-coordinator-test-fixture' -const NOW = 10_000_000 -const LEAF = '11111111-1111-4111-8111-111111111111' const PI_TRANSCRIPT_PATH = join(tmpdir(), 'pi-session-1.jsonl') -const mockRuntimeEnvironmentCall = vi.fn() - -vi.stubGlobal('window', { - api: { - runtimeEnvironments: { - call: mockRuntimeEnvironmentCall - } - } -}) - -function tab(): TerminalTab { - return { - id: 'tab-1', - ptyId: null, - worktreeId: 'wt-bg', - title: 'Agent', - customTitle: null, - color: null, - sortOrder: 0, - createdAt: 1 - } -} - -function layout(): TerminalLayoutSnapshot { - return { - root: { type: 'leaf', leafId: LEAF }, - activeLeafId: LEAF, - expandedLeafId: null, - ptyIdsByLeafId: { [LEAF]: 'pty-1' } - } -} - -function entry(): AgentStatusEntry { - return { - state: 'done', - prompt: 'ship it', - updatedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1, - stateStartedAt: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 1, - paneKey: `tab-1:${LEAF}`, - tabId: 'tab-1', - worktreeId: 'wt-bg', - agentType: 'claude', - providerSession: { key: 'session_id', id: 'session-1' }, - stateHistory: [] - } -} - -function installEligibleState( - shutdownCompletedAgentPaneForHibernation = vi.fn(), - overrides: Partial = {} -): typeof shutdownCompletedAgentPaneForHibernation { - const e = entry() - const runtimeOwnerEnvironmentId = overrides.settings?.activeRuntimeEnvironmentId ?? undefined - useAppStore.setState({ - settings: { - experimentalAgentHibernation: true, - agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS - } as never, - activeWorktreeId: 'wt-active', - repos: [], - worktreesByRepo: { - 'fixture-repo': [ - { id: 'wt-bg', repoId: 'fixture-repo', hostId: 'local', runtimeOwnerEnvironmentId } - ] - } as never, - detectedWorktreesByRepo: {}, - tabsByWorktree: { 'wt-bg': [tab()] }, - terminalLayoutsByTabId: { 'tab-1': layout() }, - ptyIdsByTabId: { 'tab-1': ['pty-1'] }, - agentStatusByPaneKey: { [e.paneKey]: e }, - sleepingAgentSessionsByPaneKey: {}, - lastTerminalInputAtByPaneKey: {}, - shutdownCompletedAgentPaneForHibernation: shutdownCompletedAgentPaneForHibernation as never, - shutdownWorktreeTerminals: vi.fn() as never, - ...overrides - }) - // Why: a pane idle long enough to hibernate has necessarily been observed by earlier - // coordinator passes, so its PTY binding is old. Seed that here — otherwise the - // binding-age floor (which exists to stop a wake or app restart sleeping the whole - // backlog immediately) would defer every candidate on its first observed tick. - const state = useAppStore.getState() - observeHibernationPtyBindings({ - tabsByWorktree: state.tabsByWorktree, - terminalLayoutsByTabId: state.terminalLayoutsByTabId, - now: NOW - DEFAULT_AGENT_HIBERNATION_IDLE_MS - 60_000, - idleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS - }) - return shutdownCompletedAgentPaneForHibernation -} - -function runtimeListResult(ptyIds: string[], truncated = false) { - return { - terminals: ptyIds.map((ptyId) => ({ - handle: `handle-${ptyId}`, - ptyId, - worktreeId: 'wt-bg', - worktreePath: '/tmp/wt-bg', - branch: 'feature', - tabId: `pty:${ptyId}`, - leafId: `pty:${ptyId}`, - title: 'Agent', - connected: true, - writable: true, - lastOutputAt: null, - preview: '' - })), - totalCount: ptyIds.length, - truncated - } -} - -function installRuntimeListResponses( - ...responses: (ReturnType | Error)[] -): void { - const queue = [...responses] - mockRuntimeEnvironmentCall.mockImplementation((args: { method: string }) => { - const compatible = createCompatibleRuntimeStatusResponseIfNeeded(args) - if (compatible) { - return Promise.resolve(compatible) - } - if (args.method === 'terminal.list') { - const response = queue.shift() ?? runtimeListResult(['pty-1']) - if (response instanceof Error) { - return Promise.reject(response) - } - return Promise.resolve({ - id: 'terminal-list', - ok: true, - result: response, - _meta: { runtimeId: 'runtime-1' } - }) - } - return Promise.resolve({ - id: 'default', - ok: true, - result: {}, - _meta: { runtimeId: 'runtime-1' } - }) - }) -} - -function deferred(): { - promise: Promise - resolve: (value: T) => void - reject: (error: Error) => void -} { - let resolve!: (value: T) => void - let reject!: (error: Error) => void - const promise = new Promise((res, rej) => { - resolve = res - reject = rej - }) - return { promise, resolve, reject } -} - -afterEach(() => { - resetAgentHibernationCoordinatorForTests() - clearRuntimeCompatibilityCacheForTests() - resetForegroundTerminalTabIdsForTests() - resetAgentHibernationOutputActivityForTests() - resetHibernationPaneAgeForTests() - hydrateDrivers([]) - mockRuntimeEnvironmentCall.mockReset() - vi.useRealTimers() -}) +afterEach(resetAgentHibernationCoordinatorFixture) describe('agent sleep coordinator', () => { it('hibernates an eligible background worktree after two stable ticks', async () => { @@ -484,40 +315,50 @@ describe('agent sleep coordinator', () => { expect(shutdown).not.toHaveBeenCalled() }) - it('hibernates a runtime-backed candidate with fresh liveness and exact PTYs', async () => { - vi.useFakeTimers() - installRuntimeListResponses( - runtimeListResult(['pty-1']), - runtimeListResult(['pty-1']), - runtimeListResult(['pty-1']) - ) - const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), { - settings: { - experimentalAgentHibernation: true, - agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS, - activeRuntimeEnvironmentId: 'runtime-1' - } as never, - ptyIdsByTabId: { 'tab-1': [] } - }) - startAgentHibernationCoordinator({ intervalMs: 1000, now: () => NOW }) - - await vi.advanceTimersByTimeAsync(1000) - await vi.advanceTimersByTimeAsync(1000) - - expect(shutdown).toHaveBeenCalledWith('wt-bg', { - paneKey: `tab-1:${LEAF}`, - tabId: 'tab-1', - leafId: LEAF, - ptyId: 'pty-1', - expectedRuntimePtyId: 'pty-1' - }) - expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith( - expect.objectContaining({ - method: 'terminal.list', - params: expect.objectContaining({ requireFreshPtyLiveness: true }) + it.each(['wt-bg', 'folder:folder-1'])( + 'hibernates a runtime-backed candidate in %s with fresh liveness and exact PTYs', + async (worktreeId) => { + vi.useFakeTimers() + const result = runtimeListResult(['pty-1']) + result.terminals[0].worktreeId = worktreeId + installRuntimeListResponses(result, result, result) + const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), { + settings: { + experimentalAgentHibernation: true, + agentHibernationIdleMs: DEFAULT_AGENT_HIBERNATION_IDLE_MS, + activeRuntimeEnvironmentId: 'runtime-1' + } as never, + folderWorkspaces: [ + { + id: 'folder-1', + folderPath: tmpdir(), + executionHostId: 'runtime:runtime-1' + } + ] as never, + tabsByWorktree: { [worktreeId]: [{ ...tab(), worktreeId }] }, + agentStatusByPaneKey: { [entry().paneKey]: { ...entry(), worktreeId } }, + ptyIdsByTabId: { 'tab-1': [] } }) - ) - }) + startAgentHibernationCoordinator({ intervalMs: 1000, now: () => NOW }) + + await vi.advanceTimersByTimeAsync(1000) + await vi.advanceTimersByTimeAsync(1000) + + expect(shutdown).toHaveBeenCalledWith(worktreeId, { + paneKey: `tab-1:${LEAF}`, + tabId: 'tab-1', + leafId: LEAF, + ptyId: 'pty-1', + expectedRuntimePtyId: 'pty-1' + }) + expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith( + expect.objectContaining({ + method: 'terminal.list', + params: expect.objectContaining({ requireFreshPtyLiveness: true }) + }) + ) + } + ) it('requires fresh runtime liveness for confirmation and pre-shutdown recheck', async () => { vi.useFakeTimers() @@ -546,7 +387,7 @@ describe('agent sleep coordinator', () => { }) it('revalidates a confirmed pane without listing unrelated runtime worktrees', async () => { - installRuntimeListResponses(...Array.from({ length: 5 }, () => runtimeListResult(['pty-1']))) + installRuntimeListResponses(...Array.from({ length: 3 }, () => runtimeListResult(['pty-1']))) const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), { settings: { experimentalAgentHibernation: true, @@ -583,15 +424,104 @@ describe('agent sleep coordinator', () => { const listCalls = mockRuntimeEnvironmentCall.mock.calls.filter( ([args]) => args.method === 'terminal.list' ) - // Two global confirmation samples list both worktrees; the destructive - // recheck lists only the candidate's owner: 2W + C, not 2W + C×W. - expect(listCalls).toHaveLength(5) + // Both confirmation samples and the destructive recheck query only the completed agent's owner. + expect(listCalls).toHaveLength(3) expect(listCalls.at(-1)?.[0]).toMatchObject({ selector: 'runtime-1', params: { worktree: expect.anything() } }) }) + it('does not request runtime inventories for 100 workspaces without completed agents', async () => { + installRuntimeListResponses() + const tabs = Array.from({ length: 100 }, (_, index) => ({ + ...tab(), + id: `tab-${index}`, + worktreeId: `wt-${index}` + })) + const shutdown = installEligibleState(vi.fn(), { + worktreesByRepo: { + 'fixture-repo': tabs.map((t) => ({ + id: t.worktreeId, + repoId: 'fixture-repo', + hostId: 'runtime:runtime-1', + runtimeOwnerEnvironmentId: 'runtime-1' + })) + } as never, + tabsByWorktree: Object.fromEntries(tabs.map((t) => [t.worktreeId, [t]])), + agentStatusByPaneKey: Object.fromEntries( + tabs.map((t, index) => [ + `${t.id}:${LEAF}`, + { + ...entry(), + tabId: t.id, + worktreeId: t.worktreeId, + paneKey: `${t.id}:${LEAF}`, + state: index % 2 === 0 ? 'working' : 'waiting' + } + ]) + ) + }) + + await runAgentHibernationTick() + + expect(mockRuntimeEnvironmentCall).not.toHaveBeenCalled() + expect(shutdown).not.toHaveBeenCalled() + }) + + it('requires host evidence after a skipped workspace completes during another inventory request', async () => { + const delayed = deferred>() + installRuntimeListResponses() + const respond = mockRuntimeEnvironmentCall.getMockImplementation()! + mockRuntimeEnvironmentCall.mockImplementation((args: { method: string }) => + args.method === 'terminal.list' + ? delayed.promise.then((result) => ({ id: 'delayed', ok: true, result })) + : respond(args) + ) + const first = entry() + const second = { ...entry(), tabId: 'tab-2', paneKey: `tab-2:${LEAF}`, worktreeId: 'wt-other' } + const shutdown = installEligibleState(vi.fn(), { + worktreesByRepo: { + 'fixture-repo': ['wt-bg', 'wt-other'].map((id) => ({ + id, + repoId: 'fixture-repo', + hostId: 'runtime:runtime-1', + runtimeOwnerEnvironmentId: 'runtime-1' + })) + } as never, + tabsByWorktree: { + 'wt-bg': [tab()], + 'wt-other': [{ ...tab(), id: 'tab-2', worktreeId: 'wt-other' }] + }, + agentStatusByPaneKey: { + [first.paneKey]: { ...first, state: 'working' }, + [second.paneKey]: second + } + }) + + const tick = runAgentHibernationTick() + await vi.waitFor(() => + expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith( + expect.objectContaining({ method: 'terminal.list' }) + ) + ) + useAppStore.setState({ agentStatusByPaneKey: { [first.paneKey]: first } }) + delayed.resolve(runtimeListResult([])) + await tick + + installRuntimeListResponses() + await runAgentHibernationTick() + expect(shutdown).not.toHaveBeenCalled() + await runAgentHibernationTick() + expect(shutdown).toHaveBeenCalledTimes(1) + expect(shutdown).toHaveBeenCalledWith( + 'wt-bg', + expect.objectContaining({ + expectedRuntimePtyId: 'pty-1' + }) + ) + }) + it('uses fresh store state after awaiting runtime liveness before shutdown', async () => { vi.useFakeTimers() const delayed = deferred>() diff --git a/src/renderer/src/lib/agent-hibernation-coordinator.ts b/src/renderer/src/lib/agent-hibernation-coordinator.ts index 5f17e1e4834..c7287aed26c 100644 --- a/src/renderer/src/lib/agent-hibernation-coordinator.ts +++ b/src/renderer/src/lib/agent-hibernation-coordinator.ts @@ -30,6 +30,7 @@ import type { RuntimeTerminalSummary } from '../../../shared/runtime-types' import { getWindowParkVisible, subscribeWindowParkVisibility } from './window-park-visibility' +import { getEntryTabId } from './agent-hibernation-pane-eligibility' export const AGENT_HIBERNATION_TICK_MS = 60 * 1000 @@ -79,7 +80,17 @@ function snapshotFromState( terminalLayoutsByTabId: state.terminalLayoutsByTabId, ptyIdsByTabId: state.ptyIdsByTabId, runtimeLivePtyIdsByWorktreeId: runtimeLiveness.runtimeLivePtyIdsByWorktreeId, - runtimeLivenessRequiredWorktreeIds: runtimeLiveness.runtimeLivenessRequiredWorktreeIds, + // Why: a workspace can gain tabs or resolve its runtime owner while the inventory above + // is in flight, and the plan is built from this later state. Union the fresh targets in + // so such a workspace is required-but-absent and the planner skips it, rather than + // answering for the execution host from client PTYs. Union, never replace: dropping a + // pre-await target would narrow the fail-closed set instead of widening it. + runtimeLivenessRequiredWorktreeIds: [ + ...new Set([ + ...runtimeLiveness.runtimeLivenessRequiredWorktreeIds, + ...getRuntimeLivenessTargetWorktrees(state, targetWorktreeId).keys() + ]) + ], mobileLockedPtyIds: [...getAllDrivers()] .filter(([, driver]) => driver.kind === 'mobile') .map(([ptyId]) => ptyId), @@ -132,8 +143,23 @@ async function collectRuntimePtyLiveness( const targets = getRuntimeLivenessTargetWorktrees(state, targetWorktreeId) const runtimeLivePtyIdsByWorktreeId: Record = {} const runtimeLivenessRequiredWorktreeIds = [...targets.keys()] + if (targets.size === 0) { + // Why: an all-local install has nothing to ask, so it must not pay the status scan below. + return { runtimeLivePtyIdsByWorktreeId, runtimeLivenessRequiredWorktreeIds } + } + const completedTabIds = new Set() + for (const entry of Object.values(state.agentStatusByPaneKey)) { + const tabId = entry?.state === 'done' ? getEntryTabId(entry) : null + if (tabId) { + completedTabIds.add(tabId) + } + } await Promise.all( [...targets].map(async ([worktreeId, runtimeEnvironmentId]) => { + if (!state.tabsByWorktree[worktreeId]?.some((tab) => completedTabIds.has(tab.id))) { + // Skipped owners still require host evidence if an agent completes during this pass. + return + } try { const result = await callRuntimeRpc( { kind: 'environment', environmentId: runtimeEnvironmentId }, diff --git a/src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts b/src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts new file mode 100644 index 00000000000..0c226ef0e6a --- /dev/null +++ b/src/renderer/src/lib/agent-hibernation-runtime-liveness-race.test.ts @@ -0,0 +1,127 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { useAppStore } from '@/store' +import { runAgentHibernationTick } from './agent-hibernation-coordinator' +import { + createCompatibleRuntimeStatusResponseIfNeeded, + type RuntimeEnvironmentCallRequest +} from '../runtime/runtime-compatibility-test-fixture' +import { + deferred, + entry, + installEligibleState, + layout, + LEAF, + mockRuntimeEnvironmentCall, + resetAgentHibernationCoordinatorFixture, + runtimeListResult, + tab +} from './agent-hibernation-coordinator-test-fixture' + +afterEach(resetAgentHibernationCoordinatorFixture) + +describe('agent sleep coordinator runtime-liveness races', () => { + it('requires host evidence when a workspace becomes runtime-owned during an inventory request', async () => { + const delayed = deferred>() + const lateTab = { ...tab(), id: 'tab-late', worktreeId: 'wt-late' } + const lateEntry = { + ...entry(), + tabId: 'tab-late', + worktreeId: 'wt-late', + paneKey: `tab-late:${LEAF}` + } + const lateList = { + ...runtimeListResult(['pty-late']), + terminals: runtimeListResult(['pty-late']).terminals.map((terminal) => ({ + ...terminal, + worktreeId: 'wt-late' + })) + } + let firstListPending = true + mockRuntimeEnvironmentCall.mockImplementation((args: RuntimeEnvironmentCallRequest) => { + const compatible = createCompatibleRuntimeStatusResponseIfNeeded(args) + if (compatible) { + return Promise.resolve(compatible) + } + if (args.method !== 'terminal.list') { + return Promise.resolve({ id: 'default', ok: true, result: {} }) + } + const isLate = args.params?.worktree === 'id:wt-late' + if (!isLate && firstListPending) { + firstListPending = false + return delayed.promise.then((result) => ({ + id: 'delayed', + ok: true, + result + })) + } + return Promise.resolve({ + id: 'terminal-list', + ok: true, + result: isLate ? lateList : runtimeListResult(['pty-1']) + }) + }) + // Why: `wt-late` starts local-owned, so the pre-await target sample never lists it. + const shutdown = installEligibleState(vi.fn().mockResolvedValue(undefined), { + worktreesByRepo: { + 'fixture-repo': [ + { + id: 'wt-bg', + repoId: 'fixture-repo', + hostId: 'runtime:runtime-1', + runtimeOwnerEnvironmentId: 'runtime-1' + }, + { id: 'wt-late', repoId: 'fixture-repo', hostId: 'local' } + ] + } as never, + tabsByWorktree: { 'wt-bg': [tab()], 'wt-late': [lateTab] }, + terminalLayoutsByTabId: { + 'tab-1': layout(), + 'tab-late': { + root: { type: 'leaf', leafId: LEAF }, + activeLeafId: LEAF, + expandedLeafId: null, + ptyIdsByLeafId: { [LEAF]: 'pty-late' } + } + }, + ptyIdsByTabId: { 'tab-1': ['pty-1'], 'tab-late': ['pty-late'] }, + agentStatusByPaneKey: { + [entry().paneKey]: entry(), + [lateEntry.paneKey]: lateEntry + } + }) + + const tick = runAgentHibernationTick() + await vi.waitFor(() => + expect(mockRuntimeEnvironmentCall).toHaveBeenCalledWith( + expect.objectContaining({ method: 'terminal.list' }) + ) + ) + // Runtime ownership resolves while the `wt-bg` inventory is still outstanding. + useAppStore.setState({ + worktreesByRepo: { + 'fixture-repo': ['wt-bg', 'wt-late'].map((id) => ({ + id, + repoId: 'fixture-repo', + hostId: 'runtime:runtime-1', + runtimeOwnerEnvironmentId: 'runtime-1' + })) + } as never + }) + delayed.resolve(runtimeListResult(['pty-1'])) + await tick + + // The racing pass has no host evidence for `wt-late`, so it must not count as one of + // the two confirmations; hibernating on the next tick would rest on client PTYs alone. + await runAgentHibernationTick() + expect(shutdown).not.toHaveBeenCalledWith('wt-late', expect.anything()) + + await runAgentHibernationTick() + expect(shutdown).toHaveBeenCalledWith( + 'wt-late', + expect.objectContaining({ + ptyId: 'pty-late', + expectedRuntimePtyId: 'pty-late' + }) + ) + }) +}) diff --git a/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts b/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts new file mode 100644 index 00000000000..98fcdeb8817 --- /dev/null +++ b/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.test.ts @@ -0,0 +1,252 @@ +import { createStore } from 'zustand/vanilla' +import { describe, expect, it, vi } from 'vitest' +import type { AppState } from '@/store/types' +import type { AgentProviderSessionMetadata } from '../../../shared/agent-session-resume' +import type { TerminalTab } from '../../../shared/terminal-tab-types' +import { aiVaultTitleSyncInputsChanged } from './ai-vault-tab-title-sync-inputs' +import { startAiVaultTabTitleSync } from './ai-vault-tab-title-sync' + +const COLLECTIONS = [ + 'agentStatusByPaneKey', + 'retainedAgentsByPaneKey', + 'sleepingAgentSessionsByPaneKey' +] as const +type Collection = (typeof COLLECTIONS)[number] + +function makeState(count = 1): AppState { + const live: AppState['agentStatusByPaneKey'] = {} + const retained: AppState['retainedAgentsByPaneKey'] = {} + const sleeping: AppState['sleepingAgentSessionsByPaneKey'] = {} + const tabs: TerminalTab[] = [] + for (let index = 0; index < count * 3; index++) { + const tab: TerminalTab = { + id: `tab-${index}`, + worktreeId: 'wt-title', + ptyId: null, + title: 'Agent', + customTitle: null, + color: null, + sortOrder: index, + createdAt: 1 + } + const entry = { + paneKey: `${tab.id}:00000000-0000-4000-8000-000000000001`, + tabId: tab.id, + worktreeId: tab.worktreeId, + agentType: 'codex' as const, + providerSession: { key: 'session_id' as const, id: `session-${index}` }, + state: 'done' as const, + prompt: '', + updatedAt: 1, + stateStartedAt: 1, + stateHistory: [] + } + tabs.push(tab) + if (index < count) { + live[entry.paneKey] = entry + } else if (index < count * 2) { + retained[entry.paneKey] = { + entry, + tab, + worktreeId: tab.worktreeId, + agentType: 'codex', + startedAt: 1 + } + } else { + sleeping[entry.paneKey] = { + ...entry, + agent: 'codex', + capturedAt: 1, + origin: 'worktree-sleep' + } + } + } + return { + agentStatusByPaneKey: live, + retainedAgentsByPaneKey: retained, + sleepingAgentSessionsByPaneKey: sleeping, + tabsByWorktree: { 'wt-title': tabs }, + terminalLayoutsByTabId: {}, + activeWorktreeId: 'wt-title', + activeWorkspaceExecutionHostId: 'local', + worktreesByRepo: { fixture: [{ id: 'wt-title', repoId: 'fixture', hostId: 'local' }] }, + detectedWorktreesByRepo: {}, + folderWorkspaces: [], + repos: [], + settings: {} + } as unknown as AppState +} + +function replaceProvider( + state: AppState, + collection: Collection, + providerSession: AgentProviderSessionMetadata +): AppState { + const paneKey = Object.keys(state[collection])[0] + const existing = state[collection][paneKey] + const next = + collection === 'retainedAgentsByPaneKey' + ? { ...existing, entry: { ...state.retainedAgentsByPaneKey[paneKey].entry, providerSession } } + : { ...existing, providerSession } + return { ...state, [collection]: { ...state[collection], [paneKey]: next } } +} + +describe('AI Vault title subscription inputs', () => { + it('does not enumerate unchanged retained and sleeping maps during live status writes', () => { + const state = makeState(500) + let unchangedEnumerations = 0 + const observeEnumerations = (records: T): T => + new Proxy(records, { + ownKeys(target) { + unchangedEnumerations++ + return Reflect.ownKeys(target) + } + }) + state.retainedAgentsByPaneKey = observeEnumerations(state.retainedAgentsByPaneKey) + state.sleepingAgentSessionsByPaneKey = observeEnumerations(state.sleepingAgentSessionsByPaneKey) + const store = createStore(() => state) + const scheduleReconcile = vi.fn(() => () => {}) + const stop = startAiVaultTabTitleSync({ + getState: store.getState, + subscribe: store.subscribe, + scheduleReconcile, + resolveSessionTitles: vi.fn() + }) + try { + const paneKey = Object.keys(state.agentStatusByPaneKey)[0] + for (let index = 0; index < 50; index++) { + store.setState((current) => ({ + agentStatusByPaneKey: { + ...current.agentStatusByPaneKey, + [paneKey]: { ...current.agentStatusByPaneKey[paneKey], updatedAt: index + 2 } + } + })) + } + expect(unchangedEnumerations).toBe(0) + expect(scheduleReconcile).toHaveBeenCalledTimes(1) + } finally { + stop() + } + }) + + it.each(COLLECTIONS)( + 'still detects provider changes in %s with other maps reused', + (collection) => { + const state = makeState() + const paneKey = Object.keys(state[collection])[0] + const provider = + collection === 'retainedAgentsByPaneKey' + ? state.retainedAgentsByPaneKey[paneKey].entry.providerSession! + : (state[collection][paneKey] as { providerSession: AgentProviderSessionMetadata }) + .providerSession + for (const patch of [ + { id: 'changed' }, + { key: 'conversation_id' as const }, + { transcriptPath: '/changed/session' } + ]) { + expect( + aiVaultTitleSyncInputsChanged( + replaceProvider(state, collection, { ...provider, ...patch }), + state + ) + ).toBe(true) + } + expect( + aiVaultTitleSyncInputsChanged(replaceProvider(state, collection, { ...provider }), state) + ).toBe(false) + } + ) + + it.each(COLLECTIONS)('detects additions and removals in %s', (collection) => { + const state = makeState() + const records = state[collection] + const empty = { ...state, [collection]: {} } + expect(aiVaultTitleSyncInputsChanged(empty, state)).toBe(true) + expect(aiVaultTitleSyncInputsChanged(state, empty)).toBe(true) + expect(aiVaultTitleSyncInputsChanged({ ...state, [collection]: { ...records } }, state)).toBe( + false + ) + }) + + it.each(COLLECTIONS)('detects agent and pane ownership changes in %s', (collection) => { + const state = makeState() + const records = state[collection] + const paneKey = Object.keys(records)[0] + const record = records[paneKey] + const entry = + collection === 'retainedAgentsByPaneKey' + ? state.retainedAgentsByPaneKey[paneKey].entry + : record + const agentField = collection === 'sleepingAgentSessionsByPaneKey' ? 'agent' : 'agentType' + const changedRecords = [ + { ...record, [agentField]: 'claude' }, + { ...record, [agentField]: 'gemini' }, + { ...record, worktreeId: 'other' }, + ...['paneKey', 'tabId'].map((field) => + collection === 'retainedAgentsByPaneKey' + ? { ...record, entry: { ...entry, [field]: 'other' } } + : { ...record, [field]: 'other' } + ) + ] + for (const changed of changedRecords) { + expect( + aiVaultTitleSyncInputsChanged( + { ...state, [collection]: { ...records, [paneKey]: changed } }, + state + ) + ).toBe(true) + } + }) + + it('still checks workspace ownership after unchanged record collections', () => { + const state = makeState() + for (const host of ['ssh:host-1', 'runtime:server-1'] as const) { + expect( + aiVaultTitleSyncInputsChanged( + { + ...state, + agentStatusByPaneKey: { ...state.agentStatusByPaneKey }, + activeWorkspaceExecutionHostId: host + }, + state + ) + ).toBe(true) + } + }) + + it('still checks active panes and stored titles after unchanged record collections', () => { + const state = makeState() + const agentStatusByPaneKey = { ...state.agentStatusByPaneKey } + expect( + aiVaultTitleSyncInputsChanged( + { + ...state, + agentStatusByPaneKey, + terminalLayoutsByTabId: { + 'tab-0': { root: null, activeLeafId: 'other', expandedLeafId: null } + } + }, + state + ) + ).toBe(true) + const tabs = state.tabsByWorktree['wt-title'] + expect( + aiVaultTitleSyncInputsChanged( + { + ...state, + agentStatusByPaneKey, + tabsByWorktree: { + 'wt-title': [ + { + ...tabs[0], + aiVaultTitle: { agent: 'codex', sessionId: 'session-0', title: 'New title' } + }, + ...tabs.slice(1) + ] + } + }, + state + ) + ).toBe(true) + }) +}) diff --git a/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.ts b/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.ts index 569d79b0dc8..75ff2b13a20 100644 --- a/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.ts +++ b/src/renderer/src/lib/ai-vault-tab-title-sync-inputs.ts @@ -22,6 +22,9 @@ function relevantRecordEqual( relevant: (entry: T) => boolean, equal: (current: T, previous: T) => boolean ): boolean { + if (current === previous) { + return true + } let currentCount = 0 let previousCount = 0 for (const [key, entry] of Object.entries(current)) { diff --git a/src/renderer/src/lib/workspace-doc-address-input.test.ts b/src/renderer/src/lib/workspace-doc-address-input.test.ts index e4a0e7b7477..9b8fd5fc341 100644 --- a/src/renderer/src/lib/workspace-doc-address-input.test.ts +++ b/src/renderer/src/lib/workspace-doc-address-input.test.ts @@ -144,4 +144,51 @@ describe('resolveWorkspaceDocAddressTarget', () => { resolveWorkspaceDocAddressTarget(makeState(), CURRENT, '/home/alice/wt1/x.html') ).toEqual({ status: 'unsupported', message: 'no channel' }) }) + + it('resolves a current-workspace address without enumerating unrelated workspace roots', () => { + const state = makeState() + state.allWorktrees = vi.fn(() => { + throw new Error('unnecessary global workspace enumeration') + }) + expect( + resolveWorkspaceDocAddressTarget(state, CURRENT, '/home/alice/wt1/docs/index.html').status + ).toBe('workspace-doc') + expect(state.allWorktrees).not.toHaveBeenCalled() + }) + + // The current worktree outranks every other root, including a more specific one nested inside it. + it('keeps the current worktree ahead of a workspace nested inside it', () => { + const state = makeState() + const nestedInsideCurrent = { + ...state, + allWorktrees: () => [ + ...state.allWorktrees(), + { id: 'repo3::/home/alice/wt1/vendor', path: '/home/alice/wt1/vendor' } + ] + } as typeof state + + expect( + resolveWorkspaceDocAddressTarget( + nestedInsideCurrent, + CURRENT, + '/home/alice/wt1/vendor/index.html' + ) + ).toMatchObject({ status: 'workspace-doc', docLocation: { worktreeId: CURRENT } }) + }) + + it('short-circuits for a folder workspace that is the current workspace', () => { + const state = makeState() + const folderCurrent = { + ...state, + getKnownWorktreeById: (id: string) => + id === 'folder:folder-1' ? { id: 'folder:folder-1', path: '/srv/site' } : undefined, + allWorktrees: vi.fn(() => { + throw new Error('unnecessary global workspace enumeration') + }) + } as unknown as AppState + + expect( + resolveWorkspaceDocAddressTarget(folderCurrent, 'folder:folder-1', '/srv/site/index.html') + ).toMatchObject({ status: 'workspace-doc', docLocation: { worktreeId: 'folder:folder-1' } }) + }) }) diff --git a/src/renderer/src/lib/workspace-doc-address-input.ts b/src/renderer/src/lib/workspace-doc-address-input.ts index 941a905b1cf..97be81a15b7 100644 --- a/src/renderer/src/lib/workspace-doc-address-input.ts +++ b/src/renderer/src/lib/workspace-doc-address-input.ts @@ -19,7 +19,6 @@ function ownedWorktreeRoots( currentWorktreeId: string ): { worktreeId: string; root: string }[] { const roots: { worktreeId: string; root: string }[] = [] - const currentPath = state.getKnownWorktreeById(currentWorktreeId)?.path ?? null for (const worktree of state.allWorktrees()) { if (worktree.id !== currentWorktreeId && worktree.path) { roots.push({ worktreeId: worktree.id, root: worktree.path }) @@ -34,7 +33,7 @@ function ownedWorktreeRoots( // Most-specific root first (after the current worktree), so a file inside a nested workspace is // attributed to that workspace and not to the outer one that lexically contains it too. roots.sort((a, b) => b.root.length - a.root.length) - return currentPath ? [{ worktreeId: currentWorktreeId, root: currentPath }, ...roots] : roots + return roots } function planToTarget( @@ -80,6 +79,10 @@ export function resolveWorkspaceDocAddressTarget( if (input.split(/[\\/]+/).some((segment) => segment === '..' || segment === '.')) { return { status: 'not-a-workspace-doc' } } + const currentPath = state.getKnownWorktreeById(currentWorktreeId)?.path + if (currentPath && getRelativePathInsideRoot(input, currentPath)) { + return planToTarget(state, currentWorktreeId, input) + } for (const { worktreeId, root } of ownedWorktreeRoots(state, currentWorktreeId)) { if (getRelativePathInsideRoot(input, root)) { return planToTarget(state, worktreeId, input) diff --git a/src/renderer/src/lib/worktree-activation-surface-caller-wiring.test.ts b/src/renderer/src/lib/worktree-activation-surface-caller-wiring.test.ts index 419e642adcb..25bce98af64 100644 --- a/src/renderer/src/lib/worktree-activation-surface-caller-wiring.test.ts +++ b/src/renderer/src/lib/worktree-activation-surface-caller-wiring.test.ts @@ -17,7 +17,7 @@ const SURFACE_PROVIDING_CALLERS = [ 'src/renderer/src/hooks/composer-state/full-creation-execution.ts', 'src/renderer/src/lib/fix-checks-agent-launch.ts', 'src/renderer/src/lib/launch-work-item-direct.ts', - 'src/renderer/src/lib/worktree-creation-flow-execute.ts', + 'src/renderer/src/lib/worktree-creation-structured-session.ts', 'src/renderer/src/lib/workspace-port-actions.ts', 'src/renderer/src/store/repos/repo-add-actions.ts' ] diff --git a/src/renderer/src/lib/worktree-creation-flow-execute.ts b/src/renderer/src/lib/worktree-creation-flow-execute.ts index 27ee8da4827..de60076cd31 100644 --- a/src/renderer/src/lib/worktree-creation-flow-execute.ts +++ b/src/renderer/src/lib/worktree-creation-flow-execute.ts @@ -173,21 +173,18 @@ export async function executeWorktreeCreation( let activation: ActivateAndRevealResult | false = false let primaryTabId: string | null - if (shouldActivateOnCompletion) { + if (shouldActivateOnCompletion && !structuredLaunch) { activation = activateAndRevealWorktree(worktree.id, { sidebarRevealBehavior: 'auto', ...(result.setup ? { setup: result.setup } : {}), ...(result.defaultTabs ? { defaultTabs: result.defaultTabs } : {}), ...(startupOpt ? { startup: startupOpt } : {}), ...(preparedRequest.issueCommand ? { issueCommand: preparedRequest.issueCommand } : {}), - ...(backendSpawned ? { backendStartupTerminalSpawned: true } : {}), - ...(structuredLaunch ? { providesInitialSurface: true } : {}) + ...(backendSpawned ? { backendStartupTerminalSpawned: true } : {}) }) primaryTabId = activation === false ? null : activation.primaryTabId } else { - // The user moved on. Seed the worktree's terminal + setup in the background - // (setActiveTab only writes global focus for the active worktree, so this is - // safe) without yanking them back to it. + // Keep chat creation on its pending surface until the session is ready. const hasExplicitTerminalWork = Boolean( startupOpt || result.setup || preparedRequest.issueCommand || result.defaultTabs ) diff --git a/src/renderer/src/lib/worktree-creation-structured-session.test.ts b/src/renderer/src/lib/worktree-creation-structured-session.test.ts index c58fcd40dce..161f22a3f7c 100644 --- a/src/renderer/src/lib/worktree-creation-structured-session.test.ts +++ b/src/renderer/src/lib/worktree-creation-structured-session.test.ts @@ -10,7 +10,8 @@ const mocks = vi.hoisted(() => ({ cancelStructuredAgentLaunch: vi.fn(), closeStructuredAgentSession: vi.fn(), callRuntimeRpc: vi.fn(), - activateStructuredAgentSessionById: vi.fn() + activateStructuredAgentSessionById: vi.fn(), + activateAndRevealWorktree: vi.fn() })) vi.mock('@/store', () => ({ @@ -51,7 +52,7 @@ vi.mock('@/lib/worktree-initial-terminal-seeding', () => ({ })) vi.mock('@/lib/worktree-activation', () => ({ - activateAndRevealWorktree: vi.fn() + activateAndRevealWorktree: mocks.activateAndRevealWorktree })) vi.mock('@/lib/agent-trust-preflight', () => ({ @@ -170,4 +171,41 @@ describe('launchStructuredWorktreeSession', () => { expect(releaseCallerAfterUnknownOutcome).toHaveBeenCalledOnce() expect(mocks.unsubscribe).toHaveBeenCalledOnce() }) + + it('activates the workspace before selecting a chat when creation deferred activation', async () => { + mocks.startStructuredAgentLaunch.mockReturnValue({ + sessionId: 'session-1', + launchResult: Promise.resolve({ sessionId: 'session-1', fence: 1 }), + claimDefinitiveRefusalFallback: vi.fn(() => Promise.resolve(false)) + }) + mocks.activateAndRevealWorktree.mockReturnValue({ primaryTabId: null }) + + const result = await launchStructuredWorktreeSession({ + creationId: 'creation-1', + request: { + repoId: 'repo-1', + name: 'routing-recovery', + setupDecision: 'run', + agent: 'codex', + pendingFirstAgentMessageRename: false, + note: '', + startupPlan: null, + quickPrompt: 'Fix the route', + quickTelemetry: null + }, + worktreeId: 'worktree-1', + shouldActivateOnCompletion: true, + fallbackStartupOpt: undefined, + activation: false, + primaryTabId: null + }) + + expect(mocks.activateAndRevealWorktree).toHaveBeenCalledWith('worktree-1', { + providesInitialSurface: true + }) + expect(mocks.activateAndRevealWorktree.mock.invocationCallOrder[0]).toBeLessThan( + mocks.activateStructuredAgentSessionById.mock.invocationCallOrder[0] + ) + expect(result.activation).toEqual({ primaryTabId: null }) + }) }) diff --git a/src/renderer/src/lib/worktree-creation-structured-session.ts b/src/renderer/src/lib/worktree-creation-structured-session.ts index 2e2cd07698a..32bed399078 100644 --- a/src/renderer/src/lib/worktree-creation-structured-session.ts +++ b/src/renderer/src/lib/worktree-creation-structured-session.ts @@ -139,6 +139,11 @@ export async function launchStructuredWorktreeSession(args: { return { accepted, cancelled, visibilityUnknown, activation, primaryTabId } } if (args.shouldActivateOnCompletion) { + // Chat selection requires its workspace to be active. + if (!activation) { + activation = activateAndRevealWorktree(args.worktreeId, { providesInitialSurface: true }) + primaryTabId = activation === false ? null : activation.primaryTabId + } activateStructuredAgentSessionById({ worktreeId: args.worktreeId, sessionId: receipt.sessionId diff --git a/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts b/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts index bd4f278c04e..590d0cd9c67 100644 --- a/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts +++ b/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection-join.test.ts @@ -30,7 +30,7 @@ function mapOf(indices: readonly number[]): AppState['agentStatusByPaneKey'] { return map } -/** Re-spread with the same entry objects, as a status ping does. */ +/** A collection refresh can replace its container while preserving the rows. */ function respread(map: AppState['agentStatusByPaneKey']): AppState['agentStatusByPaneKey'] { return { ...map } } @@ -76,7 +76,7 @@ describe('agent-status projection join short circuit', () => { const map = mapOf([0, 1, 2]) const first = buildRuntimeMobileAgentStatusProjectionForTests(map) - // A new map identity with identical entry references — the common ping shape. + // A new map identity with identical entry references. const { result, joins, sorts } = countJoins(() => buildRuntimeMobileAgentStatusProjectionForTests(respread(map)) ) @@ -101,6 +101,41 @@ describe('agent-status projection join short circuit', () => { expect(sorts).toBe(0) }) + it('does not rejoin accumulated previews for timestamp-only heartbeats', () => { + let map = mapOf(Array.from({ length: 500 }, (_, index) => index)) + for (const paneKey of Object.keys(map)) { + map[paneKey] = { + ...map[paneKey], + updatedAt: 30_000_000, + lastAssistantMessage: 'answer '.repeat(1_000) + } + } + const first = buildRuntimeMobileAgentStatusProjectionForTests(map) + + const { result, joins } = countJoins(() => { + let projection = first + for (let index = 0; index < 50; index++) { + const paneKey = `tab-${index}:leaf-0` + map = { ...map, [paneKey]: { ...map[paneKey], updatedAt: 30_000_001 + index } } + projection = buildRuntimeMobileAgentStatusProjectionForTests(map) + } + return projection + }) + + expect(result).toBe(first) + expect(joins).toBe(0) + + const paneKey = 'tab-0:leaf-0' + const nextBucket = { ...map, [paneKey]: { ...map[paneKey], updatedAt: 30_030_000 } } + expect(buildRuntimeMobileAgentStatusProjectionForTests(nextBucket)).not.toBe(first) + expect( + buildRuntimeMobileAgentStatusProjectionForTests({ + ...map, + [paneKey]: { ...map[paneKey], lastAssistantMessage: 'A new answer' } + }) + ).not.toBe(first) + }) + it('still rebuilds when an entry changes', () => { resetRuntimeMobileAgentStatusProjectionCacheForTests() const map = mapOf([0, 1]) @@ -132,6 +167,20 @@ describe('agent-status projection join short circuit', () => { ) }) + it('still rebuilds when key membership swaps at a constant entry count', () => { + // The entry-count check cannot see a same-size swap; only the per-key lookup does. + resetRuntimeMobileAgentStatusProjectionCacheForTests() + const map = mapOf([0, 1]) + const first = buildRuntimeMobileAgentStatusProjectionForTests(map) + + const swapped = { 'tab-0:leaf-0': map['tab-0:leaf-0'], 'tab-9:leaf-0': makeEntry(9) } + const result = buildRuntimeMobileAgentStatusProjectionForTests(swapped) + + expect(result).not.toBe(first) + expect(result).toContain('tab-9:leaf-0') + expect(result).not.toContain('tab-1:leaf-0') + }) + it('still rebuilds when a pane is added', () => { resetRuntimeMobileAgentStatusProjectionCacheForTests() const map = mapOf([0]) diff --git a/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts b/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts index 979df97762b..16c188edef5 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts @@ -60,6 +60,7 @@ export function buildRuntimeMobileAgentStatusProjection( // A status ping replaces one entry and re-spreads the map; reuse every other entry. const entries = new Map() const parts: string[] = [] + let projectionUnchanged = cached != null && nextEntries.length === cached.entries.size // Code-unit order, not `localeCompare`: this projection is only ever compared with `===`, so it // must be deterministic, not locale-correct — and an ICU collator per comparison is ~4.5k calls // per ping at the 500-entry cap. @@ -71,8 +72,10 @@ export function buildRuntimeMobileAgentStatusProjection( : { entry, projection: serializeAgentStatusEntry(paneKey, entry) } entries.set(paneKey, entryCache) parts.push(entryCache.projection) + projectionUnchanged &&= previous?.projection === entryCache.projection } - const projection = `[${parts.join(',')}]` + // Same-bucket heartbeats must not rejoin every pane's accumulated preview text. + const projection = projectionUnchanged && cached ? cached.projection : `[${parts.join(',')}]` graphState.cachedAgentStatusProjection = { source: agentStatusByPaneKey, entries, diff --git a/src/renderer/src/store/slices/agent-status-pane-keyed-records.test.ts b/src/renderer/src/store/slices/agent-status-pane-keyed-records.test.ts index 5b327561637..02d0e9e56ff 100644 --- a/src/renderer/src/store/slices/agent-status-pane-keyed-records.test.ts +++ b/src/renderer/src/store/slices/agent-status-pane-keyed-records.test.ts @@ -3,7 +3,8 @@ import { RECENTLY_CLOSED_AGENT_STATUS_TAB_IDS_MAX, RECENTLY_RETIRED_AGENT_STATUS_PANE_KEYS_MAX, boundRecentlyClosedAgentStatusTabIds, - boundRecentlyRetiredAgentStatusPaneKeys + boundRecentlyRetiredAgentStatusPaneKeys, + removePaneKeys } from './agent-status-pane-keyed-records' function keyRecord(keys: readonly string[]): Record { @@ -108,3 +109,49 @@ describe('boundRecentlyClosedAgentStatusTabIds', () => { expect(next.fresh).toBe(true) }) }) + +it('does not enumerate unrelated pane records when removing absent keys', () => { + let enumerations = 0 + const record = new Proxy( + Object.fromEntries(Array.from({ length: 1000 }, (_, index) => [`tab-${index}:leaf`, index])), + { + ownKeys(target) { + enumerations += 1 + return Reflect.ownKeys(target) + } + } + ) + for (let index = 0; index < 200; index += 1) { + expect(removePaneKeys(record, new Set(['absent:leaf']))).toBe(record) + } + expect(enumerations).toBe(0) +}) + +it('removes enumerable undefined values without removing inherited or hidden keys', () => { + const record = { visible: undefined } + Object.defineProperty(record, 'hidden', { value: 1, enumerable: false }) + expect(removePaneKeys(record, new Set(['hidden', 'toString']))).toBe(record) + expect(Object.hasOwn(removePaneKeys(record, new Set(['visible'])), 'visible')).toBe(false) +}) + +// Probing the requested keys only matches the old record-enumeration when the two sets +// disagree in both directions, and the store relies on the unchanged case staying identical. +it('drops every requested key that is present and keeps the reference when none are', () => { + const base = { a: 1, b: 2, c: 3 } + expect(removePaneKeys({ ...base }, new Set(['a', 'b', 'c', 'd', 'e']))).toEqual({}) + expect(removePaneKeys({ ...base }, new Set(['b', 'missing']))).toEqual({ a: 1, c: 3 }) + const disjoint = { ...base } + expect(removePaneKeys(disjoint, new Set(['x', 'y']))).toBe(disjoint) + const noKeys = { ...base } + expect(removePaneKeys(noKeys, new Set())).toBe(noKeys) + const empty = {} + expect(removePaneKeys(empty, new Set(['a']))).toBe(empty) +}) + +it('leaves surviving keys in their original order and does not pollute the prototype', () => { + const record: Record = { z: 1, y: 2, x: 3, w: 4 } + expect(Object.keys(removePaneKeys(record, new Set(['x', 'z'])))).toEqual(['y', 'w']) + const plain = { safe: 1 } + expect(removePaneKeys(plain, new Set(['__proto__', 'constructor']))).toBe(plain) + expect(Object.getPrototypeOf(plain)).toBe(Object.prototype) +}) diff --git a/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts b/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts index f9199140c07..3acf5258f2c 100644 --- a/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts +++ b/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts @@ -83,15 +83,20 @@ export function removePaneKeys( record: Record, paneKeys: ReadonlySet ): Record { - const matchingKeys = Object.keys(record).filter((key) => paneKeys.has(key)) - if (matchingKeys.length === 0) { - return record - } - const next = { ...record } - for (const key of matchingKeys) { + // Probe the requested keys instead of enumerating the record, and copy only once a key + // actually matches: pane retirement calls this across a dozen records that usually hold + // none of the retired keys, so the no-op path must stay allocation-free and keep returning + // the same reference. `propertyIsEnumerable` is own-only, so inherited keys such as + // `toString` or `__proto__` are never deletable. + let next: Record | null = null + for (const key of paneKeys) { + if (!Object.prototype.propertyIsEnumerable.call(record, key)) { + continue + } + next ??= { ...record } delete next[key] } - return next + return next ?? record } export function removePaneKeysByTabPrefix( diff --git a/src/shared/agent-status-osc-pending-retention.test.ts b/src/shared/agent-status-osc-pending-retention.test.ts new file mode 100644 index 00000000000..1261d96f155 --- /dev/null +++ b/src/shared/agent-status-osc-pending-retention.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from 'vitest' +import * as ownership from './own-retained-string' +import { createAgentStatusOscProcessor } from './agent-status-osc' + +const INCOMPLETE_STATUS = '\x1b]9999;{"state":"working","prompt":"fragment' + +describe('OSC 9999 pending storage', () => { + it('routes every retained pending frame through ownRetainedString', () => { + const own = vi.spyOn(ownership, 'ownRetainedString') + try { + const process = createAgentStatusOscProcessor() + expect(process('x'.repeat(32 * 1024) + INCOMPLETE_STATUS).cleanData).toHaveLength(32 * 1024) + expect(own).toHaveBeenCalledTimes(1) + expect(own).toHaveBeenLastCalledWith(INCOMPLETE_STATUS) + + // Ownership is unconditional: a growing frame is re-owned on every chunk. + for (let index = 0; index < 8; index += 1) { + process('x'.repeat(512)) + } + expect(own).toHaveBeenCalledTimes(9) + expect(own).toHaveBeenLastCalledWith(INCOMPLETE_STATUS + 'x'.repeat(8 * 512)) + } finally { + own.mockRestore() + } + }) + + it.each([ + [16 * 1024, 256], + [64 * 1024, 64], + [1024 * 1024, 16] + ])('keeps %i-character chunks with %i incomplete statuses byte-exact', (size, count) => { + const parsers: ReturnType[] = [] + let cleanChars = 0 + + for (let index = 0; index < count; index += 1) { + const process = createAgentStatusOscProcessor() + cleanChars += process(String.fromCharCode(65 + (index % 26)).repeat(size) + INCOMPLETE_STATUS) + .cleanData.length + parsers.push(process) + } + + expect(cleanChars).toBe(size * count) + for (const process of parsers) { + expect(process('"}\x07after')).toEqual({ + cleanData: 'after', + payloads: [{ state: 'working', prompt: 'fragment' }], + lastPayloadCleanOffset: 0 + }) + } + }) + + it.each(['\x07', '\x1b\\'])( + 'preserves raw UTF-16 across an owned suffix and %j', + (terminator) => { + const process = createAgentStatusOscProcessor() + const ordinary = '😀'.repeat(8 * 1024) + const promptStart = '漢字\ud800|\udc00|\ud83d' + + expect(process(`${ordinary}\x1b]9999;{"state":"working","prompt":"${promptStart}`)).toEqual({ + cleanData: ordinary, + payloads: [], + lastPayloadCleanOffset: null + }) + expect(process(`\ude00"}${terminator}after`)).toEqual({ + cleanData: 'after', + payloads: [{ state: 'working', prompt: '漢字\ud800|\udc00|😀' }], + lastPayloadCleanOffset: 0 + }) + } + ) + + it('drops an owned frame that grows past the pending cap', () => { + const process = createAgentStatusOscProcessor() + const marker = '\x1b]9999;{"state":"working"}' + const pending = marker + ' '.repeat(64 * 1024 - marker.length) + for (let offset = 0; offset < pending.length; offset += 512) { + process(pending.slice(offset, offset + 512)) + } + + expect(process('\x07after')).toEqual({ + cleanData: 'after', + payloads: [{ state: 'working', prompt: '' }], + lastPayloadCleanOffset: 0 + }) + }) +}) diff --git a/src/shared/agent-status-osc.ts b/src/shared/agent-status-osc.ts index da1610fc1c4..adb0c0eee47 100644 --- a/src/shared/agent-status-osc.ts +++ b/src/shared/agent-status-osc.ts @@ -1,5 +1,6 @@ import type { ParsedAgentStatusPayload } from './agent-status-types' import { parseAgentStatusPayload } from './agent-status-types' +import { ownRetainedString } from './own-retained-string' const OSC_AGENT_STATUS_PREFIX = '\x1b]9999;' @@ -103,7 +104,8 @@ export function createAgentStatusOscProcessor(): (data: string) => ProcessedAgen if (terminator === null) { const candidate = combined.slice(start) - pending = candidate.length > MAX_PENDING ? '' : candidate + // Own the frame so it stops pinning the consumed chunk it was sliced from. + pending = candidate.length > MAX_PENDING ? '' : ownRetainedString(candidate) break } diff --git a/src/shared/browser-network-tunnel-stream-framing.test.ts b/src/shared/browser-network-tunnel-stream-framing.test.ts index 621f297a525..4ca73810cdf 100644 --- a/src/shared/browser-network-tunnel-stream-framing.test.ts +++ b/src/shared/browser-network-tunnel-stream-framing.test.ts @@ -171,4 +171,45 @@ describe('browser network tunnel stream framing', () => { expect(writer.queuedBytes).toBe(0) expect(onError).not.toHaveBeenCalled() }) + + it('rejects saturated writer frames before encoding and copying their payloads', () => { + const writer = new BrowserNetworkTunnelStreamFrameWriter( + () => {}, + () => {}, + { maxQueuedFrames: 1 } + ) + const frame = new Uint8Array(65536) + expect(writer.send(frame)).toBe(true) + const set = vi.spyOn(Uint8Array.prototype, 'set') + try { + for (let index = 0; index < 1000; index += 1) { + expect(writer.send(frame)).toBe(false) + } + expect(set.mock.calls.length).toBe(0) + expect(writer.queuedBytes).toBe(65540) + } finally { + set.mockRestore() + writer.close() + } + }) + + it('admits a frame that exactly fills the byte cap and rejects one byte past it', () => { + const writer = new BrowserNetworkTunnelStreamFrameWriter( + () => {}, + () => {}, + { maxQueuedBytes: 158 } + ) + expect(writer.send(new Uint8Array(100))).toBe(true) + expect(writer.queuedBytes).toBe(104) + expect(writer.send(new Uint8Array(51))).toBe(false) + expect(writer.send(new Uint8Array(50))).toBe(true) + expect(writer.queuedBytes).toBe(158) + writer.close() + }) + + it('encodes exactly the byte count writer admission reserves', () => { + for (const size of [1, 2, 255, 4096, 65536, 64 * 1024 + 16]) { + expect(encodeBrowserNetworkTunnelStreamFrame(new Uint8Array(size)).byteLength).toBe(4 + size) + } + }) }) diff --git a/src/shared/browser-network-tunnel-stream-framing.ts b/src/shared/browser-network-tunnel-stream-framing.ts index 042224253ee..3f000a4d380 100644 --- a/src/shared/browser-network-tunnel-stream-framing.ts +++ b/src/shared/browser-network-tunnel-stream-framing.ts @@ -4,11 +4,16 @@ const DEFAULT_MAX_RETAINED_BYTES = 2 * 1024 * 1024 const DEFAULT_MAX_QUEUED_BYTES = 1024 * 1024 const DEFAULT_MAX_QUEUED_FRAMES = 512 +// Single source of truth so writer admission reserves exactly what encoding allocates. +function encodedFrameByteLength(frame: Uint8Array): number { + return LENGTH_BYTES + frame.byteLength +} + export function encodeBrowserNetworkTunnelStreamFrame(frame: Uint8Array): Uint8Array { if (frame.byteLength === 0 || frame.byteLength > DEFAULT_MAX_FRAME_BYTES) { throw new Error('browser_tunnel_stream_frame_invalid') } - const encoded = new Uint8Array(LENGTH_BYTES + frame.byteLength) + const encoded = new Uint8Array(encodedFrameByteLength(frame)) new DataView(encoded.buffer).setUint32(0, frame.byteLength, false) encoded.set(frame, LENGTH_BYTES) return encoded @@ -134,18 +139,18 @@ export class BrowserNetworkTunnelStreamFrameWriter { if (this.closed) { return false } + if ( + this.retainedBytes + encodedFrameByteLength(frame) > this.maxQueuedBytes || + this.frames.length + (this.writing ? 1 : 0) >= this.maxQueuedFrames + ) { + return false + } let encoded: Uint8Array try { encoded = encodeBrowserNetworkTunnelStreamFrame(frame) } catch { return false } - if ( - this.retainedBytes + encoded.byteLength > this.maxQueuedBytes || - this.frames.length + (this.writing ? 1 : 0) >= this.maxQueuedFrames - ) { - return false - } this.frames.push(encoded) this.retainedBytes += encoded.byteLength this.pump() diff --git a/src/shared/osc-title-scan-tail-retention.test.ts b/src/shared/osc-title-scan-tail-retention.test.ts new file mode 100644 index 00000000000..ab1643b634e --- /dev/null +++ b/src/shared/osc-title-scan-tail-retention.test.ts @@ -0,0 +1,65 @@ +import { describe, expect, it, vi } from 'vitest' +import * as ownership from './own-retained-string' +import { extractOscTitleScanTail } from './osc-title-scan-tail' + +const INCOMPLETE_TITLE = '\x1b]2;Working on a terminal title' + +describe('OSC title scan tail storage', () => { + it('routes every retained title tail through ownRetainedString', () => { + const own = vi.spyOn(ownership, 'ownRetainedString') + try { + const tail = extractOscTitleScanTail('x'.repeat(32 * 1024) + INCOMPLETE_TITLE) + expect(own).toHaveBeenCalledTimes(1) + expect(own).toHaveBeenLastCalledWith(INCOMPLETE_TITLE) + expect(tail).toBe(INCOMPLETE_TITLE) + + // Ownership is unconditional: a growing title is re-owned on every chunk. + let pending = tail + for (let index = 0; index < 8; index += 1) { + pending = extractOscTitleScanTail(pending + 'x'.repeat(512)) + } + expect(own).toHaveBeenCalledTimes(9) + expect(own).toHaveBeenLastCalledWith(pending) + + // An unrelated incomplete OSC still yields nothing to retain. + expect(extractOscTitleScanTail(`${'x'.repeat(32 * 1024)}\x1b]9999;incomplete`)).toBe('') + } finally { + own.mockRestore() + } + }) + + it.each([ + [16 * 1024, 256, INCOMPLETE_TITLE], + [64 * 1024, 64, INCOMPLETE_TITLE], + [1024 * 1024, 16, INCOMPLETE_TITLE], + [16 * 1024, 128, INCOMPLETE_TITLE + 'x'.repeat(16 * 1024)] + ])('keeps %i-character chunks with %i incomplete titles byte-exact', (size, count, title) => { + const tails: string[] = [] + for (let index = 0; index < count; index++) { + tails.push( + extractOscTitleScanTail(String.fromCharCode(65 + (index % 26)).repeat(size) + title) + ) + } + const expected = title.length <= 4096 ? title : title.slice(0, 4) + title.slice(-4092) + expect(tails).toEqual(Array.from({ length: count }, () => expected)) + }) + + it.each(['0', '1', '2'])('preserves title %s introducer and exact UTF-16 at the cap', (code) => { + const prefix = `\x1b]${code};` + for (const length of [4095, 4096, 4097, 16 * 1024]) { + const value = `${prefix}${'x'.repeat(length - 9)}漢\ud800|\udc00\ud83d` + const expected = value.length <= 4096 ? value : prefix + value.slice(-4092) + const tail = extractOscTitleScanTail('a'.repeat(32 * 1024) + value) + expect(tail).toBe(expected) + expect(extractOscTitleScanTail(`${tail}\ude00\x1b\\`)).toBe('') + } + }) + + it('keeps trimming an owned title that grows past the cap', () => { + let pending = '\x1b]2;' + for (let index = 0; index < 128; index++) { + pending = extractOscTitleScanTail(pending + 'x'.repeat(512)) + } + expect(pending).toBe(`\x1b]2;${'x'.repeat(4092)}`) + }) +}) diff --git a/src/shared/osc-title-scan-tail.ts b/src/shared/osc-title-scan-tail.ts index 62e49cf9236..23822f6c2a4 100644 --- a/src/shared/osc-title-scan-tail.ts +++ b/src/shared/osc-title-scan-tail.ts @@ -1,3 +1,5 @@ +import { ownRetainedString } from './own-retained-string' + const OSC_TITLE_SCAN_TAIL_LIMIT = 4096 const OSC_TITLE_PREFIX_LENGTH = 4 const OSC_TITLE_CODES = new Set(['0', '1', '2']) @@ -7,7 +9,8 @@ export function extractOscTitleScanTail(input: string): string { if (lastOsc !== -1) { const suffix = input.slice(lastOsc) if (!suffix.includes('\x07') && !suffix.includes('\x1b\\')) { - return extractIncompleteTitleOscTail(suffix) + // Own the tail so it stops pinning the consumed chunk it was sliced from. + return ownRetainedString(extractIncompleteTitleOscTail(suffix)) } return input.endsWith('\x1b') ? '\x1b' : '' } diff --git a/src/shared/own-retained-string.test.ts b/src/shared/own-retained-string.test.ts new file mode 100644 index 00000000000..02cc9256770 --- /dev/null +++ b/src/shared/own-retained-string.test.ts @@ -0,0 +1,97 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { ownRetainedString, resetOwnRetainedStringCopier } from './own-retained-string' +import { copyUtf16SuffixToOwnedString } from './owned-utf16-suffix' + +const PARENT_CHARS = 1024 * 1024 +const PARENTS = 32 +const TAIL_CHARS = 4096 + +function collectHeap(): number { + const gc = (globalThis as { gc?: () => void }).gc + if (!gc) { + throw new Error('global.gc unavailable - config/vitest.config.ts must pass --expose-gc') + } + void /reset/.test('reset') + gc() + gc() + return process.memoryUsage().heapUsed +} + +function retainedBytes(take: (parent: string) => string): number { + const kept: string[] = [] + for (let index = 0; index < 8; index += 1) { + take('warmup'.repeat(PARENT_CHARS / 6)) + } + const before = collectHeap() + for (let index = 0; index < PARENTS; index += 1) { + kept.push(take(String.fromCharCode(65 + (index % 26)).repeat(PARENT_CHARS))) + } + const used = collectHeap() - before + expect(kept).toHaveLength(PARENTS) + expect(kept[0]).toHaveLength(TAIL_CHARS) + return used +} + +describe('ownRetainedString', () => { + afterEach(() => { + resetOwnRetainedStringCopier() + }) + + it('releases the parent chunk that a retained tail was sliced from', () => { + const sliced = retainedBytes((parent) => parent.slice(parent.length - TAIL_CHARS)) + const owned = retainedBytes((parent) => + ownRetainedString(parent.slice(parent.length - TAIL_CHARS)) + ) + + // 32 x 1 Mi parents pinned by 32 x 4 Ki tails, versus the tails alone. + expect(sliced).toBeGreaterThan(16 * 1024 * 1024) + expect(owned).toBeLessThan(4 * 1024 * 1024) + expect(owned).toBeLessThan(sliced / 8) + }) + + it.each([ + ['ascii', 'x'.repeat(4096)], + ['two-byte', '漢'.repeat(4096)], + ['lone lead surrogate', `\ud800${'a'.repeat(64)}`], + ['lone trail surrogate', `${'a'.repeat(64)}\udfff`], + ['split pair around padding', `\ud83d${'a'.repeat(64)}\ude00`], + ['well-formed astral', '😀'.repeat(2048)], + ['mixed controls', `\x1b]2;${'漢\ud800|\udc00\ud83d'.repeat(512)}`] + ])('round-trips %s exactly', (_label, value) => { + expect(ownRetainedString(value)).toBe(value) + expect(ownRetainedString(value).length).toBe(value.length) + }) + + it('leaves already-flat short strings alone', () => { + // Below V8 SlicedString::kMinLength a slice is already a standalone copy. + for (const value of ['', '\x1b', '\ud800', 'x'.repeat(12)]) { + const parent = `${'y'.repeat(64 * 1024)}${value}` + const short = parent.slice(parent.length - value.length) + expect(ownRetainedString(short)).toBe(short) + } + expect(ownRetainedString('x'.repeat(13))).toBe('x'.repeat(13)) + }) + + it('matches the block copier when Buffer is unavailable', () => { + const original = (globalThis as { Buffer?: unknown }).Buffer + const values = [ + '漢\ud800|\udc00|\ud83d'.repeat(512), + `\x1b]9999;${'x'.repeat(4096)}\ude00`, + '😀'.repeat(2048) + ] + const withBuffer = values.map((value) => ownRetainedString(value)) + + resetOwnRetainedStringCopier() + // Renderer and mobile bundles have no Buffer; the fallback must be byte-identical. + delete (globalThis as { Buffer?: unknown }).Buffer + try { + values.forEach((value, index) => { + expect(ownRetainedString(value)).toBe(value) + expect(ownRetainedString(value)).toBe(withBuffer[index]) + expect(ownRetainedString(value)).toBe(copyUtf16SuffixToOwnedString(value, value.length)) + }) + } finally { + ;(globalThis as { Buffer?: unknown }).Buffer = original + } + }) +}) diff --git a/src/shared/own-retained-string.ts b/src/shared/own-retained-string.ts new file mode 100644 index 00000000000..9c4b1f0715c --- /dev/null +++ b/src/shared/own-retained-string.ts @@ -0,0 +1,50 @@ +import { copyUtf16SuffixToOwnedString } from './owned-utf16-suffix' + +// V8 SlicedString::kMinLength. Shorter slices are already flat, so copying them is pure waste. +const MIN_SLICED_STRING_LENGTH = 13 + +/** Lone surrogates and an unpaired trail must survive the round trip byte-for-byte. */ +const ROUND_TRIP_PROBE = '\ud800a\udfff\u0000\u6f22' + +let copyString: ((value: string) => string) | null = null + +function detectBufferCopier(): ((value: string) => string) | null { + const buffer = (globalThis as { Buffer?: typeof Buffer }).Buffer + if (typeof buffer?.from !== 'function') { + return null + } + try { + if (buffer.from(ROUND_TRIP_PROBE, 'utf16le').toString('utf16le') !== ROUND_TRIP_PROBE) { + return null + } + } catch { + return null + } + return (value) => buffer.from(value, 'utf16le').toString('utf16le') +} + +function resolveCopyString(): (value: string) => string { + if (!copyString) { + // Buffer is absent in the renderer and on mobile; fall back to the code-unit block copier. + copyString = + detectBufferCopier() ?? ((value: string) => copyUtf16SuffixToOwnedString(value, value.length)) + } + return copyString +} + +/** + * Return a standalone copy of a string that outlives the chunk it came from. + * Why: `bigChunk.slice(a, b)` is a V8 SlicedString that pins the whole parent, + * so retaining a few KiB of tail can pin megabytes of already-consumed output. + */ +export function ownRetainedString(value: string): string { + if (value.length < MIN_SLICED_STRING_LENGTH) { + return value + } + return resolveCopyString()(value) +} + +/** Test-only: drop the memoized copier so a fallback path can be exercised. */ +export function resetOwnRetainedStringCopier(): void { + copyString = null +} diff --git a/src/shared/terminal-partial-escape-tail-ground-scan.test.ts b/src/shared/terminal-partial-escape-tail-ground-scan.test.ts new file mode 100644 index 00000000000..fb3c29b6559 --- /dev/null +++ b/src/shared/terminal-partial-escape-tail-ground-scan.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it, vi } from 'vitest' +import { advancePartialEscapeTail, extractPartialEscapeTail } from './terminal-partial-escape-tail' + +describe('partial escape scanning between sequences', () => { + it.each([ + ['ASCII', 'ordinary output '.repeat(8192)], + ['UTF-16', '漢字😀 output '.repeat(8192)] + ])('skips ordinary %s text between completed escapes', (_name, text) => { + const pending = '\x1b[38;5;' + const stream = `\x1b[32m${text}\x1b[0m${text}${pending}` + const charCodeAt = vi.spyOn(String.prototype, 'charCodeAt') + let actual: string + let inspectedCodeUnits: number + try { + actual = extractPartialEscapeTail(stream) + inspectedCodeUnits = charCodeAt.mock.calls.length + } finally { + charCodeAt.mockRestore() + } + + expect(actual).toBe(pending) + expect(inspectedCodeUnits).toBeLessThan(32) + expect(advancePartialEscapeTail(actual, '196mready')).toBe('') + }) + + it.each([ + ['\x1b[38;5;', '196m'], + ['\x1b]2;building', '\x07'], + ['\x1b]2;building\x1b', '\\'], + ['\x1bPqpayload', '\x1b\\'], + ['\x1bPqpayload\x1b', '\\'], + ['\x1b(', 'B'] + ])('preserves %j through every split after ordinary text', (pending, completion) => { + const prefix = `\x1b[32m${'build output '.repeat(128)}\x1b[0m` + for (let split = 0; split <= pending.length; split += 1) { + const first = advancePartialEscapeTail('', prefix + pending.slice(0, split)) + const continued = advancePartialEscapeTail(first, pending.slice(split)) + expect(continued).toBe(pending) + expect(advancePartialEscapeTail(continued, `${completion}ready`)).toBe('') + } + }) + + it('handles aborts before returning to ordinary text', () => { + const text = 'ordinary text '.repeat(128) + for (const abort of ['\x18', '\x1a']) { + for (const pending of ['\x1b[38;', '\x1b]2;title', '\x1bPdata', '\x1b(']) { + expect(extractPartialEscapeTail(`${pending}${abort}${text}\x1b[1;`)).toBe('\x1b[1;') + } + } + expect(extractPartialEscapeTail(`\x1b]2;title\x1b[32m${text}\x1b[1;`)).toBe('\x1b[1;') + }) + + it('keeps one-character echo free of per-code-unit scans and escape searches', () => { + const charCodeAt = vi.spyOn(String.prototype, 'charCodeAt') + const indexOf = vi.spyOn(String.prototype, 'indexOf') + let actual: string + let inspections: number + let searches: number + try { + actual = advancePartialEscapeTail('', 'x') + inspections = charCodeAt.mock.calls.length + searches = indexOf.mock.calls.length + } finally { + charCodeAt.mockRestore() + indexOf.mockRestore() + } + + expect(actual).toBe('') + expect(inspections).toBe(0) + expect(searches).toBe(0) + }) +}) diff --git a/src/shared/terminal-partial-escape-tail.ts b/src/shared/terminal-partial-escape-tail.ts index 6976f0467ec..1a5e2dfc613 100644 --- a/src/shared/terminal-partial-escape-tail.ts +++ b/src/shared/terminal-partial-escape-tail.ts @@ -59,14 +59,20 @@ export function extractPartialEscapeTail(stream: string): string { let state: ScanState = 'ground' let start = 0 for (let i = 0; i < stream.length; i++) { - const code = stream.charCodeAt(i) if (state === 'ground') { - if (code === ESC) { - start = i - state = 'esc' + // Only ESC leaves ground; skip ordinary text without a per-code-unit walk. + // Check the current unit first: on dense escape streams it is usually the + // ESC itself, and indexOf's call + SIMD setup costs more than the compare. + const escape = stream.charCodeAt(i) === ESC ? i : stream.indexOf('\x1b', i) + if (escape === -1) { + return '' } + start = escape + i = escape + state = 'esc' continue } + const code = stream.charCodeAt(i) if ( code === ESC && state !== 'osc' && diff --git a/tests/e2e/golden-core-flows.spec.ts b/tests/e2e/golden-core-flows.spec.ts index bb3b3909c6e..e7494b3cb8c 100644 --- a/tests/e2e/golden-core-flows.spec.ts +++ b/tests/e2e/golden-core-flows.spec.ts @@ -471,12 +471,11 @@ test.describe('New-user golden core flow', () => { .locator('[data-contextual-tour-target="workspace-create-control"]') .first() await expect(createControl).toBeVisible() - await expect(createControl).toHaveAttribute('aria-label', 'Create') + await expect(createControl).toHaveAttribute('aria-label', 'New workspace') const createControlBox = await createControl.boundingBox() expect(createControlBox?.width ?? 0).toBeGreaterThan(0) expect(createControlBox?.height ?? 0).toBeGreaterThan(0) await createControl.click() - await orcaPage.getByRole('menuitem', { name: /^New workspace/ }).click() const workspaceName = `golden-new-${Date.now()}` await completeWorkspaceCreationTour(orcaPage, workspaceName) diff --git a/tests/e2e/helpers/sidebar-project-dialog.ts b/tests/e2e/helpers/sidebar-project-dialog.ts index c9586ccc9ef..746486d1bdb 100644 --- a/tests/e2e/helpers/sidebar-project-dialog.ts +++ b/tests/e2e/helpers/sidebar-project-dialog.ts @@ -1,14 +1,21 @@ import { expect, type Page } from '@stablyai/playwright-test' +// Why scoped: the Landing screen renders its own "Add project" button whenever no +// workspace is open, which is exactly the state these helpers run in. +function sidebarHeaderActions(page: Page) { + return page.locator('[data-sidebar-header-actions]') +} + export async function openSidebarProjectDialog(page: Page): Promise { - await page.getByRole('button', { name: 'Create', exact: true }).click() - await page.getByRole('menuitem', { name: 'Add project', exact: true }).click() + await sidebarHeaderActions(page).getByRole('button', { name: 'Add project', exact: true }).click() await expect(page.getByRole('dialog', { name: /Add a project/i })).toBeVisible() } export async function openSidebarWorkspaceComposer(page: Page): Promise { - await page.getByRole('button', { name: 'Create', exact: true }).click() - const newWorkspaceItem = page.getByRole('menuitem', { name: /^New workspace/ }) - await expect(newWorkspaceItem).toBeVisible() - await newWorkspaceItem.click() + const createButton = sidebarHeaderActions(page).getByRole('button', { + name: 'New workspace', + exact: true + }) + await expect(createButton).toBeVisible() + await createButton.click() }