diff --git a/.github/docker/Dockerfile.ci b/.github/docker/Dockerfile.ci
index 4dc207e6b..90e6b5d13 100644
--- a/.github/docker/Dockerfile.ci
+++ b/.github/docker/Dockerfile.ci
@@ -93,7 +93,14 @@ RUN curl --retry 5 --retry-delay 5 --retry-connrefused -fsSL https://bun.sh/inst
# skillify HOME discovery on 2.1.237, guard/freeze hooks on 2.1.162).
# Bump deliberately, via a PR that runs the PTY gate against the new TUI.
# test/ci-image-cli-pin.test.ts fails the free suite if this pin is removed.
-RUN npm i -g @anthropic-ai/claude-code@2.1.251
+# 2.1.284 is the first pin that recognizes the eval model claude-fable-5-1.
+# 2.1.251 already sent it effort "high", but ran it as an unknown model with
+# a generic system prompt. Users on stable (2.1.280) and latest (2.1.285) get
+# the fable-5-1 profile: its own system prompt, 64k max_tokens and per-turn
+# effort, which only moves the same effort value into the conversation.
+# Census 36626737820 on 2.1.284 looked slower mostly because the API was:
+# its SDK-only judge evals, which never start this CLI, were 25% slower too.
+RUN npm i -g @anthropic-ai/claude-code@2.1.284
# Playwright system deps (Chromium) — needed for browse E2E tests
RUN npx playwright install-deps chromium
diff --git a/.github/scripts/qualify-dia-macos.ts b/.github/scripts/qualify-dia-macos.ts
index f42470429..cdaaa3124 100644
--- a/.github/scripts/qualify-dia-macos.ts
+++ b/.github/scripts/qualify-dia-macos.ts
@@ -1106,6 +1106,7 @@ export async function qualifyDia(isolation: { root: string; configFile: string }
comparisonAttempted = true;
comparisonSource = await runDiaLaunchComparison(account, 'source', { assetRoot: root, executableName, executableSha256: receipt.artifact.executableSha256 },
undefined, deadline - performance.now());
+ if (!comparisonSource) throw new Error('diagnostic_source_launch_returned_no_result');
receipt.launchComparison = { mode: 'launch-only', qualificationCredit: false, source: comparisonSource };
receipt.browsers.source = { stage: 'delegated_comparison', launchReturned: comparisonSource.launchReturned,
timedOut: comparisonSource.timedOut ?? false, error: comparisonSource.error ?? null };
diff --git a/.github/scripts/run-dia-native-qualification.ts b/.github/scripts/run-dia-native-qualification.ts
index ee38c9418..f93c76d62 100644
--- a/.github/scripts/run-dia-native-qualification.ts
+++ b/.github/scripts/run-dia-native-qualification.ts
@@ -429,6 +429,7 @@ async function freshWorker(configFile: string) {
receipt.reason = 'comparison_chromium_control';
launchAttempted = true;
comparisonControl = await runDiaLaunchComparison(account, 'control');
+ if (!comparisonControl) throw new Error('comparison_control_failed');
receipt.comparisonControl = comparisonControl;
if (!comparisonControl.ready || !comparisonControl.cleanup?.confirmed) throw new Error('comparison_control_failed');
receipt.preflight.headlessChromium = true;
diff --git a/.github/workflows/evals-marathon.yml b/.github/workflows/evals-marathon.yml
new file mode 100644
index 000000000..7d3e52326
--- /dev/null
+++ b/.github/workflows/evals-marathon.yml
@@ -0,0 +1,291 @@
+name: Marathon Evals
+# The NON-BLOCKING marathon lane: complete start-to-finish flows (tier
+# 'marathon' in test/helpers/touchfiles-data.ts / describeE2ETier('marathon'))
+# that take longer than a blocking lane's ~12-minute wall. They never run in
+# the PR gate (evals.yml) or the weekly periodic + gate census
+# (evals-periodic.yml); nothing requires this workflow, so a red marathon
+# reports through its own tracking issue without gating any merge. Same engine
+# and FAIL-CLOSED report as the other lanes: one planner manifest, one file per
+# runner, a missing slice artifact is a failure. Always fresh: no result reuse.
+on:
+ schedule:
+ - cron: '0 12 * * 6' # Saturday 12:00 UTC, clear of the Monday periodic census
+ workflow_dispatch:
+
+concurrency:
+ group: evals-marathon
+ cancel-in-progress: true
+
+env:
+ IMAGE: ghcr.io/${{ github.repository }}/ci
+ EVALS_PROFILE: full
+ EVALS_FRESH: "1"
+ EVALS_CACHE_PURPOSE: marathon
+
+jobs:
+ build-image:
+ runs-on: ubicloud-standard-8
+ timeout-minutes: 15
+ permissions:
+ contents: read
+ packages: write
+ outputs:
+ image-tag: ${{ steps.meta.outputs.tag }}
+ steps:
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
+
+ - id: meta
+ # Keep in sync with evals.yml and evals-periodic.yml — key on Dockerfile + lockfile only
+ # (package.json's version field would bust the key on every ship).
+ # Byte-identity pinned by test/ci-image-tag-binding.test.ts.
+ run: echo "tag=${{ env.IMAGE }}:${{ hashFiles('.github/docker/Dockerfile.ci', 'bun.lock', 'patches/**') }}" >> "$GITHUB_OUTPUT"
+
+ - uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4
+ with:
+ registry: ghcr.io
+ username: ${{ github.actor }}
+ password: ${{ secrets.GITHUB_TOKEN }}
+
+ - name: Check if image exists
+ id: check
+ run: |
+ if docker manifest inspect ${{ steps.meta.outputs.tag }} > /dev/null 2>&1; then
+ echo "exists=true" >> "$GITHUB_OUTPUT"
+ else
+ echo "exists=false" >> "$GITHUB_OUTPUT"
+ fi
+
+ - if: steps.check.outputs.exists == 'false'
+ run: cp package.json bun.lock .github/docker/ && cp -R patches .github/docker/patches
+
+ # Registry cache export needs a docker-container builder — the default
+ # `docker` driver hard-errors on cache-to.
+ - if: steps.check.outputs.exists == 'false'
+ uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4
+
+ - if: steps.check.outputs.exists == 'false'
+ uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7
+ with:
+ context: .github/docker
+ file: .github/docker/Dockerfile.ci
+ push: true
+ # Cron-triggered in the base repo only, so cache export is always safe here.
+ cache-from: type=registry,ref=${{ env.IMAGE }}:buildcache
+ cache-to: type=registry,ref=${{ env.IMAGE }}:buildcache,mode=max
+ tags: |
+ ${{ steps.meta.outputs.tag }}
+ ${{ env.IMAGE }}:latest
+
+
+ plan-slices:
+ runs-on: ubicloud-standard-8
+ timeout-minutes: 10
+ permissions:
+ contents: read
+ outputs:
+ slices: ${{ steps.matrix.outputs.slices }}
+ timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
+ steps:
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
+ with:
+ persist-credentials: false
+
+ - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
+ with:
+ bun-version: 1.4.0
+
+ # One marathon file per runner: a 1-second budget never packs two
+ # recorded files together.
+ - name: Emit run manifest (ALL marathon tests)
+ env:
+ EVALS_ALL: "1"
+ run: EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --emit-plan /tmp/marathon-plan/manifest.json --slice-budget 1 --jobs 1
+
+ - name: Derive the executor matrix from the plan
+ id: matrix
+ run: |
+ echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
+ echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/marathon-plan/manifest.json)" >> "$GITHUB_OUTPUT"
+
+ - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: marathon-plan
+ path: /tmp/marathon-plan/manifest.json
+ retention-days: 30
+
+ eval-slices:
+ runs-on: ubicloud-standard-8
+ needs: [build-image, plan-slices]
+ env:
+ EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-marathon-${{ matrix.slice }}
+ # One marathon file per runner; the job timeout is the plan's supervised
+ # worst case plus 20 minutes setup/upload.
+ timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
+ permissions:
+ contents: read
+ packages: read
+ container:
+ image: ${{ needs.build-image.outputs.image-tag }}
+ credentials:
+ username: ${{ github.actor }}
+ password: ${{ secrets.GITHUB_TOKEN }}
+ options: --user runner
+ strategy:
+ fail-fast: false
+ max-parallel: 8
+ matrix:
+ slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
+ steps:
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
+ with:
+ # Full history: files with SELF-derived selection (the LLM-judge
+ # map, routing) walk git at module load, and selection is
+ # fail-closed on git errors — a shallow checkout crashed those
+ # shards on the lane's first live run ("ambiguous argument
+ # 'main...HEAD'"). The manifest still governs WHICH shards run.
+ fetch-depth: 0
+ persist-credentials: false
+
+ - name: Fix bun temp
+ uses: ./.github/actions/fix-bun-temp
+
+ - name: Restore deps
+ uses: ./.github/actions/restore-deps
+
+ - run: bun run build
+
+ # Any slice can host a PTY test — seed + registration run
+ # unconditionally (idempotent; mirrors evals.yml's sliced lane). The
+ # register composite carries the fail-fast dangling-symlink/frontmatter
+ # verification loop — this lane previously LACKED it, so a moved skill
+ # target surfaced as a silent "Unknown command" + wedged PTY session.
+ - name: Seed claude interactive config
+ uses: ./.github/actions/seed-claude-config
+ with:
+ anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
+
+ - name: Register gstack skills for PTY tests
+ uses: ./.github/actions/register-gstack-skills
+
+ - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
+ with:
+ name: marathon-plan
+ path: /tmp/marathon-plan
+
+ - name: Run marathon slice ${{ matrix.slice }}
+ env:
+ ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
+ OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
+ GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
+ PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
+ EVALS_JOBS: "1"
+ EVALS_CONCURRENCY: "2"
+ GSTACK_EVAL_DIR: /tmp/marathon-slice-results
+ run: EVALS_TIER=marathon bun run scripts/test-paid-shards.ts --tier marathon --plan /tmp/marathon-plan/manifest.json --slice ${{ matrix.slice }}
+
+ - name: Upload slice results
+ if: always()
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: marathon-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
+ path: /tmp/marathon-slice-results
+ retention-days: 90
+
+ - name: Upload native capture evidence
+ if: always()
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: native-captures-${{ env.EVALS_RUN_ID }}
+ include-hidden-files: true
+ path: |
+ ~/.gstack/projects/*/e2e-runs
+ ~/.gstack/projects/*/evals/qa-callers
+ ~/.gstack-dev/e2e-runs
+ ~/.gstack-dev/evals/qa-callers
+ if-no-files-found: ignore
+ retention-days: 90
+
+ - name: Upload shard logs on failure
+ if: failure()
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: marathon-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
+ include-hidden-files: true
+ # The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
+ # runner's spool lands THERE, not /tmp — the original /tmp glob
+ # uploaded nothing and a red slice's diagnostics were unreachable.
+ path: |
+ /home/runner/.cache/gstack-paid-shard-*.log
+ /tmp/gstack-paid-shard-*.log
+ if-no-files-found: ignore
+ retention-days: 30
+
+ report:
+ runs-on: ubicloud-standard-2
+ needs: [plan-slices, eval-slices]
+ # !cancelled(): the report must run (and FAIL) when an executor died — a
+ # missing slice artifact reading as green is the class this lane kills —
+ # but a cancelled run stops here.
+ if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
+ timeout-minutes: 10
+ permissions:
+ contents: read
+ issues: write
+ steps:
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
+ with:
+ persist-credentials: false
+
+ - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
+ with:
+ bun-version: 1.4.0
+
+ - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
+ with:
+ name: marathon-plan
+ path: /tmp/marathon-report
+
+ - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
+ with:
+ pattern: marathon-slice-*
+ path: /tmp/marathon-report
+
+ - name: Reconcile slices against the manifest (fail-closed)
+ id: reconcile
+ if: always()
+ run: |
+ set +e
+ EVALS_TIER=marathon bun --no-install run scripts/test-paid-shards.ts --tier marathon --report /tmp/marathon-report | tee /tmp/report.txt
+ # PIPESTATUS[0], NOT $?: the default step shell has no pipefail.
+ echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
+
+ # One tracking issue for the whole lane (never one per week).
+ - name: Upsert tracking issue on failure
+ if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ run: |
+ set -euo pipefail
+ TITLE="Weekly marathon evals: red lane needs triage"
+ BODY_FILE=/tmp/issue-body.md
+ {
+ echo "Automated weekly marathon report (non-blocking lane) — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
+ echo
+ echo "- reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
+ echo "- marathon slices job: ${{ needs.eval-slices.result }}"
+ echo
+ echo '```'
+ tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
+ echo '```'
+ } > "$BODY_FILE"
+ EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
+ if [ -n "$EXISTING" ]; then
+ gh issue comment "$EXISTING" --repo "$GITHUB_REPOSITORY" --body-file "$BODY_FILE"
+ echo "commented on #$EXISTING"
+ else
+ gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
+ fi
+
+ - name: Fail the workflow when reconciliation failed
+ if: always() && (steps.reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success')
+ run: exit 1
diff --git a/.github/workflows/evals-periodic.yml b/.github/workflows/evals-periodic.yml
index 8272c4abb..707c2221c 100644
--- a/.github/workflows/evals-periodic.yml
+++ b/.github/workflows/evals-periodic.yml
@@ -4,8 +4,13 @@ name: Periodic Evals
# tests can't rot invisibly — the class where the autoplan-dual-voice E2E was
# silently broken for months until a lucky local diff selected it. Engine:
# scripts/test-paid-shards.ts (the same runner local eval:bg:periodic uses):
-# one planner manifest, 6 ordinary slices plus an overlay slice, and a FAIL-CLOSED report — a slice
-# whose artifact never landed is a failure, not an absence. The gate-census
+# one planner manifest packed by recorded durations into as many ~9-minute
+# executors as the work needs (one file, or a tightly packed group, per
+# runner; overlays share one final slice), and a FAIL-CLOSED report — a slice
+# whose artifact never landed is a failure, not an absence. The matrix size
+# and job timeout come from the plan, so they cannot drift from the census.
+# Full end-to-end flows run in the non-blocking marathon lane
+# (evals-marathon.yml), never here. The gate-census
# job is the weekly EVALS_ALL backstop for the gate tier (PR lanes are
# diff-billed, so without it the full gate census might never execute
# anywhere); the hollow-shard guard (exit 0 + zero executed tests under
@@ -14,9 +19,16 @@ on:
schedule:
- cron: '0 6 * * 1' # Monday 6 AM UTC (ci-image prebuilds at 4 AM)
workflow_dispatch:
+ inputs:
+ redispatch_of:
+ description: 'Run id this run re-dispatches (the one INFRA/INCOMPLETE-only re-dispatch; set by the report job)'
+ type: string
+ default: ''
+# A re-dispatch runs in its own group so it never cancels the run that
+# dispatched it; both runs are reported.
concurrency:
- group: evals-periodic
+ group: evals-periodic${{ inputs.redispatch_of && format('-redispatch-{0}', inputs.redispatch_of) || '' }}
cancel-in-progress: true
env:
@@ -84,6 +96,11 @@ jobs:
timeout-minutes: 10
permissions:
contents: read
+ outputs:
+ periodic_slices: ${{ steps.periodic-matrix.outputs.slices }}
+ periodic_timeout_minutes: ${{ steps.periodic-matrix.outputs.timeout_minutes }}
+ gate_slices: ${{ steps.gate-matrix.outputs.slices }}
+ gate_timeout_minutes: ${{ steps.gate-matrix.outputs.timeout_minutes }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -96,7 +113,13 @@ jobs:
- name: Emit run manifest (ALL periodic tests minus reasoned excludes)
env:
EVALS_ALL: "1"
- run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slices 7
+ run: EVALS_TIER=periodic bun --no-install run scripts/test-paid-shards.ts --tier periodic --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 24
+
+ - name: Derive the periodic executor matrix from the plan
+ id: periodic-matrix
+ run: |
+ echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
+ echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -107,7 +130,13 @@ jobs:
- name: Emit gate census manifest (ALL gate tests)
env:
EVALS_ALL: "1"
- run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slices 7 --skip-judges
+ run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/gate-census-plan/manifest.json --slice-budget 540 --jobs 2 --skip-judges --max-parallel 16
+
+ - name: Derive the gate census executor matrix from the plan
+ id: gate-matrix
+ run: |
+ echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
+ echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/gate-census-plan/manifest.json)" >> "$GITHUB_OUTPUT"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
@@ -120,9 +149,10 @@ jobs:
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-eval-slices-${{ matrix.slice }}
- # Seven slices retain every registered case and retry. The complete
- # census needs at most 244m40s per slice, plus 20 minutes setup/upload.
- timeout-minutes: 360
+ # The planner packs ~9 minutes of recorded work per slice; the job timeout
+ # is its supervised worst case (every shard at its wall) plus 20 minutes
+ # setup/upload, computed from the same manifest the slices execute.
+ timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.periodic_timeout_minutes) }}
permissions:
contents: read
packages: read
@@ -134,9 +164,11 @@ jobs:
options: --user runner
strategy:
fail-fast: false
- max-parallel: 8
+ # Every planned slice starts at once; test/evals-workflow-wiring.test.ts
+ # fails when the live plan outgrows this cap.
+ max-parallel: 24
matrix:
- slice: [1, 2, 3, 4, 5, 6, 7]
+ slice: ${{ fromJSON(needs.plan-slices.outputs.periodic_slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -174,7 +206,7 @@ jobs:
name: paid-plan
path: /tmp/paid-plan
- - name: Run slice ${{ matrix.slice }}/7
+ - name: Run periodic slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -189,7 +221,7 @@ jobs:
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
- name: paid-slice-${{ matrix.slice }}
+ name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/paid-slice-results
retention-days: 90
@@ -207,11 +239,13 @@ jobs:
if-no-files-found: ignore
retention-days: 90
- - name: Upload shard logs on failure
- if: failure()
+ # always(), not failure(): a failed behavior trial is a verdict and no
+ # longer reds its runner, but its full log is the diagnosis evidence.
+ - name: Upload shard logs
+ if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
- name: paid-slice-${{ matrix.slice }}-logs
+ name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
@@ -231,8 +265,8 @@ jobs:
needs: [build-image, plan-slices]
env:
EVALS_RUN_ID: ci-${{ github.run_id }}-${{ github.run_attempt }}-gate-census-${{ matrix.slice }}
- # Seven slices need at most 272m each, plus 20 minutes setup/upload.
- timeout-minutes: 352
+ # Supervised worst case of the packed plan plus 20 minutes setup/upload.
+ timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.gate_timeout_minutes) }}
permissions:
contents: read
packages: read
@@ -243,11 +277,11 @@ jobs:
password: ${{ secrets.GITHUB_TOKEN }}
options: --user runner
strategy:
- # Four file workers total, each retaining two in-file case workers.
+ # Two file workers per slice, each retaining two in-file case workers.
fail-fast: false
- max-parallel: 4
+ max-parallel: 16
matrix:
- slice: [1, 2, 3, 4, 5, 6, 7]
+ slice: ${{ fromJSON(needs.plan-slices.outputs.gate_slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -272,13 +306,13 @@ jobs:
name: gate-census-plan
path: /tmp/gate-census-plan
- - name: Run gate census slice ${{ matrix.slice }}/7
+ - name: Run gate census slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
PLAYWRIGHT_BROWSERS_PATH: /opt/playwright-browsers
- EVALS_JOBS: "1"
+ EVALS_JOBS: "2"
EVALS_CONCURRENCY: "2"
GSTACK_EVAL_DIR: /tmp/gate-census-results
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/gate-census-plan/manifest.json --slice ${{ matrix.slice }}
@@ -287,7 +321,7 @@ jobs:
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
- name: gate-census-${{ matrix.slice }}
+ name: gate-census-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/gate-census-results
retention-days: 90
@@ -312,12 +346,16 @@ jobs:
# missing slice artifact reading as green is the class this lane kills —
# but a cancelled run stops here.
if: ${{ !cancelled() && needs.plan-slices.result == 'success' }}
- timeout-minutes: 10
+ timeout-minutes: 15
permissions:
contents: read
- # The failure notification below upserts a tracking issue via
- # `gh api /issues` — gated by the issues permission.
+ # The notification below upserts (or closes) a tracking issue via
+ # `gh issue` — gated by the issues permission.
issues: write
+ # Pass-rate history downloads earlier weekly runs' trial-outcomes.
+ actions: read
+ outputs:
+ redispatch: ${{ steps.verdict.outputs.redispatch }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -332,11 +370,12 @@ jobs:
name: paid-plan
path: /tmp/paid-report
+ # One directory per attempt-scoped slice artifact (no merge): shard
+ # records never overwrite each other and the first attempt decides.
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
- pattern: paid-slice-[0-9]*
+ pattern: paid-slice-*
path: /tmp/paid-report
- merge-multiple: true
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
@@ -347,7 +386,6 @@ jobs:
with:
pattern: gate-census-[0-9]*
path: /tmp/gate-census-report
- merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
@@ -369,34 +407,114 @@ jobs:
EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --report /tmp/gate-census-report | tee /tmp/gate-report.txt
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
- # A red weekly lane nobody must action is waste — upsert ONE tracking
- # issue (never a new issue per week) with the reconciliation output, so
- # failures have an owner-visible artifact with history in one place.
- - name: Upsert tracking issue on failure
- if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
+ - name: Stamp trial history series
+ if: always()
+ run: |
+ for file in /tmp/paid-report/trial-outcomes.jsonl /tmp/gate-census-report/trial-outcomes.jsonl; do
+ if [ -f "$file" ]; then bun --no-install run scripts/eval-trial-series.ts "$file"; fi
+ done
+
+ - name: Upload trial outcomes for pass-rate history
+ if: always()
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: trial-outcomes-periodic-a${{ github.run_attempt }}
+ path: |
+ /tmp/paid-report/trial-outcomes.jsonl
+ if-no-files-found: ignore
+ retention-days: 90
+
+ - name: Upload gate census trial outcomes for pass-rate history
+ if: always()
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: trial-outcomes-gate-census-a${{ github.run_attempt }}
+ path: |
+ /tmp/gate-census-report/trial-outcomes.jsonl
+ if-no-files-found: ignore
+ retention-days: 90
+
+ # Weekly pass-rate gate over the last 10 weekly runs (drift, rule cases
+ # behaving like behavior, quarantine exit/expiry/cap). Fails closed when
+ # history cannot be fetched.
+ - name: Pass-rate history gate
+ id: pass-rates
+ if: always()
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ run: |
+ set +e
+ bun run eval:pass-rates --gate --runs 10 > /tmp/pass-rates.txt 2>&1
+ echo "exit=$?" >> "$GITHUB_OUTPUT"
+ cat /tmp/pass-rates.txt
+
+ # UC-E1 (approved): a run whose every red verdict is machine-classified
+ # INFRA or INCOMPLETE may be re-dispatched ONCE as a new run.
+ - name: Classify the census verdict
+ id: verdict
+ if: always()
+ env:
+ REDISPATCH_OF: ${{ inputs.redispatch_of }}
+ PERIODIC_EXIT: ${{ steps.reconcile.outputs.exit }}
+ GATE_EXIT: ${{ steps.gate-reconcile.outputs.exit }}
+ run: |
+ eligible() { # $1 exit, $2 report dir
+ [ "$1" = "0" ] && return 0
+ jq -e '.version == 2 and .verdict.redispatchEligible == true' "$2/collector-outcomes.json" >/dev/null 2>&1
+ }
+ if [ -z "$REDISPATCH_OF" ] && { [ "$PERIODIC_EXIT" != "0" ] || [ "$GATE_EXIT" != "0" ]; } \
+ && eligible "$PERIODIC_EXIT" /tmp/paid-report && eligible "$GATE_EXIT" /tmp/gate-census-report; then
+ echo "redispatch=true" >> "$GITHUB_OUTPUT"
+ else
+ echo "redispatch=false" >> "$GITHUB_OUTPUT"
+ fi
+
+ # A red weekly lane nobody must action is waste — upsert ONE tracking
+ # issue (never a new issue per week) with the headline and failure block
+ # of both lanes, and close it on the next green run.
+ - name: Upsert tracking issue on failure
+ if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ REDISPATCH: ${{ steps.verdict.outputs.redispatch }}
+ REDISPATCH_OF: ${{ inputs.redispatch_of }}
run: |
set -euo pipefail
TITLE="Weekly periodic evals: red lane needs triage"
BODY_FILE=/tmp/issue-body.md
+ RUN_URL="${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
{
- echo "Automated weekly report — run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
+ echo "Automated weekly report — run: ${RUN_URL}"
+ if [ -n "$REDISPATCH_OF" ]; then echo; echo "This run is the one INFRA/INCOMPLETE re-dispatch of run ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${REDISPATCH_OF}; both runs are reported."; fi
+ if [ "$REDISPATCH" = "true" ]; then echo; echo "Every red verdict is machine-classified INFRA/INCOMPLETE: re-dispatching once as a new run (EVAL_POLICY.infraRedispatch). This run stays red and reported."; fi
echo
- echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }}"
- echo "- periodic slices job: ${{ needs.eval-slices.result }}"
- echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }}"
- echo "- gate census job: ${{ needs.gate-census.result }}"
+ echo "- periodic reconciliation exit: ${{ steps.reconcile.outputs.exit }} (slices job: ${{ needs.eval-slices.result }})"
+ echo "- gate census reconciliation exit: ${{ steps.gate-reconcile.outputs.exit }} (census job: ${{ needs.gate-census.result }})"
+ echo "- pass-rate history gate exit: ${{ steps.pass-rates.outputs.exit }}"
+ echo
+ echo "### Periodic lane"
+ cat /tmp/paid-report/report-summary.md 2>/dev/null || echo "(no periodic report summary)"
+ echo
+ echo "### Gate census"
+ cat /tmp/gate-census-report/report-summary.md 2>/dev/null || echo "(no gate census report summary)"
+ echo
+ echo "### Pass-rate history (ACTION REQUIRED)"
+ echo '```'
+ { grep -E 'ACTION REQUIRED|history unavailable' /tmp/pass-rates.txt || echo "(no pass-rate alarms)"; } | sed 's/@/@\xe2\x80\x8b/g' | head -c 6000
+ echo '```'
+ echo
+ echo "Full reconciliation output
"
echo
echo '```'
- tail -c 6000 /tmp/report.txt 2>/dev/null || echo "(no reconciliation output)"
+ tail -c 6000 /tmp/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no reconciliation output)"
echo '```'
echo
echo '```'
- tail -c 6000 /tmp/gate-report.txt 2>/dev/null || echo "(no gate census reconciliation output)"
+ tail -c 6000 /tmp/gate-report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo "(no gate census reconciliation output)"
echo '```'
+ echo " "
echo
- echo "Exclusion policy: test/helpers/periodic-exclude-data.ts (every entry needs reason + tracking; removal re-activates the file next week)."
+ echo "Policy: EVAL_POLICY and CASE_QUARANTINE in test/helpers/periodic-exclude-data.ts; history: \`bun run eval:pass-rates\`."
} > "$BODY_FILE"
EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
if [ -n "$EXISTING" ]; then
@@ -406,6 +524,37 @@ jobs:
gh issue create --repo "$GITHUB_REPOSITORY" --title "$TITLE" --body-file "$BODY_FILE"
fi
+ - name: Close the tracking issue on a green run
+ if: always() && steps.reconcile.outputs.exit == '0' && steps.gate-reconcile.outputs.exit == '0' && steps.pass-rates.outputs.exit == '0' && needs.eval-slices.result == 'success' && needs.gate-census.result == 'success'
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ REDISPATCH_OF: ${{ inputs.redispatch_of }}
+ run: |
+ set -euo pipefail
+ TITLE="Weekly periodic evals: red lane needs triage"
+ EXISTING=$(gh issue list --repo "$GITHUB_REPOSITORY" --state open --search "in:title \"$TITLE\"" --json number --jq '.[0].number // empty')
+ if [ -n "$EXISTING" ]; then
+ NOTE="Green weekly run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
+ if [ -n "$REDISPATCH_OF" ]; then NOTE="${NOTE} (the INFRA re-dispatch of run ${REDISPATCH_OF}, which stays red and reported)"; fi
+ gh issue close "$EXISTING" --repo "$GITHUB_REPOSITORY" --comment "$NOTE"
+ fi
+
- name: Fail the workflow when reconciliation failed
- if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
+ if: always() && (steps.reconcile.outputs.exit != '0' || steps.gate-reconcile.outputs.exit != '0' || steps.pass-rates.outputs.exit != '0' || needs.eval-slices.result != 'success' || needs.gate-census.result != 'success')
run: exit 1
+
+ # The one INFRA/INCOMPLETE re-dispatch (UC-E1). Its own job so the report
+ # job keeps no actions:write; the new run's concurrency group differs, so it
+ # never cancels this run.
+ redispatch:
+ runs-on: ubicloud-standard-2
+ needs: report
+ if: ${{ !cancelled() && needs.report.outputs.redispatch == 'true' }}
+ timeout-minutes: 5
+ permissions:
+ actions: write
+ steps:
+ - name: Re-dispatch the weekly census once
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ run: gh workflow run evals-periodic.yml --repo "$GITHUB_REPOSITORY" --ref "$GITHUB_REF_NAME" -f redispatch_of="$GITHUB_RUN_ID"
diff --git a/.github/workflows/evals.yml b/.github/workflows/evals.yml
index b0d97ad2c..ee2875132 100644
--- a/.github/workflows/evals.yml
+++ b/.github/workflows/evals.yml
@@ -125,6 +125,9 @@ jobs:
timeout-minutes: 10
permissions:
contents: read
+ outputs:
+ slices: ${{ steps.matrix.outputs.slices }}
+ timeout_minutes: ${{ steps.matrix.outputs.timeout_minutes }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -137,11 +140,24 @@ jobs:
with:
bun-version: 1.4.0
+ # Planner-side reuse: restore this PR's newest receipt store (the report
+ # job saves one merged store per run) and ship ONE filtered set with the
+ # plan, so every trial of a panel sees the same receipts and a newer FAIL
+ # blocks any older PASS for the same inputs.
+ - name: Restore this PR's verified judge and E2E results
+ if: github.event_name == 'pull_request'
+ uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
+ with:
+ path: /tmp/gstack-eval-input-cache
+ key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-plan
+ restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
+
- name: Emit run manifest
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
env:
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
- run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 7
+ EVALS_CACHE_DIR: ${{ github.event_name == 'pull_request' && '/tmp/gstack-eval-input-cache' || '' }}
+ run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slice-budget 540 --jobs 2 --max-parallel 16
- name: Emit validation-phase manifest
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
@@ -152,27 +168,36 @@ jobs:
run: |
bun --no-install -e '
import { mkdirSync, writeFileSync } from "node:fs";
- import { buildRunManifest, collectPaidTestFiles } from "./scripts/test-paid-shards.ts";
+ import { buildRunManifest, collectPaidTestFiles, restrictManifestSelection } from "./scripts/test-paid-shards.ts";
const phase = process.env.VALIDATION_PHASE;
if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase");
const cookieBehavior = phase === "cookie-behavior";
const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"]
: cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"]
: collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality"));
- const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceCount: 6, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
+ const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceBudgetMs: 540000, jobs: 2, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) });
- if (phase === "cookie-quality") manifest.selection = { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] };
- if (cookieBehavior) manifest.selection = { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] };
- manifest.selectionReason = phase + " validation subset; " + manifest.selectionReason;
+ const subset = phase === "cookie-quality" ? { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] }
+ : cookieBehavior ? { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] } : null;
+ const restricted = subset ? restrictManifestSelection(manifest, subset, "outside the " + phase + " validation subset") : manifest;
+ restricted.selectionReason = phase + " validation subset; " + manifest.selectionReason;
mkdirSync("/tmp/paid-plan", { recursive: true });
- writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(manifest, null, 2) + "\n");
- console.log(phase + ": " + manifest.entries.filter(entry => entry.status === "planned").length + " planned shards");
+ writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(restricted, null, 2) + "\n");
+ console.log(phase + ": " + restricted.entries.filter(entry => entry.status === "planned").length + " planned shards");
'
+ - name: Derive the executor matrix from the plan
+ id: matrix
+ run: |
+ echo "slices=$(jq -c '[range(1; .sliceCount + 1)]' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
+ echo "timeout_minutes=$(jq -e '.plan.ciTimeoutMinutes' /tmp/paid-plan/manifest.json)" >> "$GITHUB_OUTPUT"
+
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-plan
- path: /tmp/paid-plan/manifest.json
+ path: |
+ /tmp/paid-plan/manifest.json
+ /tmp/paid-plan/receipts
retention-days: 30
eval-slices:
@@ -184,14 +209,11 @@ jobs:
# (image already published), but a newer push's cancel-in-progress stops
# it instead of letting a superseded run finish its paid slices first.
if: ${{ !cancelled() && needs.build-image.result == 'success' && needs.plan-slices.result == 'success' }}
- # Aggregate spawn-concurrency budget: 6 slices x EVALS_JOBS=2 x
- # EVALS_CONCURRENCY=2 = 24 concurrent tests lane-wide (the old matrix's
- # 40-way per row queued claude session STARTUP behind 39 siblings and ate
- # per-test budgets — the documented timeout-flake family). Tune with
- # parity data before raising.
- # The complete gate census needs at most 242 minutes per slice; keep
- # 20 minutes for setup/upload without preempting configured retries.
- timeout-minutes: 265
+ # The planner packs ~9 minutes of recorded work per slice (EVALS_JOBS=2 x
+ # EVALS_CONCURRENCY=2 per runner, never the old 40-way per-row fan-out
+ # that queued claude session STARTUP behind 39 siblings). The job timeout
+ # is the plan's supervised worst case plus 20 minutes setup/upload.
+ timeout-minutes: ${{ fromJSON(needs.plan-slices.outputs.timeout_minutes) }}
permissions:
contents: read
packages: read
@@ -203,9 +225,9 @@ jobs:
options: --user runner
strategy:
fail-fast: false
- max-parallel: 6
+ max-parallel: 16
matrix:
- slice: [1, 2, 3, 4, 5, 6, 7]
+ slice: ${{ fromJSON(needs.plan-slices.outputs.slices) }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -244,17 +266,15 @@ jobs:
name: paid-plan
path: /tmp/paid-plan
- # Only this PR's receipts are eligible. No base-branch or cross-PR restore
- # prefix; every receipt also verifies exact inputs and its original age.
- - name: Restore this PR's verified judge results
- if: github.event_name == 'pull_request'
- uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
- with:
- path: /tmp/gstack-eval-input-cache
- key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
- restore-keys: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-
+ # Receipts come only from the plan (this PR's store, filtered once by the
+ # planner); new receipts land beside the slice results and the report
+ # merges them into the next store.
+ - name: Seed this slice's receipts from the plan
+ run: |
+ mkdir -p /tmp/paid-slice-results/receipts
+ if [ -d /tmp/paid-plan/receipts ]; then cp -a /tmp/paid-plan/receipts/. /tmp/paid-slice-results/receipts/; fi
- - name: Run slice ${{ matrix.slice }}/7
+ - name: Run slice ${{ matrix.slice }}
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
@@ -263,40 +283,19 @@ jobs:
EVALS_JOBS: "2"
EVALS_CONCURRENCY: "2"
GSTACK_EVAL_DIR: /tmp/paid-slice-results
- EVALS_CACHE_DIR: /tmp/gstack-eval-input-cache
+ EVALS_CACHE_DIR: /tmp/paid-slice-results/receipts
EVALS_CACHE_REPOSITORY: ${{ github.repository }}
EVALS_CACHE_PR: ${{ github.event.pull_request.number }}
EVALS_CACHE_RUNTIME_ID: ${{ needs.build-image.outputs.runtime-id }}
run: EVALS_TIER=gate bun run scripts/test-paid-shards.ts --tier gate --plan /tmp/paid-plan/manifest.json --slice ${{ matrix.slice }}
- - name: Find finalized passing receipts
- id: receipts
- if: ${{ !cancelled() && github.event_name == 'pull_request' }}
- run: |
- # Only a producer publishes. A later reuse-only slice must not become
- # the newest prefix match and hide another slice's newly earned pass.
- for receipt in /tmp/gstack-eval-input-cache/*.json; do
- [ -f "$receipt" ] || continue
- if jq -e --arg run "$GITHUB_RUN_ID/$GITHUB_RUN_ATTEMPT" '.proof.source.runId == $run' "$receipt" >/dev/null 2>&1; then
- echo 'present=true' >> "$GITHUB_OUTPUT"
- break
- fi
- done
-
- # An unrelated failing case does not discard already verified passes.
- # Failed/retried/partial attempts never become receipts in the first place.
- - name: Save verified judge results for this PR
- if: ${{ !cancelled() && steps.receipts.outputs.present == 'true' }}
- uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
- with:
- path: /tmp/gstack-eval-input-cache
- key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-${{ matrix.slice }}
-
+ # Attempt-scoped: a re-run attempt's trials are reported under that
+ # attempt and never replace (or collide with) the first attempt's.
- name: Upload slice results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
- name: paid-slice-${{ matrix.slice }}
+ name: paid-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
path: /tmp/paid-slice-results
retention-days: 90
@@ -316,11 +315,13 @@ jobs:
# The spooled per-shard full logs — a red weekly/PR lane three weeks
# later needs more than a summary line.
- - name: Upload shard logs on failure
- if: failure()
+ # always(), not failure(): a failed behavior trial is a verdict and no
+ # longer reds its runner, but its full log is the diagnosis evidence.
+ - name: Upload shard logs
+ if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
- name: paid-slice-${{ matrix.slice }}-logs
+ name: paid-logs-slice-${{ matrix.slice }}-a${{ github.run_attempt }}
include-hidden-files: true
# The Fix-bun-temp step points TMPDIR at /home/runner/.cache, so the
# runner's spool lands THERE, not /tmp — the original /tmp glob
@@ -365,11 +366,13 @@ jobs:
name: paid-plan
path: /tmp/paid-report
+ # One directory per attempt-scoped slice artifact (no merge): shard
+ # records can never overwrite each other, and the report keeps the
+ # first attempt's verdict.
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
- pattern: paid-slice-[0-9]*
+ pattern: paid-slice-*
path: /tmp/paid-report
- merge-multiple: true
- name: Reconcile slices against the manifest (fail-closed)
id: reconcile
@@ -382,17 +385,50 @@ jobs:
# (caught by the ship review army; the wiring test now pins this).
echo "exit=${PIPESTATUS[0]}" >> "$GITHUB_OUTPUT"
+ - name: Stamp trial history series
+ if: always()
+ run: |
+ if [ -f /tmp/paid-report/trial-outcomes.jsonl ]; then
+ bun --no-install run scripts/eval-trial-series.ts /tmp/paid-report/trial-outcomes.jsonl
+ fi
+
+ # One merged receipt store per run: the plan's shipped set, every slice's
+ # new pass receipts, and the report's panel and negative receipts. Saved
+ # last, so the next planner restores it as the newest prefix match.
+ - name: Merge this run's receipts
+ if: always() && github.event_name == 'pull_request'
+ run: |
+ bun --no-install run scripts/e2e-shard-reuse.ts merge /tmp/gstack-eval-input-cache \
+ /tmp/paid-report/receipts /tmp/paid-report/report-receipts /tmp/paid-report/paid-slice-*/receipts
+
+ - name: Save this PR's verified judge and E2E results
+ if: always() && github.event_name == 'pull_request'
+ uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
+ with:
+ path: /tmp/gstack-eval-input-cache
+ key: eval-input-v1-${{ github.repository_id }}-pr-${{ github.event.pull_request.number }}-${{ github.run_id }}-${{ github.run_attempt }}-merged
+
- name: Upload reconciliation output for the comment job
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
- name: report-verdict
+ name: report-verdict-a${{ github.run_attempt }}
path: |
/tmp/report.txt
/tmp/paid-report/collector-outcomes.json
+ /tmp/paid-report/report-summary.md
if-no-files-found: ignore
retention-days: 30
+ - name: Upload trial outcomes for pass-rate history
+ if: always()
+ uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+ with:
+ name: trial-outcomes-pr-a${{ github.run_attempt }}
+ path: /tmp/paid-report/trial-outcomes.jsonl
+ if-no-files-found: ignore
+ retention-days: 90
+
- name: Fail the workflow when reconciliation failed
if: steps.reconcile.outputs.exit != '0'
run: exit 1
@@ -419,18 +455,13 @@ jobs:
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
- pattern: paid-slice-[0-9]*
- path: /tmp/paid-report
- merge-multiple: true
-
- - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
- with:
- name: report-verdict
+ name: report-verdict-a${{ github.run_attempt }}
path: /tmp/verdict
continue-on-error: true
- # Verified counts come from the read-only report job, not repo code in
- # this write-token job. Keeps the
+ # Every count, verdict and failure line comes from the read-only report
+ # job's collector-outcomes v2 (panelVerdict() ran there); this job runs
+ # no repo code and never recomputes a verdict. Keeps the
# "## E2E Evals" marker so the upsert keeps updating the same comment.
# Runs even when reconciliation failed — a red lane on the PR is the point.
- name: Post PR comment
@@ -439,13 +470,14 @@ jobs:
RECONCILE_EXIT: ${{ needs.slices-report.outputs.reconcile-exit }}
run: |
# shellcheck disable=SC2086,SC2059
- RESULTS=$(find /tmp/paid-report -name '*.json' ! -name 'manifest.json' ! -name 'slice-*.json' ! -name '_partial*' 2>/dev/null | sort)
- TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0"
+ TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; EXECUTED=0; REUSED=0; COST="0"
SUITE_LINES=""
VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json
if ! jq -e '
. as $summary |
- .version == 1 and (.files | type == "array") and (.totals | type == "object") and
+ .version == 2 and (.files | type == "array") and (.totals | type == "object") and
+ (.headline | type == "array") and (.failures | type == "array") and (.panels | type == "array") and
+ (.verdict.verdict == "GREEN" or .verdict.verdict == "RED") and
([.files[] | .total == (.passed + .failed + .manual_accepted) and
(.total == (.executed + .reused)) and
([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and
@@ -454,100 +486,69 @@ jobs:
all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key]))
' "$VERIFIED" >/dev/null 2>&1; then
VERIFIED=""
- echo 'Verified collector summary unavailable; manual acceptance is unavailable/unverified.'
+ echo 'Verified report summary unavailable; no verdict, counts or manual acceptance can be shown.'
fi
+ HEADLINE='(no verified report headline)'
+ FAILURES=""
if [ -n "$VERIFIED" ]; then
- while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do
+ while IFS=$'\t' read -r _FILE T P F M _FLAKY EX RE _ATTEMPTS C TIER SHARD; do
[ "$T" -eq 0 ] && continue
TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F))
- MANUAL=$((MANUAL + M)); FLAKY=$((FLAKY + FL))
+ MANUAL=$((MANUAL + M))
EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE))
COST=$(echo "$COST + $C" | bc)
STATUS_ICON="✅"
[ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored"
[ "$F" -gt 0 ] && STATUS_ICON="❌"
- [ "$F" -eq 0 ] && [ "$M" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED")
- else
- for f in $RESULTS; do
- if ! jq -e '.total_tests' "$f" >/dev/null 2>&1; then
- echo "Skipping malformed JSON: $f"
- continue
- fi
- # FINAL-attempt accounting: eval-store keeps EVERY retry attempt
- # as its own record (that's the flake telemetry), so counting raw
- # records marks a pass-on-retry as a failure and inflates totals.
- # Group by test name and judge the LAST record. Retry metadata
- # includes both passing and failing final outcomes; show it separately.
- # Guarded: a file with total_tests but a null/non-array `tests`
- # passes the -e probe, the group_by then fails, and an empty $T
- # would abort the whole step under bash -e ([ "" -eq 0 ] is an
- # error) — killing the comment on exactly the corrupted-artifact
- # runs where the red evidence matters (claude adversarial).
- STATS=$(jq -r '[.tests | group_by(.name)[] | last] as $final | "\($final | length) \([$final[] | select(.passed)] | length) \([$final[] | select(.passed | not)] | length) \(.flaky_retries // [] | length) \([$final[] | select(.execution != "reused")] | length) \([$final[] | select(.execution == "reused")] | length)"' "$f" 2>/dev/null) || { echo "Skipping malformed tests[] in: $f"; continue; }
- read -r T P F FL EX RE <<< "$STATS"
- [ -z "$T" ] && { echo "Skipping malformed tests[] in: $f"; continue; }
- C=$(jq -r '.total_cost_usd // 0' "$f")
- TIER=$(jq -r '.tier // "unknown"' "$f")
- SHARD=$(jq -r '.shard // "-"' "$f")
- [ "$T" -eq 0 ] && continue
- TOTAL=$((TOTAL + T))
- PASSED=$((PASSED + P))
- FAILED=$((FAILED + F))
- FLAKY=$((FLAKY + FL))
- EXECUTED=$((EXECUTED + EX))
- REUSED=$((REUSED + RE))
- COST=$(echo "$COST + $C" | bc)
- STATUS_ICON="✅"
- [ "$F" -gt 0 ] && STATUS_ICON="❌"
- [ "$F" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
- SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | unverified | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
- done
+ # Report-sanitized lines (no @-mentions, one capped line each), fenced here.
+ HEADLINE=$(jq -r '.headline[]' "$VERIFIED")
+ FAILURES=$(jq -r '.failures[]' "$VERIFIED")
fi
COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.'
STATUS="✅ PASS"
- if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ]; then STATUS="❌ FAIL"; fi
+ if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ] \
+ || { [ -n "$VERIFIED" ] && [ "$(jq -r '.verdict.verdict' "$VERIFIED")" != "GREEN" ]; }; then STATUS="❌ FAIL"; fi
if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi
- if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (manual acceptance unavailable/unverified)'; fi
+ if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (verified report unavailable)'; fi
BODY="## E2E Evals: ${STATUS}
- **${PASSED} automated passed / ${TOTAL} final results** | **${FAILED} failed, ${MANUAL} manual accepted (unscored; no score-cache credit)** | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY")
+ \`\`\`
+ ${HEADLINE}
+ \`\`\`
+
+ **${EXECUTED} executed, ${REUSED} reused** rule/judge records | **${MANUAL} manual accepted (unscored; no score-cache credit)** | **\$${COST}** rule/judge cost | reconcile exit: ${RECONCILE_EXIT:-missing}
${COVERAGE}
+ Rule and judge shards
+
| Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost |
|-------|------------------|-----------------|----------|--------|--------|------|
$(echo -e "$SUITE_LINES")
+
Fail-closed reconciliation
\`\`\`
- $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null || echo '(no reconciliation output)')
+ $(tail -c 4000 /tmp/verdict/report.txt 2>/dev/null | sed 's/@/@\xe2\x80\x8b/g' || echo '(no reconciliation output)')
\`\`\`
---
- *Sliced lane: declared PR profile or broad fallback via scripts/test-paid-shards.ts (planner → 6 executors → fail-closed report). Reused scores retain their original provenance and expiry.*"
+ *Sliced lane: planner → duration-packed executors → fail-closed report. Behavior cases run a pre-registered 3-trial panel (PASS at 2/3 with no contract violation); a PASS 2/3 is shown with its failed trial, never as a clean pass. Reused results retain their original provenance and expiry.*"
- if [ "$FAILED" -gt 0 ]; then
- FAILURES=""
- for f in $RESULTS; do
- if ! jq -e '.failed' "$f" >/dev/null 2>&1; then continue; fi
- if [ -n "$VERIFIED" ]; then
- FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false and (has("manual_review") | not))][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
- else
- FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
- fi
- FAILURES="${FAILURES}${FAILS}\n"
- done
+ if [ -n "$FAILURES" ]; then
BODY="${BODY}
- ### Failures
- $(echo -e "$FAILURES")"
+ ### Failures and split verdicts
+ \`\`\`
+ ${FAILURES}
+ \`\`\`"
fi
COMMENT_ID=$(gh api repos/${{ github.repository }}/issues/${{ github.event.pull_request.number }}/comments \
diff --git a/.github/workflows/free-tests.yml b/.github/workflows/free-tests.yml
index ed8d7ec50..decf06938 100644
--- a/.github/workflows/free-tests.yml
+++ b/.github/workflows/free-tests.yml
@@ -137,6 +137,25 @@ jobs:
GSTACK_CSO_DOCKER_TESTS: "1"
DOCKER_HOST: unix:///var/run/docker.sock
+ typecheck:
+ runs-on: ubuntu-24.04
+ timeout-minutes: 10
+ steps:
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
+ with:
+ persist-credentials: false
+ - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2
+ with:
+ bun-version: 1.4.0
+ - name: Install dependencies
+ run: bun install --frozen-lockfile --ignore-scripts
+ - name: Typecheck product code (zero errors)
+ run: bun run typecheck
+ - name: Test-code type-debt ratchet
+ run: bun run typecheck:test
+ - name: CSO source formatting
+ run: bun run format:cso:check
+
free-suite:
needs: free-plan
runs-on: ubicloud-standard-8
@@ -304,19 +323,21 @@ jobs:
# gate is merge-blocking without a separate branch-protection migration.
free-tests:
if: always()
- needs: [free-suite, cso-macos-launcher, cso-windows-launcher, cso-docker-integration]
+ needs: [free-suite, typecheck, cso-macos-launcher, cso-windows-launcher, cso-docker-integration]
runs-on: ubuntu-24.04
timeout-minutes: 5
steps:
- - name: Require the free suite and every CSO platform gate
+ - name: Require the free suite, typecheck, and every CSO platform gate
env:
FREE_SUITE_RESULT: ${{ needs.free-suite.result }}
+ TYPECHECK_RESULT: ${{ needs.typecheck.result }}
CSO_MACOS_RESULT: ${{ needs.cso-macos-launcher.result }}
CSO_WINDOWS_RESULT: ${{ needs.cso-windows-launcher.result }}
CSO_DOCKER_RESULT: ${{ needs.cso-docker-integration.result }}
run: |
set -eu
test "$FREE_SUITE_RESULT" = success
+ test "$TYPECHECK_RESULT" = success
test "$CSO_MACOS_RESULT" = success
test "$CSO_WINDOWS_RESULT" = success
test "$CSO_DOCKER_RESULT" = success
diff --git a/AGENTS.md b/AGENTS.md
index d86702f34..350e43237 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -149,7 +149,8 @@ When fixing failures or preparing `/ship`, follow this order:
public events in free regressions, including negative controls, before paying
for another agent run. Check behavior and acknowledgments; match exact prose
only when that prose is the contract. Do not lower thresholds, increase model
- budgets, skip cases, or rejudge a failure to manufacture a pass.
+ budgets, skip cases, or rejudge a failure to manufacture a pass. A
+ pre-registered fixed panel is not rejudging.
For policy or validation repairs, exercise the actual registered callback with
representative native input and assert that it uses the helper’s result.
When renderer or parser failures recur at the same boundary, verify the
@@ -209,7 +210,16 @@ When fixing failures or preparing `/ship`, follow this order:
result and pending permission state; diagnose a blocked actor before waiting
through its deadline. Preserve cancellation separately from a test verdict.
Skipped or unstarted cases
- do not satisfy coverage; preserve configured retries and every attempt.
+ do not satisfy coverage; preserve every attempt. Paid evals never retry. Each
+ case's kind (`E2E_KINDS`) fixes its trials before the run: `rule` one trial;
+ `behavior` a panel of 3 independent trials, PASS at >= 2 with no contract
+ violation; `judge` 3 samples on one output, gated on the mean against the
+ unchanged threshold. Never add trials, samples or dispatches after seeing a
+ result, never change a kind to change a verdict without pass-rate evidence,
+ and report every trial. Quarantine follows `CASE_QUARANTINE`'s entry and exit
+ rules only (`EVAL_POLICY`, `docs/TESTING_INTERNALS.md`). A census whose every
+ red is machine-classified INFRA or INCOMPLETE may be re-dispatched once as a
+ new run; report both runs.
7. Prove all known repairs with focused tests, including affected paid cases.
Rerun a failed case only after a concrete repair or a demonstrated launch
correction. Run the remaining required selected evaluations on the integrated
@@ -235,11 +245,16 @@ When fixing failures or preparing `/ship`, follow this order:
```bash
bun install # install dependencies
+bun run typecheck # strict tsc over product code; must report zero errors
+bun run typecheck:test # test-code type-debt ratchet (new diagnostics fail; --write-baseline locks in fixes)
+bun run format:cso # format lib/cso/*.ts (format:cso:check is the CI gate)
bun run test:quick # fast measured free subset for edit feedback (not acceptance)
bun run test # complete free suite via the strict shard runner (no API spend)
bun run test:ubicloud # same suite on an ephemeral 16-vCPU Ubicloud VM (needs UBICLOUD_API_KEY)
bun run eval:bg:pr # changed fast live probes + selected judges, with explicit deferrals
bun run eval:bg:release # fresh complete gate + periodic live coverage
+bun run eval:pass-rates # per-case trial pass rates (Wilson), drift and quarantine alarms (--case, --gate)
+bun run scripts/test-paid-shards.ts --tier periodic --list --slice-budget 540 --jobs 2 # CI slice plan preview (free)
bun run test:windows # curated Windows-safe subset (runs on windows-latest)
bun run build # generate docs + compile binaries
bun run gen:skill-docs # regenerate SKILL.md files from templates
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 0dc933a55..ad47e885f 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,69 @@
# Changelog
+## [1.91.12.0] - 2026-10-01
+
+**Weekly evals finish in minutes, not hours, and a red now means something.**
+**Two real crash bugs fixed, and product code typechecks clean in CI.**
+
+The weekly paid eval run took 2 hours 45 minutes on Sept 28, almost all of it one timed-out test retried. It now runs every test on its own machine within a 9-minute budget, and a single long case runs one case per process. Automatic retries are gone. Tests that grade a live model's choice run three trials at once and pass on two; promises users rely on (asks before deciding, leaves git alone, no writes in plan mode) fail on any single bad trial. `$B connect --supervise` finally restarts a crashed browser, compiled `/cso` installs can witness runtime-tested assertions again, and a required `typecheck` job keeps that class of bug out.
+
+### The numbers that matter
+
+Source: the Sept 28 weekly census (run 36385945043) and the final proof census on this branch (run 36633323521). `bun run scripts/test-paid-shards.ts --tier periodic --slice-budget 540 --jobs 2 --list` prints the current plan.
+
+| Measure | Before | After |
+| --- | ---: | ---: |
+| Weekly periodic census wall clock | 2h 45m | 11m 42s (gate census alongside: 10m 22s) |
+| Longest planned slice | 160 min (one test, twice) | ~10 min |
+| Automatic retries on paid evals | up to 2 per file | 0 |
+| Product-code type errors | 103 on v1.91.8.0 (no check) | 0, required in `free-tests` |
+| `lib/cso` longest source line | 2,159 chars | 785 (a string literal) |
+
+The biggest change is honesty. With about 240 live cases, a retry used to hide a failing test; now every trial is recorded, `bun run eval:pass-rates` shows each case's pass rate with a confidence range, and a case that slides gets flagged by its history instead of passing on a lucky rerun.
+
+### Fewer rotating reds
+
+Across 11 lanes the PR eval lane failed 6-8 of its 125 records per run, a different handful each time. A census of 1,827 attempts traced most of it to two sources, and this release attacks both instead of retrying:
+
+- **Runs that ran out of time.** Passing runs used 80-92% of their budgets, and the ones that timed out took 20-50% more steps, not slower steps. The heaviest cases now start at the gate they test, from recorded setup and recorded subagent results (docsync faults, shared-libs review, QA callers, Review Army), and land at roughly 35-60% of unchanged budgets.
+- **Bookkeeping the model forgot.** `gstack-qa-evidence` now enforces the checkpoint before every next probe, fills revision/runtime/cwd/learning itself, rejects placeholders, replay-only learning, missing evidence rows and evidence observed on an older input snapshot, prints report links, timing and any declared-but-unrun probes, and answers `--help`. `/deslop-shared-libs` runs every git read through `bin/gstack-safe-git`, which always applies the safety flags.
+
+### What this means for contributors
+
+Run `bun run typecheck` and `bun run typecheck:test` before you push; both are free and take seconds. A red paid run now prints a headline and one line per failure with its cause and a rerun command. New paid evals need a kind in `E2E_KINDS`: see "Add a paid eval" in CONTRIBUTING.md.
+
+### Itemized changes
+
+#### Fixed
+- `$B connect --supervise` respawned with a block-scoped env that no longer existed, so every restart threw and the supervisor gave up after five tries. The headed env is one helper used by connect and respawn, and the loop has behavioral tests.
+- Compiled `/cso` installs called an unimported `join` when launching the assertion-witness child, breaking runtime-tested witnessing for every installed user.
+- Browser-only `/qa` runs had no stated way to build the evidence file, whose rows only accept functional captures; the shared rule now says to materialize an empty evidence list with the checkpoints named in limits, matching `/qa-only`. The fix loop had spent its last minute on it and timed out.
+- `/office-hours` asks its goal question unless the user already chose a mode, then reads that mode's section before its first question; skipping both produced forcing questions with an empty recommendation. `/design-consultation` asks the memorable-thing question on its own after Q1 instead of packing it into Q1's call.
+- Free-form eval judges (docs, outcome, posture) could return JSON broken by an unescaped quote in their reasoning; they now use structured output.
+- `/qa` checkpoint receipts now print the report link for their `exploration-NNN.json` file; reports had been linking `.qa-evidence/NNN` capture folders as checkpoints instead.
+- `/review` Review Army passes checklists to specialists by path and runs web research alongside dispatch (a 12-line N+1 review went from 300 s to 212 s), and the design-lite pass always runs its detector probe; reviews had reported the detector absent without probing in 5 of 6 captured trials.
+- `/design-consultation` opens with one decision (confirm the context and choose research), not a confirm-only question; `/document-release` defines its /ship-owned inputs, exact steps and JSON result.
+- `/review` workflow ambiguities (smoke clock vs required revalidation, setup authority, plan-completion gate, findings record), `/office-hours` builder mode not loading its brainstorm section, `/sync-gbrain` Step 4 helper arguments and write path, `/plan-ceo-review` expansion framing and pacing menus, `/plan-design-review` with no designer API key, and `/deslop-shared-libs` one-file-per-turn reads.
+- Eval detectors that graded wording or step order now grade outcomes: eng batching, CEO split-overflow, mode routing, section-loading stale-fill, outside-voice-disabled attribution, design focus menus, and PTY permission dialogs with cropped titles.
+- Harness races and adapter gaps found by the proof runs: plan seeding accepted a stale empty input box when the CLI repainted after recording its reply, the third-party-actions recorder fixture lost every failure record, the autoplan dual-voice check could not read framed subagent reports from newer Claude Code, and the HOLD SCOPE routing check judged the skill's own defer/keep menu as its rigor decision, the outside-disabled check missed a correctly attributed quote of the pre-existing review record, and the plan-review judge was not told its reason length bound on the field it writes.
+
+#### Changed
+- Paid evals: one test file or case per machine within a 540-second slice budget, planned from recorded per-tier and per-case durations; case sharding for plan, design, review-army, shared-libs, shared-libs-paths, ship-docsync and qa-callers.
+- Verdict policy: no retries; `rule` cases fail on any failed trial, `behavior` cases pass on 2 of 3 parallel trials with contract assertions still strict, `judge` entries average 3 samples against unchanged thresholds. One panel-verdict function feeds the report, PR comment, weekly issue and pass-rate history. A census whose every red is infrastructure is re-dispatched once, and both runs are reported.
+- New non-blocking weekly `evals-marathon.yml` lane for full start-to-finish flows: the full `/office-hours` workflow (a focused design-draft case replaces it in the weekly lane) and the full `/plan-ceo-review` split-overflow run, which took 8 to 20 minutes on its own.
+- The CI image pins Claude Code 2.1.284, the first version that recognizes the eval model `claude-fable-5-1` and runs it with the same profile users get. Both versions send effort "high"; a census that looked slower on 2.1.284 was mostly slower API responses (its SDK-only judges, which never start the CLI, were 25% slower too), and nine previously slow cases pass on 2.1.284 within unchanged budgets.
+- `lib/cso/*.ts` is formatted with pinned Prettier; minified transpile output is byte-identical except three canonicalized regex flag orders.
+- The duplicate dispatch-only `ship-docsync` case is removed; `ship-docsync-completion` asserts the same on the same fixture.
+
+#### Added
+- `tsconfig.json`, `bun run typecheck` (strict, zero product errors) and `bun run typecheck:test` (test-code diagnostic ratchet), both in the required `free-tests` check, plus `format:cso:check`.
+- `E2E_KINDS`, `BEHAVIOR_WHY`, `EVAL_POLICY` and a data-driven `CASE_QUARANTINE` (entry below 95% per trial over 10 trials, exit at 97%, 10% cap, 8-week expiry, never for product defects), and `CASE_CI_EXCLUDE` for CI-unrunnable cases.
+- `bun run eval:pass-rates` with Wilson intervals, per-input-identity series and a weekly drift gate; `--case --trials N` for local diagnosis.
+
+#### For contributors
+- Open PRs touching `lib/cso` should run `bun run format:cso` before rebasing.
+- Builds on the typecheck work in #2447, contributed by @laddtnov.
+- Coordinated with #2994 (v1.91.8.0), which retired the never-green finding-count evals this wave had been repairing.
## [1.91.11.0] - 2026-09-30
gstack now looks up its state folder one way everywhere, and the five most copy-pasted or oversized parts of the codebase each have a single owner. Before, about 50 scripts, hooks and libraries each resolved the state folder with their own rule, and the rules disagreed. If you set `GSTACK_HOME`, `GSTACK_STATE_DIR` or `GSTACK_STATE_ROOT`, telemetry, analytics, update-check snoozes, the egress ledger and hook logs now all land in the folder you chose. Nothing is moved for you. Run `~/.claude/skills/gstack/bin/gstack-paths --explain` to see the active folder and whether `~/.gstack` still holds older state; [docs/state-root.md](docs/state-root.md) has the move recipe.
diff --git a/CLAUDE.md b/CLAUDE.md
index a25c25693..db5326d4c 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -19,6 +19,8 @@ bun run test:e2e # run E2E tests only (diff-based, ~$4.20/run max)
bun run test:e2e:all # run ALL E2E tests regardless of diff
bun run eval:select # show which tests would run based on current diff
bun run dev # run CLI in dev mode, e.g. bun run dev goto https://example.com
+bun run typecheck # strict tsc over product code (zero errors required)
+bun run typecheck:test # test-code type-debt ratchet
bun run build # gen docs + compile binaries
bun run gen:skill-docs # regenerate SKILL.md files from templates
bun run skill:check # health dashboard for all skills
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index 24874b6ff..07496edbb 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -16,6 +16,22 @@ bin/dev-setup # activate dev mode
> **Full clone vs shallow.** The README's user-facing install uses `--depth 1` for speed. As a contributor, use a full clone (no `--depth` flag) — you'll need history for `git log`, `git blame`, `git bisect`, and reviewing PRs against earlier versions. If you already have a `--depth 1` clone from following the README, promote it to a full clone with `git fetch --unshallow`.
+### First free check (no API key, no browser)
+
+```bash
+bun install --frozen-lockfile
+bun run typecheck # expect no output and exit 0 (about a second)
+bun run typecheck:test # expect "test typecheck ratchet: N known diagnostics, none new."
+```
+
+`typecheck` covers product code (`browse/src`, `lib`, `scripts`, `bin`, `hosts`, and the other
+entries in `tsconfig.json`) and must stay at zero errors. `typecheck:test` holds test code to the
+committed `scripts/typecheck-test-baseline.json`: a new or repeated diagnostic fails and names
+the file, TS code and message; fixing diagnostics also fails until you lock the smaller allowance
+in with `bun run typecheck:test --write-baseline`. Editing `lib/cso/*.ts`? Run
+`bun run format:cso` before committing; CI runs `format:cso:check`. All three run in the required
+`free-tests` check.
+
Now edit any `SKILL.md`, invoke it in Claude Code (e.g. `/review`), and see your changes live. When you're done developing:
```bash
@@ -212,9 +228,51 @@ gate and periodic censuses run fresh weekly and on manual
dispatch of `evals-periodic.yml`; `bun run eval:bg:release` runs both locally.
Some broad behavioral failures will therefore be found after the PR gate.
+Blocking paid lanes (the PR gate and the weekly periodic + gate census) aim to
+finish in about 12 minutes including setup. The planner packs recorded wall
+times (`scripts/paid-test-durations.json`, per tier) into as many ~9-minute
+runners as the work needs, one file or a tightly packed group each; files whose
+cases are short but whose total is long run one case per runner. Matrix size and
+job timeout come from that plan. Preview it for free with
+`bun run scripts/test-paid-shards.ts --tier periodic --list --slice-budget 540 --jobs 2`.
+Complete start-to-finish flows belong to the `marathon` tier
+(`describeE2ETier('marathon')`), which runs only in the non-blocking
+`evals-marathon.yml` lane (weekly and on dispatch) and never gates a merge.
+
+Verdicts: paid evals never retry. Each case's kind in `E2E_KINDS`
+(`test/helpers/touchfiles-data.ts`) fixes its trials before the run, from the
+constants in `EVAL_POLICY` (`test/helpers/periodic-exclude-data.ts`):
+
+- `rule` (the default): one trial; any failed assertion fails the case. Use it
+ when nothing stochastic decides the verdict, or when the verdict checks a
+ contract the product must meet every run (no writes in plan mode, a question
+ before a decision, a skill-mandated step, no leaked secret).
+- `behavior`: a panel of 3 independent trials run as parallel case shards,
+ PASS at 2 or more with no contract violation (`expectContract()`). Use it only
+ when a live model choice decides the verdict and an occasional deviation is
+ acceptable product behavior; the one-line reason goes in `BEHAVIOR_WHY`.
+- `judge`: an LLM judge scoring a fixed input; 3 samples of the same prompt,
+ gated on the per-dimension mean (booleans on a majority) against the
+ unchanged threshold. An erroring sample fails the panel and is never resampled.
+
+A timed-out, crashed or infrastructure-failed trial counts as a failed trial and
+is reported with its class; a missing trial makes the case INCOMPLETE, which
+fails the lane. A 2-of-3 pass is reported as `PASS 2/3` with the failed trial's
+cause, never as a clean pass. Case budgets and thresholds never change with
+this policy. Quarantine (`CASE_QUARANTINE`) and history are described in
+`docs/TESTING_INTERNALS.md`; `bun run eval:pass-rates --case ` shows a
+case's per-trial pass rate with its Wilson interval.
+
CI enables verified first-attempt reuse for 16 workflow quality judges for
-24 hours within the same PR. The cookie workflow's custom input, the other 11
-quality cases and all dynamic agent cases stay fresh. Local runs stay fresh unless
+24 hours within the same PR. The cookie workflow's custom input and the other 11
+quality cases stay fresh. PR-profile E2E shards that run once (no retry, so the
+pass is provably a first attempt) reuse a pass from the same PR when every
+consumed input is byte-identical: the test's import closure, every tracked file
+its registered cases' touchfiles and the global touchfiles match, the runner and
+workflow, the child's EVALS_/GSTACK_/CLAUDE_/ANTHROPIC_ environment (secret
+presence only), the CI image and Claude CLI version (`scripts/e2e-shard-reuse.ts`).
+A computed case registration or a touchfile pattern matching nothing keeps the
+shard fresh. The weekly census, marathon and release lanes never reuse. Local runs stay fresh unless
the complete scoped cache and runtime configuration is supplied. The key includes complete prompt bytes, generated inputs,
fixtures, runner/rubric code, installed dependencies, model settings and runtime.
The current assertions validate a reused score again. Records retain the original
@@ -376,7 +434,7 @@ When E2E tests run, they produce machine-readable artifacts in `~/.gstack-dev/`:
bun run eval:list # list all eval runs (turns, duration, cost per run)
bun run eval:compare # compare two runs — shows per-test deltas + Takeaway commentary
bun run eval:summary # aggregate stats + per-test efficiency averages across runs
-bun run eval:flake-rank # rank tests by flake signal: retried passes first, then failure rate (--json, --dir, --since-days)
+bun run eval:pass-rates # per-case trial pass rates + Wilson intervals from recent weekly runs (--case, --runs, --dir, --backfill, --json, --gate); eval:flake-rank is an alias
```
**Detached runs for agents and long suites.** When an agent (or you, for a run
@@ -424,7 +482,9 @@ Override the judge model per run with `GSTACK_EVAL_MODEL_JUDGE`:
- **Completeness** — Are all commands, flags, and usage patterns documented?
- **Actionability** — Can the agent execute tasks using only the information in the doc?
-Each dimension is scored 1-5. Threshold: every dimension must score **≥ 4**. There's also a regression test that compares generated docs against the hand-maintained baseline from `origin/main` — generated must score equal or higher.
+Each dimension is scored 1-5 by a panel of 3 samples of the same prompt, drawn
+concurrently; each dimension's panel mean must meet that judge's threshold (≥ 4
+for most dimensions; see each case). An erroring sample fails the panel. There's also a regression test that compares generated docs against the hand-maintained baseline from `origin/main` — generated must score equal or higher.
```bash
# Needs ANTHROPIC_API_KEY in .env — included in bun run test:evals
@@ -444,6 +504,24 @@ fails, add the named path to the named key and check selection with
`bun run scripts/test-paid-shards.ts --tier gate --profile pr --list`. The rule is a lower bound: a fixture
path the test builds at runtime is not visible to it, so add such paths to the key by hand.
+### Add a paid eval
+
+1. **Test file.** Write the case in a paid test file, registered with a literal
+ name (`testIfSelected('', ...)`), grading the outcome (files, git
+ state, native questions, exit status) rather than wording, unless the step
+ itself is the contract. Wrap contract assertions in `expectContract()`.
+2. **Touchfiles.** Add `'': [...]` to `E2E_TOUCHFILES`; `bun test
+ test/touchfiles.test.ts` names any missing closure path.
+3. **Tier.** Add it to `E2E_TIERS`: `gate` for cheap contracts every PR needs,
+ `periodic` for long or model-quality cases, `marathon` for complete flows.
+4. **Kind.** Add it to `E2E_KINDS` (`rule` unless a live model choice may
+ acceptably deviate; then `behavior` plus a `BEHAVIOR_WHY` line).
+ `bun test test/eval-kinds.test.ts` prints the literal to add.
+5. **PR profile.** If a PR should run it, add it to `scripts/test-pr-profile.ts`
+ and check `bun run scripts/test-paid-shards.ts --tier gate --profile pr --list`.
+6. **Try the panel locally.** `bun run scripts/test-paid-shards.ts --tier
+ --case --trials 3` runs the same panel CI runs, before you push.
+
### CI
A GitHub Action (`.github/workflows/skill-docs.yml`) generates all hosts on pushes to main and on PRs, then rejects tracked differences and nonignored untracked output. Generation errors also fail the job. Optional ignored host caches are not compared against Git.
diff --git a/TODOS.md b/TODOS.md
index b2c9f5cf5..fdd692382 100644
--- a/TODOS.md
+++ b/TODOS.md
@@ -2,6 +2,37 @@
## NEXT PRIORITY
+### P1: paid-eval follow-ups from the v1.91.12.0 proof censuses (filed 2026-09-29)
+
+- **Thin budgets on slow API days** — on Claude Code 2.1.284, review-army-perf
+ (274 of 300 s) and the ship-docsync fault cases (250-263 of 285 s) sit at
+ 88-93% of their budgets; a slow-API census can time them out on either CLI
+ version. Make those skills faster rather than raising budgets. Effort M.
+- **Recurring reds to repair, not rerun** — `plan-design-review-plan-mode`
+ (one ~250 s thinking block before its single write; times out at 300 s on
+ 2.1.251 in every recent run) and the HOLD SCOPE
+ routing case when its next brief happens not to name the mode (see the
+ handoff item below). Effort M each.
+- **`/plan-ceo-review` skips its Step 0E mode handoff** — 0 of 15 answered
+ samples sent the required `Mode: ; approved decisions: …` chat after
+ the mode answer, across four wording repairs (none shipped). The model writes
+ the handoff in its reasoning and later says it was "sent above". A prose fix
+ won't reach it; this needs a mechanism outside the prompt (a hook or a
+ tool-result gate). The HOLD SCOPE routing case fails whenever the handoff is
+ skipped and nothing else names the posture in time. Effort M.
+- **Pre-push hook tests hang behind some shard neighbors** — on the free-suite
+ plan for dfe5e733, `test/redact-prepush-hook.test.ts` timed out 6 of 28 tests
+ at 30 s in shard 12 on two attempts (the hook process was still running and
+ killed as dangling); it passes alone in 9 s and in the next plan's shard 12.
+ One of the 29 files that ran before it only in the failing plan (browse CDP/
+ stealth/tab tests, pty-workspace-trust, heredoc-pipe-deadlock among them)
+ leaves state the hook's blocking path waits on. Reproduce with that shard's
+ plan under xvfb and GSTACK_EXPECT_BINARIES=1. Effort S.
+- **Let pass-rate history decide the rest** — every census on this branch had
+ a different handful of single-trial reds. Once `eval:pass-rates` has 10 weekly
+ trials per case, apply the CASE_QUARANTINE entry rule instead of chasing one
+ run at a time. Effort S.
+
### P2/P3: impeccable interop deferrals (filed 2026-09-08, from the CEO + eng reviews of docs/designs/IMPECCABLE_INTEROP.md)
Each item was weighed during the review and deferred with a reason; none blocks
@@ -144,9 +175,10 @@ wave"). Each was explicitly deferred with rationale, not dropped:
- **#2443 AskUserQuestion numbering redesign** — real mismatch (brief letters
vs host-rendered numbers), but a prompt-behavior redesign that shifts eval
baselines; needs its own PR with baseline refresh. Effort S.
-- **#2447 typecheck infra** — tsconfig + repo-wide typecheck script + latent
- type fixes. High-value, repo-wide blast radius, own PR with bake time.
- Effort M. Re-derive on current main (several of its fixes landed since).
+- ~~**#2447 typecheck infra**~~ — superseded: the audit fix wave (v1.91.12.0)
+ added `tsconfig.json`, `bun run typecheck` (zero product errors) and the
+ `typecheck:test` ratchet inside the required `free-tests` check, reusing
+ #2447's fixes where they still applied.
- **#2492 per-project Chromium profile** — needs an on-disk migration story
for the machine-wide profile default and SingletonLock scoping. Effort M.
- **#2286 `triggers:` frontmatter** — the Claude Code router never reads the
@@ -833,13 +865,16 @@ and `test/dx-selected-navigation-ap.test.ts`. One shared table run once against
only after `engFirstReviewAUQ` checks native completion once at entry; today each branch gates it
separately, so the change alters a paid verdict and needs its own paid run.
-### P3: Re-pin the four remaining claude-opus-4-7 paid files
+### P3: Re-pin the five remaining claude-opus-4-7 paid files
**What:** The 2026-09 audit moved seven paid evals to the default capture model (`resolveEvalModel('capture')`).
`skill-e2e-design`, `skill-e2e-office-hours-phase4`, `skill-e2e-plan-prosons` and `skill-e2e-plan` keep
`claude-opus-4-7` because six cases failed on the default model in one run (plan-design-review-plan-mode timeout,
office-hours-phase4-fork format, plan-review-prosons-neutral-neg missing output, plan-ceo-review-selective and
-plan-eng-review 600 s timeouts, plan-ceo-review-expansion-energy posture score 3). They measure an old model.
+plan-eng-review 600 s timeouts, plan-ceo-review-expansion-energy posture score 3). `skill-e2e-qa-bugs` returned
+to `claude-opus-4-7` after `qa-b6-static` timed out on the default model in two of three runs (census 36597762183
+and a targeted local rerun): each time the stream stopped mid-message, with no pending tool, right after the model
+found the disabled submit button, and emitted nothing until the 300 s case deadline. They measure an old model.
**Re-entry:** fix the prompt, budget or rubric so each case passes on the default model in one run, then drop the pin.
@@ -870,7 +905,9 @@ macOS/Aside, no physical iPhone), so the weekly periodic lane scheduled them as
green shards that verified nothing. They are now in `PERIODIC_CI_EXCLUDE`
(`test/helpers/periodic-exclude-data.ts`): `codex-e2e`, `codex-e2e-sol-scope`,
`codex-e2e-shared-libs`, `codex-e2e-recommendation-substance`,
-`skill-e2e-outside-voice`, `skill-e2e-aside`, `skill-e2e-ios-device`. They still
+`skill-e2e-outside-voice`, `skill-e2e-aside`, `skill-e2e-ios-device`. One case
+inside a case-sharded file is excluded the same way through `CASE_CI_EXCLUDE`:
+`test/skill-e2e-design.test.ts#design-review-fix` (needs Aside). They still
run locally on a machine that has the CLI or device.
**Re-entry:** the CLI or device is available in the CI image. First target:
diff --git a/VERSION b/VERSION
index 0d9e9c091..df5d57be8 100644
--- a/VERSION
+++ b/VERSION
@@ -1 +1 @@
-1.91.11.0
+1.91.12.0
diff --git a/agents-digest/gstack-AGENTS.md b/agents-digest/gstack-AGENTS.md
index 97fa07c66..31c8fac5e 100644
--- a/agents-digest/gstack-AGENTS.md
+++ b/agents-digest/gstack-AGENTS.md
@@ -1,4 +1,4 @@
-# gstack digest v1.91.11.0 — regenerate/re-copy after upgrading gstack
+# gstack digest v1.91.12.0 — regenerate/re-copy after upgrading gstack
Behavioral rules from gstack (https://github.com/garrytan/gstack), compressed
for agent hosts without a full skill install. The full skills add workflows,
diff --git a/autoplan/bin/phase-publication-hook.ts b/autoplan/bin/phase-publication-hook.ts
index ce9649b06..aa3ea4dbb 100644
--- a/autoplan/bin/phase-publication-hook.ts
+++ b/autoplan/bin/phase-publication-hook.ts
@@ -13,13 +13,16 @@ const PHASES = ['ceo', 'design', 'dx', 'eng', 'tasks'] as const;
type Phase = typeof PHASES[number];
type Event = ClaudeParentPublicEvent;
type Use = Event & { kind: 'use' };
+type Tool = Extract;
+type Turn = Extract;
+const isUse = (e: Event): e is Use => e.kind === 'use';
const number: Record = { ceo: 1, design: 2, dx: 2.5, eng: 3, tasks: 4 };
const object = (x: unknown): x is Record => x !== null && typeof x === 'object' && !Array.isArray(x);
const positive = (x: unknown): x is number => Number.isSafeInteger(x) && (x as number) > 0;
const hash = (x: string | Buffer) => createHash('sha256').update(x).digest('hex');
const ownPath = (value: unknown): value is string => typeof value === 'string' && path.isAbsolute(value) && path.normalize(value) === value;
class BoundaryError extends Error {}
-const fail = (reason: string): never => { throw new BoundaryError(reason); };
+function fail(reason: string): never { throw new BoundaryError(reason); }
export interface PublicationHookInput {
hook_event_name: 'PreToolUse'; session_id: string; transcript_path: string; cwd: string;
tool_name: string; tool_use_id: string; tool_input: Record; agent_id?: string | null;
@@ -151,8 +154,9 @@ function textResult(event: Event): string | undefined {
/** Authenticate the existing direct-create result; this does not prove its shell command's origin. */
function checkpointResult(result: Event, entered: Event[], init: Invocation): { phase: Phase; path: string } | undefined {
- const use = entered.find(e => e.kind === 'use' && e.toolUseId === result.toolUseId);
- if (result.kind !== 'result' || use?.name !== 'Bash' || use.order >= result.order) return;
+ if (result.kind !== 'result') return;
+ const use = entered.find((e): e is Use => isUse(e) && e.toolUseId === result.toolUseId);
+ if (use?.name !== 'Bash' || use.order >= result.order) return;
const text = textResult(result);
if (text === undefined) return;
const output = JSON.parse(text);
@@ -197,7 +201,7 @@ function invocation(events: Event[], root: string): Invocation {
if (!object(result) || result.sourcePlan !== fs.realpathSync(args[0]!) || result.activePlan !== args[1] ||
result.restorePath !== args[2] || typeof result.reused !== 'boolean' || !positive(result.originalBytes) ||
!/^[a-f0-9]{64}$/.test(result.originalSha256)) fail('Autoplan initialization does not match the successful native request.');
- if (result.reused && bound?.activePlan === result.activePlan && bound.restorePath === result.restorePath) continue;
+ if (result.reused && bound && bound.activePlan === result.activePlan && bound.restorePath === result.restorePath) continue;
chosen = result;
bound = { activePlan: result.activePlan, restorePath: result.restorePath,
originalSha256: result.originalSha256, start: results[0]!.order };
@@ -280,7 +284,7 @@ function closePacket(file: string, phase: Phase, init: Invocation, current = tru
/** A skill hook survives end_turn; unrelated human intervals are never phase evidence. */
function disarmed(events: Event[], root: string): boolean {
- const human = events.filter(e => e.kind === 'user_turn').at(-1);
+ const human = events.filter((e): e is Turn => e.kind === 'user_turn').at(-1);
return !!human && !human.autoplan && events.some(e => e.kind === 'end_turn' && e.order < human.order) &&
!events.some(e => e.kind === 'use' && e.name === 'Bash' && e.order > human.order && initArguments(e.input?.command, root));
}
@@ -293,7 +297,7 @@ function verifyCloseEdits(events: Event[], closeOrder: number, init: Invocation)
const current = read(init.activePlan);
let prior = current;
for (const use of edits.toReversed()) {
- const results = events.filter(e => e.kind === 'result' && e.toolUseId === use.toolUseId);
+ const results = events.filter((e): e is Tool => e.kind === 'result' && e.toolUseId === use.toolUseId);
if (results.length !== 1) fail('An active-plan mutation is pending after the close Read. Wait for its result, then verify the current close input.');
if (results[0]!.isError === true) continue;
const input = use.input;
@@ -375,7 +379,7 @@ function evaluatePublication(input: PublicationHookInput, root: string, events:
// Pinned Claude retains skill hooks after end_turn. Only an authenticated
// later human request can release the old invocation; tool results and
// compaction never do. A native slash or an actual init re-arms the guard.
- const human = before.filter(e => e.kind === 'user_turn').at(-1);
+ const human = before.filter((e): e is Turn => e.kind === 'user_turn').at(-1);
if (disarmed(before, root)) {
if (pendingRead) fail('Current native phase-entry identity is unavailable after this invocation ended.');
return { allow: true };
@@ -403,8 +407,8 @@ function evaluatePublication(input: PublicationHookInput, root: string, events:
} else if (!preparedCheckpoints.has(created.phase)) preparedCheckpoints.set(created.phase, created.path);
continue;
}
- if (use.kind !== 'use' || !['Read', 'Agent'].includes(use.name ?? '')) continue;
- const results = entered.filter(e => e.kind === 'result' && e.toolUseId === use.toolUseId);
+ if (!isUse(use) || !['Read', 'Agent'].includes(use.name ?? '')) continue;
+ const results = entered.filter((e): e is Tool => e.kind === 'result' && e.toolUseId === use.toolUseId);
if (results.length !== 1 || results[0]!.isError !== false || results[0]!.order <= use.order) continue;
let next: Consumer | undefined;
try { next = consumption(use, input.cwd, root, init, true); } catch { continue; }
diff --git a/bin/gstack-design-md.ts b/bin/gstack-design-md.ts
index ec2f419f2..a5c6bbe8b 100755
--- a/bin/gstack-design-md.ts
+++ b/bin/gstack-design-md.ts
@@ -93,7 +93,7 @@ export function main(argv = process.argv.slice(2)): number {
}
case 'mark': {
const choice = positional[0] as FormatChoice | undefined;
- if (!(FORMAT_CHOICES as readonly string[]).includes(choice)) {
+ if (!choice || !(FORMAT_CHOICES as readonly string[]).includes(choice)) {
process.stderr.write(`usage: gstack-design-md.ts mark <${FORMAT_CHOICES.join('|')}> [DESIGN.md]\n`);
return 2;
}
diff --git a/bin/gstack-gbrain-sync.ts b/bin/gstack-gbrain-sync.ts
index 29d644c6c..a613e3741 100644
--- a/bin/gstack-gbrain-sync.ts
+++ b/bin/gstack-gbrain-sync.ts
@@ -76,7 +76,10 @@ interface CodeStageDetail {
| "failed"
| "refused-autopilot"
| "refused-reclone"
- | "refused-egress-receipt";
+ | "refused-egress-receipt"
+ | "skipped-policy-read-only"
+ | "refused-policy-deny"
+ | "refused-policy-unreadable";
}
interface StageResult {
diff --git a/bin/gstack-next-version b/bin/gstack-next-version
index 1df13aad8..6b480d87a 100755
--- a/bin/gstack-next-version
+++ b/bin/gstack-next-version
@@ -165,7 +165,7 @@ function zeroBaseAtLocalWidth(versionPath: string, repoRoot: string): string {
function readBaseVersion(base: string, versionPath: string, repoRoot: string, warnings: string[]): string {
// git fetch is best-effort; we tolerate failure and fall back to whatever
// origin/ currently points at.
- runCommand("git", ["fetch", "origin", base, "--quiet"], 10000);
+ runCommand("git", ["fetch", "--no-auto-maintenance", "origin", base, "--quiet"], 10000);
const r = runCommand("git", ["show", `origin/${base}:${versionPath}`]);
if (!r.ok) {
const assumed = zeroBaseAtLocalWidth(versionPath, repoRoot);
@@ -610,7 +610,7 @@ function fetchGitClaimed(
// bounded) brings every missing tip local in a single round trip.
spawnSync(
"git",
- ["fetch", "origin", ...pending.map((p) => `refs/heads/${p.branch}`), "--depth=1", "--no-tags"],
+ ["fetch", "--no-auto-maintenance", "origin", ...pending.map((p) => `refs/heads/${p.branch}`), "--depth=1", "--no-tags"],
{ encoding: "utf8", timeout: 15000, env: { ...process.env, GIT_TERMINAL_PROMPT: "0" } },
);
// One unservable ref (dangling sha on the server) fails the WHOLE batch
@@ -626,7 +626,7 @@ function fetchGitClaimed(
retries++;
spawnSync(
"git",
- ["fetch", "origin", `refs/heads/${branch}`, "--depth=1", "--no-tags"],
+ ["fetch", "--no-auto-maintenance", "origin", `refs/heads/${branch}`, "--depth=1", "--no-tags"],
{ encoding: "utf8", timeout: 5000, env: { ...process.env, GIT_TERMINAL_PROMPT: "0" } },
);
outcome = readClaim(branch, sha);
diff --git a/bin/gstack-safe-git b/bin/gstack-safe-git
new file mode 100755
index 000000000..c18d16046
--- /dev/null
+++ b/bin/gstack-safe-git
@@ -0,0 +1,149 @@
+#!/usr/bin/env bash
+# gstack-safe-git — run one allowlisted, read-only Git query for audits that
+# must not execute project-controlled code (/deslop-shared-libs).
+#
+# Usage: gstack-safe-git [-C ] [args...]
+#
+# Every invocation runs `git` with this fixed prefix; callers cannot add or
+# override it:
+# GIT_OPTIONAL_LOCKS=0 GIT_NO_LAZY_FETCH=1 GIT_TERMINAL_PROMPT=0
+# git --no-pager --no-lazy-fetch --no-replace-objects
+# -c core.fsmonitor=false -c log.showSignature=false -c diff.submodule=short
+#
+# Only query shapes that cannot run clean/process filters, textconv or external
+# diff drivers, signature verifiers, pagers, transports, or index/ref writes are
+# forwarded. log/show/diff always get --no-ext-diff --no-textconv; diff is only
+# between two explicit object IDs. Everything else is refused with exit 2 and
+# a one-line message naming the allowed forms. Git's own exit status passes
+# through unchanged, including 129 when this Git lacks --no-lazy-fetch.
+#
+# The script sources nothing and executes only `git` from PATH.
+set -euo pipefail
+
+ALLOWED='allowed: rev-parse, symbolic-ref [--short] [, branch --show-current, remote [-v | get-url ], config --get|--get-all|--get-regexp , log, show, ls-tree, cat-file, rev-list, merge-base, for-each-ref, show-ref, grep, diff [-- ...], ls-files --cached --others --exclude-standard -z [-- ...]'
+
+refuse() {
+ echo "gstack-safe-git: refused: $1; $ALLOWED" >&2
+ exit 2
+}
+
+dir_args=()
+if [ "${1:-}" = "-C" ]; then
+ [ $# -ge 2 ] || refuse "-C needs a directory"
+ dir_args=(-C "$2")
+ shift 2
+fi
+[ $# -ge 1 ] || refuse "no subcommand"
+sub=$1
+shift
+case "$sub" in
+ -*) refuse "global option '$sub' (only a leading -C is accepted; the safety -c settings are fixed)" ;;
+ rev-parse|symbolic-ref|branch|remote|config|log|show|ls-tree|cat-file|rev-list|merge-base|for-each-ref|show-ref|grep|diff|ls-files) ;;
+ *) refuse "'$sub' is not an allowlisted read" ;;
+esac
+
+for arg in "$@"; do
+ [ "$arg" = "--" ] && break
+ case "$arg" in
+ --output|--output=*) refuse "'$arg' writes files" ;;
+ --ext-diff|--textconv|--filters|--path|--path=*) refuse "'$arg' can run configured diff drivers or filters" ;;
+ --show-signature|*%G*|*'%(signature'*) refuse "'$arg' runs a signature verifier" ;;
+ --no-index|--recurse-submodules) refuse "'$arg' reads outside the repository's committed objects" ;;
+ esac
+done
+
+positional_before_dashdash() {
+ local count=0 arg
+ for arg in "$@"; do
+ [ "$arg" = "--" ] && break
+ case "$arg" in -*) ;; *) count=$((count + 1)) ;; esac
+ done
+ echo "$count"
+}
+
+extra=()
+case "$sub" in
+ rev-parse|ls-tree|cat-file|rev-list|merge-base|for-each-ref|show-ref) ;;
+ log|show) extra=(--no-ext-diff --no-textconv) ;;
+ grep)
+ for arg in "$@"; do
+ [ "$arg" = "--" ] && break
+ case "$arg" in
+ -O*|--open-files-in-pager*) refuse "'$arg' launches a pager program" ;;
+ esac
+ done
+ ;;
+ symbolic-ref)
+ for arg in "$@"; do
+ case "$arg" in
+ -q|--quiet|--short|--no-recurse) ;;
+ -*) refuse "symbolic-ref '$arg' is not a read" ;;
+ esac
+ done
+ [ "$(positional_before_dashdash "$@")" = 1 ] || refuse "symbolic-ref reads exactly one ref"
+ ;;
+ branch)
+ [ "$*" = "--show-current" ] || refuse "branch is limited to 'branch --show-current'"
+ ;;
+ remote)
+ case "$*" in
+ ''|-v|--verbose) ;;
+ *)
+ [ "${1:-}" = "get-url" ] || refuse "remote is limited to listing and get-url"
+ shift_count=0
+ for arg in "${@:2}"; do
+ case "$arg" in
+ --push|--all) ;;
+ -*) refuse "remote get-url '$arg'" ;;
+ *) shift_count=$((shift_count + 1)) ;;
+ esac
+ done
+ [ "$shift_count" = 1 ] || refuse "remote get-url takes one remote name"
+ ;;
+ esac
+ ;;
+ config)
+ case "${1:-}" in
+ --get|--get-all|--get-regexp) ;;
+ *) refuse "config is limited to --get, --get-all and --get-regexp" ;;
+ esac
+ [ $# -ge 2 ] && [ $# -le 3 ] || refuse "config reads take a key and an optional value pattern"
+ for arg in "${@:2}"; do
+ case "$arg" in -*) refuse "config '$arg'" ;; esac
+ done
+ ;;
+ diff)
+ ids=0
+ for arg in "$@"; do
+ [ "$arg" = "--" ] && break
+ case "$arg" in
+ --cached|--staged|--merge-base|--merge-base=*) refuse "diff '$arg' compares the index or derived revisions" ;;
+ -*) ;;
+ *)
+ [[ "$arg" =~ ^[0-9a-fA-F]{7,64}$ ]] || refuse "diff operand '$arg' is not an explicit object ID (put paths after --)"
+ ids=$((ids + 1))
+ ;;
+ esac
+ done
+ [ "$ids" = 2 ] || refuse "diff needs exactly two explicit committed object IDs, never the worktree or index"
+ extra=(--no-ext-diff --no-textconv)
+ ;;
+ ls-files)
+ nul=0
+ for arg in "$@"; do
+ [ "$arg" = "--" ] && break
+ case "$arg" in
+ -z) nul=1 ;;
+ --cached|--others|--exclude-standard|--stage) ;;
+ *) refuse "ls-files '$arg' (the overlay form is 'ls-files --cached --others --exclude-standard -z [-- ...]')" ;;
+ esac
+ done
+ [ "$nul" = 1 ] || refuse "ls-files output must be NUL-delimited with -z"
+ ;;
+esac
+
+unset GIT_EXTERNAL_DIFF GIT_CONFIG_PARAMETERS GIT_CONFIG_COUNT
+export GIT_OPTIONAL_LOCKS=0 GIT_NO_LAZY_FETCH=1 GIT_TERMINAL_PROMPT=0
+exec git --no-pager --no-lazy-fetch --no-replace-objects \
+ -c core.fsmonitor=false -c log.showSignature=false -c diff.submodule=short \
+ ${dir_args[@]+"${dir_args[@]}"} "$sub" ${extra[@]+"${extra[@]}"} "$@"
diff --git a/browse/sections/command-list.md b/browse/sections/command-list.md
index 24f7b93af..cd0ef01a6 100644
--- a/browse/sections/command-list.md
+++ b/browse/sections/command-list.md
@@ -163,7 +163,7 @@ Refs are invalidated on navigation — run `snapshot` again after `goto`.
### Server
| Command | Description |
|---------|-------------|
-| `connect` | Launch headed Chromium with Chrome extension |
+| `connect [--supervise]` | Launch headed Chromium with Chrome extension; --supervise keeps the CLI attached and respawns a crashed server |
| `disconnect` | Disconnect headed browser, return to headless mode |
| `focus [@ref]` | Bring headed browser window to foreground (macOS) |
| `handoff [message]` | Open visible Chrome at current page for user takeover |
diff --git a/browse/src/browser-manager.ts b/browse/src/browser-manager.ts
index b0d48bbf3..08600f31c 100644
--- a/browse/src/browser-manager.ts
+++ b/browse/src/browser-manager.ts
@@ -15,6 +15,7 @@
* restores state. Falls back to clean slate on any failure.
*/
+import type { ChildProcess } from 'node:child_process';
import { chromium, type Browser, type BrowserContext, type BrowserContextOptions, type Page, type Locator, type Cookie } from 'playwright';
import { writeSecureFile, mkdirSecure } from './file-permissions';
import { addConsoleEntry, addNetworkEntry, addDialogEntry, networkBuffer, type DialogEntry } from './buffers';
@@ -174,6 +175,12 @@ export function probePoisonedChromiumBundle(chromiumExecutablePath: string): voi
);
}
+/** Playwright's public Browser type omits `process()`, which only browsers we launched provide. */
+function launchedProcess(browser: Browser | null | undefined): ChildProcess | null {
+ const withProcess = browser as (Browser & { process?: () => ChildProcess | null }) | null | undefined;
+ return typeof withProcess?.process === 'function' ? withProcess.process() : null;
+}
+
/**
* Resolve why the underlying Chromium ChildProcess is going away.
*
@@ -196,7 +203,7 @@ export async function resolveDisconnectCause(browser: Browser | null): Promise<'
// obtained via connectOverCDP() (or a stub in tests) has no such method —
// calling it blind throws inside the disconnect handler, which killed the
// whole daemon with "browser?.process is not a function".
- const proc = typeof browser?.process === 'function' ? browser.process() : null;
+ const proc = launchedProcess(browser);
if (proc && proc.exitCode === null && proc.signalCode === null) {
await new Promise((resolve) => {
const timer = setTimeout(resolve, 1000);
@@ -599,7 +606,7 @@ export class BrowserManager {
// #2709: record the child's identity so the CLI can reap a survivor after
// daemon shutdown. `.process()` exists here — we launched this browser.
{
- const proc = typeof this.browser.process === 'function' ? this.browser.process() : null;
+ const proc = launchedProcess(this.browser);
this.chromiumProcInfo = proc?.pid
? { pid: proc.pid, startTime: readPidStartTime(proc.pid) }
: null;
@@ -955,7 +962,7 @@ export class BrowserManager {
this.context ? this.context.close() : Promise.resolve(),
raceTimeout(this.closeRaceMs),
]).catch(() => {});
- } else {
+ } else if (this.browser) {
// Launched mode: close the browser we spawned.
this.browser.removeAllListeners('disconnected');
// Grab the child handle BEFORE the race: nulling this.browser after a
@@ -963,7 +970,7 @@ export class BrowserManager {
// caller's event loop (and keep-alive connections into test servers)
// open forever — the intermittent whole-suite wedge. If graceful close
// doesn't finish in time, the child gets SIGKILL, not freedom.
- const child = this.browser.process?.();
+ const child = launchedProcess(this.browser);
const closed = await Promise.race([
this.browser.close().then(() => true as const),
raceTimeout(this.closeRaceMs),
@@ -976,7 +983,7 @@ export class BrowserManager {
}
if (previousBrowser && previousBrowser !== currentBrowser) {
previousBrowser.removeAllListeners('disconnected');
- const child = previousBrowser.process?.();
+ const child = launchedProcess(previousBrowser);
const closed = await Promise.race([
previousBrowser.close().then(() => true), raceTimeout(this.closeRaceMs),
]).catch(() => false);
@@ -2029,10 +2036,12 @@ export class BrowserManager {
tabSessions.delete(id);
console.log(`[browse] Tab closed (id=${id}, remaining=${pages.size})`);
// If the closed tab was active, switch to another
- const state = pages === this.pages ? this : this.handoffPrevious?.pages === pages ? this.handoffPrevious : null;
- if (state?.activeTabId === id) {
- const remaining = [...pages.keys()];
- state.activeTabId = remaining.length > 0 ? remaining[remaining.length - 1] : 0;
+ const remaining = [...pages.keys()];
+ const fallback = remaining.length > 0 ? remaining[remaining.length - 1]! : 0;
+ if (pages === this.pages) {
+ if (this.activeTabId === id) this.activeTabId = fallback;
+ } else if (this.handoffPrevious?.pages === pages && this.handoffPrevious.activeTabId === id) {
+ this.handoffPrevious.activeTabId = fallback;
}
break;
}
diff --git a/browse/src/cli.ts b/browse/src/cli.ts
index 980a311dd..fe44feaf7 100644
--- a/browse/src/cli.ts
+++ b/browse/src/cli.ts
@@ -130,7 +130,7 @@ interface ServerState {
configHash?: string;
/** Xvfb child PID for cleanup on disconnect. */
xvfbPid?: number;
- xvfbStartTime?: number;
+ xvfbStartTime?: string;
xvfbDisplay?: string;
/** Launched-Chromium identity for post-stop reaping (#2709). */
chromiumPid?: number;
@@ -423,6 +423,102 @@ export function buildRestartEnv(
return env;
}
+/**
+ * Build the env for the headed `$B connect` server. Used by the initial
+ * connect and by the opt-in supervisor's respawn, so a respawned server keeps
+ * the same port, watchdog setting, proxy and config hash. Pure + exported for tests.
+ */
+export function buildHeadedServerEnv(
+ globalFlags: Pick,
+): Record {
+ return {
+ BROWSE_HEADED: '1',
+ // Use a well-known port so the Chrome extension auto-connects.
+ BROWSE_PORT: '34567',
+ // Disable parent-process watchdog: the user controls the headed browser
+ // window lifecycle. The CLI exits immediately after connect, so watching
+ // it would kill the server ~15s later. Cleanup happens via browser
+ // disconnect event or $B disconnect.
+ BROWSE_PARENT_PID: '0',
+ // Apply --proxy from this invocation if present. Without this,
+ // `browse --proxy connect` would launch headed Chromium
+ // bypassing the SOCKS bridge entirely.
+ ...(globalFlags.proxyUrl ? { BROWSE_PROXY_URL: globalFlags.proxyUrl } : {}),
+ ...(globalFlags.configHash ? { BROWSE_CONFIG_HASH: globalFlags.configHash } : {}),
+ };
+}
+
+export const SUPERVISOR_GUARD_WINDOW_MS = 5 * 60_000;
+export const SUPERVISOR_GUARD_MAX = 5;
+
+export interface HeadedSupervisorDeps {
+ env: Record;
+ tickMs: number;
+ backoffMs: number[];
+ daemonLog: string;
+ readState: () => { pid?: number } | null;
+ isProcessAlive: (pid: number) => boolean;
+ startServer: (env: Record) => Promise<{ pid: number; port: number }>;
+ spawnTerminalAgent: (server: { pid: number; port: number }) => void;
+ sleep: (ms: number) => Promise;
+ now: () => number;
+ isExiting: () => boolean;
+ log: (line: string) => void;
+ warn: (line: string) => void;
+ error: (line: string) => void;
+}
+
+/**
+ * The opt-in `$B connect --supervise` loop: poll the server PID every tick and
+ * respawn it with the connect env when it dies. Five respawns inside the
+ * rolling five-minute window give up. Returns 'stopped' when a signal asked it
+ * to exit and 'gave_up' when the crash-loop guard tripped.
+ */
+export async function runHeadedSupervisor(deps: HeadedSupervisorDeps): Promise<'stopped' | 'gave_up'> {
+ const respawns: number[] = [];
+ while (!deps.isExiting()) {
+ await deps.sleep(deps.tickMs);
+ if (deps.isExiting()) break;
+ const state = deps.readState();
+ if (state?.pid && deps.isProcessAlive(state.pid)) continue;
+ // Server died. Prune rolling window and check guard.
+ const now = deps.now();
+ while (respawns.length && now - respawns[0] > SUPERVISOR_GUARD_WINDOW_MS) {
+ respawns.shift();
+ }
+ if (respawns.length >= SUPERVISOR_GUARD_MAX) {
+ deps.error(
+ `[browse] Supervisor: ${SUPERVISOR_GUARD_MAX} server crashes in ${SUPERVISOR_GUARD_WINDOW_MS / 1000}s, giving up. ` +
+ `Crash reasons: ${deps.daemonLog}. Relaunch: $B connect --supervise`,
+ );
+ return 'gave_up';
+ }
+ const attempt = respawns.length;
+ respawns.push(now);
+ const backoff = deps.backoffMs[Math.min(attempt, deps.backoffMs.length - 1)] ?? 30_000;
+ deps.warn(`[browse] Supervisor: server PID gone — respawning in ${backoff}ms (attempt ${attempt + 1}/${SUPERVISOR_GUARD_MAX})...`);
+ await deps.sleep(backoff);
+ if (deps.isExiting()) break;
+ let respawned: { pid: number; port: number };
+ try {
+ respawned = await deps.startServer(deps.env);
+ } catch (err: any) {
+ // Let the next tick try again — the crash-loop guard already
+ // bounded the retries via the rolling window.
+ deps.error(`[browse] Supervisor: server respawn failed: ${err?.message || err}. Daemon log: ${deps.daemonLog}`);
+ continue;
+ }
+ deps.log(`[browse] Supervisor: server respawned (PID ${respawned.pid}, port ${respawned.port}).`);
+ // Re-spawn the terminal-agent too; same env wiring as the initial connect.
+ try {
+ deps.spawnTerminalAgent(respawned);
+ } catch (err: any) {
+ deps.warn(`[browse] Supervisor: terminal-agent respawn failed: ${err?.message || err}`);
+ }
+ }
+ return 'stopped';
+}
+
/** macOS only: pull the headed Chromium window to the user's current Space.
* "Google Chrome for Testing" frequently opens behind the active window or on
* another Space — the first thing users read as "I can't see the browser"
@@ -1640,22 +1736,7 @@ Refs: After 'snapshot', use @e1, @e2... as selectors:
console.log('Launching headed Chromium with extension + terminal agent...');
try {
// Start server in headed mode with extension auto-loaded
- // Use a well-known port so the Chrome extension auto-connects
- const serverEnv: Record = {
- BROWSE_HEADED: '1',
- BROWSE_PORT: '34567',
- // Disable parent-process watchdog: the user controls the headed browser
- // window lifecycle. The CLI exits immediately after connect, so watching
- // it would kill the server ~15s later. Cleanup happens via browser
- // disconnect event or $B disconnect.
- BROWSE_PARENT_PID: '0',
- // Apply --proxy from this invocation if present. Without this,
- // `browse --proxy connect` would launch headed Chromium
- // bypassing the SOCKS bridge entirely.
- ...(globalFlags.proxyUrl ? { BROWSE_PROXY_URL: globalFlags.proxyUrl } : {}),
- ...(globalFlags.configHash ? { BROWSE_CONFIG_HASH: globalFlags.configHash } : {}),
- };
- const newState = await startServer(serverEnv);
+ const newState = await startServer(buildHeadedServerEnv(globalFlags));
// Print connected status
const resp = await fetch(`http://127.0.0.1:${newState.port}/command`, {
@@ -1737,58 +1818,31 @@ Refs: After 'snapshot', use @e1, @e2... as selectors:
process.on('SIGINT', () => teardownAndExit('SIGINT'));
process.on('SIGTERM', () => teardownAndExit('SIGTERM'));
- const SUPERVISOR_TICK_MS = parseInt(
- process.env.GSTACK_SUPERVISOR_TICK_MS || '30000',
- 10,
- );
- const SUPERVISOR_GUARD_WINDOW_MS = 5 * 60_000;
- const SUPERVISOR_GUARD_MAX = 5;
- const SUPERVISOR_BACKOFF_MS = (process.env.GSTACK_SUPERVISOR_BACKOFF || '1000,2000,4000,8000,30000')
- .split(',').map(s => parseInt(s.trim(), 10)).filter(n => Number.isFinite(n));
- const respawns: number[] = [];
-
- while (!supervisorExiting) {
- await new Promise(resolve => setTimeout(resolve, SUPERVISOR_TICK_MS));
- if (supervisorExiting) break;
- const state = readState();
- if (state?.pid && isProcessAlive(state.pid)) continue;
- // Server died. Prune rolling window and check guard.
- const now = Date.now();
- while (respawns.length && now - respawns[0] > SUPERVISOR_GUARD_WINDOW_MS) {
- respawns.shift();
- }
- if (respawns.length >= SUPERVISOR_GUARD_MAX) {
- console.error(
- `[browse] Supervisor: ${SUPERVISOR_GUARD_MAX} crashes in ${SUPERVISOR_GUARD_WINDOW_MS / 1000}s — giving up.`,
- );
- process.exit(1);
- }
- const attempt = respawns.length;
- respawns.push(now);
- const backoff = SUPERVISOR_BACKOFF_MS[Math.min(attempt, SUPERVISOR_BACKOFF_MS.length - 1)] ?? 30_000;
- console.warn(`[browse] Supervisor: server PID gone — respawning in ${backoff}ms (attempt ${attempt + 1}/${SUPERVISOR_GUARD_MAX})...`);
- await new Promise(resolve => setTimeout(resolve, backoff));
- if (supervisorExiting) break;
- try {
- const respawned = await startServer(serverEnv);
- console.log(`[browse] Supervisor: server respawned (PID ${respawned.pid}, port ${respawned.port}).`);
- // Re-spawn the terminal-agent too; same env wiring as the initial connect.
- try {
- spawnTerminalAgent({
- stateFile: config.stateFile,
- serverPort: respawned.port,
- ownerPid: respawned.pid,
- cwd: config.projectDir,
- });
- } catch (err: any) {
- console.warn(`[browse] Supervisor: terminal-agent respawn failed: ${err?.message || err}`);
- }
- } catch (err: any) {
- console.error(`[browse] Supervisor: server respawn failed: ${err?.message || err}`);
- // Let the next tick try again — the crash-loop guard already
- // bounded the retries via the rolling window.
- }
- }
+ const outcome = await runHeadedSupervisor({
+ env: buildHeadedServerEnv(globalFlags),
+ tickMs: parseInt(process.env.GSTACK_SUPERVISOR_TICK_MS || '30000', 10),
+ backoffMs: (process.env.GSTACK_SUPERVISOR_BACKOFF || '1000,2000,4000,8000,30000')
+ .split(',').map(s => parseInt(s.trim(), 10)).filter(n => Number.isFinite(n)),
+ daemonLog: daemonLogPath(),
+ readState,
+ isProcessAlive,
+ startServer,
+ spawnTerminalAgent: (respawned) => {
+ spawnTerminalAgent({
+ stateFile: config.stateFile,
+ serverPort: respawned.port,
+ ownerPid: respawned.pid,
+ cwd: config.projectDir,
+ });
+ },
+ sleep: (ms) => new Promise(resolve => setTimeout(resolve, ms)),
+ now: Date.now,
+ isExiting: () => supervisorExiting,
+ log: (line) => console.log(line),
+ warn: (line) => console.warn(line),
+ error: (line) => console.error(line),
+ });
+ if (outcome === 'gave_up') process.exit(1);
process.exit(0);
}
diff --git a/browse/src/commands.ts b/browse/src/commands.ts
index 8b9a7ea5d..3be130715 100644
--- a/browse/src/commands.ts
+++ b/browse/src/commands.ts
@@ -161,7 +161,7 @@ export const COMMAND_DESCRIPTIONS: Record inFlight.delete(clientSocket));
let state: State = 'greeting';
- let buf = Buffer.alloc(0);
+ let buf: Buffer = Buffer.alloc(0);
let upstreamSocket: net.Socket | null = null;
const killBoth = (reason?: string) => {
diff --git a/browse/src/terminal-agent.ts b/browse/src/terminal-agent.ts
index a15a4ba52..c5417bd38 100644
--- a/browse/src/terminal-agent.ts
+++ b/browse/src/terminal-agent.ts
@@ -516,8 +516,13 @@ function maybeSpawnPty(ws: any, session: PtySession): boolean {
return true;
}
+interface TerminalAgentWsData {
+ cookie: string;
+ sessionId: string | null;
+}
+
function buildServer(port: number) {
- return Bun.serve({
+ return Bun.serve({
hostname: '127.0.0.1',
// #2314: allocated from the SAME fixed 10000-60000 scan range the main
// server uses (port-allocator.ts, decision 8) — never `port: 0`. Binding
@@ -695,8 +700,8 @@ function buildServer(port: number) {
* after `spawned: true` is a no-op.
*/
open(ws) {
- const sessionId = (ws.data as any)?.sessionId ?? null;
- const cookie = (ws.data as any)?.cookie || '';
+ const sessionId = ws.data?.sessionId ?? null;
+ const cookie = ws.data?.cookie || '';
// Commit 3 re-attach: if this sessionId already has a detached
// PtySession in sessionsById, REPLACE its liveWs ref and replay
@@ -770,9 +775,9 @@ function buildServer(port: number) {
proc: null,
cols: 80,
rows: 24,
- cookie: (ws.data as any)?.cookie || '',
+ cookie: ws.data?.cookie || '',
liveWs: ws,
- sessionId: (ws.data as any)?.sessionId ?? null,
+ sessionId: ws.data?.sessionId ?? null,
spawned: false,
pingInterval: null,
ringBuffer: [],
@@ -850,7 +855,7 @@ function buildServer(port: number) {
// Always drop the WS-keyed map entry and the per-attach
// attachToken — the attach grant was single-use.
sessions.delete(ws);
- const cookie = (ws.data as any)?.cookie;
+ const cookie = ws.data?.cookie;
if (cookie) validTokens.delete(cookie);
// A reattach can replace liveWs before the old socket's close arrives.
// That stale callback must not retire the new socket, grant or child.
diff --git a/browse/test/cli-supervisor.test.ts b/browse/test/cli-supervisor.test.ts
index 22bdb57d9..a689a025a 100644
--- a/browse/test/cli-supervisor.test.ts
+++ b/browse/test/cli-supervisor.test.ts
@@ -1,6 +1,12 @@
import { describe, test, expect } from 'bun:test';
import * as fs from 'fs';
import * as path from 'path';
+import {
+ buildHeadedServerEnv,
+ runHeadedSupervisor,
+ SUPERVISOR_GUARD_WINDOW_MS,
+ type HeadedSupervisorDeps,
+} from '../src/cli';
// v1.44 outer supervisor — static-grep invariants.
//
@@ -11,9 +17,11 @@ import * as path from 'path';
// unexpected exit, with the same crash-loop guard shape as the v1.44
// terminal-agent watchdog.
//
-// Live respawn tests belong in the e2e tier (real Bun.spawn cycles take
-// 3-8s each). These tripwires defend the load-bearing invariants:
-// opt-in by default, signal handlers wired, crash-loop guard, env knobs.
+// The static tripwires below defend the wiring in main(): opt-in by default,
+// signal handlers, env knobs. The behavioral block drives the extracted
+// runHeadedSupervisor loop with injected clock, sleep, and process probes —
+// the respawn path shipped broken (a block-scoped env) because only source
+// text was checked.
const CLI_TS = path.resolve(import.meta.path, '..', '..', 'src', 'cli.ts');
@@ -72,6 +80,111 @@ describe('CLI outer supervisor (v1.44+)', () => {
});
});
+// A scripted world for runHeadedSupervisor: `alive` decides the PID probe per
+// tick, sleep advances the injected clock, and every side effect is recorded.
+function harness(opts: {
+ alive: (tick: number) => boolean;
+ startServer?: (call: number) => Promise<{ pid: number; port: number }>;
+ spawnTerminalAgent?: () => void;
+ tickMs?: number;
+ exitAfterSleeps?: number;
+}) {
+ let clock = 1_000_000, sleeps = 0, tick = 0, exiting = false, starts = 0;
+ const calls = { startEnv: [] as Record[], agents: [] as number[], log: [] as string[], warn: [] as string[], error: [] as string[] };
+ const deps: HeadedSupervisorDeps = {
+ env: buildHeadedServerEnv({ proxyUrl: 'socks5://127.0.0.1:9050', configHash: 'abc123' }),
+ tickMs: opts.tickMs ?? 30_000,
+ backoffMs: [1000, 2000, 4000, 8000, 30000],
+ daemonLog: '/state/browse-daemon.log',
+ readState: () => ({ pid: 4242 }),
+ isProcessAlive: () => opts.alive(tick++),
+ startServer: async (env) => {
+ calls.startEnv.push(env);
+ const call = starts++;
+ return opts.startServer ? opts.startServer(call) : { pid: 5000 + call, port: 34567 };
+ },
+ spawnTerminalAgent: (server) => { calls.agents.push(server.pid); opts.spawnTerminalAgent?.(); },
+ sleep: async (ms) => {
+ clock += ms; sleeps++;
+ if (opts.exitAfterSleeps !== undefined && sleeps >= opts.exitAfterSleeps) exiting = true;
+ },
+ now: () => clock,
+ isExiting: () => exiting,
+ log: (line) => calls.log.push(line),
+ warn: (line) => calls.warn.push(line),
+ error: (line) => calls.error.push(line),
+ };
+ return { deps, calls, stop: () => { exiting = true; } };
+}
+
+describe('runHeadedSupervisor (behavior)', () => {
+ test('a dead server is respawned with exactly the initial connect env, and its terminal agent too', async () => {
+ const h = harness({ alive: (t) => t !== 0, exitAfterSleeps: 4 });
+ expect(await runHeadedSupervisor(h.deps)).toBe('stopped');
+ expect(h.calls.startEnv).toHaveLength(1);
+ expect(h.calls.startEnv[0]).toEqual({
+ BROWSE_HEADED: '1', BROWSE_PORT: '34567', BROWSE_PARENT_PID: '0',
+ BROWSE_PROXY_URL: 'socks5://127.0.0.1:9050', BROWSE_CONFIG_HASH: 'abc123',
+ });
+ expect(h.calls.startEnv[0]).toBe(h.deps.env);
+ expect(h.calls.agents).toEqual([5000]);
+ expect(h.calls.error).toEqual([]);
+ expect(h.calls.log.join('\n')).toContain('server respawned (PID 5000, port 34567)');
+ });
+
+ test('a failed respawn is logged with the daemon log path and counted toward the guard', async () => {
+ const h = harness({ alive: () => false, startServer: async () => { throw new Error('port 34567 busy'); } });
+ expect(await runHeadedSupervisor(h.deps)).toBe('gave_up');
+ const failures = h.calls.error.filter(line => line.includes('server respawn failed'));
+ expect(failures).toHaveLength(5);
+ expect(failures[0]).toBe('[browse] Supervisor: server respawn failed: port 34567 busy. Daemon log: /state/browse-daemon.log');
+ });
+
+ test('five crashes inside the window give up with the cause and the relaunch command', async () => {
+ const h = harness({ alive: () => false });
+ expect(await runHeadedSupervisor(h.deps)).toBe('gave_up');
+ expect(h.calls.startEnv).toHaveLength(5);
+ expect(h.calls.error.at(-1)).toBe(
+ '[browse] Supervisor: 5 server crashes in 300s, giving up. Crash reasons: /state/browse-daemon.log. Relaunch: $B connect --supervise',
+ );
+ });
+
+ test('crashes spread wider than the rolling window never trip the guard', async () => {
+ // One crash per tick with a tick longer than the window: every earlier
+ // respawn is pruned before the guard is checked.
+ const h = harness({ alive: (t) => t >= 12, tickMs: SUPERVISOR_GUARD_WINDOW_MS + 1, exitAfterSleeps: 30 });
+ expect(await runHeadedSupervisor(h.deps)).toBe('stopped');
+ expect(h.calls.startEnv).toHaveLength(12);
+ expect(h.calls.error).toEqual([]);
+ });
+
+ test('a terminal-agent failure after a successful respawn warns and keeps supervising', async () => {
+ const h = harness({ alive: (t) => t !== 0, spawnTerminalAgent: () => { throw new Error('no pty'); }, exitAfterSleeps: 4 });
+ expect(await runHeadedSupervisor(h.deps)).toBe('stopped');
+ expect(h.calls.warn.some(line => line === '[browse] Supervisor: terminal-agent respawn failed: no pty')).toBe(true);
+ expect(h.calls.error).toEqual([]);
+ });
+
+ test('an exit requested during backoff stops without starting a server', async () => {
+ // Sleep 1 is the tick, sleep 2 the backoff; exiting flips during backoff.
+ const h = harness({ alive: () => false, exitAfterSleeps: 2 });
+ expect(await runHeadedSupervisor(h.deps)).toBe('stopped');
+ expect(h.calls.startEnv).toEqual([]);
+ });
+
+ test('a live server is left alone', async () => {
+ const h = harness({ alive: () => true, exitAfterSleeps: 5 });
+ expect(await runHeadedSupervisor(h.deps)).toBe('stopped');
+ expect(h.calls.startEnv).toEqual([]);
+ });
+});
+
+describe('buildHeadedServerEnv', () => {
+ test('omits proxy and config hash when this invocation has none', () => {
+ expect(buildHeadedServerEnv({ proxyUrl: null, configHash: '' })).toEqual({ BROWSE_HEADED: '1', BROWSE_PORT: '34567', BROWSE_PARENT_PID: '0' });
+ });
+});
+
function sliceBetween(source: string, start: string, end: string): string {
const i = source.indexOf(start);
if (i === -1) throw new Error(`marker not found: ${start}`);
diff --git a/browse/test/dia-macos-qualification.test.ts b/browse/test/dia-macos-qualification.test.ts
index 40e13bf67..48e483158 100644
--- a/browse/test/dia-macos-qualification.test.ts
+++ b/browse/test/dia-macos-qualification.test.ts
@@ -1214,7 +1214,7 @@ Binary Images:
{ status: 1, stdout: '', stderr: '' }, { status: 0, stdout: '', stderr: '' },
{ status: 0, stdout: 'truncated-private-row', stderr: '' },
{ status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') },
- ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync)).toEqual({ available: false });
+ ]) expect(inspectUidProcesses(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync)).toEqual({ available: false });
});
test('numeric UID process filtering runs through the real global process table', () => {
@@ -1251,7 +1251,7 @@ Binary Images:
{ status: 113, stdout: '', stderr: 'Could not find domain for user uid: 23456' },
{ status: null, stdout: null, stderr: null, error: new Error('synthetic-private-error') },
]) {
- const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as typeof spawnSync);
+ const observation = inspectUserDomain(23456, performance.now() + 10_000, {}, (() => result) as unknown as typeof spawnSync);
expect(observation.state).toBe('unavailable');
expect(observation.structure).toBeUndefined();
expect(JSON.stringify(observation)).not.toContain('synthetic-private');
diff --git a/browse/test/server-auth.test.ts b/browse/test/server-auth.test.ts
index 4fbcff020..2220be27c 100644
--- a/browse/test/server-auth.test.ts
+++ b/browse/test/server-auth.test.ts
@@ -12,6 +12,7 @@
import { describe, test, expect, beforeAll, afterAll } from 'bun:test';
import * as fs from 'fs';
import * as path from 'path';
+import { buildHeadedServerEnv } from '../src/cli';
import { GSTACK_EXTENSION_ID } from '../src/server';
import { DEFAULT_PAIR_SCOPES, createToken } from '../src/token-registry';
import { getActivityHistory } from '../src/activity';
@@ -508,15 +509,12 @@ describe('Server auth security', () => {
// The connect subprocess env must override BROWSE_PARENT_PID
expect(pairBlock).toContain("BROWSE_PARENT_PID");
expect(pairBlock).toContain("'0'");
- // The connect command must propagate BROWSE_PARENT_PID=0 via the
- // serverEnv object literal passed to startServer. The literal text
- // `serverEnv.BROWSE_PARENT_PID` is NOT in source — the value is
- // assigned via object-literal syntax (`BROWSE_PARENT_PID: '0'`)
- // inside the `const serverEnv: Record = { ... }`
- // declaration. Assert both pieces appear in the connect block.
+ // The connect command starts its server with buildHeadedServerEnv, the
+ // same env the --supervise respawn uses, and that env disables the
+ // parent-PID watchdog.
const connectBlock = sliceBetween(CLI_SRC, 'Launching headed Chromium', 'Terminal agent started');
- expect(connectBlock).toContain("const serverEnv");
- expect(connectBlock).toContain("BROWSE_PARENT_PID: '0'");
+ expect(connectBlock).toContain('startServer(buildHeadedServerEnv(globalFlags))');
+ expect(buildHeadedServerEnv({ proxyUrl: null, configHash: '' }).BROWSE_PARENT_PID).toBe('0');
});
// Regression: newtab returned 403 for scoped tokens because the tab ownership
diff --git a/bun.lock b/bun.lock
index 4ea14d8ab..42dd194ba 100644
--- a/bun.lock
+++ b/bun.lock
@@ -17,6 +17,9 @@
"devDependencies": {
"@anthropic-ai/claude-agent-sdk": "0.2.117",
"@anthropic-ai/sdk": "^0.78.0",
+ "@types/bun": "1.4.0",
+ "prettier": "3.9.9",
+ "typescript": "7.0.2",
"xterm": "^5.3.0",
"xterm-addon-fit": "^0.8.0",
},
@@ -173,8 +176,50 @@
"@protobufjs/utf8": ["@protobufjs/utf8@1.1.2", "", {}, "sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug=="],
+ "@types/bun": ["@types/bun@1.4.0", "", { "dependencies": { "bun-types": "1.4.0" } }, "sha512-K+lZULY23vRgK/CfTjFIV+tyifaNdSMlPh9j+6mQ/cLfpOznLyAuzgV/JQysyECpkBQLVMSyvjlr2fBUSA9wFQ=="],
+
"@types/node": ["@types/node@26.4.0", "", { "dependencies": { "undici-types": "~8.3.0" } }, "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ=="],
+ "@typescript/typescript-aix-ppc64": ["@typescript/typescript-aix-ppc64@7.0.2", "", { "os": "aix", "cpu": "ppc64" }, "sha512-MTKKkWB7p/0E9xi1d1tHtZ5PiLkGEMIq88pK2CubZjOsLtYTLqhgIgi6zepFa+9GHZ6h05NMCkQxGKiPXMxXtQ=="],
+
+ "@typescript/typescript-darwin-arm64": ["@typescript/typescript-darwin-arm64@7.0.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-gowzar9MwS/aRWp6f3a4KUqzRjAZjOsmGNCM6LcTgXum+dBfgsBVMN+AgvOCCbguXyick6LJhpBszxMebJ8syA=="],
+
+ "@typescript/typescript-darwin-x64": ["@typescript/typescript-darwin-x64@7.0.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-SZ9xZInqApNlNGc9s0W1VSsktYSOe9cFqNOIqmN1Gs8SmkjKZYFt017G4VwPxASInODuAdbTW7sXiFUf893RgA=="],
+
+ "@typescript/typescript-freebsd-arm64": ["@typescript/typescript-freebsd-arm64@7.0.2", "", { "os": "freebsd", "cpu": "arm64" }, "sha512-W5NH4y/J0plIIS5b2xvTEkU7JFxyqdMAOgf+Ilhl0vHQXKO5dZoxd+C/jEtq56c4F3wk71RB4BMRQ2XdI+bwYQ=="],
+
+ "@typescript/typescript-freebsd-x64": ["@typescript/typescript-freebsd-x64@7.0.2", "", { "os": "freebsd", "cpu": "x64" }, "sha512-UMGDx5sTpzNw3WiPebH7l90IWfJggEd+egHt/q6p7/Cm3zqoV7VxkGXt+3DxPIw8CcmvAB0j3sVVfbhX+M4Tpw=="],
+
+ "@typescript/typescript-linux-arm": ["@typescript/typescript-linux-arm@7.0.2", "", { "os": "linux", "cpu": "arm" }, "sha512-gffT3xPz9sR7j/YJExkyPntrI0P2EP9XbOyWzth2/Gs0RstK+90RBcO0ncXoXy/beYll1SXw846Nf2zdnEz0QQ=="],
+
+ "@typescript/typescript-linux-arm64": ["@typescript/typescript-linux-arm64@7.0.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-Qh4eU4/y3yDjnfjjyPYihMj5/ODIlmt+Bzu17OI+fiSRDW57QmU5SiN63exPRNJPKUzcc1INa1NXdrJ+MqHjUQ=="],
+
+ "@typescript/typescript-linux-loong64": ["@typescript/typescript-linux-loong64@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-uEHck9i8hoAzXPiYRib1O7miOnz23SxIeVl6F4LXox+qov1K35jHcEW6VHKvZI+pyvl7fZEP4MCU5LYvIq1GuQ=="],
+
+ "@typescript/typescript-linux-mips64el": ["@typescript/typescript-linux-mips64el@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-R4KvAMnE43W5Qeqb0Ly56O3mWMWIAgsMyz36DCaycd5nbg/9kzm0liw3JocfRqyJY0KPmzFjbswozXyW0DnIYA=="],
+
+ "@typescript/typescript-linux-ppc64": ["@typescript/typescript-linux-ppc64@7.0.2", "", { "os": "linux", "cpu": "ppc64" }, "sha512-DORx5b3sd/4S7eayxm4FQv+A7CrkUIGRaHiwI8oiHTAI1fAPWhF4J0vAlkC8biAlHSVVwxMQ3tjZ2/DVbnQiiA=="],
+
+ "@typescript/typescript-linux-riscv64": ["@typescript/typescript-linux-riscv64@7.0.2", "", { "os": "linux", "cpu": "none" }, "sha512-wf0jqEDOjrPRnKwYRyyJDRo11KMbvMFrU+q4zqKyChODBzvlkbhNQfKvLxQCcwTpdDaXSHZTVuh0JoCrKCUMHQ=="],
+
+ "@typescript/typescript-linux-s390x": ["@typescript/typescript-linux-s390x@7.0.2", "", { "os": "linux", "cpu": "s390x" }, "sha512-IkwJc3L7yhytWd/ewjyxNDfOmswCm9GWMJT/ue/dU4aZNbwZeYAetq42VyLmsmSjvoX7z74X6ZaYCtzAr0EuGw=="],
+
+ "@typescript/typescript-linux-x64": ["@typescript/typescript-linux-x64@7.0.2", "", { "os": "linux", "cpu": "x64" }, "sha512-EYdf2cNg7rgCWJnxCdJ+F3V39O8ihb37eHAu1LK8oAFizgTQbPOK7zHHXbPt8rX24COqODXeI3sIf0fCXG7H/A=="],
+
+ "@typescript/typescript-netbsd-arm64": ["@typescript/typescript-netbsd-arm64@7.0.2", "", { "os": "none", "cpu": "arm64" }, "sha512-+polYF4MF04aPpO5FTkHran9yUQDSXqy5GiSDKpsll5jy3l3+g9QLhpf39T+ePtefhXLOGrLl0QIjkQP6VnelA=="],
+
+ "@typescript/typescript-netbsd-x64": ["@typescript/typescript-netbsd-x64@7.0.2", "", { "os": "none", "cpu": "x64" }, "sha512-8YIT0EHM/3dq10ZOVF/A7pc/YSMtbcecct4rWtexrnSCHOPcpC2KTLXfTCR6vDpnSiY12heNb1GiN/wu+T/FyA=="],
+
+ "@typescript/typescript-openbsd-arm64": ["@typescript/typescript-openbsd-arm64@7.0.2", "", { "os": "openbsd", "cpu": "arm64" }, "sha512-APT8+ClYnuYm1u9+kgGXoMj2VzWzcymwh2gNSQVySHfkRDGOTVkoWLjCmOQSaO+PoqQ57B0flRp9SA+7GnnkzQ=="],
+
+ "@typescript/typescript-openbsd-x64": ["@typescript/typescript-openbsd-x64@7.0.2", "", { "os": "openbsd", "cpu": "x64" }, "sha512-yX7s+Q0Dln0Dt9tEzZsAjXXR/+ytBM7AlglaqyeMPxQszJ1JhlJdZ6jLA+IzldHtflX81em7lDao1xXu+aRRkg=="],
+
+ "@typescript/typescript-sunos-x64": ["@typescript/typescript-sunos-x64@7.0.2", "", { "os": "sunos", "cpu": "x64" }, "sha512-dLJDGaLZ1D4HPQn62u1n8mBDkJREwMsAkCdkwd4Ieqw+x3TUyTsqY0YiBCtE6H6OzzgGk3iuZ3vFWRS+E8/d1g=="],
+
+ "@typescript/typescript-win32-arm64": ["@typescript/typescript-win32-arm64@7.0.2", "", { "os": "win32", "cpu": "arm64" }, "sha512-Gyl1Vy6OsWesLzmq+EP0Fb7b4Nid5232AvcA2SFcdYreldpNtYFFofPjnt62y9hQy7VTaZp65ICJjuAQRaVcIQ=="],
+
+ "@typescript/typescript-win32-x64": ["@typescript/typescript-win32-x64@7.0.2", "", { "os": "win32", "cpu": "x64" }, "sha512-0BQ3HkAHHlKLSp1qRvf3SUhGpGsDuhB/jgFw75guyqbxJqEaS0Cw/VFO8i2nHglJUzQCRtMMR/IBAKE3ETMC4g=="],
+
"accepts": ["accepts@2.0.0", "", { "dependencies": { "mime-types": "^3.0.0", "negotiator": "^1.0.0" } }, "sha512-5cvg6CtKwfgdmVqY1WIiXKc3Q1bkRqGLi+2W/6ao+6Y7gu/RCwRuAhGEzh5B4KlszSuTLgZYuqFqo5bImjNKng=="],
"adm-zip": ["adm-zip@0.6.1", "", {}, "sha512-Xwrja8nx9e5o2N1my4DsKCeKpdrnACyr1wtbPxBDgGzKzKyE9kRtBFA8mWldI+RVlD7CBZNWY/wQ2+ydwOR6kQ=="],
@@ -189,6 +234,8 @@
"browser-split": ["browser-split@0.0.1", "", {}, "sha512-JhvgRb2ihQhsljNda3BI8/UcRHVzrVwo3Q+P8vDtSiyobXuFpuZ9mq+MbRGMnC22CjW3RrfXdg6j6ITX8M+7Ow=="],
+ "bun-types": ["bun-types@1.4.0", "", { "dependencies": { "@types/node": "*" } }, "sha512-iIKw23BspnQQYd3prITOBxeUsxBHnwzX6YJfGMuNOZzeNcMmVqzIIVGRm1l69ogaPQmb4wB6BN8mA5bE9YuC5Q=="],
+
"bytes": ["bytes@3.1.2", "", {}, "sha512-/Nf7TyzTx6S3yRJObOAV7956r8cr2+Oj8AC5dt8wSP3BQAoeX58NoHyCU8P8zGkNXStjTSi6fzO6F0pBdcYbEg=="],
"call-bind-apply-helpers": ["call-bind-apply-helpers@1.0.2", "", { "dependencies": { "es-errors": "^1.3.0", "function-bind": "^1.1.2" } }, "sha512-Sp1ablJ0ivDkSzjcaJdxEunN5/XvksFJ2sMBFfq6x0ryhQV/2b/KwFe21cMpmHtPOSij8K99/wSfoEuTObmuMQ=="],
@@ -425,6 +472,8 @@
"playwright-core": ["playwright-core@1.62.1", "", { "bin": { "playwright-core": "cli.js" } }, "sha512-wPYSwEBJY9GHraISXqyqtx0na0LpO3XEX7jNDhntbex7tzUS7kLnZsOlFruFJB4Hi/rhDMjXGqHewDZ68nYZVw=="],
+ "prettier": ["prettier@3.9.9", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-Z/CJHIkdujO/OtN7nXUii0Rf3VT5SRuhjBA82Xvu2XhBUgX3nhP67T0LHceBdQLex7OOFGTox+Q5Yg8Jk2Qivg=="],
+
"process": ["process@0.11.10", "", {}, "sha512-cdGef/drWFoydD1JsMzuFf8100nZl+GT+yacc2bEced5f9Rjk4z+WtFUTBu9PhOi9j/jfmBPu0mMEY4wIdAF8A=="],
"process-nextick-args": ["process-nextick-args@2.0.1", "", {}, "sha512-3ouUOpQhtgrbOa17J7+uxOTpITYWaGP7/AhoR3+A+/1e9skrzelGi/dXzEYyvbxubEF6Wn2ypscTKiKJFFn1ag=="],
@@ -509,6 +558,8 @@
"type-is": ["type-is@2.1.0", "", { "dependencies": { "content-type": "^2.0.0", "media-typer": "^1.1.0", "mime-types": "^3.0.0" } }, "sha512-faYHw0anBbc/kWF3zFTEnxSFOAGUX9GFbOBthvDdLsIlEoWOFOtS0zgCiQYwIskL9iGXZL3kAXD8OoZ4GmMATA=="],
+ "typescript": ["typescript@7.0.2", "", { "optionalDependencies": { "@typescript/typescript-aix-ppc64": "7.0.2", "@typescript/typescript-darwin-arm64": "7.0.2", "@typescript/typescript-darwin-x64": "7.0.2", "@typescript/typescript-freebsd-arm64": "7.0.2", "@typescript/typescript-freebsd-x64": "7.0.2", "@typescript/typescript-linux-arm": "7.0.2", "@typescript/typescript-linux-arm64": "7.0.2", "@typescript/typescript-linux-loong64": "7.0.2", "@typescript/typescript-linux-mips64el": "7.0.2", "@typescript/typescript-linux-ppc64": "7.0.2", "@typescript/typescript-linux-riscv64": "7.0.2", "@typescript/typescript-linux-s390x": "7.0.2", "@typescript/typescript-linux-x64": "7.0.2", "@typescript/typescript-netbsd-arm64": "7.0.2", "@typescript/typescript-netbsd-x64": "7.0.2", "@typescript/typescript-openbsd-arm64": "7.0.2", "@typescript/typescript-openbsd-x64": "7.0.2", "@typescript/typescript-sunos-x64": "7.0.2", "@typescript/typescript-win32-arm64": "7.0.2", "@typescript/typescript-win32-x64": "7.0.2" }, "bin": { "tsc": "bin/tsc" } }, "sha512-8FYau96o3NKOhbjKi/qNvG/W5jhzxkbdm5sj9AbZ/5T5sWqn3hJgLfGx27sRKZWTvyzCP8dLRBTf5tBTSRVUNA=="],
+
"undici-types": ["undici-types@8.3.0", "", {}, "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ=="],
"unpipe": ["unpipe@1.0.0", "", {}, "sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ=="],
diff --git a/design-consultation/SKILL.md b/design-consultation/SKILL.md
index a25908997..bfb66cfed 100644
--- a/design-consultation/SKILL.md
+++ b/design-consultation/SKILL.md
@@ -647,17 +647,14 @@ sections. Read a section in full before doing its step; do not work from memory.
## Phase 1: Product Context
-Confirm product context in Q1, pre-filled from the codebase; then ask the memorable-thing question.
+**AskUserQuestion Q1 — one brief that confirms context AND decides research.** Never ask a confirm-only question first. In the ELI10, state your pre-filled read (from README, product files or office-hours output): what the product is, who it's for, its space and project type (web app, dashboard, marketing site, editorial, internal tool, etc.). Options:
+- A) Context right — research what top products in this space do for design first
+- B) Context right — work from design knowledge only
+- C) Context wrong or incomplete — I'll correct it
-**AskUserQuestion Q1 — include ALL of these:**
-1. Confirm what the product is, who it's for, what space/industry
-2. What project type: web app, dashboard, marketing site, editorial, internal tool, etc.
-3. "Want me to research what top products in your space are doing for design, or should I work from my design knowledge?"
-4. **Explicitly say:** "At any point you can just drop into chat and we'll talk through anything — this isn't a rigid form, it's a conversation."
+Recommend A or B for this product, naming what research buys or costs here versus the other. **Explicitly say:** "At any point you can just drop into chat and we'll talk through anything — this isn't a rigid form, it's a conversation."
-Pre-fill context from README or office-hours output, then confirm it and the research preference in Q1.
-
-**Memorable-thing forcing question.** Before moving on, ask the user: *"What's the one
+**Memorable-thing forcing question.** After Q1's answer, in its own AskUserQuestion brief (never in Q1's call), ask: *"What's the one
thing you want someone to remember after they see this product for the first time?"*
Record the one-sentence answer: a feeling, visual, claim, or posture. Every subsequent design decision must serve it.
diff --git a/design-consultation/SKILL.md.tmpl b/design-consultation/SKILL.md.tmpl
index 100025a3b..a48b49a21 100644
--- a/design-consultation/SKILL.md.tmpl
+++ b/design-consultation/SKILL.md.tmpl
@@ -126,17 +126,14 @@ Phase 5: `DESIGN_READY` uses AI mockups on realistic product screens; `DESIGN_NO
## Phase 1: Product Context
-Confirm product context in Q1, pre-filled from the codebase; then ask the memorable-thing question.
+**AskUserQuestion Q1 — one brief that confirms context AND decides research.** Never ask a confirm-only question first. In the ELI10, state your pre-filled read (from README, product files or office-hours output): what the product is, who it's for, its space and project type (web app, dashboard, marketing site, editorial, internal tool, etc.). Options:
+- A) Context right — research what top products in this space do for design first
+- B) Context right — work from design knowledge only
+- C) Context wrong or incomplete — I'll correct it
-**AskUserQuestion Q1 — include ALL of these:**
-1. Confirm what the product is, who it's for, what space/industry
-2. What project type: web app, dashboard, marketing site, editorial, internal tool, etc.
-3. "Want me to research what top products in your space are doing for design, or should I work from my design knowledge?"
-4. **Explicitly say:** "At any point you can just drop into chat and we'll talk through anything — this isn't a rigid form, it's a conversation."
+Recommend A or B for this product, naming what research buys or costs here versus the other. **Explicitly say:** "At any point you can just drop into chat and we'll talk through anything — this isn't a rigid form, it's a conversation."
-Pre-fill context from README or office-hours output, then confirm it and the research preference in Q1.
-
-**Memorable-thing forcing question.** Before moving on, ask the user: *"What's the one
+**Memorable-thing forcing question.** After Q1's answer, in its own AskUserQuestion brief (never in Q1's call), ask: *"What's the one
thing you want someone to remember after they see this product for the first time?"*
Record the one-sentence answer: a feeling, visual, claim, or posture. Every subsequent design decision must serve it.
diff --git a/design/src/daemon.ts b/design/src/daemon.ts
index 8b6e4a1ed..3d4b72ee4 100644
--- a/design/src/daemon.ts
+++ b/design/src/daemon.ts
@@ -533,6 +533,7 @@ export function start(): { port: number } {
fetch: fetchHandler,
});
const actualPort = serverRef.port;
+ if (actualPort === undefined) throw new Error('design daemon did not bind a TCP port');
const state: DaemonState = {
pid: process.pid,
port: actualPort,
diff --git a/deslop-shared-libs/SKILL.md b/deslop-shared-libs/SKILL.md
index ef62064fd..9a5fbead3 100644
--- a/deslop-shared-libs/SKILL.md
+++ b/deslop-shared-libs/SKILL.md
@@ -66,9 +66,15 @@ changed after writing one.
A direct HTTP fallback must also return its response on stdout without
creating files; do not replace successful authenticated results with an
unauthenticated request and then describe the source as inaccessible.
-3. Before local object reads, probe no-lazy-fetch support using the safe Git
- prefix below and `rev-parse --is-inside-work-tree`. A successful Git version
- check alone is insufficient. If unsupported, use pinned-commit GET API source
+3. Run every Git command as `~/.claude/skills/gstack/bin/gstack-safe-git `, never bare `git`;
+ only the exact diagnostic `git --version` may run bare. The helper fixes the
+ no-lazy-fetch, lock, pager, fsmonitor, signature and replacement-object
+ protections and refuses reads that could run filters, drivers, hooks or
+ transports, naming the allowed forms. Never bypass a refusal with raw `git`.
+ `diff` takes exactly two explicit committed object IDs, then `--` and paths.
+ First probe the audited repository with `~/.claude/skills/gstack/bin/gstack-safe-git -C rev-parse --is-inside-work-tree` (use `-C ` on every call when your shell is elsewhere); a Git
+ version check alone is insufficient. If the probe fails (for example
+ `unknown option: --no-lazy-fetch`), use pinned-commit GET API source
and history reads or disclose unavailable local-history coverage. Never retry
object reads without the no-lazy-fetch protection, including by decoding loose
objects or packfiles directly. After an unsupported probe, do not inspect Git
@@ -79,31 +85,10 @@ changed after writing one.
unavailable, continue with clearly labeled raw source and unknown tracking
status and revision/history coverage.
-The exact diagnostic `git --version` may run without the prefix below: it does
-not read repository state or execute configured hooks. It never substitutes for
-the guarded capability probe. For every other Git invocation disable optional
-locks, pager, fsmonitor, signature verification, replacement objects and lazy fetch.
-Signature display can execute a
-configured project verifier. Replacement refs must not substitute different contents
-under a cited commit ID. Keep submodule diffs short rather than reading their trees.
-Use this prefix, including for the capability probe:
-
-```bash
-GIT_OPTIONAL_LOCKS=0 GIT_NO_LAZY_FETCH=1 GIT_TERMINAL_PROMPT=0 \
- git --no-pager --no-lazy-fetch --no-replace-objects \
- -c core.fsmonitor=false -c log.showSignature=false -c diff.submodule=short
-```
-
-Restrict `git diff` to **two explicit committed object IDs**, with
-`--no-ext-diff --no-textconv` and `--` before paths. Use the same disabling
-flags for patch-producing `log`/`show` commands. Never use worktree/index diffs,
-`git status`, temporary indexes, `add`, `hash-object --path`, or other
-normalization helpers: these can execute clean/process filters or alter the index.
-Do not execute scripts from the audited project, even to inspect it.
-
-For the uncommitted overlay, enumerate tracked and nonignored untracked paths with
-guarded, NUL-delimited `ls-files --cached --others --exclude-standard -z`, then
-inspect raw source with the host's read tools or isolated standard-library reads.
+Do not execute scripts from the audited project, even to inspect it. For the
+uncommitted overlay, enumerate tracked and nonignored untracked paths with
+`~/.claude/skills/gstack/bin/gstack-safe-git ls-files --cached --others --exclude-standard -z`, then inspect
+raw source with the host's read tools or isolated standard-library reads.
For Python reads, use a trusted interpreter with `python3 -I -S`: repository-local
modules can shadow standard-library imports and execute code or write bytecode.
Do not add project paths to imports, import project modules, or use runtimes that
@@ -115,6 +100,8 @@ the repo, traverse submodule worktrees, or execute filters. Note excluded symlin
submodule, ignored, unavailable or unreadable source. Handle deletions explicitly.
Do not call an absent or unreadable overlay clean. Current raw content may differ
even when a clean filter would produce the same Git tree.
+Sessions have a bounded number of turns. Read related files together: parallel
+host reads or one read-only command per step, not one file per turn.
## Start with recent work
diff --git a/deslop-shared-libs/SKILL.md.tmpl b/deslop-shared-libs/SKILL.md.tmpl
index 1d8a3329b..29cf64c5d 100644
--- a/deslop-shared-libs/SKILL.md.tmpl
+++ b/deslop-shared-libs/SKILL.md.tmpl
@@ -60,9 +60,15 @@ changed after writing one.
A direct HTTP fallback must also return its response on stdout without
creating files; do not replace successful authenticated results with an
unauthenticated request and then describe the source as inaccessible.
-3. Before local object reads, probe no-lazy-fetch support using the safe Git
- prefix below and `rev-parse --is-inside-work-tree`. A successful Git version
- check alone is insufficient. If unsupported, use pinned-commit GET API source
+3. Run every Git command as `{{SAFE_GIT}} `, never bare `git`;
+ only the exact diagnostic `git --version` may run bare. The helper fixes the
+ no-lazy-fetch, lock, pager, fsmonitor, signature and replacement-object
+ protections and refuses reads that could run filters, drivers, hooks or
+ transports, naming the allowed forms. Never bypass a refusal with raw `git`.
+ `diff` takes exactly two explicit committed object IDs, then `--` and paths.
+ First probe the audited repository with `{{SAFE_GIT}} -C rev-parse --is-inside-work-tree` (use `-C ` on every call when your shell is elsewhere); a Git
+ version check alone is insufficient. If the probe fails (for example
+ `unknown option: --no-lazy-fetch`), use pinned-commit GET API source
and history reads or disclose unavailable local-history coverage. Never retry
object reads without the no-lazy-fetch protection, including by decoding loose
objects or packfiles directly. After an unsupported probe, do not inspect Git
@@ -73,31 +79,10 @@ changed after writing one.
unavailable, continue with clearly labeled raw source and unknown tracking
status and revision/history coverage.
-The exact diagnostic `git --version` may run without the prefix below: it does
-not read repository state or execute configured hooks. It never substitutes for
-the guarded capability probe. For every other Git invocation disable optional
-locks, pager, fsmonitor, signature verification, replacement objects and lazy fetch.
-Signature display can execute a
-configured project verifier. Replacement refs must not substitute different contents
-under a cited commit ID. Keep submodule diffs short rather than reading their trees.
-Use this prefix, including for the capability probe:
-
-```bash
-GIT_OPTIONAL_LOCKS=0 GIT_NO_LAZY_FETCH=1 GIT_TERMINAL_PROMPT=0 \
- git --no-pager --no-lazy-fetch --no-replace-objects \
- -c core.fsmonitor=false -c log.showSignature=false -c diff.submodule=short
-```
-
-Restrict `git diff` to **two explicit committed object IDs**, with
-`--no-ext-diff --no-textconv` and `--` before paths. Use the same disabling
-flags for patch-producing `log`/`show` commands. Never use worktree/index diffs,
-`git status`, temporary indexes, `add`, `hash-object --path`, or other
-normalization helpers: these can execute clean/process filters or alter the index.
-Do not execute scripts from the audited project, even to inspect it.
-
-For the uncommitted overlay, enumerate tracked and nonignored untracked paths with
-guarded, NUL-delimited `ls-files --cached --others --exclude-standard -z`, then
-inspect raw source with the host's read tools or isolated standard-library reads.
+Do not execute scripts from the audited project, even to inspect it. For the
+uncommitted overlay, enumerate tracked and nonignored untracked paths with
+`{{SAFE_GIT}} ls-files --cached --others --exclude-standard -z`, then inspect
+raw source with the host's read tools or isolated standard-library reads.
For Python reads, use a trusted interpreter with `python3 -I -S`: repository-local
modules can shadow standard-library imports and execute code or write bytecode.
Do not add project paths to imports, import project modules, or use runtimes that
@@ -109,6 +94,8 @@ the repo, traverse submodule worktrees, or execute filters. Note excluded symlin
submodule, ignored, unavailable or unreadable source. Handle deletions explicitly.
Do not call an absent or unreadable overlay clean. Current raw content may differ
even when a clean filter would produce the same Git tree.
+Sessions have a bounded number of turns. Read related files together: parallel
+host reads or one read-only command per step, not one file per turn.
## Start with recent work
diff --git a/docs/TESTING_INTERNALS.md b/docs/TESTING_INTERNALS.md
index ac7819983..efcef1de5 100644
--- a/docs/TESTING_INTERNALS.md
+++ b/docs/TESTING_INTERNALS.md
@@ -252,16 +252,20 @@ processes × `EVALS_CONCURRENCY` within-shard, per-shard `GSTACK_EVAL_DIR`,
full-stream spooling to per-shard log files (path printed at START and on
failure), never-started/timed-out taxonomy, and parent-computed diff
selection propagated to children via `EVALS_SELECTION_JSON` (fail-open: a
-child that can't parse it recomputes locally with one warning). Retry parity
-lives in `RETRY_OVERRIDES` (literals; old matrix rows' earned `retries: 2`).
-Flake telemetry rides the store: every recorded test carries its 1-based
-`attempt` (a pass-on-attempt-2 stays visible forever — bun's own stream hides
-it), runs list `flaky_retries`, the report warns on passed-only-on-retry
-tests, and `bun run eval:flake-rank` ranks the series (retried passes first,
-then failure rate; 60-day recency bound on eval files; the free lane's flake
-ledger is folded in from `flakeLedgerPath()` — override with
-`GSTACK_FLAKE_LEDGER`, the same env var the CI free lane sets before
-uploading the ledger as the `flake-ledger` artifact). Census integrity is
+child that can't parse it recomputes locally with one warning). Paid evals
+never retry; each case's kind fixes its trials before the run (see "Eval verdict
+policy" below). Files in `CASE_SHARDED_FILES` run one registered case per
+process (`#`, an exact `--test-name-pattern`, exactly one executed
+case), so a long file of short cases spreads across runners and each case gets
+its own SDK semaphore.
+Trial telemetry rides the store: every recorded test carries its 1-based
+`attempt` plus, on an isolated trial shard, its `case_id`, `kind`, `trial`,
+`panel` and `policy_version`, and each lane's report uploads one
+`trial-outcomes` JSONL line per trial. `bun run eval:pass-rates`
+(`eval:flake-rank` is an alias) turns that history into per-case pass rates
+(see "Pass-rate history" below; the free lane's flake ledger is folded in from
+`flakeLedgerPath()` — override with `GSTACK_FLAKE_LEDGER`, the same env var the
+CI free lane sets before uploading the ledger as the `flake-ledger` artifact). Census integrity is
enforced from the free suite: every `E2E_TOUCHFILES` / `LLM_JUDGE_TOUCHFILES`
key must name a living paid test (`test/touchfiles.test.ts`'s reverse
invariant), and `git show :path` fixtures are banned — vendor the bytes
@@ -304,8 +308,17 @@ does not match the cache adapter and stays fresh, as do the other 11 quality cas
CI supplies the scoped cache/runtime configuration; local runs are fresh by
default. Cached scores must
pass current assertions; reused records retain their original source and time
-and cannot renew the receipt. Dynamic live-agent runs are currently ineligible.
-`EVALS_FRESH=1`, periodic and release validation bypass both lookup and publishing.
+and cannot renew the receipt. `scripts/e2e-shard-reuse.ts` extends the same receipts
+to PR-profile E2E shards (paid evals never retry, so a pass is structurally a
+first attempt): the identity hashes the test's import closure, every tracked file
+matched by the touchfiles of every case the file registers plus the global
+touchfiles, the runner/workflow/setup actions, the child's environment pins, the
+CI image and Claude CLI version, and the shard's case ids, pattern, wall and
+concurrency. A computed registration, an unmatched touchfile pattern, a retrying
+file, a preload option or a custom endpoint makes the shard ineligible. A reused
+shard reports `reused` with its source run and writes `execution: "reused"`
+collector records; the report rejects reused outcomes outside the fast PR profile.
+`EVALS_FRESH=1`, periodic, marathon and release validation bypass both lookup and publishing.
**Free test timing and isolation.** `test:quick` is an explicitly partial measured
subset for edit feedback. `test` remains complete local acceptance with its
@@ -318,18 +331,24 @@ the entire lane, not separately to every machine. Refresh the full timing list
with `bun run test:ubicloud --record-durations`. Profiling records failures faithfully
and is separate from final release acceptance.
-**CI planner/executor/report.** `--emit-plan --slices K` computes
-selection + the slice plan ONCE (killing per-slice selector divergence);
+**CI planner/executor/report.** `--emit-plan --slice-budget S --jobs J`
+(CI) or `--slices K` (local) computes selection + the slice plan ONCE (killing
+per-slice selector divergence);
`--plan --slice i` executors consume the manifest and write
slice-result artifacts; `--report ` reconciles them FAIL-CLOSED (a slice
whose artifact never landed, or a planned shard nobody reported, is a
-failure). Slices start from the supervision baseline (registered long files
-spread by budget, the rest round-robin), then are re-packed by the recorded
-wall times in `scripts/paid-test-durations.json`: a file moves or swaps out of
-the heaviest slice only if no slice's worst-case wall (`paidShardWallUpperBoundMs`
-for 1–4 workers) rises above the baseline's maximum, so CI timeout coverage is
-never weakened. Refresh the seed from a downloaded report directory with
-`--report --write-durations`. Under `EVALS_ALL` the hollow-shard guard marks exit-0 shards with
+failure). Budget mode (`packBySliceBudget`) places shards longest-recorded-first
+into the fullest slice whose estimated wall on J FIFO workers stays within S
+seconds, else a new slice; a shard with no recorded wall weighs the whole budget
+(its own runner), a shard longer than the budget runs alone, and overlays keep
+one final one-at-a-time slice. The manifest's `plan` records each slice estimate
+and `ciTimeoutMinutes` (every slice's supervised worst case plus 20 minutes
+setup); CI derives the matrix (`[range(1; .sliceCount + 1)]`) and job timeout
+from it, and executors refuse an `EVALS_JOBS` other than the planned J and run
+their shards longest first. `--slices K` keeps the supervised round-robin
+baseline re-packed by recorded times for local runs. Durations are recorded per
+tier (a file's gate and periodic cases differ); refresh one tier from a
+downloaded report directory with `--report --write-durations`. Under `EVALS_ALL` the hollow-shard guard marks exit-0 shards with
ZERO executed tests `passed-empty` (a failure) — census-health, not just
test runs. evals.yml runs the sliced gate lane per PR — the ONLY paid lane
since the legacy 17-row matrix (22.6 min/$21 per PR serialized ahead of the
@@ -337,7 +356,12 @@ slices) was deleted after demonstrated parity; its
`KNOWN_MATRIX_GAPS`/`KNOWN_TIER_UNSET` ratchets retired with it and
`test/evals-workflow-wiring.test.ts` pins the surviving wiring (slice-count
agreement, tier consistency, the shared register-skills composite with its
-fail-fast verification loop). evals-periodic.yml runs ALL
+fail-fast verification loop). Tier `marathon` (complete start-to-finish flows)
+is selected positively: a file enters the marathon plan only when it declares
+`describeE2ETier('marathon')` or registers a marathon-tier case, and the gate and
+periodic planners exclude marathon-only files; `evals-marathon.yml` runs them
+weekly and on dispatch, fresh, one file per runner, with its own fail-closed
+report and tracking issue, and nothing requires it. evals-periodic.yml runs ALL
periodic-tier files weekly (the coverage contract) minus the reasoned
exclusions in `test/helpers/periodic-exclude-data.ts` (reason + tracking
required per entry; removal re-activates the file), plus a weekly
@@ -349,6 +373,118 @@ the runner parent and handed to shard children as `GSTACK_CLAUDE_CLI_VERSION`
(never spawned on a test thread), so a TUI-drift flake hunt is a grep, not
archaeology.
+**Eval verdict policy** (`EVAL_POLICY` version 1 in
+`test/helpers/periodic-exclude-data.ts`, pre-registered 2026-09-29). Paid evals
+never retry. Each live case has exactly one kind in `E2E_KINDS`
+(`test/helpers/touchfiles-data.ts`; `test/eval-kinds.test.ts` enforces coverage),
+and the kind fixes its trials before the run:
+
+- `rule` (default): one trial; any failed assertion fails the verdict. For
+ cases where nothing stochastic decides the verdict, or where it checks a
+ contract the product must meet every run.
+- `behavior`: a panel of `n = 3` independent trials, launched together as
+ isolated case shards on different slices (key `#~t`). All three
+ always run: no early stop and no conditional extra trial. PASS when at least
+ `k = 2` pass and no trial violated a contract (`expectContract()` stamps
+ `failure_class: 'contract'`). Each behavior case names its tolerated deviation
+ in `BEHAVIOR_WHY` and must have a literal registration so it can run alone.
+- `judge`: an LLM judge scoring a fixed input. `judgePanel()`
+ (`test/helpers/llm-judge.ts`) draws 3 samples of the same prompt concurrently
+ inside the unchanged `JUDGE_MS`; numeric dimensions gate on the per-dimension
+ mean against the unchanged threshold (no dimension compensates for another),
+ booleans on a strict majority. A sample that errors (refusal, truncation,
+ non-JSON, a malformed field) fails the panel and is never resampled; a
+ refusal counts as an unscored panel only when every sample refused.
+ `callJudge`'s 429 backoff happens before any model output and is transport,
+ not a verdict retry. The workflow-judge cache stores whole panels only.
+
+`panelVerdict()` (`test/helpers/eval-store.ts`) is the single verdict
+function the report, `collector-outcomes.json`, the PR comment and pass-rates
+all use. A timed-out, crashed or infrastructure-failed trial is a failed trial
+recorded with its class; a missing or duplicate trial record makes the verdict
+INCOMPLETE, which fails the lane; a 2/3 PASS is shown as `PASS 2/3` with the
+failed trial's cause. A manual re-run adds trials under a new run attempt and
+never replaces the first attempt's verdict. A red census is never rerun on
+unchanged inputs: each red is diagnosed as product, test/detector, harness or
+infra and resolved by a concrete repair and a fresh census, or listed as a named
+red. The one exception: a census whose every red verdict is machine-classified
+INFRA or INCOMPLETE (missing slice artifact, runner loss, API error before the
+first model turn) may be re-dispatched once as a new run, and both runs are
+reported. Changing any `EVAL_POLICY` constant after seeing census results needs
+Garry's re-approval, a `version` bump and a fresh census;
+`test/periodic-exclude-policy.test.ts` pins the approved values.
+
+**Quarantine** (`CASE_QUARANTINE`, same file). An entry needs: a per-trial rate
+below 95% over at least 10 post-policy trials of the case's current input
+identity (pre-policy backfill may justify only an initial entry, labeled as
+such); a written diagnosis in `reason` whose `failureClass` is `detector`,
+`harness` or `model-latency` (a product defect is fixed or listed as a named
+red, never quarantined); unchanged case touchfiles in the change that adds it;
+and an owner, tracking pointer, `enteredAt` date and measurable `exit`. A
+quarantined case still runs its full panel and reports in every lane but cannot
+fail it, except on a hard break (0 of n) or a contract violation, and it never
+counts as passing coverage. At most 10% of a blocking tier (gate, periodic) may
+be quarantined. The weekly report fails when an entry passes its exit rule (at
+least 97% over at least 10 trials) without being removed, when an entry is 8
+weekly runs old, or when a tier is over its cap.
+
+**Pass-rate history** (`bun run eval:pass-rates`, `scripts/eval-flake-rank.ts`).
+It reads the `trial-outcomes` artifact of the last N completed
+`evals-periodic.yml` runs on the current branch and `main` (flags: `--case`,
+`--runs N`, `--branch`, `--dir`, `--backfill`, `--json`, `--gate`) and prints
+per-case per-trial pass rates with 95% Wilson intervals. A series is one case
+under one input identity, the hash of its own touchfiles minus
+`GLOBAL_TOUCHFILES` (harness edits do not restart it), per model, Claude CLI
+version and policy version; a change starts a new series and older ones stay
+visible. Labels: INCONCLUSIVE below 10 trials, BROKEN when the latest run is
+0/n after a prior interval at or above 95%, FLAKY when failures leave the
+interval straddling 95%, FAILING when the whole interval is below it, PASSING
+otherwise. `--backfill` imports legacy slice artifacts as pre-policy trials
+(first attempt only; a record that names no registry id is listed as
+unattributed, never guessed); they are display-only. `--gate` (the weekly
+report) fails with ACTION REQUIRED, on post-policy trials of the current series
+only, when a non-quarantined blocking case meets the entry rule (proposing an
+entry), when a `rule` case does (rule case behaving like behavior: fix or
+reclassify), when a blocking case's current identity is significantly below its
+previous one (one-sided Fisher exact, α = 0.05, at least 6 trials each side,
+Holm-controlled across the cases tested), and on the quarantine rules above.
+History that cannot be fetched fails the gate closed.
+
+**The arithmetic.** With per-trial pass rate p, the chance a single case goes
+red (a false red while the product works, the catch rate once it has
+regressed):
+
+| p | 1 trial | 2-of-3 panel |
+|---|---|---|
+| 0.99 | 1.0% | 0.03% |
+| 0.95 | 5.0% | 0.72% |
+| 0.90 | 10.0% | 2.8% |
+| 0.70 | 30.0% | 21.6% |
+| 0.30 | 70.0% | 78.4% |
+
+The panel removes most false reds at healthy rates, but it catches a 0.95 → 0.70
+regression in one run only 21.6% of the time (a single trial 30%, retry-until-green
+3%), so drift detection is the history rule's job, not the per-run verdict's.
+The Fisher alarm is weak at the minimum sample (5.4% power for 0.95 → 0.70 at
+6 trials a side), and ten straight passes still leave a 72% Wilson lower bound:
+after this policy lands, every series starts INCONCLUSIVE.
+
+A lane is all green with probability Π p_rule × Π P(≥2 of 3 | p_behavior) ×
+Π p_judge. For the current registry (PR gate worst case: 107 rule cases and 24
+judges; weekly census: 190 rule, 22 behavior and 25 judge verdicts), with rule
+and judge verdicts at p_rule:
+
+| p_rule | full PR gate | weekly, behavior p = 0.90 | 0.95 | 0.97 |
+|---|---|---|---|---|
+| 0.99 | 26.8% | 6.2% | 9.8% | 10.9% |
+| 0.995 | 51.9% | 18.2% | 29.0% | 32.1% |
+| 0.999 | 87.7% | 43.2% | 68.7% | 76.1% |
+
+The rule term dominates: a green lane on a working product needs rule cases to
+be near-deterministic (0.999), which is why failing detectors are converted to
+outcome checks and product defects are fixed or named, and why each census
+reports its expected lane false-red from the measured rates.
+
**Timeout policy.** Paid tests use the tiers in
`test/helpers/eval-budgets.ts` (JUDGE/CAPTURE/CAPTURE_LONG/PTY/PTY_LONG);
`test/eval-budgets-policy.test.ts` pins that every tier fits the shard wall
@@ -356,60 +492,63 @@ minus overhead and ratchets raw literals. Budget above the wall is fiction.
No paid test may exceed the ordinary tiers.
`FINDING_RETRY_BUDGETS` also registers the CEO split-overflow and Eng
-multi-finding batching files. Each retains its 25-minute case deadline and one
-retry in a 52-minute shard wall, including two minutes for cleanup. No per-case budget grows. Overlay wrappers
+multi-finding batching files. Each retains its 25-minute case deadline and runs
+once (paid evals never retry) in a 27-minute shard wall including two minutes for
+cleanup. No per-case budget grows. Overlay wrappers
have a 1,830-second minimum shard wall and run without Bun retries; see the
[overlay contract](OVERLAY_BENCHMARK_CONTRACT.md) for their unchanged work budget.
-The quality file reserves 7,180 seconds for all 28 cases and their existing
-retry, plus cleanup. Each still has 120 seconds of model work. Its 17 workflow
+The quality file reserves its whole-file wall (3,170 seconds) for every case run
+once, plus cleanup. Each still has 120 seconds of model work. Its 17 workflow
judges own their deadline and abort signal, with five seconds for terminal
recording inside a ten-second Bun grace; the other 11 retain their existing
120-second Bun timeout. Late responses cannot create records or cache passes.
-The ship documentation file reserves 10,920 seconds for five 600-second cases and
-eight 300-second fault cases, each with one retry, plus cleanup. The standalone
+The ship documentation file reserves 4,920 seconds for four 600-second cases and
+eight 300-second fault cases, run once, plus cleanup; in CI each case runs as its
+own shard. The standalone
documentation child retains its 600-second case. The five review/ship explorer
-cases reserve 3,270 seconds including their existing retry and finalization grace.
+cases reserve 1,695 seconds, run once, including finalization grace.
These are whole-file supervision limits, not additional model work per case.
-The shared-library path file reserves 3,720 seconds for its three serial
-600-second cases, each with one retry, plus 120 seconds for cleanup. Its
+The shared-library path file reserves 1,920 seconds for its three serial
+600-second cases, run once, plus 120 seconds for cleanup; in CI each case runs as
+its own shard with a 720-second wall (a registered file's case shard supervises
+`caseMs` times its allowed attempts plus the reserve). Its
registered budget keeps the file in its own shard and binds the expected wall
to both the saved plan and the execution receipt; missing or stale budget
-records fail reconciliation. Case deadlines, model budgets and retries do not grow.
+records fail reconciliation. Case deadlines and model budgets do not grow.
`resolvePaidShardBudget(files, overrideMs?)` is the canonical per-job resolver.
Each registered finding file and each overlay wrapper requires its
own shard, even with `--files-per-shard` above one. Mixed or multi-file overlay
-jobs are rejected so ordinary files retain their configured retries. An explicit
+jobs are rejected. An explicit
CLI `--timeout`, `EVALS_SHARD_TIMEOUT_MS`, or API `timeoutMs` still wins for these
policies, including a lower cap; overlay overrides below their minimum are rejected.
Planner entries and execution results record the effective wall,
its source and policy identifier. Custom drivers must resolve each job instead
of passing their ordinary 1800-second default as an explicit cap;
their outer controller/detach wall must also cover the allocated work and cleanup.
-The current paid census has 105 files: 47 gate-tier and 71 periodic-tier.
-`eval:bg:pr` and `eval:bg:periodic` have 92820/67380-second outer caps; the PR
+The paid census counts are printed by `--list` for each tier.
+`eval:bg:pr` and `eval:bg:periodic` have 92820/67380-second outer caps, above their recomputed floors (PR fallback 78,425 s, periodic 33,821 s including the trial shards); the PR
wrapper covers a full-gate fallback at its default two workers. The broad gate
-wrapper reserves 49320 seconds, and release reserves 116700 seconds for both
-tiers. Legacy monolithic
-`eval:bg`/`eval:bg:all` retain their shorter 5400/7200-second caps and do not
-promise every registered retry; use the sharded periodic path for this policy.
+wrapper reserves 49320 seconds (floor 21,725 s), and release reserves 116700 seconds for both
+tiers; free tests recompute each floor from the live shard census, case shards
+included. Legacy monolithic
+`eval:bg`/`eval:bg:all` retain their shorter 5400/7200-second caps; use the
+sharded periodic path for complete coverage.
-Periodic CI plans `--slices 7`. When overlays are selected, the seventh is
-reserved for their serial wrappers; registered finding files are distributed
-across the remaining ordinary slices by their supervised walls. Each slice job
-has a 360-minute cap. Reconciliation rejects missing, duplicated or misplaced
-registered work and absent budget records. The weekly gate census has a
-352-minute cap across seven single-worker slices with at most four running at
-once. Its longest current work wall is 272 minutes. PR slices retain seven
-two-worker slices with a 265-minute cap for their 212-minute work wall plus
-setup. Free supervision tests
-verify these bounds against the complete current census, configured retries,
-and setup reserve. Ordinary paid tiers and the default 1800-second
-shard wall remain unchanged; the registered and overlay policies above supply
-exceptions, and unregistered over-ceiling tests still fail policy checks.
+CI plans with `--slice-budget 540 --jobs 2` for the PR gate, the periodic census
+and the weekly gate census (the gate census also `--skip-judges`), and
+`--slice-budget 1 --jobs 1` (one file per runner) for marathon. The live plans
+must fit their workflow's `max-parallel` so every slice starts at once, and
+`ciTimeoutMinutes` must stay within 360; `test/evals-workflow-wiring.test.ts`
+recomputes both from the complete census. Reconciliation rejects missing,
+duplicated or misplaced registered work, absent budget records, case shards that
+did not execute exactly their case, and reused results outside the PR profile.
+Ordinary paid tiers and the default 1800-second shard wall remain unchanged; the
+registered and overlay policies above supply exceptions, and unregistered
+over-ceiling tests still fail policy checks.
Session timeouts are two-phase: a silent API dies at the startup grace (90s
local / 300s CI floor, distinct exit reason `timeout_startup`) and the work
diff --git a/document-release/SKILL.md b/document-release/SKILL.md
index 43f702aff..667b2cd26 100644
--- a/document-release/SKILL.md
+++ b/document-release/SKILL.md
@@ -426,7 +426,7 @@ Make factual updates directly; ask about risky or subjective decisions in standa
## Ship-owned documentation mode
-With a ship candidate, require the actual spawned marker and audit-scope rules below.
+With a ship candidate, follow audit-scope's inputs, steps and JSON result below.
Missing marking/inputs/assets returns `blocked`, never standalone execution. Ship
authority overrides generic spawned recommendations and standalone steps.
@@ -439,8 +439,8 @@ authority overrides generic spawned recommendations and standalone steps.
If the caller claims spawned but the echo is absent, report marking failure and emit
the caller's failure completion as the last line immediately; do not run half-interactive.
Otherwise stay interactive without the marker. Outside ship-owned mode, spawned gates
-auto-choose the RECOMMENDED option, record it in the completion report, and continue:
-never call AskUserQuestion or stop for a prose answer. The NEVER-do invariants below do
+auto-choose the RECOMMENDED option, record it in the completion report, and continue
+through Step 9: never call AskUserQuestion or stop for a prose answer. The NEVER-do invariants below do
not relax: skip any recommendation that rewrites CHANGELOG or changes VERSION and
record why. Step 8 and cross-model review refer to this rule; narrower caller scope wins.
@@ -489,10 +489,10 @@ DOC_DIFF_BASE=$(git merge-base origin/ HEAD 2>/dev/null || git merge-base
echo "DOC_DIFF_BASE: $DOC_DIFF_BASE"
```
-1. Check the current branch. In standalone mode, if on the base branch, **abort**: "You're on the base branch. Run from a feature branch." A ship-owned read-only store audit uses its supplied source scope instead.
+1. Check the current branch. In standalone mode, if on the base branch, **abort**: "You're on the base branch. Run from a feature branch." Ship-owned mode skips this gate.
-2. Gather the diff. In ship-owned mode, also read `git diff --cached`, `git diff`,
- and selected new-file content against the supplied base, not HEAD alone.
+2. Gather the diff. In ship-owned mode, `` is the supplied base SHA; also
+ read `git diff --cached`, `git diff` and the candidate's selected new files.
```bash
git diff HEAD --stat
@@ -547,16 +547,16 @@ Use these definitions:
- **Tutorial** — learning-oriented: step-by-step walkthrough for newcomers (getting started guides)
- **Explanation** — understanding-oriented: "why this works this way" (ARCHITECTURE decisions, design rationale)
-3. **Output the coverage map.** Items with zero coverage are **critical gaps** — flag them for
- Step 3. Items with reference-only coverage are **common gaps** — note them for the PR body.
+3. **Output the coverage map.** Items with zero coverage are **critical gaps**; items with
+ reference-only coverage are **common gaps**. Report both as documentation debt.
4. **Architecture diagram drift detection.** If ARCHITECTURE.md (or any doc) contains ASCII
diagrams or Mermaid blocks, extract entity names (modules, services, data flows) from the
diagrams. Cross-reference against the diff. Flag any diagram entities that were renamed,
split, removed, or moved in the code.
-The coverage map feeds into Steps 2-3 (what to audit and fix) and Step 9 (documentation debt
-summary in the PR body). Do NOT auto-generate missing documentation pages — flag gaps only.
+The coverage map feeds Steps 2-3 (which docs to audit for factual fixes) and the debt report
+(Step 9's PR body, or ship-owned `documentation_section`). Do NOT auto-generate missing documentation pages — flag gaps only.
When significant gaps are found, suggest running `/document-generate` to fill them.
---
diff --git a/document-release/SKILL.md.tmpl b/document-release/SKILL.md.tmpl
index a7f65089e..c6c6a1d64 100644
--- a/document-release/SKILL.md.tmpl
+++ b/document-release/SKILL.md.tmpl
@@ -38,7 +38,7 @@ Make factual updates directly; ask about risky or subjective decisions in standa
## Ship-owned documentation mode
-With a ship candidate, require the actual spawned marker and audit-scope rules below.
+With a ship candidate, follow audit-scope's inputs, steps and JSON result below.
Missing marking/inputs/assets returns `blocked`, never standalone execution. Ship
authority overrides generic spawned recommendations and standalone steps.
@@ -50,8 +50,8 @@ authority overrides generic spawned recommendations and standalone steps.
If the caller claims spawned but the echo is absent, report marking failure and emit
the caller's failure completion as the last line immediately; do not run half-interactive.
Otherwise stay interactive without the marker. Outside ship-owned mode, spawned gates
-auto-choose the RECOMMENDED option, record it in the completion report, and continue:
-never call AskUserQuestion or stop for a prose answer. The NEVER-do invariants below do
+auto-choose the RECOMMENDED option, record it in the completion report, and continue
+through Step 9: never call AskUserQuestion or stop for a prose answer. The NEVER-do invariants below do
not relax: skip any recommendation that rewrites CHANGELOG or changes VERSION and
record why. Step 8 and cross-model review refer to this rule; narrower caller scope wins.
@@ -92,10 +92,10 @@ DOC_DIFF_BASE=$(git merge-base origin/ HEAD 2>/dev/null || git merge-base
echo "DOC_DIFF_BASE: $DOC_DIFF_BASE"
```
-1. Check the current branch. In standalone mode, if on the base branch, **abort**: "You're on the base branch. Run from a feature branch." A ship-owned read-only store audit uses its supplied source scope instead.
+1. Check the current branch. In standalone mode, if on the base branch, **abort**: "You're on the base branch. Run from a feature branch." Ship-owned mode skips this gate.
-2. Gather the diff. In ship-owned mode, also read `git diff --cached`, `git diff`,
- and selected new-file content against the supplied base, not HEAD alone.
+2. Gather the diff. In ship-owned mode, `` is the supplied base SHA; also
+ read `git diff --cached`, `git diff` and the candidate's selected new files.
```bash
git diff HEAD --stat
@@ -150,16 +150,16 @@ Use these definitions:
- **Tutorial** — learning-oriented: step-by-step walkthrough for newcomers (getting started guides)
- **Explanation** — understanding-oriented: "why this works this way" (ARCHITECTURE decisions, design rationale)
-3. **Output the coverage map.** Items with zero coverage are **critical gaps** — flag them for
- Step 3. Items with reference-only coverage are **common gaps** — note them for the PR body.
+3. **Output the coverage map.** Items with zero coverage are **critical gaps**; items with
+ reference-only coverage are **common gaps**. Report both as documentation debt.
4. **Architecture diagram drift detection.** If ARCHITECTURE.md (or any doc) contains ASCII
diagrams or Mermaid blocks, extract entity names (modules, services, data flows) from the
diagrams. Cross-reference against the diff. Flag any diagram entities that were renamed,
split, removed, or moved in the code.
-The coverage map feeds into Steps 2-3 (what to audit and fix) and Step 9 (documentation debt
-summary in the PR body). Do NOT auto-generate missing documentation pages — flag gaps only.
+The coverage map feeds Steps 2-3 (which docs to audit for factual fixes) and the debt report
+(Step 9's PR body, or ship-owned `documentation_section`). Do NOT auto-generate missing documentation pages — flag gaps only.
When significant gaps are found, suggest running `/document-generate` to fill them.
---
diff --git a/document-release/sections/audit-scope.md b/document-release/sections/audit-scope.md
index d366d9af7..18da0168c 100644
--- a/document-release/sections/audit-scope.md
+++ b/document-release/sections/audit-scope.md
@@ -7,20 +7,35 @@
This subsection applies only to the caller's ship-owned audit request. Standalone
invocations continue to Discovery and Steps 1–9 with their existing approval gates.
-Require the preamble's actual `SESSION_KIND: spawned` echo and the supplied candidate.
-Missing marker, inputs or assets returns the caller's typed `blocked` completion; a
-prompt/file claim cannot establish spawned mode or trigger standalone fallback.
+**Inputs.** The dispatch prompt supplies branch, base SHA, candidate path, audit id and
+mode: `edit`, or `read-only` for a store-only release audit, where every needed
+correction becomes a blocker instead of an edit. Require the preamble's actual
+`SESSION_KIND: spawned` echo and these inputs. Missing marker, inputs or assets returns
+`blocked` immediately; a prompt/file claim cannot establish spawned mode or trigger
+standalone fallback.
-Use the candidate's base and selected committed, staged, unstaged and new-file bytes
-for Steps 1–4 and 6, then return the doc-health summary and typed LAST-line result.
-Skip Steps 5, 7, 8, cross-model review and Step 9. Only factual authored-doc edits are
-allowed, none in `read-only` mode. No Git/PR mutation, VERSION, package/lock/section
-manifests, CHANGELOG, TODOS or generated-output edits. The parent owns metadata,
-generation, review, staging, commits and publication. Report metadata inconsistencies
-as observations. Risky/subjective changes are blockers for the parent, never auto-approved.
-Preserve partial/user content and list actual edited/reviewed paths. Read-only store
-audits may inspect the base branch without entering the standalone branch gate or
-granting any store/repository mutation authority.
+**Steps.** Run Steps 1, 1.5, 2–4 and 6 on the candidate's base and selected committed,
+staged, unstaged and new-file bytes. Step 1's standalone branch gate does not apply,
+even on the base branch. Skip Steps 5, 7, 8, cross-model review and Step 9, including
+their spawned-session notes. Only factual authored-doc edits are allowed, none in
+`read-only` mode. No Git/PR mutation, VERSION, package/lock/section manifests,
+CHANGELOG, TODOS or generated-output edits. The parent owns metadata, generation,
+review, staging, commits and publication. Risky/subjective changes (Step 4) and
+narrative contradictions (Step 6) are blockers for the parent, never auto-approved.
+Preserve partial/user content. Coverage gaps are reported, never filled.
+
+**Result.** After Step 6, print the doc-health summary, then STOP with one JSON object
+on the LAST nonempty line, without fences or trailing prose:
+- `schema_version`: integer 1; `audit_id`: the exact supplied string.
+- `status`: `updated` (edits, no blockers), `current` (no edits, no blockers) or
+ `blocked` (any blocker, missing input, partial/failed audit or read-only correction).
+- `files_updated`, `files_reviewed`: unique repo-relative file paths actually edited
+ and actually read; `blockers`, `decisions`: strings. Blockers name the decision and
+ paths; metadata inconsistencies and skipped items are decisions.
+- `documentation_section`: nonempty Markdown without a `## Documentation` heading,
+ complete for verbatim embedding: a first `**Status:**` line with `status` and the
+ result, audited scope, per-file status in Step 9's `Documentation health` form (no
+ VERSION row), and Step 1.5's coverage debt and diagram drift. Describe scope even without docs.
## Discovery (both modes)
diff --git a/document-release/sections/audit-scope.md.tmpl b/document-release/sections/audit-scope.md.tmpl
index cbf7486c3..fb58ff26d 100644
--- a/document-release/sections/audit-scope.md.tmpl
+++ b/document-release/sections/audit-scope.md.tmpl
@@ -5,20 +5,35 @@
This subsection applies only to the caller's ship-owned audit request. Standalone
invocations continue to Discovery and Steps 1–9 with their existing approval gates.
-Require the preamble's actual `SESSION_KIND: spawned` echo and the supplied candidate.
-Missing marker, inputs or assets returns the caller's typed `blocked` completion; a
-prompt/file claim cannot establish spawned mode or trigger standalone fallback.
+**Inputs.** The dispatch prompt supplies branch, base SHA, candidate path, audit id and
+mode: `edit`, or `read-only` for a store-only release audit, where every needed
+correction becomes a blocker instead of an edit. Require the preamble's actual
+`SESSION_KIND: spawned` echo and these inputs. Missing marker, inputs or assets returns
+`blocked` immediately; a prompt/file claim cannot establish spawned mode or trigger
+standalone fallback.
-Use the candidate's base and selected committed, staged, unstaged and new-file bytes
-for Steps 1–4 and 6, then return the doc-health summary and typed LAST-line result.
-Skip Steps 5, 7, 8, cross-model review and Step 9. Only factual authored-doc edits are
-allowed, none in `read-only` mode. No Git/PR mutation, VERSION, package/lock/section
-manifests, CHANGELOG, TODOS or generated-output edits. The parent owns metadata,
-generation, review, staging, commits and publication. Report metadata inconsistencies
-as observations. Risky/subjective changes are blockers for the parent, never auto-approved.
-Preserve partial/user content and list actual edited/reviewed paths. Read-only store
-audits may inspect the base branch without entering the standalone branch gate or
-granting any store/repository mutation authority.
+**Steps.** Run Steps 1, 1.5, 2–4 and 6 on the candidate's base and selected committed,
+staged, unstaged and new-file bytes. Step 1's standalone branch gate does not apply,
+even on the base branch. Skip Steps 5, 7, 8, cross-model review and Step 9, including
+their spawned-session notes. Only factual authored-doc edits are allowed, none in
+`read-only` mode. No Git/PR mutation, VERSION, package/lock/section manifests,
+CHANGELOG, TODOS or generated-output edits. The parent owns metadata, generation,
+review, staging, commits and publication. Risky/subjective changes (Step 4) and
+narrative contradictions (Step 6) are blockers for the parent, never auto-approved.
+Preserve partial/user content. Coverage gaps are reported, never filled.
+
+**Result.** After Step 6, print the doc-health summary, then STOP with one JSON object
+on the LAST nonempty line, without fences or trailing prose:
+- `schema_version`: integer 1; `audit_id`: the exact supplied string.
+- `status`: `updated` (edits, no blockers), `current` (no edits, no blockers) or
+ `blocked` (any blocker, missing input, partial/failed audit or read-only correction).
+- `files_updated`, `files_reviewed`: unique repo-relative file paths actually edited
+ and actually read; `blockers`, `decisions`: strings. Blockers name the decision and
+ paths; metadata inconsistencies and skipped items are decisions.
+- `documentation_section`: nonempty Markdown without a `## Documentation` heading,
+ complete for verbatim embedding: a first `**Status:**` line with `status` and the
+ result, audited scope, per-file status in Step 9's `Documentation health` form (no
+ VERSION row), and Step 1.5's coverage debt and diagram drift. Describe scope even without docs.
## Discovery (both modes)
diff --git a/document-release/sections/release-body.md b/document-release/sections/release-body.md
index 3f1fc314c..424610a46 100644
--- a/document-release/sections/release-body.md
+++ b/document-release/sections/release-body.md
@@ -2,8 +2,8 @@
## Step 2: Per-File Documentation Audit
-**Ship-owned documentation mode:** execute Steps 2–4 and 6 only, under the skeleton's
-audit/edit/result boundary. Then return the caller's typed completion; all standalone
+**Ship-owned documentation mode:** after Steps 1 and 1.5, execute Steps 2–4 and 6 only,
+under audit-scope's edit boundary, then return its JSON result; all standalone
metadata, review, commit and PR steps below remain unavailable to this child.
Read each documentation file and cross-reference it against the diff. Use these generic heuristics
@@ -131,8 +131,8 @@ After auditing each file individually, do a cross-doc consistency pass:
In ship-owned mode, protected metadata/manifests stay untouched even for factual
inconsistencies, and narrative contradictions return as blockers. This is the last
-ship-child step: output the doc-health summary and typed completion, then STOP. A
-partial audit or unresolved required correction is `blocked`, never `current`.
+ship-child step: output the doc-health summary and audit-scope's JSON result, then
+STOP. A partial audit or unresolved required correction is `blocked`, never `current`.
---
@@ -193,7 +193,7 @@ git diff HEAD -- VERSION
**Spawned sessions** (per the spawned-dispatch contract at the top of this skill): the
recommendation flips — choose C (leave version as-is) and record the uncovered scope in
- your completion report (the `decisions` array when dispatched from /ship).
+ your completion report. Ship-owned children stopped at Step 6 and never reach this step.
A spawned run must never change VERSION: the dispatching workflow owns version numbering.
The key insight: a VERSION bump set for "feature A" should not silently absorb "feature B"
@@ -211,7 +211,7 @@ not an opt-in. The user turns it off only by asking explicitly
**Spawned-session skip** (per the spawned-dispatch contract at the top of this skill): in a
spawned session, skip this entire section — the dispatching workflow owns its own review
passes, and the apply gate below needs a human. Note the skip in the upcoming Step 9 doc
-health summary and continue to Step 9.
+health summary and continue to Step 9. Ship-owned children already stopped at Step 6.
**Preflight — decide whether and how the doc review runs:**
diff --git a/document-release/sections/release-body.md.tmpl b/document-release/sections/release-body.md.tmpl
index f7d6f9f41..25ca63d8d 100644
--- a/document-release/sections/release-body.md.tmpl
+++ b/document-release/sections/release-body.md.tmpl
@@ -1,7 +1,7 @@
## Step 2: Per-File Documentation Audit
-**Ship-owned documentation mode:** execute Steps 2–4 and 6 only, under the skeleton's
-audit/edit/result boundary. Then return the caller's typed completion; all standalone
+**Ship-owned documentation mode:** after Steps 1 and 1.5, execute Steps 2–4 and 6 only,
+under audit-scope's edit boundary, then return its JSON result; all standalone
metadata, review, commit and PR steps below remain unavailable to this child.
Read each documentation file and cross-reference it against the diff. Use these generic heuristics
@@ -129,8 +129,8 @@ After auditing each file individually, do a cross-doc consistency pass:
In ship-owned mode, protected metadata/manifests stay untouched even for factual
inconsistencies, and narrative contradictions return as blockers. This is the last
-ship-child step: output the doc-health summary and typed completion, then STOP. A
-partial audit or unresolved required correction is `blocked`, never `current`.
+ship-child step: output the doc-health summary and audit-scope's JSON result, then
+STOP. A partial audit or unresolved required correction is `blocked`, never `current`.
---
@@ -191,7 +191,7 @@ git diff HEAD -- VERSION
**Spawned sessions** (per the spawned-dispatch contract at the top of this skill): the
recommendation flips — choose C (leave version as-is) and record the uncovered scope in
- your completion report (the `decisions` array when dispatched from /ship).
+ your completion report. Ship-owned children stopped at Step 6 and never reach this step.
A spawned run must never change VERSION: the dispatching workflow owns version numbering.
The key insight: a VERSION bump set for "feature A" should not silently absorb "feature B"
diff --git a/gstack/llms.txt b/gstack/llms.txt
index 20e42a661..0aa42eedf 100644
--- a/gstack/llms.txt
+++ b/gstack/llms.txt
@@ -140,7 +140,7 @@ Run with `browse [args]`. Full reference: `browse/SKILL.md`.
- `text [selector|@ref]`: Cleaned visible page text, or cleaned text for a CSS selector/@ref when one is provided
### Server
-- `connect`: Launch headed Chromium with Chrome extension
+- `connect [--supervise]`: Launch headed Chromium with Chrome extension; --supervise keeps the CLI attached and respawns a crashed server
- `disconnect`: Disconnect headed browser, return to headless mode
- `focus [@ref]`: Bring headed browser window to foreground (macOS)
- `handoff [message]`: Open visible Chrome at current page for user takeover
diff --git a/hosts/claude/hooks/auq-error-fallback-hook.ts b/hosts/claude/hooks/auq-error-fallback-hook.ts
index d46ab87dd..22e40bae5 100755
--- a/hosts/claude/hooks/auq-error-fallback-hook.ts
+++ b/hosts/claude/hooks/auq-error-fallback-hook.ts
@@ -115,7 +115,7 @@ export function sessionKind(cwd?: string): 'spawned' | 'headless' | 'interactive
timeout: 3000,
cwd: cwd && fs.existsSync(cwd) ? cwd : undefined,
});
- const out = (res.stdout || '').trim();
+ const out = String(res.stdout || '').trim();
if (out === 'spawned' || out === 'headless' || out === 'interactive') return out;
} catch (e) {
logHookError(`sessionKind failed: ${(e as Error).message}`);
diff --git a/lib/aside-render.ts b/lib/aside-render.ts
index ff7aa5d76..5a3ff76de 100644
--- a/lib/aside-render.ts
+++ b/lib/aside-render.ts
@@ -296,7 +296,7 @@ export function serveDir(root: string, nonce: string = randomBytes(16).toString(
// ─── Async spawn (keeps the loopback server's event loop free) ────────────────
async function runProc(cmd: string, args: string[], timeoutMs: number): Promise<{ code: number | null; stdout: string; stderr: string; error?: string }> {
- let child: ReturnType;
+ let child: Bun.Subprocess<'ignore', 'pipe', 'pipe'>;
try {
child = Bun.spawn([cmd, ...args], { stdout: 'pipe', stderr: 'pipe', stdin: 'ignore' });
} catch (e) {
@@ -608,7 +608,7 @@ export const NO_BROWSER_HELP = "open the Aside app (macOS 15+, aside.com), or ru
export type EngineChoice =
| { engine: 'aside'; version: string }
| { engine: 'browse'; bin: string }
- | { engine: null; probe: AsideProbe; error: string };
+ | { engine: null; probe: Extract; error: string };
let chosen: EngineChoice | undefined;
diff --git a/lib/cso/.prettierrc.json b/lib/cso/.prettierrc.json
new file mode 100644
index 000000000..8d0a27da2
--- /dev/null
+++ b/lib/cso/.prettierrc.json
@@ -0,0 +1,6 @@
+{
+ "printWidth": 110,
+ "singleQuote": true,
+ "trailingComma": "all",
+ "semi": true
+}
diff --git a/lib/cso/admission.ts b/lib/cso/admission.ts
index 5c1a891e9..7d9c85d5a 100644
--- a/lib/cso/admission.ts
+++ b/lib/cso/admission.ts
@@ -6,160 +6,468 @@ import { CsoError, sha256 } from './contracts';
import { discardAtomicNoReplaceTemp, recoverAtomicNoReplaceJson, secureDirectory } from './state';
import { atomicWriteSync } from '../fs-atomic';
-export const GROUP_LIMITS = { cpu: 2, memoryMiB: 4096, pids: 256, writableMiB: 2048, outputBytes: 1024 * 1024 } as const;
+export const GROUP_LIMITS = {
+ cpu: 2,
+ memoryMiB: 4096,
+ pids: 256,
+ writableMiB: 2048,
+ outputBytes: 1024 * 1024,
+} as const;
export const ROLE_LIMITS = {
- anchor: {cpu:.05,memoryMiB:64,pids:8,writableMiB:16},
- app: {cpu:.85,memoryMiB:2304,pids:96,writableMiB:1264},
- verifier: {cpu:.55,memoryMiB:512,pids:32,writableMiB:256},
- tests: {cpu:.55,memoryMiB:1280,pids:64,writableMiB:1024},
- postgres: {cpu:.25,memoryMiB:1024,pids:96,writableMiB:512},
- browser: {cpu:.30,memoryMiB:512,pids:16,writableMiB:256},
+ anchor: { cpu: 0.05, memoryMiB: 64, pids: 8, writableMiB: 16 },
+ app: { cpu: 0.85, memoryMiB: 2304, pids: 96, writableMiB: 1264 },
+ verifier: { cpu: 0.55, memoryMiB: 512, pids: 32, writableMiB: 256 },
+ tests: { cpu: 0.55, memoryMiB: 1280, pids: 64, writableMiB: 1024 },
+ postgres: { cpu: 0.25, memoryMiB: 1024, pids: 96, writableMiB: 512 },
+ browser: { cpu: 0.3, memoryMiB: 512, pids: 16, writableMiB: 256 },
} as const;
export type Role = keyof typeof ROLE_LIMITS;
-export interface Lease { endpoint: string; slot: number; path: string; runId: string; ownerPid: number; expiresAt: number; token:string; supervised:boolean }
-function alive(pid: number): boolean { try { process.kill(pid,0); return true; } catch { return false; } }
-function processIdentity(pid:number):string|undefined{if(process.platform!=='linux')return;try{const raw=fs.readFileSync(`/proc/${pid}/stat`,'utf8'),tail=raw.slice(raw.lastIndexOf(')')+2).trim().split(/\s+/);return /^\d+$/.test(tail[19]??'')?`linux:${tail[19]}`:undefined;}catch{return;}}
-function sameDirectory(left:fs.Stats,right:fs.Stats):boolean{return left.dev===right.dev&&left.ino===right.ino&&left.uid===right.uid&&left.mode===right.mode;}
-function sameFile(left:fs.Stats,right:fs.Stats):boolean{return left.dev===right.dev&&left.ino===right.ino&&left.uid===right.uid&&left.mode===right.mode&&left.nlink===right.nlink;}
-function privateDirectory(path:string,label:string):fs.Stats{const stat=fs.lstatSync(path);if(!stat.isDirectory()||stat.isSymbolicLink()||(process.getuid&&stat.uid!==process.getuid())||(stat.mode&0o077)!==0)throw new CsoError('UNSAFE_PATH',`${label} is not a private owned directory`);return stat;}
-function privateFile(path:string,label:string):fs.Stats{const stat=fs.lstatSync(path);if(!stat.isFile()||stat.isSymbolicLink()||stat.nlink!==1||(process.getuid&&stat.uid!==process.getuid())||(stat.mode&0o077)!==0||stat.size>1024*1024)throw new CsoError('UNSAFE_PATH',`${label} is not a private regular file`);return stat;}
-type Claim={path:string;token:string;identity:fs.Stats;pid:number;processIdentity:string|null};
-type ClaimOwner={pid:number;processIdentity:string|null;token:string;createdAt:number};
-function validateClaimOwner(value:unknown,expectedToken?:string,publisherPid?:number):ClaimOwner{
- if(!value||typeof value!=='object'||Array.isArray(value))throw new CsoError('INCOMPATIBLE_INPUT','Reproduction recovery owner is invalid');
- const owner=value as Record;
- if(Object.keys(owner).sort().join(',')!=='createdAt,pid,processIdentity,token'||!Number.isSafeInteger(owner.pid)||Number(owner.pid)<=1||
- typeof owner.token!=='string'||!/^[a-f0-9]{32}$/.test(owner.token)||(expectedToken!==undefined&&owner.token!==expectedToken)||
- !Number.isFinite(owner.createdAt)||Number(owner.createdAt)<0||!(owner.processIdentity===null||(typeof owner.processIdentity==='string'&&/^linux:\d+$/.test(owner.processIdentity)))||
- (publisherPid!==undefined&&Number(owner.pid)!==publisherPid))throw new CsoError('INCOMPATIBLE_INPUT','Reproduction recovery owner is invalid');
- return{pid:Number(owner.pid),processIdentity:owner.processIdentity as string|null,token:owner.token,createdAt:Number(owner.createdAt)};
+export interface Lease {
+ endpoint: string;
+ slot: number;
+ path: string;
+ runId: string;
+ ownerPid: number;
+ expiresAt: number;
+ token: string;
+ supervised: boolean;
}
-function inspectClaim(path:string,expectedToken?:string):Claim{
- const before=privateFile(path,'Reproduction recovery claim');if(before.size<=0||before.size>4096)throw new CsoError('UNSAFE_PATH','Reproduction recovery claim has an invalid size');
- let owner:any;try{owner=JSON.parse(fs.readFileSync(path,'utf8'));}catch{throw new CsoError('INCOMPATIBLE_INPUT','Reproduction recovery owner is invalid');}
- const after=privateFile(path,'Reproduction recovery claim');if(!sameFile(before,after))throw new CsoError('INCOMPATIBLE_INPUT','Reproduction recovery owner is invalid');
- owner=validateClaimOwner(owner,expectedToken);
- return{path,token:owner.token,identity:after,pid:owner.pid,processIdentity:owner.processIdentity};
+function alive(pid: number): boolean {
+ try {
+ process.kill(pid, 0);
+ return true;
+ } catch {
+ return false;
+ }
}
-function releaseClaim(claim:Claim):void{
- const current=inspectClaim(claim.path,claim.token);if(!sameFile(current.identity,claim.identity))throw new CsoError('PERSISTENCE_FAILED','Reproduction recovery ownership changed');
- const final=privateFile(claim.path,'Reproduction recovery claim');if(!sameFile(final,claim.identity))throw new CsoError('PERSISTENCE_FAILED','Reproduction recovery ownership changed');
+function processIdentity(pid: number): string | undefined {
+ if (process.platform !== 'linux') return;
+ try {
+ const raw = fs.readFileSync(`/proc/${pid}/stat`, 'utf8'),
+ tail = raw
+ .slice(raw.lastIndexOf(')') + 2)
+ .trim()
+ .split(/\s+/);
+ return /^\d+$/.test(tail[19] ?? '') ? `linux:${tail[19]}` : undefined;
+ } catch {
+ return;
+ }
+}
+function sameDirectory(left: fs.Stats, right: fs.Stats): boolean {
+ return (
+ left.dev === right.dev && left.ino === right.ino && left.uid === right.uid && left.mode === right.mode
+ );
+}
+function sameFile(left: fs.Stats, right: fs.Stats): boolean {
+ return (
+ left.dev === right.dev &&
+ left.ino === right.ino &&
+ left.uid === right.uid &&
+ left.mode === right.mode &&
+ left.nlink === right.nlink
+ );
+}
+function privateDirectory(path: string, label: string): fs.Stats {
+ const stat = fs.lstatSync(path);
+ if (
+ !stat.isDirectory() ||
+ stat.isSymbolicLink() ||
+ (process.getuid && stat.uid !== process.getuid()) ||
+ (stat.mode & 0o077) !== 0
+ )
+ throw new CsoError('UNSAFE_PATH', `${label} is not a private owned directory`);
+ return stat;
+}
+function privateFile(path: string, label: string): fs.Stats {
+ const stat = fs.lstatSync(path);
+ if (
+ !stat.isFile() ||
+ stat.isSymbolicLink() ||
+ stat.nlink !== 1 ||
+ (process.getuid && stat.uid !== process.getuid()) ||
+ (stat.mode & 0o077) !== 0 ||
+ stat.size > 1024 * 1024
+ )
+ throw new CsoError('UNSAFE_PATH', `${label} is not a private regular file`);
+ return stat;
+}
+type Claim = { path: string; token: string; identity: fs.Stats; pid: number; processIdentity: string | null };
+type ClaimOwner = { pid: number; processIdentity: string | null; token: string; createdAt: number };
+function validateClaimOwner(value: unknown, expectedToken?: string, publisherPid?: number): ClaimOwner {
+ if (!value || typeof value !== 'object' || Array.isArray(value))
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Reproduction recovery owner is invalid');
+ const owner = value as Record;
+ if (
+ Object.keys(owner).sort().join(',') !== 'createdAt,pid,processIdentity,token' ||
+ !Number.isSafeInteger(owner.pid) ||
+ Number(owner.pid) <= 1 ||
+ typeof owner.token !== 'string' ||
+ !/^[a-f0-9]{32}$/.test(owner.token) ||
+ (expectedToken !== undefined && owner.token !== expectedToken) ||
+ !Number.isFinite(owner.createdAt) ||
+ Number(owner.createdAt) < 0 ||
+ !(
+ owner.processIdentity === null ||
+ (typeof owner.processIdentity === 'string' && /^linux:\d+$/.test(owner.processIdentity))
+ ) ||
+ (publisherPid !== undefined && Number(owner.pid) !== publisherPid)
+ )
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Reproduction recovery owner is invalid');
+ return {
+ pid: Number(owner.pid),
+ processIdentity: owner.processIdentity as string | null,
+ token: owner.token,
+ createdAt: Number(owner.createdAt),
+ };
+}
+function inspectClaim(path: string, expectedToken?: string): Claim {
+ const before = privateFile(path, 'Reproduction recovery claim');
+ if (before.size <= 0 || before.size > 4096)
+ throw new CsoError('UNSAFE_PATH', 'Reproduction recovery claim has an invalid size');
+ let owner: any;
+ try {
+ owner = JSON.parse(fs.readFileSync(path, 'utf8'));
+ } catch {
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Reproduction recovery owner is invalid');
+ }
+ const after = privateFile(path, 'Reproduction recovery claim');
+ if (!sameFile(before, after))
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Reproduction recovery owner is invalid');
+ owner = validateClaimOwner(owner, expectedToken);
+ return {
+ path,
+ token: owner.token,
+ identity: after,
+ pid: owner.pid,
+ processIdentity: owner.processIdentity,
+ };
+}
+function releaseClaim(claim: Claim): void {
+ const current = inspectClaim(claim.path, claim.token);
+ if (!sameFile(current.identity, claim.identity))
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction recovery ownership changed');
+ const final = privateFile(claim.path, 'Reproduction recovery claim');
+ if (!sameFile(final, claim.identity))
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction recovery ownership changed');
fs.unlinkSync(claim.path);
}
-function acquireClaim(parent:string,expected:fs.Stats):Claim{
- const path=join(parent,'.recovery'),assertParent=()=>{const current=privateDirectory(parent,'Reproduction lease slot');if(!sameDirectory(expected,current))throw new CsoError('INSUFFICIENT_CAPACITY','Reproduction lease changed during recovery');};
- const recoverPublications=()=>{
- const pattern=/^\.recovery\.tmp\.(\d{1,10})\.[a-f0-9]{8}$/;
- for(const name of fs.readdirSync(parent)){
- const match=name.match(pattern);if(!match)continue;
- const publisherPid=Number(match[1]),temporary=join(parent,name),options={label:'Reproduction recovery claim',maxBytes:4096,
- validate:(value:unknown,pid:number)=>{validateClaimOwner(value,undefined,pid);}};
- assertParent();if(fs.existsSync(path))recoverAtomicNoReplaceJson(path,options);if(fs.existsSync(temporary))discardAtomicNoReplaceTemp(temporary,publisherPid,options);assertParent();
+function acquireClaim(parent: string, expected: fs.Stats): Claim {
+ const path = join(parent, '.recovery'),
+ assertParent = () => {
+ const current = privateDirectory(parent, 'Reproduction lease slot');
+ if (!sameDirectory(expected, current))
+ throw new CsoError('INSUFFICIENT_CAPACITY', 'Reproduction lease changed during recovery');
+ };
+ const recoverPublications = () => {
+ const pattern = /^\.recovery\.tmp\.(\d{1,10})\.[a-f0-9]{8}$/;
+ for (const name of fs.readdirSync(parent)) {
+ const match = name.match(pattern);
+ if (!match) continue;
+ const publisherPid = Number(match[1]),
+ temporary = join(parent, name),
+ options = {
+ label: 'Reproduction recovery claim',
+ maxBytes: 4096,
+ validate: (value: unknown, pid: number) => {
+ validateClaimOwner(value, undefined, pid);
+ },
+ };
+ assertParent();
+ if (fs.existsSync(path)) recoverAtomicNoReplaceJson(path, options);
+ if (fs.existsSync(temporary)) discardAtomicNoReplaceTemp(temporary, publisherPid, options);
+ assertParent();
}
};
- for(let attempt=0;attempt<64;attempt++){
- assertParent();recoverPublications();const token=randomBytes(16).toString('hex');
- try{
- atomicWriteSync(path,JSON.stringify({pid:process.pid,processIdentity:processIdentity(process.pid)??null,token,createdAt:Date.now()})+'\n',{mode:0o600,noReplace:true});
- const claim=inspectClaim(path,token);try{assertParent();}catch(error){try{releaseClaim(claim);}catch{}throw error;}return claim;
- }catch(error:any){
- if(error instanceof CsoError)throw error;
- if(error?.code!=='EEXIST')throw new CsoError('PERSISTENCE_FAILED','Reproduction recovery claim could not be created');
+ for (let attempt = 0; attempt < 64; attempt++) {
+ assertParent();
+ recoverPublications();
+ const token = randomBytes(16).toString('hex');
+ try {
+ atomicWriteSync(
+ path,
+ JSON.stringify({
+ pid: process.pid,
+ processIdentity: processIdentity(process.pid) ?? null,
+ token,
+ createdAt: Date.now(),
+ }) + '\n',
+ { mode: 0o600, noReplace: true },
+ );
+ const claim = inspectClaim(path, token);
+ try {
+ assertParent();
+ } catch (error) {
+ try {
+ releaseClaim(claim);
+ } catch {}
+ throw error;
+ }
+ return claim;
+ } catch (error: any) {
+ if (error instanceof CsoError) throw error;
+ if (error?.code !== 'EEXIST')
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction recovery claim could not be created');
+ }
+ assertParent();
+ const observed = inspectClaim(path),
+ isAlive = alive(observed.pid),
+ identity = isAlive ? processIdentity(observed.pid) : undefined;
+ if (
+ isAlive &&
+ !(
+ typeof observed.processIdentity === 'string' &&
+ identity !== undefined &&
+ identity !== observed.processIdentity
+ )
+ )
+ throw new CsoError('INSUFFICIENT_CAPACITY', 'Another helper is recovering the reproduction lease');
+ try {
+ releaseClaim(observed);
+ } catch (error) {
+ if (error instanceof CsoError && error.code === 'PERSISTENCE_FAILED') continue;
+ throw error;
}
- assertParent();const observed=inspectClaim(path),isAlive=alive(observed.pid),identity=isAlive?processIdentity(observed.pid):undefined;
- if(isAlive&&!(typeof observed.processIdentity==='string'&&identity!==undefined&&identity!==observed.processIdentity))throw new CsoError('INSUFFICIENT_CAPACITY','Another helper is recovering the reproduction lease');
- try{releaseClaim(observed);}catch(error){if(error instanceof CsoError&&error.code==='PERSISTENCE_FAILED')continue;throw error;}
}
- throw new CsoError('INSUFFICIENT_CAPACITY','Reproduction recovery claim changed repeatedly');
+ throw new CsoError('INSUFFICIENT_CAPACITY', 'Reproduction recovery claim changed repeatedly');
}
/** One host-user pool shared by every workspace/state root on this machine. */
-export function machinePoolRoot():string{
- const uid=process.getuid?.()??userInfo().uid;
- return secureDirectory(join(fs.realpathSync(tmpdir()),`gstack-cso-pool-${uid}`));
+export function machinePoolRoot(): string {
+ const uid = process.getuid?.() ?? userInfo().uid;
+ return secureDirectory(join(fs.realpathSync(tmpdir()), `gstack-cso-pool-${uid}`));
}
-function reclaimSlot(path:string,pool:string,slot:number,observed:fs.Stats,expectedToken?:string):boolean{
- let claim:Claim;try{claim=acquireClaim(path,observed);}catch(error){if(error instanceof CsoError&&error.code==='INSUFFICIENT_CAPACITY')return false;throw error;}
- try{const current=privateDirectory(path,'Reproduction lease slot');if(!sameDirectory(observed,current)){releaseClaim(claim);return false;}if(expectedToken){privateFile(join(path,'lease.json'),'Reproduction lease');const lease=JSON.parse(fs.readFileSync(join(path,'lease.json'),'utf8'));if(lease.token!==expectedToken){releaseClaim(claim);return false;}}
- const tomb=join(pool,`.slot-${slot}.stale-${process.pid}-${randomBytes(8).toString('hex')}`);fs.renameSync(path,tomb);const moved=privateDirectory(tomb,'Reproduction lease tomb');if(!sameDirectory(observed,moved))throw new CsoError('SNAPSHOT_RACE','Reproduction lease changed while quarantined');fs.mkdirSync(path,{mode:0o700});releaseClaim({...claim,path:join(tomb,'.recovery')});for(const name of fs.readdirSync(tomb)){if(!['lease.json','lease.token'].includes(name)&&!/^lease\.json\.tmp\.\d+\.[a-f0-9]{8}$/.test(name)&&!/^\.recovery\.tmp\.\d+\.[a-f0-9]{8}$/.test(name))throw new CsoError('UNSAFE_PATH','Stale reproduction lease contains an unexpected object');privateFile(join(tomb,name),'Stale reproduction lease file');fs.unlinkSync(join(tomb,name));}fs.rmdirSync(tomb);return true;
- }catch(error){if(error instanceof CsoError)throw error;return false;}
+function reclaimSlot(
+ path: string,
+ pool: string,
+ slot: number,
+ observed: fs.Stats,
+ expectedToken?: string,
+): boolean {
+ let claim: Claim;
+ try {
+ claim = acquireClaim(path, observed);
+ } catch (error) {
+ if (error instanceof CsoError && error.code === 'INSUFFICIENT_CAPACITY') return false;
+ throw error;
+ }
+ try {
+ const current = privateDirectory(path, 'Reproduction lease slot');
+ if (!sameDirectory(observed, current)) {
+ releaseClaim(claim);
+ return false;
+ }
+ if (expectedToken) {
+ privateFile(join(path, 'lease.json'), 'Reproduction lease');
+ const lease = JSON.parse(fs.readFileSync(join(path, 'lease.json'), 'utf8'));
+ if (lease.token !== expectedToken) {
+ releaseClaim(claim);
+ return false;
+ }
+ }
+ const tomb = join(pool, `.slot-${slot}.stale-${process.pid}-${randomBytes(8).toString('hex')}`);
+ fs.renameSync(path, tomb);
+ const moved = privateDirectory(tomb, 'Reproduction lease tomb');
+ if (!sameDirectory(observed, moved))
+ throw new CsoError('SNAPSHOT_RACE', 'Reproduction lease changed while quarantined');
+ fs.mkdirSync(path, { mode: 0o700 });
+ releaseClaim({ ...claim, path: join(tomb, '.recovery') });
+ for (const name of fs.readdirSync(tomb)) {
+ if (
+ !['lease.json', 'lease.token'].includes(name) &&
+ !/^lease\.json\.tmp\.\d+\.[a-f0-9]{8}$/.test(name) &&
+ !/^\.recovery\.tmp\.\d+\.[a-f0-9]{8}$/.test(name)
+ )
+ throw new CsoError('UNSAFE_PATH', 'Stale reproduction lease contains an unexpected object');
+ privateFile(join(tomb, name), 'Stale reproduction lease file');
+ fs.unlinkSync(join(tomb, name));
+ }
+ fs.rmdirSync(tomb);
+ return true;
+ } catch (error) {
+ if (error instanceof CsoError) throw error;
+ return false;
+ }
}
-function slotControl(pool:string,slot:number):{path:string;stat:fs.Stats}{
- const path=join(pool,`.slot-${slot}.control`);
- try{fs.mkdirSync(path,{mode:0o700});}catch(error:any){if(error?.code!=='EEXIST')throw new CsoError('PERSISTENCE_FAILED','Reproduction slot control directory could not be created');}
- const stat=privateDirectory(path,'Reproduction slot control directory');for(const name of fs.readdirSync(path))if(name!=='.recovery'&&!/^\.recovery\.tmp\.\d+\.[a-f0-9]{8}$/.test(name))throw new CsoError('UNSAFE_PATH','Reproduction slot control directory contains an unexpected object');return{path,stat};
+function slotControl(pool: string, slot: number): { path: string; stat: fs.Stats } {
+ const path = join(pool, `.slot-${slot}.control`);
+ try {
+ fs.mkdirSync(path, { mode: 0o700 });
+ } catch (error: any) {
+ if (error?.code !== 'EEXIST')
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction slot control directory could not be created');
+ }
+ const stat = privateDirectory(path, 'Reproduction slot control directory');
+ for (const name of fs.readdirSync(path))
+ if (name !== '.recovery' && !/^\.recovery\.tmp\.\d+\.[a-f0-9]{8}$/.test(name))
+ throw new CsoError('UNSAFE_PATH', 'Reproduction slot control directory contains an unexpected object');
+ return { path, stat };
}
-function writeLease(lease:Lease):void{
- const keys=Object.keys(lease).sort().join(','),expected='endpoint,expiresAt,ownerPid,path,runId,slot,supervised,token';
- if(keys!==expected||!/^unix:\/\/[/.A-Za-z0-9_-]+$/.test(lease.endpoint)||![0,1].includes(lease.slot)||
- !/^[A-Za-z0-9_.-]{1,100}$/.test(lease.runId)||lease.ownerPid!==process.pid||!Number.isSafeInteger(lease.expiresAt)||
- !/^[a-f0-9]{32}$/.test(lease.token)||typeof lease.supervised!=='boolean')
- throw new CsoError('PERSISTENCE_FAILED','Reproduction lease metadata is invalid');
- const expectedPath=join(machinePoolRoot(),sha256(lease.endpoint).slice(0,24),`slot-${lease.slot}`);
- if(lease.path!==expectedPath)throw new CsoError('PERSISTENCE_FAILED','Reproduction lease path is invalid');
+function writeLease(lease: Lease): void {
+ const keys = Object.keys(lease).sort().join(','),
+ expected = 'endpoint,expiresAt,ownerPid,path,runId,slot,supervised,token';
+ if (
+ keys !== expected ||
+ !/^unix:\/\/[/.A-Za-z0-9_-]+$/.test(lease.endpoint) ||
+ ![0, 1].includes(lease.slot) ||
+ !/^[A-Za-z0-9_.-]{1,100}$/.test(lease.runId) ||
+ lease.ownerPid !== process.pid ||
+ !Number.isSafeInteger(lease.expiresAt) ||
+ !/^[a-f0-9]{32}$/.test(lease.token) ||
+ typeof lease.supervised !== 'boolean'
+ )
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction lease metadata is invalid');
+ const expectedPath = join(machinePoolRoot(), sha256(lease.endpoint).slice(0, 24), `slot-${lease.slot}`);
+ if (lease.path !== expectedPath)
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction lease path is invalid');
// This exact helper-owned schema contains only control metadata. In
// particular, its random capability may resemble a wallet address and must
// remain byte-identical to lease.token; untrusted reports still use writeJson.
- atomicWriteSync(join(lease.path,'lease.json'),JSON.stringify(lease)+'\n',{mode:0o600});
+ atomicWriteSync(join(lease.path, 'lease.json'), JSON.stringify(lease) + '\n', { mode: 0o600 });
}
export function admit(endpoint: string, runId: string, deadline: number): Lease {
- if (!/^unix:\/\/[/.A-Za-z0-9_-]+$/.test(endpoint)) throw new CsoError('ISOLATION_FAILED','Only a pinned local Unix Docker endpoint is admitted on this host');
- const pool = secureDirectory(join(machinePoolRoot(),sha256(endpoint).slice(0,24)));
- for (let slot=0;slot<2;slot++) {
- const path=join(pool,`slot-${slot}`),control=slotControl(pool,slot);let mutation:Claim;
- try{mutation=acquireClaim(control.path,control.stat);}catch(error){if(error instanceof CsoError&&error.code==='INSUFFICIENT_CAPACITY')continue;throw error;}
- try{
+ if (!/^unix:\/\/[/.A-Za-z0-9_-]+$/.test(endpoint))
+ throw new CsoError(
+ 'ISOLATION_FAILED',
+ 'Only a pinned local Unix Docker endpoint is admitted on this host',
+ );
+ const pool = secureDirectory(join(machinePoolRoot(), sha256(endpoint).slice(0, 24)));
+ for (let slot = 0; slot < 2; slot++) {
+ const path = join(pool, `slot-${slot}`),
+ control = slotControl(pool, slot);
+ let mutation: Claim;
+ try {
+ mutation = acquireClaim(control.path, control.stat);
+ } catch (error) {
+ if (error instanceof CsoError && error.code === 'INSUFFICIENT_CAPACITY') continue;
+ throw error;
+ }
+ try {
try {
- fs.mkdirSync(path,{mode:0o700});
- } catch(error:any) {
- if(error?.code!=='EEXIST')throw new CsoError('PERSISTENCE_FAILED','Reproduction lease slot could not be created');
+ fs.mkdirSync(path, { mode: 0o700 });
+ } catch (error: any) {
+ if (error?.code !== 'EEXIST')
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction lease slot could not be created');
try {
- const observed=privateDirectory(path,'Reproduction lease slot');
- const old = JSON.parse(fs.readFileSync(join(path,'lease.json'),'utf8'));
+ const observed = privateDirectory(path, 'Reproduction lease slot');
+ const old = JSON.parse(fs.readFileSync(join(path, 'lease.json'), 'utf8'));
// A supervised lease is removed only after its watchdog or owner has
// confirmed exact-resource cleanup. This preserves the two-group cap
// through supervisor death and daemon outages.
if (old.supervised === true || (typeof old.ownerPid === 'number' && alive(old.ownerPid))) continue;
// Unsupervised stale slots cannot have created containers: supervision
// is acknowledged before the anchor create call.
- if(!reclaimSlot(path,pool,slot,observed,typeof old.token==='string'?old.token:undefined))continue;
- } catch(recoveryError) {
- if(recoveryError instanceof CsoError)throw recoveryError;
+ if (!reclaimSlot(path, pool, slot, observed, typeof old.token === 'string' ? old.token : undefined))
+ continue;
+ } catch (recoveryError) {
+ if (recoveryError instanceof CsoError) throw recoveryError;
// No live initializer can publish into this path while this stable
// slot-control claim is held. Recover a crashed partial publication
// only after the compatibility grace period.
- let stat:fs.Stats;try{stat=privateDirectory(path,'Reproduction lease slot');}catch(statError){if(statError instanceof CsoError)throw statError;continue;}
- if(Date.now()-stat.mtimeMs<=5000)continue;
- if(!reclaimSlot(path,pool,slot,stat))continue;
+ let stat: fs.Stats;
+ try {
+ stat = privateDirectory(path, 'Reproduction lease slot');
+ } catch (statError) {
+ if (statError instanceof CsoError) throw statError;
+ continue;
+ }
+ if (Date.now() - stat.mtimeMs <= 5000) continue;
+ if (!reclaimSlot(path, pool, slot, stat)) continue;
}
}
// Both authenticated records become visible as one logical publication
// when the stable slot-control claim is released.
- const lease:Lease={endpoint,slot,path,runId,ownerPid:process.pid,expiresAt:deadline,token:randomBytes(16).toString('hex'),supervised:false};writeLease(lease);fs.writeFileSync(join(path,'lease.token'),lease.token+'\n',{mode:0o600,flag:'wx'});return lease;
- }finally{releaseClaim(mutation);}
+ const lease: Lease = {
+ endpoint,
+ slot,
+ path,
+ runId,
+ ownerPid: process.pid,
+ expiresAt: deadline,
+ token: randomBytes(16).toString('hex'),
+ supervised: false,
+ };
+ writeLease(lease);
+ fs.writeFileSync(join(path, 'lease.token'), lease.token + '\n', { mode: 0o600, flag: 'wx' });
+ return lease;
+ } finally {
+ releaseClaim(mutation);
+ }
}
- throw new CsoError('INSUFFICIENT_CAPACITY','Two reproduction groups are already admitted for this Docker endpoint');
+ throw new CsoError(
+ 'INSUFFICIENT_CAPACITY',
+ 'Two reproduction groups are already admitted for this Docker endpoint',
+ );
}
-export function markSupervised(lease:Lease):void{
- const current=JSON.parse(fs.readFileSync(join(lease.path,'lease.json'),'utf8'));
- if(current.token!==lease.token||current.ownerPid!==lease.ownerPid)throw new CsoError('INSUFFICIENT_CAPACITY','Reproduction lease changed before watchdog supervision');
- lease.supervised=true;writeLease(lease);
+export function markSupervised(lease: Lease): void {
+ const current = JSON.parse(fs.readFileSync(join(lease.path, 'lease.json'), 'utf8'));
+ if (current.token !== lease.token || current.ownerPid !== lease.ownerPid)
+ throw new CsoError('INSUFFICIENT_CAPACITY', 'Reproduction lease changed before watchdog supervision');
+ lease.supervised = true;
+ writeLease(lease);
}
export function release(lease: Lease): void {
- let observed:fs.Stats;try{observed=privateDirectory(lease.path,'Reproduction lease slot');}catch(error:any){if(error?.code==='ENOENT')throw new CsoError('PERSISTENCE_FAILED','Exact reproduction lease was already missing');throw error;}
- const claim=acquireClaim(lease.path,observed);
- try{
- const currentStat=privateDirectory(lease.path,'Reproduction lease slot');if(!sameDirectory(observed,currentStat))throw new CsoError('PERSISTENCE_FAILED','Reproduction lease changed before exact release');
- const names=fs.readdirSync(lease.path).filter(name=>name!=='.recovery'&&!/^\.recovery\.tmp\.\d+\.[a-f0-9]{8}$/.test(name)).sort();if(names.join('\0')!=='lease.json\0lease.token')throw new CsoError('PERSISTENCE_FAILED','Reproduction lease contents changed before exact release');
- privateFile(join(lease.path,'lease.json'),'Reproduction lease');privateFile(join(lease.path,'lease.token'),'Reproduction lease token');
- const current=JSON.parse(fs.readFileSync(join(lease.path,'lease.json'),'utf8')),token=fs.readFileSync(join(lease.path,'lease.token'),'utf8').trim();
- if(current.runId!==lease.runId||current.ownerPid!==lease.ownerPid||current.token!==lease.token||token!==lease.token)throw new CsoError('PERSISTENCE_FAILED','Reproduction lease ownership changed before exact release');
- fs.unlinkSync(join(lease.path,'lease.token'));fs.unlinkSync(join(lease.path,'lease.json'));releaseClaim(claim);fs.rmdirSync(lease.path);
- if(fs.existsSync(lease.path))throw new CsoError('PERSISTENCE_FAILED','Exact reproduction lease removal could not be proven');
- }catch(error){try{if(fs.existsSync(claim.path))releaseClaim(claim);}catch{}if(error instanceof CsoError)throw error;throw new CsoError('PERSISTENCE_FAILED','Exact reproduction lease removal failed');}
+ let observed: fs.Stats;
+ try {
+ observed = privateDirectory(lease.path, 'Reproduction lease slot');
+ } catch (error: any) {
+ if (error?.code === 'ENOENT')
+ throw new CsoError('PERSISTENCE_FAILED', 'Exact reproduction lease was already missing');
+ throw error;
+ }
+ const claim = acquireClaim(lease.path, observed);
+ try {
+ const currentStat = privateDirectory(lease.path, 'Reproduction lease slot');
+ if (!sameDirectory(observed, currentStat))
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction lease changed before exact release');
+ const names = fs
+ .readdirSync(lease.path)
+ .filter((name) => name !== '.recovery' && !/^\.recovery\.tmp\.\d+\.[a-f0-9]{8}$/.test(name))
+ .sort();
+ if (names.join('\0') !== 'lease.json\0lease.token')
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction lease contents changed before exact release');
+ privateFile(join(lease.path, 'lease.json'), 'Reproduction lease');
+ privateFile(join(lease.path, 'lease.token'), 'Reproduction lease token');
+ const current = JSON.parse(fs.readFileSync(join(lease.path, 'lease.json'), 'utf8')),
+ token = fs.readFileSync(join(lease.path, 'lease.token'), 'utf8').trim();
+ if (
+ current.runId !== lease.runId ||
+ current.ownerPid !== lease.ownerPid ||
+ current.token !== lease.token ||
+ token !== lease.token
+ )
+ throw new CsoError('PERSISTENCE_FAILED', 'Reproduction lease ownership changed before exact release');
+ fs.unlinkSync(join(lease.path, 'lease.token'));
+ fs.unlinkSync(join(lease.path, 'lease.json'));
+ releaseClaim(claim);
+ fs.rmdirSync(lease.path);
+ if (fs.existsSync(lease.path))
+ throw new CsoError('PERSISTENCE_FAILED', 'Exact reproduction lease removal could not be proven');
+ } catch (error) {
+ try {
+ if (fs.existsSync(claim.path)) releaseClaim(claim);
+ } catch {}
+ if (error instanceof CsoError) throw error;
+ throw new CsoError('PERSISTENCE_FAILED', 'Exact reproduction lease removal failed');
+ }
}
export function total(roles: Role[]) {
- const value = roles.reduce((a,r) => ({cpu:a.cpu+ROLE_LIMITS[r].cpu,memoryMiB:a.memoryMiB+ROLE_LIMITS[r].memoryMiB,pids:a.pids+ROLE_LIMITS[r].pids,writableMiB:a.writableMiB+ROLE_LIMITS[r].writableMiB}), {cpu:0,memoryMiB:0,pids:0,writableMiB:0});
- if (value.cpu > GROUP_LIMITS.cpu || value.memoryMiB > GROUP_LIMITS.memoryMiB || value.pids > GROUP_LIMITS.pids || value.writableMiB > GROUP_LIMITS.writableMiB)
- throw new CsoError('INSUFFICIENT_CAPACITY','Requested sidecars exceed the aggregate reproduction-group limit');
+ const value = roles.reduce(
+ (a, r) => ({
+ cpu: a.cpu + ROLE_LIMITS[r].cpu,
+ memoryMiB: a.memoryMiB + ROLE_LIMITS[r].memoryMiB,
+ pids: a.pids + ROLE_LIMITS[r].pids,
+ writableMiB: a.writableMiB + ROLE_LIMITS[r].writableMiB,
+ }),
+ { cpu: 0, memoryMiB: 0, pids: 0, writableMiB: 0 },
+ );
+ if (
+ value.cpu > GROUP_LIMITS.cpu ||
+ value.memoryMiB > GROUP_LIMITS.memoryMiB ||
+ value.pids > GROUP_LIMITS.pids ||
+ value.writableMiB > GROUP_LIMITS.writableMiB
+ )
+ throw new CsoError(
+ 'INSUFFICIENT_CAPACITY',
+ 'Requested sidecars exceed the aggregate reproduction-group limit',
+ );
return value;
}
diff --git a/lib/cso/bounded-file.ts b/lib/cso/bounded-file.ts
index 243e0a3e9..8ea3cc3d0 100644
--- a/lib/cso/bounded-file.ts
+++ b/lib/cso/bounded-file.ts
@@ -2,15 +2,58 @@ import * as fs from 'node:fs';
import { CsoError } from './contracts';
/** Read one caller-supplied control file without following or blocking on a raced special file. */
-export function readBoundedStable(path:string,max:number,label:string):Buffer{
- let named:fs.Stats,fd:number|undefined;try{named=fs.lstatSync(path);}catch{throw new CsoError('MISSING_INPUT',`${label} does not exist`);}
- if(named.isSymbolicLink()||!named.isFile()||named.nlink!==1||named.size>max)throw new CsoError('MISSING_INPUT',`${label} must be one bounded regular file`);
- try{
- fd=fs.openSync(path,fs.constants.O_RDONLY|(fs.constants.O_NOFOLLOW??0)|(fs.constants.O_NONBLOCK??0));const opened=fs.fstatSync(fd);
- if(!opened.isFile()||opened.nlink!==1||opened.dev!==named.dev||opened.ino!==named.ino||opened.mode!==named.mode||opened.size!==named.size)throw new CsoError('SNAPSHOT_RACE',`${label} changed before it could be read`);
- const data=Buffer.alloc(max+1);let bytes=0,count=0;while(bytes0)bytes+=count;
- const after=fs.fstatSync(fd),current=fs.lstatSync(path);if(bytes>max)throw new CsoError('MISSING_INPUT',`${label} exceeds the ${max}-byte limit`);
- if(!current.isFile()||current.isSymbolicLink()||current.nlink!==1||current.dev!==opened.dev||current.ino!==opened.ino||current.mode!==opened.mode||after.size!==opened.size||after.mtimeMs!==opened.mtimeMs||after.ctimeMs!==opened.ctimeMs)throw new CsoError('SNAPSHOT_RACE',`${label} changed while it was read`);
- return data.subarray(0,bytes);
- }catch(error){if(error instanceof CsoError)throw error;const code=(error as NodeJS.ErrnoException).code;if(['ELOOP','ENOENT','ENOTDIR','ENXIO'].includes(code??''))throw new CsoError('SNAPSHOT_RACE',`${label} changed before it could be opened`);throw new CsoError('MISSING_INPUT',`${label} is missing or unreadable`);}finally{if(fd!==undefined)fs.closeSync(fd);}
+export function readBoundedStable(path: string, max: number, label: string): Buffer {
+ let named: fs.Stats, fd: number | undefined;
+ try {
+ named = fs.lstatSync(path);
+ } catch {
+ throw new CsoError('MISSING_INPUT', `${label} does not exist`);
+ }
+ if (named.isSymbolicLink() || !named.isFile() || named.nlink !== 1 || named.size > max)
+ throw new CsoError('MISSING_INPUT', `${label} must be one bounded regular file`);
+ try {
+ fd = fs.openSync(
+ path,
+ fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0) | (fs.constants.O_NONBLOCK ?? 0),
+ );
+ const opened = fs.fstatSync(fd);
+ if (
+ !opened.isFile() ||
+ opened.nlink !== 1 ||
+ opened.dev !== named.dev ||
+ opened.ino !== named.ino ||
+ opened.mode !== named.mode ||
+ opened.size !== named.size
+ )
+ throw new CsoError('SNAPSHOT_RACE', `${label} changed before it could be read`);
+ const data = Buffer.alloc(max + 1);
+ let bytes = 0,
+ count = 0;
+ while (bytes < data.length && (count = fs.readSync(fd, data, bytes, data.length - bytes, null)) > 0)
+ bytes += count;
+ const after = fs.fstatSync(fd),
+ current = fs.lstatSync(path);
+ if (bytes > max) throw new CsoError('MISSING_INPUT', `${label} exceeds the ${max}-byte limit`);
+ if (
+ !current.isFile() ||
+ current.isSymbolicLink() ||
+ current.nlink !== 1 ||
+ current.dev !== opened.dev ||
+ current.ino !== opened.ino ||
+ current.mode !== opened.mode ||
+ after.size !== opened.size ||
+ after.mtimeMs !== opened.mtimeMs ||
+ after.ctimeMs !== opened.ctimeMs
+ )
+ throw new CsoError('SNAPSHOT_RACE', `${label} changed while it was read`);
+ return data.subarray(0, bytes);
+ } catch (error) {
+ if (error instanceof CsoError) throw error;
+ const code = (error as NodeJS.ErrnoException).code;
+ if (['ELOOP', 'ENOENT', 'ENOTDIR', 'ENXIO'].includes(code ?? ''))
+ throw new CsoError('SNAPSHOT_RACE', `${label} changed before it could be opened`);
+ throw new CsoError('MISSING_INPUT', `${label} is missing or unreadable`);
+ } finally {
+ if (fd !== undefined) fs.closeSync(fd);
+ }
}
diff --git a/lib/cso/cache.ts b/lib/cso/cache.ts
index e9a3f6ecb..d3695d59c 100644
--- a/lib/cso/cache.ts
+++ b/lib/cso/cache.ts
@@ -3,7 +3,13 @@ import { createHash, randomBytes } from 'node:crypto';
import { join, resolve, sep } from 'node:path';
import { atomicWriteSync } from '../fs-atomic';
import { CsoError } from './contracts';
-import { discardAtomicNoReplaceTemp, privateRoot, recoverAtomicNoReplaceJson, secureDirectory, withLock as withStateLock } from './state';
+import {
+ discardAtomicNoReplaceTemp,
+ privateRoot,
+ recoverAtomicNoReplaceJson,
+ secureDirectory,
+ withLock as withStateLock,
+} from './state';
export const DEFAULT_PUBLIC_ARCHIVE_CACHE_BYTES = 10 * 1024 * 1024 * 1024;
const METADATA_VERSION = 1;
@@ -82,14 +88,17 @@ function fail(code: ConstructorParameters[0], message: string):
}
function operationControl(input?: CacheOperationInput): NormalizedCacheOperationControl {
- const value = typeof input === 'number' ? { deadline: input } : input ?? {};
- if (!value || typeof value !== 'object' || Array.isArray(value)) fail('INVALID_ARGUMENT', 'Cache operation control must be an object or absolute deadline');
+ const value = typeof input === 'number' ? { deadline: input } : (input ?? {});
+ if (!value || typeof value !== 'object' || Array.isArray(value))
+ fail('INVALID_ARGUMENT', 'Cache operation control must be an object or absolute deadline');
if (value.deadline !== undefined && (!Number.isSafeInteger(value.deadline) || value.deadline <= 0))
fail('INVALID_ARGUMENT', 'Cache deadline must be an absolute millisecond timestamp');
if (value.signal !== undefined && typeof value.signal.aborted !== 'boolean')
fail('INVALID_ARGUMENT', 'Cache cancellation signal is invalid');
- const control = Object.freeze({ ...(value.deadline === undefined ? {} : { deadline: value.deadline }),
- ...(value.signal === undefined ? {} : { signal: value.signal }) });
+ const control = Object.freeze({
+ ...(value.deadline === undefined ? {} : { deadline: value.deadline }),
+ ...(value.signal === undefined ? {} : { signal: value.signal }),
+ });
checkOperation(control);
return control;
}
@@ -100,18 +109,26 @@ function checkOperation(control: NormalizedCacheOperationControl): void {
fail('DEADLINE', 'Archive-cache operation reached its deadline');
}
-function boundedDirectoryNames(path: string, label: string, control: NormalizedCacheOperationControl): string[] {
+function boundedDirectoryNames(
+ path: string,
+ label: string,
+ control: NormalizedCacheOperationControl,
+): string[] {
checkOperation(control);
- const directory = fs.opendirSync(path), names: string[] = [];
+ const directory = fs.opendirSync(path),
+ names: string[] = [];
try {
for (;;) {
checkOperation(control);
const entry = directory.readSync();
if (!entry) break;
- if (names.length >= MAX_CACHE_DIRECTORY_ENTRIES) fail('INSUFFICIENT_CAPACITY', `${label} exceeds the cache entry limit`);
+ if (names.length >= MAX_CACHE_DIRECTORY_ENTRIES)
+ fail('INSUFFICIENT_CAPACITY', `${label} exceeds the cache entry limit`);
names.push(entry.name);
}
- } finally { directory.closeSync(); }
+ } finally {
+ directory.closeSync();
+ }
checkOperation(control);
return names;
}
@@ -119,18 +136,23 @@ function boundedDirectoryNames(path: string, label: string, control: NormalizedC
function assertEmptyDirectory(path: string, control: NormalizedCacheOperationControl): void {
checkOperation(control);
const directory = fs.opendirSync(path);
- try { if (directory.readSync()) fail('UNSAFE_PATH', 'Archive materialization directory must be empty'); }
- finally { directory.closeSync(); }
+ try {
+ if (directory.readSync()) fail('UNSAFE_PATH', 'Archive materialization directory must be empty');
+ } finally {
+ directory.closeSync();
+ }
checkOperation(control);
}
function boundedPositiveInteger(value: number, name: string): number {
- if (!Number.isSafeInteger(value) || value <= 0) fail('INVALID_ARGUMENT', `${name} must be a positive safe integer`);
+ if (!Number.isSafeInteger(value) || value <= 0)
+ fail('INVALID_ARGUMENT', `${name} must be a positive safe integer`);
return value;
}
function expectedDigest(value: string): string {
- if (!SHA256.test(value)) fail('INVALID_ARGUMENT', 'Archive SHA-256 must be 64 lowercase hexadecimal characters');
+ if (!SHA256.test(value))
+ fail('INVALID_ARGUMENT', 'Archive SHA-256 must be 64 lowercase hexadecimal characters');
return value;
}
@@ -138,33 +160,54 @@ function stagedRelativePath(value: string): string {
if (typeof value !== 'string' || value.length > 4096 || !RELATIVE_STAGE_PATH.test(value))
fail('UNSAFE_PATH', 'Staged archive path must be a contained relative path');
const parts = value.split('/');
- if (parts.some(part => !part || part === '.' || part === '..')) fail('UNSAFE_PATH', 'Staged archive path must be a contained relative path');
+ if (parts.some((part) => !part || part === '.' || part === '..'))
+ fail('UNSAFE_PATH', 'Staged archive path must be a contained relative path');
return value;
}
function stableStat(stat: fs.Stats): StableStat {
return {
- dev: stat.dev, ino: stat.ino, size: stat.size, mode: stat.mode, nlink: stat.nlink,
- mtimeMs: stat.mtimeMs, ctimeMs: stat.ctimeMs, uid: stat.uid,
+ dev: stat.dev,
+ ino: stat.ino,
+ size: stat.size,
+ mode: stat.mode,
+ nlink: stat.nlink,
+ mtimeMs: stat.mtimeMs,
+ ctimeMs: stat.ctimeMs,
+ uid: stat.uid,
};
}
function sameStat(left: StableStat, right: StableStat): boolean {
- return left.dev === right.dev && left.ino === right.ino && left.size === right.size &&
- left.mode === right.mode && left.nlink === right.nlink && left.mtimeMs === right.mtimeMs &&
- left.ctimeMs === right.ctimeMs && left.uid === right.uid;
+ return (
+ left.dev === right.dev &&
+ left.ino === right.ino &&
+ left.size === right.size &&
+ left.mode === right.mode &&
+ left.nlink === right.nlink &&
+ left.mtimeMs === right.mtimeMs &&
+ left.ctimeMs === right.ctimeMs &&
+ left.uid === right.uid
+ );
}
function sameRenamedInode(left: StableStat, right: StableStat): boolean {
- return left.dev === right.dev && left.ino === right.ino && left.size === right.size &&
- left.mode === right.mode && left.nlink === right.nlink && left.mtimeMs === right.mtimeMs && left.uid === right.uid;
+ return (
+ left.dev === right.dev &&
+ left.ino === right.ino &&
+ left.size === right.size &&
+ left.mode === right.mode &&
+ left.nlink === right.nlink &&
+ left.mtimeMs === right.mtimeMs &&
+ left.uid === right.uid
+ );
}
-
function assertOwnedRegular(stat: fs.Stats, label: string, maxBytes: number, immutable = false): void {
if (!stat.isFile() || stat.isSymbolicLink()) fail('UNSAFE_PATH', `${label} must be a regular file`);
if (stat.nlink !== 1) fail('UNSAFE_PATH', `${label} must not be hard-linked`);
- if (process.getuid && stat.uid !== process.getuid()) fail('UNSAFE_PATH', `${label} must be owned by the current user`);
+ if (process.getuid && stat.uid !== process.getuid())
+ fail('UNSAFE_PATH', `${label} must be owned by the current user`);
if (stat.size > maxBytes) fail('INSUFFICIENT_CAPACITY', `${label} exceeds its byte limit`);
if (immutable && (stat.mode & 0o222) !== 0) fail('INCOMPATIBLE_INPUT', `${label} is unexpectedly writable`);
}
@@ -172,80 +215,125 @@ function assertOwnedRegular(stat: fs.Stats, label: string, maxBytes: number, imm
function assertExistingDirectory(path: string, label: string): string {
const requested = resolve(path);
let requestedStat: fs.Stats;
- try { requestedStat = fs.lstatSync(requested); }
- catch { fail('MISSING_INPUT', `${label} does not exist`); }
+ try {
+ requestedStat = fs.lstatSync(requested);
+ } catch {
+ fail('MISSING_INPUT', `${label} does not exist`);
+ }
if (requestedStat!.isSymbolicLink()) fail('UNSAFE_PATH', `${label} must not be a symlink`);
let canonical: string;
- try { canonical = fs.realpathSync(requested); }
- catch { fail('MISSING_INPUT', `${label} does not exist`); }
+ try {
+ canonical = fs.realpathSync(requested);
+ } catch {
+ fail('MISSING_INPUT', `${label} does not exist`);
+ }
const stat = fs.lstatSync(canonical!);
if (!stat.isDirectory() || stat.isSymbolicLink()) fail('UNSAFE_PATH', `${label} must be a directory`);
- if (process.getuid && stat.uid !== process.getuid()) fail('UNSAFE_PATH', `${label} must be owned by the current user`);
+ if (process.getuid && stat.uid !== process.getuid())
+ fail('UNSAFE_PATH', `${label} must be owned by the current user`);
if ((stat.mode & 0o022) !== 0) fail('UNSAFE_PATH', `${label} must not be writable by another user`);
return canonical!;
}
-function assertContainedAncestors(root: string, relativePath: string, control: NormalizedCacheOperationControl): string {
+function assertContainedAncestors(
+ root: string,
+ relativePath: string,
+ control: NormalizedCacheOperationControl,
+): string {
const parts = relativePath.split('/');
let cursor = root;
for (const part of parts.slice(0, -1)) {
checkOperation(control);
cursor = join(cursor, part);
let stat: fs.Stats;
- try { stat = fs.lstatSync(cursor); }
- catch { fail('MISSING_INPUT', `Staged archive directory is missing: ${part}`); }
- if (!stat.isDirectory() || stat.isSymbolicLink()) fail('UNSAFE_PATH', 'Staged archive has a symlink or non-directory ancestor');
- if (process.getuid && stat.uid !== process.getuid()) fail('UNSAFE_PATH', 'Staged archive ancestor has an unexpected owner');
+ try {
+ stat = fs.lstatSync(cursor);
+ } catch {
+ fail('MISSING_INPUT', `Staged archive directory is missing: ${part}`);
+ }
+ if (!stat.isDirectory() || stat.isSymbolicLink())
+ fail('UNSAFE_PATH', 'Staged archive has a symlink or non-directory ancestor');
+ if (process.getuid && stat.uid !== process.getuid())
+ fail('UNSAFE_PATH', 'Staged archive ancestor has an unexpected owner');
if ((stat.mode & 0o022) !== 0) fail('UNSAFE_PATH', 'Staged archive ancestor is writable by another user');
}
const path = resolve(root, ...parts);
- if (path !== root && !path.startsWith(`${root}${sep}`)) fail('UNSAFE_PATH', 'Staged archive escaped its staging directory');
+ if (path !== root && !path.startsWith(`${root}${sep}`))
+ fail('UNSAFE_PATH', 'Staged archive escaped its staging directory');
return path;
}
function openNoFollow(path: string, flags: number, mode?: number): number {
const noFollow = (fs.constants as Record).O_NOFOLLOW ?? 0;
const closeOnExec = (fs.constants as Record).O_CLOEXEC ?? 0;
- try { return fs.openSync(path, flags | noFollow | closeOnExec, mode); }
- catch { fail('UNSAFE_PATH', 'Archive file could not be opened without following links'); }
+ try {
+ return fs.openSync(path, flags | noFollow | closeOnExec, mode);
+ } catch {
+ fail('UNSAFE_PATH', 'Archive file could not be opened without following links');
+ }
}
function readMetadata(path: string, digest: string, control: NormalizedCacheOperationControl): Metadata {
checkOperation(control);
let stat: fs.Stats;
- try { stat = fs.lstatSync(path); }
- catch { fail('INCOMPATIBLE_INPUT', `Cache metadata is missing for ${digest}`); }
+ try {
+ stat = fs.lstatSync(path);
+ } catch {
+ fail('INCOMPATIBLE_INPUT', `Cache metadata is missing for ${digest}`);
+ }
assertOwnedRegular(stat!, 'Cache metadata', METADATA_LIMIT);
if ((stat!.mode & 0o077) !== 0) fail('INCOMPATIBLE_INPUT', 'Cache metadata permissions are not private');
let value: unknown;
- try { checkOperation(control); value = JSON.parse(fs.readFileSync(path, 'utf8')); checkOperation(control); }
- catch (error) {
+ try {
+ checkOperation(control);
+ value = JSON.parse(fs.readFileSync(path, 'utf8'));
+ checkOperation(control);
+ } catch (error) {
if (error instanceof CsoError) throw error;
fail('INCOMPATIBLE_INPUT', `Cache metadata is invalid for ${digest}`);
}
const record = value as Partial;
- if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).sort().join(',') !== 'bytes,createdAt,lastAccessedAt,sha256,version' ||
- record.version !== METADATA_VERSION || record.sha256 !== digest || !Number.isSafeInteger(record.bytes) || Number(record.bytes) < 0 ||
- !Number.isSafeInteger(record.createdAt) || Number(record.createdAt) < 0 || !Number.isSafeInteger(record.lastAccessedAt) ||
- Number(record.lastAccessedAt) < Number(record.createdAt)) fail('INCOMPATIBLE_INPUT', `Cache metadata is invalid for ${digest}`);
+ if (
+ !value ||
+ typeof value !== 'object' ||
+ Array.isArray(value) ||
+ Object.keys(value).sort().join(',') !== 'bytes,createdAt,lastAccessedAt,sha256,version' ||
+ record.version !== METADATA_VERSION ||
+ record.sha256 !== digest ||
+ !Number.isSafeInteger(record.bytes) ||
+ Number(record.bytes) < 0 ||
+ !Number.isSafeInteger(record.createdAt) ||
+ Number(record.createdAt) < 0 ||
+ !Number.isSafeInteger(record.lastAccessedAt) ||
+ Number(record.lastAccessedAt) < Number(record.createdAt)
+ )
+ fail('INCOMPATIBLE_INPUT', `Cache metadata is invalid for ${digest}`);
return record as Metadata;
}
function writeMetadata(path: string, metadata: Metadata, noReplace = false): void {
- try { atomicWriteSync(path, `${JSON.stringify(metadata)}\n`, { mode: 0o600, noReplace }); }
- catch { fail('PERSISTENCE_FAILED', 'Cache metadata could not be written atomically'); }
+ try {
+ atomicWriteSync(path, `${JSON.stringify(metadata)}\n`, { mode: 0o600, noReplace });
+ } catch {
+ fail('PERSISTENCE_FAILED', 'Cache metadata could not be written atomically');
+ }
}
function removeRegular(path: string, label: string): void {
const stat = fs.lstatSync(path);
assertOwnedRegular(stat, label, Number.MAX_SAFE_INTEGER);
- try { fs.unlinkSync(path); }
- catch { fail('PERSISTENCE_FAILED', `${label} could not be removed`); }
+ try {
+ fs.unlinkSync(path);
+ } catch {
+ fail('PERSISTENCE_FAILED', `${label} could not be removed`);
+ }
}
function existsNoFollow(path: string): boolean {
- try { fs.lstatSync(path); return true; }
- catch (error: any) {
+ try {
+ fs.lstatSync(path);
+ return true;
+ } catch (error: any) {
if (error?.code === 'ENOENT') return false;
fail('INCOMPATIBLE_INPUT', 'Cache object could not be inspected safely');
}
@@ -269,7 +357,10 @@ export class PublicArchiveCache {
constructor(options: PublicArchiveCacheOptions) {
if (!options || typeof options !== 'object') fail('INVALID_ARGUMENT', 'Cache options are required');
- this.maxBytes = boundedPositiveInteger(options.maxBytes ?? DEFAULT_PUBLIC_ARCHIVE_CACHE_BYTES, 'maxBytes');
+ this.maxBytes = boundedPositiveInteger(
+ options.maxBytes ?? DEFAULT_PUBLIC_ARCHIVE_CACHE_BYTES,
+ 'maxBytes',
+ );
this.maxEntryBytes = boundedPositiveInteger(options.maxEntryBytes ?? this.maxBytes, 'maxEntryBytes');
if (this.maxEntryBytes > this.maxBytes) fail('INVALID_ARGUMENT', 'maxEntryBytes cannot exceed maxBytes');
this.clock = options.now ?? Date.now;
@@ -281,32 +372,45 @@ export class PublicArchiveCache {
this.recoveryDir = secureDirectory(join(this.root, 'recovery'));
this.lockDir = join(this.root, '.lock');
this.stagingRoot = assertExistingDirectory(options.stagingRoot, 'Archive staging directory');
- if (this.root === this.stagingRoot || this.root.startsWith(`${this.stagingRoot}${sep}`) || this.stagingRoot.startsWith(`${this.root}${sep}`))
+ if (
+ this.root === this.stagingRoot ||
+ this.root.startsWith(`${this.stagingRoot}${sep}`) ||
+ this.stagingRoot.startsWith(`${this.root}${sep}`)
+ )
fail('UNSAFE_PATH', 'Archive staging and cache directories must be separate');
}
/** Promote a verified staging file. The staging file is never deleted. */
promote(stagedPath: string, sha256: string, operation?: CacheOperationInput): PublicArchiveCacheEntry {
const control = operationControl(operation);
- const digest = expectedDigest(sha256), relativePath = stagedRelativePath(stagedPath);
+ const digest = expectedDigest(sha256),
+ relativePath = stagedRelativePath(stagedPath);
const source = assertContainedAncestors(this.stagingRoot, relativePath, control);
return this.withLock(() => {
checkOperation(control);
this.cleanIncoming(control);
this.recoverInterruptedOperations(control);
let initial: fs.Stats;
- try { initial = fs.lstatSync(source); }
- catch { fail('MISSING_INPUT', 'Staged archive is missing'); }
+ try {
+ initial = fs.lstatSync(source);
+ } catch {
+ fail('MISSING_INPUT', 'Staged archive is missing');
+ }
assertOwnedRegular(initial!, 'Staged archive', this.maxEntryBytes);
- if ((initial!.mode & 0o022) !== 0) fail('UNSAFE_PATH', 'Staged archive must not be writable by another user');
+ if ((initial!.mode & 0o022) !== 0)
+ fail('UNSAFE_PATH', 'Staged archive must not be writable by another user');
- const target = this.entryPath(digest), metadataPath = this.metadataPath(digest);
- const targetExists = existsNoFollow(target), metadataExists = existsNoFollow(metadataPath);
- if (targetExists !== metadataExists) fail('INCOMPATIBLE_INPUT', `Cache entry is incomplete for ${digest}`);
+ const target = this.entryPath(digest),
+ metadataPath = this.metadataPath(digest);
+ const targetExists = existsNoFollow(target),
+ metadataExists = existsNoFollow(metadataPath);
+ if (targetExists !== metadataExists)
+ fail('INCOMPATIBLE_INPUT', `Cache entry is incomplete for ${digest}`);
if (targetExists) {
const staged = this.hashFile(source, this.maxEntryBytes, false, stableStat(initial!), control);
- if (staged.digest !== digest) fail('INCOMPATIBLE_INPUT', 'Staged archive does not match its caller-provided SHA-256');
+ if (staged.digest !== digest)
+ fail('INCOMPATIBLE_INPUT', 'Staged archive does not match its caller-provided SHA-256');
const metadata = this.verifiedEntry(digest, control);
return this.touch(metadata, control);
}
@@ -315,26 +419,42 @@ export class PublicArchiveCache {
// displace any already-verified cache entry. Copying below hashes it a
// second time so a staging race still fails closed.
const authenticated = this.hashFile(source, this.maxEntryBytes, false, stableStat(initial!), control);
- if (authenticated.digest !== digest) fail('INCOMPATIBLE_INPUT', 'Staged archive does not match its caller-provided SHA-256');
+ if (authenticated.digest !== digest)
+ fail('INCOMPATIBLE_INPUT', 'Staged archive does not match its caller-provided SHA-256');
this.evictToFit(initial!.size, control);
const incoming = join(this.incomingDir, `.incoming-${process.pid}-${randomBytes(12).toString('hex')}`);
let promoted = false;
try {
const staged = this.copyAndHash(source, incoming, this.maxEntryBytes, stableStat(initial!), control);
- if (staged.digest !== digest) fail('INCOMPATIBLE_INPUT', 'Staged archive does not match its caller-provided SHA-256');
+ if (staged.digest !== digest)
+ fail('INCOMPATIBLE_INPUT', 'Staged archive does not match its caller-provided SHA-256');
if (staged.bytes !== initial!.size) fail('SNAPSHOT_RACE', 'Staged archive changed during promotion');
checkOperation(control);
fs.chmodSync(incoming, 0o400);
// A hard-link followed by unlink is an atomic no-replace publication on
// the cache filesystem. rename(2) would silently replace a raced target.
- try { fs.linkSync(incoming, target); fs.unlinkSync(incoming); }
- catch { fail('PERSISTENCE_FAILED', 'Verified archive could not be promoted atomically'); }
+ try {
+ fs.linkSync(incoming, target);
+ fs.unlinkSync(incoming);
+ } catch {
+ fail('PERSISTENCE_FAILED', 'Verified archive could not be promoted atomically');
+ }
promoted = true;
const now = this.timestamp();
- const metadata: Metadata = { version: 1, sha256: digest, bytes: staged.bytes, createdAt: now, lastAccessedAt: now };
- try { checkOperation(control); writeMetadata(metadataPath, metadata, true); }
- catch (error) {
- try { this.discardObject(target, 'entry', digest); } catch {}
+ const metadata: Metadata = {
+ version: 1,
+ sha256: digest,
+ bytes: staged.bytes,
+ createdAt: now,
+ lastAccessedAt: now,
+ };
+ try {
+ checkOperation(control);
+ writeMetadata(metadataPath, metadata, true);
+ } catch (error) {
+ try {
+ this.discardObject(target, 'entry', digest);
+ } catch {}
throw error;
}
return this.entry(metadata);
@@ -355,9 +475,11 @@ export class PublicArchiveCache {
checkOperation(control);
this.cleanIncoming(control);
this.recoverInterruptedOperations(control);
- const targetExists = existsNoFollow(this.entryPath(digest)), metadataExists = existsNoFollow(this.metadataPath(digest));
+ const targetExists = existsNoFollow(this.entryPath(digest)),
+ metadataExists = existsNoFollow(this.metadataPath(digest));
if (!targetExists && !metadataExists) return undefined;
- if (targetExists !== metadataExists) fail('INCOMPATIBLE_INPUT', `Cache entry is incomplete for ${digest}`);
+ if (targetExists !== metadataExists)
+ fail('INCOMPATIBLE_INPUT', `Cache entry is incomplete for ${digest}`);
return this.touch(this.verifiedEntry(digest, control), control);
}, control);
}
@@ -371,7 +493,10 @@ export class PublicArchiveCache {
this.recoverInterruptedOperations(control);
const entries = this.inventory(control);
let bytes = 0;
- for (const entry of entries) { checkOperation(control); bytes += entry.bytes; }
+ for (const entry of entries) {
+ checkOperation(control);
+ bytes += entry.bytes;
+ }
return { entries: entries.length, bytes, maxBytes: this.maxBytes };
}, control);
}
@@ -382,46 +507,77 @@ export class PublicArchiveCache {
* cache paths. Every source and every copy is fully hashed in the same
* critical section.
*/
- materialize(digests: string[], destinationRoot: string, operation?: CacheOperationInput): MaterializedArchive[] {
+ materialize(
+ digests: string[],
+ destinationRoot: string,
+ operation?: CacheOperationInput,
+ ): MaterializedArchive[] {
const control = operationControl(operation);
if (!Array.isArray(digests) || !digests.length)
fail('INVALID_ARGUMENT', 'Archive materialization requires at least one SHA-256 digest');
if (digests.length > MAX_CACHE_DIRECTORY_ENTRIES)
fail('INSUFFICIENT_CAPACITY', 'Archive materialization exceeds the cache entry limit');
- const selectedSet=new Set();for(const digest of digests){checkOperation(control);if(typeof digest!=='string')fail('INVALID_ARGUMENT', 'Archive materialization requires SHA-256 digest strings');selectedSet.add(expectedDigest(digest));}
- checkOperation(control);const selected=[...selectedSet].sort();checkOperation(control);
+ const selectedSet = new Set();
+ for (const digest of digests) {
+ checkOperation(control);
+ if (typeof digest !== 'string')
+ fail('INVALID_ARGUMENT', 'Archive materialization requires SHA-256 digest strings');
+ selectedSet.add(expectedDigest(digest));
+ }
+ checkOperation(control);
+ const selected = [...selectedSet].sort();
+ checkOperation(control);
const destination = assertExistingDirectory(destinationRoot, 'Archive materialization directory');
assertEmptyDirectory(destination, control);
- if (destination === this.root || destination.startsWith(`${this.root}${sep}`) || this.root.startsWith(`${destination}${sep}`) ||
- destination === this.stagingRoot || destination.startsWith(`${this.stagingRoot}${sep}`) || this.stagingRoot.startsWith(`${destination}${sep}`))
+ if (
+ destination === this.root ||
+ destination.startsWith(`${this.root}${sep}`) ||
+ this.root.startsWith(`${destination}${sep}`) ||
+ destination === this.stagingRoot ||
+ destination.startsWith(`${this.stagingRoot}${sep}`) ||
+ this.stagingRoot.startsWith(`${destination}${sep}`)
+ )
fail('UNSAFE_PATH', 'Archive materialization directory must be separate from cache and staging roots');
return this.withLock(() => {
checkOperation(control);
this.cleanIncoming(control);
this.recoverInterruptedOperations(control);
- const created: string[] = [], result: MaterializedArchive[] = [];
+ const created: string[] = [],
+ result: MaterializedArchive[] = [];
try {
for (const digest of selected) {
checkOperation(control);
- const metadata = this.verifiedEntry(digest, control), source = this.entryPath(digest), initial = fs.lstatSync(source);
+ const metadata = this.verifiedEntry(digest, control),
+ source = this.entryPath(digest),
+ initial = fs.lstatSync(source);
assertOwnedRegular(initial, 'Cached archive', this.maxEntryBytes, true);
const target = join(destination, digest);
let copied: { digest: string; bytes: number };
- try { copied = this.copyAndHash(source, target, this.maxEntryBytes, stableStat(initial), control); }
- catch (error) {
- if (existsNoFollow(target)) try { removeRegular(target, 'Incomplete run-owned archive copy'); } catch {}
+ try {
+ copied = this.copyAndHash(source, target, this.maxEntryBytes, stableStat(initial), control);
+ } catch (error) {
+ if (existsNoFollow(target))
+ try {
+ removeRegular(target, 'Incomplete run-owned archive copy');
+ } catch {}
throw error;
}
created.push(target);
- if (copied.digest !== digest || copied.bytes !== metadata.bytes) fail('SNAPSHOT_RACE', 'Cached archive changed while its run-owned copy was materialized');
+ if (copied.digest !== digest || copied.bytes !== metadata.bytes)
+ fail('SNAPSHOT_RACE', 'Cached archive changed while its run-owned copy was materialized');
fs.chmodSync(target, 0o400);
const verified = this.hashFile(target, this.maxEntryBytes, true, undefined, control);
- if (verified.digest !== digest || verified.bytes !== metadata.bytes) fail('SNAPSHOT_RACE', 'Run-owned archive copy failed verification');
+ if (verified.digest !== digest || verified.bytes !== metadata.bytes)
+ fail('SNAPSHOT_RACE', 'Run-owned archive copy failed verification');
result.push(Object.freeze({ sha256: digest, path: target, bytes: verified.bytes }));
}
return result;
} catch (error) {
- for (const path of created.reverse()) { try { removeRegular(path, 'Incomplete run-owned archive copy'); } catch {} }
+ for (const path of created.reverse()) {
+ try {
+ removeRegular(path, 'Incomplete run-owned archive copy');
+ } catch {}
+ }
throw error;
}
}, control);
@@ -429,21 +585,34 @@ export class PublicArchiveCache {
private timestamp(): number {
const value = this.clock();
- if (!Number.isSafeInteger(value) || value < 0) fail('PERSISTENCE_FAILED', 'Cache clock returned an invalid timestamp');
+ if (!Number.isSafeInteger(value) || value < 0)
+ fail('PERSISTENCE_FAILED', 'Cache clock returned an invalid timestamp');
return value;
}
- private entryPath(digest: string): string { return join(this.entriesDir, digest); }
- private metadataPath(digest: string): string { return join(this.metadataDir, `${digest}.json`); }
+ private entryPath(digest: string): string {
+ return join(this.entriesDir, digest);
+ }
+ private metadataPath(digest: string): string {
+ return join(this.metadataDir, `${digest}.json`);
+ }
private entry(metadata: Metadata): PublicArchiveCacheEntry {
- return Object.freeze({ sha256: metadata.sha256, path: this.entryPath(metadata.sha256), bytes: metadata.bytes,
- createdAt: metadata.createdAt, lastAccessedAt: metadata.lastAccessedAt });
+ return Object.freeze({
+ sha256: metadata.sha256,
+ path: this.entryPath(metadata.sha256),
+ bytes: metadata.bytes,
+ createdAt: metadata.createdAt,
+ lastAccessedAt: metadata.lastAccessedAt,
+ });
}
private touch(metadata: Metadata, control: NormalizedCacheOperationControl): PublicArchiveCacheEntry {
checkOperation(control);
- const updated: Metadata = { ...metadata, lastAccessedAt: Math.max(metadata.lastAccessedAt, this.timestamp()) };
+ const updated: Metadata = {
+ ...metadata,
+ lastAccessedAt: Math.max(metadata.lastAccessedAt, this.timestamp()),
+ };
checkOperation(control);
writeMetadata(this.metadataPath(metadata.sha256), updated);
return this.entry(updated);
@@ -453,47 +622,71 @@ export class PublicArchiveCache {
checkOperation(control);
const metadata = readMetadata(this.metadataPath(digest), digest, control);
const result = this.hashFile(this.entryPath(digest), this.maxEntryBytes, true, undefined, control);
- if (result.digest !== digest || result.bytes !== metadata.bytes) fail('INCOMPATIBLE_INPUT', `Cached archive failed SHA-256 verification: ${digest}`);
+ if (result.digest !== digest || result.bytes !== metadata.bytes)
+ fail('INCOMPATIBLE_INPUT', `Cached archive failed SHA-256 verification: ${digest}`);
return metadata;
}
- private hashFile(path: string, maxBytes: number, immutable: boolean, expected: StableStat | undefined,
- control: NormalizedCacheOperationControl): { digest: string; bytes: number } {
+ private hashFile(
+ path: string,
+ maxBytes: number,
+ immutable: boolean,
+ expected: StableStat | undefined,
+ control: NormalizedCacheOperationControl,
+ ): { digest: string; bytes: number } {
checkOperation(control);
const fd = openNoFollow(path, fs.constants.O_RDONLY);
try {
const beforeStat = fs.fstatSync(fd);
assertOwnedRegular(beforeStat, immutable ? 'Cached archive' : 'Staged archive', maxBytes, immutable);
- const before = stableStat(beforeStat), hash = createHash('sha256'), buffer = Buffer.allocUnsafe(COPY_BUFFER_BYTES);
- if (expected && !sameStat(expected, before)) fail('SNAPSHOT_RACE', 'Archive changed before it could be verified');
+ const before = stableStat(beforeStat),
+ hash = createHash('sha256'),
+ buffer = Buffer.allocUnsafe(COPY_BUFFER_BYTES);
+ if (expected && !sameStat(expected, before))
+ fail('SNAPSHOT_RACE', 'Archive changed before it could be verified');
let bytes = 0;
for (;;) {
checkOperation(control);
const read = fs.readSync(fd, buffer, 0, buffer.length, null);
if (!read) break;
bytes += read;
- if (bytes > maxBytes) fail('INSUFFICIENT_CAPACITY', 'Archive exceeded its byte limit while being read');
+ if (bytes > maxBytes)
+ fail('INSUFFICIENT_CAPACITY', 'Archive exceeded its byte limit while being read');
hash.update(buffer.subarray(0, read));
checkOperation(control);
}
checkOperation(control);
const after = stableStat(fs.fstatSync(fd));
- if (!sameStat(before, after) || bytes !== before.size) fail('SNAPSHOT_RACE', 'Archive changed while it was being verified');
+ if (!sameStat(before, after) || bytes !== before.size)
+ fail('SNAPSHOT_RACE', 'Archive changed while it was being verified');
return { digest: hash.digest('hex'), bytes };
- } finally { fs.closeSync(fd); }
+ } finally {
+ fs.closeSync(fd);
+ }
}
- private copyAndHash(source: string, destination: string, maxBytes: number, expected: StableStat,
- control: NormalizedCacheOperationControl): { digest: string; bytes: number } {
+ private copyAndHash(
+ source: string,
+ destination: string,
+ maxBytes: number,
+ expected: StableStat,
+ control: NormalizedCacheOperationControl,
+ ): { digest: string; bytes: number } {
checkOperation(control);
const sourceFd = openNoFollow(source, fs.constants.O_RDONLY);
let destinationFd: number | undefined;
try {
const beforeStat = fs.fstatSync(sourceFd);
assertOwnedRegular(beforeStat, 'Staged archive', maxBytes);
- const before = stableStat(beforeStat), hash = createHash('sha256'), buffer = Buffer.allocUnsafe(COPY_BUFFER_BYTES);
+ const before = stableStat(beforeStat),
+ hash = createHash('sha256'),
+ buffer = Buffer.allocUnsafe(COPY_BUFFER_BYTES);
if (!sameStat(expected, before)) fail('SNAPSHOT_RACE', 'Staged archive changed before promotion');
- destinationFd = openNoFollow(destination, fs.constants.O_WRONLY | fs.constants.O_CREAT | fs.constants.O_EXCL, 0o600);
+ destinationFd = openNoFollow(
+ destination,
+ fs.constants.O_WRONLY | fs.constants.O_CREAT | fs.constants.O_EXCL,
+ 0o600,
+ );
let bytes = 0;
for (;;) {
checkOperation(control);
@@ -501,13 +694,15 @@ export class PublicArchiveCache {
if (!read) break;
bytes += read;
if (bytes > expected.size) fail('SNAPSHOT_RACE', 'Staged archive grew during promotion');
- if (bytes > maxBytes) fail('INSUFFICIENT_CAPACITY', 'Archive exceeded its byte limit during promotion');
+ if (bytes > maxBytes)
+ fail('INSUFFICIENT_CAPACITY', 'Archive exceeded its byte limit during promotion');
hash.update(buffer.subarray(0, read));
let offset = 0;
while (offset < read) {
checkOperation(control);
const written = fs.writeSync(destinationFd, buffer, offset, read - offset);
- if (written <= 0) fail('PERSISTENCE_FAILED', 'Archive copy stopped before the current chunk was written');
+ if (written <= 0)
+ fail('PERSISTENCE_FAILED', 'Archive copy stopped before the current chunk was written');
offset += written;
checkOperation(control);
}
@@ -516,7 +711,8 @@ export class PublicArchiveCache {
fs.fsyncSync(destinationFd);
checkOperation(control);
const after = stableStat(fs.fstatSync(sourceFd));
- if (!sameStat(before, after) || bytes !== before.size) fail('SNAPSHOT_RACE', 'Staged archive changed during promotion');
+ if (!sameStat(before, after) || bytes !== before.size)
+ fail('SNAPSHOT_RACE', 'Staged archive changed during promotion');
return { digest: hash.digest('hex'), bytes };
} finally {
if (destinationFd !== undefined) fs.closeSync(destinationFd);
@@ -528,19 +724,33 @@ export class PublicArchiveCache {
checkOperation(control);
const entryNames = boundedDirectoryNames(this.entriesDir, 'Cache entries directory', control),
metadataNames = boundedDirectoryNames(this.metadataDir, 'Cache metadata directory', control);
- const entrySet=new Set(),metadataSet=new Set();
- for (const name of entryNames) {checkOperation(control);if (!SHA256.test(name)) fail('INCOMPATIBLE_INPUT', 'Cache entries directory contains an unexpected object');entrySet.add(name);}
- for (const name of metadataNames) {checkOperation(control);if (!/^[a-f0-9]{64}\.json$/.test(name)) fail('INCOMPATIBLE_INPUT', 'Cache metadata directory contains an unexpected object');metadataSet.add(name.slice(0,-5));}
- if (entrySet.size !== metadataSet.size) fail('INCOMPATIBLE_INPUT', 'Cache entries and metadata are inconsistent');
- for(const name of entrySet){checkOperation(control);if(!metadataSet.has(name))
- fail('INCOMPATIBLE_INPUT', 'Cache entries and metadata are inconsistent');
+ const entrySet = new Set(),
+ metadataSet = new Set();
+ for (const name of entryNames) {
+ checkOperation(control);
+ if (!SHA256.test(name))
+ fail('INCOMPATIBLE_INPUT', 'Cache entries directory contains an unexpected object');
+ entrySet.add(name);
}
- const inventory = entryNames.map(digest => {
+ for (const name of metadataNames) {
+ checkOperation(control);
+ if (!/^[a-f0-9]{64}\.json$/.test(name))
+ fail('INCOMPATIBLE_INPUT', 'Cache metadata directory contains an unexpected object');
+ metadataSet.add(name.slice(0, -5));
+ }
+ if (entrySet.size !== metadataSet.size)
+ fail('INCOMPATIBLE_INPUT', 'Cache entries and metadata are inconsistent');
+ for (const name of entrySet) {
+ checkOperation(control);
+ if (!metadataSet.has(name)) fail('INCOMPATIBLE_INPUT', 'Cache entries and metadata are inconsistent');
+ }
+ const inventory = entryNames.map((digest) => {
checkOperation(control);
const stat = fs.lstatSync(this.entryPath(digest));
assertOwnedRegular(stat, 'Cached archive', this.maxEntryBytes, true);
const metadata = readMetadata(this.metadataPath(digest), digest, control);
- if (metadata.bytes !== stat.size) fail('INCOMPATIBLE_INPUT', `Cache size metadata is inconsistent for ${digest}`);
+ if (metadata.bytes !== stat.size)
+ fail('INCOMPATIBLE_INPUT', `Cache size metadata is inconsistent for ${digest}`);
return metadata;
});
checkOperation(control);
@@ -551,10 +761,18 @@ export class PublicArchiveCache {
if (!Number.isSafeInteger(incomingBytes) || incomingBytes < 0 || incomingBytes > this.maxBytes)
fail('INSUFFICIENT_CAPACITY', 'Archive cannot fit within the public-cache limit');
checkOperation(control);
- const entries = this.inventory(control);checkOperation(control);entries.sort((a, b) => a.lastAccessedAt - b.lastAccessedAt || a.createdAt - b.createdAt || a.sha256.localeCompare(b.sha256));
+ const entries = this.inventory(control);
+ checkOperation(control);
+ entries.sort(
+ (a, b) =>
+ a.lastAccessedAt - b.lastAccessedAt || a.createdAt - b.createdAt || a.sha256.localeCompare(b.sha256),
+ );
checkOperation(control);
let total = 0;
- for (const item of entries) { checkOperation(control); total += item.bytes; }
+ for (const item of entries) {
+ checkOperation(control);
+ total += item.bytes;
+ }
for (const item of entries) {
checkOperation(control);
if (total + incomingBytes <= this.maxBytes) break;
@@ -564,36 +782,60 @@ export class PublicArchiveCache {
removeRegular(metadata, 'Evicted cache metadata');
total -= item.bytes;
}
- if (total + incomingBytes > this.maxBytes) fail('INSUFFICIENT_CAPACITY', 'Archive cache could not free enough verified capacity');
+ if (total + incomingBytes > this.maxBytes)
+ fail('INSUFFICIENT_CAPACITY', 'Archive cache could not free enough verified capacity');
}
private cleanIncoming(control: NormalizedCacheOperationControl): void {
const incomingNames = boundedDirectoryNames(this.incomingDir, 'Cache incoming directory', control);
- let publishedByInode:Map|undefined;
+ let publishedByInode: Map | undefined;
for (const name of incomingNames) {
checkOperation(control);
- if (!/^\.incoming-\d+-[a-f0-9]{24}$/.test(name)) fail('INCOMPATIBLE_INPUT', 'Cache incoming directory contains an unexpected object');
- const incoming = join(this.incomingDir, name), stat = fs.lstatSync(incoming);
- if (!stat.isFile() || stat.isSymbolicLink() || ![1, 2].includes(stat.nlink) || stat.size > this.maxEntryBytes ||
- (process.getuid && stat.uid !== process.getuid())) fail('UNSAFE_PATH', 'Incomplete cache archive is not a bounded regular file');
+ if (!/^\.incoming-\d+-[a-f0-9]{24}$/.test(name))
+ fail('INCOMPATIBLE_INPUT', 'Cache incoming directory contains an unexpected object');
+ const incoming = join(this.incomingDir, name),
+ stat = fs.lstatSync(incoming);
+ if (
+ !stat.isFile() ||
+ stat.isSymbolicLink() ||
+ ![1, 2].includes(stat.nlink) ||
+ stat.size > this.maxEntryBytes ||
+ (process.getuid && stat.uid !== process.getuid())
+ )
+ fail('UNSAFE_PATH', 'Incomplete cache archive is not a bounded regular file');
if (stat.nlink === 2) {
- if(!publishedByInode){
- publishedByInode=new Map();
- for(const entry of boundedDirectoryNames(this.entriesDir, 'Cache entries directory', control)){
+ if (!publishedByInode) {
+ publishedByInode = new Map();
+ for (const entry of boundedDirectoryNames(this.entriesDir, 'Cache entries directory', control)) {
checkOperation(control);
- if (!SHA256.test(entry)) fail('INCOMPATIBLE_INPUT', 'Cache entries directory contains an unexpected object');
- const candidate=fs.lstatSync(this.entryPath(entry)),key=`${candidate.dev}:${candidate.ino}`,matches=publishedByInode.get(key)??[];
- matches.push(entry);publishedByInode.set(key,matches);
+ if (!SHA256.test(entry))
+ fail('INCOMPATIBLE_INPUT', 'Cache entries directory contains an unexpected object');
+ const candidate = fs.lstatSync(this.entryPath(entry)),
+ key = `${candidate.dev}:${candidate.ino}`,
+ matches = publishedByInode.get(key) ?? [];
+ matches.push(entry);
+ publishedByInode.set(key, matches);
}
}
- const matches=publishedByInode.get(`${stat.dev}:${stat.ino}`)??[];
- if (matches.length !== 1) fail('UNSAFE_PATH', 'Incoming archive hard link does not match one published cache entry');
+ const matches = publishedByInode.get(`${stat.dev}:${stat.ino}`) ?? [];
+ if (matches.length !== 1)
+ fail('UNSAFE_PATH', 'Incoming archive hard link does not match one published cache entry');
const target = fs.lstatSync(this.entryPath(matches[0]));
- if (!target.isFile() || target.isSymbolicLink() || target.nlink !== 2 || target.size > this.maxEntryBytes ||
- (process.getuid && target.uid !== process.getuid()) || (target.mode & 0o222) !== 0)
+ if (
+ !target.isFile() ||
+ target.isSymbolicLink() ||
+ target.nlink !== 2 ||
+ target.size > this.maxEntryBytes ||
+ (process.getuid && target.uid !== process.getuid()) ||
+ (target.mode & 0o222) !== 0
+ )
fail('SNAPSHOT_RACE', 'Incoming archive link count or identity changed during recovery');
}
- try { fs.unlinkSync(incoming); } catch { fail('PERSISTENCE_FAILED', 'Incomplete cache archive could not be removed'); }
+ try {
+ fs.unlinkSync(incoming);
+ } catch {
+ fail('PERSISTENCE_FAILED', 'Incomplete cache archive could not be removed');
+ }
}
}
@@ -614,14 +856,30 @@ export class PublicArchiveCache {
for (const name of metadataObjects) {
checkOperation(control);
if (/^[a-f0-9]{64}\.json\.tmp\.\d+\.[a-f0-9]{8}$/.test(name)) this.recoverMetadataTemp(name, control);
- else if (!/^[a-f0-9]{64}\.json$/.test(name)) fail('INCOMPATIBLE_INPUT', 'Cache metadata directory contains an unexpected object');
+ else if (!/^[a-f0-9]{64}\.json$/.test(name))
+ fail('INCOMPATIBLE_INPUT', 'Cache metadata directory contains an unexpected object');
}
const metadata = boundedDirectoryNames(this.metadataDir, 'Cache metadata directory', control);
- for (const name of entries) if (!SHA256.test(name)) fail('INCOMPATIBLE_INPUT', 'Cache entries directory contains an unexpected object');
- for (const name of metadata) if (!/^[a-f0-9]{64}\.json$/.test(name)) fail('INCOMPATIBLE_INPUT', 'Cache metadata directory contains an unexpected object');
- const entrySet=new Set(),metadataSet=new Set(),digests=new Set();
- for(const name of entries){checkOperation(control);entrySet.add(name);digests.add(name);}
- for(const name of metadata){checkOperation(control);const digest=name.slice(0,-5);metadataSet.add(digest);digests.add(digest);}
+ for (const name of entries)
+ if (!SHA256.test(name))
+ fail('INCOMPATIBLE_INPUT', 'Cache entries directory contains an unexpected object');
+ for (const name of metadata)
+ if (!/^[a-f0-9]{64}\.json$/.test(name))
+ fail('INCOMPATIBLE_INPUT', 'Cache metadata directory contains an unexpected object');
+ const entrySet = new Set(),
+ metadataSet = new Set(),
+ digests = new Set();
+ for (const name of entries) {
+ checkOperation(control);
+ entrySet.add(name);
+ digests.add(name);
+ }
+ for (const name of metadata) {
+ checkOperation(control);
+ const digest = name.slice(0, -5);
+ metadataSet.add(digest);
+ digests.add(digest);
+ }
for (const digest of digests) {
checkOperation(control);
if (entrySet.has(digest) === metadataSet.has(digest)) continue;
@@ -631,20 +889,33 @@ export class PublicArchiveCache {
}
private recoveryPath(kind: 'entry' | 'metadata', digest: string): string {
- return join(this.recoveryDir, `.recovery-${kind}-${digest}-${process.pid}-${randomBytes(12).toString('hex')}`);
+ return join(
+ this.recoveryDir,
+ `.recovery-${kind}-${digest}-${process.pid}-${randomBytes(12).toString('hex')}`,
+ );
}
private moveToRecovery(path: string, kind: 'entry' | 'metadata', digest: string): string {
const before = fs.lstatSync(path);
- assertOwnedRegular(before, kind === 'entry' ? 'Cached archive' : 'Cache metadata', kind === 'entry' ? this.maxEntryBytes : METADATA_LIMIT, kind === 'entry');
- if (kind === 'metadata' && (before.mode & 0o077) !== 0) fail('UNSAFE_PATH', 'Cache metadata permissions are not private');
+ assertOwnedRegular(
+ before,
+ kind === 'entry' ? 'Cached archive' : 'Cache metadata',
+ kind === 'entry' ? this.maxEntryBytes : METADATA_LIMIT,
+ kind === 'entry',
+ );
+ if (kind === 'metadata' && (before.mode & 0o077) !== 0)
+ fail('UNSAFE_PATH', 'Cache metadata permissions are not private');
const destination = this.recoveryPath(kind, digest);
- try { fs.renameSync(path, destination); }
- catch { fail('PERSISTENCE_FAILED', 'Interrupted cache object could not be quarantined atomically'); }
+ try {
+ fs.renameSync(path, destination);
+ } catch {
+ fail('PERSISTENCE_FAILED', 'Interrupted cache object could not be quarantined atomically');
+ }
const after = fs.lstatSync(destination);
// rename(2) can update ctime; stable inode identity, content size, mode,
// link count, mtime, and ownership prove the moved object is the one read.
- if (!sameRenamedInode(stableStat(before), stableStat(after))) fail('SNAPSHOT_RACE', 'Cache object changed while it was quarantined');
+ if (!sameRenamedInode(stableStat(before), stableStat(after)))
+ fail('SNAPSHOT_RACE', 'Cache object changed while it was quarantined');
return destination;
}
@@ -655,49 +926,95 @@ export class PublicArchiveCache {
private recoverMetadataTemp(name: string, control: NormalizedCacheOperationControl): void {
checkOperation(control);
- const path = join(this.metadataDir, name), stat = fs.lstatSync(path);
- if (!stat.isFile() || stat.isSymbolicLink() || ![1, 2].includes(stat.nlink) || stat.size > METADATA_LIMIT ||
- (process.getuid && stat.uid !== process.getuid()) || (stat.mode & 0o077) !== 0)
+ const path = join(this.metadataDir, name),
+ stat = fs.lstatSync(path);
+ if (
+ !stat.isFile() ||
+ stat.isSymbolicLink() ||
+ ![1, 2].includes(stat.nlink) ||
+ stat.size > METADATA_LIMIT ||
+ (process.getuid && stat.uid !== process.getuid()) ||
+ (stat.mode & 0o077) !== 0
+ )
fail('UNSAFE_PATH', 'Interrupted cache metadata write is not a private regular file');
if (stat.nlink === 2) {
const target = this.metadataPath(name.slice(0, 64));
let targetStat: fs.Stats;
- try { targetStat = fs.lstatSync(target); } catch { fail('INCOMPATIBLE_INPUT', 'Hard-linked metadata temp has no published target'); }
- if (!targetStat!.isFile() || targetStat!.isSymbolicLink() || targetStat!.dev !== stat.dev || targetStat!.ino !== stat.ino || targetStat!.nlink !== 2)
+ try {
+ targetStat = fs.lstatSync(target);
+ } catch {
+ fail('INCOMPATIBLE_INPUT', 'Hard-linked metadata temp has no published target');
+ }
+ if (
+ !targetStat!.isFile() ||
+ targetStat!.isSymbolicLink() ||
+ targetStat!.dev !== stat.dev ||
+ targetStat!.ino !== stat.ino ||
+ targetStat!.nlink !== 2
+ )
fail('UNSAFE_PATH', 'Interrupted metadata hard link does not match its published target');
}
- try { fs.unlinkSync(path); } catch { fail('PERSISTENCE_FAILED', 'Interrupted cache metadata write could not be removed'); }
+ try {
+ fs.unlinkSync(path);
+ } catch {
+ fail('PERSISTENCE_FAILED', 'Interrupted cache metadata write could not be removed');
+ }
}
private withLock(callback: () => T, control: NormalizedCacheOperationControl): T {
checkOperation(control);
const marker = `${JSON.stringify({ protocol: CACHE_LOCK_PROTOCOL })}\n`;
- const options={label:'Cache lock protocol',maxBytes:METADATA_LIMIT,validate:(value:unknown)=>{
- if(!value||typeof value!=='object'||Array.isArray(value)||Object.keys(value).join(',')!=='protocol'||(value as any).protocol!==CACHE_LOCK_PROTOCOL)
- fail('INCOMPATIBLE_INPUT','Archive-cache lock protocol is invalid');
- }};
- const tempPattern=/^\.lock\.tmp\.(\d{1,10})\.[a-f0-9]{8}$/;
- for(const name of boundedDirectoryNames(this.root, 'Cache root directory', control)){
+ const options = {
+ label: 'Cache lock protocol',
+ maxBytes: METADATA_LIMIT,
+ validate: (value: unknown) => {
+ if (
+ !value ||
+ typeof value !== 'object' ||
+ Array.isArray(value) ||
+ Object.keys(value).join(',') !== 'protocol' ||
+ (value as any).protocol !== CACHE_LOCK_PROTOCOL
+ )
+ fail('INCOMPATIBLE_INPUT', 'Archive-cache lock protocol is invalid');
+ },
+ };
+ const tempPattern = /^\.lock\.tmp\.(\d{1,10})\.[a-f0-9]{8}$/;
+ for (const name of boundedDirectoryNames(this.root, 'Cache root directory', control)) {
checkOperation(control);
- const match=name.match(tempPattern);if(!match)continue;
- const temporary=join(this.root,name),publisherPid=Number(match[1]);
- if(existsNoFollow(this.lockDir))recoverAtomicNoReplaceJson(this.lockDir,options);
- if(existsNoFollow(temporary))discardAtomicNoReplaceTemp(temporary,publisherPid,options);
+ const match = name.match(tempPattern);
+ if (!match) continue;
+ const temporary = join(this.root, name),
+ publisherPid = Number(match[1]);
+ if (existsNoFollow(this.lockDir)) recoverAtomicNoReplaceJson(this.lockDir, options);
+ if (existsNoFollow(temporary)) discardAtomicNoReplaceTemp(temporary, publisherPid, options);
}
- try { atomicWriteSync(this.lockDir, marker, { mode: 0o600, noReplace: true }); }
- catch (error: any) {
- if (error?.code !== 'EEXIST') fail('PERSISTENCE_FAILED', 'Archive-cache lock protocol could not be initialized');
- recoverAtomicNoReplaceJson(this.lockDir,options);
+ try {
+ atomicWriteSync(this.lockDir, marker, { mode: 0o600, noReplace: true });
+ } catch (error: any) {
+ if (error?.code !== 'EEXIST')
+ fail('PERSISTENCE_FAILED', 'Archive-cache lock protocol could not be initialized');
+ recoverAtomicNoReplaceJson(this.lockDir, options);
const stat = fs.lstatSync(this.lockDir);
if (stat.isDirectory() && !stat.isSymbolicLink())
- fail('INSUFFICIENT_CAPACITY', 'A legacy archive-cache helper may still own or initialize this cache; its lock was left intact');
+ fail(
+ 'INSUFFICIENT_CAPACITY',
+ 'A legacy archive-cache helper may still own or initialize this cache; its lock was left intact',
+ );
assertOwnedRegular(stat, 'Cache lock protocol', METADATA_LIMIT);
if ((stat.mode & 0o077) !== 0) fail('UNSAFE_PATH', 'Cache lock protocol permissions are not private');
- let protocol:unknown;try{protocol=JSON.parse(fs.readFileSync(this.lockDir,'utf8')).protocol;}catch{}
- if(protocol!==CACHE_LOCK_PROTOCOL)fail('INCOMPATIBLE_INPUT','Archive-cache lock protocol is invalid');
+ let protocol: unknown;
+ try {
+ protocol = JSON.parse(fs.readFileSync(this.lockDir, 'utf8')).protocol;
+ } catch {}
+ if (protocol !== CACHE_LOCK_PROTOCOL)
+ fail('INCOMPATIBLE_INPUT', 'Archive-cache lock protocol is invalid');
}
- const result=withStateLock(this.root,()=>{checkOperation(control);return callback();});
- if(result&&typeof (result as any).then==='function')fail('PERSISTENCE_FAILED','Archive-cache operation unexpectedly became asynchronous');
+ const result = withStateLock(this.root, () => {
+ checkOperation(control);
+ return callback();
+ });
+ if (result && typeof (result as any).then === 'function')
+ fail('PERSISTENCE_FAILED', 'Archive-cache operation unexpectedly became asynchronous');
return result as T;
}
}
diff --git a/lib/cso/cli.ts b/lib/cso/cli.ts
index 9be4ef35c..06a8dedb5 100644
--- a/lib/cso/cli.ts
+++ b/lib/cso/cli.ts
@@ -4,30 +4,115 @@ import * as os from 'node:os';
import { randomBytes } from 'node:crypto';
import { basename, dirname, isAbsolute, join, resolve } from 'node:path';
import {
- ABI, ApplicationModel, CoverageRecord, CsoError, FindingV3, PreparationProof, RunPolicy, RunReportV3, SnapshotEntry, SnapshotManifest, SubmissionV3,
- canonical, completeness, importLegacy, object, relativePath, renderReport, rootCauseIdentity, sha256, snapshotPathHandle, snapshotPathHandleId, snapshotPathId, snapshotReference, string, strings, validateCoverage, validateFinding, validateVerificationRequest,
+ ABI,
+ ApplicationModel,
+ CoverageRecord,
+ CsoError,
+ FindingV3,
+ PreparationProof,
+ RunPolicy,
+ RunReportV3,
+ SnapshotEntry,
+ SnapshotManifest,
+ SubmissionV3,
+ canonical,
+ completeness,
+ importLegacy,
+ object,
+ relativePath,
+ renderReport,
+ rootCauseIdentity,
+ sha256,
+ snapshotPathHandle,
+ snapshotPathHandleId,
+ snapshotPathId,
+ snapshotReference,
+ string,
+ strings,
+ validateCoverage,
+ validateFinding,
+ validateVerificationRequest,
} from './contracts';
import { capture, containedFile, assertSnapshot } from './snapshot';
-import { assertStateOutside, event, finalizeReplayTemporary, loadReport, newRun, privateRoot, publicReport, PUBLIC_SOURCE_ROOT, readJson, repoId, requireTime, retention, runDirectory, saveReport, secureDirectory, withLock, writeHelperJson, writeJson, writeJsonExclusive } from './state';
+import {
+ assertStateOutside,
+ event,
+ finalizeReplayTemporary,
+ loadReport,
+ newRun,
+ privateRoot,
+ publicReport,
+ PUBLIC_SOURCE_ROOT,
+ readJson,
+ repoId,
+ requireTime,
+ retention,
+ runDirectory,
+ saveReport,
+ secureDirectory,
+ withLock,
+ writeHelperJson,
+ writeJson,
+ writeJsonExclusive,
+} from './state';
import { dockerEndpoint, dockerProbe, ISOLATION_POLICY_HASH } from './docker';
import { executable, git, redact, sanitizeForJson, sanitizeHelperForJson } from './process';
import { inspectPreparation } from './preparation';
-import { assertRuntimeCompatible, RUNTIME_CATALOG, selectRuntime, validateRuntimeCatalog, type RuntimeCatalog } from './runtime-catalog';
+import {
+ assertRuntimeCompatible,
+ RUNTIME_CATALOG,
+ selectRuntime,
+ validateRuntimeCatalog,
+ type RuntimeCatalog,
+} from './runtime-catalog';
import { PublicArchiveCache, publicArchiveCacheRoot } from './cache';
-import { admitPreparationRuntime, admitPreparationSidecar, PreparationExecutor, type DependencyClosure, type RailsDatabaseSelection } from './preparation-executor';
+import {
+ admitPreparationRuntime,
+ admitPreparationSidecar,
+ PreparationExecutor,
+ type DependencyClosure,
+ type RailsDatabaseSelection,
+} from './preparation-executor';
import { DockerPreparationSandboxRunner } from './preparation-docker';
import { importSarif, SCANNER_IDS, ScannerId } from './scanners';
-import { SCANNER_CATALOG, selectScanner, validateScannerCatalog, type ScannerCatalog } from './scanner-catalog';
+import {
+ SCANNER_CATALOG,
+ selectScanner,
+ validateScannerCatalog,
+ type ScannerCatalog,
+} from './scanner-catalog';
import { executeScanner, scannerCoverage, validateScannerRequest } from './scanner-executor';
-import { canonicalStartPlan, canonicalTestPlan, DockerVerificationExecutor, makeReviewArtifact, patchHash, validateRepairBundle, validateReviewArtifact, VerificationAttemptError, verificationHarnessHash, verifyRepair, type VerificationExecutor } from './verification';
+import {
+ canonicalStartPlan,
+ canonicalTestPlan,
+ DockerVerificationExecutor,
+ makeReviewArtifact,
+ patchHash,
+ validateRepairBundle,
+ validateReviewArtifact,
+ VerificationAttemptError,
+ verificationHarnessHash,
+ verifyRepair,
+ type VerificationExecutor,
+} from './verification';
import { readBoundedStable } from './bounded-file';
import { assertionWitnessReplayHash, runAssertionWitnessChild } from './witness';
import { historyForPath } from './history';
-import { catalogImageProvisioningPolicy, inspectCatalogImages, openLocalCatalogImageSession, provisionCatalogImages, qualifiedCatalogImages, type CatalogImageSessionFactory } from './image-provisioning';
+import {
+ catalogImageProvisioningPolicy,
+ inspectCatalogImages,
+ openLocalCatalogImageSession,
+ provisionCatalogImages,
+ qualifiedCatalogImages,
+ type CatalogImageSessionFactory,
+} from './image-provisioning';
const VERSION = '3.0.0';
-const RETENTION_MAINTENANCE_MS=1_000,RETENTION_MAX_ENTRIES=100_000,REPLAY_LOOKUP_MAX_ENTRIES=10_000;
-const productionCatalogImageSession:CatalogImageSessionFactory=deadline=>openLocalCatalogImageSession(process.env,deadline);
+const RETENTION_MAINTENANCE_MS = 1_000,
+ RETENTION_MAX_ENTRIES = 100_000,
+ REPLAY_LOOKUP_MAX_ENTRIES = 10_000;
+const productionCatalogImageSession: CatalogImageSessionFactory = (deadline) =>
+ openLocalCatalogImageSession(process.env, deadline);
/** Internal qualification seam. The public entrypoint below always supplies committed dependencies. */
export interface CsoCliDependencies {
readonly runtimeCatalog: RuntimeCatalog;
@@ -35,19 +120,44 @@ export interface CsoCliDependencies {
readonly catalogImageSession?: CatalogImageSessionFactory;
readonly watchdogPath: () => string;
}
-const GENERATION_LOCK_FD=(()=>{
+const GENERATION_LOCK_FD = (() => {
// Source-mode developer/test runs use Bun directly. For installed builds,
// this catches accidental direct use of the internal payload. It is not an
// authentication mechanism against the trusted same-user host: that user
// can reproduce an inherited descriptor or environment value. The public
// native launcher is the security boundary because it scrubs runtime and
// loader variables before Bun starts.
- if(/^bun(?:\.exe)?$/i.test(basename(process.execPath)))return undefined;
- if(process.platform==='win32'){
- if(process.env.GSTACK_CSO_GENERATION_GUARD!=='inherited-windows-generation-handle-v3')throw new CsoError('ISOLATION_FAILED','Direct use of the internal CSO payload is unsupported; invoke gstack-cso-launcher');
- delete process.env.GSTACK_CSO_GENERATION_GUARD;return undefined;
+ if (/^bun(?:\.exe)?$/i.test(basename(process.execPath))) return undefined;
+ if (process.platform === 'win32') {
+ if (process.env.GSTACK_CSO_GENERATION_GUARD !== 'inherited-windows-generation-handle-v3')
+ throw new CsoError(
+ 'ISOLATION_FAILED',
+ 'Direct use of the internal CSO payload is unsupported; invoke gstack-cso-launcher',
+ );
+ delete process.env.GSTACK_CSO_GENERATION_GUARD;
+ return undefined;
}
- const raw=process.env.GSTACK_CSO_GENERATION_LOCK_FD;if(raw===undefined||!/^(?:[3-9]|[1-9][0-9]+)$/.test(raw))throw new CsoError('ISOLATION_FAILED','Direct use of the internal CSO payload is unsupported; invoke gstack-cso-launcher');const fd=Number(raw);let stat:fs.Stats;try{stat=fs.fstatSync(fd);}catch{throw new CsoError('ISOLATION_FAILED','The launcher publication lease was not preserved by the runtime');}const install=fs.statSync(dirname(process.execPath));if(!stat.isDirectory()||stat.dev!==install.dev||stat.ino!==install.ino)throw new CsoError('ISOLATION_FAILED','The launcher publication lease does not name the helper installation directory');delete process.env.GSTACK_CSO_GENERATION_LOCK_FD;return fd;
+ const raw = process.env.GSTACK_CSO_GENERATION_LOCK_FD;
+ if (raw === undefined || !/^(?:[3-9]|[1-9][0-9]+)$/.test(raw))
+ throw new CsoError(
+ 'ISOLATION_FAILED',
+ 'Direct use of the internal CSO payload is unsupported; invoke gstack-cso-launcher',
+ );
+ const fd = Number(raw);
+ let stat: fs.Stats;
+ try {
+ stat = fs.fstatSync(fd);
+ } catch {
+ throw new CsoError('ISOLATION_FAILED', 'The launcher publication lease was not preserved by the runtime');
+ }
+ const install = fs.statSync(dirname(process.execPath));
+ if (!stat.isDirectory() || stat.dev !== install.dev || stat.ino !== install.ino)
+ throw new CsoError(
+ 'ISOLATION_FAILED',
+ 'The launcher publication lease does not name the helper installation directory',
+ );
+ delete process.env.GSTACK_CSO_GENERATION_LOCK_FD;
+ return fd;
})();
void GENERATION_LOCK_FD;
const HELP = `gstack-cso ${VERSION} (helper ABI ${ABI})
@@ -78,431 +188,2547 @@ Usage:
Static runs never execute application code. Target and scanner execution is Docker-only and requires a qualified immutable catalog.`;
const SCHEMA = {
- version:3,
- scanner:{profile:'optional exact qualified scanner profile ID',api:{runtimeProfile:'qualified app runtime ID; comprehensive mode only',port:'1024..65535',start:{executable:'absolute in-container path',args:['helper-derived literal argv with optional inspect handles']},control:{name:'legitimate control',path:'/path',method:'GET|POST|PUT|PATCH|DELETE',expected:{status:'100..599',includes:'optional',excludes:'optional'}},boundaryFiles:['untransformed snapshot-relative path or inspect handle'],schema:'reviewed OpenAPI 3.0/3.1 JSON; internal references; no server overrides, hooks, callbacks or external examples',operationIds:['1..20 unique declared path operation IDs'],seed:'optional 1..2147483647',maxExamples:'optional 1..100'}},
- submission:{application:{actors:['string'],assets:['string'],entrypoints:['string'],tenantBoundaries:['string'],sensitiveOperations:['string'],invariants:['string']},findings:[{title:'string',rootCause:'stable root cause',location:{path:'exact path or opaque handle returned by inspect',line:'positive integer',symbol:'string'},advisoryIds:['normalized advisory identity or empty'],severity:'critical|high|medium|low|informational',confidence:'high|medium|low',confidenceRationale:'why available evidence supports that confidence',evidence:'supported|hypothesis',attackerControl:'specific input/control',impact:'specific consequence',scenario:'concrete attacker scenario',trace:['entrypoint','caller','sink'],references:['supporting source/advisory reference'],recommendation:'concrete root-cause repair',challenge:{reviewer:'identity or exact sequential fallback label',independent:'boolean',mode:'independent_agent|sequential_fallback',callers:'checked callers',controls:'checked controls',counterevidence:'checked counterevidence',conclusion:'reasoned outcome'},dependency:{affectedVersion:'optional exact range/version',reachability:'reachable|unreachable|unknown',exposure:'production/build/development context',exploitation:'published exploitation evidence or unknown'}}],coverage:[{domain:'string',scope:'string',status:'assessed|partial|not_assessed|not_applicable',method:'string',gaps:['required for partial/not_assessed'],exclusions:['string'],evidence:['required for assessed/partial/not_applicable'],tool:{name:'optional',version:'exact',freshness:'timestamp/status',outcome:'string'}}],gaps:['string'],modelUsage:{source:'host-reported source',tokens:'nonnegative integer',cost:'optional finite nonnegative number'},recheck:{findingId:'original stable ID',outcome:'open|resolved|unknown',evidence:[{kind:'caller|security_boundary',path:'fresh snapshot path or inspect handle',line:'positive integer',observation:'fresh source observation'}],rootCause:'same root cause'}},
- verification:{findingId:'stable ID',runtimeProfile:'qualified runtime ID',port:'1024..65535',start:{executable:'absolute in-container path',args:['literal argv']},legitimate:[{name:'control name',path:'/numeric-loopback-relative path',method:'GET|POST|PUT|PATCH|DELETE',headers:{'optional-name':'value'},body:'optional body',expected:{status:'100..599',includes:'optional',excludes:'optional'}}],security:{name:'security assertion',path:'/path',method:'GET|POST|PUT|PATCH|DELETE',expected:{status:'fixed status',includes:'optional',excludes:'optional'},vulnerable:{status:'provably mutually-exclusive vulnerable status',includes:'optional',excludes:'optional'}},existingTests:[{executable:'must exactly match the helper-derived canonical stack suite',args:['helper-derived argv']}],testFiles:['exact immutable canonical path or inspect handle'],fixtures:{'relative/path':'content mounted read-only at /fixtures'},boundaryFiles:['snapshot path or inspect handle for the security boundary'],changes:[{path:'snapshot path or inspect handle',beforeSha256:'hash or null',after:'replacement or null',effect:'source|configuration|dependency'}],review:{artifactId:'helper-issued after record-review',reviewer:'self-attested identity distinct from producer',independent:'self-attested boolean',rootCauseRepaired:'self-attested boolean',featurePreserved:'self-attested boolean',boundaryMocks:'self-attested boolean',rationale:'specific review',reviewedPatchHash:'canonical patch hash'}},
- helperOwned:['run/report completeness','stable IDs','reproduction outcome','repair result and label','current-source closure','verification/bundle hashes'],
+ version: 3,
+ scanner: {
+ profile: 'optional exact qualified scanner profile ID',
+ api: {
+ runtimeProfile: 'qualified app runtime ID; comprehensive mode only',
+ port: '1024..65535',
+ start: {
+ executable: 'absolute in-container path',
+ args: ['helper-derived literal argv with optional inspect handles'],
+ },
+ control: {
+ name: 'legitimate control',
+ path: '/path',
+ method: 'GET|POST|PUT|PATCH|DELETE',
+ expected: { status: '100..599', includes: 'optional', excludes: 'optional' },
+ },
+ boundaryFiles: ['untransformed snapshot-relative path or inspect handle'],
+ schema:
+ 'reviewed OpenAPI 3.0/3.1 JSON; internal references; no server overrides, hooks, callbacks or external examples',
+ operationIds: ['1..20 unique declared path operation IDs'],
+ seed: 'optional 1..2147483647',
+ maxExamples: 'optional 1..100',
+ },
+ },
+ submission: {
+ application: {
+ actors: ['string'],
+ assets: ['string'],
+ entrypoints: ['string'],
+ tenantBoundaries: ['string'],
+ sensitiveOperations: ['string'],
+ invariants: ['string'],
+ },
+ findings: [
+ {
+ title: 'string',
+ rootCause: 'stable root cause',
+ location: {
+ path: 'exact path or opaque handle returned by inspect',
+ line: 'positive integer',
+ symbol: 'string',
+ },
+ advisoryIds: ['normalized advisory identity or empty'],
+ severity: 'critical|high|medium|low|informational',
+ confidence: 'high|medium|low',
+ confidenceRationale: 'why available evidence supports that confidence',
+ evidence: 'supported|hypothesis',
+ attackerControl: 'specific input/control',
+ impact: 'specific consequence',
+ scenario: 'concrete attacker scenario',
+ trace: ['entrypoint', 'caller', 'sink'],
+ references: ['supporting source/advisory reference'],
+ recommendation: 'concrete root-cause repair',
+ challenge: {
+ reviewer: 'identity or exact sequential fallback label',
+ independent: 'boolean',
+ mode: 'independent_agent|sequential_fallback',
+ callers: 'checked callers',
+ controls: 'checked controls',
+ counterevidence: 'checked counterevidence',
+ conclusion: 'reasoned outcome',
+ },
+ dependency: {
+ affectedVersion: 'optional exact range/version',
+ reachability: 'reachable|unreachable|unknown',
+ exposure: 'production/build/development context',
+ exploitation: 'published exploitation evidence or unknown',
+ },
+ },
+ ],
+ coverage: [
+ {
+ domain: 'string',
+ scope: 'string',
+ status: 'assessed|partial|not_assessed|not_applicable',
+ method: 'string',
+ gaps: ['required for partial/not_assessed'],
+ exclusions: ['string'],
+ evidence: ['required for assessed/partial/not_applicable'],
+ tool: { name: 'optional', version: 'exact', freshness: 'timestamp/status', outcome: 'string' },
+ },
+ ],
+ gaps: ['string'],
+ modelUsage: {
+ source: 'host-reported source',
+ tokens: 'nonnegative integer',
+ cost: 'optional finite nonnegative number',
+ },
+ recheck: {
+ findingId: 'original stable ID',
+ outcome: 'open|resolved|unknown',
+ evidence: [
+ {
+ kind: 'caller|security_boundary',
+ path: 'fresh snapshot path or inspect handle',
+ line: 'positive integer',
+ observation: 'fresh source observation',
+ },
+ ],
+ rootCause: 'same root cause',
+ },
+ },
+ verification: {
+ findingId: 'stable ID',
+ runtimeProfile: 'qualified runtime ID',
+ port: '1024..65535',
+ start: { executable: 'absolute in-container path', args: ['literal argv'] },
+ legitimate: [
+ {
+ name: 'control name',
+ path: '/numeric-loopback-relative path',
+ method: 'GET|POST|PUT|PATCH|DELETE',
+ headers: { 'optional-name': 'value' },
+ body: 'optional body',
+ expected: { status: '100..599', includes: 'optional', excludes: 'optional' },
+ },
+ ],
+ security: {
+ name: 'security assertion',
+ path: '/path',
+ method: 'GET|POST|PUT|PATCH|DELETE',
+ expected: { status: 'fixed status', includes: 'optional', excludes: 'optional' },
+ vulnerable: {
+ status: 'provably mutually-exclusive vulnerable status',
+ includes: 'optional',
+ excludes: 'optional',
+ },
+ },
+ existingTests: [
+ {
+ executable: 'must exactly match the helper-derived canonical stack suite',
+ args: ['helper-derived argv'],
+ },
+ ],
+ testFiles: ['exact immutable canonical path or inspect handle'],
+ fixtures: { 'relative/path': 'content mounted read-only at /fixtures' },
+ boundaryFiles: ['snapshot path or inspect handle for the security boundary'],
+ changes: [
+ {
+ path: 'snapshot path or inspect handle',
+ beforeSha256: 'hash or null',
+ after: 'replacement or null',
+ effect: 'source|configuration|dependency',
+ },
+ ],
+ review: {
+ artifactId: 'helper-issued after record-review',
+ reviewer: 'self-attested identity distinct from producer',
+ independent: 'self-attested boolean',
+ rootCauseRepaired: 'self-attested boolean',
+ featurePreserved: 'self-attested boolean',
+ boundaryMocks: 'self-attested boolean',
+ rationale: 'specific review',
+ reviewedPatchHash: 'canonical patch hash',
+ },
+ },
+ helperOwned: [
+ 'run/report completeness',
+ 'stable IDs',
+ 'reproduction outcome',
+ 'repair result and label',
+ 'current-source closure',
+ 'verification/bundle hashes',
+ ],
};
-function emit(value: unknown): void { process.stdout.write((typeof value === 'string' ? redact(value) : JSON.stringify(sanitizeHelperForJson(value),null,2)) + '\n'); }
-function persistableArtifact(value:T,label:string):T{
- const sanitized=sanitizeHelperForJson(value) as T;
- if(canonical(sanitized)!==canonical(value))throw new CsoError('REDACTION_FAILED',`${label} contains material that cannot be persisted without changing its authenticated identity`);
+function emit(value: unknown): void {
+ process.stdout.write(
+ (typeof value === 'string' ? redact(value) : JSON.stringify(sanitizeHelperForJson(value), null, 2)) +
+ '\n',
+ );
+}
+function persistableArtifact(value: T, label: string): T {
+ const sanitized = sanitizeHelperForJson(value) as T;
+ if (canonical(sanitized) !== canonical(value))
+ throw new CsoError(
+ 'REDACTION_FAILED',
+ `${label} contains material that cannot be persisted without changing its authenticated identity`,
+ );
return sanitized;
}
-function rejectUnexpected(value:Record,allowed:readonly string[],name:string):void{
- for(const key of Object.keys(value))if(!allowed.includes(key))throw new CsoError('INVALID_SCHEMA',`Unexpected ${name} field: ${key}`);
+function rejectUnexpected(value: Record, allowed: readonly string[], name: string): void {
+ for (const key of Object.keys(value))
+ if (!allowed.includes(key)) throw new CsoError('INVALID_SCHEMA', `Unexpected ${name} field: ${key}`);
}
-function need(args:string[],flag:string):string { const i=args.indexOf(flag); if(i<0||i===args.length-1||args[i+1].startsWith('--')) throw new CsoError('INVALID_ARGUMENT',`${flag} requires a value`); const v=args[i+1]; args.splice(i,2); return v; }
-function take(args:string[],flag:string):boolean { const i=args.indexOf(flag); if(i<0)return false;args.splice(i,1);return true; }
-function callerPath(value:string):string {
- if(isAbsolute(value))return resolve(value);
- const cwd=process.env.GSTACK_CSO_CALLER_CWD;
- if(!cwd||!isAbsolute(cwd))throw new CsoError('INVALID_ARGUMENT','Relative paths require the trusted gstack-cso launcher');
- return resolve(cwd,value);
+function need(args: string[], flag: string): string {
+ const i = args.indexOf(flag);
+ if (i < 0 || i === args.length - 1 || args[i + 1].startsWith('--'))
+ throw new CsoError('INVALID_ARGUMENT', `${flag} requires a value`);
+ const v = args[i + 1];
+ args.splice(i, 2);
+ return v;
}
-function readInput(path:string,max=1024*1024):unknown {
- let data:Buffer;try{data=readBoundedStable(callerPath(path),max,'Input file');}catch(error){if(error instanceof CsoError)throw error;throw new CsoError('MISSING_INPUT',`Input file does not exist: ${path}`);}
- try{return JSON.parse(data.toString('utf8'));}catch{throw new CsoError('INVALID_SCHEMA','Input is not valid JSON');}
+function take(args: string[], flag: string): boolean {
+ const i = args.indexOf(flag);
+ if (i < 0) return false;
+ args.splice(i, 1);
+ return true;
}
-function model(v:unknown):ApplicationModel {
- const x=object(v,'application model');rejectUnexpected(x,['actors','assets','entrypoints','tenantBoundaries','sensitiveOperations','invariants'],'application model'); const out:ApplicationModel={actors:strings(x.actors,'actors'),assets:strings(x.assets,'assets'),entrypoints:strings(x.entrypoints,'entrypoints'),tenantBoundaries:strings(x.tenantBoundaries,'tenantBoundaries'),sensitiveOperations:strings(x.sensitiveOperations,'sensitiveOperations'),invariants:strings(x.invariants,'invariants')};
- if(Object.values(out).some(a=>!a.length))throw new CsoError('INVALID_SCHEMA','Every application-model dimension needs at least one evidence-backed entry');return out;
+function callerPath(value: string): string {
+ if (isAbsolute(value)) return resolve(value);
+ const cwd = process.env.GSTACK_CSO_CALLER_CWD;
+ if (!cwd || !isAbsolute(cwd))
+ throw new CsoError('INVALID_ARGUMENT', 'Relative paths require the trusted gstack-cso launcher');
+ return resolve(cwd, value);
}
-function planned(scope:string):CoverageRecord[]{
- const domains = scope==='infra'?['secrets','dependencies','ci-cd','infrastructure','integrations']
- : scope==='code'?['llm-agentic-mcp','owasp-2025','stride','data-classification']
- : scope==='skills'?['skill-supply-chain'] : scope==='supply-chain'?['dependencies'] : scope==='owasp'?['owasp-2025']
- : scope.startsWith('domain:')?[scope.slice(7)]
- :['secrets','dependencies','ci-cd','infrastructure','integrations','llm-agentic-mcp','skill-supply-chain','owasp-2025','stride','data-classification'];
- return ['application-model','attack-surface',...domains].map(domain=>({domain,scope,status:'not_assessed',method:'pending investigation',gaps:['Assessment has not been submitted'],exclusions:[],evidence:[]}));
+function readInput(path: string, max = 1024 * 1024): unknown {
+ let data: Buffer;
+ try {
+ data = readBoundedStable(callerPath(path), max, 'Input file');
+ } catch (error) {
+ if (error instanceof CsoError) throw error;
+ throw new CsoError('MISSING_INPUT', `Input file does not exist: ${path}`);
+ }
+ try {
+ return JSON.parse(data.toString('utf8'));
+ } catch {
+ throw new CsoError('INVALID_SCHEMA', 'Input is not valid JSON');
+ }
}
-function snapshotCoverage(manifest:Awaited>,scope:string):CoverageRecord{
- const omitted=manifest.entries.filter(entry=>!entry.executionHash),excluded=omitted.filter(entry=>entry.transformation?.startsWith('excluded:'));
+function model(v: unknown): ApplicationModel {
+ const x = object(v, 'application model');
+ rejectUnexpected(
+ x,
+ ['actors', 'assets', 'entrypoints', 'tenantBoundaries', 'sensitiveOperations', 'invariants'],
+ 'application model',
+ );
+ const out: ApplicationModel = {
+ actors: strings(x.actors, 'actors'),
+ assets: strings(x.assets, 'assets'),
+ entrypoints: strings(x.entrypoints, 'entrypoints'),
+ tenantBoundaries: strings(x.tenantBoundaries, 'tenantBoundaries'),
+ sensitiveOperations: strings(x.sensitiveOperations, 'sensitiveOperations'),
+ invariants: strings(x.invariants, 'invariants'),
+ };
+ if (Object.values(out).some((a) => !a.length))
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Every application-model dimension needs at least one evidence-backed entry',
+ );
+ return out;
+}
+function planned(scope: string): CoverageRecord[] {
+ const domains =
+ scope === 'infra'
+ ? ['secrets', 'dependencies', 'ci-cd', 'infrastructure', 'integrations']
+ : scope === 'code'
+ ? ['llm-agentic-mcp', 'owasp-2025', 'stride', 'data-classification']
+ : scope === 'skills'
+ ? ['skill-supply-chain']
+ : scope === 'supply-chain'
+ ? ['dependencies']
+ : scope === 'owasp'
+ ? ['owasp-2025']
+ : scope.startsWith('domain:')
+ ? [scope.slice(7)]
+ : [
+ 'secrets',
+ 'dependencies',
+ 'ci-cd',
+ 'infrastructure',
+ 'integrations',
+ 'llm-agentic-mcp',
+ 'skill-supply-chain',
+ 'owasp-2025',
+ 'stride',
+ 'data-classification',
+ ];
+ return ['application-model', 'attack-surface', ...domains].map((domain) => ({
+ domain,
+ scope,
+ status: 'not_assessed',
+ method: 'pending investigation',
+ gaps: ['Assessment has not been submitted'],
+ exclusions: [],
+ evidence: [],
+ }));
+}
+function snapshotCoverage(manifest: Awaited>, scope: string): CoverageRecord {
+ const omitted = manifest.entries.filter((entry) => !entry.executionHash),
+ excluded = omitted.filter((entry) => entry.transformation?.startsWith('excluded:'));
// Classify every unexplained omission as a material coverage gap. Coverage
// must not depend on transformation prose retaining a particular prefix.
- const unread=omitted.filter(entry=>!excluded.includes(entry));
- const deleted=manifest.deletedPaths??[],captured=manifest.entries.filter(entry=>entry.executionHash).length,missing=unread.length+deleted.length;
- return {domain:'snapshot-inputs',scope,status:missing?(captured?'partial':'not_assessed'):'assessed',method:'fail-closed captured source inventory',
- gaps:[...unread.map(entry=>`${publicSnapshotPath(manifest,entry.path).path}: in-scope source payload was unread and withheld from static and runtime assessment`),...deleted.map(item=>`${publicSnapshotPath(manifest,item.path).path}: tracked source is deleted from the worktree; only retained history is available for assessment`)],
- exclusions:excluded.map(entry=>`${publicSnapshotPath(manifest,entry.path).path}: ${entry.transformation}`),
- evidence:[`${captured} sanitized execution input${captured===1?'':'s'} captured; ${excluded.length} explicit non-executable exclusion${excluded.length===1?'':'s'}; ${unread.length} unread in-scope input${unread.length===1?'':'s'}; ${deleted.length} tracked deletion${deleted.length===1?'':'s'}`]};
+ const unread = omitted.filter((entry) => !excluded.includes(entry));
+ const deleted = manifest.deletedPaths ?? [],
+ captured = manifest.entries.filter((entry) => entry.executionHash).length,
+ missing = unread.length + deleted.length;
+ return {
+ domain: 'snapshot-inputs',
+ scope,
+ status: missing ? (captured ? 'partial' : 'not_assessed') : 'assessed',
+ method: 'fail-closed captured source inventory',
+ gaps: [
+ ...unread.map(
+ (entry) =>
+ `${publicSnapshotPath(manifest, entry.path).path}: in-scope source payload was unread and withheld from static and runtime assessment`,
+ ),
+ ...deleted.map(
+ (item) =>
+ `${publicSnapshotPath(manifest, item.path).path}: tracked source is deleted from the worktree; only retained history is available for assessment`,
+ ),
+ ],
+ exclusions: excluded.map(
+ (entry) => `${publicSnapshotPath(manifest, entry.path).path}: ${entry.transformation}`,
+ ),
+ evidence: [
+ `${captured} sanitized execution input${captured === 1 ? '' : 's'} captured; ${excluded.length} explicit non-executable exclusion${excluded.length === 1 ? '' : 's'}; ${unread.length} unread in-scope input${unread.length === 1 ? '' : 's'}; ${deleted.length} tracked deletion${deleted.length === 1 ? '' : 's'}`,
+ ],
+ };
}
-function historyCoverage(status:any,scope:string):CoverageRecord{return status?.status==='captured'
- ?{domain:'history-inputs',scope,status:'assessed',method:'bounded helper-retained Git history',gaps:[],exclusions:[],evidence:[`Captured ${String(status.commits??'bounded commits')} from ${String(status.range??'the pinned source range')}`]}
- :{domain:'history-inputs',scope,status:'not_assessed',method:'bounded helper-retained Git history',gaps:[typeof status?.gap==='string'&&status.gap.trim()?status.gap:'Historical evidence was not safely retained'],exclusions:[],evidence:[]};}
-function helperOwnedCoverage(domain:string):boolean{return domain==='snapshot-inputs'||domain==='history-inputs'||domain==='runtime-readiness'||domain.startsWith('scanner:')||domain.startsWith('preparation:')||domain.startsWith('execution:');}
-function parseStart(args:string[]):{repo:string;policy:RunPolicy;comparisonBase?:string}{
- const repo=callerPath(need(args,'--repo')),comprehensive=take(args,'--comprehensive'),diff=take(args,'--diff'),offline=take(args,'--offline');
- const scopeFlags=['--infra','--code','--skills','--supply-chain','--owasp'].filter(f=>args.includes(f));
- const named=args.includes('--scope')?need(args,'--scope'):undefined;
- if(scopeFlags.length+(named?1:0)>1)throw new CsoError('INVALID_ARGUMENT','Select only one scope');
- for(const f of scopeFlags)take(args,f);
- const explicitBase=args.includes('--base'),base=explicitBase?need(args,'--base'):'origin/main';
- const rawBudget=args.includes('--budget')?need(args,'--budget'):String(comprehensive?1800:600),budgetSeconds=Number(rawBudget);
- if(!Number.isInteger(budgetSeconds)||budgetSeconds<120||budgetSeconds>(comprehensive?1800:600))throw new CsoError('INVALID_ARGUMENT',`Budget must be an integer from 120 to ${comprehensive?1800:600} seconds`);
- if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown argument: ${args[0]}`);
- if(!fs.existsSync(repo)||!fs.statSync(repo).isDirectory())throw new CsoError('MISSING_INPUT','Repository directory does not exist');
- const scope=named?`domain:${string(named,'scope',100)}`:scopeFlags[0]?.slice(2)??'default';
- return {repo,policy:{mode:comprehensive?'comprehensive':'daily',scope,diff,base,offline,budgetSeconds,maxWorkers:3,maxRepairs:3},...(diff||explicitBase?{comparisonBase:base}:{})};
+function historyCoverage(status: any, scope: string): CoverageRecord {
+ return status?.status === 'captured'
+ ? {
+ domain: 'history-inputs',
+ scope,
+ status: 'assessed',
+ method: 'bounded helper-retained Git history',
+ gaps: [],
+ exclusions: [],
+ evidence: [
+ `Captured ${String(status.commits ?? 'bounded commits')} from ${String(status.range ?? 'the pinned source range')}`,
+ ],
+ }
+ : {
+ domain: 'history-inputs',
+ scope,
+ status: 'not_assessed',
+ method: 'bounded helper-retained Git history',
+ gaps: [
+ typeof status?.gap === 'string' && status.gap.trim()
+ ? status.gap
+ : 'Historical evidence was not safely retained',
+ ],
+ exclusions: [],
+ evidence: [],
+ };
}
-async function start(args:string[],dependencies:CsoCliDependencies,parent?:RunReportV3['parent'],requiredAncestor?:string,startedAt=new Date()):Promise{
- const {repo,policy,comparisonBase}=parseStart(args),createdAt=startedAt;assertStateOutside(repo);if(!parent)retention(createdAt.getTime(),{deadlineMs:Math.min(createdAt.getTime()+RETENTION_MAINTENANCE_MS,createdAt.getTime()+policy.budgetSeconds*1000-60_000),maxEntries:RETENTION_MAX_ENTRIES});const run=newRun(repo);
- let manifest:Awaited>;try{manifest=await capture(repo,run.dir,comparisonBase,requiredAncestor,{deadlineMs:createdAt.getTime()+policy.budgetSeconds*1000-60_000});}catch(error){fs.rmSync(run.dir,{recursive:true,force:true});throw error;}
- const report:RunReportV3={schemaVersion:3,runId:run.runId,repoId:run.repoId,createdAt:createdAt.toISOString(),deadline:new Date(createdAt.getTime()+policy.budgetSeconds*1000).toISOString(),status:'running',completeness:'not assessed',policy,
- source:{root:repo,snapshotHash:manifest.executionHash,originalHash:manifest.originalHash,baseCommit:manifest.baseCommit,
- transformations:[...manifest.entries.filter(entry=>entry.transformation).map(entry=>({path:publicSnapshotPath(manifest,entry.path).path,handling:entry.transformation!})),...(manifest.deletedPaths??[]).map(item=>({path:publicSnapshotPath(manifest,item.path).path,handling:'tracked source deleted; retained history only'}))]},
- application:{actors:[],assets:[],entrypoints:[],tenantBoundaries:[],sensitiveOperations:[],invariants:[]},coverage:[snapshotCoverage(manifest,policy.scope),historyCoverage(readJson(join(run.dir,'history-status.json')),policy.scope),...planned(policy.scope)],findings:[],gaps:[],events:[],...(parent?{parent}:{})};
- event(report,'snapshot',`Captured ${manifest.entries.length} source entries; ${manifest.entries.filter(e=>e.transformation).length+(manifest.deletedPaths?.length??0)} transformations disclosed`);
- if(policy.mode==='comprehensive'){
- const plan=inspectPreparation(join(run.dir,'snapshot'));writeJson(join(run.dir,'preparation.json'),plan);
- const c:CoverageRecord={domain:'runtime-readiness',scope:plan.stack,status:plan.status==='ready'?'partial':'not_assessed',method:'inert lockfile and runtime-catalog inspection',gaps:plan.prerequisites.map(p=>p.message),exclusions:[],evidence:[`Preparation metadata: ${plan.status}`]};
- try{validateRuntimeCatalog(dependencies.runtimeCatalog);selectRuntime(plan.runtimeProfile,platform(),dependencies.runtimeCatalog);c.evidence.push(`Qualified runtime catalog: ${dependencies.runtimeCatalog.revision}`);}catch(error:any){c.gaps.push(error?.message?.startsWith('MISSING_QUALIFIED_RUNTIME')?error.message:'Runtime catalog validation failed');}
- try{const runtimeHome=secureDirectory(join(run.dir,'home')),endpoint=await dockerEndpoint(runtimeHome);const probe=await dockerProbe(endpoint,runtimeHome);c.evidence.push(`Local Docker ${probe.version} admitted at ${endpoint.uri}`);}catch(error:any){c.gaps.push(error instanceof CsoError?error.message:'Local Docker isolation admission failed');}
- if(plan.status==='ready'&&!c.gaps.length)c.status='assessed';else if(c.evidence.length>1)c.status='partial';
+function helperOwnedCoverage(domain: string): boolean {
+ return (
+ domain === 'snapshot-inputs' ||
+ domain === 'history-inputs' ||
+ domain === 'runtime-readiness' ||
+ domain.startsWith('scanner:') ||
+ domain.startsWith('preparation:') ||
+ domain.startsWith('execution:')
+ );
+}
+function parseStart(args: string[]): { repo: string; policy: RunPolicy; comparisonBase?: string } {
+ const repo = callerPath(need(args, '--repo')),
+ comprehensive = take(args, '--comprehensive'),
+ diff = take(args, '--diff'),
+ offline = take(args, '--offline');
+ const scopeFlags = ['--infra', '--code', '--skills', '--supply-chain', '--owasp'].filter((f) =>
+ args.includes(f),
+ );
+ const named = args.includes('--scope') ? need(args, '--scope') : undefined;
+ if (scopeFlags.length + (named ? 1 : 0) > 1)
+ throw new CsoError('INVALID_ARGUMENT', 'Select only one scope');
+ for (const f of scopeFlags) take(args, f);
+ const explicitBase = args.includes('--base'),
+ base = explicitBase ? need(args, '--base') : 'origin/main';
+ const rawBudget = args.includes('--budget') ? need(args, '--budget') : String(comprehensive ? 1800 : 600),
+ budgetSeconds = Number(rawBudget);
+ if (!Number.isInteger(budgetSeconds) || budgetSeconds < 120 || budgetSeconds > (comprehensive ? 1800 : 600))
+ throw new CsoError(
+ 'INVALID_ARGUMENT',
+ `Budget must be an integer from 120 to ${comprehensive ? 1800 : 600} seconds`,
+ );
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown argument: ${args[0]}`);
+ if (!fs.existsSync(repo) || !fs.statSync(repo).isDirectory())
+ throw new CsoError('MISSING_INPUT', 'Repository directory does not exist');
+ const scope = named ? `domain:${string(named, 'scope', 100)}` : (scopeFlags[0]?.slice(2) ?? 'default');
+ return {
+ repo,
+ policy: {
+ mode: comprehensive ? 'comprehensive' : 'daily',
+ scope,
+ diff,
+ base,
+ offline,
+ budgetSeconds,
+ maxWorkers: 3,
+ maxRepairs: 3,
+ },
+ ...(diff || explicitBase ? { comparisonBase: base } : {}),
+ };
+}
+async function start(
+ args: string[],
+ dependencies: CsoCliDependencies,
+ parent?: RunReportV3['parent'],
+ requiredAncestor?: string,
+ startedAt = new Date(),
+): Promise {
+ const { repo, policy, comparisonBase } = parseStart(args),
+ createdAt = startedAt;
+ assertStateOutside(repo);
+ if (!parent)
+ retention(createdAt.getTime(), {
+ deadlineMs: Math.min(
+ createdAt.getTime() + RETENTION_MAINTENANCE_MS,
+ createdAt.getTime() + policy.budgetSeconds * 1000 - 60_000,
+ ),
+ maxEntries: RETENTION_MAX_ENTRIES,
+ });
+ const run = newRun(repo);
+ let manifest: Awaited>;
+ try {
+ manifest = await capture(repo, run.dir, comparisonBase, requiredAncestor, {
+ deadlineMs: createdAt.getTime() + policy.budgetSeconds * 1000 - 60_000,
+ });
+ } catch (error) {
+ fs.rmSync(run.dir, { recursive: true, force: true });
+ throw error;
+ }
+ const report: RunReportV3 = {
+ schemaVersion: 3,
+ runId: run.runId,
+ repoId: run.repoId,
+ createdAt: createdAt.toISOString(),
+ deadline: new Date(createdAt.getTime() + policy.budgetSeconds * 1000).toISOString(),
+ status: 'running',
+ completeness: 'not assessed',
+ policy,
+ source: {
+ root: repo,
+ snapshotHash: manifest.executionHash,
+ originalHash: manifest.originalHash,
+ baseCommit: manifest.baseCommit,
+ transformations: [
+ ...manifest.entries
+ .filter((entry) => entry.transformation)
+ .map((entry) => ({
+ path: publicSnapshotPath(manifest, entry.path).path,
+ handling: entry.transformation!,
+ })),
+ ...(manifest.deletedPaths ?? []).map((item) => ({
+ path: publicSnapshotPath(manifest, item.path).path,
+ handling: 'tracked source deleted; retained history only',
+ })),
+ ],
+ },
+ application: {
+ actors: [],
+ assets: [],
+ entrypoints: [],
+ tenantBoundaries: [],
+ sensitiveOperations: [],
+ invariants: [],
+ },
+ coverage: [
+ snapshotCoverage(manifest, policy.scope),
+ historyCoverage(readJson(join(run.dir, 'history-status.json')), policy.scope),
+ ...planned(policy.scope),
+ ],
+ findings: [],
+ gaps: [],
+ events: [],
+ ...(parent ? { parent } : {}),
+ };
+ event(
+ report,
+ 'snapshot',
+ `Captured ${manifest.entries.length} source entries; ${manifest.entries.filter((e) => e.transformation).length + (manifest.deletedPaths?.length ?? 0)} transformations disclosed`,
+ );
+ if (policy.mode === 'comprehensive') {
+ const plan = inspectPreparation(join(run.dir, 'snapshot'));
+ writeJson(join(run.dir, 'preparation.json'), plan);
+ const c: CoverageRecord = {
+ domain: 'runtime-readiness',
+ scope: plan.stack,
+ status: plan.status === 'ready' ? 'partial' : 'not_assessed',
+ method: 'inert lockfile and runtime-catalog inspection',
+ gaps: plan.prerequisites.map((p) => p.message),
+ exclusions: [],
+ evidence: [`Preparation metadata: ${plan.status}`],
+ };
+ try {
+ validateRuntimeCatalog(dependencies.runtimeCatalog);
+ selectRuntime(plan.runtimeProfile, platform(), dependencies.runtimeCatalog);
+ c.evidence.push(`Qualified runtime catalog: ${dependencies.runtimeCatalog.revision}`);
+ } catch (error: any) {
+ c.gaps.push(
+ error?.message?.startsWith('MISSING_QUALIFIED_RUNTIME')
+ ? error.message
+ : 'Runtime catalog validation failed',
+ );
+ }
+ try {
+ const runtimeHome = secureDirectory(join(run.dir, 'home')),
+ endpoint = await dockerEndpoint(runtimeHome);
+ const probe = await dockerProbe(endpoint, runtimeHome);
+ c.evidence.push(`Local Docker ${probe.version} admitted at ${endpoint.uri}`);
+ } catch (error: any) {
+ c.gaps.push(error instanceof CsoError ? error.message : 'Local Docker isolation admission failed');
+ }
+ if (plan.status === 'ready' && !c.gaps.length) c.status = 'assessed';
+ else if (c.evidence.length > 1) c.status = 'partial';
report.coverage.push(c);
}
- saveReport(run.dir,report);return publicReport(report);
+ saveReport(run.dir, report);
+ return publicReport(report);
}
-async function doctor(args:string[],dependencies:CsoCliDependencies){
- const started=Date.now(),repo=callerPath(need(args,'--repo'));if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown argument: ${args[0]}`);
- const staticCheck=(async()=>{let home='';try{const stat=fs.statSync(repo);if(!stat.isDirectory())throw new CsoError('MISSING_INPUT','Repository path is not a directory');const g=executable('git');home=secureDirectory(fs.mkdtempSync(join(fs.realpathSync(os.tmpdir()),'gstack-cso-doctor-git-')));if((await git(repo,['rev-parse','--is-inside-work-tree'],home)).trim()!=='true')throw new CsoError('MISSING_INPUT','Repository path is not a Git working tree');return{capability:'static-snapshot',status:'ready',detail:g};}catch(e:any){return{capability:'static-snapshot',status:'missing',detail:e instanceof CsoError?e.message:'Repository path is missing or unreadable'};}finally{if(home)fs.rmSync(home,{recursive:true,force:true});}})();
- let preparation:ReturnType|undefined;
- try{const stat=fs.statSync(repo);if(!stat.isDirectory())throw new CsoError('MISSING_INPUT','Repository path is not a directory');preparation=inspectPreparation(repo);}catch{}
- const scannerCatalog=dependencies.scannerCatalog??SCANNER_CATALOG,imageSession=dependencies.catalogImageSession??productionCatalogImageSession,deadline=started+30_000;
- let targetPlatform:'linux/amd64'|'linux/arm64'|undefined,entries:ReturnType=[];
- try{targetPlatform=platform();entries=qualifiedCatalogImages(dependencies.runtimeCatalog,scannerCatalog,targetPlatform);}catch{}
- const inspectedPromise=inspectCatalogImages(entries,imageSession,deadline);
- const checks:any[]=[await staticCheck];
- checks.push(preparation?{capability:'application-preparation',status:preparation.status==='ready'?'ready':'missing',detail:{stack:preparation.stack,prerequisites:preparation.prerequisites}}:{capability:'application-preparation',status:'missing',detail:'Repository source is unavailable for inert preparation inspection'});
- const inspected=await inspectedPromise;checks.push({capability:'local-docker-isolation',status:inspected.docker.status,detail:inspected.docker.detail});
- try{
- if(!preparation||preparation.status!=='ready')throw new CsoError('PREREQUISITE','Resolve the application-preparation prerequisites first');
- if(!targetPlatform)throw new CsoError('PREREQUISITE','Qualified runtimes require an amd64/arm64 Linux Docker platform');
- validateRuntimeCatalog(dependencies.runtimeCatalog);const runtime=selectRuntime(preparation.runtimeProfile,targetPlatform,dependencies.runtimeCatalog),availability=inspected.images.find(item=>item.kind==='runtime'&&item.id===runtime.id),requiredSidecars:Array>=[],prerequisites:string[]=[];
- if(!availability||availability.status!=='available')prerequisites.push(availability?.reason??'Exact qualified runtime image is not present in the local Docker daemon; rerun setup with Docker and public registry access');
- if(preparation.stack==='rails'&&preparation.database?.selected==='postgresql'){
- const sidecar=selectRuntime('postgresql',targetPlatform,dependencies.runtimeCatalog),sidecarAvailability=inspected.images.find(item=>item.kind==='runtime'&&item.id===sidecar.id),sidecarReady=sidecarAvailability?.status==='available',prerequisite=sidecarReady?undefined:sidecarAvailability?.reason??'Exact qualified PostgreSQL sidecar image is not present in the local Docker daemon; rerun setup with Docker and public registry access';
- if(prerequisite)prerequisites.push(prerequisite);requiredSidecars.push({kind:'postgresql',profile:sidecar.id,image:sidecar.image,platform:targetPlatform,availability:sidecarReady?'available':'unavailable',...(prerequisite?{prerequisite}:{})});
+async function doctor(args: string[], dependencies: CsoCliDependencies) {
+ const started = Date.now(),
+ repo = callerPath(need(args, '--repo'));
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown argument: ${args[0]}`);
+ const staticCheck = (async () => {
+ let home = '';
+ try {
+ const stat = fs.statSync(repo);
+ if (!stat.isDirectory()) throw new CsoError('MISSING_INPUT', 'Repository path is not a directory');
+ const g = executable('git');
+ home = secureDirectory(fs.mkdtempSync(join(fs.realpathSync(os.tmpdir()), 'gstack-cso-doctor-git-')));
+ if ((await git(repo, ['rev-parse', '--is-inside-work-tree'], home)).trim() !== 'true')
+ throw new CsoError('MISSING_INPUT', 'Repository path is not a Git working tree');
+ return { capability: 'static-snapshot', status: 'ready', detail: g };
+ } catch (e: any) {
+ return {
+ capability: 'static-snapshot',
+ status: 'missing',
+ detail: e instanceof CsoError ? e.message : 'Repository path is missing or unreadable',
+ };
+ } finally {
+ if (home) fs.rmSync(home, { recursive: true, force: true });
}
- const ready=!prerequisites.length;checks.push({capability:'qualified-runtimes',status:ready?'ready':'missing',detail:{catalog:dependencies.runtimeCatalog.revision,profile:runtime.id,image:runtime.image,platform:targetPlatform,availability:ready?'available':'unavailable',...(requiredSidecars.length?{requiredSidecars}:{}),...(prerequisites.length?{prerequisite:prerequisites[0],prerequisites}:{})}});
- }catch(error:any){checks.push({capability:'qualified-runtimes',status:'missing',detail:error instanceof CsoError?error.message:error?.message??'Qualified runtime catalog is invalid'});}
- for(const id of SCANNER_IDS){try{
- if(!targetPlatform)throw new CsoError('PREREQUISITE','Scanner containers require an amd64/arm64 Linux Docker platform');validateScannerCatalog(scannerCatalog);
- const profile=selectScanner(id,targetPlatform,undefined,scannerCatalog),availability=inspected.images.find(item=>item.kind==='scanner'&&item.id===profile.id);
- if(!availability||availability.status!=='available')checks.push({capability:`scanner:${id}`,status:'missing',detail:{catalog:scannerCatalog.revision,profile:profile.id,image:profile.image,version:profile.version,qualifiedAt:profile.qualifiedAt,availability:'unavailable',prerequisite:availability?.reason??'Exact qualified scanner image is not present in the local Docker daemon; rerun setup with Docker and public registry access'}});
- else checks.push({capability:`scanner:${id}`,status:'ready',detail:{catalog:scannerCatalog.revision,profile:profile.id,image:profile.image,version:profile.version,qualifiedAt:profile.qualifiedAt,availability:'available'}});
- }catch(error:any){checks.push({capability:`scanner:${id}`,status:'missing',detail:error instanceof CsoError?error.message:error?.message??'Qualified scanner catalog is invalid'});}}
- return{schemaVersion:3,downloads:false,elapsedMs:Date.now()-started,checks};
-}
-async function provisionImages(args:string[],dependencies:CsoCliDependencies):Promise{
- const setupSummary=take(args,'--setup-summary'),requestedSeconds=args.includes('--per-image-seconds')?need(args,'--per-image-seconds'):undefined;if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown argument: ${args[0]}`);
- if(requestedSeconds!==undefined)catalogImageProvisioningPolicy(0,requestedSeconds);
- let targetPlatform:'linux/amd64'|'linux/arm64';
- try{targetPlatform=platform();}catch(error){
- const reason=error instanceof CsoError?error.message:'Qualified image provisioning requires an amd64/arm64 Linux Docker platform',result={schemaVersion:1,status:'not_available',downloads:true,platform:'unsupported',requested:0,inspected:0,alreadyPresent:0,downloaded:0,deadlineReached:false,unavailable:[],summary:`Qualified CSO images were not preloaded: ${reason}. Static audits remain available.`};
- return setupSummary?result.summary:result;
- }
- const scannerCatalog=dependencies.scannerCatalog??SCANNER_CATALOG,entries=qualifiedCatalogImages(dependencies.runtimeCatalog,scannerCatalog,targetPlatform),policy=catalogImageProvisioningPolicy(entries.length,requestedSeconds),deadline=Date.now()+policy.aggregateMs,result=await provisionCatalogImages(entries,targetPlatform,dependencies.catalogImageSession??productionCatalogImageSession,deadline,policy.perImageMs);
- return setupSummary?result.summary:result;
-}
-function run(args:string[]){if(!args.length)throw new CsoError('INVALID_ARGUMENT','Run ID is required');return {dir:runDirectory(args.shift()!),report:null as any};}
-function recoveryEvents(dir:string):string[]{const out:string[]=[];let visited=0;const walk=(at:string,depth:number)=>{if(depth>6||visited++>4000)return;let entries:fs.Dirent[];try{entries=fs.readdirSync(at,{withFileTypes:true});}catch{return;}for(const entry of entries){if(!/^[A-Za-z0-9._-]{1,120}$/.test(entry.name))continue;const path=join(at,entry.name);let stat:fs.Stats;try{stat=fs.lstatSync(path);}catch{continue;}if(stat.isSymbolicLink())continue;if(stat.isDirectory()){walk(path,depth+1);continue;}if(!['attempt.event','watchdog.event'].includes(entry.name)||!stat.isFile()||stat.size>8192)continue;try{const message=redact(fs.readFileSync(path,'utf8').trim());if(message&&!out.includes(message))out.push(message);}catch{}}};for(const name of ['supervision','preparation-execution']){const root=join(dir,name);if(!fs.existsSync(root))continue;const stat=fs.lstatSync(root);if(stat.isSymbolicLink()||!stat.isDirectory())throw new CsoError('UNSAFE_PATH','Watchdog recovery state is not a private directory');walk(root,0);}return out;}
-function publicSnapshotPath(manifest:SnapshotManifest,path:string):{path:string;displayPath?:string}{
- const entry=manifest.entries.find(item=>item.path===path),handle=snapshotPathHandle(entry?.pathId??snapshotPathId(manifest.root,path));let displayPath:string;
- try{displayPath=redact(path);}catch{displayPath='[sensitive path withheld]';}
- return displayPath===path?{path}:{path:handle,displayPath};
-}
-function publicSnapshotManifest(manifest:SnapshotManifest):Record{
- return {...manifest,root:PUBLIC_SOURCE_ROOT,entries:manifest.entries.map(entry=>{const {path,pathId:_,...rest}=entry;return{...rest,...publicSnapshotPath(manifest,path)};}),
- ...(manifest.deletedPaths?.length?{deletedPaths:manifest.deletedPaths.map(item=>publicSnapshotPath(manifest,item.path))}:{}),
- ...(manifest.changedPaths?{changedPaths:manifest.changedPaths.map(path=>publicSnapshotPath(manifest,path).path)}:{})};
-}
-function resolveSnapshotPath(manifest:SnapshotManifest,value:unknown,requireEntry:boolean,label='Snapshot path',allowDeleted=false):{path:string;entry?:SnapshotEntry}{
- const reference=snapshotReference(value),handleId=snapshotPathHandleId(reference);
- if(handleId){const entry=manifest.entries.find(item=>item.pathId===handleId);if(entry)return{path:entry.path,entry};const deleted=manifest.deletedPaths?.find(item=>item.pathId===handleId),changed=manifest.changedPaths?.find(path=>snapshotPathId(manifest.root,path)===handleId);if(allowDeleted&&deleted)return{path:deleted.path};if(!requireEntry&&changed)return{path:changed};throw new CsoError('INVALID_SCHEMA',`${label} handle is outside the retained inventory: ${reference}`);}
- const path=relativePath(reference),entry=manifest.entries.find(item=>item.path===path),deleted=manifest.deletedPaths?.find(item=>item.path===path),changed=manifest.changedPaths?.includes(path);if(entry)return{path,entry};if(allowDeleted&&deleted)return{path};if(!requireEntry&&changed)return{path};throw new CsoError('INVALID_SCHEMA',`${label} is outside the retained inventory: ${reference}`);
-}
-type BoundRecheckEvidence={kind:'caller'|'security_boundary';path:string;line:number;observation:string;sourceState:'present'|'absent';snapshotHash:string;sourceHash?:string;executionHash?:string};
-type BoundRecheckClaim={findingId:string;outcome:'open'|'resolved'|'unknown';evidence:BoundRecheckEvidence[];rootCause:string};
-function assertRecheckLine(runDir:string,path:string,entry:SnapshotEntry,line:number,label:string):void{
- if(!entry.executionHash)throw new CsoError('INVALID_SCHEMA',`${label} must reference source available in the fresh execution snapshot`);
- const body=readBoundedStable(containedFile(join(runDir,'snapshot'),path),64*1024*1024,label).toString('utf8'),lines=body.length?(body.endsWith('\n')?body.slice(0,-1):body).split('\n').length:0;
- if(line>lines)throw new CsoError('INVALID_SCHEMA',`${label} line is outside the fresh source file`);
-}
-function originalBoundary(report:RunReportV3):{dir:string;report:RunReportV3;manifest:SnapshotManifest;finding:FindingV3;path:string}{
- if(!report.parent)throw new CsoError('INVALID_SCHEMA','Recheck evidence requires a linked original finding');
- const dir=runDirectory(report.parent.runId),original=loadReport(dir),manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest,
- finding=original.findings.find(item=>item.id===report.parent!.findingId);
- if(report.repoId!==original.repoId)throw new CsoError('INCOMPATIBLE_INPUT','Recheck repository identity differs from the original audit');
- if(!finding)throw new CsoError('MISSING_INPUT','Original recheck finding no longer exists');
- const path=resolveSnapshotPath(manifest,finding.location.path,false,'Original finding boundary',true).path;
- return{dir,report:original,manifest,finding,path};
-}
-function bindRecheckEvidence(value:unknown,runDir:string,manifest:SnapshotManifest,boundary:ReturnType,outcome:BoundRecheckClaim['outcome']):BoundRecheckEvidence[]{
- if(!Array.isArray(value)||value.length<1||value.length>20)throw new CsoError('INVALID_SCHEMA','Recheck evidence must contain 1 to 20 fresh source observations');
- const evidence=value.map((raw,index)=>{
- const item=object(raw,`recheck evidence[${index}]`);rejectUnexpected(item,['kind','path','line','observation'],`recheck evidence[${index}]`);
- const kind=string(item.kind,`recheck evidence[${index}].kind`,32);if(!['caller','security_boundary'].includes(kind))throw new CsoError('INVALID_SCHEMA','Recheck evidence kind must be caller or security_boundary');
- if(!Number.isSafeInteger(item.line)||item.line<1)throw new CsoError('INVALID_SCHEMA',`recheck evidence[${index}].line must be a positive integer`);
- const observation=redact(string(item.observation,`recheck evidence[${index}].observation`,2048)),reference=snapshotReference(item.path);
- let selected:{path:string;entry?:SnapshotEntry}|undefined;
- try{selected=resolveSnapshotPath(manifest,reference,true,`recheck evidence[${index}].path`);}catch(error){
- if(kind!=='security_boundary')throw error;
- let old:{path:string};try{old=resolveSnapshotPath(boundary.manifest,reference,false,`recheck evidence[${index}].path`,true);}catch{throw error;}
- if(old.path!==boundary.path||manifest.entries.some(entry=>entry.path===boundary.path))throw error;
- if(item.line!==boundary.finding.location.line)throw new CsoError('INVALID_SCHEMA','Absent security-boundary evidence must cite the original finding line');
- return{kind:'security_boundary' as const,path:snapshotPathHandle(snapshotPathId(manifest.root,boundary.path)),line:item.line as number,observation,sourceState:'absent' as const,snapshotHash:manifest.originalHash};
- }
- if(!selected.entry||selected.entry.originalHash==='not-read')throw new CsoError('INVALID_SCHEMA','Recheck evidence must reference freshly captured readable source');
- assertRecheckLine(runDir,selected.path,selected.entry,item.line as number,`recheck evidence[${index}]`);
- if(kind==='security_boundary'&&selected.path!==boundary.path)throw new CsoError('INVALID_SCHEMA','Security-boundary evidence must reference the original finding location');
- return{kind:kind as BoundRecheckEvidence['kind'],path:snapshotPathHandle(selected.entry.pathId),line:item.line as number,observation,sourceState:'present' as const,snapshotHash:manifest.originalHash,sourceHash:selected.entry.originalHash,executionHash:selected.entry.executionHash};
+ })();
+ let preparation: ReturnType | undefined;
+ try {
+ const stat = fs.statSync(repo);
+ if (!stat.isDirectory()) throw new CsoError('MISSING_INPUT', 'Repository path is not a directory');
+ preparation = inspectPreparation(repo);
+ } catch {}
+ const scannerCatalog = dependencies.scannerCatalog ?? SCANNER_CATALOG,
+ imageSession = dependencies.catalogImageSession ?? productionCatalogImageSession,
+ deadline = started + 30_000;
+ let targetPlatform: 'linux/amd64' | 'linux/arm64' | undefined,
+ entries: ReturnType = [];
+ try {
+ targetPlatform = platform();
+ entries = qualifiedCatalogImages(dependencies.runtimeCatalog, scannerCatalog, targetPlatform);
+ } catch {}
+ const inspectedPromise = inspectCatalogImages(entries, imageSession, deadline);
+ const checks: any[] = [await staticCheck];
+ checks.push(
+ preparation
+ ? {
+ capability: 'application-preparation',
+ status: preparation.status === 'ready' ? 'ready' : 'missing',
+ detail: { stack: preparation.stack, prerequisites: preparation.prerequisites },
+ }
+ : {
+ capability: 'application-preparation',
+ status: 'missing',
+ detail: 'Repository source is unavailable for inert preparation inspection',
+ },
+ );
+ const inspected = await inspectedPromise;
+ checks.push({
+ capability: 'local-docker-isolation',
+ status: inspected.docker.status,
+ detail: inspected.docker.detail,
});
- const identities=new Set(evidence.map(item=>canonical(item)));if(identities.size!==evidence.length)throw new CsoError('INVALID_SCHEMA','Recheck evidence contains duplicate observations');
- if(outcome==='resolved'&&(!evidence.some(item=>item.kind==='caller'&&item.sourceState==='present')||!evidence.some(item=>item.kind==='security_boundary')))
- throw new CsoError('INVALID_SCHEMA','Resolved closure requires fresh caller evidence and evidence for the original security boundary');
+ try {
+ if (!preparation || preparation.status !== 'ready')
+ throw new CsoError('PREREQUISITE', 'Resolve the application-preparation prerequisites first');
+ if (!targetPlatform)
+ throw new CsoError('PREREQUISITE', 'Qualified runtimes require an amd64/arm64 Linux Docker platform');
+ validateRuntimeCatalog(dependencies.runtimeCatalog);
+ const runtime = selectRuntime(preparation.runtimeProfile, targetPlatform, dependencies.runtimeCatalog),
+ availability = inspected.images.find((item) => item.kind === 'runtime' && item.id === runtime.id),
+ requiredSidecars: Array> = [],
+ prerequisites: string[] = [];
+ if (!availability || availability.status !== 'available')
+ prerequisites.push(
+ availability?.reason ??
+ 'Exact qualified runtime image is not present in the local Docker daemon; rerun setup with Docker and public registry access',
+ );
+ if (preparation.stack === 'rails' && preparation.database?.selected === 'postgresql') {
+ const sidecar = selectRuntime('postgresql', targetPlatform, dependencies.runtimeCatalog),
+ sidecarAvailability = inspected.images.find(
+ (item) => item.kind === 'runtime' && item.id === sidecar.id,
+ ),
+ sidecarReady = sidecarAvailability?.status === 'available',
+ prerequisite = sidecarReady
+ ? undefined
+ : (sidecarAvailability?.reason ??
+ 'Exact qualified PostgreSQL sidecar image is not present in the local Docker daemon; rerun setup with Docker and public registry access');
+ if (prerequisite) prerequisites.push(prerequisite);
+ requiredSidecars.push({
+ kind: 'postgresql',
+ profile: sidecar.id,
+ image: sidecar.image,
+ platform: targetPlatform,
+ availability: sidecarReady ? 'available' : 'unavailable',
+ ...(prerequisite ? { prerequisite } : {}),
+ });
+ }
+ const ready = !prerequisites.length;
+ checks.push({
+ capability: 'qualified-runtimes',
+ status: ready ? 'ready' : 'missing',
+ detail: {
+ catalog: dependencies.runtimeCatalog.revision,
+ profile: runtime.id,
+ image: runtime.image,
+ platform: targetPlatform,
+ availability: ready ? 'available' : 'unavailable',
+ ...(requiredSidecars.length ? { requiredSidecars } : {}),
+ ...(prerequisites.length ? { prerequisite: prerequisites[0], prerequisites } : {}),
+ },
+ });
+ } catch (error: any) {
+ checks.push({
+ capability: 'qualified-runtimes',
+ status: 'missing',
+ detail:
+ error instanceof CsoError
+ ? error.message
+ : (error?.message ?? 'Qualified runtime catalog is invalid'),
+ });
+ }
+ for (const id of SCANNER_IDS) {
+ try {
+ if (!targetPlatform)
+ throw new CsoError('PREREQUISITE', 'Scanner containers require an amd64/arm64 Linux Docker platform');
+ validateScannerCatalog(scannerCatalog);
+ const profile = selectScanner(id, targetPlatform, undefined, scannerCatalog),
+ availability = inspected.images.find((item) => item.kind === 'scanner' && item.id === profile.id);
+ if (!availability || availability.status !== 'available')
+ checks.push({
+ capability: `scanner:${id}`,
+ status: 'missing',
+ detail: {
+ catalog: scannerCatalog.revision,
+ profile: profile.id,
+ image: profile.image,
+ version: profile.version,
+ qualifiedAt: profile.qualifiedAt,
+ availability: 'unavailable',
+ prerequisite:
+ availability?.reason ??
+ 'Exact qualified scanner image is not present in the local Docker daemon; rerun setup with Docker and public registry access',
+ },
+ });
+ else
+ checks.push({
+ capability: `scanner:${id}`,
+ status: 'ready',
+ detail: {
+ catalog: scannerCatalog.revision,
+ profile: profile.id,
+ image: profile.image,
+ version: profile.version,
+ qualifiedAt: profile.qualifiedAt,
+ availability: 'available',
+ },
+ });
+ } catch (error: any) {
+ checks.push({
+ capability: `scanner:${id}`,
+ status: 'missing',
+ detail:
+ error instanceof CsoError
+ ? error.message
+ : (error?.message ?? 'Qualified scanner catalog is invalid'),
+ });
+ }
+ }
+ return { schemaVersion: 3, downloads: false, elapsedMs: Date.now() - started, checks };
+}
+async function provisionImages(args: string[], dependencies: CsoCliDependencies): Promise {
+ const setupSummary = take(args, '--setup-summary'),
+ requestedSeconds = args.includes('--per-image-seconds') ? need(args, '--per-image-seconds') : undefined;
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown argument: ${args[0]}`);
+ if (requestedSeconds !== undefined) catalogImageProvisioningPolicy(0, requestedSeconds);
+ let targetPlatform: 'linux/amd64' | 'linux/arm64';
+ try {
+ targetPlatform = platform();
+ } catch (error) {
+ const reason =
+ error instanceof CsoError
+ ? error.message
+ : 'Qualified image provisioning requires an amd64/arm64 Linux Docker platform',
+ result = {
+ schemaVersion: 1,
+ status: 'not_available',
+ downloads: true,
+ platform: 'unsupported',
+ requested: 0,
+ inspected: 0,
+ alreadyPresent: 0,
+ downloaded: 0,
+ deadlineReached: false,
+ unavailable: [],
+ summary: `Qualified CSO images were not preloaded: ${reason}. Static audits remain available.`,
+ };
+ return setupSummary ? result.summary : result;
+ }
+ const scannerCatalog = dependencies.scannerCatalog ?? SCANNER_CATALOG,
+ entries = qualifiedCatalogImages(dependencies.runtimeCatalog, scannerCatalog, targetPlatform),
+ policy = catalogImageProvisioningPolicy(entries.length, requestedSeconds),
+ deadline = Date.now() + policy.aggregateMs,
+ result = await provisionCatalogImages(
+ entries,
+ targetPlatform,
+ dependencies.catalogImageSession ?? productionCatalogImageSession,
+ deadline,
+ policy.perImageMs,
+ );
+ return setupSummary ? result.summary : result;
+}
+function run(args: string[]) {
+ if (!args.length) throw new CsoError('INVALID_ARGUMENT', 'Run ID is required');
+ return { dir: runDirectory(args.shift()!), report: null as any };
+}
+function recoveryEvents(dir: string): string[] {
+ const out: string[] = [];
+ let visited = 0;
+ const walk = (at: string, depth: number) => {
+ if (depth > 6 || visited++ > 4000) return;
+ let entries: fs.Dirent[];
+ try {
+ entries = fs.readdirSync(at, { withFileTypes: true });
+ } catch {
+ return;
+ }
+ for (const entry of entries) {
+ if (!/^[A-Za-z0-9._-]{1,120}$/.test(entry.name)) continue;
+ const path = join(at, entry.name);
+ let stat: fs.Stats;
+ try {
+ stat = fs.lstatSync(path);
+ } catch {
+ continue;
+ }
+ if (stat.isSymbolicLink()) continue;
+ if (stat.isDirectory()) {
+ walk(path, depth + 1);
+ continue;
+ }
+ if (!['attempt.event', 'watchdog.event'].includes(entry.name) || !stat.isFile() || stat.size > 8192)
+ continue;
+ try {
+ const message = redact(fs.readFileSync(path, 'utf8').trim());
+ if (message && !out.includes(message)) out.push(message);
+ } catch {}
+ }
+ };
+ for (const name of ['supervision', 'preparation-execution']) {
+ const root = join(dir, name);
+ if (!fs.existsSync(root)) continue;
+ const stat = fs.lstatSync(root);
+ if (stat.isSymbolicLink() || !stat.isDirectory())
+ throw new CsoError('UNSAFE_PATH', 'Watchdog recovery state is not a private directory');
+ walk(root, 0);
+ }
+ return out;
+}
+function publicSnapshotPath(
+ manifest: SnapshotManifest,
+ path: string,
+): { path: string; displayPath?: string } {
+ const entry = manifest.entries.find((item) => item.path === path),
+ handle = snapshotPathHandle(entry?.pathId ?? snapshotPathId(manifest.root, path));
+ let displayPath: string;
+ try {
+ displayPath = redact(path);
+ } catch {
+ displayPath = '[sensitive path withheld]';
+ }
+ return displayPath === path ? { path } : { path: handle, displayPath };
+}
+function publicSnapshotManifest(manifest: SnapshotManifest): Record {
+ return {
+ ...manifest,
+ root: PUBLIC_SOURCE_ROOT,
+ entries: manifest.entries.map((entry) => {
+ const { path, pathId: _, ...rest } = entry;
+ return { ...rest, ...publicSnapshotPath(manifest, path) };
+ }),
+ ...(manifest.deletedPaths?.length
+ ? { deletedPaths: manifest.deletedPaths.map((item) => publicSnapshotPath(manifest, item.path)) }
+ : {}),
+ ...(manifest.changedPaths
+ ? { changedPaths: manifest.changedPaths.map((path) => publicSnapshotPath(manifest, path).path) }
+ : {}),
+ };
+}
+function resolveSnapshotPath(
+ manifest: SnapshotManifest,
+ value: unknown,
+ requireEntry: boolean,
+ label = 'Snapshot path',
+ allowDeleted = false,
+): { path: string; entry?: SnapshotEntry } {
+ const reference = snapshotReference(value),
+ handleId = snapshotPathHandleId(reference);
+ if (handleId) {
+ const entry = manifest.entries.find((item) => item.pathId === handleId);
+ if (entry) return { path: entry.path, entry };
+ const deleted = manifest.deletedPaths?.find((item) => item.pathId === handleId),
+ changed = manifest.changedPaths?.find((path) => snapshotPathId(manifest.root, path) === handleId);
+ if (allowDeleted && deleted) return { path: deleted.path };
+ if (!requireEntry && changed) return { path: changed };
+ throw new CsoError('INVALID_SCHEMA', `${label} handle is outside the retained inventory: ${reference}`);
+ }
+ const path = relativePath(reference),
+ entry = manifest.entries.find((item) => item.path === path),
+ deleted = manifest.deletedPaths?.find((item) => item.path === path),
+ changed = manifest.changedPaths?.includes(path);
+ if (entry) return { path, entry };
+ if (allowDeleted && deleted) return { path };
+ if (!requireEntry && changed) return { path };
+ throw new CsoError('INVALID_SCHEMA', `${label} is outside the retained inventory: ${reference}`);
+}
+type BoundRecheckEvidence = {
+ kind: 'caller' | 'security_boundary';
+ path: string;
+ line: number;
+ observation: string;
+ sourceState: 'present' | 'absent';
+ snapshotHash: string;
+ sourceHash?: string;
+ executionHash?: string;
+};
+type BoundRecheckClaim = {
+ findingId: string;
+ outcome: 'open' | 'resolved' | 'unknown';
+ evidence: BoundRecheckEvidence[];
+ rootCause: string;
+};
+function assertRecheckLine(
+ runDir: string,
+ path: string,
+ entry: SnapshotEntry,
+ line: number,
+ label: string,
+): void {
+ if (!entry.executionHash)
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ `${label} must reference source available in the fresh execution snapshot`,
+ );
+ const body = readBoundedStable(
+ containedFile(join(runDir, 'snapshot'), path),
+ 64 * 1024 * 1024,
+ label,
+ ).toString('utf8'),
+ lines = body.length ? (body.endsWith('\n') ? body.slice(0, -1) : body).split('\n').length : 0;
+ if (line > lines) throw new CsoError('INVALID_SCHEMA', `${label} line is outside the fresh source file`);
+}
+function originalBoundary(report: RunReportV3): {
+ dir: string;
+ report: RunReportV3;
+ manifest: SnapshotManifest;
+ finding: FindingV3;
+ path: string;
+} {
+ if (!report.parent)
+ throw new CsoError('INVALID_SCHEMA', 'Recheck evidence requires a linked original finding');
+ const dir = runDirectory(report.parent.runId),
+ original = loadReport(dir),
+ manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest,
+ finding = original.findings.find((item) => item.id === report.parent!.findingId);
+ if (report.repoId !== original.repoId)
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Recheck repository identity differs from the original audit');
+ if (!finding) throw new CsoError('MISSING_INPUT', 'Original recheck finding no longer exists');
+ const path = resolveSnapshotPath(
+ manifest,
+ finding.location.path,
+ false,
+ 'Original finding boundary',
+ true,
+ ).path;
+ return { dir, report: original, manifest, finding, path };
+}
+function bindRecheckEvidence(
+ value: unknown,
+ runDir: string,
+ manifest: SnapshotManifest,
+ boundary: ReturnType,
+ outcome: BoundRecheckClaim['outcome'],
+): BoundRecheckEvidence[] {
+ if (!Array.isArray(value) || value.length < 1 || value.length > 20)
+ throw new CsoError('INVALID_SCHEMA', 'Recheck evidence must contain 1 to 20 fresh source observations');
+ const evidence = value.map((raw, index) => {
+ const item = object(raw, `recheck evidence[${index}]`);
+ rejectUnexpected(item, ['kind', 'path', 'line', 'observation'], `recheck evidence[${index}]`);
+ const kind = string(item.kind, `recheck evidence[${index}].kind`, 32);
+ if (!['caller', 'security_boundary'].includes(kind))
+ throw new CsoError('INVALID_SCHEMA', 'Recheck evidence kind must be caller or security_boundary');
+ if (!Number.isSafeInteger(item.line) || item.line < 1)
+ throw new CsoError('INVALID_SCHEMA', `recheck evidence[${index}].line must be a positive integer`);
+ const observation = redact(string(item.observation, `recheck evidence[${index}].observation`, 2048)),
+ reference = snapshotReference(item.path);
+ let selected: { path: string; entry?: SnapshotEntry } | undefined;
+ try {
+ selected = resolveSnapshotPath(manifest, reference, true, `recheck evidence[${index}].path`);
+ } catch (error) {
+ if (kind !== 'security_boundary') throw error;
+ let old: { path: string };
+ try {
+ old = resolveSnapshotPath(
+ boundary.manifest,
+ reference,
+ false,
+ `recheck evidence[${index}].path`,
+ true,
+ );
+ } catch {
+ throw error;
+ }
+ if (old.path !== boundary.path || manifest.entries.some((entry) => entry.path === boundary.path))
+ throw error;
+ if (item.line !== boundary.finding.location.line)
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Absent security-boundary evidence must cite the original finding line',
+ );
+ return {
+ kind: 'security_boundary' as const,
+ path: snapshotPathHandle(snapshotPathId(manifest.root, boundary.path)),
+ line: item.line as number,
+ observation,
+ sourceState: 'absent' as const,
+ snapshotHash: manifest.originalHash,
+ };
+ }
+ if (!selected.entry || selected.entry.originalHash === 'not-read')
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Recheck evidence must reference freshly captured readable source',
+ );
+ assertRecheckLine(
+ runDir,
+ selected.path,
+ selected.entry,
+ item.line as number,
+ `recheck evidence[${index}]`,
+ );
+ if (kind === 'security_boundary' && selected.path !== boundary.path)
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Security-boundary evidence must reference the original finding location',
+ );
+ return {
+ kind: kind as BoundRecheckEvidence['kind'],
+ path: snapshotPathHandle(selected.entry.pathId),
+ line: item.line as number,
+ observation,
+ sourceState: 'present' as const,
+ snapshotHash: manifest.originalHash,
+ sourceHash: selected.entry.originalHash,
+ executionHash: selected.entry.executionHash,
+ };
+ });
+ const identities = new Set(evidence.map((item) => canonical(item)));
+ if (identities.size !== evidence.length)
+ throw new CsoError('INVALID_SCHEMA', 'Recheck evidence contains duplicate observations');
+ if (
+ outcome === 'resolved' &&
+ (!evidence.some((item) => item.kind === 'caller' && item.sourceState === 'present') ||
+ !evidence.some((item) => item.kind === 'security_boundary'))
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Resolved closure requires fresh caller evidence and evidence for the original security boundary',
+ );
return evidence;
}
-function validateBoundRecheckClaim(value:unknown,runDir:string,manifest:SnapshotManifest,boundary:ReturnType):BoundRecheckClaim{
- const raw=object(value,'retained recheck claim');rejectUnexpected(raw,['findingId','outcome','evidence','rootCause'],'retained recheck claim');
- const findingId=string(raw.findingId,'retained recheck findingId'),rootCause=string(raw.rootCause,'retained recheck rootCause'),outcome=raw.outcome;
- if(!['open','resolved','unknown'].includes(outcome as string))throw new CsoError('INVALID_SCHEMA','Retained recheck outcome is invalid');
- if(!Array.isArray(raw.evidence)||raw.evidence.length<1||raw.evidence.length>20)throw new CsoError('INVALID_SCHEMA','Retained recheck evidence is invalid');
- const evidence=raw.evidence.map((itemRaw,index)=>{
- const item=object(itemRaw,`retained recheck evidence[${index}]`);rejectUnexpected(item,['kind','path','line','observation','sourceState','snapshotHash','sourceHash','executionHash'],`retained recheck evidence[${index}]`);
- const kind=string(item.kind,`retained recheck evidence[${index}].kind`,32),sourceState=string(item.sourceState,`retained recheck evidence[${index}].sourceState`,16);
- if(!['caller','security_boundary'].includes(kind)||!['present','absent'].includes(sourceState))throw new CsoError('INVALID_SCHEMA','Retained recheck evidence type is invalid');
- if(!Number.isSafeInteger(item.line)||item.line<1)throw new CsoError('INVALID_SCHEMA','Retained recheck evidence line is invalid');
- const observation=string(item.observation,`retained recheck evidence[${index}].observation`,2048),reference=snapshotReference(item.path);
- if(item.snapshotHash!==manifest.originalHash)throw new CsoError('INCOMPATIBLE_INPUT','Retained recheck evidence is not bound to the complete fresh snapshot inventory');
- if(sourceState==='present'){
- const selected=resolveSnapshotPath(manifest,reference,true,'Retained recheck evidence path');
- if(!selected.entry||selected.entry.originalHash==='not-read'||typeof item.sourceHash!=='string'||item.sourceHash!==selected.entry.originalHash||typeof item.executionHash!=='string'||item.executionHash!==selected.entry.executionHash)
- throw new CsoError('INCOMPATIBLE_INPUT','Retained recheck evidence is not bound to the fresh snapshot');
- assertRecheckLine(runDir,selected.path,selected.entry,item.line as number,`retained recheck evidence[${index}]`);
- if(kind==='security_boundary'&&selected.path!==boundary.path)throw new CsoError('INCOMPATIBLE_INPUT','Retained security-boundary evidence changed location');
- return{kind:kind as BoundRecheckEvidence['kind'],path:snapshotPathHandle(selected.entry.pathId),line:item.line as number,observation,sourceState:'present' as const,snapshotHash:item.snapshotHash as string,sourceHash:item.sourceHash,executionHash:item.executionHash};
+function validateBoundRecheckClaim(
+ value: unknown,
+ runDir: string,
+ manifest: SnapshotManifest,
+ boundary: ReturnType,
+): BoundRecheckClaim {
+ const raw = object(value, 'retained recheck claim');
+ rejectUnexpected(raw, ['findingId', 'outcome', 'evidence', 'rootCause'], 'retained recheck claim');
+ const findingId = string(raw.findingId, 'retained recheck findingId'),
+ rootCause = string(raw.rootCause, 'retained recheck rootCause'),
+ outcome = raw.outcome;
+ if (!['open', 'resolved', 'unknown'].includes(outcome as string))
+ throw new CsoError('INVALID_SCHEMA', 'Retained recheck outcome is invalid');
+ if (!Array.isArray(raw.evidence) || raw.evidence.length < 1 || raw.evidence.length > 20)
+ throw new CsoError('INVALID_SCHEMA', 'Retained recheck evidence is invalid');
+ const evidence = raw.evidence.map((itemRaw, index) => {
+ const item = object(itemRaw, `retained recheck evidence[${index}]`);
+ rejectUnexpected(
+ item,
+ ['kind', 'path', 'line', 'observation', 'sourceState', 'snapshotHash', 'sourceHash', 'executionHash'],
+ `retained recheck evidence[${index}]`,
+ );
+ const kind = string(item.kind, `retained recheck evidence[${index}].kind`, 32),
+ sourceState = string(item.sourceState, `retained recheck evidence[${index}].sourceState`, 16);
+ if (!['caller', 'security_boundary'].includes(kind) || !['present', 'absent'].includes(sourceState))
+ throw new CsoError('INVALID_SCHEMA', 'Retained recheck evidence type is invalid');
+ if (!Number.isSafeInteger(item.line) || item.line < 1)
+ throw new CsoError('INVALID_SCHEMA', 'Retained recheck evidence line is invalid');
+ const observation = string(item.observation, `retained recheck evidence[${index}].observation`, 2048),
+ reference = snapshotReference(item.path);
+ if (item.snapshotHash !== manifest.originalHash)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Retained recheck evidence is not bound to the complete fresh snapshot inventory',
+ );
+ if (sourceState === 'present') {
+ const selected = resolveSnapshotPath(manifest, reference, true, 'Retained recheck evidence path');
+ if (
+ !selected.entry ||
+ selected.entry.originalHash === 'not-read' ||
+ typeof item.sourceHash !== 'string' ||
+ item.sourceHash !== selected.entry.originalHash ||
+ typeof item.executionHash !== 'string' ||
+ item.executionHash !== selected.entry.executionHash
+ )
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Retained recheck evidence is not bound to the fresh snapshot',
+ );
+ assertRecheckLine(
+ runDir,
+ selected.path,
+ selected.entry,
+ item.line as number,
+ `retained recheck evidence[${index}]`,
+ );
+ if (kind === 'security_boundary' && selected.path !== boundary.path)
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Retained security-boundary evidence changed location');
+ return {
+ kind: kind as BoundRecheckEvidence['kind'],
+ path: snapshotPathHandle(selected.entry.pathId),
+ line: item.line as number,
+ observation,
+ sourceState: 'present' as const,
+ snapshotHash: item.snapshotHash as string,
+ sourceHash: item.sourceHash,
+ executionHash: item.executionHash,
+ };
}
- if(kind!=='security_boundary'||item.sourceHash!==undefined||item.executionHash!==undefined)throw new CsoError('INVALID_SCHEMA','Only an absent original security boundary can use absent evidence');
- const old=resolveSnapshotPath(boundary.manifest,reference,false,'Retained absent security boundary',true);
- if(old.path!==boundary.path||manifest.entries.some(entry=>entry.path===boundary.path)||item.line!==boundary.finding.location.line)throw new CsoError('INCOMPATIBLE_INPUT','Retained absent-boundary evidence does not match the fresh snapshot');
- return{kind:'security_boundary' as const,path:snapshotPathHandle(snapshotPathId(manifest.root,boundary.path)),line:item.line as number,observation,sourceState:'absent' as const,snapshotHash:item.snapshotHash as string};
+ if (kind !== 'security_boundary' || item.sourceHash !== undefined || item.executionHash !== undefined)
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Only an absent original security boundary can use absent evidence',
+ );
+ const old = resolveSnapshotPath(
+ boundary.manifest,
+ reference,
+ false,
+ 'Retained absent security boundary',
+ true,
+ );
+ if (
+ old.path !== boundary.path ||
+ manifest.entries.some((entry) => entry.path === boundary.path) ||
+ item.line !== boundary.finding.location.line
+ )
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Retained absent-boundary evidence does not match the fresh snapshot',
+ );
+ return {
+ kind: 'security_boundary' as const,
+ path: snapshotPathHandle(snapshotPathId(manifest.root, boundary.path)),
+ line: item.line as number,
+ observation,
+ sourceState: 'absent' as const,
+ snapshotHash: item.snapshotHash as string,
+ };
});
- if(outcome==='resolved'&&(!evidence.some(item=>item.kind==='caller'&&item.sourceState==='present')||!evidence.some(item=>item.kind==='security_boundary')))
- throw new CsoError('INVALID_SCHEMA','Resolved closure lacks caller or original security-boundary evidence');
- if(new Set(evidence.map(item=>canonical(item))).size!==evidence.length)throw new CsoError('INVALID_SCHEMA','Retained recheck evidence contains duplicates');
- return{findingId,outcome:outcome as BoundRecheckClaim['outcome'],evidence,rootCause};
+ if (
+ outcome === 'resolved' &&
+ (!evidence.some((item) => item.kind === 'caller' && item.sourceState === 'present') ||
+ !evidence.some((item) => item.kind === 'security_boundary'))
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Resolved closure lacks caller or original security-boundary evidence',
+ );
+ if (new Set(evidence.map((item) => canonical(item))).size !== evidence.length)
+ throw new CsoError('INVALID_SCHEMA', 'Retained recheck evidence contains duplicates');
+ return { findingId, outcome: outcome as BoundRecheckClaim['outcome'], evidence, rootCause };
}
-function requireReportingTime(report:RunReportV3):void{if(Date.now()>=Date.parse(report.deadline))throw new CsoError('DEADLINE','Audit deadline reached; no further evidence can be accepted');}
-function submit(args:string[]){
- const {dir}=run(args);if(args.length!==1)throw new CsoError('INVALID_ARGUMENT','submit requires one JSON file');const rawInput=object(readInput(args[0]),'submission');rejectUnexpected(rawInput,['application','findings','coverage','gaps','modelUsage','recheck'],'submission');const input=rawInput as SubmissionV3;
- return withLock(dir,()=>{const report=loadReport(dir),manifest=readJson(join(dir,'snapshot.json'));assertSnapshot(dir,manifest);requireReportingTime(report);if(report.status!=='running')throw new CsoError('INVALID_SCHEMA','Only a running audit accepts evidence');
- if(input.findings!==undefined&&!Array.isArray(input.findings))throw new CsoError('INVALID_SCHEMA','submission.findings must be an array');
- if(input.coverage!==undefined&&!Array.isArray(input.coverage))throw new CsoError('INVALID_SCHEMA','submission.coverage must be an array');
- if(input.application)report.application=model(input.application);
- for(const raw of input.findings??[]){const sourceFinding=validateFinding(raw),sourceLocation=resolveSnapshotPath(manifest,sourceFinding.location.path,false,'Finding path',true);if(report.policy.diff&&(!Array.isArray(manifest.changedPaths)||!manifest.changedPaths.includes(sourceLocation.path)))throw new CsoError('INVALID_SCHEMA',`Diff-scope finding root cause is outside the captured changed paths: ${sourceFinding.location.path}`);const safeRaw=object(sanitizeForJson(raw),'finding'),safeLocation=object(safeRaw.location,'location');safeLocation.path=publicSnapshotPath(manifest,sourceLocation.path).path;const f=validateFinding(safeRaw),normalizedLocation=resolveSnapshotPath(manifest,f.location.path,false,'Finding path',true);if(normalizedLocation.path!==sourceLocation.path)throw new CsoError('INVALID_SCHEMA','Finding path identity changed during redaction');if(report.policy.mode==='daily'&&f.evidence==='hypothesis')throw new CsoError('INVALID_SCHEMA','Daily reports contain supported findings only');const old=report.findings.findIndex(x=>x.fingerprint===f.fingerprint);if(old<0){report.findings.push(f);event(report,'early-finding',`${f.severity} ${f.evidence} finding ${f.id}`);}else report.findings[old]={...f,reproduction:report.findings[old].reproduction,repair:report.findings[old].repair,closure:report.findings[old].closure,verificationId:report.findings[old].verificationId,reproductionAttemptId:report.findings[old].reproductionAttemptId,verificationAssurance:report.findings[old].verificationAssurance};}
- for(const raw of input.coverage??[]){const c=validateCoverage(raw);if(helperOwnedCoverage(c.domain))throw new CsoError('INVALID_SCHEMA',`Coverage domain is helper-owned: ${c.domain}`);const i=report.coverage.findIndex(x=>x.domain===c.domain&&x.scope===c.scope);if(i<0)report.coverage.push(c);else report.coverage[i]=c;}
- if(input.gaps)report.gaps=strings(input.gaps,'gaps');
- if(input.modelUsage){const u=object(input.modelUsage,'model usage');rejectUnexpected(u,['source','tokens','cost'],'model usage');if(!Number.isInteger(u.tokens)||u.tokens<0||('cost'in u&&(typeof u.cost!=='number'||!Number.isFinite(u.cost)||u.cost<0)))throw new CsoError('INVALID_SCHEMA','Model usage must be host-reported finite nonnegative numbers');report.modelUsage={source:string(u.source,'usage source'),tokens:u.tokens,...(typeof u.cost==='number'?{cost:u.cost}:{})};}
- let recheckClaim:BoundRecheckClaim|undefined;
- if(input.recheck){const rawClaim=object(input.recheck,'recheck claim');rejectUnexpected(rawClaim,['findingId','outcome','evidence','rootCause'],'recheck claim');if(!report.parent||input.recheck.findingId!==report.parent.findingId)throw new CsoError('INVALID_SCHEMA','Recheck claim must target the linked original finding');const outcome=input.recheck.outcome;if(!['open','resolved','unknown'].includes(outcome))throw new CsoError('INVALID_SCHEMA','Recheck needs an open, resolved, or unknown outcome');const boundary=originalBoundary(report),claim={findingId:string(input.recheck.findingId,'findingId'),outcome,evidence:bindRecheckEvidence(input.recheck.evidence,dir,manifest,boundary,outcome),rootCause:string(input.recheck.rootCause,'rootCause')};recheckClaim=claim;}
+function requireReportingTime(report: RunReportV3): void {
+ if (Date.now() >= Date.parse(report.deadline))
+ throw new CsoError('DEADLINE', 'Audit deadline reached; no further evidence can be accepted');
+}
+function submit(args: string[]) {
+ const { dir } = run(args);
+ if (args.length !== 1) throw new CsoError('INVALID_ARGUMENT', 'submit requires one JSON file');
+ const rawInput = object(readInput(args[0]), 'submission');
+ rejectUnexpected(
+ rawInput,
+ ['application', 'findings', 'coverage', 'gaps', 'modelUsage', 'recheck'],
+ 'submission',
+ );
+ const input = rawInput as SubmissionV3;
+ return withLock(dir, () => {
+ const report = loadReport(dir),
+ manifest = readJson(join(dir, 'snapshot.json'));
+ assertSnapshot(dir, manifest);
+ requireReportingTime(report);
+ if (report.status !== 'running')
+ throw new CsoError('INVALID_SCHEMA', 'Only a running audit accepts evidence');
+ if (input.findings !== undefined && !Array.isArray(input.findings))
+ throw new CsoError('INVALID_SCHEMA', 'submission.findings must be an array');
+ if (input.coverage !== undefined && !Array.isArray(input.coverage))
+ throw new CsoError('INVALID_SCHEMA', 'submission.coverage must be an array');
+ if (input.application) report.application = model(input.application);
+ for (const raw of input.findings ?? []) {
+ const sourceFinding = validateFinding(raw),
+ sourceLocation = resolveSnapshotPath(
+ manifest,
+ sourceFinding.location.path,
+ false,
+ 'Finding path',
+ true,
+ );
+ if (
+ report.policy.diff &&
+ (!Array.isArray(manifest.changedPaths) || !manifest.changedPaths.includes(sourceLocation.path))
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ `Diff-scope finding root cause is outside the captured changed paths: ${sourceFinding.location.path}`,
+ );
+ const safeRaw = object(sanitizeForJson(raw), 'finding'),
+ safeLocation = object(safeRaw.location, 'location');
+ safeLocation.path = publicSnapshotPath(manifest, sourceLocation.path).path;
+ const f = validateFinding(safeRaw),
+ normalizedLocation = resolveSnapshotPath(manifest, f.location.path, false, 'Finding path', true);
+ if (normalizedLocation.path !== sourceLocation.path)
+ throw new CsoError('INVALID_SCHEMA', 'Finding path identity changed during redaction');
+ if (report.policy.mode === 'daily' && f.evidence === 'hypothesis')
+ throw new CsoError('INVALID_SCHEMA', 'Daily reports contain supported findings only');
+ const old = report.findings.findIndex((x) => x.fingerprint === f.fingerprint);
+ if (old < 0) {
+ report.findings.push(f);
+ event(report, 'early-finding', `${f.severity} ${f.evidence} finding ${f.id}`);
+ } else
+ report.findings[old] = {
+ ...f,
+ reproduction: report.findings[old].reproduction,
+ repair: report.findings[old].repair,
+ closure: report.findings[old].closure,
+ verificationId: report.findings[old].verificationId,
+ reproductionAttemptId: report.findings[old].reproductionAttemptId,
+ verificationAssurance: report.findings[old].verificationAssurance,
+ };
+ }
+ for (const raw of input.coverage ?? []) {
+ const c = validateCoverage(raw);
+ if (helperOwnedCoverage(c.domain))
+ throw new CsoError('INVALID_SCHEMA', `Coverage domain is helper-owned: ${c.domain}`);
+ const i = report.coverage.findIndex((x) => x.domain === c.domain && x.scope === c.scope);
+ if (i < 0) report.coverage.push(c);
+ else report.coverage[i] = c;
+ }
+ if (input.gaps) report.gaps = strings(input.gaps, 'gaps');
+ if (input.modelUsage) {
+ const u = object(input.modelUsage, 'model usage');
+ rejectUnexpected(u, ['source', 'tokens', 'cost'], 'model usage');
+ if (
+ !Number.isInteger(u.tokens) ||
+ u.tokens < 0 ||
+ ('cost' in u && (typeof u.cost !== 'number' || !Number.isFinite(u.cost) || u.cost < 0))
+ )
+ throw new CsoError('INVALID_SCHEMA', 'Model usage must be host-reported finite nonnegative numbers');
+ report.modelUsage = {
+ source: string(u.source, 'usage source'),
+ tokens: u.tokens,
+ ...(typeof u.cost === 'number' ? { cost: u.cost } : {}),
+ };
+ }
+ let recheckClaim: BoundRecheckClaim | undefined;
+ if (input.recheck) {
+ const rawClaim = object(input.recheck, 'recheck claim');
+ rejectUnexpected(rawClaim, ['findingId', 'outcome', 'evidence', 'rootCause'], 'recheck claim');
+ if (!report.parent || input.recheck.findingId !== report.parent.findingId)
+ throw new CsoError('INVALID_SCHEMA', 'Recheck claim must target the linked original finding');
+ const outcome = input.recheck.outcome;
+ if (!['open', 'resolved', 'unknown'].includes(outcome))
+ throw new CsoError('INVALID_SCHEMA', 'Recheck needs an open, resolved, or unknown outcome');
+ const boundary = originalBoundary(report),
+ claim = {
+ findingId: string(input.recheck.findingId, 'findingId'),
+ outcome,
+ evidence: bindRecheckEvidence(input.recheck.evidence, dir, manifest, boundary, outcome),
+ rootCause: string(input.recheck.rootCause, 'rootCause'),
+ };
+ recheckClaim = claim;
+ }
// A claim can close a prior finding. Publish it only after every report
// mutation it relies on is durably accepted, so a failed submission never
// leaves closure evidence behind.
- saveReport(dir,report);if(recheckClaim)writeHelperJson(join(dir,'recheck-claim.json'),recheckClaim);return {runId:report.runId,findings:report.findings.length,completeness:report.completeness};});
+ saveReport(dir, report);
+ if (recheckClaim) writeHelperJson(join(dir, 'recheck-claim.json'), recheckClaim);
+ return { runId: report.runId, findings: report.findings.length, completeness: report.completeness };
+ });
+}
+function finish(args: string[]) {
+ const { dir } = run(args);
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', 'finish takes only a run ID');
+ return withLock(dir, () => {
+ const report = loadReport(dir),
+ manifest = readJson(join(dir, 'snapshot.json'));
+ assertSnapshot(dir, manifest);
+ for (const c of report.coverage)
+ if (
+ c.status === 'not_assessed' &&
+ !c.domain.startsWith('scanner:') &&
+ !report.gaps.includes(`${c.domain}: not assessed`)
+ )
+ report.gaps.push(`${c.domain}: not assessed`);
+ if (!report.application.actors.length && !report.gaps.includes('Application model was not completed'))
+ report.gaps.push('Application model was not completed');
+ report.completeness = completeness(report);
+ const persistTerminal = () => {
+ report.status = 'finished';
+ if (!report.events.some((e) => e.kind === 'terminal'))
+ event(report, 'terminal', 'Audit finished and retained according to the private-state policy');
+ saveReport(dir, report);
+ return {
+ runId: report.runId,
+ status: report.status,
+ completeness: report.completeness,
+ report: 'report.md',
+ };
+ };
+ if (report.parent && fs.existsSync(join(dir, 'recheck-claim.json'))) {
+ const boundary = originalBoundary(report),
+ claim = validateBoundRecheckClaim(readJson(join(dir, 'recheck-claim.json')), dir, manifest, boundary);
+ if (claim.outcome === 'resolved') {
+ if (report.completeness !== 'complete')
+ throw new CsoError('INVALID_SCHEMA', 'Partial or incompatible rechecks cannot establish closure');
+ const originalDir = boundary.dir;
+ return withLock(originalDir, () => {
+ const original = loadReport(originalDir),
+ finding = original.findings.find((f) => f.id === claim.findingId);
+ if (report.repoId !== original.repoId)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Recheck repository identity differs from the original audit',
+ );
+ const survivingVariant =
+ !!finding &&
+ report.findings.some(
+ (candidate) =>
+ candidate.fingerprint === finding.fingerprint ||
+ rootCauseIdentity(candidate.rootCause) === rootCauseIdentity(finding.rootCause) ||
+ (candidate.advisoryIds.length > 0 &&
+ candidate.advisoryIds.some((id) => finding.advisoryIds.includes(id))),
+ );
+ if (
+ !finding ||
+ finding.id !== boundary.finding.id ||
+ rootCauseIdentity(finding.rootCause) !== rootCauseIdentity(claim.rootCause) ||
+ survivingVariant
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Closure needs matching root cause, fresh caller and original-boundary evidence, and no surviving root-cause or advisory variant',
+ );
+ const result = persistTerminal();
+ if (finding.closure !== 'resolved') {
+ finding.closure = 'resolved';
+ event(original, 'closure', `Fresh recheck ${report.runId} resolved ${finding.id}`);
+ saveReport(originalDir, original);
+ }
+ return result;
+ });
+ }
+ }
+ return persistTerminal();
+ });
}
-function finish(args:string[]){const {dir}=run(args);if(args.length)throw new CsoError('INVALID_ARGUMENT','finish takes only a run ID');return withLock(dir,()=>{
- const report=loadReport(dir),manifest=readJson(join(dir,'snapshot.json'));assertSnapshot(dir,manifest);for(const c of report.coverage)if(c.status==='not_assessed'&&!c.domain.startsWith('scanner:')&&!report.gaps.includes(`${c.domain}: not assessed`))report.gaps.push(`${c.domain}: not assessed`);
- if(!report.application.actors.length&&!report.gaps.includes('Application model was not completed'))report.gaps.push('Application model was not completed');report.completeness=completeness(report);
- const persistTerminal=()=>{report.status='finished';if(!report.events.some(e=>e.kind==='terminal'))event(report,'terminal','Audit finished and retained according to the private-state policy');saveReport(dir,report);return{runId:report.runId,status:report.status,completeness:report.completeness,report:'report.md'};};
- if(report.parent&&fs.existsSync(join(dir,'recheck-claim.json'))){const boundary=originalBoundary(report),claim=validateBoundRecheckClaim(readJson(join(dir,'recheck-claim.json')),dir,manifest,boundary);if(claim.outcome==='resolved'){
- if(report.completeness!=='complete')throw new CsoError('INVALID_SCHEMA','Partial or incompatible rechecks cannot establish closure');
- const originalDir=boundary.dir;return withLock(originalDir,()=>{const original=loadReport(originalDir),finding=original.findings.find(f=>f.id===claim.findingId);
- if(report.repoId!==original.repoId)throw new CsoError('INCOMPATIBLE_INPUT','Recheck repository identity differs from the original audit');
- const survivingVariant=!!finding&&report.findings.some(candidate=>candidate.fingerprint===finding.fingerprint||rootCauseIdentity(candidate.rootCause)===rootCauseIdentity(finding.rootCause)||(candidate.advisoryIds.length>0&&candidate.advisoryIds.some(id=>finding.advisoryIds.includes(id))));
- if(!finding||finding.id!==boundary.finding.id||rootCauseIdentity(finding.rootCause)!==rootCauseIdentity(claim.rootCause)||survivingVariant)throw new CsoError('INVALID_SCHEMA','Closure needs matching root cause, fresh caller and original-boundary evidence, and no surviving root-cause or advisory variant');
- const result=persistTerminal();if(finding.closure!=='resolved'){finding.closure='resolved';event(original,'closure',`Fresh recheck ${report.runId} resolved ${finding.id}`);saveReport(originalDir,original);}return result;});
- }}return persistTerminal();});}
-async function inspect(args:string[]){const {dir}=run(args);if(args.length)throw new CsoError('INVALID_ARGUMENT','inspect takes one run ID');const report=loadReport(dir),manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest;assertSnapshot(dir,manifest);const rawSensitive=readJson(join(dir,'sensitive-evidence.json')),sensitiveEvidence=Array.isArray(rawSensitive)?rawSensitive.map(item=>{if(!item||typeof item!=='object'||typeof item.path!=='string')return item;const id=snapshotPathHandleId(item.path),exact=id?manifest.entries.find(entry=>entry.pathId===id)?.path:item.path;if(!exact)throw new CsoError('INCOMPATIBLE_INPUT','Sensitive-evidence path handle is outside the snapshot');return{...item,...publicSnapshotPath(manifest,exact)};}):rawSensitive;emit({report,manifest:publicSnapshotManifest(manifest),history:readJson(join(dir,'history-status.json')),sensitiveEvidence,preparation:fs.existsSync(join(dir,'preparation.json'))?readJson(join(dir,'preparation.json')):undefined,recovery:recoveryEvents(dir)});}
-async function read(args:string[]){const {dir}=run(args);if(args.length!==1)throw new CsoError('INVALID_ARGUMENT','read requires one path or opaque handle');const manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest;assertSnapshot(dir,manifest);const selected=resolveSnapshotPath(manifest,args[0],true),full=containedFile(join(dir,'readable'),selected.path),data=readBoundedStable(full,1024*1024,'Snapshot path');emit(data.toString('utf8'));}
-async function history(args:string[]){const {dir}=run(args),manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest;assertSnapshot(dir,manifest);let selected:string|undefined;if(args.length)selected=resolveSnapshotPath(manifest,args.shift(),false,'History path',true).path;if(args.length)throw new CsoError('INVALID_ARGUMENT','history accepts at most one path or opaque handle');const status=readJson(join(dir,'history-status.json'));if(status.status!=='captured'||!fs.existsSync(join(dir,'history.txt')))throw new CsoError('MISSING_INPUT',status.gap||'Historical evidence was not retained');const raw=fs.readFileSync(join(dir,'history.txt'),'utf8');if(!selected){emit(raw);return;}
- const displayPath=publicSnapshotPath(manifest,selected).displayPath??selected,retained=historyForPath(raw,displayPath);emit(retained??`No retained patch hunks for ${displayPath}`);}
-function resume(args:string[]){const {dir}=run(args);if(args.length)throw new CsoError('INVALID_ARGUMENT','resume takes one run ID');return withLock(dir,()=>{const report=loadReport(dir),manifest=readJson(join(dir,'snapshot.json'));if(report.status==='finished')throw new CsoError('INVALID_SCHEMA','A finished audit cannot be resumed');assertSnapshot(dir,manifest);for(const message of recoveryEvents(dir))if(!report.events.some(e=>e.kind==='watchdog-recovery'&&e.message===message))event(report,'watchdog-recovery',message);if(Date.now()>=Date.parse(report.deadline)){report.status='interrupted';event(report,'deadline','Original budget is exhausted; resume did not replenish it');saveReport(dir,report);throw new CsoError('DEADLINE','Original run budget is exhausted');}report.status='running';event(report,'resume','Continued retained snapshot under original policy');saveReport(dir,report);return {runId:report.runId,deadline:report.deadline,policy:report.policy,recovery:recoveryEvents(dir)};});}
-function importV2(args:string[]){if(args.length!==1)throw new CsoError('INVALID_ARGUMENT','import-v2 requires one report file');const legacy=sanitizeForJson(importLegacy(readInput(args[0]))) as ReturnType,id=sha256(JSON.stringify(legacy)),dir=secureDirectory(join(privateRoot(),'legacy-imports'));writeJson(join(dir,`${id}.json`),legacy);return {id,path:`legacy-imports/${id}.json`,warning:legacy.warning,report:legacy};}
-function inspectV2(args:string[]){if(args.length!==1||!/^[a-f0-9]{64}$/.test(args[0]??''))throw new CsoError('INVALID_ARGUMENT','inspect-v2 requires the 64-character import ID');const id=args[0],file=join(secureDirectory(join(privateRoot(),'legacy-imports')),`${id}.json`);if(!fs.existsSync(file))throw new CsoError('MISSING_INPUT','Legacy report import was not found or expired');const report=readJson(file);if(report?.schemaVersion!==2||report?.readOnly!==true||!Array.isArray(report?.findings))throw new CsoError('INCOMPATIBLE_INPUT','Stored legacy report is incompatible');if(sha256(JSON.stringify(report))!==id)throw new CsoError('INCOMPATIBLE_INPUT','Stored legacy report identity is inconsistent');return{id,report};}
-async function scanner(args:string[],sarif=false){
- const {dir}=run(args);
- if(sarif?args.length!==1:(args.length<1||args.length>2||!SCANNER_IDS.includes(args[0] as ScannerId)))throw new CsoError('INVALID_ARGUMENT',sarif?'import-sarif needs one file':'scan requires a supported scanner ID and optional request JSON');
- const id=args[0] as ScannerId,request=!sarif?validateScannerRequest(args[1]?readInput(args[1]):{},id):undefined;
+async function inspect(args: string[]) {
+ const { dir } = run(args);
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', 'inspect takes one run ID');
+ const report = loadReport(dir),
+ manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest;
+ assertSnapshot(dir, manifest);
+ const rawSensitive = readJson(join(dir, 'sensitive-evidence.json')),
+ sensitiveEvidence = Array.isArray(rawSensitive)
+ ? rawSensitive.map((item) => {
+ if (!item || typeof item !== 'object' || typeof item.path !== 'string') return item;
+ const id = snapshotPathHandleId(item.path),
+ exact = id ? manifest.entries.find((entry) => entry.pathId === id)?.path : item.path;
+ if (!exact)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Sensitive-evidence path handle is outside the snapshot',
+ );
+ return { ...item, ...publicSnapshotPath(manifest, exact) };
+ })
+ : rawSensitive;
+ emit({
+ report,
+ manifest: publicSnapshotManifest(manifest),
+ history: readJson(join(dir, 'history-status.json')),
+ sensitiveEvidence,
+ preparation: fs.existsSync(join(dir, 'preparation.json'))
+ ? readJson(join(dir, 'preparation.json'))
+ : undefined,
+ recovery: recoveryEvents(dir),
+ });
+}
+async function read(args: string[]) {
+ const { dir } = run(args);
+ if (args.length !== 1) throw new CsoError('INVALID_ARGUMENT', 'read requires one path or opaque handle');
+ const manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest;
+ assertSnapshot(dir, manifest);
+ const selected = resolveSnapshotPath(manifest, args[0], true),
+ full = containedFile(join(dir, 'readable'), selected.path),
+ data = readBoundedStable(full, 1024 * 1024, 'Snapshot path');
+ emit(data.toString('utf8'));
+}
+async function history(args: string[]) {
+ const { dir } = run(args),
+ manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest;
+ assertSnapshot(dir, manifest);
+ let selected: string | undefined;
+ if (args.length) selected = resolveSnapshotPath(manifest, args.shift(), false, 'History path', true).path;
+ if (args.length)
+ throw new CsoError('INVALID_ARGUMENT', 'history accepts at most one path or opaque handle');
+ const status = readJson(join(dir, 'history-status.json'));
+ if (status.status !== 'captured' || !fs.existsSync(join(dir, 'history.txt')))
+ throw new CsoError('MISSING_INPUT', status.gap || 'Historical evidence was not retained');
+ const raw = fs.readFileSync(join(dir, 'history.txt'), 'utf8');
+ if (!selected) {
+ emit(raw);
+ return;
+ }
+ const displayPath = publicSnapshotPath(manifest, selected).displayPath ?? selected,
+ retained = historyForPath(raw, displayPath);
+ emit(retained ?? `No retained patch hunks for ${displayPath}`);
+}
+function resume(args: string[]) {
+ const { dir } = run(args);
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', 'resume takes one run ID');
+ return withLock(dir, () => {
+ const report = loadReport(dir),
+ manifest = readJson(join(dir, 'snapshot.json'));
+ if (report.status === 'finished')
+ throw new CsoError('INVALID_SCHEMA', 'A finished audit cannot be resumed');
+ assertSnapshot(dir, manifest);
+ for (const message of recoveryEvents(dir))
+ if (!report.events.some((e) => e.kind === 'watchdog-recovery' && e.message === message))
+ event(report, 'watchdog-recovery', message);
+ if (Date.now() >= Date.parse(report.deadline)) {
+ report.status = 'interrupted';
+ event(report, 'deadline', 'Original budget is exhausted; resume did not replenish it');
+ saveReport(dir, report);
+ throw new CsoError('DEADLINE', 'Original run budget is exhausted');
+ }
+ report.status = 'running';
+ event(report, 'resume', 'Continued retained snapshot under original policy');
+ saveReport(dir, report);
+ return {
+ runId: report.runId,
+ deadline: report.deadline,
+ policy: report.policy,
+ recovery: recoveryEvents(dir),
+ };
+ });
+}
+function importV2(args: string[]) {
+ if (args.length !== 1) throw new CsoError('INVALID_ARGUMENT', 'import-v2 requires one report file');
+ const legacy = sanitizeForJson(importLegacy(readInput(args[0]))) as ReturnType,
+ id = sha256(JSON.stringify(legacy)),
+ dir = secureDirectory(join(privateRoot(), 'legacy-imports'));
+ writeJson(join(dir, `${id}.json`), legacy);
+ return { id, path: `legacy-imports/${id}.json`, warning: legacy.warning, report: legacy };
+}
+function inspectV2(args: string[]) {
+ if (args.length !== 1 || !/^[a-f0-9]{64}$/.test(args[0] ?? ''))
+ throw new CsoError('INVALID_ARGUMENT', 'inspect-v2 requires the 64-character import ID');
+ const id = args[0],
+ file = join(secureDirectory(join(privateRoot(), 'legacy-imports')), `${id}.json`);
+ if (!fs.existsSync(file))
+ throw new CsoError('MISSING_INPUT', 'Legacy report import was not found or expired');
+ const report = readJson(file);
+ if (report?.schemaVersion !== 2 || report?.readOnly !== true || !Array.isArray(report?.findings))
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Stored legacy report is incompatible');
+ if (sha256(JSON.stringify(report)) !== id)
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Stored legacy report identity is inconsistent');
+ return { id, report };
+}
+async function scanner(args: string[], sarif = false) {
+ const { dir } = run(args);
+ if (
+ sarif
+ ? args.length !== 1
+ : args.length < 1 || args.length > 2 || !SCANNER_IDS.includes(args[0] as ScannerId)
+ )
+ throw new CsoError(
+ 'INVALID_ARGUMENT',
+ sarif
+ ? 'import-sarif needs one file'
+ : 'scan requires a supported scanner ID and optional request JSON',
+ );
+ const id = args[0] as ScannerId,
+ request = !sarif ? validateScannerRequest(args[1] ? readInput(args[1]) : {}, id) : undefined;
// A live writer owns the lock throughout the bounded scan. Finish, submit and
// concurrent imports cannot replace coverage or retire the run underneath it.
- return await withLock(dir,async()=>{
- const report=loadReport(dir);requireTime(report);if(report.status!=='running')throw new CsoError('INVALID_SCHEMA','Scanner evidence can only enter a running audit');
- let record:any;
- if(sarif){
- const file=callerPath(args[0]);
- try{
- const data=readBoundedStable(file,1024*1024,'SARIF file'),out=importSarif(data.toString('utf8'),{sourceRoot:'/source'});record={outcome:out,coverage:scannerCoverage(out,report.policy.scope),provenance:{kind:'untrusted SARIF import',sourceHash:sha256(data)}};
- }catch(error){if(error instanceof CsoError)throw error;throw new CsoError('MISSING_INPUT','SARIF file is missing or unreadable');}
- }else{
- event(report,`scanner-attempt:${id}`,'Started bounded scanner collection; completion requires an immutable outcome artifact');saveReport(dir,report);
- record=await executeScanner({id,runId:report.runId,runDir:dir,manifest:readJson(join(dir,'snapshot.json')),policy:report.policy,executionDeadline:Date.parse(report.deadline)-60_000,platform:platform(),request,watchdogPath:join(dirname(process.execPath),'gstack-cso-watchdog')});
+ return await withLock(dir, async () => {
+ const report = loadReport(dir);
+ requireTime(report);
+ if (report.status !== 'running')
+ throw new CsoError('INVALID_SCHEMA', 'Scanner evidence can only enter a running audit');
+ let record: any;
+ if (sarif) {
+ const file = callerPath(args[0]);
+ try {
+ const data = readBoundedStable(file, 1024 * 1024, 'SARIF file'),
+ out = importSarif(data.toString('utf8'), { sourceRoot: '/source' });
+ record = {
+ outcome: out,
+ coverage: scannerCoverage(out, report.policy.scope),
+ provenance: { kind: 'untrusted SARIF import', sourceHash: sha256(data) },
+ };
+ } catch (error) {
+ if (error instanceof CsoError) throw error;
+ throw new CsoError('MISSING_INPUT', 'SARIF file is missing or unreadable');
+ }
+ } else {
+ event(
+ report,
+ `scanner-attempt:${id}`,
+ 'Started bounded scanner collection; completion requires an immutable outcome artifact',
+ );
+ saveReport(dir, report);
+ record = await executeScanner({
+ id,
+ runId: report.runId,
+ runDir: dir,
+ manifest: readJson(join(dir, 'snapshot.json')),
+ policy: report.policy,
+ executionDeadline: Date.parse(report.deadline) - 60_000,
+ platform: platform(),
+ request,
+ watchdogPath: join(dirname(process.execPath), 'gstack-cso-watchdog'),
+ });
}
- const outcome=record.outcome,originalCount=outcome.candidates.length;
+ const outcome = record.outcome,
+ originalCount = outcome.candidates.length;
// Normalization can expand a 1 MiB scanner payload. Retain supported-size
// candidate evidence and disclose omissions instead of saving unreadable state.
- while(Buffer.byteLength(JSON.stringify(record,null,2))>950_000&&outcome.candidates.length)outcome.candidates.splice(Math.max(0,outcome.candidates.length-Math.max(1,Math.ceil(outcome.candidates.length/4))));
- if(outcome.candidates.length1024*1024)throw new CsoError('PERSISTENCE_FAILED','Scanner result exceeds the private artifact limit; no saved report is claimed');
- record=persistableArtifact(record,'Scanner outcome');const artifactId=`${outcome.tool}-${sha256(canonical(record)).slice(0,16)}-${randomBytes(8).toString('hex')}`,file=join(dir,'scanner-outcomes',`${artifactId}.json`);
- writeJsonExclusive(file,record);
+ while (Buffer.byteLength(JSON.stringify(record, null, 2)) > 950_000 && outcome.candidates.length)
+ outcome.candidates.splice(
+ Math.max(0, outcome.candidates.length - Math.max(1, Math.ceil(outcome.candidates.length / 4))),
+ );
+ if (outcome.candidates.length < originalCount) {
+ outcome.status = 'partial';
+ outcome.gaps.push({
+ code: 'OUTPUT_LIMIT',
+ message: `${originalCount - outcome.candidates.length} scanner candidates withheld to fit the bounded immutable artifact`,
+ });
+ record.coverage = scannerCoverage(outcome, report.policy.scope);
+ }
+ if (Buffer.byteLength(JSON.stringify(record, null, 2)) > 1024 * 1024)
+ throw new CsoError(
+ 'PERSISTENCE_FAILED',
+ 'Scanner result exceeds the private artifact limit; no saved report is claimed',
+ );
+ record = persistableArtifact(record, 'Scanner outcome');
+ const artifactId = `${outcome.tool}-${sha256(canonical(record)).slice(0, 16)}-${randomBytes(8).toString('hex')}`,
+ file = join(dir, 'scanner-outcomes', `${artifactId}.json`);
+ writeJsonExclusive(file, record);
record.coverage.evidence.push(`Immutable outcome: ${artifactId}`);
- report.coverage.push(record.coverage);event(report,'scanner-outcome',`${artifactId}: ${outcome.status}; ${outcome.candidates.length} candidates`);saveReport(dir,report);
- return {...outcome,artifactId,artifact:`scanner-outcomes/${artifactId}.json`,provenance:record.provenance};
+ report.coverage.push(record.coverage);
+ event(
+ report,
+ 'scanner-outcome',
+ `${artifactId}: ${outcome.status}; ${outcome.candidates.length} candidates`,
+ );
+ saveReport(dir, report);
+ return {
+ ...outcome,
+ artifactId,
+ artifact: `scanner-outcomes/${artifactId}.json`,
+ provenance: record.provenance,
+ };
});
}
-function scannerOutcome(args:string[]){const {dir}=run(args);if(args.length!==1||!/^[a-z0-9-]{1,40}-[a-f0-9]{16}-[a-f0-9]{16}$/.test(args[0]))throw new CsoError('INVALID_ARGUMENT','scanner-outcome requires one immutable scanner artifact ID');return readJson(join(dir,'scanner-outcomes',`${args[0]}.json`));}
-function recheckOriginalDirectory(repo:string,findingId:string,requestedRun?:string):{dir:string;runId:string}{
- const currentRepoId=repoId(repo),root=privateRoot(),repoDir=join(root,currentRepoId),runPattern=/^\d{13}-[a-f0-9]{16}$/;
- if(requestedRun){
- if(!runPattern.test(requestedRun))throw new CsoError('INVALID_ARGUMENT','Run identifier must be the ID returned by start');
- const dir=join(repoDir,requestedRun);if(!fs.existsSync(dir))throw new CsoError('MISSING_INPUT','Original run was not found for the current repository or has expired');
- return{dir:secureDirectory(dir),runId:requestedRun};
+function scannerOutcome(args: string[]) {
+ const { dir } = run(args);
+ if (args.length !== 1 || !/^[a-z0-9-]{1,40}-[a-f0-9]{16}-[a-f0-9]{16}$/.test(args[0]))
+ throw new CsoError('INVALID_ARGUMENT', 'scanner-outcome requires one immutable scanner artifact ID');
+ return readJson(join(dir, 'scanner-outcomes', `${args[0]}.json`));
+}
+function recheckOriginalDirectory(
+ repo: string,
+ findingId: string,
+ requestedRun?: string,
+): { dir: string; runId: string } {
+ const currentRepoId = repoId(repo),
+ root = privateRoot(),
+ repoDir = join(root, currentRepoId),
+ runPattern = /^\d{13}-[a-f0-9]{16}$/;
+ if (requestedRun) {
+ if (!runPattern.test(requestedRun))
+ throw new CsoError('INVALID_ARGUMENT', 'Run identifier must be the ID returned by start');
+ const dir = join(repoDir, requestedRun);
+ if (!fs.existsSync(dir))
+ throw new CsoError(
+ 'MISSING_INPUT',
+ 'Original run was not found for the current repository or has expired',
+ );
+ return { dir: secureDirectory(dir), runId: requestedRun };
}
- if(!fs.existsSync(repoDir))throw new CsoError('MISSING_INPUT','No finished original audit contains this finding in the current repository');
- const matches:{dir:string;runId:string}[]=[],directory=fs.opendirSync(secureDirectory(repoDir));let visited=0;
- try{let entry:fs.Dirent|null;while((entry=directory.readSync())!==null){
- if(++visited>REPLAY_LOOKUP_MAX_ENTRIES)throw new CsoError('INSUFFICIENT_CAPACITY',`Recheck lookup exceeded ${REPLAY_LOOKUP_MAX_ENTRIES} private state entries`);
- if(!entry.isDirectory()||!runPattern.test(entry.name))continue;const dir=join(repoDir,entry.name),reportPath=join(dir,'report.json');if(!fs.existsSync(reportPath))continue;const report=loadReport(dir);
- if(report.runId!==entry.name||report.repoId!==currentRepoId)throw new CsoError('INCOMPATIBLE_INPUT','Retained original audit identity does not match its repository state path');
- if(report.status==='finished'&&!report.parent&&report.findings.some(f=>f.id===findingId))matches.push({dir,runId:entry.name});
- }}finally{directory.closeSync();}
- if(!matches.length)throw new CsoError('MISSING_INPUT','No finished original audit contains this finding in the current repository');
- if(matches.length>1)throw new CsoError('INVALID_ARGUMENT',`Finding matches ${matches.length} finished original audits; use --run RUN to select one`);
+ if (!fs.existsSync(repoDir))
+ throw new CsoError(
+ 'MISSING_INPUT',
+ 'No finished original audit contains this finding in the current repository',
+ );
+ const matches: { dir: string; runId: string }[] = [],
+ directory = fs.opendirSync(secureDirectory(repoDir));
+ let visited = 0;
+ try {
+ let entry: fs.Dirent | null;
+ while ((entry = directory.readSync()) !== null) {
+ if (++visited > REPLAY_LOOKUP_MAX_ENTRIES)
+ throw new CsoError(
+ 'INSUFFICIENT_CAPACITY',
+ `Recheck lookup exceeded ${REPLAY_LOOKUP_MAX_ENTRIES} private state entries`,
+ );
+ if (!entry.isDirectory() || !runPattern.test(entry.name)) continue;
+ const dir = join(repoDir, entry.name),
+ reportPath = join(dir, 'report.json');
+ if (!fs.existsSync(reportPath)) continue;
+ const report = loadReport(dir);
+ if (report.runId !== entry.name || report.repoId !== currentRepoId)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Retained original audit identity does not match its repository state path',
+ );
+ if (report.status === 'finished' && !report.parent && report.findings.some((f) => f.id === findingId))
+ matches.push({ dir, runId: entry.name });
+ }
+ } finally {
+ directory.closeSync();
+ }
+ if (!matches.length)
+ throw new CsoError(
+ 'MISSING_INPUT',
+ 'No finished original audit contains this finding in the current repository',
+ );
+ if (matches.length > 1)
+ throw new CsoError(
+ 'INVALID_ARGUMENT',
+ `Finding matches ${matches.length} finished original audits; use --run RUN to select one`,
+ );
return matches[0];
}
-async function recheck(args:string[],dependencies:CsoCliDependencies){const startedAt=new Date();if(!args.length)throw new CsoError('INVALID_ARGUMENT','recheck requires a finding ID');const findingId=args.shift()!;if(!/^[a-f0-9]{32}$/.test(findingId))throw new CsoError('INVALID_ARGUMENT','Finding identifier must be the 32-character ID reported by CSO');const repo=callerPath(need(args,'--repo')),requestedRun=args.includes('--run')?need(args,'--run'):undefined;if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown recheck argument: ${args[0]}`);if(!fs.existsSync(repo)||!fs.statSync(repo).isDirectory())throw new CsoError('MISSING_INPUT','Repository directory does not exist');assertStateOutside(repo);retention(startedAt.getTime(),{deadlineMs:startedAt.getTime()+RETENTION_MAINTENANCE_MS,maxEntries:RETENTION_MAX_ENTRIES});const selected=recheckOriginalDirectory(repo,findingId,requestedRun),runId=selected.runId,originalDir=selected.dir;return await withLock(originalDir,async()=>{const original=loadReport(originalDir);if(original.runId!==runId||original.repoId!==repoId(repo))throw new CsoError('INCOMPATIBLE_INPUT','Retained original audit identity does not match its repository state path');const finding=original.findings.find(f=>f.id===findingId);if(!finding)throw new CsoError('MISSING_INPUT','Original finding does not exist');if(original.status!=='finished'||original.parent)throw new CsoError('INVALID_SCHEMA','Recheck requires a finished original audit');
+async function recheck(args: string[], dependencies: CsoCliDependencies) {
+ const startedAt = new Date();
+ if (!args.length) throw new CsoError('INVALID_ARGUMENT', 'recheck requires a finding ID');
+ const findingId = args.shift()!;
+ if (!/^[a-f0-9]{32}$/.test(findingId))
+ throw new CsoError('INVALID_ARGUMENT', 'Finding identifier must be the 32-character ID reported by CSO');
+ const repo = callerPath(need(args, '--repo')),
+ requestedRun = args.includes('--run') ? need(args, '--run') : undefined;
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown recheck argument: ${args[0]}`);
+ if (!fs.existsSync(repo) || !fs.statSync(repo).isDirectory())
+ throw new CsoError('MISSING_INPUT', 'Repository directory does not exist');
+ assertStateOutside(repo);
+ retention(startedAt.getTime(), {
+ deadlineMs: startedAt.getTime() + RETENTION_MAINTENANCE_MS,
+ maxEntries: RETENTION_MAX_ENTRIES,
+ });
+ const selected = recheckOriginalDirectory(repo, findingId, requestedRun),
+ runId = selected.runId,
+ originalDir = selected.dir;
+ return await withLock(originalDir, async () => {
+ const original = loadReport(originalDir);
+ if (original.runId !== runId || original.repoId !== repoId(repo))
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Retained original audit identity does not match its repository state path',
+ );
+ const finding = original.findings.find((f) => f.id === findingId);
+ if (!finding) throw new CsoError('MISSING_INPUT', 'Original finding does not exist');
+ if (original.status !== 'finished' || original.parent)
+ throw new CsoError('INVALID_SCHEMA', 'Recheck requires a finished original audit');
// Keep the original immutable while the fresh snapshot is captured and
// until its child lineage report has been durably published.
- const oldManifest=readJson(join(originalDir,'snapshot.json'));
- const preserveBase=original.policy.diff||Boolean(original.source.baseCommit),report=await start(['--repo',repo,...(original.policy.mode==='comprehensive'?['--comprehensive']:[]),...(original.policy.diff?['--diff']:[]),...(preserveBase?['--base',original.policy.base]:[]),'--budget',String(original.policy.budgetSeconds),...(original.policy.offline?['--offline']:[]),...(original.policy.scope==='default'?[]:original.policy.scope.startsWith('domain:')?['--scope',original.policy.scope.slice(7)]:[`--${original.policy.scope}`])],dependencies,{runId,findingId,kind:'recheck'},oldManifest.headCommit,startedAt);
- return {runId:report.runId,parent:report.parent};});}
-
-function recordReview(args:string[]){
- const {dir}=run(args);if(!args.length)throw new CsoError('INVALID_ARGUMENT','record-review requires a request JSON file and --producer ID');const raw=readInput(args.shift()!),producer=need(args,'--producer');if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown record-review argument: ${args[0]}`);const request=validateVerificationRequest(raw);
- return withLock(dir,()=>{const report=loadReport(dir);requireTime(report);if(report.policy.mode!=='comprehensive'||report.status!=='running'||!report.findings.some(f=>f.id===request.findingId&&f.evidence==='supported'))throw new CsoError('MISSING_INPUT','Review artifact must target a supported finding in a running comprehensive audit');const artifact=persistableArtifact(makeReviewArtifact(report.runId,request,string(producer,'producer identity',200)),'Repair review artifact');writeJsonExclusive(join(dir,'reviews',`${artifact.id}.json`),artifact);event(report,'repair-review',`Self-attested review artifact ${artifact.id} bound the proposed repair; reviewer independence is not host-verifiable`);saveReport(dir,report);return{reviewArtifactId:artifact.id,reviewAssurance:artifact.assurance,patchHash:artifact.patchHash,requestHash:artifact.requestHash};});
-}
-function publicPlanArgument(manifest:SnapshotManifest,arg:string,paths:string[]):string{
- for(const path of [...paths].sort((a,b)=>b.length-a.length)){const reference=publicSnapshotPath(manifest,path).path;if(reference===path)continue;if(arg===path)return reference;if(arg===`./${path}`)return `./${reference}`;}
- return arg;
-}
-function publicTestPlan(manifest:SnapshotManifest,plan:ReturnType){return{...plan,commands:plan.commands.map(command=>({...command,args:command.args.map(arg=>publicPlanArgument(manifest,arg,plan.files))})),files:plan.files.map(path=>publicSnapshotPath(manifest,path).path)};}
-function publicStartPlan(manifest:SnapshotManifest,plan:ReturnType){return{...plan,command:{...plan.command,args:plan.command.args.map(arg=>publicPlanArgument(manifest,arg,plan.entrypointFiles))},entrypointFiles:plan.entrypointFiles.map(path=>publicSnapshotPath(manifest,path).path)};}
-function testPlan(args:string[]){const {dir}=run(args);if(args.length!==1||!['node','bun','python','rails'].includes(args[0]))throw new CsoError('INVALID_ARGUMENT','test-plan requires one supported stack');const stack=args[0] as 'node'|'bun'|'python'|'rails',manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest;assertSnapshot(dir,manifest);const preparation=inspectPreparation(join(dir,'snapshot'),stack);if(preparation.status!=='ready')throw new CsoError('PREREQUISITE',preparation.prerequisites.map(item=>item.message).join('; ')||`${stack} preparation metadata is incomplete`);return{stack,runtimeProfile:preparation.runtimeProfile,...publicTestPlan(manifest,canonicalTestPlan(join(dir,'snapshot'),stack))};}
-function runtimePlan(args:string[]){const {dir}=run(args);if(!args.length||!['node','bun','python','rails'].includes(args[0]))throw new CsoError('INVALID_ARGUMENT','runtime-plan requires one supported stack and --port PORT');const stack=args.shift() as 'node'|'bun'|'python'|'rails',rawPort=need(args,'--port');if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown runtime-plan argument: ${args[0]}`);const port=Number(rawPort);if(!Number.isInteger(port)||port<1024||port>65535)throw new CsoError('INVALID_ARGUMENT','--port must be an integer from 1024 to 65535');const manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest;assertSnapshot(dir,manifest);const preparation=inspectPreparation(join(dir,'snapshot'),stack);if(preparation.status!=='ready')throw new CsoError('PREREQUISITE',preparation.prerequisites.map(item=>item.message).join('; ')||`${stack} preparation metadata is incomplete`);return{stack,runtimeProfile:preparation.runtimeProfile,start:publicStartPlan(manifest,canonicalStartPlan(join(dir,'snapshot'),stack,port)),tests:publicTestPlan(manifest,canonicalTestPlan(join(dir,'snapshot'),stack))};}
-
-function platform(): 'linux/amd64'|'linux/arm64'{if(!['linux','darwin'].includes(process.platform)||!['x64','arm64'].includes(process.arch))throw new CsoError('PREREQUISITE','Contained target execution requires a Linux or macOS host with amd64/arm64 Linux Docker images');return process.arch==='arm64'?'linux/arm64':'linux/amd64';}
-function watchdog():string{const p=join(dirname(process.execPath),process.platform==='win32'?'gstack-cso-watchdog.exe':'gstack-cso-watchdog');if(!fs.existsSync(p))throw new CsoError('ISOLATION_FAILED','Trusted detached watchdog is missing');return p;}
-
-function closureStateFile(dir:string,findingId:string,phase:'before'|'after',plan:unknown):string{
- return join(dir,'dependency-closures',`${findingId}-${phase}-${sha256(canonical(plan)).slice(0,16)}.json`);
-}
-function retainClosure(path:string,closure:DependencyClosure):void{
- if(fs.existsSync(path)){if(canonical(readJson(path))!==canonical(closure))throw new CsoError('INCOMPATIBLE_INPUT','Retained dependency closure conflicts with this preparation plan');return;}
- writeJsonExclusive(path,persistableArtifact(closure,'Dependency closure'));
-}
-function bindArchiveHashes(target:string[],closures:{before:DependencyClosure;after:DependencyClosure}):void{
- const hashes=[...new Set([...closures.before.archives,...closures.after.archives].map(archive=>archive.sha256))].sort();target.splice(0,target.length,...hashes);
-}
-function preparedVerificationExecutor(options:{dir:string;findingId:string;runtimeProfile:string;stack:'node'|'bun'|'python'|'rails';targetPlatform:'linux/amd64'|'linux/arm64';deadline:number;offline:boolean;runtimeCatalog:RuntimeCatalog;preparation:PreparationExecutor;delegate:VerificationExecutor;beforePlan:ReturnType;beforeAdmission:ReturnType;beforeClosure:DependencyClosure;closures:{before:DependencyClosure;after:DependencyClosure};archiveHashes:string[];proofs:{before:PreparationProof;after:PreparationProof};replay?:{before:DependencyClosure;after:DependencyClosure};persistClosures:boolean;}):VerificationExecutor{
- let beforeProjectToolchainHash:string|undefined;
- return {observe:async(source,phase,request,runtime,verifier,work,control,_execution,testEvidence,witness)=>{
- const plan=phase==='before'?options.beforePlan:inspectPreparation(source,options.stack);
- if(plan.status!=='ready')throw new CsoError('PREREQUISITE',plan.prerequisites.map(item=>item.message).join('; ')||`${options.stack} dependency metadata is not ready`);
- const admission=phase==='before'?options.beforeAdmission:admitPreparationRuntime({plan,platform:options.targetPlatform,profile:options.runtimeProfile,catalog:options.runtimeCatalog});
- if(admission.runtime.id!==runtime.id||admission.runtime.image!==runtime.image)throw new CsoError('INCOMPATIBLE_INPUT','Prepared verification runtime changed between source phases');
- const state=closureStateFile(options.dir,options.findingId,phase,plan),supplied=options.replay?.[phase],retained=!supplied&&fs.existsSync(state)?readJson(state) as DependencyClosure:undefined;
- const closure=phase==='before'?(supplied??options.beforeClosure):await options.preparation.acquire({plan,admission,snapshot:source,deadline:options.deadline,offline:options.offline||Boolean(supplied),existingClosure:supplied??retained});
- options.closures[phase]=closure;
- bindArchiveHashes(options.archiveHashes,options.closures);
- if(options.persistClosures)retainClosure(state,closure);
- let database:RailsDatabaseSelection|undefined;
- if(options.stack==='rails'){
- const selected=plan.database?.selected;
- if(!selected)throw new CsoError('PREREQUISITE','Rails automatic verification could not select one locked database adapter from static test configuration');
- database=selected==='postgresql'?{adapter:'postgresql',sidecar:admitPreparationSidecar({platform:options.targetPlatform,catalog:options.runtimeCatalog})}:{adapter:'sqlite'};
- }
- const prepared=await options.preparation.prepareOffline({plan,admission,snapshot:source,closure,deadline:options.deadline,database});
- try{
- options.proofs[phase]={schemaVersion:1,dependencyClosureHash:prepared.dependencyClosureHash,configurationHash:prepared.configurationHash,
- sourceProjectionHash:prepared.sourceProjectionHash,preparedManifestHash:prepared.preparedManifestHash,preparedDependencyHash:prepared.preparedDependencyHash,receiptHash:prepared.receiptHash,
- executionEnvironmentHash:sha256(canonical(prepared.executionEnvironment)),databaseHash:prepared.databaseHash,
- transformations:prepared.transformations};
- const sourceTests=canonicalTestPlan(source,options.stack),preparedTests=canonicalTestPlan(prepared.preparedRoot,options.stack),
- sourceStart=canonicalStartPlan(source,options.stack,request.port),preparedStart=canonicalStartPlan(prepared.preparedRoot,options.stack,request.port);
- if(sourceTests.signature!==preparedTests.signature||sourceStart.signature!==preparedStart.signature||verificationHarnessHash(request,source)!==verificationHarnessHash(request,prepared.preparedRoot))throw new CsoError('ISOLATION_FAILED','Offline lifecycle execution changed the canonical start, test, or harness inputs');
- if(sourceTests.toolchain==='project'){
- if(phase==='before')beforeProjectToolchainHash=prepared.preparedDependencyHash;
- else if(!beforeProjectToolchainHash||prepared.preparedDependencyHash!==beforeProjectToolchainHash)throw new CsoError('ASSERTION_FAILED','Offline preparation changed the project-installed test toolchain between source phases');
- }
- const protectedPaths=new Set([...request.boundaryFiles,...request.testFiles,...sourceStart.entrypointFiles,...request.changes.map(item=>item.path)]);
- if(prepared.transformations.some(item=>protectedPaths.has(item.path)))throw new CsoError('ISOLATION_FAILED','Synthetic preparation transformation overlaps a security boundary, startup input, or test input');
- return await options.delegate.observe(prepared.preparedRoot,phase,request,runtime,verifier,work,control,
- {environment:prepared.executionEnvironment,database:prepared.database},testEvidence,witness);
- }
- finally{await options.preparation.dispose(prepared);}
- }};
-}
-async function verify(args:string[],dependencies:CsoCliDependencies){const {dir}=run(args);if(args.length!==1)throw new CsoError('INVALID_ARGUMENT','verify requires one request JSON file');const raw=readInput(args[0]),request=validateVerificationRequest(raw);
- return await withLock(dir,async()=>{const report=loadReport(dir),manifest=readJson(join(dir,'snapshot.json')) as SnapshotManifest;assertSnapshot(dir,manifest);requireTime(report);if(report.policy.mode!=='comprehensive'||report.status!=='running')throw new CsoError('INVALID_SCHEMA','Only a running comprehensive audit can request target execution');const finding=report.findings.find(f=>f.id===request.findingId&&f.evidence==='supported');if(!finding)throw new CsoError('MISSING_INPUT','Verification must target a supported finding in this run');
- if(!request.review.artifactId)throw new CsoError('MISSING_INPUT','Verification requires a separately persisted independent repair-review artifact');const reviewArtifact=validateReviewArtifact(readJson(join(dir,'reviews',`${request.review.artifactId}.json`)),report.runId,request);
- const findingPath=resolveSnapshotPath(manifest,finding.location.path,true,'Finding path').path;if(!request.boundaryFiles.some(path=>resolveSnapshotPath(manifest,path,true,'Boundary path').path===findingPath))throw new CsoError('INVALID_SCHEMA','Boundary files must include the finding location');
- const attempts=report.events.filter(e=>e.kind===`verification-attempt:${finding.id}`).length;if(attempts>=3)throw new CsoError('DEADLINE','Three bounded harness/repair attempts have already been used for this finding');if(report.findings.filter(f=>['runtime_tested','tested'].includes(f.repair)).length>=3)throw new CsoError('INSUFFICIENT_CAPACITY','This run already produced three runtime-tested repairs');
- const targetPlatform=platform();let runtime;try{runtime=selectRuntime(request.runtimeProfile,targetPlatform,dependencies.runtimeCatalog);}catch(error:any){throw new CsoError('PREREQUISITE',error?.message||'Qualified runtime is unavailable');}const verifier=runtime;
- if(!['node','bun','python','rails'].includes(runtime.stack))throw new CsoError('INCOMPATIBLE_INPUT','Application verification requires an application runtime profile');const plan=inspectPreparation(join(dir,'snapshot'),runtime.stack as any);if(plan.status!=='ready')throw new CsoError('PREREQUISITE',plan.prerequisites.map(p=>p.message).join('; ')||'Runtime preparation metadata is incomplete');assertRuntimeCompatible(plan,runtime);writeJson(join(dir,`preparation-${runtime.stack}.json`),plan);
- const endpoint=await dockerEndpoint(secureDirectory(join(dir,'home'))),watchdogPath=dependencies.watchdogPath(),attemptDeadline=Math.min(Date.now()+300_000,Date.parse(report.deadline)-60_000);
- const admission=admitPreparationRuntime({plan,platform:targetPlatform,profile:runtime.id,catalog:dependencies.runtimeCatalog}),staging=secureDirectory(join(dir,'archive-staging')),
- runner=new DockerPreparationSandboxRunner({endpoint,watchdogPath,runRoot:dir,controlRoot:secureDirectory(join(dir,'preparation-execution')),admission}),
- preparation=new PreparationExecutor({cache:new PublicArchiveCache({root:publicArchiveCacheRoot(),stagingRoot:staging}),runner,materializationRoot:secureDirectory(join(dir,'archive-materializations'))});
- const beforeState=closureStateFile(dir,finding.id,'before',plan),retainedBefore=fs.existsSync(beforeState)?readJson(beforeState) as DependencyClosure:undefined;
- const beforeClosure=await preparation.acquire({plan,admission,snapshot:join(dir,'snapshot'),deadline:attemptDeadline,offline:report.policy.offline,existingClosure:retainedBefore});retainClosure(beforeState,beforeClosure);
- const closures={before:beforeClosure,after:beforeClosure},archiveHashes=[...new Set(beforeClosure.archives.map(archive=>archive.sha256))].sort(),proofs={} as {before:PreparationProof;after:PreparationProof},delegate=new DockerVerificationExecutor(endpoint,watchdogPath,attemptDeadline,()=>{event(report,`verification-attempt:${finding.id}`,`Started bounded repair verification attempt ${attempts+1}`);saveReport(dir,report);});
- const executor=preparedVerificationExecutor({dir,findingId:finding.id,runtimeProfile:runtime.id,stack:runtime.stack as 'node'|'bun'|'python'|'rails',targetPlatform,deadline:attemptDeadline,offline:report.policy.offline,runtimeCatalog:dependencies.runtimeCatalog,preparation,delegate,beforePlan:plan,beforeAdmission:admission,beforeClosure,closures,archiveHashes,proofs,persistClosures:true});
- let result:Awaited>;try{result=await verifyRepair({runId:report.runId,runDir:dir,manifest,rawRequest:raw,runtime,verifier,policyHash:ISOLATION_POLICY_HASH,auditPolicyHash:sha256(canonical(report.policy)),archives:archiveHashes,dependencyClosures:closures,preparation:proofs,reviewArtifact,executor,watchdogPath,attemptDeadline});}catch(error){if(error instanceof VerificationAttemptError){finding.reproduction=error.attempt.reproduction;finding.repair=error.attempt.repair;finding.reproductionAttemptId=error.attempt.id;event(report,error.attempt.repair==='proposed'?'repair-candidate':'verification-failed',error.attempt.repair==='proposed'?`${error.attempt.id}: external repair assertions passed; helper-authenticated external assertion witness required before certification`:`${error.attempt.id}: ${error.attempt.reproduction}; repair validation failed without issuing a bundle`);saveReport(dir,report);}throw error;}
- finding.reproduction=result.manifest.before.security==='intended_failure'&&result.manifest.before.booted&&result.manifest.before.legitimate?'reproduced':result.manifest.before.security==='pass'?'disproved':result.manifest.before.booted?'inconclusive':'blocked';
- finding.repair=result.manifest.result==='tested'?'tested':result.manifest.result==='runtime_tested'?'runtime_tested':'failed';if(['tested','runtime_tested'].includes(result.manifest.result)){finding.verificationId=result.manifest.id;finding.verificationAssurance={assertions:result.manifest.assertionAssurance!,testCompletion:result.manifest.testCompletionAssurance,review:result.manifest.reviewAssurance};delete finding.reproductionAttemptId;}event(report,'verification',`${result.manifest.id}: ${result.manifest.result}; assertion assurance ${result.manifest.assertionAssurance}; test completion assurance ${result.manifest.testCompletionAssurance}; review assurance ${result.manifest.reviewAssurance}`);saveReport(dir,report);return{result:result.manifest.result,verification:result.manifest,bundle:`bundles/${result.bundle.id}.json`};});}
-function replayBundle(stored:{path:string;dir:string},id:string):any{const bundle=validateRepairBundle(readJson(stored.path),id),report=loadReport(stored.dir),finding=report.findings.find(f=>f.verificationId===id&&['tested','runtime_tested'].includes(f.repair));if(!['tested','runtime_tested'].includes(bundle.verification.result)||!finding)throw new CsoError('INCOMPATIBLE_INPUT','Only a helper-recorded runtime-tested or host-reviewed repair bundle can be replayed');return bundle;}
-async function withReplayBundle(id:string,deadline:number,fn:(stored:{path:string;dir:string},bundle:any)=>Promise):Promise{
- if(!/^[a-f0-9]{32}$/.test(id))throw new CsoError('INVALID_ARGUMENT','Bundle identifier must be the 32-character ID returned by verify');
- let visited=0,stored:{path:string;dir:string}|undefined;const admit=()=>{if(Date.now()>=deadline)throw new CsoError('DEADLINE','Replay exhausted its five-minute budget while locating the recorded bundle');if(++visited>REPLAY_LOOKUP_MAX_ENTRIES)throw new CsoError('INSUFFICIENT_CAPACITY',`Replay bundle lookup exceeded ${REPLAY_LOOKUP_MAX_ENTRIES} private state entries`);};
- const root=privateRoot(),repos=fs.opendirSync(root);try{let repo:fs.Dirent|null;search:while((repo=repos.readSync())!==null){admit();if(!repo.isDirectory()||!/^[a-f0-9]{24}$/.test(repo.name))continue;const repoDir=join(root,repo.name),runs=fs.opendirSync(repoDir);try{let run:fs.Dirent|null;while((run=runs.readSync())!==null){admit();if(!run.isDirectory()||!/^\d{13}-[a-f0-9]{16}$/.test(run.name))continue;const candidate={dir:join(repoDir,run.name),path:join(repoDir,run.name,'bundles',`${id}.json`)};if(fs.existsSync(candidate.path)){stored=candidate;break search;}}}finally{runs.closeSync();}}}finally{repos.closeSync();}
- if(!stored)throw new CsoError('MISSING_INPUT','Repair bundle was not found or expired');let matched=false,value!:T;await withLock(stored.dir,async()=>{if(!fs.existsSync(stored!.path))return;const bundle=replayBundle(stored!,id);matched=true;value=await fn(stored!,bundle);});if(matched)return value;throw new CsoError('MISSING_INPUT','Repair bundle was not found or expired');
-}
-function replayManifestValue(manifest:any):unknown{const {id:_,createdAt:__,witnessHash:___,before,after,...stable}=manifest,observation=(value:any)=>{const{output:_,...rest}=value;return rest;};return{...stable,before:observation(before),after:observation(after)};}
-async function replay(args:string[],dependencies:CsoCliDependencies){const replayStarted=Date.now(),replayDeadline=replayStarted+300_000;retention(replayStarted,{deadlineMs:replayStarted+RETENTION_MAINTENANCE_MS,maxEntries:RETENTION_MAX_ENTRIES});if(!args.length)throw new CsoError('INVALID_ARGUMENT','replay requires a bundle ID');const id=args.shift()!,source=args.includes('--source')?callerPath(need(args,'--source')):undefined;if(args.length)throw new CsoError('INVALID_ARGUMENT',`Unknown replay argument: ${args[0]}`);return await withReplayBundle(id,replayDeadline,async(stored,bundle)=>{
- if(bundle.requiredInputs.archives?.length&&!bundle.requiredInputs.dependencyClosures)throw new CsoError('MISSING_INPUT','Replay bundle predates retained dependency closures and cannot substitute current dependency state');
- let workDir=stored.dir,manifest:any,temporary:string|undefined;const retained=join(stored.dir,'snapshot');
- try{
- const captureSupplied=async()=>{if(!source)throw new CsoError('MISSING_INPUT','Retained source expired; supply explicitly matching source');const temp=newRun(source);temporary=temp.dir;workDir=temp.dir;manifest=await capture(source,workDir,undefined,undefined,{deadlineMs:replayDeadline});if(manifest.executionHash!==bundle.requiredInputs.sourceHash||manifest.originalHash!==bundle.requiredInputs.originalHash)throw new CsoError('INCOMPATIBLE_INPUT','Supplied source does not match the bundle input hashes');};
- if(fs.existsSync(retained)){manifest=readJson(join(stored.dir,'snapshot.json'));const expiresAt=typeof manifest?.expiresAt==='string'?Date.parse(manifest.expiresAt):Number.NaN;if(!Number.isFinite(expiresAt)||new Date(expiresAt).toISOString()!==manifest.expiresAt)throw new CsoError('INCOMPATIBLE_INPUT','Retained snapshot expiry is invalid');if(expiresAt<=Date.now())await captureSupplied();else assertSnapshot(stored.dir,manifest);}else await captureSupplied();
- if(manifest.executionHash!==bundle.requiredInputs.sourceHash||manifest.originalHash!==bundle.requiredInputs.originalHash)throw new CsoError('INCOMPATIBLE_INPUT','Retained source hashes do not match the bundle');validateRepairBundle(bundle,id,join(workDir,'snapshot'),manifest);if(bundle.verification.policyHash!==ISOLATION_POLICY_HASH)throw new CsoError('INCOMPATIBLE_INPUT','Current helper isolation policy does not match the recorded bundle');const targetPlatform=bundle.requiredInputs.platform as 'linux/amd64'|'linux/arm64';let runtime;try{runtime=selectRuntime(bundle.verification.runtime.profile,targetPlatform,dependencies.runtimeCatalog);}catch(error:any){throw new CsoError('PREREQUISITE',error?.message||'Qualified replay runtime is unavailable');}const verifier=runtime;if(runtime.image!==bundle.requiredInputs.runtimeImage)throw new CsoError('INCOMPATIBLE_INPUT','Qualified runtime digest does not match the bundle');
- const endpoint=await dockerEndpoint(secureDirectory(join(workDir,'home'))),watchdogPath=dependencies.watchdogPath(),attemptDeadline=replayDeadline,delegate=new DockerVerificationExecutor(endpoint,watchdogPath,attemptDeadline);let executor:VerificationExecutor=delegate,archives:string[]=[],dependencyClosures:{before:DependencyClosure;after:DependencyClosure}|undefined,proofs:{before:PreparationProof;after:PreparationProof}|undefined;
- if(bundle.requiredInputs.dependencyClosures){if(!['node','bun','python','rails'].includes(runtime.stack))throw new CsoError('INCOMPATIBLE_INPUT','Replay dependency closure requires an application runtime');const replayClosures=bundle.requiredInputs.dependencyClosures as {before:DependencyClosure;after:DependencyClosure},plan=inspectPreparation(join(workDir,'snapshot'),runtime.stack as any);if(plan.status!=='ready')throw new CsoError('PREREQUISITE',plan.prerequisites.map((item:any)=>item.message).join('; ')||'Replay dependency metadata is not ready');const admission=admitPreparationRuntime({plan,platform:targetPlatform,profile:runtime.id,catalog:dependencies.runtimeCatalog}),runner=new DockerPreparationSandboxRunner({endpoint,watchdogPath,runRoot:workDir,controlRoot:secureDirectory(join(workDir,'preparation-execution')),admission}),preparation=new PreparationExecutor({cache:new PublicArchiveCache({root:publicArchiveCacheRoot(),stagingRoot:secureDirectory(join(workDir,'archive-staging'))}),runner,materializationRoot:secureDirectory(join(workDir,'archive-materializations'))}),beforeClosure=await preparation.acquire({plan,admission,snapshot:join(workDir,'snapshot'),deadline:attemptDeadline,offline:true,existingClosure:replayClosures.before});dependencyClosures={before:beforeClosure,after:replayClosures.after};archives=[...new Set([...beforeClosure.archives,...replayClosures.after.archives].map(archive=>archive.sha256))].sort();proofs={} as {before:PreparationProof;after:PreparationProof};executor=preparedVerificationExecutor({dir:workDir,findingId:bundle.request.findingId,runtimeProfile:runtime.id,stack:runtime.stack as any,targetPlatform,deadline:attemptDeadline,offline:true,runtimeCatalog:dependencies.runtimeCatalog,preparation,delegate,beforePlan:plan,beforeAdmission:admission,beforeClosure,closures:dependencyClosures,archiveHashes:archives,proofs,replay:replayClosures,persistClosures:false});}
- const result=await verifyRepair({runId:bundle.runId,runDir:workDir,manifest,rawRequest:bundle.request,runtime,verifier,policyHash:ISOLATION_POLICY_HASH,auditPolicyHash:bundle.verification.auditPolicyHash,archives,dependencyClosures,preparation:proofs,reviewArtifact:bundle.reviewArtifact,executor,persist:false,watchdogPath,attemptDeadline});if(canonical(replayManifestValue(result.manifest))!==canonical(replayManifestValue(bundle.verification))||assertionWitnessReplayHash(result.bundle.witness!)!==assertionWitnessReplayHash(bundle.witness))throw new CsoError('INCOMPATIBLE_INPUT','Replay changed verification outcomes, preparation, or provenance inputs');const replayId=`${Date.now()}-${randomBytes(8).toString('hex')}`;writeJsonExclusive(join(stored.dir,'replays',`${replayId}.json`),{replayId,bundleId:id,verification:result.manifest,witness:result.bundle.witness});return{bundle:id,result:result.manifest.result,replay:result.manifest,replayId};
- }finally{if(temporary)finalizeReplayTemporary(temporary);}
+ const oldManifest = readJson(join(originalDir, 'snapshot.json'));
+ const preserveBase = original.policy.diff || Boolean(original.source.baseCommit),
+ report = await start(
+ [
+ '--repo',
+ repo,
+ ...(original.policy.mode === 'comprehensive' ? ['--comprehensive'] : []),
+ ...(original.policy.diff ? ['--diff'] : []),
+ ...(preserveBase ? ['--base', original.policy.base] : []),
+ '--budget',
+ String(original.policy.budgetSeconds),
+ ...(original.policy.offline ? ['--offline'] : []),
+ ...(original.policy.scope === 'default'
+ ? []
+ : original.policy.scope.startsWith('domain:')
+ ? ['--scope', original.policy.scope.slice(7)]
+ : [`--${original.policy.scope}`]),
+ ],
+ dependencies,
+ { runId, findingId, kind: 'recheck' },
+ oldManifest.headCommit,
+ startedAt,
+ );
+ return { runId: report.runId, parent: report.parent };
});
}
-export async function dispatchCsoCommand(command:string,args:string[],dependencies:CsoCliDependencies):Promise{
- args=[...args];if(!['start','recheck','replay','doctor','provision-images'].includes(command)){const started=Date.now();retention(started,{deadlineMs:started+RETENTION_MAINTENANCE_MS,maxEntries:RETENTION_MAX_ENTRIES});}
- let result:unknown;
- switch(command){case'start':result=await start(args,dependencies);break;case'doctor':result=await doctor(args,dependencies);break;case'provision-images':result=await provisionImages(args,dependencies);break;case'resume':result=resume(args);break;case'inspect':await inspect(args);return;case'read':await read(args);return;case'history':await history(args);return;case'submit':result=submit(args);break;case'finish':result=finish(args);break;case'import-v2':result=importV2(args);break;case'inspect-v2':result=inspectV2(args);break;case'scan':result=await scanner(args);break;case'scanner-outcome':result=scannerOutcome(args);break;case'import-sarif':result=await scanner(args,true);break;case'record-review':result=recordReview(args);break;case'test-plan':result=testPlan(args);break;case'runtime-plan':result=runtimePlan(args);break;case'recheck':result=await recheck(args,dependencies);break;
- case'verify':result=await verify(args,dependencies);break;case'replay':result=await replay(args,dependencies);break;case'patch-hash':if(args.length!==1)throw new CsoError('INVALID_ARGUMENT','patch-hash requires one request JSON file');result={patchHash:patchHash(validateVerificationRequest(readInput(args[0])))};break;default:throw new CsoError('INVALID_ARGUMENT',`Unknown command: ${command}`);}
+function recordReview(args: string[]) {
+ const { dir } = run(args);
+ if (!args.length)
+ throw new CsoError('INVALID_ARGUMENT', 'record-review requires a request JSON file and --producer ID');
+ const raw = readInput(args.shift()!),
+ producer = need(args, '--producer');
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown record-review argument: ${args[0]}`);
+ const request = validateVerificationRequest(raw);
+ return withLock(dir, () => {
+ const report = loadReport(dir);
+ requireTime(report);
+ if (
+ report.policy.mode !== 'comprehensive' ||
+ report.status !== 'running' ||
+ !report.findings.some((f) => f.id === request.findingId && f.evidence === 'supported')
+ )
+ throw new CsoError(
+ 'MISSING_INPUT',
+ 'Review artifact must target a supported finding in a running comprehensive audit',
+ );
+ const artifact = persistableArtifact(
+ makeReviewArtifact(report.runId, request, string(producer, 'producer identity', 200)),
+ 'Repair review artifact',
+ );
+ writeJsonExclusive(join(dir, 'reviews', `${artifact.id}.json`), artifact);
+ event(
+ report,
+ 'repair-review',
+ `Self-attested review artifact ${artifact.id} bound the proposed repair; reviewer independence is not host-verifiable`,
+ );
+ saveReport(dir, report);
+ return {
+ reviewArtifactId: artifact.id,
+ reviewAssurance: artifact.assurance,
+ patchHash: artifact.patchHash,
+ requestHash: artifact.requestHash,
+ };
+ });
+}
+function publicPlanArgument(manifest: SnapshotManifest, arg: string, paths: string[]): string {
+ for (const path of [...paths].sort((a, b) => b.length - a.length)) {
+ const reference = publicSnapshotPath(manifest, path).path;
+ if (reference === path) continue;
+ if (arg === path) return reference;
+ if (arg === `./${path}`) return `./${reference}`;
+ }
+ return arg;
+}
+function publicTestPlan(manifest: SnapshotManifest, plan: ReturnType) {
+ return {
+ ...plan,
+ commands: plan.commands.map((command) => ({
+ ...command,
+ args: command.args.map((arg) => publicPlanArgument(manifest, arg, plan.files)),
+ })),
+ files: plan.files.map((path) => publicSnapshotPath(manifest, path).path),
+ };
+}
+function publicStartPlan(manifest: SnapshotManifest, plan: ReturnType) {
+ return {
+ ...plan,
+ command: {
+ ...plan.command,
+ args: plan.command.args.map((arg) => publicPlanArgument(manifest, arg, plan.entrypointFiles)),
+ },
+ entrypointFiles: plan.entrypointFiles.map((path) => publicSnapshotPath(manifest, path).path),
+ };
+}
+function testPlan(args: string[]) {
+ const { dir } = run(args);
+ if (args.length !== 1 || !['node', 'bun', 'python', 'rails'].includes(args[0]))
+ throw new CsoError('INVALID_ARGUMENT', 'test-plan requires one supported stack');
+ const stack = args[0] as 'node' | 'bun' | 'python' | 'rails',
+ manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest;
+ assertSnapshot(dir, manifest);
+ const preparation = inspectPreparation(join(dir, 'snapshot'), stack);
+ if (preparation.status !== 'ready')
+ throw new CsoError(
+ 'PREREQUISITE',
+ preparation.prerequisites.map((item) => item.message).join('; ') ||
+ `${stack} preparation metadata is incomplete`,
+ );
+ return {
+ stack,
+ runtimeProfile: preparation.runtimeProfile,
+ ...publicTestPlan(manifest, canonicalTestPlan(join(dir, 'snapshot'), stack)),
+ };
+}
+function runtimePlan(args: string[]) {
+ const { dir } = run(args);
+ if (!args.length || !['node', 'bun', 'python', 'rails'].includes(args[0]))
+ throw new CsoError('INVALID_ARGUMENT', 'runtime-plan requires one supported stack and --port PORT');
+ const stack = args.shift() as 'node' | 'bun' | 'python' | 'rails',
+ rawPort = need(args, '--port');
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown runtime-plan argument: ${args[0]}`);
+ const port = Number(rawPort);
+ if (!Number.isInteger(port) || port < 1024 || port > 65535)
+ throw new CsoError('INVALID_ARGUMENT', '--port must be an integer from 1024 to 65535');
+ const manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest;
+ assertSnapshot(dir, manifest);
+ const preparation = inspectPreparation(join(dir, 'snapshot'), stack);
+ if (preparation.status !== 'ready')
+ throw new CsoError(
+ 'PREREQUISITE',
+ preparation.prerequisites.map((item) => item.message).join('; ') ||
+ `${stack} preparation metadata is incomplete`,
+ );
+ return {
+ stack,
+ runtimeProfile: preparation.runtimeProfile,
+ start: publicStartPlan(manifest, canonicalStartPlan(join(dir, 'snapshot'), stack, port)),
+ tests: publicTestPlan(manifest, canonicalTestPlan(join(dir, 'snapshot'), stack)),
+ };
+}
+
+function platform(): 'linux/amd64' | 'linux/arm64' {
+ if (!['linux', 'darwin'].includes(process.platform) || !['x64', 'arm64'].includes(process.arch))
+ throw new CsoError(
+ 'PREREQUISITE',
+ 'Contained target execution requires a Linux or macOS host with amd64/arm64 Linux Docker images',
+ );
+ return process.arch === 'arm64' ? 'linux/arm64' : 'linux/amd64';
+}
+function watchdog(): string {
+ const p = join(
+ dirname(process.execPath),
+ process.platform === 'win32' ? 'gstack-cso-watchdog.exe' : 'gstack-cso-watchdog',
+ );
+ if (!fs.existsSync(p)) throw new CsoError('ISOLATION_FAILED', 'Trusted detached watchdog is missing');
+ return p;
+}
+
+function closureStateFile(dir: string, findingId: string, phase: 'before' | 'after', plan: unknown): string {
+ return join(
+ dir,
+ 'dependency-closures',
+ `${findingId}-${phase}-${sha256(canonical(plan)).slice(0, 16)}.json`,
+ );
+}
+function retainClosure(path: string, closure: DependencyClosure): void {
+ if (fs.existsSync(path)) {
+ if (canonical(readJson(path)) !== canonical(closure))
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Retained dependency closure conflicts with this preparation plan',
+ );
+ return;
+ }
+ writeJsonExclusive(path, persistableArtifact(closure, 'Dependency closure'));
+}
+function bindArchiveHashes(
+ target: string[],
+ closures: { before: DependencyClosure; after: DependencyClosure },
+): void {
+ const hashes = [
+ ...new Set([...closures.before.archives, ...closures.after.archives].map((archive) => archive.sha256)),
+ ].sort();
+ target.splice(0, target.length, ...hashes);
+}
+function preparedVerificationExecutor(options: {
+ dir: string;
+ findingId: string;
+ runtimeProfile: string;
+ stack: 'node' | 'bun' | 'python' | 'rails';
+ targetPlatform: 'linux/amd64' | 'linux/arm64';
+ deadline: number;
+ offline: boolean;
+ runtimeCatalog: RuntimeCatalog;
+ preparation: PreparationExecutor;
+ delegate: VerificationExecutor;
+ beforePlan: ReturnType;
+ beforeAdmission: ReturnType;
+ beforeClosure: DependencyClosure;
+ closures: { before: DependencyClosure; after: DependencyClosure };
+ archiveHashes: string[];
+ proofs: { before: PreparationProof; after: PreparationProof };
+ replay?: { before: DependencyClosure; after: DependencyClosure };
+ persistClosures: boolean;
+}): VerificationExecutor {
+ let beforeProjectToolchainHash: string | undefined;
+ return {
+ observe: async (
+ source,
+ phase,
+ request,
+ runtime,
+ verifier,
+ work,
+ control,
+ _execution,
+ testEvidence,
+ witness,
+ ) => {
+ const plan = phase === 'before' ? options.beforePlan : inspectPreparation(source, options.stack);
+ if (plan.status !== 'ready')
+ throw new CsoError(
+ 'PREREQUISITE',
+ plan.prerequisites.map((item) => item.message).join('; ') ||
+ `${options.stack} dependency metadata is not ready`,
+ );
+ const admission =
+ phase === 'before'
+ ? options.beforeAdmission
+ : admitPreparationRuntime({
+ plan,
+ platform: options.targetPlatform,
+ profile: options.runtimeProfile,
+ catalog: options.runtimeCatalog,
+ });
+ if (admission.runtime.id !== runtime.id || admission.runtime.image !== runtime.image)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Prepared verification runtime changed between source phases',
+ );
+ const state = closureStateFile(options.dir, options.findingId, phase, plan),
+ supplied = options.replay?.[phase],
+ retained = !supplied && fs.existsSync(state) ? (readJson(state) as DependencyClosure) : undefined;
+ const closure =
+ phase === 'before'
+ ? (supplied ?? options.beforeClosure)
+ : await options.preparation.acquire({
+ plan,
+ admission,
+ snapshot: source,
+ deadline: options.deadline,
+ offline: options.offline || Boolean(supplied),
+ existingClosure: supplied ?? retained,
+ });
+ options.closures[phase] = closure;
+ bindArchiveHashes(options.archiveHashes, options.closures);
+ if (options.persistClosures) retainClosure(state, closure);
+ let database: RailsDatabaseSelection | undefined;
+ if (options.stack === 'rails') {
+ const selected = plan.database?.selected;
+ if (!selected)
+ throw new CsoError(
+ 'PREREQUISITE',
+ 'Rails automatic verification could not select one locked database adapter from static test configuration',
+ );
+ database =
+ selected === 'postgresql'
+ ? {
+ adapter: 'postgresql',
+ sidecar: admitPreparationSidecar({
+ platform: options.targetPlatform,
+ catalog: options.runtimeCatalog,
+ }),
+ }
+ : { adapter: 'sqlite' };
+ }
+ const prepared = await options.preparation.prepareOffline({
+ plan,
+ admission,
+ snapshot: source,
+ closure,
+ deadline: options.deadline,
+ database,
+ });
+ try {
+ options.proofs[phase] = {
+ schemaVersion: 1,
+ dependencyClosureHash: prepared.dependencyClosureHash,
+ configurationHash: prepared.configurationHash,
+ sourceProjectionHash: prepared.sourceProjectionHash,
+ preparedManifestHash: prepared.preparedManifestHash,
+ preparedDependencyHash: prepared.preparedDependencyHash,
+ receiptHash: prepared.receiptHash,
+ executionEnvironmentHash: sha256(canonical(prepared.executionEnvironment)),
+ databaseHash: prepared.databaseHash,
+ transformations: prepared.transformations,
+ };
+ const sourceTests = canonicalTestPlan(source, options.stack),
+ preparedTests = canonicalTestPlan(prepared.preparedRoot, options.stack),
+ sourceStart = canonicalStartPlan(source, options.stack, request.port),
+ preparedStart = canonicalStartPlan(prepared.preparedRoot, options.stack, request.port);
+ if (
+ sourceTests.signature !== preparedTests.signature ||
+ sourceStart.signature !== preparedStart.signature ||
+ verificationHarnessHash(request, source) !== verificationHarnessHash(request, prepared.preparedRoot)
+ )
+ throw new CsoError(
+ 'ISOLATION_FAILED',
+ 'Offline lifecycle execution changed the canonical start, test, or harness inputs',
+ );
+ if (sourceTests.toolchain === 'project') {
+ if (phase === 'before') beforeProjectToolchainHash = prepared.preparedDependencyHash;
+ else if (
+ !beforeProjectToolchainHash ||
+ prepared.preparedDependencyHash !== beforeProjectToolchainHash
+ )
+ throw new CsoError(
+ 'ASSERTION_FAILED',
+ 'Offline preparation changed the project-installed test toolchain between source phases',
+ );
+ }
+ const protectedPaths = new Set([
+ ...request.boundaryFiles,
+ ...request.testFiles,
+ ...sourceStart.entrypointFiles,
+ ...request.changes.map((item) => item.path),
+ ]);
+ if (prepared.transformations.some((item) => protectedPaths.has(item.path)))
+ throw new CsoError(
+ 'ISOLATION_FAILED',
+ 'Synthetic preparation transformation overlaps a security boundary, startup input, or test input',
+ );
+ return await options.delegate.observe(
+ prepared.preparedRoot,
+ phase,
+ request,
+ runtime,
+ verifier,
+ work,
+ control,
+ { environment: prepared.executionEnvironment, database: prepared.database },
+ testEvidence,
+ witness,
+ );
+ } finally {
+ await options.preparation.dispose(prepared);
+ }
+ },
+ };
+}
+async function verify(args: string[], dependencies: CsoCliDependencies) {
+ const { dir } = run(args);
+ if (args.length !== 1) throw new CsoError('INVALID_ARGUMENT', 'verify requires one request JSON file');
+ const raw = readInput(args[0]),
+ request = validateVerificationRequest(raw);
+ return await withLock(dir, async () => {
+ const report = loadReport(dir),
+ manifest = readJson(join(dir, 'snapshot.json')) as SnapshotManifest;
+ assertSnapshot(dir, manifest);
+ requireTime(report);
+ if (report.policy.mode !== 'comprehensive' || report.status !== 'running')
+ throw new CsoError('INVALID_SCHEMA', 'Only a running comprehensive audit can request target execution');
+ const finding = report.findings.find((f) => f.id === request.findingId && f.evidence === 'supported');
+ if (!finding)
+ throw new CsoError('MISSING_INPUT', 'Verification must target a supported finding in this run');
+ if (!request.review.artifactId)
+ throw new CsoError(
+ 'MISSING_INPUT',
+ 'Verification requires a separately persisted independent repair-review artifact',
+ );
+ const reviewArtifact = validateReviewArtifact(
+ readJson(join(dir, 'reviews', `${request.review.artifactId}.json`)),
+ report.runId,
+ request,
+ );
+ const findingPath = resolveSnapshotPath(manifest, finding.location.path, true, 'Finding path').path;
+ if (
+ !request.boundaryFiles.some(
+ (path) => resolveSnapshotPath(manifest, path, true, 'Boundary path').path === findingPath,
+ )
+ )
+ throw new CsoError('INVALID_SCHEMA', 'Boundary files must include the finding location');
+ const attempts = report.events.filter((e) => e.kind === `verification-attempt:${finding.id}`).length;
+ if (attempts >= 3)
+ throw new CsoError(
+ 'DEADLINE',
+ 'Three bounded harness/repair attempts have already been used for this finding',
+ );
+ if (report.findings.filter((f) => ['runtime_tested', 'tested'].includes(f.repair)).length >= 3)
+ throw new CsoError('INSUFFICIENT_CAPACITY', 'This run already produced three runtime-tested repairs');
+ const targetPlatform = platform();
+ let runtime;
+ try {
+ runtime = selectRuntime(request.runtimeProfile, targetPlatform, dependencies.runtimeCatalog);
+ } catch (error: any) {
+ throw new CsoError('PREREQUISITE', error?.message || 'Qualified runtime is unavailable');
+ }
+ const verifier = runtime;
+ if (!['node', 'bun', 'python', 'rails'].includes(runtime.stack))
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Application verification requires an application runtime profile',
+ );
+ const plan = inspectPreparation(join(dir, 'snapshot'), runtime.stack as any);
+ if (plan.status !== 'ready')
+ throw new CsoError(
+ 'PREREQUISITE',
+ plan.prerequisites.map((p) => p.message).join('; ') || 'Runtime preparation metadata is incomplete',
+ );
+ assertRuntimeCompatible(plan, runtime);
+ writeJson(join(dir, `preparation-${runtime.stack}.json`), plan);
+ const endpoint = await dockerEndpoint(secureDirectory(join(dir, 'home'))),
+ watchdogPath = dependencies.watchdogPath(),
+ attemptDeadline = Math.min(Date.now() + 300_000, Date.parse(report.deadline) - 60_000);
+ const admission = admitPreparationRuntime({
+ plan,
+ platform: targetPlatform,
+ profile: runtime.id,
+ catalog: dependencies.runtimeCatalog,
+ }),
+ staging = secureDirectory(join(dir, 'archive-staging')),
+ runner = new DockerPreparationSandboxRunner({
+ endpoint,
+ watchdogPath,
+ runRoot: dir,
+ controlRoot: secureDirectory(join(dir, 'preparation-execution')),
+ admission,
+ }),
+ preparation = new PreparationExecutor({
+ cache: new PublicArchiveCache({ root: publicArchiveCacheRoot(), stagingRoot: staging }),
+ runner,
+ materializationRoot: secureDirectory(join(dir, 'archive-materializations')),
+ });
+ const beforeState = closureStateFile(dir, finding.id, 'before', plan),
+ retainedBefore = fs.existsSync(beforeState) ? (readJson(beforeState) as DependencyClosure) : undefined;
+ const beforeClosure = await preparation.acquire({
+ plan,
+ admission,
+ snapshot: join(dir, 'snapshot'),
+ deadline: attemptDeadline,
+ offline: report.policy.offline,
+ existingClosure: retainedBefore,
+ });
+ retainClosure(beforeState, beforeClosure);
+ const closures = { before: beforeClosure, after: beforeClosure },
+ archiveHashes = [...new Set(beforeClosure.archives.map((archive) => archive.sha256))].sort(),
+ proofs = {} as { before: PreparationProof; after: PreparationProof },
+ delegate = new DockerVerificationExecutor(endpoint, watchdogPath, attemptDeadline, () => {
+ event(
+ report,
+ `verification-attempt:${finding.id}`,
+ `Started bounded repair verification attempt ${attempts + 1}`,
+ );
+ saveReport(dir, report);
+ });
+ const executor = preparedVerificationExecutor({
+ dir,
+ findingId: finding.id,
+ runtimeProfile: runtime.id,
+ stack: runtime.stack as 'node' | 'bun' | 'python' | 'rails',
+ targetPlatform,
+ deadline: attemptDeadline,
+ offline: report.policy.offline,
+ runtimeCatalog: dependencies.runtimeCatalog,
+ preparation,
+ delegate,
+ beforePlan: plan,
+ beforeAdmission: admission,
+ beforeClosure,
+ closures,
+ archiveHashes,
+ proofs,
+ persistClosures: true,
+ });
+ let result: Awaited>;
+ try {
+ result = await verifyRepair({
+ runId: report.runId,
+ runDir: dir,
+ manifest,
+ rawRequest: raw,
+ runtime,
+ verifier,
+ policyHash: ISOLATION_POLICY_HASH,
+ auditPolicyHash: sha256(canonical(report.policy)),
+ archives: archiveHashes,
+ dependencyClosures: closures,
+ preparation: proofs,
+ reviewArtifact,
+ executor,
+ watchdogPath,
+ attemptDeadline,
+ });
+ } catch (error) {
+ if (error instanceof VerificationAttemptError) {
+ finding.reproduction = error.attempt.reproduction;
+ finding.repair = error.attempt.repair;
+ finding.reproductionAttemptId = error.attempt.id;
+ event(
+ report,
+ error.attempt.repair === 'proposed' ? 'repair-candidate' : 'verification-failed',
+ error.attempt.repair === 'proposed'
+ ? `${error.attempt.id}: external repair assertions passed; helper-authenticated external assertion witness required before certification`
+ : `${error.attempt.id}: ${error.attempt.reproduction}; repair validation failed without issuing a bundle`,
+ );
+ saveReport(dir, report);
+ }
+ throw error;
+ }
+ finding.reproduction =
+ result.manifest.before.security === 'intended_failure' &&
+ result.manifest.before.booted &&
+ result.manifest.before.legitimate
+ ? 'reproduced'
+ : result.manifest.before.security === 'pass'
+ ? 'disproved'
+ : result.manifest.before.booted
+ ? 'inconclusive'
+ : 'blocked';
+ finding.repair =
+ result.manifest.result === 'tested'
+ ? 'tested'
+ : result.manifest.result === 'runtime_tested'
+ ? 'runtime_tested'
+ : 'failed';
+ if (['tested', 'runtime_tested'].includes(result.manifest.result)) {
+ finding.verificationId = result.manifest.id;
+ finding.verificationAssurance = {
+ assertions: result.manifest.assertionAssurance!,
+ testCompletion: result.manifest.testCompletionAssurance,
+ review: result.manifest.reviewAssurance,
+ };
+ delete finding.reproductionAttemptId;
+ }
+ event(
+ report,
+ 'verification',
+ `${result.manifest.id}: ${result.manifest.result}; assertion assurance ${result.manifest.assertionAssurance}; test completion assurance ${result.manifest.testCompletionAssurance}; review assurance ${result.manifest.reviewAssurance}`,
+ );
+ saveReport(dir, report);
+ return {
+ result: result.manifest.result,
+ verification: result.manifest,
+ bundle: `bundles/${result.bundle.id}.json`,
+ };
+ });
+}
+function replayBundle(stored: { path: string; dir: string }, id: string): any {
+ const bundle = validateRepairBundle(readJson(stored.path), id),
+ report = loadReport(stored.dir),
+ finding = report.findings.find(
+ (f) => f.verificationId === id && ['tested', 'runtime_tested'].includes(f.repair),
+ );
+ if (!['tested', 'runtime_tested'].includes(bundle.verification.result) || !finding)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Only a helper-recorded runtime-tested or host-reviewed repair bundle can be replayed',
+ );
+ return bundle;
+}
+async function withReplayBundle(
+ id: string,
+ deadline: number,
+ fn: (stored: { path: string; dir: string }, bundle: any) => Promise,
+): Promise {
+ if (!/^[a-f0-9]{32}$/.test(id))
+ throw new CsoError(
+ 'INVALID_ARGUMENT',
+ 'Bundle identifier must be the 32-character ID returned by verify',
+ );
+ let visited = 0,
+ stored: { path: string; dir: string } | undefined;
+ const admit = () => {
+ if (Date.now() >= deadline)
+ throw new CsoError(
+ 'DEADLINE',
+ 'Replay exhausted its five-minute budget while locating the recorded bundle',
+ );
+ if (++visited > REPLAY_LOOKUP_MAX_ENTRIES)
+ throw new CsoError(
+ 'INSUFFICIENT_CAPACITY',
+ `Replay bundle lookup exceeded ${REPLAY_LOOKUP_MAX_ENTRIES} private state entries`,
+ );
+ };
+ const root = privateRoot(),
+ repos = fs.opendirSync(root);
+ try {
+ let repo: fs.Dirent | null;
+ search: while ((repo = repos.readSync()) !== null) {
+ admit();
+ if (!repo.isDirectory() || !/^[a-f0-9]{24}$/.test(repo.name)) continue;
+ const repoDir = join(root, repo.name),
+ runs = fs.opendirSync(repoDir);
+ try {
+ let run: fs.Dirent | null;
+ while ((run = runs.readSync()) !== null) {
+ admit();
+ if (!run.isDirectory() || !/^\d{13}-[a-f0-9]{16}$/.test(run.name)) continue;
+ const candidate = {
+ dir: join(repoDir, run.name),
+ path: join(repoDir, run.name, 'bundles', `${id}.json`),
+ };
+ if (fs.existsSync(candidate.path)) {
+ stored = candidate;
+ break search;
+ }
+ }
+ } finally {
+ runs.closeSync();
+ }
+ }
+ } finally {
+ repos.closeSync();
+ }
+ if (!stored) throw new CsoError('MISSING_INPUT', 'Repair bundle was not found or expired');
+ let matched = false,
+ value!: T;
+ await withLock(stored.dir, async () => {
+ if (!fs.existsSync(stored!.path)) return;
+ const bundle = replayBundle(stored!, id);
+ matched = true;
+ value = await fn(stored!, bundle);
+ });
+ if (matched) return value;
+ throw new CsoError('MISSING_INPUT', 'Repair bundle was not found or expired');
+}
+function replayManifestValue(manifest: any): unknown {
+ const { id: _, createdAt: __, witnessHash: ___, before, after, ...stable } = manifest,
+ observation = (value: any) => {
+ const { output: _, ...rest } = value;
+ return rest;
+ };
+ return { ...stable, before: observation(before), after: observation(after) };
+}
+async function replay(args: string[], dependencies: CsoCliDependencies) {
+ const replayStarted = Date.now(),
+ replayDeadline = replayStarted + 300_000;
+ retention(replayStarted, {
+ deadlineMs: replayStarted + RETENTION_MAINTENANCE_MS,
+ maxEntries: RETENTION_MAX_ENTRIES,
+ });
+ if (!args.length) throw new CsoError('INVALID_ARGUMENT', 'replay requires a bundle ID');
+ const id = args.shift()!,
+ source = args.includes('--source') ? callerPath(need(args, '--source')) : undefined;
+ if (args.length) throw new CsoError('INVALID_ARGUMENT', `Unknown replay argument: ${args[0]}`);
+ return await withReplayBundle(id, replayDeadline, async (stored, bundle) => {
+ if (bundle.requiredInputs.archives?.length && !bundle.requiredInputs.dependencyClosures)
+ throw new CsoError(
+ 'MISSING_INPUT',
+ 'Replay bundle predates retained dependency closures and cannot substitute current dependency state',
+ );
+ let workDir = stored.dir,
+ manifest: any,
+ temporary: string | undefined;
+ const retained = join(stored.dir, 'snapshot');
+ try {
+ const captureSupplied = async () => {
+ if (!source)
+ throw new CsoError('MISSING_INPUT', 'Retained source expired; supply explicitly matching source');
+ const temp = newRun(source);
+ temporary = temp.dir;
+ workDir = temp.dir;
+ manifest = await capture(source, workDir, undefined, undefined, { deadlineMs: replayDeadline });
+ if (
+ manifest.executionHash !== bundle.requiredInputs.sourceHash ||
+ manifest.originalHash !== bundle.requiredInputs.originalHash
+ )
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Supplied source does not match the bundle input hashes');
+ };
+ if (fs.existsSync(retained)) {
+ manifest = readJson(join(stored.dir, 'snapshot.json'));
+ const expiresAt =
+ typeof manifest?.expiresAt === 'string' ? Date.parse(manifest.expiresAt) : Number.NaN;
+ if (!Number.isFinite(expiresAt) || new Date(expiresAt).toISOString() !== manifest.expiresAt)
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Retained snapshot expiry is invalid');
+ if (expiresAt <= Date.now()) await captureSupplied();
+ else assertSnapshot(stored.dir, manifest);
+ } else await captureSupplied();
+ if (
+ manifest.executionHash !== bundle.requiredInputs.sourceHash ||
+ manifest.originalHash !== bundle.requiredInputs.originalHash
+ )
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Retained source hashes do not match the bundle');
+ validateRepairBundle(bundle, id, join(workDir, 'snapshot'), manifest);
+ if (bundle.verification.policyHash !== ISOLATION_POLICY_HASH)
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Current helper isolation policy does not match the recorded bundle',
+ );
+ const targetPlatform = bundle.requiredInputs.platform as 'linux/amd64' | 'linux/arm64';
+ let runtime;
+ try {
+ runtime = selectRuntime(
+ bundle.verification.runtime.profile,
+ targetPlatform,
+ dependencies.runtimeCatalog,
+ );
+ } catch (error: any) {
+ throw new CsoError('PREREQUISITE', error?.message || 'Qualified replay runtime is unavailable');
+ }
+ const verifier = runtime;
+ if (runtime.image !== bundle.requiredInputs.runtimeImage)
+ throw new CsoError('INCOMPATIBLE_INPUT', 'Qualified runtime digest does not match the bundle');
+ const endpoint = await dockerEndpoint(secureDirectory(join(workDir, 'home'))),
+ watchdogPath = dependencies.watchdogPath(),
+ attemptDeadline = replayDeadline,
+ delegate = new DockerVerificationExecutor(endpoint, watchdogPath, attemptDeadline);
+ let executor: VerificationExecutor = delegate,
+ archives: string[] = [],
+ dependencyClosures: { before: DependencyClosure; after: DependencyClosure } | undefined,
+ proofs: { before: PreparationProof; after: PreparationProof } | undefined;
+ if (bundle.requiredInputs.dependencyClosures) {
+ if (!['node', 'bun', 'python', 'rails'].includes(runtime.stack))
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Replay dependency closure requires an application runtime',
+ );
+ const replayClosures = bundle.requiredInputs.dependencyClosures as {
+ before: DependencyClosure;
+ after: DependencyClosure;
+ },
+ plan = inspectPreparation(join(workDir, 'snapshot'), runtime.stack as any);
+ if (plan.status !== 'ready')
+ throw new CsoError(
+ 'PREREQUISITE',
+ plan.prerequisites.map((item: any) => item.message).join('; ') ||
+ 'Replay dependency metadata is not ready',
+ );
+ const admission = admitPreparationRuntime({
+ plan,
+ platform: targetPlatform,
+ profile: runtime.id,
+ catalog: dependencies.runtimeCatalog,
+ }),
+ runner = new DockerPreparationSandboxRunner({
+ endpoint,
+ watchdogPath,
+ runRoot: workDir,
+ controlRoot: secureDirectory(join(workDir, 'preparation-execution')),
+ admission,
+ }),
+ preparation = new PreparationExecutor({
+ cache: new PublicArchiveCache({
+ root: publicArchiveCacheRoot(),
+ stagingRoot: secureDirectory(join(workDir, 'archive-staging')),
+ }),
+ runner,
+ materializationRoot: secureDirectory(join(workDir, 'archive-materializations')),
+ }),
+ beforeClosure = await preparation.acquire({
+ plan,
+ admission,
+ snapshot: join(workDir, 'snapshot'),
+ deadline: attemptDeadline,
+ offline: true,
+ existingClosure: replayClosures.before,
+ });
+ dependencyClosures = { before: beforeClosure, after: replayClosures.after };
+ archives = [
+ ...new Set(
+ [...beforeClosure.archives, ...replayClosures.after.archives].map((archive) => archive.sha256),
+ ),
+ ].sort();
+ proofs = {} as { before: PreparationProof; after: PreparationProof };
+ executor = preparedVerificationExecutor({
+ dir: workDir,
+ findingId: bundle.request.findingId,
+ runtimeProfile: runtime.id,
+ stack: runtime.stack as any,
+ targetPlatform,
+ deadline: attemptDeadline,
+ offline: true,
+ runtimeCatalog: dependencies.runtimeCatalog,
+ preparation,
+ delegate,
+ beforePlan: plan,
+ beforeAdmission: admission,
+ beforeClosure,
+ closures: dependencyClosures,
+ archiveHashes: archives,
+ proofs,
+ replay: replayClosures,
+ persistClosures: false,
+ });
+ }
+ const result = await verifyRepair({
+ runId: bundle.runId,
+ runDir: workDir,
+ manifest,
+ rawRequest: bundle.request,
+ runtime,
+ verifier,
+ policyHash: ISOLATION_POLICY_HASH,
+ auditPolicyHash: bundle.verification.auditPolicyHash,
+ archives,
+ dependencyClosures,
+ preparation: proofs,
+ reviewArtifact: bundle.reviewArtifact,
+ executor,
+ persist: false,
+ watchdogPath,
+ attemptDeadline,
+ });
+ if (
+ canonical(replayManifestValue(result.manifest)) !==
+ canonical(replayManifestValue(bundle.verification)) ||
+ assertionWitnessReplayHash(result.bundle.witness!) !== assertionWitnessReplayHash(bundle.witness)
+ )
+ throw new CsoError(
+ 'INCOMPATIBLE_INPUT',
+ 'Replay changed verification outcomes, preparation, or provenance inputs',
+ );
+ const replayId = `${Date.now()}-${randomBytes(8).toString('hex')}`;
+ writeJsonExclusive(join(stored.dir, 'replays', `${replayId}.json`), {
+ replayId,
+ bundleId: id,
+ verification: result.manifest,
+ witness: result.bundle.witness,
+ });
+ return { bundle: id, result: result.manifest.result, replay: result.manifest, replayId };
+ } finally {
+ if (temporary) finalizeReplayTemporary(temporary);
+ }
+ });
+}
+
+export async function dispatchCsoCommand(
+ command: string,
+ args: string[],
+ dependencies: CsoCliDependencies,
+): Promise {
+ args = [...args];
+ if (!['start', 'recheck', 'replay', 'doctor', 'provision-images'].includes(command)) {
+ const started = Date.now();
+ retention(started, { deadlineMs: started + RETENTION_MAINTENANCE_MS, maxEntries: RETENTION_MAX_ENTRIES });
+ }
+ let result: unknown;
+ switch (command) {
+ case 'start':
+ result = await start(args, dependencies);
+ break;
+ case 'doctor':
+ result = await doctor(args, dependencies);
+ break;
+ case 'provision-images':
+ result = await provisionImages(args, dependencies);
+ break;
+ case 'resume':
+ result = resume(args);
+ break;
+ case 'inspect':
+ await inspect(args);
+ return;
+ case 'read':
+ await read(args);
+ return;
+ case 'history':
+ await history(args);
+ return;
+ case 'submit':
+ result = submit(args);
+ break;
+ case 'finish':
+ result = finish(args);
+ break;
+ case 'import-v2':
+ result = importV2(args);
+ break;
+ case 'inspect-v2':
+ result = inspectV2(args);
+ break;
+ case 'scan':
+ result = await scanner(args);
+ break;
+ case 'scanner-outcome':
+ result = scannerOutcome(args);
+ break;
+ case 'import-sarif':
+ result = await scanner(args, true);
+ break;
+ case 'record-review':
+ result = recordReview(args);
+ break;
+ case 'test-plan':
+ result = testPlan(args);
+ break;
+ case 'runtime-plan':
+ result = runtimePlan(args);
+ break;
+ case 'recheck':
+ result = await recheck(args, dependencies);
+ break;
+ case 'verify':
+ result = await verify(args, dependencies);
+ break;
+ case 'replay':
+ result = await replay(args, dependencies);
+ break;
+ case 'patch-hash':
+ if (args.length !== 1)
+ throw new CsoError('INVALID_ARGUMENT', 'patch-hash requires one request JSON file');
+ result = { patchHash: patchHash(validateVerificationRequest(readInput(args[0]))) };
+ break;
+ default:
+ throw new CsoError('INVALID_ARGUMENT', `Unknown command: ${command}`);
+ }
return result;
}
-const PRODUCTION_CLI_DEPENDENCIES:CsoCliDependencies=Object.freeze({runtimeCatalog:RUNTIME_CATALOG,scannerCatalog:SCANNER_CATALOG,catalogImageSession:productionCatalogImageSession,watchdogPath:watchdog});
-async function main(){const args=process.argv.slice(2),command=args.shift();if(!command||command==='--help'||command==='help'){process.stdout.write(HELP+'\n');return;}if(command==='--version'){emit({version:VERSION,abi:ABI});return;}if(command==='schema'){emit(SCHEMA);return;}if(command==='__cso-assertion-witness'){if(args.length)throw new CsoError('INVALID_ARGUMENT','Assertion witness does not accept command arguments');await runAssertionWitnessChild();return;}const result=await dispatchCsoCommand(command,args,PRODUCTION_CLI_DEPENDENCIES);if(result!==undefined)emit(result);}
-if(import.meta.main)main().catch(error=>{const e=error instanceof CsoError?error:new CsoError('INVALID_SCHEMA','The helper rejected an unexpected or unsafe input');try{process.stderr.write(redact(JSON.stringify({ok:false,error:{code:e.code,message:e.message}}))+'\n');}catch{process.stderr.write('{"ok":false,"error":{"code":"REDACTION_FAILED","message":"Error payload withheld"}}\n');}process.exitCode=1;});
+const PRODUCTION_CLI_DEPENDENCIES: CsoCliDependencies = Object.freeze({
+ runtimeCatalog: RUNTIME_CATALOG,
+ scannerCatalog: SCANNER_CATALOG,
+ catalogImageSession: productionCatalogImageSession,
+ watchdogPath: watchdog,
+});
+async function main() {
+ const args = process.argv.slice(2),
+ command = args.shift();
+ if (!command || command === '--help' || command === 'help') {
+ process.stdout.write(HELP + '\n');
+ return;
+ }
+ if (command === '--version') {
+ emit({ version: VERSION, abi: ABI });
+ return;
+ }
+ if (command === 'schema') {
+ emit(SCHEMA);
+ return;
+ }
+ if (command === '__cso-assertion-witness') {
+ if (args.length)
+ throw new CsoError('INVALID_ARGUMENT', 'Assertion witness does not accept command arguments');
+ await runAssertionWitnessChild();
+ return;
+ }
+ const result = await dispatchCsoCommand(command, args, PRODUCTION_CLI_DEPENDENCIES);
+ if (result !== undefined) emit(result);
+}
+if (import.meta.main)
+ main().catch((error) => {
+ const e =
+ error instanceof CsoError
+ ? error
+ : new CsoError('INVALID_SCHEMA', 'The helper rejected an unexpected or unsafe input');
+ try {
+ process.stderr.write(
+ redact(JSON.stringify({ ok: false, error: { code: e.code, message: e.message } })) + '\n',
+ );
+ } catch {
+ process.stderr.write(
+ '{"ok":false,"error":{"code":"REDACTION_FAILED","message":"Error payload withheld"}}\n',
+ );
+ }
+ process.exitCode = 1;
+ });
diff --git a/lib/cso/contracts.ts b/lib/cso/contracts.ts
index d383f6c7c..ff3ed32fe 100644
--- a/lib/cso/contracts.ts
+++ b/lib/cso/contracts.ts
@@ -3,80 +3,225 @@ import { createHash } from 'node:crypto';
export const ABI = 3;
export const MAX_OUTPUT = 1024 * 1024;
-const UNSAFE_STRING_CONTROLS=/[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f\u061c\u200e\u200f\u2028-\u202e\u2066-\u2069]/u;
-const UNSAFE_PROPERTY_CONTROLS=/[\x00-\x1f\x7f-\x9f\u061c\u200e\u200f\u2028-\u202e\u2066-\u2069]/u;
+const UNSAFE_STRING_CONTROLS =
+ /[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f\u061c\u200e\u200f\u2028-\u202e\u2066-\u2069]/u;
+const UNSAFE_PROPERTY_CONTROLS = /[\x00-\x1f\x7f-\x9f\u061c\u200e\u200f\u2028-\u202e\u2066-\u2069]/u;
export type Completeness = 'complete' | 'partial' | 'not assessed';
export type Severity = 'critical' | 'high' | 'medium' | 'low' | 'informational';
-export type ErrorCode = 'INVALID_ARGUMENT' | 'INVALID_SCHEMA' | 'MISSING_INPUT' | 'SNAPSHOT_RACE' |
- 'UNSAFE_PATH' | 'REDACTION_FAILED' | 'PERSISTENCE_FAILED' | 'TOOL_UNAVAILABLE' | 'TOOL_FAILED' |
- 'ISOLATION_FAILED' | 'INSUFFICIENT_CAPACITY' | 'DEADLINE' | 'CANCELLED' | 'PREREQUISITE' |
- 'INCOMPATIBLE_INPUT' | 'ASSERTION_FAILED';
+export type ErrorCode =
+ | 'INVALID_ARGUMENT'
+ | 'INVALID_SCHEMA'
+ | 'MISSING_INPUT'
+ | 'SNAPSHOT_RACE'
+ | 'UNSAFE_PATH'
+ | 'REDACTION_FAILED'
+ | 'PERSISTENCE_FAILED'
+ | 'TOOL_UNAVAILABLE'
+ | 'TOOL_FAILED'
+ | 'ISOLATION_FAILED'
+ | 'INSUFFICIENT_CAPACITY'
+ | 'DEADLINE'
+ | 'CANCELLED'
+ | 'PREREQUISITE'
+ | 'INCOMPATIBLE_INPUT'
+ | 'ASSERTION_FAILED';
export class CsoError extends Error {
- constructor(public code: ErrorCode, message: string) { super(message); this.name = 'CsoError'; }
+ constructor(
+ public code: ErrorCode,
+ message: string,
+ ) {
+ super(message);
+ this.name = 'CsoError';
+ }
}
export const sha256 = (value: string | Buffer): string => createHash('sha256').update(value).digest('hex');
export const canonical = (value: unknown): string => {
if (Array.isArray(value)) return `[${value.map(canonical).join(',')}]`;
- if (value && typeof value === 'object') return `{${Object.keys(value).sort().map(k => `${JSON.stringify(k)}:${canonical((value as any)[k])}`).join(',')}}`;
+ if (value && typeof value === 'object')
+ return `{${Object.keys(value)
+ .sort()
+ .map((k) => `${JSON.stringify(k)}:${canonical((value as any)[k])}`)
+ .join(',')}}`;
return JSON.stringify(value);
};
export interface CoverageRecord {
- domain: string; scope: string; status: 'assessed' | 'partial' | 'not_assessed' | 'not_applicable';
- method: string; gaps: string[]; exclusions: string[]; evidence: string[];
+ domain: string;
+ scope: string;
+ status: 'assessed' | 'partial' | 'not_assessed' | 'not_applicable';
+ method: string;
+ gaps: string[];
+ exclusions: string[];
+ evidence: string[];
tool?: { name: string; version: string; freshness: string; outcome: string };
}
export interface ApplicationModel {
- actors: string[]; assets: string[]; entrypoints: string[]; tenantBoundaries: string[];
- sensitiveOperations: string[]; invariants: string[];
+ actors: string[];
+ assets: string[];
+ entrypoints: string[];
+ tenantBoundaries: string[];
+ sensitiveOperations: string[];
+ invariants: string[];
}
export interface FindingV3 {
- id: string; fingerprint: string; title: string; rootCause: string;
- location: { path: string; line: number; symbol: string }; advisoryIds: string[];
- severity: Severity; confidence: 'high' | 'medium' | 'low'; confidenceRationale:string; evidence: 'supported' | 'hypothesis' | 'legacy_review';
- attackerControl: string; impact: string; scenario: string; trace: string[]; references: string[]; recommendation: string;
- challenge: { reviewer: string; independent: boolean; mode:'independent_agent'|'sequential_fallback'; callers: string; controls: string; counterevidence: string; conclusion: string };
- dependency?: { affectedVersion: string; reachability: 'reachable' | 'unreachable' | 'unknown'; exposure: string; exploitation: string };
+ id: string;
+ fingerprint: string;
+ title: string;
+ rootCause: string;
+ location: { path: string; line: number; symbol: string };
+ advisoryIds: string[];
+ severity: Severity;
+ confidence: 'high' | 'medium' | 'low';
+ confidenceRationale: string;
+ evidence: 'supported' | 'hypothesis' | 'legacy_review';
+ attackerControl: string;
+ impact: string;
+ scenario: string;
+ trace: string[];
+ references: string[];
+ recommendation: string;
+ challenge: {
+ reviewer: string;
+ independent: boolean;
+ mode: 'independent_agent' | 'sequential_fallback';
+ callers: string;
+ controls: string;
+ counterevidence: string;
+ conclusion: string;
+ };
+ dependency?: {
+ affectedVersion: string;
+ reachability: 'reachable' | 'unreachable' | 'unknown';
+ exposure: string;
+ exploitation: string;
+ };
reproduction: 'not_attempted' | 'blocked' | 'inconclusive' | 'disproved' | 'reproduced';
repair: 'not_attempted' | 'proposed' | 'failed' | 'runtime_tested' | 'tested';
- closure: 'open' | 'resolved' | 'unknown'; verificationId?: string; reproductionAttemptId?:string;
- verificationAssurance?: { assertions:'authenticated_out_of_process'; testCompletion:'self_reported'|'authenticated_out_of_process'; review:'self_attested'|'host_verified' };
+ closure: 'open' | 'resolved' | 'unknown';
+ verificationId?: string;
+ reproductionAttemptId?: string;
+ verificationAssurance?: {
+ assertions: 'authenticated_out_of_process';
+ testCompletion: 'self_reported' | 'authenticated_out_of_process';
+ review: 'self_attested' | 'host_verified';
+ };
}
export interface RunPolicy {
- mode: 'daily' | 'comprehensive'; scope: string; diff: boolean; base: string;
- offline: boolean; budgetSeconds: number; maxWorkers: 3; maxRepairs: 3;
+ mode: 'daily' | 'comprehensive';
+ scope: string;
+ diff: boolean;
+ base: string;
+ offline: boolean;
+ budgetSeconds: number;
+ maxWorkers: 3;
+ maxRepairs: 3;
}
export interface RunReportV3 {
- schemaVersion: 3; runId: string; repoId: string; createdAt: string; deadline: string;
- status: 'running' | 'finished' | 'interrupted'; completeness: Completeness; policy: RunPolicy;
- source: { root: string; snapshotHash: string; originalHash: string; baseCommit?: string;
- transformations?: Array<{ path: string; handling: string }> };
- application: ApplicationModel; coverage: CoverageRecord[]; findings: FindingV3[];
- gaps: string[]; events: { at: string; kind: string; message: string }[];
- parent?: { runId: string; findingId: string; kind: 'recheck' }; modelUsage?: { source: string; tokens: number; cost?: number };
+ schemaVersion: 3;
+ runId: string;
+ repoId: string;
+ createdAt: string;
+ deadline: string;
+ status: 'running' | 'finished' | 'interrupted';
+ completeness: Completeness;
+ policy: RunPolicy;
+ source: {
+ root: string;
+ snapshotHash: string;
+ originalHash: string;
+ baseCommit?: string;
+ transformations?: Array<{ path: string; handling: string }>;
+ };
+ application: ApplicationModel;
+ coverage: CoverageRecord[];
+ findings: FindingV3[];
+ gaps: string[];
+ events: { at: string; kind: string; message: string }[];
+ parent?: { runId: string; findingId: string; kind: 'recheck' };
+ modelUsage?: { source: string; tokens: number; cost?: number };
+}
+export interface SnapshotEntry {
+ path: string;
+ pathId: string;
+ originalHash: string;
+ executionHash?: string;
+ bytes: number;
+ mode: number;
+ transformation?: string;
+}
+export interface SnapshotPathIdentity {
+ path: string;
+ pathId: string;
}
-export interface SnapshotEntry { path: string; pathId: string; originalHash: string; executionHash?: string; bytes: number; mode: number; transformation?: string }
-export interface SnapshotPathIdentity { path: string; pathId: string }
export interface SnapshotManifest {
- version: 3; createdAt: string; expiresAt: string; root: string; originalHash: string; executionHash: string;
- entries: SnapshotEntry[]; deletedPaths?: SnapshotPathIdentity[]; headCommit?: string; baseCommit?: string; changedPaths?: string[];
+ version: 3;
+ createdAt: string;
+ expiresAt: string;
+ root: string;
+ originalHash: string;
+ executionHash: string;
+ entries: SnapshotEntry[];
+ deletedPaths?: SnapshotPathIdentity[];
+ headCommit?: string;
+ baseCommit?: string;
+ changedPaths?: string[];
}
export interface HttpAssertion {
- name: string; path: string; method: 'GET' | 'POST' | 'PUT' | 'PATCH' | 'DELETE';
- headers?: Record; body?: string;
+ name: string;
+ path: string;
+ method: 'GET' | 'POST' | 'PUT' | 'PATCH' | 'DELETE';
+ headers?: Record;
+ body?: string;
expected: { status: number; includes?: string; excludes?: string };
vulnerable?: { status: number; includes?: string; excludes?: string };
}
-export interface Command { executable: string; args: string[] }
+export interface Command {
+ executable: string;
+ args: string[];
+}
export interface VerificationRequest {
- findingId: string; runtimeProfile: string; port: number; start: Command;
- legitimate: HttpAssertion[]; security: HttpAssertion; existingTests: Command[];
- fixtures: Record;
- boundaryFiles: string[]; testFiles:string[];
- changes: { path: string; beforeSha256: string | null; after: string | null; effect: 'source' | 'configuration' | 'dependency' }[];
- review: { reviewer: string; independent: boolean; rootCauseRepaired: boolean; featurePreserved: boolean;
- boundaryMocks: boolean; rationale: string; reviewedPatchHash: string; artifactId?:string };
+ findingId: string;
+ runtimeProfile: string;
+ port: number;
+ start: Command;
+ legitimate: HttpAssertion[];
+ security: HttpAssertion;
+ existingTests: Command[];
+ fixtures: Record;
+ boundaryFiles: string[];
+ testFiles: string[];
+ changes: {
+ path: string;
+ beforeSha256: string | null;
+ after: string | null;
+ effect: 'source' | 'configuration' | 'dependency';
+ }[];
+ review: {
+ reviewer: string;
+ independent: boolean;
+ rootCauseRepaired: boolean;
+ featurePreserved: boolean;
+ boundaryMocks: boolean;
+ rationale: string;
+ reviewedPatchHash: string;
+ artifactId?: string;
+ };
+}
+export interface RepairReviewArtifact {
+ schemaVersion: 3;
+ id: string;
+ runId: string;
+ findingId: string;
+ createdAt: string;
+ producer: string;
+ reviewer: string;
+ assurance: 'self_attested' | 'host_verified';
+ requestHash: string;
+ patchHash: string;
+ rootCauseRepaired: boolean;
+ featurePreserved: boolean;
+ boundaryMocks: boolean;
+ rationale: string;
}
-export interface RepairReviewArtifact {schemaVersion:3;id:string;runId:string;findingId:string;createdAt:string;producer:string;reviewer:string;assurance:'self_attested'|'host_verified';requestHash:string;patchHash:string;rootCauseRepaired:boolean;featurePreserved:boolean;boundaryMocks:boolean;rationale:string}
export interface RecheckEvidenceV3 {
kind: 'caller' | 'security_boundary';
path: string;
@@ -89,235 +234,640 @@ export interface SubmissionV3 {
coverage?: unknown[];
gaps?: string[];
modelUsage?: { source: string; tokens: number; cost?: number };
- recheck?: { findingId: string; outcome: 'open' | 'resolved' | 'unknown'; evidence: RecheckEvidenceV3[]; rootCause: string };
+ recheck?: {
+ findingId: string;
+ outcome: 'open' | 'resolved' | 'unknown';
+ evidence: RecheckEvidenceV3[];
+ rootCause: string;
+ };
}
export interface VerificationObservation {
- booted: boolean; legitimate: boolean; security: 'pass' | 'intended_failure' | 'inconclusive';
- existingTests: boolean; output: string; inputHash: string;
+ booted: boolean;
+ legitimate: boolean;
+ security: 'pass' | 'intended_failure' | 'inconclusive';
+ existingTests: boolean;
+ output: string;
+ inputHash: string;
}
export interface AssertionWitnessBinding {
- schemaVersion: 1; protocol: 'gstack-cso-assertion-witness-v1'; nonce: string; phase: 'before' | 'after';
- issuedAt: string; expiresAt: string; runId: string; findingId: string;
- policyHash: string; auditPolicyHash: string;
+ schemaVersion: 1;
+ protocol: 'gstack-cso-assertion-witness-v1';
+ nonce: string;
+ phase: 'before' | 'after';
+ issuedAt: string;
+ expiresAt: string;
+ runId: string;
+ findingId: string;
+ policyHash: string;
+ auditPolicyHash: string;
runtime: { image: string; verifierImage: string; platform: string; profile: string };
- runner: { testToolchain: 'runtime' | 'project'; startPlanHash: string; testPlanHash: string;
- commandsHash: string; minimumPassingTestsHash: string };
- sourceHash: string; dependencyHash: string; configurationHash: string; requestHash: string;
- patchHash: string; harnessHash: string; assertionHash: string; fixturesHash: string;
+ runner: {
+ testToolchain: 'runtime' | 'project';
+ startPlanHash: string;
+ testPlanHash: string;
+ commandsHash: string;
+ minimumPassingTestsHash: string;
+ };
+ sourceHash: string;
+ dependencyHash: string;
+ configurationHash: string;
+ requestHash: string;
+ patchHash: string;
+ harnessHash: string;
+ assertionHash: string;
+ fixturesHash: string;
}
export interface AssertionWitnessReceipt {
- schemaVersion: 1; binding: AssertionWitnessBinding; keyId: string; publicKey: string;
- observationHash: string; externalAssertionsPassed: boolean; diagnosticTestsPassed: boolean;
- executions: Array<{ commandHash: string; exitCode: number; outputHash: string; minimumPassingTests: number;
- executedTests: number; passingTests: number; reportedPassed: boolean }>;
+ schemaVersion: 1;
+ binding: AssertionWitnessBinding;
+ keyId: string;
+ publicKey: string;
+ observationHash: string;
+ externalAssertionsPassed: boolean;
+ diagnosticTestsPassed: boolean;
+ executions: Array<{
+ commandHash: string;
+ exitCode: number;
+ outputHash: string;
+ minimumPassingTests: number;
+ executedTests: number;
+ passingTests: number;
+ reportedPassed: boolean;
+ }>;
signature: string;
}
export interface PreparationProof {
- schemaVersion: 1; dependencyClosureHash: string; configurationHash: string; sourceProjectionHash: string;
- preparedManifestHash: string; preparedDependencyHash: string; receiptHash: string; executionEnvironmentHash: string; databaseHash: string;
+ schemaVersion: 1;
+ dependencyClosureHash: string;
+ configurationHash: string;
+ sourceProjectionHash: string;
+ preparedManifestHash: string;
+ preparedDependencyHash: string;
+ receiptHash: string;
+ executionEnvironmentHash: string;
+ databaseHash: string;
transformations: Array<{ path: string; sha256: string; mode: number; reason: string }>;
}
export interface VerificationManifest {
- version: 3; id: string; runId: string; findingId: string; createdAt: string;
- helperAbi: 3; runtime: { image: string; platform: string; profile: string };
+ version: 3;
+ id: string;
+ runId: string;
+ findingId: string;
+ createdAt: string;
+ helperAbi: 3;
+ runtime: { image: string; platform: string; profile: string };
testToolchain: 'runtime' | 'project';
- policyHash: string; harnessHash: string; requestHash:string; startPlanHash:string; testPlanHash:string; fixturesHash: string; patchHash: string;
- auditPolicyHash:string; originalSourceHash:string; transformationsHash:string; archivesHash:string; preparationHash?:string;
- beforeSourceHash: string; afterSourceHash: string; beforeDependencies: string; afterDependencies: string;
- beforeConfiguration: string; afterConfiguration: string;
- before: VerificationObservation; after: VerificationObservation;
- review: VerificationRequest['review']; reviewAssurance:'self_attested'|'host_verified';
- assertionAssurance?:'authenticated_out_of_process';
- testCompletionAssurance:'self_reported'|'authenticated_out_of_process';
+ policyHash: string;
+ harnessHash: string;
+ requestHash: string;
+ startPlanHash: string;
+ testPlanHash: string;
+ fixturesHash: string;
+ patchHash: string;
+ auditPolicyHash: string;
+ originalSourceHash: string;
+ transformationsHash: string;
+ archivesHash: string;
+ preparationHash?: string;
+ beforeSourceHash: string;
+ afterSourceHash: string;
+ beforeDependencies: string;
+ afterDependencies: string;
+ beforeConfiguration: string;
+ afterConfiguration: string;
+ before: VerificationObservation;
+ after: VerificationObservation;
+ review: VerificationRequest['review'];
+ reviewAssurance: 'self_attested' | 'host_verified';
+ assertionAssurance?: 'authenticated_out_of_process';
+ testCompletionAssurance: 'self_reported' | 'authenticated_out_of_process';
witnessHash?: string;
result: 'runtime_tested' | 'tested' | 'failed' | 'inconclusive';
}
export interface RepairBundle {
- schemaVersion: 3; runId: string; id: string; createdAt: string; expiresAt: string;
- requiredInputs: { sourceHash: string; originalHash: string; runtimeImage: string; platform: string; archives: string[];
- dependencyClosures?: { before: unknown; after: unknown } };
- request: VerificationRequest; verification: VerificationManifest; transformations: SnapshotEntry[];
- preparation?: { before: PreparationProof; after: PreparationProof }; reviewArtifact?:RepairReviewArtifact;
+ schemaVersion: 3;
+ runId: string;
+ id: string;
+ createdAt: string;
+ expiresAt: string;
+ requiredInputs: {
+ sourceHash: string;
+ originalHash: string;
+ runtimeImage: string;
+ platform: string;
+ archives: string[];
+ dependencyClosures?: { before: unknown; after: unknown };
+ };
+ request: VerificationRequest;
+ verification: VerificationManifest;
+ transformations: SnapshotEntry[];
+ preparation?: { before: PreparationProof; after: PreparationProof };
+ reviewArtifact?: RepairReviewArtifact;
witness?: { before: AssertionWitnessReceipt; after: AssertionWitnessReceipt };
}
export function object(value: unknown, name = 'input'): Record {
- if (!value || typeof value !== 'object' || Array.isArray(value)) throw new CsoError('INVALID_SCHEMA', `${name} must be an object`);
- return value as Record;
+ if (!value || typeof value !== 'object' || Array.isArray(value))
+ throw new CsoError('INVALID_SCHEMA', `${name} must be an object`);
+ return value as Record;
}
-function exact(value:Record,allowed:readonly string[],name:string):void{
- for(const key of Object.keys(value))if(!allowed.includes(key))throw new CsoError('INVALID_SCHEMA',`Unexpected ${name} field: ${key}`);
+function exact(value: Record, allowed: readonly string[], name: string): void {
+ for (const key of Object.keys(value))
+ if (!allowed.includes(key)) throw new CsoError('INVALID_SCHEMA', `Unexpected ${name} field: ${key}`);
}
-function boolean(value:unknown,name:string):boolean{
- if(typeof value!=='boolean')throw new CsoError('INVALID_SCHEMA',`${name} must be a boolean`);return value;
+function boolean(value: unknown, name: string): boolean {
+ if (typeof value !== 'boolean') throw new CsoError('INVALID_SCHEMA', `${name} must be a boolean`);
+ return value;
}
export function string(value: unknown, name: string, max = 8192): string {
- if (typeof value !== 'string' || !value.trim() || value.length > max || UNSAFE_STRING_CONTROLS.test(value)) throw new CsoError('INVALID_SCHEMA', `${name} must be a nonempty string without unsafe control characters (maximum ${max})`);
+ if (typeof value !== 'string' || !value.trim() || value.length > max || UNSAFE_STRING_CONTROLS.test(value))
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ `${name} must be a nonempty string without unsafe control characters (maximum ${max})`,
+ );
return value;
}
export function strings(value: unknown, name: string): string[] {
- if (!Array.isArray(value) || value.length > 1000) throw new CsoError('INVALID_SCHEMA', `${name} must be an array`);
- return value.map((v,i) => string(v, `${name}[${i}]`));
+ if (!Array.isArray(value) || value.length > 1000)
+ throw new CsoError('INVALID_SCHEMA', `${name} must be an array`);
+ return value.map((v, i) => string(v, `${name}[${i}]`));
}
export function oneOf(value: unknown, choices: readonly T[], name: string): T {
- if (!choices.includes(value as T)) throw new CsoError('INVALID_SCHEMA', `${name} must be one of ${choices.join(', ')}`);
+ if (!choices.includes(value as T))
+ throw new CsoError('INVALID_SCHEMA', `${name} must be one of ${choices.join(', ')}`);
return value as T;
}
export function relativePath(value: unknown): string {
const p = string(value, 'relative path', 4096);
- if (p.startsWith('/') || p.includes('\\') || /^[A-Za-z]:/.test(p) || p.split('/').some(x => !x || x === '.' || x === '..') || /[\x00-\x1f\x7f]/.test(p))
+ if (
+ p.startsWith('/') ||
+ p.includes('\\') ||
+ /^[A-Za-z]:/.test(p) ||
+ p.split('/').some((x) => !x || x === '.' || x === '..') ||
+ /[\x00-\x1f\x7f]/.test(p)
+ )
throw new CsoError('UNSAFE_PATH', 'Expected a contained relative path');
return p;
}
-const SNAPSHOT_PATH_HANDLE=/^@cso-path\/\/([a-f0-9]{32})$/;
-export function snapshotPathId(root:string,path:string):string{
+const SNAPSHOT_PATH_HANDLE = /^@cso-path\/\/([a-f0-9]{32})$/;
+export function snapshotPathId(root: string, path: string): string {
// Validate the root for callers, but do not salt the opaque identity with its
// absolute checkout path. Replay must resolve the same retained path after a
// matching source tree is supplied from another checkout.
- string(root,'snapshot root',8192);const relative=relativePath(path);
- return sha256(canonical({kind:'cso-path-v3',path:relative})).slice(0,32);
+ string(root, 'snapshot root', 8192);
+ const relative = relativePath(path);
+ return sha256(canonical({ kind: 'cso-path-v3', path: relative })).slice(0, 32);
}
-export function snapshotPathHandle(pathId:string):string{
- if(!/^[a-f0-9]{32}$/.test(pathId))throw new CsoError('INVALID_SCHEMA','Snapshot path ID must be 32 lowercase hexadecimal characters');
+export function snapshotPathHandle(pathId: string): string {
+ if (!/^[a-f0-9]{32}$/.test(pathId))
+ throw new CsoError('INVALID_SCHEMA', 'Snapshot path ID must be 32 lowercase hexadecimal characters');
return `@cso-path//${pathId}`;
}
-export function snapshotPathHandleId(value:unknown):string|undefined{
- if(typeof value!=='string')return;
+export function snapshotPathHandleId(value: unknown): string | undefined {
+ if (typeof value !== 'string') return;
return SNAPSHOT_PATH_HANDLE.exec(value)?.[1];
}
-export function snapshotReference(value:unknown):string{
- const reference=string(value,'snapshot path or handle',4096);
- return snapshotPathHandleId(reference)?reference:relativePath(reference);
+export function snapshotReference(value: unknown): string {
+ const reference = string(value, 'snapshot path or handle', 4096);
+ return snapshotPathHandleId(reference) ? reference : relativePath(reference);
}
-export function snapshotOriginalIdentity(entries:Array>,deletedPaths:Array>=[]):string{
- const present=entries.map(entry=>[entry.path,entry.originalHash,entry.mode]);
+export function snapshotOriginalIdentity(
+ entries: Array>,
+ deletedPaths: Array> = [],
+): string {
+ const present = entries.map((entry) => [entry.path, entry.originalHash, entry.mode]);
// Preserve the original no-deletion identity for v3 artifacts already
// retained by pre-release builds. Any deletion changes the identity and is
// therefore impossible to strip from a manifest without detection.
- return sha256(canonical(deletedPaths.length?{entries:present,deletedPaths:deletedPaths.map(item=>item.path).sort()}:present));
+ return sha256(
+ canonical(
+ deletedPaths.length
+ ? { entries: present, deletedPaths: deletedPaths.map((item) => item.path).sort() }
+ : present,
+ ),
+ );
}
-export function rootCauseIdentity(value:string):string{return value.normalize('NFKC').trim().replace(/\s+/g,' ').toLowerCase();}
-function advisoryIdentities(values:string[]):string[]{return [...new Set(values.map(value=>value.normalize('NFKC').trim().toUpperCase()))].sort();}
-export function fingerprint(f: Pick): string {
+export function rootCauseIdentity(value: string): string {
+ return value.normalize('NFKC').trim().replace(/\s+/g, ' ').toLowerCase();
+}
+function advisoryIdentities(values: string[]): string[] {
+ return [...new Set(values.map((value) => value.normalize('NFKC').trim().toUpperCase()))].sort();
+}
+export function fingerprint(f: Pick): string {
// Titles, line shifts, severity, and generated descriptions are deliberately absent.
- return sha256(canonical({ rootCause: rootCauseIdentity(f.rootCause), path: f.location.path, symbol: f.location.symbol, advisories: advisoryIdentities(f.advisoryIds) })).slice(0,32);
+ return sha256(
+ canonical({
+ rootCause: rootCauseIdentity(f.rootCause),
+ path: f.location.path,
+ symbol: f.location.symbol,
+ advisories: advisoryIdentities(f.advisoryIds),
+ }),
+ ).slice(0, 32);
}
export function validateFinding(input: unknown): FindingV3 {
- const v = object(input, 'finding'), loc = object(v.location,'location'), c = object(v.challenge,'challenge');
- for (const reserved of ['reproduction','repair','closure','verificationId','reproductionAttemptId','verificationAssurance']) if (reserved in v)
- throw new CsoError('INVALID_SCHEMA', `${reserved} is helper-owned`);
- exact(v,['title','rootCause','location','advisoryIds','severity','confidence','confidenceRationale','evidence','attackerControl','impact','scenario','trace','references','recommendation','challenge','dependency'],'finding');
- exact(loc,['path','line','symbol'],'location');
- exact(c,['reviewer','independent','mode','callers','controls','counterevidence','conclusion'],'challenge');
+ const v = object(input, 'finding'),
+ loc = object(v.location, 'location'),
+ c = object(v.challenge, 'challenge');
+ for (const reserved of [
+ 'reproduction',
+ 'repair',
+ 'closure',
+ 'verificationId',
+ 'reproductionAttemptId',
+ 'verificationAssurance',
+ ])
+ if (reserved in v) throw new CsoError('INVALID_SCHEMA', `${reserved} is helper-owned`);
+ exact(
+ v,
+ [
+ 'title',
+ 'rootCause',
+ 'location',
+ 'advisoryIds',
+ 'severity',
+ 'confidence',
+ 'confidenceRationale',
+ 'evidence',
+ 'attackerControl',
+ 'impact',
+ 'scenario',
+ 'trace',
+ 'references',
+ 'recommendation',
+ 'challenge',
+ 'dependency',
+ ],
+ 'finding',
+ );
+ exact(loc, ['path', 'line', 'symbol'], 'location');
+ exact(
+ c,
+ ['reviewer', 'independent', 'mode', 'callers', 'controls', 'counterevidence', 'conclusion'],
+ 'challenge',
+ );
const f: FindingV3 = {
- id: '', fingerprint: '', title: string(v.title,'title'), rootCause: string(v.rootCause,'rootCause'),
- location: { path: snapshotReference(loc.path), line: loc.line, symbol: string(loc.symbol,'symbol') },
- advisoryIds: advisoryIdentities(strings(v.advisoryIds ?? [],'advisoryIds')),
- severity: oneOf(v.severity,['critical','high','medium','low','informational'],'severity'),
- confidence: oneOf(v.confidence,['high','medium','low'],'confidence'),
- confidenceRationale: string(v.confidenceRationale,'confidenceRationale'),
- evidence: oneOf(v.evidence,['supported','hypothesis'],'evidence'),
- attackerControl: string(v.attackerControl,'attackerControl'), impact: string(v.impact,'impact'), scenario: string(v.scenario,'scenario'), trace: strings(v.trace,'trace'),
- references: strings(v.references,'references'), recommendation: string(v.recommendation,'recommendation'),
- challenge: { reviewer: string(c.reviewer,'reviewer'), independent: boolean(c.independent,'challenge.independent'), mode:oneOf(c.mode,['independent_agent','sequential_fallback'],'challenge.mode'), callers: string(c.callers,'callers'),
- controls: string(c.controls,'controls'), counterevidence: string(c.counterevidence,'counterevidence'), conclusion: string(c.conclusion,'conclusion') },
- reproduction: 'not_attempted', repair: 'not_attempted', closure: 'open',
+ id: '',
+ fingerprint: '',
+ title: string(v.title, 'title'),
+ rootCause: string(v.rootCause, 'rootCause'),
+ location: { path: snapshotReference(loc.path), line: loc.line, symbol: string(loc.symbol, 'symbol') },
+ advisoryIds: advisoryIdentities(strings(v.advisoryIds ?? [], 'advisoryIds')),
+ severity: oneOf(v.severity, ['critical', 'high', 'medium', 'low', 'informational'], 'severity'),
+ confidence: oneOf(v.confidence, ['high', 'medium', 'low'], 'confidence'),
+ confidenceRationale: string(v.confidenceRationale, 'confidenceRationale'),
+ evidence: oneOf(v.evidence, ['supported', 'hypothesis'], 'evidence'),
+ attackerControl: string(v.attackerControl, 'attackerControl'),
+ impact: string(v.impact, 'impact'),
+ scenario: string(v.scenario, 'scenario'),
+ trace: strings(v.trace, 'trace'),
+ references: strings(v.references, 'references'),
+ recommendation: string(v.recommendation, 'recommendation'),
+ challenge: {
+ reviewer: string(c.reviewer, 'reviewer'),
+ independent: boolean(c.independent, 'challenge.independent'),
+ mode: oneOf(c.mode, ['independent_agent', 'sequential_fallback'], 'challenge.mode'),
+ callers: string(c.callers, 'callers'),
+ controls: string(c.controls, 'controls'),
+ counterevidence: string(c.counterevidence, 'counterevidence'),
+ conclusion: string(c.conclusion, 'conclusion'),
+ },
+ reproduction: 'not_attempted',
+ repair: 'not_attempted',
+ closure: 'open',
};
- if (!Number.isInteger(f.location.line) || f.location.line < 1) throw new CsoError('INVALID_SCHEMA','line must be a positive integer');
- const fallbackLabel='sequential challenge; independent agent unavailable';
- if((f.challenge.independent&&(f.challenge.mode!=='independent_agent'||f.challenge.reviewer===fallbackLabel))||(!f.challenge.independent&&(f.challenge.mode!=='sequential_fallback'||f.challenge.reviewer!==fallbackLabel)))throw new CsoError('INVALID_SCHEMA',`Challenge mode must bind either an independent agent or the exact fallback label: ${fallbackLabel}`);
- if (f.evidence === 'supported' && (f.confidence === 'low' || !f.trace.length || !f.references.length)) throw new CsoError('INVALID_SCHEMA','Supported findings require a challenge, a trace, supporting references, and medium/high confidence');
+ if (!Number.isInteger(f.location.line) || f.location.line < 1)
+ throw new CsoError('INVALID_SCHEMA', 'line must be a positive integer');
+ const fallbackLabel = 'sequential challenge; independent agent unavailable';
+ if (
+ (f.challenge.independent &&
+ (f.challenge.mode !== 'independent_agent' || f.challenge.reviewer === fallbackLabel)) ||
+ (!f.challenge.independent &&
+ (f.challenge.mode !== 'sequential_fallback' || f.challenge.reviewer !== fallbackLabel))
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ `Challenge mode must bind either an independent agent or the exact fallback label: ${fallbackLabel}`,
+ );
+ if (f.evidence === 'supported' && (f.confidence === 'low' || !f.trace.length || !f.references.length))
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Supported findings require a challenge, a trace, supporting references, and medium/high confidence',
+ );
if (v.dependency) {
const d = object(v.dependency);
- exact(d,['affectedVersion','reachability','exposure','exploitation'],'dependency');
- f.dependency = { affectedVersion: string(d.affectedVersion,'affectedVersion'), reachability: oneOf(d.reachability,['reachable','unreachable','unknown'],'reachability'), exposure: string(d.exposure,'exposure'), exploitation: string(d.exploitation,'exploitation') };
+ exact(d, ['affectedVersion', 'reachability', 'exposure', 'exploitation'], 'dependency');
+ f.dependency = {
+ affectedVersion: string(d.affectedVersion, 'affectedVersion'),
+ reachability: oneOf(d.reachability, ['reachable', 'unreachable', 'unknown'], 'reachability'),
+ exposure: string(d.exposure, 'exposure'),
+ exploitation: string(d.exploitation, 'exploitation'),
+ };
}
- f.fingerprint = fingerprint(f); f.id = f.fingerprint; return f;
+ f.fingerprint = fingerprint(f);
+ f.id = f.fingerprint;
+ return f;
}
export function validateCoverage(input: unknown): CoverageRecord {
- const v = object(input,'coverage');
- exact(v,['domain','scope','status','method','gaps','exclusions','evidence','tool'],'coverage');
- const c: CoverageRecord = { domain: string(v.domain,'domain'), scope: string(v.scope,'scope'),
- status: oneOf(v.status,['assessed','partial','not_assessed','not_applicable'],'coverage status'),
- method: string(v.method,'method'), gaps: strings(v.gaps,'gaps'), exclusions: strings(v.exclusions,'exclusions'), evidence: strings(v.evidence,'evidence') };
- if (c.status === 'assessed' && (c.gaps.length || !c.evidence.length)) throw new CsoError('INVALID_SCHEMA','Assessed coverage needs evidence and no outstanding gaps');
- if (c.status === 'partial' && (!c.gaps.length || !c.evidence.length)) throw new CsoError('INVALID_SCHEMA','Partial coverage needs assessed evidence and a concrete gap');
- if (c.status === 'not_assessed' && !c.gaps.length) throw new CsoError('INVALID_SCHEMA','Unassessed coverage needs a concrete gap');
- if (c.status === 'not_applicable' && (!c.evidence.length || c.gaps.length)) throw new CsoError('INVALID_SCHEMA','Non-applicability requires evidence and cannot retain an assessment gap');
- if (v.tool) { const t = object(v.tool);exact(t,['name','version','freshness','outcome'],'coverage tool'); c.tool = {name:string(t.name,'tool name'), version:string(t.version,'tool version'), freshness:string(t.freshness,'freshness'), outcome:string(t.outcome,'outcome')}; }
+ const v = object(input, 'coverage');
+ exact(v, ['domain', 'scope', 'status', 'method', 'gaps', 'exclusions', 'evidence', 'tool'], 'coverage');
+ const c: CoverageRecord = {
+ domain: string(v.domain, 'domain'),
+ scope: string(v.scope, 'scope'),
+ status: oneOf(v.status, ['assessed', 'partial', 'not_assessed', 'not_applicable'], 'coverage status'),
+ method: string(v.method, 'method'),
+ gaps: strings(v.gaps, 'gaps'),
+ exclusions: strings(v.exclusions, 'exclusions'),
+ evidence: strings(v.evidence, 'evidence'),
+ };
+ if (c.status === 'assessed' && (c.gaps.length || !c.evidence.length))
+ throw new CsoError('INVALID_SCHEMA', 'Assessed coverage needs evidence and no outstanding gaps');
+ if (c.status === 'partial' && (!c.gaps.length || !c.evidence.length))
+ throw new CsoError('INVALID_SCHEMA', 'Partial coverage needs assessed evidence and a concrete gap');
+ if (c.status === 'not_assessed' && !c.gaps.length)
+ throw new CsoError('INVALID_SCHEMA', 'Unassessed coverage needs a concrete gap');
+ if (c.status === 'not_applicable' && (!c.evidence.length || c.gaps.length))
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Non-applicability requires evidence and cannot retain an assessment gap',
+ );
+ if (v.tool) {
+ const t = object(v.tool);
+ exact(t, ['name', 'version', 'freshness', 'outcome'], 'coverage tool');
+ c.tool = {
+ name: string(t.name, 'tool name'),
+ version: string(t.version, 'tool version'),
+ freshness: string(t.freshness, 'freshness'),
+ outcome: string(t.outcome, 'outcome'),
+ };
+ }
return c;
}
export function validateCommand(value: unknown, name: string): Command {
- const v=object(value,name), executable=string(v.executable,`${name}.executable`,4096), args=strings(v.args??[],`${name}.args`);
- exact(v,['executable','args'],name);
- if(!executable.startsWith('/')||executable.includes('..'))throw new CsoError('INVALID_SCHEMA',`${name}.executable must be an absolute in-container path`);
- return {executable,args};
+ const v = object(value, name),
+ executable = string(v.executable, `${name}.executable`, 4096),
+ args = strings(v.args ?? [], `${name}.args`);
+ exact(v, ['executable', 'args'], name);
+ if (!executable.startsWith('/') || executable.includes('..'))
+ throw new CsoError('INVALID_SCHEMA', `${name}.executable must be an absolute in-container path`);
+ return { executable, args };
}
-export function validateVerificationObservation(value:unknown):VerificationObservation{
- const v=object(value,'verification observation');
- for(const key of Object.keys(v))if(!['booted','legitimate','security','existingTests','output','inputHash'].includes(key))throw new CsoError('INVALID_SCHEMA',`Unexpected verification observation field: ${key}`);
- if(typeof v.booted!=='boolean'||typeof v.legitimate!=='boolean'||typeof v.existingTests!=='boolean')throw new CsoError('INVALID_SCHEMA','Verification observation outcomes must be booleans');
- if(typeof v.output!=='string'||v.output.length>8192||v.output.includes('\0'))throw new CsoError('INVALID_SCHEMA','Verification observation output must be a bounded string');
- if(typeof v.inputHash!=='string'||(!/^$/.test(v.inputHash)&&!/^[a-f0-9]{64}$/.test(v.inputHash)))throw new CsoError('INVALID_SCHEMA','Verification observation inputHash must be empty or a sha256 hash');
- return{booted:v.booted,legitimate:v.legitimate,security:oneOf(v.security,['pass','intended_failure','inconclusive'],'verification security outcome'),existingTests:v.existingTests,output:v.output,inputHash:v.inputHash};
+export function validateVerificationObservation(value: unknown): VerificationObservation {
+ const v = object(value, 'verification observation');
+ for (const key of Object.keys(v))
+ if (!['booted', 'legitimate', 'security', 'existingTests', 'output', 'inputHash'].includes(key))
+ throw new CsoError('INVALID_SCHEMA', `Unexpected verification observation field: ${key}`);
+ if (
+ typeof v.booted !== 'boolean' ||
+ typeof v.legitimate !== 'boolean' ||
+ typeof v.existingTests !== 'boolean'
+ )
+ throw new CsoError('INVALID_SCHEMA', 'Verification observation outcomes must be booleans');
+ if (typeof v.output !== 'string' || v.output.length > 8192 || v.output.includes('\0'))
+ throw new CsoError('INVALID_SCHEMA', 'Verification observation output must be a bounded string');
+ if (typeof v.inputHash !== 'string' || (!/^$/.test(v.inputHash) && !/^[a-f0-9]{64}$/.test(v.inputHash)))
+ throw new CsoError('INVALID_SCHEMA', 'Verification observation inputHash must be empty or a sha256 hash');
+ return {
+ booted: v.booted,
+ legitimate: v.legitimate,
+ security: oneOf(
+ v.security,
+ ['pass', 'intended_failure', 'inconclusive'],
+ 'verification security outcome',
+ ),
+ existingTests: v.existingTests,
+ output: v.output,
+ inputHash: v.inputHash,
+ };
}
-function assertion(value: unknown,name:string):HttpAssertion {
- const v=object(value,name), expected=object(v.expected,`${name}.expected`);
- exact(v,['name','path','method','headers','body','expected','vulnerable'],name);
- const oracle=(x:Record,n:string)=>{exact(x,['status','includes','excludes'],n);if(!Number.isInteger(x.status)||x.status<100||x.status>599)throw new CsoError('INVALID_SCHEMA',`${n}.status must be an HTTP status`);return {status:x.status,...(x.includes===undefined?{}:{includes:string(x.includes,`${n}.includes`)}),...(x.excludes===undefined?{}:{excludes:string(x.excludes,`${n}.excludes`)})};};
- const path=string(v.path,`${name}.path`,4096);if(!path.startsWith('/')||path.startsWith('//')||/[\r\n]/.test(path))throw new CsoError('INVALID_SCHEMA',`${name}.path must stay on numeric loopback`);
- const headers:Record={};if(v.headers!==undefined)for(const [k,val] of Object.entries(object(v.headers,`${name}.headers`))){if(!/^[A-Za-z0-9-]{1,100}$/.test(k)||typeof val!=='string'||val.length>8192||/[\r\n]/.test(val))throw new CsoError('INVALID_SCHEMA',`Invalid ${name} header`);headers[k]=val;}
- return {name:string(v.name,`${name}.name`),path,method:oneOf(v.method,['GET','POST','PUT','PATCH','DELETE'],`${name}.method`),...(Object.keys(headers).length?{headers}:{}),...(v.body===undefined?{}:{body:string(v.body,`${name}.body`,65536)}),expected:oracle(expected,`${name}.expected`),...(v.vulnerable===undefined?{}:{vulnerable:oracle(object(v.vulnerable),`${name}.vulnerable`)})};
+function assertion(value: unknown, name: string): HttpAssertion {
+ const v = object(value, name),
+ expected = object(v.expected, `${name}.expected`);
+ exact(v, ['name', 'path', 'method', 'headers', 'body', 'expected', 'vulnerable'], name);
+ const oracle = (x: Record, n: string) => {
+ exact(x, ['status', 'includes', 'excludes'], n);
+ if (!Number.isInteger(x.status) || x.status < 100 || x.status > 599)
+ throw new CsoError('INVALID_SCHEMA', `${n}.status must be an HTTP status`);
+ return {
+ status: x.status,
+ ...(x.includes === undefined ? {} : { includes: string(x.includes, `${n}.includes`) }),
+ ...(x.excludes === undefined ? {} : { excludes: string(x.excludes, `${n}.excludes`) }),
+ };
+ };
+ const path = string(v.path, `${name}.path`, 4096);
+ if (!path.startsWith('/') || path.startsWith('//') || /[\r\n]/.test(path))
+ throw new CsoError('INVALID_SCHEMA', `${name}.path must stay on numeric loopback`);
+ const headers: Record = {};
+ if (v.headers !== undefined)
+ for (const [k, val] of Object.entries(object(v.headers, `${name}.headers`))) {
+ if (
+ !/^[A-Za-z0-9-]{1,100}$/.test(k) ||
+ typeof val !== 'string' ||
+ val.length > 8192 ||
+ /[\r\n]/.test(val)
+ )
+ throw new CsoError('INVALID_SCHEMA', `Invalid ${name} header`);
+ headers[k] = val;
+ }
+ return {
+ name: string(v.name, `${name}.name`),
+ path,
+ method: oneOf(v.method, ['GET', 'POST', 'PUT', 'PATCH', 'DELETE'], `${name}.method`),
+ ...(Object.keys(headers).length ? { headers } : {}),
+ ...(v.body === undefined ? {} : { body: string(v.body, `${name}.body`, 65536) }),
+ expected: oracle(expected, `${name}.expected`),
+ ...(v.vulnerable === undefined ? {} : { vulnerable: oracle(object(v.vulnerable), `${name}.vulnerable`) }),
+ };
}
-export function validateVerificationRequest(input:unknown):VerificationRequest {
- const v=object(input,'verification request'),changes=v.changes,fixtures=object(v.fixtures??{},'fixtures'),review=object(v.review,'review');
- exact(v,['findingId','runtimeProfile','port','start','legitimate','security','existingTests','fixtures','boundaryFiles','testFiles','changes','review'],'verification request');
- exact(review,['reviewer','independent','rootCauseRepaired','featurePreserved','boundaryMocks','rationale','reviewedPatchHash','artifactId'],'review');
- if(!Array.isArray(changes)||!changes.length||changes.length>100)throw new CsoError('INVALID_SCHEMA','changes must contain 1..100 declared patch effects');
- const cleanFixtures:Record={};for(const [p,body] of Object.entries(fixtures)){cleanFixtures[relativePath(p)]=string(body,`fixture ${p}`,1024*1024);}
- const request:VerificationRequest={findingId:string(v.findingId,'findingId'),runtimeProfile:string(v.runtimeProfile,'runtimeProfile',100),port:v.port,
- start:validateCommand(v.start,'start'),legitimate:(Array.isArray(v.legitimate)?v.legitimate:[]).map((x,i)=>assertion(x,`legitimate[${i}]`)),security:assertion(v.security,'security'),existingTests:(Array.isArray(v.existingTests)?v.existingTests:[]).map((x,i)=>validateCommand(x,`existingTests[${i}]`)),fixtures:cleanFixtures,boundaryFiles:strings(v.boundaryFiles,'boundaryFiles').map(snapshotReference),testFiles:strings(v.testFiles,'testFiles').map(snapshotReference),
- changes:changes.map((raw:any,i:number)=>{const x=object(raw,`changes[${i}]`),before=x.beforeSha256;exact(x,['path','beforeSha256','after','effect'],`changes[${i}]`);if(before!==null&&(typeof before!=='string'||!/^[a-f0-9]{64}$/.test(before)))throw new CsoError('INVALID_SCHEMA',`changes[${i}].beforeSha256 must be a hash or null`);return{path:snapshotReference(x.path),beforeSha256:before,after:x.after===null?null:string(x.after,`changes[${i}].after`,1024*1024),effect:oneOf(x.effect,['source','configuration','dependency'],`changes[${i}].effect`)};}),
- review:{reviewer:string(review.reviewer,'reviewer'),independent:boolean(review.independent,'review.independent'),rootCauseRepaired:boolean(review.rootCauseRepaired,'review.rootCauseRepaired'),featurePreserved:boolean(review.featurePreserved,'review.featurePreserved'),boundaryMocks:boolean(review.boundaryMocks,'review.boundaryMocks'),rationale:string(review.rationale,'review rationale'),reviewedPatchHash:string(review.reviewedPatchHash,'reviewedPatchHash'),...(review.artifactId===undefined?{}:{artifactId:string(review.artifactId,'review artifact ID')})},};
- if(!/^[a-f0-9]{32}$/.test(request.findingId))throw new CsoError('INVALID_SCHEMA','findingId must be a helper-issued identifier');
- if(request.review.artifactId!==undefined&&!/^[a-f0-9]{32}$/.test(request.review.artifactId))throw new CsoError('INVALID_SCHEMA','review artifact ID must be a helper-issued identifier');
- if(!Number.isInteger(request.port)||request.port<1024||request.port>65535)throw new CsoError('INVALID_SCHEMA','port must be 1024..65535');
- if(!request.legitimate.length||!request.security.vulnerable||!request.existingTests.length||!request.boundaryFiles.length||!request.testFiles.length)throw new CsoError('INVALID_SCHEMA','Verification needs a legitimate control, distinct before/fixed security oracles, existing tests, immutable test files, and boundary files');
- const secure=request.security.expected,vulnerable=request.security.vulnerable;
- const mutuallyExclusive=secure.status!==vulnerable.status
- ||(secure.includes!==undefined&&vulnerable.excludes!==undefined&&secure.includes.includes(vulnerable.excludes))
- ||(vulnerable.includes!==undefined&&secure.excludes!==undefined&&vulnerable.includes.includes(secure.excludes));
- if(!mutuallyExclusive)throw new CsoError('INVALID_SCHEMA','The vulnerable and fixed security oracles must be provably mutually exclusive');
- if(new Set(request.changes.map(x=>x.path)).size!==request.changes.length)throw new CsoError('INVALID_SCHEMA','Patch paths must be unique');
- if(new Set(request.testFiles).size!==request.testFiles.length||request.changes.some(change=>request.testFiles.includes(change.path)))throw new CsoError('INVALID_SCHEMA','Existing-test source files must be unique and unchanged by the repair');
- if(request.existingTests.some(command=>/(?:^|\/)(?:true|false|echo|printf|env|sh|bash)$/.test(command.executable)))throw new CsoError('INVALID_SCHEMA','Generic success or shell commands cannot stand in for a project test suite');
- if(!request.changes.some(change=>change.beforeSha256===null||change.after===null||sha256(change.after)!==change.beforeSha256))throw new CsoError('INVALID_SCHEMA','A tested repair must contain at least one material patch effect');
+export function validateVerificationRequest(input: unknown): VerificationRequest {
+ const v = object(input, 'verification request'),
+ changes = v.changes,
+ fixtures = object(v.fixtures ?? {}, 'fixtures'),
+ review = object(v.review, 'review');
+ exact(
+ v,
+ [
+ 'findingId',
+ 'runtimeProfile',
+ 'port',
+ 'start',
+ 'legitimate',
+ 'security',
+ 'existingTests',
+ 'fixtures',
+ 'boundaryFiles',
+ 'testFiles',
+ 'changes',
+ 'review',
+ ],
+ 'verification request',
+ );
+ exact(
+ review,
+ [
+ 'reviewer',
+ 'independent',
+ 'rootCauseRepaired',
+ 'featurePreserved',
+ 'boundaryMocks',
+ 'rationale',
+ 'reviewedPatchHash',
+ 'artifactId',
+ ],
+ 'review',
+ );
+ if (!Array.isArray(changes) || !changes.length || changes.length > 100)
+ throw new CsoError('INVALID_SCHEMA', 'changes must contain 1..100 declared patch effects');
+ const cleanFixtures: Record = {};
+ for (const [p, body] of Object.entries(fixtures)) {
+ cleanFixtures[relativePath(p)] = string(body, `fixture ${p}`, 1024 * 1024);
+ }
+ const request: VerificationRequest = {
+ findingId: string(v.findingId, 'findingId'),
+ runtimeProfile: string(v.runtimeProfile, 'runtimeProfile', 100),
+ port: v.port,
+ start: validateCommand(v.start, 'start'),
+ legitimate: (Array.isArray(v.legitimate) ? v.legitimate : []).map((x, i) =>
+ assertion(x, `legitimate[${i}]`),
+ ),
+ security: assertion(v.security, 'security'),
+ existingTests: (Array.isArray(v.existingTests) ? v.existingTests : []).map((x, i) =>
+ validateCommand(x, `existingTests[${i}]`),
+ ),
+ fixtures: cleanFixtures,
+ boundaryFiles: strings(v.boundaryFiles, 'boundaryFiles').map(snapshotReference),
+ testFiles: strings(v.testFiles, 'testFiles').map(snapshotReference),
+ changes: changes.map((raw: any, i: number) => {
+ const x = object(raw, `changes[${i}]`),
+ before = x.beforeSha256;
+ exact(x, ['path', 'beforeSha256', 'after', 'effect'], `changes[${i}]`);
+ if (before !== null && (typeof before !== 'string' || !/^[a-f0-9]{64}$/.test(before)))
+ throw new CsoError('INVALID_SCHEMA', `changes[${i}].beforeSha256 must be a hash or null`);
+ return {
+ path: snapshotReference(x.path),
+ beforeSha256: before,
+ after: x.after === null ? null : string(x.after, `changes[${i}].after`, 1024 * 1024),
+ effect: oneOf(x.effect, ['source', 'configuration', 'dependency'], `changes[${i}].effect`),
+ };
+ }),
+ review: {
+ reviewer: string(review.reviewer, 'reviewer'),
+ independent: boolean(review.independent, 'review.independent'),
+ rootCauseRepaired: boolean(review.rootCauseRepaired, 'review.rootCauseRepaired'),
+ featurePreserved: boolean(review.featurePreserved, 'review.featurePreserved'),
+ boundaryMocks: boolean(review.boundaryMocks, 'review.boundaryMocks'),
+ rationale: string(review.rationale, 'review rationale'),
+ reviewedPatchHash: string(review.reviewedPatchHash, 'reviewedPatchHash'),
+ ...(review.artifactId === undefined
+ ? {}
+ : { artifactId: string(review.artifactId, 'review artifact ID') }),
+ },
+ };
+ if (!/^[a-f0-9]{32}$/.test(request.findingId))
+ throw new CsoError('INVALID_SCHEMA', 'findingId must be a helper-issued identifier');
+ if (request.review.artifactId !== undefined && !/^[a-f0-9]{32}$/.test(request.review.artifactId))
+ throw new CsoError('INVALID_SCHEMA', 'review artifact ID must be a helper-issued identifier');
+ if (!Number.isInteger(request.port) || request.port < 1024 || request.port > 65535)
+ throw new CsoError('INVALID_SCHEMA', 'port must be 1024..65535');
+ if (
+ !request.legitimate.length ||
+ !request.security.vulnerable ||
+ !request.existingTests.length ||
+ !request.boundaryFiles.length ||
+ !request.testFiles.length
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Verification needs a legitimate control, distinct before/fixed security oracles, existing tests, immutable test files, and boundary files',
+ );
+ const secure = request.security.expected,
+ vulnerable = request.security.vulnerable;
+ const mutuallyExclusive =
+ secure.status !== vulnerable.status ||
+ (secure.includes !== undefined &&
+ vulnerable.excludes !== undefined &&
+ secure.includes.includes(vulnerable.excludes)) ||
+ (vulnerable.includes !== undefined &&
+ secure.excludes !== undefined &&
+ vulnerable.includes.includes(secure.excludes));
+ if (!mutuallyExclusive)
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'The vulnerable and fixed security oracles must be provably mutually exclusive',
+ );
+ if (new Set(request.changes.map((x) => x.path)).size !== request.changes.length)
+ throw new CsoError('INVALID_SCHEMA', 'Patch paths must be unique');
+ if (
+ new Set(request.testFiles).size !== request.testFiles.length ||
+ request.changes.some((change) => request.testFiles.includes(change.path))
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Existing-test source files must be unique and unchanged by the repair',
+ );
+ if (
+ request.existingTests.some((command) =>
+ /(?:^|\/)(?:true|false|echo|printf|env|sh|bash)$/.test(command.executable),
+ )
+ )
+ throw new CsoError(
+ 'INVALID_SCHEMA',
+ 'Generic success or shell commands cannot stand in for a project test suite',
+ );
+ if (
+ !request.changes.some(
+ (change) =>
+ change.beforeSha256 === null || change.after === null || sha256(change.after) !== change.beforeSha256,
+ )
+ )
+ throw new CsoError('INVALID_SCHEMA', 'A tested repair must contain at least one material patch effect');
return request;
}
-export function completeness(report: Pick): Completeness {
+export function completeness(report: Pick): Completeness {
// Scanner adapters preserve operational outcomes, but scanner output is only
// candidate evidence. The corresponding investigation domain decides whether
// assessment work remains; an optional tool failure cannot override it.
// A successful snapshot is a prerequisite, not security assessment work by
// itself. Its helper-owned partial/not-assessed state remains material.
- const work = report.coverage.filter(c => c.status !== 'not_applicable'&&!c.domain.startsWith('scanner:')&&!(['snapshot-inputs','history-inputs'].includes(c.domain)&&c.status==='assessed'));
- if (!report.gaps.length && work.length && work.every(c => c.status === 'assessed')) return 'complete';
- return work.some(c => c.status === 'assessed' || c.status === 'partial') ? 'partial' : 'not assessed';
+ const work = report.coverage.filter(
+ (c) =>
+ c.status !== 'not_applicable' &&
+ !c.domain.startsWith('scanner:') &&
+ !(['snapshot-inputs', 'history-inputs'].includes(c.domain) && c.status === 'assessed'),
+ );
+ if (!report.gaps.length && work.length && work.every((c) => c.status === 'assessed')) return 'complete';
+ return work.some((c) => c.status === 'assessed' || c.status === 'partial') ? 'partial' : 'not assessed';
}
export function renderReport(report: RunReportV3): string {
- const supported = report.findings.filter(f => f.evidence === 'supported');
- const gaps = [...new Set([...report.gaps,...report.coverage.filter(c=>!c.domain.startsWith('scanner:')).flatMap(c => c.gaps)])];
- const transformations=report.source.transformations??[];
+ const supported = report.findings.filter((f) => f.evidence === 'supported');
+ const gaps = [
+ ...new Set([
+ ...report.gaps,
+ ...report.coverage.filter((c) => !c.domain.startsWith('scanner:')).flatMap((c) => c.gaps),
+ ]),
+ ];
+ const transformations = report.source.transformations ?? [];
// Report JSON is canonical evidence. Markdown is a safe plain-text view:
// collapse line breaks and escape all Markdown control characters so model,
// repository, scanner, and advisory strings cannot forge report structure.
- const plain=(value:unknown):string=>String(value).replace(/[\x00-\x1f\x7f-\x9f\u061c\u200e\u200f\u2028\u2029\u202a-\u202e\u2066-\u2069]+/gu,' ').replace(/\s{2,}/g,' ').trim().replace(/[\\`*_[\]{}()#+!|<>]/g,'\\$&');
- const list=(values:string[]):string=>values.length?values.map(plain).join('; '):'none';
- const terminal=[...report.events].reverse().find(item=>item.kind==='terminal'),startedAt=Date.parse(report.createdAt),terminalAt=terminal?Date.parse(terminal.at):NaN;
- const elapsed=Number.isFinite(startedAt)&&Number.isFinite(terminalAt)&&terminalAt>=startedAt?`; elapsed ${terminalAt-startedAt} ms`:'';
- const timing=`Timing: started ${plain(report.createdAt)}; deadline ${plain(report.deadline)}${terminal?`; terminal ${plain(terminal.at)}${elapsed}`:''}.`;
- const usage=report.modelUsage?`Model usage: ${report.modelUsage.tokens} host-reported tokens from ${plain(report.modelUsage.source)}${report.modelUsage.cost===undefined?'':`; host-reported cost ${report.modelUsage.cost}`}.`:undefined;
- const findingLines=(f:FindingV3):string[]=>[
+ const plain = (value: unknown): string =>
+ String(value)
+ .replace(/[\x00-\x1f\x7f-\x9f\u061c\u200e\u200f\u2028\u2029\u202a-\u202e\u2066-\u2069]+/gu, ' ')
+ .replace(/\s{2,}/g, ' ')
+ .trim()
+ .replace(/[\\`*_[\]{}()#+!|<>]/g, '\\$&');
+ const list = (values: string[]): string => (values.length ? values.map(plain).join('; ') : 'none');
+ const terminal = [...report.events].reverse().find((item) => item.kind === 'terminal'),
+ startedAt = Date.parse(report.createdAt),
+ terminalAt = terminal ? Date.parse(terminal.at) : NaN;
+ const elapsed =
+ Number.isFinite(startedAt) && Number.isFinite(terminalAt) && terminalAt >= startedAt
+ ? `; elapsed ${terminalAt - startedAt} ms`
+ : '';
+ const timing = `Timing: started ${plain(report.createdAt)}; deadline ${plain(report.deadline)}${terminal ? `; terminal ${plain(terminal.at)}${elapsed}` : ''}.`;
+ const usage = report.modelUsage
+ ? `Model usage: ${report.modelUsage.tokens} host-reported tokens from ${plain(report.modelUsage.source)}${report.modelUsage.cost === undefined ? '' : `; host-reported cost ${report.modelUsage.cost}`}.`
+ : undefined;
+ const findingLines = (f: FindingV3): string[] => [
`- ${plain(f.severity.toUpperCase())} ${plain(f.title)} [${plain(f.id)}]`,
` Location: ${plain(f.location.path)}:${f.location.line} (${plain(f.location.symbol)}). Confidence: ${plain(f.confidence)} — ${plain(f.confidenceRationale)}. Evidence: ${plain(f.evidence)}.`,
` Attacker scenario: ${plain(f.scenario)}`,
@@ -325,57 +875,127 @@ export function renderReport(report: RunReportV3): string {
` Trace: ${list(f.trace)}. Supporting references: ${list(f.references)}.`,
` Counterevidence considered: ${plain(f.challenge.counterevidence)}. Challenge: ${plain(f.challenge.mode)} by ${plain(f.challenge.reviewer)}. Conclusion: ${plain(f.challenge.conclusion)}.`,
` Repair recommendation: ${plain(f.recommendation)}`,
- ` Reproduction: ${plain(f.reproduction)}${f.reproductionAttemptId?` (attempt ${plain(f.reproductionAttemptId)})`:''}. Repair: ${plain(f.repair)}. Closure: ${plain(f.closure)}.`,
- ...(f.verificationId?[` Verification: ${plain(f.verificationId)}. Bundle: bundles/${plain(f.verificationId)}.json. Assertion assurance: ${plain(f.verificationAssurance?.assertions??'unknown')}. Test completion assurance: ${plain(f.verificationAssurance?.testCompletion??'unknown')}. Review assurance: ${plain(f.verificationAssurance?.review??'unknown')}.`]:[]),
+ ` Reproduction: ${plain(f.reproduction)}${f.reproductionAttemptId ? ` (attempt ${plain(f.reproductionAttemptId)})` : ''}. Repair: ${plain(f.repair)}. Closure: ${plain(f.closure)}.`,
+ ...(f.verificationId
+ ? [
+ ` Verification: ${plain(f.verificationId)}. Bundle: bundles/${plain(f.verificationId)}.json. Assertion assurance: ${plain(f.verificationAssurance?.assertions ?? 'unknown')}. Test completion assurance: ${plain(f.verificationAssurance?.testCompletion ?? 'unknown')}. Review assurance: ${plain(f.verificationAssurance?.review ?? 'unknown')}.`,
+ ]
+ : []),
];
- const model=report.application;
- return [ `${report.completeness} — ${plain(report.policy.scope)}${report.policy.diff ? ` (diff against ${plain(report.policy.base)})` : ''}`,
+ const model = report.application;
+ return [
+ `${report.completeness} — ${plain(report.policy.scope)}${report.policy.diff ? ` (diff against ${plain(report.policy.base)})` : ''}`,
`Run: ${plain(report.runId)}. Mode: ${plain(report.policy.mode)}.`,
- timing, ...(usage?[usage]:[]),
- `Material gaps: ${gaps.length ? list(gaps) : 'none reported'}.`, '',
+ timing,
+ ...(usage ? [usage] : []),
+ `Material gaps: ${gaps.length ? list(gaps) : 'none reported'}.`,
+ '',
'Application model:',
- `- Actors: ${list(model.actors)}.`, `- Assets: ${list(model.assets)}.`, `- Entrypoints: ${list(model.entrypoints)}.`,
- `- Tenant boundaries: ${list(model.tenantBoundaries)}.`, `- Sensitive operations: ${list(model.sensitiveOperations)}.`, `- Security invariants: ${list(model.invariants)}.`, '',
- ...(supported.length ? ['Supported findings:',...supported.flatMap(findingLines)] : ['No supported findings in the assessed scope.']),
- ...(report.policy.mode === 'comprehensive' ? ['', 'Hypotheses (unconfirmed):', ...report.findings.filter(f => f.evidence === 'hypothesis').flatMap(findingLines)] : []),
- '', 'Snapshot transformations:', ...(transformations.length?transformations.map(item=>`- ${plain(item.path)}: ${plain(item.handling)}`):['- none']),
- '', 'Coverage:', ...report.coverage.flatMap(c=>[
- `- ${plain(c.domain)}: ${plain(c.status)}; ${plain(c.method)}${c.tool?`; tool ${plain(c.tool.name)} ${plain(c.tool.version)}, freshness ${plain(c.tool.freshness)}, outcome ${plain(c.tool.outcome)}`:''}.`,
+ `- Actors: ${list(model.actors)}.`,
+ `- Assets: ${list(model.assets)}.`,
+ `- Entrypoints: ${list(model.entrypoints)}.`,
+ `- Tenant boundaries: ${list(model.tenantBoundaries)}.`,
+ `- Sensitive operations: ${list(model.sensitiveOperations)}.`,
+ `- Security invariants: ${list(model.invariants)}.`,
+ '',
+ ...(supported.length
+ ? ['Supported findings:', ...supported.flatMap(findingLines)]
+ : ['No supported findings in the assessed scope.']),
+ ...(report.policy.mode === 'comprehensive'
+ ? [
+ '',
+ 'Hypotheses (unconfirmed):',
+ ...report.findings.filter((f) => f.evidence === 'hypothesis').flatMap(findingLines),
+ ]
+ : []),
+ '',
+ 'Snapshot transformations:',
+ ...(transformations.length
+ ? transformations.map((item) => `- ${plain(item.path)}: ${plain(item.handling)}`)
+ : ['- none']),
+ '',
+ 'Coverage:',
+ ...report.coverage.flatMap((c) => [
+ `- ${plain(c.domain)}: ${plain(c.status)}; ${plain(c.method)}${c.tool ? `; tool ${plain(c.tool.name)} ${plain(c.tool.version)}, freshness ${plain(c.tool.freshness)}, outcome ${plain(c.tool.outcome)}` : ''}.`,
` Scope: ${plain(c.scope)}. Evidence: ${list(c.evidence)}. Gaps: ${list(c.gaps)}. Exclusions: ${list(c.exclusions)}.`,
- ]), '',
+ ]),
+ '',
].join('\n');
}
-type LegacyJson = null | boolean | number | string | LegacyJson[] | { [key:string]: LegacyJson };
-function legacyJson(value:unknown,depth=0,seen=new WeakSet]