fix(evals): cut path variance at its measured sources

- gstack-qa-evidence capture prints startedAt/completedAt/durationMs and, for
  --deadline captures, remainingMs; the functional report takes durations from
  them. The section clock notice asks for one clock read up front instead of one
  after every checkpoint (QA runs spent 7-14% of tool calls on date -u).
- ship plan-completion: skip the audit dispatch when discovery already found no
  plan (the dispatch-vs-skip conflict produced an optional 60-100 s subagent).
- materialize/checkpoint validation errors state the expected schema, so a
  rejected annotations file is fixable in one call instead of blocking the phase.
- session-runner counts turns from the transcript when a run times out, so
  timeouts stop reporting 'turn 0'.
This commit is contained in:
garrytan committed 2026-09-30 15:59:58 +00:00
1 parent a8a32e3216
commit 79092b22ea
8 files changed
+36 -22

No files matched your search

+10 -6
View File
@@ -2,7 +2,7 @@ import * as fs from 'node:fs';
import * as path from 'node:path'; import * as path from 'node:path';
import { createHash } from 'node:crypto'; import { createHash } from 'node:crypto';
import { atomicWriteSync } from './fs-atomic'; import { atomicWriteSync } from './fs-atomic';
import { runQaDeadlineCommand, runQaWindowsWorker, startQaDeadline, withQaReceiptOutput } from './qa-deadline'; import { qaDeadlineStatus, readQaDeadline, runQaDeadlineCommand, runQaWindowsWorker, startQaDeadline, withQaReceiptOutput } from './qa-deadline';
import { scan } from './redact-engine'; import { scan } from './redact-engine';
const object = (value: unknown): value is Record<string, any> => value !== null && typeof value === 'object' && !Array.isArray(value); const object = (value: unknown): value is Record<string, any> => value !== null && typeof value === 'object' && !Array.isArray(value);
@@ -174,11 +174,15 @@ async function capture(root: string, captureId: string, publicOutput: boolean, o
fs.writeFileSync(owned(root, `.qa-evidence/${captureId}/observation.json`), bytes, { flag: 'wx', mode: 0o600 }); fs.writeFileSync(owned(root, `.qa-evidence/${captureId}/observation.json`), bytes, { flag: 'wx', mode: 0o600 });
observation = { sha256: hash(bytes), bytes: Buffer.byteLength(bytes) }; observation = { sha256: hash(bytes), bytes: Buffer.byteLength(bytes) };
} }
const completedAt = new Date().toISOString();
let remainingMs: number | undefined;
if (option === '--deadline') try { remainingMs = qaDeadlineStatus(readQaDeadline(deadline)).remainingMs; } catch {}
const receipt = { version: 1, id: captureId, cwd: process.cwd(), argv: [command, ...args], deadline, timing, startedAt, const receipt = { version: 1, id: captureId, cwd: process.cwd(), argv: [command, ...args], deadline, timing, startedAt,
completedAt: new Date().toISOString(), exitCode, signal: result.signal, status, observation, publicOutput, completedAt, exitCode, signal: result.signal, status, observation, publicOutput,
...streams }; ...streams };
const sha256 = publish(root, `.qa-evidence/${captureId}/receipt.json`, receipt); const sha256 = publish(root, `.qa-evidence/${captureId}/receipt.json`, receipt);
return { action: 'capture', id: captureId, status, sha256, exitCode, signal: result.signal, publicOutput }; return { action: 'capture', id: captureId, status, sha256, exitCode, signal: result.signal, publicOutput,
startedAt, completedAt, durationMs: Date.parse(completedAt) - Date.parse(startedAt), ...(remainingMs === undefined ? {} : { remainingMs }) };
} }
function checkpoint(root: string, checkpointId: string, source: string | Record<string, string>) { function checkpoint(root: string, checkpointId: string, source: string | Record<string, string>) {
@@ -188,7 +192,7 @@ function checkpoint(root: string, checkpointId: string, source: string | Record<
if (!exact(intent, ['capture', 'observationCommand', 'hypothesis', 'nextCommand']) if (!exact(intent, ['capture', 'observationCommand', 'hypothesis', 'nextCommand'])
|| typeof intent.capture !== 'string' || typeof intent.observationCommand !== 'string' || !intent.observationCommand.trim() || typeof intent.capture !== 'string' || typeof intent.observationCommand !== 'string' || !intent.observationCommand.trim()
|| typeof intent.hypothesis !== 'string' || intent.hypothesis.trim().length <= 20 || !/[a-z]{3}/i.test(intent.hypothesis) || typeof intent.hypothesis !== 'string' || intent.hypothesis.trim().length <= 20 || !/[a-z]{3}/i.test(intent.hypothesis)
|| typeof intent.nextCommand !== 'string' || !intent.nextCommand.trim()) throw new QaEvidenceError('Invalid causal intent'); || typeof intent.nextCommand !== 'string' || !intent.nextCommand.trim()) throw new QaEvidenceError('Invalid causal intent: need exactly capture, observationCommand, hypothesis (one sentence over 20 characters) and nextCommand');
if (scan(decode(bytes)).findings.some(finding => finding.tier === 'HIGH')) throw new QaEvidenceError('Sensitive intent cannot be published'); if (scan(decode(bytes)).findings.some(finding => finding.tier === 'HIGH')) throw new QaEvidenceError('Sensitive intent cannot be published');
const captured = readQaCapture(root, intent.capture); const captured = readQaCapture(root, intent.capture);
const value = { observationCommand: intent.observationCommand, observed: captured.observed, hypothesis: intent.hypothesis, nextCommand: intent.nextCommand }; const value = { observationCommand: intent.observationCommand, observed: captured.observed, hypothesis: intent.hypothesis, nextCommand: intent.nextCommand };
@@ -204,11 +208,11 @@ function materialize(root: string, source: string) {
if (!exact(annotations, ['revision', 'runtime', 'cwd', 'limits', 'evidence', 'learning']) if (!exact(annotations, ['revision', 'runtime', 'cwd', 'limits', 'evidence', 'learning'])
|| !['revision', 'runtime', 'cwd'].every(key => typeof annotations[key] === 'string' && annotations[key].trim()) || !['revision', 'runtime', 'cwd'].every(key => typeof annotations[key] === 'string' && annotations[key].trim())
|| !Array.isArray(annotations.limits) || !annotations.limits.length || !annotations.limits.every((limit: unknown) => typeof limit === 'string' && limit.trim()) || !Array.isArray(annotations.limits) || !annotations.limits.length || !annotations.limits.every((limit: unknown) => typeof limit === 'string' && limit.trim())
|| !Array.isArray(annotations.evidence) || !Array.isArray(annotations.learning)) throw new QaEvidenceError('Invalid report annotations'); || !Array.isArray(annotations.evidence) || !Array.isArray(annotations.learning)) throw new QaEvidenceError('Invalid report annotations: need exactly revision, runtime and cwd (non-empty strings), limits (non-empty string array), evidence (row array) and learning (checkpoint ID array)');
const captures = new Set<string>(); const captures = new Set<string>();
const evidence = annotations.evidence.map((row: any) => { const evidence = annotations.evidence.map((row: any) => {
if (!exact(row, ['capture', 'command', 'contract', 'expected', 'classification']) if (!exact(row, ['capture', 'command', 'contract', 'expected', 'classification'])
|| !Object.values(row).every(value => typeof value === 'string' && value.trim()) || captures.has(row.capture)) throw new QaEvidenceError('Invalid evidence annotation'); || !Object.values(row).every(value => typeof value === 'string' && value.trim()) || captures.has(row.capture)) throw new QaEvidenceError('Invalid evidence annotation: each row needs exactly capture, command, contract, expected and classification as non-empty strings, with a unique capture');
captures.add(row.capture); captures.add(row.capture);
const captured = readQaCapture(root, row.capture); const captured = readQaCapture(root, row.capture);
return { command: row.command, contract: row.contract, expected: row.expected, classification: row.classification, observed: captured.observed }; return { command: row.command, contract: row.contract, expected: row.expected, classification: row.classification, observed: captured.observed };
+1 -1
View File
@@ -7,7 +7,7 @@
| Surfaces / scope | {API, CLI, job, worker, webhook; changed and adjacent contracts} | | Surfaces / scope | {API, CLI, job, worker, webhook; changed and adjacent contracts} |
| Runtime / native tools | {VERSIONS AND REPOSITORY-SUPPORTED COMMANDS} | | Runtime / native tools | {VERSIONS AND REPOSITORY-SUPPORTED COMMANDS} |
| Fixture ownership / destinations | {ISOLATED ROOT, STORES, DOWNSTREAM TARGETS} | | Fixture ownership / destinations | {ISOLATED ROOT, STORES, DOWNSTREAM TARGETS} |
| Duration / stop reason | {MEASURED COMMAND DURATIONS, COMPLETE OR BOUND/BLOCKER} | | Duration / stop reason | {CAPTURE durationMs TOTALS, COMPLETE OR BOUND/BLOCKER} |
## Contract outcomes ## Contract outcomes
+3 -3
View File
@@ -14,9 +14,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below. code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path **Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists, or complete text, including user-approved scope changes. If none is known, say
say so explicitly and let the child use the fallback search below. The child does so; the child runs the fallback search below. If discovery found no plan, skip
not inherit the parent's conversation. dispatch. The child does not inherit the parent's conversation.
````text ````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report. You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -3
View File
@@ -12,9 +12,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below. code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path **Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists, or complete text, including user-approved scope changes. If none is known, say
say so explicitly and let the child use the fallback search below. The child does so; the child runs the fallback search below. If discovery found no plan, skip
not inherit the parent's conversation. dispatch. The child does not inherit the parent's conversation.
````text ````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report. You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -3
View File
@@ -1556,9 +1556,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below. code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path **Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists, or complete text, including user-approved scope changes. If none is known, say
say so explicitly and let the child use the fallback search below. The child does so; the child runs the fallback search below. If discovery found no plan, skip
not inherit the parent's conversation. dispatch. The child does not inherit the parent's conversation.
````text ````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report. You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -3
View File
@@ -1536,9 +1536,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below. code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path **Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists, or complete text, including user-approved scope changes. If none is known, say
say so explicitly and let the child use the fallback search below. The child does so; the child runs the fallback search below. If discovery found no plan, skip
not inherit the parent's conversation. dispatch. The child does not inherit the parent's conversation.
````text ````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report. You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -2
View File
@@ -293,7 +293,7 @@ Runner entry UTC: ${new Date(startTime).toISOString()}
Hard deadline UTC: ${new Date(deadline).toISOString()} Hard deadline UTC: ${new Date(deadline).toISOString()}
Completion reserve starts UTC: ${new Date(deadline - reserve).toISOString()} Completion reserve starts UTC: ${new Date(deadline - reserve).toISOString()}
Setup, CLI startup and API queueing consume this same window; it never resets. Setup, CLI startup and API queueing consume this same window; it never resets.
Before source Reads and after each saved checkpoint, use Bash to run exactly \`date -u +%Y-%m-%dT%H:%M:%SZ\`. Compare that observed UTC time with the times above. When remaining time is at most ${reserve / 1000} seconds, prioritize the remaining required completion outputs and verification. No required content or gate may be skipped. If the clock read fails, report timing unavailable; do not invent remaining time or restart the deadline.`; Before source Reads, use Bash to run exactly \`date -u +%Y-%m-%dT%H:%M:%SZ\`. After each saved checkpoint, compare the latest evidence capture's printed completedAt with the times above; run that clock read again only when no capture has completed since your last clock read. When remaining time is at most ${reserve / 1000} seconds, prioritize the remaining required completion outputs and verification. No required content or gate may be skipped. If the clock read fails, report timing unavailable; do not invent remaining time or restart the deadline.`;
systemPrompt = systemPrompt ? `${systemPrompt}\n\n${notice}` : notice; systemPrompt = systemPrompt ? `${systemPrompt}\n\n${notice}` : notice;
} }
@@ -696,7 +696,8 @@ Before source Reads and after each saved checkpoint, use Bash to run exactly \`d
} }
// Cost from result line (exact) or estimate from chars // Cost from result line (exact) or estimate from chars
const turnsUsed = resultLine?.num_turns || 0; const turnsUsed = resultLine?.num_turns
|| new Set(transcript.filter(event => event.type === 'assistant' && !event.parent_tool_use_id).map(event => event.message?.id)).size;
const estimatedCost = resultLine?.total_cost_usd || 0; const estimatedCost = resultLine?.total_cost_usd || 0;
const inputChars = prompt.length; const inputChars = prompt.length;
const outputChars = (resultLine?.result || '').length; const outputChars = (resultLine?.result || '').length;
+10 -1
View File
@@ -35,6 +35,8 @@ test('native capture executes once, preserves exact JSON and stderr, and materia
expect(result.status, result.stderr).toBe(0); expect(result.status, result.stderr).toBe(0);
const captured = receipt(result.stdout); const captured = receipt(result.stdout);
expect(captured).toMatchObject({ action: 'capture', status: 'complete', id: '001', exitCode: 0 }); expect(captured).toMatchObject({ action: 'capture', status: 'complete', id: '001', exitCode: 0 });
expect(captured.durationMs).toBe(Date.parse(captured.completedAt) - Date.parse(captured.startedAt));
expect(captured.remainingMs).toBeUndefined();
expect(result.stdout).not.toContain('stateRoot'); expect(result.stdout).not.toContain('stateRoot');
expect(result.stderr).toBe(''); expect(result.stderr).toBe('');
expect(fs.readFileSync(path.join(f.root, 'effects'), 'utf8')).toBe('once'); expect(fs.readFileSync(path.join(f.root, 'effects'), 'utf8')).toBe('once');
@@ -47,6 +49,10 @@ test('native capture executes once, preserves exact JSON and stderr, and materia
expect(JSON.parse(fs.readFileSync(path.join(f.root, 'exploration-001.json'), 'utf8'))).toEqual({ expect(JSON.parse(fs.readFileSync(path.join(f.root, 'exploration-001.json'), 'utf8'))).toEqual({
observationCommand: 'first native command', observed, hypothesis: 'The successful boundary suggests testing the rejected input next.', nextCommand: 'second native command', observationCommand: 'first native command', observed, hypothesis: 'The successful boundary suggests testing the rejected input next.', nextCommand: 'second native command',
}); });
f.json('annotations.json', { revision: 'revision', runtime: 'runtime', cwd: f.root, evidence: [], learning: [] });
const rejected = f.run('materialize', f.root, 'annotations.json');
expect(rejected.status).toBe(2);
expect(receipt(rejected.stderr).message).toContain('limits (non-empty string array)');
f.json('annotations.json', { revision: 'revision', runtime: 'runtime', cwd: f.root, limits: ['Only the declared contract was checked.'], evidence: [{ capture: '001', command: 'first native command', contract: 'README.md', expected: 'Declared exact result', classification: 'pass' }], learning: ['001'] }); f.json('annotations.json', { revision: 'revision', runtime: 'runtime', cwd: f.root, limits: ['Only the declared contract was checked.'], evidence: [{ capture: '001', command: 'first native command', contract: 'README.md', expected: 'Declared exact result', classification: 'pass' }], learning: ['001'] });
const report = f.run('materialize', f.root, 'annotations.json'); const report = f.run('materialize', f.root, 'annotations.json');
expect(report.status, report.stderr).toBe(0); expect(report.status, report.stderr).toBe(0);
@@ -94,7 +100,10 @@ test('capture shares a working-directory-relative deadline without resetting or
const before = fs.readFileSync(path.join(f.root, 'reports/deadline.json')); const before = fs.readFileSync(path.join(f.root, 'reports/deadline.json'));
const result = f.run('capture', 'reports', '001', '--deadline', 'reports/deadline.json', '--', process.execPath, '-e', 'console.log("{}")'); const result = f.run('capture', 'reports', '001', '--deadline', 'reports/deadline.json', '--', process.execPath, '-e', 'console.log("{}")');
expect(result.status, result.stderr).toBe(0); expect(result.status, result.stderr).toBe(0);
expect(receipt(result.stdout)).toMatchObject({ status: 'complete', exitCode: 0 }); const captured = receipt(result.stdout);
expect(captured).toMatchObject({ status: 'complete', exitCode: 0 });
expect(captured.remainingMs).toBeGreaterThan(0);
expect(captured.remainingMs).toBeLessThanOrEqual(5000);
expect(fs.readFileSync(path.join(f.root, 'reports/deadline.json'))).toEqual(before); expect(fs.readFileSync(path.join(f.root, 'reports/deadline.json'))).toEqual(before);
expect(fs.existsSync(path.join(f.root, 'reports/.qa-evidence/001/deadline.json'))).toBe(false); expect(fs.existsSync(path.join(f.root, 'reports/.qa-evidence/001/deadline.json'))).toBe(false);
}); });