fix(evals): cut path variance at its measured sources

- gstack-qa-evidence capture prints startedAt/completedAt/durationMs and, for
  --deadline captures, remainingMs; the functional report takes durations from
  them. The section clock notice asks for one clock read up front instead of one
  after every checkpoint (QA runs spent 7-14% of tool calls on date -u).
- ship plan-completion: skip the audit dispatch when discovery already found no
  plan (the dispatch-vs-skip conflict produced an optional 60-100 s subagent).
- materialize/checkpoint validation errors state the expected schema, so a
  rejected annotations file is fixable in one call instead of blocking the phase.
- session-runner counts turns from the transcript when a run times out, so
  timeouts stop reporting 'turn 0'.
This commit is contained in:
garrytan committed 2026-09-30 15:59:58 +00:00
1 parent a8a32e3216
commit 79092b22ea
8 files changed
+36 -22

No files matched your search

+10 -6
View File
@@ -2,7 +2,7 @@ import * as fs from 'node:fs';
import * as path from 'node:path';
import { createHash } from 'node:crypto';
import { atomicWriteSync } from './fs-atomic';
import { runQaDeadlineCommand, runQaWindowsWorker, startQaDeadline, withQaReceiptOutput } from './qa-deadline';
import { qaDeadlineStatus, readQaDeadline, runQaDeadlineCommand, runQaWindowsWorker, startQaDeadline, withQaReceiptOutput } from './qa-deadline';
import { scan } from './redact-engine';
const object = (value: unknown): value is Record<string, any> => value !== null && typeof value === 'object' && !Array.isArray(value);
@@ -174,11 +174,15 @@ async function capture(root: string, captureId: string, publicOutput: boolean, o
fs.writeFileSync(owned(root, `.qa-evidence/${captureId}/observation.json`), bytes, { flag: 'wx', mode: 0o600 });
observation = { sha256: hash(bytes), bytes: Buffer.byteLength(bytes) };
}
const completedAt = new Date().toISOString();
let remainingMs: number | undefined;
if (option === '--deadline') try { remainingMs = qaDeadlineStatus(readQaDeadline(deadline)).remainingMs; } catch {}
const receipt = { version: 1, id: captureId, cwd: process.cwd(), argv: [command, ...args], deadline, timing, startedAt,
completedAt: new Date().toISOString(), exitCode, signal: result.signal, status, observation, publicOutput,
completedAt, exitCode, signal: result.signal, status, observation, publicOutput,
...streams };
const sha256 = publish(root, `.qa-evidence/${captureId}/receipt.json`, receipt);
return { action: 'capture', id: captureId, status, sha256, exitCode, signal: result.signal, publicOutput };
return { action: 'capture', id: captureId, status, sha256, exitCode, signal: result.signal, publicOutput,
startedAt, completedAt, durationMs: Date.parse(completedAt) - Date.parse(startedAt), ...(remainingMs === undefined ? {} : { remainingMs }) };
}
function checkpoint(root: string, checkpointId: string, source: string | Record<string, string>) {
@@ -188,7 +192,7 @@ function checkpoint(root: string, checkpointId: string, source: string | Record<
if (!exact(intent, ['capture', 'observationCommand', 'hypothesis', 'nextCommand'])
|| typeof intent.capture !== 'string' || typeof intent.observationCommand !== 'string' || !intent.observationCommand.trim()
|| typeof intent.hypothesis !== 'string' || intent.hypothesis.trim().length <= 20 || !/[a-z]{3}/i.test(intent.hypothesis)
|| typeof intent.nextCommand !== 'string' || !intent.nextCommand.trim()) throw new QaEvidenceError('Invalid causal intent');
|| typeof intent.nextCommand !== 'string' || !intent.nextCommand.trim()) throw new QaEvidenceError('Invalid causal intent: need exactly capture, observationCommand, hypothesis (one sentence over 20 characters) and nextCommand');
if (scan(decode(bytes)).findings.some(finding => finding.tier === 'HIGH')) throw new QaEvidenceError('Sensitive intent cannot be published');
const captured = readQaCapture(root, intent.capture);
const value = { observationCommand: intent.observationCommand, observed: captured.observed, hypothesis: intent.hypothesis, nextCommand: intent.nextCommand };
@@ -204,11 +208,11 @@ function materialize(root: string, source: string) {
if (!exact(annotations, ['revision', 'runtime', 'cwd', 'limits', 'evidence', 'learning'])
|| !['revision', 'runtime', 'cwd'].every(key => typeof annotations[key] === 'string' && annotations[key].trim())
|| !Array.isArray(annotations.limits) || !annotations.limits.length || !annotations.limits.every((limit: unknown) => typeof limit === 'string' && limit.trim())
|| !Array.isArray(annotations.evidence) || !Array.isArray(annotations.learning)) throw new QaEvidenceError('Invalid report annotations');
|| !Array.isArray(annotations.evidence) || !Array.isArray(annotations.learning)) throw new QaEvidenceError('Invalid report annotations: need exactly revision, runtime and cwd (non-empty strings), limits (non-empty string array), evidence (row array) and learning (checkpoint ID array)');
const captures = new Set<string>();
const evidence = annotations.evidence.map((row: any) => {
if (!exact(row, ['capture', 'command', 'contract', 'expected', 'classification'])
|| !Object.values(row).every(value => typeof value === 'string' && value.trim()) || captures.has(row.capture)) throw new QaEvidenceError('Invalid evidence annotation');
|| !Object.values(row).every(value => typeof value === 'string' && value.trim()) || captures.has(row.capture)) throw new QaEvidenceError('Invalid evidence annotation: each row needs exactly capture, command, contract, expected and classification as non-empty strings, with a unique capture');
captures.add(row.capture);
const captured = readQaCapture(root, row.capture);
return { command: row.command, contract: row.contract, expected: row.expected, classification: row.classification, observed: captured.observed };
+1 -1
View File
@@ -7,7 +7,7 @@
| Surfaces / scope | {API, CLI, job, worker, webhook; changed and adjacent contracts} |
| Runtime / native tools | {VERSIONS AND REPOSITORY-SUPPORTED COMMANDS} |
| Fixture ownership / destinations | {ISOLATED ROOT, STORES, DOWNSTREAM TARGETS} |
| Duration / stop reason | {MEASURED COMMAND DURATIONS, COMPLETE OR BOUND/BLOCKER} |
| Duration / stop reason | {CAPTURE durationMs TOTALS, COMPLETE OR BOUND/BLOCKER} |
## Contract outcomes
+3 -3
View File
@@ -14,9 +14,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists,
say so explicitly and let the child use the fallback search below. The child does
not inherit the parent's conversation.
or complete text, including user-approved scope changes. If none is known, say
so; the child runs the fallback search below. If discovery found no plan, skip
dispatch. The child does not inherit the parent's conversation.
````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -3
View File
@@ -12,9 +12,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists,
say so explicitly and let the child use the fallback search below. The child does
not inherit the parent's conversation.
or complete text, including user-approved scope changes. If none is known, say
so; the child runs the fallback search below. If discovery found no plan, skip
dispatch. The child does not inherit the parent's conversation.
````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -3
View File
@@ -1556,9 +1556,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists,
say so explicitly and let the child use the fallback search below. The child does
not inherit the parent's conversation.
or complete text, including user-approved scope changes. If none is known, say
so; the child runs the fallback search below. If discovery found no plan, skip
dispatch. The child does not inherit the parent's conversation.
````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -3
View File
@@ -1536,9 +1536,9 @@ The child reads the plan and every referenced
code file; the parent validates its report and applies the gates below.
**Subagent prompt:** Substitute `<base>` and supply the active plan's absolute path
or complete text, including relevant user-approved scope changes. If none exists,
say so explicitly and let the child use the fallback search below. The child does
not inherit the parent's conversation.
or complete text, including user-approved scope changes. If none is known, say
so; the child runs the fallback search below. If discovery found no plan, skip
dispatch. The child does not inherit the parent's conversation.
````text
You are running a ship-workflow plan completion audit. The base branch is `<base>`. Use `git diff origin/<base>` and inspect untracked files from `git status` to see the full proposed change. Do not commit or push. Report only: classify every item, but do not execute Gate Logic, ask the user, or advance the workflow. The parent applies those gates to your report.
+3 -2
View File
@@ -293,7 +293,7 @@ Runner entry UTC: ${new Date(startTime).toISOString()}
Hard deadline UTC: ${new Date(deadline).toISOString()}
Completion reserve starts UTC: ${new Date(deadline - reserve).toISOString()}
Setup, CLI startup and API queueing consume this same window; it never resets.
Before source Reads and after each saved checkpoint, use Bash to run exactly \`date -u +%Y-%m-%dT%H:%M:%SZ\`. Compare that observed UTC time with the times above. When remaining time is at most ${reserve / 1000} seconds, prioritize the remaining required completion outputs and verification. No required content or gate may be skipped. If the clock read fails, report timing unavailable; do not invent remaining time or restart the deadline.`;
Before source Reads, use Bash to run exactly \`date -u +%Y-%m-%dT%H:%M:%SZ\`. After each saved checkpoint, compare the latest evidence capture's printed completedAt with the times above; run that clock read again only when no capture has completed since your last clock read. When remaining time is at most ${reserve / 1000} seconds, prioritize the remaining required completion outputs and verification. No required content or gate may be skipped. If the clock read fails, report timing unavailable; do not invent remaining time or restart the deadline.`;
systemPrompt = systemPrompt ? `${systemPrompt}\n\n${notice}` : notice;
}
@@ -696,7 +696,8 @@ Before source Reads and after each saved checkpoint, use Bash to run exactly \`d
}
// Cost from result line (exact) or estimate from chars
const turnsUsed = resultLine?.num_turns || 0;
const turnsUsed = resultLine?.num_turns
|| new Set(transcript.filter(event => event.type === 'assistant' && !event.parent_tool_use_id).map(event => event.message?.id)).size;
const estimatedCost = resultLine?.total_cost_usd || 0;
const inputChars = prompt.length;
const outputChars = (resultLine?.result || '').length;
+10 -1
View File
@@ -35,6 +35,8 @@ test('native capture executes once, preserves exact JSON and stderr, and materia
expect(result.status, result.stderr).toBe(0);
const captured = receipt(result.stdout);
expect(captured).toMatchObject({ action: 'capture', status: 'complete', id: '001', exitCode: 0 });
expect(captured.durationMs).toBe(Date.parse(captured.completedAt) - Date.parse(captured.startedAt));
expect(captured.remainingMs).toBeUndefined();
expect(result.stdout).not.toContain('stateRoot');
expect(result.stderr).toBe('');
expect(fs.readFileSync(path.join(f.root, 'effects'), 'utf8')).toBe('once');
@@ -47,6 +49,10 @@ test('native capture executes once, preserves exact JSON and stderr, and materia
expect(JSON.parse(fs.readFileSync(path.join(f.root, 'exploration-001.json'), 'utf8'))).toEqual({
observationCommand: 'first native command', observed, hypothesis: 'The successful boundary suggests testing the rejected input next.', nextCommand: 'second native command',
});
f.json('annotations.json', { revision: 'revision', runtime: 'runtime', cwd: f.root, evidence: [], learning: [] });
const rejected = f.run('materialize', f.root, 'annotations.json');
expect(rejected.status).toBe(2);
expect(receipt(rejected.stderr).message).toContain('limits (non-empty string array)');
f.json('annotations.json', { revision: 'revision', runtime: 'runtime', cwd: f.root, limits: ['Only the declared contract was checked.'], evidence: [{ capture: '001', command: 'first native command', contract: 'README.md', expected: 'Declared exact result', classification: 'pass' }], learning: ['001'] });
const report = f.run('materialize', f.root, 'annotations.json');
expect(report.status, report.stderr).toBe(0);
@@ -94,7 +100,10 @@ test('capture shares a working-directory-relative deadline without resetting or
const before = fs.readFileSync(path.join(f.root, 'reports/deadline.json'));
const result = f.run('capture', 'reports', '001', '--deadline', 'reports/deadline.json', '--', process.execPath, '-e', 'console.log("{}")');
expect(result.status, result.stderr).toBe(0);
expect(receipt(result.stdout)).toMatchObject({ status: 'complete', exitCode: 0 });
const captured = receipt(result.stdout);
expect(captured).toMatchObject({ status: 'complete', exitCode: 0 });
expect(captured.remainingMs).toBeGreaterThan(0);
expect(captured.remainingMs).toBeLessThanOrEqual(5000);
expect(fs.readFileSync(path.join(f.root, 'reports/deadline.json'))).toEqual(before);
expect(fs.existsSync(path.join(f.root, 'reports/.qa-evidence/001/deadline.json'))).toBe(false);
});