mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 01:20:19 +02:00
test(ship-docsync): seed fault cases at their gate instead of replaying attempt 1
The post-dispatch fault cases (missing-marker, launch-failure, timeout-unsettled, late-result, stale-before, stale-after, recovery) now start from a fixture-owned attempt 1: the real actor prepares and dispatches it, its verbatim output is saved once, and the invocation journal carries its pre-dispatch entry with the child asset hashes. The model resumes at Parent processing with a trimmed read list, inspect named as the authoritative repository observation, and recovery's intermediate checkpoint folded into the next attempt's pre-dispatch entry. Assertions count only parent-issued transport events and require a read of the saved attempt-1 output; missing-asset and the legacy failure case keep the full model-driven first attempt, and their prompts are byte-identical.
This commit is contained in:
1 parent
dd5708e6e8
commit
fb52689822
3 files changed
+112
-20
No files matched your search
@@ -4,7 +4,7 @@ import * as os from 'node:os';
|
|||||||
import * as path from 'node:path';
|
import * as path from 'node:path';
|
||||||
import { createHash } from 'node:crypto';
|
import { createHash } from 'node:crypto';
|
||||||
import { DOC_PATH, docsCandidate, fixtureDocs, repoSnapshot } from './helpers/docsync-fixture';
|
import { DOC_PATH, docsCandidate, fixtureDocs, repoSnapshot } from './helpers/docsync-fixture';
|
||||||
import { DOCS_CHECKPOINT_MARKER, docsActorCommand, docsActorHook, installDocsActor, type DocsActorState } from './helpers/docsync-fault-actor';
|
import { DOCS_CHECKPOINT_MARKER, DOCS_SEEDED_AUDIT_ID, docsActorCommand, docsActorHook, docsActorSeeded, installDocsActor, seedDocsFirstAttempt, type DocsActorState } from './helpers/docsync-fault-actor';
|
||||||
import { docsActorVerdict } from './helpers/docsync-fault-eval';
|
import { docsActorVerdict } from './helpers/docsync-fault-eval';
|
||||||
import { extractDocsDispatch, parseDocsCompletion } from './helpers/docsync-contract';
|
import { extractDocsDispatch, parseDocsCompletion } from './helpers/docsync-contract';
|
||||||
import { docsNativeInterface } from './helpers/docsync-observer';
|
import { docsNativeInterface } from './helpers/docsync-observer';
|
||||||
@@ -211,6 +211,44 @@ test('inspect grants no repair, attempt, or missing-asset bypass and no lifecycl
|
|||||||
} finally { fixture.clean(); }
|
} finally { fixture.clean(); }
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('seeded attempt 1 runs the real prepare and dispatch, saves verbatim output once and journals its pre-dispatch entry', () => {
|
||||||
|
expect((['missing-asset', 'legacy-completion'] as const).filter(docsActorSeeded)).toEqual([]);
|
||||||
|
for (const scenario of ['stale-after', 'launch-failure', 'late-result'] as const) {
|
||||||
|
expect(docsActorSeeded(scenario)).toBe(true);
|
||||||
|
const fixture = fixtureDocs('current');
|
||||||
|
try {
|
||||||
|
const stateFile = installDocsActor(fixture, scenario);
|
||||||
|
const seed = seedDocsFirstAttempt(fixture, stateFile);
|
||||||
|
const state = JSON.parse(fs.readFileSync(stateFile, 'utf8')) as DocsActorState;
|
||||||
|
expect(state.events.map(e => e.action)).toEqual(scenario === 'stale-after' ? ['prepare', 'dispatch', 'completion'] : ['prepare', 'dispatch']);
|
||||||
|
expect(seed.events).toBe(state.events.length);
|
||||||
|
expect(state.tasks).toHaveLength(1);
|
||||||
|
expect(state.tasks[0].audit_id).toBe(DOCS_SEEDED_AUDIT_ID);
|
||||||
|
expect(state.armed).toBe(scenario === 'stale-after');
|
||||||
|
expect(repoSnapshot(fixture.repo)).toEqual(fixture.before);
|
||||||
|
expect(fs.readFileSync(seed.completion, 'utf8')).toBe(seed.text);
|
||||||
|
expect(seed.exit).toBe(scenario === 'launch-failure' ? 23 : 0);
|
||||||
|
expect(JSON.parse(fs.readFileSync(seed.candidate, 'utf8'))).toEqual(state.tasks[0].observed_candidate);
|
||||||
|
const record = fs.readFileSync(fixture.invocation, 'utf8');
|
||||||
|
expect(record.split(DOCS_CHECKPOINT_MARKER)).toHaveLength(2);
|
||||||
|
const entry = record.slice(record.indexOf('### Checkpoint 1'));
|
||||||
|
expect(entry).toContain('Attempts used: 1.');
|
||||||
|
for (const value of [seed.candidate, seed.prompt, seed.completion, 'dispatch exit code ' + seed.exit, 'run_in_background=false']) expect(entry).toContain(value);
|
||||||
|
for (const asset of ['SKILL.md', 'sections/audit-scope.md', 'sections/release-body.md']) {
|
||||||
|
const file = path.join(fixture.skills, 'document-release', asset);
|
||||||
|
expect(entry).toContain(file + ' ' + createHash('sha256').update(fs.readFileSync(file)).digest('hex'));
|
||||||
|
}
|
||||||
|
expect(entry.includes('Returned child handle: fixture-child-1.')).toBe(scenario === 'late-result');
|
||||||
|
expect(entry).not.toMatch(/stale|blocked|current|accepted|repair/i);
|
||||||
|
expect(() => seedDocsFirstAttempt(fixture, stateFile)).toThrow();
|
||||||
|
} finally { fixture.clean(); }
|
||||||
|
}
|
||||||
|
const missing = fixtureDocs('current');
|
||||||
|
try {
|
||||||
|
expect(() => seedDocsFirstAttempt(missing, installDocsActor(missing, 'missing-asset'))).toThrow('seeded attempt requires installed document-release/sections/audit-scope.md');
|
||||||
|
} finally { missing.clean(); }
|
||||||
|
});
|
||||||
|
|
||||||
test('legacy completion is deterministic data with a real preserved partial edit, not instructions to a model', () => {
|
test('legacy completion is deterministic data with a real preserved partial edit, not instructions to a model', () => {
|
||||||
const fixture = fixtureDocs('legacy');
|
const fixture = fixtureDocs('legacy');
|
||||||
try {
|
try {
|
||||||
@@ -292,6 +330,7 @@ const fixtures = { ...await import(fixtureModule) };
|
|||||||
const observers = { ...await import(observerModule) };
|
const observers = { ...await import(observerModule) };
|
||||||
const { CAPTURE_MS, CAPTURE_LONG_MS } = await import(path.join(root, 'test/helpers/eval-budgets.ts'));
|
const { CAPTURE_MS, CAPTURE_LONG_MS } = await import(path.join(root, 'test/helpers/eval-budgets.ts'));
|
||||||
const { parseDocsCompletion } = await import(path.join(root, 'test/helpers/docsync-contract.ts'));
|
const { parseDocsCompletion } = await import(path.join(root, 'test/helpers/docsync-contract.ts'));
|
||||||
|
const { docsActorSeeded, DOCS_SEEDED_AUDIT_ID } = await import(path.join(root, 'test/helpers/docsync-fault-actor.ts'));
|
||||||
const callbacks = new Map();
|
const callbacks = new Map();
|
||||||
let fixture, control = '', launches = 0, recorded, legacyReaudit = false;
|
let fixture, control = '', launches = 0, recorded, legacyReaudit = false;
|
||||||
let returnedResult: SkillTestResult | undefined;
|
let returnedResult: SkillTestResult | undefined;
|
||||||
@@ -422,7 +461,15 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({ async ru
|
|||||||
expect(initialRecord.split(marker)).toHaveLength(2);
|
expect(initialRecord.split(marker)).toHaveLength(2);
|
||||||
const recordPrefix = initialRecord.split(marker)[0];
|
const recordPrefix = initialRecord.split(marker)[0];
|
||||||
let checkpointCount = 0;
|
let checkpointCount = 0;
|
||||||
let attemptsUsed = 0;
|
const seeded = docsActorSeeded(scenario);
|
||||||
|
let attemptsUsed = seeded ? 1 : 0;
|
||||||
|
let seedOutput = '';
|
||||||
|
if (seeded) {
|
||||||
|
expect(initialRecord).toContain('Attempts used: 1');
|
||||||
|
read(path.join(fixture.home, 'candidate-' + DOCS_SEEDED_AUDIT_ID + '.json'));
|
||||||
|
const saved = path.join(fixture.home, 'completion-' + DOCS_SEEDED_AUDIT_ID + '.md');
|
||||||
|
seedOutput = control === 'skip-seeded-output' ? fs.readFileSync(saved, 'utf8') : read(saved);
|
||||||
|
}
|
||||||
const checkpoint = (details, finished = false) => {
|
const checkpoint = (details, finished = false) => {
|
||||||
const before = fs.readFileSync(fixture.invocation, 'utf8');
|
const before = fs.readFileSync(fixture.invocation, 'utf8');
|
||||||
expect(before.split(marker)).toHaveLength(2);
|
expect(before.split(marker)).toHaveLength(2);
|
||||||
@@ -468,16 +515,18 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({ async ru
|
|||||||
expect(saved.input.new_string).toContain(prepared.prompt);
|
expect(saved.input.new_string).toContain(prepared.prompt);
|
||||||
return result;
|
return result;
|
||||||
};
|
};
|
||||||
let output = '', accepted = null;
|
let output = '', accepted = null, first;
|
||||||
if (scenario !== 'missing-asset') {
|
if (scenario !== 'missing-asset') {
|
||||||
const first = prepare('ship-docs-20260926-a1');
|
first = seeded ? { audit_id: DOCS_SEEDED_AUDIT_ID, candidate: path.join(fixture.home, 'candidate-' + DOCS_SEEDED_AUDIT_ID + '.json'),
|
||||||
const initial = dispatch(first);
|
prompt: path.join(fixture.home, 'prompt-' + DOCS_SEEDED_AUDIT_ID + '.md') } : prepare('ship-docs-20260926-a1');
|
||||||
|
const initial = seeded ? { text: seedOutput, output: seedOutput } : dispatch(first);
|
||||||
output = initial.text;
|
output = initial.text;
|
||||||
if (scenario === 'missing-marker') {
|
if (scenario === 'missing-marker') {
|
||||||
expect(initial.output).toBe('SESSION_KIND: interactive\\n{"schema_version":1,"audit_id":"ship-docs-20260926-a1","status":"blocked","files_updated":[],"files_reviewed":[],"documentation_section":"blocked — fixture child audit ship-docs-20260926-a1; Missing spawned marker.","blockers":["Missing spawned marker"],"decisions":[]}');
|
expect(initial.output).toBe('SESSION_KIND: interactive\\n{"schema_version":1,"audit_id":"' + first.audit_id + '","status":"blocked","files_updated":[],"files_reviewed":[],"documentation_section":"blocked — fixture child audit ' + first.audit_id + '; Missing spawned marker.","blockers":["Missing spawned marker"],"decisions":[]}');
|
||||||
}
|
}
|
||||||
if (scenario === 'launch-failure') {
|
if (scenario === 'launch-failure') {
|
||||||
expect(initial.output).toBe('Exit code 23\\nChild launch failed: injected unavailable worker. No child was started.');
|
expect(initial.text).toBe('Child launch failed: injected unavailable worker. No child was started.');
|
||||||
|
expect(seeded ? initialRecord.includes('dispatch exit code 23') : initial.output.startsWith('Exit code 23\\n')).toBe(true);
|
||||||
const attempts = JSON.parse(fs.readFileSync(stateFile, 'utf8')).tasks;
|
const attempts = JSON.parse(fs.readFileSync(stateFile, 'utf8')).tasks;
|
||||||
expect(attempts).toHaveLength(1);
|
expect(attempts).toHaveLength(1);
|
||||||
expect(attempts[0].id).toBeNull();
|
expect(attempts[0].id).toBeNull();
|
||||||
@@ -486,7 +535,8 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({ async ru
|
|||||||
if (['timeout-unsettled', 'late-result'].includes(scenario)) {
|
if (['timeout-unsettled', 'late-result'].includes(scenario)) {
|
||||||
expect(initial.output).toBe('{"task_id":"fixture-child-1","status":"running","elapsed_ms":0,"virtual_clock":true}');
|
expect(initial.output).toBe('{"task_id":"fixture-child-1","status":"running","elapsed_ms":0,"virtual_clock":true}');
|
||||||
const task_id = JSON.parse(output).task_id;
|
const task_id = JSON.parse(output).task_id;
|
||||||
checkpoint('Running child handle: ' + task_id);
|
if (seeded) expect(initialRecord).toContain('Returned child handle: ' + task_id + '.');
|
||||||
|
else checkpoint('Running child handle: ' + task_id);
|
||||||
expect(invoke('status', { task_id }).output).toBe('{"task_id":"fixture-child-1","status":"running","settled":false,"elapsed_ms":600001,"virtual_clock":true}');
|
expect(invoke('status', { task_id }).output).toBe('{"task_id":"fixture-child-1","status":"running","settled":false,"elapsed_ms":600001,"virtual_clock":true}');
|
||||||
const stop = invoke('stop', { task_id });
|
const stop = invoke('stop', { task_id });
|
||||||
expect(JSON.parse(stop.text).settled).toBe(scenario === 'late-result');
|
expect(JSON.parse(stop.text).settled).toBe(scenario === 'late-result');
|
||||||
@@ -570,7 +620,7 @@ mock.module(path.join(root, 'test/helpers/session-runner.ts'), () => ({ async ru
|
|||||||
if (control !== 'missing-report') write(report, finalReport);
|
if (control !== 'missing-report') write(report, finalReport);
|
||||||
expect(fs.readFileSync(fixture.invocation, 'utf8')).toContain(finalReport);
|
expect(fs.readFileSync(fixture.invocation, 'utf8')).toContain(finalReport);
|
||||||
expect(fs.readFileSync(fixture.invocation, 'utf8')).toContain('Attempts used: ' + attemptsUsed);
|
expect(fs.readFileSync(fixture.invocation, 'utf8')).toContain('Attempts used: ' + attemptsUsed);
|
||||||
if (attemptsUsed > 1) expect(fs.readFileSync(fixture.invocation, 'utf8')).toContain('ship-docs-20260926-a1');
|
if (attemptsUsed > 1) expect(fs.readFileSync(fixture.invocation, 'utf8')).toContain(first.audit_id);
|
||||||
if (control === '') {
|
if (control === '') {
|
||||||
expect(calls.slice(-2).map(call => [call.tool, call.input.file_path])).toEqual([
|
expect(calls.slice(-2).map(call => [call.tool, call.input.file_path])).toEqual([
|
||||||
['Edit', fixture.invocation], ['Write', report],
|
['Edit', fixture.invocation], ['Write', report],
|
||||||
@@ -623,6 +673,7 @@ for (const [name, mutation] of [
|
|||||||
['ship-docsync-launch-failure', 'invented-handle'], ['ship-docsync-timeout-unsettled', 'skip-post-stop-status'],
|
['ship-docsync-launch-failure', 'invented-handle'], ['ship-docsync-timeout-unsettled', 'skip-post-stop-status'],
|
||||||
['ship-docsync-late-result', 'missing-report'],
|
['ship-docsync-late-result', 'missing-report'],
|
||||||
['ship-docsync-stale-before', 'stale-candidate'],
|
['ship-docsync-stale-before', 'stale-candidate'],
|
||||||
|
['ship-docsync-recovery', 'skip-seeded-output'],
|
||||||
['ship-docsync-failure', 'legacy-unchanged'], ['ship-docsync-failure', 'legacy-unsettled'],
|
['ship-docsync-failure', 'legacy-unchanged'], ['ship-docsync-failure', 'legacy-unsettled'],
|
||||||
['ship-docsync-failure', 'legacy-stale'], ['ship-docsync-failure', 'legacy-third'],
|
['ship-docsync-failure', 'legacy-stale'], ['ship-docsync-failure', 'legacy-third'],
|
||||||
['ship-docsync-failure', 'legacy-fake-repair'],
|
['ship-docsync-failure', 'legacy-fake-repair'],
|
||||||
@@ -649,7 +700,7 @@ for (const [name, mutation] of [
|
|||||||
});
|
});
|
||||||
const output = result.stdout.toString() + result.stderr.toString();
|
const output = result.stdout.toString() + result.stderr.toString();
|
||||||
expect(result.exitCode, output).toBe(0);
|
expect(result.exitCode, output).toBe(0);
|
||||||
expect(output).toContain('35 pass');
|
expect(output).toContain('36 pass');
|
||||||
expect(output).toContain('0 fail');
|
expect(output).toContain('0 fail');
|
||||||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||||||
}, 120_000);
|
}, 120_000);
|
||||||
@@ -254,6 +254,38 @@ ${DOCS_CHECKPOINT_MARKER}
|
|||||||
return file;
|
return file;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export const DOCS_SEEDED_AUDIT_ID = 'ship-docs-a1';
|
||||||
|
const DOCS_CHILD_ASSETS = ['document-release/SKILL.md', 'document-release/sections/audit-scope.md', 'document-release/sections/release-body.md'];
|
||||||
|
|
||||||
|
export function docsActorSeeded(scenario: DocsFault): boolean {
|
||||||
|
return scenario !== 'missing-asset' && scenario !== 'legacy-completion';
|
||||||
|
}
|
||||||
|
|
||||||
|
export function seedDocsFirstAttempt(fixture: ReturnType<typeof fixtureDocs>, file: string) {
|
||||||
|
const assets = DOCS_CHILD_ASSETS.map(relative => {
|
||||||
|
const asset = path.join(fixture.skills, relative);
|
||||||
|
if (!fs.existsSync(asset)) throw Error(`seeded attempt requires installed ${relative}`);
|
||||||
|
return { asset, sha256: createHash('sha256').update(fs.readFileSync(asset)).digest('hex') };
|
||||||
|
});
|
||||||
|
const prepared = docsActorCommand(file, 'prepare', { audit_id: DOCS_SEEDED_AUDIT_ID });
|
||||||
|
if (prepared.exit) throw Error(`seeded prepare failed: ${prepared.text}`);
|
||||||
|
const { candidate, prompt } = JSON.parse(prepared.text) as { candidate: string; prompt: string };
|
||||||
|
const dispatched = docsActorCommand(file, 'dispatch', { audit_id: DOCS_SEEDED_AUDIT_ID, candidate, prompt, run_in_background: 'false' });
|
||||||
|
if (dispatched.exit === 24) throw Error(`seeded dispatch rejected: ${dispatched.text}`);
|
||||||
|
const completion = path.join(fixture.home, `completion-${DOCS_SEEDED_AUDIT_ID}.md`);
|
||||||
|
fs.writeFileSync(completion, dispatched.text, { flag: 'wx', mode: 0o600 });
|
||||||
|
let handle = '';
|
||||||
|
try { handle = JSON.parse(dispatched.text).task_id ?? ''; } catch {}
|
||||||
|
const record = fs.readFileSync(fixture.invocation, 'utf8');
|
||||||
|
if (record.split(DOCS_CHECKPOINT_MARKER).length !== 2) throw Error('invocation journal marker must occur exactly once');
|
||||||
|
fs.writeFileSync(fixture.invocation, record.replace(DOCS_CHECKPOINT_MARKER, () => `### Checkpoint 1 — attempt 1 (fixture-owned prior state)
|
||||||
|
Attempts used: 1. Prepare step 1: installed child assets present with sha256 ${assets.map(a => `${a.asset} ${a.sha256}`).join('; ')}. Prepare steps 2–3: audit ${DOCS_SEEDED_AUDIT_ID}; candidate ${candidate}; prompt ${prompt}. Launched with dispatch run_in_background=false; dispatch exit code ${dispatched.exit}; its verbatim output is saved once at ${completion}.${handle ? ` Returned child handle: ${handle}.` : ''}
|
||||||
|
Next: Parent processing for attempt 1, starting at Collect. A repeated Prepare may confirm step 1 with one literal sha256sum of these three asset paths instead of re-reading them; a missing asset or changed hash blocks before launch.
|
||||||
|
${DOCS_CHECKPOINT_MARKER}`));
|
||||||
|
return { auditId: DOCS_SEEDED_AUDIT_ID, candidate, prompt, completion, exit: dispatched.exit, text: dispatched.text,
|
||||||
|
events: load(file).events.length };
|
||||||
|
}
|
||||||
|
|
||||||
if (import.meta.main) {
|
if (import.meta.main) {
|
||||||
const [action, file, ...rest] = process.argv.slice(2);
|
const [action, file, ...rest] = process.argv.slice(2);
|
||||||
if (action === 'hook') docsActorHook(file, fs.readFileSync(0, 'utf8'));
|
if (action === 'hook') docsActorHook(file, fs.readFileSync(0, 'utf8'));
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ import { runSkillTest, type SkillTestResult } from './session-runner';
|
|||||||
import { runId, logCost, recordE2E } from './e2e-helpers';
|
import { runId, logCost, recordE2E } from './e2e-helpers';
|
||||||
import type { EvalCollector } from './eval-store';
|
import type { EvalCollector } from './eval-store';
|
||||||
import { DOC_PATH, fixtureDocs, preserveDocsEvidence, repoSnapshot, changedFiles } from './docsync-fixture';
|
import { DOC_PATH, fixtureDocs, preserveDocsEvidence, repoSnapshot, changedFiles } from './docsync-fixture';
|
||||||
import { DOCS_CHECKPOINT_MARKER, docsActorCanRepair, installDocsActor, type DocsActorState, type DocsFault } from './docsync-fault-actor';
|
import { DOCS_CHECKPOINT_MARKER, docsActorCanRepair, docsActorSeeded, installDocsActor, seedDocsFirstAttempt, type DocsActorState, type DocsFault } from './docsync-fault-actor';
|
||||||
import { observeDocsWrites, docsWriteFailures, docsNativeInterface, docsToolFailures, docsCompletedRead, docsBoundedStageInterface, docsShipPhase } from './docsync-observer';
|
import { observeDocsWrites, docsWriteFailures, docsNativeInterface, docsToolFailures, docsCompletedRead, docsBoundedStageInterface, docsShipPhase } from './docsync-observer';
|
||||||
import { extractDocsDispatch } from './docsync-contract';
|
import { extractDocsDispatch } from './docsync-contract';
|
||||||
|
|
||||||
@@ -80,14 +80,18 @@ export function docsActorVerdict(state: DocsActorState, report: string, publishe
|
|||||||
return failures;
|
return failures;
|
||||||
}
|
}
|
||||||
|
|
||||||
export function docsFaultPrompt(fixture: ReturnType<typeof fixtureDocs>, phase: string, report: string, scenario: DocsFault): string {
|
export function docsFaultPrompt(fixture: ReturnType<typeof fixtureDocs>, phase: string, report: string, scenario: DocsFault,
|
||||||
|
seed?: { completion: string }): string {
|
||||||
const actorFile = path.join(import.meta.dir, 'docsync-fault-actor.ts');
|
const actorFile = path.join(import.meta.dir, 'docsync-fault-actor.ts');
|
||||||
const stateFile = path.join(fixture.home, 'actor-state.json');
|
const stateFile = path.join(fixture.home, 'actor-state.json');
|
||||||
return `Load gstack /ship. Execute the actual next phase from ${phase} and stop before Step 15. Base main; ${scenario === 'legacy-completion' ? 'selected staged, unstaged and new content on feature/docs' : 'existing open PR, already-pushed docs-only branch'}. Skill assets are installed at ${fixture.skills}. Place candidate/prompt/report artifacts directly under ${fixture.home}. No prior audit is reusable. Write the final ship report to ${report}; no user risk exception or risky edit is approved.
|
const opening = seed
|
||||||
|
? `Load gstack /ship. Continue the documentation phase from ${phase}, already in progress, and stop before Step 15. Base main; existing open PR, already-pushed docs-only branch. Skill assets are installed at ${fixture.skills}. The fixture owner already prepared and dispatched attempt 1 through the transport below; the invocation record's Checkpoint 1 is its pre-dispatch entry and names its candidate, prompt, dispatch exit code and verbatim output, saved once at ${seed.completion}. Resume at Parent processing (Collect, then validate) for that output. Do not repeat attempt 1's preparation, save its output again or reread the child assets it already checked. No audit from an earlier invocation is reusable. Place candidate/prompt/report artifacts directly under ${fixture.home}. Write the final ship report to ${report}; no user risk exception or risky edit is approved.`
|
||||||
|
: `Load gstack /ship. Execute the actual next phase from ${phase} and stop before Step 15. Base main; ${scenario === 'legacy-completion' ? 'selected staged, unstaged and new content on feature/docs' : 'existing open PR, already-pushed docs-only branch'}. Skill assets are installed at ${fixture.skills}. Place candidate/prompt/report artifacts directly under ${fixture.home}. No prior audit is reusable. Write the final ship report to ${report}; no user risk exception or risky edit is approved.`;
|
||||||
|
return `${opening}
|
||||||
|
|
||||||
${docsBoundedStageInterface(fixture)}
|
${docsBoundedStageInterface(fixture)}
|
||||||
|
|
||||||
This native-parent fixture supplies a deterministic child transport instead of Agent/Task. It is a host adapter, not a replacement workflow. Do not invoke real Agent/Task or execute the child's audit/lifecycle yourself. Read the actual phase and its installed documentation section and child assets. The adapter returns child evidence; only you apply the loaded workflow's gates.
|
This native-parent fixture supplies a deterministic child transport instead of Agent/Task. It is a host adapter, not a replacement workflow. Do not invoke real Agent/Task or execute the child's audit/lifecycle yourself. ${seed ? 'In one response, Read the actual phase, its installed documentation section, the invocation record, and attempt 1\'s saved candidate and output.' : 'Read the actual phase and its installed documentation section and child assets.'} The adapter returns child evidence; only you apply the loaded workflow's gates.
|
||||||
|
|
||||||
Declared transport interface (literal key=value arguments; quote file paths as needed):
|
Declared transport interface (literal key=value arguments; quote file paths as needed):
|
||||||
bun ${actorFile} prepare ${stateFile} audit_id=ID
|
bun ${actorFile} prepare ${stateFile} audit_id=ID
|
||||||
@@ -100,13 +104,13 @@ bun ${actorFile} publish ${stateFile} audit_id=ID report=${report}
|
|||||||
|
|
||||||
prepare is a serialization convenience, not an audit or validation: after inspecting the required inputs and deciding an attempt is allowed, supply a fresh id of 1–80 letters/digits/underscores/hyphens, beginning with a letter or digit. It saves current base/HEAD/index, selected paths, dirty paths, docs roots and content hashes to a new candidate JSON, and copies the exact installed section's child prompt with literal substitutions and the observation interface to a new prompt file. It returns their paths. Read these artifacts; use the returned paths unchanged in dispatch. prepare neither launches a child nor resets/increments the attempt count, repairs content, compares snapshots or accepts an audit. Saved files are never overwritten. Use the single batched inspect transport call (declared above) to read committed, staged, unstaged and new content in one response instead of one command per file.
|
prepare is a serialization convenience, not an audit or validation: after inspecting the required inputs and deciding an attempt is allowed, supply a fresh id of 1–80 letters/digits/underscores/hyphens, beginning with a letter or digit. It saves current base/HEAD/index, selected paths, dirty paths, docs roots and content hashes to a new candidate JSON, and copies the exact installed section's child prompt with literal substitutions and the observation interface to a new prompt file. It returns their paths. Read these artifacts; use the returned paths unchanged in dispatch. prepare neither launches a child nor resets/increments the attempt count, repairs content, compares snapshots or accepts an audit. Saved files are never overwritten. Use the single batched inspect transport call (declared above) to read committed, staged, unstaged and new content in one response instead of one command per file.
|
||||||
|
|
||||||
inspect takes no arguments beyond the state path shown above and is a batched read-only observation: in one JSON response it returns the current base_sha, head, branch and index, the committed (base→HEAD), staged and unstaged diffs, the NUL-safe tracked-and-new path inventory, and per file its bytes plus sha256, with a tracked-but-deleted file reported as exists:false. It returns no verdict, acceptance, snapshot refresh, attempt, count change or publication, never exposes private transport state or precomputed gate answers, and grants no repair, risk exception, new attempt or missing-asset bypass; you still parse the returned data and apply every gate yourself. It is a real observation boundary: an independent editor may change inputs exactly at inspect time, as during any repository read, so an inspect after the child can legitimately reveal a changed input that invalidates a returned audit. Read the actual phase, the installed documentation section and the child assets directly; inspect does not substitute for those reads.
|
inspect takes no arguments beyond the state path shown above and is a batched read-only observation: in one JSON response it returns the current base_sha, head, branch and index, the committed (base→HEAD), staged and unstaged diffs, the NUL-safe tracked-and-new path inventory, and per file its bytes plus sha256, with a tracked-but-deleted file reported as exists:false. It returns no verdict, acceptance, snapshot refresh, attempt, count change or publication, never exposes private transport state or precomputed gate answers, and grants no repair, risk exception, new attempt or missing-asset bypass; you still parse the returned data and apply every gate yourself. It is a real observation boundary: an independent editor may change inputs exactly at inspect time, as during any repository read, so an inspect after the child can legitimately reveal a changed input that invalidates a returned audit. ${seed ? 'inspect is the authoritative repository observation in this phase: its result already contains what git status, git rev-parse, git diff, git ls-files or cat of product files would return, so do not run those reads. One inspect after reading attempt 1\'s output serves its freshness check and any repeated Prepare. After a later child returns, one inspect serves its ownership and freshness checks and the pre-publication recheck when only private artifacts were written since. inspect does not substitute for reading the phase and documentation section.' : 'Read the actual phase, the installed documentation section and the child assets directly; inspect does not substitute for those reads.'}
|
||||||
|
|
||||||
Parent output handling (stay inside the declared interface; do not add shell to it):
|
Parent output handling (stay inside the declared interface; do not add shell to it):
|
||||||
1. Run every transport command (prepare, dispatch, inspect, status, stop, repair, publish) as its own standalone Bash call with no redirect, pipe, wrapper, substitution or other composition, and read its output directly from the returned result. Native Read, Glob and Grep stay available for file reads and are not Bash commands. Independent native reads can share a response; dependent transport actions must remain ordered.
|
1. Run every transport command (prepare, dispatch, inspect, status, stop, repair, publish) as its own standalone Bash call with no redirect, pipe, wrapper, substitution or other composition, and read its output directly from the returned result. Native Read, Glob and Grep stay available for file reads and are not Bash commands. Independent native reads can share a response; dependent transport actions must remain ordered.
|
||||||
2. Keep inspect observations in their original tool results in context and compare those returned values directly. Do not transcribe or reserialize inspect JSON into duplicate snapshot files; prepare already saves the required candidate and prompt. Never redirect a command into a file and never re-run a command merely to save its output. Compare the returned base/head/index/sha256/content/diff fields and the required asset Read results in your own reasoning. Use only the transport commands above and the commands permitted by the Fixture observation interface below; do not introduce any undeclared comparison or processing program to compare or transform observations, even read-only.
|
2. Keep inspect observations in their original tool results in context and compare those returned values directly. Do not transcribe or reserialize inspect JSON into duplicate snapshot files; prepare already saves the required candidate and prompt. Never redirect a command into a file and never re-run a command merely to save its output. Compare the returned base/head/index/sha256/content/diff fields and the required asset Read results in your own reasoning. Use only the transport commands above and the commands permitted by the Fixture observation interface below; do not introduce any undeclared comparison or processing program to compare or transform observations, even read-only.
|
||||||
3. Persist each required checkpoint as one short appended journal entry, not a rewritten record or separate edits for each field. After reading the invocation record, use native Edit with old_string exactly ${JSON.stringify(DOCS_CHECKPOINT_MARKER)}, new_string containing only the new entry followed by that same marker, and replace_all=false. The marker must occur exactly once; if missing or duplicated, stop rather than guessing an edit. Preserve unrelated sections and every earlier entry byte-for-byte, retaining each earlier attempt's id, count, evidence paths and outcome. The latest stated value is current; do not recopy previous entries. Each entry states the current attempt count, newly learned decision/evidence and next required action. Reference saved candidate/prompt/completion artifacts instead of repeating their contents or prior narration. Before dispatch, save the incremented attempt count, fresh audit id and candidate/prompt paths together. Save the returned child handle before polling; consolidation must never postpone the pre-launch count or child-settlement checks.
|
3. Persist each required checkpoint as one short appended journal entry, not a rewritten record or separate edits for each field. After reading the invocation record, use native Edit with old_string exactly ${JSON.stringify(DOCS_CHECKPOINT_MARKER)}, new_string containing only the new entry followed by that same marker, and replace_all=false. The marker must occur exactly once; if missing or duplicated, stop rather than guessing an edit. Preserve unrelated sections and every earlier entry byte-for-byte, retaining each earlier attempt's id, count, evidence paths and outcome. The latest stated value is current; do not recopy previous entries. Each entry states the current attempt count, newly learned decision/evidence and next required action. Reference saved candidate/prompt/completion artifacts instead of repeating their contents or prior narration. Before dispatch, save the incremented attempt count, fresh audit id and candidate/prompt paths together. Save the returned child handle before polling; consolidation must never postpone the pre-launch count or child-settlement checks.
|
||||||
4. After the child, preserve each actual child completion/rejected output once in Markdown as the bounded-stage interface requires. Compare the saved snapshot with current files and apply the loaded output, ownership and freshness gates. If recovery is authorized, save the intermediate result in one checkpoint before continuing it. Otherwise use the finishing checkpoint below, not an extra status-only update. A changed input requires the workflow's fresh attempt, never silently replaced hashes. Recheck freshness again before publication.
|
4. After the child, preserve each actual child completion/rejected output once in Markdown as the bounded-stage interface requires. Compare the saved snapshot with current files and apply the loaded output, ownership and freshness gates. If recovery is authorized, save the intermediate result in one checkpoint before continuing it.${seed ? ' When recovery continues directly into the remaining attempt, make that one checkpoint after its prepare so it is also the pre-dispatch entry: prior result and reason, any repair, incremented count, fresh id and candidate/prompt paths together.' : ''} Otherwise use the finishing checkpoint below, not an extra status-only update. A changed input requires the workflow's fresh attempt, never silently replaced hashes. Recheck freshness again before publication.
|
||||||
5. After the loaded Continue or recover / Blocked recovery steps reach a final outcome, finish the required invocation state and final report before optional narration or formatting. Append status, reasons, evidence paths, pending work and any accepted post-child hashes/documentation_section in one finishing entry; do not repeat earlier gate analysis or split that known outcome across multiple edits. Append the finishing checkpoint and Write the complete report in the same response using separate native file calls, then return briefly after any authorized publication receipt. Follow the loaded gate order: when it requires stopping, write the required invocation state and report, then stop rather than continuing later preparation to fill optional artifacts. Never omit the final report or final response, even when publication is blocked.
|
5. After the loaded Continue or recover / Blocked recovery steps reach a final outcome, finish the required invocation state and final report before optional narration or formatting. Append status, reasons, evidence paths, pending work and any accepted post-child hashes/documentation_section in one finishing entry; do not repeat earlier gate analysis or split that known outcome across multiple edits. Append the finishing checkpoint and Write the complete report in the same response using separate native file calls, then return briefly after any authorized publication receipt. Follow the loaded gate order: when it requires stopping, write the required invocation state and report, then stop rather than continuing later preparation to fill optional artifacts. Never omit the final report or final response, even when publication is blocked.
|
||||||
|
|
||||||
dispatch returns terminal final text, a launch error, or a running task_id. Terminal final text means that child is settled. A launch error saying no child started is authoritative and returns no task handle: do not probe invented ids. Use status/stop only with an actual returned task_id. The virtual clock advances to the next policy deadline on each status query; do not sleep. A stop request alone is not settlement or permission to publish. An independent fixture actor may change selected source between phases. Do not read/edit ${stateFile}; it is private transport state. Only when the actual workflow permits publication, call publish, a local receipt rather than GitHub.
|
dispatch returns terminal final text, a launch error, or a running task_id. Terminal final text means that child is settled. A launch error saying no child started is authoritative and returns no task handle: do not probe invented ids. Use status/stop only with an actual returned task_id. The virtual clock advances to the next policy deadline on each status query; do not sleep. A stop request alone is not settlement or permission to publish. An independent fixture actor may change selected source between phases. Do not read/edit ${stateFile}; it is private transport state. Only when the actual workflow permits publication, call publish, a local receipt rather than GitHub.
|
||||||
@@ -120,6 +124,7 @@ export async function runShipDocsFault(testName: string, scenario: DocsFault, co
|
|||||||
const fixture = fixtureDocs(scenario === 'legacy-completion' ? 'legacy' : 'current');
|
const fixture = fixtureDocs(scenario === 'legacy-completion' ? 'legacy' : 'current');
|
||||||
const actorFile = path.join(import.meta.dir, 'docsync-fault-actor.ts');
|
const actorFile = path.join(import.meta.dir, 'docsync-fault-actor.ts');
|
||||||
const stateFile = installDocsActor(fixture, scenario);
|
const stateFile = installDocsActor(fixture, scenario);
|
||||||
|
const seed = docsActorSeeded(scenario) ? seedDocsFirstAttempt(fixture, stateFile) : undefined;
|
||||||
const report = path.join(fixture.home, 'ship-report.md');
|
const report = path.join(fixture.home, 'ship-report.md');
|
||||||
const phase = path.join(fixture.home, 'phase.md');
|
const phase = path.join(fixture.home, 'phase.md');
|
||||||
const skeleton = fs.readFileSync(path.join(fixture.skills, 'ship/SKILL.md'), 'utf8');
|
const skeleton = fs.readFileSync(path.join(fixture.skills, 'ship/SKILL.md'), 'utf8');
|
||||||
@@ -134,7 +139,7 @@ export async function runShipDocsFault(testName: string, scenario: DocsFault, co
|
|||||||
let passed = false;
|
let passed = false;
|
||||||
try {
|
try {
|
||||||
result = await runSkillTest({
|
result = await runSkillTest({
|
||||||
prompt: docsFaultPrompt(fixture, phase, report, scenario),
|
prompt: docsFaultPrompt(fixture, phase, report, scenario, seed),
|
||||||
workingDirectory: fixture.repo, maxTurns: scenario === 'legacy-completion' ? 30 : 24,
|
workingDirectory: fixture.repo, maxTurns: scenario === 'legacy-completion' ? 30 : 24,
|
||||||
tools: ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep'],
|
tools: ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep'],
|
||||||
allowedTools: ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep'],
|
allowedTools: ['Bash', 'Read', 'Write', 'Edit', 'Glob', 'Grep'],
|
||||||
@@ -145,7 +150,9 @@ export async function runShipDocsFault(testName: string, scenario: DocsFault, co
|
|||||||
expect(docsCompletedRead(result, phase, fixture)).toBe(true);
|
expect(docsCompletedRead(result, phase, fixture)).toBe(true);
|
||||||
const documentation = path.join(fixture.skills, 'ship/sections/documentation.md');
|
const documentation = path.join(fixture.skills, 'ship/sections/documentation.md');
|
||||||
expect(docsCompletedRead(result, documentation, fixture)).toBe(true);
|
expect(docsCompletedRead(result, documentation, fixture)).toBe(true);
|
||||||
|
if (seed) expect(docsCompletedRead(result, seed.completion, fixture)).toBe(true);
|
||||||
const state = JSON.parse(fs.readFileSync(stateFile, 'utf8')) as DocsActorState;
|
const state = JSON.parse(fs.readFileSync(stateFile, 'utf8')) as DocsActorState;
|
||||||
|
const parentEvents = state.events.slice(seed?.events ?? 0);
|
||||||
const summary = fs.readFileSync(report, 'utf8');
|
const summary = fs.readFileSync(report, 'utf8');
|
||||||
expect(docsActorVerdict(state, summary, fs.existsSync(path.join(fixture.home, 'publication.json')))).toEqual([]);
|
expect(docsActorVerdict(state, summary, fs.existsSync(path.join(fixture.home, 'publication.json')))).toEqual([]);
|
||||||
expect(docsToolFailures(result, fixture, [actorFile])).toEqual([]);
|
expect(docsToolFailures(result, fixture, [actorFile])).toEqual([]);
|
||||||
@@ -175,16 +182,18 @@ export async function runShipDocsFault(testName: string, scenario: DocsFault, co
|
|||||||
const events = result.toolCalls.filter(call => call.tool === 'Bash' && String(call.input?.command).includes(actorFile));
|
const events = result.toolCalls.filter(call => call.tool === 'Bash' && String(call.input?.command).includes(actorFile));
|
||||||
for (const action of ['prepare', 'dispatch', 'inspect', 'status', 'stop', 'repair', 'publish']) {
|
for (const action of ['prepare', 'dispatch', 'inspect', 'status', 'stop', 'repair', 'publish']) {
|
||||||
expect(events.filter(call => String(call.input?.command).replaceAll("'", '').replaceAll('"', '').includes(` ${action} `)).length)
|
expect(events.filter(call => String(call.input?.command).replaceAll("'", '').replaceAll('"', '').includes(` ${action} `)).length)
|
||||||
.toBe(state.events.filter(e => e.action === action).length);
|
.toBe(parentEvents.filter(e => e.action === action).length);
|
||||||
}
|
}
|
||||||
const inspectCalls = events.filter(call => String(call.input?.command).replaceAll("'", '').replaceAll('"', '').includes(' inspect '));
|
const inspectCalls = events.filter(call => String(call.input?.command).replaceAll("'", '').replaceAll('"', '').includes(' inspect '));
|
||||||
const inspectReceipts = state.events.filter(e => e.action === 'inspect').map(e => e.detail);
|
const inspectReceipts = parentEvents.filter(e => e.action === 'inspect').map(e => e.detail);
|
||||||
expect(inspectCalls.length).toBe(inspectReceipts.length);
|
expect(inspectCalls.length).toBe(inspectReceipts.length);
|
||||||
inspectCalls.forEach((call, index) => {
|
inspectCalls.forEach((call, index) => {
|
||||||
const emitted = call.output.trim().split('\n').at(-1) ?? '';
|
const emitted = call.output.trim().split('\n').at(-1) ?? '';
|
||||||
expect(createHash('sha256').update(emitted).digest('hex')).toBe(inspectReceipts[index]);
|
expect(createHash('sha256').update(emitted).digest('hex')).toBe(inspectReceipts[index]);
|
||||||
});
|
});
|
||||||
if (scenario !== 'missing-asset') expect(events.some(call => call.output.includes('SESSION_KIND:') || call.output.includes('task_id') || call.output.includes('Child launch failed'))).toBe(true);
|
if (scenario !== 'missing-asset' && (!seed || parentEvents.some(e => e.action === 'dispatch'))) {
|
||||||
|
expect(events.some(call => call.output.includes('SESSION_KIND:') || call.output.includes('task_id') || call.output.includes('Child launch failed'))).toBe(true);
|
||||||
|
}
|
||||||
passed = true;
|
passed = true;
|
||||||
} finally {
|
} finally {
|
||||||
const observation = observer.stop();
|
const observation = observer.stop();
|
||||||
|
|||||||
Reference in new issue
Block a user