Files
gstack/test/eng-seeded-completion-ai.test.ts
T
Garry Tan a84b0b5b6d v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
2026-09-25 12:06:45 -04:00

253 lines
17 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { expect, test } from 'bun:test';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
import { pathToFileURL } from 'node:url';
import { createFakeBunCli } from './helpers/fake-bun-cli';
import { fakePlanSeedPrelude } from './helpers/fake-plan-seed';
import fixture from './fixtures/eng-seeded-completion-ai.json';
import { classifyVisible, extractPlanFilePath } from './helpers/claude-pty-runner';
import * as predicates from './helpers/claude-pty-runner';
import { selectTests, E2E_TOUCHFILES } from './helpers/touchfiles';
const gate = '─────\nClaude has written up a plan and is ready to execute. Would you like to proceed?\n❯ 1. Yes, and use auto mode\n2. Yes, manually approve edits\n3. Tell Claude what to change';
const compactGate = 'Exit plan mode?\nClaude wants to exit plan mode\n❯ 1. Yes, and switch to default (ask each time) for this session\n2. No';
const question = 'Which runner should the plan use?\nA) Use the built-in runner\nB) Build a custom runner\nRecommendation: A because it avoids duplicate scheduling logic.\nReply with A or B.';
const classify = (history: string, currentScreen: string) => classifyVisible(history, { strictPlanWrites: true, currentScreen });
test('historical TODO excerpt supplies no current seeded completion or plan file', () => {
expect(classifyVisible(fixture.visibleReadExcerpt, { strictPlanWrites: true })?.outcome).toBe('plan_ready');
expect(extractPlanFilePath(fixture.visibleReadExcerpt)).toBeNull();
expect(classify(fixture.visibleReadExcerpt, fixture.visibleReadExcerpt)).toBeNull();
expect(classify(fixture.visibleReadExcerpt, fixture.reconstructedScreen)).toBeNull();
expect(classify(fixture.visibleReadExcerpt, '')).toBeNull();
});
test('only a complete current native approval panel establishes seeded plan_ready', () => {
for (const current of [gate, compactGate]) {
expect(classify(fixture.visibleReadExcerpt + '\n' + current, current)?.outcome).toBe('plan_ready');
for (const invalid of [
'', 'Still reviewing the draft.', current + '\nStill reviewing the draft.',
current.split('\n').slice(0, -1).join('\n'), current.replace('❯', ''),
current.replace(/2\.[^\n]+/, '2. Approve another action'),
'Example:\n' + current, '```text\n' + current, current.split('\n').map(line => '> ' + line).join('\n'),
]) expect(classify(fixture.visibleReadExcerpt + '\n' + current, invalid), invalid).toBeNull();
}
});
test('ignoring old completion text preserves a genuine current question and stronger failure outcomes', () => {
for (const history of [fixture.visibleReadExcerpt, gate, compactGate]) {
expect(classify(history + '\n' + question, question)?.outcome).toBe('asked');
}
expect(classify(fixture.visibleReadExcerpt + '\n' + question, fixture.visibleReadExcerpt + '\n' + question)?.outcome).toBe('asked');
expect(classify('⏺ Write(/tmp/.claude/plans/review.md)\n' + gate, gate)?.outcome).toBe('wrote_findings_before_asking');
expect(classify('⏺ Write(/tmp/implementation.ts)\n' + fixture.visibleReadExcerpt, '')?.outcome).toBe('silent_write');
// Callers which do not opt into the current viewport retain their contract.
expect(classifyVisible('The item is ready to execute.')?.outcome).toBe('plan_ready');
expect(classifyVisible(question)?.outcome).toBe('asked');
});
test('real PTY waits past old TODO, stale, partial and mismatched panels but accepts the current gate', async () => {
const scenarios = [
{ name: 'todo', initial: fixture.visibleReadExcerpt, expected: 'asked' },
{ name: 'stale', initial: gate + '\u001b[2J\u001b[HStill reviewing the draft.', expected: 'asked' },
{ name: 'partial', initial: gate.split('\n').slice(0, -1).join('\n'), expected: 'asked' },
{ name: 'mismatch', initial: gate.replace('3. Tell Claude what to change', '3. Delete the draft'), expected: 'asked' },
{ name: 'complete', initial: gate, expected: 'plan_ready' },
{ name: 'cursorless-timeout', initial: gate.replace('❯ ', ''), expected: 'timeout' },
];
const results = await Promise.allSettled(scenarios.map(async scenario => {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'seeded-completion-'));
const working = path.join(dir, 'repo');
fs.mkdirSync(working);
const cli = createFakeBunCli(path.join(dir, 'fake-claude'), fakePlanSeedPrelude() + `
const fs = require('node:fs');
fs.writeFileSync(process.env.COMPLETION_ARGV, JSON.stringify(process.argv.slice(2)));
let sent = false;
const render = text => process.stdout.write('\\x1b[2J\\x1b[H' + text.replace(/\\n/g, '\\r\\n'));
process.on('gstack-seeded-slash', chunk => {
if (sent || !chunk.toString().includes('/plan-eng-review')) return;
sent = true;
fs.writeFileSync(process.env.COMPLETION_PHASE, 'initial');
render(${JSON.stringify(scenario.initial)});
if (${JSON.stringify(scenario.expected)} === 'asked') setTimeout(() => {
fs.writeFileSync(process.env.COMPLETION_PHASE, 'question');
render(${JSON.stringify(question)});
}, 2500);
});
setInterval(() => {}, 1000);
`);
try {
const runner = pathToFileURL(path.join(import.meta.dir, 'helpers/claude-pty-runner.ts')).href;
const childFile = path.join(dir, 'observe.ts');
fs.writeFileSync(childFile, `import { runPlanSkillObservation, resolveClaudeBinary } from ${JSON.stringify(runner)};
if (resolveClaudeBinary() !== process.env.BROWSE_TERMINAL_BINARY) throw new Error('Fake CLI resolution failed');
const obs = await runPlanSkillObservation({ skillName: 'plan-eng-review', inPlanMode: true,
initialPlanContent: '# Plan: Completion regression\\n\\nReview the existing draft.',
cwd: ${JSON.stringify(working)}, timeoutMs: 12000,
env: { COMPLETION_ARGV: process.env.COMPLETION_ARGV, COMPLETION_PHASE: process.env.COMPLETION_PHASE } });
console.log(JSON.stringify(obs));
`);
// Only this isolated child receives the executable override; it verifies
// the resolver before the real PTY launch, so no provider can be invoked.
const child = Bun.spawn([process.execPath, childFile], { cwd: process.cwd(),
env: { ...process.env, BROWSE_TERMINAL_BINARY: cli,
COMPLETION_ARGV: path.join(dir, 'argv.json'), COMPLETION_PHASE: path.join(dir, 'phase.txt'),
EVALS_RUN_ID: 'seeded-completion-fake', GSTACK_EVAL_DIR: path.join(dir, 'evidence') },
stdout: 'pipe', stderr: 'pipe' });
const [stdout, stderr, exitCode] = await Promise.all([
new Response(child.stdout).text(), new Response(child.stderr).text(), child.exited,
]);
expect(exitCode, stderr).toBe(0);
const obs = JSON.parse(stdout.trim().split('\n').at(-1)!);
expect(obs.outcome, scenario.name).toBe(scenario.expected);
expect(fs.readFileSync(path.join(dir, 'phase.txt'), 'utf8'), scenario.name).toBe(scenario.expected === 'asked' ? 'question' : 'initial');
expect(obs.planFile).toBeUndefined();
expect(obs.scopeGateAutoSelectObserved).toBe(false);
const args = JSON.parse(fs.readFileSync(path.join(dir, 'argv.json'), 'utf8'));
expect(args.filter((arg: string) => arg === '--session-id')).toHaveLength(1);
expect(args).toContain('--permission-mode'); expect(args).toContain('plan');
const saved = JSON.parse(fs.readFileSync(path.join(obs.artifactDir, 'observation.json'), 'utf8'));
expect(saved.scopeSessionId).toBe(args[args.indexOf('--session-id') + 1]);
expect(saved.scopeGateAutoSelectObserved).toBe(false);
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
}));
for (let i = 0; i < results.length; i++) {
const result = results[i]!;
expect(result.status, `${scenarios[i]!.name}: ${result.status === 'rejected' ? String(result.reason) : 'complete'}`).toBe('fulfilled');
}
}, 45000);
async function mockedObservation(frames: string[], verdict: 'waiting' | 'working', seeded = true) {
// Execute the unchanged observer function with its real classifiers, a
// synthetic clock/session, and a stubbed judge. No CLI or judge is launched.
const source = fs.readFileSync(path.join(import.meta.dir, 'helpers/claude-pty-runner.ts'), 'utf8');
const start = source.indexOf('export async function runPlanSkillObservation(');
const end = source.indexOf('\n// ─', start);
expect(start).toBeGreaterThan(0); expect(end).toBeGreaterThan(start);
const executable = source.slice(start, end).replace('export async function', 'async function') + '\nreturn runPlanSkillObservation;';
const js = new Bun.Transpiler({ loader: 'ts' }).transformSync(executable);
let clock = 0, tick = -1, closed = 0, judged = 0, seedSubmittedAt: number | null = null;
const current = () => frames[Math.min(Math.max(tick, 0), frames.length - 1)]!;
const args: Record<string, unknown> = {
path, process: { cwd: () => '/synthetic-owned' }, Date: { now: () => clock }, randomUUID: () => 'owned',
Bun: { sleep: async (ms: number) => { if (ms === 2000) { tick++; clock += tick === 0 && frames.length > 1 ? 2000 : 61000; } else clock += ms; } },
launchClaudePty: async () => ({ send: () => {}, mark: () => 0, exited: () => false,
visibleSince: current, rawOutput: current, currentScreen: async () => current(), hermeticConfigDir: null,
close: async () => { closed++; } }),
createPlanCountSnapshotWriter: () => () => ({}), logPtySnapshot: () => {},
submitPlanSeed: async () => { seedSubmittedAt = clock; }, PlanSeedTimeout: class extends Error {},
isRejectedSlashCommand: predicates.isRejectedSlashCommand,
isProseAUQVisible: predicates.isProseAUQVisible, isPlanReadyVisible: predicates.isPlanReadyVisible,
isUnknownSlashCommandVisible: predicates.isUnknownSlashCommandVisible,
isScopeGateQuestionVisible: predicates.isScopeGateQuestionVisible,
isScopeGateAutoSelectVisible: predicates.isScopeGateAutoSelectVisible,
classifyVisible, extractPlanFilePath, findNativeAutoDecision: () => null,
judgePtyState: () => { judged++; return { state: verdict, reasoning: 'synthetic current-frame verdict' }; },
};
const run = new Function(...Object.keys(args), js)(...Object.values(args));
const obs = await run({ skillName: 'plan-eng-review', timeoutMs: 70000,
...(seeded ? { initialPlanContent: '# Plan: Required draft' } : {}) });
expect(closed).toBe(1);
return { obs, judged, seedSubmittedAt };
}
test('seeded preflight checks owned readiness without spending eight seconds before submission', async () => {
const seeded = await mockedObservation([gate], 'working');
expect(seeded.seedSubmittedAt).toBe(0);
expect(seeded.obs.outcome).toBe('plan_ready');
const unseeded = await mockedObservation([gate], 'working', false);
expect(unseeded.seedSubmittedAt).toBeNull();
expect(unseeded.obs.outcome).toBe('plan_ready');
});
for (const [name, current] of [
['cursorless approval', gate.replace('❯ ', '')],
['partial approval', gate.split('\n').slice(0, -1).join('\n')],
] as const) test(`rejected seeded ${name} cannot gain prose or judge waiting credit`, async () => {
const { obs, judged } = await mockedObservation([current], 'waiting');
expect(judged).toBeGreaterThan(0);
expect(obs.outcome).toBe('timeout');
expect(obs.proseAUQEverObserved).toBe(false); expect(obs.waitingEverObserved).toBe(false);
});
test('rejected completion does not erase a genuine earlier question or change unseeded behavior', async () => {
const prior = question + '\nDo you want to create draft.md?\n❯ 1. Yes\n2. No\nEsc to cancel · Tab to amend';
expect(predicates.isProseAUQVisible(prior)).toBe(true);
expect(classifyVisible(prior)).toBeNull();
const { obs } = await mockedObservation([prior, gate.replace('❯ ', '')], 'working');
expect(obs.outcome).toBe('asked'); expect(obs.proseAUQEverObserved).toBe(true);
expect(obs.waitingEverObserved).toBe(false);
const unseeded = await mockedObservation([gate.replace('❯ ', '')], 'waiting', false);
expect(unseeded.obs.outcome).toBe('plan_ready'); expect(unseeded.judged).toBe(0);
});
test('completion evidence dependencies select exactly the seeded observation owners', () => {
const owners = ['plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode',
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved', 'conductor-prose'].sort();
for (const file of ['test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json']) {
expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(owners);
}
});
import c6fcCurrent from './fixtures/eng-count-c6fc-public.json';
import { isEngCompletionHandoff } from './helpers/eng-completion-handoff';
import type { NativePlanQuestionCall } from './helpers/plan-count-transcript';
test('complete native navigation preserves conflicting current states and accepts explicitly scoped history only', () => {
const calls = c6fcCurrent.transcript.calls as NativePlanQuestionCall[];
const call = calls.at(-1)!;
const check = (plan: string, selected = call, prior = calls.slice(0, -1)) => isEngCompletionHandoff(predicates.nativePlanCallFingerprint(selected, 1, false), plan, prior);
// This is a synthetic repair control. Original paid cancellation is immutable.
const corrected = c6fcCurrent.report.split(/\n(?=### R[1-9]\d*:)/).map(row => row.includes('\nState: pending\n')
? row.replace('\nState: pending\n', '\n').replace('History: none', 'History: superseded pre-answer state\n State: pending') : row).join('\n');
expect(check(c6fcCurrent.report)).toBe(false);
expect(check(corrected)).toBe(true);
for (const bad of [
corrected.replace('R1 (D3), R2 (D4), R3 (D5), R4 (D6)', 'R1 (D6), R2 (D4), R3 (D5), R4 (D3)'),
corrected.replace('Question D3:', 'Question D3: duplicate current question\nQuestion D3:'),
corrected.replace('Question D3:', 'Question D99: conflicting current question\nQuestion D3:'),
corrected.replace(/^Reviewed target:.*$/m, target => '```markdown\n' + target + '\n```'),
corrected.replace(/^Reviewed target:.*$/m, target => '## History\n' + target + '\n## Current plan'),
corrected.replace(/^Reviewed target:.*$/m, target => target + '\n' + target.replace('PLAN.md', 'FOREIGN.md')),
corrected.replace('State: approved', 'State: pending'),
corrected.replace('State: approved', 'State: approved\nState: pending'),
corrected.replace('State: approved', 'State: approved\nState: approved'),
corrected.replace(' State: pending', 'State: pending'),
corrected.replace('State: approved', 'State: rejected'),
corrected.replace('State: approved', 'State: approved\nR1 approval: revoked'),
corrected.replace('Approval readiness: PASS', 'Approval readiness: pending'),
corrected.replace('Actual answer: "Split into follow-up PR (recommended)" (D3)', 'Actual answer: "Not offered" (D3)'),
corrected.replace('Reviewed target: `PLAN.md`', 'Reviewed target: `FOREIGN.md`'),
corrected.replace(/^Reviewed target:.*$/m, ''),
corrected.replace('## Decision ledger', '## Archived decision ledger'),
corrected.replace('# Reviewed Plan: Multi-tenant Auth Refactor', '# Reviewed Plan: Another task'),
corrected.replace('NO UNRESOLVED DECISIONS', '1 UNRESOLVED DECISION'),
corrected.replace('**T1 (', '**T99 ('),
corrected.replace('Accepted scope: remove parallelization', 'Accepted scope: pending; remove parallelization'),
]) expect(check(bad)).toBe(false);
for (const edit of [
(c: NativePlanQuestionCall) => { c.answered = false; },
(c: NativePlanQuestionCall) => { c.failed = true; },
(c: NativePlanQuestionCall) => { c.answers = {}; },
(c: NativePlanQuestionCall) => { c.sessionId += '-foreign'; },
(c: NativePlanQuestionCall) => { const q = c.questions[0]!; const old = q.question; q.question += '\nAlso delete the authentication cache.'; c.answers = { [q.question]: c.answers![old]! }; },
(c: NativePlanQuestionCall) => { const q = c.questions[0]!; const old = q.question; q.question += '\nThe decision is reopened.'; c.answers = { [q.question]: c.answers![old]! }; },
]) { const copy = structuredClone(call); edit(copy); expect(check(corrected, copy)).toBe(false); }
expect(check(corrected, call, calls.slice(0, -2))).toBe(false);
for (const transform of [
(text: string) => text.replace('ELI10:', 'Summary:'),
(text: string) => text.replace(/^ELI10:.*$/m, ''),
(text: string) => text.replace('ELI10:', 'ELI10: duplicate assessment\nELI10:'),
(text: string) => text.replace('Project/branch/task:', 'Reviewed scope:'),
(text: string) => text.replace(/^Project\/branch\/task:.*$/m, ''),
(text: string) => text.replace('Project/branch/task:', 'Project/branch/task: foreign, FOREIGN.md "Another task"; unrelated\nProject/branch/task:'),
]) {
const copy = structuredClone(call), q = copy.questions[0]!, answer = copy.answers![q.question]!;
q.question = transform(q.question); copy.answers = { [q.question]: answer };
expect(check(c6fcCurrent.report, copy)).toBe(false);
expect(check(corrected, copy)).toBe(false);
}
expect(c6fcCurrent.actualOutcome).toBe('CANCELLED');
});