mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 07:32:14 +02:00
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
253 lines
17 KiB
TypeScript
253 lines
17 KiB
TypeScript
import { expect, test } from 'bun:test';
|
||
import * as fs from 'node:fs';
|
||
import * as os from 'node:os';
|
||
import * as path from 'node:path';
|
||
import { pathToFileURL } from 'node:url';
|
||
import { createFakeBunCli } from './helpers/fake-bun-cli';
|
||
import { fakePlanSeedPrelude } from './helpers/fake-plan-seed';
|
||
import fixture from './fixtures/eng-seeded-completion-ai.json';
|
||
import { classifyVisible, extractPlanFilePath } from './helpers/claude-pty-runner';
|
||
import * as predicates from './helpers/claude-pty-runner';
|
||
import { selectTests, E2E_TOUCHFILES } from './helpers/touchfiles';
|
||
|
||
const gate = '─────\nClaude has written up a plan and is ready to execute. Would you like to proceed?\n❯ 1. Yes, and use auto mode\n2. Yes, manually approve edits\n3. Tell Claude what to change';
|
||
const compactGate = 'Exit plan mode?\nClaude wants to exit plan mode\n❯ 1. Yes, and switch to default (ask each time) for this session\n2. No';
|
||
const question = 'Which runner should the plan use?\nA) Use the built-in runner\nB) Build a custom runner\nRecommendation: A because it avoids duplicate scheduling logic.\nReply with A or B.';
|
||
const classify = (history: string, currentScreen: string) => classifyVisible(history, { strictPlanWrites: true, currentScreen });
|
||
|
||
test('historical TODO excerpt supplies no current seeded completion or plan file', () => {
|
||
expect(classifyVisible(fixture.visibleReadExcerpt, { strictPlanWrites: true })?.outcome).toBe('plan_ready');
|
||
expect(extractPlanFilePath(fixture.visibleReadExcerpt)).toBeNull();
|
||
expect(classify(fixture.visibleReadExcerpt, fixture.visibleReadExcerpt)).toBeNull();
|
||
expect(classify(fixture.visibleReadExcerpt, fixture.reconstructedScreen)).toBeNull();
|
||
expect(classify(fixture.visibleReadExcerpt, '')).toBeNull();
|
||
});
|
||
|
||
test('only a complete current native approval panel establishes seeded plan_ready', () => {
|
||
for (const current of [gate, compactGate]) {
|
||
expect(classify(fixture.visibleReadExcerpt + '\n' + current, current)?.outcome).toBe('plan_ready');
|
||
for (const invalid of [
|
||
'', 'Still reviewing the draft.', current + '\nStill reviewing the draft.',
|
||
current.split('\n').slice(0, -1).join('\n'), current.replace('❯', ''),
|
||
current.replace(/2\.[^\n]+/, '2. Approve another action'),
|
||
'Example:\n' + current, '```text\n' + current, current.split('\n').map(line => '> ' + line).join('\n'),
|
||
]) expect(classify(fixture.visibleReadExcerpt + '\n' + current, invalid), invalid).toBeNull();
|
||
}
|
||
});
|
||
|
||
test('ignoring old completion text preserves a genuine current question and stronger failure outcomes', () => {
|
||
for (const history of [fixture.visibleReadExcerpt, gate, compactGate]) {
|
||
expect(classify(history + '\n' + question, question)?.outcome).toBe('asked');
|
||
}
|
||
expect(classify(fixture.visibleReadExcerpt + '\n' + question, fixture.visibleReadExcerpt + '\n' + question)?.outcome).toBe('asked');
|
||
expect(classify('⏺ Write(/tmp/.claude/plans/review.md)\n' + gate, gate)?.outcome).toBe('wrote_findings_before_asking');
|
||
expect(classify('⏺ Write(/tmp/implementation.ts)\n' + fixture.visibleReadExcerpt, '')?.outcome).toBe('silent_write');
|
||
// Callers which do not opt into the current viewport retain their contract.
|
||
expect(classifyVisible('The item is ready to execute.')?.outcome).toBe('plan_ready');
|
||
expect(classifyVisible(question)?.outcome).toBe('asked');
|
||
});
|
||
|
||
test('real PTY waits past old TODO, stale, partial and mismatched panels but accepts the current gate', async () => {
|
||
const scenarios = [
|
||
{ name: 'todo', initial: fixture.visibleReadExcerpt, expected: 'asked' },
|
||
{ name: 'stale', initial: gate + '\u001b[2J\u001b[HStill reviewing the draft.', expected: 'asked' },
|
||
{ name: 'partial', initial: gate.split('\n').slice(0, -1).join('\n'), expected: 'asked' },
|
||
{ name: 'mismatch', initial: gate.replace('3. Tell Claude what to change', '3. Delete the draft'), expected: 'asked' },
|
||
{ name: 'complete', initial: gate, expected: 'plan_ready' },
|
||
{ name: 'cursorless-timeout', initial: gate.replace('❯ ', ''), expected: 'timeout' },
|
||
];
|
||
const results = await Promise.allSettled(scenarios.map(async scenario => {
|
||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'seeded-completion-'));
|
||
const working = path.join(dir, 'repo');
|
||
fs.mkdirSync(working);
|
||
const cli = createFakeBunCli(path.join(dir, 'fake-claude'), fakePlanSeedPrelude() + `
|
||
const fs = require('node:fs');
|
||
fs.writeFileSync(process.env.COMPLETION_ARGV, JSON.stringify(process.argv.slice(2)));
|
||
let sent = false;
|
||
const render = text => process.stdout.write('\\x1b[2J\\x1b[H' + text.replace(/\\n/g, '\\r\\n'));
|
||
process.on('gstack-seeded-slash', chunk => {
|
||
if (sent || !chunk.toString().includes('/plan-eng-review')) return;
|
||
sent = true;
|
||
fs.writeFileSync(process.env.COMPLETION_PHASE, 'initial');
|
||
render(${JSON.stringify(scenario.initial)});
|
||
if (${JSON.stringify(scenario.expected)} === 'asked') setTimeout(() => {
|
||
fs.writeFileSync(process.env.COMPLETION_PHASE, 'question');
|
||
render(${JSON.stringify(question)});
|
||
}, 2500);
|
||
});
|
||
setInterval(() => {}, 1000);
|
||
`);
|
||
try {
|
||
const runner = pathToFileURL(path.join(import.meta.dir, 'helpers/claude-pty-runner.ts')).href;
|
||
const childFile = path.join(dir, 'observe.ts');
|
||
fs.writeFileSync(childFile, `import { runPlanSkillObservation, resolveClaudeBinary } from ${JSON.stringify(runner)};
|
||
if (resolveClaudeBinary() !== process.env.BROWSE_TERMINAL_BINARY) throw new Error('Fake CLI resolution failed');
|
||
const obs = await runPlanSkillObservation({ skillName: 'plan-eng-review', inPlanMode: true,
|
||
initialPlanContent: '# Plan: Completion regression\\n\\nReview the existing draft.',
|
||
cwd: ${JSON.stringify(working)}, timeoutMs: 12000,
|
||
env: { COMPLETION_ARGV: process.env.COMPLETION_ARGV, COMPLETION_PHASE: process.env.COMPLETION_PHASE } });
|
||
console.log(JSON.stringify(obs));
|
||
`);
|
||
// Only this isolated child receives the executable override; it verifies
|
||
// the resolver before the real PTY launch, so no provider can be invoked.
|
||
const child = Bun.spawn([process.execPath, childFile], { cwd: process.cwd(),
|
||
env: { ...process.env, BROWSE_TERMINAL_BINARY: cli,
|
||
COMPLETION_ARGV: path.join(dir, 'argv.json'), COMPLETION_PHASE: path.join(dir, 'phase.txt'),
|
||
EVALS_RUN_ID: 'seeded-completion-fake', GSTACK_EVAL_DIR: path.join(dir, 'evidence') },
|
||
stdout: 'pipe', stderr: 'pipe' });
|
||
const [stdout, stderr, exitCode] = await Promise.all([
|
||
new Response(child.stdout).text(), new Response(child.stderr).text(), child.exited,
|
||
]);
|
||
expect(exitCode, stderr).toBe(0);
|
||
const obs = JSON.parse(stdout.trim().split('\n').at(-1)!);
|
||
expect(obs.outcome, scenario.name).toBe(scenario.expected);
|
||
expect(fs.readFileSync(path.join(dir, 'phase.txt'), 'utf8'), scenario.name).toBe(scenario.expected === 'asked' ? 'question' : 'initial');
|
||
expect(obs.planFile).toBeUndefined();
|
||
expect(obs.scopeGateAutoSelectObserved).toBe(false);
|
||
const args = JSON.parse(fs.readFileSync(path.join(dir, 'argv.json'), 'utf8'));
|
||
expect(args.filter((arg: string) => arg === '--session-id')).toHaveLength(1);
|
||
expect(args).toContain('--permission-mode'); expect(args).toContain('plan');
|
||
const saved = JSON.parse(fs.readFileSync(path.join(obs.artifactDir, 'observation.json'), 'utf8'));
|
||
expect(saved.scopeSessionId).toBe(args[args.indexOf('--session-id') + 1]);
|
||
expect(saved.scopeGateAutoSelectObserved).toBe(false);
|
||
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
||
}));
|
||
for (let i = 0; i < results.length; i++) {
|
||
const result = results[i]!;
|
||
expect(result.status, `${scenarios[i]!.name}: ${result.status === 'rejected' ? String(result.reason) : 'complete'}`).toBe('fulfilled');
|
||
}
|
||
}, 45000);
|
||
|
||
async function mockedObservation(frames: string[], verdict: 'waiting' | 'working', seeded = true) {
|
||
// Execute the unchanged observer function with its real classifiers, a
|
||
// synthetic clock/session, and a stubbed judge. No CLI or judge is launched.
|
||
const source = fs.readFileSync(path.join(import.meta.dir, 'helpers/claude-pty-runner.ts'), 'utf8');
|
||
const start = source.indexOf('export async function runPlanSkillObservation(');
|
||
const end = source.indexOf('\n// ─', start);
|
||
expect(start).toBeGreaterThan(0); expect(end).toBeGreaterThan(start);
|
||
const executable = source.slice(start, end).replace('export async function', 'async function') + '\nreturn runPlanSkillObservation;';
|
||
const js = new Bun.Transpiler({ loader: 'ts' }).transformSync(executable);
|
||
let clock = 0, tick = -1, closed = 0, judged = 0, seedSubmittedAt: number | null = null;
|
||
const current = () => frames[Math.min(Math.max(tick, 0), frames.length - 1)]!;
|
||
const args: Record<string, unknown> = {
|
||
path, process: { cwd: () => '/synthetic-owned' }, Date: { now: () => clock }, randomUUID: () => 'owned',
|
||
Bun: { sleep: async (ms: number) => { if (ms === 2000) { tick++; clock += tick === 0 && frames.length > 1 ? 2000 : 61000; } else clock += ms; } },
|
||
launchClaudePty: async () => ({ send: () => {}, mark: () => 0, exited: () => false,
|
||
visibleSince: current, rawOutput: current, currentScreen: async () => current(), hermeticConfigDir: null,
|
||
close: async () => { closed++; } }),
|
||
createPlanCountSnapshotWriter: () => () => ({}), logPtySnapshot: () => {},
|
||
submitPlanSeed: async () => { seedSubmittedAt = clock; }, PlanSeedTimeout: class extends Error {},
|
||
isRejectedSlashCommand: predicates.isRejectedSlashCommand,
|
||
isProseAUQVisible: predicates.isProseAUQVisible, isPlanReadyVisible: predicates.isPlanReadyVisible,
|
||
isUnknownSlashCommandVisible: predicates.isUnknownSlashCommandVisible,
|
||
isScopeGateQuestionVisible: predicates.isScopeGateQuestionVisible,
|
||
isScopeGateAutoSelectVisible: predicates.isScopeGateAutoSelectVisible,
|
||
classifyVisible, extractPlanFilePath, findNativeAutoDecision: () => null,
|
||
judgePtyState: () => { judged++; return { state: verdict, reasoning: 'synthetic current-frame verdict' }; },
|
||
};
|
||
const run = new Function(...Object.keys(args), js)(...Object.values(args));
|
||
const obs = await run({ skillName: 'plan-eng-review', timeoutMs: 70000,
|
||
...(seeded ? { initialPlanContent: '# Plan: Required draft' } : {}) });
|
||
expect(closed).toBe(1);
|
||
return { obs, judged, seedSubmittedAt };
|
||
}
|
||
|
||
test('seeded preflight checks owned readiness without spending eight seconds before submission', async () => {
|
||
const seeded = await mockedObservation([gate], 'working');
|
||
expect(seeded.seedSubmittedAt).toBe(0);
|
||
expect(seeded.obs.outcome).toBe('plan_ready');
|
||
const unseeded = await mockedObservation([gate], 'working', false);
|
||
expect(unseeded.seedSubmittedAt).toBeNull();
|
||
expect(unseeded.obs.outcome).toBe('plan_ready');
|
||
});
|
||
|
||
for (const [name, current] of [
|
||
['cursorless approval', gate.replace('❯ ', '')],
|
||
['partial approval', gate.split('\n').slice(0, -1).join('\n')],
|
||
] as const) test(`rejected seeded ${name} cannot gain prose or judge waiting credit`, async () => {
|
||
const { obs, judged } = await mockedObservation([current], 'waiting');
|
||
expect(judged).toBeGreaterThan(0);
|
||
expect(obs.outcome).toBe('timeout');
|
||
expect(obs.proseAUQEverObserved).toBe(false); expect(obs.waitingEverObserved).toBe(false);
|
||
});
|
||
|
||
test('rejected completion does not erase a genuine earlier question or change unseeded behavior', async () => {
|
||
const prior = question + '\nDo you want to create draft.md?\n❯ 1. Yes\n2. No\nEsc to cancel · Tab to amend';
|
||
expect(predicates.isProseAUQVisible(prior)).toBe(true);
|
||
expect(classifyVisible(prior)).toBeNull();
|
||
const { obs } = await mockedObservation([prior, gate.replace('❯ ', '')], 'working');
|
||
expect(obs.outcome).toBe('asked'); expect(obs.proseAUQEverObserved).toBe(true);
|
||
expect(obs.waitingEverObserved).toBe(false);
|
||
const unseeded = await mockedObservation([gate.replace('❯ ', '')], 'waiting', false);
|
||
expect(unseeded.obs.outcome).toBe('plan_ready'); expect(unseeded.judged).toBe(0);
|
||
});
|
||
|
||
test('completion evidence dependencies select exactly the seeded observation owners', () => {
|
||
const owners = ['plan-ceo-review-plan-mode', 'plan-eng-review-plan-mode', 'plan-design-review-plan-mode',
|
||
'plan-devex-review-plan-mode', 'plan-mode-no-op', 'auto-decide-preserved', 'conductor-prose'].sort();
|
||
for (const file of ['test/eng-seeded-completion-ai.test.ts', 'test/fixtures/eng-seeded-completion-ai.json']) {
|
||
expect(selectTests([file], E2E_TOUCHFILES).selected.sort()).toEqual(owners);
|
||
}
|
||
});
|
||
|
||
import c6fcCurrent from './fixtures/eng-count-c6fc-public.json';
|
||
import { isEngCompletionHandoff } from './helpers/eng-completion-handoff';
|
||
import type { NativePlanQuestionCall } from './helpers/plan-count-transcript';
|
||
|
||
test('complete native navigation preserves conflicting current states and accepts explicitly scoped history only', () => {
|
||
const calls = c6fcCurrent.transcript.calls as NativePlanQuestionCall[];
|
||
const call = calls.at(-1)!;
|
||
const check = (plan: string, selected = call, prior = calls.slice(0, -1)) => isEngCompletionHandoff(predicates.nativePlanCallFingerprint(selected, 1, false), plan, prior);
|
||
// This is a synthetic repair control. Original paid cancellation is immutable.
|
||
const corrected = c6fcCurrent.report.split(/\n(?=### R[1-9]\d*:)/).map(row => row.includes('\nState: pending\n')
|
||
? row.replace('\nState: pending\n', '\n').replace('History: none', 'History: superseded pre-answer state\n State: pending') : row).join('\n');
|
||
expect(check(c6fcCurrent.report)).toBe(false);
|
||
expect(check(corrected)).toBe(true);
|
||
for (const bad of [
|
||
corrected.replace('R1 (D3), R2 (D4), R3 (D5), R4 (D6)', 'R1 (D6), R2 (D4), R3 (D5), R4 (D3)'),
|
||
corrected.replace('Question D3:', 'Question D3: duplicate current question\nQuestion D3:'),
|
||
corrected.replace('Question D3:', 'Question D99: conflicting current question\nQuestion D3:'),
|
||
corrected.replace(/^Reviewed target:.*$/m, target => '```markdown\n' + target + '\n```'),
|
||
corrected.replace(/^Reviewed target:.*$/m, target => '## History\n' + target + '\n## Current plan'),
|
||
corrected.replace(/^Reviewed target:.*$/m, target => target + '\n' + target.replace('PLAN.md', 'FOREIGN.md')),
|
||
corrected.replace('State: approved', 'State: pending'),
|
||
corrected.replace('State: approved', 'State: approved\nState: pending'),
|
||
corrected.replace('State: approved', 'State: approved\nState: approved'),
|
||
corrected.replace(' State: pending', 'State: pending'),
|
||
corrected.replace('State: approved', 'State: rejected'),
|
||
corrected.replace('State: approved', 'State: approved\nR1 approval: revoked'),
|
||
corrected.replace('Approval readiness: PASS', 'Approval readiness: pending'),
|
||
corrected.replace('Actual answer: "Split into follow-up PR (recommended)" (D3)', 'Actual answer: "Not offered" (D3)'),
|
||
corrected.replace('Reviewed target: `PLAN.md`', 'Reviewed target: `FOREIGN.md`'),
|
||
corrected.replace(/^Reviewed target:.*$/m, ''),
|
||
corrected.replace('## Decision ledger', '## Archived decision ledger'),
|
||
corrected.replace('# Reviewed Plan: Multi-tenant Auth Refactor', '# Reviewed Plan: Another task'),
|
||
corrected.replace('NO UNRESOLVED DECISIONS', '1 UNRESOLVED DECISION'),
|
||
corrected.replace('**T1 (', '**T99 ('),
|
||
corrected.replace('Accepted scope: remove parallelization', 'Accepted scope: pending; remove parallelization'),
|
||
]) expect(check(bad)).toBe(false);
|
||
for (const edit of [
|
||
(c: NativePlanQuestionCall) => { c.answered = false; },
|
||
(c: NativePlanQuestionCall) => { c.failed = true; },
|
||
(c: NativePlanQuestionCall) => { c.answers = {}; },
|
||
(c: NativePlanQuestionCall) => { c.sessionId += '-foreign'; },
|
||
(c: NativePlanQuestionCall) => { const q = c.questions[0]!; const old = q.question; q.question += '\nAlso delete the authentication cache.'; c.answers = { [q.question]: c.answers![old]! }; },
|
||
(c: NativePlanQuestionCall) => { const q = c.questions[0]!; const old = q.question; q.question += '\nThe decision is reopened.'; c.answers = { [q.question]: c.answers![old]! }; },
|
||
]) { const copy = structuredClone(call); edit(copy); expect(check(corrected, copy)).toBe(false); }
|
||
expect(check(corrected, call, calls.slice(0, -2))).toBe(false);
|
||
for (const transform of [
|
||
(text: string) => text.replace('ELI10:', 'Summary:'),
|
||
(text: string) => text.replace(/^ELI10:.*$/m, ''),
|
||
(text: string) => text.replace('ELI10:', 'ELI10: duplicate assessment\nELI10:'),
|
||
(text: string) => text.replace('Project/branch/task:', 'Reviewed scope:'),
|
||
(text: string) => text.replace(/^Project\/branch\/task:.*$/m, ''),
|
||
(text: string) => text.replace('Project/branch/task:', 'Project/branch/task: foreign, FOREIGN.md "Another task"; unrelated\nProject/branch/task:'),
|
||
]) {
|
||
const copy = structuredClone(call), q = copy.questions[0]!, answer = copy.answers![q.question]!;
|
||
q.question = transform(q.question); copy.answers = { [q.question]: answer };
|
||
expect(check(c6fcCurrent.report, copy)).toBe(false);
|
||
expect(check(corrected, copy)).toBe(false);
|
||
}
|
||
expect(c6fcCurrent.actualOutcome).toBe('CANCELLED');
|
||
});
|