mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 15:41:57 +02:00
v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
@@ -10,11 +10,13 @@ import {
|
||||
isPartialEval,
|
||||
listEvalJsonFiles,
|
||||
compareEvalResults,
|
||||
evalEntryOutcome,
|
||||
formatComparison,
|
||||
generateCommentary,
|
||||
judgePassed,
|
||||
} from './eval-store';
|
||||
import type { EvalResult, EvalTestEntry, ComparisonResult } from './eval-store';
|
||||
import { manualReviewFixture } from './manual-judge-review-fixture';
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
@@ -77,6 +79,36 @@ async function captureStderr(fn: () => Promise<void>): Promise<string> {
|
||||
// --- EvalCollector tests ---
|
||||
|
||||
describe('EvalCollector', () => {
|
||||
test('manual provider refusal stays unscored and executed; malformed claims stay failed', async () => {
|
||||
const manual = manualReviewFixture();
|
||||
const invalidPass = { ...manual, passed: true };
|
||||
const invalidScore = { ...manual, judge_scores: { clarity: 5 } };
|
||||
expect(evalEntryOutcome(manual)).toBe('manual-review');
|
||||
expect(evalEntryOutcome(invalidPass)).toBe('failed');
|
||||
expect(evalEntryOutcome(invalidScore)).toBe('failed');
|
||||
expect(evalEntryOutcome({ ...manual, manual_review: { ...manual.manual_review, approval: { ...manual.manual_review!.approval,
|
||||
prompt_sha256: '0'.repeat(64) } } })).toBe('failed');
|
||||
const collector = new EvalCollector('llm-judge', tmpDir);
|
||||
collector.addTest(makeEntry({ name: 'ordinary', tier: 'llm-judge' }));
|
||||
collector.addTest(manual);
|
||||
collector.addTest(invalidPass);
|
||||
collector.addTest(invalidScore);
|
||||
const partial: EvalResult = JSON.parse(fs.readFileSync(path.join(tmpDir, '_partial-e2e.json'), 'utf8'));
|
||||
expect(partial).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1,
|
||||
executed_tests: 4, reused_tests: 0 });
|
||||
const output = await captureStderr(async () => { await collector.finalize(); });
|
||||
const result: EvalResult = JSON.parse(fs.readFileSync(fs.readdirSync(tmpDir).map(name => path.join(tmpDir, name))
|
||||
.find(name => name.endsWith('.json') && !path.basename(name).startsWith('_'))!, 'utf8'));
|
||||
expect(result).toMatchObject({ passed: 1, failed: 2, manual_accepted_tests: 1, executed_tests: 4 });
|
||||
expect(result.tests[1]).toMatchObject({ passed: false, execution: 'executed', exit_reason: 'provider_refusal' });
|
||||
expect(result.tests[1].judge_scores).toBeUndefined();
|
||||
expect(output).toContain('MANUAL');
|
||||
expect(output).toContain('1 unscored provider refusal');
|
||||
expect(output).toContain(manual.manual_review!.approval.approval_url);
|
||||
expect(result.tests[2].passed).toBe(true);
|
||||
expect(output).toContain(' FAIL ');
|
||||
});
|
||||
|
||||
test('reused passing evidence preserves origin and remains separate from newly executed attempts', async () => {
|
||||
const collector = new EvalCollector('llm-judge', tmpDir);
|
||||
const reused_from = { input_key: 'a'.repeat(64), run_id: '1234/1', revision: 'b'.repeat(40),
|
||||
@@ -590,6 +622,45 @@ describe('findLatestFinalizedRun', () => {
|
||||
// --- compareEvalResults tests ---
|
||||
|
||||
describe('compareEvalResults', () => {
|
||||
test('manual acceptance is neither a score regression nor a recovery', () => {
|
||||
const manual = manualReviewFixture();
|
||||
const prior = makeResult({ tests: [makeEntry({ name: manual.name, passed: true })] });
|
||||
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
|
||||
const toManual = compareEvalResults(prior, accepted, 'prior.json', 'accepted.json');
|
||||
expect(toManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
|
||||
expect(toManual.deltas[0]).toMatchObject({ status_change: 'manual-review', before: { passed: true },
|
||||
after: { passed: false, manual_review: true } });
|
||||
expect(formatComparison(toManual)).toContain('PASS → MANUAL');
|
||||
expect(formatComparison(toManual)).not.toContain('REGRESSION:');
|
||||
const fromManual = compareEvalResults(accepted, prior, 'accepted.json', 'prior.json');
|
||||
expect(fromManual).toMatchObject({ improved: 0, regressed: 0, manual_reviewed: 1 });
|
||||
expect(formatComparison(fromManual)).toContain('MANUAL → PASS');
|
||||
const invalid = makeResult({ tests: [{ ...manual, passed: true }] });
|
||||
expect(compareEvalResults(prior, invalid, 'prior.json', 'invalid.json').regressed).toBe(1);
|
||||
});
|
||||
|
||||
test('manual acceptance followed by a real or malformed failure is a blocking regression', () => {
|
||||
const manual = manualReviewFixture();
|
||||
const accepted = makeResult({ tests: [manual], passed: 0, failed: 0, manual_accepted_tests: 1 });
|
||||
const { manual_review: _receipt, ...ordinary } = manual;
|
||||
for (const after of [
|
||||
{ ...ordinary, exit_reason: 'timeout' },
|
||||
{ ...manual, passed: true },
|
||||
{ ...manual, judge_scores: { clarity: 5 } },
|
||||
]) {
|
||||
const failed = makeResult({ tests: [after] });
|
||||
const comparison = compareEvalResults(accepted, failed, 'accepted.json', 'failed.json');
|
||||
expect(comparison).toMatchObject({ improved: 0, regressed: 1 });
|
||||
expect(comparison.manual_reviewed).toBeUndefined();
|
||||
expect(comparison.deltas[0]).toMatchObject({ status_change: 'regressed', before: { manual_review: true },
|
||||
after: { passed: false } });
|
||||
const output = formatComparison(comparison);
|
||||
expect(output).toContain('MANUAL → FAIL');
|
||||
expect(output).toContain('REGRESSION:');
|
||||
expect(output).toContain('blocking failure');
|
||||
}
|
||||
});
|
||||
|
||||
test('detects improved/regressed/unchanged per test', () => {
|
||||
const before = makeResult({
|
||||
tests: [
|
||||
|
||||
Reference in New Issue
Block a user