mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 07:32:14 +02:00
v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
@@ -29,11 +29,22 @@ import * as os from 'os';
|
||||
import * as path from 'path';
|
||||
import { runBin } from './helpers/run-bin';
|
||||
import { selectTests, E2E_TOUCHFILES, LLM_JUDGE_TOUCHFILES, GLOBAL_TOUCHFILES } from './helpers/touchfiles';
|
||||
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
|
||||
import { renderDashboard } from '../scripts/eval-watch';
|
||||
|
||||
const ROOT = path.resolve(import.meta.dir, '..');
|
||||
const SCRIPT = (name: string) => path.join(ROOT, 'scripts', name);
|
||||
const SLUG = 'eval-cli-fixture';
|
||||
|
||||
test('eval:watch distinguishes unscored manual acceptance from malformed claims', () => {
|
||||
const manual = manualReviewFixture();
|
||||
const output = renderDashboard(null, { tests: [manual, { ...manual, passed: true }], total_cost_usd: 0 });
|
||||
expect(output).toContain('MANUAL/unscored');
|
||||
expect(output).toContain(manual.manual_review!.approval.approval_url);
|
||||
expect(output).toContain('Manual accepted: 1');
|
||||
expect(output).toContain('✗');
|
||||
});
|
||||
|
||||
let tmpHome: string;
|
||||
let evalDir: string;
|
||||
|
||||
@@ -256,6 +267,22 @@ describe('eval:list CLI (scripts/eval-list.ts)', () => {
|
||||
// ── eval-compare ─────────────────────────────────────────────────────────────
|
||||
|
||||
describe('eval:compare CLI (scripts/eval-compare.ts)', () => {
|
||||
test('shows a manual transition without calling it a scored regression', () => {
|
||||
const manual = manualReviewFixture();
|
||||
const prior = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-01T01:00:00Z',
|
||||
tests: [{ name: manual.name, passed: true }] });
|
||||
const after = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-02T01:00:00Z',
|
||||
tests: [{ name: manual.name, passed: false }] });
|
||||
const body = JSON.parse(fs.readFileSync(after, 'utf8'));
|
||||
body.tests = [manual]; body.passed = 0; body.failed = 0; body.manual_accepted_tests = 1;
|
||||
fs.writeFileSync(after, JSON.stringify(body));
|
||||
const result = runEvalCli('eval-compare.ts', prior, after);
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout).toContain('PASS → MANUAL');
|
||||
expect(result.stdout).toContain('unscored manual review');
|
||||
expect(result.stdout).not.toContain('REGRESSION:');
|
||||
});
|
||||
|
||||
test('empty eval dir prints the getting-started hint and exits 0', () => {
|
||||
const result = runEvalCli('eval-compare.ts');
|
||||
expect(result.status).toBe(0);
|
||||
@@ -337,6 +364,40 @@ describe('eval:compare CLI (scripts/eval-compare.ts)', () => {
|
||||
// ── eval-summary ─────────────────────────────────────────────────────────────
|
||||
|
||||
describe('eval:summary CLI (scripts/eval-summary.ts)', () => {
|
||||
test('reports manual provenance without inventing a scored flake', () => {
|
||||
const manual = manualReviewFixture();
|
||||
writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-01T01:00:00Z',
|
||||
tests: [{ name: manual.name, passed: true }] });
|
||||
const accepted = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-02T01:00:00Z',
|
||||
tests: [{ name: manual.name, passed: false }] });
|
||||
const body = JSON.parse(fs.readFileSync(accepted, 'utf8'));
|
||||
body.tests = [manual]; body.passed = 0; body.failed = 0; body.manual_accepted_tests = 1;
|
||||
fs.writeFileSync(accepted, JSON.stringify(body));
|
||||
const result = runEvalCli('eval-summary.ts');
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout).toContain('Manual accepted: 1 unscored provider refusal');
|
||||
expect(result.stdout).toContain(manual.manual_review!.approval.approval_url);
|
||||
expect(result.stdout).not.toContain('Flaky tests');
|
||||
});
|
||||
|
||||
test('retried manual claims remain failed in summary history', () => {
|
||||
const manual = manualReviewFixture();
|
||||
writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-01T01:00:00Z',
|
||||
tests: [{ name: manual.name, passed: true }] });
|
||||
const attempted = writeRun(evalDir, { tier: 'llm-judge', timestamp: '2026-01-02T01:00:00Z',
|
||||
tests: [{ name: manual.name, passed: false }] });
|
||||
const body = JSON.parse(fs.readFileSync(attempted, 'utf8'));
|
||||
const { manual_review: _receipt, ...ordinary } = manual;
|
||||
body.tests = [{ ...ordinary, attempt: 1 }, { ...manual, attempt: 2 }];
|
||||
body.total_tests = 2; body.passed = 0; body.failed = 2;
|
||||
fs.writeFileSync(attempted, JSON.stringify(body));
|
||||
const result = runEvalCli('eval-summary.ts');
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout).not.toContain('Manual accepted:');
|
||||
expect(result.stdout).toContain('Flaky tests (1):');
|
||||
expect(result.stdout).toContain(`llm-judge:${manual.name}`);
|
||||
});
|
||||
|
||||
test('empty eval dir prints the getting-started hint and exits 0', () => {
|
||||
const result = runEvalCli('eval-summary.ts');
|
||||
expect(result.status).toBe(0);
|
||||
|
||||
Reference in New Issue
Block a user