mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 15:41:57 +02:00
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
117 lines
4.3 KiB
TypeScript
117 lines
4.3 KiB
TypeScript
import { describe, test, expect, beforeEach, afterEach } from 'bun:test';
|
|
import * as fs from 'fs';
|
|
import * as path from 'path';
|
|
import * as os from 'os';
|
|
import { spawnSync } from 'child_process';
|
|
import { manualReviewFixture } from './helpers/manual-judge-review-fixture';
|
|
|
|
const ROOT = path.resolve(import.meta.dir, '..');
|
|
|
|
let tmpHome: string;
|
|
|
|
beforeEach(() => {
|
|
tmpHome = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-eval-list-'));
|
|
const evalDir = path.join(tmpHome, '.gstack-dev', 'evals');
|
|
fs.mkdirSync(evalDir, { recursive: true });
|
|
writeEvalRun(evalDir, '2026-a.json', '2026-05-24T01:00:00Z', 2);
|
|
writeEvalRun(evalDir, '2026-b.json', '2026-05-24T02:00:00Z', 3);
|
|
});
|
|
|
|
afterEach(() => {
|
|
fs.rmSync(tmpHome, { recursive: true, force: true });
|
|
});
|
|
|
|
function writeEvalRun(evalDir: string, filename: string, timestamp: string, turns: number) {
|
|
fs.writeFileSync(
|
|
path.join(evalDir, filename),
|
|
JSON.stringify({
|
|
schema_version: 1,
|
|
version: '1.44.0.0',
|
|
branch: 'main',
|
|
git_sha: filename,
|
|
timestamp,
|
|
tier: 'e2e',
|
|
total_tests: 1,
|
|
passed: 1,
|
|
failed: 0,
|
|
total_cost_usd: 0,
|
|
total_duration_ms: 1000,
|
|
tests: [
|
|
{
|
|
name: filename,
|
|
suite: 'sample',
|
|
tier: 'e2e',
|
|
passed: true,
|
|
duration_ms: 1000,
|
|
cost_usd: 0,
|
|
turns_used: turns,
|
|
},
|
|
],
|
|
}),
|
|
);
|
|
}
|
|
|
|
function runEvalList(...args: string[]): { stdout: string; stderr: string; status: number } {
|
|
// cwd is the temp HOME, NOT the repo root: getProjectEvalDir() probes the
|
|
// cwd-relative .claude/skills/gstack/bin/gstack-slug, and on dev machines
|
|
// with the self-symlink that probe succeeds, routing reads to an (empty)
|
|
// project-scoped dir instead of the legacy ~/.gstack-dev/evals this test
|
|
// seeds. A neutral cwd makes both slug probes fail deterministically, so
|
|
// the CLI always uses the seeded legacy dir — same behavior as CI.
|
|
const result = spawnSync('bun', ['run', path.join(ROOT, 'scripts', 'eval-list.ts'), ...args], {
|
|
cwd: tmpHome,
|
|
env: {
|
|
...process.env,
|
|
HOME: tmpHome,
|
|
GSTACK_HOME: path.join(tmpHome, '.gstack'),
|
|
},
|
|
encoding: 'utf-8',
|
|
timeout: 30_000,
|
|
});
|
|
return {
|
|
stdout: result.stdout ?? '',
|
|
stderr: result.stderr ?? '',
|
|
status: result.status ?? -1,
|
|
};
|
|
}
|
|
|
|
describe('eval:list CLI', () => {
|
|
test('labels validated manual acceptance as unscored and rejects malformed pass claims', () => {
|
|
const manual = manualReviewFixture();
|
|
const dir = path.join(tmpHome, '.gstack-dev', 'evals');
|
|
const body = (entry: typeof manual, timestamp: string) => ({
|
|
schema_version: 2, version: '1.90.0', branch: 'main', git_sha: 'fixture', timestamp,
|
|
tier: 'llm-judge', total_tests: 1, passed: 0, failed: 0, total_cost_usd: 0,
|
|
total_duration_ms: 1, tests: [entry],
|
|
});
|
|
fs.writeFileSync(path.join(dir, 'manual.json'), JSON.stringify(body(manual, '2026-05-24T03:00:00Z')));
|
|
fs.writeFileSync(path.join(dir, 'invalid.json'), JSON.stringify(body({ ...manual, passed: true }, '2026-05-24T04:00:00Z')));
|
|
const result = runEvalList('--limit', '2');
|
|
expect(result.status).toBe(0);
|
|
const lines = result.stdout.split('\n');
|
|
expect(lines.find(line => line.includes('03:00'))).toContain('MANUAL/unscored');
|
|
expect(lines.find(line => line.includes('03:00'))).toContain(manual.manual_review!.approval.approval_url);
|
|
expect(lines.find(line => line.includes('04:00'))).toContain('0/1');
|
|
expect(lines.find(line => line.includes('04:00'))).not.toContain('MANUAL');
|
|
});
|
|
|
|
test('limits displayed eval runs with a valid positive integer', () => {
|
|
const result = runEvalList('--limit', '1');
|
|
|
|
expect(result.status).toBe(0);
|
|
expect(result.stdout).toContain('Eval History (2 total runs)');
|
|
expect(result.stdout).toContain('Showing: 1');
|
|
expect(result.stdout).toContain('2026-05-24 02:00');
|
|
expect(result.stdout).not.toContain('2026-05-24 01:00');
|
|
});
|
|
|
|
test('rejects malformed limit values instead of silently slicing output', () => {
|
|
for (const value of ['1abc', 'nope', '0', '-1', '1.5']) {
|
|
const result = runEvalList('--limit', value);
|
|
expect(result.status).not.toBe(0);
|
|
expect(result.stderr).toContain('--limit requires a positive integer');
|
|
expect(result.stdout).toBe('');
|
|
}
|
|
});
|
|
});
|