mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 07:32:14 +02:00
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
132 lines
4.5 KiB
TypeScript
132 lines
4.5 KiB
TypeScript
#!/usr/bin/env bun
|
|
/**
|
|
* List eval runs from the project eval dir (~/.gstack/projects/<slug>/evals;
|
|
* legacy fallback ~/.gstack-dev/evals)
|
|
*
|
|
* Usage: bun run eval:list [--branch <name>] [--tier e2e|llm-judge] [--limit N]
|
|
*/
|
|
|
|
import * as fs from 'fs';
|
|
import { evalEntryOutcome, getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
|
|
|
|
const EVAL_DIR = getProjectEvalDir();
|
|
|
|
// Parse args
|
|
const args = process.argv.slice(2);
|
|
let filterBranch: string | null = null;
|
|
let filterTier: string | null = null;
|
|
let limit = 20;
|
|
|
|
function parseLimit(raw: string | undefined): number {
|
|
if (!raw || !/^[1-9]\d*$/.test(raw)) {
|
|
console.error('eval:list: --limit requires a positive integer');
|
|
process.exit(1);
|
|
}
|
|
const parsed = Number(raw);
|
|
if (!Number.isSafeInteger(parsed)) {
|
|
console.error('eval:list: --limit requires a positive integer');
|
|
process.exit(1);
|
|
}
|
|
return parsed;
|
|
}
|
|
|
|
for (let i = 0; i < args.length; i++) {
|
|
if (args[i] === '--branch' && args[i + 1]) { filterBranch = args[++i]; }
|
|
else if (args[i] === '--tier' && args[i + 1]) { filterTier = args[++i]; }
|
|
else if (args[i] === '--limit') { limit = parseLimit(args[++i]); }
|
|
}
|
|
|
|
// Read eval files (flat dir plus one level of shards/<slug>/)
|
|
const files = listEvalJsonFiles(EVAL_DIR);
|
|
|
|
if (files.length === 0) {
|
|
console.log('No eval runs yet. Run: EVALS=1 bun run test:evals');
|
|
process.exit(0);
|
|
}
|
|
|
|
// Parse top-level fields from each file
|
|
interface RunSummary {
|
|
file: string;
|
|
timestamp: string;
|
|
branch: string;
|
|
tier: string;
|
|
version: string;
|
|
passed: number;
|
|
manual: Array<{ name: string; approvedBy: string; approvalUrl: string }>;
|
|
total: number;
|
|
cost: number;
|
|
duration: number;
|
|
turns: number;
|
|
}
|
|
|
|
const runs: RunSummary[] = [];
|
|
for (const file of files) {
|
|
try {
|
|
const data = JSON.parse(fs.readFileSync(file, 'utf-8'));
|
|
if (filterBranch && data.branch !== filterBranch) continue;
|
|
if (filterTier && data.tier !== filterTier) continue;
|
|
const totalTurns = (data.tests || []).reduce((s: number, t: any) => s + (t.turns_used || 0), 0);
|
|
const tests = Array.isArray(data.tests) ? data.tests : null;
|
|
const final = tests ? [...new Map<string, any>(tests.map((t: any) => [t.name, t] as const)).values()] : [];
|
|
const manual = final.filter((t: any) => evalEntryOutcome(t) === 'manual-review').map((t: any) => ({
|
|
name: t.name, approvedBy: t.manual_review.approval.approved_by, approvalUrl: t.manual_review.approval.approval_url,
|
|
}));
|
|
runs.push({
|
|
file,
|
|
timestamp: data.timestamp || '',
|
|
branch: data.branch || 'unknown',
|
|
tier: data.tier || 'unknown',
|
|
version: data.version || '?',
|
|
passed: tests ? tests.filter((t: any) => evalEntryOutcome(t) === 'passed').length : data.passed || 0,
|
|
manual,
|
|
total: data.total_tests || 0,
|
|
cost: data.total_cost_usd || 0,
|
|
duration: data.total_duration_ms || 0,
|
|
turns: totalTurns,
|
|
});
|
|
} catch { continue; }
|
|
}
|
|
|
|
// Sort by timestamp descending
|
|
runs.sort((a, b) => b.timestamp.localeCompare(a.timestamp));
|
|
|
|
// Apply limit
|
|
const displayed = runs.slice(0, limit);
|
|
|
|
// Print table
|
|
console.log('');
|
|
console.log(`Eval History (${runs.length} total runs)`);
|
|
console.log('═'.repeat(105));
|
|
console.log(
|
|
' ' +
|
|
'Date'.padEnd(17) +
|
|
'Branch'.padEnd(25) +
|
|
'Tier'.padEnd(12) +
|
|
'Pass'.padEnd(8) +
|
|
'Cost'.padEnd(8) +
|
|
'Turns'.padEnd(7) +
|
|
'Duration'.padEnd(10) +
|
|
'Version'
|
|
);
|
|
console.log('─'.repeat(105));
|
|
|
|
for (const run of displayed) {
|
|
const date = run.timestamp.replace('T', ' ').slice(0, 16);
|
|
const branch = run.branch.length > 23 ? run.branch.slice(0, 20) + '...' : run.branch.padEnd(25);
|
|
const pass = `${run.passed}/${run.total}`.padEnd(8);
|
|
const cost = `$${run.cost.toFixed(2)}`.padEnd(8);
|
|
const turns = run.turns > 0 ? `${run.turns}t`.padEnd(7) : ''.padEnd(7);
|
|
const dur = run.duration > 0 ? `${Math.round(run.duration / 1000)}s`.padEnd(10) : ''.padEnd(10);
|
|
const manual = run.manual.map(entry => `MANUAL/unscored ${entry.name}: approved by ${entry.approvedBy} (${entry.approvalUrl})`).join('; ');
|
|
console.log(` ${date.padEnd(17)}${branch}${run.tier.padEnd(12)}${pass}${cost}${turns}${dur}v${run.version}${manual ? ` ${manual}` : ''}`);
|
|
}
|
|
|
|
console.log('─'.repeat(105));
|
|
|
|
const totalCost = runs.reduce((s, r) => s + r.cost, 0);
|
|
const totalDur = runs.reduce((s, r) => s + r.duration, 0);
|
|
const totalTurns = runs.reduce((s, r) => s + r.turns, 0);
|
|
console.log(` ${runs.length} runs | $${totalCost.toFixed(2)} total | ${totalTurns} turns | ${Math.round(totalDur / 1000)}s | Showing: ${displayed.length}`);
|
|
console.log(` Dir: ${EVAL_DIR}`);
|
|
console.log('');
|