mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-28 15:41:57 +02:00
* fix(browse): prepare reliable cookie import wave for validation * ci: sequence quality and behavior for validation branch * fix(browse): isolate Windows qualification and preserve native diagnostics * test(browse): cover cookie workflow quality and isolate Windows user paths * test(browse): trace native member startup and initialize fresh folders * fix(browse): keep Windows member stdin alive through EOF * fix(browse): latch native timeouts and compare contained Edge startup * test(browse): verify native version metadata and actual Windows argv * test(browse): qualify Dia import on isolated macOS CI * fix(browse): require picker origin for session mutations * fix(browse): bound credential reads through stream completion * test(browse): inspect owned Windows process arguments natively * test(evals): preserve passing coverage during cookie repair reruns * test(browse): isolate Dia qualification in a fresh macOS account * test(browse): pass bounded integer timeouts to native Mac probes * test(browse): distinguish Windows profile initialization from containment * test(browse): await descendant pipe readiness before parent exit * test(browse): initialize and restore isolated macOS Keychain state * test(browse): initialize Windows fixture folders before qualification * test(ci): pin the same Node runtime across Windows checks * test(browse): distinguish native macOS browser preflight stages * test(browse): isolate Windows descendant console lifetime * test(browse): preserve native receipts and identify fixture lock holders * test(browse): prepare dependency resolution before native Mac worker startup * test(ci): include lock and close checks in native diagnostics * test(browse): preserve native owner probe stages and subprocess deadlines * fix(browse): classify Chromium profile-in-use exit precisely * test(browse): retain Mac qualification evidence through cleanup failures * test(browse): bound Mac fixture paths and retire its owned user domain * test(browse): accept vanished fixture entries without weakening cleanup * test(browse): identify probe-created macOS user domains safely * test(browse): observe Mac user domains without targeting them first * test(browse): use passive fresh-user ownership throughout Mac qualification * test(browse): distinguish profile and registered-home Keychain lookups * test(browse): qualify Dia under one registered account home * test(browse): identify Dia startup and owned process-group failures * test(browse): classify bounded Dia startup diagnostics without leaking output * fix(test): preserve native Mac sandboxing and reap owned browser children * fix(browse): preserve Chromium sandboxing for native profile imports * test(browse): inspect signed Mach-O architecture without launching Xcode tools * test(browse): sample pending Dia startup and reap on all cleanup paths * test(browse): compare protected Dia launches in fresh Bun and Node accounts * test(browse): inspect isolated Mac GUI readiness without browser access * v1.90.0.0 fix: bind cookie picker actions to their document * test: validate cookie guards and fit nested launch fixtures * ci: configure the bundled Chromium sandbox helper * fix(browse): classify Playwright authentication timeouts * test: retain bounded Windows lifecycle diagnostics * test(cso): reuse bounded NTFS precision candidates * test(review): handle explicit preservation choices safely * test(browse): remove owned fixture directories with explicit primitives * test(review): distinguish descriptive reuse from edit commitments * test: admit only the approved unscored cookie workflow refusal * test: keep the Office Hours judge mock export-complete * fix: keep dependency-free CI planners independent of the model SDK * test: observe the exact holder after a native fixture unlink failure * fix: start seeded PTY observations at owned readiness * test: acquire identity-bound Windows deletion admission before profile resets * test: preserve qualified Git index bits without authorizing mutations
197 lines
7.3 KiB
TypeScript
197 lines
7.3 KiB
TypeScript
#!/usr/bin/env bun
|
|
/**
|
|
* Aggregate summary of eval runs from the project eval dir
|
|
* (~/.gstack/projects/<slug>/evals; legacy fallback ~/.gstack-dev/evals)
|
|
*
|
|
* Usage: bun run eval:summary
|
|
*/
|
|
|
|
import * as fs from 'fs';
|
|
import type { EvalResult } from '../test/helpers/eval-store';
|
|
import { evalEntryOutcome, getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
|
|
|
|
const EVAL_DIR = getProjectEvalDir();
|
|
|
|
// Flat dir plus one level of shards/<slug>/
|
|
const files = listEvalJsonFiles(EVAL_DIR);
|
|
|
|
if (files.length === 0) {
|
|
console.log('No eval runs yet. Run: EVALS=1 bun run test:evals');
|
|
process.exit(0);
|
|
}
|
|
|
|
// Load all results
|
|
const results: EvalResult[] = [];
|
|
for (const file of files) {
|
|
try {
|
|
results.push(JSON.parse(fs.readFileSync(file, 'utf-8')));
|
|
} catch { continue; }
|
|
}
|
|
|
|
// Aggregate stats
|
|
const e2eRuns = results.filter(r => r.tier === 'e2e');
|
|
const judgeRuns = results.filter(r => r.tier === 'llm-judge');
|
|
const totalCost = results.reduce((s, r) => s + (r.total_cost_usd || 0), 0);
|
|
const avgE2ECost = e2eRuns.length > 0 ? e2eRuns.reduce((s, r) => s + r.total_cost_usd, 0) / e2eRuns.length : 0;
|
|
const avgJudgeCost = judgeRuns.length > 0 ? judgeRuns.reduce((s, r) => s + r.total_cost_usd, 0) / judgeRuns.length : 0;
|
|
|
|
// Duration + turns from E2E runs
|
|
const avgE2EDuration = e2eRuns.length > 0
|
|
? e2eRuns.reduce((s, r) => s + (r.total_duration_ms || 0), 0) / e2eRuns.length
|
|
: 0;
|
|
const e2eTurns: number[] = [];
|
|
for (const r of e2eRuns) {
|
|
const runTurns = r.tests.reduce((s, t) => s + (t.turns_used || 0), 0);
|
|
if (runTurns > 0) e2eTurns.push(runTurns);
|
|
}
|
|
const avgE2ETurns = e2eTurns.length > 0
|
|
? e2eTurns.reduce((a, b) => a + b, 0) / e2eTurns.length
|
|
: 0;
|
|
|
|
// Per-test efficiency stats (avg turns + duration across runs)
|
|
const testEfficiency = new Map<string, { turns: number[]; durations: number[]; costs: number[] }>();
|
|
for (const r of e2eRuns) {
|
|
for (const t of r.tests) {
|
|
if (!testEfficiency.has(t.name)) {
|
|
testEfficiency.set(t.name, { turns: [], durations: [], costs: [] });
|
|
}
|
|
const stats = testEfficiency.get(t.name)!;
|
|
if (t.turns_used !== undefined) stats.turns.push(t.turns_used);
|
|
if (t.duration_ms > 0) stats.durations.push(t.duration_ms);
|
|
if (t.cost_usd > 0) stats.costs.push(t.cost_usd);
|
|
}
|
|
}
|
|
|
|
// Detection rates from outcome evals
|
|
const detectionRates: number[] = [];
|
|
for (const r of e2eRuns) {
|
|
for (const t of r.tests) {
|
|
if (t.detection_rate !== undefined) {
|
|
detectionRates.push(t.detection_rate);
|
|
}
|
|
}
|
|
}
|
|
const avgDetection = detectionRates.length > 0
|
|
? detectionRates.reduce((a, b) => a + b, 0) / detectionRates.length
|
|
: null;
|
|
|
|
// Flaky tests (passed in some runs, failed in others)
|
|
const testResults = new Map<string, boolean[]>();
|
|
const manualAccepted: Array<{ name: string; approvedBy: string; approvalUrl: string }> = [];
|
|
for (const r of results) {
|
|
const final = new Map(r.tests.map(t => [t.name, t]));
|
|
const manuallyAcceptedNames = new Set([...final.values()].filter(t => evalEntryOutcome(t) === 'manual-review').map(t => t.name));
|
|
for (const t of final.values()) {
|
|
if (evalEntryOutcome(t) === 'manual-review') manualAccepted.push({ name: t.name,
|
|
approvedBy: t.manual_review!.approval.approved_by, approvalUrl: t.manual_review!.approval.approval_url });
|
|
}
|
|
for (const t of r.tests) {
|
|
if (manuallyAcceptedNames.has(t.name)) continue;
|
|
const key = `${r.tier}:${t.name}`;
|
|
const outcome = evalEntryOutcome(t);
|
|
if (outcome === 'manual-review') continue;
|
|
if (!testResults.has(key)) testResults.set(key, []);
|
|
testResults.get(key)!.push(outcome === 'passed');
|
|
}
|
|
}
|
|
const flakyTests: string[] = [];
|
|
for (const [name, outcomes] of testResults) {
|
|
if (outcomes.length >= 2) {
|
|
const hasPass = outcomes.some(o => o);
|
|
const hasFail = outcomes.some(o => !o);
|
|
if (hasPass && hasFail) flakyTests.push(name);
|
|
}
|
|
}
|
|
|
|
// Branch stats
|
|
const branchStats = new Map<string, { runs: number; avgDetection: number; detections: number[] }>();
|
|
for (const r of e2eRuns) {
|
|
if (!branchStats.has(r.branch)) {
|
|
branchStats.set(r.branch, { runs: 0, avgDetection: 0, detections: [] });
|
|
}
|
|
const stats = branchStats.get(r.branch)!;
|
|
stats.runs++;
|
|
for (const t of r.tests) {
|
|
if (t.detection_rate !== undefined) {
|
|
stats.detections.push(t.detection_rate);
|
|
}
|
|
}
|
|
}
|
|
for (const stats of branchStats.values()) {
|
|
stats.avgDetection = stats.detections.length > 0
|
|
? stats.detections.reduce((a, b) => a + b, 0) / stats.detections.length
|
|
: 0;
|
|
}
|
|
|
|
// Print summary
|
|
console.log('');
|
|
console.log('Eval Summary');
|
|
console.log('═'.repeat(70));
|
|
console.log(` Total runs: ${results.length} (${e2eRuns.length} e2e, ${judgeRuns.length} llm-judge)`);
|
|
console.log(` Total spend: $${totalCost.toFixed(2)}`);
|
|
if (manualAccepted.length) {
|
|
console.log(` Manual accepted: ${manualAccepted.length} unscored provider refusal(s)`);
|
|
for (const entry of manualAccepted) console.log(` ${entry.name}: approved by ${entry.approvedBy} (${entry.approvalUrl})`);
|
|
}
|
|
console.log(` Avg cost/e2e: $${avgE2ECost.toFixed(2)}`);
|
|
console.log(` Avg cost/judge: $${avgJudgeCost.toFixed(2)}`);
|
|
if (avgE2EDuration > 0) {
|
|
console.log(` Avg duration/e2e: ${Math.round(avgE2EDuration / 1000)}s`);
|
|
}
|
|
if (avgE2ETurns > 0) {
|
|
console.log(` Avg turns/e2e: ${Math.round(avgE2ETurns)}`);
|
|
}
|
|
if (avgDetection !== null) {
|
|
console.log(` Avg detection: ${avgDetection.toFixed(1)} bugs`);
|
|
}
|
|
console.log('─'.repeat(70));
|
|
|
|
// Per-test efficiency averages (only if we have enough data)
|
|
if (testEfficiency.size > 0 && e2eRuns.length >= 2) {
|
|
console.log(' Per-test efficiency (averages across runs):');
|
|
const sorted = [...testEfficiency.entries()]
|
|
.filter(([, s]) => s.turns.length >= 2)
|
|
.sort((a, b) => {
|
|
const avgA = a[1].costs.reduce((s, c) => s + c, 0) / a[1].costs.length;
|
|
const avgB = b[1].costs.reduce((s, c) => s + c, 0) / b[1].costs.length;
|
|
return avgB - avgA;
|
|
});
|
|
for (const [name, stats] of sorted) {
|
|
const avgT = Math.round(stats.turns.reduce((a, b) => a + b, 0) / stats.turns.length);
|
|
const avgD = Math.round(stats.durations.reduce((a, b) => a + b, 0) / stats.durations.length / 1000);
|
|
const avgC = (stats.costs.reduce((a, b) => a + b, 0) / stats.costs.length).toFixed(2);
|
|
const label = name.length > 30 ? name.slice(0, 27) + '...' : name.padEnd(30);
|
|
console.log(` ${label} $${avgC} ${avgT}t ${avgD}s (${stats.turns.length} runs)`);
|
|
}
|
|
console.log('─'.repeat(70));
|
|
}
|
|
|
|
if (flakyTests.length > 0) {
|
|
console.log(` Flaky tests (${flakyTests.length}):`);
|
|
for (const name of flakyTests) {
|
|
console.log(` - ${name}`);
|
|
}
|
|
console.log('─'.repeat(70));
|
|
}
|
|
|
|
if (branchStats.size > 0) {
|
|
console.log(' Branches:');
|
|
const sorted = [...branchStats.entries()].sort((a, b) => b[1].avgDetection - a[1].avgDetection);
|
|
for (const [branch, stats] of sorted) {
|
|
const det = stats.detections.length > 0 ? ` avg det: ${stats.avgDetection.toFixed(1)}` : '';
|
|
console.log(` ${branch.padEnd(30)} ${stats.runs} runs${det}`);
|
|
}
|
|
console.log('─'.repeat(70));
|
|
}
|
|
|
|
// Date range
|
|
const timestamps = results.map(r => r.timestamp).filter(Boolean).sort();
|
|
if (timestamps.length > 0) {
|
|
const first = timestamps[0].replace('T', ' ').slice(0, 16);
|
|
const last = timestamps[timestamps.length - 1].replace('T', ' ').slice(0, 16);
|
|
console.log(` Date range: ${first} → ${last}`);
|
|
}
|
|
|
|
console.log(` Dir: ${EVAL_DIR}`);
|
|
console.log('');
|