Files
gstack/scripts/eval-summary.ts
T
Garry Tan a84b0b5b6d v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)
* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
2026-09-25 12:06:45 -04:00

197 lines
7.3 KiB
TypeScript

#!/usr/bin/env bun
/**
* Aggregate summary of eval runs from the project eval dir
* (~/.gstack/projects/<slug>/evals; legacy fallback ~/.gstack-dev/evals)
*
* Usage: bun run eval:summary
*/
import * as fs from 'fs';
import type { EvalResult } from '../test/helpers/eval-store';
import { evalEntryOutcome, getProjectEvalDir, listEvalJsonFiles } from '../test/helpers/eval-store';
const EVAL_DIR = getProjectEvalDir();
// Flat dir plus one level of shards/<slug>/
const files = listEvalJsonFiles(EVAL_DIR);
if (files.length === 0) {
console.log('No eval runs yet. Run: EVALS=1 bun run test:evals');
process.exit(0);
}
// Load all results
const results: EvalResult[] = [];
for (const file of files) {
try {
results.push(JSON.parse(fs.readFileSync(file, 'utf-8')));
} catch { continue; }
}
// Aggregate stats
const e2eRuns = results.filter(r => r.tier === 'e2e');
const judgeRuns = results.filter(r => r.tier === 'llm-judge');
const totalCost = results.reduce((s, r) => s + (r.total_cost_usd || 0), 0);
const avgE2ECost = e2eRuns.length > 0 ? e2eRuns.reduce((s, r) => s + r.total_cost_usd, 0) / e2eRuns.length : 0;
const avgJudgeCost = judgeRuns.length > 0 ? judgeRuns.reduce((s, r) => s + r.total_cost_usd, 0) / judgeRuns.length : 0;
// Duration + turns from E2E runs
const avgE2EDuration = e2eRuns.length > 0
? e2eRuns.reduce((s, r) => s + (r.total_duration_ms || 0), 0) / e2eRuns.length
: 0;
const e2eTurns: number[] = [];
for (const r of e2eRuns) {
const runTurns = r.tests.reduce((s, t) => s + (t.turns_used || 0), 0);
if (runTurns > 0) e2eTurns.push(runTurns);
}
const avgE2ETurns = e2eTurns.length > 0
? e2eTurns.reduce((a, b) => a + b, 0) / e2eTurns.length
: 0;
// Per-test efficiency stats (avg turns + duration across runs)
const testEfficiency = new Map<string, { turns: number[]; durations: number[]; costs: number[] }>();
for (const r of e2eRuns) {
for (const t of r.tests) {
if (!testEfficiency.has(t.name)) {
testEfficiency.set(t.name, { turns: [], durations: [], costs: [] });
}
const stats = testEfficiency.get(t.name)!;
if (t.turns_used !== undefined) stats.turns.push(t.turns_used);
if (t.duration_ms > 0) stats.durations.push(t.duration_ms);
if (t.cost_usd > 0) stats.costs.push(t.cost_usd);
}
}
// Detection rates from outcome evals
const detectionRates: number[] = [];
for (const r of e2eRuns) {
for (const t of r.tests) {
if (t.detection_rate !== undefined) {
detectionRates.push(t.detection_rate);
}
}
}
const avgDetection = detectionRates.length > 0
? detectionRates.reduce((a, b) => a + b, 0) / detectionRates.length
: null;
// Flaky tests (passed in some runs, failed in others)
const testResults = new Map<string, boolean[]>();
const manualAccepted: Array<{ name: string; approvedBy: string; approvalUrl: string }> = [];
for (const r of results) {
const final = new Map(r.tests.map(t => [t.name, t]));
const manuallyAcceptedNames = new Set([...final.values()].filter(t => evalEntryOutcome(t) === 'manual-review').map(t => t.name));
for (const t of final.values()) {
if (evalEntryOutcome(t) === 'manual-review') manualAccepted.push({ name: t.name,
approvedBy: t.manual_review!.approval.approved_by, approvalUrl: t.manual_review!.approval.approval_url });
}
for (const t of r.tests) {
if (manuallyAcceptedNames.has(t.name)) continue;
const key = `${r.tier}:${t.name}`;
const outcome = evalEntryOutcome(t);
if (outcome === 'manual-review') continue;
if (!testResults.has(key)) testResults.set(key, []);
testResults.get(key)!.push(outcome === 'passed');
}
}
const flakyTests: string[] = [];
for (const [name, outcomes] of testResults) {
if (outcomes.length >= 2) {
const hasPass = outcomes.some(o => o);
const hasFail = outcomes.some(o => !o);
if (hasPass && hasFail) flakyTests.push(name);
}
}
// Branch stats
const branchStats = new Map<string, { runs: number; avgDetection: number; detections: number[] }>();
for (const r of e2eRuns) {
if (!branchStats.has(r.branch)) {
branchStats.set(r.branch, { runs: 0, avgDetection: 0, detections: [] });
}
const stats = branchStats.get(r.branch)!;
stats.runs++;
for (const t of r.tests) {
if (t.detection_rate !== undefined) {
stats.detections.push(t.detection_rate);
}
}
}
for (const stats of branchStats.values()) {
stats.avgDetection = stats.detections.length > 0
? stats.detections.reduce((a, b) => a + b, 0) / stats.detections.length
: 0;
}
// Print summary
console.log('');
console.log('Eval Summary');
console.log('═'.repeat(70));
console.log(` Total runs: ${results.length} (${e2eRuns.length} e2e, ${judgeRuns.length} llm-judge)`);
console.log(` Total spend: $${totalCost.toFixed(2)}`);
if (manualAccepted.length) {
console.log(` Manual accepted: ${manualAccepted.length} unscored provider refusal(s)`);
for (const entry of manualAccepted) console.log(` ${entry.name}: approved by ${entry.approvedBy} (${entry.approvalUrl})`);
}
console.log(` Avg cost/e2e: $${avgE2ECost.toFixed(2)}`);
console.log(` Avg cost/judge: $${avgJudgeCost.toFixed(2)}`);
if (avgE2EDuration > 0) {
console.log(` Avg duration/e2e: ${Math.round(avgE2EDuration / 1000)}s`);
}
if (avgE2ETurns > 0) {
console.log(` Avg turns/e2e: ${Math.round(avgE2ETurns)}`);
}
if (avgDetection !== null) {
console.log(` Avg detection: ${avgDetection.toFixed(1)} bugs`);
}
console.log('─'.repeat(70));
// Per-test efficiency averages (only if we have enough data)
if (testEfficiency.size > 0 && e2eRuns.length >= 2) {
console.log(' Per-test efficiency (averages across runs):');
const sorted = [...testEfficiency.entries()]
.filter(([, s]) => s.turns.length >= 2)
.sort((a, b) => {
const avgA = a[1].costs.reduce((s, c) => s + c, 0) / a[1].costs.length;
const avgB = b[1].costs.reduce((s, c) => s + c, 0) / b[1].costs.length;
return avgB - avgA;
});
for (const [name, stats] of sorted) {
const avgT = Math.round(stats.turns.reduce((a, b) => a + b, 0) / stats.turns.length);
const avgD = Math.round(stats.durations.reduce((a, b) => a + b, 0) / stats.durations.length / 1000);
const avgC = (stats.costs.reduce((a, b) => a + b, 0) / stats.costs.length).toFixed(2);
const label = name.length > 30 ? name.slice(0, 27) + '...' : name.padEnd(30);
console.log(` ${label} $${avgC} ${avgT}t ${avgD}s (${stats.turns.length} runs)`);
}
console.log('─'.repeat(70));
}
if (flakyTests.length > 0) {
console.log(` Flaky tests (${flakyTests.length}):`);
for (const name of flakyTests) {
console.log(` - ${name}`);
}
console.log('─'.repeat(70));
}
if (branchStats.size > 0) {
console.log(' Branches:');
const sorted = [...branchStats.entries()].sort((a, b) => b[1].avgDetection - a[1].avgDetection);
for (const [branch, stats] of sorted) {
const det = stats.detections.length > 0 ? ` avg det: ${stats.avgDetection.toFixed(1)}` : '';
console.log(` ${branch.padEnd(30)} ${stats.runs} runs${det}`);
}
console.log('─'.repeat(70));
}
// Date range
const timestamps = results.map(r => r.timestamp).filter(Boolean).sort();
if (timestamps.length > 0) {
const first = timestamps[0].replace('T', ' ').slice(0, 16);
const last = timestamps[timestamps.length - 1].replace('T', ' ').slice(0, 16);
console.log(` Date range: ${first} → ${last}`);
}
console.log(` Dir: ${EVAL_DIR}`);
console.log('');