mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-03 01:46:55 +02:00
Merge remote-tracking branch 'origin/capy/rel-b' into capy/rel-a
This commit is contained in:
commit
a385e5de18
16 files changed
+1632
-234
No files matched your search
@@ -23,6 +23,8 @@ export interface JudgeScore {
|
||||
reasoning: string;
|
||||
}
|
||||
|
||||
export const JUDGE_SCORE_DIMENSIONS = ['clarity', 'completeness', 'actionability'] as const;
|
||||
|
||||
export interface JudgeRefusalEvidence {
|
||||
stop_reason: 'refusal';
|
||||
response_id: string | null;
|
||||
@@ -196,6 +198,63 @@ export async function callJudge<T>(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Samples per judge panel: EVAL_POLICY.judge.samples, restated here so this
|
||||
* helper (imported by many paid tests) does not pull the quarantine registry
|
||||
* into their touchfile closure. test/judge-panel.test.ts pins the two equal.
|
||||
*/
|
||||
export const JUDGE_PANEL_SAMPLES = 3;
|
||||
|
||||
/**
|
||||
* Judge panel (EVAL_POLICY.judge): every `judge`-kind entry draws a fixed number of
|
||||
* independent samples of the SAME prompt concurrently, inside its unchanged
|
||||
* JUDGE_MS budget. Numeric dimensions gate on the per-dimension panel mean
|
||||
* against the unchanged minimum; boolean fields gate on a strict majority.
|
||||
* A sample that errors (refusal, truncation, non-JSON, malformed field) fails
|
||||
* the whole panel and is never resampled. callJudge's 429 backoff happens
|
||||
* before any model output exists, so it is transport, not a verdict retry.
|
||||
*/
|
||||
export async function judgePanel<T>(sample: () => Promise<T>): Promise<T[]> {
|
||||
const settled = await Promise.allSettled(Array.from({ length: JUDGE_PANEL_SAMPLES }, () => sample()));
|
||||
const failures = settled.flatMap((result, index) => result.status === 'rejected' ? [{ index, reason: result.reason }] : []);
|
||||
if (failures.length === 0) return settled.map(result => (result as PromiseFulfilledResult<T>).value);
|
||||
const first = failures[0]!;
|
||||
// A refusal is an unscored panel only when EVERY sample refused; a partial
|
||||
// refusal beside scored samples is an ordinary failed panel.
|
||||
if (first.reason instanceof JudgeRefusalError && failures.length < settled.length) {
|
||||
throw new Error(`Judge panel sample ${first.index + 1} of ${settled.length} failed beside scored samples: ${first.reason.message}`);
|
||||
}
|
||||
throw first.reason;
|
||||
}
|
||||
|
||||
/** Per-dimension mean over a panel; any non-finite sample value fails the panel. */
|
||||
export function judgePanelMean<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, keys: readonly K[]): Record<K, number> {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
return Object.fromEntries(keys.map(key => {
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'number' || !Number.isFinite(value));
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-numeric ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return [key, (values as number[]).reduce((sum, value) => sum + value, 0) / values.length];
|
||||
})) as Record<K, number>;
|
||||
}
|
||||
|
||||
/** Strict majority of a boolean field; any non-boolean sample value fails the panel. */
|
||||
export function judgePanelMajority<K extends string>(samples: ReadonlyArray<Record<K, unknown>>, key: K): boolean {
|
||||
if (samples.length === 0) throw new Error('Judge panel has no samples');
|
||||
const values = samples.map(sample => sample && typeof sample === 'object' ? sample[key] : undefined);
|
||||
const bad = values.findIndex(value => typeof value !== 'boolean');
|
||||
if (bad !== -1) throw new Error(`Judge panel sample ${bad + 1} has non-boolean ${key}: ${JSON.stringify(values[bad])}`);
|
||||
return values.filter(value => value === true).length * 2 > values.length;
|
||||
}
|
||||
|
||||
/** Sample reasoning lines, numbered, for the collector record. */
|
||||
export function judgePanelReasoning(samples: ReadonlyArray<unknown>): string {
|
||||
return samples.map((sample, index) => {
|
||||
const reasoning = sample && typeof sample === 'object' ? (sample as { reasoning?: unknown }).reasoning : undefined;
|
||||
return `[sample ${index + 1}] ${typeof reasoning === 'string' ? reasoning : ''}`;
|
||||
}).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Score documentation quality on clarity/completeness/actionability (1-5).
|
||||
*/
|
||||
|
||||
@@ -72,7 +72,12 @@ export const CASE_CI_EXCLUDE: Record<string, { reason: string; tracking: string
|
||||
* new-policy trials; exit at >= `exit.rate` over >=
|
||||
* `exit.minTrials`; at most `capFraction` of each tier's
|
||||
* blocking cases; an entry expires after `expiryWeeklyRuns`.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series.
|
||||
* judge - a judge case draws `samples` independent samples of one
|
||||
* prompt concurrently; numeric dimensions gate on the panel
|
||||
* mean against the unchanged threshold, booleans on a strict
|
||||
* majority; an erroring sample fails the panel, never resampled.
|
||||
* drift - one-sided Fisher exact alarm between input-identity series
|
||||
* (Holm-controlled across the cases tested in one report).
|
||||
* infraRedispatch - a census whose every red verdict is machine-classified
|
||||
* INFRA or INCOMPLETE may be re-dispatched this many times as
|
||||
* a new run; both runs are reported.
|
||||
@@ -86,6 +91,7 @@ export const EVAL_POLICY = {
|
||||
capFraction: 0.10,
|
||||
expiryWeeklyRuns: 8,
|
||||
},
|
||||
judge: { samples: 3 },
|
||||
drift: { fisherAlpha: 0.05, fisherMinPerSide: 6 },
|
||||
infraRedispatch: 1,
|
||||
} as const;
|
||||
@@ -100,14 +106,18 @@ export const EVAL_POLICY = {
|
||||
* failures (a product defect is never quarantined), and unchanged case
|
||||
* touchfiles in the change that adds it. Pinned by
|
||||
* test/periodic-exclude-policy.test.ts.
|
||||
* reason - the written diagnosis
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - ISO date the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* reason - the written diagnosis, with the pass-rate evidence
|
||||
* failureClass - what the diagnosis found; a product defect has no class here
|
||||
* tracking - issue or TODOS pointer
|
||||
* owner - who removes it
|
||||
* enteredAt - YYYY-MM-DD the entry landed (expiry counts weekly runs from here)
|
||||
* exit - the measurable exit condition
|
||||
* At most EVAL_POLICY.quarantine.capFraction of a tier's cases (gate and
|
||||
* periodic are the blocking tiers) may be quarantined at once.
|
||||
*/
|
||||
export const CASE_QUARANTINE: Record<string, {
|
||||
reason: string;
|
||||
failureClass: 'detector' | 'harness' | 'model-latency';
|
||||
tracking: string;
|
||||
owner: string;
|
||||
enteredAt: string;
|
||||
|
||||
@@ -34,6 +34,8 @@ import {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
} from './touchfiles-data';
|
||||
|
||||
/** Repo-relative path of the pure-data file (the map-diff subject). */
|
||||
@@ -145,6 +147,9 @@ export interface TouchfileMaps {
|
||||
E2E_TIERS: Record<string, string>;
|
||||
LLM_JUDGE_TOUCHFILES: Record<string, string[]>;
|
||||
GLOBAL_TOUCHFILES: string[];
|
||||
/** Absent on base revisions older than the eval-kind registry: every current key then counts as changed. */
|
||||
E2E_KINDS?: Record<string, string>;
|
||||
BEHAVIOR_WHY?: Record<string, string>;
|
||||
}
|
||||
|
||||
export type MapDiffCause =
|
||||
@@ -171,6 +176,8 @@ const CURRENT_MAPS: TouchfileMaps = {
|
||||
E2E_TIERS,
|
||||
LLM_JUDGE_TOUCHFILES,
|
||||
GLOBAL_TOUCHFILES,
|
||||
E2E_KINDS,
|
||||
BEHAVIOR_WHY,
|
||||
};
|
||||
|
||||
function isStringArray(v: unknown): v is string[] {
|
||||
@@ -193,14 +200,18 @@ function isTouchfileMaps(v: unknown): v is TouchfileMaps {
|
||||
return isRecordOfStringArrays(o.E2E_TOUCHFILES)
|
||||
&& isRecordOfStrings(o.E2E_TIERS)
|
||||
&& isRecordOfStringArrays(o.LLM_JUDGE_TOUCHFILES)
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES);
|
||||
&& isStringArray(o.GLOBAL_TOUCHFILES)
|
||||
&& (o.E2E_KINDS === undefined || isRecordOfStrings(o.E2E_KINDS))
|
||||
&& (o.BEHAVIOR_WHY === undefined || isRecordOfStrings(o.BEHAVIOR_WHY));
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure map-diff core (injectable for tests — no git, no filesystem).
|
||||
*
|
||||
* A key counts as CHANGED when it was added to any per-key map, its dep-list
|
||||
* array differs, or its tier value flipped. A key counts as REMOVED only when
|
||||
* array differs, or its tier, kind or behavior tolerance changed. A per-key
|
||||
* map missing on the old side (a base revision older than E2E_KINDS /
|
||||
* BEHAVIOR_WHY) makes every key of that map count as added. A key counts as REMOVED only when
|
||||
* it is gone from every new per-key map; a key dropped from one map but still
|
||||
* present in another (e.g. tier entry deleted, touchfile entry kept) counts
|
||||
* as changed — conservative, because the test still exists with a different
|
||||
@@ -212,7 +223,7 @@ export function diffTouchfileMapsCore(
|
||||
oldMaps: TouchfileMaps,
|
||||
newMaps: TouchfileMaps,
|
||||
): { changedTests: string[]; removedTests: string[]; globalTouchfilesChanged: boolean } {
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES'] as const;
|
||||
const perKeyMapNames = ['E2E_TOUCHFILES', 'E2E_TIERS', 'LLM_JUDGE_TOUCHFILES', 'E2E_KINDS', 'BEHAVIOR_WHY'] as const;
|
||||
const changed = new Set<string>();
|
||||
const rawRemoved = new Set<string>();
|
||||
|
||||
@@ -290,6 +301,8 @@ export function diffTouchfileMaps(
|
||||
' E2E_TIERS: m.E2E_TIERS,',
|
||||
' LLM_JUDGE_TOUCHFILES: m.LLM_JUDGE_TOUCHFILES,',
|
||||
' GLOBAL_TOUCHFILES: m.GLOBAL_TOUCHFILES,',
|
||||
' E2E_KINDS: m.E2E_KINDS,',
|
||||
' BEHAVIOR_WHY: m.BEHAVIOR_WHY,',
|
||||
'}));',
|
||||
'',
|
||||
].join('\n'));
|
||||
|
||||
@@ -1597,7 +1597,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'shared-libs-unsupported-git': 'rule',
|
||||
'shared-libs-review-lifecycle': 'rule',
|
||||
'shared-libs-review-revalidation': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'rule',
|
||||
'shared-libs-opportunity-judgment': 'behavior',
|
||||
'shared-libs-pr-coverage': 'rule',
|
||||
'shared-libs-plan-callers': 'rule',
|
||||
'browse-basic': 'rule',
|
||||
@@ -1634,7 +1634,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'review-sql-injection': 'rule',
|
||||
'review-enum-completeness': 'rule',
|
||||
'review-base-branch': 'rule',
|
||||
'review-design-lite': 'rule',
|
||||
'review-design-lite': 'behavior',
|
||||
'review-coverage-audit': 'rule',
|
||||
'review-dashboard-via': 'rule',
|
||||
'review-army-migration-safety': 'rule',
|
||||
@@ -1642,21 +1642,21 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'review-army-delivery-audit': 'rule',
|
||||
'review-army-quality-score': 'rule',
|
||||
'review-army-json-findings': 'rule',
|
||||
'review-army-red-team': 'rule',
|
||||
'review-army-consensus': 'rule',
|
||||
'review-army-simplification': 'rule',
|
||||
'review-army-simplification-precision': 'rule',
|
||||
'review-army-red-team': 'behavior',
|
||||
'review-army-consensus': 'behavior',
|
||||
'review-army-simplification': 'behavior',
|
||||
'review-army-simplification-precision': 'behavior',
|
||||
'office-hours-spec-review': 'rule',
|
||||
'office-hours-brain-writeback': 'rule',
|
||||
'office-hours-brain-writeback': 'behavior',
|
||||
'gbrain-roundtrip-local': 'rule',
|
||||
'sync-gbrain-read-ready': 'rule',
|
||||
'sync-gbrain-read-unknown': 'rule',
|
||||
'office-hours-forcing-energy': 'rule',
|
||||
'office-hours-builder-wildness': 'rule',
|
||||
'office-hours-forcing-energy': 'behavior',
|
||||
'office-hours-builder-wildness': 'behavior',
|
||||
'plan-ceo-review': 'rule',
|
||||
'plan-ceo-review-selective': 'rule',
|
||||
'plan-ceo-review-benefits': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'rule',
|
||||
'plan-ceo-review-expansion-energy': 'behavior',
|
||||
'plan-eng-review': 'rule',
|
||||
'plan-eng-review-artifact': 'rule',
|
||||
'plan-eng-coverage-audit': 'rule',
|
||||
@@ -1688,16 +1688,16 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'setup-gbrain-remote': 'rule',
|
||||
'setup-gbrain-bad-token': 'rule',
|
||||
'setup-gbrain-path4-local-pglite': 'rule',
|
||||
'plan-ceo-review-format-mode': 'rule',
|
||||
'plan-ceo-review-format-approach': 'rule',
|
||||
'plan-eng-review-format-coverage': 'rule',
|
||||
'plan-eng-review-format-kind': 'rule',
|
||||
'office-hours-phase4-fork': 'rule',
|
||||
'llm-judge-recommendation': 'rule',
|
||||
'plan-ceo-review-prosons-cadence': 'rule',
|
||||
'plan-review-prosons-format': 'rule',
|
||||
'plan-review-prosons-hardstop-neg': 'rule',
|
||||
'plan-review-prosons-neutral-neg': 'rule',
|
||||
'plan-ceo-review-format-mode': 'behavior',
|
||||
'plan-ceo-review-format-approach': 'behavior',
|
||||
'plan-eng-review-format-coverage': 'behavior',
|
||||
'plan-eng-review-format-kind': 'behavior',
|
||||
'office-hours-phase4-fork': 'behavior',
|
||||
'llm-judge-recommendation': 'judge',
|
||||
'plan-ceo-review-prosons-cadence': 'behavior',
|
||||
'plan-review-prosons-format': 'behavior',
|
||||
'plan-review-prosons-hardstop-neg': 'behavior',
|
||||
'plan-review-prosons-neutral-neg': 'behavior',
|
||||
'plan-tune-inspect': 'rule',
|
||||
'codex-offered-office-hours': 'rule',
|
||||
'codex-offered-ceo-review': 'rule',
|
||||
@@ -1758,7 +1758,7 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'design-review-detector-shim': 'rule',
|
||||
'design-review-detector-shim-dom': 'rule',
|
||||
'design-review-plugin-handoff': 'rule',
|
||||
'design-html-slop-gate': 'rule',
|
||||
'design-html-slop-gate': 'behavior',
|
||||
'diagram-triplet': 'rule',
|
||||
'diagram-authoring-quality': 'rule',
|
||||
'gstack-upgrade-happy-path': 'rule',
|
||||
@@ -1770,8 +1770,8 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'setup-deploy-workflow': 'rule',
|
||||
'autoplan-dual-voice': 'rule',
|
||||
'benchmark-providers-live': 'rule',
|
||||
'scrape-match-path': 'rule',
|
||||
'scrape-prototype-path': 'rule',
|
||||
'scrape-match-path': 'behavior',
|
||||
'scrape-prototype-path': 'behavior',
|
||||
'skillify-happy-path': 'rule',
|
||||
'skillify-provenance-refusal': 'rule',
|
||||
'skillify-approval-reject': 'rule',
|
||||
@@ -1799,30 +1799,30 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
'overlay-harness-opus-4-7-literal-interpretation': 'rule',
|
||||
'overlay-harness-claude-dedicated-tools-vs-bash-sonnet': 'rule',
|
||||
'journey-negatives': 'rule',
|
||||
'review/SKILL.md workflow': 'rule',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'rule',
|
||||
'browse/SKILL.md reference': 'rule',
|
||||
'setup block': 'rule',
|
||||
'qa/SKILL.md workflow': 'rule',
|
||||
'qa/SKILL.md health rubric': 'rule',
|
||||
'qa/SKILL.md anti-refusal': 'rule',
|
||||
'cross-skill greptile consistency': 'rule',
|
||||
'ship/SKILL.md workflow': 'rule',
|
||||
'document-release/SKILL.md workflow': 'rule',
|
||||
'plan-ceo-review/SKILL.md modes': 'rule',
|
||||
'plan-eng-review/SKILL.md sections': 'rule',
|
||||
'plan-design-review/SKILL.md passes': 'rule',
|
||||
'design-review/SKILL.md fix loop': 'rule',
|
||||
'design-consultation/SKILL.md research': 'rule',
|
||||
'land-and-deploy/SKILL.md workflow': 'rule',
|
||||
'canary/SKILL.md monitoring loop': 'rule',
|
||||
'benchmark/SKILL.md perf collection': 'rule',
|
||||
'setup-deploy/SKILL.md platform setup': 'rule',
|
||||
'retro/SKILL.md instructions': 'rule',
|
||||
'qa-only/SKILL.md workflow': 'rule',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'rule',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'rule',
|
||||
'voice directive tone': 'rule',
|
||||
'review/SKILL.md workflow': 'judge',
|
||||
'setup-browser-cookies/SKILL.md workflow': 'judge',
|
||||
'browse/SKILL.md reference': 'judge',
|
||||
'setup block': 'judge',
|
||||
'qa/SKILL.md workflow': 'judge',
|
||||
'qa/SKILL.md health rubric': 'judge',
|
||||
'qa/SKILL.md anti-refusal': 'judge',
|
||||
'cross-skill greptile consistency': 'judge',
|
||||
'ship/SKILL.md workflow': 'judge',
|
||||
'document-release/SKILL.md workflow': 'judge',
|
||||
'plan-ceo-review/SKILL.md modes': 'judge',
|
||||
'plan-eng-review/SKILL.md sections': 'judge',
|
||||
'plan-design-review/SKILL.md passes': 'judge',
|
||||
'design-review/SKILL.md fix loop': 'judge',
|
||||
'design-consultation/SKILL.md research': 'judge',
|
||||
'land-and-deploy/SKILL.md workflow': 'judge',
|
||||
'canary/SKILL.md monitoring loop': 'judge',
|
||||
'benchmark/SKILL.md perf collection': 'judge',
|
||||
'setup-deploy/SKILL.md platform setup': 'judge',
|
||||
'retro/SKILL.md instructions': 'judge',
|
||||
'qa-only/SKILL.md workflow': 'judge',
|
||||
'gstack-upgrade/SKILL.md upgrade flow': 'judge',
|
||||
'sync-gbrain/SKILL.md read-only readiness': 'judge',
|
||||
'voice directive tone': 'judge',
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1830,4 +1830,49 @@ export const E2E_KINDS: Record<string, 'rule' | 'behavior' | 'judge'> = {
|
||||
* deviation is acceptable product behavior. Keys equal the behavior ids of
|
||||
* E2E_KINDS; values are non-empty.
|
||||
*/
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {};
|
||||
export const BEHAVIOR_WHY: Record<string, string> = {
|
||||
'shared-libs-opportunity-judgment':
|
||||
"Whether a candidate extraction is worth recommending is a judgment call; the read-only invariant stays a contract.",
|
||||
'review-design-lite':
|
||||
"How many of the seven design-lite checklist items the live review flags varies run to run; the fake-engine rows it must carry stay strict.",
|
||||
'review-army-red-team':
|
||||
"Whether the red-team lens surfaces on a small diff is a live model choice, not a contract.",
|
||||
'review-army-consensus':
|
||||
"Multi-specialist agreement on the planted SQL finding is a quality benchmark that tolerates an occasional miss.",
|
||||
'review-army-simplification':
|
||||
"Flagging the planted unnecessary structure is an advisory-lens quality judgment.",
|
||||
'review-army-simplification-precision':
|
||||
"Staying silent on a lean diff is a false-flag noise benchmark; an occasional advisory is acceptable noise.",
|
||||
'office-hours-forcing-energy':
|
||||
"The Q3 posture is scored by a live judge on generated prose; a single flat phrasing is tolerable.",
|
||||
'office-hours-builder-wildness':
|
||||
"Builder-mode creativity is scored by a live judge on generated prose; one conservative riff is tolerable.",
|
||||
'office-hours-brain-writeback':
|
||||
"The model's interpretation of the gbrain writeback instruction (page shape, tags) varies; no secret or safety step rides on it.",
|
||||
'office-hours-phase4-fork':
|
||||
"Phase 4 asks the model to invent 2-3 architectures; surfacing the fork with its reasoning is open-ended generation.",
|
||||
'plan-ceo-review-expansion-energy':
|
||||
"Expansion framing is scored by a live judge on generated proposals; one flat proposal set is tolerable.",
|
||||
'plan-ceo-review-format-mode':
|
||||
"Mode-question wording (Completeness line vs kind note) is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-format-approach':
|
||||
"Approach-menu Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-coverage':
|
||||
"Coverage-issue Completeness wording is live formatting of one AskUserQuestion.",
|
||||
'plan-eng-review-format-kind':
|
||||
"Kind-note wording is live formatting of one AskUserQuestion.",
|
||||
'plan-ceo-review-prosons-cadence':
|
||||
"Pros/Cons cadence on a hard-stop question is live formatting; either the escape or the full block is accepted.",
|
||||
'plan-review-prosons-format':
|
||||
"The full Pros/Cons block (counts of pros and cons, labels) is live formatting of one question.",
|
||||
'plan-review-prosons-hardstop-neg':
|
||||
"Not using the hard-stop escape on an ordinary decision is live formatting of one question.",
|
||||
'plan-review-prosons-neutral-neg':
|
||||
"Avoiding neutral posture and naming a because-reason is live formatting of one question.",
|
||||
'design-html-slop-gate':
|
||||
"How many scan passes the one-pass slop gate takes on a fake engine's fixed output is a judgment call.",
|
||||
'scrape-match-path':
|
||||
"The /scrape fallback no longer prescribes the browser-skills match flow, so taking it is prompt compliance.",
|
||||
'scrape-prototype-path':
|
||||
"The /scrape fallback no longer prescribes the prototype flow, so taking it is prompt compliance.",
|
||||
};
|
||||
@@ -4,7 +4,7 @@ import * as path from 'node:path';
|
||||
import { spawnSync } from 'node:child_process';
|
||||
import { DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel } from '../../lib/eval-model';
|
||||
import { JUDGE_MS } from './eval-budgets';
|
||||
import type { JudgeScore } from './llm-judge';
|
||||
import { JUDGE_PANEL_SAMPLES, JUDGE_SCORE_DIMENSIONS, judgePanelMean, type JudgeScore } from './llm-judge';
|
||||
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, WORKFLOW_JUDGE_RESPONSE_SCHEMA, WORKFLOW_JUDGE_REASONING_WORD_LIMIT } from './workflow-judge-input';
|
||||
import { buildEvalInputIdentity, lookupEvalInputCache, sourceDependencyClosure, storeEvalInputCache,
|
||||
type EvalCacheValue, type EvalInputIdentity, type EvalPassingProof } from '../../scripts/eval-input-cache';
|
||||
@@ -39,17 +39,28 @@ export function validWorkflowJudgeScore(value: EvalCacheValue, thresholds: Thres
|
||||
|| typeof value.reasoning !== 'string'
|
||||
|| (structuredResponse && (!value.reasoning.trim()
|
||||
|| value.reasoning.trim().split(/\s+/).length >= WORKFLOW_JUDGE_REASONING_WORD_LIMIT))) return false;
|
||||
return (['clarity', 'completeness', 'actionability'] as const).every(key =>
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key =>
|
||||
typeof value[key] === 'number' && Number.isInteger(value[key]) && value[key] >= thresholds[key] && value[key] <= 5);
|
||||
}
|
||||
|
||||
const SAMPLE_RANGE: Thresholds = { clarity: 1, completeness: 1, actionability: 1 };
|
||||
|
||||
/** A complete judge panel: exactly JUDGE_PANEL_SAMPLES of valid samples whose per-dimension mean meets every threshold. */
|
||||
export function validWorkflowJudgePanel(value: EvalCacheValue, thresholds: Thresholds, structuredResponse = false): value is { samples: Array<JudgeScore & EvalCacheValue> } {
|
||||
if (!value || typeof value !== 'object' || Array.isArray(value) || Object.keys(value).join(',') !== 'samples'
|
||||
|| !Array.isArray(value.samples) || value.samples.length !== JUDGE_PANEL_SAMPLES
|
||||
|| !value.samples.every(sample => validWorkflowJudgeScore(sample, SAMPLE_RANGE, structuredResponse))) return false;
|
||||
const mean = judgePanelMean(value.samples as JudgeScore[], JUDGE_SCORE_DIMENSIONS);
|
||||
return JUDGE_SCORE_DIMENSIONS.every(key => mean[key] >= thresholds[key]);
|
||||
}
|
||||
|
||||
export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
lookup(): { scores: JudgeScore; reuse: WorkflowJudgeReuse } | null;
|
||||
lookup(): { samples: JudgeScore[]; reuse: WorkflowJudgeReuse } | null;
|
||||
/** The attempt guard is rechecked after synchronous input/provenance reads. */
|
||||
publish(scores: JudgeScore, isActive?: () => boolean): (() => void) | undefined;
|
||||
publish(samples: JudgeScore[], isActive?: () => boolean): (() => void) | undefined;
|
||||
} {
|
||||
const env = opts.env ?? process.env;
|
||||
const noCache = { lookup: () => null, publish: (_scores: JudgeScore) => undefined };
|
||||
const noCache = { lookup: () => null, publish: (_samples: JudgeScore[]) => undefined };
|
||||
const pr = Number(env.EVALS_CACHE_PR);
|
||||
// Runtime ID is the immutable CI image manifest, not a mutable image tag.
|
||||
// Nonstandard Node/Bun preload code or custom model endpoints need a separate
|
||||
@@ -74,7 +85,8 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
files: workflowJudgeDependencies(opts.root, input.files.map(file => file.path)),
|
||||
prompts: { [opts.testName]: prompt },
|
||||
parameters: { rootPackage, thresholds: opts.thresholds, max_tokens: opts.maxTokens ?? DEFAULT_JUDGE_MAX_TOKENS, temperature: null, budget_ms: JUDGE_MS,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 1,
|
||||
request: opts.stream ? 'messages.stream/user' : 'messages.create/user', retries: 0,
|
||||
panel: { samples: JUDGE_PANEL_SAMPLES, numeric: 'mean', boolean: 'majority' },
|
||||
...(opts.stream ? { stream: true } : {}),
|
||||
...(opts.structuredResponse ? { output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
|
||||
response_validation: { reasoning_words_below: WORKFLOW_JUDGE_REASONING_WORD_LIMIT } } : {}) },
|
||||
@@ -94,14 +106,15 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
return {
|
||||
lookup() {
|
||||
const result = lookupEvalInputCache({ ...common, identity: before,
|
||||
validateResult: value => validWorkflowJudgeScore(value, opts.thresholds, opts.structuredResponse) });
|
||||
validateResult: value => validWorkflowJudgePanel(value, opts.thresholds, opts.structuredResponse) });
|
||||
return result.status === 'reused'
|
||||
? { scores: result.result as JudgeScore, reuse: { key: result.key, source: result.source } } : null;
|
||||
? { samples: (result.result as unknown as { samples: JudgeScore[] }).samples, reuse: { key: result.key, source: result.source } } : null;
|
||||
},
|
||||
publish(scores, isActive = () => true) {
|
||||
publish(samples, isActive = () => true) {
|
||||
// Caller reaches here ONLY after its actual assertions passed. A later
|
||||
// failed case in the file does not erase this independently completed case.
|
||||
if (!isActive() || !validWorkflowJudgeScore(scores as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const panel = { samples: samples.map(({ clarity, completeness, actionability, reasoning }) => ({ clarity, completeness, actionability, reasoning })) };
|
||||
if (!isActive() || !validWorkflowJudgePanel(panel as unknown as EvalCacheValue, opts.thresholds, opts.structuredResponse)) return;
|
||||
const after = currentIdentity();
|
||||
const runId = env.GITHUB_RUN_ID ? `${env.GITHUB_RUN_ID}/${env.GITHUB_RUN_ATTEMPT ?? '1'}` : env.EVALS_RUN_ID;
|
||||
if (!after || !runId || !isActive()) return;
|
||||
@@ -112,7 +125,7 @@ export function prepareWorkflowJudgeCache(opts: WorkflowCacheOptions): {
|
||||
cancelled: false, skipped: 0, failed: 0, passed: 1,
|
||||
cases: [{ id: opts.testName, outcome: 'passed', attempt: 1 }],
|
||||
source: { runId, revision: revision.stdout.trim(), completedAt: Date.now() },
|
||||
result: { clarity: scores.clarity, completeness: scores.completeness, actionability: scores.actionability, reasoning: scores.reasoning },
|
||||
result: panel,
|
||||
} });
|
||||
// A slow synchronous write can consume the recording allowance. The
|
||||
// caller withdraws this new receipt if its final deadline check fails.
|
||||
|
||||
Reference in new issue
Block a user