Merge origin/main (v1.91.7.0) into test-audit-reduction

Keep both intents: v1.91.7.0's functional QA, docsync and exploratory
paid cases and their free owners stay; this branch's deletions stay
deleted. main's new paid keys follow the derived-closure touchfile rule
(free *.test.ts paths dropped, static helper/fixture closure added), its
new helper-only tests join the ratchet baseline, and its free selection
examples that named free test files now assert the derived selection.

Periodic CI keeps seven slices without the retired Autoplan slice; the
gate census keeps seven single-worker slices with --skip-judges. Wall
and census literals are recomputed from the merged planner, durations
are re-recorded on Ubicloud, and VERSION stays 1.91.8.0 above 1.91.7.0.
This commit is contained in:
garrytan committed 2026-09-29 13:48:02 +00:00
commit b421bba2c9
325 files changed
+42569 -8257

No files matched your search

+148 -7
View File
@@ -8,7 +8,7 @@ import * as os from 'node:os';
import * as path from 'node:path';
import { execFileSync } from 'node:child_process';
import { prepareWorkflowJudgeCache, validWorkflowJudgeScore, workflowJudgeDependencies, type WorkflowCacheOptions } from './helpers/workflow-judge-cache';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt } from './helpers/workflow-judge-input';
import { readWorkflowJudgeInput, buildWorkflowJudgePrompt, QA_DISCOVERY_REFERENCES, WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
const roots: string[] = [];
afterEach(() => { for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); });
@@ -124,6 +124,20 @@ test('runtime/model/threshold changes miss, and retries never reuse or publish',
expect(retry.lookup()).toBeNull(); retry.publish(scores); expect(f.entries()).toHaveLength(1);
});
test('frontier reader calibration cannot reuse a score from the unspecified-reader rubric', () => {
const f = fixture(); f.cache().publish(scores);
const original = f.opts.prompt;
f.opts.agentCapability = 'frontier'; f.refreshPrompt();
expect(f.opts.prompt).not.toBe(original);
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
delete f.opts.agentCapability; f.refreshPrompt();
expect(f.opts.prompt).toBe(original);
expect(f.cache().lookup()?.scores).toEqual(scores);
});
test('a pinned workflow judge model overrides the global model and changes the cache identity', () => {
const f = fixture();
f.opts.model = 'claude-sonnet-4-6';
@@ -153,7 +167,7 @@ test('workflow registration preserves model work and reserves only terminal-reco
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const body = source.split('async function runWorkflowJudge')[1]!.split('// Block 1:')[0]!;
const stages = ['workflowJudgeAttempts.set', 'readWorkflowJudgeInput(', 'cache.lookup()',
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens })',
'callJudge<JudgeScore>(prompt, opts.model, { signal: controller.signal, max_tokens: maxTokens,',
'expect(scores.clarity)', 'expect(scores.completeness)', 'expect(scores.actionability)', 'cache.publish(scores, active)']
.map(stage => body.indexOf(stage));
expect(stages.every(position => position >= 0)).toBe(true);
@@ -162,12 +176,13 @@ test('workflow registration preserves model work and reserves only terminal-reco
expect(body).toContain('const workDeadline = started + JUDGE_MS;');
expect(source).toContain('const WORKFLOW_JUDGE_RECORD_MS = 5_000;');
expect(source).toContain('const WORKFLOW_JUDGE_TEST_MS = JUDGE_MS + 10_000;');
expect(source.match(/\}, WORKFLOW_JUDGE_TEST_MS\);/g)).toHaveLength(16);
expect(source.match(/\}, WORKFLOW_JUDGE_TEST_MS\);/g)).toHaveLength(17);
expect(source).toContain("testName: 'review/SKILL.md workflow'");
expect(source.match(/\}, JUDGE_MS\);/g)).toHaveLength(7);
});
function actualCallback(f: ReturnType<typeof fixture>, overrides: {
judge?: (prompt: string, model: undefined, options: { signal: AbortSignal }) => Promise<typeof scores>;
judge?: (prompt: string, model: string | undefined, options: { signal: AbortSignal }) => Promise<typeof scores>;
read?: typeof readWorkflowJudgeInput;
prepare?: typeof prepareWorkflowJudgeCache;
clock?: () => number;
@@ -186,20 +201,55 @@ function actualCallback(f: ReturnType<typeof fixture>, overrides: {
const run = new Function('ROOT', 'readWorkflowJudgeInput',
'buildWorkflowJudgePrompt', 'prepareWorkflowJudgeCache', 'workflowJudgeAttempts', 'callJudge',
'evalCollector', 'expect', 'console', 'performance', 'JUDGE_MS', 'WORKFLOW_JUDGE_RECORD_MS',
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS',
'setTimeout', 'clearTimeout', 'JudgeRefusalError', 'getCookieWorkflowManualReview', 'DEFAULT_JUDGE_MAX_TOKENS', 'resolveEvalModel',
'WORKFLOW_JUDGE_RESPONSE_SCHEMA', 'validWorkflowJudgeScore',
`${javascript}\nreturn runWorkflowJudge;`)(
f.root, overrides.read ?? readWorkflowJudgeInput, buildWorkflowJudgePrompt,
(options: WorkflowCacheOptions) => (overrides.prepare ?? prepareWorkflowJudgeCache)({ ...options, env: f.env }),
attempts, async (prompt: string, model: undefined, options: { signal: AbortSignal }) => {
attempts, async (prompt: string, model: string | undefined, options: { signal: AbortSignal }) => {
prompts.push(prompt); signals.push(options.signal);
return overrides.judge ? overrides.judge(prompt, model, options) : scores;
}, { addTest: (entry: any) => records.push(entry) }, expect, { log() {} },
overrides.clock ? { now: overrides.clock } : performance, overrides.budget ?? 120_000, overrides.allowance ?? 5_000,
overrides.setTimer ?? setTimeout, overrides.clearTimer ?? clearTimeout,
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS);
JudgeRefusalError, getCookieWorkflowManualReview, DEFAULT_JUDGE_MAX_TOKENS, resolveEvalModel,
WORKFLOW_JUDGE_RESPONSE_SCHEMA, validWorkflowJudgeScore);
return { run, records, signals, prompts, attempts, options: { ...f.opts, suite: 'Cache regression' } };
}
test('the actual workflow callback preserves the pinned model and frontier rubric for custom inputs', async () => {
const f = fixture();
const models: Array<string | undefined> = [];
const actual = actualCallback(f, { judge: async (_prompt, model) => { models.push(model); return scores; } });
await actual.run({ ...actual.options, model: 'claude-sonnet-4-6', agentCapability: 'frontier',
readInput: () => readWorkflowJudgeInput(f.opts) });
expect(models).toEqual(['claude-sonnet-4-6']);
expect(actual.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(actual.records[0]).toMatchObject({ passed: true, model: 'claude-sonnet-4-6', prompt: actual.prompts[0] });
});
test.each(['ship', 'review'])('the registered %s callback sends the frontier rubric and still rejects subthreshold clarity', async skill => {
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const registration = source.match(new RegExp(`testIfSelected\\('${skill}/SKILL\\.md workflow',[\\s\\S]*?await runWorkflowJudge\\(\\{([\\s\\S]*?)\\n \\}\\);`));
expect(registration).not.toBeNull();
const registered = new Function('QA_DISCOVERY_REFERENCES', `return ({${registration![1]}});`)(QA_DISCOVERY_REFERENCES);
const f = fixture();
Object.assign(f.env, { EVALS_FRESH: '1' });
const options = { ...registered, skillPath: f.opts.skillPath, startMarker: f.opts.startMarker,
endMarker: f.opts.endMarker, references: [] };
const passing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 3 }) });
await passing.run(options);
expect(passing.prompts).toHaveLength(1);
expect(passing.prompts[0]).toContain('GPT-5.6 Sol-level capability or stronger');
expect(passing.records[0]).toMatchObject({ passed: true, execution: 'executed', judge_scores: { clarity: 3 } });
const failing = actualCallback(f, { judge: async () => ({ ...scores, clarity: 2 }) });
await expect(failing.run(options)).rejects.toThrow();
expect(failing.prompts).toEqual(passing.prompts);
expect(failing.records[0]).toMatchObject({ passed: false, execution: 'executed',
exit_reason: 'validation_failed', judge_scores: { clarity: 2 } });
expect(f.entries()).toHaveLength(0);
});
test('the actual workflow callback executes once, reuses with provenance, and preserves assertion failures', async () => {
const f = fixture(); const first = actualCallback(f);
const options = { ...f.opts, suite: 'Cache regression' };
@@ -332,3 +382,94 @@ test('the actual workflow callback preserves the complete public API body; cance
expect(h.records[0].passed).toBe(true);
} finally { create.mockRestore(); }
});
test('Ship sends its authorized 64k cap and compact response contract through the streaming SDK boundary', async () => {
const source = fs.readFileSync(path.join(import.meta.dir, 'skill-llm-eval.test.ts'), 'utf8');
const registration = source.match(/testIfSelected\('ship\/SKILL\.md workflow',[\s\S]*?await runWorkflowJudge\(\{([\s\S]*?)\n \}\);/);
expect(registration).not.toBeNull();
const options = new Function('QA_DISCOVERY_REFERENCES', `return ({${registration![1]}});`)(QA_DISCOVERY_REFERENCES);
expect(options.structuredResponse).toBe(true);
expect(options.maxTokens).toBe(65_536);
expect(options.stream).toBe(true);
expect(WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning).not.toHaveProperty('pattern');
expect(WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning).not.toHaveProperty('maxLength');
expect(source.match(/structuredResponse: true/g)).toHaveLength(1);
const f = fixture();
const stream = spyOn(Messages.prototype, 'stream').mockReturnValue({ finalMessage: async () => ({
stop_reason: 'end_turn', content: [{ type: 'text', text: JSON.stringify(scores) }],
}) } as any);
try {
const h = actualCallback(f, { judge: (prompt, model, request) => callJudge<typeof scores>(prompt, model, request) });
await h.run({ ...h.options, structuredResponse: options.structuredResponse, maxTokens: options.maxTokens, stream: options.stream });
expect(stream.mock.calls[0]).toEqual([{
model: resolveEvalModel('judge'), max_tokens: 65_536,
output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
messages: [{ role: 'user', content: f.opts.prompt }],
}, { signal: h.signals[0] }]);
expect(h.records[0]).toMatchObject({ passed: true, judge_scores: { clarity: 4, completeness: 5, actionability: 4 } });
expect(f.entries()).toHaveLength(1);
expect(f.cache().lookup()).toBeNull();
f.opts.structuredResponse = true;
f.opts.maxTokens = 65_536;
f.opts.stream = true;
expect(f.cache().lookup()?.scores).toEqual(scores);
} finally { stream.mockRestore(); }
});
test('changing response serialization misses the cache even when prompt and model match', () => {
const f = fixture(); f.cache().publish(scores);
f.opts.structuredResponse = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
expect(f.entries()).toHaveLength(2);
expect(f.cache().lookup()?.scores).toEqual(scores);
const description = WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description;
try {
WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description += ' Changed response contract.';
expect(f.cache().lookup()).toBeNull();
} finally { WORKFLOW_JUDGE_RESPONSE_SCHEMA.properties.reasoning.description = description; }
expect(f.cache().lookup()?.scores).toEqual(scores);
f.opts.structuredResponse = false;
expect(f.cache().lookup()?.scores).toEqual(scores);
});
test('the actual cap and streaming transport independently affect workflow cache identity', () => {
const f = fixture(); f.cache().publish(scores);
f.opts.maxTokens = 65_536;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
f.opts.stream = true;
expect(f.cache().lookup()).toBeNull();
f.cache().publish(scores);
expect(f.entries()).toHaveLength(3);
expect(f.cache().lookup()?.scores).toEqual(scores);
delete f.opts.maxTokens;
delete f.opts.stream;
expect(f.cache().lookup()?.scores).toEqual(scores);
});
test('the structured callback rejects incomplete, schema-invalid and below-threshold answers without cache credit', async () => {
const invalid = [null, {}, { ...scores, extra: 'not a score field' }, { ...scores, clarity: '4' },
{ ...scores, clarity: 4.5 }, { ...scores, clarity: 6 }, { ...scores, reasoning: '' },
{ ...scores, reasoning: Array(150).fill('word').join(' ') }, { ...scores, clarity: 2 },
{ ...scores, completeness: 2 }, { ...scores, actionability: 3 }];
const responses: Array<{ stop_reason: string; value: unknown }> = [
...['max_tokens', 'refusal', 'stop_sequence'].map(stop_reason => ({ stop_reason, value: scores })),
...invalid.map(value => ({ stop_reason: 'end_turn', value })),
];
const stream = spyOn(Messages.prototype, 'stream');
const diagnostics = spyOn(console, 'error').mockImplementation(() => {});
try {
for (const response of responses) {
const f = fixture();
stream.mockReturnValue({ finalMessage: async () => ({ stop_reason: response.stop_reason,
content: [{ type: 'text', text: JSON.stringify(response.value) }] }) } as any);
const h = actualCallback(f, { judge: (prompt, model, request) => callJudge<typeof scores>(prompt, model, request) });
await expect(h.run({ ...h.options, structuredResponse: true, maxTokens: 65_536, stream: true })).rejects.toThrow();
expect(h.records).toHaveLength(1);
expect(h.records[0].passed).toBe(false);
expect(f.entries()).toHaveLength(0);
}
expect(validWorkflowJudgeScore({ ...scores, reasoning: Array(149).fill('word').join(' ') }, { clarity: 1, completeness: 1, actionability: 1 }, true)).toBe(true);
} finally { stream.mockRestore(); diagnostics.mockRestore(); }
});