Files
gstack/test/llm-judge-stream.test.ts
T
Garry Tan dcaea52800 v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates

* test: preserve delegated QA setup authority after main integration

* fix(qa): clarify exploration order and preserve report artifacts

* test(qa): follow the shared setup reference directly

* refactor(ship): make verification and recovery routes explicit

* test(ship): align evidence and review guards with explicit routes

* fix(workflows): clarify ship recovery and functional QA evidence

* fix(workflows): clarify approval recovery and full QA coverage

* refactor(workflows): order review transactions and clarify ship state

* fix(ship): clarify final verification and fail closed at publication

* fix(evals): attribute native atomic documentation writes

* fix(ship): clarify recovery and documentation lifecycle guidance

* fix(test): preserve observed native placeholder styling in CI

* fix(codex): report watchdog timeouts without a process-exit race

* Checkpoint functional QA implementation and workflow validation repairs

* Fix documentation and shared-review fixture contracts

* docs: clarify judge reuse and evaluation supervision

* test: align review evidence and selected case contracts

* test: verify append-only documentation checkpoints and recovery

* fix: qualify QA workflows and CI validation repairs

* fix: launch shared-libs fixture scripts on Windows

* fix: qualify QA deadlines, fixture isolation, and shard cleanup

* fix: preserve qualified QA and cancellation repairs

* fix: enforce functional fixture authority and share strict event decoding

* fix: retain free-test evidence and explain recovery

* fix: reject malformed native evidence after decoder consolidation

* test: use reliable capture for telemetry privacy filters

* test: refresh measured quick coverage and document validation costs

* Fix native fixture receipts and preserve VM validation evidence

* Align negative judge controls with upstream clarity policy

* Fix report-only QA preparation and public evidence handling

* Clarify QA-only preparation and current-report preservation

* Stream Ship quality judgments with an explicit 64k response contract

* Validate compact judge reasoning locally with supported wire schema

* Align functional QA fixture instructions with evidence acceptance

* Bind native browser diagnostics to execution evidence and align review verdicts

* Preserve native diagnostic line boundaries

* Serialize functional QA evidence from native captures

* Keep large QA evidence fixture payload out of Windows argv
2026-09-29 06:07:35 -07:00

92 lines
5.0 KiB
TypeScript

import { afterEach, beforeEach, expect, spyOn, test } from 'bun:test';
import { callJudge } from './helpers/llm-judge';
import { WORKFLOW_JUDGE_RESPONSE_SCHEMA } from './helpers/workflow-judge-input';
const scores = { clarity: 4, completeness: 4, actionability: 4, reasoning: 'Complete workflow with explicit gates.' };
let originalKey: string | undefined;
let transport: ReturnType<typeof spyOn>;
let diagnostics: ReturnType<typeof spyOn>;
beforeEach(() => {
originalKey = process.env.ANTHROPIC_API_KEY;
process.env.ANTHROPIC_API_KEY = 'test-only-key';
transport = spyOn(globalThis, 'fetch');
diagnostics = spyOn(console, 'error').mockImplementation(() => {});
});
afterEach(() => {
transport.mockRestore(); diagnostics.mockRestore();
if (originalKey === undefined) delete process.env.ANTHROPIC_API_KEY;
else process.env.ANTHROPIC_API_KEY = originalKey;
});
function response(stopReason = 'end_turn', text: string | null = JSON.stringify(scores)) {
const events = [
{ type: 'message_start', message: { id: 'msg_fixture', type: 'message', role: 'assistant', model: 'claude-fable-5-1',
content: [], stop_reason: null, stop_sequence: null, usage: { input_tokens: 99023, output_tokens: 1 } } },
{ type: 'content_block_start', index: 0, content_block: { type: 'thinking', thinking: '', signature: '' } },
{ type: 'content_block_delta', index: 0, delta: { type: 'thinking_delta', thinking: 'PRIVATE_THINKING' } },
{ type: 'content_block_delta', index: 0, delta: { type: 'signature_delta', signature: 'PRIVATE_SIGNATURE' } },
{ type: 'content_block_stop', index: 0 },
...(text === null ? [] : [
{ type: 'content_block_start', index: 1, content_block: { type: 'text', text: '' } },
{ type: 'content_block_delta', index: 1, delta: { type: 'text_delta', text } },
{ type: 'content_block_stop', index: 1 },
]),
{ type: 'message_delta', delta: { stop_reason: stopReason, stop_sequence: null }, usage: { output_tokens: 65536 } },
{ type: 'message_stop' },
];
return new Response(events.map(event => `event: ${event.type}\ndata: ${JSON.stringify(event)}\n\n`).join(''),
{ headers: { 'content-type': 'text/event-stream' } });
}
test('the pinned SDK refuses a default nonstreaming 64k request before network access', async () => {
transport.mockImplementation(() => { throw new Error('Unexpected network access'); });
await expect(callJudge('score the whole bundle', 'claude-fable-5-1', { max_tokens: 65_536 }))
.rejects.toThrow('Streaming is required');
expect(transport).not.toHaveBeenCalled();
});
test('the real SDK streams 64k requests and parses only completed public text', async () => {
transport.mockResolvedValue(response());
expect(await callJudge('score the whole bundle', 'claude-fable-5-1', {
max_tokens: 65_536, stream: true, jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA,
})).toEqual(scores);
expect(transport).toHaveBeenCalledTimes(1);
const request = transport.mock.calls[0][1];
expect(JSON.parse(request.body)).toEqual({ model: 'claude-fable-5-1', max_tokens: 65_536,
output_config: { format: { type: 'json_schema', schema: WORKFLOW_JUDGE_RESPONSE_SCHEMA } },
messages: [{ role: 'user', content: 'score the whole bundle' }], stream: true });
expect(new Headers(request.headers).has('anthropic-beta')).toBe(false);
expect(diagnostics).not.toHaveBeenCalled();
});
test('streamed truncation and refusal retain public evidence but never partial scores or private thinking', async () => {
for (const [stopReason, text] of [['max_tokens', '{"clarity":4'], ['max_tokens', null], ['refusal', null]] as const) {
transport.mockResolvedValue(response(stopReason, text));
await expect(callJudge('score the whole bundle', 'claude-fable-5-1', {
max_tokens: 65_536, stream: true, jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA,
})).rejects.toThrow(stopReason === 'max_tokens' ? 'truncated at max_tokens=65536' : 'provider refused');
const diagnostic = diagnostics.mock.calls.at(-1)![0];
expect(diagnostic).not.toContain('PRIVATE_');
expect(JSON.parse(diagnostic)).toMatchObject({ stopReason, textBlocks: text === null ? [] : [text] });
}
});
test('streaming retains the caller abort signal instead of extending its deadline', async () => {
let started!: () => void;
const ready = new Promise<void>(resolve => { started = resolve; });
transport.mockImplementation((_url: string, init: RequestInit) => new Promise((_resolve, reject) => {
init.signal!.addEventListener('abort', () => reject(init.signal!.reason), { once: true });
started();
}));
const controller = new AbortController();
const pending = callJudge('score the whole bundle', 'claude-fable-5-1', {
max_tokens: 65_536, stream: true, jsonSchema: WORKFLOW_JUDGE_RESPONSE_SCHEMA, signal: controller.signal,
});
const failure = new Error('Original workflow walltime elapsed');
const rejected = pending.catch(error => error);
await ready;
controller.abort(failure);
expect(await rejected).toBe(failure);
expect(transport).toHaveBeenCalledTimes(1);
});