mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-02 17:40:02 +02:00
test: attribute quoted prior-record field lists, state the judge reason bound in its schema, move split-overflow to marathon
Census 36629958451 reds: - outside-plan-disabled-no-fallback: the model quoted the pre-existing record as a parenthesized field list with its exact timestamp; attribution now requires that exact timestamp and the record's own field values. - plan-devex-peer-comparison-classification: the judge correctly returned missing but wrote a 1069-character reason, voiding the judgment; structured outputs cannot enforce maxLength, so the bound is stated on the field. - plan-ceo-split-overflow ran 504-1188 s as one PTY flow and set the periodic lane's wall clock; it now runs weekly in the marathon lane.
This commit is contained in:
1 parent
f63e1fb7cb
commit
bf4667146c
6 files changed
+52
-8
No files matched your search
@@ -247,3 +247,24 @@ test('b176 retained history cannot override actual current log completion or nat
|
||||
expect(disabledPlanReviewEvidence(item.result, 'codex invoked\n', item.reviewLog, item.priorRecord).passed).toBe(false);
|
||||
expect(disabledPlanReviewEvidence(sourceBoundB176.cases[1]!.result, '', sourceBoundB176.cases[1]!.reviewLog, item.priorRecord).passed).toBe(false);
|
||||
});
|
||||
|
||||
describe('parenthesized field list of the retained record (census 36629958451)', () => {
|
||||
const prior = fixture.cases[0]!.priorRecord as Record<string, string>;
|
||||
const list = (items: string[]) => `One thing worth flagging: the fixture's review log already contained a prior entry (${items.map(item => '`' + item + '`').join(', ')}) that predates this run. I did not produce it and it does not count toward this review's coverage.`;
|
||||
const exact = [prior.timestamp!, `status: ${prior.status}`, `source: ${prior.source}`, 'outside_status: completed'];
|
||||
test('the exact timestamp and the record\'s own field values attribute the quotation to that record', () => {
|
||||
expect(evaluate(0, list(exact))).toMatchObject({ passed: true, falseCompletion: false });
|
||||
});
|
||||
test.each([
|
||||
['a different timestamp', [prior.timestamp!.replace(/\d(?=Z$)/, d => String((Number(d) + 1) % 10)), ...exact.slice(1)]],
|
||||
['no timestamp', exact.slice(1)],
|
||||
['a field value the record lacks', [prior.timestamp!, 'status: completed', `source: ${prior.source}`, 'outside_status: completed']],
|
||||
['free prose inside the list', [prior.timestamp!, `status: ${prior.status}`, 'outside_status: completed and this run too']],
|
||||
])('%s is not attributed', (_name, items) => {
|
||||
expect(evaluate(0, list(items as string[])).falseCompletion).toBe(true);
|
||||
});
|
||||
test('a current mutation or a separate current claim still fails', () => {
|
||||
expect(evaluate(0, list(exact).replace('already contained', 'now appended')).falseCompletion).toBe(true);
|
||||
expect(evaluate(0, `${list(exact)}\nThis run: outside_status: completed.`).falseCompletion).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -173,6 +173,24 @@ function withoutAttributedPriorRecordData(output: string, priorRecord?: Record<s
|
||||
spans.push({ start: match.index, end: match.index + match[0].length });
|
||||
}
|
||||
}
|
||||
// A parenthesized field list right after a pre-existing-record owner is that
|
||||
// record when it quotes the record's exact ISO timestamp and every other item
|
||||
// is one of its own field values; a current mutation before the owner fails.
|
||||
const owned = /\b(?:earlier|prior|previous|historical|old(?:er)?|pre[- ]existing)\s+(?:(?:review[- ]log|review|log)\s+)?(?:entry|record|line|row)\s*\(([^()\r\n]+)\)/gi;
|
||||
for (const match of output.matchAll(owned)) {
|
||||
const lineStart = output.lastIndexOf('\n', match.index) + 1;
|
||||
const local = output.slice(lineStart, match.index).split(/(?<=[.!?;])\s+/).at(-1) ?? '';
|
||||
if (/\b(?:now|currently|current|today|new|updat\w*|append\w*|chang\w*|mark\w*|set|write|wrote)\b/i.test(local.replace(/[*`]/g, ''))) continue;
|
||||
const items = match[1]!.split(',').map(item => item.replace(/[*`]/g, '').trim());
|
||||
const fieldsOk = items.every(item => {
|
||||
if (item === priorRecord.timestamp) return true;
|
||||
const field = /^["']?([a-z_]+)["']?\s*[:=]\s*["']?([a-z0-9_.:+-]+)["']?$/i.exec(item);
|
||||
return !!field && fields.has(field[1]!) && field[1] !== 'timestamp' && priorRecord[field[1]!] === field[2];
|
||||
});
|
||||
if (!fieldsOk || !items.includes(String(priorRecord.timestamp)) || !items.some(item => /^["']?outside_status\b/i.test(item))) continue;
|
||||
const start = match.index + match[0].length - match[1]!.length - 1;
|
||||
spans.push({ start, end: start + match[1]!.length + 2 });
|
||||
}
|
||||
for (const span of spans.sort((a, b) => b.start - a.start)) {
|
||||
output = output.slice(0, span.start) + output.slice(span.start, span.end).replace(/[^\r\n]/g, ' ') + output.slice(span.end);
|
||||
}
|
||||
|
||||
@@ -54,6 +54,9 @@ export interface PlanReviewDecisionJudgment {
|
||||
engReview?: EngReviewJudgment;
|
||||
}
|
||||
export type PlanReviewJudge = (prompt: string, model?: string, opts?: Pick<CallJudgeOptions, 'signal' | 'max_tokens' | 'jsonSchema'>) => Promise<unknown>;
|
||||
// Structured outputs cannot enforce maxLength, so the reason bound the local
|
||||
// validator applies is stated on the field the model writes.
|
||||
const REASON_FIELD = { type: 'string', description: '1-1000 characters: under 120 words.' } as const;
|
||||
// Only response structure is constrained. Identity, exact quotes, enum casing,
|
||||
// uncertainty, target coverage, independence and count checks remain local.
|
||||
function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = false): NonNullable<CallJudgeOptions['jsonSchema']> {
|
||||
@@ -68,7 +71,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f
|
||||
toolUseId: { type: 'string' }, questionIndex: { type: 'integer' },
|
||||
kind: { type: 'string', enum: ['finding', 'scope', 'workflow', 'backlog', 'uncertain'] },
|
||||
targetIds: { type: 'array', items: { type: 'string' } },
|
||||
independentDecisions: { type: 'integer' }, reason: { type: 'string' },
|
||||
independentDecisions: { type: 'integer' }, reason: REASON_FIELD,
|
||||
evidence: { type: 'array', items: {
|
||||
type: 'object', additionalProperties: false, required: ['field', 'optionIndex', 'quote'],
|
||||
properties: {
|
||||
@@ -86,7 +89,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f
|
||||
type: 'object', additionalProperties: false,
|
||||
required: ['status', 'regression', 'approvals', 'navigation', 'reason'],
|
||||
properties: {
|
||||
status: { type: 'string', enum: ['complete', 'missing', 'uncertain'] }, reason: { type: 'string' },
|
||||
status: { type: 'string', enum: ['complete', 'missing', 'uncertain'] }, reason: REASON_FIELD,
|
||||
regression: { type: 'array', items: { type: 'object', additionalProperties: false,
|
||||
required: ['role', 'source', 'quote'], properties: {
|
||||
role: { type: 'string', enum: ['critical', 'baseline', 'replay', 'assertions', 'approved-differences'] },
|
||||
@@ -112,7 +115,7 @@ function planReviewDecisionSchema(withPeerComparison: boolean, withEngReview = f
|
||||
properties: { name: { type: 'string' }, quote: { type: 'string' } },
|
||||
} },
|
||||
productQuote: { type: 'string' }, groundingQuote: { type: 'string' },
|
||||
implicationQuote: { type: 'string' }, reason: { type: 'string' },
|
||||
implicationQuote: { type: 'string' }, reason: REASON_FIELD,
|
||||
},
|
||||
} } : {}),
|
||||
},
|
||||
|
||||
@@ -1276,7 +1276,7 @@ export const E2E_TIERS: Record<string, 'gate' | 'periodic' | 'marathon'> = {
|
||||
'plan-design-finding-floor': 'periodic', // stochastic ask-first (see plan-mode-handshake note); periodic
|
||||
'plan-devex-finding-floor': 'gate',
|
||||
'plan-eng-multi-finding-batching': 'periodic',
|
||||
'plan-ceo-split-overflow': 'periodic',
|
||||
'plan-ceo-split-overflow': 'marathon', // Full /plan-ceo-review through split overflow (504–1188 s on 2.1.251)
|
||||
|
||||
// Privacy gate for gstack-brain-sync — periodic (non-deterministic LLM call,
|
||||
// costs ~$0.30-$0.50 per run, not needed on every commit)
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* /plan-ceo-review split-overflow regression (periodic, paid, real-PTY).
|
||||
* /plan-ceo-review split-overflow regression (marathon, paid, real-PTY).
|
||||
*
|
||||
* Catches the original failure mode the user complained about: when the
|
||||
* agent has 5+ options for ONE conceptual decision, it must split into N
|
||||
@@ -50,7 +50,7 @@ import { ceoSplitDecisionFingerprints, isCeoSplitCandidateCall, isCeoSplitCollec
|
||||
import { CEO_SCOPE_CANDIDATES } from './helpers/plan-review-cases';
|
||||
import { evaluatePlanReviewDecisions } from './helpers/plan-review-decisions';
|
||||
|
||||
const describeE2E = describeE2ETier('periodic');
|
||||
const describeE2E = describeE2ETier('marathon');
|
||||
|
||||
const N = 5;
|
||||
const FLOOR = N - 1; // 4 — must fire at least one AUQ per non-dropped option
|
||||
@@ -60,7 +60,7 @@ const FLOOR = N - 1; // 4 — must fire at least one AUQ per non-dropped option
|
||||
* EVALS_JOBS>1, sibling worktrees) never share one /tmp artifact. */
|
||||
const FIXTURE_PLAN_PATH = '/tmp/gstack-test-plan-ceo-split-overflow.md';
|
||||
|
||||
describeE2E('/plan-ceo-review split-overflow regression (periodic)', () => {
|
||||
describeE2E('/plan-ceo-review split-overflow regression (marathon)', () => {
|
||||
test(
|
||||
`5-option scope decision emits >= ${FLOOR} review-phase AskUserQuestions (no dropping)`,
|
||||
async () => {
|
||||
|
||||
Reference in new issue
Block a user