fix(evals): repair proof-run reds in design-consultation, document-release, design and QA fixtures

- design-consultation Phase 1 asks one brief that confirms context and decides
  research; the confirm-only first question scored substance 2.
- document-release defines ship-owned inputs, exact steps and the JSON result,
  and drops stale spawned-from-/ship text (judge actionability 3.67 -> 4/4/4).
- plan-design-with-ui accepts the Step 0D focus menu the same way the shared
  picker does ("focus on specific ones?").
- plan-design-review plan-mode saves in three Edits instead of one final Write.
- QA functional annotations ask for the full 40-character revision.
- Outside-disabled attribution judges quoted prior-record data by its exact
  timestamp or a dated, pre-existing-record sentence; four captured phrasings
  replay clean and current claims still fail.
- --case can select autoplan-dual-voice by its literal test name.
This commit is contained in:
garrytan committed 2026-09-29 22:36:28 +00:00
1 parent aba80c8fb6
commit a111225e78
17 files changed
+180 -92

No files matched your search

+36 -10
View File
@@ -145,22 +145,22 @@ function withoutAttributedPriorRecordData(output: string, priorRecord?: Record<s
// An inline quotation of the retained record's exact status/source/outside_status
// values is that record when its own sentence names it as pre-existing and
// makes no current claim; wording order around the quotation does not matter.
// A named record timestamp must denote the retained record's instant at the precision written.
const priorMs = typeof priorRecord.timestamp === 'string' ? Date.parse(priorRecord.timestamp) : NaN;
const sameInstant = (stamp: string): boolean => {
if (!Number.isFinite(priorMs)) return false;
const iso = priorMs ? new Date(priorMs).toISOString() : '';
const clock = /^(\d{2}:\d{2}(?::\d{2}(?:\.\d{1,3})?)?)Z?$/.exec(stamp);
if (clock) return iso.slice(11, 11 + clock[1]!.length) === clock[1];
const at = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?Z$/.test(stamp) ? Date.parse(stamp) : NaN;
return Number.isFinite(at) && iso.slice(0, stamp.includes('.') ? 23 : stamp.length - 1) === new Date(at).toISOString().slice(0, stamp.includes('.') ? 23 : stamp.length - 1);
};
const sentenceOwnsPriorValue = (index: number, length: number): boolean => {
const start = Math.max(output.lastIndexOf('\n', index - 1), ...['. ', '! ', '? ', '; '].map(end => output.lastIndexOf(end, index - 1) + 1)) + 1;
const ends = ['\n', '. ', '! ', '? ', '; '].map(end => output.indexOf(end, index + length)).filter(at => at >= 0);
const sentence = (output.slice(start, index) + ' ' + output.slice(index + length, ends.length ? Math.min(...ends) : output.length))
.replace(/[*`]/g, '').replace(/\b(?:predates|before)\s+(?:this|my)\s+(?:run|session|workflow)(?:\s+(?:started|began))?\b/gi, 'beforehand')
.replace(/\b(?:I|we)\s+(?:did\s+not|didn't|never)\s+(?:write|create|produce|record)\b/gi, 'unauthored');
// A named record timestamp must denote the retained record's instant at the precision written.
const priorMs = typeof priorRecord.timestamp === 'string' ? Date.parse(priorRecord.timestamp) : NaN;
const sameInstant = (stamp: string): boolean => {
if (!Number.isFinite(priorMs)) return false;
const iso = priorMs ? new Date(priorMs).toISOString() : '';
const clock = /^(\d{2}:\d{2}(?::\d{2})?)Z?$/.exec(stamp);
if (clock) return iso.slice(11, 11 + clock[1]!.length) === clock[1];
const at = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?Z$/.test(stamp) ? Date.parse(stamp) : NaN;
return Number.isFinite(at) && iso.slice(0, stamp.includes('.') ? 23 : stamp.length - 1) === new Date(at).toISOString().slice(0, stamp.includes('.') ? 23 : stamp.length - 1);
};
const stamps = [...sentence.matchAll(/\btimestamp(?:ed)?\s+([0-9T:.Z-]+)/gi)].map(stamp => stamp[1]!.replace(/[.,;:]+$/, ''));
if (stamps.some(stamp => !sameInstant(stamp))) return false;
return !/\b(?:after|another|other|if|unless)\b/i.test(sentence)
@@ -191,6 +191,32 @@ function withoutAttributedPriorRecordData(output: string, priorRecord?: Record<s
const start = match.index + match[0].length - match[1]!.length - 1;
spans.push({ start, end: start + match[1]!.length + 2 });
}
// A quoted fragment carrying the retained record's exact timestamp is that
// record's data when every field it quotes has that record's value.
for (const match of output.matchAll(/`([^`\r\n]+)`/g)) {
if (typeof priorRecord.timestamp !== 'string' || !match[1]!.includes(priorRecord.timestamp)) continue;
const pairs = [...match[1]!.matchAll(/["']?([a-z_]+)["']?\s*[:=]\s*["']?([^"',}\s]+)["']?/gi)].filter(pair => fields.has(pair[1]!));
if (!pairs.some(pair => pair[1] === 'outside_status') || pairs.some(pair => priorRecord[pair[1]!] !== pair[2])) continue;
spans.push({ start: match.index!, end: match.index! + match[0].length });
}
// A whole sentence that names the pre-existing record, dates it before this
// run (its exact instant or an explicit "before this run"), quotes only that
// record's own field values and makes no current claim is that record's
// report, however its fields are quoted or split.
for (const sentence of output.matchAll(/[^\n.!?;]*(?:[.!?;](?=\S)[^\n.!?;]*)*(?:[.!?;](?=\s|$)|\n|$)/g)) {
const plain = sentence[0].replace(/[*`]/g, '');
if (!/\boutside_status["']*\s*[:=]\s*["']*completed\b/i.test(plain)) continue;
if (!/\b(?:earlier|prior|previous|historical|old(?:er)?|pre[- ]existing|stale|seeded)\s+(?:(?:review[- ]log|review|log)\s+)?(?:entry|record|line|row)\b/i.test(plain)) continue;
const beforeRun = /\b(?:predates|before)\s+(?:this|my)\s+(?:run|session|workflow)(?:\s+(?:started|began))?\b/i;
const stamps = [...plain.matchAll(/\b(?:\d{4}-\d{2}-\d{2}T)?\d{2}:\d{2}(?::\d{2}(?:\.\d{1,3})?)?Z?\b/g)].map(m => m[0]);
if (stamps.some(stamp => !sameInstant(stamp)) || (!stamps.length && !beforeRun.test(plain))) continue;
const quoted = [...plain.matchAll(/\b([a-z_]+)["']?\s*[:=]\s*["']?([a-z0-9_.+-]+)["']?/gi)].filter(m => fields.has(m[1]!) && m[1] !== 'timestamp');
if (!['status', 'source', 'outside_status'].every(key => quoted.some(m => m[1] === key))
|| quoted.some(m => priorRecord[m[1]!] !== m[2])) continue;
if (/\b(?:now|currently|current|today|new|updat\w*|append\w*|chang\w*|wrote|recorded by me)\b|\bboth reviewers agree\b/i
.test(plain.replace(beforeRun, ''))) continue;
spans.push({ start: sentence.index!, end: sentence.index! + sentence[0].length });
}
for (const span of spans.sort((a, b) => b.start - a.start)) {
output = output.slice(0, span.start) + output.slice(span.start, span.end).replace(/[^\r\n]/g, ' ') + output.slice(span.end);
}
+1 -1
View File
@@ -61,7 +61,7 @@ Fixture execution boundary:
Actions outside this interface are unsupported and fail acceptance; they are not implicitly approved.
Materialize qa-reports/evidence.json first, then write a concise qa-reports/report.md using the functional report structure. Link the evidence and checkpoint files rather than repeating full probe payloads in Markdown. Both artifacts are required before completion. The resulting evidence.json schema is:
{ "revision": "<git HEAD>", "runtime": "bun <version>", "cwd": "<working directory>", "evidence": [{"command":"<exact full outer capture invocation>","contract":"README.md","expected":"<declared expected behavior>","classification":"pass|product-defect|setup-blocked|inconclusive","observed":<complete unchanged JSON emitted by the native probe>}], "learning":[{"observationCommand":"<earlier full capture invocation>","hypothesis":"<what it taught you to challenge>","nextCommand":"<later full capture invocation>"}], "limits":["<untested or blocked coverage>"] }
{ "revision": "<full 40-character git rev-parse HEAD>", "runtime": "bun <version>", "cwd": "<working directory>", "evidence": [{"command":"<exact full outer capture invocation>","contract":"README.md","expected":"<declared expected behavior>","classification":"pass|product-defect|setup-blocked|inconclusive","observed":<complete unchanged JSON emitted by the native probe>}], "learning":[{"observationCommand":"<earlier full capture invocation>","hypothesis":"<what it taught you to challenge>","nextCommand":"<later full capture invocation>"}], "limits":["<untested or blocked coverage>"] }
Evidence rows contain ONLY complete JSON actually emitted by native probes, including failures and repeats; retain pre-repair results alongside green results. Never synthesize JSON from a tool error or raw test output. Put tests, raw CLI diagnostics, launch failures and timeouts in Markdown with their actual output and limits. The learning array is a summary: choose one completed checkpoint where an observation motivated a different later command, not the required same-command replay. Select that checkpoint ID in annotations.learning; the production helper copies its observationCommand, hypothesis and nextCommand. Both commands must name exact captured probes with different native child commands, never a combined command list or a replay distinguished only by capture ID. This selects existing exploration evidence, not another probe or a duplicate of the complete checkpoint ledger. Preserve every checkpoint and link every checkpoint in Markdown; keep every executed probe and its complete JSON in evidence, including the required replay. Missing dependencies remain setup blockers, not repairs. No browser installation or execution is needed.`;
}