mirror of
https://github.com/garrytan/gstack.git
synced 2026-10-04 02:16:56 +02:00
v1.91.7.0 feat: add functional QA and pre-publication docs checks (#2983)
* feat: add surface-aware exploratory QA and ship documentation gates * test: preserve delegated QA setup authority after main integration * fix(qa): clarify exploration order and preserve report artifacts * test(qa): follow the shared setup reference directly * refactor(ship): make verification and recovery routes explicit * test(ship): align evidence and review guards with explicit routes * fix(workflows): clarify ship recovery and functional QA evidence * fix(workflows): clarify approval recovery and full QA coverage * refactor(workflows): order review transactions and clarify ship state * fix(ship): clarify final verification and fail closed at publication * fix(evals): attribute native atomic documentation writes * fix(ship): clarify recovery and documentation lifecycle guidance * fix(test): preserve observed native placeholder styling in CI * fix(codex): report watchdog timeouts without a process-exit race * Checkpoint functional QA implementation and workflow validation repairs * Fix documentation and shared-review fixture contracts * docs: clarify judge reuse and evaluation supervision * test: align review evidence and selected case contracts * test: verify append-only documentation checkpoints and recovery * fix: qualify QA workflows and CI validation repairs * fix: launch shared-libs fixture scripts on Windows * fix: qualify QA deadlines, fixture isolation, and shard cleanup * fix: preserve qualified QA and cancellation repairs * fix: enforce functional fixture authority and share strict event decoding * fix: retain free-test evidence and explain recovery * fix: reject malformed native evidence after decoder consolidation * test: use reliable capture for telemetry privacy filters * test: refresh measured quick coverage and document validation costs * Fix native fixture receipts and preserve VM validation evidence * Align negative judge controls with upstream clarity policy * Fix report-only QA preparation and public evidence handling * Clarify QA-only preparation and current-report preservation * Stream Ship quality judgments with an explicit 64k response contract * Validate compact judge reasoning locally with supported wire schema * Align functional QA fixture instructions with evidence acceptance * Bind native browser diagnostics to execution evidence and align review verdicts * Preserve native diagnostic line boundaries * Serialize functional QA evidence from native captures * Keep large QA evidence fixture payload out of Windows argv
This commit is contained in:
1 parent
65bfb0ce49
commit
dcaea52800
333 files changed
+41755
-7357
No files matched your search
@@ -53,7 +53,8 @@ test('CEO decision cycle returns to its caller with two complete persistence che
|
||||
expect(cycle.match(/\*\*(?:Pre-question|Post-answer) checkpoint:\*\*/g)).toHaveLength(2);
|
||||
expect(cycle).toContain('Start at step 1. Reuse exact prior approvals');
|
||||
expect(cycle).toContain('run steps 2–4 only when a new answer is needed, even for one option');
|
||||
expect(cycle).toContain('0D never restarts mode selection');
|
||||
expect(cycle).toContain('0D returns to its caller, not to mode selection');
|
||||
expect(cycle).toContain("For mode changes, follow 0E's **Mode change** instruction");
|
||||
expect(cycle).toContain('Return to the calling step with the saved answer; do not ask it again');
|
||||
expect(cycle).not.toContain('Save pending rows before comparing options');
|
||||
expect(cycle).toContain('complete current plan, pending rows and comparisons');
|
||||
@@ -125,8 +126,10 @@ test('CEO handoff carries all answered rows instead of one synthetic approach',
|
||||
expect(handoff).toContain('Auto-decided review mode → <selected mode> (your preference)');
|
||||
expect(handoff).toContain('Mode: <selected mode>; approved decisions: <rows or none>');
|
||||
expect(handoff).not.toContain('<approved 0D approach>');
|
||||
expect(section).toContain('Step 0E mode-handoff format and the current ledger dispositions');
|
||||
expect(section).toContain('including actual later scope-answer references');
|
||||
expect(compactProse(section)).toContain('each governing row\'s ID, disposition and answer reference');
|
||||
expect(compactProse(section)).toContain('including scope decisions after 0E');
|
||||
expect(compactProse(section)).toContain('This is a scope update, not another mode handoff');
|
||||
expect(section).not.toContain('using the Step 0E mode-handoff format');
|
||||
expect(section).toContain('do not ask or log the mode again');
|
||||
});
|
||||
|
||||
@@ -228,7 +231,7 @@ test('CEO defines pending choices and storage before its first decision procedur
|
||||
expect(persistenceStages.every(position => position >= 0)).toBe(true);
|
||||
expect(persistenceStages).toEqual([...persistenceStages].sort((a, b) => a - b));
|
||||
expect(source).toContain('## Reviewer Concerns\n- {unresolved spec-review issues with their owning input, or "None"}');
|
||||
expect(step0).toContain('0E estimates only files that will change');
|
||||
expect(step0).toContain('0E counts changed files, excluding unchanged reuse, to recommend a mode');
|
||||
expect(persistence).not.toContain('Save a chat-only plan');
|
||||
});
|
||||
|
||||
@@ -377,10 +380,10 @@ test('CEO value comparisons and decline-all outcomes stay explicit before approv
|
||||
expect(pendingSave >= 0 && pendingSave < values && values < comparedSave && comparedSave < ask).toBe(true);
|
||||
const comparison = procedure.slice(values, ask);
|
||||
expect(compactProse(comparison)).toContain('Show unchanged, shared and pending values');
|
||||
expect(compactProse(comparison)).toContain('Changes remain separate decisions even if they use the same framework');
|
||||
expect(comparison).toContain('Keep other rows fixed or pending');
|
||||
expect(compactProse(comparison)).toContain('preserve requirements, tests and fixes');
|
||||
expect(procedure).toContain('If all options are declined, continue only with a viable current approach retained by the answer');
|
||||
expect(compactProse(comparison)).toContain('Keep independent changes separate even within one framework');
|
||||
expect(comparison).toContain('other rows stay fixed or pending');
|
||||
expect(compactProse(comparison)).toContain('Preserve requirements, tests and fixes');
|
||||
expect(procedure).toContain('If all options are declined, continue only if the answer retains a viable current approach');
|
||||
expect(procedure).toContain('otherwise leave the row unresolved and stop for direction');
|
||||
});
|
||||
|
||||
@@ -409,10 +412,10 @@ test('CEO Step 0 drafts provisional contracts before menus and saves their compl
|
||||
expect(source).toContain('| ID and owner | Contract and evidence | Current | Proposed | Status | Exact approval and scope |');
|
||||
expect(source.indexOf('| ID and owner |')).toBeLessThan(source.indexOf('**2. Record the pending choice.**'));
|
||||
expect(approach).toContain('behavior, limits, test method and coverage');
|
||||
expect(approach).toContain('other rows fixed or pending');
|
||||
expect(approach).toContain('other rows stay fixed or pending');
|
||||
expect(approach).toContain('independent changes separate ledger rows');
|
||||
expect(approach.indexOf('Record owner, behavior, limits, test method and coverage in Current/Proposed')).toBeLessThan(approach.indexOf('Build one `currentDecision`'));
|
||||
expect(compactProse(approach)).toContain('Changes remain separate decisions even if they use the same framework');
|
||||
expect(compactProse(approach)).toContain('Keep independent changes separate even within one framework');
|
||||
expect(approach).toContain('A failed save stops the review');
|
||||
expect(approach).toContain('Record pending rows before comparisons; never prewrite approval or tasks');
|
||||
expect(approach).toContain('never prewrite approval or tasks');
|
||||
@@ -477,7 +480,7 @@ test('CEO outside findings reuse authority-first decisions without turning unkno
|
||||
expect(tension).toContain('Never silently trim or replace another candidate');
|
||||
expect(procedure).toContain('Record pending rows before comparisons; never prewrite approval or tasks');
|
||||
expect(skeleton).toContain('Stop with the cause; chat cannot replace a failed save');
|
||||
expect(compactProse(skeleton)).toContain('Honor user/host artifact and cleanup limits');
|
||||
expect(compactProse(skeleton)).toContain('Honor user/host write and cleanup limits');
|
||||
expect(procedure).toContain('Under the storage policy, save/present the complete current plan, pending rows and comparisons');
|
||||
expect(compactProse(procedure)).toContain('Ask one row per call with that object unchanged, without recomposing');
|
||||
expect(procedure).toContain('Save the answer reference and scope in Exact approval and scope');
|
||||
@@ -538,13 +541,13 @@ test('CEO expansion preparation feeds one pending list into the scope decisions'
|
||||
const source = compactProse(fs.readFileSync(`${SKELETON}.tmpl`, 'utf8'));
|
||||
const framing = source.split('### 0F.')[1]!.split('### 0G.')[0]!;
|
||||
const decisions = source.split('### 0G.')[1]!.split('### 0H.')[0]!;
|
||||
expect(framing).toContain('Prepare pending candidates for 0G');
|
||||
expect(framing).toContain('user experience, concrete addition, S/M/L/XL effort, risk and impact');
|
||||
expect(framing).toContain('The user decides each proposal in 0G');
|
||||
expect(framing).toContain('balance benefits and tradeoffs without unsupported promises in SELECTIVE EXPANSION');
|
||||
expect(framing).toContain('this label does not approve scope');
|
||||
expect(decisions).toContain("extend 0F's pending list with this analysis");
|
||||
expect(decisions).toContain('then resolve each proposal individually');
|
||||
expect(framing).toContain('Prepare 0G candidates');
|
||||
expect(framing).toContain('user experience, addition, S/M/L/XL effort, risk and impact');
|
||||
expect(framing).toContain('the user still decides each proposal in 0G');
|
||||
expect(framing).toContain('SELECTIVE EXPANSION balances benefits and tradeoffs without unsupported promises');
|
||||
expect(framing).toContain('Mark one option `(recommended)`; the user still decides each proposal in 0G');
|
||||
expect(decisions).toContain("extend 0F's pending list");
|
||||
expect(decisions).toContain('Run all four 0D steps for each unanswered addition or deferral, using its menu');
|
||||
expect(decisions).toContain('For both expansion modes, ask separately for each addition');
|
||||
});
|
||||
|
||||
@@ -605,7 +608,7 @@ test('CEO mode recommendation explains a plan-specific consequence without chang
|
||||
const recommendation = compactProse(mode.split('2. Recommend without selecting.')[1]!.split('3. Resolve that recommendation.')[0]!);
|
||||
expect(recommendation).toContain("In the Recommendation's `because` clause, connect a concrete plan fact or constraint");
|
||||
expect(recommendation).toContain("this mode's actual benefit or tradeoff");
|
||||
expect(recommendation).toContain('Count/category alone is not a reason');
|
||||
expect(recommendation).toContain('not just its count/category');
|
||||
expect(recommendation).toContain('For >15 planned changed files, recommend SCOPE REDUCTION');
|
||||
expect(recommendation).toContain('a new product/system (greenfield) → SCOPE EXPANSION');
|
||||
expect(recommendation).toContain('added capability → SELECTIVE EXPANSION');
|
||||
@@ -629,7 +632,7 @@ describe('CEO review decision boundaries contract', () => {
|
||||
const apply = compactProse(continuity.split('**Apply.**')[1]!);
|
||||
|
||||
test('every approach comparison preserves approvals and separates independent changes', () => {
|
||||
expect(compactProse(alternatives)).toContain('preserve requirements, tests and fixes');
|
||||
expect(compactProse(alternatives)).toContain('Preserve requirements, tests and fixes');
|
||||
expect(alternatives).toContain('behavior, limits, test method and coverage');
|
||||
expect(skeleton).toContain('Set review depth');
|
||||
const depth = compactProse(skeleton.split("**Set review depth from the user's request.**")[1]!.split('Plain terms:')[0]!);
|
||||
@@ -650,13 +653,13 @@ describe('CEO review decision boundaries contract', () => {
|
||||
expect(alternatives).toContain('Give independent changes separate ledger rows');
|
||||
expect(alternatives).toContain("Approved change with open test method/coverage | Decide once");
|
||||
expect(alternatives).toContain('every option preserves required behavior and approved tests');
|
||||
expect(compactProse(section)).toContain('complete 0D through its post-answer save, then continue to Apply below');
|
||||
expect(compactProse(section)).toContain('0D\'s Plan decision route through its post-answer save, then return here to Apply');
|
||||
expect(compactProse(section)).toContain("following 0D's test table");
|
||||
expect(alternatives).toContain('Code change and required regressions | Keep together; carry both forward once approved');
|
||||
expect(alternatives).toContain('Separate independently selectable additions. Tests for undecided behavior stay pending');
|
||||
expect(alternatives).toContain('Tests for existing behavior | Separate independently selectable additions');
|
||||
expect(alternatives).toContain('Tests for undecided behavior stay pending');
|
||||
expect(alternatives).toContain('other rows fixed or pending');
|
||||
expect(alternatives).toContain('other rows stay fixed or pending');
|
||||
expect(alternatives).toContain('reuse, verification coverage');
|
||||
expect(alternatives).not.toContain('for architecture choices');
|
||||
expect(alternatives.indexOf('Record owner, behavior, limits, test method and coverage in Current/Proposed')).toBeLessThan(alternatives.indexOf('Build one `currentDecision`'));
|
||||
@@ -683,7 +686,8 @@ describe('CEO review decision boundaries contract', () => {
|
||||
expect(approach).toContain('STOP for the actual answer, even for a lone option');
|
||||
expect(initial).toContain('With no required choice, or after those choices settle, go to 0E');
|
||||
expect(gate).toContain('Return to the calling step with the saved answer; do not ask it again');
|
||||
expect(approach).toContain('0D never restarts mode selection');
|
||||
expect(approach).toContain('0D returns to its caller, not to mode selection');
|
||||
expect(approach).toContain("For mode changes, follow 0E's **Mode change** instruction");
|
||||
expect(approach).toContain('A recommendation is not approval');
|
||||
expect(approach).not.toContain('Do NOT proceed to Step 0D or 0F until the user responds to 0C-bis');
|
||||
expect(compactProse(approach)).toContain('Ask one row per call with that object unchanged, without recomposing');
|
||||
@@ -720,14 +724,14 @@ describe('CEO review decision boundaries contract', () => {
|
||||
|
||||
test('an unresolved section decision is answered before its scoped plan amendment', () => {
|
||||
const steps = [
|
||||
'**Resolve.** If this section needs a new decision',
|
||||
'**Resolve.** Take the first applicable path for each finding',
|
||||
'Check the saved plan against each answer\'s exact scope',
|
||||
'Correct discrepancies under the storage policy',
|
||||
'Record findings and dispositions',
|
||||
].map(step => continuity.indexOf(step));
|
||||
expect(steps.every(position => position >= 0)).toBe(true);
|
||||
expect(steps).toEqual([...steps].sort((a, b) => a - b));
|
||||
expect(continuity).toContain('complete 0D through its post-answer save, then continue to Apply below');
|
||||
expect(continuity).toContain('0D\'s Plan decision route through its post-answer save, then return here to Apply');
|
||||
expect(fs.readFileSync(`${SKELETON}.tmpl`, 'utf8')).toContain('**STOP for the actual answer, even for a lone option.**');
|
||||
expect(compactProse(section)).toContain('Check input, source and actual approvals');
|
||||
expect(apply).toContain('against each answer\'s exact scope');
|
||||
@@ -777,9 +781,9 @@ describe('CEO review decision continuity contract', () => {
|
||||
// One governing checkpoint supplies the same actual-answer rule to all
|
||||
// eleven callers; settled findings do not manufacture another question.
|
||||
expect(continuity).toContain("At each section's **Decision gate**, follow Analyze → Resolve → Apply below");
|
||||
expect(continuity).toContain('complete 0D through its post-answer save, then continue to Apply below');
|
||||
expect(continuity).toContain('0D\'s Plan decision route through its post-answer save, then return here to Apply');
|
||||
expect(compactProse(compactProse(fs.readFileSync(`${SKELETON}.tmpl`, 'utf8')))).toContain('Ask one row per call with that object unchanged, without recomposing');
|
||||
expect(continuity).toContain('If all choices are settled, cite their exact answers and go straight to Apply');
|
||||
expect(continuity).toContain('An exact prior answer covers it: cite that answer and go to Apply');
|
||||
expect(continuity).toContain('Record findings and dispositions, then review the next section');
|
||||
expect(compactProse(template)).toContain('say "No issues found" only when there are zero findings');
|
||||
expect(continuity).toContain('Check the saved plan against each answer\'s exact scope');
|
||||
@@ -828,7 +832,7 @@ describe('CEO review decision continuity contract', () => {
|
||||
expect(earlyLedger).toContain('| ID and owner | Contract and evidence | Current | Proposed | Status | Exact approval and scope |');
|
||||
expect(compactProse(earlyLedger)).toContain('cite evidence, conventions and tests; mark unknowns');
|
||||
expect(compactProse(sources)).toContain('Reuse exact approvals');
|
||||
expect(continuity).toContain('complete 0D through its post-answer save, then continue to Apply below');
|
||||
expect(continuity).toContain('0D\'s Plan decision route through its post-answer save, then return here to Apply');
|
||||
for (const requirement of ['with each row\'s owner section',
|
||||
'Check the saved plan against each answer\'s exact scope',
|
||||
'an approval is not proof of implementation or verification',
|
||||
@@ -840,11 +844,11 @@ describe('CEO review decision continuity contract', () => {
|
||||
|
||||
test('ownership never defers a critical risk or merges distinct choices by topic', () => {
|
||||
const skeleton = fs.readFileSync(`${SKELETON}.tmpl`, 'utf8');
|
||||
expect(continuity).toContain('If this section needs a new decision or evidence warrants reopening one');
|
||||
expect(continuity).toContain('This section needs a new choice, or evidence warrants reopening its prior answer');
|
||||
expect(skeleton).toContain('Give independent changes separate ledger rows');
|
||||
expect(compactProse(compactProse(skeleton))).toContain('Changes remain separate decisions even if they use the same framework');
|
||||
expect(compactProse(compactProse(skeleton))).toContain('Keep independent changes separate even within one framework');
|
||||
for (const requirement of ['Resolve critical risks now',
|
||||
'complete 0D through its post-answer save, then continue to Apply below',
|
||||
'0D\'s Plan decision route through its post-answer save, then return here to Apply',
|
||||
'Keep independent safety fixes and throughput improvements in separate rows']) expect(continuity).toContain(requirement);
|
||||
const testReview = compactProse(template.split('### Section 6: Test Review')[1]!.split('### Section 7:')[0]!);
|
||||
expect(testReview).toContain('Carry requested or approved coverage forward, including directly determined tests, without re-asking');
|
||||
@@ -934,10 +938,10 @@ test('CEO closing route checks approvals before outputs and verifies artifacts b
|
||||
expect(governingStages.every(position => position >= 0)).toBe(true);
|
||||
expect(governingStages).toEqual([...governingStages].sort((a, b) => a - b));
|
||||
const questions = section.split('## CRITICAL RULE — How to ask questions')[1]!.split('## Mode Quick Reference')[0]!;
|
||||
expect(questions).toContain('`D<N>` question heading and A/B/C option labels');
|
||||
expect(questions).toContain('Use `D<N>` and A/B/C labels');
|
||||
expect(questions).toContain('Cite the stable ledger ID separately');
|
||||
const formatting = questions.split('## Formatting Rules')[1]!;
|
||||
expect(formatting).toContain("Step 0D's exact `currentDecision` fields for the question and option descriptions");
|
||||
expect(formatting).toContain("0D's exact `currentDecision` question and option descriptions");
|
||||
expect(formatting).not.toContain("put the complete comparison in the question's brief");
|
||||
expect(questions).not.toMatch(/NUMBER \+ (?:option )?LETTER|"3A"|One sentence max per option/);
|
||||
});
|
||||
@@ -1089,7 +1093,7 @@ describe('plan-ceo-review carve — static ordering', () => {
|
||||
const procedure = document.split('### Working review decisions')[1]!.split('### Section 1:')[0]!.replace(/\s+/g, ' ');
|
||||
expect(document.indexOf('### Working review decisions')).toBeLessThan(document.indexOf('### Section 6: Test Review'));
|
||||
expect(procedure).toContain("At each section's **Decision gate**, follow Analyze → Resolve → Apply below");
|
||||
expect(procedure).toContain('complete 0D through its post-answer save, then continue to Apply below');
|
||||
expect(procedure).toContain('0D\'s Plan decision route through its post-answer save, then return here to Apply');
|
||||
expect(compactProse(compactProse(fs.readFileSync(`${SKELETON}.tmpl`, 'utf8')))).toContain('Ask one row per call with that object unchanged, without recomposing');
|
||||
expect(fs.readFileSync(`${SKELETON}.tmpl`, 'utf8')).toContain('**STOP for the actual answer, even for a lone option.**');
|
||||
expect(procedure).toContain('Check the saved plan against each answer\'s exact scope');
|
||||
@@ -1340,18 +1344,18 @@ describe('CEO complete question persistence before dispatch', () => {
|
||||
for (const field of ['| `question` | Full brief:', '| `header` and option labels | Final native text within host limits', 'offer 2–3 options', '1–2 sentence summary', 'S/M/L/XL effort', 'low/medium/high risk', 'reuse, verification coverage', 'options differ in kind, not coverage — no completeness score']) expect(built).toContain(field);
|
||||
const save = cycle.slice(cycle.indexOf('**Pre-question checkpoint:**'), cycle.indexOf('**4. Ask'));
|
||||
expect(save).toContain('Copy the grid and all exact fields below');
|
||||
expect(save).toContain('without the illustrative fence delimiters');
|
||||
expect(save).toContain('omit the fence delimiters');
|
||||
for (const field of ['## currentDecision (ROW-ID)', 'Commitment comparison: <complete grid>',
|
||||
'Question: <complete currentDecision.question>', 'Header: <exact currentDecision.header>',
|
||||
'<full first option description>', '<full second option description; repeat for all offered options>']) expect(save).toContain(field);
|
||||
expect(save).toContain('A grid, summary or pointer is insufficient');
|
||||
expect(save).toContain('not a grid, summary or pointer');
|
||||
expect(compactProse(save)).toContain('Verify IDs and fields against `currentDecision`, citations against source');
|
||||
// Native 043a questions were recomposed after title-only saves; the final
|
||||
// Edit ACK also said a Read was unnecessary. These are workflow guards,
|
||||
// not proof that a model followed the instructions.
|
||||
expect(built).toContain('Project, ELI10, Stakes, Recommendation and applicable completeness/net text');
|
||||
expect(compactProse(save)).toContain('Replace the whole payload on revision');
|
||||
expect(compactProse(save)).toContain('Keep answered decisions and their answers under separate headings');
|
||||
expect(compactProse(save)).toContain('keep answered decisions under separate headings');
|
||||
expect(compactProse(save)).toContain('Read despite Edit\'s current-in-context hint');
|
||||
expect(compactProse(save)).toContain('Correct mismatches, save and Read again before dispatch');
|
||||
const dispatch = cycle.slice(cycle.indexOf('**4. Ask'), cycle.indexOf('**STOP for the actual answer'));
|
||||
|
||||
Reference in new issue
Block a user