mirror of
https://github.com/garrytan/gstack.git
synced 2026-05-01 19:25:10 +02:00
c6c3294ee9
Three root causes fixed: - QA agent killed shared test server (kill port), breaking subsequent tests - Shared outcomeDir caused cross-contamination (b8 read b7's report) - max_false_positives=2 too strict for thorough QA agents finding derivative bugs Changes: - Restart test server in planted-bug beforeAll (resilient to agent kill) - Each planted-bug test gets isolated working directory (no cross-contamination) - max_false_positives 2→5 in all ground truth files - Accept error_max_turns for /qa quick (thorough QA is not failure) - "Write early, update later" prompt pattern ensures reports always exist - maxTurns 30→40, timeout 240s→300s for planted-bug evals Result: 10/10 E2E pass, 9/9 LLM judge pass. All three planted-bug evals score 5/5 detection with evidence quality 5. Total E2E cost: $1.69. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
44 lines
1.6 KiB
JSON
44 lines
1.6 KiB
JSON
{
|
|
"fixture": "qa-eval-checkout.html",
|
|
"bugs": [
|
|
{
|
|
"id": "broken-email-regex",
|
|
"category": "functional",
|
|
"severity": "high",
|
|
"description": "Email validation accepts 'user@' as valid — regex pattern [^@]+@[^@] is missing domain requirement",
|
|
"detection_hint": "email|regex|validation|accepts|invalid|user@|pattern"
|
|
},
|
|
{
|
|
"id": "nan-total",
|
|
"category": "functional",
|
|
"severity": "high",
|
|
"description": "Clearing the quantity field shows 'Total: $NaN' — parseInt on empty string returns NaN with no fallback",
|
|
"detection_hint": "NaN|total|quantity|empty|price|calculation|clear"
|
|
},
|
|
{
|
|
"id": "cc-field-overflow",
|
|
"category": "visual",
|
|
"severity": "medium",
|
|
"description": "Credit card input has no maxlength attribute — entering >20 characters causes text to overflow the container",
|
|
"detection_hint": "credit card|maxlength|overflow|cc|input|long|container"
|
|
},
|
|
{
|
|
"id": "missing-required-zip",
|
|
"category": "functional",
|
|
"severity": "medium",
|
|
"description": "Zip code field has no 'required' attribute — form can be submitted without a zip code",
|
|
"detection_hint": "zip|required|missing|form|submit|shipping|postal"
|
|
},
|
|
{
|
|
"id": "stripe-not-defined",
|
|
"category": "console",
|
|
"severity": "high",
|
|
"description": "Form submit triggers 'Uncaught ReferenceError: stripe is not defined' — payment SDK not loaded",
|
|
"detection_hint": "stripe|ReferenceError|not defined|console|error|submit|payment"
|
|
}
|
|
],
|
|
"total_bugs": 5,
|
|
"minimum_detection": 2,
|
|
"max_false_positives": 5
|
|
}
|