v1.90.0.0 feat: make browser cookie imports explicit and safe (#2964)

* fix(browse): prepare reliable cookie import wave for validation

* ci: sequence quality and behavior for validation branch

* fix(browse): isolate Windows qualification and preserve native diagnostics

* test(browse): cover cookie workflow quality and isolate Windows user paths

* test(browse): trace native member startup and initialize fresh folders

* fix(browse): keep Windows member stdin alive through EOF

* fix(browse): latch native timeouts and compare contained Edge startup

* test(browse): verify native version metadata and actual Windows argv

* test(browse): qualify Dia import on isolated macOS CI

* fix(browse): require picker origin for session mutations

* fix(browse): bound credential reads through stream completion

* test(browse): inspect owned Windows process arguments natively

* test(evals): preserve passing coverage during cookie repair reruns

* test(browse): isolate Dia qualification in a fresh macOS account

* test(browse): pass bounded integer timeouts to native Mac probes

* test(browse): distinguish Windows profile initialization from containment

* test(browse): await descendant pipe readiness before parent exit

* test(browse): initialize and restore isolated macOS Keychain state

* test(browse): initialize Windows fixture folders before qualification

* test(ci): pin the same Node runtime across Windows checks

* test(browse): distinguish native macOS browser preflight stages

* test(browse): isolate Windows descendant console lifetime

* test(browse): preserve native receipts and identify fixture lock holders

* test(browse): prepare dependency resolution before native Mac worker startup

* test(ci): include lock and close checks in native diagnostics

* test(browse): preserve native owner probe stages and subprocess deadlines

* fix(browse): classify Chromium profile-in-use exit precisely

* test(browse): retain Mac qualification evidence through cleanup failures

* test(browse): bound Mac fixture paths and retire its owned user domain

* test(browse): accept vanished fixture entries without weakening cleanup

* test(browse): identify probe-created macOS user domains safely

* test(browse): observe Mac user domains without targeting them first

* test(browse): use passive fresh-user ownership throughout Mac qualification

* test(browse): distinguish profile and registered-home Keychain lookups

* test(browse): qualify Dia under one registered account home

* test(browse): identify Dia startup and owned process-group failures

* test(browse): classify bounded Dia startup diagnostics without leaking output

* fix(test): preserve native Mac sandboxing and reap owned browser children

* fix(browse): preserve Chromium sandboxing for native profile imports

* test(browse): inspect signed Mach-O architecture without launching Xcode tools

* test(browse): sample pending Dia startup and reap on all cleanup paths

* test(browse): compare protected Dia launches in fresh Bun and Node accounts

* test(browse): inspect isolated Mac GUI readiness without browser access

* v1.90.0.0 fix: bind cookie picker actions to their document

* test: validate cookie guards and fit nested launch fixtures

* ci: configure the bundled Chromium sandbox helper

* fix(browse): classify Playwright authentication timeouts

* test: retain bounded Windows lifecycle diagnostics

* test(cso): reuse bounded NTFS precision candidates

* test(review): handle explicit preservation choices safely

* test(browse): remove owned fixture directories with explicit primitives

* test(review): distinguish descriptive reuse from edit commitments

* test: admit only the approved unscored cookie workflow refusal

* test: keep the Office Hours judge mock export-complete

* fix: keep dependency-free CI planners independent of the model SDK

* test: observe the exact holder after a native fixture unlink failure

* fix: start seeded PTY observations at owned readiness

* test: acquire identity-bound Windows deletion admission before profile resets

* test: preserve qualified Git index bits without authorizing mutations
This commit is contained in:
Garry Tan
2026-09-25 12:06:45 -04:00
committed by GitHub
parent 730a1017d1
commit a84b0b5b6d
111 changed files with 14996 additions and 1057 deletions
+80 -10
View File
@@ -8,6 +8,11 @@ on:
description: 'Run ALL gate tests in the sliced lane (bypass diff selection; also arms the hollow-shard guard)'
type: boolean
default: true
validation_phase:
description: 'Validation branch phase; run quality before behavior on unchanged inputs'
type: choice
options: [all, quality, cookie-quality, behavior, cookie-behavior]
default: all
concurrency:
group: evals-${{ github.event.pull_request.number || github.run_id }}
@@ -133,10 +138,37 @@ jobs:
bun-version: 1.4.0
- name: Emit run manifest
if: github.event_name != 'workflow_dispatch' || inputs.validation_phase == 'all'
env:
EVALS_ALL: ${{ (github.event_name == 'workflow_dispatch' && inputs.evals_all) && '1' || '' }}
run: EVALS_TIER=gate bun --no-install run scripts/test-paid-shards.ts --tier gate --emit-plan /tmp/paid-plan/manifest.json --slices 6
- name: Emit validation-phase manifest
if: github.event_name == 'workflow_dispatch' && inputs.validation_phase != 'all'
env:
VALIDATION_PHASE: ${{ inputs.validation_phase }}
EVALS_ALL: ${{ inputs.evals_all && '1' || '' }}
EVALS_TIER: gate
run: |
bun --no-install -e '
import { mkdirSync, writeFileSync } from "node:fs";
import { buildRunManifest, collectPaidTestFiles } from "./scripts/test-paid-shards.ts";
const phase = process.env.VALIDATION_PHASE;
if (!["quality", "cookie-quality", "behavior", "cookie-behavior"].includes(phase)) throw new Error("Invalid validation phase");
const cookieBehavior = phase === "cookie-behavior";
const discovered = phase === "cookie-quality" ? ["test/skill-llm-eval.test.ts"]
: cookieBehavior ? ["test/skill-e2e-bws.test.ts", "test/skill-e2e-qa-workflow.test.ts", "test/skill-e2e-design.test.ts", "test/skill-e2e-diagram.test.ts", "test/skill-e2e-deploy.test.ts"]
: collectPaidTestFiles().filter(file => file.startsWith("test/skill-llm-eval") === (phase === "quality"));
const manifest = buildRunManifest({ tier: "gate", profile: "full", sliceCount: 6, evalsAll: !cookieBehavior && process.env.EVALS_ALL === "1", discovered,
...(cookieBehavior ? { changedFiles: ["browse/src/cookie-picker-routes.ts", "browse/src/cookie-import-browser.ts", "browse/src/bun-polyfill.cjs"], env: { ...process.env, EVALS_ALL: "" } } : {}) });
if (phase === "cookie-quality") manifest.selection = { e2e: [], judges: ["setup-browser-cookies/SKILL.md workflow"] };
if (cookieBehavior) manifest.selection = { e2e: ["browse-basic", "browse-snapshot", "qa-quick", "qa-only-no-fix", "design-review-detector-shim-dom", "diagram-triplet", "canary-workflow", "benchmark-workflow"], judges: [] };
manifest.selectionReason = phase + " validation subset; " + manifest.selectionReason;
mkdirSync("/tmp/paid-plan", { recursive: true });
writeFileSync("/tmp/paid-plan/manifest.json", JSON.stringify(manifest, null, 2) + "\n");
console.log(phase + ": " + manifest.entries.filter(entry => entry.status === "planned").length + " planned shards");
'
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: paid-plan
@@ -152,9 +184,9 @@ jobs:
# 40-way per row queued claude session STARTUP behind 39 siblings and ate
# per-test budgets — the documented timeout-flake family). Tune with
# parity data before raising.
# The complete gate census needs at most 197 minutes per slice; keep
# The complete gate census needs at most 201 minutes per slice; keep
# 20 minutes for setup/upload without preempting configured retries.
timeout-minutes: 220
timeout-minutes: 221
permissions:
contents: read
packages: read
@@ -334,7 +366,9 @@ jobs:
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: report-verdict
path: /tmp/report.txt
path: |
/tmp/report.txt
/tmp/paid-report/collector-outcomes.json
if-no-files-found: ignore
retention-days: 30
@@ -374,7 +408,8 @@ jobs:
path: /tmp/verdict
continue-on-error: true
# Sourced from the slice artifacts' eval-store JSONs. Keeps the
# Verified counts come from the read-only report job, not repo code in
# this write-token job. Keeps the
# "## E2E Evals" marker so the upsert keeps updating the same comment.
# Runs even when reconciliation failed — a red lane on the PR is the point.
- name: Post PR comment
@@ -384,8 +419,36 @@ jobs:
run: |
# shellcheck disable=SC2086,SC2059
RESULTS=$(find /tmp/paid-report -name '*.json' ! -name 'manifest.json' ! -name 'slice-*.json' ! -name '_partial*' 2>/dev/null | sort)
TOTAL=0; PASSED=0; FAILED=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0"
TOTAL=0; PASSED=0; FAILED=0; MANUAL=0; FLAKY=0; EXECUTED=0; REUSED=0; COST="0"
SUITE_LINES=""
VERIFIED=/tmp/verdict/paid-report/collector-outcomes.json
if ! jq -e '
. as $summary |
.version == 1 and (.files | type == "array") and (.totals | type == "object") and
([.files[] | .total == (.passed + .failed + .manual_accepted) and
(.total == (.executed + .reused)) and
([.total,.passed,.failed,.manual_accepted,.executed,.reused,.attempts,.flaky] | all(. >= 0 and (floor == .))) ] | all) and
(.totals | .total == (.passed + .failed + .manual_accepted) and .total == (.executed + .reused)) and
(["total","passed","failed","manual_accepted","executed","reused","attempts","flaky"] |
all(. as $key | ([$summary.files[] | .[$key]] | add // 0) == $summary.totals[$key]))
' "$VERIFIED" >/dev/null 2>&1; then
VERIFIED=""
echo 'Verified collector summary unavailable; manual acceptance is unavailable/unverified.'
fi
if [ -n "$VERIFIED" ]; then
while IFS=$'\t' read -r f T P F M FL EX RE _ATTEMPTS C TIER SHARD; do
[ "$T" -eq 0 ] && continue
TOTAL=$((TOTAL + T)); PASSED=$((PASSED + P)); FAILED=$((FAILED + F))
MANUAL=$((MANUAL + M)); FLAKY=$((FLAKY + FL))
EXECUTED=$((EXECUTED + EX)); REUSED=$((REUSED + RE))
COST=$(echo "$COST + $C" | bc)
STATUS_ICON="✅"
[ "$M" -gt 0 ] && STATUS_ICON="⚠ manual/unscored"
[ "$F" -gt 0 ] && STATUS_ICON="❌"
[ "$F" -eq 0 ] && [ "$M" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${M} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
done < <(jq -r '.files[] | [.file,.total,.passed,.failed,.manual_accepted,.flaky,.executed,.reused,.attempts,.cost,.tier,.shard] | @tsv' "$VERIFIED")
else
for f in $RESULTS; do
if ! jq -e '.total_tests' "$f" >/dev/null 2>&1; then
echo "Skipping malformed JSON: $f"
@@ -418,22 +481,25 @@ jobs:
STATUS_ICON="✅"
[ "$F" -gt 0 ] && STATUS_ICON="❌"
[ "$F" -eq 0 ] && [ "$FL" -gt 0 ] && STATUS_ICON="✅⚠"
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
SUITE_LINES="${SUITE_LINES}| ${TIER}/${SHARD} | ${P}/${T} | unverified | ${EX} | ${RE} | ${STATUS_ICON} | \$${C} |\n"
done
fi
COVERAGE=$(jq -r '"Profile: \(.profile // "full") / \(.prCoverage.mode // "broad"); selected behaviors: \(.selection.e2e | if . == null then "all" else length end), judges: \(.selection.judges | if . == null then "all" else length end). Deferred to scheduled/release coverage: \(.prCoverage.deferred // [] | length) behaviors and \(.prCoverage.deferredPromptFiles // [] | length) changed prompt files. Deferred checks did not run and receive no PR-pass credit."' /tmp/paid-report/manifest.json) || COVERAGE='Coverage manifest unavailable; no coverage claim.'
STATUS="✅ PASS"
if [ "${RECONCILE_EXIT:-1}" != "0" ] || [ "$FAILED" -gt 0 ]; then STATUS="❌ FAIL"; fi
if [ "$STATUS" = '✅ PASS' ] && [ "$MANUAL" -gt 0 ]; then STATUS='⚠ MANUAL ACCEPTED (unscored)'; fi
if [ -z "$VERIFIED" ]; then STATUS='❌ FAIL (manual acceptance unavailable/unverified)'; fi
BODY="## E2E Evals: ${STATUS}
**${PASSED}/${TOTAL}** recorded final results passed | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY")
**${PASSED} automated passed / ${TOTAL} final results** | **${FAILED} failed, ${MANUAL} manual accepted (unscored; no score-cache credit)** | **${EXECUTED} executed, ${REUSED} reused** | **\$${COST}** total cost | reconcile exit: ${RECONCILE_EXIT:-missing}$([ "$FLAKY" -gt 0 ] && printf ' | ⚠ %s cases with multiple attempts' "$FLAKY")
${COVERAGE}
| Shard | Result | Executed | Reused | Status | Cost |
|-------|--------|----------|--------|--------|------|
| Shard | Automated result | Manual/unscored | Executed | Reused | Status | Cost |
|-------|------------------|-----------------|----------|--------|--------|------|
$(echo -e "$SUITE_LINES")
<details><summary>Fail-closed reconciliation</summary>
@@ -450,7 +516,11 @@ jobs:
FAILURES=""
for f in $RESULTS; do
if ! jq -e '.failed' "$f" >/dev/null 2>&1; then continue; fi
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
if [ -n "$VERIFIED" ]; then
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false and (has("manual_review") | not))][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
else
FAILS=$(jq -r '[.tests | group_by(.name)[] | last | select(.passed == false)][] | "- ❌ \(.name): \(.exit_reason // "unknown")"' "$f" 2>/dev/null || echo "- ⚠️ parse error")
fi
FAILURES="${FAILURES}${FAILS}\n"
done
BODY="${BODY}
+15
View File
@@ -182,6 +182,21 @@ jobs:
- name: Install Playwright Chromium
run: npx playwright install --with-deps chromium
- name: Configure the bundled Chromium sandbox helper
run: |
set -euo pipefail
chrome=$(bun -e 'import { chromium } from "playwright"; import { realpathSync } from "node:fs"; console.log(realpathSync(chromium.executablePath()))')
case "$chrome" in
"$HOME"/.cache/ms-playwright/chromium-*/chrome-linux*/chrome) ;;
*) echo "Unexpected Chromium installation path" >&2; exit 1 ;;
esac
helper="${chrome%/*}/chrome_sandbox"
installed="${chrome%/*}/chrome-sandbox"
test -f "$helper" && test ! -L "$helper"
sudo install -T -o root -g root -m 4755 "$helper" "$installed"
test "$(stat -c '%u:%a' "$installed")" = '0:4755'
cmp -s "$helper" "$installed"
# Headed-browser tests (handoff, extension sidepanel DOM) need a real
# DISPLAY — first Linux run failed with Playwright's "launched a headed
# browser without an XServer" banner. xvfb-run below provides it;
+129 -2
View File
@@ -26,6 +26,23 @@ on:
pull_request:
branches: [main]
workflow_dispatch:
inputs:
dia_native_only:
description: Run disposable ARM64 macOS Dia qualification instead of Windows
type: boolean
default: false
native_diagnostics_only:
description: Run Windows launch diagnostics and credential regressions without qualification
type: boolean
default: false
dia_launch_comparison:
description: Compare protected Dia launch under Bun and Node in separate fresh Mac jobs
type: boolean
default: false
dia_gui_readiness:
description: Inspect disposable Mac GUI-session readiness without launching browsers
type: boolean
default: false
concurrency:
group: windows-free-${{ github.event.pull_request.number || github.run_id }}
@@ -37,6 +54,7 @@ permissions:
jobs:
windows-free-tests:
if: ${{ !inputs.dia_native_only && !inputs.dia_launch_comparison && !inputs.dia_gui_readiness }}
# Ubicloud Windows runner (same provider as the Linux evals workflow).
# To revert: swap to `windows-latest` (GitHub's free 4-core Windows runner).
runs-on: windows-latest
@@ -49,6 +67,10 @@ jobs:
with:
bun-version: 1.4.0
- uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38
with:
node-version: 24.18.0
# bun install was 35s of a 55s job, all network. Cache keyed on the
# lockfile; bun's install cache lives under ~/.bun/install/cache on
# every platform.
@@ -82,6 +104,7 @@ jobs:
shell: bash
- name: Generate host SKILL.md outputs (.agents, .factory)
if: ${{ !inputs.native_diagnostics_only }}
# The golden-file regression tests in test/gen-skill-docs.test.ts read
# .agents/skills/gstack-ship/SKILL.md and .factory/skills/gstack-ship/
# SKILL.md. Both are gitignored — generated on demand by gen:skill-docs.
@@ -91,6 +114,9 @@ jobs:
run: bun run gen:skill-docs --host all
shell: bash
- name: Install Chromium for the Node worker smoke
run: bunx playwright install chromium
# The Windows job verifies the new portability work this PR delivers,
# not the entire free suite. After v1.20.0.0 ships, full-suite Windows
# parity is a P4 follow-up TODO that depends on porting many tests off
@@ -110,6 +136,7 @@ jobs:
# (test/test-free-shards.test.ts)
- name: Run curated Windows-safe suite
if: ${{ !inputs.native_diagnostics_only }}
# Replaces the previous hand-listed 13-file subset, which drifted from
# the curation registry it was supposed to sample. The runner's
# --windows-only curation (scripts/test-free-shards.ts) is the single
@@ -125,15 +152,115 @@ jobs:
run: bun run test:windows
shell: bash
- name: Run focused native launch and credential diagnostics
if: inputs.native_diagnostics_only
shell: bash
run: |
set -o pipefail
status=0
bun test browse/test/cookie-import-native-job.test.ts --test-name-pattern 'native Windows launch diagnostics|a locked real Edge profile|real Edge synthetic profile' 2>&1 | tee "$RUNNER_TEMP/gstack-free-test-native-diagnostics.log" || status=1
bun test browse/test/cookie-credential-deadline.test.ts browse/test/cookie-import-node.test.ts browse/test/bun-polyfill.test.ts 2>&1 | tee "$RUNNER_TEMP/gstack-free-test-credential-diagnostics.log" || status=1
exit "$status"
# Same diagnosability contract as free-tests.yml: a red lane must
# carry the WHY (the runner's quiet console names files, not causes).
# (#2561 was written against the old hand-listed subset; its two new
# test files are pure-TS and flow into the --windows-only curation
# automatically, so no per-file entry is needed here.)
- name: Upload shard logs on failure
if: failure()
- name: Upload full shard logs
if: always()
uses: actions/upload-artifact@v7
with:
name: windows-free-test-shard-logs
path: ${{ runner.temp }}/gstack-free-test-*.log
if-no-files-found: ignore
cookie-native-qualification:
if: github.event_name == 'workflow_dispatch' && !inputs.dia_native_only && !inputs.native_diagnostics_only && !inputs.dia_launch_comparison && !inputs.dia_gui_readiness
runs-on: windows-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1
with:
persist-credentials: false
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6
with:
bun-version: 1.4.0
- uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38
with:
node-version: 24.18.0
- name: Install pinned dependencies
run: bun install --frozen-lockfile
- name: Build the qualified Node server inputs
run: bash browse/scripts/build-node-server.sh
shell: bash
- name: Qualify owned native cookie extraction
run: ./.github/scripts/run-cookie-native-qualification.ps1 -OutputRoot "$env:RUNNER_TEMP"
- name: Preserve qualification evidence
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a
with:
name: cookie-native-qualification
path: ${{ runner.temp }}/cookie-native-qualification-*/
if-no-files-found: error
dia-native-qualification:
if: github.event_name == 'workflow_dispatch' && (inputs.dia_native_only || inputs.dia_launch_comparison || inputs.dia_gui_readiness)
runs-on: macos-15
timeout-minutes: 20
strategy:
fail-fast: false
matrix:
runtime: ${{ fromJSON(inputs.dia_launch_comparison && !inputs.dia_gui_readiness && '["bun","node"]' || '["bun"]') }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1
with:
persist-credentials: false
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6
with:
bun-version: 1.4.0
- name: Validate GUI readiness selection
if: inputs.dia_gui_readiness
env:
OTHER_DIA_MODES: ${{ inputs.dia_native_only || inputs.dia_launch_comparison || inputs.native_diagnostics_only }}
run: |
bun --no-env-file --no-install --no-macros --config=/dev/null -e '
if (process.env.OTHER_DIA_MODES !== "false") {
console.error("dia_gui_readiness must be selected alone");
process.exit(1);
}
'
- uses: actions/setup-node@249970729cb0ef3589644e2896645e5dc5ba9c38
if: inputs.dia_launch_comparison && !inputs.dia_gui_readiness
with:
node-version: 24.18.0
architecture: arm64
- name: Install pinned dependencies
if: ${{ !inputs.dia_gui_readiness }}
run: bun install --frozen-lockfile
- name: Install the synthetic destination browser
if: ${{ !inputs.dia_gui_readiness }}
run: bunx --no-install playwright install chromium
- name: Inspect GUI readiness without browser or Keychain access
if: inputs.dia_gui_readiness
env:
GSTACK_DIA_NATIVE_QUALIFY: '1'
run: bun --no-env-file --no-install --no-macros --config=/dev/null .github/scripts/run-dia-native-qualification.ts --gui-readiness-only
- name: Qualify native Dia discovery, decryption, and import
if: ${{ !inputs.dia_launch_comparison && !inputs.dia_gui_readiness }}
env:
GSTACK_DIA_NATIVE_QUALIFY: '1'
run: bun --no-env-file --no-install --no-macros --config=/dev/null .github/scripts/run-dia-native-qualification.ts
- name: Compare protected native Dia launch without qualification credit
if: inputs.dia_launch_comparison && !inputs.dia_gui_readiness
env:
GSTACK_DIA_NATIVE_QUALIFY: '1'
COMPARISON_RUNTIME: ${{ matrix.runtime }}
run: bun --no-env-file --no-install --no-macros --config=/dev/null .github/scripts/run-dia-native-qualification.ts --launch-comparison "$COMPARISON_RUNTIME"
- name: Preserve only the sanitized qualification receipt
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a
with:
name: ${{ inputs.dia_gui_readiness && 'dia-gui-readiness' || inputs.dia_launch_comparison && format('dia-launch-comparison-{0}', matrix.runtime) || 'dia-native-qualification' }}
path: ${{ runner.temp }}/dia-native-qualification.json
if-no-files-found: error