mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-11 23:49:01 +02:00
* fix: default cross-model workflows to frontier models * chore: bump version and changelog (v1.82.1.0) Co-Authored-By: OpenAI Codex <noreply@openai.com> * fix: repair frontier eval budgets and workflow instructions Preserve frontier models and quality thresholds while fixing truncated judge output, ordered section expansion, consent checks, QA scoring, and ship audit gates. Add regression coverage and refresh generated docs. Co-Authored-By: OpenAI Codex <noreply@openai.com> * fix: resolve workflow gaps exposed by frontier evals Clarify plan-review ordering and fallback modes, preserve deploy readiness gates, honor configured merge methods, correct benchmark and canary contracts, and restore vendored installs on setup failure. Cover recovery with real-shell regressions. Co-Authored-By: OpenAI Codex <noreply@openai.com> * fix: use agent capture budgets for deploy evals Multi-turn deploy and benchmark sessions were incorrectly limited to the single-call judge timeout. Use the existing capture tier and leave outer-test cleanup headroom, with a free policy regression test. Keep all behavioral assertions and frontier models unchanged. Co-Authored-By: OpenAI Codex <noreply@openai.com> * fix: clarify retro workflow and evaluate compare instructions Include compare mode in the frontier judge excerpt, define metric sources and snapshot ordering, and preserve the existing prompt-size budget. Co-authored-by: OpenAI Codex <noreply@openai.com> * fix: make documentation release review and publication consistent Review before commit, clarify changelog safeguards and unavailable reviewer modes, and preserve raw PR bodies across separate shell calls. Keep title sync in one shell and add regression coverage. Co-authored-by: OpenAI Codex <noreply@openai.com> --------- Co-authored-by: OpenAI Codex <noreply@openai.com>
233 lines
8.9 KiB
TypeScript
233 lines
8.9 KiB
TypeScript
/**
|
|
* _gstack_codex_model_probe — round-trip model readiness (#2477).
|
|
*
|
|
* The auth probe accepts "auth exists" as readiness, but a ChatGPT account
|
|
* with a model it cannot use passes auth and then dies with an HTTP 400 on
|
|
* every invocation. The model probe does one short `codex exec "reply OK"`
|
|
* round trip with gstack's selected model.
|
|
*
|
|
* Contract pinned here (all runs use a STUBBED codex binary):
|
|
* - exit 0 -> MODEL_OK, result cached (1h TTL + config/auth
|
|
* mtime signature), second call does NOT re-invoke
|
|
* - model 400 output -> MODEL_UNUSABLE (exit 1) + config.toml HINT lines,
|
|
* negative-cached 15 min (same exit-1 + hints from
|
|
* cache; re-probing every preflight charged the
|
|
* affected user 30s + real tokens per section)
|
|
* - transient failure -> MODEL_PROBE_INCONCLUSIVE, FAIL-OPEN (exit 0),
|
|
* never cached
|
|
* - config.toml mtime change invalidates a cached MODEL_OK and a cached
|
|
* MODEL_UNUSABLE (editing the pin IS the fix)
|
|
*/
|
|
import { describe, test, expect } from 'bun:test';
|
|
import { spawnSync } from 'child_process';
|
|
import * as fs from 'fs';
|
|
import * as os from 'os';
|
|
import * as path from 'path';
|
|
|
|
const ROOT = path.resolve(import.meta.dir, '..');
|
|
const PROBE = path.join(ROOT, 'bin', 'gstack-codex-probe');
|
|
|
|
const STUB = `#!/usr/bin/env bash
|
|
echo "invoked" >> "$STUB_LOG"
|
|
printf '%s\\n' "$*" >> "$STUB_ARGS_LOG"
|
|
case "\${STUB_MODE:-ok}" in
|
|
ok) echo "OK"; exit 0 ;;
|
|
model400)
|
|
echo 'warning: Model metadata for \`gpt-6-astra\` not found.' >&2
|
|
echo 'ERROR: {"type":"error","status":400,"error":{"type":"invalid_request_error","message":"The '"'"'gpt-6-astra'"'"' model is not supported when using Codex with a ChatGPT account."}}' >&2
|
|
exit 1 ;;
|
|
transient) echo "stream error: network unreachable" >&2; exit 7 ;;
|
|
esac
|
|
`;
|
|
|
|
interface Fixture {
|
|
home: string;
|
|
stubDir: string;
|
|
codexHome: string;
|
|
gstackHome: string;
|
|
stubLog: string;
|
|
stubArgsLog: string;
|
|
}
|
|
|
|
function makeFixture(): Fixture {
|
|
const home = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-model-probe-'));
|
|
const stubDir = path.join(home, 'stub-bin');
|
|
const codexHome = path.join(home, '.codex');
|
|
const gstackHome = path.join(home, '.gstack');
|
|
fs.mkdirSync(stubDir, { recursive: true });
|
|
fs.mkdirSync(codexHome, { recursive: true });
|
|
fs.mkdirSync(gstackHome, { recursive: true });
|
|
fs.writeFileSync(path.join(stubDir, 'codex'), STUB, { mode: 0o755 });
|
|
fs.writeFileSync(path.join(codexHome, 'config.toml'), 'model = "gpt-5.4"\n');
|
|
fs.writeFileSync(path.join(codexHome, 'auth.json'), '{}');
|
|
const stubLog = path.join(home, 'stub.log');
|
|
const stubArgsLog = path.join(home, 'stub-args.log');
|
|
return { home, stubDir, codexHome, gstackHome, stubLog, stubArgsLog };
|
|
}
|
|
|
|
function runProbe(f: Fixture, stubMode: string, extraEnv: Record<string, string> = {}): { stdout: string; status: number } {
|
|
const result = spawnSync(
|
|
'bash',
|
|
['-c', `set +e\nsource "${PROBE}"\n_gstack_codex_model_probe`],
|
|
{
|
|
env: {
|
|
PATH: `${f.stubDir}:${process.env.PATH ?? ''}`,
|
|
HOME: f.home,
|
|
CODEX_HOME: f.codexHome,
|
|
GSTACK_HOME: f.gstackHome,
|
|
STUB_MODE: stubMode,
|
|
STUB_LOG: f.stubLog,
|
|
STUB_ARGS_LOG: f.stubArgsLog,
|
|
_TEL: 'off',
|
|
...extraEnv,
|
|
},
|
|
timeout: 10000,
|
|
},
|
|
);
|
|
return { stdout: (result.stdout ?? '').toString(), status: result.status ?? -1 };
|
|
}
|
|
|
|
function lastArgs(f: Fixture): string {
|
|
try {
|
|
const lines = fs.readFileSync(f.stubArgsLog, 'utf-8').trim().split('\n').filter(Boolean);
|
|
return lines.at(-1) ?? '';
|
|
} catch {
|
|
return '';
|
|
}
|
|
}
|
|
|
|
function invocations(f: Fixture): number {
|
|
try {
|
|
return fs.readFileSync(f.stubLog, 'utf-8').split('\n').filter(Boolean).length;
|
|
} catch {
|
|
return 0;
|
|
}
|
|
}
|
|
|
|
describe('codex model probe (#2477)', () => {
|
|
test('successful round trip -> MODEL_OK, cached, no re-invocation', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
const first = runProbe(f, 'ok');
|
|
expect(first.stdout.trim()).toBe('MODEL_OK');
|
|
expect(first.status).toBe(0);
|
|
expect(invocations(f)).toBe(1);
|
|
expect(lastArgs(f)).toContain('-c model="gpt-6-astra"');
|
|
expect(fs.existsSync(path.join(f.gstackHome, '.codex-model-probe'))).toBe(true);
|
|
|
|
const second = runProbe(f, 'ok');
|
|
expect(second.stdout.trim()).toBe('MODEL_OK (cached)');
|
|
expect(second.status).toBe(0);
|
|
expect(invocations(f)).toBe(1); // cache hit: stub not re-invoked
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
test('model 400 -> MODEL_UNUSABLE with selected-model hints, exit 1, negative-cached', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
const r = runProbe(f, 'model400');
|
|
expect(r.stdout).toContain('MODEL_UNUSABLE');
|
|
expect(r.stdout).toContain('gstack requested model');
|
|
expect(r.stdout).toContain('GSTACK_CODEX_MODEL');
|
|
// Surfaces the actual rejection so the user sees WHICH model.
|
|
expect(r.stdout).toContain('gpt-6-astra');
|
|
expect(r.status).toBe(1);
|
|
// The deterministic 400 is config-driven: re-probing every preflight
|
|
// charged the user a 30s round trip + real tokens per review section.
|
|
// A second run within the 15-min TTL must NOT re-invoke codex, and must
|
|
// keep the exit-1 + hints contract so callers can't tell the difference.
|
|
expect(invocations(f)).toBe(1);
|
|
const second = runProbe(f, 'model400');
|
|
expect(second.stdout).toContain('MODEL_UNUSABLE (cached)');
|
|
expect(second.stdout).toContain('GSTACK_CODEX_MODEL');
|
|
expect(second.status).toBe(1);
|
|
expect(invocations(f)).toBe(1);
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
test('GSTACK_CODEX_MODEL change re-probes past a cached MODEL_UNUSABLE (the recovery path)', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
runProbe(f, 'model400');
|
|
expect(invocations(f)).toBe(1);
|
|
// Fixing the gstack model override changes the cache signature — the
|
|
// negative cache must not outlive the model it condemned.
|
|
const r = runProbe(f, 'ok', { GSTACK_CODEX_MODEL: 'gpt-5.6-sol' });
|
|
expect(r.stdout.trim()).toBe('MODEL_OK');
|
|
expect(r.status).toBe(0);
|
|
expect(invocations(f)).toBe(2);
|
|
expect(lastArgs(f)).toContain('-c model="gpt-5.6-sol"');
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
test('transient failure -> inconclusive, FAIL-OPEN exit 0', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
const r = runProbe(f, 'transient');
|
|
expect(r.stdout).toContain('MODEL_PROBE_INCONCLUSIVE');
|
|
expect(r.status).toBe(0);
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
test('TTL expiry: a cached MODEL_OK older than 3600s re-probes (T5)', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
runProbe(f, 'ok');
|
|
expect(invocations(f)).toBe(1);
|
|
// Backdate the cache line's timestamp past the 1h TTL, keeping the
|
|
// signature valid — TTL alone must force the re-probe.
|
|
const cachePath = path.join(f.gstackHome, '.codex-model-probe');
|
|
const [status, ts, sig] = fs.readFileSync(cachePath, 'utf-8').trim().split(' ');
|
|
expect(status).toBe('MODEL_OK');
|
|
fs.writeFileSync(cachePath, `MODEL_OK ${Number(ts) - 3700} ${sig}\n`);
|
|
const r = runProbe(f, 'ok');
|
|
expect(r.stdout.trim()).toBe('MODEL_OK'); // not "(cached)"
|
|
expect(invocations(f)).toBe(2); // re-probed
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
test('auth.json mtime change invalidates the cached MODEL_OK (T5: re-login re-probes)', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
runProbe(f, 'ok');
|
|
expect(invocations(f)).toBe(1);
|
|
// A re-login rewrites auth.json; the mtime signature must invalidate
|
|
// the cache even though config.toml is untouched.
|
|
const future = Date.now() / 1000 + 10;
|
|
fs.utimesSync(path.join(f.codexHome, 'auth.json'), future, future);
|
|
const r = runProbe(f, 'ok');
|
|
expect(r.stdout.trim()).toBe('MODEL_OK');
|
|
expect(invocations(f)).toBe(2); // re-probed
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
test('config.toml change invalidates the cached MODEL_OK', () => {
|
|
const f = makeFixture();
|
|
try {
|
|
runProbe(f, 'ok');
|
|
expect(invocations(f)).toBe(1);
|
|
// Change the model pin; mtime signature must invalidate the cache.
|
|
fs.writeFileSync(path.join(f.codexHome, 'config.toml'), 'model = "gpt-5.5"\n');
|
|
const future = Date.now() / 1000 + 10;
|
|
fs.utimesSync(path.join(f.codexHome, 'config.toml'), future, future);
|
|
const r = runProbe(f, 'ok');
|
|
expect(r.stdout.trim()).toBe('MODEL_OK');
|
|
expect(invocations(f)).toBe(2); // re-probed
|
|
} finally {
|
|
fs.rmSync(f.home, { recursive: true, force: true });
|
|
}
|
|
});
|
|
});
|