mirror of
https://github.com/garrytan/gstack.git
synced 2026-09-26 22:51:47 +02:00
* fix(settings): preserve symlinked settings targets
Resolve the selected target for locking, mutation, backup, and rollback; refuse target changes and preserve private modes. Addresses #2830.
* fix(redact): bind masking to original detected spans
Inspired by #2929's anchored-span diagnosis; independently implemented using normalization offsets. Addresses #2930 and the relocation portion of #2912 without changing detection sensitivity.
* fix(evals): exclude operator credentials from prefix admission
Adapts the credential-suffix screen proposed in #2636, with real launched-child regression coverage and deliberate provider-auth exceptions.
* fix(artifacts): retain custom allowlist rules on reinitialization
Preserve the exact user-owned suffix and publish only a successfully assembled replacement. Independently implements the repair reported in #2907.
* test(cso): verify exact masked reads and unmaskable payload refusal
* fix(cso): preserve exact filesystem identities through lease recovery
Preserve 64-bit device/inode identity and nanosecond race checks. Add native NTFS lifecycle coverage for #2927; retain ambiguous legacy-state refusal without claiming Windows PID-reuse recovery is resolved.
* fix(redact): bind pre-push scans to destination and preserve seam context
Uses #2935 (bd07318) as source evidence for push-target range and slice-overlap defects. Independently implemented; no cherry-pick or release metadata adoption.
* test(ci): gate native agent ownership and settings links on macOS
* fix(browse): bind agent lifetimes and cleanup to owned generations
Uses #2931 by Chris Hutton / Claude Fable 5.1 as attributed design input; independently implemented without broad sweeps or copied code. Keep uncertain children and locks rather than deleting foreign state.
* test(ci): include concurrent shutdown controls in the native macOS gate
* v1.88.1.0 fix: harden credential boundaries and owned state
* fix(redact): preserve target provenance and scan boundary semantics
* test(artifacts): read managed rules from atomic allowlist assembly
* fix: preserve native exit observations and fixture prerequisites
* fix: preserve UTF-16 offsets through redaction normalization
259 lines
11 KiB
TypeScript
259 lines
11 KiB
TypeScript
import { describe, test, expect } from 'bun:test';
|
|
import { spawnSync } from 'node:child_process';
|
|
import * as fs from 'node:fs';
|
|
import * as os from 'node:os';
|
|
import * as path from 'node:path';
|
|
import { pathToFileURL } from 'node:url';
|
|
import { parseNDJSON } from './session-runner';
|
|
|
|
test('runSkillTest launches a child without operator credentials', () => {
|
|
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gstack-hermetic-session-'));
|
|
try {
|
|
const bin = path.join(root, 'claude');
|
|
fs.writeFileSync(bin, `#!/usr/bin/env node
|
|
const names = ['GITHUB_TOKEN', 'GITHUB_PERSONAL_ACCESS_TOKEN', 'GITHUB_APP_PRIVATE_KEY', 'GH_TOKEN', 'GITHUB_ACTIONS', 'GITHUB_PATH', 'GITHUB_TOKENIZER', 'EVALS_RUN_ID'];
|
|
const present = Object.fromEntries(names.map(name => [name, Object.hasOwn(process.env, name)]));
|
|
console.log(JSON.stringify({type: 'result', subtype: 'success', result: JSON.stringify(present)}));
|
|
`, { mode: 0o700 });
|
|
const script = `import { runSkillTest } from ${JSON.stringify(pathToFileURL(path.join(import.meta.dir, 'session-runner.ts')).href)};
|
|
const result = await runSkillTest({prompt: 'synthetic fixture', workingDirectory: ${JSON.stringify(root)}, model: 'fixture', timeout: 5000, startupGraceMs: 5000, allowedTools: []});
|
|
console.log(JSON.stringify({exitReason: result.exitReason, child: JSON.parse(result.output)}));`;
|
|
const result = spawnSync(process.execPath, ['-e', script], {
|
|
cwd: path.resolve(import.meta.dir, '..', '..'),
|
|
encoding: 'utf8',
|
|
timeout: 30_000,
|
|
env: {
|
|
PATH: `${root}${path.delimiter}${process.env.PATH ?? '/usr/bin:/bin'}`,
|
|
HOME: root,
|
|
TMPDIR: os.tmpdir(),
|
|
GITHUB_TOKEN: 'synthetic-token',
|
|
GITHUB_PERSONAL_ACCESS_TOKEN: 'synthetic-pat',
|
|
GITHUB_APP_PRIVATE_KEY: 'synthetic-private-key',
|
|
GH_TOKEN: 'synthetic-gh-token',
|
|
GITHUB_ACTIONS: 'true',
|
|
GITHUB_PATH: '/tmp/actions-path',
|
|
GITHUB_TOKENIZER: 'metadata-tokenizer',
|
|
EVALS_RUN_ID: 'synthetic-run',
|
|
},
|
|
});
|
|
expect(result.status, result.stderr).toBe(0);
|
|
expect(JSON.parse(result.stdout)).toEqual({
|
|
exitReason: 'success',
|
|
child: {
|
|
GITHUB_TOKEN: false,
|
|
GITHUB_PERSONAL_ACCESS_TOKEN: false,
|
|
GITHUB_APP_PRIVATE_KEY: false,
|
|
GH_TOKEN: false,
|
|
GITHUB_ACTIONS: true,
|
|
GITHUB_PATH: true,
|
|
GITHUB_TOKENIZER: true,
|
|
EVALS_RUN_ID: true,
|
|
},
|
|
});
|
|
} finally {
|
|
fs.rmSync(root, { recursive: true, force: true });
|
|
}
|
|
});
|
|
|
|
// Fixture: minimal NDJSON session (system init, assistant with tool_use, tool result, assistant text, result)
|
|
const FIXTURE_LINES = [
|
|
'{"type":"system","subtype":"init","session_id":"test-123"}',
|
|
'{"type":"assistant","message":{"content":[{"type":"tool_use","id":"tu1","name":"Bash","input":{"command":"echo hello"}}]}}',
|
|
'{"type":"user","tool_use_result":{"tool_use_id":"tu1","stdout":"hello\\n","stderr":""}}',
|
|
'{"type":"assistant","message":{"content":[{"type":"text","text":"The command printed hello."}]}}',
|
|
'{"type":"assistant","message":{"content":[{"type":"text","text":"Let me also read a file."},{"type":"tool_use","id":"tu2","name":"Read","input":{"file_path":"/tmp/test"}}]}}',
|
|
'{"type":"result","subtype":"success","total_cost_usd":0.05,"num_turns":3,"usage":{"input_tokens":100,"output_tokens":50},"result":"Done."}',
|
|
];
|
|
|
|
describe('parseNDJSON', () => {
|
|
test('parses valid NDJSON with system + assistant + result events', () => {
|
|
const parsed = parseNDJSON(FIXTURE_LINES);
|
|
expect(parsed.transcript).toHaveLength(6);
|
|
expect(parsed.transcript[0].type).toBe('system');
|
|
expect(parsed.transcript[5].type).toBe('result');
|
|
});
|
|
|
|
test('extracts tool calls from assistant.message.content[].type === tool_use', () => {
|
|
const parsed = parseNDJSON(FIXTURE_LINES);
|
|
expect(parsed.toolCalls).toHaveLength(2);
|
|
expect(parsed.toolCalls[0]).toEqual({
|
|
tool: 'Bash',
|
|
input: { command: 'echo hello' },
|
|
output: '',
|
|
});
|
|
expect(parsed.toolCalls[1]).toEqual({
|
|
tool: 'Read',
|
|
input: { file_path: '/tmp/test' },
|
|
output: '',
|
|
});
|
|
expect(parsed.toolCallCount).toBe(2);
|
|
});
|
|
|
|
test('skips malformed lines without throwing', () => {
|
|
const lines = [
|
|
'{"type":"system"}',
|
|
'this is not json',
|
|
'{"type":"assistant","message":{"content":[{"type":"text","text":"ok"}]}}',
|
|
'{incomplete json',
|
|
'{"type":"result","subtype":"success","result":"done"}',
|
|
];
|
|
const parsed = parseNDJSON(lines);
|
|
expect(parsed.transcript).toHaveLength(3); // system, assistant, result
|
|
expect(parsed.resultLine?.subtype).toBe('success');
|
|
});
|
|
|
|
test('skips empty and whitespace-only lines', () => {
|
|
const lines = [
|
|
'',
|
|
' ',
|
|
'{"type":"system"}',
|
|
'\t',
|
|
'{"type":"result","subtype":"success","result":"ok"}',
|
|
];
|
|
const parsed = parseNDJSON(lines);
|
|
expect(parsed.transcript).toHaveLength(2);
|
|
});
|
|
|
|
test('extracts resultLine from type: "result" event', () => {
|
|
const parsed = parseNDJSON(FIXTURE_LINES);
|
|
expect(parsed.resultLine).not.toBeNull();
|
|
expect(parsed.resultLine.subtype).toBe('success');
|
|
expect(parsed.resultLine.total_cost_usd).toBe(0.05);
|
|
expect(parsed.resultLine.num_turns).toBe(3);
|
|
expect(parsed.resultLine.result).toBe('Done.');
|
|
});
|
|
|
|
test('counts turns correctly — one per assistant event, not per text block', () => {
|
|
const parsed = parseNDJSON(FIXTURE_LINES);
|
|
// 3 assistant events in fixture (tool_use, text, text+tool_use)
|
|
expect(parsed.turnCount).toBe(3);
|
|
});
|
|
|
|
test('handles empty input', () => {
|
|
const parsed = parseNDJSON([]);
|
|
expect(parsed.transcript).toHaveLength(0);
|
|
expect(parsed.resultLine).toBeNull();
|
|
expect(parsed.turnCount).toBe(0);
|
|
expect(parsed.toolCallCount).toBe(0);
|
|
expect(parsed.toolCalls).toHaveLength(0);
|
|
});
|
|
|
|
test('handles assistant event with no content array', () => {
|
|
const lines = [
|
|
'{"type":"assistant","message":{}}',
|
|
'{"type":"assistant"}',
|
|
];
|
|
const parsed = parseNDJSON(lines);
|
|
expect(parsed.turnCount).toBe(2);
|
|
expect(parsed.toolCalls).toHaveLength(0);
|
|
});
|
|
|
|
test('associates Agent verdict text with its tool-use ID without transport metadata or reasoning', () => {
|
|
const lines = [
|
|
{ type: 'assistant', message: { content: [
|
|
{ type: 'tool_use', id: 'review', name: 'Agent', input: { prompt: 'Review the design.' } },
|
|
] } },
|
|
{ type: 'user', tool_use_result: { content: [
|
|
{ type: 'thinking', thinking: 'Private computation must not become tool output.' },
|
|
{ type: 'text', text: 'Completeness: missing failure handling.' },
|
|
{ type: 'text', text: 'Quality score: 7/10' },
|
|
] }, message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'review', content: [
|
|
{ type: 'text', text: 'Completeness: missing failure handling.\nQuality score: 7/10' },
|
|
{ type: 'text', text: 'agentId: child-review\n<usage>duration_ms: 1000</usage>' },
|
|
] },
|
|
] } },
|
|
].map(event => JSON.stringify(event));
|
|
|
|
expect(parseNDJSON(lines).toolCalls).toEqual([{
|
|
tool: 'Agent',
|
|
input: { prompt: 'Review the design.' },
|
|
output: 'Completeness: missing failure handling.\nQuality score: 7/10',
|
|
}]);
|
|
});
|
|
|
|
test('preserves Task error-result diagnostics as output', () => {
|
|
const lines = [
|
|
{ type: 'assistant', message: { content: [
|
|
{ type: 'tool_use', id: 'failed-review', name: 'Task', input: {} },
|
|
] } },
|
|
{ type: 'user', message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'failed-review', is_error: true, content: 'Reviewer failed: deadline exceeded.' },
|
|
] } },
|
|
].map(event => JSON.stringify(event));
|
|
|
|
expect(parseNDJSON(lines).toolCalls[0].output).toBe('Reviewer failed: deadline exceeded.');
|
|
});
|
|
|
|
test('flattens only public text blocks from matching message results', () => {
|
|
const lines = [
|
|
{ type: 'assistant', message: { content: [
|
|
{ type: 'tool_use', id: 'legacy-task', name: 'Task', input: {} },
|
|
{ type: 'tool_use', id: 'shell', name: 'Bash', input: { command: 'echo done' } },
|
|
] } },
|
|
{ type: 'user', message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'legacy-task', content: [
|
|
{ type: 'text', text: 'Consistency: PASS' },
|
|
{ type: 'image', source: { data: 'not-text' } },
|
|
{ type: 'thinking', thinking: 'Private computation.' },
|
|
{ type: 'redacted_thinking', data: 'opaque' },
|
|
null,
|
|
{ type: 'text', text: 123 },
|
|
{ type: 'text', text: 'Quality score: 10/10' },
|
|
] },
|
|
{ type: 'tool_result', tool_use_id: 'shell', content: 'done\n' },
|
|
] } },
|
|
].map(event => JSON.stringify(event));
|
|
|
|
const parsed = parseNDJSON(lines);
|
|
expect(parsed.toolCalls.map(call => call.output)).toEqual([
|
|
'Consistency: PASS\nQuality score: 10/10', 'done\n',
|
|
]);
|
|
expect(parsed.toolCallCount).toBe(2);
|
|
expect(parsed.turnCount).toBe(1);
|
|
});
|
|
|
|
test('leaves missing, unmatched, and malformed results empty', () => {
|
|
const lines = [
|
|
{ type: 'assistant', message: { content: [
|
|
{ type: 'tool_use', id: 'missing', name: 'Agent', input: {} },
|
|
{ type: 'tool_use', id: 'malformed', name: 'Task', input: {} },
|
|
{ type: 'tool_use', name: 'Read', input: {} },
|
|
] } },
|
|
{ type: 'user', message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'unknown', content: 'Do not attach to the latest call.' },
|
|
{ type: 'tool_result', tool_use_id: 'malformed', content: { text: 'Not a public content block.' } },
|
|
{ type: 'tool_result', content: 'No tool-use ID.' },
|
|
] } },
|
|
{ type: 'user', message: { content: 'Not a tool-result array.' } },
|
|
].map(event => JSON.stringify(event));
|
|
|
|
expect(parseNDJSON(lines).toolCalls.map(call => call.output)).toEqual(['', '', '']);
|
|
});
|
|
|
|
test('scopes repeated tool-use IDs to the parent so child results cannot replace the parent verdict', () => {
|
|
const lines = [
|
|
{ type: 'assistant', parent_tool_use_id: null, message: { content: [
|
|
{ type: 'tool_use', id: 'shared', name: 'Agent', input: { prompt: 'Parent review' } },
|
|
] } },
|
|
{ type: 'assistant', parent_tool_use_id: 'shared', message: { content: [
|
|
{ type: 'tool_use', id: 'shared', name: 'Read', input: { file_path: '/tmp/design.md' } },
|
|
] } },
|
|
{ type: 'user', parent_tool_use_id: 'shared', message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'shared', content: 'Child file content' },
|
|
] } },
|
|
{ type: 'user', message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'shared', content: 'Parent review verdict' },
|
|
] } },
|
|
{ type: 'user', parent_tool_use_id: 'another-child', message: { content: [
|
|
{ type: 'tool_result', tool_use_id: 'shared', content: 'Unrelated child result' },
|
|
] } },
|
|
].map(event => JSON.stringify(event));
|
|
|
|
const parsed = parseNDJSON(lines);
|
|
expect(parsed.toolCalls.map(call => call.output)).toEqual(['Parent review verdict', 'Child file content']);
|
|
expect(parsed.toolCallCount).toBe(2);
|
|
expect(parsed.turnCount).toBe(2);
|
|
});
|
|
});
|