feat: run pi agent sessions at high thinking level (#475)

This commit is contained in:
ezl-keygraph authored and GitHub committed 2026-09-30 19:49:29 +05:30
1 parent 327c10fd90
commit 57c511ff8e
6 files changed
+19 -1

No files matched your search

+1 -1
View File
@@ -165,7 +165,7 @@ Around those phases:
- **Configuration** — YAML configs in `apps/worker/configs/` use the closed JSON Schema in `config-schema.json`. Every fresh scan runs the fixed five analysis classes; there is no public class selector. `agentic_sast.enabled` is the only public agentic-SAST setting. Finding reconciliation runs on every scan and has no public setting of its own. Config also supports authentication (MFA/TOTP), URL/code rule scoping (`rules.avoid`/`rules.focus`), `exploit`, free-form `rules_of_engagement`, and post-hoc `report` options (`min_severity`, `min_confidence`, `guidance`, and exploit-only `sarif` output via `apps/worker/src/services/sarif-renderer.ts`, on by default for exploit runs and opt out with `report.sarif: "false"`). `code_path` avoid rules are enforced via the `@gotgenes/pi-permission-system` extension: `apps/worker/src/temporal/activities.ts:syncCodePathDenyRules` writes a global `path` deny config once per workflow (`apps/worker/src/ai/pi/permission-system.ts:syncPermissionSystemConfig`), and the executor loads the extension when that config is present (`apps/worker/src/ai/pi/pi-executor.ts`), so denies fire across every tool and child `task` session. Credential resolution — local mode: env vars → `./.env`; npx mode: env vars → `~/.shannon/config.toml` (via `npx @keygraph/shannon setup`)
- **Agentic SAST progress** — Capella runs as a child workflow, so its activities are absent from the parent's `pendingActivities` and invisible to the CLI. The child signals each stage boundary up via `capellaStageProgress` (`apps/worker/src/temporal/shared.ts`); the parent's handler validates the payload and writes the child-supplied `startedAt` and `durationMs` directly to `operationalStages['agentic-sast:<stage>']`, so both the live `getProgress` query and the terminal result carry per-stage rows. Signalling is best-effort and every failure is swallowed — a closed or unreachable parent must never fail a SAST run. `CAPELLA_STAGE_LABELS` in `apps/worker/src/ai/sast/types.ts` is the one label table, shared by the scan log and the status tree; `CAPELLA_PROGRESS_STAGES` omits `export`, which runs no model and so never becomes a row. Scans predating the signal keep the aggregate `agentic-sast` span and render as a bare phase line
- **Prompts** — Per-phase templates in `apps/worker/prompts/` with variable substitution (`{{TARGET_URL}}`, `{{CONFIG_CONTEXT}}`). Shared partials in `apps/worker/prompts/shared/` via `apps/worker/src/services/prompt-manager.ts`, including `_code-path-rules.txt` (focus/avoid `[FILE]`/`[GLOB]` routing) and `_rules-of-engagement.txt` (free-text engagement rules). When `exploit: false`, `apps/worker/src/services/findings-renderer.ts` deterministically converts each `*_exploitation_queue.json` into a `*_findings.md` for report assembly — no LLM in the loop
- **Agent Harness (pi)** — Uses the **pi harness** (`@earendil-works/pi-coding-agent`, requires Node ≥ 22.19) via `apps/worker/src/ai/pi/pi-executor.ts` (`runPiPrompt` → `createAgentSession`). Retry is split in `apps/worker/src/ai/pi/retry-settings.ts`: pi's agent-level loop is off so Temporal owns agent restarts, while `provider.maxRetries` stays on — pi reads the `provider` block independently of the `enabled` flag — so transport faults are absorbed in-session rather than costing a full agent re-run. `maxRetryDelayMs` is left at pi's 60s default. One model runs every phase, named by `SHANNON_AI_MODEL=<provider>:<model-id>` (default `anthropic:claude-sonnet-4-6`). `apps/worker/src/ai/models.ts` parses the spec — splitting on the **first** colon only, so Bedrock IDs keep theirs — and resolves it through pi's `ModelRuntime`. pi ships the `CredentialStore` interface but no in-memory implementation (its own reads `auth.json` from disk), so `RuntimeCredentialStore` in that file supplies one: credentials arrive as env vars in an ephemeral container and must never touch disk. `createModelRuntime(providerId, apiKey)` builds the runtime with `allowModelNetwork: true`, so `ModelRuntime.create()` refreshes the model catalogue over the network at scan start and a freshly released model resolves without a `--models-config` file. The fetch is bounded (10s) and falls back to the static catalogue on timeout, so an unreachable catalogue endpoint cannot hang the scan. The refresh does not override a `--models-config`: pi reloads and re-applies that file as a config overlay on every refresh (it reloads `this.config` at the top of `refresh()`), so custom definitions still win over the fetched catalogue; the merge semantics below are unchanged, just layered over a fresher base. `resolveModelSelection()` is **async** because `ModelRuntime.create()` is. Any pi-ai provider id is accepted — `parseModelSpec` no longer rejects against a hardcoded list, so pi's registry is the authority (an unknown provider/model surfaces as a clear "not found in pi registry" error at preflight, which points to the browsable catalogue at `pi.dev/models` — `PI_CATALOG_URL` in `apps/worker/src/ai/models.ts`, appended to the not-found errors and shown in the setup wizard's "Other provider" hint). Four providers are **curated** (`CURATED_PROVIDERS`: `anthropic`, `openai`, `xai`, `amazon-bedrock`) with their own credential variables, config sections, and setup flows; each provider's API key env var is declared once in `PROVIDER_API_KEY_ENV` — Shannon uses each vendor's own variable name (`OPENAI_API_KEY`, `XAI_API_KEY`, …), never an invented one; Bedrock's entry is `AWS_BEARER_TOKEN_BEDROCK`, paired with `AWS_REGION`, which preflight requires separately as provider config rather than a credential. Any other provider uses the **generic** credential path: `SHANNON_AI_API_KEY` (`GENERIC_API_KEY_ENV`) supplies the key for any provider whose credential is a plain API key. Curated providers' own variables take precedence over it, and it also works as a fallback for them — Bedrock is the sole exception (it authenticates through its AWS_ variables, so the generic key never stands in for it). The CLI forwards `SHANNON_AI_API_KEY` in `COMMON_FORWARD_VARS` (it is provider-neutral, binding to whatever `SHANNON_AI_MODEL` names, so the "only one provider configured" guard counts only named credentials), and stores it under a generic `[provider]` config.toml section (`provider.api_key`). `npx @keygraph/shannon setup` exposes this as the "Other provider" option: free-text provider id + model id + key (a curated provider id is rejected there, since it has its own option). A model pi's catalogue does not carry, such as a self-hosted model, is reachable without an SDK bump: `--models-config <file>` mounts a pi `models.json` read-only at `/app/models.json`. The mount is the entire CLI→worker protocol: nothing is forwarded through the environment, and `modelsConfigPath()` detects the file at that fixed path, exactly as `piAuthPresent()` detects the pi auth mount whose flag is likewise not forwarded (`MODELS_CONFIG_CONTAINER_PATH` in the CLI and `MODELS_CONFIG_PATH` in `apps/worker/src/paths.ts` must stay in sync). `createModelRuntime` always names `modelsPath` explicitly — the mounted path, or **`null` when no config was supplied**, which switches models.json off outright. It is never left to pi's default of `<agent dir>/models.json`, because that dir is shared with the pi auth mount, so a file landing there must not silently contribute model definitions to a scan that did not ask for one. `modelsStorePath` is pinned to the agent dir alongside it, since pi otherwise derives it from `dirname(modelsPath)` and would try to write beside a read-only mount. Custom definitions merge over the built-in catalogue: a matching model id replaces the built-in entry, a new id is added alongside, and `modelOverrides` adjusts a built-in without replacing the proviLine truncated
- **Agent Harness (pi)** — Uses the **pi harness** (`@earendil-works/pi-coding-agent`, requires Node ≥ 22.19) via `apps/worker/src/ai/pi/pi-executor.ts` (`runPiPrompt` → `createAgentSession`). Retry is split in `apps/worker/src/ai/pi/retry-settings.ts`: pi's agent-level loop is off so Temporal owns agent restarts, while `provider.maxRetries` stays on — pi reads the `provider` block independently of the `enabled` flag — so transport faults are absorbed in-session rather than costing a full agent re-run. `maxRetryDelayMs` is left at pi's 60s default. One model runs every phase, named by `SHANNON_AI_MODEL=<provider>:<model-id>` (default `anthropic:claude-sonnet-4-6`). `apps/worker/src/ai/models.ts` parses the spec — splitting on the **first** colon only, so Bedrock IDs keep theirs — and resolves it through pi's `ModelRuntime`. pi ships the `CredentialStore` interface but no in-memory implementation (its own reads `auth.json` from disk), so `RuntimeCredentialStore` in that file supplies one: credentials arrive as env vars in an ephemeral container and must never touch disk. `createModelRuntime(providerId, apiKey)` builds the runtime with `allowModelNetwork: true`, so `ModelRuntime.create()` refreshes the model catalogue over the network at scan start and a freshly released model resolves without a `--models-config` file. The fetch is bounded (10s) and falls back to the static catalogue on timeout, so an unreachable catalogue endpoint cannot hang the scan. The refresh does not override a `--models-config`: pi reloads and re-applies that file as a config overlay on every refresh (it reloads `this.config` at the top of `refresh()`), so custom definitions still win over the fetched catalogue; the merge semantics below are unchanged, just layered over a fresher base. `resolveModelSelection()` is **async** because `ModelRuntime.create()` is. Any pi-ai provider id is accepted — `parseModelSpec` no longer rejects against a hardcoded list, so pi's registry is the authority (an unknown provider/model surfaces as a clear "not found in pi registry" error at preflight, which points to the browsable catalogue at `pi.dev/models` — `PI_CATALOG_URL` in `apps/worker/src/ai/models.ts`, appended to the not-found errors and shown in the setup wizard's "Other provider" hint). Four providers are **curated** (`CURATED_PROVIDERS`: `anthropic`, `openai`, `xai`, `amazon-bedrock`) with their own credential variables, config sections, and setup flows; each provider's API key env var is declared once in `PROVIDER_API_KEY_ENV` — Shannon uses each vendor's own variable name (`OPENAI_API_KEY`, `XAI_API_KEY`, …), never an invented one; Bedrock's entry is `AWS_BEARER_TOKEN_BEDROCK`, paired with `AWS_REGION`, which preflight requires separately as provider config rather than a credential. Any other provider uses the **generic** credential path: `SHANNON_AI_API_KEY` (`GENERIC_API_KEY_ENV`) supplies the key for any provider whose credential is a plain API key. Curated providers' own variables take precedence over it, and it also works as a fallback for them — Bedrock is the sole exception (it authenticates through its AWS_ variables, so the generic key never stands in for it). The CLI forwards `SHANNON_AI_API_KEY` in `COMMON_FORWARD_VARS` (it is provider-neutral, binding to whatever `SHANNON_AI_MODEL` names, so the "only one provider configured" guard counts only named credentials), and stores it under a generic `[provider]` config.toml section (`provider.api_key`). `npx @keygraph/shannon setup` exposes this as the "Other provider" option: free-text provider id + model id + key (a curated provider id is rejected there, since it has its own option). A model pi's catalogue does not carry, such as a self-hosted model, is reachable without an SDK bump: `--models-config <file>` mounts a pi `models.json` read-only at `/app/models.json`. The mount is the entire CLI→worker protocol: nothing is forwarded through the environment, and `modelsConfigPath()` detects the file at that fixed path, exactly as `piAuthPresent()` detects the pi auth mount whose flag is likewise not forwarded (`MODELS_CONFIG_CONTAINER_PATH` in the CLI and `MODELS_CONFIG_PATH` in `apps/worker/src/paths.ts` must stay in sync). `createModelRuntime` always names `modelsPath` explicitly — the mounted path, or **`null` when no config was supplied**, which switches models.json off outright. It is never left to pi's default of `<agent dir>/models.json`, because that dir is shared with the pi auth mount, so a file landing there must not silently contribute model definitions to a scan that did not ask for one. `modelsStorePath` is pinned to the agent dir alongside it, since pi otherwise derives it from `dirname(modelsPath)` and would try to write beside a read-only mount. Custom definitions merge over the built-in catalogue: a matching model id replaces the built-in entry, a new id is added alongside, and `modelOverrides` adjusts a built-in without replacing the proviLine truncated
- **Pi Credential Reuse** — `SHANNON_USE_PI_AUTH=1` opts into reusing the host's Pi login, including an `openai-codex` ChatGPT Plus/Pro subscription (`SHANNON_AI_MODEL=openai-codex:<model-id>`) or an `xai` Grok subscription (`SHANNON_AI_MODEL=xai:<model-id>`); the mechanism is provider-agnostic and works for any Pi login. `apps/cli/src/env.ts` requires `~/.pi/agent/auth.json`; `start.ts` passes its path to `spawnWorker`, which mounts only that file read-write at `/tmp/.pi/agent/auth.json`. The flag itself is not forwarded: the worker detects the file with `piAuthPresent()` and passes its path to `ModelRuntime.create`. CLI and worker API-key presence checks are skipped on this path, but the normal preflight model probe still validates the credential. The image and UID-remapping entrypoint keep `/tmp/.pi/agent` owned by `pentest` so adjacent Pi/Shannon configuration remains writable. Refreshed OAuth state is persisted to the host for subsequent scans.
- **Audit System** — Crash-safe append-only logging in `workspaces/{hostname}_{sessionId}/`. The run directory's top level holds the human-facing report in both formats (`Security-Assessment-Report.pdf` and `Security-Assessment-Report.md`, `FINAL_REPORT_PDF_FILENAME`/`FINAL_REPORT_MD_FILENAME` in `apps/worker/src/paths.ts`); everything else — deliverables, per-agent logs, prompts, `session.json`, `workflow.log`, and browser artifacts — is nested under a hidden `.shannon/` internals dir (`INTERNAL_DIR`) so a customer sees only the report. Audit path helpers route through `generateInternalPath` (`apps/worker/src/audit/utils.ts`); the CLI nests the overlay backing dirs under the same `.shannon/` (`apps/cli/src/docker.ts`, `start.ts`). `session.json`/`workflow.log` reads use dual-read resolvers (`resolveSessionJsonPath`, `resolveRunFile`) that prefer `.shannon/` and fall back to the legacy run-root layout, so pre-restructure workspaces stay listable (`scans`/`logs`) without migration. A pre-restructure workspace cannot be resumed: `classifyWorkspaceLaunch` (`apps/cli/src/commands/start.ts`) requires `.shannon/launch.json`, and its absence fails the launch as "created by an earlier version of Shannon" before anything on disk is touched. There is no in-place migration — the workspace's files and report are left untouched, and the operator starts a new scan under a different `-w` name. The report agent writes structured findings to `report.json`, from which `report-renderer.ts` renders the assembled markdown and `report-json-adapter.ts` produces the Typst-shaped JSON that `pdf-renderer.ts` compiles into `comprehensive_security_assessment_report.pdf` using the bundled `apps/worker/templates/typst/report.typ` template (the `typst` binary is installed in the worker image). `copyReportToRunRoot` (`apps/worker/src/services/reporting.ts`) surfaces both the PDF and the markdown to the run root as `Security-Assessment-Report.pdf` and `Security-Assessment-Report.md`; the deliverables-dir copies remain as the git-checkpointed sources. PDF compilation is best-effort — a failure is logged and the run still completes. WorkflowLogger (`apps/worker/src/audit/workflow-logger.ts`) provides unified human-readable per-workflow logs, backed by LogStream (`apps/worker/src/audit/log-stream.ts`) shared stream primitive. Every combined-log line is also projected into a per-agent file under `.shannon/agents/<slug>.log` (one per pipeline agent, one per Capella stage; subagents fold into the parent's file, and a stage's concurrent sessions share its file with an inline session label). The projection boundary is `apps/worker/src/audit/actor-projection.ts` (`projectActor` maps a `TraceActor` to its combined prefix and owning file slug — slugs come only from closed fields); fan-out is best-effort and never blocks the canonical combined log. A lifecycle owner holds a `LogStream` lease per agent file (the pipeline agent's `logAgent` span, or a Capella stage activity's `try/finally`) so per-line writes ride the reference count; `CapellaStageTrace.drain()` flushes a stage's trace queue before its activity returns. The CLI tails one file with `shannon logs --agent <name>` (`--list-agents` to enumerate); the default `shannon logs` path is unchanged
- **Deliverables** — Saved to `.shannon/deliverables/` in the target repo via the `save-deliverable` CLI script (`apps/worker/src/scripts/save-deliverable.ts`)
@@ -32,6 +32,7 @@ import type {
CapellaTool,
} from './capella-agent-types.js';
import { PI_RETRY_SETTINGS } from './retry-settings.js';
import { PI_THINKING_LEVEL } from './thinking-level.js';
const MAX_ERROR_LENGTH = 2_000;
const MAX_TOOLS_PER_SESSION = 32;
@@ -393,6 +394,7 @@ class StandaloneCapellaAgentExecutor implements CapellaAgentExecutor {
cwd: request.cwd,
agentDir,
model: selection.model,
thinkingLevel: PI_THINKING_LEVEL,
modelRuntime: selection.modelRuntime,
noTools: 'all',
tools: toolNames,
+2
View File
@@ -48,6 +48,7 @@ import { permissionSystemConfigExists, permissionSystemPackageDir } from './perm
import { PI_RETRY_SETTINGS } from './retry-settings.js';
import { createGlobTool, createTodoWriteTool } from './session-tools.js';
import { createTaskTool } from './task-tool.js';
import { PI_THINKING_LEVEL } from './thinking-level.js';
import { TraceEmitter } from './trace-emitter.js';
import { providerTurnError, type SafeProviderTurnDetails, safeProviderTurnDetails } from './turn-error.js';
@@ -332,6 +333,7 @@ export async function runPiPrompt(
({ session } = await createAgentSession({
cwd: sourceDir,
model: selection.model,
thinkingLevel: PI_THINKING_LEVEL,
tools,
customTools,
modelRuntime: selection.modelRuntime,
@@ -29,6 +29,7 @@ import type { ValidatingSubmitTool } from '../reconciliation/submit-validation.j
import { ConfinementError, compileRepositoryGlob, RepositoryConfinement } from '../sast/capella/tools/confinement.js';
import { createCapellaRepositoryTools } from '../sast/capella/tools/repository-tools.js';
import { PI_RETRY_SETTINGS } from './retry-settings.js';
import { PI_THINKING_LEVEL } from './thinking-level.js';
const DEFAULT_TIMEOUT_MS = 30 * 60 * 1_000;
const DEFAULT_MAX_TURNS = 64;
@@ -527,6 +528,7 @@ class StandaloneTaskFormationExecutor implements TaskFormationExecutor {
cwd: request.cwd,
agentDir,
model: selection.model,
thinkingLevel: PI_THINKING_LEVEL,
modelRuntime: selection.modelRuntime,
noTools: 'all',
tools: toolNames,
+2
View File
@@ -19,6 +19,7 @@ import {
} from '@earendil-works/pi-coding-agent';
import { type LoggableAgentName, normalizeSemanticLabel } from '../../audit/safe-fields.js';
import { PI_RETRY_SETTINGS } from './retry-settings.js';
import { PI_THINKING_LEVEL } from './thinking-level.js';
import { TraceEmitter } from './trace-emitter.js';
export interface TaskToolContext {
@@ -135,6 +136,7 @@ export function createTaskTool(config: TaskToolContext): ToolDefinition {
agentDir,
resourceLoader,
model: config.model,
thinkingLevel: PI_THINKING_LEVEL,
tools: CHILD_TOOLS,
modelRuntime: config.modelRuntime,
sessionManager: SessionManager.inMemory(config.cwd),
+10
View File
@@ -0,0 +1,10 @@
// Copyright (C) 2026 Keygraph, Inc.
//
// This program is free software: you can redistribute it and/or modify
// it under the terms of the GNU Affero General Public License version 3
// as published by the Free Software Foundation.
import type { ThinkingLevel } from '@earendil-works/pi-agent-core';
/** Thinking level for every pi agent session, raised above pi's default for deeper analysis. */
export const PI_THINKING_LEVEL: ThinkingLevel = 'high';