Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
50 changes: 13 additions & 37 deletions src/agent/index.ts
Original file line number Diff line number Diff line change
@@ -1,15 +1,13 @@
/**
* Public entry of the agent. Code outside `src/agent` imports runtime values
* from here and types from `./types`; deeper imports fail lint and the
* architecture test. Keep this list to what callers use, and keep it cheap:
* from here and types from `./types`; deeper imports fail `pnpm typecheck`.
* Keep this list to what callers use, and keep it cheap:
* the startup chunk imports this module, so anything re-exported here loads
* before the wizard does any work. Heavy paths stay behind a lazy import.
*
* Grouped by fate, per the stack plan (sections 4.1 to 4.5 and 7).
*/

/**
* Stays. The agent's contract: the one way to run it, the marker strings
* The agent's contract: the one way to run it, the marker strings
* program prompts embed, and the tool ids programs put in allowedTools and
* disallowedTools.
*/
Expand All @@ -18,40 +16,18 @@ export { runAgent, RunOutcome } from './runner';
export { AgentSignals } from './agent-interface';
export { WIZARD_TOOL_NAMES } from './tools';

/**
* Leaves when the bindings move to programs. Bindings and program data move to
* programs: resolveBinding
* is keyed by PROGRAM_BINDINGS and the agent keeps only "run from an
* already-resolved binding"; shouldDisableAsk is a flags policy programs
* decide and pass in; LONGER_ASK_TIMEOUT_MS is a tuning number programs own
* as askTimeoutMs.
*/
export { resolveBinding, shouldDisableAsk } from './runner';
export { LONGER_ASK_TIMEOUT_MS } from './wizard-ask-bridge';

/**
* Leaves later in the refactor. buildRunTags builds the trace tags runProgram
* and agentic detection send. configureGatewayFromCIEnvironment loads the CI
* gateway token. flushScanReport becomes a progress event rather than a call.
* downloadSkill leaves once skill install becomes shared.
*/
export { buildRunTags } from './agent-interface';
export { configureGatewayFromCIEnvironment } from './gateway-session';
export { flushScanReport } from './yara-hooks';
export { downloadSkill } from './tools';
/** The frameworkContext slot the legacy adapter fills for the e2e harness. */
export { TASK_OUTCOMES_KEY } from './runner';
/** The binding a program gets when it declares none. */
export { DEFAULT_BINDING } from './runner';

/**
* Leaves in C2. The TUI receives agent data through program state. Until
* then the suggested-prompts screen streams through this wrapper, which loads
* the streaming module on first call so the startup chunk does not grow.
* The MCP tutorial's prompt stream: one prompt against the PostHog MCP server,
* streamed as chunks. The MCP tool wraps it as `runMcpPrompt`, and the TUI
* reaches it only through `@tools`. It loads the streaming module on
* first call so the startup chunk does not grow.
*/
export async function* runMcpPromptViaSdk(
args: Parameters<
typeof import('./mcp-prompt-streaming').runMcpPromptViaSdk
>[0],
): AsyncIterable<import('./types').AgentChunk> {
export async function* streamMcpPrompt(
args: Parameters<typeof import('./mcp-prompt-streaming').streamMcpPrompt>[0],
): AsyncIterable<import('./types').McpPromptChunk> {
const streaming = await import('./mcp-prompt-streaming');
yield* streaming.runMcpPromptViaSdk(args);
yield* streaming.streamMcpPrompt(args);
}
94 changes: 23 additions & 71 deletions src/agent/progress.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,64 +4,16 @@
* `runAgent` reports through one optional callback and asks through one
* optional set of capabilities. Neither reaches into a UI singleton, a store,
* or a session: every payload is copied data, every question is awaited on an
* injected answerer. The legacy adapter in `src/programs/run-agent-legacy.ts`
* maps these back onto `WizardUI` one call per event, so the terminal output of
* every existing runner is unchanged.
* injected answerer. The caller decides what each event looks like.
*/

import type { SettingsConflict } from '@shared/claude-settings';

// ── What the agent hands back and asks with ─────────────────────────

/** Outcome kind for the outro screen */
export enum OutroKind {
Success = 'success',
Error = 'error',
Cancel = 'cancel',
}

export interface OutroData {
kind: OutroKind;
/** Main headline (green check for Success, red X for Error, etc.) */
message?: string;
/** Free-form body text shown under the headline. Use \n for paragraph breaks. */
body?: string;
/** Success-only: bulleted list of "what the agent did" */
changes?: string[];
/**
* Success-only: a prominent, labeled link to where the user should go
* next (e.g. an inbox the program just configured). Rendered right under
* the headline and shown verbatim — no UTM tagging — so the URL stays
* clean and copy-pasteable. Set per-program in buildOutroData.
*/
primaryLink?: { label: string; url: string };
/**
* Success-only: a short "what to do next" checklist with its own heading,
* rendered as a bulleted list. Distinct from `changes`, which recaps what
* the agent already did.
*/
nextSteps?: { heading: string; items: string[] };
docsUrl?: string;
continueUrl?: string;
/** Report file the agent wrote (e.g. "posthog-setup-report.md") */
reportFile?: string;
/** Stable machine-readable error code from the error catalog (@lib/errors). */
errorCode?: import('@shared/errors').ErrorCode;
/** Structured context for the error code; safe for telemetry payloads. */
errorDetail?: Record<string, unknown>;
/** PostHog dashboard URL the program created on the user's behalf. */
dashboardUrl?: string;
/** PostHog notebook URL the program uploaded the report to. */
notebookUrl?: string;
/**
* Copy-paste prompt the operator hands to their coding agent to finish the
* job (work the report's checklist). Printed to the terminal's main buffer on
* exit (see getExitLine in start-tui.ts) — the TUI's alternate screen is wiped
* on exit, so the scrollback line is where it survives and can be
* triple-click-selected. Set per-program in buildOutroData.
*/
handoffPrompt?: string;
}
import type { OutroData } from '@shared/outro';
import type { ResolvedBinding } from './runner/shared/types';
export type { OutroData } from '@shared/outro';

/** A single question rendered by the WizardAsk overlay. */
export interface AskQuestion {
Expand Down Expand Up @@ -136,7 +88,7 @@ export interface PendingQuestion {
* the main session, and some programs override to Haiku, so pricing must key
* off the per-turn model rather than a single run-wide assumption. Omit only
* when the caller genuinely has no model context (falls back to Sonnet
* pricing — see `pricePerMtokForModel` in `@lib/agent/token-pricing`).
* pricing — see `pricePerMtokForModel` in `@shared/token-pricing`).
*/
export interface TokenUsageDelta {
inputTokens: number;
Expand All @@ -148,7 +100,7 @@ export interface TokenUsageDelta {
model?: string;
}

/** The run spinner as the agent drives it: `WizardUI.spinner()` returns one. */
/** The run spinner as the agent drives it. Each call is one `spinner` event. */
export interface SpinnerHandle {
start(message?: string): void;
stop(message?: string): void;
Expand Down Expand Up @@ -185,7 +137,7 @@ export interface AuthErrorDetail {
logFilePath: string;
}

/** One task as the caller renders it. The same shape `WizardUI.syncTodos` takes. */
/** One task in the run's task list, as the `tasks` event carries it. */
export interface TaskSnapshot {
id?: string;
source?: string;
Expand All @@ -197,43 +149,43 @@ export interface TaskSnapshot {
export type ProgressLogLevel = 'info' | 'warn' | 'error' | 'success' | 'step';

/**
* Everything the agent reports while it runs. One event per former
* `getUI()` call, in the same order, with the same payload, so a reducer that
* maps each case back onto `WizardUI` reproduces today's output exactly.
* Everything the agent reports while it runs, in the order it happens.
*
* Payloads are copies. Never a store, a setter, a function or a live
* collection. The callback returns nothing and the agent never branches on it.
*/
export type AgentProgress =
/** The run's main work has started (`WizardUI.startRun`). */
/** The run's resolved sequence, harness and model, once, before it starts. */
| { kind: 'binding'; binding: ResolvedBinding }
/** The run's main work has started. */
| { kind: 'lifecycle'; phase: 'started' }
/** The run finished and the caller may show its outro (`WizardUI.outro`). */
/** The run finished and the caller may show its outro. */
| { kind: 'lifecycle'; phase: 'completed'; message: string }
/** The run spinner (`WizardUI.spinner()`), one handle per run. */
/** The run spinner: start, stop or change its message. One per run. */
| {
kind: 'spinner';
action: 'start' | 'stop' | 'message';
message?: string;
}
/** A log line (`WizardUI.log[level]`). */
/** A log line at a level. */
| { kind: 'log'; level: ProgressLogLevel; message: string }
/** A `[STATUS]` line the agent printed (`WizardUI.pushStatus`). */
/** A `[STATUS]` line the agent printed. */
| { kind: 'status'; message: string }
/** The full task list, already sorted for display (`WizardUI.syncTodos`). */
/** The full task list, already sorted for display. */
| { kind: 'tasks'; tasks: TaskSnapshot[] }
/** The stage of work derived from the active tool (`WizardUI.setStage`). */
/** The stage of work derived from the active tool. */
| { kind: 'stage'; stage: string }
/** A PostHog URL the agent created (`setDashboardUrl` / `setNotebookUrl`). */
/** A PostHog dashboard or notebook URL the agent created. */
| { kind: 'url'; which: 'dashboard' | 'notebook'; url: string }
/** One assistant turn's token usage (`WizardUI.addTokenUsage`). */
/** One assistant turn's token usage. */
| { kind: 'usage'; delta: TokenUsageDelta }
/** The SDK's authoritative run cost (`WizardUI.setFinalTokenCostUsd`). */
/** The SDK's authoritative run cost, in USD. */
| { kind: 'finalCost'; usd: number }
/** The gateway returned 401; a failure follows (`WizardUI.showAuthError`). */
/** The gateway returned 401; a failure follows. */
| { kind: 'authError'; detail: AuthErrorDetail }
/** The handoff document the agent published (`WizardUI.setHandoffText`). */
/** The handoff document the agent published. */
| { kind: 'handoff'; text: string }
/** The run's final outro payload (`WizardUI.setOutroData`). */
/** The run's final outro payload. */
| { kind: 'completion'; outro: OutroData }
/** One short line per agent step, only from a run that collects its transcript. */
| { kind: 'activity'; line: string };
Expand Down
74 changes: 44 additions & 30 deletions src/agent/runner/shared/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,23 +5,23 @@
* invocation snapshot, reports through `options.onProgress`, asks through
* `options.interaction`, and returns a `RunResult`. Nothing here names a UI,
* a store, a session or a program registry: the caller resolves those and
* hands over plain data. `src/programs/run-agent-legacy.ts` is the caller
* that rebuilds today's session-driven behavior on top of this contract.
* hands over plain data.
*/

import type { CloudRegion } from '@utils/types';
import type { Credentials } from '@shared/api';
import type { AuthErrorDetail, OutroData, TaskNotice } from '@agent/progress';
import type { PromptContext } from '@agent/agent-prompt';
import type { AuthErrorDetail, OutroData, TaskNotice } from '../../progress';
import type { PromptContext } from '../../agent-prompt';
import type { PackageManagerDetector } from '@utils/package-manager';
import type { ApiProject, ApiUser } from '@shared/api';
import type { Harness, Integration, Sequence } from '@shared/constants';
import type { ErrorCode } from '@shared/errors';
import type { LLMProvider } from '@posthog/warlock';
import type { AgentInteraction, ProgressEmitter } from '@agent/progress';
import type { AgentInteraction, ProgressEmitter } from '../../progress';
import type { EffortLevel } from '../switchboard/models';
import type { SwitchboardCtx } from '../switchboard';
import type { AgentBinding, SwitchboardCtx } from '../switchboard';
import type { TranscriptTail } from './transcript-tail';
import { RunOutcome } from '@shared/run-state';

export type { PromptContext, Credentials };

Expand Down Expand Up @@ -133,7 +133,7 @@ export interface RunHooks {
) => void;
}

/** The run-level routing decision the caller made. */
/** The run-level routing decision. */
export interface ResolvedBinding {
sequence: Sequence;
harness: Harness;
Expand All @@ -144,31 +144,51 @@ export interface ResolvedBinding {
}

/**
* Resolved execution data for one agent run. The caller has already decided
* which program this is, how it is routed and which flags apply; the agent
* treats every label as opaque.
* How the caller routes a run. The agent resolves the launch overrides and the
* feature flags on top of the program's binding (CLI, then flag, then binding).
*/
export interface RunConfig {
export interface AgentRouting {
/** The program's binding; `DEFAULT_BINDING` when it declares none. */
binding: AgentBinding;
/** `--harness`, `--sequence` and `--model`; dev and test builds only. */
overrides?: { harness?: Harness; sequence?: Sequence; model?: string };
/** Record the decision in analytics tags and the `switchboard resolved` event. Default true. */
record?: boolean;
}

/**
* Execution data for one agent run. The caller decides which program this is,
* what its binding is and which flags apply; the agent treats every label as
* opaque.
*/
export type RunConfig = Omit<
ResolvedRunConfig,
'binding' | 'switchboard' | 'wizardMetadata'
> & {
routing: AgentRouting;
/** Extra gateway trace tags, laid over the ones the agent builds. */
tags?: Record<string, string>;
};

/** A run's config once its routing is resolved: what the sequences read. */
export interface ResolvedRunConfig {
/** Program id: gateway spend pin, analytics label, commandments axis. */
programId: string;
/** The run definition. A program's session-taking hooks are the caller's, see `hooks`. */
run: AgentRunDefinition;
/** A composed sub-run leaves the terminal outro to its caller. */
composed: boolean;
/** Run-level sequence, harness and model. */
/** Run-level sequence, harness and model, resolved from `routing`. */
binding: ResolvedBinding;
/**
* The inputs the run-level binding was resolved from. The orchestrator
* re-resolves the harness per task role from these; nothing else reads them.
*/
/** What the binding was resolved from; the orchestrator re-resolves each task role from it. */
switchboard: SwitchboardCtx;
/** Primary skills origin (context-mill dev or GitHub Releases). */
skillsBaseUrl: string;
/** Feature flag key → variant, evaluated before the run. */
wizardFlags: Record<string, string>;
/** Flag payloads from the same snapshot. */
wizardFlagPayloads: Record<string, unknown>;
/** Gateway trace tags for this run, already stamped with sequence and harness. */
/** Gateway trace tags for this run, stamped with sequence and harness. */
wizardMetadata: Record<string, string>;
/** Extra tools added on top of BASE_ALLOWED_TOOLS for this run. */
allowedTools?: readonly string[];
Expand Down Expand Up @@ -250,13 +270,12 @@ export interface BootstrapResult {
}

/**
* A decided failure. The same fields `wizardAbort` takes, so the legacy
* adapter passes it through untouched and the exit sequence, codes and
* messages stay exactly what they were.
* A decided failure. The same fields `wizardAbort` takes, so a host passes it
* through untouched to end the process with its code and message.
*/
export interface AgentFailure {
message: string;
/** Structured error data. Renders via `outroError` instead of `outro`. */
/** Structured error data for the outro; built from `message` when absent. */
outroData?: OutroData;
error?: Error;
exitCode?: number;
Expand All @@ -265,12 +284,7 @@ export interface AgentFailure {
authErrorDetail?: AuthErrorDetail;
}

export enum RunOutcome {
Success = 'success',
Aborted = 'aborted',
Failed = 'failed',
Crashed = 'crashed',
}
export { RunOutcome };

/** Totals of every `usage` event the run emitted. */
export interface TokenUsageTotals {
Expand All @@ -282,7 +296,7 @@ export interface TokenUsageTotals {

/** What the agent reported, accumulated independently of any observer. */
export interface RunSnapshot {
tasks: import('@agent/progress').TaskSnapshot[];
tasks: import('../../progress').TaskSnapshot[];
statusMessages: string[];
stage?: string;
usage: TokenUsageTotals;
Expand Down Expand Up @@ -319,15 +333,15 @@ export type RunResult = (

export interface RunAgentOptions {
/** Receives every progress event in emission order. Never awaited. */
onProgress?: (event: import('@agent/progress').AgentProgress) => unknown;
onProgress?: (event: import('../../progress').AgentProgress) => unknown;
/** Answers the agent's questions. Absent → no ask bridge, notices declined. */
interaction?: AgentInteraction;
signal?: AbortSignal;
}

/** What a sequence receives: the contracts plus the prepared run. */
export interface SequenceContext {
config: RunConfig;
config: ResolvedRunConfig;
input: RunInput;
boot: BootstrapResult;
emit: ProgressEmitter;
Expand Down
Loading
Loading